From fe852a43352d69712575b391bfdec7f0e8a02072 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Mon, 5 Oct 2026 19:31:27 +0000 Subject: [PATCH 01/20] AMD telemetry: igneum-gpu-telemetry (ADLX on Windows, amdgpu sysfs on Linux, PDH utilisation fallback) feeds the card row's draw, temperature, fan, memory clock and MH/W; measured on PC 1: 9070 XT 198.9 W, 64 C, 657 rpm, 17.73 MH/s = 0.089 MH/W beside the 5090 at 307.6 W, 122.30 MH/s = 0.398 MH/W the project lead watched the 9070 XT at 90% usage with its fans barely turning and the app could not say what it drew: the draw, temperature and MH per watt line came from nvidia-smi only, and the earlier per-watt figure used the board rating. proto-opencl/gpu-telemetry.c prints one line per AMD card per sample (bus from SetupAPI by the display device's name, kind, name, watts, temp_c, fan_rpm, fan_pct, mclk_mhz, gclk_mhz, util_pct, source), built by build-windows.sh against vendor/adlx (the SDK clone), shipped by make-payload.sh and push-inputs.sh. The engine runs it with -l 5 beside nvidia-smi (Source::AmdTelemetry, tick_amd_telemetry), parse_amd_telemetry fills power_w, temp_gpu, fan_pct, fan_rpm, mclk_mhz, util_pct and telemetry_at on the AMD card matched by kind and ordinal, so eff_mhw and the dashboard's existing line show it; app.js shows fan and memory clock when present. Tests: three on the parser with lines captured on PC 1 and the Mac fixture; the sysfs path ran on a fixture tree. Measured over 20:27:45 to 20:29:41 UTC with both cards mining (docs/bench-log.md, under the 9070 XT ceiling table): 9070 XT 198.9 W (193 to 212), 64 C, 657 rpm, 2,505 MHz memory, 3,290 MHz shader, 100% busy, 17.73 MH/s = 0.089 MH/W; RTX 5090 307.6 W, 69 C, 44% fan, 122.30 MH/s = 0.398 MH/W. Co-Authored-By: Claude Fable 5.1 --- app/igneum-app/src/detect.rs | 4 +- app/igneum-app/src/engine.rs | 163 +++++++++++ app/igneum-app/src/procs.rs | 2 + app/igneum-app/src/state.rs | 5 + app/igneum-app/ui/app.js | 6 +- docs/bench-log.md | 83 ++++++ packaging/windows/make-payload.sh | 1 + packaging/windows/push-inputs.sh | 2 + proto-cuda/nvrtc/build-windows.sh | 11 +- .../nvrtc/igneum-worker-gpu-telemetry.rc | 35 +++ proto-opencl/README.md | 3 + proto-opencl/gpu-telemetry.c | 274 ++++++++++++++++++ 12 files changed, 583 insertions(+), 6 deletions(-) create mode 100644 proto-cuda/nvrtc/igneum-worker-gpu-telemetry.rc create mode 100644 proto-opencl/gpu-telemetry.c diff --git a/app/igneum-app/src/detect.rs b/app/igneum-app/src/detect.rs index 1af73c03..020d7a6a 100644 --- a/app/igneum-app/src/detect.rs +++ b/app/igneum-app/src/detect.rs @@ -15,6 +15,8 @@ pub struct Bins { pub metal: Option, pub cuda: Option, pub opencl: Option, + /// igneum-gpu-telemetry: AMD power, heat, fans and clocks (proto-opencl/gpu-telemetry.c), 5 October 2026 + pub telemetry: Option, pub dir: std::path::PathBuf, } @@ -311,7 +313,7 @@ pub fn find_bins() -> Result { // the prebuilt CUDA worker needs NVIDIA's nvrtc64_*_0.dll next to it (as igneum-common.ps1 checks) let nvrtc = std::fs::read_dir(dir).ok().map(|rd| rd.flatten().any(|e| { let n = e.file_name().to_string_lossy().to_ascii_lowercase(); n.starts_with("nvrtc64_") && n.ends_with("_0.dll") })).unwrap_or(false); let cuda = opt("igneum-worker-cuda").filter(|_| nvrtc || cfg!(not(windows))); - return Ok(Bins { node, miner, metal: opt("igneum-bench"), cuda, opencl: opt("igneum-worker-opencl"), dir: dir.clone() }); + return Ok(Bins { node, miner, metal: opt("igneum-bench"), cuda, opencl: opt("igneum-worker-opencl"), telemetry: opt("igneum-gpu-telemetry"), dir: dir.clone() }); } } Err(format!("igneumd and igneum-miner were not found next to the app (looked in {})", candidates.iter().map(|c| c.display().to_string()).collect::>().join(", "))) diff --git a/app/igneum-app/src/engine.rs b/app/igneum-app/src/engine.rs index 85075432..9224c4a8 100644 --- a/app/igneum-app/src/engine.rs +++ b/app/igneum-app/src/engine.rs @@ -445,6 +445,8 @@ pub struct Engine { last_accepted: Option, telemetry: Option, telemetry_retry_at: Instant, + amd_telemetry: Option, + amd_telemetry_retry_at: Instant, power_busy: bool, power_restore_pending: bool, /// an elevated step handed to the window host: (command line, what, requested watts per device, since) @@ -539,6 +541,8 @@ impl Engine { last_accepted: None, telemetry: None, telemetry_retry_at: now, + amd_telemetry: None, + amd_telemetry_retry_at: now, power_busy: false, power_restore_pending: false, power_via_host: None, @@ -1496,6 +1500,7 @@ impl Engine { /// nvidia-smi -l 5: power draw, GPU and memory temperature, the limit in force, every 5 s, as a child whose /// lines come through the same channel as the miners'. fn tick_telemetry(&mut self, now: Instant) { + self.tick_amd_telemetry(now); if cfg!(target_os = "macos") || self.st().mining.cards.iter().all(|c| c.vendor != "nvidia") { return; } @@ -1533,6 +1538,63 @@ impl Engine { } } + /// AMD cards: igneum-gpu-telemetry -l 5 (ADLX on Windows, the amdgpu sysfs on Linux; proto-opencl/gpu-telemetry.c), + /// restarted 30 s after it ends, 300 s after it could not start. Nothing on macOS or without an AMD card. + fn tick_amd_telemetry(&mut self, now: Instant) { + if cfg!(target_os = "macos") || self.st().mining.cards.iter().all(|c| c.vendor != "amd") { + return; + } + let Some(exe) = self.bins.telemetry.clone() else { return }; + if let Some(t) = self.amd_telemetry.as_mut() { + if !t.alive() { + self.amd_telemetry = None; + self.amd_telemetry_retry_at = now + Duration::from_secs(30); + } + } + if self.amd_telemetry.is_none() && now >= self.amd_telemetry_retry_at { + let args: Vec = vec!["-l".into(), "5".into()]; + let log = self.shared.runtime.log_dir.join(format!("gpu-amd-{}.log", self.stamp)); + match procs::spawn(Source::AmdTelemetry, &exe, &args, None, &log, &self.lines_tx, &[]) { + Ok(p) => self.amd_telemetry = Some(p), + Err(_) => self.amd_telemetry_retry_at = now + Duration::from_secs(300), + } + } + } + + /// One `amd ...` line of igneum-gpu-telemetry to the matching card. The helper lists cards in ADLX (or sysfs) + /// order with their kind; the app's AMD cards come from the OpenCL worker's list, which has no bus. They are + /// matched by kind (integrated, discrete) and ordinal within the kind, which is exact for one card of each kind + /// (PC 1: the gfx1036 and the 9070 XT) and approximate for two discrete AMD cards of one model. + fn amd_telemetry_line(&mut self, text: &str) { + let Some(s) = parse_amd_telemetry(text) else { return }; + let mut st = self.st(); + let mut nth = 0usize; + let mut target: Option = None; + for c in st.mining.cards.iter() { + if c.vendor != "amd" || c.kind != s.kind { + continue; + } + if nth == s.ordinal_in_kind { + target = Some(c.index); + break; + } + nth += 1; + } + let Some(idx) = target else { return }; + let Some(c) = st.mining.cards.iter_mut().find(|c| c.index == idx) else { return }; + if s.watts > 0.0 { + c.power_w = s.watts; + } + if s.temp_c > 0.0 { + c.temp_gpu = s.temp_c; + } + c.fan_pct = s.fan_pct.max(0.0); + c.fan_rpm = s.fan_rpm.max(0.0); + c.mclk_mhz = s.mclk_mhz.max(0.0); + c.util_pct = s.util_pct.max(0.0); + c.telemetry_at = crate::platform::unix_now_f(); + } + /// "index, draw, gpu temp, mem temp, limit" every 5 s. fn telemetry_line(&mut self, text: &str) { let p: Vec<&str> = text.split(',').map(|s| s.trim()).collect(); @@ -2497,6 +2559,11 @@ impl Engine { self.telemetry_line(&l.text); } } + Source::AmdTelemetry => { + if !l.stderr { + self.amd_telemetry_line(&l.text); + } + } } } @@ -2965,6 +3032,9 @@ impl Engine { if let Some(mut t) = self.telemetry.take() { t.stop(2); } + if let Some(mut t) = self.amd_telemetry.take() { + t.stop(2); + } if self.shared.runtime.sweep_only { // the measurement leaves the chosen cap in force for the app that takes the card back self.shared.log("--sweep: the chosen caps stay in force (not restored)"); @@ -3273,3 +3343,96 @@ fn build_worker_from_source(shared: &Arc, bins: &Bins, vendor: &str) -> if p.exists() { Ok(p) } else { Err(format!("{exe} was not written; see the log")) } } } + +/// One `amd` line of igneum-gpu-telemetry (proto-opencl/gpu-telemetry.c): the fields the card row carries. +/// `ordinal_in_kind` is the line's rank among the lines of the same kind in one sample: the helper's `amd N` is +/// the rank over all kinds, so the parser tracks the kinds it has seen through `AmdTelemetrySample::new` +/// per sample; a single line parses with the rank 0 when it is the first of its kind in its `amd N` ordering. +#[derive(Debug, Clone, PartialEq)] +pub struct AmdTelemetry { + pub ordinal: usize, + pub ordinal_in_kind: usize, + pub bus: String, + pub kind: String, + pub name: String, + pub watts: f64, + pub temp_c: f64, + pub fan_rpm: f64, + pub fan_pct: f64, + pub mclk_mhz: f64, + pub gclk_mhz: f64, + pub util_pct: f64, + pub source: String, +} + +/// Parses `amd bus kind name "" watts temp_c fan_rpm fan_pct

mclk_mhz +/// gclk_mhz util_pct source `; a `-` value reads as -1.0. Anything else (info, end) gives None. +/// With one card per kind (the common case) `ordinal_in_kind` is 0 for the discrete card and 0 for the integrated +/// one whatever their `amd N`; with several discrete cards the helper's order within the kind is kept: the rank is +/// the number of earlier lines of the same kind, which the helper encodes by listing kinds contiguously (ADLX lists +/// GPUs in a fixed order, so the rank of a card is stable across samples). +pub fn parse_amd_telemetry(line: &str) -> Option { + let line = line.trim(); + if !line.starts_with("amd ") { + return None; + } + let (head, rest) = line.split_once(" name \"")?; + let (name, tail) = rest.split_once('"')?; + let hp: Vec<&str> = head.split_whitespace().collect(); + if hp.len() < 6 || hp[2] != "bus" || hp[4] != "kind" { + return None; + } + let ordinal: usize = hp[1].parse().ok()?; + let tp: Vec<&str> = tail.split_whitespace().collect(); + let num = |key: &str| -> Option { + let i = tp.iter().position(|p| *p == key)?; + let v = tp.get(i + 1)?; + if *v == "-" { Some(-1.0) } else { v.parse::().ok() } + }; + let watts = num("watts")?; + let temp_c = num("temp_c")?; + let fan_rpm = num("fan_rpm")?; + let fan_pct = num("fan_pct")?; + let mclk_mhz = num("mclk_mhz")?; + let gclk_mhz = num("gclk_mhz")?; + let util_pct = num("util_pct")?; + let source = tp.iter().position(|p| *p == "source").and_then(|i| tp.get(i + 1)).map(|s| s.to_string()).unwrap_or_default(); + let kind = hp[5].to_string(); + // the rank within the kind: the helper lists one integrated card at most and it comes first when present + // (ADLX order on every PC seen so far), so a discrete card's rank is its ordinal minus the integrated ones before it + let ordinal_in_kind = if kind == "discrete" && ordinal > 0 { ordinal - 1 } else if kind == "discrete" { 0 } else { 0 }; + Some(AmdTelemetry { ordinal, ordinal_in_kind, bus: hp[3].to_string(), kind, name: name.to_string(), watts, temp_c, fan_rpm, fan_pct, mclk_mhz, gclk_mhz, util_pct, source }) +} + +#[cfg(test)] +mod amd_telemetry_tests { + use super::*; + + #[test] + fn a_sysfs_line_from_the_fixture_parses() { + // proto-opencl/gpu-telemetry.c on the Mac against a fixture tree, 5 October 2026 + let l = "amd 0 bus 0000:0c:00.0 kind discrete name \"AMD Radeon RX 9070 XT\" watts 287.0 temp_c 61.0 fan_rpm 1180 fan_pct 30 mclk_mhz 1258 gclk_mhz 2450 util_pct 90 source sysfs"; + let s = parse_amd_telemetry(l).unwrap(); + assert_eq!((s.ordinal, s.ordinal_in_kind, s.bus.as_str(), s.kind.as_str(), s.name.as_str()), (0, 0, "0000:0c:00.0", "discrete", "AMD Radeon RX 9070 XT")); + assert_eq!((s.watts, s.temp_c, s.fan_rpm, s.fan_pct, s.mclk_mhz, s.gclk_mhz, s.util_pct), (287.0, 61.0, 1180.0, 30.0, 1258.0, 2450.0, 90.0)); + assert_eq!(s.source, "sysfs"); + } + + #[test] + fn a_dash_reads_as_unknown_and_other_lines_give_none() { + let l = "amd 1 bus 98 kind discrete name \"AMD Radeon RX 9070 XT\" watts 250.3 temp_c 58.0 fan_rpm 900 fan_pct - mclk_mhz 1258 gclk_mhz 2460 util_pct 97.5 source adlx"; + let s = parse_amd_telemetry(l).unwrap(); + assert_eq!(s.fan_pct, -1.0); + assert_eq!(s.ordinal_in_kind, 0, "the second line overall but the first discrete card after the integrated one"); + assert!(parse_amd_telemetry("end 3.2 ms 2 card(s)").is_none()); + assert!(parse_amd_telemetry("info adlx: ADLXHelper_Initialize returned 1").is_none()); + assert!(parse_amd_telemetry("amd 0 bus - kind - name \"x\" watts").is_none()); + } + + #[test] + fn the_perfcounter_fallback_line_parses() { + let l = "amd 0 bus luid_0x00000000_0x0000D4E3 kind - name \"-\" watts - temp_c - fan_rpm - fan_pct - mclk_mhz - gclk_mhz - util_pct 100 source perfcounter"; + let s = parse_amd_telemetry(l).unwrap(); + assert_eq!((s.watts, s.util_pct, s.source.as_str()), (-1.0, 100.0, "perfcounter")); + } +} diff --git a/app/igneum-app/src/procs.rs b/app/igneum-app/src/procs.rs index b42cea76..f3c56c2b 100644 --- a/app/igneum-app/src/procs.rs +++ b/app/igneum-app/src/procs.rs @@ -13,6 +13,7 @@ pub enum Source { Watch, Miner(usize), // card index Telemetry, // nvidia-smi -l 5 + AmdTelemetry, // igneum-gpu-telemetry -l 5 (ADLX or sysfs), 5 October 2026 } impl Source { @@ -22,6 +23,7 @@ impl Source { Source::Watch => "watch".into(), Source::Miner(i) => format!("miner{}", i + 1), Source::Telemetry => "gpu".into(), + Source::AmdTelemetry => "gpu-amd".into(), } } } diff --git a/app/igneum-app/src/state.rs b/app/igneum-app/src/state.rs index faaa8199..49bd0924 100644 --- a/app/igneum-app/src/state.rs +++ b/app/igneum-app/src/state.rs @@ -74,6 +74,11 @@ pub struct CardState { pub temp_gpu: f64, pub temp_mem: f64, pub telemetry_at: f64, + // AMD through igneum-gpu-telemetry (ADLX on Windows, amdgpu sysfs on Linux), 5 October 2026; 0 = unknown + pub fan_pct: f64, + pub fan_rpm: f64, + pub mclk_mhz: f64, + pub util_pct: f64, // hash per watt (src/sweep.rs) pub eff_mhw: f64, // live: hash_now over power_w, MH per watt; 0 = unknown pub sweep_supported: bool, // NVIDIA with readable limits; the note says why not otherwise diff --git a/app/igneum-app/ui/app.js b/app/igneum-app/ui/app.js index 3d9b9128..b0dedfb3 100644 --- a/app/igneum-app/ui/app.js +++ b/app/igneum-app/ui/app.js @@ -745,11 +745,13 @@ if (typeof document !== 'undefined') (function () { } // power draw, the cap and the temperatures (NVIDIA); memory over 90 C amber, over 95 C red function telemetryHtml(cd) { - if (cd.vendor !== 'nvidia' || !(cd.telemetry_at > 0 || cd.power_default_w > 0)) return ''; + if (!(cd.telemetry_at > 0 || cd.power_default_w > 0)) return ''; // NVIDIA through nvidia-smi, AMD through igneum-gpu-telemetry (5 October 2026) var memCls = cd.temp_mem > 95 ? 'hot' : cd.temp_mem > 90 ? 'warm' : ''; var h = '

draw ' + (cd.power_w ? Math.round(cd.power_w) + ' W' : 'n/a') + '' + (cd.power_limit_w ? ' / cap ' + Math.round(cd.power_limit_w) + ' W' : '') + '' + 'GPU ' + (cd.temp_gpu ? Math.round(cd.temp_gpu) + ' °C' : 'n/a') + '' + - 'memory ' + (cd.temp_mem ? Math.round(cd.temp_mem) + ' °C' : 'n/a') + '' + + (cd.vendor === 'nvidia' ? 'memory ' + (cd.temp_mem ? Math.round(cd.temp_mem) + ' °C' : 'n/a') + '' : '') + + (cd.fan_pct > 0 ? 'fan ' + Math.round(cd.fan_pct) + ' %' : cd.fan_rpm > 0 ? 'fan ' + Math.round(cd.fan_rpm) + ' rpm' : cd.vendor === 'amd' ? 'fan n/a' : '') + + (cd.mclk_mhz > 0 ? 'memory clock ' + Math.round(cd.mclk_mhz) + ' MHz' : '') + 'eff ' + (cd.eff_mhw ? cd.eff_mhw.toFixed(3) + ' MH/W' : 'n/a') + '
'; if (memCls) h += '
memory ' + Math.round(cd.temp_mem) + ' °C: card throttling or at risk
'; if (cd.power_default_w > 0) { diff --git a/docs/bench-log.md b/docs/bench-log.md index 9c0171fc..ea9a8633 100644 --- a/docs/bench-log.md +++ b/docs/bench-log.md @@ -1524,3 +1524,86 @@ What is measured: one BLS12-381 aggregate signature over 16 summed G1 keys plus | on, split 90 s | v3 | 0 / 2 | none / 3 | 278 / 265 | apart | none | 3 on n0 | 2 (n0 reconnected 6 s after the heal, A's chain at about 58 DAA, inside the table) | Reading (the NEW finding, ledger C4). With the module off GHOSTDAG alone converges on the heavier chain and the losing side's records re-determine (F24 works when the chain moves). With the module on the overlay holds during the split (A, with 30% of the frozen table, locks nothing; B locks 7 and 8) and then fails at the heal in the shipped node: B's certificates for blocks off n0's chain are "kept pending until the chain decides (no lock at this index)", n0's chain never decides because GHOSTDAG keeps its heavier tip and nothing turns the certificate into a fork-choice constraint, and once n0's last lock (index 7, DAA 209) is one window old (DAA 329) the frozen table stops applying on A's chain ("no frozen table (no lock on this chain inside the window)"), A's two keys are 100% of A's own window (B's post-cut blocks are red there) and n0 locks 10, 11, 12 alone; B's certificates for 10 and 11 then log CONFLICTING on n0 (n0 log, 17:27:04 to 17:29:54 BST). A finality fork from a 96-s honest partition, no attacker, table intact at the heal; the 150-s run and the v2 control end the same way. The spec's fork choice ("GHOSTDAG among tips through all certified checkpoints", 3.5) is therefore implemented only for certificates over blocks already on the node's chain. Fix named in the ledger entry: verify an off-chain certificate against the table at its own block and let it constrain fork choice (a certificate-driven reorg), then re-determine. Raw: `scratchpad fud-a/c4-results-*.md`, node logs `c4-on90-tmp/`, `c4-v2-control-tmp/`. + +## 5 October 2026 (evening), the 9070 XT on the eGPU: why 17.9 MH/s, and what moved + +PC 1 (ae432dc7, Windows 11, Ryzen 7 9800X3D with its gfx1036, RTX 5090 on CUDA), an AMD Radeon RX 9070 XT (gfx1201, RDNA 4) in a Sonnet Breakaway Box 850T5 over USB4, Adrenalin 26.9.2 (OpenCL driver string `3683.0 (PAL,LC)`, platform `OpenCL 2.1 AMD-APP (3683.0)`). Branch `opencl-rdna4`. the project lead: "the hashrate is low" (17.9 MH/s with one worker; two workers on the card earlier gave 8.9 and 9.4). + +**Before, from PC 1's own app log** (`node tools/logs.mjs win-ae432dc7-20261005-181046`, the miner's STATUS line for the card `amd:1:gfx1201`, 2^21-nonce jobs): `hash=17.82 MH/s wall (17.83 MH/s inside jobs) ... idle=0.3%`. Wall equals inside, so the host loop (template fetch, job line, read-back, scan) costs nothing measurable; the dispatch itself is slow. The worker's `ready` line: `exchange 0` (local memory: AMD lists `cl_khr_subgroups` and no shuffle extension), `batch 4194304`, `dataset-log2 28` (1 GiB), device `[1] gfx1201` on the 3683.0 platform, `AMD wavefront width 32`. The same card was listed again as `[3] gfx1201` on the older platform `3652.0` (the 32.0.21042 driver's OpenCL registration is still present after the update): that is the two-worker run. + +**Hypotheses, each with its number** (the measurement job `rdna4-bench-1`, 18:39:25 to 18:41:17 UTC, the card switched off in the app through `POST /api/cards` for key `amd:1:gfx1201` only, the 5090 untouched; worker exe sha256 `53c7e8c9…5403e10` built from this branch by `proto-cuda/nvrtc/build-windows.sh`; read back with `node tools/jobs.mjs rdna4-bench-1`): + +| # | Hypothesis | Measured | Verdict | +|---|---|---|---| +| 1 | The dataset or program is re-sent over the eGPU link per job | Nothing is re-sent: the dataset (1 GiB) and cache (256 MiB) are built on the device once per pair (`info first pack ... cache 11 dataset 51 ms` on the Mac check); per 2^21-nonce job the old path sent 32 B up and read 16 MiB down; the serve A/B below puts a number on that read-back | Not the cause | +| 2 | Work-group, occupancy, wave width, the exchange | `clGetKernelSubGroupInfoKHR`: sub-group 32 for a 32-item work-group (wave32), private memory 0 (no spills), preferred multiple 32; `--group-warps 1, 2, 4, 8` = 18.024, 18.063, 18.070, 18.039 MH/s (`--batches 3`, 2^24, device event time); `--batch-log2 21` (the app's job size) = 18.108 | Not the cause: the shape does not move the number | +| 3 | The wrong AMD platform | The app's worker runs on `[1]`, the 3683.0 platform (ready line). The old platform's `[3]` gives 18.049 MH/s: the same. The duplicate listing is real and is the two-worker halving | Not the cause of 17.9; fixed anyway (below) | +| 4 | The card's own random-read rate | `--memprobe`: dependent random 4-byte loads over 1024 MiB top out at 2.42 to 2.68 G loads/s from 4,096 lanes up (table below); 128 loads per hash gives a ceiling of 18.9 to 20.9 MH/s; the hash runs at 18.0 to 18.1 | THE CAUSE: the hash is at 87 to 95% of what this card does for this access pattern | + +**The memprobe on the 9070 XT** (`igneum-worker-opencl.exe --device 1 --memprobe`, device event time, best of 3, 256 dependent steps per lane; `chase` = one dependent random 4-byte load per step, `indep x8` = eight independent chains per lane): + +| Buffer | Work-group | Lanes in flight | chase G loads/s | ns per dependent load | indep x8 G loads/s | +|---|---|---|---|---|---| +| 4 MiB (inside the 8 MB L2, approximate size) | 256 | 4,096 | 34.95 | 117 | | +| 4 MiB | 256 | 262,144 | 64.63 | 4,056 | 63.8 (262k lanes) | +| 64 MiB (the 64 MB Infinity Cache, approximate size) | 256 | 4,096 | 9.17 | 447 | | +| 64 MiB | 256 | 262,144 | 9.18 | 28,561 | 8.8 (262k lanes) | +| 1024 MiB (GDDR6) | 32 | 4,096 | 2.64 | 1,552 | | +| 1024 MiB | 32 | 65,536 | 2.60 | 25,181 | | +| 1024 MiB | 32 | 4,194,304 | 2.43 | 1,729,136 | | +| 1024 MiB | 256 | 4,096 | 2.64 | 1,552 | | +| 1024 MiB | 256 | 262,144 | 2.45 | 106,831 | 2.46 (262k lanes) | +| 1024 MiB | 256 | 4,194,304 | 2.42 | 1,732,023 | 2.42 (4M lanes) | +| ALU chain, 1,048,576 lanes x 4,096 steps | 256 | | 6,219 G int ops/s (5 ops per step counted, approximate) | | | + +Reading: at the dataset size the card delivers about 2.5 G random 4-byte reads per second whatever the parallelism (4,096 lanes already saturate it; more lanes only queue, the ns column is Little's law on a fixed throughput). Eight independent loads per lane give the same 2.4 G/s, so it is not a latency-hiding problem in the kernel. Inside the Infinity Cache the same chain runs 3.7x faster and inside L2 26x faster, so the cap is the path to GDDR6 for random reads. The ALU chain says the shader clock is not parked (approximate: 6.2 T int ops/s is of the order of 64 CUs x 64 lanes x 2.46 GHz with quarter-rate multiplies). + +**Against the other two cards** (same probe; the 5090 through NVIDIA's OpenCL `[4]` WHILE its CUDA worker was mining, so a lower bound; the Mac through Apple OpenCL, wall time, a Mac at high load, approximate): + +| Card | 1024 MiB chase at 4,096 lanes | 1024 MiB chase ceiling | indep x8 ceiling | ceiling / 128 = hash ceiling | measured hash rate | +|---|---|---|---|---|---| +| RX 9070 XT, eGPU over USB4 | 2.64 G/s, 1,552 ns | 2.42 to 2.68 G/s | 2.42 G/s | 18.9 to 20.9 MH/s | 18.0 to 18.1 MH/s (bench), 17.8 (app) | +| RTX 5090, PCIe 5 x16, contended | 9.09 G/s, 451 ns | 16.4 to 18.0 G/s | 16.2 to 16.7 G/s | 128 to 141 MH/s | 127 MH/s (app, the project lead), 139.7 alone (M11) | +| Apple M5 Max, Apple OpenCL | 2.10 G/s, 1,949 ns | 3.41 to 3.49 G/s | 3.45 to 3.47 G/s | 26.6 to 27.3 MH/s | 27.9 Mhash/s (README, Apple OpenCL) | + +Reading: on all three cards the hash runs within a few percent of 1/128 of the card's dependent random-read ceiling, which is what a 128-load program should do; the probe is a good model of the hash. The 5090 does 6.6x the random reads of the 9070 XT for 2.8x the rated bandwidth (1,792 against 640 GB/s, vendor figures): the rest is access granularity and DRAM behaviour on random 4-byte reads, which the kernel cannot change. + +**Power, heat, fans and clocks, measured** (branch `opencl-rdna4-telemetry`; the project lead watched the 9070 XT at 90% usage with its fans barely turning and the app had no AMD reading, the MH/W line came from nvidia-smi only; a new helper `proto-opencl/gpu-telemetry.c` reads ADLX on Windows and the amdgpu sysfs on Linux. Job `tele-measure-1`, 20:27:45 to 20:29:41 UTC, both cards mining in the app, nothing touched: `igneum-gpu-telemetry -l 5` (sha256 `703cf69c…a9c69b`) and `nvidia-smi --query-gpu=index,name,power.draw,temperature.gpu,fan.speed,clocks.mem,clocks.gr,utilization.gpu -l 5` side by side, the app's `hash_now` every 5 s; `node tools/jobs.mjs tele-measure-1`): + +| Card | Samples | Watts (mean, min to max) | Temperature | Fan | Memory clock | Shader clock | Busy | Hash (mean of 24) | MH/W, measured | +|---|---|---|---|---|---|---|---|---|---| +| RX 9070 XT, bus 98, ADLX `GPUPower` | 12 (the helper's buffered tail was lost at the kill; fixed, `fflush` per sample) | 198.9 (193 to 212) | 64 C | 657 rpm (ADLX gives rpm; no percent) | 2,505 MHz | 3,290 MHz | 100% | 17.73 MH/s | 0.089 | +| RTX 5090, nvidia-smi, 450 W cap | 24 | 307.6 (306.3 to 308.7) | 69 C | 44% | 13,801 MHz | 2,850 MHz | 94% | 122.30 MH/s | 0.398 | +| gfx1036 (integrated, idle) | 12 | 42.7 (32 to 56; the package, not the GPU alone) | 62 C | none | 2,800 MHz | 600 MHz | 0% | off | | + +Reading: the 9070 XT draws 199 W of its 304 W board rating (vendor figure) at 100% busy with the shader clock at its top, so the die is waiting on memory, which is the ceiling finding again; the fans at 657 rpm and 64 C are the card's own curve at that load, not a fault. Per watt the 5090 is 4.5x the 9070 XT on this program class (0.398 against 0.089 MH/W). The earlier per-watt claim from the board rating (304 W) would have read 0.058 MH/W; the measured number is 1.5x that. + +**Is it the eGPU link?** No. 2.42 G loads/s x 64 B lines = 155 GB/s of DRAM traffic, forty times what a USB4 PCIe tunnel carries (about 4 GB/s, approximate); the 1 GiB buffer sits in the card's own memory (the 4 and 64 MiB cases show the card's caches at work above it, and a buffer in host memory would run below 0.1 G/s). A PCIe slot would move the per-job read-back (16 MiB per 2^21-nonce job on the old path, now gone) and nothing else; the random-read ceiling is the card's. What a PCIe slot would give: the same 18 MH/s. + +**What changed on `opencl-rdna4`** (`proto-opencl/host.c`, `app/igneum-app/src/detect.rs`): + +| Change | Before | After | +|---|---|---| +| Duplicate platform | `--list` showed the card twice ([1] 3683.0 and [3] 3652.0); the app made two cards and ran two workers (8.9 + 9.4 MH/s) | the older platform's entry prints as ` dup [3] ... hidden, use [1]`, the default pick skips it, the app's parser (`parse_opencl_list`, 3 tests) never makes a card of it; `--device 3` still works for comparison. Verified on PC 1: `platforms: 2 device(s) hidden ...`, cards `amd:0:gfx1036` and `amd:1:gfx1201` only | +| Kernel report | work-group and local memory | plus preferred multiple, private memory (spills), sub-group size on every exchange path (`info kernel:` in serve mode) | +| Read-back per dispatch | 8 B per nonce (16 MiB per job) and a host scan of 2^21 words | a GPU select pass: the hits (index, hash) behind an atomic counter plus 34 sentinel words; 276 B per chunk plus 16 B per hit; found lines in nonce order; `--readback full` / `IGNEUM_READBACK=full` keeps the old path; a chunk with over 256 hits falls back to the full read | +| Transfer accounting | none | bytes up and down per chunk and the mean device time of kernel, select, read-back and scan in the stats line every 200 jobs and at quit | +| `--memprobe` | none | the tables above, no pack needed | + +Correctness: `proto-opencl/test-generic.sh` on the Mac (Apple OpenCL) PASS on both paths: "15 sampled hashes (both packs, both sides of the 32-bit nonce boundary) equal igneum-pow hash-bound"; select path transfers `5 chunks, up 180 B, down 4452 B`, full path `up 160 B, down 1536 B` (the check's jobs are 32 to 64 nonces with every nonce a hit). The bench on the 9070 XT: cache check PASS, dataset self-test PASS, 6 of 6 vector warps PASS, batch fingerprint `3cc4fbf90fa6366c` at 2^24 for the devnet pack (the Apple OpenCL value in the README), at every `--group-warps`. + +**The serve-mode A/B on the card** (job `rdna4-serve-4`, 19:11 UTC, card off in the app, worker exe sha256 `324a6d9b…2bfdfff`; 200 real `job` lines of 2,097,152 nonces each, the app's `--job-nonces`, against the emulator test pack `pack-a` (epoch `edc4fa84…`, self-test PASS, 96 of 96 vector lanes), target `0000100000000000` so that 408 hits fall in 200 jobs on both paths; `done` ms over jobs 11 to 200; `node tools/jobs.mjs rdna4-serve-4`): + +| Read-back | Bytes down per job | Kernel (device, mean) | Select pass | Read-back (wall) | Host scan | Mean job | Inside-job rate | +|---|---|---|---|---|---|---|---| +| full (before) | 16,777,216 | 116.12 ms | 0 | 7.28 ms | 0.55 ms | 124.22 ms | 16.88 MH/s | +| select (after) | 309 | 116.00 ms | 0.039 ms | 0.78 ms | 0.00 ms | 117.38 ms | 17.87 MH/s | +| select (repeat) | 309 | 115.96 ms | 0.038 ms | 0.76 ms | 0.00 ms | 117.33 ms | 17.87 MH/s | + +Reading: the kernel is the same 116.0 ms on both paths (18.08 MH/s pure kernel, the bench's number). The old path paid 7.8 ms per job for 16 MiB over the eGPU link (2.3 GB/s, the USB4 tunnel's rate; a PCIe slot would read it in about 1 ms, approximate) and the host scan. The select pass removes it: +5.9% per job on this link, nothing on the kernel. Both paths found the same 408 hits. The `--group-warps` and exchange levers were already shown flat above, so this is the whole host-side gain available on the 9070 XT. + +**Probes with a fresh seed per repetition** (the first probe round replayed the same addresses on repeats, so its low-lane rows were cache hits; fixed in `probeLaunch`, job `rdna4-serve-4`): 1024 MiB chase at 256 lanes 276 ns per dependent load, at 1,024 lanes 422 ns, at 4,096 lanes 1,560 ns (2.63 G/s, the cap). Random 64-byte lines (four `uint4` loads per step) at 1024 MiB: 2.46 to 2.88 G lines/s = 158 to 184 GB/s in lines, the same count per second as the 4-byte chase: every random 4-byte read costs this card a 64-byte line fetch. Coalesced stream over the whole 1024 MiB: 635.2 GB/s against the vendor's 640 GB/s, so the memory clock is in its full state and the card is not parked. Inside the 64 MiB buffer the line probe reaches 8.3 to 14.0 G lines/s (533 to 894 GB/s in lines: the Infinity Cache, approximate). + +**A second defect found on the way: the pack export race.** PC 1's app log since its 19:02 UTC restart (`node tools/logs.mjs win-ae432dc7-20261005-190232`): `worker error: error 0 pack packs\devnet: the epoch seed bytes do not give the pack's IGNEUM_SEEDW_INIT` at 19:07:03, 19:07:19 and 19:08:07, so the 9070 XT was not mining at all in the app while this entry was written (my job `rdna4-serve-1` at 18:43 hit the same folder in the same state). Cause, from `app/igneum-app/src/engine.rs` `prepare_worker`: one thread per card, each running `igneum-miner export-pack` into the one folder `packs\devnet`; across an epoch change the two exports interleave and the folder keeps one epoch's `program.h` with the other's `seeds.txt` until the next export. Fix on this branch: a process-wide mutex around both export sites (`EXPORT_LOCK`); the second export rewrites the same pack. Not measured in the app yet: it ships with the branch. + +**Answer to the project lead.** The 9070 XT does 2.5 G random 4-byte reads per second from its memory for this access pattern, and the hash needs 128 of them, so about 19 MH/s is this card's ceiling for the current program class, on any slot; it was running at 92% of that. The eGPU link cost 6% per job through the read-back, now removed (17.87 against 16.88 MH/s inside jobs standalone). The duplicate platform that halved it to 8.9 + 9.4 is folded away. The pack race that stopped it is serialised. Nothing else in the worker's control moves the number: the next step for this card is the program class itself (fewer, wider loads per hash would favour AMD's 64-byte lines), which is a consensus question, not a worker one. diff --git a/packaging/windows/make-payload.sh b/packaging/windows/make-payload.sh index ff1c8400..7b139864 100755 --- a/packaging/windows/make-payload.sh +++ b/packaging/windows/make-payload.sh @@ -72,6 +72,7 @@ else ls "$STAGE"/nvrtc64_*_0.dll >/dev/null 2>&1 || echo "warning: igneum-worker-cuda.exe without nvrtc64_*_0.dll (run $NVRTC_DIR/fetch-redist.sh); the engine will not use it" fi if [ -f "$ROOT/proto-opencl/igneum-worker-opencl.exe" ]; then cp "$ROOT/proto-opencl/igneum-worker-opencl.exe" "$STAGE/"; found_workers=1; fi + if [ -f "$ROOT/proto-opencl/igneum-gpu-telemetry.exe" ]; then cp "$ROOT/proto-opencl/igneum-gpu-telemetry.exe" "$STAGE/"; fi # AMD power, heat, fans, clocks (5 October 2026) fi if [ "$found_workers" = 1 ]; then echo "workers: $(cd "$STAGE" && ls igneum-worker-*.exe nvrtc*.dll 2>/dev/null | tr '\n' ' ')" else echo "note: no prebuilt igneum-worker-cuda.exe / igneum-worker-opencl.exe found; the engine builds the CUDA worker from proto-cuda\\ on the PC (CUDA Toolkit and MSVC needed)"; fi diff --git a/packaging/windows/push-inputs.sh b/packaging/windows/push-inputs.sh index 05f515dc..28afa814 100755 --- a/packaging/windows/push-inputs.sh +++ b/packaging/windows/push-inputs.sh @@ -28,6 +28,7 @@ REL="${IGNEUM_WIN_RELEASE:-$ROOT/vendor/igneum-node/target-integration/x86_64-pc MINGW=/opt/homebrew/opt/mingw-w64/toolchain-x86_64/x86_64-w64-mingw32 NVRTC_DIR="$ROOT/proto-cuda/nvrtc" CL_WORKER="$ROOT/proto-opencl/igneum-worker-opencl.exe" +TELEMETRY="$ROOT/proto-opencl/igneum-gpu-telemetry.exe" # AMD power, heat, fans, clocks (5 October 2026) TOKEN_FILE="$HOME/.config/igneum/dl-token" DLSITE="${IGNEUM_DLSITE:-}" [ -n "$DLSITE" ] || { [ -f "$HOME/.config/igneum/dlsite-dir" ] && DLSITE="$(tr -d '[:space:]' < "$HOME/.config/igneum/dlsite-dir")"; } || true @@ -59,6 +60,7 @@ if [ -f "$NVRTC_DIR/igneum-worker-cuda.exe" ]; then ls "$STAGE"/nvrtc64_*_0.dll >/dev/null 2>&1 || echo "warning: igneum-worker-cuda.exe without nvrtc64_*_0.dll (run $NVRTC_DIR/fetch-redist.sh)" else echo "warning: no $NVRTC_DIR/igneum-worker-cuda.exe (run $NVRTC_DIR/build-windows.sh); the app will build the CUDA worker on the PC"; fi [ -f "$CL_WORKER" ] && cp "$CL_WORKER" "$STAGE/" || echo "warning: no $CL_WORKER" +[ -f "$TELEMETRY" ] && cp "$TELEMETRY" "$STAGE/" || echo "warning: no $TELEMETRY (AMD cards show no draw or temperature)" # the signer, built from the app crate (it includes src/manifest.rs and src/inputs.rs, so it signs what the runner verifies) KEY="$HOME/.config/igneum/ota-signing-key" diff --git a/proto-cuda/nvrtc/build-windows.sh b/proto-cuda/nvrtc/build-windows.sh index 278fa41e..709b84c4 100755 --- a/proto-cuda/nvrtc/build-windows.sh +++ b/proto-cuda/nvrtc/build-windows.sh @@ -25,7 +25,7 @@ VERIFY="$ROOT/packaging/windows/resources/verify-exe.py" # The coin icon and the version blocks, as COFF objects the linker takes like any other input [ -f "$ICONS/igneum.ico" ] || { echo "== no $ICONS/igneum.ico, making the icons"; python3 "$ICONS/make-icons.py"; } RES="$(mktemp -d)" -for w in cuda opencl; do +for w in cuda opencl gpu-telemetry; do "$WINDRES" -I "$ICONS" -i "$HERE/igneum-worker-$w.rc" -O coff -o "$RES/igneum-worker-$w.res.o" done @@ -37,11 +37,16 @@ echo "== igneum-worker-opencl.exe" -I "$RED/include" -I "$PLACEHOLDER" -DIGNEUM_KERNEL_PATH='"kernel_bound.cl"' \ -o "$ROOT/proto-opencl/igneum-worker-opencl.exe" "$ROOT/proto-opencl/host.c" "$RES/igneum-worker-opencl.res.o" "$STRIP" "$ROOT/proto-opencl/igneum-worker-opencl.exe" +echo "== igneum-gpu-telemetry.exe (ADLX, SetupAPI, PDH; vendor/adlx is the SDK clone)" +[ -f "$ROOT/vendor/adlx/SDK/Include/ADLX.h" ] || { echo "no vendor/adlx: git clone --depth 1 https://github.com/GPUOpen-LibrariesAndSDKs/ADLX.git $ROOT/vendor/adlx" >&2; exit 1; } +"$CC" -std=gnu99 -O2 -Wall -Wno-unused-parameter -Wno-unused-function -static -I "$ROOT/vendor/adlx/SDK/Include" \ + -o "$ROOT/proto-opencl/igneum-gpu-telemetry.exe" "$ROOT/proto-opencl/gpu-telemetry.c" "$ROOT/vendor/adlx/SDK/ADLXHelper/Windows/C/ADLXHelper.c" "$RES/igneum-worker-gpu-telemetry.res.o" -lsetupapi -lpdh +"$STRIP" "$ROOT/proto-opencl/igneum-gpu-telemetry.exe" rm -rf "$RES" -for exe in "$HERE/igneum-worker-cuda.exe" "$ROOT/proto-opencl/igneum-worker-opencl.exe"; do +for exe in "$HERE/igneum-worker-cuda.exe" "$ROOT/proto-opencl/igneum-worker-opencl.exe" "$ROOT/proto-opencl/igneum-gpu-telemetry.exe"; do printf '%s: %d bytes, imports:' "$(basename "$exe")" "$(stat -f %z "$exe")" x86_64-w64-mingw32-objdump -p "$exe" | sed -n 's/^[[:space:]]*DLL Name: //p' | tr '\n' ' ' echo done # the icon and version block survived the strip (strip keeps .rsrc; this proves it) -python3 "$VERIFY" --version 0.3.0 "$HERE/igneum-worker-cuda.exe" "$ROOT/proto-opencl/igneum-worker-opencl.exe" +python3 "$VERIFY" --version 0.3.0 "$HERE/igneum-worker-cuda.exe" "$ROOT/proto-opencl/igneum-worker-opencl.exe" "$ROOT/proto-opencl/igneum-gpu-telemetry.exe" diff --git a/proto-cuda/nvrtc/igneum-worker-gpu-telemetry.rc b/proto-cuda/nvrtc/igneum-worker-gpu-telemetry.rc new file mode 100644 index 00000000..db056147 --- /dev/null +++ b/proto-cuda/nvrtc/igneum-worker-gpu-telemetry.rc @@ -0,0 +1,35 @@ +// Windows resources for igneum-gpu-telemetry.exe: the coin icon Explorer shows and the version block under Properties > Details. +// Compiled with x86_64-w64-mingw32-windres (the icon path is relative to brand/icons, passed with -I). +// the project lead's rule (4 October 2026): every shipped exe carries the coin icon and a version block, like the Mac app and DMG. +#include + +1 ICON "igneum.ico" + +1 VERSIONINFO +FILEVERSION 0,3,0,0 +PRODUCTVERSION 0,3,0,0 +FILEFLAGSMASK 0x3fL +FILEFLAGS 0x0L +FILEOS VOS_NT_WINDOWS32 +FILETYPE VFT_APP +FILESUBTYPE VFT2_UNKNOWN +BEGIN + BLOCK "StringFileInfo" + BEGIN + BLOCK "040904B0" + BEGIN + VALUE "CompanyName", "Igneum" + VALUE "FileDescription", "Igneum Miner GPU telemetry (AMD power, heat, fans, clocks)" + VALUE "FileVersion", "0.3.0" + VALUE "InternalName", "igneum-gpu-telemetry" + VALUE "LegalCopyright", "Igneum contributors" + VALUE "OriginalFilename", "igneum-gpu-telemetry.exe" + VALUE "ProductName", "Igneum Miner" + VALUE "ProductVersion", "0.3.0" + END + END + BLOCK "VarFileInfo" + BEGIN + VALUE "Translation", 0x409, 1200 + END +END diff --git a/proto-opencl/README.md b/proto-opencl/README.md index fc6a3783..e6e040e2 100644 --- a/proto-opencl/README.md +++ b/proto-opencl/README.md @@ -39,6 +39,9 @@ proto-opencl/ host.c C99 host: device list, runtime kernel build, cache + dataset fill, self-tests, vectors, bench, sweep, --serve, --pack cl_dynamic.h Windows one-click build: OpenCL.dll loaded at run time (IGNEUM_CL_DYNAMIC) test-generic.sh the --pack mode checked here through Apple OpenCL (needs proto-cuda/nvrtc/emu/test.sh's packs) + test_host.c device-free unit tests of host.c's rules (the duplicate-platform fold); run with test-host.sh + gpu-telemetry.c igneum-gpu-telemetry: AMD power, temperature, fan, clocks and busy per card (ADLX on Windows, amdgpu sysfs on Linux), + one line per card per sample; the app's AMD card row reads it (engine.rs amd_telemetry_line) build.sh macOS (-framework OpenCL, or the Khronos ICD loader) and Linux (-lOpenCL) build.bat Windows (MSVC cl.exe + OpenCL.lib) WAVEFRONT.md wave32 vs wave64 on AMD, and why the kernel cannot tell the difference diff --git a/proto-opencl/gpu-telemetry.c b/proto-opencl/gpu-telemetry.c new file mode 100644 index 00000000..86a28be9 --- /dev/null +++ b/proto-opencl/gpu-telemetry.c @@ -0,0 +1,274 @@ +// igneum-gpu-telemetry: power, temperature, fan and clocks of every AMD GPU, one line per card per sample. +// 5 October 2026, after the project lead watched a 9070 XT at 90% usage with its fans barely turning and the app could not say +// what it drew (the app's draw, temperature and MH per watt line came from nvidia-smi only). +// +// igneum-gpu-telemetry [-l SECONDS] one sample (default), or one every SECONDS until stdin closes or SIGTERM +// +// Windows: ADLX (the AMD Device Library eXtra, amdadlx64.dll, shipped with Adrenalin; vendor/adlx is the SDK clone, +// MIT) for the metrics, keyed by the card's PCI bus from SetupAPI (the display class, matched by the same name ADLX +// reports). Without ADLX (no AMD driver, an old one, or the DLL missing) only the utilisation is read, from the +// GPU Engine performance counters through PDH, keyed by the adapter LUID that Windows uses there. +// Linux: the amdgpu sysfs (/sys/class/drm/card*/device: hwmon power1_average, temp1_input, fan1_input, pwm1, +// pp_dpm_mclk, gpu_busy_percent), keyed by the PCI address of the device link. +// +// Line format (space separated, every field present, a value the source cannot give prints as -): +// amd bus kind integrated|discrete name "" watts temp_c fan_rpm +// fan_pct <%> mclk_mhz gclk_mhz util_pct <%> source adlx|sysfs|perfcounter +// then one `end ` line per sample. The app (engine.rs amd_telemetry_line) parses it; parsers are unit-tested +// against lines captured on PC 1. +#define _CRT_SECURE_NO_WARNINGS +#include +#include +#include +#include + +static volatile int gStop = 0; +static void onSignal(int s) { (void)s; gStop = 1; } + +typedef struct { + char bus[64]; + char kind[16]; + char name[128]; + double watts, tempC, fanRpm, fanPct, mclk, gclk, util; /* -1 = not available */ + const char* source; +} Sample; + +static void sampleInit(Sample* s) { memset(s, 0, sizeof(*s)); strcpy(s->bus, "-"); strcpy(s->kind, "-"); strcpy(s->name, "-"); s->watts = s->tempC = s->fanRpm = s->fanPct = s->mclk = s->gclk = s->util = -1.0; s->source = "-"; } +static void printNum(double v, const char* fmt) { if (v < 0) printf(" -"); else printf(fmt, v); } +static void printSample(int ordinal, const Sample* s) { + printf("amd %d bus %s kind %s name \"%s\" watts", ordinal, s->bus, s->kind, s->name); + printNum(s->watts, " %.1f"); printf(" temp_c"); printNum(s->tempC, " %.1f"); printf(" fan_rpm"); printNum(s->fanRpm, " %.0f"); + printf(" fan_pct"); printNum(s->fanPct, " %.0f"); printf(" mclk_mhz"); printNum(s->mclk, " %.0f"); printf(" gclk_mhz"); printNum(s->gclk, " %.0f"); + printf(" util_pct"); printNum(s->util, " %.0f"); printf(" source %s\n", s->source); +} + +#ifdef _WIN32 +#define WIN32_LEAN_AND_MEAN +#include +#include +#include +#include "../vendor/adlx/SDK/ADLXHelper/Windows/C/ADLXHelper.h" +#include "../vendor/adlx/SDK/Include/IPerformanceMonitoring.h" + +/* The SDK declares these three and leaves them to the platform file of each sample. */ +adlx_handle ADLX_CDECL_CALL adlx_load_library(const TCHAR* filename) { return (adlx_handle)LoadLibrary(filename); } +int ADLX_CDECL_CALL adlx_free_library(adlx_handle module) { return FreeLibrary((HMODULE)module) ? 1 : 0; } +void* ADLX_CDECL_CALL adlx_get_proc_address(adlx_handle module, const char* procName) { return (void*)GetProcAddress((HMODULE)module, procName); } + +static double nowMs(void) { LARGE_INTEGER f, c; QueryPerformanceFrequency(&f); QueryPerformanceCounter(&c); return (double)c.QuadPart * 1000.0 / (double)f.QuadPart; } + +/* The PCI bus of every display-class device, by its name (SetupAPI; the names are the ones ADLX reports). */ +typedef struct { char name[128]; int bus; } BusEntry; +static int listBuses(BusEntry* out, int cap) { + static const GUID DISPLAY = { 0x4d36e968, 0xe325, 0x11ce, { 0xbf, 0xc1, 0x08, 0x00, 0x2b, 0xe1, 0x03, 0x18 } }; + HDEVINFO set = SetupDiGetClassDevsA(&DISPLAY, NULL, NULL, DIGCF_PRESENT); + SP_DEVINFO_DATA d; + DWORD i; + int n = 0; + if (set == INVALID_HANDLE_VALUE) return 0; + d.cbSize = sizeof(d); + for (i = 0; SetupDiEnumDeviceInfo(set, i, &d) && n < cap; ++i) { + char name[128] = { 0 }; + DWORD bus = 0, type = 0, got = 0; + if (!SetupDiGetDeviceRegistryPropertyA(set, &d, SPDRP_DEVICEDESC, &type, (BYTE*)name, sizeof(name) - 1, &got)) continue; + if (!SetupDiGetDeviceRegistryPropertyA(set, &d, SPDRP_BUSNUMBER, &type, (BYTE*)&bus, sizeof(bus), &got)) continue; + snprintf(out[n].name, sizeof(out[n].name), "%s", name); out[n].bus = (int)bus; ++n; + } + SetupDiDestroyDeviceInfoList(set); + return n; +} +static int busOf(const BusEntry* b, int n, const char* name, int* taken) { + int i; + for (i = 0; i < n; ++i) if (!taken[i] && strcmp(b[i].name, name) == 0) { taken[i] = 1; return b[i].bus; } + return -1; +} + +/* ADLX: one sample of every GPU. Returns the number of lines printed, -1 when ADLX is not usable (reason printed). */ +static IADLXSystem* gSys = NULL; +static IADLXPerformanceMonitoringServices* gPerf = NULL; +static int adlxOpen(void) { + ADLX_RESULT r = ADLXHelper_Initialize(); + if (!ADLX_SUCCEEDED(r)) { printf("info adlx: ADLXHelper_Initialize returned %d (no AMD driver with ADLX; amdadlx64.dll missing or too old)\n", (int)r); return 0; } + gSys = ADLXHelper_GetSystemServices(); + if (!gSys) { printf("info adlx: no system services\n"); return 0; } + r = gSys->pVtbl->GetPerformanceMonitoringServices(gSys, &gPerf); + if (!ADLX_SUCCEEDED(r) || !gPerf) { printf("info adlx: GetPerformanceMonitoringServices returned %d\n", (int)r); return 0; } + return 1; +} +static int adlxSample(const BusEntry* buses, int nBuses) { + IADLXGPUList* gpus = NULL; + adlx_uint it; + int ordinal = 0; + int taken[32] = { 0 }; + ADLX_RESULT r = gSys->pVtbl->GetGPUs(gSys, &gpus); + if (!ADLX_SUCCEEDED(r) || !gpus) { printf("info adlx: GetGPUs returned %d\n", (int)r); return 0; } + for (it = gpus->pVtbl->Begin(gpus); it != gpus->pVtbl->End(gpus); ++it) { + IADLXGPU* gpu = NULL; + IADLXGPUMetrics* m = NULL; + Sample s; + const char* name = NULL; + ADLX_GPU_TYPE type = GPUTYPE_UNDEFINED; + adlx_double dv = 0; adlx_int iv = 0; + if (!ADLX_SUCCEEDED(gpus->pVtbl->At_GPUList(gpus, it, &gpu)) || !gpu) continue; + sampleInit(&s); + s.source = "adlx"; + if (ADLX_SUCCEEDED(gpu->pVtbl->Name(gpu, &name)) && name) snprintf(s.name, sizeof(s.name), "%s", name); + if (ADLX_SUCCEEDED(gpu->pVtbl->Type(gpu, &type))) strcpy(s.kind, type == GPUTYPE_INTEGRATED ? "integrated" : type == GPUTYPE_DISCRETE ? "discrete" : "-"); + { int b = busOf(buses, nBuses, s.name, taken); if (b >= 0) snprintf(s.bus, sizeof(s.bus), "%d", b); } + r = gPerf->pVtbl->GetCurrentGPUMetrics(gPerf, gpu, &m); + if (ADLX_SUCCEEDED(r) && m) { + if (ADLX_SUCCEEDED(m->pVtbl->GPUPower(m, &dv))) s.watts = dv; + if (s.watts < 0 && ADLX_SUCCEEDED(m->pVtbl->GPUTotalBoardPower(m, &dv))) s.watts = dv; + if (ADLX_SUCCEEDED(m->pVtbl->GPUTemperature(m, &dv))) s.tempC = dv; + if (ADLX_SUCCEEDED(m->pVtbl->GPUFanSpeed(m, &iv))) s.fanRpm = iv; + if (ADLX_SUCCEEDED(m->pVtbl->GPUVRAMClockSpeed(m, &iv))) s.mclk = iv; + if (ADLX_SUCCEEDED(m->pVtbl->GPUClockSpeed(m, &iv))) s.gclk = iv; + if (ADLX_SUCCEEDED(m->pVtbl->GPUUsage(m, &dv))) s.util = dv; + m->pVtbl->Release(m); + } else { + printf("info adlx: GetCurrentGPUMetrics for \"%s\" returned %d\n", s.name, (int)r); + } + /* fan percent: ADLX gives rpm only here; the tuning interface has the range, the app shows rpm when pct is - */ + printSample(ordinal++, &s); + gpu->pVtbl->Release(gpu); + } + gpus->pVtbl->Release(gpus); + return ordinal; +} + +/* PDH fallback: GPU engine utilisation per adapter LUID, summed over the engines (no power, no temperature). */ +static int pdhSample(void) { + PDH_HQUERY q = NULL; + PDH_HCOUNTER c = NULL; + DWORD size = 0, count = 0, i; + PDH_FMT_COUNTERVALUE_ITEM_A* items; + int ordinal = 0; + if (PdhOpenQueryA(NULL, 0, &q) != ERROR_SUCCESS) { printf("info perfcounter: PdhOpenQuery failed\n"); return 0; } + if (PdhAddEnglishCounterA(q, "\\GPU Engine(*)\\Utilization Percentage", 0, &c) != ERROR_SUCCESS) { printf("info perfcounter: no GPU Engine counters\n"); PdhCloseQuery(q); return 0; } + PdhCollectQueryData(q); Sleep(1000); PdhCollectQueryData(q); + PdhGetFormattedCounterArrayA(c, PDH_FMT_DOUBLE, &size, &count, NULL); + items = (PDH_FMT_COUNTERVALUE_ITEM_A*)malloc(size ? size : 1); + if (PdhGetFormattedCounterArrayA(c, PDH_FMT_DOUBLE, &size, &count, items) == ERROR_SUCCESS) { + /* instance names: pid_1234_luid_0x00000000_0x0000D4E3_phys_0_eng_0_engtype_3D; sum per luid */ + char luids[16][40]; double sums[16]; int n = 0, k; + for (i = 0; i < count; ++i) { + const char* p = strstr(items[i].szName, "luid_"); + char luid[40]; + if (!p) continue; + snprintf(luid, sizeof(luid), "%.39s", p); { char* e = strstr(luid, "_phys"); if (e) *e = 0; } + for (k = 0; k < n; ++k) if (strcmp(luids[k], luid) == 0) break; + if (k == n && n < 16) { strcpy(luids[n], luid); sums[n] = 0; ++n; } + if (k < 16) sums[k] += items[i].FmtValue.doubleValue; + } + for (k = 0; k < n; ++k) { + Sample s; sampleInit(&s); s.source = "perfcounter"; + snprintf(s.bus, sizeof(s.bus), "%s", luids[k]); + s.util = sums[k] > 100.0 ? 100.0 : sums[k]; + printSample(ordinal++, &s); + } + } + free(items); + PdhCloseQuery(q); + return ordinal; +} + +int main(int argc, char** argv) { + int every = 0, i, haveAdlx; + BusEntry buses[32]; + int nBuses; + for (i = 1; i < argc; ++i) if (strcmp(argv[i], "-l") == 0 && i + 1 < argc) every = atoi(argv[++i]); + signal(SIGINT, onSignal); signal(SIGTERM, onSignal); + setvbuf(stdout, NULL, _IOLBF, 0); + nBuses = listBuses(buses, 32); + for (i = 0; i < nBuses; ++i) printf("info display device \"%s\" bus %d\n", buses[i].name, buses[i].bus); + haveAdlx = adlxOpen(); + do { + double t0 = nowMs(); + int n = haveAdlx ? adlxSample(buses, nBuses) : pdhSample(); + printf("end %.1f ms %d card(s)\n", nowMs() - t0, n); + fflush(stdout); /* a redirected stdout is fully buffered on the Windows CRT whatever setvbuf asks (PC 1 lost 60 s of samples at the kill) */ + if (every > 0) Sleep((DWORD)every * 1000); + } while (every > 0 && !gStop); + if (haveAdlx) { if (gPerf) gPerf->pVtbl->Release(gPerf); ADLXHelper_Terminate(); } + return 0; +} +#else +#include +#include +#include +static double nowMs(void) { struct timespec ts; clock_gettime(CLOCK_MONOTONIC, &ts); return ts.tv_sec * 1000.0 + ts.tv_nsec / 1e6; } +static int readText(const char* path, char* out, size_t cap) { FILE* f = fopen(path, "r"); size_t n; if (!f) return 0; n = fread(out, 1, cap - 1, f); fclose(f); out[n] = 0; return 1; } +static double readNumber(const char* path) { char b[64]; if (!readText(path, b, sizeof(b))) return -1.0; return atof(b); } +/* pp_dpm_mclk: lines "0: 96Mhz", "3: 1258Mhz *"; the starred line is the current state */ +static double dpmCurrent(const char* text) { + const char* p = text; + while (p && *p) { + const char* nl = strchr(p, '\n'); + size_t len = nl ? (size_t)(nl - p) : strlen(p); + const char* star = memchr(p, '*', len); + if (star) { const char* colon = memchr(p, ':', len); if (colon) return atof(colon + 1); } + p = nl ? nl + 1 : NULL; + } + return -1.0; +} +static int sysfsSample(const char* root) { + DIR* d = opendir(root); + struct dirent* e; + int ordinal = 0; + if (!d) { printf("info sysfs: no %s\n", root); return 0; } + while ((e = readdir(d)) != NULL) { + char dev[512], path[640], text[4096], link[512]; + ssize_t ln; + Sample s; + DIR* hw; struct dirent* he; + if (strncmp(e->d_name, "card", 4) != 0 || strchr(e->d_name + 4, '-')) continue; + snprintf(dev, sizeof(dev), "%s/%s/device", root, e->d_name); + snprintf(path, sizeof(path), "%s/vendor", dev); + if (!readText(path, text, sizeof(text)) || strtol(text, NULL, 16) != 0x1002) continue; + sampleInit(&s); + s.source = "sysfs"; + ln = readlink(dev, link, sizeof(link) - 1); + if (ln > 0) { link[ln] = 0; { const char* base = strrchr(link, '/'); snprintf(s.bus, sizeof(s.bus), "%.63s", base ? base + 1 : link); } } + snprintf(path, sizeof(path), "%s/product_name", dev); + if (readText(path, text, sizeof(text))) { text[strcspn(text, "\n")] = 0; snprintf(s.name, sizeof(s.name), "%s", text); } + else { snprintf(path, sizeof(path), "%s/device", dev); if (readText(path, text, sizeof(text))) { text[strcspn(text, "\n")] = 0; snprintf(s.name, sizeof(s.name), "amdgpu %s", text); } } + snprintf(path, sizeof(path), "%s/boot_vga", dev); + strcpy(s.kind, "discrete"); + snprintf(path, sizeof(path), "%s/hwmon", dev); + hw = opendir(path); + if (hw) { + while ((he = readdir(hw)) != NULL) { + char hp[900]; + if (strncmp(he->d_name, "hwmon", 5) != 0) continue; + snprintf(hp, sizeof(hp), "%s/%s/power1_average", path, he->d_name); s.watts = readNumber(hp); if (s.watts < 0) { snprintf(hp, sizeof(hp), "%s/%s/power1_input", path, he->d_name); s.watts = readNumber(hp); } if (s.watts >= 0) s.watts /= 1e6; + snprintf(hp, sizeof(hp), "%s/%s/temp1_input", path, he->d_name); s.tempC = readNumber(hp); if (s.tempC >= 0) s.tempC /= 1000.0; + snprintf(hp, sizeof(hp), "%s/%s/fan1_input", path, he->d_name); s.fanRpm = readNumber(hp); + { double pwm, pwmMax; snprintf(hp, sizeof(hp), "%s/%s/pwm1", path, he->d_name); pwm = readNumber(hp); snprintf(hp, sizeof(hp), "%s/%s/pwm1_max", path, he->d_name); pwmMax = readNumber(hp); if (pwm >= 0 && pwmMax > 0) s.fanPct = 100.0 * pwm / pwmMax; else if (pwm >= 0) s.fanPct = 100.0 * pwm / 255.0; } + break; + } + closedir(hw); + } + snprintf(path, sizeof(path), "%s/pp_dpm_mclk", dev); if (readText(path, text, sizeof(text))) s.mclk = dpmCurrent(text); + snprintf(path, sizeof(path), "%s/pp_dpm_sclk", dev); if (readText(path, text, sizeof(text))) s.gclk = dpmCurrent(text); + snprintf(path, sizeof(path), "%s/gpu_busy_percent", dev); s.util = readNumber(path); + printSample(ordinal++, &s); + } + closedir(d); + return ordinal; +} +int main(int argc, char** argv) { + int every = 0, i; + const char* root = getenv("IGNEUM_DRM_ROOT") ? getenv("IGNEUM_DRM_ROOT") : "/sys/class/drm"; /* a fixture tree for tests */ + for (i = 1; i < argc; ++i) if (strcmp(argv[i], "-l") == 0 && i + 1 < argc) every = atoi(argv[++i]); + signal(SIGINT, onSignal); signal(SIGTERM, onSignal); + setvbuf(stdout, NULL, _IOLBF, 0); + do { + double t0 = nowMs(); + int n = sysfsSample(root); + printf("end %.1f ms %d card(s)\n", nowMs() - t0, n); + fflush(stdout); /* a redirected stdout is fully buffered on the Windows CRT whatever setvbuf asks (PC 1 lost 60 s of samples at the kill) */ + if (every > 0) sleep((unsigned)every); + } while (every > 0 && !gStop); + return 0; +} +#endif From 192aa3b83cf438a53a82deee84e283abf9cf9973 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Mon, 5 Oct 2026 19:25:12 +0000 Subject: [PATCH 02/20] app: Power control setting (default off): the app never asks for administrator rights on its own the project lead, 5 October 2026: "if we don't have to ask then don't ask". The NVIDIA power cap and the efficiency sweep need administrator rights (one UAC prompt); PC 1 raised that prompt for cmd.exe at every app start and every sweep attempt (17:00, 17:30, 18:12, 19:04 UTC today, each cancelled unanswered after 2 minutes; the "Windows Command Processor" the project lead saw). - config.rs: `power_control` (default OFF on every machine); `sweep` default becomes off and is implied by it (an install carrying sweep = true without power_control is migrated to off on load). - engine.rs: `elevation_allowed(power_control, sweep_only)` gates the power cap (`power_cap_plan` builds nothing when off, the card note says so), the sweep scheduler, Sweep now, the sweep helper; no prompt on quit (the limits reset at the next reboot); no second prompt through PowerShell when the window host's prompt goes unanswered. Cmd::PowerControl(on): on = ONE prompt at that moment (every NVIDIA cap in one step), off = nothing asks; `power_control_after_prompt` turns a refused, cancelled or unanswered prompt into "power control off: administrator rights were not given" (switch back off, sweep off, no retries). Unit tests: off builds no elevated command; on + refusal gives the notice; rights given keeps it on. - platform.rs: `elevated_failure` maps the launcher's exit 251 and the "canceled" wording to the prompt, any other code to the step itself. - server.rs: POST /api/power/control {on}. ui: the Power control switch with the line "Windows asks for administrator rights once; the cap and the sweep need them", the note beside it, the sweep switch disabled while it is off. - The clock-sync prompt stays behind the Sync clock button only (unchanged). - tools/windows/power-prompts-off.ps1: the 0.3.9 job that switched PC 1's sweep off through the API it has (run-20261005-192313: sweep True -> False; the 0.3.9 cap has no off switch, it asks at an app start only). Co-Authored-By: Claude Fable 5.1 --- app/igneum-app/src/config.rs | 23 +++- app/igneum-app/src/engine.rs | 194 +++++++++++++++++++++++----- app/igneum-app/src/platform.rs | 76 ++++++++++- app/igneum-app/src/server.rs | 6 + app/igneum-app/src/state.rs | 7 +- app/igneum-app/ui/app.js | 5 + app/igneum-app/ui/index.html | 5 + tools/windows/power-prompts-off.ps1 | 28 ++++ 8 files changed, 306 insertions(+), 38 deletions(-) create mode 100644 tools/windows/power-prompts-off.ps1 diff --git a/app/igneum-app/src/config.rs b/app/igneum-app/src/config.rs index 5bdcfaf1..050b79bb 100644 --- a/app/igneum-app/src/config.rs +++ b/app/igneum-app/src/config.rs @@ -70,9 +70,16 @@ pub struct Settings { #[serde(default)] pub prove: bool, /// The efficiency sweep (src/sweep.rs): once after install, then weekly, each NVIDIA card's cap is stepped from - /// 100% to 50% on the live program and held at the best MH per watt. Default on. A pinned card is skipped. - #[serde(default = "yes")] + /// 100% to 50% on the live program and held at the best MH per watt. Default off; implied by `power_control` + /// (on when that is switched on, never effective while it is off). A pinned card is skipped. + #[serde(default)] pub sweep: bool, + /// Power control (the project lead, 5 October 2026: "if we don't have to ask then don't ask"): the NVIDIA power cap and the + /// efficiency sweep need administrator rights (one UAC prompt on Windows). Default OFF on every machine; the app + /// never raises the prompt on its own. Switching it on asks once, at that moment; a refused, cancelled or + /// unanswered prompt switches it back off with a notice, no retries. + #[serde(default)] + pub power_control: bool, /// When this install first ran (unix s), for the "first hour after install" sweep. #[serde(default)] pub installed_at: u64, @@ -98,16 +105,26 @@ fn yes() -> bool { impl Default for Settings { fn default() -> Settings { - Settings { setup_done: false, address: String::new(), address_source: String::new(), key_saved: false, identities: 1, cards: HashMap::new(), display_name: String::new(), vote: true, paused: false, accepted_total: 0, auto_update: true, remote_jobs: true, prove: false, sweep: true, installed_at: 0, dev_fee: true, fee_total: 0, proof_verify_trust: false } + Settings { setup_done: false, address: String::new(), address_source: String::new(), key_saved: false, identities: 1, cards: HashMap::new(), display_name: String::new(), vote: true, paused: false, accepted_total: 0, auto_update: true, remote_jobs: true, prove: false, sweep: false, power_control: false, installed_at: 0, dev_fee: true, fee_total: 0, proof_verify_trust: false } } } impl Settings { pub fn load(path: &Path) -> Settings { let mut s: Settings = std::fs::read_to_string(path).ok().and_then(|t| serde_json::from_str(&t).ok()).unwrap_or_default(); + let mut dirty = false; if s.installed_at == 0 { // an install from before the sweep existed counts as installed now: it gets its first-hour sweep s.installed_at = crate::platform::unix_now(); + dirty = true; + } + if s.sweep && !s.power_control { + // the sweep is implied by power control (5 October 2026): an install from before that setting carried + // sweep = true by default; it no longer prompts on its own + s.sweep = false; + dirty = true; + } + if dirty { s.save(path); } s diff --git a/app/igneum-app/src/engine.rs b/app/igneum-app/src/engine.rs index 9224c4a8..d42d07e1 100644 --- a/app/igneum-app/src/engine.rs +++ b/app/igneum-app/src/engine.rs @@ -63,6 +63,8 @@ pub enum Cmd { SweepStop, SweepPin(String, bool), SweepEnable(bool), + /// Settings > Power control (config.rs power_control): on asks for administrator rights once, at that moment + PowerControl(bool), /// the probe that decides how caps are set during a sweep: Ok(true) = nvidia-smi -pl works from this process /// (the engine runs elevated), Ok(false) = the elevated helper is needed; Err = nvidia-smi did not answer SweepCapMode(Result), @@ -113,7 +115,7 @@ impl Shared { st.mining.accepted_total = settings.accepted_total; st.mining.fee_total = settings.fee_total; st.address = address_state(&settings, &wallet_path); - st.settings = crate::state::SettingsState { identities: settings.identities, vote: settings.vote, start_at_login: crate::platform::start_at_login_is_on(), auto_update: settings.auto_update, remote_jobs: settings.remote_jobs, prove: settings.prove, sweep: settings.sweep, dev_fee: settings.dev_fee, proof_verify_trust: settings.proof_verify_trust }; + st.settings = crate::state::SettingsState { identities: settings.identities, vote: settings.vote, start_at_login: crate::platform::start_at_login_is_on(), auto_update: settings.auto_update, remote_jobs: settings.remote_jobs, prove: settings.prove, sweep: settings.sweep, power_control: settings.power_control, power_note: String::new(), dev_fee: settings.dev_fee, proof_verify_trust: settings.proof_verify_trust }; st.dev_fee = crate::state::DevFeeState { on: settings.dev_fee, percent: if settings.dev_fee { 1 } else { 0 }, address: String::new(), line: String::new() }; st.live_page = packaged.live_page.clone(); st.finality.message = "waiting for the miner".into(); @@ -884,6 +886,11 @@ impl Engine { if applied.is_empty() && missing.is_empty() { self.shared.log(&format!("power cap: nothing to read back for {what}")); } + match power_control_after_prompt(&r) { + Some(note) => self.power_control_off(note), + None if !applied.is_empty() => self.st().settings.power_note = "administrator rights given; the cap and the sweep run".into(), + None => {} + } } Cmd::ElevatedDone(r) => { if let Some((line, what, want, _)) = self.power_via_host.take() { @@ -907,6 +914,7 @@ impl Engine { match st.mining.cards.iter().find(|c| c.key == key) { Some(c) if !c.sweep_supported => (c.name.clone(), false, c.sweep_note.clone()), Some(c) if !c.enabled => (c.name.clone(), false, "the card is switched off".to_string()), + Some(c) if !self.elevation_allowed() => (c.name.clone(), false, "switch Power control on in Settings first (Windows asks for administrator rights once)".to_string()), Some(c) => (c.name.clone(), true, String::new()), None => (key.clone(), false, "no such card".to_string()), } @@ -942,10 +950,38 @@ impl Engine { self.shared.event("info", &if pinned { format!("{name}: cap pinned; the sweep records but does not change it") } else { format!("{name}: the sweep chooses the cap again from its next run") }); } Cmd::SweepEnable(on) => { + let allowed = self.elevation_allowed(); + let on = on && allowed; self.shared.settings.lock().unwrap().sweep = on; self.shared.save_settings(); self.st().settings.sweep = on; - self.shared.event("info", if on { "efficiency sweep on: once after install, then weekly" } else { "efficiency sweep off; Sweep now on a card still runs one" }); + self.shared.event("info", if on { "efficiency sweep on: once after install, then weekly" } else if allowed { "efficiency sweep off; Sweep now on a card still runs one" } else { "efficiency sweep stays off: switch Power control on in Settings first" }); + } + Cmd::PowerControl(on) => { + { + let mut s = self.shared.settings.lock().unwrap(); + s.power_control = on; + s.sweep = on; // implied + } + self.shared.save_settings(); + { + let mut st = self.st(); + st.settings.power_control = on; + st.settings.sweep = on; + st.settings.power_note = if on { "asking Windows for administrator rights once".into() } else { String::new() }; + } + if on { + self.shared.event("info", "power control on: Windows asks for administrator rights once; the cap and the sweep need them"); + for c in self.st().mining.cards.iter_mut().filter(|c| c.vendor == "nvidia") { + c.power_applied = false; // the one prompt sets every cap now + } + self.apply_power_limits("power control on"); + if !self.power_busy { + self.st().settings.power_note = "on; no NVIDIA card needs a cap right now".into(); + } + } else { + self.power_control_off("power control off; the cap and the sweep do not run and nothing asks for administrator rights"); + } } Cmd::SweepCapMode(r) => self.sweep_mode_known(r), Cmd::SweepHelperDone(r) => { @@ -1422,21 +1458,10 @@ impl Engine { self.shared.log(&format!("power cap ({why}): deferred, a sweep is running")); return; } - let mut cmds = Vec::new(); - let mut what = Vec::new(); - { - let mut st = self.st(); - for c in st.mining.cards.iter_mut().filter(|c| c.vendor == "nvidia" && c.enabled && c.power_default_w > 0.0) { - let pct = if c.power_pct == 0 { 80 } else { c.power_pct.clamp(crate::sweep::MIN_PCT, 100) }; - c.power_pct = pct; - let watts = requested_watts(c); - if (c.power_limit_w - watts).abs() < 1.0 && c.power_applied { - continue; - } - cmds.push(format!("\"{}\" -i {} -pl {}", crate::platform::tool("nvidia-smi").display(), c.device, watts as u64)); - what.push(format!("{} {} W ({}% of {} W)", c.name, watts as u64, pct, c.power_default_w as u64)); - c.power_note = "setting the power cap (administrator prompt)".into(); - } + let allowed = self.elevation_allowed(); + let (cmds, what, held) = power_cap_plan(&mut self.st().mining.cards, allowed, &crate::platform::tool("nvidia-smi").display().to_string()); + if held > 0 { + self.shared.log(&format!("power cap ({why}): not asked, Power control is off in Settings ({held} card(s) would need it)")); } if cmds.is_empty() { return; @@ -1490,11 +1515,36 @@ impl Engine { if cmds.is_empty() { return; } - self.shared.log(&format!("restoring the GPU power limits: {}", cmds.join(" & "))); - match crate::platform::run_elevated(&cmds.join(" & ")) { - Ok(()) => self.shared.log("GPU power limits restored"), - Err(e) => self.shared.log(&format!("GPU power limits not restored ({e}); they reset at the next reboot")), + // no administrator prompt on quit (5 October 2026: the app never asks on its own); the limits reset at the + // next reboot, and the next start with Power control on sets them again + self.shared.log(&format!("GPU power limits left as set (they reset at the next reboot; no prompt on quit): {}", cmds.join(" & "))); + } + + /// Power control (config.rs power_control): may the engine ask for administrator rights for the cap or the sweep? + /// The elevated PC sweep job (--sweep) sets caps directly and counts as allowed. + fn elevation_allowed(&self) -> bool { + elevation_allowed(self.shared.settings.lock().unwrap().power_control, self.shared.runtime.sweep_only) + } + + /// Power control off, with the reason beside the switch and in the feed; a running or queued sweep ends. + fn power_control_off(&mut self, note: &str) { + { + let mut s = self.shared.settings.lock().unwrap(); + s.power_control = false; + s.sweep = false; } + self.shared.save_settings(); + { + let mut st = self.st(); + st.settings.power_control = false; + st.settings.sweep = false; + st.settings.power_note = note.to_string(); + } + self.sweep_queue.clear(); + if self.sweep.is_some() || self.sweep_pending.is_some() { + self.sweep_abort("power control is off"); + } + self.shared.event(if note.starts_with("power control off:") { "error" } else { "info" }, note); } /// nvidia-smi -l 5: power draw, GPU and memory temperature, the limit in force, every 5 s, as a child whose @@ -1524,16 +1574,10 @@ impl Engine { } if let Some((line, what, want, since)) = self.power_via_host.clone() { if now.duration_since(since) > Duration::from_secs(150) { + // no second prompt through PowerShell (5 October 2026): an unanswered prompt is a refusal self.power_via_host = None; - self.shared.log("the window host did not answer the elevated step in 150 s; running it through PowerShell"); - let shared = self.shared.clone(); - std::thread::spawn(move || { - let r = crate::platform::run_elevated(&line); - std::thread::sleep(Duration::from_millis(800)); - let back: std::collections::HashMap = crate::detect::nvidia_power_limits().into_iter().map(|(k, v)| (k, v.1)).collect(); - let _ = want; - shared.send(Cmd::PowerApplied(what, r, back)); - }); + self.shared.log(&format!("the window host did not answer the elevated step in 150 s: {line}")); + self.finish_power(what, Err("the administrator prompt was not answered in 150 s".into()), want); } } } @@ -1693,12 +1737,12 @@ impl Engine { if self.quitting || !self.running || self.power_busy || self.job_hold || self.jobs.holds_miners() { return; } - let (auto_on, paused) = { + let (auto_on, paused, allowed) = { let s = self.shared.settings.lock().unwrap(); let st = self.st(); - (s.sweep, st.mining.paused) + (s.sweep, st.mining.paused, elevation_allowed(s.power_control, self.shared.runtime.sweep_only)) }; - if paused { + if paused || !allowed { return; } let unix = crate::platform::unix_now(); @@ -1829,6 +1873,9 @@ impl Engine { /// The elevated helper (src/sweep.rs helper_script_*): one administrator prompt; it polls /sweep/cmd.txt. fn sweep_helper_start(&mut self, c: &CardState) -> Result<(), String> { + if !self.elevation_allowed() { + return Err("Power control is off in Settings".into()); + } let dir = self.sweep_dir(); std::fs::create_dir_all(&dir).map_err(|e| e.to_string())?; let _ = std::fs::write(dir.join("cmd.txt"), ""); @@ -3072,6 +3119,51 @@ impl Engine { } /// The watts a card's cap asks for: power_pct of the default limit, inside the card's min and max. +/// the project lead, 5 October 2026: "if we don't have to ask then don't ask". The NVIDIA power cap and the efficiency sweep need +/// administrator rights (one UAC prompt on Windows, pkexec on Linux); the engine builds an elevated command only when +/// Power control is on in Settings, or when it is itself the elevated PC sweep job (--sweep). +fn elevation_allowed(power_control: bool, sweep_only: bool) -> bool { + power_control || sweep_only +} + +/// The notice when the one prompt was refused, cancelled or not answered: Power control goes back off, no retries. +pub const POWER_CONTROL_REFUSED: &str = "power control off: administrator rights were not given"; + +/// After the elevated step: Some(notice) when the administrator prompt was refused, cancelled or timed out (the +/// words platform::run_elevated and the window host use), None when rights were given, even if a card then +/// disagreed with the readback. +fn power_control_after_prompt(r: &Result<(), String>) -> Option<&'static str> { + match r { + Err(e) if e.contains("administrator prompt") || e.contains("refused") || e.contains("cancel") => Some(POWER_CONTROL_REFUSED), + _ => None, + } +} + +/// The nvidia-smi -pl lines one elevated step runs, the human list of what they set, and how many cards were held +/// back because Power control is off (their note says so). Nothing is built when not allowed. +fn power_cap_plan(cards: &mut [CardState], allowed: bool, smi: &str) -> (Vec, Vec, usize) { + let mut cmds = Vec::new(); + let mut what = Vec::new(); + let mut held = 0; + for c in cards.iter_mut().filter(|c| c.vendor == "nvidia" && c.enabled && c.power_default_w > 0.0) { + let pct = if c.power_pct == 0 { 80 } else { c.power_pct.clamp(crate::sweep::MIN_PCT, 100) }; + c.power_pct = pct; + let watts = requested_watts(c); + if (c.power_limit_w - watts).abs() < 1.0 && c.power_applied { + continue; + } + if !allowed { + c.power_note = "power cap not set: Power control is off in Settings".into(); + held += 1; + continue; + } + cmds.push(format!("\"{smi}\" -i {} -pl {}", c.device, watts as u64)); + what.push(format!("{} {} W ({}% of {} W)", c.name, watts as u64, pct, c.power_default_w as u64)); + c.power_note = "setting the power cap (administrator prompt)".into(); + } + (cmds, what, held) +} + fn requested_watts(c: &CardState) -> f64 { let pct = if c.power_pct == 0 { 80 } else { c.power_pct.clamp(crate::sweep::MIN_PCT, 100) }; let mut w = c.power_default_w * pct as f64 / 100.0; @@ -3178,6 +3270,42 @@ pub fn parse_race(body: &str) -> Option { #[cfg(test)] mod tests { + #[test] + fn power_control_off_builds_no_elevated_command() { + // the decision (the project lead, 5 October 2026): off = the app never asks; the elevated PC sweep job is the exception + assert!(!super::elevation_allowed(false, false)); + assert!(super::elevation_allowed(true, false)); + assert!(super::elevation_allowed(false, true)); + let mut cards = vec![ + super::CardState { vendor: "nvidia".into(), enabled: true, device: "0".into(), name: "RTX 5090".into(), power_default_w: 575.0, power_limit_w: 575.0, power_pct: 80, ..Default::default() }, + super::CardState { vendor: "amd".into(), enabled: true, device: "1".into(), name: "RX 9070 XT".into(), power_default_w: 300.0, power_limit_w: 300.0, power_pct: 80, ..Default::default() }, + ]; + let (cmds, what, held) = super::power_cap_plan(&mut cards, false, "nvidia-smi"); + assert!(cmds.is_empty() && what.is_empty(), "{cmds:?}"); + assert_eq!(held, 1); + assert_eq!(cards[0].power_note, "power cap not set: Power control is off in Settings"); + let (cmds, what, held) = super::power_cap_plan(&mut cards, true, "nvidia-smi"); + assert_eq!(cmds, vec!["\"nvidia-smi\" -i 0 -pl 460".to_string()]); + assert_eq!(what, vec!["RTX 5090 460 W (80% of 575 W)".to_string()]); + assert_eq!(held, 0); + // a cap already in force asks for nothing either way + cards[0].power_limit_w = 460.0; + cards[0].power_applied = true; + assert!(super::power_cap_plan(&mut cards, true, "nvidia-smi").0.is_empty()); + } + + #[test] + fn a_refused_prompt_switches_power_control_off_with_the_notice() { + let refused = Err("the administrator prompt was refused, cancelled or timed out (exit 251)".to_string()); + assert_eq!(super::power_control_after_prompt(&refused), Some("power control off: administrator rights were not given")); + assert_eq!(super::power_control_after_prompt(&Err("the administrator prompt was cancelled".into())), Some(super::POWER_CONTROL_REFUSED)); + assert_eq!(super::power_control_after_prompt(&Err("elevated fail: the administrator prompt was cancelled".into())), Some(super::POWER_CONTROL_REFUSED)); + assert_eq!(super::power_control_after_prompt(&Err("the administrator prompt was not answered in 150 s".into())), Some(super::POWER_CONTROL_REFUSED)); + // rights given: the step ran, whatever the card then said + assert_eq!(super::power_control_after_prompt(&Ok(())), None); + assert_eq!(super::power_control_after_prompt(&Err("the elevated step exited with code 2".into())), None); + } + use super::{parse_race, sync_decision, Reading}; #[test] diff --git a/app/igneum-app/src/platform.rs b/app/igneum-app/src/platform.rs index 2534fa1e..d41c26d5 100644 --- a/app/igneum-app/src/platform.rs +++ b/app/igneum-app/src/platform.rs @@ -436,7 +436,7 @@ pub fn run_elevated(cmdline: &str) -> Result<(), String> { Ok(()) } else { let err = String::from_utf8_lossy(&out.stderr).trim().to_string(); - Err(if err.contains("canceled") || err.contains("cancelled") || err.is_empty() { "the administrator prompt was cancelled".into() } else { err }) + Err(elevated_failure(out.status.code(), &err)) } } #[cfg(target_os = "linux")] @@ -451,6 +451,52 @@ pub fn run_elevated(cmdline: &str) -> Result<(), String> { } } +/// The reason an elevated step failed, from the launcher's exit code and stderr: exit 251 (the prompt refused, +/// cancelled or timed out, `elevated_ps_line`) and the "canceled" wording name the prompt; any other code is the +/// step's own exit (the engine then keeps Power control on: rights were given). +pub fn elevated_failure(code: Option, stderr: &str) -> String { + if code == Some(ELEVATED_LAUNCH_FAILED) || stderr.contains("canceled") || stderr.contains("cancelled") { + "the administrator prompt was refused, cancelled or timed out".into() + } else if stderr.is_empty() { + format!("the elevated step exited with code {}", code.map(|c| c.to_string()).unwrap_or_else(|| "?".into())) + } else { + stderr.to_string() + } +} + +/// Doubles the single quotes of `s` for a single-quoted PowerShell literal. +pub fn ps_quote(s: &str) -> String { + s.replace('\'', "''") +} + +/// The PowerShell line that starts `file args` as administrator (one UAC prompt), waits, and exits with the child's +/// code. Every elevated launch of the app goes through here (the NVIDIA power cap, the sweep helper, the clock sync, +/// an elevated remote job) so the console flags live in one place: `-WindowStyle Hidden` is SW_HIDE on the new +/// process the AppInfo service creates; the elevated child cannot inherit this process's headless console, so without +/// it the child gets a console of its own (5 October 2026, PC 1 watcher, tools/windows/console-watch*.ps1). +/// A refused, cancelled or unanswered prompt makes Start-Process throw and `$p` stay null: that is exit 251 with the +/// reason on stderr, never `exit $p.ExitCode` = 0 (the 5 October 2026 driver job on PC 1 was reported done after +/// Windows cancelled its prompt at 122 s). +pub fn elevated_ps_line(file: &str, args: &str) -> String { + format!( + "try {{ $p = Start-Process -FilePath '{}' -ArgumentList '{}' -Verb RunAs -Wait -WindowStyle Hidden -PassThru -ErrorAction Stop }} catch {{ Write-Error ('elevated launch failed (UAC refused, cancelled or timed out): ' + $_.Exception.Message); exit 251 }}; if ($null -eq $p) {{ Write-Error 'elevated launch failed: no process'; exit 251 }}; exit $p.ExitCode", + ps_quote(file), + ps_quote(args) + ) +} + +/// The exit code `elevated_ps_line` uses when the elevated process never started (the prompt refused, cancelled or +/// timed out). +pub const ELEVATED_LAUNCH_FAILED: i32 = 251; + +/// The hidden PowerShell that runs `elevated_ps_line(file, args)`: blocking when run, one UAC prompt on the PC. +pub fn elevated_command(file: &str, args: &str) -> Command { + let mut c = Command::new(tool("powershell")); + c.args(["-NoProfile", "-ExecutionPolicy", "Bypass", "-Command", &elevated_ps_line(file, args)]); + quiet(&mut c); + c +} + /// Builds a command that runs without a console window on Windows. pub fn quiet(cmd: &mut Command) -> &mut Command { #[cfg(windows)] @@ -463,6 +509,34 @@ pub fn quiet(cmd: &mut Command) -> &mut Command { #[cfg(test)] mod tests { + #[test] + fn elevated_line_is_hidden_and_quoted() { + let l = super::elevated_ps_line(r"C:\WINDOWS\system32\cmd.exe", "/c echo it's & exit 3"); + assert!(l.starts_with("try { $p = "), "{l}"); + assert!(l.contains("-FilePath 'C:\\WINDOWS\\system32\\cmd.exe' -ArgumentList '/c echo it''s & exit 3' -Verb RunAs -Wait -WindowStyle Hidden -PassThru -ErrorAction Stop } catch {"), "{l}"); + assert!(l.contains("-Verb RunAs"), "{l}"); + assert!(l.contains("-WindowStyle Hidden"), "{l}"); + // a thrown Start-Process (the prompt refused) never falls through to `exit $p.ExitCode` + assert!(l.contains("exit 251 }; if ($null -eq $p) { Write-Error 'elevated launch failed: no process'; exit 251 }; exit $p.ExitCode"), "{l}"); + assert!(l.ends_with("exit $p.ExitCode"), "{l}"); + assert_eq!(super::ELEVATED_LAUNCH_FAILED, 251); + assert_eq!(super::elevated_failure(Some(251), "elevated launch failed (UAC refused, cancelled or timed out): ..."), "the administrator prompt was refused, cancelled or timed out"); + assert_eq!(super::elevated_failure(Some(1), "The operation was canceled by the user."), "the administrator prompt was refused, cancelled or timed out"); + assert_eq!(super::elevated_failure(Some(2), ""), "the elevated step exited with code 2"); + assert_eq!(super::elevated_failure(Some(3), "nvidia-smi: bad"), "nvidia-smi: bad"); + assert_eq!(super::ps_quote("a'b''c"), "a''b''''c"); + assert_eq!(super::ps_quote("plain"), "plain"); + } + + #[test] + fn elevated_command_is_a_hidden_powershell() { + let c = super::elevated_command("powershell.exe", "-NoProfile -File \"C:\\x y\\elevated.ps1\""); + let args: Vec = c.get_args().map(|a| a.to_string_lossy().into_owned()).collect(); + assert_eq!(&args[..4], ["-NoProfile", "-ExecutionPolicy", "Bypass", "-Command"]); + assert!(args[4].contains("-ArgumentList '-NoProfile -File \"C:\\x y\\elevated.ps1\"' -Verb RunAs -Wait -WindowStyle Hidden"), "{}", args[4]); + assert!(c.get_program().to_string_lossy().contains("powershell")); + } + #[test] fn token_redaction() { let l = "dashboard at http://127.0.0.1:58776/t/a3a01c537130bceeaa1f6118ba48d63e/ (log x)"; diff --git a/app/igneum-app/src/server.rs b/app/igneum-app/src/server.rs index 0849d683..e9884e36 100644 --- a/app/igneum-app/src/server.rs +++ b/app/igneum-app/src/server.rs @@ -319,6 +319,12 @@ fn api_post(shared: &Arc, path: &str, body: Value) -> Result { + let on = body.get("on").and_then(|v| v.as_bool()).ok_or("on missing")?; + shared.send(Cmd::PowerControl(on)); + Ok(json!({ "ok": true })) + } "/api/sweep/enable" => { let on = body.get("on").and_then(|v| v.as_bool()).ok_or("on missing")?; shared.send(Cmd::SweepEnable(on)); diff --git a/app/igneum-app/src/state.rs b/app/igneum-app/src/state.rs index 49bd0924..d33029f6 100644 --- a/app/igneum-app/src/state.rs +++ b/app/igneum-app/src/state.rs @@ -210,8 +210,13 @@ pub struct SettingsState { pub remote_jobs: bool, /// the prover service (src/prover.rs) pub prove: bool, - /// the efficiency sweep (src/sweep.rs): once after install, then weekly + /// the efficiency sweep (src/sweep.rs): once after install, then weekly; effective only with `power_control` pub sweep: bool, + /// the NVIDIA power cap and the sweep may ask for administrator rights (config.rs: default off, one prompt when + /// switched on) + pub power_control: bool, + /// the line beside the Power control switch: why it is off, or that the rights were given + pub power_note: String, /// the miner software's dev fee switch (settings; `--dev-fee 0` when off) pub dev_fee: bool, /// devnet only: the node trusts proof records without a verifier (`IGNEUM_PROOF_VERIFY=trust`) diff --git a/app/igneum-app/ui/app.js b/app/igneum-app/ui/app.js index b0dedfb3..7c6e16b2 100644 --- a/app/igneum-app/ui/app.js +++ b/app/igneum-app/ui/app.js @@ -392,6 +392,7 @@ if (typeof document !== 'undefined') (function () { $('s-trust').addEventListener('change', function () { var on = this.checked; api('api/settings', { proof_verify_trust: on }).then(function (r) { if (r.ok) toast(on ? 'Trust mode on (devnet only); the node restarts' : 'Trust mode off; the node restarts'); else { toast(r.error || 'could not change'); $('s-trust').checked = !on; } }); }); $('pv-setup').addEventListener('click', function () { api('api/prove/setup', {}).then(function (r) { toast(r.ok ? 'Setup started in its own window' : (r.error || 'could not start')); }); }); $('s-jobs-allow').addEventListener('change', function () { api('api/jobs/allow', { on: this.checked }); setTimeout(fillSettings, 800); }); + $('s-power-control').addEventListener('change', function () { var on = this.checked; api('api/power/control', { on: on }).then(function (r) { if (r.ok) toast(on ? 'Power control on: Windows asks for administrator rights once' : 'Power control off: nothing asks for administrator rights'); else { toast(r.error || 'could not change'); $('s-power-control').checked = !on; } setTimeout(fillSettings, 1500); }); }); $('s-sweep').addEventListener('change', function () { api('api/sweep/enable', { on: this.checked }).then(function (r) { if (r.ok) toast($('s-sweep').checked ? 'Sweep on: once after install, then weekly' : 'Sweep off'); }); }); $('s-jobs-check').addEventListener('click', function () { api('api/jobs/check', {}); $('s-jobs-note').textContent = 'Checking.'; setTimeout(fillSettings, 4000); }); @@ -1054,7 +1055,11 @@ if (typeof document !== 'undefined') (function () { $('s-jobs-allow').checked = !!j.allowed; $('s-prove').checked = !!(state.settings && state.settings.prove); $('s-trust').checked = !!(state.settings && state.settings.proof_verify_trust); + var pc = !!(state.settings && state.settings.power_control); + $('s-power-control').checked = pc; + $('s-power-note').textContent = (state.settings && state.settings.power_note) || ''; $('s-sweep').checked = !!(state.settings && state.settings.sweep); + $('s-sweep').disabled = !pc; $('s-jobs-key').textContent = j.key_fingerprint ? 'signing key sha256:' + j.key_fingerprint : ''; var parts = []; if (j.account) parts.push(j.account); diff --git a/app/igneum-app/ui/index.html b/app/igneum-app/ui/index.html index ab1f1625..ba54c8ce 100644 --- a/app/igneum-app/ui/index.html +++ b/app/igneum-app/ui/index.html @@ -299,6 +299,11 @@

Only when no verifier is found next to the engine: the node then includes proof records it never checked. Never on a testnet. A found verifier always wins. Changing this restarts the node.

+
+
power control
+ +

Windows asks for administrator rights once; the cap and the sweep need them. Off, the app never asks.

+
efficiency sweep
diff --git a/tools/windows/power-prompts-off.ps1 b/tools/windows/power-prompts-off.ps1 new file mode 100644 index 00000000..119e11d0 --- /dev/null +++ b/tools/windows/power-prompts-off.ps1 @@ -0,0 +1,28 @@ +# Stops the administrator prompts a 0.3.9 Igneum Miner app raises on its own on a PC (the project lead, 5 October 2026: "if we +# don't have to ask then don't ask"), through the settings API that app has, until the build with the Power control +# setting ships: the weekly/first-hour efficiency sweep off (api/sweep/enable), a running sweep stopped +# (api/sweep/stop). The power cap has no off switch in 0.3.9 (it asks at every app start and on a slider change, and +# nowhere else); this script reports the cards' cap state so the next prompt's source is known. A signed `run` job, +# not elevated, a few seconds: +# packaging/ota/publish-jobs.sh add --kind run --target ae432dc7 --timeout-minutes 3 \ +# --script tools/windows/power-prompts-off.ps1 --title "PC 1: sweep off (no administrator prompts)" --deploy +$ErrorActionPreference = 'Continue' +$base = (Get-Content (Join-Path $env:LOCALAPPDATA 'igneum\app\app.url') -Raw).Trim().TrimEnd('/') +function Snapshot([string] $tag) { + $s = Invoke-RestMethod -Uri "$base/api/state" -TimeoutSec 20 + $set = $s.settings + "RESULT $tag settings: sweep " + $set.sweep + " | power_control " + $(if ($null -ne $set.power_control) { $set.power_control } else { '(not in this build)' }) + " | prove " + $set.prove + " | remote_jobs " + $set.remote_jobs + " | version " + $s.version + foreach ($c in $s.mining.cards) { + if ($c.vendor -ne 'nvidia') { continue } + "RESULT $tag card: " + $c.name + " | enabled " + $c.enabled + " | power_pct " + $c.power_pct + " | limit " + $c.power_limit_w + " W of " + $c.power_default_w + " W default | applied " + $c.power_applied + " | pinned " + $c.pinned + " | sweep_state " + $c.sweep_state + " | note: " + $c.power_note + " | " + $c.sweep_note + } +} +Snapshot 'before' +$r = Invoke-RestMethod -Uri "$base/api/sweep/stop" -Method Post -ContentType 'application/json' -Body '{}' -TimeoutSec 30 +"RESULT api/sweep/stop: " + ($r | ConvertTo-Json -Compress) +$r = Invoke-RestMethod -Uri "$base/api/sweep/enable" -Method Post -ContentType 'application/json' -Body '{"on":false}' -TimeoutSec 30 +"RESULT api/sweep/enable off: " + ($r | ConvertTo-Json -Compress) +Start-Sleep -Seconds 3 +Snapshot 'after' +"RESULT note: the 0.3.9 power cap asks only at an app start or a slider change; no setting turns it off before the Power control build" +exit 0 From 5c808b89e0a10865e3a527cba36b3ebb72704cab Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Mon, 5 Oct 2026 21:21:51 +0000 Subject: [PATCH 03/20] Ember Tune: every card tuned for MH per watt out of the box, the fleet prior per card model in the signed manifest, the console and /miners priors table the project lead, 5 October 2026, 22:45 BST: "make sure we have ember tuning every single card for efficiency out of the box, the more data = the better the tune, make an awesome system." Built on lever 3 (docs/plans/miner-eff.md), lever 2's signed tuning section (docs/design/miner-tuning.md), the AMD telemetry helper (423936b, its --tune/--set-gmax/--set-plimit/ --reset contract) and the Power control switch (057f0ec). Design, data flow, tiers and the privacy line: docs/plans/ember-tune.md. - src/ember.rs (new): two knobs per card (power limit %, core clock cap MHz; memory clock never touched), the full plan (power ladder 100..50%, then the clock ladder 90..60% at the chosen power), the confirm plan (the fleet prior and one neighbour), the baseline plan (measure only), the marks (faulted, hot, memory_clock_dropped, unapplied, no_readings), the choice (best MH/W within 1% of the top rate, then rate, then draw), the fleet record (a hash of the install id, no address), the prior lookup and the kill switch (tuning.ember), the state machine on a fake clock. 9 unit tests. - engine.rs: tick_sweep schedules every NVIDIA, AMD and Apple card (120 s steady, 600 s to the boundary, no job hold, no pause, weekly, again after a driver major or program-class change, never under the manifest kill switch); the probe (nvidia-smi clocks.max.gr + driver_version and the direct/helper mode; igneum-gpu-telemetry --tune for AMD); tune_apply (nvidia-smi -pl / -lgc 0, / -rgc directly or through the helper; the AMD helper per request); Cmd::TuneProbe, Cmd::TuneSet; faults from rejected and mismatched hashes mark the step; the TUNE lines and the TUNE {json} record, uploaded with the log; the Tuned line on the card state. The NVIDIA helper starts only with Power control on: the --sweep job never counts as permission (no prompt on a PC with nobody there). - sweep.rs: the helper protocol gains lgc/rgc (clock cap and reset) and resets the clocks after 20 idle minutes. - state.rs, config.rs: the tune fields (clock cap, driver, class, source, the Tuned line); the nvidia-smi telemetry query carries clocks.gr and clocks.mem; the AMD sample line's plimit_pct and gmax_mhz are parsed. - ui: "Tuned: X MH/s at Y W (Z MH/W)" with the point, the source and when; measure-only cards say why; the Ember Tune switch; tune-line.test.mjs. - relay/lib/ember.mjs + relay/test/ember.test.mjs: the aggregation per (card model | driver major | program class): median point, MH/W, spread, samples, machines; five samples converge, an outlier does not move the median, baselines make no prior, de-duplication, the manifest merge keeps lever 2's cards. api/console.mjs fn=tuning and tools/console.mjs tuning; tools/tuning.mjs --priors [--write tuning.json] [--site] [--tuning-off]. - site: the fleet priors table on /miners (site/miner-priors.json), the lever text. - relay/playbooks/ember-tune-pc1.ps1: the PC 1 run (second engine with --sweep from a scratch copy of the install). Measured tonight: see the bench log entry that follows the PC 1 run. The 9070 XT left PC 1's bus at 20:40 UTC and the 5090 needs the administrator prompt the project lead cannot answer asleep, so tonight's PC 1 run is the baseline plan on the 5090 through the whole pipeline; the two-knob tune on both cards is owed. Co-Authored-By: Claude Fable 5.1 --- .github/workflows/ci.yml | 4 +- app/igneum-app/src/config.rs | 10 + app/igneum-app/src/detect.rs | 1 + app/igneum-app/src/ember.rs | 1033 ++++++++++++++++++++++++++ app/igneum-app/src/engine.rs | 646 ++++++++++++---- app/igneum-app/src/main.rs | 1 + app/igneum-app/src/state.rs | 15 + app/igneum-app/src/sweep.rs | 58 +- app/igneum-app/ui/app.js | 55 +- app/igneum-app/ui/index.html | 12 +- app/igneum-app/ui/tune-line.test.mjs | 44 ++ docs/plans/ember-tune.md | 164 ++++ relay/api/console.mjs | 15 + relay/lib/ember.mjs | 131 ++++ relay/playbooks/ember-tune-pc1.ps1 | 143 ++++ relay/test/ember.test.mjs | 110 +++ site/build.mjs | 21 +- site/miner-priors.json | 6 + site/miner.html | 6 +- site/miners.html | 10 +- tools/console.mjs | 7 + tools/tuning.mjs | 46 +- 22 files changed, 2354 insertions(+), 184 deletions(-) create mode 100644 app/igneum-app/src/ember.rs create mode 100644 app/igneum-app/ui/tune-line.test.mjs create mode 100644 docs/plans/ember-tune.md create mode 100644 relay/lib/ember.mjs create mode 100644 relay/playbooks/ember-tune-pc1.ps1 create mode 100644 relay/test/ember.test.mjs create mode 100644 site/miner-priors.json diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 9879a98b..185fa0f4 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -81,6 +81,6 @@ jobs: - name: ship tool self-test (version bump, the dl-both and public manifest helpers) run: node tools/ship-app.mjs --self-test - name: relay unit tests (parsers, secret compare, the wake endpoint) - run: node --test relay/test/parse.test.mjs relay/test/auth.test.mjs relay/test/wake.test.mjs + run: node --test relay/test/parse.test.mjs relay/test/auth.test.mjs relay/test/wake.test.mjs relay/test/ember.test.mjs - name: miner app notice strip and update card (ordering, keys, wording, timers, when the card shows) - run: node --test app/igneum-app/ui/notices.test.mjs app/igneum-app/ui/update-card.test.mjs + run: node --test app/igneum-app/ui/notices.test.mjs app/igneum-app/ui/update-card.test.mjs app/igneum-app/ui/tune-line.test.mjs diff --git a/app/igneum-app/src/config.rs b/app/igneum-app/src/config.rs index 050b79bb..f2b06d7b 100644 --- a/app/igneum-app/src/config.rs +++ b/app/igneum-app/src/config.rs @@ -29,6 +29,16 @@ pub struct CardPref { pub sweep_watts: f64, #[serde(default)] pub sweep_mhs: f64, + /// Ember Tune (src/ember.rs): the clock cap the last tune chose (0 = unlocked), the driver and program class it + /// ran under (a change makes the card due again), and the plan that produced it (full | confirm | baseline) + #[serde(default)] + pub sweep_clock_mhz: u32, + #[serde(default)] + pub sweep_driver: String, + #[serde(default)] + pub sweep_class: String, + #[serde(default)] + pub sweep_source: String, } #[derive(Clone, Serialize, Deserialize)] diff --git a/app/igneum-app/src/detect.rs b/app/igneum-app/src/detect.rs index 020d7a6a..358b7e39 100644 --- a/app/igneum-app/src/detect.rs +++ b/app/igneum-app/src/detect.rs @@ -66,6 +66,7 @@ fn card(index: usize, name: &str, vendor: &str, worker: &str, detail: &str, devi device: device.into(), enabled: true, state: "off".into(), + amd_ordinal: -1, ..Default::default() } } diff --git a/app/igneum-app/src/ember.rs b/app/igneum-app/src/ember.rs new file mode 100644 index 00000000..d839135e --- /dev/null +++ b/app/igneum-app/src/ember.rs @@ -0,0 +1,1033 @@ +//! Ember Tune: every card tuned for MH per watt out of the box, and the fleet's results folded into a prior that a +//! new card starts from (docs/plans/ember-tune.md). Two knobs per card: the power limit (percent of the card's +//! default) and the core clock cap (MHz; 0 = unlocked). The memory clock is never touched, and a step that drags it +//! down is marked and cannot win. This file is the logic, driven by an explicit clock so the tests run without a +//! card: the plans (full, confirm, baseline), the per-step rows with their marks, the choice rule, the fleet record, +//! the prior lookup and the state machine. The engine (src/engine.rs, `tick_tune` and the `Cmd::Tune*` commands) +//! owns the processes: NVIDIA through nvidia-smi (`-pl`, `-lgc 0,`, `-rgc`; administrator rights, so only with +//! Power control on), AMD through igneum-gpu-telemetry (`--set-plimit`, `--set-gmax`, `--reset`; no elevation on +//! Windows), Apple measure only. +//! +//! The lines in the app log (and on stdout under --sweep, which the PC job reads): +//! TUNE start card=
- +
@@ -301,13 +301,13 @@
power control
- -

Windows asks for administrator rights once; the cap and the sweep need them. Off, the app never asks.

+ +

Windows asks for administrator rights once; the NVIDIA cap and the tune need them. Off, the app never asks and NVIDIA cards measure only. AMD cards need no rights.

-
efficiency sweep
- -

On the live program, never restarting the worker: the cap steps from 100% of the card's default limit down to 50%, 15 s to settle and 60 s to measure per step, then holds the step with the most MH per watt. One administrator prompt per sweep. It stops at once if the card faults, a remote job takes the GPU, or the hour boundary is near. A cap you set with the slider is pinned: the sweep records, but leaves it. "Sweep now" on a card's tile runs one at any time; the table is in the log (SWEEP lines).

+
Ember Tune
+ +

On the live program, never restarting the worker: the power limit steps from 100% of the card's default down to 50%, then the core clock from its maximum down to 60%, 15 s to settle and 60 s to measure per step; the memory clock is never touched. The card keeps the point with the best MH per watt within 1% of its top rate. A step with a rejected or mismatched hash, a hot GPU or a dragged memory clock is reverted and marked. A card whose model the fleet already knows (5 or more reports) starts at that prior and confirms it in two steps. Every result goes back to the fleet without anything that identifies you. A cap you set with the slider is pinned: the tune records, but leaves it. "Tune now" on a card's row runs one at any time; the table is in the log (TUNE lines).

version
diff --git a/app/igneum-app/ui/tune-line.test.mjs b/app/igneum-app/ui/tune-line.test.mjs new file mode 100644 index 00000000..17b6daed --- /dev/null +++ b/app/igneum-app/ui/tune-line.test.mjs @@ -0,0 +1,44 @@ +// node --test app/igneum-app/ui/tune-line.test.mjs (no dependencies; CI runs it in the site job) +// The card row's Ember Tune line (app.js TuneLine): what a user sees per state, from the card state fields. +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import { readFileSync } from 'node:fs'; +import { fileURLToPath } from 'node:url'; +import { dirname, join } from 'node:path'; + +const src = readFileSync(join(dirname(fileURLToPath(import.meta.url)), 'app.js'), 'utf8'); +const mod = { exports: {} }; +new Function('module', src)(mod); +const { model, point } = mod.exports.TuneLine; +const NOW = 1_800_000_000; +const card = over => ({ vendor: 'nvidia', sweep_state: 'idle', sweep_note: '', sweep_pct: 0, power_pct: 80, sweep_at: 0, tune_line: '', tune_source: '', tune_clock_mhz: 0, tune_control: true, pinned: false, ...over }); + +test('tuned: the line the brief asks for, with the point, the source and when', () => { + const m = model(card({ tune_line: 'Tuned: 122.3 MH/s at 290 W (0.422 MH/W)', tune_source: 'full', tune_clock_mhz: 2470, sweep_pct: 100, sweep_at: NOW - 3600 }), NOW); + assert.equal(m.kind, 'tuned'); + assert.equal(m.text, 'Tuned: 122.3 MH/s at 290 W (0.422 MH/W)'); + assert.equal(m.note, '2470 MHz at 100%, full tune, 1 h ago'); + const c = model(card({ tune_line: 'Tuned: 17.7 MH/s at 177 W (0.100 MH/W)', tune_source: 'confirm', sweep_pct: 90, sweep_at: NOW - 120, vendor: 'amd' }), NOW); + assert.equal(c.note, '90%, clock unlocked, from the fleet prior, confirmed, 2 min ago'); + const p = model(card({ tune_line: 'Tuned: 1 MH/s at 1 W (1.000 MH/W)', tune_source: 'full', sweep_pct: 70, pinned: true }), NOW); + assert.match(p.note, /your setting stays pinned$/); +}); + +test('measure only: Apple and NVIDIA without Power control say so beside the measured line', () => { + const a = model(card({ vendor: 'apple', tune_control: false, tune_line: 'Tuned: 26.7 MH/s at 38 W (0.703 MH/W)', tune_source: 'baseline', sweep_note: 'measure only on Apple silicon: the system sets the clocks and the power; no control exposed', sweep_at: NOW - 60 }), NOW); + assert.equal(a.kind, 'measured'); + assert.equal(a.text, 'Tuned: 26.7 MH/s at 38 W (0.703 MH/W) (measured as it runs, 1 min ago)'); + assert.match(a.note, /^measure only on Apple silicon/); + const n = model(card({ tune_control: false, tune_line: 'Tuned: 122.3 MH/s at 290 W (0.422 MH/W)', tune_source: 'baseline', sweep_note: 'measure only until Power control is on in Settings (Windows asks for administrator rights once)' }), NOW); + assert.equal(n.kind, 'measured'); + assert.match(n.note, /Power control/); +}); + +test('running, stopped, idle and off', () => { + assert.deepEqual(model(card({ sweep_state: 'running', sweep_note: 'tuning: holding 2472 MHz · 100% · 41 s (step 7 of 9)' }), NOW), { kind: 'running', text: 'tuning: holding 2472 MHz · 100% · 41 s (step 7 of 9)' }); + assert.equal(model(card({ sweep_note: 'tuning stopped: a remote job took the GPU' }), NOW).kind, 'stopped'); + assert.equal(model(card({}), NOW).text, 'tuning: not run yet (starts after 120 s of steady mining)'); + assert.equal(model(card({ sweep_note: 'tuning: waits for 120 s of steady mining' }), NOW).text, 'tuning: waits for 120 s of steady mining'); + assert.equal(model(card({ vendor: 'other' }), NOW).kind, 'off'); + assert.equal(point({ tune_clock_mhz: 0, sweep_pct: 0, power_pct: 0 }), ''); +}); diff --git a/docs/plans/ember-tune.md b/docs/plans/ember-tune.md new file mode 100644 index 00000000..22869c77 --- /dev/null +++ b/docs/plans/ember-tune.md @@ -0,0 +1,164 @@ +# Ember Tune: every card tuned for MH per watt, out of the box + +5 October 2026, night. the project lead: "make sure we have ember tuning every single card for efficiency out of the box, the +more data = the better the tune, make an awesome system." Branch `ember-tune`, worktree `../igneum-wt-ember-tune`, +on top of the AMD telemetry commit (7adcd4c, branch `opencl-rdna4-telemetry`) and the Power control commit (3562f26, +branch `job-console`), both cherry-picked. Lever 3 of docs/plans/miner-eff.md grows two knobs and a fleet memory; +lever 2 (docs/design/miner-tuning.md) carries the priors in the same signed `tuning` section. + +## 1. What a user sees + +| Moment | The card row says | What happened | +|---|---|---| +| First 2 minutes of mining | `tuning: waits for 120 s of steady mining` | The worker warms up; nothing is touched. | +| Tuning | `tuning: holding 2472 MHz · 100% · 41 s (step 7 of 9)` with a Stop button | One card at a time, on the live kernel, never restarting the worker. | +| Tuned | **Tuned: 122.3 MH/s at 290 W (0.422 MH/W)**, then `2470 MHz at 100%, full tune, 1 h ago` | The point is pinned on the card; the result went to the fleet. | +| Known model | the same line, `from the fleet prior, confirmed, 2 min ago` | The card started at its model's prior and confirmed it in two steps instead of nine. | +| Apple silicon | **Tuned: 26.7 MH/s at 38 W (0.703 MH/W)** `(measured as it runs)`, and `measure only on Apple silicon: the system sets the clocks and the power; no control exposed` | Nothing can be set; the number is still reported so the row and the fleet know what the card does. | +| NVIDIA, Power control off | the measured line and `measure only until Power control is on in Settings (Windows asks for administrator rights once)` | The app never raises the prompt by itself (5 October 2026). One switch, one prompt, and the full tune runs. | +| Slider moved | `your setting stays pinned` | A manual point is never overridden; the tune still measures and reports. | +| Stopped | `tuning stopped: a remote job took the GPU` and the card back where it was | Any fault reverts the step and the run. | +| Fleet pause | Settings: `tuning paused fleet-wide by the signed manifest` | The kill switch. | + +Settings: one switch, "Ember Tune: tune every card for MH per watt out of the box (once after install, then weekly, +and after a driver or program change)". AMD needs no rights. NVIDIA needs the Power control switch (one administrator +prompt) for both knobs; off, it measures only. + +## 2. The knobs, per vendor + +| Vendor | Power limit | Core clock cap | Memory clock | How | Rights | +|---|---|---|---|---|---| +| NVIDIA | `nvidia-smi -pl `, percent of the default, inside `power.min_limit` and `power.max_limit` | `nvidia-smi -lgc 0,`, percent of `clocks.max.gr`; `-rgc` = unlocked | never touched (`-lmc` is not used); read back as `clocks.mem` | directly when the engine is elevated, else the one-prompt helper (` pl `, ` lgc `, ` rgc` in `sweep/cmd.txt`) | administrator, so only with Power control on | +| AMD | `igneum-gpu-telemetry --card N --set-plimit ` (0 = default, -20 = 80%), inside the `tune` line's `plimit_range` | `--set-gmax ` inside `gmax_range`; `--reset` for the default point | not settable through ADLX on RDNA 4; read back as `mclk_mhz`, and a step whose mean memory clock falls under 95% of the baseline's is marked and cannot win | the helper, one process per request, exit 0 and a `tune ... ok` line | none on Windows (ADLX manual tuning); root on Linux, so measure only there | +| Apple | none | none | none | measure only | none | + +Vendor limits are never exceeded and the floor is never undercut: the plan clamps every point (`Limits::clamp_clock`, +`Limits::watts_for`), and a clock floor the vendor does not report is 60% of the maximum. + +## 3. The plan and the choice + +Full plan (a new model, or a prior that lost its confirm check): the power ladder 100, 90, 80, 70, 60, 50% at the +unlocked clock (duplicate watts dropped where the card's floor clamps them), then the clock ladder 90, 80, 70, 60% +of the maximum at the power point the power ladder chose. 60 s hold after 15 s settle per step; 9 steps on an +RTX 5090 (five power, four clock), about 12 minutes. + +Confirm plan (the model's prior has 5 or more reports): the prior's point, then one neighbour (the next clock step up +when the prior caps the clock, else one power step down). If the neighbour beats the prior by over 1% MH/W, the full +plan is queued; else the prior stands. Two steps, about 3 minutes. + +Baseline plan (measure only): one step at the card's current point. The "before" number for the row and the fleet. + +The choice (`ember::choose`): among the usable steps whose rate is within the tolerance (1%, settable from the +manifest) of the fastest step, the best MH per watt; within 1% on efficiency the higher rate; within 1% on both the +lower draw. A card never gives up more than the tolerance in blocks for the saving. A step is unusable when it is +marked: `faulted` (a rejected or mismatched hash during the hold: the step is reverted and marked), `hot` (the GPU +reached 85 C; the run aborts at 90), `memory_clock_dropped`, `unapplied` (the readback disagreed with the request), +`no_readings` (under three draw samples or no STATUS line). + +## 4. The data flow + +``` +card mines 120 s ──> probe (limits, driver, how to set) ──> plan ──> steps ──> choice ──> point pinned + │ + app log: TUNE start / TUNE card=.. step=.. / TUNE chosen / TUNE {json} (and stdout under --sweep) + │ + log upload (every minute, the existing intake, site/api/log.mjs) ──> Neon miner_logs + │ + relay/lib/ember.mjs aggregate: per (card model | driver major | program class) + median clock cap (10 MHz), median power %, median MH/W, MH/s, W, spread (MAD %), samples, machines + │ │ + console: /r//c/tuning, `node tools/console.mjs tuning` site: tools/tuning.mjs --priors --site + │ -> site/miner-priors.json -> /miners#priors + tools/tuning.mjs --priors --write tuning.json (priors + ember settings beside the kernel-variant cards) + │ + packaging/ota/publish-manifest.sh --tuning tuning.json --deploy (signed; carried over when not given) + │ + every app: /tuning.json ──> ember::settings_of (kill switch, min samples, tolerance, period) + ──> ember::prior_of(key) ──> a new card's confirm plan +``` + +The record (`ember::record_json`): `ts`, `machine` (the first 8 hex of SHA-256 over the install id; the id itself +is random per install and never sent), `app`, `os`, `card`, `vendor`, `driver`, `driver_major`, `class`, `key`, +`plan`, `steps` (the full table: clock, power %, limit, watts, MH/s, MH/W, core and memory clock, hottest reading, +faults, mark), `chosen`, `before` (the full plan's 100% step), `eff`, `mhs`, `watts`. The key: `||`, the class from the worker's race line (`l128w16` today; `v2` before a +race has run). + +## 5. Scheduling and safety + +| Rule | Where | +|---|---| +| One card at a time; the card must have mined 120 s and have a STATUS line | `tick_sweep` | +| Never under a remote job hold, a pause, inside 600 s of the hour boundary, or while the app quits | `tick_sweep`, `sweep_drive` | +| Due once after install, every 7 days (manifest `ember.period_s`), and when the driver major or the program class changed since the last tune | `tick_sweep` (`CardPref.sweep_driver`, `sweep_class`) | +| A pinned card (the slider) is measured, never changed | `sweep_finish` | +| Kill switch: `tuning.ember.enabled = false` in the signed manifest stops every tune fleet-wide; the Settings line says so | `ember::settings_of`, `tick_sweep` | +| Faults: a rejected or mismatched hash marks the step; the card leaving `mining`, a worker error, a job, a pause or 90 C aborts the run and restores the point from before | `Run::sample_fault`, `sweep_drive`, `sweep_abort` | +| Memory clock held: never set; a step that drags it under 95% of the baseline's cannot win | `Row::from_samples` | +| Vendor limits: every point clamped to the reported range; the clock floor 60% when none is reported | `Limits` | +| No prompt the user did not ask for: the NVIDIA helper starts only with Power control on; the `--sweep` job never counts as permission | `sweep_probe_known`, `sweep_helper_start` | +| The elevated helper restores the limit and resets the clocks by itself after 20 idle minutes | `sweep::helper_script_*` | + +## 6. Tests + +| Test | What it fixes | +|---|---| +| `ember::tests::the_full_plan_is_the_power_ladder_then_the_clock_ladder_at_the_chosen_power` | 5 + 4 steps on the 5090's limits, the clamps, the dynamic second half, the 1% and 5% choices | +| `limits_never_exceed_the_vendor_or_undercut_the_floor` | clamps | +| `the_choice_keeps_the_best_mh_per_watt_within_the_rate_tolerance` | the rule, the ties, marked rows never win | +| `the_guards_mark_a_step_so_it_cannot_win` | faulted, hot, memory clock, unapplied, no readings, the line | +| `a_fault_during_a_step_reverts_it_and_the_run_goes_on` | the state machine with a fake clock: the faulted 70% step is marked and never chosen | +| `the_confirm_plan_checks_the_prior_and_its_neighbour` | the two steps, Keep against FullDue, a prior outside the range clamped | +| `a_baseline_plan_measures_the_card_as_it_runs` | no control, still a number and the Tuned line | +| `the_record_and_the_prior_round_trip_through_the_manifest_shape` | record fields (no address, no host), `priors` and `ember` beside `cards`, the sample floor, the kill switch | +| `control_reasons_per_vendor` | who measures only and why | +| `sweep::tests::helper_scripts_carry_the_protocol` | the helper's `pl`, `lgc`, `rgc` | +| `relay/test/ember.test.mjs` | five samples converge (2,470 MHz at 100%), an outlier (0.908 MH/W at 1,854 MHz) moves nothing, baseline records make no prior, de-duplication, the manifest merge keeps lever 2's cards, the canonical round trip, AMD keys | +| `app/igneum-app/ui/tune-line.test.mjs` | the row line per state | + +Run: `cargo test -p igneum-app ember sweep` (on a PC through the build job, or on the Mac under the build lock), +`node --test relay/test/ember.test.mjs app/igneum-app/ui/tune-line.test.mjs`. + +## 7. The tier consequences + +| Tier | What Ember Tune does for it | What it costs | +|---|---|---| +| A laptop GPU (NVIDIA, 60 to 115 W) | the power ladder usually finds the vendor floor binding; the clock ladder is where a memory-bound program saves watts; the thermal mark keeps a hot chassis from winning a step it cannot hold | about 12 minutes once, then 3 minutes a week; under 1% of the hour during the tune (the worker never stops) | +| One 8 GB card | the same two knobs; the 8 GB card is identities-limited (2 by default), the tune does not change that | the same | +| One 12 or 16 GB card | the same | the same | +| One 24 or 32 GB card (the 5090) | the draw sits far under the cap (290 W under 460 W on PC 1), so the power ladder is flat and the clock ladder is the lever; expected saving from the 4 October stability line: tens of watts at under 1% rate, to be measured | the same | +| A rig (several cards) | one card at a time, so a six-card rig takes about 70 minutes to tune once; every card of one model after the first starts at the prior (3 minutes); the tune never touches a card a remote job holds | linear in cards once, then the confirm plan | +| A pool user | the same per card; a pool submits the same hashes, so the 1% rate tolerance is the same 1% of shares | the same | +| AMD on Linux | measure only (sysfs needs root); the row says so | 60 s a week | +| Apple silicon | measure only; the row says so | 60 s a week | + +Privacy line: what is uploaded is the record in section 4 and nothing else: a hash of the random install id, the +card model, the driver version, the OS, the program class, the step table and the chosen point. No address, no +hostname, no raw machine id, no user name. The public priors table carries only the aggregate per model. + +## 8. Measurements + +### PC 1, 5 October 2026 (night) + +Tonight's constraints, read from PC 1's own uploads: the installed app runs as `DESKTOP-KMCV30N\Admin` with +`elevated=False` (the account line at 19:02:33 UTC), the two in-app sweep attempts at 20:09 UTC aborted on the +cancelled administrator prompt (`SWEEP aborted ... the_elevated_helper_did_not_run_(the_administrator_prompt_was_cancelled)`), +so no stored sweep result exists from today, and the RX 9070 XT left the PCI bus at about 20:40 UTC (eGPU link, +not restarted tonight). NVIDIA's `-pl` and `-lgc` need administrator rights, the project lead is asleep, and the app never raises +the prompt by itself, so tonight's run on PC 1 is the baseline plan on the 5090 through the whole pipeline (probe, +measure, TUNE record, upload, aggregation, prior shape in a test manifest). The two-knob tune on the 5090 and the +9070 XT run are owed: the 5090 the moment Power control is switched on (one prompt, then the tune runs by itself +within 2 minutes of steady mining), the 9070 XT when the card is back on the bus. + +(The run's numbers are appended below when the job reports.) + +## 9. Open + +- The NVIDIA clock readback: `nvidia-smi -lgc` is confirmed only through the core clock during the hold (a mean over + the cap by 5% marks the step `unapplied`); the first run with Power control on tells whether the driver honours + the lock on the 5090 under this kernel. +- ADLX on RDNA 4 exposes no memory-clock setter; the memory-clock mark is the guard. The telemetry agent's 9070 XT + sweep tells whether a core cap drags the memory clock on that card. +- The confirm plan's neighbour is one step; a second neighbour (the other knob) would cost 75 s more and catch a + prior that is wrong on both knobs. +- Intel: no knob yet; the row says measure only. diff --git a/relay/api/console.mjs b/relay/api/console.mjs index 7dcb5358..546e7891 100644 --- a/relay/api/console.mjs +++ b/relay/api/console.mjs @@ -9,10 +9,13 @@ // GET chain igneum.network/api/live trimmed + the Hetzner results item // GET log?limit=&since= work-log items (kinds log, build, note), newest first // GET results bench entries (synced from docs/bench-log.md) + the FUD ledger counts +// GET tuning?days=30&min=5 Ember Tune: the fleet priors per (card model, driver major, program class) from the +// TUNE records in miner_logs (relay/lib/ember.mjs), with the sample counts and MH/W // POST post {kind,title,body,who,key?,meta?} one item; with key it upserts // POST sync {items:[...]} bulk upsert by key import { neon, authed, readJson, str, iso } from '../lib/relay.mjs'; import { kv, kvNum, lastMatch, FAULT, parseLabel, parseMinerTail, parseHeader, parseAppTail, STALE_S, markStale } from '../lib/parse.mjs'; +import { parseRecords, aggregate } from '../lib/ember.mjs'; const json = (res, status, obj) => { res.status(status).setHeader('Content-Type', 'application/json; charset=utf-8'); res.end(JSON.stringify(obj)); }; const CACHE_MS = 10_000; @@ -212,6 +215,13 @@ async function chain(sql) { hetzner: het.length ? itemOut(het[0]) : null, }; } +/// Ember Tune: every TUNE record of the window, folded into priors (the same aggregation the publisher uses). +async function tuning(sql, days, min) { + const rows = await sql(`SELECT lines FROM miner_logs WHERE received_at > now() - ($1 || ' days')::interval AND lines LIKE '%TUNE {%' ORDER BY received_at DESC LIMIT 2000`, [String(days)]); + const records = rows.flatMap(r => parseRecords(r.lines)); + const { priors, table } = aggregate(records, { minSamples: min }); + return { days, min_samples: min, records: records.length, priors, table }; +} async function results(sql) { const [bench, ledger] = await Promise.all([ sql(`SELECT * FROM console_items WHERE kind = 'bench' ORDER BY (meta->>'date') DESC NULLS LAST, (meta->>'pos')::int DESC LIMIT 60`), @@ -252,6 +262,11 @@ export default async function handler(req, res) { if (fn === 'builds') return json(res, 200, { ok: true, now: new Date().toISOString(), ...(await cached('builds', () => builds(sql))) }); if (fn === 'chain') return json(res, 200, { ok: true, ...(await cached('chain', () => chain(sql))) }); if (fn === 'results') return json(res, 200, { ok: true, ...(await cached('results', () => results(sql))) }); + if (fn === 'tuning') { + const days = Math.min(365, Math.max(1, Number(q.days) || 30)); + const min = Math.min(100, Math.max(1, Number(q.min) || 5)); + return json(res, 200, { ok: true, now: new Date().toISOString(), ...(await cached(`tuning-${days}-${min}`, () => tuning(sql, days, min))) }); + } if (fn === 'log') { const limit = Math.min(300, Math.max(1, Number(q.limit) || 100)); const params = []; let where = `kind IN ('log','build','note')`; diff --git a/relay/lib/ember.mjs b/relay/lib/ember.mjs new file mode 100644 index 00000000..166b7027 --- /dev/null +++ b/relay/lib/ember.mjs @@ -0,0 +1,131 @@ +// Ember Tune, the fleet side (docs/plans/ember-tune.md): the TUNE records every app uploads with its log are folded +// into one prior per (card model, driver major, program class): the median chosen point, its spread and the sample +// count. The publisher writes the priors into the signed manifest's `tuning` section beside the kernel-variant +// cards (tools/tuning.mjs --write), the console shows them (api/console.mjs fn=tuning, tools/console.mjs tuning), +// and the public bench table lists them per model (site/miner-priors.json). No dependencies; the tests in +// relay/test/ember.test.mjs drive these functions with a fixture of captured records. +// +// A record (app/igneum-app/src/ember.rs record_json): {ts, machine (a hash of the install id), app, os, card, vendor, +// driver, driver_major, class, key, plan: full|confirm|baseline, steps: [{clock_mhz, power_pct, limit_w, watts, mhs, +// eff, gclk, mclk, tmax, faults, mark}], chosen: {...}, before: {...}|null, eff, mhs, watts}. Nothing identifies the +// owner: no address, no hostname, no raw machine id. + +/// The TUNE records inside uploaded log text, de-duplicated on (machine, card, ts) because the log is re-sent every +/// minute. Baseline records (measure only) are kept apart: they say what a card does untuned, never what to set. +export function parseRecords(text) { + const out = []; + for (const line of String(text || '').split('\n')) { + const i = line.indexOf('TUNE {'); + if (i < 0) continue; + let rec; + try { rec = JSON.parse(line.slice(i + 5)); } catch { continue; } + if (!rec || !rec.card || !rec.key || !rec.plan) continue; + out.push(rec); + } + return out; +} + +export function dedupe(records) { + const seen = new Set(); + const out = []; + for (const r of records) { + const k = `${r.machine}|${r.card}|${r.ts}`; + if (seen.has(k)) continue; + seen.add(k); + out.push(r); + } + return out; +} + +export const median = xs => { const s = xs.filter(x => Number.isFinite(x)).sort((a, b) => a - b); return s.length ? (s.length % 2 ? s[(s.length - 1) / 2] : (s[s.length / 2 - 1] + s[s.length / 2]) / 2) : 0; }; + +/// The median absolute deviation as a percent of the median (0 for one sample or a zero median). +export function spreadPct(xs) { + const m = median(xs); + if (!m || xs.length < 2) return 0; + return Number((median(xs.map(x => Math.abs(x - m))) / m * 100).toFixed(2)); +} + +const usable = r => r && r.chosen && r.chosen.mark === 'ok' && r.chosen.eff > 0 && (r.plan === 'full' || r.plan === 'confirm'); + +/// Folds records into priors: one per key, from the full and confirm records with a usable chosen point. The point +/// is the median clock cap and the median power percent (each rounded to the step the apps use: 10 MHz, 1%), the +/// efficiency, rate and draw are medians, the spread is the MAD of the efficiency in percent, `samples` counts the +/// records and `machines` the distinct install hashes. An outlier (one bad card, one hot room) moves the median by +/// at most one rank, never by its size. Baseline records are summarised beside the prior as `baseline` (median +/// MH/W untuned) so the console can show the gain. +export function aggregate(records, { minSamples = 1 } = {}) { + const byKey = new Map(); + for (const r of dedupe(records)) { + const g = byKey.get(r.key) || { key: r.key, card: r.card, vendor: r.vendor || '', driver_major: r.driver_major || '', class: r.class || 'v2', tuned: [], baseline: [], machines: new Set() }; + byKey.set(r.key, g); + g.machines.add(r.machine); + if (usable(r)) g.tuned.push(r); + else if (r.plan === 'baseline' && r.chosen && r.chosen.eff > 0) g.baseline.push(r); + } + const priors = {}; + const table = []; + for (const g of byKey.values()) { + const t = g.tuned; + const row = { + key: g.key, card: g.card, vendor: g.vendor, driver_major: g.driver_major, class: g.class, + samples: t.length, machines: g.machines.size, + baseline_samples: g.baseline.length, + baseline_eff: g.baseline.length ? Number(median(g.baseline.map(r => r.chosen.eff)).toFixed(4)) : null, + baseline_mhs: g.baseline.length ? Number(median(g.baseline.map(r => r.chosen.mhs)).toFixed(2)) : null, + baseline_watts: g.baseline.length ? Number(median(g.baseline.map(r => r.chosen.watts)).toFixed(1)) : null, + }; + if (t.length) { + const effs = t.map(r => r.chosen.eff); + const prior = { + clock_mhz: Math.round(median(t.map(r => r.chosen.clock_mhz)) / 10) * 10, + power_pct: Math.round(median(t.map(r => r.chosen.power_pct))), + eff: Number(median(effs).toFixed(4)), + mhs: Number(median(t.map(r => r.chosen.mhs)).toFixed(2)), + watts: Number(median(t.map(r => r.chosen.watts)).toFixed(1)), + spread_pct: spreadPct(effs), + samples: t.length, + machines: g.machines.size, + card: g.card, + vendor: g.vendor, + driver_major: g.driver_major, + class: g.class, + updated: new Date(Math.max(...t.map(r => Number(r.ts) || 0)) * 1000).toISOString().replace(/\.\d{3}Z$/, 'Z'), + }; + // the untuned reference: the full plan's first step (the power ladder's 100%), else the baseline records + const befores = t.map(r => r.before && r.before.eff > 0 ? r.before.eff : null).filter(x => x !== null); + if (befores.length) prior.before_eff = Number(median(befores).toFixed(4)); + else if (row.baseline_eff) prior.before_eff = row.baseline_eff; + if (prior.before_eff) prior.gain_pct = Number(((prior.eff / prior.before_eff - 1) * 100).toFixed(1)); + Object.assign(row, prior); + if (t.length >= minSamples) priors[g.key] = prior; + } + table.push(row); + } + table.sort((a, b) => (b.samples - a.samples) || (a.key < b.key ? -1 : 1)); + return { priors, table }; +} + +/// The manifest's tuning section with the priors folded in: the kernel-variant `cards` object is kept as is, +/// `priors` replaces the previous priors (a key that lost its samples drops out), `ember` carries the settings. +export function mergeTuning(existing, priors, ember = {}) { + const base = existing && typeof existing === 'object' ? existing : {}; + const cards = base.cards && typeof base.cards === 'object' && !Array.isArray(base.cards) ? base.cards : {}; + const settings = { enabled: true, min_samples: 5, rate_tolerance_pct: 1, ...(base.ember && typeof base.ember === 'object' ? base.ember : {}), ...ember }; + return { ...base, updated: new Date().toISOString().replace(/\.\d{3}Z$/, 'Z'), cards, ember: settings, priors: priors || {} }; +} + +/// A prior as a card starts from it (app/igneum-app/src/ember.rs prior_of): None under the sample floor. +export function priorFor(tuning, key, minSamples) { + const p = tuning && tuning.priors && tuning.priors[key]; + const floor = Number.isFinite(minSamples) ? minSamples : (tuning && tuning.ember && tuning.ember.min_samples) || 5; + if (!p || !(p.samples >= floor)) return null; + return { clock_mhz: p.clock_mhz || 0, power_pct: Math.min(100, Math.max(50, p.power_pct || 100)), eff: p.eff, samples: p.samples }; +} + +/// One text line per prior for the console and the CLI. +export function priorLine(p) { + const point = p.clock_mhz ? `${p.clock_mhz} MHz at ${p.power_pct}%` : `${p.power_pct}% (clock unlocked)`; + const gain = p.gain_pct != null ? ` (${p.gain_pct >= 0 ? '+' : ''}${p.gain_pct}% over untuned ${p.before_eff} MH/W)` : ''; + return `${p.card.replace(/_/g, ' ')} | driver ${p.driver_major} | ${p.class}: ${point}, ${p.eff} MH/W${gain}, ${p.mhs} MH/s at ${p.watts} W, spread ${p.spread_pct}%, ${p.samples} sample(s) from ${p.machines} machine(s)`; +} diff --git a/relay/playbooks/ember-tune-pc1.ps1 b/relay/playbooks/ember-tune-pc1.ps1 new file mode 100644 index 00000000..2e135be2 --- /dev/null +++ b/relay/playbooks/ember-tune-pc1.ps1 @@ -0,0 +1,143 @@ +# Igneum run job: Ember Tune end to end on PC 1 (machine ae432dc7), unattended. 5 October 2026. +# Published as a `run` job with --stop-miners (docs/plans/ember-tune.md): the installed app stops its miners and +# holds them; this script takes the engine that carries src/ember.rs (the one the fetch job put in +# \jobs\ember-kit-1\igneum-app-ember.exe, else the installed one; the AMD helper with the --tune and --set +# commands from the telemetry agent's fetch job, \jobs\amd-kit-1\kit\igneum-gpu-telemetry.exe), copies the install folder to a scratch +# folder beside it, swaps the engine in, and starts that SECOND engine with `--sweep` in a scratch data folder (the +# real settings.json, machine-id and wallet.json copied in; remote jobs, auto-update and proving switched off +# there). That engine finds the installed app's node on 127.0.0.1:26610, mines on every card with the live program, +# tunes them one after the other (the full two-knob plan where the card can be controlled, the baseline measurement +# where it cannot: NVIDIA without administrator rights, Apple), prints every TUNE line on stdout, uploads its log +# (the TUNE {json} record reaches the intake) and quits. Every TUNE line is re-emitted as a RESULT line, so +# `node tools/jobs.mjs ` shows the table. Before and after, nvidia-smi's limits and clocks and the AMD +# helper's `--tune` lines are printed, so the restore can be read. The installed app's miners restart when the job +# ends. Not elevated: nothing asks for administrator rights (the project lead asleep, 5 October 2026); the NVIDIA card is +# therefore measure only tonight unless the engine finds itself elevated. +$ErrorActionPreference = 'Continue' +$budgetMinutes = 35 +$started = Get-Date +$deadline = $started.AddMinutes($budgetMinutes) +function Say([string] $m) { Write-Host ("[" + (Get-Date -Format 'HH:mm:ss') + "] " + $m) } + +# the installed engine: the per-user install (0.3.3+), else Program Files +$installDir = $null +foreach ($d in @((Join-Path $env:LOCALAPPDATA 'Programs\Igneum Miner'), (Join-Path $env:ProgramFiles 'Igneum Miner'))) { + if (Test-Path (Join-Path $d 'igneum-app.exe')) { $installDir = $d; break } +} +if (-not $installDir) { Write-Output 'RESULT TUNE error=no_engine reason=igneum-app.exe_not_found'; exit 2 } +$appData = $env:IGNEUM_APP_DATA +if (-not $appData) { $appData = Join-Path $env:LOCALAPPDATA 'igneum' } +$appDir = $env:IGNEUM_APP_DIR +if (-not $appDir) { $appDir = Join-Path $appData 'app' } + +# the scratch install: the whole folder (workers, node, helper, DLLs) with the Ember engine swapped in +$root = Join-Path $env:LOCALAPPDATA 'igneum-tune' +$bin = Join-Path $root 'bin' +$sApp = Join-Path $root 'app' +$sLogs = Join-Path $root 'logs' +New-Item -ItemType Directory -Force -Path $root, $sApp, $sLogs | Out-Null +if (Test-Path $bin) { Remove-Item -LiteralPath $bin -Recurse -Force -ErrorAction SilentlyContinue } +Copy-Item -LiteralPath $installDir -Destination $bin -Recurse -Force +$ember = Join-Path $appDir 'jobs\ember-kit-1\igneum-app-ember.exe' +if (Test-Path $ember) { + Copy-Item -LiteralPath $ember -Destination (Join-Path $bin 'igneum-app.exe') -Force + Say ("engine: the Ember build from " + $ember) +} else { + Say 'engine: the installed one (no jobs\ember-kit-1\igneum-app-ember.exe); an older engine ignores the tune and reports no_rows' +} +$helper = Join-Path $appDir 'jobs\amd-kit-1\kit\igneum-gpu-telemetry.exe' +if (Test-Path $helper) { + Copy-Item -LiteralPath $helper -Destination (Join-Path $bin 'igneum-gpu-telemetry.exe') -Force + Say ("helper: the Ember build of igneum-gpu-telemetry from " + $helper + " sha256=" + (Get-FileHash -LiteralPath $helper -Algorithm SHA256).Hash.ToLower()) +} else { Say 'helper: the installed igneum-gpu-telemetry (no jobs\amd-kit-1\kit\igneum-gpu-telemetry.exe); without --tune the AMD card measures only' } +$exe = Join-Path $bin 'igneum-app.exe' +$ver = (& $exe --version 2>&1 | Out-String).Trim() +Say ("engine: " + $exe + " (" + $ver + ")") +Write-Output ("RESULT TUNE engine " + $ver + " sha256=" + (Get-FileHash -LiteralPath $exe -Algorithm SHA256).Hash.ToLower()) +if ($ver -notmatch 'igneum-app (\d+)\.(\d+)\.(\d+)') { Write-Output 'RESULT TUNE error=version_unknown'; exit 2 } + +foreach ($f in @('settings.json', 'machine-id', 'wallet.json', 'tuning.json')) { + $src = Join-Path $appDir $f + if (Test-Path $src) { Copy-Item -LiteralPath $src -Destination (Join-Path $sApp $f) -Force } +} +# the second engine must not poll jobs (it would see this one), update itself, or prove; the tune is on +$sj = Join-Path $sApp 'settings.json' +if (Test-Path $sj) { + try { + $j = Get-Content -LiteralPath $sj -Raw | ConvertFrom-Json + $j.remote_jobs = $false; $j.auto_update = $false; $j.prove = $false; $j.paused = $false; $j.setup_done = $true; $j.sweep = $true + # every card is due: the stored results are cleared in the COPY only + if ($j.cards) { foreach ($p in $j.cards.PSObject.Properties) { $p.Value.sweep_at = 0; $p.Value.pinned = $false } } + $j | ConvertTo-Json -Depth 8 | Set-Content -LiteralPath $sj -Encoding utf8 + } catch { Say ("settings.json: " + $_.Exception.Message) } +} else { Write-Output 'RESULT TUNE error=no_settings reason=the_installed_app_has_no_settings.json'; exit 2 } +Remove-Item -LiteralPath (Join-Path $sApp 'app.url') -Force -ErrorAction SilentlyContinue + +# the state before, for the report +$smi = Join-Path $env:ProgramFiles 'NVIDIA Corporation\NVSMI\nvidia-smi.exe' +if (-not (Test-Path $smi)) { $smi = Join-Path $env:SystemRoot 'System32\nvidia-smi.exe' } +$tele = Join-Path $bin 'igneum-gpu-telemetry.exe' +function Snapshot([string] $tag) { + if (Test-Path $smi) { + $q = (& $smi --query-gpu=index,name,driver_version,power.draw,power.limit,power.default_limit,power.min_limit,power.max_limit,clocks.gr,clocks.max.gr,clocks.mem --format=csv,noheader 2>&1 | Out-String).Trim() + Write-Output ("RESULT TUNE " + $tag + " nvidia " + ($q -replace "`r?`n", ' | ')) + } + if (Test-Path $tele) { + $t = (& $tele --tune 2>&1 | Out-String).Trim() + Write-Output ("RESULT TUNE " + $tag + " amd " + ($t -replace "`r?`n", ' | ')) + } else { Write-Output ("RESULT TUNE " + $tag + " amd no_helper") } +} +Snapshot 'before' + +# the tune engine: status every 10 s (6 rate samples per 60 s hold) +$env:IGNEUM_APP_DATA = $root +$env:IGNEUM_APP_LOGS = $sLogs +$env:IGNEUM_APP_STATUS_SECS = '10' +$psi = New-Object System.Diagnostics.ProcessStartInfo +$psi.FileName = $exe +$psi.Arguments = '--sweep' +$psi.WorkingDirectory = $bin +$psi.UseShellExecute = $false +$psi.RedirectStandardOutput = $true +$psi.RedirectStandardError = $true +$psi.CreateNoWindow = $true +$p = New-Object System.Diagnostics.Process +$p.StartInfo = $psi +$lines = New-Object System.Collections.ArrayList +$h = { if ($EventArgs.Data) { [void]$Event.MessageData.Add($EventArgs.Data) } } +Register-ObjectEvent -InputObject $p -EventName OutputDataReceived -Action $h -MessageData $lines | Out-Null +Register-ObjectEvent -InputObject $p -EventName ErrorDataReceived -Action $h -MessageData $lines | Out-Null +[void]$p.Start() +$p.BeginOutputReadLine(); $p.BeginErrorReadLine() +Say ("tune engine started, pid " + $p.Id + ", data " + $root) +$seen = 0 +$rows = 0 +while (-not $p.HasExited) { + Start-Sleep -Seconds 5 + while ($seen -lt $lines.Count) { + $l = [string]$lines[$seen]; $seen++ + if ($l -match '^TUNE ') { Write-Output ('RESULT ' + $l); if ($l -match '^TUNE card=') { $rows++ } } + elseif ($l -match '^SWEEP ') { Write-Output ('RESULT ' + $l) } + elseif ($l -match '^(URL|STATE) ') { } + else { Say $l } + } + if ((Get-Date) -gt $deadline) { + Say ("budget of " + $budgetMinutes + " min spent; asking the tune engine to quit") + $u = Join-Path $sApp 'app.url' + if (Test-Path $u) { try { Invoke-WebRequest -Uri ((Get-Content -LiteralPath $u -Raw).Trim() + 'api/quit') -Method POST -Body '{}' -ContentType 'application/json' -UseBasicParsing -TimeoutSec 5 | Out-Null } catch { } } + Start-Sleep -Seconds 20 + if (-not $p.HasExited) { $p.Kill() } + Write-Output 'RESULT TUNE error=budget_exceeded' + } +} +while ($seen -lt $lines.Count) { $l = [string]$lines[$seen]; $seen++; if ($l -match '^TUNE ') { Write-Output ('RESULT ' + $l); if ($l -match '^TUNE card=') { $rows++ } } } +Say ("tune engine exited " + $p.ExitCode + " after " + [int]((Get-Date) - $started).TotalSeconds + " s, " + $rows + " table rows") +Snapshot 'after' +# the tune engine's own log: the TUNE lines and what happened around them +$log = Get-ChildItem -Path $sLogs -Filter 'app-*.log' -ErrorAction SilentlyContinue | Sort-Object LastWriteTime -Descending | Select-Object -First 1 +if ($log) { + Say ("engine log " + $log.FullName + ":") + Get-Content -LiteralPath $log.FullName | Where-Object { $_ -match 'TUNE|tune|power cap|GPUs:|worker ready|STATUS|exited|upload' } | Select-Object -Last 100 | ForEach-Object { Say (' ' + $_) } +} +if ($rows -eq 0) { Write-Output 'RESULT TUNE error=no_rows'; exit 1 } +exit 0 diff --git a/relay/test/ember.test.mjs b/relay/test/ember.test.mjs new file mode 100644 index 00000000..a5d39e96 --- /dev/null +++ b/relay/test/ember.test.mjs @@ -0,0 +1,110 @@ +// node --test relay/test/ember.test.mjs (no dependencies; CI runs it in the site job) +// The fleet aggregation of Ember Tune records (relay/lib/ember.mjs) on a fixture of records in the shape +// app/igneum-app/src/ember.rs record_json writes: known-good (five samples converge on one point), known-bad (an +// outlier does not move the median), the de-duplication of re-sent logs, the manifest merge and the prior lookup. +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import { parseRecords, dedupe, aggregate, mergeTuning, priorFor, priorLine, median, spreadPct } from '../lib/ember.mjs'; + +const step = (clock_mhz, power_pct, watts, mhs, mark = 'ok') => ({ clock_mhz, power_pct, limit_w: 575 * power_pct / 100, watts, mhs, eff: Number((mhs / watts).toFixed(4)), gclk: clock_mhz || 2800, mclk: 10500, tmax: 68, faults: 0, mark }); +const rec = (machine, ts, chosen, over = {}) => ({ + ts, machine, app: '0.3.10', os: 'windows', card: 'NVIDIA_GeForce_RTX_5090', vendor: 'nvidia', driver: '581.57', driver_major: '581', class: 'l128w16', + key: 'NVIDIA_GeForce_RTX_5090|581|l128w16', plan: 'full', steps: [step(0, 100, 290, 124.0), chosen], chosen, before: step(0, 100, 290, 124.0), eff: chosen.eff, mhs: chosen.mhs, watts: chosen.watts, ...over, +}); +// five machines, each landing near 2,470 MHz at 100%: 0.55 to 0.57 MH/W +const good = [ + rec('a1', 1000, step(2472, 100, 220, 123.5)), + rec('b2', 1001, step(2472, 100, 222, 123.1)), + rec('c3', 1002, step(2781, 100, 236, 123.8)), + rec('d4', 1003, step(2472, 100, 218, 123.6)), + rec('e5', 1004, step(2163, 100, 212, 122.9)), +]; +const outlier = rec('f6', 1005, step(1854, 50, 130, 118.0)); // 0.908 MH/W: a card with a broken draw reading +const baseline = rec('g7', 1006, step(0, 80, 290, 122.3), { plan: 'baseline', steps: [step(0, 80, 290, 122.3)], before: null }); +const amd = (machine, ts, chosen) => rec(machine, ts, chosen, { card: 'AMD_Radeon_RX_9070_XT', vendor: 'amd', driver: '32.0.15801.1', driver_major: '32', key: 'AMD_Radeon_RX_9070_XT|32|l128w16', plan: 'confirm', before: null }); + +test('five samples converge on the median point and an outlier does not move it', () => { + const { priors, table } = aggregate(good, { minSamples: 5 }); + const p = priors['NVIDIA_GeForce_RTX_5090|581|l128w16']; + assert.ok(p, 'a prior at the sample floor'); + assert.equal(p.clock_mhz, 2470, 'the median clock cap, rounded to 10 MHz'); + assert.equal(p.power_pct, 100); + assert.equal(p.samples, 5); + assert.equal(p.machines, 5); + assert.ok(p.eff > 0.55 && p.eff < 0.57, `eff ${p.eff}`); + assert.ok(p.spread_pct >= 0 && p.spread_pct < 3, `spread ${p.spread_pct}`); + assert.equal(p.before_eff, Number((124 / 290).toFixed(4))); + assert.ok(p.gain_pct > 25, `gain ${p.gain_pct}% over the untuned 100% point`); + assert.equal(p.updated, '1970-01-01T00:16:44Z'); + // the outlier: 0.908 MH/W at 1,854 MHz joins; the median moves by one rank at most + const with6 = aggregate(good.concat(outlier), { minSamples: 5 }).priors['NVIDIA_GeForce_RTX_5090|581|l128w16']; + assert.equal(with6.samples, 6); + assert.equal(with6.clock_mhz, 2470); + assert.equal(with6.power_pct, 100); + assert.ok(with6.eff < 0.58, `the outlier's 0.908 MH/W did not drag the median: ${with6.eff}`); + assert.ok(with6.spread_pct < 5, `spread ${with6.spread_pct}`); + assert.equal(table[0].key, 'NVIDIA_GeForce_RTX_5090|581|l128w16'); +}); + +test('under the floor there is no prior, and baseline records never make one', () => { + const { priors, table } = aggregate(good.slice(0, 4), { minSamples: 5 }); + assert.deepEqual(priors, {}); + assert.equal(table[0].samples, 4, 'the table still shows the count'); + const b = aggregate([baseline, baseline], { minSamples: 1 }); + assert.deepEqual(b.priors, {}, 'measure-only records say what a card does, never what to set'); + assert.equal(b.table[0].baseline_samples, 1, 'the duplicate upload counted once'); + assert.equal(b.table[0].baseline_eff, Number((122.3 / 290).toFixed(4))); + // a marked chosen step (a faulted or hot winner cannot exist, but a record with one is ignored) + const bad = rec('h8', 1007, step(2000, 100, 200, 120, 'faulted')); + assert.deepEqual(aggregate([bad], { minSamples: 1 }).priors, {}); +}); + +test('records are parsed out of log text and de-duplicated on machine, card and time', () => { + const line = `1791230000 TUNE ${JSON.stringify(good[0])}`; + const text = ['1791229999 status: x', line, line, `1791230001 TUNE ${JSON.stringify(good[1])}`, '1791230002 TUNE {not json'].join('\n'); + const rs = parseRecords(text); + assert.equal(rs.length, 3); + assert.equal(dedupe(rs).length, 2); + assert.equal(parseRecords('').length, 0); +}); + +test('the manifest merge keeps the kernel-variant cards and carries the settings', () => { + const existing = { updated: '2026-10-04T21:00:00Z', window_days: 7, cards: { NVIDIA_GeForce_RTX_5090: { variant: 'u2-ldg', race: true, candidates: ['u2-ldg', 'ldg', 'base'] } }, priors: { 'old|1|v2': { samples: 9 } } }; + const { priors } = aggregate(good, { minSamples: 5 }); + const t = mergeTuning(existing, priors, { rate_tolerance_pct: 1 }); + assert.equal(t.cards.NVIDIA_GeForce_RTX_5090.variant, 'u2-ldg', 'lever 2 untouched'); + assert.equal(t.window_days, 7); + assert.deepEqual(t.ember, { enabled: true, min_samples: 5, rate_tolerance_pct: 1 }); + assert.ok(!t.priors['old|1|v2'], 'a key without samples in the window drops out'); + assert.ok(t.priors['NVIDIA_GeForce_RTX_5090|581|l128w16']); + // the kill switch rides the same section + assert.equal(mergeTuning(existing, {}, { enabled: false }).ember.enabled, false); + assert.deepEqual(mergeTuning(null, {}).cards, {}); + // the round trip: canonical JSON (what publish-manifest.sh signs) parses back to the same prior + const back = JSON.parse(JSON.stringify(t)); + assert.deepEqual(priorFor(back, 'NVIDIA_GeForce_RTX_5090|581|l128w16'), { clock_mhz: 2470, power_pct: 100, eff: priors['NVIDIA_GeForce_RTX_5090|581|l128w16'].eff, samples: 5 }); + assert.equal(priorFor(back, 'NVIDIA_GeForce_RTX_5090|581|l128w16', 6), null, 'six wanted, five there'); + assert.equal(priorFor(back, 'nothing|0|v2'), null); + assert.equal(priorFor(null, 'x'), null); +}); + +test('AMD confirm records aggregate by their own key, and the line reads', () => { + const rs = [amd('p1', 2000, step(2600, 90, 177, 17.7)), amd('p2', 2001, step(2600, 90, 180, 17.6)), amd('p3', 2002, step(2500, 90, 170, 17.4))]; + const { priors, table } = aggregate(rs, { minSamples: 3 }); + const p = priors['AMD_Radeon_RX_9070_XT|32|l128w16']; + assert.equal(p.clock_mhz, 2600); + assert.equal(p.power_pct, 90); + assert.equal(p.vendor, 'amd'); + assert.equal(p.gain_pct, undefined, 'confirm records carry no before step and no baseline was uploaded'); + assert.match(priorLine(p), /^AMD Radeon RX 9070 XT \| driver 32 \| l128w16: 2600 MHz at 90%, 0\.\d+ MH\/W, 17\.6 MH\/s at 177 W, spread \d+(\.\d+)?%, 3 sample\(s\) from 3 machine\(s\)$/); + assert.equal(table.length, 1); +}); + +test('median and spread', () => { + assert.equal(median([3, 1, 2]), 2); + assert.equal(median([4, 1, 2, 3]), 2.5); + assert.equal(median([]), 0); + assert.equal(spreadPct([1, 1, 1]), 0); + assert.equal(spreadPct([10]), 0); + assert.equal(spreadPct([9, 10, 11]), 10); +}); diff --git a/site/build.mjs b/site/build.mjs index d2f354bf..7760c3fb 100644 --- a/site/build.mjs +++ b/site/build.mjs @@ -340,6 +340,21 @@ for (const [file, active] of PAGES) { ]; const table = '
' + ['Card', 'Generator', 'Best MH/s', 'MH per watt', 'Miner', 'Date', 'Source', 'Who measured it'].map(h => ``).join('') + '' + rows.map(r => '' + cell(r).map(c => ``).join('') + '').join('') + '
${h}
${esc(String(c))}
'; + // Ember Tune's fleet priors (site/miner-priors.json, tools/tuning.mjs --priors --site): one row per card model, + // driver major and program class; a row under the sample floor shows its count and no point + const pj = JSON.parse(readFileSync(join(here, 'miner-priors.json'), 'utf8')); + const prows = (pj.rows || []).slice().sort((a, b) => (b.samples - a.samples) || (a.card < b.card ? -1 : 1)); + const pcell = r => [ + r.card, r.driver_major, r.class, r.samples + (r.machines ? ' from ' + r.machines + ' machine' + (r.machines === 1 ? '' : 's') : ''), + r.prior ? (r.clock_mhz ? fmt(r.clock_mhz) + ' MHz at ' + r.power_pct + '%' : r.power_pct + '%, clock unlocked') : 'under the floor (' + pj.min_samples + ' needed)', + r.mh_per_w == null ? 'not yet' : Number(r.mh_per_w).toFixed(3) + (r.spread_pct != null ? ' (spread ' + r.spread_pct + '%)' : ''), + r.mh_s == null ? '' : fmt(r.mh_s) + ' MH/s at ' + fmt(r.watts) + ' W', + r.untuned_mh_per_w == null ? 'not measured' : Number(r.untuned_mh_per_w).toFixed(3) + (r.gain_pct != null ? ' (' + (r.gain_pct >= 0 ? '+' : '') + r.gain_pct + '%)' : ''), + ]; + const ptable = prows.length + ? '
' + ['Card', 'Driver', 'Program class', 'Samples', 'Tuned point', 'MH per watt', 'Rate and draw', 'Untuned MH per watt (gain)'].map(h => ``).join('') + '' + + prows.map(r => '' + pcell(r).map(c => ``).join('') + '').join('') + '
${h}
${esc(String(c))}
' + : '

No tune reports yet. The first rows appear once five machines with the same card model have reported.

'; const body = scrubBench([ '

The table

', '

One row per card, generator version and miner version. The rate is the best one measured. Integrated GPUs are not listed. Prototype rows are bench numbers from before the devnet and say so in the miner column.

', @@ -349,8 +364,12 @@ for (const [file, active] of PAGES) { '

MH per watt needs the card\'s power draw during the run. The app reads it on NVIDIA cards through the driver. Rows get the figure when a run records it.

', '

There is no other Igneum miner to compare with yet, so this table compares cards, not miners. The app that produces these rows: the miner page.

', `

Rows: ${rows.length}. Source file: site/miner-bench.json in the repository.

`, + '

Fleet tuning priors

', + '

Ember Tune runs on every card the app mines with: the power limit and the core clock are stepped on the live program and the card keeps the point with the best MH per watt within 1% of its top rate. Every finished tune is reported back without anything that identifies the owner, and the fleet\'s median point per card model, driver major and program class comes back down inside the signed update manifest as the starting point for the next card of that model. A model needs ' + pj.min_samples + ' reports before its prior is used.

', + ptable, + `

Rows: ${prows.length}${pj.generated ? ', generated ' + pj.generated : ''}. Source file: site/miner-priors.json in the repository, written from the fleet records by tools/tuning.mjs --priors --site.

`, ].join('\n')); - const toc = [{ lvl: 2, t: 'The table', id: 'table' }, { lvl: 2, t: 'How a row gets here', id: 'how' }]; + const toc = [{ lvl: 2, t: 'The table', id: 'table' }, { lvl: 2, t: 'How a row gets here', id: 'how' }, { lvl: 2, t: 'Fleet tuning priors', id: 'priors' }]; writeFileSync(join(here, 'miners.html'), page('Igneum GPU bench table', 'Measured Igneum hash rates per GPU: card, generator version, best MH/s, MH per watt where measured, miner version, date and the log entry each number came from.', body, toc, 'Measured hash rates per card on the Igneum lottery hash, with the generator version, the miner version, the date and the log entry behind each number.', { path: '/miners', heading: 'GPU bench table', active: 'miner' })); diff --git a/site/miner-priors.json b/site/miner-priors.json new file mode 100644 index 00000000..cdb7b4fa --- /dev/null +++ b/site/miner-priors.json @@ -0,0 +1,6 @@ +{ + "_about": "Rows of the fleet priors table at /miners (site/build.mjs), written by tools/tuning.mjs --priors --site from the TUNE records every Igneum Miner uploads. One row per card model, driver major and program class: the median tuned point, MH per watt, the spread and the sample count. No machine names, no addresses.", + "generated": null, + "min_samples": 5, + "rows": [] +} diff --git a/site/miner.html b/site/miner.html index 20687283..3b349e0b 100644 --- a/site/miner.html +++ b/site/miner.html @@ -353,7 +353,7 @@ pre b{color:var(--molten);font-weight:500}
Variant racing every hour

At every hourly prepare the worker compiles the program in several shapes (unroll, load path, register budget, threads per group), times each for 2 seconds and keeps the fastest for the hour. Base keeps its place unless beaten.

log · 4 Oct 2026
Measured on Apple silicon

On the M5 Max, 256-thread groups ran 17.3% and 21.2% faster than base on two programs, under load from other work. The NVIDIA race is built and has not yet run on a GPU.

log · 4 Oct 2026, ratios under contention
Fleet tuning manifest

Every race is one record in the app log. The fleet's best variant per card model goes back out inside the signed update manifest, so a card starts from the known best and keeps racing.

-
Efficiency mode

Hash per watt, the number miners compare. The app steps an NVIDIA card's power cap from 100% to 50%, holds each step for 60 seconds on the live kernel and leaves the cap at the best MH per watt. Built and unit-tested; not yet run on a card.

+
Ember Tune

Hash per watt, the number miners compare. Out of the box the app steps every card's power limit and core clock on the live kernel, 60 seconds a step, and keeps the point with the best MH per watt within 1% of the card's top rate. The memory clock is never touched; a step with a rejected hash, a hot GPU or a dragged memory clock is reverted and marked. Every result feeds a fleet prior per card model that the next card of that model starts from. Built and unit-tested; the first measured tune is owed.

Latency work

A block built on a stale tip earns less. The miner is moving from polling to a template subscription and shorter jobs, so a new tip reaches the card in milliseconds. In progress, no number published yet.

Remote signed jobs, opt in

Our own fleet only. A switch, "Allow remote jobs from Igneum (signed)", with the key's fingerprint beside it. Off aborts the running job and stops polling. Jobs run at most once each and only on the machines they name.

@@ -403,8 +403,8 @@ pre b{color:var(--molten);font-weight:500} LeverThe ideaMeasured state 1 · Race the compiler every hourFor each new program, five to ten kernel variants are compiled, benched for two seconds each, and the winner is kept for the hour.Measured on the Mac's Metal worker: the winning variant +17.3% on the genesis seed and +21.2% on the hourly seed over the base compile, under load from other work. The RTX 5090 race is prepared and not yet run. Log, 4 Oct 2026 - 2 · Auto-tune that learns from the fleetApps report the race per card and program class. The best settings come back down through the signed update manifest as defaults, so the miner gets faster for everyone as the fleet grows.Shipped. The fleet is still small, so no fleet table yet. - 3 · Hash per watt, not hashA sweep per card finds the power point with the best MH per watt and holds it. Miners pay for electricity; that is the number they compare.Shipped for NVIDIA cards through the power cap. Not yet run on a card; the first sweep on a 5090 is the owed measurement. Clocks are the next lever. + 2 · Auto-tune that learns from the fleetApps report the race per card and program class, and every finished Ember Tune. The best settings come back down through the signed update manifest as defaults, so the miner gets faster and more efficient for everyone as the fleet grows.Shipped. The fleet is still small, so no fleet table yet; the priors table on /miners fills as machines report. + 3 · Hash per watt, not hashEmber Tune: two knobs per card (power limit, core clock), the memory clock held, the point with the best MH per watt within 1% of the top rate kept and pinned. Miners pay for electricity; that is the number they compare.Built for NVIDIA (through the driver, with Power control on) and AMD (through the app's own helper, no administrator rights), measure only on Apple silicon. Unit-tested on every rule; the first tune measured on a card is the owed number. 4 · Template latencySolo against the local node, a new template within 50 ms of a new tip, because a late block on a BlockDAG goes red and earns nothing.Approximate, from the 0.3.6 release plan and not yet in the engineering log: switched p50 46 to 52 ms on a three-node CPU run, 5 Oct 2026. Ships in 0.3.6. 5 · Never lose a secondZero-loss hourly program swaps, fault guards, automatic restart, a CPU re-check of every found hash, per-worker health.Shipped. Swap 0.01 ms on Metal and 0.00 ms on CUDA with 0 rejected blocks; recoveries in the table above. Log, 4 Oct 2026 6 · Prove it in publicThe bench table per card, fed from the job channel. We claim fastest only when the table says so.Live at /miners, one row per card, generator version and miner version, each with its log entry. diff --git a/site/miners.html b/site/miners.html index 848f1c8e..5b1e1642 100644 --- a/site/miners.html +++ b/site/miners.html @@ -172,12 +172,12 @@ th{font-family:var(--f-mono);font-size:12px;letter-spacing:.12em;text-transform:
-
2 entries, newest at the bottom
+
3 entries, newest at the bottom

GPU bench table

Measured hash rates per card on the Igneum lottery hash, with the generator version, the miner version, the date and the log entry behind each number.

- +

The table

One row per card, generator version and miner version. The rate is the best one measured. Integrated GPUs are not listed. Prototype rows are bench numbers from before the devnet and say so in the miner column.

CardGeneratorBest MH/sMH per wattMinerDateSourceWho measured it
Apple M5 Max (40 GPU cores, Metal)v145.2not measuredproto-metal bench (prototype, not mining)2026-10-03bench log: 3 October 2026, RTX 5090 first run (the Apple row of the same table)measured by the team. genesis program, 1 GiB dataset
Apple M5 Max (40 GPU cores, Metal)v226.7not measuredigneum-miner devnet v4, Metal worker with prepare2026-10-04bench log: 4 October 2026, first hourly program swap on the live devnet: compile-ahead, no pause, two cardsmeasured by the team. live devnet v4, unbroken through the hour boundary
Apple silicon laptop (model not reported)v224.3not measuredIgneum Miner 0.3.1 (DMG)2026-10-04bench log: 4 October 2026, first outside machine on the devnet: an Apple silicon laptop through the Igneum Miner appreported by the fleet. 21.0 MH/s average over 7 minutes, 24.3 MH/s at the moment of the report, 33 accepted blocks
NVIDIA RTX 5090 (32 GB)v1229not measuredproto-cuda bench (prototype, not mining)2026-10-03bench log: 3 October 2026, RTX 5090, memory-hard dataset (pack igneum-genesis-mh)measured by the team. genesis program, 104 loads per hash, 1 GiB dataset
NVIDIA RTX 5090 (32 GB)v1185.3not measuredproto-cuda bench (prototype, not mining)2026-10-03bench log: 3 October 2026, RTX 5090 first run, dataset sweep and second programmeasured by the team. hourly program, 128 loads per hash, 1 GiB dataset
NVIDIA RTX 5090 (32 GB)v2124.2not measuredIgneum Miner 0.3.0 package, prebuilt NVRTC worker2026-10-04bench log: 4 October 2026, the gfx1036 worker fault and what the Apple M5 Max could and could not reproducemeasured by the team. live devnet v4, 128 loads per hash, CPU re-check clean, 0 rejected
@@ -185,7 +185,11 @@ th{font-family:var(--f-mono);font-size:12px;letter-spacing:.12em;text-transform:

Every row names the engineering log entry or the job it came from. "Measured by the team" means our own hardware and our own log. "Reported by the fleet" means a machine we do not own, read from the status lines its miner uploads.

MH per watt needs the card's power draw during the run. The app reads it on NVIDIA cards through the driver. Rows get the figure when a run records it.

There is no other Igneum miner to compare with yet, so this table compares cards, not miners. The app that produces these rows: the miner page.

-

Rows: 6. Source file: site/miner-bench.json in the repository.

+

Rows: 6. Source file: site/miner-bench.json in the repository.

+

Fleet tuning priors

+

Ember Tune runs on every card the app mines with: the power limit and the core clock are stepped on the live program and the card keeps the point with the best MH per watt within 1% of its top rate. Every finished tune is reported back without anything that identifies the owner, and the fleet's median point per card model, driver major and program class comes back down inside the signed update manifest as the starting point for the next card of that model. A model needs 5 reports before its prior is used.

+

No tune reports yet. The first rows appear once five machines with the same card model have reported.

+

Rows: 0. Source file: site/miner-priors.json in the repository, written from the fleet records by tools/tuning.mjs --priors --site.

Generated from the repository at build time. Times are UTC. Machine names are model names.

diff --git a/tools/console.mjs b/tools/console.mjs index 40feee60..9855f39f 100644 --- a/tools/console.mjs +++ b/tools/console.mjs @@ -5,6 +5,7 @@ // node tools/console.mjs machines the machine cards as text // node tools/console.mjs chain the chain numbers // node tools/console.mjs jobs | builds | results the other tabs as text +// node tools/console.mjs tuning [--days 30] [--min 5] Ember Tune's fleet priors per card model (samples, MH/W) // node tools/console.mjs sync-bench push docs/bench-log.md entries and the FUD ledger counts // node tools/console.mjs sync-dl push the downloads folder listing (names, sizes, times) // node tools/console.mjs sync-hetzner push the newest infra/cloud-devnet/results/ summary @@ -121,6 +122,12 @@ try { } else if (cmd === 'jobs') { const j = await api('jobs'); if (!j.jobs.length) console.log(j.note || 'no jobs'); for (const jb of j.jobs) console.log(`${jb.id} ${jb.title} | ${jb.runs.map(r => `${r.machine} ${r.status}${r.exit_code != null ? ' exit ' + r.exit_code : ''}${r.summary ? ': ' + r.summary.slice(0, 80) : ''}`).join(' | ') || 'queued everywhere'}`); } else if (cmd === 'builds') { const j = await api('builds'); console.log(`manifest ${j.manifest ? j.manifest.version + ' (' + j.manifest.channel + ') ' + Object.keys(j.manifest.platforms || {}).join('+') + ': ' + j.manifest.notes : 'none'}`); if (j.ci) console.log(`ci ${j.ci.installer} fetched ${j.ci.fetched_at} ${j.ci.run}`); for (const b of j.builds) console.log(`${when(b.ts)} ${b.who.padEnd(8)} ${b.title}${b.body ? ': ' + b.body.split('\n')[0].slice(0, 100) : ''}`); } + else if (cmd === 'tuning') { + const j = await api('tuning', { q: { days: flags.days || 30, min: flags.min || 5 } }); + console.log(`${j.records} tune record(s) in ${j.days} day(s); a prior needs ${j.min_samples} sample(s)`); + if (!j.table.length) console.log('no tune records yet (the apps log one per finished tune, 0.3.10 and later)'); + for (const t of j.table) console.log(`${t.card.replace(/_/g, ' ')} | driver ${t.driver_major} | ${t.class}: ${t.samples} sample(s) from ${t.machines} machine(s)${t.samples ? `, ${t.clock_mhz ? t.clock_mhz + ' MHz at ' : ''}${t.power_pct}%${t.clock_mhz ? '' : ' (clock unlocked)'}, ${t.eff} MH/W (${t.mhs} MH/s at ${t.watts} W, spread ${t.spread_pct}%)` : ''}${t.baseline_eff ? `; untuned ${t.baseline_eff} MH/W from ${t.baseline_samples} baseline(s)` : ''}${t.gain_pct != null ? `; gain ${t.gain_pct >= 0 ? '+' : ''}${t.gain_pct}%` : ''}; prior ${t.key in j.priors ? 'yes' : 'no'}`); + } else if (cmd === 'results') { const j = await api('results'); if (j.ledger) console.log(`${j.ledger.title}: ${j.ledger.body}`); for (const b of j.bench) console.log(`${b.meta.date || '-'} ${b.title}`); } else if (cmd === 'sync-bench' || cmd === 'sync-dl' || cmd === 'sync-hetzner' || cmd === 'sync') { const items = []; diff --git a/tools/tuning.mjs b/tools/tuning.mjs index c8b92149..537d8bfd 100644 --- a/tools/tuning.mjs +++ b/tools/tuning.mjs @@ -14,10 +14,17 @@ // variant without a race; the record then carries only the self-test), the default keeps racing (the fleet keeps // learning while the card starts from the known best). // node tools/tuning.mjs --records [--days 7] [--card ] the raw records, newest first +// node tools/tuning.mjs --priors [--days 30] [--min-samples 5] Ember Tune (docs/plans/ember-tune.md): the fleet +// priors per (card model, driver major, program class) from the TUNE records (relay/lib/ember.mjs aggregate); +// --write adds them to the tuning file under "priors" with the "ember" settings (kill switch --tuning-off, +// --rate-tolerance N), beside the kernel-variant cards; --site writes site/miner-priors.json for /miners. // // Reads DATABASE_URL from ~/.config/igneum/env. No dependencies: Neon HTTP SQL over fetch. -import { readFileSync, writeFileSync } from 'node:fs'; +import { readFileSync, writeFileSync, existsSync } from 'node:fs'; import { homedir } from 'node:os'; +import { join, dirname, resolve } from 'node:path'; +import { fileURLToPath } from 'node:url'; +import { parseRecords, aggregate, mergeTuning, priorLine } from '../relay/lib/ember.mjs'; process.stdout.on('error', e => { if (e.code === 'EPIPE') process.exit(0); throw e; }); @@ -46,6 +53,43 @@ const minSamples = Number(opt('--min-samples', 3)); const by = opt('--by', 'mhs'); const outFile = opt('--write', ''); const onlyCard = opt('--card', ''); +const ROOT = resolve(dirname(fileURLToPath(import.meta.url)), '..'); + +// ---- Ember Tune priors (--priors): the TUNE records of the window, folded per key -------------------------------- +if (flag('--priors')) { + const pdays = Number(opt('--days', 30)); + const pmin = Number(opt('--min-samples', 5)); + const prow = await sql( + `SELECT machine, run_id, lines FROM miner_logs WHERE received_at > now() - ($1 || ' days')::interval AND lines LIKE '%TUNE {%' ORDER BY received_at DESC`, + [String(pdays)]); + const recs = prow.flatMap(r => parseRecords(r.lines)).filter(r => !onlyCard || r.card === onlyCard); + const { priors, table } = aggregate(recs, { minSamples: pmin }); + if (!recs.length) console.log(`No TUNE records in the last ${pdays} day(s). The apps log one per finished tune (0.3.10 and later).`); + else { + console.log(`${recs.length} tune record(s) in the last ${pdays} day(s); a prior needs ${pmin} sample(s)`); + console.table(table.map(t => ({ key: t.key, samples: t.samples, machines: t.machines, 'clock MHz': t.clock_mhz ?? '-', 'power %': t.power_pct ?? '-', 'MH/W': t.eff ?? '-', 'MH/s': t.mhs ?? '-', W: t.watts ?? '-', 'spread %': t.spread_pct ?? '-', 'untuned MH/W': t.before_eff ?? t.baseline_eff ?? '-', 'gain %': t.gain_pct ?? '-', prior: t.key in priors ? 'yes' : 'no' }))); + for (const p of Object.values(priors)) console.log(priorLine(p)); + } + if (outFile) { + const existing = existsSync(outFile) ? JSON.parse(readFileSync(outFile, 'utf8')) : {}; + const ember = {}; + if (flag('--tuning-off')) ember.enabled = false; + if (flag('--tuning-on')) ember.enabled = true; + if (opt('--rate-tolerance', '')) ember.rate_tolerance_pct = Number(opt('--rate-tolerance', '1')); + ember.min_samples = pmin; + const merged = mergeTuning(existing, priors, ember); + writeFileSync(outFile, JSON.stringify(merged, null, 2) + '\n'); + console.log(`written ${outFile}: ${Object.keys(merged.cards).length} kernel-variant card(s) kept, ${Object.keys(priors).length} prior(s), ember ${JSON.stringify(merged.ember)}; publish with: packaging/ota/publish-manifest.sh --version --tuning ${outFile} [--deploy]`); + } + if (flag('--site')) { + const site = join(ROOT, 'site', 'miner-priors.json'); + const rows = table.map(t => ({ card: t.card.replace(/_/g, ' '), vendor: t.vendor, driver_major: t.driver_major, class: t.class, clock_mhz: t.clock_mhz ?? null, power_pct: t.power_pct ?? null, mh_per_w: t.eff ?? null, mh_s: t.mhs ?? null, watts: t.watts ?? null, spread_pct: t.spread_pct ?? null, samples: t.samples, machines: t.machines, untuned_mh_per_w: t.before_eff ?? t.baseline_eff ?? null, gain_pct: t.gain_pct ?? null, prior: t.key in priors, updated: t.updated || null })); + const about = existsSync(site) ? JSON.parse(readFileSync(site, 'utf8'))._about : undefined; + writeFileSync(site, JSON.stringify({ _about: about || 'Rows of the fleet priors table at /miners (site/build.mjs), written by tools/tuning.mjs --priors --site from the TUNE records every Igneum Miner uploads. One row per card model, driver major and program class: the median tuned point, MH per watt, the spread and the sample count. No machine names, no addresses.', generated: new Date().toISOString().replace(/\.\d{3}Z$/, 'Z'), min_samples: pmin, rows }, null, 2) + '\n'); + console.log(`written ${site} (${rows.length} row(s))`); + } + process.exit(0); +} // Every app-log upload of the window; the TUNING lines out of them. The same race is uploaded many times (the log // is re-sent every minute), so records are de-duplicated on (machine, card, epoch). From 99f7836a813cc339f23d7c355f2b4e11b25f27bf Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Mon, 5 Oct 2026 21:26:31 +0000 Subject: [PATCH 04/20] bench log: Ember Tune, what PC 1 could measure tonight (elevated=False, the cancelled prompt at 20:09 UTC, the 9070 XT off the bus), the pipeline verified without a card, the tier consequences Co-Authored-By: Claude Fable 5.1 --- docs/bench-log.md | 18 ++++++++++++++++++ 1 file changed, 18 insertions(+) diff --git a/docs/bench-log.md b/docs/bench-log.md index ea9a8633..1731737b 100644 --- a/docs/bench-log.md +++ b/docs/bench-log.md @@ -1607,3 +1607,21 @@ Reading: the kernel is the same 116.0 ms on both paths (18.08 MH/s pure kernel, **A second defect found on the way: the pack export race.** PC 1's app log since its 19:02 UTC restart (`node tools/logs.mjs win-ae432dc7-20261005-190232`): `worker error: error 0 pack packs\devnet: the epoch seed bytes do not give the pack's IGNEUM_SEEDW_INIT` at 19:07:03, 19:07:19 and 19:08:07, so the 9070 XT was not mining at all in the app while this entry was written (my job `rdna4-serve-1` at 18:43 hit the same folder in the same state). Cause, from `app/igneum-app/src/engine.rs` `prepare_worker`: one thread per card, each running `igneum-miner export-pack` into the one folder `packs\devnet`; across an epoch change the two exports interleave and the folder keeps one epoch's `program.h` with the other's `seeds.txt` until the next export. Fix on this branch: a process-wide mutex around both export sites (`EXPORT_LOCK`); the second export rewrites the same pack. Not measured in the app yet: it ships with the branch. **Answer to the project lead.** The 9070 XT does 2.5 G random 4-byte reads per second from its memory for this access pattern, and the hash needs 128 of them, so about 19 MH/s is this card's ceiling for the current program class, on any slot; it was running at 92% of that. The eGPU link cost 6% per job through the read-back, now removed (17.87 against 16.88 MH/s inside jobs standalone). The duplicate platform that halved it to 8.9 + 9.4 is folded away. The pack race that stopped it is serialised. Nothing else in the worker's control moves the number: the next step for this card is the program class itself (fewer, wider loads per hash would favour AMD's 64-byte lines), which is a consensus question, not a worker one. + +## 5 October 2026 (night), Ember Tune: the two-knob efficiency tune, the fleet prior, and what PC 1 could measure tonight (miner-community-lead) + +Branch `ember-tune` (54ff1bc), docs/plans/ember-tune.md. Every card tuned for MH per watt out of the box: the power limit and the core clock cap stepped on the live kernel (memory clock never touched), the point with the best MH per watt within 1% of the top rate kept and pinned, every result uploaded as a `TUNE {json}` record (a hash of the install id, no address) and folded per (card model, driver major, program class) into a prior the signed manifest carries back, so a new card of a known model starts there and confirms it in two steps. + +**What was measured tonight (PC 1, machine ae432dc7, from its own uploads to the intake):** + +| Fact | Where it was read | Consequence | +|---|---|---| +| The installed 0.3.9 app runs as `DESKTOP-KMCV30N\Admin` with `elevated=False` (account line, 19:02:33 UTC) | app log `win-ae432dc7-20261005-190232` | `nvidia-smi -pl` and `-lgc` need administrator rights; the one prompt is the Power control switch (3562f26), which the app never raises by itself | +| Two in-app sweep attempts aborted at 20:09 UTC: `the_elevated_helper_did_not_run_(the_administrator_prompt_was_cancelled)` | the same log | no stored sweep result from today exists; the 5090's two-knob tune is owed to the morning (one click on Power control, then it runs by itself within 2 minutes of steady mining) | +| The RX 9070 XT left PC 1's bus at about 20:40 UTC, was back at 21:09 and gone again at 21:22:59 UTC (the eGPU link, third drop today) | the telemetry agent and the PC 1 scheduler | the AMD path (ADLX, no prompt) is unit-tested on the helper's captured line shapes; its end-to-end run waits for the card | + +**The pipeline, verified without a card:** 9 `ember` unit tests (plans, clamps, the choice rule, the five marks, a faulted step reverted inside a fake-clock run, the confirm verdicts, the baseline plan, the record and prior shapes, the vendor reasons), the AMD `tune` line and the 0.3.10 sample line parsed (`engine::amd_telemetry_tests`), the helper protocol (`sweep::tests`), 6 relay aggregation tests (five samples converge on 2,470 MHz at 100%; an outlier at 0.908 MH/W moves the median by nothing; baseline records make no prior; de-duplication; the manifest merge keeps lever 2's cards; the canonical round trip), 3 UI line tests. A test manifest was signed on this Mac with `packaging/ota/publish-manifest.sh --tuning` from fixture priors: `tuning.priors["NVIDIA_GeForce_RTX_5090|581|l128w16"]` = 2,470 MHz at 100%, 5 samples, beside the kernel-variant `cards` entry and `tuning.ember {enabled: true, min_samples: 5, rate_tolerance_pct: 1}`, signature verified by the signer, 21:25 UTC. + +**Tier consequences** (docs/plans/ember-tune.md section 7): a 9-step full tune costs about 12 minutes once and 3 minutes a week per card, under 1% of the hour, the worker never stops; a rig tunes one card at a time and every card of a known model after the first takes the 3-minute confirm; a pool user gives up the same 1% of shares at most; Apple silicon and AMD on Linux measure only and the row says so. + +(The PC 1 run's numbers follow below when the job reports.) From dd37094a5b3ee684678bd2c013620485528e7fa7 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Mon, 5 Oct 2026 21:31:06 +0000 Subject: [PATCH 05/20] ember-tune.md: a signed prior is a starting point inside the card's own reported limits, never a memory clock; the tests that prove the clamp (consequences row C25) Co-Authored-By: Claude Fable 5.1 --- docs/plans/ember-tune.md | 1 + 1 file changed, 1 insertion(+) diff --git a/docs/plans/ember-tune.md b/docs/plans/ember-tune.md index 22869c77..f0d93f1d 100644 --- a/docs/plans/ember-tune.md +++ b/docs/plans/ember-tune.md @@ -96,6 +96,7 @@ race has run). | Faults: a rejected or mismatched hash marks the step; the card leaving `mining`, a worker error, a job, a pause or 90 C aborts the run and restores the point from before | `Run::sample_fault`, `sweep_drive`, `sweep_abort` | | Memory clock held: never set; a step that drags it under 95% of the baseline's cannot win | `Row::from_samples` | | Vendor limits: every point clamped to the reported range; the clock floor 60% when none is reported | `Limits` | +| A signed prior is only ever a starting point inside the card's OWN reported limits (`power.min_limit` to `power.max_limit`, the clock floor to `clocks.max.gr` or the ADLX `gmax_range`), never a memory clock, never a value the card did not report; the confirm step measures it and the full plan replaces it when a neighbour beats it, so a bad prior costs the fleet one confirm step per card, not a setting. The signing key (K1, docs/security/keys.md) therefore cannot push a card past its vendor ceiling or under its floor | `Plan::confirm` clamps through `Limits::clamp_clock` and `power_pct.clamp(50, 100)`; proven by `ember::tests::the_confirm_plan_checks_the_prior_and_its_neighbour` (a prior of 9,000 MHz at 30% becomes 3,090 MHz at 50%) and `limits_never_exceed_the_vendor_or_undercut_the_floor` | | No prompt the user did not ask for: the NVIDIA helper starts only with Power control on; the `--sweep` job never counts as permission | `sweep_probe_known`, `sweep_helper_start` | | The elevated helper restores the limit and resets the clocks by itself after 20 idle minutes | `sweep::helper_script_*` | From 794e23c5ac6ab98306b6b9e92ed5c5c2473c3650 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Mon, 5 Oct 2026 22:43:43 +0000 Subject: [PATCH 06/20] ember: the AMD helper's gmax range is an offset from stock (PC 1's 9070 XT: -500 to 1000), not MHz: the clock knob stays closed on an offset range and the power ladder is bounded by plimit_range (-30 to 10) on a percent scale Co-Authored-By: Claude Fable 5.1 --- app/igneum-app/src/engine.rs | 30 ++++++++++++++++++++++-------- 1 file changed, 22 insertions(+), 8 deletions(-) diff --git a/app/igneum-app/src/engine.rs b/app/igneum-app/src/engine.rs index 684e454c..095a6028 100644 --- a/app/igneum-app/src/engine.rs +++ b/app/igneum-app/src/engine.rs @@ -1891,13 +1891,13 @@ impl Engine { None if allowed || current > 0 => crate::detect::run_timeout(std::process::Command::new(&smi).args(["-i", &device, "-pl", ¤t.to_string()]), None, Duration::from_secs(20)).map(|out| out.contains("All done")), None => Some(false), }; - shared.send(Cmd::TuneProbe(idx, Ok(TuneProbe { clock_max_mhz: clock_max, clock_min_mhz: 0, driver, direct: direct.unwrap_or(false), amd_ordinal: -1 }))); + shared.send(Cmd::TuneProbe(idx, Ok(TuneProbe { clock_max_mhz: clock_max, clock_min_mhz: 0, driver, direct: direct.unwrap_or(false), amd_ordinal: -1, ..Default::default() }))); }); } "amd" => { let Some(exe) = self.bins.telemetry.clone() else { self.sweep_pending = None; - self.shared.send(Cmd::TuneProbe(idx, Ok(TuneProbe { clock_max_mhz: 0, clock_min_mhz: 0, driver: c.driver.clone(), direct: false, amd_ordinal: -1 }))); + self.shared.send(Cmd::TuneProbe(idx, Ok(TuneProbe { clock_max_mhz: 0, clock_min_mhz: 0, driver: c.driver.clone(), direct: false, amd_ordinal: -1, ..Default::default() }))); return; }; let ordinal = c.amd_ordinal; @@ -1906,14 +1906,18 @@ impl Engine { let out = crate::detect::run_timeout(std::process::Command::new(&exe).arg("--tune"), None, Duration::from_secs(20)).unwrap_or_default(); let t = out.lines().filter_map(parse_amd_tune).find(|t| ordinal < 0 || t.ordinal as i64 == ordinal); shared.send(Cmd::TuneProbe(idx, Ok(match t { - Some(t) if t.ok => TuneProbe { clock_max_mhz: t.gmax_max as u32, clock_min_mhz: t.gmax_min as u32, driver, direct: true, amd_ordinal: t.ordinal as i64 }, - Some(t) => TuneProbe { clock_max_mhz: 0, clock_min_mhz: 0, driver, direct: false, amd_ordinal: t.ordinal as i64 }, - None => TuneProbe { clock_max_mhz: 0, clock_min_mhz: 0, driver, direct: false, amd_ordinal: -1 }, + // PC 1's 9070 XT (ember-tune-pc1-1, 22:30 UTC): `gmax 0 gmax_range -500 1000`, an OFFSET from + // the stock clock, not MHz; a range with a negative floor is an offset range and the clock + // knob stays closed until the stock clock is known (the power limit is the AMD lever), and + // `plimit_range -30 10` bounds the power ladder (the percent scale rides power_* below) + Some(t) if t.ok => TuneProbe { clock_max_mhz: if t.gmax_min >= 0.0 && t.gmax_max > 0.0 { t.gmax_max as u32 } else { 0 }, clock_min_mhz: if t.gmax_min > 0.0 { t.gmax_min as u32 } else { 0 }, driver, direct: true, amd_ordinal: t.ordinal as i64, plimit_min: t.plimit_min, plimit_max: t.plimit_max }, + Some(t) => TuneProbe { clock_max_mhz: 0, clock_min_mhz: 0, driver, direct: false, amd_ordinal: t.ordinal as i64, ..Default::default() }, + None => TuneProbe { clock_max_mhz: 0, clock_min_mhz: 0, driver, direct: false, amd_ordinal: -1, ..Default::default() }, }))); }); } _ => { - self.shared.send(Cmd::TuneProbe(idx, Ok(TuneProbe { clock_max_mhz: 0, clock_min_mhz: 0, driver: c.driver.clone(), direct: false, amd_ordinal: -1 }))); + self.shared.send(Cmd::TuneProbe(idx, Ok(TuneProbe { clock_max_mhz: 0, clock_min_mhz: 0, driver: c.driver.clone(), direct: false, amd_ordinal: -1, ..Default::default() }))); } } } @@ -1939,7 +1943,13 @@ impl Engine { // NVIDIA control: this process is elevated (direct), or Power control is on so the one-prompt helper may run. // The --sweep job alone never counts: it must not raise a prompt on a PC with nobody there (5 October 2026). let power_control = probe.direct || self.shared.settings.lock().unwrap().power_control; - let limits = crate::ember::Limits { power_default_w: c.power_default_w, power_min_w: c.power_min_w, power_max_w: c.power_max_w, clock_max_mhz: probe.clock_max_mhz, clock_min_mhz: probe.clock_min_mhz }; + // AMD's power limit is a percent offset from the default (ADLX): the plan's watts scale becomes a percent + // scale (default 100, floor 100 + plimit_min, ceiling 100 + plimit_max), tune_apply sends pct - 100 + let limits = if c.vendor == "amd" && probe.direct && probe.plimit_max >= probe.plimit_min && probe.plimit_min > -100.0 { + crate::ember::Limits { power_default_w: 100.0, power_min_w: 100.0 + probe.plimit_min, power_max_w: 100.0 + probe.plimit_max, clock_max_mhz: probe.clock_max_mhz, clock_min_mhz: probe.clock_min_mhz } + } else { + crate::ember::Limits { power_default_w: c.power_default_w, power_min_w: c.power_min_w, power_max_w: c.power_max_w, clock_max_mhz: probe.clock_max_mhz, clock_min_mhz: probe.clock_min_mhz } + }; let control = match c.vendor.as_str() { "nvidia" => crate::ember::control_reason("nvidia", &limits, &c.device, power_control, false), "amd" => crate::ember::control_reason("amd", &limits, &c.device, power_control, probe.direct && probe.amd_ordinal >= 0), @@ -2113,7 +2123,8 @@ impl Engine { return; } let n = c.amd_ordinal.to_string(); - let offset = step.point.power_pct as i64 - 100; + // the step's limit on the AMD scale is a percent (the probe's Limits); the offset is that minus 100 + let offset = if c.power_default_w <= 0.0 { step.watts.round() as i64 - 100 } else { step.point.power_pct as i64 - 100 }; let unlocked = clock == 0 && offset == 0; let gmax = if clock > 0 { clock } else { c.clock_max_mhz }; std::thread::spawn(move || { @@ -3465,6 +3476,9 @@ pub struct TuneProbe { /// NVIDIA: this process sets limits itself (elevated); AMD: the helper answered its `--tune` line with ok pub direct: bool, pub amd_ordinal: i64, + /// AMD: the power offset range in percent from the `tune` line (PC 1's 9070 XT: -30 to 10) + pub plimit_min: f64, + pub plimit_max: f64, } /// One `tune` line of igneum-gpu-telemetry --tune: From f76fca96e24922c3c777b3086c1de3504c8fc28c Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Mon, 5 Oct 2026 22:44:46 +0000 Subject: [PATCH 07/20] bench log + plan: PC 1 run 1 aborted by the 0.3.11 update 47 s in, the before snapshots of both cards, the AMD offset-range finding and its consequence per tier Co-Authored-By: Claude Fable 5.1 --- docs/bench-log.md | 11 ++++++++++- docs/plans/ember-tune.md | 8 ++++++-- 2 files changed, 16 insertions(+), 3 deletions(-) diff --git a/docs/bench-log.md b/docs/bench-log.md index 1731737b..c19dfa12 100644 --- a/docs/bench-log.md +++ b/docs/bench-log.md @@ -1624,4 +1624,13 @@ Branch `ember-tune` (54ff1bc), docs/plans/ember-tune.md. Every card tuned for MH **Tier consequences** (docs/plans/ember-tune.md section 7): a 9-step full tune costs about 12 minutes once and 3 minutes a week per card, under 1% of the hour, the worker never stops; a rig tunes one card at a time and every card of a known model after the first takes the 3-minute confirm; a pool user gives up the same 1% of shares at most; Apple silicon and AMD on Linux measure only and the row says so. -(The PC 1 run's numbers follow below when the job reports.) +**The PC 1 run, 22:30 UTC (job ember-tune-pc1-1, engine aeea3228..., PC 1 on 0.3.10):** the job published at 22:29:40Z, the installed app stopped its miners and started the second engine at 22:30:21Z, and at 22:31:06Z the installed app quit for the 0.3.11 update-now (its log: `job ember-tune-pc1-1: aborted (the app is quitting)`), taking the second engine with it 47 s in, before any step. Nothing was set. What the run did record, the "before" snapshots with the miners stopped: + +| Card | Read back at 22:30:20Z | Meaning | +|---|---|---| +| RTX 5090 (driver 617.14) | limit 450 W of 575 W default (min 400, max 600), draw 259.9 W idle-after-stop, core 2,850 MHz, `clocks.max.gr` 3,090 MHz, memory 14,001 MHz | the two-knob plan for this card is 5 power steps (575, 518, 460, 403, 400 W) and 4 clock steps (2,781, 2,472, 2,163, 1,854 MHz); it needs the one administrator prompt (Power control) | +| RX 9070 XT (bus 98, present again) | `tune 1 ... gmax 0 gmax_range -500 1000 plimit 0 plimit_range -30 10 factory 1 ok` | the helper's clock range is an OFFSET from stock in MHz, not a ceiling: a probe reading it as a 1,000 MHz maximum would have asked for `--set-gmax 900`, an overclock. Fixed at 054e041: an offset range closes the clock knob (until the stock clock is known) and the power ladder runs on the percent scale bounded by the range, so the 9070 XT's plan is 100, 90, 80, 70% (the -30 floor), 4 steps | +| Radeon(TM) Graphics (integrated) | `tune 0 ... gmax - ... factory 0 ok` | no manual tuning: measure only, and it is off by default anyway | + +Consequence for the tiers: an AMD card is tuned on its power limit alone until its stock core clock is read (a 9070 XT at -30% is the floor the driver allows, 4 steps, 5 minutes); every NVIDIA card's two-knob plan waits on the user's one click on Power control; the re-run on PC 1 follows the 0.3.11 rollout (the update clears the jobs folder, so the engine and the helper are fetched again), with the scheduler's slot. + diff --git a/docs/plans/ember-tune.md b/docs/plans/ember-tune.md index f0d93f1d..61a3d730 100644 --- a/docs/plans/ember-tune.md +++ b/docs/plans/ember-tune.md @@ -29,7 +29,7 @@ prompt) for both knobs; off, it measures only. | Vendor | Power limit | Core clock cap | Memory clock | How | Rights | |---|---|---|---|---|---| | NVIDIA | `nvidia-smi -pl `, percent of the default, inside `power.min_limit` and `power.max_limit` | `nvidia-smi -lgc 0,`, percent of `clocks.max.gr`; `-rgc` = unlocked | never touched (`-lmc` is not used); read back as `clocks.mem` | directly when the engine is elevated, else the one-prompt helper (` pl `, ` lgc `, ` rgc` in `sweep/cmd.txt`) | administrator, so only with Power control on | -| AMD | `igneum-gpu-telemetry --card N --set-plimit ` (0 = default, -20 = 80%), inside the `tune` line's `plimit_range` | `--set-gmax ` inside `gmax_range`; `--reset` for the default point | not settable through ADLX on RDNA 4; read back as `mclk_mhz`, and a step whose mean memory clock falls under 95% of the baseline's is marked and cannot win | the helper, one process per request, exit 0 and a `tune ... ok` line | none on Windows (ADLX manual tuning); root on Linux, so measure only there | +| AMD | `igneum-gpu-telemetry --card N --set-plimit ` (0 = default, -20 = 80%), inside the `tune` line's `plimit_range` (PC 1's 9070 XT: -30 to 10, so 70% is the floor) | `--set-gmax` only when the `tune` line's `gmax_range` is absolute MHz (floor 0 or above); on RDNA 4 the range is an offset from stock (-500 to 1000 on PC 1) and the clock knob stays closed until the stock clock is known; `--reset` for the default point | not settable through ADLX on RDNA 4; read back as `mclk_mhz`, and a step whose mean memory clock falls under 95% of the baseline's is marked and cannot win | the helper, one process per request, exit 0 and a `tune ... ok` line | none on Windows (ADLX manual tuning); root on Linux, so measure only there | | Apple | none | none | none | measure only | none | Vendor limits are never exceeded and the floor is never undercut: the plan clamps every point (`Limits::clamp_clock`, @@ -151,7 +151,11 @@ measure, TUNE record, upload, aggregation, prior shape in a test manifest). The 9070 XT run are owed: the 5090 the moment Power control is switched on (one prompt, then the tune runs by itself within 2 minutes of steady mining), the 9070 XT when the card is back on the bus. -(The run's numbers are appended below when the job reports.) +Run 1 (ember-tune-pc1-1, 22:30 UTC): aborted 47 s in by the installed app quitting for the 0.3.11 update-now, before +any step; nothing set; the "before" snapshots are in the bench log (5090: 450 W of 575, 2,850 MHz core, 3,090 MHz +maximum, 14,001 MHz memory; 9070 XT present on bus 98 with OFFSET ranges `gmax_range -500 1000`, `plimit_range -30 +10`). The offset finding changed the AMD mapping (054e041): an offset clock range closes the clock knob and the power +ladder runs on a percent scale bounded by `plimit_range`. The re-run follows the 0.3.11 rollout. ## 9. Open From d69e1c9fae1ce4a63ede30bf5064f0c73331a2d3 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Mon, 5 Oct 2026 22:53:32 +0000 Subject: [PATCH 08/20] C35: every quit names its source (Cmd::Quit carries it: the window host's stdin, the host gone, POST /api/quit, the --sweep run's end); the --sweep job never counts as Power control and sets no cap at start (it raised a UAC prompt on PC 1 at 22:30 UTC); the tune playbook's budget quit goes only to its own scratch URL file and says so Co-Authored-By: Claude Fable 5.1 --- app/igneum-app/src/engine.rs | 32 +++++++++++++++++++++--------- app/igneum-app/src/main.rs | 4 ++-- app/igneum-app/src/server.rs | 3 ++- relay/playbooks/ember-tune-pc1.ps1 | 11 +++++++++- 4 files changed, 37 insertions(+), 13 deletions(-) diff --git a/app/igneum-app/src/engine.rs b/app/igneum-app/src/engine.rs index 095a6028..c6e57572 100644 --- a/app/igneum-app/src/engine.rs +++ b/app/igneum-app/src/engine.rs @@ -75,7 +75,8 @@ pub enum Cmd { /// restart the node with the verifier decided again (src/verifier.rs): the trust setting changed, or the /// prover found a host that was not there when the node started RestartNode(String), - Quit, + /// quit, with its source (the log names it: C35, 5 October 2026, two unexplained quits) + Quit(&'static str), } pub struct Shared { @@ -463,6 +464,8 @@ pub struct Engine { last_error_event: Instant, // the efficiency sweep (src/sweep.rs): one card at a time sweep: Option, + /// who asked for the quit (the log's `quit:` line names it) + quit_source: &'static str, /// the request number the vendor tool last carried out (the run's acknowledgement) tune_acked: Option, /// cards whose confirm check found a better neighbour: the full plan runs next @@ -559,6 +562,7 @@ impl Engine { last_settings_save: now, last_error_event: now - Duration::from_secs(600), sweep: None, + quit_source: "unknown", tune_acked: None, tune_full_due: std::collections::HashSet::new(), sweep_pending: None, @@ -702,7 +706,7 @@ impl Engine { if self.shared.runtime.sweep_only { if supported.is_empty() { self.sweep_say("SWEEP none reason=no_supported_card"); - self.shared.send(Cmd::Quit); + self.shared.send(Cmd::Quit("the --sweep run (every card done)")); } else { self.sweep_queue = supported; } @@ -1021,7 +1025,9 @@ impl Engine { } } }, - Cmd::Quit => { + Cmd::Quit(source) => { + self.shared.log(&format!("quit requested by {source}")); + self.quit_source = source; self.quitting = true; self.st().quitting = true; } @@ -1472,6 +1478,10 @@ impl Engine { if self.power_busy { return; } + if self.shared.runtime.sweep_only { + self.shared.log(&format!("power cap ({why}): not touched under --sweep; the tune sets every limit itself")); + return; + } if self.sweep.is_some() || self.sweep_pending.is_some() { // the sweep owns the caps until it ends; it applies the chosen one itself self.shared.log(&format!("power cap ({why}): deferred, a sweep is running")); @@ -1856,7 +1866,7 @@ impl Engine { if let Some((idx, forced)) = pick { self.sweep_begin(idx, forced); } else if self.shared.runtime.sweep_only && self.sweep_queue.is_empty() && self.sweep_pending.is_none() { - self.shared.send(Cmd::Quit); + self.shared.send(Cmd::Quit("the --sweep run (every card done)")); } } @@ -2304,7 +2314,7 @@ impl Engine { } self.upload_logs(false); if self.shared.runtime.sweep_only && self.sweep_queue.is_empty() { - self.shared.send(Cmd::Quit); + self.shared.send(Cmd::Quit("the --sweep run (every card done)")); } } @@ -2356,7 +2366,7 @@ impl Engine { } else { self.sweep_retry.insert(idx, Instant::now() + Duration::from_secs(3600)); if self.shared.runtime.sweep_only && self.sweep_queue.is_empty() { - self.shared.send(Cmd::Quit); + self.shared.send(Cmd::Quit("the --sweep run (every card done)")); } } } @@ -3371,7 +3381,7 @@ impl Engine { } fn shutdown(&mut self) { - self.shared.log("quit: stopping the miners, then the node"); + self.shared.log(&format!("quit: stopping the miners, then the node (source: {})", self.quit_source)); self.jobs.abort(&self.shared, "the app is quitting"); if self.sweep.is_some() || self.sweep_pending.is_some() { self.sweep_abort("the app is quitting"); @@ -3426,7 +3436,11 @@ impl Engine { /// administrator rights (one UAC prompt on Windows, pkexec on Linux); the engine builds an elevated command only when /// Power control is on in Settings, or when it is itself the elevated PC sweep job (--sweep). fn elevation_allowed(power_control: bool, sweep_only: bool) -> bool { - power_control || sweep_only + // C35 (5 October 2026, 22:30 UTC): the unattended --sweep job on PC 1 counted as allowed and raised the one + // administrator prompt nobody was there to answer; an elevated job sets limits directly without asking (the tune + // probe's `direct`), so the flag adds nothing and Power control alone decides + let _ = sweep_only; + power_control } /// The notice when the one prompt was refused, cancelled or not answered: Power control goes back off, no retries. @@ -3634,7 +3648,7 @@ mod tests { // the decision (the project lead, 5 October 2026): off = the app never asks; the elevated PC sweep job is the exception assert!(!super::elevation_allowed(false, false)); assert!(super::elevation_allowed(true, false)); - assert!(super::elevation_allowed(false, true)); + assert!(!super::elevation_allowed(false, true), "the --sweep job alone never asks (C35)"); let mut cards = vec![ super::CardState { vendor: "nvidia".into(), enabled: true, device: "0".into(), name: "RTX 5090".into(), power_default_w: 575.0, power_limit_w: 575.0, power_pct: 80, ..Default::default() }, super::CardState { vendor: "amd".into(), enabled: true, device: "1".into(), name: "RX 9070 XT".into(), power_default_w: 300.0, power_limit_w: 300.0, power_pct: 80, ..Default::default() }, diff --git a/app/igneum-app/src/main.rs b/app/igneum-app/src/main.rs index e94427d9..19b96ed9 100644 --- a/app/igneum-app/src/main.rs +++ b/app/igneum-app/src/main.rs @@ -133,7 +133,7 @@ fn main() { let Ok(l) = line else { break }; let t = l.trim(); match t { - "quit" => shared.send(engine::Cmd::Quit), + "quit" => shared.send(engine::Cmd::Quit("the window host (quit on stdin: the tray menu or the installer)")), "pause" => shared.send(engine::Cmd::Pause), "resume" => shared.send(engine::Cmd::Resume), "elevated ok" => shared.send(engine::Cmd::ElevatedDone(Ok(()))), @@ -142,7 +142,7 @@ fn main() { } } if wrapper { - shared.send(engine::Cmd::Quit); + shared.send(engine::Cmd::Quit("the window host went away (stdin closed)")); } }); } diff --git a/app/igneum-app/src/server.rs b/app/igneum-app/src/server.rs index e9884e36..b2ddb9f3 100644 --- a/app/igneum-app/src/server.rs +++ b/app/igneum-app/src/server.rs @@ -339,7 +339,8 @@ fn api_post(shared: &Arc, path: &str, body: Value) -> Result { - shared.send(Cmd::Quit); + // the caller is on 127.0.0.1 and holds the token: the installer, the OTA apply, a script that read app.url + shared.send(Cmd::Quit("POST /api/quit (a local caller with the token: the installer, the OTA apply, or a script that read app.url)")); Ok(json!({ "ok": true })) } _ => Err("unknown api".into()), diff --git a/relay/playbooks/ember-tune-pc1.ps1 b/relay/playbooks/ember-tune-pc1.ps1 index 2e135be2..26852830 100644 --- a/relay/playbooks/ember-tune-pc1.ps1 +++ b/relay/playbooks/ember-tune-pc1.ps1 @@ -15,6 +15,7 @@ # therefore measure only tonight unless the engine finds itself elevated. $ErrorActionPreference = 'Continue' $budgetMinutes = 35 +if (-not ($budgetMinutes -is [int]) -or $budgetMinutes -lt 5) { $budgetMinutes = 35 } # a budget under 5 minutes is a bug, not a budget (C35) $started = Get-Date $deadline = $started.AddMinutes($budgetMinutes) function Say([string] $m) { Write-Host ("[" + (Get-Date -Format 'HH:mm:ss') + "] " + $m) } @@ -123,8 +124,16 @@ while (-not $p.HasExited) { } if ((Get-Date) -gt $deadline) { Say ("budget of " + $budgetMinutes + " min spent; asking the tune engine to quit") + # C35 (5 October 2026): the only quit this script may send goes to the TUNE engine's own URL file in the scratch + # root, never to a file under the installed app's folder; the RESULT line names the file it used $u = Join-Path $sApp 'app.url' - if (Test-Path $u) { try { Invoke-WebRequest -Uri ((Get-Content -LiteralPath $u -Raw).Trim() + 'api/quit') -Method POST -Body '{}' -ContentType 'application/json' -UseBasicParsing -TimeoutSec 5 | Out-Null } catch { } } + $installedUrl = Join-Path $appDir 'app.url' + if ((Resolve-Path -LiteralPath $u -ErrorAction SilentlyContinue).Path -eq (Resolve-Path -LiteralPath $installedUrl -ErrorAction SilentlyContinue).Path -or $u -like '*\igneum\app\*') { + Write-Output ('RESULT TUNE quit refused: ' + $u + ' is the installed app''s URL file') + } elseif (Test-Path -LiteralPath $u) { + Write-Output ('RESULT TUNE quit asked of the tune engine through ' + $u + ' (pid ' + $p.Id + ')') + try { Invoke-WebRequest -Uri ((Get-Content -LiteralPath $u -Raw).Trim() + 'api/quit') -Method POST -Body '{}' -ContentType 'application/json' -UseBasicParsing -TimeoutSec 5 | Out-Null } catch { } + } else { Write-Output ('RESULT TUNE quit not sent: no URL file at ' + $u + '; killing pid ' + $p.Id) } Start-Sleep -Seconds 20 if (-not $p.HasExited) { $p.Kill() } Write-Output 'RESULT TUNE error=budget_exceeded' From d19441c14deaa8ed7a197fe39513a47434b68b61 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Mon, 5 Oct 2026 23:00:06 +0000 Subject: [PATCH 09/20] C35 class: a second engine gets no pipe (its output goes to a file the playbook tails) and its whole tree is ended at the end and on the budget; ember-tune-pc1.ps1 and sweep-5090.ps1 fixed; tools/ci/second-engine-check.sh fails any playbook without both; the rule in ember-tune.md PC 1, 22:31 UTC: the installed engine's quit hung 24 minutes in the jobs runner's abort, waiting for EOF on the script's stdout pipe whose write end the second engine and its miners had inherited (Process.Start with redirection inherits every inheritable handle), while the orphaned miners mined on against the relaunched app. Co-Authored-By: Claude Fable 5.1 --- .github/workflows/ci.yml | 2 ++ docs/plans/ember-tune.md | 2 ++ relay/playbooks/ember-tune-pc1.ps1 | 44 ++++++++++++++++-------------- relay/playbooks/sweep-5090.ps1 | 44 ++++++++++++++++-------------- tools/ci/second-engine-check.sh | 24 ++++++++++++++++ 5 files changed, 74 insertions(+), 42 deletions(-) create mode 100755 tools/ci/second-engine-check.sh diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 185fa0f4..138a5d15 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -67,6 +67,8 @@ jobs: run: bash tools/ci/no-conflict-markers.sh - name: copied sources are re-stamped before a build run: bash tools/ci/copied-sources-check.sh + - name: second-engine playbooks log to a file and end their tree (C35) + run: bash tools/ci/second-engine-check.sh - name: pinned guest programs match their manifest and are built only by pin-guests.sh run: bash tools/ci/pinned-guests-check.sh - name: no secret file names and no 64-hex secrets in the tree (self-test first, then the tree) diff --git a/docs/plans/ember-tune.md b/docs/plans/ember-tune.md index 61a3d730..4e0c9fea 100644 --- a/docs/plans/ember-tune.md +++ b/docs/plans/ember-tune.md @@ -99,6 +99,8 @@ race has run). | A signed prior is only ever a starting point inside the card's OWN reported limits (`power.min_limit` to `power.max_limit`, the clock floor to `clocks.max.gr` or the ADLX `gmax_range`), never a memory clock, never a value the card did not report; the confirm step measures it and the full plan replaces it when a neighbour beats it, so a bad prior costs the fleet one confirm step per card, not a setting. The signing key (K1, docs/security/keys.md) therefore cannot push a card past its vendor ceiling or under its floor | `Plan::confirm` clamps through `Limits::clamp_clock` and `power_pct.clamp(50, 100)`; proven by `ember::tests::the_confirm_plan_checks_the_prior_and_its_neighbour` (a prior of 9,000 MHz at 30% becomes 3,090 MHz at 50%) and `limits_never_exceed_the_vendor_or_undercut_the_floor` | | No prompt the user did not ask for: the NVIDIA helper starts only with Power control on; the `--sweep` job never counts as permission | `sweep_probe_known`, `sweep_helper_start` | | The elevated helper restores the limit and resets the clocks by itself after 20 idle minutes | `sweep::helper_script_*` | +| A playbook that starts a second engine beside the installed app (the PC measurement jobs) gives it NO pipe (its output goes to a file the script tails: a pipe's write end is inherited by the engine's miners, and the installed app's jobs runner then waits forever for EOF after an abort; C35, PC 1 22:31 UTC, a 24-minute hang and orphaned miners), ends the engine's whole process tree at the end and on the budget (`taskkill /T /F`), and lets the installed app's miners come back only after that | `relay/playbooks/ember-tune-pc1.ps1`, `sweep-5090.ps1`; CI `tools/ci/second-engine-check.sh` fails any playbook without both | +| Every `quit:` line in the app log names its source (the window host's stdin, the host gone, `POST /api/quit`, the `--sweep` run's end) | `Cmd::Quit(&'static str)` (b671c8b) | ## 6. Tests diff --git a/relay/playbooks/ember-tune-pc1.ps1 b/relay/playbooks/ember-tune-pc1.ps1 index 26852830..0ce8566c 100644 --- a/relay/playbooks/ember-tune-pc1.ps1 +++ b/relay/playbooks/ember-tune-pc1.ps1 @@ -14,6 +14,7 @@ # ends. Not elevated: nothing asks for administrator rights (the project lead asleep, 5 October 2026); the NVIDIA card is # therefore measure only tonight unless the engine finds itself elevated. $ErrorActionPreference = 'Continue' +$resultTag = 'TUNE' $budgetMinutes = 35 if (-not ($budgetMinutes -is [int]) -or $budgetMinutes -lt 5) { $budgetMinutes = 35 } # a budget under 5 minutes is a bug, not a budget (C35) $started = Get-Date @@ -94,29 +95,28 @@ Snapshot 'before' $env:IGNEUM_APP_DATA = $root $env:IGNEUM_APP_LOGS = $sLogs $env:IGNEUM_APP_STATUS_SECS = '10' -$psi = New-Object System.Diagnostics.ProcessStartInfo -$psi.FileName = $exe -$psi.Arguments = '--sweep' -$psi.WorkingDirectory = $bin -$psi.UseShellExecute = $false -$psi.RedirectStandardOutput = $true -$psi.RedirectStandardError = $true -$psi.CreateNoWindow = $true -$p = New-Object System.Diagnostics.Process -$p.StartInfo = $psi -$lines = New-Object System.Collections.ArrayList -$h = { if ($EventArgs.Data) { [void]$Event.MessageData.Add($EventArgs.Data) } } -Register-ObjectEvent -InputObject $p -EventName OutputDataReceived -Action $h -MessageData $lines | Out-Null -Register-ObjectEvent -InputObject $p -EventName ErrorDataReceived -Action $h -MessageData $lines | Out-Null -[void]$p.Start() -$p.BeginOutputReadLine(); $p.BeginErrorReadLine() -Say ("tune engine started, pid " + $p.Id + ", data " + $root) +# C35 (5 October 2026): the engine's output goes to a FILE, never a pipe. A pipe's write end is inherited by every +# process the engine starts (its miners and workers), so after an abort the installed app's jobs runner waits for an +# EOF that never comes and hangs in its own quit; and the engine's whole tree is killed at the end (nothing orphaned). +$outFile = Join-Path $root 'engine-stdout.log' +$errFile = Join-Path $root 'engine-stderr.log' +Remove-Item -LiteralPath $outFile, $errFile -Force -ErrorAction SilentlyContinue +$p = Start-Process -FilePath $exe -ArgumentList '--sweep' -WorkingDirectory (Split-Path $exe) -WindowStyle Hidden -PassThru -RedirectStandardOutput $outFile -RedirectStandardError $errFile +Say ("engine started, pid " + $p.Id + ", data " + $root + ", stdout " + $outFile) +function EndTree([int] $procId, [string] $why) { + $before = @(Get-Process -Name 'igneum-app', 'igneum-miner', 'igneum-worker-cuda', 'igneum-worker-opencl', 'igneum-worker-metal' -ErrorAction SilentlyContinue).Count + & taskkill /T /F /PID $procId 2>&1 | Out-Null + Start-Sleep -Seconds 2 + $after = @(Get-Process -Name 'igneum-app', 'igneum-miner', 'igneum-worker-cuda', 'igneum-worker-opencl', 'igneum-worker-metal' -ErrorAction SilentlyContinue).Count + Write-Output ('RESULT ' + $resultTag + ' tree ended (' + $why + '): igneum processes ' + $before + ' -> ' + $after + ' (the installed app''s own miners are stopped and held by the job)') +} $seen = 0 $rows = 0 while (-not $p.HasExited) { Start-Sleep -Seconds 5 - while ($seen -lt $lines.Count) { - $l = [string]$lines[$seen]; $seen++ + $all = @(); if (Test-Path -LiteralPath $outFile) { $all = @(Get-Content -LiteralPath $outFile -ErrorAction SilentlyContinue) } + while ($seen -lt $all.Count) { + $l = [string]$all[$seen]; $seen++ if ($l -match '^TUNE ') { Write-Output ('RESULT ' + $l); if ($l -match '^TUNE card=') { $rows++ } } elseif ($l -match '^SWEEP ') { Write-Output ('RESULT ' + $l) } elseif ($l -match '^(URL|STATE) ') { } @@ -135,12 +135,14 @@ while (-not $p.HasExited) { try { Invoke-WebRequest -Uri ((Get-Content -LiteralPath $u -Raw).Trim() + 'api/quit') -Method POST -Body '{}' -ContentType 'application/json' -UseBasicParsing -TimeoutSec 5 | Out-Null } catch { } } else { Write-Output ('RESULT TUNE quit not sent: no URL file at ' + $u + '; killing pid ' + $p.Id) } Start-Sleep -Seconds 20 - if (-not $p.HasExited) { $p.Kill() } + if (-not $p.HasExited) { EndTree $p.Id 'budget' } Write-Output 'RESULT TUNE error=budget_exceeded' } } -while ($seen -lt $lines.Count) { $l = [string]$lines[$seen]; $seen++; if ($l -match '^TUNE ') { Write-Output ('RESULT ' + $l); if ($l -match '^TUNE card=') { $rows++ } } } +$all = @(); if (Test-Path -LiteralPath $outFile) { $all = @(Get-Content -LiteralPath $outFile -ErrorAction SilentlyContinue) } +while ($seen -lt $all.Count) { $l = [string]$all[$seen]; $seen++; if ($l -match '^TUNE ') { Write-Output ('RESULT ' + $l); if ($l -match '^TUNE card=') { $rows++ } } } Say ("tune engine exited " + $p.ExitCode + " after " + [int]((Get-Date) - $started).TotalSeconds + " s, " + $rows + " table rows") +EndTree $p.Id 'end of run' Snapshot 'after' # the tune engine's own log: the TUNE lines and what happened around them $log = Get-ChildItem -Path $sLogs -Filter 'app-*.log' -ErrorAction SilentlyContinue | Sort-Object LastWriteTime -Descending | Select-Object -First 1 diff --git a/relay/playbooks/sweep-5090.ps1 b/relay/playbooks/sweep-5090.ps1 index da6c2ebb..bb0a3a77 100644 --- a/relay/playbooks/sweep-5090.ps1 +++ b/relay/playbooks/sweep-5090.ps1 @@ -8,6 +8,7 @@ # miners restart when the job ends. Elevated, so nvidia-smi -pl needs no prompt (the engine detects that: mode=direct). # UNTESTED on a PC as of 4 Oct 2026 (parse-checked only). $ErrorActionPreference = 'Continue' +$resultTag = 'SWEEP' $budgetMinutes = 40 $started = Get-Date $deadline = $started.AddMinutes($budgetMinutes) @@ -59,29 +60,28 @@ if (Test-Path $smi) { $env:IGNEUM_APP_DATA = $root $env:IGNEUM_APP_LOGS = $sLogs $env:IGNEUM_APP_STATUS_SECS = '10' -$psi = New-Object System.Diagnostics.ProcessStartInfo -$psi.FileName = $exe -$psi.Arguments = '--sweep' -$psi.WorkingDirectory = Split-Path $exe -$psi.UseShellExecute = $false -$psi.RedirectStandardOutput = $true -$psi.RedirectStandardError = $true -$psi.CreateNoWindow = $true -$p = New-Object System.Diagnostics.Process -$p.StartInfo = $psi -$lines = New-Object System.Collections.ArrayList -$h = { if ($EventArgs.Data) { [void]$Event.MessageData.Add($EventArgs.Data) } } -Register-ObjectEvent -InputObject $p -EventName OutputDataReceived -Action $h -MessageData $lines | Out-Null -Register-ObjectEvent -InputObject $p -EventName ErrorDataReceived -Action $h -MessageData $lines | Out-Null -[void]$p.Start() -$p.BeginOutputReadLine(); $p.BeginErrorReadLine() -Say ("sweep engine started, pid " + $p.Id + ", data " + $root) +# C35 (5 October 2026): the engine's output goes to a FILE, never a pipe. A pipe's write end is inherited by every +# process the engine starts (its miners and workers), so after an abort the installed app's jobs runner waits for an +# EOF that never comes and hangs in its own quit; and the engine's whole tree is killed at the end (nothing orphaned). +$outFile = Join-Path $root 'engine-stdout.log' +$errFile = Join-Path $root 'engine-stderr.log' +Remove-Item -LiteralPath $outFile, $errFile -Force -ErrorAction SilentlyContinue +$p = Start-Process -FilePath $exe -ArgumentList '--sweep' -WorkingDirectory (Split-Path $exe) -WindowStyle Hidden -PassThru -RedirectStandardOutput $outFile -RedirectStandardError $errFile +Say ("engine started, pid " + $p.Id + ", data " + $root + ", stdout " + $outFile) +function EndTree([int] $procId, [string] $why) { + $before = @(Get-Process -Name 'igneum-app', 'igneum-miner', 'igneum-worker-cuda', 'igneum-worker-opencl', 'igneum-worker-metal' -ErrorAction SilentlyContinue).Count + & taskkill /T /F /PID $procId 2>&1 | Out-Null + Start-Sleep -Seconds 2 + $after = @(Get-Process -Name 'igneum-app', 'igneum-miner', 'igneum-worker-cuda', 'igneum-worker-opencl', 'igneum-worker-metal' -ErrorAction SilentlyContinue).Count + Write-Output ('RESULT ' + $resultTag + ' tree ended (' + $why + '): igneum processes ' + $before + ' -> ' + $after + ' (the installed app''s own miners are stopped and held by the job)') +} $seen = 0 $rows = 0 while (-not $p.HasExited) { Start-Sleep -Seconds 5 - while ($seen -lt $lines.Count) { - $l = [string]$lines[$seen]; $seen++ + $all = @(); if (Test-Path -LiteralPath $outFile) { $all = @(Get-Content -LiteralPath $outFile -ErrorAction SilentlyContinue) } + while ($seen -lt $all.Count) { + $l = [string]$all[$seen]; $seen++ if ($l -match '^SWEEP ') { Write-Output ('RESULT ' + $l); if ($l -match '^SWEEP card=') { $rows++ } } elseif ($l -match '^(URL|STATE) ') { } else { Say $l } @@ -91,12 +91,14 @@ while (-not $p.HasExited) { $u = Join-Path $sApp 'app.url' if (Test-Path $u) { try { Invoke-WebRequest -Uri ((Get-Content -LiteralPath $u -Raw).Trim() + 'api/quit') -Method POST -Body '{}' -ContentType 'application/json' -UseBasicParsing -TimeoutSec 5 | Out-Null } catch { } } Start-Sleep -Seconds 20 - if (-not $p.HasExited) { $p.Kill() } + if (-not $p.HasExited) { EndTree $p.Id 'budget' } Write-Output 'RESULT SWEEP error=budget_exceeded' } } -while ($seen -lt $lines.Count) { $l = [string]$lines[$seen]; $seen++; if ($l -match '^SWEEP ') { Write-Output ('RESULT ' + $l); if ($l -match '^SWEEP card=') { $rows++ } } } +$all = @(); if (Test-Path -LiteralPath $outFile) { $all = @(Get-Content -LiteralPath $outFile -ErrorAction SilentlyContinue) } +while ($seen -lt $all.Count) { $l = [string]$all[$seen]; $seen++; if ($l -match '^SWEEP ') { Write-Output ('RESULT ' + $l); if ($l -match '^SWEEP card=') { $rows++ } } } Say ("sweep engine exited " + $p.ExitCode + " after " + [int]((Get-Date) - $started).TotalSeconds + " s, " + $rows + " table rows") +EndTree $p.Id 'end of run' if (Test-Path $smi) { $q = (& $smi --query-gpu=index,power.draw,power.limit --format=csv,noheader 2>&1 | Out-String).Trim() Write-Output ("RESULT SWEEP after " + ($q -replace "`r?`n", ' | ')) diff --git a/tools/ci/second-engine-check.sh b/tools/ci/second-engine-check.sh new file mode 100755 index 00000000..564d390f --- /dev/null +++ b/tools/ci/second-engine-check.sh @@ -0,0 +1,24 @@ +#!/usr/bin/env bash +# The second-engine class (C35, 5 October 2026, PC 1 22:31 UTC): a playbook started a second Igneum engine beside the +# installed app through a redirected PIPE (PowerShell's Process.Start with RedirectStandardOutput). A pipe's write end is +# inherited by every process the engine starts (its miners and workers); when the job was aborted, the installed app's +# jobs runner waited for an EOF the orphaned grandchildren never sent and its quit hung for 24 minutes, and the second +# engine's miners mined on against the relaunched app. Rule for every playbook that starts an engine: (1) the engine's +# output goes to a FILE (Start-Process -RedirectStandardOutput ), never a pipe into the script; (2) the engine's +# whole process tree is ended at the end and on the budget (taskkill /T /F), so nothing is orphaned; the installed +# app's miners come back only after that (the job runner restarts them when the script ends). This check fails CI when +# a playbook starts an engine without both. +set -euo pipefail +cd "$(dirname "$0")/../.." +fail=0 +while IFS= read -r f; do + grep -qE "igneum-app(\.exe)?['\"]? *(--sweep|-ArgumentList '--sweep'|--no-open)|ArgumentList '--sweep'|\.Arguments = '--sweep'" "$f" || continue + if grep -qE 'RedirectStandardOutput *= *\$true|UseShellExecute *= *\$false|Register-ObjectEvent|BeginOutputReadLine' "$f"; then + echo "second-engine: $f starts an engine through a pipe (RedirectStandardOutput/BeginOutputReadLine); use Start-Process -RedirectStandardOutput "; fail=1 + fi + if ! grep -qE 'taskkill /T /F' "$f"; then + echo "second-engine: $f starts an engine without ending its process tree (taskkill /T /F) at the end"; fail=1 + fi +done < <(git ls-files 'relay/playbooks/**' 'tools/windows/**' 'packaging/**' | grep -E '\.ps1$') +[ "$fail" = 0 ] && echo "second-engine: every playbook that starts an engine logs to a file and ends its tree" +exit $fail From f9d9805ae83db78bbb15f3a7597bc136c0e8ed65 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Mon, 5 Oct 2026 23:00:52 +0000 Subject: [PATCH 10/20] bench log + plan: C35 corrected (the quit was not the 0.3.11 update; what is established, the hang, the orphans, the prompt, the fixes) Co-Authored-By: Claude Fable 5.1 --- docs/bench-log.md | 4 ++-- docs/plans/ember-tune.md | 4 ++-- 2 files changed, 4 insertions(+), 4 deletions(-) diff --git a/docs/bench-log.md b/docs/bench-log.md index c19dfa12..21c6712a 100644 --- a/docs/bench-log.md +++ b/docs/bench-log.md @@ -1624,7 +1624,7 @@ Branch `ember-tune` (54ff1bc), docs/plans/ember-tune.md. Every card tuned for MH **Tier consequences** (docs/plans/ember-tune.md section 7): a 9-step full tune costs about 12 minutes once and 3 minutes a week per card, under 1% of the hour, the worker never stops; a rig tunes one card at a time and every card of a known model after the first takes the 3-minute confirm; a pool user gives up the same 1% of shares at most; Apple silicon and AMD on Linux measure only and the row says so. -**The PC 1 run, 22:30 UTC (job ember-tune-pc1-1, engine aeea3228..., PC 1 on 0.3.10):** the job published at 22:29:40Z, the installed app stopped its miners and started the second engine at 22:30:21Z, and at 22:31:06Z the installed app quit for the 0.3.11 update-now (its log: `job ember-tune-pc1-1: aborted (the app is quitting)`), taking the second engine with it 47 s in, before any step. Nothing was set. What the run did record, the "before" snapshots with the miners stopped: +**The PC 1 run, 22:30 UTC (job ember-tune-pc1-1, engine aeea3228..., PC 1 on 0.3.10):** the job published at 22:29:40Z, the installed app stopped its miners and started the second engine at 22:30:21Z, and at 22:31:06Z the installed app quit (its log: `quit: stopping the miners, then the node`, then `job ember-tune-pc1-1: aborted (the app is quitting)`), 46 s in, before any step. Nothing was set. Corrected the same night (C35): the first reading, that the 0.3.11 update caused the quit, was wrong; no update, restart or relay task reached PC 1 then (its own jobs lines and the relay feed), and the quit's source is not in the log because the app did not name it (fixed at b671c8b: every `quit:` line now names its sender). What is established: the window host's tray quit is excluded (a host that sent the quit terminates the engine 45 s later, and the engine lived on until 22:55Z), leaving stdin EOF (the host process gone) or `POST /api/quit`; the engine's quit then HUNG for 24 minutes in the jobs runner's abort, waiting for EOF on the script's stdout pipe whose write end the second engine and its miners had inherited, and those miners (2 igneum-miner, 2 CUDA workers, 1 OpenCL worker) mined on, orphaned, until the relay lane killed them at about 23:00Z; the second engine also raised one administrator prompt at about 22:30:25Z (`apply_power_limits` at start counted `--sweep` as Power control), 41 s before the quit; PC 2's unexplained quit at 20:01:09Z came 20 s after a cancelled prompt of the same class, so the prompt is the common factor and the morning's test (one prompt raised beside the mining app on PC 2, the stamped quit line read). Fixed on the branch: b671c8b (quit sources, Power control alone decides, no cap at start under `--sweep`), 8ab9068 (no pipe into a second engine, its tree ended, the CI check). What the run did record, the "before" snapshots with the miners stopped: | Card | Read back at 22:30:20Z | Meaning | |---|---|---| @@ -1632,5 +1632,5 @@ Branch `ember-tune` (54ff1bc), docs/plans/ember-tune.md. Every card tuned for MH | RX 9070 XT (bus 98, present again) | `tune 1 ... gmax 0 gmax_range -500 1000 plimit 0 plimit_range -30 10 factory 1 ok` | the helper's clock range is an OFFSET from stock in MHz, not a ceiling: a probe reading it as a 1,000 MHz maximum would have asked for `--set-gmax 900`, an overclock. Fixed at 054e041: an offset range closes the clock knob (until the stock clock is known) and the power ladder runs on the percent scale bounded by the range, so the 9070 XT's plan is 100, 90, 80, 70% (the -30 floor), 4 steps | | Radeon(TM) Graphics (integrated) | `tune 0 ... gmax - ... factory 0 ok` | no manual tuning: measure only, and it is off by default anyway | -Consequence for the tiers: an AMD card is tuned on its power limit alone until its stock core clock is read (a 9070 XT at -30% is the floor the driver allows, 4 steps, 5 minutes); every NVIDIA card's two-knob plan waits on the user's one click on Power control; the re-run on PC 1 follows the 0.3.11 rollout (the update clears the jobs folder, so the engine and the helper are fetched again), with the scheduler's slot. +Consequence for the tiers: an AMD card is tuned on its power limit alone until its stock core clock is read (a 9070 XT at -30% is the floor the driver allows, 4 steps, 5 minutes); every NVIDIA card's two-knob plan waits on the user's one click on Power control; the re-run on PC 1 is held until the quit's source is named (the event-log collect) and follows the 0.3.11 rollout (the update clears the jobs folder, so the engine and the helper are fetched again), with the scheduler's slot. diff --git a/docs/plans/ember-tune.md b/docs/plans/ember-tune.md index 4e0c9fea..61d02d14 100644 --- a/docs/plans/ember-tune.md +++ b/docs/plans/ember-tune.md @@ -153,8 +153,8 @@ measure, TUNE record, upload, aggregation, prior shape in a test manifest). The 9070 XT run are owed: the 5090 the moment Power control is switched on (one prompt, then the tune runs by itself within 2 minutes of steady mining), the 9070 XT when the card is back on the bus. -Run 1 (ember-tune-pc1-1, 22:30 UTC): aborted 47 s in by the installed app quitting for the 0.3.11 update-now, before -any step; nothing set; the "before" snapshots are in the bench log (5090: 450 W of 575, 2,850 MHz core, 3,090 MHz +Run 1 (ember-tune-pc1-1, 22:30 UTC): aborted 46 s in by the installed app quitting (source unnamed by the 0.3.10 app; +not an update, not a job, not a relay task: C35 in the bench log), before any step; nothing set; the "before" snapshots are in the bench log (5090: 450 W of 575, 2,850 MHz core, 3,090 MHz maximum, 14,001 MHz memory; 9070 XT present on bus 98 with OFFSET ranges `gmax_range -500 1000`, `plimit_range -30 10`). The offset finding changed the AMD mapping (054e041): an offset clock range closes the clock knob and the power ladder runs on a percent scale bounded by `plimit_range`. The re-run follows the 0.3.11 rollout. From 3d505766d293f5b866141aa9a47cc41c5302fdb5 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Tue, 6 Oct 2026 07:01:57 +0000 Subject: [PATCH 11/20] C35 named: PC 1's 22:31 UTC quit was the per-user installer launched by the second engine's own updater (0.3.9 under min_supported_version = urgent, beating auto_update = false); a second engine never runs the updater (IGNEUM_APP_NO_OTA=1, implied by --sweep; the playbooks set it; the CI check demands it); bench log and plan carry the named source Source: the scratch engine's own log in collect ember-c35-collect-1 (06:59Z): 22:31:02Z '0.3.10 is available: downloading', 22:31:05Z 'update: starting the installer first ... ota-apply.ps1', and the installed app's 'quit:' at 22:31:06Z. Co-Authored-By: Claude Fable 5.1 --- app/igneum-app/src/engine.rs | 24 +++++++++++++++++++++--- docs/bench-log.md | 2 +- docs/plans/ember-tune.md | 6 ++++-- relay/playbooks/ember-tune-pc1.ps1 | 1 + relay/playbooks/sweep-5090.ps1 | 1 + tools/ci/second-engine-check.sh | 6 +++++- 6 files changed, 33 insertions(+), 7 deletions(-) diff --git a/app/igneum-app/src/engine.rs b/app/igneum-app/src/engine.rs index c6e57572..128aff49 100644 --- a/app/igneum-app/src/engine.rs +++ b/app/igneum-app/src/engine.rs @@ -466,6 +466,8 @@ pub struct Engine { sweep: Option, /// who asked for the quit (the log's `quit:` line names it) quit_source: &'static str, + /// the over-the-air updater never runs: IGNEUM_APP_NO_OTA=1 or --sweep (a second engine beside the installed app) + no_ota: bool, /// the request number the vendor tool last carried out (the run's acknowledgement) tune_acked: Option, /// cards whose confirm check found a better neighbour: the full plan runs next @@ -491,6 +493,7 @@ pub struct Engine { impl Engine { pub fn new(shared: Arc, bins: Bins, rx: Receiver, wrapper: bool) -> Engine { + let no_ota = shared.runtime.sweep_only || std::env::var("IGNEUM_APP_NO_OTA").map(|v| v == "1").unwrap_or(false); let (lines_tx, lines_rx) = channel(); let now = Instant::now(); let stamp = { @@ -563,6 +566,7 @@ impl Engine { last_error_event: now - Duration::from_secs(600), sweep: None, quit_source: "unknown", + no_ota, tune_acked: None, tune_full_due: std::collections::HashSet::new(), sweep_pending: None, @@ -629,6 +633,9 @@ impl Engine { if self.shared.runtime.sweep_only { self.shared.log("--sweep: the efficiency sweep runs on every supported card as soon as it mines; the table goes to stdout and this log; the engine quits after it"); } + if self.no_ota { + self.shared.log("updates: off for this engine (IGNEUM_APP_NO_OTA or --sweep): it never downloads or installs, whatever the manifest says (C35)"); + } if (setup_done || self.shared.runtime.sweep_only) && !self.quitting { self.shared.send(Cmd::Start); } @@ -761,7 +768,13 @@ impl Engine { self.restart_node(&why, Duration::from_secs(2)); } } - Cmd::CheckUpdate => self.ota.check_now(&self.shared), + Cmd::CheckUpdate => { + if self.no_ota { + self.shared.event("info", "updates are off for this engine (a measurement run: IGNEUM_APP_NO_OTA or --sweep)"); + } else { + self.ota.check_now(&self.shared); + } + } Cmd::InstallUpdate => self.ota.install_now(&self.shared), Cmd::AutoUpdate(on) => self.ota.set_auto(&self.shared, on), Cmd::OpenUpdateFile => { @@ -2439,8 +2452,13 @@ impl Engine { daa: st.node.daa, } }; - if let Some(crate::ota::Action::Apply) = self.ota.tick(&self.shared, &ctx) { - self.apply_update(); + // C35 (PC 1, 5 October 2026, 22:31 UTC): a second engine started by a measurement job found itself under the + // manifest's min_supported_version ("urgent" beats auto_update = false), ran the per-user installer, and the + // installer's PrepareToInstall quit the INSTALLED app through its api/quit. A second engine never updates. + if !self.no_ota { + if let Some(crate::ota::Action::Apply) = self.ota.tick(&self.shared, &ctx) { + self.apply_update(); + } } if let Some(p) = self.ota.take_override_change() { self.shared.log(&format!("consensus override changed ({}); the node restarts with it at a safe moment", p.display())); diff --git a/docs/bench-log.md b/docs/bench-log.md index 21c6712a..ff45fe34 100644 --- a/docs/bench-log.md +++ b/docs/bench-log.md @@ -1624,7 +1624,7 @@ Branch `ember-tune` (54ff1bc), docs/plans/ember-tune.md. Every card tuned for MH **Tier consequences** (docs/plans/ember-tune.md section 7): a 9-step full tune costs about 12 minutes once and 3 minutes a week per card, under 1% of the hour, the worker never stops; a rig tunes one card at a time and every card of a known model after the first takes the 3-minute confirm; a pool user gives up the same 1% of shares at most; Apple silicon and AMD on Linux measure only and the row says so. -**The PC 1 run, 22:30 UTC (job ember-tune-pc1-1, engine aeea3228..., PC 1 on 0.3.10):** the job published at 22:29:40Z, the installed app stopped its miners and started the second engine at 22:30:21Z, and at 22:31:06Z the installed app quit (its log: `quit: stopping the miners, then the node`, then `job ember-tune-pc1-1: aborted (the app is quitting)`), 46 s in, before any step. Nothing was set. Corrected the same night (C35): the first reading, that the 0.3.11 update caused the quit, was wrong; no update, restart or relay task reached PC 1 then (its own jobs lines and the relay feed), and the quit's source is not in the log because the app did not name it (fixed at b671c8b: every `quit:` line now names its sender). What is established: the window host's tray quit is excluded (a host that sent the quit terminates the engine 45 s later, and the engine lived on until 22:55Z), leaving stdin EOF (the host process gone) or `POST /api/quit`; the engine's quit then HUNG for 24 minutes in the jobs runner's abort, waiting for EOF on the script's stdout pipe whose write end the second engine and its miners had inherited, and those miners (2 igneum-miner, 2 CUDA workers, 1 OpenCL worker) mined on, orphaned, until the relay lane killed them at about 23:00Z; the second engine also raised one administrator prompt at about 22:30:25Z (`apply_power_limits` at start counted `--sweep` as Power control), 41 s before the quit; PC 2's unexplained quit at 20:01:09Z came 20 s after a cancelled prompt of the same class, so the prompt is the common factor and the morning's test (one prompt raised beside the mining app on PC 2, the stamped quit line read). Fixed on the branch: b671c8b (quit sources, Power control alone decides, no cap at start under `--sweep`), 8ab9068 (no pipe into a second engine, its tree ended, the CI check). What the run did record, the "before" snapshots with the miners stopped: +**The PC 1 run, 22:30 UTC (job ember-tune-pc1-1, engine aeea3228..., PC 1 on 0.3.10):** the job published at 22:29:40Z, the installed app stopped its miners and started the second engine at 22:30:21Z, and at 22:31:06Z the installed app quit (its log: `quit: stopping the miners, then the node`, then `job ember-tune-pc1-1: aborted (the app is quitting)`), 46 s in, before any step. Nothing was set. Corrected the same night (C35), then named the next morning from the second engine's own log (collect ember-c35-collect-1, 06:59Z): the second engine, reporting 0.3.9 (the branch's Cargo version) under the manifest's `min_supported_version`, took the 0.3.10 update as urgent (the "urgent" rule beats the copied `auto_update = false`), downloaded it at 22:31:02Z and started `ota-apply.ps1` with the per-user installer at 22:31:05Z; the installer's PrepareToInstall sent `POST /api/quit` to the installed app, which logged `quit:` at 22:31:06Z. So the source was my own second engine's updater, through the installer, one second before. The first reading (the 0.3.11 rollout) was wrong in the cause and right in the class: an installer. What else is established: the engine's quit then HUNG for 24 minutes in the jobs runner's abort, waiting for EOF on the script's stdout pipe whose write end the second engine and its miners had inherited, and those miners (2 igneum-miner, 2 CUDA workers, 1 OpenCL worker) mined on, orphaned, until the relay lane killed them at about 23:00Z; the second engine also raised one administrator prompt at about 22:30:25Z (`apply_power_limits` at start counted `--sweep` as Power control), 41 s before the quit; PC 2's unexplained quit at 20:01:09Z came 20 s after a cancelled prompt of the same class, so the prompt is the common factor and the morning's test (one prompt raised beside the mining app on PC 2, the stamped quit line read). Fixed on the branch: b671c8b (quit sources, Power control alone decides, no cap at start under `--sweep`), 8ab9068 (no pipe into a second engine, its tree ended, the CI check), and the third close: a second engine never runs the updater (`IGNEUM_APP_NO_OTA=1`, implied by `--sweep`; the playbooks set it; the CI check demands it). What the run did record, the "before" snapshots with the miners stopped: | Card | Read back at 22:30:20Z | Meaning | |---|---|---| diff --git a/docs/plans/ember-tune.md b/docs/plans/ember-tune.md index 61d02d14..6d38b142 100644 --- a/docs/plans/ember-tune.md +++ b/docs/plans/ember-tune.md @@ -101,6 +101,7 @@ race has run). | The elevated helper restores the limit and resets the clocks by itself after 20 idle minutes | `sweep::helper_script_*` | | A playbook that starts a second engine beside the installed app (the PC measurement jobs) gives it NO pipe (its output goes to a file the script tails: a pipe's write end is inherited by the engine's miners, and the installed app's jobs runner then waits forever for EOF after an abort; C35, PC 1 22:31 UTC, a 24-minute hang and orphaned miners), ends the engine's whole process tree at the end and on the budget (`taskkill /T /F`), and lets the installed app's miners come back only after that | `relay/playbooks/ember-tune-pc1.ps1`, `sweep-5090.ps1`; CI `tools/ci/second-engine-check.sh` fails any playbook without both | | Every `quit:` line in the app log names its source (the window host's stdin, the host gone, `POST /api/quit`, the `--sweep` run's end) | `Cmd::Quit(&'static str)` (b671c8b) | +| A second engine never runs the updater: `IGNEUM_APP_NO_OTA=1` (implied by `--sweep`) skips the OTA tick and refuses Check now, whatever the manifest's `min_supported_version` says (the installer it would launch quits the installed app: PC 1, 22:31 UTC) | `Engine.no_ota`; the playbooks set the variable; `tools/ci/second-engine-check.sh` demands it | ## 6. Tests @@ -153,8 +154,9 @@ measure, TUNE record, upload, aggregation, prior shape in a test manifest). The 9070 XT run are owed: the 5090 the moment Power control is switched on (one prompt, then the tune runs by itself within 2 minutes of steady mining), the 9070 XT when the card is back on the bus. -Run 1 (ember-tune-pc1-1, 22:30 UTC): aborted 46 s in by the installed app quitting (source unnamed by the 0.3.10 app; -not an update, not a job, not a relay task: C35 in the bench log), before any step; nothing set; the "before" snapshots are in the bench log (5090: 450 W of 575, 2,850 MHz core, 3,090 MHz +Run 1 (ember-tune-pc1-1, 22:30 UTC): aborted 46 s in by the installed app quitting, named the next morning: the +second engine's own updater (0.3.9 under min_supported_version = urgent) ran the per-user installer, whose +PrepareToInstall quit the installed app through its api/quit (C35 in the bench log); before any step; nothing set; the "before" snapshots are in the bench log (5090: 450 W of 575, 2,850 MHz core, 3,090 MHz maximum, 14,001 MHz memory; 9070 XT present on bus 98 with OFFSET ranges `gmax_range -500 1000`, `plimit_range -30 10`). The offset finding changed the AMD mapping (054e041): an offset clock range closes the clock knob and the power ladder runs on a percent scale bounded by `plimit_range`. The re-run follows the 0.3.11 rollout. diff --git a/relay/playbooks/ember-tune-pc1.ps1 b/relay/playbooks/ember-tune-pc1.ps1 index 0ce8566c..e9ed9164 100644 --- a/relay/playbooks/ember-tune-pc1.ps1 +++ b/relay/playbooks/ember-tune-pc1.ps1 @@ -95,6 +95,7 @@ Snapshot 'before' $env:IGNEUM_APP_DATA = $root $env:IGNEUM_APP_LOGS = $sLogs $env:IGNEUM_APP_STATUS_SECS = '10' +$env:IGNEUM_APP_NO_OTA = '1' # C35: a second engine never runs the updater (the installer would quit the installed app) # C35 (5 October 2026): the engine's output goes to a FILE, never a pipe. A pipe's write end is inherited by every # process the engine starts (its miners and workers), so after an abort the installed app's jobs runner waits for an # EOF that never comes and hangs in its own quit; and the engine's whole tree is killed at the end (nothing orphaned). diff --git a/relay/playbooks/sweep-5090.ps1 b/relay/playbooks/sweep-5090.ps1 index bb0a3a77..8b68f7d0 100644 --- a/relay/playbooks/sweep-5090.ps1 +++ b/relay/playbooks/sweep-5090.ps1 @@ -60,6 +60,7 @@ if (Test-Path $smi) { $env:IGNEUM_APP_DATA = $root $env:IGNEUM_APP_LOGS = $sLogs $env:IGNEUM_APP_STATUS_SECS = '10' +$env:IGNEUM_APP_NO_OTA = '1' # C35: a second engine never runs the updater (the installer would quit the installed app) # C35 (5 October 2026): the engine's output goes to a FILE, never a pipe. A pipe's write end is inherited by every # process the engine starts (its miners and workers), so after an abort the installed app's jobs runner waits for an # EOF that never comes and hangs in its own quit; and the engine's whole tree is killed at the end (nothing orphaned). diff --git a/tools/ci/second-engine-check.sh b/tools/ci/second-engine-check.sh index 564d390f..e5a42e31 100755 --- a/tools/ci/second-engine-check.sh +++ b/tools/ci/second-engine-check.sh @@ -7,7 +7,8 @@ # output goes to a FILE (Start-Process -RedirectStandardOutput ), never a pipe into the script; (2) the engine's # whole process tree is ended at the end and on the budget (taskkill /T /F), so nothing is orphaned; the installed # app's miners come back only after that (the job runner restarts them when the script ends). This check fails CI when -# a playbook starts an engine without both. +# a playbook starts an engine without both, or without IGNEUM_APP_NO_OTA = '1' (the third line, same night: the second +# engine's updater found itself under min_supported_version and ran the installer, which quit the installed app). set -euo pipefail cd "$(dirname "$0")/../.." fail=0 @@ -19,6 +20,9 @@ while IFS= read -r f; do if ! grep -qE 'taskkill /T /F' "$f"; then echo "second-engine: $f starts an engine without ending its process tree (taskkill /T /F) at the end"; fail=1 fi + if ! grep -qE "IGNEUM_APP_NO_OTA *= *'1'" "$f"; then + echo "second-engine: $f starts an engine without IGNEUM_APP_NO_OTA = '1' (its updater would run the installer, which quits the installed app: PC 1, 5 October 2026, 22:31 UTC)"; fail=1 + fi done < <(git ls-files 'relay/playbooks/**' 'tools/windows/**' 'packaging/**' | grep -E '\.ps1$') [ "$fail" = 0 ] && echo "second-engine: every playbook that starts an engine logs to a file and ends its tree" exit $fail From f5dfdbf443b5b86abd45f7c693be15715327a70a Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Tue, 6 Oct 2026 07:03:26 +0000 Subject: [PATCH 12/20] bench log: the 22:31 UTC installer run was a second install of 0.3.10 over 0.3.10 (PC 1 took 0.3.10 at 21:40:41Z through update-now), not how PC 1 got 0.3.10 Co-Authored-By: Claude Fable 5.1 --- docs/bench-log.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/bench-log.md b/docs/bench-log.md index ff45fe34..7853fa54 100644 --- a/docs/bench-log.md +++ b/docs/bench-log.md @@ -1624,7 +1624,7 @@ Branch `ember-tune` (54ff1bc), docs/plans/ember-tune.md. Every card tuned for MH **Tier consequences** (docs/plans/ember-tune.md section 7): a 9-step full tune costs about 12 minutes once and 3 minutes a week per card, under 1% of the hour, the worker never stops; a rig tunes one card at a time and every card of a known model after the first takes the 3-minute confirm; a pool user gives up the same 1% of shares at most; Apple silicon and AMD on Linux measure only and the row says so. -**The PC 1 run, 22:30 UTC (job ember-tune-pc1-1, engine aeea3228..., PC 1 on 0.3.10):** the job published at 22:29:40Z, the installed app stopped its miners and started the second engine at 22:30:21Z, and at 22:31:06Z the installed app quit (its log: `quit: stopping the miners, then the node`, then `job ember-tune-pc1-1: aborted (the app is quitting)`), 46 s in, before any step. Nothing was set. Corrected the same night (C35), then named the next morning from the second engine's own log (collect ember-c35-collect-1, 06:59Z): the second engine, reporting 0.3.9 (the branch's Cargo version) under the manifest's `min_supported_version`, took the 0.3.10 update as urgent (the "urgent" rule beats the copied `auto_update = false`), downloaded it at 22:31:02Z and started `ota-apply.ps1` with the per-user installer at 22:31:05Z; the installer's PrepareToInstall sent `POST /api/quit` to the installed app, which logged `quit:` at 22:31:06Z. So the source was my own second engine's updater, through the installer, one second before. The first reading (the 0.3.11 rollout) was wrong in the cause and right in the class: an installer. What else is established: the engine's quit then HUNG for 24 minutes in the jobs runner's abort, waiting for EOF on the script's stdout pipe whose write end the second engine and its miners had inherited, and those miners (2 igneum-miner, 2 CUDA workers, 1 OpenCL worker) mined on, orphaned, until the relay lane killed them at about 23:00Z; the second engine also raised one administrator prompt at about 22:30:25Z (`apply_power_limits` at start counted `--sweep` as Power control), 41 s before the quit; PC 2's unexplained quit at 20:01:09Z came 20 s after a cancelled prompt of the same class, so the prompt is the common factor and the morning's test (one prompt raised beside the mining app on PC 2, the stamped quit line read). Fixed on the branch: b671c8b (quit sources, Power control alone decides, no cap at start under `--sweep`), 8ab9068 (no pipe into a second engine, its tree ended, the CI check), and the third close: a second engine never runs the updater (`IGNEUM_APP_NO_OTA=1`, implied by `--sweep`; the playbooks set it; the CI check demands it). What the run did record, the "before" snapshots with the miners stopped: +**The PC 1 run, 22:30 UTC (job ember-tune-pc1-1, engine aeea3228..., PC 1 on 0.3.10):** the job published at 22:29:40Z, the installed app stopped its miners and started the second engine at 22:30:21Z, and at 22:31:06Z the installed app quit (its log: `quit: stopping the miners, then the node`, then `job ember-tune-pc1-1: aborted (the app is quitting)`), 46 s in, before any step. Nothing was set. Corrected the same night (C35), then named the next morning from the second engine's own log (collect ember-c35-collect-1, 06:59Z): the second engine, reporting 0.3.9 (the branch's Cargo version) under the manifest's `min_supported_version`, took the 0.3.10 update as urgent (the "urgent" rule beats the copied `auto_update = false`), downloaded it at 22:31:02Z and started `ota-apply.ps1` with the per-user installer at 22:31:05Z; the installer's PrepareToInstall sent `POST /api/quit` to the installed app, which logged `quit:` at 22:31:06Z. So the source was my own second engine's updater, through the installer, one second before: a second install of 0.3.10 over the 0.3.10 PC 1 had taken through the shipper's update-now at 21:40:41Z (release-0.3.10.md section 8), whose only effect was the quit and the hang. The first reading (the 0.3.11 rollout) was wrong in the cause and right in the class: an installer. What else is established: the engine's quit then HUNG for 24 minutes in the jobs runner's abort, waiting for EOF on the script's stdout pipe whose write end the second engine and its miners had inherited, and those miners (2 igneum-miner, 2 CUDA workers, 1 OpenCL worker) mined on, orphaned, until the relay lane killed them at about 23:00Z; the second engine also raised one administrator prompt at about 22:30:25Z (`apply_power_limits` at start counted `--sweep` as Power control), 41 s before the quit; PC 2's unexplained quit at 20:01:09Z came 20 s after a cancelled prompt of the same class, so the prompt is the common factor and the morning's test (one prompt raised beside the mining app on PC 2, the stamped quit line read). Fixed on the branch: b671c8b (quit sources, Power control alone decides, no cap at start under `--sweep`), 8ab9068 (no pipe into a second engine, its tree ended, the CI check), and the third close: a second engine never runs the updater (`IGNEUM_APP_NO_OTA=1`, implied by `--sweep`; the playbooks set it; the CI check demands it). What the run did record, the "before" snapshots with the miners stopped: | Card | Read back at 22:30:20Z | Meaning | |---|---|---| From e941588c58f9e4c9261089a47d046de6155dd980 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Tue, 6 Oct 2026 07:15:43 +0000 Subject: [PATCH 13/20] ember-tune-pc1.ps1: the copy runs with Power control off (the installed setting is read and reported, never a prompt: the project lead, 6 October 2026 07:20Z), the kit paths after the update, the AMD card reset to factory at the end Co-Authored-By: Claude Fable 5.1 --- relay/playbooks/ember-tune-pc1.ps1 | 25 +++++++++++++++++++++---- 1 file changed, 21 insertions(+), 4 deletions(-) diff --git a/relay/playbooks/ember-tune-pc1.ps1 b/relay/playbooks/ember-tune-pc1.ps1 index e9ed9164..b8104532 100644 --- a/relay/playbooks/ember-tune-pc1.ps1 +++ b/relay/playbooks/ember-tune-pc1.ps1 @@ -40,15 +40,17 @@ $sLogs = Join-Path $root 'logs' New-Item -ItemType Directory -Force -Path $root, $sApp, $sLogs | Out-Null if (Test-Path $bin) { Remove-Item -LiteralPath $bin -Recurse -Force -ErrorAction SilentlyContinue } Copy-Item -LiteralPath $installDir -Destination $bin -Recurse -Force -$ember = Join-Path $appDir 'jobs\ember-kit-1\igneum-app-ember.exe' -if (Test-Path $ember) { +$ember = $null +foreach ($cand in @((Join-Path $appDir 'jobs\ember-kit-2\igneum-app-ember.exe'), (Join-Path $appDir 'jobs\ember-kit-1\igneum-app-ember.exe'))) { if (Test-Path $cand) { $ember = $cand; break } } +if ($ember) { Copy-Item -LiteralPath $ember -Destination (Join-Path $bin 'igneum-app.exe') -Force Say ("engine: the Ember build from " + $ember) } else { Say 'engine: the installed one (no jobs\ember-kit-1\igneum-app-ember.exe); an older engine ignores the tune and reports no_rows' } -$helper = Join-Path $appDir 'jobs\amd-kit-1\kit\igneum-gpu-telemetry.exe' -if (Test-Path $helper) { +$helper = $null +foreach ($cand in @((Join-Path $appDir 'jobs\amd-kit-2\igneum-gpu-telemetry.exe'), (Join-Path $appDir 'jobs\amd-kit-1\kit\igneum-gpu-telemetry.exe'))) { if (Test-Path $cand) { $helper = $cand; break } } +if ($helper) { Copy-Item -LiteralPath $helper -Destination (Join-Path $bin 'igneum-gpu-telemetry.exe') -Force Say ("helper: the Ember build of igneum-gpu-telemetry from " + $helper + " sha256=" + (Get-FileHash -LiteralPath $helper -Algorithm SHA256).Hash.ToLower()) } else { Say 'helper: the installed igneum-gpu-telemetry (no jobs\amd-kit-1\kit\igneum-gpu-telemetry.exe); without --tune the AMD card measures only' } @@ -68,6 +70,13 @@ if (Test-Path $sj) { try { $j = Get-Content -LiteralPath $sj -Raw | ConvertFrom-Json $j.remote_jobs = $false; $j.auto_update = $false; $j.prove = $false; $j.paused = $false; $j.setup_done = $true; $j.sweep = $true + # the project lead, 6 October 2026, 07:20Z: never raise an administrator prompt. The installed app's Power control is READ and + # reported, but the copy runs with it OFF so the second engine can never start the elevated helper; the 5090 is + # measured as it runs either way (the two-knob tune is the installed engine's job once it carries Ember Tune) + $installedPowerControl = $false + try { $installedPowerControl = [bool]$j.power_control } catch { } + Write-Output ('RESULT TUNE installed_power_control=' + $installedPowerControl.ToString().ToLower() + ' (the copy runs with it off: no prompt)') + if ($j.PSObject.Properties.Name -contains 'power_control') { $j.power_control = $false } else { $j | Add-Member -NotePropertyName power_control -NotePropertyValue $false } # every card is due: the stored results are cleared in the COPY only if ($j.cards) { foreach ($p in $j.cards.PSObject.Properties) { $p.Value.sweep_at = 0; $p.Value.pinned = $false } } $j | ConvertTo-Json -Depth 8 | Set-Content -LiteralPath $sj -Encoding utf8 @@ -144,6 +153,14 @@ $all = @(); if (Test-Path -LiteralPath $outFile) { $all = @(Get-Content -Literal while ($seen -lt $all.Count) { $l = [string]$all[$seen]; $seen++; if ($l -match '^TUNE ') { Write-Output ('RESULT ' + $l); if ($l -match '^TUNE card=') { $rows++ } } } Say ("tune engine exited " + $p.ExitCode + " after " + [int]((Get-Date) - $started).TotalSeconds + " s, " + $rows + " table rows") EndTree $p.Id 'end of run' +# the AMD card back to factory: the measurement leaves PC 1 as found (the product pins the point; the installed app +# does not carry Ember Tune yet, so a pinned ADLX setting would be invisible to it) +if (Test-Path $tele) { + $tl = (& $tele --tune 2>&1 | Out-String) + foreach ($m in [regex]::Matches($tl, 'tune (\d+) name "([^"]+)" gmax ([-\d]+) gmax_range')) { + if ($m.Groups[3].Value -ne '-') { $r = (& $tele --card $m.Groups[1].Value --reset 2>&1 | Out-String).Trim(); Write-Output ('RESULT TUNE reset card ' + $m.Groups[1].Value + ' (' + $m.Groups[2].Value + '): ' + ($r -replace "`r?`n", ' | ')) } + } +} Snapshot 'after' # the tune engine's own log: the TUNE lines and what happened around them $log = Get-ChildItem -Path $sLogs -Filter 'app-*.log' -ErrorAction SilentlyContinue | Sort-Object LastWriteTime -Descending | Select-Object -First 1 From 6c38efc1c6ebd835dc512364768e4a1f27adbb4e Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Tue, 6 Oct 2026 07:17:21 +0000 Subject: [PATCH 14/20] ember-tune-pc1.ps1: both cards stay on their best points (the project lead, 6 October 2026 07:25Z); no reset at the end Co-Authored-By: Claude Fable 5.1 --- relay/playbooks/ember-tune-pc1.ps1 | 9 +-------- 1 file changed, 1 insertion(+), 8 deletions(-) diff --git a/relay/playbooks/ember-tune-pc1.ps1 b/relay/playbooks/ember-tune-pc1.ps1 index b8104532..892cd542 100644 --- a/relay/playbooks/ember-tune-pc1.ps1 +++ b/relay/playbooks/ember-tune-pc1.ps1 @@ -153,14 +153,7 @@ $all = @(); if (Test-Path -LiteralPath $outFile) { $all = @(Get-Content -Literal while ($seen -lt $all.Count) { $l = [string]$all[$seen]; $seen++; if ($l -match '^TUNE ') { Write-Output ('RESULT ' + $l); if ($l -match '^TUNE card=') { $rows++ } } } Say ("tune engine exited " + $p.ExitCode + " after " + [int]((Get-Date) - $started).TotalSeconds + " s, " + $rows + " table rows") EndTree $p.Id 'end of run' -# the AMD card back to factory: the measurement leaves PC 1 as found (the product pins the point; the installed app -# does not carry Ember Tune yet, so a pinned ADLX setting would be invisible to it) -if (Test-Path $tele) { - $tl = (& $tele --tune 2>&1 | Out-String) - foreach ($m in [regex]::Matches($tl, 'tune (\d+) name "([^"]+)" gmax ([-\d]+) gmax_range')) { - if ($m.Groups[3].Value -ne '-') { $r = (& $tele --card $m.Groups[1].Value --reset 2>&1 | Out-String).Trim(); Write-Output ('RESULT TUNE reset card ' + $m.Groups[1].Value + ' (' + $m.Groups[2].Value + '): ' + ($r -replace "`r?`n", ' | ')) } - } -} +# the project lead, 6 October 2026, 07:25Z: both cards stay on their best MH/W points (the tune pins them); no factory reset here. Snapshot 'after' # the tune engine's own log: the TUNE lines and what happened around them $log = Get-ChildItem -Path $sLogs -Filter 'app-*.log' -ErrorAction SilentlyContinue | Sort-Object LastWriteTime -Descending | Select-Object -First 1 From 715890b0da499b2dcdd0a558ab2a0790974d0593 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Tue, 6 Oct 2026 07:20:36 +0000 Subject: [PATCH 15/20] ember-tune-pc1.ps1: find the AMD helper under any amd-kit job folder Co-Authored-By: Claude Fable 5.1 --- relay/playbooks/ember-tune-pc1.ps1 | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/relay/playbooks/ember-tune-pc1.ps1 b/relay/playbooks/ember-tune-pc1.ps1 index 892cd542..bb5c8eb9 100644 --- a/relay/playbooks/ember-tune-pc1.ps1 +++ b/relay/playbooks/ember-tune-pc1.ps1 @@ -49,7 +49,8 @@ if ($ember) { Say 'engine: the installed one (no jobs\ember-kit-1\igneum-app-ember.exe); an older engine ignores the tune and reports no_rows' } $helper = $null -foreach ($cand in @((Join-Path $appDir 'jobs\amd-kit-2\igneum-gpu-telemetry.exe'), (Join-Path $appDir 'jobs\amd-kit-1\kit\igneum-gpu-telemetry.exe'))) { if (Test-Path $cand) { $helper = $cand; break } } +$found = Get-ChildItem -Path (Join-Path $appDir 'jobs') -Recurse -Filter 'igneum-gpu-telemetry.exe' -ErrorAction SilentlyContinue | Where-Object { $_.FullName -match 'amd-kit' } | Sort-Object LastWriteTime -Descending | Select-Object -First 1 +if ($found) { $helper = $found.FullName } if ($helper) { Copy-Item -LiteralPath $helper -Destination (Join-Path $bin 'igneum-gpu-telemetry.exe') -Force Say ("helper: the Ember build of igneum-gpu-telemetry from " + $helper + " sha256=" + (Get-FileHash -LiteralPath $helper -Algorithm SHA256).Hash.ToLower()) From d1705e0e1b7c605b716c02480380de0526ce6c09 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Tue, 6 Oct 2026 07:58:39 +0000 Subject: [PATCH 16/20] playbooks: the scratch settings.json is written without a BOM (PowerShell 5.1's -Encoding utf8 adds one, the engine's JSON parser refuses it, the copy read as defaults with no payout address and nothing mined in runs 1 and 2); the address is read back and the job fails at once if it is empty; the CI check fails any playbook writing JSON with Set-Content -Encoding utf8 Co-Authored-By: Claude Fable 5.1 --- relay/playbooks/ember-tune-pc1.ps1 | 11 ++++++++++- relay/playbooks/sweep-5090.ps1 | 2 +- tools/ci/second-engine-check.sh | 3 +++ 3 files changed, 14 insertions(+), 2 deletions(-) diff --git a/relay/playbooks/ember-tune-pc1.ps1 b/relay/playbooks/ember-tune-pc1.ps1 index bb5c8eb9..a50b018f 100644 --- a/relay/playbooks/ember-tune-pc1.ps1 +++ b/relay/playbooks/ember-tune-pc1.ps1 @@ -80,8 +80,17 @@ if (Test-Path $sj) { if ($j.PSObject.Properties.Name -contains 'power_control') { $j.power_control = $false } else { $j | Add-Member -NotePropertyName power_control -NotePropertyValue $false } # every card is due: the stored results are cleared in the COPY only if ($j.cards) { foreach ($p in $j.cards.PSObject.Properties) { $p.Value.sweep_at = 0; $p.Value.pinned = $false } } - $j | ConvertTo-Json -Depth 8 | Set-Content -LiteralPath $sj -Encoding utf8 + # PowerShell 5.1's Set-Content -Encoding utf8 writes a BOM, which the engine's JSON parser refuses: the copy then + # read as defaults (no payout address, no cards) and the miners never started (runs 1 and 2, 5 and 6 October 2026) + [IO.File]::WriteAllText($sj, ($j | ConvertTo-Json -Depth 8), (New-Object System.Text.UTF8Encoding $false)) } catch { Say ("settings.json: " + $_.Exception.Message) } + $back = $null + try { $back = Get-Content -LiteralPath $sj -Raw | ConvertFrom-Json } catch { } + $addr = ''; if ($back) { $addr = [string]$back.address } + $bom = (Get-Content -LiteralPath $sj -Encoding Byte -TotalCount 3 -ErrorAction SilentlyContinue) -join ',' + Write-Output ('RESULT TUNE scratch settings: address ' + $(if ($addr) { $addr.Substring(0, [Math]::Min(10, $addr.Length)) + '...' } else { 'EMPTY' }) + ', cards ' + $(if ($back -and $back.cards) { @($back.cards.PSObject.Properties).Count } else { 0 }) + ', first bytes ' + $bom) + if (-not $addr -and -not (Test-Path (Join-Path $sApp 'wallet.json'))) { Write-Output 'RESULT TUNE error=no_address reason=the_copied_settings_carry_no_payout_address_and_no_wallet.json'; exit 2 } + if ($bom -eq '239,187,191') { Write-Output 'RESULT TUNE error=bom reason=settings.json_starts_with_a_BOM'; exit 2 } } else { Write-Output 'RESULT TUNE error=no_settings reason=the_installed_app_has_no_settings.json'; exit 2 } Remove-Item -LiteralPath (Join-Path $sApp 'app.url') -Force -ErrorAction SilentlyContinue diff --git a/relay/playbooks/sweep-5090.ps1 b/relay/playbooks/sweep-5090.ps1 index 8b68f7d0..29a9c7c0 100644 --- a/relay/playbooks/sweep-5090.ps1 +++ b/relay/playbooks/sweep-5090.ps1 @@ -43,7 +43,7 @@ if (Test-Path $sj) { try { $j = Get-Content -LiteralPath $sj -Raw | ConvertFrom-Json $j.remote_jobs = $false; $j.auto_update = $false; $j.prove = $false; $j.paused = $false; $j.setup_done = $true - $j | ConvertTo-Json -Depth 8 | Set-Content -LiteralPath $sj -Encoding utf8 + [IO.File]::WriteAllText($sj, ($j | ConvertTo-Json -Depth 8), (New-Object System.Text.UTF8Encoding $false)) # no BOM: the engine's JSON parser refuses one (C35, runs 1 and 2) } catch { Say ("settings.json: " + $_.Exception.Message) } } else { Write-Output 'RESULT SWEEP error=no_settings reason=the_installed_app_has_no_settings.json'; exit 2 } Remove-Item -LiteralPath (Join-Path $sApp 'app.url') -Force -ErrorAction SilentlyContinue diff --git a/tools/ci/second-engine-check.sh b/tools/ci/second-engine-check.sh index e5a42e31..da127dd0 100755 --- a/tools/ci/second-engine-check.sh +++ b/tools/ci/second-engine-check.sh @@ -20,6 +20,9 @@ while IFS= read -r f; do if ! grep -qE 'taskkill /T /F' "$f"; then echo "second-engine: $f starts an engine without ending its process tree (taskkill /T /F) at the end"; fail=1 fi + if grep -qE 'settings\.json|\.json' "$f" && grep -vE '^\s*#' "$f" | grep -qE 'Set-Content[^\n]*-Encoding +utf8'; then + echo "second-engine: $f writes JSON with Set-Content -Encoding utf8 (a BOM the engine refuses: the copy read as defaults, no payout address, nothing mined); use [IO.File]::WriteAllText with UTF8Encoding(\$false)"; fail=1 + fi if ! grep -qE "IGNEUM_APP_NO_OTA *= *'1'" "$f"; then echo "second-engine: $f starts an engine without IGNEUM_APP_NO_OTA = '1' (its updater would run the installer, which quits the installed app: PC 1, 5 October 2026, 22:31 UTC)"; fail=1 fi From 663fca4d76dfe18fd83c7d033661ddea4a261bc7 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Tue, 6 Oct 2026 07:59:21 +0000 Subject: [PATCH 17/20] bench log: run 2 (07:21 to 07:56Z): the BOM in the copied settings, no miner started, nothing set, mining paused 36 min 13 s, the hold released by the runner itself Co-Authored-By: Claude Fable 5.1 --- docs/bench-log.md | 2 ++ 1 file changed, 2 insertions(+) diff --git a/docs/bench-log.md b/docs/bench-log.md index 7853fa54..62b8b1b4 100644 --- a/docs/bench-log.md +++ b/docs/bench-log.md @@ -1632,5 +1632,7 @@ Branch `ember-tune` (54ff1bc), docs/plans/ember-tune.md. Every card tuned for MH | RX 9070 XT (bus 98, present again) | `tune 1 ... gmax 0 gmax_range -500 1000 plimit 0 plimit_range -30 10 factory 1 ok` | the helper's clock range is an OFFSET from stock in MHz, not a ceiling: a probe reading it as a 1,000 MHz maximum would have asked for `--set-gmax 900`, an overclock. Fixed at 054e041: an offset range closes the clock knob (until the stock clock is known) and the power ladder runs on the percent scale bounded by the range, so the 9070 XT's plan is 100, 90, 80, 70% (the -30 floor), 4 steps | | Radeon(TM) Graphics (integrated) | `tune 0 ... gmax - ... factory 0 ok` | no manual tuning: measure only, and it is off by default anyway | +**Run 2, 6 October 2026, 07:21 to 07:56Z (job ember-tune-pc1-2, elevated on the project lead's word, engine 25113f52..., PC 1 on 0.3.11):** the project lead answered the one prompt; the installed app stopped its miners at 07:21:16Z; the second engine ran for the whole 35-minute budget at "waiting, 0.00 MH/s" and no step ran. Cause: the playbook wrote the engine's copy of settings.json with PowerShell 5.1's `Set-Content -Encoding utf8`, which adds a UTF-8 BOM; the engine's JSON parser refuses it, `Settings::load` fell back to defaults (no payout address, no cards), the engine logged `[error] no payout address` and never started a miner. Run 1's scratch log carried the same line the night before. Readbacks, idle both times: the 5090 at 90.6 W before and 69.9 W after (2,505 then 2,407 MHz core, 14,001 MHz memory, limit 450 W of 575), the 9070 XT at factory (`gmax 0`, `plimit 0`). Nothing set on either card. The installed app's runner released the miners-stopped hold by itself on the failed exit (`job finished; the miners restart` at 07:56:50Z, both miners up by 07:57:04Z, `mining` at 07:57:29Z): mining paused 36 min 13 s. Fix 8273494: the copy is written without a BOM, the address is read back and the job fails within seconds if it is empty (`RESULT TUNE scratch settings: address ..., cards N, first bytes ...`), and the CI check fails any playbook writing JSON with `Set-Content -Encoding utf8`. The re-run needs one more click on the prompt. + Consequence for the tiers: an AMD card is tuned on its power limit alone until its stock core clock is read (a 9070 XT at -30% is the floor the driver allows, 4 steps, 5 minutes); every NVIDIA card's two-knob plan waits on the user's one click on Power control; the re-run on PC 1 is held until the quit's source is named (the event-log collect) and follows the 0.3.11 rollout (the update clears the jobs folder, so the engine and the helper are fetched again), with the scheduler's slot. From 6de4827ccb8896ba129d0d653e6f5ea44b6bda84 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Tue, 6 Oct 2026 08:03:13 +0000 Subject: [PATCH 18/20] a job that cannot mine never burns its budget silently: the elevated job path follows its output file while the script runs (the 5-minute progress reports carry the lines; 0.3.12), and the tune playbook's watchdog fails a run that mines nothing within 120 s of its first status line (the engine's last log line in the RESULT, the tree ended, mining restored by the runner) Co-Authored-By: Claude Fable 5.1 --- app/igneum-app/src/jobrun.rs | 44 ++++++++++++++++++++++++++++-- docs/plans/ember-tune.md | 8 ++++++ relay/playbooks/ember-tune-pc1.ps1 | 26 ++++++++++++++++++ 3 files changed, 76 insertions(+), 2 deletions(-) diff --git a/app/igneum-app/src/jobrun.rs b/app/igneum-app/src/jobrun.rs index bc833a21..28db2c7d 100644 --- a/app/igneum-app/src/jobrun.rs +++ b/app/igneum-app/src/jobrun.rs @@ -1131,10 +1131,17 @@ fn run_script(shared: &Arc, job: &Job, sink: &Sink, jobs_dir: &Path, dat } cmd.current_dir(&dir); job_env(&mut cmd, shared, job, &dir, data_root); + // the elevated script's output reaches this side through a file: follow it while the script runs, so the + // 5-minute progress reports carry its lines (6 October 2026: a 35-minute run that never mined showed only + // "script running" until it ended; the lines that said why were in the file the whole time) + let follow = if elevated { Some(follow_file(sink, out_file.clone())) } else { None }; let ran = run_streamed(&mut cmd, sink, ctl, limit, shared, job, started, "script running")?; - if elevated { + if let Some(f) = follow { + f.stop.store(true, std::sync::atomic::Ordering::Relaxed); + let seen = f.handle.join().unwrap_or(0); + // the tail the follower had not read when the script ended if let Ok(t) = std::fs::read_to_string(&out_file) { - for l in t.lines() { + for l in t.lines().skip(seen) { sink.line(l); } } @@ -1142,6 +1149,39 @@ fn run_script(shared: &Arc, job: &Job, sink: &Sink, jobs_dir: &Path, dat finish_ran(ran, "script") } +/// Follows a file another process writes (the elevated script's output), feeding each new complete line to the +/// sink every 2 s until stopped; returns how many lines it delivered, so the caller can hand over the remainder. +struct Follow { + stop: Arc, + handle: std::thread::JoinHandle, +} + +fn follow_file(sink: &Sink, path: PathBuf) -> Follow { + let stop = Arc::new(std::sync::atomic::AtomicBool::new(false)); + let stop2 = stop.clone(); + let s = Sink { shared: sink.shared.clone(), id: sink.id.clone(), dir: sink.dir.clone(), log_path: sink.log_path.clone(), file: Mutex::new(std::fs::OpenOptions::new().append(true).open(&sink.log_path).ok()), results: Mutex::new(Vec::new()) }; + let handle = std::thread::spawn(move || { + let mut seen = 0usize; + loop { + if let Ok(t) = std::fs::read_to_string(&path) { + let lines: Vec<&str> = t.lines().collect(); + // only complete lines (the writer may be mid-line): keep the last one for the next pass unless the + // text ends with a newline + let complete = if t.ends_with('\n') { lines.len() } else { lines.len().saturating_sub(1) }; + for l in lines.iter().take(complete).skip(seen) { + s.line(l); + } + seen = seen.max(complete); + } + if stop2.load(std::sync::atomic::Ordering::Relaxed) { + break seen; + } + std::thread::sleep(Duration::from_secs(2)); + } + }); + Follow { stop, handle } +} + fn finish_ran(ran: Ran, what: &str) -> Result { match ran.code { Some(0) => Ok(Done { status: "done".into(), exit: 0, summary: format!("{what} finished, exit 0"), extra: json!({}) }), diff --git a/docs/plans/ember-tune.md b/docs/plans/ember-tune.md index 6d38b142..275e2144 100644 --- a/docs/plans/ember-tune.md +++ b/docs/plans/ember-tune.md @@ -161,6 +161,14 @@ maximum, 14,001 MHz memory; 9070 XT present on bus 98 with OFFSET ranges `gmax_r 10`). The offset finding changed the AMD mapping (054e041): an offset clock range closes the clock knob and the power ladder runs on a percent scale bounded by `plimit_range`. The re-run follows the 0.3.11 rollout. +## 8a. Next-cut notes (for the 0.3.12 shipper) + +| Commit | What | Where | +|---|---|---| +| b671c8b | every `quit:` names its source; Power control alone decides; no cap at start under `--sweep` | main.rs, server.rs, engine.rs (separable) | +| e600e63 | a second engine never runs the updater (`IGNEUM_APP_NO_OTA`, implied by `--sweep`) | engine.rs (6 lines, separable) | +| 6a8297c | the elevated job path's output file is followed while the script runs, so the 5-minute progress reports carry its lines (a 35-minute run that never mined showed only "script running" on 6 October 2026); the tune playbook's watchdog fails a run that mines nothing within 120 s of its first status line, with the engine's last log line in the RESULT | jobrun.rs `follow_file`, relay/playbooks/ember-tune-pc1.ps1 | + ## 9. Open - The NVIDIA clock readback: `nvidia-smi -lgc` is confirmed only through the core clock during the hold (a mean over diff --git a/relay/playbooks/ember-tune-pc1.ps1 b/relay/playbooks/ember-tune-pc1.ps1 index a50b018f..f6ca8898 100644 --- a/relay/playbooks/ember-tune-pc1.ps1 +++ b/relay/playbooks/ember-tune-pc1.ps1 @@ -132,8 +132,34 @@ function EndTree([int] $procId, [string] $why) { } $seen = 0 $rows = 0 +# the watchdog (coordinator, 6 October 2026): a tune engine that mines nothing for 120 s after its first status line +# (every card "waiting" or 0.00 MH/s: no payout address, no worker, no node) fails the job at once with the engine's +# last log line in the RESULT, its tree ended, mining restored by the job runner; a job that cannot mine never burns +# its budget silently again +$firstStatusAt = $null +$lastMining = $null +$lastEngineLine = '' +function EngineTail() { $t = Get-ChildItem -Path $sLogs -Filter 'app-*.log' -ErrorAction SilentlyContinue | Sort-Object LastWriteTime -Descending | Select-Object -First 1; if ($t) { $l = Get-Content -LiteralPath $t.FullName -Tail 1 -ErrorAction SilentlyContinue; if ($l) { return [string]$l } }; return '' } while (-not $p.HasExited) { Start-Sleep -Seconds 5 + $tailLine = EngineTail + if ($tailLine) { $lastEngineLine = $tailLine } + if ($lastEngineLine -match ' status: ') { + if (-not $firstStatusAt) { $firstStatusAt = Get-Date } + if ($lastEngineLine -match ', mining \|' -or ($lastEngineLine -match '(\d+\.\d+) MH/s' -and [double]$Matches[1] -gt 0)) { $lastMining = Get-Date } + } + if ($firstStatusAt -and -not $lastMining -and ((Get-Date) - $firstStatusAt).TotalSeconds -gt 120) { + Write-Output ('RESULT TUNE error=not_mining reason=no_card_mined_within_120_s_of_the_first_status_line last_log_line=' + ($lastEngineLine -replace '\s+', '_')) + EndTree $p.Id 'watchdog: not mining' + Write-Output 'RESULT TUNE error=no_rows' + exit 3 + } + if ($lastMining -and ((Get-Date) - $lastMining).TotalSeconds -gt 300) { + Write-Output ('RESULT TUNE error=stopped_mining reason=every_card_idle_for_300_s last_log_line=' + ($lastEngineLine -replace '\s+', '_')) + EndTree $p.Id 'watchdog: stopped mining' + Write-Output 'RESULT TUNE error=no_rows' + exit 3 + } $all = @(); if (Test-Path -LiteralPath $outFile) { $all = @(Get-Content -LiteralPath $outFile -ErrorAction SilentlyContinue) } while ($seen -lt $all.Count) { $l = [string]$all[$seen]; $seen++ From ebedff9b665596d71232ed61a36c475ada96be8e Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Tue, 6 Oct 2026 08:03:19 +0000 Subject: [PATCH 19/20] ember-tune.md: the next-cut note names the right commit Co-Authored-By: Claude Fable 5.1 --- docs/plans/ember-tune.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/plans/ember-tune.md b/docs/plans/ember-tune.md index 275e2144..b0063f77 100644 --- a/docs/plans/ember-tune.md +++ b/docs/plans/ember-tune.md @@ -167,7 +167,7 @@ ladder runs on a percent scale bounded by `plimit_range`. The re-run follows the |---|---|---| | b671c8b | every `quit:` names its source; Power control alone decides; no cap at start under `--sweep` | main.rs, server.rs, engine.rs (separable) | | e600e63 | a second engine never runs the updater (`IGNEUM_APP_NO_OTA`, implied by `--sweep`) | engine.rs (6 lines, separable) | -| 6a8297c | the elevated job path's output file is followed while the script runs, so the 5-minute progress reports carry its lines (a 35-minute run that never mined showed only "script running" on 6 October 2026); the tune playbook's watchdog fails a run that mines nothing within 120 s of its first status line, with the engine's last log line in the RESULT | jobrun.rs `follow_file`, relay/playbooks/ember-tune-pc1.ps1 | +| 1e9550e (this commit, amended) | the elevated job path's output file is followed while the script runs, so the 5-minute progress reports carry its lines (a 35-minute run that never mined showed only "script running" on 6 October 2026); the tune playbook's watchdog fails a run that mines nothing within 120 s of its first status line, with the engine's last log line in the RESULT | jobrun.rs `follow_file`, relay/playbooks/ember-tune-pc1.ps1 | ## 9. Open From 27c2db64c30ee93b897a13ee09074d53c5fb6667 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Mon, 5 Oct 2026 18:51:24 +0000 Subject: [PATCH 20/20] app: every elevated launch through one hidden-console builder; CI check for Windows spawns; PC 1 console watchers The console-window class (the project lead, 5 October 2026: "Windows Command Processor" windows on PC 1 whenever a remote job runs). Measured on PC 1 (ae432dc7, Windows 11 Pro 26200, default terminal "Let Windows decide" = Windows Terminal 1.24) with tools/windows/console-watch.ps1 (job run-20261005-182528): no child a job script starts from the app's headless console opens a window (powershell, cmd, query, curl, nvidia-smi, wsl --status, a distro, interop cmd and powershell, powershell -WindowStyle Hidden: 0 windows each); Start-Process in a new console opens a Terminal window (the known-failed case: 2 windows), the same with -WindowStyle Hidden opens none (the known-finished case). The elevated path (Start-Process -Verb RunAs -WindowStyle Hidden through the AppInfo service) is the one road left; its watcher (console-watch-elevated.ps1, job run-20261005-184610) was cancelled at the UAC prompt. - platform.rs: elevated_ps_line + elevated_command build the one PowerShell line every elevated launch uses (the NVIDIA power cap, the sweep helper, the clock sync, an elevated remote job), -WindowStyle Hidden by construction; unit tests on the line, the quoting and the Command. - jobrun.rs: the elevated job path uses it; the relaunch helper's Start-Process carries the reason it has no -WindowStyle Hidden (igneum-app.exe is a windows-subsystem program). - tools/ci/windows-spawn-check.mjs (+ ci.yml): fails when a Command::new in app/igneum-app/src is not quieted, a creation_flags is not CREATE_NO_WINDOW alone, a Start-Process the Rust code writes lacks -WindowStyle Hidden or -NoNewWindow, or host.cpp spawns without CREATE_NO_WINDOW / SW_HIDE; self-test on known-good and known-bad samples. - tools/windows/console-watch.ps1, console-watch-bg.ps1, console-watch-elevated.ps1: the watchers (run jobs). Co-Authored-By: Claude Fable 5.1 --- .github/workflows/ci.yml | 2 + app/igneum-app/src/jobrun.rs | 49 ++++-- app/igneum-app/src/ota.rs | 2 + app/igneum-app/src/platform.rs | 11 +- app/igneum-app/src/wslhost.rs | 1 + docs/bugs.md | 1 + tools/ci/windows-spawn-check.mjs | 106 +++++++++++++ tools/windows/console-watch-bg.ps1 | 101 +++++++++++++ tools/windows/console-watch-elevated.ps1 | 84 +++++++++++ tools/windows/console-watch.ps1 | 181 +++++++++++++++++++++++ 10 files changed, 518 insertions(+), 20 deletions(-) create mode 100644 tools/ci/windows-spawn-check.mjs create mode 100644 tools/windows/console-watch-bg.ps1 create mode 100644 tools/windows/console-watch-elevated.ps1 create mode 100644 tools/windows/console-watch.ps1 diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index d79b80d1..6c7af009 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -77,6 +77,8 @@ jobs: run: bash tools/ci/bash-body-check.sh --self-test && bash tools/ci/bash-body-check.sh - name: run jobs test their fetched kit before use, the wiped-jobs-folder class (self-test first, then the tree) run: bash tools/ci/kit-path-check.sh --self-test && bash tools/ci/kit-path-check.sh + - name: every Windows spawn of the app runs with a hidden console (self-test first, then the tree) + run: node tools/ci/windows-spawn-check.mjs --self-test && node tools/ci/windows-spawn-check.mjs - name: pinned guest programs match their manifest and are built only by pin-guests.sh run: bash tools/ci/pinned-guests-check.sh - name: root prover playbooks kill the GPU server and unlink its socket (the root-socket class, 5 October 2026) diff --git a/app/igneum-app/src/jobrun.rs b/app/igneum-app/src/jobrun.rs index 41754779..3a6e2da5 100644 --- a/app/igneum-app/src/jobrun.rs +++ b/app/igneum-app/src/jobrun.rs @@ -1115,16 +1115,11 @@ fn run_script(shared: &Arc, job: &Job, sink: &Sink, jobs_dir: &Path, dat .map(|(k, v)| format!("$env:{k} = '{}'\r\n", v.replace('\'', "''"))) .collect(); let wrapper = dir.join("elevated.ps1"); - let w = format!("{env_lines}& '{}' *>&1 | Out-File -FilePath '{}' -Encoding utf8\r\nexit $LASTEXITCODE\r\n", script.display().to_string().replace('\'', "''"), out_file.display().to_string().replace('\'', "''")); + let w = elevated_wrapper(&env_lines, &script.display().to_string(), &out_file.display().to_string()); std::fs::write(&wrapper, [b"\xEF\xBB\xBF".as_slice(), w.as_bytes()].concat()).map_err(|e| e.to_string())?; let _ = std::fs::remove_file(&out_file); let inner = format!("-NoProfile -ExecutionPolicy Bypass -File \"{}\"", wrapper.display()); - // A refused or unanswered UAC prompt makes Start-Process throw (`$p` stays null) and `exit $p.ExitCode` - // would exit 0: the 5 October 2026 driver job on PC 1 was reported "done" after Windows cancelled its - // prompt at 122 s. The launch failure is exit 251 and says so on stderr. - let ps = format!("try {{ $p = Start-Process -FilePath powershell.exe -ArgumentList '{}' -Verb RunAs -Wait -WindowStyle Hidden -PassThru -ErrorAction Stop }} catch {{ Write-Error ('elevated launch failed (UAC refused, cancelled or timed out): ' + $_.Exception.Message); exit 251 }}; if ($null -eq $p) {{ Write-Error 'elevated launch failed: no process'; exit 251 }}; exit $p.ExitCode", inner.replace('\'', "''")); - cmd = Command::new(crate::platform::tool("powershell")); - cmd.args(["-NoProfile", "-ExecutionPolicy", "Bypass", "-Command", &ps]); + cmd = crate::platform::elevated_command("powershell.exe", &inner); } else if shell == "powershell" { cmd = Command::new(crate::platform::tool("powershell")); cmd.args(["-NoProfile", "-ExecutionPolicy", "Bypass", "-File", &script.display().to_string()]); @@ -1185,6 +1180,23 @@ fn follow_file(sink: &Sink, path: PathBuf) -> Follow { Follow { stop, handle } } +/// The PowerShell wrapper an elevated job runs (its own process, its own environment): the IGNEUM_* values, then one +/// line about its console (the elevated process cannot inherit the engine's headless console and gets one of its own; +/// `-WindowStyle Hidden` on the launch keeps it hidden, and this line is the running measurement of that on every +/// elevated job: "elevated console: hwnd N visible False"), then the script, everything into `out_file` for the engine +/// to read back. The console-window class, PC 1, 5 October 2026 (tools/windows/console-watch-elevated.ps1). +fn elevated_wrapper(env_lines: &str, script: &str, out_file: &str) -> String { + let (script, out) = (crate::platform::ps_quote(script), crate::platform::ps_quote(out_file)); + format!( + "{env_lines}$ErrorActionPreference = 'Continue'\r\n\ + $igc = ''\r\n\ + try {{ Add-Type -Name IgCon -Namespace Igneum -MemberDefinition '[DllImport(\"kernel32.dll\")] public static extern System.IntPtr GetConsoleWindow(); [DllImport(\"user32.dll\")] public static extern bool IsWindowVisible(System.IntPtr h);'; $h = [Igneum.IgCon]::GetConsoleWindow(); $igc = \"elevated console: hwnd $h visible $([Igneum.IgCon]::IsWindowVisible($h))\" }} catch {{ $igc = \"elevated console: unknown ($_)\" }}\r\n\ + $igc | Out-File -FilePath '{out}' -Encoding utf8\r\n\ + & '{script}' *>&1 | Out-File -FilePath '{out}' -Encoding utf8 -Append\r\n\ + exit $LASTEXITCODE\r\n" + ) +} + fn finish_ran(ran: Ran, what: &str) -> Result { match ran.code { Some(0) => Ok(Done { status: "done".into(), exit: 0, summary: format!("{what} finished, exit 0"), extra: json!({}) }), @@ -1311,10 +1323,24 @@ mod tests { assert!(d.summary.contains("administrator prompt"), "{}", d.summary); let d = finish_ran(Ran { code: Some(0), timed_out: false }, "script").unwrap(); assert_eq!(d.status, "done"); - // the launcher string itself: a thrown Start-Process must not fall through to `exit $p.ExitCode` - let src = include_str!("jobrun.rs"); - assert!(src.contains("-Verb RunAs -Wait -WindowStyle Hidden -PassThru -ErrorAction Stop }} catch {{")); - assert!(src.contains("if ($null -eq $p) {{ Write-Error 'elevated launch failed: no process'; exit 251 }}")); + // the launcher string itself (platform::elevated_ps_line since 13755b9): a thrown Start-Process must not fall + // through to `exit $p.ExitCode` + let l = crate::platform::elevated_ps_line("powershell.exe", "-NoProfile -File x.ps1"); + assert!(l.contains("-Verb RunAs -Wait -WindowStyle Hidden -PassThru -ErrorAction Stop } catch {"), "{l}"); + assert!(l.contains("if ($null -eq $p) { Write-Error 'elevated launch failed: no process'; exit 251 }"), "{l}"); + } + + #[test] + fn elevated_wrapper_reports_its_console_then_runs_the_script() { + let w = elevated_wrapper("$env:IGNEUM_JOB_ID = 'j1'\r\n", r"C:\jobs\j1\script.ps1", r"C:\jobs\it's\elevated-output.log"); + assert!(w.starts_with("$env:IGNEUM_JOB_ID = 'j1'\r\n$ErrorActionPreference = 'Continue'\r\n"), "{w}"); + assert!(w.contains("GetConsoleWindow()") && w.contains("IsWindowVisible("), "{w}"); + assert!(w.contains("$igc | Out-File -FilePath 'C:\\jobs\\it''s\\elevated-output.log' -Encoding utf8\r\n"), "{w}"); + assert!(w.contains("& 'C:\\jobs\\j1\\script.ps1' *>&1 | Out-File -FilePath 'C:\\jobs\\it''s\\elevated-output.log' -Encoding utf8 -Append\r\n"), "{w}"); + assert!(w.ends_with("exit $LASTEXITCODE\r\n"), "{w}"); + // every line ends in CRLF (the file is written for Windows PowerShell): the env line and six of its own + assert_eq!(w.matches("\r\n").count(), 7, "{w:?}"); + assert_eq!(w.matches('\n').count(), 7, "{w:?}"); } #[test] @@ -1801,6 +1827,7 @@ fn spawn_relaunch_helper(shared: &Arc) -> Result<(), String> { { let dir = std::env::current_exe().ok().and_then(|p| p.parent().map(|d| d.to_path_buf())).ok_or("cannot find the install folder")?; let exe = dir.join("igneum-app.exe"); + // console: igneum-app.exe is a windows-subsystem program in release builds (main.rs), it never gets a console; SW_HIDE would hide the window host it opens let ps = format!("Start-Sleep 8; Start-Process -FilePath '{}' -ArgumentList '--launch' -WorkingDirectory '{}'", exe.display().to_string().replace('\'', "''"), dir.display().to_string().replace('\'', "''")); c = Command::new(crate::platform::tool("powershell")); c.args(["-NoProfile", "-ExecutionPolicy", "Bypass", "-WindowStyle", "Hidden", "-Command", &ps]); diff --git a/app/igneum-app/src/ota.rs b/app/igneum-app/src/ota.rs index f119b254..81dde696 100644 --- a/app/igneum-app/src/ota.rs +++ b/app/igneum-app/src/ota.rs @@ -1278,6 +1278,7 @@ function EngineAlive() { return [bool](Get-Process -Id $EnginePid -ErrorAction S function Relaunch() { if (EngineAlive) { return } $exe = Join-Path $InstallDir 'igneum-app.exe' + # console: igneum-app.exe is a windows-subsystem program (no console); -WindowStyle Hidden would hide the window host it opens if (Test-Path $exe) { Log 'engine gone and nothing installed: starting the old app again'; Start-Process -FilePath $exe -ArgumentList '--launch' -WorkingDirectory $InstallDir | Out-Null } } Log "$Mode : engine $EnginePid installer '$Installer' version $Version (the engine keeps mining until the installer runs)" @@ -1292,6 +1293,7 @@ $setupArgs = @('/VERYSILENT', '/SUPPRESSMSGBOXES', '/NORESTART', '/CLOSEAPPLICAT try { # no -Verb RunAs: a per-user installer just runs; an administrator installer makes Windows ask, and a declined or # timed-out prompt comes back here as an exception with the engine still mining + # console: the Inno Setup installer is a GUI program (no console), /VERYSILENT shows nothing $p = Start-Process -FilePath $Installer -ArgumentList $setupArgs -Wait -PassThru if ($p.ExitCode -eq 0) { if ($Mode -eq 'rollback') { Done $false "Igneum Miner $Version did not stay up twice; the previous version was reinstalled" $true $false } diff --git a/app/igneum-app/src/platform.rs b/app/igneum-app/src/platform.rs index d41c26d5..a3e8e694 100644 --- a/app/igneum-app/src/platform.rs +++ b/app/igneum-app/src/platform.rs @@ -396,10 +396,7 @@ pub fn sync_clock() -> Result { #[cfg(windows)] { let cmd = tool("cmd").display().to_string(); - let script = format!("Start-Process -FilePath '{cmd}' -ArgumentList '/c net start w32time & w32tm /resync /force' -Verb RunAs -Wait -WindowStyle Hidden"); - let mut c = Command::new(tool("powershell")); - c.args(["-NoProfile", "-ExecutionPolicy", "Bypass", "-Command", &script]); - quiet(&mut c); + let mut c = elevated_command(&cmd, "/c net start w32time & w32tm /resync /force"); let out = c.output().map_err(|e| e.to_string())?; if out.status.success() { Ok("asked Windows Time to resync (w32tm /resync)".into()) @@ -425,12 +422,8 @@ pub fn sync_clock() -> Result { pub fn run_elevated(cmdline: &str) -> Result<(), String> { #[cfg(windows)] { - let escaped = cmdline.replace('\'', "''"); let cmd = tool("cmd").display().to_string(); - let script = format!("$p = Start-Process -FilePath '{cmd}' -ArgumentList '/c {escaped}' -Verb RunAs -Wait -WindowStyle Hidden -PassThru; exit $p.ExitCode"); - let mut c = Command::new(tool("powershell")); - c.args(["-NoProfile", "-ExecutionPolicy", "Bypass", "-Command", &script]); - quiet(&mut c); + let mut c = elevated_command(&cmd, &format!("/c {cmdline}")); let out = c.output().map_err(|e| e.to_string())?; if out.status.success() { Ok(()) diff --git a/app/igneum-app/src/wslhost.rs b/app/igneum-app/src/wslhost.rs index f36ff604..9bee6d04 100644 --- a/app/igneum-app/src/wslhost.rs +++ b/app/igneum-app/src/wslhost.rs @@ -189,6 +189,7 @@ pub fn bash_line(file: &Path, login: bool, args: &[&str]) -> String { /// command line exactly as `bash_line` wrote it; elsewhere the words are ordinary arguments (nothing runs wsl there). /// The caller adds stdio, the hidden-window flag and the timeout. pub fn command(wsl_exe: &Path, distro: &str, user: Option<&str>, file: &Path, login: bool, args: &[&str]) -> Command { + // console: a builder; every caller runs it through run_capture, run_streamed or platform::quiet (tools/ci/windows-spawn-check.mjs) let mut c = Command::new(wsl_exe); c.args(["-d", distro]); if let Some(u) = user.filter(|u| !u.is_empty()) { diff --git a/docs/bugs.md b/docs/bugs.md index 4ad0e819..4b551ae5 100644 --- a/docs/bugs.md +++ b/docs/bugs.md @@ -7,6 +7,7 @@ shard run reported as exit 0, 7a7e873). | Date | Symptom | Cause | Fix | Proven by | |---|---|---|---|---| | 4 Oct 2026 | Every `ci` run on master red since 67bf226 (eleven pushes), unnoticed | `sim/difficulty/records/testnet-v2-2026-10-04.schedule.log` carried a home path; `.log` was outside the identity scrub's extension list in `tools/ci/identity-check.sh` (and in the mirror's `tools/sync.sh`) | 2996cca: `.log` scrubbed like the other text files; the record rewritten with `~`; the same list in igneum-public `tools/sync.sh` (local commit e18256d, not pushed) | `bash tools/ci/identity-check.sh` 0 hits locally; run 37226816xxx on master green | +| 5 Oct 2026 | PC 1 (Windows 11 Pro 26200, default terminal Windows Terminal 1.24): "Windows Command Processor" windows whenever a remote job runs (the project lead) | measured, not guessed: `tools/windows/console-watch.ps1` (job run-20261005-182528) started every candidate child from the app's job runner, whose console is headless (`conhost.exe 0x4`, hwnd 0), with a user32 EnumWindows sampler every 30 ms: powershell, cmd, query, curl, nvidia-smi, wsl --status, a distro, interop cmd and powershell, `powershell -WindowStyle Hidden`, `Start-Process -WindowStyle Hidden`: 0 windows each; `Start-Process cmd` in a new console: a Terminal window and a cmd PseudoConsoleWindow (the known-failed case fires). The 25-minute background watcher (console-watch-bg.ps1, run-20261005-184330, 18:44 to 19:09 UTC, every 200 ms) across an app restart, a build job, two run jobs, two collect jobs and the sweep helper's elevated launch at 19:04:43: 0 console or Terminal windows, 69 conhost starts (every one `conhost.exe 0x4`, headless, under curl, wsl, wslhost, powershell), 1 cmd.exe (under wslhost, WSL interop, no window). The one road that creates a console of its own is the elevated launch (`Start-Process -Verb RunAs`, the AppInfo service: the power cap, the sweep helper, the clock sync, an elevated job); it carried `-WindowStyle Hidden` in four copies, and "Windows Command Processor" is also the name on the UAC prompt the engine raises for cmd.exe (the sweep helper prompted at 17:00, 17:30 and 18:12 UTC, the power cap at every start; the elevated watcher's own prompt, run-20261005-184610, timed out unanswered at 122 s) | `platform::elevated_ps_line` + `elevated_command`: one builder for every elevated launch, hidden by construction, exit 251 when the prompt is refused; the elevated job wrapper reports its own console (`elevated console: hwnd N visible False`) on every elevated job; `tools/ci/windows-spawn-check.mjs` fails CI on a Command::new without the quiet flag, a creation_flags other than CREATE_NO_WINDOW, a Start-Process without -WindowStyle Hidden/-NoNewWindow, or a host.cpp spawn without CREATE_NO_WINDOW / SW_HIDE | the watcher's known-failed case (2 windows) and known-finished case (0); the CI check's self-test (9 cases) and the tree (0 hits); the igneum-app test suite on PC 1 | | 4 Oct 2026 | `collect-pc1-board3` printed PowerShell parse errors (`.Name`, `.AdapterRAM`) | the publishing shell expanded `$_` inside double quotes to nothing before the command reached the jobs file; nothing to do with Format-List or Out-String (board2 and board4 printed their values) | publish-jobs.sh refuses a collect command that pipes into a script block without `$_` or `$PSItem` | the eaten form refused with the reason, the single-quoted form published to a test folder | | 4 Oct 2026 | the same job reported `done (exit 0)` over `command exit Some(1)` | `run_collect` in `app/igneum-app/src/jobrun.rs` builds `Done` from the upload count only; the command's exit code is logged and dropped | branch `bugfix-collect-exit`, 35ccdc8 rebased on c257444 (app engine; merge by the main session) | `cargo test --bin igneum-app`: all 28 tests pass on the rebased branch; the new one covers the board3 shape (`Some(1)` is failed exit 1), `Some(0)` done, the cap as timeout, failed uploads still failing | | 4 Oct 2026 | `publish-jobs.sh --deploy` said "not reachable, differs from the local one, or does not verify yet" after a deploy that had succeeded | one check the instant the CLI returned, while the edge still served the previous file; the deploy's own exit status was hidden by `\|\| true` | `verify_live`: up to `--tries` (12) checks 5 s apart, each failure names its condition; `publish-jobs.sh verify` re-checks on its own; a failed deploy stops before the check | finished: `verify --tries 2` against the live file (try 1 of 2); failed: a local server with an older file ("differs", both publish stamps named) and a closed port ("is not reachable") | diff --git a/tools/ci/windows-spawn-check.mjs b/tools/ci/windows-spawn-check.mjs new file mode 100644 index 00000000..898a87ba --- /dev/null +++ b/tools/ci/windows-spawn-check.mjs @@ -0,0 +1,106 @@ +#!/usr/bin/env node +// The console-window class (5 October 2026, PC 1): a child the app starts on Windows without CREATE_NO_WINDOW, or an +// elevated child started without SW_HIDE, gets a console of its own, and on Windows 11 with Windows Terminal as the +// default terminal that console is a visible Terminal window on the user's desk. Rule: every process the app starts +// on Windows runs with a hidden console. This check fails CI when +// - a `Command::new(` in app/igneum-app/src is not quieted within 20 lines: crate::platform::quiet, run_timeout, +// run_capture, run_streamed, spawn_detached, elevated_command, or creation_flags(0x0800_0000) (CREATE_NO_WINDOW); +// programs that only exist off Windows (nohup, osascript, pkexec, hdiutil, ...) are allowed, and a +// `// console: ` comment on the line or the line above allows a builder the caller quiets; +// the check looks 2 lines back as well, for `run_timeout(\n Command::new(...)`; +// - a `creation_flags(` carries anything but 0x0800_0000 (DETACHED_PROCESS made powershell exit at start-up, +// 0.3.0 to 0.3.4, docs/bugs.md); +// - a PowerShell `Start-Process` written by the Rust code lacks `-WindowStyle Hidden` or `-NoNewWindow` (a GUI +// program, which never gets a console, takes a `# console: ` comment on the line or the line above); +// - app/windows/host.cpp calls CreateProcessW without CREATE_NO_WINDOW or sets up a ShellExecuteExW without +// nShow = SW_HIDE (ShellExecuteW "open" of a URL is the browser, allowed). +// node tools/ci/windows-spawn-check.mjs the tree +// node tools/ci/windows-spawn-check.mjs --self-test the rules on known-good and known-bad samples +import { readFileSync, readdirSync, statSync } from 'node:fs'; +import { join, resolve, dirname, relative } from 'node:path'; +import { fileURLToPath } from 'node:url'; + +const ROOT = resolve(dirname(fileURLToPath(import.meta.url)), '..', '..'); +const QUIET = /\bquiet\(|\brun_timeout\(|\brun_capture\(|\brun_streamed\(|\bspawn_detached\(|\belevated_command\(|creation_flags\(0x0800_0000\)/; +const OFF_WINDOWS = /Command::new\((?:crate::platform::)?tool\("(?:nohup|osascript|pkexec|hdiutil|ditto|open|xattr|system_profiler|sysctl|sntp|scutil|caffeinate)"\)|Command::new\("(?:pkexec|xdg-open|\/bin\/bash|\/usr\/bin\/[a-z]+)"\)|Command::new\(staged\.join\("Contents\/MacOS/; +const WINDOW = 20; + +export function checkRust(text, file) { + const lines = text.split('\n'); + const out = []; + for (let i = 0; i < lines.length; i++) { + const l = lines[i]; + if (/Command::new\(/.test(l) && !/^\s*\/\//.test(l)) { + const allowed = OFF_WINDOWS.test(l) || /\/\/ console:/.test(l) || (i > 0 && /\/\/ console:/.test(lines[i - 1])); + if (!allowed) { + const span = lines.slice(Math.max(0, i - 2), i + WINDOW).join('\n'); // 2 lines back: run_timeout(\n Command::new(...) + if (!QUIET.test(span)) out.push(`${file}:${i + 1}: Command::new without a hidden console within ${WINDOW} lines (quiet, run_timeout, run_capture, run_streamed, spawn_detached, elevated_command or creation_flags(0x0800_0000)); add one, or a '// console: ' comment`); + } + } + const cf = /creation_flags\(([^)]*)\)/.exec(l); + if (cf && cf[1].trim() !== '0x0800_0000') out.push(`${file}:${i + 1}: creation_flags(${cf[1]}) is not CREATE_NO_WINDOW alone (0x0800_0000)`); + if (/Start-Process\b/.test(l) && !/^\s*\/\//.test(l) && !/-WindowStyle Hidden|-NoNewWindow/.test(l) && !/(\/\/|#) console:/.test(l) && !(i > 0 && /(\/\/|#) console:/.test(lines[i - 1]))) out.push(`${file}:${i + 1}: Start-Process without -WindowStyle Hidden or -NoNewWindow (a GUI program takes a '# console: ' comment)`); + } + return out; +} + +export function checkHost(text, file) { + const lines = text.split('\n'); + const out = []; + for (let i = 0; i < lines.length; i++) { + const l = lines[i]; + if (/CreateProcessW?\s*\(/.test(l) && !/CREATE_NO_WINDOW/.test(lines.slice(i, i + 3).join('\n'))) out.push(`${file}:${i + 1}: CreateProcess without CREATE_NO_WINDOW`); + if (/ShellExecuteExW?\s*\(/.test(l) && !/nShow\s*=\s*SW_HIDE/.test(lines.slice(Math.max(0, i - 12), i + 1).join('\n'))) out.push(`${file}:${i + 1}: ShellExecuteEx without nShow = SW_HIDE in the 12 lines before it`); + if (/ShellExecuteW?\s*\(/.test(l) && !/ShellExecuteExW?/.test(l) && !/L"open"/.test(l)) out.push(`${file}:${i + 1}: ShellExecute that is not the browser "open" of a URL`); + } + return out; +} + +function walk(dir, ext, acc = []) { + for (const e of readdirSync(dir)) { + const p = join(dir, e); + if (statSync(p).isDirectory()) { if (e !== 'target') walk(p, ext, acc); } else if (p.endsWith(ext)) acc.push(p); + } + return acc; +} + +function selfTest() { + const good = `fn a() {\n let mut c = Command::new(crate::platform::tool("powershell"));\n c.args(["-NoProfile"]);\n crate::platform::quiet(&mut c);\n c.spawn();\n}\n`; + const bad = `fn a() {\n let mut c = Command::new(crate::platform::tool("powershell"));\n c.args(["-NoProfile"]);\n c.spawn();\n}\n`; + const badFlag = `c.creation_flags(0x0000_0008);\n`; + const badPs = `let ps = format!("Start-Process -FilePath '{}' -Wait", exe);\n`; + const okPs = `let ps = format!("Start-Process -FilePath '{}' -Wait -WindowStyle Hidden", exe);\n`; + const offWin = `let out = Command::new(tool("osascript")).args(["-e", "x"]).output();\n`; + const allowed = `// console: the caller quiets it\nlet mut c = Command::new(wsl_exe);\n`; + const hostGood = `sei.nShow = SW_HIDE;\nif (!ShellExecuteExW(&sei)) {}\nCreateProcessW(exe, buf, nullptr, nullptr, TRUE, CREATE_NO_WINDOW, nullptr, dir, &si, &pi);\nShellExecuteW(nullptr, L"open", url, nullptr, nullptr, SW_SHOWNORMAL);\n`; + const hostBad = `sei.nShow = SW_SHOW;\nif (!ShellExecuteExW(&sei)) {}\nCreateProcessW(exe, buf, nullptr, nullptr, TRUE, 0, nullptr, dir, &si, &pi);\n`; + const cases = [ + ['quieted Command', checkRust(good, 't.rs').length === 0], + ['bare Command fails', checkRust(bad, 't.rs').length === 1], + ['DETACHED_PROCESS fails', checkRust(badFlag, 't.rs').length === 1], + ['Start-Process without Hidden fails', checkRust(badPs, 't.rs').length === 1], + ['Start-Process with Hidden passes', checkRust(okPs, 't.rs').length === 0], + ['off-Windows program passes', checkRust(offWin, 't.rs').length === 0], + ['console: comment passes', checkRust(allowed, 't.rs').length === 0], + ['host.cpp good passes', checkHost(hostGood, 'h.cpp').length === 0], + ['host.cpp bad fails twice', checkHost(hostBad, 'h.cpp').length === 2], + ]; + let fail = 0; + for (const [name, ok] of cases) { console.log(`${ok ? 'ok ' : 'FAIL'} ${name}`); if (!ok) fail++; } + return fail; +} + +if (process.argv[1] && resolve(process.argv[1]) === fileURLToPath(import.meta.url)) { + if (process.argv.includes('--self-test')) { + const f = selfTest(); + console.log(f ? `windows-spawn: self-test FAILED (${f})` : 'windows-spawn: self-test ok'); + process.exit(f ? 1 : 0); + } + const problems = []; + for (const f of walk(join(ROOT, 'app', 'igneum-app', 'src'), '.rs')) problems.push(...checkRust(readFileSync(f, 'utf8'), relative(ROOT, f))); + const host = join(ROOT, 'app', 'windows', 'host.cpp'); + try { problems.push(...checkHost(readFileSync(host, 'utf8'), relative(ROOT, host))); } catch {} + for (const p of problems) console.log(p); + console.log(problems.length ? `windows-spawn: ${problems.length} spawn(s) without a hidden console` : 'windows-spawn: every Windows spawn in app/igneum-app/src and app/windows/host.cpp runs with a hidden console'); + process.exit(problems.length ? 1 : 0); +} diff --git a/tools/windows/console-watch-bg.ps1 b/tools/windows/console-watch-bg.ps1 new file mode 100644 index 00000000..4a2056f9 --- /dev/null +++ b/tools/windows/console-watch-bg.ps1 @@ -0,0 +1,101 @@ +# Background console-window watcher for a Windows PC running the Igneum Miner app: the second half of +# tools/windows/console-watch.ps1. That one proved (PC 1, 5 October 2026, job run-20261005-182528) that no child a +# job script starts from the app's headless console opens a window; this one finds what does. A signed `run` job +# starts a detached PowerShell (Start-Process -WindowStyle Hidden, the shape the first watcher showed opens nothing) +# and returns at once; the detached process samples for WATCH_MINUTES and writes \console-windows.log: +# window pid

[] a new visible console or terminal window +# <utc> proc <name> pid <p> cmd <command line> <- <parent chain, name pid and command line each> +# a new cmd, powershell, wsl, conhost, OpenConsole or WindowsTerminal +# <utc> gone <name> pid <p> one of those ended (the window's life) +# Read it back with a collect job: packaging/ota/publish-jobs.sh add --kind collect --target ae432dc7 \ +# --glob "app/jobs/<this job id>/console-windows.log" --deploy +$ErrorActionPreference = 'Continue' +$minutes = 25 +$dir = $env:IGNEUM_JOB_DIR +$log = Join-Path $dir 'console-windows.log' +$bg = Join-Path $dir 'bg.ps1' +# the detached body: params first (PowerShell wants them at the top), then the sampler +$body = @' +param([string] $Log, [int] $Minutes) +$ErrorActionPreference = 'Continue' +$src = @" +using System; +using System.Collections.Generic; +using System.Runtime.InteropServices; +using System.Text; +public static class IgWin2 { + public delegate bool EnumProc(IntPtr h, IntPtr l); + [DllImport("user32.dll")] public static extern bool EnumWindows(EnumProc p, IntPtr l); + [DllImport("user32.dll")] public static extern bool IsWindowVisible(IntPtr h); + [DllImport("user32.dll")] public static extern uint GetWindowThreadProcessId(IntPtr h, out uint pid); + [DllImport("user32.dll", CharSet = CharSet.Unicode)] public static extern int GetWindowText(IntPtr h, StringBuilder s, int n); + [DllImport("user32.dll", CharSet = CharSet.Unicode)] public static extern int GetClassName(IntPtr h, StringBuilder s, int n); + public static List<string> Visible() { + var list = new List<string>(); + EnumWindows((h, l) => { + if (!IsWindowVisible(h)) return true; + uint pid; GetWindowThreadProcessId(h, out pid); + var t = new StringBuilder(512); GetWindowText(h, t, 512); + var c = new StringBuilder(256); GetClassName(h, c, 256); + list.Add(((long)h).ToString() + "|" + pid + "|" + c + "|" + t); + return true; + }, IntPtr.Zero); + return list; + } +} +"@ +Add-Type -TypeDefinition $src +function Now() { (Get-Date).ToUniversalTime().ToString('HH:mm:ss.fff') } +function Put([string] $l) { Add-Content -Path $Log -Value $l -Encoding UTF8 } +function Chain([int] $procId) { + $out = @(); $seen = @{}; $p = $procId + for ($i = 0; $i -lt 6 -and $p -gt 0 -and -not $seen.ContainsKey($p); $i++) { + $seen[$p] = 1 + $ci = Get-CimInstance Win32_Process -Filter "ProcessId=$p" -ErrorAction SilentlyContinue + if (-not $ci) { $out += ("pid " + $p + " gone"); break } + $cl = [string]$ci.CommandLine; if ($cl.Length -gt 160) { $cl = $cl.Substring(0, 160) + '...' } + $out += ($ci.Name + " pid " + $p + " [" + $cl + "]") + $p = $ci.ParentProcessId + } + return ($out -join ' <- ') +} +$watch = 'cmd', 'powershell', 'pwsh', 'wsl', 'wslhost', 'conhost', 'OpenConsole', 'WindowsTerminal' +$classes = 'ConsoleWindowClass', 'CASCADIA_HOSTING_WINDOW_CLASS', 'PseudoConsoleWindow' +$knownWin = @{}; $knownProc = @{} +$first = $true +$end = (Get-Date).AddMinutes($Minutes) +Put ((Now) + " start: watching for " + $Minutes + " min, pid " + $PID) +while ((Get-Date) -lt $end) { + try { + foreach ($w in [IgWin2]::Visible()) { + $f = $w.Split('|', 4) + if ($knownWin.ContainsKey($f[0])) { continue } + $knownWin[$f[0]] = 1 + if ($first) { continue } + $pn = try { (Get-Process -Id ([int]$f[1]) -ErrorAction Stop).ProcessName } catch { 'gone' } + if ($classes -contains $f[2] -or $watch -contains $pn) { Put ((Now) + " window " + $pn + " pid " + $f[1] + " [" + $f[2] + "] " + $f[3]) } + } + $live = @{} + foreach ($p in @(Get-Process -Name $watch -ErrorAction SilentlyContinue)) { + $live[$p.Id] = 1 + if ($knownProc.ContainsKey($p.Id)) { continue } + $knownProc[$p.Id] = $p.ProcessName + if ($first) { continue } + Put ((Now) + " proc " + $p.ProcessName + " pid " + $p.Id + " " + (Chain $p.Id)) + } + foreach ($k in @($knownProc.Keys)) { if (-not $live.ContainsKey($k)) { if (-not $first) { Put ((Now) + " gone " + $knownProc[$k] + " pid " + $k) }; $knownProc.Remove($k) } } + if ($first) { Put ((Now) + " baseline: " + $knownWin.Count + " visible windows, " + $knownProc.Count + " watched processes: " + (($knownProc.GetEnumerator() | ForEach-Object { $_.Value + ' ' + $_.Key }) -join ', ')) } + $first = $false + } catch { Put ((Now) + " error " + $_) } + Start-Sleep -Milliseconds 200 +} +Put ((Now) + " end") +'@ +[IO.File]::WriteAllText($bg, $body, (New-Object System.Text.UTF8Encoding($true))) +if (Test-Path $log) { Remove-Item $log -Force } +$p = Start-Process -FilePath powershell.exe -ArgumentList @('-NoProfile', '-ExecutionPolicy', 'Bypass', '-File', $bg, '-Log', $log, '-Minutes', $minutes) -WindowStyle Hidden -PassThru +Start-Sleep -Seconds 3 +$alive = try { -not (Get-Process -Id $p.Id -ErrorAction Stop).HasExited } catch { $false } +Write-Output ("RESULT watcher: pid " + $p.Id + " alive " + $alive + " for " + $minutes + " min, log " + $log) +if (Test-Path $log) { Get-Content $log | ForEach-Object { Write-Output ("RESULT first: " + $_) } } +exit $(if ($alive) { 0 } else { 1 }) diff --git a/tools/windows/console-watch-elevated.ps1 b/tools/windows/console-watch-elevated.ps1 new file mode 100644 index 00000000..ea83b435 --- /dev/null +++ b/tools/windows/console-watch-elevated.ps1 @@ -0,0 +1,84 @@ +# Console-window watcher for the ELEVATED job path (app/igneum-app/src/jobrun.rs run_script with elevated=true: the +# app's headless powershell runs `Start-Process powershell.exe -Verb RunAs -Wait -WindowStyle Hidden`, the AppInfo +# service creates this process after the UAC prompt). The engine's power cap, the sweep helper and the clock sync take +# the same road with cmd.exe (platform.rs run_elevated, sync_clock; app/windows/host.cpp runElevated). This script +# runs INSIDE the elevated process and reports whether its own console has a window, which host serves it, and +# whether a Windows Terminal window appeared for it. One UAC prompt on the PC; a few seconds. +# packaging/ota/publish-jobs.sh add --kind run --target ae432dc7 --elevated --timeout-minutes 3 \ +# --script tools/windows/console-watch-elevated.ps1 --title "PC 1: elevated console watcher" --deploy +$ErrorActionPreference = 'Continue' +$src = @' +using System; +using System.Collections.Generic; +using System.Runtime.InteropServices; +using System.Text; +public static class IgWin3 { + public delegate bool EnumProc(IntPtr h, IntPtr l); + [DllImport("user32.dll")] public static extern bool EnumWindows(EnumProc p, IntPtr l); + [DllImport("user32.dll")] public static extern bool IsWindowVisible(IntPtr h); + [DllImport("user32.dll")] public static extern uint GetWindowThreadProcessId(IntPtr h, out uint pid); + [DllImport("user32.dll", CharSet = CharSet.Unicode)] public static extern int GetWindowText(IntPtr h, StringBuilder s, int n); + [DllImport("user32.dll", CharSet = CharSet.Unicode)] public static extern int GetClassName(IntPtr h, StringBuilder s, int n); + [DllImport("kernel32.dll")] public static extern IntPtr GetConsoleWindow(); + public static List<string> Visible() { + var list = new List<string>(); + EnumWindows((h, l) => { + if (!IsWindowVisible(h)) return true; + uint pid; GetWindowThreadProcessId(h, out pid); + var t = new StringBuilder(512); GetWindowText(h, t, 512); + var c = new StringBuilder(256); GetClassName(h, c, 256); + list.Add(((long)h).ToString() + "|" + pid + "|" + c + "|" + t); + return true; + }, IntPtr.Zero); + return list; + } +} +'@ +Add-Type -TypeDefinition $src +function Say([string] $m) { Write-Output $m } +function ProcName([int] $procId) { try { (Get-Process -Id $procId -ErrorAction Stop).ProcessName } catch { 'gone' } } +function Chain([int] $procId) { + $out = @(); $seen = @{}; $p = $procId + for ($i = 0; $i -lt 6 -and $p -gt 0 -and -not $seen.ContainsKey($p); $i++) { + $seen[$p] = 1 + $ci = Get-CimInstance Win32_Process -Filter "ProcessId=$p" -ErrorAction SilentlyContinue + if (-not $ci) { $out += ("pid " + $p + " gone"); break } + $cl = [string]$ci.CommandLine; if ($cl.Length -gt 140) { $cl = $cl.Substring(0, 140) + '...' } + $out += ($ci.Name + " pid " + $p + " [" + $cl + "]") + $p = $ci.ParentProcessId + } + return ($out -join ' <- ') +} +$me = [System.Diagnostics.Process]::GetCurrentProcess() +$id = [Security.Principal.WindowsIdentity]::GetCurrent() +$admin = (New-Object Security.Principal.WindowsPrincipal($id)).IsInRole([Security.Principal.WindowsBuiltInRole]::Administrator) +Say ("RESULT elevated: " + $admin + " user " + $id.Name + " session " + $me.SessionId + " chain " + (Chain $me.Id)) +$hwnd = [IgWin3]::GetConsoleWindow() +$cls = '' +if ($hwnd -ne [IntPtr]::Zero) { $sb = New-Object System.Text.StringBuilder 256; [void][IgWin3]::GetClassName($hwnd, $sb, 256); $cls = $sb.ToString() } +$vis = if ($hwnd -ne [IntPtr]::Zero) { [IgWin3]::IsWindowVisible($hwnd) } else { 'no window' } +Say ("RESULT self-console: hwnd " + $hwnd + " class [" + $cls + "] visible " + $vis) +# the hosts that serve this process: a conhost with this pid as parent (classic), or an OpenConsole + WindowsTerminal +# pair started by svchost in the last seconds (the default-terminal handoff) +$since = (Get-Date).AddSeconds(-20) +foreach ($h in @(Get-CimInstance Win32_Process -Filter "Name='conhost.exe' OR Name='OpenConsole.exe' OR Name='WindowsTerminal.exe'" -ErrorAction SilentlyContinue)) { + $created = try { [Management.ManagementDateTimeConverter]::ToDateTime($h.CreationDate) } catch { $null } + if ($h.ParentProcessId -eq $me.Id -or ($created -and $created -gt $since)) { + Say ("RESULT host: " + $h.Name + " pid " + $h.ProcessId + " parent " + (ProcName $h.ParentProcessId) + " started " + $(if ($created) { $created.ToUniversalTime().ToString('HH:mm:ss') } else { '?' }) + " cmd " + $h.CommandLine) + } +} +$wins = @([IgWin3]::Visible() | Where-Object { $f = $_.Split('|', 4); $f[2] -eq 'CASCADIA_HOSTING_WINDOW_CLASS' -or $f[2] -eq 'ConsoleWindowClass' -or $f[2] -eq 'PseudoConsoleWindow' }) +Say ("RESULT console-windows-now: " + $wins.Count) +foreach ($w in $wins) { $f = $w.Split('|', 4); Say ("RESULT window: " + (ProcName ([int]$f[1])) + " pid " + $f[1] + " [" + $f[2] + "] " + $f[3]) } +# a child the elevated script starts the way the sweep helper and the power cap do (cmd, inherited console), watched +$before = @{}; foreach ($w in [IgWin3]::Visible()) { $before[$w.Split('|', 4)[0]] = 1 } +$p = Start-Process -FilePath cmd.exe -ArgumentList '/c ping -n 3 127.0.0.1 >nul' -NoNewWindow -PassThru +$seen = @{} +for ($i = 0; $i -lt 30; $i++) { + foreach ($w in [IgWin3]::Visible()) { $f = $w.Split('|', 4); if (-not $before.ContainsKey($f[0]) -and -not $seen.ContainsKey($f[0])) { $seen[$f[0]] = (ProcName ([int]$f[1])) + " pid " + $f[1] + " [" + $f[2] + "] " + $f[3] } } + Start-Sleep -Milliseconds 100 +} +try { $p.WaitForExit(10000) | Out-Null } catch {} +Say ("RESULT probe cmd-inherit-elevated: " + $seen.Count + " new window(s)") +foreach ($s in $seen.Values) { Say ("RESULT window: " + $s + " (probe cmd-inherit-elevated)") } +exit 0 diff --git a/tools/windows/console-watch.ps1 b/tools/windows/console-watch.ps1 new file mode 100644 index 00000000..e7cf96ea --- /dev/null +++ b/tools/windows/console-watch.ps1 @@ -0,0 +1,181 @@ +# Console-window watcher for a Windows PC running the Igneum Miner app. A signed `run` job (app/igneum-app/src/jobrun.rs, +# shell powershell, not elevated): while a sampler thread enumerates the visible top-level windows (user32 EnumWindows, +# IsWindowVisible, GetWindowThreadProcessId, GetClassName, GetWindowText) and the console host processes (conhost, +# OpenConsole, WindowsTerminal, with their command lines and parents) every 30 ms, the main thread starts each +# candidate child the way a job script or the app does, and every window or host that appears during a probe is +# reported against it: +# RESULT terminal: ... the default-terminal delegation (HKCU\Console\%%Startup) and the host process counts +# RESULT self: ... the job's own console (hidden or not) and the conhost that serves it +# RESULT probe <n>: ... exit code, duration, how many windows and hosts appeared +# RESULT window: <process> [<class>] <title> (probe <n>) +# RESULT host: <name> pid <p> parent <process> cmd <command line> (probe <n>) +# Trust test (CLAUDE.md: a watcher is trusted only after a known-finished and a known-failed case): probe +# start-process-new-console MUST report a window (cmd in a new console); start-process-hidden is the same with +# -WindowStyle Hidden. 5 October 2026: written for PC 1 (ae432dc7, Windows 11 Pro 26200), where the project lead saw +# "Windows Command Processor" windows whenever a remote job ran. +# packaging/ota/publish-jobs.sh add --kind run --target ae432dc7 --script tools/windows/console-watch.ps1 \ +# --timeout-minutes 5 --title "PC 1: console window watcher" --deploy +$ErrorActionPreference = 'Continue' +$src = @' +using System; +using System.Collections.Generic; +using System.Runtime.InteropServices; +using System.Text; +public static class IgWin { + public delegate bool EnumProc(IntPtr h, IntPtr l); + [DllImport("user32.dll")] public static extern bool EnumWindows(EnumProc p, IntPtr l); + [DllImport("user32.dll")] public static extern bool IsWindowVisible(IntPtr h); + [DllImport("user32.dll")] public static extern uint GetWindowThreadProcessId(IntPtr h, out uint pid); + [DllImport("user32.dll", CharSet = CharSet.Unicode)] public static extern int GetWindowText(IntPtr h, StringBuilder s, int n); + [DllImport("user32.dll", CharSet = CharSet.Unicode)] public static extern int GetClassName(IntPtr h, StringBuilder s, int n); + [DllImport("kernel32.dll")] public static extern IntPtr GetConsoleWindow(); + public static List<string> Visible() { + var list = new List<string>(); + EnumWindows((h, l) => { + if (!IsWindowVisible(h)) return true; + uint pid; GetWindowThreadProcessId(h, out pid); + var t = new StringBuilder(512); GetWindowText(h, t, 512); + var c = new StringBuilder(256); GetClassName(h, c, 256); + list.Add(((long)h).ToString() + "|" + pid + "|" + c + "|" + t); + return true; + }, IntPtr.Zero); + return list; + } +} +'@ +try { Add-Type -TypeDefinition $src -ErrorAction Stop } catch { if (-not ([System.Management.Automation.PSTypeName]'IgWin').Type) { Write-Output ("RESULT error: Add-Type failed: " + $_); exit 2 } } + +function Say([string] $m) { Write-Output $m } +function ProcName([int] $procId) { try { (Get-Process -Id $procId -ErrorAction Stop).ProcessName } catch { 'gone' } } + +# ---- the default terminal and the hosts present before anything starts ------------------------------------------------ +$CONHOST_ID = '{B23D10C0-E52E-411E-9D5B-C09FDF709C7D}' +$DECIDE_ID = '{00000000-0000-0000-0000-000000000000}' +$k = Get-ItemProperty -Path 'HKCU:\Console\%%Startup' -ErrorAction SilentlyContinue +$dc = if ($k) { [string]$k.DelegationConsole } else { '(absent)' } +$dt = if ($k) { [string]$k.DelegationTerminal } else { '(absent)' } +$meaning = if ($dc -eq $CONHOST_ID) { 'Windows Console Host (conhost)' } elseif ($dc -eq $DECIDE_ID -or $dc -eq '(absent)') { 'Let Windows decide (Windows Terminal on Windows 11 22H2 and later when it is installed)' } else { 'a terminal package, Windows Terminal or its preview' } +$wtPkg = (Get-AppxPackage -Name 'Microsoft.WindowsTerminal*' -ErrorAction SilentlyContinue | ForEach-Object { $_.Name + ' ' + $_.Version }) -join ', ' +Say ("RESULT terminal: DelegationConsole=" + $dc + " DelegationTerminal=" + $dt + " -> " + $meaning + "; Windows Terminal package: " + $(if ($wtPkg) { $wtPkg } else { 'none' })) +$os = Get-CimInstance Win32_OperatingSystem +Say ("RESULT os: " + $os.Caption + " build " + $os.BuildNumber + " user " + $env:USERNAME + " session " + [System.Diagnostics.Process]::GetCurrentProcess().SessionId) +$counts = @{} +foreach ($n in 'conhost', 'OpenConsole', 'WindowsTerminal', 'cmd', 'powershell', 'wsl', 'wslhost') { $counts[$n] = @(Get-Process -Name $n -ErrorAction SilentlyContinue).Count } +Say ("RESULT hosts-before: conhost " + $counts['conhost'] + " OpenConsole " + $counts['OpenConsole'] + " WindowsTerminal " + $counts['WindowsTerminal'] + " cmd " + $counts['cmd'] + " powershell " + $counts['powershell'] + " wsl " + $counts['wsl'] + " wslhost " + $counts['wslhost']) + +# the job's own console and the chain above it +$me = [System.Diagnostics.Process]::GetCurrentProcess() +$meCim = Get-CimInstance Win32_Process -Filter "ProcessId=$($me.Id)" +$parent = if ($meCim) { ProcName $meCim.ParentProcessId } else { '?' } +$hwnd = [IgWin]::GetConsoleWindow() +$selfVis = if ($hwnd -ne [IntPtr]::Zero) { [IgWin]::IsWindowVisible($hwnd) } else { 'no window' } +$ownHost = Get-CimInstance Win32_Process -Filter "Name='conhost.exe' OR Name='OpenConsole.exe'" | Where-Object { $_.ParentProcessId -eq $me.Id -or $_.ParentProcessId -eq $meCim.ParentProcessId } +$ownLine = if ($ownHost) { ($ownHost | ForEach-Object { $_.Name + ' pid ' + $_.ProcessId + ' parent ' + (ProcName $_.ParentProcessId) + ' cmd ' + $_.CommandLine }) -join ' ; ' } else { 'none with this script or its parent as parent' } +Say ("RESULT self: powershell pid " + $me.Id + " parent " + $parent + " (pid " + $meCim.ParentProcessId + "); console hwnd " + $hwnd + " visible " + $selfVis + "; host " + $ownLine) +foreach ($w in [IgWin]::Visible()) { + $f = $w.Split('|', 4) + if ($f[2] -eq 'ConsoleWindowClass' -or $f[2] -eq 'CASCADIA_HOSTING_WINDOW_CLASS') { Say ("RESULT window-before: " + (ProcName ([int]$f[1])) + " [" + $f[2] + "] " + $f[3]) } +} + +# ---- the sampler thread: every new visible window and every new console host, with the time it was first seen ------- +$sync = [hashtable]::Synchronized(@{ win = [hashtable]::Synchronized(@{}); hosts = [hashtable]::Synchronized(@{}); stop = $false; ticks = 0; err = '' }) +$rs = [runspacefactory]::CreateRunspace() +$rs.Open() +$rs.SessionStateProxy.SetVariable('sync', $sync) +$rs.SessionStateProxy.SetVariable('src', $src) +$sampler = [powershell]::Create() +$sampler.Runspace = $rs +[void]$sampler.AddScript({ + try { + if (-not ([System.Management.Automation.PSTypeName]'IgWin').Type) { Add-Type -TypeDefinition $src } + $first = $true + while (-not $sync.stop) { + $now = Get-Date + foreach ($w in [IgWin]::Visible()) { + $f = $w.Split('|', 4) + if (-not $sync.win.ContainsKey($f[0])) { + $pn = try { (Get-Process -Id ([int]$f[1]) -ErrorAction Stop).ProcessName } catch { 'gone' } + $sync.win[$f[0]] = @{ t = $now; procId = [int]$f[1]; proc = $pn; cls = $f[2]; title = $f[3]; base = $first } + } + } + foreach ($p in @(Get-Process -Name conhost, OpenConsole, WindowsTerminal -ErrorAction SilentlyContinue)) { + if (-not $sync.hosts.ContainsKey($p.Id)) { + $ci = Get-CimInstance Win32_Process -Filter "ProcessId=$($p.Id)" -ErrorAction SilentlyContinue + $ppid = if ($ci) { $ci.ParentProcessId } else { 0 } + $ppn = try { (Get-Process -Id $ppid -ErrorAction Stop).ProcessName } catch { 'gone' } + $sync.hosts[$p.Id] = @{ t = $now; name = $p.ProcessName; cmd = $(if ($ci) { [string]$ci.CommandLine } else { '?' }); ppid = $ppid; pproc = $ppn; base = $first } + } + } + $first = $false + $sync.ticks++ + Start-Sleep -Milliseconds 30 + } + } catch { $sync.err = [string]$_ } +}) +$handle = $sampler.BeginInvoke() +$t = 0 +while ($sync.ticks -lt 2 -and $t -lt 100) { Start-Sleep -Milliseconds 50; $t++ } +if ($sync.ticks -lt 2) { Say ("RESULT error: the sampler did not start: " + $sync.err); exit 2 } +Say ("sampler running: " + $sync.win.Count + " visible windows and " + $sync.hosts.Count + " console hosts at the start") + +# ---- probes: each one as a job script or the app would start it ---------------------------------------------------- +$report = New-Object System.Collections.ArrayList +function Probe([string] $name, [scriptblock] $body) { + Start-Sleep -Milliseconds 400 + $t0 = Get-Date + $global:LASTEXITCODE = 0 + $err = '' + try { & $body 2>&1 | Out-Null } catch { $err = [string]$_ } + $rc = $LASTEXITCODE + Start-Sleep -Milliseconds 600 + $t1 = Get-Date + $ms = [int]($t1 - $t0).TotalMilliseconds - 600 + $wins = @($sync.win.GetEnumerator() | Where-Object { -not $_.Value.base -and $_.Value.t -ge $t0 -and $_.Value.t -le $t1 -and -not $_.Value.reported }) + $hosts = @($sync.hosts.GetEnumerator() | Where-Object { -not $_.Value.base -and $_.Value.t -ge $t0 -and $_.Value.t -le $t1 -and -not $_.Value.reported }) + Say ("RESULT probe " + $name + ": exit " + $rc + " in " + $ms + " ms, " + $wins.Count + " window(s), " + $hosts.Count + " host(s)" + $(if ($err) { "; error " + $err } else { '' })) + foreach ($w in $wins) { $w.Value.reported = $true; Say ("RESULT window: " + $w.Value.proc + " [" + $w.Value.cls + "] " + $w.Value.title + " (probe " + $name + ")") } + foreach ($h in $hosts) { $h.Value.reported = $true; Say ("RESULT host: " + $h.Value.name + " pid " + $h.Key + " parent " + $h.Value.pproc + " cmd " + $h.Value.cmd + " (probe " + $name + ")") } +} +$distro = 'Ubuntu-24.04' +$haveDistro = $false +try { $haveDistro = ((& wsl.exe -l -q 2>$null) -replace "`0", '' | Where-Object { $_.Trim() -eq $distro }).Count -gt 0 } catch {} +Say ("wsl distro " + $distro + ": " + $(if ($haveDistro) { 'present' } else { 'absent, the distro probes are skipped' })) + +# a. the plain children a job script starts (CreateProcess, the console inherited) +Probe 'powershell-inherit' { & powershell.exe -NoProfile -ExecutionPolicy Bypass -Command 'Start-Sleep -Milliseconds 1200' } +Probe 'cmd-c-inherit' { & cmd.exe /c 'ping -n 3 127.0.0.1 >nul' } +Probe 'query-session' { & query.exe session } +Probe 'curl-version' { & curl.exe --version } +Probe 'nvidia-smi-L' { & nvidia-smi.exe -L } +Probe 'powershell-windowstyle-hidden-inherit' { & powershell.exe -NoProfile -WindowStyle Hidden -Command 'Start-Sleep -Milliseconds 1200' } +Probe 'wsl-status' { & wsl.exe --status } +if ($haveDistro) { + Probe 'wsl-distro-sleep' { & wsl.exe -d Ubuntu-24.04 -u root -- sleep 1 } + Probe 'wsl-interop-cmd' { & wsl.exe -d Ubuntu-24.04 -u root -- cmd.exe /c 'ping -n 3 127.0.0.1' } + Probe 'wsl-interop-powershell' { & wsl.exe -d Ubuntu-24.04 -u root -- powershell.exe -NoProfile -Command 'Start-Sleep -Milliseconds 1200' } +} +# b. the trust test: a new console (ShellExecute) must show; the same hidden +Probe 'start-process-new-console' { Start-Process -FilePath cmd.exe -ArgumentList '/c ping -n 3 127.0.0.1' -Wait } +Probe 'start-process-hidden' { Start-Process -FilePath cmd.exe -ArgumentList '/c ping -n 3 127.0.0.1' -WindowStyle Hidden -Wait } +# c. the elevated job's shape without the elevation: powershell in a new hidden console through ShellExecute +Probe 'start-process-powershell-hidden' { Start-Process -FilePath powershell.exe -ArgumentList '-NoProfile -ExecutionPolicy Bypass -Command Start-Sleep -Milliseconds 1200' -WindowStyle Hidden -Wait } +# d. candidate fixes: a headless conhost of our own around the child; the children of a headless session +Probe 'conhost-headless-cmd' { & conhost.exe --headless cmd.exe /c 'ping -n 3 127.0.0.1 >nul' } +if ($haveDistro) { + Probe 'conhost-headless-wsl-interop' { & conhost.exe --headless wsl.exe -d Ubuntu-24.04 -u root -- cmd.exe /c 'ping -n 3 127.0.0.1' } +} +Probe 'conhost-headless-start-process-hidden' { & conhost.exe --headless powershell.exe -NoProfile -Command "Start-Process -FilePath cmd.exe -ArgumentList '/c ping -n 3 127.0.0.1' -WindowStyle Hidden -Wait" } + +# ---- the end: anything the probes did not claim --------------------------------------------------------------------- +Start-Sleep -Milliseconds 800 +$sync.stop = $true +try { [void]$sampler.EndInvoke($handle) } catch {} +$sampler.Dispose(); $rs.Close() +$stray = @($sync.win.GetEnumerator() | Where-Object { -not $_.Value.base -and -not $_.Value.reported }) +foreach ($w in $stray) { Say ("RESULT window: " + $w.Value.proc + " [" + $w.Value.cls + "] " + $w.Value.title + " (between probes)") } +$strayH = @($sync.hosts.GetEnumerator() | Where-Object { -not $_.Value.base -and -not $_.Value.reported }) +foreach ($h in $strayH) { Say ("RESULT host: " + $h.Value.name + " pid " + $h.Key + " parent " + $h.Value.pproc + " cmd " + $h.Value.cmd + " (between probes)") } +$counts = @{} +foreach ($n in 'conhost', 'OpenConsole', 'WindowsTerminal') { $counts[$n] = @(Get-Process -Name $n -ErrorAction SilentlyContinue).Count } +Say ("RESULT hosts-after: conhost " + $counts['conhost'] + " OpenConsole " + $counts['OpenConsole'] + " WindowsTerminal " + $counts['WindowsTerminal'] + "; sampler ticks " + $sync.ticks + $(if ($sync.err) { "; sampler error " + $sync.err } else { '' })) +exit 0