diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 9879a98b0..6c7af009e 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -25,6 +25,8 @@ jobs: - name: igneum-pow tests (release) working-directory: igneum-pow run: cargo test --release + - name: pack loader seed rule (packfile.h on a known-good and a known-mismatched pack) + run: bash proto-cuda/nvrtc/emu/packfile-test.sh - name: igneum-census build (release) working-directory: igneum-census run: cargo build --release @@ -67,8 +69,20 @@ jobs: run: bash tools/ci/no-conflict-markers.sh - name: copied sources are re-stamped before a build run: bash tools/ci/copied-sources-check.sh + - name: second-engine playbooks log to a file and end their tree (C35) + run: bash tools/ci/second-engine-check.sh + - name: the signer is never piped into head + run: bash tools/ci/signer-pipe-check.sh + - name: bash bodies in PowerShell job scripts pass bash -n, the lost-quote class (self-test first, then the tree) + run: bash tools/ci/bash-body-check.sh --self-test && bash tools/ci/bash-body-check.sh + - name: run jobs test their fetched kit before use, the wiped-jobs-folder class (self-test first, then the tree) + run: bash tools/ci/kit-path-check.sh --self-test && bash tools/ci/kit-path-check.sh + - name: every Windows spawn of the app runs with a hidden console (self-test first, then the tree) + run: node tools/ci/windows-spawn-check.mjs --self-test && node tools/ci/windows-spawn-check.mjs - name: pinned guest programs match their manifest and are built only by pin-guests.sh run: bash tools/ci/pinned-guests-check.sh + - name: root prover playbooks kill the GPU server and unlink its socket (the root-socket class, 5 October 2026) + run: bash tools/ci/prover-socket-check.sh - name: no secret file names and no 64-hex secrets in the tree (self-test first, then the tree) run: bash tools/ci/no-secrets-check.sh --self-test && bash tools/ci/no-secrets-check.sh - name: faucet unit tests (validation, the daily limits, the signed transaction; keccak, RLP and secp256k1 vectors) @@ -81,6 +95,6 @@ jobs: - name: ship tool self-test (version bump, the dl-both and public manifest helpers) run: node tools/ship-app.mjs --self-test - name: relay unit tests (parsers, secret compare, the wake endpoint) - run: node --test relay/test/parse.test.mjs relay/test/auth.test.mjs relay/test/wake.test.mjs + run: node --test relay/test/parse.test.mjs relay/test/auth.test.mjs relay/test/wake.test.mjs relay/test/ember.test.mjs - name: miner app notice strip and update card (ordering, keys, wording, timers, when the card shows) - run: node --test app/igneum-app/ui/notices.test.mjs app/igneum-app/ui/update-card.test.mjs + run: node --test app/igneum-app/ui/notices.test.mjs app/igneum-app/ui/update-card.test.mjs app/igneum-app/ui/view.test.mjs app/igneum-app/ui/tune-line.test.mjs diff --git a/app/igneum-app/Cargo.lock b/app/igneum-app/Cargo.lock index 601d35429..da972e82b 100644 --- a/app/igneum-app/Cargo.lock +++ b/app/igneum-app/Cargo.lock @@ -219,7 +219,7 @@ dependencies = [ [[package]] name = "igneum-app" -version = "0.3.9" +version = "0.3.12" dependencies = [ "ed25519-dalek", "getrandom", diff --git a/app/igneum-app/Cargo.toml b/app/igneum-app/Cargo.toml index e33046711..3d0e075a7 100644 --- a/app/igneum-app/Cargo.toml +++ b/app/igneum-app/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "igneum-app" -version = "0.3.9" +version = "0.3.12" edition = "2021" description = "Igneum Miner engine: supervises the node, the miner and the GPU workers, and serves the dashboard on 127.0.0.1" license = "MIT" diff --git a/app/igneum-app/resources/igneum-app.rc b/app/igneum-app/resources/igneum-app.rc index 3881fb70e..82f2543a4 100644 --- a/app/igneum-app/resources/igneum-app.rc +++ b/app/igneum-app/resources/igneum-app.rc @@ -6,8 +6,8 @@ 1 ICON "igneum.ico" 1 VERSIONINFO -FILEVERSION 0,3,9,0 -PRODUCTVERSION 0,3,9,0 +FILEVERSION 0,3,12,0 +PRODUCTVERSION 0,3,12,0 FILEFLAGSMASK 0x3fL FILEFLAGS 0x0L FILEOS VOS_NT_WINDOWS32 @@ -20,12 +20,12 @@ BEGIN BEGIN VALUE "CompanyName", "Igneum" VALUE "FileDescription", "Igneum Miner engine" - VALUE "FileVersion", "0.3.9" + VALUE "FileVersion", "0.3.12" VALUE "InternalName", "igneum-app" VALUE "LegalCopyright", "Igneum contributors" VALUE "OriginalFilename", "igneum-app.exe" VALUE "ProductName", "Igneum Miner" - VALUE "ProductVersion", "0.3.9" + VALUE "ProductVersion", "0.3.12" END END BLOCK "VarFileInfo" diff --git a/app/igneum-app/src/config.rs b/app/igneum-app/src/config.rs index 5bdcfaf1b..fe287df57 100644 --- a/app/igneum-app/src/config.rs +++ b/app/igneum-app/src/config.rs @@ -29,6 +29,16 @@ pub struct CardPref { pub sweep_watts: f64, #[serde(default)] pub sweep_mhs: f64, + /// Ember Tune (src/ember.rs): the clock cap the last tune chose (0 = unlocked), the driver and program class it + /// ran under (a change makes the card due again), and the plan that produced it (full | confirm | baseline) + #[serde(default)] + pub sweep_clock_mhz: u32, + #[serde(default)] + pub sweep_driver: String, + #[serde(default)] + pub sweep_class: String, + #[serde(default)] + pub sweep_source: String, } #[derive(Clone, Serialize, Deserialize)] @@ -70,9 +80,16 @@ pub struct Settings { #[serde(default)] pub prove: bool, /// The efficiency sweep (src/sweep.rs): once after install, then weekly, each NVIDIA card's cap is stepped from - /// 100% to 50% on the live program and held at the best MH per watt. Default on. A pinned card is skipped. - #[serde(default = "yes")] + /// 100% to 50% on the live program and held at the best MH per watt. Default off; implied by `power_control` + /// (on when that is switched on, never effective while it is off). A pinned card is skipped. + #[serde(default)] pub sweep: bool, + /// Power control (the project lead, 5 October 2026: "if we don't have to ask then don't ask"): the NVIDIA power cap and the + /// efficiency sweep need administrator rights (one UAC prompt on Windows). Default OFF on every machine; the app + /// never raises the prompt on its own. Switching it on asks once, at that moment; a refused, cancelled or + /// unanswered prompt switches it back off with a notice, no retries. + #[serde(default)] + pub power_control: bool, /// When this install first ran (unix s), for the "first hour after install" sweep. #[serde(default)] pub installed_at: u64, @@ -87,6 +104,10 @@ pub struct Settings { /// so it includes proof records it never verified (src/verifier.rs). Default off; a found verifier always wins. #[serde(default)] pub proof_verify_trust: bool, + /// Proving v1 step 1 (5 October 2026): the install-time default for `prove` has been applied once (src/provedefault.rs: + /// on when the machine can prove, never switching an explicit on back off). Older installs apply it at their next start. + #[serde(default)] + pub prove_default_applied: bool, } fn one() -> u32 { @@ -98,16 +119,26 @@ fn yes() -> bool { impl Default for Settings { fn default() -> Settings { - Settings { setup_done: false, address: String::new(), address_source: String::new(), key_saved: false, identities: 1, cards: HashMap::new(), display_name: String::new(), vote: true, paused: false, accepted_total: 0, auto_update: true, remote_jobs: true, prove: false, sweep: true, installed_at: 0, dev_fee: true, fee_total: 0, proof_verify_trust: false } + Settings { setup_done: false, address: String::new(), address_source: String::new(), key_saved: false, identities: 1, cards: HashMap::new(), display_name: String::new(), vote: true, paused: false, accepted_total: 0, auto_update: true, remote_jobs: true, prove: false, sweep: true, power_control: false, installed_at: 0, dev_fee: true, fee_total: 0, proof_verify_trust: false, prove_default_applied: false } } } impl Settings { pub fn load(path: &Path) -> Settings { let mut s: Settings = std::fs::read_to_string(path).ok().and_then(|t| serde_json::from_str(&t).ok()).unwrap_or_default(); + let mut dirty = false; if s.installed_at == 0 { // an install from before the sweep existed counts as installed now: it gets its first-hour sweep s.installed_at = crate::platform::unix_now(); + dirty = true; + } + if s.sweep && !s.power_control { + // the sweep is implied by power control (5 October 2026): an install from before that setting carried + // sweep = true by default; it no longer prompts on its own + s.sweep = false; + dirty = true; + } + if dirty { s.save(path); } s diff --git a/app/igneum-app/src/detect.rs b/app/igneum-app/src/detect.rs index 1af73c031..9e69fa419 100644 --- a/app/igneum-app/src/detect.rs +++ b/app/igneum-app/src/detect.rs @@ -1,6 +1,8 @@ //! GPU detection with the real names. macOS: the Metal worker's ready line (the device Metal reports) plus the core //! count from system_profiler. Windows: nvidia-smi for NVIDIA cards, the OpenCL worker's --list for the rest -//! (AMD, Intel), and the WMI name list as a last resort when neither tool runs. +//! (AMD, Intel), the Windows adapter list (Win32_VideoController: status, problem code, memory) for the cards no +//! worker can drive and for the integrated-or-discrete call, and that list's names as a last resort when neither +//! tool runs. The engine runs this at start and again every minute (src/hotplug.rs compares the two lists). use crate::state::CardState; use std::io::Write; @@ -15,9 +17,43 @@ pub struct Bins { pub metal: Option, pub cuda: Option, pub opencl: Option, + /// igneum-gpu-telemetry: AMD power, heat, fans and clocks (proto-opencl/gpu-telemetry.c), 5 October 2026 + pub telemetry: Option, pub dir: std::path::PathBuf, } +/// One enumeration: the cards, the notes for the setup screen, and which tools answered. A tool that did not answer +/// (nvidia-smi timed out, the OpenCL worker crashed) says nothing about its cards: the engine keeps them rather +/// than calling them removed (src/hotplug.rs). +#[derive(Clone, Default)] +pub struct Detection { + pub cards: Vec, + pub notes: Vec, + /// duplicate OpenCL platform entries left out (one line each, for the log) + pub dropped: Vec, + pub nvidia_listed: bool, + pub opencl_listed: bool, + pub adapters_listed: bool, + pub metal_listed: bool, +} + +impl Detection { + /// Whether this enumeration can say that `c` is gone: the tool that lists its vendor answered. + pub fn listed(&self, c: &CardState) -> bool { + if !c.problem.is_empty() { + return self.adapters_listed; + } + match c.vendor.as_str() { + "apple" => self.metal_listed, + "nvidia" => self.nvidia_listed, + _ => self.opencl_listed || (c.device.is_empty() && self.adapters_listed), + } + } +} + +/// The hint on a card the OS reports as faulty (Windows Code 43, 12, 31 and friends). +pub const PROBLEM_HINT: &str = "reboot with the card attached; if it persists, reinstall the driver with the card attached"; + /// Runs a command with a time limit; returns stdout (and stderr appended) or None. pub fn run_timeout(cmd: &mut Command, stdin_text: Option<&str>, limit: Duration) -> Option { cmd.stdout(Stdio::piped()).stderr(Stdio::piped()); @@ -56,7 +92,8 @@ pub fn run_timeout(cmd: &mut Command, stdin_text: Option<&str>, limit: Duration) fn card(index: usize, name: &str, vendor: &str, worker: &str, detail: &str, device: &str) -> CardState { CardState { index, - key: format!("{vendor}:{device}:{name}"), + key: format!("{vendor}:{name}"), + code: name.to_string(), name: name.to_string(), vendor: vendor.into(), worker: worker.into(), @@ -64,18 +101,146 @@ fn card(index: usize, name: &str, vendor: &str, worker: &str, detail: &str, devi device: device.into(), enabled: true, state: "off".into(), + amd_ordinal: -1, ..Default::default() } } -/// Integrated GPUs by name: AMD APUs ("Radeon Graphics", "Vega 8"), Intel iGPUs (Iris, UHD, HD Graphics, Arc A3xx is discrete). -#[allow(dead_code)] +/// Integrated GPUs by name: AMD APUs ("Radeon Graphics", "Vega 8"), Intel iGPUs (Iris, UHD, HD Graphics, Arc A3xx is +/// discrete), and the gfx codes AMD's OpenCL runtime reports instead of a marketing name (the worker's --list prints +/// CL_DEVICE_NAME: PC 1's Ryzen iGPU is "gfx1036", 5 October 2026). pub fn looks_integrated(name: &str) -> bool { let n = name.to_ascii_lowercase(); let integrated = ["radeon(tm) graphics", "radeon graphics", "vega 8", "vega 7", "vega 6", "vega 3", "vega 11", "iris", "uhd graphics", "hd graphics", "intel(r) graphics", "intel graphics", "apu", "780m", "760m", "680m", "610m", "890m", "880m"]; - integrated.iter().any(|k| n.contains(k)) && !n.contains("arc ") + if integrated.iter().any(|k| n.contains(k)) && !n.contains("arc ") { + return true; + } + // AMD APU graphics by gfx code (approximate list from AMD's ROCm and Mesa target tables): Raven/Picasso gfx902 and + // gfx909, Renoir/Cezanne/Lucienne gfx90c, Van Gogh gfx1033, Rembrandt gfx1035, Raphael/Granite Ridge gfx1036, + // Mendocino gfx1037, Phoenix gfx1103, Strix gfx1150 to gfx1152. Discrete codes (gfx1030 and so on) are not here. + let apu = ["gfx902", "gfx909", "gfx90c", "gfx1033", "gfx1035", "gfx1036", "gfx1037", "gfx1103", "gfx1150", "gfx1151", "gfx1152"]; + let code = n.trim(); + apu.iter().any(|k| code == *k || code.starts_with(&format!("{k}:")) || code.starts_with(&format!("{k} "))) } +/// One row of Windows' adapter list (Win32_VideoController), the part this app reads. +#[derive(Clone, Debug, Default, PartialEq)] +pub struct Adapter { + pub name: String, + /// "OK", "Error", "Degraded", ... (the Status property) + pub status: String, + /// the PnP problem code (ConfigManagerErrorCode): 0 = fine, 43 = the driver stopped it, 12 = no resources, 31 = not loaded + pub code: u32, + /// AdapterRAM in MB; 0 = unknown (a faulty card reports 0, and the property caps at 4 GB on 32-bit values) + pub ram_mb: u64, + pub processor: String, + pub pnp_id: String, + /// "01:00.0" from DEVPKEY_Device_BusNumber and DEVPKEY_Device_Address; empty when PowerShell could not read them + pub bus: String, +} + +impl Adapter { + /// The PCI device id from the PnP id ("PCI\\VEN_1002&DEV_7550&..." gives 0x7550); 0 when there is none. + pub fn device_id(&self) -> u16 { + let up = self.pnp_id.to_ascii_uppercase(); + up.find("DEV_").and_then(|i| u16::from_str_radix(up.get(i + 4..i + 8)?, 16).ok()).unwrap_or(0) + } + /// "Code 43" for a problem code, "status Error" for a bad status without one, None when the device is fine. + pub fn problem(&self) -> Option { + if self.code != 0 { + return Some(format!("Code {}", self.code)); + } + let st = self.status.trim(); + if !st.is_empty() && !st.eq_ignore_ascii_case("ok") { + return Some(format!("status {st}")); + } + None + } +} + +/// Integrated or discrete, from the name (APU and iGPU names, AMD gfx codes) and, when Windows' adapter row is +/// known, its processor string or a dedicated memory under 1 GB (a shared-memory iGPU; 0 = unknown, says nothing). +pub fn classify_kind(name: &str, adapter: Option<&Adapter>) -> &'static str { + if looks_integrated(name) { + return "integrated"; + } + if let Some(a) = adapter { + // the processor string names Intel iGPUs ("Intel(R) Iris(R) Xe Graphics Family"); AMD's reads "AMD Radeon + // Graphics Processor (0x7550)" for discrete cards too, so only the Intel markers count here + let proc_ = a.processor.to_ascii_lowercase(); + if looks_integrated(&a.name) || (["iris", "uhd graphics", "hd graphics"].iter().any(|k| proc_.contains(k)) && !proc_.contains("arc")) { + return "integrated"; + } + if a.ram_mb > 0 && a.ram_mb < 1024 { + return "integrated"; + } + } + "discrete" +} + +pub fn vendor_of(name: &str) -> &'static str { + let n = name.to_ascii_lowercase(); + if n.contains("nvidia") || n.contains("geforce") { + "nvidia" + } else if n.contains("amd") || n.contains("radeon") || n.starts_with("gfx") { + "amd" + } else if n.contains("apple") { + "apple" + } else { + "other" + } +} + +/// Parses `Get-CimInstance Win32_VideoController | Select-Object ... | ConvertTo-Json` (one object or an array). +pub fn parse_adapters(json: &str) -> Vec { + let Ok(v) = serde_json::from_str::(json.trim()) else { return Vec::new() }; + let rows: Vec = match v { + serde_json::Value::Array(a) => a, + o @ serde_json::Value::Object(_) => vec![o], + _ => Vec::new(), + }; + let s = |r: &serde_json::Value, k: &str| r.get(k).and_then(|x| x.as_str()).unwrap_or("").trim().to_string(); + let n = |r: &serde_json::Value, k: &str| r.get(k).and_then(|x| x.as_u64().or_else(|| x.as_str().and_then(|t| t.trim().parse::().ok()))).unwrap_or(0); + rows.iter() + .map(|r| { + // DEVPKEY_Device_Address on PCI is (device << 16) | function + let bus = match (r.get("BusNumber").and_then(|x| x.as_u64()), r.get("Address").and_then(|x| x.as_u64())) { + (Some(b), Some(a)) => format!("{:02x}:{:02x}.{:x}", b & 0xff, (a >> 16) & 0xff, a & 0xffff), + _ => String::new(), + }; + Adapter { name: s(r, "Name"), status: s(r, "Status"), code: n(r, "ConfigManagerErrorCode") as u32, ram_mb: n(r, "AdapterRAM") / (1024 * 1024), processor: s(r, "VideoProcessor"), pnp_id: s(r, "PNPDeviceID"), bus } + }) + .filter(|a| !a.name.is_empty()) + .collect() +} + +/// Windows' adapter list through PowerShell (about a second); None when PowerShell did not answer. +#[cfg(windows)] +pub fn adapters() -> Option> { + // one object per adapter, with the PCI bus number and address from the PnP properties (they name the card + // the OpenCL worker's "pci" field names); @() keeps a single adapter an array + let script = "$v = Get-CimInstance Win32_VideoController | ForEach-Object { $id = $_.PNPDeviceID; $bus = $null; $addr = $null; try { foreach ($x in (Get-PnpDeviceProperty -InstanceId $id -KeyName 'DEVPKEY_Device_BusNumber','DEVPKEY_Device_Address' -ErrorAction Stop)) { if ($x.KeyName -eq 'DEVPKEY_Device_BusNumber') { $bus = $x.Data } elseif ($x.KeyName -eq 'DEVPKEY_Device_Address') { $addr = $x.Data } } } catch {}; [pscustomobject]@{ Name = $_.Name; Status = $_.Status; ConfigManagerErrorCode = $_.ConfigManagerErrorCode; AdapterRAM = $_.AdapterRAM; VideoProcessor = $_.VideoProcessor; PNPDeviceID = $id; BusNumber = $bus; Address = $addr } }; ConvertTo-Json -InputObject @($v) -Compress"; + let out = run_timeout(Command::new(crate::platform::tool("powershell")).args(["-NoProfile", "-Command", script]), None, Duration::from_secs(15))?; + let start = out.find(|c| c == '[' || c == '{')?; + Some(parse_adapters(&out[start..])) +} + +#[cfg(not(windows))] +#[allow(dead_code)] +pub fn adapters() -> Option> { + None +} + +/// Windows' row for a detected card, by name (nvidia-smi and Windows agree on NVIDIA names; AMD's OpenCL runtime +/// reports gfx codes, which match nothing here and fall back to the name rules). +pub fn adapter_for<'a>(name: &str, adapters: &'a [Adapter]) -> Option<&'a Adapter> { + let n = name.trim().to_ascii_lowercase(); + adapters.iter().find(|a| a.name.trim().to_ascii_lowercase() == n) +} + +/// The row's words for an integrated GPU that is off by default (the switch turns it on; the choice is kept). +pub const INTEGRATED_REASON: &str = "integrated GPU, off by default (2 to 3 MH/s for 30 W)"; + /// The defaults the launchers use: discrete cards on (8 identities on a big card, 2 on a small one), integrated off /// (1 identity), Apple silicon on with 1. pub fn apply_defaults(c: &mut CardState) { @@ -87,7 +252,7 @@ pub fn apply_defaults(c: &mut CardState) { "integrated" => { c.enabled = false; c.identities = 1; - c.reason = "integrated: about 3 MH/s and it shares your system memory. Switch it on if you want it.".into(); + c.reason = INTEGRATED_REASON.into(); } _ => { c.enabled = true; @@ -148,19 +313,20 @@ pub fn nvidia_power_limits() -> std::collections::HashMap) -> Vec { - let mut cards = Vec::new(); +pub fn detect(bins: &Bins) -> Detection { + let mut d = Detection::default(); let Some(metal) = bins.metal.as_ref() else { - notes.push("the Metal worker (igneum-bench) is missing from the app".into()); - return cards; + d.notes.push("the Metal worker (igneum-bench) is missing from the app".into()); + return d; }; // the worker's own ready line: "ready metal Apple_M5_Max dataset-log2 28 batch 4194304 prepare 1" let out = run_timeout(Command::new(metal).arg("--serve"), Some("quit\n"), Duration::from_secs(20)).unwrap_or_default(); let ready = out.lines().find(|l| l.starts_with("ready ")); let Some(ready) = ready else { - notes.push(format!("the Metal worker did not report ready: {}", out.lines().last().unwrap_or("no output"))); - return cards; + d.notes.push(format!("the Metal worker did not report ready: {}", out.lines().last().unwrap_or("no output"))); + return d; }; + d.metal_listed = true; let fields: Vec<&str> = ready.split_whitespace().collect(); let name = fields.get(2).map(|s| s.replace('_', " ")).unwrap_or_else(|| "Apple GPU".into()); let prepare = fields.windows(2).any(|w| w[0] == "prepare" && w[1] == "1"); @@ -175,58 +341,280 @@ pub fn detect(bins: &Bins, notes: &mut Vec) -> Vec { } } } - if let Some(mem) = run_timeout(Command::new(crate::platform::tool("sysctl")).args(["-n", "hw.memsize"]), None, Duration::from_secs(3)) { - if let Ok(b) = mem.trim().parse::() { - let gb = b / (1024 * 1024 * 1024); - detail = if detail.is_empty() { format!("{gb} GB unified memory") } else { format!("{detail}, {gb} GB unified memory") }; - } + let mem = run_timeout(Command::new(crate::platform::tool("sysctl")).args(["-n", "hw.memsize"]), None, Duration::from_secs(3)).and_then(|m| m.trim().parse::().ok()); + if let Some(b) = mem { + let gb = b / (1024 * 1024 * 1024); + detail = if detail.is_empty() { format!("{gb} GB unified memory") } else { format!("{detail}, {gb} GB unified memory") }; } if !prepare { - notes.push("this Metal worker has no prepare support; the miner restarts at the hour boundary".into()); + d.notes.push("this Metal worker has no prepare support; the miner restarts at the hour boundary".into()); } let mut c = card(0, &name, "apple", "Metal", &detail, ""); c.kind = "apple".into(); c.path = "prebuilt".into(); - if let Some(mem) = run_timeout(Command::new(crate::platform::tool("sysctl")).args(["-n", "hw.memsize"]), None, Duration::from_secs(3)) { - c.vram_mb = mem.trim().parse::().map(|b| b / (1024 * 1024)).unwrap_or(0); - } + c.vram_mb = mem.map(|b| b / (1024 * 1024)).unwrap_or(0); apply_defaults(&mut c); mark_sweep_support(&mut c); - cards.push(c); - cards + d.cards.push(c); + assign_keys(&mut d.cards); + d } -#[cfg(not(target_os = "macos"))] -pub fn detect(bins: &Bins, notes: &mut Vec) -> Vec { - let mut cards: Vec = Vec::new(); - // NVIDIA: nvidia-smi ships with the driver - let smi = run_timeout(Command::new(crate::platform::tool("nvidia-smi")).args(["--query-gpu=index,name,memory.total", "--format=csv,noheader"]), None, Duration::from_secs(10)); - match smi { +/// AMD gfx codes the OpenCL runtime reports as the device name, with the card names Windows uses, the words for +/// the row when no adapter matches, and the PCI device ids (approximate, from AMD's public ROCm and Linux driver +/// tables; add a line when a card is seen). gfx1036 is the Ryzen desktop iGPU (PC 1: DEV_13C0, 5 October 2026). +const GFX: &[(&str, &str, &[&str], &[u16])] = &[ + ("gfx1201", "Radeon RX 9070 XT / 9070", &["Radeon RX 9070 XT", "Radeon RX 9070"], &[0x7550]), + ("gfx1200", "Radeon RX 9060 XT", &["Radeon RX 9060 XT", "Radeon RX 9060"], &[0x7590]), + ("gfx1100", "Radeon RX 7900 XTX / XT", &["Radeon RX 7900 XTX", "Radeon RX 7900 XT", "Radeon RX 7900 GRE"], &[0x744C]), + ("gfx1101", "Radeon RX 7800 XT / 7700 XT", &["Radeon RX 7800 XT", "Radeon RX 7700 XT"], &[0x747E]), + ("gfx1102", "Radeon RX 7600", &["Radeon RX 7600 XT", "Radeon RX 7600"], &[0x7480]), + ("gfx1030", "Radeon RX 6800 / 6900", &["Radeon RX 6900 XT", "Radeon RX 6950 XT", "Radeon RX 6800 XT", "Radeon RX 6800"], &[0x73BF]), + ("gfx1031", "Radeon RX 6700 XT", &["Radeon RX 6750 XT", "Radeon RX 6700 XT", "Radeon RX 6700"], &[0x73DF]), + ("gfx1032", "Radeon RX 6600", &["Radeon RX 6650 XT", "Radeon RX 6600 XT", "Radeon RX 6600"], &[0x73FF]), + ("gfx1036", "Ryzen integrated Radeon Graphics", &["Radeon(TM) Graphics", "Radeon Graphics"], &[0x164E, 0x13C0]), + ("gfx1035", "Radeon 680M (integrated)", &["Radeon 680M", "Radeon 660M"], &[0x1681]), + ("gfx1103", "Radeon 780M (integrated)", &["Radeon 780M", "Radeon 760M"], &[0x15BF, 0x15C8]), + ("gfx1150", "Radeon 890M (integrated)", &["Radeon 890M", "Radeon 880M"], &[0x150E]), + ("gfx90c", "Radeon Graphics (Renoir / Cezanne, integrated)", &["Radeon(TM) Graphics", "Radeon Graphics"], &[0x1636, 0x1638]), +]; + +fn gfx_entry(code: &str) -> Option<&'static (&'static str, &'static str, &'static [&'static str], &'static [u16])> { + let c = code.trim().to_ascii_lowercase(); + let c = c.split(|ch: char| ch == ':' || ch == ' ').next().unwrap_or(""); + GFX.iter().find(|e| e.0 == c) +} + +/// The name on the row for a device the tool knows by `code`: Windows' adapter name when one matches (by PCI bus, +/// else by the gfx code's device ids, else by the card names in the table), else the table's words, else the code. +/// `used` holds the adapters already given to another card, so two gfx1036 entries never share one. +pub fn resolve_name(code: &str, vendor: &str, bus: &str, adapters: &[Adapter], used: &mut Vec) -> (String, Option) { + let free = |i: &usize| !used.contains(i); + let fine = |a: &Adapter| a.problem().is_none(); + let vendor_ok = |a: &Adapter| vendor == "other" || vendor_of(&a.name) == vendor; + if !bus.is_empty() { + if let Some(i) = (0..adapters.len()).filter(free).find(|&i| adapters[i].bus == bus && vendor_ok(&adapters[i])) { + used.push(i); + return (adapters[i].name.clone(), Some(i)); + } + } + // the name is already a marketing name (nvidia-smi, Windows): the adapter with the same name + let same: Vec = (0..adapters.len()).filter(free).filter(|&i| adapters[i].name.trim().eq_ignore_ascii_case(code.trim())).collect(); + if same.len() == 1 { + used.push(same[0]); + return (adapters[same[0]].name.clone(), Some(same[0])); + } + let Some(entry) = gfx_entry(code) else { return (code.to_string(), None) }; + let by_id: Vec = (0..adapters.len()).filter(free).filter(|&i| fine(&adapters[i]) && entry.3.contains(&adapters[i].device_id())).collect(); + if by_id.len() == 1 { + used.push(by_id[0]); + return (adapters[by_id[0]].name.clone(), Some(by_id[0])); + } + let by_name: Vec = (0..adapters.len()).filter(free).filter(|&i| fine(&adapters[i]) && vendor_ok(&adapters[i]) && { let n = adapters[i].name.to_ascii_lowercase(); entry.2.iter().any(|m| n.contains(&m.to_ascii_lowercase())) }).collect(); + if by_name.len() == 1 { + used.push(by_name[0]); + return (adapters[by_name[0]].name.clone(), Some(by_name[0])); + } + (entry.1.to_string(), None) +} + +/// One device line pair of the OpenCL worker's --list. +#[derive(Clone, Debug, Default, PartialEq)] +pub struct ClDevice { + pub index: String, + pub name: String, + pub platform: String, + pub platform_version: String, + pub is_gpu: bool, + pub vendor: String, + pub driver: String, + pub units: String, + /// "01:00.0" when the worker printed `pci` (workers from 5 October 2026 on), else empty + pub bus: String, + pub mem_mb: u64, +} + +impl ClDevice { + pub fn vendor_word(&self) -> &'static str { + if self.vendor.contains("NVIDIA") || self.name.contains("NVIDIA") { + "nvidia" + } else if self.vendor.contains("Advanced Micro") || self.vendor.contains("AMD") || self.name.contains("Radeon") || self.name.contains("AMD") || self.name.to_ascii_lowercase().starts_with("gfx") { + "amd" + } else { + "other" + } + } + fn platform_key(&self) -> String { + format!("{} ({}) driver {}", self.platform, self.platform_version, self.driver) + } +} + +/// Parses `igneum-worker-opencl --list`: `[idx] name | platform (version)` then `GPU, vendor V, driver D, OpenCL C +/// x.y, N compute units, M MHz[, pci bb:dd.f]` then `global N MiB, ...`. The bool says the worker printed its header. +pub fn parse_opencl_list(text: &str) -> (Vec, bool) { + let lines: Vec<&str> = text.lines().collect(); + let listed = lines.iter().any(|l| l.starts_with("OpenCL devices")); + let mut out = Vec::new(); + for (i, line) in lines.iter().enumerate() { + let t = line.trim_start_matches(|c| c == ' ' || c == '*').trim(); + if !t.starts_with('[') { + continue; + } + let Some(close) = t.find(']') else { continue }; + let rest = &t[close + 1..]; + let mut halves = rest.splitn(2, " |"); + let name = halves.next().unwrap_or("").trim().to_string(); + let plat = halves.next().unwrap_or("").trim(); + // the first " (" opens the version: AMD's version string carries brackets of its own, "OpenCL 2.1 AMD-APP (3617.0)" + let (platform, platform_version) = match plat.find(" (") { + Some(p) if plat.ends_with(')') => (plat[..p].to_string(), plat[p + 2..plat.len() - 1].to_string()), + _ => (plat.to_string(), String::new()), + }; + let info = lines.get(i + 1).map(|l| l.trim()).unwrap_or(""); + let parts: Vec<&str> = info.split(", ").collect(); + let mem_mb = lines.get(i + 2).map(|l| l.trim()).and_then(|l| l.strip_prefix("global ")).and_then(|l| l.split_whitespace().next()).and_then(|n| n.parse::().ok()).unwrap_or(0); + out.push(ClDevice { + index: t[1..close].to_string(), + name, + platform, + platform_version, + is_gpu: info.starts_with("GPU"), + vendor: parts.iter().find_map(|p| p.strip_prefix("vendor ")).unwrap_or("").trim().to_string(), + driver: parts.iter().find_map(|p| p.strip_prefix("driver ")).unwrap_or("").trim().to_string(), + units: parts.iter().find(|p| p.contains("compute units")).unwrap_or(&"").to_string(), + bus: parts.iter().find_map(|p| p.strip_prefix("pci ")).unwrap_or("").trim().to_string(), + mem_mb, + }); + } + (out, listed) +} + +fn version_tuple(s: &str) -> Vec { + s.split(|c: char| !c.is_ascii_digit()).filter(|p| !p.is_empty()).map(|p| p.parse::().unwrap_or(0)).collect() +} + +/// One entry per physical card across OpenCL platforms. Two AMD ICDs after a driver upgrade each list every AMD +/// card (PC 1, 5 October 2026: gfx1036 and gfx1201 twice, two workers on one 9070 XT). Per vendor, the fuller +/// platform wins (most GPUs, then the newest driver, then the first listed); a device on another platform is kept +/// only when the winner has no device at the same PCI address (or, without addresses, the same code and ordinal). +/// Returns the kept devices and one note per dropped duplicate. +pub fn dedupe_platforms(devs: Vec) -> (Vec, Vec) { + let gpus: Vec = devs.into_iter().filter(|d| d.is_gpu).collect(); + let mut kept: Vec = Vec::new(); + let mut dropped = Vec::new(); + let mut vendors: Vec<&'static str> = Vec::new(); + for d in &gpus { + let v = d.vendor_word(); + if !vendors.contains(&v) { + vendors.push(v); + } + } + for v in vendors { + let mine: Vec<&ClDevice> = gpus.iter().filter(|d| d.vendor_word() == v).collect(); + let mut plats: Vec = Vec::new(); + for d in &mine { + let k = d.platform_key(); + if !plats.contains(&k) { + plats.push(k); + } + } + let score = |k: &String| { + let count = mine.iter().filter(|d| &d.platform_key() == k).count(); + let driver = mine.iter().find(|d| &d.platform_key() == k).map(|d| version_tuple(&d.driver)).unwrap_or_default(); + (count, driver) + }; + let winner = plats.iter().max_by(|a, b| score(a).cmp(&score(b))).cloned().unwrap_or_default(); + let identity = |d: &ClDevice, ordinal: usize| if d.bus.is_empty() { format!("{}#{ordinal}", d.name.to_ascii_lowercase()) } else { d.bus.clone() }; + let mut have: Vec = Vec::new(); + let mut seen_codes: std::collections::HashMap = std::collections::HashMap::new(); + let mut ordinal = |d: &ClDevice| { + let n = seen_codes.entry(format!("{}|{}", d.platform_key(), d.name.to_ascii_lowercase())).or_insert(0); + *n += 1; + *n + }; + for d in mine.iter().filter(|d| d.platform_key() == winner) { + let o = ordinal(d); + have.push(identity(d, o)); + kept.push((*d).clone()); + } + for d in mine.iter().filter(|d| d.platform_key() != winner) { + let o = ordinal(d); + let id = identity(d, o); + if have.contains(&id) { + dropped.push(format!("[{}] {} on {} ({}) is the same card as the one on {}: no worker", d.index, d.name, d.platform, d.platform_version, winner)); + } else { + have.push(id); + kept.push((*d).clone()); + } + } + } + kept.sort_by_key(|d| d.index.parse::().unwrap_or(u64::MAX)); + (kept, dropped) +} + +/// Keys without an index (it moves when a card arrives): vendor:code, "#2" and up for identical cards in list order. +pub fn assign_keys(cards: &mut [CardState]) { + let mut seen: std::collections::HashMap = std::collections::HashMap::new(); + for c in cards.iter_mut() { + if c.code.is_empty() { + c.code = c.name.clone(); + } + let base = format!("{}:{}", c.vendor, c.code); + let n = seen.entry(base.clone()).or_insert(0); + *n += 1; + c.key = if *n == 1 { base } else { format!("{base}#{n}") }; + } +} + +/// What one Windows enumeration gathered; `assemble` turns it into the list (pure, so the PC 1 cases are tests). +#[derive(Default)] +pub struct Inputs { + /// `nvidia-smi --query-gpu=index,name,memory.total,pci.bus_id --format=csv,noheader`; None = nvidia-smi did not run + pub nvidia: Option, + pub nvidia_limits: std::collections::HashMap, + pub cuda_worker: bool, + /// the OpenCL worker's --list; None = no worker installed or it did not answer + pub opencl: Option, + pub opencl_installed: bool, + /// Windows' adapter list; None = PowerShell did not answer + pub adapters: Option>, +} + +pub fn assemble(inp: Inputs) -> Detection { + let mut d = Detection::default(); + d.adapters_listed = inp.adapters.is_some(); + let adapters = inp.adapters.unwrap_or_default(); + let mut used: Vec = Vec::new(); + match inp.nvidia { Some(out) => { + d.nvidia_listed = true; for line in out.lines() { let parts: Vec<&str> = line.split(',').map(|s| s.trim()).collect(); if parts.len() >= 2 && parts[0].chars().all(|c| c.is_ascii_digit()) && !parts[0].is_empty() { let mem_mb: u64 = parts.get(2).and_then(|m| m.split_whitespace().next()).and_then(|n| n.parse::().ok()).map(|v| v as u64).unwrap_or(0); let detail = if mem_mb > 0 { format!("{} GB", (mem_mb + 512) / 1024) } else { String::new() }; - let worker_ok = bins.cuda.is_some(); - let mut c = card(cards.len(), parts[1], "nvidia", "CUDA", &detail, parts[0]); - c.kind = if looks_integrated(parts[1]) { "integrated".into() } else { "discrete".into() }; + // nvidia-smi prints 00000000:01:00.0; the worker and Windows say 01:00.0 + let bus = parts.get(3).map(|b| b.trim().to_ascii_lowercase()).map(|b| b.rsplit_once(':').map(|(d, r)| format!("{}:{r}", d.rsplit(':').next().unwrap_or(d))).unwrap_or(b)).unwrap_or_default(); + let (name, adapter) = resolve_name(parts[1], "nvidia", &bus, &adapters, &mut used); + let mut c = card(d.cards.len(), &name, "nvidia", "CUDA", &detail, parts[0]); + c.code = parts[1].to_string(); + c.bus = bus; + c.kind = classify_kind(parts[1], adapter.map(|i| &adapters[i])).into(); c.vram_mb = mem_mb; - c.path = if worker_ok { "prebuilt".into() } else { "build".into() }; - if !worker_ok { + c.path = if inp.cuda_worker { "prebuilt".into() } else { "build".into() }; + if !inp.cuda_worker { c.message = "no prebuilt CUDA worker in the package; built from source on first run (needs the CUDA Toolkit and Visual Studio)".into(); } apply_defaults(&mut c); - cards.push(c); + d.cards.push(c); } } - if cards.is_empty() { - notes.push("nvidia-smi ran but listed no card".into()); + if d.cards.is_empty() { + d.notes.push("nvidia-smi ran but listed no card".into()); } - let limits = nvidia_power_limits(); - for c in cards.iter_mut() { - if let Some((d, cur, lo, hi)) = limits.get(&c.device) { - c.power_default_w = *d; + for c in d.cards.iter_mut() { + if let Some((dflt, cur, lo, hi)) = inp.nvidia_limits.get(&c.device) { + c.power_default_w = *dflt; c.power_limit_w = *cur; c.power_before_w = *cur; c.power_min_w = *lo; @@ -235,55 +623,84 @@ pub fn detect(bins: &Bins, notes: &mut Vec) -> Vec { mark_sweep_support(c); } } - None => notes.push("nvidia-smi is not on this PC (no NVIDIA driver): no NVIDIA card".into()), + None => d.notes.push("nvidia-smi is not on this PC (no NVIDIA driver): no NVIDIA card".into()), } - // OpenCL: the worker's own device list (AMD, Intel; NVIDIA shows there too and is skipped) - if let Some(cl) = bins.opencl.as_ref() { - if let Some(out) = run_timeout(Command::new(cl).arg("--list"), None, Duration::from_secs(15)) { - let lines: Vec<&str> = out.lines().collect(); - for (i, line) in lines.iter().enumerate() { - let t = line.trim_start_matches(|c| c == ' ' || c == '*').trim(); - if !t.starts_with('[') { - continue; - } - let Some(close) = t.find(']') else { continue }; - let idx = &t[1..close]; - let rest = &t[close + 1..]; - let name = rest.split(" |").next().unwrap_or("").trim(); - let info = lines.get(i + 1).map(|l| l.trim()).unwrap_or(""); - let is_gpu = info.starts_with("GPU"); - let vendor_s = info.split("vendor ").nth(1).unwrap_or("").split(", driver").next().unwrap_or("").trim(); - if !is_gpu || name.contains("NVIDIA") || vendor_s.contains("NVIDIA") { - continue; - } - let vendor = if vendor_s.contains("Advanced Micro") || name.contains("Radeon") || name.contains("AMD") { "amd" } else { "other" }; - let units = info.split(", ").find(|p| p.contains("compute units")).unwrap_or("").to_string(); - let mut c = card(cards.len(), name, vendor, "OpenCL", &units, idx); - c.device = idx.to_string(); - c.kind = if looks_integrated(name) { "integrated".into() } else { "discrete".into() }; - c.path = "prebuilt".into(); - apply_defaults(&mut c); - mark_sweep_support(&mut c); - cards.push(c); - } + if let Some(out) = inp.opencl { + let (devs, listed) = parse_opencl_list(&out); + d.opencl_listed = listed; + let (kept, dropped) = dedupe_platforms(devs.into_iter().filter(|dv| dv.vendor_word() != "nvidia").collect()); + d.dropped = dropped; + for dv in kept { + let vendor = dv.vendor_word(); + let (name, adapter) = resolve_name(&dv.name, vendor, &dv.bus, &adapters, &mut used); + let mut c = card(d.cards.len(), &name, vendor, "OpenCL", &dv.units, &dv.index); + c.code = dv.name.clone(); + c.bus = dv.bus.clone(); + c.platform = format!("{} ({}), driver {}", dv.platform, dv.platform_version, dv.driver); + c.kind = classify_kind(&dv.name, adapter.map(|i| &adapters[i])).into(); + c.vram_mb = if c.kind == "integrated" { 0 } else { dv.mem_mb }; + c.path = "prebuilt".into(); + apply_defaults(&mut c); + mark_sweep_support(&mut c); + d.cards.push(c); } - } else if cards.is_empty() { - notes.push("the OpenCL worker is not installed; AMD and Intel cards cannot be listed".into()); + } else if !inp.opencl_installed && d.cards.is_empty() { + d.notes.push("the OpenCL worker is not installed; AMD and Intel cards cannot be listed".into()); } - if cards.is_empty() { + if d.cards.is_empty() { // last resort: the names Windows knows, so the screen can at least say what is in the PC - if let Some(out) = run_timeout(Command::new(crate::platform::tool("powershell")).args(["-NoProfile", "-Command", "Get-CimInstance Win32_VideoController | ForEach-Object { $_.Name }"]), None, Duration::from_secs(15)) { - for n in out.lines().map(|l| l.trim()).filter(|l| !l.is_empty()) { - let vendor = if n.contains("NVIDIA") { "nvidia" } else if n.contains("AMD") || n.contains("Radeon") { "amd" } else { "other" }; - let mut c = card(cards.len(), n, vendor, if vendor == "nvidia" { "CUDA" } else { "OpenCL" }, "", "0"); - c.kind = if looks_integrated(n) { "integrated".into() } else { "unknown".into() }; - c.enabled = false; - c.reason = "seen by Windows, but no worker can drive it (no NVIDIA driver and no OpenCL worker)".into(); - cards.push(c); - } + for (i, a) in adapters.iter().enumerate().filter(|(_, a)| a.problem().is_none()) { + let vendor = vendor_of(&a.name); + let mut c = card(d.cards.len(), &a.name, vendor, if vendor == "nvidia" { "CUDA" } else { "OpenCL" }, "", ""); + c.bus = a.bus.clone(); + c.kind = if looks_integrated(&a.name) { "integrated".into() } else { "unknown".into() }; + c.enabled = false; + c.reason = "seen by Windows, but no worker can drive it (no NVIDIA driver and no OpenCL worker)".into(); + used.push(i); + d.cards.push(c); } } - cards + // the cards Windows lists with a problem (Code 43 after an eGPU hot-plug on PC 1, 5 October 2026): shown, never driven + for (i, a) in adapters.iter().enumerate() { + let Some(problem) = a.problem() else { continue }; + if used.contains(&i) || d.cards.iter().any(|c| c.name.trim().eq_ignore_ascii_case(a.name.trim())) { + continue; + } + let vendor = vendor_of(&a.name); + let mut c = card(d.cards.len(), &a.name, vendor, if vendor == "nvidia" { "CUDA" } else { "OpenCL" }, "", ""); + c.bus = a.bus.clone(); + c.kind = classify_kind(&a.name, Some(a)).into(); + mark_unusable(&mut c, &problem); + d.cards.push(c); + } + assign_keys(&mut d.cards); + d +} + +#[cfg(not(target_os = "macos"))] +pub fn detect(bins: &Bins) -> Detection { + // Windows' own view of every adapter: status and problem code (a Code 43 card is in no tool's list), memory and + // processor for the integrated call, the PCI address and the names for the rows + let adapters = adapters(); + // NVIDIA: nvidia-smi ships with the driver + let nvidia = run_timeout(Command::new(crate::platform::tool("nvidia-smi")).args(["--query-gpu=index,name,memory.total,pci.bus_id", "--format=csv,noheader"]), None, Duration::from_secs(10)); + let nvidia_limits = if nvidia.is_some() { nvidia_power_limits() } else { Default::default() }; + // OpenCL: the worker's own device list (AMD, Intel; NVIDIA shows there too and is skipped) + let opencl = bins.opencl.as_ref().and_then(|cl| run_timeout(Command::new(cl).arg("--list"), None, Duration::from_secs(15))); + assemble(Inputs { nvidia, nvidia_limits, cuda_worker: bins.cuda.is_some(), opencl, opencl_installed: bins.opencl.is_some(), adapters }) +} + +/// A listed card no worker can drive: off, no switch, the problem on the row and the hint under it. +pub fn mark_unusable(c: &mut CardState, problem: &str) { + c.problem = problem.to_string(); + c.enabled = false; + c.identities = 1; + c.state = "unusable".into(); + c.message = format!("not usable ({problem})"); + c.reason = PROBLEM_HINT.into(); + c.sweep_supported = false; + c.sweep_state = "unsupported".into(); + c.sweep_note = c.message.clone(); } /// Finds the binaries next to the engine (Windows, a plain folder) or in Contents/Resources/bin (macOS bundle). @@ -311,7 +728,7 @@ pub fn find_bins() -> Result { // the prebuilt CUDA worker needs NVIDIA's nvrtc64_*_0.dll next to it (as igneum-common.ps1 checks) let nvrtc = std::fs::read_dir(dir).ok().map(|rd| rd.flatten().any(|e| { let n = e.file_name().to_string_lossy().to_ascii_lowercase(); n.starts_with("nvrtc64_") && n.ends_with("_0.dll") })).unwrap_or(false); let cuda = opt("igneum-worker-cuda").filter(|_| nvrtc || cfg!(not(windows))); - return Ok(Bins { node, miner, metal: opt("igneum-bench"), cuda, opencl: opt("igneum-worker-opencl"), dir: dir.clone() }); + return Ok(Bins { node, miner, metal: opt("igneum-bench"), cuda, opencl: opt("igneum-worker-opencl"), telemetry: opt("igneum-gpu-telemetry"), dir: dir.clone() }); } } Err(format!("igneumd and igneum-miner were not found next to the app (looked in {})", candidates.iter().map(|c| c.display().to_string()).collect::>().join(", "))) @@ -322,3 +739,266 @@ pub fn node_version(node: &Path) -> String { .and_then(|o| o.lines().next().map(|l| l.trim().to_string())) .unwrap_or_default() } + +#[cfg(test)] +mod tests { + use super::*; + + // PC 1's adapter list on the evening of 5 October 2026, after the RX 9070 XT went in through the Sonnet eGPU box + // while the app ran: "AMD Radeon RX 9070 XT | status Error | ram 0 GB", "AMD Radeon(TM) Graphics | status OK | + // ram 2 GB" (the Ryzen iGPU, gfx1036 to OpenCL), "NVIDIA GeForce RTX 5090 | status OK". + fn pc1() -> Vec { + parse_adapters(r#"[{"Name":"AMD Radeon RX 9070 XT","Status":"Error","ConfigManagerErrorCode":43,"AdapterRAM":0,"VideoProcessor":"AMD Radeon Graphics Processor (0x7550)","PNPDeviceID":"PCI\\VEN_1002&DEV_7550&SUBSYS_0E4E1002&REV_C0\\6&1A2B3C4D&0&00000008"}, + {"Name":"AMD Radeon(TM) Graphics","Status":"OK","ConfigManagerErrorCode":0,"AdapterRAM":2147483648,"VideoProcessor":"AMD Radeon Graphics Processor (0x164E)","PNPDeviceID":"PCI\\VEN_1002&DEV_164E&SUBSYS_00000000&REV_C1\\4&2E5A1B3&0&0041"}, + {"Name":"NVIDIA GeForce RTX 5090","Status":"OK","ConfigManagerErrorCode":0,"AdapterRAM":4293918720,"VideoProcessor":"NVIDIA GeForce RTX 5090","PNPDeviceID":"PCI\\VEN_10DE&DEV_2B85&SUBSYS_10621043&REV_A1\\4&1F2E3D4C&0&0019"}]"#) + } + + #[test] + fn adapters_parse_with_status_code_and_memory() { + let a = pc1(); + assert_eq!(a.len(), 3); + assert_eq!(a[0].name, "AMD Radeon RX 9070 XT"); + assert_eq!(a[0].status, "Error"); + assert_eq!(a[0].code, 43); + assert_eq!(a[0].ram_mb, 0); + assert_eq!(a[0].problem().as_deref(), Some("Code 43")); + assert_eq!(a[1].ram_mb, 2048); + assert_eq!(a[1].problem(), None); + assert_eq!(a[2].problem(), None); + // a single adapter: ConvertTo-Json gives one object, not an array + let one = parse_adapters(r#"{"Name":"NVIDIA GeForce RTX 5090","Status":"OK","ConfigManagerErrorCode":0,"AdapterRAM":4293918720,"VideoProcessor":"NVIDIA GeForce RTX 5090","PNPDeviceID":"PCI\\VEN_10DE"}"#); + assert_eq!(one.len(), 1); + assert!(parse_adapters("not json").is_empty()); + } + + #[test] + fn problem_without_a_code_is_the_status_word() { + let a = Adapter { name: "x".into(), status: "Degraded".into(), ..Default::default() }; + assert_eq!(a.problem().as_deref(), Some("status Degraded")); + let fine = Adapter { name: "x".into(), status: "OK".into(), ..Default::default() }; + assert_eq!(fine.problem(), None); + let code12 = Adapter { name: "x".into(), status: "Error".into(), code: 12, ..Default::default() }; + assert_eq!(code12.problem().as_deref(), Some("Code 12")); + } + + #[test] + fn kind_from_pc1_lines_and_the_mac() { + let a = pc1(); + // the discrete cards, with and without their adapter row + assert_eq!(classify_kind("AMD Radeon RX 9070 XT", adapter_for("AMD Radeon RX 9070 XT", &a)), "discrete"); + assert_eq!(classify_kind("NVIDIA GeForce RTX 5090", adapter_for("NVIDIA GeForce RTX 5090", &a)), "discrete"); + assert_eq!(classify_kind("NVIDIA GeForce RTX 5090", None), "discrete"); + // the Ryzen iGPU: by its Windows name, and by the gfx code the OpenCL worker prints (no adapter row matches a code) + assert_eq!(classify_kind("AMD Radeon(TM) Graphics", adapter_for("AMD Radeon(TM) Graphics", &a)), "integrated"); + assert_eq!(classify_kind("gfx1036", adapter_for("gfx1036", &a)), "integrated"); + assert_eq!(classify_kind("gfx1036", None), "integrated"); + assert_eq!(classify_kind("gfx1036:xnack-", None), "integrated"); + // a discrete gfx code stays discrete; an Arc card is discrete despite "Intel" + assert_eq!(classify_kind("gfx1201", None), "discrete"); + assert_eq!(classify_kind("gfx1030", None), "discrete"); + assert_eq!(classify_kind("Intel(R) Arc(TM) A770 Graphics", None), "discrete"); + // an Intel iGPU by its processor string; an AMD discrete card's processor string ("AMD Radeon Graphics Processor") does not count + let intel = Adapter { name: "Intel(R) Iris(R) Xe Graphics".into(), processor: "Intel(R) Iris(R) Xe Graphics Family".into(), ram_mb: 1024, ..Default::default() }; + assert_eq!(classify_kind("Intel(R) Iris(R) Xe Graphics", Some(&intel)), "integrated"); + assert_eq!(a[0].processor, "AMD Radeon Graphics Processor (0x7550)"); + // shared memory under 1 GB on the adapter row makes an unknown name integrated; 0 says nothing + let small = Adapter { name: "Some iGPU".into(), ram_mb: 512, ..Default::default() }; + assert_eq!(classify_kind("Some iGPU", Some(&small)), "integrated"); + let unknown = Adapter { name: "Some card".into(), ram_mb: 0, ..Default::default() }; + assert_eq!(classify_kind("Some card", Some(&unknown)), "discrete"); + // the Mac: detect() labels Apple silicon "apple" itself; the name rules do not call it integrated + assert!(!looks_integrated("Apple M5 Max")); + assert_eq!(vendor_of("Apple M5 Max"), "apple"); + assert_eq!(vendor_of("gfx1036"), "amd"); + assert_eq!(vendor_of("AMD Radeon RX 9070 XT"), "amd"); + assert_eq!(vendor_of("NVIDIA GeForce RTX 5090"), "nvidia"); + } + + // PC 1 after Adrenalin 26.9.2 and a reboot (5 October 2026, evening): all three adapters OK, the Windows names, + // DEV ids and PCI addresses as the coordinator read them (the 5090's bus 01:00.0; the two AMD cards' addresses are + // not in that reading, so this fixture leaves them empty, as an old worker's --list would) + fn pc1_rebooted() -> Vec { + parse_adapters(r#"[{"Name":"AMD Radeon(TM) Graphics","Status":"OK","ConfigManagerErrorCode":0,"AdapterRAM":2147483648,"VideoProcessor":"AMD Radeon Graphics Processor (0x13C0)","PNPDeviceID":"PCI\\VEN_1002&DEV_13C0&SUBSYS_00000000&REV_C1\\4&2E5A1B3&0&0041","BusNumber":null,"Address":null}, + {"Name":"NVIDIA GeForce RTX 5090","Status":"OK","ConfigManagerErrorCode":0,"AdapterRAM":4293918720,"VideoProcessor":"NVIDIA GeForce RTX 5090","PNPDeviceID":"PCI\\VEN_10DE&DEV_2B85&SUBSYS_10621043&REV_A1\\4&1F2E3D4C&0&0019","BusNumber":1,"Address":0}, + {"Name":"AMD Radeon RX 9070 XT","Status":"OK","ConfigManagerErrorCode":0,"AdapterRAM":4293918720,"VideoProcessor":"AMD Radeon Graphics Processor (0x7550)","PNPDeviceID":"PCI\\VEN_1002&DEV_7550&SUBSYS_0E4E1002&REV_C0\\6&1A2B3C4D&0&00000008","BusNumber":null,"Address":null}]"#) + } + // the 0.3.9 app's five rows came from this shape of --list: two AMD platforms, each listing both AMD GPUs (the old + // 32.0.21042 ICD and the new 32.0.32015 one); the driver strings are the shape AMD's runtime prints, the numbers + // are the Windows driver builds (approximate: the OpenCL CL_DRIVER_VERSION was not captured) + const PC1_LIST: &str = "OpenCL devices (4):\n\ + [0] gfx1036 | AMD Accelerated Parallel Processing (OpenCL 2.1 AMD-APP (3617.0))\n\ + GPU, vendor Advanced Micro Devices, Inc., driver 3617.0 (PAL,HSAIL), OpenCL C 2.0, 2 compute units, 2200 MHz\n\ + global 16384 MiB, max alloc 13926 MiB, local 64 KiB, max work-group 256, sub-group extension: cl_khr_subgroups (no shuffle extension), AMD wavefront width 32\n\ + [1] gfx1201 | AMD Accelerated Parallel Processing (OpenCL 2.1 AMD-APP (3617.0))\n\ + GPU, vendor Advanced Micro Devices, Inc., driver 3617.0 (PAL,HSAIL), OpenCL C 2.0, 32 compute units, 2970 MHz\n\ + global 16368 MiB, max alloc 13912 MiB, local 64 KiB, max work-group 256, sub-group extension: cl_khr_subgroups (no shuffle extension), AMD wavefront width 32\n\ + [2] gfx1036 | AMD Accelerated Parallel Processing (OpenCL 2.1 AMD-APP (3649.0))\n\ + GPU, vendor Advanced Micro Devices, Inc., driver 3649.0 (PAL,HSAIL), OpenCL C 2.0, 2 compute units, 2200 MHz\n\ + global 16384 MiB, max alloc 13926 MiB, local 64 KiB, max work-group 256, sub-group extension: cl_khr_subgroups (no shuffle extension), AMD wavefront width 32\n\ + [3] gfx1201 | AMD Accelerated Parallel Processing (OpenCL 2.1 AMD-APP (3649.0))\n\ + GPU, vendor Advanced Micro Devices, Inc., driver 3649.0 (PAL,HSAIL), OpenCL C 2.0, 32 compute units, 2970 MHz\n\ + global 16368 MiB, max alloc 13912 MiB, local 64 KiB, max work-group 256, sub-group extension: cl_khr_subgroups (no shuffle extension), AMD wavefront width 32\n"; + const PC1_SMI: &str = "0, NVIDIA GeForce RTX 5090, 32607 MiB, 00000000:01:00.0\n"; + + fn pc1_inputs(list: &str) -> Inputs { + Inputs { nvidia: Some(PC1_SMI.into()), nvidia_limits: Default::default(), cuda_worker: true, opencl: Some(list.into()), opencl_installed: true, adapters: Some(pc1_rebooted()) } + } + + #[test] + fn opencl_list_parses_both_platforms_and_the_pci_field() { + let (devs, listed) = parse_opencl_list(PC1_LIST); + assert!(listed); + assert_eq!(devs.len(), 4); + assert_eq!(devs[1].index, "1"); + assert_eq!(devs[1].name, "gfx1201"); + assert_eq!(devs[1].platform, "AMD Accelerated Parallel Processing"); + assert_eq!(devs[1].platform_version, "OpenCL 2.1 AMD-APP (3617.0)"); + assert_eq!(devs[1].driver, "3617.0 (PAL,HSAIL)"); + assert_eq!(devs[1].units, "32 compute units"); + assert_eq!(devs[1].mem_mb, 16368); + assert_eq!(devs[1].bus, ""); + assert!(devs[1].is_gpu); + assert_eq!(devs[1].vendor_word(), "amd"); + let with_pci = PC1_LIST.replace("32 compute units, 2970 MHz\n", "32 compute units, 2970 MHz, pci 05:00.0\n"); + let (devs, _) = parse_opencl_list(&with_pci); + assert_eq!(devs[1].bus, "05:00.0"); + assert_eq!(devs[3].bus, "05:00.0"); + assert!(!parse_opencl_list("").1); + } + + #[test] + fn two_amd_platforms_give_one_entry_per_card() { + // without PCI addresses: by code and ordinal, the newer driver wins + let (devs, _) = parse_opencl_list(PC1_LIST); + let (kept, dropped) = dedupe_platforms(devs); + assert_eq!(kept.iter().map(|d| d.index.as_str()).collect::>(), vec!["2", "3"]); + assert_eq!(dropped.len(), 2); + assert!(dropped[0].starts_with("[0] gfx1036 on AMD Accelerated Parallel Processing (OpenCL 2.1 AMD-APP (3617.0)) is the same card as"), "{}", dropped[0]); + // with PCI addresses: by address; a card the winner does not list (the old ICD still serving a third card) is kept + let text = PC1_LIST + .replace("2 compute units, 2200 MHz\n", "2 compute units, 2200 MHz, pci 0c:00.0\n") + .replace("32 compute units, 2970 MHz\n", "32 compute units, 2970 MHz, pci 05:00.0\n") + + " [4] gfx1100 | AMD Accelerated Parallel Processing (OpenCL 2.1 AMD-APP (3617.0))\n GPU, vendor Advanced Micro Devices, Inc., driver 3617.0 (PAL,HSAIL), OpenCL C 2.0, 96 compute units, 2500 MHz, pci 09:00.0\n global 24560 MiB\n"; + let (devs, _) = parse_opencl_list(&text); + let (kept, dropped) = dedupe_platforms(devs); + // the old platform now lists three and wins on count: its three stay, the new platform's two are duplicates + assert_eq!(kept.iter().map(|d| d.index.as_str()).collect::>(), vec!["0", "1", "4"]); + assert_eq!(dropped.len(), 2); + // two real twins on one platform keep both entries (same code, different ordinal or address) + let twins = "OpenCL devices (2):\n [0] gfx1201 | P (v)\n GPU, vendor Advanced Micro Devices, Inc., driver 1.0, OpenCL C 2.0, 32 compute units, 2970 MHz\n [1] gfx1201 | P (v)\n GPU, vendor Advanced Micro Devices, Inc., driver 1.0, OpenCL C 2.0, 32 compute units, 2970 MHz\n"; + let (kept, dropped) = dedupe_platforms(parse_opencl_list(twins).0); + assert_eq!(kept.len(), 2); + assert!(dropped.is_empty()); + } + + #[test] + fn names_come_from_windows_by_bus_then_device_id_then_the_table() { + let a = pc1_rebooted(); + let mut used = Vec::new(); + assert_eq!(resolve_name("NVIDIA GeForce RTX 5090", "nvidia", "01:00.0", &a, &mut used).0, "NVIDIA GeForce RTX 5090"); + assert_eq!(resolve_name("gfx1201", "amd", "", &a, &mut used).0, "AMD Radeon RX 9070 XT"); + assert_eq!(resolve_name("gfx1036", "amd", "", &a, &mut used).0, "AMD Radeon(TM) Graphics"); + assert_eq!(used.len(), 3); + // every adapter is taken: a second gfx1036 gets the table's words, a code the table lacks stays a code + assert_eq!(resolve_name("gfx1036", "amd", "", &a, &mut used).0, "Ryzen integrated Radeon Graphics"); + assert_eq!(resolve_name("gfx9999", "amd", "", &a, &mut used).0, "gfx9999"); + assert_eq!(resolve_name("gfx1100", "amd", "", &[], &mut Vec::new()).0, "Radeon RX 7900 XTX / XT"); + // by PCI address when both sides have one, before any table + let mut b = pc1_rebooted(); + b[2].bus = "05:00.0".into(); + let mut used = Vec::new(); + assert_eq!(resolve_name("gfx1201", "amd", "05:00.0", &b, &mut used), ("AMD Radeon RX 9070 XT".to_string(), Some(2))); + // the device id alone names a card whose adapter name the table does not know + let mut c = pc1_rebooted(); + c[2].name = "AMD Radeon RX 9070 XT OC Edition".into(); + assert_eq!(resolve_name("gfx1201", "amd", "", &c, &mut Vec::new()).0, "AMD Radeon RX 9070 XT OC Edition"); + assert_eq!(a[2].device_id(), 0x7550); + assert_eq!(a[0].device_id(), 0x13C0); + assert_eq!(a[1].bus, "01:00.0"); + } + + #[test] + fn pc1_five_rows_become_three_cards_named_properly() { + let d = assemble(pc1_inputs(PC1_LIST)); + assert!(d.nvidia_listed && d.opencl_listed && d.adapters_listed); + let rows: Vec<(String, String, String, String, bool, String)> = d.cards.iter().map(|c| (c.name.clone(), c.key.clone(), c.kind.clone(), c.device.clone(), c.enabled, c.code.clone())).collect(); + assert_eq!(rows, vec![ + ("NVIDIA GeForce RTX 5090".into(), "nvidia:NVIDIA GeForce RTX 5090".into(), "discrete".into(), "0".into(), true, "NVIDIA GeForce RTX 5090".into()), + ("AMD Radeon(TM) Graphics".into(), "amd:gfx1036".into(), "integrated".into(), "2".into(), false, "gfx1036".into()), + ("AMD Radeon RX 9070 XT".into(), "amd:gfx1201".into(), "discrete".into(), "3".into(), true, "gfx1201".into()), + ]); + assert_eq!(d.cards[0].bus, "01:00.0"); + assert_eq!(d.cards[1].reason, INTEGRATED_REASON); + assert_eq!(d.cards[1].identities, 1); + assert_eq!(d.cards[2].identities, 8, "16 GB: 8 identities"); + assert_eq!(d.cards[2].vram_mb, 16368); + assert!(d.cards[2].platform.contains("3649.0")); + assert_eq!(d.dropped.len(), 2); + assert!(d.notes.is_empty(), "{:?}", d.notes); + assert_eq!(crate::hotplug::cards_line(&d.cards), "cards: NVIDIA GeForce RTX 5090 [discrete, off] | AMD Radeon(TM) Graphics [integrated, off] | AMD Radeon RX 9070 XT [discrete, off]"); + // the same machine before the eGPU: one platform, the iGPU alone; the keys do not depend on the index + let before = "OpenCL devices (1):\n [0] gfx1036 | AMD Accelerated Parallel Processing (OpenCL 2.1 AMD-APP (3617.0))\n GPU, vendor Advanced Micro Devices, Inc., driver 3617.0 (PAL,HSAIL), OpenCL C 2.0, 2 compute units, 2200 MHz\n global 16384 MiB\n"; + let d0 = assemble(pc1_inputs(before)); + assert_eq!(d0.cards[1].key, "amd:gfx1036"); + assert_eq!(d0.cards[1].device, "0"); + // and the diff between the two lists: the iGPU moved (device 0 to 2), the 9070 XT is new, nothing is removed + let diff = crate::hotplug::diff(&d0.cards, &d.cards, &|c| d.listed(c)); + assert_eq!(diff.unchanged, vec![0]); + assert_eq!(diff.moved.len(), 1); + assert_eq!(diff.moved[0].0, 1); + assert_eq!(diff.added.len(), 1); + assert_eq!(diff.added[0].name, "AMD Radeon RX 9070 XT"); + assert!(diff.removed.is_empty()); + } + + #[test] + fn keys_number_identical_cards() { + let mut cards = vec![card(0, "NVIDIA GeForce RTX 5090", "nvidia", "CUDA", "", "0"), card(1, "NVIDIA GeForce RTX 5090", "nvidia", "CUDA", "", "1"), card(2, "gfx1201", "amd", "OpenCL", "", "2")]; + cards[2].name = "AMD Radeon RX 9070 XT".into(); + assign_keys(&mut cards); + assert_eq!(cards.iter().map(|c| c.key.as_str()).collect::>(), vec!["nvidia:NVIDIA GeForce RTX 5090", "nvidia:NVIDIA GeForce RTX 5090#2", "amd:gfx1201"]); + } + + #[test] + fn unusable_card_row() { + let mut c = card(2, "AMD Radeon RX 9070 XT", "amd", "OpenCL", "", ""); + mark_unusable(&mut c, "Code 43"); + assert!(!c.enabled); + assert_eq!(c.state, "unusable"); + assert_eq!(c.message, "not usable (Code 43)"); + assert_eq!(c.reason, PROBLEM_HINT); + assert!(!c.present()); + } + + #[test] + fn integrated_default_is_off_with_the_row_words() { + let mut c = card(1, "gfx1036", "amd", "OpenCL", "2 compute units", "1"); + c.kind = classify_kind("gfx1036", None).into(); + apply_defaults(&mut c); + assert!(!c.enabled); + assert_eq!(c.identities, 1); + assert_eq!(c.reason, INTEGRATED_REASON); + let mut big = card(0, "NVIDIA GeForce RTX 5090", "nvidia", "CUDA", "32 GB", "0"); + big.kind = "discrete".into(); + big.vram_mb = 32768; + apply_defaults(&mut big); + assert!(big.enabled); + assert_eq!(big.identities, 8); + } +} + +/// The machine's RAM in MB (the prover default's RAM gate, src/provedefault.rs): Windows through +/// `Win32_OperatingSystem.TotalVisibleMemorySize` (KB), Linux through `/proc/meminfo`, macOS through `sysctl hw.memsize`; +/// None when unreadable (no gate). +pub fn total_ram_mb() -> Option { + if cfg!(windows) { + let out = run_timeout(Command::new(crate::platform::tool("powershell")).args(["-NoProfile", "-Command", "(Get-CimInstance Win32_OperatingSystem).TotalVisibleMemorySize"]), None, Duration::from_secs(20))?; + return out.replace('\0', "").trim().parse::().ok().map(|kb| kb / 1024); + } + if cfg!(target_os = "linux") { + let text = std::fs::read_to_string("/proc/meminfo").ok()?; + return text.lines().find(|l| l.starts_with("MemTotal:")).and_then(|l| l.split_whitespace().nth(1)).and_then(|kb| kb.parse::().ok()).map(|kb| kb / 1024); + } + let out = run_timeout(Command::new("sysctl").args(["-n", "hw.memsize"]), None, Duration::from_secs(5))?; + out.trim().parse::().ok().map(|b| b / (1024 * 1024)) +} diff --git a/app/igneum-app/src/ember.rs b/app/igneum-app/src/ember.rs new file mode 100644 index 000000000..d839135e5 --- /dev/null +++ b/app/igneum-app/src/ember.rs @@ -0,0 +1,1033 @@ +//! Ember Tune: every card tuned for MH per watt out of the box, and the fleet's results folded into a prior that a +//! new card starts from (docs/plans/ember-tune.md). Two knobs per card: the power limit (percent of the card's +//! default) and the core clock cap (MHz; 0 = unlocked). The memory clock is never touched, and a step that drags it +//! down is marked and cannot win. This file is the logic, driven by an explicit clock so the tests run without a +//! card: the plans (full, confirm, baseline), the per-step rows with their marks, the choice rule, the fleet record, +//! the prior lookup and the state machine. The engine (src/engine.rs, `tick_tune` and the `Cmd::Tune*` commands) +//! owns the processes: NVIDIA through nvidia-smi (`-pl`, `-lgc 0,`, `-rgc`; administrator rights, so only with +//! Power control on), AMD through igneum-gpu-telemetry (`--set-plimit`, `--set-gmax`, `--reset`; no elevation on +//! Windows), Apple measure only. +//! +//! The lines in the app log (and on stdout under --sweep, which the PC job reads): +//! TUNE start card=