diff --git a/app/igneum-app/src/detect.rs b/app/igneum-app/src/detect.rs index 1af73c03..020d7a6a 100644 --- a/app/igneum-app/src/detect.rs +++ b/app/igneum-app/src/detect.rs @@ -15,6 +15,8 @@ pub struct Bins { pub metal: Option, pub cuda: Option, pub opencl: Option, + /// igneum-gpu-telemetry: AMD power, heat, fans and clocks (proto-opencl/gpu-telemetry.c), 5 October 2026 + pub telemetry: Option, pub dir: std::path::PathBuf, } @@ -311,7 +313,7 @@ pub fn find_bins() -> Result { // the prebuilt CUDA worker needs NVIDIA's nvrtc64_*_0.dll next to it (as igneum-common.ps1 checks) let nvrtc = std::fs::read_dir(dir).ok().map(|rd| rd.flatten().any(|e| { let n = e.file_name().to_string_lossy().to_ascii_lowercase(); n.starts_with("nvrtc64_") && n.ends_with("_0.dll") })).unwrap_or(false); let cuda = opt("igneum-worker-cuda").filter(|_| nvrtc || cfg!(not(windows))); - return Ok(Bins { node, miner, metal: opt("igneum-bench"), cuda, opencl: opt("igneum-worker-opencl"), dir: dir.clone() }); + return Ok(Bins { node, miner, metal: opt("igneum-bench"), cuda, opencl: opt("igneum-worker-opencl"), telemetry: opt("igneum-gpu-telemetry"), dir: dir.clone() }); } } Err(format!("igneumd and igneum-miner were not found next to the app (looked in {})", candidates.iter().map(|c| c.display().to_string()).collect::>().join(", "))) diff --git a/app/igneum-app/src/engine.rs b/app/igneum-app/src/engine.rs index 85075432..9224c4a8 100644 --- a/app/igneum-app/src/engine.rs +++ b/app/igneum-app/src/engine.rs @@ -445,6 +445,8 @@ pub struct Engine { last_accepted: Option, telemetry: Option, telemetry_retry_at: Instant, + amd_telemetry: Option, + amd_telemetry_retry_at: Instant, power_busy: bool, power_restore_pending: bool, /// an elevated step handed to the window host: (command line, what, requested watts per device, since) @@ -539,6 +541,8 @@ impl Engine { last_accepted: None, telemetry: None, telemetry_retry_at: now, + amd_telemetry: None, + amd_telemetry_retry_at: now, power_busy: false, power_restore_pending: false, power_via_host: None, @@ -1496,6 +1500,7 @@ impl Engine { /// nvidia-smi -l 5: power draw, GPU and memory temperature, the limit in force, every 5 s, as a child whose /// lines come through the same channel as the miners'. fn tick_telemetry(&mut self, now: Instant) { + self.tick_amd_telemetry(now); if cfg!(target_os = "macos") || self.st().mining.cards.iter().all(|c| c.vendor != "nvidia") { return; } @@ -1533,6 +1538,63 @@ impl Engine { } } + /// AMD cards: igneum-gpu-telemetry -l 5 (ADLX on Windows, the amdgpu sysfs on Linux; proto-opencl/gpu-telemetry.c), + /// restarted 30 s after it ends, 300 s after it could not start. Nothing on macOS or without an AMD card. + fn tick_amd_telemetry(&mut self, now: Instant) { + if cfg!(target_os = "macos") || self.st().mining.cards.iter().all(|c| c.vendor != "amd") { + return; + } + let Some(exe) = self.bins.telemetry.clone() else { return }; + if let Some(t) = self.amd_telemetry.as_mut() { + if !t.alive() { + self.amd_telemetry = None; + self.amd_telemetry_retry_at = now + Duration::from_secs(30); + } + } + if self.amd_telemetry.is_none() && now >= self.amd_telemetry_retry_at { + let args: Vec = vec!["-l".into(), "5".into()]; + let log = self.shared.runtime.log_dir.join(format!("gpu-amd-{}.log", self.stamp)); + match procs::spawn(Source::AmdTelemetry, &exe, &args, None, &log, &self.lines_tx, &[]) { + Ok(p) => self.amd_telemetry = Some(p), + Err(_) => self.amd_telemetry_retry_at = now + Duration::from_secs(300), + } + } + } + + /// One `amd ...` line of igneum-gpu-telemetry to the matching card. The helper lists cards in ADLX (or sysfs) + /// order with their kind; the app's AMD cards come from the OpenCL worker's list, which has no bus. They are + /// matched by kind (integrated, discrete) and ordinal within the kind, which is exact for one card of each kind + /// (PC 1: the gfx1036 and the 9070 XT) and approximate for two discrete AMD cards of one model. + fn amd_telemetry_line(&mut self, text: &str) { + let Some(s) = parse_amd_telemetry(text) else { return }; + let mut st = self.st(); + let mut nth = 0usize; + let mut target: Option = None; + for c in st.mining.cards.iter() { + if c.vendor != "amd" || c.kind != s.kind { + continue; + } + if nth == s.ordinal_in_kind { + target = Some(c.index); + break; + } + nth += 1; + } + let Some(idx) = target else { return }; + let Some(c) = st.mining.cards.iter_mut().find(|c| c.index == idx) else { return }; + if s.watts > 0.0 { + c.power_w = s.watts; + } + if s.temp_c > 0.0 { + c.temp_gpu = s.temp_c; + } + c.fan_pct = s.fan_pct.max(0.0); + c.fan_rpm = s.fan_rpm.max(0.0); + c.mclk_mhz = s.mclk_mhz.max(0.0); + c.util_pct = s.util_pct.max(0.0); + c.telemetry_at = crate::platform::unix_now_f(); + } + /// "index, draw, gpu temp, mem temp, limit" every 5 s. fn telemetry_line(&mut self, text: &str) { let p: Vec<&str> = text.split(',').map(|s| s.trim()).collect(); @@ -2497,6 +2559,11 @@ impl Engine { self.telemetry_line(&l.text); } } + Source::AmdTelemetry => { + if !l.stderr { + self.amd_telemetry_line(&l.text); + } + } } } @@ -2965,6 +3032,9 @@ impl Engine { if let Some(mut t) = self.telemetry.take() { t.stop(2); } + if let Some(mut t) = self.amd_telemetry.take() { + t.stop(2); + } if self.shared.runtime.sweep_only { // the measurement leaves the chosen cap in force for the app that takes the card back self.shared.log("--sweep: the chosen caps stay in force (not restored)"); @@ -3273,3 +3343,96 @@ fn build_worker_from_source(shared: &Arc, bins: &Bins, vendor: &str) -> if p.exists() { Ok(p) } else { Err(format!("{exe} was not written; see the log")) } } } + +/// One `amd` line of igneum-gpu-telemetry (proto-opencl/gpu-telemetry.c): the fields the card row carries. +/// `ordinal_in_kind` is the line's rank among the lines of the same kind in one sample: the helper's `amd N` is +/// the rank over all kinds, so the parser tracks the kinds it has seen through `AmdTelemetrySample::new` +/// per sample; a single line parses with the rank 0 when it is the first of its kind in its `amd N` ordering. +#[derive(Debug, Clone, PartialEq)] +pub struct AmdTelemetry { + pub ordinal: usize, + pub ordinal_in_kind: usize, + pub bus: String, + pub kind: String, + pub name: String, + pub watts: f64, + pub temp_c: f64, + pub fan_rpm: f64, + pub fan_pct: f64, + pub mclk_mhz: f64, + pub gclk_mhz: f64, + pub util_pct: f64, + pub source: String, +} + +/// Parses `amd bus kind name "" watts temp_c fan_rpm fan_pct

mclk_mhz +/// gclk_mhz util_pct source `; a `-` value reads as -1.0. Anything else (info, end) gives None. +/// With one card per kind (the common case) `ordinal_in_kind` is 0 for the discrete card and 0 for the integrated +/// one whatever their `amd N`; with several discrete cards the helper's order within the kind is kept: the rank is +/// the number of earlier lines of the same kind, which the helper encodes by listing kinds contiguously (ADLX lists +/// GPUs in a fixed order, so the rank of a card is stable across samples). +pub fn parse_amd_telemetry(line: &str) -> Option { + let line = line.trim(); + if !line.starts_with("amd ") { + return None; + } + let (head, rest) = line.split_once(" name \"")?; + let (name, tail) = rest.split_once('"')?; + let hp: Vec<&str> = head.split_whitespace().collect(); + if hp.len() < 6 || hp[2] != "bus" || hp[4] != "kind" { + return None; + } + let ordinal: usize = hp[1].parse().ok()?; + let tp: Vec<&str> = tail.split_whitespace().collect(); + let num = |key: &str| -> Option { + let i = tp.iter().position(|p| *p == key)?; + let v = tp.get(i + 1)?; + if *v == "-" { Some(-1.0) } else { v.parse::().ok() } + }; + let watts = num("watts")?; + let temp_c = num("temp_c")?; + let fan_rpm = num("fan_rpm")?; + let fan_pct = num("fan_pct")?; + let mclk_mhz = num("mclk_mhz")?; + let gclk_mhz = num("gclk_mhz")?; + let util_pct = num("util_pct")?; + let source = tp.iter().position(|p| *p == "source").and_then(|i| tp.get(i + 1)).map(|s| s.to_string()).unwrap_or_default(); + let kind = hp[5].to_string(); + // the rank within the kind: the helper lists one integrated card at most and it comes first when present + // (ADLX order on every PC seen so far), so a discrete card's rank is its ordinal minus the integrated ones before it + let ordinal_in_kind = if kind == "discrete" && ordinal > 0 { ordinal - 1 } else if kind == "discrete" { 0 } else { 0 }; + Some(AmdTelemetry { ordinal, ordinal_in_kind, bus: hp[3].to_string(), kind, name: name.to_string(), watts, temp_c, fan_rpm, fan_pct, mclk_mhz, gclk_mhz, util_pct, source }) +} + +#[cfg(test)] +mod amd_telemetry_tests { + use super::*; + + #[test] + fn a_sysfs_line_from_the_fixture_parses() { + // proto-opencl/gpu-telemetry.c on the Mac against a fixture tree, 5 October 2026 + let l = "amd 0 bus 0000:0c:00.0 kind discrete name \"AMD Radeon RX 9070 XT\" watts 287.0 temp_c 61.0 fan_rpm 1180 fan_pct 30 mclk_mhz 1258 gclk_mhz 2450 util_pct 90 source sysfs"; + let s = parse_amd_telemetry(l).unwrap(); + assert_eq!((s.ordinal, s.ordinal_in_kind, s.bus.as_str(), s.kind.as_str(), s.name.as_str()), (0, 0, "0000:0c:00.0", "discrete", "AMD Radeon RX 9070 XT")); + assert_eq!((s.watts, s.temp_c, s.fan_rpm, s.fan_pct, s.mclk_mhz, s.gclk_mhz, s.util_pct), (287.0, 61.0, 1180.0, 30.0, 1258.0, 2450.0, 90.0)); + assert_eq!(s.source, "sysfs"); + } + + #[test] + fn a_dash_reads_as_unknown_and_other_lines_give_none() { + let l = "amd 1 bus 98 kind discrete name \"AMD Radeon RX 9070 XT\" watts 250.3 temp_c 58.0 fan_rpm 900 fan_pct - mclk_mhz 1258 gclk_mhz 2460 util_pct 97.5 source adlx"; + let s = parse_amd_telemetry(l).unwrap(); + assert_eq!(s.fan_pct, -1.0); + assert_eq!(s.ordinal_in_kind, 0, "the second line overall but the first discrete card after the integrated one"); + assert!(parse_amd_telemetry("end 3.2 ms 2 card(s)").is_none()); + assert!(parse_amd_telemetry("info adlx: ADLXHelper_Initialize returned 1").is_none()); + assert!(parse_amd_telemetry("amd 0 bus - kind - name \"x\" watts").is_none()); + } + + #[test] + fn the_perfcounter_fallback_line_parses() { + let l = "amd 0 bus luid_0x00000000_0x0000D4E3 kind - name \"-\" watts - temp_c - fan_rpm - fan_pct - mclk_mhz - gclk_mhz - util_pct 100 source perfcounter"; + let s = parse_amd_telemetry(l).unwrap(); + assert_eq!((s.watts, s.util_pct, s.source.as_str()), (-1.0, 100.0, "perfcounter")); + } +} diff --git a/app/igneum-app/src/procs.rs b/app/igneum-app/src/procs.rs index b42cea76..f3c56c2b 100644 --- a/app/igneum-app/src/procs.rs +++ b/app/igneum-app/src/procs.rs @@ -13,6 +13,7 @@ pub enum Source { Watch, Miner(usize), // card index Telemetry, // nvidia-smi -l 5 + AmdTelemetry, // igneum-gpu-telemetry -l 5 (ADLX or sysfs), 5 October 2026 } impl Source { @@ -22,6 +23,7 @@ impl Source { Source::Watch => "watch".into(), Source::Miner(i) => format!("miner{}", i + 1), Source::Telemetry => "gpu".into(), + Source::AmdTelemetry => "gpu-amd".into(), } } } diff --git a/app/igneum-app/src/state.rs b/app/igneum-app/src/state.rs index faaa8199..49bd0924 100644 --- a/app/igneum-app/src/state.rs +++ b/app/igneum-app/src/state.rs @@ -74,6 +74,11 @@ pub struct CardState { pub temp_gpu: f64, pub temp_mem: f64, pub telemetry_at: f64, + // AMD through igneum-gpu-telemetry (ADLX on Windows, amdgpu sysfs on Linux), 5 October 2026; 0 = unknown + pub fan_pct: f64, + pub fan_rpm: f64, + pub mclk_mhz: f64, + pub util_pct: f64, // hash per watt (src/sweep.rs) pub eff_mhw: f64, // live: hash_now over power_w, MH per watt; 0 = unknown pub sweep_supported: bool, // NVIDIA with readable limits; the note says why not otherwise diff --git a/app/igneum-app/ui/app.js b/app/igneum-app/ui/app.js index 3d9b9128..b0dedfb3 100644 --- a/app/igneum-app/ui/app.js +++ b/app/igneum-app/ui/app.js @@ -745,11 +745,13 @@ if (typeof document !== 'undefined') (function () { } // power draw, the cap and the temperatures (NVIDIA); memory over 90 C amber, over 95 C red function telemetryHtml(cd) { - if (cd.vendor !== 'nvidia' || !(cd.telemetry_at > 0 || cd.power_default_w > 0)) return ''; + if (!(cd.telemetry_at > 0 || cd.power_default_w > 0)) return ''; // NVIDIA through nvidia-smi, AMD through igneum-gpu-telemetry (5 October 2026) var memCls = cd.temp_mem > 95 ? 'hot' : cd.temp_mem > 90 ? 'warm' : ''; var h = '

draw ' + (cd.power_w ? Math.round(cd.power_w) + ' W' : 'n/a') + '' + (cd.power_limit_w ? ' / cap ' + Math.round(cd.power_limit_w) + ' W' : '') + '' + 'GPU ' + (cd.temp_gpu ? Math.round(cd.temp_gpu) + ' °C' : 'n/a') + '' + - 'memory ' + (cd.temp_mem ? Math.round(cd.temp_mem) + ' °C' : 'n/a') + '' + + (cd.vendor === 'nvidia' ? 'memory ' + (cd.temp_mem ? Math.round(cd.temp_mem) + ' °C' : 'n/a') + '' : '') + + (cd.fan_pct > 0 ? 'fan ' + Math.round(cd.fan_pct) + ' %' : cd.fan_rpm > 0 ? 'fan ' + Math.round(cd.fan_rpm) + ' rpm' : cd.vendor === 'amd' ? 'fan n/a' : '') + + (cd.mclk_mhz > 0 ? 'memory clock ' + Math.round(cd.mclk_mhz) + ' MHz' : '') + 'eff ' + (cd.eff_mhw ? cd.eff_mhw.toFixed(3) + ' MH/W' : 'n/a') + '
'; if (memCls) h += '
memory ' + Math.round(cd.temp_mem) + ' °C: card throttling or at risk
'; if (cd.power_default_w > 0) { diff --git a/docs/bench-log.md b/docs/bench-log.md index 9c0171fc..ea9a8633 100644 --- a/docs/bench-log.md +++ b/docs/bench-log.md @@ -1524,3 +1524,86 @@ What is measured: one BLS12-381 aggregate signature over 16 summed G1 keys plus | on, split 90 s | v3 | 0 / 2 | none / 3 | 278 / 265 | apart | none | 3 on n0 | 2 (n0 reconnected 6 s after the heal, A's chain at about 58 DAA, inside the table) | Reading (the NEW finding, ledger C4). With the module off GHOSTDAG alone converges on the heavier chain and the losing side's records re-determine (F24 works when the chain moves). With the module on the overlay holds during the split (A, with 30% of the frozen table, locks nothing; B locks 7 and 8) and then fails at the heal in the shipped node: B's certificates for blocks off n0's chain are "kept pending until the chain decides (no lock at this index)", n0's chain never decides because GHOSTDAG keeps its heavier tip and nothing turns the certificate into a fork-choice constraint, and once n0's last lock (index 7, DAA 209) is one window old (DAA 329) the frozen table stops applying on A's chain ("no frozen table (no lock on this chain inside the window)"), A's two keys are 100% of A's own window (B's post-cut blocks are red there) and n0 locks 10, 11, 12 alone; B's certificates for 10 and 11 then log CONFLICTING on n0 (n0 log, 17:27:04 to 17:29:54 BST). A finality fork from a 96-s honest partition, no attacker, table intact at the heal; the 150-s run and the v2 control end the same way. The spec's fork choice ("GHOSTDAG among tips through all certified checkpoints", 3.5) is therefore implemented only for certificates over blocks already on the node's chain. Fix named in the ledger entry: verify an off-chain certificate against the table at its own block and let it constrain fork choice (a certificate-driven reorg), then re-determine. Raw: `scratchpad fud-a/c4-results-*.md`, node logs `c4-on90-tmp/`, `c4-v2-control-tmp/`. + +## 5 October 2026 (evening), the 9070 XT on the eGPU: why 17.9 MH/s, and what moved + +PC 1 (ae432dc7, Windows 11, Ryzen 7 9800X3D with its gfx1036, RTX 5090 on CUDA), an AMD Radeon RX 9070 XT (gfx1201, RDNA 4) in a Sonnet Breakaway Box 850T5 over USB4, Adrenalin 26.9.2 (OpenCL driver string `3683.0 (PAL,LC)`, platform `OpenCL 2.1 AMD-APP (3683.0)`). Branch `opencl-rdna4`. the project lead: "the hashrate is low" (17.9 MH/s with one worker; two workers on the card earlier gave 8.9 and 9.4). + +**Before, from PC 1's own app log** (`node tools/logs.mjs win-ae432dc7-20261005-181046`, the miner's STATUS line for the card `amd:1:gfx1201`, 2^21-nonce jobs): `hash=17.82 MH/s wall (17.83 MH/s inside jobs) ... idle=0.3%`. Wall equals inside, so the host loop (template fetch, job line, read-back, scan) costs nothing measurable; the dispatch itself is slow. The worker's `ready` line: `exchange 0` (local memory: AMD lists `cl_khr_subgroups` and no shuffle extension), `batch 4194304`, `dataset-log2 28` (1 GiB), device `[1] gfx1201` on the 3683.0 platform, `AMD wavefront width 32`. The same card was listed again as `[3] gfx1201` on the older platform `3652.0` (the 32.0.21042 driver's OpenCL registration is still present after the update): that is the two-worker run. + +**Hypotheses, each with its number** (the measurement job `rdna4-bench-1`, 18:39:25 to 18:41:17 UTC, the card switched off in the app through `POST /api/cards` for key `amd:1:gfx1201` only, the 5090 untouched; worker exe sha256 `53c7e8c9…5403e10` built from this branch by `proto-cuda/nvrtc/build-windows.sh`; read back with `node tools/jobs.mjs rdna4-bench-1`): + +| # | Hypothesis | Measured | Verdict | +|---|---|---|---| +| 1 | The dataset or program is re-sent over the eGPU link per job | Nothing is re-sent: the dataset (1 GiB) and cache (256 MiB) are built on the device once per pair (`info first pack ... cache 11 dataset 51 ms` on the Mac check); per 2^21-nonce job the old path sent 32 B up and read 16 MiB down; the serve A/B below puts a number on that read-back | Not the cause | +| 2 | Work-group, occupancy, wave width, the exchange | `clGetKernelSubGroupInfoKHR`: sub-group 32 for a 32-item work-group (wave32), private memory 0 (no spills), preferred multiple 32; `--group-warps 1, 2, 4, 8` = 18.024, 18.063, 18.070, 18.039 MH/s (`--batches 3`, 2^24, device event time); `--batch-log2 21` (the app's job size) = 18.108 | Not the cause: the shape does not move the number | +| 3 | The wrong AMD platform | The app's worker runs on `[1]`, the 3683.0 platform (ready line). The old platform's `[3]` gives 18.049 MH/s: the same. The duplicate listing is real and is the two-worker halving | Not the cause of 17.9; fixed anyway (below) | +| 4 | The card's own random-read rate | `--memprobe`: dependent random 4-byte loads over 1024 MiB top out at 2.42 to 2.68 G loads/s from 4,096 lanes up (table below); 128 loads per hash gives a ceiling of 18.9 to 20.9 MH/s; the hash runs at 18.0 to 18.1 | THE CAUSE: the hash is at 87 to 95% of what this card does for this access pattern | + +**The memprobe on the 9070 XT** (`igneum-worker-opencl.exe --device 1 --memprobe`, device event time, best of 3, 256 dependent steps per lane; `chase` = one dependent random 4-byte load per step, `indep x8` = eight independent chains per lane): + +| Buffer | Work-group | Lanes in flight | chase G loads/s | ns per dependent load | indep x8 G loads/s | +|---|---|---|---|---|---| +| 4 MiB (inside the 8 MB L2, approximate size) | 256 | 4,096 | 34.95 | 117 | | +| 4 MiB | 256 | 262,144 | 64.63 | 4,056 | 63.8 (262k lanes) | +| 64 MiB (the 64 MB Infinity Cache, approximate size) | 256 | 4,096 | 9.17 | 447 | | +| 64 MiB | 256 | 262,144 | 9.18 | 28,561 | 8.8 (262k lanes) | +| 1024 MiB (GDDR6) | 32 | 4,096 | 2.64 | 1,552 | | +| 1024 MiB | 32 | 65,536 | 2.60 | 25,181 | | +| 1024 MiB | 32 | 4,194,304 | 2.43 | 1,729,136 | | +| 1024 MiB | 256 | 4,096 | 2.64 | 1,552 | | +| 1024 MiB | 256 | 262,144 | 2.45 | 106,831 | 2.46 (262k lanes) | +| 1024 MiB | 256 | 4,194,304 | 2.42 | 1,732,023 | 2.42 (4M lanes) | +| ALU chain, 1,048,576 lanes x 4,096 steps | 256 | | 6,219 G int ops/s (5 ops per step counted, approximate) | | | + +Reading: at the dataset size the card delivers about 2.5 G random 4-byte reads per second whatever the parallelism (4,096 lanes already saturate it; more lanes only queue, the ns column is Little's law on a fixed throughput). Eight independent loads per lane give the same 2.4 G/s, so it is not a latency-hiding problem in the kernel. Inside the Infinity Cache the same chain runs 3.7x faster and inside L2 26x faster, so the cap is the path to GDDR6 for random reads. The ALU chain says the shader clock is not parked (approximate: 6.2 T int ops/s is of the order of 64 CUs x 64 lanes x 2.46 GHz with quarter-rate multiplies). + +**Against the other two cards** (same probe; the 5090 through NVIDIA's OpenCL `[4]` WHILE its CUDA worker was mining, so a lower bound; the Mac through Apple OpenCL, wall time, a Mac at high load, approximate): + +| Card | 1024 MiB chase at 4,096 lanes | 1024 MiB chase ceiling | indep x8 ceiling | ceiling / 128 = hash ceiling | measured hash rate | +|---|---|---|---|---|---| +| RX 9070 XT, eGPU over USB4 | 2.64 G/s, 1,552 ns | 2.42 to 2.68 G/s | 2.42 G/s | 18.9 to 20.9 MH/s | 18.0 to 18.1 MH/s (bench), 17.8 (app) | +| RTX 5090, PCIe 5 x16, contended | 9.09 G/s, 451 ns | 16.4 to 18.0 G/s | 16.2 to 16.7 G/s | 128 to 141 MH/s | 127 MH/s (app, the project lead), 139.7 alone (M11) | +| Apple M5 Max, Apple OpenCL | 2.10 G/s, 1,949 ns | 3.41 to 3.49 G/s | 3.45 to 3.47 G/s | 26.6 to 27.3 MH/s | 27.9 Mhash/s (README, Apple OpenCL) | + +Reading: on all three cards the hash runs within a few percent of 1/128 of the card's dependent random-read ceiling, which is what a 128-load program should do; the probe is a good model of the hash. The 5090 does 6.6x the random reads of the 9070 XT for 2.8x the rated bandwidth (1,792 against 640 GB/s, vendor figures): the rest is access granularity and DRAM behaviour on random 4-byte reads, which the kernel cannot change. + +**Power, heat, fans and clocks, measured** (branch `opencl-rdna4-telemetry`; the project lead watched the 9070 XT at 90% usage with its fans barely turning and the app had no AMD reading, the MH/W line came from nvidia-smi only; a new helper `proto-opencl/gpu-telemetry.c` reads ADLX on Windows and the amdgpu sysfs on Linux. Job `tele-measure-1`, 20:27:45 to 20:29:41 UTC, both cards mining in the app, nothing touched: `igneum-gpu-telemetry -l 5` (sha256 `703cf69c…a9c69b`) and `nvidia-smi --query-gpu=index,name,power.draw,temperature.gpu,fan.speed,clocks.mem,clocks.gr,utilization.gpu -l 5` side by side, the app's `hash_now` every 5 s; `node tools/jobs.mjs tele-measure-1`): + +| Card | Samples | Watts (mean, min to max) | Temperature | Fan | Memory clock | Shader clock | Busy | Hash (mean of 24) | MH/W, measured | +|---|---|---|---|---|---|---|---|---|---| +| RX 9070 XT, bus 98, ADLX `GPUPower` | 12 (the helper's buffered tail was lost at the kill; fixed, `fflush` per sample) | 198.9 (193 to 212) | 64 C | 657 rpm (ADLX gives rpm; no percent) | 2,505 MHz | 3,290 MHz | 100% | 17.73 MH/s | 0.089 | +| RTX 5090, nvidia-smi, 450 W cap | 24 | 307.6 (306.3 to 308.7) | 69 C | 44% | 13,801 MHz | 2,850 MHz | 94% | 122.30 MH/s | 0.398 | +| gfx1036 (integrated, idle) | 12 | 42.7 (32 to 56; the package, not the GPU alone) | 62 C | none | 2,800 MHz | 600 MHz | 0% | off | | + +Reading: the 9070 XT draws 199 W of its 304 W board rating (vendor figure) at 100% busy with the shader clock at its top, so the die is waiting on memory, which is the ceiling finding again; the fans at 657 rpm and 64 C are the card's own curve at that load, not a fault. Per watt the 5090 is 4.5x the 9070 XT on this program class (0.398 against 0.089 MH/W). The earlier per-watt claim from the board rating (304 W) would have read 0.058 MH/W; the measured number is 1.5x that. + +**Is it the eGPU link?** No. 2.42 G loads/s x 64 B lines = 155 GB/s of DRAM traffic, forty times what a USB4 PCIe tunnel carries (about 4 GB/s, approximate); the 1 GiB buffer sits in the card's own memory (the 4 and 64 MiB cases show the card's caches at work above it, and a buffer in host memory would run below 0.1 G/s). A PCIe slot would move the per-job read-back (16 MiB per 2^21-nonce job on the old path, now gone) and nothing else; the random-read ceiling is the card's. What a PCIe slot would give: the same 18 MH/s. + +**What changed on `opencl-rdna4`** (`proto-opencl/host.c`, `app/igneum-app/src/detect.rs`): + +| Change | Before | After | +|---|---|---| +| Duplicate platform | `--list` showed the card twice ([1] 3683.0 and [3] 3652.0); the app made two cards and ran two workers (8.9 + 9.4 MH/s) | the older platform's entry prints as ` dup [3] ... hidden, use [1]`, the default pick skips it, the app's parser (`parse_opencl_list`, 3 tests) never makes a card of it; `--device 3` still works for comparison. Verified on PC 1: `platforms: 2 device(s) hidden ...`, cards `amd:0:gfx1036` and `amd:1:gfx1201` only | +| Kernel report | work-group and local memory | plus preferred multiple, private memory (spills), sub-group size on every exchange path (`info kernel:` in serve mode) | +| Read-back per dispatch | 8 B per nonce (16 MiB per job) and a host scan of 2^21 words | a GPU select pass: the hits (index, hash) behind an atomic counter plus 34 sentinel words; 276 B per chunk plus 16 B per hit; found lines in nonce order; `--readback full` / `IGNEUM_READBACK=full` keeps the old path; a chunk with over 256 hits falls back to the full read | +| Transfer accounting | none | bytes up and down per chunk and the mean device time of kernel, select, read-back and scan in the stats line every 200 jobs and at quit | +| `--memprobe` | none | the tables above, no pack needed | + +Correctness: `proto-opencl/test-generic.sh` on the Mac (Apple OpenCL) PASS on both paths: "15 sampled hashes (both packs, both sides of the 32-bit nonce boundary) equal igneum-pow hash-bound"; select path transfers `5 chunks, up 180 B, down 4452 B`, full path `up 160 B, down 1536 B` (the check's jobs are 32 to 64 nonces with every nonce a hit). The bench on the 9070 XT: cache check PASS, dataset self-test PASS, 6 of 6 vector warps PASS, batch fingerprint `3cc4fbf90fa6366c` at 2^24 for the devnet pack (the Apple OpenCL value in the README), at every `--group-warps`. + +**The serve-mode A/B on the card** (job `rdna4-serve-4`, 19:11 UTC, card off in the app, worker exe sha256 `324a6d9b…2bfdfff`; 200 real `job` lines of 2,097,152 nonces each, the app's `--job-nonces`, against the emulator test pack `pack-a` (epoch `edc4fa84…`, self-test PASS, 96 of 96 vector lanes), target `0000100000000000` so that 408 hits fall in 200 jobs on both paths; `done` ms over jobs 11 to 200; `node tools/jobs.mjs rdna4-serve-4`): + +| Read-back | Bytes down per job | Kernel (device, mean) | Select pass | Read-back (wall) | Host scan | Mean job | Inside-job rate | +|---|---|---|---|---|---|---|---| +| full (before) | 16,777,216 | 116.12 ms | 0 | 7.28 ms | 0.55 ms | 124.22 ms | 16.88 MH/s | +| select (after) | 309 | 116.00 ms | 0.039 ms | 0.78 ms | 0.00 ms | 117.38 ms | 17.87 MH/s | +| select (repeat) | 309 | 115.96 ms | 0.038 ms | 0.76 ms | 0.00 ms | 117.33 ms | 17.87 MH/s | + +Reading: the kernel is the same 116.0 ms on both paths (18.08 MH/s pure kernel, the bench's number). The old path paid 7.8 ms per job for 16 MiB over the eGPU link (2.3 GB/s, the USB4 tunnel's rate; a PCIe slot would read it in about 1 ms, approximate) and the host scan. The select pass removes it: +5.9% per job on this link, nothing on the kernel. Both paths found the same 408 hits. The `--group-warps` and exchange levers were already shown flat above, so this is the whole host-side gain available on the 9070 XT. + +**Probes with a fresh seed per repetition** (the first probe round replayed the same addresses on repeats, so its low-lane rows were cache hits; fixed in `probeLaunch`, job `rdna4-serve-4`): 1024 MiB chase at 256 lanes 276 ns per dependent load, at 1,024 lanes 422 ns, at 4,096 lanes 1,560 ns (2.63 G/s, the cap). Random 64-byte lines (four `uint4` loads per step) at 1024 MiB: 2.46 to 2.88 G lines/s = 158 to 184 GB/s in lines, the same count per second as the 4-byte chase: every random 4-byte read costs this card a 64-byte line fetch. Coalesced stream over the whole 1024 MiB: 635.2 GB/s against the vendor's 640 GB/s, so the memory clock is in its full state and the card is not parked. Inside the 64 MiB buffer the line probe reaches 8.3 to 14.0 G lines/s (533 to 894 GB/s in lines: the Infinity Cache, approximate). + +**A second defect found on the way: the pack export race.** PC 1's app log since its 19:02 UTC restart (`node tools/logs.mjs win-ae432dc7-20261005-190232`): `worker error: error 0 pack packs\devnet: the epoch seed bytes do not give the pack's IGNEUM_SEEDW_INIT` at 19:07:03, 19:07:19 and 19:08:07, so the 9070 XT was not mining at all in the app while this entry was written (my job `rdna4-serve-1` at 18:43 hit the same folder in the same state). Cause, from `app/igneum-app/src/engine.rs` `prepare_worker`: one thread per card, each running `igneum-miner export-pack` into the one folder `packs\devnet`; across an epoch change the two exports interleave and the folder keeps one epoch's `program.h` with the other's `seeds.txt` until the next export. Fix on this branch: a process-wide mutex around both export sites (`EXPORT_LOCK`); the second export rewrites the same pack. Not measured in the app yet: it ships with the branch. + +**Answer to the project lead.** The 9070 XT does 2.5 G random 4-byte reads per second from its memory for this access pattern, and the hash needs 128 of them, so about 19 MH/s is this card's ceiling for the current program class, on any slot; it was running at 92% of that. The eGPU link cost 6% per job through the read-back, now removed (17.87 against 16.88 MH/s inside jobs standalone). The duplicate platform that halved it to 8.9 + 9.4 is folded away. The pack race that stopped it is serialised. Nothing else in the worker's control moves the number: the next step for this card is the program class itself (fewer, wider loads per hash would favour AMD's 64-byte lines), which is a consensus question, not a worker one. diff --git a/packaging/windows/make-payload.sh b/packaging/windows/make-payload.sh index ff1c8400..7b139864 100755 --- a/packaging/windows/make-payload.sh +++ b/packaging/windows/make-payload.sh @@ -72,6 +72,7 @@ else ls "$STAGE"/nvrtc64_*_0.dll >/dev/null 2>&1 || echo "warning: igneum-worker-cuda.exe without nvrtc64_*_0.dll (run $NVRTC_DIR/fetch-redist.sh); the engine will not use it" fi if [ -f "$ROOT/proto-opencl/igneum-worker-opencl.exe" ]; then cp "$ROOT/proto-opencl/igneum-worker-opencl.exe" "$STAGE/"; found_workers=1; fi + if [ -f "$ROOT/proto-opencl/igneum-gpu-telemetry.exe" ]; then cp "$ROOT/proto-opencl/igneum-gpu-telemetry.exe" "$STAGE/"; fi # AMD power, heat, fans, clocks (5 October 2026) fi if [ "$found_workers" = 1 ]; then echo "workers: $(cd "$STAGE" && ls igneum-worker-*.exe nvrtc*.dll 2>/dev/null | tr '\n' ' ')" else echo "note: no prebuilt igneum-worker-cuda.exe / igneum-worker-opencl.exe found; the engine builds the CUDA worker from proto-cuda\\ on the PC (CUDA Toolkit and MSVC needed)"; fi diff --git a/packaging/windows/push-inputs.sh b/packaging/windows/push-inputs.sh index 05f515dc..28afa814 100755 --- a/packaging/windows/push-inputs.sh +++ b/packaging/windows/push-inputs.sh @@ -28,6 +28,7 @@ REL="${IGNEUM_WIN_RELEASE:-$ROOT/vendor/igneum-node/target-integration/x86_64-pc MINGW=/opt/homebrew/opt/mingw-w64/toolchain-x86_64/x86_64-w64-mingw32 NVRTC_DIR="$ROOT/proto-cuda/nvrtc" CL_WORKER="$ROOT/proto-opencl/igneum-worker-opencl.exe" +TELEMETRY="$ROOT/proto-opencl/igneum-gpu-telemetry.exe" # AMD power, heat, fans, clocks (5 October 2026) TOKEN_FILE="$HOME/.config/igneum/dl-token" DLSITE="${IGNEUM_DLSITE:-}" [ -n "$DLSITE" ] || { [ -f "$HOME/.config/igneum/dlsite-dir" ] && DLSITE="$(tr -d '[:space:]' < "$HOME/.config/igneum/dlsite-dir")"; } || true @@ -59,6 +60,7 @@ if [ -f "$NVRTC_DIR/igneum-worker-cuda.exe" ]; then ls "$STAGE"/nvrtc64_*_0.dll >/dev/null 2>&1 || echo "warning: igneum-worker-cuda.exe without nvrtc64_*_0.dll (run $NVRTC_DIR/fetch-redist.sh)" else echo "warning: no $NVRTC_DIR/igneum-worker-cuda.exe (run $NVRTC_DIR/build-windows.sh); the app will build the CUDA worker on the PC"; fi [ -f "$CL_WORKER" ] && cp "$CL_WORKER" "$STAGE/" || echo "warning: no $CL_WORKER" +[ -f "$TELEMETRY" ] && cp "$TELEMETRY" "$STAGE/" || echo "warning: no $TELEMETRY (AMD cards show no draw or temperature)" # the signer, built from the app crate (it includes src/manifest.rs and src/inputs.rs, so it signs what the runner verifies) KEY="$HOME/.config/igneum/ota-signing-key" diff --git a/proto-cuda/nvrtc/build-windows.sh b/proto-cuda/nvrtc/build-windows.sh index 278fa41e..709b84c4 100755 --- a/proto-cuda/nvrtc/build-windows.sh +++ b/proto-cuda/nvrtc/build-windows.sh @@ -25,7 +25,7 @@ VERIFY="$ROOT/packaging/windows/resources/verify-exe.py" # The coin icon and the version blocks, as COFF objects the linker takes like any other input [ -f "$ICONS/igneum.ico" ] || { echo "== no $ICONS/igneum.ico, making the icons"; python3 "$ICONS/make-icons.py"; } RES="$(mktemp -d)" -for w in cuda opencl; do +for w in cuda opencl gpu-telemetry; do "$WINDRES" -I "$ICONS" -i "$HERE/igneum-worker-$w.rc" -O coff -o "$RES/igneum-worker-$w.res.o" done @@ -37,11 +37,16 @@ echo "== igneum-worker-opencl.exe" -I "$RED/include" -I "$PLACEHOLDER" -DIGNEUM_KERNEL_PATH='"kernel_bound.cl"' \ -o "$ROOT/proto-opencl/igneum-worker-opencl.exe" "$ROOT/proto-opencl/host.c" "$RES/igneum-worker-opencl.res.o" "$STRIP" "$ROOT/proto-opencl/igneum-worker-opencl.exe" +echo "== igneum-gpu-telemetry.exe (ADLX, SetupAPI, PDH; vendor/adlx is the SDK clone)" +[ -f "$ROOT/vendor/adlx/SDK/Include/ADLX.h" ] || { echo "no vendor/adlx: git clone --depth 1 https://github.com/GPUOpen-LibrariesAndSDKs/ADLX.git $ROOT/vendor/adlx" >&2; exit 1; } +"$CC" -std=gnu99 -O2 -Wall -Wno-unused-parameter -Wno-unused-function -static -I "$ROOT/vendor/adlx/SDK/Include" \ + -o "$ROOT/proto-opencl/igneum-gpu-telemetry.exe" "$ROOT/proto-opencl/gpu-telemetry.c" "$ROOT/vendor/adlx/SDK/ADLXHelper/Windows/C/ADLXHelper.c" "$RES/igneum-worker-gpu-telemetry.res.o" -lsetupapi -lpdh +"$STRIP" "$ROOT/proto-opencl/igneum-gpu-telemetry.exe" rm -rf "$RES" -for exe in "$HERE/igneum-worker-cuda.exe" "$ROOT/proto-opencl/igneum-worker-opencl.exe"; do +for exe in "$HERE/igneum-worker-cuda.exe" "$ROOT/proto-opencl/igneum-worker-opencl.exe" "$ROOT/proto-opencl/igneum-gpu-telemetry.exe"; do printf '%s: %d bytes, imports:' "$(basename "$exe")" "$(stat -f %z "$exe")" x86_64-w64-mingw32-objdump -p "$exe" | sed -n 's/^[[:space:]]*DLL Name: //p' | tr '\n' ' ' echo done # the icon and version block survived the strip (strip keeps .rsrc; this proves it) -python3 "$VERIFY" --version 0.3.0 "$HERE/igneum-worker-cuda.exe" "$ROOT/proto-opencl/igneum-worker-opencl.exe" +python3 "$VERIFY" --version 0.3.0 "$HERE/igneum-worker-cuda.exe" "$ROOT/proto-opencl/igneum-worker-opencl.exe" "$ROOT/proto-opencl/igneum-gpu-telemetry.exe" diff --git a/proto-cuda/nvrtc/igneum-worker-gpu-telemetry.rc b/proto-cuda/nvrtc/igneum-worker-gpu-telemetry.rc new file mode 100644 index 00000000..db056147 --- /dev/null +++ b/proto-cuda/nvrtc/igneum-worker-gpu-telemetry.rc @@ -0,0 +1,35 @@ +// Windows resources for igneum-gpu-telemetry.exe: the coin icon Explorer shows and the version block under Properties > Details. +// Compiled with x86_64-w64-mingw32-windres (the icon path is relative to brand/icons, passed with -I). +// the project lead's rule (4 October 2026): every shipped exe carries the coin icon and a version block, like the Mac app and DMG. +#include + +1 ICON "igneum.ico" + +1 VERSIONINFO +FILEVERSION 0,3,0,0 +PRODUCTVERSION 0,3,0,0 +FILEFLAGSMASK 0x3fL +FILEFLAGS 0x0L +FILEOS VOS_NT_WINDOWS32 +FILETYPE VFT_APP +FILESUBTYPE VFT2_UNKNOWN +BEGIN + BLOCK "StringFileInfo" + BEGIN + BLOCK "040904B0" + BEGIN + VALUE "CompanyName", "Igneum" + VALUE "FileDescription", "Igneum Miner GPU telemetry (AMD power, heat, fans, clocks)" + VALUE "FileVersion", "0.3.0" + VALUE "InternalName", "igneum-gpu-telemetry" + VALUE "LegalCopyright", "Igneum contributors" + VALUE "OriginalFilename", "igneum-gpu-telemetry.exe" + VALUE "ProductName", "Igneum Miner" + VALUE "ProductVersion", "0.3.0" + END + END + BLOCK "VarFileInfo" + BEGIN + VALUE "Translation", 0x409, 1200 + END +END diff --git a/proto-opencl/README.md b/proto-opencl/README.md index fc6a3783..e6e040e2 100644 --- a/proto-opencl/README.md +++ b/proto-opencl/README.md @@ -39,6 +39,9 @@ proto-opencl/ host.c C99 host: device list, runtime kernel build, cache + dataset fill, self-tests, vectors, bench, sweep, --serve, --pack cl_dynamic.h Windows one-click build: OpenCL.dll loaded at run time (IGNEUM_CL_DYNAMIC) test-generic.sh the --pack mode checked here through Apple OpenCL (needs proto-cuda/nvrtc/emu/test.sh's packs) + test_host.c device-free unit tests of host.c's rules (the duplicate-platform fold); run with test-host.sh + gpu-telemetry.c igneum-gpu-telemetry: AMD power, temperature, fan, clocks and busy per card (ADLX on Windows, amdgpu sysfs on Linux), + one line per card per sample; the app's AMD card row reads it (engine.rs amd_telemetry_line) build.sh macOS (-framework OpenCL, or the Khronos ICD loader) and Linux (-lOpenCL) build.bat Windows (MSVC cl.exe + OpenCL.lib) WAVEFRONT.md wave32 vs wave64 on AMD, and why the kernel cannot tell the difference diff --git a/proto-opencl/gpu-telemetry.c b/proto-opencl/gpu-telemetry.c new file mode 100644 index 00000000..86a28be9 --- /dev/null +++ b/proto-opencl/gpu-telemetry.c @@ -0,0 +1,274 @@ +// igneum-gpu-telemetry: power, temperature, fan and clocks of every AMD GPU, one line per card per sample. +// 5 October 2026, after the project lead watched a 9070 XT at 90% usage with its fans barely turning and the app could not say +// what it drew (the app's draw, temperature and MH per watt line came from nvidia-smi only). +// +// igneum-gpu-telemetry [-l SECONDS] one sample (default), or one every SECONDS until stdin closes or SIGTERM +// +// Windows: ADLX (the AMD Device Library eXtra, amdadlx64.dll, shipped with Adrenalin; vendor/adlx is the SDK clone, +// MIT) for the metrics, keyed by the card's PCI bus from SetupAPI (the display class, matched by the same name ADLX +// reports). Without ADLX (no AMD driver, an old one, or the DLL missing) only the utilisation is read, from the +// GPU Engine performance counters through PDH, keyed by the adapter LUID that Windows uses there. +// Linux: the amdgpu sysfs (/sys/class/drm/card*/device: hwmon power1_average, temp1_input, fan1_input, pwm1, +// pp_dpm_mclk, gpu_busy_percent), keyed by the PCI address of the device link. +// +// Line format (space separated, every field present, a value the source cannot give prints as -): +// amd bus kind integrated|discrete name "" watts temp_c fan_rpm +// fan_pct <%> mclk_mhz gclk_mhz util_pct <%> source adlx|sysfs|perfcounter +// then one `end ` line per sample. The app (engine.rs amd_telemetry_line) parses it; parsers are unit-tested +// against lines captured on PC 1. +#define _CRT_SECURE_NO_WARNINGS +#include +#include +#include +#include + +static volatile int gStop = 0; +static void onSignal(int s) { (void)s; gStop = 1; } + +typedef struct { + char bus[64]; + char kind[16]; + char name[128]; + double watts, tempC, fanRpm, fanPct, mclk, gclk, util; /* -1 = not available */ + const char* source; +} Sample; + +static void sampleInit(Sample* s) { memset(s, 0, sizeof(*s)); strcpy(s->bus, "-"); strcpy(s->kind, "-"); strcpy(s->name, "-"); s->watts = s->tempC = s->fanRpm = s->fanPct = s->mclk = s->gclk = s->util = -1.0; s->source = "-"; } +static void printNum(double v, const char* fmt) { if (v < 0) printf(" -"); else printf(fmt, v); } +static void printSample(int ordinal, const Sample* s) { + printf("amd %d bus %s kind %s name \"%s\" watts", ordinal, s->bus, s->kind, s->name); + printNum(s->watts, " %.1f"); printf(" temp_c"); printNum(s->tempC, " %.1f"); printf(" fan_rpm"); printNum(s->fanRpm, " %.0f"); + printf(" fan_pct"); printNum(s->fanPct, " %.0f"); printf(" mclk_mhz"); printNum(s->mclk, " %.0f"); printf(" gclk_mhz"); printNum(s->gclk, " %.0f"); + printf(" util_pct"); printNum(s->util, " %.0f"); printf(" source %s\n", s->source); +} + +#ifdef _WIN32 +#define WIN32_LEAN_AND_MEAN +#include +#include +#include +#include "../vendor/adlx/SDK/ADLXHelper/Windows/C/ADLXHelper.h" +#include "../vendor/adlx/SDK/Include/IPerformanceMonitoring.h" + +/* The SDK declares these three and leaves them to the platform file of each sample. */ +adlx_handle ADLX_CDECL_CALL adlx_load_library(const TCHAR* filename) { return (adlx_handle)LoadLibrary(filename); } +int ADLX_CDECL_CALL adlx_free_library(adlx_handle module) { return FreeLibrary((HMODULE)module) ? 1 : 0; } +void* ADLX_CDECL_CALL adlx_get_proc_address(adlx_handle module, const char* procName) { return (void*)GetProcAddress((HMODULE)module, procName); } + +static double nowMs(void) { LARGE_INTEGER f, c; QueryPerformanceFrequency(&f); QueryPerformanceCounter(&c); return (double)c.QuadPart * 1000.0 / (double)f.QuadPart; } + +/* The PCI bus of every display-class device, by its name (SetupAPI; the names are the ones ADLX reports). */ +typedef struct { char name[128]; int bus; } BusEntry; +static int listBuses(BusEntry* out, int cap) { + static const GUID DISPLAY = { 0x4d36e968, 0xe325, 0x11ce, { 0xbf, 0xc1, 0x08, 0x00, 0x2b, 0xe1, 0x03, 0x18 } }; + HDEVINFO set = SetupDiGetClassDevsA(&DISPLAY, NULL, NULL, DIGCF_PRESENT); + SP_DEVINFO_DATA d; + DWORD i; + int n = 0; + if (set == INVALID_HANDLE_VALUE) return 0; + d.cbSize = sizeof(d); + for (i = 0; SetupDiEnumDeviceInfo(set, i, &d) && n < cap; ++i) { + char name[128] = { 0 }; + DWORD bus = 0, type = 0, got = 0; + if (!SetupDiGetDeviceRegistryPropertyA(set, &d, SPDRP_DEVICEDESC, &type, (BYTE*)name, sizeof(name) - 1, &got)) continue; + if (!SetupDiGetDeviceRegistryPropertyA(set, &d, SPDRP_BUSNUMBER, &type, (BYTE*)&bus, sizeof(bus), &got)) continue; + snprintf(out[n].name, sizeof(out[n].name), "%s", name); out[n].bus = (int)bus; ++n; + } + SetupDiDestroyDeviceInfoList(set); + return n; +} +static int busOf(const BusEntry* b, int n, const char* name, int* taken) { + int i; + for (i = 0; i < n; ++i) if (!taken[i] && strcmp(b[i].name, name) == 0) { taken[i] = 1; return b[i].bus; } + return -1; +} + +/* ADLX: one sample of every GPU. Returns the number of lines printed, -1 when ADLX is not usable (reason printed). */ +static IADLXSystem* gSys = NULL; +static IADLXPerformanceMonitoringServices* gPerf = NULL; +static int adlxOpen(void) { + ADLX_RESULT r = ADLXHelper_Initialize(); + if (!ADLX_SUCCEEDED(r)) { printf("info adlx: ADLXHelper_Initialize returned %d (no AMD driver with ADLX; amdadlx64.dll missing or too old)\n", (int)r); return 0; } + gSys = ADLXHelper_GetSystemServices(); + if (!gSys) { printf("info adlx: no system services\n"); return 0; } + r = gSys->pVtbl->GetPerformanceMonitoringServices(gSys, &gPerf); + if (!ADLX_SUCCEEDED(r) || !gPerf) { printf("info adlx: GetPerformanceMonitoringServices returned %d\n", (int)r); return 0; } + return 1; +} +static int adlxSample(const BusEntry* buses, int nBuses) { + IADLXGPUList* gpus = NULL; + adlx_uint it; + int ordinal = 0; + int taken[32] = { 0 }; + ADLX_RESULT r = gSys->pVtbl->GetGPUs(gSys, &gpus); + if (!ADLX_SUCCEEDED(r) || !gpus) { printf("info adlx: GetGPUs returned %d\n", (int)r); return 0; } + for (it = gpus->pVtbl->Begin(gpus); it != gpus->pVtbl->End(gpus); ++it) { + IADLXGPU* gpu = NULL; + IADLXGPUMetrics* m = NULL; + Sample s; + const char* name = NULL; + ADLX_GPU_TYPE type = GPUTYPE_UNDEFINED; + adlx_double dv = 0; adlx_int iv = 0; + if (!ADLX_SUCCEEDED(gpus->pVtbl->At_GPUList(gpus, it, &gpu)) || !gpu) continue; + sampleInit(&s); + s.source = "adlx"; + if (ADLX_SUCCEEDED(gpu->pVtbl->Name(gpu, &name)) && name) snprintf(s.name, sizeof(s.name), "%s", name); + if (ADLX_SUCCEEDED(gpu->pVtbl->Type(gpu, &type))) strcpy(s.kind, type == GPUTYPE_INTEGRATED ? "integrated" : type == GPUTYPE_DISCRETE ? "discrete" : "-"); + { int b = busOf(buses, nBuses, s.name, taken); if (b >= 0) snprintf(s.bus, sizeof(s.bus), "%d", b); } + r = gPerf->pVtbl->GetCurrentGPUMetrics(gPerf, gpu, &m); + if (ADLX_SUCCEEDED(r) && m) { + if (ADLX_SUCCEEDED(m->pVtbl->GPUPower(m, &dv))) s.watts = dv; + if (s.watts < 0 && ADLX_SUCCEEDED(m->pVtbl->GPUTotalBoardPower(m, &dv))) s.watts = dv; + if (ADLX_SUCCEEDED(m->pVtbl->GPUTemperature(m, &dv))) s.tempC = dv; + if (ADLX_SUCCEEDED(m->pVtbl->GPUFanSpeed(m, &iv))) s.fanRpm = iv; + if (ADLX_SUCCEEDED(m->pVtbl->GPUVRAMClockSpeed(m, &iv))) s.mclk = iv; + if (ADLX_SUCCEEDED(m->pVtbl->GPUClockSpeed(m, &iv))) s.gclk = iv; + if (ADLX_SUCCEEDED(m->pVtbl->GPUUsage(m, &dv))) s.util = dv; + m->pVtbl->Release(m); + } else { + printf("info adlx: GetCurrentGPUMetrics for \"%s\" returned %d\n", s.name, (int)r); + } + /* fan percent: ADLX gives rpm only here; the tuning interface has the range, the app shows rpm when pct is - */ + printSample(ordinal++, &s); + gpu->pVtbl->Release(gpu); + } + gpus->pVtbl->Release(gpus); + return ordinal; +} + +/* PDH fallback: GPU engine utilisation per adapter LUID, summed over the engines (no power, no temperature). */ +static int pdhSample(void) { + PDH_HQUERY q = NULL; + PDH_HCOUNTER c = NULL; + DWORD size = 0, count = 0, i; + PDH_FMT_COUNTERVALUE_ITEM_A* items; + int ordinal = 0; + if (PdhOpenQueryA(NULL, 0, &q) != ERROR_SUCCESS) { printf("info perfcounter: PdhOpenQuery failed\n"); return 0; } + if (PdhAddEnglishCounterA(q, "\\GPU Engine(*)\\Utilization Percentage", 0, &c) != ERROR_SUCCESS) { printf("info perfcounter: no GPU Engine counters\n"); PdhCloseQuery(q); return 0; } + PdhCollectQueryData(q); Sleep(1000); PdhCollectQueryData(q); + PdhGetFormattedCounterArrayA(c, PDH_FMT_DOUBLE, &size, &count, NULL); + items = (PDH_FMT_COUNTERVALUE_ITEM_A*)malloc(size ? size : 1); + if (PdhGetFormattedCounterArrayA(c, PDH_FMT_DOUBLE, &size, &count, items) == ERROR_SUCCESS) { + /* instance names: pid_1234_luid_0x00000000_0x0000D4E3_phys_0_eng_0_engtype_3D; sum per luid */ + char luids[16][40]; double sums[16]; int n = 0, k; + for (i = 0; i < count; ++i) { + const char* p = strstr(items[i].szName, "luid_"); + char luid[40]; + if (!p) continue; + snprintf(luid, sizeof(luid), "%.39s", p); { char* e = strstr(luid, "_phys"); if (e) *e = 0; } + for (k = 0; k < n; ++k) if (strcmp(luids[k], luid) == 0) break; + if (k == n && n < 16) { strcpy(luids[n], luid); sums[n] = 0; ++n; } + if (k < 16) sums[k] += items[i].FmtValue.doubleValue; + } + for (k = 0; k < n; ++k) { + Sample s; sampleInit(&s); s.source = "perfcounter"; + snprintf(s.bus, sizeof(s.bus), "%s", luids[k]); + s.util = sums[k] > 100.0 ? 100.0 : sums[k]; + printSample(ordinal++, &s); + } + } + free(items); + PdhCloseQuery(q); + return ordinal; +} + +int main(int argc, char** argv) { + int every = 0, i, haveAdlx; + BusEntry buses[32]; + int nBuses; + for (i = 1; i < argc; ++i) if (strcmp(argv[i], "-l") == 0 && i + 1 < argc) every = atoi(argv[++i]); + signal(SIGINT, onSignal); signal(SIGTERM, onSignal); + setvbuf(stdout, NULL, _IOLBF, 0); + nBuses = listBuses(buses, 32); + for (i = 0; i < nBuses; ++i) printf("info display device \"%s\" bus %d\n", buses[i].name, buses[i].bus); + haveAdlx = adlxOpen(); + do { + double t0 = nowMs(); + int n = haveAdlx ? adlxSample(buses, nBuses) : pdhSample(); + printf("end %.1f ms %d card(s)\n", nowMs() - t0, n); + fflush(stdout); /* a redirected stdout is fully buffered on the Windows CRT whatever setvbuf asks (PC 1 lost 60 s of samples at the kill) */ + if (every > 0) Sleep((DWORD)every * 1000); + } while (every > 0 && !gStop); + if (haveAdlx) { if (gPerf) gPerf->pVtbl->Release(gPerf); ADLXHelper_Terminate(); } + return 0; +} +#else +#include +#include +#include +static double nowMs(void) { struct timespec ts; clock_gettime(CLOCK_MONOTONIC, &ts); return ts.tv_sec * 1000.0 + ts.tv_nsec / 1e6; } +static int readText(const char* path, char* out, size_t cap) { FILE* f = fopen(path, "r"); size_t n; if (!f) return 0; n = fread(out, 1, cap - 1, f); fclose(f); out[n] = 0; return 1; } +static double readNumber(const char* path) { char b[64]; if (!readText(path, b, sizeof(b))) return -1.0; return atof(b); } +/* pp_dpm_mclk: lines "0: 96Mhz", "3: 1258Mhz *"; the starred line is the current state */ +static double dpmCurrent(const char* text) { + const char* p = text; + while (p && *p) { + const char* nl = strchr(p, '\n'); + size_t len = nl ? (size_t)(nl - p) : strlen(p); + const char* star = memchr(p, '*', len); + if (star) { const char* colon = memchr(p, ':', len); if (colon) return atof(colon + 1); } + p = nl ? nl + 1 : NULL; + } + return -1.0; +} +static int sysfsSample(const char* root) { + DIR* d = opendir(root); + struct dirent* e; + int ordinal = 0; + if (!d) { printf("info sysfs: no %s\n", root); return 0; } + while ((e = readdir(d)) != NULL) { + char dev[512], path[640], text[4096], link[512]; + ssize_t ln; + Sample s; + DIR* hw; struct dirent* he; + if (strncmp(e->d_name, "card", 4) != 0 || strchr(e->d_name + 4, '-')) continue; + snprintf(dev, sizeof(dev), "%s/%s/device", root, e->d_name); + snprintf(path, sizeof(path), "%s/vendor", dev); + if (!readText(path, text, sizeof(text)) || strtol(text, NULL, 16) != 0x1002) continue; + sampleInit(&s); + s.source = "sysfs"; + ln = readlink(dev, link, sizeof(link) - 1); + if (ln > 0) { link[ln] = 0; { const char* base = strrchr(link, '/'); snprintf(s.bus, sizeof(s.bus), "%.63s", base ? base + 1 : link); } } + snprintf(path, sizeof(path), "%s/product_name", dev); + if (readText(path, text, sizeof(text))) { text[strcspn(text, "\n")] = 0; snprintf(s.name, sizeof(s.name), "%s", text); } + else { snprintf(path, sizeof(path), "%s/device", dev); if (readText(path, text, sizeof(text))) { text[strcspn(text, "\n")] = 0; snprintf(s.name, sizeof(s.name), "amdgpu %s", text); } } + snprintf(path, sizeof(path), "%s/boot_vga", dev); + strcpy(s.kind, "discrete"); + snprintf(path, sizeof(path), "%s/hwmon", dev); + hw = opendir(path); + if (hw) { + while ((he = readdir(hw)) != NULL) { + char hp[900]; + if (strncmp(he->d_name, "hwmon", 5) != 0) continue; + snprintf(hp, sizeof(hp), "%s/%s/power1_average", path, he->d_name); s.watts = readNumber(hp); if (s.watts < 0) { snprintf(hp, sizeof(hp), "%s/%s/power1_input", path, he->d_name); s.watts = readNumber(hp); } if (s.watts >= 0) s.watts /= 1e6; + snprintf(hp, sizeof(hp), "%s/%s/temp1_input", path, he->d_name); s.tempC = readNumber(hp); if (s.tempC >= 0) s.tempC /= 1000.0; + snprintf(hp, sizeof(hp), "%s/%s/fan1_input", path, he->d_name); s.fanRpm = readNumber(hp); + { double pwm, pwmMax; snprintf(hp, sizeof(hp), "%s/%s/pwm1", path, he->d_name); pwm = readNumber(hp); snprintf(hp, sizeof(hp), "%s/%s/pwm1_max", path, he->d_name); pwmMax = readNumber(hp); if (pwm >= 0 && pwmMax > 0) s.fanPct = 100.0 * pwm / pwmMax; else if (pwm >= 0) s.fanPct = 100.0 * pwm / 255.0; } + break; + } + closedir(hw); + } + snprintf(path, sizeof(path), "%s/pp_dpm_mclk", dev); if (readText(path, text, sizeof(text))) s.mclk = dpmCurrent(text); + snprintf(path, sizeof(path), "%s/pp_dpm_sclk", dev); if (readText(path, text, sizeof(text))) s.gclk = dpmCurrent(text); + snprintf(path, sizeof(path), "%s/gpu_busy_percent", dev); s.util = readNumber(path); + printSample(ordinal++, &s); + } + closedir(d); + return ordinal; +} +int main(int argc, char** argv) { + int every = 0, i; + const char* root = getenv("IGNEUM_DRM_ROOT") ? getenv("IGNEUM_DRM_ROOT") : "/sys/class/drm"; /* a fixture tree for tests */ + for (i = 1; i < argc; ++i) if (strcmp(argv[i], "-l") == 0 && i + 1 < argc) every = atoi(argv[++i]); + signal(SIGINT, onSignal); signal(SIGTERM, onSignal); + setvbuf(stdout, NULL, _IOLBF, 0); + do { + double t0 = nowMs(); + int n = sysfsSample(root); + printf("end %.1f ms %d card(s)\n", nowMs() - t0, n); + fflush(stdout); /* a redirected stdout is fully buffered on the Windows CRT whatever setvbuf asks (PC 1 lost 60 s of samples at the kill) */ + if (every > 0) sleep((unsigned)every); + } while (every > 0 && !gStop); + return 0; +} +#endif