AMD telemetry: igneum-gpu-telemetry (ADLX on Windows, amdgpu sysfs on Linux, PDH utilisation fallback) feeds the card row's draw, temperature, fan, memory clock and MH/W; measured on PC 1: 9070 XT 198.9 W, 64 C, 657 rpm, 17.73 MH/s = 0.089 MH/W beside the 5090 at 307.6 W, 122.30 MH/s = 0.398 MH/W
the project lead watched the 9070 XT at 90% usage with its fans barely turning and the app could not say what it drew: the draw, temperature and MH per watt line came from nvidia-smi only, and the earlier per-watt figure used the board rating. proto-opencl/gpu-telemetry.c prints one line per AMD card per sample (bus from SetupAPI by the display device's name, kind, name, watts, temp_c, fan_rpm, fan_pct, mclk_mhz, gclk_mhz, util_pct, source), built by build-windows.sh against vendor/adlx (the SDK clone), shipped by make-payload.sh and push-inputs.sh. The engine runs it with -l 5 beside nvidia-smi (Source::AmdTelemetry, tick_amd_telemetry), parse_amd_telemetry fills power_w, temp_gpu, fan_pct, fan_rpm, mclk_mhz, util_pct and telemetry_at on the AMD card matched by kind and ordinal, so eff_mhw and the dashboard's existing line show it; app.js shows fan and memory clock when present. Tests: three on the parser with lines captured on PC 1 and the Mac fixture; the sysfs path ran on a fixture tree. Measured over 20:27:45 to 20:29:41 UTC with both cards mining (docs/bench-log.md, under the 9070 XT ceiling table): 9070 XT 198.9 W (193 to 212), 64 C, 657 rpm, 2,505 MHz memory, 3,290 MHz shader, 100% busy, 17.73 MH/s = 0.089 MH/W; RTX 5090 307.6 W, 69 C, 44% fan, 122.30 MH/s = 0.398 MH/W. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
This commit is contained in:
parent
856ecd47ba
commit
35e3d260a4
12 changed files with 509 additions and 6 deletions
|
|
@ -15,6 +15,8 @@ pub struct Bins {
|
|||
pub metal: Option<std::path::PathBuf>,
|
||||
pub cuda: Option<std::path::PathBuf>,
|
||||
pub opencl: Option<std::path::PathBuf>,
|
||||
/// igneum-gpu-telemetry: AMD power, heat, fans and clocks (proto-opencl/gpu-telemetry.c), 5 October 2026
|
||||
pub telemetry: Option<std::path::PathBuf>,
|
||||
pub dir: std::path::PathBuf,
|
||||
}
|
||||
|
||||
|
|
@ -396,7 +398,7 @@ pub fn find_bins() -> Result<Bins, String> {
|
|||
// the prebuilt CUDA worker needs NVIDIA's nvrtc64_*_0.dll next to it (as igneum-common.ps1 checks)
|
||||
let nvrtc = std::fs::read_dir(dir).ok().map(|rd| rd.flatten().any(|e| { let n = e.file_name().to_string_lossy().to_ascii_lowercase(); n.starts_with("nvrtc64_") && n.ends_with("_0.dll") })).unwrap_or(false);
|
||||
let cuda = opt("igneum-worker-cuda").filter(|_| nvrtc || cfg!(not(windows)));
|
||||
return Ok(Bins { node, miner, metal: opt("igneum-bench"), cuda, opencl: opt("igneum-worker-opencl"), dir: dir.clone() });
|
||||
return Ok(Bins { node, miner, metal: opt("igneum-bench"), cuda, opencl: opt("igneum-worker-opencl"), telemetry: opt("igneum-gpu-telemetry"), dir: dir.clone() });
|
||||
}
|
||||
}
|
||||
Err(format!("igneumd and igneum-miner were not found next to the app (looked in {})", candidates.iter().map(|c| c.display().to_string()).collect::<Vec<_>>().join(", ")))
|
||||
|
|
|
|||
|
|
@ -445,6 +445,8 @@ pub struct Engine {
|
|||
last_accepted: Option<Instant>,
|
||||
telemetry: Option<Proc>,
|
||||
telemetry_retry_at: Instant,
|
||||
amd_telemetry: Option<Proc>,
|
||||
amd_telemetry_retry_at: Instant,
|
||||
power_busy: bool,
|
||||
power_restore_pending: bool,
|
||||
/// an elevated step handed to the window host: (command line, what, requested watts per device, since)
|
||||
|
|
@ -539,6 +541,8 @@ impl Engine {
|
|||
last_accepted: None,
|
||||
telemetry: None,
|
||||
telemetry_retry_at: now,
|
||||
amd_telemetry: None,
|
||||
amd_telemetry_retry_at: now,
|
||||
power_busy: false,
|
||||
power_restore_pending: false,
|
||||
power_via_host: None,
|
||||
|
|
@ -1496,6 +1500,7 @@ impl Engine {
|
|||
/// nvidia-smi -l 5: power draw, GPU and memory temperature, the limit in force, every 5 s, as a child whose
|
||||
/// lines come through the same channel as the miners'.
|
||||
fn tick_telemetry(&mut self, now: Instant) {
|
||||
self.tick_amd_telemetry(now);
|
||||
if cfg!(target_os = "macos") || self.st().mining.cards.iter().all(|c| c.vendor != "nvidia") {
|
||||
return;
|
||||
}
|
||||
|
|
@ -1533,6 +1538,63 @@ impl Engine {
|
|||
}
|
||||
}
|
||||
|
||||
/// AMD cards: igneum-gpu-telemetry -l 5 (ADLX on Windows, the amdgpu sysfs on Linux; proto-opencl/gpu-telemetry.c),
|
||||
/// restarted 30 s after it ends, 300 s after it could not start. Nothing on macOS or without an AMD card.
|
||||
fn tick_amd_telemetry(&mut self, now: Instant) {
|
||||
if cfg!(target_os = "macos") || self.st().mining.cards.iter().all(|c| c.vendor != "amd") {
|
||||
return;
|
||||
}
|
||||
let Some(exe) = self.bins.telemetry.clone() else { return };
|
||||
if let Some(t) = self.amd_telemetry.as_mut() {
|
||||
if !t.alive() {
|
||||
self.amd_telemetry = None;
|
||||
self.amd_telemetry_retry_at = now + Duration::from_secs(30);
|
||||
}
|
||||
}
|
||||
if self.amd_telemetry.is_none() && now >= self.amd_telemetry_retry_at {
|
||||
let args: Vec<String> = vec!["-l".into(), "5".into()];
|
||||
let log = self.shared.runtime.log_dir.join(format!("gpu-amd-{}.log", self.stamp));
|
||||
match procs::spawn(Source::AmdTelemetry, &exe, &args, None, &log, &self.lines_tx, &[]) {
|
||||
Ok(p) => self.amd_telemetry = Some(p),
|
||||
Err(_) => self.amd_telemetry_retry_at = now + Duration::from_secs(300),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// One `amd ...` line of igneum-gpu-telemetry to the matching card. The helper lists cards in ADLX (or sysfs)
|
||||
/// order with their kind; the app's AMD cards come from the OpenCL worker's list, which has no bus. They are
|
||||
/// matched by kind (integrated, discrete) and ordinal within the kind, which is exact for one card of each kind
|
||||
/// (PC 1: the gfx1036 and the 9070 XT) and approximate for two discrete AMD cards of one model.
|
||||
fn amd_telemetry_line(&mut self, text: &str) {
|
||||
let Some(s) = parse_amd_telemetry(text) else { return };
|
||||
let mut st = self.st();
|
||||
let mut nth = 0usize;
|
||||
let mut target: Option<usize> = None;
|
||||
for c in st.mining.cards.iter() {
|
||||
if c.vendor != "amd" || c.kind != s.kind {
|
||||
continue;
|
||||
}
|
||||
if nth == s.ordinal_in_kind {
|
||||
target = Some(c.index);
|
||||
break;
|
||||
}
|
||||
nth += 1;
|
||||
}
|
||||
let Some(idx) = target else { return };
|
||||
let Some(c) = st.mining.cards.iter_mut().find(|c| c.index == idx) else { return };
|
||||
if s.watts > 0.0 {
|
||||
c.power_w = s.watts;
|
||||
}
|
||||
if s.temp_c > 0.0 {
|
||||
c.temp_gpu = s.temp_c;
|
||||
}
|
||||
c.fan_pct = s.fan_pct.max(0.0);
|
||||
c.fan_rpm = s.fan_rpm.max(0.0);
|
||||
c.mclk_mhz = s.mclk_mhz.max(0.0);
|
||||
c.util_pct = s.util_pct.max(0.0);
|
||||
c.telemetry_at = crate::platform::unix_now_f();
|
||||
}
|
||||
|
||||
/// "index, draw, gpu temp, mem temp, limit" every 5 s.
|
||||
fn telemetry_line(&mut self, text: &str) {
|
||||
let p: Vec<&str> = text.split(',').map(|s| s.trim()).collect();
|
||||
|
|
@ -2497,6 +2559,11 @@ impl Engine {
|
|||
self.telemetry_line(&l.text);
|
||||
}
|
||||
}
|
||||
Source::AmdTelemetry => {
|
||||
if !l.stderr {
|
||||
self.amd_telemetry_line(&l.text);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
|
@ -2965,6 +3032,9 @@ impl Engine {
|
|||
if let Some(mut t) = self.telemetry.take() {
|
||||
t.stop(2);
|
||||
}
|
||||
if let Some(mut t) = self.amd_telemetry.take() {
|
||||
t.stop(2);
|
||||
}
|
||||
if self.shared.runtime.sweep_only {
|
||||
// the measurement leaves the chosen cap in force for the app that takes the card back
|
||||
self.shared.log("--sweep: the chosen caps stay in force (not restored)");
|
||||
|
|
@ -3282,3 +3352,96 @@ fn build_worker_from_source(shared: &Arc<Shared>, bins: &Bins, vendor: &str) ->
|
|||
if p.exists() { Ok(p) } else { Err(format!("{exe} was not written; see the log")) }
|
||||
}
|
||||
}
|
||||
|
||||
/// One `amd` line of igneum-gpu-telemetry (proto-opencl/gpu-telemetry.c): the fields the card row carries.
|
||||
/// `ordinal_in_kind` is the line's rank among the lines of the same kind in one sample: the helper's `amd N` is
|
||||
/// the rank over all kinds, so the parser tracks the kinds it has seen through `AmdTelemetrySample::new`
|
||||
/// per sample; a single line parses with the rank 0 when it is the first of its kind in its `amd N` ordering.
|
||||
#[derive(Debug, Clone, PartialEq)]
|
||||
pub struct AmdTelemetry {
|
||||
pub ordinal: usize,
|
||||
pub ordinal_in_kind: usize,
|
||||
pub bus: String,
|
||||
pub kind: String,
|
||||
pub name: String,
|
||||
pub watts: f64,
|
||||
pub temp_c: f64,
|
||||
pub fan_rpm: f64,
|
||||
pub fan_pct: f64,
|
||||
pub mclk_mhz: f64,
|
||||
pub gclk_mhz: f64,
|
||||
pub util_pct: f64,
|
||||
pub source: String,
|
||||
}
|
||||
|
||||
/// Parses `amd <n> bus <b> kind <k> name "<name>" watts <w> temp_c <t> fan_rpm <r> fan_pct <p> mclk_mhz <m>
|
||||
/// gclk_mhz <g> util_pct <u> source <s>`; a `-` value reads as -1.0. Anything else (info, end) gives None.
|
||||
/// With one card per kind (the common case) `ordinal_in_kind` is 0 for the discrete card and 0 for the integrated
|
||||
/// one whatever their `amd N`; with several discrete cards the helper's order within the kind is kept: the rank is
|
||||
/// the number of earlier lines of the same kind, which the helper encodes by listing kinds contiguously (ADLX lists
|
||||
/// GPUs in a fixed order, so the rank of a card is stable across samples).
|
||||
pub fn parse_amd_telemetry(line: &str) -> Option<AmdTelemetry> {
|
||||
let line = line.trim();
|
||||
if !line.starts_with("amd ") {
|
||||
return None;
|
||||
}
|
||||
let (head, rest) = line.split_once(" name \"")?;
|
||||
let (name, tail) = rest.split_once('"')?;
|
||||
let hp: Vec<&str> = head.split_whitespace().collect();
|
||||
if hp.len() < 6 || hp[2] != "bus" || hp[4] != "kind" {
|
||||
return None;
|
||||
}
|
||||
let ordinal: usize = hp[1].parse().ok()?;
|
||||
let tp: Vec<&str> = tail.split_whitespace().collect();
|
||||
let num = |key: &str| -> Option<f64> {
|
||||
let i = tp.iter().position(|p| *p == key)?;
|
||||
let v = tp.get(i + 1)?;
|
||||
if *v == "-" { Some(-1.0) } else { v.parse::<f64>().ok() }
|
||||
};
|
||||
let watts = num("watts")?;
|
||||
let temp_c = num("temp_c")?;
|
||||
let fan_rpm = num("fan_rpm")?;
|
||||
let fan_pct = num("fan_pct")?;
|
||||
let mclk_mhz = num("mclk_mhz")?;
|
||||
let gclk_mhz = num("gclk_mhz")?;
|
||||
let util_pct = num("util_pct")?;
|
||||
let source = tp.iter().position(|p| *p == "source").and_then(|i| tp.get(i + 1)).map(|s| s.to_string()).unwrap_or_default();
|
||||
let kind = hp[5].to_string();
|
||||
// the rank within the kind: the helper lists one integrated card at most and it comes first when present
|
||||
// (ADLX order on every PC seen so far), so a discrete card's rank is its ordinal minus the integrated ones before it
|
||||
let ordinal_in_kind = if kind == "discrete" && ordinal > 0 { ordinal - 1 } else if kind == "discrete" { 0 } else { 0 };
|
||||
Some(AmdTelemetry { ordinal, ordinal_in_kind, bus: hp[3].to_string(), kind, name: name.to_string(), watts, temp_c, fan_rpm, fan_pct, mclk_mhz, gclk_mhz, util_pct, source })
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod amd_telemetry_tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn a_sysfs_line_from_the_fixture_parses() {
|
||||
// proto-opencl/gpu-telemetry.c on the Mac against a fixture tree, 5 October 2026
|
||||
let l = "amd 0 bus 0000:0c:00.0 kind discrete name \"AMD Radeon RX 9070 XT\" watts 287.0 temp_c 61.0 fan_rpm 1180 fan_pct 30 mclk_mhz 1258 gclk_mhz 2450 util_pct 90 source sysfs";
|
||||
let s = parse_amd_telemetry(l).unwrap();
|
||||
assert_eq!((s.ordinal, s.ordinal_in_kind, s.bus.as_str(), s.kind.as_str(), s.name.as_str()), (0, 0, "0000:0c:00.0", "discrete", "AMD Radeon RX 9070 XT"));
|
||||
assert_eq!((s.watts, s.temp_c, s.fan_rpm, s.fan_pct, s.mclk_mhz, s.gclk_mhz, s.util_pct), (287.0, 61.0, 1180.0, 30.0, 1258.0, 2450.0, 90.0));
|
||||
assert_eq!(s.source, "sysfs");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_dash_reads_as_unknown_and_other_lines_give_none() {
|
||||
let l = "amd 1 bus 98 kind discrete name \"AMD Radeon RX 9070 XT\" watts 250.3 temp_c 58.0 fan_rpm 900 fan_pct - mclk_mhz 1258 gclk_mhz 2460 util_pct 97.5 source adlx";
|
||||
let s = parse_amd_telemetry(l).unwrap();
|
||||
assert_eq!(s.fan_pct, -1.0);
|
||||
assert_eq!(s.ordinal_in_kind, 0, "the second line overall but the first discrete card after the integrated one");
|
||||
assert!(parse_amd_telemetry("end 3.2 ms 2 card(s)").is_none());
|
||||
assert!(parse_amd_telemetry("info adlx: ADLXHelper_Initialize returned 1").is_none());
|
||||
assert!(parse_amd_telemetry("amd 0 bus - kind - name \"x\" watts").is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_perfcounter_fallback_line_parses() {
|
||||
let l = "amd 0 bus luid_0x00000000_0x0000D4E3 kind - name \"-\" watts - temp_c - fan_rpm - fan_pct - mclk_mhz - gclk_mhz - util_pct 100 source perfcounter";
|
||||
let s = parse_amd_telemetry(l).unwrap();
|
||||
assert_eq!((s.watts, s.util_pct, s.source.as_str()), (-1.0, 100.0, "perfcounter"));
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -13,6 +13,7 @@ pub enum Source {
|
|||
Watch,
|
||||
Miner(usize), // card index
|
||||
Telemetry, // nvidia-smi -l 5
|
||||
AmdTelemetry, // igneum-gpu-telemetry -l 5 (ADLX or sysfs), 5 October 2026
|
||||
}
|
||||
|
||||
impl Source {
|
||||
|
|
@ -22,6 +23,7 @@ impl Source {
|
|||
Source::Watch => "watch".into(),
|
||||
Source::Miner(i) => format!("miner{}", i + 1),
|
||||
Source::Telemetry => "gpu".into(),
|
||||
Source::AmdTelemetry => "gpu-amd".into(),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -74,6 +74,11 @@ pub struct CardState {
|
|||
pub temp_gpu: f64,
|
||||
pub temp_mem: f64,
|
||||
pub telemetry_at: f64,
|
||||
// AMD through igneum-gpu-telemetry (ADLX on Windows, amdgpu sysfs on Linux), 5 October 2026; 0 = unknown
|
||||
pub fan_pct: f64,
|
||||
pub fan_rpm: f64,
|
||||
pub mclk_mhz: f64,
|
||||
pub util_pct: f64,
|
||||
// hash per watt (src/sweep.rs)
|
||||
pub eff_mhw: f64, // live: hash_now over power_w, MH per watt; 0 = unknown
|
||||
pub sweep_supported: bool, // NVIDIA with readable limits; the note says why not otherwise
|
||||
|
|
|
|||
|
|
@ -745,11 +745,13 @@ if (typeof document !== 'undefined') (function () {
|
|||
}
|
||||
// power draw, the cap and the temperatures (NVIDIA); memory over 90 C amber, over 95 C red
|
||||
function telemetryHtml(cd) {
|
||||
if (cd.vendor !== 'nvidia' || !(cd.telemetry_at > 0 || cd.power_default_w > 0)) return '';
|
||||
if (!(cd.telemetry_at > 0 || cd.power_default_w > 0)) return ''; // NVIDIA through nvidia-smi, AMD through igneum-gpu-telemetry (5 October 2026)
|
||||
var memCls = cd.temp_mem > 95 ? 'hot' : cd.temp_mem > 90 ? 'warm' : '';
|
||||
var h = '<div class="m tele"><span>draw <b>' + (cd.power_w ? Math.round(cd.power_w) + ' W' : 'n/a') + '</b>' + (cd.power_limit_w ? ' <span class="dim">/ cap ' + Math.round(cd.power_limit_w) + ' W</span>' : '') + '</span>' +
|
||||
'<span>GPU <b>' + (cd.temp_gpu ? Math.round(cd.temp_gpu) + ' °C' : 'n/a') + '</b></span>' +
|
||||
'<span class="' + memCls + '">memory <b>' + (cd.temp_mem ? Math.round(cd.temp_mem) + ' °C' : 'n/a') + '</b></span>' +
|
||||
(cd.vendor === 'nvidia' ? '<span class="' + memCls + '">memory <b>' + (cd.temp_mem ? Math.round(cd.temp_mem) + ' °C' : 'n/a') + '</b></span>' : '') +
|
||||
(cd.fan_pct > 0 ? '<span>fan <b>' + Math.round(cd.fan_pct) + ' %</b></span>' : cd.fan_rpm > 0 ? '<span>fan <b>' + Math.round(cd.fan_rpm) + ' rpm</b></span>' : cd.vendor === 'amd' ? '<span>fan <b>n/a</b></span>' : '') +
|
||||
(cd.mclk_mhz > 0 ? '<span>memory clock <b>' + Math.round(cd.mclk_mhz) + ' MHz</b></span>' : '') +
|
||||
'<span>eff <b>' + (cd.eff_mhw ? cd.eff_mhw.toFixed(3) + ' MH/W' : 'n/a') + '</b></span></div>';
|
||||
if (memCls) h += '<div class="msg ' + memCls + '">memory ' + Math.round(cd.temp_mem) + ' °C: card throttling or at risk</div>';
|
||||
if (cd.power_default_w > 0) {
|
||||
|
|
|
|||
|
|
@ -1568,6 +1568,16 @@ Reading: at the dataset size the card delivers about 2.5 G random 4-byte reads p
|
|||
|
||||
Reading: on all three cards the hash runs within a few percent of 1/128 of the card's dependent random-read ceiling, which is what a 128-load program should do; the probe is a good model of the hash. The 5090 does 6.6x the random reads of the 9070 XT for 2.8x the rated bandwidth (1,792 against 640 GB/s, vendor figures): the rest is access granularity and DRAM behaviour on random 4-byte reads, which the kernel cannot change.
|
||||
|
||||
**Power, heat, fans and clocks, measured** (branch `opencl-rdna4-telemetry`; the project lead watched the 9070 XT at 90% usage with its fans barely turning and the app had no AMD reading, the MH/W line came from nvidia-smi only; a new helper `proto-opencl/gpu-telemetry.c` reads ADLX on Windows and the amdgpu sysfs on Linux. Job `tele-measure-1`, 20:27:45 to 20:29:41 UTC, both cards mining in the app, nothing touched: `igneum-gpu-telemetry -l 5` (sha256 `703cf69c…a9c69b`) and `nvidia-smi --query-gpu=index,name,power.draw,temperature.gpu,fan.speed,clocks.mem,clocks.gr,utilization.gpu -l 5` side by side, the app's `hash_now` every 5 s; `node tools/jobs.mjs tele-measure-1`):
|
||||
|
||||
| Card | Samples | Watts (mean, min to max) | Temperature | Fan | Memory clock | Shader clock | Busy | Hash (mean of 24) | MH/W, measured |
|
||||
|---|---|---|---|---|---|---|---|---|---|
|
||||
| RX 9070 XT, bus 98, ADLX `GPUPower` | 12 (the helper's buffered tail was lost at the kill; fixed, `fflush` per sample) | 198.9 (193 to 212) | 64 C | 657 rpm (ADLX gives rpm; no percent) | 2,505 MHz | 3,290 MHz | 100% | 17.73 MH/s | 0.089 |
|
||||
| RTX 5090, nvidia-smi, 450 W cap | 24 | 307.6 (306.3 to 308.7) | 69 C | 44% | 13,801 MHz | 2,850 MHz | 94% | 122.30 MH/s | 0.398 |
|
||||
| gfx1036 (integrated, idle) | 12 | 42.7 (32 to 56; the package, not the GPU alone) | 62 C | none | 2,800 MHz | 600 MHz | 0% | off | |
|
||||
|
||||
Reading: the 9070 XT draws 199 W of its 304 W board rating (vendor figure) at 100% busy with the shader clock at its top, so the die is waiting on memory, which is the ceiling finding again; the fans at 657 rpm and 64 C are the card's own curve at that load, not a fault. Per watt the 5090 is 4.5x the 9070 XT on this program class (0.398 against 0.089 MH/W). The earlier per-watt claim from the board rating (304 W) would have read 0.058 MH/W; the measured number is 1.5x that.
|
||||
|
||||
**Is it the eGPU link?** No. 2.42 G loads/s x 64 B lines = 155 GB/s of DRAM traffic, forty times what a USB4 PCIe tunnel carries (about 4 GB/s, approximate); the 1 GiB buffer sits in the card's own memory (the 4 and 64 MiB cases show the card's caches at work above it, and a buffer in host memory would run below 0.1 G/s). A PCIe slot would move the per-job read-back (16 MiB per 2^21-nonce job on the old path, now gone) and nothing else; the random-read ceiling is the card's. What a PCIe slot would give: the same 18 MH/s.
|
||||
|
||||
**What changed on `opencl-rdna4`** (`proto-opencl/host.c`, `app/igneum-app/src/detect.rs`):
|
||||
|
|
|
|||
|
|
@ -72,6 +72,7 @@ else
|
|||
ls "$STAGE"/nvrtc64_*_0.dll >/dev/null 2>&1 || echo "warning: igneum-worker-cuda.exe without nvrtc64_*_0.dll (run $NVRTC_DIR/fetch-redist.sh); the engine will not use it"
|
||||
fi
|
||||
if [ -f "$ROOT/proto-opencl/igneum-worker-opencl.exe" ]; then cp "$ROOT/proto-opencl/igneum-worker-opencl.exe" "$STAGE/"; found_workers=1; fi
|
||||
if [ -f "$ROOT/proto-opencl/igneum-gpu-telemetry.exe" ]; then cp "$ROOT/proto-opencl/igneum-gpu-telemetry.exe" "$STAGE/"; fi # AMD power, heat, fans, clocks (5 October 2026)
|
||||
fi
|
||||
if [ "$found_workers" = 1 ]; then echo "workers: $(cd "$STAGE" && ls igneum-worker-*.exe nvrtc*.dll 2>/dev/null | tr '\n' ' ')"
|
||||
else echo "note: no prebuilt igneum-worker-cuda.exe / igneum-worker-opencl.exe found; the engine builds the CUDA worker from proto-cuda\\ on the PC (CUDA Toolkit and MSVC needed)"; fi
|
||||
|
|
|
|||
|
|
@ -28,6 +28,7 @@ REL="${IGNEUM_WIN_RELEASE:-$ROOT/vendor/igneum-node/target-integration/x86_64-pc
|
|||
MINGW=/opt/homebrew/opt/mingw-w64/toolchain-x86_64/x86_64-w64-mingw32
|
||||
NVRTC_DIR="$ROOT/proto-cuda/nvrtc"
|
||||
CL_WORKER="$ROOT/proto-opencl/igneum-worker-opencl.exe"
|
||||
TELEMETRY="$ROOT/proto-opencl/igneum-gpu-telemetry.exe" # AMD power, heat, fans, clocks (5 October 2026)
|
||||
TOKEN_FILE="$HOME/.config/igneum/dl-token"
|
||||
DLSITE="${IGNEUM_DLSITE:-}"
|
||||
[ -n "$DLSITE" ] || { [ -f "$HOME/.config/igneum/dlsite-dir" ] && DLSITE="$(tr -d '[:space:]' < "$HOME/.config/igneum/dlsite-dir")"; } || true
|
||||
|
|
@ -59,6 +60,7 @@ if [ -f "$NVRTC_DIR/igneum-worker-cuda.exe" ]; then
|
|||
ls "$STAGE"/nvrtc64_*_0.dll >/dev/null 2>&1 || echo "warning: igneum-worker-cuda.exe without nvrtc64_*_0.dll (run $NVRTC_DIR/fetch-redist.sh)"
|
||||
else echo "warning: no $NVRTC_DIR/igneum-worker-cuda.exe (run $NVRTC_DIR/build-windows.sh); the app will build the CUDA worker on the PC"; fi
|
||||
[ -f "$CL_WORKER" ] && cp "$CL_WORKER" "$STAGE/" || echo "warning: no $CL_WORKER"
|
||||
[ -f "$TELEMETRY" ] && cp "$TELEMETRY" "$STAGE/" || echo "warning: no $TELEMETRY (AMD cards show no draw or temperature)"
|
||||
|
||||
# the signer, built from the app crate (it includes src/manifest.rs and src/inputs.rs, so it signs what the runner verifies)
|
||||
KEY="$HOME/.config/igneum/ota-signing-key"
|
||||
|
|
|
|||
|
|
@ -25,7 +25,7 @@ VERIFY="$ROOT/packaging/windows/resources/verify-exe.py"
|
|||
# The coin icon and the version blocks, as COFF objects the linker takes like any other input
|
||||
[ -f "$ICONS/igneum.ico" ] || { echo "== no $ICONS/igneum.ico, making the icons"; python3 "$ICONS/make-icons.py"; }
|
||||
RES="$(mktemp -d)"
|
||||
for w in cuda opencl; do
|
||||
for w in cuda opencl gpu-telemetry; do
|
||||
"$WINDRES" -I "$ICONS" -i "$HERE/igneum-worker-$w.rc" -O coff -o "$RES/igneum-worker-$w.res.o"
|
||||
done
|
||||
|
||||
|
|
@ -37,11 +37,16 @@ echo "== igneum-worker-opencl.exe"
|
|||
-I "$RED/include" -I "$PLACEHOLDER" -DIGNEUM_KERNEL_PATH='"kernel_bound.cl"' \
|
||||
-o "$ROOT/proto-opencl/igneum-worker-opencl.exe" "$ROOT/proto-opencl/host.c" "$RES/igneum-worker-opencl.res.o"
|
||||
"$STRIP" "$ROOT/proto-opencl/igneum-worker-opencl.exe"
|
||||
echo "== igneum-gpu-telemetry.exe (ADLX, SetupAPI, PDH; vendor/adlx is the SDK clone)"
|
||||
[ -f "$ROOT/vendor/adlx/SDK/Include/ADLX.h" ] || { echo "no vendor/adlx: git clone --depth 1 https://github.com/GPUOpen-LibrariesAndSDKs/ADLX.git $ROOT/vendor/adlx" >&2; exit 1; }
|
||||
"$CC" -std=gnu99 -O2 -Wall -Wno-unused-parameter -Wno-unused-function -static -I "$ROOT/vendor/adlx/SDK/Include" \
|
||||
-o "$ROOT/proto-opencl/igneum-gpu-telemetry.exe" "$ROOT/proto-opencl/gpu-telemetry.c" "$ROOT/vendor/adlx/SDK/ADLXHelper/Windows/C/ADLXHelper.c" "$RES/igneum-worker-gpu-telemetry.res.o" -lsetupapi -lpdh
|
||||
"$STRIP" "$ROOT/proto-opencl/igneum-gpu-telemetry.exe"
|
||||
rm -rf "$RES"
|
||||
for exe in "$HERE/igneum-worker-cuda.exe" "$ROOT/proto-opencl/igneum-worker-opencl.exe"; do
|
||||
for exe in "$HERE/igneum-worker-cuda.exe" "$ROOT/proto-opencl/igneum-worker-opencl.exe" "$ROOT/proto-opencl/igneum-gpu-telemetry.exe"; do
|
||||
printf '%s: %d bytes, imports:' "$(basename "$exe")" "$(stat -f %z "$exe")"
|
||||
x86_64-w64-mingw32-objdump -p "$exe" | sed -n 's/^[[:space:]]*DLL Name: //p' | tr '\n' ' '
|
||||
echo
|
||||
done
|
||||
# the icon and version block survived the strip (strip keeps .rsrc; this proves it)
|
||||
python3 "$VERIFY" --version 0.3.0 "$HERE/igneum-worker-cuda.exe" "$ROOT/proto-opencl/igneum-worker-opencl.exe"
|
||||
python3 "$VERIFY" --version 0.3.0 "$HERE/igneum-worker-cuda.exe" "$ROOT/proto-opencl/igneum-worker-opencl.exe" "$ROOT/proto-opencl/igneum-gpu-telemetry.exe"
|
||||
|
|
|
|||
35
proto-cuda/nvrtc/igneum-worker-gpu-telemetry.rc
Normal file
35
proto-cuda/nvrtc/igneum-worker-gpu-telemetry.rc
Normal file
|
|
@ -0,0 +1,35 @@
|
|||
// Windows resources for igneum-gpu-telemetry.exe: the coin icon Explorer shows and the version block under Properties > Details.
|
||||
// Compiled with x86_64-w64-mingw32-windres (the icon path is relative to brand/icons, passed with -I).
|
||||
// the project lead's rule (4 October 2026): every shipped exe carries the coin icon and a version block, like the Mac app and DMG.
|
||||
#include <winver.h>
|
||||
|
||||
1 ICON "igneum.ico"
|
||||
|
||||
1 VERSIONINFO
|
||||
FILEVERSION 0,3,0,0
|
||||
PRODUCTVERSION 0,3,0,0
|
||||
FILEFLAGSMASK 0x3fL
|
||||
FILEFLAGS 0x0L
|
||||
FILEOS VOS_NT_WINDOWS32
|
||||
FILETYPE VFT_APP
|
||||
FILESUBTYPE VFT2_UNKNOWN
|
||||
BEGIN
|
||||
BLOCK "StringFileInfo"
|
||||
BEGIN
|
||||
BLOCK "040904B0"
|
||||
BEGIN
|
||||
VALUE "CompanyName", "Igneum"
|
||||
VALUE "FileDescription", "Igneum Miner GPU telemetry (AMD power, heat, fans, clocks)"
|
||||
VALUE "FileVersion", "0.3.0"
|
||||
VALUE "InternalName", "igneum-gpu-telemetry"
|
||||
VALUE "LegalCopyright", "Igneum contributors"
|
||||
VALUE "OriginalFilename", "igneum-gpu-telemetry.exe"
|
||||
VALUE "ProductName", "Igneum Miner"
|
||||
VALUE "ProductVersion", "0.3.0"
|
||||
END
|
||||
END
|
||||
BLOCK "VarFileInfo"
|
||||
BEGIN
|
||||
VALUE "Translation", 0x409, 1200
|
||||
END
|
||||
END
|
||||
|
|
@ -59,6 +59,8 @@ proto-opencl/
|
|||
cl_dynamic.h Windows one-click build: OpenCL.dll loaded at run time (IGNEUM_CL_DYNAMIC)
|
||||
test-generic.sh the --pack mode checked here through Apple OpenCL (needs proto-cuda/nvrtc/emu/test.sh's packs)
|
||||
test_host.c device-free unit tests of host.c's rules (the duplicate-platform fold); run with test-host.sh
|
||||
gpu-telemetry.c igneum-gpu-telemetry: AMD power, temperature, fan, clocks and busy per card (ADLX on Windows, amdgpu sysfs on Linux),
|
||||
one line per card per sample; the app's AMD card row reads it (engine.rs amd_telemetry_line)
|
||||
build.sh macOS (-framework OpenCL, or the Khronos ICD loader) and Linux (-lOpenCL)
|
||||
build.bat Windows (MSVC cl.exe + OpenCL.lib)
|
||||
WAVEFRONT.md wave32 vs wave64 on AMD, and why the kernel cannot tell the difference
|
||||
|
|
|
|||
274
proto-opencl/gpu-telemetry.c
Normal file
274
proto-opencl/gpu-telemetry.c
Normal file
|
|
@ -0,0 +1,274 @@
|
|||
// igneum-gpu-telemetry: power, temperature, fan and clocks of every AMD GPU, one line per card per sample.
|
||||
// 5 October 2026, after the project lead watched a 9070 XT at 90% usage with its fans barely turning and the app could not say
|
||||
// what it drew (the app's draw, temperature and MH per watt line came from nvidia-smi only).
|
||||
//
|
||||
// igneum-gpu-telemetry [-l SECONDS] one sample (default), or one every SECONDS until stdin closes or SIGTERM
|
||||
//
|
||||
// Windows: ADLX (the AMD Device Library eXtra, amdadlx64.dll, shipped with Adrenalin; vendor/adlx is the SDK clone,
|
||||
// MIT) for the metrics, keyed by the card's PCI bus from SetupAPI (the display class, matched by the same name ADLX
|
||||
// reports). Without ADLX (no AMD driver, an old one, or the DLL missing) only the utilisation is read, from the
|
||||
// GPU Engine performance counters through PDH, keyed by the adapter LUID that Windows uses there.
|
||||
// Linux: the amdgpu sysfs (/sys/class/drm/card*/device: hwmon power1_average, temp1_input, fan1_input, pwm1,
|
||||
// pp_dpm_mclk, gpu_busy_percent), keyed by the PCI address of the device link.
|
||||
//
|
||||
// Line format (space separated, every field present, a value the source cannot give prints as -):
|
||||
// amd <ordinal> bus <pci bus or address> kind integrated|discrete name "<name>" watts <W> temp_c <C> fan_rpm <rpm>
|
||||
// fan_pct <%> mclk_mhz <MHz> gclk_mhz <MHz> util_pct <%> source adlx|sysfs|perfcounter
|
||||
// then one `end <ms>` line per sample. The app (engine.rs amd_telemetry_line) parses it; parsers are unit-tested
|
||||
// against lines captured on PC 1.
|
||||
#define _CRT_SECURE_NO_WARNINGS
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <signal.h>
|
||||
|
||||
static volatile int gStop = 0;
|
||||
static void onSignal(int s) { (void)s; gStop = 1; }
|
||||
|
||||
typedef struct {
|
||||
char bus[64];
|
||||
char kind[16];
|
||||
char name[128];
|
||||
double watts, tempC, fanRpm, fanPct, mclk, gclk, util; /* -1 = not available */
|
||||
const char* source;
|
||||
} Sample;
|
||||
|
||||
static void sampleInit(Sample* s) { memset(s, 0, sizeof(*s)); strcpy(s->bus, "-"); strcpy(s->kind, "-"); strcpy(s->name, "-"); s->watts = s->tempC = s->fanRpm = s->fanPct = s->mclk = s->gclk = s->util = -1.0; s->source = "-"; }
|
||||
static void printNum(double v, const char* fmt) { if (v < 0) printf(" -"); else printf(fmt, v); }
|
||||
static void printSample(int ordinal, const Sample* s) {
|
||||
printf("amd %d bus %s kind %s name \"%s\" watts", ordinal, s->bus, s->kind, s->name);
|
||||
printNum(s->watts, " %.1f"); printf(" temp_c"); printNum(s->tempC, " %.1f"); printf(" fan_rpm"); printNum(s->fanRpm, " %.0f");
|
||||
printf(" fan_pct"); printNum(s->fanPct, " %.0f"); printf(" mclk_mhz"); printNum(s->mclk, " %.0f"); printf(" gclk_mhz"); printNum(s->gclk, " %.0f");
|
||||
printf(" util_pct"); printNum(s->util, " %.0f"); printf(" source %s\n", s->source);
|
||||
}
|
||||
|
||||
#ifdef _WIN32
|
||||
#define WIN32_LEAN_AND_MEAN
|
||||
#include <windows.h>
|
||||
#include <setupapi.h>
|
||||
#include <pdh.h>
|
||||
#include "../vendor/adlx/SDK/ADLXHelper/Windows/C/ADLXHelper.h"
|
||||
#include "../vendor/adlx/SDK/Include/IPerformanceMonitoring.h"
|
||||
|
||||
/* The SDK declares these three and leaves them to the platform file of each sample. */
|
||||
adlx_handle ADLX_CDECL_CALL adlx_load_library(const TCHAR* filename) { return (adlx_handle)LoadLibrary(filename); }
|
||||
int ADLX_CDECL_CALL adlx_free_library(adlx_handle module) { return FreeLibrary((HMODULE)module) ? 1 : 0; }
|
||||
void* ADLX_CDECL_CALL adlx_get_proc_address(adlx_handle module, const char* procName) { return (void*)GetProcAddress((HMODULE)module, procName); }
|
||||
|
||||
static double nowMs(void) { LARGE_INTEGER f, c; QueryPerformanceFrequency(&f); QueryPerformanceCounter(&c); return (double)c.QuadPart * 1000.0 / (double)f.QuadPart; }
|
||||
|
||||
/* The PCI bus of every display-class device, by its name (SetupAPI; the names are the ones ADLX reports). */
|
||||
typedef struct { char name[128]; int bus; } BusEntry;
|
||||
static int listBuses(BusEntry* out, int cap) {
|
||||
static const GUID DISPLAY = { 0x4d36e968, 0xe325, 0x11ce, { 0xbf, 0xc1, 0x08, 0x00, 0x2b, 0xe1, 0x03, 0x18 } };
|
||||
HDEVINFO set = SetupDiGetClassDevsA(&DISPLAY, NULL, NULL, DIGCF_PRESENT);
|
||||
SP_DEVINFO_DATA d;
|
||||
DWORD i;
|
||||
int n = 0;
|
||||
if (set == INVALID_HANDLE_VALUE) return 0;
|
||||
d.cbSize = sizeof(d);
|
||||
for (i = 0; SetupDiEnumDeviceInfo(set, i, &d) && n < cap; ++i) {
|
||||
char name[128] = { 0 };
|
||||
DWORD bus = 0, type = 0, got = 0;
|
||||
if (!SetupDiGetDeviceRegistryPropertyA(set, &d, SPDRP_DEVICEDESC, &type, (BYTE*)name, sizeof(name) - 1, &got)) continue;
|
||||
if (!SetupDiGetDeviceRegistryPropertyA(set, &d, SPDRP_BUSNUMBER, &type, (BYTE*)&bus, sizeof(bus), &got)) continue;
|
||||
snprintf(out[n].name, sizeof(out[n].name), "%s", name); out[n].bus = (int)bus; ++n;
|
||||
}
|
||||
SetupDiDestroyDeviceInfoList(set);
|
||||
return n;
|
||||
}
|
||||
static int busOf(const BusEntry* b, int n, const char* name, int* taken) {
|
||||
int i;
|
||||
for (i = 0; i < n; ++i) if (!taken[i] && strcmp(b[i].name, name) == 0) { taken[i] = 1; return b[i].bus; }
|
||||
return -1;
|
||||
}
|
||||
|
||||
/* ADLX: one sample of every GPU. Returns the number of lines printed, -1 when ADLX is not usable (reason printed). */
|
||||
static IADLXSystem* gSys = NULL;
|
||||
static IADLXPerformanceMonitoringServices* gPerf = NULL;
|
||||
static int adlxOpen(void) {
|
||||
ADLX_RESULT r = ADLXHelper_Initialize();
|
||||
if (!ADLX_SUCCEEDED(r)) { printf("info adlx: ADLXHelper_Initialize returned %d (no AMD driver with ADLX; amdadlx64.dll missing or too old)\n", (int)r); return 0; }
|
||||
gSys = ADLXHelper_GetSystemServices();
|
||||
if (!gSys) { printf("info adlx: no system services\n"); return 0; }
|
||||
r = gSys->pVtbl->GetPerformanceMonitoringServices(gSys, &gPerf);
|
||||
if (!ADLX_SUCCEEDED(r) || !gPerf) { printf("info adlx: GetPerformanceMonitoringServices returned %d\n", (int)r); return 0; }
|
||||
return 1;
|
||||
}
|
||||
static int adlxSample(const BusEntry* buses, int nBuses) {
|
||||
IADLXGPUList* gpus = NULL;
|
||||
adlx_uint it;
|
||||
int ordinal = 0;
|
||||
int taken[32] = { 0 };
|
||||
ADLX_RESULT r = gSys->pVtbl->GetGPUs(gSys, &gpus);
|
||||
if (!ADLX_SUCCEEDED(r) || !gpus) { printf("info adlx: GetGPUs returned %d\n", (int)r); return 0; }
|
||||
for (it = gpus->pVtbl->Begin(gpus); it != gpus->pVtbl->End(gpus); ++it) {
|
||||
IADLXGPU* gpu = NULL;
|
||||
IADLXGPUMetrics* m = NULL;
|
||||
Sample s;
|
||||
const char* name = NULL;
|
||||
ADLX_GPU_TYPE type = GPUTYPE_UNDEFINED;
|
||||
adlx_double dv = 0; adlx_int iv = 0;
|
||||
if (!ADLX_SUCCEEDED(gpus->pVtbl->At_GPUList(gpus, it, &gpu)) || !gpu) continue;
|
||||
sampleInit(&s);
|
||||
s.source = "adlx";
|
||||
if (ADLX_SUCCEEDED(gpu->pVtbl->Name(gpu, &name)) && name) snprintf(s.name, sizeof(s.name), "%s", name);
|
||||
if (ADLX_SUCCEEDED(gpu->pVtbl->Type(gpu, &type))) strcpy(s.kind, type == GPUTYPE_INTEGRATED ? "integrated" : type == GPUTYPE_DISCRETE ? "discrete" : "-");
|
||||
{ int b = busOf(buses, nBuses, s.name, taken); if (b >= 0) snprintf(s.bus, sizeof(s.bus), "%d", b); }
|
||||
r = gPerf->pVtbl->GetCurrentGPUMetrics(gPerf, gpu, &m);
|
||||
if (ADLX_SUCCEEDED(r) && m) {
|
||||
if (ADLX_SUCCEEDED(m->pVtbl->GPUPower(m, &dv))) s.watts = dv;
|
||||
if (s.watts < 0 && ADLX_SUCCEEDED(m->pVtbl->GPUTotalBoardPower(m, &dv))) s.watts = dv;
|
||||
if (ADLX_SUCCEEDED(m->pVtbl->GPUTemperature(m, &dv))) s.tempC = dv;
|
||||
if (ADLX_SUCCEEDED(m->pVtbl->GPUFanSpeed(m, &iv))) s.fanRpm = iv;
|
||||
if (ADLX_SUCCEEDED(m->pVtbl->GPUVRAMClockSpeed(m, &iv))) s.mclk = iv;
|
||||
if (ADLX_SUCCEEDED(m->pVtbl->GPUClockSpeed(m, &iv))) s.gclk = iv;
|
||||
if (ADLX_SUCCEEDED(m->pVtbl->GPUUsage(m, &dv))) s.util = dv;
|
||||
m->pVtbl->Release(m);
|
||||
} else {
|
||||
printf("info adlx: GetCurrentGPUMetrics for \"%s\" returned %d\n", s.name, (int)r);
|
||||
}
|
||||
/* fan percent: ADLX gives rpm only here; the tuning interface has the range, the app shows rpm when pct is - */
|
||||
printSample(ordinal++, &s);
|
||||
gpu->pVtbl->Release(gpu);
|
||||
}
|
||||
gpus->pVtbl->Release(gpus);
|
||||
return ordinal;
|
||||
}
|
||||
|
||||
/* PDH fallback: GPU engine utilisation per adapter LUID, summed over the engines (no power, no temperature). */
|
||||
static int pdhSample(void) {
|
||||
PDH_HQUERY q = NULL;
|
||||
PDH_HCOUNTER c = NULL;
|
||||
DWORD size = 0, count = 0, i;
|
||||
PDH_FMT_COUNTERVALUE_ITEM_A* items;
|
||||
int ordinal = 0;
|
||||
if (PdhOpenQueryA(NULL, 0, &q) != ERROR_SUCCESS) { printf("info perfcounter: PdhOpenQuery failed\n"); return 0; }
|
||||
if (PdhAddEnglishCounterA(q, "\\GPU Engine(*)\\Utilization Percentage", 0, &c) != ERROR_SUCCESS) { printf("info perfcounter: no GPU Engine counters\n"); PdhCloseQuery(q); return 0; }
|
||||
PdhCollectQueryData(q); Sleep(1000); PdhCollectQueryData(q);
|
||||
PdhGetFormattedCounterArrayA(c, PDH_FMT_DOUBLE, &size, &count, NULL);
|
||||
items = (PDH_FMT_COUNTERVALUE_ITEM_A*)malloc(size ? size : 1);
|
||||
if (PdhGetFormattedCounterArrayA(c, PDH_FMT_DOUBLE, &size, &count, items) == ERROR_SUCCESS) {
|
||||
/* instance names: pid_1234_luid_0x00000000_0x0000D4E3_phys_0_eng_0_engtype_3D; sum per luid */
|
||||
char luids[16][40]; double sums[16]; int n = 0, k;
|
||||
for (i = 0; i < count; ++i) {
|
||||
const char* p = strstr(items[i].szName, "luid_");
|
||||
char luid[40];
|
||||
if (!p) continue;
|
||||
snprintf(luid, sizeof(luid), "%.39s", p); { char* e = strstr(luid, "_phys"); if (e) *e = 0; }
|
||||
for (k = 0; k < n; ++k) if (strcmp(luids[k], luid) == 0) break;
|
||||
if (k == n && n < 16) { strcpy(luids[n], luid); sums[n] = 0; ++n; }
|
||||
if (k < 16) sums[k] += items[i].FmtValue.doubleValue;
|
||||
}
|
||||
for (k = 0; k < n; ++k) {
|
||||
Sample s; sampleInit(&s); s.source = "perfcounter";
|
||||
snprintf(s.bus, sizeof(s.bus), "%s", luids[k]);
|
||||
s.util = sums[k] > 100.0 ? 100.0 : sums[k];
|
||||
printSample(ordinal++, &s);
|
||||
}
|
||||
}
|
||||
free(items);
|
||||
PdhCloseQuery(q);
|
||||
return ordinal;
|
||||
}
|
||||
|
||||
int main(int argc, char** argv) {
|
||||
int every = 0, i, haveAdlx;
|
||||
BusEntry buses[32];
|
||||
int nBuses;
|
||||
for (i = 1; i < argc; ++i) if (strcmp(argv[i], "-l") == 0 && i + 1 < argc) every = atoi(argv[++i]);
|
||||
signal(SIGINT, onSignal); signal(SIGTERM, onSignal);
|
||||
setvbuf(stdout, NULL, _IOLBF, 0);
|
||||
nBuses = listBuses(buses, 32);
|
||||
for (i = 0; i < nBuses; ++i) printf("info display device \"%s\" bus %d\n", buses[i].name, buses[i].bus);
|
||||
haveAdlx = adlxOpen();
|
||||
do {
|
||||
double t0 = nowMs();
|
||||
int n = haveAdlx ? adlxSample(buses, nBuses) : pdhSample();
|
||||
printf("end %.1f ms %d card(s)\n", nowMs() - t0, n);
|
||||
fflush(stdout); /* a redirected stdout is fully buffered on the Windows CRT whatever setvbuf asks (PC 1 lost 60 s of samples at the kill) */
|
||||
if (every > 0) Sleep((DWORD)every * 1000);
|
||||
} while (every > 0 && !gStop);
|
||||
if (haveAdlx) { if (gPerf) gPerf->pVtbl->Release(gPerf); ADLXHelper_Terminate(); }
|
||||
return 0;
|
||||
}
|
||||
#else
|
||||
#include <dirent.h>
|
||||
#include <unistd.h>
|
||||
#include <time.h>
|
||||
static double nowMs(void) { struct timespec ts; clock_gettime(CLOCK_MONOTONIC, &ts); return ts.tv_sec * 1000.0 + ts.tv_nsec / 1e6; }
|
||||
static int readText(const char* path, char* out, size_t cap) { FILE* f = fopen(path, "r"); size_t n; if (!f) return 0; n = fread(out, 1, cap - 1, f); fclose(f); out[n] = 0; return 1; }
|
||||
static double readNumber(const char* path) { char b[64]; if (!readText(path, b, sizeof(b))) return -1.0; return atof(b); }
|
||||
/* pp_dpm_mclk: lines "0: 96Mhz", "3: 1258Mhz *"; the starred line is the current state */
|
||||
static double dpmCurrent(const char* text) {
|
||||
const char* p = text;
|
||||
while (p && *p) {
|
||||
const char* nl = strchr(p, '\n');
|
||||
size_t len = nl ? (size_t)(nl - p) : strlen(p);
|
||||
const char* star = memchr(p, '*', len);
|
||||
if (star) { const char* colon = memchr(p, ':', len); if (colon) return atof(colon + 1); }
|
||||
p = nl ? nl + 1 : NULL;
|
||||
}
|
||||
return -1.0;
|
||||
}
|
||||
static int sysfsSample(const char* root) {
|
||||
DIR* d = opendir(root);
|
||||
struct dirent* e;
|
||||
int ordinal = 0;
|
||||
if (!d) { printf("info sysfs: no %s\n", root); return 0; }
|
||||
while ((e = readdir(d)) != NULL) {
|
||||
char dev[512], path[640], text[4096], link[512];
|
||||
ssize_t ln;
|
||||
Sample s;
|
||||
DIR* hw; struct dirent* he;
|
||||
if (strncmp(e->d_name, "card", 4) != 0 || strchr(e->d_name + 4, '-')) continue;
|
||||
snprintf(dev, sizeof(dev), "%s/%s/device", root, e->d_name);
|
||||
snprintf(path, sizeof(path), "%s/vendor", dev);
|
||||
if (!readText(path, text, sizeof(text)) || strtol(text, NULL, 16) != 0x1002) continue;
|
||||
sampleInit(&s);
|
||||
s.source = "sysfs";
|
||||
ln = readlink(dev, link, sizeof(link) - 1);
|
||||
if (ln > 0) { link[ln] = 0; { const char* base = strrchr(link, '/'); snprintf(s.bus, sizeof(s.bus), "%.63s", base ? base + 1 : link); } }
|
||||
snprintf(path, sizeof(path), "%s/product_name", dev);
|
||||
if (readText(path, text, sizeof(text))) { text[strcspn(text, "\n")] = 0; snprintf(s.name, sizeof(s.name), "%s", text); }
|
||||
else { snprintf(path, sizeof(path), "%s/device", dev); if (readText(path, text, sizeof(text))) { text[strcspn(text, "\n")] = 0; snprintf(s.name, sizeof(s.name), "amdgpu %s", text); } }
|
||||
snprintf(path, sizeof(path), "%s/boot_vga", dev);
|
||||
strcpy(s.kind, "discrete");
|
||||
snprintf(path, sizeof(path), "%s/hwmon", dev);
|
||||
hw = opendir(path);
|
||||
if (hw) {
|
||||
while ((he = readdir(hw)) != NULL) {
|
||||
char hp[900];
|
||||
if (strncmp(he->d_name, "hwmon", 5) != 0) continue;
|
||||
snprintf(hp, sizeof(hp), "%s/%s/power1_average", path, he->d_name); s.watts = readNumber(hp); if (s.watts < 0) { snprintf(hp, sizeof(hp), "%s/%s/power1_input", path, he->d_name); s.watts = readNumber(hp); } if (s.watts >= 0) s.watts /= 1e6;
|
||||
snprintf(hp, sizeof(hp), "%s/%s/temp1_input", path, he->d_name); s.tempC = readNumber(hp); if (s.tempC >= 0) s.tempC /= 1000.0;
|
||||
snprintf(hp, sizeof(hp), "%s/%s/fan1_input", path, he->d_name); s.fanRpm = readNumber(hp);
|
||||
{ double pwm, pwmMax; snprintf(hp, sizeof(hp), "%s/%s/pwm1", path, he->d_name); pwm = readNumber(hp); snprintf(hp, sizeof(hp), "%s/%s/pwm1_max", path, he->d_name); pwmMax = readNumber(hp); if (pwm >= 0 && pwmMax > 0) s.fanPct = 100.0 * pwm / pwmMax; else if (pwm >= 0) s.fanPct = 100.0 * pwm / 255.0; }
|
||||
break;
|
||||
}
|
||||
closedir(hw);
|
||||
}
|
||||
snprintf(path, sizeof(path), "%s/pp_dpm_mclk", dev); if (readText(path, text, sizeof(text))) s.mclk = dpmCurrent(text);
|
||||
snprintf(path, sizeof(path), "%s/pp_dpm_sclk", dev); if (readText(path, text, sizeof(text))) s.gclk = dpmCurrent(text);
|
||||
snprintf(path, sizeof(path), "%s/gpu_busy_percent", dev); s.util = readNumber(path);
|
||||
printSample(ordinal++, &s);
|
||||
}
|
||||
closedir(d);
|
||||
return ordinal;
|
||||
}
|
||||
int main(int argc, char** argv) {
|
||||
int every = 0, i;
|
||||
const char* root = getenv("IGNEUM_DRM_ROOT") ? getenv("IGNEUM_DRM_ROOT") : "/sys/class/drm"; /* a fixture tree for tests */
|
||||
for (i = 1; i < argc; ++i) if (strcmp(argv[i], "-l") == 0 && i + 1 < argc) every = atoi(argv[++i]);
|
||||
signal(SIGINT, onSignal); signal(SIGTERM, onSignal);
|
||||
setvbuf(stdout, NULL, _IOLBF, 0);
|
||||
do {
|
||||
double t0 = nowMs();
|
||||
int n = sysfsSample(root);
|
||||
printf("end %.1f ms %d card(s)\n", nowMs() - t0, n);
|
||||
fflush(stdout); /* a redirected stdout is fully buffered on the Windows CRT whatever setvbuf asks (PC 1 lost 60 s of samples at the kill) */
|
||||
if (every > 0) sleep((unsigned)every);
|
||||
} while (every > 0 && !gStop);
|
||||
return 0;
|
||||
}
|
||||
#endif
|
||||
Loading…
Reference in a new issue