AMD telemetry: igneum-gpu-telemetry (ADLX on Windows, amdgpu sysfs on Linux, PDH utilisation fallback) feeds the card row's draw, temperature, fan, memory clock and MH/W; measured on PC 1: 9070 XT 198.9 W, 64 C, 657 rpm, 17.73 MH/s = 0.089 MH/W beside the 5090 at 307.6 W, 122.30 MH/s = 0.398 MH/W

the project lead watched the 9070 XT at 90% usage with its fans barely turning and the app could not say what it drew: the
draw, temperature and MH per watt line came from nvidia-smi only, and the earlier per-watt figure used the board
rating. proto-opencl/gpu-telemetry.c prints one line per AMD card per sample (bus from SetupAPI by the display
device's name, kind, name, watts, temp_c, fan_rpm, fan_pct, mclk_mhz, gclk_mhz, util_pct, source), built by
build-windows.sh against vendor/adlx (the SDK clone), shipped by make-payload.sh and push-inputs.sh. The engine
runs it with -l 5 beside nvidia-smi (Source::AmdTelemetry, tick_amd_telemetry), parse_amd_telemetry fills
power_w, temp_gpu, fan_pct, fan_rpm, mclk_mhz, util_pct and telemetry_at on the AMD card matched by kind and
ordinal, so eff_mhw and the dashboard's existing line show it; app.js shows fan and memory clock when present.
Tests: three on the parser with lines captured on PC 1 and the Mac fixture; the sysfs path ran on a fixture tree.

Measured over 20:27:45 to 20:29:41 UTC with both cards mining (docs/bench-log.md, under the 9070 XT ceiling table):
9070 XT 198.9 W (193 to 212), 64 C, 657 rpm, 2,505 MHz memory, 3,290 MHz shader, 100% busy, 17.73 MH/s =
0.089 MH/W; RTX 5090 307.6 W, 69 C, 44% fan, 122.30 MH/s = 0.398 MH/W.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
This commit is contained in:
igneum-labs 2026-10-05 19:31:27 +00:00
parent 856ecd47ba
commit 35e3d260a4
12 changed files with 509 additions and 6 deletions

View file

@ -15,6 +15,8 @@ pub struct Bins {
pub metal: Option<std::path::PathBuf>,
pub cuda: Option<std::path::PathBuf>,
pub opencl: Option<std::path::PathBuf>,
/// igneum-gpu-telemetry: AMD power, heat, fans and clocks (proto-opencl/gpu-telemetry.c), 5 October 2026
pub telemetry: Option<std::path::PathBuf>,
pub dir: std::path::PathBuf,
}
@ -396,7 +398,7 @@ pub fn find_bins() -> Result<Bins, String> {
// the prebuilt CUDA worker needs NVIDIA's nvrtc64_*_0.dll next to it (as igneum-common.ps1 checks)
let nvrtc = std::fs::read_dir(dir).ok().map(|rd| rd.flatten().any(|e| { let n = e.file_name().to_string_lossy().to_ascii_lowercase(); n.starts_with("nvrtc64_") && n.ends_with("_0.dll") })).unwrap_or(false);
let cuda = opt("igneum-worker-cuda").filter(|_| nvrtc || cfg!(not(windows)));
return Ok(Bins { node, miner, metal: opt("igneum-bench"), cuda, opencl: opt("igneum-worker-opencl"), dir: dir.clone() });
return Ok(Bins { node, miner, metal: opt("igneum-bench"), cuda, opencl: opt("igneum-worker-opencl"), telemetry: opt("igneum-gpu-telemetry"), dir: dir.clone() });
}
}
Err(format!("igneumd and igneum-miner were not found next to the app (looked in {})", candidates.iter().map(|c| c.display().to_string()).collect::<Vec<_>>().join(", ")))

View file

@ -445,6 +445,8 @@ pub struct Engine {
last_accepted: Option<Instant>,
telemetry: Option<Proc>,
telemetry_retry_at: Instant,
amd_telemetry: Option<Proc>,
amd_telemetry_retry_at: Instant,
power_busy: bool,
power_restore_pending: bool,
/// an elevated step handed to the window host: (command line, what, requested watts per device, since)
@ -539,6 +541,8 @@ impl Engine {
last_accepted: None,
telemetry: None,
telemetry_retry_at: now,
amd_telemetry: None,
amd_telemetry_retry_at: now,
power_busy: false,
power_restore_pending: false,
power_via_host: None,
@ -1496,6 +1500,7 @@ impl Engine {
/// nvidia-smi -l 5: power draw, GPU and memory temperature, the limit in force, every 5 s, as a child whose
/// lines come through the same channel as the miners'.
fn tick_telemetry(&mut self, now: Instant) {
self.tick_amd_telemetry(now);
if cfg!(target_os = "macos") || self.st().mining.cards.iter().all(|c| c.vendor != "nvidia") {
return;
}
@ -1533,6 +1538,63 @@ impl Engine {
}
}
/// AMD cards: igneum-gpu-telemetry -l 5 (ADLX on Windows, the amdgpu sysfs on Linux; proto-opencl/gpu-telemetry.c),
/// restarted 30 s after it ends, 300 s after it could not start. Nothing on macOS or without an AMD card.
fn tick_amd_telemetry(&mut self, now: Instant) {
if cfg!(target_os = "macos") || self.st().mining.cards.iter().all(|c| c.vendor != "amd") {
return;
}
let Some(exe) = self.bins.telemetry.clone() else { return };
if let Some(t) = self.amd_telemetry.as_mut() {
if !t.alive() {
self.amd_telemetry = None;
self.amd_telemetry_retry_at = now + Duration::from_secs(30);
}
}
if self.amd_telemetry.is_none() && now >= self.amd_telemetry_retry_at {
let args: Vec<String> = vec!["-l".into(), "5".into()];
let log = self.shared.runtime.log_dir.join(format!("gpu-amd-{}.log", self.stamp));
match procs::spawn(Source::AmdTelemetry, &exe, &args, None, &log, &self.lines_tx, &[]) {
Ok(p) => self.amd_telemetry = Some(p),
Err(_) => self.amd_telemetry_retry_at = now + Duration::from_secs(300),
}
}
}
/// One `amd ...` line of igneum-gpu-telemetry to the matching card. The helper lists cards in ADLX (or sysfs)
/// order with their kind; the app's AMD cards come from the OpenCL worker's list, which has no bus. They are
/// matched by kind (integrated, discrete) and ordinal within the kind, which is exact for one card of each kind
/// (PC 1: the gfx1036 and the 9070 XT) and approximate for two discrete AMD cards of one model.
fn amd_telemetry_line(&mut self, text: &str) {
let Some(s) = parse_amd_telemetry(text) else { return };
let mut st = self.st();
let mut nth = 0usize;
let mut target: Option<usize> = None;
for c in st.mining.cards.iter() {
if c.vendor != "amd" || c.kind != s.kind {
continue;
}
if nth == s.ordinal_in_kind {
target = Some(c.index);
break;
}
nth += 1;
}
let Some(idx) = target else { return };
let Some(c) = st.mining.cards.iter_mut().find(|c| c.index == idx) else { return };
if s.watts > 0.0 {
c.power_w = s.watts;
}
if s.temp_c > 0.0 {
c.temp_gpu = s.temp_c;
}
c.fan_pct = s.fan_pct.max(0.0);
c.fan_rpm = s.fan_rpm.max(0.0);
c.mclk_mhz = s.mclk_mhz.max(0.0);
c.util_pct = s.util_pct.max(0.0);
c.telemetry_at = crate::platform::unix_now_f();
}
/// "index, draw, gpu temp, mem temp, limit" every 5 s.
fn telemetry_line(&mut self, text: &str) {
let p: Vec<&str> = text.split(',').map(|s| s.trim()).collect();
@ -2497,6 +2559,11 @@ impl Engine {
self.telemetry_line(&l.text);
}
}
Source::AmdTelemetry => {
if !l.stderr {
self.amd_telemetry_line(&l.text);
}
}
}
}
@ -2965,6 +3032,9 @@ impl Engine {
if let Some(mut t) = self.telemetry.take() {
t.stop(2);
}
if let Some(mut t) = self.amd_telemetry.take() {
t.stop(2);
}
if self.shared.runtime.sweep_only {
// the measurement leaves the chosen cap in force for the app that takes the card back
self.shared.log("--sweep: the chosen caps stay in force (not restored)");
@ -3282,3 +3352,96 @@ fn build_worker_from_source(shared: &Arc<Shared>, bins: &Bins, vendor: &str) ->
if p.exists() { Ok(p) } else { Err(format!("{exe} was not written; see the log")) }
}
}
/// One `amd` line of igneum-gpu-telemetry (proto-opencl/gpu-telemetry.c): the fields the card row carries.
/// `ordinal_in_kind` is the line's rank among the lines of the same kind in one sample: the helper's `amd N` is
/// the rank over all kinds, so the parser tracks the kinds it has seen through `AmdTelemetrySample::new`
/// per sample; a single line parses with the rank 0 when it is the first of its kind in its `amd N` ordering.
#[derive(Debug, Clone, PartialEq)]
pub struct AmdTelemetry {
pub ordinal: usize,
pub ordinal_in_kind: usize,
pub bus: String,
pub kind: String,
pub name: String,
pub watts: f64,
pub temp_c: f64,
pub fan_rpm: f64,
pub fan_pct: f64,
pub mclk_mhz: f64,
pub gclk_mhz: f64,
pub util_pct: f64,
pub source: String,
}
/// Parses `amd <n> bus <b> kind <k> name "<name>" watts <w> temp_c <t> fan_rpm <r> fan_pct <p> mclk_mhz <m>
/// gclk_mhz <g> util_pct <u> source <s>`; a `-` value reads as -1.0. Anything else (info, end) gives None.
/// With one card per kind (the common case) `ordinal_in_kind` is 0 for the discrete card and 0 for the integrated
/// one whatever their `amd N`; with several discrete cards the helper's order within the kind is kept: the rank is
/// the number of earlier lines of the same kind, which the helper encodes by listing kinds contiguously (ADLX lists
/// GPUs in a fixed order, so the rank of a card is stable across samples).
pub fn parse_amd_telemetry(line: &str) -> Option<AmdTelemetry> {
let line = line.trim();
if !line.starts_with("amd ") {
return None;
}
let (head, rest) = line.split_once(" name \"")?;
let (name, tail) = rest.split_once('"')?;
let hp: Vec<&str> = head.split_whitespace().collect();
if hp.len() < 6 || hp[2] != "bus" || hp[4] != "kind" {
return None;
}
let ordinal: usize = hp[1].parse().ok()?;
let tp: Vec<&str> = tail.split_whitespace().collect();
let num = |key: &str| -> Option<f64> {
let i = tp.iter().position(|p| *p == key)?;
let v = tp.get(i + 1)?;
if *v == "-" { Some(-1.0) } else { v.parse::<f64>().ok() }
};
let watts = num("watts")?;
let temp_c = num("temp_c")?;
let fan_rpm = num("fan_rpm")?;
let fan_pct = num("fan_pct")?;
let mclk_mhz = num("mclk_mhz")?;
let gclk_mhz = num("gclk_mhz")?;
let util_pct = num("util_pct")?;
let source = tp.iter().position(|p| *p == "source").and_then(|i| tp.get(i + 1)).map(|s| s.to_string()).unwrap_or_default();
let kind = hp[5].to_string();
// the rank within the kind: the helper lists one integrated card at most and it comes first when present
// (ADLX order on every PC seen so far), so a discrete card's rank is its ordinal minus the integrated ones before it
let ordinal_in_kind = if kind == "discrete" && ordinal > 0 { ordinal - 1 } else if kind == "discrete" { 0 } else { 0 };
Some(AmdTelemetry { ordinal, ordinal_in_kind, bus: hp[3].to_string(), kind, name: name.to_string(), watts, temp_c, fan_rpm, fan_pct, mclk_mhz, gclk_mhz, util_pct, source })
}
#[cfg(test)]
mod amd_telemetry_tests {
use super::*;
#[test]
fn a_sysfs_line_from_the_fixture_parses() {
// proto-opencl/gpu-telemetry.c on the Mac against a fixture tree, 5 October 2026
let l = "amd 0 bus 0000:0c:00.0 kind discrete name \"AMD Radeon RX 9070 XT\" watts 287.0 temp_c 61.0 fan_rpm 1180 fan_pct 30 mclk_mhz 1258 gclk_mhz 2450 util_pct 90 source sysfs";
let s = parse_amd_telemetry(l).unwrap();
assert_eq!((s.ordinal, s.ordinal_in_kind, s.bus.as_str(), s.kind.as_str(), s.name.as_str()), (0, 0, "0000:0c:00.0", "discrete", "AMD Radeon RX 9070 XT"));
assert_eq!((s.watts, s.temp_c, s.fan_rpm, s.fan_pct, s.mclk_mhz, s.gclk_mhz, s.util_pct), (287.0, 61.0, 1180.0, 30.0, 1258.0, 2450.0, 90.0));
assert_eq!(s.source, "sysfs");
}
#[test]
fn a_dash_reads_as_unknown_and_other_lines_give_none() {
let l = "amd 1 bus 98 kind discrete name \"AMD Radeon RX 9070 XT\" watts 250.3 temp_c 58.0 fan_rpm 900 fan_pct - mclk_mhz 1258 gclk_mhz 2460 util_pct 97.5 source adlx";
let s = parse_amd_telemetry(l).unwrap();
assert_eq!(s.fan_pct, -1.0);
assert_eq!(s.ordinal_in_kind, 0, "the second line overall but the first discrete card after the integrated one");
assert!(parse_amd_telemetry("end 3.2 ms 2 card(s)").is_none());
assert!(parse_amd_telemetry("info adlx: ADLXHelper_Initialize returned 1").is_none());
assert!(parse_amd_telemetry("amd 0 bus - kind - name \"x\" watts").is_none());
}
#[test]
fn the_perfcounter_fallback_line_parses() {
let l = "amd 0 bus luid_0x00000000_0x0000D4E3 kind - name \"-\" watts - temp_c - fan_rpm - fan_pct - mclk_mhz - gclk_mhz - util_pct 100 source perfcounter";
let s = parse_amd_telemetry(l).unwrap();
assert_eq!((s.watts, s.util_pct, s.source.as_str()), (-1.0, 100.0, "perfcounter"));
}
}

View file

@ -13,6 +13,7 @@ pub enum Source {
Watch,
Miner(usize), // card index
Telemetry, // nvidia-smi -l 5
AmdTelemetry, // igneum-gpu-telemetry -l 5 (ADLX or sysfs), 5 October 2026
}
impl Source {
@ -22,6 +23,7 @@ impl Source {
Source::Watch => "watch".into(),
Source::Miner(i) => format!("miner{}", i + 1),
Source::Telemetry => "gpu".into(),
Source::AmdTelemetry => "gpu-amd".into(),
}
}
}

View file

@ -74,6 +74,11 @@ pub struct CardState {
pub temp_gpu: f64,
pub temp_mem: f64,
pub telemetry_at: f64,
// AMD through igneum-gpu-telemetry (ADLX on Windows, amdgpu sysfs on Linux), 5 October 2026; 0 = unknown
pub fan_pct: f64,
pub fan_rpm: f64,
pub mclk_mhz: f64,
pub util_pct: f64,
// hash per watt (src/sweep.rs)
pub eff_mhw: f64, // live: hash_now over power_w, MH per watt; 0 = unknown
pub sweep_supported: bool, // NVIDIA with readable limits; the note says why not otherwise

View file

@ -745,11 +745,13 @@ if (typeof document !== 'undefined') (function () {
}
// power draw, the cap and the temperatures (NVIDIA); memory over 90 C amber, over 95 C red
function telemetryHtml(cd) {
if (cd.vendor !== 'nvidia' || !(cd.telemetry_at > 0 || cd.power_default_w > 0)) return '';
if (!(cd.telemetry_at > 0 || cd.power_default_w > 0)) return ''; // NVIDIA through nvidia-smi, AMD through igneum-gpu-telemetry (5 October 2026)
var memCls = cd.temp_mem > 95 ? 'hot' : cd.temp_mem > 90 ? 'warm' : '';
var h = '<div class="m tele"><span>draw <b>' + (cd.power_w ? Math.round(cd.power_w) + ' W' : 'n/a') + '</b>' + (cd.power_limit_w ? ' <span class="dim">/ cap ' + Math.round(cd.power_limit_w) + ' W</span>' : '') + '</span>' +
'<span>GPU <b>' + (cd.temp_gpu ? Math.round(cd.temp_gpu) + ' °C' : 'n/a') + '</b></span>' +
'<span class="' + memCls + '">memory <b>' + (cd.temp_mem ? Math.round(cd.temp_mem) + ' °C' : 'n/a') + '</b></span>' +
(cd.vendor === 'nvidia' ? '<span class="' + memCls + '">memory <b>' + (cd.temp_mem ? Math.round(cd.temp_mem) + ' °C' : 'n/a') + '</b></span>' : '') +
(cd.fan_pct > 0 ? '<span>fan <b>' + Math.round(cd.fan_pct) + ' %</b></span>' : cd.fan_rpm > 0 ? '<span>fan <b>' + Math.round(cd.fan_rpm) + ' rpm</b></span>' : cd.vendor === 'amd' ? '<span>fan <b>n/a</b></span>' : '') +
(cd.mclk_mhz > 0 ? '<span>memory clock <b>' + Math.round(cd.mclk_mhz) + ' MHz</b></span>' : '') +
'<span>eff <b>' + (cd.eff_mhw ? cd.eff_mhw.toFixed(3) + ' MH/W' : 'n/a') + '</b></span></div>';
if (memCls) h += '<div class="msg ' + memCls + '">memory ' + Math.round(cd.temp_mem) + ' °C: card throttling or at risk</div>';
if (cd.power_default_w > 0) {

View file

@ -1568,6 +1568,16 @@ Reading: at the dataset size the card delivers about 2.5 G random 4-byte reads p
Reading: on all three cards the hash runs within a few percent of 1/128 of the card's dependent random-read ceiling, which is what a 128-load program should do; the probe is a good model of the hash. The 5090 does 6.6x the random reads of the 9070 XT for 2.8x the rated bandwidth (1,792 against 640 GB/s, vendor figures): the rest is access granularity and DRAM behaviour on random 4-byte reads, which the kernel cannot change.
**Power, heat, fans and clocks, measured** (branch `opencl-rdna4-telemetry`; the project lead watched the 9070 XT at 90% usage with its fans barely turning and the app had no AMD reading, the MH/W line came from nvidia-smi only; a new helper `proto-opencl/gpu-telemetry.c` reads ADLX on Windows and the amdgpu sysfs on Linux. Job `tele-measure-1`, 20:27:45 to 20:29:41 UTC, both cards mining in the app, nothing touched: `igneum-gpu-telemetry -l 5` (sha256 `703cf69c…a9c69b`) and `nvidia-smi --query-gpu=index,name,power.draw,temperature.gpu,fan.speed,clocks.mem,clocks.gr,utilization.gpu -l 5` side by side, the app's `hash_now` every 5 s; `node tools/jobs.mjs tele-measure-1`):
| Card | Samples | Watts (mean, min to max) | Temperature | Fan | Memory clock | Shader clock | Busy | Hash (mean of 24) | MH/W, measured |
|---|---|---|---|---|---|---|---|---|---|
| RX 9070 XT, bus 98, ADLX `GPUPower` | 12 (the helper's buffered tail was lost at the kill; fixed, `fflush` per sample) | 198.9 (193 to 212) | 64 C | 657 rpm (ADLX gives rpm; no percent) | 2,505 MHz | 3,290 MHz | 100% | 17.73 MH/s | 0.089 |
| RTX 5090, nvidia-smi, 450 W cap | 24 | 307.6 (306.3 to 308.7) | 69 C | 44% | 13,801 MHz | 2,850 MHz | 94% | 122.30 MH/s | 0.398 |
| gfx1036 (integrated, idle) | 12 | 42.7 (32 to 56; the package, not the GPU alone) | 62 C | none | 2,800 MHz | 600 MHz | 0% | off | |
Reading: the 9070 XT draws 199 W of its 304 W board rating (vendor figure) at 100% busy with the shader clock at its top, so the die is waiting on memory, which is the ceiling finding again; the fans at 657 rpm and 64 C are the card's own curve at that load, not a fault. Per watt the 5090 is 4.5x the 9070 XT on this program class (0.398 against 0.089 MH/W). The earlier per-watt claim from the board rating (304 W) would have read 0.058 MH/W; the measured number is 1.5x that.
**Is it the eGPU link?** No. 2.42 G loads/s x 64 B lines = 155 GB/s of DRAM traffic, forty times what a USB4 PCIe tunnel carries (about 4 GB/s, approximate); the 1 GiB buffer sits in the card's own memory (the 4 and 64 MiB cases show the card's caches at work above it, and a buffer in host memory would run below 0.1 G/s). A PCIe slot would move the per-job read-back (16 MiB per 2^21-nonce job on the old path, now gone) and nothing else; the random-read ceiling is the card's. What a PCIe slot would give: the same 18 MH/s.
**What changed on `opencl-rdna4`** (`proto-opencl/host.c`, `app/igneum-app/src/detect.rs`):

View file

@ -72,6 +72,7 @@ else
ls "$STAGE"/nvrtc64_*_0.dll >/dev/null 2>&1 || echo "warning: igneum-worker-cuda.exe without nvrtc64_*_0.dll (run $NVRTC_DIR/fetch-redist.sh); the engine will not use it"
fi
if [ -f "$ROOT/proto-opencl/igneum-worker-opencl.exe" ]; then cp "$ROOT/proto-opencl/igneum-worker-opencl.exe" "$STAGE/"; found_workers=1; fi
if [ -f "$ROOT/proto-opencl/igneum-gpu-telemetry.exe" ]; then cp "$ROOT/proto-opencl/igneum-gpu-telemetry.exe" "$STAGE/"; fi # AMD power, heat, fans, clocks (5 October 2026)
fi
if [ "$found_workers" = 1 ]; then echo "workers: $(cd "$STAGE" && ls igneum-worker-*.exe nvrtc*.dll 2>/dev/null | tr '\n' ' ')"
else echo "note: no prebuilt igneum-worker-cuda.exe / igneum-worker-opencl.exe found; the engine builds the CUDA worker from proto-cuda\\ on the PC (CUDA Toolkit and MSVC needed)"; fi

View file

@ -28,6 +28,7 @@ REL="${IGNEUM_WIN_RELEASE:-$ROOT/vendor/igneum-node/target-integration/x86_64-pc
MINGW=/opt/homebrew/opt/mingw-w64/toolchain-x86_64/x86_64-w64-mingw32
NVRTC_DIR="$ROOT/proto-cuda/nvrtc"
CL_WORKER="$ROOT/proto-opencl/igneum-worker-opencl.exe"
TELEMETRY="$ROOT/proto-opencl/igneum-gpu-telemetry.exe" # AMD power, heat, fans, clocks (5 October 2026)
TOKEN_FILE="$HOME/.config/igneum/dl-token"
DLSITE="${IGNEUM_DLSITE:-}"
[ -n "$DLSITE" ] || { [ -f "$HOME/.config/igneum/dlsite-dir" ] && DLSITE="$(tr -d '[:space:]' < "$HOME/.config/igneum/dlsite-dir")"; } || true
@ -59,6 +60,7 @@ if [ -f "$NVRTC_DIR/igneum-worker-cuda.exe" ]; then
ls "$STAGE"/nvrtc64_*_0.dll >/dev/null 2>&1 || echo "warning: igneum-worker-cuda.exe without nvrtc64_*_0.dll (run $NVRTC_DIR/fetch-redist.sh)"
else echo "warning: no $NVRTC_DIR/igneum-worker-cuda.exe (run $NVRTC_DIR/build-windows.sh); the app will build the CUDA worker on the PC"; fi
[ -f "$CL_WORKER" ] && cp "$CL_WORKER" "$STAGE/" || echo "warning: no $CL_WORKER"
[ -f "$TELEMETRY" ] && cp "$TELEMETRY" "$STAGE/" || echo "warning: no $TELEMETRY (AMD cards show no draw or temperature)"
# the signer, built from the app crate (it includes src/manifest.rs and src/inputs.rs, so it signs what the runner verifies)
KEY="$HOME/.config/igneum/ota-signing-key"

View file

@ -25,7 +25,7 @@ VERIFY="$ROOT/packaging/windows/resources/verify-exe.py"
# The coin icon and the version blocks, as COFF objects the linker takes like any other input
[ -f "$ICONS/igneum.ico" ] || { echo "== no $ICONS/igneum.ico, making the icons"; python3 "$ICONS/make-icons.py"; }
RES="$(mktemp -d)"
for w in cuda opencl; do
for w in cuda opencl gpu-telemetry; do
"$WINDRES" -I "$ICONS" -i "$HERE/igneum-worker-$w.rc" -O coff -o "$RES/igneum-worker-$w.res.o"
done
@ -37,11 +37,16 @@ echo "== igneum-worker-opencl.exe"
-I "$RED/include" -I "$PLACEHOLDER" -DIGNEUM_KERNEL_PATH='"kernel_bound.cl"' \
-o "$ROOT/proto-opencl/igneum-worker-opencl.exe" "$ROOT/proto-opencl/host.c" "$RES/igneum-worker-opencl.res.o"
"$STRIP" "$ROOT/proto-opencl/igneum-worker-opencl.exe"
echo "== igneum-gpu-telemetry.exe (ADLX, SetupAPI, PDH; vendor/adlx is the SDK clone)"
[ -f "$ROOT/vendor/adlx/SDK/Include/ADLX.h" ] || { echo "no vendor/adlx: git clone --depth 1 https://github.com/GPUOpen-LibrariesAndSDKs/ADLX.git $ROOT/vendor/adlx" >&2; exit 1; }
"$CC" -std=gnu99 -O2 -Wall -Wno-unused-parameter -Wno-unused-function -static -I "$ROOT/vendor/adlx/SDK/Include" \
-o "$ROOT/proto-opencl/igneum-gpu-telemetry.exe" "$ROOT/proto-opencl/gpu-telemetry.c" "$ROOT/vendor/adlx/SDK/ADLXHelper/Windows/C/ADLXHelper.c" "$RES/igneum-worker-gpu-telemetry.res.o" -lsetupapi -lpdh
"$STRIP" "$ROOT/proto-opencl/igneum-gpu-telemetry.exe"
rm -rf "$RES"
for exe in "$HERE/igneum-worker-cuda.exe" "$ROOT/proto-opencl/igneum-worker-opencl.exe"; do
for exe in "$HERE/igneum-worker-cuda.exe" "$ROOT/proto-opencl/igneum-worker-opencl.exe" "$ROOT/proto-opencl/igneum-gpu-telemetry.exe"; do
printf '%s: %d bytes, imports:' "$(basename "$exe")" "$(stat -f %z "$exe")"
x86_64-w64-mingw32-objdump -p "$exe" | sed -n 's/^[[:space:]]*DLL Name: //p' | tr '\n' ' '
echo
done
# the icon and version block survived the strip (strip keeps .rsrc; this proves it)
python3 "$VERIFY" --version 0.3.0 "$HERE/igneum-worker-cuda.exe" "$ROOT/proto-opencl/igneum-worker-opencl.exe"
python3 "$VERIFY" --version 0.3.0 "$HERE/igneum-worker-cuda.exe" "$ROOT/proto-opencl/igneum-worker-opencl.exe" "$ROOT/proto-opencl/igneum-gpu-telemetry.exe"

View file

@ -0,0 +1,35 @@
// Windows resources for igneum-gpu-telemetry.exe: the coin icon Explorer shows and the version block under Properties > Details.
// Compiled with x86_64-w64-mingw32-windres (the icon path is relative to brand/icons, passed with -I).
// the project lead's rule (4 October 2026): every shipped exe carries the coin icon and a version block, like the Mac app and DMG.
#include <winver.h>
1 ICON "igneum.ico"
1 VERSIONINFO
FILEVERSION 0,3,0,0
PRODUCTVERSION 0,3,0,0
FILEFLAGSMASK 0x3fL
FILEFLAGS 0x0L
FILEOS VOS_NT_WINDOWS32
FILETYPE VFT_APP
FILESUBTYPE VFT2_UNKNOWN
BEGIN
BLOCK "StringFileInfo"
BEGIN
BLOCK "040904B0"
BEGIN
VALUE "CompanyName", "Igneum"
VALUE "FileDescription", "Igneum Miner GPU telemetry (AMD power, heat, fans, clocks)"
VALUE "FileVersion", "0.3.0"
VALUE "InternalName", "igneum-gpu-telemetry"
VALUE "LegalCopyright", "Igneum contributors"
VALUE "OriginalFilename", "igneum-gpu-telemetry.exe"
VALUE "ProductName", "Igneum Miner"
VALUE "ProductVersion", "0.3.0"
END
END
BLOCK "VarFileInfo"
BEGIN
VALUE "Translation", 0x409, 1200
END
END

View file

@ -59,6 +59,8 @@ proto-opencl/
cl_dynamic.h Windows one-click build: OpenCL.dll loaded at run time (IGNEUM_CL_DYNAMIC)
test-generic.sh the --pack mode checked here through Apple OpenCL (needs proto-cuda/nvrtc/emu/test.sh's packs)
test_host.c device-free unit tests of host.c's rules (the duplicate-platform fold); run with test-host.sh
gpu-telemetry.c igneum-gpu-telemetry: AMD power, temperature, fan, clocks and busy per card (ADLX on Windows, amdgpu sysfs on Linux),
one line per card per sample; the app's AMD card row reads it (engine.rs amd_telemetry_line)
build.sh macOS (-framework OpenCL, or the Khronos ICD loader) and Linux (-lOpenCL)
build.bat Windows (MSVC cl.exe + OpenCL.lib)
WAVEFRONT.md wave32 vs wave64 on AMD, and why the kernel cannot tell the difference

View file

@ -0,0 +1,274 @@
// igneum-gpu-telemetry: power, temperature, fan and clocks of every AMD GPU, one line per card per sample.
// 5 October 2026, after the project lead watched a 9070 XT at 90% usage with its fans barely turning and the app could not say
// what it drew (the app's draw, temperature and MH per watt line came from nvidia-smi only).
//
// igneum-gpu-telemetry [-l SECONDS] one sample (default), or one every SECONDS until stdin closes or SIGTERM
//
// Windows: ADLX (the AMD Device Library eXtra, amdadlx64.dll, shipped with Adrenalin; vendor/adlx is the SDK clone,
// MIT) for the metrics, keyed by the card's PCI bus from SetupAPI (the display class, matched by the same name ADLX
// reports). Without ADLX (no AMD driver, an old one, or the DLL missing) only the utilisation is read, from the
// GPU Engine performance counters through PDH, keyed by the adapter LUID that Windows uses there.
// Linux: the amdgpu sysfs (/sys/class/drm/card*/device: hwmon power1_average, temp1_input, fan1_input, pwm1,
// pp_dpm_mclk, gpu_busy_percent), keyed by the PCI address of the device link.
//
// Line format (space separated, every field present, a value the source cannot give prints as -):
// amd <ordinal> bus <pci bus or address> kind integrated|discrete name "<name>" watts <W> temp_c <C> fan_rpm <rpm>
// fan_pct <%> mclk_mhz <MHz> gclk_mhz <MHz> util_pct <%> source adlx|sysfs|perfcounter
// then one `end <ms>` line per sample. The app (engine.rs amd_telemetry_line) parses it; parsers are unit-tested
// against lines captured on PC 1.
#define _CRT_SECURE_NO_WARNINGS
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#include <signal.h>
static volatile int gStop = 0;
static void onSignal(int s) { (void)s; gStop = 1; }
typedef struct {
char bus[64];
char kind[16];
char name[128];
double watts, tempC, fanRpm, fanPct, mclk, gclk, util; /* -1 = not available */
const char* source;
} Sample;
static void sampleInit(Sample* s) { memset(s, 0, sizeof(*s)); strcpy(s->bus, "-"); strcpy(s->kind, "-"); strcpy(s->name, "-"); s->watts = s->tempC = s->fanRpm = s->fanPct = s->mclk = s->gclk = s->util = -1.0; s->source = "-"; }
static void printNum(double v, const char* fmt) { if (v < 0) printf(" -"); else printf(fmt, v); }
static void printSample(int ordinal, const Sample* s) {
printf("amd %d bus %s kind %s name \"%s\" watts", ordinal, s->bus, s->kind, s->name);
printNum(s->watts, " %.1f"); printf(" temp_c"); printNum(s->tempC, " %.1f"); printf(" fan_rpm"); printNum(s->fanRpm, " %.0f");
printf(" fan_pct"); printNum(s->fanPct, " %.0f"); printf(" mclk_mhz"); printNum(s->mclk, " %.0f"); printf(" gclk_mhz"); printNum(s->gclk, " %.0f");
printf(" util_pct"); printNum(s->util, " %.0f"); printf(" source %s\n", s->source);
}
#ifdef _WIN32
#define WIN32_LEAN_AND_MEAN
#include <windows.h>
#include <setupapi.h>
#include <pdh.h>
#include "../vendor/adlx/SDK/ADLXHelper/Windows/C/ADLXHelper.h"
#include "../vendor/adlx/SDK/Include/IPerformanceMonitoring.h"
/* The SDK declares these three and leaves them to the platform file of each sample. */
adlx_handle ADLX_CDECL_CALL adlx_load_library(const TCHAR* filename) { return (adlx_handle)LoadLibrary(filename); }
int ADLX_CDECL_CALL adlx_free_library(adlx_handle module) { return FreeLibrary((HMODULE)module) ? 1 : 0; }
void* ADLX_CDECL_CALL adlx_get_proc_address(adlx_handle module, const char* procName) { return (void*)GetProcAddress((HMODULE)module, procName); }
static double nowMs(void) { LARGE_INTEGER f, c; QueryPerformanceFrequency(&f); QueryPerformanceCounter(&c); return (double)c.QuadPart * 1000.0 / (double)f.QuadPart; }
/* The PCI bus of every display-class device, by its name (SetupAPI; the names are the ones ADLX reports). */
typedef struct { char name[128]; int bus; } BusEntry;
static int listBuses(BusEntry* out, int cap) {
static const GUID DISPLAY = { 0x4d36e968, 0xe325, 0x11ce, { 0xbf, 0xc1, 0x08, 0x00, 0x2b, 0xe1, 0x03, 0x18 } };
HDEVINFO set = SetupDiGetClassDevsA(&DISPLAY, NULL, NULL, DIGCF_PRESENT);
SP_DEVINFO_DATA d;
DWORD i;
int n = 0;
if (set == INVALID_HANDLE_VALUE) return 0;
d.cbSize = sizeof(d);
for (i = 0; SetupDiEnumDeviceInfo(set, i, &d) && n < cap; ++i) {
char name[128] = { 0 };
DWORD bus = 0, type = 0, got = 0;
if (!SetupDiGetDeviceRegistryPropertyA(set, &d, SPDRP_DEVICEDESC, &type, (BYTE*)name, sizeof(name) - 1, &got)) continue;
if (!SetupDiGetDeviceRegistryPropertyA(set, &d, SPDRP_BUSNUMBER, &type, (BYTE*)&bus, sizeof(bus), &got)) continue;
snprintf(out[n].name, sizeof(out[n].name), "%s", name); out[n].bus = (int)bus; ++n;
}
SetupDiDestroyDeviceInfoList(set);
return n;
}
static int busOf(const BusEntry* b, int n, const char* name, int* taken) {
int i;
for (i = 0; i < n; ++i) if (!taken[i] && strcmp(b[i].name, name) == 0) { taken[i] = 1; return b[i].bus; }
return -1;
}
/* ADLX: one sample of every GPU. Returns the number of lines printed, -1 when ADLX is not usable (reason printed). */
static IADLXSystem* gSys = NULL;
static IADLXPerformanceMonitoringServices* gPerf = NULL;
static int adlxOpen(void) {
ADLX_RESULT r = ADLXHelper_Initialize();
if (!ADLX_SUCCEEDED(r)) { printf("info adlx: ADLXHelper_Initialize returned %d (no AMD driver with ADLX; amdadlx64.dll missing or too old)\n", (int)r); return 0; }
gSys = ADLXHelper_GetSystemServices();
if (!gSys) { printf("info adlx: no system services\n"); return 0; }
r = gSys->pVtbl->GetPerformanceMonitoringServices(gSys, &gPerf);
if (!ADLX_SUCCEEDED(r) || !gPerf) { printf("info adlx: GetPerformanceMonitoringServices returned %d\n", (int)r); return 0; }
return 1;
}
static int adlxSample(const BusEntry* buses, int nBuses) {
IADLXGPUList* gpus = NULL;
adlx_uint it;
int ordinal = 0;
int taken[32] = { 0 };
ADLX_RESULT r = gSys->pVtbl->GetGPUs(gSys, &gpus);
if (!ADLX_SUCCEEDED(r) || !gpus) { printf("info adlx: GetGPUs returned %d\n", (int)r); return 0; }
for (it = gpus->pVtbl->Begin(gpus); it != gpus->pVtbl->End(gpus); ++it) {
IADLXGPU* gpu = NULL;
IADLXGPUMetrics* m = NULL;
Sample s;
const char* name = NULL;
ADLX_GPU_TYPE type = GPUTYPE_UNDEFINED;
adlx_double dv = 0; adlx_int iv = 0;
if (!ADLX_SUCCEEDED(gpus->pVtbl->At_GPUList(gpus, it, &gpu)) || !gpu) continue;
sampleInit(&s);
s.source = "adlx";
if (ADLX_SUCCEEDED(gpu->pVtbl->Name(gpu, &name)) && name) snprintf(s.name, sizeof(s.name), "%s", name);
if (ADLX_SUCCEEDED(gpu->pVtbl->Type(gpu, &type))) strcpy(s.kind, type == GPUTYPE_INTEGRATED ? "integrated" : type == GPUTYPE_DISCRETE ? "discrete" : "-");
{ int b = busOf(buses, nBuses, s.name, taken); if (b >= 0) snprintf(s.bus, sizeof(s.bus), "%d", b); }
r = gPerf->pVtbl->GetCurrentGPUMetrics(gPerf, gpu, &m);
if (ADLX_SUCCEEDED(r) && m) {
if (ADLX_SUCCEEDED(m->pVtbl->GPUPower(m, &dv))) s.watts = dv;
if (s.watts < 0 && ADLX_SUCCEEDED(m->pVtbl->GPUTotalBoardPower(m, &dv))) s.watts = dv;
if (ADLX_SUCCEEDED(m->pVtbl->GPUTemperature(m, &dv))) s.tempC = dv;
if (ADLX_SUCCEEDED(m->pVtbl->GPUFanSpeed(m, &iv))) s.fanRpm = iv;
if (ADLX_SUCCEEDED(m->pVtbl->GPUVRAMClockSpeed(m, &iv))) s.mclk = iv;
if (ADLX_SUCCEEDED(m->pVtbl->GPUClockSpeed(m, &iv))) s.gclk = iv;
if (ADLX_SUCCEEDED(m->pVtbl->GPUUsage(m, &dv))) s.util = dv;
m->pVtbl->Release(m);
} else {
printf("info adlx: GetCurrentGPUMetrics for \"%s\" returned %d\n", s.name, (int)r);
}
/* fan percent: ADLX gives rpm only here; the tuning interface has the range, the app shows rpm when pct is - */
printSample(ordinal++, &s);
gpu->pVtbl->Release(gpu);
}
gpus->pVtbl->Release(gpus);
return ordinal;
}
/* PDH fallback: GPU engine utilisation per adapter LUID, summed over the engines (no power, no temperature). */
static int pdhSample(void) {
PDH_HQUERY q = NULL;
PDH_HCOUNTER c = NULL;
DWORD size = 0, count = 0, i;
PDH_FMT_COUNTERVALUE_ITEM_A* items;
int ordinal = 0;
if (PdhOpenQueryA(NULL, 0, &q) != ERROR_SUCCESS) { printf("info perfcounter: PdhOpenQuery failed\n"); return 0; }
if (PdhAddEnglishCounterA(q, "\\GPU Engine(*)\\Utilization Percentage", 0, &c) != ERROR_SUCCESS) { printf("info perfcounter: no GPU Engine counters\n"); PdhCloseQuery(q); return 0; }
PdhCollectQueryData(q); Sleep(1000); PdhCollectQueryData(q);
PdhGetFormattedCounterArrayA(c, PDH_FMT_DOUBLE, &size, &count, NULL);
items = (PDH_FMT_COUNTERVALUE_ITEM_A*)malloc(size ? size : 1);
if (PdhGetFormattedCounterArrayA(c, PDH_FMT_DOUBLE, &size, &count, items) == ERROR_SUCCESS) {
/* instance names: pid_1234_luid_0x00000000_0x0000D4E3_phys_0_eng_0_engtype_3D; sum per luid */
char luids[16][40]; double sums[16]; int n = 0, k;
for (i = 0; i < count; ++i) {
const char* p = strstr(items[i].szName, "luid_");
char luid[40];
if (!p) continue;
snprintf(luid, sizeof(luid), "%.39s", p); { char* e = strstr(luid, "_phys"); if (e) *e = 0; }
for (k = 0; k < n; ++k) if (strcmp(luids[k], luid) == 0) break;
if (k == n && n < 16) { strcpy(luids[n], luid); sums[n] = 0; ++n; }
if (k < 16) sums[k] += items[i].FmtValue.doubleValue;
}
for (k = 0; k < n; ++k) {
Sample s; sampleInit(&s); s.source = "perfcounter";
snprintf(s.bus, sizeof(s.bus), "%s", luids[k]);
s.util = sums[k] > 100.0 ? 100.0 : sums[k];
printSample(ordinal++, &s);
}
}
free(items);
PdhCloseQuery(q);
return ordinal;
}
int main(int argc, char** argv) {
int every = 0, i, haveAdlx;
BusEntry buses[32];
int nBuses;
for (i = 1; i < argc; ++i) if (strcmp(argv[i], "-l") == 0 && i + 1 < argc) every = atoi(argv[++i]);
signal(SIGINT, onSignal); signal(SIGTERM, onSignal);
setvbuf(stdout, NULL, _IOLBF, 0);
nBuses = listBuses(buses, 32);
for (i = 0; i < nBuses; ++i) printf("info display device \"%s\" bus %d\n", buses[i].name, buses[i].bus);
haveAdlx = adlxOpen();
do {
double t0 = nowMs();
int n = haveAdlx ? adlxSample(buses, nBuses) : pdhSample();
printf("end %.1f ms %d card(s)\n", nowMs() - t0, n);
fflush(stdout); /* a redirected stdout is fully buffered on the Windows CRT whatever setvbuf asks (PC 1 lost 60 s of samples at the kill) */
if (every > 0) Sleep((DWORD)every * 1000);
} while (every > 0 && !gStop);
if (haveAdlx) { if (gPerf) gPerf->pVtbl->Release(gPerf); ADLXHelper_Terminate(); }
return 0;
}
#else
#include <dirent.h>
#include <unistd.h>
#include <time.h>
static double nowMs(void) { struct timespec ts; clock_gettime(CLOCK_MONOTONIC, &ts); return ts.tv_sec * 1000.0 + ts.tv_nsec / 1e6; }
static int readText(const char* path, char* out, size_t cap) { FILE* f = fopen(path, "r"); size_t n; if (!f) return 0; n = fread(out, 1, cap - 1, f); fclose(f); out[n] = 0; return 1; }
static double readNumber(const char* path) { char b[64]; if (!readText(path, b, sizeof(b))) return -1.0; return atof(b); }
/* pp_dpm_mclk: lines "0: 96Mhz", "3: 1258Mhz *"; the starred line is the current state */
static double dpmCurrent(const char* text) {
const char* p = text;
while (p && *p) {
const char* nl = strchr(p, '\n');
size_t len = nl ? (size_t)(nl - p) : strlen(p);
const char* star = memchr(p, '*', len);
if (star) { const char* colon = memchr(p, ':', len); if (colon) return atof(colon + 1); }
p = nl ? nl + 1 : NULL;
}
return -1.0;
}
static int sysfsSample(const char* root) {
DIR* d = opendir(root);
struct dirent* e;
int ordinal = 0;
if (!d) { printf("info sysfs: no %s\n", root); return 0; }
while ((e = readdir(d)) != NULL) {
char dev[512], path[640], text[4096], link[512];
ssize_t ln;
Sample s;
DIR* hw; struct dirent* he;
if (strncmp(e->d_name, "card", 4) != 0 || strchr(e->d_name + 4, '-')) continue;
snprintf(dev, sizeof(dev), "%s/%s/device", root, e->d_name);
snprintf(path, sizeof(path), "%s/vendor", dev);
if (!readText(path, text, sizeof(text)) || strtol(text, NULL, 16) != 0x1002) continue;
sampleInit(&s);
s.source = "sysfs";
ln = readlink(dev, link, sizeof(link) - 1);
if (ln > 0) { link[ln] = 0; { const char* base = strrchr(link, '/'); snprintf(s.bus, sizeof(s.bus), "%.63s", base ? base + 1 : link); } }
snprintf(path, sizeof(path), "%s/product_name", dev);
if (readText(path, text, sizeof(text))) { text[strcspn(text, "\n")] = 0; snprintf(s.name, sizeof(s.name), "%s", text); }
else { snprintf(path, sizeof(path), "%s/device", dev); if (readText(path, text, sizeof(text))) { text[strcspn(text, "\n")] = 0; snprintf(s.name, sizeof(s.name), "amdgpu %s", text); } }
snprintf(path, sizeof(path), "%s/boot_vga", dev);
strcpy(s.kind, "discrete");
snprintf(path, sizeof(path), "%s/hwmon", dev);
hw = opendir(path);
if (hw) {
while ((he = readdir(hw)) != NULL) {
char hp[900];
if (strncmp(he->d_name, "hwmon", 5) != 0) continue;
snprintf(hp, sizeof(hp), "%s/%s/power1_average", path, he->d_name); s.watts = readNumber(hp); if (s.watts < 0) { snprintf(hp, sizeof(hp), "%s/%s/power1_input", path, he->d_name); s.watts = readNumber(hp); } if (s.watts >= 0) s.watts /= 1e6;
snprintf(hp, sizeof(hp), "%s/%s/temp1_input", path, he->d_name); s.tempC = readNumber(hp); if (s.tempC >= 0) s.tempC /= 1000.0;
snprintf(hp, sizeof(hp), "%s/%s/fan1_input", path, he->d_name); s.fanRpm = readNumber(hp);
{ double pwm, pwmMax; snprintf(hp, sizeof(hp), "%s/%s/pwm1", path, he->d_name); pwm = readNumber(hp); snprintf(hp, sizeof(hp), "%s/%s/pwm1_max", path, he->d_name); pwmMax = readNumber(hp); if (pwm >= 0 && pwmMax > 0) s.fanPct = 100.0 * pwm / pwmMax; else if (pwm >= 0) s.fanPct = 100.0 * pwm / 255.0; }
break;
}
closedir(hw);
}
snprintf(path, sizeof(path), "%s/pp_dpm_mclk", dev); if (readText(path, text, sizeof(text))) s.mclk = dpmCurrent(text);
snprintf(path, sizeof(path), "%s/pp_dpm_sclk", dev); if (readText(path, text, sizeof(text))) s.gclk = dpmCurrent(text);
snprintf(path, sizeof(path), "%s/gpu_busy_percent", dev); s.util = readNumber(path);
printSample(ordinal++, &s);
}
closedir(d);
return ordinal;
}
int main(int argc, char** argv) {
int every = 0, i;
const char* root = getenv("IGNEUM_DRM_ROOT") ? getenv("IGNEUM_DRM_ROOT") : "/sys/class/drm"; /* a fixture tree for tests */
for (i = 1; i < argc; ++i) if (strcmp(argv[i], "-l") == 0 && i + 1 < argc) every = atoi(argv[++i]);
signal(SIGINT, onSignal); signal(SIGTERM, onSignal);
setvbuf(stdout, NULL, _IOLBF, 0);
do {
double t0 = nowMs();
int n = sysfsSample(root);
printf("end %.1f ms %d card(s)\n", nowMs() - t0, n);
fflush(stdout); /* a redirected stdout is fully buffered on the Windows CRT whatever setvbuf asks (PC 1 lost 60 s of samples at the kill) */
if (every > 0) sleep((unsigned)every);
} while (every > 0 && !gStop);
return 0;
}
#endif