diff --git a/app/igneum-app/src/detect.rs b/app/igneum-app/src/detect.rs index 5af1581d0..c6c300dc1 100644 --- a/app/igneum-app/src/detect.rs +++ b/app/igneum-app/src/detect.rs @@ -104,7 +104,7 @@ pub fn apply_defaults(c: &mut CardState) { /// Marks whether the efficiency sweep (src/sweep.rs) can run on a card, with the reason when it cannot. Called /// once the power limits are known. pub fn mark_sweep_support(c: &mut CardState) { - match crate::sweep::unsupported_reason(&c.vendor, c.power_default_w, &c.device) { + match crate::sweep::unsupported_reason_for(&c.vendor, c.power_default_w, &c.device, c.amd_tunable) { None => { c.sweep_supported = true; if c.sweep_state.is_empty() || c.sweep_state == "unsupported" { diff --git a/app/igneum-app/src/engine.rs b/app/igneum-app/src/engine.rs index 480a8b7b9..fb12ffbf9 100644 --- a/app/igneum-app/src/engine.rs +++ b/app/igneum-app/src/engine.rs @@ -962,7 +962,12 @@ impl Engine { } } Cmd::SweepCapSet(text) => { - if !text.contains("All done") { + if text.starts_with("adlx ") { + self.shared.log(&format!("sweep: {}", short(&text, 220))); + if text.contains(" error ") { + self.sweep_abort(&format!("ADLX refused the setting ({})", short(&text, 160))); + } + } else if !text.contains("All done") { self.shared.log(&format!("sweep: nvidia-smi -pl answered: {}", short(&text.replace('\n', " "), 200))); } } @@ -1552,6 +1557,14 @@ impl Engine { } } if self.amd_telemetry.is_none() && now >= self.amd_telemetry_retry_at { + // what the cards' manual tuning allows, once per child start (the sweep's ranges and the stock max clock) + if let Some(out) = crate::detect::run_timeout(std::process::Command::new(&exe).arg("--tune"), None, Duration::from_secs(20)) { + for line in out.lines() { + if let Some(tu) = parse_amd_tune(line) { + self.amd_tune_line(&tu); + } + } + } let args: Vec = vec!["-l".into(), "5".into()]; let log = self.shared.runtime.log_dir.join(format!("gpu-amd-{}.log", self.stamp)); match procs::spawn(Source::AmdTelemetry, &exe, &args, None, &log, &self.lines_tx, &[]) { @@ -1582,6 +1595,9 @@ impl Engine { } let Some(idx) = target else { return }; let Some(c) = st.mining.cards.iter_mut().find(|c| c.index == idx) else { return }; + c.amd_ordinal = s.ordinal as i32; + c.amd_plimit_pct = s.plimit_pct.max(0.0); + c.amd_gmax_mhz = s.gmax_mhz.max(0.0); if s.watts > 0.0 { c.power_w = s.watts; } @@ -1593,6 +1609,49 @@ impl Engine { c.mclk_mhz = s.mclk_mhz.max(0.0); c.util_pct = s.util_pct.max(0.0); c.telemetry_at = crate::platform::unix_now_f(); + let draw = c.power_w; + drop(st); + // the sweep's draw samples, as the NVIDIA line feeds them + if let Some(run) = self.sweep.as_mut() { + if run.card == idx && draw > 0.0 { + run.sample_draw(draw); + } + } + } + + /// One `tune` line of igneum-gpu-telemetry --tune: the card it describes (matched as the sample lines are, by + /// the ordinal among the discrete AMD cards; integrated ones have no manual tuning) learns its ranges and whether + /// the sweep can run on it. + fn amd_tune_line(&mut self, tu: &AmdTune) { + let mut st = self.st(); + let mut nth = 0usize; + let mut target: Option = None; + for c in st.mining.cards.iter() { + if c.vendor != "amd" || c.kind != "discrete" { + continue; + } + if nth == tu.ordinal_in_kind { + target = Some(c.index); + break; + } + nth += 1; + } + let Some(idx) = target else { return }; + let Some(c) = st.mining.cards.iter_mut().find(|c| c.index == idx) else { return }; + c.amd_ordinal = tu.ordinal as i32; + c.amd_tunable = tu.plimit_range.is_some(); + if let Some((lo, hi)) = tu.plimit_range { + c.amd_plimit_min = lo; + c.amd_plimit_max = hi; + } + if let Some((lo, hi)) = tu.gmax_range { + c.amd_gmax_min = lo; + c.amd_gmax_max = hi; + } + if tu.gmax > 0.0 && c.amd_gmax_stock <= 0.0 { + c.amd_gmax_stock = tu.gmax; + } + crate::detect::mark_sweep_support(c); } /// "index, draw, gpu temp, mem temp, limit" every 5 s. @@ -1711,7 +1770,7 @@ impl Engine { let cards = self.st().mining.cards.clone(); let mut pick: Option<(usize, bool)> = None; for c in cards.iter() { - if !(c.enabled && c.sweep_supported && c.vendor == "nvidia") { + if !(c.enabled && c.sweep_supported && (c.vendor == "nvidia" || (c.vendor == "amd" && c.amd_tunable))) { continue; } let forced = self.sweep_queue.contains(&c.key); @@ -1764,6 +1823,11 @@ impl Engine { cc.sweep_state = "running".into(); cc.sweep_note = "sweep: checking how the cap is set".into(); } + if c.vendor == "amd" { + // ADLX sets the limit from this process with no elevation (igneum-gpu-telemetry --card N --set-plimit) + self.sweep_mode_known(Ok(true)); + return; + } if let Some(direct) = self.sweep_direct { self.sweep_mode_known(Ok(direct)); return; @@ -1791,36 +1855,41 @@ impl Engine { return; } }; - self.sweep_direct = Some(direct); let Some(c) = self.st().mining.cards.get(idx).cloned() else { self.sweep_pending = None; return; }; - if !direct && !self.sweep_helper { + let amd = c.vendor == "amd"; + if !amd { + self.sweep_direct = Some(direct); + } + if !amd && !direct && !self.sweep_helper { if let Err(e) = self.sweep_helper_start(&c) { self.sweep_abort(&format!("the elevated helper could not start: {e}")); return; } } - let steps = crate::sweep::plan_steps(c.power_default_w, c.power_min_w, c.power_max_w); + let lever = if amd { crate::sweep::Lever::AmdPowerPct } else { crate::sweep::Lever::NvidiaWatts }; + let steps = if amd { crate::sweep::plan_amd_power_steps(c.amd_plimit_min, c.amd_plimit_max) } else { crate::sweep::plan_steps(c.power_default_w, c.power_min_w, c.power_max_w) }; if steps.is_empty() { self.sweep_abort("no steps: the card reported no default power limit"); return; } let label = self.miners.iter().find(|m| m.card == idx).map(|m| m.label.clone()).unwrap_or_else(|| format!("card-{idx}")); - let before_w = if c.power_limit_w > 0.0 { c.power_limit_w } else { requested_watts(&c) }; + let before_w = if amd { if c.amd_plimit_pct > 0.0 { c.amd_plimit_pct } else { 100.0 } } else if c.power_limit_w > 0.0 { c.power_limit_w } else { requested_watts(&c) }; let now = Instant::now(); - let run = crate::sweep::Run::new(idx, &c.key, &c.device, &label, steps.clone(), c.power_pct, before_w, forced, crate::sweep::Timing::from_env(), now); + let run = crate::sweep::Run::new(idx, &c.key, &c.device, &label, steps.clone(), c.power_pct, before_w, forced, crate::sweep::Timing::from_env(), now).with_lever(lever); self.sweep_pending = None; self.sweep_say(&format!( - "SWEEP start card={label} name={} steps={} default={:.0} min={:.0} max={:.0} before={:.0} mode={}", + "SWEEP start card={label} name={} lever={} steps={} default={:.0} min={:.0} max={:.0} before={:.0} mode={}", c.name.replace(' ', "_"), + lever.name(), steps.iter().map(|s| s.pct.to_string()).collect::>().join(","), - c.power_default_w, - c.power_min_w, - c.power_max_w, + if amd { 100.0 } else { c.power_default_w }, + if amd { 100.0 + c.amd_plimit_min } else { c.power_min_w }, + if amd { 100.0 + c.amd_plimit_max } else { c.power_max_w }, before_w, - if direct { "direct" } else { "helper" } + if amd { "adlx" } else if direct { "direct" } else { "helper" } )); self.shared.event("info", &format!("{}: efficiency sweep started: {} caps from {}% down, {} s each on the live program", c.name, steps.len(), steps[0].pct, (run.timing.settle + run.timing.hold).as_secs())); self.sweep = Some(run); @@ -1864,6 +1933,25 @@ impl Engine { fn sweep_set_cap(&mut self, device: &str, watts: f64) { self.sweep_seq += 1; let w = watts.round() as u64; + let amd = self.sweep.as_ref().map(|r| r.lever != crate::sweep::Lever::NvidiaWatts).unwrap_or(false); + if amd { + // igneum-gpu-telemetry --card --set-plimit (or --set-gmax); the readback comes with the + // next sample line (plimit_pct / gmax_mhz, 5 s) + let (lever, idx) = self.sweep.as_ref().map(|r| (r.lever, r.card)).unwrap(); + let ordinal = self.st().mining.cards.get(idx).map(|c| c.amd_ordinal).unwrap_or(-1); + let Some(exe) = self.bins.telemetry.clone() else { return }; + let args: Vec = match lever { + crate::sweep::Lever::AmdPowerPct => vec!["--card".into(), ordinal.to_string(), "--set-plimit".into(), (watts.round() as i64 - 100).to_string()], + _ => vec!["--card".into(), ordinal.to_string(), "--set-gmax".into(), w.to_string()], + }; + let shared = self.shared.clone(); + std::thread::spawn(move || { + let out = crate::detect::run_timeout(std::process::Command::new(&exe).args(&args), None, Duration::from_secs(20)).unwrap_or_else(|| "igneum-gpu-telemetry did not answer".into()); + shared.send(Cmd::SweepCapSet(format!("adlx {}: {}", args.join(" "), out.lines().find(|l| l.starts_with("tune ")).unwrap_or("no tune line")))); + }); + self.shared.log(&format!("sweep: {} {} requested on AMD card {} (ordinal {ordinal})", lever.name(), if lever == crate::sweep::Lever::AmdPowerPct { format!("{w}%") } else { format!("{w} MHz") }, device)); + return; + } if self.sweep_direct == Some(true) { let smi = crate::platform::tool("nvidia-smi"); let device = device.to_string(); @@ -1906,13 +1994,18 @@ impl Engine { self.sweep_abort(&why); return; } - let outs = self.sweep.as_mut().map(|r| r.tick(now, c.power_limit_w)).unwrap_or_default(); - let label = self.sweep.as_ref().map(|r| r.label.clone()).unwrap_or_default(); + let readback = match self.sweep.as_ref().map(|r| r.lever).unwrap_or(crate::sweep::Lever::NvidiaWatts) { + crate::sweep::Lever::NvidiaWatts => c.power_limit_w, + crate::sweep::Lever::AmdPowerPct => c.amd_plimit_pct, + crate::sweep::Lever::AmdMaxClockMhz => c.amd_gmax_mhz, + }; + let outs = self.sweep.as_mut().map(|r| r.tick(now, readback)).unwrap_or_default(); + let (label, lever) = self.sweep.as_ref().map(|r| (r.label.clone(), r.lever)).unwrap_or((String::new(), crate::sweep::Lever::NvidiaWatts)); for o in outs { match o { crate::sweep::Out::Apply(w) => self.sweep_set_cap(&device, w), crate::sweep::Out::Row(row) => { - self.sweep_say(&row.line(&label)); + self.sweep_say(&row.line_for(&label, lever)); if let Some(cc) = self.st().mining.cards.get_mut(idx) { if row.usable() { cc.sweep_note = format!("sweep: {}% done ยท {:.3} MH/W", row.pct, row.eff); @@ -3371,11 +3464,15 @@ pub struct AmdTelemetry { pub mclk_mhz: f64, pub gclk_mhz: f64, pub util_pct: f64, + /// the limits in force (5 October 2026 tuning lines): the power limit as percent of default, the max GPU clock; + /// -1 when the line has no such field (an older helper) or the card has no manual tuning + pub plimit_pct: f64, + pub gmax_mhz: f64, pub source: String, } /// Parses `amd bus kind name "" watts temp_c fan_rpm fan_pct

mclk_mhz -/// gclk_mhz util_pct source `; a `-` value reads as -1.0. Anything else (info, end) gives None. +/// gclk_mhz util_pct [plimit_pct gmax_mhz ] source `; a `-` value reads as -1.0. Anything else (info, end) gives None. /// With one card per kind (the common case) `ordinal_in_kind` is 0 for the discrete card and 0 for the integrated /// one whatever their `amd N`; with several discrete cards the helper's order within the kind is kept: the rank is /// the number of earlier lines of the same kind, which the helper encodes by listing kinds contiguously (ADLX lists @@ -3405,12 +3502,14 @@ pub fn parse_amd_telemetry(line: &str) -> Option { let mclk_mhz = num("mclk_mhz")?; let gclk_mhz = num("gclk_mhz")?; let util_pct = num("util_pct")?; + let plimit_pct = num("plimit_pct").unwrap_or(-1.0); + let gmax_mhz = num("gmax_mhz").unwrap_or(-1.0); let source = tp.iter().position(|p| *p == "source").and_then(|i| tp.get(i + 1)).map(|s| s.to_string()).unwrap_or_default(); let kind = hp[5].to_string(); // the rank within the kind: the helper lists one integrated card at most and it comes first when present // (ADLX order on every PC seen so far), so a discrete card's rank is its ordinal minus the integrated ones before it let ordinal_in_kind = if kind == "discrete" && ordinal > 0 { ordinal - 1 } else if kind == "discrete" { 0 } else { 0 }; - Some(AmdTelemetry { ordinal, ordinal_in_kind, bus: hp[3].to_string(), kind, name: name.to_string(), watts, temp_c, fan_rpm, fan_pct, mclk_mhz, gclk_mhz, util_pct, source }) + Some(AmdTelemetry { ordinal, ordinal_in_kind, bus: hp[3].to_string(), kind, name: name.to_string(), watts, temp_c, fan_rpm, fan_pct, mclk_mhz, gclk_mhz, util_pct, plimit_pct, gmax_mhz, source }) } #[cfg(test)] @@ -3420,8 +3519,9 @@ mod amd_telemetry_tests { #[test] fn a_sysfs_line_from_the_fixture_parses() { // proto-opencl/gpu-telemetry.c on the Mac against a fixture tree, 5 October 2026 - let l = "amd 0 bus 0000:0c:00.0 kind discrete name \"AMD Radeon RX 9070 XT\" watts 287.0 temp_c 61.0 fan_rpm 1180 fan_pct 30 mclk_mhz 1258 gclk_mhz 2450 util_pct 90 source sysfs"; + let l = "amd 0 bus 0000:0c:00.0 kind discrete name \"AMD Radeon RX 9070 XT\" watts 287.0 temp_c 61.0 fan_rpm 1180 fan_pct 30 mclk_mhz 1258 gclk_mhz 2450 util_pct 90 plimit_pct 80 gmax_mhz 2450 source sysfs"; let s = parse_amd_telemetry(l).unwrap(); + assert_eq!((s.plimit_pct, s.gmax_mhz), (80.0, 2450.0)); assert_eq!((s.ordinal, s.ordinal_in_kind, s.bus.as_str(), s.kind.as_str(), s.name.as_str()), (0, 0, "0000:0c:00.0", "discrete", "AMD Radeon RX 9070 XT")); assert_eq!((s.watts, s.temp_c, s.fan_rpm, s.fan_pct, s.mclk_mhz, s.gclk_mhz, s.util_pct), (287.0, 61.0, 1180.0, 30.0, 1258.0, 2450.0, 90.0)); assert_eq!(s.source, "sysfs"); @@ -3429,9 +3529,11 @@ mod amd_telemetry_tests { #[test] fn a_dash_reads_as_unknown_and_other_lines_give_none() { - let l = "amd 1 bus 98 kind discrete name \"AMD Radeon RX 9070 XT\" watts 250.3 temp_c 58.0 fan_rpm 900 fan_pct - mclk_mhz 1258 gclk_mhz 2460 util_pct 97.5 source adlx"; + // PC 1, 20:27 UTC, the first helper build (no plimit_pct / gmax_mhz fields yet): they read as -1 + let l = "amd 1 bus 98 kind discrete name \"AMD Radeon RX 9070 XT\" watts 214.0 temp_c 64.0 fan_rpm 659 fan_pct - mclk_mhz 2505 gclk_mhz 3289 util_pct 100 source adlx"; let s = parse_amd_telemetry(l).unwrap(); assert_eq!(s.fan_pct, -1.0); + assert_eq!((s.plimit_pct, s.gmax_mhz, s.watts, s.mclk_mhz), (-1.0, -1.0, 214.0, 2505.0)); assert_eq!(s.ordinal_in_kind, 0, "the second line overall but the first discrete card after the integrated one"); assert!(parse_amd_telemetry("end 3.2 ms 2 card(s)").is_none()); assert!(parse_amd_telemetry("info adlx: ADLXHelper_Initialize returned 1").is_none()); @@ -3445,3 +3547,73 @@ mod amd_telemetry_tests { assert_eq!((s.watts, s.util_pct, s.source.as_str()), (-1.0, 100.0, "perfcounter")); } } + +/// One `tune` line of igneum-gpu-telemetry --tune: `tune N name "..." gmax X gmax_range MIN MAX plimit OFF +/// plimit_range MIN MAX factory 0|1 ok|error ...`; a `-` means the card has no such tuning. +#[derive(Debug, Clone, PartialEq)] +pub struct AmdTune { + pub ordinal: usize, + pub ordinal_in_kind: usize, + pub name: String, + pub gmax: f64, + pub gmax_range: Option<(f64, f64)>, + pub plimit: f64, + pub plimit_range: Option<(f64, f64)>, + pub factory: Option, + pub ok: bool, +} + +pub fn parse_amd_tune(line: &str) -> Option { + let line = line.trim(); + let rest = line.strip_prefix("tune ")?; + let (ord_s, rest) = rest.split_once(" name \"")?; + let ordinal: usize = ord_s.trim().parse().ok()?; + let (name, tail) = rest.split_once('"')?; + let tp: Vec<&str> = tail.split_whitespace().collect(); + let at = |key: &str| tp.iter().position(|p| *p == key); + let f = |s: Option<&&str>| -> Option { let v = *s?; if v == "-" { None } else { v.parse::().ok() } }; + let gi = at("gmax")?; + let gri = at("gmax_range")?; + let pi = at("plimit")?; + let pri = at("plimit_range")?; + let gmax = f(tp.get(gi + 1)).unwrap_or(-1.0); + let gmax_range = match (f(tp.get(gri + 1)), f(tp.get(gri + 2))) { (Some(a), Some(b)) => Some((a, b)), _ => None }; + let plimit = f(tp.get(pi + 1)).unwrap_or(0.0); + let plimit_range = match (f(tp.get(pri + 1)), f(tp.get(pri + 2))) { (Some(a), Some(b)) => Some((a, b)), _ => None }; + let factory = at("factory").and_then(|i| tp.get(i + 1)).and_then(|v| match *v { "1" => Some(true), "0" => Some(false), _ => None }); + let ok = tp.last().map(|v| *v == "ok").unwrap_or(false); + // integrated cards come first in ADLX order and carry no manual tuning: the rank among discrete cards is the + // ordinal minus the integrated ones before it; a tune line with no ranges at all is taken as integrated + let integrated_before = if ordinal > 0 && gmax_range.is_none() && plimit_range.is_none() { 0 } else { ordinal.min(1) }; + let _ = integrated_before; + let ordinal_in_kind = if ordinal > 0 { ordinal - 1 } else { 0 }; + Some(AmdTune { ordinal, ordinal_in_kind, name: name.to_string(), gmax, gmax_range, plimit, plimit_range, factory, ok }) +} + +#[cfg(test)] +mod amd_tune_tests { + use super::*; + + #[test] + fn a_tune_line_with_ranges_parses() { + // the shape igneum-gpu-telemetry --tune prints (ranges are the card's; a 9070 XT's own are owed, it left the bus) + let l = "tune 1 name \"AMD Radeon RX 9070 XT\" gmax 3300 gmax_range 500 3450 plimit 0 plimit_range -30 10 factory 1 ok"; + let tu = parse_amd_tune(l).unwrap(); + assert_eq!((tu.ordinal, tu.ordinal_in_kind, tu.name.as_str(), tu.gmax, tu.plimit), (1, 0, "AMD Radeon RX 9070 XT", 3300.0, 0.0)); + assert_eq!(tu.gmax_range, Some((500.0, 3450.0))); + assert_eq!(tu.plimit_range, Some((-30.0, 10.0))); + assert_eq!(tu.factory, Some(true)); + assert!(tu.ok); + } + + #[test] + fn an_integrated_card_without_tuning_and_an_error_verdict() { + let l = "tune 0 name \"AMD Radeon(TM) Graphics\" gmax - gmax_range - - plimit - plimit_range - - factory 1 ok"; + let tu = parse_amd_tune(l).unwrap(); + assert!(tu.gmax_range.is_none() && tu.plimit_range.is_none()); + let e = parse_amd_tune("tune 1 name \"x\" gmax 2800 gmax_range 500 3450 plimit -20 plimit_range -30 10 factory 0 error SetPowerLimit(-50) returned 1").unwrap(); + assert!(!e.ok); + assert_eq!(e.factory, Some(false)); + assert!(parse_amd_tune("end 1.0 ms 2 card(s)").is_none()); + } +} diff --git a/app/igneum-app/src/state.rs b/app/igneum-app/src/state.rs index 49bd09240..b11458d0e 100644 --- a/app/igneum-app/src/state.rs +++ b/app/igneum-app/src/state.rs @@ -79,6 +79,18 @@ pub struct CardState { pub fan_rpm: f64, pub mclk_mhz: f64, pub util_pct: f64, + /// the AMD card's line in the helper's output (`amd N`), -1 until a line matched; what --card N addresses + pub amd_ordinal: i32, + /// the AMD limits in force from the helper's line: the power limit as percent of default (100 = default), the max GPU clock + pub amd_plimit_pct: f64, + pub amd_gmax_mhz: f64, + /// from the helper's `tune` line: manual tuning present, the power offset range (percent) and the max clock range and stock + pub amd_tunable: bool, + pub amd_plimit_min: f64, + pub amd_plimit_max: f64, + pub amd_gmax_min: f64, + pub amd_gmax_max: f64, + pub amd_gmax_stock: f64, // hash per watt (src/sweep.rs) pub eff_mhw: f64, // live: hash_now over power_w, MH per watt; 0 = unknown pub sweep_supported: bool, // NVIDIA with readable limits; the note says why not otherwise diff --git a/app/igneum-app/src/sweep.rs b/app/igneum-app/src/sweep.rs index 52c0adbbf..99dbf97f7 100644 --- a/app/igneum-app/src/sweep.rs +++ b/app/igneum-app/src/sweep.rs @@ -16,8 +16,14 @@ //! SWEEP card=