0.3.25: the AMD core-clock and power-cap knob for Ember Tune (main's order of 8 October 2026: PC 1's RX 9070 XT measured 18.9 MH/s at 202 W and stopped, "no knob on AMD"; the card is latency-bound on the whole class v4 ladder, so a core-clock cap is the lever as on NVIDIA). Two findings behind the stop: the app's AMD lever (--tune, --set-gmax, --set-plimit, --reset in igneum-gpu-telemetry: ADLX IADLXManualGraphicsTuning2 and IADLXManualPowerTuning on Windows, pp_od_clk_voltage and hwmon power1_cap on Linux) was built on 5 October (620296b2) and never left branch opencl-rdna4-telemetry, so the kit's exe answers no tune line and every AMD tune fell to "measure only"; and the 9070 XT's max clock is an OFFSET range (gmax 0, gmax_range -500 1000), which the engine read as "no clock knob". Now: proto-opencl/gpu-telemetry.c takes 620296b2's tool whole (a superset of the shipped one); src/ember.rs amd_knob turns the tune line and the card's stock clock under load into the knob (clock ladder stock down to stock + gmax_min, power ladder on the percent scale 100 + plimit range; absolute-MHz drivers pass through), amd_limits, amd_gmax_arg (the apply sends the offset), and "not available (<reason>)" with nothing set for no AMD device, a tune line that read an error, Linux (root under /sys, a later cut), or a stock clock not yet known; Plan::full takes a narrow range (the floor above the 45 percent rung) as the fine ladder alone in 100 MHz steps, so the 9070 XT runs 2870, 2770, 2670, 2570, 2470 under a 2,970 MHz stock, the stop rule at the knee or a faulted row, lock_result and the lock_* fields as on NVIDIA. ADLX manual tuning needs no elevation (PC 1, 5 October 2026, an unelevated job), so the no-prompt rule holds with no Power Helper verb; nothing of the engine runs elevated. engine.rs carries the whole tune line in TuneProbe.amd_tune and the card's amd_stock_mhz and amd_gmax_offset. Tests known-failed first: the_amd_knob_reports_not_available_with_its_reason_and_sets_nothing, the_9070_xt_gets_a_power_ladder_and_a_clock_ladder_from_its_stock_clock, the_9070_xt_ladder_locks_at_the_knee (a declared ladder in the card's shape, not a measurement). Owed: the kit's igneum-gpu-telemetry.exe rebuilt from this source (MSVC on a PC or build-1, the ADLX SDK at vendor/adlx beside the tree), then the first measured grid on PC 1 by job when the hash lane's queue is clear
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
This commit is contained in:
parent
1411411e00
commit
d55dfd003f
5 changed files with 384 additions and 20 deletions
|
|
@ -102,6 +102,8 @@ fn card(index: usize, name: &str, vendor: &str, worker: &str, detail: &str, devi
|
|||
enabled: true,
|
||||
state: "off".into(),
|
||||
amd_ordinal: -1,
|
||||
amd_stock_mhz: 0,
|
||||
amd_gmax_offset: false,
|
||||
..Default::default()
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -336,11 +336,21 @@ impl Plan {
|
|||
power.push(Step { point: Point { clock_mhz: 0, power_pct: pct, mem_mhz: 0 }, watts: w, kind: Kind::Power });
|
||||
}
|
||||
}
|
||||
let clock_pcts = if limits.clock_max_mhz > 0 { CLOCK_STEPS_PCT[1..].to_vec() } else { Vec::new() };
|
||||
// a narrow range (the AMD knob, 0.3.25: the 9070 XT's floor sits 500 MHz under stock, above the 45 percent rung) takes
|
||||
// no percent rungs, only the fine ladder from one step under the maximum down to the floor
|
||||
let narrow = limits.clock_max_mhz > 0 && limits.clock_floor() > limits.clock_max_mhz * CLOCK_STEPS_PCT[CLOCK_STEPS_PCT.len() - 1] / 100;
|
||||
let clock_pcts = if limits.clock_max_mhz > 0 && !narrow { CLOCK_STEPS_PCT[1..].to_vec() } else { Vec::new() };
|
||||
// the fine ladder: from the last percent rung down to the floor in CLOCK_FINE_STEP_MHZ steps (the knob of
|
||||
// 7 October 2026; the stop rule in `next` ends it at the knee)
|
||||
let mut clock_fine = Vec::new();
|
||||
if limits.clock_max_mhz > 0 {
|
||||
if narrow {
|
||||
let floor = limits.clock_floor();
|
||||
let mut m = limits.clock_max_mhz.saturating_sub(CLOCK_FINE_STEP_MHZ);
|
||||
while m >= floor && m > 0 {
|
||||
clock_fine.push(m);
|
||||
m = m.saturating_sub(CLOCK_FINE_STEP_MHZ);
|
||||
}
|
||||
} else if limits.clock_max_mhz > 0 {
|
||||
let last_pct = limits.clamp_clock(limits.clock_max_mhz * CLOCK_STEPS_PCT[CLOCK_STEPS_PCT.len() - 1] / 100);
|
||||
let floor = limits.clock_floor();
|
||||
let mut m = (last_pct / CLOCK_FINE_STEP_MHZ) * CLOCK_FINE_STEP_MHZ;
|
||||
|
|
@ -1176,13 +1186,81 @@ pub fn lock_result(plan: &Plan, rows: &[Row], chosen: &Row) -> LockResult {
|
|||
LockResult { lock_mhz: chosen.point.clock_mhz, lock_mhs: chosen.mhs, lock_w: chosen.watts, lock_mhw: chosen.eff, unlocked_mhs, unlocked_w, lock_note: note }
|
||||
}
|
||||
|
||||
/// What igneum-gpu-telemetry's `tune` line says about an AMD card (src/engine.rs parse_amd_tune, the fields the knob
|
||||
/// needs): the max-clock value and range, the power-limit offset range, and whether the line read ok.
|
||||
#[derive(Clone, Debug, Default, PartialEq)]
|
||||
pub struct AmdTuneShape {
|
||||
pub ok: bool,
|
||||
pub error: String,
|
||||
pub gmax: f64,
|
||||
pub gmax_min: f64,
|
||||
pub gmax_max: f64,
|
||||
pub plimit_min: f64,
|
||||
pub plimit_max: f64,
|
||||
}
|
||||
|
||||
/// The AMD core-clock and power-cap knob (0.3.25, main's order of 8 October 2026: PC 1's RX 9070 XT measured 18.9 MH/s
|
||||
/// at 202 W and stopped, no lever). ADLX gives two levers, both offsets: the power limit in percent of the default
|
||||
/// (`plimit_range -30 10` on the 9070 XT) and the max GPU clock in MHz from the stock boost (`gmax 0 gmax_range -500
|
||||
/// 1000`); older drivers give the clock as absolute MHz. The knob carries the stock clock, so the ladder runs in
|
||||
/// absolute MHz like NVIDIA's and the apply sends the offset. Set from an unelevated process (ADLX manual tuning needs
|
||||
/// no elevation, confirmed on PC 1, 5 October 2026): the no-prompt rule holds with no Power Helper verb.
|
||||
#[derive(Clone, Debug, Default, PartialEq)]
|
||||
pub struct AmdKnob {
|
||||
/// the stock boost clock the offsets apply to (the card's clock under load when the probe answered); 0 = absolute MHz
|
||||
pub stock_mhz: u32,
|
||||
pub gmax_offset: bool,
|
||||
pub clock_max_mhz: u32,
|
||||
pub clock_min_mhz: u32,
|
||||
pub plimit_min: f64,
|
||||
pub plimit_max: f64,
|
||||
}
|
||||
|
||||
/// The knob, or "not available (<reason>)" and nothing set. Known-failed first on no AMD device.
|
||||
pub fn amd_knob(shape: Option<&AmdTuneShape>, gclk_now_mhz: f64, linux: bool) -> Result<AmdKnob, String> {
|
||||
let Some(t) = shape else { return Err("not available (no AMD device answered igneum-gpu-telemetry --tune: no card, the tool not next to the app, or no ADLX)".into()) };
|
||||
if !t.ok {
|
||||
return Err(format!("not available (igneum-gpu-telemetry: {})", if t.error.is_empty() { "the tune line read no ok" } else { t.error.as_str() }));
|
||||
}
|
||||
if linux {
|
||||
return Err("not available (Linux: the clock and power limits under /sys need root; the tune's helper takes them in a later cut)".into());
|
||||
}
|
||||
if t.plimit_max < t.plimit_min || t.plimit_min <= -100.0 {
|
||||
return Err(format!("not available (the power-limit range {} to {} percent is not usable)", t.plimit_min, t.plimit_max));
|
||||
}
|
||||
let offset = t.gmax_min < 0.0;
|
||||
let (clock_max, clock_min, stock) = if offset {
|
||||
let stock = gclk_now_mhz.round() as u32;
|
||||
if stock == 0 {
|
||||
return Err("not available (the max clock is an offset from the stock clock and the stock clock is not known until the card mines)".into());
|
||||
}
|
||||
(stock, (stock as f64 + t.gmax_min).max(1.0) as u32, stock)
|
||||
} else if t.gmax_max > 0.0 {
|
||||
(t.gmax_max as u32, if t.gmax_min > 0.0 { t.gmax_min as u32 } else { 0 }, 0)
|
||||
} else {
|
||||
(0, 0, 0)
|
||||
};
|
||||
Ok(AmdKnob { stock_mhz: stock, gmax_offset: offset, clock_max_mhz: clock_max, clock_min_mhz: clock_min, plimit_min: t.plimit_min, plimit_max: t.plimit_max })
|
||||
}
|
||||
|
||||
/// The plan's limits for the knob: the power ladder on the percent scale (default 100, floor 100 + plimit_min, ceiling
|
||||
/// 100 + plimit_max), the clock ladder from the stock clock down to stock + gmax_min.
|
||||
pub fn amd_limits(k: &AmdKnob) -> Limits {
|
||||
Limits { power_default_w: 100.0, power_min_w: 100.0 + k.plimit_min, power_max_w: 100.0 + k.plimit_max, clock_max_mhz: k.clock_max_mhz, clock_min_mhz: k.clock_min_mhz, mem_default_mhz: 0, mem_max_mhz: 0, power_ceiling_w: 0.0 }
|
||||
}
|
||||
|
||||
/// The `--set-gmax` argument for a clock cap: the offset from stock when the driver speaks offsets, else the MHz.
|
||||
pub fn amd_gmax_arg(clock_mhz: u32, stock_mhz: u32, offset: bool) -> i64 {
|
||||
if offset { clock_mhz as i64 - stock_mhz as i64 } else { clock_mhz as i64 }
|
||||
}
|
||||
|
||||
/// Why a card cannot be tuned beyond measuring, or None when both knobs are available.
|
||||
pub fn control_reason(vendor: &str, limits: &Limits, device: &str, power_control: bool, amd_helper: bool) -> Option<String> {
|
||||
match vendor {
|
||||
"nvidia" if limits.power_default_w <= 0.0 || device.is_empty() => Some("measure only: nvidia-smi did not report this card's limits".into()),
|
||||
"nvidia" if !power_control => Some("measure only until Power control is on in Settings (Windows asks for administrator rights once)".into()),
|
||||
"nvidia" => None,
|
||||
"amd" if !amd_helper => Some("measure only: igneum-gpu-telemetry gave no tune line for this card (not next to the app, an older helper, or no ADLX)".into()),
|
||||
"amd" if !amd_helper => Some("measure only: not available (igneum-gpu-telemetry gave no tune line for this card: not next to the app, an older helper, or no ADLX)".into()),
|
||||
"amd" if cfg!(target_os = "linux") => Some("measure only on Linux: the clock and power limits under /sys need root".into()),
|
||||
"amd" => None,
|
||||
"apple" => Some("measure only on Apple silicon: the system sets the clocks and the power; no control exposed".into()),
|
||||
|
|
@ -1194,6 +1272,73 @@ pub fn control_reason(vendor: &str, limits: &Limits, device: &str, power_control
|
|||
mod tests {
|
||||
use super::*;
|
||||
|
||||
/// The 9070 XT's tune line as PC 1 answered it (ember-tune-pc1-1, 6 October 2026, 22:30 UTC): offsets, not MHz.
|
||||
fn t9070() -> AmdTuneShape {
|
||||
AmdTuneShape { ok: true, error: String::new(), gmax: 0.0, gmax_min: -500.0, gmax_max: 1000.0, plimit_min: -30.0, plimit_max: 10.0 }
|
||||
}
|
||||
|
||||
/// Known-failed first (0.3.25, main's order of 8 October 2026): no AMD device means "not available (<reason>)" and
|
||||
/// nothing set; the 9070 XT's offset range used to close the clock knob (0.3.20 to 0.3.24: `clock_max_mhz: 0` when
|
||||
/// gmax_min < 0), so the card measured one point and stopped.
|
||||
#[test]
|
||||
fn the_amd_knob_reports_not_available_with_its_reason_and_sets_nothing() {
|
||||
let e = amd_knob(None, 2970.0, false).unwrap_err();
|
||||
assert!(e.starts_with("not available (no AMD device"), "{e}");
|
||||
let bad = AmdTuneShape { ok: false, error: "no tuning services (ADLX too old, or no AMD driver)".into(), ..Default::default() };
|
||||
let e = amd_knob(Some(&bad), 2970.0, false).unwrap_err();
|
||||
assert_eq!(e, "not available (igneum-gpu-telemetry: no tuning services (ADLX too old, or no AMD driver))");
|
||||
let e = amd_knob(Some(&t9070()), 0.0, false).unwrap_err();
|
||||
assert!(e.contains("stock clock is not known until the card mines"), "{e}");
|
||||
let e = amd_knob(Some(&t9070()), 2970.0, true).unwrap_err();
|
||||
assert!(e.starts_with("not available (Linux"), "{e}");
|
||||
assert!(control_reason("amd", &Limits::default(), "amd:gfx1201", true, false).unwrap().contains("not available ("));
|
||||
}
|
||||
|
||||
/// The 9070 XT under load at 2,970 MHz (the shape of its boost clock, not a measurement): the plan carries the power
|
||||
/// ladder on the percent scale and the clock ladder from stock down to stock minus 500 in 100 MHz steps.
|
||||
#[test]
|
||||
fn the_9070_xt_gets_a_power_ladder_and_a_clock_ladder_from_its_stock_clock() {
|
||||
let k = amd_knob(Some(&t9070()), 2970.0, false).unwrap();
|
||||
assert_eq!(k, AmdKnob { stock_mhz: 2970, gmax_offset: true, clock_max_mhz: 2970, clock_min_mhz: 2470, plimit_min: -30.0, plimit_max: 10.0 });
|
||||
let l = amd_limits(&k);
|
||||
assert_eq!((l.power_default_w, l.power_min_w, l.power_max_w), (100.0, 70.0, 110.0));
|
||||
assert_eq!(l.clock_floor(), 2470, "the floor is the range's, not 20 percent of the stock");
|
||||
let p = Plan::full(&l, Point { clock_mhz: 0, power_pct: 100, mem_mhz: 0 }, 1.0);
|
||||
let pw: Vec<u32> = p.power.iter().map(|s| s.point.power_pct).collect();
|
||||
assert!(pw.contains(&100) && pw.contains(&70) && !pw.contains(&60), "{pw:?}: the AMD power ladder stops at the range floor");
|
||||
let clocks: Vec<u32> = p.clock_pcts.iter().map(|pct| l.clamp_clock(l.clock_max_mhz * pct / 100)).chain(p.clock_fine.iter().copied()).collect();
|
||||
assert_eq!(clocks, vec![2870, 2770, 2670, 2570, 2470], "a narrow range: no percent rungs (they would all clamp to the floor), the fine ladder in 100 MHz steps to the floor");
|
||||
assert_eq!(amd_gmax_arg(2670, 2970, true), -300, "the apply sends the offset");
|
||||
assert_eq!(amd_gmax_arg(2670, 0, false), 2670, "an absolute driver gets the MHz");
|
||||
// an absolute range (older drivers): MHz straight through
|
||||
let abs = AmdTuneShape { ok: true, gmax: 2400.0, gmax_min: 500.0, gmax_max: 2600.0, plimit_min: -20.0, plimit_max: 15.0, ..Default::default() };
|
||||
let k = amd_knob(Some(&abs), 0.0, false).unwrap();
|
||||
assert_eq!((k.stock_mhz, k.gmax_offset, k.clock_max_mhz, k.clock_min_mhz), (0, false, 2600, 500));
|
||||
}
|
||||
|
||||
/// A declared ladder in the 9070 XT's shape (18.9 MH/s at 202 W unlocked; flat to the knee, then the rate falls): the
|
||||
/// knob's result names the lock, the unlocked row and the stop reason, as on NVIDIA.
|
||||
#[test]
|
||||
fn the_9070_xt_ladder_locks_at_the_knee() {
|
||||
let k = amd_knob(Some(&t9070()), 2970.0, false).unwrap();
|
||||
let l = amd_limits(&k);
|
||||
let p = Plan::full(&l, Point { clock_mhz: 0, power_pct: 100, mem_mhz: 0 }, 1.0);
|
||||
// the power ladder first (100 to 70 percent on the AMD scale; 70 is the knee), then the clock ladder at the cap point
|
||||
let mut rows = Vec::new();
|
||||
for (pct, w, mhs) in [(100, 202.0, 18.9), (90, 190.0, 18.9), (80, 175.0, 18.85), (70, 160.0, 17.5)] {
|
||||
rows.push(row_at(Point { clock_mhz: 0, power_pct: pct, mem_mhz: 0 }, w, mhs));
|
||||
}
|
||||
assert_eq!(p.power.len(), 4, "{:?}", p.power.iter().map(|s| s.point.power_pct).collect::<Vec<_>>());
|
||||
for (mhz, w, mhs) in [(2870, 170.0, 18.85), (2770, 165.0, 18.85), (2670, 160.0, 18.8), (2570, 152.0, 17.5)] {
|
||||
rows.push(row_at(Point { clock_mhz: mhz, power_pct: 80, mem_mhz: 0 }, w, mhs));
|
||||
}
|
||||
let best = rows.iter().cloned().max_by(|a, b| a.eff.partial_cmp(&b.eff).unwrap()).unwrap();
|
||||
let r = lock_result(&p, &rows, &best);
|
||||
assert_eq!(r.lock_mhz, 2670, "{r:?}");
|
||||
assert!((r.unlocked_mhs - 18.85).abs() < 1e-9 && (r.unlocked_w - 175.0).abs() < 1e-9, "the cap point is the 80 percent row: {r:?}");
|
||||
assert_ne!(r.lock_note, "no lever", "{r:?}");
|
||||
}
|
||||
|
||||
/// PC 1's RTX 5090 (nvidia-smi, 4 and 5 October 2026): default 575 W, min 400 W, max 600 W; clocks.max.gr is
|
||||
/// read at the first tune (3,090 MHz is the shape used here, not a measurement).
|
||||
fn l5090() -> Limits {
|
||||
|
|
|
|||
|
|
@ -3132,11 +3132,10 @@ impl Engine {
|
|||
let t = out.lines().filter_map(parse_amd_tune).find(|t| ordinal < 0 || t.ordinal as i64 == ordinal);
|
||||
shared.send(Cmd::TuneProbe(idx, Ok(match t {
|
||||
// PC 1's 9070 XT (ember-tune-pc1-1, 22:30 UTC): `gmax 0 gmax_range -500 1000`, an OFFSET from
|
||||
// the stock clock, not MHz; a range with a negative floor is an offset range and the clock
|
||||
// knob stays closed until the stock clock is known (the power limit is the AMD lever), and
|
||||
// `plimit_range -30 10` bounds the power ladder (the percent scale rides power_* below)
|
||||
Some(t) if t.ok => TuneProbe { clock_max_mhz: if t.gmax_min >= 0.0 && t.gmax_max > 0.0 { t.gmax_max as u32 } else { 0 }, clock_min_mhz: if t.gmax_min > 0.0 { t.gmax_min as u32 } else { 0 }, driver, direct: true, amd_ordinal: t.ordinal as i64, plimit_min: t.plimit_min, plimit_max: t.plimit_max, mem_default_mhz: 0, mem_max_mhz: 0 },
|
||||
Some(t) => TuneProbe { clock_max_mhz: 0, clock_min_mhz: 0, driver, direct: false, amd_ordinal: t.ordinal as i64, ..Default::default() },
|
||||
// the stock clock, not MHz. 0.3.25 (src/ember.rs amd_knob): the line rides whole in `amd_tune`,
|
||||
// and the knob turns the offset range into a ladder from the card's stock clock under load
|
||||
Some(t) if t.ok => TuneProbe { clock_max_mhz: if t.gmax_min >= 0.0 && t.gmax_max > 0.0 { t.gmax_max as u32 } else { 0 }, clock_min_mhz: if t.gmax_min > 0.0 { t.gmax_min as u32 } else { 0 }, driver, direct: true, amd_ordinal: t.ordinal as i64, plimit_min: t.plimit_min, plimit_max: t.plimit_max, mem_default_mhz: 0, mem_max_mhz: 0, amd_tune: Some(crate::ember::AmdTuneShape { ok: true, error: String::new(), gmax: t.gmax, gmax_min: t.gmax_min, gmax_max: t.gmax_max, plimit_min: t.plimit_min, plimit_max: t.plimit_max }) },
|
||||
Some(t) => TuneProbe { clock_max_mhz: 0, clock_min_mhz: 0, driver, direct: false, amd_ordinal: t.ordinal as i64, amd_tune: Some(crate::ember::AmdTuneShape { ok: false, error: t.error.clone(), ..Default::default() }), ..Default::default() },
|
||||
None => TuneProbe { clock_max_mhz: 0, clock_min_mhz: 0, driver, direct: false, amd_ordinal: -1, ..Default::default() },
|
||||
})));
|
||||
});
|
||||
|
|
@ -3172,7 +3171,13 @@ impl Engine {
|
|||
let power_control = probe.direct || self.shared.settings.lock().unwrap().power_control || (cfg!(windows) && self.power_task_registered());
|
||||
// AMD's power limit is a percent offset from the default (ADLX): the plan's watts scale becomes a percent
|
||||
// scale (default 100, floor 100 + plimit_min, ceiling 100 + plimit_max), tune_apply sends pct - 100
|
||||
let mut limits = if c.vendor == "amd" && probe.direct && probe.plimit_max >= probe.plimit_min && probe.plimit_min > -100.0 {
|
||||
// 0.3.25: the AMD knob (src/ember.rs amd_knob): the limits from the tune line and the stock clock under load, or
|
||||
// "not available (<reason>)" and nothing set
|
||||
let amd_knob = if c.vendor == "amd" { Some(crate::ember::amd_knob(probe.amd_tune.as_ref(), c.gclk_mhz, cfg!(target_os = "linux"))) } else { None };
|
||||
let mut limits = if let Some(Ok(k)) = amd_knob.as_ref() {
|
||||
self.shared.log(&format!("tune: AMD knob for {}: stock {} MHz, clock {} to {} MHz{}, power {} to {} percent", c.name, k.stock_mhz, k.clock_min_mhz, k.clock_max_mhz, if k.gmax_offset { " (offsets)" } else { "" }, 100.0 + k.plimit_min, 100.0 + k.plimit_max));
|
||||
crate::ember::amd_limits(k)
|
||||
} else if c.vendor == "amd" && probe.direct && probe.plimit_max >= probe.plimit_min && probe.plimit_min > -100.0 {
|
||||
crate::ember::Limits { power_default_w: 100.0, power_min_w: 100.0 + probe.plimit_min, power_max_w: 100.0 + probe.plimit_max, clock_max_mhz: probe.clock_max_mhz, clock_min_mhz: probe.clock_min_mhz, mem_default_mhz: 0, mem_max_mhz: 0, power_ceiling_w: 0.0 }
|
||||
} else {
|
||||
crate::ember::Limits { power_default_w: c.power_default_w, power_min_w: c.power_min_w, power_max_w: c.power_max_w, clock_max_mhz: probe.clock_max_mhz, clock_min_mhz: probe.clock_min_mhz, mem_default_mhz: probe.mem_default_mhz, mem_max_mhz: probe.mem_max_mhz, power_ceiling_w: 0.0 }
|
||||
|
|
@ -3187,7 +3192,10 @@ impl Engine {
|
|||
}
|
||||
let control = match c.vendor.as_str() {
|
||||
"nvidia" => crate::ember::control_reason("nvidia", &limits, &c.device, power_control, false),
|
||||
"amd" => crate::ember::control_reason("amd", &limits, &c.device, power_control, probe.direct && probe.amd_ordinal >= 0),
|
||||
"amd" => match amd_knob.as_ref() {
|
||||
Some(Err(reason)) => Some(format!("measure only: {reason}")),
|
||||
_ => crate::ember::control_reason("amd", &limits, &c.device, power_control, probe.direct && probe.amd_ordinal >= 0),
|
||||
},
|
||||
v => crate::ember::control_reason(v, &limits, &c.device, power_control, false),
|
||||
};
|
||||
if c.vendor == "nvidia" {
|
||||
|
|
@ -3224,6 +3232,12 @@ impl Engine {
|
|||
if probe.amd_ordinal >= 0 {
|
||||
cc.amd_ordinal = probe.amd_ordinal;
|
||||
}
|
||||
if let Some(Ok(k)) = amd_knob.as_ref() {
|
||||
cc.clock_max_mhz = k.clock_max_mhz;
|
||||
cc.clock_min_mhz = k.clock_min_mhz;
|
||||
cc.amd_stock_mhz = k.stock_mhz;
|
||||
cc.amd_gmax_offset = k.gmax_offset;
|
||||
}
|
||||
cc.tune_control = control.is_none();
|
||||
}
|
||||
}
|
||||
|
|
@ -3462,7 +3476,8 @@ impl Engine {
|
|||
// the step's limit on the AMD scale is a percent (the probe's Limits); the offset is that minus 100
|
||||
let offset = if c.power_default_w <= 0.0 { step.watts.round() as i64 - 100 } else { step.point.power_pct as i64 - 100 };
|
||||
let unlocked = clock == 0 && offset == 0;
|
||||
let gmax = if clock > 0 { clock } else { c.clock_max_mhz };
|
||||
// 0.3.25: a clock cap goes as the offset from stock when the driver speaks offsets (the 9070 XT), else MHz
|
||||
let gmax = crate::ember::amd_gmax_arg(if clock > 0 { clock } else { c.clock_max_mhz }, c.amd_stock_mhz, c.amd_gmax_offset);
|
||||
std::thread::spawn(move || {
|
||||
let run = |args: &[&str]| crate::detect::run_timeout(std::process::Command::new(&exe).args(args), None, Duration::from_secs(20)).unwrap_or_else(|| "igneum-gpu-telemetry did not answer".into());
|
||||
let mut text = String::new();
|
||||
|
|
@ -3472,7 +3487,7 @@ impl Engine {
|
|||
ok &= out.lines().any(|l| l.starts_with("tune ") && l.contains(" ok"));
|
||||
text.push_str(out.trim());
|
||||
} else {
|
||||
if gmax > 0 {
|
||||
if gmax != 0 || clock > 0 {
|
||||
let out = run(&["--card", &n, "--set-gmax", &gmax.to_string()]);
|
||||
ok &= out.lines().any(|l| l.starts_with("tune ") && l.contains(" ok"));
|
||||
text.push_str(out.trim());
|
||||
|
|
@ -5567,6 +5582,8 @@ pub struct TuneProbe {
|
|||
/// Ember 2, NVIDIA: the memory clock under load and the vendor's maximum (nvidia-smi clocks.mem, clocks.max.mem)
|
||||
pub mem_default_mhz: u32,
|
||||
pub mem_max_mhz: u32,
|
||||
/// 0.3.25, AMD: the whole tune line for the knob (src/ember.rs amd_knob); None when no line came
|
||||
pub amd_tune: Option<crate::ember::AmdTuneShape>,
|
||||
}
|
||||
|
||||
/// One `tune` line of igneum-gpu-telemetry --tune:
|
||||
|
|
|
|||
|
|
@ -143,6 +143,9 @@ pub struct CardState {
|
|||
pub clock_cap_mhz: u32, // the cap in force (0 = unlocked)
|
||||
pub mem_cap_mhz: u32, // Ember 2: the memory clock set by the tune (0 = the driver's default)
|
||||
pub amd_ordinal: i64, // the `amd N` ordinal of igneum-gpu-telemetry (-1 = unknown)
|
||||
// 0.3.25 (src/ember.rs amd_knob): the stock clock the driver's max-clock offsets apply to, and whether it speaks offsets
|
||||
pub amd_stock_mhz: u32,
|
||||
pub amd_gmax_offset: bool,
|
||||
pub driver: String, // the driver version (nvidia-smi, or the worker's race line)
|
||||
/// driver-check (7 October 2026): the comparable version the OS reports (nvidia-smi's for NVIDIA, Windows'
|
||||
/// DriverVersion for AMD and Intel; empty = none found) and what the row says about it against the manifest's table
|
||||
|
|
|
|||
|
|
@ -3,6 +3,14 @@
|
|||
// what it drew (the app's draw, temperature and MH per watt line came from nvidia-smi only).
|
||||
//
|
||||
// igneum-gpu-telemetry [-l SECONDS] one sample (default), or one every SECONDS until stdin closes or SIGTERM
|
||||
// igneum-gpu-telemetry --tune what each card's manual tuning allows: the max GPU clock range and value,
|
||||
// the power limit range and value (percent offset from the default), one `tune` line per card
|
||||
// igneum-gpu-telemetry --card N --set-gmax MHZ | --set-plimit PCT | --reset
|
||||
// set the max GPU clock, the power limit (offset percent, 0 = default, -20 = 80%),
|
||||
// or every tuning value back to factory, on card N (the `amd N` ordinal); prints the
|
||||
// `tune` line read back after the change. ADLX manual tuning needs no elevation
|
||||
// (confirmed on PC 1, 5 October 2026, from an unelevated job). Linux: pp_od_clk_voltage
|
||||
// ("s 1 MHZ" then "c") and hwmon power1_cap (microwatts), which need root.
|
||||
//
|
||||
// Windows: ADLX (the AMD Device Library eXtra, amdadlx64.dll, shipped with Adrenalin; vendor/adlx is the SDK clone,
|
||||
// MIT) for the metrics, keyed by the card's PCI bus from SetupAPI (the display class, matched by the same name ADLX
|
||||
|
|
@ -13,7 +21,8 @@
|
|||
//
|
||||
// Line format (space separated, every field present, a value the source cannot give prints as -):
|
||||
// amd <ordinal> bus <pci bus or address> kind integrated|discrete name "<name>" watts <W> temp_c <C> fan_rpm <rpm>
|
||||
// fan_pct <%> mclk_mhz <MHz> gclk_mhz <MHz> util_pct <%> source adlx|sysfs|perfcounter
|
||||
// fan_pct <%> mclk_mhz <MHz> gclk_mhz <MHz> util_pct <%> plimit_pct <100 + offset> gmax_mhz <MHz> source adlx|sysfs|perfcounter
|
||||
// tune <ordinal> name "<name>" gmax <MHz> gmax_range <min> <max> plimit <offset %> plimit_range <min> <max> factory 0|1 ok|<error>
|
||||
// then one `end <ms>` line per sample. The app (engine.rs amd_telemetry_line) parses it; parsers are unit-tested
|
||||
// against lines captured on PC 1.
|
||||
#define _CRT_SECURE_NO_WARNINGS
|
||||
|
|
@ -30,16 +39,18 @@ typedef struct {
|
|||
char kind[16];
|
||||
char name[128];
|
||||
double watts, tempC, fanRpm, fanPct, mclk, gclk, util; /* -1 = not available */
|
||||
double plimitPct, gmax; /* the limits in force: 100 + the power offset; the max GPU clock */
|
||||
const char* source;
|
||||
} Sample;
|
||||
|
||||
static void sampleInit(Sample* s) { memset(s, 0, sizeof(*s)); strcpy(s->bus, "-"); strcpy(s->kind, "-"); strcpy(s->name, "-"); s->watts = s->tempC = s->fanRpm = s->fanPct = s->mclk = s->gclk = s->util = -1.0; s->source = "-"; }
|
||||
static void sampleInit(Sample* s) { memset(s, 0, sizeof(*s)); strcpy(s->bus, "-"); strcpy(s->kind, "-"); strcpy(s->name, "-"); s->watts = s->tempC = s->fanRpm = s->fanPct = s->mclk = s->gclk = s->util = s->plimitPct = s->gmax = -1.0; s->source = "-"; }
|
||||
static void printNum(double v, const char* fmt) { if (v < 0) printf(" -"); else printf(fmt, v); }
|
||||
static void printSample(int ordinal, const Sample* s) {
|
||||
printf("amd %d bus %s kind %s name \"%s\" watts", ordinal, s->bus, s->kind, s->name);
|
||||
printNum(s->watts, " %.1f"); printf(" temp_c"); printNum(s->tempC, " %.1f"); printf(" fan_rpm"); printNum(s->fanRpm, " %.0f");
|
||||
printf(" fan_pct"); printNum(s->fanPct, " %.0f"); printf(" mclk_mhz"); printNum(s->mclk, " %.0f"); printf(" gclk_mhz"); printNum(s->gclk, " %.0f");
|
||||
printf(" util_pct"); printNum(s->util, " %.0f"); printf(" source %s\n", s->source);
|
||||
printf(" util_pct"); printNum(s->util, " %.0f"); printf(" plimit_pct"); printNum(s->plimitPct, " %.0f"); printf(" gmax_mhz"); printNum(s->gmax, " %.0f");
|
||||
printf(" source %s\n", s->source);
|
||||
}
|
||||
|
||||
#ifdef _WIN32
|
||||
|
|
@ -49,6 +60,9 @@ static void printSample(int ordinal, const Sample* s) {
|
|||
#include <pdh.h>
|
||||
#include "../vendor/adlx/SDK/ADLXHelper/Windows/C/ADLXHelper.h"
|
||||
#include "../vendor/adlx/SDK/Include/IPerformanceMonitoring.h"
|
||||
#include "../vendor/adlx/SDK/Include/IGPUTuning.h"
|
||||
#include "../vendor/adlx/SDK/Include/IGPUManualGFXTuning.h"
|
||||
#include "../vendor/adlx/SDK/Include/IGPUManualPowerTuning.h"
|
||||
|
||||
/* The SDK declares these three and leaves them to the platform file of each sample. */
|
||||
adlx_handle ADLX_CDECL_CALL adlx_load_library(const TCHAR* filename) { return (adlx_handle)LoadLibrary(filename); }
|
||||
|
|
@ -86,6 +100,107 @@ static int busOf(const BusEntry* b, int n, const char* name, int* taken) {
|
|||
/* ADLX: one sample of every GPU. Returns the number of lines printed, -1 when ADLX is not usable (reason printed). */
|
||||
static IADLXSystem* gSys = NULL;
|
||||
static IADLXPerformanceMonitoringServices* gPerf = NULL;
|
||||
static IADLXGPUTuningServices* gTune = NULL; /* NULL when the driver has no tuning services */
|
||||
|
||||
/* The manual tuning interfaces of one GPU; either may be NULL (not supported on the card, or an iGPU). */
|
||||
typedef struct { IADLXManualGraphicsTuning2* gfx; IADLXManualPowerTuning* power; } Tuning;
|
||||
static Tuning tuningOf(IADLXGPU* gpu) {
|
||||
Tuning tu = { NULL, NULL };
|
||||
adlx_bool ok = 0;
|
||||
IADLXInterface* raw = NULL;
|
||||
if (!gTune) return tu;
|
||||
if (ADLX_SUCCEEDED(gTune->pVtbl->IsSupportedManualGFXTuning(gTune, gpu, &ok)) && ok && ADLX_SUCCEEDED(gTune->pVtbl->GetManualGFXTuning(gTune, gpu, &raw)) && raw) {
|
||||
if (!ADLX_SUCCEEDED(raw->pVtbl->QueryInterface(raw, IID_IADLXManualGraphicsTuning2(), (void**)&tu.gfx))) tu.gfx = NULL;
|
||||
raw->pVtbl->Release(raw); raw = NULL;
|
||||
}
|
||||
if (ADLX_SUCCEEDED(gTune->pVtbl->IsSupportedManualPowerTuning(gTune, gpu, &ok)) && ok && ADLX_SUCCEEDED(gTune->pVtbl->GetManualPowerTuning(gTune, gpu, &raw)) && raw) {
|
||||
if (!ADLX_SUCCEEDED(raw->pVtbl->QueryInterface(raw, IID_IADLXManualPowerTuning(), (void**)&tu.power))) tu.power = NULL;
|
||||
raw->pVtbl->Release(raw);
|
||||
}
|
||||
return tu;
|
||||
}
|
||||
static void tuningRelease(Tuning* tu) { if (tu->gfx) tu->gfx->pVtbl->Release(tu->gfx); if (tu->power) tu->power->pVtbl->Release(tu->power); tu->gfx = NULL; tu->power = NULL; }
|
||||
|
||||
/* The limits in force, for a sample line (-1 when the card has no manual tuning). */
|
||||
static void tuningNow(IADLXGPU* gpu, double* plimitPct, double* gmax) {
|
||||
Tuning tu = tuningOf(gpu);
|
||||
adlx_int v = 0;
|
||||
*plimitPct = -1.0; *gmax = -1.0;
|
||||
if (tu.power && ADLX_SUCCEEDED(tu.power->pVtbl->GetPowerLimit(tu.power, &v))) *plimitPct = 100.0 + v;
|
||||
if (tu.gfx && ADLX_SUCCEEDED(tu.gfx->pVtbl->GetGPUMaxFrequency(tu.gfx, &v))) *gmax = v;
|
||||
tuningRelease(&tu);
|
||||
}
|
||||
|
||||
/* One `tune` line: ranges, values, whether the card is at factory settings, and the verdict of the last request. */
|
||||
static void printTune(int ordinal, IADLXGPU* gpu, const char* verdict) {
|
||||
Tuning tu = tuningOf(gpu);
|
||||
const char* name = NULL;
|
||||
ADLX_IntRange r = { 0, 0, 0 };
|
||||
adlx_int v = 0;
|
||||
adlx_bool factory = 0;
|
||||
char buf[128] = "-";
|
||||
if (ADLX_SUCCEEDED(gpu->pVtbl->Name(gpu, &name)) && name) snprintf(buf, sizeof(buf), "%s", name);
|
||||
printf("tune %d name \"%s\"", ordinal, buf);
|
||||
if (tu.gfx && ADLX_SUCCEEDED(tu.gfx->pVtbl->GetGPUMaxFrequency(tu.gfx, &v)) && ADLX_SUCCEEDED(tu.gfx->pVtbl->GetGPUMaxFrequencyRange(tu.gfx, &r))) printf(" gmax %d gmax_range %d %d", (int)v, (int)r.minValue, (int)r.maxValue);
|
||||
else printf(" gmax - gmax_range - -");
|
||||
if (tu.power && ADLX_SUCCEEDED(tu.power->pVtbl->GetPowerLimit(tu.power, &v)) && ADLX_SUCCEEDED(tu.power->pVtbl->GetPowerLimitRange(tu.power, &r))) printf(" plimit %d plimit_range %d %d", (int)v, (int)r.minValue, (int)r.maxValue);
|
||||
else printf(" plimit - plimit_range - -");
|
||||
if (gTune && ADLX_SUCCEEDED(gTune->pVtbl->IsAtFactory(gTune, gpu, &factory))) printf(" factory %d", factory ? 1 : 0); else printf(" factory -");
|
||||
printf(" %s\n", verdict);
|
||||
tuningRelease(&tu);
|
||||
}
|
||||
|
||||
/* --card N with --set-gmax, --set-plimit or --reset: apply, then print the tune line read back. */
|
||||
static int tuneCard(int card, int setGmax, int gmax, int setPlimit, int plimit, int reset) {
|
||||
IADLXGPUList* gpus = NULL;
|
||||
adlx_uint it;
|
||||
int ordinal = 0, done = 0;
|
||||
if (!gTune) { printf("tune %d error no tuning services (ADLX too old, or no AMD driver)\n", card); return 2; }
|
||||
if (!ADLX_SUCCEEDED(gSys->pVtbl->GetGPUs(gSys, &gpus)) || !gpus) { printf("tune %d error GetGPUs failed\n", card); return 2; }
|
||||
for (it = gpus->pVtbl->Begin(gpus); it != gpus->pVtbl->End(gpus); ++it, ++ordinal) {
|
||||
IADLXGPU* gpu = NULL;
|
||||
char verdict[160] = "ok";
|
||||
ADLX_RESULT r = ADLX_OK;
|
||||
if (ordinal != card) continue;
|
||||
if (!ADLX_SUCCEEDED(gpus->pVtbl->At_GPUList(gpus, it, &gpu)) || !gpu) continue;
|
||||
if (reset) {
|
||||
r = gTune->pVtbl->ResetToFactory(gTune, gpu);
|
||||
if (!ADLX_SUCCEEDED(r)) snprintf(verdict, sizeof(verdict), "error ResetToFactory returned %d", (int)r);
|
||||
} else {
|
||||
Tuning tu = tuningOf(gpu);
|
||||
if (setGmax) {
|
||||
if (!tu.gfx) snprintf(verdict, sizeof(verdict), "error no manual GFX tuning on this card");
|
||||
else { r = tu.gfx->pVtbl->SetGPUMaxFrequency(tu.gfx, gmax); if (!ADLX_SUCCEEDED(r)) snprintf(verdict, sizeof(verdict), "error SetGPUMaxFrequency(%d) returned %d", gmax, (int)r); }
|
||||
}
|
||||
if (setPlimit && verdict[0] == 'o') {
|
||||
if (!tu.power) snprintf(verdict, sizeof(verdict), "error no manual power tuning on this card");
|
||||
else { r = tu.power->pVtbl->SetPowerLimit(tu.power, plimit); if (!ADLX_SUCCEEDED(r)) snprintf(verdict, sizeof(verdict), "error SetPowerLimit(%d) returned %d", plimit, (int)r); }
|
||||
}
|
||||
tuningRelease(&tu);
|
||||
}
|
||||
printTune(ordinal, gpu, verdict);
|
||||
done = verdict[0] == 'o';
|
||||
gpu->pVtbl->Release(gpu);
|
||||
}
|
||||
gpus->pVtbl->Release(gpus);
|
||||
if (!done && ordinal <= card) printf("tune %d error no such card (%d listed)\n", card, ordinal);
|
||||
return done ? 0 : 1;
|
||||
}
|
||||
static int tuneReport(void) {
|
||||
IADLXGPUList* gpus = NULL;
|
||||
adlx_uint it;
|
||||
int ordinal = 0;
|
||||
if (!gTune) { printf("info adlx: no tuning services\n"); return 1; }
|
||||
if (!ADLX_SUCCEEDED(gSys->pVtbl->GetGPUs(gSys, &gpus)) || !gpus) return 1;
|
||||
for (it = gpus->pVtbl->Begin(gpus); it != gpus->pVtbl->End(gpus); ++it, ++ordinal) {
|
||||
IADLXGPU* gpu = NULL;
|
||||
if (!ADLX_SUCCEEDED(gpus->pVtbl->At_GPUList(gpus, it, &gpu)) || !gpu) continue;
|
||||
printTune(ordinal, gpu, "ok");
|
||||
gpu->pVtbl->Release(gpu);
|
||||
}
|
||||
gpus->pVtbl->Release(gpus);
|
||||
return 0;
|
||||
}
|
||||
static int adlxOpen(void) {
|
||||
ADLX_RESULT r = ADLXHelper_Initialize();
|
||||
if (!ADLX_SUCCEEDED(r)) { printf("info adlx: ADLXHelper_Initialize returned %d (no AMD driver with ADLX; amdadlx64.dll missing or too old)\n", (int)r); return 0; }
|
||||
|
|
@ -93,6 +208,8 @@ static int adlxOpen(void) {
|
|||
if (!gSys) { printf("info adlx: no system services\n"); return 0; }
|
||||
r = gSys->pVtbl->GetPerformanceMonitoringServices(gSys, &gPerf);
|
||||
if (!ADLX_SUCCEEDED(r) || !gPerf) { printf("info adlx: GetPerformanceMonitoringServices returned %d\n", (int)r); return 0; }
|
||||
r = gSys->pVtbl->GetGPUTuningServices(gSys, &gTune);
|
||||
if (!ADLX_SUCCEEDED(r) || !gTune) { printf("info adlx: GetGPUTuningServices returned %d (no manual tuning)\n", (int)r); gTune = NULL; }
|
||||
return 1;
|
||||
}
|
||||
static int adlxSample(const BusEntry* buses, int nBuses) {
|
||||
|
|
@ -129,6 +246,7 @@ static int adlxSample(const BusEntry* buses, int nBuses) {
|
|||
printf("info adlx: GetCurrentGPUMetrics for \"%s\" returned %d\n", s.name, (int)r);
|
||||
}
|
||||
/* fan percent: ADLX gives rpm only here; the tuning interface has the range, the app shows rpm when pct is - */
|
||||
tuningNow(gpu, &s.plimitPct, &s.gmax);
|
||||
printSample(ordinal++, &s);
|
||||
gpu->pVtbl->Release(gpu);
|
||||
}
|
||||
|
|
@ -173,15 +291,32 @@ static int pdhSample(void) {
|
|||
}
|
||||
|
||||
int main(int argc, char** argv) {
|
||||
int every = 0, i, haveAdlx;
|
||||
int every = 0, i, haveAdlx, tune = 0, card = -1, setGmax = 0, gmax = 0, setPlimit = 0, plimit = 0, reset = 0;
|
||||
BusEntry buses[32];
|
||||
int nBuses;
|
||||
for (i = 1; i < argc; ++i) if (strcmp(argv[i], "-l") == 0 && i + 1 < argc) every = atoi(argv[++i]);
|
||||
for (i = 1; i < argc; ++i) {
|
||||
if (strcmp(argv[i], "-l") == 0 && i + 1 < argc) every = atoi(argv[++i]);
|
||||
else if (strcmp(argv[i], "--tune") == 0) tune = 1;
|
||||
else if (strcmp(argv[i], "--card") == 0 && i + 1 < argc) card = atoi(argv[++i]);
|
||||
else if (strcmp(argv[i], "--set-gmax") == 0 && i + 1 < argc) { setGmax = 1; gmax = atoi(argv[++i]); }
|
||||
else if (strcmp(argv[i], "--set-plimit") == 0 && i + 1 < argc) { setPlimit = 1; plimit = atoi(argv[++i]); }
|
||||
else if (strcmp(argv[i], "--reset") == 0) reset = 1;
|
||||
}
|
||||
signal(SIGINT, onSignal); signal(SIGTERM, onSignal);
|
||||
setvbuf(stdout, NULL, _IOLBF, 0);
|
||||
nBuses = listBuses(buses, 32);
|
||||
for (i = 0; i < nBuses; ++i) printf("info display device \"%s\" bus %d\n", buses[i].name, buses[i].bus);
|
||||
haveAdlx = adlxOpen();
|
||||
if (card >= 0 || tune) {
|
||||
int rc;
|
||||
if (!haveAdlx) { printf("tune %d error no ADLX\n", card); return 2; }
|
||||
rc = (card >= 0 && (setGmax || setPlimit || reset)) ? tuneCard(card, setGmax, gmax, setPlimit, plimit, reset) : tuneReport();
|
||||
fflush(stdout);
|
||||
if (gTune) gTune->pVtbl->Release(gTune);
|
||||
if (gPerf) gPerf->pVtbl->Release(gPerf);
|
||||
ADLXHelper_Terminate();
|
||||
return rc;
|
||||
}
|
||||
do {
|
||||
double t0 = nowMs();
|
||||
int n = haveAdlx ? adlxSample(buses, nBuses) : pdhSample();
|
||||
|
|
@ -189,7 +324,7 @@ int main(int argc, char** argv) {
|
|||
fflush(stdout); /* a redirected stdout is fully buffered on the Windows CRT whatever setvbuf asks (PC 1 lost 60 s of samples at the kill) */
|
||||
if (every > 0) Sleep((DWORD)every * 1000);
|
||||
} while (every > 0 && !gStop);
|
||||
if (haveAdlx) { if (gPerf) gPerf->pVtbl->Release(gPerf); ADLXHelper_Terminate(); }
|
||||
if (haveAdlx) { if (gTune) gTune->pVtbl->Release(gTune); if (gPerf) gPerf->pVtbl->Release(gPerf); ADLXHelper_Terminate(); }
|
||||
return 0;
|
||||
}
|
||||
#else
|
||||
|
|
@ -211,6 +346,54 @@ static double dpmCurrent(const char* text) {
|
|||
}
|
||||
return -1.0;
|
||||
}
|
||||
static int writeText(const char* path, const char* text) { FILE* f = fopen(path, "w"); if (!f) return 0; fputs(text, f); return fclose(f) == 0; }
|
||||
/* The device directory of the n-th amdgpu card under root (the `amd N` ordinal of the sample lines), or 0. */
|
||||
static int sysfsCardDir(const char* root, int card, char* dev, size_t cap) {
|
||||
DIR* d = opendir(root);
|
||||
struct dirent* e;
|
||||
int ordinal = 0, found = 0;
|
||||
if (!d) return 0;
|
||||
while ((e = readdir(d)) != NULL) {
|
||||
char path[640], text[64];
|
||||
if (strncmp(e->d_name, "card", 4) != 0 || strchr(e->d_name + 4, '-')) continue;
|
||||
snprintf(path, sizeof(path), "%s/%s/device/vendor", root, e->d_name);
|
||||
if (!readText(path, text, sizeof(text)) || strtol(text, NULL, 16) != 0x1002) continue;
|
||||
if (ordinal++ == card) { snprintf(dev, cap, "%s/%s/device", root, e->d_name); found = 1; break; }
|
||||
}
|
||||
closedir(d);
|
||||
return found;
|
||||
}
|
||||
/* --card N --set-gmax MHZ: pp_od_clk_voltage "s 1 MHZ" then "c"; --set-plimit PCT: hwmon power1_cap = default x (100 + PCT) / 100;
|
||||
* --reset: "r" then "c" and power1_cap = power1_cap_default. Root is needed for every write; the verdict says when it is not. */
|
||||
static int sysfsTune(const char* root, int card, int setGmax, int gmax, int setPlimit, int plimit, int reset) {
|
||||
char dev[640], path[900], text[4096], hw[700] = "";
|
||||
DIR* d;
|
||||
struct dirent* e;
|
||||
const char* verdict = "ok";
|
||||
if (!sysfsCardDir(root, card, dev, sizeof(dev))) { printf("tune %d error no such card\n", card); return 1; }
|
||||
snprintf(path, sizeof(path), "%s/hwmon", dev);
|
||||
d = opendir(path);
|
||||
if (d) { while ((e = readdir(d)) != NULL) if (strncmp(e->d_name, "hwmon", 5) == 0) { snprintf(hw, sizeof(hw), "%s/%s", path, e->d_name); break; } closedir(d); }
|
||||
snprintf(path, sizeof(path), "%s/pp_od_clk_voltage", dev);
|
||||
if (reset) {
|
||||
if (!writeText(path, "r\n") || !writeText(path, "c\n")) verdict = "error pp_od_clk_voltage reset (root needed)";
|
||||
if (hw[0]) { char cap[64], dp[900]; snprintf(dp, sizeof(dp), "%s/power1_cap_default", hw); if (readText(dp, cap, sizeof(cap))) { snprintf(dp, sizeof(dp), "%s/power1_cap", hw); if (!writeText(dp, cap)) verdict = "error power1_cap reset (root needed)"; } }
|
||||
} else {
|
||||
if (setGmax) { char cmd[64]; snprintf(cmd, sizeof(cmd), "s 1 %d\n", gmax); if (!writeText(path, cmd) || !writeText(path, "c\n")) verdict = "error pp_od_clk_voltage write (root needed, or the clock is outside the card's range)"; }
|
||||
if (setPlimit && verdict[0] == 'o') {
|
||||
char dp[900], cap[64];
|
||||
if (!hw[0]) verdict = "error no hwmon";
|
||||
else { snprintf(dp, sizeof(dp), "%s/power1_cap_default", hw); if (!readText(dp, cap, sizeof(cap))) verdict = "error no power1_cap_default"; else { double uw = atof(cap) * (100.0 + plimit) / 100.0; snprintf(cap, sizeof(cap), "%.0f\n", uw); snprintf(dp, sizeof(dp), "%s/power1_cap", hw); if (!writeText(dp, cap)) verdict = "error power1_cap write (root needed)"; } }
|
||||
}
|
||||
}
|
||||
{
|
||||
double g = -1, pc = -1, pd = -1;
|
||||
snprintf(path, sizeof(path), "%s/pp_od_clk_voltage", dev); if (readText(path, text, sizeof(text))) { const char* s1 = strstr(text, "1:"); if (s1) g = atof(s1 + 2); }
|
||||
if (hw[0]) { snprintf(path, sizeof(path), "%s/power1_cap", hw); pc = readNumber(path); snprintf(path, sizeof(path), "%s/power1_cap_default", hw); pd = readNumber(path); }
|
||||
printf("tune %d name \"amdgpu\" gmax %.0f gmax_range - - plimit %.0f plimit_range - - factory - %s\n", card, g, (pc > 0 && pd > 0) ? 100.0 * pc / pd - 100.0 : -1.0, verdict);
|
||||
}
|
||||
return verdict[0] == 'o' ? 0 : 1;
|
||||
}
|
||||
static int sysfsSample(const char* root) {
|
||||
DIR* d = opendir(root);
|
||||
struct dirent* e;
|
||||
|
|
@ -251,17 +434,31 @@ static int sysfsSample(const char* root) {
|
|||
snprintf(path, sizeof(path), "%s/pp_dpm_mclk", dev); if (readText(path, text, sizeof(text))) s.mclk = dpmCurrent(text);
|
||||
snprintf(path, sizeof(path), "%s/pp_dpm_sclk", dev); if (readText(path, text, sizeof(text))) s.gclk = dpmCurrent(text);
|
||||
snprintf(path, sizeof(path), "%s/gpu_busy_percent", dev); s.util = readNumber(path);
|
||||
{ /* the limits in force: power1_cap against its default, the OD max clock from pp_od_clk_voltage */
|
||||
char hp[900]; double pc = -1, pd = -1;
|
||||
snprintf(hp, sizeof(hp), "%s/hwmon", dev); hw = opendir(hp);
|
||||
if (hw) { while ((he = readdir(hw)) != NULL) if (strncmp(he->d_name, "hwmon", 5) == 0) { char q[1000]; snprintf(q, sizeof(q), "%s/%s/power1_cap", hp, he->d_name); pc = readNumber(q); snprintf(q, sizeof(q), "%s/%s/power1_cap_default", hp, he->d_name); pd = readNumber(q); break; } closedir(hw); }
|
||||
if (pc > 0 && pd > 0) s.plimitPct = 100.0 * pc / pd;
|
||||
snprintf(hp, sizeof(hp), "%s/pp_od_clk_voltage", dev); if (readText(hp, text, sizeof(text))) { const char* s1 = strstr(text, "1:"); if (s1) s.gmax = atof(s1 + 2); }
|
||||
}
|
||||
printSample(ordinal++, &s);
|
||||
}
|
||||
closedir(d);
|
||||
return ordinal;
|
||||
}
|
||||
int main(int argc, char** argv) {
|
||||
int every = 0, i;
|
||||
int every = 0, i, card = -1, setGmax = 0, gmax = 0, setPlimit = 0, plimit = 0, reset = 0;
|
||||
const char* root = getenv("IGNEUM_DRM_ROOT") ? getenv("IGNEUM_DRM_ROOT") : "/sys/class/drm"; /* a fixture tree for tests */
|
||||
for (i = 1; i < argc; ++i) if (strcmp(argv[i], "-l") == 0 && i + 1 < argc) every = atoi(argv[++i]);
|
||||
for (i = 1; i < argc; ++i) {
|
||||
if (strcmp(argv[i], "-l") == 0 && i + 1 < argc) every = atoi(argv[++i]);
|
||||
else if (strcmp(argv[i], "--card") == 0 && i + 1 < argc) card = atoi(argv[++i]);
|
||||
else if (strcmp(argv[i], "--set-gmax") == 0 && i + 1 < argc) { setGmax = 1; gmax = atoi(argv[++i]); }
|
||||
else if (strcmp(argv[i], "--set-plimit") == 0 && i + 1 < argc) { setPlimit = 1; plimit = atoi(argv[++i]); }
|
||||
else if (strcmp(argv[i], "--reset") == 0) reset = 1;
|
||||
}
|
||||
signal(SIGINT, onSignal); signal(SIGTERM, onSignal);
|
||||
setvbuf(stdout, NULL, _IOLBF, 0);
|
||||
if (card >= 0) return sysfsTune(root, card, setGmax, gmax, setPlimit, plimit, reset);
|
||||
do {
|
||||
double t0 = nowMs();
|
||||
int n = sysfsSample(root);
|
||||
|
|
|
|||
Loading…
Reference in a new issue