AMD clock and power control (ADLX / amdgpu sysfs) in igneum-gpu-telemetry, the AMD lever in the app's efficiency sweep, the one-card test playbook; the 9070 XT sweep is OWED (the eGPU left PC 1's bus at 20:40 UTC)
the project lead: "can anything be done to redeem AMD?". The control is built and unit-tested; the measurement was not taken because the RX 9070 XT (and the Sonnet box itself) vanished from PC 1's USB4 and PnP lists before the run (relay probes #202 and #203, 20:40 and 20:45:34 UTC; a PnP rescan did not bring it back; nobody restarts PC 1 tonight). proto-opencl/gpu-telemetry.c: --tune (per card: max GPU clock range and value, power-limit range and offset, factory flag), --card N --set-gmax MHZ, --card N --set-plimit PCT (offset from default), --card N --reset (ResetToFactory); ADLX IADLXManualGraphicsTuning2 and IADLXManualPowerTuning on Windows, pp_od_clk_voltage and hwmon power1_cap on Linux (fixture-verified); every sample line carries plimit_pct and gmax_mhz, the limits in force. host.c: --probe-mib takes a list (4,64,96,1024 for the 7900 XTX's 96 MB cache). sweep.rs: Lever (NVIDIA watts, AMD power percent, AMD max clock), plan_amd_power_steps, plan_amd_clock_steps, the read-back tolerance per lever, Row::line_for, unsupported_reason_for with the AMD case; engine.rs: parse_amd_tune, the --tune read at the AMD child's start, the AMD paths in sweep_begin (no nvidia-smi probe, no elevated helper), sweep_set_cap (the helper command) and sweep_drive (the read-back by lever), the AMD draw feeding the samples, the start gate; state.rs: the AMD ordinal, limits in force, ranges and amd_tunable. Tests: 87 pass (the planners, the machine on percent read-backs, the tune and sample parsers on captured lines). relay/playbooks/amd-card-test.ps1: one AMD card through the list fold check, memprobe, a 120 s hash run, the telemetry window, the clock and power sweeps with the memory clock watched, factory reset with read-back, the bench-log table and the MH/W and MH per pound line; parameter $Gfx (gfx1201, gfx1100). docs/bench-log.md: the owed rows with the reason and time, the gfx1100 expectation by reasoning from gfx1201. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
This commit is contained in:
parent
423936bf89
commit
9931dcb119
9 changed files with 766 additions and 42 deletions
|
|
@ -104,7 +104,7 @@ pub fn apply_defaults(c: &mut CardState) {
|
|||
/// Marks whether the efficiency sweep (src/sweep.rs) can run on a card, with the reason when it cannot. Called
|
||||
/// once the power limits are known.
|
||||
pub fn mark_sweep_support(c: &mut CardState) {
|
||||
match crate::sweep::unsupported_reason(&c.vendor, c.power_default_w, &c.device) {
|
||||
match crate::sweep::unsupported_reason_for(&c.vendor, c.power_default_w, &c.device, c.amd_tunable) {
|
||||
None => {
|
||||
c.sweep_supported = true;
|
||||
if c.sweep_state.is_empty() || c.sweep_state == "unsupported" {
|
||||
|
|
|
|||
|
|
@ -962,7 +962,12 @@ impl Engine {
|
|||
}
|
||||
}
|
||||
Cmd::SweepCapSet(text) => {
|
||||
if !text.contains("All done") {
|
||||
if text.starts_with("adlx ") {
|
||||
self.shared.log(&format!("sweep: {}", short(&text, 220)));
|
||||
if text.contains(" error ") {
|
||||
self.sweep_abort(&format!("ADLX refused the setting ({})", short(&text, 160)));
|
||||
}
|
||||
} else if !text.contains("All done") {
|
||||
self.shared.log(&format!("sweep: nvidia-smi -pl answered: {}", short(&text.replace('\n', " "), 200)));
|
||||
}
|
||||
}
|
||||
|
|
@ -1552,6 +1557,14 @@ impl Engine {
|
|||
}
|
||||
}
|
||||
if self.amd_telemetry.is_none() && now >= self.amd_telemetry_retry_at {
|
||||
// what the cards' manual tuning allows, once per child start (the sweep's ranges and the stock max clock)
|
||||
if let Some(out) = crate::detect::run_timeout(std::process::Command::new(&exe).arg("--tune"), None, Duration::from_secs(20)) {
|
||||
for line in out.lines() {
|
||||
if let Some(tu) = parse_amd_tune(line) {
|
||||
self.amd_tune_line(&tu);
|
||||
}
|
||||
}
|
||||
}
|
||||
let args: Vec<String> = vec!["-l".into(), "5".into()];
|
||||
let log = self.shared.runtime.log_dir.join(format!("gpu-amd-{}.log", self.stamp));
|
||||
match procs::spawn(Source::AmdTelemetry, &exe, &args, None, &log, &self.lines_tx, &[]) {
|
||||
|
|
@ -1582,6 +1595,9 @@ impl Engine {
|
|||
}
|
||||
let Some(idx) = target else { return };
|
||||
let Some(c) = st.mining.cards.iter_mut().find(|c| c.index == idx) else { return };
|
||||
c.amd_ordinal = s.ordinal as i32;
|
||||
c.amd_plimit_pct = s.plimit_pct.max(0.0);
|
||||
c.amd_gmax_mhz = s.gmax_mhz.max(0.0);
|
||||
if s.watts > 0.0 {
|
||||
c.power_w = s.watts;
|
||||
}
|
||||
|
|
@ -1593,6 +1609,49 @@ impl Engine {
|
|||
c.mclk_mhz = s.mclk_mhz.max(0.0);
|
||||
c.util_pct = s.util_pct.max(0.0);
|
||||
c.telemetry_at = crate::platform::unix_now_f();
|
||||
let draw = c.power_w;
|
||||
drop(st);
|
||||
// the sweep's draw samples, as the NVIDIA line feeds them
|
||||
if let Some(run) = self.sweep.as_mut() {
|
||||
if run.card == idx && draw > 0.0 {
|
||||
run.sample_draw(draw);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// One `tune` line of igneum-gpu-telemetry --tune: the card it describes (matched as the sample lines are, by
|
||||
/// the ordinal among the discrete AMD cards; integrated ones have no manual tuning) learns its ranges and whether
|
||||
/// the sweep can run on it.
|
||||
fn amd_tune_line(&mut self, tu: &AmdTune) {
|
||||
let mut st = self.st();
|
||||
let mut nth = 0usize;
|
||||
let mut target: Option<usize> = None;
|
||||
for c in st.mining.cards.iter() {
|
||||
if c.vendor != "amd" || c.kind != "discrete" {
|
||||
continue;
|
||||
}
|
||||
if nth == tu.ordinal_in_kind {
|
||||
target = Some(c.index);
|
||||
break;
|
||||
}
|
||||
nth += 1;
|
||||
}
|
||||
let Some(idx) = target else { return };
|
||||
let Some(c) = st.mining.cards.iter_mut().find(|c| c.index == idx) else { return };
|
||||
c.amd_ordinal = tu.ordinal as i32;
|
||||
c.amd_tunable = tu.plimit_range.is_some();
|
||||
if let Some((lo, hi)) = tu.plimit_range {
|
||||
c.amd_plimit_min = lo;
|
||||
c.amd_plimit_max = hi;
|
||||
}
|
||||
if let Some((lo, hi)) = tu.gmax_range {
|
||||
c.amd_gmax_min = lo;
|
||||
c.amd_gmax_max = hi;
|
||||
}
|
||||
if tu.gmax > 0.0 && c.amd_gmax_stock <= 0.0 {
|
||||
c.amd_gmax_stock = tu.gmax;
|
||||
}
|
||||
crate::detect::mark_sweep_support(c);
|
||||
}
|
||||
|
||||
/// "index, draw, gpu temp, mem temp, limit" every 5 s.
|
||||
|
|
@ -1711,7 +1770,7 @@ impl Engine {
|
|||
let cards = self.st().mining.cards.clone();
|
||||
let mut pick: Option<(usize, bool)> = None;
|
||||
for c in cards.iter() {
|
||||
if !(c.enabled && c.sweep_supported && c.vendor == "nvidia") {
|
||||
if !(c.enabled && c.sweep_supported && (c.vendor == "nvidia" || (c.vendor == "amd" && c.amd_tunable))) {
|
||||
continue;
|
||||
}
|
||||
let forced = self.sweep_queue.contains(&c.key);
|
||||
|
|
@ -1764,6 +1823,11 @@ impl Engine {
|
|||
cc.sweep_state = "running".into();
|
||||
cc.sweep_note = "sweep: checking how the cap is set".into();
|
||||
}
|
||||
if c.vendor == "amd" {
|
||||
// ADLX sets the limit from this process with no elevation (igneum-gpu-telemetry --card N --set-plimit)
|
||||
self.sweep_mode_known(Ok(true));
|
||||
return;
|
||||
}
|
||||
if let Some(direct) = self.sweep_direct {
|
||||
self.sweep_mode_known(Ok(direct));
|
||||
return;
|
||||
|
|
@ -1791,36 +1855,41 @@ impl Engine {
|
|||
return;
|
||||
}
|
||||
};
|
||||
self.sweep_direct = Some(direct);
|
||||
let Some(c) = self.st().mining.cards.get(idx).cloned() else {
|
||||
self.sweep_pending = None;
|
||||
return;
|
||||
};
|
||||
if !direct && !self.sweep_helper {
|
||||
let amd = c.vendor == "amd";
|
||||
if !amd {
|
||||
self.sweep_direct = Some(direct);
|
||||
}
|
||||
if !amd && !direct && !self.sweep_helper {
|
||||
if let Err(e) = self.sweep_helper_start(&c) {
|
||||
self.sweep_abort(&format!("the elevated helper could not start: {e}"));
|
||||
return;
|
||||
}
|
||||
}
|
||||
let steps = crate::sweep::plan_steps(c.power_default_w, c.power_min_w, c.power_max_w);
|
||||
let lever = if amd { crate::sweep::Lever::AmdPowerPct } else { crate::sweep::Lever::NvidiaWatts };
|
||||
let steps = if amd { crate::sweep::plan_amd_power_steps(c.amd_plimit_min, c.amd_plimit_max) } else { crate::sweep::plan_steps(c.power_default_w, c.power_min_w, c.power_max_w) };
|
||||
if steps.is_empty() {
|
||||
self.sweep_abort("no steps: the card reported no default power limit");
|
||||
return;
|
||||
}
|
||||
let label = self.miners.iter().find(|m| m.card == idx).map(|m| m.label.clone()).unwrap_or_else(|| format!("card-{idx}"));
|
||||
let before_w = if c.power_limit_w > 0.0 { c.power_limit_w } else { requested_watts(&c) };
|
||||
let before_w = if amd { if c.amd_plimit_pct > 0.0 { c.amd_plimit_pct } else { 100.0 } } else if c.power_limit_w > 0.0 { c.power_limit_w } else { requested_watts(&c) };
|
||||
let now = Instant::now();
|
||||
let run = crate::sweep::Run::new(idx, &c.key, &c.device, &label, steps.clone(), c.power_pct, before_w, forced, crate::sweep::Timing::from_env(), now);
|
||||
let run = crate::sweep::Run::new(idx, &c.key, &c.device, &label, steps.clone(), c.power_pct, before_w, forced, crate::sweep::Timing::from_env(), now).with_lever(lever);
|
||||
self.sweep_pending = None;
|
||||
self.sweep_say(&format!(
|
||||
"SWEEP start card={label} name={} steps={} default={:.0} min={:.0} max={:.0} before={:.0} mode={}",
|
||||
"SWEEP start card={label} name={} lever={} steps={} default={:.0} min={:.0} max={:.0} before={:.0} mode={}",
|
||||
c.name.replace(' ', "_"),
|
||||
lever.name(),
|
||||
steps.iter().map(|s| s.pct.to_string()).collect::<Vec<_>>().join(","),
|
||||
c.power_default_w,
|
||||
c.power_min_w,
|
||||
c.power_max_w,
|
||||
if amd { 100.0 } else { c.power_default_w },
|
||||
if amd { 100.0 + c.amd_plimit_min } else { c.power_min_w },
|
||||
if amd { 100.0 + c.amd_plimit_max } else { c.power_max_w },
|
||||
before_w,
|
||||
if direct { "direct" } else { "helper" }
|
||||
if amd { "adlx" } else if direct { "direct" } else { "helper" }
|
||||
));
|
||||
self.shared.event("info", &format!("{}: efficiency sweep started: {} caps from {}% down, {} s each on the live program", c.name, steps.len(), steps[0].pct, (run.timing.settle + run.timing.hold).as_secs()));
|
||||
self.sweep = Some(run);
|
||||
|
|
@ -1864,6 +1933,25 @@ impl Engine {
|
|||
fn sweep_set_cap(&mut self, device: &str, watts: f64) {
|
||||
self.sweep_seq += 1;
|
||||
let w = watts.round() as u64;
|
||||
let amd = self.sweep.as_ref().map(|r| r.lever != crate::sweep::Lever::NvidiaWatts).unwrap_or(false);
|
||||
if amd {
|
||||
// igneum-gpu-telemetry --card <ordinal> --set-plimit <offset> (or --set-gmax); the readback comes with the
|
||||
// next sample line (plimit_pct / gmax_mhz, 5 s)
|
||||
let (lever, idx) = self.sweep.as_ref().map(|r| (r.lever, r.card)).unwrap();
|
||||
let ordinal = self.st().mining.cards.get(idx).map(|c| c.amd_ordinal).unwrap_or(-1);
|
||||
let Some(exe) = self.bins.telemetry.clone() else { return };
|
||||
let args: Vec<String> = match lever {
|
||||
crate::sweep::Lever::AmdPowerPct => vec!["--card".into(), ordinal.to_string(), "--set-plimit".into(), (watts.round() as i64 - 100).to_string()],
|
||||
_ => vec!["--card".into(), ordinal.to_string(), "--set-gmax".into(), w.to_string()],
|
||||
};
|
||||
let shared = self.shared.clone();
|
||||
std::thread::spawn(move || {
|
||||
let out = crate::detect::run_timeout(std::process::Command::new(&exe).args(&args), None, Duration::from_secs(20)).unwrap_or_else(|| "igneum-gpu-telemetry did not answer".into());
|
||||
shared.send(Cmd::SweepCapSet(format!("adlx {}: {}", args.join(" "), out.lines().find(|l| l.starts_with("tune ")).unwrap_or("no tune line"))));
|
||||
});
|
||||
self.shared.log(&format!("sweep: {} {} requested on AMD card {} (ordinal {ordinal})", lever.name(), if lever == crate::sweep::Lever::AmdPowerPct { format!("{w}%") } else { format!("{w} MHz") }, device));
|
||||
return;
|
||||
}
|
||||
if self.sweep_direct == Some(true) {
|
||||
let smi = crate::platform::tool("nvidia-smi");
|
||||
let device = device.to_string();
|
||||
|
|
@ -1906,13 +1994,18 @@ impl Engine {
|
|||
self.sweep_abort(&why);
|
||||
return;
|
||||
}
|
||||
let outs = self.sweep.as_mut().map(|r| r.tick(now, c.power_limit_w)).unwrap_or_default();
|
||||
let label = self.sweep.as_ref().map(|r| r.label.clone()).unwrap_or_default();
|
||||
let readback = match self.sweep.as_ref().map(|r| r.lever).unwrap_or(crate::sweep::Lever::NvidiaWatts) {
|
||||
crate::sweep::Lever::NvidiaWatts => c.power_limit_w,
|
||||
crate::sweep::Lever::AmdPowerPct => c.amd_plimit_pct,
|
||||
crate::sweep::Lever::AmdMaxClockMhz => c.amd_gmax_mhz,
|
||||
};
|
||||
let outs = self.sweep.as_mut().map(|r| r.tick(now, readback)).unwrap_or_default();
|
||||
let (label, lever) = self.sweep.as_ref().map(|r| (r.label.clone(), r.lever)).unwrap_or((String::new(), crate::sweep::Lever::NvidiaWatts));
|
||||
for o in outs {
|
||||
match o {
|
||||
crate::sweep::Out::Apply(w) => self.sweep_set_cap(&device, w),
|
||||
crate::sweep::Out::Row(row) => {
|
||||
self.sweep_say(&row.line(&label));
|
||||
self.sweep_say(&row.line_for(&label, lever));
|
||||
if let Some(cc) = self.st().mining.cards.get_mut(idx) {
|
||||
if row.usable() {
|
||||
cc.sweep_note = format!("sweep: {}% done · {:.3} MH/W", row.pct, row.eff);
|
||||
|
|
@ -3371,11 +3464,15 @@ pub struct AmdTelemetry {
|
|||
pub mclk_mhz: f64,
|
||||
pub gclk_mhz: f64,
|
||||
pub util_pct: f64,
|
||||
/// the limits in force (5 October 2026 tuning lines): the power limit as percent of default, the max GPU clock;
|
||||
/// -1 when the line has no such field (an older helper) or the card has no manual tuning
|
||||
pub plimit_pct: f64,
|
||||
pub gmax_mhz: f64,
|
||||
pub source: String,
|
||||
}
|
||||
|
||||
/// Parses `amd <n> bus <b> kind <k> name "<name>" watts <w> temp_c <t> fan_rpm <r> fan_pct <p> mclk_mhz <m>
|
||||
/// gclk_mhz <g> util_pct <u> source <s>`; a `-` value reads as -1.0. Anything else (info, end) gives None.
|
||||
/// gclk_mhz <g> util_pct <u> [plimit_pct <l> gmax_mhz <x>] source <s>`; a `-` value reads as -1.0. Anything else (info, end) gives None.
|
||||
/// With one card per kind (the common case) `ordinal_in_kind` is 0 for the discrete card and 0 for the integrated
|
||||
/// one whatever their `amd N`; with several discrete cards the helper's order within the kind is kept: the rank is
|
||||
/// the number of earlier lines of the same kind, which the helper encodes by listing kinds contiguously (ADLX lists
|
||||
|
|
@ -3405,12 +3502,14 @@ pub fn parse_amd_telemetry(line: &str) -> Option<AmdTelemetry> {
|
|||
let mclk_mhz = num("mclk_mhz")?;
|
||||
let gclk_mhz = num("gclk_mhz")?;
|
||||
let util_pct = num("util_pct")?;
|
||||
let plimit_pct = num("plimit_pct").unwrap_or(-1.0);
|
||||
let gmax_mhz = num("gmax_mhz").unwrap_or(-1.0);
|
||||
let source = tp.iter().position(|p| *p == "source").and_then(|i| tp.get(i + 1)).map(|s| s.to_string()).unwrap_or_default();
|
||||
let kind = hp[5].to_string();
|
||||
// the rank within the kind: the helper lists one integrated card at most and it comes first when present
|
||||
// (ADLX order on every PC seen so far), so a discrete card's rank is its ordinal minus the integrated ones before it
|
||||
let ordinal_in_kind = if kind == "discrete" && ordinal > 0 { ordinal - 1 } else if kind == "discrete" { 0 } else { 0 };
|
||||
Some(AmdTelemetry { ordinal, ordinal_in_kind, bus: hp[3].to_string(), kind, name: name.to_string(), watts, temp_c, fan_rpm, fan_pct, mclk_mhz, gclk_mhz, util_pct, source })
|
||||
Some(AmdTelemetry { ordinal, ordinal_in_kind, bus: hp[3].to_string(), kind, name: name.to_string(), watts, temp_c, fan_rpm, fan_pct, mclk_mhz, gclk_mhz, util_pct, plimit_pct, gmax_mhz, source })
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
|
|
@ -3420,8 +3519,9 @@ mod amd_telemetry_tests {
|
|||
#[test]
|
||||
fn a_sysfs_line_from_the_fixture_parses() {
|
||||
// proto-opencl/gpu-telemetry.c on the Mac against a fixture tree, 5 October 2026
|
||||
let l = "amd 0 bus 0000:0c:00.0 kind discrete name \"AMD Radeon RX 9070 XT\" watts 287.0 temp_c 61.0 fan_rpm 1180 fan_pct 30 mclk_mhz 1258 gclk_mhz 2450 util_pct 90 source sysfs";
|
||||
let l = "amd 0 bus 0000:0c:00.0 kind discrete name \"AMD Radeon RX 9070 XT\" watts 287.0 temp_c 61.0 fan_rpm 1180 fan_pct 30 mclk_mhz 1258 gclk_mhz 2450 util_pct 90 plimit_pct 80 gmax_mhz 2450 source sysfs";
|
||||
let s = parse_amd_telemetry(l).unwrap();
|
||||
assert_eq!((s.plimit_pct, s.gmax_mhz), (80.0, 2450.0));
|
||||
assert_eq!((s.ordinal, s.ordinal_in_kind, s.bus.as_str(), s.kind.as_str(), s.name.as_str()), (0, 0, "0000:0c:00.0", "discrete", "AMD Radeon RX 9070 XT"));
|
||||
assert_eq!((s.watts, s.temp_c, s.fan_rpm, s.fan_pct, s.mclk_mhz, s.gclk_mhz, s.util_pct), (287.0, 61.0, 1180.0, 30.0, 1258.0, 2450.0, 90.0));
|
||||
assert_eq!(s.source, "sysfs");
|
||||
|
|
@ -3429,9 +3529,11 @@ mod amd_telemetry_tests {
|
|||
|
||||
#[test]
|
||||
fn a_dash_reads_as_unknown_and_other_lines_give_none() {
|
||||
let l = "amd 1 bus 98 kind discrete name \"AMD Radeon RX 9070 XT\" watts 250.3 temp_c 58.0 fan_rpm 900 fan_pct - mclk_mhz 1258 gclk_mhz 2460 util_pct 97.5 source adlx";
|
||||
// PC 1, 20:27 UTC, the first helper build (no plimit_pct / gmax_mhz fields yet): they read as -1
|
||||
let l = "amd 1 bus 98 kind discrete name \"AMD Radeon RX 9070 XT\" watts 214.0 temp_c 64.0 fan_rpm 659 fan_pct - mclk_mhz 2505 gclk_mhz 3289 util_pct 100 source adlx";
|
||||
let s = parse_amd_telemetry(l).unwrap();
|
||||
assert_eq!(s.fan_pct, -1.0);
|
||||
assert_eq!((s.plimit_pct, s.gmax_mhz, s.watts, s.mclk_mhz), (-1.0, -1.0, 214.0, 2505.0));
|
||||
assert_eq!(s.ordinal_in_kind, 0, "the second line overall but the first discrete card after the integrated one");
|
||||
assert!(parse_amd_telemetry("end 3.2 ms 2 card(s)").is_none());
|
||||
assert!(parse_amd_telemetry("info adlx: ADLXHelper_Initialize returned 1").is_none());
|
||||
|
|
@ -3445,3 +3547,73 @@ mod amd_telemetry_tests {
|
|||
assert_eq!((s.watts, s.util_pct, s.source.as_str()), (-1.0, 100.0, "perfcounter"));
|
||||
}
|
||||
}
|
||||
|
||||
/// One `tune` line of igneum-gpu-telemetry --tune: `tune N name "..." gmax X gmax_range MIN MAX plimit OFF
|
||||
/// plimit_range MIN MAX factory 0|1 ok|error ...`; a `-` means the card has no such tuning.
|
||||
#[derive(Debug, Clone, PartialEq)]
|
||||
pub struct AmdTune {
|
||||
pub ordinal: usize,
|
||||
pub ordinal_in_kind: usize,
|
||||
pub name: String,
|
||||
pub gmax: f64,
|
||||
pub gmax_range: Option<(f64, f64)>,
|
||||
pub plimit: f64,
|
||||
pub plimit_range: Option<(f64, f64)>,
|
||||
pub factory: Option<bool>,
|
||||
pub ok: bool,
|
||||
}
|
||||
|
||||
pub fn parse_amd_tune(line: &str) -> Option<AmdTune> {
|
||||
let line = line.trim();
|
||||
let rest = line.strip_prefix("tune ")?;
|
||||
let (ord_s, rest) = rest.split_once(" name \"")?;
|
||||
let ordinal: usize = ord_s.trim().parse().ok()?;
|
||||
let (name, tail) = rest.split_once('"')?;
|
||||
let tp: Vec<&str> = tail.split_whitespace().collect();
|
||||
let at = |key: &str| tp.iter().position(|p| *p == key);
|
||||
let f = |s: Option<&&str>| -> Option<f64> { let v = *s?; if v == "-" { None } else { v.parse::<f64>().ok() } };
|
||||
let gi = at("gmax")?;
|
||||
let gri = at("gmax_range")?;
|
||||
let pi = at("plimit")?;
|
||||
let pri = at("plimit_range")?;
|
||||
let gmax = f(tp.get(gi + 1)).unwrap_or(-1.0);
|
||||
let gmax_range = match (f(tp.get(gri + 1)), f(tp.get(gri + 2))) { (Some(a), Some(b)) => Some((a, b)), _ => None };
|
||||
let plimit = f(tp.get(pi + 1)).unwrap_or(0.0);
|
||||
let plimit_range = match (f(tp.get(pri + 1)), f(tp.get(pri + 2))) { (Some(a), Some(b)) => Some((a, b)), _ => None };
|
||||
let factory = at("factory").and_then(|i| tp.get(i + 1)).and_then(|v| match *v { "1" => Some(true), "0" => Some(false), _ => None });
|
||||
let ok = tp.last().map(|v| *v == "ok").unwrap_or(false);
|
||||
// integrated cards come first in ADLX order and carry no manual tuning: the rank among discrete cards is the
|
||||
// ordinal minus the integrated ones before it; a tune line with no ranges at all is taken as integrated
|
||||
let integrated_before = if ordinal > 0 && gmax_range.is_none() && plimit_range.is_none() { 0 } else { ordinal.min(1) };
|
||||
let _ = integrated_before;
|
||||
let ordinal_in_kind = if ordinal > 0 { ordinal - 1 } else { 0 };
|
||||
Some(AmdTune { ordinal, ordinal_in_kind, name: name.to_string(), gmax, gmax_range, plimit, plimit_range, factory, ok })
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod amd_tune_tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn a_tune_line_with_ranges_parses() {
|
||||
// the shape igneum-gpu-telemetry --tune prints (ranges are the card's; a 9070 XT's own are owed, it left the bus)
|
||||
let l = "tune 1 name \"AMD Radeon RX 9070 XT\" gmax 3300 gmax_range 500 3450 plimit 0 plimit_range -30 10 factory 1 ok";
|
||||
let tu = parse_amd_tune(l).unwrap();
|
||||
assert_eq!((tu.ordinal, tu.ordinal_in_kind, tu.name.as_str(), tu.gmax, tu.plimit), (1, 0, "AMD Radeon RX 9070 XT", 3300.0, 0.0));
|
||||
assert_eq!(tu.gmax_range, Some((500.0, 3450.0)));
|
||||
assert_eq!(tu.plimit_range, Some((-30.0, 10.0)));
|
||||
assert_eq!(tu.factory, Some(true));
|
||||
assert!(tu.ok);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn an_integrated_card_without_tuning_and_an_error_verdict() {
|
||||
let l = "tune 0 name \"AMD Radeon(TM) Graphics\" gmax - gmax_range - - plimit - plimit_range - - factory 1 ok";
|
||||
let tu = parse_amd_tune(l).unwrap();
|
||||
assert!(tu.gmax_range.is_none() && tu.plimit_range.is_none());
|
||||
let e = parse_amd_tune("tune 1 name \"x\" gmax 2800 gmax_range 500 3450 plimit -20 plimit_range -30 10 factory 0 error SetPowerLimit(-50) returned 1").unwrap();
|
||||
assert!(!e.ok);
|
||||
assert_eq!(e.factory, Some(false));
|
||||
assert!(parse_amd_tune("end 1.0 ms 2 card(s)").is_none());
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -79,6 +79,18 @@ pub struct CardState {
|
|||
pub fan_rpm: f64,
|
||||
pub mclk_mhz: f64,
|
||||
pub util_pct: f64,
|
||||
/// the AMD card's line in the helper's output (`amd N`), -1 until a line matched; what --card N addresses
|
||||
pub amd_ordinal: i32,
|
||||
/// the AMD limits in force from the helper's line: the power limit as percent of default (100 = default), the max GPU clock
|
||||
pub amd_plimit_pct: f64,
|
||||
pub amd_gmax_mhz: f64,
|
||||
/// from the helper's `tune` line: manual tuning present, the power offset range (percent) and the max clock range and stock
|
||||
pub amd_tunable: bool,
|
||||
pub amd_plimit_min: f64,
|
||||
pub amd_plimit_max: f64,
|
||||
pub amd_gmax_min: f64,
|
||||
pub amd_gmax_max: f64,
|
||||
pub amd_gmax_stock: f64,
|
||||
// hash per watt (src/sweep.rs)
|
||||
pub eff_mhw: f64, // live: hash_now over power_w, MH per watt; 0 = unknown
|
||||
pub sweep_supported: bool, // NVIDIA with readable limits; the note says why not otherwise
|
||||
|
|
|
|||
|
|
@ -16,8 +16,14 @@
|
|||
//! SWEEP card=<label> cap=<pct> limit=<set W> watts=<mean draw W> mhs=<MH/s> eff=<MH per W>
|
||||
//! SWEEP chosen card=<label> cap=<pct> limit=<W> watts=<W> mhs=<x> eff=<MH per W>
|
||||
//! SWEEP aborted card=<label> reason=<text>
|
||||
//! Apple silicon and AMD are reported unsupported with the reason (no cap to set; powermetrics needs root; no AMD
|
||||
//! power reading in this app).
|
||||
//! Apple silicon is reported unsupported with the reason (no cap to set; powermetrics needs root).
|
||||
//!
|
||||
//! AMD (5 October 2026): the same machine with another lever. igneum-gpu-telemetry (proto-opencl/gpu-telemetry.c)
|
||||
//! sets the ADLX power limit (a percent offset from the card's default, 0 = 100%) or the max GPU clock, with no
|
||||
//! elevation, and its sample lines carry the limit in force (plimit_pct, gmax_mhz) as the readback. A Step's
|
||||
//! `watts` then holds the percent (or the MHz), `Row.limit` prints it, and `Row.watts` stays the measured draw.
|
||||
//! The lever a card sweeps is `Lever`; the AMD default is the power limit (the clock lever is for the card test
|
||||
//! playbook until a measured sweep says which lever gives the better MH per watt: owed, the 9070 XT left the bus).
|
||||
|
||||
use std::collections::HashMap;
|
||||
use std::time::{Duration, Instant};
|
||||
|
|
@ -32,6 +38,89 @@ pub const PERIOD_S: u64 = 7 * 86_400;
|
|||
pub const NEEDS_S: i64 = 600;
|
||||
/// The worker must have been mining this long before a sweep starts (the first status lines are warm-up).
|
||||
pub const STABLE_S: u64 = 120;
|
||||
/// The AMD core-clock steps, MHz, highest first; the stock max clock is put in front when it is above the first.
|
||||
pub const AMD_CLOCK_STEPS: [u32; 5] = [2800, 2400, 2000, 1600, 1200];
|
||||
|
||||
/// What a sweep turns: the value a Step carries, how it is set and read back, and how close the readback must be.
|
||||
#[derive(Clone, Copy, Debug, PartialEq)]
|
||||
pub enum Lever {
|
||||
/// NVIDIA: nvidia-smi -pl, watts; readback power.limit
|
||||
NvidiaWatts,
|
||||
/// AMD: the ADLX power limit, as percent of the default (100 = default); readback plimit_pct
|
||||
AmdPowerPct,
|
||||
/// AMD: the ADLX max GPU clock, MHz; readback gmax_mhz (the card-test playbook's lever; the app sweeps the
|
||||
/// power limit until a measured sweep says the clock lever gives more MH per watt)
|
||||
#[allow(dead_code)]
|
||||
AmdMaxClockMhz,
|
||||
}
|
||||
|
||||
impl Lever {
|
||||
/// How far the readback may sit from the value set and still count as applied.
|
||||
pub fn tolerance(self) -> f64 {
|
||||
match self {
|
||||
Lever::NvidiaWatts => 1.5,
|
||||
Lever::AmdPowerPct => 0.5,
|
||||
Lever::AmdMaxClockMhz => 5.0,
|
||||
}
|
||||
}
|
||||
pub fn unit(self) -> &'static str {
|
||||
match self {
|
||||
Lever::NvidiaWatts => "W",
|
||||
Lever::AmdPowerPct => "%",
|
||||
Lever::AmdMaxClockMhz => "MHz",
|
||||
}
|
||||
}
|
||||
pub fn name(self) -> &'static str {
|
||||
match self {
|
||||
Lever::NvidiaWatts => "watts",
|
||||
Lever::AmdPowerPct => "amd_power_pct",
|
||||
Lever::AmdMaxClockMhz => "amd_max_clock",
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// The AMD power-limit steps: STEPS_PCT clamped to what ADLX allows (the offset range is `min_off..max_off`
|
||||
/// percent around 0; a 9070 XT range is read at run time), duplicates from the clamp dropped. `watts` holds the
|
||||
/// percent of default.
|
||||
pub fn plan_amd_power_steps(min_off: f64, max_off: f64) -> Vec<Step> {
|
||||
let mut out: Vec<Step> = Vec::new();
|
||||
let lo = 100.0 + min_off.min(0.0);
|
||||
let hi = 100.0 + max_off.max(0.0);
|
||||
for pct in STEPS_PCT {
|
||||
let v = (pct as f64).clamp(lo, hi).round();
|
||||
if out.last().map(|s| (s.watts - v).abs() < 0.5).unwrap_or(false) {
|
||||
continue;
|
||||
}
|
||||
out.push(Step { pct, watts: v });
|
||||
}
|
||||
out
|
||||
}
|
||||
|
||||
/// The AMD max-clock steps: the stock max clock first, then AMD_CLOCK_STEPS below it, each clamped to the card's
|
||||
/// range, duplicates dropped. `pct` is the step's share of the stock clock, for the lines and the pinned setting.
|
||||
pub fn plan_amd_clock_steps(stock: f64, min: f64, max: f64) -> Vec<Step> {
|
||||
let mut out: Vec<Step> = Vec::new();
|
||||
if stock <= 0.0 {
|
||||
return out;
|
||||
}
|
||||
let mut want: Vec<f64> = vec![stock];
|
||||
want.extend(AMD_CLOCK_STEPS.iter().map(|m| *m as f64).filter(|m| *m < stock));
|
||||
for mhz in want {
|
||||
let mut v = mhz;
|
||||
if min > 0.0 {
|
||||
v = v.max(min);
|
||||
}
|
||||
if max > 0.0 {
|
||||
v = v.min(max);
|
||||
}
|
||||
let v = v.round();
|
||||
if out.last().map(|s| (s.watts - v).abs() < 0.5).unwrap_or(false) {
|
||||
continue;
|
||||
}
|
||||
out.push(Step { pct: (100.0 * v / stock).round() as u32, watts: v });
|
||||
}
|
||||
out
|
||||
}
|
||||
|
||||
#[derive(Clone, Copy, Debug)]
|
||||
pub struct Timing {
|
||||
|
|
@ -113,10 +202,15 @@ impl Row {
|
|||
}
|
||||
/// `SWEEP card=<label> cap=<pct> limit=<W> watts=<W> mhs=<x> eff=<MH/W>`; a step without readings says so.
|
||||
pub fn line(&self, card: &str) -> String {
|
||||
self.line_for(card, Lever::NvidiaWatts)
|
||||
}
|
||||
/// The same line with the lever named when it is not the NVIDIA cap (`lever=amd_power_pct limit=80`).
|
||||
pub fn line_for(&self, card: &str, lever: Lever) -> String {
|
||||
let lv = if lever == Lever::NvidiaWatts { String::new() } else { format!(" lever={}", lever.name()) };
|
||||
if self.usable() {
|
||||
format!("SWEEP card={card} cap={} limit={:.0} watts={:.1} mhs={:.2} eff={:.4}", self.pct, self.limit, self.watts, self.mhs, self.eff)
|
||||
format!("SWEEP card={card}{lv} cap={} limit={:.0} watts={:.1} mhs={:.2} eff={:.4}", self.pct, self.limit, self.watts, self.mhs, self.eff)
|
||||
} else {
|
||||
format!("SWEEP card={card} cap={} limit={:.0} watts={:.1} mhs={:.2} eff=0 reason=no_readings draws={} rates={}", self.pct, self.limit, self.watts, self.mhs, self.draws, self.rates)
|
||||
format!("SWEEP card={card}{lv} cap={} limit={:.0} watts={:.1} mhs={:.2} eff=0 reason=no_readings draws={} rates={}", self.pct, self.limit, self.watts, self.mhs, self.draws, self.rates)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
|
@ -209,6 +303,7 @@ pub struct Run {
|
|||
pub started: Instant,
|
||||
pub forced: bool,
|
||||
pub timing: Timing,
|
||||
pub lever: Lever,
|
||||
draws: Vec<f64>,
|
||||
rates: Vec<f64>,
|
||||
}
|
||||
|
|
@ -230,11 +325,18 @@ impl Run {
|
|||
started: now,
|
||||
forced,
|
||||
timing,
|
||||
lever: Lever::NvidiaWatts,
|
||||
draws: Vec::new(),
|
||||
rates: Vec::new(),
|
||||
}
|
||||
}
|
||||
|
||||
/// The same run on another lever (AMD).
|
||||
pub fn with_lever(mut self, lever: Lever) -> Run {
|
||||
self.lever = lever;
|
||||
self
|
||||
}
|
||||
|
||||
pub fn current(&self) -> Option<&Step> {
|
||||
self.steps.get(self.i)
|
||||
}
|
||||
|
|
@ -265,12 +367,15 @@ impl Run {
|
|||
}
|
||||
}
|
||||
|
||||
/// Drives the state machine. `limit_w` is the limit the card reports now (the telemetry readback; 0 = unknown).
|
||||
/// Drives the state machine. `limit_w` is the limit the card reports now in the lever's unit (the telemetry
|
||||
/// readback; 0 = unknown).
|
||||
pub fn tick(&mut self, now: Instant, limit_w: f64) -> Vec<Out> {
|
||||
let mut out = Vec::new();
|
||||
let settle = self.timing.settle;
|
||||
let hold = self.timing.hold;
|
||||
let apply = self.timing.apply;
|
||||
let tol = self.lever.tolerance();
|
||||
let unit = self.lever.unit();
|
||||
match self.phase.clone() {
|
||||
Phase::Applying { since, sent } => {
|
||||
let Some(step) = self.current().cloned() else {
|
||||
|
|
@ -281,11 +386,11 @@ impl Run {
|
|||
if !sent {
|
||||
self.phase = Phase::Applying { since: now, sent: true };
|
||||
out.push(Out::Apply(step.watts));
|
||||
} else if limit_w > 0.0 && (limit_w - step.watts).abs() < 1.5 {
|
||||
} else if limit_w > 0.0 && (limit_w - step.watts).abs() < tol {
|
||||
self.phase = Phase::Settling { since: now };
|
||||
} else if now.duration_since(since) > apply {
|
||||
self.phase = Phase::Done;
|
||||
out.push(Out::Failed(format!("the {}% cap ({:.0} W) did not take within {} s (card reports {:.0} W)", step.pct, step.watts, apply.as_secs(), limit_w)));
|
||||
out.push(Out::Failed(format!("the {}% step ({:.0} {unit}) did not take within {} s (card reports {:.0} {unit})", step.pct, step.watts, apply.as_secs(), limit_w)));
|
||||
}
|
||||
}
|
||||
Phase::Settling { since } => {
|
||||
|
|
@ -326,12 +431,12 @@ impl Run {
|
|||
if !sent {
|
||||
self.phase = Phase::Finishing { since: now, sent: true };
|
||||
out.push(Out::Apply(best.limit));
|
||||
} else if limit_w > 0.0 && (limit_w - best.limit).abs() < 1.5 {
|
||||
} else if limit_w > 0.0 && (limit_w - best.limit).abs() < tol {
|
||||
self.phase = Phase::Done;
|
||||
out.push(Out::Finished(best));
|
||||
} else if now.duration_since(since) > apply {
|
||||
self.phase = Phase::Done;
|
||||
out.push(Out::Failed(format!("the chosen cap ({:.0} W) did not take within {} s", best.limit, apply.as_secs())));
|
||||
out.push(Out::Failed(format!("the chosen step ({:.0} {unit}) did not take within {} s", best.limit, apply.as_secs())));
|
||||
}
|
||||
}
|
||||
Phase::Done => {}
|
||||
|
|
@ -402,11 +507,18 @@ exit 0
|
|||
|
||||
/// Why a card cannot be swept, or None when it can (NVIDIA with a readable default limit and a device index).
|
||||
pub fn unsupported_reason(vendor: &str, power_default_w: f64, device: &str) -> Option<&'static str> {
|
||||
unsupported_reason_for(vendor, power_default_w, device, false)
|
||||
}
|
||||
|
||||
/// The same with the AMD case: supported once igneum-gpu-telemetry has reported manual power tuning on the card
|
||||
/// (`amd_tunable`, from its `tune` line); the 5 October 2026 reading path gives the draw.
|
||||
pub fn unsupported_reason_for(vendor: &str, power_default_w: f64, device: &str, amd_tunable: bool) -> Option<&'static str> {
|
||||
match vendor {
|
||||
"nvidia" if power_default_w > 0.0 && !device.is_empty() => None,
|
||||
"nvidia" => Some("not available: nvidia-smi did not report this card's power limits"),
|
||||
"apple" => Some("not available on Apple silicon: there is no power cap to set, and powermetrics needs administrator rights for the draw"),
|
||||
"amd" => Some("not available for AMD in this version: the app has no power reading or cap for AMD cards (nothing like nvidia-smi ships with the driver)"),
|
||||
"amd" if amd_tunable => None,
|
||||
"amd" => Some("not available: igneum-gpu-telemetry has not reported manual power tuning on this card (no ADLX, an integrated GPU, or the helper is not installed)"),
|
||||
_ => Some("not available: no power reading or cap for this card"),
|
||||
}
|
||||
}
|
||||
|
|
@ -576,12 +688,66 @@ mod tests {
|
|||
assert!(matches!(out.as_slice(), [Out::Row(_), Out::Failed(_)]), "{out:?}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn amd_power_steps_clamp_to_the_adlx_range() {
|
||||
// an offset range of -30..+10 percent (a plausible RDNA 4 range; the 9070 XT's own range is owed, the card left the bus)
|
||||
let s = plan_amd_power_steps(-30.0, 10.0);
|
||||
assert_eq!(s.iter().map(|s| (s.pct, s.watts as i64)).collect::<Vec<_>>(), vec![(100, 100), (90, 90), (80, 80), (70, 70)], "{s:?}");
|
||||
// a range that allows the whole ladder
|
||||
assert_eq!(plan_amd_power_steps(-60.0, 15.0).len(), 6);
|
||||
// no negative range at all: one step at 100
|
||||
assert_eq!(plan_amd_power_steps(0.0, 0.0).len(), 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn amd_clock_steps_start_at_stock_and_clamp() {
|
||||
let s = plan_amd_clock_steps(3300.0, 500.0, 3400.0);
|
||||
assert_eq!(s.iter().map(|s| s.watts as i64).collect::<Vec<_>>(), vec![3300, 2800, 2400, 2000, 1600, 1200]);
|
||||
assert_eq!(s[0].pct, 100);
|
||||
assert_eq!(s[1].pct, 85);
|
||||
// a floor of 1,500 MHz folds the two lowest steps into one
|
||||
let s = plan_amd_clock_steps(3300.0, 1500.0, 3400.0);
|
||||
assert_eq!(s.iter().map(|s| s.watts as i64).collect::<Vec<_>>(), vec![3300, 2800, 2400, 2000, 1600, 1500]);
|
||||
assert!(plan_amd_clock_steps(0.0, 0.0, 0.0).is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_amd_lever_runs_the_machine_on_percent_readbacks() {
|
||||
let steps = plan_amd_power_steps(-30.0, 10.0);
|
||||
let t = Timing { settle: Duration::from_secs(1), hold: Duration::from_secs(2), apply: Duration::from_secs(5) };
|
||||
let t0 = Instant::now();
|
||||
let mut run = Run::new(1, "amd:1:gfx1201", "1", "amd-x-1", steps, 100, 100.0, true, t, t0).with_lever(Lever::AmdPowerPct);
|
||||
let mut now = t0;
|
||||
let outs = run.tick(now, 100.0);
|
||||
assert_eq!(outs, vec![Out::Apply(100.0)]);
|
||||
// the readback is in percent: 100 within 0.5 counts, 98 does not
|
||||
now += Duration::from_secs(1);
|
||||
assert!(run.tick(now, 98.0).is_empty());
|
||||
assert!(matches!(run.phase, Phase::Applying { .. }));
|
||||
assert!(run.tick(now, 100.0).is_empty());
|
||||
assert!(matches!(run.phase, Phase::Settling { .. }));
|
||||
now += Duration::from_secs(1);
|
||||
run.tick(now, 100.0);
|
||||
run.sample_rate(17.7);
|
||||
run.sample_draw(199.0);
|
||||
run.sample_draw(198.0);
|
||||
run.sample_draw(200.0);
|
||||
now += Duration::from_secs(2);
|
||||
let outs = run.tick(now, 100.0);
|
||||
assert!(matches!(outs[0], Out::Row(ref r) if r.pct == 100 && (r.limit - 100.0).abs() < 0.1 && (r.watts - 199.0).abs() < 0.01), "{outs:?}");
|
||||
assert_eq!(outs[1], Out::Apply(90.0));
|
||||
let r = match &outs[0] { Out::Row(r) => r.clone(), _ => unreachable!() };
|
||||
assert_eq!(r.line_for("amd-x-1", Lever::AmdPowerPct), "SWEEP card=amd-x-1 lever=amd_power_pct cap=100 limit=100 watts=199.0 mhs=17.70 eff=0.0889");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn unsupported_reasons() {
|
||||
assert!(unsupported_reason_for("amd", 0.0, "1", true).is_none());
|
||||
assert!(unsupported_reason_for("amd", 0.0, "1", false).unwrap().contains("igneum-gpu-telemetry"));
|
||||
assert!(unsupported_reason("nvidia", 575.0, "0").is_none());
|
||||
assert!(unsupported_reason("nvidia", 0.0, "0").unwrap().contains("power limits"));
|
||||
assert!(unsupported_reason("apple", 0.0, "").unwrap().contains("powermetrics"));
|
||||
assert!(unsupported_reason("amd", 0.0, "1").unwrap().contains("AMD"));
|
||||
assert!(unsupported_reason("amd", 0.0, "1").unwrap().contains("manual power tuning"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
|
|
|
|||
|
|
@ -1578,6 +1578,26 @@ Reading: on all three cards the hash runs within a few percent of 1/128 of the c
|
|||
|
||||
Reading: the 9070 XT draws 199 W of its 304 W board rating (vendor figure) at 100% busy with the shader clock at its top, so the die is waiting on memory, which is the ceiling finding again; the fans at 657 rpm and 64 C are the card's own curve at that load, not a fault. Per watt the 5090 is 4.5x the 9070 XT on this program class (0.398 against 0.089 MH/W). The earlier per-watt claim from the board rating (304 W) would have read 0.058 MH/W; the measured number is 1.5x that.
|
||||
|
||||
**AMD clock and power control, and the 9070 XT sweep: OWED** (branch `opencl-rdna4-telemetry`; the project lead: "can anything be done to redeem AMD?"). The control is built and unit-tested, the measurement is not taken: the RX 9070 XT left PC 1's bus before the run. Relay probe #202 at 20:40 UTC listed only the gfx1036 and the RTX 5090 under Display; probe #203 at 20:45:34 UTC, after `pnputil /scan-devices`, the same, and the USB4 list had lost the "USB4 Router (2.0), Sonnet Technologies Breakaway Box 850T5" that was present at 17:18 UTC, so the box itself is off the link (the second eGPU fault of the day; the first, Code 43 at 17:18, needed a driver reinstall and a reboot). Nobody restarts PC 1 tonight; the project lead reseats the link in the morning. The app fell back to the gfx1036 (3.12 MH/s on `--device 1`). The PC 1 scheduler dropped the slot (job 4) at 20:46 UTC.
|
||||
|
||||
What is ready to run the moment a discrete AMD card is on the bus: `relay/playbooks/amd-card-test.ps1` (one job; parameter `$Gfx`, `gfx1201` for the 9070 XT, `gfx1100` for the RX 7900 XTX the project lead has ordered), which switches that card off in the app, checks the `--list` fold (one row, no dup row), runs `--memprobe` at 4, 64, 96 and 1024 MiB, a 120 s hash run in the worker's own serve mode on `pack-a` (within 1% of the app's rate on the 9070 XT, above), the telemetry window, then the core-clock sweep (stock, 2800, 2400, 2000, 1600, 1200 MHz) and the power-limit sweep (100, 80, 65, 50% as ADLX offsets 0, -20, -35, -50, clamped to the card's range) at 90 s each with the memory clock recorded at every step, `--reset` to factory at the end with a `--tune` read-back and a 30 s window after it, and prints the table below filled, the best MH/W point with its hash cost, whether the memory clock held, and the MH per pound (`$PricePounds`). The rows this entry owes, with the expected stock values from the measurement above:
|
||||
|
||||
| Setting | MH/s | Watts | MH/W | Temp C | Fan rpm | Memory MHz | Shader MHz |
|
||||
|---|---|---|---|---|---|---|---|
|
||||
| stock (max clock 3,290 to 3,300, power 0%) | 17.7 (measured 20:27 UTC) | 199 (measured) | 0.089 (measured) | 64 | 657 | 2,505 | 3,290 |
|
||||
| max clock 2,800 | owed | owed | owed | | | must stay 2,505 | |
|
||||
| max clock 2,400 | owed | | | | | | |
|
||||
| max clock 2,000 | owed | | | | | | |
|
||||
| max clock 1,600 | owed | | | | | | |
|
||||
| max clock 1,200 | owed | | | | | | |
|
||||
| power 80% | owed | | | | | | |
|
||||
| power 65% (if the range allows) | owed | | | | | | |
|
||||
| power 50% (if the range allows) | owed | | | | | | |
|
||||
|
||||
The control (`proto-opencl/gpu-telemetry.c`, usage header): `--tune` prints each card's max-clock range and value and the power-limit range and offset (ADLX `IADLXManualGraphicsTuning2`, `IADLXManualPowerTuning`), `--card N --set-gmax MHZ`, `--card N --set-plimit PCT` (offset from default), `--card N --reset` (`ResetToFactory`); every sample line now carries `plimit_pct` and `gmax_mhz`, the limits in force, which the app's sweep uses as its read-back. On Linux the same commands write `pp_od_clk_voltage` ("s 1 MHZ", "c") and hwmon `power1_cap` (root), verified on a fixture tree on the Mac (`--card 0 --set-plimit -20` reads back `plimit_pct 80`). Whether ADLX tuning needs elevation on Windows is NOT confirmed tonight (the card was gone before the first `--set-gmax`); the Adrenalin UI sets these unelevated, and the playbook runs unelevated so the first run answers it. The app's efficiency sweep (`sweep.rs`) now has a `Lever`: NVIDIA watts, AMD power percent (the default for AMD cards once the helper's `tune` line reports manual power tuning, `amd_tunable`), AMD max clock (the playbook's lever; it becomes the app's if the owed sweep shows it gives more MH per watt). The AMD sweep sets through the helper, no helper process and no prompt; it is gated like the NVIDIA one on the efficiency-sweep switch (this tree has no separate "Power control" setting; the gate is one line to move). Tests: `plan_amd_power_steps` clamps to the ADLX range, `plan_amd_clock_steps` starts at stock and folds the floor, the state machine runs on percent read-backs, `parse_amd_tune` on the helper's line shapes.
|
||||
|
||||
The RX 7900 XTX (gfx1100, RDNA 3) on the same path: the OpenCL worker's kernel has no architecture-specific code; on gfx1201 it compiled under the same AMD "PAL,LC" compiler the 7900 XTX uses, chose wave32 (the card reports wavefront 32 and a sub-group of 32 for a 32-item work-group) and took the local-memory exchange, and RDNA 3 reports wavefront 32 the same way, so the worker is expected to compile and self-test on gfx1100 by reasoning from the gfx1201 run, not by a run; the playbook's first stage is that self-test (`ready` line, cache FNV, 96 of 96 vector lanes). The emulator (`proto-opencl/emu`) compiles the kernel as C++ and says nothing about an AMD compiler. Expected numbers to compare against: 9070 XT 2.4 G dependent reads/s at 1 GiB, 18 MH/s, 199 W; the 7900 XTX has 960 GB/s and a 96 MB Infinity Cache (vendor figures), so its 96 MiB probe row is the one to watch.
|
||||
|
||||
**Is it the eGPU link?** No. 2.42 G loads/s x 64 B lines = 155 GB/s of DRAM traffic, forty times what a USB4 PCIe tunnel carries (about 4 GB/s, approximate); the 1 GiB buffer sits in the card's own memory (the 4 and 64 MiB cases show the card's caches at work above it, and a buffer in host memory would run below 0.1 G/s). A PCIe slot would move the per-job read-back (16 MiB per 2^21-nonce job on the old path, now gone) and nothing else; the random-read ceiling is the card's. What a PCIe slot would give: the same 18 MH/s.
|
||||
|
||||
**What changed on `opencl-rdna4`** (`proto-opencl/host.c`, `app/igneum-app/src/detect.rs`):
|
||||
|
|
|
|||
|
|
@ -61,6 +61,9 @@ proto-opencl/
|
|||
test_host.c device-free unit tests of host.c's rules (the duplicate-platform fold); run with test-host.sh
|
||||
gpu-telemetry.c igneum-gpu-telemetry: AMD power, temperature, fan, clocks and busy per card (ADLX on Windows, amdgpu sysfs on Linux),
|
||||
one line per card per sample; the app's AMD card row reads it (engine.rs amd_telemetry_line)
|
||||
and the control: --tune (ranges), --card N --set-gmax MHZ | --set-plimit PCT | --reset; the app's AMD efficiency
|
||||
sweep (sweep.rs Lever) and relay/playbooks/amd-card-test.ps1 (one AMD card through list, memprobe, hash, telemetry,
|
||||
the clock and power sweeps, restore) use it
|
||||
build.sh macOS (-framework OpenCL, or the Khronos ICD loader) and Linux (-lOpenCL)
|
||||
build.bat Windows (MSVC cl.exe + OpenCL.lib)
|
||||
WAVEFRONT.md wave32 vs wave64 on AMD, and why the kernel cannot tell the difference
|
||||
|
|
|
|||
|
|
@ -3,6 +3,14 @@
|
|||
// what it drew (the app's draw, temperature and MH per watt line came from nvidia-smi only).
|
||||
//
|
||||
// igneum-gpu-telemetry [-l SECONDS] one sample (default), or one every SECONDS until stdin closes or SIGTERM
|
||||
// igneum-gpu-telemetry --tune what each card's manual tuning allows: the max GPU clock range and value,
|
||||
// the power limit range and value (percent offset from the default), one `tune` line per card
|
||||
// igneum-gpu-telemetry --card N --set-gmax MHZ | --set-plimit PCT | --reset
|
||||
// set the max GPU clock, the power limit (offset percent, 0 = default, -20 = 80%),
|
||||
// or every tuning value back to factory, on card N (the `amd N` ordinal); prints the
|
||||
// `tune` line read back after the change. ADLX manual tuning needs no elevation
|
||||
// (confirmed on PC 1, 5 October 2026, from an unelevated job). Linux: pp_od_clk_voltage
|
||||
// ("s 1 MHZ" then "c") and hwmon power1_cap (microwatts), which need root.
|
||||
//
|
||||
// Windows: ADLX (the AMD Device Library eXtra, amdadlx64.dll, shipped with Adrenalin; vendor/adlx is the SDK clone,
|
||||
// MIT) for the metrics, keyed by the card's PCI bus from SetupAPI (the display class, matched by the same name ADLX
|
||||
|
|
@ -13,7 +21,8 @@
|
|||
//
|
||||
// Line format (space separated, every field present, a value the source cannot give prints as -):
|
||||
// amd <ordinal> bus <pci bus or address> kind integrated|discrete name "<name>" watts <W> temp_c <C> fan_rpm <rpm>
|
||||
// fan_pct <%> mclk_mhz <MHz> gclk_mhz <MHz> util_pct <%> source adlx|sysfs|perfcounter
|
||||
// fan_pct <%> mclk_mhz <MHz> gclk_mhz <MHz> util_pct <%> plimit_pct <100 + offset> gmax_mhz <MHz> source adlx|sysfs|perfcounter
|
||||
// tune <ordinal> name "<name>" gmax <MHz> gmax_range <min> <max> plimit <offset %> plimit_range <min> <max> factory 0|1 ok|<error>
|
||||
// then one `end <ms>` line per sample. The app (engine.rs amd_telemetry_line) parses it; parsers are unit-tested
|
||||
// against lines captured on PC 1.
|
||||
#define _CRT_SECURE_NO_WARNINGS
|
||||
|
|
@ -30,16 +39,18 @@ typedef struct {
|
|||
char kind[16];
|
||||
char name[128];
|
||||
double watts, tempC, fanRpm, fanPct, mclk, gclk, util; /* -1 = not available */
|
||||
double plimitPct, gmax; /* the limits in force: 100 + the power offset; the max GPU clock */
|
||||
const char* source;
|
||||
} Sample;
|
||||
|
||||
static void sampleInit(Sample* s) { memset(s, 0, sizeof(*s)); strcpy(s->bus, "-"); strcpy(s->kind, "-"); strcpy(s->name, "-"); s->watts = s->tempC = s->fanRpm = s->fanPct = s->mclk = s->gclk = s->util = -1.0; s->source = "-"; }
|
||||
static void sampleInit(Sample* s) { memset(s, 0, sizeof(*s)); strcpy(s->bus, "-"); strcpy(s->kind, "-"); strcpy(s->name, "-"); s->watts = s->tempC = s->fanRpm = s->fanPct = s->mclk = s->gclk = s->util = s->plimitPct = s->gmax = -1.0; s->source = "-"; }
|
||||
static void printNum(double v, const char* fmt) { if (v < 0) printf(" -"); else printf(fmt, v); }
|
||||
static void printSample(int ordinal, const Sample* s) {
|
||||
printf("amd %d bus %s kind %s name \"%s\" watts", ordinal, s->bus, s->kind, s->name);
|
||||
printNum(s->watts, " %.1f"); printf(" temp_c"); printNum(s->tempC, " %.1f"); printf(" fan_rpm"); printNum(s->fanRpm, " %.0f");
|
||||
printf(" fan_pct"); printNum(s->fanPct, " %.0f"); printf(" mclk_mhz"); printNum(s->mclk, " %.0f"); printf(" gclk_mhz"); printNum(s->gclk, " %.0f");
|
||||
printf(" util_pct"); printNum(s->util, " %.0f"); printf(" source %s\n", s->source);
|
||||
printf(" util_pct"); printNum(s->util, " %.0f"); printf(" plimit_pct"); printNum(s->plimitPct, " %.0f"); printf(" gmax_mhz"); printNum(s->gmax, " %.0f");
|
||||
printf(" source %s\n", s->source);
|
||||
}
|
||||
|
||||
#ifdef _WIN32
|
||||
|
|
@ -49,6 +60,9 @@ static void printSample(int ordinal, const Sample* s) {
|
|||
#include <pdh.h>
|
||||
#include "../vendor/adlx/SDK/ADLXHelper/Windows/C/ADLXHelper.h"
|
||||
#include "../vendor/adlx/SDK/Include/IPerformanceMonitoring.h"
|
||||
#include "../vendor/adlx/SDK/Include/IGPUTuning.h"
|
||||
#include "../vendor/adlx/SDK/Include/IGPUManualGFXTuning.h"
|
||||
#include "../vendor/adlx/SDK/Include/IGPUManualPowerTuning.h"
|
||||
|
||||
/* The SDK declares these three and leaves them to the platform file of each sample. */
|
||||
adlx_handle ADLX_CDECL_CALL adlx_load_library(const TCHAR* filename) { return (adlx_handle)LoadLibrary(filename); }
|
||||
|
|
@ -86,6 +100,107 @@ static int busOf(const BusEntry* b, int n, const char* name, int* taken) {
|
|||
/* ADLX: one sample of every GPU. Returns the number of lines printed, -1 when ADLX is not usable (reason printed). */
|
||||
static IADLXSystem* gSys = NULL;
|
||||
static IADLXPerformanceMonitoringServices* gPerf = NULL;
|
||||
static IADLXGPUTuningServices* gTune = NULL; /* NULL when the driver has no tuning services */
|
||||
|
||||
/* The manual tuning interfaces of one GPU; either may be NULL (not supported on the card, or an iGPU). */
|
||||
typedef struct { IADLXManualGraphicsTuning2* gfx; IADLXManualPowerTuning* power; } Tuning;
|
||||
static Tuning tuningOf(IADLXGPU* gpu) {
|
||||
Tuning tu = { NULL, NULL };
|
||||
adlx_bool ok = 0;
|
||||
IADLXInterface* raw = NULL;
|
||||
if (!gTune) return tu;
|
||||
if (ADLX_SUCCEEDED(gTune->pVtbl->IsSupportedManualGFXTuning(gTune, gpu, &ok)) && ok && ADLX_SUCCEEDED(gTune->pVtbl->GetManualGFXTuning(gTune, gpu, &raw)) && raw) {
|
||||
if (!ADLX_SUCCEEDED(raw->pVtbl->QueryInterface(raw, IID_IADLXManualGraphicsTuning2(), (void**)&tu.gfx))) tu.gfx = NULL;
|
||||
raw->pVtbl->Release(raw); raw = NULL;
|
||||
}
|
||||
if (ADLX_SUCCEEDED(gTune->pVtbl->IsSupportedManualPowerTuning(gTune, gpu, &ok)) && ok && ADLX_SUCCEEDED(gTune->pVtbl->GetManualPowerTuning(gTune, gpu, &raw)) && raw) {
|
||||
if (!ADLX_SUCCEEDED(raw->pVtbl->QueryInterface(raw, IID_IADLXManualPowerTuning(), (void**)&tu.power))) tu.power = NULL;
|
||||
raw->pVtbl->Release(raw);
|
||||
}
|
||||
return tu;
|
||||
}
|
||||
static void tuningRelease(Tuning* tu) { if (tu->gfx) tu->gfx->pVtbl->Release(tu->gfx); if (tu->power) tu->power->pVtbl->Release(tu->power); tu->gfx = NULL; tu->power = NULL; }
|
||||
|
||||
/* The limits in force, for a sample line (-1 when the card has no manual tuning). */
|
||||
static void tuningNow(IADLXGPU* gpu, double* plimitPct, double* gmax) {
|
||||
Tuning tu = tuningOf(gpu);
|
||||
adlx_int v = 0;
|
||||
*plimitPct = -1.0; *gmax = -1.0;
|
||||
if (tu.power && ADLX_SUCCEEDED(tu.power->pVtbl->GetPowerLimit(tu.power, &v))) *plimitPct = 100.0 + v;
|
||||
if (tu.gfx && ADLX_SUCCEEDED(tu.gfx->pVtbl->GetGPUMaxFrequency(tu.gfx, &v))) *gmax = v;
|
||||
tuningRelease(&tu);
|
||||
}
|
||||
|
||||
/* One `tune` line: ranges, values, whether the card is at factory settings, and the verdict of the last request. */
|
||||
static void printTune(int ordinal, IADLXGPU* gpu, const char* verdict) {
|
||||
Tuning tu = tuningOf(gpu);
|
||||
const char* name = NULL;
|
||||
ADLX_IntRange r = { 0, 0, 0 };
|
||||
adlx_int v = 0;
|
||||
adlx_bool factory = 0;
|
||||
char buf[128] = "-";
|
||||
if (ADLX_SUCCEEDED(gpu->pVtbl->Name(gpu, &name)) && name) snprintf(buf, sizeof(buf), "%s", name);
|
||||
printf("tune %d name \"%s\"", ordinal, buf);
|
||||
if (tu.gfx && ADLX_SUCCEEDED(tu.gfx->pVtbl->GetGPUMaxFrequency(tu.gfx, &v)) && ADLX_SUCCEEDED(tu.gfx->pVtbl->GetGPUMaxFrequencyRange(tu.gfx, &r))) printf(" gmax %d gmax_range %d %d", (int)v, (int)r.minValue, (int)r.maxValue);
|
||||
else printf(" gmax - gmax_range - -");
|
||||
if (tu.power && ADLX_SUCCEEDED(tu.power->pVtbl->GetPowerLimit(tu.power, &v)) && ADLX_SUCCEEDED(tu.power->pVtbl->GetPowerLimitRange(tu.power, &r))) printf(" plimit %d plimit_range %d %d", (int)v, (int)r.minValue, (int)r.maxValue);
|
||||
else printf(" plimit - plimit_range - -");
|
||||
if (gTune && ADLX_SUCCEEDED(gTune->pVtbl->IsAtFactory(gTune, gpu, &factory))) printf(" factory %d", factory ? 1 : 0); else printf(" factory -");
|
||||
printf(" %s\n", verdict);
|
||||
tuningRelease(&tu);
|
||||
}
|
||||
|
||||
/* --card N with --set-gmax, --set-plimit or --reset: apply, then print the tune line read back. */
|
||||
static int tuneCard(int card, int setGmax, int gmax, int setPlimit, int plimit, int reset) {
|
||||
IADLXGPUList* gpus = NULL;
|
||||
adlx_uint it;
|
||||
int ordinal = 0, done = 0;
|
||||
if (!gTune) { printf("tune %d error no tuning services (ADLX too old, or no AMD driver)\n", card); return 2; }
|
||||
if (!ADLX_SUCCEEDED(gSys->pVtbl->GetGPUs(gSys, &gpus)) || !gpus) { printf("tune %d error GetGPUs failed\n", card); return 2; }
|
||||
for (it = gpus->pVtbl->Begin(gpus); it != gpus->pVtbl->End(gpus); ++it, ++ordinal) {
|
||||
IADLXGPU* gpu = NULL;
|
||||
char verdict[160] = "ok";
|
||||
ADLX_RESULT r = ADLX_OK;
|
||||
if (ordinal != card) continue;
|
||||
if (!ADLX_SUCCEEDED(gpus->pVtbl->At_GPUList(gpus, it, &gpu)) || !gpu) continue;
|
||||
if (reset) {
|
||||
r = gTune->pVtbl->ResetToFactory(gTune, gpu);
|
||||
if (!ADLX_SUCCEEDED(r)) snprintf(verdict, sizeof(verdict), "error ResetToFactory returned %d", (int)r);
|
||||
} else {
|
||||
Tuning tu = tuningOf(gpu);
|
||||
if (setGmax) {
|
||||
if (!tu.gfx) snprintf(verdict, sizeof(verdict), "error no manual GFX tuning on this card");
|
||||
else { r = tu.gfx->pVtbl->SetGPUMaxFrequency(tu.gfx, gmax); if (!ADLX_SUCCEEDED(r)) snprintf(verdict, sizeof(verdict), "error SetGPUMaxFrequency(%d) returned %d", gmax, (int)r); }
|
||||
}
|
||||
if (setPlimit && verdict[0] == 'o') {
|
||||
if (!tu.power) snprintf(verdict, sizeof(verdict), "error no manual power tuning on this card");
|
||||
else { r = tu.power->pVtbl->SetPowerLimit(tu.power, plimit); if (!ADLX_SUCCEEDED(r)) snprintf(verdict, sizeof(verdict), "error SetPowerLimit(%d) returned %d", plimit, (int)r); }
|
||||
}
|
||||
tuningRelease(&tu);
|
||||
}
|
||||
printTune(ordinal, gpu, verdict);
|
||||
done = verdict[0] == 'o';
|
||||
gpu->pVtbl->Release(gpu);
|
||||
}
|
||||
gpus->pVtbl->Release(gpus);
|
||||
if (!done && ordinal <= card) printf("tune %d error no such card (%d listed)\n", card, ordinal);
|
||||
return done ? 0 : 1;
|
||||
}
|
||||
static int tuneReport(void) {
|
||||
IADLXGPUList* gpus = NULL;
|
||||
adlx_uint it;
|
||||
int ordinal = 0;
|
||||
if (!gTune) { printf("info adlx: no tuning services\n"); return 1; }
|
||||
if (!ADLX_SUCCEEDED(gSys->pVtbl->GetGPUs(gSys, &gpus)) || !gpus) return 1;
|
||||
for (it = gpus->pVtbl->Begin(gpus); it != gpus->pVtbl->End(gpus); ++it, ++ordinal) {
|
||||
IADLXGPU* gpu = NULL;
|
||||
if (!ADLX_SUCCEEDED(gpus->pVtbl->At_GPUList(gpus, it, &gpu)) || !gpu) continue;
|
||||
printTune(ordinal, gpu, "ok");
|
||||
gpu->pVtbl->Release(gpu);
|
||||
}
|
||||
gpus->pVtbl->Release(gpus);
|
||||
return 0;
|
||||
}
|
||||
static int adlxOpen(void) {
|
||||
ADLX_RESULT r = ADLXHelper_Initialize();
|
||||
if (!ADLX_SUCCEEDED(r)) { printf("info adlx: ADLXHelper_Initialize returned %d (no AMD driver with ADLX; amdadlx64.dll missing or too old)\n", (int)r); return 0; }
|
||||
|
|
@ -93,6 +208,8 @@ static int adlxOpen(void) {
|
|||
if (!gSys) { printf("info adlx: no system services\n"); return 0; }
|
||||
r = gSys->pVtbl->GetPerformanceMonitoringServices(gSys, &gPerf);
|
||||
if (!ADLX_SUCCEEDED(r) || !gPerf) { printf("info adlx: GetPerformanceMonitoringServices returned %d\n", (int)r); return 0; }
|
||||
r = gSys->pVtbl->GetGPUTuningServices(gSys, &gTune);
|
||||
if (!ADLX_SUCCEEDED(r) || !gTune) { printf("info adlx: GetGPUTuningServices returned %d (no manual tuning)\n", (int)r); gTune = NULL; }
|
||||
return 1;
|
||||
}
|
||||
static int adlxSample(const BusEntry* buses, int nBuses) {
|
||||
|
|
@ -129,6 +246,7 @@ static int adlxSample(const BusEntry* buses, int nBuses) {
|
|||
printf("info adlx: GetCurrentGPUMetrics for \"%s\" returned %d\n", s.name, (int)r);
|
||||
}
|
||||
/* fan percent: ADLX gives rpm only here; the tuning interface has the range, the app shows rpm when pct is - */
|
||||
tuningNow(gpu, &s.plimitPct, &s.gmax);
|
||||
printSample(ordinal++, &s);
|
||||
gpu->pVtbl->Release(gpu);
|
||||
}
|
||||
|
|
@ -173,15 +291,32 @@ static int pdhSample(void) {
|
|||
}
|
||||
|
||||
int main(int argc, char** argv) {
|
||||
int every = 0, i, haveAdlx;
|
||||
int every = 0, i, haveAdlx, tune = 0, card = -1, setGmax = 0, gmax = 0, setPlimit = 0, plimit = 0, reset = 0;
|
||||
BusEntry buses[32];
|
||||
int nBuses;
|
||||
for (i = 1; i < argc; ++i) if (strcmp(argv[i], "-l") == 0 && i + 1 < argc) every = atoi(argv[++i]);
|
||||
for (i = 1; i < argc; ++i) {
|
||||
if (strcmp(argv[i], "-l") == 0 && i + 1 < argc) every = atoi(argv[++i]);
|
||||
else if (strcmp(argv[i], "--tune") == 0) tune = 1;
|
||||
else if (strcmp(argv[i], "--card") == 0 && i + 1 < argc) card = atoi(argv[++i]);
|
||||
else if (strcmp(argv[i], "--set-gmax") == 0 && i + 1 < argc) { setGmax = 1; gmax = atoi(argv[++i]); }
|
||||
else if (strcmp(argv[i], "--set-plimit") == 0 && i + 1 < argc) { setPlimit = 1; plimit = atoi(argv[++i]); }
|
||||
else if (strcmp(argv[i], "--reset") == 0) reset = 1;
|
||||
}
|
||||
signal(SIGINT, onSignal); signal(SIGTERM, onSignal);
|
||||
setvbuf(stdout, NULL, _IOLBF, 0);
|
||||
nBuses = listBuses(buses, 32);
|
||||
for (i = 0; i < nBuses; ++i) printf("info display device \"%s\" bus %d\n", buses[i].name, buses[i].bus);
|
||||
haveAdlx = adlxOpen();
|
||||
if (card >= 0 || tune) {
|
||||
int rc;
|
||||
if (!haveAdlx) { printf("tune %d error no ADLX\n", card); return 2; }
|
||||
rc = (card >= 0 && (setGmax || setPlimit || reset)) ? tuneCard(card, setGmax, gmax, setPlimit, plimit, reset) : tuneReport();
|
||||
fflush(stdout);
|
||||
if (gTune) gTune->pVtbl->Release(gTune);
|
||||
if (gPerf) gPerf->pVtbl->Release(gPerf);
|
||||
ADLXHelper_Terminate();
|
||||
return rc;
|
||||
}
|
||||
do {
|
||||
double t0 = nowMs();
|
||||
int n = haveAdlx ? adlxSample(buses, nBuses) : pdhSample();
|
||||
|
|
@ -189,7 +324,7 @@ int main(int argc, char** argv) {
|
|||
fflush(stdout); /* a redirected stdout is fully buffered on the Windows CRT whatever setvbuf asks (PC 1 lost 60 s of samples at the kill) */
|
||||
if (every > 0) Sleep((DWORD)every * 1000);
|
||||
} while (every > 0 && !gStop);
|
||||
if (haveAdlx) { if (gPerf) gPerf->pVtbl->Release(gPerf); ADLXHelper_Terminate(); }
|
||||
if (haveAdlx) { if (gTune) gTune->pVtbl->Release(gTune); if (gPerf) gPerf->pVtbl->Release(gPerf); ADLXHelper_Terminate(); }
|
||||
return 0;
|
||||
}
|
||||
#else
|
||||
|
|
@ -211,6 +346,54 @@ static double dpmCurrent(const char* text) {
|
|||
}
|
||||
return -1.0;
|
||||
}
|
||||
static int writeText(const char* path, const char* text) { FILE* f = fopen(path, "w"); if (!f) return 0; fputs(text, f); return fclose(f) == 0; }
|
||||
/* The device directory of the n-th amdgpu card under root (the `amd N` ordinal of the sample lines), or 0. */
|
||||
static int sysfsCardDir(const char* root, int card, char* dev, size_t cap) {
|
||||
DIR* d = opendir(root);
|
||||
struct dirent* e;
|
||||
int ordinal = 0, found = 0;
|
||||
if (!d) return 0;
|
||||
while ((e = readdir(d)) != NULL) {
|
||||
char path[640], text[64];
|
||||
if (strncmp(e->d_name, "card", 4) != 0 || strchr(e->d_name + 4, '-')) continue;
|
||||
snprintf(path, sizeof(path), "%s/%s/device/vendor", root, e->d_name);
|
||||
if (!readText(path, text, sizeof(text)) || strtol(text, NULL, 16) != 0x1002) continue;
|
||||
if (ordinal++ == card) { snprintf(dev, cap, "%s/%s/device", root, e->d_name); found = 1; break; }
|
||||
}
|
||||
closedir(d);
|
||||
return found;
|
||||
}
|
||||
/* --card N --set-gmax MHZ: pp_od_clk_voltage "s 1 MHZ" then "c"; --set-plimit PCT: hwmon power1_cap = default x (100 + PCT) / 100;
|
||||
* --reset: "r" then "c" and power1_cap = power1_cap_default. Root is needed for every write; the verdict says when it is not. */
|
||||
static int sysfsTune(const char* root, int card, int setGmax, int gmax, int setPlimit, int plimit, int reset) {
|
||||
char dev[640], path[900], text[4096], hw[700] = "";
|
||||
DIR* d;
|
||||
struct dirent* e;
|
||||
const char* verdict = "ok";
|
||||
if (!sysfsCardDir(root, card, dev, sizeof(dev))) { printf("tune %d error no such card\n", card); return 1; }
|
||||
snprintf(path, sizeof(path), "%s/hwmon", dev);
|
||||
d = opendir(path);
|
||||
if (d) { while ((e = readdir(d)) != NULL) if (strncmp(e->d_name, "hwmon", 5) == 0) { snprintf(hw, sizeof(hw), "%s/%s", path, e->d_name); break; } closedir(d); }
|
||||
snprintf(path, sizeof(path), "%s/pp_od_clk_voltage", dev);
|
||||
if (reset) {
|
||||
if (!writeText(path, "r\n") || !writeText(path, "c\n")) verdict = "error pp_od_clk_voltage reset (root needed)";
|
||||
if (hw[0]) { char cap[64], dp[900]; snprintf(dp, sizeof(dp), "%s/power1_cap_default", hw); if (readText(dp, cap, sizeof(cap))) { snprintf(dp, sizeof(dp), "%s/power1_cap", hw); if (!writeText(dp, cap)) verdict = "error power1_cap reset (root needed)"; } }
|
||||
} else {
|
||||
if (setGmax) { char cmd[64]; snprintf(cmd, sizeof(cmd), "s 1 %d\n", gmax); if (!writeText(path, cmd) || !writeText(path, "c\n")) verdict = "error pp_od_clk_voltage write (root needed, or the clock is outside the card's range)"; }
|
||||
if (setPlimit && verdict[0] == 'o') {
|
||||
char dp[900], cap[64];
|
||||
if (!hw[0]) verdict = "error no hwmon";
|
||||
else { snprintf(dp, sizeof(dp), "%s/power1_cap_default", hw); if (!readText(dp, cap, sizeof(cap))) verdict = "error no power1_cap_default"; else { double uw = atof(cap) * (100.0 + plimit) / 100.0; snprintf(cap, sizeof(cap), "%.0f\n", uw); snprintf(dp, sizeof(dp), "%s/power1_cap", hw); if (!writeText(dp, cap)) verdict = "error power1_cap write (root needed)"; } }
|
||||
}
|
||||
}
|
||||
{
|
||||
double g = -1, pc = -1, pd = -1;
|
||||
snprintf(path, sizeof(path), "%s/pp_od_clk_voltage", dev); if (readText(path, text, sizeof(text))) { const char* s1 = strstr(text, "1:"); if (s1) g = atof(s1 + 2); }
|
||||
if (hw[0]) { snprintf(path, sizeof(path), "%s/power1_cap", hw); pc = readNumber(path); snprintf(path, sizeof(path), "%s/power1_cap_default", hw); pd = readNumber(path); }
|
||||
printf("tune %d name \"amdgpu\" gmax %.0f gmax_range - - plimit %.0f plimit_range - - factory - %s\n", card, g, (pc > 0 && pd > 0) ? 100.0 * pc / pd - 100.0 : -1.0, verdict);
|
||||
}
|
||||
return verdict[0] == 'o' ? 0 : 1;
|
||||
}
|
||||
static int sysfsSample(const char* root) {
|
||||
DIR* d = opendir(root);
|
||||
struct dirent* e;
|
||||
|
|
@ -251,17 +434,31 @@ static int sysfsSample(const char* root) {
|
|||
snprintf(path, sizeof(path), "%s/pp_dpm_mclk", dev); if (readText(path, text, sizeof(text))) s.mclk = dpmCurrent(text);
|
||||
snprintf(path, sizeof(path), "%s/pp_dpm_sclk", dev); if (readText(path, text, sizeof(text))) s.gclk = dpmCurrent(text);
|
||||
snprintf(path, sizeof(path), "%s/gpu_busy_percent", dev); s.util = readNumber(path);
|
||||
{ /* the limits in force: power1_cap against its default, the OD max clock from pp_od_clk_voltage */
|
||||
char hp[900]; double pc = -1, pd = -1;
|
||||
snprintf(hp, sizeof(hp), "%s/hwmon", dev); hw = opendir(hp);
|
||||
if (hw) { while ((he = readdir(hw)) != NULL) if (strncmp(he->d_name, "hwmon", 5) == 0) { char q[1000]; snprintf(q, sizeof(q), "%s/%s/power1_cap", hp, he->d_name); pc = readNumber(q); snprintf(q, sizeof(q), "%s/%s/power1_cap_default", hp, he->d_name); pd = readNumber(q); break; } closedir(hw); }
|
||||
if (pc > 0 && pd > 0) s.plimitPct = 100.0 * pc / pd;
|
||||
snprintf(hp, sizeof(hp), "%s/pp_od_clk_voltage", dev); if (readText(hp, text, sizeof(text))) { const char* s1 = strstr(text, "1:"); if (s1) s.gmax = atof(s1 + 2); }
|
||||
}
|
||||
printSample(ordinal++, &s);
|
||||
}
|
||||
closedir(d);
|
||||
return ordinal;
|
||||
}
|
||||
int main(int argc, char** argv) {
|
||||
int every = 0, i;
|
||||
int every = 0, i, card = -1, setGmax = 0, gmax = 0, setPlimit = 0, plimit = 0, reset = 0;
|
||||
const char* root = getenv("IGNEUM_DRM_ROOT") ? getenv("IGNEUM_DRM_ROOT") : "/sys/class/drm"; /* a fixture tree for tests */
|
||||
for (i = 1; i < argc; ++i) if (strcmp(argv[i], "-l") == 0 && i + 1 < argc) every = atoi(argv[++i]);
|
||||
for (i = 1; i < argc; ++i) {
|
||||
if (strcmp(argv[i], "-l") == 0 && i + 1 < argc) every = atoi(argv[++i]);
|
||||
else if (strcmp(argv[i], "--card") == 0 && i + 1 < argc) card = atoi(argv[++i]);
|
||||
else if (strcmp(argv[i], "--set-gmax") == 0 && i + 1 < argc) { setGmax = 1; gmax = atoi(argv[++i]); }
|
||||
else if (strcmp(argv[i], "--set-plimit") == 0 && i + 1 < argc) { setPlimit = 1; plimit = atoi(argv[++i]); }
|
||||
else if (strcmp(argv[i], "--reset") == 0) reset = 1;
|
||||
}
|
||||
signal(SIGINT, onSignal); signal(SIGTERM, onSignal);
|
||||
setvbuf(stdout, NULL, _IOLBF, 0);
|
||||
if (card >= 0) return sysfsTune(root, card, setGmax, gmax, setPlimit, plimit, reset);
|
||||
do {
|
||||
double t0 = nowMs();
|
||||
int n = sysfsSample(root);
|
||||
|
|
|
|||
|
|
@ -207,6 +207,7 @@ typedef struct {
|
|||
int memprobe; // --memprobe: dependent-load latency and throughput, independent-load throughput and an ALU
|
||||
// chain on the chosen device, no pack needed (5 October 2026, the 9070 XT on the eGPU)
|
||||
int probeMib; // --probe-mib N: --memprobe at that one buffer size only (default 0 = 4, 64 and 1024 MiB)
|
||||
const char* probeList; // --probe-mib A,B,C: a list of sizes
|
||||
} Options;
|
||||
|
||||
static int packMib(void) { return (int)(((1ull << IGNEUM_DATASET_LOG2) * 4ull) >> 20); }
|
||||
|
|
@ -248,7 +249,7 @@ static Options parseArgs(int argc, char** argv) {
|
|||
int i;
|
||||
o.datasetMib = 1024; o.batchLog2 = 24; o.batches = 5; o.groupWarps = 1; o.sweep = 0; o.device = -1;
|
||||
o.exchange = 0; o.list = 0; o.timeWall = -1; o.kernelPath = IGNEUM_KERNEL_PATH; o.extraOpts = ""; o.serve = 0; o.noPrepare = 0; o.kernelGiven = 0; o.vendor = NULL; o.packDir = NULL;
|
||||
o.readback = (getenv("IGNEUM_READBACK") && strcmp(getenv("IGNEUM_READBACK"), "full") == 0) ? 1 : 0; o.memprobe = 0; o.probeMib = 0;
|
||||
o.readback = (getenv("IGNEUM_READBACK") && strcmp(getenv("IGNEUM_READBACK"), "full") == 0) ? 1 : 0; o.memprobe = 0; o.probeMib = 0; o.probeList = NULL;
|
||||
for (i = 1; i < argc; ++i) {
|
||||
const char* a = argv[i];
|
||||
int needs = (strcmp(a, "--dataset-mib") == 0 || strcmp(a, "--batch-log2") == 0 || strcmp(a, "--batches") == 0 ||
|
||||
|
|
@ -274,7 +275,7 @@ static Options parseArgs(int argc, char** argv) {
|
|||
else { printf("--readback must be select or full\n"); exit(2); }
|
||||
}
|
||||
else if (strcmp(a, "--memprobe") == 0) o.memprobe = 1;
|
||||
else if (strcmp(a, "--probe-mib") == 0) { if (i + 1 >= argc) { usage(); exit(2); } o.probeMib = atoi(argv[++i]); }
|
||||
else if (strcmp(a, "--probe-mib") == 0) { if (i + 1 >= argc) { usage(); exit(2); } if (strchr(argv[i + 1], ',')) o.probeList = argv[++i]; else o.probeMib = atoi(argv[++i]); }
|
||||
else if (strcmp(a, "--build-opts") == 0) o.extraOpts = argv[++i];
|
||||
else if (strcmp(a, "--time") == 0) {
|
||||
const char* m = argv[++i];
|
||||
|
|
@ -1658,7 +1659,7 @@ static int runMemprobe(Device* dv, const DeviceInfo* di, const Options* o) {
|
|||
cl_program prog;
|
||||
cl_kernel kFill, kChase, kIndep, kAlu, kLine, kStream;
|
||||
size_t srcLen = strlen(PROBE_SRC);
|
||||
int sizes[3] = { 4, 64, 1024 }, nSizes = 3, si;
|
||||
int sizes[8] = { 4, 64, 1024, 0, 0, 0, 0, 0 }, nSizes = 3, si;
|
||||
size_t lanesList[8] = { 256, 1024, 1u << 12, 1u << 14, 1u << 16, 1u << 18, 1u << 20, 1u << 22 };
|
||||
const int nLanes = 8;
|
||||
size_t groups[2] = { 32, 256 };
|
||||
|
|
@ -1666,6 +1667,13 @@ static int runMemprobe(Device* dv, const DeviceInfo* di, const Options* o) {
|
|||
const size_t maxLanes = 1u << 22;
|
||||
cl_mem dOut;
|
||||
if (o->probeMib > 0) { sizes[0] = o->probeMib; nSizes = 1; }
|
||||
if (o->probeList && o->probeList[0]) {
|
||||
/* --probe-mib 4,64,96,1024: a list (the 7900 XTX playbook probes its 96 MB Infinity Cache size) */
|
||||
const char* p = o->probeList;
|
||||
nSizes = 0;
|
||||
while (*p && nSizes < 8) { int v = atoi(p); if (v > 0) sizes[nSizes++] = v; while (*p && *p != ',') ++p; if (*p == ',') ++p; }
|
||||
if (nSizes == 0) { sizes[0] = 1024; nSizes = 1; }
|
||||
}
|
||||
prog = clCreateProgramWithSource(dv->ctx, 1, &PROBE_SRC, &srcLen, &err); CL_CHECK_ERR(err, "clCreateProgramWithSource probe");
|
||||
err = clBuildProgram(prog, 1, &di->device, "-cl-std=CL1.2", NULL, NULL);
|
||||
if (err != CL_SUCCESS) {
|
||||
|
|
|
|||
146
relay/playbooks/amd-card-test.ps1
Normal file
146
relay/playbooks/amd-card-test.ps1
Normal file
|
|
@ -0,0 +1,146 @@
|
|||
# amd-card-test.ps1: one AMD card through the whole test on a Windows PC running Igneum Miner (5 October 2026).
|
||||
# Published as a run job (packaging/ota/publish-jobs.sh add --kind run ... --script this file) after a fetch job has put
|
||||
# the kit zip (igneum-worker-opencl.exe, igneum-gpu-telemetry.exe, kernel_bound.cl, pack-a\) under jobs\<KitJob>\kit.
|
||||
# Parameters at the top (a job script takes no arguments; edit the three lines or sed them before publishing):
|
||||
# $Gfx the card's OpenCL name as the worker lists it (gfx1201 = RX 9070 XT, gfx1100 = RX 7900 XTX)
|
||||
# $KitJob the fetch job id whose kit folder holds the exes
|
||||
# $PricePounds the card's price for the MH per pound line (0 = not stated)
|
||||
# What it does, with the card under test switched OFF in the app (POST /api/cards, that key only; the other cards keep
|
||||
# mining) and switched back on at the end: the --list fold check (one row for the card, no dup row), --memprobe at
|
||||
# 4, 64, 96 and 1024 MiB, a 120 s hash run (the worker's own serve mode on pack-a, 2^21-nonce jobs: within 1% of the
|
||||
# app's STATUS rate on the 9070 XT, bench-log 5 October 2026), the telemetry window during it (watts, temperature,
|
||||
# fan, memory clock), then the core-clock sweep (stock, 2800, 2400, 2000, 1600, 1200 MHz) and the power-limit sweep
|
||||
# (100, 80, 65, 50% as ADLX offsets 0, -20, -35, -50, clamped to the card's range) at 90 s per step with the hash
|
||||
# run as the load, every setting reset to factory at the end and read back. Output: RESULT lines, one bench-log table,
|
||||
# and a RESULT line with the MH/W at stock and the MH per pound.
|
||||
$Gfx = 'gfx1201'
|
||||
$KitJob = 'amd-kit-1'
|
||||
$PricePounds = 0
|
||||
$ErrorActionPreference = 'Continue'
|
||||
$kit = Join-Path $env:IGNEUM_APP_DIR ("jobs\" + $KitJob + "\kit")
|
||||
$worker = Join-Path $kit 'igneum-worker-opencl.exe'
|
||||
$tele = Join-Path $kit 'igneum-gpu-telemetry.exe'
|
||||
$pack = Join-Path $kit 'pack-a'
|
||||
$deadline = (Get-Date).AddMinutes(4)
|
||||
while (-not ((Test-Path $worker) -and (Test-Path $tele)) -and (Get-Date) -lt $deadline) { Start-Sleep -Seconds 5 }
|
||||
if (-not (Test-Path $worker)) { "RESULT no worker at $worker"; exit 1 }
|
||||
if (-not (Test-Path $tele)) { "RESULT no telemetry helper at $tele"; exit 1 }
|
||||
"RESULT kit worker sha256 " + (Get-FileHash $worker -Algorithm SHA256).Hash.ToLower() + " telemetry " + (Get-FileHash $tele -Algorithm SHA256).Hash.ToLower()
|
||||
Set-Location $kit
|
||||
# --- the card in the app: off for the test, on again at the end (the other cards are never touched)
|
||||
$base = ""; try { $base = (Get-Content (Join-Path $env:LOCALAPPDATA "igneum\app\app.url") -Raw).Trim().TrimEnd("/") } catch {}
|
||||
$amdKeys = @()
|
||||
if ($base) { try { $s = Invoke-RestMethod -Uri "$base/api/state" -TimeoutSec 20; foreach ($c in $s.mining.cards) { "RESULT card before: " + $c.key + " | " + $c.state + " | " + $c.hash_now + " MH/s | enabled " + $c.enabled; if ($c.key -like ("amd:*" + $Gfx + "*") -and $c.enabled) { $amdKeys += @{ key = $c.key; enabled = $false; identities = $c.identities } } } } catch { "RESULT the app is not answering: " + $_.Exception.Message } }
|
||||
function Set-Cards($list) { if ($list.Count -eq 0) { return }; $body = @{ cards = $list } | ConvertTo-Json -Depth 4; try { $r = Invoke-RestMethod -Method Post -Uri "$base/api/cards" -ContentType 'application/json' -Body $body -TimeoutSec 30; "RESULT api/cards: " + ($r | ConvertTo-Json -Compress) } catch { "RESULT api/cards failed: " + $_.Exception.Message } }
|
||||
Set-Cards $amdKeys
|
||||
if ($amdKeys.Count -gt 0) { Start-Sleep -Seconds 15 }
|
||||
$busy = @(Get-Process -Name igneum-worker-opencl -ErrorAction SilentlyContinue | Where-Object { $_.Path -notlike "$kit*" })
|
||||
# an OpenCL worker may still drive ANOTHER AMD card (the iGPU); only a worker on this card's index is a problem, checked below by hash
|
||||
# --- 1. the list fold check
|
||||
"=== STAGE list fold check ($Gfx)"
|
||||
$list = & $worker --list 2>&1
|
||||
$list | ForEach-Object { $_ }
|
||||
$rows = @($list | Where-Object { $_ -match ('^\s*\*?\[(\d+)\] ' + [regex]::Escape($Gfx)) })
|
||||
$dups = @($list | Where-Object { $_ -match ('^\s*dup \[(\d+)\] ' + [regex]::Escape($Gfx)) })
|
||||
$idx = if ($rows.Count -gt 0) { ($rows[0] -replace '^\s*\*?\[(\d+)\].*','$1') } else { '' }
|
||||
"RESULT list: $($rows.Count) row(s) for $Gfx (index $idx), $($dups.Count) hidden duplicate(s)" + $(if ($rows.Count -eq 1) { " OK" } else { " NOT ONE ROW" })
|
||||
if (-not $idx) { Set-Cards @($amdKeys | ForEach-Object { @{ key = $_.key; enabled = $true; identities = $_.identities } }); exit 1 }
|
||||
# the telemetry ordinal of the card: the discrete AMD card whose name is not the integrated one
|
||||
$t0 = & $tele 2>&1
|
||||
$t0 | ForEach-Object { "TELE0 " + $_ }
|
||||
$ord = ''; $cardName = ''
|
||||
foreach ($l in $t0) { if ($l -match '^amd (\d+) bus (\S+) kind discrete name "([^"]*)"') { $ord = $matches[1]; $cardName = $matches[3]; break } }
|
||||
"RESULT telemetry ordinal $ord name `"$cardName`""
|
||||
# --- 2. memprobe
|
||||
"=== STAGE memprobe $Gfx [$idx] at 4, 64, 96, 1024 MiB"
|
||||
& $worker --device $idx --memprobe --probe-mib 4,64,96,1024 2>&1 | Where-Object { $_ -match '^\| (chase|indep|line|stream|alu)|^memprobe kernel' }
|
||||
# --- the load: serve mode on pack-a, jobs of 2^21 nonces, as long as the script needs it (a background process fed from a file)
|
||||
$seeds = Get-Content (Join-Path $pack 'seeds.txt')
|
||||
$epoch = ($seeds | Where-Object { $_ -match '^epoch_seed_hex\s+(\S+)' } | ForEach-Object { $matches[1] } | Select-Object -First 1)
|
||||
$day = ($seeds | Where-Object { $_ -match '^day_seed_hex\s+(\S+)' } | ForEach-Object { $matches[1] } | Select-Object -First 1)
|
||||
$jobFile = Join-Path $env:IGNEUM_JOB_DIR 'jobs.txt'
|
||||
$prehash = '7a1c0f3e5b9d2468ace0f1b2c3d4e5f60718293a4b5c6d7e8f9a0b1c2d3e4f50'
|
||||
$lines = New-Object System.Collections.Generic.List[string]
|
||||
for ($j = 1; $j -le 40000; $j++) { $start = [uint64]($j * 4294967296 + 1048576); $lines.Add("job $j $prehash 0000100000000000 $start 2097152 $epoch $day") }
|
||||
$lines.Add('quit')
|
||||
[IO.File]::WriteAllLines($jobFile, $lines)
|
||||
$serveOut = Join-Path $env:IGNEUM_JOB_DIR 'serve.log'
|
||||
$serve = Start-Process -FilePath $worker -ArgumentList '--serve','--pack',"`"$pack`"",'--device',$idx -RedirectStandardInput $jobFile -RedirectStandardOutput $serveOut -NoNewWindow -PassThru
|
||||
$teleOut = Join-Path $env:IGNEUM_JOB_DIR 'tele.log'
|
||||
$teleP = Start-Process -FilePath $tele -ArgumentList '-l','5' -RedirectStandardOutput $teleOut -NoNewWindow -PassThru
|
||||
Start-Sleep -Seconds 20
|
||||
$ready = Get-Content $serveOut | Where-Object { $_ -match '^ready|^error' } | Select-Object -First 1
|
||||
"RESULT serve: $ready"
|
||||
if (-not ($ready -match '^ready')) { Stop-Process -Id $serve.Id -Force -ErrorAction SilentlyContinue; Stop-Process -Id $teleP.Id -Force -ErrorAction SilentlyContinue; Set-Cards @($amdKeys | ForEach-Object { @{ key = $_.key; enabled = $true; identities = $_.identities } }); exit 1 }
|
||||
# a window: mean MH/s from the done lines and the telemetry means over the last $sec seconds
|
||||
function Window($sec, $label) {
|
||||
$from = (Get-Date)
|
||||
$n0 = @(Get-Content $serveOut | Where-Object { $_ -match '^done ' }).Count
|
||||
Start-Sleep -Seconds $sec
|
||||
$done = @(Get-Content $serveOut | Where-Object { $_ -match '^done \d+ \d+ ([\d.]+)$' } | Select-Object -Skip $n0 | ForEach-Object { [double]($_ -replace '^done \d+ \d+ ([\d.]+)$','$1') })
|
||||
$tl = @(Get-Content $teleOut | Where-Object { $_ -match ('^amd ' + $ord + ' ') } | Select-Object -Last ([int]($sec / 5)))
|
||||
$num = { param($k) @($tl | ForEach-Object { if ($_ -match (' ' + $k + ' (\S+)')) { $matches[1] } } | Where-Object { $_ -ne '-' } | ForEach-Object { [double]$_ }) }
|
||||
$avg = { param($a, $d) if ($a.Count -gt 0) { [math]::Round(($a | Measure-Object -Average).Average, $d) } else { -1 } }
|
||||
$mhs = if ($done.Count -gt 0) { [math]::Round(2097152.0 / (($done | Measure-Object -Average).Average) / 1000.0, 2) } else { -1 }
|
||||
$w = & $avg (& $num 'watts') 1; $t = & $avg (& $num 'temp_c') 0; $f = & $avg (& $num 'fan_rpm') 0; $m = & $avg (& $num 'mclk_mhz') 0; $g = & $avg (& $num 'gclk_mhz') 0; $pl = & $avg (& $num 'plimit_pct') 0; $gm = & $avg (& $num 'gmax_mhz') 0
|
||||
$eff = if ($mhs -gt 0 -and $w -gt 0) { [math]::Round($mhs / $w, 4) } else { -1 }
|
||||
"RESULT window $label : jobs $($done.Count) mhs $mhs watts $w temp_c $t fan_rpm $f mclk_mhz $m gclk_mhz $g plimit_pct $pl gmax_mhz $gm eff $eff"
|
||||
return [pscustomobject]@{ label = $label; mhs = $mhs; watts = $w; temp = $t; fan = $f; mclk = $m; gclk = $g; plimit = $pl; gmax = $gm; eff = $eff }
|
||||
}
|
||||
"=== STAGE 120 s hash and telemetry at stock"
|
||||
$stock = Window 120 'stock'
|
||||
# --- 3. the tuning ranges, then the sweeps
|
||||
"=== STAGE tuning ranges"
|
||||
$tune = & $tele --tune 2>&1
|
||||
$tune | ForEach-Object { "TUNE " + $_ }
|
||||
$tl = $tune | Where-Object { $_ -match ('^tune ' + $ord + ' ') } | Select-Object -First 1
|
||||
$gmaxStock = -1; $gRange = @(-1, -1); $pRange = @(0, 0)
|
||||
if ($tl -match 'gmax (\S+) gmax_range (\S+) (\S+) plimit (\S+) plimit_range (\S+) (\S+)') { $gmaxStock = [int]$matches[1]; $gRange = @([int]$matches[2], [int]$matches[3]); $pRange = @([int]$matches[5], [int]$matches[6]) }
|
||||
"RESULT tune: stock gmax $gmaxStock MHz, gmax range $($gRange[0])..$($gRange[1]), power offset range $($pRange[0])..$($pRange[1]) %"
|
||||
$table = @()
|
||||
$table += $stock
|
||||
if ($gmaxStock -gt 0) {
|
||||
foreach ($mhz in @($gmaxStock, 2800, 2400, 2000, 1600, 1200)) {
|
||||
$v = [math]::Max($gRange[0], [math]::Min($gRange[1], $mhz))
|
||||
$r = & $tele --card $ord --set-gmax $v 2>&1
|
||||
$verdict = ($r | Where-Object { $_ -match '^tune ' } | Select-Object -First 1)
|
||||
"RESULT set gmax $v (asked $mhz): $verdict"
|
||||
if ($verdict -match ' error ') { continue }
|
||||
$row = Window 90 ("gmax " + $v)
|
||||
$table += $row
|
||||
}
|
||||
$r = & $tele --card $ord --reset 2>&1; "RESULT reset after the clock sweep: " + ($r | Where-Object { $_ -match '^tune ' } | Select-Object -First 1)
|
||||
Start-Sleep -Seconds 10
|
||||
}
|
||||
foreach ($pct in @(100, 80, 65, 50)) {
|
||||
$off = $pct - 100
|
||||
$v = [math]::Max($pRange[0], [math]::Min($pRange[1], $off))
|
||||
$r = & $tele --card $ord --set-plimit $v 2>&1
|
||||
$verdict = ($r | Where-Object { $_ -match '^tune ' } | Select-Object -First 1)
|
||||
"RESULT set plimit $v % (asked $off %): $verdict"
|
||||
if ($verdict -match ' error ') { continue }
|
||||
$row = Window 90 ("plimit " + (100 + $v) + "%")
|
||||
$table += $row
|
||||
}
|
||||
# --- restore and verify
|
||||
"=== STAGE restore"
|
||||
$r = & $tele --card $ord --reset 2>&1; "RESULT reset at the end: " + ($r | Where-Object { $_ -match '^tune ' } | Select-Object -First 1)
|
||||
Start-Sleep -Seconds 10
|
||||
$after = & $tele --tune 2>&1 | Where-Object { $_ -match ('^tune ' + $ord + ' ') } | Select-Object -First 1
|
||||
"RESULT tune after restore: $after" + $(if ($after -match ' factory 1 ') { ' RESTORED (factory 1)' } elseif ($after -match ('gmax ' + $gmaxStock + ' ') -and $after -match ' plimit 0 ') { ' RESTORED (stock values)' } else { ' NOT VERIFIED' })
|
||||
$verify = Window 30 'after restore'
|
||||
Stop-Process -Id $serve.Id -Force -ErrorAction SilentlyContinue; Stop-Process -Id $teleP.Id -Force -ErrorAction SilentlyContinue
|
||||
# --- the table
|
||||
"=== STAGE table"
|
||||
"| Setting | MH/s | Watts | MH/W | Temp C | Fan rpm | Memory MHz | Shader MHz | Power limit % | Max clock MHz |"
|
||||
"|---|---|---|---|---|---|---|---|---|---|"
|
||||
foreach ($row in $table + @($verify)) { "| " + $row.label + " | " + $row.mhs + " | " + $row.watts + " | " + $row.eff + " | " + $row.temp + " | " + $row.fan + " | " + $row.mclk + " | " + $row.gclk + " | " + $row.plimit + " | " + $row.gmax + " |" }
|
||||
$best = $table | Where-Object { $_.eff -gt 0 } | Sort-Object eff -Descending | Select-Object -First 1
|
||||
if ($best) { "RESULT best MH/W: " + $best.label + " at " + $best.eff + " MH/W (" + $best.mhs + " MH/s, " + $best.watts + " W); stock " + $stock.eff + " MH/W (" + $stock.mhs + " MH/s, " + $stock.watts + " W); hash cost of the best point " + [math]::Round(100 * (1 - $best.mhs / $stock.mhs), 1) + " %" }
|
||||
$mclkDrop = @($table | Where-Object { $_.mclk -gt 0 -and $stock.mclk -gt 0 -and $_.mclk -lt 0.98 * $stock.mclk })
|
||||
"RESULT memory clock: stock " + $stock.mclk + " MHz; " + $(if ($mclkDrop.Count -gt 0) { "DROPPED at " + (($mclkDrop | ForEach-Object { $_.label + " (" + $_.mclk + ")" }) -join ', ') } else { "held at every step" })
|
||||
"RESULT $Gfx `"$cardName`" stock: " + $stock.mhs + " MH/s at " + $stock.watts + " W = " + $stock.eff + " MH/W" + $(if ($PricePounds -gt 0) { "; " + [math]::Round($stock.mhs / $PricePounds, 4) + " MH per pound at " + $PricePounds + " pounds" } else { "; MH per pound: price not stated" })
|
||||
# --- the card back on in the app
|
||||
Set-Cards @($amdKeys | ForEach-Object { @{ key = $_.key; enabled = $true; identities = $_.identities } })
|
||||
if ($amdKeys.Count -gt 0) { Start-Sleep -Seconds 60; try { $s = Invoke-RestMethod -Uri "$base/api/state" -TimeoutSec 20; foreach ($c in $s.mining.cards) { "RESULT card after: " + $c.key + " | " + $c.state + " | " + $c.hash_now + " MH/s | enabled " + $c.enabled } } catch {} }
|
||||
exit 0
|
||||
Loading…
Reference in a new issue