1324 lines
66 KiB
Rust
1324 lines
66 KiB
Rust
//! Ember Tune: every card tuned for MH per watt out of the box, and the fleet's results folded into a prior that a
|
||
//! new card starts from (docs/plans/ember-tune.md). Two knobs per card: the power limit (percent of the card's
|
||
//! default) and the core clock cap (MHz; 0 = unlocked). The memory clock is never touched, and a step that drags it
|
||
//! down is marked and cannot win. This file is the logic, driven by an explicit clock so the tests run without a
|
||
//! card: the plans (full, confirm, baseline), the per-step rows with their marks, the choice rule, the fleet record,
|
||
//! the prior lookup and the state machine. The engine (src/engine.rs, `tick_tune` and the `Cmd::Tune*` commands)
|
||
//! owns the processes: NVIDIA through nvidia-smi (`-pl`, `-lgc 0,<mhz>`, `-rgc`; administrator rights, so only with
|
||
//! Power control on), AMD through igneum-gpu-telemetry (`--set-plimit`, `--set-gmax`, `--reset`; no elevation on
|
||
//! Windows), Apple measure only.
|
||
//!
|
||
//! The lines in the app log (and on stdout under --sweep, which the PC job reads):
|
||
//! TUNE start card=<label> name=<name> plan=full|confirm|baseline steps=<n> clock_max=<MHz> default=<W> ...
|
||
//! TUNE card=<label> step=<i> clock=<MHz|0> cap=<pct> limit=<W> watts=<W> mhs=<x> eff=<MH/W> gclk=<MHz> mclk=<MHz> tmax=<C> mark=ok|...
|
||
//! TUNE chosen card=<label> clock=<MHz|0> cap=<pct> limit=<W> watts=<W> mhs=<x> eff=<MH/W>
|
||
//! TUNE {json} the fleet record (one per finished plan; the intake receives it with the log upload)
|
||
//! TUNE aborted card=<label> reason=<text>
|
||
|
||
use std::time::{Duration, Instant};
|
||
|
||
/// The power steps, percent of the card's default limit, highest first (the same ladder as src/sweep.rs).
|
||
pub const POWER_STEPS_PCT: [u32; 6] = [100, 90, 80, 70, 60, 50];
|
||
/// The clock steps, percent of the card's maximum core clock, highest first; 100 = unlocked (the card's own boost).
|
||
/// 6 October 2026, run 6: the 5090's best MH/W sat on the 60% floor (1,854 MHz: 0.563 MH/W, the rate within 0.15%),
|
||
/// so the ladder and the floor go to 45% of the maximum; the 1% rate tolerance is the guard below that
|
||
pub const CLOCK_STEPS_PCT: [u32; 7] = [100, 90, 80, 70, 60, 50, 45];
|
||
/// A card's clock floor when the vendor reports none: this share of its maximum core clock.
|
||
pub const CLOCK_FLOOR_PCT: u32 = 45;
|
||
/// A point may lose this much rate against the fastest point and still win on MH per watt (the manifest can change it).
|
||
pub const RATE_TOLERANCE_PCT: f64 = 1.0;
|
||
/// A step whose hottest GPU reading reaches this is marked hot and cannot win (the engine aborts at 90).
|
||
pub const HOT_C: f64 = 85.0;
|
||
/// A step whose mean memory clock falls below this share of the baseline's is marked and cannot win.
|
||
pub const MCLK_HOLD: f64 = 0.95;
|
||
/// A fleet prior is used by a new card when it carries at least this many samples (the manifest can change it).
|
||
pub const PRIOR_MIN_SAMPLES: u32 = 5;
|
||
/// A confirm check that beats its prior by more than this on MH per watt asks for the full plan.
|
||
pub const CONFIRM_GAIN_PCT: f64 = 1.0;
|
||
/// A tune repeats this often (the manifest can change it).
|
||
pub const PERIOD_S: u64 = 7 * 86_400;
|
||
/// The worker must have mined this long before a tune starts, and this much must remain to the hour boundary.
|
||
pub const STABLE_S: u64 = 120;
|
||
pub const NEEDS_S: i64 = 600;
|
||
|
||
/// What a card's vendor allows (never exceeded, never undercut).
|
||
#[derive(Clone, Debug, Default, PartialEq)]
|
||
pub struct Limits {
|
||
pub power_default_w: f64,
|
||
pub power_min_w: f64,
|
||
pub power_max_w: f64,
|
||
/// the maximum core clock the vendor reports (nvidia-smi clocks.max.gr; the ADLX gmax_range top); 0 = unknown
|
||
pub clock_max_mhz: u32,
|
||
/// the lowest cap the vendor allows (the ADLX gmax_range floor); 0 = CLOCK_FLOOR_PCT of the maximum
|
||
pub clock_min_mhz: u32,
|
||
/// Ember 2: the memory clock the card runs at by default and the vendor's maximum (nvidia-smi clocks.mem and
|
||
/// clocks.max.mem); 0 = no memory knob (AMD through ADLX on RDNA 4 exposes none)
|
||
pub mem_default_mhz: u32,
|
||
pub mem_max_mhz: u32,
|
||
}
|
||
|
||
impl Limits {
|
||
pub fn clock_floor(&self) -> u32 {
|
||
if self.clock_min_mhz > 0 {
|
||
self.clock_min_mhz
|
||
} else {
|
||
self.clock_max_mhz * CLOCK_FLOOR_PCT / 100
|
||
}
|
||
}
|
||
/// A clock cap inside the vendor's range; 0 stays 0 (unlocked).
|
||
pub fn clamp_clock(&self, mhz: u32) -> u32 {
|
||
if mhz == 0 || self.clock_max_mhz == 0 {
|
||
return 0;
|
||
}
|
||
mhz.clamp(self.clock_floor(), self.clock_max_mhz)
|
||
}
|
||
/// A memory clock inside the vendor's range; 0 stays 0 (the default).
|
||
pub fn clamp_mem(&self, mhz: u32) -> u32 {
|
||
if mhz == 0 || self.mem_max_mhz == 0 || self.mem_default_mhz == 0 {
|
||
return 0;
|
||
}
|
||
mhz.clamp(self.mem_default_mhz, self.mem_max_mhz)
|
||
}
|
||
/// The watts a power percent asks for, inside the card's min and max, rounded to a watt.
|
||
pub fn watts_for(&self, pct: u32) -> f64 {
|
||
let mut w = self.power_default_w * pct as f64 / 100.0;
|
||
if self.power_min_w > 0.0 {
|
||
w = w.max(self.power_min_w);
|
||
}
|
||
if self.power_max_w > 0.0 {
|
||
w = w.min(self.power_max_w);
|
||
}
|
||
w.round()
|
||
}
|
||
}
|
||
|
||
/// One setting of the knobs (Ember 2 adds the memory clock: 0 = the driver's default).
|
||
#[derive(Clone, Copy, Debug, Default, PartialEq, Eq, Hash)]
|
||
pub struct Point {
|
||
/// the core clock cap in MHz; 0 = unlocked
|
||
pub clock_mhz: u32,
|
||
/// the power limit, percent of the default
|
||
pub power_pct: u32,
|
||
/// the memory clock in MHz (NVIDIA `-lmc`, locked to one value); 0 = the driver's default
|
||
pub mem_mhz: u32,
|
||
}
|
||
|
||
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
|
||
pub enum Kind {
|
||
/// the card as it runs now (the before number; the only step a measure-only card gets)
|
||
Baseline,
|
||
Power,
|
||
Clock,
|
||
/// the fleet prior and one neighbour
|
||
Confirm,
|
||
/// Ember 2: a hill-climb probe
|
||
Climb,
|
||
}
|
||
|
||
#[derive(Clone, Debug, PartialEq)]
|
||
pub struct Step {
|
||
pub point: Point,
|
||
/// the limit to set, watts
|
||
pub watts: f64,
|
||
pub kind: Kind,
|
||
}
|
||
|
||
/// What the tune is for (Settings > Ember Tune > goal). The rate floor is the share of the best rate seen a point
|
||
/// must keep to win on MH per watt: efficiency keeps 90%, balanced 99% (the 1% rule of lever 3), maximum rate
|
||
/// takes the fastest point and uses MH per watt only to break ties.
|
||
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
|
||
pub enum Goal {
|
||
Efficiency,
|
||
Balanced,
|
||
MaxRate,
|
||
}
|
||
|
||
impl Goal {
|
||
pub fn parse(s: &str) -> Goal {
|
||
match s {
|
||
"efficiency" | "eff" => Goal::Efficiency,
|
||
"rate" | "max_rate" | "maximum" => Goal::MaxRate,
|
||
_ => Goal::Balanced,
|
||
}
|
||
}
|
||
pub fn name(&self) -> &'static str {
|
||
match self {
|
||
Goal::Efficiency => "efficiency",
|
||
Goal::Balanced => "balanced",
|
||
Goal::MaxRate => "rate",
|
||
}
|
||
}
|
||
/// The rate tolerance the choice rule uses, percent under the best rate.
|
||
pub fn tolerance_pct(&self, manifest_default: f64) -> f64 {
|
||
match self {
|
||
Goal::Efficiency => 10.0,
|
||
Goal::Balanced => manifest_default,
|
||
Goal::MaxRate => 0.0,
|
||
}
|
||
}
|
||
}
|
||
|
||
/// Pounds a day for a draw at a price in pence per kWh: watts × 24 h / 1000 × price / 100.
|
||
pub fn pounds_per_day(watts: f64, pence_per_kwh: f64) -> f64 {
|
||
watts * 24.0 / 1000.0 * pence_per_kwh / 100.0
|
||
}
|
||
|
||
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
|
||
pub enum PlanKind {
|
||
Full,
|
||
Confirm,
|
||
Baseline,
|
||
/// Ember 2: the hill-climb over memory up and core down from the start point
|
||
Climb,
|
||
}
|
||
|
||
impl PlanKind {
|
||
pub fn name(&self) -> &'static str {
|
||
match self {
|
||
PlanKind::Full => "full",
|
||
PlanKind::Confirm => "confirm",
|
||
PlanKind::Baseline => "baseline",
|
||
PlanKind::Climb => "climb",
|
||
}
|
||
}
|
||
}
|
||
|
||
/// The steps of a tune. The full plan is a coordinate search: the power ladder at the unlocked clock, then the clock
|
||
/// ladder at the power point that ladder chose (asked for through `next`, so the second half depends on the rows).
|
||
#[derive(Clone, Debug, PartialEq)]
|
||
pub struct Plan {
|
||
pub kind: PlanKind,
|
||
pub limits: Limits,
|
||
pub before: Point,
|
||
pub tolerance_pct: f64,
|
||
power: Vec<Step>,
|
||
clock_pcts: Vec<u32>,
|
||
fixed: Vec<Step>,
|
||
/// Ember 2 (Climb): the start point, the step sizes and the step budget
|
||
climb: Option<Climb>,
|
||
}
|
||
|
||
/// The hill-climb's shape: from `start`, each probe moves the memory clock up by `mem_step` or the core clock down
|
||
/// by `core_step` (or both), keeps the move when the goal's score improves, else turns to the other knob; a
|
||
/// refused step (a fault on it) backs that knob off for good. At most `budget` steps including the start.
|
||
#[derive(Clone, Debug, PartialEq)]
|
||
pub struct Climb {
|
||
pub start: Point,
|
||
pub mem_step: u32,
|
||
pub core_step: u32,
|
||
pub budget: usize,
|
||
pub goal: Goal,
|
||
}
|
||
|
||
impl Plan {
|
||
/// Ember 2: the hill-climb. The start is the fleet prior (or the card's current point); a probe step is 5% of
|
||
/// the memory range above the default (0 when the card has no memory knob) and 5% of the maximum core clock;
|
||
/// five steps of 60 s converge in under 10 minutes.
|
||
pub fn climb(limits: &Limits, start: Point, goal: Goal, tolerance_pct: f64) -> Plan {
|
||
let mem_step = if limits.mem_max_mhz > limits.mem_default_mhz { ((limits.mem_max_mhz - limits.mem_default_mhz) / 20).max(25) } else { 0 };
|
||
let core_step = if limits.clock_max_mhz > 0 { (limits.clock_max_mhz / 20).max(25) } else { 0 };
|
||
let start = Point { clock_mhz: limits.clamp_clock(start.clock_mhz), power_pct: start.power_pct.clamp(50, 100), mem_mhz: limits.clamp_mem(start.mem_mhz) };
|
||
Plan { kind: PlanKind::Climb, limits: limits.clone(), before: start, tolerance_pct: goal.tolerance_pct(tolerance_pct), power: Vec::new(), clock_pcts: Vec::new(), fixed: Vec::new(), climb: Some(Climb { start, mem_step, core_step, budget: 5, goal }) }
|
||
}
|
||
|
||
/// The goal's score of a row: MH per watt for efficiency and balanced, the rate for maximum rate.
|
||
pub fn score(&self, r: &Row) -> f64 {
|
||
match self.climb.as_ref().map(|c| c.goal) {
|
||
Some(Goal::MaxRate) => r.mhs,
|
||
_ => r.eff,
|
||
}
|
||
}
|
||
|
||
/// The climb's next probe from the rows so far: None when the budget is spent or no move is left.
|
||
fn climb_next(&self, rows: &[Row]) -> Option<Step> {
|
||
let c = self.climb.as_ref()?;
|
||
let step = |p: Point| Step { point: p, watts: self.limits.watts_for(p.power_pct), kind: Kind::Climb };
|
||
if rows.is_empty() {
|
||
return Some(step(c.start));
|
||
}
|
||
if rows.len() >= c.budget {
|
||
return None;
|
||
}
|
||
// the best usable row so far is the hill's top; a knob that produced a marked (refused) row is backed off
|
||
let best = rows.iter().filter(|r| r.usable()).max_by(|a, b| self.score(a).partial_cmp(&self.score(b)).unwrap_or(std::cmp::Ordering::Equal))?;
|
||
let refused_mem = rows.iter().any(|r| !r.usable() && r.point.mem_mhz > best.point.mem_mhz);
|
||
let refused_core = rows.iter().any(|r| !r.usable() && r.point.clock_mhz != 0 && (best.point.clock_mhz == 0 || r.point.clock_mhz < best.point.clock_mhz));
|
||
let last = rows.last()?;
|
||
let mem_up = |p: Point| -> Option<Point> {
|
||
if c.mem_step == 0 || refused_mem { return None; }
|
||
let base = if p.mem_mhz == 0 { self.limits.mem_default_mhz } else { p.mem_mhz };
|
||
let m = self.limits.clamp_mem(base + c.mem_step);
|
||
(m > 0 && m != p.mem_mhz && m != base).then_some(Point { mem_mhz: m, ..p })
|
||
};
|
||
let core_down = |p: Point| -> Option<Point> {
|
||
if c.core_step == 0 || refused_core { return None; }
|
||
let base = if p.clock_mhz == 0 { self.limits.clock_max_mhz } else { p.clock_mhz };
|
||
let k = self.limits.clamp_clock(base.saturating_sub(c.core_step));
|
||
(k > 0 && k != p.clock_mhz).then_some(Point { clock_mhz: k, ..p })
|
||
};
|
||
let tried = |p: Point| rows.iter().any(|r| r.point == p);
|
||
// the last move improved: keep going the same way from the top; else turn: memory first, then core, then both
|
||
let last_improved = last.usable() && last.point == best.point && rows.len() > 1;
|
||
let last_was_mem = rows.len() > 1 && last.point.mem_mhz != rows[rows.len() - 2].point.mem_mhz;
|
||
let candidates: Vec<Option<Point>> = if last_improved && last_was_mem {
|
||
vec![mem_up(best.point), core_down(best.point)]
|
||
} else if last_improved {
|
||
vec![core_down(best.point), mem_up(best.point)]
|
||
} else {
|
||
vec![mem_up(best.point), core_down(best.point), mem_up(best.point).and_then(core_down)]
|
||
};
|
||
candidates.into_iter().flatten().find(|p| !tried(*p)).map(step)
|
||
}
|
||
|
||
/// Power 100% to 50% at the unlocked clock (duplicate watts dropped, as the card's floor clamps them), then the
|
||
/// clock ladder 90% to the floor of the maximum core clock at the chosen power. A card without a readable
|
||
/// maximum clock gets the power ladder only; a card without a default limit gets the clock ladder only.
|
||
pub fn full(limits: &Limits, before: Point, tolerance_pct: f64) -> Plan {
|
||
let mut power = Vec::new();
|
||
if limits.power_default_w > 0.0 {
|
||
for pct in POWER_STEPS_PCT {
|
||
let w = limits.watts_for(pct);
|
||
if power.last().map(|s: &Step| (s.watts - w).abs() < 0.5).unwrap_or(false) {
|
||
continue;
|
||
}
|
||
power.push(Step { point: Point { clock_mhz: 0, power_pct: pct, mem_mhz: 0 }, watts: w, kind: Kind::Power });
|
||
}
|
||
}
|
||
let clock_pcts = if limits.clock_max_mhz > 0 { CLOCK_STEPS_PCT[1..].to_vec() } else { Vec::new() };
|
||
Plan { kind: PlanKind::Full, limits: limits.clone(), before, tolerance_pct, power, clock_pcts, fixed: Vec::new(), climb: None }
|
||
}
|
||
|
||
/// The prior's point, then one neighbour: the next clock step up when the prior caps the clock (is the cap
|
||
/// costing rate?), else one power step down (is there efficiency left?).
|
||
pub fn confirm(limits: &Limits, prior: Point, before: Point, tolerance_pct: f64) -> Plan {
|
||
let p = Point { clock_mhz: limits.clamp_clock(prior.clock_mhz), power_pct: prior.power_pct.clamp(50, 100), mem_mhz: limits.clamp_mem(prior.mem_mhz) };
|
||
let first = Step { point: p, watts: limits.watts_for(p.power_pct), kind: Kind::Confirm };
|
||
let neighbour = if p.clock_mhz > 0 && limits.clock_max_mhz > 0 {
|
||
let up = p.clock_mhz + limits.clock_max_mhz / 10;
|
||
let clock = if up >= limits.clock_max_mhz { 0 } else { limits.clamp_clock(up) };
|
||
Point { clock_mhz: clock, power_pct: p.power_pct, mem_mhz: p.mem_mhz }
|
||
} else {
|
||
Point { clock_mhz: p.clock_mhz, power_pct: (p.power_pct.saturating_sub(10)).max(50), mem_mhz: p.mem_mhz }
|
||
};
|
||
let mut fixed = vec![first];
|
||
if neighbour != p {
|
||
fixed.push(Step { point: neighbour, watts: limits.watts_for(neighbour.power_pct), kind: Kind::Confirm });
|
||
}
|
||
Plan { kind: PlanKind::Confirm, limits: limits.clone(), before, tolerance_pct, power: Vec::new(), clock_pcts: Vec::new(), fixed, climb: None }
|
||
}
|
||
|
||
/// One step at the card's current point: the before number, and all a measure-only card (Apple, or NVIDIA
|
||
/// with Power control off) reports.
|
||
pub fn baseline(limits: &Limits, before: Point, tolerance_pct: f64) -> Plan {
|
||
let fixed = vec![Step { point: before, watts: limits.watts_for(before.power_pct), kind: Kind::Baseline }];
|
||
Plan { kind: PlanKind::Baseline, limits: limits.clone(), before, tolerance_pct, power: Vec::new(), clock_pcts: Vec::new(), fixed, climb: None }
|
||
}
|
||
|
||
/// How many steps the plan has at most (the clock ladder counts whether or not it runs).
|
||
pub fn len(&self) -> usize {
|
||
if let Some(c) = &self.climb {
|
||
return c.budget;
|
||
}
|
||
self.fixed.len() + self.power.len() + self.clock_pcts.len()
|
||
}
|
||
pub fn is_empty(&self) -> bool {
|
||
self.len() == 0
|
||
}
|
||
|
||
/// The next step after `rows` (one row per step done so far), or None when the plan is complete.
|
||
pub fn next(&self, rows: &[Row]) -> Option<Step> {
|
||
if self.climb.is_some() {
|
||
return self.climb_next(rows);
|
||
}
|
||
let i = rows.len();
|
||
if !self.fixed.is_empty() {
|
||
return self.fixed.get(i).cloned();
|
||
}
|
||
if i < self.power.len() {
|
||
return Some(self.power[i].clone());
|
||
}
|
||
let k = i - self.power.len();
|
||
let pct = *self.clock_pcts.get(k)?;
|
||
// the clock ladder rides the power point the power ladder chose (the before point when nothing won)
|
||
let power_pct = if self.power.is_empty() { self.before.power_pct } else { choose(&rows[..self.power.len()], self.tolerance_pct).map(|r| r.point.power_pct).unwrap_or(self.before.power_pct) };
|
||
let clock = self.limits.clamp_clock(self.limits.clock_max_mhz * pct / 100);
|
||
// a step whose clamp lands on the previous step's clock is dropped (the floor was reached)
|
||
if rows.last().map(|r| r.point.clock_mhz == clock).unwrap_or(false) {
|
||
return None;
|
||
}
|
||
Some(Step { point: Point { clock_mhz: clock, power_pct, mem_mhz: 0 }, watts: self.limits.watts_for(power_pct), kind: Kind::Clock })
|
||
}
|
||
}
|
||
|
||
/// Why a step cannot win, or Ok.
|
||
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
|
||
pub enum Mark {
|
||
Ok,
|
||
NoReadings,
|
||
/// a rejected or mismatched hash during the hold: the step was reverted and marked
|
||
Faulted,
|
||
/// the GPU reached HOT_C during the hold
|
||
Hot,
|
||
/// the memory clock fell under MCLK_HOLD of the baseline's: the knob dragged it
|
||
MemoryClock,
|
||
/// the setting did not take (the readback disagreed)
|
||
Unapplied,
|
||
}
|
||
|
||
impl Mark {
|
||
pub fn name(&self) -> &'static str {
|
||
match self {
|
||
Mark::Ok => "ok",
|
||
Mark::NoReadings => "no_readings",
|
||
Mark::Faulted => "faulted",
|
||
Mark::Hot => "hot",
|
||
Mark::MemoryClock => "memory_clock_dropped",
|
||
Mark::Unapplied => "unapplied",
|
||
}
|
||
}
|
||
}
|
||
|
||
/// One step's result. `watts` is the mean draw during the hold (what the miner pays for), `limit` the cap set,
|
||
/// `eff` MH per watt of draw (0 when the step had no readings).
|
||
#[derive(Clone, Debug, Default, PartialEq)]
|
||
pub struct Row {
|
||
pub point: Point,
|
||
pub limit: f64,
|
||
pub watts: f64,
|
||
pub mhs: f64,
|
||
pub eff: f64,
|
||
pub draws: usize,
|
||
pub rates: usize,
|
||
pub gclk: f64,
|
||
pub mclk: f64,
|
||
pub tmax: f64,
|
||
pub faults: u32,
|
||
pub mark: Option<Mark>,
|
||
}
|
||
|
||
/// What a hold collected.
|
||
#[derive(Clone, Debug, Default)]
|
||
pub struct Samples {
|
||
pub draws: Vec<f64>,
|
||
pub rates: Vec<f64>,
|
||
pub gclks: Vec<f64>,
|
||
pub mclks: Vec<f64>,
|
||
pub tmax: f64,
|
||
pub faults: u32,
|
||
pub unapplied: bool,
|
||
}
|
||
|
||
fn mean(v: &[f64]) -> f64 {
|
||
if v.is_empty() {
|
||
0.0
|
||
} else {
|
||
v.iter().sum::<f64>() / v.len() as f64
|
||
}
|
||
}
|
||
|
||
impl Row {
|
||
/// `baseline_mclk` is the memory clock of the first row (0 = unknown): the memory-clock guard.
|
||
pub fn from_samples(step: &Step, s: &Samples, baseline_mclk: f64) -> Row {
|
||
let watts = mean(&s.draws);
|
||
let mhs = mean(&s.rates);
|
||
let mclk = mean(&s.mclks);
|
||
let usable = s.draws.len() >= 3 && !s.rates.is_empty() && watts > 1.0;
|
||
let mark = if s.faults > 0 {
|
||
Mark::Faulted
|
||
} else if s.unapplied {
|
||
Mark::Unapplied
|
||
} else if !usable {
|
||
Mark::NoReadings
|
||
} else if s.tmax >= HOT_C {
|
||
Mark::Hot
|
||
} else if baseline_mclk > 0.0 && mclk > 0.0 && mclk < MCLK_HOLD * baseline_mclk {
|
||
Mark::MemoryClock
|
||
} else {
|
||
Mark::Ok
|
||
};
|
||
Row { point: step.point, limit: step.watts, watts, mhs, eff: if usable { mhs / watts } else { 0.0 }, draws: s.draws.len(), rates: s.rates.len(), gclk: mean(&s.gclks), mclk, tmax: s.tmax, faults: s.faults, mark: Some(mark) }
|
||
}
|
||
pub fn usable(&self) -> bool {
|
||
self.eff > 0.0 && self.mark == Some(Mark::Ok)
|
||
}
|
||
/// `TUNE card=<label> step=<i> clock=<MHz> cap=<pct> limit=<W> watts=<W> mhs=<x> eff=<MH/W> gclk=<MHz> mclk=<MHz> tmax=<C> mark=<m>`
|
||
pub fn line(&self, card: &str, i: usize) -> String {
|
||
format!(
|
||
"TUNE card={card} step={i} clock={} cap={} mem={} limit={:.0} watts={:.1} mhs={:.2} eff={:.4} gclk={:.0} mclk={:.0} tmax={:.0} mark={}{}",
|
||
self.point.clock_mhz,
|
||
self.point.power_pct,
|
||
self.point.mem_mhz,
|
||
self.limit,
|
||
self.watts,
|
||
self.mhs,
|
||
self.eff,
|
||
self.gclk,
|
||
self.mclk,
|
||
self.tmax,
|
||
self.mark.unwrap_or(Mark::NoReadings).name(),
|
||
if self.faults > 0 { format!(" faults={}", self.faults) } else { String::new() }
|
||
)
|
||
}
|
||
pub fn json(&self) -> serde_json::Value {
|
||
serde_json::json!({ "clock_mhz": self.point.clock_mhz, "power_pct": self.point.power_pct, "mem_mhz": self.point.mem_mhz, "limit_w": self.limit, "watts": r1(self.watts), "mhs": r2(self.mhs), "eff": r4(self.eff), "gclk": self.gclk.round(), "mclk": self.mclk.round(), "tmax": self.tmax.round(), "faults": self.faults, "mark": self.mark.unwrap_or(Mark::NoReadings).name() })
|
||
}
|
||
}
|
||
|
||
fn r1(x: f64) -> f64 {
|
||
(x * 10.0).round() / 10.0
|
||
}
|
||
fn r2(x: f64) -> f64 {
|
||
(x * 100.0).round() / 100.0
|
||
}
|
||
fn r4(x: f64) -> f64 {
|
||
(x * 10_000.0).round() / 10_000.0
|
||
}
|
||
|
||
/// The choice: among the usable rows within `tolerance_pct` of the fastest row's rate, the best MH per watt; within
|
||
/// 1% on efficiency the higher rate wins; within 1% on both, the lower draw. A card never gives up more than the
|
||
/// tolerance in blocks for the saving. Marked rows never win.
|
||
pub fn choose(rows: &[Row], tolerance_pct: f64) -> Option<Row> {
|
||
let top = rows.iter().filter(|r| r.usable()).map(|r| r.mhs).fold(0.0, f64::max);
|
||
if top <= 0.0 {
|
||
return None;
|
||
}
|
||
let floor = top * (1.0 - tolerance_pct.max(0.0) / 100.0);
|
||
let mut best: Option<&Row> = None;
|
||
for r in rows.iter().filter(|r| r.usable() && r.mhs >= floor) {
|
||
best = Some(match best {
|
||
None => r,
|
||
Some(b) => {
|
||
let eff_tie = (r.eff - b.eff).abs() <= 0.01 * b.eff.max(r.eff);
|
||
if !eff_tie {
|
||
if r.eff > b.eff { r } else { b }
|
||
} else {
|
||
let mhs_tie = (r.mhs - b.mhs).abs() <= 0.01 * b.mhs.max(r.mhs);
|
||
if !mhs_tie {
|
||
if r.mhs > b.mhs { r } else { b }
|
||
} else if r.watts < b.watts {
|
||
r
|
||
} else {
|
||
b
|
||
}
|
||
}
|
||
}
|
||
});
|
||
}
|
||
best.cloned()
|
||
}
|
||
|
||
/// A confirm check's verdict: the prior stands, or the neighbour beat it by over CONFIRM_GAIN_PCT (the full plan is
|
||
/// due), or nothing could be read.
|
||
#[derive(Clone, Debug, PartialEq)]
|
||
pub enum Verdict {
|
||
Keep(Row),
|
||
FullDue { prior: Row, better: Row },
|
||
NoReadings,
|
||
}
|
||
|
||
pub fn confirm_verdict(rows: &[Row], tolerance_pct: f64) -> Verdict {
|
||
let Some(prior) = rows.first().filter(|r| r.usable()).cloned() else {
|
||
return match choose(rows, tolerance_pct) {
|
||
Some(r) => Verdict::FullDue { prior: Row::default(), better: r },
|
||
None => Verdict::NoReadings,
|
||
};
|
||
};
|
||
match choose(rows, tolerance_pct) {
|
||
Some(best) if best.point != prior.point && best.eff > prior.eff * (1.0 + CONFIRM_GAIN_PCT / 100.0) => Verdict::FullDue { prior, better: best },
|
||
_ => Verdict::Keep(prior),
|
||
}
|
||
}
|
||
|
||
// ---- the fleet prior ------------------------------------------------------------------------------------------
|
||
|
||
/// The key a prior is filed under: card model (spaces as underscores), driver major, program class.
|
||
pub fn prior_key(card: &str, driver: &str, class: &str) -> String {
|
||
format!("{}|{}|{}", card.trim().replace(' ', "_"), driver_major(driver), if class.is_empty() { "v2" } else { class })
|
||
}
|
||
|
||
/// "581.57" -> "581", "32.0.15801" -> "32", "" -> "0".
|
||
pub fn driver_major(driver: &str) -> String {
|
||
let d: String = driver.trim().chars().take_while(|c| c.is_ascii_digit()).collect();
|
||
if d.is_empty() { "0".into() } else { d }
|
||
}
|
||
|
||
/// The program class from a race line's features (loads and wide loads per hash); the generator fixes both today.
|
||
pub fn program_class(loads: u32, wide: u32) -> String {
|
||
if loads == 0 { "v2".into() } else { format!("l{loads}w{wide}") }
|
||
}
|
||
|
||
/// A fleet prior as the signed manifest carries it under tuning.priors.<key>.
|
||
#[derive(Clone, Debug, Default, PartialEq)]
|
||
pub struct Prior {
|
||
pub point: Point,
|
||
pub eff: f64,
|
||
pub mhs: f64,
|
||
pub watts: f64,
|
||
/// the spread of the chosen efficiency across the samples, percent of the median
|
||
pub spread_pct: f64,
|
||
pub samples: u32,
|
||
}
|
||
|
||
/// The tuning section's Ember settings: the kill switch and the thresholds (defaults when absent).
|
||
#[derive(Clone, Debug, PartialEq)]
|
||
pub struct Settings {
|
||
pub enabled: bool,
|
||
pub min_samples: u32,
|
||
pub tolerance_pct: f64,
|
||
pub period_s: u64,
|
||
}
|
||
|
||
impl Default for Settings {
|
||
fn default() -> Settings {
|
||
Settings { enabled: true, min_samples: PRIOR_MIN_SAMPLES, tolerance_pct: RATE_TOLERANCE_PCT, period_s: PERIOD_S }
|
||
}
|
||
}
|
||
|
||
/// Reads `tuning.ember` (absent = the defaults; `enabled: false` is the fleet-wide kill switch).
|
||
pub fn settings_of(tuning: Option<&serde_json::Value>) -> Settings {
|
||
let d = Settings::default();
|
||
let Some(e) = tuning.and_then(|t| t.get("ember")).filter(|e| e.is_object()) else { return d };
|
||
Settings {
|
||
enabled: e.get("enabled").and_then(|v| v.as_bool()).unwrap_or(d.enabled),
|
||
min_samples: e.get("min_samples").and_then(|v| v.as_u64()).map(|v| v as u32).unwrap_or(d.min_samples),
|
||
tolerance_pct: e.get("rate_tolerance_pct").and_then(|v| v.as_f64()).filter(|v| (0.0..=25.0).contains(v)).unwrap_or(d.tolerance_pct),
|
||
period_s: e.get("period_s").and_then(|v| v.as_u64()).filter(|v| *v >= 3600).unwrap_or(d.period_s),
|
||
}
|
||
}
|
||
|
||
/// The prior for a key, or None when the manifest has none or it carries under `min_samples` samples.
|
||
pub fn prior_of(tuning: Option<&serde_json::Value>, key: &str, min_samples: u32) -> Option<Prior> {
|
||
let p = tuning?.get("priors")?.get(key)?;
|
||
let n = |k: &str| p.get(k).and_then(|v| v.as_f64()).unwrap_or(0.0);
|
||
let samples = p.get("samples").and_then(|v| v.as_u64()).unwrap_or(0) as u32;
|
||
if samples < min_samples.max(1) {
|
||
return None;
|
||
}
|
||
let point = Point { clock_mhz: n("clock_mhz") as u32, power_pct: (n("power_pct") as u32).clamp(50, 100), mem_mhz: n("mem_mhz") as u32 };
|
||
if point.power_pct == 0 && point.clock_mhz == 0 {
|
||
return None;
|
||
}
|
||
Some(Prior { point, eff: n("eff"), mhs: n("mhs"), watts: n("watts"), spread_pct: n("spread_pct"), samples })
|
||
}
|
||
|
||
/// The fleet record: one `TUNE {json}` line. `machine` is a hash of the install id (never the id, never an address).
|
||
#[allow(clippy::too_many_arguments)]
|
||
pub fn record_json(ts: f64, machine_hash: &str, app: &str, os: &str, card: &str, vendor: &str, driver: &str, class: &str, plan: PlanKind, rows: &[Row], chosen: Option<&Row>, before: Option<&Row>, floor: bool) -> serde_json::Value {
|
||
serde_json::json!({
|
||
"ts": ts.round(),
|
||
"machine": machine_hash,
|
||
"app": app,
|
||
"os": os,
|
||
"card": card.replace(' ', "_"),
|
||
"vendor": vendor,
|
||
"driver": driver,
|
||
"driver_major": driver_major(driver),
|
||
"class": if class.is_empty() { "v2" } else { class },
|
||
"key": prior_key(card, driver, class),
|
||
"plan": plan.name(),
|
||
"steps": rows.iter().map(|r| r.json()).collect::<Vec<_>>(),
|
||
"chosen": chosen.map(|r| r.json()),
|
||
"before": before.map(|r| r.json()),
|
||
"eff": chosen.map(|r| r4(r.eff)).unwrap_or(0.0),
|
||
"mhs": chosen.map(|r| r2(r.mhs)).unwrap_or(0.0),
|
||
"watts": chosen.map(|r| r1(r.watts)).unwrap_or(0.0),
|
||
"floor": floor,
|
||
"note": if floor { "floor, not optimum" } else { "" },
|
||
})
|
||
}
|
||
|
||
// ---- the state machine -----------------------------------------------------------------------------------------
|
||
|
||
pub use crate::sweep::Timing;
|
||
|
||
/// Where the tune stands, for the card row.
|
||
#[derive(Clone, Debug, PartialEq)]
|
||
pub enum Phase {
|
||
/// the step's setting was (or is about to be) requested; waiting for the readback or the acknowledgement
|
||
Applying { since: Instant, sent: bool },
|
||
Settling { since: Instant },
|
||
Holding { since: Instant },
|
||
/// the chosen point was requested; waiting for the readback
|
||
Finishing { since: Instant, sent: bool },
|
||
Done,
|
||
}
|
||
|
||
/// What the engine must do after a tick.
|
||
#[derive(Clone, Debug, PartialEq)]
|
||
pub enum Out {
|
||
/// set this point on the card (the watts are the limit for its power percent)
|
||
Apply(Step),
|
||
/// a step finished: log its line (its index is the row count before it)
|
||
Row(Row),
|
||
/// the plan finished and the chosen point is in force
|
||
Finished(Row),
|
||
/// the plan failed; the engine restores the point from before
|
||
Failed(String),
|
||
}
|
||
|
||
/// What the card reports now, for the readback.
|
||
#[derive(Clone, Copy, Debug, Default)]
|
||
pub struct Readback {
|
||
/// the power limit in force, watts (0 = unknown)
|
||
pub limit_w: f64,
|
||
/// the set command for the current request was acknowledged (nvidia-smi answered, or the helper logged it)
|
||
pub acked: bool,
|
||
}
|
||
|
||
pub struct Run {
|
||
pub card: usize,
|
||
pub device: String,
|
||
pub label: String,
|
||
pub plan: Plan,
|
||
pub current: Option<Step>,
|
||
pub phase: Phase,
|
||
pub rows: Vec<Row>,
|
||
pub chosen: Option<Row>,
|
||
pub before: Point,
|
||
pub before_w: f64,
|
||
pub started: Instant,
|
||
pub forced: bool,
|
||
pub timing: Timing,
|
||
/// the request number of the setting in flight (the engine acknowledges it by number)
|
||
pub seq: u64,
|
||
samples: Samples,
|
||
}
|
||
|
||
impl Run {
|
||
#[allow(clippy::too_many_arguments)]
|
||
pub fn new(card: usize, device: &str, label: &str, plan: Plan, before_w: f64, forced: bool, timing: Timing, now: Instant) -> Run {
|
||
let before = plan.before;
|
||
let current = plan.next(&[]);
|
||
Run { card, device: device.into(), label: label.into(), plan, current, phase: Phase::Applying { since: now, sent: false }, rows: Vec::new(), chosen: None, before, before_w, started: now, forced, timing, seq: 0, samples: Samples::default() }
|
||
}
|
||
|
||
/// A worker STATUS interval rate (MH/s); counted while holding.
|
||
pub fn sample_rate(&mut self, mhs: f64) {
|
||
if matches!(self.phase, Phase::Holding { .. }) && mhs > 0.0 {
|
||
self.samples.rates.push(mhs);
|
||
}
|
||
}
|
||
/// A telemetry reading (draw, core clock, memory clock, GPU temperature); counted while holding.
|
||
pub fn sample_telemetry(&mut self, draw_w: f64, gclk: f64, mclk: f64, temp_c: f64) {
|
||
if !matches!(self.phase, Phase::Holding { .. }) {
|
||
return;
|
||
}
|
||
if draw_w > 0.0 {
|
||
self.samples.draws.push(draw_w);
|
||
}
|
||
if gclk > 0.0 {
|
||
self.samples.gclks.push(gclk);
|
||
}
|
||
if mclk > 0.0 {
|
||
self.samples.mclks.push(mclk);
|
||
}
|
||
if temp_c > self.samples.tmax {
|
||
self.samples.tmax = temp_c;
|
||
}
|
||
}
|
||
/// A rejected or mismatched hash while this step holds: the step is marked and cannot win.
|
||
pub fn sample_fault(&mut self) {
|
||
if matches!(self.phase, Phase::Holding { .. } | Phase::Settling { .. }) {
|
||
self.samples.faults += 1;
|
||
}
|
||
}
|
||
/// The engine's readback said the setting in force is not the one requested (checked during the hold).
|
||
pub fn mark_unapplied(&mut self) {
|
||
if matches!(self.phase, Phase::Holding { .. }) {
|
||
self.samples.unapplied = true;
|
||
}
|
||
}
|
||
|
||
pub fn baseline_mclk(&self) -> f64 {
|
||
self.rows.first().map(|r| r.mclk).unwrap_or(0.0)
|
||
}
|
||
|
||
/// Seconds left in the plan from here: the rest of this step plus (settle + hold + a 10 s apply) per step to come.
|
||
pub fn eta_s(&self, now: Instant) -> i64 {
|
||
let per = (self.timing.settle + self.timing.hold).as_secs() as i64 + 10;
|
||
let this = match &self.phase {
|
||
Phase::Applying { .. } => per,
|
||
Phase::Settling { since } => per - 10 - now.duration_since(*since).as_secs().min(per as u64) as i64,
|
||
Phase::Holding { since } => (self.timing.hold.as_secs() as i64 - now.duration_since(*since).as_secs() as i64).max(0),
|
||
Phase::Finishing { .. } | Phase::Done => 0,
|
||
};
|
||
let left = self.plan.len().saturating_sub(self.rows.len() + 1) as i64;
|
||
this.max(0) + left * per
|
||
}
|
||
|
||
/// The phase in words for the card row.
|
||
pub fn words(&self, now: Instant) -> String {
|
||
let what = |p: &Point| {
|
||
let mut w = if p.clock_mhz > 0 { format!("{} MHz · {}%", p.clock_mhz, p.power_pct) } else { format!("{}%", p.power_pct) };
|
||
if p.mem_mhz > 0 {
|
||
w.push_str(&format!(" · mem {}", p.mem_mhz));
|
||
}
|
||
w
|
||
};
|
||
let cur = self.current.as_ref().map(|s| what(&s.point)).unwrap_or_default();
|
||
let n = self.rows.len() + 1;
|
||
let of = self.plan.len();
|
||
match &self.phase {
|
||
Phase::Applying { .. } => format!("tuning: setting {cur} (step {n} of {of})"),
|
||
Phase::Settling { since } => format!("tuning: {cur} settling · {} s (step {n} of {of})", self.timing.settle.as_secs().saturating_sub(now.duration_since(*since).as_secs())),
|
||
Phase::Holding { since } => format!("tuning: holding {cur} · {} s (step {n} of {of})", self.timing.hold.as_secs().saturating_sub(now.duration_since(*since).as_secs())),
|
||
Phase::Finishing { .. } => format!("tuning: setting the best point ({})", self.chosen.as_ref().map(|r| what(&r.point)).unwrap_or_default()),
|
||
Phase::Done => "tuning: done".into(),
|
||
}
|
||
}
|
||
|
||
/// A step is applied when its power limit reads back (within 1.5 W) and, when it caps the clock or changes
|
||
/// nothing in watts, the set command was acknowledged.
|
||
fn applied(step: &Step, rb: Readback, before_w: f64) -> bool {
|
||
let power_ok = rb.limit_w <= 0.0 || (rb.limit_w - step.watts).abs() < 1.5;
|
||
let power_changed = (step.watts - before_w).abs() >= 1.5;
|
||
if step.point.clock_mhz > 0 || step.point.mem_mhz > 0 || !power_changed {
|
||
rb.acked && power_ok
|
||
} else if rb.limit_w <= 0.0 {
|
||
// a card with no limit readback at all (AMD through ADLX reports an offset, never watts; run 6 on
|
||
// 6 October 2026 aborted the 9070 XT's ladder at its first step on "card reports 0 W"): the
|
||
// acknowledgement is the proof
|
||
rb.acked
|
||
} else {
|
||
power_ok
|
||
}
|
||
}
|
||
|
||
/// Drives the state machine. `rb` is what the card reports now.
|
||
pub fn tick(&mut self, now: Instant, rb: Readback) -> Vec<Out> {
|
||
let mut out = Vec::new();
|
||
let settle = self.timing.settle;
|
||
let hold = self.timing.hold;
|
||
let apply = self.timing.apply;
|
||
match self.phase.clone() {
|
||
Phase::Applying { since, sent } => {
|
||
let Some(step) = self.current.clone() else {
|
||
self.phase = Phase::Done;
|
||
out.push(Out::Failed("no steps".into()));
|
||
return out;
|
||
};
|
||
if !sent {
|
||
self.phase = Phase::Applying { since: now, sent: true };
|
||
self.seq += 1;
|
||
out.push(Out::Apply(step));
|
||
} else if Run::applied(&step, rb, self.last_watts()) {
|
||
self.phase = Phase::Settling { since: now };
|
||
} else if now.duration_since(since) > apply {
|
||
if step.kind == Kind::Baseline {
|
||
// the baseline is the card as it runs: nothing to apply, nothing to wait for
|
||
self.phase = Phase::Settling { since: now };
|
||
} else {
|
||
self.phase = Phase::Done;
|
||
out.push(Out::Failed(format!("the setting ({} MHz, {}% = {:.0} W) did not take within {} s (card reports {:.0} W, acknowledged {})", step.point.clock_mhz, step.point.power_pct, step.watts, apply.as_secs(), rb.limit_w, rb.acked)));
|
||
}
|
||
}
|
||
}
|
||
Phase::Settling { since } => {
|
||
if now.duration_since(since) >= settle {
|
||
let faults = self.samples.faults;
|
||
self.samples = Samples { faults, ..Samples::default() };
|
||
self.phase = Phase::Holding { since: now };
|
||
}
|
||
}
|
||
Phase::Holding { since } => {
|
||
if now.duration_since(since) >= hold {
|
||
let step = self.current.clone().expect("a step while holding");
|
||
let row = Row::from_samples(&step, &self.samples, self.baseline_mclk());
|
||
self.rows.push(row.clone());
|
||
out.push(Out::Row(row));
|
||
self.current = self.plan.next(&self.rows);
|
||
match self.current.clone() {
|
||
Some(next) => {
|
||
self.phase = Phase::Applying { since: now, sent: true };
|
||
self.seq += 1;
|
||
out.push(Out::Apply(next));
|
||
}
|
||
None => match self.decide() {
|
||
Some(best) => {
|
||
self.chosen = Some(best.clone());
|
||
self.phase = Phase::Finishing { since: now, sent: true };
|
||
self.seq += 1;
|
||
out.push(Out::Apply(Step { point: best.point, watts: best.limit, kind: Kind::Confirm }));
|
||
}
|
||
None => {
|
||
self.phase = Phase::Done;
|
||
out.push(Out::Failed("no step had readings (no draw from the telemetry or no STATUS line from the worker)".into()));
|
||
}
|
||
},
|
||
}
|
||
}
|
||
}
|
||
Phase::Finishing { since, sent } => {
|
||
let best = self.chosen.clone().expect("a chosen row while finishing");
|
||
let step = Step { point: best.point, watts: best.limit, kind: Kind::Confirm };
|
||
if !sent {
|
||
self.phase = Phase::Finishing { since: now, sent: true };
|
||
self.seq += 1;
|
||
out.push(Out::Apply(step));
|
||
} else if Run::applied(&step, rb, self.last_watts()) || self.plan.kind == PlanKind::Baseline {
|
||
self.phase = Phase::Done;
|
||
out.push(Out::Finished(best));
|
||
} else if now.duration_since(since) > apply {
|
||
self.phase = Phase::Done;
|
||
out.push(Out::Failed(format!("the chosen point ({} MHz, {:.0} W) did not take within {} s", best.point.clock_mhz, best.limit, apply.as_secs())));
|
||
}
|
||
}
|
||
Phase::Done => {}
|
||
}
|
||
out
|
||
}
|
||
|
||
/// The limit the previous step left (the before limit for the first).
|
||
fn last_watts(&self) -> f64 {
|
||
self.rows.last().map(|r| r.limit).unwrap_or(self.before_w)
|
||
}
|
||
|
||
/// What the plan concludes: the full plan's choice; the confirm plan keeps its prior unless the neighbour beat
|
||
/// it (then the prior still holds the card and the engine schedules the full plan); the baseline is itself.
|
||
fn decide(&self) -> Option<Row> {
|
||
match self.plan.kind {
|
||
PlanKind::Baseline => self.rows.first().cloned().filter(|r| r.usable()),
|
||
PlanKind::Confirm => match confirm_verdict(&self.rows, self.plan.tolerance_pct) {
|
||
Verdict::Keep(r) => Some(r),
|
||
Verdict::FullDue { prior, .. } if prior.usable() => Some(prior),
|
||
Verdict::FullDue { better, .. } => Some(better),
|
||
Verdict::NoReadings => None,
|
||
},
|
||
PlanKind::Full => choose(&self.rows, self.plan.tolerance_pct),
|
||
PlanKind::Climb => {
|
||
if self.plan.climb.as_ref().map(|c| c.goal) == Some(Goal::MaxRate) {
|
||
self.rows.iter().filter(|r| r.usable()).max_by(|a, b| a.mhs.partial_cmp(&b.mhs).unwrap_or(std::cmp::Ordering::Equal)).cloned()
|
||
} else {
|
||
choose(&self.rows, self.plan.tolerance_pct)
|
||
}
|
||
}
|
||
}
|
||
}
|
||
|
||
/// After a confirm plan: did the neighbour beat the prior (the full plan is due)?
|
||
pub fn full_due(&self) -> bool {
|
||
self.plan.kind == PlanKind::Confirm && matches!(confirm_verdict(&self.rows, self.plan.tolerance_pct), Verdict::FullDue { .. })
|
||
}
|
||
}
|
||
|
||
/// The card row's line once a card is tuned: "Tuned: 122.3 MH/s at 290 W (0.422 MH/W)".
|
||
pub fn tuned_line(mhs: f64, watts: f64, eff: f64) -> String {
|
||
format!("Tuned: {mhs:.1} MH/s at {watts:.0} W ({eff:.3} MH/W)")
|
||
}
|
||
|
||
/// The row's line for a baseline (measure-only) result: the card was measured as it runs, nothing was set.
|
||
/// (6 October 2026: PC 1's rows read "Tuned: 114.2 MH/s at 305 W" for a baseline, which is not a tune.)
|
||
pub fn measured_line(mhs: f64, watts: f64, eff: f64) -> String {
|
||
format!("Measured: {mhs:.1} MH/s at {watts:.0} W ({eff:.3} MH/W)")
|
||
}
|
||
|
||
/// The row's line for a plan's result.
|
||
pub fn result_line(kind: PlanKind, mhs: f64, watts: f64, eff: f64) -> String {
|
||
if kind == PlanKind::Baseline {
|
||
measured_line(mhs, watts, eff)
|
||
} else {
|
||
tuned_line(mhs, watts, eff)
|
||
}
|
||
}
|
||
|
||
/// Why a card cannot be tuned beyond measuring, or None when both knobs are available.
|
||
pub fn control_reason(vendor: &str, limits: &Limits, device: &str, power_control: bool, amd_helper: bool) -> Option<String> {
|
||
match vendor {
|
||
"nvidia" if limits.power_default_w <= 0.0 || device.is_empty() => Some("measure only: nvidia-smi did not report this card's limits".into()),
|
||
"nvidia" if !power_control => Some("measure only until Power control is on in Settings (Windows asks for administrator rights once)".into()),
|
||
"nvidia" => None,
|
||
"amd" if !amd_helper => Some("measure only: igneum-gpu-telemetry gave no tune line for this card (not next to the app, an older helper, or no ADLX)".into()),
|
||
"amd" if cfg!(target_os = "linux") => Some("measure only on Linux: the clock and power limits under /sys need root".into()),
|
||
"amd" => None,
|
||
"apple" => Some("measure only on Apple silicon: the system sets the clocks and the power; no control exposed".into()),
|
||
_ => Some("measure only: no power or clock control for this card".into()),
|
||
}
|
||
}
|
||
|
||
#[cfg(test)]
|
||
mod tests {
|
||
use super::*;
|
||
|
||
/// PC 1's RTX 5090 (nvidia-smi, 4 and 5 October 2026): default 575 W, min 400 W, max 600 W; clocks.max.gr is
|
||
/// read at the first tune (3,090 MHz is the shape used here, not a measurement).
|
||
fn l5090() -> Limits {
|
||
Limits { power_default_w: 575.0, power_min_w: 400.0, power_max_w: 600.0, clock_max_mhz: 3090, clock_min_mhz: 0, mem_default_mhz: 13801, mem_max_mhz: 14001 }
|
||
}
|
||
|
||
fn row_at(p: Point, watts: f64, mhs: f64) -> Row {
|
||
Row { point: p, limit: watts, watts, mhs, eff: if watts > 0.0 { mhs / watts } else { 0.0 }, draws: 12, rates: 6, gclk: 0.0, mclk: 0.0, tmax: 60.0, faults: 0, mark: Some(Mark::Ok) }
|
||
}
|
||
|
||
#[test]
|
||
fn the_full_plan_is_the_power_ladder_then_the_clock_ladder_at_the_chosen_power() {
|
||
let plan = Plan::full(&l5090(), Point { clock_mhz: 0, power_pct: 80, mem_mhz: 0 }, 1.0);
|
||
assert_eq!(plan.len(), 5 + 6, "five power steps (60% and 50% clamp to 400 W; one kept) and six clock steps (90% down to 45%)");
|
||
let first = plan.next(&[]).unwrap();
|
||
assert_eq!((first.point, first.watts, first.kind), (Point { clock_mhz: 0, power_pct: 100, mem_mhz: 0 }, 575.0, Kind::Power));
|
||
// the power ladder: 575, 518, 460, 403, 400
|
||
let mut rows = Vec::new();
|
||
let mut watts_seen = Vec::new();
|
||
for i in 0..5 {
|
||
let s = plan.next(&rows).unwrap();
|
||
assert_eq!(s.kind, Kind::Power, "step {i}");
|
||
watts_seen.push(s.watts);
|
||
// the draw never reaches the cap (PC 1's shape: 290 W under every limit); the rate is flat
|
||
rows.push(row_at(s.point, 290.0 + i as f64 * 0.1, 124.0));
|
||
}
|
||
assert_eq!(watts_seen, vec![575.0, 518.0, 460.0, 403.0, 400.0]);
|
||
// the clock ladder rides the chosen power point: a flat ladder ties on efficiency and rate, the lowest draw wins (100%)
|
||
let s = plan.next(&rows).unwrap();
|
||
assert_eq!(s.kind, Kind::Clock);
|
||
assert_eq!(s.point, Point { clock_mhz: 2781, power_pct: 100, mem_mhz: 0 });
|
||
rows.push(row_at(s.point, 250.0, 123.8));
|
||
let s = plan.next(&rows).unwrap();
|
||
assert_eq!(s.point.clock_mhz, 2472);
|
||
rows.push(row_at(s.point, 220.0, 123.5));
|
||
rows.push(row_at(plan.next(&rows).unwrap().point, 200.0, 118.0));
|
||
let s = plan.next(&rows).unwrap();
|
||
assert_eq!(s.point.clock_mhz, 1854, "60% of 3,090");
|
||
rows.push(row_at(s.point, 180.0, 100.0));
|
||
let s = plan.next(&rows).unwrap();
|
||
assert_eq!(s.point.clock_mhz, 1545, "50%");
|
||
rows.push(row_at(s.point, 170.0, 90.0));
|
||
let s = plan.next(&rows).unwrap();
|
||
assert_eq!(s.point.clock_mhz, 1390, "45% of 3,090 is the floor (6 October 2026)");
|
||
rows.push(row_at(s.point, 160.0, 80.0));
|
||
assert_eq!(plan.next(&rows), None);
|
||
// the choice: 2,472 MHz keeps 99.6% of the top rate at 220 W = 0.561 MH/W; 2,163 MHz (118 MH/s) is outside the 1% tolerance
|
||
let best = choose(&rows, 1.0).unwrap();
|
||
assert_eq!(best.point, Point { clock_mhz: 2472, power_pct: 100, mem_mhz: 0 });
|
||
// a wider tolerance lets the 2,163 MHz step (0.590 MH/W, 4.8% slower) win
|
||
assert_eq!(choose(&rows, 5.0).unwrap().point.clock_mhz, 2163);
|
||
// no power limits, clocks only; no clocks, power only; nothing, empty
|
||
assert_eq!(Plan::full(&Limits { clock_max_mhz: 2000, ..Default::default() }, Point::default(), 1.0).len(), 6);
|
||
assert_eq!(Plan::full(&Limits { power_default_w: 300.0, ..Default::default() }, Point::default(), 1.0).len(), 6);
|
||
assert!(Plan::full(&Limits::default(), Point::default(), 1.0).is_empty());
|
||
}
|
||
|
||
#[test]
|
||
fn a_power_step_on_a_card_without_a_limit_readback_is_applied_on_the_acknowledgement() {
|
||
// AMD through ADLX: the limit reads back as an offset, never watts (run 6, 6 October 2026)
|
||
let step = Step { point: Point { clock_mhz: 0, power_pct: 90, mem_mhz: 0 }, watts: 90.0, kind: Kind::Power };
|
||
assert!(Run::applied(&step, Readback { limit_w: 0.0, acked: true }, 100.0));
|
||
assert!(!Run::applied(&step, Readback { limit_w: 0.0, acked: false }, 100.0));
|
||
// NVIDIA: the watts must read back
|
||
assert!(!Run::applied(&step, Readback { limit_w: 100.0, acked: true }, 100.0));
|
||
assert!(Run::applied(&step, Readback { limit_w: 90.0, acked: false }, 100.0));
|
||
}
|
||
|
||
#[test]
|
||
fn a_baseline_result_reads_measured_and_a_tune_reads_tuned() {
|
||
assert_eq!(result_line(PlanKind::Baseline, 114.2, 305.0, 0.3744), "Measured: 114.2 MH/s at 305 W (0.374 MH/W)");
|
||
assert_eq!(result_line(PlanKind::Full, 127.71, 226.8, 0.5632), "Tuned: 127.7 MH/s at 227 W (0.563 MH/W)");
|
||
assert_eq!(result_line(PlanKind::Confirm, 127.71, 226.8, 0.5632), "Tuned: 127.7 MH/s at 227 W (0.563 MH/W)");
|
||
}
|
||
|
||
#[test]
|
||
fn limits_never_exceed_the_vendor_or_undercut_the_floor() {
|
||
let l = l5090();
|
||
assert_eq!(l.clock_floor(), 1390);
|
||
assert_eq!(l.clamp_clock(1000), 1390);
|
||
assert_eq!(l.clamp_clock(5000), 3090);
|
||
assert_eq!(l.clamp_clock(0), 0, "unlocked stays unlocked");
|
||
assert_eq!(Limits { clock_max_mhz: 3000, clock_min_mhz: 2100, ..Default::default() }.clamp_clock(1500), 2100, "the vendor's floor wins over the 45% rule");
|
||
assert_eq!(l.watts_for(100), 575.0);
|
||
assert_eq!(l.watts_for(50), 400.0);
|
||
assert_eq!(Limits { power_default_w: 300.0, power_max_w: 250.0, ..Default::default() }.watts_for(100), 250.0);
|
||
}
|
||
|
||
#[test]
|
||
fn the_choice_keeps_the_best_mh_per_watt_within_the_rate_tolerance() {
|
||
let p = |c: u32, pct: u32| Point { clock_mhz: c, power_pct: pct, mem_mhz: 0 };
|
||
// a card that gets more efficient as the cap drops until it collapses: 70% wins (0.280 MH/W), 60% is 35% slower
|
||
let rows = vec![row_at(p(0, 100), 560.0, 124.0), row_at(p(0, 90), 510.0, 123.5), row_at(p(0, 80), 455.0, 123.2), row_at(p(0, 70), 400.0, 122.9), row_at(p(0, 60), 345.0, 80.0)];
|
||
assert_eq!(choose(&rows, 1.0).unwrap().point, p(0, 70));
|
||
// the tolerance: a 2% slower point with better MH/W loses at 1% and wins at 3%
|
||
let rows = vec![row_at(p(0, 100), 300.0, 100.0), row_at(p(2000, 100), 240.0, 98.0)];
|
||
assert_eq!(choose(&rows, 1.0).unwrap().point, p(0, 100));
|
||
assert_eq!(choose(&rows, 3.0).unwrap().point, p(2000, 100));
|
||
// within 1% on efficiency the higher rate wins; within 1% on both the lower draw
|
||
let rows = vec![row_at(p(0, 100), 500.0, 125.0), row_at(p(0, 80), 400.5, 100.3)];
|
||
assert_eq!(choose(&rows, 25.0).unwrap().point, p(0, 100));
|
||
let flat = vec![row_at(p(0, 100), 290.0, 124.0), row_at(p(0, 90), 290.5, 123.5), row_at(p(0, 80), 289.8, 124.2), row_at(p(0, 70), 290.2, 123.9)];
|
||
assert_eq!(choose(&flat, 1.0).unwrap().point, p(0, 80), "289.8 W is the lowest draw");
|
||
// marked rows never win, whatever their numbers
|
||
let mut faulted = row_at(p(1800, 100), 150.0, 123.0);
|
||
faulted.mark = Some(Mark::Faulted);
|
||
let mut hot = row_at(p(1900, 100), 160.0, 123.0);
|
||
hot.mark = Some(Mark::Hot);
|
||
let mut mem = row_at(p(2000, 100), 170.0, 123.0);
|
||
mem.mark = Some(Mark::MemoryClock);
|
||
assert_eq!(choose(&[faulted.clone(), hot, mem, row_at(p(0, 100), 290.0, 124.0)], 1.0).unwrap().point, p(0, 100));
|
||
assert!(choose(&[faulted], 1.0).is_none());
|
||
assert!(choose(&[], 1.0).is_none());
|
||
}
|
||
|
||
#[test]
|
||
fn the_guards_mark_a_step_so_it_cannot_win() {
|
||
let step = Step { point: Point { clock_mhz: 2000, power_pct: 100, mem_mhz: 0 }, watts: 575.0, kind: Kind::Clock };
|
||
let good = Samples { draws: vec![250.0, 251.0, 249.0], rates: vec![123.0], gclks: vec![1998.0], mclks: vec![2505.0], tmax: 70.0, faults: 0, unapplied: false };
|
||
assert_eq!(Row::from_samples(&step, &good, 2505.0).mark, Some(Mark::Ok));
|
||
// one rejected or mismatched hash: faulted
|
||
assert_eq!(Row::from_samples(&step, &Samples { faults: 1, ..good.clone() }, 2505.0).mark, Some(Mark::Faulted));
|
||
// the GPU at 85 C: hot
|
||
assert_eq!(Row::from_samples(&step, &Samples { tmax: 85.0, ..good.clone() }, 2505.0).mark, Some(Mark::Hot));
|
||
// the memory clock dragged under 95% of the baseline's: marked (the AMD gmax path on RDNA 4 cannot hold it)
|
||
assert_eq!(Row::from_samples(&step, &Samples { mclks: vec![2300.0], ..good.clone() }, 2505.0).mark, Some(Mark::MemoryClock));
|
||
assert_eq!(Row::from_samples(&step, &Samples { mclks: vec![2400.0], ..good.clone() }, 2505.0).mark, Some(Mark::Ok), "2,400 is 95.8%");
|
||
// the readback disagreed: unapplied
|
||
assert_eq!(Row::from_samples(&step, &Samples { unapplied: true, ..good.clone() }, 2505.0).mark, Some(Mark::Unapplied));
|
||
// under 3 draws or no rate: no readings, and the line says so
|
||
let r = Row::from_samples(&step, &Samples { draws: vec![250.0], ..good.clone() }, 0.0);
|
||
assert_eq!(r.mark, Some(Mark::NoReadings));
|
||
assert!(!r.usable());
|
||
assert!(r.line("c", 3).contains("mark=no_readings"), "{}", r.line("c", 3));
|
||
let r = Row::from_samples(&step, &good, 0.0);
|
||
assert_eq!(r.line("nvidia-ae432dc7-1", 7), "TUNE card=nvidia-ae432dc7-1 step=7 clock=2000 cap=100 mem=0 limit=575 watts=250.0 mhs=123.00 eff=0.4920 gclk=1998 mclk=2505 tmax=70 mark=ok");
|
||
}
|
||
|
||
#[test]
|
||
fn a_fault_during_a_step_reverts_it_and_the_run_goes_on() {
|
||
let t0 = Instant::now();
|
||
let timing = Timing { settle: Duration::from_secs(2), hold: Duration::from_secs(4), apply: Duration::from_secs(30) };
|
||
let limits = Limits { power_default_w: 300.0, power_min_w: 150.0, power_max_w: 300.0, clock_max_mhz: 0, clock_min_mhz: 0, mem_default_mhz: 0, mem_max_mhz: 0 };
|
||
let plan = Plan::full(&limits, Point { clock_mhz: 0, power_pct: 80, mem_mhz: 0 }, 1.0);
|
||
let mut run = Run::new(0, "0", "c", plan, 240.0, false, timing, t0);
|
||
let mut t = t0;
|
||
let mut limit = 240.0;
|
||
let mut applied: Vec<f64> = Vec::new();
|
||
let mut rows = Vec::new();
|
||
let mut finished = None;
|
||
for _ in 0..600 {
|
||
for o in run.tick(t, Readback { limit_w: limit, acked: true }) {
|
||
match o {
|
||
Out::Apply(s) => {
|
||
applied.push(s.watts);
|
||
limit = s.watts;
|
||
}
|
||
Out::Row(r) => rows.push(r),
|
||
Out::Finished(r) => finished = Some(r),
|
||
Out::Failed(e) => panic!("{e}"),
|
||
}
|
||
}
|
||
if matches!(run.phase, Phase::Holding { .. }) {
|
||
run.sample_telemetry(limit.min(260.0), 2500.0, 1300.0, 60.0);
|
||
run.sample_telemetry(limit.min(260.0) + 1.0, 2500.0, 1300.0, 60.0);
|
||
run.sample_telemetry(limit.min(260.0) - 1.0, 2500.0, 1300.0, 60.0);
|
||
run.sample_rate(100.0);
|
||
// the 70% step (210 W) sees a mismatched hash
|
||
if (limit - 210.0).abs() < 0.5 {
|
||
run.sample_fault();
|
||
}
|
||
}
|
||
if run.phase == Phase::Done {
|
||
break;
|
||
}
|
||
t += Duration::from_secs(1);
|
||
}
|
||
assert_eq!(rows.len(), 6, "300, 270, 240, 210, 180, 150 W");
|
||
assert_eq!(rows[3].point.power_pct, 70);
|
||
assert_eq!(rows[3].mark, Some(Mark::Faulted), "the faulted step is marked");
|
||
assert!(rows[3].faults >= 1);
|
||
// 100, 90 and 80% read the same 260 W draw at the same rate; 60% (150 W, the clamp) too in this model
|
||
// because the draw model caps at 260 W: the lowest draw wins among the ties, never the faulted 70%
|
||
let f = finished.expect("finished");
|
||
assert_ne!(f.point.power_pct, 70);
|
||
assert_eq!(applied.len(), 7, "every step's cap, then the chosen one");
|
||
assert_eq!(*applied.last().unwrap(), f.limit);
|
||
}
|
||
|
||
#[test]
|
||
fn the_confirm_plan_checks_the_prior_and_its_neighbour() {
|
||
let l = l5090();
|
||
let prior = Point { clock_mhz: 2472, power_pct: 100, mem_mhz: 0 };
|
||
let plan = Plan::confirm(&l, prior, Point { clock_mhz: 0, power_pct: 80, mem_mhz: 0 }, 1.0);
|
||
assert_eq!(plan.len(), 2);
|
||
let a = plan.next(&[]).unwrap();
|
||
assert_eq!((a.point, a.kind), (prior, Kind::Confirm));
|
||
let b = plan.next(&[row_at(prior, 220.0, 123.5)]).unwrap();
|
||
assert_eq!(b.point, Point { clock_mhz: 2781, power_pct: 100, mem_mhz: 0 }, "one clock step up");
|
||
// the neighbour within 1%: the prior stands
|
||
let rows = vec![row_at(prior, 220.0, 123.5), row_at(b.point, 250.0, 123.8)];
|
||
assert!(matches!(confirm_verdict(&rows, 1.0), Verdict::Keep(r) if r.point == prior));
|
||
// the neighbour 3% better on MH/W: the full plan is due
|
||
let rows = vec![row_at(prior, 220.0, 120.0), row_at(b.point, 212.0, 120.0)];
|
||
assert!(matches!(confirm_verdict(&rows, 1.0), Verdict::FullDue { .. }));
|
||
assert_eq!(confirm_verdict(&[], 1.0), Verdict::NoReadings);
|
||
// an unlocked prior at 100%: the neighbour is one power step down; at the top clock the neighbour is unlocked
|
||
let plan = Plan::confirm(&l, Point { clock_mhz: 0, power_pct: 100, mem_mhz: 0 }, Point::default(), 1.0);
|
||
assert_eq!(plan.next(&[row_at(Point { clock_mhz: 0, power_pct: 100, mem_mhz: 0 }, 1.0, 1.0)]).unwrap().point, Point { clock_mhz: 0, power_pct: 90, mem_mhz: 0 });
|
||
let plan = Plan::confirm(&l, Point { clock_mhz: 2900, power_pct: 100, mem_mhz: 0 }, Point::default(), 1.0);
|
||
assert_eq!(plan.next(&[row_at(Point { clock_mhz: 2900, power_pct: 100, mem_mhz: 0 }, 1.0, 1.0)]).unwrap().point.clock_mhz, 0);
|
||
// a prior outside the vendor's range is clamped, never applied as is
|
||
let plan = Plan::confirm(&l, Point { clock_mhz: 9000, power_pct: 30, mem_mhz: 0 }, Point::default(), 1.0);
|
||
assert_eq!(plan.next(&[]).unwrap().point, Point { clock_mhz: 3090, power_pct: 50, mem_mhz: 0 });
|
||
}
|
||
|
||
#[test]
|
||
fn a_baseline_plan_measures_the_card_as_it_runs() {
|
||
let t0 = Instant::now();
|
||
let timing = Timing { settle: Duration::from_secs(1), hold: Duration::from_secs(2), apply: Duration::from_secs(3) };
|
||
let before = Point { clock_mhz: 0, power_pct: 80, mem_mhz: 0 };
|
||
let mut run = Run::new(0, "0", "c", Plan::baseline(&l5090(), before, 1.0), 460.0, false, timing, t0);
|
||
// nothing acknowledges the request (Power control is off): the apply window passes and the hold starts anyway
|
||
let mut t = t0;
|
||
let mut finished = None;
|
||
for _ in 0..40 {
|
||
for o in run.tick(t, Readback { limit_w: 460.0, acked: false }) {
|
||
match o {
|
||
Out::Finished(r) => finished = Some(r),
|
||
Out::Failed(e) => panic!("{e}"),
|
||
_ => {}
|
||
}
|
||
}
|
||
if matches!(run.phase, Phase::Holding { .. }) {
|
||
for w in [290.0, 291.0, 289.0] {
|
||
run.sample_telemetry(w, 2700.0, 10500.0, 66.0);
|
||
}
|
||
run.sample_rate(122.3);
|
||
}
|
||
if run.phase == Phase::Done {
|
||
break;
|
||
}
|
||
t += Duration::from_secs(1);
|
||
}
|
||
let f = finished.expect("the baseline finished");
|
||
assert_eq!(f.point, before);
|
||
assert!((f.eff - 122.3 / 290.0).abs() < 1e-6);
|
||
assert_eq!(tuned_line(f.mhs, f.watts, f.eff), "Tuned: 122.3 MH/s at 290 W (0.422 MH/W)");
|
||
}
|
||
|
||
#[test]
|
||
fn the_record_and_the_prior_round_trip_through_the_manifest_shape() {
|
||
let chosen = row_at(Point { clock_mhz: 2472, power_pct: 100, mem_mhz: 0 }, 220.0, 123.5);
|
||
let rec = record_json(1_791_230_000.0, "8f3a2c1d", "0.3.10", "windows", "NVIDIA GeForce RTX 5090", "nvidia", "581.57", "l128w16", PlanKind::Full, &[chosen.clone()], Some(&chosen), None, false);
|
||
assert_eq!(rec["floor"], false);
|
||
assert_eq!(rec["card"], "NVIDIA_GeForce_RTX_5090");
|
||
assert_eq!(rec["key"], "NVIDIA_GeForce_RTX_5090|581|l128w16");
|
||
assert_eq!(rec["driver_major"], "581");
|
||
assert_eq!(rec["plan"], "full");
|
||
assert_eq!(rec["chosen"]["clock_mhz"], 2472);
|
||
assert_eq!(rec["eff"], 0.5614);
|
||
assert!(rec.get("address").is_none() && rec.get("host").is_none());
|
||
// the manifest's tuning section: the kernel-variant cards object stays, priors and ember sit beside it
|
||
let tuning: serde_json::Value = serde_json::json!({
|
||
"updated": "2026-10-05T22:00:00Z",
|
||
"cards": { "NVIDIA_GeForce_RTX_5090": { "variant": "u2-ldg", "race": true, "candidates": ["u2-ldg", "ldg", "base"] } },
|
||
"ember": { "enabled": true, "min_samples": 5, "rate_tolerance_pct": 1.0 },
|
||
"priors": { "NVIDIA_GeForce_RTX_5090|581|l128w16": { "clock_mhz": 2472, "power_pct": 100, "eff": 0.561, "mhs": 123.5, "watts": 220.0, "spread_pct": 2.1, "samples": 7 },
|
||
"AMD_Radeon_RX_9070_XT|32|l128w16": { "clock_mhz": 2600, "power_pct": 90, "eff": 0.1, "mhs": 17.7, "watts": 177.0, "spread_pct": 4.0, "samples": 3 } }
|
||
});
|
||
let s = settings_of(Some(&tuning));
|
||
assert_eq!(s, Settings { enabled: true, min_samples: 5, tolerance_pct: 1.0, period_s: PERIOD_S });
|
||
let p = prior_of(Some(&tuning), "NVIDIA_GeForce_RTX_5090|581|l128w16", s.min_samples).unwrap();
|
||
assert_eq!(p.point, Point { clock_mhz: 2472, power_pct: 100, mem_mhz: 0 });
|
||
assert_eq!(p.samples, 7);
|
||
assert!(prior_of(Some(&tuning), "AMD_Radeon_RX_9070_XT|32|l128w16", 5).is_none(), "3 samples are under the floor");
|
||
assert!(prior_of(Some(&tuning), "AMD_Radeon_RX_9070_XT|32|l128w16", 3).is_some());
|
||
assert!(prior_of(Some(&tuning), "nothing|0|v2", 5).is_none());
|
||
assert!(prior_of(None, "x", 5).is_none());
|
||
// the kill switch, and the defaults without a section
|
||
let off: serde_json::Value = serde_json::json!({ "cards": {}, "ember": { "enabled": false } });
|
||
assert!(!settings_of(Some(&off)).enabled);
|
||
assert_eq!(settings_of(None), Settings::default());
|
||
assert_eq!(settings_of(Some(&serde_json::json!({ "cards": {}, "ember": { "rate_tolerance_pct": 99 } }))).tolerance_pct, 1.0, "out of range: the default");
|
||
assert_eq!(prior_key("AMD Radeon RX 9070 XT", "32.0.15801.1", ""), "AMD_Radeon_RX_9070_XT|32|v2");
|
||
assert_eq!(driver_major(""), "0");
|
||
assert_eq!(program_class(128, 16), "l128w16");
|
||
assert_eq!(program_class(0, 0), "v2");
|
||
}
|
||
|
||
/// Ember 2: a synthetic memory-latency-bound card. Rate rises 1% per 100 MHz of memory above the default and
|
||
/// falls only 0.3% per 100 MHz of core below the maximum; draw falls 8 W per 100 MHz of core and rises 2 W per
|
||
/// 100 MHz of memory. The climb must walk memory up and core down and converge in under five probes.
|
||
fn synthetic(p: Point) -> Row {
|
||
let mem = if p.mem_mhz == 0 { 13801.0 } else { p.mem_mhz as f64 };
|
||
let core = if p.clock_mhz == 0 { 3090.0 } else { p.clock_mhz as f64 };
|
||
let mhs = 127.0 * (1.0 + 0.015 * (mem - 13801.0) / 100.0) * (1.0 - 0.003 * (3090.0 - core) / 100.0);
|
||
let watts = 310.0 - 8.0 * (3090.0 - core) / 100.0 + 2.0 * (mem - 13801.0) / 100.0;
|
||
Row { point: p, limit: 575.0, watts, mhs, eff: mhs / watts, draws: 12, rates: 6, gclk: core, mclk: mem, tmax: 65.0, faults: 0, mark: Some(Mark::Ok) }
|
||
}
|
||
|
||
#[test]
|
||
fn the_climb_walks_memory_up_and_core_down_and_converges_in_five_probes() {
|
||
// a 2 GHz memory range above the default (the headroom a 5090 has in practice; PC 1's driver reports 14,001)
|
||
let l = Limits { mem_max_mhz: 15801, ..l5090() };
|
||
let plan = Plan::climb(&l, Point { clock_mhz: 0, power_pct: 100, mem_mhz: 0 }, Goal::Balanced, 1.0);
|
||
assert_eq!(plan.len(), 5);
|
||
let mut rows = Vec::new();
|
||
while let Some(step) = plan.next(&rows) {
|
||
assert_eq!(step.kind, Kind::Climb);
|
||
rows.push(synthetic(step.point));
|
||
}
|
||
assert_eq!(rows.len(), 5, "the budget");
|
||
assert_eq!(rows[0].point, Point { clock_mhz: 0, power_pct: 100, mem_mhz: 0 }, "it starts at the start point");
|
||
// every probe moved one knob the right way: memory never down, core never up, and both inside the vendor's range
|
||
for w in rows.windows(2) {
|
||
let (a, b) = (w[0].point, w[1].point);
|
||
assert!(b.mem_mhz >= a.mem_mhz || b.mem_mhz == 0, "{a:?} -> {b:?}");
|
||
assert!(b.mem_mhz <= 15801 && (b.clock_mhz == 0 || b.clock_mhz >= 1854), "{b:?}");
|
||
}
|
||
let best = choose(&rows, plan.tolerance_pct).unwrap();
|
||
assert!(best.eff > rows[0].eff, "the chosen point beats the start: {:.4} > {:.4}", best.eff, rows[0].eff);
|
||
assert!(best.point.mem_mhz > 13801 || (best.point.clock_mhz > 0 && best.point.clock_mhz < 3090), "it moved a knob");
|
||
}
|
||
|
||
#[test]
|
||
fn a_refused_probe_backs_that_knob_off_for_good() {
|
||
let l = Limits { mem_max_mhz: 15801, ..l5090() };
|
||
let plan = Plan::climb(&l, Point { clock_mhz: 0, power_pct: 100, mem_mhz: 0 }, Goal::Balanced, 1.0);
|
||
let mut rows = vec![synthetic(plan.next(&[]).unwrap().point)];
|
||
// the first probe (memory up) draws a mismatched hash: refused
|
||
let probe = plan.next(&rows).unwrap();
|
||
assert!(probe.point.mem_mhz > 13801, "memory first: {:?}", probe.point);
|
||
let mut bad = synthetic(probe.point);
|
||
bad.faults = 1;
|
||
bad.mark = Some(Mark::Faulted);
|
||
rows.push(bad);
|
||
// from here every probe leaves the memory clock alone
|
||
while let Some(step) = plan.next(&rows) {
|
||
assert_eq!(step.point.mem_mhz, 0, "memory backed off: {:?}", step.point);
|
||
rows.push(synthetic(step.point));
|
||
}
|
||
assert!(rows.len() >= 3 && rows.len() <= 5);
|
||
assert!(choose(&rows, 1.0).unwrap().point.mem_mhz == 0);
|
||
}
|
||
|
||
#[test]
|
||
fn each_goal_picks_its_point() {
|
||
let l = l5090();
|
||
let p = |c: u32, m: u32| Point { clock_mhz: c, power_pct: 100, mem_mhz: m };
|
||
// three points: the fastest (slightly less efficient), the most efficient (4% slower), and the start
|
||
let rows = vec![synthetic(p(0, 0)), synthetic(p(0, 14001)), synthetic(p(1854, 0))];
|
||
let eff_plan = Plan::climb(&l, p(0, 0), Goal::Efficiency, 1.0);
|
||
let bal_plan = Plan::climb(&l, p(0, 0), Goal::Balanced, 1.0);
|
||
let rate_plan = Plan::climb(&l, p(0, 0), Goal::MaxRate, 1.0);
|
||
assert_eq!(eff_plan.tolerance_pct, 10.0);
|
||
assert_eq!(bal_plan.tolerance_pct, 1.0);
|
||
assert_eq!(rate_plan.tolerance_pct, 0.0);
|
||
let by = |plan: &Plan| rows.iter().filter(|r| r.usable() && r.mhs >= rows.iter().map(|x| x.mhs).fold(0.0, f64::max) * (1.0 - plan.tolerance_pct / 100.0)).max_by(|a, b| plan.score(a).partial_cmp(&plan.score(b)).unwrap()).unwrap().point;
|
||
assert_eq!(by(&rate_plan), p(0, 14001), "maximum rate takes the fastest point");
|
||
assert_eq!(by(&eff_plan), p(1854, 0), "efficiency takes the most MH/W inside 10%");
|
||
assert_eq!(by(&bal_plan), p(0, 14001), "balanced keeps within 1% of the top rate");
|
||
assert_eq!(Goal::parse("efficiency"), Goal::Efficiency);
|
||
assert_eq!(Goal::parse("rate"), Goal::MaxRate);
|
||
assert_eq!(Goal::parse("anything"), Goal::Balanced);
|
||
assert!((pounds_per_day(310.0, 28.5) - 2.1204).abs() < 1e-3, "310 W a day at 28.5 p/kWh = £2.12");
|
||
}
|
||
|
||
#[test]
|
||
fn control_reasons_per_vendor() {
|
||
let l = l5090();
|
||
assert!(control_reason("nvidia", &l, "0", true, false).is_none());
|
||
assert!(control_reason("nvidia", &l, "0", false, false).unwrap().contains("Power control"));
|
||
assert!(control_reason("nvidia", &Limits::default(), "0", true, false).unwrap().contains("limits"));
|
||
assert!(control_reason("apple", &Limits::default(), "", true, false).unwrap().contains("Apple silicon"));
|
||
assert!(control_reason("amd", &Limits::default(), "1", false, false).unwrap().contains("igneum-gpu-telemetry"));
|
||
if cfg!(target_os = "linux") {
|
||
assert!(control_reason("amd", &Limits::default(), "1", false, true).unwrap().contains("root"));
|
||
} else {
|
||
assert!(control_reason("amd", &Limits::default(), "1", false, true).is_none());
|
||
}
|
||
}
|
||
}
|