igneum/app/igneum-app/src/ember.rs

1324 lines
66 KiB
Rust
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

//! Ember Tune: every card tuned for MH per watt out of the box, and the fleet's results folded into a prior that a
//! new card starts from (docs/plans/ember-tune.md). Two knobs per card: the power limit (percent of the card's
//! default) and the core clock cap (MHz; 0 = unlocked). The memory clock is never touched, and a step that drags it
//! down is marked and cannot win. This file is the logic, driven by an explicit clock so the tests run without a
//! card: the plans (full, confirm, baseline), the per-step rows with their marks, the choice rule, the fleet record,
//! the prior lookup and the state machine. The engine (src/engine.rs, `tick_tune` and the `Cmd::Tune*` commands)
//! owns the processes: NVIDIA through nvidia-smi (`-pl`, `-lgc 0,<mhz>`, `-rgc`; administrator rights, so only with
//! Power control on), AMD through igneum-gpu-telemetry (`--set-plimit`, `--set-gmax`, `--reset`; no elevation on
//! Windows), Apple measure only.
//!
//! The lines in the app log (and on stdout under --sweep, which the PC job reads):
//! TUNE start card=<label> name=<name> plan=full|confirm|baseline steps=<n> clock_max=<MHz> default=<W> ...
//! TUNE card=<label> step=<i> clock=<MHz|0> cap=<pct> limit=<W> watts=<W> mhs=<x> eff=<MH/W> gclk=<MHz> mclk=<MHz> tmax=<C> mark=ok|...
//! TUNE chosen card=<label> clock=<MHz|0> cap=<pct> limit=<W> watts=<W> mhs=<x> eff=<MH/W>
//! TUNE {json} the fleet record (one per finished plan; the intake receives it with the log upload)
//! TUNE aborted card=<label> reason=<text>
use std::time::{Duration, Instant};
/// The power steps, percent of the card's default limit, highest first (the same ladder as src/sweep.rs).
pub const POWER_STEPS_PCT: [u32; 6] = [100, 90, 80, 70, 60, 50];
/// The clock steps, percent of the card's maximum core clock, highest first; 100 = unlocked (the card's own boost).
/// 6 October 2026, run 6: the 5090's best MH/W sat on the 60% floor (1,854 MHz: 0.563 MH/W, the rate within 0.15%),
/// so the ladder and the floor go to 45% of the maximum; the 1% rate tolerance is the guard below that
pub const CLOCK_STEPS_PCT: [u32; 7] = [100, 90, 80, 70, 60, 50, 45];
/// A card's clock floor when the vendor reports none: this share of its maximum core clock.
pub const CLOCK_FLOOR_PCT: u32 = 45;
/// A point may lose this much rate against the fastest point and still win on MH per watt (the manifest can change it).
pub const RATE_TOLERANCE_PCT: f64 = 1.0;
/// A step whose hottest GPU reading reaches this is marked hot and cannot win (the engine aborts at 90).
pub const HOT_C: f64 = 85.0;
/// A step whose mean memory clock falls below this share of the baseline's is marked and cannot win.
pub const MCLK_HOLD: f64 = 0.95;
/// A fleet prior is used by a new card when it carries at least this many samples (the manifest can change it).
pub const PRIOR_MIN_SAMPLES: u32 = 5;
/// A confirm check that beats its prior by more than this on MH per watt asks for the full plan.
pub const CONFIRM_GAIN_PCT: f64 = 1.0;
/// A tune repeats this often (the manifest can change it).
pub const PERIOD_S: u64 = 7 * 86_400;
/// The worker must have mined this long before a tune starts, and this much must remain to the hour boundary.
pub const STABLE_S: u64 = 120;
pub const NEEDS_S: i64 = 600;
/// What a card's vendor allows (never exceeded, never undercut).
#[derive(Clone, Debug, Default, PartialEq)]
pub struct Limits {
pub power_default_w: f64,
pub power_min_w: f64,
pub power_max_w: f64,
/// the maximum core clock the vendor reports (nvidia-smi clocks.max.gr; the ADLX gmax_range top); 0 = unknown
pub clock_max_mhz: u32,
/// the lowest cap the vendor allows (the ADLX gmax_range floor); 0 = CLOCK_FLOOR_PCT of the maximum
pub clock_min_mhz: u32,
/// Ember 2: the memory clock the card runs at by default and the vendor's maximum (nvidia-smi clocks.mem and
/// clocks.max.mem); 0 = no memory knob (AMD through ADLX on RDNA 4 exposes none)
pub mem_default_mhz: u32,
pub mem_max_mhz: u32,
}
impl Limits {
pub fn clock_floor(&self) -> u32 {
if self.clock_min_mhz > 0 {
self.clock_min_mhz
} else {
self.clock_max_mhz * CLOCK_FLOOR_PCT / 100
}
}
/// A clock cap inside the vendor's range; 0 stays 0 (unlocked).
pub fn clamp_clock(&self, mhz: u32) -> u32 {
if mhz == 0 || self.clock_max_mhz == 0 {
return 0;
}
mhz.clamp(self.clock_floor(), self.clock_max_mhz)
}
/// A memory clock inside the vendor's range; 0 stays 0 (the default).
pub fn clamp_mem(&self, mhz: u32) -> u32 {
if mhz == 0 || self.mem_max_mhz == 0 || self.mem_default_mhz == 0 {
return 0;
}
mhz.clamp(self.mem_default_mhz, self.mem_max_mhz)
}
/// The watts a power percent asks for, inside the card's min and max, rounded to a watt.
pub fn watts_for(&self, pct: u32) -> f64 {
let mut w = self.power_default_w * pct as f64 / 100.0;
if self.power_min_w > 0.0 {
w = w.max(self.power_min_w);
}
if self.power_max_w > 0.0 {
w = w.min(self.power_max_w);
}
w.round()
}
}
/// One setting of the knobs (Ember 2 adds the memory clock: 0 = the driver's default).
#[derive(Clone, Copy, Debug, Default, PartialEq, Eq, Hash)]
pub struct Point {
/// the core clock cap in MHz; 0 = unlocked
pub clock_mhz: u32,
/// the power limit, percent of the default
pub power_pct: u32,
/// the memory clock in MHz (NVIDIA `-lmc`, locked to one value); 0 = the driver's default
pub mem_mhz: u32,
}
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
pub enum Kind {
/// the card as it runs now (the before number; the only step a measure-only card gets)
Baseline,
Power,
Clock,
/// the fleet prior and one neighbour
Confirm,
/// Ember 2: a hill-climb probe
Climb,
}
#[derive(Clone, Debug, PartialEq)]
pub struct Step {
pub point: Point,
/// the limit to set, watts
pub watts: f64,
pub kind: Kind,
}
/// What the tune is for (Settings > Ember Tune > goal). The rate floor is the share of the best rate seen a point
/// must keep to win on MH per watt: efficiency keeps 90%, balanced 99% (the 1% rule of lever 3), maximum rate
/// takes the fastest point and uses MH per watt only to break ties.
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
pub enum Goal {
Efficiency,
Balanced,
MaxRate,
}
impl Goal {
pub fn parse(s: &str) -> Goal {
match s {
"efficiency" | "eff" => Goal::Efficiency,
"rate" | "max_rate" | "maximum" => Goal::MaxRate,
_ => Goal::Balanced,
}
}
pub fn name(&self) -> &'static str {
match self {
Goal::Efficiency => "efficiency",
Goal::Balanced => "balanced",
Goal::MaxRate => "rate",
}
}
/// The rate tolerance the choice rule uses, percent under the best rate.
pub fn tolerance_pct(&self, manifest_default: f64) -> f64 {
match self {
Goal::Efficiency => 10.0,
Goal::Balanced => manifest_default,
Goal::MaxRate => 0.0,
}
}
}
/// Pounds a day for a draw at a price in pence per kWh: watts × 24 h / 1000 × price / 100.
pub fn pounds_per_day(watts: f64, pence_per_kwh: f64) -> f64 {
watts * 24.0 / 1000.0 * pence_per_kwh / 100.0
}
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
pub enum PlanKind {
Full,
Confirm,
Baseline,
/// Ember 2: the hill-climb over memory up and core down from the start point
Climb,
}
impl PlanKind {
pub fn name(&self) -> &'static str {
match self {
PlanKind::Full => "full",
PlanKind::Confirm => "confirm",
PlanKind::Baseline => "baseline",
PlanKind::Climb => "climb",
}
}
}
/// The steps of a tune. The full plan is a coordinate search: the power ladder at the unlocked clock, then the clock
/// ladder at the power point that ladder chose (asked for through `next`, so the second half depends on the rows).
#[derive(Clone, Debug, PartialEq)]
pub struct Plan {
pub kind: PlanKind,
pub limits: Limits,
pub before: Point,
pub tolerance_pct: f64,
power: Vec<Step>,
clock_pcts: Vec<u32>,
fixed: Vec<Step>,
/// Ember 2 (Climb): the start point, the step sizes and the step budget
climb: Option<Climb>,
}
/// The hill-climb's shape: from `start`, each probe moves the memory clock up by `mem_step` or the core clock down
/// by `core_step` (or both), keeps the move when the goal's score improves, else turns to the other knob; a
/// refused step (a fault on it) backs that knob off for good. At most `budget` steps including the start.
#[derive(Clone, Debug, PartialEq)]
pub struct Climb {
pub start: Point,
pub mem_step: u32,
pub core_step: u32,
pub budget: usize,
pub goal: Goal,
}
impl Plan {
/// Ember 2: the hill-climb. The start is the fleet prior (or the card's current point); a probe step is 5% of
/// the memory range above the default (0 when the card has no memory knob) and 5% of the maximum core clock;
/// five steps of 60 s converge in under 10 minutes.
pub fn climb(limits: &Limits, start: Point, goal: Goal, tolerance_pct: f64) -> Plan {
let mem_step = if limits.mem_max_mhz > limits.mem_default_mhz { ((limits.mem_max_mhz - limits.mem_default_mhz) / 20).max(25) } else { 0 };
let core_step = if limits.clock_max_mhz > 0 { (limits.clock_max_mhz / 20).max(25) } else { 0 };
let start = Point { clock_mhz: limits.clamp_clock(start.clock_mhz), power_pct: start.power_pct.clamp(50, 100), mem_mhz: limits.clamp_mem(start.mem_mhz) };
Plan { kind: PlanKind::Climb, limits: limits.clone(), before: start, tolerance_pct: goal.tolerance_pct(tolerance_pct), power: Vec::new(), clock_pcts: Vec::new(), fixed: Vec::new(), climb: Some(Climb { start, mem_step, core_step, budget: 5, goal }) }
}
/// The goal's score of a row: MH per watt for efficiency and balanced, the rate for maximum rate.
pub fn score(&self, r: &Row) -> f64 {
match self.climb.as_ref().map(|c| c.goal) {
Some(Goal::MaxRate) => r.mhs,
_ => r.eff,
}
}
/// The climb's next probe from the rows so far: None when the budget is spent or no move is left.
fn climb_next(&self, rows: &[Row]) -> Option<Step> {
let c = self.climb.as_ref()?;
let step = |p: Point| Step { point: p, watts: self.limits.watts_for(p.power_pct), kind: Kind::Climb };
if rows.is_empty() {
return Some(step(c.start));
}
if rows.len() >= c.budget {
return None;
}
// the best usable row so far is the hill's top; a knob that produced a marked (refused) row is backed off
let best = rows.iter().filter(|r| r.usable()).max_by(|a, b| self.score(a).partial_cmp(&self.score(b)).unwrap_or(std::cmp::Ordering::Equal))?;
let refused_mem = rows.iter().any(|r| !r.usable() && r.point.mem_mhz > best.point.mem_mhz);
let refused_core = rows.iter().any(|r| !r.usable() && r.point.clock_mhz != 0 && (best.point.clock_mhz == 0 || r.point.clock_mhz < best.point.clock_mhz));
let last = rows.last()?;
let mem_up = |p: Point| -> Option<Point> {
if c.mem_step == 0 || refused_mem { return None; }
let base = if p.mem_mhz == 0 { self.limits.mem_default_mhz } else { p.mem_mhz };
let m = self.limits.clamp_mem(base + c.mem_step);
(m > 0 && m != p.mem_mhz && m != base).then_some(Point { mem_mhz: m, ..p })
};
let core_down = |p: Point| -> Option<Point> {
if c.core_step == 0 || refused_core { return None; }
let base = if p.clock_mhz == 0 { self.limits.clock_max_mhz } else { p.clock_mhz };
let k = self.limits.clamp_clock(base.saturating_sub(c.core_step));
(k > 0 && k != p.clock_mhz).then_some(Point { clock_mhz: k, ..p })
};
let tried = |p: Point| rows.iter().any(|r| r.point == p);
// the last move improved: keep going the same way from the top; else turn: memory first, then core, then both
let last_improved = last.usable() && last.point == best.point && rows.len() > 1;
let last_was_mem = rows.len() > 1 && last.point.mem_mhz != rows[rows.len() - 2].point.mem_mhz;
let candidates: Vec<Option<Point>> = if last_improved && last_was_mem {
vec![mem_up(best.point), core_down(best.point)]
} else if last_improved {
vec![core_down(best.point), mem_up(best.point)]
} else {
vec![mem_up(best.point), core_down(best.point), mem_up(best.point).and_then(core_down)]
};
candidates.into_iter().flatten().find(|p| !tried(*p)).map(step)
}
/// Power 100% to 50% at the unlocked clock (duplicate watts dropped, as the card's floor clamps them), then the
/// clock ladder 90% to the floor of the maximum core clock at the chosen power. A card without a readable
/// maximum clock gets the power ladder only; a card without a default limit gets the clock ladder only.
pub fn full(limits: &Limits, before: Point, tolerance_pct: f64) -> Plan {
let mut power = Vec::new();
if limits.power_default_w > 0.0 {
for pct in POWER_STEPS_PCT {
let w = limits.watts_for(pct);
if power.last().map(|s: &Step| (s.watts - w).abs() < 0.5).unwrap_or(false) {
continue;
}
power.push(Step { point: Point { clock_mhz: 0, power_pct: pct, mem_mhz: 0 }, watts: w, kind: Kind::Power });
}
}
let clock_pcts = if limits.clock_max_mhz > 0 { CLOCK_STEPS_PCT[1..].to_vec() } else { Vec::new() };
Plan { kind: PlanKind::Full, limits: limits.clone(), before, tolerance_pct, power, clock_pcts, fixed: Vec::new(), climb: None }
}
/// The prior's point, then one neighbour: the next clock step up when the prior caps the clock (is the cap
/// costing rate?), else one power step down (is there efficiency left?).
pub fn confirm(limits: &Limits, prior: Point, before: Point, tolerance_pct: f64) -> Plan {
let p = Point { clock_mhz: limits.clamp_clock(prior.clock_mhz), power_pct: prior.power_pct.clamp(50, 100), mem_mhz: limits.clamp_mem(prior.mem_mhz) };
let first = Step { point: p, watts: limits.watts_for(p.power_pct), kind: Kind::Confirm };
let neighbour = if p.clock_mhz > 0 && limits.clock_max_mhz > 0 {
let up = p.clock_mhz + limits.clock_max_mhz / 10;
let clock = if up >= limits.clock_max_mhz { 0 } else { limits.clamp_clock(up) };
Point { clock_mhz: clock, power_pct: p.power_pct, mem_mhz: p.mem_mhz }
} else {
Point { clock_mhz: p.clock_mhz, power_pct: (p.power_pct.saturating_sub(10)).max(50), mem_mhz: p.mem_mhz }
};
let mut fixed = vec![first];
if neighbour != p {
fixed.push(Step { point: neighbour, watts: limits.watts_for(neighbour.power_pct), kind: Kind::Confirm });
}
Plan { kind: PlanKind::Confirm, limits: limits.clone(), before, tolerance_pct, power: Vec::new(), clock_pcts: Vec::new(), fixed, climb: None }
}
/// One step at the card's current point: the before number, and all a measure-only card (Apple, or NVIDIA
/// with Power control off) reports.
pub fn baseline(limits: &Limits, before: Point, tolerance_pct: f64) -> Plan {
let fixed = vec![Step { point: before, watts: limits.watts_for(before.power_pct), kind: Kind::Baseline }];
Plan { kind: PlanKind::Baseline, limits: limits.clone(), before, tolerance_pct, power: Vec::new(), clock_pcts: Vec::new(), fixed, climb: None }
}
/// How many steps the plan has at most (the clock ladder counts whether or not it runs).
pub fn len(&self) -> usize {
if let Some(c) = &self.climb {
return c.budget;
}
self.fixed.len() + self.power.len() + self.clock_pcts.len()
}
pub fn is_empty(&self) -> bool {
self.len() == 0
}
/// The next step after `rows` (one row per step done so far), or None when the plan is complete.
pub fn next(&self, rows: &[Row]) -> Option<Step> {
if self.climb.is_some() {
return self.climb_next(rows);
}
let i = rows.len();
if !self.fixed.is_empty() {
return self.fixed.get(i).cloned();
}
if i < self.power.len() {
return Some(self.power[i].clone());
}
let k = i - self.power.len();
let pct = *self.clock_pcts.get(k)?;
// the clock ladder rides the power point the power ladder chose (the before point when nothing won)
let power_pct = if self.power.is_empty() { self.before.power_pct } else { choose(&rows[..self.power.len()], self.tolerance_pct).map(|r| r.point.power_pct).unwrap_or(self.before.power_pct) };
let clock = self.limits.clamp_clock(self.limits.clock_max_mhz * pct / 100);
// a step whose clamp lands on the previous step's clock is dropped (the floor was reached)
if rows.last().map(|r| r.point.clock_mhz == clock).unwrap_or(false) {
return None;
}
Some(Step { point: Point { clock_mhz: clock, power_pct, mem_mhz: 0 }, watts: self.limits.watts_for(power_pct), kind: Kind::Clock })
}
}
/// Why a step cannot win, or Ok.
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
pub enum Mark {
Ok,
NoReadings,
/// a rejected or mismatched hash during the hold: the step was reverted and marked
Faulted,
/// the GPU reached HOT_C during the hold
Hot,
/// the memory clock fell under MCLK_HOLD of the baseline's: the knob dragged it
MemoryClock,
/// the setting did not take (the readback disagreed)
Unapplied,
}
impl Mark {
pub fn name(&self) -> &'static str {
match self {
Mark::Ok => "ok",
Mark::NoReadings => "no_readings",
Mark::Faulted => "faulted",
Mark::Hot => "hot",
Mark::MemoryClock => "memory_clock_dropped",
Mark::Unapplied => "unapplied",
}
}
}
/// One step's result. `watts` is the mean draw during the hold (what the miner pays for), `limit` the cap set,
/// `eff` MH per watt of draw (0 when the step had no readings).
#[derive(Clone, Debug, Default, PartialEq)]
pub struct Row {
pub point: Point,
pub limit: f64,
pub watts: f64,
pub mhs: f64,
pub eff: f64,
pub draws: usize,
pub rates: usize,
pub gclk: f64,
pub mclk: f64,
pub tmax: f64,
pub faults: u32,
pub mark: Option<Mark>,
}
/// What a hold collected.
#[derive(Clone, Debug, Default)]
pub struct Samples {
pub draws: Vec<f64>,
pub rates: Vec<f64>,
pub gclks: Vec<f64>,
pub mclks: Vec<f64>,
pub tmax: f64,
pub faults: u32,
pub unapplied: bool,
}
fn mean(v: &[f64]) -> f64 {
if v.is_empty() {
0.0
} else {
v.iter().sum::<f64>() / v.len() as f64
}
}
impl Row {
/// `baseline_mclk` is the memory clock of the first row (0 = unknown): the memory-clock guard.
pub fn from_samples(step: &Step, s: &Samples, baseline_mclk: f64) -> Row {
let watts = mean(&s.draws);
let mhs = mean(&s.rates);
let mclk = mean(&s.mclks);
let usable = s.draws.len() >= 3 && !s.rates.is_empty() && watts > 1.0;
let mark = if s.faults > 0 {
Mark::Faulted
} else if s.unapplied {
Mark::Unapplied
} else if !usable {
Mark::NoReadings
} else if s.tmax >= HOT_C {
Mark::Hot
} else if baseline_mclk > 0.0 && mclk > 0.0 && mclk < MCLK_HOLD * baseline_mclk {
Mark::MemoryClock
} else {
Mark::Ok
};
Row { point: step.point, limit: step.watts, watts, mhs, eff: if usable { mhs / watts } else { 0.0 }, draws: s.draws.len(), rates: s.rates.len(), gclk: mean(&s.gclks), mclk, tmax: s.tmax, faults: s.faults, mark: Some(mark) }
}
pub fn usable(&self) -> bool {
self.eff > 0.0 && self.mark == Some(Mark::Ok)
}
/// `TUNE card=<label> step=<i> clock=<MHz> cap=<pct> limit=<W> watts=<W> mhs=<x> eff=<MH/W> gclk=<MHz> mclk=<MHz> tmax=<C> mark=<m>`
pub fn line(&self, card: &str, i: usize) -> String {
format!(
"TUNE card={card} step={i} clock={} cap={} mem={} limit={:.0} watts={:.1} mhs={:.2} eff={:.4} gclk={:.0} mclk={:.0} tmax={:.0} mark={}{}",
self.point.clock_mhz,
self.point.power_pct,
self.point.mem_mhz,
self.limit,
self.watts,
self.mhs,
self.eff,
self.gclk,
self.mclk,
self.tmax,
self.mark.unwrap_or(Mark::NoReadings).name(),
if self.faults > 0 { format!(" faults={}", self.faults) } else { String::new() }
)
}
pub fn json(&self) -> serde_json::Value {
serde_json::json!({ "clock_mhz": self.point.clock_mhz, "power_pct": self.point.power_pct, "mem_mhz": self.point.mem_mhz, "limit_w": self.limit, "watts": r1(self.watts), "mhs": r2(self.mhs), "eff": r4(self.eff), "gclk": self.gclk.round(), "mclk": self.mclk.round(), "tmax": self.tmax.round(), "faults": self.faults, "mark": self.mark.unwrap_or(Mark::NoReadings).name() })
}
}
fn r1(x: f64) -> f64 {
(x * 10.0).round() / 10.0
}
fn r2(x: f64) -> f64 {
(x * 100.0).round() / 100.0
}
fn r4(x: f64) -> f64 {
(x * 10_000.0).round() / 10_000.0
}
/// The choice: among the usable rows within `tolerance_pct` of the fastest row's rate, the best MH per watt; within
/// 1% on efficiency the higher rate wins; within 1% on both, the lower draw. A card never gives up more than the
/// tolerance in blocks for the saving. Marked rows never win.
pub fn choose(rows: &[Row], tolerance_pct: f64) -> Option<Row> {
let top = rows.iter().filter(|r| r.usable()).map(|r| r.mhs).fold(0.0, f64::max);
if top <= 0.0 {
return None;
}
let floor = top * (1.0 - tolerance_pct.max(0.0) / 100.0);
let mut best: Option<&Row> = None;
for r in rows.iter().filter(|r| r.usable() && r.mhs >= floor) {
best = Some(match best {
None => r,
Some(b) => {
let eff_tie = (r.eff - b.eff).abs() <= 0.01 * b.eff.max(r.eff);
if !eff_tie {
if r.eff > b.eff { r } else { b }
} else {
let mhs_tie = (r.mhs - b.mhs).abs() <= 0.01 * b.mhs.max(r.mhs);
if !mhs_tie {
if r.mhs > b.mhs { r } else { b }
} else if r.watts < b.watts {
r
} else {
b
}
}
}
});
}
best.cloned()
}
/// A confirm check's verdict: the prior stands, or the neighbour beat it by over CONFIRM_GAIN_PCT (the full plan is
/// due), or nothing could be read.
#[derive(Clone, Debug, PartialEq)]
pub enum Verdict {
Keep(Row),
FullDue { prior: Row, better: Row },
NoReadings,
}
pub fn confirm_verdict(rows: &[Row], tolerance_pct: f64) -> Verdict {
let Some(prior) = rows.first().filter(|r| r.usable()).cloned() else {
return match choose(rows, tolerance_pct) {
Some(r) => Verdict::FullDue { prior: Row::default(), better: r },
None => Verdict::NoReadings,
};
};
match choose(rows, tolerance_pct) {
Some(best) if best.point != prior.point && best.eff > prior.eff * (1.0 + CONFIRM_GAIN_PCT / 100.0) => Verdict::FullDue { prior, better: best },
_ => Verdict::Keep(prior),
}
}
// ---- the fleet prior ------------------------------------------------------------------------------------------
/// The key a prior is filed under: card model (spaces as underscores), driver major, program class.
pub fn prior_key(card: &str, driver: &str, class: &str) -> String {
format!("{}|{}|{}", card.trim().replace(' ', "_"), driver_major(driver), if class.is_empty() { "v2" } else { class })
}
/// "581.57" -> "581", "32.0.15801" -> "32", "" -> "0".
pub fn driver_major(driver: &str) -> String {
let d: String = driver.trim().chars().take_while(|c| c.is_ascii_digit()).collect();
if d.is_empty() { "0".into() } else { d }
}
/// The program class from a race line's features (loads and wide loads per hash); the generator fixes both today.
pub fn program_class(loads: u32, wide: u32) -> String {
if loads == 0 { "v2".into() } else { format!("l{loads}w{wide}") }
}
/// A fleet prior as the signed manifest carries it under tuning.priors.<key>.
#[derive(Clone, Debug, Default, PartialEq)]
pub struct Prior {
pub point: Point,
pub eff: f64,
pub mhs: f64,
pub watts: f64,
/// the spread of the chosen efficiency across the samples, percent of the median
pub spread_pct: f64,
pub samples: u32,
}
/// The tuning section's Ember settings: the kill switch and the thresholds (defaults when absent).
#[derive(Clone, Debug, PartialEq)]
pub struct Settings {
pub enabled: bool,
pub min_samples: u32,
pub tolerance_pct: f64,
pub period_s: u64,
}
impl Default for Settings {
fn default() -> Settings {
Settings { enabled: true, min_samples: PRIOR_MIN_SAMPLES, tolerance_pct: RATE_TOLERANCE_PCT, period_s: PERIOD_S }
}
}
/// Reads `tuning.ember` (absent = the defaults; `enabled: false` is the fleet-wide kill switch).
pub fn settings_of(tuning: Option<&serde_json::Value>) -> Settings {
let d = Settings::default();
let Some(e) = tuning.and_then(|t| t.get("ember")).filter(|e| e.is_object()) else { return d };
Settings {
enabled: e.get("enabled").and_then(|v| v.as_bool()).unwrap_or(d.enabled),
min_samples: e.get("min_samples").and_then(|v| v.as_u64()).map(|v| v as u32).unwrap_or(d.min_samples),
tolerance_pct: e.get("rate_tolerance_pct").and_then(|v| v.as_f64()).filter(|v| (0.0..=25.0).contains(v)).unwrap_or(d.tolerance_pct),
period_s: e.get("period_s").and_then(|v| v.as_u64()).filter(|v| *v >= 3600).unwrap_or(d.period_s),
}
}
/// The prior for a key, or None when the manifest has none or it carries under `min_samples` samples.
pub fn prior_of(tuning: Option<&serde_json::Value>, key: &str, min_samples: u32) -> Option<Prior> {
let p = tuning?.get("priors")?.get(key)?;
let n = |k: &str| p.get(k).and_then(|v| v.as_f64()).unwrap_or(0.0);
let samples = p.get("samples").and_then(|v| v.as_u64()).unwrap_or(0) as u32;
if samples < min_samples.max(1) {
return None;
}
let point = Point { clock_mhz: n("clock_mhz") as u32, power_pct: (n("power_pct") as u32).clamp(50, 100), mem_mhz: n("mem_mhz") as u32 };
if point.power_pct == 0 && point.clock_mhz == 0 {
return None;
}
Some(Prior { point, eff: n("eff"), mhs: n("mhs"), watts: n("watts"), spread_pct: n("spread_pct"), samples })
}
/// The fleet record: one `TUNE {json}` line. `machine` is a hash of the install id (never the id, never an address).
#[allow(clippy::too_many_arguments)]
pub fn record_json(ts: f64, machine_hash: &str, app: &str, os: &str, card: &str, vendor: &str, driver: &str, class: &str, plan: PlanKind, rows: &[Row], chosen: Option<&Row>, before: Option<&Row>, floor: bool) -> serde_json::Value {
serde_json::json!({
"ts": ts.round(),
"machine": machine_hash,
"app": app,
"os": os,
"card": card.replace(' ', "_"),
"vendor": vendor,
"driver": driver,
"driver_major": driver_major(driver),
"class": if class.is_empty() { "v2" } else { class },
"key": prior_key(card, driver, class),
"plan": plan.name(),
"steps": rows.iter().map(|r| r.json()).collect::<Vec<_>>(),
"chosen": chosen.map(|r| r.json()),
"before": before.map(|r| r.json()),
"eff": chosen.map(|r| r4(r.eff)).unwrap_or(0.0),
"mhs": chosen.map(|r| r2(r.mhs)).unwrap_or(0.0),
"watts": chosen.map(|r| r1(r.watts)).unwrap_or(0.0),
"floor": floor,
"note": if floor { "floor, not optimum" } else { "" },
})
}
// ---- the state machine -----------------------------------------------------------------------------------------
pub use crate::sweep::Timing;
/// Where the tune stands, for the card row.
#[derive(Clone, Debug, PartialEq)]
pub enum Phase {
/// the step's setting was (or is about to be) requested; waiting for the readback or the acknowledgement
Applying { since: Instant, sent: bool },
Settling { since: Instant },
Holding { since: Instant },
/// the chosen point was requested; waiting for the readback
Finishing { since: Instant, sent: bool },
Done,
}
/// What the engine must do after a tick.
#[derive(Clone, Debug, PartialEq)]
pub enum Out {
/// set this point on the card (the watts are the limit for its power percent)
Apply(Step),
/// a step finished: log its line (its index is the row count before it)
Row(Row),
/// the plan finished and the chosen point is in force
Finished(Row),
/// the plan failed; the engine restores the point from before
Failed(String),
}
/// What the card reports now, for the readback.
#[derive(Clone, Copy, Debug, Default)]
pub struct Readback {
/// the power limit in force, watts (0 = unknown)
pub limit_w: f64,
/// the set command for the current request was acknowledged (nvidia-smi answered, or the helper logged it)
pub acked: bool,
}
pub struct Run {
pub card: usize,
pub device: String,
pub label: String,
pub plan: Plan,
pub current: Option<Step>,
pub phase: Phase,
pub rows: Vec<Row>,
pub chosen: Option<Row>,
pub before: Point,
pub before_w: f64,
pub started: Instant,
pub forced: bool,
pub timing: Timing,
/// the request number of the setting in flight (the engine acknowledges it by number)
pub seq: u64,
samples: Samples,
}
impl Run {
#[allow(clippy::too_many_arguments)]
pub fn new(card: usize, device: &str, label: &str, plan: Plan, before_w: f64, forced: bool, timing: Timing, now: Instant) -> Run {
let before = plan.before;
let current = plan.next(&[]);
Run { card, device: device.into(), label: label.into(), plan, current, phase: Phase::Applying { since: now, sent: false }, rows: Vec::new(), chosen: None, before, before_w, started: now, forced, timing, seq: 0, samples: Samples::default() }
}
/// A worker STATUS interval rate (MH/s); counted while holding.
pub fn sample_rate(&mut self, mhs: f64) {
if matches!(self.phase, Phase::Holding { .. }) && mhs > 0.0 {
self.samples.rates.push(mhs);
}
}
/// A telemetry reading (draw, core clock, memory clock, GPU temperature); counted while holding.
pub fn sample_telemetry(&mut self, draw_w: f64, gclk: f64, mclk: f64, temp_c: f64) {
if !matches!(self.phase, Phase::Holding { .. }) {
return;
}
if draw_w > 0.0 {
self.samples.draws.push(draw_w);
}
if gclk > 0.0 {
self.samples.gclks.push(gclk);
}
if mclk > 0.0 {
self.samples.mclks.push(mclk);
}
if temp_c > self.samples.tmax {
self.samples.tmax = temp_c;
}
}
/// A rejected or mismatched hash while this step holds: the step is marked and cannot win.
pub fn sample_fault(&mut self) {
if matches!(self.phase, Phase::Holding { .. } | Phase::Settling { .. }) {
self.samples.faults += 1;
}
}
/// The engine's readback said the setting in force is not the one requested (checked during the hold).
pub fn mark_unapplied(&mut self) {
if matches!(self.phase, Phase::Holding { .. }) {
self.samples.unapplied = true;
}
}
pub fn baseline_mclk(&self) -> f64 {
self.rows.first().map(|r| r.mclk).unwrap_or(0.0)
}
/// Seconds left in the plan from here: the rest of this step plus (settle + hold + a 10 s apply) per step to come.
pub fn eta_s(&self, now: Instant) -> i64 {
let per = (self.timing.settle + self.timing.hold).as_secs() as i64 + 10;
let this = match &self.phase {
Phase::Applying { .. } => per,
Phase::Settling { since } => per - 10 - now.duration_since(*since).as_secs().min(per as u64) as i64,
Phase::Holding { since } => (self.timing.hold.as_secs() as i64 - now.duration_since(*since).as_secs() as i64).max(0),
Phase::Finishing { .. } | Phase::Done => 0,
};
let left = self.plan.len().saturating_sub(self.rows.len() + 1) as i64;
this.max(0) + left * per
}
/// The phase in words for the card row.
pub fn words(&self, now: Instant) -> String {
let what = |p: &Point| {
let mut w = if p.clock_mhz > 0 { format!("{} MHz · {}%", p.clock_mhz, p.power_pct) } else { format!("{}%", p.power_pct) };
if p.mem_mhz > 0 {
w.push_str(&format!(" · mem {}", p.mem_mhz));
}
w
};
let cur = self.current.as_ref().map(|s| what(&s.point)).unwrap_or_default();
let n = self.rows.len() + 1;
let of = self.plan.len();
match &self.phase {
Phase::Applying { .. } => format!("tuning: setting {cur} (step {n} of {of})"),
Phase::Settling { since } => format!("tuning: {cur} settling · {} s (step {n} of {of})", self.timing.settle.as_secs().saturating_sub(now.duration_since(*since).as_secs())),
Phase::Holding { since } => format!("tuning: holding {cur} · {} s (step {n} of {of})", self.timing.hold.as_secs().saturating_sub(now.duration_since(*since).as_secs())),
Phase::Finishing { .. } => format!("tuning: setting the best point ({})", self.chosen.as_ref().map(|r| what(&r.point)).unwrap_or_default()),
Phase::Done => "tuning: done".into(),
}
}
/// A step is applied when its power limit reads back (within 1.5 W) and, when it caps the clock or changes
/// nothing in watts, the set command was acknowledged.
fn applied(step: &Step, rb: Readback, before_w: f64) -> bool {
let power_ok = rb.limit_w <= 0.0 || (rb.limit_w - step.watts).abs() < 1.5;
let power_changed = (step.watts - before_w).abs() >= 1.5;
if step.point.clock_mhz > 0 || step.point.mem_mhz > 0 || !power_changed {
rb.acked && power_ok
} else if rb.limit_w <= 0.0 {
// a card with no limit readback at all (AMD through ADLX reports an offset, never watts; run 6 on
// 6 October 2026 aborted the 9070 XT's ladder at its first step on "card reports 0 W"): the
// acknowledgement is the proof
rb.acked
} else {
power_ok
}
}
/// Drives the state machine. `rb` is what the card reports now.
pub fn tick(&mut self, now: Instant, rb: Readback) -> Vec<Out> {
let mut out = Vec::new();
let settle = self.timing.settle;
let hold = self.timing.hold;
let apply = self.timing.apply;
match self.phase.clone() {
Phase::Applying { since, sent } => {
let Some(step) = self.current.clone() else {
self.phase = Phase::Done;
out.push(Out::Failed("no steps".into()));
return out;
};
if !sent {
self.phase = Phase::Applying { since: now, sent: true };
self.seq += 1;
out.push(Out::Apply(step));
} else if Run::applied(&step, rb, self.last_watts()) {
self.phase = Phase::Settling { since: now };
} else if now.duration_since(since) > apply {
if step.kind == Kind::Baseline {
// the baseline is the card as it runs: nothing to apply, nothing to wait for
self.phase = Phase::Settling { since: now };
} else {
self.phase = Phase::Done;
out.push(Out::Failed(format!("the setting ({} MHz, {}% = {:.0} W) did not take within {} s (card reports {:.0} W, acknowledged {})", step.point.clock_mhz, step.point.power_pct, step.watts, apply.as_secs(), rb.limit_w, rb.acked)));
}
}
}
Phase::Settling { since } => {
if now.duration_since(since) >= settle {
let faults = self.samples.faults;
self.samples = Samples { faults, ..Samples::default() };
self.phase = Phase::Holding { since: now };
}
}
Phase::Holding { since } => {
if now.duration_since(since) >= hold {
let step = self.current.clone().expect("a step while holding");
let row = Row::from_samples(&step, &self.samples, self.baseline_mclk());
self.rows.push(row.clone());
out.push(Out::Row(row));
self.current = self.plan.next(&self.rows);
match self.current.clone() {
Some(next) => {
self.phase = Phase::Applying { since: now, sent: true };
self.seq += 1;
out.push(Out::Apply(next));
}
None => match self.decide() {
Some(best) => {
self.chosen = Some(best.clone());
self.phase = Phase::Finishing { since: now, sent: true };
self.seq += 1;
out.push(Out::Apply(Step { point: best.point, watts: best.limit, kind: Kind::Confirm }));
}
None => {
self.phase = Phase::Done;
out.push(Out::Failed("no step had readings (no draw from the telemetry or no STATUS line from the worker)".into()));
}
},
}
}
}
Phase::Finishing { since, sent } => {
let best = self.chosen.clone().expect("a chosen row while finishing");
let step = Step { point: best.point, watts: best.limit, kind: Kind::Confirm };
if !sent {
self.phase = Phase::Finishing { since: now, sent: true };
self.seq += 1;
out.push(Out::Apply(step));
} else if Run::applied(&step, rb, self.last_watts()) || self.plan.kind == PlanKind::Baseline {
self.phase = Phase::Done;
out.push(Out::Finished(best));
} else if now.duration_since(since) > apply {
self.phase = Phase::Done;
out.push(Out::Failed(format!("the chosen point ({} MHz, {:.0} W) did not take within {} s", best.point.clock_mhz, best.limit, apply.as_secs())));
}
}
Phase::Done => {}
}
out
}
/// The limit the previous step left (the before limit for the first).
fn last_watts(&self) -> f64 {
self.rows.last().map(|r| r.limit).unwrap_or(self.before_w)
}
/// What the plan concludes: the full plan's choice; the confirm plan keeps its prior unless the neighbour beat
/// it (then the prior still holds the card and the engine schedules the full plan); the baseline is itself.
fn decide(&self) -> Option<Row> {
match self.plan.kind {
PlanKind::Baseline => self.rows.first().cloned().filter(|r| r.usable()),
PlanKind::Confirm => match confirm_verdict(&self.rows, self.plan.tolerance_pct) {
Verdict::Keep(r) => Some(r),
Verdict::FullDue { prior, .. } if prior.usable() => Some(prior),
Verdict::FullDue { better, .. } => Some(better),
Verdict::NoReadings => None,
},
PlanKind::Full => choose(&self.rows, self.plan.tolerance_pct),
PlanKind::Climb => {
if self.plan.climb.as_ref().map(|c| c.goal) == Some(Goal::MaxRate) {
self.rows.iter().filter(|r| r.usable()).max_by(|a, b| a.mhs.partial_cmp(&b.mhs).unwrap_or(std::cmp::Ordering::Equal)).cloned()
} else {
choose(&self.rows, self.plan.tolerance_pct)
}
}
}
}
/// After a confirm plan: did the neighbour beat the prior (the full plan is due)?
pub fn full_due(&self) -> bool {
self.plan.kind == PlanKind::Confirm && matches!(confirm_verdict(&self.rows, self.plan.tolerance_pct), Verdict::FullDue { .. })
}
}
/// The card row's line once a card is tuned: "Tuned: 122.3 MH/s at 290 W (0.422 MH/W)".
pub fn tuned_line(mhs: f64, watts: f64, eff: f64) -> String {
format!("Tuned: {mhs:.1} MH/s at {watts:.0} W ({eff:.3} MH/W)")
}
/// The row's line for a baseline (measure-only) result: the card was measured as it runs, nothing was set.
/// (6 October 2026: PC 1's rows read "Tuned: 114.2 MH/s at 305 W" for a baseline, which is not a tune.)
pub fn measured_line(mhs: f64, watts: f64, eff: f64) -> String {
format!("Measured: {mhs:.1} MH/s at {watts:.0} W ({eff:.3} MH/W)")
}
/// The row's line for a plan's result.
pub fn result_line(kind: PlanKind, mhs: f64, watts: f64, eff: f64) -> String {
if kind == PlanKind::Baseline {
measured_line(mhs, watts, eff)
} else {
tuned_line(mhs, watts, eff)
}
}
/// Why a card cannot be tuned beyond measuring, or None when both knobs are available.
pub fn control_reason(vendor: &str, limits: &Limits, device: &str, power_control: bool, amd_helper: bool) -> Option<String> {
match vendor {
"nvidia" if limits.power_default_w <= 0.0 || device.is_empty() => Some("measure only: nvidia-smi did not report this card's limits".into()),
"nvidia" if !power_control => Some("measure only until Power control is on in Settings (Windows asks for administrator rights once)".into()),
"nvidia" => None,
"amd" if !amd_helper => Some("measure only: igneum-gpu-telemetry gave no tune line for this card (not next to the app, an older helper, or no ADLX)".into()),
"amd" if cfg!(target_os = "linux") => Some("measure only on Linux: the clock and power limits under /sys need root".into()),
"amd" => None,
"apple" => Some("measure only on Apple silicon: the system sets the clocks and the power; no control exposed".into()),
_ => Some("measure only: no power or clock control for this card".into()),
}
}
#[cfg(test)]
mod tests {
use super::*;
/// PC 1's RTX 5090 (nvidia-smi, 4 and 5 October 2026): default 575 W, min 400 W, max 600 W; clocks.max.gr is
/// read at the first tune (3,090 MHz is the shape used here, not a measurement).
fn l5090() -> Limits {
Limits { power_default_w: 575.0, power_min_w: 400.0, power_max_w: 600.0, clock_max_mhz: 3090, clock_min_mhz: 0, mem_default_mhz: 13801, mem_max_mhz: 14001 }
}
fn row_at(p: Point, watts: f64, mhs: f64) -> Row {
Row { point: p, limit: watts, watts, mhs, eff: if watts > 0.0 { mhs / watts } else { 0.0 }, draws: 12, rates: 6, gclk: 0.0, mclk: 0.0, tmax: 60.0, faults: 0, mark: Some(Mark::Ok) }
}
#[test]
fn the_full_plan_is_the_power_ladder_then_the_clock_ladder_at_the_chosen_power() {
let plan = Plan::full(&l5090(), Point { clock_mhz: 0, power_pct: 80, mem_mhz: 0 }, 1.0);
assert_eq!(plan.len(), 5 + 6, "five power steps (60% and 50% clamp to 400 W; one kept) and six clock steps (90% down to 45%)");
let first = plan.next(&[]).unwrap();
assert_eq!((first.point, first.watts, first.kind), (Point { clock_mhz: 0, power_pct: 100, mem_mhz: 0 }, 575.0, Kind::Power));
// the power ladder: 575, 518, 460, 403, 400
let mut rows = Vec::new();
let mut watts_seen = Vec::new();
for i in 0..5 {
let s = plan.next(&rows).unwrap();
assert_eq!(s.kind, Kind::Power, "step {i}");
watts_seen.push(s.watts);
// the draw never reaches the cap (PC 1's shape: 290 W under every limit); the rate is flat
rows.push(row_at(s.point, 290.0 + i as f64 * 0.1, 124.0));
}
assert_eq!(watts_seen, vec![575.0, 518.0, 460.0, 403.0, 400.0]);
// the clock ladder rides the chosen power point: a flat ladder ties on efficiency and rate, the lowest draw wins (100%)
let s = plan.next(&rows).unwrap();
assert_eq!(s.kind, Kind::Clock);
assert_eq!(s.point, Point { clock_mhz: 2781, power_pct: 100, mem_mhz: 0 });
rows.push(row_at(s.point, 250.0, 123.8));
let s = plan.next(&rows).unwrap();
assert_eq!(s.point.clock_mhz, 2472);
rows.push(row_at(s.point, 220.0, 123.5));
rows.push(row_at(plan.next(&rows).unwrap().point, 200.0, 118.0));
let s = plan.next(&rows).unwrap();
assert_eq!(s.point.clock_mhz, 1854, "60% of 3,090");
rows.push(row_at(s.point, 180.0, 100.0));
let s = plan.next(&rows).unwrap();
assert_eq!(s.point.clock_mhz, 1545, "50%");
rows.push(row_at(s.point, 170.0, 90.0));
let s = plan.next(&rows).unwrap();
assert_eq!(s.point.clock_mhz, 1390, "45% of 3,090 is the floor (6 October 2026)");
rows.push(row_at(s.point, 160.0, 80.0));
assert_eq!(plan.next(&rows), None);
// the choice: 2,472 MHz keeps 99.6% of the top rate at 220 W = 0.561 MH/W; 2,163 MHz (118 MH/s) is outside the 1% tolerance
let best = choose(&rows, 1.0).unwrap();
assert_eq!(best.point, Point { clock_mhz: 2472, power_pct: 100, mem_mhz: 0 });
// a wider tolerance lets the 2,163 MHz step (0.590 MH/W, 4.8% slower) win
assert_eq!(choose(&rows, 5.0).unwrap().point.clock_mhz, 2163);
// no power limits, clocks only; no clocks, power only; nothing, empty
assert_eq!(Plan::full(&Limits { clock_max_mhz: 2000, ..Default::default() }, Point::default(), 1.0).len(), 6);
assert_eq!(Plan::full(&Limits { power_default_w: 300.0, ..Default::default() }, Point::default(), 1.0).len(), 6);
assert!(Plan::full(&Limits::default(), Point::default(), 1.0).is_empty());
}
#[test]
fn a_power_step_on_a_card_without_a_limit_readback_is_applied_on_the_acknowledgement() {
// AMD through ADLX: the limit reads back as an offset, never watts (run 6, 6 October 2026)
let step = Step { point: Point { clock_mhz: 0, power_pct: 90, mem_mhz: 0 }, watts: 90.0, kind: Kind::Power };
assert!(Run::applied(&step, Readback { limit_w: 0.0, acked: true }, 100.0));
assert!(!Run::applied(&step, Readback { limit_w: 0.0, acked: false }, 100.0));
// NVIDIA: the watts must read back
assert!(!Run::applied(&step, Readback { limit_w: 100.0, acked: true }, 100.0));
assert!(Run::applied(&step, Readback { limit_w: 90.0, acked: false }, 100.0));
}
#[test]
fn a_baseline_result_reads_measured_and_a_tune_reads_tuned() {
assert_eq!(result_line(PlanKind::Baseline, 114.2, 305.0, 0.3744), "Measured: 114.2 MH/s at 305 W (0.374 MH/W)");
assert_eq!(result_line(PlanKind::Full, 127.71, 226.8, 0.5632), "Tuned: 127.7 MH/s at 227 W (0.563 MH/W)");
assert_eq!(result_line(PlanKind::Confirm, 127.71, 226.8, 0.5632), "Tuned: 127.7 MH/s at 227 W (0.563 MH/W)");
}
#[test]
fn limits_never_exceed_the_vendor_or_undercut_the_floor() {
let l = l5090();
assert_eq!(l.clock_floor(), 1390);
assert_eq!(l.clamp_clock(1000), 1390);
assert_eq!(l.clamp_clock(5000), 3090);
assert_eq!(l.clamp_clock(0), 0, "unlocked stays unlocked");
assert_eq!(Limits { clock_max_mhz: 3000, clock_min_mhz: 2100, ..Default::default() }.clamp_clock(1500), 2100, "the vendor's floor wins over the 45% rule");
assert_eq!(l.watts_for(100), 575.0);
assert_eq!(l.watts_for(50), 400.0);
assert_eq!(Limits { power_default_w: 300.0, power_max_w: 250.0, ..Default::default() }.watts_for(100), 250.0);
}
#[test]
fn the_choice_keeps_the_best_mh_per_watt_within_the_rate_tolerance() {
let p = |c: u32, pct: u32| Point { clock_mhz: c, power_pct: pct, mem_mhz: 0 };
// a card that gets more efficient as the cap drops until it collapses: 70% wins (0.280 MH/W), 60% is 35% slower
let rows = vec![row_at(p(0, 100), 560.0, 124.0), row_at(p(0, 90), 510.0, 123.5), row_at(p(0, 80), 455.0, 123.2), row_at(p(0, 70), 400.0, 122.9), row_at(p(0, 60), 345.0, 80.0)];
assert_eq!(choose(&rows, 1.0).unwrap().point, p(0, 70));
// the tolerance: a 2% slower point with better MH/W loses at 1% and wins at 3%
let rows = vec![row_at(p(0, 100), 300.0, 100.0), row_at(p(2000, 100), 240.0, 98.0)];
assert_eq!(choose(&rows, 1.0).unwrap().point, p(0, 100));
assert_eq!(choose(&rows, 3.0).unwrap().point, p(2000, 100));
// within 1% on efficiency the higher rate wins; within 1% on both the lower draw
let rows = vec![row_at(p(0, 100), 500.0, 125.0), row_at(p(0, 80), 400.5, 100.3)];
assert_eq!(choose(&rows, 25.0).unwrap().point, p(0, 100));
let flat = vec![row_at(p(0, 100), 290.0, 124.0), row_at(p(0, 90), 290.5, 123.5), row_at(p(0, 80), 289.8, 124.2), row_at(p(0, 70), 290.2, 123.9)];
assert_eq!(choose(&flat, 1.0).unwrap().point, p(0, 80), "289.8 W is the lowest draw");
// marked rows never win, whatever their numbers
let mut faulted = row_at(p(1800, 100), 150.0, 123.0);
faulted.mark = Some(Mark::Faulted);
let mut hot = row_at(p(1900, 100), 160.0, 123.0);
hot.mark = Some(Mark::Hot);
let mut mem = row_at(p(2000, 100), 170.0, 123.0);
mem.mark = Some(Mark::MemoryClock);
assert_eq!(choose(&[faulted.clone(), hot, mem, row_at(p(0, 100), 290.0, 124.0)], 1.0).unwrap().point, p(0, 100));
assert!(choose(&[faulted], 1.0).is_none());
assert!(choose(&[], 1.0).is_none());
}
#[test]
fn the_guards_mark_a_step_so_it_cannot_win() {
let step = Step { point: Point { clock_mhz: 2000, power_pct: 100, mem_mhz: 0 }, watts: 575.0, kind: Kind::Clock };
let good = Samples { draws: vec![250.0, 251.0, 249.0], rates: vec![123.0], gclks: vec![1998.0], mclks: vec![2505.0], tmax: 70.0, faults: 0, unapplied: false };
assert_eq!(Row::from_samples(&step, &good, 2505.0).mark, Some(Mark::Ok));
// one rejected or mismatched hash: faulted
assert_eq!(Row::from_samples(&step, &Samples { faults: 1, ..good.clone() }, 2505.0).mark, Some(Mark::Faulted));
// the GPU at 85 C: hot
assert_eq!(Row::from_samples(&step, &Samples { tmax: 85.0, ..good.clone() }, 2505.0).mark, Some(Mark::Hot));
// the memory clock dragged under 95% of the baseline's: marked (the AMD gmax path on RDNA 4 cannot hold it)
assert_eq!(Row::from_samples(&step, &Samples { mclks: vec![2300.0], ..good.clone() }, 2505.0).mark, Some(Mark::MemoryClock));
assert_eq!(Row::from_samples(&step, &Samples { mclks: vec![2400.0], ..good.clone() }, 2505.0).mark, Some(Mark::Ok), "2,400 is 95.8%");
// the readback disagreed: unapplied
assert_eq!(Row::from_samples(&step, &Samples { unapplied: true, ..good.clone() }, 2505.0).mark, Some(Mark::Unapplied));
// under 3 draws or no rate: no readings, and the line says so
let r = Row::from_samples(&step, &Samples { draws: vec![250.0], ..good.clone() }, 0.0);
assert_eq!(r.mark, Some(Mark::NoReadings));
assert!(!r.usable());
assert!(r.line("c", 3).contains("mark=no_readings"), "{}", r.line("c", 3));
let r = Row::from_samples(&step, &good, 0.0);
assert_eq!(r.line("nvidia-ae432dc7-1", 7), "TUNE card=nvidia-ae432dc7-1 step=7 clock=2000 cap=100 mem=0 limit=575 watts=250.0 mhs=123.00 eff=0.4920 gclk=1998 mclk=2505 tmax=70 mark=ok");
}
#[test]
fn a_fault_during_a_step_reverts_it_and_the_run_goes_on() {
let t0 = Instant::now();
let timing = Timing { settle: Duration::from_secs(2), hold: Duration::from_secs(4), apply: Duration::from_secs(30) };
let limits = Limits { power_default_w: 300.0, power_min_w: 150.0, power_max_w: 300.0, clock_max_mhz: 0, clock_min_mhz: 0, mem_default_mhz: 0, mem_max_mhz: 0 };
let plan = Plan::full(&limits, Point { clock_mhz: 0, power_pct: 80, mem_mhz: 0 }, 1.0);
let mut run = Run::new(0, "0", "c", plan, 240.0, false, timing, t0);
let mut t = t0;
let mut limit = 240.0;
let mut applied: Vec<f64> = Vec::new();
let mut rows = Vec::new();
let mut finished = None;
for _ in 0..600 {
for o in run.tick(t, Readback { limit_w: limit, acked: true }) {
match o {
Out::Apply(s) => {
applied.push(s.watts);
limit = s.watts;
}
Out::Row(r) => rows.push(r),
Out::Finished(r) => finished = Some(r),
Out::Failed(e) => panic!("{e}"),
}
}
if matches!(run.phase, Phase::Holding { .. }) {
run.sample_telemetry(limit.min(260.0), 2500.0, 1300.0, 60.0);
run.sample_telemetry(limit.min(260.0) + 1.0, 2500.0, 1300.0, 60.0);
run.sample_telemetry(limit.min(260.0) - 1.0, 2500.0, 1300.0, 60.0);
run.sample_rate(100.0);
// the 70% step (210 W) sees a mismatched hash
if (limit - 210.0).abs() < 0.5 {
run.sample_fault();
}
}
if run.phase == Phase::Done {
break;
}
t += Duration::from_secs(1);
}
assert_eq!(rows.len(), 6, "300, 270, 240, 210, 180, 150 W");
assert_eq!(rows[3].point.power_pct, 70);
assert_eq!(rows[3].mark, Some(Mark::Faulted), "the faulted step is marked");
assert!(rows[3].faults >= 1);
// 100, 90 and 80% read the same 260 W draw at the same rate; 60% (150 W, the clamp) too in this model
// because the draw model caps at 260 W: the lowest draw wins among the ties, never the faulted 70%
let f = finished.expect("finished");
assert_ne!(f.point.power_pct, 70);
assert_eq!(applied.len(), 7, "every step's cap, then the chosen one");
assert_eq!(*applied.last().unwrap(), f.limit);
}
#[test]
fn the_confirm_plan_checks_the_prior_and_its_neighbour() {
let l = l5090();
let prior = Point { clock_mhz: 2472, power_pct: 100, mem_mhz: 0 };
let plan = Plan::confirm(&l, prior, Point { clock_mhz: 0, power_pct: 80, mem_mhz: 0 }, 1.0);
assert_eq!(plan.len(), 2);
let a = plan.next(&[]).unwrap();
assert_eq!((a.point, a.kind), (prior, Kind::Confirm));
let b = plan.next(&[row_at(prior, 220.0, 123.5)]).unwrap();
assert_eq!(b.point, Point { clock_mhz: 2781, power_pct: 100, mem_mhz: 0 }, "one clock step up");
// the neighbour within 1%: the prior stands
let rows = vec![row_at(prior, 220.0, 123.5), row_at(b.point, 250.0, 123.8)];
assert!(matches!(confirm_verdict(&rows, 1.0), Verdict::Keep(r) if r.point == prior));
// the neighbour 3% better on MH/W: the full plan is due
let rows = vec![row_at(prior, 220.0, 120.0), row_at(b.point, 212.0, 120.0)];
assert!(matches!(confirm_verdict(&rows, 1.0), Verdict::FullDue { .. }));
assert_eq!(confirm_verdict(&[], 1.0), Verdict::NoReadings);
// an unlocked prior at 100%: the neighbour is one power step down; at the top clock the neighbour is unlocked
let plan = Plan::confirm(&l, Point { clock_mhz: 0, power_pct: 100, mem_mhz: 0 }, Point::default(), 1.0);
assert_eq!(plan.next(&[row_at(Point { clock_mhz: 0, power_pct: 100, mem_mhz: 0 }, 1.0, 1.0)]).unwrap().point, Point { clock_mhz: 0, power_pct: 90, mem_mhz: 0 });
let plan = Plan::confirm(&l, Point { clock_mhz: 2900, power_pct: 100, mem_mhz: 0 }, Point::default(), 1.0);
assert_eq!(plan.next(&[row_at(Point { clock_mhz: 2900, power_pct: 100, mem_mhz: 0 }, 1.0, 1.0)]).unwrap().point.clock_mhz, 0);
// a prior outside the vendor's range is clamped, never applied as is
let plan = Plan::confirm(&l, Point { clock_mhz: 9000, power_pct: 30, mem_mhz: 0 }, Point::default(), 1.0);
assert_eq!(plan.next(&[]).unwrap().point, Point { clock_mhz: 3090, power_pct: 50, mem_mhz: 0 });
}
#[test]
fn a_baseline_plan_measures_the_card_as_it_runs() {
let t0 = Instant::now();
let timing = Timing { settle: Duration::from_secs(1), hold: Duration::from_secs(2), apply: Duration::from_secs(3) };
let before = Point { clock_mhz: 0, power_pct: 80, mem_mhz: 0 };
let mut run = Run::new(0, "0", "c", Plan::baseline(&l5090(), before, 1.0), 460.0, false, timing, t0);
// nothing acknowledges the request (Power control is off): the apply window passes and the hold starts anyway
let mut t = t0;
let mut finished = None;
for _ in 0..40 {
for o in run.tick(t, Readback { limit_w: 460.0, acked: false }) {
match o {
Out::Finished(r) => finished = Some(r),
Out::Failed(e) => panic!("{e}"),
_ => {}
}
}
if matches!(run.phase, Phase::Holding { .. }) {
for w in [290.0, 291.0, 289.0] {
run.sample_telemetry(w, 2700.0, 10500.0, 66.0);
}
run.sample_rate(122.3);
}
if run.phase == Phase::Done {
break;
}
t += Duration::from_secs(1);
}
let f = finished.expect("the baseline finished");
assert_eq!(f.point, before);
assert!((f.eff - 122.3 / 290.0).abs() < 1e-6);
assert_eq!(tuned_line(f.mhs, f.watts, f.eff), "Tuned: 122.3 MH/s at 290 W (0.422 MH/W)");
}
#[test]
fn the_record_and_the_prior_round_trip_through_the_manifest_shape() {
let chosen = row_at(Point { clock_mhz: 2472, power_pct: 100, mem_mhz: 0 }, 220.0, 123.5);
let rec = record_json(1_791_230_000.0, "8f3a2c1d", "0.3.10", "windows", "NVIDIA GeForce RTX 5090", "nvidia", "581.57", "l128w16", PlanKind::Full, &[chosen.clone()], Some(&chosen), None, false);
assert_eq!(rec["floor"], false);
assert_eq!(rec["card"], "NVIDIA_GeForce_RTX_5090");
assert_eq!(rec["key"], "NVIDIA_GeForce_RTX_5090|581|l128w16");
assert_eq!(rec["driver_major"], "581");
assert_eq!(rec["plan"], "full");
assert_eq!(rec["chosen"]["clock_mhz"], 2472);
assert_eq!(rec["eff"], 0.5614);
assert!(rec.get("address").is_none() && rec.get("host").is_none());
// the manifest's tuning section: the kernel-variant cards object stays, priors and ember sit beside it
let tuning: serde_json::Value = serde_json::json!({
"updated": "2026-10-05T22:00:00Z",
"cards": { "NVIDIA_GeForce_RTX_5090": { "variant": "u2-ldg", "race": true, "candidates": ["u2-ldg", "ldg", "base"] } },
"ember": { "enabled": true, "min_samples": 5, "rate_tolerance_pct": 1.0 },
"priors": { "NVIDIA_GeForce_RTX_5090|581|l128w16": { "clock_mhz": 2472, "power_pct": 100, "eff": 0.561, "mhs": 123.5, "watts": 220.0, "spread_pct": 2.1, "samples": 7 },
"AMD_Radeon_RX_9070_XT|32|l128w16": { "clock_mhz": 2600, "power_pct": 90, "eff": 0.1, "mhs": 17.7, "watts": 177.0, "spread_pct": 4.0, "samples": 3 } }
});
let s = settings_of(Some(&tuning));
assert_eq!(s, Settings { enabled: true, min_samples: 5, tolerance_pct: 1.0, period_s: PERIOD_S });
let p = prior_of(Some(&tuning), "NVIDIA_GeForce_RTX_5090|581|l128w16", s.min_samples).unwrap();
assert_eq!(p.point, Point { clock_mhz: 2472, power_pct: 100, mem_mhz: 0 });
assert_eq!(p.samples, 7);
assert!(prior_of(Some(&tuning), "AMD_Radeon_RX_9070_XT|32|l128w16", 5).is_none(), "3 samples are under the floor");
assert!(prior_of(Some(&tuning), "AMD_Radeon_RX_9070_XT|32|l128w16", 3).is_some());
assert!(prior_of(Some(&tuning), "nothing|0|v2", 5).is_none());
assert!(prior_of(None, "x", 5).is_none());
// the kill switch, and the defaults without a section
let off: serde_json::Value = serde_json::json!({ "cards": {}, "ember": { "enabled": false } });
assert!(!settings_of(Some(&off)).enabled);
assert_eq!(settings_of(None), Settings::default());
assert_eq!(settings_of(Some(&serde_json::json!({ "cards": {}, "ember": { "rate_tolerance_pct": 99 } }))).tolerance_pct, 1.0, "out of range: the default");
assert_eq!(prior_key("AMD Radeon RX 9070 XT", "32.0.15801.1", ""), "AMD_Radeon_RX_9070_XT|32|v2");
assert_eq!(driver_major(""), "0");
assert_eq!(program_class(128, 16), "l128w16");
assert_eq!(program_class(0, 0), "v2");
}
/// Ember 2: a synthetic memory-latency-bound card. Rate rises 1% per 100 MHz of memory above the default and
/// falls only 0.3% per 100 MHz of core below the maximum; draw falls 8 W per 100 MHz of core and rises 2 W per
/// 100 MHz of memory. The climb must walk memory up and core down and converge in under five probes.
fn synthetic(p: Point) -> Row {
let mem = if p.mem_mhz == 0 { 13801.0 } else { p.mem_mhz as f64 };
let core = if p.clock_mhz == 0 { 3090.0 } else { p.clock_mhz as f64 };
let mhs = 127.0 * (1.0 + 0.015 * (mem - 13801.0) / 100.0) * (1.0 - 0.003 * (3090.0 - core) / 100.0);
let watts = 310.0 - 8.0 * (3090.0 - core) / 100.0 + 2.0 * (mem - 13801.0) / 100.0;
Row { point: p, limit: 575.0, watts, mhs, eff: mhs / watts, draws: 12, rates: 6, gclk: core, mclk: mem, tmax: 65.0, faults: 0, mark: Some(Mark::Ok) }
}
#[test]
fn the_climb_walks_memory_up_and_core_down_and_converges_in_five_probes() {
// a 2 GHz memory range above the default (the headroom a 5090 has in practice; PC 1's driver reports 14,001)
let l = Limits { mem_max_mhz: 15801, ..l5090() };
let plan = Plan::climb(&l, Point { clock_mhz: 0, power_pct: 100, mem_mhz: 0 }, Goal::Balanced, 1.0);
assert_eq!(plan.len(), 5);
let mut rows = Vec::new();
while let Some(step) = plan.next(&rows) {
assert_eq!(step.kind, Kind::Climb);
rows.push(synthetic(step.point));
}
assert_eq!(rows.len(), 5, "the budget");
assert_eq!(rows[0].point, Point { clock_mhz: 0, power_pct: 100, mem_mhz: 0 }, "it starts at the start point");
// every probe moved one knob the right way: memory never down, core never up, and both inside the vendor's range
for w in rows.windows(2) {
let (a, b) = (w[0].point, w[1].point);
assert!(b.mem_mhz >= a.mem_mhz || b.mem_mhz == 0, "{a:?} -> {b:?}");
assert!(b.mem_mhz <= 15801 && (b.clock_mhz == 0 || b.clock_mhz >= 1854), "{b:?}");
}
let best = choose(&rows, plan.tolerance_pct).unwrap();
assert!(best.eff > rows[0].eff, "the chosen point beats the start: {:.4} > {:.4}", best.eff, rows[0].eff);
assert!(best.point.mem_mhz > 13801 || (best.point.clock_mhz > 0 && best.point.clock_mhz < 3090), "it moved a knob");
}
#[test]
fn a_refused_probe_backs_that_knob_off_for_good() {
let l = Limits { mem_max_mhz: 15801, ..l5090() };
let plan = Plan::climb(&l, Point { clock_mhz: 0, power_pct: 100, mem_mhz: 0 }, Goal::Balanced, 1.0);
let mut rows = vec![synthetic(plan.next(&[]).unwrap().point)];
// the first probe (memory up) draws a mismatched hash: refused
let probe = plan.next(&rows).unwrap();
assert!(probe.point.mem_mhz > 13801, "memory first: {:?}", probe.point);
let mut bad = synthetic(probe.point);
bad.faults = 1;
bad.mark = Some(Mark::Faulted);
rows.push(bad);
// from here every probe leaves the memory clock alone
while let Some(step) = plan.next(&rows) {
assert_eq!(step.point.mem_mhz, 0, "memory backed off: {:?}", step.point);
rows.push(synthetic(step.point));
}
assert!(rows.len() >= 3 && rows.len() <= 5);
assert!(choose(&rows, 1.0).unwrap().point.mem_mhz == 0);
}
#[test]
fn each_goal_picks_its_point() {
let l = l5090();
let p = |c: u32, m: u32| Point { clock_mhz: c, power_pct: 100, mem_mhz: m };
// three points: the fastest (slightly less efficient), the most efficient (4% slower), and the start
let rows = vec![synthetic(p(0, 0)), synthetic(p(0, 14001)), synthetic(p(1854, 0))];
let eff_plan = Plan::climb(&l, p(0, 0), Goal::Efficiency, 1.0);
let bal_plan = Plan::climb(&l, p(0, 0), Goal::Balanced, 1.0);
let rate_plan = Plan::climb(&l, p(0, 0), Goal::MaxRate, 1.0);
assert_eq!(eff_plan.tolerance_pct, 10.0);
assert_eq!(bal_plan.tolerance_pct, 1.0);
assert_eq!(rate_plan.tolerance_pct, 0.0);
let by = |plan: &Plan| rows.iter().filter(|r| r.usable() && r.mhs >= rows.iter().map(|x| x.mhs).fold(0.0, f64::max) * (1.0 - plan.tolerance_pct / 100.0)).max_by(|a, b| plan.score(a).partial_cmp(&plan.score(b)).unwrap()).unwrap().point;
assert_eq!(by(&rate_plan), p(0, 14001), "maximum rate takes the fastest point");
assert_eq!(by(&eff_plan), p(1854, 0), "efficiency takes the most MH/W inside 10%");
assert_eq!(by(&bal_plan), p(0, 14001), "balanced keeps within 1% of the top rate");
assert_eq!(Goal::parse("efficiency"), Goal::Efficiency);
assert_eq!(Goal::parse("rate"), Goal::MaxRate);
assert_eq!(Goal::parse("anything"), Goal::Balanced);
assert!((pounds_per_day(310.0, 28.5) - 2.1204).abs() < 1e-3, "310 W a day at 28.5 p/kWh = £2.12");
}
#[test]
fn control_reasons_per_vendor() {
let l = l5090();
assert!(control_reason("nvidia", &l, "0", true, false).is_none());
assert!(control_reason("nvidia", &l, "0", false, false).unwrap().contains("Power control"));
assert!(control_reason("nvidia", &Limits::default(), "0", true, false).unwrap().contains("limits"));
assert!(control_reason("apple", &Limits::default(), "", true, false).unwrap().contains("Apple silicon"));
assert!(control_reason("amd", &Limits::default(), "1", false, false).unwrap().contains("igneum-gpu-telemetry"));
if cfg!(target_os = "linux") {
assert!(control_reason("amd", &Limits::default(), "1", false, true).unwrap().contains("root"));
} else {
assert!(control_reason("amd", &Limits::default(), "1", false, true).is_none());
}
}
}