From c5e5de67525571ab00732d193e8b029868aa31eb Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Tue, 6 Oct 2026 15:43:53 +0000 Subject: [PATCH] Run 5 (15:28 to 15:39Z): the 5090's five power rows in the bench log (the cap does not bind: 0.41 MH/W flat at 310 W); the playbook's watchdog killed the live three-card tune at the first clock step, so: idle = three consecutive status lines reading 0.00 MH/s and 300 s, never a missing match; the after snapshot and the engine-log dump on every exit; an elevated tune engine registers the Igneum Power Helper itself before its first step (the --sweep engine skipped the cap path where the registration lived); Ember 2 groundwork: the memory-clock knob in Point and Limits, the goal, the hill-climb plan (not yet wired to the engine) Co-Authored-By: Claude Fable 5.1 --- app/igneum-app/src/ember.rs | 223 ++++++++++++++++++++++++----- app/igneum-app/src/engine.rs | 43 ++++-- app/igneum-app/src/state.rs | 1 + docs/bench-log.md | 12 ++ relay/playbooks/ember-tune-pc1.ps1 | 39 ++--- 5 files changed, 258 insertions(+), 60 deletions(-) diff --git a/app/igneum-app/src/ember.rs b/app/igneum-app/src/ember.rs index dcd96daf..e5a4584b 100644 --- a/app/igneum-app/src/ember.rs +++ b/app/igneum-app/src/ember.rs @@ -49,6 +49,10 @@ pub struct Limits { pub clock_max_mhz: u32, /// the lowest cap the vendor allows (the ADLX gmax_range floor); 0 = CLOCK_FLOOR_PCT of the maximum pub clock_min_mhz: u32, + /// Ember 2: the memory clock the card runs at by default and the vendor's maximum (nvidia-smi clocks.mem and + /// clocks.max.mem); 0 = no memory knob (AMD through ADLX on RDNA 4 exposes none) + pub mem_default_mhz: u32, + pub mem_max_mhz: u32, } impl Limits { @@ -66,6 +70,13 @@ impl Limits { } mhz.clamp(self.clock_floor(), self.clock_max_mhz) } + /// A memory clock inside the vendor's range; 0 stays 0 (the default). + pub fn clamp_mem(&self, mhz: u32) -> u32 { + if mhz == 0 || self.mem_max_mhz == 0 || self.mem_default_mhz == 0 { + return 0; + } + mhz.clamp(self.mem_default_mhz, self.mem_max_mhz) + } /// The watts a power percent asks for, inside the card's min and max, rounded to a watt. pub fn watts_for(&self, pct: u32) -> f64 { let mut w = self.power_default_w * pct as f64 / 100.0; @@ -79,13 +90,15 @@ impl Limits { } } -/// One setting of the two knobs. +/// One setting of the knobs (Ember 2 adds the memory clock: 0 = the driver's default). #[derive(Clone, Copy, Debug, Default, PartialEq, Eq, Hash)] pub struct Point { /// the core clock cap in MHz; 0 = unlocked pub clock_mhz: u32, /// the power limit, percent of the default pub power_pct: u32, + /// the memory clock in MHz (NVIDIA `-lmc`, locked to one value); 0 = the driver's default + pub mem_mhz: u32, } #[derive(Clone, Copy, Debug, PartialEq, Eq)] @@ -96,6 +109,8 @@ pub enum Kind { Clock, /// the fleet prior and one neighbour Confirm, + /// Ember 2: a hill-climb probe + Climb, } #[derive(Clone, Debug, PartialEq)] @@ -106,11 +121,53 @@ pub struct Step { pub kind: Kind, } +/// What the tune is for (Settings > Ember Tune > goal). The rate floor is the share of the best rate seen a point +/// must keep to win on MH per watt: efficiency keeps 90%, balanced 99% (the 1% rule of lever 3), maximum rate +/// takes the fastest point and uses MH per watt only to break ties. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum Goal { + Efficiency, + Balanced, + MaxRate, +} + +impl Goal { + pub fn parse(s: &str) -> Goal { + match s { + "efficiency" | "eff" => Goal::Efficiency, + "rate" | "max_rate" | "maximum" => Goal::MaxRate, + _ => Goal::Balanced, + } + } + pub fn name(&self) -> &'static str { + match self { + Goal::Efficiency => "efficiency", + Goal::Balanced => "balanced", + Goal::MaxRate => "rate", + } + } + /// The rate tolerance the choice rule uses, percent under the best rate. + pub fn tolerance_pct(&self, manifest_default: f64) -> f64 { + match self { + Goal::Efficiency => 10.0, + Goal::Balanced => manifest_default, + Goal::MaxRate => 0.0, + } + } +} + +/// Pounds a day for a draw at a price in pence per kWh: watts × 24 h / 1000 × price / 100. +pub fn pounds_per_day(watts: f64, pence_per_kwh: f64) -> f64 { + watts * 24.0 / 1000.0 * pence_per_kwh / 100.0 +} + #[derive(Clone, Copy, Debug, PartialEq, Eq)] pub enum PlanKind { Full, Confirm, Baseline, + /// Ember 2: the hill-climb over memory up and core down from the start point + Climb, } impl PlanKind { @@ -119,6 +176,7 @@ impl PlanKind { PlanKind::Full => "full", PlanKind::Confirm => "confirm", PlanKind::Baseline => "baseline", + PlanKind::Climb => "climb", } } } @@ -134,9 +192,82 @@ pub struct Plan { power: Vec, clock_pcts: Vec, fixed: Vec, + /// Ember 2 (Climb): the start point, the step sizes and the step budget + climb: Option, +} + +/// The hill-climb's shape: from `start`, each probe moves the memory clock up by `mem_step` or the core clock down +/// by `core_step` (or both), keeps the move when the goal's score improves, else turns to the other knob; a +/// refused step (a fault on it) backs that knob off for good. At most `budget` steps including the start. +#[derive(Clone, Debug, PartialEq)] +pub struct Climb { + pub start: Point, + pub mem_step: u32, + pub core_step: u32, + pub budget: usize, + pub goal: Goal, } impl Plan { + /// Ember 2: the hill-climb. The start is the fleet prior (or the card's current point); a probe step is 5% of + /// the memory range above the default (0 when the card has no memory knob) and 5% of the maximum core clock; + /// five steps of 60 s converge in under 10 minutes. + pub fn climb(limits: &Limits, start: Point, goal: Goal, tolerance_pct: f64) -> Plan { + let mem_step = if limits.mem_max_mhz > limits.mem_default_mhz { ((limits.mem_max_mhz - limits.mem_default_mhz) / 20).max(25) } else { 0 }; + let core_step = if limits.clock_max_mhz > 0 { (limits.clock_max_mhz / 20).max(25) } else { 0 }; + let start = Point { clock_mhz: limits.clamp_clock(start.clock_mhz), power_pct: start.power_pct.clamp(50, 100), mem_mhz: limits.clamp_mem(start.mem_mhz) }; + Plan { kind: PlanKind::Climb, limits: limits.clone(), before: start, tolerance_pct: goal.tolerance_pct(tolerance_pct), power: Vec::new(), clock_pcts: Vec::new(), fixed: Vec::new(), climb: Some(Climb { start, mem_step, core_step, budget: 5, goal }) } + } + + /// The goal's score of a row: MH per watt for efficiency and balanced, the rate for maximum rate. + pub fn score(&self, r: &Row) -> f64 { + match self.climb.as_ref().map(|c| c.goal) { + Some(Goal::MaxRate) => r.mhs, + _ => r.eff, + } + } + + /// The climb's next probe from the rows so far: None when the budget is spent or no move is left. + fn climb_next(&self, rows: &[Row]) -> Option { + let c = self.climb.as_ref()?; + let step = |p: Point| Step { point: p, watts: self.limits.watts_for(p.power_pct), kind: Kind::Climb }; + if rows.is_empty() { + return Some(step(c.start)); + } + if rows.len() >= c.budget { + return None; + } + // the best usable row so far is the hill's top; a knob that produced a marked (refused) row is backed off + let best = rows.iter().filter(|r| r.usable()).max_by(|a, b| self.score(a).partial_cmp(&self.score(b)).unwrap_or(std::cmp::Ordering::Equal))?; + let refused_mem = rows.iter().any(|r| !r.usable() && r.point.mem_mhz > best.point.mem_mhz); + let refused_core = rows.iter().any(|r| !r.usable() && r.point.clock_mhz != 0 && (best.point.clock_mhz == 0 || r.point.clock_mhz < best.point.clock_mhz)); + let last = rows.last()?; + let mem_up = |p: Point| -> Option { + if c.mem_step == 0 || refused_mem { return None; } + let base = if p.mem_mhz == 0 { self.limits.mem_default_mhz } else { p.mem_mhz }; + let m = self.limits.clamp_mem(base + c.mem_step); + (m > 0 && m != p.mem_mhz && m != base).then_some(Point { mem_mhz: m, ..p }) + }; + let core_down = |p: Point| -> Option { + if c.core_step == 0 || refused_core { return None; } + let base = if p.clock_mhz == 0 { self.limits.clock_max_mhz } else { p.clock_mhz }; + let k = self.limits.clamp_clock(base.saturating_sub(c.core_step)); + (k > 0 && k != p.clock_mhz).then_some(Point { clock_mhz: k, ..p }) + }; + let tried = |p: Point| rows.iter().any(|r| r.point == p); + // the last move improved: keep going the same way from the top; else turn: memory first, then core, then both + let last_improved = last.usable() && last.point == best.point && rows.len() > 1; + let last_was_mem = rows.len() > 1 && last.point.mem_mhz != rows[rows.len() - 2].point.mem_mhz; + let candidates: Vec> = if last_improved && last_was_mem { + vec![mem_up(best.point), core_down(best.point)] + } else if last_improved { + vec![core_down(best.point), mem_up(best.point)] + } else { + vec![mem_up(best.point), core_down(best.point), mem_up(best.point).and_then(core_down)] + }; + candidates.into_iter().flatten().find(|p| !tried(*p)).map(step) + } + /// Power 100% to 50% at the unlocked clock (duplicate watts dropped, as the card's floor clamps them), then the /// clock ladder 90% to the floor of the maximum core clock at the chosen power. A card without a readable /// maximum clock gets the power ladder only; a card without a default limit gets the clock ladder only. @@ -148,41 +279,44 @@ impl Plan { if power.last().map(|s: &Step| (s.watts - w).abs() < 0.5).unwrap_or(false) { continue; } - power.push(Step { point: Point { clock_mhz: 0, power_pct: pct }, watts: w, kind: Kind::Power }); + power.push(Step { point: Point { clock_mhz: 0, power_pct: pct, mem_mhz: 0 }, watts: w, kind: Kind::Power }); } } let clock_pcts = if limits.clock_max_mhz > 0 { CLOCK_STEPS_PCT[1..].to_vec() } else { Vec::new() }; - Plan { kind: PlanKind::Full, limits: limits.clone(), before, tolerance_pct, power, clock_pcts, fixed: Vec::new() } + Plan { kind: PlanKind::Full, limits: limits.clone(), before, tolerance_pct, power, clock_pcts, fixed: Vec::new(), climb: None } } /// The prior's point, then one neighbour: the next clock step up when the prior caps the clock (is the cap /// costing rate?), else one power step down (is there efficiency left?). pub fn confirm(limits: &Limits, prior: Point, before: Point, tolerance_pct: f64) -> Plan { - let p = Point { clock_mhz: limits.clamp_clock(prior.clock_mhz), power_pct: prior.power_pct.clamp(50, 100) }; + let p = Point { clock_mhz: limits.clamp_clock(prior.clock_mhz), power_pct: prior.power_pct.clamp(50, 100), mem_mhz: limits.clamp_mem(prior.mem_mhz) }; let first = Step { point: p, watts: limits.watts_for(p.power_pct), kind: Kind::Confirm }; let neighbour = if p.clock_mhz > 0 && limits.clock_max_mhz > 0 { let up = p.clock_mhz + limits.clock_max_mhz / 10; let clock = if up >= limits.clock_max_mhz { 0 } else { limits.clamp_clock(up) }; - Point { clock_mhz: clock, power_pct: p.power_pct } + Point { clock_mhz: clock, power_pct: p.power_pct, mem_mhz: p.mem_mhz } } else { - Point { clock_mhz: p.clock_mhz, power_pct: (p.power_pct.saturating_sub(10)).max(50) } + Point { clock_mhz: p.clock_mhz, power_pct: (p.power_pct.saturating_sub(10)).max(50), mem_mhz: p.mem_mhz } }; let mut fixed = vec![first]; if neighbour != p { fixed.push(Step { point: neighbour, watts: limits.watts_for(neighbour.power_pct), kind: Kind::Confirm }); } - Plan { kind: PlanKind::Confirm, limits: limits.clone(), before, tolerance_pct, power: Vec::new(), clock_pcts: Vec::new(), fixed } + Plan { kind: PlanKind::Confirm, limits: limits.clone(), before, tolerance_pct, power: Vec::new(), clock_pcts: Vec::new(), fixed, climb: None } } /// One step at the card's current point: the before number, and all a measure-only card (Apple, or NVIDIA /// with Power control off) reports. pub fn baseline(limits: &Limits, before: Point, tolerance_pct: f64) -> Plan { let fixed = vec![Step { point: before, watts: limits.watts_for(before.power_pct), kind: Kind::Baseline }]; - Plan { kind: PlanKind::Baseline, limits: limits.clone(), before, tolerance_pct, power: Vec::new(), clock_pcts: Vec::new(), fixed } + Plan { kind: PlanKind::Baseline, limits: limits.clone(), before, tolerance_pct, power: Vec::new(), clock_pcts: Vec::new(), fixed, climb: None } } /// How many steps the plan has at most (the clock ladder counts whether or not it runs). pub fn len(&self) -> usize { + if let Some(c) = &self.climb { + return c.budget; + } self.fixed.len() + self.power.len() + self.clock_pcts.len() } pub fn is_empty(&self) -> bool { @@ -191,6 +325,9 @@ impl Plan { /// The next step after `rows` (one row per step done so far), or None when the plan is complete. pub fn next(&self, rows: &[Row]) -> Option { + if self.climb.is_some() { + return self.climb_next(rows); + } let i = rows.len(); if !self.fixed.is_empty() { return self.fixed.get(i).cloned(); @@ -207,7 +344,7 @@ impl Plan { if rows.last().map(|r| r.point.clock_mhz == clock).unwrap_or(false) { return None; } - Some(Step { point: Point { clock_mhz: clock, power_pct }, watts: self.limits.watts_for(power_pct), kind: Kind::Clock }) + Some(Step { point: Point { clock_mhz: clock, power_pct, mem_mhz: 0 }, watts: self.limits.watts_for(power_pct), kind: Kind::Clock }) } } @@ -305,9 +442,10 @@ impl Row { /// `TUNE card=