From 229ca3f091c3c4554c818849ef27d34e82e25d5e Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Tue, 6 Oct 2026 15:50:34 +0000 Subject: [PATCH] =?UTF-8?q?Ember=202:=20the=20climb's=20tests=20(a=20synth?= =?UTF-8?q?etic=20memory-bound=20card=20converges=20in=20five=20probes;=20?= =?UTF-8?q?a=20refused=20probe=20backs=20the=20knob=20off=20for=20good;=20?= =?UTF-8?q?each=20goal=20picks=20its=20point;=20the=20=C2=A3/day=20formula?= =?UTF-8?q?)=20and=20the=20fleet=20prior=20carrying=20the=20memory=20clock?= =?UTF-8?q?=20(climb=20records=20aggregate)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Co-Authored-By: Claude Fable 5.1 --- app/igneum-app/src/ember.rs | 78 +++++++++++++++++++++++++++++++++++++ relay/lib/ember.mjs | 6 ++- relay/test/ember.test.mjs | 11 +++++- 3 files changed, 92 insertions(+), 3 deletions(-) diff --git a/app/igneum-app/src/ember.rs b/app/igneum-app/src/ember.rs index e5a4584b1..9e9bf54cc 100644 --- a/app/igneum-app/src/ember.rs +++ b/app/igneum-app/src/ember.rs @@ -1180,6 +1180,84 @@ mod tests { assert_eq!(program_class(0, 0), "v2"); } + /// Ember 2: a synthetic memory-latency-bound card. Rate rises 1% per 100 MHz of memory above the default and + /// falls only 0.3% per 100 MHz of core below the maximum; draw falls 8 W per 100 MHz of core and rises 2 W per + /// 100 MHz of memory. The climb must walk memory up and core down and converge in under five probes. + fn synthetic(p: Point) -> Row { + let mem = if p.mem_mhz == 0 { 13801.0 } else { p.mem_mhz as f64 }; + let core = if p.clock_mhz == 0 { 3090.0 } else { p.clock_mhz as f64 }; + let mhs = 127.0 * (1.0 + 0.015 * (mem - 13801.0) / 100.0) * (1.0 - 0.003 * (3090.0 - core) / 100.0); + let watts = 310.0 - 8.0 * (3090.0 - core) / 100.0 + 2.0 * (mem - 13801.0) / 100.0; + Row { point: p, limit: 575.0, watts, mhs, eff: mhs / watts, draws: 12, rates: 6, gclk: core, mclk: mem, tmax: 65.0, faults: 0, mark: Some(Mark::Ok) } + } + + #[test] + fn the_climb_walks_memory_up_and_core_down_and_converges_in_five_probes() { + // a 2 GHz memory range above the default (the headroom a 5090 has in practice; PC 1's driver reports 14,001) + let l = Limits { mem_max_mhz: 15801, ..l5090() }; + let plan = Plan::climb(&l, Point { clock_mhz: 0, power_pct: 100, mem_mhz: 0 }, Goal::Balanced, 1.0); + assert_eq!(plan.len(), 5); + let mut rows = Vec::new(); + while let Some(step) = plan.next(&rows) { + assert_eq!(step.kind, Kind::Climb); + rows.push(synthetic(step.point)); + } + assert_eq!(rows.len(), 5, "the budget"); + assert_eq!(rows[0].point, Point { clock_mhz: 0, power_pct: 100, mem_mhz: 0 }, "it starts at the start point"); + // every probe moved one knob the right way: memory never down, core never up, and both inside the vendor's range + for w in rows.windows(2) { + let (a, b) = (w[0].point, w[1].point); + assert!(b.mem_mhz >= a.mem_mhz || b.mem_mhz == 0, "{a:?} -> {b:?}"); + assert!(b.mem_mhz <= 15801 && (b.clock_mhz == 0 || b.clock_mhz >= 1854), "{b:?}"); + } + let best = choose(&rows, plan.tolerance_pct).unwrap(); + assert!(best.eff > rows[0].eff, "the chosen point beats the start: {:.4} > {:.4}", best.eff, rows[0].eff); + assert!(best.point.mem_mhz > 13801 || (best.point.clock_mhz > 0 && best.point.clock_mhz < 3090), "it moved a knob"); + } + + #[test] + fn a_refused_probe_backs_that_knob_off_for_good() { + let l = Limits { mem_max_mhz: 15801, ..l5090() }; + let plan = Plan::climb(&l, Point { clock_mhz: 0, power_pct: 100, mem_mhz: 0 }, Goal::Balanced, 1.0); + let mut rows = vec![synthetic(plan.next(&[]).unwrap().point)]; + // the first probe (memory up) draws a mismatched hash: refused + let probe = plan.next(&rows).unwrap(); + assert!(probe.point.mem_mhz > 13801, "memory first: {:?}", probe.point); + let mut bad = synthetic(probe.point); + bad.faults = 1; + bad.mark = Some(Mark::Faulted); + rows.push(bad); + // from here every probe leaves the memory clock alone + while let Some(step) = plan.next(&rows) { + assert_eq!(step.point.mem_mhz, 0, "memory backed off: {:?}", step.point); + rows.push(synthetic(step.point)); + } + assert!(rows.len() >= 3 && rows.len() <= 5); + assert!(choose(&rows, 1.0).unwrap().point.mem_mhz == 0); + } + + #[test] + fn each_goal_picks_its_point() { + let l = l5090(); + let p = |c: u32, m: u32| Point { clock_mhz: c, power_pct: 100, mem_mhz: m }; + // three points: the fastest (slightly less efficient), the most efficient (4% slower), and the start + let rows = vec![synthetic(p(0, 0)), synthetic(p(0, 14001)), synthetic(p(1854, 0))]; + let eff_plan = Plan::climb(&l, p(0, 0), Goal::Efficiency, 1.0); + let bal_plan = Plan::climb(&l, p(0, 0), Goal::Balanced, 1.0); + let rate_plan = Plan::climb(&l, p(0, 0), Goal::MaxRate, 1.0); + assert_eq!(eff_plan.tolerance_pct, 10.0); + assert_eq!(bal_plan.tolerance_pct, 1.0); + assert_eq!(rate_plan.tolerance_pct, 0.0); + let by = |plan: &Plan| rows.iter().filter(|r| r.usable() && r.mhs >= rows.iter().map(|x| x.mhs).fold(0.0, f64::max) * (1.0 - plan.tolerance_pct / 100.0)).max_by(|a, b| plan.score(a).partial_cmp(&plan.score(b)).unwrap()).unwrap().point; + assert_eq!(by(&rate_plan), p(0, 14001), "maximum rate takes the fastest point"); + assert_eq!(by(&eff_plan), p(1854, 0), "efficiency takes the most MH/W inside 10%"); + assert_eq!(by(&bal_plan), p(0, 14001), "balanced keeps within 1% of the top rate"); + assert_eq!(Goal::parse("efficiency"), Goal::Efficiency); + assert_eq!(Goal::parse("rate"), Goal::MaxRate); + assert_eq!(Goal::parse("anything"), Goal::Balanced); + assert!((pounds_per_day(310.0, 28.5) - 2.1204).abs() < 1e-3, "310 W a day at 28.5 p/kWh = £2.12"); + } + #[test] fn control_reasons_per_vendor() { let l = l5090(); diff --git a/relay/lib/ember.mjs b/relay/lib/ember.mjs index 166b70274..5e4604352 100644 --- a/relay/lib/ember.mjs +++ b/relay/lib/ember.mjs @@ -46,7 +46,7 @@ export function spreadPct(xs) { return Number((median(xs.map(x => Math.abs(x - m))) / m * 100).toFixed(2)); } -const usable = r => r && r.chosen && r.chosen.mark === 'ok' && r.chosen.eff > 0 && (r.plan === 'full' || r.plan === 'confirm'); +const usable = r => r && r.chosen && r.chosen.mark === 'ok' && r.chosen.eff > 0 && (r.plan === 'full' || r.plan === 'confirm' || r.plan === 'climb'); /// Folds records into priors: one per key, from the full and confirm records with a usable chosen point. The point /// is the median clock cap and the median power percent (each rounded to the step the apps use: 10 MHz, 1%), the @@ -80,6 +80,8 @@ export function aggregate(records, { minSamples = 1 } = {}) { const prior = { clock_mhz: Math.round(median(t.map(r => r.chosen.clock_mhz)) / 10) * 10, power_pct: Math.round(median(t.map(r => r.chosen.power_pct))), + // Ember 2: the memory clock (0 = the driver's default); older records carry none + mem_mhz: Math.round(median(t.map(r => r.chosen.mem_mhz || 0)) / 10) * 10, eff: Number(median(effs).toFixed(4)), mhs: Number(median(t.map(r => r.chosen.mhs)).toFixed(2)), watts: Number(median(t.map(r => r.chosen.watts)).toFixed(1)), @@ -120,7 +122,7 @@ export function priorFor(tuning, key, minSamples) { const p = tuning && tuning.priors && tuning.priors[key]; const floor = Number.isFinite(minSamples) ? minSamples : (tuning && tuning.ember && tuning.ember.min_samples) || 5; if (!p || !(p.samples >= floor)) return null; - return { clock_mhz: p.clock_mhz || 0, power_pct: Math.min(100, Math.max(50, p.power_pct || 100)), eff: p.eff, samples: p.samples }; + return { clock_mhz: p.clock_mhz || 0, power_pct: Math.min(100, Math.max(50, p.power_pct || 100)), mem_mhz: p.mem_mhz || 0, eff: p.eff, samples: p.samples }; } /// One text line per prior for the console and the CLI. diff --git a/relay/test/ember.test.mjs b/relay/test/ember.test.mjs index a5d39e96e..c8ed38170 100644 --- a/relay/test/ember.test.mjs +++ b/relay/test/ember.test.mjs @@ -82,7 +82,7 @@ test('the manifest merge keeps the kernel-variant cards and carries the settings assert.deepEqual(mergeTuning(null, {}).cards, {}); // the round trip: canonical JSON (what publish-manifest.sh signs) parses back to the same prior const back = JSON.parse(JSON.stringify(t)); - assert.deepEqual(priorFor(back, 'NVIDIA_GeForce_RTX_5090|581|l128w16'), { clock_mhz: 2470, power_pct: 100, eff: priors['NVIDIA_GeForce_RTX_5090|581|l128w16'].eff, samples: 5 }); + assert.deepEqual(priorFor(back, 'NVIDIA_GeForce_RTX_5090|581|l128w16'), { clock_mhz: 2470, power_pct: 100, mem_mhz: 0, eff: priors['NVIDIA_GeForce_RTX_5090|581|l128w16'].eff, samples: 5 }); assert.equal(priorFor(back, 'NVIDIA_GeForce_RTX_5090|581|l128w16', 6), null, 'six wanted, five there'); assert.equal(priorFor(back, 'nothing|0|v2'), null); assert.equal(priorFor(null, 'x'), null); @@ -100,6 +100,15 @@ test('AMD confirm records aggregate by their own key, and the line reads', () => assert.equal(table.length, 1); }); +test('Ember 2: climb records carry the memory clock into the prior', () => { + const climb = (m, ts, mem) => ({ ...rec(m, ts, { ...step(2472, 100, 220, 123.5), mem_mhz: mem }), plan: 'climb' }); + const { priors } = aggregate([climb('a1', 1, 14001), climb('b2', 2, 14001), climb('c3', 3, 13901), climb('d4', 4, 14001), climb('e5', 5, 14001)], { minSamples: 5 }); + const p = priors['NVIDIA_GeForce_RTX_5090|581|l128w16']; + assert.equal(p.mem_mhz, 14000, 'the median memory clock, rounded to 10 MHz'); + assert.equal(p.clock_mhz, 2470); + assert.equal(priorFor({ priors, ember: { min_samples: 5 } }, 'NVIDIA_GeForce_RTX_5090|581|l128w16').mem_mhz, 14000); +}); + test('median and spread', () => { assert.equal(median([3, 1, 2]), 2); assert.equal(median([4, 1, 2, 3]), 2.5);