Ember 2: the climb's tests (a synthetic memory-bound card converges in five probes; a refused probe backs the knob off for good; each goal picks its point; the £/day formula) and the fleet prior carrying the memory clock (climb records aggregate)
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
This commit is contained in:
parent
0bf27b1d78
commit
229ca3f091
3 changed files with 92 additions and 3 deletions
|
|
@ -1180,6 +1180,84 @@ mod tests {
|
|||
assert_eq!(program_class(0, 0), "v2");
|
||||
}
|
||||
|
||||
/// Ember 2: a synthetic memory-latency-bound card. Rate rises 1% per 100 MHz of memory above the default and
|
||||
/// falls only 0.3% per 100 MHz of core below the maximum; draw falls 8 W per 100 MHz of core and rises 2 W per
|
||||
/// 100 MHz of memory. The climb must walk memory up and core down and converge in under five probes.
|
||||
fn synthetic(p: Point) -> Row {
|
||||
let mem = if p.mem_mhz == 0 { 13801.0 } else { p.mem_mhz as f64 };
|
||||
let core = if p.clock_mhz == 0 { 3090.0 } else { p.clock_mhz as f64 };
|
||||
let mhs = 127.0 * (1.0 + 0.015 * (mem - 13801.0) / 100.0) * (1.0 - 0.003 * (3090.0 - core) / 100.0);
|
||||
let watts = 310.0 - 8.0 * (3090.0 - core) / 100.0 + 2.0 * (mem - 13801.0) / 100.0;
|
||||
Row { point: p, limit: 575.0, watts, mhs, eff: mhs / watts, draws: 12, rates: 6, gclk: core, mclk: mem, tmax: 65.0, faults: 0, mark: Some(Mark::Ok) }
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_climb_walks_memory_up_and_core_down_and_converges_in_five_probes() {
|
||||
// a 2 GHz memory range above the default (the headroom a 5090 has in practice; PC 1's driver reports 14,001)
|
||||
let l = Limits { mem_max_mhz: 15801, ..l5090() };
|
||||
let plan = Plan::climb(&l, Point { clock_mhz: 0, power_pct: 100, mem_mhz: 0 }, Goal::Balanced, 1.0);
|
||||
assert_eq!(plan.len(), 5);
|
||||
let mut rows = Vec::new();
|
||||
while let Some(step) = plan.next(&rows) {
|
||||
assert_eq!(step.kind, Kind::Climb);
|
||||
rows.push(synthetic(step.point));
|
||||
}
|
||||
assert_eq!(rows.len(), 5, "the budget");
|
||||
assert_eq!(rows[0].point, Point { clock_mhz: 0, power_pct: 100, mem_mhz: 0 }, "it starts at the start point");
|
||||
// every probe moved one knob the right way: memory never down, core never up, and both inside the vendor's range
|
||||
for w in rows.windows(2) {
|
||||
let (a, b) = (w[0].point, w[1].point);
|
||||
assert!(b.mem_mhz >= a.mem_mhz || b.mem_mhz == 0, "{a:?} -> {b:?}");
|
||||
assert!(b.mem_mhz <= 15801 && (b.clock_mhz == 0 || b.clock_mhz >= 1854), "{b:?}");
|
||||
}
|
||||
let best = choose(&rows, plan.tolerance_pct).unwrap();
|
||||
assert!(best.eff > rows[0].eff, "the chosen point beats the start: {:.4} > {:.4}", best.eff, rows[0].eff);
|
||||
assert!(best.point.mem_mhz > 13801 || (best.point.clock_mhz > 0 && best.point.clock_mhz < 3090), "it moved a knob");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_refused_probe_backs_that_knob_off_for_good() {
|
||||
let l = Limits { mem_max_mhz: 15801, ..l5090() };
|
||||
let plan = Plan::climb(&l, Point { clock_mhz: 0, power_pct: 100, mem_mhz: 0 }, Goal::Balanced, 1.0);
|
||||
let mut rows = vec![synthetic(plan.next(&[]).unwrap().point)];
|
||||
// the first probe (memory up) draws a mismatched hash: refused
|
||||
let probe = plan.next(&rows).unwrap();
|
||||
assert!(probe.point.mem_mhz > 13801, "memory first: {:?}", probe.point);
|
||||
let mut bad = synthetic(probe.point);
|
||||
bad.faults = 1;
|
||||
bad.mark = Some(Mark::Faulted);
|
||||
rows.push(bad);
|
||||
// from here every probe leaves the memory clock alone
|
||||
while let Some(step) = plan.next(&rows) {
|
||||
assert_eq!(step.point.mem_mhz, 0, "memory backed off: {:?}", step.point);
|
||||
rows.push(synthetic(step.point));
|
||||
}
|
||||
assert!(rows.len() >= 3 && rows.len() <= 5);
|
||||
assert!(choose(&rows, 1.0).unwrap().point.mem_mhz == 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn each_goal_picks_its_point() {
|
||||
let l = l5090();
|
||||
let p = |c: u32, m: u32| Point { clock_mhz: c, power_pct: 100, mem_mhz: m };
|
||||
// three points: the fastest (slightly less efficient), the most efficient (4% slower), and the start
|
||||
let rows = vec![synthetic(p(0, 0)), synthetic(p(0, 14001)), synthetic(p(1854, 0))];
|
||||
let eff_plan = Plan::climb(&l, p(0, 0), Goal::Efficiency, 1.0);
|
||||
let bal_plan = Plan::climb(&l, p(0, 0), Goal::Balanced, 1.0);
|
||||
let rate_plan = Plan::climb(&l, p(0, 0), Goal::MaxRate, 1.0);
|
||||
assert_eq!(eff_plan.tolerance_pct, 10.0);
|
||||
assert_eq!(bal_plan.tolerance_pct, 1.0);
|
||||
assert_eq!(rate_plan.tolerance_pct, 0.0);
|
||||
let by = |plan: &Plan| rows.iter().filter(|r| r.usable() && r.mhs >= rows.iter().map(|x| x.mhs).fold(0.0, f64::max) * (1.0 - plan.tolerance_pct / 100.0)).max_by(|a, b| plan.score(a).partial_cmp(&plan.score(b)).unwrap()).unwrap().point;
|
||||
assert_eq!(by(&rate_plan), p(0, 14001), "maximum rate takes the fastest point");
|
||||
assert_eq!(by(&eff_plan), p(1854, 0), "efficiency takes the most MH/W inside 10%");
|
||||
assert_eq!(by(&bal_plan), p(0, 14001), "balanced keeps within 1% of the top rate");
|
||||
assert_eq!(Goal::parse("efficiency"), Goal::Efficiency);
|
||||
assert_eq!(Goal::parse("rate"), Goal::MaxRate);
|
||||
assert_eq!(Goal::parse("anything"), Goal::Balanced);
|
||||
assert!((pounds_per_day(310.0, 28.5) - 2.1204).abs() < 1e-3, "310 W a day at 28.5 p/kWh = £2.12");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn control_reasons_per_vendor() {
|
||||
let l = l5090();
|
||||
|
|
|
|||
|
|
@ -46,7 +46,7 @@ export function spreadPct(xs) {
|
|||
return Number((median(xs.map(x => Math.abs(x - m))) / m * 100).toFixed(2));
|
||||
}
|
||||
|
||||
const usable = r => r && r.chosen && r.chosen.mark === 'ok' && r.chosen.eff > 0 && (r.plan === 'full' || r.plan === 'confirm');
|
||||
const usable = r => r && r.chosen && r.chosen.mark === 'ok' && r.chosen.eff > 0 && (r.plan === 'full' || r.plan === 'confirm' || r.plan === 'climb');
|
||||
|
||||
/// Folds records into priors: one per key, from the full and confirm records with a usable chosen point. The point
|
||||
/// is the median clock cap and the median power percent (each rounded to the step the apps use: 10 MHz, 1%), the
|
||||
|
|
@ -80,6 +80,8 @@ export function aggregate(records, { minSamples = 1 } = {}) {
|
|||
const prior = {
|
||||
clock_mhz: Math.round(median(t.map(r => r.chosen.clock_mhz)) / 10) * 10,
|
||||
power_pct: Math.round(median(t.map(r => r.chosen.power_pct))),
|
||||
// Ember 2: the memory clock (0 = the driver's default); older records carry none
|
||||
mem_mhz: Math.round(median(t.map(r => r.chosen.mem_mhz || 0)) / 10) * 10,
|
||||
eff: Number(median(effs).toFixed(4)),
|
||||
mhs: Number(median(t.map(r => r.chosen.mhs)).toFixed(2)),
|
||||
watts: Number(median(t.map(r => r.chosen.watts)).toFixed(1)),
|
||||
|
|
@ -120,7 +122,7 @@ export function priorFor(tuning, key, minSamples) {
|
|||
const p = tuning && tuning.priors && tuning.priors[key];
|
||||
const floor = Number.isFinite(minSamples) ? minSamples : (tuning && tuning.ember && tuning.ember.min_samples) || 5;
|
||||
if (!p || !(p.samples >= floor)) return null;
|
||||
return { clock_mhz: p.clock_mhz || 0, power_pct: Math.min(100, Math.max(50, p.power_pct || 100)), eff: p.eff, samples: p.samples };
|
||||
return { clock_mhz: p.clock_mhz || 0, power_pct: Math.min(100, Math.max(50, p.power_pct || 100)), mem_mhz: p.mem_mhz || 0, eff: p.eff, samples: p.samples };
|
||||
}
|
||||
|
||||
/// One text line per prior for the console and the CLI.
|
||||
|
|
|
|||
|
|
@ -82,7 +82,7 @@ test('the manifest merge keeps the kernel-variant cards and carries the settings
|
|||
assert.deepEqual(mergeTuning(null, {}).cards, {});
|
||||
// the round trip: canonical JSON (what publish-manifest.sh signs) parses back to the same prior
|
||||
const back = JSON.parse(JSON.stringify(t));
|
||||
assert.deepEqual(priorFor(back, 'NVIDIA_GeForce_RTX_5090|581|l128w16'), { clock_mhz: 2470, power_pct: 100, eff: priors['NVIDIA_GeForce_RTX_5090|581|l128w16'].eff, samples: 5 });
|
||||
assert.deepEqual(priorFor(back, 'NVIDIA_GeForce_RTX_5090|581|l128w16'), { clock_mhz: 2470, power_pct: 100, mem_mhz: 0, eff: priors['NVIDIA_GeForce_RTX_5090|581|l128w16'].eff, samples: 5 });
|
||||
assert.equal(priorFor(back, 'NVIDIA_GeForce_RTX_5090|581|l128w16', 6), null, 'six wanted, five there');
|
||||
assert.equal(priorFor(back, 'nothing|0|v2'), null);
|
||||
assert.equal(priorFor(null, 'x'), null);
|
||||
|
|
@ -100,6 +100,15 @@ test('AMD confirm records aggregate by their own key, and the line reads', () =>
|
|||
assert.equal(table.length, 1);
|
||||
});
|
||||
|
||||
test('Ember 2: climb records carry the memory clock into the prior', () => {
|
||||
const climb = (m, ts, mem) => ({ ...rec(m, ts, { ...step(2472, 100, 220, 123.5), mem_mhz: mem }), plan: 'climb' });
|
||||
const { priors } = aggregate([climb('a1', 1, 14001), climb('b2', 2, 14001), climb('c3', 3, 13901), climb('d4', 4, 14001), climb('e5', 5, 14001)], { minSamples: 5 });
|
||||
const p = priors['NVIDIA_GeForce_RTX_5090|581|l128w16'];
|
||||
assert.equal(p.mem_mhz, 14000, 'the median memory clock, rounded to 10 MHz');
|
||||
assert.equal(p.clock_mhz, 2470);
|
||||
assert.equal(priorFor({ priors, ember: { min_samples: 5 } }, 'NVIDIA_GeForce_RTX_5090|581|l128w16').mem_mhz, 14000);
|
||||
});
|
||||
|
||||
test('median and spread', () => {
|
||||
assert.equal(median([3, 1, 2]), 2);
|
||||
assert.equal(median([4, 1, 2, 3]), 2.5);
|
||||
|
|
|
|||
Loading…
Reference in a new issue