the project lead, 5 October 2026, 22:45 BST: "make sure we have ember tuning every single card for efficiency out of the box, the more data = the better the tune, make an awesome system." Built on lever 3 (docs/plans/miner-eff.md), lever 2's signed tuning section (docs/design/miner-tuning.md), the AMD telemetry helper (ce28158, its --tune/--set-gmax/--set-plimit/ --reset contract) and the Power control switch (22b34e0). Design, data flow, tiers and the privacy line: docs/plans/ember-tune.md. - src/ember.rs (new): two knobs per card (power limit %, core clock cap MHz; memory clock never touched), the full plan (power ladder 100..50%, then the clock ladder 90..60% at the chosen power), the confirm plan (the fleet prior and one neighbour), the baseline plan (measure only), the marks (faulted, hot, memory_clock_dropped, unapplied, no_readings), the choice (best MH/W within 1% of the top rate, then rate, then draw), the fleet record (a hash of the install id, no address), the prior lookup and the kill switch (tuning.ember), the state machine on a fake clock. 9 unit tests. - engine.rs: tick_sweep schedules every NVIDIA, AMD and Apple card (120 s steady, 600 s to the boundary, no job hold, no pause, weekly, again after a driver major or program-class change, never under the manifest kill switch); the probe (nvidia-smi clocks.max.gr + driver_version and the direct/helper mode; igneum-gpu-telemetry --tune for AMD); tune_apply (nvidia-smi -pl / -lgc 0,<MHz> / -rgc directly or through the helper; the AMD helper per request); Cmd::TuneProbe, Cmd::TuneSet; faults from rejected and mismatched hashes mark the step; the TUNE lines and the TUNE {json} record, uploaded with the log; the Tuned line on the card state. The NVIDIA helper starts only with Power control on: the --sweep job never counts as permission (no prompt on a PC with nobody there). - sweep.rs: the helper protocol gains lgc/rgc (clock cap and reset) and resets the clocks after 20 idle minutes. - state.rs, config.rs: the tune fields (clock cap, driver, class, source, the Tuned line); the nvidia-smi telemetry query carries clocks.gr and clocks.mem; the AMD sample line's plimit_pct and gmax_mhz are parsed. - ui: "Tuned: X MH/s at Y W (Z MH/W)" with the point, the source and when; measure-only cards say why; the Ember Tune switch; tune-line.test.mjs. - relay/lib/ember.mjs + relay/test/ember.test.mjs: the aggregation per (card model | driver major | program class): median point, MH/W, spread, samples, machines; five samples converge, an outlier does not move the median, baselines make no prior, de-duplication, the manifest merge keeps lever 2's cards. api/console.mjs fn=tuning and tools/console.mjs tuning; tools/tuning.mjs --priors [--write tuning.json] [--site] [--tuning-off]. - site: the fleet priors table on /miners (site/miner-priors.json), the lever text. - relay/playbooks/ember-tune-pc1.ps1: the PC 1 run (second engine with --sweep from a scratch copy of the install). Measured tonight: see the bench log entry that follows the PC 1 run. The 9070 XT left PC 1's bus at 20:40 UTC and the 5090 needs the administrator prompt the project lead cannot answer asleep, so tonight's PC 1 run is the baseline plan on the 5090 through the whole pipeline; the two-knob tune on both cards is owed. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
131 lines
7.4 KiB
JavaScript
131 lines
7.4 KiB
JavaScript
// Ember Tune, the fleet side (docs/plans/ember-tune.md): the TUNE records every app uploads with its log are folded
|
|
// into one prior per (card model, driver major, program class): the median chosen point, its spread and the sample
|
|
// count. The publisher writes the priors into the signed manifest's `tuning` section beside the kernel-variant
|
|
// cards (tools/tuning.mjs --write), the console shows them (api/console.mjs fn=tuning, tools/console.mjs tuning),
|
|
// and the public bench table lists them per model (site/miner-priors.json). No dependencies; the tests in
|
|
// relay/test/ember.test.mjs drive these functions with a fixture of captured records.
|
|
//
|
|
// A record (app/igneum-app/src/ember.rs record_json): {ts, machine (a hash of the install id), app, os, card, vendor,
|
|
// driver, driver_major, class, key, plan: full|confirm|baseline, steps: [{clock_mhz, power_pct, limit_w, watts, mhs,
|
|
// eff, gclk, mclk, tmax, faults, mark}], chosen: {...}, before: {...}|null, eff, mhs, watts}. Nothing identifies the
|
|
// owner: no address, no hostname, no raw machine id.
|
|
|
|
/// The TUNE records inside uploaded log text, de-duplicated on (machine, card, ts) because the log is re-sent every
|
|
/// minute. Baseline records (measure only) are kept apart: they say what a card does untuned, never what to set.
|
|
export function parseRecords(text) {
|
|
const out = [];
|
|
for (const line of String(text || '').split('\n')) {
|
|
const i = line.indexOf('TUNE {');
|
|
if (i < 0) continue;
|
|
let rec;
|
|
try { rec = JSON.parse(line.slice(i + 5)); } catch { continue; }
|
|
if (!rec || !rec.card || !rec.key || !rec.plan) continue;
|
|
out.push(rec);
|
|
}
|
|
return out;
|
|
}
|
|
|
|
export function dedupe(records) {
|
|
const seen = new Set();
|
|
const out = [];
|
|
for (const r of records) {
|
|
const k = `${r.machine}|${r.card}|${r.ts}`;
|
|
if (seen.has(k)) continue;
|
|
seen.add(k);
|
|
out.push(r);
|
|
}
|
|
return out;
|
|
}
|
|
|
|
export const median = xs => { const s = xs.filter(x => Number.isFinite(x)).sort((a, b) => a - b); return s.length ? (s.length % 2 ? s[(s.length - 1) / 2] : (s[s.length / 2 - 1] + s[s.length / 2]) / 2) : 0; };
|
|
|
|
/// The median absolute deviation as a percent of the median (0 for one sample or a zero median).
|
|
export function spreadPct(xs) {
|
|
const m = median(xs);
|
|
if (!m || xs.length < 2) return 0;
|
|
return Number((median(xs.map(x => Math.abs(x - m))) / m * 100).toFixed(2));
|
|
}
|
|
|
|
const usable = r => r && r.chosen && r.chosen.mark === 'ok' && r.chosen.eff > 0 && (r.plan === 'full' || r.plan === 'confirm');
|
|
|
|
/// Folds records into priors: one per key, from the full and confirm records with a usable chosen point. The point
|
|
/// is the median clock cap and the median power percent (each rounded to the step the apps use: 10 MHz, 1%), the
|
|
/// efficiency, rate and draw are medians, the spread is the MAD of the efficiency in percent, `samples` counts the
|
|
/// records and `machines` the distinct install hashes. An outlier (one bad card, one hot room) moves the median by
|
|
/// at most one rank, never by its size. Baseline records are summarised beside the prior as `baseline` (median
|
|
/// MH/W untuned) so the console can show the gain.
|
|
export function aggregate(records, { minSamples = 1 } = {}) {
|
|
const byKey = new Map();
|
|
for (const r of dedupe(records)) {
|
|
const g = byKey.get(r.key) || { key: r.key, card: r.card, vendor: r.vendor || '', driver_major: r.driver_major || '', class: r.class || 'v2', tuned: [], baseline: [], machines: new Set() };
|
|
byKey.set(r.key, g);
|
|
g.machines.add(r.machine);
|
|
if (usable(r)) g.tuned.push(r);
|
|
else if (r.plan === 'baseline' && r.chosen && r.chosen.eff > 0) g.baseline.push(r);
|
|
}
|
|
const priors = {};
|
|
const table = [];
|
|
for (const g of byKey.values()) {
|
|
const t = g.tuned;
|
|
const row = {
|
|
key: g.key, card: g.card, vendor: g.vendor, driver_major: g.driver_major, class: g.class,
|
|
samples: t.length, machines: g.machines.size,
|
|
baseline_samples: g.baseline.length,
|
|
baseline_eff: g.baseline.length ? Number(median(g.baseline.map(r => r.chosen.eff)).toFixed(4)) : null,
|
|
baseline_mhs: g.baseline.length ? Number(median(g.baseline.map(r => r.chosen.mhs)).toFixed(2)) : null,
|
|
baseline_watts: g.baseline.length ? Number(median(g.baseline.map(r => r.chosen.watts)).toFixed(1)) : null,
|
|
};
|
|
if (t.length) {
|
|
const effs = t.map(r => r.chosen.eff);
|
|
const prior = {
|
|
clock_mhz: Math.round(median(t.map(r => r.chosen.clock_mhz)) / 10) * 10,
|
|
power_pct: Math.round(median(t.map(r => r.chosen.power_pct))),
|
|
eff: Number(median(effs).toFixed(4)),
|
|
mhs: Number(median(t.map(r => r.chosen.mhs)).toFixed(2)),
|
|
watts: Number(median(t.map(r => r.chosen.watts)).toFixed(1)),
|
|
spread_pct: spreadPct(effs),
|
|
samples: t.length,
|
|
machines: g.machines.size,
|
|
card: g.card,
|
|
vendor: g.vendor,
|
|
driver_major: g.driver_major,
|
|
class: g.class,
|
|
updated: new Date(Math.max(...t.map(r => Number(r.ts) || 0)) * 1000).toISOString().replace(/\.\d{3}Z$/, 'Z'),
|
|
};
|
|
// the untuned reference: the full plan's first step (the power ladder's 100%), else the baseline records
|
|
const befores = t.map(r => r.before && r.before.eff > 0 ? r.before.eff : null).filter(x => x !== null);
|
|
if (befores.length) prior.before_eff = Number(median(befores).toFixed(4));
|
|
else if (row.baseline_eff) prior.before_eff = row.baseline_eff;
|
|
if (prior.before_eff) prior.gain_pct = Number(((prior.eff / prior.before_eff - 1) * 100).toFixed(1));
|
|
Object.assign(row, prior);
|
|
if (t.length >= minSamples) priors[g.key] = prior;
|
|
}
|
|
table.push(row);
|
|
}
|
|
table.sort((a, b) => (b.samples - a.samples) || (a.key < b.key ? -1 : 1));
|
|
return { priors, table };
|
|
}
|
|
|
|
/// The manifest's tuning section with the priors folded in: the kernel-variant `cards` object is kept as is,
|
|
/// `priors` replaces the previous priors (a key that lost its samples drops out), `ember` carries the settings.
|
|
export function mergeTuning(existing, priors, ember = {}) {
|
|
const base = existing && typeof existing === 'object' ? existing : {};
|
|
const cards = base.cards && typeof base.cards === 'object' && !Array.isArray(base.cards) ? base.cards : {};
|
|
const settings = { enabled: true, min_samples: 5, rate_tolerance_pct: 1, ...(base.ember && typeof base.ember === 'object' ? base.ember : {}), ...ember };
|
|
return { ...base, updated: new Date().toISOString().replace(/\.\d{3}Z$/, 'Z'), cards, ember: settings, priors: priors || {} };
|
|
}
|
|
|
|
/// A prior as a card starts from it (app/igneum-app/src/ember.rs prior_of): None under the sample floor.
|
|
export function priorFor(tuning, key, minSamples) {
|
|
const p = tuning && tuning.priors && tuning.priors[key];
|
|
const floor = Number.isFinite(minSamples) ? minSamples : (tuning && tuning.ember && tuning.ember.min_samples) || 5;
|
|
if (!p || !(p.samples >= floor)) return null;
|
|
return { clock_mhz: p.clock_mhz || 0, power_pct: Math.min(100, Math.max(50, p.power_pct || 100)), eff: p.eff, samples: p.samples };
|
|
}
|
|
|
|
/// One text line per prior for the console and the CLI.
|
|
export function priorLine(p) {
|
|
const point = p.clock_mhz ? `${p.clock_mhz} MHz at ${p.power_pct}%` : `${p.power_pct}% (clock unlocked)`;
|
|
const gain = p.gain_pct != null ? ` (${p.gain_pct >= 0 ? '+' : ''}${p.gain_pct}% over untuned ${p.before_eff} MH/W)` : '';
|
|
return `${p.card.replace(/_/g, ' ')} | driver ${p.driver_major} | ${p.class}: ${point}, ${p.eff} MH/W${gain}, ${p.mhs} MH/s at ${p.watts} W, spread ${p.spread_pct}%, ${p.samples} sample(s) from ${p.machines} machine(s)`;
|
|
}
|