Run 5 (15:28 to 15:39Z): the 5090's five power rows in the bench log (the cap does not bind: 0.41 MH/W flat at 310 W); the playbook's watchdog killed the live three-card tune at the first clock step, so: idle = three consecutive status lines reading 0.00 MH/s and 300 s, never a missing match; the after snapshot and the engine-log dump on every exit; an elevated tune engine registers the Igneum Power Helper itself before its first step (the --sweep engine skipped the cap path where the registration lived); Ember 2 groundwork: the memory-clock knob in Point and Limits, the goal, the hill-climb plan (not yet wired to the engine)

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
This commit is contained in:
igneum-labs 2026-10-06 15:43:53 +00:00
parent 7ed0d75160
commit c5e5de6752
5 changed files with 258 additions and 60 deletions

View file

@ -49,6 +49,10 @@ pub struct Limits {
pub clock_max_mhz: u32,
/// the lowest cap the vendor allows (the ADLX gmax_range floor); 0 = CLOCK_FLOOR_PCT of the maximum
pub clock_min_mhz: u32,
/// Ember 2: the memory clock the card runs at by default and the vendor's maximum (nvidia-smi clocks.mem and
/// clocks.max.mem); 0 = no memory knob (AMD through ADLX on RDNA 4 exposes none)
pub mem_default_mhz: u32,
pub mem_max_mhz: u32,
}
impl Limits {
@ -66,6 +70,13 @@ impl Limits {
}
mhz.clamp(self.clock_floor(), self.clock_max_mhz)
}
/// A memory clock inside the vendor's range; 0 stays 0 (the default).
pub fn clamp_mem(&self, mhz: u32) -> u32 {
if mhz == 0 || self.mem_max_mhz == 0 || self.mem_default_mhz == 0 {
return 0;
}
mhz.clamp(self.mem_default_mhz, self.mem_max_mhz)
}
/// The watts a power percent asks for, inside the card's min and max, rounded to a watt.
pub fn watts_for(&self, pct: u32) -> f64 {
let mut w = self.power_default_w * pct as f64 / 100.0;
@ -79,13 +90,15 @@ impl Limits {
}
}
/// One setting of the two knobs.
/// One setting of the knobs (Ember 2 adds the memory clock: 0 = the driver's default).
#[derive(Clone, Copy, Debug, Default, PartialEq, Eq, Hash)]
pub struct Point {
/// the core clock cap in MHz; 0 = unlocked
pub clock_mhz: u32,
/// the power limit, percent of the default
pub power_pct: u32,
/// the memory clock in MHz (NVIDIA `-lmc`, locked to one value); 0 = the driver's default
pub mem_mhz: u32,
}
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
@ -96,6 +109,8 @@ pub enum Kind {
Clock,
/// the fleet prior and one neighbour
Confirm,
/// Ember 2: a hill-climb probe
Climb,
}
#[derive(Clone, Debug, PartialEq)]
@ -106,11 +121,53 @@ pub struct Step {
pub kind: Kind,
}
/// What the tune is for (Settings > Ember Tune > goal). The rate floor is the share of the best rate seen a point
/// must keep to win on MH per watt: efficiency keeps 90%, balanced 99% (the 1% rule of lever 3), maximum rate
/// takes the fastest point and uses MH per watt only to break ties.
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
pub enum Goal {
Efficiency,
Balanced,
MaxRate,
}
impl Goal {
pub fn parse(s: &str) -> Goal {
match s {
"efficiency" | "eff" => Goal::Efficiency,
"rate" | "max_rate" | "maximum" => Goal::MaxRate,
_ => Goal::Balanced,
}
}
pub fn name(&self) -> &'static str {
match self {
Goal::Efficiency => "efficiency",
Goal::Balanced => "balanced",
Goal::MaxRate => "rate",
}
}
/// The rate tolerance the choice rule uses, percent under the best rate.
pub fn tolerance_pct(&self, manifest_default: f64) -> f64 {
match self {
Goal::Efficiency => 10.0,
Goal::Balanced => manifest_default,
Goal::MaxRate => 0.0,
}
}
}
/// Pounds a day for a draw at a price in pence per kWh: watts × 24 h / 1000 × price / 100.
pub fn pounds_per_day(watts: f64, pence_per_kwh: f64) -> f64 {
watts * 24.0 / 1000.0 * pence_per_kwh / 100.0
}
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
pub enum PlanKind {
Full,
Confirm,
Baseline,
/// Ember 2: the hill-climb over memory up and core down from the start point
Climb,
}
impl PlanKind {
@ -119,6 +176,7 @@ impl PlanKind {
PlanKind::Full => "full",
PlanKind::Confirm => "confirm",
PlanKind::Baseline => "baseline",
PlanKind::Climb => "climb",
}
}
}
@ -134,9 +192,82 @@ pub struct Plan {
power: Vec<Step>,
clock_pcts: Vec<u32>,
fixed: Vec<Step>,
/// Ember 2 (Climb): the start point, the step sizes and the step budget
climb: Option<Climb>,
}
/// The hill-climb's shape: from `start`, each probe moves the memory clock up by `mem_step` or the core clock down
/// by `core_step` (or both), keeps the move when the goal's score improves, else turns to the other knob; a
/// refused step (a fault on it) backs that knob off for good. At most `budget` steps including the start.
#[derive(Clone, Debug, PartialEq)]
pub struct Climb {
pub start: Point,
pub mem_step: u32,
pub core_step: u32,
pub budget: usize,
pub goal: Goal,
}
impl Plan {
/// Ember 2: the hill-climb. The start is the fleet prior (or the card's current point); a probe step is 5% of
/// the memory range above the default (0 when the card has no memory knob) and 5% of the maximum core clock;
/// five steps of 60 s converge in under 10 minutes.
pub fn climb(limits: &Limits, start: Point, goal: Goal, tolerance_pct: f64) -> Plan {
let mem_step = if limits.mem_max_mhz > limits.mem_default_mhz { ((limits.mem_max_mhz - limits.mem_default_mhz) / 20).max(25) } else { 0 };
let core_step = if limits.clock_max_mhz > 0 { (limits.clock_max_mhz / 20).max(25) } else { 0 };
let start = Point { clock_mhz: limits.clamp_clock(start.clock_mhz), power_pct: start.power_pct.clamp(50, 100), mem_mhz: limits.clamp_mem(start.mem_mhz) };
Plan { kind: PlanKind::Climb, limits: limits.clone(), before: start, tolerance_pct: goal.tolerance_pct(tolerance_pct), power: Vec::new(), clock_pcts: Vec::new(), fixed: Vec::new(), climb: Some(Climb { start, mem_step, core_step, budget: 5, goal }) }
}
/// The goal's score of a row: MH per watt for efficiency and balanced, the rate for maximum rate.
pub fn score(&self, r: &Row) -> f64 {
match self.climb.as_ref().map(|c| c.goal) {
Some(Goal::MaxRate) => r.mhs,
_ => r.eff,
}
}
/// The climb's next probe from the rows so far: None when the budget is spent or no move is left.
fn climb_next(&self, rows: &[Row]) -> Option<Step> {
let c = self.climb.as_ref()?;
let step = |p: Point| Step { point: p, watts: self.limits.watts_for(p.power_pct), kind: Kind::Climb };
if rows.is_empty() {
return Some(step(c.start));
}
if rows.len() >= c.budget {
return None;
}
// the best usable row so far is the hill's top; a knob that produced a marked (refused) row is backed off
let best = rows.iter().filter(|r| r.usable()).max_by(|a, b| self.score(a).partial_cmp(&self.score(b)).unwrap_or(std::cmp::Ordering::Equal))?;
let refused_mem = rows.iter().any(|r| !r.usable() && r.point.mem_mhz > best.point.mem_mhz);
let refused_core = rows.iter().any(|r| !r.usable() && r.point.clock_mhz != 0 && (best.point.clock_mhz == 0 || r.point.clock_mhz < best.point.clock_mhz));
let last = rows.last()?;
let mem_up = |p: Point| -> Option<Point> {
if c.mem_step == 0 || refused_mem { return None; }
let base = if p.mem_mhz == 0 { self.limits.mem_default_mhz } else { p.mem_mhz };
let m = self.limits.clamp_mem(base + c.mem_step);
(m > 0 && m != p.mem_mhz && m != base).then_some(Point { mem_mhz: m, ..p })
};
let core_down = |p: Point| -> Option<Point> {
if c.core_step == 0 || refused_core { return None; }
let base = if p.clock_mhz == 0 { self.limits.clock_max_mhz } else { p.clock_mhz };
let k = self.limits.clamp_clock(base.saturating_sub(c.core_step));
(k > 0 && k != p.clock_mhz).then_some(Point { clock_mhz: k, ..p })
};
let tried = |p: Point| rows.iter().any(|r| r.point == p);
// the last move improved: keep going the same way from the top; else turn: memory first, then core, then both
let last_improved = last.usable() && last.point == best.point && rows.len() > 1;
let last_was_mem = rows.len() > 1 && last.point.mem_mhz != rows[rows.len() - 2].point.mem_mhz;
let candidates: Vec<Option<Point>> = if last_improved && last_was_mem {
vec![mem_up(best.point), core_down(best.point)]
} else if last_improved {
vec![core_down(best.point), mem_up(best.point)]
} else {
vec![mem_up(best.point), core_down(best.point), mem_up(best.point).and_then(core_down)]
};
candidates.into_iter().flatten().find(|p| !tried(*p)).map(step)
}
/// Power 100% to 50% at the unlocked clock (duplicate watts dropped, as the card's floor clamps them), then the
/// clock ladder 90% to the floor of the maximum core clock at the chosen power. A card without a readable
/// maximum clock gets the power ladder only; a card without a default limit gets the clock ladder only.
@ -148,41 +279,44 @@ impl Plan {
if power.last().map(|s: &Step| (s.watts - w).abs() < 0.5).unwrap_or(false) {
continue;
}
power.push(Step { point: Point { clock_mhz: 0, power_pct: pct }, watts: w, kind: Kind::Power });
power.push(Step { point: Point { clock_mhz: 0, power_pct: pct, mem_mhz: 0 }, watts: w, kind: Kind::Power });
}
}
let clock_pcts = if limits.clock_max_mhz > 0 { CLOCK_STEPS_PCT[1..].to_vec() } else { Vec::new() };
Plan { kind: PlanKind::Full, limits: limits.clone(), before, tolerance_pct, power, clock_pcts, fixed: Vec::new() }
Plan { kind: PlanKind::Full, limits: limits.clone(), before, tolerance_pct, power, clock_pcts, fixed: Vec::new(), climb: None }
}
/// The prior's point, then one neighbour: the next clock step up when the prior caps the clock (is the cap
/// costing rate?), else one power step down (is there efficiency left?).
pub fn confirm(limits: &Limits, prior: Point, before: Point, tolerance_pct: f64) -> Plan {
let p = Point { clock_mhz: limits.clamp_clock(prior.clock_mhz), power_pct: prior.power_pct.clamp(50, 100) };
let p = Point { clock_mhz: limits.clamp_clock(prior.clock_mhz), power_pct: prior.power_pct.clamp(50, 100), mem_mhz: limits.clamp_mem(prior.mem_mhz) };
let first = Step { point: p, watts: limits.watts_for(p.power_pct), kind: Kind::Confirm };
let neighbour = if p.clock_mhz > 0 && limits.clock_max_mhz > 0 {
let up = p.clock_mhz + limits.clock_max_mhz / 10;
let clock = if up >= limits.clock_max_mhz { 0 } else { limits.clamp_clock(up) };
Point { clock_mhz: clock, power_pct: p.power_pct }
Point { clock_mhz: clock, power_pct: p.power_pct, mem_mhz: p.mem_mhz }
} else {
Point { clock_mhz: p.clock_mhz, power_pct: (p.power_pct.saturating_sub(10)).max(50) }
Point { clock_mhz: p.clock_mhz, power_pct: (p.power_pct.saturating_sub(10)).max(50), mem_mhz: p.mem_mhz }
};
let mut fixed = vec![first];
if neighbour != p {
fixed.push(Step { point: neighbour, watts: limits.watts_for(neighbour.power_pct), kind: Kind::Confirm });
}
Plan { kind: PlanKind::Confirm, limits: limits.clone(), before, tolerance_pct, power: Vec::new(), clock_pcts: Vec::new(), fixed }
Plan { kind: PlanKind::Confirm, limits: limits.clone(), before, tolerance_pct, power: Vec::new(), clock_pcts: Vec::new(), fixed, climb: None }
}
/// One step at the card's current point: the before number, and all a measure-only card (Apple, or NVIDIA
/// with Power control off) reports.
pub fn baseline(limits: &Limits, before: Point, tolerance_pct: f64) -> Plan {
let fixed = vec![Step { point: before, watts: limits.watts_for(before.power_pct), kind: Kind::Baseline }];
Plan { kind: PlanKind::Baseline, limits: limits.clone(), before, tolerance_pct, power: Vec::new(), clock_pcts: Vec::new(), fixed }
Plan { kind: PlanKind::Baseline, limits: limits.clone(), before, tolerance_pct, power: Vec::new(), clock_pcts: Vec::new(), fixed, climb: None }
}
/// How many steps the plan has at most (the clock ladder counts whether or not it runs).
pub fn len(&self) -> usize {
if let Some(c) = &self.climb {
return c.budget;
}
self.fixed.len() + self.power.len() + self.clock_pcts.len()
}
pub fn is_empty(&self) -> bool {
@ -191,6 +325,9 @@ impl Plan {
/// The next step after `rows` (one row per step done so far), or None when the plan is complete.
pub fn next(&self, rows: &[Row]) -> Option<Step> {
if self.climb.is_some() {
return self.climb_next(rows);
}
let i = rows.len();
if !self.fixed.is_empty() {
return self.fixed.get(i).cloned();
@ -207,7 +344,7 @@ impl Plan {
if rows.last().map(|r| r.point.clock_mhz == clock).unwrap_or(false) {
return None;
}
Some(Step { point: Point { clock_mhz: clock, power_pct }, watts: self.limits.watts_for(power_pct), kind: Kind::Clock })
Some(Step { point: Point { clock_mhz: clock, power_pct, mem_mhz: 0 }, watts: self.limits.watts_for(power_pct), kind: Kind::Clock })
}
}
@ -305,9 +442,10 @@ impl Row {
/// `TUNE card=<label> step=<i> clock=<MHz> cap=<pct> limit=<W> watts=<W> mhs=<x> eff=<MH/W> gclk=<MHz> mclk=<MHz> tmax=<C> mark=<m>`
pub fn line(&self, card: &str, i: usize) -> String {
format!(
"TUNE card={card} step={i} clock={} cap={} limit={:.0} watts={:.1} mhs={:.2} eff={:.4} gclk={:.0} mclk={:.0} tmax={:.0} mark={}{}",
"TUNE card={card} step={i} clock={} cap={} mem={} limit={:.0} watts={:.1} mhs={:.2} eff={:.4} gclk={:.0} mclk={:.0} tmax={:.0} mark={}{}",
self.point.clock_mhz,
self.point.power_pct,
self.point.mem_mhz,
self.limit,
self.watts,
self.mhs,
@ -320,7 +458,7 @@ impl Row {
)
}
pub fn json(&self) -> serde_json::Value {
serde_json::json!({ "clock_mhz": self.point.clock_mhz, "power_pct": self.point.power_pct, "limit_w": self.limit, "watts": r1(self.watts), "mhs": r2(self.mhs), "eff": r4(self.eff), "gclk": self.gclk.round(), "mclk": self.mclk.round(), "tmax": self.tmax.round(), "faults": self.faults, "mark": self.mark.unwrap_or(Mark::NoReadings).name() })
serde_json::json!({ "clock_mhz": self.point.clock_mhz, "power_pct": self.point.power_pct, "mem_mhz": self.point.mem_mhz, "limit_w": self.limit, "watts": r1(self.watts), "mhs": r2(self.mhs), "eff": r4(self.eff), "gclk": self.gclk.round(), "mclk": self.mclk.round(), "tmax": self.tmax.round(), "faults": self.faults, "mark": self.mark.unwrap_or(Mark::NoReadings).name() })
}
}
@ -454,7 +592,7 @@ pub fn prior_of(tuning: Option<&serde_json::Value>, key: &str, min_samples: u32)
if samples < min_samples.max(1) {
return None;
}
let point = Point { clock_mhz: n("clock_mhz") as u32, power_pct: (n("power_pct") as u32).clamp(50, 100) };
let point = Point { clock_mhz: n("clock_mhz") as u32, power_pct: (n("power_pct") as u32).clamp(50, 100), mem_mhz: n("mem_mhz") as u32 };
if point.power_pct == 0 && point.clock_mhz == 0 {
return None;
}
@ -606,7 +744,13 @@ impl Run {
/// The phase in words for the card row.
pub fn words(&self, now: Instant) -> String {
let what = |p: &Point| if p.clock_mhz > 0 { format!("{} MHz · {}%", p.clock_mhz, p.power_pct) } else { format!("{}%", p.power_pct) };
let what = |p: &Point| {
let mut w = if p.clock_mhz > 0 { format!("{} MHz · {}%", p.clock_mhz, p.power_pct) } else { format!("{}%", p.power_pct) };
if p.mem_mhz > 0 {
w.push_str(&format!(" · mem {}", p.mem_mhz));
}
w
};
let cur = self.current.as_ref().map(|s| what(&s.point)).unwrap_or_default();
let n = self.rows.len() + 1;
let of = self.plan.len();
@ -624,7 +768,7 @@ impl Run {
fn applied(step: &Step, rb: Readback, before_w: f64) -> bool {
let power_ok = rb.limit_w <= 0.0 || (rb.limit_w - step.watts).abs() < 1.5;
let power_changed = (step.watts - before_w).abs() >= 1.5;
if step.point.clock_mhz > 0 || !power_changed {
if step.point.clock_mhz > 0 || step.point.mem_mhz > 0 || !power_changed {
rb.acked && power_ok
} else {
rb.limit_w > 0.0 && power_ok
@ -732,6 +876,13 @@ impl Run {
Verdict::NoReadings => None,
},
PlanKind::Full => choose(&self.rows, self.plan.tolerance_pct),
PlanKind::Climb => {
if self.plan.climb.as_ref().map(|c| c.goal) == Some(Goal::MaxRate) {
self.rows.iter().filter(|r| r.usable()).max_by(|a, b| a.mhs.partial_cmp(&b.mhs).unwrap_or(std::cmp::Ordering::Equal)).cloned()
} else {
choose(&self.rows, self.plan.tolerance_pct)
}
}
}
}
@ -767,7 +918,7 @@ mod tests {
/// PC 1's RTX 5090 (nvidia-smi, 4 and 5 October 2026): default 575 W, min 400 W, max 600 W; clocks.max.gr is
/// read at the first tune (3,090 MHz is the shape used here, not a measurement).
fn l5090() -> Limits {
Limits { power_default_w: 575.0, power_min_w: 400.0, power_max_w: 600.0, clock_max_mhz: 3090, clock_min_mhz: 0 }
Limits { power_default_w: 575.0, power_min_w: 400.0, power_max_w: 600.0, clock_max_mhz: 3090, clock_min_mhz: 0, mem_default_mhz: 13801, mem_max_mhz: 14001 }
}
fn row_at(p: Point, watts: f64, mhs: f64) -> Row {
@ -776,10 +927,10 @@ mod tests {
#[test]
fn the_full_plan_is_the_power_ladder_then_the_clock_ladder_at_the_chosen_power() {
let plan = Plan::full(&l5090(), Point { clock_mhz: 0, power_pct: 80 }, 1.0);
let plan = Plan::full(&l5090(), Point { clock_mhz: 0, power_pct: 80, mem_mhz: 0 }, 1.0);
assert_eq!(plan.len(), 5 + 4, "five power steps (60% and 50% clamp to 400 W; one kept) and four clock steps");
let first = plan.next(&[]).unwrap();
assert_eq!((first.point, first.watts, first.kind), (Point { clock_mhz: 0, power_pct: 100 }, 575.0, Kind::Power));
assert_eq!((first.point, first.watts, first.kind), (Point { clock_mhz: 0, power_pct: 100, mem_mhz: 0 }, 575.0, Kind::Power));
// the power ladder: 575, 518, 460, 403, 400
let mut rows = Vec::new();
let mut watts_seen = Vec::new();
@ -794,7 +945,7 @@ mod tests {
// the clock ladder rides the chosen power point: a flat ladder ties on efficiency and rate, the lowest draw wins (100%)
let s = plan.next(&rows).unwrap();
assert_eq!(s.kind, Kind::Clock);
assert_eq!(s.point, Point { clock_mhz: 2781, power_pct: 100 });
assert_eq!(s.point, Point { clock_mhz: 2781, power_pct: 100, mem_mhz: 0 });
rows.push(row_at(s.point, 250.0, 123.8));
let s = plan.next(&rows).unwrap();
assert_eq!(s.point.clock_mhz, 2472);
@ -806,7 +957,7 @@ mod tests {
assert_eq!(plan.next(&rows), None);
// the choice: 2,472 MHz keeps 99.6% of the top rate at 220 W = 0.561 MH/W; 2,163 MHz (118 MH/s) is outside the 1% tolerance
let best = choose(&rows, 1.0).unwrap();
assert_eq!(best.point, Point { clock_mhz: 2472, power_pct: 100 });
assert_eq!(best.point, Point { clock_mhz: 2472, power_pct: 100, mem_mhz: 0 });
// a wider tolerance lets the 2,163 MHz step (0.590 MH/W, 4.8% slower) win
assert_eq!(choose(&rows, 5.0).unwrap().point.clock_mhz, 2163);
// no power limits, clocks only; no clocks, power only; nothing, empty
@ -830,7 +981,7 @@ mod tests {
#[test]
fn the_choice_keeps_the_best_mh_per_watt_within_the_rate_tolerance() {
let p = |c: u32, pct: u32| Point { clock_mhz: c, power_pct: pct };
let p = |c: u32, pct: u32| Point { clock_mhz: c, power_pct: pct, mem_mhz: 0 };
// a card that gets more efficient as the cap drops until it collapses: 70% wins (0.280 MH/W), 60% is 35% slower
let rows = vec![row_at(p(0, 100), 560.0, 124.0), row_at(p(0, 90), 510.0, 123.5), row_at(p(0, 80), 455.0, 123.2), row_at(p(0, 70), 400.0, 122.9), row_at(p(0, 60), 345.0, 80.0)];
assert_eq!(choose(&rows, 1.0).unwrap().point, p(0, 70));
@ -857,7 +1008,7 @@ mod tests {
#[test]
fn the_guards_mark_a_step_so_it_cannot_win() {
let step = Step { point: Point { clock_mhz: 2000, power_pct: 100 }, watts: 575.0, kind: Kind::Clock };
let step = Step { point: Point { clock_mhz: 2000, power_pct: 100, mem_mhz: 0 }, watts: 575.0, kind: Kind::Clock };
let good = Samples { draws: vec![250.0, 251.0, 249.0], rates: vec![123.0], gclks: vec![1998.0], mclks: vec![2505.0], tmax: 70.0, faults: 0, unapplied: false };
assert_eq!(Row::from_samples(&step, &good, 2505.0).mark, Some(Mark::Ok));
// one rejected or mismatched hash: faulted
@ -875,15 +1026,15 @@ mod tests {
assert!(!r.usable());
assert!(r.line("c", 3).contains("mark=no_readings"), "{}", r.line("c", 3));
let r = Row::from_samples(&step, &good, 0.0);
assert_eq!(r.line("nvidia-ae432dc7-1", 7), "TUNE card=nvidia-ae432dc7-1 step=7 clock=2000 cap=100 limit=575 watts=250.0 mhs=123.00 eff=0.4920 gclk=1998 mclk=2505 tmax=70 mark=ok");
assert_eq!(r.line("nvidia-ae432dc7-1", 7), "TUNE card=nvidia-ae432dc7-1 step=7 clock=2000 cap=100 mem=0 limit=575 watts=250.0 mhs=123.00 eff=0.4920 gclk=1998 mclk=2505 tmax=70 mark=ok");
}
#[test]
fn a_fault_during_a_step_reverts_it_and_the_run_goes_on() {
let t0 = Instant::now();
let timing = Timing { settle: Duration::from_secs(2), hold: Duration::from_secs(4), apply: Duration::from_secs(30) };
let limits = Limits { power_default_w: 300.0, power_min_w: 150.0, power_max_w: 300.0, clock_max_mhz: 0, clock_min_mhz: 0 };
let plan = Plan::full(&limits, Point { clock_mhz: 0, power_pct: 80 }, 1.0);
let limits = Limits { power_default_w: 300.0, power_min_w: 150.0, power_max_w: 300.0, clock_max_mhz: 0, clock_min_mhz: 0, mem_default_mhz: 0, mem_max_mhz: 0 };
let plan = Plan::full(&limits, Point { clock_mhz: 0, power_pct: 80, mem_mhz: 0 }, 1.0);
let mut run = Run::new(0, "0", "c", plan, 240.0, false, timing, t0);
let mut t = t0;
let mut limit = 240.0;
@ -932,13 +1083,13 @@ mod tests {
#[test]
fn the_confirm_plan_checks_the_prior_and_its_neighbour() {
let l = l5090();
let prior = Point { clock_mhz: 2472, power_pct: 100 };
let plan = Plan::confirm(&l, prior, Point { clock_mhz: 0, power_pct: 80 }, 1.0);
let prior = Point { clock_mhz: 2472, power_pct: 100, mem_mhz: 0 };
let plan = Plan::confirm(&l, prior, Point { clock_mhz: 0, power_pct: 80, mem_mhz: 0 }, 1.0);
assert_eq!(plan.len(), 2);
let a = plan.next(&[]).unwrap();
assert_eq!((a.point, a.kind), (prior, Kind::Confirm));
let b = plan.next(&[row_at(prior, 220.0, 123.5)]).unwrap();
assert_eq!(b.point, Point { clock_mhz: 2781, power_pct: 100 }, "one clock step up");
assert_eq!(b.point, Point { clock_mhz: 2781, power_pct: 100, mem_mhz: 0 }, "one clock step up");
// the neighbour within 1%: the prior stands
let rows = vec![row_at(prior, 220.0, 123.5), row_at(b.point, 250.0, 123.8)];
assert!(matches!(confirm_verdict(&rows, 1.0), Verdict::Keep(r) if r.point == prior));
@ -947,20 +1098,20 @@ mod tests {
assert!(matches!(confirm_verdict(&rows, 1.0), Verdict::FullDue { .. }));
assert_eq!(confirm_verdict(&[], 1.0), Verdict::NoReadings);
// an unlocked prior at 100%: the neighbour is one power step down; at the top clock the neighbour is unlocked
let plan = Plan::confirm(&l, Point { clock_mhz: 0, power_pct: 100 }, Point::default(), 1.0);
assert_eq!(plan.next(&[row_at(Point { clock_mhz: 0, power_pct: 100 }, 1.0, 1.0)]).unwrap().point, Point { clock_mhz: 0, power_pct: 90 });
let plan = Plan::confirm(&l, Point { clock_mhz: 2900, power_pct: 100 }, Point::default(), 1.0);
assert_eq!(plan.next(&[row_at(Point { clock_mhz: 2900, power_pct: 100 }, 1.0, 1.0)]).unwrap().point.clock_mhz, 0);
let plan = Plan::confirm(&l, Point { clock_mhz: 0, power_pct: 100, mem_mhz: 0 }, Point::default(), 1.0);
assert_eq!(plan.next(&[row_at(Point { clock_mhz: 0, power_pct: 100, mem_mhz: 0 }, 1.0, 1.0)]).unwrap().point, Point { clock_mhz: 0, power_pct: 90, mem_mhz: 0 });
let plan = Plan::confirm(&l, Point { clock_mhz: 2900, power_pct: 100, mem_mhz: 0 }, Point::default(), 1.0);
assert_eq!(plan.next(&[row_at(Point { clock_mhz: 2900, power_pct: 100, mem_mhz: 0 }, 1.0, 1.0)]).unwrap().point.clock_mhz, 0);
// a prior outside the vendor's range is clamped, never applied as is
let plan = Plan::confirm(&l, Point { clock_mhz: 9000, power_pct: 30 }, Point::default(), 1.0);
assert_eq!(plan.next(&[]).unwrap().point, Point { clock_mhz: 3090, power_pct: 50 });
let plan = Plan::confirm(&l, Point { clock_mhz: 9000, power_pct: 30, mem_mhz: 0 }, Point::default(), 1.0);
assert_eq!(plan.next(&[]).unwrap().point, Point { clock_mhz: 3090, power_pct: 50, mem_mhz: 0 });
}
#[test]
fn a_baseline_plan_measures_the_card_as_it_runs() {
let t0 = Instant::now();
let timing = Timing { settle: Duration::from_secs(1), hold: Duration::from_secs(2), apply: Duration::from_secs(3) };
let before = Point { clock_mhz: 0, power_pct: 80 };
let before = Point { clock_mhz: 0, power_pct: 80, mem_mhz: 0 };
let mut run = Run::new(0, "0", "c", Plan::baseline(&l5090(), before, 1.0), 460.0, false, timing, t0);
// nothing acknowledges the request (Power control is off): the apply window passes and the hold starts anyway
let mut t = t0;
@ -992,7 +1143,7 @@ mod tests {
#[test]
fn the_record_and_the_prior_round_trip_through_the_manifest_shape() {
let chosen = row_at(Point { clock_mhz: 2472, power_pct: 100 }, 220.0, 123.5);
let chosen = row_at(Point { clock_mhz: 2472, power_pct: 100, mem_mhz: 0 }, 220.0, 123.5);
let rec = record_json(1_791_230_000.0, "8f3a2c1d", "0.3.10", "windows", "NVIDIA GeForce RTX 5090", "nvidia", "581.57", "l128w16", PlanKind::Full, &[chosen.clone()], Some(&chosen), None);
assert_eq!(rec["card"], "NVIDIA_GeForce_RTX_5090");
assert_eq!(rec["key"], "NVIDIA_GeForce_RTX_5090|581|l128w16");
@ -1012,7 +1163,7 @@ mod tests {
let s = settings_of(Some(&tuning));
assert_eq!(s, Settings { enabled: true, min_samples: 5, tolerance_pct: 1.0, period_s: PERIOD_S });
let p = prior_of(Some(&tuning), "NVIDIA_GeForce_RTX_5090|581|l128w16", s.min_samples).unwrap();
assert_eq!(p.point, Point { clock_mhz: 2472, power_pct: 100 });
assert_eq!(p.point, Point { clock_mhz: 2472, power_pct: 100, mem_mhz: 0 });
assert_eq!(p.samples, 7);
assert!(prior_of(Some(&tuning), "AMD_Radeon_RX_9070_XT|32|l128w16", 5).is_none(), "3 samples are under the floor");
assert!(prior_of(Some(&tuning), "AMD_Radeon_RX_9070_XT|32|l128w16", 3).is_some());

View file

@ -2239,16 +2239,19 @@ impl Engine {
let current = if c.power_limit_w > 0.0 { c.power_limit_w } else { c.power_default_w }.round() as u64;
let known_direct = self.sweep_direct;
std::thread::spawn(move || {
let q = crate::detect::run_timeout(std::process::Command::new(&smi).args(["-i", &device, "--query-gpu=clocks.max.gr,driver_version", "--format=csv,noheader,nounits"]), None, Duration::from_secs(10)).unwrap_or_default();
let q = crate::detect::run_timeout(std::process::Command::new(&smi).args(["-i", &device, "--query-gpu=clocks.max.gr,driver_version,clocks.mem,clocks.max.mem", "--format=csv,noheader,nounits"]), None, Duration::from_secs(10)).unwrap_or_default();
let p: Vec<&str> = q.trim().split(',').map(|s| s.trim()).collect();
let clock_max = p.first().and_then(|s| s.parse::<f64>().ok()).unwrap_or(0.0) as u32;
let driver = p.get(1).map(|s| s.to_string()).unwrap_or_default();
// Ember 2: the memory clock now (the default under load) and the vendor's maximum
let mem_default = p.get(2).and_then(|s| s.parse::<f64>().ok()).unwrap_or(0.0) as u32;
let mem_max = p.get(3).and_then(|s| s.parse::<f64>().ok()).unwrap_or(0.0) as u32;
let direct = match known_direct {
Some(d) => Some(d),
None if allowed || current > 0 => crate::detect::run_timeout(std::process::Command::new(&smi).args(["-i", &device, "-pl", &current.to_string()]), None, Duration::from_secs(20)).map(|out| out.contains("All done")),
None => Some(false),
};
shared.send(Cmd::TuneProbe(idx, Ok(TuneProbe { clock_max_mhz: clock_max, clock_min_mhz: 0, driver, direct: direct.unwrap_or(false), amd_ordinal: -1, ..Default::default() })));
shared.send(Cmd::TuneProbe(idx, Ok(TuneProbe { clock_max_mhz: clock_max, clock_min_mhz: 0, driver, direct: direct.unwrap_or(false), amd_ordinal: -1, mem_default_mhz: mem_default, mem_max_mhz: mem_max, ..Default::default() })));
});
}
"amd" => {
@ -2267,7 +2270,7 @@ impl Engine {
// the stock clock, not MHz; a range with a negative floor is an offset range and the clock
// knob stays closed until the stock clock is known (the power limit is the AMD lever), and
// `plimit_range -30 10` bounds the power ladder (the percent scale rides power_* below)
Some(t) if t.ok => TuneProbe { clock_max_mhz: if t.gmax_min >= 0.0 && t.gmax_max > 0.0 { t.gmax_max as u32 } else { 0 }, clock_min_mhz: if t.gmax_min > 0.0 { t.gmax_min as u32 } else { 0 }, driver, direct: true, amd_ordinal: t.ordinal as i64, plimit_min: t.plimit_min, plimit_max: t.plimit_max },
Some(t) if t.ok => TuneProbe { clock_max_mhz: if t.gmax_min >= 0.0 && t.gmax_max > 0.0 { t.gmax_max as u32 } else { 0 }, clock_min_mhz: if t.gmax_min > 0.0 { t.gmax_min as u32 } else { 0 }, driver, direct: true, amd_ordinal: t.ordinal as i64, plimit_min: t.plimit_min, plimit_max: t.plimit_max, mem_default_mhz: 0, mem_max_mhz: 0 },
Some(t) => TuneProbe { clock_max_mhz: 0, clock_min_mhz: 0, driver, direct: false, amd_ordinal: t.ordinal as i64, ..Default::default() },
None => TuneProbe { clock_max_mhz: 0, clock_min_mhz: 0, driver, direct: false, amd_ordinal: -1, ..Default::default() },
})));
@ -2305,9 +2308,9 @@ impl Engine {
// AMD's power limit is a percent offset from the default (ADLX): the plan's watts scale becomes a percent
// scale (default 100, floor 100 + plimit_min, ceiling 100 + plimit_max), tune_apply sends pct - 100
let limits = if c.vendor == "amd" && probe.direct && probe.plimit_max >= probe.plimit_min && probe.plimit_min > -100.0 {
crate::ember::Limits { power_default_w: 100.0, power_min_w: 100.0 + probe.plimit_min, power_max_w: 100.0 + probe.plimit_max, clock_max_mhz: probe.clock_max_mhz, clock_min_mhz: probe.clock_min_mhz }
crate::ember::Limits { power_default_w: 100.0, power_min_w: 100.0 + probe.plimit_min, power_max_w: 100.0 + probe.plimit_max, clock_max_mhz: probe.clock_max_mhz, clock_min_mhz: probe.clock_min_mhz, mem_default_mhz: 0, mem_max_mhz: 0 }
} else {
crate::ember::Limits { power_default_w: c.power_default_w, power_min_w: c.power_min_w, power_max_w: c.power_max_w, clock_max_mhz: probe.clock_max_mhz, clock_min_mhz: probe.clock_min_mhz }
crate::ember::Limits { power_default_w: c.power_default_w, power_min_w: c.power_min_w, power_max_w: c.power_max_w, clock_max_mhz: probe.clock_max_mhz, clock_min_mhz: probe.clock_min_mhz, mem_default_mhz: probe.mem_default_mhz, mem_max_mhz: probe.mem_max_mhz }
};
let control = match c.vendor.as_str() {
"nvidia" => crate::ember::control_reason("nvidia", &limits, &c.device, power_control, false),
@ -2316,6 +2319,26 @@ impl Engine {
};
if c.vendor == "nvidia" {
self.sweep_direct = Some(probe.direct);
// the approved step that made this engine elevated registers the Igneum Power Helper here too (run 5,
// 6 October 2026: a --sweep engine skipped the cap path where the registration lived, so the one
// approved click registered nothing); no prompt: this process already holds the rights
if probe.direct && cfg!(windows) && !self.power_task_registered() {
let dir = self.sweep_dir();
let _ = std::fs::create_dir_all(&dir);
let script = dir.join("register-power-task.ps1");
let installed: Vec<PathBuf> = [std::env::var_os("LOCALAPPDATA").map(|l| PathBuf::from(l).join("Programs").join("Igneum Miner").join("igneum-app.exe")), std::env::var_os("ProgramFiles").map(|p| PathBuf::from(p).join("Igneum Miner").join("igneum-app.exe"))].into_iter().flatten().collect();
if let Ok(exe) = std::env::current_exe() {
let target = crate::powertask::task_exe(&exe, &installed);
if std::fs::write(&script, [b"\xEF\xBB\xBF".as_slice(), crate::powertask::register_script(&target).as_bytes()].concat()).is_ok() {
let mut p = std::process::Command::new(crate::platform::tool("powershell"));
p.args(["-NoProfile", "-ExecutionPolicy", "Bypass", "-File", &script.display().to_string()]);
crate::platform::quiet(&mut p);
let ok = p.status().map(|s| s.success()).unwrap_or(false);
self.power_task = None;
self.sweep_say(&format!("TUNE helper registered={} action={}", ok, target.display()));
}
}
}
}
{
let mut st = self.st();
@ -2333,7 +2356,7 @@ impl Engine {
}
let tuning = self.tuning_object();
let ember = crate::ember::settings_of(tuning.as_ref());
let before = crate::ember::Point { clock_mhz: c.clock_cap_mhz, power_pct: if c.power_pct == 0 { 80 } else { c.power_pct } };
let before = crate::ember::Point { clock_mhz: c.clock_cap_mhz, power_pct: if c.power_pct == 0 { 80 } else { c.power_pct }, mem_mhz: c.mem_cap_mhz };
let before_w = if c.power_limit_w > 0.0 { c.power_limit_w } else { requested_watts(&c) };
let key = crate::ember::prior_key(&c.name, &probe.driver, &c.program_class);
let prior = crate::ember::prior_of(tuning.as_ref(), &key, ember.min_samples);
@ -2386,6 +2409,7 @@ impl Engine {
crate::ember::PlanKind::Full => format!("{steps} steps over the power limit and the core clock, {} s each on the live program", (run.timing.settle + run.timing.hold).as_secs()),
crate::ember::PlanKind::Confirm => format!("the fleet prior ({} samples) and one neighbour, {} s each", prior.as_ref().map(|p| p.samples).unwrap_or(0), (run.timing.settle + run.timing.hold).as_secs()),
crate::ember::PlanKind::Baseline => format!("measuring the card as it runs ({})", control.clone().unwrap_or_default()),
crate::ember::PlanKind::Climb => format!("the hill-climb: memory up, core down, up to {steps} probes of {} s", (run.timing.settle + run.timing.hold).as_secs()),
};
self.shared.event("info", &format!("{}: tuning started: {what}", c.name));
self.sweep = Some(run);
@ -2725,7 +2749,7 @@ impl Engine {
_ => self.shared.event("ok", &format!("{name}: {line}, {} held", point_words(&row.point))),
}
if pinned && kind != crate::ember::PlanKind::Baseline {
let p = self.st().mining.cards.get(idx).map(|c| crate::ember::Point { clock_mhz: c.clock_cap_mhz, power_pct: c.power_pct }).unwrap_or(run.before);
let p = self.st().mining.cards.get(idx).map(|c| crate::ember::Point { clock_mhz: c.clock_cap_mhz, power_pct: c.power_pct, mem_mhz: c.mem_cap_mhz }).unwrap_or(run.before);
let w = self.st().mining.cards.get(idx).map(requested_watts).unwrap_or(run.before_w);
self.tune_apply(idx, &run.device, &crate::ember::Step { point: p, watts: w, kind: crate::ember::Kind::Confirm });
}
@ -2753,7 +2777,7 @@ impl Engine {
(Some(r), _) => (r.card, r.label.clone(), r.device.clone(), r.before, r.before_w, r.forced, r.seq > 0 && r.plan.kind != crate::ember::PlanKind::Baseline),
(None, Some((idx, forced))) => {
let c = self.st().mining.cards.get(idx).cloned();
let (label, device, w, p) = c.map(|c| (format!("card-{idx}"), c.device.clone(), if c.power_limit_w > 0.0 { c.power_limit_w } else { requested_watts(&c) }, crate::ember::Point { clock_mhz: c.clock_cap_mhz, power_pct: c.power_pct })).unwrap_or_default();
let (label, device, w, p) = c.map(|c| (format!("card-{idx}"), c.device.clone(), if c.power_limit_w > 0.0 { c.power_limit_w } else { requested_watts(&c) }, crate::ember::Point { clock_mhz: c.clock_cap_mhz, power_pct: c.power_pct, mem_mhz: c.mem_cap_mhz })).unwrap_or_default();
(idx, label, device, p, w, forced, false)
}
_ => return,
@ -4030,6 +4054,9 @@ pub struct TuneProbe {
/// AMD: the power offset range in percent from the `tune` line (PC 1's 9070 XT: -30 to 10)
pub plimit_min: f64,
pub plimit_max: f64,
/// Ember 2, NVIDIA: the memory clock under load and the vendor's maximum (nvidia-smi clocks.mem, clocks.max.mem)
pub mem_default_mhz: u32,
pub mem_max_mhz: u32,
}
/// One `tune` line of igneum-gpu-telemetry --tune:

View file

@ -118,6 +118,7 @@ pub struct CardState {
pub clock_min_mhz: u32, // the vendor's floor for a cap (0 = 60% of the maximum)
pub gclk_mhz: f64, // core clock now
pub clock_cap_mhz: u32, // the cap in force (0 = unlocked)
pub mem_cap_mhz: u32, // Ember 2: the memory clock set by the tune (0 = the driver's default)
pub amd_ordinal: i64, // the `amd N` ordinal of igneum-gpu-telemetry (-1 = unknown)
pub driver: String, // the driver version (nvidia-smi, or the worker's race line)
pub program_class: String, // the program class of the race line (loads and wide loads per hash); "" = unknown

View file

@ -1669,6 +1669,18 @@ Branch `ember-tune` (54ff1bc), docs/plans/ember-tune.md. Every card tuned for MH
Nothing set; the installed app's miners back after 350 s. Why every earlier run (5 and 6 October, runs 1 to 4 and dry runs 1 and 2) read its copied settings as defaults, measured on PC 1 (collect ember-acl-2): the engine's own start locks its app folder with `icacls /inheritance:r /grant:r <user>:F`; cutting the folder's inheritance propagates down, the non-inheritable grant gives the children nothing, so a file COPIED in before the start (settings.json, machine-id, wallet.json) is left with no access entry and its owner cannot read it (`ReadAllText`: access denied), while the engine's own files written after the lock inherit fine, which hid it for a day. A first fix with `(OI)(CI)F /T` left the file empty too: `/T` re-applies `/inheritance:r` to each file after the propagation and an `(OI)(CI)` entry on a file is inherit-only. The right form is the inheritable grant without `/T` (07d5a72). Consequence for every tier on Windows: nothing changes for the installed app (its files were always its own); any tool that drops files into the app folder before the app starts (an installer's seed, a migration, a support script) was unreadable to the app until now and is readable from 0.3.13 on.
**Run 5, 6 October 2026, 15:28 to 15:39Z (job ember-tune-pc1-5, elevated on the project lead's click, PC 1 on 0.3.13, the tune engine = kit ember-kit-5 from 07d5a72, mode=direct):** the first run that set limits. The 5090's power ladder, 75 s a step, the clock unlocked (2,850 MHz core, 13,801 MHz memory), the rate = the worker's STATUS wall rate, the draw = nvidia-smi every 5 s:
| Cap | Limit | MH/s | W | MH/W | GPU C |
|---|---|---|---|---|---|
| 100% | 575 W | 127.38 | 309.9 | 0.411 | 64 |
| 90% | 518 W | 99.32 | 313.6 | 0.317 | 65 |
| 80% | 460 W | 123.11 | 312.2 | 0.394 | 65 |
| 70% | 403 W | 127.38 | 310.9 | 0.410 | 65 |
| 60% (floor 400 W) | 400 W | 127.38 | 311.3 | 0.409 | 65 |
Reading: the cap does not bind on this hash (310 to 314 W under every limit, as the 4 October stability line said), so the power knob is flat at 0.41 MH/W on the 5090 and the saving must come from the clocks; the 90% and 80% rows' rate dips at the same draw are stalls inside those holds (a worker restart or a template wait), not the cap. The clock ladder's first step (2,781 MHz at 575 W) was requested at 15:37:47Z and never measured: the playbook's own watchdog killed the live engine at 15:39:03Z (all three cards mining at 177 MH/s) because its idle clause sampled one log line and read "idle" from a missing match; the 4070's and the 9070 XT's plans never ran. Consequences: the 5090 is probably left with its core clock locked at 2,781 MHz (an `-lgc` lock persists until `-rgc` or a reboot) under the 575 W cap, which costs little rate; freeing it needs administrator rights; and the Power Helper was NOT registered by this run (the registration lived only in the installed app's cap path, which a `--sweep` engine skips). Fixed the same hour: the watchdog's idle rule (three consecutive status lines reading 0.00 MH/s and 300 s, never a missing match), the `after` snapshot and the engine-log dump on every exit, and an elevated tune engine registering the task itself before its first step. The decided way out: the project lead switches Power control ON in the 0.3.13 app (its one prompt registers the task from the install folder), a job frees the clock through the task (`rgc`), the tune runs unelevated through the task.
Consequence for the tiers: an AMD card is tuned on its power limit alone until its stock core clock is read (a 9070 XT at -30% is the floor the driver allows, 4 steps, 5 minutes); every NVIDIA card's two-knob plan waits on the user's one click on Power control; the re-run on PC 1 is held until the quit's source is named (the event-log collect) and follows the 0.3.11 rollout (the update clears the jobs folder, so the engine and the helper are fetched again), with the scheduler's slot.
## 5 October 2026 (night), read width of the lottery hash: 4, 16 and 64-byte loads, a per-load mix, a written scratch; three cards (gate 1 experiment, cryptographer)

View file

@ -150,6 +150,9 @@ function Forward([string] $line) {
# its budget silently again
$firstStatusAt = $null
$lastMining = $null
$idleSince = $null
$idleSamples = 0
$lastIdleStatus = ''
$lastEngineLine = ''
function EngineLogDump([string] $why) {
Write-Output ('===== engine log tail (' + $why + ')')
@ -157,29 +160,32 @@ function EngineLogDump([string] $why) {
if ($t) { Get-Content -LiteralPath $t.FullName -ErrorAction SilentlyContinue | Where-Object { $_ -notmatch 'status: accepted 0 blocks' } | Select-Object -Last 80 | ForEach-Object { Write-Output (' ' + $_) } } else { Write-Output ' (no app-*.log in the scratch logs folder)' }
if (Test-Path -LiteralPath $errFile) { Write-Output '===== engine stderr'; Get-Content -LiteralPath $errFile -ErrorAction SilentlyContinue | Select-Object -Last 20 | ForEach-Object { Write-Output (' ' + $_) } }
}
# the newest `status:` line among the engine log's last 60 lines (run 5, 15:39Z: a single sampled tail line missed the
# status lines between TUNE and telemetry lines for 300 s and the watchdog killed a live three-card tune)
function EngineStatus() { $t = Get-ChildItem -Path $sLogs -Filter 'app-*.log' -ErrorAction SilentlyContinue | Sort-Object LastWriteTime -Descending | Select-Object -First 1; if ($t) { $ls = @(Get-Content -LiteralPath $t.FullName -Tail 60 -ErrorAction SilentlyContinue); for ($i = $ls.Count - 1; $i -ge 0; $i--) { if ([string]$ls[$i] -match ' status: ') { return [string]$ls[$i] } } }; return '' }
function EngineTail() { $t = Get-ChildItem -Path $sLogs -Filter 'app-*.log' -ErrorAction SilentlyContinue | Sort-Object LastWriteTime -Descending | Select-Object -First 1; if ($t) { $l = Get-Content -LiteralPath $t.FullName -Tail 1 -ErrorAction SilentlyContinue; if ($l) { return [string]$l } }; return '' }
function Exit3([string] $why, [string] $reason) {
Write-Output ('RESULT TUNE error=' + $reason + ' last_status_line=' + ((EngineStatus) -replace '\s+', '_') + ' last_log_line=' + ($lastEngineLine -replace '\s+', '_'))
EngineLogDump $why
EndTree $p.Id $why
Snapshot 'after'
Write-Output 'RESULT TUNE error=no_rows'
exit 3
}
while (-not $p.HasExited) {
Start-Sleep -Seconds 5
$tailLine = EngineTail
if ($tailLine) { $lastEngineLine = $tailLine }
if ($lastEngineLine -match ' status: ') {
$statusLine = EngineStatus
if ($statusLine) {
if (-not $firstStatusAt) { $firstStatusAt = Get-Date }
if ($lastEngineLine -match ', mining \|' -or ($lastEngineLine -match '(\d+\.\d+) MH/s' -and [double]$Matches[1] -gt 0)) { $lastMining = Get-Date }
}
if ($firstStatusAt -and -not $lastMining -and ((Get-Date) - $firstStatusAt).TotalSeconds -gt 120) {
Write-Output ('RESULT TUNE error=not_mining reason=no_card_mined_within_120_s_of_the_first_status_line last_log_line=' + ($lastEngineLine -replace '\s+', '_'))
EngineLogDump 'watchdog: not mining'
EndTree $p.Id 'watchdog: not mining'
Write-Output 'RESULT TUNE error=no_rows'
exit 3
}
if ($lastMining -and ((Get-Date) - $lastMining).TotalSeconds -gt 300) {
Write-Output ('RESULT TUNE error=stopped_mining reason=every_card_idle_for_300_s last_log_line=' + ($lastEngineLine -replace '\s+', '_'))
EngineLogDump 'watchdog: stopped mining'
EndTree $p.Id 'watchdog: stopped mining'
Write-Output 'RESULT TUNE error=no_rows'
exit 3
$rate = 0.0; if ($statusLine -match '(\d+\.\d+) MH/s') { $rate = [double]::Parse($Matches[1], [Globalization.CultureInfo]::InvariantCulture) }
# idle = three consecutive 10-s samples whose newest status line reads 0.00 MH/s (never a missing match)
if ($rate -gt 0) { $lastMining = Get-Date; $idleSamples = 0; $idleSince = $null }
elseif ($statusLine -ne $lastIdleStatus) { $lastIdleStatus = $statusLine; $idleSamples++; if (-not $idleSince) { $idleSince = Get-Date } }
}
if ($firstStatusAt -and -not $lastMining -and ((Get-Date) - $firstStatusAt).TotalSeconds -gt 120) { Exit3 'watchdog: not mining' 'not_mining reason=no_card_mined_within_120_s_of_the_first_status_line' }
if ($lastMining -and $idleSamples -ge 3 -and $idleSince -and ((Get-Date) - $idleSince).TotalSeconds -gt 300) { Exit3 'watchdog: stopped mining' 'stopped_mining reason=three_status_lines_read_0.00_MH/s_and_300_s_passed' }
$all = @(); if (Test-Path -LiteralPath $outFile) { $all = @(Get-Content -LiteralPath $outFile -ErrorAction SilentlyContinue) }
while ($seen -lt $all.Count) {
$l = [string]$all[$seen]; $seen++
@ -203,6 +209,7 @@ while (-not $p.HasExited) {
} else { Write-Output ('RESULT TUNE quit not sent: no URL file at ' + $u + '; killing pid ' + $p.Id) }
Start-Sleep -Seconds 20
if (-not $p.HasExited) { EndTree $p.Id 'budget' }
EngineLogDump 'budget'
Write-Output 'RESULT TUNE error=budget_exceeded'
}
}