diff --git a/.github/workflows/ci-red.yml b/.github/workflows/ci-red.yml index dddb758b..48b3deff 100644 --- a/.github/workflows/ci-red.yml +++ b/.github/workflows/ci-red.yml @@ -17,7 +17,10 @@ jobs: red: name: red watcher (every branch; one line per failed run, with the branch, commit, red check and pushing author, to the updates channel and the box file) if: ${{ github.event.workflow_run.conclusion == 'failure' }} - runs-on: [self-hosted, linux, x64, igneum-build-1] + # the label ci-red is on igneum-build-1 only (added through the runners API on 7 October 2026; the default of + # RUNNER_LABELS in provision.sh carries it): the record file and the poster (igneum-ci-red.timer, the webhook file) + # live on that box, and the pool label igneum-build-1 is shared with igneum-build-2 since the same day + runs-on: [self-hosted, linux, x64, ci-red] timeout-minutes: 5 permissions: actions: read # the failed run's jobs API (the first real red run, 21:19Z on 6 October: the default token answered 403 and the line carried no step) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 5311c6dc..755e799e 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -8,7 +8,9 @@ # local gate and CI cannot drift (6 October 2026: 131 red `ci` runs in three days, 92 of them on master, every one a # tree check that would have failed on the pushing machine in under 25 s; docs/analysis/ci-failures-2026-10-06.md). # -# Where it runs: `pow` and `sims` go to the box's runner (igneum-build-1, rustc pinned, sccache read-only, 48 jobs) +# Where it runs: `pow` and `sims` go to the self-hosted pool (label igneum-build-1: the runners on igneum-build-1 and, since +# 7 October 2026, igneum-build-2, which carries that label too; rustc pinned, sccache read-only) and only when the push +# touched code (the `changes` job; a docs-only push skips them) # when the repository variable IGNEUM_CI_RUNNER is `box`, else to ubuntu-latest (docs/plans/ci-self-hosted.md; GitHub # has no fallback in runs-on, the variable is the switch). The `site` job stays on GitHub's machines. The red watcher # is its own workflow, .github/workflows/ci-red.yml (workflow_run, so the copy on master watches every branch's run @@ -26,8 +28,39 @@ on: push: pull_request: jobs: + changes: + # What the push touched (tools/ci/docs-only-check.sh): a push of documents only (docs/, site/, *.md) skips the two + # compile-or-compute jobs below, which read none of those paths, so the self-hosted queue carries only runs that can + # change their result (7 October 2026: 31 runs queued on one runner, most of them status-document pushes). The tree + # gate (the `site` job) runs on ubuntu-latest for every push. A pull request, a new branch or a force push answers + # code=true (no `before` to compare from), as does any error reading the compare API: when in doubt, run. + name: what the push touched (docs-only runs skip the Rust and simulator jobs) + runs-on: ubuntu-latest + outputs: + code: ${{ steps.classify.outputs.code }} + steps: + - uses: actions/checkout@v4 + with: + sparse-checkout: tools/ci + - id: classify + env: + GH_TOKEN: ${{ github.token }} + BEFORE: ${{ github.event.before }} + AFTER: ${{ github.sha }} + REPO: ${{ github.repository }} + EVENT: ${{ github.event_name }} + run: | + if [ "$EVENT" != push ] || [ -z "$BEFORE" ] || [ "$BEFORE" = 0000000000000000000000000000000000000000 ]; then + echo "code=true" >> "$GITHUB_OUTPUT"; echo "no base to compare from ($EVENT): the compile jobs run"; exit 0 + fi + files="$(gh api "repos/$REPO/compare/$BEFORE...$AFTER" --paginate --jq '.files[].filename' 2>/dev/null || true)" + line="$(printf '%s\n' "$files" | bash tools/ci/docs-only-check.sh)" + echo "$line" >> "$GITHUB_OUTPUT" + echo "$line: $(printf '%s\n' "$files" | grep -c .) changed path(s) between ${BEFORE:0:8} and ${AFTER:0:8}" pow: name: igneum-pow tests, igneum-census build + needs: changes + if: ${{ needs.changes.outputs.code == 'true' }} runs-on: ${{ vars.IGNEUM_CI_RUNNER == 'box' && fromJSON('["self-hosted", "linux", "x64", "igneum-build-1"]') || 'ubuntu-latest' }} steps: - uses: actions/checkout@v4 @@ -43,6 +76,10 @@ jobs: run: cargo build --release sims: name: simulators, quick modes + needs: changes + # master and release-* pushes, and pull requests into them, only (main, 7 October 2026: every code push cost two box jobs and the + # queue read 22); a feature-branch code push runs the igneum-pow tests alone. tools/ci/sims-branch-check.sh holds this rule. + if: ${{ needs.changes.outputs.code == 'true' && ((github.event_name == 'push' && (github.ref == 'refs/heads/master' || startsWith(github.ref, 'refs/heads/release-'))) || (github.event_name == 'pull_request' && (github.base_ref == 'master' || startsWith(github.base_ref, 'release-')))) }} runs-on: ${{ vars.IGNEUM_CI_RUNNER == 'box' && fromJSON('["self-hosted", "linux", "x64", "igneum-build-1"]') || 'ubuntu-latest' }} steps: - uses: actions/checkout@v4 diff --git a/brand/marks/vendor-marks.mjs b/brand/marks/vendor-marks.mjs new file mode 100644 index 00000000..08de8b84 --- /dev/null +++ b/brand/marks/vendor-marks.mjs @@ -0,0 +1,65 @@ +// Igneum vendor and OS marks (gpu-logos, 7 October 2026): the strings the miner app ships in app/igneum-app/ui/app.js +// (View.MARKS and View.VENDORS), exported verbatim for the site and anything else that names the hardware. +// view.test.mjs fails when this file and app.js drift apart, so edit app.js first and regenerate this file with +// `node brand/marks/regen.mjs` (or copy the strings by hand; the test says which one moved). +// +// Each glyph: a hand-drawn simplified monochrome mark of the vendor's public geometry (never a copied logo file, +// never a raster), 24 x 24 viewBox at 22 px, under 460 bytes, fill or stroke through currentColor so the element's +// colour tints it. Nominative use that names the hardware; the ember accent is for state and never tints a brand. +// +// The treatment in the app (app.css, the block at the end): a 44 x 44 well, radius 12, background the vendor colour +// at .14 alpha (dark) or .10 (light), a 1 px ring in the vendor colour at .45 alpha, the glyph in the full colour; +// hover and focus-within add a 3 px halo of the well colour; nothing animates. The light hex of every vendor reads at +// 3:1 or better on its well over white (nvidia 3.87, amd 5.01, intel 4.29, apple 7.52, gpu 4.65). +// +// Class names in the app: .badge. (nvidia | amd | intel | apple | gpu), .badge.mini for a 26 px inline mark, +// .gen for the series line under the name. Tokens: --mark-, --mark--well, --mark--ring. + +export const VENDORS = { + "nvidia": { + "label": "NVIDIA", + "dark": "#8BE37A", + "light": "#2F8A22" + }, + "amd": { + "label": "AMD Radeon", + "dark": "#FF5A5A", + "light": "#C41E2A" + }, + "intel": { + "label": "Intel", + "dark": "#7CC4FF", + "light": "#1C6FD6" + }, + "apple": { + "label": "Apple", + "dark": "#E6E3DD", + "light": "#4A4A50" + }, + "gpu": { + "label": "GPU", + "dark": "#9A9A9E", + "light": "#6B6B70" + } +}; +export const WELL_ALPHA = { dark: 0.14, light: 0.1 }; +export const MARKS = { + "nvidia": "", + "amd": "", + "intel": "", + "apple": "", + "gpu": "" +}; +// OS marks in the same treatment: Apple is the vendor glyph; Windows is the four slanted panes +export const OS_MARKS = { + macos: MARKS.apple, + windows: "" +}; +export const OS_COLOURS = { macos: VENDORS.apple, windows: { label: 'Windows', dark: '#7CC4FF', light: '#1C6FD6' } }; +// the CSS tokens for both themes, as the app declares them +export function tokensCss() { + const line = (theme) => Object.entries(VENDORS).map(([v, c]) => { const h = c[theme], r = parseInt(h.slice(1, 3), 16), g = parseInt(h.slice(3, 5), 16), b = parseInt(h.slice(5, 7), 16), a = String(WELL_ALPHA[theme]).replace(/^0/, ''); return `--mark-${v}:${h};--mark-${v}-well:rgba(${r},${g},${b},${a});--mark-${v}-ring:rgba(${r},${g},${b},.45)`; }).join(';'); + return `:root{${line('dark')}}\n@media (prefers-color-scheme:light){:root:not([data-theme="dark"]){${line('light')}}}\n:root[data-theme="light"]{${line('light')}}`; +} +// the well:
+export function markHtml(vendor, size) { const v = MARKS[vendor] ? vendor : 'gpu'; return '
' + MARKS[v] + '
'; } diff --git a/docs/plans/build-server.md b/docs/plans/build-server.md index 24c5ca04..a7e907d5 100644 --- a/docs/plans/build-server.md +++ b/docs/plans/build-server.md @@ -377,3 +377,28 @@ spill-over. Now (lib.sh `bs_route_spill`, master from this commit): them keeps its number; since this commit a bounded run takes the band its slot owns (slot 0 the last 32 cores, slot 1 the 32 below, slot 2 the 32 below that), so three bounded runs never share a core. A gate still takes its slot ahead of queued suites. - `--box N` still pins. Self-test: tools/ci/route-spill-check.sh (thirteen cases through `BS_ROUTE_STATE_`, no ssh), in the gate. + +## 8. The measure file is retired: per-core leases and the quiet class (7 October 2026, 15:07 UK) + +Main's reading at 15:07 UK: on build-1 a sync-fuzz probe (the capacity lane's, SIGSTOPped since 09:45Z, no owner) held the global +measure flock for five and a half hours; beside it attack-f6's phase2b waited exclusive on the same file behind seven F2 solvers +holding it shared on cores 6-11,54-59, and every new shared taker (every build) queued behind the exclusive waiter: load 120, slots +free, nine waiting. On build-2 the era VDF bench held the file exclusive while pinned to one core and five jobs waited. One global +exclusive lock across unrelated measurements was the wrong design. Now: + +- `infra/build-server/lease.sh`, installed on every box at `/srv/builds/_bin/lease` (provision.sh; by hand on build-1 and build-2 + at 14:2x BST): `lease cores --label "..." [--owner ] -- ` takes a lease on THOSE CORES ONLY (one flock per + core, `_locks/core-`, a `wait-` file with the label while it waits), runs the command under nice 10 and taskset, and + releases. Nothing else is excluded. `lease quiet --label "..." --owner -- ` is the whole-box class: refused (exit + 73) while any slot or core lease is held, capped at 20 minutes, holder line with the owner in `_locks/quiet`. `lease status`, + `lease reap`. +- remote-run.sh: an unbounded run (nice 0, the full set) takes `quiet` shared and waits for it; a bounded run (suites, benches, + everything on box 2) never takes it. Every run keeps off leased cores (its set minus the `core-` flocks, said once). A run's + keeper refreshes its holder file's mtime every 20 s and calls `lease reap`: a holder of a lease, the quiet file or a slot whose + process has been STOPPED for 5 minutes is killed and its file cleared, one line each in `_log/reaped.log`. The keeper closes the + lock descriptors it inherits (an orphaned `sleep 20` held a slot and a worktree lock 20 s past the release). BR_MEASURE=1 is the + quiet class with the same refusals; it needs a named owner (IGNEUM_AGENT). +- The lanes' own `flock -s /srv/builds/_locks/measure -c "nice -n 10 taskset -c ..."` lines no longer hold anything a build + waits for; they become `/srv/builds/_bin/lease cores --label "..." -- `, and `flock -x .../measure` becomes + `lease quiet`. Self-tests: lease.sh --self-test and remote-run.sh --self-test-slots, run on build-1 by tools/ci/box-locks-check.sh + in the gate. diff --git a/docs/plans/counter-asic-3-status.md b/docs/plans/counter-asic-3-status.md index ca109c25..2ac31041 100644 --- a/docs/plans/counter-asic-3-status.md +++ b/docs/plans/counter-asic-3-status.md @@ -367,7 +367,37 @@ Three residual classes, all a constant delivered through a writer the rule admit | era-4 | dd8fdf6ff4f59eed | | era-5 | 8bf40f5cb858d835 | -Packs zip (eight packs, packs-ca3-v4-sub2) sha256 69c36772cd79e44e2ddd589466d9c64a94a13c9e970e9f27bd76feabb9b4581b. The suite re-runs through master's build-remote on box 2 (the worktree's own script predates --box; the first run died on the flag), line to follow; G1 on PC 2 under --cards-off after it. The sub-version-2 pairing waits on the node lane's re-pin to a788661687db4bb3 and byte 7. For the record, sub-version 1's pairing: igneum-pow 8c728ca3 against b7cc37e7 (8097d600's assert_ne in) 17 passed, 0 failed, rc 0, 12:21Z. 0.3.20's CUT SET AS IT STANDS (the shipper, 14:1x UK): pin c4459193, igneum-pow 8c728ca3 at object byte 5 (sub-version 1), the floor-moved file publishing with it (the project lead's word; the floor from the live DAA at the publish plus 604,800, the digest read on c4459193), publish about 15:15Z (16:15 BST) on CASES END, the sweep from then with PC 1 first. 0.3.21's clock tonight: the node lane stages release-0.3.21-node at the shipper's sweep-end word (about 15:45Z, 16:45 BST) with the sub-version-2 re-pin (07a809a7, byte 7, id a788661687db4bb3) as its own commit, held pending the F8 census on 07a809a7; the census's clock about 14:00Z (15:00 BST) by the attack-pass lane's within-the-hour line from 13:01Z; the pass line every one of the 64 seeds under 1.2x of the window model on the chain path. If the census fails or slips past 19:00Z (20:00 BST), main's standing ruling applies (nothing on sub-version 2 is proposed until the census is green): 0.3.21's node ships byte 5 again with the re-pin dropped and the rest of its line kept. CI NOTE (13:1x UTC): master's ci runs since 12dc5c97 (nineteen of mine) sit queued behind one self-hosted runner (igneum-build-1, busy; 31 queued across branches, one in progress); the last completed master runs (439a233f to 5f990a09) are success; no red exists, the conclusions are unread until the queue drains. MAIN'S WORD (14:2x UK): the floor move is the project lead's word already and ships with 0.3.20 at the cut; the CI lever is a second self-hosted runner on build-2 plus ubuntu-latest for docs-only pushes, ordered to the CI lane; the rule reads "own a red when the conclusion lands", never holding pushes; the census verdict about 14:00Z (15:00 UK) decides 0.3.21's byte. SUB-VERSION 2's STATIC CENSUS (the hash lane, tools/ca3-v4-uniform on box 2, 13:12Z, 1,024 chain-shaped seeds plus F8's p1 to p3): 0 lossy-sourced load sites of 16,432 (14,329 injecting, 2,103 bijective); 0 programs with an or-, mul- or mulhi-sourced load; the no-era draw path gives the devnet epoch-0 seed the pack's own id a788661687db4bb3, so every draw path reads one stream. Cost of (a') and (c'): 1.99 attempts per seed on average against 0.05 before (p2's seed five), about 2 ms of generation per rejected attempt on one core; nothing a miner or node notices. Ledger entry 715f14be. Master 36e08c80 merged into ca3-v4-amend as f5244ad7 (igneum-pow untouched, re-export 0 differing files), pushed with the gate GREEN, its CI runs queued; the box-2 suite runs through the merged tools (the first attempt died on test initialisers missing the (c') field, fixed in the same push; library and packs unaffected). G1 BLOCKED: PC 2 has not picked up fetch-ca3-v4-sub2-20261007 and run-ca3-v4-sub2-g1-pc2-20261007 (published 13:04:02Z, signature OK); the intake shows nothing from 1ccfe586 since job-update-now-0319 at 10:34:21Z; the PC 2 lock releases when the job closes or in 30 minutes; the run is republished when the app reports. PC 2 READS SILENT on the console (the shipper, 14:4x UK): last seen 2 h ago, app 0.3.19 on node 5899f603, its last line the 0.3.19 update-now at 10:34:21Z; the app went down or stopped polling on that update (the UI lane's update-now; PC 1 took the same update and reports). The build-server lane's PC 2 kept-datadir job (run-20261007-125433, published 12:54Z) is unpicked for the same reason, and it is the Windows kept-datadir gate for 0.3.20's PC 1 step. The sweep cannot bring PC 2 back (the app's poller applies updates; a silent app does not poll); a Windows restart of the app is a hand action, the project lead's or by main's word; the shipper has asked main. The hash lane's G1 and the Windows kept-datadir line wait on that answer; the PC 2 lock stays. SUB-VERSION 2's SUITE LINE (the hash lane): ca3-v4-amend 526fa757 (the code; the ledger tip f0fbd9da), box 2 through the merged tools, route line "13:15:07 build-remote: box 2 (build@142.132.249.238) for class suite, priority normal", rc 0, 66 s: 61 lib + 7 derive + 4 mixer + 19 packs + 2 recheck + 7 scratch = 100 passed, 0 failed; the pinned v2 and v3 packs byte-identical; the three v4 ids as pinned (a788661687db4bb3 equal; c120d7963abdcd96 and 1a4230699a6b9c60 differ). The two commits between 07a809a7 and 526fa757 touch tests only (the (c') field in the accept.rs initialisers; the two generator unit tests follow the amended facts); the library, the packs, the id and the seven fingerprints are unchanged from 07a809a7, so the attack-pass re-gate on 07a809a7 stands for 526fa757. CI: the merge commit's run cancelled by the next push (superseded, not red); 526fa757's run queued (37626985216), its conclusion owned by the hash lane. The 0.3.21 re-pin commit is therefore 526fa757 (or the branch tip at the node lane's archive, code-identical). THE INTEROP FACT stands from the void run: the 5899f603 hub accepted 235 object-byte-5 blocks from the 8097d600 node with 0 rejected, one digest on all five nodes on the live sixteen-field file. The gates: the digest test and the kaspa-pow vector test (the amended devnet epoch-0 id 1a4230699a6b9c60 must equal, c120d7963abdcd96 must differ, the v3 control unchanged) on the box; the mixed-version Devnet 2 gate (the amended 0.3.20 node beside a 5899f603 node for ten minutes on the live file without the v4 fields) after the Mac build; the fresh-join canary the 0.3.20 cut's | +Packs zip (eight packs, packs-ca3-v4-sub2) sha256 69c36772cd79e44e2ddd589466d9c64a94a13c9e970e9f27bd76feabb9b4581b. The suite re-runs through master's build-remote on box 2 (the worktree's own script predates --box; the first run died on the flag), line to follow; G1 on PC 2 under --cards-off after it. The sub-version-2 pairing waits on the node lane's re-pin to a788661687db4bb3 and byte 7. For the record, sub-version 1's pairing: igneum-pow 8c728ca3 against b7cc37e7 (8097d600's assert_ne in) 17 passed, 0 failed, rc 0, 12:21Z. 0.3.20's CUT SET AS IT STANDS (the shipper, 14:1x UK): pin c4459193, igneum-pow 8c728ca3 at object byte 5 (sub-version 1), the floor-moved file publishing with it (the project lead's word; the floor from the live DAA at the publish plus 604,800, the digest read on c4459193), publish about 15:15Z (16:15 BST) on CASES END, the sweep from then with PC 1 first. 0.3.21's clock tonight: the node lane stages release-0.3.21-node at the shipper's sweep-end word (about 15:45Z, 16:45 BST) with the sub-version-2 re-pin (07a809a7, byte 7, id a788661687db4bb3) as its own commit, held pending the F8 census on 07a809a7; the census's clock about 14:00Z (15:00 BST) by the attack-pass lane's within-the-hour line from 13:01Z; the pass line every one of the 64 seeds under 1.2x of the window model on the chain path. If the census fails or slips past 19:00Z (20:00 BST), main's standing ruling applies (nothing on sub-version 2 is proposed until the census is green): 0.3.21's node ships byte 5 again with the re-pin dropped and the rest of its line kept. CI NOTE (13:1x UTC): master's ci runs since 12dc5c97 (nineteen of mine) sit queued behind one self-hosted runner (igneum-build-1, busy; 31 queued across branches, one in progress); the last completed master runs (439a233f to 5f990a09) are success; no red exists, the conclusions are unread until the queue drains. MAIN'S WORD (14:2x UK): the floor move is the project lead's word already and ships with 0.3.20 at the cut; the CI lever is a second self-hosted runner on build-2 plus ubuntu-latest for docs-only pushes, ordered to the CI lane; the rule reads "own a red when the conclusion lands", never holding pushes; the census verdict about 14:00Z (15:00 UK) decides 0.3.21's byte. SUB-VERSION 2's STATIC CENSUS (the hash lane, tools/ca3-v4-uniform on box 2, 13:12Z, 1,024 chain-shaped seeds plus F8's p1 to p3): 0 lossy-sourced load sites of 16,432 (14,329 injecting, 2,103 bijective); 0 programs with an or-, mul- or mulhi-sourced load; the no-era draw path gives the devnet epoch-0 seed the pack's own id a788661687db4bb3, so every draw path reads one stream. Cost of (a') and (c'): 1.99 attempts per seed on average against 0.05 before (p2's seed five), about 2 ms of generation per rejected attempt on one core; nothing a miner or node notices. Ledger entry 715f14be. Master 36e08c80 merged into ca3-v4-amend as f5244ad7 (igneum-pow untouched, re-export 0 differing files), pushed with the gate GREEN, its CI runs queued; the box-2 suite runs through the merged tools (the first attempt died on test initialisers missing the (c') field, fixed in the same push; library and packs unaffected). G1 BLOCKED: PC 2 has not picked up fetch-ca3-v4-sub2-20261007 and run-ca3-v4-sub2-g1-pc2-20261007 (published 13:04:02Z, signature OK); the intake shows nothing from 1ccfe586 since job-update-now-0319 at 10:34:21Z; the PC 2 lock releases when the job closes or in 30 minutes; the run is republished when the app reports. PC 2 READS SILENT on the console (the shipper, 14:4x UK): last seen 2 h ago, app 0.3.19 on node 5899f603, its last line the 0.3.19 update-now at 10:34:21Z; the app went down or stopped polling on that update (the UI lane's update-now; PC 1 took the same update and reports). The build-server lane's PC 2 kept-datadir job (run-20261007-125433, published 12:54Z) is unpicked for the same reason, and it is the Windows kept-datadir gate for 0.3.20's PC 1 step. The sweep cannot bring PC 2 back (the app's poller applies updates; a silent app does not poll); a Windows restart of the app is a hand action, the project lead's or by main's word; the shipper has asked main. The hash lane's G1 and the Windows kept-datadir line wait on that answer; the PC 2 lock stays. SUB-VERSION 2's SUITE LINE (the hash lane): ca3-v4-amend 526fa757 (the code; the ledger tip f0fbd9da), box 2 through the merged tools, route line "13:15:07 build-remote: box 2 (build@142.132.249.238) for class suite, priority normal", rc 0, 66 s: 61 lib + 7 derive + 4 mixer + 19 packs + 2 recheck + 7 scratch = 100 passed, 0 failed; the pinned v2 and v3 packs byte-identical; the three v4 ids as pinned (a788661687db4bb3 equal; c120d7963abdcd96 and 1a4230699a6b9c60 differ). The two commits between 07a809a7 and 526fa757 touch tests only (the (c') field in the accept.rs initialisers; the two generator unit tests follow the amended facts); the library, the packs, the id and the seven fingerprints are unchanged from 07a809a7, so the attack-pass re-gate on 07a809a7 stands for 526fa757. CI: the merge commit's run cancelled by the next push (superseded, not red); 526fa757's run queued (37626985216), its conclusion owned by the hash lane. The 0.3.21 re-pin commit is therefore 526fa757 (or the branch tip at the node lane's archive, code-identical). MAIN'S WORD ON PC 2 (15:0x UK): the hand restart of PC 2's app is asked of the project lead. If PC 2 has not polled by 15:00Z (16:00 UK), "go PC 1": the sub-version 2 G1 on PC 1 under the runner's --cards-off after the 0.3.20 sweep step lands there, one job, read back, nothing else on PC 1. The PC 1 job publishes only after the shipper's line "PC 1 on 0.3.20" (app 0.3.20, node 2.1.0-c4459193, the published file's digest, synced, the sweep step closed with its lock line), about 15:30 to 15:40Z by the clock, and only if PC 2 is still silent; The PC 1 variant is on ca3-v4-amend at a2714d43 (tools/ca3-v4-amend/pc1-v4-sub2-g1.ps1; CI checks pass; no api/quit, pause, resume or cards call in the script): the RTX 5090 on PC 1 (ae432dc7) by its index-free key nvidia:NVIDIA GeForce RTX 5090 on the runner's --cards-off, restored by the runner after; the AMD card and the Intel Arc not named and mining on; the card confirmed quiet by the process list (90 s wait) and nvidia-smi's compute-apps; the installed worker's --bench on the eight packs (the v3 control and the seven sub-version 2 packs) for the 2^24 fingerprint and self-test, rates a reference only; the kit the same zip (69c36772...), as fetch-ca3-v4-sub2-pc1-20261007 immediately before run-ca3-v4-sub2-g1-pc1-20261007 (--timeout-minutes 20, --expires-hours 12); published only on the go-PC-1 line. AP-F8-2 ON SUB-VERSION 2 (the attack-pass lane, 15:1x UK, before anything is proposed): a chain-shaped epoch seed can exhaust all 32 draw attempts under rule (a'), and the generator treats exhaustion as a consensus fault (panic): seed igneum-f9/331672, "32 consecutive candidates rejected, last: (a') load at 16 reads r6, not fresh by dataflow in the loop's steady state". One such seed in the first 331,672 chain-shaped seeds (300,000 drew clean), a rate of order 10^-6 to 10^-5 per epoch seed; the 10^6-seed measurement with the attempts distribution runs on box 2. Meaning: an exhausted epoch seed is an epoch no node can draw a program for, a liveness halt, the era and epoch seeds being VDF outputs nobody can steer; at one epoch an hour, one halt per 11 to 40 years at the bracketed rate, which the firms would compute from the rule as written and file. Sub-version 1 had 0 exhausted in 10^6 chain-shaped seeds. The fix (to the hash lane): the draw enforces the freshness fixpoint itself so (a') never fires (no exhaustion by construction), or MAX_ATTEMPTS sized to the measured rate with the exhaustion probability stated in the spec. The 64-seed hot-set gate on sub-version 2 is separate: 13 of 64 seeds read, none over 1.2x so far. Sub-version 2 is NOT GREEN until both are settled. MAIN'S RULING ON AP-F8-2 (15:2x UK): the draw must be total and no consensus path may panic; preferred fix a deterministic repair instead of rejection (the draw rewrites the offending load's source to the nearest fresh register, or inserts a fresh mix, so every seed yields a program on the first attempt and (a') becomes a check that can never fire); if the repair changes the stream's statistics, the fallback is a stated attempt bound with a deterministic last-resort draw after it, never a panic, the probability in the spec and the ledger; either way a test walking seed igneum-f9/331672 and the exhausting class, the 10^6-seed exhaustion count at zero, and the 64-seed hot-set gate re-run on the fixed commit; the hash lane builds it now on the coordinator's direction; 0.3.21 ships byte 5 if not green by 19:00Z (20:00 UK). The 07a809a7 kit is not published to any PC. + +AP-F8-2 FIXED (the hash lane, ca3-v4-amend 8bdcbdd8, 13:31:10Z, origin and build, gate GREEN): sub-version number unchanged at 2 because the stream is unchanged for every non-exhausting seed (re-export diff 0 on v4-devnet-epoch0, v4-era-0 and v4-era-5; the id a788661687db4bb3 and the seven fingerprints stand; the packs zip sha256 69c36772... holds), so the attack-pass census at 13 of 64 continues on it. The route: the deterministic repair would have rewritten every seed whose attempt 0 fails (a'), about two thirds of seeds, a new stream and a restarted census against the 19:00Z line, so the second route main allowed was taken. The bound: MAX_ATTEMPTS_V4 = 256 for the class v4 shape (v2 and v3 keep 32, keyed on the shape); at the measured rejection rate of about two thirds per attempt, 32 attempts exhaust at about 2e-6 per epoch seed (one undrawable epoch every few decades at one an hour), 256 at under 1e-45, the worst-case draw about half a second on one core. After the cap the draw is total: the seed takes the last-resort program, deterministic and accepted as drawn, the candidate at attempt 256 with every or, mul and mulhi of the base program and the shadow block rewritten to xor, so every register stays fresh from the init words on and (a') holds by construction; no consensus path panics for the v4 shape. The test class_v4_draw_is_total_with_the_last_resort (the last resort on real (a')-rejected candidates, every load fresh after it, no lossy op left, the chain path over 64 seeds without a panic, the cap per class asserted) green on the Mac, the box-2 suite running; the hash lane's 4,096-seed census with the attempt histogram on box 2; the attack-pass lane's 10^6 count is the control; the string with the attack-pass lane with seed igneum-f9/331672 named. CI: 526fa757 success; 8bdcbdd8 queued. The spec and ledger text for the bound and the probability in the ledger entry. The PC 1 variant and fetch job carry 8bdcbdd8's packs (byte-identical to 07a809a7's); the PC 2 kit published at 13:04Z is the same packs. Per tier: a miner never sees the draw (the node draws once an hour, half a second at worst); a chip maker gains nothing from the last resort (a 1e-45 event); the auditor reads the bound and its probability in the spec. SUITE LINE AT 8bdcbdd8: box 2, route line "13:31:21 build-remote: box 2 (build@142.132.249.238) for class suite, priority normal", rc 0, 80 s, 62 lib + 7 derive + 4 mixer + 19 packs + 2 recheck + 7 scratch = 101 passed, 0 failed, the total-draw test included; ledger entry 32e96c9a; CI on 526fa757 success, the newest push's run queued and owned by the hash lane. + +THE 12 GB SETTLED-CLAIM LINE ON c4459193 (the fleet lane, 13:27Z): the floor reads right and the card proves, THE NODE REFUSES. p12-vast on c4459193 (sha 45be9b02, string read back) synced 13:22:47Z on the kept copy; first claim 13:23:01Z "segment 164198..164205 (8 shards, fresh) margin=414 tip=287564 settledNumber=0x28185 settledDaa=0x46317 settledBy=finality candidates=5"; proof complete 13:25:19Z on the 3060 (8 of 8 shards, 135.9 s, peak 8,487 MiB); submission refused "statement differs from the node's at hex offset 472 (lengths 616 vs 616); ours ...2b1a81cb413236cf... node ...0000". The node's start line on this binary reads "proving v1: ... shard program id unknown, aggregator id unknown" where 5899f603 on the same override file reads the ids; the statement is built with zeros and every proof fails its check. This blocks the paid line on any card and, after the sweep, every standing prover; with the node lane (the fix) and the shipper (the cut) since 13:27Z; the pod and proof files held for the node lane's read. The cut set is therefore NOT complete on c4459193. THE CAUSE (the node lane, 13:4xZ): no commit on the line lost the ids and the binary is not the difference; on every build including 5899f603 the statement's ids come from the override object's two fields (absent on the live sixteen-field file), else IGNEUM_PROOF_PROGRAM_IDS, else the verifier host's --mode id when IGNEUM_PROOF_VERIFIER is set. hub-1 runs with IGNEUM_PROOF_VERIFIER=/opt/igneum-floor/bin/igneum-prove-host in its process environment, so the host hands it the ids; the pod's c4459193 node was started bare for the kept-datadir read, so program_ids answered None and the statement carried zeros; a bare 5899f603 reads "unknown" too. The start environment, not the binary. What the child commit fixes anyway: the binary embeds the verifying keys it verifies carried proofs with, so it knows the ids it expects; the child resolves them in that order and last from the embedded keys, with a test whose known-failed shape is the bare node's None and the zero-id statement; the diff two files (resolve_program_ids in igneum/exec/src/proving.rs and its call in kaspad/src/daemon.rs; nothing in acceptance, the version check, relay or peer handling); its exec suite running, the string and sha256 about 13:52Z. Gate carry: the digest and mixed-version gates are outside the diff's territory and carry from c4459193, but rule 4a runs every gate on every candidate from its build, so both run again on the child's binary and the fleet's set with them; the re-run that bears on the diff is the 12 GB line itself, the pod's already-proved claim submitted against a node that names the ids. Per tier: a solo miner on a bare node never proves, so feels nothing; a standing prover beside hub-1's environment was never affected; a prover on a bare node could not be paid on any build until the child. THE CONTROL (the fleet lane, p12-vast): the same c4459193 (sha256 45be9b02d1b002f5, string read back) on the same kept copy, killed and restarted 13:30:28Z with IGNEUM_PROOF_VERIFIER=/opt/igneum-floor/bin-0317/igneum-prove-host (sha 71bc2438) and HOME on the floor: the start line reads "proving v1: ... shard program id 0x2b1a81cb413236cf..., aggregator id 0x474678f3...", no panic, synced 13:31:49Z (155,725 blocks, 4 peers); the bare start of the same binary on the same copy at 12:36:11Z read "unknown". The ids are the start environment on the pinned build, not the binary. The submission verdict against the ids-naming node: the prover restarted 13:31:56Z and claims fresh (the 13:23Z proof held as evidence, past its margin), the first chain proof about 2.5 minutes on the 3060, the verdict line about 13:36Z; the bare-child line runs the minute the child's string and sha land (about 13:52Z). THE CONTROL'S VERDICT (13:33:56Z): against c4459193 restarted with the verifier set, the 3060's first proof "RESULT seg 164286 chain ... 8 shard records, chain_len 8, proof 1272909 bytes, shards 44.9 s, aggregation 39.7 s, wall 110.6 s, peak 8423 MiB", "shards accepted 8 of 8", no "FAILED: statement": the statement check passes, so the zero-id refusal at 13:25Z was the bare start's environment and nothing in the binary, the pin included. The submission then read "segment_refused ... segment already paid; end to end 113.1 s": the claim race (a 24 GB box proved the same fresh segment inside the 3060's 113 s, the shape of p2-4070-1 this morning); the settled floor does not reserve a claim per key, so a 12 GB prover on the open devnet wins only when no faster box picks its segment; the prover runs on (claim 164342..164349, settledNumber 0x28208 by finality, margin 470) and the paid line goes out the minute a race is won, minutes to tens of minutes of races, not the floor. Per tier: a 12 GB card proves and verifies on the pin; whether it is paid on the open devnet is the race against bigger cards, which is the settled floor's design today and a question for the claim rule, not this cut. THE PROVING-IDS CHILD: 55768f88 on release-0.3.20-node (c4459193's child; resolve_program_ids reads the override's fields, then the env or the host, then the embedded verifying keys; one call in the daemon), pairing igneum-pow 8c728ca3 at byte 5; the exec suite 32 passed at 13:29Z with a_bare_node_resolves_its_program_ids_from_the_embedded_keys (known failed first: the bare node's None and the zero-id statement), kaspad check green 13:33Z; igneumd and igneum-miner built on build-1 at 13:35Z, sha256 279b1b690e854fc9, the string read back; the node lane's digest and mixed-version gates on it from 13:37Z (lines about 13:52Z); the fleet has the path, sha and string for its full set from the same minute; ledger N14 on ca3-v4-node. MAIN'S WORD THROUGH THE SHIPPER (15:5x UK): THE PIN IS c4459193 (the control showed the environment names the ids on it; the ids commit 55768f88 is 0.3.21's first node commit, with the ids gate on every candidate from then); the publish about 14:55Z (15:55 BST) on CASES END. PC 1 is offline for the project lead's cable work, out of the sweep's waves; its app updates on its poller when it returns; the "PC 1 on 0.3.20" line comes after the cable work, not at the publish. PC 2 still silent. The sub-version 2 G1 waits for whichever PC returns first; nothing on either before the shipper's line. 0.3.21's node line stages tonight on 55768f88 with the re-pin held for the census. MAIN'S RULING ON 0.3.21's BYTE (16:0x UK): neither PC is needed for the CUDA half; G1 for 8bdcbdd8 runs on a fleet 5090 now (p1-5090 or a one-shot 5090 pod through the fleet lane; the Linux CUDA worker's 2^24 fingerprints and self-test against the Mac's seven; no --cards-off on a pod; ordered to the fleet lane with the kit's sha256 and the pass line). If that G1, the attack-pass lane's two gates on 8bdcbdd8 and the node lane's re-pin and pairing are all green by 19:00Z (20:00 UK), byte 7 ships in 0.3.21 with the Windows G1 owed and run on the first PC that returns (a Windows-only CUDA mismatch would be a 0.3.22 re-pin; nothing flips before the moved floor); if any is not green by then, byte 5 ships and the re-pin stages for 0.3.22. THE FLEET G1 PACKAGE (the hash lane to the fleet lane, 13:4xZ): the kit packs-ca3-v4-sub2-20261007.zip on the dl host (sha256 69c36772...), the eight packs; the worker proto-cuda/nvrtc/worker.cpp built for Linux by infra/cross/build-workers-linux.sh (dlopens libcuda and libnvrtc, compiles each pack's own text; any 0.3.20-tree build is the right binary, its string from --help); the bench `igneum-worker-cuda --bench --pack --batches 5 --batch-log2 24 --block-warps 1`, the pass per pack self-test PASS plus the fingerprint equal; the eight expected fingerprints. THE PC 2 RUN JOB WITHDRAWN: run-ca3-v4-sub2-g1-pc2-20261007 had been live in the signed file since 13:04Z and would have fired the moment PC 2's app polled, before any sweep step; removed and deployed 13:39:54Z, the file verified; the fetch kit stays; the PC 1 variant never published; the PC 2 lock released 13:39:22Z. PC 2 BACK (the shipper, 13:4xZ): polling since 13:38:32Z, app 0.3.19, non-elevated, node synced; the silence from 10:46Z was the whole PC losing power (Kernel-Power 41, no bugcheck), not the update; the build-server lane's kept-datadir job ran on it at 13:38:32Z and passed; PC 2 takes 0.3.20 on its poller in wave 1 at the publish. The Windows G1 on PC 2 ordered now under the PC 2 lock with --cards-off on the 5090, the 0.3.19 worker (nvrtc compiles each pack's text), to close well before the 14:55Z publish (a job under an app relaunch is the shape that killed the 9070 XT on 6 October); if it cannot start by 14:20Z it waits for the shipper's line that PC 2 reads 0.3.20. The fleet 5090 G1 runs regardless. THE LINUX CUDA G1 FOR SUB-VERSION 2: PASS (the fleet lane, p1-5090, the fleet's standing RTX 5090, no rent, 13:43:22Z to 13:44:07Z; kit sha256 69c36772... asserted; the box's igneum-worker-cuda 1.0 of 4 October 2026, sha256 97e036e23f4ace66; --batches 5 --batch-log2 24 --block-warps 1). + +| Pack | Fingerprint on the 5090 | Equal to the Mac | +|---|---|---| +| mx8-devnet-epoch0 (the v3 control) | 90f794dd556f7a3b | yes | +| v4-devnet-epoch0 | e370fb2080b7dbb1 | yes | +| v4-era-0 | b7237555d31fc3cf | yes | +| v4-era-1 | b6b167fa15dfe2c9 | yes | +| v4-era-2 | 28bdf65eff33f2c4 | yes | +| v4-era-3 | e26d38c46f3f1b16 | yes | +| v4-era-4 | dd8fdf6ff4f59eed | yes | +| v4-era-5 | 8bf40f5cb858d835 | yes | + +Self-test PASS on each (FNV-1a 448274a57f508cbc); rates 120 to 142 MH/s with the box's miner loop sharing the card, a reference only; p1-5090's node untouched, its supervisor restarted after. The CUDA half of G1 is green. THE WINDOWS G1 ON PC 2: GREEN (job run-ca3-v4-sub2-g1-pc2-20261007b, published 13:46:33Z under the PC 2 lock taken 13:45:03Z, the runner's --cards-off on the 5090, 15-minute timeout, nothing of the proving lane's running; start 13:46:36Z, end 13:46:50Z, exit 0; lock released 13:47:18Z; app 0.3.19, the installed worker sha256 14b6637e..., the card off before the script and restored on exit, igneum-worker-cuda 0 before and after, the prover untouched). + +| Pack | Fingerprint on PC 2's 5090 | MH/s (card alone, 5 batches) | Equal to the Mac and the fleet 5090 | +|---|---|---|---| +| mx8-devnet-epoch0 (the v3 control) | 90f794dd556f7a3b | 118.1 | yes | +| v4-devnet-epoch0 | e370fb2080b7dbb1 | 119.3 | yes | +| v4-era-0 | b7237555d31fc3cf | 115.8 | yes | +| v4-era-1 | b6b167fa15dfe2c9 | 116.7 | yes | +| v4-era-2 | 28bdf65eff33f2c4 | 119.3 | yes | +| v4-era-3 | e26d38c46f3f1b16 | 129.5 | yes | +| v4-era-4 | dd8fdf6ff4f59eed | 115.2 | yes | +| v4-era-5 | 8bf40f5cb858d835 | 117.1 | yes | + +Self-test PASS on every pack (cache FNV 448274a57f508cbc); both rows in the ledger entry 43c5bf5b. G1 FOR SUB-VERSION 2 IS COMPLETE on three platforms (Metal, Linux CUDA, Windows CUDA), nothing owed. Per tier: the 5090 rate on sub-version 2 is the same band as sub-version 1 (115 to 130 MH/s), so a miner's rate does not move with the class amendment. The lines left for byte 7 by 19:00Z: the attack-pass lane's two gates and 10^6 count on 8bdcbdd8, the node lane's re-pin and the pairing. PC 2 DOWN AGAIN (main, 16:5x UK): the project lead takes PC 2 down for cable work (PC 1 back but his desk); both PCs out of the sweep's waves, each updates on its poller on return; no PC job to PC 1; the Windows G1 completed before the outage, nothing reruns. 0.3.21's SECOND GATE LINE on 55768f88 (sha256 279b1b690e854fc9): the ten-minute mixed-version gate beside the 5899f603 pair, 13:37:40Z to 13:47:52Z, SUMMARY PASS (one digest b0afb2ee on five nodes; 223 new and 381 old blocks accepted by the old hub, 0 rejected; counts equal at 319, 486 and 604 through both clean joins and the restart step at 13:45:22Z; no panic); the node lane's two lines on 0.3.21's first candidate complete, in plan 6.9 on ca3-v4-node; the fleet's set on it (the bare-child 12 GB line, the wipe, the kept read, the cases) is the fleet's. 0.3.21's FIRST GATE LINE on 55768f88 (sha256 279b1b690e854fc9, the string read back; pairing igneum-pow 8c728ca3 at byte 5): the digest gate 13:35:41Z to 13:37:19Z SUMMARY PASS (a89be8a7 on both binaries with the peers; db9a85f9 refused, no peer; the live file's eada4bda unmoved); the ten-minute mixed-version gate from 13:37:40Z, line about 13:50Z. The 0.3.21 order as the shipper sent it: 55768f88; f067f7c1 and 70e4601e; b0444f51; 6eb21fc9; db28d331; then the re-pin from 8bdcbdd8 on the coordinator's word; suites between, the digest read after every one; the mirror's release-0.3.20-node back at the pin c4459193, release-0.3.21-node open at 55768f88. THE LATE-JOIN COMMIT (N9's second half, the node lane): 70e4601e on the box mirror as branch proof-hold-fix, from c4459193, two files (igneum/exec/src/proving.rs, protocol/flows/src/v10/proving.rs); the gap was the fetch side on the joiner (the served record ran the native check against the joiner's trailing exec state before anything was stored, the check refused it, the proof was never held, the body rule read "not held" for 20 s and failed the IBD); the fix holds the proof by hash before the checks (the pool entry still needs them) and the serve side says when it holds fewer than asked; the exec suite 32 passed at 13:26Z with the known-failed shape first, the flows check green 13:28Z, igneumd on build-1 at the 0321 worktree path built 13:32Z, sha256 17649eeb2f7d1290, string read back; with the testnet lane (the resume form, B alone); it joins the 0.3.21 staging as its own commit. THE WIPE CANARY ON c19-1, c4459193 (sha 45be9b02d1b002f5, string read back): FORM END rc 0 at 13:50:53Z. Wipe synced 13:35:50Z (57 minutes, inside the 98-minute class); mining 13:36:00Z to 13:47:07Z, 66 mined, 66 accepted, 0 rejected, isSynced true at the tip throughout; the hub holds 41 of its blocks in its last 700 with 0 rejects (13:47:09Z); the restart on its kept datadir at 13:47:15Z: the old process stopped at once (the new process's first lock line seven seconds after the marker; the watchdog held nothing, the b7cc37e7 fault closed), synced again at 13:48:39Z after 84 s, 109 templates read with max 3,432 ms and 0 timeouts; the kept read on pool-1's 0.3.17 copy on the same pod passed at 13:38Z (the rewrite line once, a clean second start). The pin's set on c4459193: the digest gate PASS, the mixed-version gate PASS, the wipe canary PASS, the kept read PASS, the restart PASS, the 12 GB line proves and verifies (paid is a race, not a gate); CASES END from c20-1 (about 14:50Z) is the last pin line. THE INTEROP FACT stands from the void run: the 5899f603 hub accepted 235 object-byte-5 blocks from the 8097d600 node with 0 rejected, one digest on all five nodes on the live sixteen-field file. The gates: the digest test and the kaspa-pow vector test (the amended devnet epoch-0 id 1a4230699a6b9c60 must equal, c120d7963abdcd96 must differ, the v3 control unchanged) on the box; the mixed-version Devnet 2 gate (the amended 0.3.20 node beside a 5899f603 node for ten minutes on the live file without the v4 fields) after the Mac build; the fresh-join canary the 0.3.20 cut's | | Main's rulings (7 October, morning) | no generator change to v4 on the live devnet; the record's null is the window model with numbers, sent by the hash lane to the attack-pass lane so AP-F8-1 re-gates against it; a fault beyond the model (a low-entropy source at site 15) stops at the coordinator with the two options priced (a 0.3.19 class amendment before the flip, or the flip held at the floor), nothing shipping without the project lead's word; the tighter tail, an acceptance bound on the hot-set share, is a CLASS V5 item (sent to the v5 lane a6410f3b8abefb762 with the 64-seed census as its gate; the bound's number follows from the model) | ### AP-F4-1, the weak-day MUL draw (the attack-pass lane, 7 October, morning): PASS against v4, a class v5 rule diff --git a/docs/plans/site-ui-5-shots/home__1440-dark.png b/docs/plans/site-ui-5-shots/home__1440-dark.png new file mode 100644 index 00000000..10cc7272 Binary files /dev/null and b/docs/plans/site-ui-5-shots/home__1440-dark.png differ diff --git a/docs/plans/site-ui-5-shots/home__1440-light.png b/docs/plans/site-ui-5-shots/home__1440-light.png new file mode 100644 index 00000000..fe912a6a Binary files /dev/null and b/docs/plans/site-ui-5-shots/home__1440-light.png differ diff --git a/docs/plans/site-ui-5-shots/home__390-dark.png b/docs/plans/site-ui-5-shots/home__390-dark.png new file mode 100644 index 00000000..61a09b67 Binary files /dev/null and b/docs/plans/site-ui-5-shots/home__390-dark.png differ diff --git a/docs/plans/site-ui-5-shots/home__390-light.png b/docs/plans/site-ui-5-shots/home__390-light.png new file mode 100644 index 00000000..108a4ba6 Binary files /dev/null and b/docs/plans/site-ui-5-shots/home__390-light.png differ diff --git a/infra/build-server/capacity/lib.sh b/infra/build-server/capacity/lib.sh index 7418d4b2..a7d36d3e 100755 --- a/infra/build-server/capacity/lib.sh +++ b/infra/build-server/capacity/lib.sh @@ -37,11 +37,12 @@ CAP_NODE_BRANCH="${CAP_NODE_BRANCH:-}"; CAP_REPO_SHA="${CAP_REPO_SHA:-}"; CAP_FO # ---- the build-slot and measure hold the layer yields to -------------------------------------------------------------- # a slot or the measure file is "held" when flock -n cannot take it (another process holds the fd). The same probe the -# collector uses (tools/workers/collect.mjs flockHeld). Returns 0 (true) when ANY build slot or the measure hold is taken. +# collector uses (tools/workers/collect.mjs flockHeld). Returns 0 (true) when ANY build slot, the quiet hold or a core lease is taken. cap_build_active() { local slots k f slots=$(cat "$LOCKS/slots" 2>/dev/null || echo 1); [ "$slots" -ge 1 ] 2>/dev/null || slots=1 - if [ -e "$LOCKS/measure" ] && ! flock -n "$LOCKS/measure" true 2>/dev/null; then return 0; fi + if [ -e "$LOCKS/quiet" ] && ! flock -n "$LOCKS/quiet" true 2>/dev/null; then return 0; fi # 7 Oct 2026: the quiet class replaced the measure file + for f in "$LOCKS"/core-*; do [ -e "$f" ] && ! flock -n "$f" true 2>/dev/null && return 0; done for k in $(seq 0 $((slots - 1))); do f="$LOCKS/build-$k" [ -e "$f" ] || continue @@ -51,7 +52,8 @@ cap_build_active() { } cap_hold_reason() { local slots k - if [ -e "$LOCKS/measure" ] && ! flock -n "$LOCKS/measure" true 2>/dev/null; then echo "measure hold"; return; fi + if [ -e "$LOCKS/quiet" ] && ! flock -n "$LOCKS/quiet" true 2>/dev/null; then echo "quiet hold"; return; fi + for f in "$LOCKS"/core-*; do [ -e "$f" ] && ! flock -n "$f" true 2>/dev/null && { echo "core lease"; return; }; done slots=$(cat "$LOCKS/slots" 2>/dev/null || echo 1); [ "$slots" -ge 1 ] 2>/dev/null || slots=1 for k in $(seq 0 $((slots - 1))); do [ -e "$LOCKS/build-$k" ] || continue diff --git a/infra/build-server/lease.sh b/infra/build-server/lease.sh new file mode 100755 index 00000000..09baf355 --- /dev/null +++ b/infra/build-server/lease.sh @@ -0,0 +1,155 @@ +#!/usr/bin/env bash +# Measurement leases on a build box (main, 7 October 2026, 15:07 UK: one global exclusive "measure" flock across unrelated +# measurements stalled build-1 at load 120 with free slots, an exclusive waiter queueing every new shared taker behind it, and +# a stopped probe held the box for five and a half hours). The measure file is retired. In its place: +# +# lease cores --label "" [--owner ] [--nice N] -- +# A measurement that pins cores takes a lease on THOSE CORES ONLY (one flock per core, _locks/core-, taken in ascending +# order, waited for up to 2 h with a wait- file carrying the label), runs the command under nice N (default 10) and +# taskset on the set, and releases. Builds and suites keep off leased cores (remote-run.sh reads the core files before it +# pins its own set). Nothing else is excluded: the box stays open. +# lease quiet --label "" --owner [--cap-s N] -- +# A WHOLE-BOX quiet measurement: its own class, refused (exit 73) while any build slot or any core lease is held, capped at +# 20 minutes (timeout; --cap-s at most 1200), holder line with the owner in _locks/quiet. Unbounded builds take quiet +# shared and wait for it; bounded suites (nice 10, a 32-core band) never take it. +# lease status every lease, the quiet holder and every waiter with its label +# lease reap a holder (lease, quiet or build slot) whose process has been STOPPED (state T) for 5 minutes or more is +# killed and its file cleared, one line each in _log/reaped.log; remote-run.sh's keeper calls this every +# 20 s while any run is on the box (LEASE_REAP_S overrides the 300 s for the self-test) +# lease --self-test the known cases against a scratch lock directory (in the gate: tools/ci/pre-push.sh) +# +# Installed on every box at /srv/builds/_bin/lease by provision.sh (and copied by hand on 7 October 2026); the lock directory is +# IGNEUM_BUILD_SLOTS_DIR (the profile sets /srv/builds/_locks), the log directory IGNEUM_BUILD_LOG_DIR. Files: core- (flock), +# lease- (holder line "pid N since HH:MM:SSZ cores :