Merge branch 'ca3-pc1-amd' into ca3-coord

This commit is contained in:
igneum-labs 2026-10-06 17:04:35 +00:00
commit fc0441d4cb
4 changed files with 119 additions and 56 deletions

View file

@ -36,7 +36,9 @@
* Build (Mac, Apple OpenCL): cc -std=c99 -O2 -Wno-deprecated-declarations -o family-probe-cl family-probe.c -framework OpenCL
* Build (Windows, mingw, no SDK): x86_64-w64-mingw32-gcc -std=c99 -O2 -static -DIGNEUM_CL_DYNAMIC -DCL_TARGET_OPENCL_VERSION=120 \
* -I <redist>/include -o family-probe-cl.exe family-probe.c (OpenCL.dll loaded at run time)
* Run: family-probe-cl [--list] [--device N] [--lanes N] [--steps N] [--reps N] [--only name[,name]]
* Run: family-probe-cl [--list] --device-name <substring> | --device N [--lanes N] [--steps N] [--reps N] [--only name[,name]]
* (--device-name picks the card by name on its newest platform and is the job's way; a bare
* ordinal is never the default)
*/
#define CL_TARGET_OPENCL_VERSION 120
#ifdef __APPLE__
@ -259,7 +261,7 @@ static int group_ref(const char* family, uint32_t g0, uint32_t seed, uint32_t st
}
/* ---- devices ---- */
typedef struct { cl_platform_id p; cl_device_id d; char pname[128], dname[128], driver[64], ver[64]; } Dev;
typedef struct { cl_platform_id p; cl_device_id d; char pname[128], dname[128], driver[64], ver[64]; cl_uint cus; } Dev;
static Dev devs[32]; static int ndevs = 0;
static void enumerate(void) {
cl_platform_id ps[8]; cl_uint np = 0, i;
@ -273,11 +275,13 @@ static void enumerate(void) {
clGetDeviceInfo(ds[j], CL_DEVICE_NAME, sizeof v->dname, v->dname, NULL);
clGetDeviceInfo(ds[j], CL_DRIVER_VERSION, sizeof v->driver, v->driver, NULL);
clGetDeviceInfo(ds[j], CL_DEVICE_VERSION, sizeof v->ver, v->ver, NULL);
v->cus = 0; clGetDeviceInfo(ds[j], CL_DEVICE_MAX_COMPUTE_UNITS, sizeof v->cus, &v->cus, NULL);
}
}
}
typedef struct { double bestMs; int built; int exact; } Run; /* exact: 1 yes, 0 no, -1 unverified */
static const Dev* gDev = NULL; /* the device under test, for the RESULT lines */
static Run run_variant(cl_context ctx, cl_command_queue q, cl_device_id dev, const Variant* v, cl_uint lanes, cl_uint steps, int reps,
cl_mem out, uint32_t* host, const char* vendor, const char* dname, double aluBest) {
@ -319,8 +323,8 @@ static Run run_variant(cl_context ctx, cl_command_queue q, cl_device_id dev, con
double sps = (double)lanes * (double)steps / (best / 1000.0);
double ratio = aluBest > 0 ? best / aluBest : 0;
const char* ex = hasRef ? (exact ? "yes" : "no") : "unverified";
printf("RESULT FAMILY name=%s ms=%.3f gsteps=%.2f ratio=%.3f exact=%s path=%s variant=%s ns=%.3f lanes=%u steps=%u vendor=%s device=\"%s\"\n",
v->family, best, sps / 1e9, ratio, ex, v->path, v->variant, best * 1e6 / (double)steps, lanes, steps, vendor, dname);
printf("RESULT FAMILY name=%s ms=%.3f gsteps=%.2f ratio=%.3f exact=%s path=%s variant=%s ns=%.3f lanes=%u steps=%u vendor=%s device=\"%s\" cus=%u platform=\"%s\"\n",
v->family, best, sps / 1e9, ratio, ex, v->path, v->variant, best * 1e6 / (double)steps, lanes, steps, vendor, dname, gDev->cus, gDev->pname);
res.bestMs = best; res.built = 1; res.exact = hasRef ? exact : -1;
}
clReleaseKernel(k); clReleaseProgram(prog); return res;
@ -329,7 +333,7 @@ static Run run_variant(cl_context ctx, cl_command_queue q, cl_device_id dev, con
static int pathRank(const char* p) { return !strcmp(p, "native") ? 0 : !strcmp(p, "sequence") ? 1 : 2; }
int main(int argc, char** argv) {
cl_uint lanes = 1u << 20, steps = 4096u; int reps = 3, device = 0, list = 0, i; Dev* v; cl_int err; cl_context ctx; cl_command_queue q; cl_mem out; uint32_t* host; const char* vendor; const char* only = NULL;
cl_uint lanes = 1u << 20, steps = 4096u; int reps = 3, device = -1, list = 0, i; Dev* v; cl_int err; cl_context ctx; cl_command_queue q; cl_mem out; uint32_t* host; const char* vendor; const char* only = NULL; const char* devName = NULL;
Run runs[64]; double aluBest = 0; int isAmd;
for (i = 1; i < argc; ++i) {
if (!strcmp(argv[i], "--list")) list = 1;
@ -338,6 +342,7 @@ int main(int argc, char** argv) {
else if (!strcmp(argv[i], "--steps") && i + 1 < argc) steps = (cl_uint)strtoul(argv[++i], 0, 10);
else if (!strcmp(argv[i], "--reps") && i + 1 < argc) reps = atoi(argv[++i]);
else if (!strcmp(argv[i], "--only") && i + 1 < argc) only = argv[++i];
else if (!strcmp(argv[i], "--device-name") && i + 1 < argc) devName = argv[++i];
else { printf("unknown argument %s\n", argv[i]); return 2; }
}
if (lanes < 64 || (lanes & 255u)) { printf("--lanes must be a multiple of 256\n"); return 2; }
@ -345,16 +350,28 @@ int main(int argc, char** argv) {
if (!ig_cl_load()) { printf("%s\n", ig_cl_error); return 1; }
#endif
enumerate();
if (list || ndevs == 0) { for (i = 0; i < ndevs; ++i) printf("[%d] %s | %s | driver %s | %s\n", i, devs[i].dname, devs[i].pname, devs[i].driver, devs[i].ver); if (ndevs == 0) printf("no OpenCL GPU devices\n"); return ndevs ? 0 : 1; }
if (device < 0 || device >= ndevs) { printf("no device %d (have %d)\n", device, ndevs); return 2; }
v = &devs[device];
if (list || ndevs == 0) { for (i = 0; i < ndevs; ++i) printf("[%d] %s | %s | driver %s | %s | %u CUs\n", i, devs[i].dname, devs[i].pname, devs[i].driver, devs[i].ver, devs[i].cus); if (ndevs == 0) printf("no OpenCL GPU devices\n"); return ndevs ? 0 : 1; }
/* --device-name <substring>: the device whose name holds it, on the NEWEST platform when two platforms list the same
card (PC 1 lists the 9070 XT on AMD-APP 3683.0 and again on the older 3652.0: the kit worker hides the older one as
dup; here the highest driver version string wins). A bare ordinal is never the default: 6 October 2026, the first
family job ran ordinal 0, PC 1's integrated gfx1036, and reported one-CU figures as the 9070 XT's. */
if (devName) {
int best = -1;
for (i = 0; i < ndevs; ++i) if (strstr(devs[i].dname, devName) && (best < 0 || strcmp(devs[i].driver, devs[best].driver) > 0)) best = i;
if (best < 0) { printf("RESULT error no OpenCL GPU device whose name holds \"%s\" (listed: ", devName); for (i = 0; i < ndevs; ++i) printf("%s[%d] %s on %s", i ? "; " : "", i, devs[i].dname, devs[i].pname); printf(")\n"); return 2; }
device = best;
printf("RESULT device_choice name=\"%s\" index=%d platform=\"%s\" driver=%s cus=%u (of %d devices; the newest platform for that name)\n", devs[best].dname, best, devs[best].pname, devs[best].driver, devs[best].cus, ndevs);
}
if (device < 0) { printf("RESULT error no device chosen: pass --device-name <substring> (the card by name) or --device N\n"); return 2; }
if (device >= ndevs) { printf("RESULT error no device %d (have %d)\n", device, ndevs); return 2; }
v = &devs[device]; gDev = v;
vendor = strstr(v->pname, "NVIDIA") ? "nvidia" : (strstr(v->pname, "AMD") ? "amd" : (strstr(v->pname, "Apple") ? "apple" : "other"));
isAmd = !strcmp(vendor, "amd");
ctx = clCreateContext(NULL, 1, &v->d, NULL, NULL, &err); if (err != CL_SUCCESS) { printf("clCreateContext %d\n", (int)err); return 1; }
q = clCreateCommandQueue(ctx, v->d, CL_QUEUE_PROFILING_ENABLE, &err); if (err != CL_SUCCESS) { printf("clCreateCommandQueue %d\n", (int)err); return 1; }
out = clCreateBuffer(ctx, CL_MEM_READ_WRITE, (size_t)lanes * 4, NULL, &err); if (err != CL_SUCCESS) { printf("clCreateBuffer %d\n", (int)err); return 1; }
host = (uint32_t*)malloc((size_t)lanes * 4);
printf("family-probe (OpenCL) on [%d] %s | %s | driver %s | %s, lanes %u, steps %u, best of %d, device event time\n", device, v->dname, v->pname, v->driver, v->ver, lanes, steps, reps);
printf("family-probe (OpenCL) on [%d] %s | %s | driver %s | %s | %u CUs, lanes %u, steps %u, best of %d, device event time\n", device, v->dname, v->pname, v->driver, v->ver, v->cus, lanes, steps, reps);
{
char ext[16384]; ext[0] = 0; clGetDeviceInfo(v->d, CL_DEVICE_EXTENSIONS, sizeof ext, ext, NULL);
printf("RESULT EXT device=\"%s\" khr_subgroups=%s khr_subgroup_shuffle=%s intel_subgroups=%s amd_media_ops2=%s khr_integer_dot_product=%s\n", v->dname,
@ -383,9 +400,9 @@ int main(int argc, char** argv) {
if (rank < bestRank || (rank == bestRank && runs[j].bestMs < bestMs)) { bestRank = rank; bestMs = runs[j].bestMs; bestIdx = j; }
}
if (bestIdx < 0) { printf("RESULT FAMILYBEST name=%s built=none\n", VARIANTS[i].family); continue; }
printf("RESULT FAMILYBEST name=%s ms=%.3f gsteps=%.2f ratio=%.3f exact=%s path=%s variant=%s\n", VARIANTS[i].family, runs[bestIdx].bestMs,
printf("RESULT FAMILYBEST name=%s ms=%.3f gsteps=%.2f ratio=%.3f exact=%s path=%s variant=%s device=\"%s\" cus=%u platform=\"%s\"\n", VARIANTS[i].family, runs[bestIdx].bestMs,
(double)lanes * (double)steps / (runs[bestIdx].bestMs / 1000.0) / 1e9, aluBest > 0 ? runs[bestIdx].bestMs / aluBest : 0,
runs[bestIdx].exact == 1 ? "yes" : runs[bestIdx].exact == -1 ? "unverified" : "no", VARIANTS[bestIdx].path, VARIANTS[bestIdx].variant);
runs[bestIdx].exact == 1 ? "yes" : runs[bestIdx].exact == -1 ? "unverified" : "no", VARIANTS[bestIdx].path, VARIANTS[bestIdx].variant, v->dname, v->cus, v->pname);
}
clReleaseMemObject(out); clReleaseCommandQueue(q); clReleaseContext(ctx); free(host);
printf("family-probe: done\n");

View file

@ -40,7 +40,7 @@ The fetch lands at `<jobs>\fetch-ca3-pc1-amd-20261006\{bin,packs,src,SHA256SUMS}
| 0b (first in the queue after "Ember closed") | `pc1-amd-reset.ps1` | `packaging/ota/publish-jobs.sh add --kind run --target ae432dc7 --id run-ca3-pc1-amd-reset-20261006 --script tools/ca3-pc1-amd/pc1-amd-reset.ps1 --timeout-minutes 5 --title "CA3 PC 1: 9070 XT ADLX back to factory, identities read" --deploy` | about 1 min | beside the miners, no card switch | the 9070 XT's ADLX tuning state after Ember run 6 (`gmax 0 plimit 0 factory 0`): `RESULT amd_state before gmax= plimit= factory=`, the factory reset as the Ember playbook did it (ember-tune ca990c2: `igneum-gpu-telemetry --card <ordinal> --reset`, the helper that answers `--tune`: the installed one first, else the newest amd-kit copy, which one printed), `RESULT amd_state after ...`, FAILED unless the after line reads factory 1; then READ ONLY the identities question of the 15:56Z restore: every gfx1201 settings.json entry (`RESULT settings_entry key= enabled= identities=`) and the app's live card list (`RESULT state_card key= identities= state= mhs_now= watts=`), `RESULT identities_answer live_key= identities=`; no POST of identities (the coordinator's call after the read) |
| 1b (the watts re-run, LAST in the queue) | `pc1-amd-watts.ps1` | `packaging/ota/publish-jobs.sh add --kind run --target ae432dc7 --id run-ca3-pc1-amd-watts-20261006 --script tools/ca3-pc1-amd/pc1-amd-watts.ps1 --timeout-minutes 20 --title "CA3 PC 1: watts on the 9070 XT (app control + sh256x27 alone)" --deploy` | about 5 min | the control row with the card ON in the app; the candidate row with the 9070 XT ALONE | the two `watts= uj=` cells job 1 left owed, taken the way main trusts after Ember run 6 (the app's path: engine telemetry 363 of 363 nonzero, /api/state 456 of 456; a direct helper call alone is not trusted): (1) the CONTROL row with the card ON in the app mining class v3, `/api/state` polled at 1 Hz for 90 s (`RESULT APPROW class=v3 mhs_wall= watts_mean= watts_min= watts_max= samples= uj=`; the app's watts come from its own helper at -l 5, so they move every 5 s: `distinct_watts=` says how many values), the helper sampler running beside over the same 90 s (`RESULT sampler_vs_app helper_mean= app_mean= ratio=`); (2) the CANDIDATE row (sh256x27, the card alone through the kit worker, which /api/state cannot see) with the sampler as its only source and the ratio as its calibration (`RESULT LADDER ... watts_source=adlx calibration_ratio=`); a ratio outside 0.95 to 1.05 prints the candidate's watts as owed with the reason. The sampler is proven on the 12 idle seconds first (`RESULT sampler_raw` x3, `RESULT sampler_idle lines= lines_9070_with_watts=`); without a 9070 XT watts line no row is taken (`RESULT watts error=<why>`, status failed). About 5 min: 12 s proof, 90 s app window, the switch, about 90 s of bench, the restore |
| G2 (job 2's slot, after "Ember closed") | `pc1-amd-g2.ps1` | its own small kit first: `tools/ca3-pc1-amd/make-g2-kit.sh "$TMPDIR/igneum-ca3-pc1-amd-g2-kit.zip"` (rebuilt 6 October 16:4x UTC from the merged tree with the generator-4 packs and the merged packfile.h: sha256 `a16165483cbef9c9001897e2e965b95051c4e683940e0dbc9bded002ed0a0e7e`, 198,597 bytes, 35 files; worker exe `7dc3b1ee…`; the 16:2x build `f4c029c7…` carried the generator-3 copies and is superseded), then `packaging/ota/publish-jobs.sh add --kind fetch --target ae432dc7 --id fetch-ca3-pc1-amd-g2-20261006 --file "$TMPDIR/igneum-ca3-pc1-amd-g2-kit.zip" --dir jobs --extract --title "CA3 PC 1 G2 kit" --deploy`, then `packaging/ota/publish-jobs.sh add --kind run --target ae432dc7 --id run-ca3-pc1-amd-g2-20261006 --script tools/ca3-pc1-amd/pc1-amd-g2.ps1 --timeout-minutes 10 --title "CA3 PC 1: gate G2 on the 9070 XT" --deploy` | about 2 min (two serve-mode jobs of 1,024 nonces; the Apple OpenCL dry run took 2.6 s a pack) | beside the miners (a correctness gate) | the "RX 9070 XT" cell of G2 in `docs/plans/counter-asic-3-gate/hash-gates.md` and the AMD line of G2 in status section 5: `RESULT g2 <pack> found <n> of 1024 distinct=<d> sha256 <digest> file <path> exit=<code>` (the hash lane's form) plus `RESULT G2 pack= found= distinct= digest= digest_match=yes|no`; the found lines go to `g2-<pack>.found` in the job folder, never to stdout; the Mac re-check is `tools/ca3-v4/g2-recheck.sh proto-cuda/packs-ca3-v4/<pack> --digest <hex>` (a digest match is 1,024 of 1,024), or `--file` after `publish-jobs.sh add --kind collect --target ae432dc7 --glob "app/jobs/run-ca3-pc1-amd-g2-20261006/g2-*.found"`. Packs: `packs-ca3-v4/mx8-devnet-epoch0` (control, class v3, id 73bcbfe8ccf988f1) and `packs-ca3-v4/v4-devnet-epoch0` (the candidate with the era at the devnet epoch-0 chain seeds; since the program-id fix 7c22d0d generator 4, class "v4", id c120d7963abdcd96, kernels and vectors byte-identical), NOT the first kit's `sh256x27`: that one is the genesis string-seed pack (`IGNEUM_SEED_BYTES_HEX` of 14 bytes) and the worker's job protocol takes only a 64-hex epoch seed, so it cannot be served (the script refuses such a pack with a RESULT line instead of a malformed job). The job line's class token is the pack's own `IGNEUM_PROGRAM_CLASS` (`class=v3` for the control, `class=v4` for the candidate), as `pc2-v4-gates.ps1` does since ca61dec: the worker refuses a token that disagrees with the pack (Mac check 16:41Z: `class=v3` on the v4 pack found 0, the known-bad case). Expected digests (the verifier through g2-recheck.sh on the Mac, equal to the kit worker's serve mode on Apple OpenCL at 16:21Z on the generator-3 copies and again at 16:41Z on the generator-4 packs through the merged host.c and packfile.h with `class=v4`, 1,024 of 1,024 every time; UNCHANGED by the re-export, as the byte-identical kernels require): mx8 `2a1824a2e0834287cf53b33f84cb7322014c85efa9788bece574e4e0730940bd`, v4 `435b976a4de57c5a26c8bada8b3b9e003504687c790d4b856bc56619d3aabd18`; the script carries them and prints `digest_match` itself |
| 2 | `pc1-amd-family.ps1` | `packaging/ota/publish-jobs.sh add --kind run --target ae432dc7 --id run-ca3-pc1-amd-family-20261006 --script tools/ca3-pc1-amd/pc1-amd-family.ps1 --timeout-minutes 15 --title "CA3 PC 1: family step costs on the 9070 XT" --deploy` | about 3 min (3 runs x up to 3 AMD devices, 26 variants each, builds dominate) | beside the miners (ratios; every row says `card_state=`) | the "RX 9070 XT" column of the item 6 table (section 3: `RESULT FAMILYBEST name=<f> ratio=<r> exact= path=` per family, `RESULT FAMILY ... variant=` for the form each took: `shfla_bperm` = `ds_bpermute_b32`, the one number that could move R3; `shflx_swz` / `shflx_bperm`; `perm_amd` = `v_perm_b32` through `amd_perm` or `perm_c` emulated; `bfe_amd`; `dot4_amd` = the 5 October sudot4 row re-measured; `mm8_gfx12` / `mm8_gfx11` = the WMMA builtins, exact=unverified); section 7 row "item 6: the 9070 XT step costs" |
| 2 (run b) | `pc1-amd-family.ps1` | its own kit first: `tools/ca3-pc1-amd/make-family-kit.sh "$TMPDIR/igneum-ca3-pc1-amd-family-kit.zip"` (built 6 October 18:0x UTC: sha256 `c8337e74af5f589d651603eddd3561700219bb5077f531a600a21719bb87ef58`; the probe exe `e7ad93a4d5b4d977…`), then `packaging/ota/publish-jobs.sh add --kind fetch --target ae432dc7 --id fetch-ca3-pc1-amd-family-20261006 --file "$TMPDIR/igneum-ca3-pc1-amd-family-kit.zip" --dir jobs --extract --title "CA3 PC 1 family kit" --deploy`, then `packaging/ota/publish-jobs.sh add --kind run --target ae432dc7 --id run-ca3-pc1-amd-family-20261006-b --script tools/ca3-pc1-amd/pc1-amd-family.ps1 --timeout-minutes 15 --title "CA3 PC 1: family step costs on the 9070 XT (run b, by name)" --deploy` | about 3 min (3 runs on the 9070 XT by name, then 3 each on the older-platform duplicate and the gfx1036 by ordinal; builds dominate) | beside the miners (ratios; every row says `card_state=`) | the "RX 9070 XT" column of the item 6 table (section 3: `RESULT FAMILYBEST name=<f> ratio=<r> exact= path=` per family, `RESULT FAMILY ... variant=` for the form each took: `shfla_bperm` = `ds_bpermute_b32`, the one number that could move R3; `shflx_swz` / `shflx_bperm`; `perm_amd` = `v_perm_b32` through `amd_perm` or `perm_c` emulated; `bfe_amd`; `dot4_amd` = the 5 October sudot4 row re-measured; `mm8_gfx12` / `mm8_gfx11` = the WMMA builtins, exact=unverified); section 7 row "item 6: the 9070 XT step costs" |
| 3 | `pc1-amd-derive.ps1` | `packaging/ota/publish-jobs.sh add --kind run --target ae432dc7 --id run-ca3-pc1-amd-derive-20261006 --script tools/ca3-pc1-amd/pc1-amd-derive.ps1 --timeout-minutes 15 --title "CA3 PC 1: dr736 on the 9070 XT" --deploy` | about 4 min (3 packs x a 60-dispatch pass and a 3-dispatch pass) | beside the miners (build, compile and bit-exact rows; the rate as a ratio to mx8 in the same job) | the item 2 table's 9070 XT cells (section 3: `RESULT DERIVE pack=dr736-genesis build_ms= build_ms_cached= build_ms_over_mx8= dataset_ms= fingerprint= match= mhs= ratio_to_mx8=`): the OpenCL compile cost of the derivation program against mx8's (the NVRTC +1.1 s finding on AMD), the 1 GiB daily build against mx8's 72 to 77 ms (5 October), bit-exactness on the AMD vendor; section 7 row "item 2: the 9070 XT rows" |
| 4 (optional, confirmed by main) | `pc1-4070-shadow.ps1` | `packaging/ota/publish-jobs.sh add --kind run --target ae432dc7 --id run-ca3-pc1-4070-shadow-20261006 --script tools/ca3-pc1-amd/pc1-4070-shadow.ps1 --timeout-minutes 30 --title "CA3 PC 1: item 8 ladder on the RTX 4070" --deploy` | about 11 min (7 benches of 100 to 200 dispatches at an expected 0.4 to 1 s each, approximate: no 4070 row exists yet; NVRTC per pack; the card switch) | 4070 ALONE (its CUDA worker stopped; 5090 and 9070 XT keep mining) | the "small NVIDIA card binds near 100,000" row of item 8 (section 3, consequences per tier at N = 100,000; section 7 "the 4060-class rows"): `RESULT LADDER pack=<p> ops=<N> mhs= watts= uj= sm_mhz=` with nvidia-smi at 1 Hz on that index; G1 fingerprints on a second NVIDIA card |
@ -55,6 +55,12 @@ G1 on the AMD vendor 7 of 7 (every fingerprint equal to the Mac's and the 5090's
| every LADDER row `watts=owed samples=0` though the one-shot helper call printed the 9070 XT at 140.0 W | `Start-Process powershell.exe -ArgumentList @(..., $tele, $samples)`: Windows PowerShell 5.1 joins an ArgumentList with spaces and does not quote, and the helper's path holds "Igneum Miner", so the wrapper's `$Exe` was `...\Programs\Igneum`, its `& $Exe -l 1` failed and nothing was written | the wrapper reads its paths from `IGNEUM_SAMPLER_EXE` and `IGNEUM_SAMPLER_OUT` in the environment (inherited by the child, never on a command line), the `-File` path is quoted, the wrapper logs its own start line with `exists=`; the sampler must prove itself in the 12 idle seconds (first three raw lines printed, a 9070 XT line with watts required) or the benches are not run and the job fails loudly with `RESULT watts error=<why>` |
| the restore posted one entry's identities (2) to every key form while settings.json holds three gfx1201 entries (`amd:3:gfx1201` off / 2, `amd:gfx1201` on / 8, `amd:1:gfx1201` on / 2); the worker came back (pid 4900), but if the app's live key is `amd:gfx1201` its identities moved from 8 to 2 | one `$cardIdent` for every form | every entry is restored with ITS OWN enabled flag and identities (and cap, job 4), the derived forms first and the key as written last so an entry's own settings win on the live key; job 4 carries the same shape. OWED: a read of PC 1's settings.json (or the app's card tile) to confirm which gfx1201 key the app runs and whether its identities read 8 or 2 after the 15:56Z restore; if 2 and the app runs `amd:gfx1201`, one `POST api/cards {key: amd:gfx1201, enabled: true, identities: 8}` puts it back (the coordinator's call, not done by any job here) |
## Job 2's first run (run-ca3-pc1-amd-family-20261006, 150 s, exit 0) and what it found
Every probe ran ordinal 0, PC 1's integrated gfx1036 (one CU: alu 40.59 G steps/s against the Mac's 881 and the 5090's 7,941; the 9070 XT, 32 CUs, should read about 1,000 to 2,000): the script's device-list parse ran a second `-match` after the capturing one, which overwrote `$Matches`, so every device read `idx 0` and an empty name, and the probe took a bare ordinal. The rows are a gfx1036 (RDNA 2 iGPU) column, kept: alu 1.00, shifts 1.15, bfe 1.15 native, andn 1.16, popc 1.55, clz 1.75, sel 1.81, shfla and shflx 1.89 through `ds_bpermute`, perm 2.43 emulated, dot4 2.57 emulated, mm8 none. The 9070 XT column stays owed to run b. Fixed (commit after c38dfef): the probe host takes `--device-name gfx1201` (the match on the newest AMD platform by driver version, the kit worker's dedup rule; `RESULT device_choice name= index= platform= driver= cus=`; `RESULT error` and exit 2 when no name matches, a bare ordinal never the default), prints the device's name, CUs and platform on every `RESULT FAMILY` and `FAMILYBEST` line; the script runs the 9070 XT by name first and the older-platform duplicate and the gfx1036 by explicit ordinal after it as their own labelled columns, and fails the job when the 9070 XT gave no row. Mac check (18:0x UTC, `with-lock.sh run`, load 9.9): `--device-name M5` chose the M5 Max with 40 CUs and ran bit-exact; `--device-name gfx1201` on the Mac printed `RESULT error no OpenCL GPU device whose name holds "gfx1201"` and exit 2 (the known-bad case); no device given printed `RESULT error no device chosen`.
The finding on AMD's OpenCL C from run a (the same compiler serves the 9070 XT): `amd_perm` (cl_amd_media_ops2) does not compile, `__builtin_amdgcn_sudot4` does not compile under OpenCL C (the 5 October dot4 row used it through the same probe shape, so the LC compiler's builtin set differs between that build and this; the row is owed a second look), and the WMMA builtins (`__builtin_amdgcn_wmma_i32_16x16x16_iu8_w32_gfx12` and `_w32`) do not compile; `__builtin_amdgcn_ds_bpermute` and `ds_swizzle` do. So on AMD the byte-permute, dot4 and mm8 families are the generic C sequence until a path exists (inline assembly through the LC compiler, or a HIP probe), and the reserve order's R1 (perm) and R8 (mm8) rows read emulated on AMD; the shuffle rows (R3) are native through `ds_bpermute`. Consequence for the AMD tier: a reserve family whose only native path is a builtin AMD's OpenCL C does not expose costs an AMD miner the emulated sequence on every step until the worker gains an ISA path; the 9070 XT ratios of run b say how much.
## The AMD watts readback on PC 1 (what exists)
`igneum-gpu-telemetry.exe` (proto-opencl/gpu-telemetry.c on master, shipped by `packaging/windows/make-payload.sh` since 0.3.10; the Ember Tune playbook found it in the install folder or under a `jobs\amd-kit-*\kit\` folder) prints one line per AMD card per sample: `amd <n> bus <pci> kind discrete name "AMD Radeon RX 9070 XT" watts <W> temp_c <C> fan_rpm <r> fan_pct <p> mclk_mhz <m> gclk_mhz <g> util_pct <u> source adlx`, then `end <ms> N card(s)`, at `-l 1` once a second with `fflush` per sample. The watts field is ADLX `GPUPower` with `GPUTotalBoardPower` as the fallback (gpu-telemetry.c lines 120 to 121 on master); on 5 October it read the 9070 XT at 198.9 W at 17.73 MH/s (release-0.3.10.md, job `tele-measure-1`), the figure the app's MH per watt line uses. So a board-watts readback EXISTS and job 1 uses it: a second copy of the helper at 1 Hz through a wrapper that stamps each line (UTC, ms) so the samples window onto each bench (after its first 12 s, before its last 2 s, the PC 2 shape), ended by the wrapper's pid in `finally`. The job prints `RESULT tele helper=<path> sha256=<hex> watts_source=adlx` after one sample; if the helper is missing or prints `watts -` for the 9070 XT the rows carry `watts=owed uj=owed` and the SUMMARY says `watts: owed`, never a guess. Whether ADLX `GPUPower` on RDNA 4 is total board power or ASIC power is not verified here: the row says `adlx` and the number is what the app itself reports.

View file

@ -0,0 +1,25 @@
#!/usr/bin/env bash
# The family kit for tools/ca3-pc1-amd/pc1-amd-family.ps1 run b (6 October 2026): the OpenCL family probe
# (proto-opencl/family-probe.c, with --device-name) cross-compiled with mingw, its source and SHA256SUMS. A kit of its
# own because the app runs a fetch id once and the first kit's probe took a bare ordinal.
# tools/ca3-pc1-amd/make-family-kit.sh [out.zip] default $TMPDIR/igneum-ca3-pc1-amd-family-kit.zip
set -euo pipefail
HERE="$(cd "$(dirname "$0")" && pwd)"
ROOT="$(cd "$HERE/../.." && pwd)"
OUT="${1:-${TMPDIR:-/tmp}/igneum-ca3-pc1-amd-family-kit.zip}"
RED="${IGNEUM_REDIST:-$ROOT/proto-cuda/nvrtc/redist}"
[ -f "$RED/include/CL/cl.h" ] || { echo "no OpenCL headers at $RED/include/CL/cl.h (run proto-cuda/nvrtc/fetch-redist.sh or set IGNEUM_REDIST)" >&2; exit 1; }
CC=x86_64-w64-mingw32-gcc; STRIP=x86_64-w64-mingw32-strip
command -v "$CC" >/dev/null || { echo "$CC not found (brew install mingw-w64)" >&2; exit 1; }
STAGE="$(mktemp -d)"
mkdir -p "$STAGE/bin" "$STAGE/src"
echo "== family-probe-cl.exe (proto-opencl/family-probe.c)"
"$CC" -std=c99 -O2 -Wall -Wextra -static -DIGNEUM_CL_DYNAMIC -DCL_TARGET_OPENCL_VERSION=120 -I "$RED/include" -o "$STAGE/bin/family-probe-cl.exe" "$ROOT/proto-opencl/family-probe.c"
"$STRIP" "$STAGE/bin/family-probe-cl.exe"
cp "$ROOT/proto-opencl/family-probe.c" "$ROOT/proto-opencl/cl_dynamic.h" "$STAGE/src/"
( cd "$STAGE" && find . -type f ! -name SHA256SUMS | sort | xargs shasum -a 256 > SHA256SUMS )
rm -f "$OUT"
( cd "$STAGE" && zip -X -q -r "$OUT" . )
rm -rf "$STAGE"
echo "kit $OUT"
echo "sha256 $(shasum -a 256 "$OUT" | cut -c1-64) bytes $(wc -c < "$OUT" | tr -d ' ')"

View file

@ -1,23 +1,26 @@
# Counter ASIC 3.0, PC 1 AMD job 2 (6 October 2026): item 6's step cost of every reserve candidate family of spec
# Counter ASIC 3.0, PC 1 AMD job 2 (6 October 2026, run b): item 6's step cost of every reserve candidate family of spec
# 1.13.2 on PC 1's RX 9070 XT (machine ae432dc7, gfx1201), docs/plans/counter-asic-3-reserve.md, status file section
# 3 "Item 6" (the OWED AMD column). Published as a plain signed `run` job (NOT --stop-miners) and run BESIDE THE MINERS:
# a dependent-chain ratio (each family against the add-xor-rotate chain, both measured in the same minute on the
# same card) survives a shared card, so no card is switched here; the process list says what the card carried and
# every line says loaded or quiet. The probe is the kit's family-probe-cl.exe (proto-opencl/family-probe.c of branch
# ca3-pc1-amd, cross-compiled with mingw, OpenCL.dll loaded at run time; the Mac ran the same source bit-exact on
# Apple OpenCL on 6 October 2026, 15:2x UTC), run three times on every AMD OpenCL device the machine lists (the
# gfx1201 on the current platform, its older-platform duplicate, the gfx1036 when present), each run best of 3 with
# a fresh seed per repetition, device event time, bit-exact against the CPU reference on two whole 32-lane groups.
# Every family is tried through every form AMD's OpenCL C offers (clang builtins ds_bpermute / ds_swizzle / sudot4 /
# wmma, cl_khr_subgroup_shuffle, cl_intel_subgroups, cl_amd_media_ops2 amd_bfe / amd_perm, the plain C form, the
# __local emulation); a form that does not compile prints build=failed with the first line of the build log and
# the run goes on. The script never quits, pauses, resumes or updates the installed app and never writes settings.json.
# Lines: the probe's own `RESULT FAMILY name=<f> ms=<best> gsteps=<x> ratio=<r> exact=<yes/no/unverified>
# path=<native|sequence|emulated> variant=<v> ...` and `RESULT FAMILYBEST ...` (the row per family the status file
# takes), each prefixed `RESULT run=<n> dev=<d>`; a `SUMMARY {json}` line at the end. Read back with `node tools/jobs.mjs <job id>`.
# 3 "Item 6" (the OWED AMD column). Run a (run-ca3-pc1-amd-family-20261006, 150 s, exit 0) measured ordinal 0, PC 1's
# integrated gfx1036 (one CU: alu 40.59 G steps/s against the 9070 XT's expected 1,000 to 2,000): the device list parse
# ran a second -match after the capturing one, which overwrote $Matches, so every device read idx 0 and an empty name,
# and the probe took a bare ordinal. Run b: the probe picks the card BY NAME (`--device-name gfx1201`, the match on the
# newest AMD platform by driver version, the kit worker's dedup rule; `RESULT device_choice name= index= platform= driver=
# cus=`), prints the device's name, CUs and platform on every RESULT line, and refuses with `RESULT error` when no
# gfx1201 is listed; the older-platform duplicate and the gfx1036 run after it by explicit ordinal, as their own labelled
# columns. Published as a plain signed `run` job (NOT --stop-miners) and run BESIDE THE MINERS: a dependent-chain ratio
# survives a shared card; no card is switched; every line says loaded or quiet. The probe is the family kit's
# family-probe-cl.exe (proto-opencl/family-probe.c of branch ca3-pc1-amd, cross-compiled with mingw, OpenCL.dll at run
# time; the Mac ran the same source bit-exact on Apple OpenCL), three runs per device, each best of 3 with a fresh seed
# per repetition, device event time, bit-exact against the CPU reference on two whole 32-lane groups. Every family is
# tried through every form AMD's OpenCL C offers; a form that does not compile prints build=failed with the first line
# of the build log and the run goes on (run a's finding: AMD's OpenCL C compiles no amd_perm, no sudot4, no WMMA
# builtin; ds_bpermute does compile). Never quits, pauses, resumes or updates the installed app; never writes
# settings.json. Lines: the probe's own `RESULT FAMILY name= ms= gsteps= ratio= exact= path= variant= ... device= cus=
# platform=` and `RESULT FAMILYBEST ...`, each prefixed `RESULT run=<n> dev=<label>`; a `SUMMARY {json}` line at the end.
# Read back with `node tools/jobs.mjs <job id>`.
$ErrorActionPreference = 'Continue'
$jobName = 'pc1-amd-family'
$kitId = 'fetch-ca3-pc1-amd-20261006'
$kitId = 'fetch-ca3-pc1-amd-family-20261006'
$started = Get-Date
function Stamp { (Get-Date).ToUniversalTime().ToString('yyyy-MM-ddTHH:mm:ssZ') }
function Summary([string] $status, [hashtable] $extra) {
@ -39,41 +42,53 @@ if (Test-Path $src) { "RESULT source family-probe.c sha256 $((Get-FileHash -Algo
function Workers { @(Get-CimInstance Win32_Process -Filter "Name = 'igneum-worker-opencl.exe' OR Name = 'igneum-worker-cuda.exe'" -ErrorAction SilentlyContinue | ForEach-Object { "$($_.Name):$($_.ProcessId):[$($_.CommandLine -replace '\s+', ' ')]" }) }
$w = @(Workers)
$amdLoaded = ($w | Where-Object { $_ -match 'igneum-worker-opencl' }).Count -gt 0
$cardState = if ($amdLoaded) { 'loaded' } else { 'quiet' }
"RESULT workers_before $(Stamp) $($w -join ' ')"
"RESULT context card_state=$(if ($amdLoaded) { 'LOADED (an igneum-worker-opencl process mines on an AMD card: absolutes are shared-card figures, ratios stand)' } else { 'quiet (no igneum-worker-opencl process)' })"
# every AMD device the probe lists (both platform entries of the 9070 XT, the gfx1036 when present)
# the probe's own device list: the 9070 XT by name first (the probe picks the newest platform itself), then the
# older-platform duplicate and the gfx1036 by explicit ordinal as their own columns
$list = @(& $probe --list 2>&1 | ForEach-Object { "$_" })
$list | ForEach-Object { "RESULT list $_" }
$devs = @()
foreach ($l in $list) { if ($l -match '^\[(\d+)\] (.*?) \| (.*?) \|' -and $l -match 'gfx1201|gfx1036|AMD|Radeon') { if ($l -notmatch 'NVIDIA') { $devs += [pscustomobject]@{ idx = [int]$Matches[1]; name = $Matches[2]; platform = $Matches[3] } } } }
if ($devs.Count -eq 0) { "RESULT error no AMD OpenCL device in the probe's list (the eGPU is off the bus and no gfx1036: every row OWED)"; Summary 'failed' @{ error = 'no AMD device' } ; exit 2 }
"RESULT devices $(($devs | ForEach-Object { "[$($_.idx)] $($_.name) on $($_.platform)" }) -join ' ; ')"
# the gfx1201 on the current platform first (the row the status file takes), then the rest
$ordered = @($devs | Where-Object { $_.name -match 'gfx1201' } | Sort-Object { $_.platform } -Descending) + @($devs | Where-Object { $_.name -notmatch 'gfx1201' })
$runs = 0; $exactRows = 0; $rowsTotal = 0; $built = 0; $failedBuilds = 0
foreach ($d in $ordered) {
for ($r = 1; $r -le 3; $r++) {
"RESULT run=$r dev=$($d.idx) start $(Stamp) name=$($d.name) platform=$($d.platform) card_state=$(if ($amdLoaded) { 'loaded' } else { 'quiet' })"
$t0 = Get-Date
$lines = @(& $probe --device $d.idx --reps 3 2>&1 | ForEach-Object { "$_" })
$rc = $LASTEXITCODE
foreach ($l in $lines) {
if ($l -match '^RESULT ') { "RESULT run=$r dev=$($d.idx) $($l.Substring(7))" } else { "RESULT run=$r dev=$($d.idx) text $l" }
if ($l -match '^RESULT FAMILY name=\S+ ms=') { $rowsTotal++; if ($l -match ' exact=yes ') { $exactRows++ } }
if ($l -match '^RESULT FAMILY .*build=failed') { $failedBuilds++ }
if ($l -match '^RESULT FAMILYBEST name=\S+ ms=') { $built++ }
}
"RESULT run=$r dev=$($d.idx) end $(Stamp) exit=$rc wall_s=$([int]((Get-Date) - $t0).TotalSeconds)"
if ($rc -ne 0) { "RESULT run=$r dev=$($d.idx) error=probe exit $rc" }
$runs++
Start-Sleep -Seconds 2
$extra = @()
$gfx1201Count = 0; $newest = ''
foreach ($l in $list) {
if ($l -match '^\[(\d+)\] (.*?) \| (.*?) \| driver (\S+)') {
$idx = [int] $Matches[1]; $name = $Matches[2]; $plat = $Matches[3]; $drv = $Matches[4]
if ($name -match 'gfx1201') { $gfx1201Count++; if ($drv -gt $newest) { $newest = $drv } }
if ($name -match 'gfx1201|gfx1036|Radeon' -and $name -notmatch 'NVIDIA') { $extra += [pscustomobject]@{ idx = $idx; name = $name; platform = $plat; driver = $drv } }
}
}
if ($gfx1201Count -eq 0) { "RESULT error no gfx1201 in the probe's list (the eGPU is off the bus: the 9070 XT column stays OWED); listed AMD devices: $(($extra | ForEach-Object { "[$($_.idx)] $($_.name) on $($_.platform) driver $($_.driver)" }) -join ' ; ')"; Summary 'failed' @{ error = 'no gfx1201' }; exit 2 }
# the by-ordinal runs after the by-name one: the gfx1201 entries NOT on the newest driver (the older platform), then the rest
$byOrdinal = @($extra | Where-Object { -not ($_.name -match 'gfx1201' -and $_.driver -eq $newest) })
"RESULT devices by_name=gfx1201 (newest driver $newest, $gfx1201Count gfx1201 entries) then by_ordinal=[$(($byOrdinal | ForEach-Object { "[$($_.idx)] $($_.name) on $($_.platform) driver $($_.driver)" }) -join ' ; ')]"
$runs = 0; $exactRows = 0; $rowsTotal = 0; $built = 0; $failedBuilds = 0; $probeErrors = 0; $mainRows = 0
function RunProbe([string] $label, [string[]] $args, [int] $r) {
"RESULT run=$r dev=$label start $(Stamp) args=[$($args -join ' ')] card_state=$cardState"
$t0 = Get-Date
$lines = @(& $probe @args --reps 3 2>&1 | ForEach-Object { "$_" })
$rc = $LASTEXITCODE
foreach ($l in $lines) {
if ($l -match '^RESULT ') { "RESULT run=$r dev=$label $($l.Substring(7))" } else { "RESULT run=$r dev=$label text $l" }
if ($l -match '^RESULT FAMILY name=\S+ ms=') { $script:rowsTotal++; if ($l -match ' exact=yes ') { $script:exactRows++ }; if ($label -eq '9070xt') { $script:mainRows++ } }
if ($l -match '^RESULT FAMILY .*build=failed') { $script:failedBuilds++ }
if ($l -match '^RESULT FAMILYBEST name=\S+ ms=') { $script:built++ }
if ($l -match '^RESULT error') { $script:probeErrors++ }
}
"RESULT run=$r dev=$label end $(Stamp) exit=$rc wall_s=$([int]((Get-Date) - $t0).TotalSeconds)"
if ($rc -ne 0) { "RESULT run=$r dev=$label error=probe exit $rc" }
$script:runs++
}
for ($r = 1; $r -le 3; $r++) { RunProbe '9070xt' @('--device-name', 'gfx1201') $r; Start-Sleep -Seconds 2 }
foreach ($d in $byOrdinal) {
$label = if ($d.name -match 'gfx1201') { 'gfx1201-old-platform' } elseif ($d.name -match 'gfx1036') { 'gfx1036' } else { 'amd-' + $d.idx }
for ($r = 1; $r -le 3; $r++) { RunProbe $label @('--device', "$($d.idx)") $r; Start-Sleep -Seconds 2 }
}
"RESULT workers_after $(Stamp) $((Workers) -join ' ')"
"RESULT end $(Stamp)"
$status = if ($rowsTotal -gt 0) { 'done' } else { 'failed' }
Summary $status @{ runs = $runs; devices = $ordered.Count; family_rows = $rowsTotal; exact_rows = $exactRows; failed_builds = $failedBuilds; familybest_rows = $built; card_state = $(if ($amdLoaded) { 'loaded' } else { 'quiet' }) }
$status = if ($mainRows -gt 0) { 'done' } elseif ($rowsTotal -gt 0) { 'partial' } else { 'failed' }
Summary $status @{ runs = $runs; family_rows = $rowsTotal; rows_9070xt = $mainRows; exact_rows = $exactRows; failed_builds = $failedBuilds; familybest_rows = $built; probe_errors = $probeErrors; card_state = $cardState }
if ($status -eq 'failed') { exit 1 }
exit 0