Merge branch 'ca3-pc1-amd' into ca3-coord
This commit is contained in:
commit
afc61c92ba
8 changed files with 1155 additions and 0 deletions
393
proto-opencl/family-probe.c
Normal file
393
proto-opencl/family-probe.c
Normal file
|
|
@ -0,0 +1,393 @@
|
|||
/* family-probe (OpenCL): the step cost of every reserve candidate family of spec 1.13.2 on AMD (and any OpenCL GPU),
|
||||
* standalone (no pack, no lottery kernel). Counter ASIC 3.0 item 6 (docs/plans/counter-asic-3-reserve.md), the AMD
|
||||
* port of proto-cuda/family-probe.cu and proto-metal/family-probe.swift, 6 October 2026. PC 1 job on the RX 9070 XT;
|
||||
* the Mac runs it on Apple OpenCL for the C-form and emulated rows only (every vendor builtin fails to build there,
|
||||
* which is reported as a row, not an error).
|
||||
*
|
||||
* Same method as the CUDA probe: a dependent chain of one op per step per lane, 1,048,576 lanes x 4,096 steps, best
|
||||
* of N, device event time, bit-exact against a CPU reference on two whole 32-lane groups (lanes 0..31 and the last
|
||||
* 32). Every chain has the dot4 probe's glue: acc = OP(acc, x, y); x = x * K + acc; y = rotl(y, 7) ^ (acc + s).
|
||||
*
|
||||
* A family is measured through every form AMD's OpenCL C might give it, each built on its own (a variant the
|
||||
* platform cannot compile prints one build=failed row with the first line of the build log and the run goes on):
|
||||
* path=native an explicit intrinsic or builtin: __builtin_amdgcn_ds_bpermute / ds_swizzle (the clang builtins the
|
||||
* LC compiler accepts, the same path the 5 October dot4 row took through __builtin_amdgcn_sudot4),
|
||||
* sub_group_shuffle[_xor] (cl_khr_subgroup_shuffle), intel_sub_group_shuffle[_xor]
|
||||
* (cl_intel_subgroups), amd_bfe / amd_perm (cl_amd_media_ops2), __builtin_amdgcn_sudot4, the
|
||||
* cl_khr_integer_dot_product dot(), and the WMMA builtins for mm8
|
||||
* path=sequence the plain OpenCL C form (rotate, shifts, masks, popcount, clz, the ternary select): whatever
|
||||
* instruction sequence the compiler emits; one instruction or several is not read from the ISA here
|
||||
* (no disassembler in the job), it is inferred from the step cost against rotr
|
||||
* path=emulated a form that is certainly several instructions: the byte permute by shifts and masks, the lane
|
||||
* shuffles through __local memory and two barriers, the dot4 by four byte products
|
||||
* The families (the names of counter-asic-3-reserve.md and the CUDA probe): alu (reference chain, 5 ops a step, no
|
||||
* acc), rotr, shflx (lane XOR 8), shl, shr, bfe (bits 7..19 of y), andn, perm (bytes y.b1, y.b3, y.b0, y.b2), popc,
|
||||
* clz (clz(0) = 32), sel (bit 5 of y ? x : acc), shfla (lane + 3 mod 32), dot4 (signed, dp4a.s32.s32 semantics),
|
||||
* mm8 (one 16x16x16 int8 WMMA per step per lane: the fragment layout of RDNA 4's WMMA is not verified against a CPU
|
||||
* reference here, so that row is exact=unverified and OWED for exactness; its step cost stands if it builds).
|
||||
*
|
||||
* Output: one line per variant
|
||||
* RESULT FAMILY name=<family> ms=<best> gsteps=<G steps/s> ratio=<best / alu best> exact=yes|no|unverified
|
||||
* path=native|sequence|emulated variant=<name> ns=<ns per step> vendor=<v> device="<name>"
|
||||
* RESULT FAMILY name=<family> variant=<name> path=<p> build=failed log="<first line>"
|
||||
* and, per family, the row the status file takes (the fastest exact variant, native before sequence before emulated):
|
||||
* RESULT FAMILYBEST name=<family> ms=<best> gsteps=<x> ratio=<r> exact=<e> path=<p> variant=<name>
|
||||
*
|
||||
* Build (Mac, Apple OpenCL): cc -std=c99 -O2 -Wno-deprecated-declarations -o family-probe-cl family-probe.c -framework OpenCL
|
||||
* Build (Windows, mingw, no SDK): x86_64-w64-mingw32-gcc -std=c99 -O2 -static -DIGNEUM_CL_DYNAMIC -DCL_TARGET_OPENCL_VERSION=120 \
|
||||
* -I <redist>/include -o family-probe-cl.exe family-probe.c (OpenCL.dll loaded at run time)
|
||||
* Run: family-probe-cl [--list] [--device N] [--lanes N] [--steps N] [--reps N] [--only name[,name]]
|
||||
*/
|
||||
#define CL_TARGET_OPENCL_VERSION 120
|
||||
#ifdef __APPLE__
|
||||
#include <OpenCL/cl.h>
|
||||
#else
|
||||
#include <CL/cl.h>
|
||||
#endif
|
||||
#ifdef IGNEUM_CL_DYNAMIC
|
||||
#include "cl_dynamic.h"
|
||||
#endif
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <stdint.h>
|
||||
|
||||
/* ---- the kernel text: one COMMON block and one body per variant, every variant its own program ---- */
|
||||
static const char* COMMON =
|
||||
"static inline uint pm_mix(uint x) { x ^= x >> 16; x *= 0x7feb352du; x ^= x >> 15; x *= 0x846ca68bu; x ^= x >> 16; return x; }\n"
|
||||
"static inline uint rotr_var(uint x, uint n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }\n"
|
||||
"#define K32 0x9E3779B1u\n"
|
||||
"#define CHAIN_HEAD uint g = (uint)get_global_id(0); uint lid = (uint)get_local_id(0); uint lane = lid & 31u; (void)lane; uint x = pm_mix(g ^ seed), y = x ^ 0x5bd1e995u; uint acc = pm_mix(x);\n"
|
||||
"#define CHAIN_TAIL x = x * K32 + acc; y = rotate(y, 7u) ^ (acc + s);\n"
|
||||
"#define CHAIN_OUT out[g] = acc ^ x ^ y;\n"
|
||||
"#define CHAIN(OP) __kernel void probe(uint steps, uint seed, __global uint* out) { CHAIN_HEAD for (uint s = 0u; s < steps; ++s) { OP; CHAIN_TAIL } CHAIN_OUT }\n"
|
||||
/* the lane id inside the wave on AMD (mbcnt of the full mask), so a ds_bpermute address stays inside the lane's
|
||||
own 32-lane group whether the compiler chose wave32 or wave64 */
|
||||
"#define AMD_WAVE_LANE (__builtin_amdgcn_mbcnt_hi(~0u, __builtin_amdgcn_mbcnt_lo(~0u, 0u)))\n";
|
||||
|
||||
static const char* K_ALU =
|
||||
"__kernel void probe(uint steps, uint seed, __global uint* out) {\n"
|
||||
" uint g = (uint)get_global_id(0); uint x = pm_mix(g ^ seed), y = x ^ 0x5bd1e995u;\n"
|
||||
" for (uint s = 0u; s < steps; ++s) { x = x * K32 + rotate(y, 7u); y = (y ^ x) + s; }\n"
|
||||
" out[g] = x ^ y;\n"
|
||||
"}\n";
|
||||
static const char* K_ROTR = "CHAIN(acc = rotr_var(y, x))\n";
|
||||
static const char* K_SHL = "CHAIN(acc = y << (x & 31u))\n";
|
||||
static const char* K_SHR = "CHAIN(acc = y >> (x & 31u))\n";
|
||||
static const char* K_ANDN = "CHAIN(acc = y & ~x)\n";
|
||||
static const char* K_POPC = "CHAIN(acc = acc + popcount(x))\n";
|
||||
static const char* K_CLZ = "CHAIN(acc = acc + clz(x))\n";
|
||||
static const char* K_SEL = "CHAIN(acc = ((y >> 5u) & 1u) ? x : acc)\n";
|
||||
static const char* K_BFE_C = "CHAIN(acc = (y >> 7u) & 0x1fffu)\n";
|
||||
static const char* K_BFE_AMD =
|
||||
"#pragma OPENCL EXTENSION cl_amd_media_ops2 : enable\n"
|
||||
"CHAIN(acc = amd_bfe(y, 7u, 13u))\n";
|
||||
static const char* K_PERM_C =
|
||||
"CHAIN(acc = ((y >> 8) & 0xffu) | (((y >> 24) & 0xffu) << 8) | ((y & 0xffu) << 16) | (((y >> 16) & 0xffu) << 24))\n";
|
||||
/* v_perm_b32 through cl_amd_media_ops2: both sources are y, so the selector bytes 1, 3, 0, 2 pick y's bytes whichever
|
||||
source the extension puts in the low half; the CPU check settles the byte order */
|
||||
static const char* K_PERM_AMD =
|
||||
"#pragma OPENCL EXTENSION cl_amd_media_ops2 : enable\n"
|
||||
"CHAIN(acc = amd_perm(y, y, 0x02000301u))\n";
|
||||
/* the lane shuffles */
|
||||
static const char* K_SHFLX_BPERM =
|
||||
"CHAIN(uint wl = AMD_WAVE_LANE; acc = acc ^ (uint)__builtin_amdgcn_ds_bpermute((int)((wl ^ 8u) << 2), (int)x))\n";
|
||||
static const char* K_SHFLX_SWZ =
|
||||
/* ds_swizzle bitmask mode: and_mask 0x1f, or_mask 0, xor_mask 8 -> offset 0x201f (lanes inside each 32-group) */
|
||||
"CHAIN(acc = acc ^ (uint)__builtin_amdgcn_ds_swizzle((int)x, 0x201f))\n";
|
||||
static const char* K_SHFLX_KHR =
|
||||
"#pragma OPENCL EXTENSION cl_khr_subgroup_shuffle : enable\n"
|
||||
"CHAIN(acc = acc ^ sub_group_shuffle_xor(x, 8u))\n";
|
||||
static const char* K_SHFLX_INTEL =
|
||||
"#pragma OPENCL EXTENSION cl_intel_subgroups : enable\n"
|
||||
"CHAIN(acc = acc ^ intel_sub_group_shuffle_xor(x, 8u))\n";
|
||||
static const char* K_SHFLX_LOCAL =
|
||||
"__kernel void probe(uint steps, uint seed, __global uint* out) {\n"
|
||||
" __local uint buf[256];\n"
|
||||
" CHAIN_HEAD\n"
|
||||
" for (uint s = 0u; s < steps; ++s) { buf[lid] = x; barrier(CLK_LOCAL_MEM_FENCE); uint v = buf[lid ^ 8u]; barrier(CLK_LOCAL_MEM_FENCE); acc = acc ^ v; CHAIN_TAIL }\n"
|
||||
" CHAIN_OUT\n"
|
||||
"}\n";
|
||||
static const char* K_SHFLA_BPERM =
|
||||
"CHAIN(uint wl = AMD_WAVE_LANE; acc = acc ^ (uint)__builtin_amdgcn_ds_bpermute((int)(((wl & ~31u) | ((wl + 3u) & 31u)) << 2), (int)x))\n";
|
||||
static const char* K_SHFLA_KHR =
|
||||
"#pragma OPENCL EXTENSION cl_khr_subgroup_shuffle : enable\n"
|
||||
"CHAIN(uint sl = get_sub_group_local_id(); acc = acc ^ sub_group_shuffle(x, (sl & ~31u) | ((sl + 3u) & 31u)))\n";
|
||||
static const char* K_SHFLA_INTEL =
|
||||
"#pragma OPENCL EXTENSION cl_intel_subgroups : enable\n"
|
||||
"CHAIN(uint sl = get_sub_group_local_id(); acc = acc ^ intel_sub_group_shuffle(x, (sl & ~31u) | ((sl + 3u) & 31u)))\n";
|
||||
static const char* K_SHFLA_LOCAL =
|
||||
"__kernel void probe(uint steps, uint seed, __global uint* out) {\n"
|
||||
" __local uint buf[256];\n"
|
||||
" CHAIN_HEAD\n"
|
||||
" for (uint s = 0u; s < steps; ++s) { buf[lid] = x; barrier(CLK_LOCAL_MEM_FENCE); uint v = buf[(lid & ~31u) | ((lid + 3u) & 31u)]; barrier(CLK_LOCAL_MEM_FENCE); acc = acc ^ v; CHAIN_TAIL }\n"
|
||||
" CHAIN_OUT\n"
|
||||
"}\n";
|
||||
/* dot4, signed bytes, wrapping accumulate (dp4a.s32.s32 semantics, the CUDA probe's dot4i row) */
|
||||
static const char* K_DOT4_C =
|
||||
"static inline uint dot4e(uint a, uint b, uint acc) { int4 va = convert_int4(as_char4(a)); int4 vb = convert_int4(as_char4(b)); return acc + (uint)(va.x * vb.x + va.y * vb.y + va.z * vb.z + va.w * vb.w); }\n"
|
||||
"CHAIN(acc = dot4e(x, y, acc))\n";
|
||||
static const char* K_DOT4_AMD =
|
||||
"CHAIN(acc = (uint)__builtin_amdgcn_sudot4(true, (int)x, true, (int)y, (int)acc, false))\n";
|
||||
static const char* K_DOT4_KHR =
|
||||
"#pragma OPENCL EXTENSION cl_khr_integer_dot_product : enable\n"
|
||||
"CHAIN(acc = acc + (uint)dot(as_char4(x), as_char4(y)))\n";
|
||||
/* mm8: one 16x16x16 int8 WMMA per step per lane through the clang builtins (gfx12 first, the gfx11 shape second).
|
||||
8 bytes of A and B per lane on gfx12 (int2), 16 (int4) on gfx11; 8 accumulators per lane (int8); the chain carries
|
||||
c.s0 forward as acc. No CPU reference for the fragment layout here: exact=unverified. */
|
||||
static const char* K_MM8_GFX12 =
|
||||
"__kernel void probe(uint steps, uint seed, __global uint* out) {\n"
|
||||
" CHAIN_HEAD int8 c = (int8)((int)acc, (int)pm_mix(acc), 0, 0, 0, 0, 0, 0);\n"
|
||||
" for (uint s = 0u; s < steps; ++s) { int2 a = (int2)((int)x, (int)y); int2 b = (int2)((int)(y ^ s), (int)x);\n"
|
||||
" c = __builtin_amdgcn_wmma_i32_16x16x16_iu8_w32_gfx12(true, a, true, b, c, false); acc = (uint)c.s0; CHAIN_TAIL }\n"
|
||||
" CHAIN_OUT\n"
|
||||
"}\n";
|
||||
static const char* K_MM8_GFX11 =
|
||||
"__kernel void probe(uint steps, uint seed, __global uint* out) {\n"
|
||||
" CHAIN_HEAD int8 c = (int8)((int)acc, (int)pm_mix(acc), 0, 0, 0, 0, 0, 0);\n"
|
||||
" for (uint s = 0u; s < steps; ++s) { int4 a = (int4)((int)x, (int)y, (int)(x ^ s), (int)(y + s)); int4 b = (int4)((int)(y ^ s), (int)x, (int)y, (int)x);\n"
|
||||
" c = __builtin_amdgcn_wmma_i32_16x16x16_iu8_w32(true, a, true, b, c, false); acc = (uint)c.s0; CHAIN_TAIL }\n"
|
||||
" CHAIN_OUT\n"
|
||||
"}\n";
|
||||
|
||||
typedef struct { const char* family; const char* variant; const char* path; const char* body; int vendorOnly; } Variant;
|
||||
/* vendorOnly: 0 = every platform, 1 = AMD builtins (clang), 2 = NVIDIA only (none here) */
|
||||
static const Variant VARIANTS[] = {
|
||||
{ "alu", "alu", "sequence", NULL, 0 },
|
||||
{ "rotr", "rotr_c", "sequence", NULL, 0 },
|
||||
{ "shflx", "shflx_bperm", "native", NULL, 1 },
|
||||
{ "shflx", "shflx_swz", "native", NULL, 1 },
|
||||
{ "shflx", "shflx_khr", "native", NULL, 0 },
|
||||
{ "shflx", "shflx_intel", "native", NULL, 0 },
|
||||
{ "shflx", "shflx_local", "emulated", NULL, 0 },
|
||||
{ "shl", "shl_c", "sequence", NULL, 0 },
|
||||
{ "shr", "shr_c", "sequence", NULL, 0 },
|
||||
{ "bfe", "bfe_amd", "native", NULL, 0 },
|
||||
{ "bfe", "bfe_c", "sequence", NULL, 0 },
|
||||
{ "andn", "andn_c", "sequence", NULL, 0 },
|
||||
{ "perm", "perm_amd", "native", NULL, 0 },
|
||||
{ "perm", "perm_c", "emulated", NULL, 0 },
|
||||
{ "popc", "popc_c", "sequence", NULL, 0 },
|
||||
{ "clz", "clz_c", "sequence", NULL, 0 },
|
||||
{ "sel", "sel_c", "sequence", NULL, 0 },
|
||||
{ "shfla", "shfla_bperm", "native", NULL, 1 },
|
||||
{ "shfla", "shfla_khr", "native", NULL, 0 },
|
||||
{ "shfla", "shfla_intel", "native", NULL, 0 },
|
||||
{ "shfla", "shfla_local", "emulated", NULL, 0 },
|
||||
{ "dot4", "dot4_amd", "native", NULL, 1 },
|
||||
{ "dot4", "dot4_khr", "native", NULL, 0 },
|
||||
{ "dot4", "dot4_c", "emulated", NULL, 0 },
|
||||
{ "mm8", "mm8_gfx12", "native", NULL, 1 },
|
||||
{ "mm8", "mm8_gfx11", "native", NULL, 1 },
|
||||
};
|
||||
static const int NVARIANTS = (int)(sizeof(VARIANTS) / sizeof(VARIANTS[0]));
|
||||
static const char* bodyOf(const char* variant) {
|
||||
if (!strcmp(variant, "alu")) return K_ALU;
|
||||
if (!strcmp(variant, "rotr_c")) return K_ROTR;
|
||||
if (!strcmp(variant, "shflx_bperm")) return K_SHFLX_BPERM;
|
||||
if (!strcmp(variant, "shflx_swz")) return K_SHFLX_SWZ;
|
||||
if (!strcmp(variant, "shflx_khr")) return K_SHFLX_KHR;
|
||||
if (!strcmp(variant, "shflx_intel")) return K_SHFLX_INTEL;
|
||||
if (!strcmp(variant, "shflx_local")) return K_SHFLX_LOCAL;
|
||||
if (!strcmp(variant, "shl_c")) return K_SHL;
|
||||
if (!strcmp(variant, "shr_c")) return K_SHR;
|
||||
if (!strcmp(variant, "bfe_amd")) return K_BFE_AMD;
|
||||
if (!strcmp(variant, "bfe_c")) return K_BFE_C;
|
||||
if (!strcmp(variant, "andn_c")) return K_ANDN;
|
||||
if (!strcmp(variant, "perm_amd")) return K_PERM_AMD;
|
||||
if (!strcmp(variant, "perm_c")) return K_PERM_C;
|
||||
if (!strcmp(variant, "popc_c")) return K_POPC;
|
||||
if (!strcmp(variant, "clz_c")) return K_CLZ;
|
||||
if (!strcmp(variant, "sel_c")) return K_SEL;
|
||||
if (!strcmp(variant, "shfla_bperm")) return K_SHFLA_BPERM;
|
||||
if (!strcmp(variant, "shfla_khr")) return K_SHFLA_KHR;
|
||||
if (!strcmp(variant, "shfla_intel")) return K_SHFLA_INTEL;
|
||||
if (!strcmp(variant, "shfla_local")) return K_SHFLA_LOCAL;
|
||||
if (!strcmp(variant, "dot4_amd")) return K_DOT4_AMD;
|
||||
if (!strcmp(variant, "dot4_khr")) return K_DOT4_KHR;
|
||||
if (!strcmp(variant, "dot4_c")) return K_DOT4_C;
|
||||
if (!strcmp(variant, "mm8_gfx12")) return K_MM8_GFX12;
|
||||
if (!strcmp(variant, "mm8_gfx11")) return K_MM8_GFX11;
|
||||
return NULL;
|
||||
}
|
||||
|
||||
/* ---- CPU reference: one whole 32-lane group (lanes g0 .. g0+31), the CUDA probe's warp_ref without mm8 ---- */
|
||||
static uint32_t pm_mix(uint32_t x) { x ^= x >> 16; x *= 0x7feb352du; x ^= x >> 15; x *= 0x846ca68bu; x ^= x >> 16; return x; }
|
||||
static uint32_t rotl32(uint32_t v, uint32_t n) { return (v << n) | (v >> (32u - n)); }
|
||||
static uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
|
||||
static uint32_t clz32(uint32_t x) { uint32_t n = 0; if (!x) return 32; while (!(x & 0x80000000u)) { x <<= 1; ++n; } return n; }
|
||||
static uint32_t popc32(uint32_t x) { uint32_t n = 0; while (x) { n += x & 1u; x >>= 1; } return n; }
|
||||
static uint32_t perm_ref(uint32_t y) { return ((y >> 8) & 0xffu) | (((y >> 24) & 0xffu) << 8) | ((y & 0xffu) << 16) | (((y >> 16) & 0xffu) << 24); }
|
||||
static uint32_t dot4s_ref(uint32_t a, uint32_t b, uint32_t acc) {
|
||||
int32_t r = (int32_t)acc; int i;
|
||||
for (i = 0; i < 4; ++i) { int32_t ba = (int32_t)(int8_t)((a >> (8 * i)) & 0xffu); int32_t bb = (int32_t)(int8_t)((b >> (8 * i)) & 0xffu); r = (int32_t)((uint32_t)r + (uint32_t)(ba * bb)); }
|
||||
return (uint32_t)r;
|
||||
}
|
||||
#define K32 0x9E3779B1u
|
||||
static int group_ref(const char* family, uint32_t g0, uint32_t seed, uint32_t steps, uint32_t* res) {
|
||||
uint32_t x[32], y[32], acc[32], xs[32], s; int l;
|
||||
for (l = 0; l < 32; ++l) { x[l] = pm_mix((g0 + (uint32_t)l) ^ seed); y[l] = x[l] ^ 0x5bd1e995u; acc[l] = pm_mix(x[l]); }
|
||||
if (!strcmp(family, "alu")) {
|
||||
for (s = 0; s < steps; ++s) for (l = 0; l < 32; ++l) { x[l] = x[l] * K32 + rotl32(y[l], 7u); y[l] = (y[l] ^ x[l]) + s; }
|
||||
for (l = 0; l < 32; ++l) res[l] = x[l] ^ y[l];
|
||||
return 1;
|
||||
}
|
||||
if (!strcmp(family, "mm8")) return 0; /* no reference: the WMMA fragment layout is not modelled here */
|
||||
for (s = 0; s < steps; ++s) {
|
||||
memcpy(xs, x, sizeof xs);
|
||||
for (l = 0; l < 32; ++l) {
|
||||
uint32_t xv = xs[l], yv = y[l];
|
||||
if (!strcmp(family, "rotr")) acc[l] = rotr_var(yv, xv);
|
||||
else if (!strcmp(family, "shflx")) acc[l] = acc[l] ^ xs[l ^ 8];
|
||||
else if (!strcmp(family, "shl")) acc[l] = yv << (xv & 31u);
|
||||
else if (!strcmp(family, "shr")) acc[l] = yv >> (xv & 31u);
|
||||
else if (!strcmp(family, "bfe")) acc[l] = (yv >> 7u) & 0x1fffu;
|
||||
else if (!strcmp(family, "andn")) acc[l] = yv & ~xv;
|
||||
else if (!strcmp(family, "perm")) acc[l] = perm_ref(yv);
|
||||
else if (!strcmp(family, "popc")) acc[l] = acc[l] + popc32(xv);
|
||||
else if (!strcmp(family, "clz")) acc[l] = acc[l] + clz32(xv);
|
||||
else if (!strcmp(family, "sel")) acc[l] = ((yv >> 5u) & 1u) ? xv : acc[l];
|
||||
else if (!strcmp(family, "shfla")) acc[l] = acc[l] ^ xs[(l + 3) & 31];
|
||||
else if (!strcmp(family, "dot4")) acc[l] = dot4s_ref(xv, yv, acc[l]);
|
||||
else { printf("no reference for %s\n", family); exit(3); }
|
||||
x[l] = xv * K32 + acc[l];
|
||||
y[l] = rotl32(yv, 7u) ^ (acc[l] + s);
|
||||
}
|
||||
}
|
||||
for (l = 0; l < 32; ++l) res[l] = acc[l] ^ x[l] ^ y[l];
|
||||
return 1;
|
||||
}
|
||||
|
||||
/* ---- devices ---- */
|
||||
typedef struct { cl_platform_id p; cl_device_id d; char pname[128], dname[128], driver[64], ver[64]; } Dev;
|
||||
static Dev devs[32]; static int ndevs = 0;
|
||||
static void enumerate(void) {
|
||||
cl_platform_id ps[8]; cl_uint np = 0, i;
|
||||
if (clGetPlatformIDs(8, ps, &np) != CL_SUCCESS) return;
|
||||
for (i = 0; i < np; ++i) {
|
||||
cl_device_id ds[8]; cl_uint nd = 0, j;
|
||||
if (clGetDeviceIDs(ps[i], CL_DEVICE_TYPE_GPU, 8, ds, &nd) != CL_SUCCESS) continue;
|
||||
for (j = 0; j < nd && ndevs < 32; ++j) {
|
||||
Dev* v = &devs[ndevs++]; v->p = ps[i]; v->d = ds[j];
|
||||
clGetPlatformInfo(ps[i], CL_PLATFORM_NAME, sizeof v->pname, v->pname, NULL);
|
||||
clGetDeviceInfo(ds[j], CL_DEVICE_NAME, sizeof v->dname, v->dname, NULL);
|
||||
clGetDeviceInfo(ds[j], CL_DRIVER_VERSION, sizeof v->driver, v->driver, NULL);
|
||||
clGetDeviceInfo(ds[j], CL_DEVICE_VERSION, sizeof v->ver, v->ver, NULL);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
typedef struct { double bestMs; int built; int exact; } Run; /* exact: 1 yes, 0 no, -1 unverified */
|
||||
|
||||
static Run run_variant(cl_context ctx, cl_command_queue q, cl_device_id dev, const Variant* v, cl_uint lanes, cl_uint steps, int reps,
|
||||
cl_mem out, uint32_t* host, const char* vendor, const char* dname, double aluBest) {
|
||||
const char* srcs[2]; cl_int err; cl_program prog; cl_kernel k; int r, exact = 1, hasRef = 1; double best = -1; Run res = { -1, 0, 0 };
|
||||
srcs[0] = COMMON; srcs[1] = v->body;
|
||||
prog = clCreateProgramWithSource(ctx, 2, srcs, NULL, &err);
|
||||
if (err != CL_SUCCESS) { printf("RESULT FAMILY name=%s variant=%s path=%s build=failed log=\"clCreateProgramWithSource %d\"\n", v->family, v->variant, v->path, (int)err); return res; }
|
||||
err = clBuildProgram(prog, 1, &dev, "-cl-std=CL1.2", NULL, NULL);
|
||||
if (err != CL_SUCCESS) {
|
||||
size_t n = 0; char* log; char* p; char* nl;
|
||||
clGetProgramBuildInfo(prog, dev, CL_PROGRAM_BUILD_LOG, 0, NULL, &n); log = (char*)calloc(n + 2, 1);
|
||||
if (n) clGetProgramBuildInfo(prog, dev, CL_PROGRAM_BUILD_LOG, n, log, NULL);
|
||||
p = log; while (*p == '\n' || *p == '\r' || *p == ' ') ++p;
|
||||
nl = strpbrk(p, "\r\n"); if (nl) *nl = 0;
|
||||
for (nl = p; *nl; ++nl) if (*nl == '"') *nl = '\'';
|
||||
printf("RESULT FAMILY name=%s variant=%s path=%s build=failed log=\"%.160s\"\n", v->family, v->variant, v->path, p);
|
||||
free(log); clReleaseProgram(prog); return res;
|
||||
}
|
||||
k = clCreateKernel(prog, "probe", &err);
|
||||
if (err != CL_SUCCESS) { printf("RESULT FAMILY name=%s variant=%s path=%s build=failed log=\"clCreateKernel %d\"\n", v->family, v->variant, v->path, (int)err); clReleaseProgram(prog); return res; }
|
||||
for (r = 0; r < reps; ++r) {
|
||||
cl_uint seed = 0x2468aceu + (cl_uint)r * 0x9E3779B9u; size_t global = lanes, local = 256; cl_event ev; cl_ulong t0, t1; double ms; uint32_t g0s[2]; int j;
|
||||
clSetKernelArg(k, 0, sizeof(cl_uint), &steps); clSetKernelArg(k, 1, sizeof(cl_uint), &seed); clSetKernelArg(k, 2, sizeof(cl_mem), &out);
|
||||
err = clEnqueueNDRangeKernel(q, k, 1, NULL, &global, &local, 0, NULL, &ev);
|
||||
if (err != CL_SUCCESS) { printf("RESULT FAMILY name=%s variant=%s path=%s build=failed log=\"launch failed %d\"\n", v->family, v->variant, v->path, (int)err); clReleaseKernel(k); clReleaseProgram(prog); return res; }
|
||||
clWaitForEvents(1, &ev);
|
||||
clGetEventProfilingInfo(ev, CL_PROFILING_COMMAND_START, sizeof t0, &t0, NULL); clGetEventProfilingInfo(ev, CL_PROFILING_COMMAND_END, sizeof t1, &t1, NULL);
|
||||
ms = (double)(t1 - t0) / 1e6; clReleaseEvent(ev);
|
||||
if (best < 0 || ms < best) best = ms;
|
||||
clEnqueueReadBuffer(q, out, CL_TRUE, 0, (size_t)lanes * 4, host, 0, NULL, NULL);
|
||||
g0s[0] = 0; g0s[1] = lanes - 32u;
|
||||
for (j = 0; j < 2; ++j) {
|
||||
uint32_t want[32]; int l;
|
||||
if (!group_ref(v->family, g0s[j], seed, steps, want)) { hasRef = 0; break; }
|
||||
for (l = 0; l < 32; ++l) if (host[g0s[j] + (uint32_t)l] != want[l]) { exact = 0; printf("MISMATCH %s lane %u: gpu %08x cpu %08x\n", v->variant, g0s[j] + (uint32_t)l, host[g0s[j] + (uint32_t)l], want[l]); break; }
|
||||
}
|
||||
}
|
||||
{
|
||||
double sps = (double)lanes * (double)steps / (best / 1000.0);
|
||||
double ratio = aluBest > 0 ? best / aluBest : 0;
|
||||
const char* ex = hasRef ? (exact ? "yes" : "no") : "unverified";
|
||||
printf("RESULT FAMILY name=%s ms=%.3f gsteps=%.2f ratio=%.3f exact=%s path=%s variant=%s ns=%.3f lanes=%u steps=%u vendor=%s device=\"%s\"\n",
|
||||
v->family, best, sps / 1e9, ratio, ex, v->path, v->variant, best * 1e6 / (double)steps, lanes, steps, vendor, dname);
|
||||
res.bestMs = best; res.built = 1; res.exact = hasRef ? exact : -1;
|
||||
}
|
||||
clReleaseKernel(k); clReleaseProgram(prog); return res;
|
||||
}
|
||||
|
||||
static int pathRank(const char* p) { return !strcmp(p, "native") ? 0 : !strcmp(p, "sequence") ? 1 : 2; }
|
||||
|
||||
int main(int argc, char** argv) {
|
||||
cl_uint lanes = 1u << 20, steps = 4096u; int reps = 3, device = 0, list = 0, i; Dev* v; cl_int err; cl_context ctx; cl_command_queue q; cl_mem out; uint32_t* host; const char* vendor; const char* only = NULL;
|
||||
Run runs[64]; double aluBest = 0; int isAmd;
|
||||
for (i = 1; i < argc; ++i) {
|
||||
if (!strcmp(argv[i], "--list")) list = 1;
|
||||
else if (!strcmp(argv[i], "--device") && i + 1 < argc) device = atoi(argv[++i]);
|
||||
else if (!strcmp(argv[i], "--lanes") && i + 1 < argc) lanes = (cl_uint)strtoul(argv[++i], 0, 10);
|
||||
else if (!strcmp(argv[i], "--steps") && i + 1 < argc) steps = (cl_uint)strtoul(argv[++i], 0, 10);
|
||||
else if (!strcmp(argv[i], "--reps") && i + 1 < argc) reps = atoi(argv[++i]);
|
||||
else if (!strcmp(argv[i], "--only") && i + 1 < argc) only = argv[++i];
|
||||
else { printf("unknown argument %s\n", argv[i]); return 2; }
|
||||
}
|
||||
if (lanes < 64 || (lanes & 255u)) { printf("--lanes must be a multiple of 256\n"); return 2; }
|
||||
#ifdef IGNEUM_CL_DYNAMIC
|
||||
if (!ig_cl_load()) { printf("%s\n", ig_cl_error); return 1; }
|
||||
#endif
|
||||
enumerate();
|
||||
if (list || ndevs == 0) { for (i = 0; i < ndevs; ++i) printf("[%d] %s | %s | driver %s | %s\n", i, devs[i].dname, devs[i].pname, devs[i].driver, devs[i].ver); if (ndevs == 0) printf("no OpenCL GPU devices\n"); return ndevs ? 0 : 1; }
|
||||
if (device < 0 || device >= ndevs) { printf("no device %d (have %d)\n", device, ndevs); return 2; }
|
||||
v = &devs[device];
|
||||
vendor = strstr(v->pname, "NVIDIA") ? "nvidia" : (strstr(v->pname, "AMD") ? "amd" : (strstr(v->pname, "Apple") ? "apple" : "other"));
|
||||
isAmd = !strcmp(vendor, "amd");
|
||||
ctx = clCreateContext(NULL, 1, &v->d, NULL, NULL, &err); if (err != CL_SUCCESS) { printf("clCreateContext %d\n", (int)err); return 1; }
|
||||
q = clCreateCommandQueue(ctx, v->d, CL_QUEUE_PROFILING_ENABLE, &err); if (err != CL_SUCCESS) { printf("clCreateCommandQueue %d\n", (int)err); return 1; }
|
||||
out = clCreateBuffer(ctx, CL_MEM_READ_WRITE, (size_t)lanes * 4, NULL, &err); if (err != CL_SUCCESS) { printf("clCreateBuffer %d\n", (int)err); return 1; }
|
||||
host = (uint32_t*)malloc((size_t)lanes * 4);
|
||||
printf("family-probe (OpenCL) on [%d] %s | %s | driver %s | %s, lanes %u, steps %u, best of %d, device event time\n", device, v->dname, v->pname, v->driver, v->ver, lanes, steps, reps);
|
||||
{
|
||||
char ext[16384]; ext[0] = 0; clGetDeviceInfo(v->d, CL_DEVICE_EXTENSIONS, sizeof ext, ext, NULL);
|
||||
printf("RESULT EXT device=\"%s\" khr_subgroups=%s khr_subgroup_shuffle=%s intel_subgroups=%s amd_media_ops2=%s khr_integer_dot_product=%s\n", v->dname,
|
||||
strstr(ext, "cl_khr_subgroups") ? "yes" : "no", strstr(ext, "cl_khr_subgroup_shuffle") ? "yes" : "no", strstr(ext, "cl_intel_subgroups") ? "yes" : "no",
|
||||
strstr(ext, "cl_amd_media_ops2") ? "yes" : "no", strstr(ext, "cl_khr_integer_dot_product") ? "yes" : "no");
|
||||
}
|
||||
for (i = 0; i < NVARIANTS; ++i) {
|
||||
Variant vv = VARIANTS[i]; vv.body = bodyOf(vv.variant);
|
||||
runs[i].bestMs = -1; runs[i].built = 0; runs[i].exact = 0;
|
||||
if (!vv.body) continue;
|
||||
if (only && !strstr(only, vv.family)) continue;
|
||||
if (vv.vendorOnly == 1 && !isAmd) { printf("RESULT FAMILY name=%s variant=%s path=%s build=skipped log=\"AMD clang builtin, platform is %s\"\n", vv.family, vv.variant, vv.path, vendor); continue; }
|
||||
runs[i] = run_variant(ctx, q, v->d, &vv, lanes, steps, reps, out, host, vendor, v->dname, aluBest);
|
||||
if (i == 0 && runs[i].built) aluBest = runs[i].bestMs;
|
||||
}
|
||||
/* the row per family: the fastest EXACT variant, native before sequence before emulated at equal exactness; a
|
||||
family with no exact variant takes its fastest unverified one and says so */
|
||||
for (i = 0; i < NVARIANTS; ++i) {
|
||||
int j, bestIdx = -1, bestRank = 9; double bestMs = 1e30;
|
||||
if (i > 0 && !strcmp(VARIANTS[i].family, VARIANTS[i - 1].family)) continue; /* one row per family, at its first variant */
|
||||
if (only && !strstr(only, VARIANTS[i].family)) continue;
|
||||
for (j = i; j < NVARIANTS && !strcmp(VARIANTS[j].family, VARIANTS[i].family); ++j) {
|
||||
int rank;
|
||||
if (!runs[j].built) continue;
|
||||
rank = (runs[j].exact == 1 ? 0 : runs[j].exact == -1 ? 3 : 6) + pathRank(VARIANTS[j].path);
|
||||
if (rank < bestRank || (rank == bestRank && runs[j].bestMs < bestMs)) { bestRank = rank; bestMs = runs[j].bestMs; bestIdx = j; }
|
||||
}
|
||||
if (bestIdx < 0) { printf("RESULT FAMILYBEST name=%s built=none\n", VARIANTS[i].family); continue; }
|
||||
printf("RESULT FAMILYBEST name=%s ms=%.3f gsteps=%.2f ratio=%.3f exact=%s path=%s variant=%s\n", VARIANTS[i].family, runs[bestIdx].bestMs,
|
||||
(double)lanes * (double)steps / (runs[bestIdx].bestMs / 1000.0) / 1e9, aluBest > 0 ? runs[bestIdx].bestMs / aluBest : 0,
|
||||
runs[bestIdx].exact == 1 ? "yes" : runs[bestIdx].exact == -1 ? "unverified" : "no", VARIANTS[bestIdx].path, VARIANTS[bestIdx].variant);
|
||||
}
|
||||
clReleaseMemObject(out); clReleaseCommandQueue(q); clReleaseContext(ctx); free(host);
|
||||
printf("family-probe: done\n");
|
||||
return 0;
|
||||
}
|
||||
|
|
@ -510,6 +510,7 @@ static const char* exchangeName(int m) { return m == 1 ? "sub_group_shuffle_xor
|
|||
// Returns 0 on success, 1 on build failure (log printed).
|
||||
static int buildProgram(Device* dv, const DeviceInfo* di, const char* src, size_t srcLen, int exchangeMode, int groupSize, const char* extra) {
|
||||
cl_int err = 0;
|
||||
double tb = wallMs(); /* the pack's compile cost (Counter ASIC 3.0 item 2, 6 October 2026): printed as one line below */
|
||||
const char* std;
|
||||
// The sub-group built-ins need OpenCL C 2.0 or 3.0. OpenCL 3.0 devices may report "OpenCL C 1.2" as the default
|
||||
// CL_DEVICE_OPENCL_C_VERSION while supporting 3.0 (the 3.0 API lists all versions; the 1.2 API cannot ask), so
|
||||
|
|
@ -534,6 +535,10 @@ static int buildProgram(Device* dv, const DeviceInfo* di, const char* src, size_
|
|||
clReleaseProgram(dv->prog); dv->prog = NULL;
|
||||
return 1;
|
||||
}
|
||||
/* The OpenCL build time of this pack's kernel text (clBuildProgram alone), the equivalent of the CUDA worker's
|
||||
* NVRTC line: a per-day item-derivation program (item 2) sits inside every hash-kernel compile, so its cost is
|
||||
* read here. Wall time, printed before the kernels are created. */
|
||||
printf("build %.1f ms clBuildProgram (exchange %d, group %d)\n", wallMs() - tb, exchangeMode, groupSize);
|
||||
dv->kHash = clCreateKernel(dv->prog, "igneum_hash", &err); CL_CHECK_ERR(err, "clCreateKernel igneum_hash");
|
||||
dv->kHashBound = clCreateKernel(dv->prog, "igneum_hash_bound", &err);
|
||||
if (err != CL_SUCCESS) dv->kHashBound = NULL; /* kernel.cl without the bound kernel: fine outside --serve */
|
||||
|
|
|
|||
64
tools/ca3-pc1-amd/README.md
Normal file
64
tools/ca3-pc1-amd/README.md
Normal file
|
|
@ -0,0 +1,64 @@
|
|||
# Counter ASIC 3.0: the PC 1 AMD jobs (6 October 2026)
|
||||
|
||||
Four signed `run` jobs for PC 1 (machine `ae432dc7`: RTX 5090, RTX 4070, RX 9070 XT `amd:gfx1201` on the eGPU, all on the installed app) that fill the OWED AMD rows of `docs/plans/counter-asic-3-status.md` sections 3, 5 and 7, plus the 4070 row of item 8. Prepared on branch `ca3-pc1-amd`; NOTHING here is published by this branch. The coordinator publishes on "go PC 1 AMD", one job at a time, in the order below, and reads each closing report before the next (`node tools/jobs.mjs <id>`, `node tools/jobs.mjs watch <id>`).
|
||||
|
||||
## Rules every script keeps
|
||||
|
||||
| Rule | How |
|
||||
|---|---|
|
||||
| One job at a time on PC 1 | the publish order below; each job's closing report read before the next |
|
||||
| The card alone only where the fixture needs it | job 1 and job 4 switch ONLY the card under test off through `POST <app.url>api/cards` (plain `run` jobs, not `--stop-miners`: the other two cards keep mining); jobs 2 and 3 run beside the miners and print `card_state=loaded|quiet` on every row |
|
||||
| Card keys from settings.json, never `/api/state`, BOTH forms | every form is posted: the key as written (`amd:1:gfx1201`, `nvidia:1:NVIDIA GeForce RTX 4070` or whatever the file holds), `vendor:index`, `vendor:code`; `apply_cards` (engine.rs) skips a key that matches no live card, so the extra forms are harmless and nothing is written for them |
|
||||
| Quiet confirmed by the process list | the worker process carrying the card's `--device <index>` must be gone (Win32_Process command lines; job 4 also reads `nvidia-smi -i <idx> --query-compute-apps`); an unconfirmed card prints `UNCONFIRMED` and the rows say `card_alone=False` |
|
||||
| Restored in `finally` | the card posted back enabled (only when settings.json had it enabled), the worker's return polled for 90 s and printed; the sampler ended by its own pid (never by name: the app runs its own copy of the AMD helper) |
|
||||
| The installed app untouched | no `/api/quit`, `/api/pause`, `/api/resume`, no manifest, no settings.json write; the only writes are POST `api/cards` on the card under test |
|
||||
| Copied sources re-stamped | nothing is built on the PC by these jobs (no cargo; the OpenCL kernels are compiled by the driver at run time); `tools/ci/copied-sources-check.sh` passes |
|
||||
| Prover socket | not applicable (no prover host in these jobs) |
|
||||
| Every failure a `RESULT ... error=<text>` line, a `SUMMARY {json}` line at the end | all four scripts |
|
||||
|
||||
If `--stop-miners` IS added at publish time, jobs 1 and 4 see the card already quiet, say so (`card already quiet before the switch`) and skip the switch: the scripts work either way.
|
||||
|
||||
## The kit (one fetch job, before job 1)
|
||||
|
||||
`tools/ca3-pc1-amd/make-kit.sh` builds `igneum-ca3-pc1-amd-kit.zip`: `bin/igneum-worker-opencl.exe` (THIS tree's `proto-opencl/host.c`, the installed worker's source plus one `build <ms> clBuildProgram` line, cross-compiled with mingw as `proto-cuda/nvrtc/build-windows.sh` does), `bin/family-probe-cl.exe` (`proto-opencl/family-probe.c`), `src/` (both sources and `cl_dynamic.h` for the sha256 record), `packs/` (mx8-devnet-epoch0, sh256x13, sh256x27, sh256x53, sh256x88, sh64x52, dr736-genesis, dr736-devnet-epoch0, each with its kernel_bound.cl, program.json and vectors), `SHA256SUMS`.
|
||||
|
||||
Built 6 October 2026, 15:4x UTC (`IGNEUM_REDIST=/Users/joshm/Projects/igneum-wt-ca2-mixer/proto-cuda/nvrtc/redist`): zip sha256 `a70fce5be672f33c3af08a5699488c09aef2848010f468ec99165fd4ab9ca61f`, 884,381 bytes, 113 files; `bin/igneum-worker-opencl.exe` `a572948e86c23317a80d9d7bb9e4bf70d5f7dfc40470974e2f08da1892016ac6`, `bin/family-probe-cl.exe` `d51a36bcae13ab36c40c2f347e3bdbe28f010499ccf84e423ecc04a9b4f151e3`, `src/host.c` `5e23ac94…`, `src/family-probe.c` `c293be9d…`. The zip is in the session scratchpad (`pc1amd/igneum-ca3-pc1-amd-kit.zip`), not in the repository; re-running `make-kit.sh` rebuilds it and prints a new sha256 (mingw output is not byte-reproducible), and the publish line takes the printed one. Every job prints `RESULT kitfile <path> sha256 <hex>` for the exes and sources it used, so the record closes on the PC side.
|
||||
|
||||
```
|
||||
tools/ca3-pc1-amd/make-kit.sh "$TMPDIR/igneum-ca3-pc1-amd-kit.zip"
|
||||
packaging/ota/publish-jobs.sh add --kind fetch --target ae432dc7 --id fetch-ca3-pc1-amd-20261006 \
|
||||
--file "$TMPDIR/igneum-ca3-pc1-amd-kit.zip" --dir jobs --extract --title "CA3 PC 1 AMD kit" --expires-hours 48 --deploy
|
||||
```
|
||||
|
||||
The fetch lands at `<jobs>\fetch-ca3-pc1-amd-20261006\{bin,packs,src,SHA256SUMS}` (jobrun.rs `fetch_base`: the job id under the jobs folder). The wiped-jobs-folder class: an app update clears the jobs folder, so if the 0.3.12 install lands on PC 1 between jobs, republish the fetch (each run job tests the kit before use and fails in seconds with `kit missing` if it is gone).
|
||||
|
||||
## The jobs, in order
|
||||
|
||||
| # | Script | Publish (after the fetch; `--id` fixed so the read-back is one command) | Length | Card | Rows it fills |
|
||||
|---|---|---|---|---|---|
|
||||
| 1 | `pc1-amd-g1-shadow.ps1` | `packaging/ota/publish-jobs.sh add --kind run --target ae432dc7 --id run-ca3-pc1-amd-g1-shadow-20261006 --script tools/ca3-pc1-amd/pc1-amd-g1-shadow.ps1 --timeout-minutes 40 --title "CA3 PC 1: G1 + AMD ladder on the 9070 XT" --deploy` | about 15 min (7 benches of 60 to 90 dispatches of 2^24 at 0.9 to 2 s each, the card switch up to 150 s, the installed-worker cross-check 40 dispatches) | 9070 XT ALONE (its worker stopped; 5090 and 4070 keep mining) | G1 for `mx8+sh256x27` on the AMD vendor (section 5: `RESULT G1 pack=sh256x27 fingerprint=3d2e8245cc084d07 match=yes` turns "AMD vendor NOT RUN" green; the other five packs give the same for the ladder); the 9070 XT column of the item 8 tables (section 3: `RESULT LADDER pack=<p> ops=<N> mhs= watts= uj= gclk_mhz=`), where the card leaves the latency bound by the 5 percent rule; item 1's AMD per-joule row from the control (`RESULT LADDER pack=mx8-devnet-epoch0 ... uj=`: section 3 "Consequences per tier (item 1)", 16 GB AMD row); section 7 rows "item 8: the 9070 XT rows", "item 1: the 9070 XT rows" |
|
||||
| 2 | `pc1-amd-family.ps1` | `packaging/ota/publish-jobs.sh add --kind run --target ae432dc7 --id run-ca3-pc1-amd-family-20261006 --script tools/ca3-pc1-amd/pc1-amd-family.ps1 --timeout-minutes 15 --title "CA3 PC 1: family step costs on the 9070 XT" --deploy` | about 3 min (3 runs x up to 3 AMD devices, 26 variants each, builds dominate) | beside the miners (ratios; every row says `card_state=`) | the "RX 9070 XT" column of the item 6 table (section 3: `RESULT FAMILYBEST name=<f> ratio=<r> exact= path=` per family, `RESULT FAMILY ... variant=` for the form each took: `shfla_bperm` = `ds_bpermute_b32`, the one number that could move R3; `shflx_swz` / `shflx_bperm`; `perm_amd` = `v_perm_b32` through `amd_perm` or `perm_c` emulated; `bfe_amd`; `dot4_amd` = the 5 October sudot4 row re-measured; `mm8_gfx12` / `mm8_gfx11` = the WMMA builtins, exact=unverified); section 7 row "item 6: the 9070 XT step costs" |
|
||||
| 3 | `pc1-amd-derive.ps1` | `packaging/ota/publish-jobs.sh add --kind run --target ae432dc7 --id run-ca3-pc1-amd-derive-20261006 --script tools/ca3-pc1-amd/pc1-amd-derive.ps1 --timeout-minutes 15 --title "CA3 PC 1: dr736 on the 9070 XT" --deploy` | about 4 min (3 packs x a 60-dispatch pass and a 3-dispatch pass) | beside the miners (build, compile and bit-exact rows; the rate as a ratio to mx8 in the same job) | the item 2 table's 9070 XT cells (section 3: `RESULT DERIVE pack=dr736-genesis build_ms= build_ms_cached= build_ms_over_mx8= dataset_ms= fingerprint= match= mhs= ratio_to_mx8=`): the OpenCL compile cost of the derivation program against mx8's (the NVRTC +1.1 s finding on AMD), the 1 GiB daily build against mx8's 72 to 77 ms (5 October), bit-exactness on the AMD vendor; section 7 row "item 2: the 9070 XT rows" |
|
||||
| 4 (optional, confirmed by main) | `pc1-4070-shadow.ps1` | `packaging/ota/publish-jobs.sh add --kind run --target ae432dc7 --id run-ca3-pc1-4070-shadow-20261006 --script tools/ca3-pc1-amd/pc1-4070-shadow.ps1 --timeout-minutes 30 --title "CA3 PC 1: item 8 ladder on the RTX 4070" --deploy` | about 11 min (7 benches of 100 to 200 dispatches at an expected 0.4 to 1 s each, approximate: no 4070 row exists yet; NVRTC per pack; the card switch) | 4070 ALONE (its CUDA worker stopped; 5090 and 9070 XT keep mining) | the "small NVIDIA card binds near 100,000" row of item 8 (section 3, consequences per tier at N = 100,000; section 7 "the 4060-class rows"): `RESULT LADDER pack=<p> ops=<N> mhs= watts= uj= sm_mhz=` with nvidia-smi at 1 Hz on that index; G1 fingerprints on a second NVIDIA card |
|
||||
|
||||
Read back: `node tools/jobs.mjs run-ca3-pc1-amd-g1-shadow-20261006` (and `--all` for the 5-minute progress uploads). The run_id on the intake is `job-<id>-ae432dc7`.
|
||||
|
||||
## The AMD watts readback on PC 1 (what exists)
|
||||
|
||||
`igneum-gpu-telemetry.exe` (proto-opencl/gpu-telemetry.c on master, shipped by `packaging/windows/make-payload.sh` since 0.3.10; the Ember Tune playbook found it in the install folder or under a `jobs\amd-kit-*\kit\` folder) prints one line per AMD card per sample: `amd <n> bus <pci> kind discrete name "AMD Radeon RX 9070 XT" watts <W> temp_c <C> fan_rpm <r> fan_pct <p> mclk_mhz <m> gclk_mhz <g> util_pct <u> source adlx`, then `end <ms> N card(s)`, at `-l 1` once a second with `fflush` per sample. The watts field is ADLX `GPUPower` with `GPUTotalBoardPower` as the fallback (gpu-telemetry.c lines 120 to 121 on master); on 5 October it read the 9070 XT at 198.9 W at 17.73 MH/s (release-0.3.10.md, job `tele-measure-1`), the figure the app's MH per watt line uses. So a board-watts readback EXISTS and job 1 uses it: a second copy of the helper at 1 Hz through a wrapper that stamps each line (UTC, ms) so the samples window onto each bench (after its first 12 s, before its last 2 s, the PC 2 shape), ended by the wrapper's pid in `finally`. The job prints `RESULT tele helper=<path> sha256=<hex> watts_source=adlx` after one sample; if the helper is missing or prints `watts -` for the 9070 XT the rows carry `watts=owed uj=owed` and the SUMMARY says `watts: owed`, never a guess. Whether ADLX `GPUPower` on RDNA 4 is total board power or ASIC power is not verified here: the row says `adlx` and the number is what the app itself reports.
|
||||
|
||||
## What the Mac tested (6 October 2026, 15:27 to 15:28 UTC, `with-lock.sh run`, load average 3.9 at the start)
|
||||
|
||||
| Check | Result |
|
||||
|---|---|
|
||||
| `proto-opencl/family-probe.c` on Apple OpenCL (`cc -framework OpenCL`, M5 Max) | every C-form and emulated row bit-exact on both 32-lane groups in 3 repetitions; ratios to the alu chain beside the Metal probe's (rotr 1.14 against Metal 1.13, shl 0.89 / 0.85, shr 0.89 / 0.86, bfe 0.82 / 0.77, andn 0.84 / 0.75, popc 1.00 / 0.87, clz 1.11 / 1.01, sel 0.82 / 0.76, perm emulated 1.52 / 1.13, shflx and shfla through `__local` 1.96 and 2.04 (Metal native 0.86 and 1.91), dot4 emulated signed 4.66 / 4.73); the Apple `dot(char4,char4)` row exact=no as the 5 October finding said; every AMD builtin skipped (`build=skipped`), the khr, intel and amd_media_ops2 forms `build=failed` with the log line, the run went on; `FAMILYBEST` picked the exact rows |
|
||||
| `proto-opencl/host.c` with the build line, `--bench-pack` on all eight kit packs on Apple OpenCL | self-test PASS (96 of 96 lanes) and the 2^24 fingerprint at base 0 equal to the recorded one on every pack: mx8-devnet-epoch0 `90f794dd556f7a3b`, sh256x13 `59ac286fe2a5a9ef`, sh256x27 `3d2e8245cc084d07`, sh256x53 `4f824b15cf2b124a`, sh256x88 `0572522e39a94d8a`, sh64x52 `9dd010f79d8ca9f4`, dr736-genesis `50e3eaa779da4f1e`, dr736-devnet-epoch0 `9553f6d5c667205a`; `build` line: mx8 70 ms, shadow packs 70 to 88 ms, dr736 337 to 356 ms (+270 ms on Apple OpenCL: item 2's compile cost shows on this harness too) |
|
||||
| mingw cross-builds of both exes | clean (`-Wall -Wextra`), 116 KB and 35 KB |
|
||||
| `tools/ci/bash-body-check.sh` on the four scripts | 0 bash bodies, all parse |
|
||||
| `tools/ci/kit-path-check.sh` on the four scripts | 4 of 4 kit paths checked before use |
|
||||
| `tools/ci/prover-socket-check.sh`, `copied-sources-check.sh`, `no-conflict-markers.sh` | pass |
|
||||
| The Windows PowerShell 5.1 parse (`windows.yml` parse job) | the real parser runs only on the Windows runner and only over `relay/playbooks` and the other listed folders (not `tools/`); on the Mac the same rule's approximation (`tools/ci/check-workflow-shell.mjs`'s drive-reference regex, applied to these four files by hand) fired once on `"$dev: --stop-miners"` and the line was fixed to `${dev}:`; 0 findings after; brace, paren and bracket depth 0 by the bash-body-check tokenizer |
|
||||
|
||||
## Not tested on the Mac (the PC side)
|
||||
|
||||
The PowerShell itself never ran (no PowerShell on this Mac): the `api/cards` switch and its process-list confirmation, the wrapper sampler, `Get-CimInstance` command-line matching, the AMD helper's presence and its watts on the 9070 XT, the AMD OpenCL compiler's answer to each builtin (`ds_bpermute`, `ds_swizzle`, `amd_perm`, `amd_bfe`, the WMMA builtins), the driver's kernel cache (pass 2 of job 3), the 4070's nvidia-smi index against the app's `--device`, and the lengths above (estimates from the PC 2 runs and the 5 October 9070 XT rates). A failure of any of those prints a `RESULT ... error=` line and the SUMMARY carries `failed` or `partial`; the card restore in `finally` does not depend on them.
|
||||
43
tools/ca3-pc1-amd/make-kit.sh
Executable file
43
tools/ca3-pc1-amd/make-kit.sh
Executable file
|
|
@ -0,0 +1,43 @@
|
|||
#!/usr/bin/env bash
|
||||
# Builds the PC 1 AMD kit for the Counter ASIC 3.0 jobs in this folder (6 October 2026): the OpenCL worker from
|
||||
# THIS tree's proto-opencl/host.c (the installed worker's source plus the `build <ms> clBuildProgram` line), the
|
||||
# OpenCL family probe (proto-opencl/family-probe.c), both cross-compiled with mingw exactly as
|
||||
# proto-cuda/nvrtc/build-windows.sh builds the shipped worker (static, IGNEUM_CL_DYNAMIC, OpenCL.dll at run time;
|
||||
# no icon or version block: this is a bench kit, not a shipped exe), the eight packs the four jobs read, the two
|
||||
# sources for the sha256 record, and SHA256SUMS. Prints the zip's sha256 and size: the README's publish line takes it.
|
||||
# tools/ca3-pc1-amd/make-kit.sh [out.zip] default $TMPDIR/igneum-ca3-pc1-amd-kit.zip
|
||||
# Needs x86_64-w64-mingw32-gcc (brew install mingw-w64) and the Khronos OpenCL headers at $IGNEUM_REDIST/include
|
||||
# (default proto-cuda/nvrtc/redist, written by proto-cuda/nvrtc/fetch-redist.sh; the ca2-mixer worktree holds a copy:
|
||||
# IGNEUM_REDIST=/Users/joshm/Projects/igneum-wt-ca2-mixer/proto-cuda/nvrtc/redist). Nothing third-party is committed.
|
||||
set -euo pipefail
|
||||
HERE="$(cd "$(dirname "$0")" && pwd)"
|
||||
ROOT="$(cd "$HERE/../.." && pwd)"
|
||||
OUT="${1:-${TMPDIR:-/tmp}/igneum-ca3-pc1-amd-kit.zip}"
|
||||
RED="${IGNEUM_REDIST:-$ROOT/proto-cuda/nvrtc/redist}"
|
||||
[ -f "$RED/include/CL/cl.h" ] || { echo "no OpenCL headers at $RED/include/CL/cl.h (run proto-cuda/nvrtc/fetch-redist.sh or set IGNEUM_REDIST)" >&2; exit 1; }
|
||||
CC=x86_64-w64-mingw32-gcc; STRIP=x86_64-w64-mingw32-strip
|
||||
command -v "$CC" >/dev/null || { echo "$CC not found (brew install mingw-w64)" >&2; exit 1; }
|
||||
PLACEHOLDER="$ROOT/proto-cuda/packs/igneum-devnet-v4-epoch0" # the memory-hard placeholder build-windows.sh uses; --bench-pack reads the real pack at run time
|
||||
STAGE="$(mktemp -d)"
|
||||
mkdir -p "$STAGE/bin" "$STAGE/src" "$STAGE/packs"
|
||||
echo "== igneum-worker-opencl.exe (kit build of proto-opencl/host.c)"
|
||||
"$CC" -std=c99 -O2 -Wall -Wextra -Wno-stringop-truncation -Wno-format-truncation -static -DIGNEUM_CL_DYNAMIC -DCL_TARGET_OPENCL_VERSION=120 \
|
||||
-I "$RED/include" -I "$PLACEHOLDER" -DIGNEUM_KERNEL_PATH='"kernel_bound.cl"' -o "$STAGE/bin/igneum-worker-opencl.exe" "$ROOT/proto-opencl/host.c"
|
||||
"$STRIP" "$STAGE/bin/igneum-worker-opencl.exe"
|
||||
echo "== family-probe-cl.exe (proto-opencl/family-probe.c)"
|
||||
"$CC" -std=c99 -O2 -Wall -Wextra -static -DIGNEUM_CL_DYNAMIC -DCL_TARGET_OPENCL_VERSION=120 -I "$RED/include" -o "$STAGE/bin/family-probe-cl.exe" "$ROOT/proto-opencl/family-probe.c"
|
||||
"$STRIP" "$STAGE/bin/family-probe-cl.exe"
|
||||
cp "$ROOT/proto-opencl/host.c" "$ROOT/proto-opencl/family-probe.c" "$ROOT/proto-opencl/cl_dynamic.h" "$STAGE/src/"
|
||||
for p in packs-ca2-mixer/mx8-devnet-epoch0 packs-ca3-shadow/sh256x13 packs-ca3-shadow/sh256x27 packs-ca3-shadow/sh256x53 packs-ca3-shadow/sh256x88 packs-ca3-shadow/sh64x52 packs-ca3-derive/dr736-genesis packs-ca3-derive/dr736-devnet-epoch0; do
|
||||
name="$(basename "$p")"
|
||||
[ -f "$ROOT/proto-cuda/$p/kernel_bound.cl" ] || { echo "pack $p has no kernel_bound.cl" >&2; exit 1; }
|
||||
mkdir -p "$STAGE/packs/$name"
|
||||
cp "$ROOT/proto-cuda/$p"/* "$STAGE/packs/$name/"
|
||||
done
|
||||
( cd "$STAGE" && find . -type f ! -name SHA256SUMS | sort | xargs shasum -a 256 > SHA256SUMS )
|
||||
rm -f "$OUT"
|
||||
( cd "$STAGE" && zip -X -q -r "$OUT" . )
|
||||
rm -rf "$STAGE"
|
||||
echo "kit $OUT"
|
||||
echo "sha256 $(shasum -a 256 "$OUT" | cut -c1-64) bytes $(wc -c < "$OUT" | tr -d ' ')"
|
||||
unzip -l "$OUT" | tail -1
|
||||
193
tools/ca3-pc1-amd/pc1-4070-shadow.ps1
Normal file
193
tools/ca3-pc1-amd/pc1-4070-shadow.ps1
Normal file
|
|
@ -0,0 +1,193 @@
|
|||
# Counter ASIC 3.0, PC 1 job 4 (6 October 2026, optional, confirmed by main at 16:xx UTC): item 8's ladder on PC 1's
|
||||
# RTX 4070 (machine ae432dc7) through the INSTALLED igneum-worker-cuda.exe (NVRTC, the pack's own kernel text, the
|
||||
# same harness as the PC 2 job tools/ca3-shadow/pc2-shadow-bench.ps1), nvidia-smi at 1 Hz on that card alone
|
||||
# (power.draw, clocks.sm, clocks.mem, utilization, temperature): the "small NVIDIA card binds near 100,000 ops" row of
|
||||
# the status file (section 3 "Item 8", section 7). Published as a plain signed `run` job (NOT --stop-miners): the
|
||||
# RTX 5090 and the RX 9070 XT keep mining; this script switches ONLY the 4070 off in the installed app through POST
|
||||
# <app.url>api/cards with its key read from settings.json (never api/state) in every form (as written, vendor:index,
|
||||
# vendor:name; the 6 October finding), confirms by the process list that the CUDA worker carrying the 4070's --device
|
||||
# index is gone and by nvidia-smi's compute-apps list on that index, benches, and restores the card in a finally block
|
||||
# whatever happens. It never quits, pauses, resumes or updates the installed app, never writes settings.json, sets no
|
||||
# power cap and no clock (nothing elevated). The 4070's nvidia-smi index is read from nvidia-smi by name; the app
|
||||
# passes that index as the worker's --device (detect.rs), and the worker's own RESULT line names the device it ran on,
|
||||
# which the script checks for "4070" before it takes a row. Packs: mx8-devnet-epoch0 (the class v3 control, first and
|
||||
# last), sh256x13, sh256x27, sh256x53, sh256x88, sh64x52. Lines: `RESULT G1 ...` (the fingerprint against the Mac's
|
||||
# and the 5090's), `RESULT LADDER pack=<name> ops=<N> mhs=<x> watts=<y> uj=<z> sm_mhz=<m> ...`, a `SUMMARY {json}` line.
|
||||
# Read back with `node tools/jobs.mjs <job id>`. The kit: fetch job fetch-ca3-pc1-amd-20261006 (the packs).
|
||||
$ErrorActionPreference = 'Continue'
|
||||
$jobName = 'pc1-4070-shadow'
|
||||
$kitId = 'fetch-ca3-pc1-amd-20261006'
|
||||
$started = Get-Date
|
||||
function Stamp { (Get-Date).ToUniversalTime().ToString('yyyy-MM-ddTHH:mm:ssZ') }
|
||||
function Summary([string] $status, [hashtable] $extra) {
|
||||
$o = [ordered]@{ job = $jobName; status = $status; duration_s = [int]((Get-Date) - $started).TotalSeconds; finished_at = (Stamp) }
|
||||
foreach ($k in $extra.Keys) { $o[$k] = $extra[$k] }
|
||||
'SUMMARY ' + ($o | ConvertTo-Json -Compress -Depth 4)
|
||||
}
|
||||
$expected = @{ 'mx8-devnet-epoch0' = '90f794dd556f7a3b'; 'sh256x13' = '59ac286fe2a5a9ef'; 'sh256x27' = '3d2e8245cc084d07'; 'sh256x53' = '4f824b15cf2b124a'; 'sh256x88' = '0572522e39a94d8a'; 'sh64x52' = '9dd010f79d8ca9f4' }
|
||||
$opsPerHash = @{ 'mx8-devnet-epoch0' = 930; 'sh256x13' = 49700; 'sh256x27' = 102100; 'sh256x53' = 199600; 'sh256x88' = 330700; 'sh64x52' = 49700 }
|
||||
# timed dispatches of 2^24 nonces (wall time, as the PC 2 job): a 4070-class card is expected near 40 to 50 MH/s on
|
||||
# the control (approximate, no 4070 row exists yet), so 200 dispatches are about 70 to 85 s; fewer where the ALU budget binds
|
||||
$batches = @{ 'mx8-devnet-epoch0' = 200; 'sh256x13' = 200; 'sh256x27' = 180; 'sh256x53' = 150; 'sh256x88' = 100; 'sh64x52' = 200 }
|
||||
$order = @('mx8-devnet-epoch0', 'sh256x13', 'sh256x27', 'sh256x53', 'sh256x88', 'sh64x52', 'mx8-devnet-epoch0')
|
||||
"RESULT start $(Stamp) job=$jobName machine=$env:COMPUTERNAME app_version=$env:IGNEUM_APP_VERSION"
|
||||
$jobs = Split-Path $env:IGNEUM_JOB_DIR
|
||||
$kit = Join-Path $jobs $kitId
|
||||
if (-not (Test-Path $kit)) { "RESULT error kit missing at $kit (the fetch job $kitId runs first; republish it after any app update)"; Summary 'failed' @{ error = 'kit missing' }; exit 2 }
|
||||
$packs = Join-Path $kit 'packs'
|
||||
if (-not (Test-Path $packs)) { "RESULT error packs missing at $packs"; Summary 'failed' @{ error = 'packs missing' }; exit 2 }
|
||||
$job = $env:IGNEUM_JOB_DIR; if (-not $job) { $job = Join-Path $env:TEMP 'igneum-ca3-pc1-4070' }; New-Item -ItemType Directory -Force -Path $job | Out-Null
|
||||
$inst = @("$env:LOCALAPPDATA\Programs\Igneum Miner", "$env:ProgramFiles\Igneum Miner") | Where-Object { Test-Path (Join-Path $_ 'igneum-worker-cuda.exe') } | Select-Object -First 1
|
||||
if (-not $inst) { "RESULT error no installed igneum-worker-cuda.exe"; Summary 'failed' @{ error = 'no cuda worker' }; exit 2 }
|
||||
$exe = Join-Path $inst 'igneum-worker-cuda.exe'
|
||||
"RESULT worker $exe sha256 $((Get-FileHash -Algorithm SHA256 $exe).Hash.ToLower()) bytes $((Get-Item $exe).Length) nvrtc_dlls $((Get-ChildItem $inst -Filter 'nvrtc*.dll').Count)"
|
||||
$appDir = if ($env:IGNEUM_APP_DIR) { $env:IGNEUM_APP_DIR } else { Join-Path $env:LOCALAPPDATA 'igneum\app' }
|
||||
$urlFile = Join-Path $appDir 'app.url'
|
||||
$base = $null
|
||||
if (Test-Path $urlFile) { $base = (Get-Content -LiteralPath $urlFile -Raw).Trim().TrimEnd('/') }
|
||||
|
||||
# ---- the 4070 by name: nvidia-smi index (= the app's --device for the CUDA worker, detect.rs) ----
|
||||
function Smi([string] $q, $idx) { try { $a = @('--query-gpu=' + $q, '--format=csv,noheader,nounits'); if ($null -ne $idx) { $a += @('-i', "$idx") }; (& nvidia-smi @a 2>$null | ForEach-Object { "$_" }) -join ' | ' } catch { 'nvidia-smi failed' } }
|
||||
$gpus = @(& nvidia-smi --query-gpu=index,name,pci.bus_id,memory.total --format=csv,noheader 2>$null | ForEach-Object { "$_" })
|
||||
$gpus | ForEach-Object { "RESULT gpu $_" }
|
||||
$idx = $null; $gpuName = ''
|
||||
foreach ($g in $gpus) { if ($g -match '^\s*(\d+),\s*([^,]*4070[^,]*),') { $idx = [int]$Matches[1]; $gpuName = $Matches[2].Trim(); break } }
|
||||
if ($null -eq $idx) { "RESULT error no RTX 4070 in nvidia-smi's list"; Summary 'failed' @{ error = 'no 4070' }; exit 2 }
|
||||
"RESULT device nvidia-smi index $idx name '$gpuName' limits $(Smi 'power.limit,power.min_limit,power.max_limit,clocks.max.sm' $idx)"
|
||||
|
||||
# ---- the card's keys from settings.json (READ ONLY), every form ----
|
||||
$cardKeys = @(); $cardEnabled = $false; $cardIdent = 1; $cardPower = 0
|
||||
try {
|
||||
$sj = Get-Content -LiteralPath (Join-Path $appDir 'settings.json') -Raw | ConvertFrom-Json
|
||||
foreach ($prop in $sj.cards.PSObject.Properties) {
|
||||
if ($prop.Name -like 'nvidia:*' -and $prop.Name -match '4070') {
|
||||
$parts = $prop.Name -split ':'
|
||||
$cardKeys += $prop.Name
|
||||
if ($parts.Count -ge 3) { $cardKeys += ($parts[0] + ':' + $parts[1]); $cardKeys += ($parts[0] + ':' + (($parts | Select-Object -Skip 2) -join ':')) }
|
||||
$cardEnabled = [bool] $prop.Value.enabled
|
||||
if ($prop.Value.identities) { $cardIdent = [int] $prop.Value.identities }
|
||||
if ($prop.Value.power_pct) { $cardPower = [int] $prop.Value.power_pct }
|
||||
"RESULT card settings_key=$($prop.Name) enabled=$cardEnabled identities=$cardIdent power_pct=$cardPower forms=[$($cardKeys -join ' | ')]"
|
||||
}
|
||||
}
|
||||
} catch { "RESULT settings_read_failed $($_.Exception.Message)" }
|
||||
if ($cardKeys.Count -eq 0) { "RESULT card none in settings.json matching nvidia:*4070*: no switch; the rows below carry whatever load the card has" }
|
||||
function PostCards([bool] $enabled) {
|
||||
if (-not $base) { "RESULT cards no app.url: nothing posted"; return }
|
||||
foreach ($k in $cardKeys) {
|
||||
$c = @{ key = $k; enabled = $enabled; identities = $cardIdent }
|
||||
if ($cardPower -gt 0) { $c.power_pct = $cardPower }
|
||||
$body = @{ cards = @($c) } | ConvertTo-Json -Depth 5
|
||||
try { Invoke-RestMethod -Uri "$base/api/cards" -Method POST -Body $body -ContentType 'application/json' -TimeoutSec 10 | Out-Null; "RESULT cards key=$k enabled=$enabled posted" } catch { "RESULT cards key=$k enabled=$enabled error=$($_.Exception.Message)" }
|
||||
}
|
||||
}
|
||||
function Workers { @(Get-CimInstance Win32_Process -Filter "Name = 'igneum-worker-opencl.exe' OR Name = 'igneum-worker-cuda.exe'" -ErrorAction SilentlyContinue | ForEach-Object { "$($_.Name):$($_.ProcessId):[$($_.CommandLine -replace '\s+', ' ')]" }) }
|
||||
function CardWorkers { @(Get-CimInstance Win32_Process -Filter "Name = 'igneum-worker-cuda.exe'" -ErrorAction SilentlyContinue | Where-Object { $_.CommandLine -match "--device\s+$idx(\s|$)" } | ForEach-Object { $_.ProcessId }) }
|
||||
function CudaApps { @(& nvidia-smi -i $idx --query-compute-apps=pid,process_name,used_memory --format=csv,noheader 2>$null | ForEach-Object { "$_" } | Where-Object { $_ -match 'igneum|sp1|prove' }) }
|
||||
"RESULT workers_before $(Stamp) $((Workers) -join ' ') cuda_apps_on_$idx=[$((CudaApps) -join '; ')]"
|
||||
|
||||
$samples = Join-Path $job 'smi-samples.csv'
|
||||
$sampler = $null
|
||||
$results = @()
|
||||
$cardAlone = $false
|
||||
$switched = $false
|
||||
try {
|
||||
if ($cardKeys.Count -gt 0 -and $base) {
|
||||
$before = @(CardWorkers)
|
||||
if ($before.Count -eq 0 -and (@(CudaApps)).Count -eq 0) { "RESULT card already quiet before the switch (no cuda worker with --device $idx, no compute app on index $idx)"; $cardAlone = $true }
|
||||
else {
|
||||
PostCards $false; $switched = $true
|
||||
$t = 0
|
||||
while ($t -lt 150) { Start-Sleep -Seconds 5; $t += 5; if ((@(CardWorkers)).Count -eq 0 -and (@(CudaApps)).Count -eq 0) { break } }
|
||||
$left = @(CardWorkers); $apps = @(CudaApps)
|
||||
if ($left.Count -eq 0 -and $apps.Count -eq 0) { "RESULT card $(Stamp) quiet after $t s: no cuda worker with --device $idx and no igneum compute app on index $idx (the card to itself)"; $cardAlone = $true }
|
||||
else { "RESULT card $(Stamp) UNCONFIRMED after $t s: worker pids [$($left -join ' ')] compute apps [$($apps -join '; ')]; the rows below are LOADED-card figures" }
|
||||
Start-Sleep -Seconds 5
|
||||
}
|
||||
} else { "RESULT card not switched (no key or no app.url): the rows below carry whatever load the card has" }
|
||||
"RESULT workers_during $(Stamp) $((Workers) -join ' ')"
|
||||
"RESULT gpu_before $(Stamp) $(Smi 'power.draw,power.limit,clocks.sm,clocks.mem,temperature.gpu,memory.used' $idx)"
|
||||
# the sampler: nvidia-smi on this index every second into a file (its own timestamps), ended in finally
|
||||
$smiExe = (Get-Command nvidia-smi -ErrorAction SilentlyContinue).Source
|
||||
if (-not $smiExe) { $smiExe = Join-Path $env:SystemRoot 'System32\nvidia-smi.exe' }
|
||||
Remove-Item -LiteralPath $samples -Force -ErrorAction SilentlyContinue
|
||||
$sampler = Start-Process -FilePath $smiExe -ArgumentList @('-i', "$idx", '--query-gpu=timestamp,power.draw,utilization.gpu,clocks.sm,clocks.mem,temperature.gpu,memory.used', '--format=csv,noheader,nounits', '-l', '1') -RedirectStandardOutput $samples -NoNewWindow -PassThru
|
||||
Start-Sleep -Seconds 12
|
||||
"RESULT idle $(Stamp) 12 s of idle samples before the first bench: $(Smi 'power.draw,utilization.gpu,clocks.sm' $idx)"
|
||||
$i = 0
|
||||
foreach ($pk in $order) {
|
||||
$i++
|
||||
$d = Join-Path $packs $pk
|
||||
if (-not (Test-Path $d)) { "RESULT G1 pack=$pk error=pack missing at $d"; "RESULT LADDER pack=$pk error=pack missing"; continue }
|
||||
$instrs = 0; $cls = ''
|
||||
try { $pj = Get-Content -LiteralPath (Join-Path $d 'program.json') -Raw | ConvertFrom-Json; $cls = [string] $pj.load_class; if ($pj.shadow) { $instrs = [int] $pj.shadow.instrs_per_hash } } catch { }
|
||||
$n = $batches[$pk]
|
||||
$tag = "$pk#$i"
|
||||
$t0 = Get-Date
|
||||
"RESULT bench $tag start $(Stamp) batches=$n class=$cls shadow_instrs_per_hash=$instrs"
|
||||
$lines = @(& $exe --bench --pack $d --batches $n --batch-log2 24 --block-warps 1 --device $idx 2>&1 | ForEach-Object { "$_" })
|
||||
$rc = $LASTEXITCODE
|
||||
$t1 = Get-Date
|
||||
$lines | ForEach-Object { "RESULT bench $tag $_" }
|
||||
"RESULT bench $tag end $(Stamp) exit=$rc wall_s=$([int]($t1 - $t0).TotalSeconds)"
|
||||
$res = $lines | Where-Object { $_ -match '^RESULT pack=' } | Select-Object -First 1
|
||||
$fp = ''; $mhs = ''; $check = ''; $devName = ''; $nvrtc = ''; $ds = ''
|
||||
if ($res -and $res -match 'fingerprint=([0-9a-f]{16})') { $fp = $Matches[1] }
|
||||
if ($res -and $res -match 'mhs=([\d.]+)') { $mhs = $Matches[1] }
|
||||
if ($res -and $res -match 'check=(\S+)') { $check = $Matches[1] }
|
||||
if ($res -and $res -match 'device=(\S+)') { $devName = $Matches[1] }
|
||||
$tl = $lines | Where-Object { $_ -match 'nvrtc (\d+) cache \d+ dataset (\d+)' } | Select-Object -First 1
|
||||
if ($tl -and $tl -match 'nvrtc (\d+) cache \d+ dataset (\d+)') { $nvrtc = $Matches[1]; $ds = $Matches[2] }
|
||||
if (-not $res -or $rc -ne 0) {
|
||||
$err = ($lines | Where-Object { $_ -match 'FAIL|error|MISMATCH' } | Select-Object -First 1); if (-not $err) { $err = "exit $rc, no RESULT line" }
|
||||
"RESULT G1 pack=$pk error=$($err -replace '\s+', ' ')"; "RESULT LADDER pack=$pk error=$($err -replace '\s+', ' ')"; continue
|
||||
}
|
||||
if ($devName -notmatch '4070') { "RESULT G1 pack=$pk error=the worker ran on device '$devName', not the 4070 (CUDA ordinal $idx is another card: the row is not taken)"; "RESULT LADDER pack=$pk error=wrong device $devName"; continue }
|
||||
$want = $expected[$pk]
|
||||
$match = if ($fp -eq $want) { 'yes' } else { 'no' }
|
||||
"RESULT G1 pack=$pk fingerprint=$fp match=$match expected=$want check=$check card=RTX4070 harness=cuda-installed"
|
||||
$results += [pscustomobject]@{ tag = $tag; pack = $pk; instrs = $instrs; ops = $opsPerHash[$pk]; mhs = $mhs; fp = $fp; match = $match; check = $check; nvrtc_ms = $nvrtc; dataset_ms = $ds; dev = $devName; t0 = $t0; t1 = $t1 }
|
||||
Start-Sleep -Seconds 3
|
||||
}
|
||||
} finally {
|
||||
if ($sampler) { try { Stop-Process -Id $sampler.Id -Force -ErrorAction SilentlyContinue; "RESULT sampler ended pid=$($sampler.Id)" } catch { } }
|
||||
if ($switched) {
|
||||
if ($cardEnabled) { PostCards $true; "RESULT card restore posted enabled=true (settings.json had it enabled)" } else { "RESULT card restore skipped: settings.json had the card disabled before this job" }
|
||||
$t = 0; $back = @()
|
||||
while ($t -lt 90 -and $cardEnabled) { Start-Sleep -Seconds 5; $t += 5; $back = @(CardWorkers); if ($back.Count -gt 0) { break } }
|
||||
"RESULT card_workers_after $(Stamp) pids=[$($back -join ' ')] after $t s $(if ($back.Count -gt 0) { '(the card mines again)' } elseif ($cardEnabled) { '(NOT back yet: check the app)' } else { '' })"
|
||||
}
|
||||
"RESULT workers_after $(Stamp) $((Workers) -join ' ')"
|
||||
"RESULT gpu_after $(Stamp) $(Smi 'power.draw,power.limit,clocks.sm,clocks.mem,temperature.gpu,memory.used' $idx)"
|
||||
}
|
||||
# ---- the samples: the mean over each bench window after its first 12 s (NVRTC, fills, warm-up) and before its last 2 s ----
|
||||
$rows = @()
|
||||
if (Test-Path $samples) {
|
||||
foreach ($l in (Get-Content -LiteralPath $samples)) {
|
||||
$p = $l -split ',\s*'
|
||||
if ($p.Count -lt 7) { continue }
|
||||
try { $ts = [datetime]::ParseExact($p[0].Trim(), 'yyyy/MM/dd HH:mm:ss.fff', $null) } catch { continue }
|
||||
try { $rows += [pscustomobject]@{ ts = $ts; w = [double]$p[1]; util = [double]$p[2]; sm = [double]$p[3]; mem = [double]$p[4]; temp = [double]$p[5]; used = [double]$p[6] } } catch { }
|
||||
}
|
||||
}
|
||||
"RESULT samples total $($rows.Count) file $samples"
|
||||
$g1Match = 0
|
||||
foreach ($r in $results) {
|
||||
$in = @($rows | Where-Object { $_.ts -gt $r.t0.AddSeconds(12) -and $_.ts -lt $r.t1.AddSeconds(-2) })
|
||||
$watts = 'owed'; $uj = 'owed'; $sm = 'owed'; $mem = 'owed'; $tmax = 'owed'; $util = 'owed'; $wmin = ''; $wmax = ''
|
||||
if ($in.Count -gt 0) {
|
||||
$m = $in | Measure-Object -Property w -Average -Minimum -Maximum
|
||||
$watts = [math]::Round($m.Average, 1); $wmin = $m.Minimum; $wmax = $m.Maximum
|
||||
$sm = [math]::Round(($in | Measure-Object -Property sm -Average).Average); $mem = [math]::Round(($in | Measure-Object -Property mem -Average).Average)
|
||||
$util = [math]::Round(($in | Measure-Object -Property util -Average).Average, 1); $tmax = ($in | Measure-Object -Property temp -Maximum).Maximum
|
||||
if ($r.mhs -and [double] $r.mhs -gt 0) { $uj = [math]::Round($m.Average / [double] $r.mhs, 3) }
|
||||
}
|
||||
if ($r.match -eq 'yes') { $g1Match++ }
|
||||
"RESULT LADDER pack=$($r.pack) ops=$($r.ops) shadow_instrs=$($r.instrs) mhs=$($r.mhs) watts=$watts uj=$uj watts_min=$wmin watts_max=$wmax sm_mhz=$sm mem_mhz=$mem util=$util temp_max=$tmax samples=$($in.Count) window_s=$([int]($r.t1 - $r.t0).TotalSeconds) nvrtc_ms=$($r.nvrtc_ms) dataset_ms=$($r.dataset_ms) fingerprint=$($r.fp) match=$($r.match) card_alone=$cardAlone device=$($r.dev) tag=$($r.tag)"
|
||||
}
|
||||
foreach ($r in $rows) { "RESULT sample $($r.ts.ToString('HH:mm:ss')) $($r.w) $($r.util) $($r.sm) $($r.mem) $($r.temp) $($r.used)" }
|
||||
$status = if ($results.Count -eq $order.Count) { 'done' } elseif ($results.Count -gt 0) { 'partial' } else { 'failed' }
|
||||
"RESULT end $(Stamp)"
|
||||
Summary $status @{ packs_run = $results.Count; packs_wanted = $order.Count; g1_match = $g1Match; card_alone = $cardAlone; smi_index = $idx; device = $gpuName }
|
||||
if ($status -eq 'failed') { exit 1 }
|
||||
exit 0
|
||||
108
tools/ca3-pc1-amd/pc1-amd-derive.ps1
Normal file
108
tools/ca3-pc1-amd/pc1-amd-derive.ps1
Normal file
|
|
@ -0,0 +1,108 @@
|
|||
# Counter ASIC 3.0, PC 1 AMD job 3 (6 October 2026): item 2's per-day item-derivation packs (dr736-genesis and
|
||||
# dr736-devnet-epoch0, docs/plans/counter-asic-3-derivation.md) on PC 1's RX 9070 XT (machine ae432dc7, gfx1201)
|
||||
# through the kit's igneum-worker-opencl.exe, with the class v3 control mx8-devnet-epoch0 in the same job. Published
|
||||
# as a plain signed `run` job (NOT --stop-miners) and run BESIDE THE MINERS: the rows this job is for are build and
|
||||
# compile times and bit-exactness, which a shared card does not change, plus a 60 s rate whose ratio to the control
|
||||
# in the same job stands on a shared card (absolutes do not; the process list says what the card carried). Nothing
|
||||
# is switched; the script never quits, pauses, resumes or updates the installed app and never writes settings.json.
|
||||
# Per pack, two passes: pass 1 is the cold compile (the first clBuildProgram of that kernel text on this PC today;
|
||||
# AMD's driver caches compiled kernels on disk, so pass 2 is the cached figure and is printed beside it), the 1 GiB
|
||||
# daily build (the worker's `cache .. dataset .. ms` line), the self-test (cache FNV, dataset head, 64 samples, 96
|
||||
# vector lanes) and the 2^24 fingerprint at base nonce 0 against the Mac's (dr736-genesis 50e3eaa779da4f1e,
|
||||
# dr736-devnet-epoch0 9553f6d5c667205a, mx8-devnet-epoch0 90f794dd556f7a3b) with 60 timed dispatches for the rate;
|
||||
# pass 2 takes 3 dispatches. The compile time is the worker's own `build <ms> clBuildProgram` line (host.c of branch
|
||||
# ca3-pc1-amd), the OpenCL equivalent of the CUDA worker's NVRTC line (+1.1 s per dr736 pack on the 5090).
|
||||
# Lines: `RESULT DERIVE pack=<name> build_ms=<x> build_ms_cached=<x2> dataset_ms=<y> fingerprint=<hex> match=<yes/no>
|
||||
# mhs=<z> ratio_to_mx8=<r> ...`; a `SUMMARY {json}` line at the end. Read back with `node tools/jobs.mjs <job id>`.
|
||||
$ErrorActionPreference = 'Continue'
|
||||
$jobName = 'pc1-amd-derive'
|
||||
$kitId = 'fetch-ca3-pc1-amd-20261006'
|
||||
$started = Get-Date
|
||||
function Stamp { (Get-Date).ToUniversalTime().ToString('yyyy-MM-ddTHH:mm:ssZ') }
|
||||
function Summary([string] $status, [hashtable] $extra) {
|
||||
$o = [ordered]@{ job = $jobName; status = $status; duration_s = [int]((Get-Date) - $started).TotalSeconds; finished_at = (Stamp) }
|
||||
foreach ($k in $extra.Keys) { $o[$k] = $extra[$k] }
|
||||
'SUMMARY ' + ($o | ConvertTo-Json -Compress -Depth 4)
|
||||
}
|
||||
$expected = @{ 'mx8-devnet-epoch0' = '90f794dd556f7a3b'; 'dr736-genesis' = '50e3eaa779da4f1e'; 'dr736-devnet-epoch0' = '9553f6d5c667205a' }
|
||||
$order = @('mx8-devnet-epoch0', 'dr736-genesis', 'dr736-devnet-epoch0')
|
||||
"RESULT start $(Stamp) job=$jobName machine=$env:COMPUTERNAME app_version=$env:IGNEUM_APP_VERSION"
|
||||
$jobs = Split-Path $env:IGNEUM_JOB_DIR
|
||||
$kit = Join-Path $jobs $kitId
|
||||
if (-not (Test-Path $kit)) { "RESULT error kit missing at $kit (the fetch job $kitId runs first; republish it after any app update)"; Summary 'failed' @{ error = 'kit missing' }; exit 2 }
|
||||
$exe = Join-Path $kit 'bin\igneum-worker-opencl.exe'
|
||||
$packs = Join-Path $kit 'packs'
|
||||
if (-not (Test-Path $exe)) { "RESULT error worker missing at $exe"; Summary 'failed' @{ error = 'worker missing' }; exit 2 }
|
||||
if (-not (Test-Path $packs)) { "RESULT error packs missing at $packs"; Summary 'failed' @{ error = 'packs missing' }; exit 2 }
|
||||
"RESULT worker kit $exe sha256 $((Get-FileHash -Algorithm SHA256 $exe).Hash.ToLower()) bytes $((Get-Item $exe).Length)"
|
||||
foreach ($pk in $order) {
|
||||
$d = Join-Path $packs $pk
|
||||
if (-not (Test-Path (Join-Path $d 'kernel_bound.cl'))) { "RESULT error pack $pk missing at $d"; Summary 'failed' @{ error = "pack $pk missing" }; exit 2 }
|
||||
"RESULT pack $pk kernel_bound.cl sha256 $((Get-FileHash -Algorithm SHA256 (Join-Path $d 'kernel_bound.cl')).Hash.ToLower()) program.json sha256 $((Get-FileHash -Algorithm SHA256 (Join-Path $d 'program.json')).Hash.ToLower())"
|
||||
}
|
||||
|
||||
# the OpenCL device index of the 9070 XT (the installed worker's list when present: the app's own indices)
|
||||
$inst = @("$env:LOCALAPPDATA\Programs\Igneum Miner", "$env:ProgramFiles\Igneum Miner") | Where-Object { Test-Path (Join-Path $_ 'igneum-app.exe') } | Select-Object -First 1
|
||||
$listExe = $exe
|
||||
if ($inst -and (Test-Path (Join-Path $inst 'igneum-worker-opencl.exe'))) { $listExe = Join-Path $inst 'igneum-worker-opencl.exe' }
|
||||
$list = @(& $listExe --list 2>&1 | ForEach-Object { "$_" })
|
||||
$list | ForEach-Object { "RESULT list $_" }
|
||||
$dev = $null
|
||||
foreach ($l in $list) { if ($l -match '^\s*\[(\d+)\].*gfx1201' -and $l -notmatch 'dup') { $dev = [int]$Matches[1]; break } }
|
||||
if ($null -eq $dev) { "RESULT error no gfx1201 device in --list (the eGPU is off the bus: every row OWED)"; Summary 'failed' @{ error = 'no gfx1201' }; exit 2 }
|
||||
"RESULT device $dev gfx1201 (list from $listExe)"
|
||||
|
||||
function Workers { @(Get-CimInstance Win32_Process -Filter "Name = 'igneum-worker-opencl.exe' OR Name = 'igneum-worker-cuda.exe'" -ErrorAction SilentlyContinue | ForEach-Object { "$($_.Name):$($_.ProcessId):[$($_.CommandLine -replace '\s+', ' ')]" }) }
|
||||
$w = @(Workers)
|
||||
$loaded = ($w | Where-Object { $_ -match "igneum-worker-opencl.*--device\s+$dev(\s|$)" }).Count -gt 0
|
||||
"RESULT workers_before $(Stamp) $($w -join ' ')"
|
||||
"RESULT context card_state=$(if ($loaded) { 'LOADED (the card mines beside this job: build, compile and bit-exact rows stand, the rate is a ratio to the control only)' } else { 'quiet (no opencl worker with --device ' + $dev + ')' })"
|
||||
|
||||
function RunPack([string] $pk, [int] $batches, [string] $tag) {
|
||||
$d = Join-Path $packs $pk
|
||||
$t0 = Get-Date
|
||||
"RESULT bench $tag start $(Stamp) batches=$batches"
|
||||
$lines = @(& $exe --bench-pack --pack $d --batches $batches --batch-log2 24 --device $dev 2>&1 | ForEach-Object { "$_" })
|
||||
$rc = $LASTEXITCODE
|
||||
$lines | ForEach-Object { "RESULT bench $tag $_" }
|
||||
"RESULT bench $tag end $(Stamp) exit=$rc wall_s=$([int]((Get-Date) - $t0).TotalSeconds)"
|
||||
$o = [ordered]@{ pack = $pk; rc = $rc; build_ms = ''; cache_ms = ''; dataset_ms = ''; check_ms = ''; all_ms = ''; check = ''; fp = ''; mhs = ''; error = '' }
|
||||
foreach ($l in $lines) {
|
||||
if ($l -match '^build ([\d.]+) ms clBuildProgram') { $o.build_ms = $Matches[1] }
|
||||
if ($l -match '^pack .*: cache (\d+) dataset (\d+) hot \d+ check (\d+) ms \((\d+) ms in all\)') { $o.cache_ms = $Matches[1]; $o.dataset_ms = $Matches[2]; $o.check_ms = $Matches[3]; $o.all_ms = $Matches[4] }
|
||||
if ($l -match '^RESULT pack=') { if ($l -match 'fingerprint=([0-9a-f]{16})') { $o.fp = $Matches[1] }; if ($l -match 'mhs=([\d.]+)') { $o.mhs = $Matches[1] }; if ($l -match 'check=(\S+)') { $o.check = $Matches[1] } }
|
||||
}
|
||||
if (-not $o.fp -or $rc -ne 0) { $e = ($lines | Where-Object { $_ -match 'FAIL|error|MISMATCH' } | Select-Object -First 1); if (-not $e) { $e = "exit $rc, no RESULT line" }; $o.error = ($e -replace '\s+', ' ') }
|
||||
return $o
|
||||
}
|
||||
|
||||
$rows = @{}
|
||||
$ok = 0
|
||||
foreach ($pk in $order) {
|
||||
$p1 = RunPack $pk 60 "$pk#1"
|
||||
Start-Sleep -Seconds 2
|
||||
$p2 = RunPack $pk 3 "$pk#2"
|
||||
$rows[$pk] = @{ p1 = $p1; p2 = $p2 }
|
||||
if ($p1.error) { "RESULT DERIVE pack=$pk error=$($p1.error)" } else { $ok++ }
|
||||
Start-Sleep -Seconds 2
|
||||
}
|
||||
# the rows: build (cold and cached), dataset, fingerprint, rate and its ratio to the control measured in the same job
|
||||
$ctrl = $rows['mx8-devnet-epoch0'].p1
|
||||
foreach ($pk in $order) {
|
||||
$p1 = $rows[$pk].p1; $p2 = $rows[$pk].p2
|
||||
if ($p1.error) { continue }
|
||||
$want = $expected[$pk]
|
||||
$match = if ($p1.fp -eq $want) { 'yes' } else { 'no' }
|
||||
$match2 = if ($p2.fp -eq $want) { 'yes' } else { 'no' }
|
||||
$ratio = ''
|
||||
if ($ctrl -and -not $ctrl.error -and $ctrl.mhs -and $p1.mhs -and [double] $ctrl.mhs -gt 0) { $ratio = [math]::Round([double] $p1.mhs / [double] $ctrl.mhs, 3) }
|
||||
$buildDelta = ''
|
||||
if ($ctrl -and -not $ctrl.error -and $ctrl.build_ms -and $p1.build_ms) { $buildDelta = [math]::Round([double] $p1.build_ms - [double] $ctrl.build_ms, 1) }
|
||||
"RESULT DERIVE pack=$pk build_ms=$($p1.build_ms) build_ms_cached=$($p2.build_ms) build_ms_over_mx8=$buildDelta dataset_ms=$($p1.dataset_ms) dataset_ms_pass2=$($p2.dataset_ms) cache_ms=$($p1.cache_ms) check_ms=$($p1.check_ms) fingerprint=$($p1.fp) match=$match fingerprint_pass2=$($p2.fp) match_pass2=$match2 check=$($p1.check) mhs=$($p1.mhs) ratio_to_mx8=$ratio card_state=$(if ($loaded) { 'loaded' } else { 'quiet' }) card=gfx1201 harness=opencl-kit expected=$want"
|
||||
}
|
||||
"RESULT workers_after $(Stamp) $((Workers) -join ' ')"
|
||||
"RESULT end $(Stamp)"
|
||||
$status = if ($ok -eq $order.Count) { 'done' } elseif ($ok -gt 0) { 'partial' } else { 'failed' }
|
||||
Summary $status @{ packs_ok = $ok; packs_wanted = $order.Count; device = $dev; card_state = $(if ($loaded) { 'loaded' } else { 'quiet' }) }
|
||||
if ($status -eq 'failed') { exit 1 }
|
||||
exit 0
|
||||
79
tools/ca3-pc1-amd/pc1-amd-family.ps1
Normal file
79
tools/ca3-pc1-amd/pc1-amd-family.ps1
Normal file
|
|
@ -0,0 +1,79 @@
|
|||
# Counter ASIC 3.0, PC 1 AMD job 2 (6 October 2026): item 6's step cost of every reserve candidate family of spec
|
||||
# 1.13.2 on PC 1's RX 9070 XT (machine ae432dc7, gfx1201), docs/plans/counter-asic-3-reserve.md, status file section
|
||||
# 3 "Item 6" (the OWED AMD column). Published as a plain signed `run` job (NOT --stop-miners) and run BESIDE THE MINERS:
|
||||
# a dependent-chain ratio (each family against the add-xor-rotate chain, both measured in the same minute on the
|
||||
# same card) survives a shared card, so no card is switched here; the process list says what the card carried and
|
||||
# every line says loaded or quiet. The probe is the kit's family-probe-cl.exe (proto-opencl/family-probe.c of branch
|
||||
# ca3-pc1-amd, cross-compiled with mingw, OpenCL.dll loaded at run time; the Mac ran the same source bit-exact on
|
||||
# Apple OpenCL on 6 October 2026, 15:2x UTC), run three times on every AMD OpenCL device the machine lists (the
|
||||
# gfx1201 on the current platform, its older-platform duplicate, the gfx1036 when present), each run best of 3 with
|
||||
# a fresh seed per repetition, device event time, bit-exact against the CPU reference on two whole 32-lane groups.
|
||||
# Every family is tried through every form AMD's OpenCL C offers (clang builtins ds_bpermute / ds_swizzle / sudot4 /
|
||||
# wmma, cl_khr_subgroup_shuffle, cl_intel_subgroups, cl_amd_media_ops2 amd_bfe / amd_perm, the plain C form, the
|
||||
# __local emulation); a form that does not compile prints build=failed with the first line of the build log and
|
||||
# the run goes on. The script never quits, pauses, resumes or updates the installed app and never writes settings.json.
|
||||
# Lines: the probe's own `RESULT FAMILY name=<f> ms=<best> gsteps=<x> ratio=<r> exact=<yes/no/unverified>
|
||||
# path=<native|sequence|emulated> variant=<v> ...` and `RESULT FAMILYBEST ...` (the row per family the status file
|
||||
# takes), each prefixed `RESULT run=<n> dev=<d>`; a `SUMMARY {json}` line at the end. Read back with `node tools/jobs.mjs <job id>`.
|
||||
$ErrorActionPreference = 'Continue'
|
||||
$jobName = 'pc1-amd-family'
|
||||
$kitId = 'fetch-ca3-pc1-amd-20261006'
|
||||
$started = Get-Date
|
||||
function Stamp { (Get-Date).ToUniversalTime().ToString('yyyy-MM-ddTHH:mm:ssZ') }
|
||||
function Summary([string] $status, [hashtable] $extra) {
|
||||
$o = [ordered]@{ job = $jobName; status = $status; duration_s = [int]((Get-Date) - $started).TotalSeconds; finished_at = (Stamp) }
|
||||
foreach ($k in $extra.Keys) { $o[$k] = $extra[$k] }
|
||||
'SUMMARY ' + ($o | ConvertTo-Json -Compress -Depth 4)
|
||||
}
|
||||
"RESULT start $(Stamp) job=$jobName machine=$env:COMPUTERNAME app_version=$env:IGNEUM_APP_VERSION"
|
||||
$jobs = Split-Path $env:IGNEUM_JOB_DIR
|
||||
$kit = Join-Path $jobs $kitId
|
||||
if (-not (Test-Path $kit)) { "RESULT error kit missing at $kit (the fetch job $kitId runs first; republish it after any app update)"; Summary 'failed' @{ error = 'kit missing' }; exit 2 }
|
||||
$probe = Join-Path $kit 'bin\family-probe-cl.exe'
|
||||
$src = Join-Path $kit 'src\family-probe.c'
|
||||
if (-not (Test-Path $probe)) { "RESULT error probe missing at $probe"; Summary 'failed' @{ error = 'probe missing' }; exit 2 }
|
||||
"RESULT probe $probe sha256 $((Get-FileHash -Algorithm SHA256 $probe).Hash.ToLower()) bytes $((Get-Item $probe).Length)"
|
||||
if (Test-Path $src) { "RESULT source family-probe.c sha256 $((Get-FileHash -Algorithm SHA256 $src).Hash.ToLower()) bytes $((Get-Item $src).Length)" }
|
||||
|
||||
# what the card carries during the run (read only; nothing is switched)
|
||||
function Workers { @(Get-CimInstance Win32_Process -Filter "Name = 'igneum-worker-opencl.exe' OR Name = 'igneum-worker-cuda.exe'" -ErrorAction SilentlyContinue | ForEach-Object { "$($_.Name):$($_.ProcessId):[$($_.CommandLine -replace '\s+', ' ')]" }) }
|
||||
$w = @(Workers)
|
||||
$amdLoaded = ($w | Where-Object { $_ -match 'igneum-worker-opencl' }).Count -gt 0
|
||||
"RESULT workers_before $(Stamp) $($w -join ' ')"
|
||||
"RESULT context card_state=$(if ($amdLoaded) { 'LOADED (an igneum-worker-opencl process mines on an AMD card: absolutes are shared-card figures, ratios stand)' } else { 'quiet (no igneum-worker-opencl process)' })"
|
||||
|
||||
# every AMD device the probe lists (both platform entries of the 9070 XT, the gfx1036 when present)
|
||||
$list = @(& $probe --list 2>&1 | ForEach-Object { "$_" })
|
||||
$list | ForEach-Object { "RESULT list $_" }
|
||||
$devs = @()
|
||||
foreach ($l in $list) { if ($l -match '^\[(\d+)\] (.*?) \| (.*?) \|' -and $l -match 'gfx1201|gfx1036|AMD|Radeon') { if ($l -notmatch 'NVIDIA') { $devs += [pscustomobject]@{ idx = [int]$Matches[1]; name = $Matches[2]; platform = $Matches[3] } } } }
|
||||
if ($devs.Count -eq 0) { "RESULT error no AMD OpenCL device in the probe's list (the eGPU is off the bus and no gfx1036: every row OWED)"; Summary 'failed' @{ error = 'no AMD device' } ; exit 2 }
|
||||
"RESULT devices $(($devs | ForEach-Object { "[$($_.idx)] $($_.name) on $($_.platform)" }) -join ' ; ')"
|
||||
# the gfx1201 on the current platform first (the row the status file takes), then the rest
|
||||
$ordered = @($devs | Where-Object { $_.name -match 'gfx1201' } | Sort-Object { $_.platform } -Descending) + @($devs | Where-Object { $_.name -notmatch 'gfx1201' })
|
||||
|
||||
$runs = 0; $exactRows = 0; $rowsTotal = 0; $built = 0; $failedBuilds = 0
|
||||
foreach ($d in $ordered) {
|
||||
for ($r = 1; $r -le 3; $r++) {
|
||||
"RESULT run=$r dev=$($d.idx) start $(Stamp) name=$($d.name) platform=$($d.platform) card_state=$(if ($amdLoaded) { 'loaded' } else { 'quiet' })"
|
||||
$t0 = Get-Date
|
||||
$lines = @(& $probe --device $d.idx --reps 3 2>&1 | ForEach-Object { "$_" })
|
||||
$rc = $LASTEXITCODE
|
||||
foreach ($l in $lines) {
|
||||
if ($l -match '^RESULT ') { "RESULT run=$r dev=$($d.idx) $($l.Substring(7))" } else { "RESULT run=$r dev=$($d.idx) text $l" }
|
||||
if ($l -match '^RESULT FAMILY name=\S+ ms=') { $rowsTotal++; if ($l -match ' exact=yes ') { $exactRows++ } }
|
||||
if ($l -match '^RESULT FAMILY .*build=failed') { $failedBuilds++ }
|
||||
if ($l -match '^RESULT FAMILYBEST name=\S+ ms=') { $built++ }
|
||||
}
|
||||
"RESULT run=$r dev=$($d.idx) end $(Stamp) exit=$rc wall_s=$([int]((Get-Date) - $t0).TotalSeconds)"
|
||||
if ($rc -ne 0) { "RESULT run=$r dev=$($d.idx) error=probe exit $rc" }
|
||||
$runs++
|
||||
Start-Sleep -Seconds 2
|
||||
}
|
||||
}
|
||||
"RESULT workers_after $(Stamp) $((Workers) -join ' ')"
|
||||
"RESULT end $(Stamp)"
|
||||
$status = if ($rowsTotal -gt 0) { 'done' } else { 'failed' }
|
||||
Summary $status @{ runs = $runs; devices = $ordered.Count; family_rows = $rowsTotal; exact_rows = $exactRows; failed_builds = $failedBuilds; familybest_rows = $built; card_state = $(if ($amdLoaded) { 'loaded' } else { 'quiet' }) }
|
||||
if ($status -eq 'failed') { exit 1 }
|
||||
exit 0
|
||||
270
tools/ca3-pc1-amd/pc1-amd-g1-shadow.ps1
Normal file
270
tools/ca3-pc1-amd/pc1-amd-g1-shadow.ps1
Normal file
|
|
@ -0,0 +1,270 @@
|
|||
# Counter ASIC 3.0, PC 1 AMD job 1 (6 October 2026): gate G1 for the class v4 candidate and item 8's AMD ladder on PC 1's
|
||||
# RX 9070 XT (machine ae432dc7, gfx1201 on the eGPU), docs/plans/counter-asic-3-status.md sections 3, 5 and 7.
|
||||
# Published as a plain signed `run` job (NOT --stop-miners): the RTX 5090 and the RTX 4070 keep mining; this script
|
||||
# switches ONLY the 9070 XT off in the installed app through POST <app.url>api/cards with the card key read from the
|
||||
# app's settings.json (never api/state) in EVERY form (the key as written, vendor:index, vendor:code; the 6 October
|
||||
# finding: the index-less form switched nothing on 5 October), confirms by the process list that the card's
|
||||
# igneum-worker-opencl process (the one with its --device index) is gone, runs the kit's igneum-worker-opencl.exe
|
||||
# (proto-opencl/host.c of branch ca3-pc1-amd: the installed worker's source plus one `build <ms> clBuildProgram` line,
|
||||
# cross-compiled with mingw as proto-cuda/nvrtc/build-windows.sh does) with --bench-pack on every pack of the kit while
|
||||
# the AMD helper igneum-gpu-telemetry.exe (ADLX board watts, the installed app's own helper, run as a second copy at
|
||||
# 1 Hz through a wrapper this script starts and ends by its own pid) samples the card, and restores the card in a
|
||||
# finally block whatever happens. It never quits, pauses, resumes or updates the installed app, never writes
|
||||
# settings.json and never touches the manifest. If --stop-miners IS used at publish time the card is already quiet
|
||||
# and the switch is a no-op: the script works either way and says which it saw.
|
||||
# Per pack: the self-test (cache, dataset, 64 samples, 96 vector lanes, from the worker's own line), the 2^24
|
||||
# fingerprint at base nonce 0 against the Mac's and the 5090's (G1), and a 75 to 120 s rate with the card alone for
|
||||
# MH/s; the mean board watts over the bench window give microjoules per hash. The control row doubles as item 1's AMD
|
||||
# per-joule row. Lines: `RESULT G1 pack=<name> fingerprint=<hex> match=<yes/no> ...`,
|
||||
# `RESULT LADDER pack=<name> ops=<N> mhs=<x> watts=<y|owed> uj=<z|owed> ...`, a `SUMMARY {json}` line at the end.
|
||||
# Read back with `node tools/jobs.mjs <job id>`. The kit: fetch job fetch-ca3-pc1-amd-20261006 (tools/ca3-pc1-amd/README.md).
|
||||
$ErrorActionPreference = 'Continue'
|
||||
$jobName = 'pc1-amd-g1-shadow'
|
||||
$kitId = 'fetch-ca3-pc1-amd-20261006'
|
||||
$started = Get-Date
|
||||
function Stamp { (Get-Date).ToUniversalTime().ToString('yyyy-MM-ddTHH:mm:ssZ') }
|
||||
function Summary([string] $status, [hashtable] $extra) {
|
||||
$o = [ordered]@{ job = $jobName; status = $status; duration_s = [int]((Get-Date) - $started).TotalSeconds; finished_at = (Stamp) }
|
||||
foreach ($k in $extra.Keys) { $o[$k] = $extra[$k] }
|
||||
'SUMMARY ' + ($o | ConvertTo-Json -Compress -Depth 4)
|
||||
}
|
||||
# the expected 2^24 fingerprints at base nonce 0 (the Mac on Metal and Apple OpenCL, the 5090 on CUDA: status file
|
||||
# section 3, docs/analysis/latency-shadow-2026-10-06.md sections 2 and 5, counter-asic-2-status.md 22:16 for the
|
||||
# control; re-checked on Apple OpenCL through this kit's host.c on 6 October 2026, 15:2x UTC)
|
||||
$expected = @{
|
||||
'mx8-devnet-epoch0' = '90f794dd556f7a3b'
|
||||
'sh256x13' = '59ac286fe2a5a9ef'
|
||||
'sh256x27' = '3d2e8245cc084d07'
|
||||
'sh256x53' = '4f824b15cf2b124a'
|
||||
'sh256x88' = '0572522e39a94d8a'
|
||||
'sh64x52' = '9dd010f79d8ca9f4'
|
||||
}
|
||||
# ops per hash as the status file counts them (x1.83 counted ops per shadow instruction + 930 for the base program);
|
||||
# the shadow instruction count is read from each pack's program.json at run time and printed beside it
|
||||
$opsPerHash = @{ 'mx8-devnet-epoch0' = 930; 'sh256x13' = 49700; 'sh256x27' = 102100; 'sh256x53' = 199600; 'sh256x88' = 330700; 'sh64x52' = 49700 }
|
||||
# timed dispatches of 2^24 nonces per pack: about 0.9 s each at 19 MH/s, longer where the shadow binds the card
|
||||
$batches = @{ 'mx8-devnet-epoch0' = 90; 'sh256x13' = 90; 'sh256x27' = 90; 'sh256x53' = 80; 'sh256x88' = 60; 'sh64x52' = 90 }
|
||||
$order = @('mx8-devnet-epoch0', 'sh256x13', 'sh256x27', 'sh256x53', 'sh256x88', 'sh64x52', 'mx8-devnet-epoch0')
|
||||
|
||||
"RESULT start $(Stamp) job=$jobName machine=$env:COMPUTERNAME app_version=$env:IGNEUM_APP_VERSION"
|
||||
# ---- the kit (the wiped-jobs-folder rule: tested before use; republish the fetch after any app update) ----
|
||||
$jobs = Split-Path $env:IGNEUM_JOB_DIR
|
||||
$kit = Join-Path $jobs $kitId
|
||||
if (-not (Test-Path $kit)) { "RESULT error kit missing at $kit (the fetch job $kitId runs first; republish it after any app update)"; Summary 'failed' @{ error = 'kit missing' }; exit 2 }
|
||||
$exe = Join-Path $kit 'bin\igneum-worker-opencl.exe'
|
||||
$packs = Join-Path $kit 'packs'
|
||||
if (-not (Test-Path $exe)) { "RESULT error worker missing at $exe"; Summary 'failed' @{ error = 'worker missing' }; exit 2 }
|
||||
if (-not (Test-Path $packs)) { "RESULT error packs missing at $packs"; Summary 'failed' @{ error = 'packs missing' }; exit 2 }
|
||||
"RESULT worker kit $exe sha256 $((Get-FileHash -Algorithm SHA256 $exe).Hash.ToLower()) bytes $((Get-Item $exe).Length)"
|
||||
foreach ($f in (Get-ChildItem $kit -Recurse -File | Where-Object { $_.Name -match '\.(exe|c|json)$|SHA256SUMS' } | Sort-Object FullName)) { "RESULT kitfile $($f.FullName.Substring($kit.Length + 1).Replace('\', '/')) sha256 $((Get-FileHash -Algorithm SHA256 $f.FullName).Hash.ToLower()) bytes $($f.Length)" }
|
||||
$job = $env:IGNEUM_JOB_DIR; if (-not $job) { $job = Join-Path $env:TEMP 'igneum-ca3-pc1-amd' }; New-Item -ItemType Directory -Force -Path $job | Out-Null
|
||||
|
||||
# ---- the installed app: install folder (the installed worker for the cross-check and the AMD helper), app dir, URL ----
|
||||
$inst = @("$env:LOCALAPPDATA\Programs\Igneum Miner", "$env:ProgramFiles\Igneum Miner") | Where-Object { Test-Path (Join-Path $_ 'igneum-app.exe') } | Select-Object -First 1
|
||||
$appDir = if ($env:IGNEUM_APP_DIR) { $env:IGNEUM_APP_DIR } else { Join-Path $env:LOCALAPPDATA 'igneum\app' }
|
||||
$urlFile = Join-Path $appDir 'app.url'
|
||||
$base = $null
|
||||
if (Test-Path $urlFile) { $base = (Get-Content -LiteralPath $urlFile -Raw).Trim().TrimEnd('/') }
|
||||
"RESULT app install=$inst app_dir=$appDir url_present=$([bool]$base)"
|
||||
|
||||
# ---- the OpenCL device index of the 9070 XT, from the worker's own list (the current platform; the older platform's duplicate is marked dup) ----
|
||||
$listExe = $exe
|
||||
if ($inst -and (Test-Path (Join-Path $inst 'igneum-worker-opencl.exe'))) { $listExe = Join-Path $inst 'igneum-worker-opencl.exe' }
|
||||
$list = @(& $listExe --list 2>&1 | ForEach-Object { "$_" })
|
||||
$list | ForEach-Object { "RESULT list $_" }
|
||||
$dev = $null
|
||||
foreach ($l in $list) { if ($l -match '^\s*\[(\d+)\].*gfx1201' -and $l -notmatch 'dup') { $dev = [int]$Matches[1]; break } }
|
||||
if ($null -eq $dev) { "RESULT error no gfx1201 device in --list (the eGPU is off the bus: every row OWED)"; Summary 'failed' @{ error = 'no gfx1201' }; exit 2 }
|
||||
"RESULT device $dev gfx1201 (list from $listExe)"
|
||||
|
||||
# ---- the card's keys from settings.json (READ ONLY), every form ----
|
||||
$cardKeys = @(); $cardEnabled = $false; $cardIdent = 1
|
||||
try {
|
||||
$sj = Get-Content -LiteralPath (Join-Path $appDir 'settings.json') -Raw | ConvertFrom-Json
|
||||
foreach ($prop in $sj.cards.PSObject.Properties) {
|
||||
if ($prop.Name -like 'amd:*' -and $prop.Name -match 'gfx1201') {
|
||||
$parts = $prop.Name -split ':'
|
||||
$cardKeys += $prop.Name
|
||||
if ($parts.Count -ge 3) { $cardKeys += ($parts[0] + ':' + $parts[1]); $cardKeys += ($parts[0] + ':' + $parts[2]) }
|
||||
$cardEnabled = [bool] $prop.Value.enabled
|
||||
if ($prop.Value.identities) { $cardIdent = [int] $prop.Value.identities }
|
||||
"RESULT card settings_key=$($prop.Name) enabled=$cardEnabled identities=$cardIdent forms=[$($cardKeys -join ' | ')]"
|
||||
}
|
||||
}
|
||||
} catch { "RESULT settings_read_failed $($_.Exception.Message)" }
|
||||
if ($cardKeys.Count -eq 0) { "RESULT card none in settings.json matching amd:*gfx1201*: no switch; the rows below carry whatever load the card has" }
|
||||
function PostCards([bool] $enabled) {
|
||||
if (-not $base) { "RESULT cards no app.url: nothing posted"; return }
|
||||
foreach ($k in $cardKeys) {
|
||||
$body = @{ cards = @(@{ key = $k; enabled = $enabled; identities = $cardIdent }) } | ConvertTo-Json -Depth 5
|
||||
try { Invoke-RestMethod -Uri "$base/api/cards" -Method POST -Body $body -ContentType 'application/json' -TimeoutSec 10 | Out-Null; "RESULT cards key=$k enabled=$enabled posted" } catch { "RESULT cards key=$k enabled=$enabled error=$($_.Exception.Message)" }
|
||||
}
|
||||
}
|
||||
function Workers {
|
||||
@(Get-CimInstance Win32_Process -Filter "Name = 'igneum-worker-opencl.exe' OR Name = 'igneum-worker-cuda.exe' OR Name = 'igneum-worker-metal.exe'" -ErrorAction SilentlyContinue | ForEach-Object { "$($_.Name):$($_.ProcessId):[$($_.CommandLine -replace '\s+', ' ')]" })
|
||||
}
|
||||
function CardWorkers { @(Get-CimInstance Win32_Process -Filter "Name = 'igneum-worker-opencl.exe'" -ErrorAction SilentlyContinue | Where-Object { $_.CommandLine -match "--device\s+$dev(\s|$)" } | ForEach-Object { $_.ProcessId }) }
|
||||
"RESULT workers_before $(Stamp) $((Workers) -join ' ')"
|
||||
|
||||
# ---- the AMD helper: the installed app's igneum-gpu-telemetry.exe (0.3.10+), else the newest one under an amd-kit job folder ----
|
||||
$tele = $null
|
||||
if ($inst -and (Test-Path (Join-Path $inst 'igneum-gpu-telemetry.exe'))) { $tele = Join-Path $inst 'igneum-gpu-telemetry.exe' }
|
||||
if (-not $tele) { $found = Get-ChildItem -Path (Join-Path $appDir 'jobs') -Recurse -Filter 'igneum-gpu-telemetry.exe' -ErrorAction SilentlyContinue | Where-Object { $_.FullName -match 'amd-kit' } | Sort-Object LastWriteTime -Descending | Select-Object -First 1; if ($found) { $tele = $found.FullName } }
|
||||
$wattsSource = 'owed'
|
||||
if ($tele) {
|
||||
$one = @(& $tele 2>&1 | ForEach-Object { "$_" })
|
||||
$one | ForEach-Object { "RESULT tele_once $_" }
|
||||
$card9070 = $one | Where-Object { $_ -match '^amd \d+ .*name "[^"]*9070[^"]*" watts (\S+) .* source (\S+)' } | Select-Object -First 1
|
||||
if ($card9070 -and $card9070 -match ' watts (\S+) .* source (\S+)$') { if ($Matches[1] -ne '-') { $wattsSource = $Matches[2] } }
|
||||
"RESULT tele helper=$tele sha256=$((Get-FileHash -Algorithm SHA256 $tele).Hash.ToLower()) watts_source=$wattsSource (board watts through ADLX GPUPower, GPUTotalBoardPower as the fallback; 'owed' = the helper gives no watts for the 9070 XT)"
|
||||
} else { "RESULT tele none (no igneum-gpu-telemetry.exe in the install folder or under an amd-kit job folder): watts OWED" }
|
||||
|
||||
$samples = Join-Path $job 'amd-samples.log'
|
||||
$sampler = $null
|
||||
$windows = @()
|
||||
$results = @()
|
||||
$cardAlone = $false
|
||||
$switched = $false
|
||||
try {
|
||||
# ---- the card off, confirmed by the process list (never api/state) ----
|
||||
if ($cardKeys.Count -gt 0 -and $base) {
|
||||
$before = @(CardWorkers)
|
||||
"RESULT card_workers_before pids=[$($before -join ' ')]"
|
||||
if ($before.Count -eq 0) { "RESULT card already quiet before the switch (no opencl worker with --device ${dev}: --stop-miners or the card is off in the app)"; $cardAlone = $true }
|
||||
else {
|
||||
PostCards $false; $switched = $true
|
||||
$t = 0
|
||||
while ($t -lt 150) { Start-Sleep -Seconds 5; $t += 5; if ((@(CardWorkers)).Count -eq 0) { break } }
|
||||
$left = @(CardWorkers)
|
||||
if ($left.Count -eq 0) { "RESULT card $(Stamp) quiet after $t s: no igneum-worker-opencl process with --device $dev (the card to itself)"; $cardAlone = $true }
|
||||
else { "RESULT card $(Stamp) UNCONFIRMED after $t s: opencl worker pids [$($left -join ' ')] still carry --device $dev; the rows below are LOADED-card figures" }
|
||||
Start-Sleep -Seconds 5
|
||||
}
|
||||
} else { "RESULT card not switched (no key or no app.url): the rows below carry whatever load the card has" }
|
||||
"RESULT workers_during $(Stamp) $((Workers) -join ' ')"
|
||||
|
||||
# ---- the sampler: the helper at 1 Hz through a wrapper (timestamps added per line), ended by its own pid in finally ----
|
||||
if ($tele -and $wattsSource -ne 'owed') {
|
||||
$wrapper = Join-Path $job 'sampler.ps1'
|
||||
$wrapperText = @'
|
||||
param([string] $Exe, [string] $Out)
|
||||
& $Exe -l 1 2>&1 | ForEach-Object { Add-Content -LiteralPath $Out -Value ((Get-Date).ToUniversalTime().ToString('yyyy-MM-ddTHH:mm:ss.fff') + ' ' + $_) }
|
||||
'@
|
||||
[IO.File]::WriteAllText($wrapper, $wrapperText, (New-Object System.Text.UTF8Encoding $false))
|
||||
Remove-Item -LiteralPath $samples -Force -ErrorAction SilentlyContinue
|
||||
$sampler = Start-Process -FilePath 'powershell.exe' -ArgumentList @('-NoProfile', '-ExecutionPolicy', 'Bypass', '-File', $wrapper, $tele, $samples) -WindowStyle Hidden -PassThru
|
||||
"RESULT sampler pid=$($sampler.Id) file=$samples"
|
||||
Start-Sleep -Seconds 12
|
||||
"RESULT idle $(Stamp) 12 s of idle samples taken before the first bench"
|
||||
}
|
||||
|
||||
# ---- the ladder ----
|
||||
$i = 0
|
||||
foreach ($pk in $order) {
|
||||
$i++
|
||||
$d = Join-Path $packs $pk
|
||||
if (-not (Test-Path $d)) { "RESULT G1 pack=$pk error=pack missing at $d"; "RESULT LADDER pack=$pk error=pack missing"; continue }
|
||||
$instrs = 0; $cls = ''
|
||||
try { $pj = Get-Content -LiteralPath (Join-Path $d 'program.json') -Raw | ConvertFrom-Json; $cls = [string] $pj.load_class; if ($pj.shadow) { $instrs = [int] $pj.shadow.instrs_per_hash } } catch { "RESULT note program.json of $pk unreadable: $($_.Exception.Message)" }
|
||||
$n = $batches[$pk]
|
||||
$tag = "$pk#$i"
|
||||
$t0 = (Get-Date).ToUniversalTime()
|
||||
"RESULT bench $tag start $(Stamp) batches=$n class=$cls shadow_instrs_per_hash=$instrs"
|
||||
$lines = @(& $exe --bench-pack --pack $d --batches $n --batch-log2 24 --device $dev 2>&1 | ForEach-Object { "$_" })
|
||||
$rc = $LASTEXITCODE
|
||||
$t1 = (Get-Date).ToUniversalTime()
|
||||
$lines | ForEach-Object { "RESULT bench $tag $_" }
|
||||
"RESULT bench $tag end $(Stamp) exit=$rc wall_s=$([int]($t1 - $t0).TotalSeconds)"
|
||||
$res = $lines | Where-Object { $_ -match '^RESULT pack=' } | Select-Object -First 1
|
||||
$fp = ''; $mhs = ''; $check = ''
|
||||
if ($res -and $res -match 'fingerprint=([0-9a-f]{16})') { $fp = $Matches[1] }
|
||||
if ($res -and $res -match 'mhs=([\d.]+)') { $mhs = $Matches[1] }
|
||||
if ($res -and $res -match 'check=(\S+)') { $check = $Matches[1] }
|
||||
$buildMs = ''; if (($lines | Where-Object { $_ -match '^build ([\d.]+) ms clBuildProgram' } | Select-Object -First 1) -match '^build ([\d.]+) ms') { $buildMs = $Matches[1] }
|
||||
$dsMs = ''; if (($lines | Where-Object { $_ -match '^pack .*: cache \d+ dataset (\d+) ' } | Select-Object -First 1) -match ' dataset (\d+) ') { $dsMs = $Matches[1] }
|
||||
if (-not $res -or $rc -ne 0) {
|
||||
$err = ($lines | Where-Object { $_ -match 'FAIL|error|MISMATCH' } | Select-Object -First 1); if (-not $err) { $err = "exit $rc, no RESULT line" }
|
||||
"RESULT G1 pack=$pk error=$($err -replace '\s+', ' ')"
|
||||
"RESULT LADDER pack=$pk error=$($err -replace '\s+', ' ')"
|
||||
continue
|
||||
}
|
||||
$want = $expected[$pk]
|
||||
$match = if ($want -and $fp -eq $want) { 'yes' } elseif ($want) { 'no' } else { 'unknown' }
|
||||
"RESULT G1 pack=$pk fingerprint=$fp match=$match expected=$want check=$check card=gfx1201 harness=opencl-kit"
|
||||
$results += [pscustomobject]@{ tag = $tag; pack = $pk; instrs = $instrs; ops = $opsPerHash[$pk]; mhs = $mhs; fp = $fp; match = $match; check = $check; build_ms = $buildMs; dataset_ms = $dsMs; t0 = $t0; t1 = $t1 }
|
||||
Start-Sleep -Seconds 3
|
||||
}
|
||||
|
||||
# ---- the cross-check: the INSTALLED worker on the control pack, if it carries --bench-pack ----
|
||||
if ($inst -and (Test-Path (Join-Path $inst 'igneum-worker-opencl.exe'))) {
|
||||
$iexe = Join-Path $inst 'igneum-worker-opencl.exe'
|
||||
$help = (& $iexe --help 2>&1 | Out-String)
|
||||
if ($help -match '--bench-pack') {
|
||||
$d = Join-Path $packs 'mx8-devnet-epoch0'
|
||||
$t0 = (Get-Date).ToUniversalTime()
|
||||
"RESULT xcheck start $(Stamp) worker=installed sha256=$((Get-FileHash -Algorithm SHA256 $iexe).Hash.ToLower())"
|
||||
$lines = @(& $iexe --bench-pack --pack $d --batches 40 --batch-log2 24 --device $dev 2>&1 | ForEach-Object { "$_" })
|
||||
$rc = $LASTEXITCODE
|
||||
$t1 = (Get-Date).ToUniversalTime()
|
||||
$lines | Where-Object { $_ -match '^RESULT|^pack |^warm-up|FAIL|error' } | ForEach-Object { "RESULT xcheck $_" }
|
||||
$res = $lines | Where-Object { $_ -match '^RESULT pack=' } | Select-Object -First 1
|
||||
$fp = ''; $mhs = ''
|
||||
if ($res -and $res -match 'fingerprint=([0-9a-f]{16})') { $fp = $Matches[1] }
|
||||
if ($res -and $res -match 'mhs=([\d.]+)') { $mhs = $Matches[1] }
|
||||
if ($res) { "RESULT XCHECK worker=installed pack=mx8-devnet-epoch0 fingerprint=$fp match=$(if ($fp -eq $expected['mx8-devnet-epoch0']) { 'yes' } else { 'no' }) mhs=$mhs exit=$rc (the kit worker's control rows are the LADDER lines)"; $windows += [pscustomobject]@{ tag = 'xcheck-installed'; t0 = $t0; t1 = $t1 } }
|
||||
else { "RESULT XCHECK worker=installed error=exit $rc, no RESULT line" }
|
||||
} else { "RESULT XCHECK worker=installed skipped: the installed igneum-worker-opencl.exe has no --bench-pack" }
|
||||
}
|
||||
} finally {
|
||||
# ---- whatever happened: the sampler ended by its own pid (never by name: the app runs its own copy of the helper), the card back ----
|
||||
if ($sampler) { try { & taskkill /T /F /PID $sampler.Id 2>&1 | Out-Null; "RESULT sampler ended pid=$($sampler.Id)" } catch { "RESULT sampler end error=$($_.Exception.Message)" } }
|
||||
if ($switched) {
|
||||
if ($cardEnabled) { PostCards $true; "RESULT card restore posted enabled=true (settings.json had it enabled)" } else { "RESULT card restore skipped: settings.json had the card disabled before this job" }
|
||||
$t = 0; $back = @()
|
||||
while ($t -lt 90 -and $cardEnabled) { Start-Sleep -Seconds 5; $t += 5; $back = @(CardWorkers); if ($back.Count -gt 0) { break } }
|
||||
"RESULT card_workers_after $(Stamp) pids=[$($back -join ' ')] after $t s $(if ($back.Count -gt 0) { '(the card mines again)' } elseif ($cardEnabled) { '(NOT back yet: check the app)' } else { '' })"
|
||||
}
|
||||
"RESULT workers_after $(Stamp) $((Workers) -join ' ')"
|
||||
}
|
||||
|
||||
# ---- the samples: board watts and clocks of the 9070 XT inside each bench window (after its first 12 s, before its last 2 s) ----
|
||||
$rows = @()
|
||||
if (Test-Path $samples) {
|
||||
foreach ($l in (Get-Content -LiteralPath $samples)) {
|
||||
if ($l -match '^(\S+) amd (\d+) bus (\S+) kind (\S+) name "([^"]*)" watts (\S+) temp_c (\S+) fan_rpm (\S+) fan_pct (\S+) mclk_mhz (\S+) gclk_mhz (\S+) util_pct (\S+) source (\S+)') {
|
||||
if ($Matches[5] -notmatch '9070') { continue }
|
||||
try { $ts = [datetime]::ParseExact($Matches[1], 'yyyy-MM-ddTHH:mm:ss.fff', [Globalization.CultureInfo]::InvariantCulture) } catch { continue }
|
||||
$w = if ($Matches[6] -eq '-') { $null } else { [double] $Matches[6] }
|
||||
$rows += [pscustomobject]@{ ts = $ts; w = $w; temp = $Matches[7]; fan = $Matches[9]; mclk = $Matches[10]; gclk = $Matches[11]; util = $Matches[12]; src = $Matches[13] }
|
||||
}
|
||||
}
|
||||
}
|
||||
"RESULT samples total $($rows.Count) file $samples (9070 XT lines)"
|
||||
function Window([datetime] $t0, [datetime] $t1) { @($rows | Where-Object { $_.ts -gt $t0.AddSeconds(12) -and $_.ts -lt $t1.AddSeconds(-2) -and $null -ne $_.w }) }
|
||||
$g1Match = 0; $ladderRows = 0
|
||||
foreach ($r in $results) {
|
||||
$in = @(Window $r.t0 $r.t1)
|
||||
$watts = 'owed'; $uj = 'owed'; $gclk = 'owed'; $mclk = 'owed'; $tmax = 'owed'; $wmin = ''; $wmax = ''
|
||||
if ($in.Count -gt 0) {
|
||||
$m = $in | Measure-Object -Property w -Average -Minimum -Maximum
|
||||
$watts = [math]::Round($m.Average, 1); $wmin = $m.Minimum; $wmax = $m.Maximum
|
||||
$gclk = [math]::Round((($in | ForEach-Object { [double] $_.gclk }) | Measure-Object -Average).Average)
|
||||
$mclk = [math]::Round((($in | ForEach-Object { [double] $_.mclk }) | Measure-Object -Average).Average)
|
||||
$tmax = (($in | ForEach-Object { [double] $_.temp }) | Measure-Object -Maximum).Maximum
|
||||
if ($r.mhs -and [double] $r.mhs -gt 0) { $uj = [math]::Round($m.Average / [double] $r.mhs, 3) }
|
||||
}
|
||||
if ($r.match -eq 'yes') { $g1Match++ }
|
||||
$ladderRows++
|
||||
"RESULT LADDER pack=$($r.pack) ops=$($r.ops) shadow_instrs=$($r.instrs) mhs=$($r.mhs) watts=$watts uj=$uj watts_min=$wmin watts_max=$wmax gclk_mhz=$gclk mclk_mhz=$mclk temp_max=$tmax samples=$($in.Count) window_s=$([int]($r.t1 - $r.t0).TotalSeconds) build_ms=$($r.build_ms) dataset_ms=$($r.dataset_ms) fingerprint=$($r.fp) match=$($r.match) card_alone=$cardAlone tag=$($r.tag)"
|
||||
}
|
||||
foreach ($win in $windows) { $in = @(Window $win.t0 $win.t1); if ($in.Count -gt 0) { $m = $in | Measure-Object -Property w -Average; "RESULT power $($win.tag) samples=$($in.Count) watts_mean=$([math]::Round($m.Average, 1))" } }
|
||||
if ($rows.Count -gt 0) {
|
||||
$idle = @($rows | Where-Object { $results.Count -gt 0 -and $_.ts -lt $results[0].t0 -and $null -ne $_.w })
|
||||
if ($idle.Count -gt 0) { "RESULT idle_power samples=$($idle.Count) watts_mean=$([math]::Round(($idle | Measure-Object -Property w -Average).Average, 1)) (the card switched off in the app, before the first bench)" }
|
||||
foreach ($r in $rows) { "RESULT sample $($r.ts.ToString('HH:mm:ss')) $($r.w) $($r.temp) $($r.fan) $($r.mclk) $($r.gclk) $($r.util) $($r.src)" }
|
||||
}
|
||||
$status = if ($results.Count -eq $order.Count) { 'done' } elseif ($results.Count -gt 0) { 'partial' } else { 'failed' }
|
||||
"RESULT end $(Stamp)"
|
||||
Summary $status @{ packs_run = $results.Count; packs_wanted = $order.Count; g1_match = $g1Match; ladder_rows = $ladderRows; watts = $wattsSource; card_alone = $cardAlone; device = $dev }
|
||||
if ($status -eq 'failed') { exit 1 }
|
||||
exit 0
|
||||
Loading…
Reference in a new issue