diff --git a/proto-opencl/host.c b/proto-opencl/host.c index 4abef5654..dab006f07 100644 --- a/proto-opencl/host.c +++ b/proto-opencl/host.c @@ -508,8 +508,17 @@ typedef struct { static const char* exchangeName(int m) { return m == 1 ? "sub_group_shuffle_xor (cl_khr_subgroup_shuffle)" : m == 2 ? "intel_sub_group_shuffle_xor (cl_intel_subgroups)" : "local-memory exchange with barrier"; } // Returns 0 on success, 1 on build failure (log printed). +#include "intel_rotr.h" /* the Intel rotate fold: rotr_var rewritten on an Intel platform (7 October 2026) */ +static char* intelRotrPatch(const DeviceInfo* di, const char* src, size_t* srcLen, int* patched) { + if (!di) { *patched = 0; return (char*)src; } + return igneum_intel_rotr_patch(di->vendor, di->platformName, src, srcLen, patched, 0); +} + static int buildProgram(Device* dv, const DeviceInfo* di, const char* src, size_t srcLen, int exchangeMode, int groupSize, const char* extra) { cl_int err = 0; + int patched = 0; + char* psrc = intelRotrPatch(di, src, &srcLen, &patched); + src = psrc; double tb = wallMs(); /* the pack's compile cost (Counter ASIC 3.0 item 2, 6 October 2026): printed as one line below */ const char* std; // The sub-group built-ins need OpenCL C 2.0 or 3.0. OpenCL 3.0 devices may report "OpenCL C 1.2" as the default @@ -522,6 +531,7 @@ static int buildProgram(Device* dv, const DeviceInfo* di, const char* src, size_ else std = "-cl-std=CL1.2"; snprintf(dv->buildOptions, sizeof(dv->buildOptions), "%s -D IGNEUM_GROUP=%d -D IGNEUM_EXCHANGE=%d %s", std, groupSize, exchangeMode, extra); dv->prog = clCreateProgramWithSource(dv->ctx, 1, &src, &srcLen, &err); + if (patched) free(psrc); CL_CHECK_ERR(err, "clCreateProgramWithSource"); err = clBuildProgram(dv->prog, 1, &di->device, dv->buildOptions, NULL, NULL); if (err != CL_SUCCESS) { @@ -1315,7 +1325,12 @@ static void prepareRun(PrepareTask* t) { src = readFile(path, &srcLen); if (!src) { snprintf(t->error, sizeof(t->error), "cannot read %s", path); releasePair(p); t->doneAt = wallMs(); t->done = 1; return; } tb = wallMs(); - p->prog = clCreateProgramWithSource(t->dv->ctx, 1, (const char**)&src, &srcLen, &err); + { + int patched = 0; + char* psrc = intelRotrPatch(t->di, src, &srcLen, &patched); + p->prog = clCreateProgramWithSource(t->dv->ctx, 1, (const char**)&psrc, &srcLen, &err); + if (patched) free(psrc); + } free(src); if (err != CL_SUCCESS) { prepareFail(t, "clCreateProgramWithSource", err); releasePair(p); t->doneAt = wallMs(); t->done = 1; return; } err = clBuildProgram(p->prog, 1, &t->di->device, t->dv->buildOptions, NULL, NULL); diff --git a/proto-opencl/intel_rotr.h b/proto-opencl/intel_rotr.h new file mode 100644 index 000000000..f960efc7d --- /dev/null +++ b/proto-opencl/intel_rotr.h @@ -0,0 +1,40 @@ +/* intel_rotr.h: the Intel rotate fold (7 October 2026, the Arc B580 register-trace bisect, docs/plans/intel-arc.md + * section 7). Intel's OpenCL compiler turns the pack's `rotr_var(x, n) = rotate(x, (0u - n) & 31u)` into a rotate LEFT + * by n (the negation dropped), so every variable right-rotate of every program came out wrong on the Arc while every + * other family and the dataset kernels were bit-exact (lane 0's trace diverged at instruction 6 of iteration 0, rotr, + * and nowhere before; job run-ia-arc-trace-20261007-b). On an Intel platform the worker rewrites that one helper line + * to the shift form the CPU interpreter and the family probe compute, before clCreateProgramWithSource; the text is + * otherwise untouched and no other vendor sees a change. Shared by host.c and test_intel_rotr.c (the gate's test). */ +#ifndef IGNEUM_INTEL_ROTR_H +#define IGNEUM_INTEL_ROTR_H +#include +#include +#include + +static const char* IGNEUM_ROTR_BUILTIN = "static inline uint rotr_var(uint x, uint n) { return rotate(x, (0u - n) & 31u); }"; +static const char* IGNEUM_ROTR_SHIFTS = "static inline uint rotr_var(uint x, uint n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }"; + +/* Returns the source to build: the original pointer when nothing applies, else a new buffer (the caller frees it when + * `*patched` is set) with the one line replaced and `*srcLen` updated. `vendor` and `platform` are the device's + * CL_DEVICE_VENDOR and the platform name; only a string holding "Intel" is rewritten. `quiet` suppresses the line. */ +static char* igneum_intel_rotr_patch(const char* vendor, const char* platform, const char* src, size_t* srcLen, int* patched, int quiet) { + const char* at; + char* out; + size_t a = strlen(IGNEUM_ROTR_BUILTIN), b = strlen(IGNEUM_ROTR_SHIFTS), pre; + *patched = 0; + if (!((vendor && strstr(vendor, "Intel")) || (platform && strstr(platform, "Intel")))) return (char*)src; + at = strstr(src, IGNEUM_ROTR_BUILTIN); + if (!at) return (char*)src; + pre = (size_t)(at - src); + out = (char*)malloc(*srcLen - a + b + 1); + if (!out) return (char*)src; + memcpy(out, src, pre); + memcpy(out + pre, IGNEUM_ROTR_SHIFTS, b); + memcpy(out + pre + b, at + a, *srcLen - pre - a); + out[*srcLen - a + b] = 0; + *srcLen = *srcLen - a + b; + *patched = 1; + if (!quiet) printf("intel: rotr_var rewritten to the shift form before the build (the rotate fold of 7 October 2026)\n"); + return out; +} +#endif diff --git a/proto-opencl/test_intel_rotr.c b/proto-opencl/test_intel_rotr.c new file mode 100644 index 000000000..e495722c2 --- /dev/null +++ b/proto-opencl/test_intel_rotr.c @@ -0,0 +1,47 @@ +/* test_intel_rotr.c: the gate's test of intel_rotr.h (7 October 2026). Feeds a kernel text holding the pack's rotr_var + * helper line through the rewrite under an Intel vendor string and under AMD's and NVIDIA's, and asserts: the Intel + * output carries the shift form and nothing else changed (the bytes before and after the line equal), the AMD and + * NVIDIA outputs are the input pointer, untouched; a text without the line is returned untouched on Intel too. + * cc -std=c99 -Wall -Wextra -o test_intel_rotr proto-opencl/test_intel_rotr.c && ./test_intel_rotr */ +#include "intel_rotr.h" +#include + +int main(void) { + const char* pre = "static inline uint rotl_imm(uint x, uint n) { return rotate(x, n); }\n"; + const char* post = "\nstatic inline uint ds_elem(uint i, uint d0, uint d1) { return i ^ d0 ^ d1; }\n"; + char text[1024]; + size_t len, plen, blen = strlen(IGNEUM_ROTR_BUILTIN), slen = strlen(IGNEUM_ROTR_SHIFTS); + int patched = -1; + char* out; + snprintf(text, sizeof(text), "%s%s%s", pre, IGNEUM_ROTR_BUILTIN, post); + len = strlen(text); + /* Intel: the one line becomes the shift form, the rest is byte for byte the same */ + plen = len; + out = igneum_intel_rotr_patch("Intel(R) Corporation", "Intel(R) OpenCL Graphics", text, &plen, &patched, 1); + assert(patched == 1); + assert(out != text); + assert(plen == len - blen + slen); + assert(strlen(out) == plen); + assert(memcmp(out, pre, strlen(pre)) == 0); + assert(memcmp(out + strlen(pre), IGNEUM_ROTR_SHIFTS, slen) == 0); + assert(strcmp(out + strlen(pre) + slen, post) == 0); + assert(strstr(out, "rotate(x, (0u - n)") == NULL); + free(out); + /* the platform name alone names Intel too (a vendor string of another spelling) */ + plen = len; patched = -1; + out = igneum_intel_rotr_patch("GenuineIntel", "Intel(R) OpenCL Graphics", text, &plen, &patched, 1); + assert(patched == 1 && out != text); free(out); + /* AMD and NVIDIA: untouched, the input pointer back, the length unchanged */ + plen = len; patched = -1; + out = igneum_intel_rotr_patch("Advanced Micro Devices, Inc.", "AMD Accelerated Parallel Processing", text, &plen, &patched, 1); + assert(patched == 0 && out == text && plen == len && strstr(text, IGNEUM_ROTR_BUILTIN) != NULL); + plen = len; patched = -1; + out = igneum_intel_rotr_patch("NVIDIA Corporation", "NVIDIA CUDA", text, &plen, &patched, 1); + assert(patched == 0 && out == text && plen == len); + /* Intel with no helper line (a probe kernel): untouched */ + plen = strlen(pre); patched = -1; + out = igneum_intel_rotr_patch("Intel(R) Corporation", "Intel(R) OpenCL Graphics", pre, &plen, &patched, 1); + assert(patched == 0 && out == pre && plen == strlen(pre)); + printf("test_intel_rotr: ok (Intel rewritten to the shift form, AMD and NVIDIA untouched, no line untouched)\n"); + return 0; +} diff --git a/tools/ci/pre-push.sh b/tools/ci/pre-push.sh index 095f7eb73..b7b8d8afb 100755 --- a/tools/ci/pre-push.sh +++ b/tools/ci/pre-push.sh @@ -83,6 +83,7 @@ tree_checks() { run "heat gate reader self-test (a known hold passes, a known drift fails)" node tools/heat-gate.mjs --self-test run "relay unit tests" node --test relay/test/parse.test.mjs relay/test/auth.test.mjs relay/test/wake.test.mjs relay/test/ember.test.mjs run "miner app notice strip and update card tests" node --test app/igneum-app/ui/notices.test.mjs app/igneum-app/ui/update-card.test.mjs app/igneum-app/ui/view.test.mjs app/igneum-app/ui/tune-line.test.mjs app/igneum-app/ui/ui-ota.test.mjs + run "the Intel rotate-fold rewrite: Intel gets the shift form, AMD and NVIDIA untouched (proto-opencl/test_intel_rotr.c)" bash -c 'cc -std=c99 -Wall -Wextra -o "${TMPDIR:-/tmp}/test_intel_rotr.$$" proto-opencl/test_intel_rotr.c && "${TMPDIR:-/tmp}/test_intel_rotr.$$"; s=$?; rm -f "${TMPDIR:-/tmp}/test_intel_rotr.$$"; exit $s' run "no secret file names and no 64-hex secrets in the tree" bash -c 'bash tools/ci/no-secrets-check.sh --self-test && bash tools/ci/no-secrets-check.sh' }