Intel's compiler turns rotate(x, (0u - n) & 31u) into a rotate LEFT by n: lane 0's register trace on the B580 diverged at instruction 6 of iteration 0 (rotr) and nowhere before, in both exchange modes, with every other family and the dataset kernels bit-exact. proto-opencl/intel_rotr.h rewrites the one helper line when the device's vendor or platform string holds Intel (host.c's buildProgram and the prepare path), no other vendor sees a change, no pack or consensus text moves. proto-opencl/test_intel_rotr.c (the pre-push gate runs it) feeds the line through the rewrite under Intel, AMD and NVIDIA strings and asserts the outputs. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com> (cherry picked from commit 26e135a362718a68a842da59080b43e93afd9dc2)
47 lines
2.8 KiB
C
47 lines
2.8 KiB
C
/* test_intel_rotr.c: the gate's test of intel_rotr.h (7 October 2026). Feeds a kernel text holding the pack's rotr_var
|
|
* helper line through the rewrite under an Intel vendor string and under AMD's and NVIDIA's, and asserts: the Intel
|
|
* output carries the shift form and nothing else changed (the bytes before and after the line equal), the AMD and
|
|
* NVIDIA outputs are the input pointer, untouched; a text without the line is returned untouched on Intel too.
|
|
* cc -std=c99 -Wall -Wextra -o test_intel_rotr proto-opencl/test_intel_rotr.c && ./test_intel_rotr */
|
|
#include "intel_rotr.h"
|
|
#include <assert.h>
|
|
|
|
int main(void) {
|
|
const char* pre = "static inline uint rotl_imm(uint x, uint n) { return rotate(x, n); }\n";
|
|
const char* post = "\nstatic inline uint ds_elem(uint i, uint d0, uint d1) { return i ^ d0 ^ d1; }\n";
|
|
char text[1024];
|
|
size_t len, plen, blen = strlen(IGNEUM_ROTR_BUILTIN), slen = strlen(IGNEUM_ROTR_SHIFTS);
|
|
int patched = -1;
|
|
char* out;
|
|
snprintf(text, sizeof(text), "%s%s%s", pre, IGNEUM_ROTR_BUILTIN, post);
|
|
len = strlen(text);
|
|
/* Intel: the one line becomes the shift form, the rest is byte for byte the same */
|
|
plen = len;
|
|
out = igneum_intel_rotr_patch("Intel(R) Corporation", "Intel(R) OpenCL Graphics", text, &plen, &patched, 1);
|
|
assert(patched == 1);
|
|
assert(out != text);
|
|
assert(plen == len - blen + slen);
|
|
assert(strlen(out) == plen);
|
|
assert(memcmp(out, pre, strlen(pre)) == 0);
|
|
assert(memcmp(out + strlen(pre), IGNEUM_ROTR_SHIFTS, slen) == 0);
|
|
assert(strcmp(out + strlen(pre) + slen, post) == 0);
|
|
assert(strstr(out, "rotate(x, (0u - n)") == NULL);
|
|
free(out);
|
|
/* the platform name alone names Intel too (a vendor string of another spelling) */
|
|
plen = len; patched = -1;
|
|
out = igneum_intel_rotr_patch("GenuineIntel", "Intel(R) OpenCL Graphics", text, &plen, &patched, 1);
|
|
assert(patched == 1 && out != text); free(out);
|
|
/* AMD and NVIDIA: untouched, the input pointer back, the length unchanged */
|
|
plen = len; patched = -1;
|
|
out = igneum_intel_rotr_patch("Advanced Micro Devices, Inc.", "AMD Accelerated Parallel Processing", text, &plen, &patched, 1);
|
|
assert(patched == 0 && out == text && plen == len && strstr(text, IGNEUM_ROTR_BUILTIN) != NULL);
|
|
plen = len; patched = -1;
|
|
out = igneum_intel_rotr_patch("NVIDIA Corporation", "NVIDIA CUDA", text, &plen, &patched, 1);
|
|
assert(patched == 0 && out == text && plen == len);
|
|
/* Intel with no helper line (a probe kernel): untouched */
|
|
plen = strlen(pre); patched = -1;
|
|
out = igneum_intel_rotr_patch("Intel(R) Corporation", "Intel(R) OpenCL Graphics", pre, &plen, &patched, 1);
|
|
assert(patched == 0 && out == pre && plen == strlen(pre));
|
|
printf("test_intel_rotr: ok (Intel rewritten to the shift form, AMD and NVIDIA untouched, no line untouched)\n");
|
|
return 0;
|
|
}
|