igneum/proto-cuda/emu/shim.cpp
2026-10-03 16:00:43 +00:00

39 lines
1.1 KiB
C++

#include "cuda_runtime.h"
thread_local uint3 threadIdx{0u, 0u, 0u};
thread_local uint3 blockIdx{0u, 0u, 0u};
thread_local dim3 blockDim;
thread_local dim3 gridDim;
thread_local EmuWarp* emu_current_warp = nullptr;
static void warp_barrier(EmuWarp* w) {
std::unique_lock<std::mutex> lk(w->m);
unsigned gen = w->generation;
if (++w->arrived == w->size) {
w->arrived = 0;
++w->generation;
w->cv.notify_all();
} else {
w->cv.wait(lk, [&] { return gen != w->generation; });
}
}
unsigned int __shfl_xor_sync(unsigned int, unsigned int v, int laneMask, int) {
EmuWarp* w = emu_current_warp;
unsigned lane = threadIdx.x & 31u;
w->slot[lane] = v;
warp_barrier(w);
unsigned int r = w->slot[lane ^ (unsigned)laneMask];
warp_barrier(w);
return r;
}
unsigned int __shfl_sync(unsigned int, unsigned int v, int srcLane, int) {
EmuWarp* w = emu_current_warp;
unsigned lane = threadIdx.x & 31u;
w->slot[lane] = v;
warp_barrier(w);
unsigned int r = w->slot[(unsigned)srcLane & 31u];
warp_barrier(w);
return r;
}