#include "cuda_runtime.h" thread_local uint3 threadIdx{0u, 0u, 0u}; thread_local uint3 blockIdx{0u, 0u, 0u}; thread_local dim3 blockDim; thread_local dim3 gridDim; thread_local EmuWarp* emu_current_warp = nullptr; static void warp_barrier(EmuWarp* w) { std::unique_lock lk(w->m); unsigned gen = w->generation; if (++w->arrived == w->size) { w->arrived = 0; ++w->generation; w->cv.notify_all(); } else { w->cv.wait(lk, [&] { return gen != w->generation; }); } } unsigned int __shfl_xor_sync(unsigned int, unsigned int v, int laneMask, int) { EmuWarp* w = emu_current_warp; unsigned lane = threadIdx.x & 31u; w->slot[lane] = v; warp_barrier(w); unsigned int r = w->slot[lane ^ (unsigned)laneMask]; warp_barrier(w); return r; } unsigned int __shfl_sync(unsigned int, unsigned int v, int srcLane, int) { EmuWarp* w = emu_current_warp; unsigned lane = threadIdx.x & 31u; w->slot[lane] = v; warp_barrier(w); unsigned int r = w->slot[(unsigned)srcLane & 31u]; warp_barrier(w); return r; }