yum-mirror/slang
Making it easier to work with shaders
git clone https://git.yummers.dev/yum-mirror/slang
52b91231c
master
1//TEST:SIMPLE(filecheck=CUDA): -target cuda -line-directive-mode none 2//TEST:SIMPLE(filecheck=TORCH): -target torch -line-directive-mode none 3 4 5[AutoPyBindCUDA] 6[CUDAKernel] 7void plain_copy(float3[4] input, TensorView<float> output) 8{ 9 // CUDA: __global__ void __kernel__plain_copy(FixedArray<_VectorStorage_float3_0, 4> input_0, TensorView output_0) 10 // TORCH: void __kernel__plain_copy(FixedArray<_VectorStorage_float3_0, 4> _0, TensorView _1); 11 12 // Get the 'global' index of this thread. 13 uint3 dispatchIdx = cudaThreadIdx() + cudaBlockIdx() * cudaBlockDim(); 14 15 // If the thread index is beyond the input size, exit early. 16 if (dispatchIdx.x >= 1) 17 return; 18 19 output[0] = input[0].x; 20 output[1] = input[2].y; 21 output[2] = input[3].z; 22}