yum-mirror/slang
Making it easier to work with shaders
git clone https://git.yummers.dev/yum-mirror/slang
f02b08490
master
1//TEST(compute, vulkan):COMPARE_COMPUTE_EX:-vk -compute -shaderobj -output-using-type 2// Does not run on DX11 as SM 6.4 is required. 3//DISABLE_TEST(compute):COMPARE_COMPUTE_EX:-slang -compute -dx11 4//TEST(compute):COMPARE_COMPUTE_EX:-slang -compute -dx12 -profile cs_6_4 -shaderobj -output-using-type 5//TEST(compute):COMPARE_COMPUTE_EX:-metal -compute -shaderobj -output-using-type 6//TEST(compute):COMPARE_COMPUTE_EX:-wgsl -compute -shaderobj -render-feature half -output-using-type 7//TEST(compute):COMPARE_COMPUTE_EX:-cuda -compute -shaderobj -g0 -output-using-type 8//TEST(compute):COMPARE_COMPUTE_EX:-cpu -compute -shaderobj -output-using-type 9 10//TEST_INPUT:ubuffer(data=[0 0 0], stride=4):out,name outputBuffer 11RWStructuredBuffer<int> outputBuffer; 12 13[numthreads(1, 1, 1)] 14void computeMain(uint3 dispatchThreadID : SV_DispatchThreadID) 15{ 16 uint outputIndex = 0; 17 18 // 19 // dot4add_u8packed() 20 // [4 3 2 1] dot [1 2 4 2] + 5 21 // (4 * 1) + (3 * 2) + (2 * 4) + (1 * 2) + 5 = 25 22 // 23 uint unsignedX = 0x01020304U; 24 uint unsignedY = 0x02040201U; 25 uint unsignedAcc = 5U; 26 uint unsignedResult = dot4add_u8packed(unsignedX, unsignedY, unsignedAcc); 27 outputBuffer[outputIndex++] = unsignedResult; 28 29 // 30 // dot4add_i8packed() 31 // [6 2 3 -1] dot [-2 -6 2 6] - 100 32 // (6 * -2) + (2 * -6) + (3 * 2) + (-1 * 6) - 100 = -124 33 // 34 int signedX = 0xFF030206; 35 int signedY = 0x0602FAFE; 36 int signedAcc = -100; 37 int signedResult = dot4add_i8packed(signedX, signedY, signedAcc); 38 outputBuffer[outputIndex++] = signedResult; 39 40 // 41 // dot2add() 42 // [10.8 -3.3] dot [1.4 -20.3] - 2.11 43 // (10.8 * 1.4) + (-3.3 * -20.3) - 2.0 = 80.11 44 // 45 half2 half2X = half2(half(10.8), half(-3.3)); 46 half2 half2Y = half2(half(1.4), half(-20.3)); 47 48 // `half2Acc` is assigned -2.0 here. 49 // Thread index is used so that `half2Acc` will not be implicitly emitted as literal `-2.0` which 50 // may be treated as a double by DXC and cause it to fail to compile because no overload exists for `dot2add` that 51 // accepts double. 52 float half2Acc = float(dispatchThreadID.x + 1) * -2.0f; 53 float half2Result = dot2add(half2X, half2Y, half2Acc); 54 outputBuffer[outputIndex++] = int(half2Result); 55}