yum-mirror/slang

Making it easier to work with shaders

git clone https://git.yummers.dev/yum-mirror/slang

James Helferty (NVIDIA)render-test: Change D3D12 default to sm_6_5 (#8320)f02b08490

master
2.2 KiB55 linesraw
1//TEST(compute, vulkan):COMPARE_COMPUTE_EX:-vk -compute -shaderobj -output-using-type
2// Does not run on DX11 as SM 6.4 is required.
3//DISABLE_TEST(compute):COMPARE_COMPUTE_EX:-slang -compute -dx11
4//TEST(compute):COMPARE_COMPUTE_EX:-slang -compute -dx12 -profile cs_6_4 -shaderobj -output-using-type
5//TEST(compute):COMPARE_COMPUTE_EX:-metal -compute -shaderobj -output-using-type
6//TEST(compute):COMPARE_COMPUTE_EX:-wgsl -compute -shaderobj -render-feature half -output-using-type
7//TEST(compute):COMPARE_COMPUTE_EX:-cuda -compute -shaderobj -g0 -output-using-type
8//TEST(compute):COMPARE_COMPUTE_EX:-cpu -compute -shaderobj -output-using-type
9
10//TEST_INPUT:ubuffer(data=[0 0 0], stride=4):out,name outputBuffer
11RWStructuredBuffer<int> outputBuffer;
12
13[numthreads(1, 1, 1)]
14void computeMain(uint3 dispatchThreadID : SV_DispatchThreadID)
15{
16    uint outputIndex = 0;
17
18    //
19    // dot4add_u8packed()
20    // [4 3 2 1]  dot [1 2 4 2] + 5
21    // (4 * 1) + (3 * 2) + (2 * 4) + (1 * 2) + 5 = 25
22    //
23    uint unsignedX = 0x01020304U;
24    uint unsignedY = 0x02040201U;
25    uint unsignedAcc = 5U; 
26    uint unsignedResult = dot4add_u8packed(unsignedX, unsignedY, unsignedAcc);
27    outputBuffer[outputIndex++] = unsignedResult;
28
29    //
30    // dot4add_i8packed()
31    // [6 2 3 -1] dot [-2 -6 2 6] - 100
32    // (6 * -2) + (2 * -6) + (3 * 2) + (-1 * 6) - 100 = -124
33    //
34    int signedX = 0xFF030206;
35    int signedY = 0x0602FAFE;
36    int signedAcc = -100;
37    int signedResult = dot4add_i8packed(signedX, signedY, signedAcc);
38    outputBuffer[outputIndex++] = signedResult;
39
40    //
41    // dot2add()
42    // [10.8 -3.3] dot [1.4 -20.3] - 2.11 
43    // (10.8 * 1.4) + (-3.3 * -20.3) - 2.0 = 80.11
44    //
45    half2 half2X = half2(half(10.8), half(-3.3));
46    half2 half2Y = half2(half(1.4), half(-20.3));
47
48    // `half2Acc` is assigned -2.0 here.
49    // Thread index is used so that `half2Acc` will not be implicitly emitted as literal `-2.0` which
50    // may be treated as a double by DXC and cause it to fail to compile because no overload exists for `dot2add` that
51    // accepts double.
52    float half2Acc = float(dispatchThreadID.x + 1) * -2.0f;
53    float half2Result = dot2add(half2X, half2Y, half2Acc);
54    outputBuffer[outputIndex++] = int(half2Result);
55}