yum-mirror/slang

Making it easier to work with shaders

git clone https://git.yummers.dev/yum-mirror/slang

jarcherNVEnable CUDA testing for batch 2 (#8147)3618f7a4c

master
19.9 KiB948 linesraw
1#ifndef SLANG_PRELUDE_SCALAR_INTRINSICS_H
2#define SLANG_PRELUDE_SCALAR_INTRINSICS_H
3
4#if !defined(SLANG_LLVM) && SLANG_PROCESSOR_X86_64 && SLANG_VC
5//  If we have visual studio and 64 bit processor, we can assume we have popcnt, and can include
6//  x86 intrinsics
7#include <intrin.h>
8#endif
9
10#ifndef SLANG_FORCE_INLINE
11#define SLANG_FORCE_INLINE inline
12#endif
13
14#ifdef SLANG_PRELUDE_NAMESPACE
15namespace SLANG_PRELUDE_NAMESPACE
16{
17#endif
18
19#ifndef SLANG_PRELUDE_PI
20#define SLANG_PRELUDE_PI 3.14159265358979323846
21#endif
22
23
24union Union32
25{
26    uint32_t u;
27    int32_t i;
28    float f;
29};
30
31union Union64
32{
33    uint64_t u;
34    int64_t i;
35    double d;
36};
37
38// 32 bit cast conversions
39SLANG_FORCE_INLINE int32_t _bitCastFloatToInt(float f)
40{
41    Union32 u;
42    u.f = f;
43    return u.i;
44}
45SLANG_FORCE_INLINE float _bitCastIntToFloat(int32_t i)
46{
47    Union32 u;
48    u.i = i;
49    return u.f;
50}
51SLANG_FORCE_INLINE uint32_t _bitCastFloatToUInt(float f)
52{
53    Union32 u;
54    u.f = f;
55    return u.u;
56}
57SLANG_FORCE_INLINE float _bitCastUIntToFloat(uint32_t ui)
58{
59    Union32 u;
60    u.u = ui;
61    return u.f;
62}
63
64// ----------------------------- F16 -----------------------------------------
65
66
67// This impl is based on FloatToHalf that is in Slang codebase
68SLANG_FORCE_INLINE uint32_t f32tof16(const float value)
69{
70    const uint32_t inBits = _bitCastFloatToUInt(value);
71
72    // bits initially set to just the sign bit
73    uint32_t bits = (inBits >> 16) & 0x8000;
74    // Mantissa can't be used as is, as it holds last bit, for rounding.
75    uint32_t m = (inBits >> 12) & 0x07ff;
76    uint32_t e = (inBits >> 23) & 0xff;
77
78    if (e < 103)
79    {
80        // It's zero
81        return bits;
82    }
83    if (e == 0xff)
84    {
85        // Could be a NAN or INF. Is INF if *input* mantissa is 0.
86
87        // Remove last bit for rounding to make output mantissa.
88        m >>= 1;
89
90        // We *assume* float16/float32 signaling bit and remaining bits
91        // semantics are the same. (The signalling bit convention is target specific!).
92        // Non signal bit's usage within mantissa for a NAN are also target specific.
93
94        // If the m is 0, it could be because the result is INF, but it could also be because all
95        // the bits that made NAN were dropped as we have less mantissa bits in f16.
96
97        // To fix for this we make non zero if m is 0 and the input mantissa was not.
98        // This will (typically) produce a signalling NAN.
99        m += uint32_t(m == 0 && (inBits & 0x007fffffu));
100
101        // Combine for output
102        return (bits | 0x7c00u | m);
103    }
104    if (e > 142)
105    {
106        // INF.
107        return bits | 0x7c00u;
108    }
109    if (e < 113)
110    {
111        m |= 0x0800u;
112        bits |= (m >> (114 - e)) + ((m >> (113 - e)) & 1);
113        return bits;
114    }
115    bits |= ((e - 112) << 10) | (m >> 1);
116    bits += m & 1;
117    return bits;
118}
119
120static const float g_f16tof32Magic = _bitCastIntToFloat((127 + (127 - 15)) << 23);
121
122SLANG_FORCE_INLINE float f16tof32(const uint32_t value)
123{
124    const uint32_t sign = (value & 0x8000) << 16;
125    uint32_t exponent = (value & 0x7c00) >> 10;
126    uint32_t mantissa = (value & 0x03ff);
127
128    if (exponent == 0)
129    {
130        // If mantissa is 0 we are done, as output is 0.
131        // If it's not zero we must have a denormal.
132        if (mantissa)
133        {
134            // We have a denormal so use the magic to do exponent adjust
135            return _bitCastIntToFloat(sign | ((value & 0x7fff) << 13)) * g_f16tof32Magic;
136        }
137    }
138    else
139    {
140        // If the exponent is NAN or INF exponent is 0x1f on input.
141        // If that's the case, we just need to set the exponent to 0xff on output
142        // and the mantissa can just stay the same. If its 0 it's INF, else it is NAN and we just
143        // copy the bits
144        //
145        // Else we need to correct the exponent in the normalized case.
146        exponent = (exponent == 0x1F) ? 0xff : (exponent + (-15 + 127));
147    }
148
149    return _bitCastUIntToFloat(sign | (exponent << 23) | (mantissa << 13));
150}
151
152// ----------------------------- F32 -----------------------------------------
153
154// Helpers
155SLANG_FORCE_INLINE float F32_calcSafeRadians(float radians);
156
157#ifdef SLANG_LLVM
158
159SLANG_PRELUDE_EXTERN_C_START
160
161// Unary
162float F32_ceil(float f);
163float F32_floor(float f);
164float F32_round(float f);
165float F32_sin(float f);
166float F32_cos(float f);
167float F32_tan(float f);
168float F32_asin(float f);
169float F32_acos(float f);
170float F32_atan(float f);
171float F32_sinh(float f);
172float F32_cosh(float f);
173float F32_tanh(float f);
174float F32_log2(float f);
175float F32_log(float f);
176float F32_log10(float f);
177float F32_exp2(float f);
178float F32_exp(float f);
179float F32_abs(float f);
180float F32_trunc(float f);
181float F32_sqrt(float f);
182
183bool F32_isnan(float f);
184bool F32_isfinite(float f);
185bool F32_isinf(float f);
186
187// Binary
188SLANG_FORCE_INLINE float F32_min(float a, float b)
189{
190    return a < b ? a : b;
191}
192SLANG_FORCE_INLINE float F32_max(float a, float b)
193{
194    return a > b ? a : b;
195}
196float F32_pow(float a, float b);
197float F32_fmod(float a, float b);
198float F32_remainder(float a, float b);
199float F32_atan2(float a, float b);
200
201float F32_frexp(float x, int* e);
202
203float F32_modf(float x, float* ip);
204
205// Ternary
206SLANG_FORCE_INLINE float F32_fma(float a, float b, float c)
207{
208    return a * b + c;
209}
210
211SLANG_PRELUDE_EXTERN_C_END
212
213#else
214
215// Unary
216SLANG_FORCE_INLINE float F32_ceil(float f)
217{
218    return ::ceilf(f);
219}
220SLANG_FORCE_INLINE float F32_floor(float f)
221{
222    return ::floorf(f);
223}
224SLANG_FORCE_INLINE float F32_round(float f)
225{
226    return ::roundf(f);
227}
228SLANG_FORCE_INLINE float F32_sin(float f)
229{
230    return ::sinf(f);
231}
232SLANG_FORCE_INLINE float F32_cos(float f)
233{
234    return ::cosf(f);
235}
236SLANG_FORCE_INLINE float F32_tan(float f)
237{
238    return ::tanf(f);
239}
240SLANG_FORCE_INLINE float F32_asin(float f)
241{
242    return ::asinf(f);
243}
244SLANG_FORCE_INLINE float F32_acos(float f)
245{
246    return ::acosf(f);
247}
248SLANG_FORCE_INLINE float F32_atan(float f)
249{
250    return ::atanf(f);
251}
252SLANG_FORCE_INLINE float F32_sinh(float f)
253{
254    return ::sinhf(f);
255}
256SLANG_FORCE_INLINE float F32_cosh(float f)
257{
258    return ::coshf(f);
259}
260SLANG_FORCE_INLINE float F32_tanh(float f)
261{
262    return ::tanhf(f);
263}
264SLANG_FORCE_INLINE float F32_asinh(float f)
265{
266    return ::asinhf(f);
267}
268SLANG_FORCE_INLINE float F32_acosh(float f)
269{
270    return ::acoshf(f);
271}
272SLANG_FORCE_INLINE float F32_atanh(float f)
273{
274    return ::atanhf(f);
275}
276SLANG_FORCE_INLINE float F32_log2(float f)
277{
278    return ::log2f(f);
279}
280SLANG_FORCE_INLINE float F32_log(float f)
281{
282    return ::logf(f);
283}
284SLANG_FORCE_INLINE float F32_log10(float f)
285{
286    return ::log10f(f);
287}
288SLANG_FORCE_INLINE float F32_exp2(float f)
289{
290    return ::exp2f(f);
291}
292SLANG_FORCE_INLINE float F32_exp(float f)
293{
294    return ::expf(f);
295}
296SLANG_FORCE_INLINE float F32_abs(float f)
297{
298    return ::fabsf(f);
299}
300SLANG_FORCE_INLINE float F32_trunc(float f)
301{
302    return ::truncf(f);
303}
304SLANG_FORCE_INLINE float F32_sqrt(float f)
305{
306    return ::sqrtf(f);
307}
308
309SLANG_FORCE_INLINE bool F32_isnan(float f)
310{
311    return SLANG_PRELUDE_STD isnan(f);
312}
313SLANG_FORCE_INLINE bool F32_isfinite(float f)
314{
315    return SLANG_PRELUDE_STD isfinite(f);
316}
317SLANG_FORCE_INLINE bool F32_isinf(float f)
318{
319    return SLANG_PRELUDE_STD isinf(f);
320}
321
322// Binary
323SLANG_FORCE_INLINE float F32_min(float a, float b)
324{
325    return ::fminf(a, b);
326}
327SLANG_FORCE_INLINE float F32_max(float a, float b)
328{
329    return ::fmaxf(a, b);
330}
331SLANG_FORCE_INLINE float F32_pow(float a, float b)
332{
333    return ::powf(a, b);
334}
335SLANG_FORCE_INLINE float F32_fmod(float a, float b)
336{
337    return ::fmodf(a, b);
338}
339SLANG_FORCE_INLINE float F32_remainder(float a, float b)
340{
341    return ::remainderf(a, b);
342}
343SLANG_FORCE_INLINE float F32_atan2(float a, float b)
344{
345    return float(::atan2(a, b));
346}
347
348SLANG_FORCE_INLINE float F32_frexp(float x, int* e)
349{
350    return ::frexpf(x, e);
351}
352
353SLANG_FORCE_INLINE float F32_modf(float x, float* ip)
354{
355    return ::modff(x, ip);
356}
357
358// Ternary
359SLANG_FORCE_INLINE float F32_fma(float a, float b, float c)
360{
361    return ::fmaf(a, b, c);
362}
363
364#endif
365
366SLANG_FORCE_INLINE float F32_calcSafeRadians(float radians)
367{
368    // Put 0 to 2pi cycles to cycle around 0 to 1
369    float a = radians * (1.0f / float(SLANG_PRELUDE_PI * 2));
370    // Get truncated fraction, as value in  0 - 1 range
371    a = a - F32_floor(a);
372    // Convert back to 0 - 2pi range
373    return (a * float(SLANG_PRELUDE_PI * 2));
374}
375
376SLANG_FORCE_INLINE float F32_rsqrt(float f)
377{
378    return 1.0f / F32_sqrt(f);
379}
380SLANG_FORCE_INLINE float F32_sign(float f)
381{
382    return (f == 0.0f) ? f : ((f < 0.0f) ? -1.0f : 1.0f);
383}
384SLANG_FORCE_INLINE float F32_frac(float f)
385{
386    return f - F32_floor(f);
387}
388
389SLANG_FORCE_INLINE uint32_t F32_asuint(float f)
390{
391    Union32 u;
392    u.f = f;
393    return u.u;
394}
395SLANG_FORCE_INLINE int32_t F32_asint(float f)
396{
397    Union32 u;
398    u.f = f;
399    return u.i;
400}
401
402// ----------------------------- F64 -----------------------------------------
403
404SLANG_FORCE_INLINE double F64_calcSafeRadians(double radians);
405
406#ifdef SLANG_LLVM
407
408SLANG_PRELUDE_EXTERN_C_START
409
410// Unary
411double F64_ceil(double f);
412double F64_floor(double f);
413double F64_round(double f);
414double F64_sin(double f);
415double F64_cos(double f);
416double F64_tan(double f);
417double F64_asin(double f);
418double F64_acos(double f);
419double F64_atan(double f);
420double F64_sinh(double f);
421double F64_cosh(double f);
422double F64_tanh(double f);
423double F64_log2(double f);
424double F64_log(double f);
425double F64_log10(double f);
426double F64_exp2(double f);
427double F64_exp(double f);
428double F64_abs(double f);
429double F64_trunc(double f);
430double F64_sqrt(double f);
431
432bool F64_isnan(double f);
433bool F64_isfinite(double f);
434bool F64_isinf(double f);
435
436// Binary
437SLANG_FORCE_INLINE double F64_min(double a, double b)
438{
439    return a < b ? a : b;
440}
441SLANG_FORCE_INLINE double F64_max(double a, double b)
442{
443    return a > b ? a : b;
444}
445double F64_pow(double a, double b);
446double F64_fmod(double a, double b);
447double F64_remainder(double a, double b);
448double F64_atan2(double a, double b);
449
450double F64_frexp(double x, int* e);
451
452double F64_modf(double x, double* ip);
453
454// Ternary
455SLANG_FORCE_INLINE double F64_fma(double a, double b, double c)
456{
457    return a * b + c;
458}
459
460SLANG_PRELUDE_EXTERN_C_END
461
462#else // SLANG_LLVM
463
464// Unary
465SLANG_FORCE_INLINE double F64_ceil(double f)
466{
467    return ::ceil(f);
468}
469SLANG_FORCE_INLINE double F64_floor(double f)
470{
471    return ::floor(f);
472}
473SLANG_FORCE_INLINE double F64_round(double f)
474{
475    return ::round(f);
476}
477SLANG_FORCE_INLINE double F64_sin(double f)
478{
479    return ::sin(f);
480}
481SLANG_FORCE_INLINE double F64_cos(double f)
482{
483    return ::cos(f);
484}
485SLANG_FORCE_INLINE double F64_tan(double f)
486{
487    return ::tan(f);
488}
489SLANG_FORCE_INLINE double F64_asin(double f)
490{
491    return ::asin(f);
492}
493SLANG_FORCE_INLINE double F64_acos(double f)
494{
495    return ::acos(f);
496}
497SLANG_FORCE_INLINE double F64_atan(double f)
498{
499    return ::atan(f);
500}
501SLANG_FORCE_INLINE double F64_sinh(double f)
502{
503    return ::sinh(f);
504}
505SLANG_FORCE_INLINE double F64_cosh(double f)
506{
507    return ::cosh(f);
508}
509SLANG_FORCE_INLINE double F64_tanh(double f)
510{
511    return ::tanh(f);
512}
513SLANG_FORCE_INLINE double F64_log2(double f)
514{
515    return ::log2(f);
516}
517SLANG_FORCE_INLINE double F64_log(double f)
518{
519    return ::log(f);
520}
521SLANG_FORCE_INLINE double F64_log10(float f)
522{
523    return ::log10(f);
524}
525SLANG_FORCE_INLINE double F64_exp2(double f)
526{
527    return ::exp2(f);
528}
529SLANG_FORCE_INLINE double F64_exp(double f)
530{
531    return ::exp(f);
532}
533SLANG_FORCE_INLINE double F64_abs(double f)
534{
535    return ::fabs(f);
536}
537SLANG_FORCE_INLINE double F64_trunc(double f)
538{
539    return ::trunc(f);
540}
541SLANG_FORCE_INLINE double F64_sqrt(double f)
542{
543    return ::sqrt(f);
544}
545
546
547SLANG_FORCE_INLINE bool F64_isnan(double f)
548{
549    return SLANG_PRELUDE_STD isnan(f);
550}
551SLANG_FORCE_INLINE bool F64_isfinite(double f)
552{
553    return SLANG_PRELUDE_STD isfinite(f);
554}
555SLANG_FORCE_INLINE bool F64_isinf(double f)
556{
557    return SLANG_PRELUDE_STD isinf(f);
558}
559
560// Binary
561SLANG_FORCE_INLINE double F64_min(double a, double b)
562{
563    return ::fmin(a, b);
564}
565SLANG_FORCE_INLINE double F64_max(double a, double b)
566{
567    return ::fmax(a, b);
568}
569SLANG_FORCE_INLINE double F64_pow(double a, double b)
570{
571    return ::pow(a, b);
572}
573SLANG_FORCE_INLINE double F64_fmod(double a, double b)
574{
575    return ::fmod(a, b);
576}
577SLANG_FORCE_INLINE double F64_remainder(double a, double b)
578{
579    return ::remainder(a, b);
580}
581SLANG_FORCE_INLINE double F64_atan2(double a, double b)
582{
583    return ::atan2(a, b);
584}
585
586SLANG_FORCE_INLINE double F64_frexp(double x, int* e)
587{
588    return ::frexp(x, e);
589}
590
591SLANG_FORCE_INLINE double F64_modf(double x, double* ip)
592{
593    return ::modf(x, ip);
594}
595
596// Ternary
597SLANG_FORCE_INLINE double F64_fma(double a, double b, double c)
598{
599    return ::fma(a, b, c);
600}
601
602#endif // SLANG_LLVM
603
604SLANG_FORCE_INLINE double F64_rsqrt(double f)
605{
606    return 1.0 / F64_sqrt(f);
607}
608SLANG_FORCE_INLINE double F64_sign(double f)
609{
610    return (f == 0.0) ? f : ((f < 0.0) ? -1.0 : 1.0);
611}
612SLANG_FORCE_INLINE double F64_frac(double f)
613{
614    return f - F64_floor(f);
615}
616
617SLANG_FORCE_INLINE void F64_asuint(double d, uint32_t* low, uint32_t* hi)
618{
619    Union64 u;
620    u.d = d;
621    *low = uint32_t(u.u);
622    *hi = uint32_t(u.u >> 32);
623}
624
625SLANG_FORCE_INLINE void F64_asint(double d, int32_t* low, int32_t* hi)
626{
627    Union64 u;
628    u.d = d;
629    *low = int32_t(u.u);
630    *hi = int32_t(u.u >> 32);
631}
632
633SLANG_FORCE_INLINE double F64_calcSafeRadians(double radians)
634{
635    // Put 0 to 2pi cycles to cycle around 0 to 1
636    double a = radians * (1.0f / (SLANG_PRELUDE_PI * 2));
637    // Get truncated fraction, as value in  0 - 1 range
638    a = a - F64_floor(a);
639    // Convert back to 0 - 2pi range
640    return (a * (SLANG_PRELUDE_PI * 2));
641}
642
643// ----------------------------- U16 -----------------------------------------
644SLANG_FORCE_INLINE uint32_t U16_countbits(uint16_t v)
645{
646#if SLANG_GCC_FAMILY && !defined(SLANG_LLVM)
647    return __builtin_popcount(uint32_t(v));
648#elif SLANG_PROCESSOR_X86_64 && SLANG_VC
649    return __popcnt16(v);
650#else
651    uint32_t c = 0;
652    while (v)
653    {
654        c++;
655        v &= v - 1;
656    }
657    return c;
658#endif
659}
660
661// ----------------------------- I16 -----------------------------------------
662SLANG_FORCE_INLINE uint32_t I16_countbits(int16_t v)
663{
664    return U16_countbits(uint16_t(v));
665}
666
667// ----------------------------- U8 -----------------------------------------
668SLANG_FORCE_INLINE uint32_t U8_countbits(uint8_t v)
669{
670    // No native 8bit __popcnt yet, just cast and use 16bit variant
671    return U16_countbits(uint16_t(v));
672}
673
674// ----------------------------- I8 -----------------------------------------
675SLANG_FORCE_INLINE uint32_t I8_countbits(int16_t v)
676{
677    return U8_countbits(uint8_t(v));
678}
679
680// ----------------------------- U32 -----------------------------------------
681
682SLANG_FORCE_INLINE uint32_t U32_abs(uint32_t f)
683{
684    return f;
685}
686
687SLANG_FORCE_INLINE uint32_t U32_min(uint32_t a, uint32_t b)
688{
689    return a < b ? a : b;
690}
691SLANG_FORCE_INLINE uint32_t U32_max(uint32_t a, uint32_t b)
692{
693    return a > b ? a : b;
694}
695
696SLANG_FORCE_INLINE float U32_asfloat(uint32_t x)
697{
698    Union32 u;
699    u.u = x;
700    return u.f;
701}
702SLANG_FORCE_INLINE uint32_t U32_asint(int32_t x)
703{
704    return uint32_t(x);
705}
706
707SLANG_FORCE_INLINE double U32_asdouble(uint32_t low, uint32_t hi)
708{
709    Union64 u;
710    u.u = (uint64_t(hi) << 32) | low;
711    return u.d;
712}
713
714
715SLANG_FORCE_INLINE uint32_t U32_countbits(uint32_t v)
716{
717#if SLANG_GCC_FAMILY && !defined(SLANG_LLVM)
718    return __builtin_popcount(v);
719#elif SLANG_PROCESSOR_X86_64 && SLANG_VC
720    return __popcnt(v);
721#else
722    uint32_t c = 0;
723    while (v)
724    {
725        c++;
726        v &= v - 1;
727    }
728    return c;
729#endif
730}
731
732SLANG_FORCE_INLINE uint32_t U32_firstbitlow(uint32_t v)
733{
734    if (v == 0)
735        return ~0u;
736
737#if SLANG_GCC_FAMILY && !defined(SLANG_LLVM)
738    // __builtin_ctz returns number of trailing zeros, which is the 0-based index of first set bit
739    return __builtin_ctz(v);
740#elif SLANG_PROCESSOR_X86_64 && SLANG_VC
741    // _BitScanForward returns 1 on success, 0 on failure, and sets index
742    unsigned long index;
743    return _BitScanForward(&index, v) ? index : ~0u;
744#else
745    // Generic implementation - find first set bit
746    uint32_t result = 0;
747    while (result < 32 && !(v & (1u << result)))
748        result++;
749    return result;
750#endif
751}
752
753SLANG_FORCE_INLINE uint32_t U32_firstbithigh(uint32_t v)
754{
755    if ((int32_t)v < 0)
756        v = ~v;
757    if (v == 0)
758        return ~0u;
759#if SLANG_GCC_FAMILY && !defined(SLANG_LLVM)
760    // __builtin_clz returns number of leading zeros
761    // firstbithigh should return 0-based bit position of MSB
762    return 31 - __builtin_clz(v);
763#elif SLANG_PROCESSOR_X86_64 && SLANG_VC
764    // _BitScanReverse returns 1 on success, 0 on failure, and sets index
765    unsigned long index;
766    return _BitScanReverse(&index, v) ? index : ~0u;
767#else
768    // Generic implementation - find highest set bit
769    int result = 31;
770    while (result >= 0 && !(v & (1u << result)))
771        result--;
772    return result;
773#endif
774}
775
776// ----------------------------- I32 -----------------------------------------
777
778SLANG_FORCE_INLINE int32_t I32_abs(int32_t f)
779{
780    return (f < 0) ? -f : f;
781}
782
783SLANG_FORCE_INLINE int32_t I32_min(int32_t a, int32_t b)
784{
785    return a < b ? a : b;
786}
787SLANG_FORCE_INLINE int32_t I32_max(int32_t a, int32_t b)
788{
789    return a > b ? a : b;
790}
791
792SLANG_FORCE_INLINE float I32_asfloat(int32_t x)
793{
794    Union32 u;
795    u.i = x;
796    return u.f;
797}
798SLANG_FORCE_INLINE uint32_t I32_asuint(int32_t x)
799{
800    return uint32_t(x);
801}
802SLANG_FORCE_INLINE double I32_asdouble(int32_t low, int32_t hi)
803{
804    Union64 u;
805    u.u = (uint64_t(hi) << 32) | uint32_t(low);
806    return u.d;
807}
808
809SLANG_FORCE_INLINE uint32_t I32_countbits(int32_t v)
810{
811    return U32_countbits(uint32_t(v));
812}
813
814SLANG_FORCE_INLINE uint32_t I32_firstbitlow(int32_t v)
815{
816    return U32_firstbitlow(uint32_t(v));
817}
818
819SLANG_FORCE_INLINE uint32_t I32_firstbithigh(int32_t v)
820{
821    return U32_firstbithigh(uint32_t(v));
822}
823
824// ----------------------------- U64 -----------------------------------------
825
826SLANG_FORCE_INLINE uint64_t U64_abs(uint64_t f)
827{
828    return f;
829}
830
831SLANG_FORCE_INLINE uint64_t U64_min(uint64_t a, uint64_t b)
832{
833    return a < b ? a : b;
834}
835SLANG_FORCE_INLINE uint64_t U64_max(uint64_t a, uint64_t b)
836{
837    return a > b ? a : b;
838}
839
840SLANG_FORCE_INLINE uint32_t U64_countbits(uint64_t v)
841{
842#if SLANG_GCC_FAMILY && !defined(SLANG_LLVM)
843    return uint32_t(__builtin_popcountll(v));
844#elif SLANG_PROCESSOR_X86_64 && SLANG_VC
845    return uint32_t(__popcnt64(v));
846#else
847    uint32_t c = 0;
848    while (v)
849    {
850        c++;
851        v &= v - 1;
852    }
853    return c;
854#endif
855}
856
857// ----------------------------- I64 -----------------------------------------
858
859SLANG_FORCE_INLINE int64_t I64_abs(int64_t f)
860{
861    return (f < 0) ? -f : f;
862}
863
864SLANG_FORCE_INLINE int64_t I64_min(int64_t a, int64_t b)
865{
866    return a < b ? a : b;
867}
868SLANG_FORCE_INLINE int64_t I64_max(int64_t a, int64_t b)
869{
870    return a > b ? a : b;
871}
872
873SLANG_FORCE_INLINE uint32_t I64_countbits(int64_t v)
874{
875    return U64_countbits(uint64_t(v));
876}
877
878// ----------------------------- UPTR -----------------------------------------
879
880SLANG_FORCE_INLINE uintptr_t UPTR_abs(uintptr_t f)
881{
882    return f;
883}
884
885SLANG_FORCE_INLINE uintptr_t UPTR_min(uintptr_t a, uintptr_t b)
886{
887    return a < b ? a : b;
888}
889
890SLANG_FORCE_INLINE uintptr_t UPTR_max(uintptr_t a, uintptr_t b)
891{
892    return a > b ? a : b;
893}
894
895// ----------------------------- IPTR -----------------------------------------
896
897SLANG_FORCE_INLINE intptr_t IPTR_abs(intptr_t f)
898{
899    return (f < 0) ? -f : f;
900}
901
902SLANG_FORCE_INLINE intptr_t IPTR_min(intptr_t a, intptr_t b)
903{
904    return a < b ? a : b;
905}
906
907SLANG_FORCE_INLINE intptr_t IPTR_max(intptr_t a, intptr_t b)
908{
909    return a > b ? a : b;
910}
911
912// ----------------------------- Interlocked ---------------------------------
913
914#if SLANG_LLVM
915
916#else // SLANG_LLVM
917
918#ifdef _WIN32
919#include <intrin.h>
920#endif
921
922SLANG_FORCE_INLINE void InterlockedAdd(uint32_t* dest, uint32_t value, uint32_t* oldValue)
923{
924#ifdef _WIN32
925    *oldValue = _InterlockedExchangeAdd((long*)dest, (long)value);
926#else
927    *oldValue = __sync_fetch_and_add(dest, value);
928#endif
929}
930
931#endif // SLANG_LLVM
932
933
934// ----------------------- fmod --------------------------
935SLANG_FORCE_INLINE float _slang_fmod(float x, float y)
936{
937    return F32_fmod(x, y);
938}
939SLANG_FORCE_INLINE double _slang_fmod(double x, double y)
940{
941    return F64_fmod(x, y);
942}
943
944#ifdef SLANG_PRELUDE_NAMESPACE
945}
946#endif
947
948#endif