1 ; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
2 ; RUN: llc < %s -mtriple=x86_64-unknown-unknown -mattr=avx,+slow-unaligned-mem-32 | FileCheck %s --check-prefix=AVXSLOW
3 ; RUN: llc < %s -mtriple=x86_64-unknown-unknown -mattr=avx,-slow-unaligned-mem-32 | FileCheck %s --check-prefix=AVXFAST
4 ; RUN: llc < %s -mtriple=x86_64-unknown-unknown -mattr=avx2 | FileCheck %s --check-prefix=AVX2
6 ; Don't generate an unaligned 32-byte load on this test if that is slower than two 16-byte loads.
8 define <8 x float> @load32bytes(ptr %Ap) {
9 ; AVXSLOW-LABEL: load32bytes:
11 ; AVXSLOW-NEXT: vmovaps (%rdi), %xmm0
12 ; AVXSLOW-NEXT: vinsertf128 $1, 16(%rdi), %ymm0, %ymm0
15 ; AVXFAST-LABEL: load32bytes:
17 ; AVXFAST-NEXT: vmovups (%rdi), %ymm0
20 ; AVX2-LABEL: load32bytes:
22 ; AVX2-NEXT: vmovups (%rdi), %ymm0
24 %A = load <8 x float>, ptr %Ap, align 16
28 ; Don't generate an unaligned 32-byte store on this test if that is slower than two 16-byte loads.
30 define void @store32bytes(<8 x float> %A, ptr %P) {
31 ; AVXSLOW-LABEL: store32bytes:
33 ; AVXSLOW-NEXT: vextractf128 $1, %ymm0, 16(%rdi)
34 ; AVXSLOW-NEXT: vmovaps %xmm0, (%rdi)
35 ; AVXSLOW-NEXT: vzeroupper
38 ; AVXFAST-LABEL: store32bytes:
40 ; AVXFAST-NEXT: vmovups %ymm0, (%rdi)
41 ; AVXFAST-NEXT: vzeroupper
44 ; AVX2-LABEL: store32bytes:
46 ; AVX2-NEXT: vmovups %ymm0, (%rdi)
47 ; AVX2-NEXT: vzeroupper
49 store <8 x float> %A, ptr %P, align 16
53 ; Merge two consecutive 16-byte subvector loads into a single 32-byte load if it's faster.
55 define <8 x float> @combine_16_byte_loads_no_intrinsic(ptr %ptr) {
56 ; AVXSLOW-LABEL: combine_16_byte_loads_no_intrinsic:
58 ; AVXSLOW-NEXT: vmovups 48(%rdi), %xmm0
59 ; AVXSLOW-NEXT: vinsertf128 $1, 64(%rdi), %ymm0, %ymm0
62 ; AVXFAST-LABEL: combine_16_byte_loads_no_intrinsic:
64 ; AVXFAST-NEXT: vmovups 48(%rdi), %ymm0
67 ; AVX2-LABEL: combine_16_byte_loads_no_intrinsic:
69 ; AVX2-NEXT: vmovups 48(%rdi), %ymm0
71 %ptr1 = getelementptr inbounds <4 x float>, ptr %ptr, i64 3
72 %ptr2 = getelementptr inbounds <4 x float>, ptr %ptr, i64 4
73 %v1 = load <4 x float>, ptr %ptr1, align 1
74 %v2 = load <4 x float>, ptr %ptr2, align 1
75 %v3 = shufflevector <4 x float> %v1, <4 x float> %v2, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>
79 ; If the first load is 32-byte aligned, then the loads should be merged in all cases.
81 define <8 x float> @combine_16_byte_loads_aligned(ptr %ptr) {
82 ; AVXSLOW-LABEL: combine_16_byte_loads_aligned:
84 ; AVXSLOW-NEXT: vmovaps 48(%rdi), %ymm0
87 ; AVXFAST-LABEL: combine_16_byte_loads_aligned:
89 ; AVXFAST-NEXT: vmovaps 48(%rdi), %ymm0
92 ; AVX2-LABEL: combine_16_byte_loads_aligned:
94 ; AVX2-NEXT: vmovaps 48(%rdi), %ymm0
96 %ptr1 = getelementptr inbounds <4 x float>, ptr %ptr, i64 3
97 %ptr2 = getelementptr inbounds <4 x float>, ptr %ptr, i64 4
98 %v1 = load <4 x float>, ptr %ptr1, align 32
99 %v2 = load <4 x float>, ptr %ptr2, align 1
100 %v3 = shufflevector <4 x float> %v1, <4 x float> %v2, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>
104 ; Swap the order of the shufflevector operands to ensure that the pattern still matches.
106 define <8 x float> @combine_16_byte_loads_no_intrinsic_swap(ptr %ptr) {
107 ; AVXSLOW-LABEL: combine_16_byte_loads_no_intrinsic_swap:
109 ; AVXSLOW-NEXT: vmovups 64(%rdi), %xmm0
110 ; AVXSLOW-NEXT: vinsertf128 $1, 80(%rdi), %ymm0, %ymm0
113 ; AVXFAST-LABEL: combine_16_byte_loads_no_intrinsic_swap:
115 ; AVXFAST-NEXT: vmovups 64(%rdi), %ymm0
118 ; AVX2-LABEL: combine_16_byte_loads_no_intrinsic_swap:
120 ; AVX2-NEXT: vmovups 64(%rdi), %ymm0
122 %ptr1 = getelementptr inbounds <4 x float>, ptr %ptr, i64 4
123 %ptr2 = getelementptr inbounds <4 x float>, ptr %ptr, i64 5
124 %v1 = load <4 x float>, ptr %ptr1, align 1
125 %v2 = load <4 x float>, ptr %ptr2, align 1
126 %v3 = shufflevector <4 x float> %v2, <4 x float> %v1, <8 x i32> <i32 4, i32 5, i32 6, i32 7, i32 0, i32 1, i32 2, i32 3>
130 ; Check each element type other than float to make sure it is handled correctly.
131 ; Use the loaded values with an 'add' to make sure we're using the correct load type.
132 ; Don't generate 32-byte loads for integer ops unless we have AVX2.
134 define <4 x i64> @combine_16_byte_loads_i64(ptr %ptr, <4 x i64> %x) {
135 ; AVXSLOW-LABEL: combine_16_byte_loads_i64:
137 ; AVXSLOW-NEXT: vextractf128 $1, %ymm0, %xmm1
138 ; AVXSLOW-NEXT: vpaddq 96(%rdi), %xmm1, %xmm1
139 ; AVXSLOW-NEXT: vpaddq 80(%rdi), %xmm0, %xmm0
140 ; AVXSLOW-NEXT: vinsertf128 $1, %xmm1, %ymm0, %ymm0
143 ; AVXFAST-LABEL: combine_16_byte_loads_i64:
145 ; AVXFAST-NEXT: vextractf128 $1, %ymm0, %xmm1
146 ; AVXFAST-NEXT: vpaddq 96(%rdi), %xmm1, %xmm1
147 ; AVXFAST-NEXT: vpaddq 80(%rdi), %xmm0, %xmm0
148 ; AVXFAST-NEXT: vinsertf128 $1, %xmm1, %ymm0, %ymm0
151 ; AVX2-LABEL: combine_16_byte_loads_i64:
153 ; AVX2-NEXT: vpaddq 80(%rdi), %ymm0, %ymm0
155 %ptr1 = getelementptr inbounds <2 x i64>, ptr %ptr, i64 5
156 %ptr2 = getelementptr inbounds <2 x i64>, ptr %ptr, i64 6
157 %v1 = load <2 x i64>, ptr %ptr1, align 1
158 %v2 = load <2 x i64>, ptr %ptr2, align 1
159 %v3 = shufflevector <2 x i64> %v1, <2 x i64> %v2, <4 x i32> <i32 0, i32 1, i32 2, i32 3>
160 %v4 = add <4 x i64> %v3, %x
164 define <8 x i32> @combine_16_byte_loads_i32(ptr %ptr, <8 x i32> %x) {
165 ; AVXSLOW-LABEL: combine_16_byte_loads_i32:
167 ; AVXSLOW-NEXT: vextractf128 $1, %ymm0, %xmm1
168 ; AVXSLOW-NEXT: vpaddd 112(%rdi), %xmm1, %xmm1
169 ; AVXSLOW-NEXT: vpaddd 96(%rdi), %xmm0, %xmm0
170 ; AVXSLOW-NEXT: vinsertf128 $1, %xmm1, %ymm0, %ymm0
173 ; AVXFAST-LABEL: combine_16_byte_loads_i32:
175 ; AVXFAST-NEXT: vextractf128 $1, %ymm0, %xmm1
176 ; AVXFAST-NEXT: vpaddd 112(%rdi), %xmm1, %xmm1
177 ; AVXFAST-NEXT: vpaddd 96(%rdi), %xmm0, %xmm0
178 ; AVXFAST-NEXT: vinsertf128 $1, %xmm1, %ymm0, %ymm0
181 ; AVX2-LABEL: combine_16_byte_loads_i32:
183 ; AVX2-NEXT: vpaddd 96(%rdi), %ymm0, %ymm0
185 %ptr1 = getelementptr inbounds <4 x i32>, ptr %ptr, i64 6
186 %ptr2 = getelementptr inbounds <4 x i32>, ptr %ptr, i64 7
187 %v1 = load <4 x i32>, ptr %ptr1, align 1
188 %v2 = load <4 x i32>, ptr %ptr2, align 1
189 %v3 = shufflevector <4 x i32> %v1, <4 x i32> %v2, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>
190 %v4 = add <8 x i32> %v3, %x
194 define <16 x i16> @combine_16_byte_loads_i16(ptr %ptr, <16 x i16> %x) {
195 ; AVXSLOW-LABEL: combine_16_byte_loads_i16:
197 ; AVXSLOW-NEXT: vextractf128 $1, %ymm0, %xmm1
198 ; AVXSLOW-NEXT: vpaddw 128(%rdi), %xmm1, %xmm1
199 ; AVXSLOW-NEXT: vpaddw 112(%rdi), %xmm0, %xmm0
200 ; AVXSLOW-NEXT: vinsertf128 $1, %xmm1, %ymm0, %ymm0
203 ; AVXFAST-LABEL: combine_16_byte_loads_i16:
205 ; AVXFAST-NEXT: vextractf128 $1, %ymm0, %xmm1
206 ; AVXFAST-NEXT: vpaddw 128(%rdi), %xmm1, %xmm1
207 ; AVXFAST-NEXT: vpaddw 112(%rdi), %xmm0, %xmm0
208 ; AVXFAST-NEXT: vinsertf128 $1, %xmm1, %ymm0, %ymm0
211 ; AVX2-LABEL: combine_16_byte_loads_i16:
213 ; AVX2-NEXT: vpaddw 112(%rdi), %ymm0, %ymm0
215 %ptr1 = getelementptr inbounds <8 x i16>, ptr %ptr, i64 7
216 %ptr2 = getelementptr inbounds <8 x i16>, ptr %ptr, i64 8
217 %v1 = load <8 x i16>, ptr %ptr1, align 1
218 %v2 = load <8 x i16>, ptr %ptr2, align 1
219 %v3 = shufflevector <8 x i16> %v1, <8 x i16> %v2, <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15>
220 %v4 = add <16 x i16> %v3, %x
224 define <32 x i8> @combine_16_byte_loads_i8(ptr %ptr, <32 x i8> %x) {
225 ; AVXSLOW-LABEL: combine_16_byte_loads_i8:
227 ; AVXSLOW-NEXT: vextractf128 $1, %ymm0, %xmm1
228 ; AVXSLOW-NEXT: vpaddb 144(%rdi), %xmm1, %xmm1
229 ; AVXSLOW-NEXT: vpaddb 128(%rdi), %xmm0, %xmm0
230 ; AVXSLOW-NEXT: vinsertf128 $1, %xmm1, %ymm0, %ymm0
233 ; AVXFAST-LABEL: combine_16_byte_loads_i8:
235 ; AVXFAST-NEXT: vextractf128 $1, %ymm0, %xmm1
236 ; AVXFAST-NEXT: vpaddb 144(%rdi), %xmm1, %xmm1
237 ; AVXFAST-NEXT: vpaddb 128(%rdi), %xmm0, %xmm0
238 ; AVXFAST-NEXT: vinsertf128 $1, %xmm1, %ymm0, %ymm0
241 ; AVX2-LABEL: combine_16_byte_loads_i8:
243 ; AVX2-NEXT: vpaddb 128(%rdi), %ymm0, %ymm0
245 %ptr1 = getelementptr inbounds <16 x i8>, ptr %ptr, i64 8
246 %ptr2 = getelementptr inbounds <16 x i8>, ptr %ptr, i64 9
247 %v1 = load <16 x i8>, ptr %ptr1, align 1
248 %v2 = load <16 x i8>, ptr %ptr2, align 1
249 %v3 = shufflevector <16 x i8> %v1, <16 x i8> %v2, <32 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15, i32 16, i32 17, i32 18, i32 19, i32 20, i32 21, i32 22, i32 23, i32 24, i32 25, i32 26, i32 27, i32 28, i32 29, i32 30, i32 31>
250 %v4 = add <32 x i8> %v3, %x
254 define <4 x double> @combine_16_byte_loads_double(ptr %ptr, <4 x double> %x) {
255 ; AVXSLOW-LABEL: combine_16_byte_loads_double:
257 ; AVXSLOW-NEXT: vmovups 144(%rdi), %xmm1
258 ; AVXSLOW-NEXT: vinsertf128 $1, 160(%rdi), %ymm1, %ymm1
259 ; AVXSLOW-NEXT: vaddpd %ymm0, %ymm1, %ymm0
262 ; AVXFAST-LABEL: combine_16_byte_loads_double:
264 ; AVXFAST-NEXT: vaddpd 144(%rdi), %ymm0, %ymm0
267 ; AVX2-LABEL: combine_16_byte_loads_double:
269 ; AVX2-NEXT: vaddpd 144(%rdi), %ymm0, %ymm0
271 %ptr1 = getelementptr inbounds <2 x double>, ptr %ptr, i64 9
272 %ptr2 = getelementptr inbounds <2 x double>, ptr %ptr, i64 10
273 %v1 = load <2 x double>, ptr %ptr1, align 1
274 %v2 = load <2 x double>, ptr %ptr2, align 1
275 %v3 = shufflevector <2 x double> %v1, <2 x double> %v2, <4 x i32> <i32 0, i32 1, i32 2, i32 3>
276 %v4 = fadd <4 x double> %v3, %x