blob: bbe948ce3ad2dc5b2bdbb21c255f6b28cd4d600b [file] [log] [blame]
Dan Gohmanfce288f2009-09-09 00:09:15 +00001; RUN: llc < %s -march=arm -mattr=+neon | FileCheck %s
Bob Wilsonc0110052009-09-01 04:27:10 +00002
Bob Wilsonec1d81c2009-10-06 21:16:19 +00003%struct.__neon_int8x8x2_t = type { <8 x i8>, <8 x i8> }
4%struct.__neon_int16x4x2_t = type { <4 x i16>, <4 x i16> }
5%struct.__neon_int32x2x2_t = type { <2 x i32>, <2 x i32> }
6%struct.__neon_float32x2x2_t = type { <2 x float>, <2 x float> }
Bob Wilsonc0110052009-09-01 04:27:10 +00007
8define <8 x i8> @vld2lanei8(i8* %A, <8 x i8>* %B) nounwind {
9;CHECK: vld2lanei8:
10;CHECK: vld2.8
11 %tmp1 = load <8 x i8>* %B
Bob Wilsonec1d81c2009-10-06 21:16:19 +000012 %tmp2 = call %struct.__neon_int8x8x2_t @llvm.arm.neon.vld2lane.v8i8(i8* %A, <8 x i8> %tmp1, <8 x i8> %tmp1, i32 1)
13 %tmp3 = extractvalue %struct.__neon_int8x8x2_t %tmp2, 0
14 %tmp4 = extractvalue %struct.__neon_int8x8x2_t %tmp2, 1
Bob Wilsonc0110052009-09-01 04:27:10 +000015 %tmp5 = add <8 x i8> %tmp3, %tmp4
16 ret <8 x i8> %tmp5
17}
18
19define <4 x i16> @vld2lanei16(i16* %A, <4 x i16>* %B) nounwind {
20;CHECK: vld2lanei16:
21;CHECK: vld2.16
22 %tmp1 = load <4 x i16>* %B
Bob Wilsonec1d81c2009-10-06 21:16:19 +000023 %tmp2 = call %struct.__neon_int16x4x2_t @llvm.arm.neon.vld2lane.v4i16(i16* %A, <4 x i16> %tmp1, <4 x i16> %tmp1, i32 1)
24 %tmp3 = extractvalue %struct.__neon_int16x4x2_t %tmp2, 0
25 %tmp4 = extractvalue %struct.__neon_int16x4x2_t %tmp2, 1
Bob Wilsonc0110052009-09-01 04:27:10 +000026 %tmp5 = add <4 x i16> %tmp3, %tmp4
27 ret <4 x i16> %tmp5
28}
29
30define <2 x i32> @vld2lanei32(i32* %A, <2 x i32>* %B) nounwind {
31;CHECK: vld2lanei32:
32;CHECK: vld2.32
33 %tmp1 = load <2 x i32>* %B
Bob Wilsonec1d81c2009-10-06 21:16:19 +000034 %tmp2 = call %struct.__neon_int32x2x2_t @llvm.arm.neon.vld2lane.v2i32(i32* %A, <2 x i32> %tmp1, <2 x i32> %tmp1, i32 1)
35 %tmp3 = extractvalue %struct.__neon_int32x2x2_t %tmp2, 0
36 %tmp4 = extractvalue %struct.__neon_int32x2x2_t %tmp2, 1
Bob Wilsonc0110052009-09-01 04:27:10 +000037 %tmp5 = add <2 x i32> %tmp3, %tmp4
38 ret <2 x i32> %tmp5
39}
40
41define <2 x float> @vld2lanef(float* %A, <2 x float>* %B) nounwind {
42;CHECK: vld2lanef:
43;CHECK: vld2.32
44 %tmp1 = load <2 x float>* %B
Bob Wilsonec1d81c2009-10-06 21:16:19 +000045 %tmp2 = call %struct.__neon_float32x2x2_t @llvm.arm.neon.vld2lane.v2f32(float* %A, <2 x float> %tmp1, <2 x float> %tmp1, i32 1)
46 %tmp3 = extractvalue %struct.__neon_float32x2x2_t %tmp2, 0
47 %tmp4 = extractvalue %struct.__neon_float32x2x2_t %tmp2, 1
Bob Wilsonc0110052009-09-01 04:27:10 +000048 %tmp5 = add <2 x float> %tmp3, %tmp4
49 ret <2 x float> %tmp5
50}
51
Bob Wilsonec1d81c2009-10-06 21:16:19 +000052declare %struct.__neon_int8x8x2_t @llvm.arm.neon.vld2lane.v8i8(i8*, <8 x i8>, <8 x i8>, i32) nounwind readonly
53declare %struct.__neon_int16x4x2_t @llvm.arm.neon.vld2lane.v4i16(i8*, <4 x i16>, <4 x i16>, i32) nounwind readonly
54declare %struct.__neon_int32x2x2_t @llvm.arm.neon.vld2lane.v2i32(i8*, <2 x i32>, <2 x i32>, i32) nounwind readonly
55declare %struct.__neon_float32x2x2_t @llvm.arm.neon.vld2lane.v2f32(i8*, <2 x float>, <2 x float>, i32) nounwind readonly
Bob Wilsonc0110052009-09-01 04:27:10 +000056
Bob Wilsonec1d81c2009-10-06 21:16:19 +000057%struct.__neon_int8x8x3_t = type { <8 x i8>, <8 x i8>, <8 x i8> }
58%struct.__neon_int16x4x3_t = type { <4 x i16>, <4 x i16>, <4 x i16> }
59%struct.__neon_int32x2x3_t = type { <2 x i32>, <2 x i32>, <2 x i32> }
60%struct.__neon_float32x2x3_t = type { <2 x float>, <2 x float>, <2 x float> }
Bob Wilsonc0110052009-09-01 04:27:10 +000061
62define <8 x i8> @vld3lanei8(i8* %A, <8 x i8>* %B) nounwind {
63;CHECK: vld3lanei8:
64;CHECK: vld3.8
65 %tmp1 = load <8 x i8>* %B
Bob Wilsonec1d81c2009-10-06 21:16:19 +000066 %tmp2 = call %struct.__neon_int8x8x3_t @llvm.arm.neon.vld3lane.v8i8(i8* %A, <8 x i8> %tmp1, <8 x i8> %tmp1, <8 x i8> %tmp1, i32 1)
67 %tmp3 = extractvalue %struct.__neon_int8x8x3_t %tmp2, 0
68 %tmp4 = extractvalue %struct.__neon_int8x8x3_t %tmp2, 1
69 %tmp5 = extractvalue %struct.__neon_int8x8x3_t %tmp2, 2
Bob Wilsonc0110052009-09-01 04:27:10 +000070 %tmp6 = add <8 x i8> %tmp3, %tmp4
71 %tmp7 = add <8 x i8> %tmp5, %tmp6
72 ret <8 x i8> %tmp7
73}
74
75define <4 x i16> @vld3lanei16(i16* %A, <4 x i16>* %B) nounwind {
76;CHECK: vld3lanei16:
77;CHECK: vld3.16
78 %tmp1 = load <4 x i16>* %B
Bob Wilsonec1d81c2009-10-06 21:16:19 +000079 %tmp2 = call %struct.__neon_int16x4x3_t @llvm.arm.neon.vld3lane.v4i16(i16* %A, <4 x i16> %tmp1, <4 x i16> %tmp1, <4 x i16> %tmp1, i32 1)
80 %tmp3 = extractvalue %struct.__neon_int16x4x3_t %tmp2, 0
81 %tmp4 = extractvalue %struct.__neon_int16x4x3_t %tmp2, 1
82 %tmp5 = extractvalue %struct.__neon_int16x4x3_t %tmp2, 2
Bob Wilsonc0110052009-09-01 04:27:10 +000083 %tmp6 = add <4 x i16> %tmp3, %tmp4
84 %tmp7 = add <4 x i16> %tmp5, %tmp6
85 ret <4 x i16> %tmp7
86}
87
88define <2 x i32> @vld3lanei32(i32* %A, <2 x i32>* %B) nounwind {
89;CHECK: vld3lanei32:
90;CHECK: vld3.32
91 %tmp1 = load <2 x i32>* %B
Bob Wilsonec1d81c2009-10-06 21:16:19 +000092 %tmp2 = call %struct.__neon_int32x2x3_t @llvm.arm.neon.vld3lane.v2i32(i32* %A, <2 x i32> %tmp1, <2 x i32> %tmp1, <2 x i32> %tmp1, i32 1)
93 %tmp3 = extractvalue %struct.__neon_int32x2x3_t %tmp2, 0
94 %tmp4 = extractvalue %struct.__neon_int32x2x3_t %tmp2, 1
95 %tmp5 = extractvalue %struct.__neon_int32x2x3_t %tmp2, 2
Bob Wilsonc0110052009-09-01 04:27:10 +000096 %tmp6 = add <2 x i32> %tmp3, %tmp4
97 %tmp7 = add <2 x i32> %tmp5, %tmp6
98 ret <2 x i32> %tmp7
99}
100
101define <2 x float> @vld3lanef(float* %A, <2 x float>* %B) nounwind {
102;CHECK: vld3lanef:
103;CHECK: vld3.32
104 %tmp1 = load <2 x float>* %B
Bob Wilsonec1d81c2009-10-06 21:16:19 +0000105 %tmp2 = call %struct.__neon_float32x2x3_t @llvm.arm.neon.vld3lane.v2f32(float* %A, <2 x float> %tmp1, <2 x float> %tmp1, <2 x float> %tmp1, i32 1)
106 %tmp3 = extractvalue %struct.__neon_float32x2x3_t %tmp2, 0
107 %tmp4 = extractvalue %struct.__neon_float32x2x3_t %tmp2, 1
108 %tmp5 = extractvalue %struct.__neon_float32x2x3_t %tmp2, 2
Bob Wilsonc0110052009-09-01 04:27:10 +0000109 %tmp6 = add <2 x float> %tmp3, %tmp4
110 %tmp7 = add <2 x float> %tmp5, %tmp6
111 ret <2 x float> %tmp7
112}
113
Bob Wilsonec1d81c2009-10-06 21:16:19 +0000114declare %struct.__neon_int8x8x3_t @llvm.arm.neon.vld3lane.v8i8(i8*, <8 x i8>, <8 x i8>, <8 x i8>, i32) nounwind readonly
115declare %struct.__neon_int16x4x3_t @llvm.arm.neon.vld3lane.v4i16(i8*, <4 x i16>, <4 x i16>, <4 x i16>, i32) nounwind readonly
116declare %struct.__neon_int32x2x3_t @llvm.arm.neon.vld3lane.v2i32(i8*, <2 x i32>, <2 x i32>, <2 x i32>, i32) nounwind readonly
117declare %struct.__neon_float32x2x3_t @llvm.arm.neon.vld3lane.v2f32(i8*, <2 x float>, <2 x float>, <2 x float>, i32) nounwind readonly
Bob Wilsonc0110052009-09-01 04:27:10 +0000118
Bob Wilsonec1d81c2009-10-06 21:16:19 +0000119%struct.__neon_int8x8x4_t = type { <8 x i8>, <8 x i8>, <8 x i8>, <8 x i8> }
120%struct.__neon_int16x4x4_t = type { <4 x i16>, <4 x i16>, <4 x i16>, <4 x i16> }
121%struct.__neon_int32x2x4_t = type { <2 x i32>, <2 x i32>, <2 x i32>, <2 x i32> }
122%struct.__neon_float32x2x4_t = type { <2 x float>, <2 x float>, <2 x float>, <2 x float> }
Bob Wilsonc0110052009-09-01 04:27:10 +0000123
124define <8 x i8> @vld4lanei8(i8* %A, <8 x i8>* %B) nounwind {
125;CHECK: vld4lanei8:
126;CHECK: vld4.8
127 %tmp1 = load <8 x i8>* %B
Bob Wilsonec1d81c2009-10-06 21:16:19 +0000128 %tmp2 = call %struct.__neon_int8x8x4_t @llvm.arm.neon.vld4lane.v8i8(i8* %A, <8 x i8> %tmp1, <8 x i8> %tmp1, <8 x i8> %tmp1, <8 x i8> %tmp1, i32 1)
129 %tmp3 = extractvalue %struct.__neon_int8x8x4_t %tmp2, 0
130 %tmp4 = extractvalue %struct.__neon_int8x8x4_t %tmp2, 1
131 %tmp5 = extractvalue %struct.__neon_int8x8x4_t %tmp2, 2
132 %tmp6 = extractvalue %struct.__neon_int8x8x4_t %tmp2, 3
Bob Wilsonc0110052009-09-01 04:27:10 +0000133 %tmp7 = add <8 x i8> %tmp3, %tmp4
134 %tmp8 = add <8 x i8> %tmp5, %tmp6
135 %tmp9 = add <8 x i8> %tmp7, %tmp8
136 ret <8 x i8> %tmp9
137}
138
139define <4 x i16> @vld4lanei16(i16* %A, <4 x i16>* %B) nounwind {
140;CHECK: vld4lanei16:
141;CHECK: vld4.16
142 %tmp1 = load <4 x i16>* %B
Bob Wilsonec1d81c2009-10-06 21:16:19 +0000143 %tmp2 = call %struct.__neon_int16x4x4_t @llvm.arm.neon.vld4lane.v4i16(i16* %A, <4 x i16> %tmp1, <4 x i16> %tmp1, <4 x i16> %tmp1, <4 x i16> %tmp1, i32 1)
144 %tmp3 = extractvalue %struct.__neon_int16x4x4_t %tmp2, 0
145 %tmp4 = extractvalue %struct.__neon_int16x4x4_t %tmp2, 1
146 %tmp5 = extractvalue %struct.__neon_int16x4x4_t %tmp2, 2
147 %tmp6 = extractvalue %struct.__neon_int16x4x4_t %tmp2, 3
Bob Wilsonc0110052009-09-01 04:27:10 +0000148 %tmp7 = add <4 x i16> %tmp3, %tmp4
149 %tmp8 = add <4 x i16> %tmp5, %tmp6
150 %tmp9 = add <4 x i16> %tmp7, %tmp8
151 ret <4 x i16> %tmp9
152}
153
154define <2 x i32> @vld4lanei32(i32* %A, <2 x i32>* %B) nounwind {
155;CHECK: vld4lanei32:
156;CHECK: vld4.32
157 %tmp1 = load <2 x i32>* %B
Bob Wilsonec1d81c2009-10-06 21:16:19 +0000158 %tmp2 = call %struct.__neon_int32x2x4_t @llvm.arm.neon.vld4lane.v2i32(i32* %A, <2 x i32> %tmp1, <2 x i32> %tmp1, <2 x i32> %tmp1, <2 x i32> %tmp1, i32 1)
159 %tmp3 = extractvalue %struct.__neon_int32x2x4_t %tmp2, 0
160 %tmp4 = extractvalue %struct.__neon_int32x2x4_t %tmp2, 1
161 %tmp5 = extractvalue %struct.__neon_int32x2x4_t %tmp2, 2
162 %tmp6 = extractvalue %struct.__neon_int32x2x4_t %tmp2, 3
Bob Wilsonc0110052009-09-01 04:27:10 +0000163 %tmp7 = add <2 x i32> %tmp3, %tmp4
164 %tmp8 = add <2 x i32> %tmp5, %tmp6
165 %tmp9 = add <2 x i32> %tmp7, %tmp8
166 ret <2 x i32> %tmp9
167}
168
169define <2 x float> @vld4lanef(float* %A, <2 x float>* %B) nounwind {
170;CHECK: vld4lanef:
171;CHECK: vld4.32
172 %tmp1 = load <2 x float>* %B
Bob Wilsonec1d81c2009-10-06 21:16:19 +0000173 %tmp2 = call %struct.__neon_float32x2x4_t @llvm.arm.neon.vld4lane.v2f32(float* %A, <2 x float> %tmp1, <2 x float> %tmp1, <2 x float> %tmp1, <2 x float> %tmp1, i32 1)
174 %tmp3 = extractvalue %struct.__neon_float32x2x4_t %tmp2, 0
175 %tmp4 = extractvalue %struct.__neon_float32x2x4_t %tmp2, 1
176 %tmp5 = extractvalue %struct.__neon_float32x2x4_t %tmp2, 2
177 %tmp6 = extractvalue %struct.__neon_float32x2x4_t %tmp2, 3
Bob Wilsonc0110052009-09-01 04:27:10 +0000178 %tmp7 = add <2 x float> %tmp3, %tmp4
179 %tmp8 = add <2 x float> %tmp5, %tmp6
180 %tmp9 = add <2 x float> %tmp7, %tmp8
181 ret <2 x float> %tmp9
182}
183
Bob Wilsonec1d81c2009-10-06 21:16:19 +0000184declare %struct.__neon_int8x8x4_t @llvm.arm.neon.vld4lane.v8i8(i8*, <8 x i8>, <8 x i8>, <8 x i8>, <8 x i8>, i32) nounwind readonly
185declare %struct.__neon_int16x4x4_t @llvm.arm.neon.vld4lane.v4i16(i8*, <4 x i16>, <4 x i16>, <4 x i16>, <4 x i16>, i32) nounwind readonly
186declare %struct.__neon_int32x2x4_t @llvm.arm.neon.vld4lane.v2i32(i8*, <2 x i32>, <2 x i32>, <2 x i32>, <2 x i32>, i32) nounwind readonly
187declare %struct.__neon_float32x2x4_t @llvm.arm.neon.vld4lane.v2f32(i8*, <2 x float>, <2 x float>, <2 x float>, <2 x float>, i32) nounwind readonly