blob: ef6d4e6d22d4bbfd5e233ff68ce753a6244571b9 [file] [log] [blame]
Simon Pilgrimc1794352016-04-30 20:41:52 +00001; NOTE: Assertions have been autogenerated by utils/update_test_checks.py
2; RUN: opt < %s -instcombine -S | FileCheck %s
3target datalayout = "e-m:e-i64:64-f80:128-n8:16:32:64-S128"
4
5; FIXME: Verify that instcombine is able to fold identity shuffles.
6
7define <8 x i32> @identity_test_vpermd(<8 x i32> %a0) {
8; CHECK-LABEL: @identity_test_vpermd(
9; CHECK-NEXT: [[A:%.*]] = tail call <8 x i32> @llvm.x86.avx2.permd(<8 x i32> [[A:%.*]]0, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>)
10; CHECK-NEXT: ret <8 x i32> [[A]]
11;
12 %a = tail call <8 x i32> @llvm.x86.avx2.permd(<8 x i32> %a0, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>)
13 ret <8 x i32> %a
14}
15
16define <8 x float> @identity_test_vpermps(<8 x float> %a0) {
17; CHECK-LABEL: @identity_test_vpermps(
18; CHECK-NEXT: [[A:%.*]] = tail call <8 x float> @llvm.x86.avx2.permps(<8 x float> [[A:%.*]]0, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>)
19; CHECK-NEXT: ret <8 x float> [[A]]
20;
21 %a = tail call <8 x float> @llvm.x86.avx2.permps(<8 x float> %a0, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>)
22 ret <8 x float> %a
23}
24
25; FIXME: Instcombine should be able to fold the following shuffle to a builtin shufflevector
26; with a shuffle mask of all zeroes.
27
28define <8 x i32> @zero_test_vpermd(<8 x i32> %a0) {
29; CHECK-LABEL: @zero_test_vpermd(
30; CHECK-NEXT: [[A:%.*]] = tail call <8 x i32> @llvm.x86.avx2.permd(<8 x i32> [[A:%.*]]0, <8 x i32> zeroinitializer)
31; CHECK-NEXT: ret <8 x i32> [[A]]
32;
33 %a = tail call <8 x i32> @llvm.x86.avx2.permd(<8 x i32> %a0, <8 x i32> zeroinitializer)
34 ret <8 x i32> %a
35}
36
37define <8 x float> @zero_test_vpermps(<8 x float> %a0) {
38; CHECK-LABEL: @zero_test_vpermps(
39; CHECK-NEXT: [[A:%.*]] = tail call <8 x float> @llvm.x86.avx2.permps(<8 x float> [[A:%.*]]0, <8 x i32> zeroinitializer)
40; CHECK-NEXT: ret <8 x float> [[A]]
41;
42 %a = tail call <8 x float> @llvm.x86.avx2.permps(<8 x float> %a0, <8 x i32> zeroinitializer)
43 ret <8 x float> %a
44}
45
46; FIXME: Verify that instcombine is able to fold constant shuffles.
47
48define <8 x i32> @shuffle_test_vpermd(<8 x i32> %a0) {
49; CHECK-LABEL: @shuffle_test_vpermd(
50; CHECK-NEXT: [[A:%.*]] = tail call <8 x i32> @llvm.x86.avx2.permd(<8 x i32> [[A:%.*]]0, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>)
51; CHECK-NEXT: ret <8 x i32> [[A]]
52;
53 %a = tail call <8 x i32> @llvm.x86.avx2.permd(<8 x i32> %a0, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>)
54 ret <8 x i32> %a
55}
56
57define <8 x float> @shuffle_test_vpermps(<8 x float> %a0) {
58; CHECK-LABEL: @shuffle_test_vpermps(
59; CHECK-NEXT: [[A:%.*]] = tail call <8 x float> @llvm.x86.avx2.permps(<8 x float> [[A:%.*]]0, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>)
60; CHECK-NEXT: ret <8 x float> [[A]]
61;
62 %a = tail call <8 x float> @llvm.x86.avx2.permps(<8 x float> %a0, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>)
63 ret <8 x float> %a
64}
65
66; FIXME: Verify that instcombine is able to fold constant shuffles with undef mask elements.
67
68define <8 x i32> @undef_test_vpermd(<8 x i32> %a0) {
69; CHECK-LABEL: @undef_test_vpermd(
70; CHECK-NEXT: [[A:%.*]] = tail call <8 x i32> @llvm.x86.avx2.permd(<8 x i32> [[A:%.*]]0, <8 x i32> <i32 undef, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>)
71; CHECK-NEXT: ret <8 x i32> [[A]]
72;
73 %a = tail call <8 x i32> @llvm.x86.avx2.permd(<8 x i32> %a0, <8 x i32> <i32 undef, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>)
74 ret <8 x i32> %a
75}
76
77define <8 x float> @undef_test_vpermps(<8 x float> %a0) {
78; CHECK-LABEL: @undef_test_vpermps(
79; CHECK-NEXT: [[A:%.*]] = tail call <8 x float> @llvm.x86.avx2.permps(<8 x float> [[A:%.*]]0, <8 x i32> <i32 undef, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>)
80; CHECK-NEXT: ret <8 x float> [[A]]
81;
82 %a = tail call <8 x float> @llvm.x86.avx2.permps(<8 x float> %a0, <8 x i32> <i32 undef, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>)
83 ret <8 x float> %a
84}
85
86declare <8 x i32> @llvm.x86.avx2.permd(<8 x i32>, <8 x i32>)
87declare <8 x float> @llvm.x86.avx2.permps(<8 x float>, <8 x i32>)