| Simon Pilgrim | 5e79ea8 | 2015-10-25 11:42:46 +0000 | [diff] [blame] | 1 | ; RUN: llc -O3 -disable-peephole -mtriple=x86_64-unknown-unknown -mattr=+avx,+aes,+pclmul < %s | FileCheck %s |
| Simon Pilgrim | 0177fa3 | 2015-01-20 23:54:17 +0000 | [diff] [blame] | 2 | |
| 3 | target datalayout = "e-m:e-i64:64-f80:128-n8:16:32:64-S128" |
| 4 | target triple = "x86_64-unknown-unknown" |
| 5 | |
| 6 | ; Stack reload folding tests. |
| 7 | ; |
| 8 | ; By including a nop call with sideeffects we can force a partial register spill of the |
| 9 | ; relevant registers and check that the reload is correctly folded into the instruction. |
| 10 | |
| 11 | define <2 x i64> @stack_fold_aesdec(<2 x i64> %a0, <2 x i64> %a1) { |
| 12 | ;CHECK-LABEL: stack_fold_aesdec |
| 13 | ;CHECK: vaesdec {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 14 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 15 | %2 = call <2 x i64> @llvm.x86.aesni.aesdec(<2 x i64> %a0, <2 x i64> %a1) |
| 16 | ret <2 x i64> %2 |
| 17 | } |
| 18 | declare <2 x i64> @llvm.x86.aesni.aesdec(<2 x i64>, <2 x i64>) nounwind readnone |
| 19 | |
| 20 | define <2 x i64> @stack_fold_aesdeclast(<2 x i64> %a0, <2 x i64> %a1) { |
| 21 | ;CHECK-LABEL: stack_fold_aesdeclast |
| 22 | ;CHECK: vaesdeclast {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 23 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 24 | %2 = call <2 x i64> @llvm.x86.aesni.aesdeclast(<2 x i64> %a0, <2 x i64> %a1) |
| 25 | ret <2 x i64> %2 |
| 26 | } |
| 27 | declare <2 x i64> @llvm.x86.aesni.aesdeclast(<2 x i64>, <2 x i64>) nounwind readnone |
| 28 | |
| 29 | define <2 x i64> @stack_fold_aesenc(<2 x i64> %a0, <2 x i64> %a1) { |
| 30 | ;CHECK-LABEL: stack_fold_aesenc |
| 31 | ;CHECK: vaesenc {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 32 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 33 | %2 = call <2 x i64> @llvm.x86.aesni.aesenc(<2 x i64> %a0, <2 x i64> %a1) |
| 34 | ret <2 x i64> %2 |
| 35 | } |
| 36 | declare <2 x i64> @llvm.x86.aesni.aesenc(<2 x i64>, <2 x i64>) nounwind readnone |
| 37 | |
| 38 | define <2 x i64> @stack_fold_aesenclast(<2 x i64> %a0, <2 x i64> %a1) { |
| 39 | ;CHECK-LABEL: stack_fold_aesenclast |
| 40 | ;CHECK: vaesenclast {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 41 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 42 | %2 = call <2 x i64> @llvm.x86.aesni.aesenclast(<2 x i64> %a0, <2 x i64> %a1) |
| 43 | ret <2 x i64> %2 |
| 44 | } |
| 45 | declare <2 x i64> @llvm.x86.aesni.aesenclast(<2 x i64>, <2 x i64>) nounwind readnone |
| 46 | |
| 47 | define <2 x i64> @stack_fold_aesimc(<2 x i64> %a0) { |
| 48 | ;CHECK-LABEL: stack_fold_aesimc |
| 49 | ;CHECK: vaesimc {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 50 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm1},~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 51 | %2 = call <2 x i64> @llvm.x86.aesni.aesimc(<2 x i64> %a0) |
| 52 | ret <2 x i64> %2 |
| 53 | } |
| 54 | declare <2 x i64> @llvm.x86.aesni.aesimc(<2 x i64>) nounwind readnone |
| 55 | |
| 56 | define <2 x i64> @stack_fold_aeskeygenassist(<2 x i64> %a0) { |
| 57 | ;CHECK-LABEL: stack_fold_aeskeygenassist |
| 58 | ;CHECK: vaeskeygenassist $7, {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 59 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm1},~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 60 | %2 = call <2 x i64> @llvm.x86.aesni.aeskeygenassist(<2 x i64> %a0, i8 7) |
| 61 | ret <2 x i64> %2 |
| 62 | } |
| 63 | declare <2 x i64> @llvm.x86.aesni.aeskeygenassist(<2 x i64>, i8) nounwind readnone |
| 64 | |
| Simon Pilgrim | 5fa0fb2 | 2015-01-21 23:43:30 +0000 | [diff] [blame] | 65 | define <4 x i32> @stack_fold_movd_load(i32 %a0) { |
| 66 | ;CHECK-LABEL: stack_fold_movd_load |
| 67 | ;CHECK: movd {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}} {{.*#+}} 4-byte Folded Reload |
| 68 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{rax},~{rbx},~{rcx},~{rdx},~{rsi},~{rdi},~{rbp},~{r8},~{r9},~{r10},~{r11},~{r12},~{r13},~{r14},~{r15}"() |
| 69 | %2 = insertelement <4 x i32> zeroinitializer, i32 %a0, i32 0 |
| Simon Pilgrim | 6eb925a | 2015-02-14 14:10:44 +0000 | [diff] [blame] | 70 | ; add forces execution domain |
| 71 | %3 = add <4 x i32> %2, <i32 1, i32 1, i32 1, i32 1> |
| 72 | ret <4 x i32> %3 |
| Simon Pilgrim | 5fa0fb2 | 2015-01-21 23:43:30 +0000 | [diff] [blame] | 73 | } |
| 74 | |
| 75 | define i32 @stack_fold_movd_store(<4 x i32> %a0) { |
| 76 | ;CHECK-LABEL: stack_fold_movd_store |
| 77 | ;CHECK: movd {{%xmm[0-9][0-9]*}}, {{-?[0-9]*}}(%rsp) {{.*#+}} 4-byte Folded Spill |
| Simon Pilgrim | 6eb925a | 2015-02-14 14:10:44 +0000 | [diff] [blame] | 78 | ; add forces execution domain |
| 79 | %1 = add <4 x i32> %a0, <i32 1, i32 1, i32 1, i32 1> |
| 80 | %2 = extractelement <4 x i32> %1, i32 0 |
| 81 | %3 = tail call <2 x i64> asm sideeffect "nop", "=x,~{rax},~{rbx},~{rcx},~{rdx},~{rsi},~{rdi},~{rbp},~{r8},~{r9},~{r10},~{r11},~{r12},~{r13},~{r14},~{r15}"() |
| 82 | ret i32 %2 |
| Simon Pilgrim | 5fa0fb2 | 2015-01-21 23:43:30 +0000 | [diff] [blame] | 83 | } |
| 84 | |
| 85 | define <2 x i64> @stack_fold_movq_load(<2 x i64> %a0) { |
| 86 | ;CHECK-LABEL: stack_fold_movq_load |
| 87 | ;CHECK: movq {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 88 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm1},~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 89 | %2 = shufflevector <2 x i64> %a0, <2 x i64> zeroinitializer, <2 x i32> <i32 0, i32 2> |
| Simon Pilgrim | 2711b74 | 2015-03-30 15:25:51 +0000 | [diff] [blame] | 90 | ; add forces execution domain |
| 91 | %3 = add <2 x i64> %2, <i64 1, i64 1> |
| 92 | ret <2 x i64> %3 |
| Simon Pilgrim | 5fa0fb2 | 2015-01-21 23:43:30 +0000 | [diff] [blame] | 93 | } |
| 94 | |
| 95 | define i64 @stack_fold_movq_store(<2 x i64> %a0) { |
| 96 | ;CHECK-LABEL: stack_fold_movq_store |
| 97 | ;CHECK: movq {{%xmm[0-9][0-9]*}}, {{-?[0-9]*}}(%rsp) {{.*#+}} 8-byte Folded Spill |
| Simon Pilgrim | 2711b74 | 2015-03-30 15:25:51 +0000 | [diff] [blame] | 98 | ; add forces execution domain |
| 99 | %1 = add <2 x i64> %a0, <i64 1, i64 1> |
| 100 | %2 = extractelement <2 x i64> %1, i32 0 |
| 101 | %3 = tail call <2 x i64> asm sideeffect "nop", "=x,~{rax},~{rbx},~{rcx},~{rdx},~{rsi},~{rdi},~{rbp},~{r8},~{r9},~{r10},~{r11},~{r12},~{r13},~{r14},~{r15}"() |
| 102 | ret i64 %2 |
| Simon Pilgrim | 5fa0fb2 | 2015-01-21 23:43:30 +0000 | [diff] [blame] | 103 | } |
| 104 | |
| Simon Pilgrim | cbc3b2f | 2015-02-07 21:20:11 +0000 | [diff] [blame] | 105 | define <8 x i16> @stack_fold_mpsadbw(<16 x i8> %a0, <16 x i8> %a1) { |
| 106 | ;CHECK-LABEL: stack_fold_mpsadbw |
| 107 | ;CHECK: vmpsadbw $7, {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 108 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 109 | %2 = call <8 x i16> @llvm.x86.sse41.mpsadbw(<16 x i8> %a0, <16 x i8> %a1, i8 7) |
| 110 | ret <8 x i16> %2 |
| 111 | } |
| 112 | declare <8 x i16> @llvm.x86.sse41.mpsadbw(<16 x i8>, <16 x i8>, i8) nounwind readnone |
| 113 | |
| Simon Pilgrim | 0177fa3 | 2015-01-20 23:54:17 +0000 | [diff] [blame] | 114 | define <16 x i8> @stack_fold_pabsb(<16 x i8> %a0) { |
| 115 | ;CHECK-LABEL: stack_fold_pabsb |
| 116 | ;CHECK: vpabsb {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 117 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm1},~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 118 | %2 = call <16 x i8> @llvm.x86.ssse3.pabs.b.128(<16 x i8> %a0) |
| 119 | ret <16 x i8> %2 |
| 120 | } |
| 121 | declare <16 x i8> @llvm.x86.ssse3.pabs.b.128(<16 x i8>) nounwind readnone |
| 122 | |
| 123 | define <4 x i32> @stack_fold_pabsd(<4 x i32> %a0) { |
| 124 | ;CHECK-LABEL: stack_fold_pabsd |
| 125 | ;CHECK: vpabsd {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 126 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm1},~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 127 | %2 = call <4 x i32> @llvm.x86.ssse3.pabs.d.128(<4 x i32> %a0) |
| 128 | ret <4 x i32> %2 |
| 129 | } |
| 130 | declare <4 x i32> @llvm.x86.ssse3.pabs.d.128(<4 x i32>) nounwind readnone |
| 131 | |
| 132 | define <8 x i16> @stack_fold_pabsw(<8 x i16> %a0) { |
| 133 | ;CHECK-LABEL: stack_fold_pabsw |
| 134 | ;CHECK: vpabsw {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 135 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm1},~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 136 | %2 = call <8 x i16> @llvm.x86.ssse3.pabs.w.128(<8 x i16> %a0) |
| 137 | ret <8 x i16> %2 |
| 138 | } |
| 139 | declare <8 x i16> @llvm.x86.ssse3.pabs.w.128(<8 x i16>) nounwind readnone |
| 140 | |
| 141 | define <8 x i16> @stack_fold_packssdw(<4 x i32> %a0, <4 x i32> %a1) { |
| 142 | ;CHECK-LABEL: stack_fold_packssdw |
| 143 | ;CHECK: vpackssdw {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 144 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 145 | %2 = call <8 x i16> @llvm.x86.sse2.packssdw.128(<4 x i32> %a0, <4 x i32> %a1) |
| 146 | ret <8 x i16> %2 |
| 147 | } |
| 148 | declare <8 x i16> @llvm.x86.sse2.packssdw.128(<4 x i32>, <4 x i32>) nounwind readnone |
| 149 | |
| 150 | define <16 x i8> @stack_fold_packsswb(<8 x i16> %a0, <8 x i16> %a1) { |
| 151 | ;CHECK-LABEL: stack_fold_packsswb |
| 152 | ;CHECK: vpacksswb {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 153 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 154 | %2 = call <16 x i8> @llvm.x86.sse2.packsswb.128(<8 x i16> %a0, <8 x i16> %a1) |
| 155 | ret <16 x i8> %2 |
| 156 | } |
| 157 | declare <16 x i8> @llvm.x86.sse2.packsswb.128(<8 x i16>, <8 x i16>) nounwind readnone |
| 158 | |
| 159 | define <8 x i16> @stack_fold_packusdw(<4 x i32> %a0, <4 x i32> %a1) { |
| 160 | ;CHECK-LABEL: stack_fold_packusdw |
| 161 | ;CHECK: vpackusdw {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 162 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 163 | %2 = call <8 x i16> @llvm.x86.sse41.packusdw(<4 x i32> %a0, <4 x i32> %a1) |
| 164 | ret <8 x i16> %2 |
| 165 | } |
| 166 | declare <8 x i16> @llvm.x86.sse41.packusdw(<4 x i32>, <4 x i32>) nounwind readnone |
| 167 | |
| 168 | define <16 x i8> @stack_fold_packuswb(<8 x i16> %a0, <8 x i16> %a1) { |
| 169 | ;CHECK-LABEL: stack_fold_packuswb |
| 170 | ;CHECK: vpackuswb {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 171 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 172 | %2 = call <16 x i8> @llvm.x86.sse2.packuswb.128(<8 x i16> %a0, <8 x i16> %a1) |
| 173 | ret <16 x i8> %2 |
| 174 | } |
| 175 | declare <16 x i8> @llvm.x86.sse2.packuswb.128(<8 x i16>, <8 x i16>) nounwind readnone |
| 176 | |
| 177 | define <16 x i8> @stack_fold_paddb(<16 x i8> %a0, <16 x i8> %a1) { |
| 178 | ;CHECK-LABEL: stack_fold_paddb |
| 179 | ;CHECK: vpaddb {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 180 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 181 | %2 = add <16 x i8> %a0, %a1 |
| 182 | ret <16 x i8> %2 |
| 183 | } |
| 184 | |
| 185 | define <4 x i32> @stack_fold_paddd(<4 x i32> %a0, <4 x i32> %a1) { |
| 186 | ;CHECK-LABEL: stack_fold_paddd |
| 187 | ;CHECK: vpaddd {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 188 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 189 | %2 = add <4 x i32> %a0, %a1 |
| 190 | ret <4 x i32> %2 |
| 191 | } |
| 192 | |
| 193 | define <2 x i64> @stack_fold_paddq(<2 x i64> %a0, <2 x i64> %a1) { |
| 194 | ;CHECK-LABEL: stack_fold_paddq |
| 195 | ;CHECK: vpaddq {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 196 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 197 | %2 = add <2 x i64> %a0, %a1 |
| 198 | ret <2 x i64> %2 |
| 199 | } |
| 200 | |
| 201 | define <16 x i8> @stack_fold_paddsb(<16 x i8> %a0, <16 x i8> %a1) { |
| 202 | ;CHECK-LABEL: stack_fold_paddsb |
| 203 | ;CHECK: vpaddsb {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 204 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 205 | %2 = call <16 x i8> @llvm.x86.sse2.padds.b(<16 x i8> %a0, <16 x i8> %a1) |
| 206 | ret <16 x i8> %2 |
| 207 | } |
| 208 | declare <16 x i8> @llvm.x86.sse2.padds.b(<16 x i8>, <16 x i8>) nounwind readnone |
| 209 | |
| 210 | define <8 x i16> @stack_fold_paddsw(<8 x i16> %a0, <8 x i16> %a1) { |
| 211 | ;CHECK-LABEL: stack_fold_paddsw |
| 212 | ;CHECK: vpaddsw {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 213 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 214 | %2 = call <8 x i16> @llvm.x86.sse2.padds.w(<8 x i16> %a0, <8 x i16> %a1) |
| 215 | ret <8 x i16> %2 |
| 216 | } |
| 217 | declare <8 x i16> @llvm.x86.sse2.padds.w(<8 x i16>, <8 x i16>) nounwind readnone |
| 218 | |
| 219 | define <16 x i8> @stack_fold_paddusb(<16 x i8> %a0, <16 x i8> %a1) { |
| 220 | ;CHECK-LABEL: stack_fold_paddusb |
| 221 | ;CHECK: vpaddusb {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 222 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 223 | %2 = call <16 x i8> @llvm.x86.sse2.paddus.b(<16 x i8> %a0, <16 x i8> %a1) |
| 224 | ret <16 x i8> %2 |
| 225 | } |
| 226 | declare <16 x i8> @llvm.x86.sse2.paddus.b(<16 x i8>, <16 x i8>) nounwind readnone |
| 227 | |
| 228 | define <8 x i16> @stack_fold_paddusw(<8 x i16> %a0, <8 x i16> %a1) { |
| 229 | ;CHECK-LABEL: stack_fold_paddusw |
| 230 | ;CHECK: vpaddusw {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 231 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 232 | %2 = call <8 x i16> @llvm.x86.sse2.paddus.w(<8 x i16> %a0, <8 x i16> %a1) |
| 233 | ret <8 x i16> %2 |
| 234 | } |
| 235 | declare <8 x i16> @llvm.x86.sse2.paddus.w(<8 x i16>, <8 x i16>) nounwind readnone |
| 236 | |
| 237 | define <8 x i16> @stack_fold_paddw(<8 x i16> %a0, <8 x i16> %a1) { |
| 238 | ;CHECK-LABEL: stack_fold_paddw |
| 239 | ;CHECK: vpaddw {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 240 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 241 | %2 = add <8 x i16> %a0, %a1 |
| 242 | ret <8 x i16> %2 |
| 243 | } |
| 244 | |
| 245 | define <16 x i8> @stack_fold_palignr(<16 x i8> %a0, <16 x i8> %a1) { |
| 246 | ;CHECK-LABEL: stack_fold_palignr |
| 247 | ;CHECK: vpalignr $1, {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 248 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 249 | %2 = shufflevector <16 x i8> %a1, <16 x i8> %a0, <16 x i32> <i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15, i32 16> |
| 250 | ret <16 x i8> %2 |
| 251 | } |
| 252 | |
| 253 | define <16 x i8> @stack_fold_pand(<16 x i8> %a0, <16 x i8> %a1) { |
| 254 | ;CHECK-LABEL: stack_fold_pand |
| 255 | ;CHECK: vpand {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 256 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 257 | %2 = and <16 x i8> %a0, %a1 |
| 258 | ; add forces execution domain |
| 259 | %3 = add <16 x i8> %2, <i8 1, i8 1, i8 1, i8 1, i8 1, i8 1, i8 1, i8 1, i8 1, i8 1, i8 1, i8 1, i8 1, i8 1, i8 1, i8 1> |
| 260 | ret <16 x i8> %3 |
| 261 | } |
| 262 | |
| 263 | define <16 x i8> @stack_fold_pandn(<16 x i8> %a0, <16 x i8> %a1) { |
| 264 | ;CHECK-LABEL: stack_fold_pandn |
| 265 | ;CHECK: vpandn {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 266 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 267 | %2 = xor <16 x i8> %a0, <i8 -1, i8 -1, i8 -1, i8 -1, i8 -1, i8 -1, i8 -1, i8 -1, i8 -1, i8 -1, i8 -1, i8 -1, i8 -1, i8 -1, i8 -1, i8 -1> |
| 268 | %3 = and <16 x i8> %2, %a1 |
| 269 | ; add forces execution domain |
| 270 | %4 = add <16 x i8> %3, <i8 1, i8 1, i8 1, i8 1, i8 1, i8 1, i8 1, i8 1, i8 1, i8 1, i8 1, i8 1, i8 1, i8 1, i8 1, i8 1> |
| 271 | ret <16 x i8> %4 |
| 272 | } |
| 273 | |
| 274 | define <16 x i8> @stack_fold_pavgb(<16 x i8> %a0, <16 x i8> %a1) { |
| 275 | ;CHECK-LABEL: stack_fold_pavgb |
| 276 | ;CHECK: vpavgb {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 277 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 278 | %2 = call <16 x i8> @llvm.x86.sse2.pavg.b(<16 x i8> %a0, <16 x i8> %a1) |
| 279 | ret <16 x i8> %2 |
| 280 | } |
| 281 | declare <16 x i8> @llvm.x86.sse2.pavg.b(<16 x i8>, <16 x i8>) nounwind readnone |
| 282 | |
| 283 | define <8 x i16> @stack_fold_pavgw(<8 x i16> %a0, <8 x i16> %a1) { |
| 284 | ;CHECK-LABEL: stack_fold_pavgw |
| 285 | ;CHECK: vpavgw {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 286 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 287 | %2 = call <8 x i16> @llvm.x86.sse2.pavg.w(<8 x i16> %a0, <8 x i16> %a1) |
| 288 | ret <8 x i16> %2 |
| 289 | } |
| 290 | declare <8 x i16> @llvm.x86.sse2.pavg.w(<8 x i16>, <8 x i16>) nounwind readnone |
| 291 | |
| Simon Pilgrim | 5fa0fb2 | 2015-01-21 23:43:30 +0000 | [diff] [blame] | 292 | define <16 x i8> @stack_fold_pblendvb(<16 x i8> %a0, <16 x i8> %a1, <16 x i8> %c) { |
| 293 | ;CHECK-LABEL: stack_fold_pblendvb |
| 294 | ;CHECK: vpblendvb {{%xmm[0-9][0-9]*}}, {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 295 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 296 | %2 = call <16 x i8> @llvm.x86.sse41.pblendvb(<16 x i8> %a1, <16 x i8> %c, <16 x i8> %a0) |
| 297 | ret <16 x i8> %2 |
| 298 | } |
| Simon Pilgrim | 0177fa3 | 2015-01-20 23:54:17 +0000 | [diff] [blame] | 299 | declare <16 x i8> @llvm.x86.sse41.pblendvb(<16 x i8>, <16 x i8>, <16 x i8>) nounwind readnone |
| 300 | |
| 301 | define <8 x i16> @stack_fold_pblendw(<8 x i16> %a0, <8 x i16> %a1) { |
| 302 | ;CHECK-LABEL: stack_fold_pblendw |
| 303 | ;CHECK: vpblendw $7, {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 304 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 305 | %2 = call <8 x i16> @llvm.x86.sse41.pblendw(<8 x i16> %a0, <8 x i16> %a1, i8 7) |
| 306 | ret <8 x i16> %2 |
| 307 | } |
| 308 | declare <8 x i16> @llvm.x86.sse41.pblendw(<8 x i16>, <8 x i16>, i8) nounwind readnone |
| 309 | |
| Simon Pilgrim | 5fa0fb2 | 2015-01-21 23:43:30 +0000 | [diff] [blame] | 310 | define <2 x i64> @stack_fold_pclmulqdq(<2 x i64> %a0, <2 x i64> %a1) { |
| 311 | ;CHECK-LABEL: stack_fold_pclmulqdq |
| 312 | ;CHECK: vpclmulqdq $0, {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 313 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 314 | %2 = call <2 x i64> @llvm.x86.pclmulqdq(<2 x i64> %a0, <2 x i64> %a1, i8 0) |
| 315 | ret <2 x i64> %2 |
| 316 | } |
| Simon Pilgrim | 0177fa3 | 2015-01-20 23:54:17 +0000 | [diff] [blame] | 317 | declare <2 x i64> @llvm.x86.pclmulqdq(<2 x i64>, <2 x i64>, i8) nounwind readnone |
| 318 | |
| 319 | define <16 x i8> @stack_fold_pcmpeqb(<16 x i8> %a0, <16 x i8> %a1) { |
| 320 | ;CHECK-LABEL: stack_fold_pcmpeqb |
| 321 | ;CHECK: vpcmpeqb {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 322 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 323 | %2 = icmp eq <16 x i8> %a0, %a1 |
| 324 | %3 = sext <16 x i1> %2 to <16 x i8> |
| 325 | ret <16 x i8> %3 |
| 326 | } |
| 327 | |
| 328 | define <4 x i32> @stack_fold_pcmpeqd(<4 x i32> %a0, <4 x i32> %a1) { |
| 329 | ;CHECK-LABEL: stack_fold_pcmpeqd |
| 330 | ;CHECK: vpcmpeqd {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 331 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 332 | %2 = icmp eq <4 x i32> %a0, %a1 |
| 333 | %3 = sext <4 x i1> %2 to <4 x i32> |
| 334 | ret <4 x i32> %3 |
| 335 | } |
| 336 | |
| 337 | define <2 x i64> @stack_fold_pcmpeqq(<2 x i64> %a0, <2 x i64> %a1) { |
| 338 | ;CHECK-LABEL: stack_fold_pcmpeqq |
| 339 | ;CHECK: vpcmpeqq {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 340 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 341 | %2 = icmp eq <2 x i64> %a0, %a1 |
| 342 | %3 = sext <2 x i1> %2 to <2 x i64> |
| 343 | ret <2 x i64> %3 |
| 344 | } |
| 345 | |
| 346 | define <8 x i16> @stack_fold_pcmpeqw(<8 x i16> %a0, <8 x i16> %a1) { |
| 347 | ;CHECK-LABEL: stack_fold_pcmpeqw |
| 348 | ;CHECK: vpcmpeqw {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 349 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 350 | %2 = icmp eq <8 x i16> %a0, %a1 |
| 351 | %3 = sext <8 x i1> %2 to <8 x i16> |
| 352 | ret <8 x i16> %3 |
| 353 | } |
| 354 | |
| Simon Pilgrim | 5fa0fb2 | 2015-01-21 23:43:30 +0000 | [diff] [blame] | 355 | define i32 @stack_fold_pcmpestri(<16 x i8> %a0, <16 x i8> %a1) { |
| 356 | ;CHECK-LABEL: stack_fold_pcmpestri |
| 357 | ;CHECK: vpcmpestri $7, {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 358 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{rax},~{flags}"() |
| 359 | %2 = call i32 @llvm.x86.sse42.pcmpestri128(<16 x i8> %a0, i32 7, <16 x i8> %a1, i32 7, i8 7) |
| 360 | ret i32 %2 |
| 361 | } |
| Simon Pilgrim | 0177fa3 | 2015-01-20 23:54:17 +0000 | [diff] [blame] | 362 | declare i32 @llvm.x86.sse42.pcmpestri128(<16 x i8>, i32, <16 x i8>, i32, i8) nounwind readnone |
| 363 | |
| Simon Pilgrim | 5fa0fb2 | 2015-01-21 23:43:30 +0000 | [diff] [blame] | 364 | define <16 x i8> @stack_fold_pcmpestrm(<16 x i8> %a0, <16 x i8> %a1) { |
| 365 | ;CHECK-LABEL: stack_fold_pcmpestrm |
| 366 | ;CHECK: vpcmpestrm $7, {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 367 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{rax},~{flags}"() |
| 368 | %2 = call <16 x i8> @llvm.x86.sse42.pcmpestrm128(<16 x i8> %a0, i32 7, <16 x i8> %a1, i32 7, i8 7) |
| 369 | ret <16 x i8> %2 |
| 370 | } |
| Simon Pilgrim | 0177fa3 | 2015-01-20 23:54:17 +0000 | [diff] [blame] | 371 | declare <16 x i8> @llvm.x86.sse42.pcmpestrm128(<16 x i8>, i32, <16 x i8>, i32, i8) nounwind readnone |
| 372 | |
| 373 | define <16 x i8> @stack_fold_pcmpgtb(<16 x i8> %a0, <16 x i8> %a1) { |
| 374 | ;CHECK-LABEL: stack_fold_pcmpgtb |
| 375 | ;CHECK: vpcmpgtb {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 376 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 377 | %2 = icmp sgt <16 x i8> %a0, %a1 |
| 378 | %3 = sext <16 x i1> %2 to <16 x i8> |
| 379 | ret <16 x i8> %3 |
| 380 | } |
| 381 | |
| 382 | define <4 x i32> @stack_fold_pcmpgtd(<4 x i32> %a0, <4 x i32> %a1) { |
| 383 | ;CHECK-LABEL: stack_fold_pcmpgtd |
| 384 | ;CHECK: vpcmpgtd {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 385 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 386 | %2 = icmp sgt <4 x i32> %a0, %a1 |
| 387 | %3 = sext <4 x i1> %2 to <4 x i32> |
| 388 | ret <4 x i32> %3 |
| 389 | } |
| 390 | |
| 391 | define <2 x i64> @stack_fold_pcmpgtq(<2 x i64> %a0, <2 x i64> %a1) { |
| 392 | ;CHECK-LABEL: stack_fold_pcmpgtq |
| 393 | ;CHECK: vpcmpgtq {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 394 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 395 | %2 = icmp sgt <2 x i64> %a0, %a1 |
| 396 | %3 = sext <2 x i1> %2 to <2 x i64> |
| 397 | ret <2 x i64> %3 |
| 398 | } |
| 399 | |
| 400 | define <8 x i16> @stack_fold_pcmpgtw(<8 x i16> %a0, <8 x i16> %a1) { |
| 401 | ;CHECK-LABEL: stack_fold_pcmpgtw |
| 402 | ;CHECK: vpcmpgtw {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 403 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 404 | %2 = icmp sgt <8 x i16> %a0, %a1 |
| 405 | %3 = sext <8 x i1> %2 to <8 x i16> |
| 406 | ret <8 x i16> %3 |
| 407 | } |
| 408 | |
| Simon Pilgrim | 5fa0fb2 | 2015-01-21 23:43:30 +0000 | [diff] [blame] | 409 | define i32 @stack_fold_pcmpistri(<16 x i8> %a0, <16 x i8> %a1) { |
| 410 | ;CHECK-LABEL: stack_fold_pcmpistri |
| 411 | ;CHECK: vpcmpistri $7, {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 412 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 413 | %2 = call i32 @llvm.x86.sse42.pcmpistri128(<16 x i8> %a0, <16 x i8> %a1, i8 7) |
| 414 | ret i32 %2 |
| 415 | } |
| Simon Pilgrim | 0177fa3 | 2015-01-20 23:54:17 +0000 | [diff] [blame] | 416 | declare i32 @llvm.x86.sse42.pcmpistri128(<16 x i8>, <16 x i8>, i8) nounwind readnone |
| 417 | |
| Simon Pilgrim | 5fa0fb2 | 2015-01-21 23:43:30 +0000 | [diff] [blame] | 418 | define <16 x i8> @stack_fold_pcmpistrm(<16 x i8> %a0, <16 x i8> %a1) { |
| 419 | ;CHECK-LABEL: stack_fold_pcmpistrm |
| 420 | ;CHECK: vpcmpistrm $7, {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 421 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 422 | %2 = call <16 x i8> @llvm.x86.sse42.pcmpistrm128(<16 x i8> %a0, <16 x i8> %a1, i8 7) |
| 423 | ret <16 x i8> %2 |
| 424 | } |
| Simon Pilgrim | 0177fa3 | 2015-01-20 23:54:17 +0000 | [diff] [blame] | 425 | declare <16 x i8> @llvm.x86.sse42.pcmpistrm128(<16 x i8>, <16 x i8>, i8) nounwind readnone |
| 426 | |
| 427 | ; TODO stack_fold_pextrb |
| Simon Pilgrim | 5fa0fb2 | 2015-01-21 23:43:30 +0000 | [diff] [blame] | 428 | |
| 429 | define i32 @stack_fold_pextrd(<4 x i32> %a0) { |
| 430 | ;CHECK-LABEL: stack_fold_pextrd |
| 431 | ;CHECK: pextrd $1, {{%xmm[0-9][0-9]*}}, {{-?[0-9]*}}(%rsp) {{.*#+}} 4-byte Folded Spill |
| 432 | ;CHECK: movl {{-?[0-9]*}}(%rsp), %eax {{.*#+}} 4-byte Reload |
| 433 | %1 = extractelement <4 x i32> %a0, i32 1 |
| 434 | %2 = tail call <2 x i64> asm sideeffect "nop", "=x,~{rax},~{rbx},~{rcx},~{rdx},~{rsi},~{rdi},~{rbp},~{r8},~{r9},~{r10},~{r11},~{r12},~{r13},~{r14},~{r15}"() |
| 435 | ret i32 %1 |
| 436 | } |
| 437 | |
| 438 | define i64 @stack_fold_pextrq(<2 x i64> %a0) { |
| 439 | ;CHECK-LABEL: stack_fold_pextrq |
| 440 | ;CHECK: pextrq $1, {{%xmm[0-9][0-9]*}}, {{-?[0-9]*}}(%rsp) {{.*#+}} 8-byte Folded Spill |
| 441 | ;CHECK: movq {{-?[0-9]*}}(%rsp), %rax {{.*#+}} 8-byte Reload |
| 442 | %1 = extractelement <2 x i64> %a0, i32 1 |
| 443 | %2 = tail call <2 x i64> asm sideeffect "nop", "=x,~{rax},~{rbx},~{rcx},~{rdx},~{rsi},~{rdi},~{rbp},~{r8},~{r9},~{r10},~{r11},~{r12},~{r13},~{r14},~{r15}"() |
| 444 | ret i64 %1 |
| 445 | } |
| 446 | |
| Simon Pilgrim | 0177fa3 | 2015-01-20 23:54:17 +0000 | [diff] [blame] | 447 | ; TODO stack_fold_pextrw |
| 448 | |
| 449 | define <4 x i32> @stack_fold_phaddd(<4 x i32> %a0, <4 x i32> %a1) { |
| 450 | ;CHECK-LABEL: stack_fold_phaddd |
| 451 | ;CHECK: vphaddd {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 452 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 453 | %2 = call <4 x i32> @llvm.x86.ssse3.phadd.d.128(<4 x i32> %a0, <4 x i32> %a1) |
| 454 | ret <4 x i32> %2 |
| 455 | } |
| 456 | declare <4 x i32> @llvm.x86.ssse3.phadd.d.128(<4 x i32>, <4 x i32>) nounwind readnone |
| 457 | |
| 458 | define <8 x i16> @stack_fold_phaddsw(<8 x i16> %a0, <8 x i16> %a1) { |
| 459 | ;CHECK-LABEL: stack_fold_phaddsw |
| 460 | ;CHECK: vphaddsw {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 461 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 462 | %2 = call <8 x i16> @llvm.x86.ssse3.phadd.sw.128(<8 x i16> %a0, <8 x i16> %a1) |
| 463 | ret <8 x i16> %2 |
| 464 | } |
| 465 | declare <8 x i16> @llvm.x86.ssse3.phadd.sw.128(<8 x i16>, <8 x i16>) nounwind readnone |
| 466 | |
| 467 | define <8 x i16> @stack_fold_phaddw(<8 x i16> %a0, <8 x i16> %a1) { |
| 468 | ;CHECK-LABEL: stack_fold_phaddw |
| 469 | ;CHECK: vphaddw {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 470 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 471 | %2 = call <8 x i16> @llvm.x86.ssse3.phadd.w.128(<8 x i16> %a0, <8 x i16> %a1) |
| 472 | ret <8 x i16> %2 |
| 473 | } |
| 474 | declare <8 x i16> @llvm.x86.ssse3.phadd.w.128(<8 x i16>, <8 x i16>) nounwind readnone |
| 475 | |
| Simon Pilgrim | 5fa0fb2 | 2015-01-21 23:43:30 +0000 | [diff] [blame] | 476 | define <8 x i16> @stack_fold_phminposuw(<8 x i16> %a0) { |
| 477 | ;CHECK-LABEL: stack_fold_phminposuw |
| 478 | ;CHECK: vphminposuw {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 479 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm1},~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 480 | %2 = call <8 x i16> @llvm.x86.sse41.phminposuw(<8 x i16> %a0) |
| 481 | ret <8 x i16> %2 |
| 482 | } |
| Simon Pilgrim | 0177fa3 | 2015-01-20 23:54:17 +0000 | [diff] [blame] | 483 | declare <8 x i16> @llvm.x86.sse41.phminposuw(<8 x i16>) nounwind readnone |
| 484 | |
| 485 | define <4 x i32> @stack_fold_phsubd(<4 x i32> %a0, <4 x i32> %a1) { |
| 486 | ;CHECK-LABEL: stack_fold_phsubd |
| 487 | ;CHECK: vphsubd {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 488 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 489 | %2 = call <4 x i32> @llvm.x86.ssse3.phsub.d.128(<4 x i32> %a0, <4 x i32> %a1) |
| 490 | ret <4 x i32> %2 |
| 491 | } |
| 492 | declare <4 x i32> @llvm.x86.ssse3.phsub.d.128(<4 x i32>, <4 x i32>) nounwind readnone |
| 493 | |
| 494 | define <8 x i16> @stack_fold_phsubsw(<8 x i16> %a0, <8 x i16> %a1) { |
| 495 | ;CHECK-LABEL: stack_fold_phsubsw |
| 496 | ;CHECK: vphsubsw {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 497 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 498 | %2 = call <8 x i16> @llvm.x86.ssse3.phsub.sw.128(<8 x i16> %a0, <8 x i16> %a1) |
| 499 | ret <8 x i16> %2 |
| 500 | } |
| 501 | declare <8 x i16> @llvm.x86.ssse3.phsub.sw.128(<8 x i16>, <8 x i16>) nounwind readnone |
| 502 | |
| 503 | define <8 x i16> @stack_fold_phsubw(<8 x i16> %a0, <8 x i16> %a1) { |
| 504 | ;CHECK-LABEL: stack_fold_phsubw |
| 505 | ;CHECK: vphsubw {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 506 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 507 | %2 = call <8 x i16> @llvm.x86.ssse3.phsub.w.128(<8 x i16> %a0, <8 x i16> %a1) |
| 508 | ret <8 x i16> %2 |
| 509 | } |
| 510 | declare <8 x i16> @llvm.x86.ssse3.phsub.w.128(<8 x i16>, <8 x i16>) nounwind readnone |
| 511 | |
| Simon Pilgrim | 5fa0fb2 | 2015-01-21 23:43:30 +0000 | [diff] [blame] | 512 | define <16 x i8> @stack_fold_pinsrb(<16 x i8> %a0, i8 %a1) { |
| 513 | ;CHECK-LABEL: stack_fold_pinsrb |
| 514 | ;CHECK: vpinsrb $1, {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 4-byte Folded Reload |
| 515 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{rax},~{rbx},~{rcx},~{rdx},~{rsi},~{rdi},~{rbp},~{r8},~{r9},~{r10},~{r11},~{r12},~{r13},~{r14},~{r15}"() |
| 516 | %2 = insertelement <16 x i8> %a0, i8 %a1, i32 1 |
| 517 | ret <16 x i8> %2 |
| 518 | } |
| 519 | |
| 520 | define <4 x i32> @stack_fold_pinsrd(<4 x i32> %a0, i32 %a1) { |
| 521 | ;CHECK-LABEL: stack_fold_pinsrd |
| 522 | ;CHECK: vpinsrd $1, {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 4-byte Folded Reload |
| 523 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{rax},~{rbx},~{rcx},~{rdx},~{rsi},~{rdi},~{rbp},~{r8},~{r9},~{r10},~{r11},~{r12},~{r13},~{r14},~{r15}"() |
| 524 | %2 = insertelement <4 x i32> %a0, i32 %a1, i32 1 |
| 525 | ret <4 x i32> %2 |
| 526 | } |
| 527 | |
| 528 | define <2 x i64> @stack_fold_pinsrq(<2 x i64> %a0, i64 %a1) { |
| 529 | ;CHECK-LABEL: stack_fold_pinsrq |
| 530 | ;CHECK: vpinsrq $1, {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 8-byte Folded Reload |
| 531 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{rax},~{rbx},~{rcx},~{rdx},~{rsi},~{rdi},~{rbp},~{r8},~{r9},~{r10},~{r11},~{r12},~{r13},~{r14},~{r15}"() |
| 532 | %2 = insertelement <2 x i64> %a0, i64 %a1, i32 1 |
| 533 | ret <2 x i64> %2 |
| 534 | } |
| 535 | |
| 536 | define <8 x i16> @stack_fold_pinsrw(<8 x i16> %a0, i16 %a1) { |
| 537 | ;CHECK-LABEL: stack_fold_pinsrw |
| 538 | ;CHECK: vpinsrw $1, {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 4-byte Folded Reload |
| 539 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{rax},~{rbx},~{rcx},~{rdx},~{rsi},~{rdi},~{rbp},~{r8},~{r9},~{r10},~{r11},~{r12},~{r13},~{r14},~{r15}"() |
| 540 | %2 = insertelement <8 x i16> %a0, i16 %a1, i32 1 |
| 541 | ret <8 x i16> %2 |
| 542 | } |
| Simon Pilgrim | 0177fa3 | 2015-01-20 23:54:17 +0000 | [diff] [blame] | 543 | |
| 544 | define <8 x i16> @stack_fold_pmaddubsw(<16 x i8> %a0, <16 x i8> %a1) { |
| 545 | ;CHECK-LABEL: stack_fold_pmaddubsw |
| 546 | ;CHECK: vpmaddubsw {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 547 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 548 | %2 = call <8 x i16> @llvm.x86.ssse3.pmadd.ub.sw.128(<16 x i8> %a0, <16 x i8> %a1) |
| 549 | ret <8 x i16> %2 |
| 550 | } |
| 551 | declare <8 x i16> @llvm.x86.ssse3.pmadd.ub.sw.128(<16 x i8>, <16 x i8>) nounwind readnone |
| 552 | |
| 553 | define <4 x i32> @stack_fold_pmaddwd(<8 x i16> %a0, <8 x i16> %a1) { |
| 554 | ;CHECK-LABEL: stack_fold_pmaddwd |
| 555 | ;CHECK: vpmaddwd {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 556 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 557 | %2 = call <4 x i32> @llvm.x86.sse2.pmadd.wd(<8 x i16> %a0, <8 x i16> %a1) |
| 558 | ret <4 x i32> %2 |
| 559 | } |
| 560 | declare <4 x i32> @llvm.x86.sse2.pmadd.wd(<8 x i16>, <8 x i16>) nounwind readnone |
| 561 | |
| 562 | define <16 x i8> @stack_fold_pmaxsb(<16 x i8> %a0, <16 x i8> %a1) { |
| 563 | ;CHECK-LABEL: stack_fold_pmaxsb |
| 564 | ;CHECK: vpmaxsb {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 565 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 566 | %2 = call <16 x i8> @llvm.x86.sse41.pmaxsb(<16 x i8> %a0, <16 x i8> %a1) |
| 567 | ret <16 x i8> %2 |
| 568 | } |
| 569 | declare <16 x i8> @llvm.x86.sse41.pmaxsb(<16 x i8>, <16 x i8>) nounwind readnone |
| 570 | |
| 571 | define <4 x i32> @stack_fold_pmaxsd(<4 x i32> %a0, <4 x i32> %a1) { |
| 572 | ;CHECK-LABEL: stack_fold_pmaxsd |
| 573 | ;CHECK: vpmaxsd {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 574 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 575 | %2 = call <4 x i32> @llvm.x86.sse41.pmaxsd(<4 x i32> %a0, <4 x i32> %a1) |
| 576 | ret <4 x i32> %2 |
| 577 | } |
| 578 | declare <4 x i32> @llvm.x86.sse41.pmaxsd(<4 x i32>, <4 x i32>) nounwind readnone |
| 579 | |
| 580 | define <8 x i16> @stack_fold_pmaxsw(<8 x i16> %a0, <8 x i16> %a1) { |
| 581 | ;CHECK-LABEL: stack_fold_pmaxsw |
| 582 | ;CHECK: vpmaxsw {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 583 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 584 | %2 = call <8 x i16> @llvm.x86.sse2.pmaxs.w(<8 x i16> %a0, <8 x i16> %a1) |
| 585 | ret <8 x i16> %2 |
| 586 | } |
| 587 | declare <8 x i16> @llvm.x86.sse2.pmaxs.w(<8 x i16>, <8 x i16>) nounwind readnone |
| 588 | |
| 589 | define <16 x i8> @stack_fold_pmaxub(<16 x i8> %a0, <16 x i8> %a1) { |
| 590 | ;CHECK-LABEL: stack_fold_pmaxub |
| 591 | ;CHECK: vpmaxub {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 592 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 593 | %2 = call <16 x i8> @llvm.x86.sse2.pmaxu.b(<16 x i8> %a0, <16 x i8> %a1) |
| 594 | ret <16 x i8> %2 |
| 595 | } |
| 596 | declare <16 x i8> @llvm.x86.sse2.pmaxu.b(<16 x i8>, <16 x i8>) nounwind readnone |
| 597 | |
| 598 | define <4 x i32> @stack_fold_pmaxud(<4 x i32> %a0, <4 x i32> %a1) { |
| 599 | ;CHECK-LABEL: stack_fold_pmaxud |
| 600 | ;CHECK: vpmaxud {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 601 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 602 | %2 = call <4 x i32> @llvm.x86.sse41.pmaxud(<4 x i32> %a0, <4 x i32> %a1) |
| 603 | ret <4 x i32> %2 |
| 604 | } |
| 605 | declare <4 x i32> @llvm.x86.sse41.pmaxud(<4 x i32>, <4 x i32>) nounwind readnone |
| 606 | |
| 607 | define <8 x i16> @stack_fold_pmaxuw(<8 x i16> %a0, <8 x i16> %a1) { |
| 608 | ;CHECK-LABEL: stack_fold_pmaxuw |
| 609 | ;CHECK: vpmaxuw {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 610 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 611 | %2 = call <8 x i16> @llvm.x86.sse41.pmaxuw(<8 x i16> %a0, <8 x i16> %a1) |
| 612 | ret <8 x i16> %2 |
| 613 | } |
| 614 | declare <8 x i16> @llvm.x86.sse41.pmaxuw(<8 x i16>, <8 x i16>) nounwind readnone |
| 615 | |
| 616 | define <16 x i8> @stack_fold_pminsb(<16 x i8> %a0, <16 x i8> %a1) { |
| 617 | ;CHECK-LABEL: stack_fold_pminsb |
| 618 | ;CHECK: vpminsb {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 619 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 620 | %2 = call <16 x i8> @llvm.x86.sse41.pminsb(<16 x i8> %a0, <16 x i8> %a1) |
| 621 | ret <16 x i8> %2 |
| 622 | } |
| 623 | declare <16 x i8> @llvm.x86.sse41.pminsb(<16 x i8>, <16 x i8>) nounwind readnone |
| 624 | |
| 625 | define <4 x i32> @stack_fold_pminsd(<4 x i32> %a0, <4 x i32> %a1) { |
| 626 | ;CHECK-LABEL: stack_fold_pminsd |
| 627 | ;CHECK: vpminsd {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 628 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 629 | %2 = call <4 x i32> @llvm.x86.sse41.pminsd(<4 x i32> %a0, <4 x i32> %a1) |
| 630 | ret <4 x i32> %2 |
| 631 | } |
| 632 | declare <4 x i32> @llvm.x86.sse41.pminsd(<4 x i32>, <4 x i32>) nounwind readnone |
| 633 | |
| 634 | define <8 x i16> @stack_fold_pminsw(<8 x i16> %a0, <8 x i16> %a1) { |
| 635 | ;CHECK-LABEL: stack_fold_pminsw |
| 636 | ;CHECK: vpminsw {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 637 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 638 | %2 = call <8 x i16> @llvm.x86.sse2.pmins.w(<8 x i16> %a0, <8 x i16> %a1) |
| 639 | ret <8 x i16> %2 |
| 640 | } |
| 641 | declare <8 x i16> @llvm.x86.sse2.pmins.w(<8 x i16>, <8 x i16>) nounwind readnone |
| 642 | |
| 643 | define <16 x i8> @stack_fold_pminub(<16 x i8> %a0, <16 x i8> %a1) { |
| 644 | ;CHECK-LABEL: stack_fold_pminub |
| 645 | ;CHECK: vpminub {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 646 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 647 | %2 = call <16 x i8> @llvm.x86.sse2.pminu.b(<16 x i8> %a0, <16 x i8> %a1) |
| 648 | ret <16 x i8> %2 |
| 649 | } |
| 650 | declare <16 x i8> @llvm.x86.sse2.pminu.b(<16 x i8>, <16 x i8>) nounwind readnone |
| 651 | |
| 652 | define <4 x i32> @stack_fold_pminud(<4 x i32> %a0, <4 x i32> %a1) { |
| 653 | ;CHECK-LABEL: stack_fold_pminud |
| 654 | ;CHECK: vpminud {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 655 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 656 | %2 = call <4 x i32> @llvm.x86.sse41.pminud(<4 x i32> %a0, <4 x i32> %a1) |
| 657 | ret <4 x i32> %2 |
| 658 | } |
| 659 | declare <4 x i32> @llvm.x86.sse41.pminud(<4 x i32>, <4 x i32>) nounwind readnone |
| 660 | |
| 661 | define <8 x i16> @stack_fold_pminuw(<8 x i16> %a0, <8 x i16> %a1) { |
| 662 | ;CHECK-LABEL: stack_fold_pminuw |
| 663 | ;CHECK: vpminuw {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 664 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 665 | %2 = call <8 x i16> @llvm.x86.sse41.pminuw(<8 x i16> %a0, <8 x i16> %a1) |
| 666 | ret <8 x i16> %2 |
| 667 | } |
| 668 | declare <8 x i16> @llvm.x86.sse41.pminuw(<8 x i16>, <8 x i16>) nounwind readnone |
| 669 | |
| Simon Pilgrim | 5fa0fb2 | 2015-01-21 23:43:30 +0000 | [diff] [blame] | 670 | define <4 x i32> @stack_fold_pmovsxbd(<16 x i8> %a0) { |
| 671 | ;CHECK-LABEL: stack_fold_pmovsxbd |
| 672 | ;CHECK: vpmovsxbd {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 673 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm1},~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| Simon Pilgrim | 8ee50e9 | 2015-09-12 11:45:24 +0000 | [diff] [blame] | 674 | %2 = shufflevector <16 x i8> %a0, <16 x i8> undef, <4 x i32> <i32 0, i32 1, i32 2, i32 3> |
| 675 | %3 = sext <4 x i8> %2 to <4 x i32> |
| 676 | ret <4 x i32> %3 |
| Simon Pilgrim | 5fa0fb2 | 2015-01-21 23:43:30 +0000 | [diff] [blame] | 677 | } |
| Simon Pilgrim | 0177fa3 | 2015-01-20 23:54:17 +0000 | [diff] [blame] | 678 | |
| Simon Pilgrim | 5fa0fb2 | 2015-01-21 23:43:30 +0000 | [diff] [blame] | 679 | define <2 x i64> @stack_fold_pmovsxbq(<16 x i8> %a0) { |
| 680 | ;CHECK-LABEL: stack_fold_pmovsxbq |
| Simon Pilgrim | 8ee50e9 | 2015-09-12 11:45:24 +0000 | [diff] [blame] | 681 | ;CHECK: vpmovsxbq {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| Simon Pilgrim | 5fa0fb2 | 2015-01-21 23:43:30 +0000 | [diff] [blame] | 682 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm1},~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| Simon Pilgrim | 8ee50e9 | 2015-09-12 11:45:24 +0000 | [diff] [blame] | 683 | %2 = shufflevector <16 x i8> %a0, <16 x i8> undef, <2 x i32> <i32 0, i32 1> |
| 684 | %3 = sext <2 x i8> %2 to <2 x i64> |
| 685 | ret <2 x i64> %3 |
| Simon Pilgrim | 5fa0fb2 | 2015-01-21 23:43:30 +0000 | [diff] [blame] | 686 | } |
| Simon Pilgrim | 0177fa3 | 2015-01-20 23:54:17 +0000 | [diff] [blame] | 687 | |
| Simon Pilgrim | 5fa0fb2 | 2015-01-21 23:43:30 +0000 | [diff] [blame] | 688 | define <8 x i16> @stack_fold_pmovsxbw(<16 x i8> %a0) { |
| 689 | ;CHECK-LABEL: stack_fold_pmovsxbw |
| 690 | ;CHECK: vpmovsxbw {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 691 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm1},~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| Simon Pilgrim | 8ee50e9 | 2015-09-12 11:45:24 +0000 | [diff] [blame] | 692 | %2 = shufflevector <16 x i8> %a0, <16 x i8> undef, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7> |
| 693 | %3 = sext <8 x i8> %2 to <8 x i16> |
| 694 | ret <8 x i16> %3 |
| Simon Pilgrim | 5fa0fb2 | 2015-01-21 23:43:30 +0000 | [diff] [blame] | 695 | } |
| Simon Pilgrim | 0177fa3 | 2015-01-20 23:54:17 +0000 | [diff] [blame] | 696 | |
| Simon Pilgrim | 5fa0fb2 | 2015-01-21 23:43:30 +0000 | [diff] [blame] | 697 | define <2 x i64> @stack_fold_pmovsxdq(<4 x i32> %a0) { |
| 698 | ;CHECK-LABEL: stack_fold_pmovsxdq |
| 699 | ;CHECK: vpmovsxdq {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 700 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm1},~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| Simon Pilgrim | 8ee50e9 | 2015-09-12 11:45:24 +0000 | [diff] [blame] | 701 | %2 = shufflevector <4 x i32> %a0, <4 x i32> undef, <2 x i32> <i32 0, i32 1> |
| 702 | %3 = sext <2 x i32> %2 to <2 x i64> |
| 703 | ret <2 x i64> %3 |
| Simon Pilgrim | 5fa0fb2 | 2015-01-21 23:43:30 +0000 | [diff] [blame] | 704 | } |
| Simon Pilgrim | 0177fa3 | 2015-01-20 23:54:17 +0000 | [diff] [blame] | 705 | |
| Simon Pilgrim | 5fa0fb2 | 2015-01-21 23:43:30 +0000 | [diff] [blame] | 706 | define <4 x i32> @stack_fold_pmovsxwd(<8 x i16> %a0) { |
| 707 | ;CHECK-LABEL: stack_fold_pmovsxwd |
| 708 | ;CHECK: vpmovsxwd {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 709 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm1},~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| Simon Pilgrim | 8ee50e9 | 2015-09-12 11:45:24 +0000 | [diff] [blame] | 710 | %2 = shufflevector <8 x i16> %a0, <8 x i16> undef, <4 x i32> <i32 0, i32 1, i32 2, i32 3> |
| 711 | %3 = sext <4 x i16> %2 to <4 x i32> |
| 712 | ret <4 x i32> %3 |
| Simon Pilgrim | 5fa0fb2 | 2015-01-21 23:43:30 +0000 | [diff] [blame] | 713 | } |
| Simon Pilgrim | 0177fa3 | 2015-01-20 23:54:17 +0000 | [diff] [blame] | 714 | |
| Simon Pilgrim | 5fa0fb2 | 2015-01-21 23:43:30 +0000 | [diff] [blame] | 715 | define <2 x i64> @stack_fold_pmovsxwq(<8 x i16> %a0) { |
| 716 | ;CHECK-LABEL: stack_fold_pmovsxwq |
| 717 | ;CHECK: vpmovsxwq {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 718 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm1},~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| Simon Pilgrim | 8ee50e9 | 2015-09-12 11:45:24 +0000 | [diff] [blame] | 719 | %2 = shufflevector <8 x i16> %a0, <8 x i16> undef, <2 x i32> <i32 0, i32 1> |
| 720 | %3 = sext <2 x i16> %2 to <2 x i64> |
| 721 | ret <2 x i64> %3 |
| Simon Pilgrim | 5fa0fb2 | 2015-01-21 23:43:30 +0000 | [diff] [blame] | 722 | } |
| Simon Pilgrim | 0177fa3 | 2015-01-20 23:54:17 +0000 | [diff] [blame] | 723 | |
| Simon Pilgrim | 5fa0fb2 | 2015-01-21 23:43:30 +0000 | [diff] [blame] | 724 | define <4 x i32> @stack_fold_pmovzxbd(<16 x i8> %a0) { |
| 725 | ;CHECK-LABEL: stack_fold_pmovzxbd |
| 726 | ;CHECK: vpmovzxbd {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 727 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm1},~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| Simon Pilgrim | ece7475 | 2015-03-14 23:16:43 +0000 | [diff] [blame] | 728 | %2 = shufflevector <16 x i8> %a0, <16 x i8> zeroinitializer, <16 x i32> <i32 0, i32 16, i32 17, i32 18, i32 1, i32 19, i32 20, i32 21, i32 2, i32 22, i32 23, i32 24, i32 3, i32 25, i32 26, i32 27> |
| 729 | %3 = bitcast <16 x i8> %2 to <4 x i32> |
| 730 | ret <4 x i32> %3 |
| Simon Pilgrim | 5fa0fb2 | 2015-01-21 23:43:30 +0000 | [diff] [blame] | 731 | } |
| Simon Pilgrim | 0177fa3 | 2015-01-20 23:54:17 +0000 | [diff] [blame] | 732 | |
| Simon Pilgrim | 5fa0fb2 | 2015-01-21 23:43:30 +0000 | [diff] [blame] | 733 | define <2 x i64> @stack_fold_pmovzxbq(<16 x i8> %a0) { |
| 734 | ;CHECK-LABEL: stack_fold_pmovzxbq |
| 735 | ;CHECK: vpmovzxbq {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 736 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm1},~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| Simon Pilgrim | ece7475 | 2015-03-14 23:16:43 +0000 | [diff] [blame] | 737 | %2 = shufflevector <16 x i8> %a0, <16 x i8> zeroinitializer, <16 x i32> <i32 0, i32 16, i32 17, i32 18, i32 19, i32 20, i32 21, i32 22, i32 1, i32 22, i32 23, i32 24, i32 25, i32 26, i32 27, i32 28> |
| 738 | %3 = bitcast <16 x i8> %2 to <2 x i64> |
| 739 | ret <2 x i64> %3 |
| Simon Pilgrim | 5fa0fb2 | 2015-01-21 23:43:30 +0000 | [diff] [blame] | 740 | } |
| Simon Pilgrim | 0177fa3 | 2015-01-20 23:54:17 +0000 | [diff] [blame] | 741 | |
| Simon Pilgrim | 5fa0fb2 | 2015-01-21 23:43:30 +0000 | [diff] [blame] | 742 | define <8 x i16> @stack_fold_pmovzxbw(<16 x i8> %a0) { |
| 743 | ;CHECK-LABEL: stack_fold_pmovzxbw |
| 744 | ;CHECK: vpmovzxbw {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 745 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm1},~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| Simon Pilgrim | ece7475 | 2015-03-14 23:16:43 +0000 | [diff] [blame] | 746 | %2 = shufflevector <16 x i8> %a0, <16 x i8> zeroinitializer, <16 x i32> <i32 0, i32 16, i32 1, i32 17, i32 2, i32 18, i32 3, i32 19, i32 4, i32 20, i32 5, i32 21, i32 6, i32 22, i32 7, i32 23> |
| 747 | %3 = bitcast <16 x i8> %2 to <8 x i16> |
| 748 | ret <8 x i16> %3 |
| Simon Pilgrim | 5fa0fb2 | 2015-01-21 23:43:30 +0000 | [diff] [blame] | 749 | } |
| Simon Pilgrim | 0177fa3 | 2015-01-20 23:54:17 +0000 | [diff] [blame] | 750 | |
| Simon Pilgrim | 5fa0fb2 | 2015-01-21 23:43:30 +0000 | [diff] [blame] | 751 | define <2 x i64> @stack_fold_pmovzxdq(<4 x i32> %a0) { |
| 752 | ;CHECK-LABEL: stack_fold_pmovzxdq |
| 753 | ;CHECK: vpmovzxdq {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 754 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm1},~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| Simon Pilgrim | ece7475 | 2015-03-14 23:16:43 +0000 | [diff] [blame] | 755 | %2 = shufflevector <4 x i32> %a0, <4 x i32> zeroinitializer, <4 x i32> <i32 0, i32 4, i32 1, i32 5> |
| 756 | %3 = bitcast <4 x i32> %2 to <2 x i64> |
| 757 | ret <2 x i64> %3 |
| Simon Pilgrim | 5fa0fb2 | 2015-01-21 23:43:30 +0000 | [diff] [blame] | 758 | } |
| Simon Pilgrim | 0177fa3 | 2015-01-20 23:54:17 +0000 | [diff] [blame] | 759 | |
| Simon Pilgrim | 5fa0fb2 | 2015-01-21 23:43:30 +0000 | [diff] [blame] | 760 | define <4 x i32> @stack_fold_pmovzxwd(<8 x i16> %a0) { |
| 761 | ;CHECK-LABEL: stack_fold_pmovzxwd |
| 762 | ;CHECK: vpmovzxwd {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 763 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm1},~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| Simon Pilgrim | ece7475 | 2015-03-14 23:16:43 +0000 | [diff] [blame] | 764 | %2 = shufflevector <8 x i16> %a0, <8 x i16> zeroinitializer, <8 x i32> <i32 0, i32 8, i32 1, i32 9, i32 2, i32 10, i32 3, i32 11> |
| 765 | %3 = bitcast <8 x i16> %2 to <4 x i32> |
| 766 | ret <4 x i32> %3 |
| Simon Pilgrim | 5fa0fb2 | 2015-01-21 23:43:30 +0000 | [diff] [blame] | 767 | } |
| Simon Pilgrim | 0177fa3 | 2015-01-20 23:54:17 +0000 | [diff] [blame] | 768 | |
| Simon Pilgrim | 5fa0fb2 | 2015-01-21 23:43:30 +0000 | [diff] [blame] | 769 | define <2 x i64> @stack_fold_pmovzxwq(<8 x i16> %a0) { |
| 770 | ;CHECK-LABEL: stack_fold_pmovzxwq |
| 771 | ;CHECK: vpmovzxwq {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 772 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm1},~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| Simon Pilgrim | ece7475 | 2015-03-14 23:16:43 +0000 | [diff] [blame] | 773 | %2 = shufflevector <8 x i16> %a0, <8 x i16> zeroinitializer, <8 x i32> <i32 0, i32 8, i32 9, i32 10, i32 1, i32 11, i32 12, i32 13> |
| 774 | %3 = bitcast <8 x i16> %2 to <2 x i64> |
| 775 | ret <2 x i64> %3 |
| Simon Pilgrim | 5fa0fb2 | 2015-01-21 23:43:30 +0000 | [diff] [blame] | 776 | } |
| Simon Pilgrim | 0177fa3 | 2015-01-20 23:54:17 +0000 | [diff] [blame] | 777 | |
| 778 | define <2 x i64> @stack_fold_pmuldq(<4 x i32> %a0, <4 x i32> %a1) { |
| 779 | ;CHECK-LABEL: stack_fold_pmuldq |
| 780 | ;CHECK: vpmuldq {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 781 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 782 | %2 = call <2 x i64> @llvm.x86.sse41.pmuldq(<4 x i32> %a0, <4 x i32> %a1) |
| 783 | ret <2 x i64> %2 |
| 784 | } |
| 785 | declare <2 x i64> @llvm.x86.sse41.pmuldq(<4 x i32>, <4 x i32>) nounwind readnone |
| 786 | |
| 787 | define <8 x i16> @stack_fold_pmulhrsw(<8 x i16> %a0, <8 x i16> %a1) { |
| 788 | ;CHECK-LABEL: stack_fold_pmulhrsw |
| 789 | ;CHECK: vpmulhrsw {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 790 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 791 | %2 = call <8 x i16> @llvm.x86.ssse3.pmul.hr.sw.128(<8 x i16> %a0, <8 x i16> %a1) |
| 792 | ret <8 x i16> %2 |
| 793 | } |
| 794 | declare <8 x i16> @llvm.x86.ssse3.pmul.hr.sw.128(<8 x i16>, <8 x i16>) nounwind readnone |
| 795 | |
| 796 | define <8 x i16> @stack_fold_pmulhuw(<8 x i16> %a0, <8 x i16> %a1) { |
| 797 | ;CHECK-LABEL: stack_fold_pmulhuw |
| 798 | ;CHECK: vpmulhuw {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 799 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 800 | %2 = call <8 x i16> @llvm.x86.sse2.pmulhu.w(<8 x i16> %a0, <8 x i16> %a1) |
| 801 | ret <8 x i16> %2 |
| 802 | } |
| 803 | declare <8 x i16> @llvm.x86.sse2.pmulhu.w(<8 x i16>, <8 x i16>) nounwind readnone |
| 804 | |
| 805 | define <8 x i16> @stack_fold_pmulhw(<8 x i16> %a0, <8 x i16> %a1) { |
| 806 | ;CHECK-LABEL: stack_fold_pmulhw |
| 807 | ;CHECK: vpmulhw {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 808 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 809 | %2 = call <8 x i16> @llvm.x86.sse2.pmulh.w(<8 x i16> %a0, <8 x i16> %a1) |
| 810 | ret <8 x i16> %2 |
| 811 | } |
| 812 | declare <8 x i16> @llvm.x86.sse2.pmulh.w(<8 x i16>, <8 x i16>) nounwind readnone |
| 813 | |
| 814 | define <4 x i32> @stack_fold_pmulld(<4 x i32> %a0, <4 x i32> %a1) { |
| 815 | ;CHECK-LABEL: stack_fold_pmulld |
| 816 | ;CHECK: vpmulld {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 817 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 818 | %2 = mul <4 x i32> %a0, %a1 |
| 819 | ret <4 x i32> %2 |
| 820 | } |
| 821 | |
| 822 | define <8 x i16> @stack_fold_pmullw(<8 x i16> %a0, <8 x i16> %a1) { |
| 823 | ;CHECK-LABEL: stack_fold_pmullw |
| 824 | ;CHECK: vpmullw {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 825 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 826 | %2 = mul <8 x i16> %a0, %a1 |
| 827 | ret <8 x i16> %2 |
| 828 | } |
| 829 | |
| 830 | define <2 x i64> @stack_fold_pmuludq(<4 x i32> %a0, <4 x i32> %a1) { |
| 831 | ;CHECK-LABEL: stack_fold_pmuludq |
| 832 | ;CHECK: vpmuludq {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 833 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 834 | %2 = call <2 x i64> @llvm.x86.sse2.pmulu.dq(<4 x i32> %a0, <4 x i32> %a1) |
| 835 | ret <2 x i64> %2 |
| 836 | } |
| 837 | declare <2 x i64> @llvm.x86.sse2.pmulu.dq(<4 x i32>, <4 x i32>) nounwind readnone |
| 838 | |
| 839 | define <16 x i8> @stack_fold_por(<16 x i8> %a0, <16 x i8> %a1) { |
| 840 | ;CHECK-LABEL: stack_fold_por |
| 841 | ;CHECK: vpor {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 842 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 843 | %2 = or <16 x i8> %a0, %a1 |
| 844 | ; add forces execution domain |
| 845 | %3 = add <16 x i8> %2, <i8 1, i8 1, i8 1, i8 1, i8 1, i8 1, i8 1, i8 1, i8 1, i8 1, i8 1, i8 1, i8 1, i8 1, i8 1, i8 1> |
| 846 | ret <16 x i8> %3 |
| 847 | } |
| 848 | |
| 849 | define <2 x i64> @stack_fold_psadbw(<16 x i8> %a0, <16 x i8> %a1) { |
| 850 | ;CHECK-LABEL: stack_fold_psadbw |
| 851 | ;CHECK: vpsadbw {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 852 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 853 | %2 = call <2 x i64> @llvm.x86.sse2.psad.bw(<16 x i8> %a0, <16 x i8> %a1) |
| 854 | ret <2 x i64> %2 |
| 855 | } |
| 856 | declare <2 x i64> @llvm.x86.sse2.psad.bw(<16 x i8>, <16 x i8>) nounwind readnone |
| 857 | |
| 858 | define <16 x i8> @stack_fold_pshufb(<16 x i8> %a0, <16 x i8> %a1) { |
| 859 | ;CHECK-LABEL: stack_fold_pshufb |
| 860 | ;CHECK: vpshufb {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 861 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 862 | %2 = call <16 x i8> @llvm.x86.ssse3.pshuf.b.128(<16 x i8> %a0, <16 x i8> %a1) |
| 863 | ret <16 x i8> %2 |
| 864 | } |
| 865 | declare <16 x i8> @llvm.x86.ssse3.pshuf.b.128(<16 x i8>, <16 x i8>) nounwind readnone |
| 866 | |
| 867 | define <4 x i32> @stack_fold_pshufd(<4 x i32> %a0) { |
| 868 | ;CHECK-LABEL: stack_fold_pshufd |
| 869 | ;CHECK: vpshufd $27, {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 870 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm1},~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 871 | %2 = shufflevector <4 x i32> %a0, <4 x i32> undef, <4 x i32> <i32 3, i32 2, i32 1, i32 0> |
| 872 | ret <4 x i32> %2 |
| 873 | } |
| 874 | |
| 875 | define <8 x i16> @stack_fold_pshufhw(<8 x i16> %a0) { |
| 876 | ;CHECK-LABEL: stack_fold_pshufhw |
| 877 | ;CHECK: vpshufhw $11, {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 878 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm1},~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 879 | %2 = shufflevector <8 x i16> %a0, <8 x i16> undef, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 7, i32 6, i32 4, i32 4> |
| 880 | ret <8 x i16> %2 |
| 881 | } |
| 882 | |
| 883 | define <8 x i16> @stack_fold_pshuflw(<8 x i16> %a0) { |
| 884 | ;CHECK-LABEL: stack_fold_pshuflw |
| 885 | ;CHECK: vpshuflw $27, {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 886 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm1},~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 887 | %2 = shufflevector <8 x i16> %a0, <8 x i16> undef, <8 x i32> <i32 3, i32 2, i32 1, i32 0, i32 4, i32 5, i32 6, i32 7> |
| 888 | ret <8 x i16> %2 |
| 889 | } |
| 890 | |
| 891 | define <16 x i8> @stack_fold_psignb(<16 x i8> %a0, <16 x i8> %a1) { |
| 892 | ;CHECK-LABEL: stack_fold_psignb |
| 893 | ;CHECK: vpsignb {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 894 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 895 | %2 = call <16 x i8> @llvm.x86.ssse3.psign.b.128(<16 x i8> %a0, <16 x i8> %a1) |
| 896 | ret <16 x i8> %2 |
| 897 | } |
| 898 | declare <16 x i8> @llvm.x86.ssse3.psign.b.128(<16 x i8>, <16 x i8>) nounwind readnone |
| 899 | |
| 900 | define <4 x i32> @stack_fold_psignd(<4 x i32> %a0, <4 x i32> %a1) { |
| 901 | ;CHECK-LABEL: stack_fold_psignd |
| 902 | ;CHECK: vpsignd {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 903 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 904 | %2 = call <4 x i32> @llvm.x86.ssse3.psign.d.128(<4 x i32> %a0, <4 x i32> %a1) |
| 905 | ret <4 x i32> %2 |
| 906 | } |
| 907 | declare <4 x i32> @llvm.x86.ssse3.psign.d.128(<4 x i32>, <4 x i32>) nounwind readnone |
| 908 | |
| 909 | define <8 x i16> @stack_fold_psignw(<8 x i16> %a0, <8 x i16> %a1) { |
| 910 | ;CHECK-LABEL: stack_fold_psignw |
| 911 | ;CHECK: vpsignw {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 912 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 913 | %2 = call <8 x i16> @llvm.x86.ssse3.psign.w.128(<8 x i16> %a0, <8 x i16> %a1) |
| 914 | ret <8 x i16> %2 |
| 915 | } |
| 916 | declare <8 x i16> @llvm.x86.ssse3.psign.w.128(<8 x i16>, <8 x i16>) nounwind readnone |
| 917 | |
| 918 | define <4 x i32> @stack_fold_pslld(<4 x i32> %a0, <4 x i32> %a1) { |
| 919 | ;CHECK-LABEL: stack_fold_pslld |
| 920 | ;CHECK: vpslld {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 921 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 922 | %2 = call <4 x i32> @llvm.x86.sse2.psll.d(<4 x i32> %a0, <4 x i32> %a1) |
| 923 | ret <4 x i32> %2 |
| 924 | } |
| 925 | declare <4 x i32> @llvm.x86.sse2.psll.d(<4 x i32>, <4 x i32>) nounwind readnone |
| 926 | |
| 927 | define <2 x i64> @stack_fold_psllq(<2 x i64> %a0, <2 x i64> %a1) { |
| 928 | ;CHECK-LABEL: stack_fold_psllq |
| 929 | ;CHECK: vpsllq {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 930 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 931 | %2 = call <2 x i64> @llvm.x86.sse2.psll.q(<2 x i64> %a0, <2 x i64> %a1) |
| 932 | ret <2 x i64> %2 |
| 933 | } |
| 934 | declare <2 x i64> @llvm.x86.sse2.psll.q(<2 x i64>, <2 x i64>) nounwind readnone |
| 935 | |
| 936 | define <8 x i16> @stack_fold_psllw(<8 x i16> %a0, <8 x i16> %a1) { |
| 937 | ;CHECK-LABEL: stack_fold_psllw |
| 938 | ;CHECK: vpsllw {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 939 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 940 | %2 = call <8 x i16> @llvm.x86.sse2.psll.w(<8 x i16> %a0, <8 x i16> %a1) |
| 941 | ret <8 x i16> %2 |
| 942 | } |
| 943 | declare <8 x i16> @llvm.x86.sse2.psll.w(<8 x i16>, <8 x i16>) nounwind readnone |
| 944 | |
| 945 | define <4 x i32> @stack_fold_psrad(<4 x i32> %a0, <4 x i32> %a1) { |
| 946 | ;CHECK-LABEL: stack_fold_psrad |
| 947 | ;CHECK: vpsrad {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 948 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 949 | %2 = call <4 x i32> @llvm.x86.sse2.psra.d(<4 x i32> %a0, <4 x i32> %a1) |
| 950 | ret <4 x i32> %2 |
| 951 | } |
| 952 | declare <4 x i32> @llvm.x86.sse2.psra.d(<4 x i32>, <4 x i32>) nounwind readnone |
| 953 | |
| 954 | define <8 x i16> @stack_fold_psraw(<8 x i16> %a0, <8 x i16> %a1) { |
| 955 | ;CHECK-LABEL: stack_fold_psraw |
| 956 | ;CHECK: vpsraw {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 957 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 958 | %2 = call <8 x i16> @llvm.x86.sse2.psra.w(<8 x i16> %a0, <8 x i16> %a1) |
| 959 | ret <8 x i16> %2 |
| 960 | } |
| 961 | declare <8 x i16> @llvm.x86.sse2.psra.w(<8 x i16>, <8 x i16>) nounwind readnone |
| 962 | |
| 963 | define <4 x i32> @stack_fold_psrld(<4 x i32> %a0, <4 x i32> %a1) { |
| 964 | ;CHECK-LABEL: stack_fold_psrld |
| 965 | ;CHECK: vpsrld {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 966 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 967 | %2 = call <4 x i32> @llvm.x86.sse2.psrl.d(<4 x i32> %a0, <4 x i32> %a1) |
| 968 | ret <4 x i32> %2 |
| 969 | } |
| 970 | declare <4 x i32> @llvm.x86.sse2.psrl.d(<4 x i32>, <4 x i32>) nounwind readnone |
| 971 | |
| 972 | define <2 x i64> @stack_fold_psrlq(<2 x i64> %a0, <2 x i64> %a1) { |
| 973 | ;CHECK-LABEL: stack_fold_psrlq |
| 974 | ;CHECK: vpsrlq {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 975 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 976 | %2 = call <2 x i64> @llvm.x86.sse2.psrl.q(<2 x i64> %a0, <2 x i64> %a1) |
| 977 | ret <2 x i64> %2 |
| 978 | } |
| 979 | declare <2 x i64> @llvm.x86.sse2.psrl.q(<2 x i64>, <2 x i64>) nounwind readnone |
| 980 | |
| 981 | define <8 x i16> @stack_fold_psrlw(<8 x i16> %a0, <8 x i16> %a1) { |
| 982 | ;CHECK-LABEL: stack_fold_psrlw |
| 983 | ;CHECK: vpsrlw {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 984 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 985 | %2 = call <8 x i16> @llvm.x86.sse2.psrl.w(<8 x i16> %a0, <8 x i16> %a1) |
| 986 | ret <8 x i16> %2 |
| 987 | } |
| 988 | declare <8 x i16> @llvm.x86.sse2.psrl.w(<8 x i16>, <8 x i16>) nounwind readnone |
| 989 | |
| 990 | define <16 x i8> @stack_fold_psubb(<16 x i8> %a0, <16 x i8> %a1) { |
| 991 | ;CHECK-LABEL: stack_fold_psubb |
| 992 | ;CHECK: vpsubb {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 993 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 994 | %2 = sub <16 x i8> %a0, %a1 |
| 995 | ret <16 x i8> %2 |
| 996 | } |
| 997 | |
| 998 | define <4 x i32> @stack_fold_psubd(<4 x i32> %a0, <4 x i32> %a1) { |
| 999 | ;CHECK-LABEL: stack_fold_psubd |
| 1000 | ;CHECK: vpsubd {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 1001 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 1002 | %2 = sub <4 x i32> %a0, %a1 |
| 1003 | ret <4 x i32> %2 |
| 1004 | } |
| 1005 | |
| Simon Pilgrim | 5fa0fb2 | 2015-01-21 23:43:30 +0000 | [diff] [blame] | 1006 | define <2 x i64> @stack_fold_psubq(<2 x i64> %a0, <2 x i64> %a1) { |
| 1007 | ;CHECK-LABEL: stack_fold_psubq |
| 1008 | ;CHECK: vpsubq {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 1009 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 1010 | %2 = sub <2 x i64> %a0, %a1 |
| 1011 | ret <2 x i64> %2 |
| 1012 | } |
| Simon Pilgrim | 0177fa3 | 2015-01-20 23:54:17 +0000 | [diff] [blame] | 1013 | |
| 1014 | define <16 x i8> @stack_fold_psubsb(<16 x i8> %a0, <16 x i8> %a1) { |
| 1015 | ;CHECK-LABEL: stack_fold_psubsb |
| 1016 | ;CHECK: vpsubsb {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 1017 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 1018 | %2 = call <16 x i8> @llvm.x86.sse2.psubs.b(<16 x i8> %a0, <16 x i8> %a1) |
| 1019 | ret <16 x i8> %2 |
| 1020 | } |
| 1021 | declare <16 x i8> @llvm.x86.sse2.psubs.b(<16 x i8>, <16 x i8>) nounwind readnone |
| 1022 | |
| 1023 | define <8 x i16> @stack_fold_psubsw(<8 x i16> %a0, <8 x i16> %a1) { |
| 1024 | ;CHECK-LABEL: stack_fold_psubsw |
| 1025 | ;CHECK: vpsubsw {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 1026 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 1027 | %2 = call <8 x i16> @llvm.x86.sse2.psubs.w(<8 x i16> %a0, <8 x i16> %a1) |
| 1028 | ret <8 x i16> %2 |
| 1029 | } |
| 1030 | declare <8 x i16> @llvm.x86.sse2.psubs.w(<8 x i16>, <8 x i16>) nounwind readnone |
| 1031 | |
| Simon Pilgrim | 5fa0fb2 | 2015-01-21 23:43:30 +0000 | [diff] [blame] | 1032 | define <16 x i8> @stack_fold_psubusb(<16 x i8> %a0, <16 x i8> %a1) { |
| 1033 | ;CHECK-LABEL: stack_fold_psubusb |
| 1034 | ;CHECK: vpsubusb {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 1035 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 1036 | %2 = call <16 x i8> @llvm.x86.sse2.psubus.b(<16 x i8> %a0, <16 x i8> %a1) |
| 1037 | ret <16 x i8> %2 |
| 1038 | } |
| Simon Pilgrim | 0177fa3 | 2015-01-20 23:54:17 +0000 | [diff] [blame] | 1039 | declare <16 x i8> @llvm.x86.sse2.psubus.b(<16 x i8>, <16 x i8>) nounwind readnone |
| 1040 | |
| Simon Pilgrim | 5fa0fb2 | 2015-01-21 23:43:30 +0000 | [diff] [blame] | 1041 | define <8 x i16> @stack_fold_psubusw(<8 x i16> %a0, <8 x i16> %a1) { |
| 1042 | ;CHECK-LABEL: stack_fold_psubusw |
| 1043 | ;CHECK: vpsubusw {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 1044 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 1045 | %2 = call <8 x i16> @llvm.x86.sse2.psubus.w(<8 x i16> %a0, <8 x i16> %a1) |
| 1046 | ret <8 x i16> %2 |
| 1047 | } |
| Simon Pilgrim | 0177fa3 | 2015-01-20 23:54:17 +0000 | [diff] [blame] | 1048 | declare <8 x i16> @llvm.x86.sse2.psubus.w(<8 x i16>, <8 x i16>) nounwind readnone |
| 1049 | |
| 1050 | define <8 x i16> @stack_fold_psubw(<8 x i16> %a0, <8 x i16> %a1) { |
| 1051 | ;CHECK-LABEL: stack_fold_psubw |
| 1052 | ;CHECK: vpsubw {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 1053 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 1054 | %2 = sub <8 x i16> %a0, %a1 |
| 1055 | ret <8 x i16> %2 |
| 1056 | } |
| 1057 | |
| Simon Pilgrim | 5fa0fb2 | 2015-01-21 23:43:30 +0000 | [diff] [blame] | 1058 | define i32 @stack_fold_ptest(<2 x i64> %a0, <2 x i64> %a1) { |
| 1059 | ;CHECK-LABEL: stack_fold_ptest |
| 1060 | ;CHECK: vptest {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 1061 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 1062 | %2 = call i32 @llvm.x86.sse41.ptestc(<2 x i64> %a0, <2 x i64> %a1) |
| 1063 | ret i32 %2 |
| 1064 | } |
| Simon Pilgrim | 0177fa3 | 2015-01-20 23:54:17 +0000 | [diff] [blame] | 1065 | declare i32 @llvm.x86.sse41.ptestc(<2 x i64>, <2 x i64>) nounwind readnone |
| 1066 | |
| Simon Pilgrim | a261867 | 2015-02-07 21:44:06 +0000 | [diff] [blame] | 1067 | define i32 @stack_fold_ptest_ymm(<4 x i64> %a0, <4 x i64> %a1) { |
| 1068 | ;CHECK-LABEL: stack_fold_ptest_ymm |
| 1069 | ;CHECK: vptest {{-?[0-9]*}}(%rsp), {{%ymm[0-9][0-9]*}} {{.*#+}} 32-byte Folded Reload |
| 1070 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 1071 | %2 = call i32 @llvm.x86.avx.ptestc.256(<4 x i64> %a0, <4 x i64> %a1) |
| 1072 | ret i32 %2 |
| 1073 | } |
| 1074 | declare i32 @llvm.x86.avx.ptestc.256(<4 x i64>, <4 x i64>) nounwind readnone |
| 1075 | |
| Simon Pilgrim | 0177fa3 | 2015-01-20 23:54:17 +0000 | [diff] [blame] | 1076 | define <16 x i8> @stack_fold_punpckhbw(<16 x i8> %a0, <16 x i8> %a1) { |
| 1077 | ;CHECK-LABEL: stack_fold_punpckhbw |
| 1078 | ;CHECK: vpunpckhbw {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 1079 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 1080 | %2 = shufflevector <16 x i8> %a0, <16 x i8> %a1, <16 x i32> <i32 8, i32 24, i32 9, i32 25, i32 10, i32 26, i32 11, i32 27, i32 12, i32 28, i32 13, i32 29, i32 14, i32 30, i32 15, i32 31> |
| 1081 | ret <16 x i8> %2 |
| 1082 | } |
| 1083 | |
| 1084 | define <4 x i32> @stack_fold_punpckhdq(<4 x i32> %a0, <4 x i32> %a1) { |
| 1085 | ;CHECK-LABEL: stack_fold_punpckhdq |
| 1086 | ;CHECK: vpunpckhdq {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 1087 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 1088 | %2 = shufflevector <4 x i32> %a0, <4 x i32> %a1, <4 x i32> <i32 2, i32 6, i32 3, i32 7> |
| Simon Pilgrim | b4a0df9 | 2015-02-12 22:47:45 +0000 | [diff] [blame] | 1089 | ; add forces execution domain |
| 1090 | %3 = add <4 x i32> %2, <i32 1, i32 1, i32 1, i32 1> |
| 1091 | ret <4 x i32> %3 |
| Simon Pilgrim | 0177fa3 | 2015-01-20 23:54:17 +0000 | [diff] [blame] | 1092 | } |
| 1093 | |
| 1094 | define <2 x i64> @stack_fold_punpckhqdq(<2 x i64> %a0, <2 x i64> %a1) { |
| 1095 | ;CHECK-LABEL: stack_fold_punpckhqdq |
| 1096 | ;CHECK: vpunpckhqdq {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 1097 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 1098 | %2 = shufflevector <2 x i64> %a0, <2 x i64> %a1, <2 x i32> <i32 1, i32 3> |
| Simon Pilgrim | b4a0df9 | 2015-02-12 22:47:45 +0000 | [diff] [blame] | 1099 | ; add forces execution domain |
| 1100 | %3 = add <2 x i64> %2, <i64 1, i64 1> |
| 1101 | ret <2 x i64> %3 |
| Simon Pilgrim | 0177fa3 | 2015-01-20 23:54:17 +0000 | [diff] [blame] | 1102 | } |
| 1103 | |
| 1104 | define <8 x i16> @stack_fold_punpckhwd(<8 x i16> %a0, <8 x i16> %a1) { |
| 1105 | ;CHECK-LABEL: stack_fold_punpckhwd |
| 1106 | ;CHECK: vpunpckhwd {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 1107 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 1108 | %2 = shufflevector <8 x i16> %a0, <8 x i16> %a1, <8 x i32> <i32 4, i32 12, i32 5, i32 13, i32 6, i32 14, i32 7, i32 15> |
| 1109 | ret <8 x i16> %2 |
| 1110 | } |
| 1111 | |
| 1112 | define <16 x i8> @stack_fold_punpcklbw(<16 x i8> %a0, <16 x i8> %a1) { |
| 1113 | ;CHECK-LABEL: stack_fold_punpcklbw |
| 1114 | ;CHECK: vpunpcklbw {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 1115 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 1116 | %2 = shufflevector <16 x i8> %a0, <16 x i8> %a1, <16 x i32> <i32 0, i32 16, i32 1, i32 17, i32 2, i32 18, i32 3, i32 19, i32 4, i32 20, i32 5, i32 21, i32 6, i32 22, i32 7, i32 23> |
| 1117 | ret <16 x i8> %2 |
| 1118 | } |
| 1119 | |
| 1120 | define <4 x i32> @stack_fold_punpckldq(<4 x i32> %a0, <4 x i32> %a1) { |
| 1121 | ;CHECK-LABEL: stack_fold_punpckldq |
| 1122 | ;CHECK: vpunpckldq {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 1123 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 1124 | %2 = shufflevector <4 x i32> %a0, <4 x i32> %a1, <4 x i32> <i32 0, i32 4, i32 1, i32 5> |
| Simon Pilgrim | b4a0df9 | 2015-02-12 22:47:45 +0000 | [diff] [blame] | 1125 | ; add forces execution domain |
| 1126 | %3 = add <4 x i32> %2, <i32 1, i32 1, i32 1, i32 1> |
| 1127 | ret <4 x i32> %3 |
| Simon Pilgrim | 0177fa3 | 2015-01-20 23:54:17 +0000 | [diff] [blame] | 1128 | } |
| 1129 | |
| 1130 | define <2 x i64> @stack_fold_punpcklqdq(<2 x i64> %a0, <2 x i64> %a1) { |
| 1131 | ;CHECK-LABEL: stack_fold_punpcklqdq |
| 1132 | ;CHECK: vpunpcklqdq {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 1133 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 1134 | %2 = shufflevector <2 x i64> %a0, <2 x i64> %a1, <2 x i32> <i32 0, i32 2> |
| Simon Pilgrim | b4a0df9 | 2015-02-12 22:47:45 +0000 | [diff] [blame] | 1135 | ; add forces execution domain |
| 1136 | %3 = add <2 x i64> %2, <i64 1, i64 1> |
| 1137 | ret <2 x i64> %3 |
| Simon Pilgrim | 0177fa3 | 2015-01-20 23:54:17 +0000 | [diff] [blame] | 1138 | } |
| 1139 | |
| 1140 | define <8 x i16> @stack_fold_punpcklwd(<8 x i16> %a0, <8 x i16> %a1) { |
| 1141 | ;CHECK-LABEL: stack_fold_punpcklwd |
| 1142 | ;CHECK: vpunpcklwd {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 1143 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 1144 | %2 = shufflevector <8 x i16> %a0, <8 x i16> %a1, <8 x i32> <i32 0, i32 8, i32 1, i32 9, i32 2, i32 10, i32 3, i32 11> |
| 1145 | ret <8 x i16> %2 |
| 1146 | } |
| 1147 | |
| 1148 | define <16 x i8> @stack_fold_pxor(<16 x i8> %a0, <16 x i8> %a1) { |
| 1149 | ;CHECK-LABEL: stack_fold_pxor |
| 1150 | ;CHECK: vpxor {{-?[0-9]*}}(%rsp), {{%xmm[0-9][0-9]*}}, {{%xmm[0-9][0-9]*}} {{.*#+}} 16-byte Folded Reload |
| 1151 | %1 = tail call <2 x i64> asm sideeffect "nop", "=x,~{xmm2},~{xmm3},~{xmm4},~{xmm5},~{xmm6},~{xmm7},~{xmm8},~{xmm9},~{xmm10},~{xmm11},~{xmm12},~{xmm13},~{xmm14},~{xmm15},~{flags}"() |
| 1152 | %2 = xor <16 x i8> %a0, %a1 |
| 1153 | ; add forces execution domain |
| 1154 | %3 = add <16 x i8> %2, <i8 1, i8 1, i8 1, i8 1, i8 1, i8 1, i8 1, i8 1, i8 1, i8 1, i8 1, i8 1, i8 1, i8 1, i8 1, i8 1> |
| 1155 | ret <16 x i8> %3 |
| 1156 | } |