summaryrefslogtreecommitdiff
path: root/test/CodeGen/X86
diff options
context:
space:
mode:
authorDimitry Andric <dim@FreeBSD.org>2016-11-25 19:05:59 +0000
committerDimitry Andric <dim@FreeBSD.org>2016-11-25 19:05:59 +0000
commit6449741f4c1842221757c062f4abbae7bb524ba9 (patch)
tree5a2ca31d10f5ca2e8fb9c1ade59c306526de8329 /test/CodeGen/X86
parent60a9e02f5509f102642299ee408fab21b2ee30e4 (diff)
Notes
Diffstat (limited to 'test/CodeGen/X86')
-rw-r--r--test/CodeGen/X86/avx-vbroadcast.ll61
-rw-r--r--test/CodeGen/X86/branchfolding-undef.mir29
-rw-r--r--test/CodeGen/X86/no-and8ri8.ll18
-rw-r--r--test/CodeGen/X86/pr30298.ll43
4 files changed, 151 insertions, 0 deletions
diff --git a/test/CodeGen/X86/avx-vbroadcast.ll b/test/CodeGen/X86/avx-vbroadcast.ll
index b312be9aa6b2..534670795894 100644
--- a/test/CodeGen/X86/avx-vbroadcast.ll
+++ b/test/CodeGen/X86/avx-vbroadcast.ll
@@ -546,3 +546,64 @@ define <4 x double> @splat_concat4(double* %p) {
%6 = shufflevector <2 x double> %3, <2 x double> %5, <4 x i32> <i32 0, i32 1, i32 2, i32 3>
ret <4 x double> %6
}
+
+;
+; When VBROADCAST replaces an existing load, ensure it still respects lifetime dependencies.
+;
+define float @broadcast_lifetime() nounwind {
+; X32-LABEL: broadcast_lifetime:
+; X32: ## BB#0:
+; X32-NEXT: pushl %esi
+; X32-NEXT: subl $56, %esp
+; X32-NEXT: leal {{[0-9]+}}(%esp), %esi
+; X32-NEXT: movl %esi, (%esp)
+; X32-NEXT: calll _gfunc
+; X32-NEXT: vbroadcastss {{[0-9]+}}(%esp), %xmm0
+; X32-NEXT: vmovaps %xmm0, {{[0-9]+}}(%esp) ## 16-byte Spill
+; X32-NEXT: movl %esi, (%esp)
+; X32-NEXT: calll _gfunc
+; X32-NEXT: vbroadcastss {{[0-9]+}}(%esp), %xmm0
+; X32-NEXT: vsubss {{[0-9]+}}(%esp), %xmm0, %xmm0 ## 16-byte Folded Reload
+; X32-NEXT: vmovss %xmm0, {{[0-9]+}}(%esp)
+; X32-NEXT: flds {{[0-9]+}}(%esp)
+; X32-NEXT: addl $56, %esp
+; X32-NEXT: popl %esi
+; X32-NEXT: retl
+;
+; X64-LABEL: broadcast_lifetime:
+; X64: ## BB#0:
+; X64-NEXT: subq $40, %rsp
+; X64-NEXT: leaq (%rsp), %rdi
+; X64-NEXT: callq _gfunc
+; X64-NEXT: vbroadcastss {{[0-9]+}}(%rsp), %xmm0
+; X64-NEXT: vmovaps %xmm0, {{[0-9]+}}(%rsp) ## 16-byte Spill
+; X64-NEXT: leaq (%rsp), %rdi
+; X64-NEXT: callq _gfunc
+; X64-NEXT: vbroadcastss {{[0-9]+}}(%rsp), %xmm0
+; X64-NEXT: vsubss {{[0-9]+}}(%rsp), %xmm0, %xmm0 ## 16-byte Folded Reload
+; X64-NEXT: addq $40, %rsp
+; X64-NEXT: retq
+ %1 = alloca <4 x float>, align 16
+ %2 = alloca <4 x float>, align 16
+ %3 = bitcast <4 x float>* %1 to i8*
+ %4 = bitcast <4 x float>* %2 to i8*
+
+ call void @llvm.lifetime.start(i64 16, i8* %3)
+ call void @gfunc(<4 x float>* %1)
+ %5 = load <4 x float>, <4 x float>* %1, align 16
+ call void @llvm.lifetime.end(i64 16, i8* %3)
+
+ call void @llvm.lifetime.start(i64 16, i8* %4)
+ call void @gfunc(<4 x float>* %2)
+ %6 = load <4 x float>, <4 x float>* %2, align 16
+ call void @llvm.lifetime.end(i64 16, i8* %4)
+
+ %7 = extractelement <4 x float> %5, i32 1
+ %8 = extractelement <4 x float> %6, i32 1
+ %9 = fsub float %8, %7
+ ret float %9
+}
+
+declare void @gfunc(<4 x float>*)
+declare void @llvm.lifetime.start(i64, i8*)
+declare void @llvm.lifetime.end(i64, i8*)
diff --git a/test/CodeGen/X86/branchfolding-undef.mir b/test/CodeGen/X86/branchfolding-undef.mir
new file mode 100644
index 000000000000..0da167b33257
--- /dev/null
+++ b/test/CodeGen/X86/branchfolding-undef.mir
@@ -0,0 +1,29 @@
+# RUN: llc -o - %s -march=x86 -run-pass branch-folder | FileCheck %s
+# Test that tail merging drops undef flags that aren't present on all
+# instructions to be merged.
+--- |
+ define void @func() { ret void }
+...
+---
+# CHECK-LABEL: name: func
+# CHECK: bb.1:
+# CHECK: %eax = MOV32ri 2
+# CHECK-NOT: RET
+# CHECK: bb.2:
+# CHECK-NOT: RET 0, undef %eax
+# CHECK: RET 0, %eax
+name: func
+tracksRegLiveness: true
+body: |
+ bb.0:
+ successors: %bb.1, %bb.2
+ JE_1 %bb.1, implicit undef %eflags
+ JMP_1 %bb.2
+
+ bb.1:
+ %eax = MOV32ri 2
+ RET 0, %eax
+
+ bb.2:
+ RET 0, undef %eax
+...
diff --git a/test/CodeGen/X86/no-and8ri8.ll b/test/CodeGen/X86/no-and8ri8.ll
new file mode 100644
index 000000000000..57f33226602e
--- /dev/null
+++ b/test/CodeGen/X86/no-and8ri8.ll
@@ -0,0 +1,18 @@
+; RUN: llc -mtriple=x86_64-pc-linux -mattr=+avx512f --show-mc-encoding < %s | FileCheck %s
+
+declare i1 @bar()
+
+; CHECK-LABEL: @foo
+; CHECK-NOT: andb {{.*}} # encoding: [0x82,
+define i1 @foo(i1 %i) nounwind {
+entry:
+ br i1 %i, label %if, label %else
+
+if:
+ %r = call i1 @bar()
+ br label %else
+
+else:
+ %ret = phi i1 [%r, %if], [true, %entry]
+ ret i1 %ret
+}
diff --git a/test/CodeGen/X86/pr30298.ll b/test/CodeGen/X86/pr30298.ll
new file mode 100644
index 000000000000..1e6dad0b20d1
--- /dev/null
+++ b/test/CodeGen/X86/pr30298.ll
@@ -0,0 +1,43 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
+; RUN: llc -mtriple=i386-pc-linux-gnu -mattr=+sse < %s | FileCheck %s
+
+@c = external global i32*, align 8
+
+define void @mul_2xi8(i8* nocapture readonly %a, i8* nocapture readonly %b, i64 %index) nounwind {
+; CHECK-LABEL: mul_2xi8:
+; CHECK: # BB#0: # %entry
+; CHECK-NEXT: pushl %ebx
+; CHECK-NEXT: pushl %edi
+; CHECK-NEXT: pushl %esi
+; CHECK-NEXT: movl {{[0-9]+}}(%esp), %eax
+; CHECK-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; CHECK-NEXT: movl {{[0-9]+}}(%esp), %edx
+; CHECK-NEXT: movl c, %esi
+; CHECK-NEXT: movzbl 1(%edx,%ecx), %edi
+; CHECK-NEXT: movzbl (%edx,%ecx), %edx
+; CHECK-NEXT: movzbl 1(%eax,%ecx), %ebx
+; CHECK-NEXT: movzbl (%eax,%ecx), %eax
+; CHECK-NEXT: imull %edx, %eax
+; CHECK-NEXT: imull %edi, %ebx
+; CHECK-NEXT: movl %ebx, 4(%esi,%ecx,4)
+; CHECK-NEXT: movl %eax, (%esi,%ecx,4)
+; CHECK-NEXT: popl %esi
+; CHECK-NEXT: popl %edi
+; CHECK-NEXT: popl %ebx
+; CHECK-NEXT: retl
+entry:
+ %pre = load i32*, i32** @c
+ %tmp6 = getelementptr inbounds i8, i8* %a, i64 %index
+ %tmp7 = bitcast i8* %tmp6 to <2 x i8>*
+ %wide.load = load <2 x i8>, <2 x i8>* %tmp7, align 1
+ %tmp8 = zext <2 x i8> %wide.load to <2 x i32>
+ %tmp10 = getelementptr inbounds i8, i8* %b, i64 %index
+ %tmp11 = bitcast i8* %tmp10 to <2 x i8>*
+ %wide.load17 = load <2 x i8>, <2 x i8>* %tmp11, align 1
+ %tmp12 = zext <2 x i8> %wide.load17 to <2 x i32>
+ %tmp13 = mul nuw nsw <2 x i32> %tmp12, %tmp8
+ %tmp14 = getelementptr inbounds i32, i32* %pre, i64 %index
+ %tmp15 = bitcast i32* %tmp14 to <2 x i32>*
+ store <2 x i32> %tmp13, <2 x i32>* %tmp15, align 4
+ ret void
+}