aboutsummaryrefslogtreecommitdiff
path: root/test
diff options
context:
space:
mode:
authorDimitry Andric <dim@FreeBSD.org>2016-02-24 21:32:58 +0000
committerDimitry Andric <dim@FreeBSD.org>2016-02-24 21:32:58 +0000
commitd9c9bd8485071afb22adcd2bb08f6a8e5e587ed6 (patch)
tree53c036d35173ba19f107d9afe07678667d270bee /test
parent3f4bde29a30d8c43db5cbe8f5541ebc5d1fdc6af (diff)
Notes
Diffstat (limited to 'test')
-rw-r--r--test/CodeGen/AArch64/aarch64-dynamic-stack-layout.ll2
-rw-r--r--test/CodeGen/AArch64/arm64-shrink-wrapping.ll85
-rw-r--r--test/CodeGen/ARM/Windows/alloca.ll4
-rw-r--r--test/CodeGen/PowerPC/pr26690.ll118
-rw-r--r--test/CodeGen/X86/i386-tlscall-fastregalloc.ll26
-rw-r--r--test/CodeGen/X86/tls-shrink-wrapping.ll60
6 files changed, 293 insertions, 2 deletions
diff --git a/test/CodeGen/AArch64/aarch64-dynamic-stack-layout.ll b/test/CodeGen/AArch64/aarch64-dynamic-stack-layout.ll
index 1820b8163a90..90093f94d0ad 100644
--- a/test/CodeGen/AArch64/aarch64-dynamic-stack-layout.ll
+++ b/test/CodeGen/AArch64/aarch64-dynamic-stack-layout.ll
@@ -522,10 +522,10 @@ bb1:
; CHECK-LABEL: realign_conditional2
; Extra realignment in the prologue (performance issue).
-; CHECK: tbz {{.*}} .[[LABEL:.*]]
; CHECK: sub x9, sp, #32 // =32
; CHECK: and sp, x9, #0xffffffffffffffe0
; CHECK: mov x19, sp
+; CHECK: tbz {{.*}} .[[LABEL:.*]]
; Stack is realigned in a non-entry BB.
; CHECK: sub [[REG:x[01-9]+]], sp, #64
; CHECK: and sp, [[REG]], #0xffffffffffffffe0
diff --git a/test/CodeGen/AArch64/arm64-shrink-wrapping.ll b/test/CodeGen/AArch64/arm64-shrink-wrapping.ll
index 2ecd66ddf5d4..4d751f501d4a 100644
--- a/test/CodeGen/AArch64/arm64-shrink-wrapping.ll
+++ b/test/CodeGen/AArch64/arm64-shrink-wrapping.ll
@@ -630,3 +630,88 @@ loop2b: ; preds = %loop1
end:
ret void
}
+
+; Don't do shrink-wrapping when we need to re-align the stack pointer.
+; See bug 26642.
+; CHECK-LABEL: stack_realign:
+; CHECK-NOT: lsl w[[LSL1:[0-9]+]], w0, w1
+; CHECK-NOT: lsl w[[LSL2:[0-9]+]], w1, w0
+; CHECK: stp x29, x30, [sp, #-16]!
+; CHECK: mov x29, sp
+; CHECK: sub x{{[0-9]+}}, sp, #16
+; CHECK-DAG: lsl w[[LSL1:[0-9]+]], w0, w1
+; CHECK-DAG: lsl w[[LSL2:[0-9]+]], w1, w0
+; CHECK-DAG: str w[[LSL1]],
+; CHECK-DAG: str w[[LSL2]],
+
+define i32 @stack_realign(i32 %a, i32 %b, i32* %ptr1, i32* %ptr2) {
+ %tmp = alloca i32, align 32
+ %shl1 = shl i32 %a, %b
+ %shl2 = shl i32 %b, %a
+ %tmp2 = icmp slt i32 %a, %b
+ br i1 %tmp2, label %true, label %false
+
+true:
+ store i32 %a, i32* %tmp, align 4
+ %tmp4 = load i32, i32* %tmp
+ br label %false
+
+false:
+ %tmp.0 = phi i32 [ %tmp4, %true ], [ %a, %0 ]
+ store i32 %shl1, i32* %ptr1
+ store i32 %shl2, i32* %ptr2
+ ret i32 %tmp.0
+}
+
+; Re-aligned stack pointer with all caller-save regs live. See bug
+; 26642. In this case we currently avoid shrink wrapping because
+; ensuring we have a scratch register to re-align the stack pointer is
+; too complicated. Output should be the same for both enabled and
+; disabled shrink wrapping.
+; CHECK-LABEL: stack_realign2:
+; CHECK: stp {{x[0-9]+}}, {{x[0-9]+}}, [sp, #-{{[0-9]+}}]!
+; CHECK: add x29, sp, #{{[0-9]+}}
+; CHECK: lsl {{w[0-9]+}}, w0, w1
+
+define void @stack_realign2(i32 %a, i32 %b, i32* %ptr1, i32* %ptr2, i32* %ptr3, i32* %ptr4, i32* %ptr5, i32* %ptr6) {
+ %tmp = alloca i32, align 32
+ %tmp1 = shl i32 %a, %b
+ %tmp2 = shl i32 %b, %a
+ %tmp3 = lshr i32 %a, %b
+ %tmp4 = lshr i32 %b, %a
+ %tmp5 = add i32 %b, %a
+ %tmp6 = sub i32 %b, %a
+ %tmp7 = add i32 %tmp1, %tmp2
+ %tmp8 = sub i32 %tmp2, %tmp3
+ %tmp9 = add i32 %tmp3, %tmp4
+ %tmp10 = add i32 %tmp4, %tmp5
+ %cmp = icmp slt i32 %a, %b
+ br i1 %cmp, label %true, label %false
+
+true:
+ store i32 %a, i32* %tmp, align 4
+ call void asm sideeffect "nop", "~{x19},~{x20},~{x21},~{x22},~{x23},~{x24},~{x25},~{x26},~{x27},~{x28}"() nounwind
+ br label %false
+
+false:
+ store i32 %tmp1, i32* %ptr1, align 4
+ store i32 %tmp2, i32* %ptr2, align 4
+ store i32 %tmp3, i32* %ptr3, align 4
+ store i32 %tmp4, i32* %ptr4, align 4
+ store i32 %tmp5, i32* %ptr5, align 4
+ store i32 %tmp6, i32* %ptr6, align 4
+ %idx1 = getelementptr inbounds i32, i32* %ptr1, i64 1
+ store i32 %a, i32* %idx1, align 4
+ %idx2 = getelementptr inbounds i32, i32* %ptr1, i64 2
+ store i32 %b, i32* %idx2, align 4
+ %idx3 = getelementptr inbounds i32, i32* %ptr1, i64 3
+ store i32 %tmp7, i32* %idx3, align 4
+ %idx4 = getelementptr inbounds i32, i32* %ptr1, i64 4
+ store i32 %tmp8, i32* %idx4, align 4
+ %idx5 = getelementptr inbounds i32, i32* %ptr1, i64 5
+ store i32 %tmp9, i32* %idx5, align 4
+ %idx6 = getelementptr inbounds i32, i32* %ptr1, i64 6
+ store i32 %tmp10, i32* %idx6, align 4
+
+ ret void
+}
diff --git a/test/CodeGen/ARM/Windows/alloca.ll b/test/CodeGen/ARM/Windows/alloca.ll
index 6a3d002ab3b3..0f20ffbd36db 100644
--- a/test/CodeGen/ARM/Windows/alloca.ll
+++ b/test/CodeGen/ARM/Windows/alloca.ll
@@ -13,7 +13,9 @@ entry:
}
; CHECK: bl num_entries
-; CHECK: movs [[R1:r[0-9]+]], #7
+; Any register is actually valid here, but turns out we use lr,
+; because we do not have the kill flag on R0.
+; CHECK: mov.w [[R1:lr]], #7
; CHECK: add.w [[R0:r[0-9]+]], [[R1]], [[R0]], lsl #2
; CHECK: bic [[R0]], [[R0]], #7
; CHECK: lsrs r4, [[R0]], #2
diff --git a/test/CodeGen/PowerPC/pr26690.ll b/test/CodeGen/PowerPC/pr26690.ll
new file mode 100644
index 000000000000..3e7662409d51
--- /dev/null
+++ b/test/CodeGen/PowerPC/pr26690.ll
@@ -0,0 +1,118 @@
+; RUN: llc -mcpu=pwr8 -mtriple=powerpc64le-unknown-linux-gnu < %s | FileCheck %s
+
+%struct.anon = type { %struct.anon.0, %struct.anon.1 }
+%struct.anon.0 = type { i32 }
+%struct.anon.1 = type { i32 }
+
+@i = common global i32 0, align 4
+@b = common global i32* null, align 8
+@c = common global i32 0, align 4
+@a = common global i32 0, align 4
+@h = common global i32 0, align 4
+@g = common global i32 0, align 4
+@j = common global i32 0, align 4
+@f = common global %struct.anon zeroinitializer, align 4
+@d = common global i32 0, align 4
+@e = common global i32 0, align 4
+
+; Function Attrs: norecurse nounwind
+define signext i32 @fn1(i32* nocapture %p1, i32 signext %p2, i32* nocapture %p3) {
+entry:
+ %0 = load i32, i32* @i, align 4, !tbaa !1
+ %cond = icmp eq i32 %0, 8
+ br i1 %cond, label %if.end16, label %while.cond.preheader
+
+while.cond.preheader: ; preds = %entry
+ %1 = load i32*, i32** @b, align 8, !tbaa !5
+ %2 = load i32, i32* %1, align 4, !tbaa !1
+ %tobool18 = icmp eq i32 %2, 0
+ br i1 %tobool18, label %while.end, label %while.body.lr.ph
+
+while.body.lr.ph: ; preds = %while.cond.preheader
+ %.pre = load i32, i32* @c, align 4, !tbaa !1
+ br label %while.body
+
+while.body: ; preds = %while.body.backedge, %while.body.lr.ph
+ switch i32 %.pre, label %while.body.backedge [
+ i32 0, label %sw.bb1
+ i32 8, label %sw.bb1
+ i32 6, label %sw.bb1
+ i32 24, label %while.cond.backedge
+ ]
+
+while.body.backedge: ; preds = %while.body, %while.cond.backedge
+ br label %while.body
+
+sw.bb1: ; preds = %while.body, %while.body, %while.body
+ store i32 2, i32* @a, align 4, !tbaa !1
+ br label %while.cond.backedge
+
+while.cond.backedge: ; preds = %while.body, %sw.bb1
+ store i32 4, i32* @a, align 4, !tbaa !1
+ %.pre19 = load i32, i32* %1, align 4, !tbaa !1
+ %tobool = icmp eq i32 %.pre19, 0
+ br i1 %tobool, label %while.end.loopexit, label %while.body.backedge
+
+while.end.loopexit: ; preds = %while.cond.backedge
+ br label %while.end
+
+while.end: ; preds = %while.end.loopexit, %while.cond.preheader
+ %3 = load i32, i32* @h, align 4, !tbaa !1
+ %mul = mul nsw i32 %0, %3
+ %4 = load i32, i32* @g, align 4, !tbaa !1
+ %mul4 = mul nsw i32 %mul, %4
+ store i32 %mul4, i32* @j, align 4, !tbaa !1
+ %5 = load i32, i32* getelementptr inbounds (%struct.anon, %struct.anon* @f, i64 0, i32 0, i32 0), align 4, !tbaa !7
+ %tobool5 = icmp eq i32 %5, 0
+ br i1 %tobool5, label %if.end, label %if.then
+
+if.then: ; preds = %while.end
+ %div = sdiv i32 %5, %mul
+ store i32 %div, i32* @g, align 4, !tbaa !1
+ br label %if.end
+
+if.end: ; preds = %while.end, %if.then
+ %6 = phi i32 [ %4, %while.end ], [ %div, %if.then ]
+ %7 = load i32, i32* getelementptr inbounds (%struct.anon, %struct.anon* @f, i64 0, i32 1, i32 0), align 4, !tbaa !10
+ %tobool7 = icmp ne i32 %7, 0
+ %tobool8 = icmp ne i32 %mul4, 0
+ %or.cond = and i1 %tobool7, %tobool8
+ %tobool10 = icmp ne i32 %0, 0
+ %or.cond17 = and i1 %or.cond, %tobool10
+ br i1 %or.cond17, label %if.then11, label %if.end13
+
+if.then11: ; preds = %if.end
+ store i32 %3, i32* @d, align 4, !tbaa !1
+ %8 = load i32, i32* @e, align 4, !tbaa !1
+ store i32 %8, i32* %p3, align 4, !tbaa !1
+ %.pre20 = load i32, i32* @g, align 4, !tbaa !1
+ br label %if.end13
+
+if.end13: ; preds = %if.then11, %if.end
+ %9 = phi i32 [ %.pre20, %if.then11 ], [ %6, %if.end ]
+ %tobool14 = icmp eq i32 %9, 0
+ br i1 %tobool14, label %if.end16, label %if.then15
+
+if.then15: ; preds = %if.end13
+ store i32 %p2, i32* %p1, align 4, !tbaa !1
+ br label %if.end16
+
+if.end16: ; preds = %entry, %if.end13, %if.then15
+ ret i32 2
+}
+
+; CHECK: mfcr {{[0-9]+}}
+
+!llvm.ident = !{!0}
+
+!0 = !{!"clang version 3.9.0 (trunk 261520)"}
+!1 = !{!2, !2, i64 0}
+!2 = !{!"int", !3, i64 0}
+!3 = !{!"omnipotent char", !4, i64 0}
+!4 = !{!"Simple C/C++ TBAA"}
+!5 = !{!6, !6, i64 0}
+!6 = !{!"any pointer", !3, i64 0}
+!7 = !{!8, !2, i64 0}
+!8 = !{!"", !9, i64 0, !9, i64 4}
+!9 = !{!"", !2, i64 0}
+!10 = !{!8, !2, i64 4}
diff --git a/test/CodeGen/X86/i386-tlscall-fastregalloc.ll b/test/CodeGen/X86/i386-tlscall-fastregalloc.ll
new file mode 100644
index 000000000000..775c0c1b3784
--- /dev/null
+++ b/test/CodeGen/X86/i386-tlscall-fastregalloc.ll
@@ -0,0 +1,26 @@
+; RUN: llc %s -o - -O0 -regalloc=fast | FileCheck %s
+target datalayout = "e-m:o-p:32:32-f64:32:64-f80:128-n8:16:32-S128"
+target triple = "i386-apple-macosx10.10"
+
+@c = external global i8, align 1
+@p = thread_local global i8* null, align 4
+
+; Check that regalloc fast correctly preserves EAX that is set by the TLS call
+; until the actual use.
+; PR26485.
+;
+; CHECK-LABEL: f:
+; Get p.
+; CHECK: movl _p@{{[0-9a-zA-Z]+}}, [[P_ADDR:%[a-z]+]]
+; CHECK-NEXT: calll *([[P_ADDR]])
+; At this point eax contiains the address of p.
+; Load c address.
+; Make sure we do not clobber eax.
+; CHECK-NEXT: movl L_c{{[^,]*}}, [[C_ADDR:%e[b-z]x+]]
+; Store c address into p.
+; CHECK-NEXT: movl [[C_ADDR]], (%eax)
+define void @f() #0 {
+entry:
+ store i8* @c, i8** @p, align 4
+ ret void
+}
diff --git a/test/CodeGen/X86/tls-shrink-wrapping.ll b/test/CodeGen/X86/tls-shrink-wrapping.ll
new file mode 100644
index 000000000000..37c1754c0be8
--- /dev/null
+++ b/test/CodeGen/X86/tls-shrink-wrapping.ll
@@ -0,0 +1,60 @@
+; Testcase generated from the following code:
+; extern __thread int i;
+; void f();
+; int g(void) {
+; if (i) {
+; i = 0;
+; f();
+; }
+; return i;
+; }
+; We want to make sure that TLS variables are not accessed before
+; the stack frame is set up.
+
+; RUN: llc < %s -relocation-model=pic | FileCheck %s
+
+target datalayout = "e-m:e-i64:64-f80:128-n8:16:32:64-S128"
+target triple = "x86_64-unknown-freebsd11.0"
+
+@i = external thread_local global i32, align 4
+
+define i32 @g() #0 {
+entry:
+ %tmp = load i32, i32* @i, align 4
+ %tobool = icmp eq i32 %tmp, 0
+ br i1 %tobool, label %if.end, label %if.then
+
+if.then: ; preds = %entry
+ store i32 0, i32* @i, align 4
+ tail call void (...) @f() #2
+ %.pre = load i32, i32* @i, align 4
+ br label %if.end
+
+if.end: ; preds = %if.then, %entry
+ %tmp1 = phi i32 [ 0, %entry ], [ %.pre, %if.then ]
+ ret i32 %tmp1
+}
+
+; CHECK: g: # @g
+; CHECK-NEXT: .cfi_startproc
+; CHECK-NEXT: # BB#0: # %entry
+; CHECK-NEXT: pushq %rbp
+; CHECK-NEXT: .Ltmp0:
+; CHECK-NEXT: .cfi_def_cfa_offset 16
+; CHECK-NEXT: .Ltmp1:
+; CHECK-NEXT: .cfi_offset %rbp, -16
+; CHECK-NEXT: movq %rsp, %rbp
+; CHECK-NEXT: .Ltmp2:
+; CHECK-NEXT: .cfi_def_cfa_register %rbp
+; CHECK-NEXT: pushq %rbx
+; CHECK-NEXT: pushq %rax
+; CHECK-NEXT: .Ltmp3:
+; CHECK-NEXT: .cfi_offset %rbx, -24
+; CHECK-NEXT: data16
+; CHECK-NEXT: leaq i@TLSGD(%rip), %rdi
+
+declare void @f(...) #1
+
+attributes #0 = { nounwind uwtable "disable-tail-calls"="false" "less-precise-fpmad"="false" "no-frame-pointer-elim"="true" "no-frame-pointer-elim-non-leaf" "no-infs-fp-math"="false" "no-nans-fp-math"="false" "stack-protector-buffer-size"="8" "target-cpu"="x86-64" "target-features"="+fxsr,+mmx,+sse,+sse2" "unsafe-fp-math"="false" "use-soft-float"="false" }
+attributes #1 = { "disable-tail-calls"="false" "less-precise-fpmad"="false" "no-frame-pointer-elim"="true" "no-frame-pointer-elim-non-leaf" "no-infs-fp-math"="false" "no-nans-fp-math"="false" "stack-protector-buffer-size"="8" "target-cpu"="x86-64" "target-features"="+fxsr,+mmx,+sse,+sse2" "unsafe-fp-math"="false" "use-soft-float"="false" }
+attributes #2 = { nounwind }