diff options
Diffstat (limited to 'test/Transforms/ScalarRepl')
| -rw-r--r-- | test/Transforms/ScalarRepl/2008-01-29-PromoteBug.ll | 2 | ||||
| -rw-r--r-- | test/Transforms/ScalarRepl/2008-06-05-loadstore-agg.ll | 4 | ||||
| -rw-r--r-- | test/Transforms/ScalarRepl/dg.exp | 2 | ||||
| -rw-r--r-- | test/Transforms/ScalarRepl/inline-vector.ll | 53 | ||||
| -rw-r--r-- | test/Transforms/ScalarRepl/only-memcpy-uses.ll | 27 | ||||
| -rw-r--r-- | test/Transforms/ScalarRepl/union-pointer.ll | 2 | ||||
| -rw-r--r-- | test/Transforms/ScalarRepl/vector_promote.ll | 167 |
7 files changed, 251 insertions, 6 deletions
diff --git a/test/Transforms/ScalarRepl/2008-01-29-PromoteBug.ll b/test/Transforms/ScalarRepl/2008-01-29-PromoteBug.ll index d799bd77e458..8bc4ff0b3ffc 100644 --- a/test/Transforms/ScalarRepl/2008-01-29-PromoteBug.ll +++ b/test/Transforms/ScalarRepl/2008-01-29-PromoteBug.ll @@ -1,6 +1,6 @@ ; RUN: opt < %s -scalarrepl -instcombine -S | grep {ret i8 17} ; rdar://5707076 -target datalayout = "e-p:32:32:32-i1:8:8-i8:8:8-i16:16:16-i32:32:32-i64:32:64-f32:32:32-f64:32:64-v64:64:64-v128:128:128-a0:0:64-f80:128:128" +target datalayout = "e-p:32:32:32-i1:8:8-i8:8:8-i16:16:16-i32:32:32-i64:32:64-f32:32:32-f64:32:64-v64:64:64-v128:128:128-a0:0:64-f80:128:128-n8:16:32" target triple = "i386-apple-darwin9.1.0" %struct.T = type <{ i8, [3 x i8] }> diff --git a/test/Transforms/ScalarRepl/2008-06-05-loadstore-agg.ll b/test/Transforms/ScalarRepl/2008-06-05-loadstore-agg.ll index 87a08b7eaaf2..ce70a1b13b81 100644 --- a/test/Transforms/ScalarRepl/2008-06-05-loadstore-agg.ll +++ b/test/Transforms/ScalarRepl/2008-06-05-loadstore-agg.ll @@ -13,7 +13,7 @@ define i32 @foo() { %res2 = insertvalue { i32, i32 } %res1, i32 2, 1 ; <{ i32, i32 }> [#uses=1] ; And store it store { i32, i32 } %res2, { i32, i32 }* %target - ; Actually use %target, so it doesn't get removed alltogether + ; Actually use %target, so it doesn't get removed altogether %ptr = getelementptr { i32, i32 }* %target, i32 0, i32 0 %val = load i32* %ptr ret i32 %val @@ -26,7 +26,7 @@ define i32 @bar() { %res2 = insertvalue [ 2 x i32 ] %res1, i32 2, 1 ; <{ i32, i32 }> [#uses=1] ; And store it store [ 2 x i32 ] %res2, [ 2 x i32 ]* %target - ; Actually use %target, so it doesn't get removed alltogether + ; Actually use %target, so it doesn't get removed altogether %ptr = getelementptr [ 2 x i32 ]* %target, i32 0, i32 0 %val = load i32* %ptr ret i32 %val diff --git a/test/Transforms/ScalarRepl/dg.exp b/test/Transforms/ScalarRepl/dg.exp index f2005891a59a..39954d8a498d 100644 --- a/test/Transforms/ScalarRepl/dg.exp +++ b/test/Transforms/ScalarRepl/dg.exp @@ -1,3 +1,3 @@ load_lib llvm.exp -RunLLVMTests [lsort [glob -nocomplain $srcdir/$subdir/*.{ll,c,cpp}]] +RunLLVMTests [lsort [glob -nocomplain $srcdir/$subdir/*.{ll}]] diff --git a/test/Transforms/ScalarRepl/inline-vector.ll b/test/Transforms/ScalarRepl/inline-vector.ll new file mode 100644 index 000000000000..2f51cc7cf59c --- /dev/null +++ b/test/Transforms/ScalarRepl/inline-vector.ll @@ -0,0 +1,53 @@ +; RUN: opt < %s -scalarrepl -S | FileCheck %s +; RUN: opt < %s -scalarrepl-ssa -S | FileCheck %s +target datalayout = "e-p:32:32:32-i1:8:32-i8:8:32-i16:16:32-i32:32:32-i64:32:32-f32:32:32-f64:32:32-v64:64:64-v128:128:128-a0:0:32-n32" +target triple = "thumbv7-apple-darwin10.0.0" + +%struct.Vector4 = type { float, float, float, float } +@f.vector = internal constant %struct.Vector4 { float 1.000000e+00, float 2.000000e+00, float 3.000000e+00, float 4.000000e+00 }, align 16 + +; CHECK: define void @f +; CHECK-NOT: alloca +; CHECK: phi <4 x float> + +define void @f() nounwind ssp { +entry: + %i = alloca i32, align 4 + %vector = alloca %struct.Vector4, align 16 + %agg.tmp = alloca %struct.Vector4, align 16 + %tmp = bitcast %struct.Vector4* %vector to i8* + call void @llvm.memcpy.p0i8.p0i8.i32(i8* %tmp, i8* bitcast (%struct.Vector4* @f.vector to i8*), i32 16, i32 16, i1 false) + br label %for.cond + +for.cond: ; preds = %for.body, %entry + %storemerge = phi i32 [ 0, %entry ], [ %inc, %for.body ] + store i32 %storemerge, i32* %i, align 4 + %cmp = icmp slt i32 %storemerge, 1000000 + br i1 %cmp, label %for.body, label %for.end + +for.body: ; preds = %for.cond + %tmp2 = bitcast %struct.Vector4* %agg.tmp to i8* + %tmp3 = bitcast %struct.Vector4* %vector to i8* + call void @llvm.memcpy.p0i8.p0i8.i32(i8* %tmp2, i8* %tmp3, i32 16, i32 16, i1 false) + %0 = bitcast %struct.Vector4* %agg.tmp to [2 x i64]* + %1 = load [2 x i64]* %0, align 16 + %tmp2.i = extractvalue [2 x i64] %1, 0 + %tmp3.i = zext i64 %tmp2.i to i128 + %tmp10.i = bitcast i128 %tmp3.i to <4 x float> + %sub.i.i = fsub <4 x float> <float -0.000000e+00, float -0.000000e+00, float -0.000000e+00, float -0.000000e+00>, %tmp10.i + %2 = bitcast %struct.Vector4* %vector to <4 x float>* + store <4 x float> %sub.i.i, <4 x float>* %2, align 16 + %tmp4 = load i32* %i, align 4 + %inc = add nsw i32 %tmp4, 1 + br label %for.cond + +for.end: ; preds = %for.cond + %x = getelementptr inbounds %struct.Vector4* %vector, i32 0, i32 0 + %tmp5 = load float* %x, align 16 + %conv = fpext float %tmp5 to double + %call = call i32 (...)* @printf(double %conv) nounwind + ret void +} + +declare void @llvm.memcpy.p0i8.p0i8.i32(i8* nocapture, i8* nocapture, i32, i32, i1) nounwind +declare i32 @printf(...) diff --git a/test/Transforms/ScalarRepl/only-memcpy-uses.ll b/test/Transforms/ScalarRepl/only-memcpy-uses.ll new file mode 100644 index 000000000000..cfb88bd80d60 --- /dev/null +++ b/test/Transforms/ScalarRepl/only-memcpy-uses.ll @@ -0,0 +1,27 @@ +; RUN: opt < %s -scalarrepl -S | FileCheck %s +target datalayout = "e-p:64:64:64-i1:8:8-i8:8:8-i16:16:16-i32:32:32-i64:64:64-f32:32:32-f64:64:64-v64:64:64-v128:128:128-a0:0:64-s0:64:64-f80:128:128-n8:16:32:64" +target triple = "x86_64-apple-darwin10.0.0" + +%struct.S = type { [12 x i32] } + +; CHECK: @bar4 +define void @bar4(%struct.S* byval %s) nounwind ssp { +entry: +; CHECK: alloca +; CHECK-NOT: load +; CHECK: memcpy + %t = alloca %struct.S, align 4 + %agg.tmp = alloca %struct.S, align 4 + %tmp = bitcast %struct.S* %t to i8* + %tmp1 = bitcast %struct.S* %s to i8* + call void @llvm.memcpy.p0i8.p0i8.i64(i8* %tmp, i8* %tmp1, i64 48, i32 4, i1 false) + %tmp2 = bitcast %struct.S* %agg.tmp to i8* + %tmp3 = bitcast %struct.S* %t to i8* + call void @llvm.memcpy.p0i8.p0i8.i64(i8* %tmp2, i8* %tmp3, i64 48, i32 4, i1 false) + %call = call i32 (...)* @bazz(%struct.S* byval %agg.tmp) + ret void +} + +declare void @llvm.memcpy.p0i8.p0i8.i64(i8* nocapture, i8* nocapture, i64, i32, i1) nounwind + +declare i32 @bazz(...) diff --git a/test/Transforms/ScalarRepl/union-pointer.ll b/test/Transforms/ScalarRepl/union-pointer.ll index fe702fa21772..ea4ec14e5621 100644 --- a/test/Transforms/ScalarRepl/union-pointer.ll +++ b/test/Transforms/ScalarRepl/union-pointer.ll @@ -3,7 +3,7 @@ ; RUN: not grep alloca ; RUN: opt < %s -scalarrepl -S | grep {ret i8} -target datalayout = "e-p:32:32" +target datalayout = "e-p:32:32-n8:16:32" target triple = "i686-apple-darwin8.7.2" %struct.Val = type { i32*, i32 } diff --git a/test/Transforms/ScalarRepl/vector_promote.ll b/test/Transforms/ScalarRepl/vector_promote.ll index 37cb49f539d6..c51ef109c0b2 100644 --- a/test/Transforms/ScalarRepl/vector_promote.ll +++ b/test/Transforms/ScalarRepl/vector_promote.ll @@ -94,7 +94,172 @@ define i64 @test6(<2 x float> %X) { %tmp = load i64* %P ret i64 %tmp ; CHECK: @test6 -; CHECK: bitcast <2 x float> %X to <1 x i64> +; CHECK: bitcast <2 x float> %X to i64 ; CHECK: ret i64 } +define float @test7(<4 x float> %x) { + %a = alloca <4 x float> + store <4 x float> %x, <4 x float>* %a + %p = bitcast <4 x float>* %a to <2 x float>* + %b = load <2 x float>* %p + %q = getelementptr <4 x float>* %a, i32 0, i32 2 + %c = load float* %q + ret float %c +; CHECK: @test7 +; CHECK-NOT: alloca +; CHECK: bitcast <4 x float> %x to <2 x double> +; CHECK-NEXT: extractelement <2 x double> +; CHECK-NEXT: bitcast double %tmp4 to <2 x float> +; CHECK-NEXT: extractelement <4 x float> +} + +define void @test8(<4 x float> %x, <2 x float> %y) { + %a = alloca <4 x float> + store <4 x float> %x, <4 x float>* %a + %p = bitcast <4 x float>* %a to <2 x float>* + store <2 x float> %y, <2 x float>* %p + ret void +; CHECK: @test8 +; CHECK-NOT: alloca +; CHECK: bitcast <4 x float> %x to <2 x double> +; CHECK-NEXT: bitcast <2 x float> %y to double +; CHECK-NEXT: insertelement <2 x double> +; CHECK-NEXT: bitcast <2 x double> %tmp2 to <4 x float> +} + +define i256 @test9(<4 x i256> %x) { + %a = alloca <4 x i256> + store <4 x i256> %x, <4 x i256>* %a + %p = bitcast <4 x i256>* %a to <2 x i256>* + %b = load <2 x i256>* %p + %q = getelementptr <4 x i256>* %a, i32 0, i32 2 + %c = load i256* %q + ret i256 %c +; CHECK: @test9 +; CHECK-NOT: alloca +; CHECK: bitcast <4 x i256> %x to <2 x i512> +; CHECK-NEXT: extractelement <2 x i512> +; CHECK-NEXT: bitcast i512 %tmp4 to <2 x i256> +; CHECK-NEXT: extractelement <4 x i256> +} + +define void @test10(<4 x i256> %x, <2 x i256> %y) { + %a = alloca <4 x i256> + store <4 x i256> %x, <4 x i256>* %a + %p = bitcast <4 x i256>* %a to <2 x i256>* + store <2 x i256> %y, <2 x i256>* %p + ret void +; CHECK: @test10 +; CHECK-NOT: alloca +; CHECK: bitcast <4 x i256> %x to <2 x i512> +; CHECK-NEXT: bitcast <2 x i256> %y to i512 +; CHECK-NEXT: insertelement <2 x i512> +; CHECK-NEXT: bitcast <2 x i512> %tmp2 to <4 x i256> +} + +%union.v = type { <2 x i64> } + +define void @test11(<2 x i64> %x) { + %a = alloca %union.v + %p = getelementptr inbounds %union.v* %a, i32 0, i32 0 + store <2 x i64> %x, <2 x i64>* %p, align 16 + %q = getelementptr inbounds %union.v* %a, i32 0, i32 0 + %r = bitcast <2 x i64>* %q to <4 x float>* + %b = load <4 x float>* %r, align 16 + ret void +; CHECK: @test11 +; CHECK-NOT: alloca +} + +define void @test12() { +entry: + %a = alloca <64 x i8>, align 64 + store <64 x i8> undef, <64 x i8>* %a, align 64 + %p = bitcast <64 x i8>* %a to <16 x i8>* + %0 = load <16 x i8>* %p, align 64 + store <16 x i8> undef, <16 x i8>* %p, align 64 + %q = bitcast <16 x i8>* %p to <64 x i8>* + %1 = load <64 x i8>* %q, align 64 + ret void +; CHECK: @test12 +; CHECK-NOT: alloca +; CHECK: extractelement <4 x i128> +; CHECK: insertelement <4 x i128> +} + +define float @test13(<4 x float> %x, <2 x i32> %y) { + %a = alloca <4 x float> + store <4 x float> %x, <4 x float>* %a + %p = bitcast <4 x float>* %a to <2 x float>* + %b = load <2 x float>* %p + %q = getelementptr <4 x float>* %a, i32 0, i32 2 + %c = load float* %q + %r = bitcast <4 x float>* %a to <2 x i32>* + store <2 x i32> %y, <2 x i32>* %r + ret float %c +; CHECK: @test13 +; CHECK-NOT: alloca +; CHECK: bitcast <4 x float> %x to i128 +} + +define <3 x float> @test14(<3 x float> %x) { +entry: + %x.addr = alloca <3 x float>, align 16 + %r = alloca <3 x i32>, align 16 + %extractVec = shufflevector <3 x float> %x, <3 x float> undef, <4 x i32> <i32 0, i32 1, i32 2, i32 undef> + %storetmp = bitcast <3 x float>* %x.addr to <4 x float>* + store <4 x float> %extractVec, <4 x float>* %storetmp, align 16 + %tmp = load <3 x float>* %x.addr, align 16 + %cmp = fcmp une <3 x float> %tmp, zeroinitializer + %sext = sext <3 x i1> %cmp to <3 x i32> + %and = and <3 x i32> <i32 1065353216, i32 1065353216, i32 1065353216>, %sext + %extractVec1 = shufflevector <3 x i32> %and, <3 x i32> undef, <4 x i32> <i32 0, i32 1, i32 2, i32 undef> + %storetmp2 = bitcast <3 x i32>* %r to <4 x i32>* + store <4 x i32> %extractVec1, <4 x i32>* %storetmp2, align 16 + %tmp3 = load <3 x i32>* %r, align 16 + %0 = bitcast <3 x i32> %tmp3 to <3 x float> + %tmp4 = load <3 x float>* %x.addr, align 16 + ret <3 x float> %tmp4 +; CHECK: @test14 +; CHECK-NOT: alloca +; CHECK: shufflevector <4 x i32> %extractVec1, <4 x i32> undef, <3 x i32> <i32 0, i32 1, i32 2> +} + +define void @test15(<3 x i64>* sret %agg.result, <3 x i64> %x, <3 x i64> %min) { +entry: + %x.addr = alloca <3 x i64>, align 32 + %min.addr = alloca <3 x i64>, align 32 + %extractVec = shufflevector <3 x i64> %x, <3 x i64> undef, <4 x i32> <i32 0, i32 1, i32 2, i32 undef> + %storetmp = bitcast <3 x i64>* %x.addr to <4 x i64>* + store <4 x i64> %extractVec, <4 x i64>* %storetmp, align 32 + %extractVec1 = shufflevector <3 x i64> %min, <3 x i64> undef, <4 x i32> <i32 0, i32 1, i32 2, i32 undef> + %storetmp2 = bitcast <3 x i64>* %min.addr to <4 x i64>* + store <4 x i64> %extractVec1, <4 x i64>* %storetmp2, align 32 + %tmp = load <3 x i64>* %x.addr + %tmp5 = extractelement <3 x i64> %tmp, i32 0 + %tmp11 = insertelement <3 x i64> %tmp, i64 %tmp5, i32 0 + store <3 x i64> %tmp11, <3 x i64>* %x.addr + %tmp30 = load <3 x i64>* %x.addr, align 32 + store <3 x i64> %tmp30, <3 x i64>* %agg.result + ret void +; CHECK: @test15 +; CHECK-NOT: alloca +; CHECK: shufflevector <4 x i64> %tmpV2, <4 x i64> undef, <3 x i32> <i32 0, i32 1, i32 2> +} + +define <4 x float> @test16(<4 x float> %x, i64 %y0, i64 %y1) { +entry: + %tmp8 = bitcast <4 x float> undef to <2 x double> + %tmp9 = bitcast i64 %y0 to double + %tmp10 = insertelement <2 x double> %tmp8, double %tmp9, i32 0 + %tmp11 = bitcast <2 x double> %tmp10 to <4 x float> + %tmp3 = bitcast <4 x float> %tmp11 to <2 x double> + %tmp4 = bitcast i64 %y1 to double + %tmp5 = insertelement <2 x double> %tmp3, double %tmp4, i32 1 + %tmp6 = bitcast <2 x double> %tmp5 to <4 x float> + ret <4 x float> %tmp6 +; CHECK: @test16 +; CHECK-NOT: alloca +; CHECK: bitcast <4 x float> %tmp11 to <2 x double> +} |
