diff options
Diffstat (limited to 'test/Transforms/LoopStrengthReduce')
12 files changed, 753 insertions, 29 deletions
diff --git a/test/Transforms/LoopStrengthReduce/AMDGPU/different-addrspace-addressing-mode-loops.ll b/test/Transforms/LoopStrengthReduce/AMDGPU/different-addrspace-addressing-mode-loops.ll new file mode 100644 index 000000000000..bf61112a3c3e --- /dev/null +++ b/test/Transforms/LoopStrengthReduce/AMDGPU/different-addrspace-addressing-mode-loops.ll @@ -0,0 +1,156 @@ +; RUN: opt -S -mtriple=amdgcn-- -mcpu=bonaire -loop-reduce < %s | FileCheck -check-prefix=OPT %s + +; Test that loops with different maximum offsets for different address +; spaces are correctly handled. + +target datalayout = "e-p:32:32-p1:64:64-p2:64:64-p3:32:32-p4:64:64-p5:32:32-p24:64:64-i64:64-v16:16-v24:32-v32:32-v48:64-v96:128-v192:256-v256:256-v512:512-v1024:1024-v2048:2048-n32:64" + +; OPT-LABEL: @test_global_addressing_loop_uniform_index_max_offset_i32( +; OPT: {{^}}.lr.ph: +; OPT: %lsr.iv2 = phi i8 addrspace(1)* [ %scevgep3, %.lr.ph ], [ %arg1, %.lr.ph.preheader ] +; OPT: %scevgep4 = getelementptr i8, i8 addrspace(1)* %lsr.iv2, i64 4095 +; OPT: load i8, i8 addrspace(1)* %scevgep4, align 1 +define void @test_global_addressing_loop_uniform_index_max_offset_i32(i32 addrspace(1)* noalias nocapture %arg0, i8 addrspace(1)* noalias nocapture readonly %arg1, i32 %n) #0 { +bb: + %tmp = icmp sgt i32 %n, 0 + br i1 %tmp, label %.lr.ph.preheader, label %._crit_edge + +.lr.ph.preheader: ; preds = %bb + br label %.lr.ph + +._crit_edge.loopexit: ; preds = %.lr.ph + br label %._crit_edge + +._crit_edge: ; preds = %._crit_edge.loopexit, %bb + ret void + +.lr.ph: ; preds = %.lr.ph, %.lr.ph.preheader + %indvars.iv = phi i64 [ %indvars.iv.next, %.lr.ph ], [ 0, %.lr.ph.preheader ] + %tmp1 = add nuw nsw i64 %indvars.iv, 4095 + %tmp2 = getelementptr inbounds i8, i8 addrspace(1)* %arg1, i64 %tmp1 + %tmp3 = load i8, i8 addrspace(1)* %tmp2, align 1 + %tmp4 = sext i8 %tmp3 to i32 + %tmp5 = getelementptr inbounds i32, i32 addrspace(1)* %arg0, i64 %indvars.iv + %tmp6 = load i32, i32 addrspace(1)* %tmp5, align 4 + %tmp7 = add nsw i32 %tmp6, %tmp4 + store i32 %tmp7, i32 addrspace(1)* %tmp5, align 4 + %indvars.iv.next = add nuw nsw i64 %indvars.iv, 1 + %lftr.wideiv = trunc i64 %indvars.iv.next to i32 + %exitcond = icmp eq i32 %lftr.wideiv, %n + br i1 %exitcond, label %._crit_edge.loopexit, label %.lr.ph +} + +; OPT-LABEL: @test_global_addressing_loop_uniform_index_max_offset_p1_i32( +; OPT: {{^}}.lr.ph.preheader: +; OPT: %scevgep2 = getelementptr i8, i8 addrspace(1)* %arg1, i64 4096 +; OPT: br label %.lr.ph + +; OPT: {{^}}.lr.ph: +; OPT: %lsr.iv3 = phi i8 addrspace(1)* [ %scevgep4, %.lr.ph ], [ %scevgep2, %.lr.ph.preheader ] +; OPT: %scevgep4 = getelementptr i8, i8 addrspace(1)* %lsr.iv3, i64 1 +define void @test_global_addressing_loop_uniform_index_max_offset_p1_i32(i32 addrspace(1)* noalias nocapture %arg0, i8 addrspace(1)* noalias nocapture readonly %arg1, i32 %n) #0 { +bb: + %tmp = icmp sgt i32 %n, 0 + br i1 %tmp, label %.lr.ph.preheader, label %._crit_edge + +.lr.ph.preheader: ; preds = %bb + br label %.lr.ph + +._crit_edge.loopexit: ; preds = %.lr.ph + br label %._crit_edge + +._crit_edge: ; preds = %._crit_edge.loopexit, %bb + ret void + +.lr.ph: ; preds = %.lr.ph, %.lr.ph.preheader + %indvars.iv = phi i64 [ %indvars.iv.next, %.lr.ph ], [ 0, %.lr.ph.preheader ] + %tmp1 = add nuw nsw i64 %indvars.iv, 4096 + %tmp2 = getelementptr inbounds i8, i8 addrspace(1)* %arg1, i64 %tmp1 + %tmp3 = load i8, i8 addrspace(1)* %tmp2, align 1 + %tmp4 = sext i8 %tmp3 to i32 + %tmp5 = getelementptr inbounds i32, i32 addrspace(1)* %arg0, i64 %indvars.iv + %tmp6 = load i32, i32 addrspace(1)* %tmp5, align 4 + %tmp7 = add nsw i32 %tmp6, %tmp4 + store i32 %tmp7, i32 addrspace(1)* %tmp5, align 4 + %indvars.iv.next = add nuw nsw i64 %indvars.iv, 1 + %lftr.wideiv = trunc i64 %indvars.iv.next to i32 + %exitcond = icmp eq i32 %lftr.wideiv, %n + br i1 %exitcond, label %._crit_edge.loopexit, label %.lr.ph +} + +; OPT-LABEL: @test_local_addressing_loop_uniform_index_max_offset_i32( +; OPT: {{^}}.lr.ph +; OPT: %lsr.iv2 = phi i8 addrspace(3)* [ %scevgep3, %.lr.ph ], [ %arg1, %.lr.ph.preheader ] +; OPT: %scevgep4 = getelementptr i8, i8 addrspace(3)* %lsr.iv2, i32 65535 +; OPT: %tmp4 = load i8, i8 addrspace(3)* %scevgep4, align 1 +define void @test_local_addressing_loop_uniform_index_max_offset_i32(i32 addrspace(1)* noalias nocapture %arg0, i8 addrspace(3)* noalias nocapture readonly %arg1, i32 %n) #0 { +bb: + %tmp = icmp sgt i32 %n, 0 + br i1 %tmp, label %.lr.ph.preheader, label %._crit_edge + +.lr.ph.preheader: ; preds = %bb + br label %.lr.ph + +._crit_edge.loopexit: ; preds = %.lr.ph + br label %._crit_edge + +._crit_edge: ; preds = %._crit_edge.loopexit, %bb + ret void + +.lr.ph: ; preds = %.lr.ph, %.lr.ph.preheader + %indvars.iv = phi i64 [ %indvars.iv.next, %.lr.ph ], [ 0, %.lr.ph.preheader ] + %tmp1 = add nuw nsw i64 %indvars.iv, 65535 + %tmp2 = trunc i64 %tmp1 to i32 + %tmp3 = getelementptr inbounds i8, i8 addrspace(3)* %arg1, i32 %tmp2 + %tmp4 = load i8, i8 addrspace(3)* %tmp3, align 1 + %tmp5 = sext i8 %tmp4 to i32 + %tmp6 = getelementptr inbounds i32, i32 addrspace(1)* %arg0, i64 %indvars.iv + %tmp7 = load i32, i32 addrspace(1)* %tmp6, align 4 + %tmp8 = add nsw i32 %tmp7, %tmp5 + store i32 %tmp8, i32 addrspace(1)* %tmp6, align 4 + %indvars.iv.next = add nuw nsw i64 %indvars.iv, 1 + %lftr.wideiv = trunc i64 %indvars.iv.next to i32 + %exitcond = icmp eq i32 %lftr.wideiv, %n + br i1 %exitcond, label %._crit_edge.loopexit, label %.lr.ph +} + +; OPT-LABEL: @test_local_addressing_loop_uniform_index_max_offset_p1_i32( +; OPT: {{^}}.lr.ph.preheader: +; OPT: %scevgep2 = getelementptr i8, i8 addrspace(3)* %arg1, i32 65536 +; OPT: br label %.lr.ph + +; OPT: {{^}}.lr.ph: +; OPT: %lsr.iv3 = phi i8 addrspace(3)* [ %scevgep4, %.lr.ph ], [ %scevgep2, %.lr.ph.preheader ] +; OPT: %scevgep4 = getelementptr i8, i8 addrspace(3)* %lsr.iv3, i32 1 +define void @test_local_addressing_loop_uniform_index_max_offset_p1_i32(i32 addrspace(1)* noalias nocapture %arg0, i8 addrspace(3)* noalias nocapture readonly %arg1, i32 %n) #0 { +bb: + %tmp = icmp sgt i32 %n, 0 + br i1 %tmp, label %.lr.ph.preheader, label %._crit_edge + +.lr.ph.preheader: ; preds = %bb + br label %.lr.ph + +._crit_edge.loopexit: ; preds = %.lr.ph + br label %._crit_edge + +._crit_edge: ; preds = %._crit_edge.loopexit, %bb + ret void + +.lr.ph: ; preds = %.lr.ph, %.lr.ph.preheader + %indvars.iv = phi i64 [ %indvars.iv.next, %.lr.ph ], [ 0, %.lr.ph.preheader ] + %tmp1 = add nuw nsw i64 %indvars.iv, 65536 + %tmp2 = trunc i64 %tmp1 to i32 + %tmp3 = getelementptr inbounds i8, i8 addrspace(3)* %arg1, i32 %tmp2 + %tmp4 = load i8, i8 addrspace(3)* %tmp3, align 1 + %tmp5 = sext i8 %tmp4 to i32 + %tmp6 = getelementptr inbounds i32, i32 addrspace(1)* %arg0, i64 %indvars.iv + %tmp7 = load i32, i32 addrspace(1)* %tmp6, align 4 + %tmp8 = add nsw i32 %tmp7, %tmp5 + store i32 %tmp8, i32 addrspace(1)* %tmp6, align 4 + %indvars.iv.next = add nuw nsw i64 %indvars.iv, 1 + %lftr.wideiv = trunc i64 %indvars.iv.next to i32 + %exitcond = icmp eq i32 %lftr.wideiv, %n + br i1 %exitcond, label %._crit_edge.loopexit, label %.lr.ph +} + +attributes #0 = { nounwind "less-precise-fpmad"="false" "no-frame-pointer-elim"="true" "no-frame-pointer-elim-non-leaf" "no-infs-fp-math"="false" "no-nans-fp-math"="false" "stack-protector-buffer-size"="8" "target-cpu"="hawaii" "unsafe-fp-math"="false" "use-soft-float"="false" } diff --git a/test/Transforms/LoopStrengthReduce/AMDGPU/lit.local.cfg b/test/Transforms/LoopStrengthReduce/AMDGPU/lit.local.cfg new file mode 100644 index 000000000000..6baccf05fff0 --- /dev/null +++ b/test/Transforms/LoopStrengthReduce/AMDGPU/lit.local.cfg @@ -0,0 +1,3 @@ +if not 'AMDGPU' in config.root.targets: + config.unsupported = True + diff --git a/test/Transforms/LoopStrengthReduce/AMDGPU/lsr-postinc-pos-addrspace.ll b/test/Transforms/LoopStrengthReduce/AMDGPU/lsr-postinc-pos-addrspace.ll new file mode 100644 index 000000000000..bd80302a68b8 --- /dev/null +++ b/test/Transforms/LoopStrengthReduce/AMDGPU/lsr-postinc-pos-addrspace.ll @@ -0,0 +1,113 @@ +; RUN: llc -march=amdgcn -mcpu=bonaire -print-lsr-output < %s 2>&1 | FileCheck %s + +; Test various conditions where OptimizeLoopTermCond doesn't look at a +; memory instruction use and fails to find the address space. + +target datalayout = "e-p:32:32-p1:64:64-p2:64:64-p3:32:32-p4:64:64-p5:32:32-p24:64:64-i64:64-v16:16-v24:32-v32:32-v48:64-v96:128-v192:256-v256:256-v512:512-v1024:1024-v2048:2048-n32:64" + +; CHECK-LABEL: @local_cmp_user( +; CHECK: bb11: +; CHECK: %lsr.iv1 = phi i32 [ %lsr.iv.next2, %bb ], [ -2, %entry ] +; CHECK: %lsr.iv = phi i32 [ %lsr.iv.next, %bb ], [ undef, %entry ] + +; CHECK: bb: +; CHECK: %lsr.iv.next = add i32 %lsr.iv, -1 +; CHECK: %lsr.iv.next2 = add i32 %lsr.iv1, 2 +; CHECK: %scevgep = getelementptr i8, i8 addrspace(3)* %t, i32 %lsr.iv.next2 +; CHECK: %c1 = icmp ult i8 addrspace(3)* %scevgep, undef +define void @local_cmp_user() nounwind { +entry: + br label %bb11 + +bb11: + %i = phi i32 [ 0, %entry ], [ %i.next, %bb ] + %ii = shl i32 %i, 1 + %c0 = icmp eq i32 %i, undef + br i1 %c0, label %bb13, label %bb + +bb: + %t = load i8 addrspace(3)*, i8 addrspace(3)* addrspace(3)* undef + %p = getelementptr i8, i8 addrspace(3)* %t, i32 %ii + %c1 = icmp ult i8 addrspace(3)* %p, undef + %i.next = add i32 %i, 1 + br i1 %c1, label %bb11, label %bb13 + +bb13: + unreachable +} + +; CHECK-LABEL: @global_cmp_user( +; CHECK: %lsr.iv.next = add i64 %lsr.iv, -1 +; CHECK: %lsr.iv.next2 = add i64 %lsr.iv1, 2 +; CHECK: %scevgep = getelementptr i8, i8 addrspace(1)* %t, i64 %lsr.iv.next2 +define void @global_cmp_user() nounwind { +entry: + br label %bb11 + +bb11: + %i = phi i64 [ 0, %entry ], [ %i.next, %bb ] + %ii = shl i64 %i, 1 + %c0 = icmp eq i64 %i, undef + br i1 %c0, label %bb13, label %bb + +bb: + %t = load i8 addrspace(1)*, i8 addrspace(1)* addrspace(1)* undef + %p = getelementptr i8, i8 addrspace(1)* %t, i64 %ii + %c1 = icmp ult i8 addrspace(1)* %p, undef + %i.next = add i64 %i, 1 + br i1 %c1, label %bb11, label %bb13 + +bb13: + unreachable +} + +; CHECK-LABEL: @global_gep_user( +; CHECK: %p = getelementptr i8, i8 addrspace(1)* %t, i32 %lsr.iv1 +; CHECK: %lsr.iv.next = add i32 %lsr.iv, -1 +; CHECK: %lsr.iv.next2 = add i32 %lsr.iv1, 2 +define void @global_gep_user() nounwind { +entry: + br label %bb11 + +bb11: + %i = phi i32 [ 0, %entry ], [ %i.next, %bb ] + %ii = shl i32 %i, 1 + %c0 = icmp eq i32 %i, undef + br i1 %c0, label %bb13, label %bb + +bb: + %t = load i8 addrspace(1)*, i8 addrspace(1)* addrspace(1)* undef + %p = getelementptr i8, i8 addrspace(1)* %t, i32 %ii + %c1 = icmp ult i8 addrspace(1)* %p, undef + %i.next = add i32 %i, 1 + br i1 %c1, label %bb11, label %bb13 + +bb13: + unreachable +} + +; CHECK-LABEL: @global_sext_scale_user( +; CHECK: %p = getelementptr i8, i8 addrspace(1)* %t, i64 %ii.ext +; CHECK: %lsr.iv.next = add i32 %lsr.iv, -1 +; CHECK: %lsr.iv.next2 = add i32 %lsr.iv1, 2 +define void @global_sext_scale_user() nounwind { +entry: + br label %bb11 + +bb11: + %i = phi i32 [ 0, %entry ], [ %i.next, %bb ] + %ii = shl i32 %i, 1 + %ii.ext = sext i32 %ii to i64 + %c0 = icmp eq i32 %i, undef + br i1 %c0, label %bb13, label %bb + +bb: + %t = load i8 addrspace(1)*, i8 addrspace(1)* addrspace(1)* undef + %p = getelementptr i8, i8 addrspace(1)* %t, i64 %ii.ext + %c1 = icmp ult i8 addrspace(1)* %p, undef + %i.next = add i32 %i, 1 + br i1 %c1, label %bb11, label %bb13 + +bb13: + unreachable +} diff --git a/test/Transforms/LoopStrengthReduce/ARM/ivchain-ARM.ll b/test/Transforms/LoopStrengthReduce/ARM/ivchain-ARM.ll index 2ad6c2ea52da..788842101080 100644 --- a/test/Transforms/LoopStrengthReduce/ARM/ivchain-ARM.ll +++ b/test/Transforms/LoopStrengthReduce/ARM/ivchain-ARM.ll @@ -239,33 +239,33 @@ define hidden void @testNeon(i8* %ref_data, i32 %ref_stride, i32 %limit, <16 x i %counter.04 = phi i32 [ 0, %.lr.ph ], [ %44, %11 ] %result.03 = phi <16 x i8> [ zeroinitializer, %.lr.ph ], [ %41, %11 ] %.012 = phi <16 x i8>* [ %data, %.lr.ph ], [ %43, %11 ] - %12 = tail call <1 x i64> @llvm.arm.neon.vld1.v1i64(i8* %.05, i32 1) nounwind + %12 = tail call <1 x i64> @llvm.arm.neon.vld1.v1i64.p0i8(i8* %.05, i32 1) nounwind %13 = getelementptr inbounds i8, i8* %.05, i32 %ref_stride - %14 = tail call <1 x i64> @llvm.arm.neon.vld1.v1i64(i8* %13, i32 1) nounwind + %14 = tail call <1 x i64> @llvm.arm.neon.vld1.v1i64.p0i8(i8* %13, i32 1) nounwind %15 = shufflevector <1 x i64> %12, <1 x i64> %14, <2 x i32> <i32 0, i32 1> %16 = bitcast <2 x i64> %15 to <16 x i8> %17 = getelementptr inbounds <16 x i8>, <16 x i8>* %.012, i32 1 store <16 x i8> %16, <16 x i8>* %.012, align 4 %18 = getelementptr inbounds i8, i8* %.05, i32 %2 - %19 = tail call <1 x i64> @llvm.arm.neon.vld1.v1i64(i8* %18, i32 1) nounwind + %19 = tail call <1 x i64> @llvm.arm.neon.vld1.v1i64.p0i8(i8* %18, i32 1) nounwind %20 = getelementptr inbounds i8, i8* %.05, i32 %3 - %21 = tail call <1 x i64> @llvm.arm.neon.vld1.v1i64(i8* %20, i32 1) nounwind + %21 = tail call <1 x i64> @llvm.arm.neon.vld1.v1i64.p0i8(i8* %20, i32 1) nounwind %22 = shufflevector <1 x i64> %19, <1 x i64> %21, <2 x i32> <i32 0, i32 1> %23 = bitcast <2 x i64> %22 to <16 x i8> %24 = getelementptr inbounds <16 x i8>, <16 x i8>* %.012, i32 2 store <16 x i8> %23, <16 x i8>* %17, align 4 %25 = getelementptr inbounds i8, i8* %.05, i32 %4 - %26 = tail call <1 x i64> @llvm.arm.neon.vld1.v1i64(i8* %25, i32 1) nounwind + %26 = tail call <1 x i64> @llvm.arm.neon.vld1.v1i64.p0i8(i8* %25, i32 1) nounwind %27 = getelementptr inbounds i8, i8* %.05, i32 %5 - %28 = tail call <1 x i64> @llvm.arm.neon.vld1.v1i64(i8* %27, i32 1) nounwind + %28 = tail call <1 x i64> @llvm.arm.neon.vld1.v1i64.p0i8(i8* %27, i32 1) nounwind %29 = shufflevector <1 x i64> %26, <1 x i64> %28, <2 x i32> <i32 0, i32 1> %30 = bitcast <2 x i64> %29 to <16 x i8> %31 = getelementptr inbounds <16 x i8>, <16 x i8>* %.012, i32 3 store <16 x i8> %30, <16 x i8>* %24, align 4 %32 = getelementptr inbounds i8, i8* %.05, i32 %6 - %33 = tail call <1 x i64> @llvm.arm.neon.vld1.v1i64(i8* %32, i32 1) nounwind + %33 = tail call <1 x i64> @llvm.arm.neon.vld1.v1i64.p0i8(i8* %32, i32 1) nounwind %34 = getelementptr inbounds i8, i8* %.05, i32 %7 - %35 = tail call <1 x i64> @llvm.arm.neon.vld1.v1i64(i8* %34, i32 1) nounwind + %35 = tail call <1 x i64> @llvm.arm.neon.vld1.v1i64.p0i8(i8* %34, i32 1) nounwind %36 = shufflevector <1 x i64> %33, <1 x i64> %35, <2 x i32> <i32 0, i32 1> %37 = bitcast <2 x i64> %36 to <16 x i8> store <16 x i8> %37, <16 x i8>* %31, align 4 @@ -290,7 +290,7 @@ define hidden void @testNeon(i8* %ref_data, i32 %ref_stride, i32 %limit, <16 x i ret void } -declare <1 x i64> @llvm.arm.neon.vld1.v1i64(i8*, i32) nounwind readonly +declare <1 x i64> @llvm.arm.neon.vld1.v1i64.p0i8(i8*, i32) nounwind readonly ; Handle chains in which the same offset is used for both loads and ; stores to the same array. @@ -328,32 +328,32 @@ for.body: ; preds = %for.body, %entry %i.0110 = phi i32 [ 0, %entry ], [ %inc, %for.body ] %src.addr = phi i8* [ %src, %entry ], [ %add.ptr45, %for.body ] %add.ptr = getelementptr inbounds i8, i8* %src.addr, i32 %idx.neg - %vld1 = tail call <8 x i8> @llvm.arm.neon.vld1.v8i8(i8* %add.ptr, i32 1) + %vld1 = tail call <8 x i8> @llvm.arm.neon.vld1.v8i8.p0i8(i8* %add.ptr, i32 1) %add.ptr3 = getelementptr inbounds i8, i8* %src.addr, i32 %idx.neg2 - %vld2 = tail call <8 x i8> @llvm.arm.neon.vld1.v8i8(i8* %add.ptr3, i32 1) + %vld2 = tail call <8 x i8> @llvm.arm.neon.vld1.v8i8.p0i8(i8* %add.ptr3, i32 1) %add.ptr7 = getelementptr inbounds i8, i8* %src.addr, i32 %idx.neg6 - %vld3 = tail call <8 x i8> @llvm.arm.neon.vld1.v8i8(i8* %add.ptr7, i32 1) + %vld3 = tail call <8 x i8> @llvm.arm.neon.vld1.v8i8.p0i8(i8* %add.ptr7, i32 1) %add.ptr11 = getelementptr inbounds i8, i8* %src.addr, i32 %idx.neg10 - %vld4 = tail call <8 x i8> @llvm.arm.neon.vld1.v8i8(i8* %add.ptr11, i32 1) - %vld5 = tail call <8 x i8> @llvm.arm.neon.vld1.v8i8(i8* %src.addr, i32 1) + %vld4 = tail call <8 x i8> @llvm.arm.neon.vld1.v8i8.p0i8(i8* %add.ptr11, i32 1) + %vld5 = tail call <8 x i8> @llvm.arm.neon.vld1.v8i8.p0i8(i8* %src.addr, i32 1) %add.ptr17 = getelementptr inbounds i8, i8* %src.addr, i32 %stride - %vld6 = tail call <8 x i8> @llvm.arm.neon.vld1.v8i8(i8* %add.ptr17, i32 1) + %vld6 = tail call <8 x i8> @llvm.arm.neon.vld1.v8i8.p0i8(i8* %add.ptr17, i32 1) %add.ptr20 = getelementptr inbounds i8, i8* %src.addr, i32 %mul5 - %vld7 = tail call <8 x i8> @llvm.arm.neon.vld1.v8i8(i8* %add.ptr20, i32 1) + %vld7 = tail call <8 x i8> @llvm.arm.neon.vld1.v8i8.p0i8(i8* %add.ptr20, i32 1) %add.ptr23 = getelementptr inbounds i8, i8* %src.addr, i32 %mul1 - %vld8 = tail call <8 x i8> @llvm.arm.neon.vld1.v8i8(i8* %add.ptr23, i32 1) + %vld8 = tail call <8 x i8> @llvm.arm.neon.vld1.v8i8.p0i8(i8* %add.ptr23, i32 1) %vadd1 = tail call <8 x i8> @llvm.arm.neon.vhaddu.v8i8(<8 x i8> %vld1, <8 x i8> %vld2) nounwind %vadd2 = tail call <8 x i8> @llvm.arm.neon.vhaddu.v8i8(<8 x i8> %vld2, <8 x i8> %vld3) nounwind %vadd3 = tail call <8 x i8> @llvm.arm.neon.vhaddu.v8i8(<8 x i8> %vld3, <8 x i8> %vld4) nounwind %vadd4 = tail call <8 x i8> @llvm.arm.neon.vhaddu.v8i8(<8 x i8> %vld4, <8 x i8> %vld5) nounwind %vadd5 = tail call <8 x i8> @llvm.arm.neon.vhaddu.v8i8(<8 x i8> %vld5, <8 x i8> %vld6) nounwind %vadd6 = tail call <8 x i8> @llvm.arm.neon.vhaddu.v8i8(<8 x i8> %vld6, <8 x i8> %vld7) nounwind - tail call void @llvm.arm.neon.vst1.v8i8(i8* %add.ptr3, <8 x i8> %vadd1, i32 1) - tail call void @llvm.arm.neon.vst1.v8i8(i8* %add.ptr7, <8 x i8> %vadd2, i32 1) - tail call void @llvm.arm.neon.vst1.v8i8(i8* %add.ptr11, <8 x i8> %vadd3, i32 1) - tail call void @llvm.arm.neon.vst1.v8i8(i8* %src.addr, <8 x i8> %vadd4, i32 1) - tail call void @llvm.arm.neon.vst1.v8i8(i8* %add.ptr17, <8 x i8> %vadd5, i32 1) - tail call void @llvm.arm.neon.vst1.v8i8(i8* %add.ptr20, <8 x i8> %vadd6, i32 1) + tail call void @llvm.arm.neon.vst1.p0i8.v8i8(i8* %add.ptr3, <8 x i8> %vadd1, i32 1) + tail call void @llvm.arm.neon.vst1.p0i8.v8i8(i8* %add.ptr7, <8 x i8> %vadd2, i32 1) + tail call void @llvm.arm.neon.vst1.p0i8.v8i8(i8* %add.ptr11, <8 x i8> %vadd3, i32 1) + tail call void @llvm.arm.neon.vst1.p0i8.v8i8(i8* %src.addr, <8 x i8> %vadd4, i32 1) + tail call void @llvm.arm.neon.vst1.p0i8.v8i8(i8* %add.ptr17, <8 x i8> %vadd5, i32 1) + tail call void @llvm.arm.neon.vst1.p0i8.v8i8(i8* %add.ptr20, <8 x i8> %vadd6, i32 1) %inc = add nsw i32 %i.0110, 1 %add.ptr45 = getelementptr inbounds i8, i8* %src.addr, i32 8 %exitcond = icmp eq i32 %inc, 4 @@ -363,8 +363,8 @@ for.end: ; preds = %for.body ret void } -declare <8 x i8> @llvm.arm.neon.vld1.v8i8(i8*, i32) nounwind readonly +declare <8 x i8> @llvm.arm.neon.vld1.v8i8.p0i8(i8*, i32) nounwind readonly -declare void @llvm.arm.neon.vst1.v8i8(i8*, <8 x i8>, i32) nounwind +declare void @llvm.arm.neon.vst1.p0i8.v8i8(i8*, <8 x i8>, i32) nounwind declare <8 x i8> @llvm.arm.neon.vhaddu.v8i8(<8 x i8>, <8 x i8>) nounwind readnone diff --git a/test/Transforms/LoopStrengthReduce/NVPTX/lit.local.cfg b/test/Transforms/LoopStrengthReduce/NVPTX/lit.local.cfg new file mode 100644 index 000000000000..2cb98eb371b2 --- /dev/null +++ b/test/Transforms/LoopStrengthReduce/NVPTX/lit.local.cfg @@ -0,0 +1,2 @@ +if not 'NVPTX' in config.root.targets: + config.unsupported = True diff --git a/test/Transforms/LoopStrengthReduce/NVPTX/trunc.ll b/test/Transforms/LoopStrengthReduce/NVPTX/trunc.ll new file mode 100644 index 000000000000..a16065b4dfbd --- /dev/null +++ b/test/Transforms/LoopStrengthReduce/NVPTX/trunc.ll @@ -0,0 +1,45 @@ +; RUN: opt < %s -loop-reduce -S | FileCheck %s + +target datalayout = "e-i64:64-v16:16-v32:32-n16:32:64" +target triple = "nvptx64-nvidia-cuda" + +; This confirms that NVPTXTTI considers a 64-to-32 integer trunc free. If such +; truncs were not considered free, LSR would promote (int)i as a separate +; induction variable in the following example. +; +; for (long i = begin; i != end; i += stride) +; use((int)i); +; +; That would be worthless, because "i" is simulated by two 32-bit registers and +; truncating it to 32-bit is as simple as directly using the register that +; contains the low bits. +define void @trunc_is_free(i64 %begin, i64 %stride, i64 %end) { +; CHECK-LABEL: @trunc_is_free( +entry: + %cmp.4 = icmp eq i64 %begin, %end + br i1 %cmp.4, label %for.cond.cleanup, label %for.body.preheader + +for.body.preheader: ; preds = %entry + br label %for.body + +for.cond.cleanup.loopexit: ; preds = %for.body + br label %for.cond.cleanup + +for.cond.cleanup: ; preds = %for.cond.cleanup.loopexit, %entry + ret void + +for.body: ; preds = %for.body.preheader, %for.body +; CHECK: for.body: + %i.05 = phi i64 [ %add, %for.body ], [ %begin, %for.body.preheader ] + %conv = trunc i64 %i.05 to i32 +; CHECK: trunc i64 %{{[^ ]+}} to i32 + tail call void @_Z3usei(i32 %conv) #2 + %add = add nsw i64 %i.05, %stride + %cmp = icmp eq i64 %add, %end + br i1 %cmp, label %for.cond.cleanup.loopexit, label %for.body +} + +declare void @_Z3usei(i32) + +!nvvm.annotations = !{!0} +!0 = !{void (i64, i64, i64)* @trunc_is_free, !"kernel", i32 1} diff --git a/test/Transforms/LoopStrengthReduce/X86/ivchain-stress-X86.ll b/test/Transforms/LoopStrengthReduce/X86/ivchain-stress-X86.ll index 24be0dc42d6d..7925bf01020e 100644 --- a/test/Transforms/LoopStrengthReduce/X86/ivchain-stress-X86.ll +++ b/test/Transforms/LoopStrengthReduce/X86/ivchain-stress-X86.ll @@ -23,7 +23,7 @@ ; X32: add ; X32: add ; X32: add -; X32: leal +; X32: add ; X32: %for.body.3 define void @sharedidx(i8* nocapture %a, i8* nocapture %b, i8* nocapture %c, i32 %s, i32 %len) nounwind ssp { entry: diff --git a/test/Transforms/LoopStrengthReduce/funclet.ll b/test/Transforms/LoopStrengthReduce/funclet.ll new file mode 100644 index 000000000000..5d20646141c4 --- /dev/null +++ b/test/Transforms/LoopStrengthReduce/funclet.ll @@ -0,0 +1,216 @@ +; RUN: opt < %s -loop-reduce -S | FileCheck %s + +target datalayout = "e-m:x-p:32:32-i64:64-f80:32-n8:16:32-a:0:32-S32" +target triple = "i686-pc-windows-msvc" + +declare i32 @_except_handler3(...) +declare i32 @__CxxFrameHandler3(...) + +declare void @external(i32*) +declare void @reserve() + +define void @f() personality i32 (...)* @_except_handler3 { +entry: + br label %throw + +throw: ; preds = %throw, %entry + %tmp96 = getelementptr inbounds i8, i8* undef, i32 1 + invoke void @reserve() + to label %throw unwind label %pad + +pad: ; preds = %throw + %phi2 = phi i8* [ %tmp96, %throw ] + %cs = catchswitch within none [label %unreachable] unwind label %blah2 + +unreachable: + catchpad within %cs [] + unreachable + +blah2: + %cleanuppadi4.i.i.i = cleanuppad within none [] + br label %loop_body + +loop_body: ; preds = %iter, %pad + %tmp99 = phi i8* [ %tmp101, %iter ], [ %phi2, %blah2 ] + %tmp100 = icmp eq i8* %tmp99, undef + br i1 %tmp100, label %unwind_out, label %iter + +iter: ; preds = %loop_body + %tmp101 = getelementptr inbounds i8, i8* %tmp99, i32 1 + br i1 undef, label %unwind_out, label %loop_body + +unwind_out: ; preds = %iter, %loop_body + cleanupret from %cleanuppadi4.i.i.i unwind to caller +} + +; CHECK-LABEL: define void @f( +; CHECK: cleanuppad within none [] +; CHECK-NEXT: ptrtoint i8* %phi2 to i32 + +define void @g() personality i32 (...)* @_except_handler3 { +entry: + br label %throw + +throw: ; preds = %throw, %entry + %tmp96 = getelementptr inbounds i8, i8* undef, i32 1 + invoke void @reserve() + to label %throw unwind label %pad + +pad: + %phi2 = phi i8* [ %tmp96, %throw ] + %cs = catchswitch within none [label %unreachable, label %blah] unwind to caller + +unreachable: + catchpad within %cs [] + unreachable + +blah: + %catchpad = catchpad within %cs [] + br label %loop_body + +unwind_out: + catchret from %catchpad to label %leave + +leave: + ret void + +loop_body: ; preds = %iter, %pad + %tmp99 = phi i8* [ %tmp101, %iter ], [ %phi2, %blah ] + %tmp100 = icmp eq i8* %tmp99, undef + br i1 %tmp100, label %unwind_out, label %iter + +iter: ; preds = %loop_body + %tmp101 = getelementptr inbounds i8, i8* %tmp99, i32 1 + br i1 undef, label %unwind_out, label %loop_body +} + +; CHECK-LABEL: define void @g( +; CHECK: blah: +; CHECK-NEXT: catchpad within %cs [] +; CHECK-NEXT: ptrtoint i8* %phi2 to i32 + + +define void @h() personality i32 (...)* @_except_handler3 { +entry: + br label %throw + +throw: ; preds = %throw, %entry + %tmp96 = getelementptr inbounds i8, i8* undef, i32 1 + invoke void @reserve() + to label %throw unwind label %pad + +pad: + %cs = catchswitch within none [label %unreachable, label %blug] unwind to caller + +unreachable: + catchpad within %cs [] + unreachable + +blug: + %phi2 = phi i8* [ %tmp96, %pad ] + %catchpad = catchpad within %cs [] + br label %loop_body + +unwind_out: + catchret from %catchpad to label %leave + +leave: + ret void + +loop_body: ; preds = %iter, %pad + %tmp99 = phi i8* [ %tmp101, %iter ], [ %phi2, %blug ] + %tmp100 = icmp eq i8* %tmp99, undef + br i1 %tmp100, label %unwind_out, label %iter + +iter: ; preds = %loop_body + %tmp101 = getelementptr inbounds i8, i8* %tmp99, i32 1 + br i1 undef, label %unwind_out, label %loop_body +} + +; CHECK-LABEL: define void @h( +; CHECK: blug: +; CHECK: catchpad within %cs [] +; CHECK-NEXT: ptrtoint i8* %phi2 to i32 + +define void @i() personality i32 (...)* @_except_handler3 { +entry: + br label %throw + +throw: ; preds = %throw, %entry + %tmp96 = getelementptr inbounds i8, i8* undef, i32 1 + invoke void @reserve() + to label %throw unwind label %catchpad + +catchpad: ; preds = %throw + %phi2 = phi i8* [ %tmp96, %throw ] + %cs = catchswitch within none [label %cp_body] unwind label %cleanuppad + +cp_body: + catchpad within %cs [] + br label %loop_head + +cleanuppad: + cleanuppad within none [] + br label %loop_head + +loop_head: + br label %loop_body + +loop_body: ; preds = %iter, %catchpad + %tmp99 = phi i8* [ %tmp101, %iter ], [ %phi2, %loop_head ] + %tmp100 = icmp eq i8* %tmp99, undef + br i1 %tmp100, label %unwind_out, label %iter + +iter: ; preds = %loop_body + %tmp101 = getelementptr inbounds i8, i8* %tmp99, i32 1 + br i1 undef, label %unwind_out, label %loop_body + +unwind_out: ; preds = %iter, %loop_body + unreachable +} + +; CHECK-LABEL: define void @i( +; CHECK: ptrtoint i8* %phi2 to i32 + +define void @test1(i32* %b, i32* %c) personality i32 (...)* @__CxxFrameHandler3 { +entry: + br label %for.cond + +for.cond: ; preds = %for.inc, %entry + %d.0 = phi i32* [ %b, %entry ], [ %incdec.ptr, %for.inc ] + invoke void @external(i32* %d.0) + to label %for.inc unwind label %catch.dispatch + +for.inc: ; preds = %for.cond + %incdec.ptr = getelementptr inbounds i32, i32* %d.0, i32 1 + br label %for.cond + +catch.dispatch: ; preds = %for.cond + %cs = catchswitch within none [label %catch] unwind label %catch.dispatch.2 + +catch: ; preds = %catch.dispatch + %0 = catchpad within %cs [i8* null, i32 64, i8* null] + catchret from %0 to label %try.cont + +try.cont: ; preds = %catch + invoke void @external(i32* %c) + to label %try.cont.7 unwind label %catch.dispatch.2 + +catch.dispatch.2: ; preds = %try.cont, %catchendblock + %e.0 = phi i32* [ %c, %try.cont ], [ %b, %catch.dispatch ] + %cs2 = catchswitch within none [label %catch.4] unwind to caller + +catch.4: ; preds = %catch.dispatch.2 + catchpad within %cs2 [i8* null, i32 64, i8* null] + unreachable + +try.cont.7: ; preds = %try.cont + ret void +} + +; CHECK-LABEL: define void @test1( +; CHECK: for.cond: +; CHECK: %d.0 = phi i32* [ %b, %entry ], [ %incdec.ptr, %for.inc ] + +; CHECK: catch.dispatch.2: +; CHECK: %e.0 = phi i32* [ %c, %try.cont ], [ %b, %catch.dispatch ] diff --git a/test/Transforms/LoopStrengthReduce/pr12018.ll b/test/Transforms/LoopStrengthReduce/pr12018.ll index b15961a77904..bb5d1654fada 100644 --- a/test/Transforms/LoopStrengthReduce/pr12018.ll +++ b/test/Transforms/LoopStrengthReduce/pr12018.ll @@ -16,7 +16,7 @@ for.body: ; preds = %_ZN8nsTArray9Elemen %tmp = bitcast %struct.nsTArrayHeader* %add.ptr.i to %struct.nsTArray* %arrayidx = getelementptr inbounds %struct.nsTArray, %struct.nsTArray* %tmp, i32 %i.06 %add = add nsw i32 %i.06, 1 - call void @llvm.dbg.value(metadata %struct.nsTArray* %aValues, i64 0, metadata !0, metadata !DIExpression()) nounwind, !dbg !DILocation(scope: !DISubprogram()) + call void @llvm.dbg.value(metadata %struct.nsTArray* %aValues, i64 0, metadata !0, metadata !DIExpression()) nounwind, !dbg !DILocation(scope: !1) br label %_ZN8nsTArray9ElementAtEi.exit _ZN8nsTArray9ElementAtEi.exit: ; preds = %for.body @@ -35,4 +35,5 @@ declare %struct.nsTArrayHeader* @_ZN8nsTArray4Hdr2Ev() declare void @llvm.dbg.value(metadata, i64, metadata, metadata) nounwind readnone -!0 = !DILocalVariable(tag: DW_TAG_arg_variable, scope: !DISubprogram()) +!0 = !DILocalVariable(scope: !1) +!1 = distinct !DISubprogram() diff --git a/test/Transforms/LoopStrengthReduce/pr25541.ll b/test/Transforms/LoopStrengthReduce/pr25541.ll new file mode 100644 index 000000000000..011998b90893 --- /dev/null +++ b/test/Transforms/LoopStrengthReduce/pr25541.ll @@ -0,0 +1,48 @@ +; RUN: opt < %s -loop-reduce -S | FileCheck %s +target datalayout = "e-m:w-i64:64-f80:128-n8:16:32:64-S128" +target triple = "x86_64-pc-windows-msvc" + +define void @f() personality i32 (...)* @__CxxFrameHandler3 { +entry: + br label %for.cond.i + +for.cond.i: ; preds = %for.inc.i, %entry + %_First.addr.0.i = phi i32* [ null, %entry ], [ %incdec.ptr.i, %for.inc.i ] + invoke void @g() + to label %for.inc.i unwind label %catch.dispatch.i + +catch.dispatch.i: ; preds = %for.cond.i + %cs = catchswitch within none [label %for.cond.1.preheader.i] unwind to caller + +for.cond.1.preheader.i: ; preds = %catch.dispatch.i + %0 = catchpad within %cs [i8* null, i32 64, i8* null] + %cmp.i = icmp eq i32* %_First.addr.0.i, null + br label %for.cond.1.i + +for.cond.1.i: ; preds = %for.body.i, %for.cond.1.preheader.i + br i1 %cmp.i, label %for.end.i, label %for.body.i + +for.body.i: ; preds = %for.cond.1.i + call void @g() + br label %for.cond.1.i + +for.inc.i: ; preds = %for.cond.i + %incdec.ptr.i = getelementptr inbounds i32, i32* %_First.addr.0.i, i64 1 + br label %for.cond.i + +for.end.i: ; preds = %for.cond.1.i + catchret from %0 to label %leave + +leave: ; preds = %for.end.i + ret void +} + +; CHECK-LABEL: define void @f( +; CHECK: %[[PHI:.*]] = phi i64 [ %[[IV_NEXT:.*]], {{.*}} ], [ 0, {{.*}} ] +; CHECK: %[[ITOP:.*]] = inttoptr i64 %[[PHI]] to i32* +; CHECK: %[[CMP:.*]] = icmp eq i32* %[[ITOP]], null +; CHECK: %[[IV_NEXT]] = add i64 %[[PHI]], -4 + +declare void @g() + +declare i32 @__CxxFrameHandler3(...) diff --git a/test/Transforms/LoopStrengthReduce/quadradic-exit-value.ll b/test/Transforms/LoopStrengthReduce/quadradic-exit-value.ll index 483becc0e7b8..c6d6690e4302 100644 --- a/test/Transforms/LoopStrengthReduce/quadradic-exit-value.ll +++ b/test/Transforms/LoopStrengthReduce/quadradic-exit-value.ll @@ -28,7 +28,7 @@ exit: ; sure they aren't marked as post-inc users. ; ; CHECK-LABEL: IV Users for loop %test2.loop -; CHECK: %sext.us = {0,+,(16777216 + (-16777216 * %sub.us)),+,33554432}<%test2.loop> in %f = ashr i32 %sext.us, 24 +; CHECK: %sext.us = {0,+,(16777216 + (-16777216 * %sub.us))<nuw><nsw>,+,33554432}<%test2.loop> in %f = ashr i32 %sext.us, 24 define i32 @test2() { entry: br label %test2.loop diff --git a/test/Transforms/LoopStrengthReduce/sext-ind-var.ll b/test/Transforms/LoopStrengthReduce/sext-ind-var.ll new file mode 100644 index 000000000000..3cf8f536fa71 --- /dev/null +++ b/test/Transforms/LoopStrengthReduce/sext-ind-var.ll @@ -0,0 +1,140 @@ +; RUN: opt -loop-reduce -S < %s | FileCheck %s + +target datalayout = "e-i64:64-v16:16-v32:32-n16:32:64" +target triple = "nvptx64-unknown-unknown" + +; LSR used not to be able to generate a float* induction variable in +; these cases due to scalar evolution not propagating nsw from an +; instruction to the SCEV, preventing distributing sext into the +; corresponding addrec. + +; Test this pattern: +; +; for (int i = 0; i < numIterations; ++i) +; sum += ptr[i + offset]; +; +define float @testadd(float* %input, i32 %offset, i32 %numIterations) { +; CHECK-LABEL: @testadd +; CHECK: sext i32 %offset to i64 +; CHECK: loop: +; CHECK-DAG: phi float* +; CHECK-DAG: phi i32 +; CHECK-NOT: sext + +entry: + br label %loop + +loop: + %i = phi i32 [ %nexti, %loop ], [ 0, %entry ] + %sum = phi float [ %nextsum, %loop ], [ 0.000000e+00, %entry ] + %index32 = add nuw nsw i32 %i, %offset + %index64 = sext i32 %index32 to i64 + %ptr = getelementptr inbounds float, float* %input, i64 %index64 + %addend = load float, float* %ptr, align 4 + %nextsum = fadd float %sum, %addend + %nexti = add nuw nsw i32 %i, 1 + %exitcond = icmp eq i32 %nexti, %numIterations + br i1 %exitcond, label %exit, label %loop + +exit: + ret float %nextsum +} + +; Test this pattern: +; +; for (int i = 0; i < numIterations; ++i) +; sum += ptr[i - offset]; +; +define float @testsub(float* %input, i32 %offset, i32 %numIterations) { +; CHECK-LABEL: @testsub +; CHECK: sub i32 0, %offset +; CHECK: sext i32 +; CHECK: loop: +; CHECK-DAG: phi float* +; CHECK-DAG: phi i32 +; CHECK-NOT: sext + +entry: + br label %loop + +loop: + %i = phi i32 [ %nexti, %loop ], [ 0, %entry ] + %sum = phi float [ %nextsum, %loop ], [ 0.000000e+00, %entry ] + %index32 = sub nuw nsw i32 %i, %offset + %index64 = sext i32 %index32 to i64 + %ptr = getelementptr inbounds float, float* %input, i64 %index64 + %addend = load float, float* %ptr, align 4 + %nextsum = fadd float %sum, %addend + %nexti = add nuw nsw i32 %i, 1 + %exitcond = icmp eq i32 %nexti, %numIterations + br i1 %exitcond, label %exit, label %loop + +exit: + ret float %nextsum +} + +; Test this pattern: +; +; for (int i = 0; i < numIterations; ++i) +; sum += ptr[i * stride]; +; +define float @testmul(float* %input, i32 %stride, i32 %numIterations) { +; CHECK-LABEL: @testmul +; CHECK: sext i32 %stride to i64 +; CHECK: loop: +; CHECK-DAG: phi float* +; CHECK-DAG: phi i32 +; CHECK-NOT: sext + +entry: + br label %loop + +loop: + %i = phi i32 [ %nexti, %loop ], [ 0, %entry ] + %sum = phi float [ %nextsum, %loop ], [ 0.000000e+00, %entry ] + %index32 = mul nuw nsw i32 %i, %stride + %index64 = sext i32 %index32 to i64 + %ptr = getelementptr inbounds float, float* %input, i64 %index64 + %addend = load float, float* %ptr, align 4 + %nextsum = fadd float %sum, %addend + %nexti = add nuw nsw i32 %i, 1 + %exitcond = icmp eq i32 %nexti, %numIterations + br i1 %exitcond, label %exit, label %loop + +exit: + ret float %nextsum +} + +; Test this pattern: +; +; for (int i = 0; i < numIterations; ++i) +; sum += ptr[3 * (i << 7)]; +; +; The multiplication by 3 is to make the address calculation expensive +; enough to force the introduction of a pointer induction variable. +define float @testshl(float* %input, i32 %numIterations) { +; CHECK-LABEL: @testshl +; CHECK: loop: +; CHECK-DAG: phi float* +; CHECK-DAG: phi i32 +; CHECK-NOT: sext + +entry: + br label %loop + +loop: + %i = phi i32 [ %nexti, %loop ], [ 0, %entry ] + %sum = phi float [ %nextsum, %loop ], [ 0.000000e+00, %entry ] + %index32 = shl nuw nsw i32 %i, 7 + %index32mul = mul nuw nsw i32 %index32, 3 + %index64 = sext i32 %index32mul to i64 + %ptr = getelementptr inbounds float, float* %input, i64 %index64 + %addend = load float, float* %ptr, align 4 + %nextsum = fadd float %sum, %addend + %nexti = add nuw nsw i32 %i, 1 + %exitcond = icmp eq i32 %nexti, %numIterations + br i1 %exitcond, label %exit, label %loop + +exit: + ret float %nextsum +} |
