From b915e9e0fc85ba6f398b3fab0db6a81a8913af94 Mon Sep 17 00:00:00 2001 From: Dimitry Andric Date: Mon, 2 Jan 2017 19:17:04 +0000 Subject: Vendor import of llvm trunk r290819: https://llvm.org/svn/llvm-project/llvm/trunk@290819 --- test/CodeGen/Hexagon/SUnit-boundary-prob.ll | 202 +++++++++++++++++++ test/CodeGen/Hexagon/addh-sext-trunc.ll | 30 +-- test/CodeGen/Hexagon/addr-calc-opt.ll | 54 ++++++ test/CodeGen/Hexagon/anti-dep-partial.mir | 34 ++++ test/CodeGen/Hexagon/bit-gen-rseq.ll | 43 +++++ test/CodeGen/Hexagon/bit-loop-rc-mismatch.ll | 30 +++ test/CodeGen/Hexagon/bit-rie.ll | 208 ++++++++++++++++++++ test/CodeGen/Hexagon/bit-skip-byval.ll | 11 ++ test/CodeGen/Hexagon/bit-validate-reg.ll | 21 ++ test/CodeGen/Hexagon/bit-visit-flowq.ll | 47 +++++ test/CodeGen/Hexagon/block-addr.ll | 7 +- test/CodeGen/Hexagon/branchfolder-keep-impdef.ll | 29 +++ test/CodeGen/Hexagon/build-vector-shuffle.ll | 21 ++ test/CodeGen/Hexagon/combine.ll | 2 +- test/CodeGen/Hexagon/const-pool-tf.ll | 40 ++++ test/CodeGen/Hexagon/constp-clb.ll | 23 +++ test/CodeGen/Hexagon/constp-combine-neg.ll | 27 +++ test/CodeGen/Hexagon/constp-ctb.ll | 26 +++ test/CodeGen/Hexagon/constp-extract.ll | 31 +++ test/CodeGen/Hexagon/constp-physreg.ll | 21 ++ test/CodeGen/Hexagon/constp-rewrite-branches.ll | 17 ++ test/CodeGen/Hexagon/constp-rseq.ll | 19 ++ test/CodeGen/Hexagon/constp-vsplat.ll | 18 ++ test/CodeGen/Hexagon/copy-to-combine-dbg.ll | 57 ++++++ test/CodeGen/Hexagon/dead-store-stack.ll | 131 +++++++++++++ test/CodeGen/Hexagon/early-if-vecpi.ll | 69 +++++++ test/CodeGen/Hexagon/expand-condsets-def-undef.mir | 41 ++++ test/CodeGen/Hexagon/expand-condsets-extend.ll | 112 +++++++++++ test/CodeGen/Hexagon/expand-condsets-impuse.mir | 78 ++++++++ test/CodeGen/Hexagon/expand-condsets-rm-reg.mir | 49 +++++ .../Hexagon/expand-condsets-same-inputs.mir | 32 +++ test/CodeGen/Hexagon/expand-condsets-undef2.ll | 47 +++++ test/CodeGen/Hexagon/expand-vstorerw-undef.ll | 95 +++++++++ test/CodeGen/Hexagon/fixed-spill-mutable.ll | 69 +++++++ test/CodeGen/Hexagon/float-amode.ll | 89 +++++++++ test/CodeGen/Hexagon/fminmax.ll | 27 +++ test/CodeGen/Hexagon/frame-offset-overflow.ll | 163 ++++++++++++++++ test/CodeGen/Hexagon/fsel.ll | 22 +++ test/CodeGen/Hexagon/hwloop-crit-edge.ll | 1 + test/CodeGen/Hexagon/hwloop-loop1.ll | 2 - test/CodeGen/Hexagon/hwloop-noreturn-call.ll | 63 ++++++ test/CodeGen/Hexagon/hwloop-preh.ll | 44 +++++ test/CodeGen/Hexagon/hwloop1.ll | 2 +- .../Hexagon/ifcvt-diamond-bug-2016-08-26.ll | 37 ++++ test/CodeGen/Hexagon/ifcvt-impuse-livein.mir | 42 ++++ test/CodeGen/Hexagon/ifcvt-live-subreg.mir | 50 +++++ test/CodeGen/Hexagon/inline-asm-hexagon.ll | 16 ++ test/CodeGen/Hexagon/inline-asm-i1.ll | 14 ++ test/CodeGen/Hexagon/insert4.ll | 10 +- test/CodeGen/Hexagon/intrinsics/llsc_bundling.ll | 12 ++ test/CodeGen/Hexagon/is-legal-void.ll | 58 ++++++ test/CodeGen/Hexagon/livephysregs-lane-masks.mir | 40 ++++ test/CodeGen/Hexagon/livephysregs-lane-masks2.mir | 55 ++++++ test/CodeGen/Hexagon/long-calls.ll | 73 +++++++ test/CodeGen/Hexagon/loop-prefetch.ll | 27 +++ test/CodeGen/Hexagon/lower-extract-subvector.ll | 47 +++++ .../misaligned_double_vector_store_not_fast.ll | 47 +++++ test/CodeGen/Hexagon/mulhs.ll | 23 +++ test/CodeGen/Hexagon/newvalueSameReg.ll | 63 ++++++ test/CodeGen/Hexagon/opt-spill-volatile.ll | 29 +++ test/CodeGen/Hexagon/packetize-cfi-location.ll | 72 +++++++ test/CodeGen/Hexagon/packetize-return-arg.ll | 37 ++++ test/CodeGen/Hexagon/peephole-kill-flags.ll | 27 +++ test/CodeGen/Hexagon/pic-simple.ll | 2 +- test/CodeGen/Hexagon/pic-static.ll | 2 +- test/CodeGen/Hexagon/post-inc-aa-metadata.ll | 37 ++++ test/CodeGen/Hexagon/post-ra-kill-update.mir | 37 ++++ test/CodeGen/Hexagon/propagate-vcombine.ll | 48 +++++ test/CodeGen/Hexagon/rdf-copy.ll | 2 +- test/CodeGen/Hexagon/rdf-extra-livein.ll | 73 +++++++ test/CodeGen/Hexagon/rdf-filter-defs.ll | 214 +++++++++++++++++++++ test/CodeGen/Hexagon/rdf-ignore-undef.ll | 55 ++++++ test/CodeGen/Hexagon/rdf-multiple-phis-up.ll | 40 ++++ test/CodeGen/Hexagon/rdf-phi-shadows.ll | 64 ++++++ test/CodeGen/Hexagon/rdf-phi-up.ll | 60 ++++++ test/CodeGen/Hexagon/regalloc-bad-undef.mir | 204 ++++++++++++++++++++ test/CodeGen/Hexagon/sf-min-max.ll | 67 +++++++ test/CodeGen/Hexagon/sffms.ll | 25 +++ test/CodeGen/Hexagon/split-const32-const64.ll | 18 +- test/CodeGen/Hexagon/storerd-io-over-rr.ll | 12 ++ test/CodeGen/Hexagon/struct_args.ll | 8 +- test/CodeGen/Hexagon/subi-asl.ll | 70 +++++++ test/CodeGen/Hexagon/swp-const-tc.ll | 51 +++++ test/CodeGen/Hexagon/swp-dag-phi.ll | 42 ++++ test/CodeGen/Hexagon/swp-epilog-phi10.ll | 88 +++++++++ test/CodeGen/Hexagon/swp-epilog-reuse-1.ll | 44 +++++ test/CodeGen/Hexagon/swp-epilog-reuse.ll | 65 +++++++ test/CodeGen/Hexagon/swp-matmul-bitext.ll | 75 ++++++++ test/CodeGen/Hexagon/swp-max.ll | 42 ++++ test/CodeGen/Hexagon/swp-multi-loops.ll | 75 ++++++++ test/CodeGen/Hexagon/swp-prolog-phi4.ll | 65 +++++++ test/CodeGen/Hexagon/swp-vect-dotprod.ll | 41 ++++ test/CodeGen/Hexagon/swp-vmult.ll | 33 ++++ test/CodeGen/Hexagon/swp-vsum.ll | 29 +++ test/CodeGen/Hexagon/tailcall_fastcc_ccc.ll | 22 +++ test/CodeGen/Hexagon/tls_static.ll | 2 +- test/CodeGen/Hexagon/two-crash.ll | 23 +++ test/CodeGen/Hexagon/v60-cur.ll | 2 +- test/CodeGen/Hexagon/v60-vsel1.ll | 69 +++++++ test/CodeGen/Hexagon/v6vec-vprint.ll | 36 ++++ test/CodeGen/Hexagon/vassign-to-combine.ll | 56 ++++++ test/CodeGen/Hexagon/vdmpy-halide-test.ll | 167 ++++++++++++++++ test/CodeGen/Hexagon/vect/vect-vsplatb.ll | 2 +- test/CodeGen/Hexagon/vect/vect-vsplath.ll | 2 +- test/CodeGen/Hexagon/vector-ext-load.ll | 10 + test/CodeGen/Hexagon/vmpa-halide-test.ll | 145 ++++++++++++++ test/CodeGen/Hexagon/vpack_eo.ll | 73 +++++++ 107 files changed, 5174 insertions(+), 56 deletions(-) create mode 100644 test/CodeGen/Hexagon/SUnit-boundary-prob.ll create mode 100644 test/CodeGen/Hexagon/addr-calc-opt.ll create mode 100644 test/CodeGen/Hexagon/anti-dep-partial.mir create mode 100644 test/CodeGen/Hexagon/bit-gen-rseq.ll create mode 100644 test/CodeGen/Hexagon/bit-loop-rc-mismatch.ll create mode 100644 test/CodeGen/Hexagon/bit-rie.ll create mode 100644 test/CodeGen/Hexagon/bit-skip-byval.ll create mode 100644 test/CodeGen/Hexagon/bit-validate-reg.ll create mode 100644 test/CodeGen/Hexagon/bit-visit-flowq.ll create mode 100644 test/CodeGen/Hexagon/branchfolder-keep-impdef.ll create mode 100644 test/CodeGen/Hexagon/build-vector-shuffle.ll create mode 100644 test/CodeGen/Hexagon/const-pool-tf.ll create mode 100644 test/CodeGen/Hexagon/constp-clb.ll create mode 100644 test/CodeGen/Hexagon/constp-combine-neg.ll create mode 100644 test/CodeGen/Hexagon/constp-ctb.ll create mode 100644 test/CodeGen/Hexagon/constp-extract.ll create mode 100644 test/CodeGen/Hexagon/constp-physreg.ll create mode 100644 test/CodeGen/Hexagon/constp-rewrite-branches.ll create mode 100644 test/CodeGen/Hexagon/constp-rseq.ll create mode 100644 test/CodeGen/Hexagon/constp-vsplat.ll create mode 100644 test/CodeGen/Hexagon/copy-to-combine-dbg.ll create mode 100644 test/CodeGen/Hexagon/dead-store-stack.ll create mode 100644 test/CodeGen/Hexagon/early-if-vecpi.ll create mode 100644 test/CodeGen/Hexagon/expand-condsets-def-undef.mir create mode 100644 test/CodeGen/Hexagon/expand-condsets-extend.ll create mode 100644 test/CodeGen/Hexagon/expand-condsets-impuse.mir create mode 100644 test/CodeGen/Hexagon/expand-condsets-rm-reg.mir create mode 100644 test/CodeGen/Hexagon/expand-condsets-same-inputs.mir create mode 100644 test/CodeGen/Hexagon/expand-condsets-undef2.ll create mode 100644 test/CodeGen/Hexagon/expand-vstorerw-undef.ll create mode 100644 test/CodeGen/Hexagon/fixed-spill-mutable.ll create mode 100644 test/CodeGen/Hexagon/float-amode.ll create mode 100644 test/CodeGen/Hexagon/fminmax.ll create mode 100644 test/CodeGen/Hexagon/frame-offset-overflow.ll create mode 100644 test/CodeGen/Hexagon/fsel.ll create mode 100644 test/CodeGen/Hexagon/hwloop-noreturn-call.ll create mode 100644 test/CodeGen/Hexagon/hwloop-preh.ll create mode 100644 test/CodeGen/Hexagon/ifcvt-diamond-bug-2016-08-26.ll create mode 100644 test/CodeGen/Hexagon/ifcvt-impuse-livein.mir create mode 100644 test/CodeGen/Hexagon/ifcvt-live-subreg.mir create mode 100644 test/CodeGen/Hexagon/inline-asm-hexagon.ll create mode 100644 test/CodeGen/Hexagon/inline-asm-i1.ll create mode 100644 test/CodeGen/Hexagon/intrinsics/llsc_bundling.ll create mode 100644 test/CodeGen/Hexagon/is-legal-void.ll create mode 100644 test/CodeGen/Hexagon/livephysregs-lane-masks.mir create mode 100644 test/CodeGen/Hexagon/livephysregs-lane-masks2.mir create mode 100644 test/CodeGen/Hexagon/long-calls.ll create mode 100644 test/CodeGen/Hexagon/loop-prefetch.ll create mode 100644 test/CodeGen/Hexagon/lower-extract-subvector.ll create mode 100644 test/CodeGen/Hexagon/misaligned_double_vector_store_not_fast.ll create mode 100644 test/CodeGen/Hexagon/mulhs.ll create mode 100644 test/CodeGen/Hexagon/newvalueSameReg.ll create mode 100644 test/CodeGen/Hexagon/opt-spill-volatile.ll create mode 100644 test/CodeGen/Hexagon/packetize-cfi-location.ll create mode 100644 test/CodeGen/Hexagon/packetize-return-arg.ll create mode 100644 test/CodeGen/Hexagon/peephole-kill-flags.ll create mode 100644 test/CodeGen/Hexagon/post-inc-aa-metadata.ll create mode 100644 test/CodeGen/Hexagon/post-ra-kill-update.mir create mode 100644 test/CodeGen/Hexagon/propagate-vcombine.ll create mode 100644 test/CodeGen/Hexagon/rdf-extra-livein.ll create mode 100644 test/CodeGen/Hexagon/rdf-filter-defs.ll create mode 100644 test/CodeGen/Hexagon/rdf-ignore-undef.ll create mode 100644 test/CodeGen/Hexagon/rdf-multiple-phis-up.ll create mode 100644 test/CodeGen/Hexagon/rdf-phi-shadows.ll create mode 100644 test/CodeGen/Hexagon/rdf-phi-up.ll create mode 100644 test/CodeGen/Hexagon/regalloc-bad-undef.mir create mode 100644 test/CodeGen/Hexagon/sf-min-max.ll create mode 100644 test/CodeGen/Hexagon/sffms.ll create mode 100644 test/CodeGen/Hexagon/storerd-io-over-rr.ll create mode 100644 test/CodeGen/Hexagon/subi-asl.ll create mode 100644 test/CodeGen/Hexagon/swp-const-tc.ll create mode 100644 test/CodeGen/Hexagon/swp-dag-phi.ll create mode 100644 test/CodeGen/Hexagon/swp-epilog-phi10.ll create mode 100644 test/CodeGen/Hexagon/swp-epilog-reuse-1.ll create mode 100644 test/CodeGen/Hexagon/swp-epilog-reuse.ll create mode 100644 test/CodeGen/Hexagon/swp-matmul-bitext.ll create mode 100644 test/CodeGen/Hexagon/swp-max.ll create mode 100644 test/CodeGen/Hexagon/swp-multi-loops.ll create mode 100644 test/CodeGen/Hexagon/swp-prolog-phi4.ll create mode 100644 test/CodeGen/Hexagon/swp-vect-dotprod.ll create mode 100644 test/CodeGen/Hexagon/swp-vmult.ll create mode 100644 test/CodeGen/Hexagon/swp-vsum.ll create mode 100644 test/CodeGen/Hexagon/tailcall_fastcc_ccc.ll create mode 100644 test/CodeGen/Hexagon/two-crash.ll create mode 100644 test/CodeGen/Hexagon/v60-vsel1.ll create mode 100644 test/CodeGen/Hexagon/v6vec-vprint.ll create mode 100644 test/CodeGen/Hexagon/vassign-to-combine.ll create mode 100644 test/CodeGen/Hexagon/vdmpy-halide-test.ll create mode 100644 test/CodeGen/Hexagon/vector-ext-load.ll create mode 100644 test/CodeGen/Hexagon/vmpa-halide-test.ll create mode 100644 test/CodeGen/Hexagon/vpack_eo.ll (limited to 'test/CodeGen/Hexagon') diff --git a/test/CodeGen/Hexagon/SUnit-boundary-prob.ll b/test/CodeGen/Hexagon/SUnit-boundary-prob.ll new file mode 100644 index 000000000000..9df178f9907c --- /dev/null +++ b/test/CodeGen/Hexagon/SUnit-boundary-prob.ll @@ -0,0 +1,202 @@ +; RUN: llc -march=hexagon -O2 -mcpu=hexagonv60 < %s | FileCheck %s +; This was aborting while processing SUnits. + +; CHECK: vmem + +source_filename = "bugpoint-output-bdb0052.bc" +target datalayout = "e-m:e-p:32:32:32-a:0-n16:32-i64:64:64-i32:32:32-i16:16:16-i1:8:8-f32:32:32-f64:64:64-v32:32:32-v64:64:64-v512:512:512-v1024:1024:1024-v2048:2048:2048" +target triple = "hexagon-unknown--elf" + +; Function Attrs: nounwind readnone +declare <16 x i32> @llvm.hexagon.V6.lo(<32 x i32>) #0 + +; Function Attrs: nounwind readnone +declare <16 x i32> @llvm.hexagon.V6.hi(<32 x i32>) #0 + +; Function Attrs: nounwind readnone +declare <32 x i32> @llvm.hexagon.V6.vshuffvdd(<16 x i32>, <16 x i32>, i32) #0 + +; Function Attrs: nounwind readnone +declare <32 x i32> @llvm.hexagon.V6.vdealvdd(<16 x i32>, <16 x i32>, i32) #0 + +; Function Attrs: nounwind readnone +declare <32 x i32> @llvm.hexagon.V6.vcombine(<16 x i32>, <16 x i32>) #0 + +; Function Attrs: nounwind readnone +declare <16 x i32> @llvm.hexagon.V6.vshufeh(<16 x i32>, <16 x i32>) #0 + +; Function Attrs: nounwind readnone +declare <16 x i32> @llvm.hexagon.V6.vshufoh(<16 x i32>, <16 x i32>) #0 + +; Function Attrs: nounwind readnone +declare <32 x i32> @llvm.hexagon.V6.vmpyuhv(<16 x i32>, <16 x i32>) #0 + +; Function Attrs: nounwind readnone +declare <16 x i32> @llvm.hexagon.V6.vaslw.acc(<16 x i32>, <16 x i32>, i32) #0 + +define void @__error_op_vmpy_v__uh_v__uh__1() #1 { +entry: + %in_u16.host181 = load i16*, i16** undef, align 4 + %in_u32.host182 = load i32*, i32** undef, align 4 + br label %"for op_vmpy_v__uh_v__uh__1.s0.y" + +"for op_vmpy_v__uh_v__uh__1.s0.y": ; preds = %"end for op_vmpy_v__uh_v__uh__1.s0.x.x", %entry + %op_vmpy_v__uh_v__uh__1.s0.y = phi i32 [ 0, %entry ], [ %63, %"end for op_vmpy_v__uh_v__uh__1.s0.x.x" ] + %0 = mul nuw nsw i32 %op_vmpy_v__uh_v__uh__1.s0.y, 768 + %1 = add nuw nsw i32 %0, 32 + %2 = add nuw nsw i32 %0, 64 + %3 = add nuw nsw i32 %0, 96 + br label %"for op_vmpy_v__uh_v__uh__1.s0.x.x" + +"for op_vmpy_v__uh_v__uh__1.s0.x.x": ; preds = %"for op_vmpy_v__uh_v__uh__1.s0.x.x", %"for op_vmpy_v__uh_v__uh__1.s0.y" + %.phi210 = phi i32* [ %in_u32.host182, %"for op_vmpy_v__uh_v__uh__1.s0.y" ], [ %.inc211.3, %"for op_vmpy_v__uh_v__uh__1.s0.x.x" ] + %.phi213 = phi i16* [ %in_u16.host181, %"for op_vmpy_v__uh_v__uh__1.s0.y" ], [ %.inc214.3, %"for op_vmpy_v__uh_v__uh__1.s0.x.x" ] + %op_vmpy_v__uh_v__uh__1.s0.x.x = phi i32 [ 0, %"for op_vmpy_v__uh_v__uh__1.s0.y" ], [ %61, %"for op_vmpy_v__uh_v__uh__1.s0.x.x" ] + %4 = mul nuw nsw i32 %op_vmpy_v__uh_v__uh__1.s0.x.x, 32 + %5 = bitcast i32* %.phi210 to <16 x i32>* + %6 = load <16 x i32>, <16 x i32>* %5, align 64, !tbaa !1 + %7 = add nuw nsw i32 %4, 16 + %8 = getelementptr inbounds i32, i32* %in_u32.host182, i32 %7 + %9 = bitcast i32* %8 to <16 x i32>* + %10 = load <16 x i32>, <16 x i32>* %9, align 64, !tbaa !1 + %11 = tail call <32 x i32> @llvm.hexagon.V6.vcombine(<16 x i32> %10, <16 x i32> %6) + %e.i = tail call <16 x i32> @llvm.hexagon.V6.lo(<32 x i32> %11) #2 + %o.i = tail call <16 x i32> @llvm.hexagon.V6.hi(<32 x i32> %11) #2 + %r.i = tail call <32 x i32> @llvm.hexagon.V6.vdealvdd(<16 x i32> %o.i, <16 x i32> %e.i, i32 -4) #2 + %12 = bitcast i16* %.phi213 to <16 x i32>* + %13 = load <16 x i32>, <16 x i32>* %12, align 64, !tbaa !4 + %a_lo.i = tail call <16 x i32> @llvm.hexagon.V6.lo(<32 x i32> %r.i) #2 + %a_hi.i = tail call <16 x i32> @llvm.hexagon.V6.hi(<32 x i32> %r.i) #2 + %a_e.i = tail call <16 x i32> @llvm.hexagon.V6.vshufeh(<16 x i32> %a_hi.i, <16 x i32> %a_lo.i) #2 + %a_o.i = tail call <16 x i32> @llvm.hexagon.V6.vshufoh(<16 x i32> %a_hi.i, <16 x i32> %a_lo.i) #2 + %ab_e.i = tail call <32 x i32> @llvm.hexagon.V6.vmpyuhv(<16 x i32> %a_e.i, <16 x i32> %13) #2 + %ab_o.i = tail call <32 x i32> @llvm.hexagon.V6.vmpyuhv(<16 x i32> %a_o.i, <16 x i32> %13) #2 + %a_lo.i.i = tail call <16 x i32> @llvm.hexagon.V6.lo(<32 x i32> %ab_e.i) #2 + %l_lo.i.i = tail call <16 x i32> @llvm.hexagon.V6.lo(<32 x i32> %ab_o.i) #2 + %s_lo.i.i = tail call <16 x i32> @llvm.hexagon.V6.vaslw.acc(<16 x i32> %a_lo.i.i, <16 x i32> %l_lo.i.i, i32 16) #2 + %l_hi.i.i = tail call <16 x i32> @llvm.hexagon.V6.hi(<32 x i32> %ab_o.i) #2 + %s_hi.i.i = tail call <16 x i32> @llvm.hexagon.V6.vaslw.acc(<16 x i32> undef, <16 x i32> %l_hi.i.i, i32 16) #2 + %s.i.i = tail call <32 x i32> @llvm.hexagon.V6.vcombine(<16 x i32> %s_hi.i.i, <16 x i32> %s_lo.i.i) #2 + %e.i189 = tail call <16 x i32> @llvm.hexagon.V6.lo(<32 x i32> %s.i.i) #2 + %o.i190 = tail call <16 x i32> @llvm.hexagon.V6.hi(<32 x i32> %s.i.i) #2 + %r.i191 = tail call <32 x i32> @llvm.hexagon.V6.vshuffvdd(<16 x i32> %o.i190, <16 x i32> %e.i189, i32 -4) #2 + %14 = tail call <16 x i32> @llvm.hexagon.V6.lo(<32 x i32> %r.i191) + %15 = add nuw nsw i32 %4, %0 + %16 = getelementptr inbounds i32, i32* undef, i32 %15 + %17 = bitcast i32* %16 to <16 x i32>* + store <16 x i32> %14, <16 x i32>* %17, align 64, !tbaa !6 + %18 = tail call <16 x i32> @llvm.hexagon.V6.hi(<32 x i32> %r.i191) + store <16 x i32> %18, <16 x i32>* undef, align 64, !tbaa !6 + %.inc211 = getelementptr i32, i32* %.phi210, i32 32 + %.inc214 = getelementptr i16, i16* %.phi213, i32 32 + %19 = bitcast i32* %.inc211 to <16 x i32>* + %20 = load <16 x i32>, <16 x i32>* %19, align 64, !tbaa !1 + %21 = add nuw nsw i32 %4, 48 + %22 = getelementptr inbounds i32, i32* %in_u32.host182, i32 %21 + %23 = bitcast i32* %22 to <16 x i32>* + %24 = load <16 x i32>, <16 x i32>* %23, align 64, !tbaa !1 + %25 = tail call <32 x i32> @llvm.hexagon.V6.vcombine(<16 x i32> %24, <16 x i32> %20) + %e.i.1 = tail call <16 x i32> @llvm.hexagon.V6.lo(<32 x i32> %25) #2 + %r.i.1 = tail call <32 x i32> @llvm.hexagon.V6.vdealvdd(<16 x i32> undef, <16 x i32> %e.i.1, i32 -4) #2 + %26 = bitcast i16* %.inc214 to <16 x i32>* + %27 = load <16 x i32>, <16 x i32>* %26, align 64, !tbaa !4 + %a_lo.i.1 = tail call <16 x i32> @llvm.hexagon.V6.lo(<32 x i32> %r.i.1) #2 + %a_e.i.1 = tail call <16 x i32> @llvm.hexagon.V6.vshufeh(<16 x i32> undef, <16 x i32> %a_lo.i.1) #2 + %a_o.i.1 = tail call <16 x i32> @llvm.hexagon.V6.vshufoh(<16 x i32> undef, <16 x i32> %a_lo.i.1) #2 + %ab_e.i.1 = tail call <32 x i32> @llvm.hexagon.V6.vmpyuhv(<16 x i32> %a_e.i.1, <16 x i32> %27) #2 + %ab_o.i.1 = tail call <32 x i32> @llvm.hexagon.V6.vmpyuhv(<16 x i32> %a_o.i.1, <16 x i32> %27) #2 + %a_lo.i.i.1 = tail call <16 x i32> @llvm.hexagon.V6.lo(<32 x i32> %ab_e.i.1) #2 + %s_lo.i.i.1 = tail call <16 x i32> @llvm.hexagon.V6.vaslw.acc(<16 x i32> %a_lo.i.i.1, <16 x i32> undef, i32 16) #2 + %a_hi.i.i.1 = tail call <16 x i32> @llvm.hexagon.V6.hi(<32 x i32> %ab_e.i.1) #2 + %l_hi.i.i.1 = tail call <16 x i32> @llvm.hexagon.V6.hi(<32 x i32> %ab_o.i.1) #2 + %s_hi.i.i.1 = tail call <16 x i32> @llvm.hexagon.V6.vaslw.acc(<16 x i32> %a_hi.i.i.1, <16 x i32> %l_hi.i.i.1, i32 16) #2 + %s.i.i.1 = tail call <32 x i32> @llvm.hexagon.V6.vcombine(<16 x i32> %s_hi.i.i.1, <16 x i32> %s_lo.i.i.1) #2 + %e.i189.1 = tail call <16 x i32> @llvm.hexagon.V6.lo(<32 x i32> %s.i.i.1) #2 + %o.i190.1 = tail call <16 x i32> @llvm.hexagon.V6.hi(<32 x i32> %s.i.i.1) #2 + %r.i191.1 = tail call <32 x i32> @llvm.hexagon.V6.vshuffvdd(<16 x i32> %o.i190.1, <16 x i32> %e.i189.1, i32 -4) #2 + %28 = tail call <16 x i32> @llvm.hexagon.V6.lo(<32 x i32> %r.i191.1) + %29 = add nuw nsw i32 %1, %4 + %30 = getelementptr inbounds i32, i32* undef, i32 %29 + %31 = bitcast i32* %30 to <16 x i32>* + store <16 x i32> %28, <16 x i32>* %31, align 64, !tbaa !6 + %32 = tail call <16 x i32> @llvm.hexagon.V6.hi(<32 x i32> %r.i191.1) + %33 = add nuw nsw i32 %29, 16 + %34 = getelementptr inbounds i32, i32* undef, i32 %33 + %35 = bitcast i32* %34 to <16 x i32>* + store <16 x i32> %32, <16 x i32>* %35, align 64, !tbaa !6 + %.inc211.1 = getelementptr i32, i32* %.phi210, i32 64 + %.inc214.1 = getelementptr i16, i16* %.phi213, i32 64 + %36 = bitcast i32* %.inc211.1 to <16 x i32>* + %37 = load <16 x i32>, <16 x i32>* %36, align 64, !tbaa !1 + %38 = add nuw nsw i32 %4, 80 + %39 = getelementptr inbounds i32, i32* %in_u32.host182, i32 %38 + %40 = bitcast i32* %39 to <16 x i32>* + %41 = load <16 x i32>, <16 x i32>* %40, align 64, !tbaa !1 + %42 = tail call <32 x i32> @llvm.hexagon.V6.vcombine(<16 x i32> %41, <16 x i32> %37) + %e.i.2 = tail call <16 x i32> @llvm.hexagon.V6.lo(<32 x i32> %42) #2 + %o.i.2 = tail call <16 x i32> @llvm.hexagon.V6.hi(<32 x i32> %42) #2 + %r.i.2 = tail call <32 x i32> @llvm.hexagon.V6.vdealvdd(<16 x i32> %o.i.2, <16 x i32> %e.i.2, i32 -4) #2 + %43 = bitcast i16* %.inc214.1 to <16 x i32>* + %44 = load <16 x i32>, <16 x i32>* %43, align 64, !tbaa !4 + %a_lo.i.2 = tail call <16 x i32> @llvm.hexagon.V6.lo(<32 x i32> %r.i.2) #2 + %a_hi.i.2 = tail call <16 x i32> @llvm.hexagon.V6.hi(<32 x i32> %r.i.2) #2 + %a_e.i.2 = tail call <16 x i32> @llvm.hexagon.V6.vshufeh(<16 x i32> %a_hi.i.2, <16 x i32> %a_lo.i.2) #2 + %a_o.i.2 = tail call <16 x i32> @llvm.hexagon.V6.vshufoh(<16 x i32> %a_hi.i.2, <16 x i32> %a_lo.i.2) #2 + %ab_e.i.2 = tail call <32 x i32> @llvm.hexagon.V6.vmpyuhv(<16 x i32> %a_e.i.2, <16 x i32> %44) #2 + %ab_o.i.2 = tail call <32 x i32> @llvm.hexagon.V6.vmpyuhv(<16 x i32> %a_o.i.2, <16 x i32> %44) #2 + %l_lo.i.i.2 = tail call <16 x i32> @llvm.hexagon.V6.lo(<32 x i32> %ab_o.i.2) #2 + %s_lo.i.i.2 = tail call <16 x i32> @llvm.hexagon.V6.vaslw.acc(<16 x i32> undef, <16 x i32> %l_lo.i.i.2, i32 16) #2 + %a_hi.i.i.2 = tail call <16 x i32> @llvm.hexagon.V6.hi(<32 x i32> %ab_e.i.2) #2 + %l_hi.i.i.2 = tail call <16 x i32> @llvm.hexagon.V6.hi(<32 x i32> %ab_o.i.2) #2 + %s_hi.i.i.2 = tail call <16 x i32> @llvm.hexagon.V6.vaslw.acc(<16 x i32> %a_hi.i.i.2, <16 x i32> %l_hi.i.i.2, i32 16) #2 + %s.i.i.2 = tail call <32 x i32> @llvm.hexagon.V6.vcombine(<16 x i32> %s_hi.i.i.2, <16 x i32> %s_lo.i.i.2) #2 + %e.i189.2 = tail call <16 x i32> @llvm.hexagon.V6.lo(<32 x i32> %s.i.i.2) #2 + %o.i190.2 = tail call <16 x i32> @llvm.hexagon.V6.hi(<32 x i32> %s.i.i.2) #2 + %r.i191.2 = tail call <32 x i32> @llvm.hexagon.V6.vshuffvdd(<16 x i32> %o.i190.2, <16 x i32> %e.i189.2, i32 -4) #2 + %45 = tail call <16 x i32> @llvm.hexagon.V6.lo(<32 x i32> %r.i191.2) + %46 = add nuw nsw i32 %2, %4 + %47 = getelementptr inbounds i32, i32* undef, i32 %46 + %48 = bitcast i32* %47 to <16 x i32>* + store <16 x i32> %45, <16 x i32>* %48, align 64, !tbaa !6 + %49 = tail call <16 x i32> @llvm.hexagon.V6.hi(<32 x i32> %r.i191.2) + %50 = add nuw nsw i32 %46, 16 + %51 = getelementptr inbounds i32, i32* undef, i32 %50 + %52 = bitcast i32* %51 to <16 x i32>* + store <16 x i32> %49, <16 x i32>* %52, align 64, !tbaa !6 + %e.i189.3 = tail call <16 x i32> @llvm.hexagon.V6.lo(<32 x i32> undef) #2 + %r.i191.3 = tail call <32 x i32> @llvm.hexagon.V6.vshuffvdd(<16 x i32> undef, <16 x i32> %e.i189.3, i32 -4) #2 + %53 = tail call <16 x i32> @llvm.hexagon.V6.lo(<32 x i32> %r.i191.3) + %54 = add nuw nsw i32 %3, %4 + %55 = getelementptr inbounds i32, i32* undef, i32 %54 + %56 = bitcast i32* %55 to <16 x i32>* + store <16 x i32> %53, <16 x i32>* %56, align 64, !tbaa !6 + %57 = tail call <16 x i32> @llvm.hexagon.V6.hi(<32 x i32> %r.i191.3) + %58 = add nuw nsw i32 %54, 16 + %59 = getelementptr inbounds i32, i32* undef, i32 %58 + %60 = bitcast i32* %59 to <16 x i32>* + store <16 x i32> %57, <16 x i32>* %60, align 64, !tbaa !6 + %61 = add nuw nsw i32 %op_vmpy_v__uh_v__uh__1.s0.x.x, 4 + %62 = icmp eq i32 %61, 24 + %.inc211.3 = getelementptr i32, i32* %.phi210, i32 128 + %.inc214.3 = getelementptr i16, i16* %.phi213, i32 128 + br i1 %62, label %"end for op_vmpy_v__uh_v__uh__1.s0.x.x", label %"for op_vmpy_v__uh_v__uh__1.s0.x.x" + +"end for op_vmpy_v__uh_v__uh__1.s0.x.x": ; preds = %"for op_vmpy_v__uh_v__uh__1.s0.x.x" + %63 = add nuw nsw i32 %op_vmpy_v__uh_v__uh__1.s0.y, 1 + br label %"for op_vmpy_v__uh_v__uh__1.s0.y" +} + +attributes #0 = { nounwind readnone } +attributes #1 = { "target-cpu"="hexagonv60" "target-features"="+hvx" } +attributes #2 = { nounwind } + +!llvm.module.flags = !{!0} + +!0 = !{i32 2, !"halide_mattrs", !"+hvx"} +!1 = !{!2, !2, i64 0} +!2 = !{!"in_u32", !3} +!3 = !{!"Halide buffer"} +!4 = !{!5, !5, i64 0} +!5 = !{!"in_u16", !3} +!6 = !{!7, !7, i64 0} +!7 = !{!"op_vmpy_v__uh_v__uh__1", !3} diff --git a/test/CodeGen/Hexagon/addh-sext-trunc.ll b/test/CodeGen/Hexagon/addh-sext-trunc.ll index 094932933fbc..7f219944436b 100644 --- a/test/CodeGen/Hexagon/addh-sext-trunc.ll +++ b/test/CodeGen/Hexagon/addh-sext-trunc.ll @@ -4,37 +4,17 @@ target datalayout = "e-p:32:32:32-i64:64:64-i32:32:32-i16:16:16-i1:32:32-f64:64:64-f32:32:32-v64:64:64-v32:32:32-a0:0-n16:32" target triple = "hexagon-unknown-none" -%struct.aDataType = type { i16, i16, i16, i16, i16, i16*, i16*, i16*, i8*, i16*, i16*, i16*, i8* } -define i8* @a_get_score(%struct.aDataType* nocapture %pData, i16 signext %gmmModelIndex, i16* nocapture %pGmmScoreL16Q4) #0 { -entry: - %numSubVector = getelementptr inbounds %struct.aDataType, %struct.aDataType* %pData, i32 0, i32 3 - %0 = load i16, i16* %numSubVector, align 2, !tbaa !0 - %and = and i16 %0, -4 - %b = getelementptr inbounds %struct.aDataType, %struct.aDataType* %pData, i32 0, i32 8 - %1 = load i8*, i8** %b, align 4, !tbaa !3 +define i32 @foo(i16 %a, i32 %b) #0 { + %and = and i16 %a, -4 %conv3 = sext i16 %and to i32 - %cmp21 = icmp sgt i16 %and, 0 - br i1 %cmp21, label %for.inc.preheader, label %for.end - -for.inc.preheader: ; preds = %entry - br label %for.inc - -for.inc: ; preds = %for.inc.preheader, %for.inc - %j.022 = phi i32 [ %phitmp, %for.inc ], [ 0, %for.inc.preheader ] - %add13 = mul i32 %j.022, 65536 + %add13 = mul i32 %b, 65536 %sext = add i32 %add13, 262144 %phitmp = ashr exact i32 %sext, 16 - %cmp = icmp slt i32 %phitmp, %conv3 - br i1 %cmp, label %for.inc, label %for.end.loopexit - -for.end.loopexit: ; preds = %for.inc - br label %for.end - -for.end: ; preds = %for.end.loopexit, %entry - ret i8* %1 + ret i32 %phitmp } + attributes #0 = { nounwind readonly "less-precise-fpmad"="false" "no-frame-pointer-elim"="false" "no-frame-pointer-elim-non-leaf"="true" "no-infs-fp-math"="false" "no-nans-fp-math"="false" "unsafe-fp-math"="false" "use-soft-float"="false" } !0 = !{!"short", !1} diff --git a/test/CodeGen/Hexagon/addr-calc-opt.ll b/test/CodeGen/Hexagon/addr-calc-opt.ll new file mode 100644 index 000000000000..9bd010d52364 --- /dev/null +++ b/test/CodeGen/Hexagon/addr-calc-opt.ll @@ -0,0 +1,54 @@ +; RUN: llc -march=hexagon -mcpu=hexagonv5 < %s | FileCheck %s +; +; Test whether we can produce minimal code for this complex address +; calculation. +; + +; CHECK: r0 = memub(r{{[0-9]+}}<<#3{{ *}}+{{ *}}##the_global+516) + +%0 = type { [3 x %1] } +%1 = type { %2, i8, i8, i8, i8, i8, [4 x i8], i8, [10 x i8], [10 x i8], [10 x i8], i8, [3 x %4], i16, i16, i16, i16, i32, i8, [4 x i8], i8, i8, i8, i8, %5, i8, i8, i8, i8, i8, i16, i8, i8, i8, i16, i16, i8, i8, [2 x i8], [2 x i8], i8, i8, i8, i8, i8, i16, i16, i8, i8, i8, i8, i8, i8, %9, i8, [6 x [2 x i8]], i16, i32, %10, [28 x i8], [4 x %17] } +%2 = type { %3 } +%3 = type { i8, i8, i8, i8, i8, i16, i16, i16, i16, i16 } +%4 = type { i16, i16 } +%5 = type { [10 x %6] } +%6 = type { [2 x %7] } +%7 = type { i8, [2 x %8] } +%8 = type { [4 x i8] } +%9 = type { i8 } +%10 = type { %11, %13 } +%11 = type { [2 x [2 x i8]], [2 x [8 x %12]], [6 x i16], [6 x i16] } +%12 = type { i8, i8 } +%13 = type { [4 x %12], [4 x %12], [2 x [4 x %14]], [6 x i16] } +%14 = type { %15, %16 } +%15 = type { i8, i8 } +%16 = type { i8, i8 } +%17 = type { i8, i8, %1*, i16, i16, i16, i64, i32, i32, %18, i8, %21, i8, [2 x i16], i8, i8, i8, i8, i8, i8, i8, i8, i8, i8, i8, i8, i8, i8, i8, i8, i8, i8, i8, i8, i8, i8, i16, i16, i16, i8, i8, i8, i16, i16, [2 x i16], i16, [2 x i32], [2 x i16], [2 x i16], i8, i8, [6 x %23], i8, i8, i8, %24, %25, %26, %28 } +%18 = type { %19, [10 x %20] } +%19 = type { i32 } +%20 = type { [2 x i8], [2 x i8], i8, i8, i8, i8 } +%21 = type { i8, i8, i8, [8 x %22] } +%22 = type { i8, i8, i8, i32 } +%23 = type { i32, i16, i16, [2 x i16], [2 x i16], [2 x i16], i32 } +%24 = type { [2 x i32], [2 x i64*], [2 x i64*], [2 x i64*], [2 x i32], [2 x i32], i32 } +%25 = type { [2 x i32], [2 x i32], [2 x i32] } +%26 = type { i8, i8, i8, i16, i16, %27, i32, i32, i32, i16 } +%27 = type { i64 } +%28 = type { %29, %31, [24 x i8] } +%29 = type { [2 x %30], [16 x i32] } +%30 = type { [16 x i32], [8 x i32], [16 x i32], [64 x i32], [2 x i32], i64, i32, i32, i32, i32 } +%31 = type { [2 x %32] } +%32 = type { [4 x %33], i32 } +%33 = type { i32, i32 } + +@the_global = external global %0 + +; Function Attrs: nounwind optsize readonly ssp +define zeroext i8 @myFun(i8 zeroext, i8 zeroext) { + %3 = zext i8 %1 to i32 + %4 = zext i8 %0 to i32 + %5 = getelementptr inbounds %0, %0* @the_global, i32 0, i32 0, i32 %4, i32 60, i32 0, i32 9, i32 1, i32 %3, i32 0, i32 0 + %6 = load i8, i8* %5, align 4 + ret i8 %6 +} + diff --git a/test/CodeGen/Hexagon/anti-dep-partial.mir b/test/CodeGen/Hexagon/anti-dep-partial.mir new file mode 100644 index 000000000000..09bc49c508a2 --- /dev/null +++ b/test/CodeGen/Hexagon/anti-dep-partial.mir @@ -0,0 +1,34 @@ +# RUN: llc -march=hexagon -post-RA-scheduler -run-pass post-RA-sched %s -o - | FileCheck %s + +--- | + declare void @check(i64, i32, i32, i64) + define void @foo() { + ret void + } +... + +--- +name: foo +tracksRegLiveness: true +body: | + bb.0: + successors: + liveins: %r0, %r1, %d1, %d2, %r16, %r17, %r19, %r22, %r23 + %r2 = A2_add %r23, killed %r17 + %r6 = M2_mpyi %r16, %r16 + %r22 = M2_accii %r22, killed %r2, 2 + %r7 = A2_tfrsi 12345678 + %r3 = A2_tfr killed %r16 + %d2 = A2_tfrp killed %d0 + %r2 = L2_loadri_io %r29, 28 + %r2 = M2_mpyi killed %r6, killed %r2 + %r23 = S2_asr_i_r %r22, 31 + S2_storeri_io killed %r29, 0, killed %r7 + ; The anti-dependency on r23 between the first A2_add and the + ; S2_asr_i_r was causing d11 to be renamed, while r22 remained + ; unchanged. Check that the renaming of d11 does not happen. + ; CHECK: d11 + %d0 = A2_tfrp killed %d11 + J2_call @check, implicit-def %d0, implicit-def %d1, implicit-def %d2, implicit %d0, implicit %d1, implicit %d2 +... + diff --git a/test/CodeGen/Hexagon/bit-gen-rseq.ll b/test/CodeGen/Hexagon/bit-gen-rseq.ll new file mode 100644 index 000000000000..08d4b7877159 --- /dev/null +++ b/test/CodeGen/Hexagon/bit-gen-rseq.ll @@ -0,0 +1,43 @@ +; RUN: llc -march=hexagon -disable-hsdr -hexagon-subreg-liveness < %s | FileCheck %s +; Check that we don't generate any bitwise operations. + +; CHECK-NOT: = or( +; CHECK-NOT: = and( + +target triple = "hexagon" + +define i32 @fred(i32* nocapture readonly %p, i32 %n) #0 { +entry: + %t.sroa.0.048 = load i32, i32* %p, align 4 + %cmp49 = icmp ugt i32 %n, 1 + br i1 %cmp49, label %for.body, label %for.end + +for.body: ; preds = %entry, %for.body + %t.sroa.0.052 = phi i32 [ %t.sroa.0.0, %for.body ], [ %t.sroa.0.048, %entry ] + %t.sroa.11.051 = phi i64 [ %t.sroa.11.0.extract.shift, %for.body ], [ 0, %entry ] + %i.050 = phi i32 [ %inc, %for.body ], [ 1, %entry ] + %t.sroa.0.0.insert.ext = zext i32 %t.sroa.0.052 to i64 + %t.sroa.0.0.insert.insert = or i64 %t.sroa.0.0.insert.ext, %t.sroa.11.051 + %0 = tail call i64 @llvm.hexagon.A2.addp(i64 %t.sroa.0.0.insert.insert, i64 %t.sroa.0.0.insert.insert) + %t.sroa.11.0.extract.shift = and i64 %0, -4294967296 + %arrayidx4 = getelementptr inbounds i32, i32* %p, i32 %i.050 + %inc = add nuw i32 %i.050, 1 + %t.sroa.0.0 = load i32, i32* %arrayidx4, align 4 + %exitcond = icmp eq i32 %inc, %n + br i1 %exitcond, label %for.end, label %for.body + +for.end: ; preds = %for.body, %entry + %t.sroa.0.0.lcssa = phi i32 [ %t.sroa.0.048, %entry ], [ %t.sroa.0.0, %for.body ] + %t.sroa.11.0.lcssa = phi i64 [ 0, %entry ], [ %t.sroa.11.0.extract.shift, %for.body ] + %t.sroa.0.0.insert.ext17 = zext i32 %t.sroa.0.0.lcssa to i64 + %t.sroa.0.0.insert.insert19 = or i64 %t.sroa.0.0.insert.ext17, %t.sroa.11.0.lcssa + %1 = tail call i64 @llvm.hexagon.A2.addp(i64 %t.sroa.0.0.insert.insert19, i64 %t.sroa.0.0.insert.insert19) + %t.sroa.11.0.extract.shift41 = lshr i64 %1, 32 + %t.sroa.11.0.extract.trunc42 = trunc i64 %t.sroa.11.0.extract.shift41 to i32 + ret i32 %t.sroa.11.0.extract.trunc42 +} + +declare i64 @llvm.hexagon.A2.addp(i64, i64) #1 + +attributes #0 = { norecurse nounwind readonly } +attributes #1 = { nounwind readnone } diff --git a/test/CodeGen/Hexagon/bit-loop-rc-mismatch.ll b/test/CodeGen/Hexagon/bit-loop-rc-mismatch.ll new file mode 100644 index 000000000000..db57998aeb66 --- /dev/null +++ b/test/CodeGen/Hexagon/bit-loop-rc-mismatch.ll @@ -0,0 +1,30 @@ +; RUN: llc -march=hexagon < %s +; REQUIRES: asserts + +target datalayout = "e-m:e-p:32:32:32-a:0-n16:32-i64:64:64-i32:32:32-i16:16:16-i1:8:8-f32:32:32-f64:64:64-v32:32:32-v64:64:64-v512:512:512-v1024:1024:1024-v2048:2048:2048" +target triple = "hexagon" + +define weak_odr hidden i32 @fred(i32* %this, i32* nocapture readonly dereferenceable(4) %__k) #0 align 2 { +entry: + %call = tail call i64 @danny(i32* %this, i32* nonnull dereferenceable(4) %__k) #2 + %__p.sroa.0.0.extract.trunc = trunc i64 %call to i32 + br i1 undef, label %for.end, label %for.body + +for.body: ; preds = %for.body, %entry + %__p.sroa.0.018 = phi i32 [ %call8, %for.body ], [ %__p.sroa.0.0.extract.trunc, %entry ] + %call8 = tail call i32 @sammy(i32* %this, i32 %__p.sroa.0.018) #2 + %0 = inttoptr i32 %call8 to i32* + %lnot.i = icmp eq i32* %0, undef + br i1 %lnot.i, label %for.end, label %for.body + +for.end: ; preds = %for.body, %entry + ret i32 0 +} + +declare hidden i64 @danny(i32*, i32* nocapture readonly dereferenceable(4)) #1 align 2 +declare hidden i32 @sammy(i32* nocapture, i32) #0 align 2 + +attributes #0 = { nounwind optsize "disable-tail-calls"="false" "less-precise-fpmad"="false" "no-frame-pointer-elim"="true" "no-frame-pointer-elim-non-leaf" "no-infs-fp-math"="false" "no-nans-fp-math"="false" "stack-protector-buffer-size"="8" "target-cpu"="hexagonv60" "target-features"="+hvx,-hvx-double" "unsafe-fp-math"="false" "use-soft-float"="false" } +attributes #1 = { nounwind optsize readonly "disable-tail-calls"="false" "less-precise-fpmad"="false" "no-frame-pointer-elim"="true" "no-frame-pointer-elim-non-leaf" "no-infs-fp-math"="false" "no-nans-fp-math"="false" "stack-protector-buffer-size"="8" "target-cpu"="hexagonv60" "target-features"="+hvx,-hvx-double" "unsafe-fp-math"="false" "use-soft-float"="false" } +attributes #2 = { optsize } + diff --git a/test/CodeGen/Hexagon/bit-rie.ll b/test/CodeGen/Hexagon/bit-rie.ll new file mode 100644 index 000000000000..6bd0558f580c --- /dev/null +++ b/test/CodeGen/Hexagon/bit-rie.ll @@ -0,0 +1,208 @@ +; RUN: llc -march=hexagon < %s | FileCheck %s +; CHECK-LABEL: LBB0{{.*}}if.end +; CHECK: r[[REG:[0-9]+]] = zxth +; CHECK: lsr(r[[REG]], + +target triple = "hexagon" + +@g0 = external constant [146 x i16], align 8 +@g1 = external constant [0 x i16], align 2 + +define void @fred(i32* nocapture readonly %p0, i16 signext %p1, i16* nocapture %p2, i16 signext %p3, i16 signext %p4, i16 signext %p5) #0 { +entry: + %conv = sext i16 %p1 to i32 + %0 = tail call i32 @llvm.hexagon.S2.asl.r.r.sat(i32 %conv, i32 1) + %1 = tail call i32 @llvm.hexagon.A2.sath(i32 %0) + %2 = tail call i32 @llvm.hexagon.A2.addh.l16.sat.ll(i32 3, i32 %1) + %conv3 = sext i16 %p4 to i32 + %cmp144 = icmp sgt i16 %p4, 0 + br i1 %cmp144, label %for.body, label %for.end + +for.body: ; preds = %entry, %for.body + %arrayidx.phi = phi i32* [ %arrayidx.inc, %for.body ], [ %p0, %entry ] + %i.0146.apmt = phi i32 [ %inc.apmt, %for.body ], [ 0, %entry ] + %L_temp1.0145 = phi i32 [ %5, %for.body ], [ 1, %entry ] + %3 = load i32, i32* %arrayidx.phi, align 4, !tbaa !1 + %4 = tail call i32 @llvm.hexagon.A2.abssat(i32 %3) + %5 = tail call i32 @llvm.hexagon.A2.max(i32 %L_temp1.0145, i32 %4) + %inc.apmt = add nuw nsw i32 %i.0146.apmt, 1 + %exitcond151 = icmp eq i32 %inc.apmt, %conv3 + %arrayidx.inc = getelementptr i32, i32* %arrayidx.phi, i32 1 + br i1 %exitcond151, label %for.end, label %for.body, !llvm.loop !5 + +for.end: ; preds = %for.body, %entry + %L_temp1.0.lcssa = phi i32 [ 1, %entry ], [ %5, %for.body ] + %6 = tail call i32 @llvm.hexagon.S2.clbnorm(i32 %L_temp1.0.lcssa) + %arrayidx6 = getelementptr inbounds [146 x i16], [146 x i16]* @g0, i32 0, i32 %conv3 + %7 = load i16, i16* %arrayidx6, align 2, !tbaa !7 + %conv7 = sext i16 %7 to i32 + %8 = tail call i32 @llvm.hexagon.A2.subh.l16.sat.ll(i32 %6, i32 %conv7) + br i1 %cmp144, label %for.body14.lr.ph, label %for.end29 + +for.body14.lr.ph: ; preds = %for.end + %sext132 = shl i32 %8, 16 + %conv17 = ashr exact i32 %sext132, 16 + br label %for.body14 + +for.body14: ; preds = %for.body14, %for.body14.lr.ph + %arrayidx16.phi = phi i32* [ %p0, %for.body14.lr.ph ], [ %arrayidx16.inc, %for.body14 ] + %i.1143.apmt = phi i32 [ 0, %for.body14.lr.ph ], [ %inc28.apmt, %for.body14 ] + %L_temp.0142 = phi i32 [ 0, %for.body14.lr.ph ], [ %12, %for.body14 ] + %9 = load i32, i32* %arrayidx16.phi, align 4, !tbaa !1 + %10 = tail call i32 @llvm.hexagon.S2.asl.r.r.sat(i32 %9, i32 %conv17) + %11 = tail call i32 @llvm.hexagon.A2.asrh(i32 %10) + %sext133 = shl i32 %11, 16 + %conv23 = ashr exact i32 %sext133, 16 + %12 = tail call i32 @llvm.hexagon.M2.mpy.acc.sat.ll.s0(i32 %L_temp.0142, i32 %conv23, i32 %conv23) + %inc28.apmt = add nuw nsw i32 %i.1143.apmt, 1 + %exitcond = icmp eq i32 %inc28.apmt, %conv3 + %arrayidx16.inc = getelementptr i32, i32* %arrayidx16.phi, i32 1 + br i1 %exitcond, label %for.end29, label %for.body14 + +for.end29: ; preds = %for.body14, %for.end + %L_temp.0.lcssa = phi i32 [ 0, %for.end ], [ %12, %for.body14 ] + %13 = tail call i32 @llvm.hexagon.A2.subh.l16.sat.ll(i32 %conv3, i32 1) + %cmp31 = icmp sgt i32 %13, 0 + br i1 %cmp31, label %if.then, label %if.end + +if.then: ; preds = %for.end29 + %arrayidx34 = getelementptr inbounds [0 x i16], [0 x i16]* @g1, i32 0, i32 %conv3 + %14 = load i16, i16* %arrayidx34, align 2, !tbaa !7 + %cmp.i = icmp eq i32 %L_temp.0.lcssa, -2147483648 + %cmp1.i = icmp eq i16 %14, -32768 + %or.cond.i = and i1 %cmp.i, %cmp1.i + br i1 %or.cond.i, label %if.end, label %if.else.i + +if.else.i: ; preds = %if.then + %conv3.i = sext i16 %14 to i32 + %15 = tail call i32 @llvm.hexagon.M2.hmmpyl.s1(i32 %L_temp.0.lcssa, i32 %conv3.i) #2 + %16 = tail call i64 @llvm.hexagon.M2.mpyd.ll.s1(i32 %conv3.i, i32 %L_temp.0.lcssa) #2 + %conv5.i = trunc i64 %16 to i32 + %phitmp = and i32 %conv5.i, 65535 + br label %if.end + +if.end: ; preds = %if.else.i, %if.then, %for.end29 + %L_temp.2 = phi i32 [ %L_temp.0.lcssa, %for.end29 ], [ %15, %if.else.i ], [ 2147483647, %if.then ] + %lsb.0 = phi i32 [ 0, %for.end29 ], [ %phitmp, %if.else.i ], [ 65535, %if.then ] + %sext = shl i32 %8, 16 + %conv35 = ashr exact i32 %sext, 16 + %17 = tail call i32 @llvm.hexagon.A2.subh.l16.sat.ll(i32 %conv35, i32 16) + %18 = tail call i32 @llvm.hexagon.S2.asl.r.r.sat(i32 %17, i32 1) + %19 = tail call i32 @llvm.hexagon.A2.sath(i32 %18) + %20 = tail call i32 @llvm.hexagon.S2.clbnorm(i32 %L_temp.2) + %sext123 = shl i32 %20, 16 + %conv38 = ashr exact i32 %sext123, 16 + %sext124 = shl i32 %19, 16 + %conv39 = ashr exact i32 %sext124, 16 + %21 = tail call i32 @llvm.hexagon.A2.addh.l16.sat.ll(i32 %conv38, i32 %conv39) + %22 = tail call i32 @llvm.hexagon.S2.asl.r.r.sat(i32 %L_temp.2, i32 %conv38) + %23 = tail call i32 @llvm.hexagon.A2.zxth(i32 %lsb.0) + %24 = tail call i32 @llvm.hexagon.A2.subh.l16.sat.ll(i32 16, i32 %conv38) + %25 = tail call i32 @llvm.hexagon.S2.lsr.r.r(i32 %23, i32 %24) + %sext125 = shl i32 %25, 16 + %conv45 = ashr exact i32 %sext125, 16 + %26 = tail call i32 @llvm.hexagon.A2.addsat(i32 %22, i32 %conv45) + %sext126 = shl i32 %2, 16 + %conv46 = ashr exact i32 %sext126, 16 + %sext127 = shl i32 %21, 16 + %conv47 = ashr exact i32 %sext127, 16 + %27 = tail call i32 @llvm.hexagon.A2.addh.l16.sat.ll(i32 %conv46, i32 %conv47) + %sext128 = shl i32 %27, 16 + %conv49 = ashr exact i32 %sext128, 16 + %cmp50 = icmp sgt i32 %sext128, 327679 + %tobool = icmp eq i16 %p5, 0 + %or.cond = or i1 %tobool, %cmp50 + br i1 %or.cond, label %if.else68, label %if.then53 + +if.then53: ; preds = %if.end + %28 = tail call i32 @llvm.hexagon.S2.asl.r.r.sat(i32 %conv49, i32 1) + %29 = tail call i32 @llvm.hexagon.A2.sath(i32 %28) + %30 = tail call i32 @llvm.hexagon.A2.subsat(i32 %26, i32 1276901417) + %cmp56 = icmp slt i32 %30, 0 + br i1 %cmp56, label %if.then58, label %if.else + +if.then58: ; preds = %if.then53 + %sext131 = shl i32 %29, 16 + %conv59 = ashr exact i32 %sext131, 16 + %31 = tail call i32 @llvm.hexagon.A2.addh.l16.sat.ll(i32 %conv59, i32 2) + br label %if.end80 + +if.else: ; preds = %if.then53 + %32 = tail call i32 @llvm.hexagon.A2.subsat(i32 %26, i32 1805811301) + %cmp61 = icmp slt i32 %32, 0 + br i1 %cmp61, label %if.then63, label %if.end80 + +if.then63: ; preds = %if.else + %sext130 = shl i32 %29, 16 + %conv64 = ashr exact i32 %sext130, 16 + %33 = tail call i32 @llvm.hexagon.A2.addh.l16.sat.ll(i32 %conv64, i32 1) + br label %if.end80 + +if.else68: ; preds = %if.end + %34 = tail call i32 @llvm.hexagon.A2.subsat(i32 %26, i32 1518500250) + %cmp69 = icmp slt i32 %34, 0 + br i1 %cmp69, label %if.then71, label %if.end74 + +if.then71: ; preds = %if.else68 + %35 = tail call i32 @llvm.hexagon.A2.addh.l16.sat.ll(i32 %conv49, i32 1) + br label %if.end74 + +if.end74: ; preds = %if.then71, %if.else68 + %m.0.in = phi i32 [ %35, %if.then71 ], [ %27, %if.else68 ] + br i1 %tobool, label %if.end80, label %if.then76 + +if.then76: ; preds = %if.end74 + %sext129 = shl i32 %m.0.in, 16 + %conv77 = ashr exact i32 %sext129, 16 + %36 = tail call i32 @llvm.hexagon.A2.addh.l16.sat.ll(i32 %conv77, i32 5) + br label %if.end80 + +if.end80: ; preds = %if.end74, %if.then76, %if.then58, %if.then63, %if.else + %m.1.in = phi i32 [ %31, %if.then58 ], [ %33, %if.then63 ], [ %29, %if.else ], [ %36, %if.then76 ], [ %m.0.in, %if.end74 ] + %m.1 = trunc i32 %m.1.in to i16 + %cmp.i135 = icmp slt i16 %m.1, 0 + %var_out.0.i136 = select i1 %cmp.i135, i16 0, i16 %m.1 + %conv81 = sext i16 %p3 to i32 + %37 = tail call i32 @llvm.hexagon.A2.subh.l16.sat.ll(i32 %conv81, i32 1) + %conv82 = trunc i32 %37 to i16 + %cmp.i134 = icmp sgt i16 %var_out.0.i136, %conv82 + %var_out.0.i = select i1 %cmp.i134, i16 %conv82, i16 %var_out.0.i136 + store i16 %var_out.0.i, i16* %p2, align 2, !tbaa !7 + ret void +} + +declare i32 @llvm.hexagon.A2.abssat(i32) #2 +declare i32 @llvm.hexagon.A2.addh.l16.sat.ll(i32, i32) #2 +declare i32 @llvm.hexagon.A2.addsat(i32, i32) #2 +declare i32 @llvm.hexagon.A2.asrh(i32) #2 +declare i32 @llvm.hexagon.A2.max(i32, i32) #2 +declare i32 @llvm.hexagon.A2.sath(i32) #2 +declare i32 @llvm.hexagon.A2.subh.l16.sat.ll(i32, i32) #2 +declare i32 @llvm.hexagon.A2.subsat(i32, i32) #2 +declare i32 @llvm.hexagon.A2.zxth(i32) #2 +declare i32 @llvm.hexagon.M2.hmmpyl.s1(i32, i32) #2 +declare i32 @llvm.hexagon.M2.mpy.acc.sat.ll.s0(i32, i32, i32) #2 +declare i32 @llvm.hexagon.S2.asl.r.r.sat(i32, i32) #2 +declare i32 @llvm.hexagon.S2.asr.r.r.sat(i32, i32) #2 +declare i32 @llvm.hexagon.S2.clbnorm(i32) #2 +declare i32 @llvm.hexagon.S2.lsr.r.r(i32, i32) #2 +declare i64 @llvm.hexagon.M2.mpyd.ll.s1(i32, i32) #2 +declare void @llvm.lifetime.end(i64, i8* nocapture) #1 +declare void @llvm.lifetime.start(i64, i8* nocapture) #1 + +attributes #0 = { norecurse nounwind "target-cpu"="hexagonv60" "target-features"="+hvx,-hvx-double" "unsafe-fp-math"="false" "use-soft-float"="false" } +attributes #1 = { argmemonly nounwind } +attributes #2 = { nounwind readnone } + + +!1 = !{!2, !2, i64 0} +!2 = !{!"int", !3, i64 0} +!3 = !{!"omnipotent char", !4, i64 0} +!4 = !{!"Simple C/C++ TBAA"} +!5 = distinct !{!5, !6} +!6 = !{!"llvm.loop.threadify", i32 81508608} +!7 = !{!8, !8, i64 0} +!8 = !{!"short", !3, i64 0} +!9 = distinct !{!9, !10} +!10 = !{!"llvm.loop.threadify", i32 1441813} +!11 = distinct !{!11, !10} diff --git a/test/CodeGen/Hexagon/bit-skip-byval.ll b/test/CodeGen/Hexagon/bit-skip-byval.ll new file mode 100644 index 000000000000..d6c1aad94007 --- /dev/null +++ b/test/CodeGen/Hexagon/bit-skip-byval.ll @@ -0,0 +1,11 @@ +; RUN: llc -march=hexagon < %s | FileCheck %s +; +; Either and or zxtb. +; CHECK: r0 = and(r1, #255) + +%struct.t0 = type { i32 } + +define i32 @foo(%struct.t0* byval align 8 %s, i8 zeroext %t, i8 %u) #0 { + %a = zext i8 %u to i32 + ret i32 %a +} diff --git a/test/CodeGen/Hexagon/bit-validate-reg.ll b/test/CodeGen/Hexagon/bit-validate-reg.ll new file mode 100644 index 000000000000..16d4a5e4484d --- /dev/null +++ b/test/CodeGen/Hexagon/bit-validate-reg.ll @@ -0,0 +1,21 @@ +; RUN: llc -march=hexagon < %s | FileCheck %s + +; Make sure we don't generate zxtb to transfer a predicate register into +; a general purpose register. + +; CHECK: r0 = p0 +; CHECK-NOT: zxtb(p + +target triple = "hexagon" + +; Function Attrs: nounwind +define i32 @fred() local_unnamed_addr #0 { +entry: + %0 = tail call i32 @llvm.hexagon.C4.and.and(i32 undef, i32 undef, i32 undef) + ret i32 %0 +} + +declare i32 @llvm.hexagon.C4.and.and(i32, i32, i32) #1 + +attributes #0 = { nounwind "target-cpu"="hexagonv5" } +attributes #1 = { nounwind readnone } diff --git a/test/CodeGen/Hexagon/bit-visit-flowq.ll b/test/CodeGen/Hexagon/bit-visit-flowq.ll new file mode 100644 index 000000000000..b44847dee68e --- /dev/null +++ b/test/CodeGen/Hexagon/bit-visit-flowq.ll @@ -0,0 +1,47 @@ +; RUN: llc -march=hexagon < %s | FileCheck %s +; REQUIRES: asserts + +; Check that we don't crash. +; CHECK: call bar + +target triple = "hexagon" + +@debug = external hidden unnamed_addr global i1, align 4 + +; Function Attrs: nounwind +define void @foo() local_unnamed_addr #0 { +entry: + br label %if.end5 + +if.end5: ; preds = %entry + br i1 undef, label %if.then12, label %if.end13 + +if.then12: ; preds = %if.end5 + unreachable + +if.end13: ; preds = %if.end5 + br label %for.cond + +for.cond: ; preds = %if.end13 + %or.cond288 = or i1 undef, undef + br i1 undef, label %if.then44, label %if.end51 + +if.then44: ; preds = %for.cond + tail call void @bar() #0 + br label %if.end51 + +if.end51: ; preds = %if.then44, %for.cond + %.b433 = load i1, i1* @debug, align 4 + %or.cond290 = and i1 %or.cond288, %.b433 + br i1 %or.cond290, label %if.then55, label %if.end63 + +if.then55: ; preds = %if.end51 + unreachable + +if.end63: ; preds = %if.end51 + unreachable +} + +declare void @bar() local_unnamed_addr #0 + +attributes #0 = { nounwind } diff --git a/test/CodeGen/Hexagon/block-addr.ll b/test/CodeGen/Hexagon/block-addr.ll index 420af2fee1c9..c0db2cef545e 100644 --- a/test/CodeGen/Hexagon/block-addr.ll +++ b/test/CodeGen/Hexagon/block-addr.ll @@ -1,9 +1,8 @@ ; RUN: llc -march=hexagon < %s | FileCheck %s -; Allow combine(..##JTI..): -; CHECK: r{{[0-9]+}}{{.*}} = {{.*}}#.LJTI -; CHECK: r{{[0-9]+}} = memw(r{{[0-9]+}}{{ *}}+{{ *}}r{{[0-9]+<<#[0-9]+}}) -; CHECK: jumpr:nt r{{[0-9]+}} +; CHECK: .LJTI +; CHECK-DAG: r[[REG:[0-9]+]] = memw(r{{[0-9]+}}{{ *}}+{{ *}}r{{[0-9]+<<#[0-9]+}}) +; CHECK-DAG: jumpr:nt r[[REG]] define void @main() #0 { entry: diff --git a/test/CodeGen/Hexagon/branchfolder-keep-impdef.ll b/test/CodeGen/Hexagon/branchfolder-keep-impdef.ll new file mode 100644 index 000000000000..a56680bd4399 --- /dev/null +++ b/test/CodeGen/Hexagon/branchfolder-keep-impdef.ll @@ -0,0 +1,29 @@ +; RUN: llc -march=hexagon -verify-machineinstrs < %s | FileCheck %s +; +; Check that the testcase compiles successfully. Expect that if-conversion +; took place. +; CHECK-LABEL: fred: +; CHECK: if (!p0) r1 = memw(r0 + #0) + +target triple = "hexagon" + +define void @fred(i32 %p0) local_unnamed_addr align 2 { +b0: + br i1 undef, label %b1, label %b2 + +b1: ; preds = %b0 + %t0 = load i8*, i8** undef, align 4 + br label %b2 + +b2: ; preds = %b1, %b0 + %t1 = phi i8* [ %t0, %b1 ], [ undef, %b0 ] + %t2 = getelementptr inbounds i8, i8* %t1, i32 %p0 + tail call void @llvm.memmove.p0i8.p0i8.i32(i8* undef, i8* %t2, i32 undef, i32 1, i1 false) #1 + unreachable +} + +declare void @llvm.memmove.p0i8.p0i8.i32(i8* nocapture, i8* nocapture readonly, i32, i32, i1) #0 + +attributes #0 = { argmemonly nounwind } +attributes #1 = { nounwind } + diff --git a/test/CodeGen/Hexagon/build-vector-shuffle.ll b/test/CodeGen/Hexagon/build-vector-shuffle.ll new file mode 100644 index 000000000000..1d06953ddf32 --- /dev/null +++ b/test/CodeGen/Hexagon/build-vector-shuffle.ll @@ -0,0 +1,21 @@ +; RUN: llc -march=hexagon < %s | FileCheck %s +; Check that we don't crash. +; CHECK: vshuff + +target triple = "hexagon" + +define void @hex_interleaved.s0.__outermost() local_unnamed_addr #0 { +entry: + %0 = icmp eq i32 undef, 0 + %sel2 = select i1 %0, <32 x i16> undef, <32 x i16> zeroinitializer + %1 = bitcast <32 x i16> %sel2 to <16 x i32> + %2 = tail call <16 x i32> @llvm.hexagon.V6.vshuffh(<16 x i32> %1) + store <16 x i32> %2, <16 x i32>* undef, align 2 + unreachable +} + +; Function Attrs: nounwind readnone +declare <16 x i32> @llvm.hexagon.V6.vshuffh(<16 x i32>) #1 + +attributes #0 = { nounwind "target-cpu"="hexagonv60" "target-features"="+hvx" } +attributes #1 = { nounwind readnone } diff --git a/test/CodeGen/Hexagon/combine.ll b/test/CodeGen/Hexagon/combine.ll index 8f5cec88d692..04a080fdf425 100644 --- a/test/CodeGen/Hexagon/combine.ll +++ b/test/CodeGen/Hexagon/combine.ll @@ -1,4 +1,4 @@ -; RUN: llc -march=hexagon -mcpu=hexagonv5 -disable-hsdr < %s | FileCheck %s +; RUN: llc -march=hexagon -mcpu=hexagonv5 -disable-hsdr -hexagon-bit=0 < %s | FileCheck %s ; CHECK: combine(r{{[0-9]+}}, r{{[0-9]+}}) @j = external global i32 diff --git a/test/CodeGen/Hexagon/const-pool-tf.ll b/test/CodeGen/Hexagon/const-pool-tf.ll new file mode 100644 index 000000000000..9a4569b1e4de --- /dev/null +++ b/test/CodeGen/Hexagon/const-pool-tf.ll @@ -0,0 +1,40 @@ +; RUN: llc -march=hexagon -mcpu=hexagonv60 -relocation-model pic < %s | FileCheck %s + +; CHECK: @PCREL + +target datalayout = "e-m:e-p:32:32:32-a:0-n16:32-i64:64:64-i32:32:32-i16:16:16-i1:8:8-f32:32:32-f64:64:64-v32:32:32-v64:64:64-v512:512:512-v1024:1024:1024-v2048:2048:2048" +target triple = "hexagon-unknown--elf" + +; Function Attrs: nounwind +define void @hex_h.s0.__outermost(i32 %h.stride.114) #0 { +entry: + br i1 undef, label %"for h.s0.y.preheader", label %call_destructor.exit, !prof !1 + +call_destructor.exit: ; preds = %entry + ret void + +"for h.s0.y.preheader": ; preds = %entry + %tmp22.us = mul i32 undef, %h.stride.114 + br label %"for h.s0.x.x.us" + +"for h.s0.x.x.us": ; preds = %"for h.s0.x.x.us", %"for h.s0.y.preheader" + %h.s0.x.x.us = phi i32 [ %5, %"for h.s0.x.x.us" ], [ 0, %"for h.s0.y.preheader" ] + %0 = shl nsw i32 %h.s0.x.x.us, 5 + %1 = add i32 %0, %tmp22.us + %2 = add nsw i32 %1, 16 + %3 = getelementptr inbounds i32, i32* null, i32 %2 + %4 = bitcast i32* %3 to <16 x i32>* + store <16 x i32> zeroinitializer, <16 x i32>* %4, align 4, !tbaa !2 + %5 = add nuw nsw i32 %h.s0.x.x.us, 1 + br label %"for h.s0.x.x.us" +} + +attributes #0 = { nounwind } + +!llvm.ident = !{!0, !0, !0, !0, !0, !0, !0, !0, !0, !0, !0, !0, !0, !0, !0, !0} + +!0 = !{!"Clang $LLVM_VERSION_MAJOR.$LLVM_VERSION_MINOR (based on LLVM 3.9.0)"} +!1 = !{!"branch_weights", i32 1073741824, i32 0} +!2 = !{!3, !3, i64 0} +!3 = !{!"h", !4} +!4 = !{!"Halide buffer"} diff --git a/test/CodeGen/Hexagon/constp-clb.ll b/test/CodeGen/Hexagon/constp-clb.ll new file mode 100644 index 000000000000..1a872b11aad3 --- /dev/null +++ b/test/CodeGen/Hexagon/constp-clb.ll @@ -0,0 +1,23 @@ +; RUN: llc -mcpu=hexagonv5 < %s +; REQUIRES: asserts + +target datalayout = "e-m:e-p:32:32-i1:32-i64:64-a:0-v32:32-n16:32" +target triple = "hexagon-unknown--elf" + +; Function Attrs: nounwind readnone +define i64 @foo() #0 { +entry: + %0 = tail call i32 @llvm.hexagon.S2.clbp(i64 291) + %1 = tail call i64 @llvm.hexagon.A4.combineir(i32 0, i32 %0) + ret i64 %1 +} + +; Function Attrs: nounwind readnone +declare i32 @llvm.hexagon.S2.clbp(i64) #1 + +; Function Attrs: nounwind readnone +declare i64 @llvm.hexagon.A4.combineir(i32, i32) #1 + +attributes #0 = { nounwind readnone "less-precise-fpmad"="false" "no-frame-pointer-elim"="true" "no-frame-pointer-elim-non-leaf" "no-infs-fp-math"="false" "no-nans-fp-math"="false" "stack-protector-buffer-size"="8" "unsafe-fp-math"="false" "use-soft-float"="false" } +attributes #1 = { nounwind readnone } + diff --git a/test/CodeGen/Hexagon/constp-combine-neg.ll b/test/CodeGen/Hexagon/constp-combine-neg.ll new file mode 100644 index 000000000000..18f0e81076af --- /dev/null +++ b/test/CodeGen/Hexagon/constp-combine-neg.ll @@ -0,0 +1,27 @@ +; RUN: llc -O2 -march=hexagon < %s | FileCheck %s --check-prefix=CHECK-TEST1 +; RUN: llc -O2 -march=hexagon < %s | FileCheck %s --check-prefix=CHECK-TEST2 +; RUN: llc -O2 -march=hexagon < %s | FileCheck %s --check-prefix=CHECK-TEST3 +define i32 @main() #0 { +entry: + %l = alloca [7 x i32], align 8 + %p_arrayidx45 = bitcast [7 x i32]* %l to i32* + %vector_ptr = bitcast [7 x i32]* %l to <2 x i32>* + store <2 x i32> , <2 x i32>* %vector_ptr, align 8 + %p_arrayidx.1 = getelementptr [7 x i32], [7 x i32]* %l, i32 0, i32 2 + %vector_ptr.1 = bitcast i32* %p_arrayidx.1 to <2 x i32>* + store <2 x i32> , <2 x i32>* %vector_ptr.1, align 8 + %p_arrayidx.2 = getelementptr [7 x i32], [7 x i32]* %l, i32 0, i32 4 + %vector_ptr.2 = bitcast i32* %p_arrayidx.2 to <2 x i32>* + store <2 x i32> , <2 x i32>* %vector_ptr.2, align 8 + ret i32 0 +} + +; The instructions seem to be in a different order in the .s file than +; the corresponding values in the .ll file, so just run the test three +; times and each time test for a different instruction. +; CHECK-TEST1: combine(#-2, #3) +; CHECK-TEST2: combine(#6, #-4) +; CHECK-TEST3: combine(#-10, #-8) + +attributes #0 = { "less-precise-fpmad"="false" "no-frame-pointer-elim"="false" "no-frame-pointer-elim-non-leaf"="true" "no-infs-fp-math"="false" "no-nans-fp-math"="false" "unsafe-fp-math"="false" "use-soft-float"="false" } + diff --git a/test/CodeGen/Hexagon/constp-ctb.ll b/test/CodeGen/Hexagon/constp-ctb.ll new file mode 100644 index 000000000000..76a9820583ea --- /dev/null +++ b/test/CodeGen/Hexagon/constp-ctb.ll @@ -0,0 +1,26 @@ +; RUN: llc < %s +; REQUIRES: asserts + +target datalayout = "e-m:e-p:32:32-i1:32-i64:64-a:0-v32:32-n16:32" +target triple = "hexagon-unknown--elf" + +; Function Attrs: nounwind readnone +define i64 @foo() #0 { +entry: + %0 = tail call i32 @llvm.hexagon.S2.ct0p(i64 18) + %1 = tail call i32 @llvm.hexagon.S2.ct1p(i64 27) + %2 = tail call i64 @llvm.hexagon.A2.combinew(i32 %0, i32 %1) + ret i64 %2 +} + +; Function Attrs: nounwind readnone +declare i32 @llvm.hexagon.S2.ct0p(i64) #0 + +; Function Attrs: nounwind readnone +declare i32 @llvm.hexagon.S2.ct1p(i64) #0 + +; Function Attrs: nounwind readnone +declare i64 @llvm.hexagon.A2.combinew(i32, i32) #0 + +attributes #0 = { nounwind readnone } + diff --git a/test/CodeGen/Hexagon/constp-extract.ll b/test/CodeGen/Hexagon/constp-extract.ll new file mode 100644 index 000000000000..00f176317c4c --- /dev/null +++ b/test/CodeGen/Hexagon/constp-extract.ll @@ -0,0 +1,31 @@ +; Expect the constant propagation to evaluate signed and unsigned bit extract. +; RUN: llc -march=hexagon -O2 < %s | FileCheck %s + +target triple = "hexagon" + +@x = common global i32 0, align 4 +@y = common global i32 0, align 4 + +define void @foo() #0 { +entry: + ; extractu(0x000ABCD0, 16, 4) + ; should evaluate to 0xABCD (dec 43981) + %0 = call i32 @llvm.hexagon.S2.extractu(i32 703696, i32 16, i32 4) +; CHECK: 43981 +; CHECK-NOT: extractu + store i32 %0, i32* @x, align 4 + ; extract(0x000ABCD0, 16, 4) + ; should evaluate to 0xFFFFABCD (dec 4294945741 or -21555) + %1 = call i32 @llvm.hexagon.S4.extract(i32 703696, i32 16, i32 4) +; CHECK: -21555 +; CHECK-NOT: extract + store i32 %1, i32* @y, align 4 + ret void +} + +declare i32 @llvm.hexagon.S2.extractu(i32, i32, i32) #1 + +declare i32 @llvm.hexagon.S4.extract(i32, i32, i32) #1 + +attributes #0 = { nounwind "less-precise-fpmad"="false" "no-frame-pointer-elim"="true" "no-frame-pointer-elim-non-leaf"="true" "no-infs-fp-math"="false" "no-nans-fp-math"="false" "unsafe-fp-math"="false" "use-soft-float"="false" } +attributes #1 = { nounwind readnone } diff --git a/test/CodeGen/Hexagon/constp-physreg.ll b/test/CodeGen/Hexagon/constp-physreg.ll new file mode 100644 index 000000000000..0473b96f6de4 --- /dev/null +++ b/test/CodeGen/Hexagon/constp-physreg.ll @@ -0,0 +1,21 @@ +; RUN: llc -O2 -march hexagon < %s +target datalayout = "e-p:32:32:32-i64:64:64-i32:32:32-i16:16:16-i1:32:32-f64:64:64-f32:32:32-v64:64:64-v32:32:32-a0:0-n16:32" +target triple = "hexagon" + +define signext i16 @foo(i16 signext %var1, i16 signext %var2) #0 { +entry: + %0 = or i16 %var2, %var1 + %1 = icmp slt i16 %0, 0 + %cmp8 = icmp sgt i16 %var1, %var2 + %or.cond19 = or i1 %1, %cmp8 + br i1 %or.cond19, label %return, label %if.end + +if.end: ; preds = %entry + br label %return + +return: ; preds = %if.end, %if.end15, %entry + %retval.0.reg2mem.0 = phi i16 [ 0, %entry ], [ 32767, %if.end ] + ret i16 %retval.0.reg2mem.0 +} + +attributes #0 = { nounwind readnone "less-precise-fpmad"="false" "no-frame-pointer-elim"="false" "no-frame-pointer-elim-non-leaf"="true" "no-infs-fp-math"="false" "no-nans-fp-math"="false" "unsafe-fp-math"="false" "use-soft-float"="false" } diff --git a/test/CodeGen/Hexagon/constp-rewrite-branches.ll b/test/CodeGen/Hexagon/constp-rewrite-branches.ll new file mode 100644 index 000000000000..dfac49cb553e --- /dev/null +++ b/test/CodeGen/Hexagon/constp-rewrite-branches.ll @@ -0,0 +1,17 @@ +; RUN: llc -O2 -march hexagon < %s | FileCheck %s + +define i32 @foo(i32 %x) { + %p = icmp eq i32 %x, 0 + br i1 %p, label %zero, label %nonzero +nonzero: + %v1 = add i32 %x, 1 + %c = icmp eq i32 %x, %v1 +; This branch will be rewritten by HCP. A bug would cause both branches to +; go away, leaving no path to "ret -1". + br i1 %c, label %zero, label %other +zero: + ret i32 0 +other: +; CHECK: -1 + ret i32 -1 +} diff --git a/test/CodeGen/Hexagon/constp-rseq.ll b/test/CodeGen/Hexagon/constp-rseq.ll new file mode 100644 index 000000000000..c89407e4b8e4 --- /dev/null +++ b/test/CodeGen/Hexagon/constp-rseq.ll @@ -0,0 +1,19 @@ +; RUN: llc -march=hexagon < %s | FileCheck %s +; CHECK: cmp +; Make sure that the result is not a compile-time constant. + +define i64 @foo(i32 %x) { +entry: + %c = icmp slt i32 %x, 17 + br i1 %c, label %b1, label %b2 +b1: + br label %b2 +b2: + %p = phi i32 [ 1, %entry ], [ 0, %b1 ] + %q = sub i32 %x, %x + %y = zext i32 %q to i64 + %u = shl i64 %y, 32 + %v = zext i32 %p to i64 + %w = or i64 %u, %v + ret i64 %w +} diff --git a/test/CodeGen/Hexagon/constp-vsplat.ll b/test/CodeGen/Hexagon/constp-vsplat.ll new file mode 100644 index 000000000000..6063383a5f75 --- /dev/null +++ b/test/CodeGen/Hexagon/constp-vsplat.ll @@ -0,0 +1,18 @@ +; RUN: llc < %s +; REQUIRES: asserts +target datalayout = "e-m:e-p:32:32-i1:32-i64:64-a:0-v32:32-n16:32" +target triple = "hexagon" + +; Function Attrs: nounwind readnone +define i64 @foo() #0 { +entry: + %0 = tail call i32 @llvm.hexagon.S2.vsplatrb(i32 255) + %conv = zext i32 %0 to i64 + %shl = shl nuw i64 %conv, 32 + %or = or i64 %shl, %conv + ret i64 %or +} + +declare i32 @llvm.hexagon.S2.vsplatrb(i32) #0 + +attributes #0 = { nounwind readnone } diff --git a/test/CodeGen/Hexagon/copy-to-combine-dbg.ll b/test/CodeGen/Hexagon/copy-to-combine-dbg.ll new file mode 100644 index 000000000000..9ffc7b509251 --- /dev/null +++ b/test/CodeGen/Hexagon/copy-to-combine-dbg.ll @@ -0,0 +1,57 @@ +; RUN: llc -march=hexagon < %s | FileCheck %s +; Check for some sane output (original problem was a crash). +; CHECK: DEBUG_VALUE: fred:Count <- 0 + +target triple = "hexagon" + +define i32 @fred(i32 %p) local_unnamed_addr #0 !dbg !6 { +entry: + br label %cond.end + +cond.end: ; preds = %entry + br i1 undef, label %cond.false.i, label %for.body.lr.ph.i + +for.body.lr.ph.i: ; preds = %cond.end + tail call void @llvm.dbg.value(metadata i32 0, i64 0, metadata !10, metadata !12) #0, !dbg !13 + br label %for.body.i + +cond.false.i: ; preds = %cond.end + unreachable + +for.body.i: ; preds = %for.inc.i, %for.body.lr.ph.i + %inc.sink37.i = phi i32 [ 0, %for.body.lr.ph.i ], [ %inc.i, %for.inc.i ] + %call.i = tail call i8* undef(i32 12, i8* undef) #0 + br label %for.inc.i + +for.inc.i: ; preds = %for.body.i + %inc.i = add nuw i32 %inc.sink37.i, 1 + %cmp1.i = icmp ult i32 %inc.i, %p + br i1 %cmp1.i, label %for.body.i, label %PQ_AllocMem.exit.loopexit + +PQ_AllocMem.exit.loopexit: ; preds = %for.inc.i + unreachable +} + +declare void @llvm.dbg.value(metadata, i64, metadata, metadata) #1 + +attributes #0 = { nounwind } +attributes #1 = { nounwind readnone } + +!llvm.dbg.cu = !{!0} +!llvm.module.flags = !{!3, !4} +!llvm.ident = !{!5} + +!0 = distinct !DICompileUnit(language: DW_LANG_C99, file: !1, producer: "clang version 4.0.0 (http://llvm.org/git/clang.git 37afcb099ac2b001f4c826da7ca1d077b67a508c) (http://llvm.org/git/llvm.git 5887f1c75b3ba216850c834b186efdd3e54b7d4f)", isOptimized: true, runtimeVersion: 0, emissionKind: FullDebug, enums: !2, retainedTypes: !2) +!1 = !DIFile(filename: "file.c", directory: "/") +!2 = !{} +!3 = !{i32 2, !"Dwarf Version", i32 4} +!4 = !{i32 2, !"Debug Info Version", i32 3} +!5 = !{!"clang version 4.0.0 (http://llvm.org/git/clang.git 37afcb099ac2b001f4c826da7ca1d077b67a508c) (http://llvm.org/git/llvm.git 5887f1c75b3ba216850c834b186efdd3e54b7d4f)"} +!6 = distinct !DISubprogram(name: "fred", scope: !1, file: !1, line: 116, type: !7, isLocal: false, isDefinition: true, scopeLine: 121, flags: DIFlagPrototyped, isOptimized: true, unit: !0, variables: !9) +!7 = !DISubroutineType(types: !2) +!8 = !DIBasicType(name: "int", size: 32, align: 32, encoding: DW_ATE_signed) +!9 = !{!10} +!10 = !DILocalVariable(name: "Count", scope: !6, file: !1, line: 1, type: !8) +!11 = distinct !DILocation(line: 1, column: 1, scope: !6) +!12 = !DIExpression() +!13 = !DILocation(line: 1, column: 1, scope: !6, inlinedAt: !11) diff --git a/test/CodeGen/Hexagon/dead-store-stack.ll b/test/CodeGen/Hexagon/dead-store-stack.ll new file mode 100644 index 000000000000..93d324baad9e --- /dev/null +++ b/test/CodeGen/Hexagon/dead-store-stack.ll @@ -0,0 +1,131 @@ +; RUN: llc -O2 -march=hexagon < %s | FileCheck %s +; CHECK: ParseFunc: +; CHECK: r[[ARG0:[0-9]+]] = memuh(r[[ARG1:[0-9]+]] + #[[OFFSET:[0-9]+]]) +; CHECK: memw(r[[ARG1]]+#[[OFFSET]]) = r[[ARG0]] + +@.str.3 = external unnamed_addr constant [8 x i8], align 1 +; Function Attrs: nounwind +define void @ParseFunc() local_unnamed_addr #0 { +entry: + %dataVar = alloca i32, align 4 + %0 = load i32, i32* %dataVar, align 4 + %and = and i32 %0, 65535 + store i32 %and, i32* %dataVar, align 4 + %.pr = load i32, i32* %dataVar, align 4 + switch i32 %.pr, label %sw.epilog [ + i32 4, label %sw.bb + i32 5, label %sw.bb + i32 1, label %sw.bb39 + i32 2, label %sw.bb40 + i32 3, label %sw.bb41 + i32 6, label %sw.bb42 + i32 7, label %sw.bb43 + i32 13, label %sw.bb44 + i32 0, label %sw.bb44 + i32 14, label %sw.bb45 + i32 15, label %sw.bb46 + ] + +sw.bb: + %cmp1.i = icmp eq i32 %.pr, 4 + br label %land.rhs.i + +land.rhs.i: + br label %ParseFuncNext.exit.i + +ParseFuncNext.exit.i: + br i1 %cmp1.i, label %if.then.i, label %if.else10.i + +if.then.i: + call void (i8*, i32, i8*, ...) @snprintf(i8* undef, i32 undef, i8* getelementptr inbounds ([8 x i8], [8 x i8]* @.str.3, i32 0, i32 0), i32 undef) #2 + br label %if.end27.i + +if.else10.i: + unreachable + +if.end27.i: + br label %land.rhs.i + +sw.bb39: + unreachable + +sw.bb40: + unreachable + +sw.bb41: + unreachable + +sw.bb42: + %1 = load i32, i32* undef, align 4 + %shr.i = lshr i32 %1, 16 + br label %while.cond.i.i + +while.cond.i.i: + %2 = load i8, i8* undef, align 1 + switch i8 %2, label %if.then4.i [ + i8 48, label %land.end.i.i + i8 120, label %land.end.i.i + i8 37, label %do.body.i.i + ] + +land.end.i.i: + unreachable + +do.body.i.i: + switch i8 undef, label %if.then4.i [ + i8 117, label %if.end40.i.i + i8 120, label %if.end40.i.i + i8 88, label %if.end40.i.i + i8 100, label %if.end40.i.i + i8 105, label %if.end40.i.i + ] + +if.end40.i.i: + %trunc.i = trunc i32 %shr.i to i16 + br label %land.rhs.i126 + +if.then4.i: + unreachable + +land.rhs.i126: + switch i16 %trunc.i, label %sw.epilog.i [ + i16 1, label %sw.bb.i + i16 2, label %sw.bb12.i + i16 4, label %sw.bb16.i + ] + +sw.bb.i: + unreachable + +sw.bb12.i: + unreachable + +sw.bb16.i: + unreachable + +sw.epilog.i: + call void (i8*, i32, i8*, ...) @snprintf(i8* undef, i32 undef, i8* nonnull undef, i32 undef) #2 + br label %land.rhs.i126 + +sw.bb43: + unreachable + +sw.bb44: + unreachable + +sw.bb45: + unreachable + +sw.bb46: + unreachable + +sw.epilog: + ret void +} + +; Function Attrs: nounwind +declare void @snprintf(i8* nocapture, i32, i8* nocapture readonly, ...) local_unnamed_addr #1 + +attributes #0 = { nounwind "correctly-rounded-divide-sqrt-fp-math"="false" "disable-tail-calls"="false" "less-precise-fpmad"="false" "no-frame-pointer-elim"="true" "no-frame-pointer-elim-non-leaf" "no-infs-fp-math"="false" "no-jump-tables"="false" "no-nans-fp-math"="false" "no-signed-zeros-fp-math"="false" "stack-protector-buffer-size"="8" "target-features"="+hvx" "unsafe-fp-math"="false" "use-soft-float"="false" } +attributes #1 = { nounwind "correctly-rounded-divide-sqrt-fp-math"="false" "disable-tail-calls"="false" "less-precise-fpmad"="false" "no-frame-pointer-elim"="true" "no-frame-pointer-elim-non-leaf" "no-infs-fp-math"="false" "no-nans-fp-math"="false" "no-signed-zeros-fp-math"="false" "stack-protector-buffer-size"="8" "target-features"="+hvx" "unsafe-fp-math"="false" "use-soft-float"="false" } +attributes #2 = { nounwind } diff --git a/test/CodeGen/Hexagon/early-if-vecpi.ll b/test/CodeGen/Hexagon/early-if-vecpi.ll new file mode 100644 index 000000000000..6f3ec2d5a51d --- /dev/null +++ b/test/CodeGen/Hexagon/early-if-vecpi.ll @@ -0,0 +1,69 @@ +; RUN: llc -march=hexagon < %s | FileCheck %s + +target triple = "hexagon-unknown--elf" + +; Check that we can predicate base+offset vector stores. +; CHECK-LABEL: sammy +; CHECK: if{{.*}}vmem(r{{[0-9]+}}+#0) = +define void @sammy(<16 x i32>* nocapture %p, <16 x i32>* nocapture readonly %q, i32 %n) #0 { +entry: + %0 = load <16 x i32>, <16 x i32>* %q, align 64 + %sub = add nsw i32 %n, -1 + br label %for.body + +for.body: ; preds = %if.end, %entry + %p.addr.011 = phi <16 x i32>* [ %p, %entry ], [ %incdec.ptr, %if.end ] + %i.010 = phi i32 [ 0, %entry ], [ %add, %if.end ] + %mul = mul nsw i32 %i.010, %sub + %add = add nuw nsw i32 %i.010, 1 + %mul1 = mul nsw i32 %add, %n + %cmp2 = icmp slt i32 %mul, %mul1 + br i1 %cmp2, label %if.then, label %if.end + +if.then: ; preds = %for.body + store <16 x i32> %0, <16 x i32>* %p.addr.011, align 64 + br label %if.end + +if.end: ; preds = %if.then, %for.body + %incdec.ptr = getelementptr inbounds <16 x i32>, <16 x i32>* %p.addr.011, i32 1 + %exitcond = icmp eq i32 %add, 100 + br i1 %exitcond, label %for.end, label %for.body + +for.end: ; preds = %if.end + ret void +} + +; Check that we can predicate post-increment vector stores. +; CHECK-LABEL: danny +; CHECK: if{{.*}}vmem(r{{[0-9]+}}++#1) = +define void @danny(<16 x i32>* nocapture %p, <16 x i32>* nocapture readonly %q, i32 %n) #0 { +entry: + %0 = load <16 x i32>, <16 x i32>* %q, align 64 + %sub = add nsw i32 %n, -1 + br label %for.body + +for.body: ; preds = %if.end, %entry + %p.addr.012 = phi <16 x i32>* [ %p, %entry ], [ %incdec.ptr3, %if.end ] + %i.011 = phi i32 [ 0, %entry ], [ %add, %if.end ] + %mul = mul nsw i32 %i.011, %sub + %add = add nuw nsw i32 %i.011, 1 + %mul1 = mul nsw i32 %add, %n + %cmp2 = icmp slt i32 %mul, %mul1 + br i1 %cmp2, label %if.then, label %if.end + +if.then: ; preds = %for.body + %incdec.ptr = getelementptr inbounds <16 x i32>, <16 x i32>* %p.addr.012, i32 1 + store <16 x i32> %0, <16 x i32>* %p.addr.012, align 64 + br label %if.end + +if.end: ; preds = %if.then, %for.body + %p.addr.1 = phi <16 x i32>* [ %incdec.ptr, %if.then ], [ %p.addr.012, %for.body ] + %incdec.ptr3 = getelementptr inbounds <16 x i32>, <16 x i32>* %p.addr.1, i32 1 + %exitcond = icmp eq i32 %add, 100 + br i1 %exitcond, label %for.end, label %for.body + +for.end: ; preds = %if.end + ret void +} + +attributes #0 = { norecurse nounwind "target-cpu"="hexagonv60" "target-features"="+hvx,-hvx-double" } diff --git a/test/CodeGen/Hexagon/expand-condsets-def-undef.mir b/test/CodeGen/Hexagon/expand-condsets-def-undef.mir new file mode 100644 index 000000000000..44da969bf29b --- /dev/null +++ b/test/CodeGen/Hexagon/expand-condsets-def-undef.mir @@ -0,0 +1,41 @@ +# RUN: llc -march=hexagon -run-pass expand-condsets -o - %s -verify-machineinstrs | FileCheck %s + +# CHECK-LABEL: name: fred + +# Make sure that is accounted for when validating moves +# during predication. In the code below, %2.isub_hi is invalidated +# by the C2_mux instruction, and so predicating the A2_addi as an argument +# to the C2_muxir should not happen. + +--- | + define void @fred() { ret void } + +... +--- + +name: fred +tracksRegLiveness: true +registers: + - { id: 0, class: predregs } + - { id: 1, class: intregs } + - { id: 2, class: doubleregs } + - { id: 3, class: intregs } +liveins: + - { reg: '%p0', virtual-reg: '%0' } + - { reg: '%r0', virtual-reg: '%1' } + - { reg: '%d0', virtual-reg: '%2' } + +body: | + bb.0: + liveins: %r0, %d0, %p0 + %0 = COPY %p0 + %1 = COPY %r0 + %2 = COPY %d0 + ; Check that this instruction is unchanged (remains unpredicated) + ; CHECK: %3 = A2_addi %2.isub_hi, 1 + %3 = A2_addi %2.isub_hi, 1 + undef %2.isub_lo = C2_mux %0, %2.isub_lo, %1 + %2.isub_hi = C2_muxir %0, %3, 0 + +... + diff --git a/test/CodeGen/Hexagon/expand-condsets-extend.ll b/test/CodeGen/Hexagon/expand-condsets-extend.ll new file mode 100644 index 000000000000..e716925ed8ef --- /dev/null +++ b/test/CodeGen/Hexagon/expand-condsets-extend.ll @@ -0,0 +1,112 @@ +; RUN: llc -march=hexagon < %s | FileCheck %s +; REQUIRES: asserts + +; Check for a reasonable output. This testcase used to crash. +; CHECK: .size fred + +target triple = "hexagon" + +define void @fred() local_unnamed_addr #0 { +entry: + %0 = load i64, i64* undef, align 8 + %shr.i465 = lshr i64 %0, 48 + %trunc = trunc i64 %shr.i465 to i15 + switch i15 %trunc, label %if.end26 [ + i15 -1, label %if.then14 + i15 0, label %if.then21 + ] + +if.then14: ; preds = %entry + unreachable + +if.then21: ; preds = %entry + unreachable + +if.end26: ; preds = %entry + br label %if.end36 + +if.end36: ; preds = %if.end26 + %or.i335 = or i64 undef, undef + %shl2.i322 = or i64 undef, -9223372036854775808 + br i1 undef, label %if.then44, label %lor.rhs.i + +lor.rhs.i: ; preds = %if.end36 + br label %le128.exit + +le128.exit: ; preds = %lor.rhs.i + br i1 undef, label %if.then44, label %while.cond.preheader + +if.then44: ; preds = %le128.exit, %if.end36 + %conv42544 = phi i64 [ 0, %le128.exit ], [ 1, %if.end36 ] + br label %while.cond.preheader + +while.cond.preheader: ; preds = %if.then44, %le128.exit + %aSig0.3.ph = phi i64 [ undef, %if.then44 ], [ %or.i335, %le128.exit ] + %q.0.ph = phi i64 [ %conv42544, %if.then44 ], [ 0, %le128.exit ] + br i1 undef, label %while.body.lr.ph, label %while.end + +while.body.lr.ph: ; preds = %while.cond.preheader + %shr.i263 = lshr i64 %shl2.i322, 32 + br label %while.body + +while.body: ; preds = %exit312, %while.body.lr.ph + %aSig0.3554 = phi i64 [ %aSig0.3.ph, %while.body.lr.ph ], [ %sub3.i205, %exit312 ] + br label %while.body.i297 + +while.body.i297: ; preds = %while.body.i297, %while.body + %z.045.i287 = phi i64 [ %sub.i290, %while.body.i297 ], [ undef, %while.body ] + %sub.i290 = add i64 %z.045.i287, -4294967296 + %cmp3.i296 = icmp slt i64 undef, 0 + br i1 %cmp3.i296, label %while.body.i297, label %while.end.i305.loopexit + +while.end.i305.loopexit: ; preds = %while.body.i297 + %or14.i309 = or i64 0, %sub.i290 + br label %exit312 + +exit312: ; preds = %while.end.i305.loopexit + %cmp50 = icmp ugt i64 %or14.i309, 4 + %cond = select i1 %cmp50, i64 undef, i64 0 + %shr3.i.i221 = lshr i64 %cond, 32 + %mul15.i11.i243 = mul nuw i64 %shr3.i.i221, %shr.i263 + %add20.i18.i250 = add i64 0, %mul15.i11.i243 + %add26.i23.i255 = add i64 %add20.i18.i250, 0 + %add3.i.i261 = add i64 %add26.i23.i255, 0 + %shl4.i215 = shl i64 %add3.i.i261, 61 + %or10.i = or i64 %shl4.i215, 0 + %shl2.i207 = shl i64 %aSig0.3554, 61 + %or.i209 = or i64 %shl2.i207, 0 + %sub1.i202 = add i64 0, %or.i209 + %sub3.i205 = sub i64 %sub1.i202, %or10.i + %cmp47 = icmp sgt i32 undef, 61 + br i1 %cmp47, label %while.body, label %while.end.loopexit + +while.end.loopexit: ; preds = %exit312 + br label %while.end + +while.end: ; preds = %while.end.loopexit, %while.cond.preheader + %aSig0.3.lcssa = phi i64 [ %aSig0.3.ph, %while.cond.preheader ], [ %sub3.i205, %while.end.loopexit ] + %q.0.lcssa = phi i64 [ %q.0.ph, %while.cond.preheader ], [ %cond, %while.end.loopexit ] + br i1 undef, label %if.then56, label %if.else71 + +if.then56: ; preds = %while.end + unreachable + +if.else71: ; preds = %while.end + %shr8.i155 = lshr i64 %aSig0.3.lcssa, 12 + br label %do.body + +do.body: ; preds = %do.body, %if.else71 + %aSig0.5 = phi i64 [ %sub3.i151, %do.body ], [ %shr8.i155, %if.else71 ] + %q.1 = phi i64 [ %inc, %do.body ], [ %q.0.lcssa, %if.else71 ] + %inc = add i64 %q.1, 1 + %sub1.i148 = sub i64 %aSig0.5, 0 + %sub3.i151 = add i64 %sub1.i148, 0 + %cmp73 = icmp sgt i64 %sub3.i151, -1 + br i1 %cmp73, label %do.body, label %do.end + +do.end: ; preds = %do.body + %and = and i64 %inc, 1 + unreachable +} + +attributes #0 = { nounwind } diff --git a/test/CodeGen/Hexagon/expand-condsets-impuse.mir b/test/CodeGen/Hexagon/expand-condsets-impuse.mir new file mode 100644 index 000000000000..08b6798aa2fb --- /dev/null +++ b/test/CodeGen/Hexagon/expand-condsets-impuse.mir @@ -0,0 +1,78 @@ +# RUN: llc -march=hexagon -run-pass expand-condsets -o - %s -verify-machineinstrs | FileCheck %s + +# CHECK-LABEL: name: fred + +--- | + define void @fred() { ret void } + +... +--- + +name: fred +tracksRegLiveness: true +registers: + - { id: 0, class: intregs } + - { id: 1, class: intregs } + - { id: 2, class: intregs } + - { id: 3, class: intregs } + - { id: 4, class: predregs } + - { id: 5, class: intregs } + - { id: 6, class: intregs } + - { id: 7, class: intregs } + - { id: 8, class: predregs } + - { id: 9, class: intregs } + - { id: 10, class: intregs } + - { id: 11, class: intregs } + - { id: 12, class: predregs } + - { id: 13, class: intregs } + - { id: 14, class: intregs } + - { id: 99, class: intregs } +liveins: + - { reg: '%r0', virtual-reg: '%99' } + +body: | + bb.0: + liveins: %r0 + successors: %bb.298, %bb.301 + %99 = COPY %r0 + J2_jumpr %99, implicit-def %pc + + bb.298: + liveins: %r0 + successors: %bb.299, %bb.301, %bb.309 + %0 = A2_tfrsi 123 + %1 = A2_tfrsi -1 + %3 = L2_loadri_io %99, 8 + %4 = C2_cmpeqi %3, 33 + %5 = A2_tfrsi -2 + %6 = C2_mux %4, %5, %1 + J2_jumpr %6, implicit-def %pc + + bb.299: + successors: %bb.300, %bb.309 + %7 = L2_loadrb_io %99, 12 + %8 = C2_cmpeqi %7, 9 + %9 = A2_tfrsi -999 + ; CHECK: %10 = C2_cmoveit killed %8, -999, implicit %10 + %10 = C2_mux %8, %9, %1 + J2_jumpr %10, implicit-def %pc + + bb.300: + successors: %bb.309 + S2_storeri_io %99, 0, %0 + J2_jump %bb.309, implicit-def %pc + + bb.301: + successors: %bb.299, %bb.309 + %0 = A2_tfrsi 124 + %1 = A2_tfrsi -4 + %11 = L2_loadri_io %99, 8 + %12 = C2_cmpeqi %11, 33 + %13 = A2_tfrsi -2 + %14 = C2_mux %12, %13, %1 + J2_jumpr %14, implicit-def %pc + + bb.309: + +... + diff --git a/test/CodeGen/Hexagon/expand-condsets-rm-reg.mir b/test/CodeGen/Hexagon/expand-condsets-rm-reg.mir new file mode 100644 index 000000000000..983035e228cc --- /dev/null +++ b/test/CodeGen/Hexagon/expand-condsets-rm-reg.mir @@ -0,0 +1,49 @@ +# RUN: llc -march=hexagon -run-pass expand-condsets -o - 2>&1 %s -verify-machineinstrs -debug-only=expand-condsets | FileCheck %s +# REQUIRES: asserts + +# Check that coalesced registers are removed from live intervals. +# +# Check that vreg3 is coalesced into vreg4, and that after coalescing +# it is no longer in live intervals. + +# CHECK-LABEL: After expand-condsets +# CHECK: INTERVALS +# CHECK-NOT: vreg3 +# CHECK: MACHINEINSTRS + + +--- | + define void @fred() { ret void } + +... +--- + +name: fred +tracksRegLiveness: true +registers: + - { id: 0, class: intregs } + - { id: 1, class: intregs } + - { id: 2, class: predregs } + - { id: 3, class: intregs } + - { id: 4, class: intregs } +liveins: + - { reg: '%r0', virtual-reg: '%0' } + - { reg: '%r1', virtual-reg: '%1' } + - { reg: '%p0', virtual-reg: '%2' } + +body: | + bb.0: + liveins: %r0, %r1, %p0 + %0 = COPY %r0 + %0 = COPY %r0 ; Force isSSA = false. + %1 = COPY %r1 + %2 = COPY %p0 + ; Check that %3 was coalesced into %4. + ; CHECK: %4 = A2_abs %1 + ; CHECK: %4 = A2_tfrt killed %2, killed %0, implicit %4 + %3 = A2_abs %1 + %4 = C2_mux %2, %0, %3 + %r0 = COPY %4 + J2_jumpr %r31, implicit %r0, implicit-def %pc +... + diff --git a/test/CodeGen/Hexagon/expand-condsets-same-inputs.mir b/test/CodeGen/Hexagon/expand-condsets-same-inputs.mir new file mode 100644 index 000000000000..83938d1b774a --- /dev/null +++ b/test/CodeGen/Hexagon/expand-condsets-same-inputs.mir @@ -0,0 +1,32 @@ +# RUN: llc -march=hexagon -run-pass expand-condsets -expand-condsets-coa-limit=0 -o - %s -verify-machineinstrs | FileCheck %s + +# CHECK-LABEL: name: fred + +--- | + define void @fred() { ret void } + +... +--- + +name: fred +tracksRegLiveness: true +registers: + - { id: 0, class: predregs } + - { id: 1, class: intregs } + - { id: 2, class: intregs } + - { id: 3, class: intregs } + +body: | + bb.0: + liveins: %r0, %r1, %r2, %p0 + %0 = COPY %p0 + %0 = COPY %p0 ; Cheat: convince MIR parser that this is not SSA. + %1 = COPY %r1 + ; Make sure we do not expand/predicate a mux with identical inputs. + ; CHECK-NOT: A2_paddit + %2 = A2_addi %1, 1 + %3 = C2_mux %0, killed %2, %2 + %r0 = COPY %3 + +... + diff --git a/test/CodeGen/Hexagon/expand-condsets-undef2.ll b/test/CodeGen/Hexagon/expand-condsets-undef2.ll new file mode 100644 index 000000000000..d62d50d83613 --- /dev/null +++ b/test/CodeGen/Hexagon/expand-condsets-undef2.ll @@ -0,0 +1,47 @@ +; RUN: llc -march=hexagon < %s +; REQUIRES: asserts + +; Test that the HexagonExpandCondsets pass does not assert due to +; attempting to shrink a live interval incorrectly. + + +define void @test() #0 { +entry: + br i1 undef, label %cleanup, label %if.end + +if.end: + %0 = load i32, i32* undef, align 4 + %sext = shl i32 %0, 16 + %conv19 = ashr exact i32 %sext, 16 + br i1 undef, label %cleanup, label %for.body.lr.ph + +for.body.lr.ph: + br label %for.body + +for.body: + %bestScoreL16Q4.0278 = phi i16 [ 32767, %for.body.lr.ph ], [ %.sink, %early_termination ] + br i1 false, label %for.body44.lr.ph, label %for.cond90.preheader + +for.body44.lr.ph: + %conv77 = sext i16 %bestScoreL16Q4.0278 to i32 + unreachable + +for.cond90.preheader: + br i1 undef, label %early_termination, label %for.body97 + +for.body97: + br i1 undef, label %for.body97, label %early_termination + +early_termination: + %.sink = select i1 undef, i16 undef, i16 %bestScoreL16Q4.0278 + %cmp27 = icmp slt i32 undef, %conv19 + br i1 %cmp27, label %for.body, label %for.end124 + +for.end124: + unreachable + +cleanup: + ret void +} + +attributes #0 = { nounwind "target-cpu"="hexagonv60" } diff --git a/test/CodeGen/Hexagon/expand-vstorerw-undef.ll b/test/CodeGen/Hexagon/expand-vstorerw-undef.ll new file mode 100644 index 000000000000..8524bf33de18 --- /dev/null +++ b/test/CodeGen/Hexagon/expand-vstorerw-undef.ll @@ -0,0 +1,95 @@ +; RUN: llc -march=hexagon < %s | FileCheck %s + +; After register allocation it is possible to have a spill of a register +; that is only partially defined. That in itself it fine, but creates a +; problem for double vector registers. Stores of such registers are pseudo +; instructions that are expanded into pairs of individual vector stores, +; and in case of a partially defined source, one of the stores may use +; an entirely undefined register. +; +; This testcase used to crash. Make sure we can handle it, and that we +; do generate a store for the defined part of W0: + +; CHECK-LABEL: fred: +; CHECK: v[[REG:[0-9]+]] = vsplat +; CHECK: vmem(r29+#6) = v[[REG]] + + +target triple = "hexagon" + +declare void @danny() local_unnamed_addr #0 +declare void @sammy() local_unnamed_addr #0 +declare <32 x i32> @llvm.hexagon.V6.lo.128B(<64 x i32>) #1 +declare <32 x i32> @llvm.hexagon.V6.lvsplatw.128B(i32) #1 +declare <64 x i32> @llvm.hexagon.V6.vcombine.128B(<32 x i32>, <32 x i32>) #1 +declare <32 x i32> @llvm.hexagon.V6.vshuffeb.128B(<32 x i32>, <32 x i32>) #1 +declare <32 x i32> @llvm.hexagon.V6.vlsrh.128B(<32 x i32>, i32) #1 +declare <64 x i32> @llvm.hexagon.V6.vaddh.dv.128B(<64 x i32>, <64 x i32>) #1 + +define hidden void @fred() #2 { +b0: + %v1 = load i32, i32* null, align 4 + %v2 = icmp ult i64 0, 2147483648 + br i1 %v2, label %b3, label %b5 + +b3: ; preds = %b0 + %v4 = icmp sgt i32 0, -1 + br i1 %v4, label %b6, label %b5 + +b5: ; preds = %b3, %b0 + ret void + +b6: ; preds = %b3 + tail call void @danny() + br label %b7 + +b7: ; preds = %b21, %b6 + %v8 = icmp sgt i32 %v1, 0 + %v9 = select i1 %v8, i32 %v1, i32 0 + %v10 = select i1 false, i32 0, i32 %v9 + %v11 = icmp slt i32 %v10, 0 + %v12 = select i1 %v11, i32 %v10, i32 0 + %v13 = icmp slt i32 0, %v12 + br i1 %v13, label %b14, label %b18 + +b14: ; preds = %b16, %b7 + br i1 false, label %b15, label %b16 + +b15: ; preds = %b14 + br label %b16 + +b16: ; preds = %b15, %b14 + %v17 = icmp eq i32 0, %v12 + br i1 %v17, label %b18, label %b14 + +b18: ; preds = %b16, %b7 + tail call void @danny() + %v19 = tail call <32 x i32> @llvm.hexagon.V6.lvsplatw.128B(i32 524296) #0 + %v20 = tail call <64 x i32> @llvm.hexagon.V6.vcombine.128B(<32 x i32> %v19, <32 x i32> %v19) + br label %b22 + +b21: ; preds = %b22 + tail call void @sammy() #3 + br label %b7 + +b22: ; preds = %b22, %b18 + %v23 = tail call <64 x i32> @llvm.hexagon.V6.vaddh.dv.128B(<64 x i32> zeroinitializer, <64 x i32> %v20) #0 + %v24 = tail call <32 x i32> @llvm.hexagon.V6.lo.128B(<64 x i32> %v23) + %v25 = tail call <32 x i32> @llvm.hexagon.V6.vlsrh.128B(<32 x i32> %v24, i32 4) #0 + %v26 = tail call <64 x i32> @llvm.hexagon.V6.vcombine.128B(<32 x i32> zeroinitializer, <32 x i32> %v25) + %v27 = tail call <32 x i32> @llvm.hexagon.V6.vshuffeb.128B(<32 x i32> zeroinitializer, <32 x i32> zeroinitializer) #0 + %v28 = tail call <32 x i32> @llvm.hexagon.V6.lo.128B(<64 x i32> %v26) #0 + %v29 = tail call <32 x i32> @llvm.hexagon.V6.vshuffeb.128B(<32 x i32> zeroinitializer, <32 x i32> %v28) #0 + store <32 x i32> %v27, <32 x i32>* null, align 128 + %v30 = add nsw i32 0, 128 + %v31 = getelementptr inbounds i8, i8* null, i32 %v30 + %v32 = bitcast i8* %v31 to <32 x i32>* + store <32 x i32> %v29, <32 x i32>* %v32, align 128 + %v33 = icmp eq i32 0, 0 + br i1 %v33, label %b21, label %b22 +} + +attributes #0 = { nounwind } +attributes #1 = { nounwind readnone } +attributes #2 = { nounwind "reciprocal-estimates"="none" "target-cpu"="hexagonv60" "target-features"="+hvx-double" } +attributes #3 = { nobuiltin nounwind } diff --git a/test/CodeGen/Hexagon/fixed-spill-mutable.ll b/test/CodeGen/Hexagon/fixed-spill-mutable.ll new file mode 100644 index 000000000000..03aa72bc5a8c --- /dev/null +++ b/test/CodeGen/Hexagon/fixed-spill-mutable.ll @@ -0,0 +1,69 @@ +; RUN: llc -march=hexagon < %s | FileCheck %s + +; The early return is predicated, and the save-restore code is mixed together: +; { +; p0 = cmp.eq(r0, #0) +; if (p0.new) r17:16 = memd(r29 + #0) +; memd(r29+#0) = r17:16 +; } +; { +; if (p0) dealloc_return +; } +; The problem is that the load will execute before the store, clobbering the +; pair r17:16. +; +; Check that the store and the load are not in the same packet. +; CHECK: memd{{.*}} = r17:16 +; CHECK: } +; CHECK: r17:16 = memd +; CHECK-LABEL: LBB0_1: + +target triple = "hexagon" + +%struct.0 = type { i8*, %struct.1*, %struct.2*, %struct.0*, %struct.0* } +%struct.1 = type { [60 x i8], i32, %struct.1* } +%struct.2 = type { i8, i8, i8, i8, %union.anon } +%union.anon = type { %struct.3* } +%struct.3 = type { %struct.3*, %struct.2* } + +@var = external hidden unnamed_addr global %struct.0*, align 4 + +declare void @bar(i8*, i32) local_unnamed_addr #0 + +define void @foo() local_unnamed_addr #1 { +entry: + %.pr = load %struct.0*, %struct.0** @var, align 4, !tbaa !1 + %cmp2 = icmp eq %struct.0* %.pr, null + br i1 %cmp2, label %while.end, label %while.body.preheader + +while.body.preheader: ; preds = %entry + br label %while.body + +while.body: ; preds = %while.body.preheader, %while.body + %0 = phi %struct.0* [ %4, %while.body ], [ %.pr, %while.body.preheader ] + %right = getelementptr inbounds %struct.0, %struct.0* %0, i32 0, i32 4 + %1 = bitcast %struct.0** %right to i32* + %2 = load i32, i32* %1, align 4, !tbaa !5 + %3 = bitcast %struct.0* %0 to i8* + tail call void @bar(i8* %3, i32 20) #1 + store i32 %2, i32* bitcast (%struct.0** @var to i32*), align 4, !tbaa !1 + %4 = inttoptr i32 %2 to %struct.0* + %cmp = icmp eq i32 %2, 0 + br i1 %cmp, label %while.end.loopexit, label %while.body + +while.end.loopexit: ; preds = %while.body + br label %while.end + +while.end: ; preds = %while.end.loopexit, %entry + ret void +} + +attributes #0 = { optsize } +attributes #1 = { nounwind optsize } + +!1 = !{!2, !2, i64 0} +!2 = !{!"any pointer", !3, i64 0} +!3 = !{!"omnipotent char", !4, i64 0} +!4 = !{!"Simple C/C++ TBAA"} +!5 = !{!6, !2, i64 16} +!6 = !{!"0", !2, i64 0, !2, i64 4, !2, i64 8, !2, i64 12, !2, i64 16} diff --git a/test/CodeGen/Hexagon/float-amode.ll b/test/CodeGen/Hexagon/float-amode.ll new file mode 100644 index 000000000000..9804f48349f8 --- /dev/null +++ b/test/CodeGen/Hexagon/float-amode.ll @@ -0,0 +1,89 @@ +; RUN: llc -march=hexagon -fp-contract=fast -disable-hexagon-peephole -disable-hexagon-amodeopt < %s | FileCheck %s + +; The test checks for various addressing modes for floating point loads/stores. + +%struct.matrix_paramsGlob = type { [50 x i8], i16, [50 x float] } +%struct.matrix_params = type { [50 x i8], i16, float** } +%struct.matrix_params2 = type { i16, [50 x [50 x float]] } + +@globB = common global %struct.matrix_paramsGlob zeroinitializer, align 4 +@globA = common global %struct.matrix_paramsGlob zeroinitializer, align 4 +@b = common global float 0.000000e+00, align 4 +@a = common global float 0.000000e+00, align 4 + +; CHECK-LABEL: test1 +; CHECK: [[REG11:(r[0-9]+)]]{{ *}}={{ *}}memw(r{{[0-9]+}} + r{{[0-9]+}}<<#2) +; CHECK: [[REG12:(r[0-9]+)]] += sfmpy({{.*}}[[REG11]] +; CHECK: memw(r{{[0-9]+}} + r{{[0-9]+}}<<#2) = [[REG12]].new + +; Function Attrs: norecurse nounwind +define void @test1(%struct.matrix_params* nocapture readonly %params, i32 %col1) { +entry: + %matrixA = getelementptr inbounds %struct.matrix_params, %struct.matrix_params* %params, i32 0, i32 2 + %0 = load float**, float*** %matrixA, align 4 + %arrayidx = getelementptr inbounds float*, float** %0, i32 2 + %1 = load float*, float** %arrayidx, align 4 + %arrayidx1 = getelementptr inbounds float, float* %1, i32 %col1 + %2 = load float, float* %arrayidx1, align 4 + %mul = fmul float %2, 2.000000e+01 + %add = fadd float %mul, 1.000000e+01 + %arrayidx3 = getelementptr inbounds float*, float** %0, i32 5 + %3 = load float*, float** %arrayidx3, align 4 + %arrayidx4 = getelementptr inbounds float, float* %3, i32 %col1 + store float %add, float* %arrayidx4, align 4 + ret void +} + +; CHECK-LABEL: test2 +; CHECK: [[REG21:(r[0-9]+)]]{{ *}}={{ *}}memw(##globB+92) +; CHECK: [[REG22:(r[0-9]+)]] = sfadd({{.*}}[[REG21]] +; CHECK: memw(##globA+84) = [[REG22]] + +; Function Attrs: norecurse nounwind +define void @test2(%struct.matrix_params* nocapture readonly %params, i32 %col1) { +entry: + %matrixA = getelementptr inbounds %struct.matrix_params, %struct.matrix_params* %params, i32 0, i32 2 + %0 = load float**, float*** %matrixA, align 4 + %1 = load float*, float** %0, align 4 + %arrayidx1 = getelementptr inbounds float, float* %1, i32 %col1 + %2 = load float, float* %arrayidx1, align 4 + %3 = load float, float* getelementptr inbounds (%struct.matrix_paramsGlob, %struct.matrix_paramsGlob* @globB, i32 0, i32 2, i32 10), align 4 + %add = fadd float %2, %3 + store float %add, float* getelementptr inbounds (%struct.matrix_paramsGlob, %struct.matrix_paramsGlob* @globA, i32 0, i32 2, i32 8), align 4 + ret void +} + +; CHECK-LABEL: test3 +; CHECK: [[REG31:(r[0-9]+)]]{{ *}}={{ *}}memw(#b) +; CHECK: [[REG32:(r[0-9]+)]] = sfadd({{.*}}[[REG31]] +; CHECK: memw(#a) = [[REG32]] + +; Function Attrs: norecurse nounwind +define void @test3(%struct.matrix_params* nocapture readonly %params, i32 %col1) { +entry: + %matrixA = getelementptr inbounds %struct.matrix_params, %struct.matrix_params* %params, i32 0, i32 2 + %0 = load float**, float*** %matrixA, align 4 + %1 = load float*, float** %0, align 4 + %arrayidx1 = getelementptr inbounds float, float* %1, i32 %col1 + %2 = load float, float* %arrayidx1, align 4 + %3 = load float, float* @b, align 4 + %add = fadd float %2, %3 + store float %add, float* @a, align 4 + ret void +} + +; CHECK-LABEL: test4 +; CHECK: [[REG41:(r[0-9]+)]]{{ *}}={{ *}}memw(r0<<#2 + ##globB+52) +; CHECK: [[REG42:(r[0-9]+)]] = sfadd({{.*}}[[REG41]] +; CHECK: memw(r0<<#2 + ##globA+60) = [[REG42]] +; Function Attrs: noinline norecurse nounwind +define void @test4(i32 %col1) { +entry: + %arrayidx = getelementptr inbounds %struct.matrix_paramsGlob, %struct.matrix_paramsGlob* @globB, i32 0, i32 2, i32 %col1 + %0 = load float, float* %arrayidx, align 4 + %add = fadd float %0, 0.000000e+00 + %add1 = add nsw i32 %col1, 2 + %arrayidx2 = getelementptr inbounds %struct.matrix_paramsGlob, %struct.matrix_paramsGlob* @globA, i32 0, i32 2, i32 %add1 + store float %add, float* %arrayidx2, align 4 + ret void +} diff --git a/test/CodeGen/Hexagon/fminmax.ll b/test/CodeGen/Hexagon/fminmax.ll new file mode 100644 index 000000000000..7c1a9fb42f23 --- /dev/null +++ b/test/CodeGen/Hexagon/fminmax.ll @@ -0,0 +1,27 @@ +; RUN: llc -march=hexagon < %s | FileCheck %s + +target datalayout = "e-m:e-p:32:32:32-a:0-n16:32-i64:64:64-i32:32:32-i16:16:16-i1:8:8-f32:32:32-f64:64:64-v32:32:32-v64:64:64-v512:512:512-v1024:1024:1024-v2048:2048:2048" +target triple = "hexagon" + +; CHECK-LABEL: minimum +; CHECK: sfmin +define float @minimum(float %x, float %y) #0 { +entry: + %call = tail call float @fminf(float %x, float %y) #1 + ret float %call +} + +; CHECK-LABEL: maximum +; CHECK: sfmax +define float @maximum(float %x, float %y) #0 { +entry: + %call = tail call float @fmaxf(float %x, float %y) #1 + ret float %call +} + +declare float @fminf(float, float) #0 +declare float @fmaxf(float, float) #0 + +attributes #0 = { nounwind readnone "disable-tail-calls"="false" "less-precise-fpmad"="false" "no-frame-pointer-elim"="true" "no-frame-pointer-elim-non-leaf" "no-infs-fp-math"="false" "no-nans-fp-math"="false" "stack-protector-buffer-size"="8" "target-cpu"="hexagonv60" "target-features"="+hvx,-hvx-double" "unsafe-fp-math"="false" "use-soft-float"="false" } +attributes #1 = { nounwind readnone } + diff --git a/test/CodeGen/Hexagon/frame-offset-overflow.ll b/test/CodeGen/Hexagon/frame-offset-overflow.ll new file mode 100644 index 000000000000..43d5fd5ad0f0 --- /dev/null +++ b/test/CodeGen/Hexagon/frame-offset-overflow.ll @@ -0,0 +1,163 @@ +; RUN: llc -march=hexagon < %s | FileCheck %s + +; In reality, check that the compilation succeeded and that some code was +; generated. +; CHECK: vadd + +target triple = "hexagon" + +define void @fred(i16* noalias nocapture readonly %p0, i32 %p1, i32 %p2, i16* noalias nocapture %p3, i32 %p4) local_unnamed_addr #1 { +entry: + %mul = mul i32 %p4, %p1 + %add.ptr = getelementptr inbounds i16, i16* %p0, i32 %mul + %add = add nsw i32 %p4, 1 + %rem = srem i32 %add, 5 + %mul1 = mul i32 %rem, %p1 + %add.ptr2 = getelementptr inbounds i16, i16* %p0, i32 %mul1 + %add.ptr6 = getelementptr inbounds i16, i16* %p0, i32 0 + %add7 = add nsw i32 %p4, 3 + %rem8 = srem i32 %add7, 5 + %mul9 = mul i32 %rem8, %p1 + %add.ptr10 = getelementptr inbounds i16, i16* %p0, i32 %mul9 + %add.ptr14 = getelementptr inbounds i16, i16* %p0, i32 0 + %incdec.ptr18 = getelementptr inbounds i16, i16* %add.ptr14, i32 32 + %0 = bitcast i16* %incdec.ptr18 to <16 x i32>* + %incdec.ptr17 = getelementptr inbounds i16, i16* %add.ptr10, i32 32 + %1 = bitcast i16* %incdec.ptr17 to <16 x i32>* + %incdec.ptr16 = getelementptr inbounds i16, i16* %add.ptr6, i32 32 + %2 = bitcast i16* %incdec.ptr16 to <16 x i32>* + %incdec.ptr15 = getelementptr inbounds i16, i16* %add.ptr2, i32 32 + %3 = bitcast i16* %incdec.ptr15 to <16 x i32>* + %incdec.ptr = getelementptr inbounds i16, i16* %add.ptr, i32 32 + %4 = bitcast i16* %incdec.ptr to <16 x i32>* + %5 = bitcast i16* %p3 to <16 x i32>* + br i1 undef, label %for.end.loopexit.unr-lcssa, label %for.body + +for.body: ; preds = %for.body, %entry + %optr.0102 = phi <16 x i32>* [ %incdec.ptr24.3, %for.body ], [ %5, %entry ] + %iptr4.0101 = phi <16 x i32>* [ %incdec.ptr23.3, %for.body ], [ %0, %entry ] + %iptr3.0100 = phi <16 x i32>* [ %incdec.ptr22.3, %for.body ], [ %1, %entry ] + %iptr2.099 = phi <16 x i32>* [ undef, %for.body ], [ %2, %entry ] + %iptr1.098 = phi <16 x i32>* [ %incdec.ptr20.3, %for.body ], [ %3, %entry ] + %iptr0.097 = phi <16 x i32>* [ %incdec.ptr19.3, %for.body ], [ %4, %entry ] + %dVsumv1.096 = phi <32 x i32> [ %66, %for.body ], [ undef, %entry ] + %niter = phi i32 [ %niter.nsub.3, %for.body ], [ undef, %entry ] + %6 = load <16 x i32>, <16 x i32>* %iptr0.097, align 64, !tbaa !1 + %7 = load <16 x i32>, <16 x i32>* %iptr1.098, align 64, !tbaa !1 + %8 = load <16 x i32>, <16 x i32>* %iptr2.099, align 64, !tbaa !1 + %9 = load <16 x i32>, <16 x i32>* %iptr3.0100, align 64, !tbaa !1 + %10 = load <16 x i32>, <16 x i32>* %iptr4.0101, align 64, !tbaa !1 + %11 = tail call <32 x i32> @llvm.hexagon.V6.vaddhw(<16 x i32> %6, <16 x i32> %10) + %12 = tail call <32 x i32> @llvm.hexagon.V6.vmpyhsat.acc(<32 x i32> %11, <16 x i32> %8, i32 393222) + %13 = tail call <32 x i32> @llvm.hexagon.V6.vcombine(<16 x i32> %9, <16 x i32> %7) + %14 = tail call <32 x i32> @llvm.hexagon.V6.vmpahb.acc(<32 x i32> %12, <32 x i32> %13, i32 67372036) + %15 = tail call <16 x i32> @llvm.hexagon.V6.lo(<32 x i32> %dVsumv1.096) + %16 = tail call <16 x i32> @llvm.hexagon.V6.lo(<32 x i32> %14) + %17 = tail call <16 x i32> @llvm.hexagon.V6.valignbi(<16 x i32> %16, <16 x i32> %15, i32 4) + %18 = tail call <16 x i32> @llvm.hexagon.V6.hi(<32 x i32> %14) + %19 = tail call <16 x i32> @llvm.hexagon.V6.valignb(<16 x i32> %16, <16 x i32> %15, i32 8) + %20 = tail call <16 x i32> @llvm.hexagon.V6.valignb(<16 x i32> %18, <16 x i32> undef, i32 8) + %21 = tail call <16 x i32> @llvm.hexagon.V6.vaddw(<16 x i32> %17, <16 x i32> %19) + %22 = tail call <16 x i32> @llvm.hexagon.V6.vaddw(<16 x i32> %15, <16 x i32> %19) + %23 = tail call <16 x i32> @llvm.hexagon.V6.vmpyiwb.acc(<16 x i32> %22, <16 x i32> %17, i32 101058054) + %24 = tail call <16 x i32> @llvm.hexagon.V6.vmpyiwb.acc(<16 x i32> %23, <16 x i32> zeroinitializer, i32 67372036) + %25 = tail call <16 x i32> @llvm.hexagon.V6.vaddw(<16 x i32> undef, <16 x i32> %20) + %26 = tail call <16 x i32> @llvm.hexagon.V6.vmpyiwb.acc(<16 x i32> %25, <16 x i32> undef, i32 101058054) + %27 = tail call <16 x i32> @llvm.hexagon.V6.vmpyiwb.acc(<16 x i32> %26, <16 x i32> %21, i32 67372036) + %28 = tail call <16 x i32> @llvm.hexagon.V6.vasrwh(<16 x i32> %27, <16 x i32> %24, i32 8) + %incdec.ptr24 = getelementptr inbounds <16 x i32>, <16 x i32>* %optr.0102, i32 1 + store <16 x i32> %28, <16 x i32>* %optr.0102, align 64, !tbaa !1 + %incdec.ptr19.1 = getelementptr inbounds <16 x i32>, <16 x i32>* %iptr0.097, i32 2 + %incdec.ptr23.1 = getelementptr inbounds <16 x i32>, <16 x i32>* %iptr4.0101, i32 2 + %29 = tail call <16 x i32> @llvm.hexagon.V6.lo(<32 x i32> %14) + %30 = tail call <16 x i32> @llvm.hexagon.V6.hi(<32 x i32> %14) + %31 = tail call <16 x i32> @llvm.hexagon.V6.valignbi(<16 x i32> undef, <16 x i32> %29, i32 4) + %32 = tail call <16 x i32> @llvm.hexagon.V6.valignbi(<16 x i32> undef, <16 x i32> %30, i32 4) + %33 = tail call <16 x i32> @llvm.hexagon.V6.valignb(<16 x i32> undef, <16 x i32> %29, i32 8) + %34 = tail call <16 x i32> @llvm.hexagon.V6.valignb(<16 x i32> undef, <16 x i32> %30, i32 8) + %35 = tail call <16 x i32> @llvm.hexagon.V6.vaddw(<16 x i32> %31, <16 x i32> %33) + %36 = tail call <16 x i32> @llvm.hexagon.V6.vaddw(<16 x i32> %29, <16 x i32> %33) + %37 = tail call <16 x i32> @llvm.hexagon.V6.vmpyiwb.acc(<16 x i32> %36, <16 x i32> %31, i32 101058054) + %38 = tail call <16 x i32> @llvm.hexagon.V6.vmpyiwb.acc(<16 x i32> %37, <16 x i32> undef, i32 67372036) + %39 = tail call <16 x i32> @llvm.hexagon.V6.vaddw(<16 x i32> %30, <16 x i32> %34) + %40 = tail call <16 x i32> @llvm.hexagon.V6.vmpyiwb.acc(<16 x i32> %39, <16 x i32> %32, i32 101058054) + %41 = tail call <16 x i32> @llvm.hexagon.V6.vmpyiwb.acc(<16 x i32> %40, <16 x i32> %35, i32 67372036) + %42 = tail call <16 x i32> @llvm.hexagon.V6.vasrwh(<16 x i32> %41, <16 x i32> %38, i32 8) + %incdec.ptr24.1 = getelementptr inbounds <16 x i32>, <16 x i32>* %optr.0102, i32 2 + store <16 x i32> %42, <16 x i32>* %incdec.ptr24, align 64, !tbaa !1 + %incdec.ptr19.2 = getelementptr inbounds <16 x i32>, <16 x i32>* %iptr0.097, i32 3 + %43 = load <16 x i32>, <16 x i32>* %incdec.ptr19.1, align 64, !tbaa !1 + %incdec.ptr20.2 = getelementptr inbounds <16 x i32>, <16 x i32>* %iptr1.098, i32 3 + %incdec.ptr21.2 = getelementptr inbounds <16 x i32>, <16 x i32>* %iptr2.099, i32 3 + %incdec.ptr22.2 = getelementptr inbounds <16 x i32>, <16 x i32>* %iptr3.0100, i32 3 + %incdec.ptr23.2 = getelementptr inbounds <16 x i32>, <16 x i32>* %iptr4.0101, i32 3 + %44 = load <16 x i32>, <16 x i32>* %incdec.ptr23.1, align 64, !tbaa !1 + %45 = tail call <32 x i32> @llvm.hexagon.V6.vaddhw(<16 x i32> %43, <16 x i32> %44) + %46 = tail call <32 x i32> @llvm.hexagon.V6.vmpyhsat.acc(<32 x i32> %45, <16 x i32> undef, i32 393222) + %47 = tail call <32 x i32> @llvm.hexagon.V6.vmpahb.acc(<32 x i32> %46, <32 x i32> undef, i32 67372036) + %48 = tail call <16 x i32> @llvm.hexagon.V6.lo(<32 x i32> %47) + %49 = tail call <16 x i32> @llvm.hexagon.V6.valignbi(<16 x i32> %48, <16 x i32> undef, i32 4) + %50 = tail call <16 x i32> @llvm.hexagon.V6.valignb(<16 x i32> %48, <16 x i32> undef, i32 8) + %51 = tail call <16 x i32> @llvm.hexagon.V6.vaddw(<16 x i32> zeroinitializer, <16 x i32> undef) + %52 = tail call <16 x i32> @llvm.hexagon.V6.vaddw(<16 x i32> %49, <16 x i32> %50) + %53 = tail call <16 x i32> @llvm.hexagon.V6.vaddw(<16 x i32> undef, <16 x i32> %50) + %54 = tail call <16 x i32> @llvm.hexagon.V6.vmpyiwb.acc(<16 x i32> %53, <16 x i32> %49, i32 101058054) + %55 = tail call <16 x i32> @llvm.hexagon.V6.vmpyiwb.acc(<16 x i32> %54, <16 x i32> %51, i32 67372036) + %56 = tail call <16 x i32> @llvm.hexagon.V6.vmpyiwb.acc(<16 x i32> undef, <16 x i32> %52, i32 67372036) + %57 = tail call <16 x i32> @llvm.hexagon.V6.vasrwh(<16 x i32> %56, <16 x i32> %55, i32 8) + %incdec.ptr24.2 = getelementptr inbounds <16 x i32>, <16 x i32>* %optr.0102, i32 3 + store <16 x i32> %57, <16 x i32>* %incdec.ptr24.1, align 64, !tbaa !1 + %incdec.ptr19.3 = getelementptr inbounds <16 x i32>, <16 x i32>* %iptr0.097, i32 4 + %58 = load <16 x i32>, <16 x i32>* %incdec.ptr19.2, align 64, !tbaa !1 + %incdec.ptr20.3 = getelementptr inbounds <16 x i32>, <16 x i32>* %iptr1.098, i32 4 + %59 = load <16 x i32>, <16 x i32>* %incdec.ptr20.2, align 64, !tbaa !1 + %60 = load <16 x i32>, <16 x i32>* %incdec.ptr21.2, align 64, !tbaa !1 + %incdec.ptr22.3 = getelementptr inbounds <16 x i32>, <16 x i32>* %iptr3.0100, i32 4 + %61 = load <16 x i32>, <16 x i32>* %incdec.ptr22.2, align 64, !tbaa !1 + %incdec.ptr23.3 = getelementptr inbounds <16 x i32>, <16 x i32>* %iptr4.0101, i32 4 + %62 = load <16 x i32>, <16 x i32>* %incdec.ptr23.2, align 64, !tbaa !1 + %63 = tail call <32 x i32> @llvm.hexagon.V6.vaddhw(<16 x i32> %58, <16 x i32> %62) + %64 = tail call <32 x i32> @llvm.hexagon.V6.vmpyhsat.acc(<32 x i32> %63, <16 x i32> %60, i32 393222) + %65 = tail call <32 x i32> @llvm.hexagon.V6.vcombine(<16 x i32> %61, <16 x i32> %59) + %66 = tail call <32 x i32> @llvm.hexagon.V6.vmpahb.acc(<32 x i32> %64, <32 x i32> %65, i32 67372036) + %67 = tail call <16 x i32> @llvm.hexagon.V6.hi(<32 x i32> %47) + %68 = tail call <16 x i32> @llvm.hexagon.V6.lo(<32 x i32> %66) + %69 = tail call <16 x i32> @llvm.hexagon.V6.valignbi(<16 x i32> %68, <16 x i32> undef, i32 4) + %70 = tail call <16 x i32> @llvm.hexagon.V6.hi(<32 x i32> %66) + %71 = tail call <16 x i32> @llvm.hexagon.V6.valignbi(<16 x i32> %70, <16 x i32> %67, i32 4) + %72 = tail call <16 x i32> @llvm.hexagon.V6.valignb(<16 x i32> %70, <16 x i32> %67, i32 8) + %73 = tail call <16 x i32> @llvm.hexagon.V6.vaddw(<16 x i32> %67, <16 x i32> %71) + %74 = tail call <16 x i32> @llvm.hexagon.V6.vmpyiwb.acc(<16 x i32> undef, <16 x i32> %69, i32 101058054) + %75 = tail call <16 x i32> @llvm.hexagon.V6.vmpyiwb.acc(<16 x i32> %74, <16 x i32> %73, i32 67372036) + %76 = tail call <16 x i32> @llvm.hexagon.V6.vaddw(<16 x i32> %67, <16 x i32> %72) + %77 = tail call <16 x i32> @llvm.hexagon.V6.vmpyiwb.acc(<16 x i32> %76, <16 x i32> %71, i32 101058054) + %78 = tail call <16 x i32> @llvm.hexagon.V6.vmpyiwb.acc(<16 x i32> %77, <16 x i32> undef, i32 67372036) + %79 = tail call <16 x i32> @llvm.hexagon.V6.vasrwh(<16 x i32> %78, <16 x i32> %75, i32 8) + %incdec.ptr24.3 = getelementptr inbounds <16 x i32>, <16 x i32>* %optr.0102, i32 4 + store <16 x i32> %79, <16 x i32>* %incdec.ptr24.2, align 64, !tbaa !1 + %niter.nsub.3 = add i32 %niter, -4 + %niter.ncmp.3 = icmp eq i32 %niter.nsub.3, 0 + br i1 %niter.ncmp.3, label %for.end.loopexit.unr-lcssa, label %for.body + +for.end.loopexit.unr-lcssa: ; preds = %for.body, %entry + ret void +} + +declare <16 x i32> @llvm.hexagon.V6.hi(<32 x i32>) #0 +declare <16 x i32> @llvm.hexagon.V6.lo(<32 x i32>) #0 +declare <16 x i32> @llvm.hexagon.V6.vaddw(<16 x i32>, <16 x i32>) #0 +declare <16 x i32> @llvm.hexagon.V6.valignb(<16 x i32>, <16 x i32>, i32) #0 +declare <16 x i32> @llvm.hexagon.V6.valignbi(<16 x i32>, <16 x i32>, i32) #0 +declare <16 x i32> @llvm.hexagon.V6.vasrwh(<16 x i32>, <16 x i32>, i32) #0 +declare <16 x i32> @llvm.hexagon.V6.vmpyiwb.acc(<16 x i32>, <16 x i32>, i32) #0 +declare <32 x i32> @llvm.hexagon.V6.vaddhw(<16 x i32>, <16 x i32>) #0 +declare <32 x i32> @llvm.hexagon.V6.vcombine(<16 x i32>, <16 x i32>) #0 +declare <32 x i32> @llvm.hexagon.V6.vmpahb.acc(<32 x i32>, <32 x i32>, i32) #0 +declare <32 x i32> @llvm.hexagon.V6.vmpyhsat.acc(<32 x i32>, <16 x i32>, i32) #0 + +attributes #0 = { nounwind readnone } +attributes #1 = { nounwind "target-cpu"="hexagonv60" "target-features"="+hvx,-hvx-double" } + +!1 = !{!2, !2, i64 0} +!2 = !{!"omnipotent char", !3, i64 0} +!3 = !{!"Simple C/C++ TBAA"} diff --git a/test/CodeGen/Hexagon/fsel.ll b/test/CodeGen/Hexagon/fsel.ll new file mode 100644 index 000000000000..247249da50b1 --- /dev/null +++ b/test/CodeGen/Hexagon/fsel.ll @@ -0,0 +1,22 @@ +; RUN: llc -march=hexagon -O0 < %s | FileCheck %s + +; CHECK-LABEL: danny: +; CHECK: mux(p0, r1, ##1065353216) + +define float @danny(i32 %x, float %f) #0 { + %t = icmp sgt i32 %x, 0 + %u = select i1 %t, float %f, float 1.0 + ret float %u +} + +; CHECK-LABEL: sammy: +; CHECK: mux(p0, ##1069547520, r1) + +define float @sammy(i32 %x, float %f) #0 { + %t = icmp sgt i32 %x, 0 + %u = select i1 %t, float 1.5, float %f + ret float %u +} + +attributes #0 = { nounwind "target-cpu"="hexagonv5" } + diff --git a/test/CodeGen/Hexagon/hwloop-crit-edge.ll b/test/CodeGen/Hexagon/hwloop-crit-edge.ll index 4de4540c142e..f6e08eaf2e0c 100644 --- a/test/CodeGen/Hexagon/hwloop-crit-edge.ll +++ b/test/CodeGen/Hexagon/hwloop-crit-edge.ll @@ -1,4 +1,5 @@ ; RUN: llc -O3 -march=hexagon -mcpu=hexagonv5 < %s | FileCheck %s +; XFAIL: * ; ; Generate hardware loop when loop 'latch' block is different ; from the loop 'exiting' block. diff --git a/test/CodeGen/Hexagon/hwloop-loop1.ll b/test/CodeGen/Hexagon/hwloop-loop1.ll index 8b02736e0374..238d34e7ea15 100644 --- a/test/CodeGen/Hexagon/hwloop-loop1.ll +++ b/test/CodeGen/Hexagon/hwloop-loop1.ll @@ -2,8 +2,6 @@ ; ; Generate loop1 instruction for double loop sequence. -; CHECK: loop0(.LBB{{.}}_{{.}}, #100) -; CHECK: endloop0 ; CHECK: loop1(.LBB{{.}}_{{.}}, #100) ; CHECK: loop0(.LBB{{.}}_{{.}}, #100) ; CHECK: endloop0 diff --git a/test/CodeGen/Hexagon/hwloop-noreturn-call.ll b/test/CodeGen/Hexagon/hwloop-noreturn-call.ll new file mode 100644 index 000000000000..1045e2ed80a7 --- /dev/null +++ b/test/CodeGen/Hexagon/hwloop-noreturn-call.ll @@ -0,0 +1,63 @@ +; RUN: llc -march=hexagon < %s | FileCheck %s + +target triple = "hexagon" + +; CHECK-LABEL: danny: +; CHECK-DAG: loop0 +; CHECK-DAG: call trap +define void @danny(i32* %p, i32 %n, i32 %k) #0 { +entry: + br label %for.body + +for.body: ; preds = %entry + %t0 = phi i32 [ 0, %entry ], [ %t1, %for.cont ] + %t1 = add i32 %t0, 1 + %t2 = getelementptr i32, i32* %p, i32 %t0 + store i32 %t1, i32* %t2, align 4 + %c = icmp sgt i32 %t1, %k + br i1 %c, label %noret, label %for.cont + +for.cont: + %cmp = icmp slt i32 %t0, %n + br i1 %cmp, label %for.body, label %for.end + +for.end: ; preds = %for.cond + ret void + +noret: + call void @trap() #1 + br label %for.cont +} + +; CHECK-LABEL: sammy: +; CHECK-DAG: loop0 +; CHECK-DAG: callr +define void @sammy(i32* %p, i32 %n, i32 %k, void (...)* %f) #0 { +entry: + br label %for.body + +for.body: ; preds = %entry + %t0 = phi i32 [ 0, %entry ], [ %t1, %for.cont ] + %t1 = add i32 %t0, 1 + %t2 = getelementptr i32, i32* %p, i32 %t0 + store i32 %t1, i32* %t2, align 4 + %c = icmp sgt i32 %t1, %k + br i1 %c, label %noret, label %for.cont + +for.cont: + %cmp = icmp slt i32 %t0, %n + br i1 %cmp, label %for.body, label %for.end + +for.end: ; preds = %for.cond + ret void + +noret: + call void (...) %f() #1 + br label %for.cont +} + +declare void @trap() #1 + +attributes #0 = { nounwind "target-cpu"="hexagonv60" "target-features"="+hvx,-hvx-double" } +attributes #1 = { nounwind noreturn } + diff --git a/test/CodeGen/Hexagon/hwloop-preh.ll b/test/CodeGen/Hexagon/hwloop-preh.ll new file mode 100644 index 000000000000..e92461f43da5 --- /dev/null +++ b/test/CodeGen/Hexagon/hwloop-preh.ll @@ -0,0 +1,44 @@ +; RUN: llc -march=hexagon -disable-machine-licm -hwloop-spec-preheader=1 < %s | FileCheck %s +; CHECK: loop0 + +target triple = "hexagon" + +define i32 @foo(i32 %x, i32 %n, i32* nocapture %A, i32* nocapture %B) #0 { +entry: + %cmp = icmp sgt i32 %x, 0 + br i1 %cmp, label %for.cond.preheader, label %return + +for.cond.preheader: ; preds = %entry + %cmp16 = icmp sgt i32 %n, 0 + br i1 %cmp16, label %for.body.preheader, label %return + +for.body.preheader: ; preds = %for.cond.preheader + br label %for.body + +for.body: ; preds = %for.body.preheader, %for.body + %arrayidx.phi = phi i32* [ %arrayidx.inc, %for.body ], [ %B, %for.body.preheader ] + %arrayidx2.phi = phi i32* [ %arrayidx2.inc, %for.body ], [ %A, %for.body.preheader ] + %i.07 = phi i32 [ %inc, %for.body ], [ 0, %for.body.preheader ] + %0 = load i32, i32* %arrayidx.phi, align 4, !tbaa !0 + %1 = load i32, i32* %arrayidx2.phi, align 4, !tbaa !0 + %add = add nsw i32 %1, %0 + store i32 %add, i32* %arrayidx2.phi, align 4, !tbaa !0 + %inc = add nsw i32 %i.07, 1 + %exitcond = icmp eq i32 %inc, %n + %arrayidx.inc = getelementptr i32, i32* %arrayidx.phi, i32 1 + %arrayidx2.inc = getelementptr i32, i32* %arrayidx2.phi, i32 1 + br i1 %exitcond, label %return.loopexit, label %for.body + +return.loopexit: ; preds = %for.body + br label %return + +return: ; preds = %return.loopexit, %for.cond.preheader, %entry + %retval.0 = phi i32 [ 2, %entry ], [ 0, %for.cond.preheader ], [ 0, %return.loopexit ] + ret i32 %retval.0 +} + +!0 = !{!"int", !1} +!1 = !{!"omnipotent char", !2} +!2 = !{!"Simple C/C++ TBAA"} + +attributes #0 = { nounwind "target-cpu"="hexagonv60" "target-features"="-hvx,-hvx-double" } diff --git a/test/CodeGen/Hexagon/hwloop1.ll b/test/CodeGen/Hexagon/hwloop1.ll index 97b779cf9628..68af3b34eeeb 100644 --- a/test/CodeGen/Hexagon/hwloop1.ll +++ b/test/CodeGen/Hexagon/hwloop1.ll @@ -1,4 +1,4 @@ -; RUN: llc -march=hexagon < %s | FileCheck %s +; RUN: llc -march=hexagon -enable-pipeliner=false < %s | FileCheck %s ; Check that we generate hardware loop instructions. ; Case 1 : Loop with a constant number of iterations. diff --git a/test/CodeGen/Hexagon/ifcvt-diamond-bug-2016-08-26.ll b/test/CodeGen/Hexagon/ifcvt-diamond-bug-2016-08-26.ll new file mode 100644 index 000000000000..68a5dc16ecff --- /dev/null +++ b/test/CodeGen/Hexagon/ifcvt-diamond-bug-2016-08-26.ll @@ -0,0 +1,37 @@ +; RUN: llc -march=hexagon -o - %s | FileCheck %s +target triple = "hexagon" + +%struct.0 = type { i16, i16 } + +@t = external local_unnamed_addr global %struct.0, align 2 + +define void @foo(i32 %p) local_unnamed_addr #0 { +entry: + %conv90 = trunc i32 %p to i16 + %call105 = call signext i16 @bar(i16 signext 16384, i16 signext undef) #0 + %call175 = call signext i16 @bar(i16 signext %conv90, i16 signext 4) #0 + %call197 = call signext i16 @bar(i16 signext %conv90, i16 signext 4) #0 + %cmp199 = icmp eq i16 %call197, 0 + br i1 %cmp199, label %if.then200, label %if.else201 + +; CHECK-DAG: [[R4:r[0-9]+]] = #4 +; CHECK: p0 = cmp.eq(r0, #0) +; CHECK: if (!p0.new) [[R3:r[0-9]+]] = #3 +; CHECK-DAG: if (!p0) memh(##t) = [[R3]] +; CHECK-DAG: if (p0) memh(##t) = [[R4]] +if.then200: ; preds = %entry + store i16 4, i16* getelementptr inbounds (%struct.0, %struct.0* @t, i32 0, i32 0), align 2 + store i16 0, i16* getelementptr inbounds (%struct.0, %struct.0* @t, i32 0, i32 1), align 2 + br label %if.end202 + +if.else201: ; preds = %entry + store i16 3, i16* getelementptr inbounds (%struct.0, %struct.0* @t, i32 0, i32 0), align 2 + br label %if.end202 + +if.end202: ; preds = %if.else201, %if.then200 + ret void +} + +declare signext i16 @bar(i16 signext, i16 signext) local_unnamed_addr #0 + +attributes #0 = { optsize "target-cpu"="hexagonv55" } diff --git a/test/CodeGen/Hexagon/ifcvt-impuse-livein.mir b/test/CodeGen/Hexagon/ifcvt-impuse-livein.mir new file mode 100644 index 000000000000..780b9cedf7fa --- /dev/null +++ b/test/CodeGen/Hexagon/ifcvt-impuse-livein.mir @@ -0,0 +1,42 @@ +# RUN: llc -march=hexagon -run-pass if-converter %s -o - | FileCheck %s + +# Make sure that the necessary implicit uses are added to predicated +# instructions. + +# CHECK-LABEL: name: foo + +--- | + define void @foo() { + ret void + } +... + +--- +name: foo +tracksRegLiveness: true +body: | + bb.0: + successors: %bb.1, %bb.2 + liveins: %r0, %r2, %p1 + J2_jumpf %p1, %bb.1, implicit-def %pc + J2_jump %bb.2, implicit-def %pc + bb.1: + successors: %bb.3 + liveins: %r2 + %r0 = A2_tfrsi 2 + J2_jump %bb.3, implicit-def %pc + bb.2: + successors: %bb.3 + liveins: %r0 + ; Even though r2 was not live on entry to this block, it was live across + ; block bb.1 in the original diamond. After if-conversion, the diamond + ; became a single block, and so r2 is now live on entry to the instructions + ; originating from bb.2. + ; CHECK: %r2 = C2_cmoveit %p1, 1, implicit %r2 + %r2 = A2_tfrsi 1 + bb.3: + liveins: %r0, %r2 + %r0 = A2_add %r0, %r2 + J2_jumpr %r31, implicit-def %pc +... + diff --git a/test/CodeGen/Hexagon/ifcvt-live-subreg.mir b/test/CodeGen/Hexagon/ifcvt-live-subreg.mir new file mode 100644 index 000000000000..12cc086bd34b --- /dev/null +++ b/test/CodeGen/Hexagon/ifcvt-live-subreg.mir @@ -0,0 +1,50 @@ +# RUN: llc -march=hexagon -run-pass if-converter -o - %s | FileCheck %s +# Check that an implicit use is generated for a predicated instruction +# when a subregister of the redefined register is live. + +# CHECK-LABEL: name: foo + +# Verify the predicated block: +# CHECK-LABEL: bb.0: +# CHECK: liveins: %r0, %r1, %p0, %d8 +# CHECK: %d8 = A2_combinew killed %r0, killed %r1 +# CHECK: %d8 = L2_ploadrdf_io %p0, %r29, 0, implicit %d8 +# CHECK: J2_jumprf %p0, killed %r31, implicit-def %pc, implicit-def %pc, implicit killed %d8 + +--- | + define void @foo() { + ret void + } +... + + +--- +name: foo +alignment: 4 +tracksRegLiveness: true +liveins: + - { reg: '%r0' } + - { reg: '%r1' } + - { reg: '%p0' } + - { reg: '%d8' } +body: | + bb.0: + successors: %bb.1, %bb.2 + liveins: %r0, %r1, %p0, %d8 + %d8 = A2_combinew killed %r0, killed %r1 + J2_jumpf killed %p0, %bb.2, implicit-def %pc + + bb.1: + liveins: %d0, %r17 + %r0 = A2_tfrsi 0 + %r1 = A2_tfrsi 0 + A2_nop ; non-predicable + J2_jumpr killed %r31, implicit-def dead %pc, implicit killed %d0 + + bb.2: + ; Predicate this block. + %d8 = L2_loadrd_io %r29, 0 + J2_jumpr killed %r31, implicit-def dead %pc, implicit killed %d8 + +... + diff --git a/test/CodeGen/Hexagon/inline-asm-hexagon.ll b/test/CodeGen/Hexagon/inline-asm-hexagon.ll new file mode 100644 index 000000000000..302096d49b3e --- /dev/null +++ b/test/CodeGen/Hexagon/inline-asm-hexagon.ll @@ -0,0 +1,16 @@ +; RUN: llc -march=hexagon < %s | FileCheck %s + +target triple = "hexagon" + +;CHECK: [[REGH:r[0-9]]]:[[REGL:[0-9]]] = memd_locked +;CHECK: HIGH([[REGH]]) +;CHECK: LOW(r[[REGL]]) +define i32 @fred(i64* %free_list_ptr, i32** %item_ptr, i8** %free_item_ptr) nounwind { +entry: + %free_list_ptr.addr = alloca i64*, align 4 + store i64* %free_list_ptr, i64** %free_list_ptr.addr, align 4 + %0 = load i32*, i32** %item_ptr, align 4 + %1 = call { i64, i32 } asm sideeffect "1: $0 = memd_locked($5)\0A\09 $1 = HIGH(${0:H}) \0A\09 $1 = add($1,#1) \0A\09 memw($6) = LOW(${0:L}) \0A\09 $0 = combine($7,$1) \0A\09 memd_locked($5,p0) = $0 \0A\09 if !p0 jump 1b\0A\09", "=&r,=&r,=*m,=*m,r,r,r,r,*m,*m,~{p0}"(i64** %free_list_ptr.addr, i8** %free_item_ptr, i64 0, i64* %free_list_ptr, i8** %free_item_ptr, i32* %0, i64** %free_list_ptr.addr, i8** %free_item_ptr) nounwind + %asmresult1 = extractvalue { i64, i32 } %1, 1 + ret i32 %asmresult1 +} diff --git a/test/CodeGen/Hexagon/inline-asm-i1.ll b/test/CodeGen/Hexagon/inline-asm-i1.ll new file mode 100644 index 000000000000..cc1f4ce55dad --- /dev/null +++ b/test/CodeGen/Hexagon/inline-asm-i1.ll @@ -0,0 +1,14 @@ +; RUN: llc -march=hexagon < %s | FileCheck %s +; CHECK: r[[REG0:[0-9]+]] = usr +; CHECK: [[REG0]] = insert(r{{[0-9]+}}, #1, #16) + +target triple = "hexagon" + +define hidden void @fred() #0 { +entry: + %0 = call { i32, i32 } asm sideeffect " $0 = usr\0A $1 = $2\0A $0 = insert($1, #1, #16)\0Ausr = $0 \0A", "=&r,=&r,r"(i1 undef) #1 + ret void +} + +attributes #0 = { nounwind "target-cpu"="hexagonv60" } +attributes #1 = { nounwind } diff --git a/test/CodeGen/Hexagon/insert4.ll b/test/CodeGen/Hexagon/insert4.ll index 96c8bba24d7c..c4d575dd4060 100644 --- a/test/CodeGen/Hexagon/insert4.ll +++ b/test/CodeGen/Hexagon/insert4.ll @@ -1,9 +1,9 @@ ; RUN: llc -march=hexagon < %s | FileCheck %s -; Check that we are generating insert instructions. -; CHECK: insert -; CHECK: insert -; CHECK: insert -; CHECK: insert +; +; Check that we no longer generate 4 inserts. +; CHECK: combine(r{{[0-9]+}}.l, r{{[0-9]+}}.l) +; CHECK: combine(r{{[0-9]+}}.l, r{{[0-9]+}}.l) +; CHECK-NOT: insert target datalayout = "e-p:32:32:32-i64:64:64-i32:32:32-i16:16:16-i1:32:32-f64:64:64-f32:32:32-v64:64:64-v32:32:32-a0:0-n16:32" target triple = "hexagon" diff --git a/test/CodeGen/Hexagon/intrinsics/llsc_bundling.ll b/test/CodeGen/Hexagon/intrinsics/llsc_bundling.ll new file mode 100644 index 000000000000..966945b66f47 --- /dev/null +++ b/test/CodeGen/Hexagon/intrinsics/llsc_bundling.ll @@ -0,0 +1,12 @@ +; RUN: llc -march=hexagon < %s +target triple = "hexagon-unknown--elf" + +; Function Attrs: norecurse nounwind +define void @_Z4lockv() #0 { +entry: + %__shared_owners = alloca i32, align 4 + %0 = cmpxchg weak i32* %__shared_owners, i32 0, i32 1 seq_cst seq_cst + ret void +} + +attributes #0 = { nounwind } diff --git a/test/CodeGen/Hexagon/is-legal-void.ll b/test/CodeGen/Hexagon/is-legal-void.ll new file mode 100644 index 000000000000..222934abb82d --- /dev/null +++ b/test/CodeGen/Hexagon/is-legal-void.ll @@ -0,0 +1,58 @@ +; RUN: llc -march=hexagon < %s +; REQUIRES: asserts + +; The two loads based on %struct.0, loading two different data types +; cause LSR to assume type "void" for the memory type. This would then +; cause an assert in isLegalAddressingMode. Make sure we no longer crash. + +target triple = "hexagon" + +%struct.0 = type { i8*, i8, %union.anon.0 } +%union.anon.0 = type { i8* } + +define hidden fastcc void @fred() unnamed_addr #0 { +entry: + br i1 undef, label %while.end, label %while.body.lr.ph + +while.body.lr.ph: ; preds = %entry + br label %while.body + +while.body: ; preds = %exit.2, %while.body.lr.ph + %lsr.iv = phi %struct.0* [ %cgep22, %exit.2 ], [ undef, %while.body.lr.ph ] + switch i32 undef, label %exit [ + i32 1, label %sw.bb.i + i32 2, label %sw.bb3.i + ] + +sw.bb.i: ; preds = %while.body + unreachable + +sw.bb3.i: ; preds = %while.body + unreachable + +exit: ; preds = %while.body + switch i32 undef, label %exit.2 [ + i32 1, label %sw.bb.i17 + i32 2, label %sw.bb3.i20 + ] + +sw.bb.i17: ; preds = %.exit + %0 = bitcast %struct.0* %lsr.iv to i32* + %1 = load i32, i32* %0, align 4 + unreachable + +sw.bb3.i20: ; preds = %exit + %2 = bitcast %struct.0* %lsr.iv to i8** + %3 = load i8*, i8** %2, align 4 + unreachable + +exit.2: ; preds = %exit + %cgep22 = getelementptr %struct.0, %struct.0* %lsr.iv, i32 1 + br label %while.body + +while.end: ; preds = %entry + ret void +} + +attributes #0 = { nounwind optsize "target-cpu"="hexagonv55" } + diff --git a/test/CodeGen/Hexagon/livephysregs-lane-masks.mir b/test/CodeGen/Hexagon/livephysregs-lane-masks.mir new file mode 100644 index 000000000000..b2e1968bb59a --- /dev/null +++ b/test/CodeGen/Hexagon/livephysregs-lane-masks.mir @@ -0,0 +1,40 @@ +# RUN: llc -march=hexagon -run-pass if-converter -verify-machineinstrs -o - %s | FileCheck %s + +# CHECK-LABEL: name: foo +# CHECK: %p0 = C2_cmpeqi %r16, 0 +# Make sure there is no implicit use of r1. +# CHECK: %r1 = L2_ploadruhf_io %p0, %r29, 6 + +--- | + define void @foo() { + ret void + } +... + + +--- +name: foo +tracksRegLiveness: true + +body: | + bb.0: + liveins: %r16 + successors: %bb.1, %bb.2 + %p0 = C2_cmpeqi %r16, 0 + J2_jumpt %p0, %bb.2, implicit-def %pc + + bb.1: + ; The lane mask %d0:0002 is equivalent to %r0. LivePhysRegs would ignore + ; it and treat it as the whole %d0, which is a pair %r1, %r0. The extra + ; %r1 would cause an (undefined) implicit use to be added during + ; if-conversion. + liveins: %d0:0x00000002, %d15:0x00000001, %r16 + successors: %bb.2 + %r1 = L2_loadruh_io %r29, 6 + S2_storeri_io killed %r16, 0, %r1 + + bb.2: + liveins: %r0 + %d8 = L2_loadrd_io %r29, 8 + L4_return implicit-def %r29, implicit-def %r30, implicit-def %r31, implicit-def %pc, implicit %r30 + diff --git a/test/CodeGen/Hexagon/livephysregs-lane-masks2.mir b/test/CodeGen/Hexagon/livephysregs-lane-masks2.mir new file mode 100644 index 000000000000..586857016551 --- /dev/null +++ b/test/CodeGen/Hexagon/livephysregs-lane-masks2.mir @@ -0,0 +1,55 @@ +# RUN: llc -march=hexagon -verify-machineinstrs -run-pass branch-folder -o - %s | FileCheck %s + +# CHECK-LABEL: name: fred + +--- | + define void @fred() { ret void } + +... +--- + +name: fred +tracksRegLiveness: true + +body: | + bb.0: + liveins: %p2, %r0 + successors: %bb.1, %bb.2 + J2_jumpt killed %p2, %bb.1, implicit-def %pc + J2_jump %bb.2, implicit-def %pc + + bb.1: + liveins: %r0, %r19 + successors: %bb.3 + %r2 = A2_tfrsi 4 + %r1 = COPY %r19 + %r0 = S2_asl_r_r killed %r0, killed %r2 + %r0 = A2_asrh killed %r0 + J2_jump %bb.3, implicit-def %pc + + bb.2: + liveins: %r0, %r18 + successors: %bb.3 + %r2 = A2_tfrsi 5 + %r1 = L2_loadrh_io %r18, 0 + %r0 = S2_asl_r_r killed %r0, killed %r2 + %r0 = A2_asrh killed %r0 + + bb.3: + ; A live-in register without subregs, but with a lane mask that is not ~0 + ; is not recognized by LivePhysRegs. Branch folding exposes this problem + ; (through tail merging). + ; + ; CHECK: bb.3: + ; CHECK: liveins:{{.*}}%p0 + ; CHECK: %r0 = S2_asl_r_r killed %r0, killed %r2 + ; CHECK: %r0 = A2_asrh killed %r0 + ; CHECK: %r0 = C2_cmoveit killed %p0, 1 + ; CHECK: J2_jumpr %r31, implicit-def %pc, implicit %r0 + ; + liveins: %p0:0x1 + %r0 = C2_cmoveit killed %p0, 1 + J2_jumpr %r31, implicit-def %pc, implicit %r0 +... + + diff --git a/test/CodeGen/Hexagon/long-calls.ll b/test/CodeGen/Hexagon/long-calls.ll new file mode 100644 index 000000000000..9f9a527a542f --- /dev/null +++ b/test/CodeGen/Hexagon/long-calls.ll @@ -0,0 +1,73 @@ +; RUN: llc -march=hexagon -enable-save-restore-long < %s | FileCheck %s + +; Check that the -long-calls feature is supported by the backend. + +; CHECK: call ##foo +; CHECK: jump ##__restore +define i64 @test_longcall(i32 %x, i32 %y) #0 { +entry: + %add = add nsw i32 %x, 5 + %call = tail call i64 @foo(i32 %add) #6 + %conv = sext i32 %y to i64 + %add1 = add nsw i64 %call, %conv + ret i64 %add1 +} + +; CHECK: jump ##foo +define i64 @test_longtailcall(i32 %x, i32 %y) #1 { +entry: + %add = add nsw i32 %x, 5 + %call = tail call i64 @foo(i32 %add) #6 + ret i64 %call +} + +; CHECK: call ##bar +define i64 @test_longnoret(i32 %x, i32 %y) #2 { +entry: + %add = add nsw i32 %x, 5 + %0 = tail call i64 @bar(i32 %add) #7 + unreachable +} + +; CHECK: call foo +; CHECK: jump ##__restore +; The restore call will still be long because of the enable-save-restore-long +; option being used. +define i64 @test_shortcall(i32 %x, i32 %y) #3 { +entry: + %add = add nsw i32 %x, 5 + %call = tail call i64 @foo(i32 %add) #6 + %conv = sext i32 %y to i64 + %add1 = add nsw i64 %call, %conv + ret i64 %add1 +} + +; CHECK: jump foo +define i64 @test_shorttailcall(i32 %x, i32 %y) #4 { +entry: + %add = add nsw i32 %x, 5 + %call = tail call i64 @foo(i32 %add) #6 + ret i64 %call +} + +; CHECK: call bar +define i64 @test_shortnoret(i32 %x, i32 %y) #5 { +entry: + %add = add nsw i32 %x, 5 + %0 = tail call i64 @bar(i32 %add) #7 + unreachable +} + +declare i64 @foo(i32) #6 +declare i64 @bar(i32) #7 + +attributes #0 = { minsize nounwind "target-cpu"="hexagonv60" "target-features"="+long-calls" } +attributes #1 = { nounwind "target-cpu"="hexagonv60" "target-features"="+long-calls" } +attributes #2 = { noreturn nounwind "target-cpu"="hexagonv60" "target-features"="+long-calls" } + +attributes #3 = { minsize nounwind "target-cpu"="hexagonv60" "target-features"="-long-calls" } +attributes #4 = { nounwind "target-cpu"="hexagonv60" "target-features"="-long-calls" } +attributes #5 = { noreturn nounwind "target-cpu"="hexagonv60" "target-features"="-long-calls" } + +attributes #6 = { noreturn "target-cpu"="hexagonv60" } +attributes #7 = { noreturn nounwind "target-cpu"="hexagonv60" } diff --git a/test/CodeGen/Hexagon/loop-prefetch.ll b/test/CodeGen/Hexagon/loop-prefetch.ll new file mode 100644 index 000000000000..0c6e4581a71f --- /dev/null +++ b/test/CodeGen/Hexagon/loop-prefetch.ll @@ -0,0 +1,27 @@ +; RUN: llc -march=hexagon -hexagon-loop-prefetch < %s | FileCheck %s +; CHECK: dcfetch + +target triple = "hexagon" + +define void @copy(i32* nocapture %d, i32* nocapture readonly %s, i32 %n) local_unnamed_addr #0 { +entry: + %tobool2 = icmp eq i32 %n, 0 + br i1 %tobool2, label %while.end, label %while.body + +while.body: ; preds = %entry, %while.body + %n.addr.05 = phi i32 [ %dec, %while.body ], [ %n, %entry ] + %s.addr.04 = phi i32* [ %incdec.ptr, %while.body ], [ %s, %entry ] + %d.addr.03 = phi i32* [ %incdec.ptr1, %while.body ], [ %d, %entry ] + %dec = add i32 %n.addr.05, -1 + %incdec.ptr = getelementptr inbounds i32, i32* %s.addr.04, i32 1 + %0 = load i32, i32* %s.addr.04, align 4 + %incdec.ptr1 = getelementptr inbounds i32, i32* %d.addr.03, i32 1 + store i32 %0, i32* %d.addr.03, align 4 + %tobool = icmp eq i32 %dec, 0 + br i1 %tobool, label %while.end, label %while.body + +while.end: ; preds = %while.body, %entry + ret void +} + +attributes #0 = { norecurse nounwind "target-cpu"="hexagonv60" "target-features"="-hvx,-hvx-double" } diff --git a/test/CodeGen/Hexagon/lower-extract-subvector.ll b/test/CodeGen/Hexagon/lower-extract-subvector.ll new file mode 100644 index 000000000000..ba67de9e00a4 --- /dev/null +++ b/test/CodeGen/Hexagon/lower-extract-subvector.ll @@ -0,0 +1,47 @@ +; RUN: llc -march=hexagon -O3 < %s | FileCheck %s + +; This test checks if we custom lower extract_subvector. If we cannot +; custom lower extract_subvector this test makes the compiler crash. + +; CHECK: vmem +target triple = "hexagon-unknown--elf" + +; Function Attrs: nounwind +define void @__processed() #0 { +entry: + br label %"for matrix.s0.y" + +"for matrix.s0.y": ; preds = %"for matrix.s0.y", %entry + br i1 undef, label %"produce processed", label %"for matrix.s0.y" + +"produce processed": ; preds = %"for matrix.s0.y" + br i1 undef, label %"for processed.s0.ty.ty.preheader", label %"consume processed" + +"for processed.s0.ty.ty.preheader": ; preds = %"produce processed" + br i1 undef, label %"for denoised.s0.y.preheader", label %"consume denoised" + +"for denoised.s0.y.preheader": ; preds = %"for processed.s0.ty.ty.preheader" + unreachable + +"consume denoised": ; preds = %"for processed.s0.ty.ty.preheader" + br i1 undef, label %"consume deinterleaved", label %if.then.i164 + +if.then.i164: ; preds = %"consume denoised" + unreachable + +"consume deinterleaved": ; preds = %"consume denoised" + %0 = tail call <64 x i32> @llvm.hexagon.V6.vshuffvdd.128B(<32 x i32> undef, <32 x i32> undef, i32 -2) + %1 = bitcast <64 x i32> %0 to <128 x i16> + %2 = shufflevector <128 x i16> %1, <128 x i16> undef, <64 x i32> + store <64 x i16> %2, <64 x i16>* undef, align 128 + unreachable + +"consume processed": ; preds = %"produce processed" + ret void +} + +; Function Attrs: nounwind readnone +declare <64 x i32> @llvm.hexagon.V6.vshuffvdd.128B(<32 x i32>, <32 x i32>, i32) #1 + +attributes #0 = { nounwind "target-cpu"="hexagonv60" "target-features"="+hvx,+hvx-double" } +attributes #1 = { nounwind readnone "target-cpu"="hexagonv60" "target-features"="+hvx,+hvx-double" } diff --git a/test/CodeGen/Hexagon/misaligned_double_vector_store_not_fast.ll b/test/CodeGen/Hexagon/misaligned_double_vector_store_not_fast.ll new file mode 100644 index 000000000000..25cb14e8514e --- /dev/null +++ b/test/CodeGen/Hexagon/misaligned_double_vector_store_not_fast.ll @@ -0,0 +1,47 @@ +; RUN: llc -march=hexagon -O3 -debug-only=isel 2>&1 < %s | FileCheck %s +; REQUIRES: asserts + +; DAGCombiner converts the two vector stores to a double vector store, +; even if the double vector store is unaligned. This is not good. If it +; is unaligned, we should let the DAGCombiner know that it is slow via +; the allowsMisalignedAccess function in HexagonISelLowering. + +; CHECK-NOT: store + +target triple = "hexagon-unknown--elf" + +; Function Attrs: nounwind +define void @__processed() #0 { +entry: + br label %"for demosaiced.s0.y.y" + +"for demosaiced.s0.y.y": ; preds = %"for demosaiced.s0.y.y", %entry + %demosaiced.s0.y.y = phi i32 [ 0, %entry ], [ %0, %"for demosaiced.s0.y.y" ] + %0 = add nuw nsw i32 %demosaiced.s0.y.y, 1 + %1 = mul nuw nsw i32 %demosaiced.s0.y.y, 256 + %2 = tail call <64 x i32> @llvm.hexagon.V6.vshuffvdd.128B(<32 x i32> undef, <32 x i32> undef, i32 -2) + %3 = bitcast <64 x i32> %2 to <128 x i16> + %4 = shufflevector <128 x i16> %3, <128 x i16> undef, <64 x i32> + %5 = add nuw nsw i32 %1, 32896 + %6 = getelementptr inbounds i16, i16* undef, i32 %5 + %7 = bitcast i16* %6 to <64 x i16>* + store <64 x i16> %4, <64 x i16>* %7, align 128 + %8 = shufflevector <128 x i16> %3, <128 x i16> undef, <64 x i32> + %9 = add nuw nsw i32 %1, 32960 + %10 = getelementptr inbounds i16, i16* undef, i32 %9 + %11 = bitcast i16* %10 to <64 x i16>* + store <64 x i16> %8, <64 x i16>* %11, align 128 + br i1 false, label %"consume demosaiced", label %"for demosaiced.s0.y.y" + +"consume demosaiced": ; preds = %"for demosaiced.s0.y.y" + unreachable + +"consume processed": ; preds = %"produce processed" + ret void +} + +declare <64 x i32> @llvm.hexagon.V6.vshuffvdd.128B(<32 x i32>, <32 x i32>, i32) #1 + +attributes #0 = { nounwind "target-cpu"="hexagonv60" "target-features"="+hvx,+hvx-double" } +attributes #1 = { nounwind readnone "target-cpu"="hexagonv60" "target-features"="+hvx,+hvx-double" } + diff --git a/test/CodeGen/Hexagon/mulhs.ll b/test/CodeGen/Hexagon/mulhs.ll new file mode 100644 index 000000000000..b8727da8a7e2 --- /dev/null +++ b/test/CodeGen/Hexagon/mulhs.ll @@ -0,0 +1,23 @@ +; RUN: llc -march=hexagon < %s | FileCheck %s + +; CHECK: mpy +; CHECK-NOT: call + +target triple = "hexagon" + +; Function Attrs: nounwind +define i32 @fred(i64 %x, i64 %y, i64* nocapture %z) #0 { +entry: + %0 = tail call { i64, i1 } @llvm.smul.with.overflow.i64(i64 %x, i64 %y) + %1 = extractvalue { i64, i1 } %0, 1 + %2 = extractvalue { i64, i1 } %0, 0 + store i64 %2, i64* %z, align 8 + %conv = zext i1 %1 to i32 + ret i32 %conv +} + +; Function Attrs: nounwind readnone +declare { i64, i1 } @llvm.smul.with.overflow.i64(i64, i64) #1 + +attributes #0 = { nounwind } +attributes #1 = { nounwind readnone } diff --git a/test/CodeGen/Hexagon/newvalueSameReg.ll b/test/CodeGen/Hexagon/newvalueSameReg.ll new file mode 100644 index 000000000000..0fc4df22eb32 --- /dev/null +++ b/test/CodeGen/Hexagon/newvalueSameReg.ll @@ -0,0 +1,63 @@ +; RUN: llc -march=hexagon -hexagon-expand-condsets=0 < %s | FileCheck %s +; +; Expand-condsets eliminates the "mux" instruction, which is what this +; testcase is checking. + +%struct._Dnk_filet.1 = type { i16, i8, i32, i8*, i8*, i8*, i8*, i8*, i8*, i32*, [2 x i32], i8*, i8*, i8*, %struct._Mbstatet.0, i8*, [8 x i8], i8 } +%struct._Mbstatet.0 = type { i32, i16, i16 } + +@_Stdout = external global %struct._Dnk_filet.1 +@.str = external unnamed_addr constant [23 x i8], align 8 + +; Test that we don't generate a new value compare if the operands are +; the same register. + +; CHECK-NOT: cmp.eq([[REG0:(r[0-9]+)]].new, [[REG0]]) +; CHECK: cmp.eq([[REG1:(r[0-9]+)]], [[REG1]]) + +; Function Attrs: nounwind +declare void @fprintf(%struct._Dnk_filet.1* nocapture, i8* nocapture readonly, ...) #1 + +define void @main() #0 { +entry: + %0 = load i32*, i32** undef, align 4 + %1 = load i32, i32* undef, align 4 + br i1 undef, label %if.end, label %_ZNSt6vectorIbSaIbEE3endEv.exit + +_ZNSt6vectorIbSaIbEE3endEv.exit: + %2 = icmp slt i32 %1, 0 + %sub5.i.i.i = lshr i32 %1, 5 + %add619.i.i.i = add i32 %sub5.i.i.i, -134217728 + %sub5.i.pn.i.i = select i1 %2, i32 %add619.i.i.i, i32 %sub5.i.i.i + %storemerge2.i.i = getelementptr inbounds i32, i32* %0, i32 %sub5.i.pn.i.i + %cmp.i.i = icmp ult i32* %storemerge2.i.i, %0 + %.mux = select i1 %cmp.i.i, i32 0, i32 1 + br i1 undef, label %_ZNSt6vectorIbSaIbEE3endEv.exit57, label %if.end + +_ZNSt6vectorIbSaIbEE3endEv.exit57: + %3 = icmp slt i32 %1, 0 + %sub5.i.i.i44 = lshr i32 %1, 5 + %add619.i.i.i45 = add i32 %sub5.i.i.i44, -134217728 + %sub5.i.pn.i.i46 = select i1 %3, i32 %add619.i.i.i45, i32 %sub5.i.i.i44 + %storemerge2.i.i47 = getelementptr inbounds i32, i32* %0, i32 %sub5.i.pn.i.i46 + %cmp.i38 = icmp ult i32* %storemerge2.i.i47, %0 + %.reg2mem.sroa.0.sroa.0.0.load14.i.reload = select i1 %cmp.i38, i32 0, i32 1 + %cmp = icmp eq i32 %.mux, %.reg2mem.sroa.0.sroa.0.0.load14.i.reload + br i1 %cmp, label %if.end, label %if.then + +if.then: + call void (%struct._Dnk_filet.1*, i8*, ...) @fprintf(%struct._Dnk_filet.1* @_Stdout, i8* getelementptr inbounds ([23 x i8], [23 x i8]* @.str, i32 0, i32 0), i32 %.mux, i32 %.reg2mem.sroa.0.sroa.0.0.load14.i.reload) #1 + unreachable + +if.end: + br i1 undef, label %_ZNSt6vectorIbSaIbEED2Ev.exit, label %if.then.i.i.i + +if.then.i.i.i: + unreachable + +_ZNSt6vectorIbSaIbEED2Ev.exit: + ret void +} + +attributes #0 = { "target-cpu"="hexagonv5" } +attributes #1 = { nounwind "target-cpu"="hexagonv5" } diff --git a/test/CodeGen/Hexagon/opt-spill-volatile.ll b/test/CodeGen/Hexagon/opt-spill-volatile.ll new file mode 100644 index 000000000000..99dd4646d743 --- /dev/null +++ b/test/CodeGen/Hexagon/opt-spill-volatile.ll @@ -0,0 +1,29 @@ +; RUN: llc -march=hexagon < %s | FileCheck %s +; Check that the load/store to the volatile stack object has not been +; optimized away. + +target triple = "hexagon" + +; CHECK-LABEL: foo +; CHECK: memw(r29+#4) = +; CHECK: = memw(r29 + #4) +define i32 @foo(i32 %a) #0 { +entry: + %x = alloca i32, align 4 + %x.0.x.0..sroa_cast = bitcast i32* %x to i8* + call void @llvm.lifetime.start(i64 4, i8* %x.0.x.0..sroa_cast) + store volatile i32 0, i32* %x, align 4 + %call = tail call i32 bitcast (i32 (...)* @bar to i32 ()*)() #0 + %x.0.x.0. = load volatile i32, i32* %x, align 4 + %add = add nsw i32 %x.0.x.0., %a + call void @llvm.lifetime.end(i64 4, i8* %x.0.x.0..sroa_cast) + ret i32 %add +} + +declare void @llvm.lifetime.start(i64, i8* nocapture) #1 +declare void @llvm.lifetime.end(i64, i8* nocapture) #1 + +declare i32 @bar(...) #0 + +attributes #0 = { nounwind } +attributes #1 = { argmemonly nounwind } diff --git a/test/CodeGen/Hexagon/packetize-cfi-location.ll b/test/CodeGen/Hexagon/packetize-cfi-location.ll new file mode 100644 index 000000000000..0d80a7bb289d --- /dev/null +++ b/test/CodeGen/Hexagon/packetize-cfi-location.ll @@ -0,0 +1,72 @@ +; RUN: llc -march=hexagon < %s | FileCheck %s + +target triple = "hexagon" +%type.0 = type { i32, i8**, i32, i32, i32 } + +; Check that CFI is before the packet with call+allocframe. +; CHECK-LABEL: danny: +; CHECK: cfi_def_cfa +; CHECK: call throw +; CHECK-NEXT: allocframe + +; Expect packet: +; { +; call throw +; allocframe(#0) +; } + +define i8* @danny(%type.0* %p0, i32 %p1) #0 { +entry: + %t0 = getelementptr inbounds %type.0, %type.0* %p0, i32 0, i32 4 + %t1 = load i32, i32* %t0, align 4 + %th = icmp ugt i32 %t1, %p1 + br i1 %th, label %if.end, label %if.then + +if.then: ; preds = %entry + tail call void @throw(%type.0* nonnull %p0) + unreachable + +if.end: ; preds = %entry + %t6 = getelementptr inbounds %type.0, %type.0* %p0, i32 0, i32 3 + %t2 = load i32, i32* %t6, align 4 + %t9 = add i32 %t2, %p1 + %ta = lshr i32 %t9, 4 + %tb = and i32 %t9, 15 + %t7 = getelementptr inbounds %type.0, %type.0* %p0, i32 0, i32 2 + %t3 = load i32, i32* %t7, align 4 + %tc = icmp ult i32 %ta, %t3 + %td = select i1 %tc, i32 0, i32 %t3 + %te = sub i32 %ta, %td + %t8 = getelementptr inbounds %type.0, %type.0* %p0, i32 0, i32 1 + %t4 = load i8**, i8*** %t8, align 4 + %tf = getelementptr inbounds i8*, i8** %t4, i32 %te + %t5 = load i8*, i8** %tf, align 4 + %tg = getelementptr inbounds i8, i8* %t5, i32 %tb + ret i8* %tg +} + +; Check that CFI is after allocframe. +; CHECK-LABEL: sammy: +; CHECK: allocframe +; CHECK: cfi_def_cfa + +define void @sammy(%type.0* %p0, i32 %p1) #0 { +entry: + %t0 = icmp sgt i32 %p1, 0 + br i1 %t0, label %if.then, label %if.else +if.then: + call void @throw(%type.0* nonnull %p0) + br label %if.end +if.else: + call void @nothrow() #2 + br label %if.end +if.end: + ret void +} + +declare void @throw(%type.0*) #1 +declare void @nothrow() #2 + +attributes #0 = { "target-cpu"="hexagonv55" } +attributes #1 = { noreturn "target-cpu"="hexagonv55" } +attributes #2 = { nounwind "target-cpu"="hexagonv55" } diff --git a/test/CodeGen/Hexagon/packetize-return-arg.ll b/test/CodeGen/Hexagon/packetize-return-arg.ll new file mode 100644 index 000000000000..b18fc23eca81 --- /dev/null +++ b/test/CodeGen/Hexagon/packetize-return-arg.ll @@ -0,0 +1,37 @@ +; RUN: llc -march=hexagon < %s | FileCheck %s +; Check that "r0 = rN" is packetized together with dealloc_return. +; CHECK: r0 = r +; CHECK-NOT: { +; CHECK: dealloc_return + +target triple = "hexagon-unknown--elf" + +; Function Attrs: nounwind +define i8* @fred(i8* %user_context, i32 %x) #0 { +entry: + %and14 = add i32 %x, 255 + %add1 = and i32 %and14, -128 + %call = tail call i8* @malloc(i32 %add1) #1 + %cmp = icmp eq i8* %call, null + br i1 %cmp, label %cleanup, label %if.end + +if.end: ; preds = %entry + %0 = ptrtoint i8* %call to i32 + %sub4 = add i32 %0, 131 + %and5 = and i32 %sub4, -128 + %1 = inttoptr i32 %and5 to i8* + %2 = inttoptr i32 %and5 to i8** + %arrayidx = getelementptr inbounds i8*, i8** %2, i32 -1 + store i8* %call, i8** %arrayidx, align 4 + br label %cleanup + +cleanup: ; preds = %if.end, %entry + %retval.0 = phi i8* [ %1, %if.end ], [ null, %entry ] + ret i8* %retval.0 +} + +; Function Attrs: nounwind +declare noalias i8* @malloc(i32) local_unnamed_addr #1 + +attributes #0 = { nounwind } +attributes #1 = { nobuiltin nounwind } diff --git a/test/CodeGen/Hexagon/peephole-kill-flags.ll b/test/CodeGen/Hexagon/peephole-kill-flags.ll new file mode 100644 index 000000000000..03de15323528 --- /dev/null +++ b/test/CodeGen/Hexagon/peephole-kill-flags.ll @@ -0,0 +1,27 @@ +; RUN: llc -march=hexagon -verify-machineinstrs < %s | FileCheck %s +; CHECK: memw + +; Check that the testcase compiles without errors. + +target triple = "hexagon" + +; Function Attrs: nounwind +define void @fred() #0 { +entry: + br label %for.cond + +for.cond: ; preds = %entry + %0 = load i32, i32* undef, align 4 + %mul = mul nsw i32 2, %0 + %cmp = icmp slt i32 undef, %mul + br i1 %cmp, label %for.body, label %for.end13 + +for.body: ; preds = %for.cond + unreachable + +for.end13: ; preds = %for.cond + ret void +} + +attributes #0 = { nounwind "target-cpu"="hexagonv60" "target-features"="+hvx,-hvx-double" } + diff --git a/test/CodeGen/Hexagon/pic-simple.ll b/test/CodeGen/Hexagon/pic-simple.ll index fa223d5372e1..46d95204f2e7 100644 --- a/test/CodeGen/Hexagon/pic-simple.ll +++ b/test/CodeGen/Hexagon/pic-simple.ll @@ -1,4 +1,4 @@ -; RUN: llc -march=hexagon -mcpu=hexagonv5 -relocation-model=pic < %s | FileCheck %s +; RUN: llc -mtriple=hexagon-- -mcpu=hexagonv5 -relocation-model=pic < %s | FileCheck %s ; CHECK: r{{[0-9]+}} = add({{pc|PC}}, ##_GLOBAL_OFFSET_TABLE_@PCREL) ; CHECK: r{{[0-9]+}} = memw(r{{[0-9]+}}{{.*}}+{{.*}}##src@GOT) diff --git a/test/CodeGen/Hexagon/pic-static.ll b/test/CodeGen/Hexagon/pic-static.ll index f4ccc6b9ee73..66d7734f2cf2 100644 --- a/test/CodeGen/Hexagon/pic-static.ll +++ b/test/CodeGen/Hexagon/pic-static.ll @@ -1,4 +1,4 @@ -; RUN: llc -march=hexagon -mcpu=hexagonv5 -relocation-model=pic < %s | FileCheck %s +; RUN: llc -mtriple=hexagon-- -mcpu=hexagonv5 -relocation-model=pic < %s | FileCheck %s ; CHECK-DAG: r{{[0-9]+}} = add({{pc|PC}}, ##_GLOBAL_OFFSET_TABLE_@PCREL) ; CHECK-DAG: r{{[0-9]+}} = add({{pc|PC}}, ##x@PCREL) diff --git a/test/CodeGen/Hexagon/post-inc-aa-metadata.ll b/test/CodeGen/Hexagon/post-inc-aa-metadata.ll new file mode 100644 index 000000000000..fb2f038e6e59 --- /dev/null +++ b/test/CodeGen/Hexagon/post-inc-aa-metadata.ll @@ -0,0 +1,37 @@ +; RUN: llc -march=hexagon -debug-only=isel < %s 2>&1 | FileCheck %s +; REQUIRES: asserts + +; Check that the generated post-increment load has TBAA information. +; CHECK-LABEL: Machine code for function fred: +; CHECK: = V6_vL32b_pi %vreg{{[0-9]+}}, 64; mem:LD64[{{.*}}](tbaa= + +target triple = "hexagon" + +; Function Attrs: norecurse nounwind +define void @fred(<16 x i32>* nocapture %p, <16 x i32>* nocapture readonly %q, i32 %n) local_unnamed_addr #0 { +entry: + %tobool2 = icmp eq i32 %n, 0 + br i1 %tobool2, label %while.end, label %while.body + +while.body: ; preds = %entry, %while.body + %n.addr.05 = phi i32 [ %dec, %while.body ], [ %n, %entry ] + %q.addr.04 = phi <16 x i32>* [ %incdec.ptr, %while.body ], [ %q, %entry ] + %p.addr.03 = phi <16 x i32>* [ %incdec.ptr1, %while.body ], [ %p, %entry ] + %dec = add i32 %n.addr.05, -1 + %incdec.ptr = getelementptr inbounds <16 x i32>, <16 x i32>* %q.addr.04, i32 1 + %0 = load <16 x i32>, <16 x i32>* %q.addr.04, align 64, !tbaa !1 + %incdec.ptr1 = getelementptr inbounds <16 x i32>, <16 x i32>* %p.addr.03, i32 1 + store <16 x i32> %0, <16 x i32>* %p.addr.03, align 64, !tbaa !1 + %tobool = icmp eq i32 %dec, 0 + br i1 %tobool, label %while.end, label %while.body + +while.end: ; preds = %while.body, %entry + ret void +} + +attributes #0 = { norecurse nounwind "target-cpu"="hexagonv60" "target-features"="+hvx,-hvx-double" } + + +!1 = !{!2, !2, i64 0} +!2 = !{!"omnipotent char", !3, i64 0} +!3 = !{!"Simple C/C++ TBAA"} diff --git a/test/CodeGen/Hexagon/post-ra-kill-update.mir b/test/CodeGen/Hexagon/post-ra-kill-update.mir new file mode 100644 index 000000000000..c43624d7a8d3 --- /dev/null +++ b/test/CodeGen/Hexagon/post-ra-kill-update.mir @@ -0,0 +1,37 @@ +# RUN: llc -march=hexagon -mcpu=hexagonv60 -run-pass post-RA-sched -o - %s | FileCheck %s + +# The post-RA scheduler reorders S2_lsr_r_p and S2_lsr_r_p_or. Both of them +# use r9, and the last of the two kills it. The kill flag fixup did not +# correctly update the flag, resulting in both instructions killing r9. + +# CHECK-LABEL: name: foo +# Check for no-kill of r9 in the first instruction, after reordering: +# CHECK: %d7 = S2_lsr_r_p_or %d7, killed %d1, %r9 +# CHECK: %d13 = S2_lsr_r_p killed %d0, killed %r9 + +--- | + define void @foo() { + ret void + } +... + +--- +name: foo +tracksRegLiveness: true +body: | + bb.0: + successors: %bb.1 + liveins: %d0, %d1, %r9, %r13 + + %d7 = S2_asl_r_p %d0, %r13 + %d5 = S2_asl_r_p %d1, killed %r13 + %d6 = S2_lsr_r_p killed %d0, %r9 + %d7 = S2_lsr_r_p_or killed %d7, killed %d1, killed %r9 + %d1 = A2_combinew killed %r11, killed %r10 + %d0 = A2_combinew killed %r15, killed %r14 + J2_jump %bb.1, implicit-def %pc + + bb.1: + A2_nop +... + diff --git a/test/CodeGen/Hexagon/propagate-vcombine.ll b/test/CodeGen/Hexagon/propagate-vcombine.ll new file mode 100644 index 000000000000..4948a89b73e8 --- /dev/null +++ b/test/CodeGen/Hexagon/propagate-vcombine.ll @@ -0,0 +1,48 @@ +; RUN: llc -march=hexagon < %s | FileCheck %s + +@v0 = global <16 x i32> zeroinitializer, align 64 +@v1 = global <16 x i32> zeroinitializer, align 64 + +; CHECK-LABEL: danny: +; CHECK-NOT: vcombine + +define void @danny() #0 { + %t0 = load <16 x i32>, <16 x i32>* @v0, align 64 + %t1 = load <16 x i32>, <16 x i32>* @v1, align 64 + %t2 = call <32 x i32> @llvm.hexagon.V6.vcombine(<16 x i32> %t0, <16 x i32> %t1) + %t3 = tail call <16 x i32> @llvm.hexagon.V6.lo(<32 x i32> %t2) + %t4 = tail call <16 x i32> @llvm.hexagon.V6.hi(<32 x i32> %t2) + store <16 x i32> %t3, <16 x i32>* @v0, align 64 + store <16 x i32> %t4, <16 x i32>* @v1, align 64 + ret void +} + +@w0 = global <32 x i32> zeroinitializer, align 128 +@w1 = global <32 x i32> zeroinitializer, align 128 + +; CHECK-LABEL: sammy: +; CHECK-NOT: vcombine + +define void @sammy() #1 { + %t0 = load <32 x i32>, <32 x i32>* @w0, align 128 + %t1 = load <32 x i32>, <32 x i32>* @w1, align 128 + %t2 = call <64 x i32> @llvm.hexagon.V6.vcombine.128B(<32 x i32> %t0, <32 x i32> %t1) + %t3 = tail call <32 x i32> @llvm.hexagon.V6.lo.128B(<64 x i32> %t2) + %t4 = tail call <32 x i32> @llvm.hexagon.V6.hi.128B(<64 x i32> %t2) + store <32 x i32> %t3, <32 x i32>* @w0, align 128 + store <32 x i32> %t4, <32 x i32>* @w1, align 128 + ret void +} + +declare <32 x i32> @llvm.hexagon.V6.vcombine(<16 x i32>, <16 x i32>) #2 +declare <16 x i32> @llvm.hexagon.V6.lo(<32 x i32>) #2 +declare <16 x i32> @llvm.hexagon.V6.hi(<32 x i32>) #2 + +declare <64 x i32> @llvm.hexagon.V6.vcombine.128B(<32 x i32>, <32 x i32>) #3 +declare <32 x i32> @llvm.hexagon.V6.lo.128B(<64 x i32>) #3 +declare <32 x i32> @llvm.hexagon.V6.hi.128B(<64 x i32>) #3 + +attributes #0 = { nounwind "target-cpu"="hexagonv60" "target-features"="+hvx" } +attributes #1 = { nounwind "target-cpu"="hexagonv60" "target-features"="+hvx,+hvx-double" } +attributes #2 = { nounwind readnone "target-cpu"="hexagonv60" "target-features"="+hvx" } +attributes #3 = { nounwind readnone "target-cpu"="hexagonv60" "target-features"="+hvx,+hvx-double" } diff --git a/test/CodeGen/Hexagon/rdf-copy.ll b/test/CodeGen/Hexagon/rdf-copy.ll index afb03a6315d7..ce47cf672d79 100644 --- a/test/CodeGen/Hexagon/rdf-copy.ll +++ b/test/CodeGen/Hexagon/rdf-copy.ll @@ -17,7 +17,7 @@ ; CHECK: [[DST:r[0-9]+]] = [[SRC:r[0-9]+]] ; CHECK-DAG: memw([[SRC]] ; CHECK-NOT: memw([[DST]] -; CHECK-LABEL: LBB0_2 +; CHECK: %if.end target datalayout = "e-p:32:32:32-i64:64:64-i32:32:32-i16:16:16-i1:32:32-f64:64:64-f32:32:32-v64:64:64-v32:32:32-a0:0-n16:32" target triple = "hexagon" diff --git a/test/CodeGen/Hexagon/rdf-extra-livein.ll b/test/CodeGen/Hexagon/rdf-extra-livein.ll new file mode 100644 index 000000000000..5d947e6fe45d --- /dev/null +++ b/test/CodeGen/Hexagon/rdf-extra-livein.ll @@ -0,0 +1,73 @@ +; RUN: llc -march=hexagon -verify-machineinstrs < %s | FileCheck %s +; Verify that the code compiles successfully. +; CHECK: call printf + +target triple = "hexagon" + +%struct.0 = type { i32, i32, i32, i32, i32, i32, i32, i32, i32 } + +@.str.13 = external unnamed_addr constant [60 x i8], align 1 + +declare void @printf(i8* nocapture readonly, ...) local_unnamed_addr #0 + +declare void @danny() local_unnamed_addr #0 +declare zeroext i8 @sammy() local_unnamed_addr #0 + +; Function Attrs: nounwind +define void @main() local_unnamed_addr #0 { +entry: + br i1 undef, label %if.then8, label %if.end10 + +if.then8: ; preds = %entry + ret void + +if.end10: ; preds = %entry + br label %do.body + +do.body: ; preds = %if.end88.do.body_crit_edge, %if.end10 + %cond = icmp eq i32 undef, 0 + br i1 %cond, label %if.end49, label %if.then124 + +if.end49: ; preds = %do.body + br i1 undef, label %if.end55, label %if.then53 + +if.then53: ; preds = %if.end49 + call void @danny() + br label %if.end55 + +if.end55: ; preds = %if.then53, %if.end49 + %call76 = call zeroext i8 @sammy() #0 + switch i8 %call76, label %sw.epilog79 [ + i8 0, label %sw.bb77 + i8 3, label %sw.bb77 + ] + +sw.bb77: ; preds = %if.end55, %if.end55 + unreachable + +sw.epilog79: ; preds = %if.end55 + br i1 undef, label %if.end88, label %if.then81 + +if.then81: ; preds = %sw.epilog79 + %div87 = fdiv float 0.000000e+00, undef + br label %if.end88 + +if.end88: ; preds = %if.then81, %sw.epilog79 + %t.1 = phi float [ undef, %sw.epilog79 ], [ %div87, %if.then81 ] + %div89 = fdiv float 1.000000e+00, %t.1 + %mul92 = fmul float undef, %div89 + %div93 = fdiv float %mul92, 1.000000e+06 + %conv107 = fpext float %div93 to double + call void (i8*, ...) @printf(i8* getelementptr inbounds ([60 x i8], [60 x i8]* @.str.13, i32 0, i32 0), double %conv107, double undef, i64 undef, i32 undef) #0 + br i1 undef, label %if.end88.do.body_crit_edge, label %if.then124 + +if.end88.do.body_crit_edge: ; preds = %if.end88 + br label %do.body + +if.then124: ; preds = %if.end88, %do.body + unreachable +} + + +attributes #0 = { nounwind } + diff --git a/test/CodeGen/Hexagon/rdf-filter-defs.ll b/test/CodeGen/Hexagon/rdf-filter-defs.ll new file mode 100644 index 000000000000..735b20e697fd --- /dev/null +++ b/test/CodeGen/Hexagon/rdf-filter-defs.ll @@ -0,0 +1,214 @@ +; RUN: llc -march=hexagon -verify-machineinstrs < %s | FileCheck %s + +; Check that this testcase compiles successfully. +; CHECK: dealloc_return + +target triple = "hexagon" + +%type.0 = type { %type.1, %type.3, i32, i32 } +%type.1 = type { %type.2 } +%type.2 = type { i8 } +%type.3 = type { i8*, [12 x i8] } +%type.4 = type { i8 } + +define weak_odr dereferenceable(28) %type.0* @fred(%type.0* %p0, i32 %p1, %type.0* dereferenceable(28) %p2, i32 %p3, i32 %p4) local_unnamed_addr align 2 { +b0: + %t0 = getelementptr inbounds %type.0, %type.0* %p0, i32 0, i32 2 + %t1 = load i32, i32* %t0, align 4 + %t2 = icmp ult i32 %t1, %p1 + %t3 = getelementptr inbounds %type.0, %type.0* %p2, i32 0, i32 2 + br i1 %t2, label %b2, label %b1 + +b1: + %t4 = load i32, i32* %t3, align 4 + %t5 = icmp ult i32 %t4, %p3 + br i1 %t5, label %b2, label %b3 + +b2: + %t6 = bitcast %type.0* %p0 to %type.4* + tail call void @blah(%type.4* %t6) + %t7 = load i32, i32* %t3, align 4 + %t8 = load i32, i32* %t0, align 4 + br label %b3 + +b3: + %t9 = phi i32 [ %t8, %b2 ], [ %t1, %b1 ] + %t10 = phi i32 [ %t7, %b2 ], [ %t4, %b1 ] + %t11 = sub i32 %t10, %p3 + %t12 = icmp ult i32 %t11, %p4 + %t13 = select i1 %t12, i32 %t11, i32 %p4 + %t14 = xor i32 %t9, -1 + %t15 = icmp ult i32 %t13, %t14 + br i1 %t15, label %b5, label %b4 + +b4: + %t16 = bitcast %type.0* %p0 to %type.4* + tail call void @danny(%type.4* %t16) + br label %b5 + +b5: + %t17 = icmp eq i32 %t13, 0 + br i1 %t17, label %b33, label %b6 + +b6: + %t18 = load i32, i32* %t0, align 4 + %t19 = add i32 %t18, %t13 + %t20 = icmp eq i32 %t19, -1 + br i1 %t20, label %b7, label %b8 + +b7: + %t21 = bitcast %type.0* %p0 to %type.4* + tail call void @danny(%type.4* %t21) + br label %b8 + +b8: + %t22 = getelementptr inbounds %type.0, %type.0* %p0, i32 0, i32 3 + %t23 = load i32, i32* %t22, align 4 + %t24 = icmp ult i32 %t23, %t19 + br i1 %t24, label %b9, label %b10 + +b9: + %t25 = load i32, i32* %t0, align 4 + tail call void @sammy(%type.0* nonnull %p0, i32 %t19, i32 %t25) + %t26 = load i32, i32* %t22, align 4 + br label %b15 + +b10: + %t27 = icmp eq i32 %t19, 0 + br i1 %t27, label %b11, label %b15 + +b11: + %t28 = icmp ugt i32 %t23, 15 + %t29 = getelementptr inbounds %type.0, %type.0* %p0, i32 0, i32 1 + br i1 %t28, label %b12, label %b13 + +b12: + %t30 = getelementptr inbounds %type.3, %type.3* %t29, i32 0, i32 0 + %t31 = load i8*, i8** %t30, align 4 + br label %b14 + +b13: + %t32 = bitcast %type.3* %t29 to i8* + br label %b14 + +b14: + %t33 = phi i8* [ %t31, %b12 ], [ %t32, %b13 ] + store i32 0, i32* %t0, align 4 + br label %b31 + +b15: + %t34 = phi i32 [ %t26, %b9 ], [ %t23, %b10 ] + %t35 = icmp ugt i32 %t34, 15 + %t36 = getelementptr inbounds %type.0, %type.0* %p0, i32 0, i32 1 + br i1 %t35, label %b16, label %b17 + +b16: + %t37 = getelementptr inbounds %type.3, %type.3* %t36, i32 0, i32 0 + %t38 = load i8*, i8** %t37, align 4 + br label %b18 + +b17: + %t39 = bitcast %type.3* %t36 to i8* + %t40 = bitcast %type.3* %t36 to i8* + br label %b18 + +b18: + %t41 = phi i8* [ %t38, %b16 ], [ %t39, %b17 ] + %t42 = phi i8* [ %t38, %b16 ], [ %t40, %b17 ] + %t43 = getelementptr inbounds i8, i8* %t41, i32 %p1 + %t44 = getelementptr inbounds i8, i8* %t43, i32 %t13 + %t45 = getelementptr inbounds i8, i8* %t42, i32 %p1 + %t46 = load i32, i32* %t0, align 4 + %t47 = sub i32 %t46, %p1 + tail call void @llvm.memmove.p0i8.p0i8.i32(i8* %t44, i8* %t45, i32 %t47, i32 1, i1 false) #1 + %t48 = icmp eq %type.0* %p0, %p2 + %t49 = load i32, i32* %t22, align 4 + %t50 = icmp ugt i32 %t49, 15 + br i1 %t50, label %b19, label %b20 + +b19: + %t51 = getelementptr inbounds %type.3, %type.3* %t36, i32 0, i32 0 + %t52 = load i8*, i8** %t51, align 4 + br label %b21 + +b20: + %t53 = bitcast %type.3* %t36 to i8* + br label %b21 + +b21: + %t54 = phi i8* [ %t52, %b19 ], [ %t53, %b20 ] + %t55 = getelementptr inbounds i8, i8* %t54, i32 %p1 + br i1 %t48, label %b22, label %b26 + +b22: + br i1 %t50, label %b23, label %b24 + +b23: + %t56 = getelementptr inbounds %type.3, %type.3* %t36, i32 0, i32 0 + %t57 = load i8*, i8** %t56, align 4 + br label %b25 + +b24: + %t58 = bitcast %type.3* %t36 to i8* + br label %b25 + +b25: + %t59 = phi i8* [ %t57, %b23 ], [ %t58, %b24 ] + %t60 = icmp ult i32 %p1, %p3 + %t61 = select i1 %t60, i32 %t13, i32 0 + %t62 = add i32 %t61, %p3 + %t63 = getelementptr inbounds i8, i8* %t59, i32 %t62 + tail call void @llvm.memmove.p0i8.p0i8.i32(i8* %t55, i8* %t63, i32 %t13, i32 1, i1 false) #1 + br label %b27 + +b26: + %t64 = getelementptr inbounds %type.0, %type.0* %p2, i32 0, i32 3 + %t65 = load i32, i32* %t64, align 4 + %t66 = icmp ugt i32 %t65, 15 + %t67 = getelementptr inbounds %type.0, %type.0* %p2, i32 0, i32 1 + %t68 = getelementptr inbounds %type.3, %type.3* %t67, i32 0, i32 0 + %t69 = load i8*, i8** %t68, align 4 + %t70 = bitcast %type.3* %t67 to i8* + %t71 = select i1 %t66, i8* %t69, i8* %t70 + %t72 = getelementptr inbounds i8, i8* %t71, i32 %p3 + tail call void @llvm.memcpy.p0i8.p0i8.i32(i8* %t55, i8* %t72, i32 %t13, i32 1, i1 false) #1 + br label %b27 + +b27: + %t73 = load i32, i32* %t22, align 4 + %t74 = icmp ugt i32 %t73, 15 + br i1 %t74, label %b28, label %b29 + +b28: + %t75 = getelementptr inbounds %type.3, %type.3* %t36, i32 0, i32 0 + %t76 = load i8*, i8** %t75, align 4 + br label %b30 + +b29: + %t77 = bitcast %type.3* %t36 to i8* + br label %b30 + +b30: + %t78 = phi i8* [ %t76, %b28 ], [ %t77, %b29 ] + store i32 %t19, i32* %t0, align 4 + %t79 = getelementptr inbounds i8, i8* %t78, i32 %t19 + br label %b31 + +b31: + %t80 = phi i8* [ %t33, %b14 ], [ %t79, %b30 ] + store i8 0, i8* %t80, align 1 + br label %b33 + +b33: + ret %type.0* %p0 +} + +declare void @llvm.memcpy.p0i8.p0i8.i32(i8* nocapture writeonly, i8* nocapture readonly, i32, i32, i1) #0 +declare void @llvm.memmove.p0i8.p0i8.i32(i8* nocapture, i8* nocapture readonly, i32, i32, i1) #0 + +declare void @blah(%type.4*) local_unnamed_addr +declare void @danny(%type.4*) local_unnamed_addr +declare void @sammy(%type.0*, i32, i32) local_unnamed_addr align 2 + +attributes #0 = { argmemonly nounwind } +attributes #1 = { nounwind } diff --git a/test/CodeGen/Hexagon/rdf-ignore-undef.ll b/test/CodeGen/Hexagon/rdf-ignore-undef.ll new file mode 100644 index 000000000000..5d72318f420f --- /dev/null +++ b/test/CodeGen/Hexagon/rdf-ignore-undef.ll @@ -0,0 +1,55 @@ +; RUN: llc -march=hexagon -verify-machineinstrs < %s | FileCheck %s +; Check that we don't crash. +; CHECK: call foo + +target triple = "hexagon" + +%struct.1 = type { i16, i8, i32, i8*, i8*, i8*, i8*, i8*, i8*, i32* } +%struct.0 = type { i32, i32, i32, i32, i32, i32, i32, i32, i32 } + +declare void @foo(i8*, %struct.0*) local_unnamed_addr #0 +declare void @bar(%struct.1*, %struct.0* readonly) local_unnamed_addr #0 + +define i32 @fred(i32 %argc, i8** nocapture readonly %argv) local_unnamed_addr #0 { +entry: + br label %do.body + +do.body: ; preds = %if.end88.do.body_crit_edge, %entry + %cond = icmp eq i32 undef, 0 + br i1 %cond, label %if.end49, label %if.then124 + +if.end49: ; preds = %do.body + call void @foo(i8* nonnull undef, %struct.0* nonnull undef) #0 + br i1 undef, label %if.end55, label %if.then53 + +if.then53: ; preds = %if.end49 + call void @bar(%struct.1* null, %struct.0* nonnull undef) + br label %if.end55 + +if.end55: ; preds = %if.then53, %if.end49 + switch i8 undef, label %sw.epilog79 [ + i8 0, label %sw.bb77 + i8 3, label %sw.bb77 + ] + +sw.bb77: ; preds = %if.end55, %if.end55 + br label %sw.epilog79 + +sw.epilog79: ; preds = %sw.bb77, %if.end55 + br i1 undef, label %if.end88, label %if.then81 + +if.then81: ; preds = %sw.epilog79 + br label %if.end88 + +if.end88: ; preds = %if.then81, %sw.epilog79 + store float 0.000000e+00, float* undef, align 4 + br i1 undef, label %if.end88.do.body_crit_edge, label %if.then124 + +if.end88.do.body_crit_edge: ; preds = %if.end88 + br label %do.body + +if.then124: ; preds = %if.end88, %do.body + unreachable +} + +attributes #0 = { nounwind } diff --git a/test/CodeGen/Hexagon/rdf-multiple-phis-up.ll b/test/CodeGen/Hexagon/rdf-multiple-phis-up.ll new file mode 100644 index 000000000000..d23846ac6ed4 --- /dev/null +++ b/test/CodeGen/Hexagon/rdf-multiple-phis-up.ll @@ -0,0 +1,40 @@ +; RUN: llc -march=hexagon < %s | FileCheck %s +; REQUIRES: asserts + +; Check that we do not crash. +; CHECK: call foo + +target triple = "hexagon" + +%struct.0 = type { i8*, i8*, [2 x i8*], i32, i32, i8*, i32, i32, i32, i32, i32, [2 x i32], i32, i32, i32, i32, i32, i32, i32, i32, i32, i32 } + +define i32 @fred(i8* %p0) local_unnamed_addr #0 { +entry: + %0 = bitcast i8* %p0 to %struct.0* + br i1 undef, label %if.then21, label %for.body.i + +if.then21: ; preds = %entry + %.pr = load i32, i32* undef, align 4 + switch i32 %.pr, label %cleanup [ + i32 1, label %for.body.i + i32 3, label %if.then60 + ] + +for.body.i: ; preds = %for.body.i, %if.then21, %entry + %1 = load i8, i8* undef, align 1 + %cmp7.i = icmp ugt i8 %1, -17 + br i1 %cmp7.i, label %cleanup, label %for.body.i + +if.then60: ; preds = %if.then21 + %call61 = call i32 @foo(%struct.0* nonnull %0) #0 + br label %cleanup + +cleanup: ; preds = %if.then60, %for.body.i, %if.then21 + ret i32 undef +} + +declare i32 @foo(%struct.0*) local_unnamed_addr #0 + + +attributes #0 = { nounwind } + diff --git a/test/CodeGen/Hexagon/rdf-phi-shadows.ll b/test/CodeGen/Hexagon/rdf-phi-shadows.ll new file mode 100644 index 000000000000..f26ab9b0cef2 --- /dev/null +++ b/test/CodeGen/Hexagon/rdf-phi-shadows.ll @@ -0,0 +1,64 @@ +; RUN: llc -march=hexagon -verify-machineinstrs < %s | FileCheck %s +; Check that we don't crash. +; CHECK: call printf +target triple = "hexagon" + +%struct.1 = type { i16, i8, i32, i8*, i8*, i8*, i8*, i8*, i8*, i32* } +%struct.0 = type { i32, i32, i32, i32, i32, i32, i32, i32, i32 } + +declare void @foo(%struct.1*, %struct.0* readonly) local_unnamed_addr #0 +declare zeroext i8 @bar() local_unnamed_addr #0 +declare i32 @printf(i8* nocapture readonly, ...) local_unnamed_addr #0 + +@.str = private unnamed_addr constant [5 x i8] c"blah\00", align 1 + +define i32 @main(i32 %argc, i8** nocapture readonly %argv) local_unnamed_addr #0 { +entry: + %t0 = alloca %struct.0, align 4 + br label %do.body + +do.body: ; preds = %if.end88.do.body_crit_edge, %entry + %cond = icmp eq i32 undef, 0 + br i1 %cond, label %if.end49, label %if.then124 + +if.end49: ; preds = %do.body + br i1 undef, label %if.end55, label %if.then53 + +if.then53: ; preds = %if.end49 + call void @foo(%struct.1* null, %struct.0* nonnull %t0) + br label %if.end55 + +if.end55: ; preds = %if.then53, %if.end49 + %call76 = call zeroext i8 @bar() #0 + switch i8 %call76, label %sw.epilog79 [ + i8 0, label %sw.bb77 + i8 3, label %sw.bb77 + ] + +sw.bb77: ; preds = %if.end55, %if.end55 + unreachable + +sw.epilog79: ; preds = %if.end55 + br i1 undef, label %if.end88, label %if.then81 + +if.then81: ; preds = %sw.epilog79 + %div87 = fdiv float 0.000000e+00, undef + br label %if.end88 + +if.end88: ; preds = %if.then81, %sw.epilog79 + %t1 = phi float [ undef, %sw.epilog79 ], [ %div87, %if.then81 ] + %div89 = fdiv float 1.000000e+00, %t1 + %.sroa.speculated = select i1 undef, float 0.000000e+00, float undef + %conv108 = fpext float %.sroa.speculated to double + %call113 = call i32 (i8*, ...) @printf(i8* getelementptr inbounds ([5 x i8], [5 x i8]* @.str, i32 0, i32 0), double undef, double %conv108, i64 undef, i32 undef) #0 + br i1 undef, label %if.end88.do.body_crit_edge, label %if.then124 + +if.end88.do.body_crit_edge: ; preds = %if.end88 + br label %do.body + +if.then124: ; preds = %if.end88, %do.body + %t2 = phi float [ undef, %do.body ], [ %t1, %if.end88 ] + ret i32 0 +} + +attributes #0 = { nounwind } diff --git a/test/CodeGen/Hexagon/rdf-phi-up.ll b/test/CodeGen/Hexagon/rdf-phi-up.ll new file mode 100644 index 000000000000..28f4c90c174d --- /dev/null +++ b/test/CodeGen/Hexagon/rdf-phi-up.ll @@ -0,0 +1,60 @@ +; RUN: llc -march=hexagon -verify-machineinstrs < %s | FileCheck %s +; Check that this testcase compiles successfully. +; CHECK-LABEL: fred: +; CHECK: call foo + +target triple = "hexagon" + +%struct.0 = type { i32, i16, i8* } + +declare void @llvm.lifetime.start(i64, i8* nocapture) #1 +declare void @llvm.lifetime.end(i64, i8* nocapture) #1 + +define i32 @fred(i8* readonly %p0, i32* %p1) local_unnamed_addr #0 { +entry: + %v0 = alloca i16, align 2 + %v1 = icmp eq i8* %p0, null + br i1 %v1, label %if.then, label %lor.lhs.false + +lor.lhs.false: ; preds = %entry + %v2 = bitcast i8* %p0 to %struct.0** + %v3 = load %struct.0*, %struct.0** %v2, align 4 + %v4 = icmp eq %struct.0* %v3, null + br i1 %v4, label %if.then, label %if.else + +if.then: ; preds = %lor.lhs.false, %ent + %v5 = icmp eq i32* %p1, null + br i1 %v5, label %cleanup, label %if.then3 + +if.then3: ; preds = %if.then + store i32 0, i32* %p1, align 4 + br label %cleanup + +if.else: ; preds = %lor.lhs.false + %v6 = bitcast i16* %v0 to i8* + call void @llvm.lifetime.start(i64 2, i8* nonnull %v6) #0 + store i16 0, i16* %v0, align 2 + %v7 = call i32 @foo(%struct.0* nonnull %v3, i16* nonnull %v0) #0 + %v8 = icmp eq i32* %p1, null + br i1 %v8, label %if.end7, label %if.then6 + +if.then6: ; preds = %if.else + %v9 = load i16, i16* %v0, align 2 + %v10 = zext i16 %v9 to i32 + store i32 %v10, i32* %p1, align 4 + br label %if.end7 + +if.end7: ; preds = %if.else, %if.then6 + call void @llvm.lifetime.end(i64 2, i8* nonnull %v6) #0 + br label %cleanup + +cleanup: ; preds = %if.then3, %if.then, + %v11 = phi i32 [ %v7, %if.end7 ], [ -2147024809, %if.then ], [ -2147024809, %if.then3 ] + ret i32 %v11 +} + +declare i32 @foo(%struct.0*, i16*) local_unnamed_addr #0 + +attributes #0 = { nounwind } +attributes #1 = { argmemonly nounwind } + diff --git a/test/CodeGen/Hexagon/regalloc-bad-undef.mir b/test/CodeGen/Hexagon/regalloc-bad-undef.mir new file mode 100644 index 000000000000..d8fbb92b0d50 --- /dev/null +++ b/test/CodeGen/Hexagon/regalloc-bad-undef.mir @@ -0,0 +1,204 @@ +# RUN: llc -march=hexagon -hexagon-subreg-liveness -start-after machine-scheduler -stop-after stack-slot-coloring -o - %s | FileCheck %s + +--- | + target triple = "hexagon" + + ; Function Attrs: nounwind optsize + define void @main() #0 { + entry: + br label %for.body + + for.body: ; preds = %if.end82, %entry + %lsr.iv = phi i32 [ %lsr.iv.next, %if.end82 ], [ 524288, %entry ] + %call9 = tail call i32 @lrand48() #0 + %conv10 = sext i32 %call9 to i64 + %shr11 = lshr i64 %conv10, 9 + %or12 = or i64 0, %shr11 + %call14 = tail call i32 @lrand48() #0 + %conv15138 = zext i32 %call14 to i64 + %shr16 = lshr i64 %conv15138, 9 + %0 = call i64 @llvm.hexagon.S2.extractup(i64 %conv15138, i32 22, i32 9) + %1 = shl i64 %0, 42 + %shl17 = shl i64 %shr16, 42 + %or22 = or i64 0, %1 + %or26 = or i64 %or22, 0 + %shr30 = lshr i64 undef, 25 + %2 = call i64 @llvm.hexagon.S2.extractup(i64 undef, i32 6, i32 25) + %and = and i64 %shr30, 63 + %sub = shl i64 2, %2 + %add = add i64 %sub, -1 + %shl37 = shl i64 %add, 0 + %call38 = tail call i32 @lrand48() #0 + %conv39141 = zext i32 %call38 to i64 + %shr40 = lshr i64 %conv39141, 25 + %3 = call i64 @llvm.hexagon.S2.extractup(i64 %conv39141, i32 6, i32 25) + %and41 = and i64 %shr40, 63 + %sub43 = shl i64 2, %3 + %add45 = add i64 %sub43, -1 + %call46 = tail call i32 @lrand48() #0 + %shr48 = lshr i64 undef, 25 + %4 = call i64 @llvm.hexagon.S2.extractup(i64 undef, i32 6, i32 25) + %and49 = and i64 %shr48, 63 + %shl50 = shl i64 %add45, %4 + %and52 = and i64 %shl37, %or12 + %and54 = and i64 %shl50, %or26 + store i64 %and54, i64* undef, align 8 + %cmp56 = icmp eq i64 %and52, 0 + br i1 %cmp56, label %for.end, label %if.end82 + + if.end82: ; preds = %for.body + %lsr.iv.next = add nsw i32 %lsr.iv, -1 + %exitcond = icmp eq i32 %lsr.iv.next, 0 + br i1 %exitcond, label %for.end, label %for.body + + for.end: ; preds = %if.end82, %for.body + unreachable + } + + declare i32 @lrand48() #0 + declare i64 @llvm.hexagon.S2.extractup(i64, i32, i32) #1 + + attributes #0 = { nounwind optsize "target-cpu"="hexagonv55" "target-features"="-hvx,-hvx-double" } + attributes #1 = { nounwind readnone } + +... +--- +name: main +alignment: 2 +tracksRegLiveness: true +registers: + - { id: 0, class: intregs } + - { id: 1, class: intregs } + - { id: 2, class: intregs } + - { id: 3, class: intregs } + - { id: 4, class: doubleregs } + - { id: 5, class: intregs } + - { id: 6, class: doubleregs } + - { id: 7, class: doubleregs } + - { id: 8, class: doubleregs } + - { id: 9, class: doubleregs } + - { id: 10, class: intregs } + - { id: 11, class: doubleregs } + - { id: 12, class: doubleregs } + - { id: 13, class: doubleregs } + - { id: 14, class: intregs } + - { id: 15, class: doubleregs } + - { id: 16, class: doubleregs } + - { id: 17, class: intregs } + - { id: 18, class: doubleregs } + - { id: 19, class: intregs } + - { id: 20, class: doubleregs } + - { id: 21, class: doubleregs } + - { id: 22, class: doubleregs } + - { id: 23, class: intregs } + - { id: 24, class: doubleregs } + - { id: 25, class: predregs } + - { id: 26, class: predregs } + - { id: 27, class: intregs } + - { id: 28, class: intregs } + - { id: 29, class: doubleregs } + - { id: 30, class: intregs } + - { id: 31, class: intregs } + - { id: 32, class: doubleregs } + - { id: 33, class: intregs } + - { id: 34, class: intregs } + - { id: 35, class: doubleregs } + - { id: 36, class: doubleregs } + - { id: 37, class: intregs } + - { id: 38, class: intregs } + - { id: 39, class: doubleregs } + - { id: 40, class: doubleregs } + - { id: 41, class: intregs } + - { id: 42, class: intregs } + - { id: 43, class: doubleregs } + - { id: 44, class: intregs } + - { id: 45, class: intregs } + - { id: 46, class: doubleregs } + - { id: 47, class: doubleregs } + - { id: 48, class: doubleregs } + - { id: 49, class: doubleregs } + - { id: 50, class: doubleregs } + - { id: 51, class: doubleregs } + - { id: 52, class: intregs } + - { id: 53, class: intregs } + - { id: 54, class: intregs } + - { id: 55, class: doubleregs } + - { id: 56, class: doubleregs } + - { id: 57, class: intregs } + - { id: 58, class: intregs } + - { id: 59, class: intregs } +frameInfo: + isFrameAddressTaken: false + isReturnAddressTaken: false + hasStackMap: false + hasPatchPoint: false + stackSize: 0 + offsetAdjustment: 0 + maxAlignment: 0 + adjustsStack: false + hasCalls: true + maxCallFrameSize: 0 + hasOpaqueSPAdjustment: false + hasVAStart: false + hasMustTailInVarArgFunc: false +body: | + bb.0.entry: + successors: %bb.1.for.body + + %59 = A2_tfrsi 524288 + undef %32.isub_hi = A2_tfrsi 0 + %8 = S2_extractup undef %9, 6, 25 + %47 = A2_tfrpi 2 + %13 = A2_tfrpi -1 + %13 = S2_asl_r_p_acc %13, %47, %8.isub_lo + %51 = A2_tfrpi 0 + + ; CHECK: %d2 = S2_extractup undef %d0, 6, 25 + ; CHECK: %d0 = A2_tfrpi 2 + ; CHECK: %d13 = A2_tfrpi -1 + ; CHECK-NOT: undef %r4 + + bb.1.for.body: + successors: %bb.3.for.end, %bb.2.if.end82 + + ADJCALLSTACKDOWN 0, implicit-def dead %r29, implicit-def dead %r30, implicit %r31, implicit %r30, implicit %r29 + J2_call @lrand48, implicit-def dead %d0, implicit-def dead %d1, implicit-def dead %d2, implicit-def dead %d3, implicit-def dead %d4, implicit-def dead %d5, implicit-def dead %d6, implicit-def dead %d7, implicit-def dead %r28, implicit-def dead %r31, implicit-def dead %p0, implicit-def dead %p1, implicit-def dead %p2, implicit-def dead %p3, implicit-def dead %m0, implicit-def dead %m1, implicit-def dead %lc0, implicit-def dead %lc1, implicit-def dead %sa0, implicit-def dead %sa1, implicit-def dead %usr, implicit-def %usr_ovf, implicit-def dead %cs0, implicit-def dead %cs1, implicit-def dead %w0, implicit-def dead %w1, implicit-def dead %w2, implicit-def dead %w3, implicit-def dead %w4, implicit-def dead %w5, implicit-def dead %w6, implicit-def dead %w7, implicit-def dead %w8, implicit-def dead %w9, implicit-def dead %w10, implicit-def dead %w11, implicit-def dead %w12, implicit-def dead %w13, implicit-def dead %w14, implicit-def dead %w15, implicit-def dead %q0, implicit-def dead %q1, implicit-def dead %q2, implicit-def dead %q3, implicit-def %r0 + ADJCALLSTACKUP 0, 0, implicit-def dead %r29, implicit-def dead %r30, implicit-def dead %r31, implicit %r29 + undef %29.isub_lo = COPY killed %r0 + %29.isub_hi = S2_asr_i_r %29.isub_lo, 31 + ADJCALLSTACKDOWN 0, implicit-def dead %r29, implicit-def dead %r30, implicit %r31, implicit %r30, implicit %r29 + J2_call @lrand48, implicit-def dead %d0, implicit-def dead %d1, implicit-def dead %d2, implicit-def dead %d3, implicit-def dead %d4, implicit-def dead %d5, implicit-def dead %d6, implicit-def dead %d7, implicit-def dead %r28, implicit-def dead %r31, implicit-def dead %p0, implicit-def dead %p1, implicit-def dead %p2, implicit-def dead %p3, implicit-def dead %m0, implicit-def dead %m1, implicit-def dead %lc0, implicit-def dead %lc1, implicit-def dead %sa0, implicit-def dead %sa1, implicit-def dead %usr, implicit-def %usr_ovf, implicit-def dead %cs0, implicit-def dead %cs1, implicit-def dead %w0, implicit-def dead %w1, implicit-def dead %w2, implicit-def dead %w3, implicit-def dead %w4, implicit-def dead %w5, implicit-def dead %w6, implicit-def dead %w7, implicit-def dead %w8, implicit-def dead %w9, implicit-def dead %w10, implicit-def dead %w11, implicit-def dead %w12, implicit-def dead %w13, implicit-def dead %w14, implicit-def dead %w15, implicit-def dead %q0, implicit-def dead %q1, implicit-def dead %q2, implicit-def dead %q3, implicit-def %r0 + ADJCALLSTACKUP 0, 0, implicit-def dead %r29, implicit-def dead %r30, implicit-def dead %r31, implicit %r29 + %32.isub_lo = COPY killed %r0 + %7 = S2_extractup %32, 22, 9 + ADJCALLSTACKDOWN 0, implicit-def dead %r29, implicit-def dead %r30, implicit %r31, implicit %r30, implicit %r29 + J2_call @lrand48, implicit-def dead %d0, implicit-def dead %d1, implicit-def dead %d2, implicit-def dead %d3, implicit-def dead %d4, implicit-def dead %d5, implicit-def dead %d6, implicit-def dead %d7, implicit-def dead %r28, implicit-def dead %r31, implicit-def dead %p0, implicit-def dead %p1, implicit-def dead %p2, implicit-def dead %p3, implicit-def dead %m0, implicit-def dead %m1, implicit-def dead %lc0, implicit-def dead %lc1, implicit-def dead %sa0, implicit-def dead %sa1, implicit-def dead %usr, implicit-def %usr_ovf, implicit-def dead %cs0, implicit-def dead %cs1, implicit-def dead %w0, implicit-def dead %w1, implicit-def dead %w2, implicit-def dead %w3, implicit-def dead %w4, implicit-def dead %w5, implicit-def dead %w6, implicit-def dead %w7, implicit-def dead %w8, implicit-def dead %w9, implicit-def dead %w10, implicit-def dead %w11, implicit-def dead %w12, implicit-def dead %w13, implicit-def dead %w14, implicit-def dead %w15, implicit-def dead %q0, implicit-def dead %q1, implicit-def dead %q2, implicit-def dead %q3, implicit-def %r0 + ADJCALLSTACKUP 0, 0, implicit-def dead %r29, implicit-def dead %r30, implicit-def dead %r31, implicit %r29 + undef %43.isub_lo = COPY killed %r0 + %43.isub_hi = COPY %32.isub_hi + %16 = S2_extractup %43, 6, 25 + %18 = A2_tfrpi -1 + %18 = S2_asl_r_p_acc %18, %47, %16.isub_lo + ADJCALLSTACKDOWN 0, implicit-def dead %r29, implicit-def dead %r30, implicit %r31, implicit %r30, implicit %r29 + J2_call @lrand48, implicit-def dead %d0, implicit-def dead %d1, implicit-def dead %d2, implicit-def dead %d3, implicit-def dead %d4, implicit-def dead %d5, implicit-def dead %d6, implicit-def dead %d7, implicit-def dead %r28, implicit-def dead %r31, implicit-def dead %p0, implicit-def dead %p1, implicit-def dead %p2, implicit-def dead %p3, implicit-def dead %m0, implicit-def dead %m1, implicit-def dead %lc0, implicit-def dead %lc1, implicit-def dead %sa0, implicit-def dead %sa1, implicit-def dead %usr, implicit-def %usr_ovf, implicit-def dead %cs0, implicit-def dead %cs1, implicit-def dead %w0, implicit-def dead %w1, implicit-def dead %w2, implicit-def dead %w3, implicit-def dead %w4, implicit-def dead %w5, implicit-def dead %w6, implicit-def dead %w7, implicit-def dead %w8, implicit-def dead %w9, implicit-def dead %w10, implicit-def dead %w11, implicit-def dead %w12, implicit-def dead %w13, implicit-def dead %w14, implicit-def dead %w15, implicit-def dead %q0, implicit-def dead %q1, implicit-def dead %q2, implicit-def dead %q3 + ADJCALLSTACKUP 0, 0, implicit-def dead %r29, implicit-def dead %r30, implicit-def dead %r31, implicit %r29 + %22 = S2_asl_r_p %18, %8.isub_lo + %21 = COPY %13 + %21 = S2_lsr_i_p_and %21, %29, 9 + %22 = S2_asl_i_p_and %22, %7, 42 + S2_storerd_io undef %23, 0, %22 :: (store 8 into `i64* undef`) + %25 = C2_cmpeqp %21, %51 + J2_jumpt %25, %bb.3.for.end, implicit-def dead %pc + J2_jump %bb.2.if.end82, implicit-def dead %pc + + bb.2.if.end82: + successors: %bb.3.for.end, %bb.1.for.body + + %59 = A2_addi %59, -1 + %26 = C2_cmpeqi %59, 0 + J2_jumpf %26, %bb.1.for.body, implicit-def dead %pc + J2_jump %bb.3.for.end, implicit-def dead %pc + + bb.3.for.end: + +... diff --git a/test/CodeGen/Hexagon/sf-min-max.ll b/test/CodeGen/Hexagon/sf-min-max.ll new file mode 100644 index 000000000000..e795cb42d6a2 --- /dev/null +++ b/test/CodeGen/Hexagon/sf-min-max.ll @@ -0,0 +1,67 @@ +; RUN: llc -march=hexagon < %s | FileCheck %s + +; CHECK-LABEL: sf_min_olt: +; CHECK: sfmin +define float @sf_min_olt(float %x, float %y) #0 { + %t = fcmp olt float %x, %y + %u = select i1 %t, float %x, float %y + ret float %u +} + +; CHECK-LABEL: sf_min_ole: +; CHECK: sfmin +define float @sf_min_ole(float %x, float %y) #0 { + %t = fcmp ole float %x, %y + %u = select i1 %t, float %x, float %y + ret float %u +} + +; CHECK-LABEL: sf_max_ogt: +; CHECK: sfmax +define float @sf_max_ogt(float %x, float %y) #0 { + %t = fcmp ogt float %x, %y + %u = select i1 %t, float %x, float %y + ret float %u +} + +; CHECK-LABEL: sf_max_oge: +; CHECK: sfmax +define float @sf_max_oge(float %x, float %y) #0 { + %t = fcmp oge float %x, %y + %u = select i1 %t, float %x, float %y + ret float %u +} + +; CHECK-LABEL: sf_max_olt: +; CHECK: sfmax +define float @sf_max_olt(float %x, float %y) #0 { + %t = fcmp olt float %x, %y + %u = select i1 %t, float %y, float %x + ret float %u +} + +; CHECK-LABEL: sf_max_ole: +; CHECK: sfmax +define float @sf_max_ole(float %x, float %y) #0 { + %t = fcmp ole float %x, %y + %u = select i1 %t, float %y, float %x + ret float %u +} + +; CHECK-LABEL: sf_min_ogt: +; CHECK: sfmin +define float @sf_min_ogt(float %x, float %y) #0 { + %t = fcmp ogt float %x, %y + %u = select i1 %t, float %y, float %x + ret float %u +} + +; CHECK-LABEL: sf_min_oge: +; CHECK: sfmin +define float @sf_min_oge(float %x, float %y) #0 { + %t = fcmp oge float %x, %y + %u = select i1 %t, float %y, float %x + ret float %u +} + +attributes #0 = { nounwind "target-cpu"="hexagonv5" } diff --git a/test/CodeGen/Hexagon/sffms.ll b/test/CodeGen/Hexagon/sffms.ll new file mode 100644 index 000000000000..ef47976ab3bb --- /dev/null +++ b/test/CodeGen/Hexagon/sffms.ll @@ -0,0 +1,25 @@ +; RUN: llc -march=hexagon -fp-contract=fast < %s | FileCheck %s + +; Check that "Rx-=sfmpy(Rs,Rt)" is being generated for "fsub(fmul(..))" + +; CHECK: r{{[0-9]+}} -= sfmpy + +%struct.matrix_params = type { float** } + +; Function Attrs: norecurse nounwind +define void @loop2_1(%struct.matrix_params* nocapture readonly %params, i32 %col1) #0 { +entry: + %matrixA = getelementptr inbounds %struct.matrix_params, %struct.matrix_params* %params, i32 0, i32 0 + %0 = load float**, float*** %matrixA, align 4 + %1 = load float*, float** %0, align 4 + %arrayidx1 = getelementptr inbounds float, float* %1, i32 %col1 + %2 = load float, float* %arrayidx1, align 4 + %arrayidx3 = getelementptr inbounds float*, float** %0, i32 %col1 + %3 = load float*, float** %arrayidx3, align 4 + %4 = load float, float* %3, align 4 + %mul = fmul float %2, %4 + %sub = fsub float %2, %mul + %arrayidx10 = getelementptr inbounds float, float* %3, i32 %col1 + store float %sub, float* %arrayidx10, align 4 + ret void +} diff --git a/test/CodeGen/Hexagon/split-const32-const64.ll b/test/CodeGen/Hexagon/split-const32-const64.ll index 2815253545c5..95741462e508 100644 --- a/test/CodeGen/Hexagon/split-const32-const64.ll +++ b/test/CodeGen/Hexagon/split-const32-const64.ll @@ -1,24 +1,28 @@ -; RUN: llc -march=hexagon -mcpu=hexagonv5 -hexagon-small-data-threshold=0 < %s | FileCheck %s +; RUN: llc -march=hexagon -hexagon-small-data-threshold=0 < %s | FileCheck %s -; Check that CONST32/CONST64 instructions are 'not' generated when -; small-data-threshold is set to 0. +; Check that CONST32/CONST64 instructions are 'not' generated when the +; small data threshold is set to 0. -; with immediate value. @a = external global i32 @b = external global i32 @la = external global i64 @lb = external global i64 -define void @test1() nounwind { +; CHECK-LABEL: test1: ; CHECK-NOT: CONST32 +define void @test1() nounwind { entry: + br label %block +block: store i32 12345670, i32* @a, align 4 - store i32 12345670, i32* @b, align 4 + %q = ptrtoint i8* blockaddress (@test1, %block) to i32 + store i32 %q, i32* @b, align 4 ret void } -define void @test2() nounwind { +; CHECK-LABEL: test2: ; CHECK-NOT: CONST64 +define void @test2() nounwind { entry: store i64 1234567890123, i64* @la, align 8 store i64 1234567890123, i64* @lb, align 8 diff --git a/test/CodeGen/Hexagon/storerd-io-over-rr.ll b/test/CodeGen/Hexagon/storerd-io-over-rr.ll new file mode 100644 index 000000000000..8727330ca5bd --- /dev/null +++ b/test/CodeGen/Hexagon/storerd-io-over-rr.ll @@ -0,0 +1,12 @@ +; RUN: llc -march=hexagon < %s | FileCheck %s +; Check for memd(base + #offset), instead of memd(base + reg<<#c). +; CHECK: memd(r{{[0-9]+}}+# + +define void @fred(i32 %p, i64 %v) #0 { + %t0 = add i32 %p, 4 + %t1 = inttoptr i32 %t0 to i64* + store i64 %v, i64* %t1 + ret void +} + +attributes #0 = { nounwind "target-cpu"="hexagonv60" } diff --git a/test/CodeGen/Hexagon/struct_args.ll b/test/CodeGen/Hexagon/struct_args.ll index 2ac1f8eadbb7..11c23b82ec4a 100644 --- a/test/CodeGen/Hexagon/struct_args.ll +++ b/test/CodeGen/Hexagon/struct_args.ll @@ -1,6 +1,6 @@ -; RUN: llc -march=hexagon -mcpu=hexagonv4 -disable-hsdr < %s | FileCheck %s -; CHECK: r{{[0-9]}}:{{[0-9]}} = combine({{r[0-9]|#0}}, r{{[0-9]}}) -; CHECK: r{{[0-9]}}:{{[0-9]}} |= asl(r{{[0-9]}}:{{[0-9]}}, #32) +; RUN: llc -march=hexagon -disable-hsdr < %s | FileCheck %s +; CHECK-DAG: r0 = memw +; CHECK-DAG: r1 = memw %struct.small = type { i32, i32 } @@ -8,7 +8,7 @@ define void @foo() nounwind { entry: - %0 = load i64, i64* bitcast (%struct.small* @s1 to i64*), align 1 + %0 = load i64, i64* bitcast (%struct.small* @s1 to i64*), align 4 call void @bar(i64 %0) ret void } diff --git a/test/CodeGen/Hexagon/subi-asl.ll b/test/CodeGen/Hexagon/subi-asl.ll new file mode 100644 index 000000000000..f0b27e828f50 --- /dev/null +++ b/test/CodeGen/Hexagon/subi-asl.ll @@ -0,0 +1,70 @@ +; RUN: llc -march=hexagon < %s | FileCheck %s + +; Check if S4_subi_asl_ri is being generated correctly. + +; CHECK-LABEL: yes_sub_asl +; CHECK: [[REG1:(r[0-9]+)]] = sub(#0, asl([[REG1]], #1)) + +; CHECK-LABEL: no_sub_asl +; CHECK: [[REG2:(r[0-9]+)]] = asl(r{{[0-9]+}}, #1) +; CHECK: r{{[0-9]+}} = sub([[REG2]], r{{[0-9]+}}) + +%struct.rtx_def = type { i16, i8 } + +@this_insn_number = external global i32, align 4 + +; Function Attrs: nounwind +define void @yes_sub_asl(%struct.rtx_def* %reg, %struct.rtx_def* nocapture readonly %setter) #0 { +entry: + %code = getelementptr inbounds %struct.rtx_def, %struct.rtx_def* %reg, i32 0, i32 0 + %0 = load i16, i16* %code, align 4 + switch i16 %0, label %return [ + i16 2, label %if.end + i16 5, label %if.end + ] + +if.end: + %code6 = getelementptr inbounds %struct.rtx_def, %struct.rtx_def* %setter, i32 0, i32 0 + %1 = load i16, i16* %code6, align 4 + %cmp8 = icmp eq i16 %1, 56 + %conv9 = zext i1 %cmp8 to i32 + %2 = load i32, i32* @this_insn_number, align 4 + %3 = mul i32 %2, -2 + %sub = add nsw i32 %conv9, %3 + tail call void @reg_is_born(%struct.rtx_def* nonnull %reg, i32 %sub) #2 + br label %return + +return: + ret void +} + +declare void @reg_is_born(%struct.rtx_def*, i32) #1 + +; Function Attrs: nounwind +define void @no_sub_asl(%struct.rtx_def* %reg, %struct.rtx_def* nocapture readonly %setter) #0 { +entry: + %code = getelementptr inbounds %struct.rtx_def, %struct.rtx_def* %reg, i32 0, i32 0 + %0 = load i16, i16* %code, align 4 + switch i16 %0, label %return [ + i16 2, label %if.end + i16 5, label %if.end + ] + +if.end: + %1 = load i32, i32* @this_insn_number, align 4 + %mul = mul nsw i32 %1, 2 + %code6 = getelementptr inbounds %struct.rtx_def, %struct.rtx_def* %setter, i32 0, i32 0 + %2 = load i16, i16* %code6, align 4 + %cmp8 = icmp eq i16 %2, 56 + %conv9 = zext i1 %cmp8 to i32 + %sub = sub nsw i32 %mul, %conv9 + tail call void @reg_is_born(%struct.rtx_def* nonnull %reg, i32 %sub) #2 + br label %return + +return: + ret void +} + +attributes #0 = { nounwind "target-cpu"="hexagonv5" } +attributes #1 = { "target-cpu"="hexagonv5" } +attributes #2 = { nounwind } diff --git a/test/CodeGen/Hexagon/swp-const-tc.ll b/test/CodeGen/Hexagon/swp-const-tc.ll new file mode 100644 index 000000000000..3113094d2ba3 --- /dev/null +++ b/test/CodeGen/Hexagon/swp-const-tc.ll @@ -0,0 +1,51 @@ +; RUN: llc -march=hexagon -mcpu=hexagonv5 -enable-pipeliner -verify-machineinstrs < %s | FileCheck %s + +; If the trip count is a compile-time constant, then decrement it instead +; of computing a new LC0 value. + +; CHECK-LABEL: @test +; CHECK: loop0(.LBB0_1, #998) + +define i32 @test(i32* %A, i32* %B, i32 %count) { +entry: + br label %for.body + +for.body: + %sum.02 = phi i32 [ 0, %entry ], [ %add, %for.body ] + %arrayidx.phi = phi i32* [ %A, %entry ], [ %arrayidx.inc, %for.body ] + %i.01 = phi i32 [ 0, %entry ], [ %inc, %for.body ] + %0 = load i32, i32* %arrayidx.phi, align 4 + %add = add nsw i32 %0, %sum.02 + %inc = add nsw i32 %i.01, 1 + %exitcond = icmp eq i32 %inc, 1000 + %arrayidx.inc = getelementptr i32, i32* %arrayidx.phi, i32 1 + br i1 %exitcond, label %for.end, label %for.body + +for.end: + ret i32 %add +} + +; The constant trip count is small enough that the kernel is not executed. + +; CHECK-LABEL: @test1 +; CHECK-NOT: loop0( + +define i32 @test1(i32* %A, i32* %B, i32 %count) { +entry: + br label %for.body + +for.body: + %sum.02 = phi i32 [ 0, %entry ], [ %add, %for.body ] + %arrayidx.phi = phi i32* [ %A, %entry ], [ %arrayidx.inc, %for.body ] + %i.01 = phi i32 [ 0, %entry ], [ %inc, %for.body ] + %0 = load i32, i32* %arrayidx.phi, align 4 + %add = add nsw i32 %0, %sum.02 + %inc = add nsw i32 %i.01, 1 + %exitcond = icmp eq i32 %inc, 1 + %arrayidx.inc = getelementptr i32, i32* %arrayidx.phi, i32 1 + br i1 %exitcond, label %for.end, label %for.body + +for.end: + ret i32 %add +} + diff --git a/test/CodeGen/Hexagon/swp-dag-phi.ll b/test/CodeGen/Hexagon/swp-dag-phi.ll new file mode 100644 index 000000000000..54d9492ebac6 --- /dev/null +++ b/test/CodeGen/Hexagon/swp-dag-phi.ll @@ -0,0 +1,42 @@ +; RUN: llc -march=hexagon -mcpu=hexagonv5 -enable-pipeliner -pipeliner-max-stages=2 < %s +; REQUIRES: asserts + +; This tests check that a dependence is created between a Phi and it's uses. +; An assert occurs if the Phi dependences are not correct. + +define void @test1(i32* %f2, i32 %nc) { +entry: + %i.011 = add i32 %nc, -1 + %cmp12 = icmp sgt i32 %i.011, 1 + br i1 %cmp12, label %for.body.preheader, label %for.end + +for.body.preheader: + %0 = add i32 %nc, -2 + %scevgep = getelementptr i32, i32* %f2, i32 %0 + %sri = load i32, i32* %scevgep, align 4 + %scevgep15 = getelementptr i32, i32* %f2, i32 %i.011 + %sri16 = load i32, i32* %scevgep15, align 4 + br label %for.body + +for.body: + %i.014 = phi i32 [ %i.0, %for.body ], [ %i.011, %for.body.preheader ] + %i.0.in13 = phi i32 [ %i.014, %for.body ], [ %nc, %for.body.preheader ] + %sr = phi i32 [ %1, %for.body ], [ %sri, %for.body.preheader ] + %sr17 = phi i32 [ %sr, %for.body ], [ %sri16, %for.body.preheader ] + %arrayidx = getelementptr inbounds i32, i32* %f2, i32 %i.014 + %sub1 = add nsw i32 %i.0.in13, -3 + %arrayidx2 = getelementptr inbounds i32, i32* %f2, i32 %sub1 + %1 = load i32, i32* %arrayidx2, align 4 + %sub3 = sub nsw i32 %sr17, %1 + store i32 %sub3, i32* %arrayidx, align 4 + %i.0 = add nsw i32 %i.014, -1 + %cmp = icmp sgt i32 %i.0, 1 + br i1 %cmp, label %for.body, label %for.end.loopexit + +for.end.loopexit: + br label %for.end + +for.end: + ret void +} + diff --git a/test/CodeGen/Hexagon/swp-epilog-phi10.ll b/test/CodeGen/Hexagon/swp-epilog-phi10.ll new file mode 100644 index 000000000000..ff35e0b30f35 --- /dev/null +++ b/test/CodeGen/Hexagon/swp-epilog-phi10.ll @@ -0,0 +1,88 @@ +; RUN: llc -march=hexagon -mcpu=hexagonv5 < %s +; REQUIRES: asserts + +define void @test(i8* noalias nocapture readonly %src, i32 %srcStride) local_unnamed_addr #0 { +entry: + %add.ptr = getelementptr inbounds i8, i8* %src, i32 %srcStride + %add.ptr2 = getelementptr inbounds i8, i8* %add.ptr, i32 %srcStride + %add.ptr3 = getelementptr inbounds i8, i8* %add.ptr2, i32 %srcStride + br label %for.body9.epil + +for.body9.epil: + %inc.sink385.epil = phi i32 [ %add17.epil, %for.body9.epil ], [ 2, %entry ] + %sr.epil = phi i8 [ %0, %for.body9.epil ], [ undef, %entry ] + %sr431.epil = phi i8 [ %2, %for.body9.epil ], [ 0, %entry ] + %sr432.epil = phi i8 [ %sr431.epil, %for.body9.epil ], [ 0, %entry ] + %epil.iter = phi i32 [ %epil.iter.sub, %for.body9.epil ], [ undef, %entry ] + %sub11.epil = add i32 %inc.sink385.epil, -1 + %add17.epil = add nuw i32 %inc.sink385.epil, 1 + %conv19.epil = zext i8 %sr.epil to i32 + %add21.epil = add i32 %inc.sink385.epil, 2 + %arrayidx22.epil = getelementptr inbounds i8, i8* %src, i32 %add21.epil + %0 = load i8, i8* %arrayidx22.epil, align 1 + %conv23.epil = zext i8 %0 to i32 + %1 = load i8, i8* undef, align 1 + %conv42.epil = zext i8 %1 to i32 + %conv53.epil = zext i8 %sr432.epil to i32 + %2 = load i8, i8* undef, align 1 + %conv61.epil = zext i8 %2 to i32 + %3 = load i8, i8* undef, align 1 + %conv65.epil = zext i8 %3 to i32 + %4 = load i8, i8* null, align 1 + %conv69.epil = zext i8 %4 to i32 + %5 = load i8, i8* undef, align 1 + %conv72.epil = zext i8 %5 to i32 + %6 = load i8, i8* undef, align 1 + %conv76.epil = zext i8 %6 to i32 + %7 = load i8, i8* undef, align 1 + %conv80.epil = zext i8 %7 to i32 + %8 = load i8, i8* undef, align 1 + %conv84.epil = zext i8 %8 to i32 + %9 = load i8, i8* undef, align 1 + %conv88.epil = zext i8 %9 to i32 + %10 = load i8, i8* undef, align 1 + %conv91.epil = zext i8 %10 to i32 + %11 = load i8, i8* undef, align 1 + %conv95.epil = zext i8 %11 to i32 + %12 = load i8, i8* undef, align 1 + %conv99.epil = zext i8 %12 to i32 + %add.epil = add nuw nsw i32 0, %conv19.epil + %add16.epil = add nuw nsw i32 %add.epil, 0 + %add20.epil = add nuw nsw i32 %add16.epil, 0 + %add24.epil = add nuw nsw i32 %add20.epil, 0 + %add28.epil = add nuw nsw i32 %add24.epil, 0 + %add32.epil = add nuw nsw i32 %add28.epil, 0 + %add35.epil = add i32 %add32.epil, 0 + %add39.epil = add i32 %add35.epil, 0 + %add43.epil = add i32 %add39.epil, %conv53.epil + %add47.epil = add i32 %add43.epil, 0 + %add51.epil = add i32 %add47.epil, 0 + %add54.epil = add i32 %add51.epil, %conv23.epil + %add58.epil = add i32 %add54.epil, %conv42.epil + %add62.epil = add i32 %add58.epil, %conv61.epil + %add66.epil = add i32 %add62.epil, %conv65.epil + %add70.epil = add i32 %add66.epil, %conv69.epil + %add73.epil = add i32 %add70.epil, %conv72.epil + %add77.epil = add i32 %add73.epil, %conv76.epil + %add81.epil = add i32 %add77.epil, %conv80.epil + %add85.epil = add i32 %add81.epil, %conv84.epil + %add89.epil = add i32 %add85.epil, %conv88.epil + %add92.epil = add i32 %add89.epil, %conv91.epil + %add96.epil = add i32 %add92.epil, %conv95.epil + %add100.epil = add i32 %add96.epil, %conv99.epil + %mul.epil = mul nsw i32 %add100.epil, 2621 + %add101.epil = add nsw i32 %mul.epil, 32768 + %shr369.epil = lshr i32 %add101.epil, 16 + %conv102.epil = trunc i32 %shr369.epil to i8 + %arrayidx103.epil = getelementptr inbounds i8, i8* undef, i32 %inc.sink385.epil + store i8 %conv102.epil, i8* %arrayidx103.epil, align 1 + %epil.iter.sub = add i32 %epil.iter, -1 + %epil.iter.cmp = icmp eq i32 %epil.iter.sub, 0 + br i1 %epil.iter.cmp, label %for.end, label %for.body9.epil + +for.end: + unreachable +} + +attributes #0 = { norecurse nounwind "correctly-rounded-divide-sqrt-fp-math"="false" "disable-tail-calls"="false" "less-precise-fpmad"="false" "no-frame-pointer-elim"="true" "no-frame-pointer-elim-non-leaf" "no-infs-fp-math"="false" "no-jump-tables"="false" "no-nans-fp-math"="false" "no-signed-zeros-fp-math"="false" "stack-protector-buffer-size"="8" "target-cpu"="hexagonv5" "unsafe-fp-math"="false" "use-soft-float"="false" } + diff --git a/test/CodeGen/Hexagon/swp-epilog-reuse-1.ll b/test/CodeGen/Hexagon/swp-epilog-reuse-1.ll new file mode 100644 index 000000000000..3b5dbe698cec --- /dev/null +++ b/test/CodeGen/Hexagon/swp-epilog-reuse-1.ll @@ -0,0 +1,44 @@ +; RUN: llc -march=hexagon -mcpu=hexagonv60 < %s +; REQUIRES: asserts + +; Test that the pipeliner reuses an existing Phi when generating the epilog +; block. In this case, the original loops has a Phi whose operand is another +; Phi. When the loop is pipelined, the Phi that generates the operand value +; is used in two stages. This means the the Phi for the second stage can +; be reused. The bug causes an assert due to an invalid virtual register error +; in the live variable analysis. + +define void @test(i8* %a, i8* %b) #0 { +entry: + br label %for.body6.us.prol + +for.body6.us.prol: + %i.065.us.prol = phi i32 [ 0, %entry ], [ %inc.us.prol, %for.body6.us.prol ] + %im1.064.us.prol = phi i32 [ undef, %entry ], [ %i.065.us.prol, %for.body6.us.prol ] + %prol.iter = phi i32 [ undef, %entry ], [ %prol.iter.sub, %for.body6.us.prol ] + %arrayidx8.us.prol = getelementptr inbounds i8, i8* %b, i32 %im1.064.us.prol + %0 = load i8, i8* %arrayidx8.us.prol, align 1 + %conv9.us.prol = sext i8 %0 to i32 + %add.us.prol = add nsw i32 %conv9.us.prol, 0 + %add12.us.prol = add nsw i32 %add.us.prol, 0 + %mul.us.prol = mul nsw i32 %add12.us.prol, 3 + %conv13.us.prol = trunc i32 %mul.us.prol to i8 + %arrayidx14.us.prol = getelementptr inbounds i8, i8* %a, i32 %i.065.us.prol + store i8 %conv13.us.prol, i8* %arrayidx14.us.prol, align 1 + %inc.us.prol = add nuw nsw i32 %i.065.us.prol, 1 + %prol.iter.sub = add i32 %prol.iter, -1 + %prol.iter.cmp = icmp eq i32 %prol.iter.sub, 0 + br i1 %prol.iter.cmp, label %for.body6.us, label %for.body6.us.prol + +for.body6.us: + %im2.063.us = phi i32 [ undef, %for.body6.us ], [ %im1.064.us.prol, %for.body6.us.prol ] + %arrayidx10.us = getelementptr inbounds i8, i8* %b, i32 %im2.063.us + %1 = load i8, i8* %arrayidx10.us, align 1 + %conv11.us = sext i8 %1 to i32 + %add12.us = add nsw i32 0, %conv11.us + %mul.us = mul nsw i32 %add12.us, 3 + %conv13.us = trunc i32 %mul.us to i8 + store i8 %conv13.us, i8* undef, align 1 + br label %for.body6.us +} + diff --git a/test/CodeGen/Hexagon/swp-epilog-reuse.ll b/test/CodeGen/Hexagon/swp-epilog-reuse.ll new file mode 100644 index 000000000000..6a2ad73f2092 --- /dev/null +++ b/test/CodeGen/Hexagon/swp-epilog-reuse.ll @@ -0,0 +1,65 @@ +; RUN: llc -fp-contract=fast -O3 -march=hexagon -mcpu=hexagonv5 < %s +; REQUIRES: asserts + +; Test that the pipeliner doesn't ICE due because the PHI generation +; code in the epilog does not attempt to reuse an existing PHI. + +define void @test(float* noalias %srcImg, i32 %width, float* noalias %dstImg) { +entry.split: + %shr = lshr i32 %width, 1 + %incdec.ptr253 = getelementptr inbounds float, float* %dstImg, i32 2 + br i1 undef, label %for.body, label %for.end + +for.body: + %dst.21518.reg2mem.0 = phi float* [ null, %while.end712 ], [ %incdec.ptr253, %entry.split ] + %dstEnd.01519 = phi float* [ %add.ptr725, %while.end712 ], [ undef, %entry.split ] + %add.ptr367 = getelementptr inbounds float, float* %srcImg, i32 undef + %dst.31487 = getelementptr inbounds float, float* %dst.21518.reg2mem.0, i32 1 + br i1 undef, label %while.body661.preheader, label %while.end712 + +while.body661.preheader: + %scevgep1941 = getelementptr float, float* %add.ptr367, i32 1 + br label %while.body661.ur + +while.body661.ur: + %lsr.iv1942 = phi float* [ %scevgep1941, %while.body661.preheader ], [ undef, %while.body661.ur ] + %col1.31508.reg2mem.0.ur = phi float [ %col3.31506.reg2mem.0.ur, %while.body661.ur ], [ undef, %while.body661.preheader ] + %col4.31507.reg2mem.0.ur = phi float [ %add710.ur, %while.body661.ur ], [ 0.000000e+00, %while.body661.preheader ] + %col3.31506.reg2mem.0.ur = phi float [ %add689.ur, %while.body661.ur ], [ undef, %while.body661.preheader ] + %dst.41511.ur = phi float* [ %incdec.ptr674.ur, %while.body661.ur ], [ %dst.31487, %while.body661.preheader ] + %mul662.ur = fmul float %col1.31508.reg2mem.0.ur, 4.000000e+00 + %add663.ur = fadd float undef, %mul662.ur + %add665.ur = fadd float %add663.ur, undef + %add667.ur = fadd float undef, %add665.ur + %add669.ur = fadd float undef, %add667.ur + %add670.ur = fadd float %col4.31507.reg2mem.0.ur, %add669.ur + %conv673.ur = fmul float %add670.ur, 3.906250e-03 + %incdec.ptr674.ur = getelementptr inbounds float, float* %dst.41511.ur, i32 1 + store float %conv673.ur, float* %dst.41511.ur, align 4 + %scevgep1959 = getelementptr float, float* %lsr.iv1942, i32 -1 + %0 = load float, float* %scevgep1959, align 4 + %mul680.ur = fmul float %0, 4.000000e+00 + %add681.ur = fadd float undef, %mul680.ur + %add684.ur = fadd float undef, %add681.ur + %add687.ur = fadd float undef, %add684.ur + %add689.ur = fadd float undef, %add687.ur + %add699.ur = fadd float undef, undef + %add703.ur = fadd float undef, %add699.ur + %add707.ur = fadd float undef, %add703.ur + %add710.ur = fadd float undef, %add707.ur + %cmp660.ur = icmp ult float* %incdec.ptr674.ur, %dstEnd.01519 + br i1 %cmp660.ur, label %while.body661.ur, label %while.end712 + +while.end712: + %dst.4.lcssa.reg2mem.0 = phi float* [ %dst.31487, %for.body ], [ undef, %while.body661.ur ] + %conv721 = fpext float undef to double + %mul722 = fmul double %conv721, 0x3F7111112119E8FB + %conv723 = fptrunc double %mul722 to float + store float %conv723, float* %dst.4.lcssa.reg2mem.0, align 4 + %add.ptr725 = getelementptr inbounds float, float* %dstEnd.01519, i32 %shr + %cmp259 = icmp ult i32 undef, undef + br i1 %cmp259, label %for.body, label %for.end + +for.end: + ret void +} diff --git a/test/CodeGen/Hexagon/swp-matmul-bitext.ll b/test/CodeGen/Hexagon/swp-matmul-bitext.ll new file mode 100644 index 000000000000..db5bb96d0bc9 --- /dev/null +++ b/test/CodeGen/Hexagon/swp-matmul-bitext.ll @@ -0,0 +1,75 @@ +; RUN: llc -march=hexagon -mcpu=hexagonv60 -enable-bsb-sched=0 -enable-pipeliner < %s | FileCheck %s +; RUN: llc -march=hexagon -mcpu=hexagonv5 -enable-pipeliner < %s | FileCheck %s + +; From coremark. Test that we pipeline the matrix multiplication bitextract +; function. The pipelined code should have two packets. + +; CHECK: loop0(.LBB0_[[LOOP:.]], +; CHECK: .LBB0_[[LOOP]]: +; CHECK: = extractu([[REG2:(r[0-9]+)]], +; CHECK: = extractu([[REG2]], +; CHECK: [[REG0:(r[0-9]+)]] = memh +; CHECK: [[REG1:(r[0-9]+)]] = memh +; CHECK: += mpyi +; CHECK: [[REG2]] = mpyi([[REG0]], [[REG1]]) +; CHECK: endloop0 + +%union_h2_sem_t = type { i32 } + +@sem_i = common global [0 x %union_h2_sem_t] zeroinitializer, align 4 + +define void @matrix_mul_matrix_bitextract(i32 %N, i32* %C, i16* %A, i16* %B) { +entry: + %cmp53 = icmp eq i32 %N, 0 + br i1 %cmp53, label %for_end27, label %for_body3_lr_ph_us + +for_body3_lr_ph_us: + %i_054_us = phi i32 [ %inc26_us, %for_cond1_for_inc25_crit_edge_us ], [ 0, %entry ] + %0 = mul i32 %i_054_us, %N + %arrayidx9_us_us_gep = getelementptr i16, i16* %A, i32 %0 + br label %for_body3_us_us + +for_cond1_for_inc25_crit_edge_us: + %inc26_us = add i32 %i_054_us, 1 + %exitcond89 = icmp eq i32 %inc26_us, %N + br i1 %exitcond89, label %for_end27, label %for_body3_lr_ph_us + +for_body3_us_us: + %j_052_us_us = phi i32 [ %inc23_us_us, %for_cond4_for_inc22_crit_edge_us_us ], [ 0, %for_body3_lr_ph_us ] + %add_us_us = add i32 %j_052_us_us, %0 + %arrayidx_us_us = getelementptr inbounds i32, i32* %C, i32 %add_us_us + store i32 0, i32* %arrayidx_us_us, align 4 + br label %for_body6_us_us + +for_cond4_for_inc22_crit_edge_us_us: + store i32 %add21_us_us, i32* %arrayidx_us_us, align 4 + %inc23_us_us = add i32 %j_052_us_us, 1 + %exitcond88 = icmp eq i32 %inc23_us_us, %N + br i1 %exitcond88, label %for_cond1_for_inc25_crit_edge_us, label %for_body3_us_us + +for_body6_us_us: + %1 = phi i32 [ 0, %for_body3_us_us ], [ %add21_us_us, %for_body6_us_us ] + %arrayidx9_us_us_phi = phi i16* [ %arrayidx9_us_us_gep, %for_body3_us_us ], [ %arrayidx9_us_us_inc, %for_body6_us_us ] + %k_050_us_us = phi i32 [ 0, %for_body3_us_us ], [ %inc_us_us, %for_body6_us_us ] + %2 = load i16, i16* %arrayidx9_us_us_phi, align 2 + %conv_us_us = sext i16 %2 to i32 + %mul10_us_us = mul i32 %k_050_us_us, %N + %add11_us_us = add i32 %mul10_us_us, %j_052_us_us + %arrayidx12_us_us = getelementptr inbounds i16, i16* %B, i32 %add11_us_us + %3 = load i16, i16* %arrayidx12_us_us, align 2 + %conv13_us_us = sext i16 %3 to i32 + %mul14_us_us = mul nsw i32 %conv13_us_us, %conv_us_us + %shr47_us_us = lshr i32 %mul14_us_us, 2 + %and_us_us = and i32 %shr47_us_us, 15 + %shr1548_us_us = lshr i32 %mul14_us_us, 5 + %and16_us_us = and i32 %shr1548_us_us, 127 + %mul17_us_us = mul i32 %and_us_us, %and16_us_us + %add21_us_us = add i32 %mul17_us_us, %1 + %inc_us_us = add i32 %k_050_us_us, 1 + %exitcond87 = icmp eq i32 %inc_us_us, %N + %arrayidx9_us_us_inc = getelementptr i16, i16* %arrayidx9_us_us_phi, i32 1 + br i1 %exitcond87, label %for_cond4_for_inc22_crit_edge_us_us, label %for_body6_us_us + +for_end27: + ret void +} diff --git a/test/CodeGen/Hexagon/swp-max.ll b/test/CodeGen/Hexagon/swp-max.ll new file mode 100644 index 000000000000..038138ff2561 --- /dev/null +++ b/test/CodeGen/Hexagon/swp-max.ll @@ -0,0 +1,42 @@ +; RUN: llc -march=hexagon -mcpu=hexagonv5 -enable-pipeliner \ +; RUN: -pipeliner-max-stages=2 < %s | FileCheck %s + +@A = global [8 x i32] [i32 4, i32 -3, i32 5, i32 -2, i32 -1, i32 2, i32 6, i32 -2], align 8 + +define i32 @test(i32 %Left, i32 %Right) { +entry: + %add = add nsw i32 %Right, %Left + %div = sdiv i32 %add, 2 + %cmp9 = icmp slt i32 %div, %Left + br i1 %cmp9, label %for.end, label %for.body.preheader + +for.body.preheader: + br label %for.body + +; CHECK: loop0(.LBB0_[[LOOP:.]], +; CHECK: .LBB0_[[LOOP]]: +; CHECK: [[REG1:(r[0-9]+)]] = max(r{{[0-9]+}}, [[REG1]]) +; CHECK: [[REG0:(r[0-9]+)]] = add([[REG2:(r[0-9]+)]], [[REG0]]) +; CHECK: [[REG2]] = memw +; CHECK: endloop0 + +for.body: + %MaxLeftBorderSum.012 = phi i32 [ %MaxLeftBorderSum.1, %for.body ], [ 0, %for.body.preheader ] + %i.011 = phi i32 [ %dec, %for.body ], [ %div, %for.body.preheader ] + %LeftBorderSum.010 = phi i32 [ %add1, %for.body ], [ 0, %for.body.preheader ] + %arrayidx = getelementptr inbounds [8 x i32], [8 x i32]* @A, i32 0, i32 %i.011 + %0 = load i32, i32* %arrayidx, align 4 + %add1 = add nsw i32 %0, %LeftBorderSum.010 + %cmp2 = icmp sgt i32 %add1, %MaxLeftBorderSum.012 + %MaxLeftBorderSum.1 = select i1 %cmp2, i32 %add1, i32 %MaxLeftBorderSum.012 + %dec = add nsw i32 %i.011, -1 + %cmp = icmp slt i32 %dec, %Left + br i1 %cmp, label %for.end.loopexit, label %for.body + +for.end.loopexit: + br label %for.end + +for.end: + %MaxLeftBorderSum.0.lcssa = phi i32 [ 0, %entry ], [ %MaxLeftBorderSum.1, %for.end.loopexit ] + ret i32 %MaxLeftBorderSum.0.lcssa +} diff --git a/test/CodeGen/Hexagon/swp-multi-loops.ll b/test/CodeGen/Hexagon/swp-multi-loops.ll new file mode 100644 index 000000000000..56e8c6511000 --- /dev/null +++ b/test/CodeGen/Hexagon/swp-multi-loops.ll @@ -0,0 +1,75 @@ +; RUN: llc -march=hexagon -mcpu=hexagonv5 -enable-pipeliner < %s | FileCheck %s + +; Make sure we attempt to pipeline all inner most loops. + +; Check if the first loop is pipelined. +; CHECK: loop0(.LBB0_[[LOOP:.]], +; CHECK: .LBB0_[[LOOP]]: +; CHECK: add(r{{[0-9]+}}, r{{[0-9]+}}) +; CHECK-NEXT: memw(r{{[0-9]+}}{{.*}}++{{.*}}#4) +; CHECK-NEXT: endloop0 + +; Check if the second loop is pipelined. +; CHECK: loop0(.LBB0_[[LOOP:.]], +; CHECK: .LBB0_[[LOOP]]: +; CHECK: add(r{{[0-9]+}}, r{{[0-9]+}}) +; CHECK-NEXT: memw(r{{[0-9]+}}{{.*}}++{{.*}}#4) +; CHECK-NEXT: endloop0 + +define i32 @test(i32* %a, i32 %n, i32 %l) { +entry: + %cmp23 = icmp sgt i32 %n, 0 + br i1 %cmp23, label %for.body3.lr.ph.preheader, label %for.end14 + +for.body3.lr.ph.preheader: + br label %for.body3.lr.ph + +for.body3.lr.ph: + %sum1.026 = phi i32 [ %add8, %for.inc12 ], [ 0, %for.body3.lr.ph.preheader ] + %sum.025 = phi i32 [ %add, %for.inc12 ], [ 0, %for.body3.lr.ph.preheader ] + %j.024 = phi i32 [ %inc13, %for.inc12 ], [ 0, %for.body3.lr.ph.preheader ] + br label %for.body3 + +for.body3: + %sum.118 = phi i32 [ %sum.025, %for.body3.lr.ph ], [ %add, %for.body3 ] + %arrayidx.phi = phi i32* [ %a, %for.body3.lr.ph ], [ %arrayidx.inc, %for.body3 ] + %i.017 = phi i32 [ 0, %for.body3.lr.ph ], [ %inc, %for.body3 ] + %0 = load i32, i32* %arrayidx.phi, align 4 + %add = add nsw i32 %0, %sum.118 + %inc = add nsw i32 %i.017, 1 + %exitcond = icmp eq i32 %inc, %n + %arrayidx.inc = getelementptr i32, i32* %arrayidx.phi, i32 1 + br i1 %exitcond, label %for.end, label %for.body3 + +for.end: + tail call void @bar(i32* %a) #2 + br label %for.body6 + +for.body6: + %sum1.121 = phi i32 [ %sum1.026, %for.end ], [ %add8, %for.body6 ] + %arrayidx7.phi = phi i32* [ %a, %for.end ], [ %arrayidx7.inc, %for.body6 ] + %i.120 = phi i32 [ 0, %for.end ], [ %inc10, %for.body6 ] + %1 = load i32, i32* %arrayidx7.phi, align 4 + %add8 = add nsw i32 %1, %sum1.121 + %inc10 = add nsw i32 %i.120, 1 + %exitcond29 = icmp eq i32 %inc10, %n + %arrayidx7.inc = getelementptr i32, i32* %arrayidx7.phi, i32 1 + br i1 %exitcond29, label %for.inc12, label %for.body6 + +for.inc12: + %inc13 = add nsw i32 %j.024, 1 + %exitcond30 = icmp eq i32 %inc13, %n + br i1 %exitcond30, label %for.end14.loopexit, label %for.body3.lr.ph + +for.end14.loopexit: + br label %for.end14 + +for.end14: + %sum1.0.lcssa = phi i32 [ 0, %entry ], [ %add8, %for.end14.loopexit ] + %sum.0.lcssa = phi i32 [ 0, %entry ], [ %add, %for.end14.loopexit ] + %add15 = add nsw i32 %sum1.0.lcssa, %sum.0.lcssa + ret i32 %add15 +} + +declare void @bar(i32*) + diff --git a/test/CodeGen/Hexagon/swp-prolog-phi4.ll b/test/CodeGen/Hexagon/swp-prolog-phi4.ll new file mode 100644 index 000000000000..5ed0514ef74c --- /dev/null +++ b/test/CodeGen/Hexagon/swp-prolog-phi4.ll @@ -0,0 +1,65 @@ +; RUN: llc -march=hexagon -mcpu=hexagonv5 -verify-machineinstrs < %s + +; Test that the name rewriter code doesn't chase the Phi operands for +; Phis that do not occur in the loop that is being pipelined. + +define void @test(i32 %srcStride) local_unnamed_addr #0 { +entry: + br label %for.body + +for.body: + %add.ptr3.pn = phi i8* [ undef, %entry ], [ %src4.0394, %for.end ] + %src2.0390 = phi i8* [ undef, %entry ], [ %add.ptr3.pn, %for.end ] + %src4.0394 = getelementptr inbounds i8, i8* %add.ptr3.pn, i32 %srcStride + %sri414 = load i8, i8* undef, align 1 + br i1 undef, label %for.body9.epil, label %for.body9.preheader.new + +for.body9.preheader.new: + br label %for.body9.epil + +for.body9.epil: + %inc.sink385.epil = phi i32 [ %add17.epil, %for.body9.epil ], [ 2, %for.body ], [ undef, %for.body9.preheader.new ] + %sr420.epil = phi i8 [ undef, %for.body9.epil ], [ %sri414, %for.body ], [ undef, %for.body9.preheader.new ] + %sr421.epil = phi i8 [ %sr420.epil, %for.body9.epil ], [ undef, %for.body ], [ undef, %for.body9.preheader.new ] + %sr422.epil = phi i8 [ %sr421.epil, %for.body9.epil ], [ 0, %for.body ], [ undef, %for.body9.preheader.new ] + %epil.iter = phi i32 [ %epil.iter.sub, %for.body9.epil ], [ undef, %for.body9.preheader.new ], [ undef, %for.body ] + %add17.epil = add nuw i32 %inc.sink385.epil, 1 + %add21.epil = add i32 %inc.sink385.epil, 2 + %arrayidx22.epil = getelementptr inbounds i8, i8* undef, i32 %add21.epil + %conv27.epil = zext i8 %sr422.epil to i32 + %0 = load i8, i8* null, align 1 + %conv61.epil = zext i8 %0 to i32 + %arrayidx94.epil = getelementptr inbounds i8, i8* %src4.0394, i32 %add17.epil + %1 = load i8, i8* %arrayidx94.epil, align 1 + %add35.epil = add i32 0, %conv27.epil + %add39.epil = add i32 %add35.epil, 0 + %add43.epil = add i32 %add39.epil, 0 + %add47.epil = add i32 %add43.epil, 0 + %add51.epil = add i32 %add47.epil, 0 + %add54.epil = add i32 %add51.epil, 0 + %add58.epil = add i32 %add54.epil, 0 + %add62.epil = add i32 %add58.epil, %conv61.epil + %add66.epil = add i32 %add62.epil, 0 + %add70.epil = add i32 %add66.epil, 0 + %add73.epil = add i32 %add70.epil, 0 + %add77.epil = add i32 %add73.epil, 0 + %add81.epil = add i32 %add77.epil, 0 + %add85.epil = add i32 %add81.epil, 0 + %add89.epil = add i32 %add85.epil, 0 + %add92.epil = add i32 %add89.epil, 0 + %add96.epil = add i32 %add92.epil, 0 + %add100.epil = add i32 %add96.epil, 0 + %mul.epil = mul nsw i32 %add100.epil, 2621 + %add101.epil = add nsw i32 %mul.epil, 32768 + %shr369.epil = lshr i32 %add101.epil, 16 + %conv102.epil = trunc i32 %shr369.epil to i8 + store i8 %conv102.epil, i8* undef, align 1 + %epil.iter.sub = add i32 %epil.iter, -1 + %epil.iter.cmp = icmp eq i32 %epil.iter.sub, 0 + br i1 %epil.iter.cmp, label %for.end, label %for.body9.epil + +for.end: + br label %for.body +} + +attributes #0 = { norecurse nounwind "correctly-rounded-divide-sqrt-fp-math"="false" "disable-tail-calls"="false" "less-precise-fpmad"="false" "no-frame-pointer-elim"="true" "no-frame-pointer-elim-non-leaf" "no-infs-fp-math"="false" "no-jump-tables"="false" "no-nans-fp-math"="false" "no-signed-zeros-fp-math"="false" "stack-protector-buffer-size"="8" "target-cpu"="hexagonv5" "unsafe-fp-math"="false" "use-soft-float"="false" } diff --git a/test/CodeGen/Hexagon/swp-vect-dotprod.ll b/test/CodeGen/Hexagon/swp-vect-dotprod.ll new file mode 100644 index 000000000000..3ff88452499e --- /dev/null +++ b/test/CodeGen/Hexagon/swp-vect-dotprod.ll @@ -0,0 +1,41 @@ +; RUN: llc -march=hexagon -mcpu=hexagonv5 -enable-pipeliner < %s | FileCheck %s +; RUN: llc -march=hexagon -mcpu=hexagonv5 -O2 < %s | FileCheck %s +; RUN: llc -march=hexagon -mcpu=hexagonv5 -O3 < %s | FileCheck %s +; +; Check that we pipeline a vectorized dot product in a single packet. +; +; CHECK: { +; CHECK: += mpyi +; CHECK: += mpyi +; CHECK: memd +; CHECK: memd +; CHECK: } :endloop0 + +@a = common global [5000 x i32] zeroinitializer, align 8 +@b = common global [5000 x i32] zeroinitializer, align 8 + +define i32 @vecMultGlobal() { +entry: + br label %polly.loop_body + +polly.loop_after: + %0 = extractelement <2 x i32> %addp_vec, i32 0 + %1 = extractelement <2 x i32> %addp_vec, i32 1 + %add_sum = add i32 %0, %1 + ret i32 %add_sum + +polly.loop_body: + %polly.loopiv13 = phi i32 [ 0, %entry ], [ %polly.next_loopiv, %polly.loop_body ] + %reduction.012 = phi <2 x i32> [ zeroinitializer, %entry ], [ %addp_vec, %polly.loop_body ] + %polly.next_loopiv = add nsw i32 %polly.loopiv13, 2 + %p_arrayidx1 = getelementptr [5000 x i32], [5000 x i32]* @b, i32 0, i32 %polly.loopiv13 + %p_arrayidx = getelementptr [5000 x i32], [5000 x i32]* @a, i32 0, i32 %polly.loopiv13 + %vector_ptr = bitcast i32* %p_arrayidx1 to <2 x i32>* + %_p_vec_full = load <2 x i32>, <2 x i32>* %vector_ptr, align 8 + %vector_ptr7 = bitcast i32* %p_arrayidx to <2 x i32>* + %_p_vec_full8 = load <2 x i32>, <2 x i32>* %vector_ptr7, align 8 + %mulp_vec = mul <2 x i32> %_p_vec_full8, %_p_vec_full + %addp_vec = add <2 x i32> %mulp_vec, %reduction.012 + %2 = icmp slt i32 %polly.next_loopiv, 5000 + br i1 %2, label %polly.loop_body, label %polly.loop_after +} diff --git a/test/CodeGen/Hexagon/swp-vmult.ll b/test/CodeGen/Hexagon/swp-vmult.ll new file mode 100644 index 000000000000..9018405274cd --- /dev/null +++ b/test/CodeGen/Hexagon/swp-vmult.ll @@ -0,0 +1,33 @@ +; RUN: llc -march=hexagon -mcpu=hexagonv5 -enable-pipeliner < %s | FileCheck %s +; RUN: llc -march=hexagon -mcpu=hexagonv5 -O3 < %s | FileCheck %s + +; Multiply and accumulate +; CHECK: mpyi([[REG0:r([0-9]+)]], [[REG1:r([0-9]+)]]) +; CHECK-NEXT: add(r{{[0-9]+}}, #4) +; CHECK-NEXT: [[REG0]] = memw(r{{[0-9]+}} + r{{[0-9]+}}<<#0) +; CHECK-NEXT: [[REG1]] = memw(r{{[0-9]+}} + r{{[0-9]+}}<<#0) +; CHECK-NEXT: endloop0 + +define i32 @foo(i32* %a, i32* %b, i32 %n) { +entry: + br label %for.body + +for.body: + %sum.03 = phi i32 [ 0, %entry ], [ %add, %for.body ] + %arrayidx.phi = phi i32* [ %a, %entry ], [ %arrayidx.inc, %for.body ] + %arrayidx1.phi = phi i32* [ %b, %entry ], [ %arrayidx1.inc, %for.body ] + %i.02 = phi i32 [ 0, %entry ], [ %inc, %for.body ] + %0 = load i32, i32* %arrayidx.phi, align 4 + %1 = load i32, i32* %arrayidx1.phi, align 4 + %mul = mul nsw i32 %1, %0 + %add = add nsw i32 %mul, %sum.03 + %inc = add nsw i32 %i.02, 1 + %exitcond = icmp eq i32 %inc, 10000 + %arrayidx.inc = getelementptr i32, i32* %arrayidx.phi, i32 1 + %arrayidx1.inc = getelementptr i32, i32* %arrayidx1.phi, i32 1 + br i1 %exitcond, label %for.end, label %for.body + +for.end: + ret i32 %add +} + diff --git a/test/CodeGen/Hexagon/swp-vsum.ll b/test/CodeGen/Hexagon/swp-vsum.ll new file mode 100644 index 000000000000..4756c644709f --- /dev/null +++ b/test/CodeGen/Hexagon/swp-vsum.ll @@ -0,0 +1,29 @@ +; RUN: llc -march=hexagon -mcpu=hexagonv5 -enable-pipeliner < %s | FileCheck %s +; RUN: llc -march=hexagon -mcpu=hexagonv5 -O3 < %s | FileCheck %s + +; Simple vector total. +; CHECK: loop0(.LBB0_[[LOOP:.]], +; CHECK: .LBB0_[[LOOP]]: +; CHECK: add([[REG:r([0-9]+)]], r{{[0-9]+}}) +; CHECK-NEXT: add(r{{[0-9]+}}, #4) +; CHECK-NEXT: [[REG]] = memw(r{{[0-9]+}} + r{{[0-9]+}}<<#0) +; CHECK-NEXT: endloop0 + +define i32 @foo(i32* %a, i32 %n) { +entry: + br label %for.body + +for.body: + %sum.02 = phi i32 [ 0, %entry ], [ %add, %for.body ] + %arrayidx.phi = phi i32* [ %a, %entry ], [ %arrayidx.inc, %for.body ] + %i.01 = phi i32 [ 0, %entry ], [ %inc, %for.body ] + %0 = load i32, i32* %arrayidx.phi, align 4 + %add = add nsw i32 %0, %sum.02 + %inc = add nsw i32 %i.01, 1 + %exitcond = icmp eq i32 %inc, 10000 + %arrayidx.inc = getelementptr i32, i32* %arrayidx.phi, i32 1 + br i1 %exitcond, label %for.end, label %for.body + +for.end: + ret i32 %add +} diff --git a/test/CodeGen/Hexagon/tailcall_fastcc_ccc.ll b/test/CodeGen/Hexagon/tailcall_fastcc_ccc.ll new file mode 100644 index 000000000000..479fc15a8e8a --- /dev/null +++ b/test/CodeGen/Hexagon/tailcall_fastcc_ccc.ll @@ -0,0 +1,22 @@ +; RUN: llc -march=hexagon < %s | FileCheck %s + +target triple = "hexagon" + +declare hidden fastcc void @callee(i32, i32) #0 +declare hidden void @callee2(i32, i32) #0 + +; CHECK: jump callee +define void @caller(i32 %pp) #0 { +entry: + tail call fastcc void @callee(i32 %pp, i32 0) + ret void +} + +; CHECK: jump callee2 +define void @caller2(i32 %pp) #0 { +entry: + tail call fastcc void @callee2(i32 %pp, i32 0) + ret void +} + +attributes #0 = { nounwind } diff --git a/test/CodeGen/Hexagon/tls_static.ll b/test/CodeGen/Hexagon/tls_static.ll index ad2ca716b70d..dbd3bd7b4ba8 100644 --- a/test/CodeGen/Hexagon/tls_static.ll +++ b/test/CodeGen/Hexagon/tls_static.ll @@ -1,4 +1,4 @@ -; RUN: llc -O0 -march=hexagon -relocation-model=static < %s | FileCheck %s +; RUN: llc -O0 -mtriple=hexagon-- -relocation-model=static < %s | FileCheck %s @dst_le = thread_local global i32 0, align 4 @src_le = thread_local global i32 0, align 4 diff --git a/test/CodeGen/Hexagon/two-crash.ll b/test/CodeGen/Hexagon/two-crash.ll new file mode 100644 index 000000000000..0ab02cda8a07 --- /dev/null +++ b/test/CodeGen/Hexagon/two-crash.ll @@ -0,0 +1,23 @@ +; RUN: llc -march=hexagon < %s | FileCheck %s +; This testcase crashed, because we propagated a reg:sub into a tied use. +; The two-address pass rewrote it in a way that generated incorrect code. +; CHECK: r{{[0-9]+}} += lsr(r{{[0-9]+}}, #16) + +target triple = "hexagon" + +define i64 @fred(i64 %x) local_unnamed_addr #0 { +entry: + %t.sroa.0.0.extract.trunc = trunc i64 %x to i32 + %t4.sroa.4.0.extract.shift = lshr i64 %x, 16 + %add11 = add i32 0, %t.sroa.0.0.extract.trunc + %t14.sroa.3.0.extract.trunc = trunc i64 %t4.sroa.4.0.extract.shift to i32 + %t14.sroa.4.0.extract.shift = lshr i64 %x, 24 + %add21 = add i32 %add11, %t14.sroa.3.0.extract.trunc + %t24.sroa.3.0.extract.trunc = trunc i64 %t14.sroa.4.0.extract.shift to i32 + %add31 = add i32 %add21, %t24.sroa.3.0.extract.trunc + %conv32.mask = and i32 %add31, 255 + %conv33 = zext i32 %conv32.mask to i64 + ret i64 %conv33 +} + +attributes #0 = { norecurse nounwind readnone } diff --git a/test/CodeGen/Hexagon/v60-cur.ll b/test/CodeGen/Hexagon/v60-cur.ll index fe24309f5b87..a7d4f6d310e4 100644 --- a/test/CodeGen/Hexagon/v60-cur.ll +++ b/test/CodeGen/Hexagon/v60-cur.ll @@ -1,4 +1,4 @@ -; RUN: llc -march=hexagon < %s | FileCheck %s +; RUN: llc -march=hexagon -enable-pipeliner=false < %s | FileCheck %s ; Test that we generate a .cur diff --git a/test/CodeGen/Hexagon/v60-vsel1.ll b/test/CodeGen/Hexagon/v60-vsel1.ll new file mode 100644 index 000000000000..e673145c9d14 --- /dev/null +++ b/test/CodeGen/Hexagon/v60-vsel1.ll @@ -0,0 +1,69 @@ +; RUN: llc -march=hexagon < %s | FileCheck %s + +; CHECK: if (p{{[0-3]}}) v{{[0-9]+}} = v{{[0-9]+}} + +target triple = "hexagon" + +; Function Attrs: nounwind +define void @fast9_detect_coarse(i8* nocapture readnone %img, i32 %xsize, i32 %stride, i32 %barrier, i32* nocapture %bitmask, i32 %boundary) #0 { +entry: + %0 = bitcast i32* %bitmask to <16 x i32>* + %1 = mul i32 %boundary, -2 + %sub = add i32 %1, %xsize + %rem = and i32 %boundary, 63 + %add = add i32 %sub, %rem + %2 = tail call <16 x i32> @llvm.hexagon.V6.lvsplatw(i32 -1) + %3 = tail call <16 x i32> @llvm.hexagon.V6.lvsplatw(i32 1) + %4 = tail call <512 x i1> @llvm.hexagon.V6.pred.scalar2(i32 %add) + %5 = tail call <16 x i32> @llvm.hexagon.V6.vandqrt.acc(<16 x i32> %3, <512 x i1> %4, i32 12) + %and4 = and i32 %add, 511 + %cmp = icmp eq i32 %and4, 0 + %sMaskR.0 = select i1 %cmp, <16 x i32> %2, <16 x i32> %5 + %cmp547 = icmp sgt i32 %add, 0 + br i1 %cmp547, label %for.body.lr.ph, label %for.end + +for.body.lr.ph: ; preds = %entry + %6 = tail call <512 x i1> @llvm.hexagon.V6.pred.scalar2(i32 %boundary) + %7 = tail call <16 x i32> @llvm.hexagon.V6.vandqrt(<512 x i1> %6, i32 16843009) + %8 = tail call <16 x i32> @llvm.hexagon.V6.vnot(<16 x i32> %7) + %9 = add i32 %rem, %xsize + %10 = add i32 %9, -1 + %11 = add i32 %10, %1 + %12 = lshr i32 %11, 9 + %13 = mul i32 %12, 16 + %14 = add nuw nsw i32 %13, 16 + %scevgep = getelementptr i32, i32* %bitmask, i32 %14 + br label %for.body + +for.body: ; preds = %for.body.lr.ph, %for.body + %i.050 = phi i32 [ %add, %for.body.lr.ph ], [ %sub6, %for.body ] + %sMask.049 = phi <16 x i32> [ %8, %for.body.lr.ph ], [ %2, %for.body ] + %optr.048 = phi <16 x i32>* [ %0, %for.body.lr.ph ], [ %incdec.ptr, %for.body ] + %15 = tail call <16 x i32> @llvm.hexagon.V6.vand(<16 x i32> undef, <16 x i32> %sMask.049) + %incdec.ptr = getelementptr inbounds <16 x i32>, <16 x i32>* %optr.048, i32 1 + store <16 x i32> %15, <16 x i32>* %optr.048, align 64 + %sub6 = add nsw i32 %i.050, -512 + %cmp5 = icmp sgt i32 %sub6, 0 + br i1 %cmp5, label %for.body, label %for.cond.for.end_crit_edge + +for.cond.for.end_crit_edge: ; preds = %for.body + %scevgep51 = bitcast i32* %scevgep to <16 x i32>* + br label %for.end + +for.end: ; preds = %for.cond.for.end_crit_edge, %entry + %optr.0.lcssa = phi <16 x i32>* [ %scevgep51, %for.cond.for.end_crit_edge ], [ %0, %entry ] + %16 = load <16 x i32>, <16 x i32>* %optr.0.lcssa, align 64 + %17 = tail call <16 x i32> @llvm.hexagon.V6.vand(<16 x i32> %16, <16 x i32> %sMaskR.0) + store <16 x i32> %17, <16 x i32>* %optr.0.lcssa, align 64 + ret void +} + +declare <16 x i32> @llvm.hexagon.V6.lvsplatw(i32) #1 +declare <512 x i1> @llvm.hexagon.V6.pred.scalar2(i32) #1 +declare <16 x i32> @llvm.hexagon.V6.vandqrt.acc(<16 x i32>, <512 x i1>, i32) #1 +declare <16 x i32> @llvm.hexagon.V6.vandqrt(<512 x i1>, i32) #1 +declare <16 x i32> @llvm.hexagon.V6.vnot(<16 x i32>) #1 +declare <16 x i32> @llvm.hexagon.V6.vand(<16 x i32>, <16 x i32>) #1 + +attributes #0 = { nounwind "target-cpu"="hexagonv60" "target-features"="+hvx" } +attributes #1 = { nounwind readnone } diff --git a/test/CodeGen/Hexagon/v6vec-vprint.ll b/test/CodeGen/Hexagon/v6vec-vprint.ll new file mode 100644 index 000000000000..224547c24b75 --- /dev/null +++ b/test/CodeGen/Hexagon/v6vec-vprint.ll @@ -0,0 +1,36 @@ +; RUN: llc -march=hexagon -mcpu=hexagonv60 -enable-hexagon-hvx -disable-hexagon-shuffle=0 -O2 -enable-hexagon-vector-print < %s | FileCheck --check-prefix=CHECK %s +; RUN: llc -march=hexagon -mcpu=hexagonv60 -enable-hexagon-hvx -disable-hexagon-shuffle=0 -O2 -enable-hexagon-vector-print -trace-hex-vector-stores-only < %s | FileCheck --check-prefix=VSTPRINT %s +; generate .long XXXX which is a vector debug print instruction. +; CHECK: .long 0x1dffe0 +; CHECK: .long 0x1dffe0 +; CHECK: .long 0x1dffe0 +; VSTPRINT: .long 0x1dffe0 +; VSTPRINT-NOT: .long 0x1dffe0 +target datalayout = "e-p:32:32:32-i64:64:64-i32:32:32-i16:16:16-i1:32:32-f64:64:64-f32:32:32-v64:64:64-v32:32:32-a:0-n16:32" +target triple = "hexagon" + +; Function Attrs: nounwind +define void @do_vecs(i8* nocapture readonly %a, i8* nocapture readonly %b, i8* nocapture %c) #0 { +entry: + %0 = bitcast i8* %a to <16 x i32>* + %1 = load <16 x i32>, <16 x i32>* %0, align 4, !tbaa !1 + %2 = bitcast i8* %b to <16 x i32>* + %3 = load <16 x i32>, <16 x i32>* %2, align 4, !tbaa !1 + %4 = tail call <16 x i32> @llvm.hexagon.V6.vaddw(<16 x i32> %1, <16 x i32> %3) + %5 = bitcast i8* %c to <16 x i32>* + store <16 x i32> %4, <16 x i32>* %5, align 4, !tbaa !1 + ret void +} + +; Function Attrs: nounwind readnone +declare <16 x i32> @llvm.hexagon.V6.vaddw(<16 x i32>, <16 x i32>) #1 + +attributes #0 = { nounwind "less-precise-fpmad"="false" "no-frame-pointer-elim"="true" "no-frame-pointer-elim-non-leaf" "no-infs-fp-math"="false" "no-nans-fp-math"="false" "stack-protector-buffer-size"="8" "unsafe-fp-math"="false" "use-soft-float"="false" } +attributes #1 = { nounwind readnone } + +!llvm.ident = !{!0} + +!0 = !{!"QuIC LLVM Hexagon Clang version 7.x-pre-unknown"} +!1 = !{!2, !2, i64 0} +!2 = !{!"omnipotent char", !3, i64 0} +!3 = !{!"Simple C/C++ TBAA"} diff --git a/test/CodeGen/Hexagon/vassign-to-combine.ll b/test/CodeGen/Hexagon/vassign-to-combine.ll new file mode 100644 index 000000000000..a9a0d51e43b6 --- /dev/null +++ b/test/CodeGen/Hexagon/vassign-to-combine.ll @@ -0,0 +1,56 @@ +; RUN: llc -march=hexagon < %s | FileCheck %s + +; This testcase is known to generate an opportunity for creating vcombine +; in HexagonCopyToCombine. + +; CHECK: vcombine + +target triple = "hexagon-unknown--elf" + +declare <32 x i32> @llvm.hexagon.V6.hi.128B(<64 x i32>) #0 +declare <32 x i32> @llvm.hexagon.V6.vabsdiffuh.128B(<32 x i32>, <32 x i32>) #0 +declare <32 x i32> @llvm.hexagon.V6.vlalignbi.128B(<32 x i32>, <32 x i32>, i32) #0 +declare <32 x i32> @llvm.hexagon.V6.vsathub.128B(<32 x i32>, <32 x i32>) #0 +declare <64 x i32> @llvm.hexagon.V6.vaddh.dv.128B(<64 x i32>, <64 x i32>) #0 +declare <64 x i32> @llvm.hexagon.V6.vadduhsat.dv.128B(<64 x i32>, <64 x i32>) #0 +declare <64 x i32> @llvm.hexagon.V6.vaddubh.128B(<32 x i32>, <32 x i32>) #0 +declare <64 x i32> @llvm.hexagon.V6.vmpyub.128B(<32 x i32>, i32) #0 + +define void @foo() local_unnamed_addr #1 { +entry: + %0 = load <32 x i32>, <32 x i32>* undef, align 128 + %1 = load <32 x i32>, <32 x i32>* null, align 128 + br i1 undef, label %b2, label %b1 + +b1: ; preds = %entry + %2 = tail call <32 x i32> @llvm.hexagon.V6.vlalignbi.128B(<32 x i32> %0, <32 x i32> %1, i32 1) + %3 = tail call <64 x i32> @llvm.hexagon.V6.vmpyub.128B(<32 x i32> %2, i32 33686018) #1 + %4 = tail call <64 x i32> @llvm.hexagon.V6.vadduhsat.dv.128B(<64 x i32> undef, <64 x i32> %3) #1 + %5 = tail call <32 x i32> @llvm.hexagon.V6.hi.128B(<64 x i32> %4) + %6 = tail call <32 x i32> @llvm.hexagon.V6.vabsdiffuh.128B(<32 x i32> %5, <32 x i32> undef) #1 + %7 = tail call <64 x i32> @llvm.hexagon.V6.vaddubh.128B(<32 x i32> %6, <32 x i32> undef) + %8 = tail call <64 x i32> @llvm.hexagon.V6.vaddh.dv.128B(<64 x i32> undef, <64 x i32> %7) #1 + %9 = tail call <32 x i32> @llvm.hexagon.V6.hi.128B(<64 x i32> %8) #1 + %10 = tail call <32 x i32> @llvm.hexagon.V6.vsathub.128B(<32 x i32> %9, <32 x i32> undef) #1 + store <32 x i32> %10, <32 x i32>* undef, align 128 + br label %b2 + +b2: ; preds = %b1, %entry + %c2.host31.sroa.3.2.unr.ph = phi <32 x i32> [ zeroinitializer, %b1 ], [ %0, %entry ] + %c2.host31.sroa.0.2.unr.ph = phi <32 x i32> [ %0, %b1 ], [ %1, %entry ] + %11 = tail call <32 x i32> @llvm.hexagon.V6.vlalignbi.128B(<32 x i32> %c2.host31.sroa.3.2.unr.ph, <32 x i32> %c2.host31.sroa.0.2.unr.ph, i32 1) + %12 = tail call <64 x i32> @llvm.hexagon.V6.vmpyub.128B(<32 x i32> %11, i32 33686018) #1 + %13 = tail call <64 x i32> @llvm.hexagon.V6.vadduhsat.dv.128B(<64 x i32> undef, <64 x i32> %12) #1 + %14 = tail call <32 x i32> @llvm.hexagon.V6.hi.128B(<64 x i32> %13) + %15 = tail call <32 x i32> @llvm.hexagon.V6.vabsdiffuh.128B(<32 x i32> %14, <32 x i32> undef) #1 + %16 = tail call <64 x i32> @llvm.hexagon.V6.vaddubh.128B(<32 x i32> %15, <32 x i32> undef) + %17 = tail call <64 x i32> @llvm.hexagon.V6.vaddh.dv.128B(<64 x i32> undef, <64 x i32> %16) #1 + %18 = tail call <32 x i32> @llvm.hexagon.V6.hi.128B(<64 x i32> %17) #1 + %19 = tail call <32 x i32> @llvm.hexagon.V6.vsathub.128B(<32 x i32> %18, <32 x i32> undef) #1 + store <32 x i32> %19, <32 x i32>* undef, align 128 + ret void +} + +attributes #0 = { nounwind readnone } +attributes #1 = { nounwind "target-cpu"="hexagonv60" "target-features"="+hvx,+hvx-double" } + diff --git a/test/CodeGen/Hexagon/vdmpy-halide-test.ll b/test/CodeGen/Hexagon/vdmpy-halide-test.ll new file mode 100644 index 000000000000..7e41bd4d20d4 --- /dev/null +++ b/test/CodeGen/Hexagon/vdmpy-halide-test.ll @@ -0,0 +1,167 @@ +; RUN: llc -march=hexagon < %s +; REQUIRES: asserts + +; Thie tests checks a compiler assert. So the test just needs to compile for it to pass +target triple = "hexagon-unknown--elf" + +%struct.buffer_t = type { i64, i8*, [4 x i32], [4 x i32], [4 x i32], i32, i8, i8, [6 x i8] } + +; Function Attrs: norecurse nounwind +define i32 @__testOne(%struct.buffer_t* noalias nocapture readonly %inputOne.buffer, %struct.buffer_t* noalias nocapture readonly %inputTwo.buffer, %struct.buffer_t* noalias nocapture readonly %testOne.buffer) #0 { +entry: + %buf_host = getelementptr inbounds %struct.buffer_t, %struct.buffer_t* %inputOne.buffer, i32 0, i32 1 + %inputOne.host = load i8*, i8** %buf_host, align 4 + %buf_min = getelementptr inbounds %struct.buffer_t, %struct.buffer_t* %inputOne.buffer, i32 0, i32 4, i32 0 + %inputOne.min.0 = load i32, i32* %buf_min, align 4 + %buf_host10 = getelementptr inbounds %struct.buffer_t, %struct.buffer_t* %inputTwo.buffer, i32 0, i32 1 + %inputTwo.host = load i8*, i8** %buf_host10, align 4 + %buf_min22 = getelementptr inbounds %struct.buffer_t, %struct.buffer_t* %inputTwo.buffer, i32 0, i32 4, i32 0 + %inputTwo.min.0 = load i32, i32* %buf_min22, align 4 + %buf_host27 = getelementptr inbounds %struct.buffer_t, %struct.buffer_t* %testOne.buffer, i32 0, i32 1 + %testOne.host = load i8*, i8** %buf_host27, align 4 + %buf_extent31 = getelementptr inbounds %struct.buffer_t, %struct.buffer_t* %testOne.buffer, i32 0, i32 2, i32 0 + %testOne.extent.0 = load i32, i32* %buf_extent31, align 4 + %buf_min39 = getelementptr inbounds %struct.buffer_t, %struct.buffer_t* %testOne.buffer, i32 0, i32 4, i32 0 + %testOne.min.0 = load i32, i32* %buf_min39, align 4 + %0 = ashr i32 %testOne.extent.0, 4 + %1 = icmp sgt i32 %0, 0 + br i1 %1, label %"for testOne.s0.x.x.preheader", label %"end for testOne.s0.x.x" + +"for testOne.s0.x.x.preheader": ; preds = %entry + %2 = bitcast i8* %inputOne.host to i16* + %3 = bitcast i8* %inputTwo.host to i16* + %4 = bitcast i8* %testOne.host to i32* + br label %"for testOne.s0.x.x" + +"for testOne.s0.x.x": ; preds = %"for testOne.s0.x.x", %"for testOne.s0.x.x.preheader" + %.phi = phi i32* [ %4, %"for testOne.s0.x.x.preheader" ], [ %.inc, %"for testOne.s0.x.x" ] + %testOne.s0.x.x = phi i32 [ 0, %"for testOne.s0.x.x.preheader" ], [ %50, %"for testOne.s0.x.x" ] + %5 = shl nsw i32 %testOne.s0.x.x, 4 + %6 = add nsw i32 %5, %testOne.min.0 + %7 = shl nsw i32 %6, 1 + %8 = sub nsw i32 %7, %inputOne.min.0 + %9 = getelementptr inbounds i16, i16* %2, i32 %8 + %10 = bitcast i16* %9 to <16 x i16>* + %11 = load <16 x i16>, <16 x i16>* %10, align 2, !tbaa !5 + %12 = add nsw i32 %8, 15 + %13 = getelementptr inbounds i16, i16* %2, i32 %12 + %14 = bitcast i16* %13 to <16 x i16>* + %15 = load <16 x i16>, <16 x i16>* %14, align 2, !tbaa !5 + %16 = shufflevector <16 x i16> %11, <16 x i16> %15, <16 x i32> + %17 = add nsw i32 %8, 1 + %18 = getelementptr inbounds i16, i16* %2, i32 %17 + %19 = bitcast i16* %18 to <16 x i16>* + %20 = load <16 x i16>, <16 x i16>* %19, align 2, !tbaa !5 + %21 = add nsw i32 %8, 16 + %22 = getelementptr inbounds i16, i16* %2, i32 %21 + %23 = bitcast i16* %22 to <16 x i16>* + %24 = load <16 x i16>, <16 x i16>* %23, align 2, !tbaa !5 + %25 = shufflevector <16 x i16> %20, <16 x i16> %24, <16 x i32> + %26 = shufflevector <16 x i16> %16, <16 x i16> %25, <32 x i32> + %27 = sub nsw i32 %7, %inputTwo.min.0 + %28 = getelementptr inbounds i16, i16* %3, i32 %27 + %29 = bitcast i16* %28 to <16 x i16>* + %30 = load <16 x i16>, <16 x i16>* %29, align 2, !tbaa !8 + %31 = add nsw i32 %27, 15 + %32 = getelementptr inbounds i16, i16* %3, i32 %31 + %33 = bitcast i16* %32 to <16 x i16>* + %34 = load <16 x i16>, <16 x i16>* %33, align 2, !tbaa !8 + %35 = shufflevector <16 x i16> %30, <16 x i16> %34, <16 x i32> + %36 = add nsw i32 %27, 1 + %37 = getelementptr inbounds i16, i16* %3, i32 %36 + %38 = bitcast i16* %37 to <16 x i16>* + %39 = load <16 x i16>, <16 x i16>* %38, align 2, !tbaa !8 + %40 = add nsw i32 %27, 16 + %41 = getelementptr inbounds i16, i16* %3, i32 %40 + %42 = bitcast i16* %41 to <16 x i16>* + %43 = load <16 x i16>, <16 x i16>* %42, align 2, !tbaa !8 + %44 = shufflevector <16 x i16> %39, <16 x i16> %43, <16 x i32> + %45 = shufflevector <16 x i16> %35, <16 x i16> %44, <32 x i32> + %46 = bitcast <32 x i16> %26 to <16 x i32> + %47 = bitcast <32 x i16> %45 to <16 x i32> + %48 = tail call <16 x i32> @llvm.hexagon.V6.vdmpyhvsat(<16 x i32> %46, <16 x i32> %47) + %49 = bitcast i32* %.phi to <16 x i32>* + store <16 x i32> %48, <16 x i32>* %49, align 4, !tbaa !10 + %50 = add nuw nsw i32 %testOne.s0.x.x, 1 + %51 = icmp eq i32 %50, %0 + %.inc = getelementptr i32, i32* %.phi, i32 16 + br i1 %51, label %"end for testOne.s0.x.x", label %"for testOne.s0.x.x" + +"end for testOne.s0.x.x": ; preds = %"for testOne.s0.x.x", %entry + %52 = add nsw i32 %testOne.extent.0, 15 + %53 = ashr i32 %52, 4 + %54 = icmp sgt i32 %53, %0 + br i1 %54, label %"for testOne.s0.x.x44.preheader", label %destructor_block + +"for testOne.s0.x.x44.preheader": ; preds = %"end for testOne.s0.x.x" + %55 = add nsw i32 %testOne.min.0, %testOne.extent.0 + %56 = shl nsw i32 %55, 1 + %57 = sub nsw i32 %56, %inputOne.min.0 + %58 = add nsw i32 %57, -32 + %59 = bitcast i8* %inputOne.host to i16* + %60 = getelementptr inbounds i16, i16* %59, i32 %58 + %61 = bitcast i16* %60 to <16 x i16>* + %62 = load <16 x i16>, <16 x i16>* %61, align 2 + %63 = add nsw i32 %57, -17 + %64 = getelementptr inbounds i16, i16* %59, i32 %63 + %65 = bitcast i16* %64 to <16 x i16>* + %66 = load <16 x i16>, <16 x i16>* %65, align 2 + %67 = add nsw i32 %57, -31 + %68 = getelementptr inbounds i16, i16* %59, i32 %67 + %69 = bitcast i16* %68 to <16 x i16>* + %70 = load <16 x i16>, <16 x i16>* %69, align 2 + %71 = add nsw i32 %57, -16 + %72 = getelementptr inbounds i16, i16* %59, i32 %71 + %73 = bitcast i16* %72 to <16 x i16>* + %74 = load <16 x i16>, <16 x i16>* %73, align 2 + %75 = shufflevector <16 x i16> %70, <16 x i16> %74, <16 x i32> + %76 = sub nsw i32 %56, %inputTwo.min.0 + %77 = add nsw i32 %76, -32 + %78 = bitcast i8* %inputTwo.host to i16* + %79 = getelementptr inbounds i16, i16* %78, i32 %77 + %80 = bitcast i16* %79 to <16 x i16>* + %81 = load <16 x i16>, <16 x i16>* %80, align 2 + %82 = add nsw i32 %76, -17 + %83 = getelementptr inbounds i16, i16* %78, i32 %82 + %84 = bitcast i16* %83 to <16 x i16>* + %85 = load <16 x i16>, <16 x i16>* %84, align 2 + %86 = shufflevector <16 x i16> %81, <16 x i16> %85, <16 x i32> + %87 = add nsw i32 %76, -31 + %88 = getelementptr inbounds i16, i16* %78, i32 %87 + %89 = bitcast i16* %88 to <16 x i16>* + %90 = load <16 x i16>, <16 x i16>* %89, align 2 + %91 = add nsw i32 %76, -16 + %92 = getelementptr inbounds i16, i16* %78, i32 %91 + %93 = bitcast i16* %92 to <16 x i16>* + %94 = load <16 x i16>, <16 x i16>* %93, align 2 + %95 = shufflevector <16 x i16> %90, <16 x i16> %94, <16 x i32> + %96 = shufflevector <16 x i16> %86, <16 x i16> %95, <32 x i32> + %97 = bitcast <32 x i16> %96 to <16 x i32> + %98 = add nsw i32 %testOne.extent.0, -16 + %99 = bitcast i8* %testOne.host to i32* + %100 = getelementptr inbounds i32, i32* %99, i32 %98 + %101 = shufflevector <16 x i16> %62, <16 x i16> %66, <16 x i32> + %102 = shufflevector <16 x i16> %101, <16 x i16> %75, <32 x i32> + %103 = bitcast <32 x i16> %102 to <16 x i32> + %104 = tail call <16 x i32> @llvm.hexagon.V6.vdmpyhvsat(<16 x i32> %103, <16 x i32> %97) + %105 = bitcast i32* %100 to <16 x i32>* + store <16 x i32> %104, <16 x i32>* %105, align 4, !tbaa !10 + br label %destructor_block + +destructor_block: ; preds = %"for testOne.s0.x.x44.preheader", %"end for testOne.s0.x.x" + ret i32 0 +} + +; Function Attrs: nounwind readnone +declare <16 x i32> @llvm.hexagon.V6.vdmpyhvsat(<16 x i32>, <16 x i32>) #1 + +attributes #0 = { norecurse nounwind "target-cpu"="hexagonv60" "target-features"="+hvx,-hvx-double" } +attributes #1 = { nounwind readnone "target-cpu"="hexagonv60" "target-features"="+hvx,-hvx-double" } + +!5 = !{!6, !6, i64 0} +!6 = !{!"inputOne", !7} +!7 = !{!"Halide buffer"} +!8 = !{!9, !9, i64 0} +!9 = !{!"inputTwo", !7} +!10 = !{!11, !11, i64 0} +!11 = !{!"testOne", !7} diff --git a/test/CodeGen/Hexagon/vect/vect-vsplatb.ll b/test/CodeGen/Hexagon/vect/vect-vsplatb.ll index 6996dd144eba..097e2ccd600a 100644 --- a/test/CodeGen/Hexagon/vect/vect-vsplatb.ll +++ b/test/CodeGen/Hexagon/vect/vect-vsplatb.ll @@ -1,4 +1,4 @@ -; RUN: llc -march=hexagon < %s | FileCheck %s +; RUN: llc -march=hexagon -disable-hcp < %s | FileCheck %s ; Make sure we build the constant vector <7, 7, 7, 7> with a vsplatb. ; CHECK: vsplatb @B = common global [400 x i8] zeroinitializer, align 8 diff --git a/test/CodeGen/Hexagon/vect/vect-vsplath.ll b/test/CodeGen/Hexagon/vect/vect-vsplath.ll index f5207109773e..db90bf42be2a 100644 --- a/test/CodeGen/Hexagon/vect/vect-vsplath.ll +++ b/test/CodeGen/Hexagon/vect/vect-vsplath.ll @@ -1,4 +1,4 @@ -; RUN: llc -march=hexagon < %s | FileCheck %s +; RUN: llc -march=hexagon -disable-hcp < %s | FileCheck %s ; Make sure we build the constant vector <7, 7, 7, 7> with a vsplath. ; CHECK: vsplath @B = common global [400 x i16] zeroinitializer, align 8 diff --git a/test/CodeGen/Hexagon/vector-ext-load.ll b/test/CodeGen/Hexagon/vector-ext-load.ll new file mode 100644 index 000000000000..536dad165ef8 --- /dev/null +++ b/test/CodeGen/Hexagon/vector-ext-load.ll @@ -0,0 +1,10 @@ +; A copy of 2012-06-08-APIntCrash.ll with arch explicitly set to hexagon. + +; RUN: llc -march=hexagon < %s + +define void @test1(<8 x i32>* %ptr) { + %1 = load <8 x i32>, <8 x i32>* %ptr, align 32 + %2 = and <8 x i32> %1, + store <8 x i32> %2, <8 x i32>* %ptr, align 16 + ret void +} diff --git a/test/CodeGen/Hexagon/vmpa-halide-test.ll b/test/CodeGen/Hexagon/vmpa-halide-test.ll new file mode 100644 index 000000000000..9c359900ba42 --- /dev/null +++ b/test/CodeGen/Hexagon/vmpa-halide-test.ll @@ -0,0 +1,145 @@ +; RUN: llc -march=hexagon < %s +; Thie tests checks a compiler assert. So the test just needs to compile +; for it to pass + +target triple = "hexagon-unknown--elf" + +%struct.buffer_t = type { i64, i8*, [4 x i32], [4 x i32], [4 x i32], i32, i8, i8, [6 x i8] } + +; Function Attrs: norecurse nounwind +define i32 @__testOne(%struct.buffer_t* noalias nocapture readonly %inputOne.buffer, %struct.buffer_t* noalias nocapture readonly %inputTwo.buffer, %struct.buffer_t* noalias nocapture readonly %testOne.buffer) #0 { +entry: + %buf_host = getelementptr inbounds %struct.buffer_t, %struct.buffer_t* %inputOne.buffer, i32 0, i32 1 + %inputOne.host = load i8*, i8** %buf_host, align 4 + %buf_min = getelementptr inbounds %struct.buffer_t, %struct.buffer_t* %inputOne.buffer, i32 0, i32 4, i32 0 + %inputOne.min.0 = load i32, i32* %buf_min, align 4 + %buf_host10 = getelementptr inbounds %struct.buffer_t, %struct.buffer_t* %inputTwo.buffer, i32 0, i32 1 + %inputTwo.host = load i8*, i8** %buf_host10, align 4 + %buf_min22 = getelementptr inbounds %struct.buffer_t, %struct.buffer_t* %inputTwo.buffer, i32 0, i32 4, i32 0 + %inputTwo.min.0 = load i32, i32* %buf_min22, align 4 + %buf_host27 = getelementptr inbounds %struct.buffer_t, %struct.buffer_t* %testOne.buffer, i32 0, i32 1 + %testOne.host = load i8*, i8** %buf_host27, align 4 + %buf_extent31 = getelementptr inbounds %struct.buffer_t, %struct.buffer_t* %testOne.buffer, i32 0, i32 2, i32 0 + %testOne.extent.0 = load i32, i32* %buf_extent31, align 4 + %buf_min39 = getelementptr inbounds %struct.buffer_t, %struct.buffer_t* %testOne.buffer, i32 0, i32 4, i32 0 + %testOne.min.0 = load i32, i32* %buf_min39, align 4 + %0 = ashr i32 %testOne.extent.0, 6 + %1 = icmp sgt i32 %0, 0 + br i1 %1, label %"for testOne.s0.x.x.preheader", label %"end for testOne.s0.x.x" + +"for testOne.s0.x.x.preheader": ; preds = %entry + %2 = bitcast i8* %testOne.host to i16* + br label %"for testOne.s0.x.x" + +"for testOne.s0.x.x": ; preds = %"for testOne.s0.x.x", %"for testOne.s0.x.x.preheader" + %.phi = phi i16* [ %2, %"for testOne.s0.x.x.preheader" ], [ %.inc, %"for testOne.s0.x.x" ] + %testOne.s0.x.x = phi i32 [ 0, %"for testOne.s0.x.x.preheader" ], [ %38, %"for testOne.s0.x.x" ] + %3 = shl nsw i32 %testOne.s0.x.x, 6 + %4 = add nsw i32 %3, %testOne.min.0 + %5 = shl nsw i32 %4, 1 + %6 = sub nsw i32 %5, %inputOne.min.0 + %7 = getelementptr inbounds i8, i8* %inputOne.host, i32 %6 + %8 = bitcast i8* %7 to <64 x i8>* + %9 = load <64 x i8>, <64 x i8>* %8, align 1, !tbaa !5 + %10 = add nsw i32 %6, 64 + %11 = getelementptr inbounds i8, i8* %inputOne.host, i32 %10 + %12 = bitcast i8* %11 to <64 x i8>* + %13 = load <64 x i8>, <64 x i8>* %12, align 1, !tbaa !5 + %14 = shufflevector <64 x i8> %9, <64 x i8> %13, <64 x i32> + %15 = shufflevector <64 x i8> %9, <64 x i8> %13, <64 x i32> + %16 = shufflevector <64 x i8> %14, <64 x i8> %15, <128 x i32> + %17 = sub nsw i32 %5, %inputTwo.min.0 + %18 = getelementptr inbounds i8, i8* %inputTwo.host, i32 %17 + %19 = bitcast i8* %18 to <64 x i8>* + %20 = load <64 x i8>, <64 x i8>* %19, align 1, !tbaa !8 + %21 = add nsw i32 %17, 64 + %22 = getelementptr inbounds i8, i8* %inputTwo.host, i32 %21 + %23 = bitcast i8* %22 to <64 x i8>* + %24 = load <64 x i8>, <64 x i8>* %23, align 1, !tbaa !8 + %25 = shufflevector <64 x i8> %20, <64 x i8> %24, <64 x i32> + %26 = shufflevector <64 x i8> %20, <64 x i8> %24, <64 x i32> + %27 = shufflevector <64 x i8> %25, <64 x i8> %26, <128 x i32> + %28 = bitcast <128 x i8> %16 to <32 x i32> + %29 = bitcast <128 x i8> %27 to <32 x i32> + %30 = tail call <32 x i32> @llvm.hexagon.V6.vmpabuuv(<32 x i32> %28, <32 x i32> %29) + %31 = bitcast <32 x i32> %30 to <64 x i16> + %32 = shufflevector <64 x i16> %31, <64 x i16> undef, <32 x i32> + %33 = bitcast i16* %.phi to <32 x i16>* + store <32 x i16> %32, <32 x i16>* %33, align 2, !tbaa !10 + %34 = shufflevector <64 x i16> %31, <64 x i16> undef, <32 x i32> + %35 = or i32 %3, 32 + %36 = getelementptr inbounds i16, i16* %2, i32 %35 + %37 = bitcast i16* %36 to <32 x i16>* + store <32 x i16> %34, <32 x i16>* %37, align 2, !tbaa !10 + %38 = add nuw nsw i32 %testOne.s0.x.x, 1 + %39 = icmp eq i32 %38, %0 + %.inc = getelementptr i16, i16* %.phi, i32 64 + br i1 %39, label %"end for testOne.s0.x.x", label %"for testOne.s0.x.x" + +"end for testOne.s0.x.x": ; preds = %"for testOne.s0.x.x", %entry + %40 = add nsw i32 %testOne.extent.0, 63 + %41 = ashr i32 %40, 6 + %42 = icmp sgt i32 %41, %0 + br i1 %42, label %"for testOne.s0.x.x44.preheader", label %destructor_block + +"for testOne.s0.x.x44.preheader": ; preds = %"end for testOne.s0.x.x" + %43 = add nsw i32 %testOne.min.0, %testOne.extent.0 + %44 = shl nsw i32 %43, 1 + %45 = sub nsw i32 %44, %inputOne.min.0 + %46 = add nsw i32 %45, -128 + %47 = getelementptr inbounds i8, i8* %inputOne.host, i32 %46 + %48 = bitcast i8* %47 to <64 x i8>* + %49 = load <64 x i8>, <64 x i8>* %48, align 1 + %50 = add nsw i32 %45, -64 + %51 = getelementptr inbounds i8, i8* %inputOne.host, i32 %50 + %52 = bitcast i8* %51 to <64 x i8>* + %53 = load <64 x i8>, <64 x i8>* %52, align 1 + %54 = shufflevector <64 x i8> %49, <64 x i8> %53, <64 x i32> + %55 = shufflevector <64 x i8> %49, <64 x i8> %53, <64 x i32> + %56 = shufflevector <64 x i8> %54, <64 x i8> %55, <128 x i32> + %57 = sub nsw i32 %44, %inputTwo.min.0 + %58 = add nsw i32 %57, -128 + %59 = getelementptr inbounds i8, i8* %inputTwo.host, i32 %58 + %60 = bitcast i8* %59 to <64 x i8>* + %61 = load <64 x i8>, <64 x i8>* %60, align 1 + %62 = add nsw i32 %57, -64 + %63 = getelementptr inbounds i8, i8* %inputTwo.host, i32 %62 + %64 = bitcast i8* %63 to <64 x i8>* + %65 = load <64 x i8>, <64 x i8>* %64, align 1 + %66 = shufflevector <64 x i8> %61, <64 x i8> %65, <64 x i32> + %67 = shufflevector <64 x i8> %61, <64 x i8> %65, <64 x i32> + %68 = shufflevector <64 x i8> %66, <64 x i8> %67, <128 x i32> + %69 = bitcast <128 x i8> %56 to <32 x i32> + %70 = bitcast <128 x i8> %68 to <32 x i32> + %71 = tail call <32 x i32> @llvm.hexagon.V6.vmpabuuv(<32 x i32> %69, <32 x i32> %70) + %72 = bitcast <32 x i32> %71 to <64 x i16> + %73 = add nsw i32 %testOne.extent.0, -64 + %74 = bitcast i8* %testOne.host to i16* + %75 = getelementptr inbounds i16, i16* %74, i32 %73 + %76 = bitcast i16* %75 to <32 x i16>* + %77 = add nsw i32 %testOne.extent.0, -32 + %78 = getelementptr inbounds i16, i16* %74, i32 %77 + %79 = shufflevector <64 x i16> %72, <64 x i16> undef, <32 x i32> + %80 = shufflevector <64 x i16> %72, <64 x i16> undef, <32 x i32> + %81 = bitcast i16* %78 to <32 x i16>* + store <32 x i16> %79, <32 x i16>* %76, align 2, !tbaa !10 + store <32 x i16> %80, <32 x i16>* %81, align 2, !tbaa !10 + br label %destructor_block + +destructor_block: ; preds = %"for testOne.s0.x.x44.preheader", %"end for testOne.s0.x.x" + ret i32 0 +} + +; Function Attrs: nounwind readnone +declare <32 x i32> @llvm.hexagon.V6.vmpabuuv(<32 x i32>, <32 x i32>) #1 + +attributes #0 = { norecurse nounwind "target-cpu"="hexagonv60" "target-features"="+hvx,-hvx-double" } +attributes #1 = { nounwind readnone "target-cpu"="hexagonv60" "target-features"="+hvx,-hvx-double" } + +!5 = !{!6, !6, i64 0} +!6 = !{!"inputOne", !7} +!7 = !{!"Halide buffer"} +!8 = !{!9, !9, i64 0} +!9 = !{!"inputTwo", !7} +!10 = !{!11, !11, i64 0} +!11 = !{!"testOne", !7} diff --git a/test/CodeGen/Hexagon/vpack_eo.ll b/test/CodeGen/Hexagon/vpack_eo.ll new file mode 100644 index 000000000000..7238ca84a42e --- /dev/null +++ b/test/CodeGen/Hexagon/vpack_eo.ll @@ -0,0 +1,73 @@ +; RUN: llc -march=hexagon < %s | FileCheck %s +target triple = "hexagon-unknown--elf" + +; CHECK-DAG: vpacke +; CHECK-DAG: vpacko + +%struct.buffer_t = type { i64, i8*, [4 x i32], [4 x i32], [4 x i32], i32, i8, i8, [6 x i8] } + +; Function Attrs: norecurse nounwind +define i32 @__Strided_LoadTest(%struct.buffer_t* noalias nocapture readonly %InputOne.buffer, %struct.buffer_t* noalias nocapture readonly %InputTwo.buffer, %struct.buffer_t* noalias nocapture readonly %Strided_LoadTest.buffer) #0 { +entry: + %buf_host = getelementptr inbounds %struct.buffer_t, %struct.buffer_t* %InputOne.buffer, i32 0, i32 1 + %0 = bitcast i8** %buf_host to i16** + %InputOne.host45 = load i16*, i16** %0, align 4 + %buf_host10 = getelementptr inbounds %struct.buffer_t, %struct.buffer_t* %InputTwo.buffer, i32 0, i32 1 + %1 = bitcast i8** %buf_host10 to i16** + %InputTwo.host46 = load i16*, i16** %1, align 4 + %buf_host27 = getelementptr inbounds %struct.buffer_t, %struct.buffer_t* %Strided_LoadTest.buffer, i32 0, i32 1 + %2 = bitcast i8** %buf_host27 to i16** + %Strided_LoadTest.host44 = load i16*, i16** %2, align 4 + %3 = bitcast i16* %InputOne.host45 to <32 x i16>* + %4 = load <32 x i16>, <32 x i16>* %3, align 2, !tbaa !4 + %5 = getelementptr inbounds i16, i16* %InputOne.host45, i32 32 + %6 = bitcast i16* %5 to <32 x i16>* + %7 = load <32 x i16>, <32 x i16>* %6, align 2, !tbaa !4 + %8 = shufflevector <32 x i16> %4, <32 x i16> %7, <32 x i32> + %9 = bitcast i16* %InputTwo.host46 to <32 x i16>* + %10 = load <32 x i16>, <32 x i16>* %9, align 2, !tbaa !7 + %11 = getelementptr inbounds i16, i16* %InputTwo.host46, i32 32 + %12 = bitcast i16* %11 to <32 x i16>* + %13 = load <32 x i16>, <32 x i16>* %12, align 2, !tbaa !7 + %14 = shufflevector <32 x i16> %10, <32 x i16> %13, <32 x i32> + %15 = bitcast <32 x i16> %8 to <16 x i32> + %16 = bitcast <32 x i16> %14 to <16 x i32> + %17 = tail call <16 x i32> @llvm.hexagon.V6.vaddh(<16 x i32> %15, <16 x i32> %16) + %18 = bitcast i16* %Strided_LoadTest.host44 to <16 x i32>* + store <16 x i32> %17, <16 x i32>* %18, align 2, !tbaa !9 + %.inc = getelementptr i16, i16* %InputOne.host45, i32 64 + %.inc49 = getelementptr i16, i16* %InputTwo.host46, i32 64 + %.inc52 = getelementptr i16, i16* %Strided_LoadTest.host44, i32 32 + %19 = bitcast i16* %.inc to <32 x i16>* + %20 = load <32 x i16>, <32 x i16>* %19, align 2, !tbaa !4 + %21 = getelementptr inbounds i16, i16* %InputOne.host45, i32 96 + %22 = bitcast i16* %21 to <32 x i16>* + %23 = load <32 x i16>, <32 x i16>* %22, align 2, !tbaa !4 + %24 = shufflevector <32 x i16> %20, <32 x i16> %23, <32 x i32> + %25 = bitcast i16* %.inc49 to <32 x i16>* + %26 = load <32 x i16>, <32 x i16>* %25, align 2, !tbaa !7 + %27 = getelementptr inbounds i16, i16* %InputTwo.host46, i32 96 + %28 = bitcast i16* %27 to <32 x i16>* + %29 = load <32 x i16>, <32 x i16>* %28, align 2, !tbaa !7 + %30 = shufflevector <32 x i16> %26, <32 x i16> %29, <32 x i32> + %31 = bitcast <32 x i16> %24 to <16 x i32> + %32 = bitcast <32 x i16> %30 to <16 x i32> + %33 = tail call <16 x i32> @llvm.hexagon.V6.vaddh(<16 x i32> %31, <16 x i32> %32) + %34 = bitcast i16* %.inc52 to <16 x i32>* + store <16 x i32> %33, <16 x i32>* %34, align 2, !tbaa !9 + ret i32 0 +} + +; Function Attrs: nounwind readnone +declare <16 x i32> @llvm.hexagon.V6.vaddh(<16 x i32>, <16 x i32>) #1 + +attributes #0 = { norecurse nounwind "target-cpu"="hexagonv60" "target-features"="+hvx,-hvx-double" } +attributes #1 = { nounwind readnone "target-cpu"="hexagonv60" "target-features"="+hvx,-hvx-double" } + +!4 = !{!5, !5, i64 0} +!5 = !{!"InputOne", !6} +!6 = !{!"Halide buffer"} +!7 = !{!8, !8, i64 0} +!8 = !{!"InputTwo", !6} +!9 = !{!10, !10, i64 0} +!10 = !{!"Strided_LoadTest", !6} -- cgit v1.3