summaryrefslogtreecommitdiff
path: root/test/CodeGen/Hexagon
diff options
context:
space:
mode:
authorDimitry Andric <dim@FreeBSD.org>2017-01-02 19:17:04 +0000
committerDimitry Andric <dim@FreeBSD.org>2017-01-02 19:17:04 +0000
commitb915e9e0fc85ba6f398b3fab0db6a81a8913af94 (patch)
tree98b8f811c7aff2547cab8642daf372d6c59502fb /test/CodeGen/Hexagon
parent6421cca32f69ac849537a3cff78c352195e99f1b (diff)
Notes
Diffstat (limited to 'test/CodeGen/Hexagon')
-rw-r--r--test/CodeGen/Hexagon/SUnit-boundary-prob.ll202
-rw-r--r--test/CodeGen/Hexagon/addh-sext-trunc.ll30
-rw-r--r--test/CodeGen/Hexagon/addr-calc-opt.ll54
-rw-r--r--test/CodeGen/Hexagon/anti-dep-partial.mir34
-rw-r--r--test/CodeGen/Hexagon/bit-gen-rseq.ll43
-rw-r--r--test/CodeGen/Hexagon/bit-loop-rc-mismatch.ll30
-rw-r--r--test/CodeGen/Hexagon/bit-rie.ll208
-rw-r--r--test/CodeGen/Hexagon/bit-skip-byval.ll11
-rw-r--r--test/CodeGen/Hexagon/bit-validate-reg.ll21
-rw-r--r--test/CodeGen/Hexagon/bit-visit-flowq.ll47
-rw-r--r--test/CodeGen/Hexagon/block-addr.ll7
-rw-r--r--test/CodeGen/Hexagon/branchfolder-keep-impdef.ll29
-rw-r--r--test/CodeGen/Hexagon/build-vector-shuffle.ll21
-rw-r--r--test/CodeGen/Hexagon/combine.ll2
-rw-r--r--test/CodeGen/Hexagon/const-pool-tf.ll40
-rw-r--r--test/CodeGen/Hexagon/constp-clb.ll23
-rw-r--r--test/CodeGen/Hexagon/constp-combine-neg.ll27
-rw-r--r--test/CodeGen/Hexagon/constp-ctb.ll26
-rw-r--r--test/CodeGen/Hexagon/constp-extract.ll31
-rw-r--r--test/CodeGen/Hexagon/constp-physreg.ll21
-rw-r--r--test/CodeGen/Hexagon/constp-rewrite-branches.ll17
-rw-r--r--test/CodeGen/Hexagon/constp-rseq.ll19
-rw-r--r--test/CodeGen/Hexagon/constp-vsplat.ll18
-rw-r--r--test/CodeGen/Hexagon/copy-to-combine-dbg.ll57
-rw-r--r--test/CodeGen/Hexagon/dead-store-stack.ll131
-rw-r--r--test/CodeGen/Hexagon/early-if-vecpi.ll69
-rw-r--r--test/CodeGen/Hexagon/expand-condsets-def-undef.mir41
-rw-r--r--test/CodeGen/Hexagon/expand-condsets-extend.ll112
-rw-r--r--test/CodeGen/Hexagon/expand-condsets-impuse.mir78
-rw-r--r--test/CodeGen/Hexagon/expand-condsets-rm-reg.mir49
-rw-r--r--test/CodeGen/Hexagon/expand-condsets-same-inputs.mir32
-rw-r--r--test/CodeGen/Hexagon/expand-condsets-undef2.ll47
-rw-r--r--test/CodeGen/Hexagon/expand-vstorerw-undef.ll95
-rw-r--r--test/CodeGen/Hexagon/fixed-spill-mutable.ll69
-rw-r--r--test/CodeGen/Hexagon/float-amode.ll89
-rw-r--r--test/CodeGen/Hexagon/fminmax.ll27
-rw-r--r--test/CodeGen/Hexagon/frame-offset-overflow.ll163
-rw-r--r--test/CodeGen/Hexagon/fsel.ll22
-rw-r--r--test/CodeGen/Hexagon/hwloop-crit-edge.ll1
-rw-r--r--test/CodeGen/Hexagon/hwloop-loop1.ll2
-rw-r--r--test/CodeGen/Hexagon/hwloop-noreturn-call.ll63
-rw-r--r--test/CodeGen/Hexagon/hwloop-preh.ll44
-rw-r--r--test/CodeGen/Hexagon/hwloop1.ll2
-rw-r--r--test/CodeGen/Hexagon/ifcvt-diamond-bug-2016-08-26.ll37
-rw-r--r--test/CodeGen/Hexagon/ifcvt-impuse-livein.mir42
-rw-r--r--test/CodeGen/Hexagon/ifcvt-live-subreg.mir50
-rw-r--r--test/CodeGen/Hexagon/inline-asm-hexagon.ll16
-rw-r--r--test/CodeGen/Hexagon/inline-asm-i1.ll14
-rw-r--r--test/CodeGen/Hexagon/insert4.ll10
-rw-r--r--test/CodeGen/Hexagon/intrinsics/llsc_bundling.ll12
-rw-r--r--test/CodeGen/Hexagon/is-legal-void.ll58
-rw-r--r--test/CodeGen/Hexagon/livephysregs-lane-masks.mir40
-rw-r--r--test/CodeGen/Hexagon/livephysregs-lane-masks2.mir55
-rw-r--r--test/CodeGen/Hexagon/long-calls.ll73
-rw-r--r--test/CodeGen/Hexagon/loop-prefetch.ll27
-rw-r--r--test/CodeGen/Hexagon/lower-extract-subvector.ll47
-rw-r--r--test/CodeGen/Hexagon/misaligned_double_vector_store_not_fast.ll47
-rw-r--r--test/CodeGen/Hexagon/mulhs.ll23
-rw-r--r--test/CodeGen/Hexagon/newvalueSameReg.ll63
-rw-r--r--test/CodeGen/Hexagon/opt-spill-volatile.ll29
-rw-r--r--test/CodeGen/Hexagon/packetize-cfi-location.ll72
-rw-r--r--test/CodeGen/Hexagon/packetize-return-arg.ll37
-rw-r--r--test/CodeGen/Hexagon/peephole-kill-flags.ll27
-rw-r--r--test/CodeGen/Hexagon/pic-simple.ll2
-rw-r--r--test/CodeGen/Hexagon/pic-static.ll2
-rw-r--r--test/CodeGen/Hexagon/post-inc-aa-metadata.ll37
-rw-r--r--test/CodeGen/Hexagon/post-ra-kill-update.mir37
-rw-r--r--test/CodeGen/Hexagon/propagate-vcombine.ll48
-rw-r--r--test/CodeGen/Hexagon/rdf-copy.ll2
-rw-r--r--test/CodeGen/Hexagon/rdf-extra-livein.ll73
-rw-r--r--test/CodeGen/Hexagon/rdf-filter-defs.ll214
-rw-r--r--test/CodeGen/Hexagon/rdf-ignore-undef.ll55
-rw-r--r--test/CodeGen/Hexagon/rdf-multiple-phis-up.ll40
-rw-r--r--test/CodeGen/Hexagon/rdf-phi-shadows.ll64
-rw-r--r--test/CodeGen/Hexagon/rdf-phi-up.ll60
-rw-r--r--test/CodeGen/Hexagon/regalloc-bad-undef.mir204
-rw-r--r--test/CodeGen/Hexagon/sf-min-max.ll67
-rw-r--r--test/CodeGen/Hexagon/sffms.ll25
-rw-r--r--test/CodeGen/Hexagon/split-const32-const64.ll18
-rw-r--r--test/CodeGen/Hexagon/storerd-io-over-rr.ll12
-rw-r--r--test/CodeGen/Hexagon/struct_args.ll8
-rw-r--r--test/CodeGen/Hexagon/subi-asl.ll70
-rw-r--r--test/CodeGen/Hexagon/swp-const-tc.ll51
-rw-r--r--test/CodeGen/Hexagon/swp-dag-phi.ll42
-rw-r--r--test/CodeGen/Hexagon/swp-epilog-phi10.ll88
-rw-r--r--test/CodeGen/Hexagon/swp-epilog-reuse-1.ll44
-rw-r--r--test/CodeGen/Hexagon/swp-epilog-reuse.ll65
-rw-r--r--test/CodeGen/Hexagon/swp-matmul-bitext.ll75
-rw-r--r--test/CodeGen/Hexagon/swp-max.ll42
-rw-r--r--test/CodeGen/Hexagon/swp-multi-loops.ll75
-rw-r--r--test/CodeGen/Hexagon/swp-prolog-phi4.ll65
-rw-r--r--test/CodeGen/Hexagon/swp-vect-dotprod.ll41
-rw-r--r--test/CodeGen/Hexagon/swp-vmult.ll33
-rw-r--r--test/CodeGen/Hexagon/swp-vsum.ll29
-rw-r--r--test/CodeGen/Hexagon/tailcall_fastcc_ccc.ll22
-rw-r--r--test/CodeGen/Hexagon/tls_static.ll2
-rw-r--r--test/CodeGen/Hexagon/two-crash.ll23
-rw-r--r--test/CodeGen/Hexagon/v60-cur.ll2
-rw-r--r--test/CodeGen/Hexagon/v60-vsel1.ll69
-rw-r--r--test/CodeGen/Hexagon/v6vec-vprint.ll36
-rw-r--r--test/CodeGen/Hexagon/vassign-to-combine.ll56
-rw-r--r--test/CodeGen/Hexagon/vdmpy-halide-test.ll167
-rw-r--r--test/CodeGen/Hexagon/vect/vect-vsplatb.ll2
-rw-r--r--test/CodeGen/Hexagon/vect/vect-vsplath.ll2
-rw-r--r--test/CodeGen/Hexagon/vector-ext-load.ll10
-rw-r--r--test/CodeGen/Hexagon/vmpa-halide-test.ll145
-rw-r--r--test/CodeGen/Hexagon/vpack_eo.ll73
107 files changed, 5174 insertions, 56 deletions
diff --git a/test/CodeGen/Hexagon/SUnit-boundary-prob.ll b/test/CodeGen/Hexagon/SUnit-boundary-prob.ll
new file mode 100644
index 000000000000..9df178f9907c
--- /dev/null
+++ b/test/CodeGen/Hexagon/SUnit-boundary-prob.ll
@@ -0,0 +1,202 @@
+; RUN: llc -march=hexagon -O2 -mcpu=hexagonv60 < %s | FileCheck %s
+; This was aborting while processing SUnits.
+
+; CHECK: vmem
+
+source_filename = "bugpoint-output-bdb0052.bc"
+target datalayout = "e-m:e-p:32:32:32-a:0-n16:32-i64:64:64-i32:32:32-i16:16:16-i1:8:8-f32:32:32-f64:64:64-v32:32:32-v64:64:64-v512:512:512-v1024:1024:1024-v2048:2048:2048"
+target triple = "hexagon-unknown--elf"
+
+; Function Attrs: nounwind readnone
+declare <16 x i32> @llvm.hexagon.V6.lo(<32 x i32>) #0
+
+; Function Attrs: nounwind readnone
+declare <16 x i32> @llvm.hexagon.V6.hi(<32 x i32>) #0
+
+; Function Attrs: nounwind readnone
+declare <32 x i32> @llvm.hexagon.V6.vshuffvdd(<16 x i32>, <16 x i32>, i32) #0
+
+; Function Attrs: nounwind readnone
+declare <32 x i32> @llvm.hexagon.V6.vdealvdd(<16 x i32>, <16 x i32>, i32) #0
+
+; Function Attrs: nounwind readnone
+declare <32 x i32> @llvm.hexagon.V6.vcombine(<16 x i32>, <16 x i32>) #0
+
+; Function Attrs: nounwind readnone
+declare <16 x i32> @llvm.hexagon.V6.vshufeh(<16 x i32>, <16 x i32>) #0
+
+; Function Attrs: nounwind readnone
+declare <16 x i32> @llvm.hexagon.V6.vshufoh(<16 x i32>, <16 x i32>) #0
+
+; Function Attrs: nounwind readnone
+declare <32 x i32> @llvm.hexagon.V6.vmpyuhv(<16 x i32>, <16 x i32>) #0
+
+; Function Attrs: nounwind readnone
+declare <16 x i32> @llvm.hexagon.V6.vaslw.acc(<16 x i32>, <16 x i32>, i32) #0
+
+define void @__error_op_vmpy_v__uh_v__uh__1() #1 {
+entry:
+ %in_u16.host181 = load i16*, i16** undef, align 4
+ %in_u32.host182 = load i32*, i32** undef, align 4
+ br label %"for op_vmpy_v__uh_v__uh__1.s0.y"
+
+"for op_vmpy_v__uh_v__uh__1.s0.y": ; preds = %"end for op_vmpy_v__uh_v__uh__1.s0.x.x", %entry
+ %op_vmpy_v__uh_v__uh__1.s0.y = phi i32 [ 0, %entry ], [ %63, %"end for op_vmpy_v__uh_v__uh__1.s0.x.x" ]
+ %0 = mul nuw nsw i32 %op_vmpy_v__uh_v__uh__1.s0.y, 768
+ %1 = add nuw nsw i32 %0, 32
+ %2 = add nuw nsw i32 %0, 64
+ %3 = add nuw nsw i32 %0, 96
+ br label %"for op_vmpy_v__uh_v__uh__1.s0.x.x"
+
+"for op_vmpy_v__uh_v__uh__1.s0.x.x": ; preds = %"for op_vmpy_v__uh_v__uh__1.s0.x.x", %"for op_vmpy_v__uh_v__uh__1.s0.y"
+ %.phi210 = phi i32* [ %in_u32.host182, %"for op_vmpy_v__uh_v__uh__1.s0.y" ], [ %.inc211.3, %"for op_vmpy_v__uh_v__uh__1.s0.x.x" ]
+ %.phi213 = phi i16* [ %in_u16.host181, %"for op_vmpy_v__uh_v__uh__1.s0.y" ], [ %.inc214.3, %"for op_vmpy_v__uh_v__uh__1.s0.x.x" ]
+ %op_vmpy_v__uh_v__uh__1.s0.x.x = phi i32 [ 0, %"for op_vmpy_v__uh_v__uh__1.s0.y" ], [ %61, %"for op_vmpy_v__uh_v__uh__1.s0.x.x" ]
+ %4 = mul nuw nsw i32 %op_vmpy_v__uh_v__uh__1.s0.x.x, 32
+ %5 = bitcast i32* %.phi210 to <16 x i32>*
+ %6 = load <16 x i32>, <16 x i32>* %5, align 64, !tbaa !1
+ %7 = add nuw nsw i32 %4, 16
+ %8 = getelementptr inbounds i32, i32* %in_u32.host182, i32 %7
+ %9 = bitcast i32* %8 to <16 x i32>*
+ %10 = load <16 x i32>, <16 x i32>* %9, align 64, !tbaa !1
+ %11 = tail call <32 x i32> @llvm.hexagon.V6.vcombine(<16 x i32> %10, <16 x i32> %6)
+ %e.i = tail call <16 x i32> @llvm.hexagon.V6.lo(<32 x i32> %11) #2
+ %o.i = tail call <16 x i32> @llvm.hexagon.V6.hi(<32 x i32> %11) #2
+ %r.i = tail call <32 x i32> @llvm.hexagon.V6.vdealvdd(<16 x i32> %o.i, <16 x i32> %e.i, i32 -4) #2
+ %12 = bitcast i16* %.phi213 to <16 x i32>*
+ %13 = load <16 x i32>, <16 x i32>* %12, align 64, !tbaa !4
+ %a_lo.i = tail call <16 x i32> @llvm.hexagon.V6.lo(<32 x i32> %r.i) #2
+ %a_hi.i = tail call <16 x i32> @llvm.hexagon.V6.hi(<32 x i32> %r.i) #2
+ %a_e.i = tail call <16 x i32> @llvm.hexagon.V6.vshufeh(<16 x i32> %a_hi.i, <16 x i32> %a_lo.i) #2
+ %a_o.i = tail call <16 x i32> @llvm.hexagon.V6.vshufoh(<16 x i32> %a_hi.i, <16 x i32> %a_lo.i) #2
+ %ab_e.i = tail call <32 x i32> @llvm.hexagon.V6.vmpyuhv(<16 x i32> %a_e.i, <16 x i32> %13) #2
+ %ab_o.i = tail call <32 x i32> @llvm.hexagon.V6.vmpyuhv(<16 x i32> %a_o.i, <16 x i32> %13) #2
+ %a_lo.i.i = tail call <16 x i32> @llvm.hexagon.V6.lo(<32 x i32> %ab_e.i) #2
+ %l_lo.i.i = tail call <16 x i32> @llvm.hexagon.V6.lo(<32 x i32> %ab_o.i) #2
+ %s_lo.i.i = tail call <16 x i32> @llvm.hexagon.V6.vaslw.acc(<16 x i32> %a_lo.i.i, <16 x i32> %l_lo.i.i, i32 16) #2
+ %l_hi.i.i = tail call <16 x i32> @llvm.hexagon.V6.hi(<32 x i32> %ab_o.i) #2
+ %s_hi.i.i = tail call <16 x i32> @llvm.hexagon.V6.vaslw.acc(<16 x i32> undef, <16 x i32> %l_hi.i.i, i32 16) #2
+ %s.i.i = tail call <32 x i32> @llvm.hexagon.V6.vcombine(<16 x i32> %s_hi.i.i, <16 x i32> %s_lo.i.i) #2
+ %e.i189 = tail call <16 x i32> @llvm.hexagon.V6.lo(<32 x i32> %s.i.i) #2
+ %o.i190 = tail call <16 x i32> @llvm.hexagon.V6.hi(<32 x i32> %s.i.i) #2
+ %r.i191 = tail call <32 x i32> @llvm.hexagon.V6.vshuffvdd(<16 x i32> %o.i190, <16 x i32> %e.i189, i32 -4) #2
+ %14 = tail call <16 x i32> @llvm.hexagon.V6.lo(<32 x i32> %r.i191)
+ %15 = add nuw nsw i32 %4, %0
+ %16 = getelementptr inbounds i32, i32* undef, i32 %15
+ %17 = bitcast i32* %16 to <16 x i32>*
+ store <16 x i32> %14, <16 x i32>* %17, align 64, !tbaa !6
+ %18 = tail call <16 x i32> @llvm.hexagon.V6.hi(<32 x i32> %r.i191)
+ store <16 x i32> %18, <16 x i32>* undef, align 64, !tbaa !6
+ %.inc211 = getelementptr i32, i32* %.phi210, i32 32
+ %.inc214 = getelementptr i16, i16* %.phi213, i32 32
+ %19 = bitcast i32* %.inc211 to <16 x i32>*
+ %20 = load <16 x i32>, <16 x i32>* %19, align 64, !tbaa !1
+ %21 = add nuw nsw i32 %4, 48
+ %22 = getelementptr inbounds i32, i32* %in_u32.host182, i32 %21
+ %23 = bitcast i32* %22 to <16 x i32>*
+ %24 = load <16 x i32>, <16 x i32>* %23, align 64, !tbaa !1
+ %25 = tail call <32 x i32> @llvm.hexagon.V6.vcombine(<16 x i32> %24, <16 x i32> %20)
+ %e.i.1 = tail call <16 x i32> @llvm.hexagon.V6.lo(<32 x i32> %25) #2
+ %r.i.1 = tail call <32 x i32> @llvm.hexagon.V6.vdealvdd(<16 x i32> undef, <16 x i32> %e.i.1, i32 -4) #2
+ %26 = bitcast i16* %.inc214 to <16 x i32>*
+ %27 = load <16 x i32>, <16 x i32>* %26, align 64, !tbaa !4
+ %a_lo.i.1 = tail call <16 x i32> @llvm.hexagon.V6.lo(<32 x i32> %r.i.1) #2
+ %a_e.i.1 = tail call <16 x i32> @llvm.hexagon.V6.vshufeh(<16 x i32> undef, <16 x i32> %a_lo.i.1) #2
+ %a_o.i.1 = tail call <16 x i32> @llvm.hexagon.V6.vshufoh(<16 x i32> undef, <16 x i32> %a_lo.i.1) #2
+ %ab_e.i.1 = tail call <32 x i32> @llvm.hexagon.V6.vmpyuhv(<16 x i32> %a_e.i.1, <16 x i32> %27) #2
+ %ab_o.i.1 = tail call <32 x i32> @llvm.hexagon.V6.vmpyuhv(<16 x i32> %a_o.i.1, <16 x i32> %27) #2
+ %a_lo.i.i.1 = tail call <16 x i32> @llvm.hexagon.V6.lo(<32 x i32> %ab_e.i.1) #2
+ %s_lo.i.i.1 = tail call <16 x i32> @llvm.hexagon.V6.vaslw.acc(<16 x i32> %a_lo.i.i.1, <16 x i32> undef, i32 16) #2
+ %a_hi.i.i.1 = tail call <16 x i32> @llvm.hexagon.V6.hi(<32 x i32> %ab_e.i.1) #2
+ %l_hi.i.i.1 = tail call <16 x i32> @llvm.hexagon.V6.hi(<32 x i32> %ab_o.i.1) #2
+ %s_hi.i.i.1 = tail call <16 x i32> @llvm.hexagon.V6.vaslw.acc(<16 x i32> %a_hi.i.i.1, <16 x i32> %l_hi.i.i.1, i32 16) #2
+ %s.i.i.1 = tail call <32 x i32> @llvm.hexagon.V6.vcombine(<16 x i32> %s_hi.i.i.1, <16 x i32> %s_lo.i.i.1) #2
+ %e.i189.1 = tail call <16 x i32> @llvm.hexagon.V6.lo(<32 x i32> %s.i.i.1) #2
+ %o.i190.1 = tail call <16 x i32> @llvm.hexagon.V6.hi(<32 x i32> %s.i.i.1) #2
+ %r.i191.1 = tail call <32 x i32> @llvm.hexagon.V6.vshuffvdd(<16 x i32> %o.i190.1, <16 x i32> %e.i189.1, i32 -4) #2
+ %28 = tail call <16 x i32> @llvm.hexagon.V6.lo(<32 x i32> %r.i191.1)
+ %29 = add nuw nsw i32 %1, %4
+ %30 = getelementptr inbounds i32, i32* undef, i32 %29
+ %31 = bitcast i32* %30 to <16 x i32>*
+ store <16 x i32> %28, <16 x i32>* %31, align 64, !tbaa !6
+ %32 = tail call <16 x i32> @llvm.hexagon.V6.hi(<32 x i32> %r.i191.1)
+ %33 = add nuw nsw i32 %29, 16
+ %34 = getelementptr inbounds i32, i32* undef, i32 %33
+ %35 = bitcast i32* %34 to <16 x i32>*
+ store <16 x i32> %32, <16 x i32>* %35, align 64, !tbaa !6
+ %.inc211.1 = getelementptr i32, i32* %.phi210, i32 64
+ %.inc214.1 = getelementptr i16, i16* %.phi213, i32 64
+ %36 = bitcast i32* %.inc211.1 to <16 x i32>*
+ %37 = load <16 x i32>, <16 x i32>* %36, align 64, !tbaa !1
+ %38 = add nuw nsw i32 %4, 80
+ %39 = getelementptr inbounds i32, i32* %in_u32.host182, i32 %38
+ %40 = bitcast i32* %39 to <16 x i32>*
+ %41 = load <16 x i32>, <16 x i32>* %40, align 64, !tbaa !1
+ %42 = tail call <32 x i32> @llvm.hexagon.V6.vcombine(<16 x i32> %41, <16 x i32> %37)
+ %e.i.2 = tail call <16 x i32> @llvm.hexagon.V6.lo(<32 x i32> %42) #2
+ %o.i.2 = tail call <16 x i32> @llvm.hexagon.V6.hi(<32 x i32> %42) #2
+ %r.i.2 = tail call <32 x i32> @llvm.hexagon.V6.vdealvdd(<16 x i32> %o.i.2, <16 x i32> %e.i.2, i32 -4) #2
+ %43 = bitcast i16* %.inc214.1 to <16 x i32>*
+ %44 = load <16 x i32>, <16 x i32>* %43, align 64, !tbaa !4
+ %a_lo.i.2 = tail call <16 x i32> @llvm.hexagon.V6.lo(<32 x i32> %r.i.2) #2
+ %a_hi.i.2 = tail call <16 x i32> @llvm.hexagon.V6.hi(<32 x i32> %r.i.2) #2
+ %a_e.i.2 = tail call <16 x i32> @llvm.hexagon.V6.vshufeh(<16 x i32> %a_hi.i.2, <16 x i32> %a_lo.i.2) #2
+ %a_o.i.2 = tail call <16 x i32> @llvm.hexagon.V6.vshufoh(<16 x i32> %a_hi.i.2, <16 x i32> %a_lo.i.2) #2
+ %ab_e.i.2 = tail call <32 x i32> @llvm.hexagon.V6.vmpyuhv(<16 x i32> %a_e.i.2, <16 x i32> %44) #2
+ %ab_o.i.2 = tail call <32 x i32> @llvm.hexagon.V6.vmpyuhv(<16 x i32> %a_o.i.2, <16 x i32> %44) #2
+ %l_lo.i.i.2 = tail call <16 x i32> @llvm.hexagon.V6.lo(<32 x i32> %ab_o.i.2) #2
+ %s_lo.i.i.2 = tail call <16 x i32> @llvm.hexagon.V6.vaslw.acc(<16 x i32> undef, <16 x i32> %l_lo.i.i.2, i32 16) #2
+ %a_hi.i.i.2 = tail call <16 x i32> @llvm.hexagon.V6.hi(<32 x i32> %ab_e.i.2) #2
+ %l_hi.i.i.2 = tail call <16 x i32> @llvm.hexagon.V6.hi(<32 x i32> %ab_o.i.2) #2
+ %s_hi.i.i.2 = tail call <16 x i32> @llvm.hexagon.V6.vaslw.acc(<16 x i32> %a_hi.i.i.2, <16 x i32> %l_hi.i.i.2, i32 16) #2
+ %s.i.i.2 = tail call <32 x i32> @llvm.hexagon.V6.vcombine(<16 x i32> %s_hi.i.i.2, <16 x i32> %s_lo.i.i.2) #2
+ %e.i189.2 = tail call <16 x i32> @llvm.hexagon.V6.lo(<32 x i32> %s.i.i.2) #2
+ %o.i190.2 = tail call <16 x i32> @llvm.hexagon.V6.hi(<32 x i32> %s.i.i.2) #2
+ %r.i191.2 = tail call <32 x i32> @llvm.hexagon.V6.vshuffvdd(<16 x i32> %o.i190.2, <16 x i32> %e.i189.2, i32 -4) #2
+ %45 = tail call <16 x i32> @llvm.hexagon.V6.lo(<32 x i32> %r.i191.2)
+ %46 = add nuw nsw i32 %2, %4
+ %47 = getelementptr inbounds i32, i32* undef, i32 %46
+ %48 = bitcast i32* %47 to <16 x i32>*
+ store <16 x i32> %45, <16 x i32>* %48, align 64, !tbaa !6
+ %49 = tail call <16 x i32> @llvm.hexagon.V6.hi(<32 x i32> %r.i191.2)
+ %50 = add nuw nsw i32 %46, 16
+ %51 = getelementptr inbounds i32, i32* undef, i32 %50
+ %52 = bitcast i32* %51 to <16 x i32>*
+ store <16 x i32> %49, <16 x i32>* %52, align 64, !tbaa !6
+ %e.i189.3 = tail call <16 x i32> @llvm.hexagon.V6.lo(<32 x i32> undef) #2
+ %r.i191.3 = tail call <32 x i32> @llvm.hexagon.V6.vshuffvdd(<16 x i32> undef, <16 x i32> %e.i189.3, i32 -4) #2
+ %53 = tail call <16 x i32> @llvm.hexagon.V6.lo(<32 x i32> %r.i191.3)
+ %54 = add nuw nsw i32 %3, %4
+ %55 = getelementptr inbounds i32, i32* undef, i32 %54
+ %56 = bitcast i32* %55 to <16 x i32>*
+ store <16 x i32> %53, <16 x i32>* %56, align 64, !tbaa !6
+ %57 = tail call <16 x i32> @llvm.hexagon.V6.hi(<32 x i32> %r.i191.3)
+ %58 = add nuw nsw i32 %54, 16
+ %59 = getelementptr inbounds i32, i32* undef, i32 %58
+ %60 = bitcast i32* %59 to <16 x i32>*
+ store <16 x i32> %57, <16 x i32>* %60, align 64, !tbaa !6
+ %61 = add nuw nsw i32 %op_vmpy_v__uh_v__uh__1.s0.x.x, 4
+ %62 = icmp eq i32 %61, 24
+ %.inc211.3 = getelementptr i32, i32* %.phi210, i32 128
+ %.inc214.3 = getelementptr i16, i16* %.phi213, i32 128
+ br i1 %62, label %"end for op_vmpy_v__uh_v__uh__1.s0.x.x", label %"for op_vmpy_v__uh_v__uh__1.s0.x.x"
+
+"end for op_vmpy_v__uh_v__uh__1.s0.x.x": ; preds = %"for op_vmpy_v__uh_v__uh__1.s0.x.x"
+ %63 = add nuw nsw i32 %op_vmpy_v__uh_v__uh__1.s0.y, 1
+ br label %"for op_vmpy_v__uh_v__uh__1.s0.y"
+}
+
+attributes #0 = { nounwind readnone }
+attributes #1 = { "target-cpu"="hexagonv60" "target-features"="+hvx" }
+attributes #2 = { nounwind }
+
+!llvm.module.flags = !{!0}
+
+!0 = !{i32 2, !"halide_mattrs", !"+hvx"}
+!1 = !{!2, !2, i64 0}
+!2 = !{!"in_u32", !3}
+!3 = !{!"Halide buffer"}
+!4 = !{!5, !5, i64 0}
+!5 = !{!"in_u16", !3}
+!6 = !{!7, !7, i64 0}
+!7 = !{!"op_vmpy_v__uh_v__uh__1", !3}
diff --git a/test/CodeGen/Hexagon/addh-sext-trunc.ll b/test/CodeGen/Hexagon/addh-sext-trunc.ll
index 094932933fbc..7f219944436b 100644
--- a/test/CodeGen/Hexagon/addh-sext-trunc.ll
+++ b/test/CodeGen/Hexagon/addh-sext-trunc.ll
@@ -4,37 +4,17 @@
target datalayout = "e-p:32:32:32-i64:64:64-i32:32:32-i16:16:16-i1:32:32-f64:64:64-f32:32:32-v64:64:64-v32:32:32-a0:0-n16:32"
target triple = "hexagon-unknown-none"
-%struct.aDataType = type { i16, i16, i16, i16, i16, i16*, i16*, i16*, i8*, i16*, i16*, i16*, i8* }
-define i8* @a_get_score(%struct.aDataType* nocapture %pData, i16 signext %gmmModelIndex, i16* nocapture %pGmmScoreL16Q4) #0 {
-entry:
- %numSubVector = getelementptr inbounds %struct.aDataType, %struct.aDataType* %pData, i32 0, i32 3
- %0 = load i16, i16* %numSubVector, align 2, !tbaa !0
- %and = and i16 %0, -4
- %b = getelementptr inbounds %struct.aDataType, %struct.aDataType* %pData, i32 0, i32 8
- %1 = load i8*, i8** %b, align 4, !tbaa !3
+define i32 @foo(i16 %a, i32 %b) #0 {
+ %and = and i16 %a, -4
%conv3 = sext i16 %and to i32
- %cmp21 = icmp sgt i16 %and, 0
- br i1 %cmp21, label %for.inc.preheader, label %for.end
-
-for.inc.preheader: ; preds = %entry
- br label %for.inc
-
-for.inc: ; preds = %for.inc.preheader, %for.inc
- %j.022 = phi i32 [ %phitmp, %for.inc ], [ 0, %for.inc.preheader ]
- %add13 = mul i32 %j.022, 65536
+ %add13 = mul i32 %b, 65536
%sext = add i32 %add13, 262144
%phitmp = ashr exact i32 %sext, 16
- %cmp = icmp slt i32 %phitmp, %conv3
- br i1 %cmp, label %for.inc, label %for.end.loopexit
-
-for.end.loopexit: ; preds = %for.inc
- br label %for.end
-
-for.end: ; preds = %for.end.loopexit, %entry
- ret i8* %1
+ ret i32 %phitmp
}
+
attributes #0 = { nounwind readonly "less-precise-fpmad"="false" "no-frame-pointer-elim"="false" "no-frame-pointer-elim-non-leaf"="true" "no-infs-fp-math"="false" "no-nans-fp-math"="false" "unsafe-fp-math"="false" "use-soft-float"="false" }
!0 = !{!"short", !1}
diff --git a/test/CodeGen/Hexagon/addr-calc-opt.ll b/test/CodeGen/Hexagon/addr-calc-opt.ll
new file mode 100644
index 000000000000..9bd010d52364
--- /dev/null
+++ b/test/CodeGen/Hexagon/addr-calc-opt.ll
@@ -0,0 +1,54 @@
+; RUN: llc -march=hexagon -mcpu=hexagonv5 < %s | FileCheck %s
+;
+; Test whether we can produce minimal code for this complex address
+; calculation.
+;
+
+; CHECK: r0 = memub(r{{[0-9]+}}<<#3{{ *}}+{{ *}}##the_global+516)
+
+%0 = type { [3 x %1] }
+%1 = type { %2, i8, i8, i8, i8, i8, [4 x i8], i8, [10 x i8], [10 x i8], [10 x i8], i8, [3 x %4], i16, i16, i16, i16, i32, i8, [4 x i8], i8, i8, i8, i8, %5, i8, i8, i8, i8, i8, i16, i8, i8, i8, i16, i16, i8, i8, [2 x i8], [2 x i8], i8, i8, i8, i8, i8, i16, i16, i8, i8, i8, i8, i8, i8, %9, i8, [6 x [2 x i8]], i16, i32, %10, [28 x i8], [4 x %17] }
+%2 = type { %3 }
+%3 = type { i8, i8, i8, i8, i8, i16, i16, i16, i16, i16 }
+%4 = type { i16, i16 }
+%5 = type { [10 x %6] }
+%6 = type { [2 x %7] }
+%7 = type { i8, [2 x %8] }
+%8 = type { [4 x i8] }
+%9 = type { i8 }
+%10 = type { %11, %13 }
+%11 = type { [2 x [2 x i8]], [2 x [8 x %12]], [6 x i16], [6 x i16] }
+%12 = type { i8, i8 }
+%13 = type { [4 x %12], [4 x %12], [2 x [4 x %14]], [6 x i16] }
+%14 = type { %15, %16 }
+%15 = type { i8, i8 }
+%16 = type { i8, i8 }
+%17 = type { i8, i8, %1*, i16, i16, i16, i64, i32, i32, %18, i8, %21, i8, [2 x i16], i8, i8, i8, i8, i8, i8, i8, i8, i8, i8, i8, i8, i8, i8, i8, i8, i8, i8, i8, i8, i8, i8, i16, i16, i16, i8, i8, i8, i16, i16, [2 x i16], i16, [2 x i32], [2 x i16], [2 x i16], i8, i8, [6 x %23], i8, i8, i8, %24, %25, %26, %28 }
+%18 = type { %19, [10 x %20] }
+%19 = type { i32 }
+%20 = type { [2 x i8], [2 x i8], i8, i8, i8, i8 }
+%21 = type { i8, i8, i8, [8 x %22] }
+%22 = type { i8, i8, i8, i32 }
+%23 = type { i32, i16, i16, [2 x i16], [2 x i16], [2 x i16], i32 }
+%24 = type { [2 x i32], [2 x i64*], [2 x i64*], [2 x i64*], [2 x i32], [2 x i32], i32 }
+%25 = type { [2 x i32], [2 x i32], [2 x i32] }
+%26 = type { i8, i8, i8, i16, i16, %27, i32, i32, i32, i16 }
+%27 = type { i64 }
+%28 = type { %29, %31, [24 x i8] }
+%29 = type { [2 x %30], [16 x i32] }
+%30 = type { [16 x i32], [8 x i32], [16 x i32], [64 x i32], [2 x i32], i64, i32, i32, i32, i32 }
+%31 = type { [2 x %32] }
+%32 = type { [4 x %33], i32 }
+%33 = type { i32, i32 }
+
+@the_global = external global %0
+
+; Function Attrs: nounwind optsize readonly ssp
+define zeroext i8 @myFun(i8 zeroext, i8 zeroext) {
+ %3 = zext i8 %1 to i32
+ %4 = zext i8 %0 to i32
+ %5 = getelementptr inbounds %0, %0* @the_global, i32 0, i32 0, i32 %4, i32 60, i32 0, i32 9, i32 1, i32 %3, i32 0, i32 0
+ %6 = load i8, i8* %5, align 4
+ ret i8 %6
+}
+
diff --git a/test/CodeGen/Hexagon/anti-dep-partial.mir b/test/CodeGen/Hexagon/anti-dep-partial.mir
new file mode 100644
index 000000000000..09bc49c508a2
--- /dev/null
+++ b/test/CodeGen/Hexagon/anti-dep-partial.mir
@@ -0,0 +1,34 @@
+# RUN: llc -march=hexagon -post-RA-scheduler -run-pass post-RA-sched %s -o - | FileCheck %s
+
+--- |
+ declare void @check(i64, i32, i32, i64)
+ define void @foo() {
+ ret void
+ }
+...
+
+---
+name: foo
+tracksRegLiveness: true
+body: |
+ bb.0:
+ successors:
+ liveins: %r0, %r1, %d1, %d2, %r16, %r17, %r19, %r22, %r23
+ %r2 = A2_add %r23, killed %r17
+ %r6 = M2_mpyi %r16, %r16
+ %r22 = M2_accii %r22, killed %r2, 2
+ %r7 = A2_tfrsi 12345678
+ %r3 = A2_tfr killed %r16
+ %d2 = A2_tfrp killed %d0
+ %r2 = L2_loadri_io %r29, 28
+ %r2 = M2_mpyi killed %r6, killed %r2
+ %r23 = S2_asr_i_r %r22, 31
+ S2_storeri_io killed %r29, 0, killed %r7
+ ; The anti-dependency on r23 between the first A2_add and the
+ ; S2_asr_i_r was causing d11 to be renamed, while r22 remained
+ ; unchanged. Check that the renaming of d11 does not happen.
+ ; CHECK: d11
+ %d0 = A2_tfrp killed %d11
+ J2_call @check, implicit-def %d0, implicit-def %d1, implicit-def %d2, implicit %d0, implicit %d1, implicit %d2
+...
+
diff --git a/test/CodeGen/Hexagon/bit-gen-rseq.ll b/test/CodeGen/Hexagon/bit-gen-rseq.ll
new file mode 100644
index 000000000000..08d4b7877159
--- /dev/null
+++ b/test/CodeGen/Hexagon/bit-gen-rseq.ll
@@ -0,0 +1,43 @@
+; RUN: llc -march=hexagon -disable-hsdr -hexagon-subreg-liveness < %s | FileCheck %s
+; Check that we don't generate any bitwise operations.
+
+; CHECK-NOT: = or(
+; CHECK-NOT: = and(
+
+target triple = "hexagon"
+
+define i32 @fred(i32* nocapture readonly %p, i32 %n) #0 {
+entry:
+ %t.sroa.0.048 = load i32, i32* %p, align 4
+ %cmp49 = icmp ugt i32 %n, 1
+ br i1 %cmp49, label %for.body, label %for.end
+
+for.body: ; preds = %entry, %for.body
+ %t.sroa.0.052 = phi i32 [ %t.sroa.0.0, %for.body ], [ %t.sroa.0.048, %entry ]
+ %t.sroa.11.051 = phi i64 [ %t.sroa.11.0.extract.shift, %for.body ], [ 0, %entry ]
+ %i.050 = phi i32 [ %inc, %for.body ], [ 1, %entry ]
+ %t.sroa.0.0.insert.ext = zext i32 %t.sroa.0.052 to i64
+ %t.sroa.0.0.insert.insert = or i64 %t.sroa.0.0.insert.ext, %t.sroa.11.051
+ %0 = tail call i64 @llvm.hexagon.A2.addp(i64 %t.sroa.0.0.insert.insert, i64 %t.sroa.0.0.insert.insert)
+ %t.sroa.11.0.extract.shift = and i64 %0, -4294967296
+ %arrayidx4 = getelementptr inbounds i32, i32* %p, i32 %i.050
+ %inc = add nuw i32 %i.050, 1
+ %t.sroa.0.0 = load i32, i32* %arrayidx4, align 4
+ %exitcond = icmp eq i32 %inc, %n
+ br i1 %exitcond, label %for.end, label %for.body
+
+for.end: ; preds = %for.body, %entry
+ %t.sroa.0.0.lcssa = phi i32 [ %t.sroa.0.048, %entry ], [ %t.sroa.0.0, %for.body ]
+ %t.sroa.11.0.lcssa = phi i64 [ 0, %entry ], [ %t.sroa.11.0.extract.shift, %for.body ]
+ %t.sroa.0.0.insert.ext17 = zext i32 %t.sroa.0.0.lcssa to i64
+ %t.sroa.0.0.insert.insert19 = or i64 %t.sroa.0.0.insert.ext17, %t.sroa.11.0.lcssa
+ %1 = tail call i64 @llvm.hexagon.A2.addp(i64 %t.sroa.0.0.insert.insert19, i64 %t.sroa.0.0.insert.insert19)
+ %t.sroa.11.0.extract.shift41 = lshr i64 %1, 32
+ %t.sroa.11.0.extract.trunc42 = trunc i64 %t.sroa.11.0.extract.shift41 to i32
+ ret i32 %t.sroa.11.0.extract.trunc42
+}
+
+declare i64 @llvm.hexagon.A2.addp(i64, i64) #1
+
+attributes #0 = { norecurse nounwind readonly }
+attributes #1 = { nounwind readnone }
diff --git a/test/CodeGen/Hexagon/bit-loop-rc-mismatch.ll b/test/CodeGen/Hexagon/bit-loop-rc-mismatch.ll
new file mode 100644
index 000000000000..db57998aeb66
--- /dev/null
+++ b/test/CodeGen/Hexagon/bit-loop-rc-mismatch.ll
@@ -0,0 +1,30 @@
+; RUN: llc -march=hexagon < %s
+; REQUIRES: asserts
+
+target datalayout = "e-m:e-p:32:32:32-a:0-n16:32-i64:64:64-i32:32:32-i16:16:16-i1:8:8-f32:32:32-f64:64:64-v32:32:32-v64:64:64-v512:512:512-v1024:1024:1024-v2048:2048:2048"
+target triple = "hexagon"
+
+define weak_odr hidden i32 @fred(i32* %this, i32* nocapture readonly dereferenceable(4) %__k) #0 align 2 {
+entry:
+ %call = tail call i64 @danny(i32* %this, i32* nonnull dereferenceable(4) %__k) #2
+ %__p.sroa.0.0.extract.trunc = trunc i64 %call to i32
+ br i1 undef, label %for.end, label %for.body
+
+for.body: ; preds = %for.body, %entry
+ %__p.sroa.0.018 = phi i32 [ %call8, %for.body ], [ %__p.sroa.0.0.extract.trunc, %entry ]
+ %call8 = tail call i32 @sammy(i32* %this, i32 %__p.sroa.0.018) #2
+ %0 = inttoptr i32 %call8 to i32*
+ %lnot.i = icmp eq i32* %0, undef
+ br i1 %lnot.i, label %for.end, label %for.body
+
+for.end: ; preds = %for.body, %entry
+ ret i32 0
+}
+
+declare hidden i64 @danny(i32*, i32* nocapture readonly dereferenceable(4)) #1 align 2
+declare hidden i32 @sammy(i32* nocapture, i32) #0 align 2
+
+attributes #0 = { nounwind optsize "disable-tail-calls"="false" "less-precise-fpmad"="false" "no-frame-pointer-elim"="true" "no-frame-pointer-elim-non-leaf" "no-infs-fp-math"="false" "no-nans-fp-math"="false" "stack-protector-buffer-size"="8" "target-cpu"="hexagonv60" "target-features"="+hvx,-hvx-double" "unsafe-fp-math"="false" "use-soft-float"="false" }
+attributes #1 = { nounwind optsize readonly "disable-tail-calls"="false" "less-precise-fpmad"="false" "no-frame-pointer-elim"="true" "no-frame-pointer-elim-non-leaf" "no-infs-fp-math"="false" "no-nans-fp-math"="false" "stack-protector-buffer-size"="8" "target-cpu"="hexagonv60" "target-features"="+hvx,-hvx-double" "unsafe-fp-math"="false" "use-soft-float"="false" }
+attributes #2 = { optsize }
+
diff --git a/test/CodeGen/Hexagon/bit-rie.ll b/test/CodeGen/Hexagon/bit-rie.ll
new file mode 100644
index 000000000000..6bd0558f580c
--- /dev/null
+++ b/test/CodeGen/Hexagon/bit-rie.ll
@@ -0,0 +1,208 @@
+; RUN: llc -march=hexagon < %s | FileCheck %s
+; CHECK-LABEL: LBB0{{.*}}if.end
+; CHECK: r[[REG:[0-9]+]] = zxth
+; CHECK: lsr(r[[REG]],
+
+target triple = "hexagon"
+
+@g0 = external constant [146 x i16], align 8
+@g1 = external constant [0 x i16], align 2
+
+define void @fred(i32* nocapture readonly %p0, i16 signext %p1, i16* nocapture %p2, i16 signext %p3, i16 signext %p4, i16 signext %p5) #0 {
+entry:
+ %conv = sext i16 %p1 to i32
+ %0 = tail call i32 @llvm.hexagon.S2.asl.r.r.sat(i32 %conv, i32 1)
+ %1 = tail call i32 @llvm.hexagon.A2.sath(i32 %0)
+ %2 = tail call i32 @llvm.hexagon.A2.addh.l16.sat.ll(i32 3, i32 %1)
+ %conv3 = sext i16 %p4 to i32
+ %cmp144 = icmp sgt i16 %p4, 0
+ br i1 %cmp144, label %for.body, label %for.end
+
+for.body: ; preds = %entry, %for.body
+ %arrayidx.phi = phi i32* [ %arrayidx.inc, %for.body ], [ %p0, %entry ]
+ %i.0146.apmt = phi i32 [ %inc.apmt, %for.body ], [ 0, %entry ]
+ %L_temp1.0145 = phi i32 [ %5, %for.body ], [ 1, %entry ]
+ %3 = load i32, i32* %arrayidx.phi, align 4, !tbaa !1
+ %4 = tail call i32 @llvm.hexagon.A2.abssat(i32 %3)
+ %5 = tail call i32 @llvm.hexagon.A2.max(i32 %L_temp1.0145, i32 %4)
+ %inc.apmt = add nuw nsw i32 %i.0146.apmt, 1
+ %exitcond151 = icmp eq i32 %inc.apmt, %conv3
+ %arrayidx.inc = getelementptr i32, i32* %arrayidx.phi, i32 1
+ br i1 %exitcond151, label %for.end, label %for.body, !llvm.loop !5
+
+for.end: ; preds = %for.body, %entry
+ %L_temp1.0.lcssa = phi i32 [ 1, %entry ], [ %5, %for.body ]
+ %6 = tail call i32 @llvm.hexagon.S2.clbnorm(i32 %L_temp1.0.lcssa)
+ %arrayidx6 = getelementptr inbounds [146 x i16], [146 x i16]* @g0, i32 0, i32 %conv3
+ %7 = load i16, i16* %arrayidx6, align 2, !tbaa !7
+ %conv7 = sext i16 %7 to i32
+ %8 = tail call i32 @llvm.hexagon.A2.subh.l16.sat.ll(i32 %6, i32 %conv7)
+ br i1 %cmp144, label %for.body14.lr.ph, label %for.end29
+
+for.body14.lr.ph: ; preds = %for.end
+ %sext132 = shl i32 %8, 16
+ %conv17 = ashr exact i32 %sext132, 16
+ br label %for.body14
+
+for.body14: ; preds = %for.body14, %for.body14.lr.ph
+ %arrayidx16.phi = phi i32* [ %p0, %for.body14.lr.ph ], [ %arrayidx16.inc, %for.body14 ]
+ %i.1143.apmt = phi i32 [ 0, %for.body14.lr.ph ], [ %inc28.apmt, %for.body14 ]
+ %L_temp.0142 = phi i32 [ 0, %for.body14.lr.ph ], [ %12, %for.body14 ]
+ %9 = load i32, i32* %arrayidx16.phi, align 4, !tbaa !1
+ %10 = tail call i32 @llvm.hexagon.S2.asl.r.r.sat(i32 %9, i32 %conv17)
+ %11 = tail call i32 @llvm.hexagon.A2.asrh(i32 %10)
+ %sext133 = shl i32 %11, 16
+ %conv23 = ashr exact i32 %sext133, 16
+ %12 = tail call i32 @llvm.hexagon.M2.mpy.acc.sat.ll.s0(i32 %L_temp.0142, i32 %conv23, i32 %conv23)
+ %inc28.apmt = add nuw nsw i32 %i.1143.apmt, 1
+ %exitcond = icmp eq i32 %inc28.apmt, %conv3
+ %arrayidx16.inc = getelementptr i32, i32* %arrayidx16.phi, i32 1
+ br i1 %exitcond, label %for.end29, label %for.body14
+
+for.end29: ; preds = %for.body14, %for.end
+ %L_temp.0.lcssa = phi i32 [ 0, %for.end ], [ %12, %for.body14 ]
+ %13 = tail call i32 @llvm.hexagon.A2.subh.l16.sat.ll(i32 %conv3, i32 1)
+ %cmp31 = icmp sgt i32 %13, 0
+ br i1 %cmp31, label %if.then, label %if.end
+
+if.then: ; preds = %for.end29
+ %arrayidx34 = getelementptr inbounds [0 x i16], [0 x i16]* @g1, i32 0, i32 %conv3
+ %14 = load i16, i16* %arrayidx34, align 2, !tbaa !7
+ %cmp.i = icmp eq i32 %L_temp.0.lcssa, -2147483648
+ %cmp1.i = icmp eq i16 %14, -32768
+ %or.cond.i = and i1 %cmp.i, %cmp1.i
+ br i1 %or.cond.i, label %if.end, label %if.else.i
+
+if.else.i: ; preds = %if.then
+ %conv3.i = sext i16 %14 to i32
+ %15 = tail call i32 @llvm.hexagon.M2.hmmpyl.s1(i32 %L_temp.0.lcssa, i32 %conv3.i) #2
+ %16 = tail call i64 @llvm.hexagon.M2.mpyd.ll.s1(i32 %conv3.i, i32 %L_temp.0.lcssa) #2
+ %conv5.i = trunc i64 %16 to i32
+ %phitmp = and i32 %conv5.i, 65535
+ br label %if.end
+
+if.end: ; preds = %if.else.i, %if.then, %for.end29
+ %L_temp.2 = phi i32 [ %L_temp.0.lcssa, %for.end29 ], [ %15, %if.else.i ], [ 2147483647, %if.then ]
+ %lsb.0 = phi i32 [ 0, %for.end29 ], [ %phitmp, %if.else.i ], [ 65535, %if.then ]
+ %sext = shl i32 %8, 16
+ %conv35 = ashr exact i32 %sext, 16
+ %17 = tail call i32 @llvm.hexagon.A2.subh.l16.sat.ll(i32 %conv35, i32 16)
+ %18 = tail call i32 @llvm.hexagon.S2.asl.r.r.sat(i32 %17, i32 1)
+ %19 = tail call i32 @llvm.hexagon.A2.sath(i32 %18)
+ %20 = tail call i32 @llvm.hexagon.S2.clbnorm(i32 %L_temp.2)
+ %sext123 = shl i32 %20, 16
+ %conv38 = ashr exact i32 %sext123, 16
+ %sext124 = shl i32 %19, 16
+ %conv39 = ashr exact i32 %sext124, 16
+ %21 = tail call i32 @llvm.hexagon.A2.addh.l16.sat.ll(i32 %conv38, i32 %conv39)
+ %22 = tail call i32 @llvm.hexagon.S2.asl.r.r.sat(i32 %L_temp.2, i32 %conv38)
+ %23 = tail call i32 @llvm.hexagon.A2.zxth(i32 %lsb.0)
+ %24 = tail call i32 @llvm.hexagon.A2.subh.l16.sat.ll(i32 16, i32 %conv38)
+ %25 = tail call i32 @llvm.hexagon.S2.lsr.r.r(i32 %23, i32 %24)
+ %sext125 = shl i32 %25, 16
+ %conv45 = ashr exact i32 %sext125, 16
+ %26 = tail call i32 @llvm.hexagon.A2.addsat(i32 %22, i32 %conv45)
+ %sext126 = shl i32 %2, 16
+ %conv46 = ashr exact i32 %sext126, 16
+ %sext127 = shl i32 %21, 16
+ %conv47 = ashr exact i32 %sext127, 16
+ %27 = tail call i32 @llvm.hexagon.A2.addh.l16.sat.ll(i32 %conv46, i32 %conv47)
+ %sext128 = shl i32 %27, 16
+ %conv49 = ashr exact i32 %sext128, 16
+ %cmp50 = icmp sgt i32 %sext128, 327679
+ %tobool = icmp eq i16 %p5, 0
+ %or.cond = or i1 %tobool, %cmp50
+ br i1 %or.cond, label %if.else68, label %if.then53
+
+if.then53: ; preds = %if.end
+ %28 = tail call i32 @llvm.hexagon.S2.asl.r.r.sat(i32 %conv49, i32 1)
+ %29 = tail call i32 @llvm.hexagon.A2.sath(i32 %28)
+ %30 = tail call i32 @llvm.hexagon.A2.subsat(i32 %26, i32 1276901417)
+ %cmp56 = icmp slt i32 %30, 0
+ br i1 %cmp56, label %if.then58, label %if.else
+
+if.then58: ; preds = %if.then53
+ %sext131 = shl i32 %29, 16
+ %conv59 = ashr exact i32 %sext131, 16
+ %31 = tail call i32 @llvm.hexagon.A2.addh.l16.sat.ll(i32 %conv59, i32 2)
+ br label %if.end80
+
+if.else: ; preds = %if.then53
+ %32 = tail call i32 @llvm.hexagon.A2.subsat(i32 %26, i32 1805811301)
+ %cmp61 = icmp slt i32 %32, 0
+ br i1 %cmp61, label %if.then63, label %if.end80
+
+if.then63: ; preds = %if.else
+ %sext130 = shl i32 %29, 16
+ %conv64 = ashr exact i32 %sext130, 16
+ %33 = tail call i32 @llvm.hexagon.A2.addh.l16.sat.ll(i32 %conv64, i32 1)
+ br label %if.end80
+
+if.else68: ; preds = %if.end
+ %34 = tail call i32 @llvm.hexagon.A2.subsat(i32 %26, i32 1518500250)
+ %cmp69 = icmp slt i32 %34, 0
+ br i1 %cmp69, label %if.then71, label %if.end74
+
+if.then71: ; preds = %if.else68
+ %35 = tail call i32 @llvm.hexagon.A2.addh.l16.sat.ll(i32 %conv49, i32 1)
+ br label %if.end74
+
+if.end74: ; preds = %if.then71, %if.else68
+ %m.0.in = phi i32 [ %35, %if.then71 ], [ %27, %if.else68 ]
+ br i1 %tobool, label %if.end80, label %if.then76
+
+if.then76: ; preds = %if.end74
+ %sext129 = shl i32 %m.0.in, 16
+ %conv77 = ashr exact i32 %sext129, 16
+ %36 = tail call i32 @llvm.hexagon.A2.addh.l16.sat.ll(i32 %conv77, i32 5)
+ br label %if.end80
+
+if.end80: ; preds = %if.end74, %if.then76, %if.then58, %if.then63, %if.else
+ %m.1.in = phi i32 [ %31, %if.then58 ], [ %33, %if.then63 ], [ %29, %if.else ], [ %36, %if.then76 ], [ %m.0.in, %if.end74 ]
+ %m.1 = trunc i32 %m.1.in to i16
+ %cmp.i135 = icmp slt i16 %m.1, 0
+ %var_out.0.i136 = select i1 %cmp.i135, i16 0, i16 %m.1
+ %conv81 = sext i16 %p3 to i32
+ %37 = tail call i32 @llvm.hexagon.A2.subh.l16.sat.ll(i32 %conv81, i32 1)
+ %conv82 = trunc i32 %37 to i16
+ %cmp.i134 = icmp sgt i16 %var_out.0.i136, %conv82
+ %var_out.0.i = select i1 %cmp.i134, i16 %conv82, i16 %var_out.0.i136
+ store i16 %var_out.0.i, i16* %p2, align 2, !tbaa !7
+ ret void
+}
+
+declare i32 @llvm.hexagon.A2.abssat(i32) #2
+declare i32 @llvm.hexagon.A2.addh.l16.sat.ll(i32, i32) #2
+declare i32 @llvm.hexagon.A2.addsat(i32, i32) #2
+declare i32 @llvm.hexagon.A2.asrh(i32) #2
+declare i32 @llvm.hexagon.A2.max(i32, i32) #2
+declare i32 @llvm.hexagon.A2.sath(i32) #2
+declare i32 @llvm.hexagon.A2.subh.l16.sat.ll(i32, i32) #2
+declare i32 @llvm.hexagon.A2.subsat(i32, i32) #2
+declare i32 @llvm.hexagon.A2.zxth(i32) #2
+declare i32 @llvm.hexagon.M2.hmmpyl.s1(i32, i32) #2
+declare i32 @llvm.hexagon.M2.mpy.acc.sat.ll.s0(i32, i32, i32) #2
+declare i32 @llvm.hexagon.S2.asl.r.r.sat(i32, i32) #2
+declare i32 @llvm.hexagon.S2.asr.r.r.sat(i32, i32) #2
+declare i32 @llvm.hexagon.S2.clbnorm(i32) #2
+declare i32 @llvm.hexagon.S2.lsr.r.r(i32, i32) #2
+declare i64 @llvm.hexagon.M2.mpyd.ll.s1(i32, i32) #2
+declare void @llvm.lifetime.end(i64, i8* nocapture) #1
+declare void @llvm.lifetime.start(i64, i8* nocapture) #1
+
+attributes #0 = { norecurse nounwind "target-cpu"="hexagonv60" "target-features"="+hvx,-hvx-double" "unsafe-fp-math"="false" "use-soft-float"="false" }
+attributes #1 = { argmemonly nounwind }
+attributes #2 = { nounwind readnone }
+
+
+!1 = !{!2, !2, i64 0}
+!2 = !{!"int", !3, i64 0}
+!3 = !{!"omnipotent char", !4, i64 0}
+!4 = !{!"Simple C/C++ TBAA"}
+!5 = distinct !{!5, !6}
+!6 = !{!"llvm.loop.threadify", i32 81508608}
+!7 = !{!8, !8, i64 0}
+!8 = !{!"short", !3, i64 0}
+!9 = distinct !{!9, !10}
+!10 = !{!"llvm.loop.threadify", i32 1441813}
+!11 = distinct !{!11, !10}
diff --git a/test/CodeGen/Hexagon/bit-skip-byval.ll b/test/CodeGen/Hexagon/bit-skip-byval.ll
new file mode 100644
index 000000000000..d6c1aad94007
--- /dev/null
+++ b/test/CodeGen/Hexagon/bit-skip-byval.ll
@@ -0,0 +1,11 @@
+; RUN: llc -march=hexagon < %s | FileCheck %s
+;
+; Either and or zxtb.
+; CHECK: r0 = and(r1, #255)
+
+%struct.t0 = type { i32 }
+
+define i32 @foo(%struct.t0* byval align 8 %s, i8 zeroext %t, i8 %u) #0 {
+ %a = zext i8 %u to i32
+ ret i32 %a
+}
diff --git a/test/CodeGen/Hexagon/bit-validate-reg.ll b/test/CodeGen/Hexagon/bit-validate-reg.ll
new file mode 100644
index 000000000000..16d4a5e4484d
--- /dev/null
+++ b/test/CodeGen/Hexagon/bit-validate-reg.ll
@@ -0,0 +1,21 @@
+; RUN: llc -march=hexagon < %s | FileCheck %s
+
+; Make sure we don't generate zxtb to transfer a predicate register into
+; a general purpose register.
+
+; CHECK: r0 = p0
+; CHECK-NOT: zxtb(p
+
+target triple = "hexagon"
+
+; Function Attrs: nounwind
+define i32 @fred() local_unnamed_addr #0 {
+entry:
+ %0 = tail call i32 @llvm.hexagon.C4.and.and(i32 undef, i32 undef, i32 undef)
+ ret i32 %0
+}
+
+declare i32 @llvm.hexagon.C4.and.and(i32, i32, i32) #1
+
+attributes #0 = { nounwind "target-cpu"="hexagonv5" }
+attributes #1 = { nounwind readnone }
diff --git a/test/CodeGen/Hexagon/bit-visit-flowq.ll b/test/CodeGen/Hexagon/bit-visit-flowq.ll
new file mode 100644
index 000000000000..b44847dee68e
--- /dev/null
+++ b/test/CodeGen/Hexagon/bit-visit-flowq.ll
@@ -0,0 +1,47 @@
+; RUN: llc -march=hexagon < %s | FileCheck %s
+; REQUIRES: asserts
+
+; Check that we don't crash.
+; CHECK: call bar
+
+target triple = "hexagon"
+
+@debug = external hidden unnamed_addr global i1, align 4
+
+; Function Attrs: nounwind
+define void @foo() local_unnamed_addr #0 {
+entry:
+ br label %if.end5
+
+if.end5: ; preds = %entry
+ br i1 undef, label %if.then12, label %if.end13
+
+if.then12: ; preds = %if.end5
+ unreachable
+
+if.end13: ; preds = %if.end5
+ br label %for.cond
+
+for.cond: ; preds = %if.end13
+ %or.cond288 = or i1 undef, undef
+ br i1 undef, label %if.then44, label %if.end51
+
+if.then44: ; preds = %for.cond
+ tail call void @bar() #0
+ br label %if.end51
+
+if.end51: ; preds = %if.then44, %for.cond
+ %.b433 = load i1, i1* @debug, align 4
+ %or.cond290 = and i1 %or.cond288, %.b433
+ br i1 %or.cond290, label %if.then55, label %if.end63
+
+if.then55: ; preds = %if.end51
+ unreachable
+
+if.end63: ; preds = %if.end51
+ unreachable
+}
+
+declare void @bar() local_unnamed_addr #0
+
+attributes #0 = { nounwind }
diff --git a/test/CodeGen/Hexagon/block-addr.ll b/test/CodeGen/Hexagon/block-addr.ll
index 420af2fee1c9..c0db2cef545e 100644
--- a/test/CodeGen/Hexagon/block-addr.ll
+++ b/test/CodeGen/Hexagon/block-addr.ll
@@ -1,9 +1,8 @@
; RUN: llc -march=hexagon < %s | FileCheck %s
-; Allow combine(..##JTI..):
-; CHECK: r{{[0-9]+}}{{.*}} = {{.*}}#.LJTI
-; CHECK: r{{[0-9]+}} = memw(r{{[0-9]+}}{{ *}}+{{ *}}r{{[0-9]+<<#[0-9]+}})
-; CHECK: jumpr:nt r{{[0-9]+}}
+; CHECK: .LJTI
+; CHECK-DAG: r[[REG:[0-9]+]] = memw(r{{[0-9]+}}{{ *}}+{{ *}}r{{[0-9]+<<#[0-9]+}})
+; CHECK-DAG: jumpr:nt r[[REG]]
define void @main() #0 {
entry:
diff --git a/test/CodeGen/Hexagon/branchfolder-keep-impdef.ll b/test/CodeGen/Hexagon/branchfolder-keep-impdef.ll
new file mode 100644
index 000000000000..a56680bd4399
--- /dev/null
+++ b/test/CodeGen/Hexagon/branchfolder-keep-impdef.ll
@@ -0,0 +1,29 @@
+; RUN: llc -march=hexagon -verify-machineinstrs < %s | FileCheck %s
+;
+; Check that the testcase compiles successfully. Expect that if-conversion
+; took place.
+; CHECK-LABEL: fred:
+; CHECK: if (!p0) r1 = memw(r0 + #0)
+
+target triple = "hexagon"
+
+define void @fred(i32 %p0) local_unnamed_addr align 2 {
+b0:
+ br i1 undef, label %b1, label %b2
+
+b1: ; preds = %b0
+ %t0 = load i8*, i8** undef, align 4
+ br label %b2
+
+b2: ; preds = %b1, %b0
+ %t1 = phi i8* [ %t0, %b1 ], [ undef, %b0 ]
+ %t2 = getelementptr inbounds i8, i8* %t1, i32 %p0
+ tail call void @llvm.memmove.p0i8.p0i8.i32(i8* undef, i8* %t2, i32 undef, i32 1, i1 false) #1
+ unreachable
+}
+
+declare void @llvm.memmove.p0i8.p0i8.i32(i8* nocapture, i8* nocapture readonly, i32, i32, i1) #0
+
+attributes #0 = { argmemonly nounwind }
+attributes #1 = { nounwind }
+
diff --git a/test/CodeGen/Hexagon/build-vector-shuffle.ll b/test/CodeGen/Hexagon/build-vector-shuffle.ll
new file mode 100644
index 000000000000..1d06953ddf32
--- /dev/null
+++ b/test/CodeGen/Hexagon/build-vector-shuffle.ll
@@ -0,0 +1,21 @@
+; RUN: llc -march=hexagon < %s | FileCheck %s
+; Check that we don't crash.
+; CHECK: vshuff
+
+target triple = "hexagon"
+
+define void @hex_interleaved.s0.__outermost() local_unnamed_addr #0 {
+entry:
+ %0 = icmp eq i32 undef, 0
+ %sel2 = select i1 %0, <32 x i16> undef, <32 x i16> zeroinitializer
+ %1 = bitcast <32 x i16> %sel2 to <16 x i32>
+ %2 = tail call <16 x i32> @llvm.hexagon.V6.vshuffh(<16 x i32> %1)
+ store <16 x i32> %2, <16 x i32>* undef, align 2
+ unreachable
+}
+
+; Function Attrs: nounwind readnone
+declare <16 x i32> @llvm.hexagon.V6.vshuffh(<16 x i32>) #1
+
+attributes #0 = { nounwind "target-cpu"="hexagonv60" "target-features"="+hvx" }
+attributes #1 = { nounwind readnone }
diff --git a/test/CodeGen/Hexagon/combine.ll b/test/CodeGen/Hexagon/combine.ll
index 8f5cec88d692..04a080fdf425 100644
--- a/test/CodeGen/Hexagon/combine.ll
+++ b/test/CodeGen/Hexagon/combine.ll
@@ -1,4 +1,4 @@
-; RUN: llc -march=hexagon -mcpu=hexagonv5 -disable-hsdr < %s | FileCheck %s
+; RUN: llc -march=hexagon -mcpu=hexagonv5 -disable-hsdr -hexagon-bit=0 < %s | FileCheck %s
; CHECK: combine(r{{[0-9]+}}, r{{[0-9]+}})
@j = external global i32
diff --git a/test/CodeGen/Hexagon/const-pool-tf.ll b/test/CodeGen/Hexagon/const-pool-tf.ll
new file mode 100644
index 000000000000..9a4569b1e4de
--- /dev/null
+++ b/test/CodeGen/Hexagon/const-pool-tf.ll
@@ -0,0 +1,40 @@
+; RUN: llc -march=hexagon -mcpu=hexagonv60 -relocation-model pic < %s | FileCheck %s
+
+; CHECK: @PCREL
+
+target datalayout = "e-m:e-p:32:32:32-a:0-n16:32-i64:64:64-i32:32:32-i16:16:16-i1:8:8-f32:32:32-f64:64:64-v32:32:32-v64:64:64-v512:512:512-v1024:1024:1024-v2048:2048:2048"
+target triple = "hexagon-unknown--elf"
+
+; Function Attrs: nounwind
+define void @hex_h.s0.__outermost(i32 %h.stride.114) #0 {
+entry:
+ br i1 undef, label %"for h.s0.y.preheader", label %call_destructor.exit, !prof !1
+
+call_destructor.exit: ; preds = %entry
+ ret void
+
+"for h.s0.y.preheader": ; preds = %entry
+ %tmp22.us = mul i32 undef, %h.stride.114
+ br label %"for h.s0.x.x.us"
+
+"for h.s0.x.x.us": ; preds = %"for h.s0.x.x.us", %"for h.s0.y.preheader"
+ %h.s0.x.x.us = phi i32 [ %5, %"for h.s0.x.x.us" ], [ 0, %"for h.s0.y.preheader" ]
+ %0 = shl nsw i32 %h.s0.x.x.us, 5
+ %1 = add i32 %0, %tmp22.us
+ %2 = add nsw i32 %1, 16
+ %3 = getelementptr inbounds i32, i32* null, i32 %2
+ %4 = bitcast i32* %3 to <16 x i32>*
+ store <16 x i32> zeroinitializer, <16 x i32>* %4, align 4, !tbaa !2
+ %5 = add nuw nsw i32 %h.s0.x.x.us, 1
+ br label %"for h.s0.x.x.us"
+}
+
+attributes #0 = { nounwind }
+
+!llvm.ident = !{!0, !0, !0, !0, !0, !0, !0, !0, !0, !0, !0, !0, !0, !0, !0, !0}
+
+!0 = !{!"Clang $LLVM_VERSION_MAJOR.$LLVM_VERSION_MINOR (based on LLVM 3.9.0)"}
+!1 = !{!"branch_weights", i32 1073741824, i32 0}
+!2 = !{!3, !3, i64 0}
+!3 = !{!"h", !4}
+!4 = !{!"Halide buffer"}
diff --git a/test/CodeGen/Hexagon/constp-clb.ll b/test/CodeGen/Hexagon/constp-clb.ll
new file mode 100644
index 000000000000..1a872b11aad3
--- /dev/null
+++ b/test/CodeGen/Hexagon/constp-clb.ll
@@ -0,0 +1,23 @@
+; RUN: llc -mcpu=hexagonv5 < %s
+; REQUIRES: asserts
+
+target datalayout = "e-m:e-p:32:32-i1:32-i64:64-a:0-v32:32-n16:32"
+target triple = "hexagon-unknown--elf"
+
+; Function Attrs: nounwind readnone
+define i64 @foo() #0 {
+entry:
+ %0 = tail call i32 @llvm.hexagon.S2.clbp(i64 291)
+ %1 = tail call i64 @llvm.hexagon.A4.combineir(i32 0, i32 %0)
+ ret i64 %1
+}
+
+; Function Attrs: nounwind readnone
+declare i32 @llvm.hexagon.S2.clbp(i64) #1
+
+; Function Attrs: nounwind readnone
+declare i64 @llvm.hexagon.A4.combineir(i32, i32) #1
+
+attributes #0 = { nounwind readnone "less-precise-fpmad"="false" "no-frame-pointer-elim"="true" "no-frame-pointer-elim-non-leaf" "no-infs-fp-math"="false" "no-nans-fp-math"="false" "stack-protector-buffer-size"="8" "unsafe-fp-math"="false" "use-soft-float"="false" }
+attributes #1 = { nounwind readnone }
+
diff --git a/test/CodeGen/Hexagon/constp-combine-neg.ll b/test/CodeGen/Hexagon/constp-combine-neg.ll
new file mode 100644
index 000000000000..18f0e81076af
--- /dev/null
+++ b/test/CodeGen/Hexagon/constp-combine-neg.ll
@@ -0,0 +1,27 @@
+; RUN: llc -O2 -march=hexagon < %s | FileCheck %s --check-prefix=CHECK-TEST1
+; RUN: llc -O2 -march=hexagon < %s | FileCheck %s --check-prefix=CHECK-TEST2
+; RUN: llc -O2 -march=hexagon < %s | FileCheck %s --check-prefix=CHECK-TEST3
+define i32 @main() #0 {
+entry:
+ %l = alloca [7 x i32], align 8
+ %p_arrayidx45 = bitcast [7 x i32]* %l to i32*
+ %vector_ptr = bitcast [7 x i32]* %l to <2 x i32>*
+ store <2 x i32> <i32 3, i32 -2>, <2 x i32>* %vector_ptr, align 8
+ %p_arrayidx.1 = getelementptr [7 x i32], [7 x i32]* %l, i32 0, i32 2
+ %vector_ptr.1 = bitcast i32* %p_arrayidx.1 to <2 x i32>*
+ store <2 x i32> <i32 -4, i32 6>, <2 x i32>* %vector_ptr.1, align 8
+ %p_arrayidx.2 = getelementptr [7 x i32], [7 x i32]* %l, i32 0, i32 4
+ %vector_ptr.2 = bitcast i32* %p_arrayidx.2 to <2 x i32>*
+ store <2 x i32> <i32 -8, i32 -10>, <2 x i32>* %vector_ptr.2, align 8
+ ret i32 0
+}
+
+; The instructions seem to be in a different order in the .s file than
+; the corresponding values in the .ll file, so just run the test three
+; times and each time test for a different instruction.
+; CHECK-TEST1: combine(#-2, #3)
+; CHECK-TEST2: combine(#6, #-4)
+; CHECK-TEST3: combine(#-10, #-8)
+
+attributes #0 = { "less-precise-fpmad"="false" "no-frame-pointer-elim"="false" "no-frame-pointer-elim-non-leaf"="true" "no-infs-fp-math"="false" "no-nans-fp-math"="false" "unsafe-fp-math"="false" "use-soft-float"="false" }
+
diff --git a/test/CodeGen/Hexagon/constp-ctb.ll b/test/CodeGen/Hexagon/constp-ctb.ll
new file mode 100644
index 000000000000..76a9820583ea
--- /dev/null
+++ b/test/CodeGen/Hexagon/constp-ctb.ll
@@ -0,0 +1,26 @@
+; RUN: llc < %s
+; REQUIRES: asserts
+
+target datalayout = "e-m:e-p:32:32-i1:32-i64:64-a:0-v32:32-n16:32"
+target triple = "hexagon-unknown--elf"
+
+; Function Attrs: nounwind readnone
+define i64 @foo() #0 {
+entry:
+ %0 = tail call i32 @llvm.hexagon.S2.ct0p(i64 18)
+ %1 = tail call i32 @llvm.hexagon.S2.ct1p(i64 27)
+ %2 = tail call i64 @llvm.hexagon.A2.combinew(i32 %0, i32 %1)
+ ret i64 %2
+}
+
+; Function Attrs: nounwind readnone
+declare i32 @llvm.hexagon.S2.ct0p(i64) #0
+
+; Function Attrs: nounwind readnone
+declare i32 @llvm.hexagon.S2.ct1p(i64) #0
+
+; Function Attrs: nounwind readnone
+declare i64 @llvm.hexagon.A2.combinew(i32, i32) #0
+
+attributes #0 = { nounwind readnone }
+
diff --git a/test/CodeGen/Hexagon/constp-extract.ll b/test/CodeGen/Hexagon/constp-extract.ll
new file mode 100644
index 000000000000..00f176317c4c
--- /dev/null
+++ b/test/CodeGen/Hexagon/constp-extract.ll
@@ -0,0 +1,31 @@
+; Expect the constant propagation to evaluate signed and unsigned bit extract.
+; RUN: llc -march=hexagon -O2 < %s | FileCheck %s
+
+target triple = "hexagon"
+
+@x = common global i32 0, align 4
+@y = common global i32 0, align 4
+
+define void @foo() #0 {
+entry:
+ ; extractu(0x000ABCD0, 16, 4)
+ ; should evaluate to 0xABCD (dec 43981)
+ %0 = call i32 @llvm.hexagon.S2.extractu(i32 703696, i32 16, i32 4)
+; CHECK: 43981
+; CHECK-NOT: extractu
+ store i32 %0, i32* @x, align 4
+ ; extract(0x000ABCD0, 16, 4)
+ ; should evaluate to 0xFFFFABCD (dec 4294945741 or -21555)
+ %1 = call i32 @llvm.hexagon.S4.extract(i32 703696, i32 16, i32 4)
+; CHECK: -21555
+; CHECK-NOT: extract
+ store i32 %1, i32* @y, align 4
+ ret void
+}
+
+declare i32 @llvm.hexagon.S2.extractu(i32, i32, i32) #1
+
+declare i32 @llvm.hexagon.S4.extract(i32, i32, i32) #1
+
+attributes #0 = { nounwind "less-precise-fpmad"="false" "no-frame-pointer-elim"="true" "no-frame-pointer-elim-non-leaf"="true" "no-infs-fp-math"="false" "no-nans-fp-math"="false" "unsafe-fp-math"="false" "use-soft-float"="false" }
+attributes #1 = { nounwind readnone }
diff --git a/test/CodeGen/Hexagon/constp-physreg.ll b/test/CodeGen/Hexagon/constp-physreg.ll
new file mode 100644
index 000000000000..0473b96f6de4
--- /dev/null
+++ b/test/CodeGen/Hexagon/constp-physreg.ll
@@ -0,0 +1,21 @@
+; RUN: llc -O2 -march hexagon < %s
+target datalayout = "e-p:32:32:32-i64:64:64-i32:32:32-i16:16:16-i1:32:32-f64:64:64-f32:32:32-v64:64:64-v32:32:32-a0:0-n16:32"
+target triple = "hexagon"
+
+define signext i16 @foo(i16 signext %var1, i16 signext %var2) #0 {
+entry:
+ %0 = or i16 %var2, %var1
+ %1 = icmp slt i16 %0, 0
+ %cmp8 = icmp sgt i16 %var1, %var2
+ %or.cond19 = or i1 %1, %cmp8
+ br i1 %or.cond19, label %return, label %if.end
+
+if.end: ; preds = %entry
+ br label %return
+
+return: ; preds = %if.end, %if.end15, %entry
+ %retval.0.reg2mem.0 = phi i16 [ 0, %entry ], [ 32767, %if.end ]
+ ret i16 %retval.0.reg2mem.0
+}
+
+attributes #0 = { nounwind readnone "less-precise-fpmad"="false" "no-frame-pointer-elim"="false" "no-frame-pointer-elim-non-leaf"="true" "no-infs-fp-math"="false" "no-nans-fp-math"="false" "unsafe-fp-math"="false" "use-soft-float"="false" }
diff --git a/test/CodeGen/Hexagon/constp-rewrite-branches.ll b/test/CodeGen/Hexagon/constp-rewrite-branches.ll
new file mode 100644
index 000000000000..dfac49cb553e
--- /dev/null
+++ b/test/CodeGen/Hexagon/constp-rewrite-branches.ll
@@ -0,0 +1,17 @@
+; RUN: llc -O2 -march hexagon < %s | FileCheck %s
+
+define i32 @foo(i32 %x) {
+ %p = icmp eq i32 %x, 0
+ br i1 %p, label %zero, label %nonzero
+nonzero:
+ %v1 = add i32 %x, 1
+ %c = icmp eq i32 %x, %v1
+; This branch will be rewritten by HCP. A bug would cause both branches to
+; go away, leaving no path to "ret -1".
+ br i1 %c, label %zero, label %other
+zero:
+ ret i32 0
+other:
+; CHECK: -1
+ ret i32 -1
+}
diff --git a/test/CodeGen/Hexagon/constp-rseq.ll b/test/CodeGen/Hexagon/constp-rseq.ll
new file mode 100644
index 000000000000..c89407e4b8e4
--- /dev/null
+++ b/test/CodeGen/Hexagon/constp-rseq.ll
@@ -0,0 +1,19 @@
+; RUN: llc -march=hexagon < %s | FileCheck %s
+; CHECK: cmp
+; Make sure that the result is not a compile-time constant.
+
+define i64 @foo(i32 %x) {
+entry:
+ %c = icmp slt i32 %x, 17
+ br i1 %c, label %b1, label %b2
+b1:
+ br label %b2
+b2:
+ %p = phi i32 [ 1, %entry ], [ 0, %b1 ]
+ %q = sub i32 %x, %x
+ %y = zext i32 %q to i64
+ %u = shl i64 %y, 32
+ %v = zext i32 %p to i64
+ %w = or i64 %u, %v
+ ret i64 %w
+}
diff --git a/test/CodeGen/Hexagon/constp-vsplat.ll b/test/CodeGen/Hexagon/constp-vsplat.ll
new file mode 100644
index 000000000000..6063383a5f75
--- /dev/null
+++ b/test/CodeGen/Hexagon/constp-vsplat.ll
@@ -0,0 +1,18 @@
+; RUN: llc < %s
+; REQUIRES: asserts
+target datalayout = "e-m:e-p:32:32-i1:32-i64:64-a:0-v32:32-n16:32"
+target triple = "hexagon"
+
+; Function Attrs: nounwind readnone
+define i64 @foo() #0 {
+entry:
+ %0 = tail call i32 @llvm.hexagon.S2.vsplatrb(i32 255)
+ %conv = zext i32 %0 to i64
+ %shl = shl nuw i64 %conv, 32
+ %or = or i64 %shl, %conv
+ ret i64 %or
+}
+
+declare i32 @llvm.hexagon.S2.vsplatrb(i32) #0
+
+attributes #0 = { nounwind readnone }
diff --git a/test/CodeGen/Hexagon/copy-to-combine-dbg.ll b/test/CodeGen/Hexagon/copy-to-combine-dbg.ll
new file mode 100644
index 000000000000..9ffc7b509251
--- /dev/null
+++ b/test/CodeGen/Hexagon/copy-to-combine-dbg.ll
@@ -0,0 +1,57 @@
+; RUN: llc -march=hexagon < %s | FileCheck %s
+; Check for some sane output (original problem was a crash).
+; CHECK: DEBUG_VALUE: fred:Count <- 0
+
+target triple = "hexagon"
+
+define i32 @fred(i32 %p) local_unnamed_addr #0 !dbg !6 {
+entry:
+ br label %cond.end
+
+cond.end: ; preds = %entry
+ br i1 undef, label %cond.false.i, label %for.body.lr.ph.i
+
+for.body.lr.ph.i: ; preds = %cond.end
+ tail call void @llvm.dbg.value(metadata i32 0, i64 0, metadata !10, metadata !12) #0, !dbg !13
+ br label %for.body.i
+
+cond.false.i: ; preds = %cond.end
+ unreachable
+
+for.body.i: ; preds = %for.inc.i, %for.body.lr.ph.i
+ %inc.sink37.i = phi i32 [ 0, %for.body.lr.ph.i ], [ %inc.i, %for.inc.i ]
+ %call.i = tail call i8* undef(i32 12, i8* undef) #0
+ br label %for.inc.i
+
+for.inc.i: ; preds = %for.body.i
+ %inc.i = add nuw i32 %inc.sink37.i, 1
+ %cmp1.i = icmp ult i32 %inc.i, %p
+ br i1 %cmp1.i, label %for.body.i, label %PQ_AllocMem.exit.loopexit
+
+PQ_AllocMem.exit.loopexit: ; preds = %for.inc.i
+ unreachable
+}
+
+declare void @llvm.dbg.value(metadata, i64, metadata, metadata) #1
+
+attributes #0 = { nounwind }
+attributes #1 = { nounwind readnone }
+
+!llvm.dbg.cu = !{!0}
+!llvm.module.flags = !{!3, !4}
+!llvm.ident = !{!5}
+
+!0 = distinct !DICompileUnit(language: DW_LANG_C99, file: !1, producer: "clang version 4.0.0 (http://llvm.org/git/clang.git 37afcb099ac2b001f4c826da7ca1d077b67a508c) (http://llvm.org/git/llvm.git 5887f1c75b3ba216850c834b186efdd3e54b7d4f)", isOptimized: true, runtimeVersion: 0, emissionKind: FullDebug, enums: !2, retainedTypes: !2)
+!1 = !DIFile(filename: "file.c", directory: "/")
+!2 = !{}
+!3 = !{i32 2, !"Dwarf Version", i32 4}
+!4 = !{i32 2, !"Debug Info Version", i32 3}
+!5 = !{!"clang version 4.0.0 (http://llvm.org/git/clang.git 37afcb099ac2b001f4c826da7ca1d077b67a508c) (http://llvm.org/git/llvm.git 5887f1c75b3ba216850c834b186efdd3e54b7d4f)"}
+!6 = distinct !DISubprogram(name: "fred", scope: !1, file: !1, line: 116, type: !7, isLocal: false, isDefinition: true, scopeLine: 121, flags: DIFlagPrototyped, isOptimized: true, unit: !0, variables: !9)
+!7 = !DISubroutineType(types: !2)
+!8 = !DIBasicType(name: "int", size: 32, align: 32, encoding: DW_ATE_signed)
+!9 = !{!10}
+!10 = !DILocalVariable(name: "Count", scope: !6, file: !1, line: 1, type: !8)
+!11 = distinct !DILocation(line: 1, column: 1, scope: !6)
+!12 = !DIExpression()
+!13 = !DILocation(line: 1, column: 1, scope: !6, inlinedAt: !11)
diff --git a/test/CodeGen/Hexagon/dead-store-stack.ll b/test/CodeGen/Hexagon/dead-store-stack.ll
new file mode 100644
index 000000000000..93d324baad9e
--- /dev/null
+++ b/test/CodeGen/Hexagon/dead-store-stack.ll
@@ -0,0 +1,131 @@
+; RUN: llc -O2 -march=hexagon < %s | FileCheck %s
+; CHECK: ParseFunc:
+; CHECK: r[[ARG0:[0-9]+]] = memuh(r[[ARG1:[0-9]+]] + #[[OFFSET:[0-9]+]])
+; CHECK: memw(r[[ARG1]]+#[[OFFSET]]) = r[[ARG0]]
+
+@.str.3 = external unnamed_addr constant [8 x i8], align 1
+; Function Attrs: nounwind
+define void @ParseFunc() local_unnamed_addr #0 {
+entry:
+ %dataVar = alloca i32, align 4
+ %0 = load i32, i32* %dataVar, align 4
+ %and = and i32 %0, 65535
+ store i32 %and, i32* %dataVar, align 4
+ %.pr = load i32, i32* %dataVar, align 4
+ switch i32 %.pr, label %sw.epilog [
+ i32 4, label %sw.bb
+ i32 5, label %sw.bb
+ i32 1, label %sw.bb39
+ i32 2, label %sw.bb40
+ i32 3, label %sw.bb41
+ i32 6, label %sw.bb42
+ i32 7, label %sw.bb43
+ i32 13, label %sw.bb44
+ i32 0, label %sw.bb44
+ i32 14, label %sw.bb45
+ i32 15, label %sw.bb46
+ ]
+
+sw.bb:
+ %cmp1.i = icmp eq i32 %.pr, 4
+ br label %land.rhs.i
+
+land.rhs.i:
+ br label %ParseFuncNext.exit.i
+
+ParseFuncNext.exit.i:
+ br i1 %cmp1.i, label %if.then.i, label %if.else10.i
+
+if.then.i:
+ call void (i8*, i32, i8*, ...) @snprintf(i8* undef, i32 undef, i8* getelementptr inbounds ([8 x i8], [8 x i8]* @.str.3, i32 0, i32 0), i32 undef) #2
+ br label %if.end27.i
+
+if.else10.i:
+ unreachable
+
+if.end27.i:
+ br label %land.rhs.i
+
+sw.bb39:
+ unreachable
+
+sw.bb40:
+ unreachable
+
+sw.bb41:
+ unreachable
+
+sw.bb42:
+ %1 = load i32, i32* undef, align 4
+ %shr.i = lshr i32 %1, 16
+ br label %while.cond.i.i
+
+while.cond.i.i:
+ %2 = load i8, i8* undef, align 1
+ switch i8 %2, label %if.then4.i [
+ i8 48, label %land.end.i.i
+ i8 120, label %land.end.i.i
+ i8 37, label %do.body.i.i
+ ]
+
+land.end.i.i:
+ unreachable
+
+do.body.i.i:
+ switch i8 undef, label %if.then4.i [
+ i8 117, label %if.end40.i.i
+ i8 120, label %if.end40.i.i
+ i8 88, label %if.end40.i.i
+ i8 100, label %if.end40.i.i
+ i8 105, label %if.end40.i.i
+ ]
+
+if.end40.i.i:
+ %trunc.i = trunc i32 %shr.i to i16
+ br label %land.rhs.i126
+
+if.then4.i:
+ unreachable
+
+land.rhs.i126:
+ switch i16 %trunc.i, label %sw.epilog.i [
+ i16 1, label %sw.bb.i
+ i16 2, label %sw.bb12.i
+ i16 4, label %sw.bb16.i
+ ]
+
+sw.bb.i:
+ unreachable
+
+sw.bb12.i:
+ unreachable
+
+sw.bb16.i:
+ unreachable
+
+sw.epilog.i:
+ call void (i8*, i32, i8*, ...) @snprintf(i8* undef, i32 undef, i8* nonnull undef, i32 undef) #2
+ br label %land.rhs.i126
+
+sw.bb43:
+ unreachable
+
+sw.bb44:
+ unreachable
+
+sw.bb45:
+ unreachable
+
+sw.bb46:
+ unreachable
+
+sw.epilog:
+ ret void
+}
+
+; Function Attrs: nounwind
+declare void @snprintf(i8* nocapture, i32, i8* nocapture readonly, ...) local_unnamed_addr #1
+
+attributes #0 = { nounwind "correctly-rounded-divide-sqrt-fp-math"="false" "disable-tail-calls"="false" "less-precise-fpmad"="false" "no-frame-pointer-elim"="true" "no-frame-pointer-elim-non-leaf" "no-infs-fp-math"="false" "no-jump-tables"="false" "no-nans-fp-math"="false" "no-signed-zeros-fp-math"="false" "stack-protector-buffer-size"="8" "target-features"="+hvx" "unsafe-fp-math"="false" "use-soft-float"="false" }
+attributes #1 = { nounwind "correctly-rounded-divide-sqrt-fp-math"="false" "disable-tail-calls"="false" "less-precise-fpmad"="false" "no-frame-pointer-elim"="true" "no-frame-pointer-elim-non-leaf" "no-infs-fp-math"="false" "no-nans-fp-math"="false" "no-signed-zeros-fp-math"="false" "stack-protector-buffer-size"="8" "target-features"="+hvx" "unsafe-fp-math"="false" "use-soft-float"="false" }
+attributes #2 = { nounwind }
diff --git a/test/CodeGen/Hexagon/early-if-vecpi.ll b/test/CodeGen/Hexagon/early-if-vecpi.ll
new file mode 100644
index 000000000000..6f3ec2d5a51d
--- /dev/null
+++ b/test/CodeGen/Hexagon/early-if-vecpi.ll
@@ -0,0 +1,69 @@
+; RUN: llc -march=hexagon < %s | FileCheck %s
+
+target triple = "hexagon-unknown--elf"
+
+; Check that we can predicate base+offset vector stores.
+; CHECK-LABEL: sammy
+; CHECK: if{{.*}}vmem(r{{[0-9]+}}+#0) =
+define void @sammy(<16 x i32>* nocapture %p, <16 x i32>* nocapture readonly %q, i32 %n) #0 {
+entry:
+ %0 = load <16 x i32>, <16 x i32>* %q, align 64
+ %sub = add nsw i32 %n, -1
+ br label %for.body
+
+for.body: ; preds = %if.end, %entry
+ %p.addr.011 = phi <16 x i32>* [ %p, %entry ], [ %incdec.ptr, %if.end ]
+ %i.010 = phi i32 [ 0, %entry ], [ %add, %if.end ]
+ %mul = mul nsw i32 %i.010, %sub
+ %add = add nuw nsw i32 %i.010, 1
+ %mul1 = mul nsw i32 %add, %n
+ %cmp2 = icmp slt i32 %mul, %mul1
+ br i1 %cmp2, label %if.then, label %if.end
+
+if.then: ; preds = %for.body
+ store <16 x i32> %0, <16 x i32>* %p.addr.011, align 64
+ br label %if.end
+
+if.end: ; preds = %if.then, %for.body
+ %incdec.ptr = getelementptr inbounds <16 x i32>, <16 x i32>* %p.addr.011, i32 1
+ %exitcond = icmp eq i32 %add, 100
+ br i1 %exitcond, label %for.end, label %for.body
+
+for.end: ; preds = %if.end
+ ret void
+}
+
+; Check that we can predicate post-increment vector stores.
+; CHECK-LABEL: danny
+; CHECK: if{{.*}}vmem(r{{[0-9]+}}++#1) =
+define void @danny(<16 x i32>* nocapture %p, <16 x i32>* nocapture readonly %q, i32 %n) #0 {
+entry:
+ %0 = load <16 x i32>, <16 x i32>* %q, align 64
+ %sub = add nsw i32 %n, -1
+ br label %for.body
+
+for.body: ; preds = %if.end, %entry
+ %p.addr.012 = phi <16 x i32>* [ %p, %entry ], [ %incdec.ptr3, %if.end ]
+ %i.011 = phi i32 [ 0, %entry ], [ %add, %if.end ]
+ %mul = mul nsw i32 %i.011, %sub
+ %add = add nuw nsw i32 %i.011, 1
+ %mul1 = mul nsw i32 %add, %n
+ %cmp2 = icmp slt i32 %mul, %mul1
+ br i1 %cmp2, label %if.then, label %if.end
+
+if.then: ; preds = %for.body
+ %incdec.ptr = getelementptr inbounds <16 x i32>, <16 x i32>* %p.addr.012, i32 1
+ store <16 x i32> %0, <16 x i32>* %p.addr.012, align 64
+ br label %if.end
+
+if.end: ; preds = %if.then, %for.body
+ %p.addr.1 = phi <16 x i32>* [ %incdec.ptr, %if.then ], [ %p.addr.012, %for.body ]
+ %incdec.ptr3 = getelementptr inbounds <16 x i32>, <16 x i32>* %p.addr.1, i32 1
+ %exitcond = icmp eq i32 %add, 100
+ br i1 %exitcond, label %for.end, label %for.body
+
+for.end: ; preds = %if.end
+ ret void
+}
+
+attributes #0 = { norecurse nounwind "target-cpu"="hexagonv60" "target-features"="+hvx,-hvx-double" }
diff --git a/test/CodeGen/Hexagon/expand-condsets-def-undef.mir b/test/CodeGen/Hexagon/expand-condsets-def-undef.mir
new file mode 100644
index 000000000000..44da969bf29b
--- /dev/null
+++ b/test/CodeGen/Hexagon/expand-condsets-def-undef.mir
@@ -0,0 +1,41 @@
+# RUN: llc -march=hexagon -run-pass expand-condsets -o - %s -verify-machineinstrs | FileCheck %s
+
+# CHECK-LABEL: name: fred
+
+# Make sure that <def,read-undef> is accounted for when validating moves
+# during predication. In the code below, %2.isub_hi is invalidated
+# by the C2_mux instruction, and so predicating the A2_addi as an argument
+# to the C2_muxir should not happen.
+
+--- |
+ define void @fred() { ret void }
+
+...
+---
+
+name: fred
+tracksRegLiveness: true
+registers:
+ - { id: 0, class: predregs }
+ - { id: 1, class: intregs }
+ - { id: 2, class: doubleregs }
+ - { id: 3, class: intregs }
+liveins:
+ - { reg: '%p0', virtual-reg: '%0' }
+ - { reg: '%r0', virtual-reg: '%1' }
+ - { reg: '%d0', virtual-reg: '%2' }
+
+body: |
+ bb.0:
+ liveins: %r0, %d0, %p0
+ %0 = COPY %p0
+ %1 = COPY %r0
+ %2 = COPY %d0
+ ; Check that this instruction is unchanged (remains unpredicated)
+ ; CHECK: %3 = A2_addi %2.isub_hi, 1
+ %3 = A2_addi %2.isub_hi, 1
+ undef %2.isub_lo = C2_mux %0, %2.isub_lo, %1
+ %2.isub_hi = C2_muxir %0, %3, 0
+
+...
+
diff --git a/test/CodeGen/Hexagon/expand-condsets-extend.ll b/test/CodeGen/Hexagon/expand-condsets-extend.ll
new file mode 100644
index 000000000000..e716925ed8ef
--- /dev/null
+++ b/test/CodeGen/Hexagon/expand-condsets-extend.ll
@@ -0,0 +1,112 @@
+; RUN: llc -march=hexagon < %s | FileCheck %s
+; REQUIRES: asserts
+
+; Check for a reasonable output. This testcase used to crash.
+; CHECK: .size fred
+
+target triple = "hexagon"
+
+define void @fred() local_unnamed_addr #0 {
+entry:
+ %0 = load i64, i64* undef, align 8
+ %shr.i465 = lshr i64 %0, 48
+ %trunc = trunc i64 %shr.i465 to i15
+ switch i15 %trunc, label %if.end26 [
+ i15 -1, label %if.then14
+ i15 0, label %if.then21
+ ]
+
+if.then14: ; preds = %entry
+ unreachable
+
+if.then21: ; preds = %entry
+ unreachable
+
+if.end26: ; preds = %entry
+ br label %if.end36
+
+if.end36: ; preds = %if.end26
+ %or.i335 = or i64 undef, undef
+ %shl2.i322 = or i64 undef, -9223372036854775808
+ br i1 undef, label %if.then44, label %lor.rhs.i
+
+lor.rhs.i: ; preds = %if.end36
+ br label %le128.exit
+
+le128.exit: ; preds = %lor.rhs.i
+ br i1 undef, label %if.then44, label %while.cond.preheader
+
+if.then44: ; preds = %le128.exit, %if.end36
+ %conv42544 = phi i64 [ 0, %le128.exit ], [ 1, %if.end36 ]
+ br label %while.cond.preheader
+
+while.cond.preheader: ; preds = %if.then44, %le128.exit
+ %aSig0.3.ph = phi i64 [ undef, %if.then44 ], [ %or.i335, %le128.exit ]
+ %q.0.ph = phi i64 [ %conv42544, %if.then44 ], [ 0, %le128.exit ]
+ br i1 undef, label %while.body.lr.ph, label %while.end
+
+while.body.lr.ph: ; preds = %while.cond.preheader
+ %shr.i263 = lshr i64 %shl2.i322, 32
+ br label %while.body
+
+while.body: ; preds = %exit312, %while.body.lr.ph
+ %aSig0.3554 = phi i64 [ %aSig0.3.ph, %while.body.lr.ph ], [ %sub3.i205, %exit312 ]
+ br label %while.body.i297
+
+while.body.i297: ; preds = %while.body.i297, %while.body
+ %z.045.i287 = phi i64 [ %sub.i290, %while.body.i297 ], [ undef, %while.body ]
+ %sub.i290 = add i64 %z.045.i287, -4294967296
+ %cmp3.i296 = icmp slt i64 undef, 0
+ br i1 %cmp3.i296, label %while.body.i297, label %while.end.i305.loopexit
+
+while.end.i305.loopexit: ; preds = %while.body.i297
+ %or14.i309 = or i64 0, %sub.i290
+ br label %exit312
+
+exit312: ; preds = %while.end.i305.loopexit
+ %cmp50 = icmp ugt i64 %or14.i309, 4
+ %cond = select i1 %cmp50, i64 undef, i64 0
+ %shr3.i.i221 = lshr i64 %cond, 32
+ %mul15.i11.i243 = mul nuw i64 %shr3.i.i221, %shr.i263
+ %add20.i18.i250 = add i64 0, %mul15.i11.i243
+ %add26.i23.i255 = add i64 %add20.i18.i250, 0
+ %add3.i.i261 = add i64 %add26.i23.i255, 0
+ %shl4.i215 = shl i64 %add3.i.i261, 61
+ %or10.i = or i64 %shl4.i215, 0
+ %shl2.i207 = shl i64 %aSig0.3554, 61
+ %or.i209 = or i64 %shl2.i207, 0
+ %sub1.i202 = add i64 0, %or.i209
+ %sub3.i205 = sub i64 %sub1.i202, %or10.i
+ %cmp47 = icmp sgt i32 undef, 61
+ br i1 %cmp47, label %while.body, label %while.end.loopexit
+
+while.end.loopexit: ; preds = %exit312
+ br label %while.end
+
+while.end: ; preds = %while.end.loopexit, %while.cond.preheader
+ %aSig0.3.lcssa = phi i64 [ %aSig0.3.ph, %while.cond.preheader ], [ %sub3.i205, %while.end.loopexit ]
+ %q.0.lcssa = phi i64 [ %q.0.ph, %while.cond.preheader ], [ %cond, %while.end.loopexit ]
+ br i1 undef, label %if.then56, label %if.else71
+
+if.then56: ; preds = %while.end
+ unreachable
+
+if.else71: ; preds = %while.end
+ %shr8.i155 = lshr i64 %aSig0.3.lcssa, 12
+ br label %do.body
+
+do.body: ; preds = %do.body, %if.else71
+ %aSig0.5 = phi i64 [ %sub3.i151, %do.body ], [ %shr8.i155, %if.else71 ]
+ %q.1 = phi i64 [ %inc, %do.body ], [ %q.0.lcssa, %if.else71 ]
+ %inc = add i64 %q.1, 1
+ %sub1.i148 = sub i64 %aSig0.5, 0
+ %sub3.i151 = add i64 %sub1.i148, 0
+ %cmp73 = icmp sgt i64 %sub3.i151, -1
+ br i1 %cmp73, label %do.body, label %do.end
+
+do.end: ; preds = %do.body
+ %and = and i64 %inc, 1
+ unreachable
+}
+
+attributes #0 = { nounwind }
diff --git a/test/CodeGen/Hexagon/expand-condsets-impuse.mir b/test/CodeGen/Hexagon/expand-condsets-impuse.mir
new file mode 100644
index 000000000000..08b6798aa2fb
--- /dev/null
+++ b/test/CodeGen/Hexagon/expand-condsets-impuse.mir
@@ -0,0 +1,78 @@
+# RUN: llc -march=hexagon -run-pass expand-condsets -o - %s -verify-machineinstrs | FileCheck %s
+
+# CHECK-LABEL: name: fred
+
+--- |
+ define void @fred() { ret void }
+
+...
+---
+
+name: fred
+tracksRegLiveness: true
+registers:
+ - { id: 0, class: intregs }
+ - { id: 1, class: intregs }
+ - { id: 2, class: intregs }
+ - { id: 3, class: intregs }
+ - { id: 4, class: predregs }
+ - { id: 5, class: intregs }
+ - { id: 6, class: intregs }
+ - { id: 7, class: intregs }
+ - { id: 8, class: predregs }
+ - { id: 9, class: intregs }
+ - { id: 10, class: intregs }
+ - { id: 11, class: intregs }
+ - { id: 12, class: predregs }
+ - { id: 13, class: intregs }
+ - { id: 14, class: intregs }
+ - { id: 99, class: intregs }
+liveins:
+ - { reg: '%r0', virtual-reg: '%99' }
+
+body: |
+ bb.0:
+ liveins: %r0
+ successors: %bb.298, %bb.301
+ %99 = COPY %r0
+ J2_jumpr %99, implicit-def %pc
+
+ bb.298:
+ liveins: %r0
+ successors: %bb.299, %bb.301, %bb.309
+ %0 = A2_tfrsi 123
+ %1 = A2_tfrsi -1
+ %3 = L2_loadri_io %99, 8
+ %4 = C2_cmpeqi %3, 33
+ %5 = A2_tfrsi -2
+ %6 = C2_mux %4, %5, %1
+ J2_jumpr %6, implicit-def %pc
+
+ bb.299:
+ successors: %bb.300, %bb.309
+ %7 = L2_loadrb_io %99, 12
+ %8 = C2_cmpeqi %7, 9
+ %9 = A2_tfrsi -999
+ ; CHECK: %10 = C2_cmoveit killed %8, -999, implicit %10
+ %10 = C2_mux %8, %9, %1
+ J2_jumpr %10, implicit-def %pc
+
+ bb.300:
+ successors: %bb.309
+ S2_storeri_io %99, 0, %0
+ J2_jump %bb.309, implicit-def %pc
+
+ bb.301:
+ successors: %bb.299, %bb.309
+ %0 = A2_tfrsi 124
+ %1 = A2_tfrsi -4
+ %11 = L2_loadri_io %99, 8
+ %12 = C2_cmpeqi %11, 33
+ %13 = A2_tfrsi -2
+ %14 = C2_mux %12, %13, %1
+ J2_jumpr %14, implicit-def %pc
+
+ bb.309:
+
+...
+
diff --git a/test/CodeGen/Hexagon/expand-condsets-rm-reg.mir b/test/CodeGen/Hexagon/expand-condsets-rm-reg.mir
new file mode 100644
index 000000000000..983035e228cc
--- /dev/null
+++ b/test/CodeGen/Hexagon/expand-condsets-rm-reg.mir
@@ -0,0 +1,49 @@
+# RUN: llc -march=hexagon -run-pass expand-condsets -o - 2>&1 %s -verify-machineinstrs -debug-only=expand-condsets | FileCheck %s
+# REQUIRES: asserts
+
+# Check that coalesced registers are removed from live intervals.
+#
+# Check that vreg3 is coalesced into vreg4, and that after coalescing
+# it is no longer in live intervals.
+
+# CHECK-LABEL: After expand-condsets
+# CHECK: INTERVALS
+# CHECK-NOT: vreg3
+# CHECK: MACHINEINSTRS
+
+
+--- |
+ define void @fred() { ret void }
+
+...
+---
+
+name: fred
+tracksRegLiveness: true
+registers:
+ - { id: 0, class: intregs }
+ - { id: 1, class: intregs }
+ - { id: 2, class: predregs }
+ - { id: 3, class: intregs }
+ - { id: 4, class: intregs }
+liveins:
+ - { reg: '%r0', virtual-reg: '%0' }
+ - { reg: '%r1', virtual-reg: '%1' }
+ - { reg: '%p0', virtual-reg: '%2' }
+
+body: |
+ bb.0:
+ liveins: %r0, %r1, %p0
+ %0 = COPY %r0
+ %0 = COPY %r0 ; Force isSSA = false.
+ %1 = COPY %r1
+ %2 = COPY %p0
+ ; Check that %3 was coalesced into %4.
+ ; CHECK: %4 = A2_abs %1
+ ; CHECK: %4 = A2_tfrt killed %2, killed %0, implicit %4
+ %3 = A2_abs %1
+ %4 = C2_mux %2, %0, %3
+ %r0 = COPY %4
+ J2_jumpr %r31, implicit %r0, implicit-def %pc
+...
+
diff --git a/test/CodeGen/Hexagon/expand-condsets-same-inputs.mir b/test/CodeGen/Hexagon/expand-condsets-same-inputs.mir
new file mode 100644
index 000000000000..83938d1b774a
--- /dev/null
+++ b/test/CodeGen/Hexagon/expand-condsets-same-inputs.mir
@@ -0,0 +1,32 @@
+# RUN: llc -march=hexagon -run-pass expand-condsets -expand-condsets-coa-limit=0 -o - %s -verify-machineinstrs | FileCheck %s
+
+# CHECK-LABEL: name: fred
+
+--- |
+ define void @fred() { ret void }
+
+...
+---
+
+name: fred
+tracksRegLiveness: true
+registers:
+ - { id: 0, class: predregs }
+ - { id: 1, class: intregs }
+ - { id: 2, class: intregs }
+ - { id: 3, class: intregs }
+
+body: |
+ bb.0:
+ liveins: %r0, %r1, %r2, %p0
+ %0 = COPY %p0
+ %0 = COPY %p0 ; Cheat: convince MIR parser that this is not SSA.
+ %1 = COPY %r1
+ ; Make sure we do not expand/predicate a mux with identical inputs.
+ ; CHECK-NOT: A2_paddit
+ %2 = A2_addi %1, 1
+ %3 = C2_mux %0, killed %2, %2
+ %r0 = COPY %3
+
+...
+
diff --git a/test/CodeGen/Hexagon/expand-condsets-undef2.ll b/test/CodeGen/Hexagon/expand-condsets-undef2.ll
new file mode 100644
index 000000000000..d62d50d83613
--- /dev/null
+++ b/test/CodeGen/Hexagon/expand-condsets-undef2.ll
@@ -0,0 +1,47 @@
+; RUN: llc -march=hexagon < %s
+; REQUIRES: asserts
+
+; Test that the HexagonExpandCondsets pass does not assert due to
+; attempting to shrink a live interval incorrectly.
+
+
+define void @test() #0 {
+entry:
+ br i1 undef, label %cleanup, label %if.end
+
+if.end:
+ %0 = load i32, i32* undef, align 4
+ %sext = shl i32 %0, 16
+ %conv19 = ashr exact i32 %sext, 16
+ br i1 undef, label %cleanup, label %for.body.lr.ph
+
+for.body.lr.ph:
+ br label %for.body
+
+for.body:
+ %bestScoreL16Q4.0278 = phi i16 [ 32767, %for.body.lr.ph ], [ %.sink, %early_termination ]
+ br i1 false, label %for.body44.lr.ph, label %for.cond90.preheader
+
+for.body44.lr.ph:
+ %conv77 = sext i16 %bestScoreL16Q4.0278 to i32
+ unreachable
+
+for.cond90.preheader:
+ br i1 undef, label %early_termination, label %for.body97
+
+for.body97:
+ br i1 undef, label %for.body97, label %early_termination
+
+early_termination:
+ %.sink = select i1 undef, i16 undef, i16 %bestScoreL16Q4.0278
+ %cmp27 = icmp slt i32 undef, %conv19
+ br i1 %cmp27, label %for.body, label %for.end124
+
+for.end124:
+ unreachable
+
+cleanup:
+ ret void
+}
+
+attributes #0 = { nounwind "target-cpu"="hexagonv60" }
diff --git a/test/CodeGen/Hexagon/expand-vstorerw-undef.ll b/test/CodeGen/Hexagon/expand-vstorerw-undef.ll
new file mode 100644
index 000000000000..8524bf33de18
--- /dev/null
+++ b/test/CodeGen/Hexagon/expand-vstorerw-undef.ll
@@ -0,0 +1,95 @@
+; RUN: llc -march=hexagon < %s | FileCheck %s
+
+; After register allocation it is possible to have a spill of a register
+; that is only partially defined. That in itself it fine, but creates a
+; problem for double vector registers. Stores of such registers are pseudo
+; instructions that are expanded into pairs of individual vector stores,
+; and in case of a partially defined source, one of the stores may use
+; an entirely undefined register.
+;
+; This testcase used to crash. Make sure we can handle it, and that we
+; do generate a store for the defined part of W0:
+
+; CHECK-LABEL: fred:
+; CHECK: v[[REG:[0-9]+]] = vsplat
+; CHECK: vmem(r29+#6) = v[[REG]]
+
+
+target triple = "hexagon"
+
+declare void @danny() local_unnamed_addr #0
+declare void @sammy() local_unnamed_addr #0
+declare <32 x i32> @llvm.hexagon.V6.lo.128B(<64 x i32>) #1
+declare <32 x i32> @llvm.hexagon.V6.lvsplatw.128B(i32) #1
+declare <64 x i32> @llvm.hexagon.V6.vcombine.128B(<32 x i32>, <32 x i32>) #1
+declare <32 x i32> @llvm.hexagon.V6.vshuffeb.128B(<32 x i32>, <32 x i32>) #1
+declare <32 x i32> @llvm.hexagon.V6.vlsrh.128B(<32 x i32>, i32) #1
+declare <64 x i32> @llvm.hexagon.V6.vaddh.dv.128B(<64 x i32>, <64 x i32>) #1
+
+define hidden void @fred() #2 {
+b0:
+ %v1 = load i32, i32* null, align 4
+ %v2 = icmp ult i64 0, 2147483648
+ br i1 %v2, label %b3, label %b5
+
+b3: ; preds = %b0
+ %v4 = icmp sgt i32 0, -1
+ br i1 %v4, label %b6, label %b5
+
+b5: ; preds = %b3, %b0
+ ret void
+
+b6: ; preds = %b3
+ tail call void @danny()
+ br label %b7
+
+b7: ; preds = %b21, %b6
+ %v8 = icmp sgt i32 %v1, 0
+ %v9 = select i1 %v8, i32 %v1, i32 0
+ %v10 = select i1 false, i32 0, i32 %v9
+ %v11 = icmp slt i32 %v10, 0
+ %v12 = select i1 %v11, i32 %v10, i32 0
+ %v13 = icmp slt i32 0, %v12
+ br i1 %v13, label %b14, label %b18
+
+b14: ; preds = %b16, %b7
+ br i1 false, label %b15, label %b16
+
+b15: ; preds = %b14
+ br label %b16
+
+b16: ; preds = %b15, %b14
+ %v17 = icmp eq i32 0, %v12
+ br i1 %v17, label %b18, label %b14
+
+b18: ; preds = %b16, %b7
+ tail call void @danny()
+ %v19 = tail call <32 x i32> @llvm.hexagon.V6.lvsplatw.128B(i32 524296) #0
+ %v20 = tail call <64 x i32> @llvm.hexagon.V6.vcombine.128B(<32 x i32> %v19, <32 x i32> %v19)
+ br label %b22
+
+b21: ; preds = %b22
+ tail call void @sammy() #3
+ br label %b7
+
+b22: ; preds = %b22, %b18
+ %v23 = tail call <64 x i32> @llvm.hexagon.V6.vaddh.dv.128B(<64 x i32> zeroinitializer, <64 x i32> %v20) #0
+ %v24 = tail call <32 x i32> @llvm.hexagon.V6.lo.128B(<64 x i32> %v23)
+ %v25 = tail call <32 x i32> @llvm.hexagon.V6.vlsrh.128B(<32 x i32> %v24, i32 4) #0
+ %v26 = tail call <64 x i32> @llvm.hexagon.V6.vcombine.128B(<32 x i32> zeroinitializer, <32 x i32> %v25)
+ %v27 = tail call <32 x i32> @llvm.hexagon.V6.vshuffeb.128B(<32 x i32> zeroinitializer, <32 x i32> zeroinitializer) #0
+ %v28 = tail call <32 x i32> @llvm.hexagon.V6.lo.128B(<64 x i32> %v26) #0
+ %v29 = tail call <32 x i32> @llvm.hexagon.V6.vshuffeb.128B(<32 x i32> zeroinitializer, <32 x i32> %v28) #0
+ store <32 x i32> %v27, <32 x i32>* null, align 128
+ %v30 = add nsw i32 0, 128
+ %v31 = getelementptr inbounds i8, i8* null, i32 %v30
+ %v32 = bitcast i8* %v31 to <32 x i32>*
+ store <32 x i32> %v29, <32 x i32>* %v32, align 128
+ %v33 = icmp eq i32 0, 0
+ br i1 %v33, label %b21, label %b22
+}
+
+attributes #0 = { nounwind }
+attributes #1 = { nounwind readnone }
+attributes #2 = { nounwind "reciprocal-estimates"="none" "target-cpu"="hexagonv60" "target-features"="+hvx-double" }
+attributes #3 = { nobuiltin nounwind }
diff --git a/test/CodeGen/Hexagon/fixed-spill-mutable.ll b/test/CodeGen/Hexagon/fixed-spill-mutable.ll
new file mode 100644
index 000000000000..03aa72bc5a8c
--- /dev/null
+++ b/test/CodeGen/Hexagon/fixed-spill-mutable.ll
@@ -0,0 +1,69 @@
+; RUN: llc -march=hexagon < %s | FileCheck %s
+
+; The early return is predicated, and the save-restore code is mixed together:
+; {
+; p0 = cmp.eq(r0, #0)
+; if (p0.new) r17:16 = memd(r29 + #0)
+; memd(r29+#0) = r17:16
+; }
+; {
+; if (p0) dealloc_return
+; }
+; The problem is that the load will execute before the store, clobbering the
+; pair r17:16.
+;
+; Check that the store and the load are not in the same packet.
+; CHECK: memd{{.*}} = r17:16
+; CHECK: }
+; CHECK: r17:16 = memd
+; CHECK-LABEL: LBB0_1:
+
+target triple = "hexagon"
+
+%struct.0 = type { i8*, %struct.1*, %struct.2*, %struct.0*, %struct.0* }
+%struct.1 = type { [60 x i8], i32, %struct.1* }
+%struct.2 = type { i8, i8, i8, i8, %union.anon }
+%union.anon = type { %struct.3* }
+%struct.3 = type { %struct.3*, %struct.2* }
+
+@var = external hidden unnamed_addr global %struct.0*, align 4
+
+declare void @bar(i8*, i32) local_unnamed_addr #0
+
+define void @foo() local_unnamed_addr #1 {
+entry:
+ %.pr = load %struct.0*, %struct.0** @var, align 4, !tbaa !1
+ %cmp2 = icmp eq %struct.0* %.pr, null
+ br i1 %cmp2, label %while.end, label %while.body.preheader
+
+while.body.preheader: ; preds = %entry
+ br label %while.body
+
+while.body: ; preds = %while.body.preheader, %while.body
+ %0 = phi %struct.0* [ %4, %while.body ], [ %.pr, %while.body.preheader ]
+ %right = getelementptr inbounds %struct.0, %struct.0* %0, i32 0, i32 4
+ %1 = bitcast %struct.0** %right to i32*
+ %2 = load i32, i32* %1, align 4, !tbaa !5
+ %3 = bitcast %struct.0* %0 to i8*
+ tail call void @bar(i8* %3, i32 20) #1
+ store i32 %2, i32* bitcast (%struct.0** @var to i32*), align 4, !tbaa !1
+ %4 = inttoptr i32 %2 to %struct.0*
+ %cmp = icmp eq i32 %2, 0
+ br i1 %cmp, label %while.end.loopexit, label %while.body
+
+while.end.loopexit: ; preds = %while.body
+ br label %while.end
+
+while.end: ; preds = %while.end.loopexit, %entry
+ ret void
+}
+
+attributes #0 = { optsize }
+attributes #1 = { nounwind optsize }
+
+!1 = !{!2, !2, i64 0}
+!2 = !{!"any pointer", !3, i64 0}
+!3 = !{!"omnipotent char", !4, i64 0}
+!4 = !{!"Simple C/C++ TBAA"}
+!5 = !{!6, !2, i64 16}
+!6 = !{!"0", !2, i64 0, !2, i64 4, !2, i64 8, !2, i64 12, !2, i64 16}
diff --git a/test/CodeGen/Hexagon/float-amode.ll b/test/CodeGen/Hexagon/float-amode.ll
new file mode 100644
index 000000000000..9804f48349f8
--- /dev/null
+++ b/test/CodeGen/Hexagon/float-amode.ll
@@ -0,0 +1,89 @@
+; RUN: llc -march=hexagon -fp-contract=fast -disable-hexagon-peephole -disable-hexagon-amodeopt < %s | FileCheck %s
+
+; The test checks for various addressing modes for floating point loads/stores.
+
+%struct.matrix_paramsGlob = type { [50 x i8], i16, [50 x float] }
+%struct.matrix_params = type { [50 x i8], i16, float** }
+%struct.matrix_params2 = type { i16, [50 x [50 x float]] }
+
+@globB = common global %struct.matrix_paramsGlob zeroinitializer, align 4
+@globA = common global %struct.matrix_paramsGlob zeroinitializer, align 4
+@b = common global float 0.000000e+00, align 4
+@a = common global float 0.000000e+00, align 4
+
+; CHECK-LABEL: test1
+; CHECK: [[REG11:(r[0-9]+)]]{{ *}}={{ *}}memw(r{{[0-9]+}} + r{{[0-9]+}}<<#2)
+; CHECK: [[REG12:(r[0-9]+)]] += sfmpy({{.*}}[[REG11]]
+; CHECK: memw(r{{[0-9]+}} + r{{[0-9]+}}<<#2) = [[REG12]].new
+
+; Function Attrs: norecurse nounwind
+define void @test1(%struct.matrix_params* nocapture readonly %params, i32 %col1) {
+entry:
+ %matrixA = getelementptr inbounds %struct.matrix_params, %struct.matrix_params* %params, i32 0, i32 2
+ %0 = load float**, float*** %matrixA, align 4
+ %arrayidx = getelementptr inbounds float*, float** %0, i32 2
+ %1 = load float*, float** %arrayidx, align 4
+ %arrayidx1 = getelementptr inbounds float, float* %1, i32 %col1
+ %2 = load float, float* %arrayidx1, align 4
+ %mul = fmul float %2, 2.000000e+01
+ %add = fadd float %mul, 1.000000e+01
+ %arrayidx3 = getelementptr inbounds float*, float** %0, i32 5
+ %3 = load float*, float** %arrayidx3, align 4
+ %arrayidx4 = getelementptr inbounds float, float* %3, i32 %col1
+ store float %add, float* %arrayidx4, align 4
+ ret void
+}
+
+; CHECK-LABEL: test2
+; CHECK: [[REG21:(r[0-9]+)]]{{ *}}={{ *}}memw(##globB+92)
+; CHECK: [[REG22:(r[0-9]+)]] = sfadd({{.*}}[[REG21]]
+; CHECK: memw(##globA+84) = [[REG22]]
+
+; Function Attrs: norecurse nounwind
+define void @test2(%struct.matrix_params* nocapture readonly %params, i32 %col1) {
+entry:
+ %matrixA = getelementptr inbounds %struct.matrix_params, %struct.matrix_params* %params, i32 0, i32 2
+ %0 = load float**, float*** %matrixA, align 4
+ %1 = load float*, float** %0, align 4
+ %arrayidx1 = getelementptr inbounds float, float* %1, i32 %col1
+ %2 = load float, float* %arrayidx1, align 4
+ %3 = load float, float* getelementptr inbounds (%struct.matrix_paramsGlob, %struct.matrix_paramsGlob* @globB, i32 0, i32 2, i32 10), align 4
+ %add = fadd float %2, %3
+ store float %add, float* getelementptr inbounds (%struct.matrix_paramsGlob, %struct.matrix_paramsGlob* @globA, i32 0, i32 2, i32 8), align 4
+ ret void
+}
+
+; CHECK-LABEL: test3
+; CHECK: [[REG31:(r[0-9]+)]]{{ *}}={{ *}}memw(#b)
+; CHECK: [[REG32:(r[0-9]+)]] = sfadd({{.*}}[[REG31]]
+; CHECK: memw(#a) = [[REG32]]
+
+; Function Attrs: norecurse nounwind
+define void @test3(%struct.matrix_params* nocapture readonly %params, i32 %col1) {
+entry:
+ %matrixA = getelementptr inbounds %struct.matrix_params, %struct.matrix_params* %params, i32 0, i32 2
+ %0 = load float**, float*** %matrixA, align 4
+ %1 = load float*, float** %0, align 4
+ %arrayidx1 = getelementptr inbounds float, float* %1, i32 %col1
+ %2 = load float, float* %arrayidx1, align 4
+ %3 = load float, float* @b, align 4
+ %add = fadd float %2, %3
+ store float %add, float* @a, align 4
+ ret void
+}
+
+; CHECK-LABEL: test4
+; CHECK: [[REG41:(r[0-9]+)]]{{ *}}={{ *}}memw(r0<<#2 + ##globB+52)
+; CHECK: [[REG42:(r[0-9]+)]] = sfadd({{.*}}[[REG41]]
+; CHECK: memw(r0<<#2 + ##globA+60) = [[REG42]]
+; Function Attrs: noinline norecurse nounwind
+define void @test4(i32 %col1) {
+entry:
+ %arrayidx = getelementptr inbounds %struct.matrix_paramsGlob, %struct.matrix_paramsGlob* @globB, i32 0, i32 2, i32 %col1
+ %0 = load float, float* %arrayidx, align 4
+ %add = fadd float %0, 0.000000e+00
+ %add1 = add nsw i32 %col1, 2
+ %arrayidx2 = getelementptr inbounds %struct.matrix_paramsGlob, %struct.matrix_paramsGlob* @globA, i32 0, i32 2, i32 %add1
+ store float %add, float* %arrayidx2, align 4
+ ret void
+}
diff --git a/test/CodeGen/Hexagon/fminmax.ll b/test/CodeGen/Hexagon/fminmax.ll
new file mode 100644
index 000000000000..7c1a9fb42f23
--- /dev/null
+++ b/test/CodeGen/Hexagon/fminmax.ll
@@ -0,0 +1,27 @@
+; RUN: llc -march=hexagon < %s | FileCheck %s
+
+target datalayout = "e-m:e-p:32:32:32-a:0-n16:32-i64:64:64-i32:32:32-i16:16:16-i1:8:8-f32:32:32-f64:64:64-v32:32:32-v64:64:64-v512:512:512-v1024:1024:1024-v2048:2048:2048"
+target triple = "hexagon"
+
+; CHECK-LABEL: minimum
+; CHECK: sfmin
+define float @minimum(float %x, float %y) #0 {
+entry:
+ %call = tail call float @fminf(float %x, float %y) #1
+ ret float %call
+}
+
+; CHECK-LABEL: maximum
+; CHECK: sfmax
+define float @maximum(float %x, float %y) #0 {
+entry:
+ %call = tail call float @fmaxf(float %x, float %y) #1
+ ret float %call
+}
+
+declare float @fminf(float, float) #0
+declare float @fmaxf(float, float) #0
+
+attributes #0 = { nounwind readnone "disable-tail-calls"="false" "less-precise-fpmad"="false" "no-frame-pointer-elim"="true" "no-frame-pointer-elim-non-leaf" "no-infs-fp-math"="false" "no-nans-fp-math"="false" "stack-protector-buffer-size"="8" "target-cpu"="hexagonv60" "target-features"="+hvx,-hvx-double" "unsafe-fp-math"="false" "use-soft-float"="false" }
+attributes #1 = { nounwind readnone }
+
diff --git a/test/CodeGen/Hexagon/frame-offset-overflow.ll b/test/CodeGen/Hexagon/frame-offset-overflow.ll
new file mode 100644
index 000000000000..43d5fd5ad0f0
--- /dev/null
+++ b/test/CodeGen/Hexagon/frame-offset-overflow.ll
@@ -0,0 +1,163 @@
+; RUN: llc -march=hexagon < %s | FileCheck %s
+
+; In reality, check that the compilation succeeded and that some code was
+; generated.
+; CHECK: vadd
+
+target triple = "hexagon"
+
+define void @fred(i16* noalias nocapture readonly %p0, i32 %p1, i32 %p2, i16* noalias nocapture %p3, i32 %p4) local_unnamed_addr #1 {
+entry:
+ %mul = mul i32 %p4, %p1
+ %add.ptr = getelementptr inbounds i16, i16* %p0, i32 %mul
+ %add = add nsw i32 %p4, 1
+ %rem = srem i32 %add, 5
+ %mul1 = mul i32 %rem, %p1
+ %add.ptr2 = getelementptr inbounds i16, i16* %p0, i32 %mul1
+ %add.ptr6 = getelementptr inbounds i16, i16* %p0, i32 0
+ %add7 = add nsw i32 %p4, 3
+ %rem8 = srem i32 %add7, 5
+ %mul9 = mul i32 %rem8, %p1
+ %add.ptr10 = getelementptr inbounds i16, i16* %p0, i32 %mul9
+ %add.ptr14 = getelementptr inbounds i16, i16* %p0, i32 0
+ %incdec.ptr18 = getelementptr inbounds i16, i16* %add.ptr14, i32 32
+ %0 = bitcast i16* %incdec.ptr18 to <16 x i32>*
+ %incdec.ptr17 = getelementptr inbounds i16, i16* %add.ptr10, i32 32
+ %1 = bitcast i16* %incdec.ptr17 to <16 x i32>*
+ %incdec.ptr16 = getelementptr inbounds i16, i16* %add.ptr6, i32 32
+ %2 = bitcast i16* %incdec.ptr16 to <16 x i32>*
+ %incdec.ptr15 = getelementptr inbounds i16, i16* %add.ptr2, i32 32
+ %3 = bitcast i16* %incdec.ptr15 to <16 x i32>*
+ %incdec.ptr = getelementptr inbounds i16, i16* %add.ptr, i32 32
+ %4 = bitcast i16* %incdec.ptr to <16 x i32>*
+ %5 = bitcast i16* %p3 to <16 x i32>*
+ br i1 undef, label %for.end.loopexit.unr-lcssa, label %for.body
+
+for.body: ; preds = %for.body, %entry
+ %optr.0102 = phi <16 x i32>* [ %incdec.ptr24.3, %for.body ], [ %5, %entry ]
+ %iptr4.0101 = phi <16 x i32>* [ %incdec.ptr23.3, %for.body ], [ %0, %entry ]
+ %iptr3.0100 = phi <16 x i32>* [ %incdec.ptr22.3, %for.body ], [ %1, %entry ]
+ %iptr2.099 = phi <16 x i32>* [ undef, %for.body ], [ %2, %entry ]
+ %iptr1.098 = phi <16 x i32>* [ %incdec.ptr20.3, %for.body ], [ %3, %entry ]
+ %iptr0.097 = phi <16 x i32>* [ %incdec.ptr19.3, %for.body ], [ %4, %entry ]
+ %dVsumv1.096 = phi <32 x i32> [ %66, %for.body ], [ undef, %entry ]
+ %niter = phi i32 [ %niter.nsub.3, %for.body ], [ undef, %entry ]
+ %6 = load <16 x i32>, <16 x i32>* %iptr0.097, align 64, !tbaa !1
+ %7 = load <16 x i32>, <16 x i32>* %iptr1.098, align 64, !tbaa !1
+ %8 = load <16 x i32>, <16 x i32>* %iptr2.099, align 64, !tbaa !1
+ %9 = load <16 x i32>, <16 x i32>* %iptr3.0100, align 64, !tbaa !1
+ %10 = load <16 x i32>, <16 x i32>* %iptr4.0101, align 64, !tbaa !1
+ %11 = tail call <32 x i32> @llvm.hexagon.V6.vaddhw(<16 x i32> %6, <16 x i32> %10)
+ %12 = tail call <32 x i32> @llvm.hexagon.V6.vmpyhsat.acc(<32 x i32> %11, <16 x i32> %8, i32 393222)
+ %13 = tail call <32 x i32> @llvm.hexagon.V6.vcombine(<16 x i32> %9, <16 x i32> %7)
+ %14 = tail call <32 x i32> @llvm.hexagon.V6.vmpahb.acc(<32 x i32> %12, <32 x i32> %13, i32 67372036)
+ %15 = tail call <16 x i32> @llvm.hexagon.V6.lo(<32 x i32> %dVsumv1.096)
+ %16 = tail call <16 x i32> @llvm.hexagon.V6.lo(<32 x i32> %14)
+ %17 = tail call <16 x i32> @llvm.hexagon.V6.valignbi(<16 x i32> %16, <16 x i32> %15, i32 4)
+ %18 = tail call <16 x i32> @llvm.hexagon.V6.hi(<32 x i32> %14)
+ %19 = tail call <16 x i32> @llvm.hexagon.V6.valignb(<16 x i32> %16, <16 x i32> %15, i32 8)
+ %20 = tail call <16 x i32> @llvm.hexagon.V6.valignb(<16 x i32> %18, <16 x i32> undef, i32 8)
+ %21 = tail call <16 x i32> @llvm.hexagon.V6.vaddw(<16 x i32> %17, <16 x i32> %19)
+ %22 = tail call <16 x i32> @llvm.hexagon.V6.vaddw(<16 x i32> %15, <16 x i32> %19)
+ %23 = tail call <16 x i32> @llvm.hexagon.V6.vmpyiwb.acc(<16 x i32> %22, <16 x i32> %17, i32 101058054)
+ %24 = tail call <16 x i32> @llvm.hexagon.V6.vmpyiwb.acc(<16 x i32> %23, <16 x i32> zeroinitializer, i32 67372036)
+ %25 = tail call <16 x i32> @llvm.hexagon.V6.vaddw(<16 x i32> undef, <16 x i32> %20)
+ %26 = tail call <16 x i32> @llvm.hexagon.V6.vmpyiwb.acc(<16 x i32> %25, <16 x i32> undef, i32 101058054)
+ %27 = tail call <16 x i32> @llvm.hexagon.V6.vmpyiwb.acc(<16 x i32> %26, <16 x i32> %21, i32 67372036)
+ %28 = tail call <16 x i32> @llvm.hexagon.V6.vasrwh(<16 x i32> %27, <16 x i32> %24, i32 8)
+ %incdec.ptr24 = getelementptr inbounds <16 x i32>, <16 x i32>* %optr.0102, i32 1
+ store <16 x i32> %28, <16 x i32>* %optr.0102, align 64, !tbaa !1
+ %incdec.ptr19.1 = getelementptr inbounds <16 x i32>, <16 x i32>* %iptr0.097, i32 2
+ %incdec.ptr23.1 = getelementptr inbounds <16 x i32>, <16 x i32>* %iptr4.0101, i32 2
+ %29 = tail call <16 x i32> @llvm.hexagon.V6.lo(<32 x i32> %14)
+ %30 = tail call <16 x i32> @llvm.hexagon.V6.hi(<32 x i32> %14)
+ %31 = tail call <16 x i32> @llvm.hexagon.V6.valignbi(<16 x i32> undef, <16 x i32> %29, i32 4)
+ %32 = tail call <16 x i32> @llvm.hexagon.V6.valignbi(<16 x i32> undef, <16 x i32> %30, i32 4)
+ %33 = tail call <16 x i32> @llvm.hexagon.V6.valignb(<16 x i32> undef, <16 x i32> %29, i32 8)
+ %34 = tail call <16 x i32> @llvm.hexagon.V6.valignb(<16 x i32> undef, <16 x i32> %30, i32 8)
+ %35 = tail call <16 x i32> @llvm.hexagon.V6.vaddw(<16 x i32> %31, <16 x i32> %33)
+ %36 = tail call <16 x i32> @llvm.hexagon.V6.vaddw(<16 x i32> %29, <16 x i32> %33)
+ %37 = tail call <16 x i32> @llvm.hexagon.V6.vmpyiwb.acc(<16 x i32> %36, <16 x i32> %31, i32 101058054)
+ %38 = tail call <16 x i32> @llvm.hexagon.V6.vmpyiwb.acc(<16 x i32> %37, <16 x i32> undef, i32 67372036)
+ %39 = tail call <16 x i32> @llvm.hexagon.V6.vaddw(<16 x i32> %30, <16 x i32> %34)
+ %40 = tail call <16 x i32> @llvm.hexagon.V6.vmpyiwb.acc(<16 x i32> %39, <16 x i32> %32, i32 101058054)
+ %41 = tail call <16 x i32> @llvm.hexagon.V6.vmpyiwb.acc(<16 x i32> %40, <16 x i32> %35, i32 67372036)
+ %42 = tail call <16 x i32> @llvm.hexagon.V6.vasrwh(<16 x i32> %41, <16 x i32> %38, i32 8)
+ %incdec.ptr24.1 = getelementptr inbounds <16 x i32>, <16 x i32>* %optr.0102, i32 2
+ store <16 x i32> %42, <16 x i32>* %incdec.ptr24, align 64, !tbaa !1
+ %incdec.ptr19.2 = getelementptr inbounds <16 x i32>, <16 x i32>* %iptr0.097, i32 3
+ %43 = load <16 x i32>, <16 x i32>* %incdec.ptr19.1, align 64, !tbaa !1
+ %incdec.ptr20.2 = getelementptr inbounds <16 x i32>, <16 x i32>* %iptr1.098, i32 3
+ %incdec.ptr21.2 = getelementptr inbounds <16 x i32>, <16 x i32>* %iptr2.099, i32 3
+ %incdec.ptr22.2 = getelementptr inbounds <16 x i32>, <16 x i32>* %iptr3.0100, i32 3
+ %incdec.ptr23.2 = getelementptr inbounds <16 x i32>, <16 x i32>* %iptr4.0101, i32 3
+ %44 = load <16 x i32>, <16 x i32>* %incdec.ptr23.1, align 64, !tbaa !1
+ %45 = tail call <32 x i32> @llvm.hexagon.V6.vaddhw(<16 x i32> %43, <16 x i32> %44)
+ %46 = tail call <32 x i32> @llvm.hexagon.V6.vmpyhsat.acc(<32 x i32> %45, <16 x i32> undef, i32 393222)
+ %47 = tail call <32 x i32> @llvm.hexagon.V6.vmpahb.acc(<32 x i32> %46, <32 x i32> undef, i32 67372036)
+ %48 = tail call <16 x i32> @llvm.hexagon.V6.lo(<32 x i32> %47)
+ %49 = tail call <16 x i32> @llvm.hexagon.V6.valignbi(<16 x i32> %48, <16 x i32> undef, i32 4)
+ %50 = tail call <16 x i32> @llvm.hexagon.V6.valignb(<16 x i32> %48, <16 x i32> undef, i32 8)
+ %51 = tail call <16 x i32> @llvm.hexagon.V6.vaddw(<16 x i32> zeroinitializer, <16 x i32> undef)
+ %52 = tail call <16 x i32> @llvm.hexagon.V6.vaddw(<16 x i32> %49, <16 x i32> %50)
+ %53 = tail call <16 x i32> @llvm.hexagon.V6.vaddw(<16 x i32> undef, <16 x i32> %50)
+ %54 = tail call <16 x i32> @llvm.hexagon.V6.vmpyiwb.acc(<16 x i32> %53, <16 x i32> %49, i32 101058054)
+ %55 = tail call <16 x i32> @llvm.hexagon.V6.vmpyiwb.acc(<16 x i32> %54, <16 x i32> %51, i32 67372036)
+ %56 = tail call <16 x i32> @llvm.hexagon.V6.vmpyiwb.acc(<16 x i32> undef, <16 x i32> %52, i32 67372036)
+ %57 = tail call <16 x i32> @llvm.hexagon.V6.vasrwh(<16 x i32> %56, <16 x i32> %55, i32 8)
+ %incdec.ptr24.2 = getelementptr inbounds <16 x i32>, <16 x i32>* %optr.0102, i32 3
+ store <16 x i32> %57, <16 x i32>* %incdec.ptr24.1, align 64, !tbaa !1
+ %incdec.ptr19.3 = getelementptr inbounds <16 x i32>, <16 x i32>* %iptr0.097, i32 4
+ %58 = load <16 x i32>, <16 x i32>* %incdec.ptr19.2, align 64, !tbaa !1
+ %incdec.ptr20.3 = getelementptr inbounds <16 x i32>, <16 x i32>* %iptr1.098, i32 4
+ %59 = load <16 x i32>, <16 x i32>* %incdec.ptr20.2, align 64, !tbaa !1
+ %60 = load <16 x i32>, <16 x i32>* %incdec.ptr21.2, align 64, !tbaa !1
+ %incdec.ptr22.3 = getelementptr inbounds <16 x i32>, <16 x i32>* %iptr3.0100, i32 4
+ %61 = load <16 x i32>, <16 x i32>* %incdec.ptr22.2, align 64, !tbaa !1
+ %incdec.ptr23.3 = getelementptr inbounds <16 x i32>, <16 x i32>* %iptr4.0101, i32 4
+ %62 = load <16 x i32>, <16 x i32>* %incdec.ptr23.2, align 64, !tbaa !1
+ %63 = tail call <32 x i32> @llvm.hexagon.V6.vaddhw(<16 x i32> %58, <16 x i32> %62)
+ %64 = tail call <32 x i32> @llvm.hexagon.V6.vmpyhsat.acc(<32 x i32> %63, <16 x i32> %60, i32 393222)
+ %65 = tail call <32 x i32> @llvm.hexagon.V6.vcombine(<16 x i32> %61, <16 x i32> %59)
+ %66 = tail call <32 x i32> @llvm.hexagon.V6.vmpahb.acc(<32 x i32> %64, <32 x i32> %65, i32 67372036)
+ %67 = tail call <16 x i32> @llvm.hexagon.V6.hi(<32 x i32> %47)
+ %68 = tail call <16 x i32> @llvm.hexagon.V6.lo(<32 x i32> %66)
+ %69 = tail call <16 x i32> @llvm.hexagon.V6.valignbi(<16 x i32> %68, <16 x i32> undef, i32 4)
+ %70 = tail call <16 x i32> @llvm.hexagon.V6.hi(<32 x i32> %66)
+ %71 = tail call <16 x i32> @llvm.hexagon.V6.valignbi(<16 x i32> %70, <16 x i32> %67, i32 4)
+ %72 = tail call <16 x i32> @llvm.hexagon.V6.valignb(<16 x i32> %70, <16 x i32> %67, i32 8)
+ %73 = tail call <16 x i32> @llvm.hexagon.V6.vaddw(<16 x i32> %67, <16 x i32> %71)
+ %74 = tail call <16 x i32> @llvm.hexagon.V6.vmpyiwb.acc(<16 x i32> undef, <16 x i32> %69, i32 101058054)
+ %75 = tail call <16 x i32> @llvm.hexagon.V6.vmpyiwb.acc(<16 x i32> %74, <16 x i32> %73, i32 67372036)
+ %76 = tail call <16 x i32> @llvm.hexagon.V6.vaddw(<16 x i32> %67, <16 x i32> %72)
+ %77 = tail call <16 x i32> @llvm.hexagon.V6.vmpyiwb.acc(<16 x i32> %76, <16 x i32> %71, i32 101058054)
+ %78 = tail call <16 x i32> @llvm.hexagon.V6.vmpyiwb.acc(<16 x i32> %77, <16 x i32> undef, i32 67372036)
+ %79 = tail call <16 x i32> @llvm.hexagon.V6.vasrwh(<16 x i32> %78, <16 x i32> %75, i32 8)
+ %incdec.ptr24.3 = getelementptr inbounds <16 x i32>, <16 x i32>* %optr.0102, i32 4
+ store <16 x i32> %79, <16 x i32>* %incdec.ptr24.2, align 64, !tbaa !1
+ %niter.nsub.3 = add i32 %niter, -4
+ %niter.ncmp.3 = icmp eq i32 %niter.nsub.3, 0
+ br i1 %niter.ncmp.3, label %for.end.loopexit.unr-lcssa, label %for.body
+
+for.end.loopexit.unr-lcssa: ; preds = %for.body, %entry
+ ret void
+}
+
+declare <16 x i32> @llvm.hexagon.V6.hi(<32 x i32>) #0
+declare <16 x i32> @llvm.hexagon.V6.lo(<32 x i32>) #0
+declare <16 x i32> @llvm.hexagon.V6.vaddw(<16 x i32>, <16 x i32>) #0
+declare <16 x i32> @llvm.hexagon.V6.valignb(<16 x i32>, <16 x i32>, i32) #0
+declare <16 x i32> @llvm.hexagon.V6.valignbi(<16 x i32>, <16 x i32>, i32) #0
+declare <16 x i32> @llvm.hexagon.V6.vasrwh(<16 x i32>, <16 x i32>, i32) #0
+declare <16 x i32> @llvm.hexagon.V6.vmpyiwb.acc(<16 x i32>, <16 x i32>, i32) #0
+declare <32 x i32> @llvm.hexagon.V6.vaddhw(<16 x i32>, <16 x i32>) #0
+declare <32 x i32> @llvm.hexagon.V6.vcombine(<16 x i32>, <16 x i32>) #0
+declare <32 x i32> @llvm.hexagon.V6.vmpahb.acc(<32 x i32>, <32 x i32>, i32) #0
+declare <32 x i32> @llvm.hexagon.V6.vmpyhsat.acc(<32 x i32>, <16 x i32>, i32) #0
+
+attributes #0 = { nounwind readnone }
+attributes #1 = { nounwind "target-cpu"="hexagonv60" "target-features"="+hvx,-hvx-double" }
+
+!1 = !{!2, !2, i64 0}
+!2 = !{!"omnipotent char", !3, i64 0}
+!3 = !{!"Simple C/C++ TBAA"}
diff --git a/test/CodeGen/Hexagon/fsel.ll b/test/CodeGen/Hexagon/fsel.ll
new file mode 100644
index 000000000000..247249da50b1
--- /dev/null
+++ b/test/CodeGen/Hexagon/fsel.ll
@@ -0,0 +1,22 @@
+; RUN: llc -march=hexagon -O0 < %s | FileCheck %s
+
+; CHECK-LABEL: danny:
+; CHECK: mux(p0, r1, ##1065353216)
+
+define float @danny(i32 %x, float %f) #0 {
+ %t = icmp sgt i32 %x, 0
+ %u = select i1 %t, float %f, float 1.0
+ ret float %u
+}
+
+; CHECK-LABEL: sammy:
+; CHECK: mux(p0, ##1069547520, r1)
+
+define float @sammy(i32 %x, float %f) #0 {
+ %t = icmp sgt i32 %x, 0
+ %u = select i1 %t, float 1.5, float %f
+ ret float %u
+}
+
+attributes #0 = { nounwind "target-cpu"="hexagonv5" }
+
diff --git a/test/CodeGen/Hexagon/hwloop-crit-edge.ll b/test/CodeGen/Hexagon/hwloop-crit-edge.ll
index 4de4540c142e..f6e08eaf2e0c 100644
--- a/test/CodeGen/Hexagon/hwloop-crit-edge.ll
+++ b/test/CodeGen/Hexagon/hwloop-crit-edge.ll
@@ -1,4 +1,5 @@
; RUN: llc -O3 -march=hexagon -mcpu=hexagonv5 < %s | FileCheck %s
+; XFAIL: *
;
; Generate hardware loop when loop 'latch' block is different
; from the loop 'exiting' block.
diff --git a/test/CodeGen/Hexagon/hwloop-loop1.ll b/test/CodeGen/Hexagon/hwloop-loop1.ll
index 8b02736e0374..238d34e7ea15 100644
--- a/test/CodeGen/Hexagon/hwloop-loop1.ll
+++ b/test/CodeGen/Hexagon/hwloop-loop1.ll
@@ -2,8 +2,6 @@
;
; Generate loop1 instruction for double loop sequence.
-; CHECK: loop0(.LBB{{.}}_{{.}}, #100)
-; CHECK: endloop0
; CHECK: loop1(.LBB{{.}}_{{.}}, #100)
; CHECK: loop0(.LBB{{.}}_{{.}}, #100)
; CHECK: endloop0
diff --git a/test/CodeGen/Hexagon/hwloop-noreturn-call.ll b/test/CodeGen/Hexagon/hwloop-noreturn-call.ll
new file mode 100644
index 000000000000..1045e2ed80a7
--- /dev/null
+++ b/test/CodeGen/Hexagon/hwloop-noreturn-call.ll
@@ -0,0 +1,63 @@
+; RUN: llc -march=hexagon < %s | FileCheck %s
+
+target triple = "hexagon"
+
+; CHECK-LABEL: danny:
+; CHECK-DAG: loop0
+; CHECK-DAG: call trap
+define void @danny(i32* %p, i32 %n, i32 %k) #0 {
+entry:
+ br label %for.body
+
+for.body: ; preds = %entry
+ %t0 = phi i32 [ 0, %entry ], [ %t1, %for.cont ]
+ %t1 = add i32 %t0, 1
+ %t2 = getelementptr i32, i32* %p, i32 %t0
+ store i32 %t1, i32* %t2, align 4
+ %c = icmp sgt i32 %t1, %k
+ br i1 %c, label %noret, label %for.cont
+
+for.cont:
+ %cmp = icmp slt i32 %t0, %n
+ br i1 %cmp, label %for.body, label %for.end
+
+for.end: ; preds = %for.cond
+ ret void
+
+noret:
+ call void @trap() #1
+ br label %for.cont
+}
+
+; CHECK-LABEL: sammy:
+; CHECK-DAG: loop0
+; CHECK-DAG: callr
+define void @sammy(i32* %p, i32 %n, i32 %k, void (...)* %f) #0 {
+entry:
+ br label %for.body
+
+for.body: ; preds = %entry
+ %t0 = phi i32 [ 0, %entry ], [ %t1, %for.cont ]
+ %t1 = add i32 %t0, 1
+ %t2 = getelementptr i32, i32* %p, i32 %t0
+ store i32 %t1, i32* %t2, align 4
+ %c = icmp sgt i32 %t1, %k
+ br i1 %c, label %noret, label %for.cont
+
+for.cont:
+ %cmp = icmp slt i32 %t0, %n
+ br i1 %cmp, label %for.body, label %for.end
+
+for.end: ; preds = %for.cond
+ ret void
+
+noret:
+ call void (...) %f() #1
+ br label %for.cont
+}
+
+declare void @trap() #1
+
+attributes #0 = { nounwind "target-cpu"="hexagonv60" "target-features"="+hvx,-hvx-double" }
+attributes #1 = { nounwind noreturn }
+
diff --git a/test/CodeGen/Hexagon/hwloop-preh.ll b/test/CodeGen/Hexagon/hwloop-preh.ll
new file mode 100644
index 000000000000..e92461f43da5
--- /dev/null
+++ b/test/CodeGen/Hexagon/hwloop-preh.ll
@@ -0,0 +1,44 @@
+; RUN: llc -march=hexagon -disable-machine-licm -hwloop-spec-preheader=1 < %s | FileCheck %s
+; CHECK: loop0
+
+target triple = "hexagon"
+
+define i32 @foo(i32 %x, i32 %n, i32* nocapture %A, i32* nocapture %B) #0 {
+entry:
+ %cmp = icmp sgt i32 %x, 0
+ br i1 %cmp, label %for.cond.preheader, label %return
+
+for.cond.preheader: ; preds = %entry
+ %cmp16 = icmp sgt i32 %n, 0
+ br i1 %cmp16, label %for.body.preheader, label %return
+
+for.body.preheader: ; preds = %for.cond.preheader
+ br label %for.body
+
+for.body: ; preds = %for.body.preheader, %for.body
+ %arrayidx.phi = phi i32* [ %arrayidx.inc, %for.body ], [ %B, %for.body.preheader ]
+ %arrayidx2.phi = phi i32* [ %arrayidx2.inc, %for.body ], [ %A, %for.body.preheader ]
+ %i.07 = phi i32 [ %inc, %for.body ], [ 0, %for.body.preheader ]
+ %0 = load i32, i32* %arrayidx.phi, align 4, !tbaa !0
+ %1 = load i32, i32* %arrayidx2.phi, align 4, !tbaa !0
+ %add = add nsw i32 %1, %0
+ store i32 %add, i32* %arrayidx2.phi, align 4, !tbaa !0
+ %inc = add nsw i32 %i.07, 1
+ %exitcond = icmp eq i32 %inc, %n
+ %arrayidx.inc = getelementptr i32, i32* %arrayidx.phi, i32 1
+ %arrayidx2.inc = getelementptr i32, i32* %arrayidx2.phi, i32 1
+ br i1 %exitcond, label %return.loopexit, label %for.body
+
+return.loopexit: ; preds = %for.body
+ br label %return
+
+return: ; preds = %return.loopexit, %for.cond.preheader, %entry
+ %retval.0 = phi i32 [ 2, %entry ], [ 0, %for.cond.preheader ], [ 0, %return.loopexit ]
+ ret i32 %retval.0
+}
+
+!0 = !{!"int", !1}
+!1 = !{!"omnipotent char", !2}
+!2 = !{!"Simple C/C++ TBAA"}
+
+attributes #0 = { nounwind "target-cpu"="hexagonv60" "target-features"="-hvx,-hvx-double" }
diff --git a/test/CodeGen/Hexagon/hwloop1.ll b/test/CodeGen/Hexagon/hwloop1.ll
index 97b779cf9628..68af3b34eeeb 100644
--- a/test/CodeGen/Hexagon/hwloop1.ll
+++ b/test/CodeGen/Hexagon/hwloop1.ll
@@ -1,4 +1,4 @@
-; RUN: llc -march=hexagon < %s | FileCheck %s
+; RUN: llc -march=hexagon -enable-pipeliner=false < %s | FileCheck %s
; Check that we generate hardware loop instructions.
; Case 1 : Loop with a constant number of iterations.
diff --git a/test/CodeGen/Hexagon/ifcvt-diamond-bug-2016-08-26.ll b/test/CodeGen/Hexagon/ifcvt-diamond-bug-2016-08-26.ll
new file mode 100644
index 000000000000..68a5dc16ecff
--- /dev/null
+++ b/test/CodeGen/Hexagon/ifcvt-diamond-bug-2016-08-26.ll
@@ -0,0 +1,37 @@
+; RUN: llc -march=hexagon -o - %s | FileCheck %s
+target triple = "hexagon"
+
+%struct.0 = type { i16, i16 }
+
+@t = external local_unnamed_addr global %struct.0, align 2
+
+define void @foo(i32 %p) local_unnamed_addr #0 {
+entry:
+ %conv90 = trunc i32 %p to i16
+ %call105 = call signext i16 @bar(i16 signext 16384, i16 signext undef) #0
+ %call175 = call signext i16 @bar(i16 signext %conv90, i16 signext 4) #0
+ %call197 = call signext i16 @bar(i16 signext %conv90, i16 signext 4) #0
+ %cmp199 = icmp eq i16 %call197, 0
+ br i1 %cmp199, label %if.then200, label %if.else201
+
+; CHECK-DAG: [[R4:r[0-9]+]] = #4
+; CHECK: p0 = cmp.eq(r0, #0)
+; CHECK: if (!p0.new) [[R3:r[0-9]+]] = #3
+; CHECK-DAG: if (!p0) memh(##t) = [[R3]]
+; CHECK-DAG: if (p0) memh(##t) = [[R4]]
+if.then200: ; preds = %entry
+ store i16 4, i16* getelementptr inbounds (%struct.0, %struct.0* @t, i32 0, i32 0), align 2
+ store i16 0, i16* getelementptr inbounds (%struct.0, %struct.0* @t, i32 0, i32 1), align 2
+ br label %if.end202
+
+if.else201: ; preds = %entry
+ store i16 3, i16* getelementptr inbounds (%struct.0, %struct.0* @t, i32 0, i32 0), align 2
+ br label %if.end202
+
+if.end202: ; preds = %if.else201, %if.then200
+ ret void
+}
+
+declare signext i16 @bar(i16 signext, i16 signext) local_unnamed_addr #0
+
+attributes #0 = { optsize "target-cpu"="hexagonv55" }
diff --git a/test/CodeGen/Hexagon/ifcvt-impuse-livein.mir b/test/CodeGen/Hexagon/ifcvt-impuse-livein.mir
new file mode 100644
index 000000000000..780b9cedf7fa
--- /dev/null
+++ b/test/CodeGen/Hexagon/ifcvt-impuse-livein.mir
@@ -0,0 +1,42 @@
+# RUN: llc -march=hexagon -run-pass if-converter %s -o - | FileCheck %s
+
+# Make sure that the necessary implicit uses are added to predicated
+# instructions.
+
+# CHECK-LABEL: name: foo
+
+--- |
+ define void @foo() {
+ ret void
+ }
+...
+
+---
+name: foo
+tracksRegLiveness: true
+body: |
+ bb.0:
+ successors: %bb.1, %bb.2
+ liveins: %r0, %r2, %p1
+ J2_jumpf %p1, %bb.1, implicit-def %pc
+ J2_jump %bb.2, implicit-def %pc
+ bb.1:
+ successors: %bb.3
+ liveins: %r2
+ %r0 = A2_tfrsi 2
+ J2_jump %bb.3, implicit-def %pc
+ bb.2:
+ successors: %bb.3
+ liveins: %r0
+ ; Even though r2 was not live on entry to this block, it was live across
+ ; block bb.1 in the original diamond. After if-conversion, the diamond
+ ; became a single block, and so r2 is now live on entry to the instructions
+ ; originating from bb.2.
+ ; CHECK: %r2 = C2_cmoveit %p1, 1, implicit %r2
+ %r2 = A2_tfrsi 1
+ bb.3:
+ liveins: %r0, %r2
+ %r0 = A2_add %r0, %r2
+ J2_jumpr %r31, implicit-def %pc
+...
+
diff --git a/test/CodeGen/Hexagon/ifcvt-live-subreg.mir b/test/CodeGen/Hexagon/ifcvt-live-subreg.mir
new file mode 100644
index 000000000000..12cc086bd34b
--- /dev/null
+++ b/test/CodeGen/Hexagon/ifcvt-live-subreg.mir
@@ -0,0 +1,50 @@
+# RUN: llc -march=hexagon -run-pass if-converter -o - %s | FileCheck %s
+# Check that an implicit use is generated for a predicated instruction
+# when a subregister of the redefined register is live.
+
+# CHECK-LABEL: name: foo
+
+# Verify the predicated block:
+# CHECK-LABEL: bb.0:
+# CHECK: liveins: %r0, %r1, %p0, %d8
+# CHECK: %d8 = A2_combinew killed %r0, killed %r1
+# CHECK: %d8 = L2_ploadrdf_io %p0, %r29, 0, implicit %d8
+# CHECK: J2_jumprf %p0, killed %r31, implicit-def %pc, implicit-def %pc, implicit killed %d8
+
+--- |
+ define void @foo() {
+ ret void
+ }
+...
+
+
+---
+name: foo
+alignment: 4
+tracksRegLiveness: true
+liveins:
+ - { reg: '%r0' }
+ - { reg: '%r1' }
+ - { reg: '%p0' }
+ - { reg: '%d8' }
+body: |
+ bb.0:
+ successors: %bb.1, %bb.2
+ liveins: %r0, %r1, %p0, %d8
+ %d8 = A2_combinew killed %r0, killed %r1
+ J2_jumpf killed %p0, %bb.2, implicit-def %pc
+
+ bb.1:
+ liveins: %d0, %r17
+ %r0 = A2_tfrsi 0
+ %r1 = A2_tfrsi 0
+ A2_nop ; non-predicable
+ J2_jumpr killed %r31, implicit-def dead %pc, implicit killed %d0
+
+ bb.2:
+ ; Predicate this block.
+ %d8 = L2_loadrd_io %r29, 0
+ J2_jumpr killed %r31, implicit-def dead %pc, implicit killed %d8
+
+...
+
diff --git a/test/CodeGen/Hexagon/inline-asm-hexagon.ll b/test/CodeGen/Hexagon/inline-asm-hexagon.ll
new file mode 100644
index 000000000000..302096d49b3e
--- /dev/null
+++ b/test/CodeGen/Hexagon/inline-asm-hexagon.ll
@@ -0,0 +1,16 @@
+; RUN: llc -march=hexagon < %s | FileCheck %s
+
+target triple = "hexagon"
+
+;CHECK: [[REGH:r[0-9]]]:[[REGL:[0-9]]] = memd_locked
+;CHECK: HIGH([[REGH]])
+;CHECK: LOW(r[[REGL]])
+define i32 @fred(i64* %free_list_ptr, i32** %item_ptr, i8** %free_item_ptr) nounwind {
+entry:
+ %free_list_ptr.addr = alloca i64*, align 4
+ store i64* %free_list_ptr, i64** %free_list_ptr.addr, align 4
+ %0 = load i32*, i32** %item_ptr, align 4
+ %1 = call { i64, i32 } asm sideeffect "1: $0 = memd_locked($5)\0A\09 $1 = HIGH(${0:H}) \0A\09 $1 = add($1,#1) \0A\09 memw($6) = LOW(${0:L}) \0A\09 $0 = combine($7,$1) \0A\09 memd_locked($5,p0) = $0 \0A\09 if !p0 jump 1b\0A\09", "=&r,=&r,=*m,=*m,r,r,r,r,*m,*m,~{p0}"(i64** %free_list_ptr.addr, i8** %free_item_ptr, i64 0, i64* %free_list_ptr, i8** %free_item_ptr, i32* %0, i64** %free_list_ptr.addr, i8** %free_item_ptr) nounwind
+ %asmresult1 = extractvalue { i64, i32 } %1, 1
+ ret i32 %asmresult1
+}
diff --git a/test/CodeGen/Hexagon/inline-asm-i1.ll b/test/CodeGen/Hexagon/inline-asm-i1.ll
new file mode 100644
index 000000000000..cc1f4ce55dad
--- /dev/null
+++ b/test/CodeGen/Hexagon/inline-asm-i1.ll
@@ -0,0 +1,14 @@
+; RUN: llc -march=hexagon < %s | FileCheck %s
+; CHECK: r[[REG0:[0-9]+]] = usr
+; CHECK: [[REG0]] = insert(r{{[0-9]+}}, #1, #16)
+
+target triple = "hexagon"
+
+define hidden void @fred() #0 {
+entry:
+ %0 = call { i32, i32 } asm sideeffect " $0 = usr\0A $1 = $2\0A $0 = insert($1, #1, #16)\0Ausr = $0 \0A", "=&r,=&r,r"(i1 undef) #1
+ ret void
+}
+
+attributes #0 = { nounwind "target-cpu"="hexagonv60" }
+attributes #1 = { nounwind }
diff --git a/test/CodeGen/Hexagon/insert4.ll b/test/CodeGen/Hexagon/insert4.ll
index 96c8bba24d7c..c4d575dd4060 100644
--- a/test/CodeGen/Hexagon/insert4.ll
+++ b/test/CodeGen/Hexagon/insert4.ll
@@ -1,9 +1,9 @@
; RUN: llc -march=hexagon < %s | FileCheck %s
-; Check that we are generating insert instructions.
-; CHECK: insert
-; CHECK: insert
-; CHECK: insert
-; CHECK: insert
+;
+; Check that we no longer generate 4 inserts.
+; CHECK: combine(r{{[0-9]+}}.l, r{{[0-9]+}}.l)
+; CHECK: combine(r{{[0-9]+}}.l, r{{[0-9]+}}.l)
+; CHECK-NOT: insert
target datalayout = "e-p:32:32:32-i64:64:64-i32:32:32-i16:16:16-i1:32:32-f64:64:64-f32:32:32-v64:64:64-v32:32:32-a0:0-n16:32"
target triple = "hexagon"
diff --git a/test/CodeGen/Hexagon/intrinsics/llsc_bundling.ll b/test/CodeGen/Hexagon/intrinsics/llsc_bundling.ll
new file mode 100644
index 000000000000..966945b66f47
--- /dev/null
+++ b/test/CodeGen/Hexagon/intrinsics/llsc_bundling.ll
@@ -0,0 +1,12 @@
+; RUN: llc -march=hexagon < %s
+target triple = "hexagon-unknown--elf"
+
+; Function Attrs: norecurse nounwind
+define void @_Z4lockv() #0 {
+entry:
+ %__shared_owners = alloca i32, align 4
+ %0 = cmpxchg weak i32* %__shared_owners, i32 0, i32 1 seq_cst seq_cst
+ ret void
+}
+
+attributes #0 = { nounwind }
diff --git a/test/CodeGen/Hexagon/is-legal-void.ll b/test/CodeGen/Hexagon/is-legal-void.ll
new file mode 100644
index 000000000000..222934abb82d
--- /dev/null
+++ b/test/CodeGen/Hexagon/is-legal-void.ll
@@ -0,0 +1,58 @@
+; RUN: llc -march=hexagon < %s
+; REQUIRES: asserts
+
+; The two loads based on %struct.0, loading two different data types
+; cause LSR to assume type "void" for the memory type. This would then
+; cause an assert in isLegalAddressingMode. Make sure we no longer crash.
+
+target triple = "hexagon"
+
+%struct.0 = type { i8*, i8, %union.anon.0 }
+%union.anon.0 = type { i8* }
+
+define hidden fastcc void @fred() unnamed_addr #0 {
+entry:
+ br i1 undef, label %while.end, label %while.body.lr.ph
+
+while.body.lr.ph: ; preds = %entry
+ br label %while.body
+
+while.body: ; preds = %exit.2, %while.body.lr.ph
+ %lsr.iv = phi %struct.0* [ %cgep22, %exit.2 ], [ undef, %while.body.lr.ph ]
+ switch i32 undef, label %exit [
+ i32 1, label %sw.bb.i
+ i32 2, label %sw.bb3.i
+ ]
+
+sw.bb.i: ; preds = %while.body
+ unreachable
+
+sw.bb3.i: ; preds = %while.body
+ unreachable
+
+exit: ; preds = %while.body
+ switch i32 undef, label %exit.2 [
+ i32 1, label %sw.bb.i17
+ i32 2, label %sw.bb3.i20
+ ]
+
+sw.bb.i17: ; preds = %.exit
+ %0 = bitcast %struct.0* %lsr.iv to i32*
+ %1 = load i32, i32* %0, align 4
+ unreachable
+
+sw.bb3.i20: ; preds = %exit
+ %2 = bitcast %struct.0* %lsr.iv to i8**
+ %3 = load i8*, i8** %2, align 4
+ unreachable
+
+exit.2: ; preds = %exit
+ %cgep22 = getelementptr %struct.0, %struct.0* %lsr.iv, i32 1
+ br label %while.body
+
+while.end: ; preds = %entry
+ ret void
+}
+
+attributes #0 = { nounwind optsize "target-cpu"="hexagonv55" }
+
diff --git a/test/CodeGen/Hexagon/livephysregs-lane-masks.mir b/test/CodeGen/Hexagon/livephysregs-lane-masks.mir
new file mode 100644
index 000000000000..b2e1968bb59a
--- /dev/null
+++ b/test/CodeGen/Hexagon/livephysregs-lane-masks.mir
@@ -0,0 +1,40 @@
+# RUN: llc -march=hexagon -run-pass if-converter -verify-machineinstrs -o - %s | FileCheck %s
+
+# CHECK-LABEL: name: foo
+# CHECK: %p0 = C2_cmpeqi %r16, 0
+# Make sure there is no implicit use of r1.
+# CHECK: %r1 = L2_ploadruhf_io %p0, %r29, 6
+
+--- |
+ define void @foo() {
+ ret void
+ }
+...
+
+
+---
+name: foo
+tracksRegLiveness: true
+
+body: |
+ bb.0:
+ liveins: %r16
+ successors: %bb.1, %bb.2
+ %p0 = C2_cmpeqi %r16, 0
+ J2_jumpt %p0, %bb.2, implicit-def %pc
+
+ bb.1:
+ ; The lane mask %d0:0002 is equivalent to %r0. LivePhysRegs would ignore
+ ; it and treat it as the whole %d0, which is a pair %r1, %r0. The extra
+ ; %r1 would cause an (undefined) implicit use to be added during
+ ; if-conversion.
+ liveins: %d0:0x00000002, %d15:0x00000001, %r16
+ successors: %bb.2
+ %r1 = L2_loadruh_io %r29, 6
+ S2_storeri_io killed %r16, 0, %r1
+
+ bb.2:
+ liveins: %r0
+ %d8 = L2_loadrd_io %r29, 8
+ L4_return implicit-def %r29, implicit-def %r30, implicit-def %r31, implicit-def %pc, implicit %r30
+
diff --git a/test/CodeGen/Hexagon/livephysregs-lane-masks2.mir b/test/CodeGen/Hexagon/livephysregs-lane-masks2.mir
new file mode 100644
index 000000000000..586857016551
--- /dev/null
+++ b/test/CodeGen/Hexagon/livephysregs-lane-masks2.mir
@@ -0,0 +1,55 @@
+# RUN: llc -march=hexagon -verify-machineinstrs -run-pass branch-folder -o - %s | FileCheck %s
+
+# CHECK-LABEL: name: fred
+
+--- |
+ define void @fred() { ret void }
+
+...
+---
+
+name: fred
+tracksRegLiveness: true
+
+body: |
+ bb.0:
+ liveins: %p2, %r0
+ successors: %bb.1, %bb.2
+ J2_jumpt killed %p2, %bb.1, implicit-def %pc
+ J2_jump %bb.2, implicit-def %pc
+
+ bb.1:
+ liveins: %r0, %r19
+ successors: %bb.3
+ %r2 = A2_tfrsi 4
+ %r1 = COPY %r19
+ %r0 = S2_asl_r_r killed %r0, killed %r2
+ %r0 = A2_asrh killed %r0
+ J2_jump %bb.3, implicit-def %pc
+
+ bb.2:
+ liveins: %r0, %r18
+ successors: %bb.3
+ %r2 = A2_tfrsi 5
+ %r1 = L2_loadrh_io %r18, 0
+ %r0 = S2_asl_r_r killed %r0, killed %r2
+ %r0 = A2_asrh killed %r0
+
+ bb.3:
+ ; A live-in register without subregs, but with a lane mask that is not ~0
+ ; is not recognized by LivePhysRegs. Branch folding exposes this problem
+ ; (through tail merging).
+ ;
+ ; CHECK: bb.3:
+ ; CHECK: liveins:{{.*}}%p0
+ ; CHECK: %r0 = S2_asl_r_r killed %r0, killed %r2
+ ; CHECK: %r0 = A2_asrh killed %r0
+ ; CHECK: %r0 = C2_cmoveit killed %p0, 1
+ ; CHECK: J2_jumpr %r31, implicit-def %pc, implicit %r0
+ ;
+ liveins: %p0:0x1
+ %r0 = C2_cmoveit killed %p0, 1
+ J2_jumpr %r31, implicit-def %pc, implicit %r0
+...
+
+
diff --git a/test/CodeGen/Hexagon/long-calls.ll b/test/CodeGen/Hexagon/long-calls.ll
new file mode 100644
index 000000000000..9f9a527a542f
--- /dev/null
+++ b/test/CodeGen/Hexagon/long-calls.ll
@@ -0,0 +1,73 @@
+; RUN: llc -march=hexagon -enable-save-restore-long < %s | FileCheck %s
+
+; Check that the -long-calls feature is supported by the backend.
+
+; CHECK: call ##foo
+; CHECK: jump ##__restore
+define i64 @test_longcall(i32 %x, i32 %y) #0 {
+entry:
+ %add = add nsw i32 %x, 5
+ %call = tail call i64 @foo(i32 %add) #6
+ %conv = sext i32 %y to i64
+ %add1 = add nsw i64 %call, %conv
+ ret i64 %add1
+}
+
+; CHECK: jump ##foo
+define i64 @test_longtailcall(i32 %x, i32 %y) #1 {
+entry:
+ %add = add nsw i32 %x, 5
+ %call = tail call i64 @foo(i32 %add) #6
+ ret i64 %call
+}
+
+; CHECK: call ##bar
+define i64 @test_longnoret(i32 %x, i32 %y) #2 {
+entry:
+ %add = add nsw i32 %x, 5
+ %0 = tail call i64 @bar(i32 %add) #7
+ unreachable
+}
+
+; CHECK: call foo
+; CHECK: jump ##__restore
+; The restore call will still be long because of the enable-save-restore-long
+; option being used.
+define i64 @test_shortcall(i32 %x, i32 %y) #3 {
+entry:
+ %add = add nsw i32 %x, 5
+ %call = tail call i64 @foo(i32 %add) #6
+ %conv = sext i32 %y to i64
+ %add1 = add nsw i64 %call, %conv
+ ret i64 %add1
+}
+
+; CHECK: jump foo
+define i64 @test_shorttailcall(i32 %x, i32 %y) #4 {
+entry:
+ %add = add nsw i32 %x, 5
+ %call = tail call i64 @foo(i32 %add) #6
+ ret i64 %call
+}
+
+; CHECK: call bar
+define i64 @test_shortnoret(i32 %x, i32 %y) #5 {
+entry:
+ %add = add nsw i32 %x, 5
+ %0 = tail call i64 @bar(i32 %add) #7
+ unreachable
+}
+
+declare i64 @foo(i32) #6
+declare i64 @bar(i32) #7
+
+attributes #0 = { minsize nounwind "target-cpu"="hexagonv60" "target-features"="+long-calls" }
+attributes #1 = { nounwind "target-cpu"="hexagonv60" "target-features"="+long-calls" }
+attributes #2 = { noreturn nounwind "target-cpu"="hexagonv60" "target-features"="+long-calls" }
+
+attributes #3 = { minsize nounwind "target-cpu"="hexagonv60" "target-features"="-long-calls" }
+attributes #4 = { nounwind "target-cpu"="hexagonv60" "target-features"="-long-calls" }
+attributes #5 = { noreturn nounwind "target-cpu"="hexagonv60" "target-features"="-long-calls" }
+
+attributes #6 = { noreturn "target-cpu"="hexagonv60" }
+attributes #7 = { noreturn nounwind "target-cpu"="hexagonv60" }
diff --git a/test/CodeGen/Hexagon/loop-prefetch.ll b/test/CodeGen/Hexagon/loop-prefetch.ll
new file mode 100644
index 000000000000..0c6e4581a71f
--- /dev/null
+++ b/test/CodeGen/Hexagon/loop-prefetch.ll
@@ -0,0 +1,27 @@
+; RUN: llc -march=hexagon -hexagon-loop-prefetch < %s | FileCheck %s
+; CHECK: dcfetch
+
+target triple = "hexagon"
+
+define void @copy(i32* nocapture %d, i32* nocapture readonly %s, i32 %n) local_unnamed_addr #0 {
+entry:
+ %tobool2 = icmp eq i32 %n, 0
+ br i1 %tobool2, label %while.end, label %while.body
+
+while.body: ; preds = %entry, %while.body
+ %n.addr.05 = phi i32 [ %dec, %while.body ], [ %n, %entry ]
+ %s.addr.04 = phi i32* [ %incdec.ptr, %while.body ], [ %s, %entry ]
+ %d.addr.03 = phi i32* [ %incdec.ptr1, %while.body ], [ %d, %entry ]
+ %dec = add i32 %n.addr.05, -1
+ %incdec.ptr = getelementptr inbounds i32, i32* %s.addr.04, i32 1
+ %0 = load i32, i32* %s.addr.04, align 4
+ %incdec.ptr1 = getelementptr inbounds i32, i32* %d.addr.03, i32 1
+ store i32 %0, i32* %d.addr.03, align 4
+ %tobool = icmp eq i32 %dec, 0
+ br i1 %tobool, label %while.end, label %while.body
+
+while.end: ; preds = %while.body, %entry
+ ret void
+}
+
+attributes #0 = { norecurse nounwind "target-cpu"="hexagonv60" "target-features"="-hvx,-hvx-double" }
diff --git a/test/CodeGen/Hexagon/lower-extract-subvector.ll b/test/CodeGen/Hexagon/lower-extract-subvector.ll
new file mode 100644
index 000000000000..ba67de9e00a4
--- /dev/null
+++ b/test/CodeGen/Hexagon/lower-extract-subvector.ll
@@ -0,0 +1,47 @@
+; RUN: llc -march=hexagon -O3 < %s | FileCheck %s
+
+; This test checks if we custom lower extract_subvector. If we cannot
+; custom lower extract_subvector this test makes the compiler crash.
+
+; CHECK: vmem
+target triple = "hexagon-unknown--elf"
+
+; Function Attrs: nounwind
+define void @__processed() #0 {
+entry:
+ br label %"for matrix.s0.y"
+
+"for matrix.s0.y": ; preds = %"for matrix.s0.y", %entry
+ br i1 undef, label %"produce processed", label %"for matrix.s0.y"
+
+"produce processed": ; preds = %"for matrix.s0.y"
+ br i1 undef, label %"for processed.s0.ty.ty.preheader", label %"consume processed"
+
+"for processed.s0.ty.ty.preheader": ; preds = %"produce processed"
+ br i1 undef, label %"for denoised.s0.y.preheader", label %"consume denoised"
+
+"for denoised.s0.y.preheader": ; preds = %"for processed.s0.ty.ty.preheader"
+ unreachable
+
+"consume denoised": ; preds = %"for processed.s0.ty.ty.preheader"
+ br i1 undef, label %"consume deinterleaved", label %if.then.i164
+
+if.then.i164: ; preds = %"consume denoised"
+ unreachable
+
+"consume deinterleaved": ; preds = %"consume denoised"
+ %0 = tail call <64 x i32> @llvm.hexagon.V6.vshuffvdd.128B(<32 x i32> undef, <32 x i32> undef, i32 -2)
+ %1 = bitcast <64 x i32> %0 to <128 x i16>
+ %2 = shufflevector <128 x i16> %1, <128 x i16> undef, <64 x i32> <i32 64, i32 65, i32 66, i32 67, i32 68, i32 69, i32 70, i32 71, i32 72, i32 73, i32 74, i32 75, i32 76, i32 77, i32 78, i32 79, i32 80, i32 81, i32 82, i32 83, i32 84, i32 85, i32 86, i32 87, i32 88, i32 89, i32 90, i32 91, i32 92, i32 93, i32 94, i32 95, i32 96, i32 97, i32 98, i32 99, i32 100, i32 101, i32 102, i32 103, i32 104, i32 105, i32 106, i32 107, i32 108, i32 109, i32 110, i32 111, i32 112, i32 113, i32 114, i32 115, i32 116, i32 117, i32 118, i32 119, i32 120, i32 121, i32 122, i32 123, i32 124, i32 125, i32 126, i32 127>
+ store <64 x i16> %2, <64 x i16>* undef, align 128
+ unreachable
+
+"consume processed": ; preds = %"produce processed"
+ ret void
+}
+
+; Function Attrs: nounwind readnone
+declare <64 x i32> @llvm.hexagon.V6.vshuffvdd.128B(<32 x i32>, <32 x i32>, i32) #1
+
+attributes #0 = { nounwind "target-cpu"="hexagonv60" "target-features"="+hvx,+hvx-double" }
+attributes #1 = { nounwind readnone "target-cpu"="hexagonv60" "target-features"="+hvx,+hvx-double" }
diff --git a/test/CodeGen/Hexagon/misaligned_double_vector_store_not_fast.ll b/test/CodeGen/Hexagon/misaligned_double_vector_store_not_fast.ll
new file mode 100644
index 000000000000..25cb14e8514e
--- /dev/null
+++ b/test/CodeGen/Hexagon/misaligned_double_vector_store_not_fast.ll
@@ -0,0 +1,47 @@
+; RUN: llc -march=hexagon -O3 -debug-only=isel 2>&1 < %s | FileCheck %s
+; REQUIRES: asserts
+
+; DAGCombiner converts the two vector stores to a double vector store,
+; even if the double vector store is unaligned. This is not good. If it
+; is unaligned, we should let the DAGCombiner know that it is slow via
+; the allowsMisalignedAccess function in HexagonISelLowering.
+
+; CHECK-NOT: store<ST256{{.*}}(align=128)>
+
+target triple = "hexagon-unknown--elf"
+
+; Function Attrs: nounwind
+define void @__processed() #0 {
+entry:
+ br label %"for demosaiced.s0.y.y"
+
+"for demosaiced.s0.y.y": ; preds = %"for demosaiced.s0.y.y", %entry
+ %demosaiced.s0.y.y = phi i32 [ 0, %entry ], [ %0, %"for demosaiced.s0.y.y" ]
+ %0 = add nuw nsw i32 %demosaiced.s0.y.y, 1
+ %1 = mul nuw nsw i32 %demosaiced.s0.y.y, 256
+ %2 = tail call <64 x i32> @llvm.hexagon.V6.vshuffvdd.128B(<32 x i32> undef, <32 x i32> undef, i32 -2)
+ %3 = bitcast <64 x i32> %2 to <128 x i16>
+ %4 = shufflevector <128 x i16> %3, <128 x i16> undef, <64 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15, i32 16, i32 17, i32 18, i32 19, i32 20, i32 21, i32 22, i32 23, i32 24, i32 25, i32 26, i32 27, i32 28, i32 29, i32 30, i32 31, i32 32, i32 33, i32 34, i32 35, i32 36, i32 37, i32 38, i32 39, i32 40, i32 41, i32 42, i32 43, i32 44, i32 45, i32 46, i32 47, i32 48, i32 49, i32 50, i32 51, i32 52, i32 53, i32 54, i32 55, i32 56, i32 57, i32 58, i32 59, i32 60, i32 61, i32 62, i32 63>
+ %5 = add nuw nsw i32 %1, 32896
+ %6 = getelementptr inbounds i16, i16* undef, i32 %5
+ %7 = bitcast i16* %6 to <64 x i16>*
+ store <64 x i16> %4, <64 x i16>* %7, align 128
+ %8 = shufflevector <128 x i16> %3, <128 x i16> undef, <64 x i32> <i32 64, i32 65, i32 66, i32 67, i32 68, i32 69, i32 70, i32 71, i32 72, i32 73, i32 74, i32 75, i32 76, i32 77, i32 78, i32 79, i32 80, i32 81, i32 82, i32 83, i32 84, i32 85, i32 86, i32 87, i32 88, i32 89, i32 90, i32 91, i32 92, i32 93, i32 94, i32 95, i32 96, i32 97, i32 98, i32 99, i32 100, i32 101, i32 102, i32 103, i32 104, i32 105, i32 106, i32 107, i32 108, i32 109, i32 110, i32 111, i32 112, i32 113, i32 114, i32 115, i32 116, i32 117, i32 118, i32 119, i32 120, i32 121, i32 122, i32 123, i32 124, i32 125, i32 126, i32 127>
+ %9 = add nuw nsw i32 %1, 32960
+ %10 = getelementptr inbounds i16, i16* undef, i32 %9
+ %11 = bitcast i16* %10 to <64 x i16>*
+ store <64 x i16> %8, <64 x i16>* %11, align 128
+ br i1 false, label %"consume demosaiced", label %"for demosaiced.s0.y.y"
+
+"consume demosaiced": ; preds = %"for demosaiced.s0.y.y"
+ unreachable
+
+"consume processed": ; preds = %"produce processed"
+ ret void
+}
+
+declare <64 x i32> @llvm.hexagon.V6.vshuffvdd.128B(<32 x i32>, <32 x i32>, i32) #1
+
+attributes #0 = { nounwind "target-cpu"="hexagonv60" "target-features"="+hvx,+hvx-double" }
+attributes #1 = { nounwind readnone "target-cpu"="hexagonv60" "target-features"="+hvx,+hvx-double" }
+
diff --git a/test/CodeGen/Hexagon/mulhs.ll b/test/CodeGen/Hexagon/mulhs.ll
new file mode 100644
index 000000000000..b8727da8a7e2
--- /dev/null
+++ b/test/CodeGen/Hexagon/mulhs.ll
@@ -0,0 +1,23 @@
+; RUN: llc -march=hexagon < %s | FileCheck %s
+
+; CHECK: mpy
+; CHECK-NOT: call
+
+target triple = "hexagon"
+
+; Function Attrs: nounwind
+define i32 @fred(i64 %x, i64 %y, i64* nocapture %z) #0 {
+entry:
+ %0 = tail call { i64, i1 } @llvm.smul.with.overflow.i64(i64 %x, i64 %y)
+ %1 = extractvalue { i64, i1 } %0, 1
+ %2 = extractvalue { i64, i1 } %0, 0
+ store i64 %2, i64* %z, align 8
+ %conv = zext i1 %1 to i32
+ ret i32 %conv
+}
+
+; Function Attrs: nounwind readnone
+declare { i64, i1 } @llvm.smul.with.overflow.i64(i64, i64) #1
+
+attributes #0 = { nounwind }
+attributes #1 = { nounwind readnone }
diff --git a/test/CodeGen/Hexagon/newvalueSameReg.ll b/test/CodeGen/Hexagon/newvalueSameReg.ll
new file mode 100644
index 000000000000..0fc4df22eb32
--- /dev/null
+++ b/test/CodeGen/Hexagon/newvalueSameReg.ll
@@ -0,0 +1,63 @@
+; RUN: llc -march=hexagon -hexagon-expand-condsets=0 < %s | FileCheck %s
+;
+; Expand-condsets eliminates the "mux" instruction, which is what this
+; testcase is checking.
+
+%struct._Dnk_filet.1 = type { i16, i8, i32, i8*, i8*, i8*, i8*, i8*, i8*, i32*, [2 x i32], i8*, i8*, i8*, %struct._Mbstatet.0, i8*, [8 x i8], i8 }
+%struct._Mbstatet.0 = type { i32, i16, i16 }
+
+@_Stdout = external global %struct._Dnk_filet.1
+@.str = external unnamed_addr constant [23 x i8], align 8
+
+; Test that we don't generate a new value compare if the operands are
+; the same register.
+
+; CHECK-NOT: cmp.eq([[REG0:(r[0-9]+)]].new, [[REG0]])
+; CHECK: cmp.eq([[REG1:(r[0-9]+)]], [[REG1]])
+
+; Function Attrs: nounwind
+declare void @fprintf(%struct._Dnk_filet.1* nocapture, i8* nocapture readonly, ...) #1
+
+define void @main() #0 {
+entry:
+ %0 = load i32*, i32** undef, align 4
+ %1 = load i32, i32* undef, align 4
+ br i1 undef, label %if.end, label %_ZNSt6vectorIbSaIbEE3endEv.exit
+
+_ZNSt6vectorIbSaIbEE3endEv.exit:
+ %2 = icmp slt i32 %1, 0
+ %sub5.i.i.i = lshr i32 %1, 5
+ %add619.i.i.i = add i32 %sub5.i.i.i, -134217728
+ %sub5.i.pn.i.i = select i1 %2, i32 %add619.i.i.i, i32 %sub5.i.i.i
+ %storemerge2.i.i = getelementptr inbounds i32, i32* %0, i32 %sub5.i.pn.i.i
+ %cmp.i.i = icmp ult i32* %storemerge2.i.i, %0
+ %.mux = select i1 %cmp.i.i, i32 0, i32 1
+ br i1 undef, label %_ZNSt6vectorIbSaIbEE3endEv.exit57, label %if.end
+
+_ZNSt6vectorIbSaIbEE3endEv.exit57:
+ %3 = icmp slt i32 %1, 0
+ %sub5.i.i.i44 = lshr i32 %1, 5
+ %add619.i.i.i45 = add i32 %sub5.i.i.i44, -134217728
+ %sub5.i.pn.i.i46 = select i1 %3, i32 %add619.i.i.i45, i32 %sub5.i.i.i44
+ %storemerge2.i.i47 = getelementptr inbounds i32, i32* %0, i32 %sub5.i.pn.i.i46
+ %cmp.i38 = icmp ult i32* %storemerge2.i.i47, %0
+ %.reg2mem.sroa.0.sroa.0.0.load14.i.reload = select i1 %cmp.i38, i32 0, i32 1
+ %cmp = icmp eq i32 %.mux, %.reg2mem.sroa.0.sroa.0.0.load14.i.reload
+ br i1 %cmp, label %if.end, label %if.then
+
+if.then:
+ call void (%struct._Dnk_filet.1*, i8*, ...) @fprintf(%struct._Dnk_filet.1* @_Stdout, i8* getelementptr inbounds ([23 x i8], [23 x i8]* @.str, i32 0, i32 0), i32 %.mux, i32 %.reg2mem.sroa.0.sroa.0.0.load14.i.reload) #1
+ unreachable
+
+if.end:
+ br i1 undef, label %_ZNSt6vectorIbSaIbEED2Ev.exit, label %if.then.i.i.i
+
+if.then.i.i.i:
+ unreachable
+
+_ZNSt6vectorIbSaIbEED2Ev.exit:
+ ret void
+}
+
+attributes #0 = { "target-cpu"="hexagonv5" }
+attributes #1 = { nounwind "target-cpu"="hexagonv5" }
diff --git a/test/CodeGen/Hexagon/opt-spill-volatile.ll b/test/CodeGen/Hexagon/opt-spill-volatile.ll
new file mode 100644
index 000000000000..99dd4646d743
--- /dev/null
+++ b/test/CodeGen/Hexagon/opt-spill-volatile.ll
@@ -0,0 +1,29 @@
+; RUN: llc -march=hexagon < %s | FileCheck %s
+; Check that the load/store to the volatile stack object has not been
+; optimized away.
+
+target triple = "hexagon"
+
+; CHECK-LABEL: foo
+; CHECK: memw(r29+#4) =
+; CHECK: = memw(r29 + #4)
+define i32 @foo(i32 %a) #0 {
+entry:
+ %x = alloca i32, align 4
+ %x.0.x.0..sroa_cast = bitcast i32* %x to i8*
+ call void @llvm.lifetime.start(i64 4, i8* %x.0.x.0..sroa_cast)
+ store volatile i32 0, i32* %x, align 4
+ %call = tail call i32 bitcast (i32 (...)* @bar to i32 ()*)() #0
+ %x.0.x.0. = load volatile i32, i32* %x, align 4
+ %add = add nsw i32 %x.0.x.0., %a
+ call void @llvm.lifetime.end(i64 4, i8* %x.0.x.0..sroa_cast)
+ ret i32 %add
+}
+
+declare void @llvm.lifetime.start(i64, i8* nocapture) #1
+declare void @llvm.lifetime.end(i64, i8* nocapture) #1
+
+declare i32 @bar(...) #0
+
+attributes #0 = { nounwind }
+attributes #1 = { argmemonly nounwind }
diff --git a/test/CodeGen/Hexagon/packetize-cfi-location.ll b/test/CodeGen/Hexagon/packetize-cfi-location.ll
new file mode 100644
index 000000000000..0d80a7bb289d
--- /dev/null
+++ b/test/CodeGen/Hexagon/packetize-cfi-location.ll
@@ -0,0 +1,72 @@
+; RUN: llc -march=hexagon < %s | FileCheck %s
+
+target triple = "hexagon"
+%type.0 = type { i32, i8**, i32, i32, i32 }
+
+; Check that CFI is before the packet with call+allocframe.
+; CHECK-LABEL: danny:
+; CHECK: cfi_def_cfa
+; CHECK: call throw
+; CHECK-NEXT: allocframe
+
+; Expect packet:
+; {
+; call throw
+; allocframe(#0)
+; }
+
+define i8* @danny(%type.0* %p0, i32 %p1) #0 {
+entry:
+ %t0 = getelementptr inbounds %type.0, %type.0* %p0, i32 0, i32 4
+ %t1 = load i32, i32* %t0, align 4
+ %th = icmp ugt i32 %t1, %p1
+ br i1 %th, label %if.end, label %if.then
+
+if.then: ; preds = %entry
+ tail call void @throw(%type.0* nonnull %p0)
+ unreachable
+
+if.end: ; preds = %entry
+ %t6 = getelementptr inbounds %type.0, %type.0* %p0, i32 0, i32 3
+ %t2 = load i32, i32* %t6, align 4
+ %t9 = add i32 %t2, %p1
+ %ta = lshr i32 %t9, 4
+ %tb = and i32 %t9, 15
+ %t7 = getelementptr inbounds %type.0, %type.0* %p0, i32 0, i32 2
+ %t3 = load i32, i32* %t7, align 4
+ %tc = icmp ult i32 %ta, %t3
+ %td = select i1 %tc, i32 0, i32 %t3
+ %te = sub i32 %ta, %td
+ %t8 = getelementptr inbounds %type.0, %type.0* %p0, i32 0, i32 1
+ %t4 = load i8**, i8*** %t8, align 4
+ %tf = getelementptr inbounds i8*, i8** %t4, i32 %te
+ %t5 = load i8*, i8** %tf, align 4
+ %tg = getelementptr inbounds i8, i8* %t5, i32 %tb
+ ret i8* %tg
+}
+
+; Check that CFI is after allocframe.
+; CHECK-LABEL: sammy:
+; CHECK: allocframe
+; CHECK: cfi_def_cfa
+
+define void @sammy(%type.0* %p0, i32 %p1) #0 {
+entry:
+ %t0 = icmp sgt i32 %p1, 0
+ br i1 %t0, label %if.then, label %if.else
+if.then:
+ call void @throw(%type.0* nonnull %p0)
+ br label %if.end
+if.else:
+ call void @nothrow() #2
+ br label %if.end
+if.end:
+ ret void
+}
+
+declare void @throw(%type.0*) #1
+declare void @nothrow() #2
+
+attributes #0 = { "target-cpu"="hexagonv55" }
+attributes #1 = { noreturn "target-cpu"="hexagonv55" }
+attributes #2 = { nounwind "target-cpu"="hexagonv55" }
diff --git a/test/CodeGen/Hexagon/packetize-return-arg.ll b/test/CodeGen/Hexagon/packetize-return-arg.ll
new file mode 100644
index 000000000000..b18fc23eca81
--- /dev/null
+++ b/test/CodeGen/Hexagon/packetize-return-arg.ll
@@ -0,0 +1,37 @@
+; RUN: llc -march=hexagon < %s | FileCheck %s
+; Check that "r0 = rN" is packetized together with dealloc_return.
+; CHECK: r0 = r
+; CHECK-NOT: {
+; CHECK: dealloc_return
+
+target triple = "hexagon-unknown--elf"
+
+; Function Attrs: nounwind
+define i8* @fred(i8* %user_context, i32 %x) #0 {
+entry:
+ %and14 = add i32 %x, 255
+ %add1 = and i32 %and14, -128
+ %call = tail call i8* @malloc(i32 %add1) #1
+ %cmp = icmp eq i8* %call, null
+ br i1 %cmp, label %cleanup, label %if.end
+
+if.end: ; preds = %entry
+ %0 = ptrtoint i8* %call to i32
+ %sub4 = add i32 %0, 131
+ %and5 = and i32 %sub4, -128
+ %1 = inttoptr i32 %and5 to i8*
+ %2 = inttoptr i32 %and5 to i8**
+ %arrayidx = getelementptr inbounds i8*, i8** %2, i32 -1
+ store i8* %call, i8** %arrayidx, align 4
+ br label %cleanup
+
+cleanup: ; preds = %if.end, %entry
+ %retval.0 = phi i8* [ %1, %if.end ], [ null, %entry ]
+ ret i8* %retval.0
+}
+
+; Function Attrs: nounwind
+declare noalias i8* @malloc(i32) local_unnamed_addr #1
+
+attributes #0 = { nounwind }
+attributes #1 = { nobuiltin nounwind }
diff --git a/test/CodeGen/Hexagon/peephole-kill-flags.ll b/test/CodeGen/Hexagon/peephole-kill-flags.ll
new file mode 100644
index 000000000000..03de15323528
--- /dev/null
+++ b/test/CodeGen/Hexagon/peephole-kill-flags.ll
@@ -0,0 +1,27 @@
+; RUN: llc -march=hexagon -verify-machineinstrs < %s | FileCheck %s
+; CHECK: memw
+
+; Check that the testcase compiles without errors.
+
+target triple = "hexagon"
+
+; Function Attrs: nounwind
+define void @fred() #0 {
+entry:
+ br label %for.cond
+
+for.cond: ; preds = %entry
+ %0 = load i32, i32* undef, align 4
+ %mul = mul nsw i32 2, %0
+ %cmp = icmp slt i32 undef, %mul
+ br i1 %cmp, label %for.body, label %for.end13
+
+for.body: ; preds = %for.cond
+ unreachable
+
+for.end13: ; preds = %for.cond
+ ret void
+}
+
+attributes #0 = { nounwind "target-cpu"="hexagonv60" "target-features"="+hvx,-hvx-double" }
+
diff --git a/test/CodeGen/Hexagon/pic-simple.ll b/test/CodeGen/Hexagon/pic-simple.ll
index fa223d5372e1..46d95204f2e7 100644
--- a/test/CodeGen/Hexagon/pic-simple.ll
+++ b/test/CodeGen/Hexagon/pic-simple.ll
@@ -1,4 +1,4 @@
-; RUN: llc -march=hexagon -mcpu=hexagonv5 -relocation-model=pic < %s | FileCheck %s
+; RUN: llc -mtriple=hexagon-- -mcpu=hexagonv5 -relocation-model=pic < %s | FileCheck %s
; CHECK: r{{[0-9]+}} = add({{pc|PC}}, ##_GLOBAL_OFFSET_TABLE_@PCREL)
; CHECK: r{{[0-9]+}} = memw(r{{[0-9]+}}{{.*}}+{{.*}}##src@GOT)
diff --git a/test/CodeGen/Hexagon/pic-static.ll b/test/CodeGen/Hexagon/pic-static.ll
index f4ccc6b9ee73..66d7734f2cf2 100644
--- a/test/CodeGen/Hexagon/pic-static.ll
+++ b/test/CodeGen/Hexagon/pic-static.ll
@@ -1,4 +1,4 @@
-; RUN: llc -march=hexagon -mcpu=hexagonv5 -relocation-model=pic < %s | FileCheck %s
+; RUN: llc -mtriple=hexagon-- -mcpu=hexagonv5 -relocation-model=pic < %s | FileCheck %s
; CHECK-DAG: r{{[0-9]+}} = add({{pc|PC}}, ##_GLOBAL_OFFSET_TABLE_@PCREL)
; CHECK-DAG: r{{[0-9]+}} = add({{pc|PC}}, ##x@PCREL)
diff --git a/test/CodeGen/Hexagon/post-inc-aa-metadata.ll b/test/CodeGen/Hexagon/post-inc-aa-metadata.ll
new file mode 100644
index 000000000000..fb2f038e6e59
--- /dev/null
+++ b/test/CodeGen/Hexagon/post-inc-aa-metadata.ll
@@ -0,0 +1,37 @@
+; RUN: llc -march=hexagon -debug-only=isel < %s 2>&1 | FileCheck %s
+; REQUIRES: asserts
+
+; Check that the generated post-increment load has TBAA information.
+; CHECK-LABEL: Machine code for function fred:
+; CHECK: = V6_vL32b_pi %vreg{{[0-9]+}}<tied1>, 64; mem:LD64[{{.*}}](tbaa=
+
+target triple = "hexagon"
+
+; Function Attrs: norecurse nounwind
+define void @fred(<16 x i32>* nocapture %p, <16 x i32>* nocapture readonly %q, i32 %n) local_unnamed_addr #0 {
+entry:
+ %tobool2 = icmp eq i32 %n, 0
+ br i1 %tobool2, label %while.end, label %while.body
+
+while.body: ; preds = %entry, %while.body
+ %n.addr.05 = phi i32 [ %dec, %while.body ], [ %n, %entry ]
+ %q.addr.04 = phi <16 x i32>* [ %incdec.ptr, %while.body ], [ %q, %entry ]
+ %p.addr.03 = phi <16 x i32>* [ %incdec.ptr1, %while.body ], [ %p, %entry ]
+ %dec = add i32 %n.addr.05, -1
+ %incdec.ptr = getelementptr inbounds <16 x i32>, <16 x i32>* %q.addr.04, i32 1
+ %0 = load <16 x i32>, <16 x i32>* %q.addr.04, align 64, !tbaa !1
+ %incdec.ptr1 = getelementptr inbounds <16 x i32>, <16 x i32>* %p.addr.03, i32 1
+ store <16 x i32> %0, <16 x i32>* %p.addr.03, align 64, !tbaa !1
+ %tobool = icmp eq i32 %dec, 0
+ br i1 %tobool, label %while.end, label %while.body
+
+while.end: ; preds = %while.body, %entry
+ ret void
+}
+
+attributes #0 = { norecurse nounwind "target-cpu"="hexagonv60" "target-features"="+hvx,-hvx-double" }
+
+
+!1 = !{!2, !2, i64 0}
+!2 = !{!"omnipotent char", !3, i64 0}
+!3 = !{!"Simple C/C++ TBAA"}
diff --git a/test/CodeGen/Hexagon/post-ra-kill-update.mir b/test/CodeGen/Hexagon/post-ra-kill-update.mir
new file mode 100644
index 000000000000..c43624d7a8d3
--- /dev/null
+++ b/test/CodeGen/Hexagon/post-ra-kill-update.mir
@@ -0,0 +1,37 @@
+# RUN: llc -march=hexagon -mcpu=hexagonv60 -run-pass post-RA-sched -o - %s | FileCheck %s
+
+# The post-RA scheduler reorders S2_lsr_r_p and S2_lsr_r_p_or. Both of them
+# use r9, and the last of the two kills it. The kill flag fixup did not
+# correctly update the flag, resulting in both instructions killing r9.
+
+# CHECK-LABEL: name: foo
+# Check for no-kill of r9 in the first instruction, after reordering:
+# CHECK: %d7 = S2_lsr_r_p_or %d7, killed %d1, %r9
+# CHECK: %d13 = S2_lsr_r_p killed %d0, killed %r9
+
+--- |
+ define void @foo() {
+ ret void
+ }
+...
+
+---
+name: foo
+tracksRegLiveness: true
+body: |
+ bb.0:
+ successors: %bb.1
+ liveins: %d0, %d1, %r9, %r13
+
+ %d7 = S2_asl_r_p %d0, %r13
+ %d5 = S2_asl_r_p %d1, killed %r13
+ %d6 = S2_lsr_r_p killed %d0, %r9
+ %d7 = S2_lsr_r_p_or killed %d7, killed %d1, killed %r9
+ %d1 = A2_combinew killed %r11, killed %r10
+ %d0 = A2_combinew killed %r15, killed %r14
+ J2_jump %bb.1, implicit-def %pc
+
+ bb.1:
+ A2_nop
+...
+
diff --git a/test/CodeGen/Hexagon/propagate-vcombine.ll b/test/CodeGen/Hexagon/propagate-vcombine.ll
new file mode 100644
index 000000000000..4948a89b73e8
--- /dev/null
+++ b/test/CodeGen/Hexagon/propagate-vcombine.ll
@@ -0,0 +1,48 @@
+; RUN: llc -march=hexagon < %s | FileCheck %s
+
+@v0 = global <16 x i32> zeroinitializer, align 64
+@v1 = global <16 x i32> zeroinitializer, align 64
+
+; CHECK-LABEL: danny:
+; CHECK-NOT: vcombine
+
+define void @danny() #0 {
+ %t0 = load <16 x i32>, <16 x i32>* @v0, align 64
+ %t1 = load <16 x i32>, <16 x i32>* @v1, align 64
+ %t2 = call <32 x i32> @llvm.hexagon.V6.vcombine(<16 x i32> %t0, <16 x i32> %t1)
+ %t3 = tail call <16 x i32> @llvm.hexagon.V6.lo(<32 x i32> %t2)
+ %t4 = tail call <16 x i32> @llvm.hexagon.V6.hi(<32 x i32> %t2)
+ store <16 x i32> %t3, <16 x i32>* @v0, align 64
+ store <16 x i32> %t4, <16 x i32>* @v1, align 64
+ ret void
+}
+
+@w0 = global <32 x i32> zeroinitializer, align 128
+@w1 = global <32 x i32> zeroinitializer, align 128
+
+; CHECK-LABEL: sammy:
+; CHECK-NOT: vcombine
+
+define void @sammy() #1 {
+ %t0 = load <32 x i32>, <32 x i32>* @w0, align 128
+ %t1 = load <32 x i32>, <32 x i32>* @w1, align 128
+ %t2 = call <64 x i32> @llvm.hexagon.V6.vcombine.128B(<32 x i32> %t0, <32 x i32> %t1)
+ %t3 = tail call <32 x i32> @llvm.hexagon.V6.lo.128B(<64 x i32> %t2)
+ %t4 = tail call <32 x i32> @llvm.hexagon.V6.hi.128B(<64 x i32> %t2)
+ store <32 x i32> %t3, <32 x i32>* @w0, align 128
+ store <32 x i32> %t4, <32 x i32>* @w1, align 128
+ ret void
+}
+
+declare <32 x i32> @llvm.hexagon.V6.vcombine(<16 x i32>, <16 x i32>) #2
+declare <16 x i32> @llvm.hexagon.V6.lo(<32 x i32>) #2
+declare <16 x i32> @llvm.hexagon.V6.hi(<32 x i32>) #2
+
+declare <64 x i32> @llvm.hexagon.V6.vcombine.128B(<32 x i32>, <32 x i32>) #3
+declare <32 x i32> @llvm.hexagon.V6.lo.128B(<64 x i32>) #3
+declare <32 x i32> @llvm.hexagon.V6.hi.128B(<64 x i32>) #3
+
+attributes #0 = { nounwind "target-cpu"="hexagonv60" "target-features"="+hvx" }
+attributes #1 = { nounwind "target-cpu"="hexagonv60" "target-features"="+hvx,+hvx-double" }
+attributes #2 = { nounwind readnone "target-cpu"="hexagonv60" "target-features"="+hvx" }
+attributes #3 = { nounwind readnone "target-cpu"="hexagonv60" "target-features"="+hvx,+hvx-double" }
diff --git a/test/CodeGen/Hexagon/rdf-copy.ll b/test/CodeGen/Hexagon/rdf-copy.ll
index afb03a6315d7..ce47cf672d79 100644
--- a/test/CodeGen/Hexagon/rdf-copy.ll
+++ b/test/CodeGen/Hexagon/rdf-copy.ll
@@ -17,7 +17,7 @@
; CHECK: [[DST:r[0-9]+]] = [[SRC:r[0-9]+]]
; CHECK-DAG: memw([[SRC]]
; CHECK-NOT: memw([[DST]]
-; CHECK-LABEL: LBB0_2
+; CHECK: %if.end
target datalayout = "e-p:32:32:32-i64:64:64-i32:32:32-i16:16:16-i1:32:32-f64:64:64-f32:32:32-v64:64:64-v32:32:32-a0:0-n16:32"
target triple = "hexagon"
diff --git a/test/CodeGen/Hexagon/rdf-extra-livein.ll b/test/CodeGen/Hexagon/rdf-extra-livein.ll
new file mode 100644
index 000000000000..5d947e6fe45d
--- /dev/null
+++ b/test/CodeGen/Hexagon/rdf-extra-livein.ll
@@ -0,0 +1,73 @@
+; RUN: llc -march=hexagon -verify-machineinstrs < %s | FileCheck %s
+; Verify that the code compiles successfully.
+; CHECK: call printf
+
+target triple = "hexagon"
+
+%struct.0 = type { i32, i32, i32, i32, i32, i32, i32, i32, i32 }
+
+@.str.13 = external unnamed_addr constant [60 x i8], align 1
+
+declare void @printf(i8* nocapture readonly, ...) local_unnamed_addr #0
+
+declare void @danny() local_unnamed_addr #0
+declare zeroext i8 @sammy() local_unnamed_addr #0
+
+; Function Attrs: nounwind
+define void @main() local_unnamed_addr #0 {
+entry:
+ br i1 undef, label %if.then8, label %if.end10
+
+if.then8: ; preds = %entry
+ ret void
+
+if.end10: ; preds = %entry
+ br label %do.body
+
+do.body: ; preds = %if.end88.do.body_crit_edge, %if.end10
+ %cond = icmp eq i32 undef, 0
+ br i1 %cond, label %if.end49, label %if.then124
+
+if.end49: ; preds = %do.body
+ br i1 undef, label %if.end55, label %if.then53
+
+if.then53: ; preds = %if.end49
+ call void @danny()
+ br label %if.end55
+
+if.end55: ; preds = %if.then53, %if.end49
+ %call76 = call zeroext i8 @sammy() #0
+ switch i8 %call76, label %sw.epilog79 [
+ i8 0, label %sw.bb77
+ i8 3, label %sw.bb77
+ ]
+
+sw.bb77: ; preds = %if.end55, %if.end55
+ unreachable
+
+sw.epilog79: ; preds = %if.end55
+ br i1 undef, label %if.end88, label %if.then81
+
+if.then81: ; preds = %sw.epilog79
+ %div87 = fdiv float 0.000000e+00, undef
+ br label %if.end88
+
+if.end88: ; preds = %if.then81, %sw.epilog79
+ %t.1 = phi float [ undef, %sw.epilog79 ], [ %div87, %if.then81 ]
+ %div89 = fdiv float 1.000000e+00, %t.1
+ %mul92 = fmul float undef, %div89
+ %div93 = fdiv float %mul92, 1.000000e+06
+ %conv107 = fpext float %div93 to double
+ call void (i8*, ...) @printf(i8* getelementptr inbounds ([60 x i8], [60 x i8]* @.str.13, i32 0, i32 0), double %conv107, double undef, i64 undef, i32 undef) #0
+ br i1 undef, label %if.end88.do.body_crit_edge, label %if.then124
+
+if.end88.do.body_crit_edge: ; preds = %if.end88
+ br label %do.body
+
+if.then124: ; preds = %if.end88, %do.body
+ unreachable
+}
+
+
+attributes #0 = { nounwind }
+
diff --git a/test/CodeGen/Hexagon/rdf-filter-defs.ll b/test/CodeGen/Hexagon/rdf-filter-defs.ll
new file mode 100644
index 000000000000..735b20e697fd
--- /dev/null
+++ b/test/CodeGen/Hexagon/rdf-filter-defs.ll
@@ -0,0 +1,214 @@
+; RUN: llc -march=hexagon -verify-machineinstrs < %s | FileCheck %s
+
+; Check that this testcase compiles successfully.
+; CHECK: dealloc_return
+
+target triple = "hexagon"
+
+%type.0 = type { %type.1, %type.3, i32, i32 }
+%type.1 = type { %type.2 }
+%type.2 = type { i8 }
+%type.3 = type { i8*, [12 x i8] }
+%type.4 = type { i8 }
+
+define weak_odr dereferenceable(28) %type.0* @fred(%type.0* %p0, i32 %p1, %type.0* dereferenceable(28) %p2, i32 %p3, i32 %p4) local_unnamed_addr align 2 {
+b0:
+ %t0 = getelementptr inbounds %type.0, %type.0* %p0, i32 0, i32 2
+ %t1 = load i32, i32* %t0, align 4
+ %t2 = icmp ult i32 %t1, %p1
+ %t3 = getelementptr inbounds %type.0, %type.0* %p2, i32 0, i32 2
+ br i1 %t2, label %b2, label %b1
+
+b1:
+ %t4 = load i32, i32* %t3, align 4
+ %t5 = icmp ult i32 %t4, %p3
+ br i1 %t5, label %b2, label %b3
+
+b2:
+ %t6 = bitcast %type.0* %p0 to %type.4*
+ tail call void @blah(%type.4* %t6)
+ %t7 = load i32, i32* %t3, align 4
+ %t8 = load i32, i32* %t0, align 4
+ br label %b3
+
+b3:
+ %t9 = phi i32 [ %t8, %b2 ], [ %t1, %b1 ]
+ %t10 = phi i32 [ %t7, %b2 ], [ %t4, %b1 ]
+ %t11 = sub i32 %t10, %p3
+ %t12 = icmp ult i32 %t11, %p4
+ %t13 = select i1 %t12, i32 %t11, i32 %p4
+ %t14 = xor i32 %t9, -1
+ %t15 = icmp ult i32 %t13, %t14
+ br i1 %t15, label %b5, label %b4
+
+b4:
+ %t16 = bitcast %type.0* %p0 to %type.4*
+ tail call void @danny(%type.4* %t16)
+ br label %b5
+
+b5:
+ %t17 = icmp eq i32 %t13, 0
+ br i1 %t17, label %b33, label %b6
+
+b6:
+ %t18 = load i32, i32* %t0, align 4
+ %t19 = add i32 %t18, %t13
+ %t20 = icmp eq i32 %t19, -1
+ br i1 %t20, label %b7, label %b8
+
+b7:
+ %t21 = bitcast %type.0* %p0 to %type.4*
+ tail call void @danny(%type.4* %t21)
+ br label %b8
+
+b8:
+ %t22 = getelementptr inbounds %type.0, %type.0* %p0, i32 0, i32 3
+ %t23 = load i32, i32* %t22, align 4
+ %t24 = icmp ult i32 %t23, %t19
+ br i1 %t24, label %b9, label %b10
+
+b9:
+ %t25 = load i32, i32* %t0, align 4
+ tail call void @sammy(%type.0* nonnull %p0, i32 %t19, i32 %t25)
+ %t26 = load i32, i32* %t22, align 4
+ br label %b15
+
+b10:
+ %t27 = icmp eq i32 %t19, 0
+ br i1 %t27, label %b11, label %b15
+
+b11:
+ %t28 = icmp ugt i32 %t23, 15
+ %t29 = getelementptr inbounds %type.0, %type.0* %p0, i32 0, i32 1
+ br i1 %t28, label %b12, label %b13
+
+b12:
+ %t30 = getelementptr inbounds %type.3, %type.3* %t29, i32 0, i32 0
+ %t31 = load i8*, i8** %t30, align 4
+ br label %b14
+
+b13:
+ %t32 = bitcast %type.3* %t29 to i8*
+ br label %b14
+
+b14:
+ %t33 = phi i8* [ %t31, %b12 ], [ %t32, %b13 ]
+ store i32 0, i32* %t0, align 4
+ br label %b31
+
+b15:
+ %t34 = phi i32 [ %t26, %b9 ], [ %t23, %b10 ]
+ %t35 = icmp ugt i32 %t34, 15
+ %t36 = getelementptr inbounds %type.0, %type.0* %p0, i32 0, i32 1
+ br i1 %t35, label %b16, label %b17
+
+b16:
+ %t37 = getelementptr inbounds %type.3, %type.3* %t36, i32 0, i32 0
+ %t38 = load i8*, i8** %t37, align 4
+ br label %b18
+
+b17:
+ %t39 = bitcast %type.3* %t36 to i8*
+ %t40 = bitcast %type.3* %t36 to i8*
+ br label %b18
+
+b18:
+ %t41 = phi i8* [ %t38, %b16 ], [ %t39, %b17 ]
+ %t42 = phi i8* [ %t38, %b16 ], [ %t40, %b17 ]
+ %t43 = getelementptr inbounds i8, i8* %t41, i32 %p1
+ %t44 = getelementptr inbounds i8, i8* %t43, i32 %t13
+ %t45 = getelementptr inbounds i8, i8* %t42, i32 %p1
+ %t46 = load i32, i32* %t0, align 4
+ %t47 = sub i32 %t46, %p1
+ tail call void @llvm.memmove.p0i8.p0i8.i32(i8* %t44, i8* %t45, i32 %t47, i32 1, i1 false) #1
+ %t48 = icmp eq %type.0* %p0, %p2
+ %t49 = load i32, i32* %t22, align 4
+ %t50 = icmp ugt i32 %t49, 15
+ br i1 %t50, label %b19, label %b20
+
+b19:
+ %t51 = getelementptr inbounds %type.3, %type.3* %t36, i32 0, i32 0
+ %t52 = load i8*, i8** %t51, align 4
+ br label %b21
+
+b20:
+ %t53 = bitcast %type.3* %t36 to i8*
+ br label %b21
+
+b21:
+ %t54 = phi i8* [ %t52, %b19 ], [ %t53, %b20 ]
+ %t55 = getelementptr inbounds i8, i8* %t54, i32 %p1
+ br i1 %t48, label %b22, label %b26
+
+b22:
+ br i1 %t50, label %b23, label %b24
+
+b23:
+ %t56 = getelementptr inbounds %type.3, %type.3* %t36, i32 0, i32 0
+ %t57 = load i8*, i8** %t56, align 4
+ br label %b25
+
+b24:
+ %t58 = bitcast %type.3* %t36 to i8*
+ br label %b25
+
+b25:
+ %t59 = phi i8* [ %t57, %b23 ], [ %t58, %b24 ]
+ %t60 = icmp ult i32 %p1, %p3
+ %t61 = select i1 %t60, i32 %t13, i32 0
+ %t62 = add i32 %t61, %p3
+ %t63 = getelementptr inbounds i8, i8* %t59, i32 %t62
+ tail call void @llvm.memmove.p0i8.p0i8.i32(i8* %t55, i8* %t63, i32 %t13, i32 1, i1 false) #1
+ br label %b27
+
+b26:
+ %t64 = getelementptr inbounds %type.0, %type.0* %p2, i32 0, i32 3
+ %t65 = load i32, i32* %t64, align 4
+ %t66 = icmp ugt i32 %t65, 15
+ %t67 = getelementptr inbounds %type.0, %type.0* %p2, i32 0, i32 1
+ %t68 = getelementptr inbounds %type.3, %type.3* %t67, i32 0, i32 0
+ %t69 = load i8*, i8** %t68, align 4
+ %t70 = bitcast %type.3* %t67 to i8*
+ %t71 = select i1 %t66, i8* %t69, i8* %t70
+ %t72 = getelementptr inbounds i8, i8* %t71, i32 %p3
+ tail call void @llvm.memcpy.p0i8.p0i8.i32(i8* %t55, i8* %t72, i32 %t13, i32 1, i1 false) #1
+ br label %b27
+
+b27:
+ %t73 = load i32, i32* %t22, align 4
+ %t74 = icmp ugt i32 %t73, 15
+ br i1 %t74, label %b28, label %b29
+
+b28:
+ %t75 = getelementptr inbounds %type.3, %type.3* %t36, i32 0, i32 0
+ %t76 = load i8*, i8** %t75, align 4
+ br label %b30
+
+b29:
+ %t77 = bitcast %type.3* %t36 to i8*
+ br label %b30
+
+b30:
+ %t78 = phi i8* [ %t76, %b28 ], [ %t77, %b29 ]
+ store i32 %t19, i32* %t0, align 4
+ %t79 = getelementptr inbounds i8, i8* %t78, i32 %t19
+ br label %b31
+
+b31:
+ %t80 = phi i8* [ %t33, %b14 ], [ %t79, %b30 ]
+ store i8 0, i8* %t80, align 1
+ br label %b33
+
+b33:
+ ret %type.0* %p0
+}
+
+declare void @llvm.memcpy.p0i8.p0i8.i32(i8* nocapture writeonly, i8* nocapture readonly, i32, i32, i1) #0
+declare void @llvm.memmove.p0i8.p0i8.i32(i8* nocapture, i8* nocapture readonly, i32, i32, i1) #0
+
+declare void @blah(%type.4*) local_unnamed_addr
+declare void @danny(%type.4*) local_unnamed_addr
+declare void @sammy(%type.0*, i32, i32) local_unnamed_addr align 2
+
+attributes #0 = { argmemonly nounwind }
+attributes #1 = { nounwind }
diff --git a/test/CodeGen/Hexagon/rdf-ignore-undef.ll b/test/CodeGen/Hexagon/rdf-ignore-undef.ll
new file mode 100644
index 000000000000..5d72318f420f
--- /dev/null
+++ b/test/CodeGen/Hexagon/rdf-ignore-undef.ll
@@ -0,0 +1,55 @@
+; RUN: llc -march=hexagon -verify-machineinstrs < %s | FileCheck %s
+; Check that we don't crash.
+; CHECK: call foo
+
+target triple = "hexagon"
+
+%struct.1 = type { i16, i8, i32, i8*, i8*, i8*, i8*, i8*, i8*, i32* }
+%struct.0 = type { i32, i32, i32, i32, i32, i32, i32, i32, i32 }
+
+declare void @foo(i8*, %struct.0*) local_unnamed_addr #0
+declare void @bar(%struct.1*, %struct.0* readonly) local_unnamed_addr #0
+
+define i32 @fred(i32 %argc, i8** nocapture readonly %argv) local_unnamed_addr #0 {
+entry:
+ br label %do.body
+
+do.body: ; preds = %if.end88.do.body_crit_edge, %entry
+ %cond = icmp eq i32 undef, 0
+ br i1 %cond, label %if.end49, label %if.then124
+
+if.end49: ; preds = %do.body
+ call void @foo(i8* nonnull undef, %struct.0* nonnull undef) #0
+ br i1 undef, label %if.end55, label %if.then53
+
+if.then53: ; preds = %if.end49
+ call void @bar(%struct.1* null, %struct.0* nonnull undef)
+ br label %if.end55
+
+if.end55: ; preds = %if.then53, %if.end49
+ switch i8 undef, label %sw.epilog79 [
+ i8 0, label %sw.bb77
+ i8 3, label %sw.bb77
+ ]
+
+sw.bb77: ; preds = %if.end55, %if.end55
+ br label %sw.epilog79
+
+sw.epilog79: ; preds = %sw.bb77, %if.end55
+ br i1 undef, label %if.end88, label %if.then81
+
+if.then81: ; preds = %sw.epilog79
+ br label %if.end88
+
+if.end88: ; preds = %if.then81, %sw.epilog79
+ store float 0.000000e+00, float* undef, align 4
+ br i1 undef, label %if.end88.do.body_crit_edge, label %if.then124
+
+if.end88.do.body_crit_edge: ; preds = %if.end88
+ br label %do.body
+
+if.then124: ; preds = %if.end88, %do.body
+ unreachable
+}
+
+attributes #0 = { nounwind }
diff --git a/test/CodeGen/Hexagon/rdf-multiple-phis-up.ll b/test/CodeGen/Hexagon/rdf-multiple-phis-up.ll
new file mode 100644
index 000000000000..d23846ac6ed4
--- /dev/null
+++ b/test/CodeGen/Hexagon/rdf-multiple-phis-up.ll
@@ -0,0 +1,40 @@
+; RUN: llc -march=hexagon < %s | FileCheck %s
+; REQUIRES: asserts
+
+; Check that we do not crash.
+; CHECK: call foo
+
+target triple = "hexagon"
+
+%struct.0 = type { i8*, i8*, [2 x i8*], i32, i32, i8*, i32, i32, i32, i32, i32, [2 x i32], i32, i32, i32, i32, i32, i32, i32, i32, i32, i32 }
+
+define i32 @fred(i8* %p0) local_unnamed_addr #0 {
+entry:
+ %0 = bitcast i8* %p0 to %struct.0*
+ br i1 undef, label %if.then21, label %for.body.i
+
+if.then21: ; preds = %entry
+ %.pr = load i32, i32* undef, align 4
+ switch i32 %.pr, label %cleanup [
+ i32 1, label %for.body.i
+ i32 3, label %if.then60
+ ]
+
+for.body.i: ; preds = %for.body.i, %if.then21, %entry
+ %1 = load i8, i8* undef, align 1
+ %cmp7.i = icmp ugt i8 %1, -17
+ br i1 %cmp7.i, label %cleanup, label %for.body.i
+
+if.then60: ; preds = %if.then21
+ %call61 = call i32 @foo(%struct.0* nonnull %0) #0
+ br label %cleanup
+
+cleanup: ; preds = %if.then60, %for.body.i, %if.then21
+ ret i32 undef
+}
+
+declare i32 @foo(%struct.0*) local_unnamed_addr #0
+
+
+attributes #0 = { nounwind }
+
diff --git a/test/CodeGen/Hexagon/rdf-phi-shadows.ll b/test/CodeGen/Hexagon/rdf-phi-shadows.ll
new file mode 100644
index 000000000000..f26ab9b0cef2
--- /dev/null
+++ b/test/CodeGen/Hexagon/rdf-phi-shadows.ll
@@ -0,0 +1,64 @@
+; RUN: llc -march=hexagon -verify-machineinstrs < %s | FileCheck %s
+; Check that we don't crash.
+; CHECK: call printf
+target triple = "hexagon"
+
+%struct.1 = type { i16, i8, i32, i8*, i8*, i8*, i8*, i8*, i8*, i32* }
+%struct.0 = type { i32, i32, i32, i32, i32, i32, i32, i32, i32 }
+
+declare void @foo(%struct.1*, %struct.0* readonly) local_unnamed_addr #0
+declare zeroext i8 @bar() local_unnamed_addr #0
+declare i32 @printf(i8* nocapture readonly, ...) local_unnamed_addr #0
+
+@.str = private unnamed_addr constant [5 x i8] c"blah\00", align 1
+
+define i32 @main(i32 %argc, i8** nocapture readonly %argv) local_unnamed_addr #0 {
+entry:
+ %t0 = alloca %struct.0, align 4
+ br label %do.body
+
+do.body: ; preds = %if.end88.do.body_crit_edge, %entry
+ %cond = icmp eq i32 undef, 0
+ br i1 %cond, label %if.end49, label %if.then124
+
+if.end49: ; preds = %do.body
+ br i1 undef, label %if.end55, label %if.then53
+
+if.then53: ; preds = %if.end49
+ call void @foo(%struct.1* null, %struct.0* nonnull %t0)
+ br label %if.end55
+
+if.end55: ; preds = %if.then53, %if.end49
+ %call76 = call zeroext i8 @bar() #0
+ switch i8 %call76, label %sw.epilog79 [
+ i8 0, label %sw.bb77
+ i8 3, label %sw.bb77
+ ]
+
+sw.bb77: ; preds = %if.end55, %if.end55
+ unreachable
+
+sw.epilog79: ; preds = %if.end55
+ br i1 undef, label %if.end88, label %if.then81
+
+if.then81: ; preds = %sw.epilog79
+ %div87 = fdiv float 0.000000e+00, undef
+ br label %if.end88
+
+if.end88: ; preds = %if.then81, %sw.epilog79
+ %t1 = phi float [ undef, %sw.epilog79 ], [ %div87, %if.then81 ]
+ %div89 = fdiv float 1.000000e+00, %t1
+ %.sroa.speculated = select i1 undef, float 0.000000e+00, float undef
+ %conv108 = fpext float %.sroa.speculated to double
+ %call113 = call i32 (i8*, ...) @printf(i8* getelementptr inbounds ([5 x i8], [5 x i8]* @.str, i32 0, i32 0), double undef, double %conv108, i64 undef, i32 undef) #0
+ br i1 undef, label %if.end88.do.body_crit_edge, label %if.then124
+
+if.end88.do.body_crit_edge: ; preds = %if.end88
+ br label %do.body
+
+if.then124: ; preds = %if.end88, %do.body
+ %t2 = phi float [ undef, %do.body ], [ %t1, %if.end88 ]
+ ret i32 0
+}
+
+attributes #0 = { nounwind }
diff --git a/test/CodeGen/Hexagon/rdf-phi-up.ll b/test/CodeGen/Hexagon/rdf-phi-up.ll
new file mode 100644
index 000000000000..28f4c90c174d
--- /dev/null
+++ b/test/CodeGen/Hexagon/rdf-phi-up.ll
@@ -0,0 +1,60 @@
+; RUN: llc -march=hexagon -verify-machineinstrs < %s | FileCheck %s
+; Check that this testcase compiles successfully.
+; CHECK-LABEL: fred:
+; CHECK: call foo
+
+target triple = "hexagon"
+
+%struct.0 = type { i32, i16, i8* }
+
+declare void @llvm.lifetime.start(i64, i8* nocapture) #1
+declare void @llvm.lifetime.end(i64, i8* nocapture) #1
+
+define i32 @fred(i8* readonly %p0, i32* %p1) local_unnamed_addr #0 {
+entry:
+ %v0 = alloca i16, align 2
+ %v1 = icmp eq i8* %p0, null
+ br i1 %v1, label %if.then, label %lor.lhs.false
+
+lor.lhs.false: ; preds = %entry
+ %v2 = bitcast i8* %p0 to %struct.0**
+ %v3 = load %struct.0*, %struct.0** %v2, align 4
+ %v4 = icmp eq %struct.0* %v3, null
+ br i1 %v4, label %if.then, label %if.else
+
+if.then: ; preds = %lor.lhs.false, %ent
+ %v5 = icmp eq i32* %p1, null
+ br i1 %v5, label %cleanup, label %if.then3
+
+if.then3: ; preds = %if.then
+ store i32 0, i32* %p1, align 4
+ br label %cleanup
+
+if.else: ; preds = %lor.lhs.false
+ %v6 = bitcast i16* %v0 to i8*
+ call void @llvm.lifetime.start(i64 2, i8* nonnull %v6) #0
+ store i16 0, i16* %v0, align 2
+ %v7 = call i32 @foo(%struct.0* nonnull %v3, i16* nonnull %v0) #0
+ %v8 = icmp eq i32* %p1, null
+ br i1 %v8, label %if.end7, label %if.then6
+
+if.then6: ; preds = %if.else
+ %v9 = load i16, i16* %v0, align 2
+ %v10 = zext i16 %v9 to i32
+ store i32 %v10, i32* %p1, align 4
+ br label %if.end7
+
+if.end7: ; preds = %if.else, %if.then6
+ call void @llvm.lifetime.end(i64 2, i8* nonnull %v6) #0
+ br label %cleanup
+
+cleanup: ; preds = %if.then3, %if.then,
+ %v11 = phi i32 [ %v7, %if.end7 ], [ -2147024809, %if.then ], [ -2147024809, %if.then3 ]
+ ret i32 %v11
+}
+
+declare i32 @foo(%struct.0*, i16*) local_unnamed_addr #0
+
+attributes #0 = { nounwind }
+attributes #1 = { argmemonly nounwind }
+
diff --git a/test/CodeGen/Hexagon/regalloc-bad-undef.mir b/test/CodeGen/Hexagon/regalloc-bad-undef.mir
new file mode 100644
index 000000000000..d8fbb92b0d50
--- /dev/null
+++ b/test/CodeGen/Hexagon/regalloc-bad-undef.mir
@@ -0,0 +1,204 @@
+# RUN: llc -march=hexagon -hexagon-subreg-liveness -start-after machine-scheduler -stop-after stack-slot-coloring -o - %s | FileCheck %s
+
+--- |
+ target triple = "hexagon"
+
+ ; Function Attrs: nounwind optsize
+ define void @main() #0 {
+ entry:
+ br label %for.body
+
+ for.body: ; preds = %if.end82, %entry
+ %lsr.iv = phi i32 [ %lsr.iv.next, %if.end82 ], [ 524288, %entry ]
+ %call9 = tail call i32 @lrand48() #0
+ %conv10 = sext i32 %call9 to i64
+ %shr11 = lshr i64 %conv10, 9
+ %or12 = or i64 0, %shr11
+ %call14 = tail call i32 @lrand48() #0
+ %conv15138 = zext i32 %call14 to i64
+ %shr16 = lshr i64 %conv15138, 9
+ %0 = call i64 @llvm.hexagon.S2.extractup(i64 %conv15138, i32 22, i32 9)
+ %1 = shl i64 %0, 42
+ %shl17 = shl i64 %shr16, 42
+ %or22 = or i64 0, %1
+ %or26 = or i64 %or22, 0
+ %shr30 = lshr i64 undef, 25
+ %2 = call i64 @llvm.hexagon.S2.extractup(i64 undef, i32 6, i32 25)
+ %and = and i64 %shr30, 63
+ %sub = shl i64 2, %2
+ %add = add i64 %sub, -1
+ %shl37 = shl i64 %add, 0
+ %call38 = tail call i32 @lrand48() #0
+ %conv39141 = zext i32 %call38 to i64
+ %shr40 = lshr i64 %conv39141, 25
+ %3 = call i64 @llvm.hexagon.S2.extractup(i64 %conv39141, i32 6, i32 25)
+ %and41 = and i64 %shr40, 63
+ %sub43 = shl i64 2, %3
+ %add45 = add i64 %sub43, -1
+ %call46 = tail call i32 @lrand48() #0
+ %shr48 = lshr i64 undef, 25
+ %4 = call i64 @llvm.hexagon.S2.extractup(i64 undef, i32 6, i32 25)
+ %and49 = and i64 %shr48, 63
+ %shl50 = shl i64 %add45, %4
+ %and52 = and i64 %shl37, %or12
+ %and54 = and i64 %shl50, %or26
+ store i64 %and54, i64* undef, align 8
+ %cmp56 = icmp eq i64 %and52, 0
+ br i1 %cmp56, label %for.end, label %if.end82
+
+ if.end82: ; preds = %for.body
+ %lsr.iv.next = add nsw i32 %lsr.iv, -1
+ %exitcond = icmp eq i32 %lsr.iv.next, 0
+ br i1 %exitcond, label %for.end, label %for.body
+
+ for.end: ; preds = %if.end82, %for.body
+ unreachable
+ }
+
+ declare i32 @lrand48() #0
+ declare i64 @llvm.hexagon.S2.extractup(i64, i32, i32) #1
+
+ attributes #0 = { nounwind optsize "target-cpu"="hexagonv55" "target-features"="-hvx,-hvx-double" }
+ attributes #1 = { nounwind readnone }
+
+...
+---
+name: main
+alignment: 2
+tracksRegLiveness: true
+registers:
+ - { id: 0, class: intregs }
+ - { id: 1, class: intregs }
+ - { id: 2, class: intregs }
+ - { id: 3, class: intregs }
+ - { id: 4, class: doubleregs }
+ - { id: 5, class: intregs }
+ - { id: 6, class: doubleregs }
+ - { id: 7, class: doubleregs }
+ - { id: 8, class: doubleregs }
+ - { id: 9, class: doubleregs }
+ - { id: 10, class: intregs }
+ - { id: 11, class: doubleregs }
+ - { id: 12, class: doubleregs }
+ - { id: 13, class: doubleregs }
+ - { id: 14, class: intregs }
+ - { id: 15, class: doubleregs }
+ - { id: 16, class: doubleregs }
+ - { id: 17, class: intregs }
+ - { id: 18, class: doubleregs }
+ - { id: 19, class: intregs }
+ - { id: 20, class: doubleregs }
+ - { id: 21, class: doubleregs }
+ - { id: 22, class: doubleregs }
+ - { id: 23, class: intregs }
+ - { id: 24, class: doubleregs }
+ - { id: 25, class: predregs }
+ - { id: 26, class: predregs }
+ - { id: 27, class: intregs }
+ - { id: 28, class: intregs }
+ - { id: 29, class: doubleregs }
+ - { id: 30, class: intregs }
+ - { id: 31, class: intregs }
+ - { id: 32, class: doubleregs }
+ - { id: 33, class: intregs }
+ - { id: 34, class: intregs }
+ - { id: 35, class: doubleregs }
+ - { id: 36, class: doubleregs }
+ - { id: 37, class: intregs }
+ - { id: 38, class: intregs }
+ - { id: 39, class: doubleregs }
+ - { id: 40, class: doubleregs }
+ - { id: 41, class: intregs }
+ - { id: 42, class: intregs }
+ - { id: 43, class: doubleregs }
+ - { id: 44, class: intregs }
+ - { id: 45, class: intregs }
+ - { id: 46, class: doubleregs }
+ - { id: 47, class: doubleregs }
+ - { id: 48, class: doubleregs }
+ - { id: 49, class: doubleregs }
+ - { id: 50, class: doubleregs }
+ - { id: 51, class: doubleregs }
+ - { id: 52, class: intregs }
+ - { id: 53, class: intregs }
+ - { id: 54, class: intregs }
+ - { id: 55, class: doubleregs }
+ - { id: 56, class: doubleregs }
+ - { id: 57, class: intregs }
+ - { id: 58, class: intregs }
+ - { id: 59, class: intregs }
+frameInfo:
+ isFrameAddressTaken: false
+ isReturnAddressTaken: false
+ hasStackMap: false
+ hasPatchPoint: false
+ stackSize: 0
+ offsetAdjustment: 0
+ maxAlignment: 0
+ adjustsStack: false
+ hasCalls: true
+ maxCallFrameSize: 0
+ hasOpaqueSPAdjustment: false
+ hasVAStart: false
+ hasMustTailInVarArgFunc: false
+body: |
+ bb.0.entry:
+ successors: %bb.1.for.body
+
+ %59 = A2_tfrsi 524288
+ undef %32.isub_hi = A2_tfrsi 0
+ %8 = S2_extractup undef %9, 6, 25
+ %47 = A2_tfrpi 2
+ %13 = A2_tfrpi -1
+ %13 = S2_asl_r_p_acc %13, %47, %8.isub_lo
+ %51 = A2_tfrpi 0
+
+ ; CHECK: %d2 = S2_extractup undef %d0, 6, 25
+ ; CHECK: %d0 = A2_tfrpi 2
+ ; CHECK: %d13 = A2_tfrpi -1
+ ; CHECK-NOT: undef %r4
+
+ bb.1.for.body:
+ successors: %bb.3.for.end, %bb.2.if.end82
+
+ ADJCALLSTACKDOWN 0, implicit-def dead %r29, implicit-def dead %r30, implicit %r31, implicit %r30, implicit %r29
+ J2_call @lrand48, implicit-def dead %d0, implicit-def dead %d1, implicit-def dead %d2, implicit-def dead %d3, implicit-def dead %d4, implicit-def dead %d5, implicit-def dead %d6, implicit-def dead %d7, implicit-def dead %r28, implicit-def dead %r31, implicit-def dead %p0, implicit-def dead %p1, implicit-def dead %p2, implicit-def dead %p3, implicit-def dead %m0, implicit-def dead %m1, implicit-def dead %lc0, implicit-def dead %lc1, implicit-def dead %sa0, implicit-def dead %sa1, implicit-def dead %usr, implicit-def %usr_ovf, implicit-def dead %cs0, implicit-def dead %cs1, implicit-def dead %w0, implicit-def dead %w1, implicit-def dead %w2, implicit-def dead %w3, implicit-def dead %w4, implicit-def dead %w5, implicit-def dead %w6, implicit-def dead %w7, implicit-def dead %w8, implicit-def dead %w9, implicit-def dead %w10, implicit-def dead %w11, implicit-def dead %w12, implicit-def dead %w13, implicit-def dead %w14, implicit-def dead %w15, implicit-def dead %q0, implicit-def dead %q1, implicit-def dead %q2, implicit-def dead %q3, implicit-def %r0
+ ADJCALLSTACKUP 0, 0, implicit-def dead %r29, implicit-def dead %r30, implicit-def dead %r31, implicit %r29
+ undef %29.isub_lo = COPY killed %r0
+ %29.isub_hi = S2_asr_i_r %29.isub_lo, 31
+ ADJCALLSTACKDOWN 0, implicit-def dead %r29, implicit-def dead %r30, implicit %r31, implicit %r30, implicit %r29
+ J2_call @lrand48, implicit-def dead %d0, implicit-def dead %d1, implicit-def dead %d2, implicit-def dead %d3, implicit-def dead %d4, implicit-def dead %d5, implicit-def dead %d6, implicit-def dead %d7, implicit-def dead %r28, implicit-def dead %r31, implicit-def dead %p0, implicit-def dead %p1, implicit-def dead %p2, implicit-def dead %p3, implicit-def dead %m0, implicit-def dead %m1, implicit-def dead %lc0, implicit-def dead %lc1, implicit-def dead %sa0, implicit-def dead %sa1, implicit-def dead %usr, implicit-def %usr_ovf, implicit-def dead %cs0, implicit-def dead %cs1, implicit-def dead %w0, implicit-def dead %w1, implicit-def dead %w2, implicit-def dead %w3, implicit-def dead %w4, implicit-def dead %w5, implicit-def dead %w6, implicit-def dead %w7, implicit-def dead %w8, implicit-def dead %w9, implicit-def dead %w10, implicit-def dead %w11, implicit-def dead %w12, implicit-def dead %w13, implicit-def dead %w14, implicit-def dead %w15, implicit-def dead %q0, implicit-def dead %q1, implicit-def dead %q2, implicit-def dead %q3, implicit-def %r0
+ ADJCALLSTACKUP 0, 0, implicit-def dead %r29, implicit-def dead %r30, implicit-def dead %r31, implicit %r29
+ %32.isub_lo = COPY killed %r0
+ %7 = S2_extractup %32, 22, 9
+ ADJCALLSTACKDOWN 0, implicit-def dead %r29, implicit-def dead %r30, implicit %r31, implicit %r30, implicit %r29
+ J2_call @lrand48, implicit-def dead %d0, implicit-def dead %d1, implicit-def dead %d2, implicit-def dead %d3, implicit-def dead %d4, implicit-def dead %d5, implicit-def dead %d6, implicit-def dead %d7, implicit-def dead %r28, implicit-def dead %r31, implicit-def dead %p0, implicit-def dead %p1, implicit-def dead %p2, implicit-def dead %p3, implicit-def dead %m0, implicit-def dead %m1, implicit-def dead %lc0, implicit-def dead %lc1, implicit-def dead %sa0, implicit-def dead %sa1, implicit-def dead %usr, implicit-def %usr_ovf, implicit-def dead %cs0, implicit-def dead %cs1, implicit-def dead %w0, implicit-def dead %w1, implicit-def dead %w2, implicit-def dead %w3, implicit-def dead %w4, implicit-def dead %w5, implicit-def dead %w6, implicit-def dead %w7, implicit-def dead %w8, implicit-def dead %w9, implicit-def dead %w10, implicit-def dead %w11, implicit-def dead %w12, implicit-def dead %w13, implicit-def dead %w14, implicit-def dead %w15, implicit-def dead %q0, implicit-def dead %q1, implicit-def dead %q2, implicit-def dead %q3, implicit-def %r0
+ ADJCALLSTACKUP 0, 0, implicit-def dead %r29, implicit-def dead %r30, implicit-def dead %r31, implicit %r29
+ undef %43.isub_lo = COPY killed %r0
+ %43.isub_hi = COPY %32.isub_hi
+ %16 = S2_extractup %43, 6, 25
+ %18 = A2_tfrpi -1
+ %18 = S2_asl_r_p_acc %18, %47, %16.isub_lo
+ ADJCALLSTACKDOWN 0, implicit-def dead %r29, implicit-def dead %r30, implicit %r31, implicit %r30, implicit %r29
+ J2_call @lrand48, implicit-def dead %d0, implicit-def dead %d1, implicit-def dead %d2, implicit-def dead %d3, implicit-def dead %d4, implicit-def dead %d5, implicit-def dead %d6, implicit-def dead %d7, implicit-def dead %r28, implicit-def dead %r31, implicit-def dead %p0, implicit-def dead %p1, implicit-def dead %p2, implicit-def dead %p3, implicit-def dead %m0, implicit-def dead %m1, implicit-def dead %lc0, implicit-def dead %lc1, implicit-def dead %sa0, implicit-def dead %sa1, implicit-def dead %usr, implicit-def %usr_ovf, implicit-def dead %cs0, implicit-def dead %cs1, implicit-def dead %w0, implicit-def dead %w1, implicit-def dead %w2, implicit-def dead %w3, implicit-def dead %w4, implicit-def dead %w5, implicit-def dead %w6, implicit-def dead %w7, implicit-def dead %w8, implicit-def dead %w9, implicit-def dead %w10, implicit-def dead %w11, implicit-def dead %w12, implicit-def dead %w13, implicit-def dead %w14, implicit-def dead %w15, implicit-def dead %q0, implicit-def dead %q1, implicit-def dead %q2, implicit-def dead %q3
+ ADJCALLSTACKUP 0, 0, implicit-def dead %r29, implicit-def dead %r30, implicit-def dead %r31, implicit %r29
+ %22 = S2_asl_r_p %18, %8.isub_lo
+ %21 = COPY %13
+ %21 = S2_lsr_i_p_and %21, %29, 9
+ %22 = S2_asl_i_p_and %22, %7, 42
+ S2_storerd_io undef %23, 0, %22 :: (store 8 into `i64* undef`)
+ %25 = C2_cmpeqp %21, %51
+ J2_jumpt %25, %bb.3.for.end, implicit-def dead %pc
+ J2_jump %bb.2.if.end82, implicit-def dead %pc
+
+ bb.2.if.end82:
+ successors: %bb.3.for.end, %bb.1.for.body
+
+ %59 = A2_addi %59, -1
+ %26 = C2_cmpeqi %59, 0
+ J2_jumpf %26, %bb.1.for.body, implicit-def dead %pc
+ J2_jump %bb.3.for.end, implicit-def dead %pc
+
+ bb.3.for.end:
+
+...
diff --git a/test/CodeGen/Hexagon/sf-min-max.ll b/test/CodeGen/Hexagon/sf-min-max.ll
new file mode 100644
index 000000000000..e795cb42d6a2
--- /dev/null
+++ b/test/CodeGen/Hexagon/sf-min-max.ll
@@ -0,0 +1,67 @@
+; RUN: llc -march=hexagon < %s | FileCheck %s
+
+; CHECK-LABEL: sf_min_olt:
+; CHECK: sfmin
+define float @sf_min_olt(float %x, float %y) #0 {
+ %t = fcmp olt float %x, %y
+ %u = select i1 %t, float %x, float %y
+ ret float %u
+}
+
+; CHECK-LABEL: sf_min_ole:
+; CHECK: sfmin
+define float @sf_min_ole(float %x, float %y) #0 {
+ %t = fcmp ole float %x, %y
+ %u = select i1 %t, float %x, float %y
+ ret float %u
+}
+
+; CHECK-LABEL: sf_max_ogt:
+; CHECK: sfmax
+define float @sf_max_ogt(float %x, float %y) #0 {
+ %t = fcmp ogt float %x, %y
+ %u = select i1 %t, float %x, float %y
+ ret float %u
+}
+
+; CHECK-LABEL: sf_max_oge:
+; CHECK: sfmax
+define float @sf_max_oge(float %x, float %y) #0 {
+ %t = fcmp oge float %x, %y
+ %u = select i1 %t, float %x, float %y
+ ret float %u
+}
+
+; CHECK-LABEL: sf_max_olt:
+; CHECK: sfmax
+define float @sf_max_olt(float %x, float %y) #0 {
+ %t = fcmp olt float %x, %y
+ %u = select i1 %t, float %y, float %x
+ ret float %u
+}
+
+; CHECK-LABEL: sf_max_ole:
+; CHECK: sfmax
+define float @sf_max_ole(float %x, float %y) #0 {
+ %t = fcmp ole float %x, %y
+ %u = select i1 %t, float %y, float %x
+ ret float %u
+}
+
+; CHECK-LABEL: sf_min_ogt:
+; CHECK: sfmin
+define float @sf_min_ogt(float %x, float %y) #0 {
+ %t = fcmp ogt float %x, %y
+ %u = select i1 %t, float %y, float %x
+ ret float %u
+}
+
+; CHECK-LABEL: sf_min_oge:
+; CHECK: sfmin
+define float @sf_min_oge(float %x, float %y) #0 {
+ %t = fcmp oge float %x, %y
+ %u = select i1 %t, float %y, float %x
+ ret float %u
+}
+
+attributes #0 = { nounwind "target-cpu"="hexagonv5" }
diff --git a/test/CodeGen/Hexagon/sffms.ll b/test/CodeGen/Hexagon/sffms.ll
new file mode 100644
index 000000000000..ef47976ab3bb
--- /dev/null
+++ b/test/CodeGen/Hexagon/sffms.ll
@@ -0,0 +1,25 @@
+; RUN: llc -march=hexagon -fp-contract=fast < %s | FileCheck %s
+
+; Check that "Rx-=sfmpy(Rs,Rt)" is being generated for "fsub(fmul(..))"
+
+; CHECK: r{{[0-9]+}} -= sfmpy
+
+%struct.matrix_params = type { float** }
+
+; Function Attrs: norecurse nounwind
+define void @loop2_1(%struct.matrix_params* nocapture readonly %params, i32 %col1) #0 {
+entry:
+ %matrixA = getelementptr inbounds %struct.matrix_params, %struct.matrix_params* %params, i32 0, i32 0
+ %0 = load float**, float*** %matrixA, align 4
+ %1 = load float*, float** %0, align 4
+ %arrayidx1 = getelementptr inbounds float, float* %1, i32 %col1
+ %2 = load float, float* %arrayidx1, align 4
+ %arrayidx3 = getelementptr inbounds float*, float** %0, i32 %col1
+ %3 = load float*, float** %arrayidx3, align 4
+ %4 = load float, float* %3, align 4
+ %mul = fmul float %2, %4
+ %sub = fsub float %2, %mul
+ %arrayidx10 = getelementptr inbounds float, float* %3, i32 %col1
+ store float %sub, float* %arrayidx10, align 4
+ ret void
+}
diff --git a/test/CodeGen/Hexagon/split-const32-const64.ll b/test/CodeGen/Hexagon/split-const32-const64.ll
index 2815253545c5..95741462e508 100644
--- a/test/CodeGen/Hexagon/split-const32-const64.ll
+++ b/test/CodeGen/Hexagon/split-const32-const64.ll
@@ -1,24 +1,28 @@
-; RUN: llc -march=hexagon -mcpu=hexagonv5 -hexagon-small-data-threshold=0 < %s | FileCheck %s
+; RUN: llc -march=hexagon -hexagon-small-data-threshold=0 < %s | FileCheck %s
-; Check that CONST32/CONST64 instructions are 'not' generated when
-; small-data-threshold is set to 0.
+; Check that CONST32/CONST64 instructions are 'not' generated when the
+; small data threshold is set to 0.
-; with immediate value.
@a = external global i32
@b = external global i32
@la = external global i64
@lb = external global i64
-define void @test1() nounwind {
+; CHECK-LABEL: test1:
; CHECK-NOT: CONST32
+define void @test1() nounwind {
entry:
+ br label %block
+block:
store i32 12345670, i32* @a, align 4
- store i32 12345670, i32* @b, align 4
+ %q = ptrtoint i8* blockaddress (@test1, %block) to i32
+ store i32 %q, i32* @b, align 4
ret void
}
-define void @test2() nounwind {
+; CHECK-LABEL: test2:
; CHECK-NOT: CONST64
+define void @test2() nounwind {
entry:
store i64 1234567890123, i64* @la, align 8
store i64 1234567890123, i64* @lb, align 8
diff --git a/test/CodeGen/Hexagon/storerd-io-over-rr.ll b/test/CodeGen/Hexagon/storerd-io-over-rr.ll
new file mode 100644
index 000000000000..8727330ca5bd
--- /dev/null
+++ b/test/CodeGen/Hexagon/storerd-io-over-rr.ll
@@ -0,0 +1,12 @@
+; RUN: llc -march=hexagon < %s | FileCheck %s
+; Check for memd(base + #offset), instead of memd(base + reg<<#c).
+; CHECK: memd(r{{[0-9]+}}+#
+
+define void @fred(i32 %p, i64 %v) #0 {
+ %t0 = add i32 %p, 4
+ %t1 = inttoptr i32 %t0 to i64*
+ store i64 %v, i64* %t1
+ ret void
+}
+
+attributes #0 = { nounwind "target-cpu"="hexagonv60" }
diff --git a/test/CodeGen/Hexagon/struct_args.ll b/test/CodeGen/Hexagon/struct_args.ll
index 2ac1f8eadbb7..11c23b82ec4a 100644
--- a/test/CodeGen/Hexagon/struct_args.ll
+++ b/test/CodeGen/Hexagon/struct_args.ll
@@ -1,6 +1,6 @@
-; RUN: llc -march=hexagon -mcpu=hexagonv4 -disable-hsdr < %s | FileCheck %s
-; CHECK: r{{[0-9]}}:{{[0-9]}} = combine({{r[0-9]|#0}}, r{{[0-9]}})
-; CHECK: r{{[0-9]}}:{{[0-9]}} |= asl(r{{[0-9]}}:{{[0-9]}}, #32)
+; RUN: llc -march=hexagon -disable-hsdr < %s | FileCheck %s
+; CHECK-DAG: r0 = memw
+; CHECK-DAG: r1 = memw
%struct.small = type { i32, i32 }
@@ -8,7 +8,7 @@
define void @foo() nounwind {
entry:
- %0 = load i64, i64* bitcast (%struct.small* @s1 to i64*), align 1
+ %0 = load i64, i64* bitcast (%struct.small* @s1 to i64*), align 4
call void @bar(i64 %0)
ret void
}
diff --git a/test/CodeGen/Hexagon/subi-asl.ll b/test/CodeGen/Hexagon/subi-asl.ll
new file mode 100644
index 000000000000..f0b27e828f50
--- /dev/null
+++ b/test/CodeGen/Hexagon/subi-asl.ll
@@ -0,0 +1,70 @@
+; RUN: llc -march=hexagon < %s | FileCheck %s
+
+; Check if S4_subi_asl_ri is being generated correctly.
+
+; CHECK-LABEL: yes_sub_asl
+; CHECK: [[REG1:(r[0-9]+)]] = sub(#0, asl([[REG1]], #1))
+
+; CHECK-LABEL: no_sub_asl
+; CHECK: [[REG2:(r[0-9]+)]] = asl(r{{[0-9]+}}, #1)
+; CHECK: r{{[0-9]+}} = sub([[REG2]], r{{[0-9]+}})
+
+%struct.rtx_def = type { i16, i8 }
+
+@this_insn_number = external global i32, align 4
+
+; Function Attrs: nounwind
+define void @yes_sub_asl(%struct.rtx_def* %reg, %struct.rtx_def* nocapture readonly %setter) #0 {
+entry:
+ %code = getelementptr inbounds %struct.rtx_def, %struct.rtx_def* %reg, i32 0, i32 0
+ %0 = load i16, i16* %code, align 4
+ switch i16 %0, label %return [
+ i16 2, label %if.end
+ i16 5, label %if.end
+ ]
+
+if.end:
+ %code6 = getelementptr inbounds %struct.rtx_def, %struct.rtx_def* %setter, i32 0, i32 0
+ %1 = load i16, i16* %code6, align 4
+ %cmp8 = icmp eq i16 %1, 56
+ %conv9 = zext i1 %cmp8 to i32
+ %2 = load i32, i32* @this_insn_number, align 4
+ %3 = mul i32 %2, -2
+ %sub = add nsw i32 %conv9, %3
+ tail call void @reg_is_born(%struct.rtx_def* nonnull %reg, i32 %sub) #2
+ br label %return
+
+return:
+ ret void
+}
+
+declare void @reg_is_born(%struct.rtx_def*, i32) #1
+
+; Function Attrs: nounwind
+define void @no_sub_asl(%struct.rtx_def* %reg, %struct.rtx_def* nocapture readonly %setter) #0 {
+entry:
+ %code = getelementptr inbounds %struct.rtx_def, %struct.rtx_def* %reg, i32 0, i32 0
+ %0 = load i16, i16* %code, align 4
+ switch i16 %0, label %return [
+ i16 2, label %if.end
+ i16 5, label %if.end
+ ]
+
+if.end:
+ %1 = load i32, i32* @this_insn_number, align 4
+ %mul = mul nsw i32 %1, 2
+ %code6 = getelementptr inbounds %struct.rtx_def, %struct.rtx_def* %setter, i32 0, i32 0
+ %2 = load i16, i16* %code6, align 4
+ %cmp8 = icmp eq i16 %2, 56
+ %conv9 = zext i1 %cmp8 to i32
+ %sub = sub nsw i32 %mul, %conv9
+ tail call void @reg_is_born(%struct.rtx_def* nonnull %reg, i32 %sub) #2
+ br label %return
+
+return:
+ ret void
+}
+
+attributes #0 = { nounwind "target-cpu"="hexagonv5" }
+attributes #1 = { "target-cpu"="hexagonv5" }
+attributes #2 = { nounwind }
diff --git a/test/CodeGen/Hexagon/swp-const-tc.ll b/test/CodeGen/Hexagon/swp-const-tc.ll
new file mode 100644
index 000000000000..3113094d2ba3
--- /dev/null
+++ b/test/CodeGen/Hexagon/swp-const-tc.ll
@@ -0,0 +1,51 @@
+; RUN: llc -march=hexagon -mcpu=hexagonv5 -enable-pipeliner -verify-machineinstrs < %s | FileCheck %s
+
+; If the trip count is a compile-time constant, then decrement it instead
+; of computing a new LC0 value.
+
+; CHECK-LABEL: @test
+; CHECK: loop0(.LBB0_1, #998)
+
+define i32 @test(i32* %A, i32* %B, i32 %count) {
+entry:
+ br label %for.body
+
+for.body:
+ %sum.02 = phi i32 [ 0, %entry ], [ %add, %for.body ]
+ %arrayidx.phi = phi i32* [ %A, %entry ], [ %arrayidx.inc, %for.body ]
+ %i.01 = phi i32 [ 0, %entry ], [ %inc, %for.body ]
+ %0 = load i32, i32* %arrayidx.phi, align 4
+ %add = add nsw i32 %0, %sum.02
+ %inc = add nsw i32 %i.01, 1
+ %exitcond = icmp eq i32 %inc, 1000
+ %arrayidx.inc = getelementptr i32, i32* %arrayidx.phi, i32 1
+ br i1 %exitcond, label %for.end, label %for.body
+
+for.end:
+ ret i32 %add
+}
+
+; The constant trip count is small enough that the kernel is not executed.
+
+; CHECK-LABEL: @test1
+; CHECK-NOT: loop0(
+
+define i32 @test1(i32* %A, i32* %B, i32 %count) {
+entry:
+ br label %for.body
+
+for.body:
+ %sum.02 = phi i32 [ 0, %entry ], [ %add, %for.body ]
+ %arrayidx.phi = phi i32* [ %A, %entry ], [ %arrayidx.inc, %for.body ]
+ %i.01 = phi i32 [ 0, %entry ], [ %inc, %for.body ]
+ %0 = load i32, i32* %arrayidx.phi, align 4
+ %add = add nsw i32 %0, %sum.02
+ %inc = add nsw i32 %i.01, 1
+ %exitcond = icmp eq i32 %inc, 1
+ %arrayidx.inc = getelementptr i32, i32* %arrayidx.phi, i32 1
+ br i1 %exitcond, label %for.end, label %for.body
+
+for.end:
+ ret i32 %add
+}
+
diff --git a/test/CodeGen/Hexagon/swp-dag-phi.ll b/test/CodeGen/Hexagon/swp-dag-phi.ll
new file mode 100644
index 000000000000..54d9492ebac6
--- /dev/null
+++ b/test/CodeGen/Hexagon/swp-dag-phi.ll
@@ -0,0 +1,42 @@
+; RUN: llc -march=hexagon -mcpu=hexagonv5 -enable-pipeliner -pipeliner-max-stages=2 < %s
+; REQUIRES: asserts
+
+; This tests check that a dependence is created between a Phi and it's uses.
+; An assert occurs if the Phi dependences are not correct.
+
+define void @test1(i32* %f2, i32 %nc) {
+entry:
+ %i.011 = add i32 %nc, -1
+ %cmp12 = icmp sgt i32 %i.011, 1
+ br i1 %cmp12, label %for.body.preheader, label %for.end
+
+for.body.preheader:
+ %0 = add i32 %nc, -2
+ %scevgep = getelementptr i32, i32* %f2, i32 %0
+ %sri = load i32, i32* %scevgep, align 4
+ %scevgep15 = getelementptr i32, i32* %f2, i32 %i.011
+ %sri16 = load i32, i32* %scevgep15, align 4
+ br label %for.body
+
+for.body:
+ %i.014 = phi i32 [ %i.0, %for.body ], [ %i.011, %for.body.preheader ]
+ %i.0.in13 = phi i32 [ %i.014, %for.body ], [ %nc, %for.body.preheader ]
+ %sr = phi i32 [ %1, %for.body ], [ %sri, %for.body.preheader ]
+ %sr17 = phi i32 [ %sr, %for.body ], [ %sri16, %for.body.preheader ]
+ %arrayidx = getelementptr inbounds i32, i32* %f2, i32 %i.014
+ %sub1 = add nsw i32 %i.0.in13, -3
+ %arrayidx2 = getelementptr inbounds i32, i32* %f2, i32 %sub1
+ %1 = load i32, i32* %arrayidx2, align 4
+ %sub3 = sub nsw i32 %sr17, %1
+ store i32 %sub3, i32* %arrayidx, align 4
+ %i.0 = add nsw i32 %i.014, -1
+ %cmp = icmp sgt i32 %i.0, 1
+ br i1 %cmp, label %for.body, label %for.end.loopexit
+
+for.end.loopexit:
+ br label %for.end
+
+for.end:
+ ret void
+}
+
diff --git a/test/CodeGen/Hexagon/swp-epilog-phi10.ll b/test/CodeGen/Hexagon/swp-epilog-phi10.ll
new file mode 100644
index 000000000000..ff35e0b30f35
--- /dev/null
+++ b/test/CodeGen/Hexagon/swp-epilog-phi10.ll
@@ -0,0 +1,88 @@
+; RUN: llc -march=hexagon -mcpu=hexagonv5 < %s
+; REQUIRES: asserts
+
+define void @test(i8* noalias nocapture readonly %src, i32 %srcStride) local_unnamed_addr #0 {
+entry:
+ %add.ptr = getelementptr inbounds i8, i8* %src, i32 %srcStride
+ %add.ptr2 = getelementptr inbounds i8, i8* %add.ptr, i32 %srcStride
+ %add.ptr3 = getelementptr inbounds i8, i8* %add.ptr2, i32 %srcStride
+ br label %for.body9.epil
+
+for.body9.epil:
+ %inc.sink385.epil = phi i32 [ %add17.epil, %for.body9.epil ], [ 2, %entry ]
+ %sr.epil = phi i8 [ %0, %for.body9.epil ], [ undef, %entry ]
+ %sr431.epil = phi i8 [ %2, %for.body9.epil ], [ 0, %entry ]
+ %sr432.epil = phi i8 [ %sr431.epil, %for.body9.epil ], [ 0, %entry ]
+ %epil.iter = phi i32 [ %epil.iter.sub, %for.body9.epil ], [ undef, %entry ]
+ %sub11.epil = add i32 %inc.sink385.epil, -1
+ %add17.epil = add nuw i32 %inc.sink385.epil, 1
+ %conv19.epil = zext i8 %sr.epil to i32
+ %add21.epil = add i32 %inc.sink385.epil, 2
+ %arrayidx22.epil = getelementptr inbounds i8, i8* %src, i32 %add21.epil
+ %0 = load i8, i8* %arrayidx22.epil, align 1
+ %conv23.epil = zext i8 %0 to i32
+ %1 = load i8, i8* undef, align 1
+ %conv42.epil = zext i8 %1 to i32
+ %conv53.epil = zext i8 %sr432.epil to i32
+ %2 = load i8, i8* undef, align 1
+ %conv61.epil = zext i8 %2 to i32
+ %3 = load i8, i8* undef, align 1
+ %conv65.epil = zext i8 %3 to i32
+ %4 = load i8, i8* null, align 1
+ %conv69.epil = zext i8 %4 to i32
+ %5 = load i8, i8* undef, align 1
+ %conv72.epil = zext i8 %5 to i32
+ %6 = load i8, i8* undef, align 1
+ %conv76.epil = zext i8 %6 to i32
+ %7 = load i8, i8* undef, align 1
+ %conv80.epil = zext i8 %7 to i32
+ %8 = load i8, i8* undef, align 1
+ %conv84.epil = zext i8 %8 to i32
+ %9 = load i8, i8* undef, align 1
+ %conv88.epil = zext i8 %9 to i32
+ %10 = load i8, i8* undef, align 1
+ %conv91.epil = zext i8 %10 to i32
+ %11 = load i8, i8* undef, align 1
+ %conv95.epil = zext i8 %11 to i32
+ %12 = load i8, i8* undef, align 1
+ %conv99.epil = zext i8 %12 to i32
+ %add.epil = add nuw nsw i32 0, %conv19.epil
+ %add16.epil = add nuw nsw i32 %add.epil, 0
+ %add20.epil = add nuw nsw i32 %add16.epil, 0
+ %add24.epil = add nuw nsw i32 %add20.epil, 0
+ %add28.epil = add nuw nsw i32 %add24.epil, 0
+ %add32.epil = add nuw nsw i32 %add28.epil, 0
+ %add35.epil = add i32 %add32.epil, 0
+ %add39.epil = add i32 %add35.epil, 0
+ %add43.epil = add i32 %add39.epil, %conv53.epil
+ %add47.epil = add i32 %add43.epil, 0
+ %add51.epil = add i32 %add47.epil, 0
+ %add54.epil = add i32 %add51.epil, %conv23.epil
+ %add58.epil = add i32 %add54.epil, %conv42.epil
+ %add62.epil = add i32 %add58.epil, %conv61.epil
+ %add66.epil = add i32 %add62.epil, %conv65.epil
+ %add70.epil = add i32 %add66.epil, %conv69.epil
+ %add73.epil = add i32 %add70.epil, %conv72.epil
+ %add77.epil = add i32 %add73.epil, %conv76.epil
+ %add81.epil = add i32 %add77.epil, %conv80.epil
+ %add85.epil = add i32 %add81.epil, %conv84.epil
+ %add89.epil = add i32 %add85.epil, %conv88.epil
+ %add92.epil = add i32 %add89.epil, %conv91.epil
+ %add96.epil = add i32 %add92.epil, %conv95.epil
+ %add100.epil = add i32 %add96.epil, %conv99.epil
+ %mul.epil = mul nsw i32 %add100.epil, 2621
+ %add101.epil = add nsw i32 %mul.epil, 32768
+ %shr369.epil = lshr i32 %add101.epil, 16
+ %conv102.epil = trunc i32 %shr369.epil to i8
+ %arrayidx103.epil = getelementptr inbounds i8, i8* undef, i32 %inc.sink385.epil
+ store i8 %conv102.epil, i8* %arrayidx103.epil, align 1
+ %epil.iter.sub = add i32 %epil.iter, -1
+ %epil.iter.cmp = icmp eq i32 %epil.iter.sub, 0
+ br i1 %epil.iter.cmp, label %for.end, label %for.body9.epil
+
+for.end:
+ unreachable
+}
+
+attributes #0 = { norecurse nounwind "correctly-rounded-divide-sqrt-fp-math"="false" "disable-tail-calls"="false" "less-precise-fpmad"="false" "no-frame-pointer-elim"="true" "no-frame-pointer-elim-non-leaf" "no-infs-fp-math"="false" "no-jump-tables"="false" "no-nans-fp-math"="false" "no-signed-zeros-fp-math"="false" "stack-protector-buffer-size"="8" "target-cpu"="hexagonv5" "unsafe-fp-math"="false" "use-soft-float"="false" }
+
diff --git a/test/CodeGen/Hexagon/swp-epilog-reuse-1.ll b/test/CodeGen/Hexagon/swp-epilog-reuse-1.ll
new file mode 100644
index 000000000000..3b5dbe698cec
--- /dev/null
+++ b/test/CodeGen/Hexagon/swp-epilog-reuse-1.ll
@@ -0,0 +1,44 @@
+; RUN: llc -march=hexagon -mcpu=hexagonv60 < %s
+; REQUIRES: asserts
+
+; Test that the pipeliner reuses an existing Phi when generating the epilog
+; block. In this case, the original loops has a Phi whose operand is another
+; Phi. When the loop is pipelined, the Phi that generates the operand value
+; is used in two stages. This means the the Phi for the second stage can
+; be reused. The bug causes an assert due to an invalid virtual register error
+; in the live variable analysis.
+
+define void @test(i8* %a, i8* %b) #0 {
+entry:
+ br label %for.body6.us.prol
+
+for.body6.us.prol:
+ %i.065.us.prol = phi i32 [ 0, %entry ], [ %inc.us.prol, %for.body6.us.prol ]
+ %im1.064.us.prol = phi i32 [ undef, %entry ], [ %i.065.us.prol, %for.body6.us.prol ]
+ %prol.iter = phi i32 [ undef, %entry ], [ %prol.iter.sub, %for.body6.us.prol ]
+ %arrayidx8.us.prol = getelementptr inbounds i8, i8* %b, i32 %im1.064.us.prol
+ %0 = load i8, i8* %arrayidx8.us.prol, align 1
+ %conv9.us.prol = sext i8 %0 to i32
+ %add.us.prol = add nsw i32 %conv9.us.prol, 0
+ %add12.us.prol = add nsw i32 %add.us.prol, 0
+ %mul.us.prol = mul nsw i32 %add12.us.prol, 3
+ %conv13.us.prol = trunc i32 %mul.us.prol to i8
+ %arrayidx14.us.prol = getelementptr inbounds i8, i8* %a, i32 %i.065.us.prol
+ store i8 %conv13.us.prol, i8* %arrayidx14.us.prol, align 1
+ %inc.us.prol = add nuw nsw i32 %i.065.us.prol, 1
+ %prol.iter.sub = add i32 %prol.iter, -1
+ %prol.iter.cmp = icmp eq i32 %prol.iter.sub, 0
+ br i1 %prol.iter.cmp, label %for.body6.us, label %for.body6.us.prol
+
+for.body6.us:
+ %im2.063.us = phi i32 [ undef, %for.body6.us ], [ %im1.064.us.prol, %for.body6.us.prol ]
+ %arrayidx10.us = getelementptr inbounds i8, i8* %b, i32 %im2.063.us
+ %1 = load i8, i8* %arrayidx10.us, align 1
+ %conv11.us = sext i8 %1 to i32
+ %add12.us = add nsw i32 0, %conv11.us
+ %mul.us = mul nsw i32 %add12.us, 3
+ %conv13.us = trunc i32 %mul.us to i8
+ store i8 %conv13.us, i8* undef, align 1
+ br label %for.body6.us
+}
+
diff --git a/test/CodeGen/Hexagon/swp-epilog-reuse.ll b/test/CodeGen/Hexagon/swp-epilog-reuse.ll
new file mode 100644
index 000000000000..6a2ad73f2092
--- /dev/null
+++ b/test/CodeGen/Hexagon/swp-epilog-reuse.ll
@@ -0,0 +1,65 @@
+; RUN: llc -fp-contract=fast -O3 -march=hexagon -mcpu=hexagonv5 < %s
+; REQUIRES: asserts
+
+; Test that the pipeliner doesn't ICE due because the PHI generation
+; code in the epilog does not attempt to reuse an existing PHI.
+
+define void @test(float* noalias %srcImg, i32 %width, float* noalias %dstImg) {
+entry.split:
+ %shr = lshr i32 %width, 1
+ %incdec.ptr253 = getelementptr inbounds float, float* %dstImg, i32 2
+ br i1 undef, label %for.body, label %for.end
+
+for.body:
+ %dst.21518.reg2mem.0 = phi float* [ null, %while.end712 ], [ %incdec.ptr253, %entry.split ]
+ %dstEnd.01519 = phi float* [ %add.ptr725, %while.end712 ], [ undef, %entry.split ]
+ %add.ptr367 = getelementptr inbounds float, float* %srcImg, i32 undef
+ %dst.31487 = getelementptr inbounds float, float* %dst.21518.reg2mem.0, i32 1
+ br i1 undef, label %while.body661.preheader, label %while.end712
+
+while.body661.preheader:
+ %scevgep1941 = getelementptr float, float* %add.ptr367, i32 1
+ br label %while.body661.ur
+
+while.body661.ur:
+ %lsr.iv1942 = phi float* [ %scevgep1941, %while.body661.preheader ], [ undef, %while.body661.ur ]
+ %col1.31508.reg2mem.0.ur = phi float [ %col3.31506.reg2mem.0.ur, %while.body661.ur ], [ undef, %while.body661.preheader ]
+ %col4.31507.reg2mem.0.ur = phi float [ %add710.ur, %while.body661.ur ], [ 0.000000e+00, %while.body661.preheader ]
+ %col3.31506.reg2mem.0.ur = phi float [ %add689.ur, %while.body661.ur ], [ undef, %while.body661.preheader ]
+ %dst.41511.ur = phi float* [ %incdec.ptr674.ur, %while.body661.ur ], [ %dst.31487, %while.body661.preheader ]
+ %mul662.ur = fmul float %col1.31508.reg2mem.0.ur, 4.000000e+00
+ %add663.ur = fadd float undef, %mul662.ur
+ %add665.ur = fadd float %add663.ur, undef
+ %add667.ur = fadd float undef, %add665.ur
+ %add669.ur = fadd float undef, %add667.ur
+ %add670.ur = fadd float %col4.31507.reg2mem.0.ur, %add669.ur
+ %conv673.ur = fmul float %add670.ur, 3.906250e-03
+ %incdec.ptr674.ur = getelementptr inbounds float, float* %dst.41511.ur, i32 1
+ store float %conv673.ur, float* %dst.41511.ur, align 4
+ %scevgep1959 = getelementptr float, float* %lsr.iv1942, i32 -1
+ %0 = load float, float* %scevgep1959, align 4
+ %mul680.ur = fmul float %0, 4.000000e+00
+ %add681.ur = fadd float undef, %mul680.ur
+ %add684.ur = fadd float undef, %add681.ur
+ %add687.ur = fadd float undef, %add684.ur
+ %add689.ur = fadd float undef, %add687.ur
+ %add699.ur = fadd float undef, undef
+ %add703.ur = fadd float undef, %add699.ur
+ %add707.ur = fadd float undef, %add703.ur
+ %add710.ur = fadd float undef, %add707.ur
+ %cmp660.ur = icmp ult float* %incdec.ptr674.ur, %dstEnd.01519
+ br i1 %cmp660.ur, label %while.body661.ur, label %while.end712
+
+while.end712:
+ %dst.4.lcssa.reg2mem.0 = phi float* [ %dst.31487, %for.body ], [ undef, %while.body661.ur ]
+ %conv721 = fpext float undef to double
+ %mul722 = fmul double %conv721, 0x3F7111112119E8FB
+ %conv723 = fptrunc double %mul722 to float
+ store float %conv723, float* %dst.4.lcssa.reg2mem.0, align 4
+ %add.ptr725 = getelementptr inbounds float, float* %dstEnd.01519, i32 %shr
+ %cmp259 = icmp ult i32 undef, undef
+ br i1 %cmp259, label %for.body, label %for.end
+
+for.end:
+ ret void
+}
diff --git a/test/CodeGen/Hexagon/swp-matmul-bitext.ll b/test/CodeGen/Hexagon/swp-matmul-bitext.ll
new file mode 100644
index 000000000000..db5bb96d0bc9
--- /dev/null
+++ b/test/CodeGen/Hexagon/swp-matmul-bitext.ll
@@ -0,0 +1,75 @@
+; RUN: llc -march=hexagon -mcpu=hexagonv60 -enable-bsb-sched=0 -enable-pipeliner < %s | FileCheck %s
+; RUN: llc -march=hexagon -mcpu=hexagonv5 -enable-pipeliner < %s | FileCheck %s
+
+; From coremark. Test that we pipeline the matrix multiplication bitextract
+; function. The pipelined code should have two packets.
+
+; CHECK: loop0(.LBB0_[[LOOP:.]],
+; CHECK: .LBB0_[[LOOP]]:
+; CHECK: = extractu([[REG2:(r[0-9]+)]],
+; CHECK: = extractu([[REG2]],
+; CHECK: [[REG0:(r[0-9]+)]] = memh
+; CHECK: [[REG1:(r[0-9]+)]] = memh
+; CHECK: += mpyi
+; CHECK: [[REG2]] = mpyi([[REG0]], [[REG1]])
+; CHECK: endloop0
+
+%union_h2_sem_t = type { i32 }
+
+@sem_i = common global [0 x %union_h2_sem_t] zeroinitializer, align 4
+
+define void @matrix_mul_matrix_bitextract(i32 %N, i32* %C, i16* %A, i16* %B) {
+entry:
+ %cmp53 = icmp eq i32 %N, 0
+ br i1 %cmp53, label %for_end27, label %for_body3_lr_ph_us
+
+for_body3_lr_ph_us:
+ %i_054_us = phi i32 [ %inc26_us, %for_cond1_for_inc25_crit_edge_us ], [ 0, %entry ]
+ %0 = mul i32 %i_054_us, %N
+ %arrayidx9_us_us_gep = getelementptr i16, i16* %A, i32 %0
+ br label %for_body3_us_us
+
+for_cond1_for_inc25_crit_edge_us:
+ %inc26_us = add i32 %i_054_us, 1
+ %exitcond89 = icmp eq i32 %inc26_us, %N
+ br i1 %exitcond89, label %for_end27, label %for_body3_lr_ph_us
+
+for_body3_us_us:
+ %j_052_us_us = phi i32 [ %inc23_us_us, %for_cond4_for_inc22_crit_edge_us_us ], [ 0, %for_body3_lr_ph_us ]
+ %add_us_us = add i32 %j_052_us_us, %0
+ %arrayidx_us_us = getelementptr inbounds i32, i32* %C, i32 %add_us_us
+ store i32 0, i32* %arrayidx_us_us, align 4
+ br label %for_body6_us_us
+
+for_cond4_for_inc22_crit_edge_us_us:
+ store i32 %add21_us_us, i32* %arrayidx_us_us, align 4
+ %inc23_us_us = add i32 %j_052_us_us, 1
+ %exitcond88 = icmp eq i32 %inc23_us_us, %N
+ br i1 %exitcond88, label %for_cond1_for_inc25_crit_edge_us, label %for_body3_us_us
+
+for_body6_us_us:
+ %1 = phi i32 [ 0, %for_body3_us_us ], [ %add21_us_us, %for_body6_us_us ]
+ %arrayidx9_us_us_phi = phi i16* [ %arrayidx9_us_us_gep, %for_body3_us_us ], [ %arrayidx9_us_us_inc, %for_body6_us_us ]
+ %k_050_us_us = phi i32 [ 0, %for_body3_us_us ], [ %inc_us_us, %for_body6_us_us ]
+ %2 = load i16, i16* %arrayidx9_us_us_phi, align 2
+ %conv_us_us = sext i16 %2 to i32
+ %mul10_us_us = mul i32 %k_050_us_us, %N
+ %add11_us_us = add i32 %mul10_us_us, %j_052_us_us
+ %arrayidx12_us_us = getelementptr inbounds i16, i16* %B, i32 %add11_us_us
+ %3 = load i16, i16* %arrayidx12_us_us, align 2
+ %conv13_us_us = sext i16 %3 to i32
+ %mul14_us_us = mul nsw i32 %conv13_us_us, %conv_us_us
+ %shr47_us_us = lshr i32 %mul14_us_us, 2
+ %and_us_us = and i32 %shr47_us_us, 15
+ %shr1548_us_us = lshr i32 %mul14_us_us, 5
+ %and16_us_us = and i32 %shr1548_us_us, 127
+ %mul17_us_us = mul i32 %and_us_us, %and16_us_us
+ %add21_us_us = add i32 %mul17_us_us, %1
+ %inc_us_us = add i32 %k_050_us_us, 1
+ %exitcond87 = icmp eq i32 %inc_us_us, %N
+ %arrayidx9_us_us_inc = getelementptr i16, i16* %arrayidx9_us_us_phi, i32 1
+ br i1 %exitcond87, label %for_cond4_for_inc22_crit_edge_us_us, label %for_body6_us_us
+
+for_end27:
+ ret void
+}
diff --git a/test/CodeGen/Hexagon/swp-max.ll b/test/CodeGen/Hexagon/swp-max.ll
new file mode 100644
index 000000000000..038138ff2561
--- /dev/null
+++ b/test/CodeGen/Hexagon/swp-max.ll
@@ -0,0 +1,42 @@
+; RUN: llc -march=hexagon -mcpu=hexagonv5 -enable-pipeliner \
+; RUN: -pipeliner-max-stages=2 < %s | FileCheck %s
+
+@A = global [8 x i32] [i32 4, i32 -3, i32 5, i32 -2, i32 -1, i32 2, i32 6, i32 -2], align 8
+
+define i32 @test(i32 %Left, i32 %Right) {
+entry:
+ %add = add nsw i32 %Right, %Left
+ %div = sdiv i32 %add, 2
+ %cmp9 = icmp slt i32 %div, %Left
+ br i1 %cmp9, label %for.end, label %for.body.preheader
+
+for.body.preheader:
+ br label %for.body
+
+; CHECK: loop0(.LBB0_[[LOOP:.]],
+; CHECK: .LBB0_[[LOOP]]:
+; CHECK: [[REG1:(r[0-9]+)]] = max(r{{[0-9]+}}, [[REG1]])
+; CHECK: [[REG0:(r[0-9]+)]] = add([[REG2:(r[0-9]+)]], [[REG0]])
+; CHECK: [[REG2]] = memw
+; CHECK: endloop0
+
+for.body:
+ %MaxLeftBorderSum.012 = phi i32 [ %MaxLeftBorderSum.1, %for.body ], [ 0, %for.body.preheader ]
+ %i.011 = phi i32 [ %dec, %for.body ], [ %div, %for.body.preheader ]
+ %LeftBorderSum.010 = phi i32 [ %add1, %for.body ], [ 0, %for.body.preheader ]
+ %arrayidx = getelementptr inbounds [8 x i32], [8 x i32]* @A, i32 0, i32 %i.011
+ %0 = load i32, i32* %arrayidx, align 4
+ %add1 = add nsw i32 %0, %LeftBorderSum.010
+ %cmp2 = icmp sgt i32 %add1, %MaxLeftBorderSum.012
+ %MaxLeftBorderSum.1 = select i1 %cmp2, i32 %add1, i32 %MaxLeftBorderSum.012
+ %dec = add nsw i32 %i.011, -1
+ %cmp = icmp slt i32 %dec, %Left
+ br i1 %cmp, label %for.end.loopexit, label %for.body
+
+for.end.loopexit:
+ br label %for.end
+
+for.end:
+ %MaxLeftBorderSum.0.lcssa = phi i32 [ 0, %entry ], [ %MaxLeftBorderSum.1, %for.end.loopexit ]
+ ret i32 %MaxLeftBorderSum.0.lcssa
+}
diff --git a/test/CodeGen/Hexagon/swp-multi-loops.ll b/test/CodeGen/Hexagon/swp-multi-loops.ll
new file mode 100644
index 000000000000..56e8c6511000
--- /dev/null
+++ b/test/CodeGen/Hexagon/swp-multi-loops.ll
@@ -0,0 +1,75 @@
+; RUN: llc -march=hexagon -mcpu=hexagonv5 -enable-pipeliner < %s | FileCheck %s
+
+; Make sure we attempt to pipeline all inner most loops.
+
+; Check if the first loop is pipelined.
+; CHECK: loop0(.LBB0_[[LOOP:.]],
+; CHECK: .LBB0_[[LOOP]]:
+; CHECK: add(r{{[0-9]+}}, r{{[0-9]+}})
+; CHECK-NEXT: memw(r{{[0-9]+}}{{.*}}++{{.*}}#4)
+; CHECK-NEXT: endloop0
+
+; Check if the second loop is pipelined.
+; CHECK: loop0(.LBB0_[[LOOP:.]],
+; CHECK: .LBB0_[[LOOP]]:
+; CHECK: add(r{{[0-9]+}}, r{{[0-9]+}})
+; CHECK-NEXT: memw(r{{[0-9]+}}{{.*}}++{{.*}}#4)
+; CHECK-NEXT: endloop0
+
+define i32 @test(i32* %a, i32 %n, i32 %l) {
+entry:
+ %cmp23 = icmp sgt i32 %n, 0
+ br i1 %cmp23, label %for.body3.lr.ph.preheader, label %for.end14
+
+for.body3.lr.ph.preheader:
+ br label %for.body3.lr.ph
+
+for.body3.lr.ph:
+ %sum1.026 = phi i32 [ %add8, %for.inc12 ], [ 0, %for.body3.lr.ph.preheader ]
+ %sum.025 = phi i32 [ %add, %for.inc12 ], [ 0, %for.body3.lr.ph.preheader ]
+ %j.024 = phi i32 [ %inc13, %for.inc12 ], [ 0, %for.body3.lr.ph.preheader ]
+ br label %for.body3
+
+for.body3:
+ %sum.118 = phi i32 [ %sum.025, %for.body3.lr.ph ], [ %add, %for.body3 ]
+ %arrayidx.phi = phi i32* [ %a, %for.body3.lr.ph ], [ %arrayidx.inc, %for.body3 ]
+ %i.017 = phi i32 [ 0, %for.body3.lr.ph ], [ %inc, %for.body3 ]
+ %0 = load i32, i32* %arrayidx.phi, align 4
+ %add = add nsw i32 %0, %sum.118
+ %inc = add nsw i32 %i.017, 1
+ %exitcond = icmp eq i32 %inc, %n
+ %arrayidx.inc = getelementptr i32, i32* %arrayidx.phi, i32 1
+ br i1 %exitcond, label %for.end, label %for.body3
+
+for.end:
+ tail call void @bar(i32* %a) #2
+ br label %for.body6
+
+for.body6:
+ %sum1.121 = phi i32 [ %sum1.026, %for.end ], [ %add8, %for.body6 ]
+ %arrayidx7.phi = phi i32* [ %a, %for.end ], [ %arrayidx7.inc, %for.body6 ]
+ %i.120 = phi i32 [ 0, %for.end ], [ %inc10, %for.body6 ]
+ %1 = load i32, i32* %arrayidx7.phi, align 4
+ %add8 = add nsw i32 %1, %sum1.121
+ %inc10 = add nsw i32 %i.120, 1
+ %exitcond29 = icmp eq i32 %inc10, %n
+ %arrayidx7.inc = getelementptr i32, i32* %arrayidx7.phi, i32 1
+ br i1 %exitcond29, label %for.inc12, label %for.body6
+
+for.inc12:
+ %inc13 = add nsw i32 %j.024, 1
+ %exitcond30 = icmp eq i32 %inc13, %n
+ br i1 %exitcond30, label %for.end14.loopexit, label %for.body3.lr.ph
+
+for.end14.loopexit:
+ br label %for.end14
+
+for.end14:
+ %sum1.0.lcssa = phi i32 [ 0, %entry ], [ %add8, %for.end14.loopexit ]
+ %sum.0.lcssa = phi i32 [ 0, %entry ], [ %add, %for.end14.loopexit ]
+ %add15 = add nsw i32 %sum1.0.lcssa, %sum.0.lcssa
+ ret i32 %add15
+}
+
+declare void @bar(i32*)
+
diff --git a/test/CodeGen/Hexagon/swp-prolog-phi4.ll b/test/CodeGen/Hexagon/swp-prolog-phi4.ll
new file mode 100644
index 000000000000..5ed0514ef74c
--- /dev/null
+++ b/test/CodeGen/Hexagon/swp-prolog-phi4.ll
@@ -0,0 +1,65 @@
+; RUN: llc -march=hexagon -mcpu=hexagonv5 -verify-machineinstrs < %s
+
+; Test that the name rewriter code doesn't chase the Phi operands for
+; Phis that do not occur in the loop that is being pipelined.
+
+define void @test(i32 %srcStride) local_unnamed_addr #0 {
+entry:
+ br label %for.body
+
+for.body:
+ %add.ptr3.pn = phi i8* [ undef, %entry ], [ %src4.0394, %for.end ]
+ %src2.0390 = phi i8* [ undef, %entry ], [ %add.ptr3.pn, %for.end ]
+ %src4.0394 = getelementptr inbounds i8, i8* %add.ptr3.pn, i32 %srcStride
+ %sri414 = load i8, i8* undef, align 1
+ br i1 undef, label %for.body9.epil, label %for.body9.preheader.new
+
+for.body9.preheader.new:
+ br label %for.body9.epil
+
+for.body9.epil:
+ %inc.sink385.epil = phi i32 [ %add17.epil, %for.body9.epil ], [ 2, %for.body ], [ undef, %for.body9.preheader.new ]
+ %sr420.epil = phi i8 [ undef, %for.body9.epil ], [ %sri414, %for.body ], [ undef, %for.body9.preheader.new ]
+ %sr421.epil = phi i8 [ %sr420.epil, %for.body9.epil ], [ undef, %for.body ], [ undef, %for.body9.preheader.new ]
+ %sr422.epil = phi i8 [ %sr421.epil, %for.body9.epil ], [ 0, %for.body ], [ undef, %for.body9.preheader.new ]
+ %epil.iter = phi i32 [ %epil.iter.sub, %for.body9.epil ], [ undef, %for.body9.preheader.new ], [ undef, %for.body ]
+ %add17.epil = add nuw i32 %inc.sink385.epil, 1
+ %add21.epil = add i32 %inc.sink385.epil, 2
+ %arrayidx22.epil = getelementptr inbounds i8, i8* undef, i32 %add21.epil
+ %conv27.epil = zext i8 %sr422.epil to i32
+ %0 = load i8, i8* null, align 1
+ %conv61.epil = zext i8 %0 to i32
+ %arrayidx94.epil = getelementptr inbounds i8, i8* %src4.0394, i32 %add17.epil
+ %1 = load i8, i8* %arrayidx94.epil, align 1
+ %add35.epil = add i32 0, %conv27.epil
+ %add39.epil = add i32 %add35.epil, 0
+ %add43.epil = add i32 %add39.epil, 0
+ %add47.epil = add i32 %add43.epil, 0
+ %add51.epil = add i32 %add47.epil, 0
+ %add54.epil = add i32 %add51.epil, 0
+ %add58.epil = add i32 %add54.epil, 0
+ %add62.epil = add i32 %add58.epil, %conv61.epil
+ %add66.epil = add i32 %add62.epil, 0
+ %add70.epil = add i32 %add66.epil, 0
+ %add73.epil = add i32 %add70.epil, 0
+ %add77.epil = add i32 %add73.epil, 0
+ %add81.epil = add i32 %add77.epil, 0
+ %add85.epil = add i32 %add81.epil, 0
+ %add89.epil = add i32 %add85.epil, 0
+ %add92.epil = add i32 %add89.epil, 0
+ %add96.epil = add i32 %add92.epil, 0
+ %add100.epil = add i32 %add96.epil, 0
+ %mul.epil = mul nsw i32 %add100.epil, 2621
+ %add101.epil = add nsw i32 %mul.epil, 32768
+ %shr369.epil = lshr i32 %add101.epil, 16
+ %conv102.epil = trunc i32 %shr369.epil to i8
+ store i8 %conv102.epil, i8* undef, align 1
+ %epil.iter.sub = add i32 %epil.iter, -1
+ %epil.iter.cmp = icmp eq i32 %epil.iter.sub, 0
+ br i1 %epil.iter.cmp, label %for.end, label %for.body9.epil
+
+for.end:
+ br label %for.body
+}
+
+attributes #0 = { norecurse nounwind "correctly-rounded-divide-sqrt-fp-math"="false" "disable-tail-calls"="false" "less-precise-fpmad"="false" "no-frame-pointer-elim"="true" "no-frame-pointer-elim-non-leaf" "no-infs-fp-math"="false" "no-jump-tables"="false" "no-nans-fp-math"="false" "no-signed-zeros-fp-math"="false" "stack-protector-buffer-size"="8" "target-cpu"="hexagonv5" "unsafe-fp-math"="false" "use-soft-float"="false" }
diff --git a/test/CodeGen/Hexagon/swp-vect-dotprod.ll b/test/CodeGen/Hexagon/swp-vect-dotprod.ll
new file mode 100644
index 000000000000..3ff88452499e
--- /dev/null
+++ b/test/CodeGen/Hexagon/swp-vect-dotprod.ll
@@ -0,0 +1,41 @@
+; RUN: llc -march=hexagon -mcpu=hexagonv5 -enable-pipeliner < %s | FileCheck %s
+; RUN: llc -march=hexagon -mcpu=hexagonv5 -O2 < %s | FileCheck %s
+; RUN: llc -march=hexagon -mcpu=hexagonv5 -O3 < %s | FileCheck %s
+;
+; Check that we pipeline a vectorized dot product in a single packet.
+;
+; CHECK: {
+; CHECK: += mpyi
+; CHECK: += mpyi
+; CHECK: memd
+; CHECK: memd
+; CHECK: } :endloop0
+
+@a = common global [5000 x i32] zeroinitializer, align 8
+@b = common global [5000 x i32] zeroinitializer, align 8
+
+define i32 @vecMultGlobal() {
+entry:
+ br label %polly.loop_body
+
+polly.loop_after:
+ %0 = extractelement <2 x i32> %addp_vec, i32 0
+ %1 = extractelement <2 x i32> %addp_vec, i32 1
+ %add_sum = add i32 %0, %1
+ ret i32 %add_sum
+
+polly.loop_body:
+ %polly.loopiv13 = phi i32 [ 0, %entry ], [ %polly.next_loopiv, %polly.loop_body ]
+ %reduction.012 = phi <2 x i32> [ zeroinitializer, %entry ], [ %addp_vec, %polly.loop_body ]
+ %polly.next_loopiv = add nsw i32 %polly.loopiv13, 2
+ %p_arrayidx1 = getelementptr [5000 x i32], [5000 x i32]* @b, i32 0, i32 %polly.loopiv13
+ %p_arrayidx = getelementptr [5000 x i32], [5000 x i32]* @a, i32 0, i32 %polly.loopiv13
+ %vector_ptr = bitcast i32* %p_arrayidx1 to <2 x i32>*
+ %_p_vec_full = load <2 x i32>, <2 x i32>* %vector_ptr, align 8
+ %vector_ptr7 = bitcast i32* %p_arrayidx to <2 x i32>*
+ %_p_vec_full8 = load <2 x i32>, <2 x i32>* %vector_ptr7, align 8
+ %mulp_vec = mul <2 x i32> %_p_vec_full8, %_p_vec_full
+ %addp_vec = add <2 x i32> %mulp_vec, %reduction.012
+ %2 = icmp slt i32 %polly.next_loopiv, 5000
+ br i1 %2, label %polly.loop_body, label %polly.loop_after
+}
diff --git a/test/CodeGen/Hexagon/swp-vmult.ll b/test/CodeGen/Hexagon/swp-vmult.ll
new file mode 100644
index 000000000000..9018405274cd
--- /dev/null
+++ b/test/CodeGen/Hexagon/swp-vmult.ll
@@ -0,0 +1,33 @@
+; RUN: llc -march=hexagon -mcpu=hexagonv5 -enable-pipeliner < %s | FileCheck %s
+; RUN: llc -march=hexagon -mcpu=hexagonv5 -O3 < %s | FileCheck %s
+
+; Multiply and accumulate
+; CHECK: mpyi([[REG0:r([0-9]+)]], [[REG1:r([0-9]+)]])
+; CHECK-NEXT: add(r{{[0-9]+}}, #4)
+; CHECK-NEXT: [[REG0]] = memw(r{{[0-9]+}} + r{{[0-9]+}}<<#0)
+; CHECK-NEXT: [[REG1]] = memw(r{{[0-9]+}} + r{{[0-9]+}}<<#0)
+; CHECK-NEXT: endloop0
+
+define i32 @foo(i32* %a, i32* %b, i32 %n) {
+entry:
+ br label %for.body
+
+for.body:
+ %sum.03 = phi i32 [ 0, %entry ], [ %add, %for.body ]
+ %arrayidx.phi = phi i32* [ %a, %entry ], [ %arrayidx.inc, %for.body ]
+ %arrayidx1.phi = phi i32* [ %b, %entry ], [ %arrayidx1.inc, %for.body ]
+ %i.02 = phi i32 [ 0, %entry ], [ %inc, %for.body ]
+ %0 = load i32, i32* %arrayidx.phi, align 4
+ %1 = load i32, i32* %arrayidx1.phi, align 4
+ %mul = mul nsw i32 %1, %0
+ %add = add nsw i32 %mul, %sum.03
+ %inc = add nsw i32 %i.02, 1
+ %exitcond = icmp eq i32 %inc, 10000
+ %arrayidx.inc = getelementptr i32, i32* %arrayidx.phi, i32 1
+ %arrayidx1.inc = getelementptr i32, i32* %arrayidx1.phi, i32 1
+ br i1 %exitcond, label %for.end, label %for.body
+
+for.end:
+ ret i32 %add
+}
+
diff --git a/test/CodeGen/Hexagon/swp-vsum.ll b/test/CodeGen/Hexagon/swp-vsum.ll
new file mode 100644
index 000000000000..4756c644709f
--- /dev/null
+++ b/test/CodeGen/Hexagon/swp-vsum.ll
@@ -0,0 +1,29 @@
+; RUN: llc -march=hexagon -mcpu=hexagonv5 -enable-pipeliner < %s | FileCheck %s
+; RUN: llc -march=hexagon -mcpu=hexagonv5 -O3 < %s | FileCheck %s
+
+; Simple vector total.
+; CHECK: loop0(.LBB0_[[LOOP:.]],
+; CHECK: .LBB0_[[LOOP]]:
+; CHECK: add([[REG:r([0-9]+)]], r{{[0-9]+}})
+; CHECK-NEXT: add(r{{[0-9]+}}, #4)
+; CHECK-NEXT: [[REG]] = memw(r{{[0-9]+}} + r{{[0-9]+}}<<#0)
+; CHECK-NEXT: endloop0
+
+define i32 @foo(i32* %a, i32 %n) {
+entry:
+ br label %for.body
+
+for.body:
+ %sum.02 = phi i32 [ 0, %entry ], [ %add, %for.body ]
+ %arrayidx.phi = phi i32* [ %a, %entry ], [ %arrayidx.inc, %for.body ]
+ %i.01 = phi i32 [ 0, %entry ], [ %inc, %for.body ]
+ %0 = load i32, i32* %arrayidx.phi, align 4
+ %add = add nsw i32 %0, %sum.02
+ %inc = add nsw i32 %i.01, 1
+ %exitcond = icmp eq i32 %inc, 10000
+ %arrayidx.inc = getelementptr i32, i32* %arrayidx.phi, i32 1
+ br i1 %exitcond, label %for.end, label %for.body
+
+for.end:
+ ret i32 %add
+}
diff --git a/test/CodeGen/Hexagon/tailcall_fastcc_ccc.ll b/test/CodeGen/Hexagon/tailcall_fastcc_ccc.ll
new file mode 100644
index 000000000000..479fc15a8e8a
--- /dev/null
+++ b/test/CodeGen/Hexagon/tailcall_fastcc_ccc.ll
@@ -0,0 +1,22 @@
+; RUN: llc -march=hexagon < %s | FileCheck %s
+
+target triple = "hexagon"
+
+declare hidden fastcc void @callee(i32, i32) #0
+declare hidden void @callee2(i32, i32) #0
+
+; CHECK: jump callee
+define void @caller(i32 %pp) #0 {
+entry:
+ tail call fastcc void @callee(i32 %pp, i32 0)
+ ret void
+}
+
+; CHECK: jump callee2
+define void @caller2(i32 %pp) #0 {
+entry:
+ tail call fastcc void @callee2(i32 %pp, i32 0)
+ ret void
+}
+
+attributes #0 = { nounwind }
diff --git a/test/CodeGen/Hexagon/tls_static.ll b/test/CodeGen/Hexagon/tls_static.ll
index ad2ca716b70d..dbd3bd7b4ba8 100644
--- a/test/CodeGen/Hexagon/tls_static.ll
+++ b/test/CodeGen/Hexagon/tls_static.ll
@@ -1,4 +1,4 @@
-; RUN: llc -O0 -march=hexagon -relocation-model=static < %s | FileCheck %s
+; RUN: llc -O0 -mtriple=hexagon-- -relocation-model=static < %s | FileCheck %s
@dst_le = thread_local global i32 0, align 4
@src_le = thread_local global i32 0, align 4
diff --git a/test/CodeGen/Hexagon/two-crash.ll b/test/CodeGen/Hexagon/two-crash.ll
new file mode 100644
index 000000000000..0ab02cda8a07
--- /dev/null
+++ b/test/CodeGen/Hexagon/two-crash.ll
@@ -0,0 +1,23 @@
+; RUN: llc -march=hexagon < %s | FileCheck %s
+; This testcase crashed, because we propagated a reg:sub into a tied use.
+; The two-address pass rewrote it in a way that generated incorrect code.
+; CHECK: r{{[0-9]+}} += lsr(r{{[0-9]+}}, #16)
+
+target triple = "hexagon"
+
+define i64 @fred(i64 %x) local_unnamed_addr #0 {
+entry:
+ %t.sroa.0.0.extract.trunc = trunc i64 %x to i32
+ %t4.sroa.4.0.extract.shift = lshr i64 %x, 16
+ %add11 = add i32 0, %t.sroa.0.0.extract.trunc
+ %t14.sroa.3.0.extract.trunc = trunc i64 %t4.sroa.4.0.extract.shift to i32
+ %t14.sroa.4.0.extract.shift = lshr i64 %x, 24
+ %add21 = add i32 %add11, %t14.sroa.3.0.extract.trunc
+ %t24.sroa.3.0.extract.trunc = trunc i64 %t14.sroa.4.0.extract.shift to i32
+ %add31 = add i32 %add21, %t24.sroa.3.0.extract.trunc
+ %conv32.mask = and i32 %add31, 255
+ %conv33 = zext i32 %conv32.mask to i64
+ ret i64 %conv33
+}
+
+attributes #0 = { norecurse nounwind readnone }
diff --git a/test/CodeGen/Hexagon/v60-cur.ll b/test/CodeGen/Hexagon/v60-cur.ll
index fe24309f5b87..a7d4f6d310e4 100644
--- a/test/CodeGen/Hexagon/v60-cur.ll
+++ b/test/CodeGen/Hexagon/v60-cur.ll
@@ -1,4 +1,4 @@
-; RUN: llc -march=hexagon < %s | FileCheck %s
+; RUN: llc -march=hexagon -enable-pipeliner=false < %s | FileCheck %s
; Test that we generate a .cur
diff --git a/test/CodeGen/Hexagon/v60-vsel1.ll b/test/CodeGen/Hexagon/v60-vsel1.ll
new file mode 100644
index 000000000000..e673145c9d14
--- /dev/null
+++ b/test/CodeGen/Hexagon/v60-vsel1.ll
@@ -0,0 +1,69 @@
+; RUN: llc -march=hexagon < %s | FileCheck %s
+
+; CHECK: if (p{{[0-3]}}) v{{[0-9]+}} = v{{[0-9]+}}
+
+target triple = "hexagon"
+
+; Function Attrs: nounwind
+define void @fast9_detect_coarse(i8* nocapture readnone %img, i32 %xsize, i32 %stride, i32 %barrier, i32* nocapture %bitmask, i32 %boundary) #0 {
+entry:
+ %0 = bitcast i32* %bitmask to <16 x i32>*
+ %1 = mul i32 %boundary, -2
+ %sub = add i32 %1, %xsize
+ %rem = and i32 %boundary, 63
+ %add = add i32 %sub, %rem
+ %2 = tail call <16 x i32> @llvm.hexagon.V6.lvsplatw(i32 -1)
+ %3 = tail call <16 x i32> @llvm.hexagon.V6.lvsplatw(i32 1)
+ %4 = tail call <512 x i1> @llvm.hexagon.V6.pred.scalar2(i32 %add)
+ %5 = tail call <16 x i32> @llvm.hexagon.V6.vandqrt.acc(<16 x i32> %3, <512 x i1> %4, i32 12)
+ %and4 = and i32 %add, 511
+ %cmp = icmp eq i32 %and4, 0
+ %sMaskR.0 = select i1 %cmp, <16 x i32> %2, <16 x i32> %5
+ %cmp547 = icmp sgt i32 %add, 0
+ br i1 %cmp547, label %for.body.lr.ph, label %for.end
+
+for.body.lr.ph: ; preds = %entry
+ %6 = tail call <512 x i1> @llvm.hexagon.V6.pred.scalar2(i32 %boundary)
+ %7 = tail call <16 x i32> @llvm.hexagon.V6.vandqrt(<512 x i1> %6, i32 16843009)
+ %8 = tail call <16 x i32> @llvm.hexagon.V6.vnot(<16 x i32> %7)
+ %9 = add i32 %rem, %xsize
+ %10 = add i32 %9, -1
+ %11 = add i32 %10, %1
+ %12 = lshr i32 %11, 9
+ %13 = mul i32 %12, 16
+ %14 = add nuw nsw i32 %13, 16
+ %scevgep = getelementptr i32, i32* %bitmask, i32 %14
+ br label %for.body
+
+for.body: ; preds = %for.body.lr.ph, %for.body
+ %i.050 = phi i32 [ %add, %for.body.lr.ph ], [ %sub6, %for.body ]
+ %sMask.049 = phi <16 x i32> [ %8, %for.body.lr.ph ], [ %2, %for.body ]
+ %optr.048 = phi <16 x i32>* [ %0, %for.body.lr.ph ], [ %incdec.ptr, %for.body ]
+ %15 = tail call <16 x i32> @llvm.hexagon.V6.vand(<16 x i32> undef, <16 x i32> %sMask.049)
+ %incdec.ptr = getelementptr inbounds <16 x i32>, <16 x i32>* %optr.048, i32 1
+ store <16 x i32> %15, <16 x i32>* %optr.048, align 64
+ %sub6 = add nsw i32 %i.050, -512
+ %cmp5 = icmp sgt i32 %sub6, 0
+ br i1 %cmp5, label %for.body, label %for.cond.for.end_crit_edge
+
+for.cond.for.end_crit_edge: ; preds = %for.body
+ %scevgep51 = bitcast i32* %scevgep to <16 x i32>*
+ br label %for.end
+
+for.end: ; preds = %for.cond.for.end_crit_edge, %entry
+ %optr.0.lcssa = phi <16 x i32>* [ %scevgep51, %for.cond.for.end_crit_edge ], [ %0, %entry ]
+ %16 = load <16 x i32>, <16 x i32>* %optr.0.lcssa, align 64
+ %17 = tail call <16 x i32> @llvm.hexagon.V6.vand(<16 x i32> %16, <16 x i32> %sMaskR.0)
+ store <16 x i32> %17, <16 x i32>* %optr.0.lcssa, align 64
+ ret void
+}
+
+declare <16 x i32> @llvm.hexagon.V6.lvsplatw(i32) #1
+declare <512 x i1> @llvm.hexagon.V6.pred.scalar2(i32) #1
+declare <16 x i32> @llvm.hexagon.V6.vandqrt.acc(<16 x i32>, <512 x i1>, i32) #1
+declare <16 x i32> @llvm.hexagon.V6.vandqrt(<512 x i1>, i32) #1
+declare <16 x i32> @llvm.hexagon.V6.vnot(<16 x i32>) #1
+declare <16 x i32> @llvm.hexagon.V6.vand(<16 x i32>, <16 x i32>) #1
+
+attributes #0 = { nounwind "target-cpu"="hexagonv60" "target-features"="+hvx" }
+attributes #1 = { nounwind readnone }
diff --git a/test/CodeGen/Hexagon/v6vec-vprint.ll b/test/CodeGen/Hexagon/v6vec-vprint.ll
new file mode 100644
index 000000000000..224547c24b75
--- /dev/null
+++ b/test/CodeGen/Hexagon/v6vec-vprint.ll
@@ -0,0 +1,36 @@
+; RUN: llc -march=hexagon -mcpu=hexagonv60 -enable-hexagon-hvx -disable-hexagon-shuffle=0 -O2 -enable-hexagon-vector-print < %s | FileCheck --check-prefix=CHECK %s
+; RUN: llc -march=hexagon -mcpu=hexagonv60 -enable-hexagon-hvx -disable-hexagon-shuffle=0 -O2 -enable-hexagon-vector-print -trace-hex-vector-stores-only < %s | FileCheck --check-prefix=VSTPRINT %s
+; generate .long XXXX which is a vector debug print instruction.
+; CHECK: .long 0x1dffe0
+; CHECK: .long 0x1dffe0
+; CHECK: .long 0x1dffe0
+; VSTPRINT: .long 0x1dffe0
+; VSTPRINT-NOT: .long 0x1dffe0
+target datalayout = "e-p:32:32:32-i64:64:64-i32:32:32-i16:16:16-i1:32:32-f64:64:64-f32:32:32-v64:64:64-v32:32:32-a:0-n16:32"
+target triple = "hexagon"
+
+; Function Attrs: nounwind
+define void @do_vecs(i8* nocapture readonly %a, i8* nocapture readonly %b, i8* nocapture %c) #0 {
+entry:
+ %0 = bitcast i8* %a to <16 x i32>*
+ %1 = load <16 x i32>, <16 x i32>* %0, align 4, !tbaa !1
+ %2 = bitcast i8* %b to <16 x i32>*
+ %3 = load <16 x i32>, <16 x i32>* %2, align 4, !tbaa !1
+ %4 = tail call <16 x i32> @llvm.hexagon.V6.vaddw(<16 x i32> %1, <16 x i32> %3)
+ %5 = bitcast i8* %c to <16 x i32>*
+ store <16 x i32> %4, <16 x i32>* %5, align 4, !tbaa !1
+ ret void
+}
+
+; Function Attrs: nounwind readnone
+declare <16 x i32> @llvm.hexagon.V6.vaddw(<16 x i32>, <16 x i32>) #1
+
+attributes #0 = { nounwind "less-precise-fpmad"="false" "no-frame-pointer-elim"="true" "no-frame-pointer-elim-non-leaf" "no-infs-fp-math"="false" "no-nans-fp-math"="false" "stack-protector-buffer-size"="8" "unsafe-fp-math"="false" "use-soft-float"="false" }
+attributes #1 = { nounwind readnone }
+
+!llvm.ident = !{!0}
+
+!0 = !{!"QuIC LLVM Hexagon Clang version 7.x-pre-unknown"}
+!1 = !{!2, !2, i64 0}
+!2 = !{!"omnipotent char", !3, i64 0}
+!3 = !{!"Simple C/C++ TBAA"}
diff --git a/test/CodeGen/Hexagon/vassign-to-combine.ll b/test/CodeGen/Hexagon/vassign-to-combine.ll
new file mode 100644
index 000000000000..a9a0d51e43b6
--- /dev/null
+++ b/test/CodeGen/Hexagon/vassign-to-combine.ll
@@ -0,0 +1,56 @@
+; RUN: llc -march=hexagon < %s | FileCheck %s
+
+; This testcase is known to generate an opportunity for creating vcombine
+; in HexagonCopyToCombine.
+
+; CHECK: vcombine
+
+target triple = "hexagon-unknown--elf"
+
+declare <32 x i32> @llvm.hexagon.V6.hi.128B(<64 x i32>) #0
+declare <32 x i32> @llvm.hexagon.V6.vabsdiffuh.128B(<32 x i32>, <32 x i32>) #0
+declare <32 x i32> @llvm.hexagon.V6.vlalignbi.128B(<32 x i32>, <32 x i32>, i32) #0
+declare <32 x i32> @llvm.hexagon.V6.vsathub.128B(<32 x i32>, <32 x i32>) #0
+declare <64 x i32> @llvm.hexagon.V6.vaddh.dv.128B(<64 x i32>, <64 x i32>) #0
+declare <64 x i32> @llvm.hexagon.V6.vadduhsat.dv.128B(<64 x i32>, <64 x i32>) #0
+declare <64 x i32> @llvm.hexagon.V6.vaddubh.128B(<32 x i32>, <32 x i32>) #0
+declare <64 x i32> @llvm.hexagon.V6.vmpyub.128B(<32 x i32>, i32) #0
+
+define void @foo() local_unnamed_addr #1 {
+entry:
+ %0 = load <32 x i32>, <32 x i32>* undef, align 128
+ %1 = load <32 x i32>, <32 x i32>* null, align 128
+ br i1 undef, label %b2, label %b1
+
+b1: ; preds = %entry
+ %2 = tail call <32 x i32> @llvm.hexagon.V6.vlalignbi.128B(<32 x i32> %0, <32 x i32> %1, i32 1)
+ %3 = tail call <64 x i32> @llvm.hexagon.V6.vmpyub.128B(<32 x i32> %2, i32 33686018) #1
+ %4 = tail call <64 x i32> @llvm.hexagon.V6.vadduhsat.dv.128B(<64 x i32> undef, <64 x i32> %3) #1
+ %5 = tail call <32 x i32> @llvm.hexagon.V6.hi.128B(<64 x i32> %4)
+ %6 = tail call <32 x i32> @llvm.hexagon.V6.vabsdiffuh.128B(<32 x i32> %5, <32 x i32> undef) #1
+ %7 = tail call <64 x i32> @llvm.hexagon.V6.vaddubh.128B(<32 x i32> %6, <32 x i32> undef)
+ %8 = tail call <64 x i32> @llvm.hexagon.V6.vaddh.dv.128B(<64 x i32> undef, <64 x i32> %7) #1
+ %9 = tail call <32 x i32> @llvm.hexagon.V6.hi.128B(<64 x i32> %8) #1
+ %10 = tail call <32 x i32> @llvm.hexagon.V6.vsathub.128B(<32 x i32> %9, <32 x i32> undef) #1
+ store <32 x i32> %10, <32 x i32>* undef, align 128
+ br label %b2
+
+b2: ; preds = %b1, %entry
+ %c2.host31.sroa.3.2.unr.ph = phi <32 x i32> [ zeroinitializer, %b1 ], [ %0, %entry ]
+ %c2.host31.sroa.0.2.unr.ph = phi <32 x i32> [ %0, %b1 ], [ %1, %entry ]
+ %11 = tail call <32 x i32> @llvm.hexagon.V6.vlalignbi.128B(<32 x i32> %c2.host31.sroa.3.2.unr.ph, <32 x i32> %c2.host31.sroa.0.2.unr.ph, i32 1)
+ %12 = tail call <64 x i32> @llvm.hexagon.V6.vmpyub.128B(<32 x i32> %11, i32 33686018) #1
+ %13 = tail call <64 x i32> @llvm.hexagon.V6.vadduhsat.dv.128B(<64 x i32> undef, <64 x i32> %12) #1
+ %14 = tail call <32 x i32> @llvm.hexagon.V6.hi.128B(<64 x i32> %13)
+ %15 = tail call <32 x i32> @llvm.hexagon.V6.vabsdiffuh.128B(<32 x i32> %14, <32 x i32> undef) #1
+ %16 = tail call <64 x i32> @llvm.hexagon.V6.vaddubh.128B(<32 x i32> %15, <32 x i32> undef)
+ %17 = tail call <64 x i32> @llvm.hexagon.V6.vaddh.dv.128B(<64 x i32> undef, <64 x i32> %16) #1
+ %18 = tail call <32 x i32> @llvm.hexagon.V6.hi.128B(<64 x i32> %17) #1
+ %19 = tail call <32 x i32> @llvm.hexagon.V6.vsathub.128B(<32 x i32> %18, <32 x i32> undef) #1
+ store <32 x i32> %19, <32 x i32>* undef, align 128
+ ret void
+}
+
+attributes #0 = { nounwind readnone }
+attributes #1 = { nounwind "target-cpu"="hexagonv60" "target-features"="+hvx,+hvx-double" }
+
diff --git a/test/CodeGen/Hexagon/vdmpy-halide-test.ll b/test/CodeGen/Hexagon/vdmpy-halide-test.ll
new file mode 100644
index 000000000000..7e41bd4d20d4
--- /dev/null
+++ b/test/CodeGen/Hexagon/vdmpy-halide-test.ll
@@ -0,0 +1,167 @@
+; RUN: llc -march=hexagon < %s
+; REQUIRES: asserts
+
+; Thie tests checks a compiler assert. So the test just needs to compile for it to pass
+target triple = "hexagon-unknown--elf"
+
+%struct.buffer_t = type { i64, i8*, [4 x i32], [4 x i32], [4 x i32], i32, i8, i8, [6 x i8] }
+
+; Function Attrs: norecurse nounwind
+define i32 @__testOne(%struct.buffer_t* noalias nocapture readonly %inputOne.buffer, %struct.buffer_t* noalias nocapture readonly %inputTwo.buffer, %struct.buffer_t* noalias nocapture readonly %testOne.buffer) #0 {
+entry:
+ %buf_host = getelementptr inbounds %struct.buffer_t, %struct.buffer_t* %inputOne.buffer, i32 0, i32 1
+ %inputOne.host = load i8*, i8** %buf_host, align 4
+ %buf_min = getelementptr inbounds %struct.buffer_t, %struct.buffer_t* %inputOne.buffer, i32 0, i32 4, i32 0
+ %inputOne.min.0 = load i32, i32* %buf_min, align 4
+ %buf_host10 = getelementptr inbounds %struct.buffer_t, %struct.buffer_t* %inputTwo.buffer, i32 0, i32 1
+ %inputTwo.host = load i8*, i8** %buf_host10, align 4
+ %buf_min22 = getelementptr inbounds %struct.buffer_t, %struct.buffer_t* %inputTwo.buffer, i32 0, i32 4, i32 0
+ %inputTwo.min.0 = load i32, i32* %buf_min22, align 4
+ %buf_host27 = getelementptr inbounds %struct.buffer_t, %struct.buffer_t* %testOne.buffer, i32 0, i32 1
+ %testOne.host = load i8*, i8** %buf_host27, align 4
+ %buf_extent31 = getelementptr inbounds %struct.buffer_t, %struct.buffer_t* %testOne.buffer, i32 0, i32 2, i32 0
+ %testOne.extent.0 = load i32, i32* %buf_extent31, align 4
+ %buf_min39 = getelementptr inbounds %struct.buffer_t, %struct.buffer_t* %testOne.buffer, i32 0, i32 4, i32 0
+ %testOne.min.0 = load i32, i32* %buf_min39, align 4
+ %0 = ashr i32 %testOne.extent.0, 4
+ %1 = icmp sgt i32 %0, 0
+ br i1 %1, label %"for testOne.s0.x.x.preheader", label %"end for testOne.s0.x.x"
+
+"for testOne.s0.x.x.preheader": ; preds = %entry
+ %2 = bitcast i8* %inputOne.host to i16*
+ %3 = bitcast i8* %inputTwo.host to i16*
+ %4 = bitcast i8* %testOne.host to i32*
+ br label %"for testOne.s0.x.x"
+
+"for testOne.s0.x.x": ; preds = %"for testOne.s0.x.x", %"for testOne.s0.x.x.preheader"
+ %.phi = phi i32* [ %4, %"for testOne.s0.x.x.preheader" ], [ %.inc, %"for testOne.s0.x.x" ]
+ %testOne.s0.x.x = phi i32 [ 0, %"for testOne.s0.x.x.preheader" ], [ %50, %"for testOne.s0.x.x" ]
+ %5 = shl nsw i32 %testOne.s0.x.x, 4
+ %6 = add nsw i32 %5, %testOne.min.0
+ %7 = shl nsw i32 %6, 1
+ %8 = sub nsw i32 %7, %inputOne.min.0
+ %9 = getelementptr inbounds i16, i16* %2, i32 %8
+ %10 = bitcast i16* %9 to <16 x i16>*
+ %11 = load <16 x i16>, <16 x i16>* %10, align 2, !tbaa !5
+ %12 = add nsw i32 %8, 15
+ %13 = getelementptr inbounds i16, i16* %2, i32 %12
+ %14 = bitcast i16* %13 to <16 x i16>*
+ %15 = load <16 x i16>, <16 x i16>* %14, align 2, !tbaa !5
+ %16 = shufflevector <16 x i16> %11, <16 x i16> %15, <16 x i32> <i32 0, i32 2, i32 4, i32 6, i32 8, i32 10, i32 12, i32 14, i32 17, i32 19, i32 21, i32 23, i32 25, i32 27, i32 29, i32 31>
+ %17 = add nsw i32 %8, 1
+ %18 = getelementptr inbounds i16, i16* %2, i32 %17
+ %19 = bitcast i16* %18 to <16 x i16>*
+ %20 = load <16 x i16>, <16 x i16>* %19, align 2, !tbaa !5
+ %21 = add nsw i32 %8, 16
+ %22 = getelementptr inbounds i16, i16* %2, i32 %21
+ %23 = bitcast i16* %22 to <16 x i16>*
+ %24 = load <16 x i16>, <16 x i16>* %23, align 2, !tbaa !5
+ %25 = shufflevector <16 x i16> %20, <16 x i16> %24, <16 x i32> <i32 0, i32 2, i32 4, i32 6, i32 8, i32 10, i32 12, i32 14, i32 17, i32 19, i32 21, i32 23, i32 25, i32 27, i32 29, i32 31>
+ %26 = shufflevector <16 x i16> %16, <16 x i16> %25, <32 x i32> <i32 0, i32 16, i32 1, i32 17, i32 2, i32 18, i32 3, i32 19, i32 4, i32 20, i32 5, i32 21, i32 6, i32 22, i32 7, i32 23, i32 8, i32 24, i32 9, i32 25, i32 10, i32 26, i32 11, i32 27, i32 12, i32 28, i32 13, i32 29, i32 14, i32 30, i32 15, i32 31>
+ %27 = sub nsw i32 %7, %inputTwo.min.0
+ %28 = getelementptr inbounds i16, i16* %3, i32 %27
+ %29 = bitcast i16* %28 to <16 x i16>*
+ %30 = load <16 x i16>, <16 x i16>* %29, align 2, !tbaa !8
+ %31 = add nsw i32 %27, 15
+ %32 = getelementptr inbounds i16, i16* %3, i32 %31
+ %33 = bitcast i16* %32 to <16 x i16>*
+ %34 = load <16 x i16>, <16 x i16>* %33, align 2, !tbaa !8
+ %35 = shufflevector <16 x i16> %30, <16 x i16> %34, <16 x i32> <i32 0, i32 2, i32 4, i32 6, i32 8, i32 10, i32 12, i32 14, i32 17, i32 19, i32 21, i32 23, i32 25, i32 27, i32 29, i32 31>
+ %36 = add nsw i32 %27, 1
+ %37 = getelementptr inbounds i16, i16* %3, i32 %36
+ %38 = bitcast i16* %37 to <16 x i16>*
+ %39 = load <16 x i16>, <16 x i16>* %38, align 2, !tbaa !8
+ %40 = add nsw i32 %27, 16
+ %41 = getelementptr inbounds i16, i16* %3, i32 %40
+ %42 = bitcast i16* %41 to <16 x i16>*
+ %43 = load <16 x i16>, <16 x i16>* %42, align 2, !tbaa !8
+ %44 = shufflevector <16 x i16> %39, <16 x i16> %43, <16 x i32> <i32 0, i32 2, i32 4, i32 6, i32 8, i32 10, i32 12, i32 14, i32 17, i32 19, i32 21, i32 23, i32 25, i32 27, i32 29, i32 31>
+ %45 = shufflevector <16 x i16> %35, <16 x i16> %44, <32 x i32> <i32 0, i32 16, i32 1, i32 17, i32 2, i32 18, i32 3, i32 19, i32 4, i32 20, i32 5, i32 21, i32 6, i32 22, i32 7, i32 23, i32 8, i32 24, i32 9, i32 25, i32 10, i32 26, i32 11, i32 27, i32 12, i32 28, i32 13, i32 29, i32 14, i32 30, i32 15, i32 31>
+ %46 = bitcast <32 x i16> %26 to <16 x i32>
+ %47 = bitcast <32 x i16> %45 to <16 x i32>
+ %48 = tail call <16 x i32> @llvm.hexagon.V6.vdmpyhvsat(<16 x i32> %46, <16 x i32> %47)
+ %49 = bitcast i32* %.phi to <16 x i32>*
+ store <16 x i32> %48, <16 x i32>* %49, align 4, !tbaa !10
+ %50 = add nuw nsw i32 %testOne.s0.x.x, 1
+ %51 = icmp eq i32 %50, %0
+ %.inc = getelementptr i32, i32* %.phi, i32 16
+ br i1 %51, label %"end for testOne.s0.x.x", label %"for testOne.s0.x.x"
+
+"end for testOne.s0.x.x": ; preds = %"for testOne.s0.x.x", %entry
+ %52 = add nsw i32 %testOne.extent.0, 15
+ %53 = ashr i32 %52, 4
+ %54 = icmp sgt i32 %53, %0
+ br i1 %54, label %"for testOne.s0.x.x44.preheader", label %destructor_block
+
+"for testOne.s0.x.x44.preheader": ; preds = %"end for testOne.s0.x.x"
+ %55 = add nsw i32 %testOne.min.0, %testOne.extent.0
+ %56 = shl nsw i32 %55, 1
+ %57 = sub nsw i32 %56, %inputOne.min.0
+ %58 = add nsw i32 %57, -32
+ %59 = bitcast i8* %inputOne.host to i16*
+ %60 = getelementptr inbounds i16, i16* %59, i32 %58
+ %61 = bitcast i16* %60 to <16 x i16>*
+ %62 = load <16 x i16>, <16 x i16>* %61, align 2
+ %63 = add nsw i32 %57, -17
+ %64 = getelementptr inbounds i16, i16* %59, i32 %63
+ %65 = bitcast i16* %64 to <16 x i16>*
+ %66 = load <16 x i16>, <16 x i16>* %65, align 2
+ %67 = add nsw i32 %57, -31
+ %68 = getelementptr inbounds i16, i16* %59, i32 %67
+ %69 = bitcast i16* %68 to <16 x i16>*
+ %70 = load <16 x i16>, <16 x i16>* %69, align 2
+ %71 = add nsw i32 %57, -16
+ %72 = getelementptr inbounds i16, i16* %59, i32 %71
+ %73 = bitcast i16* %72 to <16 x i16>*
+ %74 = load <16 x i16>, <16 x i16>* %73, align 2
+ %75 = shufflevector <16 x i16> %70, <16 x i16> %74, <16 x i32> <i32 0, i32 2, i32 4, i32 6, i32 8, i32 10, i32 12, i32 14, i32 17, i32 19, i32 21, i32 23, i32 25, i32 27, i32 29, i32 31>
+ %76 = sub nsw i32 %56, %inputTwo.min.0
+ %77 = add nsw i32 %76, -32
+ %78 = bitcast i8* %inputTwo.host to i16*
+ %79 = getelementptr inbounds i16, i16* %78, i32 %77
+ %80 = bitcast i16* %79 to <16 x i16>*
+ %81 = load <16 x i16>, <16 x i16>* %80, align 2
+ %82 = add nsw i32 %76, -17
+ %83 = getelementptr inbounds i16, i16* %78, i32 %82
+ %84 = bitcast i16* %83 to <16 x i16>*
+ %85 = load <16 x i16>, <16 x i16>* %84, align 2
+ %86 = shufflevector <16 x i16> %81, <16 x i16> %85, <16 x i32> <i32 0, i32 2, i32 4, i32 6, i32 8, i32 10, i32 12, i32 14, i32 17, i32 19, i32 21, i32 23, i32 25, i32 27, i32 29, i32 31>
+ %87 = add nsw i32 %76, -31
+ %88 = getelementptr inbounds i16, i16* %78, i32 %87
+ %89 = bitcast i16* %88 to <16 x i16>*
+ %90 = load <16 x i16>, <16 x i16>* %89, align 2
+ %91 = add nsw i32 %76, -16
+ %92 = getelementptr inbounds i16, i16* %78, i32 %91
+ %93 = bitcast i16* %92 to <16 x i16>*
+ %94 = load <16 x i16>, <16 x i16>* %93, align 2
+ %95 = shufflevector <16 x i16> %90, <16 x i16> %94, <16 x i32> <i32 0, i32 2, i32 4, i32 6, i32 8, i32 10, i32 12, i32 14, i32 17, i32 19, i32 21, i32 23, i32 25, i32 27, i32 29, i32 31>
+ %96 = shufflevector <16 x i16> %86, <16 x i16> %95, <32 x i32> <i32 0, i32 16, i32 1, i32 17, i32 2, i32 18, i32 3, i32 19, i32 4, i32 20, i32 5, i32 21, i32 6, i32 22, i32 7, i32 23, i32 8, i32 24, i32 9, i32 25, i32 10, i32 26, i32 11, i32 27, i32 12, i32 28, i32 13, i32 29, i32 14, i32 30, i32 15, i32 31>
+ %97 = bitcast <32 x i16> %96 to <16 x i32>
+ %98 = add nsw i32 %testOne.extent.0, -16
+ %99 = bitcast i8* %testOne.host to i32*
+ %100 = getelementptr inbounds i32, i32* %99, i32 %98
+ %101 = shufflevector <16 x i16> %62, <16 x i16> %66, <16 x i32> <i32 0, i32 2, i32 4, i32 6, i32 8, i32 10, i32 12, i32 14, i32 17, i32 19, i32 21, i32 23, i32 25, i32 27, i32 29, i32 31>
+ %102 = shufflevector <16 x i16> %101, <16 x i16> %75, <32 x i32> <i32 0, i32 16, i32 1, i32 17, i32 2, i32 18, i32 3, i32 19, i32 4, i32 20, i32 5, i32 21, i32 6, i32 22, i32 7, i32 23, i32 8, i32 24, i32 9, i32 25, i32 10, i32 26, i32 11, i32 27, i32 12, i32 28, i32 13, i32 29, i32 14, i32 30, i32 15, i32 31>
+ %103 = bitcast <32 x i16> %102 to <16 x i32>
+ %104 = tail call <16 x i32> @llvm.hexagon.V6.vdmpyhvsat(<16 x i32> %103, <16 x i32> %97)
+ %105 = bitcast i32* %100 to <16 x i32>*
+ store <16 x i32> %104, <16 x i32>* %105, align 4, !tbaa !10
+ br label %destructor_block
+
+destructor_block: ; preds = %"for testOne.s0.x.x44.preheader", %"end for testOne.s0.x.x"
+ ret i32 0
+}
+
+; Function Attrs: nounwind readnone
+declare <16 x i32> @llvm.hexagon.V6.vdmpyhvsat(<16 x i32>, <16 x i32>) #1
+
+attributes #0 = { norecurse nounwind "target-cpu"="hexagonv60" "target-features"="+hvx,-hvx-double" }
+attributes #1 = { nounwind readnone "target-cpu"="hexagonv60" "target-features"="+hvx,-hvx-double" }
+
+!5 = !{!6, !6, i64 0}
+!6 = !{!"inputOne", !7}
+!7 = !{!"Halide buffer"}
+!8 = !{!9, !9, i64 0}
+!9 = !{!"inputTwo", !7}
+!10 = !{!11, !11, i64 0}
+!11 = !{!"testOne", !7}
diff --git a/test/CodeGen/Hexagon/vect/vect-vsplatb.ll b/test/CodeGen/Hexagon/vect/vect-vsplatb.ll
index 6996dd144eba..097e2ccd600a 100644
--- a/test/CodeGen/Hexagon/vect/vect-vsplatb.ll
+++ b/test/CodeGen/Hexagon/vect/vect-vsplatb.ll
@@ -1,4 +1,4 @@
-; RUN: llc -march=hexagon < %s | FileCheck %s
+; RUN: llc -march=hexagon -disable-hcp < %s | FileCheck %s
; Make sure we build the constant vector <7, 7, 7, 7> with a vsplatb.
; CHECK: vsplatb
@B = common global [400 x i8] zeroinitializer, align 8
diff --git a/test/CodeGen/Hexagon/vect/vect-vsplath.ll b/test/CodeGen/Hexagon/vect/vect-vsplath.ll
index f5207109773e..db90bf42be2a 100644
--- a/test/CodeGen/Hexagon/vect/vect-vsplath.ll
+++ b/test/CodeGen/Hexagon/vect/vect-vsplath.ll
@@ -1,4 +1,4 @@
-; RUN: llc -march=hexagon < %s | FileCheck %s
+; RUN: llc -march=hexagon -disable-hcp < %s | FileCheck %s
; Make sure we build the constant vector <7, 7, 7, 7> with a vsplath.
; CHECK: vsplath
@B = common global [400 x i16] zeroinitializer, align 8
diff --git a/test/CodeGen/Hexagon/vector-ext-load.ll b/test/CodeGen/Hexagon/vector-ext-load.ll
new file mode 100644
index 000000000000..536dad165ef8
--- /dev/null
+++ b/test/CodeGen/Hexagon/vector-ext-load.ll
@@ -0,0 +1,10 @@
+; A copy of 2012-06-08-APIntCrash.ll with arch explicitly set to hexagon.
+
+; RUN: llc -march=hexagon < %s
+
+define void @test1(<8 x i32>* %ptr) {
+ %1 = load <8 x i32>, <8 x i32>* %ptr, align 32
+ %2 = and <8 x i32> %1, <i32 0, i32 0, i32 0, i32 -1, i32 0, i32 0, i32 0, i32 -1>
+ store <8 x i32> %2, <8 x i32>* %ptr, align 16
+ ret void
+}
diff --git a/test/CodeGen/Hexagon/vmpa-halide-test.ll b/test/CodeGen/Hexagon/vmpa-halide-test.ll
new file mode 100644
index 000000000000..9c359900ba42
--- /dev/null
+++ b/test/CodeGen/Hexagon/vmpa-halide-test.ll
@@ -0,0 +1,145 @@
+; RUN: llc -march=hexagon < %s
+; Thie tests checks a compiler assert. So the test just needs to compile
+; for it to pass
+
+target triple = "hexagon-unknown--elf"
+
+%struct.buffer_t = type { i64, i8*, [4 x i32], [4 x i32], [4 x i32], i32, i8, i8, [6 x i8] }
+
+; Function Attrs: norecurse nounwind
+define i32 @__testOne(%struct.buffer_t* noalias nocapture readonly %inputOne.buffer, %struct.buffer_t* noalias nocapture readonly %inputTwo.buffer, %struct.buffer_t* noalias nocapture readonly %testOne.buffer) #0 {
+entry:
+ %buf_host = getelementptr inbounds %struct.buffer_t, %struct.buffer_t* %inputOne.buffer, i32 0, i32 1
+ %inputOne.host = load i8*, i8** %buf_host, align 4
+ %buf_min = getelementptr inbounds %struct.buffer_t, %struct.buffer_t* %inputOne.buffer, i32 0, i32 4, i32 0
+ %inputOne.min.0 = load i32, i32* %buf_min, align 4
+ %buf_host10 = getelementptr inbounds %struct.buffer_t, %struct.buffer_t* %inputTwo.buffer, i32 0, i32 1
+ %inputTwo.host = load i8*, i8** %buf_host10, align 4
+ %buf_min22 = getelementptr inbounds %struct.buffer_t, %struct.buffer_t* %inputTwo.buffer, i32 0, i32 4, i32 0
+ %inputTwo.min.0 = load i32, i32* %buf_min22, align 4
+ %buf_host27 = getelementptr inbounds %struct.buffer_t, %struct.buffer_t* %testOne.buffer, i32 0, i32 1
+ %testOne.host = load i8*, i8** %buf_host27, align 4
+ %buf_extent31 = getelementptr inbounds %struct.buffer_t, %struct.buffer_t* %testOne.buffer, i32 0, i32 2, i32 0
+ %testOne.extent.0 = load i32, i32* %buf_extent31, align 4
+ %buf_min39 = getelementptr inbounds %struct.buffer_t, %struct.buffer_t* %testOne.buffer, i32 0, i32 4, i32 0
+ %testOne.min.0 = load i32, i32* %buf_min39, align 4
+ %0 = ashr i32 %testOne.extent.0, 6
+ %1 = icmp sgt i32 %0, 0
+ br i1 %1, label %"for testOne.s0.x.x.preheader", label %"end for testOne.s0.x.x"
+
+"for testOne.s0.x.x.preheader": ; preds = %entry
+ %2 = bitcast i8* %testOne.host to i16*
+ br label %"for testOne.s0.x.x"
+
+"for testOne.s0.x.x": ; preds = %"for testOne.s0.x.x", %"for testOne.s0.x.x.preheader"
+ %.phi = phi i16* [ %2, %"for testOne.s0.x.x.preheader" ], [ %.inc, %"for testOne.s0.x.x" ]
+ %testOne.s0.x.x = phi i32 [ 0, %"for testOne.s0.x.x.preheader" ], [ %38, %"for testOne.s0.x.x" ]
+ %3 = shl nsw i32 %testOne.s0.x.x, 6
+ %4 = add nsw i32 %3, %testOne.min.0
+ %5 = shl nsw i32 %4, 1
+ %6 = sub nsw i32 %5, %inputOne.min.0
+ %7 = getelementptr inbounds i8, i8* %inputOne.host, i32 %6
+ %8 = bitcast i8* %7 to <64 x i8>*
+ %9 = load <64 x i8>, <64 x i8>* %8, align 1, !tbaa !5
+ %10 = add nsw i32 %6, 64
+ %11 = getelementptr inbounds i8, i8* %inputOne.host, i32 %10
+ %12 = bitcast i8* %11 to <64 x i8>*
+ %13 = load <64 x i8>, <64 x i8>* %12, align 1, !tbaa !5
+ %14 = shufflevector <64 x i8> %9, <64 x i8> %13, <64 x i32> <i32 0, i32 2, i32 4, i32 6, i32 8, i32 10, i32 12, i32 14, i32 16, i32 18, i32 20, i32 22, i32 24, i32 26, i32 28, i32 30, i32 32, i32 34, i32 36, i32 38, i32 40, i32 42, i32 44, i32 46, i32 48, i32 50, i32 52, i32 54, i32 56, i32 58, i32 60, i32 62, i32 64, i32 66, i32 68, i32 70, i32 72, i32 74, i32 76, i32 78, i32 80, i32 82, i32 84, i32 86, i32 88, i32 90, i32 92, i32 94, i32 96, i32 98, i32 100, i32 102, i32 104, i32 106, i32 108, i32 110, i32 112, i32 114, i32 116, i32 118, i32 120, i32 122, i32 124, i32 126>
+ %15 = shufflevector <64 x i8> %9, <64 x i8> %13, <64 x i32> <i32 1, i32 3, i32 5, i32 7, i32 9, i32 11, i32 13, i32 15, i32 17, i32 19, i32 21, i32 23, i32 25, i32 27, i32 29, i32 31, i32 33, i32 35, i32 37, i32 39, i32 41, i32 43, i32 45, i32 47, i32 49, i32 51, i32 53, i32 55, i32 57, i32 59, i32 61, i32 63, i32 65, i32 67, i32 69, i32 71, i32 73, i32 75, i32 77, i32 79, i32 81, i32 83, i32 85, i32 87, i32 89, i32 91, i32 93, i32 95, i32 97, i32 99, i32 101, i32 103, i32 105, i32 107, i32 109, i32 111, i32 113, i32 115, i32 117, i32 119, i32 121, i32 123, i32 125, i32 127>
+ %16 = shufflevector <64 x i8> %14, <64 x i8> %15, <128 x i32> <i32 0, i32 64, i32 1, i32 65, i32 2, i32 66, i32 3, i32 67, i32 4, i32 68, i32 5, i32 69, i32 6, i32 70, i32 7, i32 71, i32 8, i32 72, i32 9, i32 73, i32 10, i32 74, i32 11, i32 75, i32 12, i32 76, i32 13, i32 77, i32 14, i32 78, i32 15, i32 79, i32 16, i32 80, i32 17, i32 81, i32 18, i32 82, i32 19, i32 83, i32 20, i32 84, i32 21, i32 85, i32 22, i32 86, i32 23, i32 87, i32 24, i32 88, i32 25, i32 89, i32 26, i32 90, i32 27, i32 91, i32 28, i32 92, i32 29, i32 93, i32 30, i32 94, i32 31, i32 95, i32 32, i32 96, i32 33, i32 97, i32 34, i32 98, i32 35, i32 99, i32 36, i32 100, i32 37, i32 101, i32 38, i32 102, i32 39, i32 103, i32 40, i32 104, i32 41, i32 105, i32 42, i32 106, i32 43, i32 107, i32 44, i32 108, i32 45, i32 109, i32 46, i32 110, i32 47, i32 111, i32 48, i32 112, i32 49, i32 113, i32 50, i32 114, i32 51, i32 115, i32 52, i32 116, i32 53, i32 117, i32 54, i32 118, i32 55, i32 119, i32 56, i32 120, i32 57, i32 121, i32 58, i32 122, i32 59, i32 123, i32 60, i32 124, i32 61, i32 125, i32 62, i32 126, i32 63, i32 127>
+ %17 = sub nsw i32 %5, %inputTwo.min.0
+ %18 = getelementptr inbounds i8, i8* %inputTwo.host, i32 %17
+ %19 = bitcast i8* %18 to <64 x i8>*
+ %20 = load <64 x i8>, <64 x i8>* %19, align 1, !tbaa !8
+ %21 = add nsw i32 %17, 64
+ %22 = getelementptr inbounds i8, i8* %inputTwo.host, i32 %21
+ %23 = bitcast i8* %22 to <64 x i8>*
+ %24 = load <64 x i8>, <64 x i8>* %23, align 1, !tbaa !8
+ %25 = shufflevector <64 x i8> %20, <64 x i8> %24, <64 x i32> <i32 0, i32 2, i32 4, i32 6, i32 8, i32 10, i32 12, i32 14, i32 16, i32 18, i32 20, i32 22, i32 24, i32 26, i32 28, i32 30, i32 32, i32 34, i32 36, i32 38, i32 40, i32 42, i32 44, i32 46, i32 48, i32 50, i32 52, i32 54, i32 56, i32 58, i32 60, i32 62, i32 64, i32 66, i32 68, i32 70, i32 72, i32 74, i32 76, i32 78, i32 80, i32 82, i32 84, i32 86, i32 88, i32 90, i32 92, i32 94, i32 96, i32 98, i32 100, i32 102, i32 104, i32 106, i32 108, i32 110, i32 112, i32 114, i32 116, i32 118, i32 120, i32 122, i32 124, i32 126>
+ %26 = shufflevector <64 x i8> %20, <64 x i8> %24, <64 x i32> <i32 1, i32 3, i32 5, i32 7, i32 9, i32 11, i32 13, i32 15, i32 17, i32 19, i32 21, i32 23, i32 25, i32 27, i32 29, i32 31, i32 33, i32 35, i32 37, i32 39, i32 41, i32 43, i32 45, i32 47, i32 49, i32 51, i32 53, i32 55, i32 57, i32 59, i32 61, i32 63, i32 65, i32 67, i32 69, i32 71, i32 73, i32 75, i32 77, i32 79, i32 81, i32 83, i32 85, i32 87, i32 89, i32 91, i32 93, i32 95, i32 97, i32 99, i32 101, i32 103, i32 105, i32 107, i32 109, i32 111, i32 113, i32 115, i32 117, i32 119, i32 121, i32 123, i32 125, i32 127>
+ %27 = shufflevector <64 x i8> %25, <64 x i8> %26, <128 x i32> <i32 0, i32 64, i32 1, i32 65, i32 2, i32 66, i32 3, i32 67, i32 4, i32 68, i32 5, i32 69, i32 6, i32 70, i32 7, i32 71, i32 8, i32 72, i32 9, i32 73, i32 10, i32 74, i32 11, i32 75, i32 12, i32 76, i32 13, i32 77, i32 14, i32 78, i32 15, i32 79, i32 16, i32 80, i32 17, i32 81, i32 18, i32 82, i32 19, i32 83, i32 20, i32 84, i32 21, i32 85, i32 22, i32 86, i32 23, i32 87, i32 24, i32 88, i32 25, i32 89, i32 26, i32 90, i32 27, i32 91, i32 28, i32 92, i32 29, i32 93, i32 30, i32 94, i32 31, i32 95, i32 32, i32 96, i32 33, i32 97, i32 34, i32 98, i32 35, i32 99, i32 36, i32 100, i32 37, i32 101, i32 38, i32 102, i32 39, i32 103, i32 40, i32 104, i32 41, i32 105, i32 42, i32 106, i32 43, i32 107, i32 44, i32 108, i32 45, i32 109, i32 46, i32 110, i32 47, i32 111, i32 48, i32 112, i32 49, i32 113, i32 50, i32 114, i32 51, i32 115, i32 52, i32 116, i32 53, i32 117, i32 54, i32 118, i32 55, i32 119, i32 56, i32 120, i32 57, i32 121, i32 58, i32 122, i32 59, i32 123, i32 60, i32 124, i32 61, i32 125, i32 62, i32 126, i32 63, i32 127>
+ %28 = bitcast <128 x i8> %16 to <32 x i32>
+ %29 = bitcast <128 x i8> %27 to <32 x i32>
+ %30 = tail call <32 x i32> @llvm.hexagon.V6.vmpabuuv(<32 x i32> %28, <32 x i32> %29)
+ %31 = bitcast <32 x i32> %30 to <64 x i16>
+ %32 = shufflevector <64 x i16> %31, <64 x i16> undef, <32 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15, i32 16, i32 17, i32 18, i32 19, i32 20, i32 21, i32 22, i32 23, i32 24, i32 25, i32 26, i32 27, i32 28, i32 29, i32 30, i32 31>
+ %33 = bitcast i16* %.phi to <32 x i16>*
+ store <32 x i16> %32, <32 x i16>* %33, align 2, !tbaa !10
+ %34 = shufflevector <64 x i16> %31, <64 x i16> undef, <32 x i32> <i32 32, i32 33, i32 34, i32 35, i32 36, i32 37, i32 38, i32 39, i32 40, i32 41, i32 42, i32 43, i32 44, i32 45, i32 46, i32 47, i32 48, i32 49, i32 50, i32 51, i32 52, i32 53, i32 54, i32 55, i32 56, i32 57, i32 58, i32 59, i32 60, i32 61, i32 62, i32 63>
+ %35 = or i32 %3, 32
+ %36 = getelementptr inbounds i16, i16* %2, i32 %35
+ %37 = bitcast i16* %36 to <32 x i16>*
+ store <32 x i16> %34, <32 x i16>* %37, align 2, !tbaa !10
+ %38 = add nuw nsw i32 %testOne.s0.x.x, 1
+ %39 = icmp eq i32 %38, %0
+ %.inc = getelementptr i16, i16* %.phi, i32 64
+ br i1 %39, label %"end for testOne.s0.x.x", label %"for testOne.s0.x.x"
+
+"end for testOne.s0.x.x": ; preds = %"for testOne.s0.x.x", %entry
+ %40 = add nsw i32 %testOne.extent.0, 63
+ %41 = ashr i32 %40, 6
+ %42 = icmp sgt i32 %41, %0
+ br i1 %42, label %"for testOne.s0.x.x44.preheader", label %destructor_block
+
+"for testOne.s0.x.x44.preheader": ; preds = %"end for testOne.s0.x.x"
+ %43 = add nsw i32 %testOne.min.0, %testOne.extent.0
+ %44 = shl nsw i32 %43, 1
+ %45 = sub nsw i32 %44, %inputOne.min.0
+ %46 = add nsw i32 %45, -128
+ %47 = getelementptr inbounds i8, i8* %inputOne.host, i32 %46
+ %48 = bitcast i8* %47 to <64 x i8>*
+ %49 = load <64 x i8>, <64 x i8>* %48, align 1
+ %50 = add nsw i32 %45, -64
+ %51 = getelementptr inbounds i8, i8* %inputOne.host, i32 %50
+ %52 = bitcast i8* %51 to <64 x i8>*
+ %53 = load <64 x i8>, <64 x i8>* %52, align 1
+ %54 = shufflevector <64 x i8> %49, <64 x i8> %53, <64 x i32> <i32 0, i32 2, i32 4, i32 6, i32 8, i32 10, i32 12, i32 14, i32 16, i32 18, i32 20, i32 22, i32 24, i32 26, i32 28, i32 30, i32 32, i32 34, i32 36, i32 38, i32 40, i32 42, i32 44, i32 46, i32 48, i32 50, i32 52, i32 54, i32 56, i32 58, i32 60, i32 62, i32 64, i32 66, i32 68, i32 70, i32 72, i32 74, i32 76, i32 78, i32 80, i32 82, i32 84, i32 86, i32 88, i32 90, i32 92, i32 94, i32 96, i32 98, i32 100, i32 102, i32 104, i32 106, i32 108, i32 110, i32 112, i32 114, i32 116, i32 118, i32 120, i32 122, i32 124, i32 126>
+ %55 = shufflevector <64 x i8> %49, <64 x i8> %53, <64 x i32> <i32 1, i32 3, i32 5, i32 7, i32 9, i32 11, i32 13, i32 15, i32 17, i32 19, i32 21, i32 23, i32 25, i32 27, i32 29, i32 31, i32 33, i32 35, i32 37, i32 39, i32 41, i32 43, i32 45, i32 47, i32 49, i32 51, i32 53, i32 55, i32 57, i32 59, i32 61, i32 63, i32 65, i32 67, i32 69, i32 71, i32 73, i32 75, i32 77, i32 79, i32 81, i32 83, i32 85, i32 87, i32 89, i32 91, i32 93, i32 95, i32 97, i32 99, i32 101, i32 103, i32 105, i32 107, i32 109, i32 111, i32 113, i32 115, i32 117, i32 119, i32 121, i32 123, i32 125, i32 127>
+ %56 = shufflevector <64 x i8> %54, <64 x i8> %55, <128 x i32> <i32 0, i32 64, i32 1, i32 65, i32 2, i32 66, i32 3, i32 67, i32 4, i32 68, i32 5, i32 69, i32 6, i32 70, i32 7, i32 71, i32 8, i32 72, i32 9, i32 73, i32 10, i32 74, i32 11, i32 75, i32 12, i32 76, i32 13, i32 77, i32 14, i32 78, i32 15, i32 79, i32 16, i32 80, i32 17, i32 81, i32 18, i32 82, i32 19, i32 83, i32 20, i32 84, i32 21, i32 85, i32 22, i32 86, i32 23, i32 87, i32 24, i32 88, i32 25, i32 89, i32 26, i32 90, i32 27, i32 91, i32 28, i32 92, i32 29, i32 93, i32 30, i32 94, i32 31, i32 95, i32 32, i32 96, i32 33, i32 97, i32 34, i32 98, i32 35, i32 99, i32 36, i32 100, i32 37, i32 101, i32 38, i32 102, i32 39, i32 103, i32 40, i32 104, i32 41, i32 105, i32 42, i32 106, i32 43, i32 107, i32 44, i32 108, i32 45, i32 109, i32 46, i32 110, i32 47, i32 111, i32 48, i32 112, i32 49, i32 113, i32 50, i32 114, i32 51, i32 115, i32 52, i32 116, i32 53, i32 117, i32 54, i32 118, i32 55, i32 119, i32 56, i32 120, i32 57, i32 121, i32 58, i32 122, i32 59, i32 123, i32 60, i32 124, i32 61, i32 125, i32 62, i32 126, i32 63, i32 127>
+ %57 = sub nsw i32 %44, %inputTwo.min.0
+ %58 = add nsw i32 %57, -128
+ %59 = getelementptr inbounds i8, i8* %inputTwo.host, i32 %58
+ %60 = bitcast i8* %59 to <64 x i8>*
+ %61 = load <64 x i8>, <64 x i8>* %60, align 1
+ %62 = add nsw i32 %57, -64
+ %63 = getelementptr inbounds i8, i8* %inputTwo.host, i32 %62
+ %64 = bitcast i8* %63 to <64 x i8>*
+ %65 = load <64 x i8>, <64 x i8>* %64, align 1
+ %66 = shufflevector <64 x i8> %61, <64 x i8> %65, <64 x i32> <i32 0, i32 2, i32 4, i32 6, i32 8, i32 10, i32 12, i32 14, i32 16, i32 18, i32 20, i32 22, i32 24, i32 26, i32 28, i32 30, i32 32, i32 34, i32 36, i32 38, i32 40, i32 42, i32 44, i32 46, i32 48, i32 50, i32 52, i32 54, i32 56, i32 58, i32 60, i32 62, i32 64, i32 66, i32 68, i32 70, i32 72, i32 74, i32 76, i32 78, i32 80, i32 82, i32 84, i32 86, i32 88, i32 90, i32 92, i32 94, i32 96, i32 98, i32 100, i32 102, i32 104, i32 106, i32 108, i32 110, i32 112, i32 114, i32 116, i32 118, i32 120, i32 122, i32 124, i32 126>
+ %67 = shufflevector <64 x i8> %61, <64 x i8> %65, <64 x i32> <i32 1, i32 3, i32 5, i32 7, i32 9, i32 11, i32 13, i32 15, i32 17, i32 19, i32 21, i32 23, i32 25, i32 27, i32 29, i32 31, i32 33, i32 35, i32 37, i32 39, i32 41, i32 43, i32 45, i32 47, i32 49, i32 51, i32 53, i32 55, i32 57, i32 59, i32 61, i32 63, i32 65, i32 67, i32 69, i32 71, i32 73, i32 75, i32 77, i32 79, i32 81, i32 83, i32 85, i32 87, i32 89, i32 91, i32 93, i32 95, i32 97, i32 99, i32 101, i32 103, i32 105, i32 107, i32 109, i32 111, i32 113, i32 115, i32 117, i32 119, i32 121, i32 123, i32 125, i32 127>
+ %68 = shufflevector <64 x i8> %66, <64 x i8> %67, <128 x i32> <i32 0, i32 64, i32 1, i32 65, i32 2, i32 66, i32 3, i32 67, i32 4, i32 68, i32 5, i32 69, i32 6, i32 70, i32 7, i32 71, i32 8, i32 72, i32 9, i32 73, i32 10, i32 74, i32 11, i32 75, i32 12, i32 76, i32 13, i32 77, i32 14, i32 78, i32 15, i32 79, i32 16, i32 80, i32 17, i32 81, i32 18, i32 82, i32 19, i32 83, i32 20, i32 84, i32 21, i32 85, i32 22, i32 86, i32 23, i32 87, i32 24, i32 88, i32 25, i32 89, i32 26, i32 90, i32 27, i32 91, i32 28, i32 92, i32 29, i32 93, i32 30, i32 94, i32 31, i32 95, i32 32, i32 96, i32 33, i32 97, i32 34, i32 98, i32 35, i32 99, i32 36, i32 100, i32 37, i32 101, i32 38, i32 102, i32 39, i32 103, i32 40, i32 104, i32 41, i32 105, i32 42, i32 106, i32 43, i32 107, i32 44, i32 108, i32 45, i32 109, i32 46, i32 110, i32 47, i32 111, i32 48, i32 112, i32 49, i32 113, i32 50, i32 114, i32 51, i32 115, i32 52, i32 116, i32 53, i32 117, i32 54, i32 118, i32 55, i32 119, i32 56, i32 120, i32 57, i32 121, i32 58, i32 122, i32 59, i32 123, i32 60, i32 124, i32 61, i32 125, i32 62, i32 126, i32 63, i32 127>
+ %69 = bitcast <128 x i8> %56 to <32 x i32>
+ %70 = bitcast <128 x i8> %68 to <32 x i32>
+ %71 = tail call <32 x i32> @llvm.hexagon.V6.vmpabuuv(<32 x i32> %69, <32 x i32> %70)
+ %72 = bitcast <32 x i32> %71 to <64 x i16>
+ %73 = add nsw i32 %testOne.extent.0, -64
+ %74 = bitcast i8* %testOne.host to i16*
+ %75 = getelementptr inbounds i16, i16* %74, i32 %73
+ %76 = bitcast i16* %75 to <32 x i16>*
+ %77 = add nsw i32 %testOne.extent.0, -32
+ %78 = getelementptr inbounds i16, i16* %74, i32 %77
+ %79 = shufflevector <64 x i16> %72, <64 x i16> undef, <32 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15, i32 16, i32 17, i32 18, i32 19, i32 20, i32 21, i32 22, i32 23, i32 24, i32 25, i32 26, i32 27, i32 28, i32 29, i32 30, i32 31>
+ %80 = shufflevector <64 x i16> %72, <64 x i16> undef, <32 x i32> <i32 32, i32 33, i32 34, i32 35, i32 36, i32 37, i32 38, i32 39, i32 40, i32 41, i32 42, i32 43, i32 44, i32 45, i32 46, i32 47, i32 48, i32 49, i32 50, i32 51, i32 52, i32 53, i32 54, i32 55, i32 56, i32 57, i32 58, i32 59, i32 60, i32 61, i32 62, i32 63>
+ %81 = bitcast i16* %78 to <32 x i16>*
+ store <32 x i16> %79, <32 x i16>* %76, align 2, !tbaa !10
+ store <32 x i16> %80, <32 x i16>* %81, align 2, !tbaa !10
+ br label %destructor_block
+
+destructor_block: ; preds = %"for testOne.s0.x.x44.preheader", %"end for testOne.s0.x.x"
+ ret i32 0
+}
+
+; Function Attrs: nounwind readnone
+declare <32 x i32> @llvm.hexagon.V6.vmpabuuv(<32 x i32>, <32 x i32>) #1
+
+attributes #0 = { norecurse nounwind "target-cpu"="hexagonv60" "target-features"="+hvx,-hvx-double" }
+attributes #1 = { nounwind readnone "target-cpu"="hexagonv60" "target-features"="+hvx,-hvx-double" }
+
+!5 = !{!6, !6, i64 0}
+!6 = !{!"inputOne", !7}
+!7 = !{!"Halide buffer"}
+!8 = !{!9, !9, i64 0}
+!9 = !{!"inputTwo", !7}
+!10 = !{!11, !11, i64 0}
+!11 = !{!"testOne", !7}
diff --git a/test/CodeGen/Hexagon/vpack_eo.ll b/test/CodeGen/Hexagon/vpack_eo.ll
new file mode 100644
index 000000000000..7238ca84a42e
--- /dev/null
+++ b/test/CodeGen/Hexagon/vpack_eo.ll
@@ -0,0 +1,73 @@
+; RUN: llc -march=hexagon < %s | FileCheck %s
+target triple = "hexagon-unknown--elf"
+
+; CHECK-DAG: vpacke
+; CHECK-DAG: vpacko
+
+%struct.buffer_t = type { i64, i8*, [4 x i32], [4 x i32], [4 x i32], i32, i8, i8, [6 x i8] }
+
+; Function Attrs: norecurse nounwind
+define i32 @__Strided_LoadTest(%struct.buffer_t* noalias nocapture readonly %InputOne.buffer, %struct.buffer_t* noalias nocapture readonly %InputTwo.buffer, %struct.buffer_t* noalias nocapture readonly %Strided_LoadTest.buffer) #0 {
+entry:
+ %buf_host = getelementptr inbounds %struct.buffer_t, %struct.buffer_t* %InputOne.buffer, i32 0, i32 1
+ %0 = bitcast i8** %buf_host to i16**
+ %InputOne.host45 = load i16*, i16** %0, align 4
+ %buf_host10 = getelementptr inbounds %struct.buffer_t, %struct.buffer_t* %InputTwo.buffer, i32 0, i32 1
+ %1 = bitcast i8** %buf_host10 to i16**
+ %InputTwo.host46 = load i16*, i16** %1, align 4
+ %buf_host27 = getelementptr inbounds %struct.buffer_t, %struct.buffer_t* %Strided_LoadTest.buffer, i32 0, i32 1
+ %2 = bitcast i8** %buf_host27 to i16**
+ %Strided_LoadTest.host44 = load i16*, i16** %2, align 4
+ %3 = bitcast i16* %InputOne.host45 to <32 x i16>*
+ %4 = load <32 x i16>, <32 x i16>* %3, align 2, !tbaa !4
+ %5 = getelementptr inbounds i16, i16* %InputOne.host45, i32 32
+ %6 = bitcast i16* %5 to <32 x i16>*
+ %7 = load <32 x i16>, <32 x i16>* %6, align 2, !tbaa !4
+ %8 = shufflevector <32 x i16> %4, <32 x i16> %7, <32 x i32> <i32 1, i32 3, i32 5, i32 7, i32 9, i32 11, i32 13, i32 15, i32 17, i32 19, i32 21, i32 23, i32 25, i32 27, i32 29, i32 31, i32 33, i32 35, i32 37, i32 39, i32 41, i32 43, i32 45, i32 47, i32 49, i32 51, i32 53, i32 55, i32 57, i32 59, i32 61, i32 63>
+ %9 = bitcast i16* %InputTwo.host46 to <32 x i16>*
+ %10 = load <32 x i16>, <32 x i16>* %9, align 2, !tbaa !7
+ %11 = getelementptr inbounds i16, i16* %InputTwo.host46, i32 32
+ %12 = bitcast i16* %11 to <32 x i16>*
+ %13 = load <32 x i16>, <32 x i16>* %12, align 2, !tbaa !7
+ %14 = shufflevector <32 x i16> %10, <32 x i16> %13, <32 x i32> <i32 0, i32 2, i32 4, i32 6, i32 8, i32 10, i32 12, i32 14, i32 16, i32 18, i32 20, i32 22, i32 24, i32 26, i32 28, i32 30, i32 32, i32 34, i32 36, i32 38, i32 40, i32 42, i32 44, i32 46, i32 48, i32 50, i32 52, i32 54, i32 56, i32 58, i32 60, i32 62>
+ %15 = bitcast <32 x i16> %8 to <16 x i32>
+ %16 = bitcast <32 x i16> %14 to <16 x i32>
+ %17 = tail call <16 x i32> @llvm.hexagon.V6.vaddh(<16 x i32> %15, <16 x i32> %16)
+ %18 = bitcast i16* %Strided_LoadTest.host44 to <16 x i32>*
+ store <16 x i32> %17, <16 x i32>* %18, align 2, !tbaa !9
+ %.inc = getelementptr i16, i16* %InputOne.host45, i32 64
+ %.inc49 = getelementptr i16, i16* %InputTwo.host46, i32 64
+ %.inc52 = getelementptr i16, i16* %Strided_LoadTest.host44, i32 32
+ %19 = bitcast i16* %.inc to <32 x i16>*
+ %20 = load <32 x i16>, <32 x i16>* %19, align 2, !tbaa !4
+ %21 = getelementptr inbounds i16, i16* %InputOne.host45, i32 96
+ %22 = bitcast i16* %21 to <32 x i16>*
+ %23 = load <32 x i16>, <32 x i16>* %22, align 2, !tbaa !4
+ %24 = shufflevector <32 x i16> %20, <32 x i16> %23, <32 x i32> <i32 1, i32 3, i32 5, i32 7, i32 9, i32 11, i32 13, i32 15, i32 17, i32 19, i32 21, i32 23, i32 25, i32 27, i32 29, i32 31, i32 33, i32 35, i32 37, i32 39, i32 41, i32 43, i32 45, i32 47, i32 49, i32 51, i32 53, i32 55, i32 57, i32 59, i32 61, i32 63>
+ %25 = bitcast i16* %.inc49 to <32 x i16>*
+ %26 = load <32 x i16>, <32 x i16>* %25, align 2, !tbaa !7
+ %27 = getelementptr inbounds i16, i16* %InputTwo.host46, i32 96
+ %28 = bitcast i16* %27 to <32 x i16>*
+ %29 = load <32 x i16>, <32 x i16>* %28, align 2, !tbaa !7
+ %30 = shufflevector <32 x i16> %26, <32 x i16> %29, <32 x i32> <i32 0, i32 2, i32 4, i32 6, i32 8, i32 10, i32 12, i32 14, i32 16, i32 18, i32 20, i32 22, i32 24, i32 26, i32 28, i32 30, i32 32, i32 34, i32 36, i32 38, i32 40, i32 42, i32 44, i32 46, i32 48, i32 50, i32 52, i32 54, i32 56, i32 58, i32 60, i32 62>
+ %31 = bitcast <32 x i16> %24 to <16 x i32>
+ %32 = bitcast <32 x i16> %30 to <16 x i32>
+ %33 = tail call <16 x i32> @llvm.hexagon.V6.vaddh(<16 x i32> %31, <16 x i32> %32)
+ %34 = bitcast i16* %.inc52 to <16 x i32>*
+ store <16 x i32> %33, <16 x i32>* %34, align 2, !tbaa !9
+ ret i32 0
+}
+
+; Function Attrs: nounwind readnone
+declare <16 x i32> @llvm.hexagon.V6.vaddh(<16 x i32>, <16 x i32>) #1
+
+attributes #0 = { norecurse nounwind "target-cpu"="hexagonv60" "target-features"="+hvx,-hvx-double" }
+attributes #1 = { nounwind readnone "target-cpu"="hexagonv60" "target-features"="+hvx,-hvx-double" }
+
+!4 = !{!5, !5, i64 0}
+!5 = !{!"InputOne", !6}
+!6 = !{!"Halide buffer"}
+!7 = !{!8, !8, i64 0}
+!8 = !{!"InputTwo", !6}
+!9 = !{!10, !10, i64 0}
+!10 = !{!"Strided_LoadTest", !6}