diff options
Diffstat (limited to 'test/CodeGen/AMDGPU')
| -rw-r--r-- | test/CodeGen/AMDGPU/ctlz.ll | 269 | ||||
| -rw-r--r-- | test/CodeGen/AMDGPU/ctlz_zero_undef.ll | 197 | ||||
| -rw-r--r-- | test/CodeGen/AMDGPU/flat-scratch-reg.ll | 8 | ||||
| -rw-r--r-- | test/CodeGen/AMDGPU/fmin_legacy.ll | 4 | ||||
| -rw-r--r-- | test/CodeGen/AMDGPU/fsub.ll | 15 | ||||
| -rw-r--r-- | test/CodeGen/AMDGPU/hsa-globals.ll | 16 | ||||
| -rw-r--r-- | test/CodeGen/AMDGPU/hsa-note-no-func.ll | 6 | ||||
| -rw-r--r-- | test/CodeGen/AMDGPU/hsa.ll | 4 | ||||
| -rw-r--r-- | test/CodeGen/AMDGPU/inline-asm.ll | 11 | ||||
| -rw-r--r-- | test/CodeGen/AMDGPU/llvm.amdgcn.dispatch.ptr.ll | 3 | ||||
| -rw-r--r-- | test/CodeGen/AMDGPU/llvm.round.f64.ll | 2 | ||||
| -rw-r--r-- | test/CodeGen/AMDGPU/ret.ll | 245 | ||||
| -rw-r--r-- | test/CodeGen/AMDGPU/si-scheduler.ll | 55 | ||||
| -rw-r--r-- | test/CodeGen/AMDGPU/sint_to_fp.i64.ll | 62 | ||||
| -rw-r--r-- | test/CodeGen/AMDGPU/sint_to_fp.ll | 91 | ||||
| -rw-r--r-- | test/CodeGen/AMDGPU/udiv.ll | 89 | ||||
| -rw-r--r-- | test/CodeGen/AMDGPU/uint_to_fp.i64.ll | 57 | ||||
| -rw-r--r-- | test/CodeGen/AMDGPU/uint_to_fp.ll | 123 |
18 files changed, 1168 insertions, 89 deletions
diff --git a/test/CodeGen/AMDGPU/ctlz.ll b/test/CodeGen/AMDGPU/ctlz.ll new file mode 100644 index 000000000000..baedf47eef0d --- /dev/null +++ b/test/CodeGen/AMDGPU/ctlz.ll @@ -0,0 +1,269 @@ +; RUN: llc -march=amdgcn -mcpu=SI -verify-machineinstrs < %s | FileCheck -check-prefix=SI -check-prefix=FUNC %s +; RUN: llc -march=amdgcn -mcpu=tonga -verify-machineinstrs < %s | FileCheck -check-prefix=SI -check-prefix=FUNC %s +; RUN: llc -march=r600 -mcpu=cypress -verify-machineinstrs < %s | FileCheck -check-prefix=EG -check-prefix=FUNC %s + +declare i7 @llvm.ctlz.i7(i7, i1) nounwind readnone +declare i8 @llvm.ctlz.i8(i8, i1) nounwind readnone +declare i16 @llvm.ctlz.i16(i16, i1) nounwind readnone + +declare i32 @llvm.ctlz.i32(i32, i1) nounwind readnone +declare <2 x i32> @llvm.ctlz.v2i32(<2 x i32>, i1) nounwind readnone +declare <4 x i32> @llvm.ctlz.v4i32(<4 x i32>, i1) nounwind readnone + +declare i64 @llvm.ctlz.i64(i64, i1) nounwind readnone +declare <2 x i64> @llvm.ctlz.v2i64(<2 x i64>, i1) nounwind readnone +declare <4 x i64> @llvm.ctlz.v4i64(<4 x i64>, i1) nounwind readnone + +declare i32 @llvm.r600.read.tidig.x() nounwind readnone + +; FUNC-LABEL: {{^}}s_ctlz_i32: +; SI: s_load_dword [[VAL:s[0-9]+]], s{{\[[0-9]+:[0-9]+\]}}, {{0xb|0x2c}} +; SI-DAG: s_flbit_i32_b32 [[CTLZ:s[0-9]+]], [[VAL]] +; SI-DAG: v_cmp_eq_i32_e64 [[CMPZ:s\[[0-9]+:[0-9]+\]]], 0, [[VAL]] +; SI-DAG: v_mov_b32_e32 [[VCTLZ:v[0-9]+]], [[CTLZ]] +; SI: v_cndmask_b32_e64 [[RESULT:v[0-9]+]], [[VCTLZ]], 32, [[CMPZ]] +; SI: buffer_store_dword [[RESULT]] +; SI: s_endpgm + +; EG: FFBH_UINT +; EG: CNDE_INT +define void @s_ctlz_i32(i32 addrspace(1)* noalias %out, i32 %val) nounwind { + %ctlz = call i32 @llvm.ctlz.i32(i32 %val, i1 false) nounwind readnone + store i32 %ctlz, i32 addrspace(1)* %out, align 4 + ret void +} + +; FUNC-LABEL: {{^}}v_ctlz_i32: +; SI: buffer_load_dword [[VAL:v[0-9]+]], +; SI-DAG: v_ffbh_u32_e32 [[CTLZ:v[0-9]+]], [[VAL]] +; SI-DAG: v_cmp_eq_i32_e32 vcc, 0, [[CTLZ]] +; SI: v_cndmask_b32_e64 [[RESULT:v[0-9]+]], [[CTLZ]], 32, vcc +; SI: buffer_store_dword [[RESULT]], +; SI: s_endpgm + +; EG: FFBH_UINT +; EG: CNDE_INT +define void @v_ctlz_i32(i32 addrspace(1)* noalias %out, i32 addrspace(1)* noalias %valptr) nounwind { + %val = load i32, i32 addrspace(1)* %valptr, align 4 + %ctlz = call i32 @llvm.ctlz.i32(i32 %val, i1 false) nounwind readnone + store i32 %ctlz, i32 addrspace(1)* %out, align 4 + ret void +} + +; FUNC-LABEL: {{^}}v_ctlz_v2i32: +; SI: buffer_load_dwordx2 +; SI: v_ffbh_u32_e32 +; SI: v_ffbh_u32_e32 +; SI: buffer_store_dwordx2 +; SI: s_endpgm + +; EG: FFBH_UINT +; EG: CNDE_INT +; EG: FFBH_UINT +; EG: CNDE_INT +define void @v_ctlz_v2i32(<2 x i32> addrspace(1)* noalias %out, <2 x i32> addrspace(1)* noalias %valptr) nounwind { + %val = load <2 x i32>, <2 x i32> addrspace(1)* %valptr, align 8 + %ctlz = call <2 x i32> @llvm.ctlz.v2i32(<2 x i32> %val, i1 false) nounwind readnone + store <2 x i32> %ctlz, <2 x i32> addrspace(1)* %out, align 8 + ret void +} + +; FUNC-LABEL: {{^}}v_ctlz_v4i32: +; SI: buffer_load_dwordx4 +; SI: v_ffbh_u32_e32 +; SI: v_ffbh_u32_e32 +; SI: v_ffbh_u32_e32 +; SI: v_ffbh_u32_e32 +; SI: buffer_store_dwordx4 +; SI: s_endpgm + + +; EG-DAG: FFBH_UINT +; EG-DAG: CNDE_INT + +; EG-DAG: FFBH_UINT +; EG-DAG: CNDE_INT + +; EG-DAG: FFBH_UINT +; EG-DAG: CNDE_INT + +; EG-DAG: FFBH_UINT +; EG-DAG: CNDE_INT +define void @v_ctlz_v4i32(<4 x i32> addrspace(1)* noalias %out, <4 x i32> addrspace(1)* noalias %valptr) nounwind { + %val = load <4 x i32>, <4 x i32> addrspace(1)* %valptr, align 16 + %ctlz = call <4 x i32> @llvm.ctlz.v4i32(<4 x i32> %val, i1 false) nounwind readnone + store <4 x i32> %ctlz, <4 x i32> addrspace(1)* %out, align 16 + ret void +} + +; FUNC-LABEL: {{^}}v_ctlz_i8: +; SI: buffer_load_ubyte [[VAL:v[0-9]+]], +; SI-DAG: v_ffbh_u32_e32 [[FFBH:v[0-9]+]], [[VAL]] +; SI-DAG: v_cmp_eq_i32_e32 vcc, 0, [[CTLZ]] +; SI-DAG: v_cndmask_b32_e64 [[CORRECTED_FFBH:v[0-9]+]], [[FFBH]], 32, vcc +; SI: v_add_i32_e32 [[RESULT:v[0-9]+]], vcc, 0xffffffe8, [[CORRECTED_FFBH]] +; SI: buffer_store_byte [[RESULT]], +define void @v_ctlz_i8(i8 addrspace(1)* noalias %out, i8 addrspace(1)* noalias %valptr) nounwind { + %val = load i8, i8 addrspace(1)* %valptr + %ctlz = call i8 @llvm.ctlz.i8(i8 %val, i1 false) nounwind readnone + store i8 %ctlz, i8 addrspace(1)* %out + ret void +} + +; FUNC-LABEL: {{^}}s_ctlz_i64: +; SI: s_load_dwordx2 s{{\[}}[[LO:[0-9]+]]:[[HI:[0-9]+]]{{\]}}, s{{\[[0-9]+:[0-9]+\]}}, {{0xb|0x2c}} +; SI-DAG: v_cmp_eq_i32_e64 vcc, 0, s[[HI]] +; SI-DAG: s_flbit_i32_b32 [[FFBH_LO:s[0-9]+]], s[[LO]] +; SI-DAG: s_add_i32 [[ADD:s[0-9]+]], [[FFBH_LO]], 32 +; SI-DAG: s_flbit_i32_b32 [[FFBH_HI:s[0-9]+]], s[[HI]] +; SI-DAG: v_mov_b32_e32 [[VFFBH_LO:v[0-9]+]], [[FFBH_LO]] +; SI-DAG: v_mov_b32_e32 [[VFFBH_HI:v[0-9]+]], [[FFBH_HI]] +; SI-DAG: v_cndmask_b32_e32 v[[CTLZ:[0-9]+]], [[VFFBH_HI]], [[VFFBH_LO]] +; SI-DAG: v_mov_b32_e32 v[[CTLZ_HI:[0-9]+]], 0{{$}} +; SI: {{buffer|flat}}_store_dwordx2 v{{\[}}[[CTLZ]]:[[CTLZ_HI]]{{\]}} +define void @s_ctlz_i64(i64 addrspace(1)* noalias %out, i64 %val) nounwind { + %ctlz = call i64 @llvm.ctlz.i64(i64 %val, i1 false) + store i64 %ctlz, i64 addrspace(1)* %out + ret void +} + +; FUNC-LABEL: {{^}}s_ctlz_i64_trunc: +define void @s_ctlz_i64_trunc(i32 addrspace(1)* noalias %out, i64 %val) nounwind { + %ctlz = call i64 @llvm.ctlz.i64(i64 %val, i1 false) + %trunc = trunc i64 %ctlz to i32 + store i32 %trunc, i32 addrspace(1)* %out + ret void +} + +; FUNC-LABEL: {{^}}v_ctlz_i64: +; SI: {{buffer|flat}}_load_dwordx2 v{{\[}}[[LO:[0-9]+]]:[[HI:[0-9]+]]{{\]}} +; SI-DAG: v_cmp_eq_i32_e64 [[CMPHI:s\[[0-9]+:[0-9]+\]]], 0, v[[HI]] +; SI-DAG: v_ffbh_u32_e32 [[FFBH_LO:v[0-9]+]], v[[LO]] +; SI-DAG: v_add_i32_e32 [[ADD:v[0-9]+]], vcc, 32, [[FFBH_LO]] +; SI-DAG: v_ffbh_u32_e32 [[FFBH_HI:v[0-9]+]], v[[HI]] +; SI-DAG: v_cndmask_b32_e64 v[[CTLZ:[0-9]+]], [[FFBH_HI]], [[ADD]], [[CMPHI]] +; SI-DAG: v_or_b32_e32 [[OR:v[0-9]+]], v[[LO]], v[[HI]] +; SI-DAG: v_cmp_eq_i32_e32 vcc, 0, [[OR]] +; SI-DAG: v_cndmask_b32_e64 v[[CLTZ_LO:[0-9]+]], v[[CTLZ:[0-9]+]], 64, vcc +; SI-DAG: v_mov_b32_e32 v[[CTLZ_HI:[0-9]+]], 0{{$}} +; SI: {{buffer|flat}}_store_dwordx2 v{{\[}}[[CLTZ_LO]]:[[CTLZ_HI]]{{\]}} +define void @v_ctlz_i64(i64 addrspace(1)* noalias %out, i64 addrspace(1)* noalias %in) nounwind { + %tid = call i32 @llvm.r600.read.tidig.x() + %in.gep = getelementptr i64, i64 addrspace(1)* %in, i32 %tid + %out.gep = getelementptr i64, i64 addrspace(1)* %out, i32 %tid + %val = load i64, i64 addrspace(1)* %in.gep + %ctlz = call i64 @llvm.ctlz.i64(i64 %val, i1 false) + store i64 %ctlz, i64 addrspace(1)* %out.gep + ret void +} + +; FUNC-LABEL: {{^}}v_ctlz_i64_trunc: +define void @v_ctlz_i64_trunc(i32 addrspace(1)* noalias %out, i64 addrspace(1)* noalias %in) nounwind { + %tid = call i32 @llvm.r600.read.tidig.x() + %in.gep = getelementptr i64, i64 addrspace(1)* %in, i32 %tid + %out.gep = getelementptr i32, i32 addrspace(1)* %out, i32 %tid + %val = load i64, i64 addrspace(1)* %in.gep + %ctlz = call i64 @llvm.ctlz.i64(i64 %val, i1 false) + %trunc = trunc i64 %ctlz to i32 + store i32 %trunc, i32 addrspace(1)* %out.gep + ret void +} + +; FUNC-LABEL: {{^}}v_ctlz_i32_sel_eq_neg1: +; SI: buffer_load_dword [[VAL:v[0-9]+]], +; SI: v_ffbh_u32_e32 [[RESULT:v[0-9]+]], [[VAL]] +; SI: buffer_store_dword [[RESULT]], +; SI: s_endpgm + define void @v_ctlz_i32_sel_eq_neg1(i32 addrspace(1)* noalias %out, i32 addrspace(1)* noalias %valptr) nounwind { + %val = load i32, i32 addrspace(1)* %valptr + %ctlz = call i32 @llvm.ctlz.i32(i32 %val, i1 false) nounwind readnone + %cmp = icmp eq i32 %val, 0 + %sel = select i1 %cmp, i32 -1, i32 %ctlz + store i32 %sel, i32 addrspace(1)* %out + ret void +} + +; FUNC-LABEL: {{^}}v_ctlz_i32_sel_ne_neg1: +; SI: buffer_load_dword [[VAL:v[0-9]+]], +; SI: v_ffbh_u32_e32 [[RESULT:v[0-9]+]], [[VAL]] +; SI: buffer_store_dword [[RESULT]], +; SI: s_endpgm +define void @v_ctlz_i32_sel_ne_neg1(i32 addrspace(1)* noalias %out, i32 addrspace(1)* noalias %valptr) nounwind { + %val = load i32, i32 addrspace(1)* %valptr + %ctlz = call i32 @llvm.ctlz.i32(i32 %val, i1 false) nounwind readnone + %cmp = icmp ne i32 %val, 0 + %sel = select i1 %cmp, i32 %ctlz, i32 -1 + store i32 %sel, i32 addrspace(1)* %out + ret void +} + +; TODO: Should be able to eliminate select here as well. +; FUNC-LABEL: {{^}}v_ctlz_i32_sel_eq_bitwidth: +; SI: buffer_load_dword +; SI: v_ffbh_u32_e32 +; SI: v_cmp +; SI: v_cndmask +; SI: s_endpgm +define void @v_ctlz_i32_sel_eq_bitwidth(i32 addrspace(1)* noalias %out, i32 addrspace(1)* noalias %valptr) nounwind { + %val = load i32, i32 addrspace(1)* %valptr + %ctlz = call i32 @llvm.ctlz.i32(i32 %val, i1 false) nounwind readnone + %cmp = icmp eq i32 %ctlz, 32 + %sel = select i1 %cmp, i32 -1, i32 %ctlz + store i32 %sel, i32 addrspace(1)* %out + ret void +} + +; FUNC-LABEL: {{^}}v_ctlz_i32_sel_ne_bitwidth: +; SI: buffer_load_dword +; SI: v_ffbh_u32_e32 +; SI: v_cmp +; SI: v_cndmask +; SI: s_endpgm +define void @v_ctlz_i32_sel_ne_bitwidth(i32 addrspace(1)* noalias %out, i32 addrspace(1)* noalias %valptr) nounwind { + %val = load i32, i32 addrspace(1)* %valptr + %ctlz = call i32 @llvm.ctlz.i32(i32 %val, i1 false) nounwind readnone + %cmp = icmp ne i32 %ctlz, 32 + %sel = select i1 %cmp, i32 %ctlz, i32 -1 + store i32 %sel, i32 addrspace(1)* %out + ret void +} + +; FUNC-LABEL: {{^}}v_ctlz_i8_sel_eq_neg1: +; SI: buffer_load_ubyte [[VAL:v[0-9]+]], +; SI: v_ffbh_u32_e32 [[FFBH:v[0-9]+]], [[VAL]] +; SI: buffer_store_byte [[FFBH]], + define void @v_ctlz_i8_sel_eq_neg1(i8 addrspace(1)* noalias %out, i8 addrspace(1)* noalias %valptr) nounwind { + %val = load i8, i8 addrspace(1)* %valptr + %ctlz = call i8 @llvm.ctlz.i8(i8 %val, i1 false) nounwind readnone + %cmp = icmp eq i8 %val, 0 + %sel = select i1 %cmp, i8 -1, i8 %ctlz + store i8 %sel, i8 addrspace(1)* %out + ret void +} + +; FUNC-LABEL: {{^}}v_ctlz_i16_sel_eq_neg1: +; SI: buffer_load_ushort [[VAL:v[0-9]+]], +; SI: v_ffbh_u32_e32 [[FFBH:v[0-9]+]], [[VAL]] +; SI: buffer_store_short [[FFBH]], + define void @v_ctlz_i16_sel_eq_neg1(i16 addrspace(1)* noalias %out, i16 addrspace(1)* noalias %valptr) nounwind { + %val = load i16, i16 addrspace(1)* %valptr + %ctlz = call i16 @llvm.ctlz.i16(i16 %val, i1 false) nounwind readnone + %cmp = icmp eq i16 %val, 0 + %sel = select i1 %cmp, i16 -1, i16 %ctlz + store i16 %sel, i16 addrspace(1)* %out + ret void +} + +; FUNC-LABEL: {{^}}v_ctlz_i7_sel_eq_neg1: +; SI: buffer_load_ubyte [[VAL:v[0-9]+]], +; SI: v_ffbh_u32_e32 [[FFBH:v[0-9]+]], [[VAL]] +; SI: v_and_b32_e32 [[TRUNC:v[0-9]+]], 0x7f, [[FFBH]] +; SI: buffer_store_byte [[TRUNC]], + define void @v_ctlz_i7_sel_eq_neg1(i7 addrspace(1)* noalias %out, i7 addrspace(1)* noalias %valptr) nounwind { + %val = load i7, i7 addrspace(1)* %valptr + %ctlz = call i7 @llvm.ctlz.i7(i7 %val, i1 false) nounwind readnone + %cmp = icmp eq i7 %val, 0 + %sel = select i1 %cmp, i7 -1, i7 %ctlz + store i7 %sel, i7 addrspace(1)* %out + ret void +} diff --git a/test/CodeGen/AMDGPU/ctlz_zero_undef.ll b/test/CodeGen/AMDGPU/ctlz_zero_undef.ll index bd26c302fe5a..c1f84cd460cf 100644 --- a/test/CodeGen/AMDGPU/ctlz_zero_undef.ll +++ b/test/CodeGen/AMDGPU/ctlz_zero_undef.ll @@ -2,10 +2,18 @@ ; RUN: llc -march=amdgcn -mcpu=tonga -verify-machineinstrs < %s | FileCheck -check-prefix=SI -check-prefix=FUNC %s ; RUN: llc -march=r600 -mcpu=cypress -verify-machineinstrs < %s | FileCheck -check-prefix=EG -check-prefix=FUNC %s +declare i8 @llvm.ctlz.i8(i8, i1) nounwind readnone + declare i32 @llvm.ctlz.i32(i32, i1) nounwind readnone declare <2 x i32> @llvm.ctlz.v2i32(<2 x i32>, i1) nounwind readnone declare <4 x i32> @llvm.ctlz.v4i32(<4 x i32>, i1) nounwind readnone +declare i64 @llvm.ctlz.i64(i64, i1) nounwind readnone +declare <2 x i64> @llvm.ctlz.v2i64(<2 x i64>, i1) nounwind readnone +declare <4 x i64> @llvm.ctlz.v4i64(<4 x i64>, i1) nounwind readnone + +declare i32 @llvm.r600.read.tidig.x() nounwind readnone + ; FUNC-LABEL: {{^}}s_ctlz_zero_undef_i32: ; SI: s_load_dword [[VAL:s[0-9]+]], ; SI: s_flbit_i32_b32 [[SRESULT:s[0-9]+]], [[VAL]] @@ -69,3 +77,192 @@ define void @v_ctlz_zero_undef_v4i32(<4 x i32> addrspace(1)* noalias %out, <4 x store <4 x i32> %ctlz, <4 x i32> addrspace(1)* %out, align 16 ret void } + +; FUNC-LABEL: {{^}}v_ctlz_zero_undef_i8: +; SI: buffer_load_ubyte [[VAL:v[0-9]+]], +; SI: v_ffbh_u32_e32 [[FFBH:v[0-9]+]], [[VAL]] +; SI: v_add_i32_e32 [[RESULT:v[0-9]+]], vcc, 0xffffffe8, [[FFBH]] +; SI: buffer_store_byte [[RESULT]], +define void @v_ctlz_zero_undef_i8(i8 addrspace(1)* noalias %out, i8 addrspace(1)* noalias %valptr) nounwind { + %val = load i8, i8 addrspace(1)* %valptr + %ctlz = call i8 @llvm.ctlz.i8(i8 %val, i1 true) nounwind readnone + store i8 %ctlz, i8 addrspace(1)* %out + ret void +} + +; FUNC-LABEL: {{^}}s_ctlz_zero_undef_i64: +; SI: s_load_dwordx2 s{{\[}}[[LO:[0-9]+]]:[[HI:[0-9]+]]{{\]}}, s{{\[[0-9]+:[0-9]+\]}}, {{0xb|0x2c}} +; SI-DAG: v_cmp_eq_i32_e64 vcc, 0, s[[HI]] +; SI-DAG: s_flbit_i32_b32 [[FFBH_LO:s[0-9]+]], s[[LO]] +; SI-DAG: s_add_i32 [[ADD:s[0-9]+]], [[FFBH_LO]], 32 +; SI-DAG: s_flbit_i32_b32 [[FFBH_HI:s[0-9]+]], s[[HI]] +; SI-DAG: v_mov_b32_e32 [[VFFBH_LO:v[0-9]+]], [[FFBH_LO]] +; SI-DAG: v_mov_b32_e32 [[VFFBH_HI:v[0-9]+]], [[FFBH_HI]] +; SI-DAG: v_cndmask_b32_e32 v[[CTLZ:[0-9]+]], [[VFFBH_HI]], [[VFFBH_LO]] +; SI-DAG: v_mov_b32_e32 v[[CTLZ_HI:[0-9]+]], 0{{$}} +; SI: {{buffer|flat}}_store_dwordx2 v{{\[}}[[CTLZ]]:[[CTLZ_HI]]{{\]}} +define void @s_ctlz_zero_undef_i64(i64 addrspace(1)* noalias %out, i64 %val) nounwind { + %ctlz = call i64 @llvm.ctlz.i64(i64 %val, i1 true) + store i64 %ctlz, i64 addrspace(1)* %out + ret void +} + +; FUNC-LABEL: {{^}}s_ctlz_zero_undef_i64_trunc: +define void @s_ctlz_zero_undef_i64_trunc(i32 addrspace(1)* noalias %out, i64 %val) nounwind { + %ctlz = call i64 @llvm.ctlz.i64(i64 %val, i1 true) + %trunc = trunc i64 %ctlz to i32 + store i32 %trunc, i32 addrspace(1)* %out + ret void +} + +; FUNC-LABEL: {{^}}v_ctlz_zero_undef_i64: +; SI: {{buffer|flat}}_load_dwordx2 v{{\[}}[[LO:[0-9]+]]:[[HI:[0-9]+]]{{\]}} +; SI-DAG: v_cmp_eq_i32_e64 [[CMPHI:s\[[0-9]+:[0-9]+\]]], 0, v[[HI]] +; SI-DAG: v_ffbh_u32_e32 [[FFBH_LO:v[0-9]+]], v[[LO]] +; SI-DAG: v_add_i32_e32 [[ADD:v[0-9]+]], vcc, 32, [[FFBH_LO]] +; SI-DAG: v_ffbh_u32_e32 [[FFBH_HI:v[0-9]+]], v[[HI]] +; SI-DAG: v_cndmask_b32_e64 v[[CTLZ:[0-9]+]], [[FFBH_HI]], [[FFBH_LO]] +; SI-DAG: v_mov_b32_e32 v[[CTLZ_HI:[0-9]+]], 0{{$}} +; SI: {{buffer|flat}}_store_dwordx2 v{{\[}}[[CTLZ]]:[[CTLZ_HI]]{{\]}} +define void @v_ctlz_zero_undef_i64(i64 addrspace(1)* noalias %out, i64 addrspace(1)* noalias %in) nounwind { + %tid = call i32 @llvm.r600.read.tidig.x() + %in.gep = getelementptr i64, i64 addrspace(1)* %in, i32 %tid + %out.gep = getelementptr i64, i64 addrspace(1)* %out, i32 %tid + %val = load i64, i64 addrspace(1)* %in.gep + %ctlz = call i64 @llvm.ctlz.i64(i64 %val, i1 true) + store i64 %ctlz, i64 addrspace(1)* %out.gep + ret void +} + +; FUNC-LABEL: {{^}}v_ctlz_zero_undef_i64_trunc: +define void @v_ctlz_zero_undef_i64_trunc(i32 addrspace(1)* noalias %out, i64 addrspace(1)* noalias %in) nounwind { + %tid = call i32 @llvm.r600.read.tidig.x() + %in.gep = getelementptr i64, i64 addrspace(1)* %in, i32 %tid + %out.gep = getelementptr i32, i32 addrspace(1)* %out, i32 %tid + %val = load i64, i64 addrspace(1)* %in.gep + %ctlz = call i64 @llvm.ctlz.i64(i64 %val, i1 true) + %trunc = trunc i64 %ctlz to i32 + store i32 %trunc, i32 addrspace(1)* %out.gep + ret void +} + +; FUNC-LABEL: {{^}}v_ctlz_zero_undef_i32_sel_eq_neg1: +; SI: buffer_load_dword [[VAL:v[0-9]+]], +; SI: v_ffbh_u32_e32 [[RESULT:v[0-9]+]], [[VAL]] +; SI-NEXT: buffer_store_dword [[RESULT]], + define void @v_ctlz_zero_undef_i32_sel_eq_neg1(i32 addrspace(1)* noalias %out, i32 addrspace(1)* noalias %valptr) nounwind { + %val = load i32, i32 addrspace(1)* %valptr + %ctlz = call i32 @llvm.ctlz.i32(i32 %val, i1 true) nounwind readnone + %cmp = icmp eq i32 %val, 0 + %sel = select i1 %cmp, i32 -1, i32 %ctlz + store i32 %sel, i32 addrspace(1)* %out + ret void +} + +; FUNC-LABEL: {{^}}v_ctlz_zero_undef_i32_sel_ne_neg1: +; SI: buffer_load_dword [[VAL:v[0-9]+]], +; SI: v_ffbh_u32_e32 [[RESULT:v[0-9]+]], [[VAL]] +; SI-NEXT: buffer_store_dword [[RESULT]], +define void @v_ctlz_zero_undef_i32_sel_ne_neg1(i32 addrspace(1)* noalias %out, i32 addrspace(1)* noalias %valptr) nounwind { + %val = load i32, i32 addrspace(1)* %valptr + %ctlz = call i32 @llvm.ctlz.i32(i32 %val, i1 true) nounwind readnone + %cmp = icmp ne i32 %val, 0 + %sel = select i1 %cmp, i32 %ctlz, i32 -1 + store i32 %sel, i32 addrspace(1)* %out + ret void +} + +; FUNC-LABEL: {{^}}v_ctlz_zero_undef_i8_sel_eq_neg1: +; SI: buffer_load_ubyte [[VAL:v[0-9]+]], +; SI: v_ffbh_u32_e32 [[FFBH:v[0-9]+]], [[VAL]] +; SI: buffer_store_byte [[FFBH]], + define void @v_ctlz_zero_undef_i8_sel_eq_neg1(i8 addrspace(1)* noalias %out, i8 addrspace(1)* noalias %valptr) nounwind { + %val = load i8, i8 addrspace(1)* %valptr + %ctlz = call i8 @llvm.ctlz.i8(i8 %val, i1 true) nounwind readnone + %cmp = icmp eq i8 %val, 0 + %sel = select i1 %cmp, i8 -1, i8 %ctlz + store i8 %sel, i8 addrspace(1)* %out + ret void +} + +; FUNC-LABEL: {{^}}v_ctlz_zero_undef_i32_sel_eq_neg1_two_use: +; SI: buffer_load_dword [[VAL:v[0-9]+]], +; SI-DAG: v_ffbh_u32_e32 [[RESULT0:v[0-9]+]], [[VAL]] +; SI-DAG: v_cmp_eq_i32_e32 vcc, 0, [[VAL]] +; SI-DAG: v_cndmask_b32_e64 [[RESULT1:v[0-9]+]], 0, 1, vcc +; SI-DAG: buffer_store_dword [[RESULT0]] +; SI-DAG: buffer_store_byte [[RESULT1]] +; SI: s_endpgm + define void @v_ctlz_zero_undef_i32_sel_eq_neg1_two_use(i32 addrspace(1)* noalias %out, i32 addrspace(1)* noalias %valptr) nounwind { + %val = load i32, i32 addrspace(1)* %valptr + %ctlz = call i32 @llvm.ctlz.i32(i32 %val, i1 true) nounwind readnone + %cmp = icmp eq i32 %val, 0 + %sel = select i1 %cmp, i32 -1, i32 %ctlz + store volatile i32 %sel, i32 addrspace(1)* %out + store volatile i1 %cmp, i1 addrspace(1)* undef + ret void +} + +; Selected on wrong constant +; FUNC-LABEL: {{^}}v_ctlz_zero_undef_i32_sel_eq_0: +; SI: buffer_load_dword +; SI: v_ffbh_u32_e32 +; SI: v_cmp +; SI: v_cndmask +; SI: buffer_store_dword + define void @v_ctlz_zero_undef_i32_sel_eq_0(i32 addrspace(1)* noalias %out, i32 addrspace(1)* noalias %valptr) nounwind { + %val = load i32, i32 addrspace(1)* %valptr + %ctlz = call i32 @llvm.ctlz.i32(i32 %val, i1 true) nounwind readnone + %cmp = icmp eq i32 %val, 0 + %sel = select i1 %cmp, i32 0, i32 %ctlz + store i32 %sel, i32 addrspace(1)* %out + ret void +} + +; Selected on wrong constant +; FUNC-LABEL: {{^}}v_ctlz_zero_undef_i32_sel_ne_0: +; SI: buffer_load_dword +; SI: v_ffbh_u32_e32 +; SI: v_cmp +; SI: v_cndmask +; SI: buffer_store_dword +define void @v_ctlz_zero_undef_i32_sel_ne_0(i32 addrspace(1)* noalias %out, i32 addrspace(1)* noalias %valptr) nounwind { + %val = load i32, i32 addrspace(1)* %valptr + %ctlz = call i32 @llvm.ctlz.i32(i32 %val, i1 true) nounwind readnone + %cmp = icmp ne i32 %val, 0 + %sel = select i1 %cmp, i32 %ctlz, i32 0 + store i32 %sel, i32 addrspace(1)* %out + ret void +} + +; Compare on wrong constant +; FUNC-LABEL: {{^}}v_ctlz_zero_undef_i32_sel_eq_cmp_non0: +; SI: buffer_load_dword +; SI: v_ffbh_u32_e32 +; SI: v_cmp +; SI: v_cndmask +; SI: buffer_store_dword + define void @v_ctlz_zero_undef_i32_sel_eq_cmp_non0(i32 addrspace(1)* noalias %out, i32 addrspace(1)* noalias %valptr) nounwind { + %val = load i32, i32 addrspace(1)* %valptr + %ctlz = call i32 @llvm.ctlz.i32(i32 %val, i1 true) nounwind readnone + %cmp = icmp eq i32 %val, 1 + %sel = select i1 %cmp, i32 0, i32 %ctlz + store i32 %sel, i32 addrspace(1)* %out + ret void +} + +; Selected on wrong constant +; FUNC-LABEL: {{^}}v_ctlz_zero_undef_i32_sel_ne_cmp_non0: +; SI: buffer_load_dword +; SI: v_ffbh_u32_e32 +; SI: v_cmp +; SI: v_cndmask +; SI: buffer_store_dword +define void @v_ctlz_zero_undef_i32_sel_ne_cmp_non0(i32 addrspace(1)* noalias %out, i32 addrspace(1)* noalias %valptr) nounwind { + %val = load i32, i32 addrspace(1)* %valptr + %ctlz = call i32 @llvm.ctlz.i32(i32 %val, i1 true) nounwind readnone + %cmp = icmp ne i32 %val, 1 + %sel = select i1 %cmp, i32 %ctlz, i32 0 + store i32 %sel, i32 addrspace(1)* %out + ret void +} diff --git a/test/CodeGen/AMDGPU/flat-scratch-reg.ll b/test/CodeGen/AMDGPU/flat-scratch-reg.ll index 9aea7c773431..b9489101f906 100644 --- a/test/CodeGen/AMDGPU/flat-scratch-reg.ll +++ b/test/CodeGen/AMDGPU/flat-scratch-reg.ll @@ -1,6 +1,6 @@ ; RUN: llc < %s -march=amdgcn -mcpu=kaveri -verify-machineinstrs | FileCheck %s --check-prefix=GCN --check-prefix=CI --check-prefix=NO-XNACK ; RUN: llc < %s -march=amdgcn -mcpu=fiji -verify-machineinstrs | FileCheck %s --check-prefix=GCN --check-prefix=VI --check-prefix=NO-XNACK -; RUN: llc < %s -march=amdgcn -mcpu=carrizo -mattr=+xnack -verify-machineinstrs | FileCheck %s --check-prefix=GCN --check-prefix=XNACK +; RUN: llc < %s -march=amdgcn -mcpu=carrizo -mattr=+xnack -verify-machineinstrs | FileCheck %s --check-prefix=GCN --check-prefix=VI --check-prefix=XNACK ; GCN-LABEL: {{^}}no_vcc_no_flat: ; NO-XNACK: ; NumSgprs: 8 @@ -22,8 +22,7 @@ entry: ; GCN-LABEL: {{^}}no_vcc_flat: ; CI: ; NumSgprs: 12 -; VI: ; NumSgprs: 12 -; XNACK: ; NumSgprs: 14 +; VI: ; NumSgprs: 14 define void @no_vcc_flat() { entry: call void asm sideeffect "", "~{SGPR7},~{FLAT_SCR}"() @@ -32,8 +31,7 @@ entry: ; GCN-LABEL: {{^}}vcc_flat: ; CI: ; NumSgprs: 12 -; VI: ; NumSgprs: 12 -; XNACK: ; NumSgprs: 14 +; VI: ; NumSgprs: 14 define void @vcc_flat() { entry: call void asm sideeffect "", "~{SGPR7},~{VCC},~{FLAT_SCR}"() diff --git a/test/CodeGen/AMDGPU/fmin_legacy.ll b/test/CodeGen/AMDGPU/fmin_legacy.ll index 52fc3d0d251a..69a0a520a476 100644 --- a/test/CodeGen/AMDGPU/fmin_legacy.ll +++ b/test/CodeGen/AMDGPU/fmin_legacy.ll @@ -8,8 +8,8 @@ declare i32 @llvm.r600.read.tidig.x() #1 ; FUNC-LABEL: @test_fmin_legacy_f32 ; EG: MIN * -; SI-SAFE: v_min_legacy_f32_e32 -; SI-NONAN: v_min_f32_e32 +; SI-SAFE: v_min_legacy_f32_e64 +; SI-NONAN: v_min_f32_e64 define void @test_fmin_legacy_f32(<4 x float> addrspace(1)* %out, <4 x float> inreg %reg0) #0 { %r0 = extractelement <4 x float> %reg0, i32 0 %r1 = extractelement <4 x float> %reg0, i32 1 diff --git a/test/CodeGen/AMDGPU/fsub.ll b/test/CodeGen/AMDGPU/fsub.ll index dfe41cb5b111..38d573258a5e 100644 --- a/test/CodeGen/AMDGPU/fsub.ll +++ b/test/CodeGen/AMDGPU/fsub.ll @@ -32,9 +32,8 @@ declare void @llvm.AMDGPU.store.output(float, i32) ; R600-DAG: ADD {{\** *}}T{{[0-9]+\.[XYZW]}}, KC0[3].X, -KC0[3].Z ; R600-DAG: ADD {{\** *}}T{{[0-9]+\.[XYZW]}}, KC0[2].W, -KC0[3].Y -; FIXME: Should be using SGPR directly for first operand -; SI: v_subrev_f32_e32 {{v[0-9]+}}, {{v[0-9]+}}, {{v[0-9]+}} -; SI: v_subrev_f32_e32 {{v[0-9]+}}, {{v[0-9]+}}, {{v[0-9]+}} +; SI: v_subrev_f32_e32 {{v[0-9]+}}, {{s[0-9]+}}, {{v[0-9]+}} +; SI: v_subrev_f32_e32 {{v[0-9]+}}, {{s[0-9]+}}, {{v[0-9]+}} define void @fsub_v2f32(<2 x float> addrspace(1)* %out, <2 x float> %a, <2 x float> %b) { %sub = fsub <2 x float> %a, %b store <2 x float> %sub, <2 x float> addrspace(1)* %out, align 8 @@ -60,13 +59,11 @@ define void @v_fsub_v4f32(<4 x float> addrspace(1)* %out, <4 x float> addrspace( ret void } -; FIXME: Should be using SGPR directly for first operand - ; FUNC-LABEL: {{^}}s_fsub_v4f32: -; SI: v_subrev_f32_e32 {{v[0-9]+}}, {{v[0-9]+}}, {{v[0-9]+}} -; SI: v_subrev_f32_e32 {{v[0-9]+}}, {{v[0-9]+}}, {{v[0-9]+}} -; SI: v_subrev_f32_e32 {{v[0-9]+}}, {{v[0-9]+}}, {{v[0-9]+}} -; SI: v_subrev_f32_e32 {{v[0-9]+}}, {{v[0-9]+}}, {{v[0-9]+}} +; SI: v_subrev_f32_e32 {{v[0-9]+}}, {{s[0-9]+}}, {{v[0-9]+}} +; SI: v_subrev_f32_e32 {{v[0-9]+}}, {{s[0-9]+}}, {{v[0-9]+}} +; SI: v_subrev_f32_e32 {{v[0-9]+}}, {{s[0-9]+}}, {{v[0-9]+}} +; SI: v_subrev_f32_e32 {{v[0-9]+}}, {{s[0-9]+}}, {{v[0-9]+}} ; SI: s_endpgm define void @s_fsub_v4f32(<4 x float> addrspace(1)* %out, <4 x float> %a, <4 x float> %b) { %result = fsub <4 x float> %a, %b diff --git a/test/CodeGen/AMDGPU/hsa-globals.ll b/test/CodeGen/AMDGPU/hsa-globals.ll index 1d76c40c042e..90322ac3dc01 100644 --- a/test/CodeGen/AMDGPU/hsa-globals.ll +++ b/test/CodeGen/AMDGPU/hsa-globals.ll @@ -17,41 +17,49 @@ define void @test() { } ; ASM: .amdgpu_hsa_module_global internal_global +; ASM: .size internal_global_program, 4 ; ASM: .hsadata_global_program ; ASM: internal_global_program: ; ASM: .long 0 ; ASM: .amdgpu_hsa_module_global common_global +; ASM: .size common_global_program, 4 ; ASM: .hsadata_global_program ; ASM: common_global_program: ; ASM: .long 0 ; ASM: .amdgpu_hsa_program_global external_global +; ASM: .size external_global_program, 4 ; ASM: .hsadata_global_program ; ASM: external_global_program: ; ASM: .long 0 ; ASM: .amdgpu_hsa_module_global internal_global +; ASM: .size internal_global_agent, 4 ; ASM: .hsadata_global_agent ; ASM: internal_global_agent: ; ASM: .long 0 ; ASM: .amdgpu_hsa_module_global common_global +; ASM: .size common_global_agent, 4 ; ASM: .hsadata_global_agent ; ASM: common_global_agent: ; ASM: .long 0 ; ASM: .amdgpu_hsa_program_global external_global +; ASM: .size external_global_agent, 4 ; ASM: .hsadata_global_agent ; ASM: external_global_agent: ; ASM: .long 0 ; ASM: .amdgpu_hsa_module_global internal_readonly +; ASM: .size internal_readonly, 4 ; ASM: .hsatext ; ASM: internal_readonly: ; ASM: .long 0 ; ASM: .amdgpu_hsa_program_global external_readonly +; ASM: .size external_readonly, 4 ; ASM: .hsatext ; ASM: external_readonly: ; ASM: .long 0 @@ -79,18 +87,21 @@ define void @test() { ; ELF: Symbol { ; ELF: Name: common_global_agent +; ELF: Size: 4 ; ELF: Binding: Local ; ELF: Section: .hsadata_global_agent ; ELF: } ; ELF: Symbol { ; ELF: Name: common_global_program +; ELF: Size: 4 ; ELF: Binding: Local ; ELF: Section: .hsadata_global_program ; ELF: } ; ELF: Symbol { ; ELF: Name: internal_global_agent +; ELF: Size: 4 ; ELF: Binding: Local ; ELF: Type: Object ; ELF: Section: .hsadata_global_agent @@ -98,6 +109,7 @@ define void @test() { ; ELF: Symbol { ; ELF: Name: internal_global_program +; ELF: Size: 4 ; ELF: Binding: Local ; ELF: Type: Object ; ELF: Section: .hsadata_global_program @@ -105,6 +117,7 @@ define void @test() { ; ELF: Symbol { ; ELF: Name: internal_readonly +; ELF: Size: 4 ; ELF: Binding: Local ; ELF: Type: Object ; ELF: Section: .hsatext @@ -112,6 +125,7 @@ define void @test() { ; ELF: Symbol { ; ELF: Name: external_global_agent +; ELF: Size: 4 ; ELF: Binding: Global ; ELF: Type: Object ; ELF: Section: .hsadata_global_agent @@ -119,6 +133,7 @@ define void @test() { ; ELF: Symbol { ; ELF: Name: external_global_program +; ELF: Size: 4 ; ELF: Binding: Global ; ELF: Type: Object ; ELF: Section: .hsadata_global_program @@ -126,6 +141,7 @@ define void @test() { ; ELF: Symbol { ; ELF: Name: external_readonly +; ELF: Size: 4 ; ELF: Binding: Global ; ELF: Type: Object ; ELF: Section: .hsatext diff --git a/test/CodeGen/AMDGPU/hsa-note-no-func.ll b/test/CodeGen/AMDGPU/hsa-note-no-func.ll new file mode 100644 index 000000000000..0e4662231b4f --- /dev/null +++ b/test/CodeGen/AMDGPU/hsa-note-no-func.ll @@ -0,0 +1,6 @@ +; RUN: llc < %s -mtriple=amdgcn--amdhsa -mcpu=kaveri | FileCheck --check-prefix=HSA --check-prefix=HSA-CI %s +; RUN: llc < %s -mtriple=amdgcn--amdhsa -mcpu=carrizo | FileCheck --check-prefix=HSA --check-prefix=HSA-VI %s + +; HSA: .hsa_code_object_version 1,0 +; HSA-CI: .hsa_code_object_isa 7,0,0,"AMD","AMDGPU" +; HSA-VI: .hsa_code_object_isa 8,0,1,"AMD","AMDGPU" diff --git a/test/CodeGen/AMDGPU/hsa.ll b/test/CodeGen/AMDGPU/hsa.ll index abc89b7fd837..c089dfd9a971 100644 --- a/test/CodeGen/AMDGPU/hsa.ll +++ b/test/CodeGen/AMDGPU/hsa.ll @@ -28,6 +28,7 @@ ; ELF: Symbol { ; ELF: Name: simple +; ELF: Size: 296 ; ELF: Type: AMDGPU_HSA_KERNEL (0xA) ; ELF: } @@ -52,6 +53,9 @@ ; Make sure we generate flat store for HSA ; HSA: flat_store_dword v{{[0-9]+}} +; HSA: .Lfunc_end0: +; HSA: .size simple, .Lfunc_end0-simple + define void @simple(i32 addrspace(1)* %out) { entry: store i32 0, i32 addrspace(1)* %out diff --git a/test/CodeGen/AMDGPU/inline-asm.ll b/test/CodeGen/AMDGPU/inline-asm.ll index efc2292de3a5..9c8d3534f8ad 100644 --- a/test/CodeGen/AMDGPU/inline-asm.ll +++ b/test/CodeGen/AMDGPU/inline-asm.ll @@ -10,3 +10,14 @@ entry: call void asm sideeffect "s_endpgm", ""() ret void } + +; CHECK: {{^}}inline_asm_shader: +; CHECK: s_endpgm +; CHECK: s_endpgm +define void @inline_asm_shader() #0 { +entry: + call void asm sideeffect "s_endpgm", ""() + ret void +} + +attributes #0 = { "ShaderType"="0" } diff --git a/test/CodeGen/AMDGPU/llvm.amdgcn.dispatch.ptr.ll b/test/CodeGen/AMDGPU/llvm.amdgcn.dispatch.ptr.ll index dc95cd1ee012..d96ea743f6ed 100644 --- a/test/CodeGen/AMDGPU/llvm.amdgcn.dispatch.ptr.ll +++ b/test/CodeGen/AMDGPU/llvm.amdgcn.dispatch.ptr.ll @@ -1,4 +1,7 @@ ; RUN: llc -mtriple=amdgcn--amdhsa -mcpu=kaveri -verify-machineinstrs < %s | FileCheck -check-prefix=GCN %s +; RUN: not llc -mtriple=amdgcn-unknown-unknown -mcpu=kaveri -verify-machineinstrs < %s 2>&1 | FileCheck -check-prefix=ERROR %s + +; ERROR: error: unsupported hsa intrinsic without hsa target in test ; GCN-LABEL: {{^}}test: ; GCN: enable_sgpr_dispatch_ptr = 1 diff --git a/test/CodeGen/AMDGPU/llvm.round.f64.ll b/test/CodeGen/AMDGPU/llvm.round.f64.ll index 6b365dc09e2a..98afbeee93e6 100644 --- a/test/CodeGen/AMDGPU/llvm.round.f64.ll +++ b/test/CodeGen/AMDGPU/llvm.round.f64.ll @@ -21,7 +21,7 @@ define void @round_f64(double addrspace(1)* %out, double %x) #0 { ; SI-DAG: v_cmp_eq_i32 ; SI-DAG: s_mov_b32 [[BFIMASK:s[0-9]+]], 0x7fffffff -; SI-DAG: v_cmp_gt_i32_e32 +; SI-DAG: v_cmp_gt_i32 ; SI-DAG: v_bfi_b32 [[COPYSIGN:v[0-9]+]], [[BFIMASK]] ; SI: buffer_store_dwordx2 diff --git a/test/CodeGen/AMDGPU/ret.ll b/test/CodeGen/AMDGPU/ret.ll new file mode 100644 index 000000000000..2bd9fd6858fe --- /dev/null +++ b/test/CodeGen/AMDGPU/ret.ll @@ -0,0 +1,245 @@ +; RUN: llc -march=amdgcn -mcpu=tahiti -verify-machineinstrs < %s | FileCheck -check-prefix=GCN %s +; RUN: llc -march=amdgcn -mcpu=tonga -verify-machineinstrs < %s | FileCheck -check-prefix=GCN %s + +attributes #0 = { "ShaderType"="1" } + +declare void @llvm.SI.export(i32, i32, i32, i32, i32, float, float, float, float) + +; GCN-LABEL: {{^}}vgpr: +; GCN: v_mov_b32_e32 v1, v0 +; GCN-DAG: v_add_f32_e32 v0, 1.0, v1 +; GCN-DAG: exp 15, 0, 1, 1, 1, v1, v1, v1, v1 +; GCN: s_waitcnt expcnt(0) +; GCN-NOT: s_endpgm +define {float, float} @vgpr([9 x <16 x i8>] addrspace(2)* byval, i32 inreg, i32 inreg, float) #0 { + call void @llvm.SI.export(i32 15, i32 1, i32 1, i32 0, i32 1, float %3, float %3, float %3, float %3) + %x = fadd float %3, 1.0 + %a = insertvalue {float, float} undef, float %x, 0 + %b = insertvalue {float, float} %a, float %3, 1 + ret {float, float} %b +} + +; GCN-LABEL: {{^}}vgpr_literal: +; GCN: v_mov_b32_e32 v4, v0 +; GCN-DAG: v_mov_b32_e32 v0, 1.0 +; GCN-DAG: v_mov_b32_e32 v1, 2.0 +; GCN-DAG: v_mov_b32_e32 v2, 4.0 +; GCN-DAG: v_mov_b32_e32 v3, -1.0 +; GCN: exp 15, 0, 1, 1, 1, v4, v4, v4, v4 +; GCN: s_waitcnt expcnt(0) +; GCN-NOT: s_endpgm +define {float, float, float, float} @vgpr_literal([9 x <16 x i8>] addrspace(2)* byval, i32 inreg, i32 inreg, float) #0 { + call void @llvm.SI.export(i32 15, i32 1, i32 1, i32 0, i32 1, float %3, float %3, float %3, float %3) + ret {float, float, float, float} {float 1.0, float 2.0, float 4.0, float -1.0} +} + + +; GCN: .long 165580 +; GCN-NEXT: .long 562 +; GCN-NEXT: .long 165584 +; GCN-NEXT: .long 562 +; GCN-LABEL: {{^}}vgpr_ps_addr0: +; GCN-NOT: v_mov_b32_e32 v0 +; GCN-NOT: v_mov_b32_e32 v1 +; GCN-NOT: v_mov_b32_e32 v2 +; GCN: v_mov_b32_e32 v3, v4 +; GCN: v_mov_b32_e32 v4, v6 +; GCN-NOT: s_endpgm +attributes #1 = { "ShaderType"="0" "InitialPSInputAddr"="0" } +define {float, float, float, float, float} @vgpr_ps_addr0([9 x <16 x i8>] addrspace(2)* byval, i32 inreg, i32 inreg, <2 x i32>, <2 x i32>, <2 x i32>, <3 x i32>, <2 x i32>, <2 x i32>, <2 x i32>, float, float, float, float, float, float, float, float, float) #1 { + %i0 = extractelement <2 x i32> %4, i32 0 + %i1 = extractelement <2 x i32> %4, i32 1 + %i2 = extractelement <2 x i32> %7, i32 0 + %i3 = extractelement <2 x i32> %8, i32 0 + %f0 = bitcast i32 %i0 to float + %f1 = bitcast i32 %i1 to float + %f2 = bitcast i32 %i2 to float + %f3 = bitcast i32 %i3 to float + %r0 = insertvalue {float, float, float, float, float} undef, float %f0, 0 + %r1 = insertvalue {float, float, float, float, float} %r0, float %f1, 1 + %r2 = insertvalue {float, float, float, float, float} %r1, float %f2, 2 + %r3 = insertvalue {float, float, float, float, float} %r2, float %f3, 3 + %r4 = insertvalue {float, float, float, float, float} %r3, float %12, 4 + ret {float, float, float, float, float} %r4 +} + + +; GCN: .long 165580 +; GCN-NEXT: .long 1 +; GCN-NEXT: .long 165584 +; GCN-NEXT: .long 1 +; GCN-LABEL: {{^}}ps_input_ena_no_inputs: +; GCN: v_mov_b32_e32 v0, 1.0 +; GCN-NOT: s_endpgm +define float @ps_input_ena_no_inputs([9 x <16 x i8>] addrspace(2)* byval, i32 inreg, i32 inreg, <2 x i32>, <2 x i32>, <2 x i32>, <3 x i32>, <2 x i32>, <2 x i32>, <2 x i32>, float, float, float, float, float, float, float, float, float) #1 { + ret float 1.0 +} + + +; GCN: .long 165580 +; GCN-NEXT: .long 2081 +; GCN-NEXT: .long 165584 +; GCN-NEXT: .long 2081 +; GCN-LABEL: {{^}}ps_input_ena_pos_w: +; GCN-DAG: v_mov_b32_e32 v0, v4 +; GCN-DAG: v_mov_b32_e32 v1, v2 +; GCN: v_mov_b32_e32 v2, v3 +; GCN-NOT: s_endpgm +define {float, <2 x float>} @ps_input_ena_pos_w([9 x <16 x i8>] addrspace(2)* byval, i32 inreg, i32 inreg, <2 x i32>, <2 x i32>, <2 x i32>, <3 x i32>, <2 x i32>, <2 x i32>, <2 x i32>, float, float, float, float, float, float, float, float, float) #1 { + %f = bitcast <2 x i32> %8 to <2 x float> + %s = insertvalue {float, <2 x float>} undef, float %14, 0 + %s1 = insertvalue {float, <2 x float>} %s, <2 x float> %f, 1 + ret {float, <2 x float>} %s1 +} + + +; GCN: .long 165580 +; GCN-NEXT: .long 562 +; GCN-NEXT: .long 165584 +; GCN-NEXT: .long 563 +; GCN-LABEL: {{^}}vgpr_ps_addr1: +; GCN-DAG: v_mov_b32_e32 v0, v2 +; GCN-DAG: v_mov_b32_e32 v1, v3 +; GCN: v_mov_b32_e32 v2, v4 +; GCN-DAG: v_mov_b32_e32 v3, v6 +; GCN-DAG: v_mov_b32_e32 v4, v8 +; GCN-NOT: s_endpgm +attributes #2 = { "ShaderType"="0" "InitialPSInputAddr"="1" } +define {float, float, float, float, float} @vgpr_ps_addr1([9 x <16 x i8>] addrspace(2)* byval, i32 inreg, i32 inreg, <2 x i32>, <2 x i32>, <2 x i32>, <3 x i32>, <2 x i32>, <2 x i32>, <2 x i32>, float, float, float, float, float, float, float, float, float) #2 { + %i0 = extractelement <2 x i32> %4, i32 0 + %i1 = extractelement <2 x i32> %4, i32 1 + %i2 = extractelement <2 x i32> %7, i32 0 + %i3 = extractelement <2 x i32> %8, i32 0 + %f0 = bitcast i32 %i0 to float + %f1 = bitcast i32 %i1 to float + %f2 = bitcast i32 %i2 to float + %f3 = bitcast i32 %i3 to float + %r0 = insertvalue {float, float, float, float, float} undef, float %f0, 0 + %r1 = insertvalue {float, float, float, float, float} %r0, float %f1, 1 + %r2 = insertvalue {float, float, float, float, float} %r1, float %f2, 2 + %r3 = insertvalue {float, float, float, float, float} %r2, float %f3, 3 + %r4 = insertvalue {float, float, float, float, float} %r3, float %12, 4 + ret {float, float, float, float, float} %r4 +} + + +; GCN: .long 165580 +; GCN-NEXT: .long 562 +; GCN-NEXT: .long 165584 +; GCN-NEXT: .long 631 +; GCN-LABEL: {{^}}vgpr_ps_addr119: +; GCN-DAG: v_mov_b32_e32 v0, v2 +; GCN-DAG: v_mov_b32_e32 v1, v3 +; GCN: v_mov_b32_e32 v2, v6 +; GCN: v_mov_b32_e32 v3, v8 +; GCN: v_mov_b32_e32 v4, v12 +; GCN-NOT: s_endpgm +attributes #3 = { "ShaderType"="0" "InitialPSInputAddr"="119" } +define {float, float, float, float, float} @vgpr_ps_addr119([9 x <16 x i8>] addrspace(2)* byval, i32 inreg, i32 inreg, <2 x i32>, <2 x i32>, <2 x i32>, <3 x i32>, <2 x i32>, <2 x i32>, <2 x i32>, float, float, float, float, float, float, float, float, float) #3 { + %i0 = extractelement <2 x i32> %4, i32 0 + %i1 = extractelement <2 x i32> %4, i32 1 + %i2 = extractelement <2 x i32> %7, i32 0 + %i3 = extractelement <2 x i32> %8, i32 0 + %f0 = bitcast i32 %i0 to float + %f1 = bitcast i32 %i1 to float + %f2 = bitcast i32 %i2 to float + %f3 = bitcast i32 %i3 to float + %r0 = insertvalue {float, float, float, float, float} undef, float %f0, 0 + %r1 = insertvalue {float, float, float, float, float} %r0, float %f1, 1 + %r2 = insertvalue {float, float, float, float, float} %r1, float %f2, 2 + %r3 = insertvalue {float, float, float, float, float} %r2, float %f3, 3 + %r4 = insertvalue {float, float, float, float, float} %r3, float %12, 4 + ret {float, float, float, float, float} %r4 +} + + +; GCN: .long 165580 +; GCN-NEXT: .long 562 +; GCN-NEXT: .long 165584 +; GCN-NEXT: .long 946 +; GCN-LABEL: {{^}}vgpr_ps_addr418: +; GCN-NOT: v_mov_b32_e32 v0 +; GCN-NOT: v_mov_b32_e32 v1 +; GCN-NOT: v_mov_b32_e32 v2 +; GCN: v_mov_b32_e32 v3, v4 +; GCN: v_mov_b32_e32 v4, v8 +; GCN-NOT: s_endpgm +attributes #4 = { "ShaderType"="0" "InitialPSInputAddr"="418" } +define {float, float, float, float, float} @vgpr_ps_addr418([9 x <16 x i8>] addrspace(2)* byval, i32 inreg, i32 inreg, <2 x i32>, <2 x i32>, <2 x i32>, <3 x i32>, <2 x i32>, <2 x i32>, <2 x i32>, float, float, float, float, float, float, float, float, float) #4 { + %i0 = extractelement <2 x i32> %4, i32 0 + %i1 = extractelement <2 x i32> %4, i32 1 + %i2 = extractelement <2 x i32> %7, i32 0 + %i3 = extractelement <2 x i32> %8, i32 0 + %f0 = bitcast i32 %i0 to float + %f1 = bitcast i32 %i1 to float + %f2 = bitcast i32 %i2 to float + %f3 = bitcast i32 %i3 to float + %r0 = insertvalue {float, float, float, float, float} undef, float %f0, 0 + %r1 = insertvalue {float, float, float, float, float} %r0, float %f1, 1 + %r2 = insertvalue {float, float, float, float, float} %r1, float %f2, 2 + %r3 = insertvalue {float, float, float, float, float} %r2, float %f3, 3 + %r4 = insertvalue {float, float, float, float, float} %r3, float %12, 4 + ret {float, float, float, float, float} %r4 +} + + +; GCN-LABEL: {{^}}sgpr: +; GCN: s_add_i32 s0, s3, 2 +; GCN: s_mov_b32 s2, s3 +; GCN-NOT: s_endpgm +define {i32, i32, i32} @sgpr([9 x <16 x i8>] addrspace(2)* byval, i32 inreg, i32 inreg, float) #0 { + %x = add i32 %2, 2 + %a = insertvalue {i32, i32, i32} undef, i32 %x, 0 + %b = insertvalue {i32, i32, i32} %a, i32 %1, 1 + %c = insertvalue {i32, i32, i32} %a, i32 %2, 2 + ret {i32, i32, i32} %c +} + + +; GCN-LABEL: {{^}}sgpr_literal: +; GCN: s_mov_b32 s0, 5 +; GCN-NOT: s_mov_b32 s0, s0 +; GCN-DAG: s_mov_b32 s1, 6 +; GCN-DAG: s_mov_b32 s2, 7 +; GCN-DAG: s_mov_b32 s3, 8 +; GCN-NOT: s_endpgm +define {i32, i32, i32, i32} @sgpr_literal([9 x <16 x i8>] addrspace(2)* byval, i32 inreg, i32 inreg, float) #0 { + %x = add i32 %2, 2 + ret {i32, i32, i32, i32} {i32 5, i32 6, i32 7, i32 8} +} + + +; GCN-LABEL: {{^}}both: +; GCN: v_mov_b32_e32 v1, v0 +; GCN-DAG: exp 15, 0, 1, 1, 1, v1, v1, v1, v1 +; GCN-DAG: v_add_f32_e32 v0, 1.0, v1 +; GCN-DAG: s_add_i32 s0, s3, 2 +; GCN-DAG: s_mov_b32 s1, s2 +; GCN: s_mov_b32 s2, s3 +; GCN: s_waitcnt expcnt(0) +; GCN-NOT: s_endpgm +define {float, i32, float, i32, i32} @both([9 x <16 x i8>] addrspace(2)* byval, i32 inreg, i32 inreg, float) #0 { + call void @llvm.SI.export(i32 15, i32 1, i32 1, i32 0, i32 1, float %3, float %3, float %3, float %3) + %v = fadd float %3, 1.0 + %s = add i32 %2, 2 + %a0 = insertvalue {float, i32, float, i32, i32} undef, float %v, 0 + %a1 = insertvalue {float, i32, float, i32, i32} %a0, i32 %s, 1 + %a2 = insertvalue {float, i32, float, i32, i32} %a1, float %3, 2 + %a3 = insertvalue {float, i32, float, i32, i32} %a2, i32 %1, 3 + %a4 = insertvalue {float, i32, float, i32, i32} %a3, i32 %2, 4 + ret {float, i32, float, i32, i32} %a4 +} + + +; GCN-LABEL: {{^}}structure_literal: +; GCN: v_mov_b32_e32 v3, v0 +; GCN-DAG: v_mov_b32_e32 v0, 1.0 +; GCN-DAG: s_mov_b32 s0, 2 +; GCN-DAG: s_mov_b32 s1, 3 +; GCN-DAG: v_mov_b32_e32 v1, 2.0 +; GCN-DAG: v_mov_b32_e32 v2, 4.0 +; GCN-DAG: exp 15, 0, 1, 1, 1, v3, v3, v3, v3 +define {{float, i32}, {i32, <2 x float>}} @structure_literal([9 x <16 x i8>] addrspace(2)* byval, i32 inreg, i32 inreg, float) #0 { + call void @llvm.SI.export(i32 15, i32 1, i32 1, i32 0, i32 1, float %3, float %3, float %3, float %3) + ret {{float, i32}, {i32, <2 x float>}} {{float, i32} {float 1.0, i32 2}, {i32, <2 x float>} {i32 3, <2 x float> <float 2.0, float 4.0>}} +} diff --git a/test/CodeGen/AMDGPU/si-scheduler.ll b/test/CodeGen/AMDGPU/si-scheduler.ll new file mode 100644 index 000000000000..66a9571d75bf --- /dev/null +++ b/test/CodeGen/AMDGPU/si-scheduler.ll @@ -0,0 +1,55 @@ +; RUN: llc -march=amdgcn -mcpu=SI --misched=si < %s | FileCheck %s + +; The test checks the "si" machine scheduler pass works correctly. + +; CHECK-LABEL: {{^}}main: +; CHECK: s_wqm +; CHECK: s_load_dwordx4 +; CHECK: s_load_dwordx8 +; CHECK: s_waitcnt lgkmcnt(0) +; CHECK: image_sample +; CHECK: s_waitcnt vmcnt(0) +; CHECK: exp +; CHECK: s_endpgm + +define void @main([6 x <16 x i8>] addrspace(2)* byval, [17 x <16 x i8>] addrspace(2)* byval, [17 x <4 x i32>] addrspace(2)* byval, [34 x <8 x i32>] addrspace(2)* byval, float inreg, i32 inreg, <2 x i32>, +<2 x i32>, <2 x i32>, <3 x i32>, <2 x i32>, <2 x i32>, <2 x i32>, float, float, float, float, float, float, i32, float, float) #0 { +main_body: + %22 = bitcast [34 x <8 x i32>] addrspace(2)* %3 to <32 x i8> addrspace(2)* + %23 = load <32 x i8>, <32 x i8> addrspace(2)* %22, align 32, !tbaa !0 + %24 = bitcast [17 x <4 x i32>] addrspace(2)* %2 to <16 x i8> addrspace(2)* + %25 = load <16 x i8>, <16 x i8> addrspace(2)* %24, align 16, !tbaa !0 + %26 = call float @llvm.SI.fs.interp(i32 0, i32 0, i32 %5, <2 x i32> %11) + %27 = call float @llvm.SI.fs.interp(i32 1, i32 0, i32 %5, <2 x i32> %11) + %28 = bitcast float %26 to i32 + %29 = bitcast float %27 to i32 + %30 = insertelement <2 x i32> undef, i32 %28, i32 0 + %31 = insertelement <2 x i32> %30, i32 %29, i32 1 + %32 = call <4 x float> @llvm.SI.sample.v2i32(<2 x i32> %31, <32 x i8> %23, <16 x i8> %25, i32 2) + %33 = extractelement <4 x float> %32, i32 0 + %34 = extractelement <4 x float> %32, i32 1 + %35 = extractelement <4 x float> %32, i32 2 + %36 = extractelement <4 x float> %32, i32 3 + %37 = call i32 @llvm.SI.packf16(float %33, float %34) + %38 = bitcast i32 %37 to float + %39 = call i32 @llvm.SI.packf16(float %35, float %36) + %40 = bitcast i32 %39 to float + call void @llvm.SI.export(i32 15, i32 1, i32 1, i32 0, i32 1, float %38, float %40, float %38, float %40) + ret void +} + +; Function Attrs: nounwind readnone +declare float @llvm.SI.fs.interp(i32, i32, i32, <2 x i32>) #1 + +; Function Attrs: nounwind readnone +declare <4 x float> @llvm.SI.sample.v2i32(<2 x i32>, <32 x i8>, <16 x i8>, i32) #1 + +; Function Attrs: nounwind readnone +declare i32 @llvm.SI.packf16(float, float) #1 + +declare void @llvm.SI.export(i32, i32, i32, i32, i32, float, float, float, float) + +attributes #0 = { "ShaderType"="0" "enable-no-nans-fp-math"="true" } +attributes #1 = { nounwind readnone } + +!0 = !{!"const", null, i32 1} diff --git a/test/CodeGen/AMDGPU/sint_to_fp.i64.ll b/test/CodeGen/AMDGPU/sint_to_fp.i64.ll new file mode 100644 index 000000000000..138b93b16d8d --- /dev/null +++ b/test/CodeGen/AMDGPU/sint_to_fp.i64.ll @@ -0,0 +1,62 @@ +; RUN: llc -march=amdgcn -mcpu=SI -verify-machineinstrs < %s | FileCheck -check-prefix=GCN -check-prefix=SI -check-prefix=FUNC %s +; RUN: llc -march=amdgcn -mcpu=tonga -verify-machineinstrs < %s | FileCheck -check-prefix=GCN -check-prefix=VI -check-prefix=FUNC %s + +; FIXME: This should be merged with sint_to_fp.ll, but s_sint_to_fp_v2i64 crashes on r600 + +; FUNC-LABEL: {{^}}s_sint_to_fp_i64_to_f32: +define void @s_sint_to_fp_i64_to_f32(float addrspace(1)* %out, i64 %in) #0 { + %result = sitofp i64 %in to float + store float %result, float addrspace(1)* %out + ret void +} + +; FUNC-LABEL: {{^}}v_sint_to_fp_i64_to_f32: +; GCN: {{buffer|flat}}_load_dwordx2 + +; SI: v_ashr_i64 {{v\[[0-9]+:[0-9]+\]}}, {{v\[[0-9]+:[0-9]+\]}}, 63 +; VI: v_ashrrev_i64 {{v\[[0-9]+:[0-9]+\]}}, 63, {{v\[[0-9]+:[0-9]+\]}} +; GCN: v_xor_b32 + +; GCN: v_ffbh_u32 +; GCN: v_ffbh_u32 +; GCN: v_cndmask +; GCN: v_cndmask + +; GCN-DAG: v_cmp_eq_i64 +; GCN-DAG: v_cmp_lt_u64 + +; GCN: v_xor_b32_e32 v{{[0-9]+}}, 0x80000000, v{{[0-9]+}} +; GCN: v_cndmask_b32_e32 [[SIGN_SEL:v[0-9]+]], +; GCN: {{buffer|flat}}_store_dword [[SIGN_SEL]] +define void @v_sint_to_fp_i64_to_f32(float addrspace(1)* %out, i64 addrspace(1)* %in) #0 { + %tid = call i32 @llvm.r600.read.tidig.x() + %in.gep = getelementptr i64, i64 addrspace(1)* %in, i32 %tid + %out.gep = getelementptr float, float addrspace(1)* %out, i32 %tid + %val = load i64, i64 addrspace(1)* %in.gep + %result = sitofp i64 %val to float + store float %result, float addrspace(1)* %out.gep + ret void +} + +; FUNC-LABEL: {{^}}s_sint_to_fp_v2i64: +define void @s_sint_to_fp_v2i64(<2 x float> addrspace(1)* %out, <2 x i64> %in) #0{ + %result = sitofp <2 x i64> %in to <2 x float> + store <2 x float> %result, <2 x float> addrspace(1)* %out + ret void +} + +; FUNC-LABEL: {{^}}v_sint_to_fp_v4i64: +define void @v_sint_to_fp_v4i64(<4 x float> addrspace(1)* %out, <4 x i64> addrspace(1)* %in) #0 { + %tid = call i32 @llvm.r600.read.tidig.x() + %in.gep = getelementptr <4 x i64>, <4 x i64> addrspace(1)* %in, i32 %tid + %out.gep = getelementptr <4 x float>, <4 x float> addrspace(1)* %out, i32 %tid + %value = load <4 x i64>, <4 x i64> addrspace(1)* %in.gep + %result = sitofp <4 x i64> %value to <4 x float> + store <4 x float> %result, <4 x float> addrspace(1)* %out.gep + ret void +} + +declare i32 @llvm.r600.read.tidig.x() #1 + +attributes #0 = { nounwind } +attributes #1 = { nounwind readnone } diff --git a/test/CodeGen/AMDGPU/sint_to_fp.ll b/test/CodeGen/AMDGPU/sint_to_fp.ll index 8506441d1361..851085c9535d 100644 --- a/test/CodeGen/AMDGPU/sint_to_fp.ll +++ b/test/CodeGen/AMDGPU/sint_to_fp.ll @@ -2,63 +2,120 @@ ; RUN: llc -march=amdgcn -mcpu=tonga -verify-machineinstrs < %s | FileCheck -check-prefix=SI -check-prefix=FUNC %s ; RUN: llc -march=r600 -mcpu=redwood < %s | FileCheck -check-prefix=R600 -check-prefix=FUNC %s - ; FUNC-LABEL: {{^}}s_sint_to_fp_i32_to_f32: -; R600: INT_TO_FLT * T{{[0-9]+\.[XYZW]}}, KC0[2].Z ; SI: v_cvt_f32_i32_e32 {{v[0-9]+}}, {{s[0-9]+$}} -define void @s_sint_to_fp_i32_to_f32(float addrspace(1)* %out, i32 %in) { + +; R600: INT_TO_FLT * T{{[0-9]+\.[XYZW]}}, KC0[2].Z +define void @s_sint_to_fp_i32_to_f32(float addrspace(1)* %out, i32 %in) #0 { %result = sitofp i32 %in to float store float %result, float addrspace(1)* %out ret void } -; FUNC-LABEL: {{^}}sint_to_fp_v2i32: -; R600-DAG: INT_TO_FLT * T{{[0-9]+\.[XYZW]}}, KC0[2].W -; R600-DAG: INT_TO_FLT * T{{[0-9]+\.[XYZW]}}, KC0[3].X +; FUNC-LABEL: {{^}}v_sint_to_fp_i32_to_f32: +; SI: v_cvt_f32_i32_e32 {{v[0-9]+}}, {{v[0-9]+$}} +; R600: INT_TO_FLT +define void @v_sint_to_fp_i32_to_f32(float addrspace(1)* %out, i32 addrspace(1)* %in) #0 { + %tid = call i32 @llvm.r600.read.tidig.x() + %in.gep = getelementptr i32, i32 addrspace(1)* %in, i32 %tid + %out.gep = getelementptr float, float addrspace(1)* %out, i32 %tid + %val = load i32, i32 addrspace(1)* %in.gep + %result = sitofp i32 %val to float + store float %result, float addrspace(1)* %out.gep + ret void +} + +; FUNC-LABEL: {{^}}s_sint_to_fp_v2i32: ; SI: v_cvt_f32_i32_e32 ; SI: v_cvt_f32_i32_e32 -define void @sint_to_fp_v2i32(<2 x float> addrspace(1)* %out, <2 x i32> %in) { + +; R600-DAG: INT_TO_FLT * T{{[0-9]+\.[XYZW]}}, KC0[2].W +; R600-DAG: INT_TO_FLT * T{{[0-9]+\.[XYZW]}}, KC0[3].X +define void @s_sint_to_fp_v2i32(<2 x float> addrspace(1)* %out, <2 x i32> %in) #0{ %result = sitofp <2 x i32> %in to <2 x float> store <2 x float> %result, <2 x float> addrspace(1)* %out ret void } -; FUNC-LABEL: {{^}}sint_to_fp_v4i32: +; FUNC-LABEL: {{^}}s_sint_to_fp_v4i32_to_v4f32: +; SI: v_cvt_f32_i32_e32 +; SI: v_cvt_f32_i32_e32 +; SI: v_cvt_f32_i32_e32 +; SI: v_cvt_f32_i32_e32 +; SI: s_endpgm + ; R600: INT_TO_FLT * T{{[0-9]+\.[XYZW], T[0-9]+\.[XYZW]}} ; R600: INT_TO_FLT * T{{[0-9]+\.[XYZW], T[0-9]+\.[XYZW]}} ; R600: INT_TO_FLT * T{{[0-9]+\.[XYZW], T[0-9]+\.[XYZW]}} ; R600: INT_TO_FLT * T{{[0-9]+\.[XYZW], T[0-9]+\.[XYZW]}} +define void @s_sint_to_fp_v4i32_to_v4f32(<4 x float> addrspace(1)* %out, <4 x i32> addrspace(1)* %in) #0 { + %value = load <4 x i32>, <4 x i32> addrspace(1) * %in + %result = sitofp <4 x i32> %value to <4 x float> + store <4 x float> %result, <4 x float> addrspace(1)* %out + ret void +} +; FUNC-LABEL: {{^}}v_sint_to_fp_v4i32: ; SI: v_cvt_f32_i32_e32 ; SI: v_cvt_f32_i32_e32 ; SI: v_cvt_f32_i32_e32 ; SI: v_cvt_f32_i32_e32 -define void @sint_to_fp_v4i32(<4 x float> addrspace(1)* %out, <4 x i32> addrspace(1)* %in) { - %value = load <4 x i32>, <4 x i32> addrspace(1) * %in + +; R600: INT_TO_FLT * T{{[0-9]+\.[XYZW], T[0-9]+\.[XYZW]}} +; R600: INT_TO_FLT * T{{[0-9]+\.[XYZW], T[0-9]+\.[XYZW]}} +; R600: INT_TO_FLT * T{{[0-9]+\.[XYZW], T[0-9]+\.[XYZW]}} +; R600: INT_TO_FLT * T{{[0-9]+\.[XYZW], T[0-9]+\.[XYZW]}} +define void @v_sint_to_fp_v4i32(<4 x float> addrspace(1)* %out, <4 x i32> addrspace(1)* %in) #0 { + %tid = call i32 @llvm.r600.read.tidig.x() + %in.gep = getelementptr <4 x i32>, <4 x i32> addrspace(1)* %in, i32 %tid + %out.gep = getelementptr <4 x float>, <4 x float> addrspace(1)* %out, i32 %tid + %value = load <4 x i32>, <4 x i32> addrspace(1)* %in.gep %result = sitofp <4 x i32> %value to <4 x float> - store <4 x float> %result, <4 x float> addrspace(1)* %out + store <4 x float> %result, <4 x float> addrspace(1)* %out.gep ret void } -; FUNC-LABEL: {{^}}sint_to_fp_i1_f32: +; FUNC-LABEL: {{^}}s_sint_to_fp_i1_f32: ; SI: v_cmp_eq_i32_e64 [[CMP:s\[[0-9]+:[0-9]\]]], ; SI-NEXT: v_cndmask_b32_e64 [[RESULT:v[0-9]+]], 0, 1.0, [[CMP]] ; SI: buffer_store_dword [[RESULT]], ; SI: s_endpgm -define void @sint_to_fp_i1_f32(float addrspace(1)* %out, i32 %in) { +define void @s_sint_to_fp_i1_f32(float addrspace(1)* %out, i32 %in) #0 { %cmp = icmp eq i32 %in, 0 %fp = uitofp i1 %cmp to float - store float %fp, float addrspace(1)* %out, align 4 + store float %fp, float addrspace(1)* %out ret void } -; FUNC-LABEL: {{^}}sint_to_fp_i1_f32_load: +; FUNC-LABEL: {{^}}s_sint_to_fp_i1_f32_load: ; SI: v_cndmask_b32_e64 [[RESULT:v[0-9]+]], 0, -1.0 ; SI: buffer_store_dword [[RESULT]], ; SI: s_endpgm -define void @sint_to_fp_i1_f32_load(float addrspace(1)* %out, i1 %in) { +define void @s_sint_to_fp_i1_f32_load(float addrspace(1)* %out, i1 %in) #0 { %fp = sitofp i1 %in to float - store float %fp, float addrspace(1)* %out, align 4 + store float %fp, float addrspace(1)* %out ret void } + +; FUNC-LABEL: {{^}}v_sint_to_fp_i1_f32_load: +; SI: {{buffer|flat}}_load_ubyte +; SI: v_and_b32_e32 {{v[0-9]+}}, 1, {{v[0-9]+}} +; SI: v_cmp_eq_i32 +; SI: v_cndmask_b32_e64 [[RESULT:v[0-9]+]], 0, -1.0 +; SI: {{buffer|flat}}_store_dword [[RESULT]], +; SI: s_endpgm +define void @v_sint_to_fp_i1_f32_load(float addrspace(1)* %out, i1 addrspace(1)* %in) #0 { + %tid = call i32 @llvm.r600.read.tidig.x() + %in.gep = getelementptr i1, i1 addrspace(1)* %in, i32 %tid + %out.gep = getelementptr float, float addrspace(1)* %out, i32 %tid + %val = load i1, i1 addrspace(1)* %in.gep + %fp = sitofp i1 %val to float + store float %fp, float addrspace(1)* %out.gep + ret void +} + +declare i32 @llvm.r600.read.tidig.x() #1 + +attributes #0 = { nounwind } +attributes #1 = { nounwind readnone } diff --git a/test/CodeGen/AMDGPU/udiv.ll b/test/CodeGen/AMDGPU/udiv.ll index de22a22e5029..2a09e0b20498 100644 --- a/test/CodeGen/AMDGPU/udiv.ll +++ b/test/CodeGen/AMDGPU/udiv.ll @@ -1,12 +1,11 @@ -;RUN: llc < %s -march=r600 -mcpu=redwood | FileCheck --check-prefix=EG %s -;RUN: llc < %s -march=amdgcn -mcpu=verde -verify-machineinstrs | FileCheck --check-prefix=SI %s -;RUN: llc < %s -march=amdgcn -mcpu=tonga -verify-machineinstrs | FileCheck --check-prefix=SI %s +; RUN: llc -march=r600 -mcpu=redwood < %s | FileCheck -check-prefix=EG -check-prefix=FUNC %s +; RUN: llc -march=amdgcn -mcpu=verde -verify-machineinstrs < %s | FileCheck -check-prefix=SI -check-prefix=FUNC %s +; RUN: llc -march=amdgcn -mcpu=tonga -verify-machineinstrs < %s | FileCheck -check-prefix=SI -check-prefix=FUNC %s -;EG-LABEL: {{^}}test: -;EG-NOT: SETGE_INT -;EG: CF_END - -define void @test(i32 addrspace(1)* %out, i32 addrspace(1)* %in) { +; FUNC-LABEL: {{^}}udiv_i32: +; EG-NOT: SETGE_INT +; EG: CF_END +define void @udiv_i32(i32 addrspace(1)* %out, i32 addrspace(1)* %in) { %b_ptr = getelementptr i32, i32 addrspace(1)* %in, i32 1 %a = load i32, i32 addrspace(1) * %in %b = load i32, i32 addrspace(1) * %b_ptr @@ -15,16 +14,24 @@ define void @test(i32 addrspace(1)* %out, i32 addrspace(1)* %in) { ret void } -;The code generated by udiv is long and complex and may frequently change. -;The goal of this test is to make sure the ISel doesn't fail when it gets -;a v4i32 udiv +; FUNC-LABEL: {{^}}s_udiv_i32: + +define void @s_udiv_i32(i32 addrspace(1)* %out, i32 %a, i32 %b) { + %result = udiv i32 %a, %b + store i32 %result, i32 addrspace(1)* %out + ret void +} + + +; The code generated by udiv is long and complex and may frequently +; change. The goal of this test is to make sure the ISel doesn't fail +; when it gets a v4i32 udiv -;EG-LABEL: {{^}}test2: -;EG: CF_END -;SI-LABEL: {{^}}test2: -;SI: s_endpgm +; FUNC-LABEL: {{^}}udiv_v2i32: +; EG: CF_END -define void @test2(<2 x i32> addrspace(1)* %out, <2 x i32> addrspace(1)* %in) { +; SI: s_endpgm +define void @udiv_v2i32(<2 x i32> addrspace(1)* %out, <2 x i32> addrspace(1)* %in) { %b_ptr = getelementptr <2 x i32>, <2 x i32> addrspace(1)* %in, i32 1 %a = load <2 x i32>, <2 x i32> addrspace(1) * %in %b = load <2 x i32>, <2 x i32> addrspace(1) * %b_ptr @@ -33,12 +40,10 @@ define void @test2(<2 x i32> addrspace(1)* %out, <2 x i32> addrspace(1)* %in) { ret void } -;EG-LABEL: {{^}}test4: -;EG: CF_END -;SI-LABEL: {{^}}test4: -;SI: s_endpgm - -define void @test4(<4 x i32> addrspace(1)* %out, <4 x i32> addrspace(1)* %in) { +; FUNC-LABEL: {{^}}udiv_v4i32: +; EG: CF_END +; SI: s_endpgm +define void @udiv_v4i32(<4 x i32> addrspace(1)* %out, <4 x i32> addrspace(1)* %in) { %b_ptr = getelementptr <4 x i32>, <4 x i32> addrspace(1)* %in, i32 1 %a = load <4 x i32>, <4 x i32> addrspace(1) * %in %b = load <4 x i32>, <4 x i32> addrspace(1) * %b_ptr @@ -46,3 +51,43 @@ define void @test4(<4 x i32> addrspace(1)* %out, <4 x i32> addrspace(1)* %in) { store <4 x i32> %result, <4 x i32> addrspace(1)* %out ret void } + +; FUNC-LABEL: {{^}}udiv_i32_div_pow2: +; SI: buffer_load_dword [[VAL:v[0-9]+]] +; SI: v_lshrrev_b32_e32 [[RESULT:v[0-9]+]], 4, [[VAL]] +; SI: buffer_store_dword [[RESULT]] +define void @udiv_i32_div_pow2(i32 addrspace(1)* %out, i32 addrspace(1)* %in) { + %b_ptr = getelementptr i32, i32 addrspace(1)* %in, i32 1 + %a = load i32, i32 addrspace(1)* %in + %result = udiv i32 %a, 16 + store i32 %result, i32 addrspace(1)* %out + ret void +} + +; FUNC-LABEL: {{^}}udiv_i32_div_k_even: +; SI-DAG: buffer_load_dword [[VAL:v[0-9]+]] +; SI-DAG: v_mov_b32_e32 [[K:v[0-9]+]], 0xfabbd9c1 +; SI: v_mul_hi_u32 [[MULHI:v[0-9]+]], [[K]], [[VAL]] +; SI: v_lshrrev_b32_e32 [[RESULT:v[0-9]+]], 25, [[MULHI]] +; SI: buffer_store_dword [[RESULT]] +define void @udiv_i32_div_k_even(i32 addrspace(1)* %out, i32 addrspace(1)* %in) { + %b_ptr = getelementptr i32, i32 addrspace(1)* %in, i32 1 + %a = load i32, i32 addrspace(1)* %in + %result = udiv i32 %a, 34259182 + store i32 %result, i32 addrspace(1)* %out + ret void +} + +; FUNC-LABEL: {{^}}udiv_i32_div_k_odd: +; SI-DAG: buffer_load_dword [[VAL:v[0-9]+]] +; SI-DAG: v_mov_b32_e32 [[K:v[0-9]+]], 0x7d5deca3 +; SI: v_mul_hi_u32 [[MULHI:v[0-9]+]], [[K]], [[VAL]] +; SI: v_lshrrev_b32_e32 [[RESULT:v[0-9]+]], 24, [[MULHI]] +; SI: buffer_store_dword [[RESULT]] +define void @udiv_i32_div_k_odd(i32 addrspace(1)* %out, i32 addrspace(1)* %in) { + %b_ptr = getelementptr i32, i32 addrspace(1)* %in, i32 1 + %a = load i32, i32 addrspace(1)* %in + %result = udiv i32 %a, 34259183 + store i32 %result, i32 addrspace(1)* %out + ret void +} diff --git a/test/CodeGen/AMDGPU/uint_to_fp.i64.ll b/test/CodeGen/AMDGPU/uint_to_fp.i64.ll new file mode 100644 index 000000000000..3ab11442d5cc --- /dev/null +++ b/test/CodeGen/AMDGPU/uint_to_fp.i64.ll @@ -0,0 +1,57 @@ +; RUN: llc -march=amdgcn -mcpu=SI -verify-machineinstrs < %s | FileCheck -check-prefix=GCN -check-prefix=SI -check-prefix=FUNC %s +; RUN: llc -march=amdgcn -mcpu=tonga -verify-machineinstrs < %s | FileCheck -check-prefix=GCN -check-prefix=VI -check-prefix=FUNC %s + +; FIXME: This should be merged with uint_to_fp.ll, but s_uint_to_fp_v2i64 crashes on r600 + +; FUNC-LABEL: {{^}}s_uint_to_fp_i64_to_f32: +define void @s_uint_to_fp_i64_to_f32(float addrspace(1)* %out, i64 %in) #0 { + %result = uitofp i64 %in to float + store float %result, float addrspace(1)* %out + ret void +} + +; FUNC-LABEL: {{^}}v_uint_to_fp_i64_to_f32: +; GCN: {{buffer|flat}}_load_dwordx2 + +; GCN: v_ffbh_u32 +; GCN: v_ffbh_u32 +; GCN: v_cndmask +; GCN: v_cndmask + +; GCN-DAG: v_cmp_eq_i64 +; GCN-DAG: v_cmp_lt_u64 + +; GCN: v_add_i32_e32 [[VR:v[0-9]+]] +; GCN: {{buffer|flat}}_store_dword [[VR]] +define void @v_uint_to_fp_i64_to_f32(float addrspace(1)* %out, i64 addrspace(1)* %in) #0 { + %tid = call i32 @llvm.r600.read.tidig.x() + %in.gep = getelementptr i64, i64 addrspace(1)* %in, i32 %tid + %out.gep = getelementptr float, float addrspace(1)* %out, i32 %tid + %val = load i64, i64 addrspace(1)* %in.gep + %result = uitofp i64 %val to float + store float %result, float addrspace(1)* %out.gep + ret void +} + +; FUNC-LABEL: {{^}}s_uint_to_fp_v2i64: +define void @s_uint_to_fp_v2i64(<2 x float> addrspace(1)* %out, <2 x i64> %in) #0{ + %result = uitofp <2 x i64> %in to <2 x float> + store <2 x float> %result, <2 x float> addrspace(1)* %out + ret void +} + +; FUNC-LABEL: {{^}}v_uint_to_fp_v4i64: +define void @v_uint_to_fp_v4i64(<4 x float> addrspace(1)* %out, <4 x i64> addrspace(1)* %in) #0 { + %tid = call i32 @llvm.r600.read.tidig.x() + %in.gep = getelementptr <4 x i64>, <4 x i64> addrspace(1)* %in, i32 %tid + %out.gep = getelementptr <4 x float>, <4 x float> addrspace(1)* %out, i32 %tid + %value = load <4 x i64>, <4 x i64> addrspace(1)* %in.gep + %result = uitofp <4 x i64> %value to <4 x float> + store <4 x float> %result, <4 x float> addrspace(1)* %out.gep + ret void +} + +declare i32 @llvm.r600.read.tidig.x() #1 + +attributes #0 = { nounwind } +attributes #1 = { nounwind readnone } diff --git a/test/CodeGen/AMDGPU/uint_to_fp.ll b/test/CodeGen/AMDGPU/uint_to_fp.ll index 00fea80b1bc8..a3343d1e2d9c 100644 --- a/test/CodeGen/AMDGPU/uint_to_fp.ll +++ b/test/CodeGen/AMDGPU/uint_to_fp.ll @@ -2,81 +2,138 @@ ; RUN: llc -march=amdgcn -mcpu=tonga -verify-machineinstrs < %s | FileCheck -check-prefix=SI -check-prefix=FUNC %s ; RUN: llc -march=r600 -mcpu=redwood < %s | FileCheck -check-prefix=R600 -check-prefix=FUNC %s -; FUNC-LABEL: {{^}}uint_to_fp_i32_to_f32: -; R600-DAG: UINT_TO_FLT * T{{[0-9]+\.[XYZW]}}, KC0[2].Z - +; FUNC-LABEL: {{^}}s_uint_to_fp_i32_to_f32: ; SI: v_cvt_f32_u32_e32 -; SI: s_endpgm -define void @uint_to_fp_i32_to_f32(float addrspace(1)* %out, i32 %in) { + +; R600: UINT_TO_FLT * T{{[0-9]+\.[XYZW]}}, KC0[2].Z +define void @s_uint_to_fp_i32_to_f32(float addrspace(1)* %out, i32 %in) #0 { %result = uitofp i32 %in to float store float %result, float addrspace(1)* %out ret void } -; FUNC-LABEL: {{^}}uint_to_fp_v2i32_to_v2f32: -; R600-DAG: UINT_TO_FLT * T{{[0-9]+\.[XYZW]}}, KC0[2].W -; R600-DAG: UINT_TO_FLT * T{{[0-9]+\.[XYZW]}}, KC0[3].X +; FUNC-LABEL: {{^}}v_uint_to_fp_i32_to_f32: +; SI: v_cvt_f32_u32_e32 {{v[0-9]+}}, {{v[0-9]+$}} + +; R600: INT_TO_FLT +define void @v_uint_to_fp_i32_to_f32(float addrspace(1)* %out, i32 addrspace(1)* %in) #0 { + %tid = call i32 @llvm.r600.read.tidig.x() + %in.gep = getelementptr i32, i32 addrspace(1)* %in, i32 %tid + %out.gep = getelementptr float, float addrspace(1)* %out, i32 %tid + %val = load i32, i32 addrspace(1)* %in.gep + %result = uitofp i32 %val to float + store float %result, float addrspace(1)* %out.gep + ret void +} +; FUNC-LABEL: {{^}}s_uint_to_fp_v2i32_to_v2f32: ; SI: v_cvt_f32_u32_e32 ; SI: v_cvt_f32_u32_e32 -; SI: s_endpgm -define void @uint_to_fp_v2i32_to_v2f32(<2 x float> addrspace(1)* %out, <2 x i32> %in) { + +; R600-DAG: UINT_TO_FLT * T{{[0-9]+\.[XYZW]}}, KC0[2].W +; R600-DAG: UINT_TO_FLT * T{{[0-9]+\.[XYZW]}}, KC0[3].X +define void @s_uint_to_fp_v2i32_to_v2f32(<2 x float> addrspace(1)* %out, <2 x i32> %in) #0 { %result = uitofp <2 x i32> %in to <2 x float> store <2 x float> %result, <2 x float> addrspace(1)* %out ret void } -; FUNC-LABEL: {{^}}uint_to_fp_v4i32_to_v4f32: -; R600: UINT_TO_FLT * T{{[0-9]+\.[XYZW], T[0-9]+\.[XYZW]}} -; R600: UINT_TO_FLT * T{{[0-9]+\.[XYZW], T[0-9]+\.[XYZW]}} -; R600: UINT_TO_FLT * T{{[0-9]+\.[XYZW], T[0-9]+\.[XYZW]}} -; R600: UINT_TO_FLT * T{{[0-9]+\.[XYZW], T[0-9]+\.[XYZW]}} - +; FUNC-LABEL: {{^}}s_uint_to_fp_v4i32_to_v4f32: ; SI: v_cvt_f32_u32_e32 ; SI: v_cvt_f32_u32_e32 ; SI: v_cvt_f32_u32_e32 ; SI: v_cvt_f32_u32_e32 ; SI: s_endpgm -define void @uint_to_fp_v4i32_to_v4f32(<4 x float> addrspace(1)* %out, <4 x i32> addrspace(1)* %in) { + +; R600: UINT_TO_FLT * T{{[0-9]+\.[XYZW], T[0-9]+\.[XYZW]}} +; R600: UINT_TO_FLT * T{{[0-9]+\.[XYZW], T[0-9]+\.[XYZW]}} +; R600: UINT_TO_FLT * T{{[0-9]+\.[XYZW], T[0-9]+\.[XYZW]}} +; R600: UINT_TO_FLT * T{{[0-9]+\.[XYZW], T[0-9]+\.[XYZW]}} +define void @s_uint_to_fp_v4i32_to_v4f32(<4 x float> addrspace(1)* %out, <4 x i32> addrspace(1)* %in) #0 { %value = load <4 x i32>, <4 x i32> addrspace(1) * %in %result = uitofp <4 x i32> %value to <4 x float> store <4 x float> %result, <4 x float> addrspace(1)* %out ret void } -; FUNC-LABEL: {{^}}uint_to_fp_i64_to_f32: -; R600: UINT_TO_FLT -; R600: UINT_TO_FLT -; R600: MULADD_IEEE +; FUNC-LABEL: {{^}}v_uint_to_fp_v4i32: ; SI: v_cvt_f32_u32_e32 ; SI: v_cvt_f32_u32_e32 -; SI: v_madmk_f32_e32 {{v[0-9]+}}, {{v[0-9]+}}, {{v[0-9]+}}, 0x4f800000 -; SI: s_endpgm -define void @uint_to_fp_i64_to_f32(float addrspace(1)* %out, i64 %in) { -entry: - %0 = uitofp i64 %in to float - store float %0, float addrspace(1)* %out +; SI: v_cvt_f32_u32_e32 +; SI: v_cvt_f32_u32_e32 + +; R600: UINT_TO_FLT * T{{[0-9]+\.[XYZW], T[0-9]+\.[XYZW]}} +; R600: UINT_TO_FLT * T{{[0-9]+\.[XYZW], T[0-9]+\.[XYZW]}} +; R600: UINT_TO_FLT * T{{[0-9]+\.[XYZW], T[0-9]+\.[XYZW]}} +; R600: UINT_TO_FLT * T{{[0-9]+\.[XYZW], T[0-9]+\.[XYZW]}} +define void @v_uint_to_fp_v4i32(<4 x float> addrspace(1)* %out, <4 x i32> addrspace(1)* %in) #0 { + %tid = call i32 @llvm.r600.read.tidig.x() + %in.gep = getelementptr <4 x i32>, <4 x i32> addrspace(1)* %in, i32 %tid + %out.gep = getelementptr <4 x float>, <4 x float> addrspace(1)* %out, i32 %tid + %value = load <4 x i32>, <4 x i32> addrspace(1)* %in.gep + %result = uitofp <4 x i32> %value to <4 x float> + store <4 x float> %result, <4 x float> addrspace(1)* %out.gep ret void } -; FUNC-LABEL: {{^}}uint_to_fp_i1_to_f32: +; FUNC-LABEL: {{^}}s_uint_to_fp_i1_to_f32: ; SI: v_cmp_eq_i32_e64 [[CMP:s\[[0-9]+:[0-9]\]]], ; SI-NEXT: v_cndmask_b32_e64 [[RESULT:v[0-9]+]], 0, 1.0, [[CMP]] ; SI: buffer_store_dword [[RESULT]], ; SI: s_endpgm -define void @uint_to_fp_i1_to_f32(float addrspace(1)* %out, i32 %in) { +define void @s_uint_to_fp_i1_to_f32(float addrspace(1)* %out, i32 %in) #0 { %cmp = icmp eq i32 %in, 0 %fp = uitofp i1 %cmp to float - store float %fp, float addrspace(1)* %out, align 4 + store float %fp, float addrspace(1)* %out ret void } -; FUNC-LABEL: {{^}}uint_to_fp_i1_to_f32_load: +; FUNC-LABEL: {{^}}s_uint_to_fp_i1_to_f32_load: ; SI: v_cndmask_b32_e64 [[RESULT:v[0-9]+]], 0, 1.0 ; SI: buffer_store_dword [[RESULT]], ; SI: s_endpgm -define void @uint_to_fp_i1_to_f32_load(float addrspace(1)* %out, i1 %in) { +define void @s_uint_to_fp_i1_to_f32_load(float addrspace(1)* %out, i1 %in) #0 { %fp = uitofp i1 %in to float - store float %fp, float addrspace(1)* %out, align 4 + store float %fp, float addrspace(1)* %out + ret void +} + +; FUNC-LABEL: {{^}}v_uint_to_fp_i1_f32_load: +; SI: {{buffer|flat}}_load_ubyte +; SI: v_and_b32_e32 {{v[0-9]+}}, 1, {{v[0-9]+}} +; SI: v_cmp_eq_i32 +; SI: v_cndmask_b32_e64 [[RESULT:v[0-9]+]], 0, 1.0 +; SI: {{buffer|flat}}_store_dword [[RESULT]], +; SI: s_endpgm +define void @v_uint_to_fp_i1_f32_load(float addrspace(1)* %out, i1 addrspace(1)* %in) #0 { + %tid = call i32 @llvm.r600.read.tidig.x() + %in.gep = getelementptr i1, i1 addrspace(1)* %in, i32 %tid + %out.gep = getelementptr float, float addrspace(1)* %out, i32 %tid + %val = load i1, i1 addrspace(1)* %in.gep + %fp = uitofp i1 %val to float + store float %fp, float addrspace(1)* %out.gep ret void } + +; FIXME: Repeated here to test r600 +; FUNC-LABEL: {{^}}s_uint_to_fp_i64_to_f32: +; R600: FFBH_UINT +; R600: FFBH_UINT +; R600: CNDE_INT +; R600: CNDE_INT + +; R600-DAG: SETGT_UINT +; R600-DAG: SETGT_UINT +; R600-DAG: SETE_INT + +define void @s_uint_to_fp_i64_to_f32(float addrspace(1)* %out, i64 %in) #0 { +entry: + %cvt = uitofp i64 %in to float + store float %cvt, float addrspace(1)* %out + ret void +} + +declare i32 @llvm.r600.read.tidig.x() #1 + +attributes #0 = { nounwind } +attributes #1 = { nounwind readnone } |
