diff options
Diffstat (limited to 'test/Transforms/EarlyCSE')
| -rw-r--r-- | test/Transforms/EarlyCSE/AArch64/ldstN.ll | 18 | ||||
| -rw-r--r-- | test/Transforms/EarlyCSE/atomics.ll | 259 | ||||
| -rw-r--r-- | test/Transforms/EarlyCSE/basic.ll | 74 | ||||
| -rw-r--r-- | test/Transforms/EarlyCSE/fence.ll | 86 |
4 files changed, 437 insertions, 0 deletions
diff --git a/test/Transforms/EarlyCSE/AArch64/ldstN.ll b/test/Transforms/EarlyCSE/AArch64/ldstN.ll new file mode 100644 index 000000000000..cc1af31429e1 --- /dev/null +++ b/test/Transforms/EarlyCSE/AArch64/ldstN.ll @@ -0,0 +1,18 @@ +; RUN: opt -S -early-cse < %s | FileCheck %s
+target datalayout = "e-m:e-i64:64-i128:128-n32:64-S128"
+target triple = "aarch64--linux-gnu"
+
+declare { <4 x i16>, <4 x i16>, <4 x i16>, <4 x i16> } @llvm.aarch64.neon.ld4.v4i16.p0v4i16(<4 x i16>*)
+
+; Although the store and the ld4 are using the same pointer, the
+; data can not be reused because ld4 accesses multiple elements.
+define { <4 x i16>, <4 x i16>, <4 x i16>, <4 x i16> } @foo() {
+entry:
+ store <4 x i16> undef, <4 x i16>* undef, align 8
+ %0 = call { <4 x i16>, <4 x i16>, <4 x i16>, <4 x i16> } @llvm.aarch64.neon.ld4.v4i16.p0v4i16(<4 x i16>* undef)
+ ret { <4 x i16>, <4 x i16>, <4 x i16>, <4 x i16> } %0
+; CHECK-LABEL: @foo(
+; CHECK: store
+; CHECK-NEXT: call
+; CHECK-NEXT: ret
+}
diff --git a/test/Transforms/EarlyCSE/atomics.ll b/test/Transforms/EarlyCSE/atomics.ll new file mode 100644 index 000000000000..21c19cd8e880 --- /dev/null +++ b/test/Transforms/EarlyCSE/atomics.ll @@ -0,0 +1,259 @@ +; RUN: opt < %s -S -early-cse | FileCheck %s + +; CHECK-LABEL: @test12( +define i32 @test12(i1 %B, i32* %P1, i32* %P2) { + %load0 = load i32, i32* %P1 + %1 = load atomic i32, i32* %P2 seq_cst, align 4 + %load1 = load i32, i32* %P1 + %sel = select i1 %B, i32 %load0, i32 %load1 + ret i32 %sel + ; CHECK: load i32, i32* %P1 + ; CHECK: load i32, i32* %P1 +} + +; CHECK-LABEL: @test13( +; atomic to non-atomic forwarding is legal +define i32 @test13(i1 %B, i32* %P1) { + %a = load atomic i32, i32* %P1 seq_cst, align 4 + %b = load i32, i32* %P1 + %res = sub i32 %a, %b + ret i32 %res + ; CHECK: load atomic i32, i32* %P1 + ; CHECK: ret i32 0 +} + +; CHECK-LABEL: @test14( +; atomic to unordered atomic forwarding is legal +define i32 @test14(i1 %B, i32* %P1) { + %a = load atomic i32, i32* %P1 seq_cst, align 4 + %b = load atomic i32, i32* %P1 unordered, align 4 + %res = sub i32 %a, %b + ret i32 %res + ; CHECK: load atomic i32, i32* %P1 seq_cst + ; CHECK-NEXT: ret i32 0 +} + +; CHECK-LABEL: @test15( +; implementation restriction: can't forward to stonger +; than unordered +define i32 @test15(i1 %B, i32* %P1, i32* %P2) { + %a = load atomic i32, i32* %P1 seq_cst, align 4 + %b = load atomic i32, i32* %P1 seq_cst, align 4 + %res = sub i32 %a, %b + ret i32 %res + ; CHECK: load atomic i32, i32* %P1 + ; CHECK: load atomic i32, i32* %P1 +} + +; CHECK-LABEL: @test16( +; forwarding non-atomic to atomic is wrong! (However, +; it would be legal to use the later value in place of the +; former in this particular example. We just don't +; do that right now.) +define i32 @test16(i1 %B, i32* %P1, i32* %P2) { + %a = load i32, i32* %P1, align 4 + %b = load atomic i32, i32* %P1 unordered, align 4 + %res = sub i32 %a, %b + ret i32 %res + ; CHECK: load i32, i32* %P1 + ; CHECK: load atomic i32, i32* %P1 +} + +; Can't DSE across a full fence +define void @fence_seq_cst_store(i1 %B, i32* %P1, i32* %P2) { +; CHECK-LABEL: @fence_seq_cst_store +; CHECK: store +; CHECK: store atomic +; CHECK: store + store i32 0, i32* %P1, align 4 + store atomic i32 0, i32* %P2 seq_cst, align 4 + store i32 0, i32* %P1, align 4 + ret void +} + +; Can't DSE across a full fence +define void @fence_seq_cst(i1 %B, i32* %P1, i32* %P2) { +; CHECK-LABEL: @fence_seq_cst +; CHECK: store +; CHECK: fence seq_cst +; CHECK: store + store i32 0, i32* %P1, align 4 + fence seq_cst + store i32 0, i32* %P1, align 4 + ret void +} + +; Can't DSE across a full fence +define void @fence_asm_sideeffect(i1 %B, i32* %P1, i32* %P2) { +; CHECK-LABEL: @fence_asm_sideeffect +; CHECK: store +; CHECK: call void asm sideeffect +; CHECK: store + store i32 0, i32* %P1, align 4 + call void asm sideeffect "", ""() + store i32 0, i32* %P1, align 4 + ret void +} + +; Can't DSE across a full fence +define void @fence_asm_memory(i1 %B, i32* %P1, i32* %P2) { +; CHECK-LABEL: @fence_asm_memory +; CHECK: store +; CHECK: call void asm +; CHECK: store + store i32 0, i32* %P1, align 4 + call void asm "", "~{memory}"() + store i32 0, i32* %P1, align 4 + ret void +} + +; Can't remove a volatile load +define i32 @volatile_load(i1 %B, i32* %P1, i32* %P2) { + %a = load i32, i32* %P1, align 4 + %b = load volatile i32, i32* %P1, align 4 + %res = sub i32 %a, %b + ret i32 %res + ; CHECK-LABEL: @volatile_load + ; CHECK: load i32, i32* %P1 + ; CHECK: load volatile i32, i32* %P1 +} + +; Can't remove redundant volatile loads +define i32 @redundant_volatile_load(i1 %B, i32* %P1, i32* %P2) { + %a = load volatile i32, i32* %P1, align 4 + %b = load volatile i32, i32* %P1, align 4 + %res = sub i32 %a, %b + ret i32 %res + ; CHECK-LABEL: @redundant_volatile_load + ; CHECK: load volatile i32, i32* %P1 + ; CHECK: load volatile i32, i32* %P1 + ; CHECK: sub +} + +; Can't DSE a volatile store +define void @volatile_store(i1 %B, i32* %P1, i32* %P2) { +; CHECK-LABEL: @volatile_store +; CHECK: store volatile +; CHECK: store + store volatile i32 0, i32* %P1, align 4 + store i32 3, i32* %P1, align 4 + ret void +} + +; Can't DSE a redundant volatile store +define void @redundant_volatile_store(i1 %B, i32* %P1, i32* %P2) { +; CHECK-LABEL: @redundant_volatile_store +; CHECK: store volatile +; CHECK: store volatile + store volatile i32 0, i32* %P1, align 4 + store volatile i32 0, i32* %P1, align 4 + ret void +} + +; Can value forward from volatiles +define i32 @test20(i1 %B, i32* %P1, i32* %P2) { + %a = load volatile i32, i32* %P1, align 4 + %b = load i32, i32* %P1, align 4 + %res = sub i32 %a, %b + ret i32 %res + ; CHECK-LABEL: @test20 + ; CHECK: load volatile i32, i32* %P1 + ; CHECK: ret i32 0 +} + +; Can DSE a non-volatile store in favor of a volatile one +; currently a missed optimization +define void @test21(i1 %B, i32* %P1, i32* %P2) { +; CHECK-LABEL: @test21 +; CHECK: store +; CHECK: store volatile + store i32 0, i32* %P1, align 4 + store volatile i32 3, i32* %P1, align 4 + ret void +} + +; Can DSE a normal store in favor of a unordered one +define void @test22(i1 %B, i32* %P1, i32* %P2) { +; CHECK-LABEL: @test22 +; CHECK-NEXT: store atomic + store i32 0, i32* %P1, align 4 + store atomic i32 3, i32* %P1 unordered, align 4 + ret void +} + +; Can also DSE a unordered store in favor of a normal one +define void @test23(i1 %B, i32* %P1, i32* %P2) { +; CHECK-LABEL: @test23 +; CHECK-NEXT: store i32 0 + store atomic i32 3, i32* %P1 unordered, align 4 + store i32 0, i32* %P1, align 4 + ret void +} + +; As an implementation limitation, can't remove ordered stores +; Note that we could remove the earlier store if we could +; represent the required ordering. +define void @test24(i1 %B, i32* %P1, i32* %P2) { +; CHECK-LABEL: @test24 +; CHECK-NEXT: store atomic +; CHECK-NEXT: store i32 0 + store atomic i32 3, i32* %P1 release, align 4 + store i32 0, i32* %P1, align 4 + ret void +} + +; Can't remove volatile stores - each is independently observable and +; the count of such stores is an observable program side effect. +define void @test25(i1 %B, i32* %P1, i32* %P2) { +; CHECK-LABEL: @test25 +; CHECK-NEXT: store volatile +; CHECK-NEXT: store volatile + store volatile i32 3, i32* %P1, align 4 + store volatile i32 0, i32* %P1, align 4 + ret void +} + +; Can DSE a unordered store in favor of a unordered one +define void @test26(i1 %B, i32* %P1, i32* %P2) { +; CHECK-LABEL: @test26 +; CHECK-NEXT: store atomic i32 3, i32* %P1 unordered, align 4 +; CHECK-NEXT: ret + store atomic i32 0, i32* %P1 unordered, align 4 + store atomic i32 3, i32* %P1 unordered, align 4 + ret void +} + +; Can DSE a unordered store in favor of a ordered one, +; but current don't due to implementation limits +define void @test27(i1 %B, i32* %P1, i32* %P2) { +; CHECK-LABEL: @test27 +; CHECK-NEXT: store atomic i32 0, i32* %P1 unordered, align 4 +; CHECK-NEXT: store atomic i32 3, i32* %P1 release, align 4 +; CHECK-NEXT: ret + store atomic i32 0, i32* %P1 unordered, align 4 + store atomic i32 3, i32* %P1 release, align 4 + ret void +} + +; Can DSE an unordered atomic store in favor of an +; ordered one, but current don't due to implementation limits +define void @test28(i1 %B, i32* %P1, i32* %P2) { +; CHECK-LABEL: @test28 +; CHECK-NEXT: store atomic i32 0, i32* %P1 unordered, align 4 +; CHECK-NEXT: store atomic i32 3, i32* %P1 release, align 4 +; CHECK-NEXT: ret + store atomic i32 0, i32* %P1 unordered, align 4 + store atomic i32 3, i32* %P1 release, align 4 + ret void +} + +; As an implementation limitation, can't remove ordered stores +; see also: @test24 +define void @test29(i1 %B, i32* %P1, i32* %P2) { +; CHECK-LABEL: @test29 +; CHECK-NEXT: store atomic +; CHECK-NEXT: store atomic + store atomic i32 3, i32* %P1 release, align 4 + store atomic i32 0, i32* %P1 unordered, align 4 + ret void +} diff --git a/test/Transforms/EarlyCSE/basic.ll b/test/Transforms/EarlyCSE/basic.ll index 43b5e6098f6a..8c9b74b4d0e1 100644 --- a/test/Transforms/EarlyCSE/basic.ll +++ b/test/Transforms/EarlyCSE/basic.ll @@ -203,3 +203,77 @@ define i32 @test12(i1 %B, i32* %P1, i32* %P2) { ; CHECK: load i32, i32* %P1 ; CHECK: load i32, i32* %P1 } + +define void @dse1(i32 *%P) { +; CHECK-LABEL: @dse1 +; CHECK-NOT: store + %v = load i32, i32* %P + store i32 %v, i32* %P + ret void +} + +define void @dse2(i32 *%P) { +; CHECK-LABEL: @dse2 +; CHECK-NOT: store + %v = load atomic i32, i32* %P seq_cst, align 4 + store i32 %v, i32* %P + ret void +} + +define void @dse3(i32 *%P) { +; CHECK-LABEL: @dse3 +; CHECK-NOT: store + %v = load atomic i32, i32* %P seq_cst, align 4 + store atomic i32 %v, i32* %P unordered, align 4 + ret void +} + +define i32 @dse4(i32 *%P, i32 *%Q) { +; CHECK-LABEL: @dse4 +; CHECK-NOT: store +; CHECK: ret i32 0 + %a = load i32, i32* %Q + %v = load atomic i32, i32* %P unordered, align 4 + store atomic i32 %v, i32* %P unordered, align 4 + %b = load i32, i32* %Q + %res = sub i32 %a, %b + ret i32 %res +} + +; Note that in this example, %P and %Q could in fact be the same +; pointer. %v could be different than the value observed for %a +; and that's okay because we're using relaxed memory ordering. +; The only guarantee we have to provide is that each of the loads +; has to observe some value written to that location. We do +; not have to respect the order in which those writes were done. +define i32 @dse5(i32 *%P, i32 *%Q) { +; CHECK-LABEL: @dse5 +; CHECK-NOT: store +; CHECK: ret i32 0 + %v = load atomic i32, i32* %P unordered, align 4 + %a = load atomic i32, i32* %Q unordered, align 4 + store atomic i32 %v, i32* %P unordered, align 4 + %b = load atomic i32, i32* %Q unordered, align 4 + %res = sub i32 %a, %b + ret i32 %res +} + + +define void @dse_neg1(i32 *%P) { +; CHECK-LABEL: @dse_neg1 +; CHECK: store + %v = load i32, i32* %P + store i32 5, i32* %P + ret void +} + +; Could remove the store, but only if ordering was somehow +; encoded. +define void @dse_neg2(i32 *%P) { +; CHECK-LABEL: @dse_neg2 +; CHECK: store + %v = load i32, i32* %P + store atomic i32 %v, i32* %P seq_cst, align 4 + ret void +} + diff --git a/test/Transforms/EarlyCSE/fence.ll b/test/Transforms/EarlyCSE/fence.ll new file mode 100644 index 000000000000..c6d47e9fb22e --- /dev/null +++ b/test/Transforms/EarlyCSE/fence.ll @@ -0,0 +1,86 @@ +; RUN: opt -S -early-cse < %s | FileCheck %s +; NOTE: This file is testing the current implementation. Some of +; the transforms used as negative tests below would be legal, but +; only if reached through a chain of logic which EarlyCSE is incapable +; of performing. To say it differently, this file tests a conservative +; version of the memory model. If we want to extend EarlyCSE to be more +; aggressive in the future, we may need to relax some of the negative tests. + +; We can value forward across the fence since we can (semantically) +; reorder the following load before the fence. +define i32 @test(i32* %addr.i) { +; CHECK-LABEL: @test +; CHECK: store +; CHECK: fence +; CHECK-NOT: load +; CHECK: ret + store i32 5, i32* %addr.i, align 4 + fence release + %a = load i32, i32* %addr.i, align 4 + ret i32 %a +} + +; Same as above +define i32 @test2(i32* noalias %addr.i, i32* noalias %otheraddr) { +; CHECK-LABEL: @test2 +; CHECK: load +; CHECK: fence +; CHECK-NOT: load +; CHECK: ret + %a = load i32, i32* %addr.i, align 4 + fence release + %a2 = load i32, i32* %addr.i, align 4 + %res = sub i32 %a, %a2 + ret i32 %a +} + +; We can not value forward across an acquire barrier since we might +; be syncronizing with another thread storing to the same variable +; followed by a release fence. If this thread observed the release +; had happened, we must present a consistent view of memory at the +; fence. Note that it would be legal to reorder '%a' after the fence +; and then remove '%a2'. The current implementation doesn't know how +; to do this, but if it learned, this test will need revised. +define i32 @test3(i32* noalias %addr.i, i32* noalias %otheraddr) { +; CHECK-LABEL: @test3 +; CHECK: load +; CHECK: fence +; CHECK: load +; CHECK: sub +; CHECK: ret + %a = load i32, i32* %addr.i, align 4 + fence acquire + %a2 = load i32, i32* %addr.i, align 4 + %res = sub i32 %a, %a2 + ret i32 %res +} + +; We can not dead store eliminate accross the fence. We could in +; principal reorder the second store above the fence and then DSE either +; store, but this is beyond the simple last-store DSE which EarlyCSE +; implements. +define void @test4(i32* %addr.i) { +; CHECK-LABEL: @test4 +; CHECK: store +; CHECK: fence +; CHECK: store +; CHECK: ret + store i32 5, i32* %addr.i, align 4 + fence release + store i32 5, i32* %addr.i, align 4 + ret void +} + +; We *could* DSE across this fence, but don't. No other thread can +; observe the order of the acquire fence and the store. +define void @test5(i32* %addr.i) { +; CHECK-LABEL: @test5 +; CHECK: store +; CHECK: fence +; CHECK: store +; CHECK: ret + store i32 5, i32* %addr.i, align 4 + fence acquire + store i32 5, i32* %addr.i, align 4 + ret void +} |
