aboutsummaryrefslogtreecommitdiffstats
path: root/test/Transforms
diff options
context:
space:
mode:
authorChris Lattner <sabre@nondot.org>2008-12-15 03:35:32 +0000
committerChris Lattner <sabre@nondot.org>2008-12-15 03:35:32 +0000
commit9e59c64c14cfe55e7cc9086c6bff8cfeecac361e (patch)
treebeb35f47a4da2768af1f2b6aecbcd93c0ea438b4 /test/Transforms
parent255dafc5de1e0e15ccebc9f5947330833008374a (diff)
downloadexternal_llvm-9e59c64c14cfe55e7cc9086c6bff8cfeecac361e.zip
external_llvm-9e59c64c14cfe55e7cc9086c6bff8cfeecac361e.tar.gz
external_llvm-9e59c64c14cfe55e7cc9086c6bff8cfeecac361e.tar.bz2
Implement initial support for PHI translation in memdep. This means that
memdep keeps track of how PHIs affect the pointer in dep queries, which allows it to eliminate the load in cases like rle-phi-translate.ll, which basically end up being: BB1: X = load P br BB3 BB2: Y = load Q br BB3 BB3: R = phi [P] [Q] load R turning "load R" into a phi of X/Y. In addition to additional exposed opportunities, this makes memdep safe in many cases that it wasn't before (which is required for load PRE) and also makes it substantially more efficient. For example, consider: bb1: // has many predecessors. P = some_operator() load P In this example, previously memdep would scan all the predecessors of BB1 to see if they had something that would mustalias P. In some cases (e.g. test/Transforms/GVN/rle-must-alias.ll) it would actually find them and end up eliminating something. In many other cases though, it would scan and not find anything useful. MemDep now stops at a block if the pointer is defined in that block and cannot be phi translated to predecessors. This causes it to miss the (rare) cases like rle-must-alias.ll, but makes it faster by not scanning tons of stuff that is unlikely to be useful. For example, this speeds up GVN as a whole from 3.928s to 2.448s (60%)!. IMO, scalar GVN should be enhanced to simplify the rle-must-alias pointer base anyway, which would allow the loads to be eliminated. In the future, this should be enhanced to phi translate through geps and bitcasts as well (as indicated by FIXMEs) making memdep even more powerful. git-svn-id: https://llvm.org/svn/llvm-project/llvm/trunk@61022 91177308-0d34-0410-b5e6-96231b3b80d8
Diffstat (limited to 'test/Transforms')
-rw-r--r--test/Transforms/GVN/rle-must-alias.ll5
-rw-r--r--test/Transforms/GVN/rle-no-phi-translate.ll3
-rw-r--r--test/Transforms/GVN/rle-phi-translate.ll32
3 files changed, 40 insertions, 0 deletions
diff --git a/test/Transforms/GVN/rle-must-alias.ll b/test/Transforms/GVN/rle-must-alias.ll
index e507556..7108916 100644
--- a/test/Transforms/GVN/rle-must-alias.ll
+++ b/test/Transforms/GVN/rle-must-alias.ll
@@ -1,4 +1,9 @@
; RUN: llvm-as < %s | opt -gvn | llvm-dis | grep {DEAD.rle = phi i32}
+; XFAIL: *
+
+; FIXME: GVN should eliminate the fully redundant %9 GEP which
+; allows DEAD to be removed. This is PR3198.
+
; The %7 and %4 loads combine to make %DEAD unneeded.
target datalayout = "e-p:32:32:32-i1:8:8-i8:8:8-i16:16:16-i32:32:32-i64:32:64-f32:32:32-f64:32:64-v64:64:64-v128:128:128-a0:0:64-f80:128:128"
target triple = "i386-apple-darwin7"
diff --git a/test/Transforms/GVN/rle-no-phi-translate.ll b/test/Transforms/GVN/rle-no-phi-translate.ll
index b7f0dbc..9ffbe21 100644
--- a/test/Transforms/GVN/rle-no-phi-translate.ll
+++ b/test/Transforms/GVN/rle-no-phi-translate.ll
@@ -1,4 +1,7 @@
; RUN: llvm-as < %s | opt -gvn | llvm-dis | grep load
+; FIXME: This should be promotable, but memdep/gvn don't track values
+; path/edge sensitively enough.
+
target datalayout = "e-p:32:32:32-i1:8:8-i8:8:8-i16:16:16-i32:32:32-i64:32:64-f32:32:32-f64:32:64-v64:64:64-v128:128:128-a0:0:64-f80:128:128"
target triple = "i386-apple-darwin7"
diff --git a/test/Transforms/GVN/rle-phi-translate.ll b/test/Transforms/GVN/rle-phi-translate.ll
new file mode 100644
index 0000000..0f9c58f
--- /dev/null
+++ b/test/Transforms/GVN/rle-phi-translate.ll
@@ -0,0 +1,32 @@
+; RUN: llvm-as < %s | opt -gvn | llvm-dis | grep {%cv.rle = phi i32}
+; RUN: llvm-as < %s | opt -gvn | llvm-dis | grep {%bv.rle = phi i32}
+target datalayout = "e-p:32:32:32-i1:8:8-i8:8:8-i16:16:16-i32:32:32-i64:32:64-f32:32:32-f64:32:64-v64:64:64-v128:128:128-a0:0:64-f80:128:128"
+target triple = "i386-apple-darwin7"
+
+define i32 @g(i32* %b, i32* %c) nounwind {
+entry:
+ %g = alloca i32 ; <i32*> [#uses=4]
+ %t1 = icmp eq i32* %b, null ; <i1> [#uses=1]
+ br i1 %t1, label %bb, label %bb1
+
+bb: ; preds = %entry
+ %t2 = load i32* %c, align 4 ; <i32> [#uses=1]
+ %t3 = add i32 %t2, 1 ; <i32> [#uses=1]
+ store i32 %t3, i32* %g, align 4
+ br label %bb2
+
+bb1: ; preds = %entry
+ %t5 = load i32* %b, align 4 ; <i32> [#uses=1]
+ %t6 = add i32 %t5, 1 ; <i32> [#uses=1]
+ store i32 %t6, i32* %g, align 4
+ br label %bb2
+
+bb2: ; preds = %bb1, %bb
+ %c_addr.0 = phi i32* [ %g, %bb1 ], [ %c, %bb ] ; <i32*> [#uses=1]
+ %b_addr.0 = phi i32* [ %b, %bb1 ], [ %g, %bb ] ; <i32*> [#uses=1]
+ %cv = load i32* %c_addr.0, align 4 ; <i32> [#uses=1]
+ %bv = load i32* %b_addr.0, align 4 ; <i32> [#uses=1]
+ %ret = add i32 %cv, %bv ; <i32> [#uses=1]
+ ret i32 %ret
+}
+