2009-11-27 07:32:59 +08:00
|
|
|
; RUN: opt < %s -gvn -S | FileCheck %s
|
|
|
|
|
Implement initial support for PHI translation in memdep. This means that
memdep keeps track of how PHIs affect the pointer in dep queries, which
allows it to eliminate the load in cases like rle-phi-translate.ll, which
basically end up being:
BB1:
X = load P
br BB3
BB2:
Y = load Q
br BB3
BB3:
R = phi [P] [Q]
load R
turning "load R" into a phi of X/Y. In addition to additional exposed
opportunities, this makes memdep safe in many cases that it wasn't before
(which is required for load PRE) and also makes it substantially more
efficient. For example, consider:
bb1: // has many predecessors.
P = some_operator()
load P
In this example, previously memdep would scan all the predecessors of BB1
to see if they had something that would mustalias P. In some cases (e.g.
test/Transforms/GVN/rle-must-alias.ll) it would actually find them and end
up eliminating something. In many other cases though, it would scan and not
find anything useful. MemDep now stops at a block if the pointer is defined
in that block and cannot be phi translated to predecessors. This causes it
to miss the (rare) cases like rle-must-alias.ll, but makes it faster by not
scanning tons of stuff that is unlikely to be useful. For example, this
speeds up GVN as a whole from 3.928s to 2.448s (60%)!. IMO, scalar GVN
should be enhanced to simplify the rle-must-alias pointer base anyway, which
would allow the loads to be eliminated.
In the future, this should be enhanced to phi translate through geps and
bitcasts as well (as indicated by FIXMEs) making memdep even more powerful.
llvm-svn: 61022
2008-12-15 11:35:32 +08:00
|
|
|
target datalayout = "e-p:32:32:32-i1:8:8-i8:8:8-i16:16:16-i32:32:32-i64:32:64-f32:32:32-f64:32:64-v64:64:64-v128:128:128-a0:0:64-f80:128:128"
|
|
|
|
target triple = "i386-apple-darwin7"
|
|
|
|
|
2009-11-27 07:32:59 +08:00
|
|
|
define i32 @test1(i32* %b, i32* %c) nounwind {
|
2009-11-27 07:41:07 +08:00
|
|
|
; CHECK: @test1
|
Implement initial support for PHI translation in memdep. This means that
memdep keeps track of how PHIs affect the pointer in dep queries, which
allows it to eliminate the load in cases like rle-phi-translate.ll, which
basically end up being:
BB1:
X = load P
br BB3
BB2:
Y = load Q
br BB3
BB3:
R = phi [P] [Q]
load R
turning "load R" into a phi of X/Y. In addition to additional exposed
opportunities, this makes memdep safe in many cases that it wasn't before
(which is required for load PRE) and also makes it substantially more
efficient. For example, consider:
bb1: // has many predecessors.
P = some_operator()
load P
In this example, previously memdep would scan all the predecessors of BB1
to see if they had something that would mustalias P. In some cases (e.g.
test/Transforms/GVN/rle-must-alias.ll) it would actually find them and end
up eliminating something. In many other cases though, it would scan and not
find anything useful. MemDep now stops at a block if the pointer is defined
in that block and cannot be phi translated to predecessors. This causes it
to miss the (rare) cases like rle-must-alias.ll, but makes it faster by not
scanning tons of stuff that is unlikely to be useful. For example, this
speeds up GVN as a whole from 3.928s to 2.448s (60%)!. IMO, scalar GVN
should be enhanced to simplify the rle-must-alias pointer base anyway, which
would allow the loads to be eliminated.
In the future, this should be enhanced to phi translate through geps and
bitcasts as well (as indicated by FIXMEs) making memdep even more powerful.
llvm-svn: 61022
2008-12-15 11:35:32 +08:00
|
|
|
entry:
|
2009-11-27 07:32:59 +08:00
|
|
|
%g = alloca i32
|
|
|
|
%t1 = icmp eq i32* %b, null
|
Implement initial support for PHI translation in memdep. This means that
memdep keeps track of how PHIs affect the pointer in dep queries, which
allows it to eliminate the load in cases like rle-phi-translate.ll, which
basically end up being:
BB1:
X = load P
br BB3
BB2:
Y = load Q
br BB3
BB3:
R = phi [P] [Q]
load R
turning "load R" into a phi of X/Y. In addition to additional exposed
opportunities, this makes memdep safe in many cases that it wasn't before
(which is required for load PRE) and also makes it substantially more
efficient. For example, consider:
bb1: // has many predecessors.
P = some_operator()
load P
In this example, previously memdep would scan all the predecessors of BB1
to see if they had something that would mustalias P. In some cases (e.g.
test/Transforms/GVN/rle-must-alias.ll) it would actually find them and end
up eliminating something. In many other cases though, it would scan and not
find anything useful. MemDep now stops at a block if the pointer is defined
in that block and cannot be phi translated to predecessors. This causes it
to miss the (rare) cases like rle-must-alias.ll, but makes it faster by not
scanning tons of stuff that is unlikely to be useful. For example, this
speeds up GVN as a whole from 3.928s to 2.448s (60%)!. IMO, scalar GVN
should be enhanced to simplify the rle-must-alias pointer base anyway, which
would allow the loads to be eliminated.
In the future, this should be enhanced to phi translate through geps and
bitcasts as well (as indicated by FIXMEs) making memdep even more powerful.
llvm-svn: 61022
2008-12-15 11:35:32 +08:00
|
|
|
br i1 %t1, label %bb, label %bb1
|
|
|
|
|
2009-11-27 07:32:59 +08:00
|
|
|
bb:
|
|
|
|
%t2 = load i32* %c, align 4
|
|
|
|
%t3 = add i32 %t2, 1
|
Implement initial support for PHI translation in memdep. This means that
memdep keeps track of how PHIs affect the pointer in dep queries, which
allows it to eliminate the load in cases like rle-phi-translate.ll, which
basically end up being:
BB1:
X = load P
br BB3
BB2:
Y = load Q
br BB3
BB3:
R = phi [P] [Q]
load R
turning "load R" into a phi of X/Y. In addition to additional exposed
opportunities, this makes memdep safe in many cases that it wasn't before
(which is required for load PRE) and also makes it substantially more
efficient. For example, consider:
bb1: // has many predecessors.
P = some_operator()
load P
In this example, previously memdep would scan all the predecessors of BB1
to see if they had something that would mustalias P. In some cases (e.g.
test/Transforms/GVN/rle-must-alias.ll) it would actually find them and end
up eliminating something. In many other cases though, it would scan and not
find anything useful. MemDep now stops at a block if the pointer is defined
in that block and cannot be phi translated to predecessors. This causes it
to miss the (rare) cases like rle-must-alias.ll, but makes it faster by not
scanning tons of stuff that is unlikely to be useful. For example, this
speeds up GVN as a whole from 3.928s to 2.448s (60%)!. IMO, scalar GVN
should be enhanced to simplify the rle-must-alias pointer base anyway, which
would allow the loads to be eliminated.
In the future, this should be enhanced to phi translate through geps and
bitcasts as well (as indicated by FIXMEs) making memdep even more powerful.
llvm-svn: 61022
2008-12-15 11:35:32 +08:00
|
|
|
store i32 %t3, i32* %g, align 4
|
|
|
|
br label %bb2
|
|
|
|
|
|
|
|
bb1: ; preds = %entry
|
2009-11-27 07:32:59 +08:00
|
|
|
%t5 = load i32* %b, align 4
|
|
|
|
%t6 = add i32 %t5, 1
|
Implement initial support for PHI translation in memdep. This means that
memdep keeps track of how PHIs affect the pointer in dep queries, which
allows it to eliminate the load in cases like rle-phi-translate.ll, which
basically end up being:
BB1:
X = load P
br BB3
BB2:
Y = load Q
br BB3
BB3:
R = phi [P] [Q]
load R
turning "load R" into a phi of X/Y. In addition to additional exposed
opportunities, this makes memdep safe in many cases that it wasn't before
(which is required for load PRE) and also makes it substantially more
efficient. For example, consider:
bb1: // has many predecessors.
P = some_operator()
load P
In this example, previously memdep would scan all the predecessors of BB1
to see if they had something that would mustalias P. In some cases (e.g.
test/Transforms/GVN/rle-must-alias.ll) it would actually find them and end
up eliminating something. In many other cases though, it would scan and not
find anything useful. MemDep now stops at a block if the pointer is defined
in that block and cannot be phi translated to predecessors. This causes it
to miss the (rare) cases like rle-must-alias.ll, but makes it faster by not
scanning tons of stuff that is unlikely to be useful. For example, this
speeds up GVN as a whole from 3.928s to 2.448s (60%)!. IMO, scalar GVN
should be enhanced to simplify the rle-must-alias pointer base anyway, which
would allow the loads to be eliminated.
In the future, this should be enhanced to phi translate through geps and
bitcasts as well (as indicated by FIXMEs) making memdep even more powerful.
llvm-svn: 61022
2008-12-15 11:35:32 +08:00
|
|
|
store i32 %t6, i32* %g, align 4
|
|
|
|
br label %bb2
|
|
|
|
|
|
|
|
bb2: ; preds = %bb1, %bb
|
2009-11-27 07:32:59 +08:00
|
|
|
%c_addr.0 = phi i32* [ %g, %bb1 ], [ %c, %bb ]
|
|
|
|
%b_addr.0 = phi i32* [ %b, %bb1 ], [ %g, %bb ]
|
|
|
|
%cv = load i32* %c_addr.0, align 4
|
|
|
|
%bv = load i32* %b_addr.0, align 4
|
|
|
|
; CHECK: %bv = phi i32
|
|
|
|
; CHECK: %cv = phi i32
|
|
|
|
; CHECK-NOT: load
|
|
|
|
; CHECK: ret i32
|
|
|
|
%ret = add i32 %cv, %bv
|
Implement initial support for PHI translation in memdep. This means that
memdep keeps track of how PHIs affect the pointer in dep queries, which
allows it to eliminate the load in cases like rle-phi-translate.ll, which
basically end up being:
BB1:
X = load P
br BB3
BB2:
Y = load Q
br BB3
BB3:
R = phi [P] [Q]
load R
turning "load R" into a phi of X/Y. In addition to additional exposed
opportunities, this makes memdep safe in many cases that it wasn't before
(which is required for load PRE) and also makes it substantially more
efficient. For example, consider:
bb1: // has many predecessors.
P = some_operator()
load P
In this example, previously memdep would scan all the predecessors of BB1
to see if they had something that would mustalias P. In some cases (e.g.
test/Transforms/GVN/rle-must-alias.ll) it would actually find them and end
up eliminating something. In many other cases though, it would scan and not
find anything useful. MemDep now stops at a block if the pointer is defined
in that block and cannot be phi translated to predecessors. This causes it
to miss the (rare) cases like rle-must-alias.ll, but makes it faster by not
scanning tons of stuff that is unlikely to be useful. For example, this
speeds up GVN as a whole from 3.928s to 2.448s (60%)!. IMO, scalar GVN
should be enhanced to simplify the rle-must-alias pointer base anyway, which
would allow the loads to be eliminated.
In the future, this should be enhanced to phi translate through geps and
bitcasts as well (as indicated by FIXMEs) making memdep even more powerful.
llvm-svn: 61022
2008-12-15 11:35:32 +08:00
|
|
|
ret i32 %ret
|
|
|
|
}
|
|
|
|
|
2009-11-27 07:41:07 +08:00
|
|
|
define i8 @test2(i1 %cond, i32* %b, i32* %c) nounwind {
|
|
|
|
; CHECK: @test2
|
|
|
|
entry:
|
2009-12-20 04:44:43 +08:00
|
|
|
br i1 %cond, label %bb, label %bb1
|
2009-11-27 07:41:07 +08:00
|
|
|
|
|
|
|
bb:
|
|
|
|
%b1 = bitcast i32* %b to i8*
|
|
|
|
store i8 4, i8* %b1
|
2009-12-20 04:44:43 +08:00
|
|
|
br label %bb2
|
2009-11-27 07:41:07 +08:00
|
|
|
|
|
|
|
bb1:
|
|
|
|
%c1 = bitcast i32* %c to i8*
|
|
|
|
store i8 92, i8* %c1
|
2009-12-20 04:44:43 +08:00
|
|
|
br label %bb2
|
2009-11-27 07:41:07 +08:00
|
|
|
|
|
|
|
bb2:
|
2009-12-20 04:44:43 +08:00
|
|
|
%d = phi i32* [ %c, %bb1 ], [ %b, %bb ]
|
2009-11-27 07:41:07 +08:00
|
|
|
%d1 = bitcast i32* %d to i8*
|
2009-12-20 04:44:43 +08:00
|
|
|
%dv = load i8* %d1
|
2009-11-27 08:07:37 +08:00
|
|
|
; CHECK: %dv = phi i8 [ 92, %bb1 ], [ 4, %bb ]
|
2009-11-27 07:41:07 +08:00
|
|
|
; CHECK-NOT: load
|
|
|
|
; CHECK: ret i8 %dv
|
2009-12-20 04:44:43 +08:00
|
|
|
ret i8 %dv
|
2009-11-27 07:41:07 +08:00
|
|
|
}
|
|
|
|
|
2009-11-27 08:07:37 +08:00
|
|
|
define i32 @test3(i1 %cond, i32* %b, i32* %c) nounwind {
|
|
|
|
; CHECK: @test3
|
|
|
|
entry:
|
2009-12-20 04:44:43 +08:00
|
|
|
br i1 %cond, label %bb, label %bb1
|
2009-11-27 08:07:37 +08:00
|
|
|
|
|
|
|
bb:
|
|
|
|
%b1 = getelementptr i32* %b, i32 17
|
|
|
|
store i32 4, i32* %b1
|
2009-12-20 04:44:43 +08:00
|
|
|
br label %bb2
|
2009-11-27 08:07:37 +08:00
|
|
|
|
|
|
|
bb1:
|
|
|
|
%c1 = getelementptr i32* %c, i32 7
|
|
|
|
store i32 82, i32* %c1
|
2009-12-20 04:44:43 +08:00
|
|
|
br label %bb2
|
2009-11-27 08:07:37 +08:00
|
|
|
|
|
|
|
bb2:
|
2009-12-20 04:44:43 +08:00
|
|
|
%d = phi i32* [ %c, %bb1 ], [ %b, %bb ]
|
|
|
|
%i = phi i32 [ 7, %bb1 ], [ 17, %bb ]
|
2009-11-27 08:07:37 +08:00
|
|
|
%d1 = getelementptr i32* %d, i32 %i
|
2009-12-20 04:44:43 +08:00
|
|
|
%dv = load i32* %d1
|
2009-11-27 14:31:14 +08:00
|
|
|
; CHECK: %dv = phi i32 [ 82, %bb1 ], [ 4, %bb ]
|
|
|
|
; CHECK-NOT: load
|
|
|
|
; CHECK: ret i32 %dv
|
2009-12-20 04:44:43 +08:00
|
|
|
ret i32 %dv
|
2009-11-27 08:07:37 +08:00
|
|
|
}
|
|
|
|
|
teach phi translation of GEPs to simplify geps like 'gep x, 0'.
This allows us to compile the example from PR5313 into:
LBB1_2: ## %bb
incl %ecx
movb %al, (%rsi)
movslq %ecx, %rax
movb (%rdi,%rax), %al
testb %al, %al
jne LBB1_2
instead of:
LBB1_2: ## %bb
movslq %eax, %rcx
incl %eax
movb (%rdi,%rcx), %cl
movb %cl, (%rsi)
movslq %eax, %rcx
cmpb $0, (%rdi,%rcx)
jne LBB1_2
llvm-svn: 89981
2009-11-27 08:34:38 +08:00
|
|
|
; PR5313
|
|
|
|
define i32 @test4(i1 %cond, i32* %b, i32* %c) nounwind {
|
|
|
|
; CHECK: @test4
|
|
|
|
entry:
|
2009-12-20 04:44:43 +08:00
|
|
|
br i1 %cond, label %bb, label %bb1
|
teach phi translation of GEPs to simplify geps like 'gep x, 0'.
This allows us to compile the example from PR5313 into:
LBB1_2: ## %bb
incl %ecx
movb %al, (%rsi)
movslq %ecx, %rax
movb (%rdi,%rax), %al
testb %al, %al
jne LBB1_2
instead of:
LBB1_2: ## %bb
movslq %eax, %rcx
incl %eax
movb (%rdi,%rcx), %cl
movb %cl, (%rsi)
movslq %eax, %rcx
cmpb $0, (%rdi,%rcx)
jne LBB1_2
llvm-svn: 89981
2009-11-27 08:34:38 +08:00
|
|
|
|
|
|
|
bb:
|
|
|
|
store i32 4, i32* %b
|
2009-12-20 04:44:43 +08:00
|
|
|
br label %bb2
|
teach phi translation of GEPs to simplify geps like 'gep x, 0'.
This allows us to compile the example from PR5313 into:
LBB1_2: ## %bb
incl %ecx
movb %al, (%rsi)
movslq %ecx, %rax
movb (%rdi,%rax), %al
testb %al, %al
jne LBB1_2
instead of:
LBB1_2: ## %bb
movslq %eax, %rcx
incl %eax
movb (%rdi,%rcx), %cl
movb %cl, (%rsi)
movslq %eax, %rcx
cmpb $0, (%rdi,%rcx)
jne LBB1_2
llvm-svn: 89981
2009-11-27 08:34:38 +08:00
|
|
|
|
|
|
|
bb1:
|
|
|
|
%c1 = getelementptr i32* %c, i32 7
|
|
|
|
store i32 82, i32* %c1
|
2009-12-20 04:44:43 +08:00
|
|
|
br label %bb2
|
teach phi translation of GEPs to simplify geps like 'gep x, 0'.
This allows us to compile the example from PR5313 into:
LBB1_2: ## %bb
incl %ecx
movb %al, (%rsi)
movslq %ecx, %rax
movb (%rdi,%rax), %al
testb %al, %al
jne LBB1_2
instead of:
LBB1_2: ## %bb
movslq %eax, %rcx
incl %eax
movb (%rdi,%rcx), %cl
movb %cl, (%rsi)
movslq %eax, %rcx
cmpb $0, (%rdi,%rcx)
jne LBB1_2
llvm-svn: 89981
2009-11-27 08:34:38 +08:00
|
|
|
|
|
|
|
bb2:
|
2009-12-20 04:44:43 +08:00
|
|
|
%d = phi i32* [ %c, %bb1 ], [ %b, %bb ]
|
|
|
|
%i = phi i32 [ 7, %bb1 ], [ 0, %bb ]
|
teach phi translation of GEPs to simplify geps like 'gep x, 0'.
This allows us to compile the example from PR5313 into:
LBB1_2: ## %bb
incl %ecx
movb %al, (%rsi)
movslq %ecx, %rax
movb (%rdi,%rax), %al
testb %al, %al
jne LBB1_2
instead of:
LBB1_2: ## %bb
movslq %eax, %rcx
incl %eax
movb (%rdi,%rcx), %cl
movb %cl, (%rsi)
movslq %eax, %rcx
cmpb $0, (%rdi,%rcx)
jne LBB1_2
llvm-svn: 89981
2009-11-27 08:34:38 +08:00
|
|
|
%d1 = getelementptr i32* %d, i32 %i
|
2009-12-20 04:44:43 +08:00
|
|
|
%dv = load i32* %d1
|
2009-11-27 14:31:14 +08:00
|
|
|
; CHECK: %dv = phi i32 [ 82, %bb1 ], [ 4, %bb ]
|
|
|
|
; CHECK-NOT: load
|
|
|
|
; CHECK: ret i32 %dv
|
2009-12-20 04:44:43 +08:00
|
|
|
ret i32 %dv
|
teach phi translation of GEPs to simplify geps like 'gep x, 0'.
This allows us to compile the example from PR5313 into:
LBB1_2: ## %bb
incl %ecx
movb %al, (%rsi)
movslq %ecx, %rax
movb (%rdi,%rax), %al
testb %al, %al
jne LBB1_2
instead of:
LBB1_2: ## %bb
movslq %eax, %rcx
incl %eax
movb (%rdi,%rcx), %cl
movb %cl, (%rsi)
movslq %eax, %rcx
cmpb $0, (%rdi,%rcx)
jne LBB1_2
llvm-svn: 89981
2009-11-27 08:34:38 +08:00
|
|
|
}
|
|
|
|
|
2009-12-20 05:29:22 +08:00
|
|
|
|
|
|
|
|
|
|
|
; void test5(int N, double* G) {
|
|
|
|
; for (long j = 1; j < 1000; j++)
|
|
|
|
; G[j] = G[j] + G[j-1];
|
|
|
|
; }
|
|
|
|
;
|
|
|
|
; Should compile into one load in the loop.
|
|
|
|
define void @test5(i32 %N, double* nocapture %G) nounwind ssp {
|
|
|
|
; CHECK: @test5
|
|
|
|
bb.nph:
|
|
|
|
br label %for.body
|
|
|
|
|
|
|
|
for.body:
|
|
|
|
%indvar = phi i64 [ 0, %bb.nph ], [ %tmp, %for.body ]
|
|
|
|
%arrayidx6 = getelementptr double* %G, i64 %indvar
|
|
|
|
%tmp = add i64 %indvar, 1
|
|
|
|
%arrayidx = getelementptr double* %G, i64 %tmp
|
|
|
|
%tmp3 = load double* %arrayidx
|
|
|
|
%tmp7 = load double* %arrayidx6
|
|
|
|
%add = fadd double %tmp3, %tmp7
|
|
|
|
store double %add, double* %arrayidx
|
|
|
|
%exitcond = icmp eq i64 %tmp, 999
|
|
|
|
br i1 %exitcond, label %for.end, label %for.body
|
|
|
|
; CHECK: for.body:
|
|
|
|
; CHECK: phi double
|
|
|
|
; CHECK: load double
|
|
|
|
; CHECK-NOT: load double
|
|
|
|
; CHECK: br i1
|
|
|
|
for.end:
|
|
|
|
ret void
|
|
|
|
}
|