It looks like select gets lowered to an unconditional write
LLVM diff:
diff --git a/llvm/lib/Transforms/Instrumentation/InstrProfiling.cpp b/llvm/lib/Transforms/Instrumentation/InstrProfiling.cpp
index fe5a0578bd97..f1b68417d575 100644
--- a/llvm/lib/Transforms/Instrumentation/InstrProfiling.cpp
+++ b/llvm/lib/Transforms/Instrumentation/InstrProfiling.cpp
@@ -908,9 +908,16 @@ Value *InstrLowerer::getBitmapAddress(InstrProfMCDCTVBitmapUpdate *I) {
void InstrLowerer::lowerCover(InstrProfCoverInst *CoverInstruction) {
auto *Addr = getCounterAddress(CoverInstruction);
+ auto &Ctx = CoverInstruction->getParent()->getContext();
+ auto *Int8Ty = llvm::Type::getInt8Ty(Ctx);
IRBuilder<> Builder(CoverInstruction);
+ Value *Load = Builder.CreateLoad(Int8Ty, Addr, "pgocount");
+ Value *CondV = Builder.CreateICmpNE(Load, ConstantInt::get(Int8Ty, 0),
+ "pgocount.ifnonzero");
+ Value *Sel = Builder.CreateSelect(CondV, Builder.getInt8(0), Load,
+ "pgocount.select");
// We store zero to represent that this block is covered.
- Builder.CreateStore(Builder.getInt8(0), Addr);
+ Builder.CreateStore(Sel, Addr);
CoverInstruction->eraseFromParent();
}
I have a test file simple.cc which contains a function foo() that does nothing:
void foo() {
}
If we compile with -O0 we see that there is indeed a select instruction:
~/src/llvm-project/build/bin/clang -c -S -emit-llvm -O0 -fprofile-instr-generate -mllvm -runtime-counter-relocation=true -mllvm -enable-single-byte-coverage=true ~/src/tests/simple.cc -o -
(other bits omitted)
; Function Attrs: mustprogress noinline nounwind optnone uwtable
define dso_local void @_Z3foov() #0 {
%1 = load i64, ptr @__llvm_profile_counter_bias, align 8
%2 = add i64 ptrtoint (ptr @__profc__Z3foov to i64), %1
%3 = inttoptr i64 %2 to ptr
%4 = load i8, ptr %3, align 1
%5 = icmp ne i8 %4, 0
%6 = select i1 %5, i8 0, i8 %4
store i8 %6, ptr %3, align 1
ret void
}
-O3 lowers the select to an unconditional store in LLVM IR
define dso_local void @_Z3foov() local_unnamed_addr #0 {
%1 = load i64, ptr @__llvm_profile_counter_bias, align 8
%2 = add i64 %1, ptrtoint (ptr @__profc__Z3foov to i64)
%3 = inttoptr i64 %2 to ptr
store i8 0, ptr %3, align 1
ret void
}
which gets lowered into mov 0 in x86:
_Z3foov: # @_Z3foov
.cfi_startproc
# %bb.0:
movq __llvm_profile_counter_bias(%rip), %rax
leaq .L__profc__Z3foov(%rip), %rcx
movb $0, (%rax,%rcx)
retq
OTOH, LLVM preserves the branch instruction:
_Z3foov: # @_Z3foov
.cfi_startproc
# %bb.0:
movq __llvm_profile_counter_bias(%rip), %rax
leaq .L__profc__Z3foov(%rip), %rcx
cmpb $0, (%rax,%rcx)
je .LBB0_2
# %bb.1:
movb $0, (%rax,%rcx)
.LBB0_2:
retq
perf stat on base_unittests also confirms that no conditional counter updates were made:
Performance counter stats for 'out/selectcondbool/base_unittests':
8,138,966.68 msec task-clock:u # 28.346 CPUs utilized
0 context-switches:u # 0.000 /sec
0 cpu-migrations:u # 0.000 /sec
8,400,131 page-faults:u # 1.032 K/sec
31,061,247,596,477 cycles:u # 3.816 GHz (83.35%)
1,312,779,320,466 stalled-cycles-frontend:u # 4.23% frontend cycles idle (83.35%)
22,075,624,483,530 stalled-cycles-backend:u # 71.07% backend cycles idle (83.36%)
1,330,903,366,283 instructions:u # 0.04 insn per cycle
# 16.59 stalled cycles per insn (83.34%)
159,776,601,808 branches:u # 19.631 M/sec (83.34%)
1,694,924,175 branch-misses:u # 1.06% of all branches (83.35%)
287.130240302 seconds time elapsed
7853.346829000 seconds user
258.224291000 seconds sys
EDIT: missed a retq instruction