diff --git a/llvm/lib/Target/X86/X86DiscriminateMemOps.cpp b/llvm/lib/Target/X86/X86DiscriminateMemOps.cpp --- a/llvm/lib/Target/X86/X86DiscriminateMemOps.cpp +++ b/llvm/lib/Target/X86/X86DiscriminateMemOps.cpp @@ -34,6 +34,14 @@ "the build of the binary consuming the profile."), cl::Hidden); +static cl::opt BypassPrefetchInstructions( + "x86-bypass-prefetch-instructions", cl::init(true), + cl::desc("When discriminating instructions with memory operands, ignore " + "prefetch instructions. This ensures the other memory operand " + "instructions have the same identifiers after inserting " + "prefetches, allowing for successive insertions."), + cl::Hidden); + namespace { using Location = std::pair; @@ -62,6 +70,10 @@ X86DiscriminateMemOps(); }; +bool IsPrefetchOpcode(unsigned Opcode) { + return Opcode == X86::PREFETCHNTA || Opcode == X86::PREFETCHT0 || + Opcode == X86::PREFETCHT1 || Opcode == X86::PREFETCHT2; +} } // end anonymous namespace //===----------------------------------------------------------------------===// @@ -98,6 +110,8 @@ const auto &DI = MI.getDebugLoc(); if (!DI) continue; + if (BypassPrefetchInstructions && IsPrefetchOpcode(MI.getDesc().Opcode)) + continue; Location Loc = diToLocation(DI); MemOpDiscriminators[Loc] = std::max(MemOpDiscriminators[Loc], DI->getBaseDiscriminator()); @@ -114,6 +128,8 @@ for (auto &MI : MBB) { if (X86II::getMemoryOperandNo(MI.getDesc().TSFlags) < 0) continue; + if (BypassPrefetchInstructions && IsPrefetchOpcode(MI.getDesc().Opcode)) + continue; const DILocation *DI = MI.getDebugLoc(); bool HasDebug = DI; if (!HasDebug) { diff --git a/llvm/test/CodeGen/X86/discriminate-mem-ops-skip-pfetch.ll b/llvm/test/CodeGen/X86/discriminate-mem-ops-skip-pfetch.ll new file mode 100644 --- /dev/null +++ b/llvm/test/CodeGen/X86/discriminate-mem-ops-skip-pfetch.ll @@ -0,0 +1,69 @@ +; RUN: llc -x86-discriminate-memops < %s | FileCheck %s +; RUN: llc -x86-discriminate-memops -x86-bypass-prefetch-instructions=0 < %s | FileCheck %s -check-prefix=NOBYPASS +; +; original source, compiled with -O3 -gmlt -fdebug-info-for-profiling: +; int sum(int* arr, int pos1, int pos2) { +; return arr[pos1] + arr[pos2]; +; } +; +; ModuleID = 'test.cc' +source_filename = "test.cc" +target datalayout = "e-m:e-i64:64-f80:128-n8:16:32:64-S128" +target triple = "x86_64-unknown-linux-gnu" + +declare void @llvm.prefetch(i8 *, i32, i32, i32) +; Function Attrs: norecurse nounwind readonly uwtable +define i32 @sum(i32* %arr, i32 %pos1, i32 %pos2) !dbg !7 { +entry: + %idxprom = sext i32 %pos1 to i64, !dbg !9 + %arrayidx = getelementptr inbounds i32, i32* %arr, i64 %idxprom, !dbg !9 + %0 = load i32, i32* %arrayidx, align 4, !dbg !9, !tbaa !10 + %idxprom1 = sext i32 %pos2 to i64, !dbg !14 + %arrayidx2 = getelementptr inbounds i32, i32* %arr, i64 %idxprom1, !dbg !14 + %addr = bitcast i32* %arrayidx2 to i8* + call void @llvm.prefetch(i8* %addr, i32 0, i32 3, i32 1) + %1 = load i32, i32* %arrayidx2, align 4, !dbg !14, !tbaa !10 + %add = add nsw i32 %1, %0, !dbg !15 + ret i32 %add, !dbg !16 +} + +attributes #0 = { "target-cpu"="x86-64" } + +!llvm.dbg.cu = !{!0} +!llvm.module.flags = !{!3, !4, !5} +!llvm.ident = !{!6} + +!0 = distinct !DICompileUnit(language: DW_LANG_C_plus_plus, file: !1, isOptimized: true, runtimeVersion: 0, emissionKind: LineTablesOnly, enums: !2, debugInfoForProfiling: true) +!1 = !DIFile(filename: "test.cc", directory: "/tmp") +!2 = !{} +!3 = !{i32 2, !"Dwarf Version", i32 4} +!4 = !{i32 2, !"Debug Info Version", i32 3} +!5 = !{i32 1, !"wchar_size", i32 4} +!6 = !{!"clang version 7.0.0 (trunk 322155) (llvm/trunk 322159)"} +!7 = distinct !DISubprogram(name: "sum", linkageName: "sum", scope: !1, file: !1, line: 1, type: !8, isLocal: false, isDefinition: true, scopeLine: 1, flags: DIFlagPrototyped, isOptimized: true, unit: !0) +!8 = !DISubroutineType(types: !2) +!9 = !DILocation(line: 2, column: 10, scope: !7) +!10 = !{!11, !11, i64 0} +!11 = !{!"int", !12, i64 0} +!12 = !{!"omnipotent char", !13, i64 0} +!13 = !{!"Simple C++ TBAA"} +!14 = !DILocation(line: 2, column: 22, scope: !7) +!15 = !DILocation(line: 2, column: 20, scope: !7) +!16 = !DILocation(line: 2, column: 3, scope: !7) + +;CHECK-LABEL: sum: +;CHECK: # %bb.0: +;CHECK: prefetcht0 (%rdi,%rax,4) +;CHECK-NEXT: movl (%rdi,%rax,4), %eax +;CHECK-NEXT: .loc 1 2 20 discriminator 2 # test.cc:2:20 +;CHECK-NEXT: addl (%rdi,%rcx,4), %eax +;CHECK-NEXT: .loc 1 2 3 # test.cc:2:3 + +;NOBYPASS-LABEL: sum: +;NOBYPASS: # %bb.0: +;NOBYPASS: prefetcht0 (%rdi,%rax,4) +;NOBYPASS-NEXT: .loc 1 2 22 +;NOBYPASS-NEXT: movl (%rdi,%rax,4), %eax +;NOBYPASS-NEXT: .loc 1 2 20 {{.*}} discriminator 2 # test.cc:2:20 +;NOBYPASS-NEXT: addl (%rdi,%rcx,4), %eax +;NOBYPASS-NEXT: .loc 1 2 3 # test.cc:2:3