We've noticed that for large builds executing thin-link can take on the order of 10s of minutes. We are only using a single thread to write the sharded indices and import files for each input bitcode file. While we need to ensure the index file produced lists modules in a deterministic order, that doesn't prevent us from executing the rest of the work in parallel. In this change we use a thread pool to execute as much of the backend's work as possible in parallel. In local testing on a machine with 80 cores, this change makes a thin-link for ~100,000 input files run in ~2 minutes. Without this change it takes upwards of 10 minutes. --------- Co-authored-by: Nuri Amari <nuriamari@fb.com>
25 lines
740 B
LLVM
25 lines
740 B
LLVM
; REQUIRES: x86, non-root-user
|
|
|
|
; Basic ThinLTO tests.
|
|
; RUN: opt -module-summary %s -o %t1.o
|
|
; RUN: opt -module-summary %p/Inputs/thinlto.ll -o %t2.o
|
|
|
|
; Ensure lld generates error if unable to write to index files
|
|
; RUN: rm -f %t2.o.thinlto.bc
|
|
; RUN: touch %t2.o.thinlto.bc
|
|
; RUN: chmod u-w %t2.o.thinlto.bc
|
|
; RUN: not ld.lld --plugin-opt=thinlto-index-only -shared %t1.o %t2.o -o /dev/null 2>&1 | FileCheck -DMSG=%errc_EACCES %s
|
|
; RUN: chmod u+w %t2.o.thinlto.bc
|
|
; CHECK: 'cannot open {{.*}}2.o.thinlto.bc': [[MSG]]
|
|
|
|
target datalayout = "e-m:e-p270:32:32-p271:32:32-p272:64:64-i64:64-f80:128-n8:16:32:64-S128"
|
|
target triple = "x86_64-unknown-linux-gnu"
|
|
|
|
declare void @g(...)
|
|
|
|
define void @f() {
|
|
entry:
|
|
call void (...) @g()
|
|
ret void
|
|
}
|