Proteus
Programmable JIT compilation and optimization for C/C++ using LLVM
Loading...
Searching...
No Matches
CoreLLVMHIP.h
Go to the documentation of this file.
1#ifndef PROTEUS_CORE_LLVM_HIP_H
2#define PROTEUS_CORE_LLVM_HIP_H
3
4#include "proteus/Error.h"
11
12#include <llvm/Bitcode/BitcodeWriter.h>
13#include <llvm/CodeGen/MachineModuleInfo.h>
14#include <llvm/IR/DiagnosticPrinter.h>
15#include <llvm/IR/Function.h>
16#include <llvm/IR/LegacyPassManager.h>
17#include <llvm/IR/Module.h>
18#include <llvm/IR/Verifier.h>
19#include <llvm/LTO/LTO.h>
20#include <llvm/MC/MCSubtargetInfo.h>
21#include <llvm/Support/CodeGen.h>
22#include <llvm/Support/FileSystem.h>
23#include <llvm/Support/MemoryBuffer.h>
24#include <llvm/Support/Path.h>
25#include <llvm/Support/Signals.h>
26#include <llvm/Support/TargetSelect.h>
27#include <llvm/Support/WithColor.h>
28#include <llvm/Target/TargetMachine.h>
29
30#include <optional>
31
32#if LLVM_VERSION_MAJOR >= 18
33#include <lld/Common/Driver.h>
34LLD_HAS_DRIVER(elf)
35#endif
36
37namespace proteus {
38
39using namespace llvm;
40
41namespace detail {
42
43inline const SmallVector<StringRef> &gridDimXFnName() {
44 static SmallVector<StringRef> Names = {
45 "_ZNK17__HIP_CoordinatesI13__HIP_GridDimE3__XcvjEv",
46 "llvm.amdgcn.num.workgroups.x", "_ZL20__hip_get_grid_dim_xv"};
47 return Names;
48}
49
50inline const SmallVector<StringRef> &gridDimYFnName() {
51 static SmallVector<StringRef> Names = {
52 "_ZNK17__HIP_CoordinatesI13__HIP_GridDimE3__YcvjEv",
53 "llvm.amdgcn.num.workgroups.y", "_ZL20__hip_get_grid_dim_yv"};
54 return Names;
55}
56
57inline const SmallVector<StringRef> &gridDimZFnName() {
58 static SmallVector<StringRef> Names = {
59 "_ZNK17__HIP_CoordinatesI13__HIP_GridDimE3__ZcvjEv",
60 "llvm.amdgcn.num.workgroups.z", "_ZL20__hip_get_grid_dim_zv"};
61 return Names;
62}
63
64inline const SmallVector<StringRef> &blockDimXFnName() {
65 static SmallVector<StringRef> Names = {
66 "_ZNK17__HIP_CoordinatesI14__HIP_BlockDimE3__XcvjEv",
67 "llvm.amdgcn.workgroup.size.x", "_ZL21__hip_get_block_dim_xv"};
68 return Names;
69}
70
71inline const SmallVector<StringRef> &blockDimYFnName() {
72 static SmallVector<StringRef> Names = {
73 "_ZNK17__HIP_CoordinatesI14__HIP_BlockDimE3__YcvjEv",
74 "llvm.amdgcn.workgroup.size.y", "_ZL21__hip_get_block_dim_yv"};
75 return Names;
76}
77
78inline const SmallVector<StringRef> &blockDimZFnName() {
79 static SmallVector<StringRef> Names = {
80 "_ZNK17__HIP_CoordinatesI14__HIP_BlockDimE3__ZcvjEv",
81 "llvm.amdgcn.workgroup.size.z", "_ZL21__hip_get_block_dim_zv"};
82 return Names;
83}
84
85inline const SmallVector<StringRef> &blockIdxXFnName() {
86 static SmallVector<StringRef> Names = {
87 "_ZNK17__HIP_CoordinatesI14__HIP_BlockIdxE3__XcvjEv",
88 "llvm.amdgcn.workgroup.id.x"};
89 return Names;
90};
91
92inline const SmallVector<StringRef> &blockIdxYFnName() {
93 static SmallVector<StringRef> Names = {
94 "_ZNK17__HIP_CoordinatesI14__HIP_BlockIdxE3__YcvjEv",
95 "llvm.amdgcn.workgroup.id.y"};
96 return Names;
97};
98
99inline const SmallVector<StringRef> &blockIdxZFnName() {
100 static SmallVector<StringRef> Names = {
101 "_ZNK17__HIP_CoordinatesI14__HIP_BlockIdxE3__ZcvjEv",
102 "llvm.amdgcn.workgroup.id.z"};
103 return Names;
104}
105
106inline const SmallVector<StringRef> &threadIdxXFnName() {
107 static SmallVector<StringRef> Names = {
108 "_ZNK17__HIP_CoordinatesI15__HIP_ThreadIdxE3__XcvjEv",
109 "llvm.amdgcn.workitem.id.x"};
110 return Names;
111};
112
113inline const SmallVector<StringRef> &threadIdxYFnName() {
114 static SmallVector<StringRef> Names = {
115 "_ZNK17__HIP_CoordinatesI15__HIP_ThreadIdxE3__YcvjEv",
116 "llvm.amdgcn.workitem.id.y"};
117 return Names;
118};
119
120inline const SmallVector<StringRef> &threadIdxZFnName() {
121 static SmallVector<StringRef> Names = {
122 "_ZNK17__HIP_CoordinatesI15__HIP_ThreadIdxE3__ZcvjEv",
123 "llvm.amdgcn.workitem.id.z"};
124 return Names;
125};
126
127inline Expected<sys::fs::TempFile> createTempFile(StringRef Prefix,
128 StringRef Suffix) {
129 SmallString<128> TmpDir;
130 sys::path::system_temp_directory(true, TmpDir);
131
132 SmallString<64> FileName;
133 FileName.append(Prefix);
134 FileName.append(Suffix.empty() ? "-%%%%%%%" : "-%%%%%%%.");
135 FileName.append(Suffix);
136 sys::path::append(TmpDir, FileName);
137 return sys::fs::TempFile::create(TmpDir);
138}
139
140#if LLVM_VERSION_MAJOR >= 18
141inline SmallVector<std::unique_ptr<sys::fs::TempFile>>
142codegenSerial(Module &M, StringRef DeviceArch,
143 [[maybe_unused]] char OptLevel = '3', int CodegenOptLevel = 3) {
144 TIMESCOPE("proteus::codegenSerial");
145 SmallVector<std::unique_ptr<sys::fs::TempFile>> ObjectFiles;
146
147 auto ExpectedTM =
148 proteus::detail::createTargetMachine(M, DeviceArch, CodegenOptLevel);
149 if (!ExpectedTM)
150 reportFatalError(toString(ExpectedTM.takeError()));
151
152 std::unique_ptr<TargetMachine> TM = std::move(*ExpectedTM);
153 TargetLibraryInfoImpl TLII(Triple(M.getTargetTriple()));
154 M.setDataLayout(TM->createDataLayout());
155
156 legacy::PassManager PM;
157 PM.add(new TargetLibraryInfoWrapperPass(TLII));
158 MachineModuleInfoWrapperPass *MMIWP =
159#if LLVM_VERSION_MAJOR >= 20
160 new MachineModuleInfoWrapperPass(TM.get());
161#else
162 new MachineModuleInfoWrapperPass(
163 reinterpret_cast<LLVMTargetMachine *>(TM.get()));
164#endif
165
166 SmallVector<char, 4096> ObjectCode;
167 raw_svector_ostream OS(ObjectCode);
168 auto ExpectedF = createTempFile("object", "o");
169 if (auto E = ExpectedF.takeError())
170 reportFatalError("Error creating object tmp file " +
171 toString(std::move(E)));
172 auto ObjectFile = std::move(*ExpectedF);
173 auto FileStream = std::make_unique<CachedFileStream>(
174 std::make_unique<llvm::raw_fd_ostream>(ObjectFile.FD, false));
175 TM->addPassesToEmitFile(PM, *FileStream->OS, nullptr,
176 CodeGenFileType::ObjectFile,
177 /* DisableVerify */ true, MMIWP);
178
179 std::unique_ptr<sys::fs::TempFile> ObjectFilePtr =
180 std::make_unique<sys::fs::TempFile>(std::move(ObjectFile));
181 ObjectFiles.emplace_back(std::move(ObjectFilePtr));
182
183 PM.run(M);
184 if (Error E = FileStream->commit())
185 reportFatalError("Error committing object tmp file stream " +
186 toString(std::move(E)));
187
188 return ObjectFiles;
189}
190
191inline SmallVector<std::unique_ptr<sys::fs::TempFile>>
192codegenParallel(Module &M, StringRef DeviceArch,
193 const OptimizationPipelineConfig &OptConfig =
194 OptimizationPipelineConfig(std::nullopt, '3', 3)) {
195 TIMESCOPE("proteus::codegenParallel");
196 // Use regular LTO with parallelism enabled to parallelize codegen.
197 std::atomic<bool> LTOError = false;
198
199 auto DiagnosticHandler = [&](const DiagnosticInfo &DI) {
200 std::string ErrStorage;
201 raw_string_ostream OS(ErrStorage);
202 DiagnosticPrinterRawOStream DP(OS);
203 DI.print(DP);
204
205 switch (DI.getSeverity()) {
206 case DS_Error:
207 WithColor::error(errs(), "[proteus codegen]") << ErrStorage << "\n";
208 LTOError = true;
209 break;
210 case DS_Warning:
211 WithColor::warning(errs(), "[proteus codegen]") << ErrStorage << "\n";
212 break;
213 case DS_Note:
214 WithColor::note(errs(), "[proteus codegen]") << ErrStorage << "\n";
215 break;
216 case DS_Remark:
217 WithColor::remark(errs()) << ErrStorage << "\n";
218 break;
219 }
220 };
221
222 // Create TargetMachine and extract options/features.
223 auto ExpectedTM = proteus::detail::createTargetMachine(
224 M, DeviceArch, OptConfig.CodegenOptLevel);
225 if (!ExpectedTM)
226 reportFatalError(toString(ExpectedTM.takeError()));
227 std::unique_ptr<TargetMachine> TM = std::move(*ExpectedTM);
228
229 lto::Config Conf;
230 Conf.CPU = DeviceArch;
231
232 // Propagate attributes from TargetMachine to LTO Config.
233 std::string FeatureStr = TM->getMCSubtargetInfo()->getFeatureString().str();
234 if (!FeatureStr.empty()) {
235 SmallVector<StringRef> Features;
236 StringRef(FeatureStr).split(Features, ',');
237 for (auto &F : Features)
238 Conf.MAttrs.push_back(F.str());
239 } else {
240 Conf.MAttrs = {};
241 }
242
243 // FIX: Propagate TargetOptions (e.g. UnsafeFPMath, etc.)
244 Conf.Options = TM->Options;
245
246 Conf.DisableVerify = true;
247 Conf.TimeTraceEnabled = false;
248 Conf.DebugPassManager = false;
249 Conf.VerifyEach = false;
250 Conf.DiagHandler = DiagnosticHandler;
251 Conf.OptLevel = OptConfig.OptLevel;
252 const auto Plugins = getJITPassPluginConfigs();
253 // Parallel codegen lets LTO own optimization, so custom textual pipelines
254 // must be forwarded to the LTO configuration instead of run beforehand.
255 if (OptConfig.PassPipeline ||
258 OptConfig.PassPipeline, OptConfig.OptLevel, Plugins);
259 for (const auto &PluginPath :
260 proteus::detail::getUniqueJITPassPluginPaths(Plugins))
261 Conf.PassPlugins.push_back(PluginPath);
262 Conf.CGOptLevel = static_cast<CodeGenOptLevel>(OptConfig.CodegenOptLevel);
263
264 unsigned ParallelCodeGenParallelismLevel =
265 std::max(1u, std::thread::hardware_concurrency());
266 lto::LTO L(std::move(Conf), {}, ParallelCodeGenParallelismLevel);
267
268 // Ensure module has the correct DataLayout prior to emitting bitcode.
269 M.setDataLayout(TM->createDataLayout());
270
271 SmallString<0> BitcodeBuf;
272 raw_svector_ostream BitcodeOS(BitcodeBuf);
273 WriteBitcodeToFile(M, BitcodeOS);
274
275 // TODO: Module identifier can be empty because you always have on module to
276 // link. However, in the general case, with multiple modules, each one must
277 // have a unique identifier for LTO to work correctly.
278 auto IF = cantFail(lto::InputFile::create(
279 MemoryBufferRef{BitcodeBuf, M.getModuleIdentifier()}));
280
281 std::set<std::string> PrevailingSymbols;
282 auto BuildResolutions = [&]() {
283 // Save the input file and the buffer associated with its memory.
284 const auto Symbols = IF->symbols();
285 SmallVector<lto::SymbolResolution, 16> Resolutions(Symbols.size());
286 size_t SymbolIdx = 0;
287 for (auto &Sym : Symbols) {
288 lto::SymbolResolution &Res = Resolutions[SymbolIdx];
289 SymbolIdx++;
290
291 // All defined symbols are prevailing.
292 Res.Prevailing = !Sym.isUndefined() &&
293 PrevailingSymbols.insert(Sym.getName().str()).second;
294
295 Res.VisibleToRegularObj =
296 Res.Prevailing &&
297 Sym.getVisibility() != GlobalValue::HiddenVisibility &&
298 !Sym.canBeOmittedFromSymbolTable();
299
300 Res.ExportDynamic =
301 Sym.getVisibility() != GlobalValue::HiddenVisibility &&
302 (!Sym.canBeOmittedFromSymbolTable());
303
304 Res.FinalDefinitionInLinkageUnit =
305 Sym.getVisibility() != GlobalValue::DefaultVisibility &&
306 (!Sym.isUndefined() && !Sym.isCommon());
307
308 // Device linking does not support linker redefined symbols (e.g.
309 // --wrap).
310 Res.LinkerRedefined = false;
311
313 auto PrintSymbol = [](const lto::InputFile::Symbol &Sym,
314 lto::SymbolResolution &Res) {
315 auto &OutStream = Logger::logs("proteus");
316 OutStream << "Vis: ";
317 switch (Sym.getVisibility()) {
318 case GlobalValue::HiddenVisibility:
319 OutStream << 'H';
320 break;
321 case GlobalValue::ProtectedVisibility:
322 OutStream << 'P';
323 break;
324 case GlobalValue::DefaultVisibility:
325 OutStream << 'D';
326 break;
327 }
328
329 OutStream << " Sym: ";
330 auto PrintBool = [&](char C, bool B) { OutStream << (B ? C : '-'); };
331 PrintBool('U', Sym.isUndefined());
332 PrintBool('C', Sym.isCommon());
333 PrintBool('W', Sym.isWeak());
334 PrintBool('I', Sym.isIndirect());
335 PrintBool('O', Sym.canBeOmittedFromSymbolTable());
336 PrintBool('T', Sym.isTLS());
337 PrintBool('X', Sym.isExecutable());
338 OutStream << ' ' << Sym.getName();
339 OutStream << "| P " << Res.Prevailing;
340 OutStream << " V " << Res.VisibleToRegularObj;
341 OutStream << " E " << Res.ExportDynamic;
342 OutStream << " F " << Res.FinalDefinitionInLinkageUnit;
343 OutStream << "\n";
344 };
345
346 PrintSymbol(Sym, Res);
347 }
348 }
349
350 // Add the bitcode file with its resolved symbols to the LTO job.
351 cantFail(L.add(std::move(IF), Resolutions));
352 };
353
354 BuildResolutions();
355
356 // Run the LTO job to compile the bitcode.
357 size_t MaxTasks = L.getMaxTasks();
358 SmallVector<std::unique_ptr<sys::fs::TempFile>> ObjectFiles{MaxTasks};
359
360 auto AddStream =
361 [&](size_t Task,
362 const Twine & /*ModuleName*/) -> std::unique_ptr<CachedFileStream> {
363 std::string TaskStr = Task ? "." + std::to_string(Task) : "";
364 auto ExpectedF = createTempFile("lto-shard" + TaskStr, "o");
365 if (auto E = ExpectedF.takeError())
366 reportFatalError("Error creating tmp file " + toString(std::move(E)));
367 ObjectFiles[Task] =
368 std::make_unique<sys::fs::TempFile>(std::move(*ExpectedF));
369 auto Ret = std::make_unique<CachedFileStream>(
370 std::make_unique<llvm::raw_fd_ostream>(ObjectFiles[Task]->FD, false));
371 if (!Ret)
372 reportFatalError("Error creating CachedFileStream");
373 return Ret;
374 };
375
376 if (Error E = L.run(AddStream))
377 reportFatalError("Error: " + toString(std::move(E)));
378
379 if (LTOError)
381 createStringError(inconvertibleErrorCode(),
382 "Errors encountered inside the LTO pipeline.")));
383
384 return ObjectFiles;
385}
386#endif
387
388inline std::unique_ptr<MemoryBuffer> codegenRTC(Module &M,
389 StringRef DeviceArch) {
390 TIMESCOPE("proteus::codegenRTC");
391 char *BinOut;
392 size_t BinSize;
393
394 SmallString<4096> ModuleBuf;
395 raw_svector_ostream ModuleBufOS(ModuleBuf);
396 WriteBitcodeToFile(M, ModuleBufOS);
397
398 hiprtcLinkState HipLinkStatePtr;
399
400 // NOTE: This code is an example of passing custom, AMD-specific
401 // options to the compiler/linker.
402 // NOTE: Unrolling can have a dramatic (time-consuming) effect on JIT
403 // compilation time and on the resulting optimization, better or worse
404 // depending on code specifics.
405 std::string MArchOpt = ("-march=" + DeviceArch).str();
406
407 // NOTE: We used to pass these options as well. "-mllvm",
408 // "-unroll-threshold=1000",
409 // We removed them cause we saw on bezier they cause slowdowns
410 const char *OptArgs[] = {MArchOpt.c_str()};
411 std::vector<hiprtcJIT_option> JITOptions = {
412 HIPRTC_JIT_IR_TO_ISA_OPT_EXT, HIPRTC_JIT_IR_TO_ISA_OPT_COUNT_EXT};
413 size_t OptArgsSize = 1;
414 const void *JITOptionsValues[] = {(void *)OptArgs, (void *)(OptArgsSize)};
416 JITOptions.size(), JITOptions.data(), (void **)JITOptionsValues,
417 &HipLinkStatePtr));
418 // NOTE: the following version of te code does not set options.
419 // proteusHiprtcErrCheck(hiprtcLinkCreate(0, nullptr, nullptr,
420 // &hip_link_state_ptr));
421
423 HipLinkStatePtr, HIPRTC_JIT_INPUT_LLVM_BITCODE, (void *)ModuleBuf.data(),
424 ModuleBuf.size(), "", 0, nullptr, nullptr));
426 HipLinkStatePtr, (void **)&BinOut, &BinSize));
427
428 return MemoryBuffer::getMemBuffer(StringRef{BinOut, BinSize});
429}
430
431} // namespace detail
432
433inline void setLaunchBoundsForKernel(Function &F, int MaxNumWorkGroups,
434 int MinBlocksPerSM = 0) {
435 // TODO: fix calculation of launch bounds.
436 // TODO: find maximum (hardcoded 1024) from device info.
437 // TODO: Setting as 1, BlockSize to replicate launch bounds settings
438 F.addFnAttr("amdgpu-flat-work-group-size",
439 "1," + std::to_string(std::min(1024, MaxNumWorkGroups)));
440 // F->addFnAttr("amdgpu-waves-per-eu", std::to_string(WavesPerEU));
441 if (MinBlocksPerSM != 0) {
442 // NOTE: We are missing a heuristic to define the `WavesPerEU`, as such we
443 // still need to study it. I restrict the waves by setting min equal to max
444 // and disallowing any heuristics that HIP will use internally.
445 // For more information please check:
446 // https://clang.llvm.org/docs/AttributeReference.html#amdgpu-waves-per-eu
447 F.addFnAttr("amdgpu-waves-per-eu", std::to_string(MinBlocksPerSM) + "," +
448 std::to_string(MinBlocksPerSM));
449 }
450
451 PROTEUS_DBG(Logger::logs("proteus")
452 << " => Set Workgroup size " << MaxNumWorkGroups
453 << " WavesPerEU (unused) " << MinBlocksPerSM << "\n");
454}
455
456inline std::unique_ptr<MemoryBuffer>
457codegenObject(Module &M, StringRef DeviceArch,
458 [[maybe_unused]] SmallPtrSetImpl<void *> &GlobalLinkedBinaries,
460 const OptimizationPipelineConfig &OptConfig =
461 OptimizationPipelineConfig(std::nullopt, '3', 3)) {
462 TIMESCOPE("proteus::codegenObjectHIP");
463 assert(GlobalLinkedBinaries.empty() &&
464 "Expected empty linked binaries for HIP");
465 Timer T(Config::get().ProteusEnableTimers);
466 SmallVector<std::unique_ptr<sys::fs::TempFile>> ObjectFiles;
467 switch (CGOption) {
468 case CodegenOption::RTC: {
469 auto Ret = detail::codegenRTC(M, DeviceArch);
471 << "Codegen RTC " << T.elapsed() << " ms\n");
472 return Ret;
473 }
474#if LLVM_VERSION_MAJOR >= 18
476 ObjectFiles = detail::codegenSerial(M, DeviceArch);
477 break;
479 ObjectFiles = detail::codegenParallel(M, DeviceArch, OptConfig);
480 break;
481#endif
482 default:
483 reportFatalError("Unknown Codegen Option");
484 }
485
486 if (ObjectFiles.empty())
487 reportFatalError("Expected non-empty vector of object files");
488
489#if LLVM_VERSION_MAJOR >= 18
490 auto ExpectedF = detail::createTempFile("proteus-jit", "o");
491 if (auto E = ExpectedF.takeError())
492 reportFatalError("Error creating shared object file " +
493 toString(std::move(E)));
494
495 auto SharedObject = std::move(*ExpectedF);
496
497 std::vector<const char *> Args{"ld.lld", "--no-undefined", "-shared", "-o",
498 SharedObject.TmpName.c_str()};
499 for (auto &File : ObjectFiles) {
500 if (!File)
501 continue;
502 Args.push_back(File->TmpName.c_str());
503 }
504
506 for (auto &Arg : Args) {
507 Logger::logs("proteus") << Arg << " ";
508 }
509 Logger::logs("proteus") << "\n";
510 }
511
513 << "Codegen object " << toString(CGOption) << "["
514 << ObjectFiles.size() << "] " << T.elapsed() << " ms\n");
515
516 T.reset();
517 // The LLD linker interface is not thread-safe, so we use a mutex.
518 static std::mutex Mutex;
519 {
520 std::lock_guard LockGuard{Mutex};
521 lld::Result S = lld::lldMain(Args, llvm::outs(), llvm::errs(),
522 {{lld::Gnu, &lld::elf::link}});
523 if (S.retCode)
524 reportFatalError("Error: lld failed");
525 }
526
527 ErrorOr<std::unique_ptr<MemoryBuffer>> Buffer =
528 MemoryBuffer::getFileAsStream(SharedObject.TmpName);
529 if (!Buffer)
530 reportFatalError("Error reading file: " + Buffer.getError().message());
531
532 // Remove temporary files.
533 for (auto &File : ObjectFiles) {
534 if (!File)
535 continue;
536 if (auto E = File->discard())
537 reportFatalError("Error removing object tmp file " +
538 toString(std::move(E)));
539 }
540 if (auto E = SharedObject.discard())
541 reportFatalError("Error removing shared object tmp file " +
542 toString(std::move(E)));
543
545 << "Codegen linking " << T.elapsed() << " ms\n");
546
547 return std::move(*Buffer);
548#else
549 reportFatalError("Expected LLVM18 for non-RTC codegen");
550#endif
551}
552
553} // namespace proteus
554
555#endif
char int void ** Args
Definition CompilerInterfaceHost.cpp:23
#define PROTEUS_TIMER_OUTPUT(x)
Definition Config.h:483
#define PROTEUS_DBG(x)
Definition Debug.h:9
#define TIMESCOPE(...)
Definition TimeTracing.h:66
#define proteusHiprtcErrCheck(CALL)
Definition UtilsHIP.h:30
static Config & get()
Definition Config.h:371
bool ProteusDebugOutput
Definition Config.h:387
static llvm::raw_ostream & outs(const std::string &Name)
Definition Logger.h:25
static llvm::raw_ostream & logs(const std::string &Name)
Definition Logger.h:19
Definition TimeTracing.h:33
void reset()
Definition TimeTracing.cpp:68
uint64_t elapsed()
Definition TimeTracing.cpp:66
Definition CompiledLibrary.h:8
const SmallVector< StringRef > & threadIdxXFnName()
Definition CoreLLVMCUDA.h:70
const SmallVector< StringRef > & gridDimYFnName()
Definition CoreLLVMCUDA.h:30
std::vector< std::string > getUniqueJITPassPluginPaths(const std::vector< JITPassPluginConfig > &Plugins)
Definition CoreLLVM.h:162
const SmallVector< StringRef > & threadIdxZFnName()
Definition CoreLLVMCUDA.h:80
std::string composeOptimizationPassPipeline(std::optional< std::string > PassPipeline, char OptLevel, const std::vector< JITPassPluginConfig > &Plugins)
Definition CoreLLVM.h:124
const SmallVector< StringRef > & blockIdxZFnName()
Definition CoreLLVMCUDA.h:65
const SmallVector< StringRef > & gridDimZFnName()
Definition CoreLLVMCUDA.h:35
std::unique_ptr< MemoryBuffer > codegenRTC(Module &M, StringRef DeviceArch)
Definition CoreLLVMHIP.h:388
const SmallVector< StringRef > & gridDimXFnName()
Definition CoreLLVMCUDA.h:25
const SmallVector< StringRef > & blockIdxXFnName()
Definition CoreLLVMCUDA.h:55
Expected< std::unique_ptr< TargetMachine > > createTargetMachine(Module &M, StringRef Arch, unsigned OptLevel=3)
Definition CoreLLVM.h:70
const SmallVector< StringRef > & threadIdxYFnName()
Definition CoreLLVMCUDA.h:75
Expected< sys::fs::TempFile > createTempFile(StringRef Prefix, StringRef Suffix)
Definition CoreLLVMHIP.h:127
const SmallVector< StringRef > & blockIdxYFnName()
Definition CoreLLVMCUDA.h:60
bool hasJITPassPluginInsertion(const std::vector< JITPassPluginConfig > &Plugins)
Definition CoreLLVM.h:154
const SmallVector< StringRef > & blockDimYFnName()
Definition CoreLLVMCUDA.h:45
const SmallVector< StringRef > & blockDimZFnName()
Definition CoreLLVMCUDA.h:50
const SmallVector< StringRef > & blockDimXFnName()
Definition CoreLLVMCUDA.h:40
hiprtcResult rtcLinkAddData(hiprtcLinkState LinkState, hiprtcJITInputType InputType, void *Image, size_t ImageSize, const char *Name, unsigned int NumOptions, hiprtcJIT_option *Options, void **OptionValues)
Definition HIPRuntimeAPI.cpp:198
hiprtcResult rtcLinkCreate(unsigned int NumOptions, hiprtcJIT_option *Options, void **OptionValues, hiprtcLinkState *LinkStateOut)
Definition HIPRuntimeAPI.cpp:191
hiprtcResult rtcLinkComplete(hiprtcLinkState LinkState, void **BinOut, size_t *SizeOut)
Definition HIPRuntimeAPI.cpp:209
Definition MemoryCache.h:27
std::vector< JITPassPluginConfig > getJITPassPluginConfigs()
Definition JITPassPluginRegistry.cpp:105
void setLaunchBoundsForKernel(Function &F, int MaxThreadsPerSM, int MinBlocksPerSM=0)
Definition CoreLLVMCUDA.h:87
void reportFatalError(const llvm::Twine &Reason, const char *FILE, unsigned Line)
Definition Error.cpp:14
CodegenOption
Definition Config.h:19
std::unique_ptr< MemoryBuffer > codegenObject(Module &M, StringRef DeviceArch, SmallPtrSetImpl< void * > &GlobalLinkedBinaries, CodegenOption CGOption=CodegenOption::RTC)
Definition CoreLLVMCUDA.h:176
std::string toString(CodegenOption Option)
Definition Config.h:31
Definition CoreLLVM.h:291