1 //===--- Cuda.cpp - Cuda Tool and ToolChain Implementations -----*- C++ -*-===//
2 //
3 //                     The LLVM Compiler Infrastructure
4 //
5 // This file is distributed under the University of Illinois Open Source
6 // License. See LICENSE.TXT for details.
7 //
8 //===----------------------------------------------------------------------===//
9 
10 #include "Cuda.h"
11 #include "CommonArgs.h"
12 #include "InputInfo.h"
13 #include "clang/Basic/Cuda.h"
14 #include "clang/Basic/VirtualFileSystem.h"
15 #include "clang/Config/config.h"
16 #include "clang/Driver/Compilation.h"
17 #include "clang/Driver/Distro.h"
18 #include "clang/Driver/Driver.h"
19 #include "clang/Driver/DriverDiagnostic.h"
20 #include "clang/Driver/Options.h"
21 #include "llvm/Option/ArgList.h"
22 #include "llvm/Support/FileSystem.h"
23 #include "llvm/Support/Path.h"
24 #include "llvm/Support/Program.h"
25 #include <system_error>
26 
27 using namespace clang::driver;
28 using namespace clang::driver::toolchains;
29 using namespace clang::driver::tools;
30 using namespace clang;
31 using namespace llvm::opt;
32 
33 // Parses the contents of version.txt in an CUDA installation.  It should
34 // contain one line of the from e.g. "CUDA Version 7.5.2".
35 static CudaVersion ParseCudaVersionFile(llvm::StringRef V) {
36   if (!V.startswith("CUDA Version "))
37     return CudaVersion::UNKNOWN;
38   V = V.substr(strlen("CUDA Version "));
39   int Major = -1, Minor = -1;
40   auto First = V.split('.');
41   auto Second = First.second.split('.');
42   if (First.first.getAsInteger(10, Major) ||
43       Second.first.getAsInteger(10, Minor))
44     return CudaVersion::UNKNOWN;
45 
46   if (Major == 7 && Minor == 0) {
47     // This doesn't appear to ever happen -- version.txt doesn't exist in the
48     // CUDA 7 installs I've seen.  But no harm in checking.
49     return CudaVersion::CUDA_70;
50   }
51   if (Major == 7 && Minor == 5)
52     return CudaVersion::CUDA_75;
53   if (Major == 8 && Minor == 0)
54     return CudaVersion::CUDA_80;
55   if (Major == 9 && Minor == 0)
56     return CudaVersion::CUDA_90;
57   if (Major == 9 && Minor == 1)
58     return CudaVersion::CUDA_91;
59   return CudaVersion::UNKNOWN;
60 }
61 
62 CudaInstallationDetector::CudaInstallationDetector(
63     const Driver &D, const llvm::Triple &HostTriple,
64     const llvm::opt::ArgList &Args)
65     : D(D) {
66   struct Candidate {
67     std::string Path;
68     bool StrictChecking;
69 
70     Candidate(std::string Path, bool StrictChecking = false)
71         : Path(Path), StrictChecking(StrictChecking) {}
72   };
73   SmallVector<Candidate, 4> Candidates;
74 
75   // In decreasing order so we prefer newer versions to older versions.
76   std::initializer_list<const char *> Versions = {"8.0", "7.5", "7.0"};
77 
78   if (Args.hasArg(clang::driver::options::OPT_cuda_path_EQ)) {
79     Candidates.emplace_back(
80         Args.getLastArgValue(clang::driver::options::OPT_cuda_path_EQ).str());
81   } else if (HostTriple.isOSWindows()) {
82     for (const char *Ver : Versions)
83       Candidates.emplace_back(
84           D.SysRoot + "/Program Files/NVIDIA GPU Computing Toolkit/CUDA/v" +
85           Ver);
86   } else {
87     if (!Args.hasArg(clang::driver::options::OPT_cuda_path_ignore_env)) {
88       // Try to find ptxas binary. If the executable is located in a directory
89       // called 'bin/', its parent directory might be a good guess for a valid
90       // CUDA installation.
91       // However, some distributions might installs 'ptxas' to /usr/bin. In that
92       // case the candidate would be '/usr' which passes the following checks
93       // because '/usr/include' exists as well. To avoid this case, we always
94       // check for the directory potentially containing files for libdevice,
95       // even if the user passes -nocudalib.
96       if (llvm::ErrorOr<std::string> ptxas =
97               llvm::sys::findProgramByName("ptxas")) {
98         SmallString<256> ptxasAbsolutePath;
99         llvm::sys::fs::real_path(*ptxas, ptxasAbsolutePath);
100 
101         StringRef ptxasDir = llvm::sys::path::parent_path(ptxasAbsolutePath);
102         if (llvm::sys::path::filename(ptxasDir) == "bin")
103           Candidates.emplace_back(llvm::sys::path::parent_path(ptxasDir),
104                                   /*StrictChecking=*/true);
105       }
106     }
107 
108     Candidates.emplace_back(D.SysRoot + "/usr/local/cuda");
109     for (const char *Ver : Versions)
110       Candidates.emplace_back(D.SysRoot + "/usr/local/cuda-" + Ver);
111 
112     if (Distro(D.getVFS()).IsDebian())
113       // Special case for Debian to have nvidia-cuda-toolkit work
114       // out of the box. More info on http://bugs.debian.org/882505
115       Candidates.emplace_back(D.SysRoot + "/usr/lib/cuda");
116   }
117 
118   bool NoCudaLib = Args.hasArg(options::OPT_nocudalib);
119 
120   for (const auto &Candidate : Candidates) {
121     InstallPath = Candidate.Path;
122     if (InstallPath.empty() || !D.getVFS().exists(InstallPath))
123       continue;
124 
125     BinPath = InstallPath + "/bin";
126     IncludePath = InstallPath + "/include";
127     LibDevicePath = InstallPath + "/nvvm/libdevice";
128 
129     auto &FS = D.getVFS();
130     if (!(FS.exists(IncludePath) && FS.exists(BinPath)))
131       continue;
132     bool CheckLibDevice = (!NoCudaLib || Candidate.StrictChecking);
133     if (CheckLibDevice && !FS.exists(LibDevicePath))
134       continue;
135 
136     // On Linux, we have both lib and lib64 directories, and we need to choose
137     // based on our triple.  On MacOS, we have only a lib directory.
138     //
139     // It's sufficient for our purposes to be flexible: If both lib and lib64
140     // exist, we choose whichever one matches our triple.  Otherwise, if only
141     // lib exists, we use it.
142     if (HostTriple.isArch64Bit() && FS.exists(InstallPath + "/lib64"))
143       LibPath = InstallPath + "/lib64";
144     else if (FS.exists(InstallPath + "/lib"))
145       LibPath = InstallPath + "/lib";
146     else
147       continue;
148 
149     llvm::ErrorOr<std::unique_ptr<llvm::MemoryBuffer>> VersionFile =
150         FS.getBufferForFile(InstallPath + "/version.txt");
151     if (!VersionFile) {
152       // CUDA 7.0 doesn't have a version.txt, so guess that's our version if
153       // version.txt isn't present.
154       Version = CudaVersion::CUDA_70;
155     } else {
156       Version = ParseCudaVersionFile((*VersionFile)->getBuffer());
157     }
158 
159     if (Version >= CudaVersion::CUDA_90) {
160       // CUDA-9+ uses single libdevice file for all GPU variants.
161       std::string FilePath = LibDevicePath + "/libdevice.10.bc";
162       if (FS.exists(FilePath)) {
163         for (const char *GpuArchName :
164              {"sm_20", "sm_30", "sm_32", "sm_35", "sm_50", "sm_52", "sm_53",
165                    "sm_60", "sm_61", "sm_62", "sm_70", "sm_72"}) {
166           const CudaArch GpuArch = StringToCudaArch(GpuArchName);
167           if (Version >= MinVersionForCudaArch(GpuArch) &&
168               Version <= MaxVersionForCudaArch(GpuArch))
169             LibDeviceMap[GpuArchName] = FilePath;
170         }
171       }
172     } else {
173       std::error_code EC;
174       for (llvm::sys::fs::directory_iterator LI(LibDevicePath, EC), LE;
175            !EC && LI != LE; LI = LI.increment(EC)) {
176         StringRef FilePath = LI->path();
177         StringRef FileName = llvm::sys::path::filename(FilePath);
178         // Process all bitcode filenames that look like
179         // libdevice.compute_XX.YY.bc
180         const StringRef LibDeviceName = "libdevice.";
181         if (!(FileName.startswith(LibDeviceName) && FileName.endswith(".bc")))
182           continue;
183         StringRef GpuArch = FileName.slice(
184             LibDeviceName.size(), FileName.find('.', LibDeviceName.size()));
185         LibDeviceMap[GpuArch] = FilePath.str();
186         // Insert map entries for specifc devices with this compute
187         // capability. NVCC's choice of the libdevice library version is
188         // rather peculiar and depends on the CUDA version.
189         if (GpuArch == "compute_20") {
190           LibDeviceMap["sm_20"] = FilePath;
191           LibDeviceMap["sm_21"] = FilePath;
192           LibDeviceMap["sm_32"] = FilePath;
193         } else if (GpuArch == "compute_30") {
194           LibDeviceMap["sm_30"] = FilePath;
195           if (Version < CudaVersion::CUDA_80) {
196             LibDeviceMap["sm_50"] = FilePath;
197             LibDeviceMap["sm_52"] = FilePath;
198             LibDeviceMap["sm_53"] = FilePath;
199           }
200           LibDeviceMap["sm_60"] = FilePath;
201           LibDeviceMap["sm_61"] = FilePath;
202           LibDeviceMap["sm_62"] = FilePath;
203         } else if (GpuArch == "compute_35") {
204           LibDeviceMap["sm_35"] = FilePath;
205           LibDeviceMap["sm_37"] = FilePath;
206         } else if (GpuArch == "compute_50") {
207           if (Version >= CudaVersion::CUDA_80) {
208             LibDeviceMap["sm_50"] = FilePath;
209             LibDeviceMap["sm_52"] = FilePath;
210             LibDeviceMap["sm_53"] = FilePath;
211           }
212         }
213       }
214     }
215 
216     // Check that we have found at least one libdevice that we can link in if
217     // -nocudalib hasn't been specified.
218     if (LibDeviceMap.empty() && !NoCudaLib)
219       continue;
220 
221     IsValid = true;
222     break;
223   }
224 }
225 
226 void CudaInstallationDetector::AddCudaIncludeArgs(
227     const ArgList &DriverArgs, ArgStringList &CC1Args) const {
228   if (!DriverArgs.hasArg(options::OPT_nobuiltininc)) {
229     // Add cuda_wrappers/* to our system include path.  This lets us wrap
230     // standard library headers.
231     SmallString<128> P(D.ResourceDir);
232     llvm::sys::path::append(P, "include");
233     llvm::sys::path::append(P, "cuda_wrappers");
234     CC1Args.push_back("-internal-isystem");
235     CC1Args.push_back(DriverArgs.MakeArgString(P));
236   }
237 
238   if (DriverArgs.hasArg(options::OPT_nocudainc))
239     return;
240 
241   if (!isValid()) {
242     D.Diag(diag::err_drv_no_cuda_installation);
243     return;
244   }
245 
246   CC1Args.push_back("-internal-isystem");
247   CC1Args.push_back(DriverArgs.MakeArgString(getIncludePath()));
248   CC1Args.push_back("-include");
249   CC1Args.push_back("__clang_cuda_runtime_wrapper.h");
250 }
251 
252 void CudaInstallationDetector::CheckCudaVersionSupportsArch(
253     CudaArch Arch) const {
254   if (Arch == CudaArch::UNKNOWN || Version == CudaVersion::UNKNOWN ||
255       ArchsWithBadVersion.count(Arch) > 0)
256     return;
257 
258   auto MinVersion = MinVersionForCudaArch(Arch);
259   auto MaxVersion = MaxVersionForCudaArch(Arch);
260   if (Version < MinVersion || Version > MaxVersion) {
261     ArchsWithBadVersion.insert(Arch);
262     D.Diag(diag::err_drv_cuda_version_unsupported)
263         << CudaArchToString(Arch) << CudaVersionToString(MinVersion)
264         << CudaVersionToString(MaxVersion) << InstallPath
265         << CudaVersionToString(Version);
266   }
267 }
268 
269 void CudaInstallationDetector::print(raw_ostream &OS) const {
270   if (isValid())
271     OS << "Found CUDA installation: " << InstallPath << ", version "
272        << CudaVersionToString(Version) << "\n";
273 }
274 
275 void NVPTX::Assembler::ConstructJob(Compilation &C, const JobAction &JA,
276                                     const InputInfo &Output,
277                                     const InputInfoList &Inputs,
278                                     const ArgList &Args,
279                                     const char *LinkingOutput) const {
280   const auto &TC =
281       static_cast<const toolchains::CudaToolChain &>(getToolChain());
282   assert(TC.getTriple().isNVPTX() && "Wrong platform");
283 
284   StringRef GPUArchName;
285   // If this is an OpenMP action we need to extract the device architecture
286   // from the -march=arch option. This option may come from -Xopenmp-target
287   // flag or the default value.
288   if (JA.isDeviceOffloading(Action::OFK_OpenMP)) {
289     GPUArchName = Args.getLastArgValue(options::OPT_march_EQ);
290     assert(!GPUArchName.empty() && "Must have an architecture passed in.");
291   } else
292     GPUArchName = JA.getOffloadingArch();
293 
294   // Obtain architecture from the action.
295   CudaArch gpu_arch = StringToCudaArch(GPUArchName);
296   assert(gpu_arch != CudaArch::UNKNOWN &&
297          "Device action expected to have an architecture.");
298 
299   // Check that our installation's ptxas supports gpu_arch.
300   if (!Args.hasArg(options::OPT_no_cuda_version_check)) {
301     TC.CudaInstallation.CheckCudaVersionSupportsArch(gpu_arch);
302   }
303 
304   ArgStringList CmdArgs;
305   CmdArgs.push_back(TC.getTriple().isArch64Bit() ? "-m64" : "-m32");
306   if (Args.hasFlag(options::OPT_cuda_noopt_device_debug,
307                    options::OPT_no_cuda_noopt_device_debug, false)) {
308     // ptxas does not accept -g option if optimization is enabled, so
309     // we ignore the compiler's -O* options if we want debug info.
310     CmdArgs.push_back("-g");
311     CmdArgs.push_back("--dont-merge-basicblocks");
312     CmdArgs.push_back("--return-at-end");
313   } else if (Arg *A = Args.getLastArg(options::OPT_O_Group)) {
314     // Map the -O we received to -O{0,1,2,3}.
315     //
316     // TODO: Perhaps we should map host -O2 to ptxas -O3. -O3 is ptxas's
317     // default, so it may correspond more closely to the spirit of clang -O2.
318 
319     // -O3 seems like the least-bad option when -Osomething is specified to
320     // clang but it isn't handled below.
321     StringRef OOpt = "3";
322     if (A->getOption().matches(options::OPT_O4) ||
323         A->getOption().matches(options::OPT_Ofast))
324       OOpt = "3";
325     else if (A->getOption().matches(options::OPT_O0))
326       OOpt = "0";
327     else if (A->getOption().matches(options::OPT_O)) {
328       // -Os, -Oz, and -O(anything else) map to -O2, for lack of better options.
329       OOpt = llvm::StringSwitch<const char *>(A->getValue())
330                  .Case("1", "1")
331                  .Case("2", "2")
332                  .Case("3", "3")
333                  .Case("s", "2")
334                  .Case("z", "2")
335                  .Default("2");
336     }
337     CmdArgs.push_back(Args.MakeArgString(llvm::Twine("-O") + OOpt));
338   } else {
339     // If no -O was passed, pass -O0 to ptxas -- no opt flag should correspond
340     // to no optimizations, but ptxas's default is -O3.
341     CmdArgs.push_back("-O0");
342   }
343 
344   // Pass -v to ptxas if it was passed to the driver.
345   if (Args.hasArg(options::OPT_v))
346     CmdArgs.push_back("-v");
347 
348   CmdArgs.push_back("--gpu-name");
349   CmdArgs.push_back(Args.MakeArgString(CudaArchToString(gpu_arch)));
350   CmdArgs.push_back("--output-file");
351   CmdArgs.push_back(Args.MakeArgString(TC.getInputFilename(Output)));
352   for (const auto& II : Inputs)
353     CmdArgs.push_back(Args.MakeArgString(II.getFilename()));
354 
355   for (const auto& A : Args.getAllArgValues(options::OPT_Xcuda_ptxas))
356     CmdArgs.push_back(Args.MakeArgString(A));
357 
358   bool Relocatable = false;
359   if (JA.isOffloading(Action::OFK_OpenMP))
360     // In OpenMP we need to generate relocatable code.
361     Relocatable = Args.hasFlag(options::OPT_fopenmp_relocatable_target,
362                                options::OPT_fnoopenmp_relocatable_target,
363                                /*Default=*/true);
364   else if (JA.isOffloading(Action::OFK_Cuda))
365     Relocatable = Args.hasFlag(options::OPT_fcuda_rdc,
366                                options::OPT_fno_cuda_rdc, /*Default=*/false);
367 
368   if (Relocatable)
369     CmdArgs.push_back("-c");
370 
371   const char *Exec;
372   if (Arg *A = Args.getLastArg(options::OPT_ptxas_path_EQ))
373     Exec = A->getValue();
374   else
375     Exec = Args.MakeArgString(TC.GetProgramPath("ptxas"));
376   C.addCommand(llvm::make_unique<Command>(JA, *this, Exec, CmdArgs, Inputs));
377 }
378 
379 // All inputs to this linker must be from CudaDeviceActions, as we need to look
380 // at the Inputs' Actions in order to figure out which GPU architecture they
381 // correspond to.
382 void NVPTX::Linker::ConstructJob(Compilation &C, const JobAction &JA,
383                                  const InputInfo &Output,
384                                  const InputInfoList &Inputs,
385                                  const ArgList &Args,
386                                  const char *LinkingOutput) const {
387   const auto &TC =
388       static_cast<const toolchains::CudaToolChain &>(getToolChain());
389   assert(TC.getTriple().isNVPTX() && "Wrong platform");
390 
391   ArgStringList CmdArgs;
392   CmdArgs.push_back("--cuda");
393   CmdArgs.push_back(TC.getTriple().isArch64Bit() ? "-64" : "-32");
394   CmdArgs.push_back(Args.MakeArgString("--create"));
395   CmdArgs.push_back(Args.MakeArgString(Output.getFilename()));
396 
397   for (const auto& II : Inputs) {
398     auto *A = II.getAction();
399     assert(A->getInputs().size() == 1 &&
400            "Device offload action is expected to have a single input");
401     const char *gpu_arch_str = A->getOffloadingArch();
402     assert(gpu_arch_str &&
403            "Device action expected to have associated a GPU architecture!");
404     CudaArch gpu_arch = StringToCudaArch(gpu_arch_str);
405 
406     // We need to pass an Arch of the form "sm_XX" for cubin files and
407     // "compute_XX" for ptx.
408     const char *Arch =
409         (II.getType() == types::TY_PP_Asm)
410             ? CudaVirtualArchToString(VirtualArchForCudaArch(gpu_arch))
411             : gpu_arch_str;
412     CmdArgs.push_back(Args.MakeArgString(llvm::Twine("--image=profile=") +
413                                          Arch + ",file=" + II.getFilename()));
414   }
415 
416   for (const auto& A : Args.getAllArgValues(options::OPT_Xcuda_fatbinary))
417     CmdArgs.push_back(Args.MakeArgString(A));
418 
419   const char *Exec = Args.MakeArgString(TC.GetProgramPath("fatbinary"));
420   C.addCommand(llvm::make_unique<Command>(JA, *this, Exec, CmdArgs, Inputs));
421 }
422 
423 void NVPTX::OpenMPLinker::ConstructJob(Compilation &C, const JobAction &JA,
424                                        const InputInfo &Output,
425                                        const InputInfoList &Inputs,
426                                        const ArgList &Args,
427                                        const char *LinkingOutput) const {
428   const auto &TC =
429       static_cast<const toolchains::CudaToolChain &>(getToolChain());
430   assert(TC.getTriple().isNVPTX() && "Wrong platform");
431 
432   ArgStringList CmdArgs;
433 
434   // OpenMP uses nvlink to link cubin files. The result will be embedded in the
435   // host binary by the host linker.
436   assert(!JA.isHostOffloading(Action::OFK_OpenMP) &&
437          "CUDA toolchain not expected for an OpenMP host device.");
438 
439   if (Output.isFilename()) {
440     CmdArgs.push_back("-o");
441     CmdArgs.push_back(Output.getFilename());
442   } else
443     assert(Output.isNothing() && "Invalid output.");
444   if (Args.hasArg(options::OPT_g_Flag))
445     CmdArgs.push_back("-g");
446 
447   if (Args.hasArg(options::OPT_v))
448     CmdArgs.push_back("-v");
449 
450   StringRef GPUArch =
451       Args.getLastArgValue(options::OPT_march_EQ);
452   assert(!GPUArch.empty() && "At least one GPU Arch required for ptxas.");
453 
454   CmdArgs.push_back("-arch");
455   CmdArgs.push_back(Args.MakeArgString(GPUArch));
456 
457   // Add paths specified in LIBRARY_PATH environment variable as -L options.
458   addDirectoryList(Args, CmdArgs, "-L", "LIBRARY_PATH");
459 
460   // Add paths for the default clang library path.
461   SmallString<256> DefaultLibPath =
462       llvm::sys::path::parent_path(TC.getDriver().Dir);
463   llvm::sys::path::append(DefaultLibPath, "lib" CLANG_LIBDIR_SUFFIX);
464   CmdArgs.push_back(Args.MakeArgString(Twine("-L") + DefaultLibPath));
465 
466   // Add linking against library implementing OpenMP calls on NVPTX target.
467   CmdArgs.push_back("-lomptarget-nvptx");
468 
469   for (const auto &II : Inputs) {
470     if (II.getType() == types::TY_LLVM_IR ||
471         II.getType() == types::TY_LTO_IR ||
472         II.getType() == types::TY_LTO_BC ||
473         II.getType() == types::TY_LLVM_BC) {
474       C.getDriver().Diag(diag::err_drv_no_linker_llvm_support)
475           << getToolChain().getTripleString();
476       continue;
477     }
478 
479     // Currently, we only pass the input files to the linker, we do not pass
480     // any libraries that may be valid only for the host.
481     if (!II.isFilename())
482       continue;
483 
484     const char *CubinF = C.addTempFile(
485         C.getArgs().MakeArgString(getToolChain().getInputFilename(II)));
486 
487     CmdArgs.push_back(CubinF);
488   }
489 
490   AddOpenMPLinkerScript(getToolChain(), C, Output, Inputs, Args, CmdArgs, JA);
491 
492   const char *Exec =
493       Args.MakeArgString(getToolChain().GetProgramPath("nvlink"));
494   C.addCommand(llvm::make_unique<Command>(JA, *this, Exec, CmdArgs, Inputs));
495 }
496 
497 /// CUDA toolchain.  Our assembler is ptxas, and our "linker" is fatbinary,
498 /// which isn't properly a linker but nonetheless performs the step of stitching
499 /// together object files from the assembler into a single blob.
500 
501 CudaToolChain::CudaToolChain(const Driver &D, const llvm::Triple &Triple,
502                              const ToolChain &HostTC, const ArgList &Args,
503                              const Action::OffloadKind OK)
504     : ToolChain(D, Triple, Args), HostTC(HostTC),
505       CudaInstallation(D, HostTC.getTriple(), Args), OK(OK) {
506   if (CudaInstallation.isValid())
507     getProgramPaths().push_back(CudaInstallation.getBinPath());
508   // Lookup binaries into the driver directory, this is used to
509   // discover the clang-offload-bundler executable.
510   getProgramPaths().push_back(getDriver().Dir);
511 }
512 
513 std::string CudaToolChain::getInputFilename(const InputInfo &Input) const {
514   // Only object files are changed, for example assembly files keep their .s
515   // extensions. CUDA also continues to use .o as they don't use nvlink but
516   // fatbinary.
517   if (!(OK == Action::OFK_OpenMP && Input.getType() == types::TY_Object))
518     return ToolChain::getInputFilename(Input);
519 
520   // Replace extension for object files with cubin because nvlink relies on
521   // these particular file names.
522   SmallString<256> Filename(ToolChain::getInputFilename(Input));
523   llvm::sys::path::replace_extension(Filename, "cubin");
524   return Filename.str();
525 }
526 
527 void CudaToolChain::addClangTargetOptions(
528     const llvm::opt::ArgList &DriverArgs,
529     llvm::opt::ArgStringList &CC1Args,
530     Action::OffloadKind DeviceOffloadingKind) const {
531   HostTC.addClangTargetOptions(DriverArgs, CC1Args, DeviceOffloadingKind);
532 
533   StringRef GpuArch = DriverArgs.getLastArgValue(options::OPT_march_EQ);
534   assert(!GpuArch.empty() && "Must have an explicit GPU arch.");
535   assert((DeviceOffloadingKind == Action::OFK_OpenMP ||
536           DeviceOffloadingKind == Action::OFK_Cuda) &&
537          "Only OpenMP or CUDA offloading kinds are supported for NVIDIA GPUs.");
538 
539   if (DeviceOffloadingKind == Action::OFK_Cuda) {
540     CC1Args.push_back("-fcuda-is-device");
541 
542     if (DriverArgs.hasFlag(options::OPT_fcuda_flush_denormals_to_zero,
543                            options::OPT_fno_cuda_flush_denormals_to_zero, false))
544       CC1Args.push_back("-fcuda-flush-denormals-to-zero");
545 
546     if (DriverArgs.hasFlag(options::OPT_fcuda_approx_transcendentals,
547                            options::OPT_fno_cuda_approx_transcendentals, false))
548       CC1Args.push_back("-fcuda-approx-transcendentals");
549 
550     if (DriverArgs.hasFlag(options::OPT_fcuda_rdc, options::OPT_fno_cuda_rdc,
551                            false))
552       CC1Args.push_back("-fcuda-rdc");
553   }
554 
555   if (DriverArgs.hasArg(options::OPT_nocudalib))
556     return;
557 
558   std::string LibDeviceFile = CudaInstallation.getLibDeviceFile(GpuArch);
559 
560   if (LibDeviceFile.empty()) {
561     if (DeviceOffloadingKind == Action::OFK_OpenMP &&
562         DriverArgs.hasArg(options::OPT_S))
563       return;
564 
565     getDriver().Diag(diag::err_drv_no_cuda_libdevice) << GpuArch;
566     return;
567   }
568 
569   CC1Args.push_back("-mlink-cuda-bitcode");
570   CC1Args.push_back(DriverArgs.MakeArgString(LibDeviceFile));
571 
572   if (CudaInstallation.version() >= CudaVersion::CUDA_90) {
573     // CUDA-9 uses new instructions that are only available in PTX6.0
574     CC1Args.push_back("-target-feature");
575     CC1Args.push_back("+ptx60");
576   } else {
577     // Libdevice in CUDA-7.0 requires PTX version that's more recent
578     // than LLVM defaults to. Use PTX4.2 which is the PTX version that
579     // came with CUDA-7.0.
580     CC1Args.push_back("-target-feature");
581     CC1Args.push_back("+ptx42");
582   }
583 }
584 
585 void CudaToolChain::AddCudaIncludeArgs(const ArgList &DriverArgs,
586                                        ArgStringList &CC1Args) const {
587   // Check our CUDA version if we're going to include the CUDA headers.
588   if (!DriverArgs.hasArg(options::OPT_nocudainc) &&
589       !DriverArgs.hasArg(options::OPT_no_cuda_version_check)) {
590     StringRef Arch = DriverArgs.getLastArgValue(options::OPT_march_EQ);
591     assert(!Arch.empty() && "Must have an explicit GPU arch.");
592     CudaInstallation.CheckCudaVersionSupportsArch(StringToCudaArch(Arch));
593   }
594   CudaInstallation.AddCudaIncludeArgs(DriverArgs, CC1Args);
595 }
596 
597 llvm::opt::DerivedArgList *
598 CudaToolChain::TranslateArgs(const llvm::opt::DerivedArgList &Args,
599                              StringRef BoundArch,
600                              Action::OffloadKind DeviceOffloadKind) const {
601   DerivedArgList *DAL =
602       HostTC.TranslateArgs(Args, BoundArch, DeviceOffloadKind);
603   if (!DAL)
604     DAL = new DerivedArgList(Args.getBaseArgs());
605 
606   const OptTable &Opts = getDriver().getOpts();
607 
608   // For OpenMP device offloading, append derived arguments. Make sure
609   // flags are not duplicated.
610   // Also append the compute capability.
611   if (DeviceOffloadKind == Action::OFK_OpenMP) {
612     for (Arg *A : Args) {
613       bool IsDuplicate = false;
614       for (Arg *DALArg : *DAL) {
615         if (A == DALArg) {
616           IsDuplicate = true;
617           break;
618         }
619       }
620       if (!IsDuplicate)
621         DAL->append(A);
622     }
623 
624     StringRef Arch = DAL->getLastArgValue(options::OPT_march_EQ);
625     if (Arch.empty())
626       DAL->AddJoinedArg(nullptr, Opts.getOption(options::OPT_march_EQ),
627                         CLANG_OPENMP_NVPTX_DEFAULT_ARCH);
628 
629     return DAL;
630   }
631 
632   for (Arg *A : Args) {
633     if (A->getOption().matches(options::OPT_Xarch__)) {
634       // Skip this argument unless the architecture matches BoundArch
635       if (BoundArch.empty() || A->getValue(0) != BoundArch)
636         continue;
637 
638       unsigned Index = Args.getBaseArgs().MakeIndex(A->getValue(1));
639       unsigned Prev = Index;
640       std::unique_ptr<Arg> XarchArg(Opts.ParseOneArg(Args, Index));
641 
642       // If the argument parsing failed or more than one argument was
643       // consumed, the -Xarch_ argument's parameter tried to consume
644       // extra arguments. Emit an error and ignore.
645       //
646       // We also want to disallow any options which would alter the
647       // driver behavior; that isn't going to work in our model. We
648       // use isDriverOption() as an approximation, although things
649       // like -O4 are going to slip through.
650       if (!XarchArg || Index > Prev + 1) {
651         getDriver().Diag(diag::err_drv_invalid_Xarch_argument_with_args)
652             << A->getAsString(Args);
653         continue;
654       } else if (XarchArg->getOption().hasFlag(options::DriverOption)) {
655         getDriver().Diag(diag::err_drv_invalid_Xarch_argument_isdriver)
656             << A->getAsString(Args);
657         continue;
658       }
659       XarchArg->setBaseArg(A);
660       A = XarchArg.release();
661       DAL->AddSynthesizedArg(A);
662     }
663     DAL->append(A);
664   }
665 
666   if (!BoundArch.empty()) {
667     DAL->eraseArg(options::OPT_march_EQ);
668     DAL->AddJoinedArg(nullptr, Opts.getOption(options::OPT_march_EQ), BoundArch);
669   }
670   return DAL;
671 }
672 
673 Tool *CudaToolChain::buildAssembler() const {
674   return new tools::NVPTX::Assembler(*this);
675 }
676 
677 Tool *CudaToolChain::buildLinker() const {
678   if (OK == Action::OFK_OpenMP)
679     return new tools::NVPTX::OpenMPLinker(*this);
680   return new tools::NVPTX::Linker(*this);
681 }
682 
683 void CudaToolChain::addClangWarningOptions(ArgStringList &CC1Args) const {
684   HostTC.addClangWarningOptions(CC1Args);
685 }
686 
687 ToolChain::CXXStdlibType
688 CudaToolChain::GetCXXStdlibType(const ArgList &Args) const {
689   return HostTC.GetCXXStdlibType(Args);
690 }
691 
692 void CudaToolChain::AddClangSystemIncludeArgs(const ArgList &DriverArgs,
693                                               ArgStringList &CC1Args) const {
694   HostTC.AddClangSystemIncludeArgs(DriverArgs, CC1Args);
695 }
696 
697 void CudaToolChain::AddClangCXXStdlibIncludeArgs(const ArgList &Args,
698                                                  ArgStringList &CC1Args) const {
699   HostTC.AddClangCXXStdlibIncludeArgs(Args, CC1Args);
700 }
701 
702 void CudaToolChain::AddIAMCUIncludeArgs(const ArgList &Args,
703                                         ArgStringList &CC1Args) const {
704   HostTC.AddIAMCUIncludeArgs(Args, CC1Args);
705 }
706 
707 SanitizerMask CudaToolChain::getSupportedSanitizers() const {
708   // The CudaToolChain only supports sanitizers in the sense that it allows
709   // sanitizer arguments on the command line if they are supported by the host
710   // toolchain. The CudaToolChain will actually ignore any command line
711   // arguments for any of these "supported" sanitizers. That means that no
712   // sanitization of device code is actually supported at this time.
713   //
714   // This behavior is necessary because the host and device toolchains
715   // invocations often share the command line, so the device toolchain must
716   // tolerate flags meant only for the host toolchain.
717   return HostTC.getSupportedSanitizers();
718 }
719 
720 VersionTuple CudaToolChain::computeMSVCVersion(const Driver *D,
721                                                const ArgList &Args) const {
722   return HostTC.computeMSVCVersion(D, Args);
723 }
724