[HIP] Support -fcuda-flush-denormals-to-zero for amdgcn

author Yaxun Liu <Yaxun.Liu@amd.com>

Sat, 21 Jul 2018 02:02:22 +0000 (02:02 +0000)

committer Yaxun Liu <Yaxun.Liu@amd.com>

Sat, 21 Jul 2018 02:02:22 +0000 (02:02 +0000)
author Yaxun Liu <Yaxun.Liu@amd.com>
Sat, 21 Jul 2018 02:02:22 +0000 (02:02 +0000)
committer Yaxun Liu <Yaxun.Liu@amd.com>
Sat, 21 Jul 2018 02:02:22 +0000 (02:02 +0000)
diff --git a/include/clang/Basic/LangOptions.def b/include/clang/Basic/LangOptions.def

index b1ac8557157e76b31c4c7058580e760edda400db..fc38af5b03f6ee6ec9119dd365d3cc3740bb4ba3 100644 (file)
--- a/include/clang/Basic/LangOptions.def
+++ b/include/clang/Basic/LangOptions.def
@@ -209,7 +209,6 @@ LANGOPT(RenderScript      , 1, 0, "RenderScript")
  LANGOPT(CUDAIsDevice      , 1, 0, "compiling for CUDA device")
  LANGOPT(CUDAAllowVariadicFunctions, 1, 0, "allowing variadic functions in CUDA device code")
  LANGOPT(CUDAHostDeviceConstexpr, 1, 1, "treating unattributed constexpr functions as __host__ __device__")
-LANGOPT(CUDADeviceFlushDenormalsToZero, 1, 0, "flushing denormals to zero")
  LANGOPT(CUDADeviceApproxTranscendentals, 1, 0, "using approximate transcendental functions")
  LANGOPT(CUDARelocatableDeviceCode, 1, 0, "generate relocatable device code")
  
diff --git a/lib/CodeGen/CGCall.cpp b/lib/CodeGen/CGCall.cpp

index fcc8a3e5f644898074bfb5f51076843af3b69148..f60136c2f17052d3c701d8717fc0e57f2f35d61e 100644 (file)
--- a/lib/CodeGen/CGCall.cpp
+++ b/lib/CodeGen/CGCall.cpp
@@ -1800,7 +1800,7 @@ void CodeGenModule::ConstructDefaultFnAttrList(StringRef Name, bool HasOptnone,
      FuncAttrs.addAttribute(llvm::Attribute::NoUnwind);
  
      // Respect -fcuda-flush-denormals-to-zero.
-    if (getLangOpts().CUDADeviceFlushDenormalsToZero)
+    if (CodeGenOpts.FlushDenorm)
        FuncAttrs.addAttribute("nvptx-f32ftz", "true");
    }
  }
diff --git a/lib/CodeGen/CodeGenModule.cpp b/lib/CodeGen/CodeGenModule.cpp

index 627a33d8b555887d6afc7166c2d3117068d58eb3..ecdf78d4b3472791fb443571dda135b667c48669 100644 (file)
--- a/lib/CodeGen/CodeGenModule.cpp
+++ b/lib/CodeGen/CodeGenModule.cpp
@@ -526,7 +526,7 @@ void CodeGenModule::Release() {
      // floating point values to 0.  (This corresponds to its "__CUDA_FTZ"
      // property.)
      getModule().addModuleFlag(llvm::Module::Override, "nvvm-reflect-ftz",
-                              LangOpts.CUDADeviceFlushDenormalsToZero ? 1 : 0);
+                              CodeGenOpts.FlushDenorm ? 1 : 0);
    }
  
    // Emit OpenCL specific module metadata: OpenCL/SPIR version.
diff --git a/lib/Frontend/CompilerInvocation.cpp b/lib/Frontend/CompilerInvocation.cpp

index a494b27a39b02fb4423db8abf6dee71fbcc66654..5878cce772b70cd2d0c143e699363b216f051540 100644 (file)
--- a/lib/Frontend/CompilerInvocation.cpp
+++ b/lib/Frontend/CompilerInvocation.cpp
@@ -690,7 +690,9 @@ static bool ParseCodeGenArgs(CodeGenOptions &Opts, ArgList &Args, InputKind IK,
                          Args.hasArg(OPT_cl_unsafe_math_optimizations) ||
                          Args.hasArg(OPT_cl_fast_relaxed_math));
    Opts.Reassociate = Args.hasArg(OPT_mreassociate);
-  Opts.FlushDenorm = Args.hasArg(OPT_cl_denorms_are_zero);
+  Opts.FlushDenorm = Args.hasArg(OPT_cl_denorms_are_zero) ||
+                     (Args.hasArg(OPT_fcuda_is_device) &&
+                      Args.hasArg(OPT_fcuda_flush_denormals_to_zero));
    Opts.CorrectlyRoundedDivSqrt =
        Args.hasArg(OPT_cl_fp32_correctly_rounded_divide_sqrt);
    Opts.UniformWGSize =
@@ -2191,9 +2193,6 @@ static void ParseLangArgs(LangOptions &Opts, ArgList &Args, InputKind IK,
    if (Args.hasArg(OPT_fno_cuda_host_device_constexpr))
      Opts.CUDAHostDeviceConstexpr = 0;
  
-  if (Opts.CUDAIsDevice && Args.hasArg(OPT_fcuda_flush_denormals_to_zero))
-    Opts.CUDADeviceFlushDenormalsToZero = 1;
-
    if (Opts.CUDAIsDevice && Args.hasArg(OPT_fcuda_approx_transcendentals))
      Opts.CUDADeviceApproxTranscendentals = 1;
  
diff --git a/test/CodeGenCUDA/flush-denormals.cu b/test/CodeGenCUDA/flush-denormals.cu

index 94285f1f33cedcf93cfd79b0abc6715c1138eaec..13f2e3b2871e10621578001f9b6569387ae1f053 100644 (file)
--- a/test/CodeGenCUDA/flush-denormals.cu
+++ b/test/CodeGenCUDA/flush-denormals.cu
@@ -5,6 +5,13 @@
  // RUN:   -triple nvptx-nvidia-cuda -emit-llvm -o - %s | \
  // RUN:   FileCheck %s -check-prefix CHECK -check-prefix FTZ
  
+// RUN: %clang_cc1 -fcuda-is-device -x hip \
+// RUN:   -triple amdgcn-amd-amdhsa -target-cpu gfx900 -emit-llvm -o - %s | \
+// RUN:   FileCheck %s -check-prefix CHECK -check-prefix AMDNOFTZ
+// RUN: %clang_cc1 -fcuda-is-device -x hip -fcuda-flush-denormals-to-zero \
+// RUN:   -triple amdgcn-amd-amdhsa -target-cpu gfx900 -emit-llvm -o - %s | \
+// RUN:   FileCheck %s -check-prefix CHECK -check-prefix AMDFTZ
+
  #include "Inputs/cuda.h"
  
  // Checks that device function calls get emitted with the "ntpvx-f32ftz"
@@ -12,11 +19,19 @@
  // -fcuda-flush-denormals-to-zero.  Further, check that we reflect the presence
  // or absence of -fcuda-flush-denormals-to-zero in a module flag.
  
+// AMDGCN targets always have +fp64-fp16-denormals.
+// AMDGCN targets without fast FMAF (e.g. gfx803) always have +fp32-denormals.
+// For AMDGCN target with fast FMAF (e.g. gfx900), it has +fp32-denormals
+// by default and -fp32-denormals when there is option
+// -fcuda-flush-denormals-to-zero.
+
  // CHECK-LABEL: define void @foo() #0
  extern "C" __device__ void foo() {}
  
  // FTZ: attributes #0 = {{.*}} "nvptx-f32ftz"="true"
  // NOFTZ-NOT: attributes #0 = {{.*}} "nvptx-f32ftz"
+// AMDNOFTZ: attributes #0 = {{.*}}+fp32-denormals{{.*}}+fp64-fp16-denormals
+// AMDFTZ: attributes #0 = {{.*}}+fp64-fp16-denormals{{.*}}-fp32-denormals
  
  // FTZ:!llvm.module.flags = !{{{.*}}[[MODFLAG:![0-9]+]]}
  // FTZ:[[MODFLAG]] = !{i32 4, !"nvvm-reflect-ftz", i32 1}
author	Yaxun Liu <Yaxun.Liu@amd.com>
	Sat, 21 Jul 2018 02:02:22 +0000 (02:02 +0000)
committer	Yaxun Liu <Yaxun.Liu@amd.com>
	Sat, 21 Jul 2018 02:02:22 +0000 (02:02 +0000)
include/clang/Basic/LangOptions.def		patch \| blob \| history
lib/CodeGen/CGCall.cpp		patch \| blob \| history
lib/CodeGen/CodeGenModule.cpp		patch \| blob \| history
lib/Frontend/CompilerInvocation.cpp		patch \| blob \| history
test/CodeGenCUDA/flush-denormals.cu		patch \| blob \| history