https://github.com/nvptm updated 
https://github.com/llvm/llvm-project/pull/227775

>From 5e0ec5bb7eab241d05efd70926c5ba7f12a96cc9 Mon Sep 17 00:00:00 2001
From: nvpm <[email protected]>
Date: Tue, 22 Sep 2026 12:54:21 -0700
Subject: [PATCH 1/2] [flang][OpenACC] Preserve DO CONCURRENT independence in
 kernels loops

---
 clang/include/clang/Options/FlangOptions.td   |  3 +++
 flang/docs/OpenACC-extensions.md              | 11 ++++++++++
 flang/include/flang/Lower/LoweringOptions.def |  6 ++++++
 flang/lib/Frontend/CompilerInvocation.cpp     |  9 ++++++++
 flang/lib/Lower/OpenACC.cpp                   | 21 +++++++++++++++----
 .../OpenACC/acc-do-concurrent-locality.f90    |  4 ++--
 ...kernels-do-concurrent-independent-flag.f90 | 20 ++++++++++++++++++
 7 files changed, 68 insertions(+), 6 deletions(-)
 create mode 100644 
flang/test/Lower/OpenACC/acc-kernels-do-concurrent-independent-flag.f90

diff --git a/clang/include/clang/Options/FlangOptions.td 
b/clang/include/clang/Options/FlangOptions.td
index 13dfbc3cde8782..5a45528a39a6d5 100644
--- a/clang/include/clang/Options/FlangOptions.td
+++ b/clang/include/clang/Options/FlangOptions.td
@@ -201,6 +201,9 @@ defm openacc_multiple_names_in_routine : 
OptOutFC1FFlag<"openacc-multiple-names-
 defm openacc_combined_loop_firstprivate : 
OptOutFC1FFlag<"openacc-combined-loop-firstprivate",
   "Attach firstprivate on combined loop in addition to compute construct 
(extension)",
   "Keep firstprivate only on the compute construct for combined loop">;
+defm openacc_acc_kernels_do_concurrent_independent : 
OptOutFC1FFlag<"openacc-acc-kernels-do-concurrent-independent",
+  "Treat DO CONCURRENT associated with an OpenACC KERNELS LOOP as independent 
(extension)",
+  "Treat DO CONCURRENT associated with an OpenACC KERNELS LOOP as auto">;
 defm prefer_intrinsic_module_use_association : 
OptOutFC1FFlag<"prefer-intrinsic-module-use-association",
   "Resolve a USE association conflict in favor of an intrinsic module generic 
(extension)",
   "Diagnose a USE association conflict with an intrinsic module generic">;
diff --git a/flang/docs/OpenACC-extensions.md b/flang/docs/OpenACC-extensions.md
index e397e1f1c936de..aa2987686628b1 100644
--- a/flang/docs/OpenACC-extensions.md
+++ b/flang/docs/OpenACC-extensions.md
@@ -54,6 +54,17 @@ These extensions require no flag.
 
 ## Extensions enabled by default
 
+### `-fopenacc-acc-kernels-do-concurrent-independent` — independent `DO 
CONCURRENT` in `KERNELS LOOP`
+
+When a `DO CONCURRENT` is associated with a combined OpenACC `KERNELS LOOP`
+construct and no explicit `seq`, `auto`, or `independent` clause is present,
+the iteration-independence assertion of `DO CONCURRENT` is preserved and the
+loop is treated as if the `independent` clause were present.
+
+Disable with
+`-fno-openacc-acc-kernels-do-concurrent-independent` to treat the loop as
+`auto` when no explicit loop parallelism mode is present.
+
 ### `-fopenacc-combined-loop-firstprivate` — combined loop firstprivate
 
 `firstprivate` is a compute-construct clause, not a `loop` clause.  On a
diff --git a/flang/include/flang/Lower/LoweringOptions.def 
b/flang/include/flang/Lower/LoweringOptions.def
index b242749ee2cb3f..4590df1c8a6f8a 100644
--- a/flang/include/flang/Lower/LoweringOptions.def
+++ b/flang/include/flang/Lower/LoweringOptions.def
@@ -108,5 +108,11 @@ ENUM_LOWERINGOPT(OpenACCCombinedLoopFirstprivate, 
unsigned, 1, 1)
 ENUM_LOWERINGOPT(InitLocalMode, Fortran::lower::InitLocalKind, 3,
                  Fortran::lower::InitLocalKind::Off)
 
+/// If true (default), treat a DO CONCURRENT associated with an OpenACC kernels
+/// loop as independent when no explicit loop parallelism mode is present.
+/// Controlled by
+/// -f[no-]openacc-acc-kernels-do-concurrent-independent.
+ENUM_LOWERINGOPT(OpenACCKernelsDoConcurrentIndependent, unsigned, 1, 1)
+
 #undef LOWERINGOPT
 #undef ENUM_LOWERINGOPT
diff --git a/flang/lib/Frontend/CompilerInvocation.cpp 
b/flang/lib/Frontend/CompilerInvocation.cpp
index 1f9be77057fb05..9a2a7cd78b32ae 100644
--- a/flang/lib/Frontend/CompilerInvocation.cpp
+++ b/flang/lib/Frontend/CompilerInvocation.cpp
@@ -1858,6 +1858,15 @@ bool CompilerInvocation::createFromArgs(
                    clang::options::OPT_fno_openacc_combined_loop_firstprivate,
                    /*default=*/true));
 
+  // -f[no-]openacc-acc-kernels-do-concurrent-independent
+  invoc.loweringOpts.setOpenACCKernelsDoConcurrentIndependent(
+      args.hasFlag(
+          clang::options::
+              OPT_fopenacc_acc_kernels_do_concurrent_independent,
+          clang::options::
+              OPT_fno_openacc_acc_kernels_do_concurrent_independent,
+          /*default=*/true));
+
   if (auto *arg = args.getLastArg(clang::options::OPT_ffp_maxmin_behavior_EQ)) 
{
     auto value = Fortran::common::parseFPMaxminBehavior(arg->getValue());
     invoc.getCodeGenOpts().setFPMaxminBehavior(value);
diff --git a/flang/lib/Lower/OpenACC.cpp b/flang/lib/Lower/OpenACC.cpp
index 0193265cf940bb..2c5821a7efe588 100644
--- a/flang/lib/Lower/OpenACC.cpp
+++ b/flang/lib/Lower/OpenACC.cpp
@@ -1617,7 +1617,8 @@ static void determineDefaultLoopParMode(
     Fortran::lower::AbstractConverter &converter, mlir::acc::LoopOp &loopOp,
     llvm::SmallVector<mlir::Attribute> &seqDeviceTypes,
     llvm::SmallVector<mlir::Attribute> &independentDeviceTypes,
-    llvm::SmallVector<mlir::Attribute> &autoDeviceTypes) {
+    llvm::SmallVector<mlir::Attribute> &autoDeviceTypes,
+    bool kernelsDoConcurrentIsIndependent) {
   auto hasDeviceNone = [](mlir::Attribute attr) -> bool {
     return mlir::dyn_cast<mlir::acc::DeviceTypeAttr>(attr).getValue() ==
            mlir::acc::DeviceType::None;
@@ -1671,8 +1672,15 @@ static void determineDefaultLoopParMode(
     // auto clause.
     assert(mlir::isa_and_present<mlir::acc::KernelsOp>(parentOp) &&
            "Expected kernels construct");
-    autoDeviceTypes.push_back(mlir::acc::DeviceTypeAttr::get(
-        builder.getContext(), mlir::acc::DeviceType::None));
+    // By default, preserve the DO CONCURRENT iteration-independence assertion
+    // when honoring an associated OpenACC kernels loop. An explicit
+    // seq/auto/independent clause has already returned above.
+    if (kernelsDoConcurrentIsIndependent)
+      independentDeviceTypes.push_back(mlir::acc::DeviceTypeAttr::get(
+          builder.getContext(), mlir::acc::DeviceType::None));
+    else
+      autoDeviceTypes.push_back(mlir::acc::DeviceTypeAttr::get(
+          builder.getContext(), mlir::acc::DeviceType::None));
   }
 }
 
@@ -2684,8 +2692,13 @@ static mlir::acc::LoopOp createLoopOp(
         builder.getDenseI32ArrayAttr(tileOperandsSegments));
 
   // Determine the loop's default par mode - either seq, independent, or auto.
+  const bool kernelsDoConcurrentIsIndependent =
+      outerDoConstruct.IsDoConcurrent() &&
+      converter.getLoweringOptions()
+          .getOpenACCKernelsDoConcurrentIndependent();
   determineDefaultLoopParMode(converter, loopOp, seqDeviceTypes,
-                              independentDeviceTypes, autoDeviceTypes);
+                              independentDeviceTypes, autoDeviceTypes,
+                              kernelsDoConcurrentIsIndependent);
   if (!seqDeviceTypes.empty())
     loopOp.setSeqAttr(builder.getArrayAttr(seqDeviceTypes));
   if (!independentDeviceTypes.empty())
diff --git a/flang/test/Lower/OpenACC/acc-do-concurrent-locality.f90 
b/flang/test/Lower/OpenACC/acc-do-concurrent-locality.f90
index 67a9008f975700..da1957af86829e 100644
--- a/flang/test/Lower/OpenACC/acc-do-concurrent-locality.f90
+++ b/flang/test/Lower/OpenACC/acc-do-concurrent-locality.f90
@@ -47,7 +47,7 @@ subroutine reduce_parallel_region()
 ! CHECK: acc.loop {{.*}}reduction(%[[RED]] : !fir.ref<f32>)
 ! CHECK: } inclusiveUpperbound(array<i1: true>) independent
 
-! Combined kernels loop with reduce (auto)
+! Combined kernels loop with reduce (independent by default extension)
 ! CHECK-LABEL: func.func @_QPreduce_kernels_loop
 subroutine reduce_kernels_loop()
   real :: a(16,16), b(16,16), s
@@ -63,7 +63,7 @@ subroutine reduce_kernels_loop()
 ! CHECK: acc.kernels combined(loop)
 ! CHECK: %[[RED:.*]] = acc.reduction varPtr(%{{.*}} : !fir.ref<f32>) 
recipe(@reduction_add{{.*}}) name("s") -> !fir.ref<f32>
 ! CHECK: acc.loop combined(kernels) {{.*}}reduction(%[[RED]] : !fir.ref<f32>)
-! CHECK: } inclusiveUpperbound(array<i1: true, true>) auto_
+! CHECK: } inclusiveUpperbound(array<i1: true, true>) independent
 
 ! Combined parallel loop with reduce (independent)
 ! CHECK-LABEL: func.func @_QPreduce_parallel_loop
diff --git 
a/flang/test/Lower/OpenACC/acc-kernels-do-concurrent-independent-flag.f90 
b/flang/test/Lower/OpenACC/acc-kernels-do-concurrent-independent-flag.f90
new file mode 100644
index 00000000000000..5ff04023be9caa
--- /dev/null
+++ b/flang/test/Lower/OpenACC/acc-kernels-do-concurrent-independent-flag.f90
@@ -0,0 +1,20 @@
+! Test that disabling the kernels-loop DO CONCURRENT independence extension
+! lowers the loop as auto.
+!
+! RUN: %flang_fc1 -fopenacc \
+! RUN:   -fno-openacc-acc-kernels-do-concurrent-independent \
+! RUN:   -emit-hlfir %s -o - | FileCheck %s
+
+subroutine kernels_loop_do_concurrent(n, x)
+  integer :: n, i
+  real :: x(n)
+  !$acc kernels loop
+  do concurrent (i = 1:n)
+    x(i) = real(i)
+  end do
+end subroutine
+
+! CHECK-LABEL: func.func @_QPkernels_loop_do_concurrent
+! CHECK: acc.kernels combined(loop)
+! CHECK: acc.loop combined(kernels)
+! CHECK: } inclusiveUpperbound(array<i1: true>) auto_

>From 50007ceee8723946c84358bc0677d4c7010e20c0 Mon Sep 17 00:00:00 2001
From: nvpm <[email protected]>
Date: Wed, 30 Sep 2026 09:50:47 -0700
Subject: [PATCH 2/2] Format

---
 flang/lib/Frontend/CompilerInvocation.cpp | 11 ++++-------
 flang/lib/Lower/OpenACC.cpp               |  3 +--
 2 files changed, 5 insertions(+), 9 deletions(-)

diff --git a/flang/lib/Frontend/CompilerInvocation.cpp 
b/flang/lib/Frontend/CompilerInvocation.cpp
index 9a2a7cd78b32ae..48a6f3927f00a4 100644
--- a/flang/lib/Frontend/CompilerInvocation.cpp
+++ b/flang/lib/Frontend/CompilerInvocation.cpp
@@ -1859,13 +1859,10 @@ bool CompilerInvocation::createFromArgs(
                    /*default=*/true));
 
   // -f[no-]openacc-acc-kernels-do-concurrent-independent
-  invoc.loweringOpts.setOpenACCKernelsDoConcurrentIndependent(
-      args.hasFlag(
-          clang::options::
-              OPT_fopenacc_acc_kernels_do_concurrent_independent,
-          clang::options::
-              OPT_fno_openacc_acc_kernels_do_concurrent_independent,
-          /*default=*/true));
+  invoc.loweringOpts.setOpenACCKernelsDoConcurrentIndependent(args.hasFlag(
+      clang::options::OPT_fopenacc_acc_kernels_do_concurrent_independent,
+      clang::options::OPT_fno_openacc_acc_kernels_do_concurrent_independent,
+      /*default=*/true));
 
   if (auto *arg = args.getLastArg(clang::options::OPT_ffp_maxmin_behavior_EQ)) 
{
     auto value = Fortran::common::parseFPMaxminBehavior(arg->getValue());
diff --git a/flang/lib/Lower/OpenACC.cpp b/flang/lib/Lower/OpenACC.cpp
index 2c5821a7efe588..eec08190aa6a1e 100644
--- a/flang/lib/Lower/OpenACC.cpp
+++ b/flang/lib/Lower/OpenACC.cpp
@@ -2694,8 +2694,7 @@ static mlir::acc::LoopOp createLoopOp(
   // Determine the loop's default par mode - either seq, independent, or auto.
   const bool kernelsDoConcurrentIsIndependent =
       outerDoConstruct.IsDoConcurrent() &&
-      converter.getLoweringOptions()
-          .getOpenACCKernelsDoConcurrentIndependent();
+      
converter.getLoweringOptions().getOpenACCKernelsDoConcurrentIndependent();
   determineDefaultLoopParMode(converter, loopOp, seqDeviceTypes,
                               independentDeviceTypes, autoDeviceTypes,
                               kernelsDoConcurrentIsIndependent);

_______________________________________________
cfe-commits mailing list
[email protected]
https://lists.llvm.org/cgi-bin/mailman/listinfo/cfe-commits

Reply via email to