https://github.com/nvptm updated https://github.com/llvm/llvm-project/pull/227775
>From 5e0ec5bb7eab241d05efd70926c5ba7f12a96cc9 Mon Sep 17 00:00:00 2001 From: nvpm <[email protected]> Date: Tue, 22 Sep 2026 12:54:21 -0700 Subject: [PATCH 1/2] [flang][OpenACC] Preserve DO CONCURRENT independence in kernels loops --- clang/include/clang/Options/FlangOptions.td | 3 +++ flang/docs/OpenACC-extensions.md | 11 ++++++++++ flang/include/flang/Lower/LoweringOptions.def | 6 ++++++ flang/lib/Frontend/CompilerInvocation.cpp | 9 ++++++++ flang/lib/Lower/OpenACC.cpp | 21 +++++++++++++++---- .../OpenACC/acc-do-concurrent-locality.f90 | 4 ++-- ...kernels-do-concurrent-independent-flag.f90 | 20 ++++++++++++++++++ 7 files changed, 68 insertions(+), 6 deletions(-) create mode 100644 flang/test/Lower/OpenACC/acc-kernels-do-concurrent-independent-flag.f90 diff --git a/clang/include/clang/Options/FlangOptions.td b/clang/include/clang/Options/FlangOptions.td index 13dfbc3cde8782..5a45528a39a6d5 100644 --- a/clang/include/clang/Options/FlangOptions.td +++ b/clang/include/clang/Options/FlangOptions.td @@ -201,6 +201,9 @@ defm openacc_multiple_names_in_routine : OptOutFC1FFlag<"openacc-multiple-names- defm openacc_combined_loop_firstprivate : OptOutFC1FFlag<"openacc-combined-loop-firstprivate", "Attach firstprivate on combined loop in addition to compute construct (extension)", "Keep firstprivate only on the compute construct for combined loop">; +defm openacc_acc_kernels_do_concurrent_independent : OptOutFC1FFlag<"openacc-acc-kernels-do-concurrent-independent", + "Treat DO CONCURRENT associated with an OpenACC KERNELS LOOP as independent (extension)", + "Treat DO CONCURRENT associated with an OpenACC KERNELS LOOP as auto">; defm prefer_intrinsic_module_use_association : OptOutFC1FFlag<"prefer-intrinsic-module-use-association", "Resolve a USE association conflict in favor of an intrinsic module generic (extension)", "Diagnose a USE association conflict with an intrinsic module generic">; diff --git a/flang/docs/OpenACC-extensions.md b/flang/docs/OpenACC-extensions.md index e397e1f1c936de..aa2987686628b1 100644 --- a/flang/docs/OpenACC-extensions.md +++ b/flang/docs/OpenACC-extensions.md @@ -54,6 +54,17 @@ These extensions require no flag. ## Extensions enabled by default +### `-fopenacc-acc-kernels-do-concurrent-independent` — independent `DO CONCURRENT` in `KERNELS LOOP` + +When a `DO CONCURRENT` is associated with a combined OpenACC `KERNELS LOOP` +construct and no explicit `seq`, `auto`, or `independent` clause is present, +the iteration-independence assertion of `DO CONCURRENT` is preserved and the +loop is treated as if the `independent` clause were present. + +Disable with +`-fno-openacc-acc-kernels-do-concurrent-independent` to treat the loop as +`auto` when no explicit loop parallelism mode is present. + ### `-fopenacc-combined-loop-firstprivate` — combined loop firstprivate `firstprivate` is a compute-construct clause, not a `loop` clause. On a diff --git a/flang/include/flang/Lower/LoweringOptions.def b/flang/include/flang/Lower/LoweringOptions.def index b242749ee2cb3f..4590df1c8a6f8a 100644 --- a/flang/include/flang/Lower/LoweringOptions.def +++ b/flang/include/flang/Lower/LoweringOptions.def @@ -108,5 +108,11 @@ ENUM_LOWERINGOPT(OpenACCCombinedLoopFirstprivate, unsigned, 1, 1) ENUM_LOWERINGOPT(InitLocalMode, Fortran::lower::InitLocalKind, 3, Fortran::lower::InitLocalKind::Off) +/// If true (default), treat a DO CONCURRENT associated with an OpenACC kernels +/// loop as independent when no explicit loop parallelism mode is present. +/// Controlled by +/// -f[no-]openacc-acc-kernels-do-concurrent-independent. +ENUM_LOWERINGOPT(OpenACCKernelsDoConcurrentIndependent, unsigned, 1, 1) + #undef LOWERINGOPT #undef ENUM_LOWERINGOPT diff --git a/flang/lib/Frontend/CompilerInvocation.cpp b/flang/lib/Frontend/CompilerInvocation.cpp index 1f9be77057fb05..9a2a7cd78b32ae 100644 --- a/flang/lib/Frontend/CompilerInvocation.cpp +++ b/flang/lib/Frontend/CompilerInvocation.cpp @@ -1858,6 +1858,15 @@ bool CompilerInvocation::createFromArgs( clang::options::OPT_fno_openacc_combined_loop_firstprivate, /*default=*/true)); + // -f[no-]openacc-acc-kernels-do-concurrent-independent + invoc.loweringOpts.setOpenACCKernelsDoConcurrentIndependent( + args.hasFlag( + clang::options:: + OPT_fopenacc_acc_kernels_do_concurrent_independent, + clang::options:: + OPT_fno_openacc_acc_kernels_do_concurrent_independent, + /*default=*/true)); + if (auto *arg = args.getLastArg(clang::options::OPT_ffp_maxmin_behavior_EQ)) { auto value = Fortran::common::parseFPMaxminBehavior(arg->getValue()); invoc.getCodeGenOpts().setFPMaxminBehavior(value); diff --git a/flang/lib/Lower/OpenACC.cpp b/flang/lib/Lower/OpenACC.cpp index 0193265cf940bb..2c5821a7efe588 100644 --- a/flang/lib/Lower/OpenACC.cpp +++ b/flang/lib/Lower/OpenACC.cpp @@ -1617,7 +1617,8 @@ static void determineDefaultLoopParMode( Fortran::lower::AbstractConverter &converter, mlir::acc::LoopOp &loopOp, llvm::SmallVector<mlir::Attribute> &seqDeviceTypes, llvm::SmallVector<mlir::Attribute> &independentDeviceTypes, - llvm::SmallVector<mlir::Attribute> &autoDeviceTypes) { + llvm::SmallVector<mlir::Attribute> &autoDeviceTypes, + bool kernelsDoConcurrentIsIndependent) { auto hasDeviceNone = [](mlir::Attribute attr) -> bool { return mlir::dyn_cast<mlir::acc::DeviceTypeAttr>(attr).getValue() == mlir::acc::DeviceType::None; @@ -1671,8 +1672,15 @@ static void determineDefaultLoopParMode( // auto clause. assert(mlir::isa_and_present<mlir::acc::KernelsOp>(parentOp) && "Expected kernels construct"); - autoDeviceTypes.push_back(mlir::acc::DeviceTypeAttr::get( - builder.getContext(), mlir::acc::DeviceType::None)); + // By default, preserve the DO CONCURRENT iteration-independence assertion + // when honoring an associated OpenACC kernels loop. An explicit + // seq/auto/independent clause has already returned above. + if (kernelsDoConcurrentIsIndependent) + independentDeviceTypes.push_back(mlir::acc::DeviceTypeAttr::get( + builder.getContext(), mlir::acc::DeviceType::None)); + else + autoDeviceTypes.push_back(mlir::acc::DeviceTypeAttr::get( + builder.getContext(), mlir::acc::DeviceType::None)); } } @@ -2684,8 +2692,13 @@ static mlir::acc::LoopOp createLoopOp( builder.getDenseI32ArrayAttr(tileOperandsSegments)); // Determine the loop's default par mode - either seq, independent, or auto. + const bool kernelsDoConcurrentIsIndependent = + outerDoConstruct.IsDoConcurrent() && + converter.getLoweringOptions() + .getOpenACCKernelsDoConcurrentIndependent(); determineDefaultLoopParMode(converter, loopOp, seqDeviceTypes, - independentDeviceTypes, autoDeviceTypes); + independentDeviceTypes, autoDeviceTypes, + kernelsDoConcurrentIsIndependent); if (!seqDeviceTypes.empty()) loopOp.setSeqAttr(builder.getArrayAttr(seqDeviceTypes)); if (!independentDeviceTypes.empty()) diff --git a/flang/test/Lower/OpenACC/acc-do-concurrent-locality.f90 b/flang/test/Lower/OpenACC/acc-do-concurrent-locality.f90 index 67a9008f975700..da1957af86829e 100644 --- a/flang/test/Lower/OpenACC/acc-do-concurrent-locality.f90 +++ b/flang/test/Lower/OpenACC/acc-do-concurrent-locality.f90 @@ -47,7 +47,7 @@ subroutine reduce_parallel_region() ! CHECK: acc.loop {{.*}}reduction(%[[RED]] : !fir.ref<f32>) ! CHECK: } inclusiveUpperbound(array<i1: true>) independent -! Combined kernels loop with reduce (auto) +! Combined kernels loop with reduce (independent by default extension) ! CHECK-LABEL: func.func @_QPreduce_kernels_loop subroutine reduce_kernels_loop() real :: a(16,16), b(16,16), s @@ -63,7 +63,7 @@ subroutine reduce_kernels_loop() ! CHECK: acc.kernels combined(loop) ! CHECK: %[[RED:.*]] = acc.reduction varPtr(%{{.*}} : !fir.ref<f32>) recipe(@reduction_add{{.*}}) name("s") -> !fir.ref<f32> ! CHECK: acc.loop combined(kernels) {{.*}}reduction(%[[RED]] : !fir.ref<f32>) -! CHECK: } inclusiveUpperbound(array<i1: true, true>) auto_ +! CHECK: } inclusiveUpperbound(array<i1: true, true>) independent ! Combined parallel loop with reduce (independent) ! CHECK-LABEL: func.func @_QPreduce_parallel_loop diff --git a/flang/test/Lower/OpenACC/acc-kernels-do-concurrent-independent-flag.f90 b/flang/test/Lower/OpenACC/acc-kernels-do-concurrent-independent-flag.f90 new file mode 100644 index 00000000000000..5ff04023be9caa --- /dev/null +++ b/flang/test/Lower/OpenACC/acc-kernels-do-concurrent-independent-flag.f90 @@ -0,0 +1,20 @@ +! Test that disabling the kernels-loop DO CONCURRENT independence extension +! lowers the loop as auto. +! +! RUN: %flang_fc1 -fopenacc \ +! RUN: -fno-openacc-acc-kernels-do-concurrent-independent \ +! RUN: -emit-hlfir %s -o - | FileCheck %s + +subroutine kernels_loop_do_concurrent(n, x) + integer :: n, i + real :: x(n) + !$acc kernels loop + do concurrent (i = 1:n) + x(i) = real(i) + end do +end subroutine + +! CHECK-LABEL: func.func @_QPkernels_loop_do_concurrent +! CHECK: acc.kernels combined(loop) +! CHECK: acc.loop combined(kernels) +! CHECK: } inclusiveUpperbound(array<i1: true>) auto_ >From 50007ceee8723946c84358bc0677d4c7010e20c0 Mon Sep 17 00:00:00 2001 From: nvpm <[email protected]> Date: Wed, 30 Sep 2026 09:50:47 -0700 Subject: [PATCH 2/2] Format --- flang/lib/Frontend/CompilerInvocation.cpp | 11 ++++------- flang/lib/Lower/OpenACC.cpp | 3 +-- 2 files changed, 5 insertions(+), 9 deletions(-) diff --git a/flang/lib/Frontend/CompilerInvocation.cpp b/flang/lib/Frontend/CompilerInvocation.cpp index 9a2a7cd78b32ae..48a6f3927f00a4 100644 --- a/flang/lib/Frontend/CompilerInvocation.cpp +++ b/flang/lib/Frontend/CompilerInvocation.cpp @@ -1859,13 +1859,10 @@ bool CompilerInvocation::createFromArgs( /*default=*/true)); // -f[no-]openacc-acc-kernels-do-concurrent-independent - invoc.loweringOpts.setOpenACCKernelsDoConcurrentIndependent( - args.hasFlag( - clang::options:: - OPT_fopenacc_acc_kernels_do_concurrent_independent, - clang::options:: - OPT_fno_openacc_acc_kernels_do_concurrent_independent, - /*default=*/true)); + invoc.loweringOpts.setOpenACCKernelsDoConcurrentIndependent(args.hasFlag( + clang::options::OPT_fopenacc_acc_kernels_do_concurrent_independent, + clang::options::OPT_fno_openacc_acc_kernels_do_concurrent_independent, + /*default=*/true)); if (auto *arg = args.getLastArg(clang::options::OPT_ffp_maxmin_behavior_EQ)) { auto value = Fortran::common::parseFPMaxminBehavior(arg->getValue()); diff --git a/flang/lib/Lower/OpenACC.cpp b/flang/lib/Lower/OpenACC.cpp index 2c5821a7efe588..eec08190aa6a1e 100644 --- a/flang/lib/Lower/OpenACC.cpp +++ b/flang/lib/Lower/OpenACC.cpp @@ -2694,8 +2694,7 @@ static mlir::acc::LoopOp createLoopOp( // Determine the loop's default par mode - either seq, independent, or auto. const bool kernelsDoConcurrentIsIndependent = outerDoConstruct.IsDoConcurrent() && - converter.getLoweringOptions() - .getOpenACCKernelsDoConcurrentIndependent(); + converter.getLoweringOptions().getOpenACCKernelsDoConcurrentIndependent(); determineDefaultLoopParMode(converter, loopOp, seqDeviceTypes, independentDeviceTypes, autoDeviceTypes, kernelsDoConcurrentIsIndependent); _______________________________________________ cfe-commits mailing list [email protected] https://lists.llvm.org/cgi-bin/mailman/listinfo/cfe-commits
