alpaka
Abstraction Library for Parallel Kernel Acceleration
Loading...
Searching...
No Matches
TaskKernelCpuOmp2Threads.hpp
Go to the documentation of this file.
1/* Copyright 2026 Axel Huebl, Benjamin Worpitz, Bert Wesarg, René Widera, Jan Stephan, Bernhard Manfred Gruber
2 * SPDX-License-Identifier: MPL-2.0
3 */
4
5#pragma once
6
7// Specialized traits.
10#include "alpaka/dim/Traits.hpp"
11#include "alpaka/idx/Traits.hpp"
13
14// Implementation details.
16#include "alpaka/core/Decay.hpp"
17#include "alpaka/dev/DevCpu.hpp"
23
24#include <functional>
25#include <stdexcept>
26#include <tuple>
27#include <type_traits>
28#if ALPAKA_DEBUG >= ALPAKA_DEBUG_MINIMAL
29# include <iostream>
30#endif
31
32#ifdef ALPAKA_ACC_CPU_B_SEQ_T_OMP2_ENABLED
33
34# if _OPENMP < 200203
35# error If ALPAKA_ACC_CPU_B_SEQ_T_OMP2_ENABLED is set, the compiler has to support OpenMP 2.0 or higher!
36# endif
37
38# include <omp.h>
39
40namespace alpaka
41{
42 //! The CPU OpenMP 2.0 thread accelerator execution task.
43 template<typename TDim, typename TIdx, typename TKernelFnObj, typename... TArgs>
44 class TaskKernelCpuOmp2Threads final : public WorkDivMembers<TDim, TIdx>
45 {
46 public:
47 template<typename TWorkDiv>
48 ALPAKA_FN_HOST TaskKernelCpuOmp2Threads(TWorkDiv&& workDiv, TKernelFnObj const& kernelFnObj, TArgs&&... args)
49 : WorkDivMembers<TDim, TIdx>(std::forward<TWorkDiv>(workDiv))
50 , m_kernelFnObj(kernelFnObj)
51 , m_args(std::forward<TArgs>(args)...)
52 {
53 static_assert(
54 Dim<std::decay_t<TWorkDiv>>::value == TDim::value,
55 "The work division and the execution task have to be of the same dimensionality!");
56 }
57
58 //! Executes the kernel function object.
59 ALPAKA_FN_HOST auto operator()() const -> void
60 {
62
63 auto const gridBlockExtent = getWorkDiv<Grid, Blocks>(*this);
64 auto const blockThreadExtent = getWorkDiv<Block, Threads>(*this);
65 auto const threadElemExtent = getWorkDiv<Thread, Elems>(*this);
66
67 // Get the size of the block shared dynamic memory.
68 auto const blockSharedMemDynSizeBytes = std::apply(
69 [&](std::decay_t<TArgs> const&... args)
70 {
72 m_kernelFnObj,
73 blockThreadExtent,
74 threadElemExtent,
75 args...);
76 },
77 m_args);
78
79# if ALPAKA_DEBUG >= ALPAKA_DEBUG_FULL
80 std::cout << __func__ << " blockSharedMemDynSizeBytes: " << blockSharedMemDynSizeBytes << " B"
81 << std::endl;
82# endif
83
85 *static_cast<WorkDivMembers<TDim, TIdx> const*>(this),
86 blockSharedMemDynSizeBytes);
87
88 // The number of threads in this block.
89 TIdx const blockThreadCount(blockThreadExtent.prod());
90 [[maybe_unused]] int const iBlockThreadCount(static_cast<int>(blockThreadCount));
91
92 if(::omp_in_parallel() != 0)
93 {
94 throw std::runtime_error(
95 "The OpenMP 2.0 thread backend can not be used within an existing parallel region!");
96 }
97
98 // Force the environment to use the given number of threads.
99 int const ompIsDynamic(::omp_get_dynamic());
100 ::omp_set_dynamic(0);
101
102 // Execute the blocks serially.
104 gridBlockExtent,
105 [&](Vec<TDim, TIdx> const& gridBlockIdx)
106 {
107 acc.m_gridBlockIdx = gridBlockIdx;
108
109// Execute the threads in parallel.
110
111// Parallel execution of the threads in a block is required because when syncBlockThreads is called all of them have to
112// be done with their work up to this line. So we have to spawn one OS thread per thread in a block. 'omp for' is not
113// useful because it is meant for cases where multiple iterations are executed by one thread but in our case a 1:1
114// mapping is required. Therefore we use 'omp parallel' with the specified number of threads in a block.
115# pragma omp parallel num_threads(iBlockThreadCount)
116 {
117# pragma omp single nowait
118 {
119 // The OpenMP runtime does not create a parallel region when only one thread is
120 // required in the num_threads clause. In all other cases we expect to be in a parallel
121 // region now.
122 if((iBlockThreadCount > 1) && (::omp_in_parallel() == 0))
123 {
124 throw std::runtime_error("The OpenMP 2.0 runtime did not create a parallel region!");
125 }
126
127 int const numThreads = ::omp_get_num_threads();
128 if(numThreads != iBlockThreadCount)
129 {
130 throw std::runtime_error("The OpenMP 2.0 runtime did not use the number of threads "
131 "that had been required!");
132 }
133 }
134
135 std::apply(
136 [&](auto&&... argsWithAcc) {
138 m_kernelFnObj,
139 std::forward<decltype(argsWithAcc)>(argsWithAcc)...);
140 },
141 std::tuple_cat(std::tie(acc), m_args));
142 std::apply(m_kernelFnObj, std::tuple_cat(std::tie(acc), m_args));
143
144 // Wait for all threads to finish before deleting the shared memory.
145 // This is done by default if the omp 'nowait' clause is missing on the omp parallel directive
146 // syncBlockThreads(acc);
147 }
148
149 // After a block has been processed, the shared memory has to be deleted.
150 freeSharedVars(acc);
151 });
152
153 // Reset the dynamic thread number setting.
154 ::omp_set_dynamic(ompIsDynamic);
155 }
156
157 private:
158 TKernelFnObj m_kernelFnObj;
159 std::tuple<std::decay_t<TArgs>...> m_args;
160 };
161
162 namespace trait
163 {
164 //! The CPU OpenMP 2.0 block thread execution task accelerator type trait specialization.
165 template<typename TDim, typename TIdx, typename TKernelFnObj, typename... TArgs>
166 struct AccType<TaskKernelCpuOmp2Threads<TDim, TIdx, TKernelFnObj, TArgs...>>
167 {
169 };
170
171 //! The CPU OpenMP 2.0 block thread execution task device type trait specialization.
172 template<typename TDim, typename TIdx, typename TKernelFnObj, typename... TArgs>
173 struct DevType<TaskKernelCpuOmp2Threads<TDim, TIdx, TKernelFnObj, TArgs...>>
174 {
175 using type = DevCpu;
176 };
177
178 //! The CPU OpenMP 2.0 block thread execution task dimension getter trait specialization.
179 template<typename TDim, typename TIdx, typename TKernelFnObj, typename... TArgs>
180 struct DimType<TaskKernelCpuOmp2Threads<TDim, TIdx, TKernelFnObj, TArgs...>>
181 {
182 using type = TDim;
183 };
184
185 //! The CPU OpenMP 2.0 block thread execution task platform type trait specialization.
186 template<typename TDim, typename TIdx, typename TKernelFnObj, typename... TArgs>
187 struct PlatformType<TaskKernelCpuOmp2Threads<TDim, TIdx, TKernelFnObj, TArgs...>>
188 {
190 };
191
192 //! The CPU OpenMP 2.0 block thread execution task idx type trait specialization.
193 template<typename TDim, typename TIdx, typename TKernelFnObj, typename... TArgs>
194 struct IdxType<TaskKernelCpuOmp2Threads<TDim, TIdx, TKernelFnObj, TArgs...>>
195 {
196 using type = TIdx;
197 };
198
199 //! \brief Specialisation of the class template FunctionAttributes
200 //! \tparam TDev The device type.
201 //! \tparam TDim The dimensionality of the accelerator device properties.
202 //! \tparam TIdx The idx type of the accelerator device properties.
203 //! \tparam TKernelFn Kernel function object type.
204 //! \tparam TArgs Kernel function object argument types as a parameter pack.
205 template<typename TDev, typename TDim, typename TIdx, typename TKernelFn, typename... TArgs>
206 struct FunctionAttributes<AccCpuOmp2Threads<TDim, TIdx>, TDev, TKernelFn, TArgs...>
207 {
208 //! \param dev The device instance
209 //! \param kernelFn The kernel function object which should be executed.
210 //! \param args The kernel invocation arguments.
211 //! \return KernelFunctionAttributes instance. The default version always returns an instance with zero
212 //! fields. For CPU, the field of max threads allowed by kernel function for the block is 1.
214 TDev const& dev,
215 [[maybe_unused]] TKernelFn const& kernelFn,
216 [[maybe_unused]] TArgs&&... args) -> alpaka::KernelFunctionAttributes
217 {
218 alpaka::KernelFunctionAttributes kernelFunctionAttributes;
219
220 // set function properties for maxThreadsPerBlock to device properties, since API doesn't have function
221 // properties function.
223 kernelFunctionAttributes.maxThreadsPerBlock = static_cast<int>(props.m_blockThreadCountMax);
224 kernelFunctionAttributes.maxDynamicSharedSizeBytes
225 = static_cast<int>(alpaka::BlockSharedDynMemberAllocKiB * 1024);
226 return kernelFunctionAttributes;
227 }
228 };
229
230 //! The CPU OMP2 threads get max active blocks for cooperative kernel specialization.
231 template<typename TDev, typename TKernelFnObj, typename TDim, typename TIdx, typename... TArgs>
232 struct MaxActiveBlocks<AccCpuOmp2Threads<TDim, TIdx>, TDev, TKernelFnObj, TDim, TIdx, TArgs...>
233 {
235 TKernelFnObj const& /*kernelFnObj*/,
236 TDev const& /*device*/,
237 alpaka::Vec<TDim, TIdx> const& /*blockThreadExtent*/,
238 alpaka::Vec<TDim, TIdx> const& /*threadElemExtent*/,
239 TArgs const&... /*args*/) -> int
240 {
241 return 1;
242 }
243 };
244
245 } // namespace trait
246} // namespace alpaka
247
248#endif
#define ALPAKA_DEBUG_MINIMAL_LOG_SCOPE
Definition Debug.hpp:55
The CPU OpenMP 2.0 thread accelerator.
The CPU device handle.
Definition DevCpu.hpp:56
The CPU OpenMP 2.0 thread accelerator execution task.
ALPAKA_FN_HOST TaskKernelCpuOmp2Threads(TWorkDiv &&workDiv, TKernelFnObj const &kernelFnObj, TArgs &&... args)
ALPAKA_FN_HOST auto operator()() const -> void
Executes the kernel function object.
A n-dimensional vector.
Definition Vec.hpp:38
A basic class holding the work division as grid block extent, block thread and thread element extent.
#define ALPAKA_FN_HOST
Definition Common.hpp:43
ALPAKA_FN_HOST_ACC auto checkKernelReturnType(TKernelFnObj const &, TAcc const &, TArgs &&...) -> void
Definition Traits.hpp:372
ALPAKA_NO_HOST_ACC_WARNING ALPAKA_FN_HOST_ACC auto ndLoopIncIdx(TExtentVec const &extent, TFnObj const &f) -> void
Loops over an n-dimensional iteration index variable calling f(idx, args...) for each iteration....
Definition NdLoop.hpp:81
The alpaka accelerator library.
ALPAKA_NO_HOST_ACC_WARNING ALPAKA_FN_HOST_ACC auto getWorkDiv(TWorkDiv const &workDiv) -> Vec< Dim< TWorkDiv >, Idx< TWorkDiv > >
Get the extent requested.
Definition Traits.hpp:33
constexpr std::uint32_t BlockSharedDynMemberAllocKiB
ALPAKA_FN_HOST auto getAccDevProps(TDev const &dev) -> AccDevProps< Dim< TAcc >, Idx< TAcc > >
Definition Traits.hpp:95
ALPAKA_NO_HOST_ACC_WARNING ALPAKA_FN_HOST_ACC auto getBlockSharedMemDynSizeBytes(TKernelFnObj const &kernelFnObj, Vec< TDim, Idx< TAcc > > const &blockThreadExtent, Vec< TDim, Idx< TAcc > > const &threadElemExtent, TArgs const &... args) -> std::size_t
Definition Traits.hpp:197
ALPAKA_NO_HOST_ACC_WARNING ALPAKA_FN_ACC auto freeSharedVars(TBlockSharedMemSt &blockSharedMemSt) -> void
Frees all memory used by block shared variables.
Definition Traits.hpp:54
typename trait::DimType< T >::type Dim
The dimension type trait alias template to remove the ::type.
Definition Traits.hpp:19
STL namespace.
Kernel function attributes struct. Attributes are filled by calling the API of the accelerator using ...
The CPU device platform.
The accelerator type trait.
Definition Traits.hpp:42
The device type trait.
Definition Traits.hpp:23
The dimension getter type trait.
Definition Traits.hpp:14
static ALPAKA_FN_HOST auto getFunctionAttributes(TDev const &dev, TKernelFn const &kernelFn, TArgs &&... args) -> alpaka::KernelFunctionAttributes
The structure template to access to the functions attributes of a kernel function object.
Definition Traits.hpp:94
The idx type trait.
Definition Traits.hpp:25
static ALPAKA_FN_HOST auto getMaxActiveBlocks(TKernelFnObj const &, TDev const &, alpaka::Vec< TDim, TIdx > const &, alpaka::Vec< TDim, TIdx > const &, TArgs const &...) -> int
Get maximum requested blocks for cooperative kernel trait.
Definition Traits.hpp:50
The platform type trait.
Definition Traits.hpp:38