19#ifndef OPM_GPUBUFFER_HEADER_HPP
20#define OPM_GPUBUFFER_HEADER_HPP
28#include <opm/common/ErrorMacros.hpp>
30#include <cuda_runtime.h>
31#include <dune/common/fvector.hh>
32#include <dune/istl/bvector.hh>
33#include <fmt/format.h>
70 using size_type = size_t;
76 GpuBuffer() =
default;
86 GpuBuffer(
const GpuBuffer<T>& other)
87 : GpuBuffer(other.m_numberOfElements)
89 assertSameSize(other);
90 if (m_numberOfElements == 0) {
104 GpuBuffer(GpuBuffer<T>&& other) noexcept
105 : m_dataOnDevice(other.m_dataOnDevice)
106 , m_numberOfElements(other.m_numberOfElements)
108 other.m_dataOnDevice =
nullptr;
109 other.m_numberOfElements = 0;
120 GpuBuffer<T>& operator=(GpuBuffer<T>&& other)
noexcept
122 if (
this != &other) {
124 m_dataOnDevice = other.m_dataOnDevice;
125 m_numberOfElements = other.m_numberOfElements;
126 other.m_dataOnDevice =
nullptr;
127 other.m_numberOfElements = 0;
141 explicit GpuBuffer(
const std::vector<T>& data)
142 : GpuBuffer(data.size())
152 explicit GpuBuffer(
const size_t numberOfElements)
153 : m_numberOfElements(numberOfElements)
168 GpuBuffer(
const T* dataOnHost,
const size_t numberOfElements)
169 : GpuBuffer(numberOfElements)
188 return m_dataOnDevice;
194 const T* data()
const
196 return m_dataOnDevice;
206 template <
int BlockDimension>
207 void copyFromHost(
const Dune::BlockVector<Dune::FieldVector<T, BlockDimension>>& bvector)
209 if (m_numberOfElements != bvector.dim()) {
210 OPM_THROW(std::runtime_error,
211 fmt::format(
"Given incompatible vector size. GpuBuffer has size {},\n however, the BlockVector "
212 "has dim() = {} (N() = {}, and size() = {}).",
218 const auto dataPointer =
static_cast<const T*
>(&(bvector[0][0]));
219 copyFromHost(dataPointer, m_numberOfElements);
229 template <
int BlockDimension>
230 void copyToHost(Dune::BlockVector<Dune::FieldVector<T, BlockDimension>>& bvector)
const
232 if (m_numberOfElements != bvector.dim()) {
233 OPM_THROW(std::runtime_error,
234 fmt::format(
"Given incompatible vector size. GpuBuffer has size {},\n however, the BlockVector "
235 "has dim() = {} (N() = {}, and size() = {}).",
241 const auto dataPointer =
static_cast<T*
>(&(bvector[0][0]));
242 copyToHost(dataPointer, m_numberOfElements);
252 void copyFromHost(
const T* dataPointer,
size_t numberOfElements)
254 if (numberOfElements > size()) {
255 OPM_THROW(std::runtime_error,
256 fmt::format(
"Requesting to copy too many elements. Buffer has {} elements, while {} was requested.",
270 void copyToHost(T* dataPointer,
size_t numberOfElements)
const
272 assertSameSize(numberOfElements);
283 void copyFromHost(
const std::vector<T>& data)
285 assertSameSize(data.size());
291 if constexpr (std::is_same_v<T, bool>)
293 auto tmp = std::make_unique<bool[]>(data.size());
294 for (
size_t i = 0; i < data.size(); ++i) {
295 tmp[i] =
static_cast<bool>(data[i]);
297 copyFromHost(tmp.get(), data.size());
300 copyFromHost(data.data(), data.size());
311 void copyToHost(std::vector<T>& data)
const
313 assertSameSize(data.size());
319 if constexpr (std::is_same_v<T, bool>)
321 auto tmp = std::make_unique<bool[]>(data.size());
322 copyToHost(tmp.get(), data.size());
323 for (
size_t i = 0; i < data.size(); ++i) {
324 data[i] =
static_cast<bool>(tmp[i]);
329 copyToHost(data.data(), data.size());
343 void copyFromHostAsync(
const T* dataPointer,
size_t numberOfElements, cudaStream_t stream =
detail::DEFAULT_STREAM)
345 if (numberOfElements > size()) {
346 OPM_THROW(std::runtime_error,
347 fmt::format(
"Requesting to copy too many elements. Buffer has {} elements, while {} was requested.",
365 void copyToHostAsync(T* dataPointer,
size_t numberOfElements, cudaStream_t stream =
detail::DEFAULT_STREAM)
const
367 assertSameSize(numberOfElements);
375 size_type size()
const
377 return m_numberOfElements;
384 void resize(
size_t newSize)
387 OPM_THROW(std::invalid_argument,
"Setting a GpuBuffer size to a non-positive number is not allowed");
390 if (newSize == m_numberOfElements) {
394 if (m_numberOfElements == 0) {
400 T* tmpBuffer =
nullptr;
404 size_t sizeOfMove = std::min({m_numberOfElements, newSize});
411 m_dataOnDevice = tmpBuffer;
415 m_numberOfElements = newSize;
422 std::vector<T> asStdVector()
const
424 std::vector<T> temporary(m_numberOfElements);
425 copyToHost(temporary);
430 T* m_dataOnDevice =
nullptr;
431 size_t m_numberOfElements = 0;
433 void assertSameSize(
const GpuBuffer<T>& other)
const
435 assertSameSize(other.m_numberOfElements);
438 void assertSameSize(
size_t size)
const
440 if (size != m_numberOfElements) {
441 OPM_THROW(std::invalid_argument,
442 fmt::format(fmt::runtime(
"Given buffer has {}, while we have {}."),
443 size, m_numberOfElements));
447 void assertHasElements()
const
449 if (m_numberOfElements <= 0) {
450 OPM_THROW(std::invalid_argument,
"We have 0 elements");
457 return GpuView<T>(buf.data(), buf.size());
461GpuView<const T>
make_view(
const GpuBuffer<T>& buf) {
462 return GpuView<const T>(buf.data(), buf.size());
#define OPM_GPU_SAFE_CALL(expression)
OPM_GPU_SAFE_CALL checks the return type of the GPU expression (function call) and throws an exceptio...
Definition: gpu_safe_call.hpp:164
#define OPM_GPU_WARN_IF_ERROR(expression)
OPM_GPU_WARN_IF_ERROR checks the return type of the GPU expression (function call) and issues a warni...
Definition: gpu_safe_call.hpp:185
void gpuMemcpyHostToDeviceAsync(T *dstDevice, const T *srcHost, std::size_t count, cudaStream_t stream)
gpuMemcpyHostToDeviceAsync copies count elements of type T from host to device asynchronously.
Definition: gpu_memcpy.hpp:123
void gpuMemcpyHostToDevice(T *dstDevice, const T *srcHost, std::size_t count)
gpuMemcpyHostToDevice copies count elements of type T from host to device.
Definition: gpu_memcpy.hpp:46
void gpuMemcpyDeviceToHostAsync(T *dstHost, const T *srcDevice, std::size_t count, cudaStream_t stream)
gpuMemcpyDeviceToHostAsync copies count elements of type T from device to host asynchronously.
Definition: gpu_memcpy.hpp:155
constexpr cudaStream_t DEFAULT_STREAM
The default GPU stream (stream 0)
Definition: gpu_constants.hpp:31
void gpuMemcpyDeviceToHost(T *dstHost, const T *srcDevice, std::size_t count)
gpuMemcpyDeviceToHost copies count elements of type T from device to host.
Definition: gpu_memcpy.hpp:69
void gpuMemcpyDeviceToDevice(T *dstDevice, const T *srcDevice, std::size_t count)
gpuMemcpyDeviceToDevice copies count elements of type T from device to device.
Definition: gpu_memcpy.hpp:92
A small, fixed‑dimension MiniVector class backed by std::array that can be used in both host and CUDA...
Definition: GpuFlowProblem.hpp:48
inline ::Opm::NoThermalLawManager make_view(::Opm::NoThermalLawManager &)
make_view overload for the no-op thermal manager.
Definition: GpuFlowProblem.hpp:101