import psutil
import pystencils as ps
import pystencils.plot as psplot
import numpy as np
import sympy as sp
import shutil
Demo: Finite differences - 2D wave equation#
In this tutorial we show how to use the finite difference module of pystencils to solve a 2D wave equations. The time derivative is discretized by a simple forward Euler method.
We begin by creating three numpy arrays for the current, the previous and the next timestep. Actually we will see later that two fields are enough, but let’s keep it simple. From these numpy arrays we create pystencils fields to formulate our update rule.
size = (60, 70) # domain size
u_arrays = [np.zeros(size), np.zeros(size), np.zeros(size)]
u_fields = [ps.Field.create_from_numpy_array("u%s" % (name,), arr)
for name, arr in zip(["0", "1", "2"], u_arrays)]
# Nicer display for fields
for i, field in enumerate(u_fields):
field.latex_name = "u^{(%d)}" % (i,)
pystencils contains already simple rules to discretize the a diffusion term. The time discretization is done manually.
discretize = ps.fd.Discretization2ndOrder()
def central2nd_time_derivative(fields):
f_next, f_current, f_last = fields
return (f_next[0, 0] - 2 * f_current[0, 0] + f_last[0, 0]) / discretize.dt**2
rhs = ps.fd.diffusion(u_fields[1], 1)
wave_eq = sp.Eq(central2nd_time_derivative(u_fields), discretize(rhs))
wave_eq = sp.simplify(wave_eq)
wave_eq
The explicit Euler scheme is now obtained by solving above equation with respect to \(u_C^{next}\).
u_next_C = u_fields[-1][0, 0]
update_rule = ps.Assignment(u_next_C, sp.solve(wave_eq, u_next_C)[0])
update_rule
Before creating the kernel, we substitute numeric values for \(dx\) and \(dt\). Then a kernel is created just like in the last tutorial.
update_rule = update_rule.subs({discretize.dx: 0.1, discretize.dt: 0.05})
ast = ps.create_kernel(update_rule)
kernel = ast.compile()
ps.inspect(ast)
void kernel (const double * RESTRICT const _data_u0, const double * RESTRICT const _data_u1, double * RESTRICT const _data_u2)
{
for(ctr_0: int64 = [1: int64]; ctr_0 < [59: int64]; ctr_0 += [1: int64])
{
for(ctr_1: int64 = [1: int64]; ctr_1 < [69: int64]; ctr_1 += [1: int64])
{
_data_u2[ctr_0 * [70: int64] + ctr_1] = [0.25: float64] * _data_u1[(ctr_0 + [1: int64]) * [70: int64] + ctr_1] + [0.25: float64] * _data_u1[ctr_0 * [70: int64] + (ctr_1 + [1: int64])] + [0.25: float64] * _data_u1[ctr_0 * [70: int64] + (ctr_1 + [-1: int64])] + [0.25: float64] * _data_u1[(ctr_0 + [-1: int64]) * [70: int64] + ctr_1] + _data_u1[ctr_0 * [70: int64] + ctr_1] + [-1.0: float64] * _data_u0[ctr_0 * [70: int64] + ctr_1];
}
}
}
To run simulation a suitable initial condition and boundary treatment is required. We chose an initial condition which is zero at the borders of the domain. The outermost layer is not changed by the update kernel, so we have an implicit homogenous Dirichlet boundary condition.
X,Y = np.meshgrid( np.linspace(0, 1, size[1]), np.linspace(0,1, size[0]))
Z = np.sin(2*X*np.pi) * np.sin(2*Y*np.pi)
# Initialize the previous and current values with the initial function
np.copyto(u_arrays[0], Z)
np.copyto(u_arrays[1], Z)
# The values for the next timesteps do not matter, since they are overwritten
u_arrays[2][:, :] = 0
One timestep now consists of applying the kernel once, then shifting the arrays.
def run(timesteps=1):
for t in range(timesteps):
kernel(u0=u_arrays[0], u1=u_arrays[1], u2=u_arrays[2])
u_arrays[0], u_arrays[1], u_arrays[2] = u_arrays[1], u_arrays[2], u_arrays[0]
return u_arrays[2]
Lets create an animation of the solution:
ani = psplot.surface_plot_animation(run, zlim=(-1, 1))
from pystencils.jupyter import display_as_html_video
display_as_html_video(ani)
assert np.isfinite(np.max(u_arrays[2]))
Runing on GPU
We can also run the same kernel on the GPU, by using the cupy package.
try:
import cupy
except ImportError:
cupy=None
print('No cupy installed')
res = None
if cupy:
gpu_ast = ps.create_kernel(update_rule, target=ps.Target.GPU)
gpu_kernel = gpu_ast.compile()
res = ps.inspect(gpu_ast)
res
__global__ void kernel (const double * RESTRICT const _data_u0, const double * RESTRICT const _data_u1, double * RESTRICT const _data_u2)
{
ctr_0: const int64 = [1: int64] + ((int64) blockIdx.y * (int64) blockDim.y + (int64) threadIdx.y);
ctr_1: const int64 = [1: int64] + ((int64) blockIdx.x * (int64) blockDim.x + (int64) threadIdx.x);
if(ctr_0 < [59: int64] && ctr_1 < [69: int64])
{
_data_u2[ctr_0 * [70: int64] + ctr_1] = [0.25: float64] * _data_u1[(ctr_0 + [1: int64]) * [70: int64] + ctr_1] + [0.25: float64] * _data_u1[ctr_0 * [70: int64] + (ctr_1 + [1: int64])] + [0.25: float64] * _data_u1[ctr_0 * [70: int64] + (ctr_1 + [-1: int64])] + [0.25: float64] * _data_u1[(ctr_0 + [-1: int64]) * [70: int64] + ctr_1] + _data_u1[ctr_0 * [70: int64] + ctr_1] + [-1.0: float64] * _data_u0[ctr_0 * [70: int64] + ctr_1];
}
}
The run function has to be changed now slightly, since the data has to be transfered to the GPU first, then the kernel can be executed, and in the end the data has to be transfered back
if cupy:
def run_on_gpu(timesteps=1):
# Transfer arrays to GPU
gpuArrs = [cupy.asarray(cpu_array) for cpu_array in u_arrays]
for t in range(timesteps):
gpu_kernel(u0=gpuArrs[0], u1=gpuArrs[1], u2=gpuArrs[2])
gpuArrs[0], gpuArrs[1], gpuArrs[2] = gpuArrs[1], gpuArrs[2], gpuArrs[0]
# Transfer arrays to CPU
for gpuArr, cpuArr in zip(gpuArrs, u_arrays):
cpuArr[:] = gpuArr.get()
assert np.isfinite(np.max(u_arrays[2]))
if cupy:
run_on_gpu(400)
---------------------------------------------------------------------------
NVRTCError Traceback (most recent call last)
File /pycodegen-dev/.venv/lib/python3.12/site-packages/cupy/cuda/compiler.py:860, in _NVRTCProgram.compile(self, options, log_stream)
859 nvrtc.addNameExpression(self.ptr, ker)
--> 860 nvrtc.compileProgram(self.ptr, options)
861 mapping = None
File cupy_backends/cuda/libs/nvrtc.pyx:125, in cupy_backends.cuda.libs.nvrtc.compileProgram()
--> 125 'Could not get source, probably due dynamically evaluated source code.'
File cupy_backends/cuda/libs/nvrtc.pyx:138, in cupy_backends.cuda.libs.nvrtc.compileProgram()
--> 138 'Could not get source, probably due dynamically evaluated source code.'
File cupy_backends/cuda/libs/nvrtc.pyx:53, in cupy_backends.cuda.libs.nvrtc.check_status()
---> 53 'Could not get source, probably due dynamically evaluated source code.'
NVRTCError: HIPRTC_ERROR_COMPILATION (6)
During handling of the above exception, another exception occurred:
CompileException Traceback (most recent call last)
Cell In[12], line 2
1 if cupy:
----> 2 run_on_gpu(400)
Cell In[11], line 7, in run_on_gpu(timesteps)
3 # Transfer arrays to GPU
4 gpuArrs = [cupy.asarray(cpu_array) for cpu_array in u_arrays]
5
6 for t in range(timesteps):
----> 7 gpu_kernel(u0=gpuArrs[0], u1=gpuArrs[1], u2=gpuArrs[2])
8 gpuArrs[0], gpuArrs[1], gpuArrs[2] = gpuArrs[1], gpuArrs[2], gpuArrs[0]
9
10 # Transfer arrays to CPU
File /pycodegen-dev/pystencils/src/pystencils/jit/jit.py:29, in KernelWrapper.__call__(self, **kwargs)
27 for name, mapper in self._view_mappings.items():
28 kwargs[name] = mapper.view_ndarray(kwargs[name])
---> 29 self._invocable(**kwargs)
File /pycodegen-dev/pystencils/src/pystencils/jit/gpu_cupy.py:62, in CupyKernelWrapper._invoke(self, **kwargs)
60 device = self._get_device(kernel_args)
61 with cp.cuda.Device(device):
---> 62 self._raw_kernel(launch_grid.grid, launch_grid.block, kernel_args)
File cupy/_core/raw.pyx:101, in cupy._core.raw.RawKernel.__call__()
--> 101 'Could not get source, probably due dynamically evaluated source code.'
File cupy/_core/raw.pyx:108, in cupy._core.raw.RawKernel.kernel.__get__()
--> 108 'Could not get source, probably due dynamically evaluated source code.'
File cupy/_core/raw.pyx:125, in cupy._core.raw.RawKernel._kernel()
--> 125 'Could not get source, probably due dynamically evaluated source code.'
File cupy/_util.pyx:67, in cupy._util.memoize.decorator.ret()
---> 67 'Could not get source, probably due dynamically evaluated source code.'
File cupy/_core/raw.pyx:549, in cupy._core.raw._get_raw_module()
--> 549 'Could not get source, probably due dynamically evaluated source code.'
File cupy/_core/core.pyx:2621, in cupy._core.core.compile_with_cache()
-> 2621 'Could not get source, probably due dynamically evaluated source code.'
File cupy/_core/core.pyx:2639, in cupy._core.core.compile_with_cache()
-> 2639 'Could not get source, probably due dynamically evaluated source code.'
File /pycodegen-dev/.venv/lib/python3.12/site-packages/cupy/cuda/compiler.py:645, in _compile_module_with_cache(source, options, arch, extra_source, backend, enable_cooperative_groups, name_expressions, log_stream, jitify, to_ltoir)
643 if runtime.is_hip:
644 backend = 'hiprtc' if backend == 'nvrtc' else 'hipcc'
--> 645 return _compile_with_cache_hip(
646 source, options, arch, extra_source, backend,
647 name_expressions, log_stream, cache_in_memory)
648 else:
649 return _compile_with_cache_cuda(
650 source, options, arch, extra_source, backend,
651 enable_cooperative_groups, name_expressions, log_stream,
652 cache_in_memory, jitify, to_ltoir)
File /pycodegen-dev/.venv/lib/python3.12/site-packages/cupy/cuda/compiler.py:1072, in _compile_with_cache_hip(source, options, arch, extra_source, backend, name_expressions, log_stream, cache_in_memory, use_converter)
1068 pass
1070 if backend == 'hiprtc':
1071 # compile_using_nvrtc calls hiprtc for hip builds
-> 1072 binary, mapping = compile_using_nvrtc(
1073 source, options, arch, name + '.cu', name_expressions,
1074 log_stream, cache_in_memory)
1075 mod._set_mapping(mapping)
1076 else:
File /pycodegen-dev/.venv/lib/python3.12/site-packages/cupy/cuda/compiler.py:414, in compile_using_nvrtc(source, options, arch, filename, name_expressions, log_stream, cache_in_memory, jitify, method)
411 if jitify is not None:
412 _jitify_deprecation_warning(jitify)
--> 414 return _compile_using_nvrtc_no_warning(
415 source, options, arch, filename, name_expressions, log_stream,
416 cache_in_memory, jitify, method)
File /pycodegen-dev/.venv/lib/python3.12/site-packages/cupy/cuda/compiler.py:402, in _compile_using_nvrtc_no_warning(source, options, arch, filename, name_expressions, log_stream, cache_in_memory, jitify, method)
399 else:
400 cu_path = '' if not jitify else filename
--> 402 return _compile(source, options, cu_path, name_expressions,
403 log_stream, jitify, method)
File /pycodegen-dev/.venv/lib/python3.12/site-packages/cupy/cuda/compiler.py:384, in _compile_using_nvrtc_no_warning.<locals>._compile(source, options, cu_path, name_expressions, log_stream, jitify, method)
381 prog = _NVRTCProgram(source, cu_path, headers, include_names,
382 name_expressions=name_expressions, method=method)
383 try:
--> 384 compiled_obj, mapping = prog.compile(options, log_stream)
385 except CompileException as e:
386 dump = _get_bool_env_variable(
387 'CUPY_DUMP_CUDA_SOURCE_ON_ERROR', False)
File /pycodegen-dev/.venv/lib/python3.12/site-packages/cupy/cuda/compiler.py:879, in _NVRTCProgram.compile(self, options, log_stream)
877 except nvrtc.NVRTCError:
878 log = nvrtc.getProgramLog(self.ptr)
--> 879 raise CompileException(log, self.src, self.name, options,
880 'nvrtc' if not runtime.is_hip else 'hiprtc')
CompileException: In file included from /tmp/comgr-345b3a/input/tmp/tmpnz1gjn5f/c8573e542c59846fdedfebf89b822ea2cba126fc.hsaco.cu:1:
In file included from /pycodegen-dev/pystencils/src/pystencils/include/pystencils_runtime/generic_gpu.hpp:4:
In file included from /pycodegen-dev/pystencils/src/pystencils/include/pystencils_runtime/./random.hpp:22:
/pycodegen-dev/pystencils/src/pystencils/include/pystencils_runtime/././bits/philox_rand.h:113:9: error: no type named 'uint32_t' in namespace 'std'; did you mean '__hip_internal::uint32_t'?
113 | typedef std::uint32_t uint32;
| ^~~~~~~~~~~~~
| __hip_internal::uint32_t
/tmp/comgr-345b3a/include/hiprtc_runtime.h:1427:22: note: '__hip_internal::uint32_t' declared here
1427 | typedef unsigned int uint32_t;
| ^
In file included from /tmp/comgr-345b3a/input/tmp/tmpnz1gjn5f/c8573e542c59846fdedfebf89b822ea2cba126fc.hsaco.cu:1:
In file included from /pycodegen-dev/pystencils/src/pystencils/include/pystencils_runtime/generic_gpu.hpp:4:
In file included from /pycodegen-dev/pystencils/src/pystencils/include/pystencils_runtime/./random.hpp:22:
/pycodegen-dev/pystencils/src/pystencils/include/pystencils_runtime/././bits/philox_rand.h:114:9: error: no type named 'uint64_t' in namespace 'std'; did you mean '__hip_internal::uint64_t'?
114 | typedef std::uint64_t uint64;
| ^~~~~~~~~~~~~
| __hip_internal::uint64_t
/tmp/comgr-345b3a/include/hiprtc_runtime.h:1428:28: note: '__hip_internal::uint64_t' declared here
1428 | typedef unsigned long long uint64_t;
| ^
In file included from /tmp/comgr-345b3a/input/tmp/tmpnz1gjn5f/c8573e542c59846fdedfebf89b822ea2cba126fc.hsaco.cu:1:
In file included from /pycodegen-dev/pystencils/src/pystencils/include/pystencils_runtime/generic_gpu.hpp:4:
/pycodegen-dev/pystencils/src/pystencils/include/pystencils_runtime/./random.hpp:32:14: error: unknown type name 'uint32_t'; did you mean '__hip_internal::uint32_t'?
32 | (uint32_t)ctr0, (uint32_t)ctr1, (uint32_t)ctr2, (uint32_t)ctr3, //
| ^~~~~~~~
| __hip_internal::uint32_t
/tmp/comgr-345b3a/include/hiprtc_runtime.h:1427:22: note: '__hip_internal::uint32_t' declared here
1427 | typedef unsigned int uint32_t;
| ^
In file included from /tmp/comgr-345b3a/input/tmp/tmpnz1gjn5f/c8573e542c59846fdedfebf89b822ea2cba126fc.hsaco.cu:1:
In file included from /pycodegen-dev/pystencils/src/pystencils/include/pystencils_runtime/generic_gpu.hpp:4:
/pycodegen-dev/pystencils/src/pystencils/include/pystencils_runtime/./random.hpp:32:30: error: unknown type name 'uint32_t'; did you mean '__hip_internal::uint32_t'?
32 | (uint32_t)ctr0, (uint32_t)ctr1, (uint32_t)ctr2, (uint32_t)ctr3, //
| ^~~~~~~~
| __hip_internal::uint32_t
/tmp/comgr-345b3a/include/hiprtc_runtime.h:1427:22: note: '__hip_internal::uint32_t' declared here
1427 | typedef unsigned int uint32_t;
| ^
In file included from /tmp/comgr-345b3a/input/tmp/tmpnz1gjn5f/c8573e542c59846fdedfebf89b822ea2cba126fc.hsaco.cu:1:
In file included from /pycodegen-dev/pystencils/src/pystencils/include/pystencils_runtime/generic_gpu.hpp:4:
/pycodegen-dev/pystencils/src/pystencils/include/pystencils_runtime/./random.hpp:32:46: error: unknown type name 'uint32_t'; did you mean '__hip_internal::uint32_t'?
32 | (uint32_t)ctr0, (uint32_t)ctr1, (uint32_t)ctr2, (uint32_t)ctr3, //
| ^~~~~~~~
| __hip_internal::uint32_t
/tmp/comgr-345b3a/include/hiprtc_runtime.h:1427:22: note: '__hip_internal::uint32_t' declared here
1427 | typedef unsigned int uint32_t;
| ^
In file included from /tmp/comgr-345b3a/input/tmp/tmpnz1gjn5f/c8573e542c59846fdedfebf89b822ea2cba126fc.hsaco.cu:1:
In file included from /pycodegen-dev/pystencils/src/pystencils/include/pystencils_runtime/generic_gpu.hpp:4:
/pycodegen-dev/pystencils/src/pystencils/include/pystencils_runtime/./random.hpp:32:62: error: unknown type name 'uint32_t'; did you mean '__hip_internal::uint32_t'?
32 | (uint32_t)ctr0, (uint32_t)ctr1, (uint32_t)ctr2, (uint32_t)ctr3, //
| ^~~~~~~~
| __hip_internal::uint32_t
/tmp/comgr-345b3a/include/hiprtc_runtime.h:1427:22: note: '__hip_internal::uint32_t' declared here
1427 | typedef unsigned int uint32_t;
| ^
In file included from /tmp/comgr-345b3a/input/tmp/tmpnz1gjn5f/c8573e542c59846fdedfebf89b822ea2cba126fc.hsaco.cu:1:
In file included from /pycodegen-dev/pystencils/src/pystencils/include/pystencils_runtime/generic_gpu.hpp:4:
/pycodegen-dev/pystencils/src/pystencils/include/pystencils_runtime/./random.hpp:33:14: error: unknown type name 'uint32_t'; did you mean '__hip_internal::uint32_t'?
33 | (uint32_t)key0, (uint32_t)key1, //
| ^~~~~~~~
| __hip_internal::uint32_t
/tmp/comgr-345b3a/include/hiprtc_runtime.h:1427:22: note: '__hip_internal::uint32_t' declared here
1427 | typedef unsigned int uint32_t;
| ^
In file included from /tmp/comgr-345b3a/input/tmp/tmpnz1gjn5f/c8573e542c59846fdedfebf89b822ea2cba126fc.hsaco.cu:1:
In file included from /pycodegen-dev/pystencils/src/pystencils/include/pystencils_runtime/generic_gpu.hpp:4:
/pycodegen-dev/pystencils/src/pystencils/include/pystencils_runtime/./random.hpp:33:30: error: unknown type name 'uint32_t'; did you mean '__hip_internal::uint32_t'?
33 | (uint32_t)key0, (uint32_t)key1, //
| ^~~~~~~~
| __hip_internal::uint32_t
/tmp/comgr-345b3a/include/hiprtc_runtime.h:1427:22: note: '__hip_internal::uint32_t' declared here
1427 | typedef unsigned int uint32_t;
| ^
In file included from /tmp/comgr-345b3a/input/tmp/tmpnz1gjn5f/c8573e542c59846fdedfebf89b822ea2cba126fc.hsaco.cu:1:
In file included from /pycodegen-dev/pystencils/src/pystencils/include/pystencils_runtime/generic_gpu.hpp:4:
/pycodegen-dev/pystencils/src/pystencils/include/pystencils_runtime/./random.hpp:44:14: error: unknown type name 'uint32_t'; did you mean '__hip_internal::uint32_t'?
44 | (uint32_t)ctr0, (uint32_t)ctr1, (uint32_t)ctr2, (uint32_t)ctr3, //
| ^~~~~~~~
| __hip_internal::uint32_t
/tmp/comgr-345b3a/include/hiprtc_runtime.h:1427:22: note: '__hip_internal::uint32_t' declared here
1427 | typedef unsigned int uint32_t;
| ^
In file included from /tmp/comgr-345b3a/input/tmp/tmpnz1gjn5f/c8573e542c59846fdedfebf89b822ea2cba126fc.hsaco.cu:1:
In file included from /pycodegen-dev/pystencils/src/pystencils/include/pystencils_runtime/generic_gpu.hpp:4:
/pycodegen-dev/pystencils/src/pystencils/include/pystencils_runtime/./random.hpp:44:30: error: unknown type name 'uint32_t'; did you mean '__hip_internal::uint32_t'?
44 | (uint32_t)ctr0, (uint32_t)ctr1, (uint32_t)ctr2, (uint32_t)ctr3, //
| ^~~~~~~~
| __hip_internal::uint32_t
/tmp/comgr-345b3a/include/hiprtc_runtime.h:1427:22: note: '__hip_internal::uint32_t' declared here
1427 | typedef unsigned int uint32_t;
| ^
In file included from /tmp/comgr-345b3a/input/tmp/tmpnz1gjn5f/c8573e542c59846fdedfebf89b822ea2cba126fc.hsaco.cu:1:
In file included from /pycodegen-dev/pystencils/src/pystencils/include/pystencils_runtime/generic_gpu.hpp:4:
/pycodegen-dev/pystencils/src/pystencils/include/pystencils_runtime/./random.hpp:44:46: error: unknown type name 'uint32_t'; did you mean '__hip_internal::uint32_t'?
44 | (uint32_t)ctr0, (uint32_t)ctr1, (uint32_t)ctr2, (uint32_t)ctr3, //
| ^~~~~~~~
| __hip_internal::uint32_t
/tmp/comgr-345b3a/include/hiprtc_runtime.h:1427:22: note: '__hip_internal::uint32_t' declared here
1427 | typedef unsigned int uint32_t;
| ^
In file included from /tmp/comgr-345b3a/input/tmp/tmpnz1gjn5f/c8573e542c59846fdedfebf89b822ea2cba126fc.hsaco.cu:1:
In file included from /pycodegen-dev/pystencils/src/pystencils/include/pystencils_runtime/generic_gpu.hpp:4:
/pycodegen-dev/pystencils/src/pystencils/include/pystencils_runtime/./random.hpp:44:62: error: unknown type name 'uint32_t'; did you mean '__hip_internal::uint32_t'?
44 | (uint32_t)ctr0, (uint32_t)ctr1, (uint32_t)ctr2, (uint32_t)ctr3, //
| ^~~~~~~~
| __hip_internal::uint32_t
/tmp/comgr-345b3a/include/hiprtc_runtime.h:1427:22: note: '__hip_internal::uint32_t' declared here
1427 | typedef unsigned int uint32_t;
| ^
In file included from /tmp/comgr-345b3a/input/tmp/tmpnz1gjn5f/c8573e542c59846fdedfebf89b822ea2cba126fc.hsaco.cu:1:
In file included from /pycodegen-dev/pystencils/src/pystencils/include/pystencils_runtime/generic_gpu.hpp:4:
/pycodegen-dev/pystencils/src/pystencils/include/pystencils_runtime/./random.hpp:45:14: error: unknown type name 'uint32_t'; did you mean '__hip_internal::uint32_t'?
45 | (uint32_t)key0, (uint32_t)key1, //
| ^~~~~~~~
| __hip_internal::uint32_t
/tmp/comgr-345b3a/include/hiprtc_runtime.h:1427:22: note: '__hip_internal::uint32_t' declared here
1427 | typedef unsigned int uint32_t;
| ^
In file included from /tmp/comgr-345b3a/input/tmp/tmpnz1gjn5f/c8573e542c59846fdedfebf89b822ea2cba126fc.hsaco.cu:1:
In file included from /pycodegen-dev/pystencils/src/pystencils/include/pystencils_runtime/generic_gpu.hpp:4:
/pycodegen-dev/pystencils/src/pystencils/include/pystencils_runtime/./random.hpp:45:30: error: unknown type name 'uint32_t'; did you mean '__hip_internal::uint32_t'?
45 | (uint32_t)key0, (uint32_t)key1, //
| ^~~~~~~~
| __hip_internal::uint32_t
/tmp/comgr-345b3a/include/hiprtc_runtime.h:1427:22: note: '__hip_internal::uint32_t' declared here
1427 | typedef unsigned int uint32_t;
| ^
/tmp/comgr-345b3a/input/tmp/tmpnz1gjn5f/c8573e542c59846fdedfebf89b822ea2cba126fc.hsaco.cu:8:10: error: unknown type name 'int64_t'; did you mean '__hip_internal::int64_t'?
8 | const int64_t ctr_0 = 1LL + ((int64_t) blockIdx.y * (int64_t) blockDim.y + (int64_t) threadIdx.y);
| ^~~~~~~
| __hip_internal::int64_t
/tmp/comgr-345b3a/include/hiprtc_runtime.h:1432:26: note: '__hip_internal::int64_t' declared here
1432 | typedef signed long long int64_t;
| ^
/tmp/comgr-345b3a/input/tmp/tmpnz1gjn5f/c8573e542c59846fdedfebf89b822ea2cba126fc.hsaco.cu:8:34: error: unknown type name 'int64_t'; did you mean '__hip_internal::int64_t'?
8 | const int64_t ctr_0 = 1LL + ((int64_t) blockIdx.y * (int64_t) blockDim.y + (int64_t) threadIdx.y);
| ^~~~~~~
| __hip_internal::int64_t
/tmp/comgr-345b3a/include/hiprtc_runtime.h:1432:26: note: '__hip_internal::int64_t' declared here
1432 | typedef signed long long int64_t;
| ^
/tmp/comgr-345b3a/input/tmp/tmpnz1gjn5f/c8573e542c59846fdedfebf89b822ea2cba126fc.hsaco.cu:8:57: error: unknown type name 'int64_t'; did you mean '__hip_internal::int64_t'?
8 | const int64_t ctr_0 = 1LL + ((int64_t) blockIdx.y * (int64_t) blockDim.y + (int64_t) threadIdx.y);
| ^~~~~~~
| __hip_internal::int64_t
/tmp/comgr-345b3a/include/hiprtc_runtime.h:1432:26: note: '__hip_internal::int64_t' declared here
1432 | typedef signed long long int64_t;
| ^
/tmp/comgr-345b3a/input/tmp/tmpnz1gjn5f/c8573e542c59846fdedfebf89b822ea2cba126fc.hsaco.cu:8:80: error: unknown type name 'int64_t'; did you mean '__hip_internal::int64_t'?
8 | const int64_t ctr_0 = 1LL + ((int64_t) blockIdx.y * (int64_t) blockDim.y + (int64_t) threadIdx.y);
| ^~~~~~~
| __hip_internal::int64_t
/tmp/comgr-345b3a/include/hiprtc_runtime.h:1432:26: note: '__hip_internal::int64_t' declared here
1432 | typedef signed long long int64_t;
| ^
/tmp/comgr-345b3a/input/tmp/tmpnz1gjn5f/c8573e542c59846fdedfebf89b822ea2cba126fc.hsaco.cu:9:10: error: unknown type name 'int64_t'; did you mean '__hip_internal::int64_t'?
9 | const int64_t ctr_1 = 1LL + ((int64_t) blockIdx.x * (int64_t) blockDim.x + (int64_t) threadIdx.x);
| ^~~~~~~
| __hip_internal::int64_t
/tmp/comgr-345b3a/include/hiprtc_runtime.h:1432:26: note: '__hip_internal::int64_t' declared here
1432 | typedef signed long long int64_t;
| ^
fatal error: too many errors emitted, stopping now [-ferror-limit=]
20 errors generated when compiling for gfx1103.