Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
185 commits
Select commit Hold shift + click to select a range
2f14646
Continued cupy port
max-models Aug 11, 2026
e1304a4
Merge branch 'devel' into cupy-full-model
max-models Aug 11, 2026
aba7b36
_to_numpy_for_kernel(self.params_numpy)
max-models Aug 11, 2026
0bec00a
Set feectools<=0.1.10
max-models Aug 11, 2026
742c0e8
Made sure params_LinearMHDDriftkineticCC.py works with cupy
max-models Aug 11, 2026
7ac8ed6
Update scope-profiler
max-models Aug 11, 2026
2f08950
Update for latest scope-profiler
max-models Aug 11, 2026
87d282a
Added --backend flag
max-models Aug 12, 2026
71569f1
Added outputs to eval_spline_mpi_matrix
max-models Aug 12, 2026
c9a7720
updated feectools
max-models Aug 12, 2026
33012d0
Added params_PressureLessSPH.py
max-models Aug 12, 2026
fa2e8d4
Added profiling regions (mostly for setup)
max-models Aug 12, 2026
7387ccd
Updated to scope-profiler 0.2.7
max-models Aug 12, 2026
f456d6b
Added more profiling regions, in particular to propagators
max-models Aug 12, 2026
2716749
Split summaries in different tables
max-models Aug 12, 2026
7a680d9
print tables with a loop
max-models Aug 12, 2026
1e7dad4
Removed profiling_trace parameter
max-models Aug 12, 2026
8d7eb6a
Removed profiling_trace from the examples
max-models Aug 12, 2026
76944c1
temporary: fix all cupy tests
max-models Aug 12, 2026
bf65fea
Merge branch 'devel' into add-more-profiling-regions
max-models Aug 13, 2026
44687e0
Speed up host to/from device transfers
max-models Aug 14, 2026
4c1cf6d
Merge branch 'devel' into cupy-full-model
max-models Aug 14, 2026
de2fc6a
np -> xp
max-models Aug 14, 2026
e4f82e6
xp->np
max-models Aug 14, 2026
56e6f85
optimization, reduce the number of .get operations in _find_outside_p…
max-models Aug 14, 2026
40a4072
Use self.name as profiling label
max-models Aug 14, 2026
5bb268b
Merge branch 'add-more-profiling-regions' into cupy-full-model
max-models Aug 14, 2026
26f0d03
Formatting
max-models Aug 14, 2026
ae1e61a
Merge remote-tracking branch 'origin/add-more-profiling-regions' into…
max-models Aug 14, 2026
63ec65c
Np=8_000_000
max-models Aug 15, 2026
578c8f8
Added cuda fusedkernels in src/struphy/pic/pushing/pusher_kernels_cud…
max-models Aug 15, 2026
145afc6
Added push_v_with_efield_cuboid_gpu
max-models Aug 15, 2026
ca928af
Added push_v_with_efield_cuboid_gpu
max-models Aug 15, 2026
c7a66f5
Added param save_restart: bool to turn off restart saving
max-models Aug 15, 2026
77da284
Fixed the scalar bug
max-models Aug 15, 2026
bdb022b
generalize push_v_with_efield for mpi like push_eta
max-models Aug 15, 2026
3b64236
Fix MPI with cupy
max-models Aug 15, 2026
41022ec
Added cuda version of eval kernels!
max-models Aug 15, 2026
78a5bf2
Ported more sph kernels
max-models Aug 15, 2026
f76af51
Merge remote-tracking branch 'origin/devel' into cupy-full-model
max-models Aug 15, 2026
41a236a
Merge remote-tracking branch 'origin/devel' into cupy-full-model
max-models Aug 15, 2026
19a3a9b
Added src/struphy/pic/sorting_kernels_cuda.py
max-models Aug 15, 2026
b3c3441
Fix params, remove profiling_trace
max-models Aug 15, 2026
a322102
Extend benchmarks
max-models Aug 15, 2026
c19ceb2
Fix numpy test error
max-models Aug 15, 2026
a7d1234
etas = tuple(_dev(e) for e in etas): numpy fix
max-models Aug 16, 2026
5b84e40
Added more pusher kernels
max-models Aug 16, 2026
b047493
Added accum kernels with cuda
max-models Aug 16, 2026
f510403
Formatting
max-models Aug 16, 2026
3e76fee
Added bench_gpu/out_* to gitignore
max-models Aug 16, 2026
ac986a1
Added encoding utf8 in format.py
max-models Aug 16, 2026
14296bf
Accum kernels
max-models Aug 16, 2026
2648673
Port linear mhd kernels
max-models Aug 16, 2026
7194b99
Cuda versions of gc kernels
max-models Aug 16, 2026
9baeb3b
added bench_gpu to gitignore
max-models Aug 16, 2026
057cb53
Added profiling/examples/LinearMHDDriftkineticCC/params_LinearMHDDrif…
max-models Aug 16, 2026
c968555
Added profiling/submit_linearmhd_numpy_vs_cupy.py
max-models Aug 16, 2026
0ac2eeb
Keep markers on GPU (removed cp.asarray(markers))
max-models Aug 16, 2026
96b87fe
Cupy fixes, removed _to_numpy_for_kernel in a bunch of places
max-models Aug 16, 2026
04014c7
Added pusher_utilities_kernels_cuda.py
max-models Aug 16, 2026
861e88e
Setup example sim
max-models Aug 16, 2026
2595b5f
Fix params
max-models Aug 16, 2026
d97a489
Updated kernels
max-models Aug 16, 2026
e18bd7e
Port particles to grid kernels
max-models Aug 16, 2026
1f497fc
Added src/struphy/pic/pushing/pusher_kernels_sph_cuda.py
max-models Aug 16, 2026
8270538
More accum kernels
max-models Aug 16, 2026
b2a15ee
Added profiling/submit_guidingcenter_numpy_vs_cupy.py
max-models Aug 16, 2026
835d21b
Added eval_kernels_gc_cuda.py
max-models Aug 16, 2026
349dd08
accum kernels
max-models Aug 16, 2026
0017fd8
pusher gc kernels
max-models Aug 16, 2026
1b19226
Updated name
max-models Aug 16, 2026
257e848
Formatting
max-models Aug 16, 2026
3080791
Updated sph kernels, added eval_kernels_sph_cuda.py
max-models Aug 17, 2026
c3d132f
Update kernels
max-models Aug 17, 2026
880a806
Fix hdf5 error: HDF5_USE_FILE_LOCKING = False by default
max-models Aug 17, 2026
f9326a8
Add remaining kernels
max-models Aug 17, 2026
c28c20d
Added setup: total
max-models Aug 17, 2026
fd243b0
Added submit_guidingcenter_cupy_scaling.py
max-models Aug 17, 2026
b564327
Improve mpi sort using cuda
max-models Aug 17, 2026
dfa554d
Updated params for bigger testcase
max-models Aug 17, 2026
dde7145
Back to 10M markers
max-models Aug 17, 2026
30b5a70
improvements of mpi sorting
max-models Aug 17, 2026
d755dda
Fixed the xp.all(axis=1) bottleneck
max-models Aug 17, 2026
395b69f
formatting
max-models Aug 17, 2026
d30b8ac
Added self._eta_bc_buf for apply_kinetic_bc
max-models Aug 17, 2026
ab71c77
Merge remote-tracking branch 'origin/improvements-of-mpi-sort' into c…
max-models Aug 17, 2026
af71f18
Added self._eta_bc_buf for apply_kinetic_bc
max-models Aug 17, 2026
eba75a1
formatting
max-models Aug 17, 2026
92ebc68
Merge remote-tracking branch 'origin/add-totals-to-profiling' into cu…
max-models Aug 17, 2026
62af538
Run the profiling on a bigger case and up to 2 nodes.
max-models Aug 17, 2026
01f7075
Merge branch 'improvements-of-mpi-sort' into cupy-full-model
max-models Aug 17, 2026
697a550
Merge branch 'column-major-fix-of-apply-markers-bc' into cupy-full-model
max-models Aug 17, 2026
8f55004
Added _compute_neighbor_ranks()
max-models Aug 17, 2026
dfee0de
Merge remote-tracking branch 'origin/devel' into fix-_sendrecv_get_de…
max-models Aug 17, 2026
980575e
Merge branch 'fix-_sendrecv_get_destinations-loop-only-over-neighbour…
max-models Aug 17, 2026
a167433
Run with save_restart=False
max-models Aug 17, 2026
4184298
Removed barriers
max-models Aug 17, 2026
1fb9ef0
Elementwise AND for matched_local
max-models Aug 17, 2026
a472135
temporary: Set OMPI_MCA_coll_hcoll_enable=0 and MPI4PY_RC_THREAD_LEVE…
max-models Aug 17, 2026
7d9094d
Extend cluster presets a bit
max-models Aug 17, 2026
fc65f56
Added VlasovAmpereOneSpecies/params_VlasovAmpere_scaling.py
max-models Aug 17, 2026
c988f71
Added sph scaling example
max-models Aug 17, 2026
d4f1647
Skip over communication if running on 1 rank
max-models Aug 17, 2026
bf225c5
Some micro optimization
max-models Aug 17, 2026
4f503dc
Added src/struphy/pic/tests/bench_mpi_sort_markers.py
max-models Aug 17, 2026
ab4309e
Optimization, special case for alpha=1
max-models Aug 17, 2026
74808a1
Reusing of not_hole_or_ghost
max-models Aug 17, 2026
d210b9a
Update scope-profiler
max-models Aug 17, 2026
879e695
Updated update_holes(self, update_valid_mks: bool = True)
max-models Aug 17, 2026
bd0de01
Use xp.to_numpy instead of .get() manually
max-models Aug 17, 2026
1c68bd9
Merge branch 'devel' into fix-_sendrecv_get_destinations-loop-only-ov…
max-models Aug 18, 2026
e48fab9
Merge branch 'devel' into cupy-full-model
max-models Aug 18, 2026
5d0c6c7
Merge remote-tracking branch 'origin/fix-_sendrecv_get_destinations-l…
max-models Aug 18, 2026
0b6eb6f
Use numpy for small arrays
max-models Aug 18, 2026
1628739
Added DirectSolver(InverseLinearOperator)
max-models Aug 18, 2026
90e4ea7
Added more profiling examples
max-models Aug 18, 2026
145da0a
Increase problem size
max-models Aug 18, 2026
895b303
Added pproc
max-models Aug 18, 2026
bf92585
Added node vs node and cupy scaling
max-models Aug 18, 2026
db1ae24
cpus_per_task=1 for gpu
max-models Aug 18, 2026
71e771f
Updated params
max-models Aug 18, 2026
06da69a
Added tests
max-models Aug 18, 2026
d708b8b
formatting
max-models Aug 18, 2026
42136e0
Updated profiling params
max-models Aug 19, 2026
3db41d8
Use xp.union1d since it works with both numpy and cupy
max-models Aug 19, 2026
53f2133
updated params_LinearMHDDriftkineticCC.py
max-models Aug 19, 2026
375c739
Run driftkinetic_cyclone_cupy_scaling with 1,2,4,8 GPUs
max-models Aug 19, 2026
365d19b
Updated feectools
max-models Aug 19, 2026
bd33ae1
update feectools
max-models Aug 19, 2026
aea1ab1
update feectools
max-models Aug 19, 2026
045ef7c
Cleanup
max-models Aug 19, 2026
eedf7e6
Hardcoded the params
max-models Aug 19, 2026
09bece4
Cleanup
max-models Aug 19, 2026
1288a65
Added topology-test-multi-rank
max-models Aug 19, 2026
2a40612
cleanup
max-models Aug 19, 2026
bc8febf
feectools: merged direct solver branch into the cupy branch
max-models Aug 19, 2026
6559d2d
Revert previous commit and run all mpi tests with 1,2,4 ranks
max-models Aug 19, 2026
9d6472e
Merge branch 'devel' into cupy-full-model
max-models Aug 20, 2026
fcdc1bf
formatting
max-models Aug 20, 2026
3bc6e34
bugfix in feectools
max-models Aug 20, 2026
34916a4
Updated feectools
max-models Aug 20, 2026
3240d4c
Merge branch 'devel' into fix-_sendrecv_get_destinations-loop-only-ov…
max-models Aug 20, 2026
98ed255
Removed the DirectSolver
max-models Aug 20, 2026
80d9490
Added mpi_pic mark
max-models Aug 20, 2026
2cf3412
mpi tests with 2, mpi pic tests with 1,2,3,4
max-models Aug 20, 2026
b9ed0dd
Merge branch 'fix-_sendrecv_get_destinations-loop-only-over-neighbour…
max-models Aug 20, 2026
e956331
Added shell bash to mpi jobs
max-models Aug 21, 2026
f932da9
Skip MPI pic tests on 3 procs
max-models Aug 21, 2026
5313057
Updated feectools
max-models Aug 21, 2026
7e5b15e
Merge remote-tracking branch 'origin/fix-_sendrecv_get_destinations-l…
max-models Aug 21, 2026
88734aa
Increase number of boxes to 6,6,6
max-models Aug 21, 2026
4fbbd77
Merge remote-tracking branch 'origin/fix-_sendrecv_get_destinations-l…
max-models Aug 21, 2026
0d99630
Increase number of elementes in test_sorting.py
max-models Aug 21, 2026
cebcba7
Merge remote-tracking branch 'origin/fix-_sendrecv_get_destinations-l…
max-models Aug 21, 2026
d43599c
Fix test_estimate_mem.py
max-models Aug 21, 2026
74814e9
Merge branch 'devel' into cupy-full-model
max-models Aug 22, 2026
1903fff
Merge branch 'devel' into cupy-full-model
max-models Aug 22, 2026
1417fee
fix
max-models Aug 26, 2026
246747f
CUDA kernels for FEEC assembly and projection
max-models Aug 31, 2026
2cd5592
run gitlab ci on gpu runners
max-models Aug 31, 2026
d5eeb3b
small optimization
max-models Sep 1, 2026
c859853
formatting
max-models Sep 1, 2026
3be59ba
Update feectools
max-models Sep 1, 2026
6dd7d3f
gpu bug fixes
max-models Sep 1, 2026
f300a21
Merge branch 'devel' into cupy-full-model
max-models Sep 1, 2026
82d92b0
formatting
max-models Sep 2, 2026
f1786c0
Formatting
max-models Sep 2, 2026
4445fe8
remove noqa
max-models Sep 2, 2026
03d431c
Moved the cuda code to .cu files
max-models Sep 2, 2026
f051292
Add profilingoptions with labels
max-models Sep 2, 2026
9944f47
Enable setting log file with environment variable
max-models Sep 3, 2026
f2a10a9
Merge branch 'enable-setting-log-file-with-environment-variable' into…
max-models Sep 3, 2026
ece72aa
Update params
max-models Sep 3, 2026
907f917
Consilidate the cuda kernels, add CudaKernel
max-models Sep 14, 2026
5c4c801
Cleanup for a MWE
max-models Sep 14, 2026
98ce1f4
Update feectools
max-models Sep 15, 2026
38f49da
Fix tests
max-models Sep 15, 2026
c59ed96
Fixes
max-models Sep 16, 2026
b120a6b
Removed CudaKernelSet
max-models Sep 16, 2026
d89ac91
Fix the tests
max-models Sep 16, 2026
b1f226c
Merge branch 'devel' into cuda-push_v_in_efield_only
max-models Sep 22, 2026
51e9d5f
Merge branch 'devel' into cuda-push_v_in_efield_only
max-models Sep 23, 2026
05ec3ef
formatting
max-models Sep 23, 2026
660e632
Merge branch 'devel' into cuda-push_v_in_efield_only
max-models Sep 27, 2026
468afc2
temporary: notes
max-models Sep 28, 2026
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 2 additions & 0 deletions .gitignore
Original file line number Diff line number Diff line change
Expand Up @@ -101,6 +101,8 @@ src/struphy/io/out/
src/struphy/state.yml
src/struphy/io/inp/params_*
*.bin
bench_gpu/out_*
bench_gpu/

# models list
bin/
Expand Down
5 changes: 5 additions & 0 deletions pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -63,6 +63,9 @@ phys = [
"gvec>=1.1.0, <=1.5.0",
"desc-opt<=0.17.1",
]
cuda = [
"cupy-cuda12x==13.6.0",
]
dev = [
"struphy[mpi]",
"notebook",
Expand Down Expand Up @@ -137,6 +140,7 @@ struphy = "struphy.console.main:struphy"
]
struphy = [
"compile_struphy.mk",
"**/*.cu",
]

[tool.autopep8]
Expand Down Expand Up @@ -182,4 +186,5 @@ markers = [
"hybrid",
"single",
"mpi_pic",
"needs_host_kernels",
]
10 changes: 10 additions & 0 deletions setup/modules.pitagora.sh
Original file line number Diff line number Diff line change
Expand Up @@ -5,12 +5,22 @@ python/3.11.7"

# openmpi/4.1.6--gcc--12.3.0
MODULES_GCC="gcc/12.3.0 \
openmpi/4.1.6--gcc--12.3.0 \
python/3.11.7 \
hdf5/1.14.3--gcc--12.3.0 \
cmake/3.27.9 \
netcdf-fortran/4.6.1--gcc--12.3.0 \
netlib-scalapack/2.2.0--openmpi--4.1.6--gcc--12.3.0-ucx1.20"

# On the Booster (GPU) partition, ARRAY_BACKEND=cupy runs need libnvrtc.so.12 for
# cupy's RawKernel/JIT compilation -- otherwise every cupy import fails as soon as
# it touches the GPU (e.g. `xp.tri()` at struphy import time). SLURM_JOB_PARTITION
# is only set inside a submitted job, so this is a no-op on the DCGP (CPU) partition
# or outside SLURM.
if [[ "${SLURM_JOB_PARTITION:-}" == *boost* ]]; then
MODULES_INTEL="$MODULES_INTEL cuda/12.6"
MODULES_GCC="$MODULES_GCC cuda/12.6"
fi

# For GVEC
# Should be fixed so it works with both gcc and intel
Expand Down
27 changes: 27 additions & 0 deletions src/struphy/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -4,6 +4,33 @@
import logging.config
import os

# HDF5's file locking relies on flock(), which is unreliable/unsupported on
# parallel filesystems such as Lustre or GPFS (common on HPC clusters) and
# causes spurious `BlockingIOError: Unable to synchronously open file` errors.
# Disable it unless the user has explicitly configured it.
os.environ.setdefault("HDF5_USE_FILE_LOCKING", "FALSE")

# mpi4py defaults to requesting MPI_THREAD_MULTIPLE (thread level 3) from
# MPI_Init_thread. On at least one cluster this repo runs on (Pitagora's Booster
# partition, OpenMPI 4.1.6 + UCX 1.20), the UCX worker does not support that level,
# which OpenMPI reports at every multi-rank run ("UCP worker does not support
# MPI_THREAD_MULTIPLE" / "failed to init ucx" / hcoll init failure) and works
# around by making hcoll (its GPU-aware collective component) fail to initialize,
# silently falling back to a different, working collective implementation. Struphy
# only ever calls MPI from the main Python thread (CuPy's internal CUDA driver
# threads don't touch MPI), so requesting the weaker MPI_THREAD_FUNNELED guarantee
# instead is sufficient and avoids the warnings -- but it ALSO lets hcoll
# successfully initialize where it previously failed to, and hcoll's own
# Alltoallv implementation on this cluster then segfaults
# (hmca_bcol_ucx_p2p_alltoallv_pairwise_chunk_progress) the first time it's
# actually used, something the failed init was silently protecting us from.
# hcoll must therefore stay disabled explicitly alongside the thread-level
# change, not just left to fail its own init. Both must be set before mpi4py.MPI
# is imported anywhere (thread level can't change after MPI_Init), and only if
# the user hasn't already configured them themselves.
os.environ.setdefault("MPI4PY_RC_THREAD_LEVEL", "funneled")
os.environ.setdefault("OMPI_MCA_coll_hcoll_enable", "0")

from feectools.ddm.mpi import mpi as MPI

from struphy.utils.mpi_launch import launched_under_mpi
Expand Down
17 changes: 17 additions & 0 deletions src/struphy/conftest.py
Original file line number Diff line number Diff line change
@@ -1,9 +1,17 @@
import logging
import os

import pytest

from struphy import set_logging_level

# Tests marked "needs_host_kernels" call a Pyccel kernel that is not yet
# ported for the CuPy backend (e.g. FEEC mass-matrix/basis-projection
# assembly, particle-to-grid accumulation). Under ARRAY_BACKEND=cupy they are
# skipped rather than run to failure, so a GPU CI run reports the state of
# the actually-ported code paths instead of drowning in known gaps.
_NEEDS_HOST_KERNELS_SKIP_REASON = "needs_host_kernels: not yet ported to the CuPy backend (ARRAY_BACKEND=cupy)"


def set_logging_level_pytest(config):
level_name = str(config.getoption("--logging-level")).upper()
Expand All @@ -25,6 +33,15 @@ def pytest_configure(config):
set_logging_level_pytest(config)


def pytest_collection_modifyitems(config, items):
if os.environ.get("ARRAY_BACKEND") != "cupy":
return
skip_host_only = pytest.mark.skip(reason=_NEEDS_HOST_KERNELS_SKIP_REASON)
for item in items:
if "needs_host_kernels" in item.keywords:
item.add_marker(skip_host_only)


def pytest_addoption(parser):
parser.addoption("--with-desc", action="store_true")
parser.addoption("--vrbose", action="store_true")
Expand Down
10 changes: 9 additions & 1 deletion src/struphy/console/format.py
Original file line number Diff line number Diff line change
Expand Up @@ -1211,7 +1211,15 @@ def confirm_formatting(python_files, linters, yes):
)
print("\n")
if not yes:
ans = input("Format files (Y/n)?\n")
try:
ans = input("Format files (Y/n)?\n")
except EOFError:
# stdin isn't an interactive terminal (piped, scripted, non-tty
# shell, ...) -- input() can't prompt, so there's no way to get
# a real answer. Fail safe (don't format) instead of crashing
# with a raw traceback, and point at the flag that avoids this.
print("\nNo interactive terminal to confirm on. Exiting... (use --yes/-y to skip this prompt)")
sys.exit(1)
if ans.lower() not in ("y", "yes", ""):
print("Exiting...")
sys.exit(1)
Expand Down
63 changes: 63 additions & 0 deletions src/struphy/cuda.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,63 @@
"""Helpers for loading CUDA C sources used by CuPy RawKernel wrappers."""

from functools import lru_cache
from pathlib import Path
from typing import Sequence


@lru_cache(maxsize=None)
def load_cuda_source(module_file: str, source_name: str) -> str:
"""Load a CUDA C source fragment stored alongside its Python wrapper."""
path = Path(module_file).with_name("cuda") / source_name
return path.read_text(encoding="utf-8")


class CudaKernel:
"""A lazily-compiled, cached ``cupy.RawKernel``.

Every hand-written CUDA replacement in ``struphy.pic.*_cuda`` /
``struphy.feec.*_cuda`` used to repeat the same boilerplate at each call
site: a module-level ``_foo_kernel = None`` sentinel, a
``_get_foo_kernel()`` function that imports ``cupy`` and compiles the
``RawKernel`` the first time it's needed (so importing these modules
under ``ARRAY_BACKEND=numpy`` never touches CuPy), and caches it back into
the global. This class replaces that boilerplate with one declaration:

_foo_kernel = CudaKernel(_FOO_SRC, "foo_cuda")

made once at module level, right next to the ``*_gpu`` function it backs.
Compilation is still deferred to first call (``import cupy`` only happens
inside :meth:`__call__`), and the compiled kernel is cached on the
instance -- identical behavior to the old pattern, but now every real GPU
kernel a module launches shows up as a ``CudaKernel(...)`` at module
scope, so `grep -n "CudaKernel("` (or just reading the top of the file)
tells you exactly which functions do device work and which don't.
"""

__slots__ = ("_source", "_name", "_kernel")

def __init__(self, source: str, name: str) -> None:
self._source = source
self._name = name
self._kernel = None

def __call__(self, grid, block, args) -> None:
self._compiled()(grid, block, args)

def _compiled(self):
if self._kernel is None:
import cupy as cp

kernel = cp.RawKernel(self._source, self._name)
kernel.compile()
self._kernel = kernel
return self._kernel


def launch_1d(kernel: CudaKernel, n: int, args: Sequence, threads: int = 256) -> None:
"""Launch ``kernel`` over a 1-D grid with one thread per element of a
length-``n`` array (markers, indices, quadrature points, ...) -- the
launch geometry shared by every kernel in ``struphy.pic.*_cuda`` /
``struphy.feec.*_cuda``."""
blocks = (n + threads - 1) // threads
kernel((blocks,), (threads,), tuple(args))
21 changes: 13 additions & 8 deletions src/struphy/feec/preconditioner.py
Original file line number Diff line number Diff line change
@@ -1,6 +1,7 @@
import logging

import cunumpy as xp
import numpy as np
from feectools.api.essential_bc import apply_essential_bc_stencil
from feectools.ddm.cart import CartDecomposition, DomainDecomposition
from feectools.ddm.mpi import MockComm
Expand Down Expand Up @@ -260,7 +261,7 @@ def fun(e):

M_local = StencilMatrix(V_local, V_local)

row_indices, col_indices = xp.nonzero(M_arr)
row_indices, col_indices = np.nonzero(M_arr)

for row_i, col_i in zip(row_indices, col_indices):
# only consider row indices on process
Expand All @@ -273,7 +274,7 @@ def fun(e):
] = M_arr[row_i, col_i]

# check if stencil matrix was built correctly
assert xp.allclose(M_local.toarray()[s : e + 1], M_arr[s : e + 1])
assert np.allclose(M_local.toarray()[s : e + 1], M_arr[s : e + 1])

matrixcells += [M_local.copy()]
# =======================================================================================================
Expand Down Expand Up @@ -625,7 +626,7 @@ def __init__(self, mass_operator, apply_bc=True):

M_local = StencilMatrix(V_local, V_local)

row_indices, col_indices = xp.nonzero(M_arr)
row_indices, col_indices = np.nonzero(M_arr)

for row_i, col_i in zip(row_indices, col_indices):
# only consider row indices on process
Expand All @@ -638,7 +639,7 @@ def __init__(self, mass_operator, apply_bc=True):
] = M_arr[row_i, col_i]

# check if stencil matrix was built correctly
assert xp.allclose(M_local.toarray()[s : e + 1], M_arr[s : e + 1])
assert np.allclose(M_local.toarray()[s : e + 1], M_arr[s : e + 1])

matrixcells += [M_local.copy()]
# =======================================================================================================
Expand Down Expand Up @@ -911,10 +912,12 @@ class FFTSolver(BandedSolver):
"""

def __init__(self, circmat):
assert isinstance(circmat, xp.ndarray)
# circmat comes from StencilMatrix.toarray(), always host numpy; the
# underlying solve() also calls scipy's solve_circulant, host-only.
assert isinstance(circmat, np.ndarray)
assert is_circulant(circmat)

self._space = xp.ndarray
self._space = np.ndarray
self._column = circmat[:, 0]

# --------------------------------------
Expand Down Expand Up @@ -979,13 +982,15 @@ def is_circulant(mat):
Whether the matrix is circulant (=True) or not (=False).
"""

assert isinstance(mat, xp.ndarray)
# mat comes from StencilMatrix.toarray(), which always densifies to a
# host numpy.ndarray regardless of the active backend.
assert isinstance(mat, np.ndarray)
assert len(mat.shape) == 2
assert mat.shape[0] == mat.shape[1]

if mat.shape[0] > 1:
for i in range(mat.shape[0] - 1):
circulant = xp.allclose(mat[i, :], xp.roll(mat[i + 1, :], -1))
circulant = np.allclose(mat[i, :], np.roll(mat[i + 1, :], -1))
if not circulant:
return circulant
else:
Expand Down
7 changes: 4 additions & 3 deletions src/struphy/feec/psydac_derham.py
Original file line number Diff line number Diff line change
Expand Up @@ -1925,11 +1925,12 @@ def _get_domain_array(self):
else:
nproc = 1

# send buffer
dom_arr_loc = xp.zeros(9, dtype=float)
# send buffer -- host-resident: mpi4py's Allgather needs a real
# buffer-protocol array, not a CuPy array, regardless of backend.
dom_arr_loc = np.zeros(9, dtype=float)

# main array (receive buffers)
dom_arr = xp.zeros(nproc * 9, dtype=float)
dom_arr = np.zeros(nproc * 9, dtype=float)

# Get global starts and ends of domain decomposition
gl_s = self.domain_decomposition.starts
Expand Down
1 change: 0 additions & 1 deletion src/struphy/kernel_arguments/pusher_args_kernels.py
Original file line number Diff line number Diff line change
Expand Up @@ -108,7 +108,6 @@ def __init__(
self.bd2 = np.empty(int(pn[1]), dtype=float)
self.bd3 = np.empty(int(pn[2]), dtype=float)


class DomainArguments:
"""Holds the mandatory arguments pertaining to :class:`~struphy.geometry.base.Domain` passed to particle pusher kernels.

Expand Down
Loading