From a016b6ba50bd44eff82fa8c34acef85b0999805a Mon Sep 17 00:00:00 2001 From: haykh Date: Mon, 18 May 2026 08:51:01 -0400 Subject: [PATCH 001/125] version bump --- CMakeLists.txt | 75 +++++++++++++++++++++++++------------------------- 1 file changed, 38 insertions(+), 37 deletions(-) diff --git a/CMakeLists.txt b/CMakeLists.txt index cd5f0c258..5482348f5 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -7,7 +7,7 @@ set(PROJECT_NAME entity) project( ${PROJECT_NAME} - VERSION 1.4.0 + VERSION 1.5.0 LANGUAGES CXX C) add_compile_options("-D ENTITY_VERSION=\"${PROJECT_VERSION}\"") set(hash_cmd "git diff --quiet src/ && echo $(git rev-parse HEAD) ") @@ -28,35 +28,35 @@ include(${CMAKE_CURRENT_SOURCE_DIR}/cmake/defaults.cmake) # defaults set(DEBUG - ${default_debug} - CACHE BOOL "Debug mode") + ${default_debug} + CACHE BOOL "Debug mode") set(precision - ${default_precision} - CACHE STRING "Precision") + ${default_precision} + CACHE STRING "Precision") set(deposit - ${default_deposit} - CACHE STRING "Deposit") + ${default_deposit} + CACHE STRING "Deposit") set(shape_order - ${default_shape_order} - CACHE STRING "Shape function") + ${default_shape_order} + CACHE STRING "Shape function") set(pgen - ${default_pgen} - CACHE STRING "Problem generator") + ${default_pgen} + CACHE STRING "Problem generator") set(output - ${default_output} - CACHE BOOL "Enable output") + ${default_output} + CACHE BOOL "Enable output") set(mpi - ${default_mpi} - CACHE BOOL "Use MPI") + ${default_mpi} + CACHE BOOL "Use MPI") set(gpu_aware_mpi - ${default_gpu_aware_mpi} - CACHE BOOL "Enable GPU-aware MPI") + ${default_gpu_aware_mpi} + CACHE BOOL "Enable GPU-aware MPI") # -------------------------- Compilation settings -------------------------- # set(CMAKE_CXX_STANDARD 20) @@ -65,42 +65,42 @@ set(CMAKE_EXPORT_COMPILE_COMMANDS ON) if(${DEBUG} STREQUAL "OFF") set(CMAKE_BUILD_TYPE - Release - CACHE STRING "CMake build type") + Release + CACHE STRING "CMake build type") set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -DNDEBUG -O3") else() set(CMAKE_BUILD_TYPE - Debug - CACHE STRING "CMake build type") + Debug + CACHE STRING "CMake build type") set(CMAKE_CXX_FLAGS - "${CMAKE_CXX_FLAGS} -DDEBUG -Wall -Wextra -Wno-unknown-pragmas") + "${CMAKE_CXX_FLAGS} -DDEBUG -Wall -Wextra -Wno-unknown-pragmas") endif() # options set(precisions - "single" "double" - CACHE STRING "Precisions") + "single" "double" + CACHE STRING "Precisions") set(deposits - "zigzag" "esirkepov" - CACHE STRING "Deposits") + "zigzag" "esirkepov" + CACHE STRING "Deposits") if(${deposit} STREQUAL "zigzag") set(shape_order - ${default_shape_order} - CACHE STRING "Shape functions") + ${default_shape_order} + CACHE STRING "Shape functions") endif() set(shape_orders - "1;2;3;4;5;6;7;8;9;10;11" - CACHE STRING "Shape orders") + "1;2;3;4;5;6;7;8;9;10;11" + CACHE STRING "Shape orders") include(${CMAKE_CURRENT_SOURCE_DIR}/cmake/config.cmake) # ------------------------- Third-Party Tests ------------------------------ # set(BUILD_TESTING - OFF - CACHE BOOL "Build tests") + OFF + CACHE BOOL "Build tests") # ------------------------ Third-party dependencies ------------------------ # include(${CMAKE_CURRENT_SOURCE_DIR}/cmake/dependencies.cmake) @@ -129,8 +129,8 @@ else() endif() if(("${Kokkos_DEVICES}" MATCHES "CUDA") - OR ("${Kokkos_DEVICES}" MATCHES "HIP") - OR ("${Kokkos_DEVICES}" MATCHES "SYCL")) + OR ("${Kokkos_DEVICES}" MATCHES "HIP") + OR ("${Kokkos_DEVICES}" MATCHES "SYCL")) set(DEVICE_ENABLED ON) else() set(DEVICE_ENABLED OFF) @@ -148,8 +148,8 @@ if(${mpi}) endif() else() set(gpu_aware_mpi - OFF - CACHE BOOL "Use explicit copy when using MPI + GPU") + OFF + CACHE BOOL "Use explicit copy when using MPI + GPU") endif() endif() @@ -195,7 +195,8 @@ else() string(REPLACE "/" "_" pg_nodir ${pg_nodir}) set(pgen_suffix "_${pg_nodir}") set_problem_generator(${pg}) - add_subdirectory(${SRC_DIR}/engines ${CMAKE_CURRENT_BINARY_DIR}/${pg_nodir}/engines) + add_subdirectory(${SRC_DIR}/engines + ${CMAKE_CURRENT_BINARY_DIR}/${pg_nodir}/engines) add_subdirectory(${SRC_DIR} ${CMAKE_CURRENT_BINARY_DIR}/${pg_nodir}/src) list(APPEND pgens_short ${PGEN}) endforeach() From 00fa0bf65a35c111c94dc8924ee4e6eda8285629 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ludwig=20B=C3=B6ss?= Date: Tue, 19 May 2026 19:25:22 +0200 Subject: [PATCH 002/125] port: team policy and vendor specific sort --- CMakeLists.txt | 66 + cmake/defaults.cmake | 15 + cmake/report.cmake | 25 + src/engines/srpic/currents.h | 135 +- src/framework/containers/particles.h | 58 + src/framework/containers/particles_sort.cpp | 341 ++++- src/global/arch/kokkos_aliases.h | 27 + src/global/utils/sort_dispatch.h | 171 +++ src/global/utils/sorting.h | 155 ++- src/kernels/currents_deposit.hpp | 1395 ++++++++++++------- tests/framework/CMakeLists.txt | 6 + tests/framework/sort_by_key.cpp | 110 ++ tests/global/tiling.cpp | 29 +- tests/kernels/CMakeLists.txt | 3 + tests/kernels/deposit_tiled.cpp | 262 ++++ 15 files changed, 2286 insertions(+), 512 deletions(-) create mode 100644 src/global/utils/sort_dispatch.h create mode 100644 tests/framework/sort_by_key.cpp create mode 100644 tests/kernels/deposit_tiled.cpp diff --git a/CMakeLists.txt b/CMakeLists.txt index cd5f0c258..acac1d7d5 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -58,6 +58,16 @@ set(gpu_aware_mpi ${default_gpu_aware_mpi} CACHE BOOL "Enable GPU-aware MPI") +set(team_policy + ${default_team_policy} + CACHE BOOL "Enable team_policy tile-blocked deposit/pusher kernels") +set(team_policy_tile_size + ${default_team_policy_tile_size} + CACHE STRING "team_policy tile edge length in cells") +set(team_policy_tile_sizes + "4;6;8;10;12" + CACHE STRING "team_policy tile-size choices") + # -------------------------- Compilation settings -------------------------- # set(CMAKE_CXX_STANDARD 20) set(CMAKE_CXX_STANDARD_REQUIRED ON) @@ -136,6 +146,62 @@ else() set(DEVICE_ENABLED OFF) endif() +# ------------------------------ team_policy wiring ------------------------ # +if(${team_policy}) + list(FIND team_policy_tile_sizes "${team_policy_tile_size}" _tps_idx) + if(_tps_idx EQUAL -1) + message(FATAL_ERROR + "${Red}team_policy_tile_size must be one of ${team_policy_tile_sizes}, " + "got '${team_policy_tile_size}'${ColorReset}") + endif() + add_compile_options("-D TEAM_POLICY") + add_compile_options("-D TEAM_POLICY_TILE_SIZE=${team_policy_tile_size}") + + # Vendor sort: oneDPL on SYCL, Thrust on CUDA. Used automatically + # when found; falls back to Kokkos::BinSort otherwise. + if("${Kokkos_DEVICES}" MATCHES "SYCL") + find_package(oneDPL QUIET) + if(oneDPL_FOUND) + message(STATUS "team_policy: oneDPL found, enabling SYCL sort_by_key") + add_compile_options("-D ONEDPL_ENABLED") + set(DEPENDENCIES ${DEPENDENCIES} oneDPL) + else() + message(STATUS "team_policy: oneDPL not found; using BinSort fallback " + "for SYCL sort_by_key") + endif() + endif() + + if("${Kokkos_DEVICES}" MATCHES "CUDA") + find_package(Thrust QUIET) + if(Thrust_FOUND) + message(STATUS "team_policy: Thrust enabled for CUDA sort_by_key") + add_compile_options("-D THRUST_ENABLED") + else() + message(STATUS "team_policy: Thrust not found; using BinSort fallback " + "for CUDA sort_by_key") + endif() + endif() + + if("${Kokkos_DEVICES}" MATCHES "HIP") + # rocThrust ships with ROCm and exposes the same thrust:: API. Using + # it lets the HIP backend build a single permutation via + # sort_by_key and gather all SoA members through one reused buffer, + # instead of the legacy per-member Kokkos::BinSort path which + # allocates a fresh `sorted_values` buffer for every member every + # step (the dominant source of allocator churn / fragmentation on + # ROCm). + find_package(rocthrust QUIET) + if(rocthrust_FOUND) + message(STATUS "team_policy: rocThrust enabled for HIP sort_by_key") + add_compile_options("-D ROCTHRUST_ENABLED") + set(DEPENDENCIES ${DEPENDENCIES} roc::rocthrust) + else() + message(STATUS "team_policy: rocThrust not found; using BinSort " + "fallback for HIP sort_by_key") + endif() + endif() +endif() + # MPI if(${mpi}) find_or_fetch_dependency(MPI FALSE REQUIRED) diff --git a/cmake/defaults.cmake b/cmake/defaults.cmake index fb8790019..a85accf84 100644 --- a/cmake/defaults.cmake +++ b/cmake/defaults.cmake @@ -92,3 +92,18 @@ else() endif() set_property(CACHE default_gpu_aware_mpi PROPERTY TYPE BOOL) + +if(DEFINED ENV{Entity_ENABLE_TEAM_POLICY}) + set(default_team_policy + $ENV{Entity_ENABLE_TEAM_POLICY} + CACHE INTERNAL "Default flag for team_policy tile-blocked kernels") +else() + set(default_team_policy + OFF + CACHE INTERNAL "Default flag for team_policy tile-blocked kernels") +endif() +set_property(CACHE default_team_policy PROPERTY TYPE BOOL) + +set(default_team_policy_tile_size + 8 + CACHE INTERNAL "Default tile edge length in cells for team_policy") diff --git a/cmake/report.cmake b/cmake/report.cmake index 7b145bfc7..65a22a7a6 100644 --- a/cmake/report.cmake +++ b/cmake/report.cmake @@ -122,6 +122,26 @@ if(${mpi} AND ${DEVICE_ENABLED}) GPU_AWARE_MPI_REPORT 46) endif() +printchoices( + "Team Policy" + "team_policy" + "${ON_OFF_VALUES}" + ${team_policy} + OFF + "${Green}" + TEAM_POLICY_REPORT + 46) +if(${team_policy}) + printchoices( + "Team Tile Size" + "team_policy_tile_size" + "${team_policy_tile_sizes}" + ${team_policy_tile_size} + ${default_team_policy_tile_size} + "${Blue}" + TEAM_POLICY_TILE_SIZE_REPORT + 46) +endif() printchoices( "Debug mode" "DEBUG" @@ -197,6 +217,11 @@ if(${mpi} AND ${DEVICE_ENABLED}) string(APPEND REPORT_TEXT " " ${GPU_AWARE_MPI_REPORT} "\n") endif() +string(APPEND REPORT_TEXT " " ${TEAM_POLICY_REPORT} "\n") +if(${team_policy}) + string(APPEND REPORT_TEXT " " ${TEAM_POLICY_TILE_SIZE_REPORT} "\n") +endif() + string( APPEND REPORT_TEXT diff --git a/src/engines/srpic/currents.h b/src/engines/srpic/currents.h index 3afabea8a..faf0bb3ad 100644 --- a/src/engines/srpic/currents.h +++ b/src/engines/srpic/currents.h @@ -2,7 +2,8 @@ * @file engines/srpic/currents.h * @brief Current deposition and filtering routines for the SRPIC engine * @implements - * - ntt::srpic::CallDepositKernel<> -> void + * - ntt::srpic::CallDepositKernel<> -> void (flat path) + * - ntt::srpic::CallDepositKernelTiled<> -> void (TEAM_POLICY) * - ntt::srpic::CurrentsDeposit<> -> void * - ntt::srpic::CurrentsFilter<> -> void * @namespaces: @@ -61,11 +62,142 @@ namespace ntt { dt)); } +#if defined(TEAM_POLICY) + /** + * @brief Tiled deposit launcher (TeamPolicy + per-team scratch). + * + * Iterates over `tile_layout.ntiles_total` teams; each team accumulates + * its tile's particle contributions in SLM scratch and atomically + * flushes to the global J. Requires the species to have been sorted + * with `team_policy` enabled (`tile_layout` populated by + * `SortSpatially`). + * + * Falls back to the flat kernel if `tile_offsets` is empty — this + * happens on the first step before the first sort, or for very small + * species that exited early in `SortSpatially`. The fallback uses the + * passed-in `scatter_cur` so the caller still composes correctly. + */ + template + void CallDepositKernelTiled( + const Particles& species, + const M& local_metric, + const ndfield_t& cur, + real_t dt) { + static_assert(O <= 11u, "Shape order must be <= 11"); + constexpr unsigned short T = static_cast( + TEAM_POLICY_TILE_SIZE); + const auto& layout = species.tile_layout(); + raise::ErrorIf(layout.ntiles_total == 0u, + "CallDepositKernelTiled: tile_layout has 0 tiles — call " + "SortSpatially before CurrentsDeposit", + HERE); + raise::ErrorIf(layout.tile_offsets.extent(0) != layout.ntiles_total + 1u, + "CallDepositKernelTiled: tile_offsets size inconsistent " + "with ntiles_total", + HERE); + + using kernel_t = kernel::DepositCurrents_kernel_tiled; + kernel_t kern { cur, + species.i1, + species.i2, + species.i3, + species.i1_prev, + species.i2_prev, + species.i3_prev, + species.dx1, + species.dx2, + species.dx3, + species.dx1_prev, + species.dx2_prev, + species.dx3_prev, + species.ux1, + species.ux2, + species.ux3, + species.phi, + species.weight, + species.tag, + local_metric, + (real_t)(species.charge()), + dt, + layout }; + + Kokkos::TeamPolicy<> policy(static_cast(layout.ntiles_total), + Kokkos::AUTO); + policy.set_scratch_size(0, Kokkos::PerTeam(kernel_t::scratch_bytes())); + Kokkos::parallel_for("CurrentsDepositTiled", policy, kern); + } +#endif // TEAM_POLICY + template void CurrentsDeposit(Domain& domain, const prm::Parameters& engine_params) { const auto dt = engine_params.get("dt"); Kokkos::deep_copy(domain.fields.cur, ZERO); + +#if defined(TEAM_POLICY) + + // First-step fallback: if any contributing species has not been + // sorted yet (tile_layout still empty), fall back to the flat + // scatter-view path for that step. Subsequent steps see populated + // layouts and use the tiled kernel. + bool any_unsorted = false; + for (auto& species : domain.species) { + if ((species.pusher() == ParticlePusher::NONE) or + (species.npart() == 0) or cmp::AlmostZero_host(species.charge())) { + continue; + } + if (species.tile_layout().ntiles_total == 0u or + species.tile_layout().tile_offsets.extent(0) == 0u) { + any_unsorted = true; + break; + } + } + if (any_unsorted) { + auto scatter_cur = Kokkos::Experimental::create_scatter_view( + domain.fields.cur); + for (auto& species : domain.species) { + if ((species.pusher() == ParticlePusher::NONE) or + (species.npart() == 0) or cmp::AlmostZero_host(species.charge())) { + continue; + } + logger::Checkpoint( + fmt::format("Launching currents deposit (flat fallback, no sort yet) " + "for %d [%s] : %lu %f", + species.index(), + species.label().c_str(), + species.npart(), + (double)species.charge()), + HERE); + CallDepositKernel(species, + domain.mesh.metric, + scatter_cur, + dt); + } + Kokkos::Experimental::contribute(domain.fields.cur, scatter_cur); + } else { + for (auto& species : domain.species) { + if ((species.pusher() == ParticlePusher::NONE) or + (species.npart() == 0) or cmp::AlmostZero_host(species.charge())) { + continue; + } + logger::Checkpoint( + fmt::format("Launching tiled currents deposit for %d [%s] : %lu %f", + species.index(), + species.label().c_str(), + species.npart(), + (double)species.charge()), + HERE); + + CallDepositKernelTiled(species, + domain.mesh.metric, + domain.fields.cur, + dt); + } + } +#else auto scatter_cur = Kokkos::Experimental::create_scatter_view( domain.fields.cur); for (auto& species : domain.species) { @@ -84,6 +216,7 @@ namespace ntt { CallDepositKernel(species, domain.mesh.metric, scatter_cur, dt); } Kokkos::Experimental::contribute(domain.fields.cur, scatter_cur); +#endif } template diff --git a/src/framework/containers/particles.h b/src/framework/containers/particles.h index 895026552..0a15cb5d9 100644 --- a/src/framework/containers/particles.h +++ b/src/framework/containers/particles.h @@ -25,6 +25,7 @@ #include "traits/metric.h" #include "utils/error.h" #include "utils/formatting.h" +#include "utils/sorting.h" #include "framework/containers/species.h" #include "framework/domain/grid.h" @@ -90,6 +91,30 @@ namespace ntt { const uint8_t m_ntags { (uint8_t)(2 + math::pow(3, (int)D) - 1) }; #endif + // team_policy: tile metadata produced by SortSpatially + // and consumed by the tiled deposit / pusher kernels. Lazily + // allocated on first sort. The sort backend itself (oneDPL on SYCL, + // Thrust on CUDA, std::sort on Host, Kokkos::BinSort otherwise) is + // selected at compile time based on the Kokkos device and the + // vendor libraries detected by CMake. + TileLayout m_tile_layout {}; + +#if defined(TEAM_POLICY) && \ + ((defined(SYCL_ENABLED) && defined(ONEDPL_ENABLED)) || \ + (defined(CUDA_ENABLED) && defined(THRUST_ENABLED)) || \ + (defined(HIP_ENABLED) && defined(ROCTHRUST_ENABLED))) + // Persistent byte scratch reused by every SoA-member gather in + // `apply_permutation_to_soa`, across all members and all timesteps. + // Without this each member would allocate (and free) its own + // transient buffer every sort; recycling one persistent buffer + // removes that allocation churn entirely — the structural fix for + // the ROCm sort slowdown / fragmentation. Grown monotonically to + // the largest required size, never shrunk. Kokkos device + // allocations are over-aligned (>= 8 B), so reinterpreting the + // bytes as any SoA element type (<= 8 B PODs) is well-defined. + array_t m_perm_scratch {}; +#endif + public: // for empty allocation Particles() {} @@ -276,9 +301,42 @@ namespace ntt { /** * @brief Sort particles spatially by their cell indices * @param grid The grid object to get the cell information for sorting + * @note In team_policy mode (compile-time `team_policy=ON`), also + * populates `m_tile_layout` with tile-offset and per-tile + * permutation metadata that the tiled deposit/pusher kernels + * consume. */ void SortSpatially(const Grid&); +#if defined(TEAM_POLICY) && \ + ((defined(SYCL_ENABLED) && defined(ONEDPL_ENABLED)) || \ + (defined(CUDA_ENABLED) && defined(THRUST_ENABLED)) || \ + (defined(HIP_ENABLED) && defined(ROCTHRUST_ENABLED))) + private: + /** + * @brief Apply a particle-index permutation (built by oneDPL/Thrust + * sort_by_key) to every SoA member array. Sequential — one + * transient buffer at a time, fenced before scope exit. + * Only compiled when a vendor sort backend is enabled; the + * BinSort path applies the permutation in place via + * `sorter.sort(view)` instead. + */ + void apply_permutation_to_soa(const prtl_perm_t& perm); + + public: +#endif + + /** + * @brief Read-only access to the tile layout produced by the most + * recent SortSpatially call. Returns a default-constructed + * layout (`ntiles_total == 0`) when the species has not yet + * been sorted. + */ + [[nodiscard]] + auto tile_layout() const -> const TileLayout& { + return m_tile_layout; + } + /** * @brief Copy particle data from device to host. */ diff --git a/src/framework/containers/particles_sort.cpp b/src/framework/containers/particles_sort.cpp index 904fb3fc7..c04813e12 100644 --- a/src/framework/containers/particles_sort.cpp +++ b/src/framework/containers/particles_sort.cpp @@ -8,10 +8,22 @@ #include "framework/containers/particles.h" #include "framework/domain/grid.h" +#if defined(TEAM_POLICY) + #if (defined(SYCL_ENABLED) && defined(ONEDPL_ENABLED)) || \ + (defined(CUDA_ENABLED) && defined(THRUST_ENABLED)) || \ + (defined(HIP_ENABLED) && defined(ROCTHRUST_ENABLED)) + #define TEAM_POLICY_USE_VENDOR_SORT + #include "utils/sort_dispatch.h" + #endif +#endif + #include #include +#include #include +#include +#include #include #include #include @@ -195,6 +207,192 @@ namespace ntt { template void Particles::SortSpatially(const Grid& grid) { +#if defined(TEAM_POLICY) + // ---------------------- team_policy: tile-based sort ------------------ // + const auto npart_local = npart(); + if (npart_local == 0u) { + m_tile_layout = TileLayout {}; + m_is_sorted = true; + return; + } + + constexpr unsigned short T = static_cast( + TEAM_POLICY_TILE_SIZE); + static_assert(T > 0u, "TEAM_POLICY_TILE_SIZE must be > 0"); + + // 1. Compute per-axis tile counts and total_tiles. + const auto ncells_active = grid.n_active(); + ncells_t ntx[3] { 1u, 1u, 1u }; + ncells_t total_tiles { 1u }; + if constexpr ((D == Dim::_1D) or (D == Dim::_2D) or (D == Dim::_3D)) { + ntx[0] = static_cast(math::ceil( + static_cast(ncells_active[0]) / static_cast(T))); + total_tiles *= ntx[0]; + } + if constexpr ((D == Dim::_2D) or (D == Dim::_3D)) { + ntx[1] = static_cast(math::ceil( + static_cast(ncells_active[1]) / static_cast(T))); + total_tiles *= ntx[1]; + } + if constexpr (D == Dim::_3D) { + ntx[2] = static_cast(math::ceil( + static_cast(ncells_active[2]) / static_cast(T))); + total_tiles *= ntx[2]; + } + + // 2. Compute per-particle tile key (with min(i, i_prev)). + array_t tile_indices { "tile_indices", npart_local }; + Kokkos::parallel_for( + "FillTileIndices", + rangeActiveParticles(), + sort::PositionToTileIndex { i1, + i2, + i3, + tag, + tile_indices, + ncells_active, + static_cast(T), + array_t {}, + i1_prev, + i2_prev, + i3_prev }); + + // 3. Sort. Vendor library (oneDPL/Thrust) when compiled in; + // Kokkos::BinSort otherwise. n_bins = total_tiles + 2 covers + // the dead-particle sentinel bin (total_tiles + 1u). + const ncells_t n_bins = total_tiles + 2u; + const auto slice = prtl_slice_t(0, npart_local); + #if defined(TEAM_POLICY_USE_VENDOR_SORT) + // Vendor path: produce an explicit permutation via sort_by_key, + // then apply it to each SoA member with a sequential one-buffer + // gather (peak transient = one `npart × sizeof(member)` buffer. + prtl_perm_t perm { "tile_perm", npart_local }; + #if defined(SYCL_ENABLED) && defined(ONEDPL_ENABLED) + sort_helpers::sort_by_key_dispatch(tile_indices, + perm, + n_bins, + sort::backend::OneDPL {}); + #elif defined(HIP_ENABLED) && defined(ROCTHRUST_ENABLED) + sort_helpers::sort_by_key_dispatch(tile_indices, + perm, + n_bins, + sort::backend::Rocthrust {}); + #else + sort_helpers::sort_by_key_dispatch(tile_indices, + perm, + n_bins, + sort::backend::Thrust {}); + #endif + Kokkos::fence("SortSpatially: pre-gather drain"); + apply_permutation_to_soa(perm); + #else + // BinSort path: same mechanism as legacy SortSpatially (BinSort + // allocates one temp View per `sorter.sort(view)` call and frees + // it before the next), so peak transient memory is bounded. + using sorter_op_t = Kokkos::BinOp1D>; + using sorter_t = Kokkos::BinSort, sorter_op_t>; + auto bin_op = sorter_op_t { static_cast(n_bins), 0u, n_bins }; + auto sorter = sorter_t { tile_indices, bin_op, false }; + sorter.create_permute_vector(); + if constexpr (D == Dim::_1D or D == Dim::_2D or D == Dim::_3D) { + sorter.sort(Kokkos::subview(i1, slice)); + sorter.sort(Kokkos::subview(i1_prev, slice)); + sorter.sort(Kokkos::subview(dx1, slice)); + sorter.sort(Kokkos::subview(dx1_prev, slice)); + } + if constexpr (D == Dim::_2D or D == Dim::_3D) { + sorter.sort(Kokkos::subview(i2, slice)); + sorter.sort(Kokkos::subview(i2_prev, slice)); + sorter.sort(Kokkos::subview(dx2, slice)); + sorter.sort(Kokkos::subview(dx2_prev, slice)); + } + if constexpr (D == Dim::_3D) { + sorter.sort(Kokkos::subview(i3, slice)); + sorter.sort(Kokkos::subview(i3_prev, slice)); + sorter.sort(Kokkos::subview(dx3, slice)); + sorter.sort(Kokkos::subview(dx3_prev, slice)); + } + sorter.sort(Kokkos::subview(ux1, slice)); + sorter.sort(Kokkos::subview(ux2, slice)); + sorter.sort(Kokkos::subview(ux3, slice)); + sorter.sort(Kokkos::subview(weight, slice)); + sorter.sort(Kokkos::subview(tag, slice)); + if constexpr (D == Dim::_2D and C != Coord::Cartesian) { + sorter.sort(Kokkos::subview(phi, slice)); + } + for (auto pldr { 0u }; pldr < npld_r(); ++pldr) { + sorter.sort(Kokkos::subview(pld_r, slice, pldr)); + } + for (auto pldi { 0u }; pldi < npld_i(); ++pldi) { + sorter.sort(Kokkos::subview(pld_i, slice, pldi)); + } + // Apply the same permutation to `tile_indices` itself so it ends + // monotonically non-decreasing for the offsets pass below. + sorter.sort(tile_indices); + #endif // TEAM_POLICY_USE_VENDOR_SORT + + // 5. Compute per-tile prefix-sum `tile_offsets` for the tiled + // pusher. `tile_indices` is now sorted (monotonically + // non-decreasing for alive particles, dead sentinel + // `total_tiles + 1` clustered at the end) — vendor sort_by_key + // sorts keys in place; the BinSort path explicitly applies the + // same permutation to `tile_indices` above. Transition-detect + // directly on it: the start of each non-empty tile is the only + // place a write happens — atomic-free in the dense branch. + // Empty tiles (no particles) are filled by a reverse pass on a + // small host mirror (`total_tiles ≈ 176K` at production scale → + // ~700 KB). + { + array_t tile_offsets { "tile_offsets", total_tiles + 1u }; + Kokkos::deep_copy(tile_offsets, static_cast(npart_local)); + + const auto total_tiles_v = total_tiles; + auto ti_v = tile_indices; + Kokkos::parallel_for( + "DetectTileBoundaries", + rangeActiveParticles(), + Lambda(prtlidx_t p) { + const auto t_curr = ti_v(p); + const bool boundary = (p == 0u) || (ti_v(p - 1u) != t_curr); + if (!boundary) { + return; + } + if (t_curr < total_tiles_v) { + tile_offsets(t_curr) = p; + } else { + // First dead particle — also marks the alive_count boundary + // stored at index total_tiles. + Kokkos::atomic_min(&tile_offsets(total_tiles_v), p); + } + }); + + auto h_offsets = Kokkos::create_mirror_view(tile_offsets); + Kokkos::deep_copy(h_offsets, tile_offsets); + for (auto t = static_cast(total_tiles); t-- > 0u;) { + if (h_offsets(t) > h_offsets(t + 1u)) { + h_offsets(t) = h_offsets(t + 1u); + } + } + Kokkos::deep_copy(tile_offsets, h_offsets); + + m_tile_layout.tile_offsets = tile_offsets; + } + + // 6. Populate `m_tile_layout` size/shape. `tile_perm` is not used + // in the current design — the SoA arrays are physically permuted + // into tile order, so consumers iterate + // `[tile_offsets(t), tile_offsets(t+1))` directly without a + // separate permutation indirection. + m_tile_layout.ntiles_per_axis[0] = ntx[0]; + m_tile_layout.ntiles_per_axis[1] = ntx[1]; + m_tile_layout.ntiles_per_axis[2] = ntx[2]; + m_tile_layout.ntiles_total = total_tiles; + m_tile_layout.tile_size = T; + m_tile_layout.tile_perm = prtl_perm_t {}; + m_is_sorted = true; + + Kokkos::fence("SortSpatially: end of team_policy path"); +#else // !TEAM_POLICY — legacy in-place BinSort by global cell index const auto nx2 = grid.n_active(in::x2); const auto nx3 = grid.n_active(in::x3); const auto total_cells = grid.num_active(); @@ -250,13 +448,153 @@ namespace ntt { for (auto pldi { 0u }; pldi < npld_i(); ++pldi) { sorter.sort(Kokkos::subview(pld_i, slice, pldi)); } +#endif // TEAM_POLICY + } + +#if defined(TEAM_POLICY_USE_VENDOR_SORT) + namespace permute_helpers { + + // Permute a 1D SoA member array `arr` in place by `perm`, gathering + // through `scratch` — a persistent byte buffer reused by every + // member and every timestep (no per-call allocation). An unmanaged + // typed view aliases the scratch bytes; the caller guarantees + // `scratch` is large enough and that Kokkos' device over-alignment + // covers the element type. + template + inline void permute_1d_inplace(V& arr, + const prtl_perm_t& perm, + npart_t n, + const array_t& scratch) { + if (n == 0u) { + return; + } + using value_t = typename V::non_const_value_type; + using buf_t = Kokkos::View>; + buf_t buf(reinterpret_cast(scratch.data()), n); + auto perm_v = perm; + auto arr_v = arr; + Kokkos::parallel_for( + "Permute1D", + n, + KOKKOS_LAMBDA(const npart_t p) { buf(p) = arr_v(perm_v(p)); }); + Kokkos::deep_copy(Kokkos::subview(arr, prtl_slice_t(0u, n)), buf); + Kokkos::fence("permute_1d_inplace: end"); + } + + // 2D analogue for `pld_r` / `pld_i`. + template + inline void permute_2d_inplace(V& arr, + const prtl_perm_t& perm, + npart_t n, + npart_t ncols, + const array_t& scratch) { + if (n == 0u or ncols == 0u) { + return; + } + using value_t = typename V::non_const_value_type; + using buf_t = Kokkos::View>; + buf_t buf(reinterpret_cast(scratch.data()), n, ncols); + auto perm_v = perm; + auto arr_v = arr; + Kokkos::parallel_for( + "Permute2D", + CreateParticleRangePolicy({ 0u, 0u }, { n, ncols }), + KOKKOS_LAMBDA(const npart_t p, const npart_t l) { + buf(p, l) = arr_v(perm_v(p), l); + }); + Kokkos::deep_copy(Kokkos::subview(arr, prtl_slice_t(0u, n), Kokkos::ALL), + buf); + Kokkos::fence("permute_2d_inplace: end"); + } + + } // namespace permute_helpers + + template + void Particles::apply_permutation_to_soa(const prtl_perm_t& perm) { + const auto n = npart(); + if (n == 0u) { + return; + } + + // Size the persistent scratch once to the largest gather any member + // needs this call: 1D members need n * sizeof(real_t) bytes (the + // widest element); the 2D payloads need n * ncols * elem bytes. + // Grown monotonically, never shrunk — so after warmup this incurs + // no allocation at all. + std::size_t need = static_cast(n) * sizeof(real_t); + if (npld_r() > 0) { + need = std::max(need, + static_cast(n) * + static_cast(npld_r()) * sizeof(real_t)); + } + if (npld_i() > 0) { + need = std::max(need, + static_cast(n) * + static_cast(npld_i()) * sizeof(npart_t)); + } + if (m_perm_scratch.extent(0) < need) { + m_perm_scratch = array_t { "perm_scratch", need }; + } + const auto& scratch = m_perm_scratch; + + using permute_helpers::permute_1d_inplace; + using permute_helpers::permute_2d_inplace; + + if constexpr (D == Dim::_1D or D == Dim::_2D or D == Dim::_3D) { + permute_1d_inplace(i1, perm, n, scratch); + permute_1d_inplace(dx1, perm, n, scratch); + permute_1d_inplace(i1_prev, perm, n, scratch); + permute_1d_inplace(dx1_prev, perm, n, scratch); + } + if constexpr (D == Dim::_2D or D == Dim::_3D) { + permute_1d_inplace(i2, perm, n, scratch); + permute_1d_inplace(dx2, perm, n, scratch); + permute_1d_inplace(i2_prev, perm, n, scratch); + permute_1d_inplace(dx2_prev, perm, n, scratch); + } + if constexpr (D == Dim::_3D) { + permute_1d_inplace(i3, perm, n, scratch); + permute_1d_inplace(dx3, perm, n, scratch); + permute_1d_inplace(i3_prev, perm, n, scratch); + permute_1d_inplace(dx3_prev, perm, n, scratch); + } + permute_1d_inplace(ux1, perm, n, scratch); + permute_1d_inplace(ux2, perm, n, scratch); + permute_1d_inplace(ux3, perm, n, scratch); + permute_1d_inplace(weight, perm, n, scratch); + permute_1d_inplace(tag, perm, n, scratch); + if constexpr (D == Dim::_2D and C != Coord::Cartesian) { + permute_1d_inplace(phi, perm, n, scratch); + } + if (npld_r() > 0) { + permute_2d_inplace(pld_r, perm, n, static_cast(npld_r()), scratch); + } + if (npld_i() > 0) { + permute_2d_inplace(pld_i, perm, n, static_cast(npld_i()), scratch); + } } +#endif // TEAM_POLICY_USE_VENDOR_SORT + +#if defined(TEAM_POLICY_USE_VENDOR_SORT) + #define APPLY_PERM_INSTANTIATE(D, C) \ + template void Particles::apply_permutation_to_soa( \ + const prtl_perm_t&); +#else + #define APPLY_PERM_INSTANTIATE(D, C) +#endif #define PARTICLES_SORT(D, C) \ template auto Particles::NpartsPerTagAndOffsets() const \ -> std::pair, array_t>; \ template void Particles::RemoveDead(); \ - template void Particles::SortSpatially(const Grid&); + template void Particles::SortSpatially(const Grid&); \ + APPLY_PERM_INSTANTIATE(D, C) PARTICLES_SORT(Dim::_1D, Coord::Cartesian) PARTICLES_SORT(Dim::_2D, Coord::Cartesian) @@ -266,5 +604,6 @@ namespace ntt { PARTICLES_SORT(Dim::_3D, Coord::Spherical) PARTICLES_SORT(Dim::_3D, Coord::Qspherical) #undef PARTICLES_SORT +#undef APPLY_PERM_INSTANTIATE } // namespace ntt diff --git a/src/global/arch/kokkos_aliases.h b/src/global/arch/kokkos_aliases.h index 314e78c0e..43f21ce47 100644 --- a/src/global/arch/kokkos_aliases.h +++ b/src/global/arch/kokkos_aliases.h @@ -10,6 +10,7 @@ * - CreateRangePolicy, CreateRangePolicyOnHost * - random_number_pool_t, random_generator_t * - Random function + * - prtl_perm_t, TileLayout<> * @cpp: * - arch/kokkos_aliases.cpp * @namespaces: @@ -258,6 +259,32 @@ template auto CreateRangePolicyOnHost(const tuple_t&, const tuple_t&) -> range_h_t; +// --------------------------- team_policy types ---------------------------- // +// Particle permutation index: maps a sorted-position p in [0, npart) to a +// pre-sort particle index. Produced by SortSpatially, consumed by tiled +// pusher and deposit kernels to walk particles tile-by-tile without +// physically re-permuting the SoA arrays in lock step every step. +using prtl_perm_t = array_t; + +// Tile layout metadata: the contract between Stream 1 (sort) and Streams +// 2/3 (tiled deposit / pusher). All members are device-resident. +// ntiles_per_axis : number of tiles along each axis (1 for unused axes). +// ntiles_total : product of ntiles_per_axis = league size for TeamPolicy. +// tile_size : tile edge length in cells (compile-time CMake knob, +// replicated here for runtime checks). +// tile_offsets : prefix-sum of per-tile particle counts; size +// ntiles_total + 1; tile t owns particles +// [tile_offsets(t), tile_offsets(t+1)). +// tile_perm : size npart, particle index sorted by tile. +template +struct TileLayout { + ncells_t ntiles_per_axis[3] { 1u, 1u, 1u }; + ncells_t ntiles_total { 0u }; + unsigned short tile_size { 0u }; + array_t tile_offsets; + prtl_perm_t tile_perm; +}; + // Random number pool/generator type alias // (using math:: instead of Kokkos:: to suppress compiler warning on unused namespace alias) using random_number_pool_t = math::Random_XorShift64_Pool; diff --git a/src/global/utils/sort_dispatch.h b/src/global/utils/sort_dispatch.h new file mode 100644 index 000000000..506a90742 --- /dev/null +++ b/src/global/utils/sort_dispatch.h @@ -0,0 +1,171 @@ +/** + * @file utils/sort_dispatch.h + * @brief Backend-dispatched sort_by_key for team_policy SortSpatially. + * @implements + * - sort_helpers::sort_by_key_dispatch -> void (BinSort, OneDPL, Thrust, StdSort) + * @namespaces: + * - ntt::sort_helpers:: + * @macros: + * - TEAM_POLICY + * - SYCL_ENABLED, ONEDPL_ENABLED (oneDPL overload) + * - CUDA_ENABLED, THRUST_ENABLED (Thrust overload) + * + * @note Each overload produces a permutation `perm` of size N such that + * keys[perm[0]] <= keys[perm[1]] <= ... in stable order. + * Always-available overloads: BinSort (uses Kokkos::BinSort) and + * StdSort (host-side std::stable_sort fallback). The vendor-library + * overloads (OneDPL on SYCL, Thrust on CUDA) are conditional on the + * respective build flags. + */ + +#ifndef GLOBAL_UTILS_SORT_DISPATCH_H +#define GLOBAL_UTILS_SORT_DISPATCH_H + +#if !defined(TEAM_POLICY) + #error "sort_dispatch.h is only meaningful when TEAM_POLICY is defined" +#endif + +#include "global.h" + +#include "arch/kokkos_aliases.h" +#include "utils/sorting.h" + +#include +#include + +#if defined(SYCL_ENABLED) && defined(ONEDPL_ENABLED) + #include + #include +#endif +#if defined(CUDA_ENABLED) && defined(THRUST_ENABLED) + #include + #include + #include +#endif +#if defined(HIP_ENABLED) && defined(ROCTHRUST_ENABLED) + #include + #include + #include + #include +#endif + +#include +#include + +namespace ntt::sort_helpers { + + // Always-available legacy fallback: Kokkos::BinSort. n_bins must be an + // upper bound on distinct key values. + inline void sort_by_key_dispatch(const array_t& keys, + prtl_perm_t& perm, + ncells_t n_bins, + ::sort::backend::BinSort) { + const auto n = static_cast(keys.extent(0)); + if (n == 0u) { + return; + } + using sorter_op_t = Kokkos::BinOp1D>; + using sorter_t = Kokkos::BinSort, sorter_op_t>; + auto bin_op = sorter_op_t { static_cast(n_bins), 0u, n_bins }; + auto sorter = sorter_t { keys, bin_op, false }; + sorter.create_permute_vector(); + auto perm_v = perm; + Kokkos::parallel_for( + "PermInitIota", + n, + KOKKOS_LAMBDA(const npart_t i) { perm_v(i) = i; }); + Kokkos::fence("sort_by_key_dispatch BinSort: pre-sort"); + sorter.sort(perm); + Kokkos::fence("sort_by_key_dispatch BinSort: post-sort"); + } + +#if defined(SYCL_ENABLED) && defined(ONEDPL_ENABLED) + inline void sort_by_key_dispatch(const array_t& keys, + prtl_perm_t& perm, + ncells_t /*n_bins*/, + ::sort::backend::OneDPL) { + const auto n = static_cast(keys.extent(0)); + if (n == 0u) { + return; + } + auto* keys_ptr = keys.data(); + auto* perm_ptr = perm.data(); + auto exec = Kokkos::DefaultExecutionSpace(); + auto perm_v = perm; + Kokkos::parallel_for( + "PermInitIota", + n, + KOKKOS_LAMBDA(const npart_t i) { perm_v(i) = i; }); + // Drain Kokkos's queue so oneDPL's policy sees the iota'd perm even + // if oneDPL submits to a different SYCL queue internally. + exec.fence("sort_by_key_dispatch OneDPL: pre-sort"); + auto queue = exec.sycl_queue(); + auto policy = oneapi::dpl::execution::make_device_policy(queue); + oneapi::dpl::sort_by_key(policy, keys_ptr, keys_ptr + n, perm_ptr); + exec.fence("sort_by_key_dispatch OneDPL: post-sort"); + } +#endif + +#if defined(CUDA_ENABLED) && defined(THRUST_ENABLED) + inline void sort_by_key_dispatch(const array_t& keys, + prtl_perm_t& perm, + ncells_t /*n_bins*/, + ::sort::backend::Thrust) { + const auto n = static_cast(keys.extent(0)); + if (n == 0u) { + return; + } + Kokkos::fence("sort_by_key_dispatch Thrust: pre-sort"); + thrust::device_ptr kp(keys.data()); + thrust::device_ptr pp(perm.data()); + thrust::sequence(pp, pp + n); + thrust::sort_by_key(kp, kp + n, pp); + Kokkos::fence("sort_by_key_dispatch Thrust: post-sort"); + } +#endif + +#if defined(HIP_ENABLED) && defined(ROCTHRUST_ENABLED) + // rocThrust exposes the same thrust:: API as CUDA Thrust; with hipcc + // device_ptr-based algorithms dispatch to the HIP backend. Mirrors + // the CUDA Thrust overload. + inline void sort_by_key_dispatch(const array_t& keys, + prtl_perm_t& perm, + ncells_t /*n_bins*/, + ::sort::backend::Rocthrust) { + const auto n = static_cast(keys.extent(0)); + if (n == 0u) { + return; + } + Kokkos::fence("sort_by_key_dispatch Rocthrust: pre-sort"); + thrust::device_ptr kp(keys.data()); + thrust::device_ptr pp(perm.data()); + thrust::sequence(pp, pp + n); + thrust::sort_by_key(kp, kp + n, pp); + Kokkos::fence("sort_by_key_dispatch Rocthrust: post-sort"); + } +#endif + + // Host fallback: indirect-sort via std::stable_sort. + inline void sort_by_key_dispatch(const array_t& keys, + prtl_perm_t& perm, + ncells_t /*n_bins*/, + ::sort::backend::StdSort) { + const auto n = static_cast(keys.extent(0)); + if (n == 0u) { + return; + } + auto keys_h = Kokkos::create_mirror_view_and_copy(Kokkos::HostSpace(), + keys); + auto perm_h = Kokkos::create_mirror_view(perm); + std::iota(perm_h.data(), perm_h.data() + n, npart_t { 0u }); + std::stable_sort(perm_h.data(), + perm_h.data() + n, + [&](npart_t a, npart_t b) { + return keys_h(a) < keys_h(b); + }); + Kokkos::deep_copy(perm, perm_h); + } + +} // namespace ntt::sort_helpers + +#endif // GLOBAL_UTILS_SORT_DISPATCH_H diff --git a/src/global/utils/sorting.h b/src/global/utils/sorting.h index dbe774402..8442f5ddb 100644 --- a/src/global/utils/sorting.h +++ b/src/global/utils/sorting.h @@ -5,6 +5,7 @@ * - sort::BinBool<> * - sort::BinTag<> * - sort::PositionToTileIndex<> + * - sort::backend tag types (compile-time tag dispatch for sort_by_key) * @namespaces: * - sort:: * @note BinBool sorts by boolean values "true" then "false" @@ -62,9 +63,27 @@ namespace sort { const int m_max_bins; }; - template + /** + * @brief Bin a particle into a tile of edge length `tile_size` cells. + * @tparam D Dimension. + * @tparam Count If true, atomic-increment `num_ppt[tile]` for each live + * particle (used to populate per-tile counts in one pass). + * @tparam UsePrev If true, the bin key uses `min(i_curr, i_prev)` instead + * of `i_curr` alone. This guarantees that a particle whose + * Esirkepov stencil straddles a tile boundary (because it + * crossed the boundary during the pusher) lands in the + * lower-indexed of the two tiles. Combined with a halo of + * `O+1` cells in the deposit's per-tile scratch, this + * keeps every particle's stencil inside its assigned + * tile's interior+halo region. See plan §S2.4. + * + * Dead particles get the sentinel `total_tiles + 1u` so they sort to the + * end (or get skipped, depending on the consumer). + */ + template struct PositionToTileIndex { const array_t i1, i2, i3; + const array_t i1_prev, i2_prev, i3_prev; const array_t tag; array_t tile_indices; ncells_t tile_size; @@ -72,20 +91,37 @@ namespace sort { ncells_t ntx2 { 0u }, ntx3 { 0u }; ncells_t total_tiles { 0u }; - - PositionToTileIndex(const array_t& i1, - const array_t& i2, - const array_t& i3, - const array_t& tag, - array_t& tile_indices, + // Active-cell extents per axis. Used to clamp the bin key when + // UsePrev=true, since `i_prev` can be transiently negative after the + // pusher's periodic wrap (`i_prev -= ni`) or out-of-range after an + // MPI receive that hasn't translated frames. Without clamping, the + // signed-to-unsigned promotion in `int(-1) / uint32_t(T)` produces + // ~1.07e9, the linearised `tile_indices(p)` overflows past `n_bins`, + // and BinSort's internal `atomic_add(&bin_count[wild_idx], 1)` + // faults on an unmapped page. + int ncells1 { 1 }, ncells2 { 1 }, ncells3 { 1 }; + + PositionToTileIndex(const array_t& i1_, + const array_t& i2_, + const array_t& i3_, + const array_t& tag_, + array_t& tile_indices_, const std::vector& ncells, - ncells_t tile_size = 1u) - : i1 { i1 } - , i2 { i2 } - , i3 { i3 } - , tag { tag } - , tile_indices { tile_indices } - , tile_size { tile_size } + ncells_t tile_size_ = 1u, + const array_t& num_ppt_ = { "num_ppt", 0u }, + const array_t& i1_prev_ = {}, + const array_t& i2_prev_ = {}, + const array_t& i3_prev_ = {}) + : i1 { i1_ } + , i2 { i2_ } + , i3 { i3_ } + , i1_prev { i1_prev_ } + , i2_prev { i2_prev_ } + , i3_prev { i3_prev_ } + , tag { tag_ } + , tile_indices { tile_indices_ } + , tile_size { tile_size_ } + , num_ppt { num_ppt_ } , ntx2 { 1u } , ntx3 { 1u } , total_tiles { 1u } { @@ -93,49 +129,122 @@ namespace sort { "ncells size must match D", HERE); if constexpr ((D == Dim::_1D) or (D == Dim::_2D) or (D == Dim::_3D)) { + ncells1 = static_cast(ncells[0]); npart_t ntx1 = static_cast(math::ceil( static_cast(ncells[0]) / static_cast(tile_size))); total_tiles *= ntx1; } if constexpr ((D == Dim::_2D) or (D == Dim::_3D)) { + ncells2 = static_cast(ncells[1]); ntx2 = static_cast(math::ceil( static_cast(ncells[1]) / static_cast(tile_size))); total_tiles *= ntx2; } if constexpr (D == Dim::_3D) { + ncells3 = static_cast(ncells[2]); ntx3 = static_cast(math::ceil( static_cast(ncells[2]) / static_cast(tile_size))); total_tiles *= ntx3; } if constexpr (Count) { - num_ppt = array_t { "num_ppt", total_tiles }; + raise::ErrorIf(num_ppt.extent(0) != total_tiles, + "num_ppt must have extent equal to total tiles", + HERE); + } + if constexpr (UsePrev) { + raise::ErrorIf( + i1_prev.extent(0) == 0u, + "PositionToTileIndex requires i1_prev to be set", + HERE); + if constexpr ((D == Dim::_2D) or (D == Dim::_3D)) { + raise::ErrorIf( + i2_prev.extent(0) == 0u, + "PositionToTileIndex requires i2_prev to be set", + HERE); + } + if constexpr (D == Dim::_3D) { + raise::ErrorIf( + i3_prev.extent(0) == 0u, + "PositionToTileIndex requires i3_prev to be set", + HERE); + } } } - Inline auto operator()(prtldx_t p) const { + Inline auto operator()(prtlidx_t p) const { if (tag(p) != ntt::ParticleTag::alive) { tile_indices(p) = total_tiles + 1u; } else { + // bin key per-axis: use min(i, i_prev) when UsePrev so that a + // particle straddling a boundary lands in the lower tile. + // Then clamp to [0, ncells_axis - 1] — `i_prev` can be negative + // (after the pusher's periodic-wrap path: `i_prev -= ni`) or + // out-of-range (after MPI receive without frame translation). + // Without the clamp, signed-to-unsigned promotion in + // `int(-1) / uint32_t(T)` makes `tile_indices(p)` overflow far + // past `n_bins`, and BinSort's `atomic_add(&bin_count[bin],1)` + // faults on an unmapped page. + const auto clamp_axis = [](int v, int ncells) -> int { + return (v < 0) ? 0 : ((v >= ncells) ? (ncells - 1) : v); + }; + const auto key1 = [&]() -> int { + if constexpr (UsePrev) { + const int raw = (i1(p) < i1_prev(p)) ? i1(p) : i1_prev(p); + return clamp_axis(raw, ncells1); + } else { + return i1(p); + } + }(); + const auto key2 = [&]() -> int { + if constexpr (UsePrev) { + const int raw = (i2(p) < i2_prev(p)) ? i2(p) : i2_prev(p); + return clamp_axis(raw, ncells2); + } else { + return i2(p); + } + }(); + const auto key3 = [&]() -> int { + if constexpr (UsePrev) { + const int raw = (i3(p) < i3_prev(p)) ? i3(p) : i3_prev(p); + return clamp_axis(raw, ncells3); + } else { + return i3(p); + } + }(); if constexpr (D == Dim::_1D) { - tile_indices(p) = static_cast(i1(p) / tile_size); + tile_indices(p) = static_cast(key1 / tile_size); } else if constexpr (D == Dim::_2D) { - tile_indices(p) = static_cast(i1(p) / tile_size) * ntx2 + - static_cast(i2(p) / tile_size); + tile_indices(p) = static_cast(key1 / tile_size) * ntx2 + + static_cast(key2 / tile_size); } else if constexpr (D == Dim::_3D) { - tile_indices(p) = (static_cast(i1(p) / tile_size) * ntx2 + - static_cast(i2(p) / tile_size)) * + tile_indices(p) = (static_cast(key1 / tile_size) * ntx2 + + static_cast(key2 / tile_size)) * ntx3 + - static_cast(i3(p) / tile_size); + static_cast(key3 / tile_size); } else { raise::KernelError(HERE, "Wrong D in SortSpatially"); } if constexpr (Count) { - Kokkos::atomic_add(&num_ppt(tile_indices(p)), 1u); + Kokkos::atomic_add(&num_ppt(tile_indices(p)), 1); } } } }; + // -------------------- Backend dispatch for sort_by_key ------------------- // + // Compile-time tags for tag-dispatch into backend-specific + // sort_by_key implementations. Selection is fully compile-time: the + // backend that resolves depends on the active Kokkos device and the + // availability of the corresponding vendor library. + namespace backend { + struct OneDPL {}; + struct Thrust {}; + struct Rocthrust {}; + struct StdSort {}; + // Always-available legacy fallback using Kokkos::BinSort. + struct BinSort {}; + } // namespace backend + } // namespace sort #endif // GLOBAL_UTILS_SORTING_H diff --git a/src/kernels/currents_deposit.hpp b/src/kernels/currents_deposit.hpp index 268187af2..5eb7ff2b6 100644 --- a/src/kernels/currents_deposit.hpp +++ b/src/kernels/currents_deposit.hpp @@ -1,10 +1,24 @@ /** * @file kernels/currents_deposit.hpp - * @brief Covariant algorithms for the current deposition + * @brief Covariant algorithms for the current deposition. + * + * Two kernels share the same per-particle body + * (`kernel::deposit::deposit_one_particle`): + * - `kernel::DepositCurrents_kernel` flat (RangePolicy over particles, + * writes into a `Kokkos::Experimental::ScatterView`). Always available. + * - `kernel::DepositCurrents_kernel_tiled` team-policy + * (one team per spatial tile, accumulates into team SLM scratch with + * atomic adds, then flushes to global J). Available when `team_policy=ON` + * (`#if defined(TEAM_POLICY)`). Stream 2 of the Pattern A plan. + * * @implements + * - kernel::deposit::PrtlPack<> + * - kernel::deposit::deposit_one_particle<> * - kernel::DepositCurrents_kernel<> + * - kernel::DepositCurrents_kernel_tiled<> (TEAM_POLICY only) * @namespaces: * - kernel:: + * - kernel::deposit:: */ #ifndef KERNELS_CURRENTS_DEPOSIT_HPP @@ -18,100 +32,99 @@ #include "utils/error.h" #include "utils/numeric.h" -#include "particle_shapes.hpp" +#include "kernels/particle_shapes.hpp" #include +#include #define i_di_to_Xi(I, DI) (static_cast((I)) + static_cast((DI))) namespace kernel { using namespace ntt; - /** - * @brief Algorithm for the current deposition - */ - template - class DepositCurrents_kernel { - static_assert(O <= 11u, "Shape function order O must be <= 11"); - static constexpr auto D = M::Dim; - - scatter_ndfield_t J; - const array_t i1, i2, i3; - const array_t i1_prev, i2_prev, i3_prev; - const array_t dx1, dx2, dx3; - const array_t dx1_prev, dx2_prev, dx3_prev; - const array_t ux1, ux2, ux3; - const array_t phi; - const array_t weight; - const array_t tag; - const M metric; - const real_t charge, inv_dt; + namespace deposit { - public: /** - * @brief explicit constructor. + * @brief Per-particle reference pack consumed by both the flat and tiled + * deposit kernels. The same set of SoA references is captured by + * each kernel; bundling them here keeps the helper's argument + * list manageable and ensures every consumer reads the same + * view aliases. */ - DepositCurrents_kernel(const scatter_ndfield_t& scatter_cur, - const array_t& i1, - const array_t& i2, - const array_t& i3, - const array_t& i1_prev, - const array_t& i2_prev, - const array_t& i3_prev, - const array_t& dx1, - const array_t& dx2, - const array_t& dx3, - const array_t& dx1_prev, - const array_t& dx2_prev, - const array_t& dx3_prev, - const array_t& ux1, - const array_t& ux2, - const array_t& ux3, - const array_t& phi, - const array_t& weight, - const array_t& tag, - const M& metric, - real_t charge, - const real_t dt) - : J { scatter_cur } - , i1 { i1 } - , i2 { i2 } - , i3 { i3 } - , i1_prev { i1_prev } - , i2_prev { i2_prev } - , i3_prev { i3_prev } - , dx1 { dx1 } - , dx2 { dx2 } - , dx3 { dx3 } - , dx1_prev { dx1_prev } - , dx2_prev { dx2_prev } - , dx3_prev { dx3_prev } - , ux1 { ux1 } - , ux2 { ux2 } - , ux3 { ux3 } - , phi { phi } - , weight { weight } - , tag { tag } - , metric { metric } - , charge { charge } - , inv_dt { ONE / dt } { - raise::ErrorIf( - (O == 2u and N_GHOSTS < 2), - "Order of interpolation is 2, but number of ghost cells is < 2", - HERE); - } + template + struct PrtlPack { + array_t i1, i2, i3; + array_t i1_prev, i2_prev, i3_prev; + array_t dx1, dx2, dx3; + array_t dx1_prev, dx2_prev, dx3_prev; + array_t ux1, ux2, ux3; + array_t phi; + array_t weight; + array_t tag; + }; /** - * @brief Iteration of the loop over particles. - * @param p index. + * @brief Per-particle deposit body, shared between the flat and tiled + * kernels. + * + * The caller supplies a `deposit_at(idx..., comp, val)` callback that + * applies the contribution `val` to the J component `comp` at the + * **global** J cell index `idx...` (already includes the `N_GHOSTS` + * offset). The flat kernel's callback simply does + * `J_acc(idx..., comp) += val` on its scatter-view accessor; the tiled + * kernel's callback translates `idx...` into per-tile scratch + * coordinates and uses `Kokkos::atomic_add` on SLM. Either way, this + * function is identical numerically and contains the only deposit math + * in the codebase. + * + * Dead particles return early. The callback is invoked once per cell + * write, with the dimension-appropriate signature: + * - 1D: `deposit_at(int g_i1, int comp, real_t val)` + * - 2D: `deposit_at(int g_i1, int g_i2, int comp, real_t val)` + * - 3D: `deposit_at(int g_i1, int g_i2, int g_i3, int comp, real_t val)` */ - Inline auto operator()(prtlidx_t p) const -> void { + template + Inline void deposit_one_particle(prtlidx_t p, + const PrtlPack& prtls, + const M& metric, + real_t charge, + real_t inv_dt, + DepositFn deposit_at) { + static_assert(O <= 11u, "Shape function order O must be <= 11"); + constexpr auto D = M::Dim; + + const auto& i1 = prtls.i1; + const auto& i2 = prtls.i2; + const auto& i3 = prtls.i3; + const auto& i1_prev = prtls.i1_prev; + const auto& i2_prev = prtls.i2_prev; + const auto& i3_prev = prtls.i3_prev; + const auto& dx1 = prtls.dx1; + const auto& dx2 = prtls.dx2; + const auto& dx3 = prtls.dx3; + const auto& dx1_prev = prtls.dx1_prev; + const auto& dx2_prev = prtls.dx2_prev; + const auto& dx3_prev = prtls.dx3_prev; + const auto& ux1 = prtls.ux1; + const auto& ux2 = prtls.ux2; + const auto& ux3 = prtls.ux3; + const auto& phi = prtls.phi; + const auto& weight = prtls.weight; + const auto& tag = prtls.tag; + if (tag(p) == ParticleTag::dead) { return; } + // recover particle velocity to deposit in unsimulated direction - vec_t vp { ZERO }; - { + [[maybe_unused]] vec_t vp { ZERO }; + // `vp` only feeds the unsimulated-direction current in the 1D + // (jx2, jx3) and 2D (jx3) branches. In 3D every J component comes + // from the Esirkepov/zigzag charge motion and `vp` is never read, + // so the metric transform + 1/sqrt + NaN/Inf guard below is pure + // dead work there — skip it (also frees xp/inv_energy registers). + if constexpr (D != Dim::_3D) { coord_t xp { ZERO }; if constexpr (D == Dim::_1D) { xp[0] = i_di_to_Xi(i1(p), dx1(p)); @@ -167,7 +180,6 @@ namespace kernel { const real_t coeff { weight(p) * charge }; - // ToDo: interpolation_order as parameter if constexpr (O == 0u) { /* Zig-zag deposit @@ -191,8 +203,6 @@ namespace kernel { dx1(p) - dxp_r_1) * coeff * inv_dt }; - auto J_acc = J.access(); - if constexpr (D == Dim::_1D) { const real_t Fx2_1 { HALF * vp[1] * coeff }; const real_t Fx2_2 { HALF * vp[1] * coeff }; @@ -200,18 +210,18 @@ namespace kernel { const real_t Fx3_1 { HALF * vp[2] * coeff }; const real_t Fx3_2 { HALF * vp[2] * coeff }; - J_acc(i1_prev(p) + N_GHOSTS, cur::jx1) += Fx1_1; - J_acc(i1(p) + N_GHOSTS, cur::jx1) += Fx1_2; + deposit_at(i1_prev(p) + N_GHOSTS, cur::jx1, Fx1_1); + deposit_at(i1(p) + N_GHOSTS, cur::jx1, Fx1_2); - J_acc(i1_prev(p) + N_GHOSTS, cur::jx2) += Fx2_1 * (ONE - Wx1_1); - J_acc(i1_prev(p) + N_GHOSTS + 1, cur::jx2) += Fx2_1 * Wx1_1; - J_acc(i1(p) + N_GHOSTS, cur::jx2) += Fx2_2 * (ONE - Wx1_2); - J_acc(i1(p) + N_GHOSTS + 1, cur::jx2) += Fx2_2 * Wx1_2; + deposit_at(i1_prev(p) + N_GHOSTS, cur::jx2, Fx2_1 * (ONE - Wx1_1)); + deposit_at(i1_prev(p) + N_GHOSTS + 1, cur::jx2, Fx2_1 * Wx1_1); + deposit_at(i1(p) + N_GHOSTS, cur::jx2, Fx2_2 * (ONE - Wx1_2)); + deposit_at(i1(p) + N_GHOSTS + 1, cur::jx2, Fx2_2 * Wx1_2); - J_acc(i1_prev(p) + N_GHOSTS, cur::jx3) += Fx3_1 * (ONE - Wx1_1); - J_acc(i1_prev(p) + N_GHOSTS + 1, cur::jx3) += Fx3_1 * Wx1_1; - J_acc(i1(p) + N_GHOSTS, cur::jx3) += Fx3_2 * (ONE - Wx1_2); - J_acc(i1(p) + N_GHOSTS + 1, cur::jx3) += Fx3_2 * Wx1_2; + deposit_at(i1_prev(p) + N_GHOSTS, cur::jx3, Fx3_1 * (ONE - Wx1_1)); + deposit_at(i1_prev(p) + N_GHOSTS + 1, cur::jx3, Fx3_1 * Wx1_1); + deposit_at(i1(p) + N_GHOSTS, cur::jx3, Fx3_2 * (ONE - Wx1_2)); + deposit_at(i1(p) + N_GHOSTS + 1, cur::jx3, Fx3_2 * Wx1_2); } else if constexpr (D == Dim::_2D || D == Dim::_3D) { const auto dxp_r_2 { static_cast(i2(p) == i2_prev(p)) * (dx2(p) + dx2_prev(p)) * @@ -236,51 +246,73 @@ namespace kernel { const real_t Fx3_1 { HALF * vp[2] * coeff }; const real_t Fx3_2 { HALF * vp[2] * coeff }; - J_acc(i1_prev(p) + N_GHOSTS, - i2_prev(p) + N_GHOSTS, - cur::jx1) += Fx1_1 * (ONE - Wx2_1); - J_acc(i1_prev(p) + N_GHOSTS, - i2_prev(p) + N_GHOSTS + 1, - cur::jx1) += Fx1_1 * Wx2_1; - J_acc(i1(p) + N_GHOSTS, i2(p) + N_GHOSTS, cur::jx1) += Fx1_2 * - (ONE - Wx2_2); - J_acc(i1(p) + N_GHOSTS, i2(p) + N_GHOSTS + 1, cur::jx1) += Fx1_2 * Wx2_2; - - J_acc(i1_prev(p) + N_GHOSTS, - i2_prev(p) + N_GHOSTS, - cur::jx2) += Fx2_1 * (ONE - Wx1_1); - J_acc(i1_prev(p) + N_GHOSTS + 1, - i2_prev(p) + N_GHOSTS, - cur::jx2) += Fx2_1 * Wx1_1; - J_acc(i1(p) + N_GHOSTS, i2(p) + N_GHOSTS, cur::jx2) += Fx2_2 * - (ONE - Wx1_2); - J_acc(i1(p) + N_GHOSTS + 1, i2(p) + N_GHOSTS, cur::jx2) += Fx2_2 * Wx1_2; - - J_acc(i1_prev(p) + N_GHOSTS, - i2_prev(p) + N_GHOSTS, - cur::jx3) += Fx3_1 * (ONE - Wx1_1) * (ONE - Wx2_1); - J_acc(i1_prev(p) + N_GHOSTS + 1, - i2_prev(p) + N_GHOSTS, - cur::jx3) += Fx3_1 * Wx1_1 * (ONE - Wx2_1); - J_acc(i1_prev(p) + N_GHOSTS, - i2_prev(p) + N_GHOSTS + 1, - cur::jx3) += Fx3_1 * (ONE - Wx1_1) * Wx2_1; - J_acc(i1_prev(p) + N_GHOSTS + 1, - i2_prev(p) + N_GHOSTS + 1, - cur::jx3) += Fx3_1 * Wx1_1 * Wx2_1; - - J_acc(i1(p) + N_GHOSTS, - i2(p) + N_GHOSTS, - cur::jx3) += Fx3_2 * (ONE - Wx1_2) * (ONE - Wx2_2); - J_acc(i1(p) + N_GHOSTS + 1, - i2(p) + N_GHOSTS, - cur::jx3) += Fx3_2 * Wx1_2 * (ONE - Wx2_2); - J_acc(i1(p) + N_GHOSTS, - i2(p) + N_GHOSTS + 1, - cur::jx3) += Fx3_2 * (ONE - Wx1_2) * Wx2_2; - J_acc(i1(p) + N_GHOSTS + 1, - i2(p) + N_GHOSTS + 1, - cur::jx3) += Fx3_2 * Wx1_2 * Wx2_2; + deposit_at(i1_prev(p) + N_GHOSTS, + i2_prev(p) + N_GHOSTS, + cur::jx1, + Fx1_1 * (ONE - Wx2_1)); + deposit_at(i1_prev(p) + N_GHOSTS, + i2_prev(p) + N_GHOSTS + 1, + cur::jx1, + Fx1_1 * Wx2_1); + deposit_at(i1(p) + N_GHOSTS, + i2(p) + N_GHOSTS, + cur::jx1, + Fx1_2 * (ONE - Wx2_2)); + deposit_at(i1(p) + N_GHOSTS, + i2(p) + N_GHOSTS + 1, + cur::jx1, + Fx1_2 * Wx2_2); + + deposit_at(i1_prev(p) + N_GHOSTS, + i2_prev(p) + N_GHOSTS, + cur::jx2, + Fx2_1 * (ONE - Wx1_1)); + deposit_at(i1_prev(p) + N_GHOSTS + 1, + i2_prev(p) + N_GHOSTS, + cur::jx2, + Fx2_1 * Wx1_1); + deposit_at(i1(p) + N_GHOSTS, + i2(p) + N_GHOSTS, + cur::jx2, + Fx2_2 * (ONE - Wx1_2)); + deposit_at(i1(p) + N_GHOSTS + 1, + i2(p) + N_GHOSTS, + cur::jx2, + Fx2_2 * Wx1_2); + + deposit_at(i1_prev(p) + N_GHOSTS, + i2_prev(p) + N_GHOSTS, + cur::jx3, + Fx3_1 * (ONE - Wx1_1) * (ONE - Wx2_1)); + deposit_at(i1_prev(p) + N_GHOSTS + 1, + i2_prev(p) + N_GHOSTS, + cur::jx3, + Fx3_1 * Wx1_1 * (ONE - Wx2_1)); + deposit_at(i1_prev(p) + N_GHOSTS, + i2_prev(p) + N_GHOSTS + 1, + cur::jx3, + Fx3_1 * (ONE - Wx1_1) * Wx2_1); + deposit_at(i1_prev(p) + N_GHOSTS + 1, + i2_prev(p) + N_GHOSTS + 1, + cur::jx3, + Fx3_1 * Wx1_1 * Wx2_1); + + deposit_at(i1(p) + N_GHOSTS, + i2(p) + N_GHOSTS, + cur::jx3, + Fx3_2 * (ONE - Wx1_2) * (ONE - Wx2_2)); + deposit_at(i1(p) + N_GHOSTS + 1, + i2(p) + N_GHOSTS, + cur::jx3, + Fx3_2 * Wx1_2 * (ONE - Wx2_2)); + deposit_at(i1(p) + N_GHOSTS, + i2(p) + N_GHOSTS + 1, + cur::jx3, + Fx3_2 * (ONE - Wx1_2) * Wx2_2); + deposit_at(i1(p) + N_GHOSTS + 1, + i2(p) + N_GHOSTS + 1, + cur::jx3, + Fx3_2 * Wx1_2 * Wx2_2); } else { const auto dxp_r_3 { static_cast(i3(p) == i3_prev(p)) * (dx3(p) + dx3_prev(p)) * @@ -300,107 +332,131 @@ namespace kernel { dx3(p) - dxp_r_3) * coeff * inv_dt }; - J_acc(i1_prev(p) + N_GHOSTS, - i2_prev(p) + N_GHOSTS, - i3_prev(p) + N_GHOSTS, - cur::jx1) += Fx1_1 * (ONE - Wx2_1) * (ONE - Wx3_1); - J_acc(i1_prev(p) + N_GHOSTS, - i2_prev(p) + N_GHOSTS + 1, - i3_prev(p) + N_GHOSTS, - cur::jx1) += Fx1_1 * Wx2_1 * (ONE - Wx3_1); - J_acc(i1_prev(p) + N_GHOSTS, - i2_prev(p) + N_GHOSTS, - i3_prev(p) + N_GHOSTS + 1, - cur::jx1) += Fx1_1 * (ONE - Wx2_1) * Wx3_1; - J_acc(i1_prev(p) + N_GHOSTS, - i2_prev(p) + N_GHOSTS + 1, - i3_prev(p) + N_GHOSTS + 1, - cur::jx1) += Fx1_1 * Wx2_1 * Wx3_1; - - J_acc(i1(p) + N_GHOSTS, - i2(p) + N_GHOSTS, - i3(p) + N_GHOSTS, - cur::jx1) += Fx1_2 * (ONE - Wx2_2) * (ONE - Wx3_2); - J_acc(i1(p) + N_GHOSTS, - i2(p) + N_GHOSTS + 1, - i3(p) + N_GHOSTS, - cur::jx1) += Fx1_2 * Wx2_2 * (ONE - Wx3_2); - J_acc(i1(p) + N_GHOSTS, - i2(p) + N_GHOSTS, - i3(p) + N_GHOSTS + 1, - cur::jx1) += Fx1_2 * (ONE - Wx2_2) * Wx3_2; - J_acc(i1(p) + N_GHOSTS, - i2(p) + N_GHOSTS + 1, - i3(p) + N_GHOSTS + 1, - cur::jx1) += Fx1_2 * Wx2_2 * Wx3_2; - - J_acc(i1_prev(p) + N_GHOSTS, - i2_prev(p) + N_GHOSTS, - i3_prev(p) + N_GHOSTS, - cur::jx2) += Fx2_1 * (ONE - Wx1_1) * (ONE - Wx3_1); - J_acc(i1_prev(p) + N_GHOSTS + 1, - i2_prev(p) + N_GHOSTS, - i3_prev(p) + N_GHOSTS, - cur::jx2) += Fx2_1 * Wx1_1 * (ONE - Wx3_1); - J_acc(i1_prev(p) + N_GHOSTS, - i2_prev(p) + N_GHOSTS, - i3_prev(p) + N_GHOSTS + 1, - cur::jx2) += Fx2_1 * (ONE - Wx1_1) * Wx3_1; - J_acc(i1_prev(p) + N_GHOSTS + 1, - i2_prev(p) + N_GHOSTS, - i3_prev(p) + N_GHOSTS + 1, - cur::jx2) += Fx2_1 * Wx1_1 * Wx3_1; - - J_acc(i1(p) + N_GHOSTS, - i2(p) + N_GHOSTS, - i3(p) + N_GHOSTS, - cur::jx2) += Fx2_2 * (ONE - Wx1_2) * (ONE - Wx3_2); - J_acc(i1(p) + N_GHOSTS + 1, - i2(p) + N_GHOSTS, - i3(p) + N_GHOSTS, - cur::jx2) += Fx2_2 * Wx1_2 * (ONE - Wx3_2); - J_acc(i1(p) + N_GHOSTS, - i2(p) + N_GHOSTS, - i3(p) + N_GHOSTS + 1, - cur::jx2) += Fx2_2 * (ONE - Wx1_2) * Wx3_2; - J_acc(i1(p) + N_GHOSTS + 1, - i2(p) + N_GHOSTS, - i3(p) + N_GHOSTS + 1, - cur::jx2) += Fx2_2 * Wx1_2 * Wx3_2; - - J_acc(i1_prev(p) + N_GHOSTS, - i2_prev(p) + N_GHOSTS, - i3_prev(p) + N_GHOSTS, - cur::jx3) += Fx3_1 * (ONE - Wx1_1) * (ONE - Wx2_1); - J_acc(i1_prev(p) + N_GHOSTS + 1, - i2_prev(p) + N_GHOSTS, - i3_prev(p) + N_GHOSTS, - cur::jx3) += Fx3_1 * Wx1_1 * (ONE - Wx2_1); - J_acc(i1_prev(p) + N_GHOSTS, - i2_prev(p) + N_GHOSTS + 1, - i3_prev(p) + N_GHOSTS, - cur::jx3) += Fx3_1 * (ONE - Wx1_1) * Wx2_1; - J_acc(i1_prev(p) + N_GHOSTS + 1, - i2_prev(p) + N_GHOSTS + 1, - i3_prev(p) + N_GHOSTS, - cur::jx3) += Fx3_1 * Wx1_1 * Wx2_1; - - J_acc(i1(p) + N_GHOSTS, - i2(p) + N_GHOSTS, - i3(p) + N_GHOSTS, - cur::jx3) += Fx3_2 * (ONE - Wx1_2) * (ONE - Wx2_2); - J_acc(i1(p) + N_GHOSTS + 1, - i2(p) + N_GHOSTS, - i3(p) + N_GHOSTS, - cur::jx3) += Fx3_2 * Wx1_2 * (ONE - Wx2_2); - J_acc(i1(p) + N_GHOSTS, - i2(p) + N_GHOSTS + 1, - i3(p) + N_GHOSTS, - cur::jx3) += Fx3_2 * (ONE - Wx1_2) * Wx2_2; - J_acc(i1(p) + N_GHOSTS + 1, - i2(p) + N_GHOSTS + 1, - i3(p) + N_GHOSTS, - cur::jx3) += Fx3_2 * Wx1_2 * Wx2_2; + deposit_at(i1_prev(p) + N_GHOSTS, + i2_prev(p) + N_GHOSTS, + i3_prev(p) + N_GHOSTS, + cur::jx1, + Fx1_1 * (ONE - Wx2_1) * (ONE - Wx3_1)); + deposit_at(i1_prev(p) + N_GHOSTS, + i2_prev(p) + N_GHOSTS + 1, + i3_prev(p) + N_GHOSTS, + cur::jx1, + Fx1_1 * Wx2_1 * (ONE - Wx3_1)); + deposit_at(i1_prev(p) + N_GHOSTS, + i2_prev(p) + N_GHOSTS, + i3_prev(p) + N_GHOSTS + 1, + cur::jx1, + Fx1_1 * (ONE - Wx2_1) * Wx3_1); + deposit_at(i1_prev(p) + N_GHOSTS, + i2_prev(p) + N_GHOSTS + 1, + i3_prev(p) + N_GHOSTS + 1, + cur::jx1, + Fx1_1 * Wx2_1 * Wx3_1); + + deposit_at(i1(p) + N_GHOSTS, + i2(p) + N_GHOSTS, + i3(p) + N_GHOSTS, + cur::jx1, + Fx1_2 * (ONE - Wx2_2) * (ONE - Wx3_2)); + deposit_at(i1(p) + N_GHOSTS, + i2(p) + N_GHOSTS + 1, + i3(p) + N_GHOSTS, + cur::jx1, + Fx1_2 * Wx2_2 * (ONE - Wx3_2)); + deposit_at(i1(p) + N_GHOSTS, + i2(p) + N_GHOSTS, + i3(p) + N_GHOSTS + 1, + cur::jx1, + Fx1_2 * (ONE - Wx2_2) * Wx3_2); + deposit_at(i1(p) + N_GHOSTS, + i2(p) + N_GHOSTS + 1, + i3(p) + N_GHOSTS + 1, + cur::jx1, + Fx1_2 * Wx2_2 * Wx3_2); + + deposit_at(i1_prev(p) + N_GHOSTS, + i2_prev(p) + N_GHOSTS, + i3_prev(p) + N_GHOSTS, + cur::jx2, + Fx2_1 * (ONE - Wx1_1) * (ONE - Wx3_1)); + deposit_at(i1_prev(p) + N_GHOSTS + 1, + i2_prev(p) + N_GHOSTS, + i3_prev(p) + N_GHOSTS, + cur::jx2, + Fx2_1 * Wx1_1 * (ONE - Wx3_1)); + deposit_at(i1_prev(p) + N_GHOSTS, + i2_prev(p) + N_GHOSTS, + i3_prev(p) + N_GHOSTS + 1, + cur::jx2, + Fx2_1 * (ONE - Wx1_1) * Wx3_1); + deposit_at(i1_prev(p) + N_GHOSTS + 1, + i2_prev(p) + N_GHOSTS, + i3_prev(p) + N_GHOSTS + 1, + cur::jx2, + Fx2_1 * Wx1_1 * Wx3_1); + + deposit_at(i1(p) + N_GHOSTS, + i2(p) + N_GHOSTS, + i3(p) + N_GHOSTS, + cur::jx2, + Fx2_2 * (ONE - Wx1_2) * (ONE - Wx3_2)); + deposit_at(i1(p) + N_GHOSTS + 1, + i2(p) + N_GHOSTS, + i3(p) + N_GHOSTS, + cur::jx2, + Fx2_2 * Wx1_2 * (ONE - Wx3_2)); + deposit_at(i1(p) + N_GHOSTS, + i2(p) + N_GHOSTS, + i3(p) + N_GHOSTS + 1, + cur::jx2, + Fx2_2 * (ONE - Wx1_2) * Wx3_2); + deposit_at(i1(p) + N_GHOSTS + 1, + i2(p) + N_GHOSTS, + i3(p) + N_GHOSTS + 1, + cur::jx2, + Fx2_2 * Wx1_2 * Wx3_2); + + deposit_at(i1_prev(p) + N_GHOSTS, + i2_prev(p) + N_GHOSTS, + i3_prev(p) + N_GHOSTS, + cur::jx3, + Fx3_1 * (ONE - Wx1_1) * (ONE - Wx2_1)); + deposit_at(i1_prev(p) + N_GHOSTS + 1, + i2_prev(p) + N_GHOSTS, + i3_prev(p) + N_GHOSTS, + cur::jx3, + Fx3_1 * Wx1_1 * (ONE - Wx2_1)); + deposit_at(i1_prev(p) + N_GHOSTS, + i2_prev(p) + N_GHOSTS + 1, + i3_prev(p) + N_GHOSTS, + cur::jx3, + Fx3_1 * (ONE - Wx1_1) * Wx2_1); + deposit_at(i1_prev(p) + N_GHOSTS + 1, + i2_prev(p) + N_GHOSTS + 1, + i3_prev(p) + N_GHOSTS, + cur::jx3, + Fx3_1 * Wx1_1 * Wx2_1); + + deposit_at(i1(p) + N_GHOSTS, + i2(p) + N_GHOSTS, + i3(p) + N_GHOSTS, + cur::jx3, + Fx3_2 * (ONE - Wx1_2) * (ONE - Wx2_2)); + deposit_at(i1(p) + N_GHOSTS + 1, + i2(p) + N_GHOSTS, + i3(p) + N_GHOSTS, + cur::jx3, + Fx3_2 * Wx1_2 * (ONE - Wx2_2)); + deposit_at(i1(p) + N_GHOSTS, + i2(p) + N_GHOSTS + 1, + i3(p) + N_GHOSTS, + cur::jx3, + Fx3_2 * (ONE - Wx1_2) * Wx2_2); + deposit_at(i1(p) + N_GHOSTS + 1, + i2(p) + N_GHOSTS + 1, + i3(p) + N_GHOSTS, + cur::jx3, + Fx3_2 * Wx1_2 * Wx2_2); } } } else if constexpr ((O >= 1u) and (O <= 11u)) { @@ -421,33 +477,14 @@ namespace kernel { fS_x1); if constexpr (D == Dim::_1D) { - // define weight vectors - real_t Wx1[O + 2]; - real_t Wx23[O + 2]; - - // Calculate weight function -#pragma unroll - for (int i = 0; i < O + 2; ++i) { - // Esirkepov 2001, Eq. 38 for 1D case - Wx1[i] = fS_x1[i] - iS_x1[i]; - Wx23[i] = HALF * (fS_x1[i] + iS_x1[i]); - } - - // contribution within the shape function stencil - real_t jx1[O + 2]; - - // prefactors for j update + // (1D): fused Esirkepov, no [O+2] temporaries. + // jx1[i] = -Qdx1dt * sum_{i'=0}^{i} (fS_x1[i'] - iS_x1[i']) + // = -Qdx1dt * P1[i] (Eq. 38, 1D) + // Wx23[i] = HALF * (fS_x1[i] + iS_x1[i]) (computed inline) const real_t Qdx1dt = coeff * inv_dt; const real_t QVx2 = coeff * vp[1]; const real_t QVx3 = coeff * vp[2]; - // Calculate current contribution - jx1[0] = -Qdx1dt * Wx1[0]; -#pragma unroll - for (int i = 1; i < O + 2; ++i) { - jx1[i] = jx1[i - 1] - Qdx1dt * Wx1[i]; - } - // account for ghost cells i1_min += N_GHOSTS; i1_max += N_GHOSTS; @@ -455,21 +492,18 @@ namespace kernel { // get number of update indices for asymmetric movement const int di_x1 = i1_max - i1_min; - /* - Current update - */ - auto J_acc = J.access(); - - for (int i = 0; i < di_x1; ++i) { - J_acc(i1_min + i, cur::jx1) += jx1[i]; - } - + // Current update — fused over the union line so the J cell + // stays L1-resident across the 3 component atomic_adds. + real_t P1 = ZERO; for (int i = 0; i <= di_x1; ++i) { - J_acc(i1_min + i, cur::jx2) += QVx2 * Wx23[i]; - } - - for (int i = 0; i <= di_x1; ++i) { - J_acc(i1_min + i, cur::jx3) += QVx3 * Wx23[i]; + P1 += fS_x1[i] - iS_x1[i]; + const int gi = i1_min + i; + const real_t Wx23 = HALF * (fS_x1[i] + iS_x1[i]); + if (i < di_x1) { + deposit_at(gi, cur::jx1, -Qdx1dt * P1); + } + deposit_at(gi, cur::jx2, QVx2 * Wx23); + deposit_at(gi, cur::jx3, QVx3 * Wx23); } } else if constexpr (D == Dim::_2D) { @@ -489,63 +523,23 @@ namespace kernel { iS_x2, fS_x2); - // define weight tensors - real_t Wx1[O + 2][O + 2]; - real_t Wx2[O + 2][O + 2]; - real_t Wx3[O + 2][O + 2]; - -// Calculate weight function -#pragma unroll - for (int i = 0; i < O + 2; ++i) { -#pragma unroll - for (int j = 0; j < O + 2; ++j) { - // Esirkepov 2001, Eq. 38 (simplified) - Wx1[i][j] = HALF * (fS_x1[i] - iS_x1[i]) * (fS_x2[j] + iS_x2[j]); - - Wx2[i][j] = HALF * (fS_x1[i] + iS_x1[i]) * (fS_x2[j] - iS_x2[j]); - - Wx3[i][j] = THIRD * (fS_x2[j] * (HALF * iS_x1[i] + fS_x1[i]) + - iS_x2[j] * (HALF * fS_x1[i] + iS_x1[i])); - } - } - - // contribution within the shape function stencil - real_t jx1[O + 2][O + 2], jx2[O + 2][O + 2]; - - // prefactors for j update - const real_t Qdx1dt = coeff * inv_dt; - const real_t Qdx2dt = coeff * inv_dt; - const real_t QVx3 = coeff * vp[2]; - - // Calculate current contribution - - // jx1 -#pragma unroll - for (int j = 0; j < O + 2; ++j) { - jx1[0][j] = -Qdx1dt * Wx1[0][j]; - } - -#pragma unroll - for (int i = 1; i < O + 2; ++i) { -#pragma unroll - for (int j = 0; j < O + 2; ++j) { - jx1[i][j] = jx1[i - 1][j] - Qdx1dt * Wx1[i][j]; - } - } - - // jx2 -#pragma unroll - for (int i = 0; i < O + 2; ++i) { - jx2[i][0] = -Qdx2dt * Wx2[i][0]; - } - -#pragma unroll - for (int j = 1; j < O + 2; ++j) { -#pragma unroll - for (int i = 0; i < O + 2; ++i) { - jx2[i][j] = jx2[i][j - 1] - Qdx2dt * Wx2[i][j]; - } - } + // (2D): fused Esirkepov, no [O+2]^2 temporaries. + // + // Esirkepov 2001 Eq. 38 (simplified) is separable: with + // P1[i] = sum_{i'=0}^{i} (fS_x1[i'] - iS_x1[i']) and + // P2[j] = sum_{j'=0}^{j} (fS_x2[j'] - iS_x2[j']), + // jx1[i][j] = -Q*HALF * P1[i] * (fS_x2[j] + iS_x2[j]) + // jx2[i][j] = -Q*HALF * P2[j] * (fS_x1[i] + iS_x1[i]) + // Wx3[i][j] = THIRD*( fS_x2[j]*(HALF*iS_x1[i]+fS_x1[i]) + // + iS_x2[j]*(HALF*fS_x1[i]+iS_x1[i]) ) + // with Q = coeff*inv_dt (Qdx1dt == Qdx2dt). Same value as the + // old explicit Wx/jx tensors up to FP reassociation; + // charge-conserving by construction. Prefix sums carried as + // running scalars, so the only per-thread state is the + // existing 1D shape arrays. + const real_t QVx3 = coeff * vp[2]; + // -Q*HALF prefactor (Qdx1dt == Qdx2dt == coeff*inv_dt) + const real_t cf = -(coeff * inv_dt) * HALF; // account for ghost cells i1_min += N_GHOSTS; @@ -557,26 +551,30 @@ namespace kernel { const int di_x1 = i1_max - i1_min; const int di_x2 = i2_max - i2_min; - /* - Current update - */ - auto J_acc = J.access(); - - for (int i = 0; i < di_x1; ++i) { - for (int j = 0; j <= di_x2; ++j) { - J_acc(i1_min + i, i2_min + j, cur::jx1) += jx1[i][j]; - } - } - - for (int i = 0; i <= di_x1; ++i) { - for (int j = 0; j < di_x2; ++j) { - J_acc(i1_min + i, i2_min + j, cur::jx2) += jx2[i][j]; - } - } - + // Current update — fused over the union plane so the J cell + // line stays L1-resident across the 3 component atomic_adds. + real_t P1 = ZERO; for (int i = 0; i <= di_x1; ++i) { + P1 += fS_x1[i] - iS_x1[i]; + const int gi = i1_min + i; + const real_t iSx1 = iS_x1[i]; + const real_t fSx1 = fS_x1[i]; + const real_t A1 = fSx1 + iSx1; // jx2 cross-factor + real_t P2 = ZERO; for (int j = 0; j <= di_x2; ++j) { - J_acc(i1_min + i, i2_min + j, cur::jx3) += QVx3 * Wx3[i][j]; + P2 += fS_x2[j] - iS_x2[j]; + const int gj = i2_min + j; + const real_t iSx2 = iS_x2[j]; + const real_t fSx2 = fS_x2[j]; + if (i < di_x1) { + deposit_at(gi, gj, cur::jx1, cf * P1 * (fSx2 + iSx2)); + } + if (j < di_x2) { + deposit_at(gi, gj, cur::jx2, cf * P2 * A1); + } + const real_t Wx3 = THIRD * (fSx2 * (HALF * iSx1 + fSx1) + + iSx2 * (HALF * fSx1 + iSx1)); + deposit_at(gi, gj, cur::jx3, QVx3 * Wx3); } } @@ -610,104 +608,33 @@ namespace kernel { iS_x3, fS_x3); - // define weight tensors - real_t Wx1[O + 2][O + 2][O + 2]; - real_t Wx2[O + 2][O + 2][O + 2]; - real_t Wx3[O + 2][O + 2][O + 2]; - -// Calculate weight function -#pragma unroll - for (int i = 0; i < O + 2; ++i) { -#pragma unroll - for (int j = 0; j < O + 2; ++j) { -#pragma unroll - for (int k = 0; k < O + 2; ++k) { - // Esirkepov 2001, Eq. 31 - Wx1[i][j][k] = THIRD * (fS_x1[i] - iS_x1[i]) * - ((iS_x2[j] * iS_x3[k] + fS_x2[j] * fS_x3[k]) + - HALF * (iS_x3[k] * fS_x2[j] + iS_x2[j] * fS_x3[k])); - - Wx2[i][j][k] = THIRD * (fS_x2[j] - iS_x2[j]) * - (iS_x1[i] * iS_x3[k] + fS_x1[i] * fS_x3[k] + - HALF * (iS_x3[k] * fS_x1[i] + iS_x1[i] * fS_x3[k])); - - Wx3[i][j][k] = THIRD * (fS_x3[k] - iS_x3[k]) * - (iS_x1[i] * iS_x2[j] + fS_x1[i] * fS_x2[j] + - HALF * (iS_x1[i] * fS_x2[j] + iS_x2[j] * fS_x1[i])); - } - } - } - - // contribution within the shape function stencil - real_t jx1[O + 2][O + 2][O + 2], jx2[O + 2][O + 2][O + 2], - jx3[O + 2][O + 2][O + 2]; - - // prefactors to j update - const real_t Qdxdt = coeff * inv_dt; - const real_t Qdydt = coeff * inv_dt; - const real_t Qdzdt = coeff * inv_dt; - - // Calculate current contribution - - // jx1 -#pragma unroll - for (int j = 0; j < O + 2; ++j) { -#pragma unroll - for (int k = 0; k < O + 2; ++k) { - jx1[0][j][k] = -Qdxdt * Wx1[0][j][k]; - } - } - -#pragma unroll - for (int i = 1; i < O + 2; ++i) { -#pragma unroll - for (int j = 0; j < O + 2; ++j) { -#pragma unroll - for (int k = 0; k < O + 2; ++k) { - jx1[i][j][k] = jx1[i - 1][j][k] - Qdxdt * Wx1[i][j][k]; - } - } - } - - // jx2 -#pragma unroll - for (int i = 0; i < O + 2; ++i) { -#pragma unroll - for (int k = 0; k < O + 2; ++k) { - jx2[i][0][k] = -Qdydt * Wx2[i][0][k]; - } - } - -#pragma unroll - for (int i = 0; i < O + 2; ++i) { -#pragma unroll - for (int j = 1; j < O + 2; ++j) { -#pragma unroll - for (int k = 0; k < O + 2; ++k) { - jx2[i][j][k] = jx2[i][j - 1][k] - Qdydt * Wx2[i][j][k]; - } - } - } - - // jx3 -#pragma unroll - for (int i = 0; i < O + 2; ++i) { -#pragma unroll - for (int j = 0; j < O + 2; ++j) { - jx3[i][j][0] = -Qdydt * Wx3[i][j][0]; - } - } - -#pragma unroll - for (int i = 0; i < O + 2; ++i) { -#pragma unroll - for (int j = 0; j < O + 2; ++j) { -#pragma unroll - for (int k = 1; k < O + 2; ++k) { - jx3[i][j][k] = jx3[i][j][k - 1] - Qdzdt * Wx3[i][j][k]; - } - } - } + // fused Esirkepov, no (O+2)^3 temporaries. + // + // The Esirkepov 3D current (2001, Eq. 31) is separable: with + // P1[i] = sum_{i'=0}^{i} (fS_x1[i'] - iS_x1[i']) (and likewise + // P2[j], P3[k]) the cumulative-sum currents collapse to + // + // jx1[i][j][k] = -Q*THIRD * P1[i] * G23(j,k) + // jx2[i][j][k] = -Q*THIRD * P2[j] * H13(i,k) + // jx3[i][j][k] = -Q*THIRD * P3[k] * F12(i,j) + // + // with the 1D-shape cross-factors + // + // G23(j,k) = iS_x2[j]*iS_x3[k] + fS_x2[j]*fS_x3[k] + // + HALF*(iS_x3[k]*fS_x2[j] + iS_x2[j]*fS_x3[k]) + // H13(i,k) = iS_x1[i]*iS_x3[k] + fS_x1[i]*fS_x3[k] + // + HALF*(iS_x3[k]*fS_x1[i] + iS_x1[i]*fS_x3[k]) + // F12(i,j) = iS_x1[i]*iS_x2[j] + fS_x1[i]*fS_x2[j] + // + HALF*(iS_x1[i]*fS_x2[j] + iS_x2[j]*fS_x1[i]) + // + // and Q = coeff*inv_dt (Qdxdt == Qdydt == Qdzdt). This is the + // same value as the old explicit Wx/jx tensors up to + // floating-point reassociation: charge-conserving by + // construction (the Esirkepov decomposition is exact). The + // prefix sums are carried as running scalars in the deposit + // loop, so the only per-thread state is the existing 1D shape + // arrays (no (O+2)^3 / (O+2)^2 locals, hence far fewer VGPRs + // and no private-memory tensor traffic). // account for ghost cells i1_min += N_GHOSTS; @@ -722,31 +649,50 @@ namespace kernel { const int di_x2 = i2_max - i2_min; const int di_x3 = i3_max - i3_min; + // -Q*THIRD prefactor (Qdxdt == Qdydt == Qdzdt == coeff*inv_dt) + const real_t cf = -(coeff * inv_dt) * THIRD; + /* - Current update + Current update — fused over the union cube so the J cell + line stays L1-resident across the 3 component atomic_adds. + Per-cell branches on (i + class DepositCurrents_kernel { + static_assert(O <= 11u, "Shape function order O must be <= 11"); + static constexpr auto D = M::Dim; + + scatter_ndfield_t J; + deposit::PrtlPack prtls; + const M metric; + const real_t charge, inv_dt; + + public: + DepositCurrents_kernel(const scatter_ndfield_t& scatter_cur, + const array_t& i1, + const array_t& i2, + const array_t& i3, + const array_t& i1_prev, + const array_t& i2_prev, + const array_t& i3_prev, + const array_t& dx1, + const array_t& dx2, + const array_t& dx3, + const array_t& dx1_prev, + const array_t& dx2_prev, + const array_t& dx3_prev, + const array_t& ux1, + const array_t& ux2, + const array_t& ux3, + const array_t& phi, + const array_t& weight, + const array_t& tag, + const M& metric, + real_t charge, + const real_t dt) + : J { scatter_cur } + , prtls { i1, i2, i3, i1_prev, i2_prev, i3_prev, + dx1, dx2, dx3, dx1_prev, dx2_prev, dx3_prev, + ux1, ux2, ux3, phi, weight, tag } + , metric { metric } + , charge { charge } + , inv_dt { ONE / dt } { + raise::ErrorIf( + (O == 2u and N_GHOSTS < 2), + "Order of interpolation is 2, but number of ghost cells is < 2", + HERE); + } + + Inline auto operator()(prtlidx_t p) const -> void { + auto J_acc = J.access(); + if constexpr (D == Dim::_1D) { + deposit::deposit_one_particle( + p, + prtls, + metric, + charge, + inv_dt, + [&](int g_i1, int comp, real_t v) { + J_acc(g_i1, comp) += v; + }); + } else if constexpr (D == Dim::_2D) { + deposit::deposit_one_particle( + p, + prtls, + metric, + charge, + inv_dt, + [&](int g_i1, int g_i2, int comp, real_t v) { + J_acc(g_i1, g_i2, comp) += v; + }); + } else if constexpr (D == Dim::_3D) { + deposit::deposit_one_particle( + p, + prtls, + metric, + charge, + inv_dt, + [&](int g_i1, int g_i2, int g_i3, int comp, real_t v) { + J_acc(g_i1, g_i2, g_i3, comp) += v; + }); + } + } }; + +#if defined(TEAM_POLICY) + + /** + * @brief Tiled current-deposition kernel. + * + * One team per spatial tile (`league_size = ntiles_total`). Each team + * accumulates particle contributions into a per-team scratch buffer of + * shape `(T_TILE + 2*HALO)^D × 3` real_t, where `HALO = O + 1` cells per + * side. Scratch atomics live in SLM (PVC: ~5–10 cycles per + * `atomic_add`); the global J is touched only once per scratch cell at + * flush time. Compared with the flat scatter-view kernel: + * - global atomic pressure ~ (T_TILE + 2*HALO)^D × 3 per tile + * instead of (stencil writes per particle × particles) + * - per-particle stencil writes are tile-local (SLM) instead of + * scattering through global HBM + * + * Supports `O ∈ {0, ..., 11}`. `O == 0` (zigzag) is wired for + * A/B benchmarking against the flat scatter-view kernel — its narrow + * stencil typically makes scratch alloc/zero/flush overhead a + * regression there, but it's good to be able to measure the + * crossover. To revert and use flat for zigzag-only builds, change + * the dispatch in `engines/srpic/currents.h` from + * `#if defined(TEAM_POLICY)` to + * `#if defined(TEAM_POLICY) && (SHAPE_ORDER > 0)`. + * + * Particle iteration order is governed by `tile_offsets`: tile `t` + * owns particles `[tile_offsets(t), tile_offsets(t+1))`, post-sort. + * `SortSpatially` (`particles_sort.cpp`) is responsible for keeping + * the SoA arrays consistent with that. + * + * **Halo sizing and escape valve.** Sort runs at the end of the + * previous step (see `srpic.hpp`), so at deposit time the particle + * has already been pushed once — its `min(i, i_prev)` may differ + * from the bin key by one cell of drift per step elapsed since the + * last sort. The scratch HALO is `STENCIL_REACH(O) + DRIFT`, where + * `STENCIL_REACH = 2` for zigzag (writes `{i_prev, i_prev+1, i, + * i+1}` ⇒ +2 above `min(i, i_prev)` with `|Δi|=1`) and `O` for + * Esirkepov, and `DRIFT` is a fixed constant (1) covering the one + * guaranteed post-sort pusher step. + * + * HALO is sized for the *common* (every-step-sorted) case, not for + * a worst-case sort cadence: correctness does **not** depend on it. + * Any particle whose stencil escapes the scratch tile — because it + * drifted further than `DRIFT` (e.g. a large runtime + * `spatial_sorting_interval`), or because the halo is otherwise + * undersized — silently falls back to a direct, bounds-clipped + * `Kokkos::atomic_add` on the global J view. That path is + * charge-conserving (each particle's stencil is deposited exactly + * once, partly to private SLM scratch and partly to global J, and + * scratch is flushed once via `atomic_add`); it is merely slower + * per write. Sorting less often than every step therefore costs + * escape-valve traffic, never accuracy. + */ + template + class DepositCurrents_kernel_tiled { + static_assert(O <= 11u, "Shape order O must be <= 11"); + static_assert(T_TILE > 0u, "T_TILE must be positive"); + static constexpr auto D = M::Dim; + + // Per-side scratch halo, derived from first principles. + // + // total halo = stencil_reach(O) + drift_between_sort_and_deposit + // + // stencil_reach(O) — maximum cells the deposit writes ABOVE + // min(i, i_prev) under CFL |v·dt/dx| ≤ 1/2: + // - O == 0 (zigzag): writes {i_prev, i_prev+1, i, i+1} ⇒ +2 + // - O >= 1 Esirkepov: `for_deposit` returns an (O+2)-wide + // array but only O+1 entries are non-zero, and the union + // window satisfies `i_max - i_min <= O+1` (see + // particle_shapes.hpp::for_deposit). The genuine one-sided + // reach above min(i, i_prev) is therefore O, not O+1 — the + // old `O+1` carried one extra cell of conservative padding + // on top of the already-conservative drift term below. + // + // drift — sort runs at end-of-step (see srpic.hpp), so a particle + // sees exactly one pusher step before the *next* step's deposit + // when sorted every step (the common case). DRIFT is therefore a + // fixed constant of 1, NOT a compile-time function of the runtime + // sort cadence. Sizing the halo for the common case (rather than a + // worst-case sort interval) is what keeps the scratch small enough + // for good occupancy; a species sorted less often than + // every step just drifts past the halo and takes the global-J + // escape valve more often — correct, only slower (see the class + // doc-comment for why this is charge-conserving). + // + static constexpr int STENCIL_REACH = (O == 0u) + ? 2 + : static_cast(O); + static constexpr int DRIFT = 1; + static constexpr int HALO = STENCIL_REACH + DRIFT; + static constexpr int TE = static_cast(T_TILE) + 2 * HALO; + + using exec_space = Kokkos::DefaultExecutionSpace; + using team_policy = Kokkos::TeamPolicy; + using member_t = typename team_policy::member_type; + using scratch_mem = typename exec_space::scratch_memory_space; + + // Scratch view types: trailing extent of 3 (jx1, jx2, jx3 components) + // is fixed by a runtime extent so we don't need a separate dimension + // template per component count. + using scratch_1d_t = Kokkos::View>; + using scratch_2d_t = Kokkos::View>; + using scratch_3d_t = Kokkos::View>; + + ndfield_t J; + deposit::PrtlPack prtls; + const M metric; + const real_t charge, inv_dt; + + // Tile metadata produced by SortSpatially. + array_t tile_offsets; + ncells_t ntx1 { 1u }, ntx2 { 1u }, ntx3 { 1u }; + ncells_t total_tiles { 0u }; + + // J's full storage extent including all ghost cells. Used to clip + // the cooperative flush so that a partial tile at the high end of + // the domain does not over-write past the J view. + int j_ext1 { 0 }, j_ext2 { 0 }, j_ext3 { 0 }; + + public: + DepositCurrents_kernel_tiled(const ndfield_t& cur, + const array_t& i1, + const array_t& i2, + const array_t& i3, + const array_t& i1_prev, + const array_t& i2_prev, + const array_t& i3_prev, + const array_t& dx1, + const array_t& dx2, + const array_t& dx3, + const array_t& dx1_prev, + const array_t& dx2_prev, + const array_t& dx3_prev, + const array_t& ux1, + const array_t& ux2, + const array_t& ux3, + const array_t& phi, + const array_t& weight, + const array_t& tag, + const M& metric, + real_t charge, + const real_t dt, + const TileLayout& layout) + : J { cur } + , prtls { i1, i2, i3, i1_prev, i2_prev, i3_prev, + dx1, dx2, dx3, dx1_prev, dx2_prev, dx3_prev, + ux1, ux2, ux3, phi, weight, tag } + , metric { metric } + , charge { charge } + , inv_dt { ONE / dt } + , tile_offsets { layout.tile_offsets } + , ntx1 { layout.ntiles_per_axis[0] } + , ntx2 { layout.ntiles_per_axis[1] } + , ntx3 { layout.ntiles_per_axis[2] } + , total_tiles { layout.ntiles_total } { + raise::ErrorIf( + layout.tile_size != T_TILE, + "Tiled deposit launched with mismatched T_TILE and runtime tile_size", + HERE); + // Note: HALO is allowed to exceed N_GHOSTS. The cooperative + // scratch→J flush and the per-particle escape valve both bounds-clip + // their writes against `j_ext*` so writes that would land past J's + // ghost stripe are silently dropped (they only ever come from a + // particle whose stencil reaches into the domain ghost region, where + // CommunicateFields will re-supply the contribution). + if constexpr (D == Dim::_1D || D == Dim::_2D || D == Dim::_3D) { + j_ext1 = static_cast(cur.extent(0)); + } + if constexpr (D == Dim::_2D || D == Dim::_3D) { + j_ext2 = static_cast(cur.extent(1)); + } + if constexpr (D == Dim::_3D) { + j_ext3 = static_cast(cur.extent(2)); + } + } + + /** + * @brief Per-team scratch size in bytes. Used by the launcher to set + * `team_policy.set_scratch_size(0, Kokkos::PerTeam(bytes))`. + */ + static constexpr std::size_t scratch_bytes() { + if constexpr (D == Dim::_1D) { + return scratch_1d_t::shmem_size(TE, 3); + } else if constexpr (D == Dim::_2D) { + return scratch_2d_t::shmem_size(TE, TE, 3); + } else { + return scratch_3d_t::shmem_size(TE, TE, TE, 3); + } + } + + KOKKOS_INLINE_FUNCTION + void operator()(const member_t& team) const { + const auto tile_id = static_cast(team.league_rank()); + // Tile coordinates (tile-grid indices) → tile origin in **active** + // cell coords (no ghost offset). Using ncells_t to match the linearised + // tile index produced by SortSpatially. + ncells_t tx1 = 0, tx2 = 0, tx3 = 0; + if constexpr (D == Dim::_1D) { + tx1 = tile_id; + } else if constexpr (D == Dim::_2D) { + tx1 = tile_id / ntx2; + tx2 = tile_id - tx1 * ntx2; + } else { + const auto plane = ntx2 * ntx3; + tx1 = tile_id / plane; + const auto rem = tile_id - tx1 * plane; + tx2 = rem / ntx3; + tx3 = rem - tx2 * ntx3; + } + // origin_active = lowest active-cell index in the tile (no ghost). + // origin_J = same value translated into J's storage coordinate + // (i.e. plus N_GHOSTS). + // origin_J_low = J coordinate of scratch index 0 (i.e. origin_J - HALO). + // local index `li` in scratch ↔ global J index `gi = li + origin_J_low`. + const int origin_J1_low = static_cast(tx1 * T_TILE) + + static_cast(N_GHOSTS) - HALO; + const int origin_J2_low = static_cast(tx2 * T_TILE) + + static_cast(N_GHOSTS) - HALO; + const int origin_J3_low = static_cast(tx3 * T_TILE) + + static_cast(N_GHOSTS) - HALO; + + // Allocate scratch and cooperatively zero-fill it. + if constexpr (D == Dim::_1D) { + scratch_1d_t scr(team.team_scratch(0), TE, 3); + Kokkos::parallel_for( + Kokkos::TeamThreadRange(team, TE * 3), + [&](const int idx) { + const int li = idx / 3; + const int c = idx - li * 3; + scr(li, c) = ZERO; + }); + team.team_barrier(); + + const auto p_begin = tile_offsets(tile_id); + const auto p_end = tile_offsets(tile_id + 1u); + const int e1_d = j_ext1; + Kokkos::parallel_for( + Kokkos::TeamThreadRange(team, p_begin, p_end), + [&](const npart_t p) { + deposit::deposit_one_particle( + p, + prtls, + metric, + charge, + inv_dt, + // Escape valve: a particle whose stencil reaches past the + // tile's scratch (e.g. exceeded the compile-time + // STENCIL_REACH + DRIFT budget) falls back to a direct + // atomic_add on the global J view. Bounds-clipped against + // J's storage extent so writes past the domain ghost stripe + // are dropped (matches the cooperative flush below; those + // contributions are re-supplied by SynchronizeFields(J)). + [&](int g_i1, int comp, real_t v) { + const int li = g_i1 - origin_J1_low; + if (li >= 0 && li < TE) { + Kokkos::atomic_add(&scr(li, comp), v); + } else if (g_i1 >= 0 && g_i1 < e1_d) { + Kokkos::atomic_add(&J(g_i1, comp), v); + } + }); + }); + team.team_barrier(); + + // Cooperative flush of scratch to global J. Bounds-clip against + // the J view extent in case a partial high-end tile (or non-zero + // halo at domain edges) would otherwise write past J. + const int e1 = j_ext1; + Kokkos::parallel_for( + Kokkos::TeamThreadRange(team, TE * 3), + [&](const int idx) { + const int li = idx / 3; + const int c = idx - li * 3; + const int gi = li + origin_J1_low; + if (gi < 0 || gi >= e1) { + return; + } + const real_t v = scr(li, c); + if (v != ZERO) { + Kokkos::atomic_add(&J(gi, c), v); + } + }); + } else if constexpr (D == Dim::_2D) { + scratch_2d_t scr(team.team_scratch(0), TE, TE, 3); + Kokkos::parallel_for( + Kokkos::TeamThreadRange(team, TE * TE * 3), + [&](const int idx) { + const int lij = idx / 3; + const int c = idx - lij * 3; + const int li = lij / TE; + const int lj = lij - li * TE; + scr(li, lj, c) = ZERO; + }); + team.team_barrier(); + + const auto p_begin = tile_offsets(tile_id); + const auto p_end = tile_offsets(tile_id + 1u); + const int e1_d = j_ext1; + const int e2_d = j_ext2; + Kokkos::parallel_for( + Kokkos::TeamThreadRange(team, p_begin, p_end), + [&](const npart_t p) { + deposit::deposit_one_particle( + p, + prtls, + metric, + charge, + inv_dt, + // See 1D branch for rationale. + [&](int g_i1, int g_i2, int comp, real_t v) { + const int li = g_i1 - origin_J1_low; + const int lj = g_i2 - origin_J2_low; + if (li >= 0 && li < TE && lj >= 0 && lj < TE) { + Kokkos::atomic_add(&scr(li, lj, comp), v); + } else if (g_i1 >= 0 && g_i1 < e1_d && g_i2 >= 0 && + g_i2 < e2_d) { + Kokkos::atomic_add(&J(g_i1, g_i2, comp), v); + } + }); + }); + team.team_barrier(); + + const int e1 = j_ext1; + const int e2 = j_ext2; + Kokkos::parallel_for( + Kokkos::TeamThreadRange(team, TE * TE * 3), + [&](const int idx) { + const int lij = idx / 3; + const int c = idx - lij * 3; + const int li = lij / TE; + const int lj = lij - li * TE; + const int gi = li + origin_J1_low; + const int gj = lj + origin_J2_low; + if (gi < 0 || gi >= e1 || gj < 0 || gj >= e2) { + return; + } + const real_t v = scr(li, lj, c); + if (v != ZERO) { + Kokkos::atomic_add(&J(gi, gj, c), v); + } + }); + } else if constexpr (D == Dim::_3D) { + scratch_3d_t scr(team.team_scratch(0), TE, TE, TE, 3); + const int cells = TE * TE * TE; + Kokkos::parallel_for( + Kokkos::TeamThreadRange(team, cells * 3), + [&](const int idx) { + const int lijk = idx / 3; + const int c = idx - lijk * 3; + const int li = lijk / (TE * TE); + const int rem = lijk - li * TE * TE; + const int lj = rem / TE; + const int lk = rem - lj * TE; + scr(li, lj, lk, c) = ZERO; + }); + team.team_barrier(); + + const auto p_begin = tile_offsets(tile_id); + const auto p_end = tile_offsets(tile_id + 1u); + const int e1_d = j_ext1; + const int e2_d = j_ext2; + const int e3_d = j_ext3; + Kokkos::parallel_for( + Kokkos::TeamThreadRange(team, p_begin, p_end), + [&](const npart_t p) { + deposit::deposit_one_particle( + p, + prtls, + metric, + charge, + inv_dt, + // See 1D branch for rationale. + [&](int g_i1, int g_i2, int g_i3, int comp, real_t v) { + const int li = g_i1 - origin_J1_low; + const int lj = g_i2 - origin_J2_low; + const int lk = g_i3 - origin_J3_low; + if (li >= 0 && li < TE && lj >= 0 && lj < TE && lk >= 0 && + lk < TE) { + Kokkos::atomic_add(&scr(li, lj, lk, comp), v); + } else if (g_i1 >= 0 && g_i1 < e1_d && g_i2 >= 0 && + g_i2 < e2_d && g_i3 >= 0 && g_i3 < e3_d) { + Kokkos::atomic_add(&J(g_i1, g_i2, g_i3, comp), v); + } + }); + }); + team.team_barrier(); + + const int e1 = j_ext1; + const int e2 = j_ext2; + const int e3 = j_ext3; + Kokkos::parallel_for( + Kokkos::TeamThreadRange(team, cells * 3), + [&](const int idx) { + const int lijk = idx / 3; + const int c = idx - lijk * 3; + const int li = lijk / (TE * TE); + const int rem = lijk - li * TE * TE; + const int lj = rem / TE; + const int lk = rem - lj * TE; + const int gi = li + origin_J1_low; + const int gj = lj + origin_J2_low; + const int gk = lk + origin_J3_low; + if (gi < 0 || gi >= e1 || gj < 0 || gj >= e2 || gk < 0 || + gk >= e3) { + return; + } + const real_t v = scr(li, lj, lk, c); + if (v != ZERO) { + Kokkos::atomic_add(&J(gi, gj, gk, c), v); + } + }); + } + } + }; + +#endif // TEAM_POLICY + } // namespace kernel #undef i_di_to_Xi diff --git a/tests/framework/CMakeLists.txt b/tests/framework/CMakeLists.txt index d6dc295af..9a2e2865e 100644 --- a/tests/framework/CMakeLists.txt +++ b/tests/framework/CMakeLists.txt @@ -44,6 +44,12 @@ else() gen_test(particles_sort false) endif() +# team_policy X-3: per-backend sort_by_key permutation test (only built +# when the compile-time team_policy toggle is on). +if(${team_policy}) + gen_test(sort_by_key false) +endif() + gen_test(fields false) gen_test(grid_mesh false) if(${DEBUG}) diff --git a/tests/framework/sort_by_key.cpp b/tests/framework/sort_by_key.cpp new file mode 100644 index 000000000..94dc51d0e --- /dev/null +++ b/tests/framework/sort_by_key.cpp @@ -0,0 +1,110 @@ +/** + * @brief X-3 (team_policy) — sort_by_key permutation test. + * + * Exercises every backend overload of `ntt::sort_helpers::sort_by_key_dispatch` + * that is compiled in for the current Kokkos device. For each backend: + * 1. Allocate keys = { 5, 2, 5, 1, 3, 5, 2 }, perm = (uninitialised). + * 2. Call sort_by_key_dispatch. + * 3. Verify that keys[perm[i]] is sorted in non-decreasing order. + * + * Stability is verified for the BinSort and StdSort backends (the others + * promise stability per their documentation but we don't bake that into + * the test). + * + * Built only when `team_policy=ON` at CMake time. + */ +#include "enums.h" +#include "global.h" + +#include "arch/kokkos_aliases.h" +#include "utils/error.h" +#include "utils/sort_dispatch.h" +#include "utils/sorting.h" + +#include + +#include +#include + +namespace { + + using namespace ntt; + + template + void test_one_backend(const char* label, Backend tag) { + const std::vector keys_host_init { 5u, 2u, 5u, 1u, 3u, 5u, 2u }; + const npart_t n = keys_host_init.size(); + const ncells_t n_max = 6u; // bin range [0, n_max) + + array_t keys { "keys", n }; + auto keys_h = Kokkos::create_mirror_view(keys); + for (npart_t i = 0u; i < n; ++i) { + keys_h(i) = keys_host_init[i]; + } + Kokkos::deep_copy(keys, keys_h); + + prtl_perm_t perm { "perm", n }; + + sort_helpers::sort_by_key_dispatch(keys, perm, n_max, tag); + + auto perm_h = Kokkos::create_mirror_view(perm); + Kokkos::deep_copy(perm_h, perm); + + // Validate: keys[perm[0]] <= keys[perm[1]] <= ... + for (npart_t i = 1u; i < n; ++i) { + const auto a = keys_host_init[perm_h(i - 1u)]; + const auto b = keys_host_init[perm_h(i)]; + raise::ErrorIf( + a > b, + std::string("sort_by_key_dispatch produced non-sorted permutation " + "for backend ") + + label, + HERE); + } + + // Validate: perm is a permutation of [0, n). + std::vector seen(n, 0); + for (npart_t i = 0u; i < n; ++i) { + const auto idx = perm_h(i); + raise::ErrorIf(idx >= n, + std::string("permutation index out of range for " + "backend ") + + label, + HERE); + seen[idx] += 1; + } + for (npart_t i = 0u; i < n; ++i) { + raise::ErrorIf(seen[i] != 1, + std::string("permutation not a bijection for backend ") + + label, + HERE); + } + + std::cout << "[OK] sort_by_key_dispatch<" << label << ">: " + << "keys[perm] sorted, perm is a bijection." << std::endl; + } + +} // namespace + +auto main(int argc, char* argv[]) -> int { + ntt::GlobalInitialize(argc, argv); + try { + // Always-available backends. + test_one_backend("BinSort", ::sort::backend::BinSort {}); +#if !defined(DEVICE_ENABLED) + test_one_backend("StdSort", ::sort::backend::StdSort {}); +#endif +#if defined(SYCL_ENABLED) && defined(ONEDPL_ENABLED) + test_one_backend("OneDPL", ::sort::backend::OneDPL {}); +#endif +#if defined(CUDA_ENABLED) && defined(THRUST_ENABLED) + test_one_backend("Thrust", ::sort::backend::Thrust {}); +#endif + } catch (std::exception& e) { + std::cerr << e.what() << std::endl; + ntt::GlobalFinalize(); + return 1; + } + ntt::GlobalFinalize(); + return 0; +} diff --git a/tests/global/tiling.cpp b/tests/global/tiling.cpp index d0f0a412c..2f926ce28 100644 --- a/tests/global/tiling.cpp +++ b/tests/global/tiling.cpp @@ -1,6 +1,5 @@ #include "arch/kokkos_aliases.h" #include "utils/error.h" -#include "utils/formatting.h" #include "utils/numeric.h" #include "utils/sorting.h" @@ -122,37 +121,27 @@ void test_tiling(const array_t& i1, const auto ntiles = nt1 * nt2 * nt3; - auto position_to_tile_kernel = sort::PositionToTileIndex { - i1, i2, i3, tag, tile_indices, ncells, ts - }; - Kokkos::parallel_for("Tiling", npart, position_to_tile_kernel); - const auto num_ppt = position_to_tile_kernel.num_ppt; + array_t num_ppt { "num_ppt", ntiles }; + Kokkos::parallel_for( + "Tiling", + npart, + sort::PositionToTileIndex { i1, i2, i3, tag, tile_indices, ncells, ts, num_ppt }); Kokkos::parallel_for( "Checking", npart, Lambda(prtlidx_t p) { CheckValue(p, i1, i2, i3, tag, tile_indices, nt1, nt2, nt3, ntiles, ts); }); - raise::ErrorIf( - num_ppt.extent(0) != ntiles, - fmt::format("num_ppt size does not match number of tiles %u != %u", - num_ppt.extent(0), - ntiles), - HERE); npart_t tot_alive = 0u; Kokkos::parallel_reduce( "CountAliveInTiles", ntiles, - Lambda(cellidx_t t, npart_t & count) { count += num_ppt(t); }, + Lambda(prtlidx_t t, npart_t & count) { count += num_ppt(t); }, tot_alive); - raise::ErrorIf( - tot_alive != npart - ndead, - fmt::format("Error in counting particles per tile: %u != %u - %u", - tot_alive, - npart, - ndead), - HERE); + raise::ErrorIf(tot_alive != npart - ndead, + "Error in counting particles per tile", + HERE); } } diff --git a/tests/kernels/CMakeLists.txt b/tests/kernels/CMakeLists.txt index f1438e108..0fe23f578 100644 --- a/tests/kernels/CMakeLists.txt +++ b/tests/kernels/CMakeLists.txt @@ -26,6 +26,9 @@ endfunction() gen_test(faraday_mink) gen_test(ampere_mink) gen_test(deposit) +if(${team_policy}) + gen_test(deposit_tiled) +endif() gen_test(digital_filter) gen_test(particle_moments) gen_test(fields_to_phys) diff --git a/tests/kernels/deposit_tiled.cpp b/tests/kernels/deposit_tiled.cpp new file mode 100644 index 000000000..87f4f3f4d --- /dev/null +++ b/tests/kernels/deposit_tiled.cpp @@ -0,0 +1,262 @@ +/** + * @file tests/kernels/deposit_tiled.cpp + * @brief X-1 numerical-equivalence test for the tiled deposit kernel. + * + * Runs the flat (`DepositCurrents_kernel`) and tiled + * (`DepositCurrents_kernel_tiled`) kernels on identical particle SoA inputs + * for shape orders O = 1..11 and asserts that the resulting J array is + * identical cell-by-cell within a small floating-point tolerance. + * + * Built only when `team_policy=ON` (`-D TEAM_POLICY` defined). The test + * matches the per-particle setup used in `deposit.cpp` so that any + * regression in the shared `kernel::deposit::deposit_one_particle` body + * is caught by both tests. + */ + +#include "enums.h" +#include "global.h" + +#include "arch/kokkos_aliases.h" +#include "utils/comparators.h" + +#include "metrics/minkowski.h" + +#include "kernels/currents_deposit.hpp" + +#include +#include + +#include +#include +#include +#include + +namespace { + + using namespace ntt; + + void errorIf(bool condition, const std::string& msg) { + if (condition) { + throw std::runtime_error(msg); + } + } + + template + void put_value(const array_t& arr, T value, int i) { + auto h = Kokkos::create_mirror_view(arr); + h(i) = value; + Kokkos::deep_copy(arr, h); + } + + // Builds tile_offsets for a single-particle test. Particle 0 is alive + // and lives in tile (tx1, tx2); slots 1..n_slots-1 carry the dead + // sentinel and are never referenced by tile_offsets — so the tiled + // kernel never iterates over them. + array_t build_tile_offsets_single_particle(ncells_t ntx1, + ncells_t ntx2, + ncells_t tx1, + ncells_t tx2) { + const ncells_t total_tiles = ntx1 * ntx2; + const ncells_t hot_tile = tx1 * ntx2 + tx2; + array_t offsets("tile_offsets", total_tiles + 1u); + auto h = Kokkos::create_mirror_view(offsets); + for (ncells_t t = 0; t <= total_tiles; ++t) { + h(t) = (t <= hot_tile) ? static_cast(0) + : static_cast(1); + } + Kokkos::deep_copy(offsets, h); + return offsets; + } + + template + void run_one_case() { + using metric_t = metric::Minkowski; + constexpr unsigned short nx1 = 50u, nx2 = 50u; + metric_t metric { { nx1, nx2 }, + { { 0.0, 55.0 }, { 0.0, 55.0 } }, + {} }; + + // Particle setup (mirrors deposit.cpp). + const int i0 = 25, j0 = 21, i0f = 24, j0f = 20; + const real_t uz = 2.5; + const prtldx_t dxi = static_cast(0.65); + const prtldx_t dxf = static_cast(0.99); + const prtldx_t dyi = static_cast(0.65); + const prtldx_t dyf = static_cast(0.80); + + array_t i1 { "i1", 10 }; + array_t i2 { "i2", 10 }; + array_t i3 { "i3", 10 }; + array_t i1_prev { "i1_prev", 10 }; + array_t i2_prev { "i2_prev", 10 }; + array_t i3_prev { "i3_prev", 10 }; + array_t dx1 { "dx1", 10 }; + array_t dx2 { "dx2", 10 }; + array_t dx3 { "dx3", 10 }; + array_t dx1_prev { "dx1_prev", 10 }; + array_t dx2_prev { "dx2_prev", 10 }; + array_t dx3_prev { "dx3_prev", 10 }; + array_t ux1 { "ux1", 10 }; + array_t ux2 { "ux2", 10 }; + array_t ux3 { "ux3", 10 }; + array_t phi { "phi", 10 }; + array_t weight { "weight", 10 }; + array_t tag { "tag", 10 }; + const real_t charge = 1.0, dt = 1.0; + + put_value(i1, i0f, 0); + put_value(i2, j0f, 0); + put_value(i1_prev, i0, 0); + put_value(i2_prev, j0, 0); + put_value(dx1, dxf, 0); + put_value(dx2, dyf, 0); + put_value(dx1_prev, dxi, 0); + put_value(dx2_prev, dyi, 0); + put_value(ux1, ZERO, 0); + put_value(ux2, ZERO, 0); + put_value(ux3, uz, 0); + put_value(weight, 1.0, 0); + put_value(tag, ParticleTag::alive, 0); + + // Run the flat kernel. + ndfield_t J_flat { "J_flat", + nx1 + 2u * N_GHOSTS, + nx2 + 2u * N_GHOSTS }; + { + auto J_scat = Kokkos::Experimental::create_scatter_view(J_flat); + Kokkos::parallel_for( + "FlatDeposit", + 10, + kernel::DepositCurrents_kernel( + J_scat, + i1, i2, i3, + i1_prev, i2_prev, i3_prev, + dx1, dx2, dx3, + dx1_prev, dx2_prev, dx3_prev, + ux1, ux2, ux3, + phi, weight, tag, + metric, charge, dt)); + Kokkos::Experimental::contribute(J_flat, J_scat); + Kokkos::fence("flat deposit done"); + } + + // Run the tiled kernel. Build a TileLayout with one alive particle + // landing in its expected tile (sort key = min(i, i_prev) / T_TILE). + ndfield_t J_tiled { "J_tiled", + nx1 + 2u * N_GHOSTS, + nx2 + 2u * N_GHOSTS }; + { + const auto sort_i1 = static_cast( + (i0 < i0f) ? i0 : i0f); // min(i, i_prev) before clamp + const auto sort_i2 = static_cast((j0 < j0f) ? j0 : j0f); + const auto ntx1 = static_cast( + std::ceil(static_cast(nx1) / static_cast(T_TILE))); + const auto ntx2 = static_cast( + std::ceil(static_cast(nx2) / static_cast(T_TILE))); + const auto tx1 = static_cast(sort_i1) / T_TILE; + const auto tx2 = static_cast(sort_i2) / T_TILE; + + TileLayout layout; + layout.ntiles_per_axis[0] = ntx1; + layout.ntiles_per_axis[1] = ntx2; + layout.ntiles_per_axis[2] = 1u; + layout.ntiles_total = ntx1 * ntx2; + layout.tile_size = T_TILE; + layout.tile_offsets = build_tile_offsets_single_particle(ntx1, + ntx2, + tx1, + tx2); + + using kernel_t = + kernel::DepositCurrents_kernel_tiled; + kernel_t kern { J_tiled, + i1, i2, i3, + i1_prev, i2_prev, i3_prev, + dx1, dx2, dx3, + dx1_prev, dx2_prev, dx3_prev, + ux1, ux2, ux3, + phi, weight, tag, + metric, charge, dt, layout }; + + Kokkos::TeamPolicy<> policy(static_cast(layout.ntiles_total), + Kokkos::AUTO); + policy.set_scratch_size(0, + Kokkos::PerTeam(kernel_t::scratch_bytes())); + Kokkos::parallel_for("TiledDeposit", policy, kern); + Kokkos::fence("tiled deposit done"); + } + + // Compare J_flat vs J_tiled cell-by-cell. + auto h_flat = Kokkos::create_mirror_view(J_flat); + auto h_tiled = Kokkos::create_mirror_view(J_tiled); + Kokkos::deep_copy(h_flat, J_flat); + Kokkos::deep_copy(h_tiled, J_tiled); + + const real_t eps = static_cast(1.0e-5); + real_t max_diff = ZERO; + int fail_count = 0; + for (ncells_t i = 0; i < h_flat.extent(0); ++i) { + for (ncells_t j = 0; j < h_flat.extent(1); ++j) { + for (int c = 0; c < 3; ++c) { + const real_t a = h_flat(i, j, c); + const real_t b = h_tiled(i, j, c); + const real_t diff = math::fabs(a - b); + const real_t mag = math::max(math::fabs(a), math::fabs(b)); + if (diff > max_diff) { + max_diff = diff; + } + if (diff > eps * math::max(mag, static_cast(1.0))) { + if (fail_count < 5) { + std::cerr << " J(" << i << "," << j << ",c=" << c + << ") flat=" << a << " tiled=" << b + << " diff=" << diff << '\n'; + } + ++fail_count; + } + } + } + } + if (fail_count > 0) { + std::cerr << "X-1 deposit_tiled equivalence FAILED for O=" << O + << " T_TILE=" << T_TILE + << " : " << fail_count << " mismatches; max_diff=" << max_diff + << '\n'; + throw std::logic_error("DepositCurrents_kernel_tiled mismatch"); + } + std::cerr << "X-1 deposit_tiled OK O=" << O << " T_TILE=" << T_TILE + << " max_diff=" << max_diff << '\n'; + } + + template + void run_all_orders() { + run_one_case<0u, T_TILE>(); + run_one_case<1u, T_TILE>(); + run_one_case<2u, T_TILE>(); + run_one_case<3u, T_TILE>(); + run_one_case<4u, T_TILE>(); + run_one_case<5u, T_TILE>(); + run_one_case<6u, T_TILE>(); + run_one_case<7u, T_TILE>(); + run_one_case<8u, T_TILE>(); + run_one_case<9u, T_TILE>(); + run_one_case<10u, T_TILE>(); + run_one_case<11u, T_TILE>(); + } + +} // namespace + +auto main(int argc, char* argv[]) -> int { + Kokkos::initialize(argc, argv); + try { + // Run with each tile-size choice from the validated CMake list. + run_all_orders<4u>(); + run_all_orders<8u>(); + run_all_orders<12u>(); + } catch (std::exception& e) { + std::cerr << e.what() << '\n'; + Kokkos::finalize(); + return 1; + } + Kokkos::finalize(); + return 0; +} From 04d403c606147e0d651cac28f930d5592f808c84 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ludwig=20B=C3=B6ss?= Date: Tue, 19 May 2026 19:25:38 +0200 Subject: [PATCH 003/125] removed redundant inner loop over species --- src/framework/domain/metadomain_sort.cpp | 4 +--- 1 file changed, 1 insertion(+), 3 deletions(-) diff --git a/src/framework/domain/metadomain_sort.cpp b/src/framework/domain/metadomain_sort.cpp index 791bf31a8..a2b08f8be 100644 --- a/src/framework/domain/metadomain_sort.cpp +++ b/src/framework/domain/metadomain_sort.cpp @@ -21,9 +21,7 @@ namespace ntt { const auto clearing_interval = species.clearing_interval(); if ((clearing_interval > 0u) and (step % clearing_interval == 0u) and (step > 0u)) { - for (auto& species : domain.species) { - species.RemoveDead(); - } + species.RemoveDead(); } const auto spatial_sorting_interval = species.spatial_sorting_interval(); if ((spatial_sorting_interval > 0u) and From 5830c0740e43851f5d26afb8e35438160be8a007 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ludwig=20B=C3=B6ss?= Date: Tue, 19 May 2026 19:26:11 +0200 Subject: [PATCH 004/125] frontier-specific memory pool allocation --- src/global/global.cpp | 43 +++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 43 insertions(+) diff --git a/src/global/global.cpp b/src/global/global.cpp index ec22fd2f3..c3aa1bcdc 100644 --- a/src/global/global.cpp +++ b/src/global/global.cpp @@ -6,8 +6,51 @@ #include #endif // MPI_ENABLED +#if defined(HIP_ENABLED) + #include + + #include +#endif // HIP_ENABLED + +namespace { +#if defined(HIP_ENABLED) + // Turn the ROCm stream-ordered allocator into a caching arena. + // + // This Kokkos build uses hipMallocAsync/hipFreeAsync (Kokkos option + // IMPL_HIP_MALLOC_ASYNC). The default memory pool has a release + // threshold of 0, so every freed block is handed back to the driver + // at the next stream sync. With ~50 GB of particle SoA permanently + // pinned and only ~14 GB free, the per-step churn of dozens of + // large, differently-sized sort/comm scratch buffers fragments that + // free space: allocation cost grows monotonically (ParticleSort + // slowdown) until no contiguous mid-size block remains and BinSort's + // `sorted_values` allocation fails (OOM). Raising the release + // threshold to "unlimited" makes the pool retain and recycle freed + // blocks instead, which stabilizes the working set and removes both + // the slowdown and the OOM. + void ConfigureHipMemPool() { + int device = 0; + if (hipGetDevice(&device) != hipSuccess) { + return; + } + hipMemPool_t pool = nullptr; + if (hipDeviceGetDefaultMemPool(&pool, device) != hipSuccess or + pool == nullptr) { + return; + } + uint64_t threshold = UINT64_MAX; + (void)hipMemPoolSetAttribute(pool, + hipMemPoolAttrReleaseThreshold, + &threshold); + } +#endif // HIP_ENABLED +} // namespace + void ntt::GlobalInitialize(int argc, char* argv[]) { Kokkos::initialize(argc, argv); +#if defined(HIP_ENABLED) + ConfigureHipMemPool(); +#endif // HIP_ENABLED #if defined(MPI_ENABLED) MPI_Init(&argc, &argv); #endif // MPI_ENABLED From cf14e9f11891afc732bfb9d289b2c16c53af770c Mon Sep 17 00:00:00 2001 From: LudwigBoess Date: Tue, 26 May 2026 17:06:27 +0000 Subject: [PATCH 005/125] support more tile sizes --- CMakeLists.txt | 2 +- tests/kernels/deposit_tiled.cpp | 4 ++++ 2 files changed, 5 insertions(+), 1 deletion(-) diff --git a/CMakeLists.txt b/CMakeLists.txt index 248a34e7b..06494a7db 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -65,7 +65,7 @@ set(team_policy_tile_size ${default_team_policy_tile_size} CACHE STRING "team_policy tile edge length in cells") set(team_policy_tile_sizes - "4;6;8;10;12" + "4;6;8;10;12;14;16" CACHE STRING "team_policy tile-size choices") # -------------------------- Compilation settings -------------------------- # diff --git a/tests/kernels/deposit_tiled.cpp b/tests/kernels/deposit_tiled.cpp index 87f4f3f4d..504cc3818 100644 --- a/tests/kernels/deposit_tiled.cpp +++ b/tests/kernels/deposit_tiled.cpp @@ -250,8 +250,12 @@ auto main(int argc, char* argv[]) -> int { try { // Run with each tile-size choice from the validated CMake list. run_all_orders<4u>(); + run_all_orders<6u>(); run_all_orders<8u>(); + run_all_orders<10u>(); run_all_orders<12u>(); + run_all_orders<14u>(); + run_all_orders<16u>(); } catch (std::exception& e) { std::cerr << e.what() << '\n'; Kokkos::finalize(); From 787aa045750fa2c582b54d02c5e8d3f4265de1e0 Mon Sep 17 00:00:00 2001 From: LudwigBoess Date: Tue, 26 May 2026 17:12:06 +0000 Subject: [PATCH 006/125] removed persistent sort scratch to reduce memory overhead --- src/framework/containers/particles.h | 16 --- src/framework/containers/particles_sort.cpp | 111 +++++++------------- 2 files changed, 38 insertions(+), 89 deletions(-) diff --git a/src/framework/containers/particles.h b/src/framework/containers/particles.h index 0a15cb5d9..4f0770729 100644 --- a/src/framework/containers/particles.h +++ b/src/framework/containers/particles.h @@ -99,22 +99,6 @@ namespace ntt { // vendor libraries detected by CMake. TileLayout m_tile_layout {}; -#if defined(TEAM_POLICY) && \ - ((defined(SYCL_ENABLED) && defined(ONEDPL_ENABLED)) || \ - (defined(CUDA_ENABLED) && defined(THRUST_ENABLED)) || \ - (defined(HIP_ENABLED) && defined(ROCTHRUST_ENABLED))) - // Persistent byte scratch reused by every SoA-member gather in - // `apply_permutation_to_soa`, across all members and all timesteps. - // Without this each member would allocate (and free) its own - // transient buffer every sort; recycling one persistent buffer - // removes that allocation churn entirely — the structural fix for - // the ROCm sort slowdown / fragmentation. Grown monotonically to - // the largest required size, never shrunk. Kokkos device - // allocations are over-aligned (>= 8 B), so reinterpreting the - // bytes as any SoA element type (<= 8 B PODs) is well-defined. - array_t m_perm_scratch {}; -#endif - public: // for empty allocation Particles() {} diff --git a/src/framework/containers/particles_sort.cpp b/src/framework/containers/particles_sort.cpp index c04813e12..0317e0b5f 100644 --- a/src/framework/containers/particles_sort.cpp +++ b/src/framework/containers/particles_sort.cpp @@ -454,28 +454,20 @@ namespace ntt { #if defined(TEAM_POLICY_USE_VENDOR_SORT) namespace permute_helpers { - // Permute a 1D SoA member array `arr` in place by `perm`, gathering - // through `scratch` — a persistent byte buffer reused by every - // member and every timestep (no per-call allocation). An unmanaged - // typed view aliases the scratch bytes; the caller guarantees - // `scratch` is large enough and that Kokkos' device over-alignment - // covers the element type. + // Permute a 1D SoA member array `arr` in place by `perm`, using a + // single transient buffer of size `n`. Buffer is freed at scope + // exit; the explicit fence right before that drains queued GPU + // work referencing it. template - inline void permute_1d_inplace(V& arr, - const prtl_perm_t& perm, - npart_t n, - const array_t& scratch) { + inline void permute_1d_inplace(V& arr, + const prtl_perm_t& perm, + npart_t n) { if (n == 0u) { return; } - using value_t = typename V::non_const_value_type; - using buf_t = Kokkos::View>; - buf_t buf(reinterpret_cast(scratch.data()), n); - auto perm_v = perm; - auto arr_v = arr; + V buf(std::string(arr.label()) + "_perm_buf", n); + auto perm_v = perm; + auto arr_v = arr; Kokkos::parallel_for( "Permute1D", n, @@ -486,22 +478,16 @@ namespace ntt { // 2D analogue for `pld_r` / `pld_i`. template - inline void permute_2d_inplace(V& arr, - const prtl_perm_t& perm, - npart_t n, - npart_t ncols, - const array_t& scratch) { + inline void permute_2d_inplace(V& arr, + const prtl_perm_t& perm, + npart_t n, + npart_t ncols) { if (n == 0u or ncols == 0u) { return; } - using value_t = typename V::non_const_value_type; - using buf_t = Kokkos::View>; - buf_t buf(reinterpret_cast(scratch.data()), n, ncols); - auto perm_v = perm; - auto arr_v = arr; + V buf(std::string(arr.label()) + "_perm_buf", n, ncols); + auto perm_v = perm; + auto arr_v = arr; Kokkos::parallel_for( "Permute2D", CreateParticleRangePolicy({ 0u, 0u }, { n, ncols }), @@ -522,61 +508,40 @@ namespace ntt { return; } - // Size the persistent scratch once to the largest gather any member - // needs this call: 1D members need n * sizeof(real_t) bytes (the - // widest element); the 2D payloads need n * ncols * elem bytes. - // Grown monotonically, never shrunk — so after warmup this incurs - // no allocation at all. - std::size_t need = static_cast(n) * sizeof(real_t); - if (npld_r() > 0) { - need = std::max(need, - static_cast(n) * - static_cast(npld_r()) * sizeof(real_t)); - } - if (npld_i() > 0) { - need = std::max(need, - static_cast(n) * - static_cast(npld_i()) * sizeof(npart_t)); - } - if (m_perm_scratch.extent(0) < need) { - m_perm_scratch = array_t { "perm_scratch", need }; - } - const auto& scratch = m_perm_scratch; - using permute_helpers::permute_1d_inplace; using permute_helpers::permute_2d_inplace; if constexpr (D == Dim::_1D or D == Dim::_2D or D == Dim::_3D) { - permute_1d_inplace(i1, perm, n, scratch); - permute_1d_inplace(dx1, perm, n, scratch); - permute_1d_inplace(i1_prev, perm, n, scratch); - permute_1d_inplace(dx1_prev, perm, n, scratch); + permute_1d_inplace(i1, perm, n); + permute_1d_inplace(dx1, perm, n); + permute_1d_inplace(i1_prev, perm, n); + permute_1d_inplace(dx1_prev, perm, n); } if constexpr (D == Dim::_2D or D == Dim::_3D) { - permute_1d_inplace(i2, perm, n, scratch); - permute_1d_inplace(dx2, perm, n, scratch); - permute_1d_inplace(i2_prev, perm, n, scratch); - permute_1d_inplace(dx2_prev, perm, n, scratch); + permute_1d_inplace(i2, perm, n); + permute_1d_inplace(dx2, perm, n); + permute_1d_inplace(i2_prev, perm, n); + permute_1d_inplace(dx2_prev, perm, n); } if constexpr (D == Dim::_3D) { - permute_1d_inplace(i3, perm, n, scratch); - permute_1d_inplace(dx3, perm, n, scratch); - permute_1d_inplace(i3_prev, perm, n, scratch); - permute_1d_inplace(dx3_prev, perm, n, scratch); - } - permute_1d_inplace(ux1, perm, n, scratch); - permute_1d_inplace(ux2, perm, n, scratch); - permute_1d_inplace(ux3, perm, n, scratch); - permute_1d_inplace(weight, perm, n, scratch); - permute_1d_inplace(tag, perm, n, scratch); + permute_1d_inplace(i3, perm, n); + permute_1d_inplace(dx3, perm, n); + permute_1d_inplace(i3_prev, perm, n); + permute_1d_inplace(dx3_prev, perm, n); + } + permute_1d_inplace(ux1, perm, n); + permute_1d_inplace(ux2, perm, n); + permute_1d_inplace(ux3, perm, n); + permute_1d_inplace(weight, perm, n); + permute_1d_inplace(tag, perm, n); if constexpr (D == Dim::_2D and C != Coord::Cartesian) { - permute_1d_inplace(phi, perm, n, scratch); + permute_1d_inplace(phi, perm, n); } if (npld_r() > 0) { - permute_2d_inplace(pld_r, perm, n, static_cast(npld_r()), scratch); + permute_2d_inplace(pld_r, perm, n, static_cast(npld_r())); } if (npld_i() > 0) { - permute_2d_inplace(pld_i, perm, n, static_cast(npld_i()), scratch); + permute_2d_inplace(pld_i, perm, n, static_cast(npld_i())); } } #endif // TEAM_POLICY_USE_VENDOR_SORT From 827cf264c2903b3179252c88f19c0a7bb090357d Mon Sep 17 00:00:00 2001 From: LudwigBoess Date: Tue, 2 Jun 2026 18:20:33 +0000 Subject: [PATCH 007/125] added team policy reporting --- src/engines/reporter.cpp | 3 +++ src/global/utils/reporter.cpp | 6 ++++++ 2 files changed, 9 insertions(+) diff --git a/src/engines/reporter.cpp b/src/engines/reporter.cpp index f56874d23..3caeda5de 100644 --- a/src/engines/reporter.cpp +++ b/src/engines/reporter.cpp @@ -32,6 +32,9 @@ namespace ntt { "%s", params.template get("simulation.name").c_str()); reporter::AddParam(report, 4, "Engine", "%s", SimEngine(S).to_string()); +#if defined(TEAM_POLICY) + reporter::AddParam(report, 4, "Tile size", "%d", TEAM_POLICY_TILE_SIZE); +#endif reporter::AddParam(report, 4, "Metric", "%s", M.to_string()); #if SHAPE_ORDER == 0 reporter::AddParam(report, 4, "Deposit", "%s", "zigzag"); diff --git a/src/global/utils/reporter.cpp b/src/global/utils/reporter.cpp index 77117c4b9..a4b10eee6 100644 --- a/src/global/utils/reporter.cpp +++ b/src/global/utils/reporter.cpp @@ -250,6 +250,12 @@ namespace reporter { #else AddParam(report, 4, "GPU_AWARE_MPI", "%s", "OFF"); #endif + +#if defined(TEAM_POLICY) + AddParam(report, 4, "TEAM_POLICY", "%s", "ON"); +#else + AddParam(report, 4, "TEAM_POLICY", "%s", "OFF"); +#endif report += "\n"; return report; } From 4b81914aae5cbc15cea19217b284113f617750f3 Mon Sep 17 00:00:00 2001 From: LudwigBoess Date: Tue, 2 Jun 2026 19:08:47 +0000 Subject: [PATCH 008/125] explicitly bind GPU Transport Layer for GPU aware MPI on Frontier --- CMakeLists.txt | 56 ++++++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 56 insertions(+) diff --git a/CMakeLists.txt b/CMakeLists.txt index 06494a7db..0137671a0 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -211,6 +211,62 @@ if(${mpi}) if(${DEVICE_ENABLED}) if(${gpu_aware_mpi}) add_compile_options("-D GPU_AWARE_MPI") + + # On Cray systems (e.g. Frontier) GPU-aware Cray MPICH can only + # handle device pointers if the GPU Transport Layer (GTL) library + # is linked. The Cray compiler wrappers (cc/CC) inject this + # automatically, but we build with hipcc/nvcc directly, so + # find_package(MPI) only finds base libmpi and the GTL is left + # out -> MPI_Sendrecv on a device pointer fails with + # "OFI ... Bad address". Add it explicitly here. + # + # Cray PE exports PE_MPICH_GTL_DIR_ / PE_MPICH_GTL_LIBS_ + # (e.g. amd_gfx90a -> -lmpi_gtl_hsa). Their absence means this is + # not a Cray MPICH build, in which case nothing extra is needed. + if("${Kokkos_DEVICES}" MATCHES "HIP") + set(_gtl_accels amd_gfx942 amd_gfx940 amd_gfx90a amd_gfx908 amd_gfx906) + elseif("${Kokkos_DEVICES}" MATCHES "CUDA") + set(_gtl_accels nvidia90 nvidia80 nvidia70) + elseif("${Kokkos_DEVICES}" MATCHES "SYCL") + set(_gtl_accels ponteVecchio) + else() + set(_gtl_accels "") + endif() + + set(_gtl_dir "") + set(_gtl_libflag "") + foreach(_accel ${_gtl_accels}) + if((NOT _gtl_dir) AND (DEFINED ENV{PE_MPICH_GTL_DIR_${_accel}})) + # strip the leading "-L" from the Cray-provided value + string(REGEX REPLACE "^-L" "" + _gtl_dir "$ENV{PE_MPICH_GTL_DIR_${_accel}}") + string(REGEX REPLACE "^-l" "" + _gtl_libflag "$ENV{PE_MPICH_GTL_LIBS_${_accel}}") + endif() + endforeach() + + if(_gtl_dir AND _gtl_libflag) + find_library(MPI_GTL_LIBRARY + NAMES ${_gtl_libflag} + HINTS "${_gtl_dir}" + NO_DEFAULT_PATH) + if(MPI_GTL_LIBRARY) + message(STATUS + "GPU-aware MPI: linking Cray GTL library ${MPI_GTL_LIBRARY}") + set(DEPENDENCIES ${DEPENDENCIES} ${MPI_GTL_LIBRARY}) + else() + message(FATAL_ERROR + "${Red}gpu_aware_mpi=ON: Cray MPICH detected but the GTL " + "library 'lib${_gtl_libflag}' was not found in '${_gtl_dir}'. " + "GPU-aware MPI will crash at runtime without it. Make sure the " + "craype-accel module is loaded, or build with gpu_aware_mpi=OFF." + "${ColorReset}") + endif() + else() + message(STATUS + "GPU-aware MPI: no Cray GTL environment found; assuming the MPI " + "implementation is GPU-aware without an extra transport library.") + endif() endif() else() set(gpu_aware_mpi From a603ecb0b11662282cfd613bf8bff085c5aead70 Mon Sep 17 00:00:00 2001 From: haykh Date: Wed, 3 Jun 2026 10:38:29 -0400 Subject: [PATCH 009/125] minor refactor --- src/engines/srpic/currents.h | 92 +- src/global/arch/kokkos_aliases.h | 73 ++ src/kernels/currents_deposit.hpp | 1760 ++++++++++++++---------------- 3 files changed, 934 insertions(+), 991 deletions(-) diff --git a/src/engines/srpic/currents.h b/src/engines/srpic/currents.h index faf0bb3ad..63746d101 100644 --- a/src/engines/srpic/currents.h +++ b/src/engines/srpic/currents.h @@ -39,30 +39,12 @@ namespace ntt { species.rangeActiveParticles(), kernel::DepositCurrents_kernel( scatter_cur, - species.i1, - species.i2, - species.i3, - species.i1_prev, - species.i2_prev, - species.i3_prev, - species.dx1, - species.dx2, - species.dx3, - species.dx1_prev, - species.dx2_prev, - species.dx3_prev, - species.ux1, - species.ux2, - species.ux3, - species.phi, - species.weight, - species.tag, + species, local_metric, (real_t)(species.charge()), dt)); } -#if defined(TEAM_POLICY) /** * @brief Tiled deposit launcher (TeamPolicy + per-team scratch). * @@ -78,11 +60,10 @@ namespace ntt { * passed-in `scatter_cur` so the caller still composes correctly. */ template - void CallDepositKernelTiled( - const Particles& species, - const M& local_metric, - const ndfield_t& cur, - real_t dt) { + void CallDepositKernelTiled(const Particles& species, + const M& local_metric, + const ndfield_t& cur, + real_t dt) { static_assert(O <= 11u, "Shape order must be <= 11"); constexpr unsigned short T = static_cast( TEAM_POLICY_TILE_SIZE); @@ -96,40 +77,18 @@ namespace ntt { "with ntiles_total", HERE); - using kernel_t = kernel::DepositCurrents_kernel_tiled; - kernel_t kern { cur, - species.i1, - species.i2, - species.i3, - species.i1_prev, - species.i2_prev, - species.i3_prev, - species.dx1, - species.dx2, - species.dx3, - species.dx1_prev, - species.dx2_prev, - species.dx3_prev, - species.ux1, - species.ux2, - species.ux3, - species.phi, - species.weight, - species.tag, - local_metric, - (real_t)(species.charge()), - dt, - layout }; + auto deposit_kernel = + kernel::DepositCurrentsTiled_kernel { + cur, species, local_metric, (real_t)(species.charge()), dt, layout + }; Kokkos::TeamPolicy<> policy(static_cast(layout.ntiles_total), Kokkos::AUTO); - policy.set_scratch_size(0, Kokkos::PerTeam(kernel_t::scratch_bytes())); - Kokkos::parallel_for("CurrentsDepositTiled", policy, kern); + policy.set_scratch_size( + 0, + Kokkos::PerTeam(decltype(deposit_kernel)::scratch_bytes())); + Kokkos::parallel_for("CurrentsDepositTiled", policy, deposit_kernel); } -#endif // TEAM_POLICY template void CurrentsDeposit(Domain& domain, @@ -164,17 +123,18 @@ namespace ntt { continue; } logger::Checkpoint( - fmt::format("Launching currents deposit (flat fallback, no sort yet) " - "for %d [%s] : %lu %f", - species.index(), - species.label().c_str(), - species.npart(), - (double)species.charge()), + fmt::format( + "Launching currents deposit (flat fallback, no sort yet) " + "for %d [%s] : %lu %f", + species.index(), + species.label().c_str(), + species.npart(), + (double)species.charge()), HERE); CallDepositKernel(species, - domain.mesh.metric, - scatter_cur, - dt); + domain.mesh.metric, + scatter_cur, + dt); } Kokkos::Experimental::contribute(domain.fields.cur, scatter_cur); } else { @@ -192,9 +152,9 @@ namespace ntt { HERE); CallDepositKernelTiled(species, - domain.mesh.metric, - domain.fields.cur, - dt); + domain.mesh.metric, + domain.fields.cur, + dt); } } #else diff --git a/src/global/arch/kokkos_aliases.h b/src/global/arch/kokkos_aliases.h index 43f21ce47..4fe88cdf7 100644 --- a/src/global/arch/kokkos_aliases.h +++ b/src/global/arch/kokkos_aliases.h @@ -226,6 +226,79 @@ namespace kokkos_aliases_hidden { template using range_h_t = typename kokkos_aliases_hidden::range_h_impl::type; +// Array aliases of arbitrary type and dimensions (up to 4) +namespace kokkos_aliases_hidden { + // c++ magic + template + struct scratch_nddata_impl { + using type = void; + }; + + template + struct scratch_nddata_impl<1, T> { + using type = Kokkos::View>; + }; + + template + struct scratch_nddata_impl<2, T> { + using type = Kokkos::View>; + }; + + template + struct scratch_nddata_impl<3, T> { + using type = Kokkos::View>; + }; + + template + struct scratch_nddata_impl<4, T> { + using type = Kokkos::View>; + }; +} // namespace kokkos_aliases_hidden + +template +using scratch_nddata_t = typename kokkos_aliases_hidden::scratch_nddata_impl::type; + +// Defining aliases for Scratch memory ndfield +namespace kokkos_aliases_hidden { + template + struct scratch_ndfield_impl { + using type = void; + }; + + template + struct scratch_ndfield_impl { + using type = Kokkos::View>; + }; + + template + struct scratch_ndfield_impl { + using type = Kokkos::View>; + }; + + template + struct scratch_ndfield_impl { + using type = Kokkos::View>; + }; +} // namespace kokkos_aliases_hidden + +template +using scratch_ndfield_t = + typename kokkos_aliases_hidden::scratch_ndfield_impl::type; + /** * @brief Function template for generating 1D Kokkos range policy for particles. * @tparam D Dimension diff --git a/src/kernels/currents_deposit.hpp b/src/kernels/currents_deposit.hpp index 5eb7ff2b6..79252621a 100644 --- a/src/kernels/currents_deposit.hpp +++ b/src/kernels/currents_deposit.hpp @@ -3,7 +3,7 @@ * @brief Covariant algorithms for the current deposition. * * Two kernels share the same per-particle body - * (`kernel::deposit::deposit_one_particle`): + * (`kernel::DepositOneParticle`): * - `kernel::DepositCurrents_kernel` flat (RangePolicy over particles, * writes into a `Kokkos::Experimental::ScatterView`). Always available. * - `kernel::DepositCurrents_kernel_tiled` team-policy @@ -13,7 +13,7 @@ * * @implements * - kernel::deposit::PrtlPack<> - * - kernel::deposit::deposit_one_particle<> + * - kernel::DepositOneParticle<> * - kernel::DepositCurrents_kernel<> * - kernel::DepositCurrents_kernel_tiled<> (TEAM_POLICY only) * @namespaces: @@ -32,6 +32,7 @@ #include "utils/error.h" #include "utils/numeric.h" +#include "framework/containers/particles.h" #include "kernels/particle_shapes.hpp" #include @@ -42,671 +43,645 @@ namespace kernel { using namespace ntt; - namespace deposit { - - /** - * @brief Per-particle reference pack consumed by both the flat and tiled - * deposit kernels. The same set of SoA references is captured by - * each kernel; bundling them here keeps the helper's argument - * list manageable and ensures every consumer reads the same - * view aliases. - */ - template - struct PrtlPack { - array_t i1, i2, i3; - array_t i1_prev, i2_prev, i3_prev; - array_t dx1, dx2, dx3; - array_t dx1_prev, dx2_prev, dx3_prev; - array_t ux1, ux2, ux3; - array_t phi; - array_t weight; - array_t tag; - }; + /** + * @brief Per-particle deposit body, shared between the flat and tiled + * kernels. + * + * The caller supplies a `deposit_at(idx..., comp, val)` callback that + * applies the contribution `val` to the J component `comp` at the + * **global** J cell index `idx...` (already includes the `N_GHOSTS` + * offset). The flat kernel's callback simply does + * `J_acc(idx..., comp) += val` on its scatter-view accessor; the tiled + * kernel's callback translates `idx...` into per-tile scratch + * coordinates and uses `Kokkos::atomic_add` on SLM. Either way, this + * function is identical numerically and contains the only deposit math + * in the codebase. + * + * Dead particles return early. The callback is invoked once per cell + * write, with the dimension-appropriate signature: + * - 1D: `deposit_at(int g_i1, int comp, real_t val)` + * - 2D: `deposit_at(int g_i1, int g_i2, int comp, real_t val)` + * - 3D: `deposit_at(int g_i1, int g_i2, int g_i3, int comp, real_t val)` + */ + template + Inline void DepositOneParticle(prtlidx_t p, + const ParticleArrays& prtls, + const M& metric, + real_t charge, + real_t inv_dt, + DepositFn deposit_at) { + static_assert(O <= 11u, "Shape function order O must be <= 11"); + constexpr auto D = M::Dim; - /** - * @brief Per-particle deposit body, shared between the flat and tiled - * kernels. - * - * The caller supplies a `deposit_at(idx..., comp, val)` callback that - * applies the contribution `val` to the J component `comp` at the - * **global** J cell index `idx...` (already includes the `N_GHOSTS` - * offset). The flat kernel's callback simply does - * `J_acc(idx..., comp) += val` on its scatter-view accessor; the tiled - * kernel's callback translates `idx...` into per-tile scratch - * coordinates and uses `Kokkos::atomic_add` on SLM. Either way, this - * function is identical numerically and contains the only deposit math - * in the codebase. - * - * Dead particles return early. The callback is invoked once per cell - * write, with the dimension-appropriate signature: - * - 1D: `deposit_at(int g_i1, int comp, real_t val)` - * - 2D: `deposit_at(int g_i1, int g_i2, int comp, real_t val)` - * - 3D: `deposit_at(int g_i1, int g_i2, int g_i3, int comp, real_t val)` - */ - template - Inline void deposit_one_particle(prtlidx_t p, - const PrtlPack& prtls, - const M& metric, - real_t charge, - real_t inv_dt, - DepositFn deposit_at) { - static_assert(O <= 11u, "Shape function order O must be <= 11"); - constexpr auto D = M::Dim; - - const auto& i1 = prtls.i1; - const auto& i2 = prtls.i2; - const auto& i3 = prtls.i3; - const auto& i1_prev = prtls.i1_prev; - const auto& i2_prev = prtls.i2_prev; - const auto& i3_prev = prtls.i3_prev; - const auto& dx1 = prtls.dx1; - const auto& dx2 = prtls.dx2; - const auto& dx3 = prtls.dx3; - const auto& dx1_prev = prtls.dx1_prev; - const auto& dx2_prev = prtls.dx2_prev; - const auto& dx3_prev = prtls.dx3_prev; - const auto& ux1 = prtls.ux1; - const auto& ux2 = prtls.ux2; - const auto& ux3 = prtls.ux3; - const auto& phi = prtls.phi; - const auto& weight = prtls.weight; - const auto& tag = prtls.tag; - - if (tag(p) == ParticleTag::dead) { - return; - } + if (prtls.tag(p) == ParticleTag::dead) { + return; + } - // recover particle velocity to deposit in unsimulated direction - [[maybe_unused]] vec_t vp { ZERO }; - // `vp` only feeds the unsimulated-direction current in the 1D - // (jx2, jx3) and 2D (jx3) branches. In 3D every J component comes - // from the Esirkepov/zigzag charge motion and `vp` is never read, - // so the metric transform + 1/sqrt + NaN/Inf guard below is pure - // dead work there — skip it (also frees xp/inv_energy registers). - if constexpr (D != Dim::_3D) { - coord_t xp { ZERO }; - if constexpr (D == Dim::_1D) { - xp[0] = i_di_to_Xi(i1(p), dx1(p)); - } else if constexpr (D == Dim::_2D) { - if constexpr (M::PrtlDim == Dim::_3D) { - xp[0] = i_di_to_Xi(i1(p), dx1(p)); - xp[1] = i_di_to_Xi(i2(p), dx2(p)); - xp[2] = phi(p); - } else { - xp[0] = i_di_to_Xi(i1(p), dx1(p)); - xp[1] = i_di_to_Xi(i2(p), dx2(p)); - } - } else { - xp[0] = i_di_to_Xi(i1(p), dx1(p)); - xp[1] = i_di_to_Xi(i2(p), dx2(p)); - xp[2] = i_di_to_Xi(i3(p), dx3(p)); - } - auto inv_energy { ZERO }; - if constexpr (S == SimEngine::SRPIC) { - metric.template transform_xyz(xp, - { ux1(p), ux2(p), ux3(p) }, - vp); - inv_energy = ONE / math::sqrt(ONE + NORM_SQR(ux1(p), ux2(p), ux3(p))); + // recover particle velocity to deposit in unsimulated direction + [[maybe_unused]] + vec_t vp { ZERO }; + // `vp` only feeds the unsimulated-direction current in the 1D + // (jx2, jx3) and 2D (jx3) branches. In 3D every J component comes + // from the Esirkepov/zigzag charge motion and `vp` is never read, + // so the metric transform + 1/sqrt + NaN/Inf guard below is pure + // dead work there — skip it (also frees xp/inv_energy registers). + if constexpr (D != Dim::_3D) { + coord_t xp { ZERO }; + if constexpr (D == Dim::_1D) { + xp[0] = i_di_to_Xi(prtls.i1(p), prtls.dx1(p)); + } else if constexpr (D == Dim::_2D) { + if constexpr (M::PrtlDim == Dim::_3D) { + xp[0] = i_di_to_Xi(prtls.i1(p), prtls.dx1(p)); + xp[1] = i_di_to_Xi(prtls.i2(p), prtls.dx2(p)); + xp[2] = prtls.phi(p); } else { - coord_t xp_ { ZERO }; - xp_[0] = xp[0]; - real_t theta_Cd { xp[1] }; - const auto theta_Ph { metric.template convert<2, Crd::Cd, Crd::Ph>( - theta_Cd) }; - const auto small_angle { static_cast(constant::SMALL_ANGLE_GR) }; - const auto large_angle { static_cast( - constant::PI - constant::SMALL_ANGLE_GR) }; - if (theta_Ph < small_angle) { - theta_Cd = metric.template convert<2, Crd::Ph, Crd::Cd>(small_angle); - } else if (theta_Ph >= large_angle) { - theta_Cd = metric.template convert<2, Crd::Ph, Crd::Cd>(large_angle); - } - xp_[1] = theta_Cd; - metric.template transform(xp_, - { ux1(p), ux2(p), ux3(p) }, - vp); - inv_energy = metric.alpha(xp_) / - math::sqrt(ONE + ux1(p) * vp[0] + ux2(p) * vp[1] + - ux3(p) * vp[2]); + xp[0] = i_di_to_Xi(prtls.i1(p), prtls.dx1(p)); + xp[1] = i_di_to_Xi(prtls.i2(p), prtls.dx2(p)); } - if (Kokkos::isnan(vp[2]) || Kokkos::isinf(vp[2])) { - vp[2] = ZERO; + } else { + xp[0] = i_di_to_Xi(prtls.i1(p), prtls.dx1(p)); + xp[1] = i_di_to_Xi(prtls.i2(p), prtls.dx2(p)); + xp[2] = i_di_to_Xi(prtls.i3(p), prtls.dx3(p)); + } + auto inv_energy { ZERO }; + if constexpr (S == SimEngine::SRPIC) { + metric.template transform_xyz( + xp, + { prtls.ux1(p), prtls.ux2(p), prtls.ux3(p) }, + vp); + inv_energy = ONE / U2GAMMA(prtls.ux1(p), prtls.ux2(p), prtls.ux3(p)); + } else { + coord_t xp_ { ZERO }; + xp_[0] = xp[0]; + real_t theta_Cd { xp[1] }; + const auto theta_Ph { metric.template convert<2, Crd::Cd, Crd::Ph>( + theta_Cd) }; + const auto small_angle { static_cast(constant::SMALL_ANGLE_GR) }; + const auto large_angle { static_cast( + constant::PI - constant::SMALL_ANGLE_GR) }; + if (theta_Ph < small_angle) { + theta_Cd = metric.template convert<2, Crd::Ph, Crd::Cd>(small_angle); + } else if (theta_Ph >= large_angle) { + theta_Cd = metric.template convert<2, Crd::Ph, Crd::Cd>(large_angle); } - vp[0] *= inv_energy; - vp[1] *= inv_energy; - vp[2] *= inv_energy; + xp_[1] = theta_Cd; + metric.template transform( + xp_, + { prtls.ux1(p), prtls.ux2(p), prtls.ux3(p) }, + vp); + inv_energy = metric.alpha(xp_) / + math::sqrt(ONE + prtls.ux1(p) * vp[0] + + prtls.ux2(p) * vp[1] + prtls.ux3(p) * vp[2]); + } + if (Kokkos::isnan(vp[2]) || Kokkos::isinf(vp[2])) { + vp[2] = ZERO; } + vp[0] *= inv_energy; + vp[1] *= inv_energy; + vp[2] *= inv_energy; + } - const real_t coeff { weight(p) * charge }; + const real_t coeff { prtls.weight(p) * charge }; + + if constexpr (O == 0u) { + /* + Zig-zag deposit + */ + const auto dxp_r_1 { static_cast(prtls.i1(p) == prtls.i1_prev(p)) * + (prtls.dx1(p) + prtls.dx1_prev(p)) * + static_cast(INV_2) }; + + const real_t Wx1_1 { INV_2 * + (dxp_r_1 + prtls.dx1_prev(p) + + static_cast(prtls.i1(p) > prtls.i1_prev(p))) }; + const real_t Wx1_2 { INV_2 * + (prtls.dx1(p) + dxp_r_1 + + static_cast( + static_cast(prtls.i1(p) > prtls.i1_prev(p)) + + prtls.i1_prev(p) - prtls.i1(p))) }; + const real_t Fx1_1 { (static_cast(prtls.i1(p) > prtls.i1_prev(p)) + + dxp_r_1 - prtls.dx1_prev(p)) * + coeff * inv_dt }; + const real_t Fx1_2 { (static_cast( + prtls.i1(p) - prtls.i1_prev(p) - + static_cast(prtls.i1(p) > prtls.i1_prev(p))) + + prtls.dx1(p) - dxp_r_1) * + coeff * inv_dt }; - if constexpr (O == 0u) { - /* - Zig-zag deposit - */ - const auto dxp_r_1 { static_cast(i1(p) == i1_prev(p)) * - (dx1(p) + dx1_prev(p)) * + if constexpr (D == Dim::_1D) { + const real_t Fx2_1 { HALF * vp[1] * coeff }; + const real_t Fx2_2 { HALF * vp[1] * coeff }; + + const real_t Fx3_1 { HALF * vp[2] * coeff }; + const real_t Fx3_2 { HALF * vp[2] * coeff }; + + deposit_at(prtls.i1_prev(p) + N_GHOSTS, cur::jx1, Fx1_1); + deposit_at(prtls.i1(p) + N_GHOSTS, cur::jx1, Fx1_2); + + deposit_at(prtls.i1_prev(p) + N_GHOSTS, cur::jx2, Fx2_1 * (ONE - Wx1_1)); + deposit_at(prtls.i1_prev(p) + N_GHOSTS + 1, cur::jx2, Fx2_1 * Wx1_1); + deposit_at(prtls.i1(p) + N_GHOSTS, cur::jx2, Fx2_2 * (ONE - Wx1_2)); + deposit_at(prtls.i1(p) + N_GHOSTS + 1, cur::jx2, Fx2_2 * Wx1_2); + + deposit_at(prtls.i1_prev(p) + N_GHOSTS, cur::jx3, Fx3_1 * (ONE - Wx1_1)); + deposit_at(prtls.i1_prev(p) + N_GHOSTS + 1, cur::jx3, Fx3_1 * Wx1_1); + deposit_at(prtls.i1(p) + N_GHOSTS, cur::jx3, Fx3_2 * (ONE - Wx1_2)); + deposit_at(prtls.i1(p) + N_GHOSTS + 1, cur::jx3, Fx3_2 * Wx1_2); + } else if constexpr (D == Dim::_2D || D == Dim::_3D) { + const auto dxp_r_2 { static_cast(prtls.i2(p) == prtls.i2_prev(p)) * + (prtls.dx2(p) + prtls.dx2_prev(p)) * static_cast(INV_2) }; - const real_t Wx1_1 { INV_2 * (dxp_r_1 + dx1_prev(p) + - static_cast(i1(p) > i1_prev(p))) }; - const real_t Wx1_2 { INV_2 * (dx1(p) + dxp_r_1 + - static_cast( - static_cast(i1(p) > i1_prev(p)) + - i1_prev(p) - i1(p))) }; - const real_t Fx1_1 { (static_cast(i1(p) > i1_prev(p)) + - dxp_r_1 - dx1_prev(p)) * - coeff * inv_dt }; - const real_t Fx1_2 { (static_cast( - i1(p) - i1_prev(p) - - static_cast(i1(p) > i1_prev(p))) + - dx1(p) - dxp_r_1) * + const real_t Wx2_1 { INV_2 * (dxp_r_2 + prtls.dx2_prev(p) + + static_cast(prtls.i2(p) > + prtls.i2_prev(p))) }; + const real_t Wx2_2 { INV_2 * + (prtls.dx2(p) + dxp_r_2 + + static_cast( + static_cast(prtls.i2(p) > prtls.i2_prev(p)) + + prtls.i2_prev(p) - prtls.i2(p))) }; + const real_t Fx2_1 { (static_cast(prtls.i2(p) > prtls.i2_prev(p)) + + dxp_r_2 - prtls.dx2_prev(p)) * coeff * inv_dt }; - - if constexpr (D == Dim::_1D) { - const real_t Fx2_1 { HALF * vp[1] * coeff }; - const real_t Fx2_2 { HALF * vp[1] * coeff }; - + const real_t Fx2_2 { + (static_cast(prtls.i2(p) - prtls.i2_prev(p) - + static_cast(prtls.i2(p) > prtls.i2_prev(p))) + + prtls.dx2(p) - dxp_r_2) * + coeff * inv_dt + }; + + if constexpr (D == Dim::_2D) { const real_t Fx3_1 { HALF * vp[2] * coeff }; const real_t Fx3_2 { HALF * vp[2] * coeff }; - deposit_at(i1_prev(p) + N_GHOSTS, cur::jx1, Fx1_1); - deposit_at(i1(p) + N_GHOSTS, cur::jx1, Fx1_2); - - deposit_at(i1_prev(p) + N_GHOSTS, cur::jx2, Fx2_1 * (ONE - Wx1_1)); - deposit_at(i1_prev(p) + N_GHOSTS + 1, cur::jx2, Fx2_1 * Wx1_1); - deposit_at(i1(p) + N_GHOSTS, cur::jx2, Fx2_2 * (ONE - Wx1_2)); - deposit_at(i1(p) + N_GHOSTS + 1, cur::jx2, Fx2_2 * Wx1_2); - - deposit_at(i1_prev(p) + N_GHOSTS, cur::jx3, Fx3_1 * (ONE - Wx1_1)); - deposit_at(i1_prev(p) + N_GHOSTS + 1, cur::jx3, Fx3_1 * Wx1_1); - deposit_at(i1(p) + N_GHOSTS, cur::jx3, Fx3_2 * (ONE - Wx1_2)); - deposit_at(i1(p) + N_GHOSTS + 1, cur::jx3, Fx3_2 * Wx1_2); - } else if constexpr (D == Dim::_2D || D == Dim::_3D) { - const auto dxp_r_2 { static_cast(i2(p) == i2_prev(p)) * - (dx2(p) + dx2_prev(p)) * - static_cast(INV_2) }; - - const real_t Wx2_1 { INV_2 * (dxp_r_2 + dx2_prev(p) + - static_cast(i2(p) > i2_prev(p))) }; - const real_t Wx2_2 { INV_2 * (dx2(p) + dxp_r_2 + + deposit_at(prtls.i1_prev(p) + N_GHOSTS, + prtls.i2_prev(p) + N_GHOSTS, + cur::jx1, + Fx1_1 * (ONE - Wx2_1)); + deposit_at(prtls.i1_prev(p) + N_GHOSTS, + prtls.i2_prev(p) + N_GHOSTS + 1, + cur::jx1, + Fx1_1 * Wx2_1); + deposit_at(prtls.i1(p) + N_GHOSTS, + prtls.i2(p) + N_GHOSTS, + cur::jx1, + Fx1_2 * (ONE - Wx2_2)); + deposit_at(prtls.i1(p) + N_GHOSTS, + prtls.i2(p) + N_GHOSTS + 1, + cur::jx1, + Fx1_2 * Wx2_2); + + deposit_at(prtls.i1_prev(p) + N_GHOSTS, + prtls.i2_prev(p) + N_GHOSTS, + cur::jx2, + Fx2_1 * (ONE - Wx1_1)); + deposit_at(prtls.i1_prev(p) + N_GHOSTS + 1, + prtls.i2_prev(p) + N_GHOSTS, + cur::jx2, + Fx2_1 * Wx1_1); + deposit_at(prtls.i1(p) + N_GHOSTS, + prtls.i2(p) + N_GHOSTS, + cur::jx2, + Fx2_2 * (ONE - Wx1_2)); + deposit_at(prtls.i1(p) + N_GHOSTS + 1, + prtls.i2(p) + N_GHOSTS, + cur::jx2, + Fx2_2 * Wx1_2); + + deposit_at(prtls.i1_prev(p) + N_GHOSTS, + prtls.i2_prev(p) + N_GHOSTS, + cur::jx3, + Fx3_1 * (ONE - Wx1_1) * (ONE - Wx2_1)); + deposit_at(prtls.i1_prev(p) + N_GHOSTS + 1, + prtls.i2_prev(p) + N_GHOSTS, + cur::jx3, + Fx3_1 * Wx1_1 * (ONE - Wx2_1)); + deposit_at(prtls.i1_prev(p) + N_GHOSTS, + prtls.i2_prev(p) + N_GHOSTS + 1, + cur::jx3, + Fx3_1 * (ONE - Wx1_1) * Wx2_1); + deposit_at(prtls.i1_prev(p) + N_GHOSTS + 1, + prtls.i2_prev(p) + N_GHOSTS + 1, + cur::jx3, + Fx3_1 * Wx1_1 * Wx2_1); + + deposit_at(prtls.i1(p) + N_GHOSTS, + prtls.i2(p) + N_GHOSTS, + cur::jx3, + Fx3_2 * (ONE - Wx1_2) * (ONE - Wx2_2)); + deposit_at(prtls.i1(p) + N_GHOSTS + 1, + prtls.i2(p) + N_GHOSTS, + cur::jx3, + Fx3_2 * Wx1_2 * (ONE - Wx2_2)); + deposit_at(prtls.i1(p) + N_GHOSTS, + prtls.i2(p) + N_GHOSTS + 1, + cur::jx3, + Fx3_2 * (ONE - Wx1_2) * Wx2_2); + deposit_at(prtls.i1(p) + N_GHOSTS + 1, + prtls.i2(p) + N_GHOSTS + 1, + cur::jx3, + Fx3_2 * Wx1_2 * Wx2_2); + } else { + const auto dxp_r_3 { + static_cast(prtls.i3(p) == prtls.i3_prev(p)) * + (prtls.dx3(p) + prtls.dx3_prev(p)) * static_cast(INV_2) + }; + const real_t Wx3_1 { INV_2 * (dxp_r_3 + prtls.dx3_prev(p) + static_cast( - static_cast(i2(p) > i2_prev(p)) + - i2_prev(p) - i2(p))) }; - const real_t Fx2_1 { (static_cast(i2(p) > i2_prev(p)) + - dxp_r_2 - dx2_prev(p)) * - coeff * inv_dt }; - const real_t Fx2_2 { (static_cast( - i2(p) - i2_prev(p) - - static_cast(i2(p) > i2_prev(p))) + - dx2(p) - dxp_r_2) * + prtls.i3(p) > prtls.i3_prev(p))) }; + const real_t Wx3_2 { + INV_2 * + (prtls.dx3(p) + dxp_r_3 + + static_cast(static_cast(prtls.i3(p) > prtls.i3_prev(p)) + + prtls.i3_prev(p) - prtls.i3(p))) + }; + const real_t Fx3_1 { (static_cast(prtls.i3(p) > prtls.i3_prev(p)) + + dxp_r_3 - prtls.dx3_prev(p)) * coeff * inv_dt }; + const real_t Fx3_2 { + (static_cast(prtls.i3(p) - prtls.i3_prev(p) - + static_cast(prtls.i3(p) > prtls.i3_prev(p))) + + prtls.dx3(p) - dxp_r_3) * + coeff * inv_dt + }; + + deposit_at(prtls.i1_prev(p) + N_GHOSTS, + prtls.i2_prev(p) + N_GHOSTS, + prtls.i3_prev(p) + N_GHOSTS, + cur::jx1, + Fx1_1 * (ONE - Wx2_1) * (ONE - Wx3_1)); + deposit_at(prtls.i1_prev(p) + N_GHOSTS, + prtls.i2_prev(p) + N_GHOSTS + 1, + prtls.i3_prev(p) + N_GHOSTS, + cur::jx1, + Fx1_1 * Wx2_1 * (ONE - Wx3_1)); + deposit_at(prtls.i1_prev(p) + N_GHOSTS, + prtls.i2_prev(p) + N_GHOSTS, + prtls.i3_prev(p) + N_GHOSTS + 1, + cur::jx1, + Fx1_1 * (ONE - Wx2_1) * Wx3_1); + deposit_at(prtls.i1_prev(p) + N_GHOSTS, + prtls.i2_prev(p) + N_GHOSTS + 1, + prtls.i3_prev(p) + N_GHOSTS + 1, + cur::jx1, + Fx1_1 * Wx2_1 * Wx3_1); + + deposit_at(prtls.i1(p) + N_GHOSTS, + prtls.i2(p) + N_GHOSTS, + prtls.i3(p) + N_GHOSTS, + cur::jx1, + Fx1_2 * (ONE - Wx2_2) * (ONE - Wx3_2)); + deposit_at(prtls.i1(p) + N_GHOSTS, + prtls.i2(p) + N_GHOSTS + 1, + prtls.i3(p) + N_GHOSTS, + cur::jx1, + Fx1_2 * Wx2_2 * (ONE - Wx3_2)); + deposit_at(prtls.i1(p) + N_GHOSTS, + prtls.i2(p) + N_GHOSTS, + prtls.i3(p) + N_GHOSTS + 1, + cur::jx1, + Fx1_2 * (ONE - Wx2_2) * Wx3_2); + deposit_at(prtls.i1(p) + N_GHOSTS, + prtls.i2(p) + N_GHOSTS + 1, + prtls.i3(p) + N_GHOSTS + 1, + cur::jx1, + Fx1_2 * Wx2_2 * Wx3_2); + + deposit_at(prtls.i1_prev(p) + N_GHOSTS, + prtls.i2_prev(p) + N_GHOSTS, + prtls.i3_prev(p) + N_GHOSTS, + cur::jx2, + Fx2_1 * (ONE - Wx1_1) * (ONE - Wx3_1)); + deposit_at(prtls.i1_prev(p) + N_GHOSTS + 1, + prtls.i2_prev(p) + N_GHOSTS, + prtls.i3_prev(p) + N_GHOSTS, + cur::jx2, + Fx2_1 * Wx1_1 * (ONE - Wx3_1)); + deposit_at(prtls.i1_prev(p) + N_GHOSTS, + prtls.i2_prev(p) + N_GHOSTS, + prtls.i3_prev(p) + N_GHOSTS + 1, + cur::jx2, + Fx2_1 * (ONE - Wx1_1) * Wx3_1); + deposit_at(prtls.i1_prev(p) + N_GHOSTS + 1, + prtls.i2_prev(p) + N_GHOSTS, + prtls.i3_prev(p) + N_GHOSTS + 1, + cur::jx2, + Fx2_1 * Wx1_1 * Wx3_1); + + deposit_at(prtls.i1(p) + N_GHOSTS, + prtls.i2(p) + N_GHOSTS, + prtls.i3(p) + N_GHOSTS, + cur::jx2, + Fx2_2 * (ONE - Wx1_2) * (ONE - Wx3_2)); + deposit_at(prtls.i1(p) + N_GHOSTS + 1, + prtls.i2(p) + N_GHOSTS, + prtls.i3(p) + N_GHOSTS, + cur::jx2, + Fx2_2 * Wx1_2 * (ONE - Wx3_2)); + deposit_at(prtls.i1(p) + N_GHOSTS, + prtls.i2(p) + N_GHOSTS, + prtls.i3(p) + N_GHOSTS + 1, + cur::jx2, + Fx2_2 * (ONE - Wx1_2) * Wx3_2); + deposit_at(prtls.i1(p) + N_GHOSTS + 1, + prtls.i2(p) + N_GHOSTS, + prtls.i3(p) + N_GHOSTS + 1, + cur::jx2, + Fx2_2 * Wx1_2 * Wx3_2); + + deposit_at(prtls.i1_prev(p) + N_GHOSTS, + prtls.i2_prev(p) + N_GHOSTS, + prtls.i3_prev(p) + N_GHOSTS, + cur::jx3, + Fx3_1 * (ONE - Wx1_1) * (ONE - Wx2_1)); + deposit_at(prtls.i1_prev(p) + N_GHOSTS + 1, + prtls.i2_prev(p) + N_GHOSTS, + prtls.i3_prev(p) + N_GHOSTS, + cur::jx3, + Fx3_1 * Wx1_1 * (ONE - Wx2_1)); + deposit_at(prtls.i1_prev(p) + N_GHOSTS, + prtls.i2_prev(p) + N_GHOSTS + 1, + prtls.i3_prev(p) + N_GHOSTS, + cur::jx3, + Fx3_1 * (ONE - Wx1_1) * Wx2_1); + deposit_at(prtls.i1_prev(p) + N_GHOSTS + 1, + prtls.i2_prev(p) + N_GHOSTS + 1, + prtls.i3_prev(p) + N_GHOSTS, + cur::jx3, + Fx3_1 * Wx1_1 * Wx2_1); + + deposit_at(prtls.i1(p) + N_GHOSTS, + prtls.i2(p) + N_GHOSTS, + prtls.i3(p) + N_GHOSTS, + cur::jx3, + Fx3_2 * (ONE - Wx1_2) * (ONE - Wx2_2)); + deposit_at(prtls.i1(p) + N_GHOSTS + 1, + prtls.i2(p) + N_GHOSTS, + prtls.i3(p) + N_GHOSTS, + cur::jx3, + Fx3_2 * Wx1_2 * (ONE - Wx2_2)); + deposit_at(prtls.i1(p) + N_GHOSTS, + prtls.i2(p) + N_GHOSTS + 1, + prtls.i3(p) + N_GHOSTS, + cur::jx3, + Fx3_2 * (ONE - Wx1_2) * Wx2_2); + deposit_at(prtls.i1(p) + N_GHOSTS + 1, + prtls.i2(p) + N_GHOSTS + 1, + prtls.i3(p) + N_GHOSTS, + cur::jx3, + Fx3_2 * Wx1_2 * Wx2_2); + } + } + } else if constexpr ((O >= 1u) and (O <= 11u)) { + + // shape function in dim1 -> always required + real_t iS_x1[O + 2], fS_x1[O + 2]; + // indices of the shape function + int i1_min, i1_max; + + // call shape function + prtl_shape::for_deposit(prtls.i1_prev(p), + static_cast(prtls.dx1_prev(p)), + prtls.i1(p), + static_cast(prtls.dx1(p)), + i1_min, + i1_max, + iS_x1, + fS_x1); - if constexpr (D == Dim::_2D) { - const real_t Fx3_1 { HALF * vp[2] * coeff }; - const real_t Fx3_2 { HALF * vp[2] * coeff }; - - deposit_at(i1_prev(p) + N_GHOSTS, - i2_prev(p) + N_GHOSTS, - cur::jx1, - Fx1_1 * (ONE - Wx2_1)); - deposit_at(i1_prev(p) + N_GHOSTS, - i2_prev(p) + N_GHOSTS + 1, - cur::jx1, - Fx1_1 * Wx2_1); - deposit_at(i1(p) + N_GHOSTS, - i2(p) + N_GHOSTS, - cur::jx1, - Fx1_2 * (ONE - Wx2_2)); - deposit_at(i1(p) + N_GHOSTS, - i2(p) + N_GHOSTS + 1, - cur::jx1, - Fx1_2 * Wx2_2); - - deposit_at(i1_prev(p) + N_GHOSTS, - i2_prev(p) + N_GHOSTS, - cur::jx2, - Fx2_1 * (ONE - Wx1_1)); - deposit_at(i1_prev(p) + N_GHOSTS + 1, - i2_prev(p) + N_GHOSTS, - cur::jx2, - Fx2_1 * Wx1_1); - deposit_at(i1(p) + N_GHOSTS, - i2(p) + N_GHOSTS, - cur::jx2, - Fx2_2 * (ONE - Wx1_2)); - deposit_at(i1(p) + N_GHOSTS + 1, - i2(p) + N_GHOSTS, - cur::jx2, - Fx2_2 * Wx1_2); - - deposit_at(i1_prev(p) + N_GHOSTS, - i2_prev(p) + N_GHOSTS, - cur::jx3, - Fx3_1 * (ONE - Wx1_1) * (ONE - Wx2_1)); - deposit_at(i1_prev(p) + N_GHOSTS + 1, - i2_prev(p) + N_GHOSTS, - cur::jx3, - Fx3_1 * Wx1_1 * (ONE - Wx2_1)); - deposit_at(i1_prev(p) + N_GHOSTS, - i2_prev(p) + N_GHOSTS + 1, - cur::jx3, - Fx3_1 * (ONE - Wx1_1) * Wx2_1); - deposit_at(i1_prev(p) + N_GHOSTS + 1, - i2_prev(p) + N_GHOSTS + 1, - cur::jx3, - Fx3_1 * Wx1_1 * Wx2_1); - - deposit_at(i1(p) + N_GHOSTS, - i2(p) + N_GHOSTS, - cur::jx3, - Fx3_2 * (ONE - Wx1_2) * (ONE - Wx2_2)); - deposit_at(i1(p) + N_GHOSTS + 1, - i2(p) + N_GHOSTS, - cur::jx3, - Fx3_2 * Wx1_2 * (ONE - Wx2_2)); - deposit_at(i1(p) + N_GHOSTS, - i2(p) + N_GHOSTS + 1, - cur::jx3, - Fx3_2 * (ONE - Wx1_2) * Wx2_2); - deposit_at(i1(p) + N_GHOSTS + 1, - i2(p) + N_GHOSTS + 1, - cur::jx3, - Fx3_2 * Wx1_2 * Wx2_2); - } else { - const auto dxp_r_3 { static_cast(i3(p) == i3_prev(p)) * - (dx3(p) + dx3_prev(p)) * - static_cast(INV_2) }; - const real_t Wx3_1 { INV_2 * (dxp_r_3 + dx3_prev(p) + - static_cast(i3(p) > i3_prev(p))) }; - const real_t Wx3_2 { INV_2 * (dx3(p) + dxp_r_3 + - static_cast( - static_cast(i3(p) > i3_prev(p)) + - i3_prev(p) - i3(p))) }; - const real_t Fx3_1 { (static_cast(i3(p) > i3_prev(p)) + - dxp_r_3 - dx3_prev(p)) * - coeff * inv_dt }; - const real_t Fx3_2 { (static_cast( - i3(p) - i3_prev(p) - - static_cast(i3(p) > i3_prev(p))) + - dx3(p) - dxp_r_3) * - coeff * inv_dt }; - - deposit_at(i1_prev(p) + N_GHOSTS, - i2_prev(p) + N_GHOSTS, - i3_prev(p) + N_GHOSTS, - cur::jx1, - Fx1_1 * (ONE - Wx2_1) * (ONE - Wx3_1)); - deposit_at(i1_prev(p) + N_GHOSTS, - i2_prev(p) + N_GHOSTS + 1, - i3_prev(p) + N_GHOSTS, - cur::jx1, - Fx1_1 * Wx2_1 * (ONE - Wx3_1)); - deposit_at(i1_prev(p) + N_GHOSTS, - i2_prev(p) + N_GHOSTS, - i3_prev(p) + N_GHOSTS + 1, - cur::jx1, - Fx1_1 * (ONE - Wx2_1) * Wx3_1); - deposit_at(i1_prev(p) + N_GHOSTS, - i2_prev(p) + N_GHOSTS + 1, - i3_prev(p) + N_GHOSTS + 1, - cur::jx1, - Fx1_1 * Wx2_1 * Wx3_1); - - deposit_at(i1(p) + N_GHOSTS, - i2(p) + N_GHOSTS, - i3(p) + N_GHOSTS, - cur::jx1, - Fx1_2 * (ONE - Wx2_2) * (ONE - Wx3_2)); - deposit_at(i1(p) + N_GHOSTS, - i2(p) + N_GHOSTS + 1, - i3(p) + N_GHOSTS, - cur::jx1, - Fx1_2 * Wx2_2 * (ONE - Wx3_2)); - deposit_at(i1(p) + N_GHOSTS, - i2(p) + N_GHOSTS, - i3(p) + N_GHOSTS + 1, - cur::jx1, - Fx1_2 * (ONE - Wx2_2) * Wx3_2); - deposit_at(i1(p) + N_GHOSTS, - i2(p) + N_GHOSTS + 1, - i3(p) + N_GHOSTS + 1, - cur::jx1, - Fx1_2 * Wx2_2 * Wx3_2); - - deposit_at(i1_prev(p) + N_GHOSTS, - i2_prev(p) + N_GHOSTS, - i3_prev(p) + N_GHOSTS, - cur::jx2, - Fx2_1 * (ONE - Wx1_1) * (ONE - Wx3_1)); - deposit_at(i1_prev(p) + N_GHOSTS + 1, - i2_prev(p) + N_GHOSTS, - i3_prev(p) + N_GHOSTS, - cur::jx2, - Fx2_1 * Wx1_1 * (ONE - Wx3_1)); - deposit_at(i1_prev(p) + N_GHOSTS, - i2_prev(p) + N_GHOSTS, - i3_prev(p) + N_GHOSTS + 1, - cur::jx2, - Fx2_1 * (ONE - Wx1_1) * Wx3_1); - deposit_at(i1_prev(p) + N_GHOSTS + 1, - i2_prev(p) + N_GHOSTS, - i3_prev(p) + N_GHOSTS + 1, - cur::jx2, - Fx2_1 * Wx1_1 * Wx3_1); - - deposit_at(i1(p) + N_GHOSTS, - i2(p) + N_GHOSTS, - i3(p) + N_GHOSTS, - cur::jx2, - Fx2_2 * (ONE - Wx1_2) * (ONE - Wx3_2)); - deposit_at(i1(p) + N_GHOSTS + 1, - i2(p) + N_GHOSTS, - i3(p) + N_GHOSTS, - cur::jx2, - Fx2_2 * Wx1_2 * (ONE - Wx3_2)); - deposit_at(i1(p) + N_GHOSTS, - i2(p) + N_GHOSTS, - i3(p) + N_GHOSTS + 1, - cur::jx2, - Fx2_2 * (ONE - Wx1_2) * Wx3_2); - deposit_at(i1(p) + N_GHOSTS + 1, - i2(p) + N_GHOSTS, - i3(p) + N_GHOSTS + 1, - cur::jx2, - Fx2_2 * Wx1_2 * Wx3_2); - - deposit_at(i1_prev(p) + N_GHOSTS, - i2_prev(p) + N_GHOSTS, - i3_prev(p) + N_GHOSTS, - cur::jx3, - Fx3_1 * (ONE - Wx1_1) * (ONE - Wx2_1)); - deposit_at(i1_prev(p) + N_GHOSTS + 1, - i2_prev(p) + N_GHOSTS, - i3_prev(p) + N_GHOSTS, - cur::jx3, - Fx3_1 * Wx1_1 * (ONE - Wx2_1)); - deposit_at(i1_prev(p) + N_GHOSTS, - i2_prev(p) + N_GHOSTS + 1, - i3_prev(p) + N_GHOSTS, - cur::jx3, - Fx3_1 * (ONE - Wx1_1) * Wx2_1); - deposit_at(i1_prev(p) + N_GHOSTS + 1, - i2_prev(p) + N_GHOSTS + 1, - i3_prev(p) + N_GHOSTS, - cur::jx3, - Fx3_1 * Wx1_1 * Wx2_1); - - deposit_at(i1(p) + N_GHOSTS, - i2(p) + N_GHOSTS, - i3(p) + N_GHOSTS, - cur::jx3, - Fx3_2 * (ONE - Wx1_2) * (ONE - Wx2_2)); - deposit_at(i1(p) + N_GHOSTS + 1, - i2(p) + N_GHOSTS, - i3(p) + N_GHOSTS, - cur::jx3, - Fx3_2 * Wx1_2 * (ONE - Wx2_2)); - deposit_at(i1(p) + N_GHOSTS, - i2(p) + N_GHOSTS + 1, - i3(p) + N_GHOSTS, - cur::jx3, - Fx3_2 * (ONE - Wx1_2) * Wx2_2); - deposit_at(i1(p) + N_GHOSTS + 1, - i2(p) + N_GHOSTS + 1, - i3(p) + N_GHOSTS, - cur::jx3, - Fx3_2 * Wx1_2 * Wx2_2); + if constexpr (D == Dim::_1D) { + // (1D): fused Esirkepov, no [O+2] temporaries. + // jx1[i] = -Qdx1dt * sum_{i'=0}^{i} (fS_x1[i'] - iS_x1[i']) + // = -Qdx1dt * P1[i] (Eq. 38, 1D) + // Wx23[i] = HALF * (fS_x1[i] + iS_x1[i]) (computed inline) + const real_t Qdx1dt = coeff * inv_dt; + const real_t QVx2 = coeff * vp[1]; + const real_t QVx3 = coeff * vp[2]; + + // account for ghost cells + i1_min += N_GHOSTS; + i1_max += N_GHOSTS; + + // get number of update indices for asymmetric movement + const int di_x1 = i1_max - i1_min; + + // Current update — fused over the union line so the J cell + // stays L1-resident across the 3 component atomic_adds. + real_t P1 = ZERO; + for (int i = 0; i <= di_x1; ++i) { + P1 += fS_x1[i] - iS_x1[i]; + const int gi = i1_min + i; + const real_t Wx23 = HALF * (fS_x1[i] + iS_x1[i]); + if (i < di_x1) { + deposit_at(gi, cur::jx1, -Qdx1dt * P1); } + deposit_at(gi, cur::jx2, QVx2 * Wx23); + deposit_at(gi, cur::jx3, QVx3 * Wx23); } - } else if constexpr ((O >= 1u) and (O <= 11u)) { + + } else if constexpr (D == Dim::_2D) { // shape function in dim1 -> always required - real_t iS_x1[O + 2], fS_x1[O + 2]; + real_t iS_x2[O + 2], fS_x2[O + 2]; // indices of the shape function - int i1_min, i1_max; + int i2_min, i2_max; // call shape function - prtl_shape::for_deposit(i1_prev(p), - static_cast(dx1_prev(p)), - i1(p), - static_cast(dx1(p)), - i1_min, - i1_max, - iS_x1, - fS_x1); - - if constexpr (D == Dim::_1D) { - // (1D): fused Esirkepov, no [O+2] temporaries. - // jx1[i] = -Qdx1dt * sum_{i'=0}^{i} (fS_x1[i'] - iS_x1[i']) - // = -Qdx1dt * P1[i] (Eq. 38, 1D) - // Wx23[i] = HALF * (fS_x1[i] + iS_x1[i]) (computed inline) - const real_t Qdx1dt = coeff * inv_dt; - const real_t QVx2 = coeff * vp[1]; - const real_t QVx3 = coeff * vp[2]; - - // account for ghost cells - i1_min += N_GHOSTS; - i1_max += N_GHOSTS; - - // get number of update indices for asymmetric movement - const int di_x1 = i1_max - i1_min; - - // Current update — fused over the union line so the J cell - // stays L1-resident across the 3 component atomic_adds. - real_t P1 = ZERO; - for (int i = 0; i <= di_x1; ++i) { - P1 += fS_x1[i] - iS_x1[i]; - const int gi = i1_min + i; - const real_t Wx23 = HALF * (fS_x1[i] + iS_x1[i]); + prtl_shape::for_deposit(prtls.i2_prev(p), + static_cast(prtls.dx2_prev(p)), + prtls.i2(p), + static_cast(prtls.dx2(p)), + i2_min, + i2_max, + iS_x2, + fS_x2); + + /** + * (2D): fused Esirkepov, no [O+2]^2 temporaries. + * + * Esirkepov 2001 Eq. 38 (simplified) is separable: with + * P1[i] = sum_{i'=0}^{i} (fS_x1[i'] - iS_x1[i']) and + * P2[j] = sum_{j'=0}^{j} (fS_x2[j'] - iS_x2[j']), + * jx1[i][j] = -Q*HALF * P1[i] * (fS_x2[j] + iS_x2[j]) + * jx2[i][j] = -Q*HALF * P2[j] * (fS_x1[i] + iS_x1[i]) + * Wx3[i][j] = THIRD*( fS_x2[j]*(HALF*iS_x1[i]+fS_x1[i]) + * + iS_x2[j]*(HALF*fS_x1[i]+iS_x1[i]) ) + * with Q = coeff*inv_dt (Qdx1dt == Qdx2dt). Same value as the + * old explicit Wx/jx tensors up to FP reassociation; + * charge-conserving by construction. Prefix sums carried as + * running scalars, so the only per-thread state is the + * existing 1D shape arrays. + */ + const real_t QVx3 = coeff * vp[2]; + // -Q*HALF prefactor (Qdx1dt == Qdx2dt == coeff*inv_dt) + const real_t cf = -(coeff * inv_dt) * HALF; + + // account for ghost cells + i1_min += N_GHOSTS; + i2_min += N_GHOSTS; + i1_max += N_GHOSTS; + i2_max += N_GHOSTS; + + // get number of update indices for asymmetric movement + const int di_x1 = i1_max - i1_min; + const int di_x2 = i2_max - i2_min; + + // Current update — fused over the union plane so the J cell + // line stays L1-resident across the 3 component atomic_adds. + real_t P1 = ZERO; + for (int i = 0; i <= di_x1; ++i) { + P1 += fS_x1[i] - iS_x1[i]; + const int gi = i1_min + i; + const real_t iSx1 = iS_x1[i]; + const real_t fSx1 = fS_x1[i]; + const real_t A1 = fSx1 + iSx1; // jx2 cross-factor + real_t P2 = ZERO; + for (int j = 0; j <= di_x2; ++j) { + P2 += fS_x2[j] - iS_x2[j]; + const int gj = i2_min + j; + const real_t iSx2 = iS_x2[j]; + const real_t fSx2 = fS_x2[j]; if (i < di_x1) { - deposit_at(gi, cur::jx1, -Qdx1dt * P1); + deposit_at(gi, gj, cur::jx1, cf * P1 * (fSx2 + iSx2)); } - deposit_at(gi, cur::jx2, QVx2 * Wx23); - deposit_at(gi, cur::jx3, QVx3 * Wx23); + if (j < di_x2) { + deposit_at(gi, gj, cur::jx2, cf * P2 * A1); + } + const real_t Wx3 = THIRD * (fSx2 * (HALF * iSx1 + fSx1) + + iSx2 * (HALF * fSx1 + iSx1)); + deposit_at(gi, gj, cur::jx3, QVx3 * Wx3); } + } + + } else if constexpr (D == Dim::_3D) { + // shape function in dim2 + real_t iS_x2[O + 2], fS_x2[O + 2]; + // indices of the shape function + int i2_min, i2_max; + // call shape function + prtl_shape::for_deposit(prtls.i2_prev(p), + static_cast(prtls.dx2_prev(p)), + prtls.i2(p), + static_cast(prtls.dx2(p)), + i2_min, + i2_max, + iS_x2, + fS_x2); + + // shape function in dim3 + real_t iS_x3[O + 2], fS_x3[O + 2]; + // indices of the shape function + int i3_min, i3_max; - } else if constexpr (D == Dim::_2D) { - - // shape function in dim1 -> always required - real_t iS_x2[O + 2], fS_x2[O + 2]; - // indices of the shape function - int i2_min, i2_max; - - // call shape function - prtl_shape::for_deposit(i2_prev(p), - static_cast(dx2_prev(p)), - i2(p), - static_cast(dx2(p)), - i2_min, - i2_max, - iS_x2, - fS_x2); - - // (2D): fused Esirkepov, no [O+2]^2 temporaries. - // - // Esirkepov 2001 Eq. 38 (simplified) is separable: with - // P1[i] = sum_{i'=0}^{i} (fS_x1[i'] - iS_x1[i']) and - // P2[j] = sum_{j'=0}^{j} (fS_x2[j'] - iS_x2[j']), - // jx1[i][j] = -Q*HALF * P1[i] * (fS_x2[j] + iS_x2[j]) - // jx2[i][j] = -Q*HALF * P2[j] * (fS_x1[i] + iS_x1[i]) - // Wx3[i][j] = THIRD*( fS_x2[j]*(HALF*iS_x1[i]+fS_x1[i]) - // + iS_x2[j]*(HALF*fS_x1[i]+iS_x1[i]) ) - // with Q = coeff*inv_dt (Qdx1dt == Qdx2dt). Same value as the - // old explicit Wx/jx tensors up to FP reassociation; - // charge-conserving by construction. Prefix sums carried as - // running scalars, so the only per-thread state is the - // existing 1D shape arrays. - const real_t QVx3 = coeff * vp[2]; - // -Q*HALF prefactor (Qdx1dt == Qdx2dt == coeff*inv_dt) - const real_t cf = -(coeff * inv_dt) * HALF; - - // account for ghost cells - i1_min += N_GHOSTS; - i2_min += N_GHOSTS; - i1_max += N_GHOSTS; - i2_max += N_GHOSTS; - - // get number of update indices for asymmetric movement - const int di_x1 = i1_max - i1_min; - const int di_x2 = i2_max - i2_min; - - // Current update — fused over the union plane so the J cell - // line stays L1-resident across the 3 component atomic_adds. - real_t P1 = ZERO; - for (int i = 0; i <= di_x1; ++i) { - P1 += fS_x1[i] - iS_x1[i]; - const int gi = i1_min + i; - const real_t iSx1 = iS_x1[i]; - const real_t fSx1 = fS_x1[i]; - const real_t A1 = fSx1 + iSx1; // jx2 cross-factor - real_t P2 = ZERO; - for (int j = 0; j <= di_x2; ++j) { - P2 += fS_x2[j] - iS_x2[j]; - const int gj = i2_min + j; - const real_t iSx2 = iS_x2[j]; - const real_t fSx2 = fS_x2[j]; + // call shape function + prtl_shape::for_deposit(prtls.i3_prev(p), + static_cast(prtls.dx3_prev(p)), + prtls.i3(p), + static_cast(prtls.dx3(p)), + i3_min, + i3_max, + iS_x3, + fS_x3); + + /** + * fused Esirkepov, no (O+2)^3 temporaries. + * + * The Esirkepov 3D current (2001, Eq. 31) is separable: with + * P1[i] = sum_{i'=0}^{i} (fS_x1[i'] - iS_x1[i']) (and likewise + * P2[j], P3[k]) the cumulative-sum currents collapse to + * + * jx1[i][j][k] = -Q*THIRD * P1[i] * G23(j,k) + * jx2[i][j][k] = -Q*THIRD * P2[j] * H13(i,k) + * jx3[i][j][k] = -Q*THIRD * P3[k] * F12(i,j) + * + * with the 1D-shape cross-factors + * + * G23(j,k) = iS_x2[j]*iS_x3[k] + fS_x2[j]*fS_x3[k] + * + HALF*(iS_x3[k]*fS_x2[j] + iS_x2[j]*fS_x3[k]) + * H13(i,k) = iS_x1[i]*iS_x3[k] + fS_x1[i]*fS_x3[k] + * + HALF*(iS_x3[k]*fS_x1[i] + iS_x1[i]*fS_x3[k]) + * F12(i,j) = iS_x1[i]*iS_x2[j] + fS_x1[i]*fS_x2[j] + * + HALF*(iS_x1[i]*fS_x2[j] + iS_x2[j]*fS_x1[i]) + * + * and Q = coeff*inv_dt (Qdxdt == Qdydt == Qdzdt). This is the + * same value as the old explicit Wx/jx tensors up to + * floating-point reassociation: charge-conserving by + * construction (the Esirkepov decomposition is exact). The + * prefix sums are carried as running scalars in the deposit + * loop, so the only per-thread state is the existing 1D shape + * arrays (no (O+2)^3 / (O+2)^2 locals, hence far fewer VGPRs + * and no private-memory tensor traffic). + */ + + // account for ghost cells + i1_min += N_GHOSTS; + i2_min += N_GHOSTS; + i3_min += N_GHOSTS; + i1_max += N_GHOSTS; + i2_max += N_GHOSTS; + i3_max += N_GHOSTS; + + // get number of update indices for asymmetric movement + const int di_x1 = i1_max - i1_min; + const int di_x2 = i2_max - i2_min; + const int di_x3 = i3_max - i3_min; + + // -Q*THIRD prefactor (Qdxdt == Qdydt == Qdzdt == coeff*inv_dt) + const real_t cf = -(coeff * inv_dt) * THIRD; + + /** + * Current update — fused over the union cube so the J cell + * line stays L1-resident across the 3 component atomic_adds. + * Per-cell branches on (i(i2_prev(p), - static_cast(dx2_prev(p)), - i2(p), - static_cast(dx2(p)), - i2_min, - i2_max, - iS_x2, - fS_x2); - - // shape function in dim3 - real_t iS_x3[O + 2], fS_x3[O + 2]; - // indices of the shape function - int i3_min, i3_max; - - // call shape function - prtl_shape::for_deposit(i3_prev(p), - static_cast(dx3_prev(p)), - i3(p), - static_cast(dx3(p)), - i3_min, - i3_max, - iS_x3, - fS_x3); - - // fused Esirkepov, no (O+2)^3 temporaries. - // - // The Esirkepov 3D current (2001, Eq. 31) is separable: with - // P1[i] = sum_{i'=0}^{i} (fS_x1[i'] - iS_x1[i']) (and likewise - // P2[j], P3[k]) the cumulative-sum currents collapse to - // - // jx1[i][j][k] = -Q*THIRD * P1[i] * G23(j,k) - // jx2[i][j][k] = -Q*THIRD * P2[j] * H13(i,k) - // jx3[i][j][k] = -Q*THIRD * P3[k] * F12(i,j) - // - // with the 1D-shape cross-factors - // - // G23(j,k) = iS_x2[j]*iS_x3[k] + fS_x2[j]*fS_x3[k] - // + HALF*(iS_x3[k]*fS_x2[j] + iS_x2[j]*fS_x3[k]) - // H13(i,k) = iS_x1[i]*iS_x3[k] + fS_x1[i]*fS_x3[k] - // + HALF*(iS_x3[k]*fS_x1[i] + iS_x1[i]*fS_x3[k]) - // F12(i,j) = iS_x1[i]*iS_x2[j] + fS_x1[i]*fS_x2[j] - // + HALF*(iS_x1[i]*fS_x2[j] + iS_x2[j]*fS_x1[i]) - // - // and Q = coeff*inv_dt (Qdxdt == Qdydt == Qdzdt). This is the - // same value as the old explicit Wx/jx tensors up to - // floating-point reassociation: charge-conserving by - // construction (the Esirkepov decomposition is exact). The - // prefix sums are carried as running scalars in the deposit - // loop, so the only per-thread state is the existing 1D shape - // arrays (no (O+2)^3 / (O+2)^2 locals, hence far fewer VGPRs - // and no private-memory tensor traffic). - - // account for ghost cells - i1_min += N_GHOSTS; - i2_min += N_GHOSTS; - i3_min += N_GHOSTS; - i1_max += N_GHOSTS; - i2_max += N_GHOSTS; - i3_max += N_GHOSTS; - - // get number of update indices for asymmetric movement - const int di_x1 = i1_max - i1_min; - const int di_x2 = i2_max - i2_min; - const int di_x3 = i3_max - i3_min; - - // -Q*THIRD prefactor (Qdxdt == Qdydt == Qdzdt == coeff*inv_dt) - const real_t cf = -(coeff * inv_dt) * THIRD; - - /* - Current update — fused over the union cube so the J cell - line stays L1-resident across the 3 component atomic_adds. - Per-cell branches on (i 11 not supported. Seriously. " - "What are you even doing here? Entity already goes to 11!"); - } + } // dim + } else { // order + raise::KernelError( + HERE, + "Unsupported interpolation order. O > 11 not supported. Seriously. " + "What are you even doing here? Entity already goes to 11!"); } - - } // namespace deposit + } /** * @brief Flat current-deposition kernel. @@ -721,38 +696,19 @@ namespace kernel { static_assert(O <= 11u, "Shape function order O must be <= 11"); static constexpr auto D = M::Dim; - scatter_ndfield_t J; - deposit::PrtlPack prtls; - const M metric; - const real_t charge, inv_dt; + scatter_ndfield_t J; + const ParticleArrays prtls; + const M metric; + const real_t charge, inv_dt; public: DepositCurrents_kernel(const scatter_ndfield_t& scatter_cur, - const array_t& i1, - const array_t& i2, - const array_t& i3, - const array_t& i1_prev, - const array_t& i2_prev, - const array_t& i3_prev, - const array_t& dx1, - const array_t& dx2, - const array_t& dx3, - const array_t& dx1_prev, - const array_t& dx2_prev, - const array_t& dx3_prev, - const array_t& ux1, - const array_t& ux2, - const array_t& ux3, - const array_t& phi, - const array_t& weight, - const array_t& tag, + const ParticleArrays& prtls, const M& metric, real_t charge, const real_t dt) : J { scatter_cur } - , prtls { i1, i2, i3, i1_prev, i2_prev, i3_prev, - dx1, dx2, dx3, dx1_prev, dx2_prev, dx3_prev, - ux1, ux2, ux3, phi, weight, tag } + , prtls { prtls } , metric { metric } , charge { charge } , inv_dt { ONE / dt } { @@ -765,27 +721,25 @@ namespace kernel { Inline auto operator()(prtlidx_t p) const -> void { auto J_acc = J.access(); if constexpr (D == Dim::_1D) { - deposit::deposit_one_particle( - p, - prtls, - metric, - charge, - inv_dt, - [&](int g_i1, int comp, real_t v) { - J_acc(g_i1, comp) += v; - }); + DepositOneParticle(p, + prtls, + metric, + charge, + inv_dt, + [&](int g_i1, int comp, real_t v) { + J_acc(g_i1, comp) += v; + }); } else if constexpr (D == Dim::_2D) { - deposit::deposit_one_particle( - p, - prtls, - metric, - charge, - inv_dt, - [&](int g_i1, int g_i2, int comp, real_t v) { - J_acc(g_i1, g_i2, comp) += v; - }); + DepositOneParticle(p, + prtls, + metric, + charge, + inv_dt, + [&](int g_i1, int g_i2, int comp, real_t v) { + J_acc(g_i1, g_i2, comp) += v; + }); } else if constexpr (D == Dim::_3D) { - deposit::deposit_one_particle( + DepositOneParticle( p, prtls, metric, @@ -798,8 +752,6 @@ namespace kernel { } }; -#if defined(TEAM_POLICY) - /** * @brief Tiled current-deposition kernel. * @@ -851,107 +803,74 @@ namespace kernel { * per write. Sorting less often than every step therefore costs * escape-valve traffic, never accuracy. */ - template - class DepositCurrents_kernel_tiled { + template + class DepositCurrentsTiled_kernel { static_assert(O <= 11u, "Shape order O must be <= 11"); static_assert(T_TILE > 0u, "T_TILE must be positive"); static constexpr auto D = M::Dim; - // Per-side scratch halo, derived from first principles. - // - // total halo = stencil_reach(O) + drift_between_sort_and_deposit - // - // stencil_reach(O) — maximum cells the deposit writes ABOVE - // min(i, i_prev) under CFL |v·dt/dx| ≤ 1/2: - // - O == 0 (zigzag): writes {i_prev, i_prev+1, i, i+1} ⇒ +2 - // - O >= 1 Esirkepov: `for_deposit` returns an (O+2)-wide - // array but only O+1 entries are non-zero, and the union - // window satisfies `i_max - i_min <= O+1` (see - // particle_shapes.hpp::for_deposit). The genuine one-sided - // reach above min(i, i_prev) is therefore O, not O+1 — the - // old `O+1` carried one extra cell of conservative padding - // on top of the already-conservative drift term below. - // - // drift — sort runs at end-of-step (see srpic.hpp), so a particle - // sees exactly one pusher step before the *next* step's deposit - // when sorted every step (the common case). DRIFT is therefore a - // fixed constant of 1, NOT a compile-time function of the runtime - // sort cadence. Sizing the halo for the common case (rather than a - // worst-case sort interval) is what keeps the scratch small enough - // for good occupancy; a species sorted less often than - // every step just drifts past the halo and takes the global-J - // escape valve more often — correct, only slower (see the class - // doc-comment for why this is charge-conserving). - // - static constexpr int STENCIL_REACH = (O == 0u) - ? 2 - : static_cast(O); - static constexpr int DRIFT = 1; - static constexpr int HALO = STENCIL_REACH + DRIFT; - static constexpr int TE = static_cast(T_TILE) + 2 * HALO; - - using exec_space = Kokkos::DefaultExecutionSpace; - using team_policy = Kokkos::TeamPolicy; - using member_t = typename team_policy::member_type; - using scratch_mem = typename exec_space::scratch_memory_space; - - // Scratch view types: trailing extent of 3 (jx1, jx2, jx3 components) - // is fixed by a runtime extent so we don't need a separate dimension - // template per component count. - using scratch_1d_t = Kokkos::View>; - using scratch_2d_t = Kokkos::View>; - using scratch_3d_t = Kokkos::View>; - - ndfield_t J; - deposit::PrtlPack prtls; - const M metric; - const real_t charge, inv_dt; + /** + * Per-side scratch halo, derived from first principles. + * + * total halo = stencil_reach(O) + drift_between_sort_and_deposit + * + * stencil_reach(O) — maximum cells the deposit writes ABOVE + * min(i, i_prev) under CFL |v * dt/dx| <= 1/2: + * - O == 0 (zigzag): writes { i_prev, i_prev+1, i, i+1 } => +2 + * - O >= 1 Esirkepov: `for_deposit` returns an (O+2)-wide + * array but only O+1 entries are non-zero, and the union + * window satisfies `i_max - i_min <= O+1` (see + * particle_shapes.hpp::for_deposit). The genuine one-sided + * reach above min(i, i_prev) is therefore O, not O+1 — the + * old `O+1` carried one extra cell of conservative padding + * on top of the already-conservative drift term below. + * + * drift — sort runs at end-of-step (see srpic.hpp), so a particle + * sees exactly one pusher step before the *next* step's deposit + * when sorted every step (the common case). DRIFT is therefore a + * fixed constant of 1, NOT a compile-time function of the runtime + * sort cadence. Sizing the halo for the common case (rather than a + * worst-case sort interval) is what keeps the scratch small enough + * for good occupancy; a species sorted less often than + * every step just drifts past the halo and takes the global-J + * escape valve more often — correct, only slower (see the class + * doc-comment for why this is charge-conserving). + */ + static constexpr int STENCIL_REACH = (O == 0u) ? 2 : static_cast(O); + static constexpr int DRIFT = 1; + static constexpr int HALO = STENCIL_REACH + DRIFT; + static constexpr int TE = static_cast(T_TILE) + 2 * HALO; + + using exec_space = Kokkos::DefaultExecutionSpace; + using team_policy = Kokkos::TeamPolicy; + using member_t = typename team_policy::member_type; + + ndfield_t J; + ParticleArrays prtls; + const M metric; + const real_t charge, inv_dt; // Tile metadata produced by SortSpatially. - array_t tile_offsets; - ncells_t ntx1 { 1u }, ntx2 { 1u }, ntx3 { 1u }; - ncells_t total_tiles { 0u }; + array_t tile_offsets; + ncells_t ntx1 { 1u }, ntx2 { 1u }, ntx3 { 1u }; + ncells_t total_tiles { 0u }; - // J's full storage extent including all ghost cells. Used to clip - // the cooperative flush so that a partial tile at the high end of - // the domain does not over-write past the J view. - int j_ext1 { 0 }, j_ext2 { 0 }, j_ext3 { 0 }; + /** + * J's full storage extent including all ghost cells. Used to clip + * the cooperative flush so that a partial tile at the high end of + * the domain does not over-write past the J view. + */ + int j_ext1 { 0 }, j_ext2 { 0 }, j_ext3 { 0 }; public: - DepositCurrents_kernel_tiled(const ndfield_t& cur, - const array_t& i1, - const array_t& i2, - const array_t& i3, - const array_t& i1_prev, - const array_t& i2_prev, - const array_t& i3_prev, - const array_t& dx1, - const array_t& dx2, - const array_t& dx3, - const array_t& dx1_prev, - const array_t& dx2_prev, - const array_t& dx3_prev, - const array_t& ux1, - const array_t& ux2, - const array_t& ux3, - const array_t& phi, - const array_t& weight, - const array_t& tag, - const M& metric, - real_t charge, - const real_t dt, - const TileLayout& layout) + DepositCurrentsTiled_kernel(const ndfield_t& cur, + const ParticleArrays& prtls, + const M& metric, + real_t charge, + real_t dt, + const TileLayout& layout) : J { cur } - , prtls { i1, i2, i3, i1_prev, i2_prev, i3_prev, - dx1, dx2, dx3, dx1_prev, dx2_prev, dx3_prev, - ux1, ux2, ux3, phi, weight, tag } + , prtls { prtls } , metric { metric } , charge { charge } , inv_dt { ONE / dt } @@ -964,16 +883,18 @@ namespace kernel { layout.tile_size != T_TILE, "Tiled deposit launched with mismatched T_TILE and runtime tile_size", HERE); - // Note: HALO is allowed to exceed N_GHOSTS. The cooperative - // scratch→J flush and the per-particle escape valve both bounds-clip - // their writes against `j_ext*` so writes that would land past J's - // ghost stripe are silently dropped (they only ever come from a - // particle whose stencil reaches into the domain ghost region, where - // CommunicateFields will re-supply the contribution). - if constexpr (D == Dim::_1D || D == Dim::_2D || D == Dim::_3D) { + /** + * @note: HALO is allowed to exceed N_GHOSTS. The cooperative + * scratch→J flush and the per-particle escape valve both bounds-clip + * their writes against `j_ext*` so writes that would land past J's + * ghost stripe are silently dropped (they only ever come from a + * particle whose stencil reaches into the domain ghost region, where + * CommunicateFields will re-supply the contribution). + */ + if constexpr (D == Dim::_1D or D == Dim::_2D or D == Dim::_3D) { j_ext1 = static_cast(cur.extent(0)); } - if constexpr (D == Dim::_2D || D == Dim::_3D) { + if constexpr (D == Dim::_2D or D == Dim::_3D) { j_ext2 = static_cast(cur.extent(1)); } if constexpr (D == Dim::_3D) { @@ -985,23 +906,24 @@ namespace kernel { * @brief Per-team scratch size in bytes. Used by the launcher to set * `team_policy.set_scratch_size(0, Kokkos::PerTeam(bytes))`. */ - static constexpr std::size_t scratch_bytes() { + static constexpr size_t scratch_bytes() { if constexpr (D == Dim::_1D) { - return scratch_1d_t::shmem_size(TE, 3); + return scratch_ndfield_t::shmem_size(TE, 3); } else if constexpr (D == Dim::_2D) { - return scratch_2d_t::shmem_size(TE, TE, 3); + return scratch_ndfield_t::shmem_size(TE, TE, 3); } else { - return scratch_3d_t::shmem_size(TE, TE, TE, 3); + return scratch_ndfield_t::shmem_size(TE, TE, TE, 3); } } - KOKKOS_INLINE_FUNCTION - void operator()(const member_t& team) const { + Inline void operator()(const member_t& team) const { const auto tile_id = static_cast(team.league_rank()); - // Tile coordinates (tile-grid indices) → tile origin in **active** - // cell coords (no ghost offset). Using ncells_t to match the linearised - // tile index produced by SortSpatially. - ncells_t tx1 = 0, tx2 = 0, tx3 = 0; + /** + * Tile coordinates (tile-grid indices) → tile origin in **active** + * cell coords (no ghost offset). Using ncells_t to match the linearised + * tile index produced by SortSpatially. + */ + ncells_t tx1 = 0, tx2 = 0, tx3 = 0; if constexpr (D == Dim::_1D) { tx1 = tile_id; } else if constexpr (D == Dim::_2D) { @@ -1014,213 +936,201 @@ namespace kernel { tx2 = rem / ntx3; tx3 = rem - tx2 * ntx3; } - // origin_active = lowest active-cell index in the tile (no ghost). - // origin_J = same value translated into J's storage coordinate - // (i.e. plus N_GHOSTS). - // origin_J_low = J coordinate of scratch index 0 (i.e. origin_J - HALO). - // local index `li` in scratch ↔ global J index `gi = li + origin_J_low`. - const int origin_J1_low = static_cast(tx1 * T_TILE) - + static_cast(N_GHOSTS) - HALO; - const int origin_J2_low = static_cast(tx2 * T_TILE) - + static_cast(N_GHOSTS) - HALO; - const int origin_J3_low = static_cast(tx3 * T_TILE) - + static_cast(N_GHOSTS) - HALO; + /** + * origin_active = lowest active-cell index in the tile (no ghost). + * origin_J = same value translated into J's storage coordinate + * (i.e. plus N_GHOSTS). + * origin_J_low = J coordinate of scratch index 0 (i.e. origin_J - HALO). + * local index `li` in scratch ↔ global J index `gi = li + origin_J_low`. + */ + const int origin_J1_low = static_cast(tx1 * T_TILE) + + static_cast(N_GHOSTS) - HALO; + const int origin_J2_low = static_cast(tx2 * T_TILE) + + static_cast(N_GHOSTS) - HALO; + const int origin_J3_low = static_cast(tx3 * T_TILE) + + static_cast(N_GHOSTS) - HALO; // Allocate scratch and cooperatively zero-fill it. if constexpr (D == Dim::_1D) { - scratch_1d_t scr(team.team_scratch(0), TE, 3); - Kokkos::parallel_for( - Kokkos::TeamThreadRange(team, TE * 3), - [&](const int idx) { - const int li = idx / 3; - const int c = idx - li * 3; - scr(li, c) = ZERO; - }); + scratch_ndfield_t scr { team.team_scratch(0), TE }; + Kokkos::parallel_for(Kokkos::TeamThreadRange(team, TE * 3), + [&](ncells_t idx) { + const auto li = idx / 3; + const auto c = idx - li * 3; + scr(li, c) = ZERO; + }); team.team_barrier(); const auto p_begin = tile_offsets(tile_id); const auto p_end = tile_offsets(tile_id + 1u); - const int e1_d = j_ext1; Kokkos::parallel_for( Kokkos::TeamThreadRange(team, p_begin, p_end), - [&](const npart_t p) { - deposit::deposit_one_particle( + [&](prtlidx_t p) { + DepositOneParticle( p, prtls, metric, charge, inv_dt, - // Escape valve: a particle whose stencil reaches past the - // tile's scratch (e.g. exceeded the compile-time - // STENCIL_REACH + DRIFT budget) falls back to a direct - // atomic_add on the global J view. Bounds-clipped against - // J's storage extent so writes past the domain ghost stripe - // are dropped (matches the cooperative flush below; those - // contributions are re-supplied by SynchronizeFields(J)). + /** + * Escape valve: a particle whose stencil reaches past the + * tile's scratch (e.g. exceeded the compile-time + * STENCIL_REACH + DRIFT budget) falls back to a direct + * atomic_add on the global J view. Bounds-clipped against + * J's storage extent so writes past the domain ghost stripe + * are dropped (matches the cooperative flush below; those + * contributions are re-supplied by SynchronizeFields(J)). + */ [&](int g_i1, int comp, real_t v) { const int li = g_i1 - origin_J1_low; - if (li >= 0 && li < TE) { + if (li >= 0 and li < TE) { Kokkos::atomic_add(&scr(li, comp), v); - } else if (g_i1 >= 0 && g_i1 < e1_d) { + } else if (g_i1 >= 0 and g_i1 < j_ext1) { Kokkos::atomic_add(&J(g_i1, comp), v); } }); }); team.team_barrier(); - // Cooperative flush of scratch to global J. Bounds-clip against - // the J view extent in case a partial high-end tile (or non-zero - // halo at domain edges) would otherwise write past J. - const int e1 = j_ext1; - Kokkos::parallel_for( - Kokkos::TeamThreadRange(team, TE * 3), - [&](const int idx) { - const int li = idx / 3; - const int c = idx - li * 3; - const int gi = li + origin_J1_low; - if (gi < 0 || gi >= e1) { - return; - } - const real_t v = scr(li, c); - if (v != ZERO) { - Kokkos::atomic_add(&J(gi, c), v); - } - }); + /** + * Cooperative flush of scratch to global J. Bounds-clip against + * the J view extent in case a partial high-end tile (or non-zero + * halo at domain edges) would otherwise write past J. + */ + Kokkos::parallel_for(Kokkos::TeamThreadRange(team, TE * 3), + [&](const int idx) { + const auto li = idx / 3; + const auto c = idx - li * 3; + const auto gi = li + origin_J1_low; + if (gi < 0 or gi >= j_ext1) { + return; + } + const real_t v = scr(li, c); + if (v != ZERO) { + Kokkos::atomic_add(&J(gi, c), v); + } + }); } else if constexpr (D == Dim::_2D) { - scratch_2d_t scr(team.team_scratch(0), TE, TE, 3); - Kokkos::parallel_for( - Kokkos::TeamThreadRange(team, TE * TE * 3), - [&](const int idx) { - const int lij = idx / 3; - const int c = idx - lij * 3; - const int li = lij / TE; - const int lj = lij - li * TE; - scr(li, lj, c) = ZERO; - }); + scratch_ndfield_t scr { team.team_scratch(0), TE, TE }; + Kokkos::parallel_for(Kokkos::TeamThreadRange(team, SQR(TE) * 3), + [&](const int idx) { + const auto lij = idx / 3; + const auto c = idx - lij * 3; + const auto li = lij / TE; + const auto lj = lij - li * TE; + scr(li, lj, c) = ZERO; + }); team.team_barrier(); const auto p_begin = tile_offsets(tile_id); const auto p_end = tile_offsets(tile_id + 1u); - const int e1_d = j_ext1; - const int e2_d = j_ext2; Kokkos::parallel_for( Kokkos::TeamThreadRange(team, p_begin, p_end), - [&](const npart_t p) { - deposit::deposit_one_particle( + [&](prtlidx_t p) { + DepositOneParticle( p, prtls, metric, charge, inv_dt, // See 1D branch for rationale. - [&](int g_i1, int g_i2, int comp, real_t v) { - const int li = g_i1 - origin_J1_low; - const int lj = g_i2 - origin_J2_low; - if (li >= 0 && li < TE && lj >= 0 && lj < TE) { + [&](const int g_i1, const int g_i2, int comp, real_t v) { + const auto li = g_i1 - origin_J1_low; + const auto lj = g_i2 - origin_J2_low; + if ((li >= 0 and li < TE) and (lj >= 0 and lj < TE)) { Kokkos::atomic_add(&scr(li, lj, comp), v); - } else if (g_i1 >= 0 && g_i1 < e1_d && g_i2 >= 0 && - g_i2 < e2_d) { + } else if ((g_i1 >= 0 and g_i1 < j_ext1) and + (g_i2 >= 0 and g_i2 < j_ext2)) { Kokkos::atomic_add(&J(g_i1, g_i2, comp), v); } }); }); team.team_barrier(); - const int e1 = j_ext1; - const int e2 = j_ext2; - Kokkos::parallel_for( - Kokkos::TeamThreadRange(team, TE * TE * 3), - [&](const int idx) { - const int lij = idx / 3; - const int c = idx - lij * 3; - const int li = lij / TE; - const int lj = lij - li * TE; - const int gi = li + origin_J1_low; - const int gj = lj + origin_J2_low; - if (gi < 0 || gi >= e1 || gj < 0 || gj >= e2) { - return; - } - const real_t v = scr(li, lj, c); - if (v != ZERO) { - Kokkos::atomic_add(&J(gi, gj, c), v); - } - }); + Kokkos::parallel_for(Kokkos::TeamThreadRange(team, SQR(TE) * 3), + [&](const int idx) { + const auto lij = idx / 3; + const auto c = idx - lij * 3; + const auto li = lij / TE; + const auto lj = lij - li * TE; + const auto gi = li + origin_J1_low; + const auto gj = lj + origin_J2_low; + if ((gi < 0 or gi >= j_ext1) or + (gj < 0 or gj >= j_ext2)) { + return; + } + const real_t v = scr(li, lj, c); + if (v != ZERO) { + Kokkos::atomic_add(&J(gi, gj, c), v); + } + }); } else if constexpr (D == Dim::_3D) { - scratch_3d_t scr(team.team_scratch(0), TE, TE, TE, 3); - const int cells = TE * TE * TE; - Kokkos::parallel_for( - Kokkos::TeamThreadRange(team, cells * 3), - [&](const int idx) { - const int lijk = idx / 3; - const int c = idx - lijk * 3; - const int li = lijk / (TE * TE); - const int rem = lijk - li * TE * TE; - const int lj = rem / TE; - const int lk = rem - lj * TE; - scr(li, lj, lk, c) = ZERO; - }); + scratch_ndfield_t scr { team.team_scratch(0), TE, TE, TE }; + Kokkos::parallel_for(Kokkos::TeamThreadRange(team, CUBE(TE) * 3), + [&](const int idx) { + const auto lijk = idx / 3; + const auto c = idx - lijk * 3; + const auto li = lijk / (TE * TE); + const auto rem = lijk - li * TE * TE; + const auto lj = rem / TE; + const auto lk = rem - lj * TE; + scr(li, lj, lk, c) = ZERO; + }); team.team_barrier(); const auto p_begin = tile_offsets(tile_id); const auto p_end = tile_offsets(tile_id + 1u); - const int e1_d = j_ext1; - const int e2_d = j_ext2; - const int e3_d = j_ext3; Kokkos::parallel_for( Kokkos::TeamThreadRange(team, p_begin, p_end), - [&](const npart_t p) { - deposit::deposit_one_particle( + [&](prtlidx_t p) { + DepositOneParticle( p, prtls, metric, charge, inv_dt, // See 1D branch for rationale. - [&](int g_i1, int g_i2, int g_i3, int comp, real_t v) { - const int li = g_i1 - origin_J1_low; - const int lj = g_i2 - origin_J2_low; - const int lk = g_i3 - origin_J3_low; - if (li >= 0 && li < TE && lj >= 0 && lj < TE && lk >= 0 && - lk < TE) { + [&](const int g_i1, const int g_i2, const int g_i3, int comp, real_t v) { + const auto li = g_i1 - origin_J1_low; + const auto lj = g_i2 - origin_J2_low; + const auto lk = g_i3 - origin_J3_low; + if ((li >= 0 and li < TE) and (lj >= 0 and lj < TE) and + (lk >= 0 and lk < TE)) { Kokkos::atomic_add(&scr(li, lj, lk, comp), v); - } else if (g_i1 >= 0 && g_i1 < e1_d && g_i2 >= 0 && - g_i2 < e2_d && g_i3 >= 0 && g_i3 < e3_d) { + } else if ((g_i1 >= 0 and g_i1 < j_ext1) and + (g_i2 >= 0 and g_i2 < j_ext2) and + (g_i3 >= 0 and g_i3 < j_ext3)) { Kokkos::atomic_add(&J(g_i1, g_i2, g_i3, comp), v); } }); }); team.team_barrier(); - const int e1 = j_ext1; - const int e2 = j_ext2; - const int e3 = j_ext3; - Kokkos::parallel_for( - Kokkos::TeamThreadRange(team, cells * 3), - [&](const int idx) { - const int lijk = idx / 3; - const int c = idx - lijk * 3; - const int li = lijk / (TE * TE); - const int rem = lijk - li * TE * TE; - const int lj = rem / TE; - const int lk = rem - lj * TE; - const int gi = li + origin_J1_low; - const int gj = lj + origin_J2_low; - const int gk = lk + origin_J3_low; - if (gi < 0 || gi >= e1 || gj < 0 || gj >= e2 || gk < 0 || - gk >= e3) { - return; - } - const real_t v = scr(li, lj, lk, c); - if (v != ZERO) { - Kokkos::atomic_add(&J(gi, gj, gk, c), v); - } - }); + Kokkos::parallel_for(Kokkos::TeamThreadRange(team, CUBE(TE) * 3), + [&](const int idx) { + const int lijk = idx / 3; + const int c = idx - lijk * 3; + const int li = lijk / (TE * TE); + const int rem = lijk - li * TE * TE; + const int lj = rem / TE; + const int lk = rem - lj * TE; + const int gi = li + origin_J1_low; + const int gj = lj + origin_J2_low; + const int gk = lk + origin_J3_low; + if ((gi < 0 or gi >= j_ext1) or + (gj < 0 or gj >= j_ext2) or + (gk < 0 or gk >= j_ext3)) { + return; + } + const real_t v = scr(li, lj, lk, c); + if (v != ZERO) { + Kokkos::atomic_add(&J(gi, gj, gk, c), v); + } + }); } } }; -#endif // TEAM_POLICY - } // namespace kernel #undef i_di_to_Xi From b5e05ad7130ca88c44ecdecdb6c46690fb2e2a9e Mon Sep 17 00:00:00 2001 From: LudwigBoess Date: Fri, 19 Jun 2026 20:13:33 +0000 Subject: [PATCH 010/125] compile-time sorting interval with team policies and sorting speedup --- CMakeLists.txt | 13 ++ cmake/defaults.cmake | 5 + cmake/report.cmake | 14 ++ src/engines/srpic/currents.h | 90 ++++--- src/framework/containers/particles.h | 13 +- src/framework/containers/particles_sort.cpp | 110 +++++---- src/framework/parameters/parameters.cpp | 7 + src/framework/parameters/particles.cpp | 9 + src/global/arch/kokkos_aliases.h | 25 +- src/kernels/currents_deposit.hpp | 245 ++++++++++++++------ tests/framework/particles_sort.cpp | 12 +- 11 files changed, 372 insertions(+), 171 deletions(-) diff --git a/CMakeLists.txt b/CMakeLists.txt index 0137671a0..d7bea74f0 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -67,6 +67,10 @@ set(team_policy_tile_size set(team_policy_tile_sizes "4;6;8;10;12;14;16" CACHE STRING "team_policy tile-size choices") +set(team_policy_sort + ${default_team_policy_sort} + CACHE STRING + "team_policy hardwired spatial sorting interval; >0 overrides the runtime spatial_sorting_interval and sizes the tiled deposit scratch halo (0 = use runtime)") # -------------------------- Compilation settings -------------------------- # set(CMAKE_CXX_STANDARD 20) @@ -157,6 +161,15 @@ if(${team_policy}) add_compile_options("-D TEAM_POLICY") add_compile_options("-D TEAM_POLICY_TILE_SIZE=${team_policy_tile_size}") + # Optional compile-time hardwired sort interval. When > 0, it (a) overrides + # the runtime spatial_sorting_interval (see framework/parameters) and (b) + # sizes the tiled deposit scratch halo to DRIFT = interval, so a particle + # drifting over a full sort interval still deposits inside its tile scratch. + if(team_policy_sort GREATER 0) + add_compile_options( + "-D TEAM_POLICY_SORT_INTERVAL=${team_policy_sort}") + endif() + # Vendor sort: oneDPL on SYCL, Thrust on CUDA. Used automatically # when found; falls back to Kokkos::BinSort otherwise. if("${Kokkos_DEVICES}" MATCHES "SYCL") diff --git a/cmake/defaults.cmake b/cmake/defaults.cmake index a85accf84..619f48dc1 100644 --- a/cmake/defaults.cmake +++ b/cmake/defaults.cmake @@ -107,3 +107,8 @@ set_property(CACHE default_team_policy PROPERTY TYPE BOOL) set(default_team_policy_tile_size 8 CACHE INTERNAL "Default tile edge length in cells for team_policy") + +set(default_team_policy_sort + 0 + CACHE INTERNAL + "Default hardwired spatial sorting interval for team_policy (0 = runtime)") diff --git a/cmake/report.cmake b/cmake/report.cmake index 65a22a7a6..e6036d366 100644 --- a/cmake/report.cmake +++ b/cmake/report.cmake @@ -141,6 +141,17 @@ if(${team_policy}) "${Blue}" TEAM_POLICY_TILE_SIZE_REPORT 46) + if(team_policy_sort GREATER 0) + printchoices( + "Team Sort Interval" + "team_policy_sort" + "${team_policy_sort}" + ${team_policy_sort} + 0 + "${Blue}" + TEAM_POLICY_SORT_INTERVAL_REPORT + 46) + endif() endif() printchoices( "Debug mode" @@ -220,6 +231,9 @@ endif() string(APPEND REPORT_TEXT " " ${TEAM_POLICY_REPORT} "\n") if(${team_policy}) string(APPEND REPORT_TEXT " " ${TEAM_POLICY_TILE_SIZE_REPORT} "\n") + if(team_policy_sort GREATER 0) + string(APPEND REPORT_TEXT " " ${TEAM_POLICY_SORT_INTERVAL_REPORT} "\n") + endif() endif() string( diff --git a/src/engines/srpic/currents.h b/src/engines/srpic/currents.h index 63746d101..eadda70b9 100644 --- a/src/engines/srpic/currents.h +++ b/src/engines/srpic/currents.h @@ -79,7 +79,8 @@ namespace ntt { auto deposit_kernel = kernel::DepositCurrentsTiled_kernel { - cur, species, local_metric, (real_t)(species.charge()), dt, layout + cur, species, local_metric, (real_t)(species.charge()), + dt, layout, species.npart() }; Kokkos::TeamPolicy<> policy(static_cast(layout.ntiles_total), @@ -88,6 +89,30 @@ namespace ntt { 0, Kokkos::PerTeam(decltype(deposit_kernel)::scratch_bytes())); Kokkos::parallel_for("CurrentsDepositTiled", policy, deposit_kernel); + + // Particles appended since the last sort (injection / MPI receive on a + // no-sort step) live past the partition and are not visited by any team + // above. Deposit that tail [npart_partitioned, npart) with the flat + // scatter-view kernel so every active particle is deposited exactly + // once. The range is empty when the species was just sorted (the + // every-step-sorted common case), so this is a no-op there. + if (species.npart() > layout.npart_partitioned) { + // `cur` is a const ref; take a non-const View handle (shallow copy, + // shares storage) so the scatter view can contribute back into it. + auto cur_nc = cur; + auto scatter_cur = Kokkos::Experimental::create_scatter_view(cur_nc); + Kokkos::parallel_for( + "CurrentsDepositTiledTail", + CreateParticleRangePolicy({ layout.npart_partitioned }, + { species.npart() }), + kernel::DepositCurrents_kernel( + scatter_cur, + species, + local_metric, + (real_t)(species.charge()), + dt)); + Kokkos::Experimental::contribute(cur_nc, scatter_cur); + } } template @@ -98,51 +123,45 @@ namespace ntt { #if defined(TEAM_POLICY) - // First-step fallback: if any contributing species has not been - // sorted yet (tile_layout still empty), fall back to the flat - // scatter-view path for that step. Subsequent steps see populated - // layouts and use the tiled kernel. - bool any_unsorted = false; + // Tiled deposit. Correctness no longer depends on the SoA being in a + // "sorted" state at deposit time — the tiled kernel handles a stale + // partition per-particle: + // - a particle whose full stencil has drifted out of its tile is + // deposited straight to the global J view (the per-particle escape + // valve); `team_policy_sort_interval` sizes the scratch halo so the + // common in-tile case stays in fast SLM (see currents_deposit.hpp); + // - particles dead-tagged in place since the sort are clamped out by + // the kernel and skipped by the dead-tag test; + // - particles appended past the partition since the sort (injection / + // MPI receive on a no-sort step) are deposited by the launcher's + // flat tail pass over [npart_partitioned, npart). + // Together these cover every active particle exactly once for any sort + // interval. The only case the tiled kernel cannot serve is the very + // first step, before any SortSpatially has populated a layout; that + // species takes the flat scatter-view path for that step alone. for (auto& species : domain.species) { if ((species.pusher() == ParticlePusher::NONE) or (species.npart() == 0) or cmp::AlmostZero_host(species.charge())) { continue; } - if (species.tile_layout().ntiles_total == 0u or - species.tile_layout().tile_offsets.extent(0) == 0u) { - any_unsorted = true; - break; - } - } - if (any_unsorted) { - auto scatter_cur = Kokkos::Experimental::create_scatter_view( - domain.fields.cur); - for (auto& species : domain.species) { - if ((species.pusher() == ParticlePusher::NONE) or - (species.npart() == 0) or cmp::AlmostZero_host(species.charge())) { - continue; - } + const auto& layout = species.tile_layout(); + if (layout.ntiles_total == 0u or layout.tile_offsets.extent(0) == 0u) { logger::Checkpoint( - fmt::format( - "Launching currents deposit (flat fallback, no sort yet) " - "for %d [%s] : %lu %f", - species.index(), - species.label().c_str(), - species.npart(), - (double)species.charge()), + fmt::format("Launching currents deposit (flat, no sort yet) for " + "%d [%s] : %lu %f", + species.index(), + species.label().c_str(), + species.npart(), + (double)species.charge()), HERE); + auto scatter_cur = Kokkos::Experimental::create_scatter_view( + domain.fields.cur); CallDepositKernel(species, domain.mesh.metric, scatter_cur, dt); - } - Kokkos::Experimental::contribute(domain.fields.cur, scatter_cur); - } else { - for (auto& species : domain.species) { - if ((species.pusher() == ParticlePusher::NONE) or - (species.npart() == 0) or cmp::AlmostZero_host(species.charge())) { - continue; - } + Kokkos::Experimental::contribute(domain.fields.cur, scatter_cur); + } else { logger::Checkpoint( fmt::format("Launching tiled currents deposit for %d [%s] : %lu %f", species.index(), @@ -150,7 +169,6 @@ namespace ntt { species.npart(), (double)species.charge()), HERE); - CallDepositKernelTiled(species, domain.mesh.metric, domain.fields.cur, diff --git a/src/framework/containers/particles.h b/src/framework/containers/particles.h index 4f0770729..0877efd15 100644 --- a/src/framework/containers/particles.h +++ b/src/framework/containers/particles.h @@ -299,11 +299,14 @@ namespace ntt { private: /** * @brief Apply a particle-index permutation (built by oneDPL/Thrust - * sort_by_key) to every SoA member array. Sequential — one - * transient buffer at a time, fenced before scope exit. - * Only compiled when a vendor sort backend is enabled; the - * BinSort path applies the permutation in place via - * `sorter.sort(view)` instead. + * sort_by_key) to the SoA member arrays. Each member is + * gathered into a fresh full-capacity buffer whose handle is + * then swapped in (no copy-back), one buffer at a time, fenced + * before the old storage is released. The *_prev arrays are + * intentionally not permuted (overwritten by the next push + * before any read). Only compiled when a vendor sort backend + * is enabled; the BinSort path applies the permutation in + * place via `sorter.sort(view)` instead. */ void apply_permutation_to_soa(const prtl_perm_t& perm); diff --git a/src/framework/containers/particles_sort.cpp b/src/framework/containers/particles_sort.cpp index 0317e0b5f..a52b013f7 100644 --- a/src/framework/containers/particles_sort.cpp +++ b/src/framework/containers/particles_sort.cpp @@ -264,8 +264,11 @@ namespace ntt { const auto slice = prtl_slice_t(0, npart_local); #if defined(TEAM_POLICY_USE_VENDOR_SORT) // Vendor path: produce an explicit permutation via sort_by_key, - // then apply it to each SoA member with a sequential one-buffer - // gather (peak transient = one `npart × sizeof(member)` buffer. + // then apply it to each SoA member by gathering into a fresh + // full-capacity buffer and swapping the View handle in (no + // copy-back). The *_prev arrays are skipped — see + // apply_permutation_to_soa. Peak transient = one + // `maxnpart × sizeof(member)` buffer at a time. prtl_perm_t perm { "tile_perm", npart_local }; #if defined(SYCL_ENABLED) && defined(ONEDPL_ENABLED) sort_helpers::sort_by_key_dispatch(tile_indices, @@ -376,6 +379,11 @@ namespace ntt { Kokkos::deep_copy(tile_offsets, h_offsets); m_tile_layout.tile_offsets = tile_offsets; + // tile_offsets(total_tiles) is the alive-particle count at sort time: + // the tiles partition exactly [0, npart_partitioned). The deposit + // launcher compares this against the live npart() to detect (and + // separately deposit) particles appended since this sort. + m_tile_layout.npart_partitioned = h_offsets(total_tiles); } // 6. Populate `m_tile_layout` size/shape. `tile_perm` is not used @@ -454,38 +462,43 @@ namespace ntt { #if defined(TEAM_POLICY_USE_VENDOR_SORT) namespace permute_helpers { - // Permute a 1D SoA member array `arr` in place by `perm`, using a - // single transient buffer of size `n`. Buffer is freed at scope - // exit; the explicit fence right before that drains queued GPU - // work referencing it. + // Permute a 1D SoA member array `arr` by `perm`. Gathers into a + // fresh buffer allocated at the member's full capacity (maxnpart), + // then swaps the View handle in. This avoids the redundant copy-back + // pass of the old gather-then-deep_copy approach (~2x less HBM + // traffic). Allocating at full capacity preserves the member's spare + // room for injection; the untouched tail [n, capacity) is + // zero-initialized by Kokkos (cleaner than the stale values the old + // deep_copy left there). The fence drains the gather (which reads the + // old storage) before the swap drops the last reference to it. template - inline void permute_1d_inplace(V& arr, - const prtl_perm_t& perm, - npart_t n) { + inline void permute_1d_swap(V& arr, + const prtl_perm_t& perm, + npart_t n) { if (n == 0u) { return; } - V buf(std::string(arr.label()) + "_perm_buf", n); + V buf(arr.label(), arr.extent(0)); auto perm_v = perm; auto arr_v = arr; Kokkos::parallel_for( "Permute1D", n, KOKKOS_LAMBDA(const npart_t p) { buf(p) = arr_v(perm_v(p)); }); - Kokkos::deep_copy(Kokkos::subview(arr, prtl_slice_t(0u, n)), buf); - Kokkos::fence("permute_1d_inplace: end"); + Kokkos::fence("permute_1d_swap: end"); + arr = buf; } // 2D analogue for `pld_r` / `pld_i`. template - inline void permute_2d_inplace(V& arr, - const prtl_perm_t& perm, - npart_t n, - npart_t ncols) { + inline void permute_2d_swap(V& arr, + const prtl_perm_t& perm, + npart_t n, + npart_t ncols) { if (n == 0u or ncols == 0u) { return; } - V buf(std::string(arr.label()) + "_perm_buf", n, ncols); + V buf(arr.label(), arr.extent(0), arr.extent(1)); auto perm_v = perm; auto arr_v = arr; Kokkos::parallel_for( @@ -494,9 +507,8 @@ namespace ntt { KOKKOS_LAMBDA(const npart_t p, const npart_t l) { buf(p, l) = arr_v(perm_v(p), l); }); - Kokkos::deep_copy(Kokkos::subview(arr, prtl_slice_t(0u, n), Kokkos::ALL), - buf); - Kokkos::fence("permute_2d_inplace: end"); + Kokkos::fence("permute_2d_swap: end"); + arr = buf; } } // namespace permute_helpers @@ -508,40 +520,50 @@ namespace ntt { return; } - using permute_helpers::permute_1d_inplace; - using permute_helpers::permute_2d_inplace; - + using permute_helpers::permute_1d_swap; + using permute_helpers::permute_2d_swap; + + // The *_prev arrays (i{1,2,3}_prev, dx{1,2,3}_prev) are intentionally + // NOT permuted. SortSpatially runs at the very end of the step loop + // (engine step_forward), and the first thing the next step's pusher + // does is overwrite prev := current (positionPush, sr.hpp / gr.hpp) + // for every active particle, before any consumer reads prev: + // - current deposit: runs after the push, which has already + // overwritten prev; species with pusher==NONE (whose prev would + // stay un-permuted) are skipped by CurrentsDeposit entirely. + // - pusher getParticlePrevPosition / piston: read prev only after + // positionPush has rewritten it within the same call. + // - checkpoint (prev is checkpoint-only, never in diagnostic + // output): on restart the first push overwrites prev before it + // is read, so restart results are unaffected; only the redundant + // prev field saved to the checkpoint differs from the old code. + // Permuting prev would therefore reorder data that is overwritten + // before it is ever observed. if constexpr (D == Dim::_1D or D == Dim::_2D or D == Dim::_3D) { - permute_1d_inplace(i1, perm, n); - permute_1d_inplace(dx1, perm, n); - permute_1d_inplace(i1_prev, perm, n); - permute_1d_inplace(dx1_prev, perm, n); + permute_1d_swap(i1, perm, n); + permute_1d_swap(dx1, perm, n); } if constexpr (D == Dim::_2D or D == Dim::_3D) { - permute_1d_inplace(i2, perm, n); - permute_1d_inplace(dx2, perm, n); - permute_1d_inplace(i2_prev, perm, n); - permute_1d_inplace(dx2_prev, perm, n); + permute_1d_swap(i2, perm, n); + permute_1d_swap(dx2, perm, n); } if constexpr (D == Dim::_3D) { - permute_1d_inplace(i3, perm, n); - permute_1d_inplace(dx3, perm, n); - permute_1d_inplace(i3_prev, perm, n); - permute_1d_inplace(dx3_prev, perm, n); - } - permute_1d_inplace(ux1, perm, n); - permute_1d_inplace(ux2, perm, n); - permute_1d_inplace(ux3, perm, n); - permute_1d_inplace(weight, perm, n); - permute_1d_inplace(tag, perm, n); + permute_1d_swap(i3, perm, n); + permute_1d_swap(dx3, perm, n); + } + permute_1d_swap(ux1, perm, n); + permute_1d_swap(ux2, perm, n); + permute_1d_swap(ux3, perm, n); + permute_1d_swap(weight, perm, n); + permute_1d_swap(tag, perm, n); if constexpr (D == Dim::_2D and C != Coord::Cartesian) { - permute_1d_inplace(phi, perm, n); + permute_1d_swap(phi, perm, n); } if (npld_r() > 0) { - permute_2d_inplace(pld_r, perm, n, static_cast(npld_r())); + permute_2d_swap(pld_r, perm, n, static_cast(npld_r())); } if (npld_i() > 0) { - permute_2d_inplace(pld_i, perm, n, static_cast(npld_i())); + permute_2d_swap(pld_i, perm, n, static_cast(npld_i())); } } #endif // TEAM_POLICY_USE_VENDOR_SORT diff --git a/src/framework/parameters/parameters.cpp b/src/framework/parameters/parameters.cpp index 7372da510..4116783a1 100644 --- a/src/framework/parameters/parameters.cpp +++ b/src/framework/parameters/parameters.cpp @@ -67,11 +67,18 @@ namespace ntt { "clear_interval", defaults::clear_interval); set("particles.clear_interval", global_clearing_interval); +#if defined(TEAM_POLICY_SORT_INTERVAL) + // See particles.cpp: the compile-time team_policy_sort_interval overrides + // the runtime value (kept consistent here for the stored global param). + const auto global_spatial_sorting_interval = static_cast( + TEAM_POLICY_SORT_INTERVAL); +#else const auto global_spatial_sorting_interval = toml::find_or( toml_data, "particles", "spatial_sorting_interval", 0u); +#endif set("particles.spatial_sorting_interval", global_spatial_sorting_interval); set("scales.n0", ppc0 / get("scales.V0")); diff --git a/src/framework/parameters/particles.cpp b/src/framework/parameters/particles.cpp index 6f1a23c03..46192035b 100644 --- a/src/framework/parameters/particles.cpp +++ b/src/framework/parameters/particles.cpp @@ -121,10 +121,19 @@ namespace ntt { sp, "clear_interval", global_clearing_interval); +#if defined(TEAM_POLICY_SORT_INTERVAL) + // Compile-time hardwired sort interval (the `team_policy_sort_interval` + // CMake knob). It overrides whatever the input file requested so the + // tiled deposit's scratch halo — sized for exactly this cadence — is + // guaranteed to contain every particle's drift between sorts. + const auto spatial_sorting_interval = static_cast( + TEAM_POLICY_SORT_INTERVAL); +#else const auto spatial_sorting_interval = toml::find_or( sp, "spatial_sorting_interval", global_spatial_sorting_interval); +#endif auto pusher_str = toml::find_or(sp, "pusher", std::string(def_pusher)); const auto npayloads_real = toml::find_or(sp, "n_payloads_real", diff --git a/src/global/arch/kokkos_aliases.h b/src/global/arch/kokkos_aliases.h index 4fe88cdf7..7212b554e 100644 --- a/src/global/arch/kokkos_aliases.h +++ b/src/global/arch/kokkos_aliases.h @@ -340,20 +340,27 @@ auto CreateRangePolicyOnHost(const tuple_t&, using prtl_perm_t = array_t; // Tile layout metadata: the contract between Stream 1 (sort) and Streams -// 2/3 (tiled deposit / pusher). All members are device-resident. -// ntiles_per_axis : number of tiles along each axis (1 for unused axes). -// ntiles_total : product of ntiles_per_axis = league size for TeamPolicy. -// tile_size : tile edge length in cells (compile-time CMake knob, -// replicated here for runtime checks). -// tile_offsets : prefix-sum of per-tile particle counts; size -// ntiles_total + 1; tile t owns particles -// [tile_offsets(t), tile_offsets(t+1)). -// tile_perm : size npart, particle index sorted by tile. +// 2/3 (tiled deposit / pusher). Scalars are host-resident; the views are +// device-resident. +// ntiles_per_axis : number of tiles along each axis (1 for unused axes). +// ntiles_total : product of ntiles_per_axis = league size for TeamPolicy. +// tile_size : tile edge length in cells (compile-time CMake knob, +// replicated here for runtime checks). +// npart_partitioned: number of (alive) particles partitioned at the last +// sort, i.e. tile_offsets(ntiles_total). The tiles cover +// exactly [0, npart_partitioned); particles appended past +// it (injection / MPI receive on a no-sort step) are not +// partitioned and must be deposited separately. +// tile_offsets : prefix-sum of per-tile particle counts; size +// ntiles_total + 1; tile t owns particles +// [tile_offsets(t), tile_offsets(t+1)). +// tile_perm : size npart, particle index sorted by tile. template struct TileLayout { ncells_t ntiles_per_axis[3] { 1u, 1u, 1u }; ncells_t ntiles_total { 0u }; unsigned short tile_size { 0u }; + npart_t npart_partitioned { 0u }; array_t tile_offsets; prtl_perm_t tile_perm; }; diff --git a/src/kernels/currents_deposit.hpp b/src/kernels/currents_deposit.hpp index 79252621a..8840a1046 100644 --- a/src/kernels/currents_deposit.hpp +++ b/src/kernels/currents_deposit.hpp @@ -780,28 +780,37 @@ namespace kernel { * `SortSpatially` (`particles_sort.cpp`) is responsible for keeping * the SoA arrays consistent with that. * - * **Halo sizing and escape valve.** Sort runs at the end of the - * previous step (see `srpic.hpp`), so at deposit time the particle - * has already been pushed once — its `min(i, i_prev)` may differ - * from the bin key by one cell of drift per step elapsed since the - * last sort. The scratch HALO is `STENCIL_REACH(O) + DRIFT`, where - * `STENCIL_REACH = 2` for zigzag (writes `{i_prev, i_prev+1, i, - * i+1}` ⇒ +2 above `min(i, i_prev)` with `|Δi|=1`) and `O` for - * Esirkepov, and `DRIFT` is a fixed constant (1) covering the one - * guaranteed post-sort pusher step. + * **Halo sizing and escape valve.** Sort runs at the end of a step + * (see `srpic.hpp`); a particle is pushed once per step thereafter, so + * its `min(i, i_prev)` may differ from the bin key by one cell of drift + * per step elapsed since the last sort. The scratch HALO is + * `STENCIL_REACH(O) + DRIFT`, where `STENCIL_REACH = 2` for zigzag + * (writes `{i_prev, i_prev+1, i, i+1}` ⇒ +2 above `min(i, i_prev)` with + * `|Δi|=1`) and `O` for Esirkepov. `DRIFT` is the + * `team_policy_sort_interval` CMake knob (macro TEAM_POLICY_SORT_INTERVAL) + * when set — the hardwired sort interval, hence the maximum drift any + * particle accrues between sorts — and `1` otherwise (the + * every-step-sorted common case). * - * HALO is sized for the *common* (every-step-sorted) case, not for - * a worst-case sort cadence: correctness does **not** depend on it. - * Any particle whose stencil escapes the scratch tile — because it - * drifted further than `DRIFT` (e.g. a large runtime - * `spatial_sorting_interval`), or because the halo is otherwise - * undersized — silently falls back to a direct, bounds-clipped - * `Kokkos::atomic_add` on the global J view. That path is - * charge-conserving (each particle's stencil is deposited exactly - * once, partly to private SLM scratch and partly to global J, and - * scratch is flushed once via `atomic_add`); it is merely slower - * per write. Sorting less often than every step therefore costs + * Correctness does **not** depend on the halo size. Any particle whose + * full stencil escapes the scratch tile — because it drifted further + * than `DRIFT`, was reordered far from its tile by a no-sort-step + * `CommunicateParticles`, or because the halo is otherwise undersized — + * is deposited *as a whole* via a direct, bounds-clipped + * `Kokkos::atomic_add` on the global J view (the per-particle escape + * valve). Each particle's stencil is therefore deposited exactly once + * (entirely to SLM scratch when it fits, entirely to global J when it + * does not), so the path is charge-conserving; it is merely slower per + * write. Sizing `DRIFT` to the sort interval keeps the common, + * within-interval drift in fast SLM; sorting less often only costs * escape-valve traffic, never accuracy. + * + * **Partition coverage.** The team iteration covers only the particles + * partitioned at the last sort, `[0, layout.npart_partitioned)`, clamped + * to the live `npart`. Particles appended past the partition since the + * sort are not seen here; the launcher (`engines/srpic/currents.h`) + * deposits that tail with the flat kernel so every active particle is + * covered exactly once regardless of sort cadence. */ template class DepositCurrentsTiled_kernel { @@ -825,21 +834,33 @@ namespace kernel { * old `O+1` carried one extra cell of conservative padding * on top of the already-conservative drift term below. * - * drift — sort runs at end-of-step (see srpic.hpp), so a particle - * sees exactly one pusher step before the *next* step's deposit - * when sorted every step (the common case). DRIFT is therefore a - * fixed constant of 1, NOT a compile-time function of the runtime - * sort cadence. Sizing the halo for the common case (rather than a - * worst-case sort interval) is what keeps the scratch small enough - * for good occupancy; a species sorted less often than - * every step just drifts past the halo and takes the global-J - * escape valve more often — correct, only slower (see the class - * doc-comment for why this is charge-conserving). + * drift — sort runs at end-of-step (see srpic.hpp), so a particle is + * pushed once per step between its last sort and a given deposit. With + * a sort interval of `K`, a particle therefore drifts at most `K` cells + * (CFL |v dt/dx| <= 1/2 ⇒ |Δi| <= 1 per step) before the next sort. The + * `team_policy_sort_interval` CMake knob (macro TEAM_POLICY_SORT_INTERVAL) + * pins that interval at compile time and feeds it here as DRIFT, sizing + * the halo so a fully-interval-drifted particle still deposits inside + * its tile scratch. When the knob is unset, DRIFT defaults to 1 (the + * sorted-every-step common case); any particle that drifts past the halo + * (e.g. a larger runtime interval, or a CFL excursion) takes the + * per-particle global-J escape valve below — correct, only slower (see + * the class doc-comment for why this is charge-conserving). */ - static constexpr int STENCIL_REACH = (O == 0u) ? 2 : static_cast(O); - static constexpr int DRIFT = 1; - static constexpr int HALO = STENCIL_REACH + DRIFT; - static constexpr int TE = static_cast(T_TILE) + 2 * HALO; + static constexpr int STENCIL_REACH = (O == 0u) ? 2 : static_cast(O); + // One-sided footprint reach for the per-particle escape valve: the + // deposit writes at most this many cells above max(i,i_prev) (and fewer + // below min), so [min - FOOTPRINT_REACH, max + FOOTPRINT_REACH] in cell + // coords conservatively bounds every deposited cell for any order + // (Esirkepov reaches max+O; O=0 zigzag reaches max+1). + static constexpr int FOOTPRINT_REACH = (O == 0u) ? 1 : static_cast(O); +#if defined(TEAM_POLICY_SORT_INTERVAL) + static constexpr int DRIFT = static_cast(TEAM_POLICY_SORT_INTERVAL); +#else + static constexpr int DRIFT = 1; +#endif + static constexpr int HALO = STENCIL_REACH + DRIFT; + static constexpr int TE = static_cast(T_TILE) + 2 * HALO; using exec_space = Kokkos::DefaultExecutionSpace; using team_policy = Kokkos::TeamPolicy; @@ -855,6 +876,17 @@ namespace kernel { ncells_t ntx1 { 1u }, ntx2 { 1u }, ntx3 { 1u }; ncells_t total_tiles { 0u }; + /** + * Current active-particle count. `tile_offsets` partitions only the + * particles that existed at the last sort ([0, layout.npart_partitioned)); + * `npart` may differ if the pusher dead-tagged particles in place since. + * Each team clamps its `[tile_offsets(t), tile_offsets(t+1))` slice to + * `npart` so stale slots past the live array are never read. Particles + * appended *beyond* the partition (npart > npart_partitioned) are not seen + * by any team here — the launcher deposits that tail separately. + */ + npart_t npart { 0u }; + /** * J's full storage extent including all ghost cells. Used to clip * the cooperative flush so that a partial tile at the high end of @@ -868,7 +900,8 @@ namespace kernel { const M& metric, real_t charge, real_t dt, - const TileLayout& layout) + const TileLayout& layout, + npart_t npart) : J { cur } , prtls { prtls } , metric { metric } @@ -878,7 +911,8 @@ namespace kernel { , ntx1 { layout.ntiles_per_axis[0] } , ntx2 { layout.ntiles_per_axis[1] } , ntx3 { layout.ntiles_per_axis[2] } - , total_tiles { layout.ntiles_total } { + , total_tiles { layout.ntiles_total } + , npart { npart } { raise::ErrorIf( layout.tile_size != T_TILE, "Tiled deposit launched with mismatched T_TILE and runtime tile_size", @@ -907,12 +941,17 @@ namespace kernel { * `team_policy.set_scratch_size(0, Kokkos::PerTeam(bytes))`. */ static constexpr size_t scratch_bytes() { + // The component count (3) is a *static* extent of scratch_ndfield_t + // (View / **[3] / ***[3]), so shmem_size() takes only the + // dynamic spatial extents — passing 3 as well trips Kokkos' + // `rank_dynamic != number of arguments` abort. This matches the + // scratch View construction below, which also omits the 3. if constexpr (D == Dim::_1D) { - return scratch_ndfield_t::shmem_size(TE, 3); + return scratch_ndfield_t::shmem_size(TE); } else if constexpr (D == Dim::_2D) { - return scratch_ndfield_t::shmem_size(TE, TE, 3); + return scratch_ndfield_t::shmem_size(TE, TE); } else { - return scratch_ndfield_t::shmem_size(TE, TE, TE, 3); + return scratch_ndfield_t::shmem_size(TE, TE, TE); } } @@ -961,31 +1000,49 @@ namespace kernel { }); team.team_barrier(); - const auto p_begin = tile_offsets(tile_id); - const auto p_end = tile_offsets(tile_id + 1u); + // Clamp the tile's particle slice to the live array: slots past + // `npart` may hold stale (possibly alive-tagged) data from a prior + // step's compaction and must not be re-deposited. + const auto t_lo = tile_offsets(tile_id); + const auto t_hi = tile_offsets(tile_id + 1u); + const auto p_begin = (t_lo < npart) ? t_lo : npart; + const auto p_end = (t_hi < npart) ? t_hi : npart; Kokkos::parallel_for( Kokkos::TeamThreadRange(team, p_begin, p_end), [&](prtlidx_t p) { + /** + * Per-particle escape valve: route the WHOLE particle to the + * global J view when its Esirkepov footprint does not fit + * inside this tile's scratch window [0,TE); only particles + * fully inside the tile touch SLM scratch. A particle drifts + * out of its tile when sorted less often than every step. + * + * The conservative footprint bound in cell coords, + * [min(i,i_prev) - O, max(i,i_prev) + O], covers + * prtl_shape::for_deposit for any order (i_min >= + * min-floor(O/2), i_max <= max+O), so when `to_scratch` is true + * every deposited cell is provably in [0,TE) and the scratch write + * needs no per-cell bounds test. The global path bounds-clips + * against J's storage extent (writes past the ghost stripe are + * re-supplied by SynchronizeFields(J)). + */ + const int i1c = prtls.i1(p), i1p = prtls.i1_prev(p); + const int lo1 = (i1c < i1p ? i1c : i1p) + static_cast(N_GHOSTS) - + FOOTPRINT_REACH - origin_J1_low; + const int hi1 = (i1c > i1p ? i1c : i1p) + static_cast(N_GHOSTS) + + FOOTPRINT_REACH - origin_J1_low; + const bool to_scratch = (lo1 >= 0 and hi1 < TE); DepositOneParticle( p, prtls, metric, charge, inv_dt, - /** - * Escape valve: a particle whose stencil reaches past the - * tile's scratch (e.g. exceeded the compile-time - * STENCIL_REACH + DRIFT budget) falls back to a direct - * atomic_add on the global J view. Bounds-clipped against - * J's storage extent so writes past the domain ghost stripe - * are dropped (matches the cooperative flush below; those - * contributions are re-supplied by SynchronizeFields(J)). - */ [&](int g_i1, int comp, real_t v) { - const int li = g_i1 - origin_J1_low; - if (li >= 0 and li < TE) { - Kokkos::atomic_add(&scr(li, comp), v); - } else if (g_i1 >= 0 and g_i1 < j_ext1) { + if (to_scratch) { + Kokkos::atomic_add(&scr(g_i1 - origin_J1_low, comp), v); + //} else if (g_i1 >= 0 and g_i1 < j_ext1) { + } else { Kokkos::atomic_add(&J(g_i1, comp), v); } }); @@ -1022,25 +1079,42 @@ namespace kernel { }); team.team_barrier(); - const auto p_begin = tile_offsets(tile_id); - const auto p_end = tile_offsets(tile_id + 1u); + // Clamp the tile's particle slice to the live array: slots past + // `npart` may hold stale (possibly alive-tagged) data from a prior + // step's compaction and must not be re-deposited. + const auto t_lo = tile_offsets(tile_id); + const auto t_hi = tile_offsets(tile_id + 1u); + const auto p_begin = (t_lo < npart) ? t_lo : npart; + const auto p_end = (t_hi < npart) ? t_hi : npart; Kokkos::parallel_for( Kokkos::TeamThreadRange(team, p_begin, p_end), [&](prtlidx_t p) { + // See 1D branch for rationale: route the whole particle to the + // global escape valve unless its full footprint fits in scratch. + const int i1c = prtls.i1(p), i1p = prtls.i1_prev(p); + const int i2c = prtls.i2(p), i2p = prtls.i2_prev(p); + const int lo1 = (i1c < i1p ? i1c : i1p) + static_cast(N_GHOSTS) - + FOOTPRINT_REACH - origin_J1_low; + const int hi1 = (i1c > i1p ? i1c : i1p) + static_cast(N_GHOSTS) + + FOOTPRINT_REACH - origin_J1_low; + const int lo2 = (i2c < i2p ? i2c : i2p) + static_cast(N_GHOSTS) - + FOOTPRINT_REACH - origin_J2_low; + const int hi2 = (i2c > i2p ? i2c : i2p) + static_cast(N_GHOSTS) + + FOOTPRINT_REACH - origin_J2_low; + const bool to_scratch = (lo1 >= 0 and hi1 < TE and lo2 >= 0 and + hi2 < TE); DepositOneParticle( p, prtls, metric, charge, inv_dt, - // See 1D branch for rationale. [&](const int g_i1, const int g_i2, int comp, real_t v) { - const auto li = g_i1 - origin_J1_low; - const auto lj = g_i2 - origin_J2_low; - if ((li >= 0 and li < TE) and (lj >= 0 and lj < TE)) { - Kokkos::atomic_add(&scr(li, lj, comp), v); - } else if ((g_i1 >= 0 and g_i1 < j_ext1) and - (g_i2 >= 0 and g_i2 < j_ext2)) { + if (to_scratch) { + Kokkos::atomic_add( + &scr(g_i1 - origin_J1_low, g_i2 - origin_J2_low, comp), + v); + } else { Kokkos::atomic_add(&J(g_i1, g_i2, comp), v); } }); @@ -1078,28 +1152,49 @@ namespace kernel { }); team.team_barrier(); - const auto p_begin = tile_offsets(tile_id); - const auto p_end = tile_offsets(tile_id + 1u); + // Clamp the tile's particle slice to the live array: slots past + // `npart` may hold stale (possibly alive-tagged) data from a prior + // step's compaction and must not be re-deposited. + const auto t_lo = tile_offsets(tile_id); + const auto t_hi = tile_offsets(tile_id + 1u); + const auto p_begin = (t_lo < npart) ? t_lo : npart; + const auto p_end = (t_hi < npart) ? t_hi : npart; Kokkos::parallel_for( Kokkos::TeamThreadRange(team, p_begin, p_end), [&](prtlidx_t p) { + // See 1D branch for rationale: route the whole particle to the + // global escape valve unless its full footprint fits in scratch. + const int i1c = prtls.i1(p), i1p = prtls.i1_prev(p); + const int i2c = prtls.i2(p), i2p = prtls.i2_prev(p); + const int i3c = prtls.i3(p), i3p = prtls.i3_prev(p); + const int lo1 = (i1c < i1p ? i1c : i1p) + static_cast(N_GHOSTS) - + FOOTPRINT_REACH - origin_J1_low; + const int hi1 = (i1c > i1p ? i1c : i1p) + static_cast(N_GHOSTS) + + FOOTPRINT_REACH - origin_J1_low; + const int lo2 = (i2c < i2p ? i2c : i2p) + static_cast(N_GHOSTS) - + FOOTPRINT_REACH - origin_J2_low; + const int hi2 = (i2c > i2p ? i2c : i2p) + static_cast(N_GHOSTS) + + FOOTPRINT_REACH - origin_J2_low; + const int lo3 = (i3c < i3p ? i3c : i3p) + static_cast(N_GHOSTS) - + FOOTPRINT_REACH - origin_J3_low; + const int hi3 = (i3c > i3p ? i3c : i3p) + static_cast(N_GHOSTS) + + FOOTPRINT_REACH - origin_J3_low; + const bool to_scratch = (lo1 >= 0 and hi1 < TE and lo2 >= 0 and + hi2 < TE and lo3 >= 0 and hi3 < TE); DepositOneParticle( p, prtls, metric, charge, inv_dt, - // See 1D branch for rationale. [&](const int g_i1, const int g_i2, const int g_i3, int comp, real_t v) { - const auto li = g_i1 - origin_J1_low; - const auto lj = g_i2 - origin_J2_low; - const auto lk = g_i3 - origin_J3_low; - if ((li >= 0 and li < TE) and (lj >= 0 and lj < TE) and - (lk >= 0 and lk < TE)) { - Kokkos::atomic_add(&scr(li, lj, lk, comp), v); - } else if ((g_i1 >= 0 and g_i1 < j_ext1) and - (g_i2 >= 0 and g_i2 < j_ext2) and - (g_i3 >= 0 and g_i3 < j_ext3)) { + if (to_scratch) { + Kokkos::atomic_add(&scr(g_i1 - origin_J1_low, + g_i2 - origin_J2_low, + g_i3 - origin_J3_low, + comp), + v); + } else { Kokkos::atomic_add(&J(g_i1, g_i2, g_i3, comp), v); } }); diff --git a/tests/framework/particles_sort.cpp b/tests/framework/particles_sort.cpp index 1ba10d828..6945c962f 100644 --- a/tests/framework/particles_sort.cpp +++ b/tests/framework/particles_sort.cpp @@ -82,7 +82,11 @@ auto main(int argc, char* argv[]) -> int { Kokkos::deep_copy(pld_r_h, prtls.pld_r); Kokkos::deep_copy(pld_i_h, prtls.pld_i); - for (auto p { 0u }; p < 75u; ++p) { + // Only [0, npart) is defined after a sort. The swap-based gather in + // apply_permutation_to_soa replaces each SoA View, zero-filling the + // spare capacity [npart, maxnpart) (a don't-care region overwritten + // by injection), so the old "tail preserved" check no longer holds. + for (auto p { 0u }; p < 66u; ++p) { if (p < 16u) { raise::ErrorIf(weight_h(p) != 3.0, "error in sorting particles", HERE); } else if (p < 33u) { @@ -180,7 +184,11 @@ auto main(int argc, char* argv[]) -> int { auto weight_h = Kokkos::create_mirror_view(prtls.weight); Kokkos::deep_copy(weight_h, prtls.weight); - for (auto p { 0u }; p < 75u; ++p) { + // Only [0, npart) is defined after a sort. The swap-based gather in + // apply_permutation_to_soa replaces each SoA View, zero-filling the + // spare capacity [npart, maxnpart) (a don't-care region overwritten + // by injection), so the old "tail preserved" check no longer holds. + for (auto p { 0u }; p < 66u; ++p) { if (p < 13u) { raise::ErrorIf(weight_h(p) != 4.0, "error in sorting particles", HERE); } else if (p < 26u) { From 4f4409d507b942e9a7a15cc3c6a3abb69af07dd7 Mon Sep 17 00:00:00 2001 From: LudwigBoess Date: Fri, 19 Jun 2026 20:13:50 +0000 Subject: [PATCH 011/125] fix test for tiled deposit --- tests/kernels/deposit_tiled.cpp | 289 ++++++++++++++++++++++++++++++-- 1 file changed, 273 insertions(+), 16 deletions(-) diff --git a/tests/kernels/deposit_tiled.cpp b/tests/kernels/deposit_tiled.cpp index 504cc3818..3cc2f62b0 100644 --- a/tests/kernels/deposit_tiled.cpp +++ b/tests/kernels/deposit_tiled.cpp @@ -3,7 +3,7 @@ * @brief X-1 numerical-equivalence test for the tiled deposit kernel. * * Runs the flat (`DepositCurrents_kernel`) and tiled - * (`DepositCurrents_kernel_tiled`) kernels on identical particle SoA inputs + * (`DepositCurrentsTiled_kernel`) kernels on identical particle SoA inputs * for shape orders O = 1..11 and asserts that the resulting J array is * identical cell-by-cell within a small floating-point tolerance. * @@ -48,6 +48,37 @@ namespace { Kokkos::deep_copy(arr, h); } + // Pack the per-test SoA arrays into a ParticleArrays — the struct both + // deposit kernels take. Payload (pld_*) members stay default; these are + // 2D Cartesian Minkowski cases, so phi/i3/dx3 are present but unread. + ParticleArrays pack_arrays(const array_t& i1, + const array_t& i2, + const array_t& i3, + const array_t& i1_prev, + const array_t& i2_prev, + const array_t& i3_prev, + const array_t& dx1, + const array_t& dx2, + const array_t& dx3, + const array_t& dx1_prev, + const array_t& dx2_prev, + const array_t& dx3_prev, + const array_t& ux1, + const array_t& ux2, + const array_t& ux3, + const array_t& phi, + const array_t& weight, + const array_t& tag) { + ParticleArrays pa; + pa.i1 = i1, pa.i2 = i2, pa.i3 = i3; + pa.i1_prev = i1_prev, pa.i2_prev = i2_prev, pa.i3_prev = i3_prev; + pa.dx1 = dx1, pa.dx2 = dx2, pa.dx3 = dx3; + pa.dx1_prev = dx1_prev, pa.dx2_prev = dx2_prev, pa.dx3_prev = dx3_prev; + pa.ux1 = ux1, pa.ux2 = ux2, pa.ux3 = ux3; + pa.phi = phi, pa.weight = weight, pa.tag = tag; + return pa; + } + // Builds tile_offsets for a single-particle test. Particle 0 is alive // and lives in tile (tx1, tx2); slots 1..n_slots-1 carry the dead // sentinel and are never referenced by tile_offsets — so the tiled @@ -68,6 +99,69 @@ namespace { return offsets; } + // Buckets ALL `n_alive` particles (slots 0..n_alive-1) into tile 0, + // modelling a maximally-stale tile layout: every particle was "sorted" + // into tile 0 but now sits anywhere in the domain (as happens when the + // SoA drifts / is reordered between sorts). Every particle except the + // few that genuinely live near the origin must therefore take the + // per-particle escape valve to global J. + array_t build_tile_offsets_all_in_tile0(ncells_t total_tiles, + npart_t n_alive) { + array_t offsets("tile_offsets", total_tiles + 1u); + auto h = Kokkos::create_mirror_view(offsets); + h(0) = static_cast(0); + for (ncells_t t = 1; t <= total_tiles; ++t) { + h(t) = n_alive; + } + Kokkos::deep_copy(offsets, h); + return offsets; + } + + // Cell-by-cell comparison of two J fields; throws on mismatch. + void compare_J_fields(const ndfield_t& J_flat, + const ndfield_t& J_tiled, + unsigned short O, + unsigned short T_TILE, + const char* label) { + auto h_flat = Kokkos::create_mirror_view(J_flat); + auto h_tiled = Kokkos::create_mirror_view(J_tiled); + Kokkos::deep_copy(h_flat, J_flat); + Kokkos::deep_copy(h_tiled, J_tiled); + + const real_t eps = static_cast(1.0e-5); + real_t max_diff = ZERO; + int fail_count = 0; + for (ncells_t i = 0; i < h_flat.extent(0); ++i) { + for (ncells_t j = 0; j < h_flat.extent(1); ++j) { + for (int c = 0; c < 3; ++c) { + const real_t a = h_flat(i, j, c); + const real_t b = h_tiled(i, j, c); + const real_t diff = math::fabs(a - b); + const real_t mag = math::max(math::fabs(a), math::fabs(b)); + if (diff > max_diff) { + max_diff = diff; + } + if (diff > eps * math::max(mag, static_cast(1.0))) { + if (fail_count < 5) { + std::cerr << " [" << label << "] J(" << i << "," << j + << ",c=" << c << ") flat=" << a << " tiled=" << b + << " diff=" << diff << '\n'; + } + ++fail_count; + } + } + } + } + if (fail_count > 0) { + std::cerr << "deposit_tiled[" << label << "] FAILED for O=" << O + << " T_TILE=" << T_TILE << " : " << fail_count + << " mismatches; max_diff=" << max_diff << '\n'; + throw std::logic_error("DepositCurrentsTiled_kernel mismatch"); + } + std::cerr << "deposit_tiled[" << label << "] OK O=" << O + << " T_TILE=" << T_TILE << " max_diff=" << max_diff << '\n'; + } + template void run_one_case() { using metric_t = metric::Minkowski; @@ -129,12 +223,12 @@ namespace { 10, kernel::DepositCurrents_kernel( J_scat, - i1, i2, i3, - i1_prev, i2_prev, i3_prev, - dx1, dx2, dx3, - dx1_prev, dx2_prev, dx3_prev, - ux1, ux2, ux3, - phi, weight, tag, + pack_arrays(i1, i2, i3, + i1_prev, i2_prev, i3_prev, + dx1, dx2, dx3, + dx1_prev, dx2_prev, dx3_prev, + ux1, ux2, ux3, + phi, weight, tag), metric, charge, dt)); Kokkos::Experimental::contribute(J_flat, J_scat); Kokkos::fence("flat deposit done"); @@ -168,15 +262,18 @@ namespace { tx2); using kernel_t = - kernel::DepositCurrents_kernel_tiled; + kernel::DepositCurrentsTiled_kernel; + // npart = full slot count (10): the lone alive particle sits in slot 0 + // and the per-tile slice clamp keeps the (dead) tail out. kernel_t kern { J_tiled, - i1, i2, i3, - i1_prev, i2_prev, i3_prev, - dx1, dx2, dx3, - dx1_prev, dx2_prev, dx3_prev, - ux1, ux2, ux3, - phi, weight, tag, - metric, charge, dt, layout }; + pack_arrays(i1, i2, i3, + i1_prev, i2_prev, i3_prev, + dx1, dx2, dx3, + dx1_prev, dx2_prev, dx3_prev, + ux1, ux2, ux3, + phi, weight, tag), + metric, charge, dt, layout, + static_cast(10) }; Kokkos::TeamPolicy<> policy(static_cast(layout.ntiles_total), Kokkos::AUTO); @@ -221,12 +318,158 @@ namespace { << " T_TILE=" << T_TILE << " : " << fail_count << " mismatches; max_diff=" << max_diff << '\n'; - throw std::logic_error("DepositCurrents_kernel_tiled mismatch"); + throw std::logic_error("DepositCurrentsTiled_kernel mismatch"); } std::cerr << "X-1 deposit_tiled OK O=" << O << " T_TILE=" << T_TILE << " max_diff=" << max_diff << '\n'; } + // Drift / stale-layout regression test for the per-particle escape valve. + // + // A population of alive particles is spread across the whole domain + // (including the near-boundary cells that deposit into the ghost stripe) + // but the tile layout buckets them ALL into tile 0 — i.e. the layout is + // maximally stale w.r.t. their real positions, exactly the situation that + // arises when the SoA is reordered/appended between sorts. The tiled + // kernel must route every out-of-tile particle to the global J view and + // reproduce the flat deposit cell-for-cell: no charge dropped or + // double-counted at any drift distance. This is the property that broke + // when `spatial_sorting_interval > 1` left drifted particles depositing + // partial stencils, producing a density/E line at the decomposition + // boundary. + template + void run_drift_case() { + using metric_t = metric::Minkowski; + constexpr unsigned short nx1 = 50u, nx2 = 50u; + metric_t metric { { nx1, nx2 }, { { 0.0, 55.0 }, { 0.0, 55.0 } }, {} }; + + constexpr int n_slots = 64; + constexpr int n_base = 5; + const int bases[n_base] = { 1, 13, 25, 37, 48 }; + const int n_alive = n_base * n_base; // 25 + + array_t i1 { "i1", n_slots }, i2 { "i2", n_slots }, + i3 { "i3", n_slots }; + array_t i1_prev { "i1_prev", n_slots }, + i2_prev { "i2_prev", n_slots }, i3_prev { "i3_prev", n_slots }; + array_t dx1 { "dx1", n_slots }, dx2 { "dx2", n_slots }, + dx3 { "dx3", n_slots }; + array_t dx1_prev { "dx1_prev", n_slots }, + dx2_prev { "dx2_prev", n_slots }, dx3_prev { "dx3_prev", n_slots }; + array_t ux1 { "ux1", n_slots }, ux2 { "ux2", n_slots }, + ux3 { "ux3", n_slots }; + array_t phi { "phi", n_slots }, weight { "weight", n_slots }; + array_t tag { "tag", n_slots }; + const real_t charge = 1.0, dt = 1.0; + + // Fill alive particles on host (slots >= n_alive stay zero == dead). + auto h_i1 = Kokkos::create_mirror_view(i1); + auto h_i2 = Kokkos::create_mirror_view(i2); + auto h_i1p = Kokkos::create_mirror_view(i1_prev); + auto h_i2p = Kokkos::create_mirror_view(i2_prev); + auto h_dx1 = Kokkos::create_mirror_view(dx1); + auto h_dx2 = Kokkos::create_mirror_view(dx2); + auto h_dx1p = Kokkos::create_mirror_view(dx1_prev); + auto h_dx2p = Kokkos::create_mirror_view(dx2_prev); + auto h_ux3 = Kokkos::create_mirror_view(ux3); + auto h_w = Kokkos::create_mirror_view(weight); + auto h_tag = Kokkos::create_mirror_view(tag); + int p = 0; + for (int a = 0; a < n_base; ++a) { + for (int b = 0; b < n_base; ++b, ++p) { + h_i1p(p) = bases[a]; + h_i1(p) = bases[a] - 1; + h_i2p(p) = bases[b]; + h_i2(p) = bases[b] - 1; + h_dx1p(p) = static_cast(0.65); + h_dx1(p) = static_cast(0.99); + h_dx2p(p) = static_cast(0.65); + h_dx2(p) = static_cast(0.80); + h_ux3(p) = static_cast(2.5); + h_w(p) = static_cast(1.0); + h_tag(p) = ParticleTag::alive; + } + } + Kokkos::deep_copy(i1, h_i1); + Kokkos::deep_copy(i2, h_i2); + Kokkos::deep_copy(i1_prev, h_i1p); + Kokkos::deep_copy(i2_prev, h_i2p); + Kokkos::deep_copy(dx1, h_dx1); + Kokkos::deep_copy(dx2, h_dx2); + Kokkos::deep_copy(dx1_prev, h_dx1p); + Kokkos::deep_copy(dx2_prev, h_dx2p); + Kokkos::deep_copy(ux3, h_ux3); + Kokkos::deep_copy(weight, h_w); + Kokkos::deep_copy(tag, h_tag); + + // Flat reference over all slots (dead slots are skipped internally). + ndfield_t J_flat { "J_flat", + nx1 + 2u * N_GHOSTS, + nx2 + 2u * N_GHOSTS }; + { + auto J_scat = Kokkos::Experimental::create_scatter_view(J_flat); + Kokkos::parallel_for( + "FlatDepositDrift", + n_slots, + kernel::DepositCurrents_kernel( + J_scat, + pack_arrays(i1, i2, i3, + i1_prev, i2_prev, i3_prev, + dx1, dx2, dx3, + dx1_prev, dx2_prev, dx3_prev, + ux1, ux2, ux3, + phi, weight, tag), + metric, charge, dt)); + Kokkos::Experimental::contribute(J_flat, J_scat); + Kokkos::fence("flat drift deposit done"); + } + + // Tiled with a maximally-stale layout: all alive particles in tile 0. + ndfield_t J_tiled { "J_tiled", + nx1 + 2u * N_GHOSTS, + nx2 + 2u * N_GHOSTS }; + { + const auto ntx1 = static_cast( + std::ceil(static_cast(nx1) / static_cast(T_TILE))); + const auto ntx2 = static_cast( + std::ceil(static_cast(nx2) / static_cast(T_TILE))); + + TileLayout layout; + layout.ntiles_per_axis[0] = ntx1; + layout.ntiles_per_axis[1] = ntx2; + layout.ntiles_per_axis[2] = 1u; + layout.ntiles_total = ntx1 * ntx2; + layout.tile_size = T_TILE; + layout.tile_offsets = build_tile_offsets_all_in_tile0( + ntx1 * ntx2, + static_cast(n_alive)); + + using kernel_t = + kernel::DepositCurrentsTiled_kernel; + // npart = n_alive: the stale layout buckets all alive particles into + // tile 0, so the team must walk [0, n_alive) and route the drifted + // ones to the global-J escape valve. + kernel_t kern { J_tiled, + pack_arrays(i1, i2, i3, + i1_prev, i2_prev, i3_prev, + dx1, dx2, dx3, + dx1_prev, dx2_prev, dx3_prev, + ux1, ux2, ux3, + phi, weight, tag), + metric, charge, dt, layout, + static_cast(n_alive) }; + + Kokkos::TeamPolicy<> policy(static_cast(layout.ntiles_total), + Kokkos::AUTO); + policy.set_scratch_size(0, + Kokkos::PerTeam(kernel_t::scratch_bytes())); + Kokkos::parallel_for("TiledDepositDrift", policy, kern); + Kokkos::fence("tiled drift deposit done"); + } + + compare_J_fields(J_flat, J_tiled, O, T_TILE, "drift"); + } + template void run_all_orders() { run_one_case<0u, T_TILE>(); @@ -241,6 +484,20 @@ namespace { run_one_case<9u, T_TILE>(); run_one_case<10u, T_TILE>(); run_one_case<11u, T_TILE>(); + + // Stale-layout / drift regression (per-particle escape valve). + run_drift_case<0u, T_TILE>(); + run_drift_case<1u, T_TILE>(); + run_drift_case<2u, T_TILE>(); + run_drift_case<3u, T_TILE>(); + run_drift_case<4u, T_TILE>(); + run_drift_case<5u, T_TILE>(); + run_drift_case<6u, T_TILE>(); + run_drift_case<7u, T_TILE>(); + run_drift_case<8u, T_TILE>(); + run_drift_case<9u, T_TILE>(); + run_drift_case<10u, T_TILE>(); + run_drift_case<11u, T_TILE>(); } } // namespace From 4dfaba918443fb6dc374f2d6c661c5836e5742dc Mon Sep 17 00:00:00 2001 From: Ludwig Boess Date: Fri, 19 Jun 2026 22:36:26 -0400 Subject: [PATCH 012/125] replaced compile-time sort intervale with compile-time drift halo size --- CMakeLists.txt | 40 ++++++++++++++----------- cmake/defaults.cmake | 6 ++-- cmake/report.cmake | 24 +++++++-------- src/engines/srpic/currents.h | 2 +- src/framework/parameters/parameters.cpp | 7 ----- src/framework/parameters/particles.cpp | 9 ------ src/kernels/currents_deposit.hpp | 39 ++++++++++++------------ 7 files changed, 56 insertions(+), 71 deletions(-) diff --git a/CMakeLists.txt b/CMakeLists.txt index d7bea74f0..f8409209b 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -67,10 +67,10 @@ set(team_policy_tile_size set(team_policy_tile_sizes "4;6;8;10;12;14;16" CACHE STRING "team_policy tile-size choices") -set(team_policy_sort - ${default_team_policy_sort} +set(team_policy_drift + ${default_team_policy_drift} CACHE STRING - "team_policy hardwired spatial sorting interval; >0 overrides the runtime spatial_sorting_interval and sizes the tiled deposit scratch halo (0 = use runtime)") + "team_policy tiled-deposit scratch halo drift in cells (max cells a particle may move between two sorts). Sizes the deposit scratch halo only; the sort cadence is set at runtime via spatial_sorting_interval. Default 1.") # -------------------------- Compilation settings -------------------------- # set(CMAKE_CXX_STANDARD 20) @@ -161,14 +161,13 @@ if(${team_policy}) add_compile_options("-D TEAM_POLICY") add_compile_options("-D TEAM_POLICY_TILE_SIZE=${team_policy_tile_size}") - # Optional compile-time hardwired sort interval. When > 0, it (a) overrides - # the runtime spatial_sorting_interval (see framework/parameters) and (b) - # sizes the tiled deposit scratch halo to DRIFT = interval, so a particle - # drifting over a full sort interval still deposits inside its tile scratch. - if(team_policy_sort GREATER 0) - add_compile_options( - "-D TEAM_POLICY_SORT_INTERVAL=${team_policy_sort}") - endif() + # Compile-time tiled-deposit scratch halo drift. Sizes the halo so a + # particle that drifts up to DRIFT cells between two sorts still deposits + # inside its tile scratch; particles drifting further take the + # per-particle global-J escape valve (correct, only slower). This is + # independent of the sort cadence, which is set at runtime via + # `spatial_sorting_interval`. Defaults to 1 (the sorted-every-step case). + add_compile_options("-D TEAM_POLICY_DRIFT=${team_policy_drift}") # Vendor sort: oneDPL on SYCL, Thrust on CUDA. Used automatically # when found; falls back to Kokkos::BinSort otherwise. @@ -196,18 +195,23 @@ if(${team_policy}) endif() if("${Kokkos_DEVICES}" MATCHES "HIP") - # rocThrust ships with ROCm and exposes the same thrust:: API. Using - # it lets the HIP backend build a single permutation via - # sort_by_key and gather all SoA members through one reused buffer, - # instead of the legacy per-member Kokkos::BinSort path which - # allocates a fresh `sorted_values` buffer for every member every - # step (the dominant source of allocator churn / fragmentation on - # ROCm). + # rocThrust ships with ROCm. The HIP sort_by_key path uses rocprim's + # bounded-bit radix sort directly (rocprim is rocThrust's own + # dependency, so its headers come in transitively; we find it + # explicitly to keep the include path robust). This builds a single + # permutation that gathers all SoA members, instead of the legacy + # per-member Kokkos::BinSort path which allocates a fresh + # `sorted_values` buffer for every member every step (the dominant + # source of allocator churn / fragmentation on ROCm). find_package(rocthrust QUIET) if(rocthrust_FOUND) message(STATUS "team_policy: rocThrust enabled for HIP sort_by_key") add_compile_options("-D ROCTHRUST_ENABLED") set(DEPENDENCIES ${DEPENDENCIES} roc::rocthrust) + find_package(rocprim QUIET) + if(rocprim_FOUND) + set(DEPENDENCIES ${DEPENDENCIES} roc::rocprim) + endif() else() message(STATUS "team_policy: rocThrust not found; using BinSort " "fallback for HIP sort_by_key") diff --git a/cmake/defaults.cmake b/cmake/defaults.cmake index 619f48dc1..01427921f 100644 --- a/cmake/defaults.cmake +++ b/cmake/defaults.cmake @@ -108,7 +108,7 @@ set(default_team_policy_tile_size 8 CACHE INTERNAL "Default tile edge length in cells for team_policy") -set(default_team_policy_sort - 0 +set(default_team_policy_drift + 1 CACHE INTERNAL - "Default hardwired spatial sorting interval for team_policy (0 = runtime)") + "Default tiled-deposit scratch halo drift for team_policy (cells between sorts)") diff --git a/cmake/report.cmake b/cmake/report.cmake index e6036d366..d90dfb085 100644 --- a/cmake/report.cmake +++ b/cmake/report.cmake @@ -141,17 +141,15 @@ if(${team_policy}) "${Blue}" TEAM_POLICY_TILE_SIZE_REPORT 46) - if(team_policy_sort GREATER 0) - printchoices( - "Team Sort Interval" - "team_policy_sort" - "${team_policy_sort}" - ${team_policy_sort} - 0 - "${Blue}" - TEAM_POLICY_SORT_INTERVAL_REPORT - 46) - endif() + printchoices( + "Team Deposit Drift" + "team_policy_drift" + "${team_policy_drift}" + ${team_policy_drift} + 1 + "${Blue}" + TEAM_POLICY_DRIFT_REPORT + 46) endif() printchoices( "Debug mode" @@ -231,9 +229,7 @@ endif() string(APPEND REPORT_TEXT " " ${TEAM_POLICY_REPORT} "\n") if(${team_policy}) string(APPEND REPORT_TEXT " " ${TEAM_POLICY_TILE_SIZE_REPORT} "\n") - if(team_policy_sort GREATER 0) - string(APPEND REPORT_TEXT " " ${TEAM_POLICY_SORT_INTERVAL_REPORT} "\n") - endif() + string(APPEND REPORT_TEXT " " ${TEAM_POLICY_DRIFT_REPORT} "\n") endif() string( diff --git a/src/engines/srpic/currents.h b/src/engines/srpic/currents.h index eadda70b9..b8cfe63ab 100644 --- a/src/engines/srpic/currents.h +++ b/src/engines/srpic/currents.h @@ -128,7 +128,7 @@ namespace ntt { // partition per-particle: // - a particle whose full stencil has drifted out of its tile is // deposited straight to the global J view (the per-particle escape - // valve); `team_policy_sort_interval` sizes the scratch halo so the + // valve); `team_policy_drift` sizes the scratch halo so the // common in-tile case stays in fast SLM (see currents_deposit.hpp); // - particles dead-tagged in place since the sort are clamped out by // the kernel and skipped by the dead-tag test; diff --git a/src/framework/parameters/parameters.cpp b/src/framework/parameters/parameters.cpp index 4116783a1..7372da510 100644 --- a/src/framework/parameters/parameters.cpp +++ b/src/framework/parameters/parameters.cpp @@ -67,18 +67,11 @@ namespace ntt { "clear_interval", defaults::clear_interval); set("particles.clear_interval", global_clearing_interval); -#if defined(TEAM_POLICY_SORT_INTERVAL) - // See particles.cpp: the compile-time team_policy_sort_interval overrides - // the runtime value (kept consistent here for the stored global param). - const auto global_spatial_sorting_interval = static_cast( - TEAM_POLICY_SORT_INTERVAL); -#else const auto global_spatial_sorting_interval = toml::find_or( toml_data, "particles", "spatial_sorting_interval", 0u); -#endif set("particles.spatial_sorting_interval", global_spatial_sorting_interval); set("scales.n0", ppc0 / get("scales.V0")); diff --git a/src/framework/parameters/particles.cpp b/src/framework/parameters/particles.cpp index 46192035b..6f1a23c03 100644 --- a/src/framework/parameters/particles.cpp +++ b/src/framework/parameters/particles.cpp @@ -121,19 +121,10 @@ namespace ntt { sp, "clear_interval", global_clearing_interval); -#if defined(TEAM_POLICY_SORT_INTERVAL) - // Compile-time hardwired sort interval (the `team_policy_sort_interval` - // CMake knob). It overrides whatever the input file requested so the - // tiled deposit's scratch halo — sized for exactly this cadence — is - // guaranteed to contain every particle's drift between sorts. - const auto spatial_sorting_interval = static_cast( - TEAM_POLICY_SORT_INTERVAL); -#else const auto spatial_sorting_interval = toml::find_or( sp, "spatial_sorting_interval", global_spatial_sorting_interval); -#endif auto pusher_str = toml::find_or(sp, "pusher", std::string(def_pusher)); const auto npayloads_real = toml::find_or(sp, "n_payloads_real", diff --git a/src/kernels/currents_deposit.hpp b/src/kernels/currents_deposit.hpp index 8840a1046..b92b02423 100644 --- a/src/kernels/currents_deposit.hpp +++ b/src/kernels/currents_deposit.hpp @@ -786,11 +786,12 @@ namespace kernel { * per step elapsed since the last sort. The scratch HALO is * `STENCIL_REACH(O) + DRIFT`, where `STENCIL_REACH = 2` for zigzag * (writes `{i_prev, i_prev+1, i, i+1}` ⇒ +2 above `min(i, i_prev)` with - * `|Δi|=1`) and `O` for Esirkepov. `DRIFT` is the - * `team_policy_sort_interval` CMake knob (macro TEAM_POLICY_SORT_INTERVAL) - * when set — the hardwired sort interval, hence the maximum drift any - * particle accrues between sorts — and `1` otherwise (the - * every-step-sorted common case). + * `|Δi|=1`) and `O` for Esirkepov. `DRIFT` is the `team_policy_drift` + * CMake knob (macro TEAM_POLICY_DRIFT) — the number of cells a particle + * may drift between two sorts that the halo is sized to absorb — and `1` + * by default (the every-step-sorted common case). It is independent of + * the sort cadence, which is set at runtime via `spatial_sorting_interval`; + * particles that drift past the halo take the escape valve below. * * Correctness does **not** depend on the halo size. Any particle whose * full stencil escapes the scratch tile — because it drifted further @@ -801,9 +802,9 @@ namespace kernel { * valve). Each particle's stencil is therefore deposited exactly once * (entirely to SLM scratch when it fits, entirely to global J when it * does not), so the path is charge-conserving; it is merely slower per - * write. Sizing `DRIFT` to the sort interval keeps the common, - * within-interval drift in fast SLM; sorting less often only costs - * escape-valve traffic, never accuracy. + * write. Sizing `DRIFT` to the typical between-sort drift keeps the + * common case in fast SLM; sorting less often (or drifting past the + * halo) only costs escape-valve traffic, never accuracy. * * **Partition coverage.** The team iteration covers only the particles * partitioned at the last sort, `[0, layout.npart_partitioned)`, clamped @@ -836,15 +837,15 @@ namespace kernel { * * drift — sort runs at end-of-step (see srpic.hpp), so a particle is * pushed once per step between its last sort and a given deposit. With - * a sort interval of `K`, a particle therefore drifts at most `K` cells - * (CFL |v dt/dx| <= 1/2 ⇒ |Δi| <= 1 per step) before the next sort. The - * `team_policy_sort_interval` CMake knob (macro TEAM_POLICY_SORT_INTERVAL) - * pins that interval at compile time and feeds it here as DRIFT, sizing - * the halo so a fully-interval-drifted particle still deposits inside - * its tile scratch. When the knob is unset, DRIFT defaults to 1 (the - * sorted-every-step common case); any particle that drifts past the halo - * (e.g. a larger runtime interval, or a CFL excursion) takes the - * per-particle global-J escape valve below — correct, only slower (see + * a runtime sort interval of `K` (spatial_sorting_interval), a particle + * drifts at most `K` cells (CFL |v dt/dx| <= 1/2 ⇒ |Δi| <= 1 per step) + * before the next sort. The `team_policy_drift` CMake knob (macro + * TEAM_POLICY_DRIFT) sets DRIFT independently of `K`, sizing the halo so + * a particle that drifts up to DRIFT cells still deposits inside its + * tile scratch. DRIFT defaults to 1 (the sorted-every-step common case); + * any particle that drifts past the halo (e.g. a larger sort interval, + * or a CFL excursion) takes the per-particle global-J escape valve + * below — correct, only slower (see * the class doc-comment for why this is charge-conserving). */ static constexpr int STENCIL_REACH = (O == 0u) ? 2 : static_cast(O); @@ -854,8 +855,8 @@ namespace kernel { // coords conservatively bounds every deposited cell for any order // (Esirkepov reaches max+O; O=0 zigzag reaches max+1). static constexpr int FOOTPRINT_REACH = (O == 0u) ? 1 : static_cast(O); -#if defined(TEAM_POLICY_SORT_INTERVAL) - static constexpr int DRIFT = static_cast(TEAM_POLICY_SORT_INTERVAL); +#if defined(TEAM_POLICY_DRIFT) + static constexpr int DRIFT = static_cast(TEAM_POLICY_DRIFT); #else static constexpr int DRIFT = 1; #endif From cd1be6b383839fb72a31e078c3da5a59f0246d71 Mon Sep 17 00:00:00 2001 From: Ludwig Boess Date: Fri, 19 Jun 2026 22:37:08 -0400 Subject: [PATCH 013/125] AMD-specific sorting improvements --- src/framework/containers/particles.h | 12 ++ src/framework/containers/particles_sort.cpp | 128 +++++++------ src/global/utils/sort_dispatch.h | 154 ++++++++++++--- tests/framework/particles_sort.cpp | 196 ++++++++++++++------ tests/framework/sort_by_key.cpp | 3 + 5 files changed, 353 insertions(+), 140 deletions(-) diff --git a/src/framework/containers/particles.h b/src/framework/containers/particles.h index 0877efd15..511a07a62 100644 --- a/src/framework/containers/particles.h +++ b/src/framework/containers/particles.h @@ -99,6 +99,18 @@ namespace ntt { // vendor libraries detected by CMake. TileLayout m_tile_layout {}; +#if defined(TEAM_POLICY) + // Build m_tile_layout.tile_offsets / npart_partitioned from the + // already-sorted tile-index keys. A separate member function (not a + // lambda local to SortSpatially) so the inner device kernel is not an + // extended __device__ lambda nested inside another lambda — which + // nvcc forbids. Lets the vendor path run the offsets pass and then + // release the keys before the SoA gather allocates its buffers. + void compute_tile_offsets(const array_t& tile_indices, + ncells_t total_tiles, + npart_t npart_local); +#endif + public: // for empty allocation Particles() {} diff --git a/src/framework/containers/particles_sort.cpp b/src/framework/containers/particles_sort.cpp index a52b013f7..3db6c1438 100644 --- a/src/framework/containers/particles_sort.cpp +++ b/src/framework/containers/particles_sort.cpp @@ -205,6 +205,61 @@ namespace ntt { m_is_sorted = true; } +#if defined(TEAM_POLICY) + template + void Particles::compute_tile_offsets( + const array_t& tile_indices, + ncells_t total_tiles, + npart_t npart_local) { + // Compute the per-tile prefix-sum `tile_offsets` for the tiled + // pusher from the (already sorted) `tile_indices` — monotonically + // non-decreasing for alive particles, with the dead sentinel + // `total_tiles + 1` clustered at the end. Transition-detect directly + // on it: the start of each non-empty tile is the only place a write + // happens — atomic-free in the dense branch. Empty tiles (no + // particles) are filled by a reverse pass on a small host mirror + // (`total_tiles ≈ 176K` at production scale → ~700 KB). + array_t tile_offsets { "tile_offsets", total_tiles + 1u }; + Kokkos::deep_copy(tile_offsets, static_cast(npart_local)); + + const auto total_tiles_v = total_tiles; + auto ti_v = tile_indices; + Kokkos::parallel_for( + "DetectTileBoundaries", + CreateParticleRangePolicy({ 0u }, { npart_local }), + Lambda(prtlidx_t p) { + const auto t_curr = ti_v(p); + const bool boundary = (p == 0u) || (ti_v(p - 1u) != t_curr); + if (!boundary) { + return; + } + if (t_curr < total_tiles_v) { + tile_offsets(t_curr) = p; + } else { + // First dead particle — also marks the alive_count boundary + // stored at index total_tiles. + Kokkos::atomic_min(&tile_offsets(total_tiles_v), p); + } + }); + + auto h_offsets = Kokkos::create_mirror_view(tile_offsets); + Kokkos::deep_copy(h_offsets, tile_offsets); + for (auto t = static_cast(total_tiles); t-- > 0u;) { + if (h_offsets(t) > h_offsets(t + 1u)) { + h_offsets(t) = h_offsets(t + 1u); + } + } + Kokkos::deep_copy(tile_offsets, h_offsets); + + m_tile_layout.tile_offsets = tile_offsets; + // tile_offsets(total_tiles) is the alive-particle count at sort time: + // the tiles partition exactly [0, npart_partitioned). The deposit + // launcher compares this against the live npart() to detect (and + // separately deposit) particles appended since this sort. + m_tile_layout.npart_partitioned = h_offsets(total_tiles); + } +#endif // TEAM_POLICY + template void Particles::SortSpatially(const Grid& grid) { #if defined(TEAM_POLICY) @@ -262,6 +317,7 @@ namespace ntt { // the dead-particle sentinel bin (total_tiles + 1u). const ncells_t n_bins = total_tiles + 2u; const auto slice = prtl_slice_t(0, npart_local); + #if defined(TEAM_POLICY_USE_VENDOR_SORT) // Vendor path: produce an explicit permutation via sort_by_key, // then apply it to each SoA member by gathering into a fresh @@ -286,6 +342,12 @@ namespace ntt { n_bins, sort::backend::Thrust {}); #endif + // `tile_indices` is sorted in place by sort_by_key. Build the tile + // offsets from it now, then drop it before the gather allocates its + // `maxnpart`-sized buffers — so the keys are not co-resident with + // them at the gather's peak (#2). + compute_tile_offsets(tile_indices, total_tiles, npart_local); + tile_indices = array_t {}; Kokkos::fence("SortSpatially: pre-gather drain"); apply_permutation_to_soa(perm); #else @@ -330,67 +392,17 @@ namespace ntt { sorter.sort(Kokkos::subview(pld_i, slice, pldi)); } // Apply the same permutation to `tile_indices` itself so it ends - // monotonically non-decreasing for the offsets pass below. + // monotonically non-decreasing for the offsets pass, then build the + // tile offsets from it (the in-place BinSort path has no separate + // gather to hoist this ahead of). sorter.sort(tile_indices); + compute_tile_offsets(tile_indices, total_tiles, npart_local); #endif // TEAM_POLICY_USE_VENDOR_SORT - // 5. Compute per-tile prefix-sum `tile_offsets` for the tiled - // pusher. `tile_indices` is now sorted (monotonically - // non-decreasing for alive particles, dead sentinel - // `total_tiles + 1` clustered at the end) — vendor sort_by_key - // sorts keys in place; the BinSort path explicitly applies the - // same permutation to `tile_indices` above. Transition-detect - // directly on it: the start of each non-empty tile is the only - // place a write happens — atomic-free in the dense branch. - // Empty tiles (no particles) are filled by a reverse pass on a - // small host mirror (`total_tiles ≈ 176K` at production scale → - // ~700 KB). - { - array_t tile_offsets { "tile_offsets", total_tiles + 1u }; - Kokkos::deep_copy(tile_offsets, static_cast(npart_local)); - - const auto total_tiles_v = total_tiles; - auto ti_v = tile_indices; - Kokkos::parallel_for( - "DetectTileBoundaries", - rangeActiveParticles(), - Lambda(prtlidx_t p) { - const auto t_curr = ti_v(p); - const bool boundary = (p == 0u) || (ti_v(p - 1u) != t_curr); - if (!boundary) { - return; - } - if (t_curr < total_tiles_v) { - tile_offsets(t_curr) = p; - } else { - // First dead particle — also marks the alive_count boundary - // stored at index total_tiles. - Kokkos::atomic_min(&tile_offsets(total_tiles_v), p); - } - }); - - auto h_offsets = Kokkos::create_mirror_view(tile_offsets); - Kokkos::deep_copy(h_offsets, tile_offsets); - for (auto t = static_cast(total_tiles); t-- > 0u;) { - if (h_offsets(t) > h_offsets(t + 1u)) { - h_offsets(t) = h_offsets(t + 1u); - } - } - Kokkos::deep_copy(tile_offsets, h_offsets); - - m_tile_layout.tile_offsets = tile_offsets; - // tile_offsets(total_tiles) is the alive-particle count at sort time: - // the tiles partition exactly [0, npart_partitioned). The deposit - // launcher compares this against the live npart() to detect (and - // separately deposit) particles appended since this sort. - m_tile_layout.npart_partitioned = h_offsets(total_tiles); - } - - // 6. Populate `m_tile_layout` size/shape. `tile_perm` is not used - // in the current design — the SoA arrays are physically permuted - // into tile order, so consumers iterate - // `[tile_offsets(t), tile_offsets(t+1))` directly without a - // separate permutation indirection. + // Populate `m_tile_layout` size/shape. `tile_perm` is not used in the + // current design — the SoA arrays are physically permuted into tile + // order, so consumers iterate `[tile_offsets(t), tile_offsets(t+1))` + // directly without a separate permutation indirection. m_tile_layout.ntiles_per_axis[0] = ntx[0]; m_tile_layout.ntiles_per_axis[1] = ntx[1]; m_tile_layout.ntiles_per_axis[2] = ntx[2]; diff --git a/src/global/utils/sort_dispatch.h b/src/global/utils/sort_dispatch.h index 506a90742..cfc389464 100644 --- a/src/global/utils/sort_dispatch.h +++ b/src/global/utils/sort_dispatch.h @@ -14,8 +14,13 @@ * keys[perm[0]] <= keys[perm[1]] <= ... in stable order. * Always-available overloads: BinSort (uses Kokkos::BinSort) and * StdSort (host-side std::stable_sort fallback). The vendor-library - * overloads (OneDPL on SYCL, Thrust on CUDA) are conditional on the - * respective build flags. + * overloads (OneDPL on SYCL, cub radix sort on CUDA, rocprim radix + * sort on HIP) are conditional on the respective build flags. + * @note The CUDA/HIP overloads bound the radix sort to the significant + * key bits (`significant_bits(n_bins)`) instead of the full 32, so + * only ceil(log2(n_bins)) bits are sorted — fewer radix passes than + * a full-width `thrust::sort_by_key`. Scratch is transient (freed at + * scope exit); no persistent buffer is retained (cf. 787aa045). */ #ifndef GLOBAL_UTILS_SORT_DISPATCH_H @@ -38,22 +43,32 @@ #include #endif #if defined(CUDA_ENABLED) && defined(THRUST_ENABLED) - #include - #include - #include + #include #endif #if defined(HIP_ENABLED) && defined(ROCTHRUST_ENABLED) - #include - #include - #include - #include + #include #endif #include +#include #include namespace ntt::sort_helpers { + // Number of low-order bits needed to represent keys in [0, n_bins). + // The radix sort only needs to scan these bits — bounding `end_bit` to + // ceil(log2(n_bins)) instead of 32 cuts the number of passes (e.g. 18 + // bits when total_tiles ~ 176K). Returns at least 1. + inline unsigned int significant_bits(ncells_t n_bins) { + unsigned int bits = 0u; + while (bits < 32u && + (static_cast(1u) << bits) < + static_cast(n_bins)) { + ++bits; + } + return (bits == 0u) ? 1u : bits; + } + // Always-available legacy fallback: Kokkos::BinSort. n_bins must be an // upper bound on distinct key values. inline void sort_by_key_dispatch(const array_t& keys, @@ -109,39 +124,126 @@ namespace ntt::sort_helpers { #if defined(CUDA_ENABLED) && defined(THRUST_ENABLED) inline void sort_by_key_dispatch(const array_t& keys, prtl_perm_t& perm, - ncells_t /*n_bins*/, + ncells_t n_bins, ::sort::backend::Thrust) { const auto n = static_cast(keys.extent(0)); if (n == 0u) { return; } - Kokkos::fence("sort_by_key_dispatch Thrust: pre-sort"); - thrust::device_ptr kp(keys.data()); - thrust::device_ptr pp(perm.data()); - thrust::sequence(pp, pp + n); - thrust::sort_by_key(kp, kp + n, pp); - Kokkos::fence("sort_by_key_dispatch Thrust: post-sort"); + auto exec = Kokkos::DefaultExecutionSpace(); + auto perm_v = perm; + Kokkos::parallel_for( + "PermInitIota", + n, + KOKKOS_LAMBDA(const npart_t i) { perm_v(i) = i; }); + + // Out-of-place radix sort bounded to the significant key bits. The + // _out buffers and temp storage are transient (freed at scope exit). + array_t keys_out("tile_keys_sorted", n); + prtl_perm_t perm_out("tile_perm_sorted", n); + const int end_bit = static_cast(significant_bits(n_bins)); + + exec.fence("sort_by_key_dispatch Thrust: pre-sort"); + auto stream = exec.cuda_stream(); + + std::size_t temp_bytes = 0; + auto err = cub::DeviceRadixSort::SortPairs(nullptr, + temp_bytes, + keys.data(), + keys_out.data(), + perm.data(), + perm_out.data(), + n, + 0, + end_bit, + stream); + raise::ErrorIf(err != cudaSuccess, + "cub::DeviceRadixSort::SortPairs (size query) failed", + HERE); + array_t temp("cub_radix_temp", + (temp_bytes == 0u) ? std::size_t { 1 } : temp_bytes); + err = cub::DeviceRadixSort::SortPairs(temp.data(), + temp_bytes, + keys.data(), + keys_out.data(), + perm.data(), + perm_out.data(), + n, + 0, + end_bit, + stream); + raise::ErrorIf(err != cudaSuccess, + "cub::DeviceRadixSort::SortPairs failed", + HERE); + exec.fence("sort_by_key_dispatch Thrust: post-sort"); + + // Publish sorted keys back in place; swap the sorted permutation in. + auto keys_nc = keys; // non-const handle aliasing the same storage + Kokkos::deep_copy(keys_nc, keys_out); + perm = perm_out; } #endif #if defined(HIP_ENABLED) && defined(ROCTHRUST_ENABLED) - // rocThrust exposes the same thrust:: API as CUDA Thrust; with hipcc - // device_ptr-based algorithms dispatch to the HIP backend. Mirrors - // the CUDA Thrust overload. + // HIP analogue of the CUDA cub overload, using rocprim's radix sort + // (which ships with rocThrust). Same bounded-bit, out-of-place, + // transient-scratch scheme. inline void sort_by_key_dispatch(const array_t& keys, prtl_perm_t& perm, - ncells_t /*n_bins*/, + ncells_t n_bins, ::sort::backend::Rocthrust) { const auto n = static_cast(keys.extent(0)); if (n == 0u) { return; } - Kokkos::fence("sort_by_key_dispatch Rocthrust: pre-sort"); - thrust::device_ptr kp(keys.data()); - thrust::device_ptr pp(perm.data()); - thrust::sequence(pp, pp + n); - thrust::sort_by_key(kp, kp + n, pp); - Kokkos::fence("sort_by_key_dispatch Rocthrust: post-sort"); + auto exec = Kokkos::DefaultExecutionSpace(); + auto perm_v = perm; + Kokkos::parallel_for( + "PermInitIota", + n, + KOKKOS_LAMBDA(const npart_t i) { perm_v(i) = i; }); + + array_t keys_out("tile_keys_sorted", n); + prtl_perm_t perm_out("tile_perm_sorted", n); + const unsigned int end_bit = significant_bits(n_bins); + + exec.fence("sort_by_key_dispatch Rocthrust: pre-sort"); + auto stream = exec.hip_stream(); + + std::size_t temp_bytes = 0; + auto err = rocprim::radix_sort_pairs(nullptr, + temp_bytes, + keys.data(), + keys_out.data(), + perm.data(), + perm_out.data(), + static_cast(n), + 0u, + end_bit, + stream); + raise::ErrorIf(err != hipSuccess, + "rocprim::radix_sort_pairs (size query) failed", + HERE); + array_t temp("rocprim_radix_temp", + (temp_bytes == 0u) ? std::size_t { 1 } : temp_bytes); + err = rocprim::radix_sort_pairs(temp.data(), + temp_bytes, + keys.data(), + keys_out.data(), + perm.data(), + perm_out.data(), + static_cast(n), + 0u, + end_bit, + stream); + raise::ErrorIf(err != hipSuccess, + "rocprim::radix_sort_pairs failed", + HERE); + exec.fence("sort_by_key_dispatch Rocthrust: post-sort"); + + auto keys_nc = keys; // non-const handle aliasing the same storage + Kokkos::deep_copy(keys_nc, keys_out); + perm = perm_out; } #endif diff --git a/tests/framework/particles_sort.cpp b/tests/framework/particles_sort.cpp index 6945c962f..4e4881d98 100644 --- a/tests/framework/particles_sort.cpp +++ b/tests/framework/particles_sort.cpp @@ -31,12 +31,14 @@ auto main(int argc, char* argv[]) -> int { ntt::EmissionType::NONE, 2u, 1u); - auto& i1_p = prtls.i1; - auto& i2_p = prtls.i2; - auto& tag_p = prtls.tag; - auto& weight_p = prtls.weight; - auto& pld_r = prtls.pld_r; - auto& pld_i = prtls.pld_i; + auto& i1_p = prtls.i1; + auto& i2_p = prtls.i2; + auto& i1_prev_p = prtls.i1_prev; + auto& i2_prev_p = prtls.i2_prev; + auto& tag_p = prtls.tag; + auto& weight_p = prtls.weight; + auto& pld_r = prtls.pld_r; + auto& pld_i = prtls.pld_i; Kokkos::parallel_for( "InitParticles", prtls.maxnpart(), @@ -61,9 +63,14 @@ auto main(int argc, char* argv[]) -> int { i2_p(p) = 23u; weight_p(p) = 3.0; } - pld_r(p, 0) = weight_p(p) + static_cast(0.5); - pld_r(p, 1) = weight_p(p) + static_cast(10.5); - pld_i(p, 0) = static_cast(weight_p(p) + 10.0); + // team_policy keys on min(i, i_prev); without a meaningful + // i_prev every key would collapse to 0. Set i_prev = i so the + // tile key reduces to the particle's current cell. + i1_prev_p(p) = i1_p(p); + i2_prev_p(p) = i2_p(p); + pld_r(p, 0) = weight_p(p) + static_cast(0.5); + pld_r(p, 1) = weight_p(p) + static_cast(10.5); + pld_i(p, 0) = static_cast(weight_p(p) + 10.0); } else { tag_p(p) = ntt::ParticleTag::dead; } @@ -75,43 +82,78 @@ auto main(int argc, char* argv[]) -> int { prtls.SortSpatially(grid); + auto i1_h = Kokkos::create_mirror_view(prtls.i1); + auto i2_h = Kokkos::create_mirror_view(prtls.i2); + auto tag_h = Kokkos::create_mirror_view(prtls.tag); auto weight_h = Kokkos::create_mirror_view(prtls.weight); auto pld_r_h = Kokkos::create_mirror_view(prtls.pld_r); auto pld_i_h = Kokkos::create_mirror_view(prtls.pld_i); + Kokkos::deep_copy(i1_h, prtls.i1); + Kokkos::deep_copy(i2_h, prtls.i2); + Kokkos::deep_copy(tag_h, prtls.tag); Kokkos::deep_copy(weight_h, prtls.weight); Kokkos::deep_copy(pld_r_h, prtls.pld_r); Kokkos::deep_copy(pld_i_h, prtls.pld_i); - // Only [0, npart) is defined after a sort. The swap-based gather in - // apply_permutation_to_soa replaces each SoA View, zero-filling the - // spare capacity [npart, maxnpart) (a don't-care region overwritten - // by injection), so the old "tail preserved" check no longer holds. + // Tile geometry, mirroring sort::PositionToTileIndex. T = 1 (no + // team_policy) reproduces the legacy per-cell ordering. +#if defined(TEAM_POLICY) + const ncells_t T = static_cast(TEAM_POLICY_TILE_SIZE); +#else + const ncells_t T = 1u; +#endif + const auto na = grid.n_active(); + const ncells_t ntx2 = (na[1] + T - 1u) / T; + const auto tile_of = [&](int a, int b) -> ncells_t { + return (static_cast(a) / T) * ntx2 + + (static_cast(b) / T); + }; + + // SortSpatially is order-by-tile, not order-by-cell: assert the + // invariants that hold for any tile size rather than a hardwired + // permutation. (1) alive particles form a prefix sorted by + // non-decreasing tile index; (2) dead particles (weight == -1) form + // the suffix; (3) every SoA member is permuted by the *same* + // permutation, so each alive slot still satisfies pld == f(weight); + // (4) no alive particle is lost. Only [0, npart) is defined after a + // sort — the swap-gather zero-fills [npart, maxnpart). + bool seen_dead = false; + bool have_prev = false; + ncells_t prev_tile = 0u; + npart_t n_alive_obs = 0u; for (auto p { 0u }; p < 66u; ++p) { - if (p < 16u) { - raise::ErrorIf(weight_h(p) != 3.0, "error in sorting particles", HERE); - } else if (p < 33u) { - raise::ErrorIf(weight_h(p) != 1.0, "error in sorting particles", HERE); - } else if (p < 46u) { - raise::ErrorIf(weight_h(p) != 2.0, "error in sorting particles", HERE); - } else if (p < 59u) { - raise::ErrorIf(weight_h(p) != 0.0, "error in sorting particles", HERE); - } else { - raise::ErrorIf(weight_h(p) != -1.0, "error in sorting particles", HERE); - } - if (p < 59u) { - raise::ErrorIf(pld_r_h(p, 0) != weight_h(p) + static_cast(0.5), - "error in sorting particle real payload 0", - HERE); - raise::ErrorIf(pld_r_h(p, 1) != weight_h(p) + static_cast(10.5), - "error in sorting particle real payload 1", + if (tag_h(p) != ntt::ParticleTag::alive) { + seen_dead = true; + raise::ErrorIf(weight_h(p) != -1.0, + "dead particle has unexpected weight", HERE); - raise::ErrorIf( - pld_i_h(p, 0) != - static_cast(weight_h(p) + static_cast(10.0)), - "error in sorting particle integer payload 0", - HERE); + continue; } + raise::ErrorIf(seen_dead, + "alive particle after a dead one (not sorted to prefix)", + HERE); + const auto tile = tile_of(i1_h(p), i2_h(p)); + raise::ErrorIf(have_prev && (tile < prev_tile), + "alive particles not sorted by tile index", + HERE); + prev_tile = tile; + have_prev = true; + ++n_alive_obs; + raise::ErrorIf(pld_r_h(p, 0) != weight_h(p) + static_cast(0.5), + "error in sorting particle real payload 0", + HERE); + raise::ErrorIf(pld_r_h(p, 1) != weight_h(p) + static_cast(10.5), + "error in sorting particle real payload 1", + HERE); + raise::ErrorIf( + pld_i_h(p, 0) != + static_cast(weight_h(p) + static_cast(10.0)), + "error in sorting particle integer payload 0", + HERE); } + raise::ErrorIf(n_alive_obs != 59u, + "wrong number of alive particles after sort", + HERE); } { // 3D @@ -133,11 +175,14 @@ auto main(int argc, char* argv[]) -> int { ntt::EmissionType::NONE, 0u, 0u); - auto& i1_p = prtls.i1; - auto& i2_p = prtls.i2; - auto& i3_p = prtls.i3; - auto& tag_p = prtls.tag; - auto& weight_p = prtls.weight; + auto& i1_p = prtls.i1; + auto& i2_p = prtls.i2; + auto& i3_p = prtls.i3; + auto& i1_prev_p = prtls.i1_prev; + auto& i2_prev_p = prtls.i2_prev; + auto& i3_prev_p = prtls.i3_prev; + auto& tag_p = prtls.tag; + auto& weight_p = prtls.weight; Kokkos::parallel_for( "InitParticles", prtls.maxnpart(), @@ -171,6 +216,11 @@ auto main(int argc, char* argv[]) -> int { i3_p(p) = 7u; weight_p(p) = 4.0; } + // see 2D block: i_prev = i so the team_policy tile key reduces + // to the particle's current cell. + i1_prev_p(p) = i1_p(p); + i2_prev_p(p) = i2_p(p); + i3_prev_p(p) = i3_p(p); } else { tag_p(p) = ntt::ParticleTag::dead; } @@ -182,27 +232,61 @@ auto main(int argc, char* argv[]) -> int { prtls.SortSpatially(grid); + auto i1_h = Kokkos::create_mirror_view(prtls.i1); + auto i2_h = Kokkos::create_mirror_view(prtls.i2); + auto i3_h = Kokkos::create_mirror_view(prtls.i3); + auto tag_h = Kokkos::create_mirror_view(prtls.tag); auto weight_h = Kokkos::create_mirror_view(prtls.weight); + Kokkos::deep_copy(i1_h, prtls.i1); + Kokkos::deep_copy(i2_h, prtls.i2); + Kokkos::deep_copy(i3_h, prtls.i3); + Kokkos::deep_copy(tag_h, prtls.tag); Kokkos::deep_copy(weight_h, prtls.weight); - // Only [0, npart) is defined after a sort. The swap-based gather in - // apply_permutation_to_soa replaces each SoA View, zero-filling the - // spare capacity [npart, maxnpart) (a don't-care region overwritten - // by injection), so the old "tail preserved" check no longer holds. + + // Same invariants as the 2D block (no payloads here): alive prefix + // sorted by non-decreasing tile index, dead (weight == -1) suffix, + // alive count preserved. T = 1 reproduces the legacy per-cell order. +#if defined(TEAM_POLICY) + const ncells_t T = static_cast(TEAM_POLICY_TILE_SIZE); +#else + const ncells_t T = 1u; +#endif + const auto na = grid.n_active(); + const ncells_t ntx2 = (na[1] + T - 1u) / T; + const ncells_t ntx3 = (na[2] + T - 1u) / T; + const auto tile_of = [&](int a, int b, int c) -> ncells_t { + return ((static_cast(a) / T) * ntx2 + + (static_cast(b) / T)) * + ntx3 + + (static_cast(c) / T); + }; + + bool seen_dead = false; + bool have_prev = false; + ncells_t prev_tile = 0u; + npart_t n_alive_obs = 0u; for (auto p { 0u }; p < 66u; ++p) { - if (p < 13u) { - raise::ErrorIf(weight_h(p) != 4.0, "error in sorting particles", HERE); - } else if (p < 26u) { - raise::ErrorIf(weight_h(p) != 1.0, "error in sorting particles", HERE); - } else if (p < 39u) { - raise::ErrorIf(weight_h(p) != 2.0, "error in sorting particles", HERE); - } else if (p < 46u) { - raise::ErrorIf(weight_h(p) != 0.0, "error in sorting particles", HERE); - } else if (p < 59u) { - raise::ErrorIf(weight_h(p) != 3.0, "error in sorting particles", HERE); - } else { - raise::ErrorIf(weight_h(p) != -1.0, "error in sorting particles", HERE); + if (tag_h(p) != ntt::ParticleTag::alive) { + seen_dead = true; + raise::ErrorIf(weight_h(p) != -1.0, + "dead particle has unexpected weight", + HERE); + continue; } + raise::ErrorIf(seen_dead, + "alive particle after a dead one (not sorted to prefix)", + HERE); + const auto tile = tile_of(i1_h(p), i2_h(p), i3_h(p)); + raise::ErrorIf(have_prev && (tile < prev_tile), + "alive particles not sorted by tile index", + HERE); + prev_tile = tile; + have_prev = true; + ++n_alive_obs; } + raise::ErrorIf(n_alive_obs != 59u, + "wrong number of alive particles after sort", + HERE); } } catch (const std::exception& e) { diff --git a/tests/framework/sort_by_key.cpp b/tests/framework/sort_by_key.cpp index 94dc51d0e..9ccecc732 100644 --- a/tests/framework/sort_by_key.cpp +++ b/tests/framework/sort_by_key.cpp @@ -99,6 +99,9 @@ auto main(int argc, char* argv[]) -> int { #endif #if defined(CUDA_ENABLED) && defined(THRUST_ENABLED) test_one_backend("Thrust", ::sort::backend::Thrust {}); +#endif +#if defined(HIP_ENABLED) && defined(ROCTHRUST_ENABLED) + test_one_backend("Rocthrust", ::sort::backend::Rocthrust {}); #endif } catch (std::exception& e) { std::cerr << e.what() << std::endl; From ed70dbd24512efb3c483f9773f67fb865eb44f97 Mon Sep 17 00:00:00 2001 From: Ludwig Boess Date: Sun, 21 Jun 2026 12:48:28 -0400 Subject: [PATCH 014/125] reduced exchange for filters --- src/engines/srpic/currents.h | 116 +++++++++++++++++++++++++++++++---- 1 file changed, 104 insertions(+), 12 deletions(-) diff --git a/src/engines/srpic/currents.h b/src/engines/srpic/currents.h index b8cfe63ab..30eb9466d 100644 --- a/src/engines/srpic/currents.h +++ b/src/engines/srpic/currents.h @@ -27,6 +27,8 @@ #include "kernels/currents_deposit.hpp" #include "kernels/digital_filter.hpp" +#include + namespace ntt { namespace srpic { @@ -202,9 +204,11 @@ namespace ntt { Domain& domain, const SimulationParams& params) { logger::Checkpoint("Launching currents filtering kernels", HERE); - auto range = srpic::RangeWithAxisBCs(domain); const auto nfilter = params.template get( "algorithms.current_filters"); + if (nfilter == 0u) { + return; + } tuple_t size; if constexpr (M::Dim == Dim::_1D || M::Dim == Dim::_2D || M::Dim == Dim::_3D) { size[0] = domain.mesh.n_active(in::x1); @@ -215,17 +219,105 @@ namespace ntt { if constexpr (M::Dim == Dim::_3D) { size[2] = domain.mesh.n_active(in::x3); } - // !TODO: this needs to be done more efficiently - for (auto i { 0u }; i < nfilter; ++i) { - Kokkos::deep_copy(domain.fields.buff, domain.fields.cur); - Kokkos::parallel_for("CurrentsFilter", - range, - kernel::DigitalFilter_kernel( - domain.fields.cur, - domain.fields.buff, - size, - domain.mesh.flds_bc())); - metadomain.CommunicateFields(domain, Comm::J); + + // The filter ping-pongs `cur` <-> scratch `buff`: one up-front copy + // seeds `buff` with valid ghost cells (the kernel writes only the + // cells in its launch range), then each pass filters the input into + // the other buffer and swaps the View handles so `cur` again names the + // result. `buff` is pure scratch (also reused by CommunicateFields), so + // the permanent handle swap is transparent and `cur` always names the + // result. The single seeding copy preserves physical-boundary ghosts + // (conductor/match/atmosphere/...), which neither the kernel nor the + // MPI exchange refresh. + Kokkos::deep_copy(domain.fields.buff, domain.fields.cur); + + const auto flds_bc = domain.mesh.flds_bc(); + + if constexpr (M::CoordType == Coord::Cartesian) { + // Reduced-exchange ghost-margin scheme. One halo exchange refreshes + // N_GHOSTS ghost layers — enough for N_GHOSTS passes of the 3-point + // binomial if each pass also recomputes the inner ghost layers it + // will need next. We therefore extend the launch range by a shrinking + // margin `m` into the ghost zone, but only on comm-refreshed sides + // (PERIODIC self-wrap or SYNC inter-domain), where the ghost cell is + // interior physics. Physical-boundary ghosts are never written or + // refreshed, exactly as in the per-pass loop, so the result is + // identical for every BC — while doing one exchange per N_GHOSTS + // passes instead of one per pass. (Entering the loop the ghosts are + // valid to distance N_GHOSTS: srpic.hpp runs CommunicateFields(J) + // immediately before CurrentsFilter.) + const int G = static_cast(N_GHOSTS); + const auto comm_side = [](FldsBC b) { + return (b == FldsBC::PERIODIC) or (b == FldsBC::SYNC); + }; + bool ext_lo[3] = { false, false, false }; + bool ext_hi[3] = { false, false, false }; + for (auto d { 0 }; d < static_cast(M::Dim); ++d) { + ext_lo[d] = comm_side(flds_bc[d].first); + ext_hi[d] = comm_side(flds_bc[d].second); + } + const auto make_range = [&](int m) -> range_t { + const auto ml = [&](int d) -> ncells_t { + return ext_lo[d] ? static_cast(m) : 0u; + }; + const auto mh = [&](int d) -> ncells_t { + return ext_hi[d] ? static_cast(m) : 0u; + }; + if constexpr (M::Dim == Dim::_1D) { + return CreateRangePolicy( + { domain.mesh.i_min(in::x1) - ml(0) }, + { domain.mesh.i_max(in::x1) + mh(0) }); + } else if constexpr (M::Dim == Dim::_2D) { + return CreateRangePolicy( + { domain.mesh.i_min(in::x1) - ml(0), + domain.mesh.i_min(in::x2) - ml(1) }, + { domain.mesh.i_max(in::x1) + mh(0), + domain.mesh.i_max(in::x2) + mh(1) }); + } else { + return CreateRangePolicy( + { domain.mesh.i_min(in::x1) - ml(0), + domain.mesh.i_min(in::x2) - ml(1), + domain.mesh.i_min(in::x3) - ml(2) }, + { domain.mesh.i_max(in::x1) + mh(0), + domain.mesh.i_max(in::x2) + mh(1), + domain.mesh.i_max(in::x3) + mh(2) }); + } + }; + int m = G - 1; + for (auto i { 0u }; i < nfilter; ++i) { + Kokkos::parallel_for( + "CurrentsFilter", + make_range(m), + kernel::DigitalFilter_kernel( + domain.fields.buff, + domain.fields.cur, + size, + flds_bc)); + std::swap(domain.fields.cur, domain.fields.buff); + --m; + if (m < 0 or i == nfilter - 1u) { + // refresh ghosts to distance G (and leave them valid for the + // downstream field solver after the final pass) + metadomain.CommunicateFields(domain, Comm::J); + m = G - 1; + } + } + } else { + // Non-Cartesian (axis BCs need the +1 range fixup): keep the + // per-pass exchange cadence, still ping-ponging the buffers. + const auto range = srpic::RangeWithAxisBCs(domain); + for (auto i { 0u }; i < nfilter; ++i) { + Kokkos::parallel_for( + "CurrentsFilter", + range, + kernel::DigitalFilter_kernel( + domain.fields.buff, + domain.fields.cur, + size, + flds_bc)); + std::swap(domain.fields.cur, domain.fields.buff); + metadomain.CommunicateFields(domain, Comm::J); + } } } From 314d1424c2d2927a119a8918a45dfce8aa9bb501 Mon Sep 17 00:00:00 2001 From: Ludwig Boess Date: Sun, 21 Jun 2026 12:49:15 -0400 Subject: [PATCH 015/125] moved tag_offsets_h outside of loop to avoid multiple device-host copies --- src/framework/containers/particles_comm.cpp | 8 +++++--- 1 file changed, 5 insertions(+), 3 deletions(-) diff --git a/src/framework/containers/particles_comm.cpp b/src/framework/containers/particles_comm.cpp index 0fa140f93..6ede3abf0 100644 --- a/src/framework/containers/particles_comm.cpp +++ b/src/framework/containers/particles_comm.cpp @@ -266,6 +266,11 @@ namespace ntt { auto iteration = 0; auto current_received = 0; + // `tag_offsets` is the same for every direction; mirror it to the host + // once here instead of re-copying it inside the direction loop. + auto tag_offsets_h = Kokkos::create_mirror_view(tag_offsets); + Kokkos::deep_copy(tag_offsets_h, tag_offsets); + for (const auto& direction : dirs_to_comm) { const auto send_rank = send_ranks.at(direction); const auto recv_rank = recv_ranks.at(direction); @@ -291,9 +296,6 @@ namespace ntt { npart_send_in * NPLDS_I }; } - auto tag_offsets_h = Kokkos::create_mirror_view(tag_offsets); - Kokkos::deep_copy(tag_offsets_h, tag_offsets); - npart_t idx_offset = npart_dead; if (tag_send > 2) { idx_offset += tag_offsets_h(tag_send - 3); From 8047a118816d27f46a577a27235ac19ac8f740b7 Mon Sep 17 00:00:00 2001 From: Ludwig Boess Date: Sun, 21 Jun 2026 12:49:52 -0400 Subject: [PATCH 016/125] batched async communication --- src/framework/domain/comm_mpi.hpp | 170 +++++++++++++++++++++++ src/framework/domain/metadomain_comm.cpp | 66 +++++++++ 2 files changed, 236 insertions(+) diff --git a/src/framework/domain/comm_mpi.hpp b/src/framework/domain/comm_mpi.hpp index 52103c170..668efdd75 100644 --- a/src/framework/domain/comm_mpi.hpp +++ b/src/framework/domain/comm_mpi.hpp @@ -365,6 +365,176 @@ namespace comm { } } + // ---- batched, non-blocking field halo exchange ------------------------ // + + // Per-direction parameters, collected once by the caller (GetSendRecvParams) + // and reused for every field communicated in the same call. + template + struct FieldCommDir { + int send_rank { -1 }, recv_rank { -1 }; + unsigned int send_ind { 0 }, recv_ind { 0 }; + std::vector send_slice {}, recv_slice {}; + int tag { 0 }; + }; + + /** + * @brief Non-additive (ghost-overwrite) halo exchange of one field across + * all directions at once. + * @details Packs every send, posts all `MPI_Irecv` then all `MPI_Isend` + * (unique per-direction tags), a single `MPI_Waitall`, then unpacks + * — overlapping the round-trips instead of serializing one blocking + * `MPI_Sendrecv` per direction. Self-communication (periodic single + * domain) is a local copy. Slicing/packing mirrors + * `comm::CommunicateField`. The tag pairs A's send in a direction + * with B's receive in the same direction (the matching neighbor), + * so it is unique per (rank-pair, direction). + */ + template + inline void CommunicateFieldBatched(unsigned int my_idx, + ndfield_t& fld, + const std::vector>& dirs, + const cell_range_t& comps) { + static constexpr unsigned short Dp1 = static_cast(D) + 1; + using buf_t = ndarray_t; + const ncells_t ncomp = comps.second - comps.first; + const auto ndirs = dirs.size(); + + const auto is_self = [&](const FieldCommDir& dd) { + return (dd.send_ind == my_idx) && (dd.recv_ind == my_idx); + }; + const auto ext = [](const std::vector& sl, int d) -> ncells_t { + return sl[d].second - sl[d].first; + }; + const auto make_buf = [&](const char* lbl, + const std::vector& sl) -> buf_t { + if constexpr (D == Dim::_1D) { + return buf_t(lbl, ext(sl, 0), ncomp); + } else if constexpr (D == Dim::_2D) { + return buf_t(lbl, ext(sl, 0), ext(sl, 1), ncomp); + } else { + return buf_t(lbl, ext(sl, 0), ext(sl, 1), ext(sl, 2), ncomp); + } + }; + const auto count = [](const buf_t& b) -> int { + ncells_t n = 1; + for (auto d { 0 }; d < static_cast(Dp1); ++d) { + n *= static_cast(b.extent(d)); + } + return static_cast(n); + }; + const auto pack = [&](buf_t& b, const std::vector& sl) { + if constexpr (D == Dim::_1D) { + Kokkos::deep_copy(b, Kokkos::subview(fld, sl[0], comps)); + } else if constexpr (D == Dim::_2D) { + Kokkos::deep_copy(b, Kokkos::subview(fld, sl[0], sl[1], comps)); + } else { + Kokkos::deep_copy(b, Kokkos::subview(fld, sl[0], sl[1], sl[2], comps)); + } + }; + const auto unpack = [&](const buf_t& b, const std::vector& sl) { + if constexpr (D == Dim::_1D) { + Kokkos::deep_copy(Kokkos::subview(fld, sl[0], comps), b); + } else if constexpr (D == Dim::_2D) { + Kokkos::deep_copy(Kokkos::subview(fld, sl[0], sl[1], comps), b); + } else { + Kokkos::deep_copy(Kokkos::subview(fld, sl[0], sl[1], sl[2], comps), b); + } + }; + const auto self_copy = [&](const std::vector& ssl, + const std::vector& rsl) { + if constexpr (D == Dim::_1D) { + Kokkos::deep_copy(Kokkos::subview(fld, rsl[0], comps), + Kokkos::subview(fld, ssl[0], comps)); + } else if constexpr (D == Dim::_2D) { + Kokkos::deep_copy(Kokkos::subview(fld, rsl[0], rsl[1], comps), + Kokkos::subview(fld, ssl[0], ssl[1], comps)); + } else { + Kokkos::deep_copy(Kokkos::subview(fld, rsl[0], rsl[1], rsl[2], comps), + Kokkos::subview(fld, ssl[0], ssl[1], ssl[2], comps)); + } + }; + + std::vector send_buf(ndirs), recv_buf(ndirs); +#if defined(DEVICE_ENABLED) && !defined(GPU_AWARE_MPI) + using hbuf_t = typename buf_t::host_mirror_type; + std::vector send_h(ndirs), recv_h(ndirs); +#endif + + // phase 1: self copies, pack sends, allocate recvs + for (auto i { 0u }; i < ndirs; ++i) { + const auto& dd = dirs[i]; + if (is_self(dd)) { + self_copy(dd.send_slice, dd.recv_slice); + continue; + } + if (dd.recv_rank >= 0) { + recv_buf[i] = make_buf("recv_fld", dd.recv_slice); +#if defined(DEVICE_ENABLED) && !defined(GPU_AWARE_MPI) + recv_h[i] = Kokkos::create_mirror_view(recv_buf[i]); +#endif + } + if (dd.send_rank >= 0) { + send_buf[i] = make_buf("send_fld", dd.send_slice); + pack(send_buf[i], dd.send_slice); +#if defined(DEVICE_ENABLED) && !defined(GPU_AWARE_MPI) + send_h[i] = Kokkos::create_mirror_view(send_buf[i]); + Kokkos::deep_copy(send_h[i], send_buf[i]); +#endif + } + } + // drain packs (and device->host staging) before MPI reads the buffers + Kokkos::fence("CommunicateFieldBatched: pre-MPI"); + + // phase 2: post all recvs, then all sends + std::vector reqs; + reqs.reserve(2 * ndirs); + for (auto i { 0u }; i < ndirs; ++i) { + const auto& dd = dirs[i]; + if (is_self(dd) || dd.recv_rank < 0) { + continue; + } + reqs.emplace_back(); +#if defined(DEVICE_ENABLED) && !defined(GPU_AWARE_MPI) + MPI_Irecv(recv_h[i].data(), count(recv_buf[i]), mpi::get_type(), + dd.recv_rank, dd.tag, MPI_COMM_WORLD, &reqs.back()); +#else + MPI_Irecv(recv_buf[i].data(), count(recv_buf[i]), mpi::get_type(), + dd.recv_rank, dd.tag, MPI_COMM_WORLD, &reqs.back()); +#endif + } + for (auto i { 0u }; i < ndirs; ++i) { + const auto& dd = dirs[i]; + if (is_self(dd) || dd.send_rank < 0) { + continue; + } + reqs.emplace_back(); +#if defined(DEVICE_ENABLED) && !defined(GPU_AWARE_MPI) + MPI_Isend(send_h[i].data(), count(send_buf[i]), mpi::get_type(), + dd.send_rank, dd.tag, MPI_COMM_WORLD, &reqs.back()); +#else + MPI_Isend(send_buf[i].data(), count(send_buf[i]), mpi::get_type(), + dd.send_rank, dd.tag, MPI_COMM_WORLD, &reqs.back()); +#endif + } + + // phase 3: complete all transfers + if (not reqs.empty()) { + MPI_Waitall(static_cast(reqs.size()), reqs.data(), MPI_STATUSES_IGNORE); + } + + // phase 4: unpack received ghosts + for (auto i { 0u }; i < ndirs; ++i) { + const auto& dd = dirs[i]; + if (is_self(dd) || dd.recv_rank < 0) { + continue; + } +#if defined(DEVICE_ENABLED) && !defined(GPU_AWARE_MPI) + Kokkos::deep_copy(recv_buf[i], recv_h[i]); +#endif + unpack(recv_buf[i], dd.recv_slice); + } + } + } // namespace comm #endif // FRAMEWORK_DOMAIN_COMM_MPI_HPP diff --git a/src/framework/domain/metadomain_comm.cpp b/src/framework/domain/metadomain_comm.cpp index f8c0f60d7..d6061d7bf 100644 --- a/src/framework/domain/metadomain_comm.cpp +++ b/src/framework/domain/metadomain_comm.cpp @@ -273,6 +273,71 @@ namespace ntt { comp_range_cur = cell_range_t(cur::jx1, cur::jx3 + 1); } // traverse in all directions and send/recv the fields +#if defined(MPI_ENABLED) + // Batched non-blocking exchange: collect the per-direction params once, + // then overlap all directions for each field (post all Irecv/Isend with + // per-direction tags + one Waitall) instead of a blocking Sendrecv per + // direction. The #else branch is the single-rank non-MPI path. + { + std::vector> dirs; + dirs.reserve(dir::Directions::all.size()); + for (auto& direction : dir::Directions::all) { + const auto [send_params, recv_params] = GetSendRecvParams(this, + domain, + direction, + false); + const auto [send_indrank, send_slice] = send_params; + const auto [recv_indrank, recv_slice] = recv_params; + const auto [send_ind, send_rank] = send_indrank; + const auto [recv_ind, recv_rank] = recv_indrank; + if (send_rank < 0 and recv_rank < 0) { + continue; + } + comm::FieldCommDir fcd; + fcd.send_rank = send_rank; + fcd.recv_rank = recv_rank; + fcd.send_ind = send_ind; + fcd.recv_ind = recv_ind; + fcd.send_slice = send_slice; + fcd.recv_slice = recv_slice; + fcd.tag = static_cast(mpi::PrtlSendTag::dir2tag(direction)); + dirs.push_back(std::move(fcd)); + } + if (comm_em) { + comm::CommunicateFieldBatched(domain.index(), + domain.fields.em, + dirs, + comp_range_fld); + } + if constexpr (S == SimEngine::GRPIC) { + if (comm_aux) { + comm::CommunicateFieldBatched(domain.index(), + domain.fields.aux, + dirs, + comp_range_fld); + } + if (comm_em0) { + comm::CommunicateFieldBatched(domain.index(), + domain.fields.em0, + dirs, + comp_range_fld); + } + if (comm_j) { + comm::CommunicateFieldBatched(domain.index(), + domain.fields.cur0, + dirs, + comp_range_cur); + } + } else { + if (comm_j) { + comm::CommunicateFieldBatched(domain.index(), + domain.fields.cur, + dirs, + comp_range_cur); + } + } + } +#else for (auto& direction : dir::Directions::all) { const auto [send_params, recv_params] = GetSendRecvParams(this, domain, direction, false); @@ -364,6 +429,7 @@ namespace ntt { } } } +#endif // MPI_ENABLED } template From 50369176dde282993a75a1c73bd89b097ffcc38d Mon Sep 17 00:00:00 2001 From: Ludwig Boess Date: Sun, 21 Jun 2026 16:45:53 -0400 Subject: [PATCH 017/125] removed async comm because of bandwidth issue --- src/framework/domain/comm_mpi.hpp | 170 ----------------------- src/framework/domain/metadomain_comm.cpp | 66 --------- 2 files changed, 236 deletions(-) diff --git a/src/framework/domain/comm_mpi.hpp b/src/framework/domain/comm_mpi.hpp index 668efdd75..52103c170 100644 --- a/src/framework/domain/comm_mpi.hpp +++ b/src/framework/domain/comm_mpi.hpp @@ -365,176 +365,6 @@ namespace comm { } } - // ---- batched, non-blocking field halo exchange ------------------------ // - - // Per-direction parameters, collected once by the caller (GetSendRecvParams) - // and reused for every field communicated in the same call. - template - struct FieldCommDir { - int send_rank { -1 }, recv_rank { -1 }; - unsigned int send_ind { 0 }, recv_ind { 0 }; - std::vector send_slice {}, recv_slice {}; - int tag { 0 }; - }; - - /** - * @brief Non-additive (ghost-overwrite) halo exchange of one field across - * all directions at once. - * @details Packs every send, posts all `MPI_Irecv` then all `MPI_Isend` - * (unique per-direction tags), a single `MPI_Waitall`, then unpacks - * — overlapping the round-trips instead of serializing one blocking - * `MPI_Sendrecv` per direction. Self-communication (periodic single - * domain) is a local copy. Slicing/packing mirrors - * `comm::CommunicateField`. The tag pairs A's send in a direction - * with B's receive in the same direction (the matching neighbor), - * so it is unique per (rank-pair, direction). - */ - template - inline void CommunicateFieldBatched(unsigned int my_idx, - ndfield_t& fld, - const std::vector>& dirs, - const cell_range_t& comps) { - static constexpr unsigned short Dp1 = static_cast(D) + 1; - using buf_t = ndarray_t; - const ncells_t ncomp = comps.second - comps.first; - const auto ndirs = dirs.size(); - - const auto is_self = [&](const FieldCommDir& dd) { - return (dd.send_ind == my_idx) && (dd.recv_ind == my_idx); - }; - const auto ext = [](const std::vector& sl, int d) -> ncells_t { - return sl[d].second - sl[d].first; - }; - const auto make_buf = [&](const char* lbl, - const std::vector& sl) -> buf_t { - if constexpr (D == Dim::_1D) { - return buf_t(lbl, ext(sl, 0), ncomp); - } else if constexpr (D == Dim::_2D) { - return buf_t(lbl, ext(sl, 0), ext(sl, 1), ncomp); - } else { - return buf_t(lbl, ext(sl, 0), ext(sl, 1), ext(sl, 2), ncomp); - } - }; - const auto count = [](const buf_t& b) -> int { - ncells_t n = 1; - for (auto d { 0 }; d < static_cast(Dp1); ++d) { - n *= static_cast(b.extent(d)); - } - return static_cast(n); - }; - const auto pack = [&](buf_t& b, const std::vector& sl) { - if constexpr (D == Dim::_1D) { - Kokkos::deep_copy(b, Kokkos::subview(fld, sl[0], comps)); - } else if constexpr (D == Dim::_2D) { - Kokkos::deep_copy(b, Kokkos::subview(fld, sl[0], sl[1], comps)); - } else { - Kokkos::deep_copy(b, Kokkos::subview(fld, sl[0], sl[1], sl[2], comps)); - } - }; - const auto unpack = [&](const buf_t& b, const std::vector& sl) { - if constexpr (D == Dim::_1D) { - Kokkos::deep_copy(Kokkos::subview(fld, sl[0], comps), b); - } else if constexpr (D == Dim::_2D) { - Kokkos::deep_copy(Kokkos::subview(fld, sl[0], sl[1], comps), b); - } else { - Kokkos::deep_copy(Kokkos::subview(fld, sl[0], sl[1], sl[2], comps), b); - } - }; - const auto self_copy = [&](const std::vector& ssl, - const std::vector& rsl) { - if constexpr (D == Dim::_1D) { - Kokkos::deep_copy(Kokkos::subview(fld, rsl[0], comps), - Kokkos::subview(fld, ssl[0], comps)); - } else if constexpr (D == Dim::_2D) { - Kokkos::deep_copy(Kokkos::subview(fld, rsl[0], rsl[1], comps), - Kokkos::subview(fld, ssl[0], ssl[1], comps)); - } else { - Kokkos::deep_copy(Kokkos::subview(fld, rsl[0], rsl[1], rsl[2], comps), - Kokkos::subview(fld, ssl[0], ssl[1], ssl[2], comps)); - } - }; - - std::vector send_buf(ndirs), recv_buf(ndirs); -#if defined(DEVICE_ENABLED) && !defined(GPU_AWARE_MPI) - using hbuf_t = typename buf_t::host_mirror_type; - std::vector send_h(ndirs), recv_h(ndirs); -#endif - - // phase 1: self copies, pack sends, allocate recvs - for (auto i { 0u }; i < ndirs; ++i) { - const auto& dd = dirs[i]; - if (is_self(dd)) { - self_copy(dd.send_slice, dd.recv_slice); - continue; - } - if (dd.recv_rank >= 0) { - recv_buf[i] = make_buf("recv_fld", dd.recv_slice); -#if defined(DEVICE_ENABLED) && !defined(GPU_AWARE_MPI) - recv_h[i] = Kokkos::create_mirror_view(recv_buf[i]); -#endif - } - if (dd.send_rank >= 0) { - send_buf[i] = make_buf("send_fld", dd.send_slice); - pack(send_buf[i], dd.send_slice); -#if defined(DEVICE_ENABLED) && !defined(GPU_AWARE_MPI) - send_h[i] = Kokkos::create_mirror_view(send_buf[i]); - Kokkos::deep_copy(send_h[i], send_buf[i]); -#endif - } - } - // drain packs (and device->host staging) before MPI reads the buffers - Kokkos::fence("CommunicateFieldBatched: pre-MPI"); - - // phase 2: post all recvs, then all sends - std::vector reqs; - reqs.reserve(2 * ndirs); - for (auto i { 0u }; i < ndirs; ++i) { - const auto& dd = dirs[i]; - if (is_self(dd) || dd.recv_rank < 0) { - continue; - } - reqs.emplace_back(); -#if defined(DEVICE_ENABLED) && !defined(GPU_AWARE_MPI) - MPI_Irecv(recv_h[i].data(), count(recv_buf[i]), mpi::get_type(), - dd.recv_rank, dd.tag, MPI_COMM_WORLD, &reqs.back()); -#else - MPI_Irecv(recv_buf[i].data(), count(recv_buf[i]), mpi::get_type(), - dd.recv_rank, dd.tag, MPI_COMM_WORLD, &reqs.back()); -#endif - } - for (auto i { 0u }; i < ndirs; ++i) { - const auto& dd = dirs[i]; - if (is_self(dd) || dd.send_rank < 0) { - continue; - } - reqs.emplace_back(); -#if defined(DEVICE_ENABLED) && !defined(GPU_AWARE_MPI) - MPI_Isend(send_h[i].data(), count(send_buf[i]), mpi::get_type(), - dd.send_rank, dd.tag, MPI_COMM_WORLD, &reqs.back()); -#else - MPI_Isend(send_buf[i].data(), count(send_buf[i]), mpi::get_type(), - dd.send_rank, dd.tag, MPI_COMM_WORLD, &reqs.back()); -#endif - } - - // phase 3: complete all transfers - if (not reqs.empty()) { - MPI_Waitall(static_cast(reqs.size()), reqs.data(), MPI_STATUSES_IGNORE); - } - - // phase 4: unpack received ghosts - for (auto i { 0u }; i < ndirs; ++i) { - const auto& dd = dirs[i]; - if (is_self(dd) || dd.recv_rank < 0) { - continue; - } -#if defined(DEVICE_ENABLED) && !defined(GPU_AWARE_MPI) - Kokkos::deep_copy(recv_buf[i], recv_h[i]); -#endif - unpack(recv_buf[i], dd.recv_slice); - } - } - } // namespace comm #endif // FRAMEWORK_DOMAIN_COMM_MPI_HPP diff --git a/src/framework/domain/metadomain_comm.cpp b/src/framework/domain/metadomain_comm.cpp index d6061d7bf..f8c0f60d7 100644 --- a/src/framework/domain/metadomain_comm.cpp +++ b/src/framework/domain/metadomain_comm.cpp @@ -273,71 +273,6 @@ namespace ntt { comp_range_cur = cell_range_t(cur::jx1, cur::jx3 + 1); } // traverse in all directions and send/recv the fields -#if defined(MPI_ENABLED) - // Batched non-blocking exchange: collect the per-direction params once, - // then overlap all directions for each field (post all Irecv/Isend with - // per-direction tags + one Waitall) instead of a blocking Sendrecv per - // direction. The #else branch is the single-rank non-MPI path. - { - std::vector> dirs; - dirs.reserve(dir::Directions::all.size()); - for (auto& direction : dir::Directions::all) { - const auto [send_params, recv_params] = GetSendRecvParams(this, - domain, - direction, - false); - const auto [send_indrank, send_slice] = send_params; - const auto [recv_indrank, recv_slice] = recv_params; - const auto [send_ind, send_rank] = send_indrank; - const auto [recv_ind, recv_rank] = recv_indrank; - if (send_rank < 0 and recv_rank < 0) { - continue; - } - comm::FieldCommDir fcd; - fcd.send_rank = send_rank; - fcd.recv_rank = recv_rank; - fcd.send_ind = send_ind; - fcd.recv_ind = recv_ind; - fcd.send_slice = send_slice; - fcd.recv_slice = recv_slice; - fcd.tag = static_cast(mpi::PrtlSendTag::dir2tag(direction)); - dirs.push_back(std::move(fcd)); - } - if (comm_em) { - comm::CommunicateFieldBatched(domain.index(), - domain.fields.em, - dirs, - comp_range_fld); - } - if constexpr (S == SimEngine::GRPIC) { - if (comm_aux) { - comm::CommunicateFieldBatched(domain.index(), - domain.fields.aux, - dirs, - comp_range_fld); - } - if (comm_em0) { - comm::CommunicateFieldBatched(domain.index(), - domain.fields.em0, - dirs, - comp_range_fld); - } - if (comm_j) { - comm::CommunicateFieldBatched(domain.index(), - domain.fields.cur0, - dirs, - comp_range_cur); - } - } else { - if (comm_j) { - comm::CommunicateFieldBatched(domain.index(), - domain.fields.cur, - dirs, - comp_range_cur); - } - } - } -#else for (auto& direction : dir::Directions::all) { const auto [send_params, recv_params] = GetSendRecvParams(this, domain, direction, false); @@ -429,7 +364,6 @@ namespace ntt { } } } -#endif // MPI_ENABLED } template From 88914209ae20bff930cf6bedacb0653053fa81f3 Mon Sep 17 00:00:00 2001 From: Ludwig Boess Date: Sun, 21 Jun 2026 16:51:00 -0400 Subject: [PATCH 018/125] bugfix in current deposit --- src/kernels/currents_deposit.hpp | 20 ++++++++++++++++---- 1 file changed, 16 insertions(+), 4 deletions(-) diff --git a/src/kernels/currents_deposit.hpp b/src/kernels/currents_deposit.hpp index b92b02423..98fa0b661 100644 --- a/src/kernels/currents_deposit.hpp +++ b/src/kernels/currents_deposit.hpp @@ -1042,8 +1042,12 @@ namespace kernel { [&](int g_i1, int comp, real_t v) { if (to_scratch) { Kokkos::atomic_add(&scr(g_i1 - origin_J1_low, comp), v); - //} else if (g_i1 >= 0 and g_i1 < j_ext1) { - } else { + } else if (g_i1 >= 0 and g_i1 < j_ext1) { + // Bounds-clip the escape-valve write against J's storage, + // exactly as the cooperative flush does. Cells past the + // ghost stripe are re-supplied by SynchronizeFields(J); an + // unclipped write here faults the GPU (an escaped boundary + // particle's stencil can reach past j_ext1). Kokkos::atomic_add(&J(g_i1, comp), v); } }); @@ -1115,7 +1119,11 @@ namespace kernel { Kokkos::atomic_add( &scr(g_i1 - origin_J1_low, g_i2 - origin_J2_low, comp), v); - } else { + } else if (g_i1 >= 0 and g_i1 < j_ext1 and g_i2 >= 0 and + g_i2 < j_ext2) { + // Bounds-clip as the cooperative flush does; an unclipped + // escape-valve write faults the GPU when an escaped boundary + // particle's stencil reaches past j_ext. Kokkos::atomic_add(&J(g_i1, g_i2, comp), v); } }); @@ -1195,7 +1203,11 @@ namespace kernel { g_i3 - origin_J3_low, comp), v); - } else { + } else if (g_i1 >= 0 and g_i1 < j_ext1 and g_i2 >= 0 and + g_i2 < j_ext2 and g_i3 >= 0 and g_i3 < j_ext3) { + // Bounds-clip as the cooperative flush does; an unclipped + // escape-valve write faults the GPU when an escaped boundary + // particle's stencil reaches past j_ext. Kokkos::atomic_add(&J(g_i1, g_i2, g_i3, comp), v); } }); From 4f3a7625ab055fec718fdac32bb1f588c980777c Mon Sep 17 00:00:00 2001 From: Ludwig Boess Date: Sun, 21 Jun 2026 19:38:24 -0400 Subject: [PATCH 019/125] double-buffer radix sort to drop the N-sized temp (fixes device OOM at scale) --- src/global/utils/sort_dispatch.h | 74 +++++++++++++++++++++----------- 1 file changed, 49 insertions(+), 25 deletions(-) diff --git a/src/global/utils/sort_dispatch.h b/src/global/utils/sort_dispatch.h index cfc389464..27a20a910 100644 --- a/src/global/utils/sort_dispatch.h +++ b/src/global/utils/sort_dispatch.h @@ -137,8 +137,8 @@ namespace ntt::sort_helpers { n, KOKKOS_LAMBDA(const npart_t i) { perm_v(i) = i; }); - // Out-of-place radix sort bounded to the significant key bits. The - // _out buffers and temp storage are transient (freed at scope exit). + // Radix sort bounded to the significant key bits. The _out buffers and + // temp storage are transient (freed at scope exit). array_t keys_out("tile_keys_sorted", n); prtl_perm_t perm_out("tile_perm_sorted", n); const int end_bit = static_cast(significant_bits(n_bins)); @@ -146,13 +146,20 @@ namespace ntt::sort_helpers { exec.fence("sort_by_key_dispatch Thrust: pre-sort"); auto stream = exec.cuda_stream(); + // DoubleBuffer radix sort: cub ping-pongs between the supplied + // (current, alternate) buffer pairs, so `temp_bytes` holds only the + // histograms (~MB) instead of an internal N-sized alternate (~8*N bytes). + // Nearly halves the sort's transient memory vs the plain out-of-place + // form, which matters at high npart where the N-sized temp can fail to + // allocate (device OOM at scale). + cub::DoubleBuffer d_keys(keys.data(), keys_out.data()); + cub::DoubleBuffer d_perm(perm.data(), perm_out.data()); + std::size_t temp_bytes = 0; auto err = cub::DeviceRadixSort::SortPairs(nullptr, temp_bytes, - keys.data(), - keys_out.data(), - perm.data(), - perm_out.data(), + d_keys, + d_perm, n, 0, end_bit, @@ -164,10 +171,8 @@ namespace ntt::sort_helpers { (temp_bytes == 0u) ? std::size_t { 1 } : temp_bytes); err = cub::DeviceRadixSort::SortPairs(temp.data(), temp_bytes, - keys.data(), - keys_out.data(), - perm.data(), - perm_out.data(), + d_keys, + d_perm, n, 0, end_bit, @@ -177,10 +182,16 @@ namespace ntt::sort_helpers { HERE); exec.fence("sort_by_key_dispatch Thrust: post-sort"); - // Publish sorted keys back in place; swap the sorted permutation in. - auto keys_nc = keys; // non-const handle aliasing the same storage - Kokkos::deep_copy(keys_nc, keys_out); - perm = perm_out; + // Publish results from whichever buffer cub left as Current() (depends on + // the pass count): copy sorted keys back into `keys`' storage if they + // ended up in the alternate, and point `perm` at its current buffer. + if (d_keys.Current() != keys.data()) { + auto keys_nc = keys; // non-const handle aliasing the same storage + Kokkos::deep_copy(keys_nc, keys_out); + } + if (d_perm.Current() == perm_out.data()) { + perm = perm_out; + } } #endif @@ -210,13 +221,20 @@ namespace ntt::sort_helpers { exec.fence("sort_by_key_dispatch Rocthrust: pre-sort"); auto stream = exec.hip_stream(); + // double_buffer radix sort: rocprim ping-pongs between the supplied + // (current, alternate) buffer pairs, so `temp_storage` holds only the + // histograms (~MB) instead of an internal N-sized alternate (~8*N bytes, + // the dominant `rocprim_radix_temp`). This nearly halves the sort's + // transient memory vs the plain out-of-place form, which matters at high + // npart where that N-sized temp can fail to allocate (device OOM at scale). + rocprim::double_buffer d_keys(keys.data(), keys_out.data()); + rocprim::double_buffer d_perm(perm.data(), perm_out.data()); + std::size_t temp_bytes = 0; auto err = rocprim::radix_sort_pairs(nullptr, temp_bytes, - keys.data(), - keys_out.data(), - perm.data(), - perm_out.data(), + d_keys, + d_perm, static_cast(n), 0u, end_bit, @@ -228,10 +246,8 @@ namespace ntt::sort_helpers { (temp_bytes == 0u) ? std::size_t { 1 } : temp_bytes); err = rocprim::radix_sort_pairs(temp.data(), temp_bytes, - keys.data(), - keys_out.data(), - perm.data(), - perm_out.data(), + d_keys, + d_perm, static_cast(n), 0u, end_bit, @@ -241,9 +257,17 @@ namespace ntt::sort_helpers { HERE); exec.fence("sort_by_key_dispatch Rocthrust: post-sort"); - auto keys_nc = keys; // non-const handle aliasing the same storage - Kokkos::deep_copy(keys_nc, keys_out); - perm = perm_out; + // Publish results from whichever buffer rocprim left as `current()` + // (depends on the pass count). If the sorted keys ended up in the + // alternate, copy them back into `keys`' storage so downstream + // (compute_tile_offsets) sees them; point `perm` at its current buffer. + if (d_keys.current() != keys.data()) { + auto keys_nc = keys; // non-const handle aliasing the same storage + Kokkos::deep_copy(keys_nc, keys_out); + } + if (d_perm.current() == perm_out.data()) { + perm = perm_out; + } } #endif From 94ba9cdb3e484dbffed0a681bb1222d873b207af Mon Sep 17 00:00:00 2001 From: Ludwig Boess Date: Wed, 24 Jun 2026 18:26:32 -0400 Subject: [PATCH 020/125] fix printing of total particles --- src/global/utils/diag.cpp | 11 ++++++----- 1 file changed, 6 insertions(+), 5 deletions(-) diff --git a/src/global/utils/diag.cpp b/src/global/utils/diag.cpp index 4cf763faf..604a872ae 100644 --- a/src/global/utils/diag.cpp +++ b/src/global/utils/diag.cpp @@ -17,6 +17,7 @@ #include #endif // MPI_ENABLED +#include #include #include #include @@ -26,8 +27,8 @@ namespace diag { auto npart_stats(npart_t npart, npart_t maxnpart) - -> std::vector> { - auto stats = std::vector>(); + -> std::vector> { + auto stats = std::vector>(); const auto percentage = [](npart_t part, npart_t maxpart) -> unsigned short { return static_cast( 100.0f * static_cast(part) / static_cast(maxpart)); @@ -59,9 +60,9 @@ namespace diag { if (rank != MPI_ROOT_RANK) { return stats; } - const npart_t tot_npart = std::accumulate(mpi_npart.begin(), - mpi_npart.end(), - static_cast(0)); + const std::size_t tot_npart = std::accumulate(mpi_npart.begin(), + mpi_npart.end(), + static_cast(0)); const npart_t max_idx = std::distance( mpi_npart.begin(), std::max_element(mpi_npart.begin(), mpi_npart.end())); From 9a4333e1798a6965054dc80407ac64f7ff3774e8 Mon Sep 17 00:00:00 2001 From: Ludwig Boess Date: Wed, 24 Jun 2026 18:47:03 -0400 Subject: [PATCH 021/125] use a persistent buffer per sort to reduce memory overhead --- src/framework/containers/particles.h | 19 +-- src/framework/containers/particles_sort.cpp | 131 ++++++++++++++------ 2 files changed, 101 insertions(+), 49 deletions(-) diff --git a/src/framework/containers/particles.h b/src/framework/containers/particles.h index 511a07a62..c3fa93137 100644 --- a/src/framework/containers/particles.h +++ b/src/framework/containers/particles.h @@ -311,14 +311,17 @@ namespace ntt { private: /** * @brief Apply a particle-index permutation (built by oneDPL/Thrust - * sort_by_key) to the SoA member arrays. Each member is - * gathered into a fresh full-capacity buffer whose handle is - * then swapped in (no copy-back), one buffer at a time, fenced - * before the old storage is released. The *_prev arrays are - * intentionally not permuted (overwritten by the next push - * before any read). Only compiled when a vendor sort backend - * is enabled; the BinSort path applies the permutation in - * place via `sorter.sort(view)` instead. + * sort_by_key) to the SoA member arrays. Members are gathered + * through `perm` into a reusable `n`-sized scratch buffer + * (one per member type, shared across members of that type) + * and copied back in place, so the large persistent member + * arrays keep their storage/address and the gather makes a + * handful of transient allocations instead of one maxnpart + * buffer per member. The *_prev arrays are intentionally not + * permuted (overwritten by the next push before any read). + * Only compiled when a vendor sort backend is enabled; the + * BinSort path applies the permutation in place via + * `sorter.sort(view)` instead. */ void apply_permutation_to_soa(const prtl_perm_t& perm); diff --git a/src/framework/containers/particles_sort.cpp b/src/framework/containers/particles_sort.cpp index 3db6c1438..8261fe5b4 100644 --- a/src/framework/containers/particles_sort.cpp +++ b/src/framework/containers/particles_sort.cpp @@ -474,53 +474,73 @@ namespace ntt { #if defined(TEAM_POLICY_USE_VENDOR_SORT) namespace permute_helpers { - // Permute a 1D SoA member array `arr` by `perm`. Gathers into a - // fresh buffer allocated at the member's full capacity (maxnpart), - // then swaps the View handle in. This avoids the redundant copy-back - // pass of the old gather-then-deep_copy approach (~2x less HBM - // traffic). Allocating at full capacity preserves the member's spare - // room for injection; the untouched tail [n, capacity) is - // zero-initialized by Kokkos (cleaner than the stale values the old - // deep_copy left there). The fence drains the gather (which reads the - // old storage) before the swap drops the last reference to it. + // Permute a 1D SoA member `arr` by `perm` in place, using a + // caller-owned reusable `scratch` buffer. Gathers the live prefix + // `arr[perm[0..n)]` into `scratch[0..n)`, then copies it back into + // `arr[0..n)`. Unlike the old swap-the-handle approach, `arr` keeps + // its original storage and full maxnpart capacity, so the large + // persistent member arrays never change address between sorts (the + // dominant source of allocator churn / fragmentation on ROCm), and + // `scratch` is shared across every member of the same type within one + // sort -- one transient allocation per type group instead of a fresh + // maxnpart buffer per member. `scratch` is sized to the live count + // `n` (not maxnpart), which also lowers the gather's peak transient. + // The tail `arr[n, capacity)` is left untouched (stale, never read: + // consumers iterate `[0, npart())`). Cost vs the swap: one extra + // copy-back pass (~2x HBM traffic), negligible when sorting is a + // small fraction of the step. The fences keep the shared scratch from + // being overwritten by the next member's gather before this member's + // copy-back has drained it. template - inline void permute_1d_swap(V& arr, + inline void permute_1d_into(V& arr, + const V& scratch, const prtl_perm_t& perm, npart_t n) { if (n == 0u) { return; } - V buf(arr.label(), arr.extent(0)); auto perm_v = perm; auto arr_v = arr; + auto buf_v = scratch; Kokkos::parallel_for( "Permute1D", n, - KOKKOS_LAMBDA(const npart_t p) { buf(p) = arr_v(perm_v(p)); }); - Kokkos::fence("permute_1d_swap: end"); - arr = buf; + KOKKOS_LAMBDA(const npart_t p) { buf_v(p) = arr_v(perm_v(p)); }); + Kokkos::fence("permute_1d_into: gather"); + Kokkos::deep_copy( + Kokkos::subview(arr, std::make_pair(static_cast(0), n)), + Kokkos::subview(scratch, std::make_pair(static_cast(0), n))); + Kokkos::fence("permute_1d_into: copy-back"); } - // 2D analogue for `pld_r` / `pld_i`. + // 2D analogue for `pld_r` / `pld_i` (`scratch` is `(>= n, ncols)`). template - inline void permute_2d_swap(V& arr, + inline void permute_2d_into(V& arr, + const V& scratch, const prtl_perm_t& perm, npart_t n, npart_t ncols) { if (n == 0u or ncols == 0u) { return; } - V buf(arr.label(), arr.extent(0), arr.extent(1)); auto perm_v = perm; auto arr_v = arr; + auto buf_v = scratch; Kokkos::parallel_for( "Permute2D", CreateParticleRangePolicy({ 0u, 0u }, { n, ncols }), KOKKOS_LAMBDA(const npart_t p, const npart_t l) { - buf(p, l) = arr_v(perm_v(p), l); + buf_v(p, l) = arr_v(perm_v(p), l); }); - Kokkos::fence("permute_2d_swap: end"); - arr = buf; + Kokkos::fence("permute_2d_into: gather"); + Kokkos::deep_copy( + Kokkos::subview(arr, + std::make_pair(static_cast(0), n), + Kokkos::ALL), + Kokkos::subview(scratch, + std::make_pair(static_cast(0), n), + Kokkos::ALL)); + Kokkos::fence("permute_2d_into: copy-back"); } } // namespace permute_helpers @@ -532,8 +552,8 @@ namespace ntt { return; } - using permute_helpers::permute_1d_swap; - using permute_helpers::permute_2d_swap; + using permute_helpers::permute_1d_into; + using permute_helpers::permute_2d_into; // The *_prev arrays (i{1,2,3}_prev, dx{1,2,3}_prev) are intentionally // NOT permuted. SortSpatially runs at the very end of the step loop @@ -551,31 +571,60 @@ namespace ntt { // prev field saved to the checkpoint differs from the old code. // Permuting prev would therefore reorder data that is overwritten // before it is ever observed. - if constexpr (D == Dim::_1D or D == Dim::_2D or D == Dim::_3D) { - permute_1d_swap(i1, perm, n); - permute_1d_swap(dx1, perm, n); + // + // Each block below allocates a single `n`-sized scratch buffer that + // is reused for every member of that type, then freed before the next + // block allocates its own. The whole gather thus makes one transient + // allocation per type group (int / prtldx_t / real_t / short / each + // payload) instead of one fresh maxnpart buffer per member, and peak + // transient is a single n-sized scratch at a time. + { + array_t scratch { "perm_scratch_int", n }; + if constexpr (D == Dim::_1D or D == Dim::_2D or D == Dim::_3D) { + permute_1d_into(i1, scratch, perm, n); + } + if constexpr (D == Dim::_2D or D == Dim::_3D) { + permute_1d_into(i2, scratch, perm, n); + } + if constexpr (D == Dim::_3D) { + permute_1d_into(i3, scratch, perm, n); + } } - if constexpr (D == Dim::_2D or D == Dim::_3D) { - permute_1d_swap(i2, perm, n); - permute_1d_swap(dx2, perm, n); + { + array_t scratch { "perm_scratch_prtldx", n }; + if constexpr (D == Dim::_1D or D == Dim::_2D or D == Dim::_3D) { + permute_1d_into(dx1, scratch, perm, n); + } + if constexpr (D == Dim::_2D or D == Dim::_3D) { + permute_1d_into(dx2, scratch, perm, n); + } + if constexpr (D == Dim::_3D) { + permute_1d_into(dx3, scratch, perm, n); + } } - if constexpr (D == Dim::_3D) { - permute_1d_swap(i3, perm, n); - permute_1d_swap(dx3, perm, n); - } - permute_1d_swap(ux1, perm, n); - permute_1d_swap(ux2, perm, n); - permute_1d_swap(ux3, perm, n); - permute_1d_swap(weight, perm, n); - permute_1d_swap(tag, perm, n); - if constexpr (D == Dim::_2D and C != Coord::Cartesian) { - permute_1d_swap(phi, perm, n); + { + array_t scratch { "perm_scratch_real", n }; + permute_1d_into(ux1, scratch, perm, n); + permute_1d_into(ux2, scratch, perm, n); + permute_1d_into(ux3, scratch, perm, n); + permute_1d_into(weight, scratch, perm, n); + if constexpr (D == Dim::_2D and C != Coord::Cartesian) { + permute_1d_into(phi, scratch, perm, n); + } + } + { + array_t scratch { "perm_scratch_tag", n }; + permute_1d_into(tag, scratch, perm, n); } if (npld_r() > 0) { - permute_2d_swap(pld_r, perm, n, static_cast(npld_r())); + const auto ncols = static_cast(npld_r()); + array_t scratch { "perm_scratch_pld_r", n, ncols }; + permute_2d_into(pld_r, scratch, perm, n, ncols); } if (npld_i() > 0) { - permute_2d_swap(pld_i, perm, n, static_cast(npld_i())); + const auto ncols = static_cast(npld_i()); + array_t scratch { "perm_scratch_pld_i", n, ncols }; + permute_2d_into(pld_i, scratch, perm, n, ncols); } } #endif // TEAM_POLICY_USE_VENDOR_SORT From 216940ce14c55be31f9851ab6217fa11ca85e2d9 Mon Sep 17 00:00:00 2001 From: Ludwig Boess Date: Wed, 24 Jun 2026 18:47:17 -0400 Subject: [PATCH 022/125] added the option to explicitly disable vendor sort --- CMakeLists.txt | 98 ++++++++++++++++++++--------------- cmake/defaults.cmake | 13 +++++ cmake/report.cmake | 10 ++++ src/global/utils/reporter.cpp | 7 +++ 4 files changed, 87 insertions(+), 41 deletions(-) diff --git a/CMakeLists.txt b/CMakeLists.txt index f8409209b..b9bbb1767 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -71,6 +71,10 @@ set(team_policy_drift ${default_team_policy_drift} CACHE STRING "team_policy tiled-deposit scratch halo drift in cells (max cells a particle may move between two sorts). Sizes the deposit scratch halo only; the sort cadence is set at runtime via spatial_sorting_interval. Default 1.") +set(vendor_sort + ${default_vendor_sort} + CACHE BOOL + "Use the vendor sort_by_key (oneDPL/Thrust/rocThrust) for the team_policy spatial sort when available. OFF forces the Kokkos::BinSort fallback, which sorts each SoA member in place (lower peak memory, no maxnpart gather buffer) at the cost of sort speed.") # -------------------------- Compilation settings -------------------------- # set(CMAKE_CXX_STANDARD 20) @@ -169,53 +173,65 @@ if(${team_policy}) # `spatial_sorting_interval`. Defaults to 1 (the sorted-every-step case). add_compile_options("-D TEAM_POLICY_DRIFT=${team_policy_drift}") - # Vendor sort: oneDPL on SYCL, Thrust on CUDA. Used automatically - # when found; falls back to Kokkos::BinSort otherwise. - if("${Kokkos_DEVICES}" MATCHES "SYCL") - find_package(oneDPL QUIET) - if(oneDPL_FOUND) - message(STATUS "team_policy: oneDPL found, enabling SYCL sort_by_key") - add_compile_options("-D ONEDPL_ENABLED") - set(DEPENDENCIES ${DEPENDENCIES} oneDPL) - else() - message(STATUS "team_policy: oneDPL not found; using BinSort fallback " - "for SYCL sort_by_key") + # Vendor sort: oneDPL on SYCL, Thrust on CUDA, rocThrust/rocprim on HIP. + # When `vendor_sort` is ON (default) the available library is detected + # and used; the spatial sort then builds a single permutation that + # gathers all SoA members. When `vendor_sort` is OFF, or no library is + # found, the code falls back to Kokkos::BinSort, which sorts each member + # in place -- lower peak memory and no maxnpart gather buffer, at the + # cost of sort speed (negligible when sorting is a small fraction of the + # step). The `vendor_sort` knob lets you force the BinSort fallback even + # when a vendor library is present. + if(${vendor_sort}) + if("${Kokkos_DEVICES}" MATCHES "SYCL") + find_package(oneDPL QUIET) + if(oneDPL_FOUND) + message(STATUS "team_policy: oneDPL found, enabling SYCL sort_by_key") + add_compile_options("-D ONEDPL_ENABLED") + set(DEPENDENCIES ${DEPENDENCIES} oneDPL) + else() + message(STATUS "team_policy: oneDPL not found; using BinSort fallback " + "for SYCL sort_by_key") + endif() endif() - endif() - if("${Kokkos_DEVICES}" MATCHES "CUDA") - find_package(Thrust QUIET) - if(Thrust_FOUND) - message(STATUS "team_policy: Thrust enabled for CUDA sort_by_key") - add_compile_options("-D THRUST_ENABLED") - else() - message(STATUS "team_policy: Thrust not found; using BinSort fallback " - "for CUDA sort_by_key") + if("${Kokkos_DEVICES}" MATCHES "CUDA") + find_package(Thrust QUIET) + if(Thrust_FOUND) + message(STATUS "team_policy: Thrust enabled for CUDA sort_by_key") + add_compile_options("-D THRUST_ENABLED") + else() + message(STATUS "team_policy: Thrust not found; using BinSort fallback " + "for CUDA sort_by_key") + endif() endif() - endif() - if("${Kokkos_DEVICES}" MATCHES "HIP") - # rocThrust ships with ROCm. The HIP sort_by_key path uses rocprim's - # bounded-bit radix sort directly (rocprim is rocThrust's own - # dependency, so its headers come in transitively; we find it - # explicitly to keep the include path robust). This builds a single - # permutation that gathers all SoA members, instead of the legacy - # per-member Kokkos::BinSort path which allocates a fresh - # `sorted_values` buffer for every member every step (the dominant - # source of allocator churn / fragmentation on ROCm). - find_package(rocthrust QUIET) - if(rocthrust_FOUND) - message(STATUS "team_policy: rocThrust enabled for HIP sort_by_key") - add_compile_options("-D ROCTHRUST_ENABLED") - set(DEPENDENCIES ${DEPENDENCIES} roc::rocthrust) - find_package(rocprim QUIET) - if(rocprim_FOUND) - set(DEPENDENCIES ${DEPENDENCIES} roc::rocprim) + if("${Kokkos_DEVICES}" MATCHES "HIP") + # rocThrust ships with ROCm. The HIP sort_by_key path uses rocprim's + # bounded-bit radix sort directly (rocprim is rocThrust's own + # dependency, so its headers come in transitively; we find it + # explicitly to keep the include path robust). This builds a single + # permutation that gathers all SoA members, instead of the legacy + # per-member Kokkos::BinSort path which allocates a fresh + # `sorted_values` buffer for every member every step (the dominant + # source of allocator churn / fragmentation on ROCm). + find_package(rocthrust QUIET) + if(rocthrust_FOUND) + message(STATUS "team_policy: rocThrust enabled for HIP sort_by_key") + add_compile_options("-D ROCTHRUST_ENABLED") + set(DEPENDENCIES ${DEPENDENCIES} roc::rocthrust) + find_package(rocprim QUIET) + if(rocprim_FOUND) + set(DEPENDENCIES ${DEPENDENCIES} roc::rocprim) + endif() + else() + message(STATUS "team_policy: rocThrust not found; using BinSort " + "fallback for HIP sort_by_key") endif() - else() - message(STATUS "team_policy: rocThrust not found; using BinSort " - "fallback for HIP sort_by_key") endif() + else() + message(STATUS "team_policy: vendor_sort=OFF; forcing Kokkos::BinSort " + "fallback for spatial sort_by_key") endif() endif() diff --git a/cmake/defaults.cmake b/cmake/defaults.cmake index 01427921f..c03ebaf81 100644 --- a/cmake/defaults.cmake +++ b/cmake/defaults.cmake @@ -104,6 +104,19 @@ else() endif() set_property(CACHE default_team_policy PROPERTY TYPE BOOL) +if(DEFINED ENV{Entity_ENABLE_VENDOR_SORT}) + set(default_vendor_sort + $ENV{Entity_ENABLE_VENDOR_SORT} + CACHE INTERNAL + "Default flag for vendor sort_by_key (oneDPL/Thrust/rocThrust)") +else() + set(default_vendor_sort + ON + CACHE INTERNAL + "Default flag for vendor sort_by_key (oneDPL/Thrust/rocThrust)") +endif() +set_property(CACHE default_vendor_sort PROPERTY TYPE BOOL) + set(default_team_policy_tile_size 8 CACHE INTERNAL "Default tile edge length in cells for team_policy") diff --git a/cmake/report.cmake b/cmake/report.cmake index d90dfb085..d972fafc4 100644 --- a/cmake/report.cmake +++ b/cmake/report.cmake @@ -150,6 +150,15 @@ if(${team_policy}) "${Blue}" TEAM_POLICY_DRIFT_REPORT 46) + printchoices( + "Vendor sort" + "vendor_sort" + "${ON_OFF_VALUES}" + ${vendor_sort} + ON + "${Green}" + VENDOR_SORT_REPORT + 46) endif() printchoices( "Debug mode" @@ -230,6 +239,7 @@ string(APPEND REPORT_TEXT " " ${TEAM_POLICY_REPORT} "\n") if(${team_policy}) string(APPEND REPORT_TEXT " " ${TEAM_POLICY_TILE_SIZE_REPORT} "\n") string(APPEND REPORT_TEXT " " ${TEAM_POLICY_DRIFT_REPORT} "\n") + string(APPEND REPORT_TEXT " " ${VENDOR_SORT_REPORT} "\n") endif() string( diff --git a/src/global/utils/reporter.cpp b/src/global/utils/reporter.cpp index a4b10eee6..6113a89c7 100644 --- a/src/global/utils/reporter.cpp +++ b/src/global/utils/reporter.cpp @@ -253,6 +253,13 @@ namespace reporter { #if defined(TEAM_POLICY) AddParam(report, 4, "TEAM_POLICY", "%s", "ON"); + #if (defined(SYCL_ENABLED) && defined(ONEDPL_ENABLED)) || \ + (defined(CUDA_ENABLED) && defined(THRUST_ENABLED)) || \ + (defined(HIP_ENABLED) && defined(ROCTHRUST_ENABLED)) + AddParam(report, 4, "VENDOR_SORT", "%s", "ON"); + #else + AddParam(report, 4, "VENDOR_SORT", "%s", "OFF (BinSort)"); + #endif #else AddParam(report, 4, "TEAM_POLICY", "%s", "OFF"); #endif From af942c82eeda02c30e15004069caf83b57583dd4 Mon Sep 17 00:00:00 2001 From: Ludwig Boess Date: Wed, 24 Jun 2026 19:08:50 -0400 Subject: [PATCH 023/125] remove dead particles after every sort --- src/framework/containers/particles.h | 8 ++++- src/framework/containers/particles_sort.cpp | 35 +++++++++++++++------ 2 files changed, 33 insertions(+), 10 deletions(-) diff --git a/src/framework/containers/particles.h b/src/framework/containers/particles.h index c3fa93137..fe4a0b17f 100644 --- a/src/framework/containers/particles.h +++ b/src/framework/containers/particles.h @@ -322,8 +322,14 @@ namespace ntt { * Only compiled when a vendor sort backend is enabled; the * BinSort path applies the permutation in place via * `sorter.sort(view)` instead. + * @param perm Permutation: sorted position -> pre-sort slot index. + * @param n Number of leading (alive) particles to gather. Pass + * `npart_partitioned` so only the alive set is moved into + * `[0, n)` (the dead were binned to the sentinel tile and sort + * to the tail); the caller then drops the dead tail via + * `set_npart(n)`. */ - void apply_permutation_to_soa(const prtl_perm_t& perm); + void apply_permutation_to_soa(const prtl_perm_t& perm, npart_t n); public: #endif diff --git a/src/framework/containers/particles_sort.cpp b/src/framework/containers/particles_sort.cpp index 8261fe5b4..5e9ae4147 100644 --- a/src/framework/containers/particles_sort.cpp +++ b/src/framework/containers/particles_sort.cpp @@ -319,12 +319,12 @@ namespace ntt { const auto slice = prtl_slice_t(0, npart_local); #if defined(TEAM_POLICY_USE_VENDOR_SORT) - // Vendor path: produce an explicit permutation via sort_by_key, - // then apply it to each SoA member by gathering into a fresh - // full-capacity buffer and swapping the View handle in (no - // copy-back). The *_prev arrays are skipped — see + // Vendor path: produce an explicit permutation via sort_by_key, then + // apply it to each SoA member by gathering the alive prefix through a + // reusable scratch buffer (one per member type, sized to the alive + // count, copied back in place). The *_prev arrays are skipped — see // apply_permutation_to_soa. Peak transient = one - // `maxnpart × sizeof(member)` buffer at a time. + // `npart_partitioned × sizeof(member)` scratch at a time. prtl_perm_t perm { "tile_perm", npart_local }; #if defined(SYCL_ENABLED) && defined(ONEDPL_ENABLED) sort_helpers::sort_by_key_dispatch(tile_indices, @@ -349,7 +349,14 @@ namespace ntt { compute_tile_offsets(tile_indices, total_tiles, npart_local); tile_indices = array_t {}; Kokkos::fence("SortSpatially: pre-gather drain"); - apply_permutation_to_soa(perm); + // Gather only the alive particles. The sort binned dead particles to + // the sentinel tile, so they occupy [npart_partitioned, npart_local) + // and `perm[0, npart_partitioned)` lists the alive slots in tile + // order. Restricting the gather to the alive count drops the dead in + // the same pass (no separate compaction), skips work on dead slots, + // and sizes the gather scratch to the alive count. The dead tail is + // released below via `set_npart`. + apply_permutation_to_soa(perm, m_tile_layout.npart_partitioned); #else // BinSort path: same mechanism as legacy SortSpatially (BinSort // allocates one temp View per `sorter.sort(view)` call and frees @@ -411,6 +418,16 @@ namespace ntt { m_tile_layout.tile_perm = prtl_perm_t {}; m_is_sorted = true; + // Compact-on-sort: drop the dead tail now instead of waiting for the + // periodic RemoveDead. The sort parked dead particles in the sentinel + // bin at [npart_partitioned, npart()), and the alive set is exactly + // [0, npart_partitioned) — gathered into tile order above (vendor) or + // sorted in place (BinSort). Shrinking npart() here keeps it, and + // every subsequent sort's transient buffers, tracking the alive count + // instead of ratcheting up with dead slots between clearing intervals. + // (RemoveDead remains the compactor when spatial sorting is disabled.) + set_npart(m_tile_layout.npart_partitioned); + Kokkos::fence("SortSpatially: end of team_policy path"); #else // !TEAM_POLICY — legacy in-place BinSort by global cell index const auto nx2 = grid.n_active(in::x2); @@ -546,8 +563,8 @@ namespace ntt { } // namespace permute_helpers template - void Particles::apply_permutation_to_soa(const prtl_perm_t& perm) { - const auto n = npart(); + void Particles::apply_permutation_to_soa(const prtl_perm_t& perm, + npart_t n) { if (n == 0u) { return; } @@ -632,7 +649,7 @@ namespace ntt { #if defined(TEAM_POLICY_USE_VENDOR_SORT) #define APPLY_PERM_INSTANTIATE(D, C) \ template void Particles::apply_permutation_to_soa( \ - const prtl_perm_t&); + const prtl_perm_t&, npart_t); #else #define APPLY_PERM_INSTANTIATE(D, C) #endif From e1de8ae56656b286bcfcfce474d7a17af8555a6a Mon Sep 17 00:00:00 2001 From: Ludwig Boess Date: Wed, 24 Jun 2026 19:09:15 -0400 Subject: [PATCH 024/125] test update --- tests/framework/particles_sort.cpp | 44 ++++++++++++++++++++++++------ 1 file changed, 35 insertions(+), 9 deletions(-) diff --git a/tests/framework/particles_sort.cpp b/tests/framework/particles_sort.cpp index 4e4881d98..a2b82456e 100644 --- a/tests/framework/particles_sort.cpp +++ b/tests/framework/particles_sort.cpp @@ -112,16 +112,30 @@ auto main(int argc, char* argv[]) -> int { // SortSpatially is order-by-tile, not order-by-cell: assert the // invariants that hold for any tile size rather than a hardwired // permutation. (1) alive particles form a prefix sorted by - // non-decreasing tile index; (2) dead particles (weight == -1) form - // the suffix; (3) every SoA member is permuted by the *same* - // permutation, so each alive slot still satisfies pld == f(weight); - // (4) no alive particle is lost. Only [0, npart) is defined after a - // sort — the swap-gather zero-fills [npart, maxnpart). + // non-decreasing tile index; (2) every SoA member is permuted by the + // *same* permutation, so each alive slot still satisfies + // pld == f(weight); (3) no alive particle is lost. Only [0, npart()) + // is defined after a sort. The team_policy path compacts — it drops + // the dead, so npart() equals the alive count and [0, npart()) is + // entirely alive; the legacy (non-team) path keeps the dead as a + // weight == -1 suffix, leaving npart() unchanged. Iterating + // [0, npart()) exercises both: the prefix-sorted / no-alive-after-dead + // checks below hold either way. +#if defined(TEAM_POLICY) + raise::ErrorIf(prtls.npart() != 59u, + "team_policy sort must compact: npart() should equal " + "the alive count", + HERE); +#else + raise::ErrorIf(prtls.npart() != 66u, + "legacy sort should leave npart() unchanged", + HERE); +#endif bool seen_dead = false; bool have_prev = false; ncells_t prev_tile = 0u; npart_t n_alive_obs = 0u; - for (auto p { 0u }; p < 66u; ++p) { + for (auto p { 0u }; p < prtls.npart(); ++p) { if (tag_h(p) != ntt::ParticleTag::alive) { seen_dead = true; raise::ErrorIf(weight_h(p) != -1.0, @@ -244,8 +258,10 @@ auto main(int argc, char* argv[]) -> int { Kokkos::deep_copy(weight_h, prtls.weight); // Same invariants as the 2D block (no payloads here): alive prefix - // sorted by non-decreasing tile index, dead (weight == -1) suffix, - // alive count preserved. T = 1 reproduces the legacy per-cell order. + // sorted by non-decreasing tile index, alive count preserved. The + // team_policy path compacts the dead away (npart() == alive count); + // the legacy path keeps them as a weight == -1 suffix. T = 1 + // reproduces the legacy per-cell order. #if defined(TEAM_POLICY) const ncells_t T = static_cast(TEAM_POLICY_TILE_SIZE); #else @@ -261,11 +277,21 @@ auto main(int argc, char* argv[]) -> int { (static_cast(c) / T); }; +#if defined(TEAM_POLICY) + raise::ErrorIf(prtls.npart() != 59u, + "team_policy sort must compact: npart() should equal " + "the alive count", + HERE); +#else + raise::ErrorIf(prtls.npart() != 66u, + "legacy sort should leave npart() unchanged", + HERE); +#endif bool seen_dead = false; bool have_prev = false; ncells_t prev_tile = 0u; npart_t n_alive_obs = 0u; - for (auto p { 0u }; p < 66u; ++p) { + for (auto p { 0u }; p < prtls.npart(); ++p) { if (tag_h(p) != ntt::ParticleTag::alive) { seen_dead = true; raise::ErrorIf(weight_h(p) != -1.0, From 592ff89012eaf3276bc6dc491193df9c329765e8 Mon Sep 17 00:00:00 2001 From: Ludwig Boess Date: Thu, 25 Jun 2026 11:11:47 -0400 Subject: [PATCH 025/125] generalized reduced exchange for current filters on any coordinate system --- src/engines/srpic/currents.h | 167 ++++++++++++++++++----------------- 1 file changed, 88 insertions(+), 79 deletions(-) diff --git a/src/engines/srpic/currents.h b/src/engines/srpic/currents.h index 30eb9466d..dbb482e83 100644 --- a/src/engines/srpic/currents.h +++ b/src/engines/srpic/currents.h @@ -233,90 +233,99 @@ namespace ntt { const auto flds_bc = domain.mesh.flds_bc(); - if constexpr (M::CoordType == Coord::Cartesian) { - // Reduced-exchange ghost-margin scheme. One halo exchange refreshes - // N_GHOSTS ghost layers — enough for N_GHOSTS passes of the 3-point - // binomial if each pass also recomputes the inner ghost layers it - // will need next. We therefore extend the launch range by a shrinking - // margin `m` into the ghost zone, but only on comm-refreshed sides - // (PERIODIC self-wrap or SYNC inter-domain), where the ghost cell is - // interior physics. Physical-boundary ghosts are never written or - // refreshed, exactly as in the per-pass loop, so the result is - // identical for every BC — while doing one exchange per N_GHOSTS - // passes instead of one per pass. (Entering the loop the ghosts are - // valid to distance N_GHOSTS: srpic.hpp runs CommunicateFields(J) - // immediately before CurrentsFilter.) - const int G = static_cast(N_GHOSTS); - const auto comm_side = [](FldsBC b) { - return (b == FldsBC::PERIODIC) or (b == FldsBC::SYNC); + // Reduced-exchange ghost-margin scheme (all coordinate types). One halo + // exchange refreshes N_GHOSTS ghost layers — enough for N_GHOSTS passes + // of the 3-point binomial if each pass also recomputes the inner ghost + // layers it will need next. We therefore extend the launch range by a + // shrinking margin `m` into the ghost zone, but only on comm-refreshed + // sides (PERIODIC self-wrap or SYNC inter-domain), where the ghost cell + // is interior physics. Physical-boundary ghosts are never written or + // refreshed, exactly as in the per-pass loop, so the result is identical + // for every BC — while doing one exchange per N_GHOSTS passes instead of + // one per pass. (Entering the loop the ghosts are valid to distance + // N_GHOSTS: srpic.hpp runs CommunicateFields(J) immediately before + // CurrentsFilter.) + // + // Non-Cartesian axis: the theta (x2) direction is self-contained in the + // filter kernel — the axis branches only ever read/write within + // [i2_min, i2_max] and never cross the axis — so the axis needs no halo + // exchange at all (the axis current fold is done once by + // SynchronizeFields(J) before this function). The single coordinate + // dependency is that the axis cell sits at i_max(x2), one past the active + // range, and must be filtered on every pass. We therefore add a fixed +1 + // to the x2 upper bound when that side is AXIS — a physical boundary, so + // never the shrinking comm margin. This folds the old RangeWithAxisBCs + // fixup into make_range, letting the same loop serve every CoordType. + const int G = static_cast(N_GHOSTS); + const auto comm_side = [](FldsBC b) { + return (b == FldsBC::PERIODIC) or (b == FldsBC::SYNC); + }; + bool ext_lo[3] = { false, false, false }; + bool ext_hi[3] = { false, false, false }; + for (auto d { 0 }; d < static_cast(M::Dim); ++d) { + ext_lo[d] = comm_side(flds_bc[d].first); + ext_hi[d] = comm_side(flds_bc[d].second); + } + // AXIS at the upper x2 boundary needs the axis cell (i_max(x2)) included + // every pass; matches srpic::RangeWithAxisBCs. (The lower-x2 axis cell is + // already i_min(x2), so no fixup is needed there.) + bool axis_hi_x2 = false; + if constexpr (M::CoordType != Coord::Cartesian and + (M::Dim == Dim::_2D or M::Dim == Dim::_3D)) { + axis_hi_x2 = (flds_bc[1].second == FldsBC::AXIS); + } + const auto make_range = [&](int m) -> range_t { + const auto ml = [&](int d) -> ncells_t { + return ext_lo[d] ? static_cast(m) : 0u; }; - bool ext_lo[3] = { false, false, false }; - bool ext_hi[3] = { false, false, false }; - for (auto d { 0 }; d < static_cast(M::Dim); ++d) { - ext_lo[d] = comm_side(flds_bc[d].first); - ext_hi[d] = comm_side(flds_bc[d].second); - } - const auto make_range = [&](int m) -> range_t { - const auto ml = [&](int d) -> ncells_t { - return ext_lo[d] ? static_cast(m) : 0u; - }; - const auto mh = [&](int d) -> ncells_t { - return ext_hi[d] ? static_cast(m) : 0u; - }; - if constexpr (M::Dim == Dim::_1D) { - return CreateRangePolicy( - { domain.mesh.i_min(in::x1) - ml(0) }, - { domain.mesh.i_max(in::x1) + mh(0) }); - } else if constexpr (M::Dim == Dim::_2D) { - return CreateRangePolicy( - { domain.mesh.i_min(in::x1) - ml(0), - domain.mesh.i_min(in::x2) - ml(1) }, - { domain.mesh.i_max(in::x1) + mh(0), - domain.mesh.i_max(in::x2) + mh(1) }); - } else { - return CreateRangePolicy( - { domain.mesh.i_min(in::x1) - ml(0), - domain.mesh.i_min(in::x2) - ml(1), - domain.mesh.i_min(in::x3) - ml(2) }, - { domain.mesh.i_max(in::x1) + mh(0), - domain.mesh.i_max(in::x2) + mh(1), - domain.mesh.i_max(in::x3) + mh(2) }); + const auto mh = [&](int d) -> ncells_t { + if (ext_hi[d]) { + return static_cast(m); } - }; - int m = G - 1; - for (auto i { 0u }; i < nfilter; ++i) { - Kokkos::parallel_for( - "CurrentsFilter", - make_range(m), - kernel::DigitalFilter_kernel( - domain.fields.buff, - domain.fields.cur, - size, - flds_bc)); - std::swap(domain.fields.cur, domain.fields.buff); - --m; - if (m < 0 or i == nfilter - 1u) { - // refresh ghosts to distance G (and leave them valid for the - // downstream field solver after the final pass) - metadomain.CommunicateFields(domain, Comm::J); - m = G - 1; + // axis cell fixup (x2 == dimension index 1); mutually exclusive with + // the comm margin since AXIS is not a comm side + if (d == 1 and axis_hi_x2) { + return 1u; } + return 0u; + }; + if constexpr (M::Dim == Dim::_1D) { + return CreateRangePolicy( + { domain.mesh.i_min(in::x1) - ml(0) }, + { domain.mesh.i_max(in::x1) + mh(0) }); + } else if constexpr (M::Dim == Dim::_2D) { + return CreateRangePolicy( + { domain.mesh.i_min(in::x1) - ml(0), + domain.mesh.i_min(in::x2) - ml(1) }, + { domain.mesh.i_max(in::x1) + mh(0), + domain.mesh.i_max(in::x2) + mh(1) }); + } else { + return CreateRangePolicy( + { domain.mesh.i_min(in::x1) - ml(0), + domain.mesh.i_min(in::x2) - ml(1), + domain.mesh.i_min(in::x3) - ml(2) }, + { domain.mesh.i_max(in::x1) + mh(0), + domain.mesh.i_max(in::x2) + mh(1), + domain.mesh.i_max(in::x3) + mh(2) }); } - } else { - // Non-Cartesian (axis BCs need the +1 range fixup): keep the - // per-pass exchange cadence, still ping-ponging the buffers. - const auto range = srpic::RangeWithAxisBCs(domain); - for (auto i { 0u }; i < nfilter; ++i) { - Kokkos::parallel_for( - "CurrentsFilter", - range, - kernel::DigitalFilter_kernel( - domain.fields.buff, - domain.fields.cur, - size, - flds_bc)); - std::swap(domain.fields.cur, domain.fields.buff); + }; + int m = G - 1; + for (auto i { 0u }; i < nfilter; ++i) { + Kokkos::parallel_for( + "CurrentsFilter", + make_range(m), + kernel::DigitalFilter_kernel( + domain.fields.buff, + domain.fields.cur, + size, + flds_bc)); + std::swap(domain.fields.cur, domain.fields.buff); + --m; + if (m < 0 or i == nfilter - 1u) { + // refresh ghosts to distance G (and leave them valid for the + // downstream field solver after the final pass) metadomain.CommunicateFields(domain, Comm::J); + m = G - 1; } } } From 6e1e23e29b82e05a3c2593657925af239af36e30 Mon Sep 17 00:00:00 2001 From: LudwigBoess Date: Thu, 25 Jun 2026 18:37:44 +0000 Subject: [PATCH 026/125] bugfix --- src/engines/srpic/currents.h | 2 ++ 1 file changed, 2 insertions(+) diff --git a/src/engines/srpic/currents.h b/src/engines/srpic/currents.h index dbb482e83..23c57a903 100644 --- a/src/engines/srpic/currents.h +++ b/src/engines/srpic/currents.h @@ -47,6 +47,7 @@ namespace ntt { dt)); } +#if defined(TEAM_POLICY) /** * @brief Tiled deposit launcher (TeamPolicy + per-team scratch). * @@ -116,6 +117,7 @@ namespace ntt { Kokkos::Experimental::contribute(cur_nc, scatter_cur); } } +#endif // TEAM_POLICY template void CurrentsDeposit(Domain& domain, From ff3ec19bcd463024fd5f78979d5135c120792228 Mon Sep 17 00:00:00 2001 From: LudwigBoess Date: Thu, 25 Jun 2026 23:08:19 +0000 Subject: [PATCH 027/125] move definittion of `compute_tile_offsets` to `public` so it compiles under nvcc --- src/framework/containers/particles.h | 26 ++++++++++++++------------ 1 file changed, 14 insertions(+), 12 deletions(-) diff --git a/src/framework/containers/particles.h b/src/framework/containers/particles.h index fe4a0b17f..396b5aec8 100644 --- a/src/framework/containers/particles.h +++ b/src/framework/containers/particles.h @@ -99,18 +99,6 @@ namespace ntt { // vendor libraries detected by CMake. TileLayout m_tile_layout {}; -#if defined(TEAM_POLICY) - // Build m_tile_layout.tile_offsets / npart_partitioned from the - // already-sorted tile-index keys. A separate member function (not a - // lambda local to SortSpatially) so the inner device kernel is not an - // extended __device__ lambda nested inside another lambda — which - // nvcc forbids. Lets the vendor path run the offsets pass and then - // release the keys before the SoA gather allocates its buffers. - void compute_tile_offsets(const array_t& tile_indices, - ncells_t total_tiles, - npart_t npart_local); -#endif - public: // for empty allocation Particles() {} @@ -219,6 +207,20 @@ namespace ntt { return m_ntags; } +#if defined(TEAM_POLICY) + // Build m_tile_layout.tile_offsets / npart_partitioned from the + // already-sorted tile-index keys. A separate member function (not a + // lambda local to SortSpatially) so the inner device kernel is not an + // extended __device__ lambda nested inside another lambda — which + // nvcc forbids. Lets the vendor path run the offsets pass and then + // release the keys before the SoA gather allocates its buffers. + // NOTE: must be public — nvcc forbids an extended __host__ __device__ + // lambda inside a member function with private/protected access. + void compute_tile_offsets(const array_t& tile_indices, + ncells_t total_tiles, + npart_t npart_local); +#endif + [[nodiscard]] auto memory_footprint() const -> std::size_t { std::size_t footprint = 0; From 03ccd200e25baae4064e1a3e8e29ad5b2b0502f6 Mon Sep 17 00:00:00 2001 From: LudwigBoess Date: Fri, 26 Jun 2026 01:02:05 +0000 Subject: [PATCH 028/125] added `team_policy_team_size` as a tunable runtime parameter --- input.example.toml | 8 +++++ src/engines/engine.hpp | 13 ++++--- src/engines/reporter.cpp | 15 ++++++++ src/engines/srpic/currents.h | 48 ++++++++++++++++++++++--- src/framework/parameters/algorithms.cpp | 7 ++++ src/framework/parameters/algorithms.h | 1 + src/global/defaults.h | 2 ++ 7 files changed, 85 insertions(+), 9 deletions(-) diff --git a/input.example.toml b/input.example.toml index 741d5c2d2..87ca48f3c 100644 --- a/input.example.toml +++ b/input.example.toml @@ -301,6 +301,14 @@ # @type: bool # @default: true enable = "" + # team_policy tiled-deposit work-group (team) size + # @type: uint [>= 0] + # @default: 0 + # @note: 0 keeps Kokkos::AUTO (backend occupancy heuristic); a positive + # value overrides it, clamped to the backend/scratch maximum at + # launch. Only used in `team_policy=ON` builds. Pick a multiple of + # the device subgroup width for best occupancy (see ideal_tile_size.py) + team_policy_team_size = "" # @inferred: # - order diff --git a/src/engines/engine.hpp b/src/engines/engine.hpp index b20e163ca..057bfcb5a 100644 --- a/src/engines/engine.hpp +++ b/src/engines/engine.hpp @@ -78,10 +78,11 @@ namespace ntt { Metadomain m_metadomain; user::PGen m_pgen; - const bool is_resuming; - const simtime_t runtime; - const real_t dt; - const timestep_t max_steps; + const bool is_resuming; + const simtime_t runtime; + const real_t dt; + const std::size_t team_policy_team_size; + const timestep_t max_steps; const timestep_t start_step; const simtime_t start_time; simtime_t time; @@ -109,6 +110,8 @@ namespace ntt { , is_resuming { m_params.get("checkpoint.is_resuming") } , runtime { m_params.get("simulation.runtime") } , dt { m_params.get("algorithms.timestep.dt") } + , team_policy_team_size { m_params.get( + "algorithms.deposit.team_policy_team_size") } , max_steps { static_cast(runtime / dt) } , start_step { m_params.get("checkpoint.start_step") } , start_time { m_params.get("checkpoint.start_time") } @@ -127,6 +130,8 @@ namespace ntt { auto parameters = prm::Parameters {}; parameters.set("dt", static_cast(dt)); parameters.set("time", static_cast(time)); + parameters.set("team_policy_team_size", + static_cast(team_policy_team_size)); return parameters; } }; diff --git a/src/engines/reporter.cpp b/src/engines/reporter.cpp index 3caeda5de..fcd46fb81 100644 --- a/src/engines/reporter.cpp +++ b/src/engines/reporter.cpp @@ -34,6 +34,21 @@ namespace ntt { reporter::AddParam(report, 4, "Engine", "%s", SimEngine(S).to_string()); #if defined(TEAM_POLICY) reporter::AddParam(report, 4, "Tile size", "%d", TEAM_POLICY_TILE_SIZE); + #if defined(TEAM_POLICY_DRIFT) + reporter::AddParam(report, 4, "Halo drift", "%d", TEAM_POLICY_DRIFT); + #endif + if (params.template get( + "algorithms.deposit.team_policy_team_size") == 0u) { + reporter::AddParam(report, 4, "Team size", "%s", "AUTO (Kokkos)"); + } else { + reporter::AddParam( + report, + 4, + "Team size", + "%d (requested; clamped to backend max at launch)", + static_cast(params.template get( + "algorithms.deposit.team_policy_team_size"))); + } #endif reporter::AddParam(report, 4, "Metric", "%s", M.to_string()); #if SHAPE_ORDER == 0 diff --git a/src/engines/srpic/currents.h b/src/engines/srpic/currents.h index 23c57a903..74946ed68 100644 --- a/src/engines/srpic/currents.h +++ b/src/engines/srpic/currents.h @@ -66,7 +66,8 @@ namespace ntt { void CallDepositKernelTiled(const Particles& species, const M& local_metric, const ndfield_t& cur, - real_t dt) { + real_t dt, + int team_size_req) { static_assert(O <= 11u, "Shape order must be <= 11"); constexpr unsigned short T = static_cast( TEAM_POLICY_TILE_SIZE); @@ -86,11 +87,41 @@ namespace ntt { dt, layout, species.npart() }; + const auto scratch = Kokkos::PerTeam( + decltype(deposit_kernel)::scratch_bytes()); + + // Team (work-group) size. The default (team_size_req == 0) leaves + // Kokkos::AUTO, which sizes the team from the backend occupancy + // heuristic. A positive `algorithms.deposit.team_policy_team_size` + // overrides it, clamped to the scratch/backend-feasible maximum so an + // over-large request cannot abort the launch (Kokkos errors when + // team_size > team_size_max). No portable subgroup rounding is applied; + // pick a multiple of the device subgroup width (printed per arch by + // ideal_tile_size.py) for the best occupancy. Kokkos::TeamPolicy<> policy(static_cast(layout.ntiles_total), Kokkos::AUTO); - policy.set_scratch_size( - 0, - Kokkos::PerTeam(decltype(deposit_kernel)::scratch_bytes())); + policy.set_scratch_size(0, scratch); + if (team_size_req > 0) { + const int ts_max = policy.team_size_max(deposit_kernel, + Kokkos::ParallelForTag {}); + int ts = team_size_req; + if (ts > ts_max) { + raise::Warning( + fmt::format("algorithms.deposit.team_policy_team_size = %d exceeds " + "the tiled-deposit maximum %d on this backend; clamping " + "to %d", + team_size_req, + ts_max, + ts_max), + HERE); + ts = ts_max; + } + policy = Kokkos::TeamPolicy<>(static_cast(layout.ntiles_total), ts); + policy.set_scratch_size(0, scratch); + logger::Checkpoint( + fmt::format("Tiled deposit: explicit team size %d", ts), + HERE); + } Kokkos::parallel_for("CurrentsDepositTiled", policy, deposit_kernel); // Particles appended since the last sort (injection / MPI receive on a @@ -126,6 +157,12 @@ namespace ntt { Kokkos::deep_copy(domain.fields.cur, ZERO); #if defined(TEAM_POLICY) + // Optional runtime override for the tiled-deposit team (work-group) size; + // 0 (default) keeps Kokkos::AUTO. Clamped to the backend max in the + // launcher (see CallDepositKernelTiled). + const auto team_size_req = static_cast( + engine_params.get("team_policy_team_size", + std::optional { 0u })); // Tiled deposit. Correctness no longer depends on the SoA being in a // "sorted" state at deposit time — the tiled kernel handles a stale @@ -176,7 +213,8 @@ namespace ntt { CallDepositKernelTiled(species, domain.mesh.metric, domain.fields.cur, - dt); + dt, + team_size_req); } } #else diff --git a/src/framework/parameters/algorithms.cpp b/src/framework/parameters/algorithms.cpp index 4766db965..87c77fbd9 100644 --- a/src/framework/parameters/algorithms.cpp +++ b/src/framework/parameters/algorithms.cpp @@ -33,6 +33,11 @@ namespace ntt { deposit_enable = toml::find_or(toml_data, "algorithms", "deposit", "enable", true); deposit_order = static_cast(SHAPE_ORDER); + deposit_team_policy_team_size = toml::find_or(toml_data, + "algorithms", + "deposit", + "team_policy_team_size", + defaults::team_policy_team_size); fieldsolver_enable = toml::find_or(toml_data, "algorithms", @@ -140,6 +145,8 @@ namespace ntt { params->set("algorithms.deposit.enable", deposit_enable.value()); params->set("algorithms.deposit.order", deposit_order.value()); + params->set("algorithms.deposit.team_policy_team_size", + deposit_team_policy_team_size.value()); params->set("algorithms.fieldsolver.enable", fieldsolver_enable.value()); for (const auto& [key, value] : fieldsolver_stencil_coeffs.value()) { diff --git a/src/framework/parameters/algorithms.h b/src/framework/parameters/algorithms.h index 97edb244a..c46f480fe 100644 --- a/src/framework/parameters/algorithms.h +++ b/src/framework/parameters/algorithms.h @@ -34,6 +34,7 @@ namespace ntt { std::optional deposit_enable; std::optional deposit_order; + std::optional deposit_team_policy_team_size; std::optional fieldsolver_enable; std::optional> fieldsolver_stencil_coeffs; diff --git a/src/global/defaults.h b/src/global/defaults.h index e1387677e..c4f9b67f4 100644 --- a/src/global/defaults.h +++ b/src/global/defaults.h @@ -22,6 +22,8 @@ namespace ntt::defaults { const unsigned short current_filters = 0; + const std::size_t team_policy_team_size = 0; + const std::string em_pusher = "Boris"; const std::string ph_pusher = "Photon"; const timestep_t clear_interval = 100; From ca7307dd7ee3b19e0997dedbc6787131ed9556ed Mon Sep 17 00:00:00 2001 From: LudwigBoess Date: Fri, 26 Jun 2026 02:38:14 +0000 Subject: [PATCH 029/125] add script to compute ideal tile size --- ideal_tile_size.py | 955 +++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 955 insertions(+) create mode 100644 ideal_tile_size.py diff --git a/ideal_tile_size.py b/ideal_tile_size.py new file mode 100644 index 000000000..e216f5981 --- /dev/null +++ b/ideal_tile_size.py @@ -0,0 +1,955 @@ +#!/usr/bin/env python3 +"""Recommend the team tile size (T_TILE) for entity's tiled current-deposit kernel. + +Each GPU work-group (team) owns a TE^dim scratch tile in shared memory (SLM on Intel, +LDS on AMD, shared mem on NVIDIA), accumulates its particles' currents into it, then +flushes once to global memory. With + + TE = T_TILE + 2*HALO + HALO = stencil_reach + drift (stencil_reach = shape_order for Esirkepov, 2 for the + O==0 zigzag deposit; drift = the compile-time + `team_policy_drift` CMake knob, NOT the runtime + spatial_sorting_interval -- see currents_deposit.hpp) + +the tile size is squeezed by three competing pressures: + + * Shared-memory capacity (HARD): TE^dim * ncomp * sizeof(real) must fit, ideally with + several work-groups resident per compute unit so latency is hidden. Binds in 3D / + double precision / on AMD's 64 KiB LDS. + * Halo overhead (push LARGER): zero-fill + flush sweep the whole TE^dim tile, so a tiny + tile is almost all halo (1-(T/TE)^dim wasted). Big HALO (infrequent sorts) makes this + worse and forces larger tiles. + * Particles per tile (push SMALLER): ppc*T^dim particles all atomic-add into one fixed + scratch tile -> SLM-atomic contention and load imbalance grow with tile size. + +Recommendation = the largest tile that respects the particle budget and shared-memory +residency; if that tile would be mostly halo, it is grown (toward lower halo) up to the +shared-memory limit. This is a first-order model -- confirm by sweeping the entity knobs + -D team_policy_tile_size= -D team_policy_drift= and re-profiling (see roofline/). +The team (work-group) size defaults to Kokkos::AUTO; override it at runtime with the + [algorithms.deposit] team_policy_team_size = (0 = AUTO) +toml knob -- clamped to the backend maximum at launch (engines/srpic/currents.h). + +Two ways to drive it: + + * Interactive TUI (no arguments): + python ideal_tile_size.py + + * Scriptable CLI (any argument): + python ideal_tile_size.py pvc --dim 2 --ppc 16 + python ideal_tile_size.py amd --dim 3 --ppc 64 + python ideal_tile_size.py all --dim 2 --ppc 16 --drift 4 # infrequent sorting +""" +import argparse +import curses +import os +import sys +from typing import Callable, List, Optional, Tuple + +# Shared-memory budget is the load-bearing number. Per compute unit (Xe-core / SM / CU); +# smem_wg_max is the largest single-work-group allocation. +ARCH = { + "pvc": dict(label="Intel Data Center GPU Max 1550 (PVC, Xe-HPC), per tile", + smem_cu=128 * 1024, smem_wg_max=128 * 1024, + subgroup=32, max_wg=1024, n_cu=64, cu="Xe-core"), + "a100": dict(label="NVIDIA A100 (Ampere)", + smem_cu=164 * 1024, smem_wg_max=163 * 1024, + subgroup=32, max_wg=1024, n_cu=108, cu="SM", + note="shared mem >48 KiB/block needs opt-in (cudaFuncAttributeMaxDynamicSharedMemorySize)"), + "h100": dict(label="NVIDIA H100 (Hopper)", + smem_cu=228 * 1024, smem_wg_max=227 * 1024, + subgroup=32, max_wg=1024, n_cu=132, cu="SM", + note="shared mem >48 KiB/block needs opt-in (cudaFuncAttributeMaxDynamicSharedMemorySize)"), + "gh200": dict(label="NVIDIA GH200 Grace Hopper (Hopper H100/H200 GPU)", + smem_cu=228 * 1024, smem_wg_max=227 * 1024, + subgroup=32, max_wg=1024, n_cu=132, cu="SM", + note="shared mem >48 KiB/block needs opt-in (cudaFuncAttributeMaxDynamicSharedMemorySize)"), + "mi250x": dict(label="AMD Instinct MI250X (CDNA2), per GCD", + smem_cu=64 * 1024, smem_wg_max=64 * 1024, + subgroup=64, max_wg=1024, n_cu=110, cu="CU"), + "mi300x": dict(label="AMD Instinct MI300X (CDNA3)", + smem_cu=64 * 1024, smem_wg_max=64 * 1024, + subgroup=64, max_wg=1024, n_cu=304, cu="CU"), +} +ALIAS = {"nvidia": "h100", "amd": "mi300x", "intel": "pvc", "mi250": "mi250x", "mi300": "mi300x", + "gracehopper": "gh200", "grace-hopper": "gh200", "gh200x": "gh200"} +PRECISION = {"single": 4, "double": 8} + +# arch choices offered in the TUI (canonical keys plus the meta-target "all") +ARCH_CHOICES = list(ARCH.keys()) + ["all"] +# "all" expands to one representative of each vendor (matches the CLI behaviour) +ALL_ARCHS = ["pvc", "nvidia", "amd"] + + +# ============================ +# core model (shared by TUI + CLI) +# ============================ + +class Settings: + """Tunable inputs for the tile-size model. + + Attribute names match the argparse dest names so `recommend`/`report_lines` accept + either a Settings instance (TUI) or an argparse namespace (CLI) interchangeably. + """ + + def __init__(self): + self.arch = "pvc" # an ARCH key or "all" + self.dim = 2 # 1 / 2 / 3 + self.ppc = 16.0 # particles per cell (per species) + self.shape_order = 2 # entity shape_order + self.precision = "single" # single / double + self.components = 3 # current-field components (J has 3) + self.drift = 1 # team_policy_drift: cells of drift the scratch halo absorbs + # (compile-time CMake knob, independent of spatial_sorting_interval) + self.target_resident = 2 # work-groups resident per compute unit + self.npart_cap = 1600.0 # particle-per-tile budget (contention / load-balance proxy) + self.halo_max = 0.70 # halo fraction above which the tile is grown + self.grid = 0 # cells per dim (0 disables the GPU-fill check) + self.balance_factor = 4 # min tiles per compute unit + self.min_tile = 4 # entity's team_policy_tile_sizes list starts at 4 + self.max_tile = 64 + + +def resolve_arch(name): + key = ALIAS.get(name.lower(), name.lower()) + if key not in ARCH: + raise SystemExit("unknown arch '%s'; choose from %s (or aliases %s)" + % (name, ", ".join(ARCH), ", ".join(ALIAS))) + return key, ARCH[key] + + +def recommend(hw, p): + """p: Settings or argparse namespace. Returns dict with rows, chosen row, binding.""" + # Matches DepositCurrentsTiled_kernel: STENCIL_REACH = O for Esirkepov (O>=1), + # 2 for the O==0 zigzag deposit; HALO = STENCIL_REACH + TEAM_POLICY_DRIFT. + stencil_reach = 2 if p.shape_order == 0 else p.shape_order + halo = stencil_reach + p.drift + real = PRECISION[p.precision] + atoms_pp = p.components * (p.shape_order + 1) ** p.dim # ~ useful atomics / particle + + rows = [] + for T in range(p.min_tile, p.max_tile + 1, 2): # entity uses even tile sizes + TE = T + 2 * halo + scratch = TE ** p.dim * p.components * real + npart = p.ppc * T ** p.dim + halo_frac = 1.0 - (T / TE) ** p.dim + ovhd = (2.0 * p.components / p.ppc) * (TE / T) ** p.dim / atoms_pp # zero+flush vs deposit + resident = int(hw["smem_cu"] // scratch) if scratch else 0 + ntiles = (p.grid / T) ** p.dim if p.grid else None + rows.append(dict(T=T, TE=TE, scratch=scratch, npart=npart, halo_frac=halo_frac, + ovhd=ovhd, resident=resident, ntiles=ntiles, + ok_cap=scratch <= hw["smem_wg_max"])) + + capfeas = [r for r in rows if r["ok_cap"]] + if not capfeas: + return dict(halo=halo, rows=rows, chosen=None, binding=None, grown=False) + + def largest(pred, default): + ts = [r["T"] for r in capfeas if pred(r)] + return max(ts) if ts else default + + T_cap = max(r["T"] for r in capfeas) + T_res = largest(lambda r: r["resident"] >= p.target_resident, p.min_tile) + T_np = largest(lambda r: r["npart"] <= p.npart_cap, p.min_tile) + T_bal = largest(lambda r: r["ntiles"] is None or r["ntiles"] >= p.balance_factor * hw["n_cu"], p.min_tile) + bounds = {"shared-memory capacity": T_cap, "shared-memory residency": T_res, + "GPU fill (too few tiles)": T_bal, "particle budget per tile": T_np} + + chosen_T = max(min(bounds.values()), p.min_tile) + binding = min(bounds, key=lambda k: bounds[k]) + + # If the particle-budget pick is mostly halo, grow the tile to cut halo, but never + # past what shared memory / GPU-fill allow (that just trades halo for contention). + grown = False + cur = next(r for r in capfeas if r["T"] == chosen_T) + if cur["halo_frac"] > p.halo_max: + halo_ok = [r["T"] for r in capfeas if r["halo_frac"] <= p.halo_max] + ceil_T = min(T_res, T_bal, T_cap) + if halo_ok: + target = max(min(halo_ok), chosen_T) # smallest tile that clears halo_max + new_T = min(max(target, chosen_T), ceil_T) + if new_T > chosen_T: + chosen_T, grown = new_T, True + binding = ("halo overhead" if min(halo_ok) <= ceil_T + else min({k: v for k, v in bounds.items() + if k != "particle budget per tile"}, + key=lambda k: bounds[k])) + else: + new_T = min(ceil_T, T_cap) # can't clear halo_max at all -> go as big as SLM allows + if new_T > chosen_T: + chosen_T, grown = new_T, True + binding = "shared-memory residency" + + chosen = next(r for r in capfeas if r["T"] == chosen_T) + return dict(halo=halo, rows=rows, chosen=chosen, binding=binding, grown=grown, atoms_pp=atoms_pp) + + +def kib(b): + return "%.1f" % (b / 1024.0) + + +def report_lines(name, key, hw, p, res): + """Build the recommendation report as a list of text lines (no printing).""" + L = [] + L.append("=" * 80) + L.append("%s [preset: %s]" % (hw["label"], key)) + L.append(" dim=%d ppc=%g shape_order=%d precision=%s(%dB) J-components=%d" + % (p.dim, p.ppc, p.shape_order, p.precision, PRECISION[p.precision], p.components)) + reach = 2 if p.shape_order == 0 else p.shape_order + reach_kind = "zigzag" if p.shape_order == 0 else "Esirkepov O" + L.append(" HALO = stencil_reach + drift = %d + %d = %d -> TE = T_TILE + %d" + " (reach %d = %s; drift = team_policy_drift)" + % (reach, p.drift, res["halo"], 2 * res["halo"], reach, reach_kind)) + L.append(" shared mem %s KiB/%s (budget %s KiB for %d resident WGs); subgroup=%d, n_cu=%d" + % (kib(hw["smem_cu"]), hw["cu"], kib(hw["smem_cu"] / p.target_resident), + p.target_resident, hw["subgroup"], hw["n_cu"])) + if hw.get("note"): + L.append(" note: %s" % hw["note"]) + L.append("-" * 80) + L.append(" T_TILE TE scratch resWG part/tile halo% zero+flush%") + for r in res["rows"]: + if not r["ok_cap"]: + continue + mark = " <== recommended" if r is res["chosen"] else "" + L.append(" %3d %4d %6s K %4d %9.0f %4.0f %7.1f%s" + % (r["T"], r["TE"], kib(r["scratch"]), r["resident"], r["npart"], + 100 * r["halo_frac"], 100 * r["ovhd"], mark)) + L.append("-" * 80) + + c = res["chosen"] + if c is None: + L.append(" INFEASIBLE: even T_TILE=%d does not fit %s KiB shared memory." + % (p.min_tile, kib(hw["smem_wg_max"]))) + L.append(" -> sort more often (smaller drift), use precision single, lower shape_order,") + L.append(" or use a non-tiled (global-atomic / ScatterView) deposit on this arch.") + return L + L.append(" RECOMMENDED T_TILE = %d (limited by: %s%s)" + % (c["T"], res["binding"], "; tile grown to reduce halo" if res["grown"] else "")) + L.append(" %.1f KiB scratch/team, %d work-groups resident/%s, %.0f particles/team, %.0f%% halo" + % (c["scratch"] / 1024.0, c["resident"], hw["cu"], c["npart"], 100 * c["halo_frac"])) + team = min(hw["max_wg"], 256 - 256 % hw["subgroup"]) + extra = "" if c["T"] <= 16 else " (entity's team_policy_tile_sizes list stops at 16; extend it)" + L.append(" entity build: -D team_policy=ON -D team_policy_tile_size=%d -D team_policy_drift=%d%s" + % (min(c["T"], 16), p.drift, extra)) + L.append(" team (work-group) size: Kokkos::AUTO by default; to override, set in the toml") + L.append(" [algorithms.deposit] team_policy_team_size = %d (0 = AUTO; keep a multiple of" + % team) + L.append(" subgroup=%d), then sweep around it and re-profile" % hw["subgroup"]) + # contextual guidance + if c["halo_frac"] > p.halo_max: + if p.drift > 1: + L.append(" !! %.0f%% of the tile is halo, inflated by team_policy_drift=%d; lower it " + "(and sort at least that often via spatial_sorting_interval)" + % (100 * c["halo_frac"], p.drift)) + else: + L.append(" !! %.0f%% halo is intrinsic at this size (shared memory caps the tile here)" + % (100 * c["halo_frac"])) + if c["npart"] > p.npart_cap: + L.append(" !! %.0f particles/team exceeds the %.0f budget -> watch SLM-atomic contention" + % (c["npart"], p.npart_cap)) + if c["resident"] < p.target_resident: + L.append(" !! only %d work-group(s) resident/%s -> limited latency hiding" + % (c["resident"], hw["cu"])) + return L + + +def archs_for(name): + """Expand a setting/arg value into the list of arch names to report on.""" + return ALL_ARCHS if name.lower() == "all" else [name] + + +def build_report(p): + """Run the model for the selected arch(es) and return the full report as lines.""" + lines = [] + for a in archs_for(p.arch): + try: + key, hw = resolve_arch(a) + except SystemExit as e: + lines.append(str(e)) + continue + lines.extend(report_lines(a, key, hw, p, recommend(hw, p))) + lines.append("=" * 80) + return lines + + +# ============================ +# colors: edit these +# ============================ + +COLOR_TITLE_FG = curses.COLOR_BLUE +COLOR_TEXT_FG = curses.COLOR_WHITE +COLOR_SELECTED_FG = curses.COLOR_WHITE +COLOR_SELECTED_BG = curses.COLOR_BLACK +COLOR_HINT_FG = curses.COLOR_YELLOW +COLOR_OK_FG = curses.COLOR_GREEN +COLOR_ERR_FG = curses.COLOR_RED +COLOR_KEY_FG = curses.COLOR_MAGENTA +COLOR_DIM_FG = curses.COLOR_CYAN + +PAIR_TITLE = 1 +PAIR_TEXT = 2 +PAIR_SELECTED = 3 +PAIR_HINT = 4 +PAIR_OK = 5 +PAIR_ERR = 6 +PAIR_KEY = 7 +PAIR_DIM = 8 + + +class MenuItem: + def __init__(self, label, hint="", right=None, on_enter=None, + on_space=None, disabled=None): + self.label = label + self.hint = hint + self.right = right + self.on_enter = on_enter + self.on_space = on_space + self.disabled = disabled + + +# ============================ +# TUI +# ============================ + +class App: + def __init__(self, stdscr): + self.stdscr = stdscr + self.s = Settings() + + self.state = "mainmenu" + self.stack: List[Tuple[str, int]] = [] + self.selected = 0 + self.scroll = 0 + self.message = "use arrows or j/k" + + self._init_curses() + + def _init_curses(self) -> None: + curses.curs_set(0) + self.stdscr.keypad(True) + curses.noecho() + curses.cbreak() + + if curses.has_colors(): + curses.start_color() + curses.use_default_colors() + curses.init_pair(PAIR_TITLE, COLOR_TITLE_FG, -1) + curses.init_pair(PAIR_TEXT, COLOR_TEXT_FG, -1) + curses.init_pair(PAIR_SELECTED, COLOR_SELECTED_FG, COLOR_SELECTED_BG) + curses.init_pair(PAIR_HINT, COLOR_HINT_FG, -1) + curses.init_pair(PAIR_OK, COLOR_OK_FG, -1) + curses.init_pair(PAIR_ERR, COLOR_ERR_FG, -1) + curses.init_pair(PAIR_KEY, COLOR_KEY_FG, -1) + curses.init_pair(PAIR_DIM, COLOR_DIM_FG, -1) + + def cp(self, pair_id: int) -> int: + return curses.color_pair(pair_id) if curses.has_colors() else 0 + + # ----- formatting helpers ----- + + def arch_label(self) -> str: + if self.s.arch == "all": + return "all (%s)" % " + ".join(ALL_ARCHS) + try: + return ARCH[resolve_arch(self.s.arch)[0]]["label"] + except SystemExit: + return self.s.arch + + # ----- nav stack ----- + + def push(self, st: str) -> None: + self.stack.append((self.state, self.selected)) + self.state = st + self.selected = 0 + self.scroll = 0 + self.message = "" + + def pop(self) -> None: + if self.stack: + self.state, self.selected = self.stack.pop() + else: + self.state, self.selected = "mainmenu", 0 + self.scroll = 0 + self.message = "" + + # ----- drawing ----- + + def add(self, y: int, x: int, s: str, attr: int = 0) -> None: + try: + self.stdscr.addstr(y, x, s, attr) + except curses.error: + pass + + def hline(self, y: int) -> None: + _, w = self.stdscr.getmaxyx() + try: + self.stdscr.hline(y, 0, curses.ACS_HLINE, max(0, w - 1)) + except curses.error: + pass + + def draw_keybar(self, y: int, x: int, pairs: List[Tuple[str, str]]) -> None: + cur_x = x + for key, action in pairs: + self.add(y, cur_x, key, self.cp(PAIR_KEY) | curses.A_BOLD) + cur_x += len(key) + self.add(y, cur_x, " ", self.cp(PAIR_DIM)) + cur_x += 1 + self.add(y, cur_x, action, self.cp(PAIR_HINT)) + cur_x += len(action) + self.add(y, cur_x, " ", self.cp(PAIR_DIM)) + cur_x += 3 + + def breadcrumb(self) -> str: + return { + "mainmenu": "mainmenu", + "arch": "mainmenu > architecture", + "physics": "mainmenu > physics & particles", + "tuning": "mainmenu > tuning knobs", + }.get(self.state, "mainmenu") + + def draw_menu(self, title: str, prompt: str, items: List[MenuItem]) -> None: + self.stdscr.erase() + h, w = self.stdscr.getmaxyx() + + self.add(0, 2, title, self.cp(PAIR_TITLE) | curses.A_BOLD) + bc = self.breadcrumb() + self.add(0, max(2, w - 2 - len(bc)), bc, self.cp(PAIR_DIM)) + + self.draw_keybar( + 1, + 2, + [ + ("up/dn/j/k", "move"), + ("enter", "select"), + ("space", "toggle/cycle"), + ("b", "back"), + ("q", "quit"), + ], + ) + self.hline(2) + + status1 = ("arch: %s dim: %d precision: %s ppc: %g" + % (self.s.arch, self.s.dim, self.s.precision, self.s.ppc)) + status2 = ("shape_order: %d drift: %d components: %d tile range: %d-%d" + % (self.s.shape_order, self.s.drift, self.s.components, + self.s.min_tile, self.s.max_tile)) + self.add(3, 2, status1[: w - 4], self.cp(PAIR_TEXT)) + self.add(4, 2, status2[: w - 4], self.cp(PAIR_TEXT)) + self.hline(5) + + self.add(6, 2, prompt[: w - 4], self.cp(PAIR_TEXT) | curses.A_BOLD) + + list_y = 8 + footer_h = 3 + view_h = max(1, h - list_y - footer_h) + n = len(items) + + if n == 0: + self.add(list_y, 2, "(empty)", self.cp(PAIR_HINT)) + else: + self.selected = max(0, min(self.selected, n - 1)) + + if self.selected < self.scroll: + self.scroll = self.selected + if self.selected >= self.scroll + view_h: + self.scroll = self.selected - view_h + 1 + self.scroll = max(0, min(self.scroll, max(0, n - view_h))) + + shown = items[self.scroll : self.scroll + view_h] + + for i, it in enumerate(shown): + idx = self.scroll + i + sel = idx == self.selected + dis = bool(it.disabled and it.disabled()) + + row_attr = ( + self.cp(PAIR_SELECTED) | curses.A_BOLD + if sel + else (self.cp(PAIR_DIM) if dis else self.cp(PAIR_TEXT)) + ) + self.add(list_y + i, 2, (" %s" % it.label)[: w - 4], row_attr) + + if it.right: + rt = (it.right() or "").strip() + if rt: + rt = rt[: max(0, w - 6)] + x = max(2, w - 2 - len(rt)) + rt_attr = ( + row_attr + if sel + else (self.cp(PAIR_HINT) if not dis else self.cp(PAIR_DIM)) + ) + self.add(list_y + i, x, rt, rt_attr) + + if sel and it.hint: + self.add( + list_y + i, + min(w - 4, 30), + (" %s" % it.hint)[: w - 4], + self.cp(PAIR_HINT), + ) + + self.hline(h - 3) + msg = self.message or "" + if msg: + is_err = msg.startswith("error") + attr = (self.cp(PAIR_ERR) if is_err else self.cp(PAIR_OK)) | curses.A_BOLD + self.add(h - 2, 2, msg[: w - 4], attr) + self.stdscr.refresh() + + # ----- modals ----- + + def input_box(self, title: str, prompt: str, initial: str) -> Optional[str]: + h, w = self.stdscr.getmaxyx() + win_h, win_w = 9, min(86, max(46, w - 6)) + top, left = max(0, (h - win_h) // 2), max(0, (w - win_w) // 2) + + win = curses.newwin(win_h, win_w, top, left) + win.keypad(True) + win.border() + + win.addstr(1, 2, title[: win_w - 4], self.cp(PAIR_TITLE) | curses.A_BOLD) + win.addstr(2, 2, prompt[: win_w - 4], self.cp(PAIR_TEXT)) + + buf = list(initial) + curses.curs_set(1) + + while True: + win.addstr(4, 2, " " * (win_w - 4), self.cp(PAIR_TEXT)) + text = "".join(buf) + if len(text) > win_w - 4: + text = text[-(win_w - 4) :] + win.addstr(4, 2, text, self.cp(PAIR_TEXT) | curses.A_BOLD) + win.addstr(6, 2, "enter=ok esc=cancel", self.cp(PAIR_DIM)) + win.refresh() + + ch = win.getch() + if ch == 27: + curses.curs_set(0) + return None + if ch in (curses.KEY_ENTER, 10, 13): + curses.curs_set(0) + return "".join(buf).strip() + if ch in (curses.KEY_BACKSPACE, 127, 8): + if buf: + buf.pop() + elif 32 <= ch <= 126: + buf.append(chr(ch)) + + # ----- value editors ----- + + def cycle_attr(self, attr: str, options: list) -> None: + cur = getattr(self.s, attr) + if cur not in options: + setattr(self.s, attr, options[0]) + else: + setattr(self.s, attr, options[(options.index(cur) + 1) % len(options)]) + + def edit_int(self, label: str, attr: str, minv: Optional[int] = None) -> None: + val = self.input_box(label, "enter an integer:", str(getattr(self.s, attr))) + if val is None or val == "": + return + try: + n = int(val) + except ValueError: + self.message = "error: '%s' is not an integer" % val + return + if minv is not None and n < minv: + self.message = "error: %s must be >= %d" % (label, minv) + return + setattr(self.s, attr, n) + self.message = "%s = %d" % (label, n) + + def edit_float(self, label: str, attr: str, + minv: Optional[float] = None, maxv: Optional[float] = None) -> None: + val = self.input_box(label, "enter a number:", str(getattr(self.s, attr))) + if val is None or val == "": + return + try: + x = float(val) + except ValueError: + self.message = "error: '%s' is not a number" % val + return + if minv is not None and x < minv: + self.message = "error: %s must be >= %g" % (label, minv) + return + if maxv is not None and x > maxv: + self.message = "error: %s must be <= %g" % (label, maxv) + return + setattr(self.s, attr, x) + self.message = "%s = %g" % (label, x) + + # ----- report pager ----- + + def _pager_attr(self, ln: str) -> int: + s = ln.strip() + if "RECOMMENDED" in ln: + return self.cp(PAIR_OK) | curses.A_BOLD + if "<== recommended" in ln: + return self.cp(PAIR_OK) + if "INFEASIBLE" in ln or "unknown arch" in ln: + return self.cp(PAIR_ERR) | curses.A_BOLD + if s.startswith("!!"): + return self.cp(PAIR_HINT) + if s.startswith("T_TILE"): + return self.cp(PAIR_TITLE) | curses.A_BOLD + if set(s) <= {"=", "-"} and s: + return self.cp(PAIR_DIM) + return self.cp(PAIR_TEXT) + + def pager(self, title: str, lines: List[str]) -> None: + top = 0 + note = "" + while True: + self.stdscr.erase() + h, w = self.stdscr.getmaxyx() + + self.add(0, 2, title, self.cp(PAIR_TITLE) | curses.A_BOLD) + self.draw_keybar( + 1, + 2, + [ + ("up/dn/j/k", "scroll"), + ("PgUp/PgDn", "page"), + ("g/G", "top/end"), + ("w", "save"), + ("b/q", "back"), + ], + ) + self.hline(2) + + list_y = 4 + view_h = max(1, h - list_y - 2) + n = len(lines) + top = max(0, min(top, max(0, n - view_h))) + + for i, ln in enumerate(lines[top : top + view_h]): + self.add(list_y + i, 2, ln[: w - 3], self._pager_attr(ln)) + + self.hline(h - 2) + footer = "line %d-%d / %d" % (top + 1, min(top + view_h, n), n) + if note: + footer += " " + note + self.add(h - 1, 2, footer[: w - 4], self.cp(PAIR_DIM)) + self.stdscr.refresh() + + ch = self.stdscr.getch() + if ch in (ord("q"), ord("Q"), ord("b"), 8, 127): + return + if ch in (curses.KEY_UP, ord("k"), ord("K")): + top -= 1 + elif ch in (curses.KEY_DOWN, ord("j"), ord("J")): + top += 1 + elif ch in (curses.KEY_PPAGE,): + top -= view_h + elif ch in (curses.KEY_NPAGE, ord(" ")): + top += view_h + elif ch in (ord("g"),): + top = 0 + elif ch in (ord("G"),): + top = n + elif ch in (ord("w"), ord("W")): + note = self._save_report(lines) + + def _save_report(self, lines: List[str]) -> str: + path = os.path.join(os.getcwd(), "ideal_tile_size_report.txt") + try: + with open(path, "w") as f: + f.write("\n".join(lines) + "\n") + return "saved to %s" % path + except OSError as e: + return "save failed: %s" % e + + def do_compute(self) -> None: + self.pager("recommendation [%s]" % self.s.arch, build_report(self.s)) + + def reset(self) -> None: + self.s = Settings() + self.message = "reset to defaults" + + # ----- menus ----- + + def menu_main(self) -> Tuple[str, str, List[MenuItem]]: + return ( + "entity tile-size advisor", + "main menu:", + [ + MenuItem( + "architecture", + "choose the target GPU (or 'all')", + right=self.arch_label, + on_enter=lambda: self.push("arch"), + ), + MenuItem( + "physics & particles", + "dim, ppc, shape order, precision, ...", + right=lambda: "dim %d / ppc %g / so %d / %s" + % (self.s.dim, self.s.ppc, self.s.shape_order, self.s.precision), + on_enter=lambda: self.push("physics"), + ), + MenuItem( + "tuning knobs", + "residency, budgets, tile range, ...", + right=lambda: "res %d / cap %g / halo %.2f" + % (self.s.target_resident, self.s.npart_cap, self.s.halo_max), + on_enter=lambda: self.push("tuning"), + ), + MenuItem( + "compute recommendation", + "run the model and show the report", + on_enter=self.do_compute, + ), + MenuItem("reset to defaults", "", on_enter=self.reset), + MenuItem("exit", "", on_enter=lambda: setattr(self, "state", "exit")), + ], + ) + + def menu_arch(self) -> Tuple[str, str, List[MenuItem]]: + def choose(name: str): + self.s.arch = name + self.pop() + + def label(name: str) -> str: + mark = "(*)" if name == self.s.arch else "( )" + return "%s %s" % (mark, name) + + def right(name: str) -> str: + if name == "all": + return " + ".join(ALL_ARCHS) + return ARCH[name]["label"] + + items = [ + MenuItem( + label(a), + "select target", + right=(lambda a=a: right(a)), + on_enter=(lambda a=a: choose(a)), + ) + for a in ARCH_CHOICES + ] + items.append(MenuItem("back", "return", on_enter=self.pop)) + return ("architecture", "pick the target GPU:", items) + + def menu_physics(self) -> Tuple[str, str, List[MenuItem]]: + def cyc_dim(): + self.cycle_attr("dim", [1, 2, 3]) + + def cyc_prec(): + self.cycle_attr("precision", ["single", "double"]) + + return ( + "physics & particles", + "set physical inputs:", + [ + MenuItem( + "dim", + "space cycles: 1 / 2 / 3", + right=lambda: str(self.s.dim), + on_enter=cyc_dim, + on_space=cyc_dim, + ), + MenuItem( + "ppc", + "particles per cell (per species)", + right=lambda: "%g" % self.s.ppc, + on_enter=lambda: self.edit_float("ppc", "ppc", minv=0.0), + ), + MenuItem( + "shape_order", + "entity particle shape order", + right=lambda: str(self.s.shape_order), + on_enter=lambda: self.edit_int("shape_order", "shape_order", minv=0), + ), + MenuItem( + "precision", + "space cycles: single / double", + right=lambda: self.s.precision, + on_enter=cyc_prec, + on_space=cyc_prec, + ), + MenuItem( + "components", + "current-field components (J has 3)", + right=lambda: str(self.s.components), + on_enter=lambda: self.edit_int("components", "components", minv=1), + ), + MenuItem( + "drift", + "team_policy_drift CMake knob: cells the scratch halo absorbs (>= spatial_sorting_interval)", + right=lambda: str(self.s.drift), + on_enter=lambda: self.edit_int("drift", "drift", minv=0), + ), + MenuItem("back", "return", on_enter=self.pop), + ], + ) + + def menu_tuning(self) -> Tuple[str, str, List[MenuItem]]: + return ( + "tuning knobs", + "set the model's budgets and ranges:", + [ + MenuItem( + "target_resident", + "work-groups resident per compute unit", + right=lambda: str(self.s.target_resident), + on_enter=lambda: self.edit_int("target_resident", "target_resident", minv=1), + ), + MenuItem( + "npart_cap", + "particle-per-tile budget (contention proxy)", + right=lambda: "%g" % self.s.npart_cap, + on_enter=lambda: self.edit_float("npart_cap", "npart_cap", minv=1.0), + ), + MenuItem( + "halo_max", + "halo fraction above which the tile is grown (0..1)", + right=lambda: "%.2f" % self.s.halo_max, + on_enter=lambda: self.edit_float("halo_max", "halo_max", minv=0.0, maxv=1.0), + ), + MenuItem( + "grid", + "cells per dim (0 disables GPU-fill check)", + right=lambda: str(self.s.grid), + on_enter=lambda: self.edit_int("grid", "grid", minv=0), + ), + MenuItem( + "balance_factor", + "min tiles per compute unit", + right=lambda: str(self.s.balance_factor), + on_enter=lambda: self.edit_int("balance_factor", "balance_factor", minv=1), + ), + MenuItem( + "min_tile", + "smallest T_TILE to consider", + right=lambda: str(self.s.min_tile), + on_enter=lambda: self.edit_int("min_tile", "min_tile", minv=1), + ), + MenuItem( + "max_tile", + "largest T_TILE to consider", + right=lambda: str(self.s.max_tile), + on_enter=lambda: self.edit_int("max_tile", "max_tile", minv=1), + ), + MenuItem("back", "return", on_enter=self.pop), + ], + ) + + def get_menu(self) -> Tuple[str, str, List[MenuItem]]: + if self.state == "mainmenu": + return self.menu_main() + if self.state == "arch": + return self.menu_arch() + if self.state == "physics": + return self.menu_physics() + if self.state == "tuning": + return self.menu_tuning() + self.state = "mainmenu" + return self.menu_main() + + # ----- navigation ----- + + def is_disabled(self, it: MenuItem) -> bool: + return bool(it.disabled and it.disabled()) + + def move_sel(self, items: List[MenuItem], delta: int) -> None: + if not items: + return + n = len(items) + start = self.selected + for _ in range(n): + self.selected = (self.selected + delta) % n + if not self.is_disabled(items[self.selected]): + return + self.selected = start + + def activate(self, items: List[MenuItem], enter: bool) -> None: + if not items: + return + it = items[self.selected] + if self.is_disabled(it): + self.message = "error: option disabled." + return + fn = it.on_enter if enter else it.on_space + if fn: + fn() + + # ----- loop ----- + + def run(self) -> None: + while True: + if self.state == "exit": + return + + title, prompt, items = self.get_menu() + self.draw_menu(title, prompt, items) + + ch = self.stdscr.getch() + + if ch in (ord("q"), ord("Q")): + self.state = "exit" + continue + if ch in (ord("b"), 8, 127): + self.pop() + continue + if ch in (curses.KEY_UP, ord("k"), ord("K")): + self.move_sel(items, -1) + continue + if ch in (curses.KEY_DOWN, ord("j"), ord("J")): + self.move_sel(items, +1) + continue + if ch in (curses.KEY_ENTER, 10, 13): + self.activate(items, enter=True) + continue + if ch == ord(" "): + self.activate(items, enter=False) + continue + + +def run_tui() -> int: + try: + curses.wrapper(lambda stdscr: App(stdscr).run()) + except KeyboardInterrupt: + return 130 + return 0 + + +# ============================ +# CLI (preserved for scripting / sweeps) +# ============================ + +def run_cli(argv) -> int: + ap = argparse.ArgumentParser(description="Recommend entity team tile size (T_TILE).") + ap.add_argument("arch", help="pvc | nvidia | amd (or a100/h100/gh200/mi250x/mi300x, or 'all')") + ap.add_argument("--dim", type=int, default=2, choices=(1, 2, 3)) + ap.add_argument("--ppc", type=float, default=16.0, help="particles per cell (per species)") + ap.add_argument("--shape-order", type=int, default=2, help="particle shape order (entity shape_order)") + ap.add_argument("--precision", choices=("single", "double"), default="single") + ap.add_argument("--components", type=int, default=3, help="current-field components (J has 3)") + ap.add_argument("--drift", type=int, default=1, + help="team_policy_drift CMake knob (compile-time): cells of drift the scratch " + "halo absorbs; size it >= spatial_sorting_interval") + ap.add_argument("--target-resident", type=int, default=2, help="work-groups resident per compute unit") + ap.add_argument("--npart-cap", type=float, default=1600, + help="particle-per-tile budget (SLM-atomic-contention / load-balance proxy)") + ap.add_argument("--halo-max", type=float, default=0.70, help="halo fraction above which the tile is grown") + ap.add_argument("--grid", type=int, default=0, help="cells per dim (optional; enables a GPU-fill check)") + ap.add_argument("--balance-factor", type=int, default=4, help="min tiles per compute unit") + ap.add_argument("--min-tile", type=int, default=4, + help="smallest T_TILE to consider (entity's team_policy_tile_sizes starts at 4)") + ap.add_argument("--max-tile", type=int, default=64) + p = ap.parse_args(argv) + + for ln in build_report(p): + print(ln) + return 0 + + +def main() -> int: + # no arguments -> interactive TUI; any argument -> scriptable CLI + if len(sys.argv) > 1: + return run_cli(sys.argv[1:]) + return run_tui() + + +if __name__ == "__main__": + raise SystemExit(main()) From 61ebebfffdf38c91456311087b94b96a20917d99 Mon Sep 17 00:00:00 2001 From: LudwigBoess Date: Fri, 26 Jun 2026 03:03:23 +0000 Subject: [PATCH 030/125] team policy for GRPIC --- src/engines/grpic/currents.h | 228 ++++++++++++++++++++++++++++++----- src/engines/grpic/grpic.hpp | 3 +- 2 files changed, 200 insertions(+), 31 deletions(-) diff --git a/src/engines/grpic/currents.h b/src/engines/grpic/currents.h index 11c847533..cb4032f54 100644 --- a/src/engines/grpic/currents.h +++ b/src/engines/grpic/currents.h @@ -2,6 +2,8 @@ * @file engines/grpic/currents.h * @brief Current deposition and filtering routines for the GRPIC engine * @implements + * - ntt::grpic::CallDepositKernel<> -> void (flat path) + * - ntt::grpic::CallDepositKernelTiled<> -> void (TEAM_POLICY) * - ntt::grpic::CurrentsDeposit<> -> void * - ntt::grpic::CurrentsFilter<> -> void * @namespaces: @@ -12,8 +14,11 @@ #define ENGINES_GRPIC_CURRENTS_H #include "enums.h" +#include "global.h" +#include "arch/kokkos_aliases.h" #include "traits/metric.h" +#include "utils/error.h" #include "utils/log.h" #include "utils/param_container.h" @@ -26,13 +31,198 @@ namespace ntt { namespace grpic { + template + void CallDepositKernel(const Particles& species, + const M& local_metric, + const scatter_ndfield_t& scatter_cur, + real_t dt) { + Kokkos::parallel_for("CurrentsDeposit", + species.rangeActiveParticles(), + kernel::DepositCurrents_kernel( + scatter_cur, + species, + local_metric, + (real_t)(species.charge()), + dt)); + } + +#if defined(TEAM_POLICY) + /** + * @brief Tiled deposit launcher (TeamPolicy + per-team scratch). + * + * Identical in structure to the SRPIC launcher (`engines/srpic/currents.h`): + * iterates over `tile_layout.ntiles_total` teams; each team accumulates its + * tile's particle contributions in SLM scratch and atomically flushes to the + * global J (here `cur0`, the GRPIC half-step current). Requires the species + * to have been sorted with `team_policy` enabled (`tile_layout` populated by + * `SortSpatially`). + * + * The deposit body (`kernel::DepositOneParticle`) is + * the same shared math used by the flat path — it already carries the GR + * velocity-recovery branch — so the only engine-specific differences from + * SRPIC are the `SimEngine::GRPIC` tag and the `cur0` target. + * + * Falls back to the flat kernel for the tail `[npart_partitioned, npart)` + * exactly as SRPIC does; see the per-step coverage note in + * `kernels/currents_deposit.hpp`. + */ + template + void CallDepositKernelTiled(const Particles& species, + const M& local_metric, + const ndfield_t& cur, + real_t dt, + int team_size_req) { + static_assert(O <= 11u, "Shape order must be <= 11"); + constexpr unsigned short T = static_cast( + TEAM_POLICY_TILE_SIZE); + const auto& layout = species.tile_layout(); + raise::ErrorIf(layout.ntiles_total == 0u, + "CallDepositKernelTiled: tile_layout has 0 tiles — call " + "SortSpatially before CurrentsDeposit", + HERE); + raise::ErrorIf(layout.tile_offsets.extent(0) != layout.ntiles_total + 1u, + "CallDepositKernelTiled: tile_offsets size inconsistent " + "with ntiles_total", + HERE); + + auto deposit_kernel = + kernel::DepositCurrentsTiled_kernel { + cur, species, local_metric, (real_t)(species.charge()), + dt, layout, species.npart() + }; + + const auto scratch = Kokkos::PerTeam( + decltype(deposit_kernel)::scratch_bytes()); + + // Team (work-group) size. The default (team_size_req == 0) leaves + // Kokkos::AUTO, which sizes the team from the backend occupancy + // heuristic. A positive `algorithms.deposit.team_policy_team_size` + // overrides it, clamped to the scratch/backend-feasible maximum so an + // over-large request cannot abort the launch (Kokkos errors when + // team_size > team_size_max). No portable subgroup rounding is applied; + // pick a multiple of the device subgroup width (printed per arch by + // ideal_tile_size.py) for the best occupancy. + Kokkos::TeamPolicy<> policy(static_cast(layout.ntiles_total), + Kokkos::AUTO); + policy.set_scratch_size(0, scratch); + if (team_size_req > 0) { + const int ts_max = policy.team_size_max(deposit_kernel, + Kokkos::ParallelForTag {}); + int ts = team_size_req; + if (ts > ts_max) { + raise::Warning( + fmt::format("algorithms.deposit.team_policy_team_size = %d exceeds " + "the tiled-deposit maximum %d on this backend; clamping " + "to %d", + team_size_req, + ts_max, + ts_max), + HERE); + ts = ts_max; + } + policy = Kokkos::TeamPolicy<>(static_cast(layout.ntiles_total), ts); + policy.set_scratch_size(0, scratch); + logger::Checkpoint( + fmt::format("Tiled deposit: explicit team size %d", ts), + HERE); + } + Kokkos::parallel_for("CurrentsDepositTiled", policy, deposit_kernel); + + // Particles appended since the last sort (injection / MPI receive on a + // no-sort step) live past the partition and are not visited by any team + // above. Deposit that tail [npart_partitioned, npart) with the flat + // scatter-view kernel so every active particle is deposited exactly + // once. The range is empty when the species was just sorted (the + // every-step-sorted common case), so this is a no-op there. + if (species.npart() > layout.npart_partitioned) { + // `cur` is a const ref; take a non-const View handle (shallow copy, + // shares storage) so the scatter view can contribute back into it. + auto cur_nc = cur; + auto scatter_cur = Kokkos::Experimental::create_scatter_view(cur_nc); + Kokkos::parallel_for( + "CurrentsDepositTiledTail", + CreateParticleRangePolicy({ layout.npart_partitioned }, + { species.npart() }), + kernel::DepositCurrents_kernel( + scatter_cur, + species, + local_metric, + (real_t)(species.charge()), + dt)); + Kokkos::Experimental::contribute(cur_nc, scatter_cur); + } + } +#endif // TEAM_POLICY + template void CurrentsDeposit(Domain& domain, const prm::Parameters& engine_params) { + const auto dt = engine_params.get("dt"); + // GRPIC deposits the half-step current into `cur0` (the engine no longer + // pre-zeros it — this is the single source of truth, matching SRPIC). + Kokkos::deep_copy(domain.fields.cur0, ZERO); + +#if defined(TEAM_POLICY) + // Optional runtime override for the tiled-deposit team (work-group) size; + // 0 (default) keeps Kokkos::AUTO. Clamped to the backend max in the + // launcher (see CallDepositKernelTiled). + const auto team_size_req = static_cast( + engine_params.get("team_policy_team_size", + std::optional { 0u })); + + // Tiled deposit. Correctness no longer depends on the SoA being in a + // "sorted" state at deposit time — the tiled kernel handles a stale + // partition per-particle (escape valve for drifted particles, dead-tag + // clamp, and the launcher's flat tail pass for appended particles). The + // only case the tiled kernel cannot serve is the very first step, before + // any SortSpatially has populated a layout; that species takes the flat + // scatter-view path for that step alone. See engines/srpic/currents.h and + // kernels/currents_deposit.hpp for the full coverage argument. + for (auto& species : domain.species) { + if ((species.pusher() == ParticlePusher::NONE) or + (species.npart() == 0) or cmp::AlmostZero_host(species.charge())) { + continue; + } + const auto& layout = species.tile_layout(); + if (layout.ntiles_total == 0u or layout.tile_offsets.extent(0) == 0u) { + logger::Checkpoint( + fmt::format("Launching currents deposit (flat, no sort yet) for " + "%d [%s] : %lu %f", + species.index(), + species.label().c_str(), + species.npart(), + (double)species.charge()), + HERE); + auto scatter_cur0 = Kokkos::Experimental::create_scatter_view( + domain.fields.cur0); + CallDepositKernel(species, + domain.mesh.metric, + scatter_cur0, + dt); + Kokkos::Experimental::contribute(domain.fields.cur0, scatter_cur0); + } else { + logger::Checkpoint( + fmt::format("Launching tiled currents deposit for %d [%s] : %lu %f", + species.index(), + species.label().c_str(), + species.npart(), + (double)species.charge()), + HERE); + CallDepositKernelTiled(species, + domain.mesh.metric, + domain.fields.cur0, + dt, + team_size_req); + } + } +#else auto scatter_cur0 = Kokkos::Experimental::create_scatter_view( domain.fields.cur0); - const auto dt = engine_params.get("dt"); for (auto& species : domain.species) { + if ((species.pusher() == ParticlePusher::NONE) or + (species.npart() == 0) or cmp::AlmostZero_host(species.charge())) { + continue; + } logger::Checkpoint( fmt::format("Launching currents deposit kernel for %d [%s] : %lu %f", species.index(), @@ -40,36 +230,14 @@ namespace ntt { species.npart(), (double)species.charge()), HERE); - if (species.npart() == 0 || cmp::AlmostZero(species.charge())) { - continue; - } - Kokkos::parallel_for("CurrentsDeposit", - species.rangeActiveParticles(), - kernel::DepositCurrents_kernel( - scatter_cur0, - species.i1, - species.i2, - species.i3, - species.i1_prev, - species.i2_prev, - species.i3_prev, - species.dx1, - species.dx2, - species.dx3, - species.dx1_prev, - species.dx2_prev, - species.dx3_prev, - species.ux1, - species.ux2, - species.ux3, - species.phi, - species.weight, - species.tag, - domain.mesh.metric, - (real_t)(species.charge()), - dt)); + + CallDepositKernel(species, + domain.mesh.metric, + scatter_cur0, + dt); } Kokkos::Experimental::contribute(domain.fields.cur0, scatter_cur0); +#endif } template @@ -103,4 +271,4 @@ namespace ntt { } // namespace grpic } // namespace ntt -#endif // ENGINES_GRPIC_CURRENTS_H \ No newline at end of file +#endif // ENGINES_GRPIC_CURRENTS_H diff --git a/src/engines/grpic/grpic.hpp b/src/engines/grpic/grpic.hpp index fc2daa4de..621592955 100644 --- a/src/engines/grpic/grpic.hpp +++ b/src/engines/grpic/grpic.hpp @@ -416,7 +416,8 @@ namespace ntt { */ if (deposit_enabled) { timers.start("CurrentDeposit"); - Kokkos::deep_copy(dom.fields.cur0, ZERO); + // `cur0` is zeroed inside grpic::CurrentsDeposit (matching SRPIC), + // so no pre-zero is needed here. grpic::CurrentsDeposit(dom, this->engineParams()); timers.stop("CurrentDeposit"); From 210454b150ce7674985c030076b3429ca746722e Mon Sep 17 00:00:00 2001 From: LudwigBoess Date: Fri, 26 Jun 2026 14:03:51 +0000 Subject: [PATCH 031/125] bugfix in deposit tests --- tests/kernels/deposit.cpp | 18 ++++++++++++------ tests/kernels/deposit_tiled.cpp | 16 ++++++++++++++++ 2 files changed, 28 insertions(+), 6 deletions(-) diff --git a/tests/kernels/deposit.cpp b/tests/kernels/deposit.cpp index 24f544202..10095d304 100644 --- a/tests/kernels/deposit.cpp +++ b/tests/kernels/deposit.cpp @@ -142,15 +142,21 @@ void testDeposit(const std::vector& res, auto J_scat = Kokkos::Experimental::create_scatter_view(J); + // The deposit kernel now takes a `ParticleArrays` SoA struct instead of + // the individual per-component arrays. Pack the per-test arrays into one; + // payload (pld_*) members stay default (unused here). + ParticleArrays pa; + pa.i1 = i1, pa.i2 = i2, pa.i3 = i3; + pa.i1_prev = i1_prev, pa.i2_prev = i2_prev, pa.i3_prev = i3_prev; + pa.dx1 = dx1, pa.dx2 = dx2, pa.dx3 = dx3; + pa.dx1_prev = dx1_prev, pa.dx2_prev = dx2_prev, pa.dx3_prev = dx3_prev; + pa.ux1 = ux1, pa.ux2 = ux2, pa.ux3 = ux3; + pa.phi = phi, pa.weight = weight, pa.tag = tag; + // clang-format off Kokkos::parallel_for("CurrentsDeposit", 10, kernel::DepositCurrents_kernel(J_scat, - i1, i2, i3, - i1_prev, i2_prev, i3_prev, - dx1, dx2, dx3, - dx1_prev, dx2_prev, dx3_prev, - ux1, ux2, ux3, - phi, weight, tag, + pa, metric, charge, inv_dt)); // clang-format on diff --git a/tests/kernels/deposit_tiled.cpp b/tests/kernels/deposit_tiled.cpp index 3cc2f62b0..1b6237ae4 100644 --- a/tests/kernels/deposit_tiled.cpp +++ b/tests/kernels/deposit_tiled.cpp @@ -339,6 +339,22 @@ namespace { // boundary. template void run_drift_case() { + // This case deposits boundary-adjacent particles (cells touching the + // ghost stripe), so an order-O stencil must fit inside the field's + // N_GHOSTS ghost layers. N_GHOSTS is a compile-time constant fixed by + // the build's SHAPE_ORDER ((SHAPE_ORDER+1)/2 + 1); a build whose ghost + // width is smaller than order O requires would deposit outside the + // field -- silent on GPU (no Kokkos View bounds guard; the overshoot + // cells carry zero shape-weight so results still match) but heap + // corruption on a host/SERIAL build. Skip those orders here; build at + // the matching SHAPE_ORDER to drift-test higher orders. The equivalence + // ("X-1") cases above stay interior, so they exercise all orders. + if constexpr ((O + 1u) / 2u + 1u > N_GHOSTS) { + std::cerr << "deposit_tiled[drift] SKIP O=" << O << " T_TILE=" << T_TILE + << " (needs N_GHOSTS>=" << ((O + 1u) / 2u + 1u) + << ", build has " << N_GHOSTS << ")\n"; + return; + } using metric_t = metric::Minkowski; constexpr unsigned short nx1 = 50u, nx2 = 50u; metric_t metric { { nx1, nx2 }, { { 0.0, 55.0 }, { 0.0, 55.0 } }, {} }; From e9cfa06efbd7a975d98644a5d5cd0955c3d0bd45 Mon Sep 17 00:00:00 2001 From: LudwigBoess Date: Fri, 26 Jun 2026 14:23:32 +0000 Subject: [PATCH 032/125] added explicit charge conservation test to the tiled deposit test --- tests/kernels/deposit_tiled.cpp | 53 +++++++++++++++++++++++++++++++++ 1 file changed, 53 insertions(+) diff --git a/tests/kernels/deposit_tiled.cpp b/tests/kernels/deposit_tiled.cpp index 1b6237ae4..f058d6f30 100644 --- a/tests/kernels/deposit_tiled.cpp +++ b/tests/kernels/deposit_tiled.cpp @@ -162,6 +162,55 @@ namespace { << " T_TILE=" << T_TILE << " max_diff=" << max_diff << '\n'; } + // Intrinsic charge-conservation check on a single deposited J field. + // Esirkepov/zigzag deposits satisfy the discrete continuity equation, so + // the spatial sum of the discrete divergence div.J = dJx/dx + dJy/dy + // vanishes whenever the summation region encloses every particle's full + // stencil (J == 0 on the region's outer boundary). This is evaluated on + // J_tiled ALONE -- it does not compare against the flat reference -- so it + // certifies the per-particle escape valve deposits each drifted particle's + // stencil as one coherent unit: no cell dropped, duplicated, or split + // between SLM scratch and global J. (The run_drift_case order guard keeps + // every stencil inside [0, j_ext), so the extreme ghost cells stay zero and + // the telescoping boundary flux is genuinely zero rather than clipped.) + // Accumulated in double regardless of build precision to keep the + // tolerance tight. + void check_charge_conservation(const ndfield_t& J, + unsigned short O, + unsigned short T_TILE, + const char* label) { + auto h = Kokkos::create_mirror_view(J); + Kokkos::deep_copy(h, J); + + double sum_div = 0.0; // Sum over the field of div.J (jx1 -> dx, jx2 -> dy). + double abs_tot = 0.0; // Total |J|, sets the relative tolerance scale. + for (ncells_t i = 1; i < h.extent(0); ++i) { + for (ncells_t j = 1; j < h.extent(1); ++j) { + sum_div += (static_cast(h(i, j, 0)) - + static_cast(h(i - 1, j, 0))) + + (static_cast(h(i, j, 1)) - + static_cast(h(i, j - 1, 1))); + } + } + for (ncells_t i = 0; i < h.extent(0); ++i) { + for (ncells_t j = 0; j < h.extent(1); ++j) { + abs_tot += std::fabs(static_cast(h(i, j, 0))) + + std::fabs(static_cast(h(i, j, 1))); + } + } + const double tol = 1.0e-5 * (abs_tot > 1.0 ? abs_tot : 1.0); + if (std::fabs(sum_div) > tol) { + std::cerr << "deposit_tiled[" << label + << "] CHARGE NON-CONSERVED for O=" << O << " T_TILE=" << T_TILE + << " : sum(div.J)=" << sum_div << " tol=" << tol + << " (abs_tot=" << abs_tot << ")\n"; + throw std::logic_error( + "DepositCurrentsTiled_kernel charge non-conservation"); + } + std::cerr << "deposit_tiled[" << label << "] charge-conserved O=" << O + << " T_TILE=" << T_TILE << " sum(div.J)=" << sum_div << '\n'; + } + template void run_one_case() { using metric_t = metric::Minkowski; @@ -484,6 +533,10 @@ namespace { } compare_J_fields(J_flat, J_tiled, O, T_TILE, "drift"); + // Self-contained conservation check on the escape-valve output: the + // drifted particles all take the per-particle global-J path, so this + // certifies that path is charge-conserving without leaning on J_flat. + check_charge_conservation(J_tiled, O, T_TILE, "drift"); } template From 4e9e9314eacb9919854ea4587f650505ae54be90 Mon Sep 17 00:00:00 2001 From: LudwigBoess Date: Sat, 27 Jun 2026 16:28:11 +0000 Subject: [PATCH 033/125] port of load balancing from my fork --- input.example.toml | 29 ++ src/engines/engine.hpp | 33 ++ src/engines/grpic/grpic.hpp | 6 +- src/engines/srpic/srpic.hpp | 7 +- src/framework/CMakeLists.txt | 2 + src/framework/containers/fields.h | 5 +- src/framework/containers/fields_io.cpp | 27 +- src/framework/domain/domain.h | 4 + src/framework/domain/mesh.h | 9 + src/framework/domain/metadomain.h | 17 + src/framework/domain/metadomain_chckpt.cpp | 13 +- src/framework/domain/metadomain_io.cpp | 3 + src/framework/domain/metadomain_loadbal.cpp | 428 ++++++++++++++++++++ src/framework/parameters/parameters.cpp | 46 +++ src/output/utils/writers.cpp | 18 +- src/output/utils/writers.h | 3 +- src/output/writer.cpp | 83 +++- src/output/writer.h | 11 + 18 files changed, 710 insertions(+), 34 deletions(-) create mode 100644 src/framework/domain/metadomain_loadbal.cpp diff --git a/input.example.toml b/input.example.toml index 87ca48f3c..8d6f25117 100644 --- a/input.example.toml +++ b/input.example.toml @@ -28,6 +28,35 @@ # @example: [2, 2, 2] (total of 8 domains) decomposition = "" + # Diffusion-style dynamic load balancing (Cartesian metrics only). + # Domain boundaries between MPI neighbors are nudged to equalize the + # active-particle count per rank. All inter-rank traffic uses only the + # existing nearest-neighbor field/particle communication paths. + [simulation.domain.load_balance] + # Enable dynamic load balancing + # @type: bool + # @default: false + enable = "" + # Run the rebalancer every `interval` timesteps (0 disables) + # @type: int + # @default: 0 + interval = "" + # Dimensions along which load is redistributed (1 = x1, 2 = x2, 3 = x3) + # @type: array of int, subset of [1, 2, 3] + # @default: [1] + dimensions = "" + # Skip rebalancing along a dim when (max - min) / mean of the per-slice + # particle count is below this fraction + # @type: float + # @default: 0.1 + tolerance = "" + # Maximum cell-shift per interior boundary per event; clamped at compile + # time to N_GHOSTS so the migrating field strip is already cached in the + # rank's ghost zone. + # @type: int + # @default: N_GHOSTS + max_shift = "" + [grid] # Spatial resolution of the grid # @required diff --git a/src/engines/engine.hpp b/src/engines/engine.hpp index 057bfcb5a..161339a9d 100644 --- a/src/engines/engine.hpp +++ b/src/engines/engine.hpp @@ -253,6 +253,7 @@ namespace ntt { "ParticlePusher", "FieldBoundaries", "ParticleBoundaries", "Communications", "Injector", "Custom", + "LoadBalance", "ParticleSort", "Output", "Checkpoint" }, []() { @@ -267,6 +268,17 @@ namespace ntt { const auto clear_interval = m_params.template get( "particles.clear_interval"); + const auto lb_enable = m_params.template get( + "simulation.domain.load_balance.enable"); + const auto lb_interval = m_params.template get( + "simulation.domain.load_balance.interval"); + const auto lb_dim_mask = m_params.template get( + "simulation.domain.load_balance.dim_mask"); + const auto lb_tolerance = m_params.template get( + "simulation.domain.load_balance.tolerance"); + const auto lb_max_shift = m_params.template get( + "simulation.domain.load_balance.max_shift"); + // main algorithm loop while (step < max_steps) { // run the engine-dependent algorithm step @@ -282,6 +294,27 @@ namespace ntt { }); timers.stop("Custom"); } + if constexpr (MetricClass) { + if (lb_enable and lb_interval > 0 and (step + 1) % lb_interval == 0 and + lb_dim_mask != 0u) { + timers.start("LoadBalance"); + m_metadomain.Rebalance(lb_dim_mask, + lb_tolerance, + static_cast(lb_max_shift)); + timers.stop("LoadBalance"); + } + } + // Sort particles last — after step_forward, CustomPostStep, and + // LoadBalance — so the tile layout (and dead-particle compaction) the + // next step's deposit relies on reflects every particle change made this + // step: moving-window shift/injection and in-place dead-tagging done in + // CustomPostStep, plus any rebalance migration. + timers.start("ParticleSort"); + m_metadomain.runOnLocalDomains([this](auto& dom) { + m_metadomain.SortParticles(time, step, m_params, dom); + }); + timers.stop("ParticleSort"); + auto print_prtl_clear = (clear_interval > 0 and step % clear_interval == 0 and step > 0); diff --git a/src/engines/grpic/grpic.hpp b/src/engines/grpic/grpic.hpp index 621592955..3487d2867 100644 --- a/src/engines/grpic/grpic.hpp +++ b/src/engines/grpic/grpic.hpp @@ -613,9 +613,9 @@ namespace ntt { timers.stop("FieldBoundaries"); } - timers.start("ParticleSort"); - m_metadomain.SortParticles(time, step, m_params, dom); - timers.stop("ParticleSort"); + // NOTE: particle sorting is intentionally NOT done here. It runs once per + // step in the engine loop (Engine::run) after CustomPostStep and + // LoadBalance — see the SRPIC engine for the rationale. /** * Finally: em0::B at n-1/2 diff --git a/src/engines/srpic/srpic.hpp b/src/engines/srpic/srpic.hpp index 94a8949c1..d0352318e 100644 --- a/src/engines/srpic/srpic.hpp +++ b/src/engines/srpic/srpic.hpp @@ -182,9 +182,10 @@ namespace ntt { timers.stop("Injector"); } - timers.start("ParticleSort"); - m_metadomain.SortParticles(time, step, m_params, dom); - timers.stop("ParticleSort"); + // NOTE: particle sorting is intentionally NOT done here. It runs once per + // step in the engine loop (Engine::run) after CustomPostStep and + // LoadBalance, so the layout the next deposit uses reflects window + // shifts/injection and dead-tagging performed in CustomPostStep. } }; diff --git a/src/framework/CMakeLists.txt b/src/framework/CMakeLists.txt index 5343f3d8a..377c1f98a 100644 --- a/src/framework/CMakeLists.txt +++ b/src/framework/CMakeLists.txt @@ -19,6 +19,7 @@ # * domain/metadomain_stats.cpp # * domain/metadomain_io.cpp # * domain/metadomain_reshape.cpp +# * domain/metadomain_loadbal.cpp # * containers/particles.cpp # * containers/particles_comm.cpp # * containers/particles_io.cpp @@ -59,6 +60,7 @@ set(SOURCES ${SRC_DIR}/domain/metadomain_sort.cpp ${SRC_DIR}/domain/metadomain_stats.cpp ${SRC_DIR}/domain/metadomain_reshape.cpp + ${SRC_DIR}/domain/metadomain_loadbal.cpp ${SRC_DIR}/containers/particles.cpp ${SRC_DIR}/containers/particles_sort.cpp ${SRC_DIR}/containers/fields.cpp) diff --git a/src/framework/containers/fields.h b/src/framework/containers/fields.h index 4acabf7b4..2dcfd0a7e 100644 --- a/src/framework/containers/fields.h +++ b/src/framework/containers/fields.h @@ -175,7 +175,10 @@ namespace ntt { void CheckpointRead(adios2::IO&, adios2::Engine&, const adios2::Box&); - void CheckpointWrite(adios2::IO&, adios2::Engine&) const; + void CheckpointWrite(adios2::IO&, + adios2::Engine&, + const std::vector&, + const std::vector&) const; #endif }; diff --git a/src/framework/containers/fields_io.cpp b/src/framework/containers/fields_io.cpp index 683519493..a17ab1430 100644 --- a/src/framework/containers/fields_io.cpp +++ b/src/framework/containers/fields_io.cpp @@ -62,13 +62,27 @@ namespace ntt { } template - void Fields::CheckpointWrite(adios2::IO& io, adios2::Engine& writer) const { + void Fields::CheckpointWrite( + adios2::IO& io, + adios2::Engine& writer, + const std::vector& local_shape, + const std::vector& local_offset) const { logger::Checkpoint("Writing fields checkpoint", HERE); - out::WriteNDField(io, writer, "em", em); + // Per-rank slab: re-set the variable selection to track the (possibly + // rebalanced) local layout. The component axis is always full. + auto build_range = [&](unsigned short ncomp) { + auto start = adios2::Dims(local_offset.begin(), local_offset.end()); + auto count = adios2::Dims(local_shape.begin(), local_shape.end()); + start.push_back(0); + count.push_back(ncomp); + return adios2::Box(start, count); + }; + + out::WriteNDField(io, writer, "em", em, build_range(6)); if (S == ntt::SimEngine::GRPIC) { - out::WriteNDField(io, writer, "em0", em0); - out::WriteNDField(io, writer, "cur", cur); + out::WriteNDField(io, writer, "em0", em0, build_range(6)); + out::WriteNDField(io, writer, "cur", cur, build_range(3)); } } @@ -81,7 +95,10 @@ namespace ntt { template void Fields::CheckpointRead(adios2::IO&, \ adios2::Engine&, \ const adios2::Box&); \ - template void Fields::CheckpointWrite(adios2::IO&, adios2::Engine&) const; + template void Fields::CheckpointWrite(adios2::IO&, \ + adios2::Engine&, \ + const std::vector&, \ + const std::vector&) const; FIELDS_CHECKPOINTS(Dim::_1D, SimEngine::SRPIC) FIELDS_CHECKPOINTS(Dim::_2D, SimEngine::SRPIC) diff --git a/src/framework/domain/domain.h b/src/framework/domain/domain.h index 747ab4576..0b708e555 100644 --- a/src/framework/domain/domain.h +++ b/src/framework/domain/domain.h @@ -165,6 +165,10 @@ namespace ntt { m_neighbor_idx[dir] = idx; } + void set_offset_ncells(const std::vector& off) { + m_offset_ncells = off; + } + /* printer overload ----------------------------------------------------- */ auto Report() const -> std::string { std::string report; diff --git a/src/framework/domain/mesh.h b/src/framework/domain/mesh.h index aa2eea726..50449ae49 100644 --- a/src/framework/domain/mesh.h +++ b/src/framework/domain/mesh.h @@ -65,6 +65,15 @@ namespace ntt { new (&metric) M { this->m_resolution, new_extent, m_metric_params_raw }; } + void set_resolution_and_extent(const std::vector& new_res, + const boundaries_t& new_extent) { + raise::ErrorIf(new_res.size() != D, "invalid resolution dim", HERE); + this->m_resolution = new_res; + m_extent = new_extent; + metric.~M(); + new (&metric) M { this->m_resolution, m_extent, m_metric_params_raw }; + } + /** * @brief Get the intersection of the mesh with a box * @param box physical extent diff --git a/src/framework/domain/metadomain.h b/src/framework/domain/metadomain.h index 8af9e6e0b..f090dade1 100644 --- a/src/framework/domain/metadomain.h +++ b/src/framework/domain/metadomain.h @@ -132,6 +132,23 @@ namespace ntt { /* domain update-related ------------------------------------------------ */ void ShiftByCells(int, in = in::x1); + /** + * @brief Rebalance the load (active particles) across MPI domains by + * shifting interior domain boundaries between neighbors. + * @param dim_mask bitmask: bit d (0,1,2) set => balance along dim x1/x2/x3 + * @param tolerance skip if (max-min)/mean of the per-slice load is below + * this fraction + * @param max_shift_cells per-event cap for any single boundary movement, + * additionally clamped to N_GHOSTS so the field strip we need is already + * present in the local ghost zone + * @note Only neighbor communication is used (CommunicateFields ghosts + + * CommunicateParticles). + */ + void Rebalance(unsigned int dim_mask, + real_t tolerance, + ncells_t max_shift_cells) + requires(MetricClass); + /* output-related ------------------------------------------------------- */ #if defined(OUTPUT_ENABLED) void InitWriter(adios2::ADIOS*, const SimulationParams&); diff --git a/src/framework/domain/metadomain_chckpt.cpp b/src/framework/domain/metadomain_chckpt.cpp index 1116df85e..fbf0f11cd 100644 --- a/src/framework/domain/metadomain_chckpt.cpp +++ b/src/framework/domain/metadomain_chckpt.cpp @@ -125,8 +125,19 @@ namespace ntt { } params.saveTOML(g_checkpoint_writer.written().back().second, current_time); + // Recompute the local with-ghosts shape/offset every step so the + // ADIOS variable selection tracks any rebalance that has happened + // since InitCheckpointWriter. + std::vector loc_off_with_ghosts; + for (auto d { 0u }; d < M::Dim; ++d) { + loc_off_with_ghosts.push_back( + local_domain->offset_ncells()[d] + + 2 * N_GHOSTS * local_domain->offset_ndomains()[d]); + } local_domain->fields.CheckpointWrite(g_checkpoint_writer.io(), - g_checkpoint_writer.writer()); + g_checkpoint_writer.writer(), + local_domain->mesh.n_all(), + loc_off_with_ghosts); #if !defined(MPI_ENABLED) const std::size_t dom_tot = 1, dom_offset = 0; #else diff --git a/src/framework/domain/metadomain_io.cpp b/src/framework/domain/metadomain_io.cpp index 3fc0f5f57..d53dc49a1 100644 --- a/src/framework/domain/metadomain_io.cpp +++ b/src/framework/domain/metadomain_io.cpp @@ -409,6 +409,9 @@ namespace ntt { } } } + // Refresh the writer's cached per-rank slab so that field/mesh writes + // pick up the (possibly rebalanced) current local layout. + g_writer.setLocalLayout(off_ncells_with_ghosts, loc_shape_with_ghosts); for (auto dim { 0u }; dim < M::Dim; ++dim) { const auto l_size = local_domain->mesh.n_active()[dim]; const auto l_offset = local_domain->offset_ncells()[dim]; diff --git a/src/framework/domain/metadomain_loadbal.cpp b/src/framework/domain/metadomain_loadbal.cpp new file mode 100644 index 000000000..0768acad6 --- /dev/null +++ b/src/framework/domain/metadomain_loadbal.cpp @@ -0,0 +1,428 @@ +#include "enums.h" +#include "global.h" + +#include "arch/kokkos_aliases.h" +#include "traits/metric.h" +#include "utils/error.h" +#include "utils/log.h" +#include "utils/numeric.h" + +#include "framework/containers/fields.h" +#include "framework/domain/metadomain.h" +#include "framework/specialization_registry.h" + +#if defined(MPI_ENABLED) + #include "arch/mpi_aliases.h" + #include "arch/mpi_tags.h" + + #include +#endif + +#include + +#include +#include +#include +#include +#include +#include + +namespace ntt { + +#if defined(MPI_ENABLED) + // Copy old field into new (already-zero) field, applying a shift in the + // active-cell offset along each dimension. + // For new (with-ghost) index i, old (with-ghost) index = i + delta[d]. + // Cells whose old index is out of bounds are left at 0 and will be filled + // by CommunicateFields() afterwards. + template + void CopyShifted(const ndfield_mirror_t& src_h, + ndfield_t& dst_dev, + const std::vector& delta) { + auto dst_h = Kokkos::create_mirror_view(dst_dev); + Kokkos::deep_copy(dst_h, ZERO); + + if constexpr (D == Dim::_1D) { + const int new_n1 = static_cast(dst_h.extent(0)); + const int old_n1 = static_cast(src_h.extent(0)); + for (int i1 = 0; i1 < new_n1; ++i1) { + const int oi1 = i1 + delta[0]; + if (oi1 < 0 or oi1 >= old_n1) { + continue; + } + for (auto c { 0u }; c < NC; ++c) { + dst_h(i1, c) = src_h(oi1, c); + } + } + } else if constexpr (D == Dim::_2D) { + const int new_n1 = static_cast(dst_h.extent(0)); + const int new_n2 = static_cast(dst_h.extent(1)); + const int old_n1 = static_cast(src_h.extent(0)); + const int old_n2 = static_cast(src_h.extent(1)); + for (int i1 = 0; i1 < new_n1; ++i1) { + const int oi1 = i1 + delta[0]; + if (oi1 < 0 or oi1 >= old_n1) { + continue; + } + for (int i2 = 0; i2 < new_n2; ++i2) { + const int oi2 = i2 + delta[1]; + if (oi2 < 0 or oi2 >= old_n2) { + continue; + } + for (auto c { 0u }; c < NC; ++c) { + dst_h(i1, i2, c) = src_h(oi1, oi2, c); + } + } + } + } else if constexpr (D == Dim::_3D) { + const int new_n1 = static_cast(dst_h.extent(0)); + const int new_n2 = static_cast(dst_h.extent(1)); + const int new_n3 = static_cast(dst_h.extent(2)); + const int old_n1 = static_cast(src_h.extent(0)); + const int old_n2 = static_cast(src_h.extent(1)); + const int old_n3 = static_cast(src_h.extent(2)); + for (int i1 = 0; i1 < new_n1; ++i1) { + const int oi1 = i1 + delta[0]; + if (oi1 < 0 or oi1 >= old_n1) { + continue; + } + for (int i2 = 0; i2 < new_n2; ++i2) { + const int oi2 = i2 + delta[1]; + if (oi2 < 0 or oi2 >= old_n2) { + continue; + } + for (int i3 = 0; i3 < new_n3; ++i3) { + const int oi3 = i3 + delta[2]; + if (oi3 < 0 or oi3 >= old_n3) { + continue; + } + for (auto c { 0u }; c < NC; ++c) { + dst_h(i1, i2, i3, c) = src_h(oi1, oi2, oi3, c); + } + } + } + } + } + Kokkos::deep_copy(dst_dev, dst_h); + } +#endif // MPI_ENABLED + + template + void Metadomain::Rebalance(unsigned int dim_mask, + real_t tolerance, + ncells_t max_shift_cells) + requires(MetricClass) + { +#if !defined(MPI_ENABLED) + (void)dim_mask; + (void)tolerance; + (void)max_shift_cells; + return; +#else + raise::ErrorIf(l_subdomain_indices().size() != 1, + "Rebalance assumes one local subdomain per rank", + HERE); + // The theta dimension (idx 1) is bounded by the polar axis for any + // non-Cartesian metric: moving an interior boundary in theta is fine + // in principle, but the safety net here forbids it pending validation. + if constexpr (M::CoordType != ntt::Coord::Cartesian) { + raise::ErrorIf((dim_mask & (1u << 1)) != 0u, + "Rebalance along the polar axis is not supported", + HERE); + } + // strip-width is constrained by N_GHOSTS so that any new active cells + // are already present in the rank's old ghost zone (after the most + // recent ghost-cell exchange). + if (max_shift_cells > N_GHOSTS) { + max_shift_cells = static_cast(N_GHOSTS); + } + if (max_shift_cells == 0 or dim_mask == 0) { + return; + } + + const auto local_idx = l_subdomain_indices()[0]; + + /* --- 1. Allgather active particle counts per domain ------------------ */ + npart_t local_npart { 0 }; + for (const auto& sp : g_subdomains[local_idx].species) { + local_npart += sp.npart(); + } + std::vector npart_per_dom(g_ndomains, 0); + MPI_Allgather(&local_npart, + 1, + mpi::get_type(), + npart_per_dom.data(), + 1, + mpi::get_type(), + MPI_COMM_WORLD); + + /* --- 2. Project ncells / load onto each balanced dim ----------------- */ + std::vector> ncells_per_pos(M::Dim); + std::vector> load_per_pos(M::Dim); + for (auto d { 0u }; d < M::Dim; ++d) { + ncells_per_pos[d].assign(g_ndomains_per_dim[d], 0); + load_per_pos[d].assign(g_ndomains_per_dim[d], 0.0); + } + for (unsigned int idx { 0 }; idx < g_ndomains; ++idx) { + const auto& off = g_domain_offsets[idx]; + const auto ncells = g_subdomains[idx].mesh.n_active(); + for (auto d { 0u }; d < M::Dim; ++d) { + ncells_per_pos[d][off[d]] = ncells[d]; + load_per_pos[d][off[d]] += static_cast(npart_per_dom[idx]); + } + } + + /* --- 3. Diffusion-style boundary shifts per balanced dim ------------- */ + std::vector> new_ncells_per_pos = ncells_per_pos; + const auto MIN_NCELLS = static_cast(2 * N_GHOSTS + 4); + bool any_shift = false; + + for (auto d { 0u }; d < M::Dim; ++d) { + if ((dim_mask & (1u << d)) == 0u) { + continue; + } + const auto N = g_ndomains_per_dim[d]; + if (N < 2) { + continue; + } + + double total_load = 0.0; + double max_load = 0.0; + double min_load = std::numeric_limits::infinity(); + for (auto p { 0u }; p < N; ++p) { + total_load += load_per_pos[d][p]; + max_load = std::max(max_load, load_per_pos[d][p]); + min_load = std::min(min_load, load_per_pos[d][p]); + } + if (total_load <= 0.0) { + continue; + } + const auto mean = total_load / static_cast(N); + if (((max_load - min_load) / mean) < + static_cast(tolerance)) { + continue; + } + + // bnd_shift[k] is the number of cells transferred from position k-1 to + // position k by moving the (interior) boundary k. bnd_shift[0] and + // bnd_shift[N] are exterior boundaries that are pinned at 0. + std::vector bnd_shift(N + 1, 0); + const int cap = static_cast(max_shift_cells); + for (auto k { 1u }; k < N; ++k) { + const auto l_load = load_per_pos[d][k - 1]; + const auto r_load = load_per_pos[d][k]; + const auto l_density = (ncells_per_pos[d][k - 1] > 0) + ? l_load / static_cast( + ncells_per_pos[d][k - 1]) + : 0.0; + const auto r_density = (ncells_per_pos[d][k] > 0) + ? r_load / static_cast( + ncells_per_pos[d][k]) + : 0.0; + const auto avg_density = std::max(0.5 * (l_density + r_density), 1.0); + // Move boundary towards the lighter side. Halve the gradient so that + // a single sweep does roughly one diffusion step. + int shift = static_cast( + std::round(0.5 * (l_load - r_load) / avg_density)); + shift = std::clamp(shift, -cap, cap); + // Don't shrink either side below MIN_NCELLS. + const int max_pos = static_cast(ncells_per_pos[d][k - 1]) - + static_cast(MIN_NCELLS); + const int max_neg = static_cast(ncells_per_pos[d][k]) - + static_cast(MIN_NCELLS); + shift = std::clamp(shift, + -std::max(max_neg, 0), + std::max(max_pos, 0)); + bnd_shift[k] = shift; + if (shift != 0) { + any_shift = true; + } + } + // new_ncells[p] = ncells[p] + bnd_shift[p] - bnd_shift[p+1] + for (auto p { 0u }; p < N; ++p) { + new_ncells_per_pos[d][p] = static_cast( + static_cast(ncells_per_pos[d][p]) + bnd_shift[p] - + bnd_shift[p + 1]); + } + } + + if (not any_shift) { + return; + } + + /* --- 4. Per-position prefix sums (offset and physical extent) -------- */ + // Face positions are queried from g_mesh.metric in the global code-coordinate + // system, so curvilinear stretches (log-r, eta-stretching) are honored. + std::vector> new_offset_per_pos(M::Dim); + std::vector>> extent_per_pos(M::Dim); + auto face_phys = [this](unsigned int d, real_t x_code) -> real_t { + if (d == 0u) { + return g_mesh.metric.template convert<1, Crd::Cd, Crd::Ph>(x_code); + } + if constexpr (M::Dim == Dim::_2D or M::Dim == Dim::_3D) { + if (d == 1u) { + return g_mesh.metric.template convert<2, Crd::Cd, Crd::Ph>(x_code); + } + } + if constexpr (M::Dim == Dim::_3D) { + if (d == 2u) { + return g_mesh.metric.template convert<3, Crd::Cd, Crd::Ph>(x_code); + } + } + raise::Error("Invalid dimension index in Rebalance face_phys", HERE); + return ZERO; + }; + for (auto d { 0u }; d < M::Dim; ++d) { + const auto N = g_ndomains_per_dim[d]; + new_offset_per_pos[d].assign(N, 0); + extent_per_pos[d].resize(N); + ncells_t running { 0 }; + for (auto p { 0u }; p < N; ++p) { + new_offset_per_pos[d][p] = running; + const auto x_lo = face_phys(d, static_cast(running)); + running += new_ncells_per_pos[d][p]; + const auto x_hi = face_phys(d, static_cast(running)); + extent_per_pos[d][p] = { x_lo, x_hi }; + } + } + + /* --- 5. Save local em (and em0/cur0 for GRPIC) to host --------------- */ + auto& local_dom = g_subdomains[local_idx]; + const auto old_offset_ncells = local_dom.offset_ncells(); + + auto em_old_h = Kokkos::create_mirror_view(local_dom.fields.em); + Kokkos::deep_copy(em_old_h, local_dom.fields.em); + auto em0_old_h = decltype(Kokkos::create_mirror_view(local_dom.fields.em0)) {}; + auto cur0_old_h = decltype(Kokkos::create_mirror_view(local_dom.fields.cur0)) {}; + if constexpr (S == SimEngine::GRPIC) { + em0_old_h = Kokkos::create_mirror_view(local_dom.fields.em0); + cur0_old_h = Kokkos::create_mirror_view(local_dom.fields.cur0); + Kokkos::deep_copy(em0_old_h, local_dom.fields.em0); + Kokkos::deep_copy(cur0_old_h, local_dom.fields.cur0); + } + + /* --- 6. Update bookkeeping for every g_subdomain --------------------- */ + std::vector new_local_ncells(M::Dim); + std::vector new_local_offset(M::Dim); + for (unsigned int idx { 0 }; idx < g_ndomains; ++idx) { + auto& sub = g_subdomains[idx]; + const auto& off_ndoms = g_domain_offsets[idx]; + std::vector ncells_d(M::Dim); + std::vector offset_d(M::Dim); + boundaries_t ext_d; + for (auto d { 0u }; d < M::Dim; ++d) { + ncells_d[d] = new_ncells_per_pos[d][off_ndoms[d]]; + offset_d[d] = new_offset_per_pos[d][off_ndoms[d]]; + ext_d.push_back(extent_per_pos[d][off_ndoms[d]]); + } + sub.mesh.set_resolution_and_extent(ncells_d, ext_d); + sub.set_offset_ncells(offset_d); + if (idx == local_idx) { + new_local_ncells = ncells_d; + new_local_offset = offset_d; + } + } + // Boundary conditions and neighbor topology are unchanged. + + /* --- 7. Reallocate local fields, copy from saved buffer with shift --- */ + // delta[d] = new_offset[d] - old_offset[d] (in the with-ghost field + // coordinate system the same delta applies). + std::vector delta(M::Dim); + for (auto d { 0u }; d < M::Dim; ++d) { + delta[d] = static_cast(new_local_offset[d]) - + static_cast(old_offset_ncells[d]); + } + + local_dom.fields = Fields { new_local_ncells }; + CopyShifted(em_old_h, local_dom.fields.em, delta); + if constexpr (S == SimEngine::GRPIC) { + CopyShifted(em0_old_h, local_dom.fields.em0, delta); + CopyShifted(cur0_old_h, local_dom.fields.cur0, delta); + } + + /* --- 8. Refill ghost zones from neighbors ---------------------------- */ + CommunicateFields(local_dom, Comm::E | Comm::B); + + /* --- 9. Shift particle indices, retag, and migrate ------------------- */ + // Particle indices i_d are in active-cell coordinates: i_d in [0, n_active) + // for an in-domain particle. After the offset moves by delta, particles + // get i_d_new = i_d_old - delta[d]. Particles whose new index falls + // outside [0, n_active) are tagged for the appropriate neighbor and + // sent by CommunicateParticles(). + for (auto& sp : local_dom.species) { + if (sp.npart() == 0) { + continue; + } + auto i1 = sp.i1, i2 = sp.i2, i3 = sp.i3; + auto i1p = sp.i1_prev, i2p = sp.i2_prev, i3p = sp.i3_prev; + auto tag = sp.tag; + const int dx1 = -delta[0]; + int dx2 = 0; + int dx3 = 0; + int new_n1 = static_cast(new_local_ncells[0]); + int new_n2 = 1; + int new_n3 = 1; + if constexpr (M::Dim == Dim::_2D or M::Dim == Dim::_3D) { + dx2 = -delta[1]; + new_n2 = static_cast(new_local_ncells[1]); + } + if constexpr (M::Dim == Dim::_3D) { + dx3 = -delta[2]; + new_n3 = static_cast(new_local_ncells[2]); + } + Kokkos::parallel_for( + "RebalanceShiftPrtls", + sp.rangeActiveParticles(), + Lambda(prtlidx_t p) { + if (tag(p) != ParticleTag::alive) { + return; + } + if constexpr (M::Dim == Dim::_1D or M::Dim == Dim::_2D or + M::Dim == Dim::_3D) { + i1(p) += dx1; + i1p(p) += dx1; + } + if constexpr (M::Dim == Dim::_2D or M::Dim == Dim::_3D) { + i2(p) += dx2; + i2p(p) += dx2; + } + if constexpr (M::Dim == Dim::_3D) { + i3(p) += dx3; + i3p(p) += dx3; + } + if constexpr (M::Dim == Dim::_1D) { + tag(p) = mpi::SendTag(tag(p), i1(p) < 0, i1(p) >= new_n1); + } else if constexpr (M::Dim == Dim::_2D) { + tag(p) = mpi::SendTag(tag(p), + i1(p) < 0, + i1(p) >= new_n1, + i2(p) < 0, + i2(p) >= new_n2); + } else if constexpr (M::Dim == Dim::_3D) { + tag(p) = mpi::SendTag(tag(p), + i1(p) < 0, + i1(p) >= new_n1, + i2(p) < 0, + i2(p) >= new_n2, + i3(p) < 0, + i3(p) >= new_n3); + } + }); + sp.set_unsorted(); + } + + CommunicateParticles(local_dom); + logger::Checkpoint("Rebalance: domains shifted, fields and particles redistributed", + HERE); +#endif // MPI_ENABLED + } + + // NOLINTBEGIN(bugprone-macro-parentheses) +#define METADOMAIN_REBAL(S, M, D) \ + template void Metadomain>::Rebalance(unsigned int, real_t, ncells_t); + + NTT_FOREACH_SPECIALIZATION(METADOMAIN_REBAL) +#undef METADOMAIN_REBAL + // NOLINTEND(bugprone-macro-parentheses) + +} // namespace ntt diff --git a/src/framework/parameters/parameters.cpp b/src/framework/parameters/parameters.cpp index 7372da510..8b525d49a 100644 --- a/src/framework/parameters/parameters.cpp +++ b/src/framework/parameters/parameters.cpp @@ -239,6 +239,52 @@ namespace ntt { alg_params.read(get("scales.dx0"), alg_extra_flags, toml_data); alg_params.setParams(alg_extra_flags, this); + /* [simulation.domain.load_balance] ------------------------------------- */ + set("simulation.domain.load_balance.enable", + toml::find_or(toml_data, "simulation", "domain", "load_balance", "enable", false)); + set("simulation.domain.load_balance.interval", + toml::find_or(toml_data, + "simulation", + "domain", + "load_balance", + "interval", + 0u)); + set("simulation.domain.load_balance.tolerance", + toml::find_or(toml_data, + "simulation", + "domain", + "load_balance", + "tolerance", + static_cast(0.1))); + set("simulation.domain.load_balance.max_shift", + toml::find_or(toml_data, + "simulation", + "domain", + "load_balance", + "max_shift", + static_cast(N_GHOSTS))); + { + // dimensions: list of 1/2/3 mapped to a bitmask + const auto dim_ints = toml::find_or>( + toml_data, + "simulation", + "domain", + "load_balance", + "dimensions", + std::vector { 1 }); + unsigned int mask = 0u; + for (const auto& d : dim_ints) { + if (d == 1 or d == 2 or d == 3) { + mask |= 1u << (d - 1); + } else { + raise::Error( + "simulation.domain.load_balance.dimensions: unknown dim, expected 1/2/3", + HERE); + } + } + set("simulation.domain.load_balance.dim_mask", mask); + } + /* extra physics ------------------------------------------------------ */ params::Extra extra_params {}; const std::map extra_extra_flags = { diff --git a/src/output/utils/writers.cpp b/src/output/utils/writers.cpp index a02f298e4..d89cd8cab 100644 --- a/src/output/utils/writers.cpp +++ b/src/output/utils/writers.cpp @@ -76,13 +76,18 @@ namespace out { } template - void WriteNDField(adios2::IO& io, - adios2::Engine& writer, - const std::string& name, - const ndfield_t& data) { + void WriteNDField(adios2::IO& io, + adios2::Engine& writer, + const std::string& name, + const ndfield_t& data, + const adios2::Box& range) { + auto var = io.InquireVariable(name); + if (not range.first.empty()) { + var.SetSelection(range); + } auto data_h = Kokkos::create_mirror_view(data); Kokkos::deep_copy(data_h, data); - writer.Put(io.InquireVariable(name), data_h.data(), adios2::Mode::Sync); + writer.Put(var, data_h.data(), adios2::Mode::Sync); } // NOLINTBEGIN(bugprone-macro-parentheses) @@ -122,7 +127,8 @@ namespace out { template void WriteNDField(adios2::IO&, \ adios2::Engine&, \ const std::string&, \ - const ndfield_t&); + const ndfield_t&, \ + const adios2::Box&); NDFIELD_WRITERS(Dim::_1D, 3) NDFIELD_WRITERS(Dim::_1D, 6) NDFIELD_WRITERS(Dim::_2D, 3) diff --git a/src/output/utils/writers.h b/src/output/utils/writers.h index cbd089452..4b8909a19 100644 --- a/src/output/utils/writers.h +++ b/src/output/utils/writers.h @@ -78,7 +78,8 @@ namespace out { void WriteNDField(adios2::IO&, adios2::Engine&, const std::string&, - const ndfield_t&); + const ndfield_t&, + const adios2::Box& = {}); } // namespace out diff --git a/src/output/writer.cpp b/src/output/writer.cpp index a84745c5b..5b9ae89e2 100644 --- a/src/output/writer.cpp +++ b/src/output/writer.cpp @@ -73,6 +73,33 @@ namespace out { m_mode = mode; } + void Writer::setLocalLayout(const std::vector& loc_corner, + const std::vector& loc_shape) { + raise::ErrorIf(loc_corner.size() != m_flds_l_corner.size() or + loc_shape.size() != m_flds_l_shape.size(), + "setLocalLayout dim mismatch with the original layout", + HERE); + m_flds_l_corner = loc_corner; + m_flds_l_shape = loc_shape; + m_flds_l_corner_dwn.clear(); + m_flds_l_shape_dwn.clear(); + m_flds_l_first.clear(); + for (auto i { 0u }; i < m_flds_g_shape.size(); ++i) { + const double d = static_cast(m_dwn[i]); + const double l = static_cast(loc_corner[i]); + const double n = static_cast(loc_shape[i]); + const double f = math::ceil(l / d) * d - l; + m_flds_l_corner_dwn.push_back(static_cast(math::ceil(l / d))); + m_flds_l_first.push_back(static_cast(f)); + m_flds_l_shape_dwn.push_back(static_cast(math::ceil((n - f) / d))); + } + if constexpr (not std::is_same::array_layout, + Kokkos::LayoutRight>::value) { + std::reverse(m_flds_l_corner_dwn.begin(), m_flds_l_corner_dwn.end()); + std::reverse(m_flds_l_shape_dwn.begin(), m_flds_l_shape_dwn.end()); + } + } + void Writer::defineMeshLayout( const std::vector& glob_shape, const std::vector& loc_corner, @@ -111,25 +138,24 @@ namespace out { for (auto i { 0u }; i < m_flds_g_shape.size(); ++i) { // cell-centers + // ConstantDims is intentionally NOT set: per-rank slab shape can change + // when dynamic load balancing shifts domain boundaries between writes. m_io.DefineVariable("X" + std::to_string(i + 1), { m_flds_g_shape_dwn[i] }, { m_flds_l_corner_dwn[i] }, - { m_flds_l_shape_dwn[i] }, - adios2::ConstantDims); + { m_flds_l_shape_dwn[i] }); // cell-edges const auto is_last = (m_flds_l_corner[i] + m_flds_l_shape[i] == m_flds_g_shape[i]); m_io.DefineVariable("X" + std::to_string(i + 1) + "e", { m_flds_g_shape_dwn[i] + 1 }, { m_flds_l_corner_dwn[i] }, - { m_flds_l_shape_dwn[i] + (is_last ? 1 : 0) }, - adios2::ConstantDims); + { m_flds_l_shape_dwn[i] + (is_last ? 1 : 0) }); m_io.DefineVariable( "N" + std::to_string(i + 1) + "l", { static_cast(2 * domain_idx.second) }, { static_cast(2 * domain_idx.first) }, - { static_cast(2) }, - adios2::ConstantDims); + { static_cast(2) }); } if constexpr (std::is_same::array_layout, @@ -154,21 +180,21 @@ namespace out { m_flds_writers.emplace_back(S, fld); } for (const auto& fld : m_flds_writers) { + // ConstantDims is intentionally NOT set: per-rank slab shape can change + // when dynamic load balancing shifts domain boundaries between writes. if (fld.comp.empty()) { // scalar m_io.DefineVariable(fld.name(), m_flds_g_shape_dwn, m_flds_l_corner_dwn, - m_flds_l_shape_dwn, - adios2::ConstantDims); + m_flds_l_shape_dwn); } else { // vector or tensor for (auto i { 0u }; i < fld.comp.size(); ++i) { m_io.DefineVariable(fld.name(i), m_flds_g_shape_dwn, m_flds_l_corner_dwn, - m_flds_l_shape_dwn, - adios2::ConstantDims); + m_flds_l_shape_dwn); } } } @@ -198,9 +224,14 @@ namespace out { std::size_t comp, std::vector dwn, std::vector first_cell, - bool ghosts) { + bool ghosts, + const adios2::Dims& loc_corner_dwn, + const adios2::Dims& loc_shape_dwn) { // when dwn != 1 in any direction, it is assumed that ghosts == false - auto var = io.InquireVariable(varname); + auto var = io.InquireVariable(varname); + // Refresh the per-step (start, count) so the slab tracks any rebalance + // that happened since the variable was declared. + var.SetSelection(adios2::Box(loc_corner_dwn, loc_shape_dwn)); const auto gh_zones = ghosts ? 0 : N_GHOSTS; ndarray_t output_field {}; @@ -332,7 +363,9 @@ namespace out { addresses[i], m_dwn, m_flds_l_first, - m_flds_ghosts); + m_flds_ghosts, + m_flds_l_corner_dwn, + m_flds_l_shape_dwn); } } @@ -410,8 +443,28 @@ namespace out { const array_t& xc, const array_t& xe, const std::vector& loc_off_sz) { + // Per-step (start, count) for the per-rank slab; tracks the (possibly + // rebalanced) layout cached by setLocalLayout(). auto varc = m_io.InquireVariable("X" + std::to_string(dim + 1)); auto vare = m_io.InquireVariable("X" + std::to_string(dim + 1) + "e"); + // m_flds_l_corner_dwn / m_flds_l_shape_dwn are reversed for non-LayoutRight + // (see defineMeshLayout / setLocalLayout); m_flds_l_corner / m_flds_l_shape + // / m_flds_g_shape are not. Map the dim-order index to the dwn-array index. + constexpr bool layout_right = std::is_same< + typename ndfield_t::array_layout, + Kokkos::LayoutRight>::value; + const auto i_dwn = layout_right + ? static_cast(dim) + : (m_flds_g_shape.size() - 1u - + static_cast(dim)); + const auto is_last = (m_flds_l_corner[dim] + m_flds_l_shape[dim] == + m_flds_g_shape[dim]); + varc.SetSelection(adios2::Box( + { m_flds_l_corner_dwn[i_dwn] }, + { m_flds_l_shape_dwn[i_dwn] })); + vare.SetSelection(adios2::Box( + { m_flds_l_corner_dwn[i_dwn] }, + { m_flds_l_shape_dwn[i_dwn] + (is_last ? 1ul : 0ul) })); auto xc_h = Kokkos::create_mirror_view(xc); auto xe_h = Kokkos::create_mirror_view(xe); Kokkos::deep_copy(xc_h, xc); @@ -506,7 +559,9 @@ namespace out { std::size_t, \ std::vector, \ std::vector, \ - bool); + bool, \ + const adios2::Dims&, \ + const adios2::Dims&); WRITE_FIELD(Dim::_1D, 3) WRITE_FIELD(Dim::_1D, 6) WRITE_FIELD(Dim::_2D, 3) diff --git a/src/output/writer.h b/src/output/writer.h index 055cb215e..7231c738b 100644 --- a/src/output/writer.h +++ b/src/output/writer.h @@ -109,6 +109,17 @@ namespace out { bool, Coord); + /** + * @brief Refresh the cached per-rank slab corner and shape. + * + * Must be called before any per-step write whenever the local domain has + * been resized (e.g. by `Metadomain::Rebalance`); the cached values are + * passed to `var.SetSelection()` inside `writeMesh` and `WriteField`. + * The global shape is preserved (load balancing conserves total cells). + */ + void setLocalLayout(const std::vector& loc_corner, + const std::vector& loc_shape); + void defineFieldOutputs(const SimEngine&, const std::vector&); void defineSpectraOutputs(const std::vector&); From ca3f162fa0bff91ffea83fcf55f6aaa50e987a50 Mon Sep 17 00:00:00 2001 From: Ludwig Boess Date: Mon, 29 Jun 2026 14:49:09 -0400 Subject: [PATCH 034/125] bugfix in for sort includes --- src/global/utils/sort_dispatch.h | 20 ++++++++++++++++++++ 1 file changed, 20 insertions(+) diff --git a/src/global/utils/sort_dispatch.h b/src/global/utils/sort_dispatch.h index 27a20a910..432b1c1b9 100644 --- a/src/global/utils/sort_dispatch.h +++ b/src/global/utils/sort_dispatch.h @@ -38,6 +38,21 @@ #include #include +// Entity's Kokkos alias macros (arch/kokkos_aliases.h) define bare words such +// as `Function`, `Inline`, `Lambda` and `ClassLambda`. These collide with +// template-parameter and member names used inside the vendor sort headers +// (rocPRIM, cub, oneDPL) and corrupt their parsing (e.g. rocPRIM's +// `template`). Suspend the aliases across the +// vendor includes only, then restore them for the rest of the translation unit. +#pragma push_macro("Function") +#pragma push_macro("Inline") +#pragma push_macro("Lambda") +#pragma push_macro("ClassLambda") +#undef Function +#undef Inline +#undef Lambda +#undef ClassLambda + #if defined(SYCL_ENABLED) && defined(ONEDPL_ENABLED) #include #include @@ -49,6 +64,11 @@ #include #endif +#pragma pop_macro("ClassLambda") +#pragma pop_macro("Lambda") +#pragma pop_macro("Inline") +#pragma pop_macro("Function") + #include #include #include From 3d22fd02f6748624c3362c69b4c46a85cc351ac1 Mon Sep 17 00:00:00 2001 From: Ludwig Boess Date: Mon, 29 Jun 2026 23:52:19 -0400 Subject: [PATCH 035/125] initial commit on internal renderer --- src/engines/engine.hpp | 9 +- src/framework/CMakeLists.txt | 1 + src/framework/domain/metadomain.h | 13 + src/framework/domain/metadomain_comm.cpp | 34 +++ src/framework/domain/metadomain_render.cpp | 315 +++++++++++++++++++ src/global/global.h | 1 + src/global/utils/diag.cpp | 4 + src/global/utils/diag.h | 2 + src/global/utils/timer.cpp | 6 +- src/output/CMakeLists.txt | 1 + src/output/render/composite.h | 77 +++++ src/output/render/png.h | 174 +++++++++++ src/output/render/raymarch.hpp | 298 ++++++++++++++++++ src/output/render/renderer.cpp | 334 +++++++++++++++++++++ src/output/render/renderer.h | 171 +++++++++++ src/output/render/transfer_fn.h | 192 ++++++++++++ 16 files changed, 1630 insertions(+), 2 deletions(-) create mode 100644 src/framework/domain/metadomain_render.cpp create mode 100644 src/output/render/composite.h create mode 100644 src/output/render/png.h create mode 100644 src/output/render/raymarch.hpp create mode 100644 src/output/render/renderer.cpp create mode 100644 src/output/render/renderer.h create mode 100644 src/output/render/transfer_fn.h diff --git a/src/engines/engine.hpp b/src/engines/engine.hpp index 161339a9d..95985717d 100644 --- a/src/engines/engine.hpp +++ b/src/engines/engine.hpp @@ -142,6 +142,7 @@ namespace ntt { #if defined(OUTPUT_ENABLED) m_metadomain.InitWriter(&m_adios, m_params); m_metadomain.InitCheckpointWriter(&m_adios, m_params); + m_metadomain.InitRenderer(m_params); #endif logger::Checkpoint("Initializing Engine", HERE); if (not is_resuming) { @@ -255,7 +256,7 @@ namespace ntt { "Injector", "Custom", "LoadBalance", "ParticleSort", "Output", - "Checkpoint" }, + "Render", "Checkpoint" }, []() { Kokkos::fence(); }, @@ -323,6 +324,7 @@ namespace ntt { ++step; auto print_output = false; + auto print_render = false; auto print_checkpoint = false; #if defined(OUTPUT_ENABLED) timers.start("Output"); @@ -368,6 +370,10 @@ namespace ntt { } timers.stop("Output"); + timers.start("Render"); + print_render = m_metadomain.Render(m_params, step, step - 1, time, time - dt); + timers.stop("Render"); + timers.start("Checkpoint"); print_checkpoint = m_metadomain.WriteCheckpoint(m_params, step, @@ -394,6 +400,7 @@ namespace ntt { m_metadomain.l_maxnpart_perspec(), print_prtl_clear, print_output, + print_render, print_checkpoint, m_params.get("diagnostics.colored_stdout")); } diff --git a/src/framework/CMakeLists.txt b/src/framework/CMakeLists.txt index 377c1f98a..a6c5317cf 100644 --- a/src/framework/CMakeLists.txt +++ b/src/framework/CMakeLists.txt @@ -66,6 +66,7 @@ set(SOURCES ${SRC_DIR}/containers/fields.cpp) if(${output}) list(APPEND SOURCES ${SRC_DIR}/domain/metadomain_io.cpp) + list(APPEND SOURCES ${SRC_DIR}/domain/metadomain_render.cpp) list(APPEND SOURCES ${SRC_DIR}/domain/metadomain_chckpt.cpp) list(APPEND SOURCES ${SRC_DIR}/containers/fields_io.cpp) list(APPEND SOURCES ${SRC_DIR}/containers/particles_io.cpp) diff --git a/src/framework/domain/metadomain.h b/src/framework/domain/metadomain.h index f090dade1..7e5d9f420 100644 --- a/src/framework/domain/metadomain.h +++ b/src/framework/domain/metadomain.h @@ -39,6 +39,7 @@ #if defined(OUTPUT_ENABLED) #include "output/checkpoint.h" + #include "output/render/renderer.h" #include "output/writer.h" #include @@ -96,6 +97,9 @@ namespace ntt { void SynchronizeFields(Domain&, CommTags, const cell_range_t& = { 0, 0 }) const; + // Halo-fill of the bckp buffer (neighbor active cells -> local ghosts), + // used by the in-situ renderer for seamless trilinear sampling. + void CommunicateBckp(Domain&, const cell_range_t&) const; #if defined(MPI_ENABLED) && defined(OUTPUT_ENABLED) void CommunicateVectorPotential(unsigned short); #endif @@ -173,6 +177,14 @@ namespace ntt { void ContinueFromCheckpoint(adios2::ADIOS*, const SimulationParams&); void redecomposeFromCheckpoint(const std::vector>&, const std::vector>&); + + /* in-situ volume renderer (see metadomain_render.cpp) ------------------ */ + void InitRenderer(const SimulationParams&); + auto Render(const SimulationParams&, + timestep_t, + timestep_t, + simtime_t, + simtime_t) -> bool; #endif void InitStatsWriter(const SimulationParams&, bool); @@ -305,6 +317,7 @@ namespace ntt { #if defined(OUTPUT_ENABLED) out::Writer g_writer; checkpoint::Writer g_checkpoint_writer; + out::Renderer g_renderer; #endif #if defined(MPI_ENABLED) diff --git a/src/framework/domain/metadomain_comm.cpp b/src/framework/domain/metadomain_comm.cpp index f8c0f60d7..dd18b8e25 100644 --- a/src/framework/domain/metadomain_comm.cpp +++ b/src/framework/domain/metadomain_comm.cpp @@ -652,6 +652,38 @@ namespace ntt { #endif } + template + void Metadomain::CommunicateBckp(Domain& domain, + const cell_range_t& components) const { + // Halo FILL of the bckp buffer: copy each neighbor's active boundary cells + // into this domain's ghost zones (additive=false). This is distinct from + // SynchronizeFields, which sums ghost-deposited values back into active + // cells. The renderer needs the ghost halo populated so trilinear sampling + // near a domain face reads valid neighbor values (C0 across the face). + for (auto& direction : dir::Directions::all) { + const auto [send_params, + recv_params] = GetSendRecvParams(this, domain, direction, false); + const auto [send_indrank, send_slice] = send_params; + const auto [recv_indrank, recv_slice] = recv_params; + const auto [send_ind, send_rank] = send_indrank; + const auto [recv_ind, recv_rank] = recv_indrank; + if (send_rank < 0 and recv_rank < 0) { + continue; + } + comm::CommunicateField(domain.index(), + domain.fields.bckp, + domain.fields.bckp, + send_ind, + recv_ind, + send_rank, + recv_rank, + send_slice, + recv_slice, + components, + false); + } + } + // NOLINTBEGIN(bugprone-macro-parentheses) #define METADOMAIN_COMM(S, M, D) \ template void Metadomain>::CommunicateFields(Domain>&, \ @@ -659,6 +691,8 @@ namespace ntt { template void Metadomain>::SynchronizeFields(Domain>&, \ CommTags, \ const cell_range_t&) const; \ + template void Metadomain>::CommunicateBckp(Domain>&, \ + const cell_range_t&) const; \ template void Metadomain>::CommunicateParticles(Domain>&) const; NTT_FOREACH_SPECIALIZATION(METADOMAIN_COMM) diff --git a/src/framework/domain/metadomain_render.cpp b/src/framework/domain/metadomain_render.cpp new file mode 100644 index 000000000..4e9cbf696 --- /dev/null +++ b/src/framework/domain/metadomain_render.cpp @@ -0,0 +1,315 @@ +/** + * @file framework/domain/metadomain_render.cpp + * @brief Metadomain driver for the in-situ volume renderer + * @implements + * - ntt::Metadomain::InitRenderer + * - ntt::Metadomain::Render + * @namespaces: + * - ntt:: + * @macros: + * - MPI_ENABLED + * - OUTPUT_ENABLED + * @note + * This is the templated counterpart of the (plain) out::Renderer: it owns the + * per-(engine, metric, dim) field preparation, the device ray-march kernel + * launch, and the device->host copy. It reuses the exact field-prep code paths + * as Metadomain::Write (ComputeMoments / FieldsToPhys), then hands the per-rank + * host image to out::Renderer for the MPI ordered composite and PNG write. + * Active only for 3D Minkowski; a no-op otherwise (the structured-order + * composite assumes an axis-aligned, affine code<->world map). + */ + +#include "enums.h" +#include "global.h" + +#include "arch/kokkos_aliases.h" +#include "traits/metric.h" +#include "utils/error.h" +#include "utils/log.h" + +#include "framework/containers/particles.h" +#include "framework/domain/domain.h" +#include "framework/domain/mesh.h" +#include "framework/domain/metadomain.h" +#include "framework/parameters/parameters.h" +#include "framework/specialization_registry.h" +#include "kernels/fields_to_phys.hpp" +#include "kernels/particle_moments.hpp" +#include "output/render/composite.h" +#include "output/render/raymarch.hpp" + +#include +#include + +#include +#include +#include + +namespace ntt { + + namespace { + + // Mirror of Metadomain::Write's ComputeMoments, kept with internal linkage + // so it does not collide with the (identically-named) one in metadomain_io. + template + void renderMoment(const SimulationParams& params, + const Mesh& mesh, + const std::vector>& prtl_species, + ndfield_t& buffer, + idx_t buff_idx) { + std::vector specs; + for (auto& sp : prtl_species) { + if (sp.mass() > 0) { + specs.push_back(sp.index()); + } + } + auto scatter_buff = Kokkos::Experimental::create_scatter_view(buffer); + const auto use_weights = params.get("particles.use_weights"); + const auto ni2 = mesh.n_active(in::x2); + const auto inv_n0 = ONE / params.get("scales.n0"); + const auto smooth_order = params.get( + "output.fields.smoothing.order"); + const auto smooth_method = OutputSmoothingType::from_string( + params.get("output.fields.smoothing.method")); + const std::vector components {}; + for (const auto& sp : specs) { + auto& prtl_spec = prtl_species[sp - 1]; + Kokkos::parallel_for( + "RenderComputeMoments", + prtl_spec.rangeActiveParticles(), + kernel::ParticleMoments_kernel(components, + scatter_buff, + buff_idx, + prtl_spec, + use_weights, + mesh.metric, + mesh.flds_bc(), + ni2, + inv_n0, + smooth_order, + smooth_method)); + } + Kokkos::Experimental::contribute(buffer, scatter_buff); + } + + } // namespace + + template + void Metadomain::InitRenderer(const SimulationParams& params) { + g_renderer.init(params, mesh().extent()); + } + + template + auto Metadomain::Render(const SimulationParams& params, + timestep_t current_step, + timestep_t finished_step, + simtime_t current_time, + simtime_t finished_time) -> bool { + if constexpr (M::Dim == Dim::_3D and M::CoordType == Coord::type::Cartesian) { + // structured-order composite assumes an axis-aligned, affine code<->world + // map; only Cartesian (Minkowski) 3D qualifies. + if (not g_renderer.enabled() or + not g_renderer.shouldRender(finished_step, finished_time)) { + return false; + } + raise::ErrorIf(l_subdomain_indices().size() != 1, + "Renderer supports one subdomain per rank only", + HERE); + auto local_domain = subdomain_ptr(l_subdomain_indices()[0]); + raise::ErrorIf(local_domain->is_placeholder(), + "local_domain is a placeholder", + HERE); + logger::Checkpoint("Rendering output", HERE); + + const auto& cam = g_renderer.camera(); + const int W = g_renderer.width(); + const int H = g_renderer.height(); + + // per-domain world AABB + const auto loc_ext = local_domain->mesh.extent(); + real_t lo[3] = { loc_ext[0].first, loc_ext[1].first, loc_ext[2].first }; + real_t hi[3] = { loc_ext[0].second, loc_ext[1].second, loc_ext[2].second }; + + // fixed global world step (identical on all ranks -> seamless) + const auto glob_ext = mesh().extent(); + real_t gdiag = ZERO; + for (auto d { 0 }; d < 3; ++d) { + const real_t s = glob_ext[d].second - glob_ext[d].first; + gdiag += s * s; + } + gdiag = math::sqrt(gdiag); + const real_t ds = (g_renderer.stepSize() > ZERO) + ? g_renderer.stepSize() + : gdiag / static_cast(g_renderer.samples()); + const int max_steps = 2 * g_renderer.samples() + 16; + + // composite order key (depends on the current decomposition offsets) + const uint64_t order_key = out::compositeOrderKey( + local_domain->offset_ndomains(), + ndomains_per_dim(), + cam.forward); + + auto& bckp = local_domain->fields.bckp; + const int ext0 = static_cast(bckp.extent(0)); + const int ext1 = static_cast(bckp.extent(1)); + const int ext2 = static_cast(bckp.extent(2)); + + const auto metric = local_domain->mesh.metric; + + bool rendered_any = false; + for (const auto& scene : g_renderer.scenes()) { + Kokkos::deep_copy(bckp, ZERO); + + if (scene.field == "N") { + renderMoment(params, + local_domain->mesh, + local_domain->species, + bckp, + 0u); + // sum boundary-crossing particle deposits back into active cells + // (particles in neighbor domains deposit into our ghost zone) + SynchronizeFields(*local_domain, Comm::Bckp, { 0, 1 }); + } else if (scene.field == "Bmag" or scene.field == "Jmag") { + const bool is_current = (scene.field == "Jmag"); + // raw vector components into bckp(:,0..2) + if (is_current) { + Kokkos::deep_copy( + Kokkos::subview(bckp, Kokkos::ALL, Kokkos::ALL, Kokkos::ALL, + cell_range_t(0, 3)), + Kokkos::subview(local_domain->fields.cur, Kokkos::ALL, Kokkos::ALL, + Kokkos::ALL, cell_range_t(cur::jx1, cur::jx3 + 1))); + } else { + Kokkos::deep_copy( + Kokkos::subview(bckp, Kokkos::ALL, Kokkos::ALL, Kokkos::ALL, + cell_range_t(0, 3)), + Kokkos::subview(local_domain->fields.em, Kokkos::ALL, Kokkos::ALL, + Kokkos::ALL, cell_range_t(em::bx1, em::bx3 + 1))); + } + // interpolate to cell centers + convert to physical basis -> (3,4,5) + PrepareOutputFlags interp = is_current + ? PrepareOutput::InterpToCellCenterFromEdges + : PrepareOutput::InterpToCellCenterFromFaces; + PrepareOutputFlags prepare = (S == SimEngine::SRPIC) + ? PrepareOutput::ConvertToHat + : PrepareOutput::ConvertToPhysCntrv; + list_t comp_from = { 0, 1, 2 }; + list_t comp_to = { 3, 4, 5 }; + Kokkos::parallel_for( + "RenderFieldsToPhys", + local_domain->mesh.rangeActiveCells(), + kernel::FieldsToPhys_kernel(bckp, + bckp, + comp_from, + comp_to, + interp | prepare, + metric)); + // magnitude -> bckp(:,0) + auto bckp_v = bckp; + Kokkos::parallel_for( + "RenderVectorMagnitude", + local_domain->mesh.rangeActiveCells(), + Lambda(cellidx_t i1, cellidx_t i2, cellidx_t i3) { + const real_t v1 = bckp_v(i1, i2, i3, 3); + const real_t v2 = bckp_v(i1, i2, i3, 4); + const real_t v3 = bckp_v(i1, i2, i3, 5); + bckp_v(i1, i2, i3, 0) = math::sqrt(v1 * v1 + v2 * v2 + v3 * v3); + }); + } else if (scene.field == "smooth_xyz") { + // continuous-by-construction regression field: x + y + z + auto bckp_v = bckp; + Kokkos::parallel_for( + "RenderSmoothXYZ", + local_domain->mesh.rangeActiveCells(), + Lambda(cellidx_t i1, cellidx_t i2, cellidx_t i3) { + coord_t x_Cd { ZERO }, x_Ph { ZERO }; + x_Cd[0] = COORD(i1) + HALF; + x_Cd[1] = COORD(i2) + HALF; + x_Cd[2] = COORD(i3) + HALF; + metric.template convert(x_Cd, x_Ph); + bckp_v(i1, i2, i3, 0) = x_Ph[0] + x_Ph[1] + x_Ph[2]; + }); + } else { + raise::Warning( + "output.render: unknown field '" + scene.field + "', skipping", + HERE); + continue; + } + + // fill the ghost halo with neighbor active values so trilinear + // sampling is C0 across domain faces (this is a halo EXCHANGE, not the + // sum-into-active that SynchronizeFields performs). + CommunicateBckp(*local_domain, { 0, 1 }); + + // ---- launch the ray-march kernel ------------------------------- // + array_t image { "render_img", + static_cast(W) * + static_cast(H) }; + randacc_ndfield_t Fld { bckp }; + Kokkos::parallel_for( + "VolumeRayMarch", + CreateRangePolicy({ 0, 0 }, + { static_cast(W), + static_cast(H) }), + kernel::VolumeRayMarch_kernel(Fld, + 0u, + metric, + cam, + lo, + hi, + ext0, + ext1, + ext2, + W, + H, + ds, + max_steps, + scene.tf.lut, + scene.tf.n_lut, + scene.tf.vmin, + scene.tf.vmax, + scene.tf.log_scale, + g_renderer.earlyAlpha(), + image)); + Kokkos::fence(); + + // device -> host, into a layout-agnostic pixel-major buffer + auto image_h = Kokkos::create_mirror_view(image); + Kokkos::deep_copy(image_h, image); + const std::size_t npix = static_cast(W) * + static_cast(H); + std::vector rgba(npix * 4); + for (std::size_t p = 0; p < npix; ++p) { + rgba[p * 4 + 0] = image_h(p, 0); + rgba[p * 4 + 1] = image_h(p, 1); + rgba[p * 4 + 2] = image_h(p, 2); + rgba[p * 4 + 3] = image_h(p, 3); + } + g_renderer.compositeAndWrite(rgba, order_key, scene.prefix, current_step); + rendered_any = true; + } + return rendered_any; + } else { + (void)params; + (void)current_step; + (void)finished_step; + (void)current_time; + (void)finished_time; + return false; + } + } + + // NOLINTBEGIN(bugprone-macro-parentheses) +#define METADOMAIN_RENDER(S, M, D) \ + template void Metadomain>::InitRenderer(const SimulationParams&); \ + template auto Metadomain>::Render(const SimulationParams&, \ + timestep_t, \ + timestep_t, \ + simtime_t, \ + simtime_t) -> bool; + + NTT_FOREACH_SPECIALIZATION(METADOMAIN_RENDER) + +#undef METADOMAIN_RENDER + // NOLINTEND(bugprone-macro-parentheses) + +} // namespace ntt diff --git a/src/global/global.h b/src/global/global.h index c3fbc5061..dafd8663c 100644 --- a/src/global/global.h +++ b/src/global/global.h @@ -254,6 +254,7 @@ namespace Timer { PrintParticleSort = 1 << 4, PrintCheckpoint = 1 << 5, PrintNormed = 1 << 6, + PrintRender = 1 << 7, Default = PrintNormed | PrintTotal | PrintTitle | AutoConvert, }; } // namespace Timer diff --git a/src/global/utils/diag.cpp b/src/global/utils/diag.cpp index 604a872ae..f22bc8137 100644 --- a/src/global/utils/diag.cpp +++ b/src/global/utils/diag.cpp @@ -90,6 +90,7 @@ namespace diag { const std::vector& species_maxnpart, bool print_prtl_clear, bool print_output, + bool print_render, bool print_checkpoint, bool print_colors) { DiagFlags diag_flags = Diag::Default; @@ -106,6 +107,9 @@ namespace diag { if (print_output) { timer_flags |= Timer::PrintOutput; } + if (print_render) { + timer_flags |= Timer::PrintRender; + } if (print_checkpoint) { timer_flags |= Timer::PrintCheckpoint; } diff --git a/src/global/utils/diag.h b/src/global/utils/diag.h index 18669c1f1..61186aa1c 100644 --- a/src/global/utils/diag.h +++ b/src/global/utils/diag.h @@ -36,6 +36,7 @@ namespace diag { * @param maxnpart (per each species) * @param particlesort (if true, dead particles were removed) * @param output (if true, output was written) + * @param render (if true, a volume render was produced) * @param checkpoint (if true, checkpoint was written) * @param colorful_print (if true, print with colors) */ @@ -52,6 +53,7 @@ namespace diag { bool, bool, bool, + bool, bool); } // namespace diag diff --git a/src/global/utils/timer.cpp b/src/global/utils/timer.cpp index 6f2045be6..85647d899 100644 --- a/src/global/utils/timer.cpp +++ b/src/global/utils/timer.cpp @@ -136,7 +136,10 @@ namespace timer { auto Timers::printAll(TimerFlags flags, npart_t npart, ncells_t ncells) const -> std::string { - const std::vector extras { "ParticleSort", "Output", "Checkpoint" }; + const std::vector extras { "ParticleSort", + "Output", + "Render", + "Checkpoint" }; const auto stats = gather(extras, npart, ncells); if (stats.empty()) { return ""; @@ -262,6 +265,7 @@ namespace timer { // print extra timers for output/checkpoint/particleSort const std::vector extras_f { Timer::PrintParticleSort, Timer::PrintOutput, + Timer::PrintRender, Timer::PrintCheckpoint }; for (auto i { 0u }; i < extras.size(); ++i) { const auto& name = extras[i]; diff --git a/src/output/CMakeLists.txt b/src/output/CMakeLists.txt index 4cf1bf410..f0a490d5e 100644 --- a/src/output/CMakeLists.txt +++ b/src/output/CMakeLists.txt @@ -31,6 +31,7 @@ set(SOURCES ${SRC_DIR}/stats.cpp ${SRC_DIR}/fields.cpp if(${output}) list(APPEND SOURCES ${SRC_DIR}/writer.cpp) list(APPEND SOURCES ${SRC_DIR}/checkpoint.cpp) + list(APPEND SOURCES ${SRC_DIR}/render/renderer.cpp) list(APPEND SOURCES ${SRC_DIR}/utils/writers.cpp) list(APPEND SOURCES ${SRC_DIR}/utils/readers.cpp) list(APPEND SOURCES ${SRC_DIR}/utils/tuning.cpp) diff --git a/src/output/render/composite.h b/src/output/render/composite.h new file mode 100644 index 000000000..39ea31aad --- /dev/null +++ b/src/output/render/composite.h @@ -0,0 +1,77 @@ +/** + * @file output/render/composite.h + * @brief Front-to-back visibility ordering for the structured decomposition + * and the premultiplied "over" compositing operator. + * @implements + * - out::compositeOrderKey + * - out::overComposite + * @namespaces: + * - out:: + * @note + * entity decomposes the global box into a regular Dx x Dy x Dz grid of domains + * (domain index == MPI rank). For a camera viewing the box from outside, the + * correct global front-to-back order is a deterministic per-axis ordering by + * which side of each split plane the camera sits on -- no general depth sort, + * no cyclic overlap. Ordered premultiplied "over" of the non-overlapping, + * correctly-ordered per-domain segments reconstructs the single-image ray + * integral, hence is seamless. + */ + +#ifndef OUTPUT_RENDER_COMPOSITE_H +#define OUTPUT_RENDER_COMPOSITE_H + +#include "global.h" + +#include "utils/numeric.h" + +#include +#include + +namespace out { + + /** + * @brief Total-order sort key placing nearer domains first (front-to-back). + * @param offset integer grid coordinate of the domain (offset_ndomains) + * @param ndoms number of domains per axis (ndomains_per_dim) + * @param forward camera view direction (world == code axes for Minkowski) + * @return a single key; ascending key == front-to-back. Smaller is nearer. + * + * For axis d: if the camera looks toward +d (forward[d] >= 0), the smaller + * grid index is nearer, so key_d = offset_d. Otherwise key_d is reversed. + * The per-axis keys are packed lexicographically (axis 0 most significant). + */ + inline auto compositeOrderKey(const std::vector& offset, + const std::vector& ndoms, + const real_t forward[3]) + -> uint64_t { + uint64_t key = 0; + for (std::size_t d = 0; d < ndoms.size(); ++d) { + const unsigned int Dd = ndoms[d]; + const unsigned int od = offset[d]; + const unsigned int kd = (forward[d] >= ZERO) ? od : (Dd - 1u - od); + key = key * static_cast(Dd) + + static_cast(kd); + } + return key; + } + + /** + * @brief Accumulate one segment into a front-to-back running composite. + * @param acc 4-element premultiplied RGBA accumulator (modified in place) + * @param seg 4-element premultiplied RGBA of the next (further) segment + * + * acc holds everything in front of seg. The "over" operator: + * C_acc += (1 - A_acc) * C_seg ; A_acc += (1 - A_acc) * A_seg + * Associative with identity (0,0,0,0); segments must be supplied front first. + */ + inline void overComposite(real_t acc[4], const real_t seg[4]) { + const real_t one_minus_a = ONE - acc[3]; + acc[0] += one_minus_a * seg[0]; + acc[1] += one_minus_a * seg[1]; + acc[2] += one_minus_a * seg[2]; + acc[3] += one_minus_a * seg[3]; + } + +} // namespace out + +#endif // OUTPUT_RENDER_COMPOSITE_H diff --git a/src/output/render/png.h b/src/output/render/png.h new file mode 100644 index 000000000..3f22cd5f5 --- /dev/null +++ b/src/output/render/png.h @@ -0,0 +1,174 @@ +/** + * @file output/render/png.h + * @brief Self-contained, dependency-free PNG (8-bit RGBA) encoder + * @implements + * - out::write_png + * @namespaces: + * - out:: + * @note + * Header-only. Emits a valid PNG using stored (uncompressed) DEFLATE blocks + * wrapped in a zlib stream, with per-scanline filter type 0 (None). This keeps + * the encoder tiny and provably correct at the cost of compression ratio; the + * resulting files are still orders of magnitude smaller than the full-field + * dumps the renderer is meant to replace. A drop-in stronger encoder (e.g. a + * fixed-Huffman DEFLATE, or vendored stb_image_write) can replace the IDAT + * producer without touching callers. + */ + +#ifndef OUTPUT_RENDER_PNG_H +#define OUTPUT_RENDER_PNG_H + +#include "global.h" + +#include +#include +#include +#include + +namespace out { + + namespace png_hidden { + + inline auto crc32(const uint8_t* data, std::size_t len) -> uint32_t { + static uint32_t table[256]; + static bool ready = false; + if (not ready) { + for (uint32_t n = 0; n < 256; ++n) { + uint32_t c = n; + for (int k = 0; k < 8; ++k) { + c = (c & 1u) ? (0xEDB88320u ^ (c >> 1)) : (c >> 1); + } + table[n] = c; + } + ready = true; + } + uint32_t c = 0xFFFFFFFFu; + for (std::size_t i = 0; i < len; ++i) { + c = table[(c ^ data[i]) & 0xFFu] ^ (c >> 8); + } + return c ^ 0xFFFFFFFFu; + } + + inline auto adler32(const uint8_t* data, std::size_t len) -> uint32_t { + constexpr uint32_t MOD = 65521u; + uint32_t a = 1u, b = 0u; + for (std::size_t i = 0; i < len; ++i) { + a = (a + data[i]) % MOD; + b = (b + a) % MOD; + } + return (b << 16) | a; + } + + inline void put_u32_be(std::vector& v, uint32_t x) { + v.push_back(static_cast((x >> 24) & 0xFFu)); + v.push_back(static_cast((x >> 16) & 0xFFu)); + v.push_back(static_cast((x >> 8) & 0xFFu)); + v.push_back(static_cast(x & 0xFFu)); + } + + inline void write_chunk(std::vector& out, + const char type[4], + const std::vector& data) { + put_u32_be(out, static_cast(data.size())); + std::vector typed_data; + typed_data.reserve(4 + data.size()); + for (int i = 0; i < 4; ++i) { + typed_data.push_back(static_cast(type[i])); + } + typed_data.insert(typed_data.end(), data.begin(), data.end()); + out.insert(out.end(), typed_data.begin(), typed_data.end()); + put_u32_be(out, crc32(typed_data.data(), typed_data.size())); + } + + // zlib stream wrapping `raw` in stored (BTYPE=00) DEFLATE blocks + inline auto zlib_store(const std::vector& raw) -> std::vector { + std::vector z; + z.push_back(0x78); // CMF: CM=8, CINFO=7 + z.push_back(0x01); // FLG: makes (CMF<<8 | FLG) % 31 == 0, no dict, level 0 + std::size_t off = 0; + const std::size_t n = raw.size(); + constexpr std::size_t BLOCK = 65535u; + if (n == 0) { + z.push_back(0x01); // final, stored + z.push_back(0x00); + z.push_back(0x00); + z.push_back(0xFF); + z.push_back(0xFF); + } + while (off < n) { + const std::size_t len = (n - off > BLOCK) ? BLOCK : (n - off); + const bool final = (off + len >= n); + z.push_back(final ? 0x01 : 0x00); + const uint16_t l = static_cast(len); + const uint16_t nl = static_cast(~l); + z.push_back(static_cast(l & 0xFFu)); + z.push_back(static_cast((l >> 8) & 0xFFu)); + z.push_back(static_cast(nl & 0xFFu)); + z.push_back(static_cast((nl >> 8) & 0xFFu)); + z.insert(z.end(), raw.begin() + off, raw.begin() + off + len); + off += len; + } + put_u32_be(z, adler32(raw.data(), raw.size())); + return z; + } + + } // namespace png_hidden + + /** + * @brief Write an 8-bit RGBA buffer to a PNG file. + * @param path output file path + * @param width image width in pixels + * @param height image height in pixels + * @param rgba pointer to width*height*4 bytes, row-major, top-left origin + * @return true on success + */ + inline auto write_png(const path_t& path, + int width, + int height, + const uint8_t* rgba) -> bool { + using namespace png_hidden; + const std::size_t w = static_cast(width); + const std::size_t h = static_cast(height); + // build filtered raw scanlines: each row prefixed with filter byte 0 (None) + std::vector raw; + raw.reserve(h * (1 + w * 4)); + for (std::size_t y = 0; y < h; ++y) { + raw.push_back(0x00); + const uint8_t* row = rgba + y * w * 4; + raw.insert(raw.end(), row, row + w * 4); + } + + std::vector file; + // PNG signature + const uint8_t sig[8] = { 137, 80, 78, 71, 13, 10, 26, 10 }; + file.insert(file.end(), sig, sig + 8); + + // IHDR + std::vector ihdr; + put_u32_be(ihdr, static_cast(width)); + put_u32_be(ihdr, static_cast(height)); + ihdr.push_back(8); // bit depth + ihdr.push_back(6); // color type: RGBA + ihdr.push_back(0); // compression + ihdr.push_back(0); // filter + ihdr.push_back(0); // interlace + write_chunk(file, "IHDR", ihdr); + + // IDAT + write_chunk(file, "IDAT", zlib_store(raw)); + + // IEND + write_chunk(file, "IEND", {}); + + std::ofstream f(path, std::ios::binary); + if (not f.good()) { + return false; + } + f.write(reinterpret_cast(file.data()), + static_cast(file.size())); + return f.good(); + } + +} // namespace out + +#endif // OUTPUT_RENDER_PNG_H diff --git a/src/output/render/raymarch.hpp b/src/output/render/raymarch.hpp new file mode 100644 index 000000000..c2403bc69 --- /dev/null +++ b/src/output/render/raymarch.hpp @@ -0,0 +1,298 @@ +/** + * @file output/render/raymarch.hpp + * @brief Header-only Kokkos volume ray-march kernel (one parallel_for over pixels) + * @implements + * - kernel::VolumeRayMarch_kernel + * @namespaces: + * - kernel:: + * @macros: + * - OUTPUT_ENABLED + * @note + * Pure Kokkos: the only device entities are Views, the (trivially-copyable) + * metric, and the POD camera. Runs in Kokkos::DefaultExecutionSpace, inheriting + * whatever backend entity was built with (HIP / CUDA / SYCL / OpenMP). + * + * Seamlessness: every rank marches at GLOBAL sample positions t_k = k*ds + * measured from the shared camera eye, with a FIXED world-space step `ds` + * identical on all ranks. Each global sample therefore lands in exactly one + * domain (half-open membership via the per-domain slab interval [t_enter, + * t_exit)), so the ordered cross-domain "over" composite reproduces the single + * full-ray integral exactly. Trilinear sampling reads into the 1-cell ghost + * halo entity already exchanges, so the per-rank field is C0-continuous up to + * the shared face. + */ + +#ifndef OUTPUT_RENDER_RAYMARCH_HPP +#define OUTPUT_RENDER_RAYMARCH_HPP + +#include "enums.h" +#include "global.h" + +#include "arch/kokkos_aliases.h" +#include "traits/metric.h" + +#include "output/render/renderer.h" + +namespace kernel { + using namespace ntt; + + template + class VolumeRayMarch_kernel { + static constexpr auto D = M::Dim; + + randacc_ndfield_t Fld; + const idx_t comp; + const M metric; + const out::CameraDevice cam; + + // local-domain world AABB and View extents (for index clamping) + const real_t lo0, lo1, lo2, hi0, hi1, hi2; + const int ext0, ext1, ext2; + + const int W, H; + const real_t ds; // fixed world step (global, identical on all ranks) + const int max_steps; // safety cap on the marching loop + + // transfer function + array_t lut; + const int n_lut; + const real_t vmin, vmax; + const bool log_scale; + const real_t early_alpha; + + array_t image; // output, (W*H, 4) premultiplied RGBA + + public: + VolumeRayMarch_kernel(const randacc_ndfield_t& Fld_, + idx_t comp_, + const M& metric_, + const out::CameraDevice& cam_, + const real_t lo[3], + const real_t hi[3], + int ext0_, + int ext1_, + int ext2_, + int W_, + int H_, + real_t ds_, + int max_steps_, + const array_t& lut_, + int n_lut_, + real_t vmin_, + real_t vmax_, + bool log_scale_, + real_t early_alpha_, + const array_t& image_) + : Fld { Fld_ } + , comp { comp_ } + , metric { metric_ } + , cam { cam_ } + , lo0 { lo[0] } + , lo1 { lo[1] } + , lo2 { lo[2] } + , hi0 { hi[0] } + , hi1 { hi[1] } + , hi2 { hi[2] } + , ext0 { ext0_ } + , ext1 { ext1_ } + , ext2 { ext2_ } + , W { W_ } + , H { H_ } + , ds { ds_ } + , max_steps { max_steps_ } + , lut { lut_ } + , n_lut { n_lut_ } + , vmin { vmin_ } + , vmax { vmax_ } + , log_scale { log_scale_ } + , early_alpha { early_alpha_ } + , image { image_ } {} + + // trilinear sample of the prepared scalar at world point p, reading the + // ghost halo for corners just outside the active box. + Inline auto sample(real_t px, real_t py, real_t pz) const -> real_t { + // world -> code (cell-center continuous index = code index - 1/2) + const real_t g0 = metric.template convert<1, Crd::Ph, Crd::Cd>(px) - HALF; + const real_t g1 = metric.template convert<2, Crd::Ph, Crd::Cd>(py) - HALF; + const real_t g2 = metric.template convert<3, Crd::Ph, Crd::Cd>(pz) - HALF; + const real_t f0 = math::floor(g0); + const real_t f1 = math::floor(g1); + const real_t f2 = math::floor(g2); + const real_t t0 = g0 - f0; + const real_t t1 = g1 - f1; + const real_t t2 = g2 - f2; + // base View index of the lower corner (active cell i -> View i + N_GHOSTS) + int b0 = static_cast(f0) + static_cast(N_GHOSTS); + int b1 = static_cast(f1) + static_cast(N_GHOSTS); + int b2 = static_cast(f2) + static_cast(N_GHOSTS); + // clamp so both corners (b, b+1) stay in range [0, ext-1] + b0 = (b0 < 0) ? 0 : ((b0 > ext0 - 2) ? ext0 - 2 : b0); + b1 = (b1 < 0) ? 0 : ((b1 > ext1 - 2) ? ext1 - 2 : b1); + b2 = (b2 < 0) ? 0 : ((b2 > ext2 - 2) ? ext2 - 2 : b2); + const real_t c000 = Fld(b0, b1, b2, comp); + const real_t c100 = Fld(b0 + 1, b1, b2, comp); + const real_t c010 = Fld(b0, b1 + 1, b2, comp); + const real_t c110 = Fld(b0 + 1, b1 + 1, b2, comp); + const real_t c001 = Fld(b0, b1, b2 + 1, comp); + const real_t c101 = Fld(b0 + 1, b1, b2 + 1, comp); + const real_t c011 = Fld(b0, b1 + 1, b2 + 1, comp); + const real_t c111 = Fld(b0 + 1, b1 + 1, b2 + 1, comp); + const real_t c00 = c000 * (ONE - t0) + c100 * t0; + const real_t c10 = c010 * (ONE - t0) + c110 * t0; + const real_t c01 = c001 * (ONE - t0) + c101 * t0; + const real_t c11 = c011 * (ONE - t0) + c111 * t0; + const real_t c0 = c00 * (ONE - t1) + c10 * t1; + const real_t c1 = c01 * (ONE - t1) + c11 * t1; + return c0 * (ONE - t2) + c1 * t2; + } + + Inline void operator()(cellidx_t px, cellidx_t py) const { + const auto pix = static_cast(py) * static_cast(W) + + static_cast(px); + // default transparent + image(pix, 0) = ZERO; + image(pix, 1) = ZERO; + image(pix, 2) = ZERO; + image(pix, 3) = ZERO; + + // ---- ray generation ------------------------------------------------ // + const real_t fx = TWO * (static_cast(px) + HALF) / + static_cast(W) - ONE; + const real_t fy = ONE - TWO * (static_cast(py) + HALF) / + static_cast(H); + real_t ox, oy, oz, dx, dy, dz; + if (cam.orthographic) { + const real_t sx = fx * cam.half_w; + const real_t sy = fy * cam.half_h; + ox = cam.eye[0] + sx * cam.right[0] + sy * cam.up[0]; + oy = cam.eye[1] + sx * cam.right[1] + sy * cam.up[1]; + oz = cam.eye[2] + sx * cam.right[2] + sy * cam.up[2]; + dx = cam.forward[0]; + dy = cam.forward[1]; + dz = cam.forward[2]; + } else { + const real_t nx = fx * cam.aspect * cam.tan_half_fov; + const real_t ny = fy * cam.tan_half_fov; + dx = cam.forward[0] + nx * cam.right[0] + ny * cam.up[0]; + dy = cam.forward[1] + nx * cam.right[1] + ny * cam.up[1]; + dz = cam.forward[2] + nx * cam.right[2] + ny * cam.up[2]; + const real_t inv = ONE / math::sqrt(dx * dx + dy * dy + dz * dz); + dx *= inv; + dy *= inv; + dz *= inv; + ox = cam.eye[0]; + oy = cam.eye[1]; + oz = cam.eye[2]; + } + + // ---- ray-AABB slab test against [lo, hi] --------------------------- // + real_t t_enter = ZERO; + real_t t_exit = static_cast(1e30); + const real_t eps = static_cast(1e-12); + // axis 0 + if (math::abs(dx) < eps) { + if (ox < lo0 or ox > hi0) { + return; + } + } else { + real_t t1 = (lo0 - ox) / dx; + real_t t2 = (hi0 - ox) / dx; + if (t1 > t2) { + const real_t tmp = t1; + t1 = t2; + t2 = tmp; + } + t_enter = (t1 > t_enter) ? t1 : t_enter; + t_exit = (t2 < t_exit) ? t2 : t_exit; + } + // axis 1 + if (math::abs(dy) < eps) { + if (oy < lo1 or oy > hi1) { + return; + } + } else { + real_t t1 = (lo1 - oy) / dy; + real_t t2 = (hi1 - oy) / dy; + if (t1 > t2) { + const real_t tmp = t1; + t1 = t2; + t2 = tmp; + } + t_enter = (t1 > t_enter) ? t1 : t_enter; + t_exit = (t2 < t_exit) ? t2 : t_exit; + } + // axis 2 + if (math::abs(dz) < eps) { + if (oz < lo2 or oz > hi2) { + return; + } + } else { + real_t t1 = (lo2 - oz) / dz; + real_t t2 = (hi2 - oz) / dz; + if (t1 > t2) { + const real_t tmp = t1; + t1 = t2; + t2 = tmp; + } + t_enter = (t1 > t_enter) ? t1 : t_enter; + t_exit = (t2 < t_exit) ? t2 : t_exit; + } + if (t_enter >= t_exit) { + return; + } + + // ---- march at global sample positions t_k = k*ds ------------------- // + const real_t inv_range = (vmax > vmin) ? (ONE / (vmax - vmin)) : ZERO; + const real_t log_vmin = log_scale ? math::log10(vmin) : ZERO; + // first global sample index inside this segment: t_k >= t_enter + const real_t k0 = math::ceil(t_enter / ds); + real_t t = k0 * ds; + real_t acc_r = ZERO, acc_g = ZERO, acc_b = ZERO, acc_a = ZERO; + int steps = 0; + while (t < t_exit and steps < max_steps) { + const real_t s = sample(ox + t * dx, oy + t * dy, oz + t * dz); + // normalize through the transfer function range + real_t u; + if (log_scale) { + u = (s > ZERO) ? (math::log10(s) - log_vmin) * inv_range + : -ONE; + } else { + u = (s - vmin) * inv_range; + } + if (u < ZERO) { + u = ZERO; + } else if (u > ONE) { + u = ONE; + } + int idx = static_cast(u * static_cast(n_lut - 1) + HALF); + if (idx < 0) { + idx = 0; + } else if (idx > n_lut - 1) { + idx = n_lut - 1; + } + const real_t cr = lut(idx, 0); // premultiplied + const real_t cg = lut(idx, 1); + const real_t cb = lut(idx, 2); + const real_t ca = lut(idx, 3); + const real_t w = ONE - acc_a; + acc_r += w * cr; + acc_g += w * cg; + acc_b += w * cb; + acc_a += w * ca; + if (acc_a >= early_alpha) { + break; + } + t += ds; + ++steps; + } + + image(pix, 0) = acc_r; + image(pix, 1) = acc_g; + image(pix, 2) = acc_b; + image(pix, 3) = acc_a; + } + }; + +} // namespace kernel + +#endif // OUTPUT_RENDER_RAYMARCH_HPP diff --git a/src/output/render/renderer.cpp b/src/output/render/renderer.cpp new file mode 100644 index 000000000..019fc7eaa --- /dev/null +++ b/src/output/render/renderer.cpp @@ -0,0 +1,334 @@ +#include "output/render/renderer.h" + +#include "global.h" + +#include "arch/kokkos_aliases.h" +#include "utils/error.h" +#include "utils/formatting.h" +#include "utils/log.h" +#include "utils/numeric.h" + +#include "output/render/composite.h" +#include "output/render/png.h" +#include "output/render/transfer_fn.h" + +#include + +#if defined(MPI_ENABLED) + #include "arch/mpi_aliases.h" + + #include +#endif + +#include +#include +#include +#include +#include +#include +#include +#include + +namespace out { + + namespace { + + inline void cross3(const real_t a[3], const real_t b[3], real_t out[3]) { + out[0] = a[1] * b[2] - a[2] * b[1]; + out[1] = a[2] * b[0] - a[0] * b[2]; + out[2] = a[0] * b[1] - a[1] * b[0]; + } + + inline auto norm3(const real_t a[3]) -> real_t { + return std::sqrt(a[0] * a[0] + a[1] * a[1] + a[2] * a[2]); + } + + inline void normalize3(real_t a[3]) { + const real_t n = norm3(a); + if (n > static_cast(1e-30)) { + a[0] /= n; + a[1] /= n; + a[2] /= n; + } + } + + inline auto quantize(real_t v) -> uint8_t { + const real_t c = (v < ZERO) ? ZERO : ((v > ONE) ? ONE : v); + return static_cast(c * static_cast(255.0) + HALF); + } + + } // namespace + + void Renderer::init(const ntt::SimulationParams& params, + const boundaries_t& global_extent) { + m_enabled = false; + const auto& td = params.data(); + + const bool enable = toml::find_or(td, "output", "render", "enable", false); + if (not enable) { + return; + } + // the renderer is a 3D feature; silently no-op otherwise. + if (global_extent.size() != 3) { + raise::Warning("output.render enabled but simulation is not 3D; " + "the volume renderer will be inactive", + HERE); + return; + } + + m_root = path_t(params.get("simulation.name")); + + m_width = toml::find_or(td, "output", "render", "width", 1024); + m_height = toml::find_or(td, "output", "render", "height", 1024); + m_samples = toml::find_or(td, "output", "render", "samples", 400); + m_step_size = toml::find_or(td, "output", "render", "step_size", ZERO); + m_early_alpha = toml::find_or(td, + "output", + "render", + "early_term_alpha", + static_cast(0.99)); + m_n_lut = toml::find_or(td, "output", "render", "n_lut", 256); + + // cadence: mirror output.* (interval in steps; interval_time in sim time) + const auto interval = toml::find_or(td, + "output", + "render", + "interval", + 0u); + const auto interval_time = toml::find_or(td, + "output", + "render", + "interval_time", + -1.0); + m_tracker.init("render", interval, interval_time); + + /* ---- camera --------------------------------------------------------- */ + real_t center[3], size[3]; + real_t maxext = ZERO; + for (int d = 0; d < 3; ++d) { + center[d] = static_cast(0.5) * + (global_extent[d].first + global_extent[d].second); + size[d] = global_extent[d].second - global_extent[d].first; + maxext = (size[d] > maxext) ? size[d] : maxext; + } + const real_t diag = std::sqrt(size[0] * size[0] + size[1] * size[1] + + size[2] * size[2]); + + const bool ortho = toml::find_or(td, + "output", + "render", + "camera", + "orthographic", + true); + auto pos = toml::find_or>(td, + "output", + "render", + "camera", + "position", + std::vector {}); + auto look = toml::find_or>(td, + "output", + "render", + "camera", + "look_at", + std::vector {}); + auto up = toml::find_or>(td, + "output", + "render", + "camera", + "up", + std::vector {}); + const real_t fov = toml::find_or(td, + "output", + "render", + "camera", + "fov", + static_cast(35.0)); + // default covers the box from any view direction (default camera looks + // down the diagonal), so nothing is clipped without explicit framing. + (void)maxext; + const real_t ortho_height = toml::find_or(td, + "output", + "render", + "camera", + "ortho_height", + diag); + + real_t eye[3], lookat[3], upv[3]; + for (int d = 0; d < 3; ++d) { + // default eye: box center pushed back along (1,1,1) by ~1.7 diagonals + eye[d] = (pos.size() == 3) + ? pos[d] + : center[d] + static_cast(1.7) * diag * + static_cast(0.57735026919); + lookat[d] = (look.size() == 3) ? look[d] : center[d]; + } + upv[0] = (up.size() == 3) ? up[0] : ZERO; + upv[1] = (up.size() == 3) ? up[1] : ZERO; + upv[2] = (up.size() == 3) ? up[2] : ONE; + + real_t forward[3] = { lookat[0] - eye[0], + lookat[1] - eye[1], + lookat[2] - eye[2] }; + normalize3(forward); + real_t right[3]; + cross3(forward, upv, right); + normalize3(right); + real_t up_cam[3]; + cross3(right, forward, up_cam); + + for (int d = 0; d < 3; ++d) { + m_camera_dev.eye[d] = eye[d]; + m_camera_dev.forward[d] = forward[d]; + m_camera_dev.right[d] = right[d]; + m_camera_dev.up[d] = up_cam[d]; + } + m_camera_dev.aspect = static_cast(m_width) / + static_cast(m_height); + m_camera_dev.tan_half_fov = std::tan(static_cast(0.5) * fov * + static_cast(constant::PI) / + static_cast(180.0)); + m_camera_dev.orthographic = ortho; + m_camera_dev.half_h = static_cast(0.5) * ortho_height; + m_camera_dev.half_w = m_camera_dev.half_h * m_camera_dev.aspect; + + /* ---- scenes --------------------------------------------------------- */ + m_scenes.clear(); + const auto scenes_arr = toml::find_or(td, + "output", + "render", + "scenes", + toml::array {}); + for (const auto& sc : scenes_arr) { + Scene scene; + scene.field = toml::find_or(sc, "field", ""); + scene.prefix = toml::find_or(sc, "prefix", scene.field + "_"); + if (scene.field.empty()) { + raise::Warning("output.render scene with no field; skipping", HERE); + continue; + } + scene.tf.vmin = toml::find_or(sc, "min", ZERO); + scene.tf.vmax = toml::find_or(sc, "max", ONE); + scene.tf.log_scale = toml::find_or(sc, "log", false); + scene.tf.n_lut = m_n_lut; + const auto colormap = toml::find_or(sc, "colormap", "viridis"); + // alpha control points: array of [position, alpha] pairs + const auto alpha_raw = toml::find_or>>( + sc, + "alpha", + std::vector> {}); + std::vector> alpha_pts; + for (const auto& p : alpha_raw) { + if (p.size() >= 2) { + alpha_pts.push_back({ p[0], p[1] }); + } + } + scene.tf.lut = buildLUT(colormap, m_n_lut, alpha_pts); + m_scenes.push_back(std::move(scene)); + } + + if (m_scenes.empty()) { + raise::Warning("output.render enabled but no valid scenes; disabling", HERE); + return; + } + + m_enabled = true; + logger::Checkpoint("Volume renderer initialized", HERE); + } + + void Renderer::compositeAndWrite(const std::vector& rgba, + uint64_t order_key, + const std::string& prefix, + timestep_t step) const { + const std::size_t npix = static_cast(m_width) * + static_cast(m_height); + const std::size_t n = npix * 4; + + auto write_image = [&](const std::vector& img) { + // ensure /renders/ exists + const auto dir = m_root / path_t("renders"); + try { + if (not std::filesystem::exists(m_root)) { + std::filesystem::create_directory(m_root); + } + if (not std::filesystem::exists(dir)) { + std::filesystem::create_directory(dir); + } + } catch (const std::exception& e) { + raise::Warning(e.what(), HERE); + } + std::vector bytes(n); + for (std::size_t i = 0; i < n; ++i) { + bytes[i] = quantize(img[i]); + } + const auto fname = dir / fmt::format("%s%08lu.png", + prefix.c_str(), + static_cast(step)); + if (not write_png(fname, m_width, m_height, bytes.data())) { + raise::Warning( + fmt::format("failed to write %s", fname.string().c_str()), + HERE); + } + }; + +#if defined(MPI_ENABLED) + int rank = 0, size = 1; + MPI_Comm_rank(MPI_COMM_WORLD, &rank); + MPI_Comm_size(MPI_COMM_WORLD, &size); + + if (size == 1) { + write_image(rgba); + return; + } + + std::vector recv; + std::vector keys; + if (rank == MPI_ROOT_RANK) { + recv.resize(static_cast(size) * n); + keys.resize(static_cast(size)); + } + const unsigned long long my_key = static_cast(order_key); + + MPI_Gather(rgba.data(), + static_cast(n), + mpi::get_type(), + (rank == MPI_ROOT_RANK) ? recv.data() : nullptr, + static_cast(n), + mpi::get_type(), + MPI_ROOT_RANK, + MPI_COMM_WORLD); + MPI_Gather(&my_key, + 1, + MPI_UNSIGNED_LONG_LONG, + (rank == MPI_ROOT_RANK) ? keys.data() : nullptr, + 1, + MPI_UNSIGNED_LONG_LONG, + MPI_ROOT_RANK, + MPI_COMM_WORLD); + + if (rank != MPI_ROOT_RANK) { + return; + } + + // front-to-back order = ranks sorted by ascending composite key + std::vector order(size); + std::iota(order.begin(), order.end(), 0); + std::stable_sort(order.begin(), order.end(), [&](int a, int b) { + return keys[a] < keys[b]; + }); + + std::vector acc(n, ZERO); + for (const int r : order) { + const real_t* seg_base = recv.data() + static_cast(r) * n; + for (std::size_t p = 0; p < npix; ++p) { + overComposite(acc.data() + p * 4, seg_base + p * 4); + } + } + write_image(acc); +#else + (void)order_key; + write_image(rgba); +#endif + } + +} // namespace out diff --git a/src/output/render/renderer.h b/src/output/render/renderer.h new file mode 100644 index 000000000..707307341 --- /dev/null +++ b/src/output/render/renderer.h @@ -0,0 +1,171 @@ +/** + * @file output/render/renderer.h + * @brief In-situ volume renderer: configuration, cadence, host composite & PNG + * @implements + * - out::Renderer + * - out::CameraDevice + * - out::TransferFunction + * - out::Scene + * @cpp: + * - render/renderer.cpp + * @namespaces: + * - out:: + * @macros: + * - MPI_ENABLED + * - OUTPUT_ENABLED + * @note + * The Renderer is intentionally NOT templated on the engine/metric: it owns + * only metric-agnostic, host-side work (config parsing, cadence tracking, the + * MPI ordered composite and PNG encode). The device ray-march kernel and the + * field preparation live in the templated `Metadomain::Render`, mirroring + * the `out::Writer` (plain) / `Metadomain::Write` (templated) split. + */ + +#ifndef OUTPUT_RENDER_RENDERER_H +#define OUTPUT_RENDER_RENDERER_H + +#include "global.h" + +#include "arch/kokkos_aliases.h" +#include "utils/tools.h" + +#include "framework/parameters/parameters.h" + +#include +#include +#include + +namespace out { + + /** + * @brief Device-friendly POD camera + precomputed per-pixel ray basis. + * @note Trivially copyable; captured by value into the Kokkos kernel. + */ + struct CameraDevice { + real_t eye[3] { ZERO, ZERO, ZERO }; + real_t right[3] { ONE, ZERO, ZERO }; + real_t up[3] { ZERO, ONE, ZERO }; + real_t forward[3] { ZERO, ZERO, -ONE }; + real_t tan_half_fov { ONE }; + real_t aspect { ONE }; + bool orthographic { true }; + real_t half_w { ONE }; + real_t half_h { ONE }; + }; + + /** + * @brief Per-scene transfer function: premultiplied RGBA device LUT + range. + */ + struct TransferFunction { + array_t lut; // device, (n_lut, 4), premultiplied RGBA + int n_lut { 256 }; + real_t vmin { ZERO }; + real_t vmax { ONE }; + bool log_scale { false }; + }; + + /** + * @brief One rendered scalar field -> one PNG stream. + */ + struct Scene { + std::string field; // "N" | "Bmag" | "Jmag" | "smooth_xyz" + std::string prefix; // PNG filename prefix, e.g. "Bmag_" + TransferFunction tf; + }; + + class Renderer { + public: + Renderer() {} + + ~Renderer() = default; + + Renderer(Renderer&&) = default; + + /** + * @brief Parse `[output.render.*]` and build the camera + per-scene LUTs. + * @param params simulation parameters (raw toml read via params.data()) + * @param global_extent global physical box, for default camera framing + */ + void init(const ntt::SimulationParams& params, + const boundaries_t& global_extent); + + [[nodiscard]] + auto shouldRender(timestep_t step, simtime_t time) -> bool { + return m_enabled and m_tracker.shouldWrite(step, time); + } + + /** + * @brief Composite the per-rank host image across MPI and write the PNG. + * @param rgba host buffer, length width*height*4, premultiplied RGBA, in + * pixel-major / channel-minor order (rgba[pix*4 + ch]) + * @param order_key this rank's front-to-back sort key (see composite.h) + * @param prefix PNG filename prefix + * @param step current timestep (for the filename cycle number) + * @note Only the MPI root rank writes the file. + */ + void compositeAndWrite(const std::vector& rgba, + uint64_t order_key, + const std::string& prefix, + timestep_t step) const; + + /* getters -------------------------------------------------------------- */ + [[nodiscard]] + auto enabled() const -> bool { + return m_enabled; + } + + [[nodiscard]] + auto width() const -> int { + return m_width; + } + + [[nodiscard]] + auto height() const -> int { + return m_height; + } + + [[nodiscard]] + auto samples() const -> int { + return m_samples; + } + + [[nodiscard]] + auto stepSize() const -> real_t { + return m_step_size; + } + + [[nodiscard]] + auto earlyAlpha() const -> real_t { + return m_early_alpha; + } + + [[nodiscard]] + auto camera() const -> const CameraDevice& { + return m_camera_dev; + } + + [[nodiscard]] + auto scenes() const -> const std::vector& { + return m_scenes; + } + + private: + bool m_enabled { false }; + + int m_width { 1024 }; + int m_height { 1024 }; + int m_samples { 400 }; + real_t m_step_size { ZERO }; // world units/step; 0 => derive from samples + real_t m_early_alpha { static_cast(0.99) }; + int m_n_lut { 256 }; + + CameraDevice m_camera_dev; + std::vector m_scenes; + + tools::Tracker m_tracker; + path_t m_root; + }; + +} // namespace out + +#endif // OUTPUT_RENDER_RENDERER_H diff --git a/src/output/render/transfer_fn.h b/src/output/render/transfer_fn.h new file mode 100644 index 000000000..a7d682cb3 --- /dev/null +++ b/src/output/render/transfer_fn.h @@ -0,0 +1,192 @@ +/** + * @file output/render/transfer_fn.h + * @brief Colormap tables and premultiplied RGBA look-up-table builder + * @implements + * - out::buildLUT + * - out::colormapRGB + * @namespaces: + * - out:: + * @note + * Colormaps are stored as a handful of anchor colors and linearly + * interpolated; this is visually indistinguishable from the full 256-entry + * matplotlib tables for volume rendering while keeping the header compact. + * The LUT is built on the host and deep-copied to a device View of shape + * (N_LUT, 4) holding premultiplied RGBA (R=r*a, G=g*a, B=b*a, A=a). + */ + +#ifndef OUTPUT_RENDER_TRANSFER_FN_H +#define OUTPUT_RENDER_TRANSFER_FN_H + +#include "global.h" + +#include "arch/kokkos_aliases.h" +#include "utils/numeric.h" + +#include +#include +#include + +namespace out { + + namespace cmap_hidden { + + // anchor colors sampled at uniform positions in [0, 1] + struct Anchors { + const float (*rgb)[3]; + int n; + }; + + inline constexpr float viridis[9][3] = { + { 0.267004f, 0.004874f, 0.329415f }, + { 0.282623f, 0.140926f, 0.457517f }, + { 0.253935f, 0.265254f, 0.529983f }, + { 0.206756f, 0.371758f, 0.553117f }, + { 0.163625f, 0.471133f, 0.558148f }, + { 0.127568f, 0.566949f, 0.550556f }, + { 0.134692f, 0.658636f, 0.517649f }, + { 0.477504f, 0.821444f, 0.318195f }, + { 0.993248f, 0.906157f, 0.143936f }, + }; + + inline constexpr float inferno[9][3] = { + { 0.001462f, 0.000466f, 0.013866f }, + { 0.087411f, 0.044556f, 0.224813f }, + { 0.258234f, 0.038571f, 0.406485f }, + { 0.416331f, 0.090203f, 0.432943f }, + { 0.578304f, 0.148039f, 0.404411f }, + { 0.735683f, 0.215906f, 0.330245f }, + { 0.865006f, 0.316822f, 0.226055f }, + { 0.954506f, 0.468744f, 0.099874f }, + { 0.988362f, 0.998364f, 0.644924f }, + }; + + inline constexpr float plasma[9][3] = { + { 0.050383f, 0.029803f, 0.527975f }, + { 0.287076f, 0.010855f, 0.627295f }, + { 0.417642f, 0.000564f, 0.658390f }, + { 0.562738f, 0.051545f, 0.641509f }, + { 0.692840f, 0.165141f, 0.564522f }, + { 0.798216f, 0.280197f, 0.469538f }, + { 0.881443f, 0.392529f, 0.383229f }, + { 0.949217f, 0.517763f, 0.295662f }, + { 0.940015f, 0.975158f, 0.131326f }, + }; + + // Moreland cool-to-warm diverging + inline constexpr float cool2warm[3][3] = { + { 0.230f, 0.299f, 0.754f }, + { 0.865f, 0.865f, 0.865f }, + { 0.706f, 0.016f, 0.150f }, + }; + + inline constexpr float gray[2][3] = { + { 0.0f, 0.0f, 0.0f }, + { 1.0f, 1.0f, 1.0f }, + }; + + inline auto lookup(const std::string& name) -> Anchors { + if (name == "inferno") { + return { inferno, 9 }; + } else if (name == "plasma") { + return { plasma, 9 }; + } else if (name == "cool2warm" or name == "coolwarm") { + return { cool2warm, 3 }; + } else if (name == "gray" or name == "grey") { + return { gray, 2 }; + } else { + // default / "viridis" + return { viridis, 9 }; + } + } + + } // namespace cmap_hidden + + /** + * @brief Sample a named colormap at u in [0, 1], returning RGB in [0, 1]. + */ + inline void colormapRGB(const std::string& name, + real_t u, + real_t& r, + real_t& g, + real_t& b) { + const auto anchors = cmap_hidden::lookup(name); + if (u <= ZERO) { + r = anchors.rgb[0][0]; + g = anchors.rgb[0][1]; + b = anchors.rgb[0][2]; + return; + } + if (u >= ONE) { + r = anchors.rgb[anchors.n - 1][0]; + g = anchors.rgb[anchors.n - 1][1]; + b = anchors.rgb[anchors.n - 1][2]; + return; + } + const real_t x = u * static_cast(anchors.n - 1); + const int i0 = static_cast(x); + const int i1 = (i0 + 1 < anchors.n) ? (i0 + 1) : i0; + const real_t t = x - static_cast(i0); + r = static_cast(anchors.rgb[i0][0]) * (ONE - t) + + static_cast(anchors.rgb[i1][0]) * t; + g = static_cast(anchors.rgb[i0][1]) * (ONE - t) + + static_cast(anchors.rgb[i1][1]) * t; + b = static_cast(anchors.rgb[i0][2]) * (ONE - t) + + static_cast(anchors.rgb[i1][2]) * t; + } + + /** + * @brief Piecewise-linear opacity from sorted (position, alpha) control points. + */ + inline auto alphaAt(const std::vector>& pts, real_t u) + -> real_t { + if (pts.empty()) { + return u; // sensible default: linear ramp + } + if (u <= pts.front()[0]) { + return pts.front()[1]; + } + if (u >= pts.back()[0]) { + return pts.back()[1]; + } + for (std::size_t i = 0; i + 1 < pts.size(); ++i) { + if (u >= pts[i][0] and u <= pts[i + 1][0]) { + const real_t span = pts[i + 1][0] - pts[i][0]; + const real_t t = (span > ZERO) ? (u - pts[i][0]) / span : ZERO; + return pts[i][1] * (ONE - t) + pts[i + 1][1] * t; + } + } + return pts.back()[1]; + } + + /** + * @brief Build a premultiplied RGBA device LUT from a colormap + alpha points. + * @param colormap name of the colormap + * @param n_lut number of entries + * @param alpha_pts sorted (position, alpha) control points in [0,1]x[0,1] + * @return device View of shape (n_lut, 4), premultiplied RGBA + */ + inline auto buildLUT(const std::string& colormap, + int n_lut, + const std::vector>& alpha_pts) + -> array_t { + array_t lut { "render_lut", static_cast(n_lut) }; + auto lut_h = Kokkos::create_mirror_view(lut); + for (int i = 0; i < n_lut; ++i) { + const real_t u = (n_lut > 1) + ? static_cast(i) / static_cast(n_lut - 1) + : ZERO; + real_t r, g, b; + colormapRGB(colormap, u, r, g, b); + const real_t a = alphaAt(alpha_pts, u); + lut_h(i, 0) = r * a; // premultiplied + lut_h(i, 1) = g * a; + lut_h(i, 2) = b * a; + lut_h(i, 3) = a; + } + Kokkos::deep_copy(lut, lut_h); + return lut; + } + +} // namespace out + +#endif // OUTPUT_RENDER_TRANSFER_FN_H From ca24afe13d19b5a8df2422d07d8233fb8de64449 Mon Sep 17 00:00:00 2001 From: Ludwig Boess Date: Tue, 30 Jun 2026 00:01:17 -0400 Subject: [PATCH 036/125] add background to rendered image --- src/output/render/renderer.cpp | 23 +++++++++++++++++++++-- src/output/render/renderer.h | 3 +++ 2 files changed, 24 insertions(+), 2 deletions(-) diff --git a/src/output/render/renderer.cpp b/src/output/render/renderer.cpp index 019fc7eaa..37c01e80e 100644 --- a/src/output/render/renderer.cpp +++ b/src/output/render/renderer.cpp @@ -89,6 +89,18 @@ namespace out { static_cast(0.99)); m_n_lut = toml::find_or(td, "output", "render", "n_lut", 256); + // opaque background color (shows through low-alpha pixels); default black + const auto bg = toml::find_or>(td, + "output", + "render", + "background", + std::vector {}); + if (bg.size() == 3) { + m_background[0] = bg[0]; + m_background[1] = bg[1]; + m_background[2] = bg[2]; + } + // cadence: mirror output.* (interval in steps; interval_time in sim time) const auto interval = toml::find_or(td, "output", @@ -257,9 +269,16 @@ namespace out { } catch (const std::exception& e) { raise::Warning(e.what(), HERE); } + // composite the premultiplied image over the opaque background: + // out = src_premult + (1 - src_alpha) * background, alpha = opaque. std::vector bytes(n); - for (std::size_t i = 0; i < n; ++i) { - bytes[i] = quantize(img[i]); + for (std::size_t p = 0; p < npix; ++p) { + const real_t a = img[p * 4 + 3]; + const real_t inv = ONE - a; + bytes[p * 4 + 0] = quantize(img[p * 4 + 0] + inv * m_background[0]); + bytes[p * 4 + 1] = quantize(img[p * 4 + 1] + inv * m_background[1]); + bytes[p * 4 + 2] = quantize(img[p * 4 + 2] + inv * m_background[2]); + bytes[p * 4 + 3] = 255; } const auto fname = dir / fmt::format("%s%08lu.png", prefix.c_str(), diff --git a/src/output/render/renderer.h b/src/output/render/renderer.h index 707307341..aaae57243 100644 --- a/src/output/render/renderer.h +++ b/src/output/render/renderer.h @@ -158,6 +158,9 @@ namespace out { real_t m_step_size { ZERO }; // world units/step; 0 => derive from samples real_t m_early_alpha { static_cast(0.99) }; int m_n_lut { 256 }; + // opaque background composited under the final image (shows through + // low-alpha pixels); defaults to black. + real_t m_background[3] { ZERO, ZERO, ZERO }; CameraDevice m_camera_dev; std::vector m_scenes; From 614aff4ef9e608f4cc10f2bf9470c41ec3b105a6 Mon Sep 17 00:00:00 2001 From: Ludwig Boess Date: Tue, 30 Jun 2026 11:07:02 -0400 Subject: [PATCH 037/125] added colorbar --- src/framework/domain/metadomain_render.cpp | 2 +- src/output/render/colorbar.h | 270 +++++++++++++++++++++ src/output/render/renderer.cpp | 61 ++++- src/output/render/renderer.h | 11 +- 4 files changed, 338 insertions(+), 6 deletions(-) create mode 100644 src/output/render/colorbar.h diff --git a/src/framework/domain/metadomain_render.cpp b/src/framework/domain/metadomain_render.cpp index 4e9cbf696..c83d7f703 100644 --- a/src/framework/domain/metadomain_render.cpp +++ b/src/framework/domain/metadomain_render.cpp @@ -284,7 +284,7 @@ namespace ntt { rgba[p * 4 + 2] = image_h(p, 2); rgba[p * 4 + 3] = image_h(p, 3); } - g_renderer.compositeAndWrite(rgba, order_key, scene.prefix, current_step); + g_renderer.compositeAndWrite(rgba, order_key, scene, current_step); rendered_any = true; } return rendered_any; diff --git a/src/output/render/colorbar.h b/src/output/render/colorbar.h new file mode 100644 index 000000000..4c311bd90 --- /dev/null +++ b/src/output/render/colorbar.h @@ -0,0 +1,270 @@ +/** + * @file output/render/colorbar.h + * @brief Draw a colorbar (gradient strip + value ticks + label) onto an + * opaque 8-bit RGBA image, using a self-contained 5x7 bitmap font. + * @implements + * - out::drawColorbar + * @namespaces: + * - out:: + * @note + * Header-only, host-only. No font dependency: a compact 5x7 ASCII font (digits, + * sign/exponent symbols, and A-Z) is embedded. Lowercase is mapped to + * uppercase; unknown glyphs render as blank. Drawn on the MPI root rank after + * the final composite, so it only ever touches the output buffer. + */ + +#ifndef OUTPUT_RENDER_COLORBAR_H +#define OUTPUT_RENDER_COLORBAR_H + +#include "global.h" + +#include "utils/numeric.h" + +#include "output/render/transfer_fn.h" + +#include +#include +#include +#include + +namespace out { + + namespace cbar_hidden { + + // 5x7 glyph: 7 rows, low 5 bits per row, bit 4 = leftmost column. + inline auto glyph(char c) -> const uint8_t* { + // map lowercase to uppercase + if (c >= 'a' and c <= 'z') { + c = static_cast(c - 'a' + 'A'); + } + switch (c) { + case '0': { static const uint8_t g[7] = { 0b01110, 0b10001, 0b10011, 0b10101, 0b11001, 0b10001, 0b01110 }; return g; } + case '1': { static const uint8_t g[7] = { 0b00100, 0b01100, 0b00100, 0b00100, 0b00100, 0b00100, 0b01110 }; return g; } + case '2': { static const uint8_t g[7] = { 0b01110, 0b10001, 0b00001, 0b00010, 0b00100, 0b01000, 0b11111 }; return g; } + case '3': { static const uint8_t g[7] = { 0b11111, 0b00010, 0b00100, 0b00010, 0b00001, 0b10001, 0b01110 }; return g; } + case '4': { static const uint8_t g[7] = { 0b00010, 0b00110, 0b01010, 0b10010, 0b11111, 0b00010, 0b00010 }; return g; } + case '5': { static const uint8_t g[7] = { 0b11111, 0b10000, 0b11110, 0b00001, 0b00001, 0b10001, 0b01110 }; return g; } + case '6': { static const uint8_t g[7] = { 0b00110, 0b01000, 0b10000, 0b11110, 0b10001, 0b10001, 0b01110 }; return g; } + case '7': { static const uint8_t g[7] = { 0b11111, 0b00001, 0b00010, 0b00100, 0b01000, 0b01000, 0b01000 }; return g; } + case '8': { static const uint8_t g[7] = { 0b01110, 0b10001, 0b10001, 0b01110, 0b10001, 0b10001, 0b01110 }; return g; } + case '9': { static const uint8_t g[7] = { 0b01110, 0b10001, 0b10001, 0b01111, 0b00001, 0b00010, 0b01100 }; return g; } + case '.': { static const uint8_t g[7] = { 0b00000, 0b00000, 0b00000, 0b00000, 0b00000, 0b00110, 0b00110 }; return g; } + case '-': { static const uint8_t g[7] = { 0b00000, 0b00000, 0b00000, 0b11111, 0b00000, 0b00000, 0b00000 }; return g; } + case '+': { static const uint8_t g[7] = { 0b00000, 0b00100, 0b00100, 0b11111, 0b00100, 0b00100, 0b00000 }; return g; } + case '_': { static const uint8_t g[7] = { 0b00000, 0b00000, 0b00000, 0b00000, 0b00000, 0b00000, 0b11111 }; return g; } + case ':': { static const uint8_t g[7] = { 0b00000, 0b00110, 0b00110, 0b00000, 0b00110, 0b00110, 0b00000 }; return g; } + case '/': { static const uint8_t g[7] = { 0b00001, 0b00010, 0b00010, 0b00100, 0b01000, 0b01000, 0b10000 }; return g; } + case 'A': { static const uint8_t g[7] = { 0b01110, 0b10001, 0b10001, 0b11111, 0b10001, 0b10001, 0b10001 }; return g; } + case 'B': { static const uint8_t g[7] = { 0b11110, 0b10001, 0b10001, 0b11110, 0b10001, 0b10001, 0b11110 }; return g; } + case 'C': { static const uint8_t g[7] = { 0b01110, 0b10001, 0b10000, 0b10000, 0b10000, 0b10001, 0b01110 }; return g; } + case 'D': { static const uint8_t g[7] = { 0b11100, 0b10010, 0b10001, 0b10001, 0b10001, 0b10010, 0b11100 }; return g; } + case 'E': { static const uint8_t g[7] = { 0b11111, 0b10000, 0b10000, 0b11110, 0b10000, 0b10000, 0b11111 }; return g; } + case 'F': { static const uint8_t g[7] = { 0b11111, 0b10000, 0b10000, 0b11110, 0b10000, 0b10000, 0b10000 }; return g; } + case 'G': { static const uint8_t g[7] = { 0b01110, 0b10001, 0b10000, 0b10111, 0b10001, 0b10001, 0b01111 }; return g; } + case 'H': { static const uint8_t g[7] = { 0b10001, 0b10001, 0b10001, 0b11111, 0b10001, 0b10001, 0b10001 }; return g; } + case 'I': { static const uint8_t g[7] = { 0b01110, 0b00100, 0b00100, 0b00100, 0b00100, 0b00100, 0b01110 }; return g; } + case 'J': { static const uint8_t g[7] = { 0b00111, 0b00010, 0b00010, 0b00010, 0b10010, 0b10010, 0b01100 }; return g; } + case 'K': { static const uint8_t g[7] = { 0b10001, 0b10010, 0b10100, 0b11000, 0b10100, 0b10010, 0b10001 }; return g; } + case 'L': { static const uint8_t g[7] = { 0b10000, 0b10000, 0b10000, 0b10000, 0b10000, 0b10000, 0b11111 }; return g; } + case 'M': { static const uint8_t g[7] = { 0b10001, 0b11011, 0b10101, 0b10101, 0b10001, 0b10001, 0b10001 }; return g; } + case 'N': { static const uint8_t g[7] = { 0b10001, 0b11001, 0b10101, 0b10011, 0b10001, 0b10001, 0b10001 }; return g; } + case 'O': { static const uint8_t g[7] = { 0b01110, 0b10001, 0b10001, 0b10001, 0b10001, 0b10001, 0b01110 }; return g; } + case 'P': { static const uint8_t g[7] = { 0b11110, 0b10001, 0b10001, 0b11110, 0b10000, 0b10000, 0b10000 }; return g; } + case 'Q': { static const uint8_t g[7] = { 0b01110, 0b10001, 0b10001, 0b10001, 0b10101, 0b10010, 0b01101 }; return g; } + case 'R': { static const uint8_t g[7] = { 0b11110, 0b10001, 0b10001, 0b11110, 0b10100, 0b10010, 0b10001 }; return g; } + case 'S': { static const uint8_t g[7] = { 0b01111, 0b10000, 0b10000, 0b01110, 0b00001, 0b00001, 0b11110 }; return g; } + case 'T': { static const uint8_t g[7] = { 0b11111, 0b00100, 0b00100, 0b00100, 0b00100, 0b00100, 0b00100 }; return g; } + case 'U': { static const uint8_t g[7] = { 0b10001, 0b10001, 0b10001, 0b10001, 0b10001, 0b10001, 0b01110 }; return g; } + case 'V': { static const uint8_t g[7] = { 0b10001, 0b10001, 0b10001, 0b10001, 0b10001, 0b01010, 0b00100 }; return g; } + case 'W': { static const uint8_t g[7] = { 0b10001, 0b10001, 0b10001, 0b10101, 0b10101, 0b11011, 0b10001 }; return g; } + case 'X': { static const uint8_t g[7] = { 0b10001, 0b10001, 0b01010, 0b00100, 0b01010, 0b10001, 0b10001 }; return g; } + case 'Y': { static const uint8_t g[7] = { 0b10001, 0b10001, 0b01010, 0b00100, 0b00100, 0b00100, 0b00100 }; return g; } + case 'Z': { static const uint8_t g[7] = { 0b11111, 0b00001, 0b00010, 0b00100, 0b01000, 0b10000, 0b11111 }; return g; } + default: { static const uint8_t g[7] = { 0, 0, 0, 0, 0, 0, 0 }; return g; } // blank + } + } + + inline void setPx(uint8_t* rgba, int W, int H, int x, int y, uint8_t r, + uint8_t g, uint8_t b) { + if (x < 0 or x >= W or y < 0 or y >= H) { + return; + } + const std::size_t i = (static_cast(y) * W + x) * 4; + rgba[i + 0] = r; + rgba[i + 1] = g; + rgba[i + 2] = b; + rgba[i + 3] = 255; + } + + inline void drawChar(uint8_t* rgba, int W, int H, int x, int y, char c, + int s, uint8_t r, uint8_t g, uint8_t b) { + const uint8_t* gl = glyph(c); + for (int row = 0; row < 7; ++row) { + for (int col = 0; col < 5; ++col) { + if (gl[row] & (1u << (4 - col))) { + for (int dy = 0; dy < s; ++dy) { + for (int dx = 0; dx < s; ++dx) { + setPx(rgba, W, H, x + col * s + dx, y + row * s + dy, r, g, b); + } + } + } + } + } + } + + inline void drawText(uint8_t* rgba, int W, int H, int x, int y, + const std::string& str, int s, uint8_t r, uint8_t g, + uint8_t b) { + int cx = x; + for (const char c : str) { + drawChar(rgba, W, H, cx, y, c, s, r, g, b); + cx += 6 * s; // 5px glyph + 1px spacing + } + } + + inline auto quant(real_t v) -> uint8_t { + const real_t c = (v < ZERO) ? ZERO : ((v > ONE) ? ONE : v); + return static_cast(c * static_cast(255.0) + HALF); + } + + inline auto fmtNum(real_t v) -> std::string { + char buf[32]; + std::snprintf(buf, sizeof(buf), "%.3g", static_cast(v)); + return std::string(buf); + } + + inline auto scale(int H) -> int { + return std::max(2, H / 400); + } + + } // namespace cbar_hidden + + /** + * @brief Pixel width of the colorbar block (bar + gap + labels + padding). + * @note Depends only on H, so it can size a canvas margin before drawing. + */ + inline auto colorbarBlockWidth(int H) -> int { + const int s = cbar_hidden::scale(H); + const int char_w = 6 * s; + const int bar_w = std::max(12, H / 50); + const int gap = 3 * s; + const int label_w = 9 * char_w; // room for e.g. "-1.23e+04" + const int pad = 4 * s; + return pad + bar_w + gap + label_w + pad; + } + + /** + * @brief Draw a vertical colorbar onto an opaque RGBA buffer. + * @param rgba width*height*4 bytes, opaque (alpha forced to 255 on drawn px) + * @param W,H image dimensions + * @param colormap name of the colormap to redraw the gradient + * @param vmin,vmax value range mapped onto the bar + * @param log_scale if true, ticks are spaced/labelled logarithmically + * @param label title drawn above the bar (e.g. the field name) + * @param bg background RGB (to auto-pick contrasting text color) + */ + inline void drawColorbar(uint8_t* rgba, + int W, + int H, + const std::string& colormap, + real_t vmin, + real_t vmax, + bool log_scale, + const std::string& label, + const real_t bg[3]) { + using namespace cbar_hidden; + + const int s = scale(H); + const int char_h = 8 * s; + const int bar_w = std::max(12, H / 50); + const int bar_h = H / 2; + const int gap = 3 * s; + const int pad = 4 * s; + const int block_w = colorbarBlockWidth(H); + + // place the bar near the right edge of the (possibly extended) canvas + int bar_x = W - block_w + pad; + if (bar_x < pad) { + bar_x = pad; + } + const int bar_y = (H - bar_h) / 2; + + // contrasting monochrome for text / frame / ticks + const real_t lum = static_cast(0.299) * bg[0] + + static_cast(0.587) * bg[1] + + static_cast(0.114) * bg[2]; + const uint8_t tc = (lum < HALF) ? 255 : 0; + + // gradient strip (top = vmax, bottom = vmin) + for (int j = 0; j < bar_h; ++j) { + const real_t u = (bar_h > 1) + ? ONE - static_cast(j) / + static_cast(bar_h - 1) + : ZERO; + real_t cr, cg, cb; + colormapRGB(colormap, u, cr, cg, cb); + const uint8_t R = quant(cr), G = quant(cg), B = quant(cb); + for (int i = 0; i < bar_w; ++i) { + setPx(rgba, W, H, bar_x + i, bar_y + j, R, G, B); + } + } + + // frame (thickness s) + for (int t = 0; t < s; ++t) { + for (int i = -t; i < bar_w + t; ++i) { + setPx(rgba, W, H, bar_x + i, bar_y - t, tc, tc, tc); + setPx(rgba, W, H, bar_x + i, bar_y + bar_h - 1 + t, tc, tc, tc); + } + for (int j = -t; j < bar_h + t; ++j) { + setPx(rgba, W, H, bar_x - t, bar_y + j, tc, tc, tc); + setPx(rgba, W, H, bar_x + bar_w - 1 + t, bar_y + j, tc, tc, tc); + } + } + + // ticks + labels + const bool can_log = log_scale and vmin > ZERO and vmax > ZERO; + const real_t lvmin = can_log ? math::log10(vmin) : ZERO; + const real_t lvmax = can_log ? math::log10(vmax) : ZERO; + const int nticks = 5; + for (int t = 0; t < nticks; ++t) { + const real_t u = static_cast(t) / + static_cast(nticks - 1); + const real_t val = can_log + ? math::pow(static_cast(10), + lvmin + (lvmax - lvmin) * u) + : (vmin + (vmax - vmin) * u); + const int ty = bar_y + + static_cast((ONE - u) * + static_cast(bar_h - 1)); + // tick line + for (int i = 0; i < gap; ++i) { + for (int w = 0; w < std::max(1, s / 2); ++w) { + setPx(rgba, W, H, bar_x + bar_w + i, ty + w, tc, tc, tc); + } + } + // label, vertically centered on the tick + drawText(rgba, + W, + H, + bar_x + bar_w + gap + 2 * s, + ty - char_h / 2, + fmtNum(val), + s, + tc, + tc, + tc); + } + + // title above the bar, left-aligned to the bar so it stays in the strip + if (not label.empty()) { + int ty = bar_y - char_h - 2 * gap; + if (ty < 0) { + ty = 0; + } + drawText(rgba, W, H, bar_x, ty, label, s, tc, tc, tc); + } + } + +} // namespace out + +#endif // OUTPUT_RENDER_COLORBAR_H diff --git a/src/output/render/renderer.cpp b/src/output/render/renderer.cpp index 37c01e80e..c28fed054 100644 --- a/src/output/render/renderer.cpp +++ b/src/output/render/renderer.cpp @@ -8,6 +8,7 @@ #include "utils/log.h" #include "utils/numeric.h" +#include "output/render/colorbar.h" #include "output/render/composite.h" #include "output/render/png.h" #include "output/render/transfer_fn.h" @@ -101,6 +102,13 @@ namespace out { m_background[2] = bg[2]; } + m_colorbar = toml::find_or(td, "output", "render", "colorbar", true); + m_colorbar_outside = toml::find_or(td, + "output", + "render", + "colorbar_outside", + true); + // cadence: mirror output.* (interval in steps; interval_time in sim time) const auto interval = toml::find_or(td, "output", @@ -219,11 +227,13 @@ namespace out { raise::Warning("output.render scene with no field; skipping", HERE); continue; } + scene.label = toml::find_or(sc, "label", scene.field); scene.tf.vmin = toml::find_or(sc, "min", ZERO); scene.tf.vmax = toml::find_or(sc, "max", ONE); scene.tf.log_scale = toml::find_or(sc, "log", false); scene.tf.n_lut = m_n_lut; const auto colormap = toml::find_or(sc, "colormap", "viridis"); + scene.tf.colormap = colormap; // alpha control points: array of [position, alpha] pairs const auto alpha_raw = toml::find_or>>( sc, @@ -250,7 +260,7 @@ namespace out { void Renderer::compositeAndWrite(const std::vector& rgba, uint64_t order_key, - const std::string& prefix, + const Scene& scene, timestep_t step) const { const std::size_t npix = static_cast(m_width) * static_cast(m_height); @@ -281,9 +291,54 @@ namespace out { bytes[p * 4 + 3] = 255; } const auto fname = dir / fmt::format("%s%08lu.png", - prefix.c_str(), + scene.prefix.c_str(), static_cast(step)); - if (not write_png(fname, m_width, m_height, bytes.data())) { + bool ok = true; + if (m_colorbar and m_colorbar_outside) { + // extend the canvas to the right so the colorbar sits in its own margin, + // outside the rendered volume. + const int strip = colorbarBlockWidth(m_height); + const int CW = m_width + strip; + const uint8_t bR = quantize(m_background[0]); + const uint8_t bG = quantize(m_background[1]); + const uint8_t bB = quantize(m_background[2]); + std::vector canvas(static_cast(CW) * m_height * 4); + for (std::size_t i = 0; i < canvas.size(); i += 4) { + canvas[i + 0] = bR; + canvas[i + 1] = bG; + canvas[i + 2] = bB; + canvas[i + 3] = 255; + } + for (int y = 0; y < m_height; ++y) { + std::copy_n(&bytes[static_cast(y) * m_width * 4], + static_cast(m_width) * 4, + &canvas[static_cast(y) * CW * 4]); + } + drawColorbar(canvas.data(), + CW, + m_height, + scene.tf.colormap, + scene.tf.vmin, + scene.tf.vmax, + scene.tf.log_scale, + scene.label, + m_background); + ok = write_png(fname, CW, m_height, canvas.data()); + } else { + if (m_colorbar) { + drawColorbar(bytes.data(), + m_width, + m_height, + scene.tf.colormap, + scene.tf.vmin, + scene.tf.vmax, + scene.tf.log_scale, + scene.label, + m_background); + } + ok = write_png(fname, m_width, m_height, bytes.data()); + } + if (not ok) { raise::Warning( fmt::format("failed to write %s", fname.string().c_str()), HERE); diff --git a/src/output/render/renderer.h b/src/output/render/renderer.h index aaae57243..88294183d 100644 --- a/src/output/render/renderer.h +++ b/src/output/render/renderer.h @@ -62,6 +62,7 @@ namespace out { real_t vmin { ZERO }; real_t vmax { ONE }; bool log_scale { false }; + std::string colormap { "viridis" }; // for redrawing the colorbar }; /** @@ -70,6 +71,7 @@ namespace out { struct Scene { std::string field; // "N" | "Bmag" | "Jmag" | "smooth_xyz" std::string prefix; // PNG filename prefix, e.g. "Bmag_" + std::string label; // colorbar title (defaults to field) TransferFunction tf; }; @@ -99,13 +101,13 @@ namespace out { * @param rgba host buffer, length width*height*4, premultiplied RGBA, in * pixel-major / channel-minor order (rgba[pix*4 + ch]) * @param order_key this rank's front-to-back sort key (see composite.h) - * @param prefix PNG filename prefix + * @param scene the scene being written (prefix, colorbar colormap/range/label) * @param step current timestep (for the filename cycle number) * @note Only the MPI root rank writes the file. */ void compositeAndWrite(const std::vector& rgba, uint64_t order_key, - const std::string& prefix, + const Scene& scene, timestep_t step) const; /* getters -------------------------------------------------------------- */ @@ -161,6 +163,11 @@ namespace out { // opaque background composited under the final image (shows through // low-alpha pixels); defaults to black. real_t m_background[3] { ZERO, ZERO, ZERO }; + // draw a colorbar (gradient + value ticks + label) on each PNG + bool m_colorbar { true }; + // draw the colorbar in an extended right margin (outside the render region) + // rather than overlaying it on top of the rendered volume + bool m_colorbar_outside { true }; CameraDevice m_camera_dev; std::vector m_scenes; From df108202d70c468aaefb03b9e18d9cdab8282c1c Mon Sep 17 00:00:00 2001 From: Ludwig Boess Date: Tue, 30 Jun 2026 11:43:36 -0400 Subject: [PATCH 038/125] significant speedup --- src/framework/domain/metadomain_render.cpp | 99 ++++++++------ src/output/render/composite.h | 143 +++++++++++++++++++ src/output/render/raymarch.hpp | 25 +++- src/output/render/renderer.cpp | 151 +++++++++++++++------ src/output/render/renderer.h | 29 ++-- 5 files changed, 347 insertions(+), 100 deletions(-) diff --git a/src/framework/domain/metadomain_render.cpp b/src/framework/domain/metadomain_render.cpp index c83d7f703..013b855a4 100644 --- a/src/framework/domain/metadomain_render.cpp +++ b/src/framework/domain/metadomain_render.cpp @@ -156,6 +156,11 @@ namespace ntt { const auto metric = local_domain->mesh.metric; + // screen-space bounding box of this domain's footprint (same for all + // scenes); we only ray-march and composite within it. + int bx0 = 0, by0 = 0, bw = 0, bh = 0; + const bool on_screen = out::screenBBox(cam, W, H, lo, hi, bx0, by0, bw, bh); + bool rendered_any = false; for (const auto& scene : g_renderer.scenes()) { Kokkos::deep_copy(bckp, ZERO); @@ -240,51 +245,59 @@ namespace ntt { // sum-into-active that SynchronizeFields performs). CommunicateBckp(*local_domain, { 0, 1 }); - // ---- launch the ray-march kernel ------------------------------- // - array_t image { "render_img", - static_cast(W) * - static_cast(H) }; - randacc_ndfield_t Fld { bckp }; - Kokkos::parallel_for( - "VolumeRayMarch", - CreateRangePolicy({ 0, 0 }, - { static_cast(W), - static_cast(H) }), - kernel::VolumeRayMarch_kernel(Fld, - 0u, - metric, - cam, - lo, - hi, - ext0, - ext1, - ext2, - W, - H, - ds, - max_steps, - scene.tf.lut, - scene.tf.n_lut, - scene.tf.vmin, - scene.tf.vmax, - scene.tf.log_scale, - g_renderer.earlyAlpha(), - image)); - Kokkos::fence(); + // ---- launch the ray-march kernel over the screen bbox ---------- // + out::SubImage sub; + if (on_screen) { + sub.x0 = bx0; + sub.y0 = by0; + sub.w = bw; + sub.h = bh; + const std::size_t bnpix = static_cast(bw) * + static_cast(bh); + array_t image { "render_img", bnpix }; + randacc_ndfield_t Fld { bckp }; + Kokkos::parallel_for( + "VolumeRayMarch", + CreateRangePolicy({ 0, 0 }, + { static_cast(bw), + static_cast(bh) }), + kernel::VolumeRayMarch_kernel(Fld, + 0u, + metric, + cam, + lo, + hi, + ext0, + ext1, + ext2, + W, + H, + bx0, + by0, + bw, + ds, + max_steps, + scene.tf.lut, + scene.tf.n_lut, + scene.tf.vmin, + scene.tf.vmax, + scene.tf.log_scale, + g_renderer.earlyAlpha(), + image)); + Kokkos::fence(); - // device -> host, into a layout-agnostic pixel-major buffer - auto image_h = Kokkos::create_mirror_view(image); - Kokkos::deep_copy(image_h, image); - const std::size_t npix = static_cast(W) * - static_cast(H); - std::vector rgba(npix * 4); - for (std::size_t p = 0; p < npix; ++p) { - rgba[p * 4 + 0] = image_h(p, 0); - rgba[p * 4 + 1] = image_h(p, 1); - rgba[p * 4 + 2] = image_h(p, 2); - rgba[p * 4 + 3] = image_h(p, 3); + // device -> host, into a layout-agnostic pixel-major buffer + auto image_h = Kokkos::create_mirror_view(image); + Kokkos::deep_copy(image_h, image); + sub.rgba.resize(bnpix * 4); + for (std::size_t p = 0; p < bnpix; ++p) { + sub.rgba[p * 4 + 0] = image_h(p, 0); + sub.rgba[p * 4 + 1] = image_h(p, 1); + sub.rgba[p * 4 + 2] = image_h(p, 2); + sub.rgba[p * 4 + 3] = image_h(p, 3); + } } - g_renderer.compositeAndWrite(rgba, order_key, scene, current_step); + g_renderer.compositeAndWrite(sub, order_key, scene, current_step); rendered_any = true; } return rendered_any; diff --git a/src/output/render/composite.h b/src/output/render/composite.h index 39ea31aad..47bcdcc15 100644 --- a/src/output/render/composite.h +++ b/src/output/render/composite.h @@ -24,6 +24,11 @@ #include "utils/numeric.h" +#include "output/render/renderer.h" + +#include +#include +#include #include #include @@ -72,6 +77,144 @@ namespace out { acc[3] += one_minus_a * seg[3]; } + /** + * @brief Project a world point to a (fractional) screen pixel, inverting the + * ray-march kernel's ray generation. + * @return false if the point is behind a perspective camera (no projection) + */ + inline auto projectToScreen(const CameraDevice& cam, + int W, + int H, + const real_t p[3], + real_t& outx, + real_t& outy) -> bool { + const real_t dx = p[0] - cam.eye[0]; + const real_t dy = p[1] - cam.eye[1]; + const real_t dz = p[2] - cam.eye[2]; + const real_t cx = dx * cam.right[0] + dy * cam.right[1] + dz * cam.right[2]; + const real_t cy = dx * cam.up[0] + dy * cam.up[1] + dz * cam.up[2]; + real_t fx, fy; + if (cam.orthographic) { + fx = cx / cam.half_w; + fy = cy / cam.half_h; + } else { + const real_t cz = dx * cam.forward[0] + dy * cam.forward[1] + + dz * cam.forward[2]; + if (cz <= static_cast(1e-6)) { + return false; + } + fx = (cx / cz) / (cam.aspect * cam.tan_half_fov); + fy = (cy / cz) / cam.tan_half_fov; + } + outx = (fx + ONE) * HALF * static_cast(W) - HALF; + outy = (ONE - fy) * HALF * static_cast(H) - HALF; + return true; + } + + /** + * @brief Screen-space bounding box (in pixels) of a world-space AABB. + * @param lo,hi world AABB corners + * @param[out] bx0,by0,bw,bh clamped pixel bbox (top-left + size) + * @return false if the box projects to an empty on-screen region + * @note Falls back to the full frame if any corner is behind the camera. + */ + inline auto screenBBox(const CameraDevice& cam, + int W, + int H, + const real_t lo[3], + const real_t hi[3], + int& bx0, + int& by0, + int& bw, + int& bh) -> bool { + real_t minx = static_cast(1e30), miny = static_cast(1e30); + real_t maxx = static_cast(-1e30), maxy = static_cast(-1e30); + for (int c = 0; c < 8; ++c) { + const real_t p[3] = { (c & 1) ? hi[0] : lo[0], + (c & 2) ? hi[1] : lo[1], + (c & 4) ? hi[2] : lo[2] }; + real_t sx, sy; + if (not projectToScreen(cam, W, H, p, sx, sy)) { + bx0 = 0; + by0 = 0; + bw = W; + bh = H; + return true; // conservative fallback + } + minx = std::min(minx, sx); + maxx = std::max(maxx, sx); + miny = std::min(miny, sy); + maxy = std::max(maxy, sy); + } + const int pad = 2; + int x0 = static_cast(std::floor(minx)) - pad; + int x1 = static_cast(std::ceil(maxx)) + pad; + int y0 = static_cast(std::floor(miny)) - pad; + int y1 = static_cast(std::ceil(maxy)) + pad; + x0 = std::max(0, std::min(W, x0)); + x1 = std::max(0, std::min(W, x1)); + y0 = std::max(0, std::min(H, y0)); + y1 = std::max(0, std::min(H, y1)); + bx0 = x0; + by0 = y0; + bw = x1 - x0; + bh = y1 - y0; + return (bw > 0 and bh > 0); + } + + /** + * @brief Composite two sparse sub-images: `front` OVER `back`. + * @return a sub-image spanning the union of the two bounding boxes + * @note premultiplied "over": out = front + (1 - front.a) * back. Associative, + * so a tree of these reproduces the sequential front-to-back composite. + */ + inline auto overSub(const SubImage& f, const SubImage& b) -> SubImage { + if (f.w == 0 or f.h == 0) { + return b; + } + if (b.w == 0 or b.h == 0) { + return f; + } + const int ux0 = std::min(f.x0, b.x0); + const int uy0 = std::min(f.y0, b.y0); + const int ux1 = std::max(f.x0 + f.w, b.x0 + b.w); + const int uy1 = std::max(f.y0 + f.h, b.y0 + b.h); + SubImage r; + r.x0 = ux0; + r.y0 = uy0; + r.w = ux1 - ux0; + r.h = uy1 - uy0; + r.rgba.assign(static_cast(r.w) * r.h * 4, ZERO); + // place `back` + for (int y = 0; y < b.h; ++y) { + for (int x = 0; x < b.w; ++x) { + const std::size_t ri = (static_cast(b.y0 + y - uy0) * r.w + + (b.x0 + x - ux0)) * + 4; + const std::size_t bi = (static_cast(y) * b.w + x) * 4; + r.rgba[ri + 0] = b.rgba[bi + 0]; + r.rgba[ri + 1] = b.rgba[bi + 1]; + r.rgba[ri + 2] = b.rgba[bi + 2]; + r.rgba[ri + 3] = b.rgba[bi + 3]; + } + } + // `front` OVER the (back-filled) result + for (int y = 0; y < f.h; ++y) { + for (int x = 0; x < f.w; ++x) { + const std::size_t ri = (static_cast(f.y0 + y - uy0) * r.w + + (f.x0 + x - ux0)) * + 4; + const std::size_t fi = (static_cast(y) * f.w + x) * 4; + const real_t inv = ONE - f.rgba[fi + 3]; + r.rgba[ri + 0] = f.rgba[fi + 0] + inv * r.rgba[ri + 0]; + r.rgba[ri + 1] = f.rgba[fi + 1] + inv * r.rgba[ri + 1]; + r.rgba[ri + 2] = f.rgba[fi + 2] + inv * r.rgba[ri + 2]; + r.rgba[ri + 3] = f.rgba[fi + 3] + inv * r.rgba[ri + 3]; + } + } + return r; + } + } // namespace out #endif // OUTPUT_RENDER_COMPOSITE_H diff --git a/src/output/render/raymarch.hpp b/src/output/render/raymarch.hpp index c2403bc69..f10702b47 100644 --- a/src/output/render/raymarch.hpp +++ b/src/output/render/raymarch.hpp @@ -49,7 +49,8 @@ namespace kernel { const real_t lo0, lo1, lo2, hi0, hi1, hi2; const int ext0, ext1, ext2; - const int W, H; + const int W, H; // full frame size (for ray generation / ndc) + const int bx0, by0, bw; // screen-bbox offset and width (output stride) const real_t ds; // fixed world step (global, identical on all ranks) const int max_steps; // safety cap on the marching loop @@ -60,7 +61,7 @@ namespace kernel { const bool log_scale; const real_t early_alpha; - array_t image; // output, (W*H, 4) premultiplied RGBA + array_t image; // output, (bw*bh, 4) premultiplied RGBA public: VolumeRayMarch_kernel(const randacc_ndfield_t& Fld_, @@ -74,6 +75,9 @@ namespace kernel { int ext2_, int W_, int H_, + int bx0_, + int by0_, + int bw_, real_t ds_, int max_steps_, const array_t& lut_, @@ -98,6 +102,9 @@ namespace kernel { , ext2 { ext2_ } , W { W_ } , H { H_ } + , bx0 { bx0_ } + , by0 { by0_ } + , bw { bw_ } , ds { ds_ } , max_steps { max_steps_ } , lut { lut_ } @@ -146,9 +153,13 @@ namespace kernel { return c0 * (ONE - t2) + c1 * t2; } - Inline void operator()(cellidx_t px, cellidx_t py) const { - const auto pix = static_cast(py) * static_cast(W) + - static_cast(px); + Inline void operator()(cellidx_t lpx, cellidx_t lpy) const { + // local bbox index -> output pixel; global pixel -> ray generation + const auto pix = static_cast(lpy) * + static_cast(bw) + + static_cast(lpx); + const int gpx = bx0 + static_cast(lpx); + const int gpy = by0 + static_cast(lpy); // default transparent image(pix, 0) = ZERO; image(pix, 1) = ZERO; @@ -156,9 +167,9 @@ namespace kernel { image(pix, 3) = ZERO; // ---- ray generation ------------------------------------------------ // - const real_t fx = TWO * (static_cast(px) + HALF) / + const real_t fx = TWO * (static_cast(gpx) + HALF) / static_cast(W) - ONE; - const real_t fy = ONE - TWO * (static_cast(py) + HALF) / + const real_t fy = ONE - TWO * (static_cast(gpy) + HALF) / static_cast(H); real_t ox, oy, oz, dx, dy, dz; if (cam.orthographic) { diff --git a/src/output/render/renderer.cpp b/src/output/render/renderer.cpp index c28fed054..051f991f0 100644 --- a/src/output/render/renderer.cpp +++ b/src/output/render/renderer.cpp @@ -258,14 +258,35 @@ namespace out { logger::Checkpoint("Volume renderer initialized", HERE); } - void Renderer::compositeAndWrite(const std::vector& rgba, - uint64_t order_key, - const Scene& scene, - timestep_t step) const { + void Renderer::compositeAndWrite(const SubImage& sub, + uint64_t order_key, + const Scene& scene, + timestep_t step) const { const std::size_t npix = static_cast(m_width) * static_cast(m_height); const std::size_t n = npix * 4; + // expand a sparse sub-image into a full transparent frame (premultiplied) + auto subToFull = [&](const SubImage& s) -> std::vector { + std::vector full(n, ZERO); + for (int y = 0; y < s.h; ++y) { + for (int x = 0; x < s.w; ++x) { + const int fx = s.x0 + x; + const int fy = s.y0 + y; + if (fx < 0 or fx >= m_width or fy < 0 or fy >= m_height) { + continue; + } + const std::size_t fi = (static_cast(fy) * m_width + fx) * 4; + const std::size_t si = (static_cast(y) * s.w + x) * 4; + full[fi + 0] = s.rgba[si + 0]; + full[fi + 1] = s.rgba[si + 1]; + full[fi + 2] = s.rgba[si + 2]; + full[fi + 3] = s.rgba[si + 3]; + } + } + return full; + }; + auto write_image = [&](const std::vector& img) { // ensure /renders/ exists const auto dir = m_root / path_t("renders"); @@ -351,57 +372,103 @@ namespace out { MPI_Comm_size(MPI_COMM_WORLD, &size); if (size == 1) { - write_image(rgba); + write_image(subToFull(sub)); return; } - std::vector recv; - std::vector keys; - if (rank == MPI_ROOT_RANK) { - recv.resize(static_cast(size) * n); - keys.resize(static_cast(size)); - } - const unsigned long long my_key = static_cast(order_key); - - MPI_Gather(rgba.data(), - static_cast(n), - mpi::get_type(), - (rank == MPI_ROOT_RANK) ? recv.data() : nullptr, - static_cast(n), - mpi::get_type(), - MPI_ROOT_RANK, - MPI_COMM_WORLD); - MPI_Gather(&my_key, - 1, - MPI_UNSIGNED_LONG_LONG, - (rank == MPI_ROOT_RANK) ? keys.data() : nullptr, - 1, - MPI_UNSIGNED_LONG_LONG, - MPI_ROOT_RANK, - MPI_COMM_WORLD); - - if (rank != MPI_ROOT_RANK) { - return; - } + constexpr int TAG_HDR = 7301; + constexpr int TAG_DATA = 7302; + + auto sendSub = [&](const SubImage& s, int dest) { + int hdr[4] = { s.x0, s.y0, s.w, s.h }; + MPI_Send(hdr, 4, MPI_INT, dest, TAG_HDR, MPI_COMM_WORLD); + const int cnt = s.w * s.h * 4; + if (cnt > 0) { + MPI_Send(s.rgba.data(), + cnt, + mpi::get_type(), + dest, + TAG_DATA, + MPI_COMM_WORLD); + } + }; + auto recvSub = [&](int src) -> SubImage { + int hdr[4]; + MPI_Recv(hdr, 4, MPI_INT, src, TAG_HDR, MPI_COMM_WORLD, MPI_STATUS_IGNORE); + SubImage s; + s.x0 = hdr[0]; + s.y0 = hdr[1]; + s.w = hdr[2]; + s.h = hdr[3]; + const int cnt = s.w * s.h * 4; + if (cnt > 0) { + s.rgba.resize(static_cast(cnt)); + MPI_Recv(s.rgba.data(), + cnt, + mpi::get_type(), + src, + TAG_DATA, + MPI_COMM_WORLD, + MPI_STATUS_IGNORE); + } + return s; + }; - // front-to-back order = ranks sorted by ascending composite key - std::vector order(size); + // Every rank learns the full key vector (one uint64 each: ~tiny) and + // derives the same global front-to-back order, so no rank needs the others' + // images to agree on the composite order. + const unsigned long long my_key = static_cast(order_key); + std::vector keys(static_cast(size)); + MPI_Allgather(&my_key, + 1, + MPI_UNSIGNED_LONG_LONG, + keys.data(), + 1, + MPI_UNSIGNED_LONG_LONG, + MPI_COMM_WORLD); + std::vector order(size); // order[position] = world rank, front-to-back std::iota(order.begin(), order.end(), 0); std::stable_sort(order.begin(), order.end(), [&](int a, int b) { return keys[a] < keys[b]; }); + std::vector pos(size); // pos[world rank] = front-to-back position + for (int i = 0; i < size; ++i) { + pos[order[i]] = i; + } - std::vector acc(n, ZERO); - for (const int r : order) { - const real_t* seg_base = recv.data() + static_cast(r) * n; - for (std::size_t p = 0; p < npix; ++p) { - overComposite(acc.data() + p * 4, seg_base + p * 4); + // Order-preserving binary tree reduction over positions. At level `s`, the + // front of each pair (lower position) receives the back partner's image and + // composites front OVER back; the back partner sends and drops out. "over" + // is associative, so this reproduces the sequential front-to-back composite + // in O(log nranks) rounds with no single-rank bottleneck. + SubImage cur = sub; + const int P = pos[rank]; + for (int s = 1; s < size; s <<= 1) { + if ((P % (2 * s)) == 0) { + const int pp = P + s; + if (pp < size) { + const SubImage back = recvSub(order[pp]); + cur = overSub(cur, back); // cur is the front + } + } else if ((P % (2 * s)) == s) { + sendSub(cur, order[P - s]); + break; // absorbed into the front partner + } + } + + // The fully composited image now lives at position 0; deliver it to root. + if (rank == order[0]) { + if (rank == MPI_ROOT_RANK) { + write_image(subToFull(cur)); + } else { + sendSub(cur, MPI_ROOT_RANK); } + } else if (rank == MPI_ROOT_RANK) { + write_image(subToFull(recvSub(order[0]))); } - write_image(acc); #else (void)order_key; - write_image(rgba); + write_image(subToFull(sub)); #endif } diff --git a/src/output/render/renderer.h b/src/output/render/renderer.h index 88294183d..91f3d67dd 100644 --- a/src/output/render/renderer.h +++ b/src/output/render/renderer.h @@ -75,6 +75,19 @@ namespace out { TransferFunction tf; }; + /** + * @brief A sparse screen-space sub-image: the bounding box of one domain's + * projected footprint plus its premultiplied RGBA pixels. + * @note Each domain covers only a small part of the screen, so compositing + * these sparse boxes (not full frames) is what lets the renderer scale to + * thousands of ranks. + */ + struct SubImage { + int x0 { 0 }, y0 { 0 }; // top-left pixel in the full frame + int w { 0 }, h { 0 }; // bbox size in pixels (0 => empty) + std::vector rgba; // w*h*4 premultiplied, pixel-major + }; + class Renderer { public: Renderer() {} @@ -97,18 +110,18 @@ namespace out { } /** - * @brief Composite the per-rank host image across MPI and write the PNG. - * @param rgba host buffer, length width*height*4, premultiplied RGBA, in - * pixel-major / channel-minor order (rgba[pix*4 + ch]) + * @brief Composite the per-rank sparse sub-image across MPI and write PNG. + * @param sub this rank's sparse screen-space sub-image (premultiplied RGBA) * @param order_key this rank's front-to-back sort key (see composite.h) * @param scene the scene being written (prefix, colorbar colormap/range/label) * @param step current timestep (for the filename cycle number) - * @note Only the MPI root rank writes the file. + * @note Uses an order-preserving distributed tree reduce; only the MPI root + * rank assembles the full frame and writes the file. */ - void compositeAndWrite(const std::vector& rgba, - uint64_t order_key, - const Scene& scene, - timestep_t step) const; + void compositeAndWrite(const SubImage& sub, + uint64_t order_key, + const Scene& scene, + timestep_t step) const; /* getters -------------------------------------------------------------- */ [[nodiscard]] From ec55fcf5044d35331af4a29590a94f4b4ed28d35 Mon Sep 17 00:00:00 2001 From: Ludwig Boess Date: Tue, 30 Jun 2026 13:35:28 -0400 Subject: [PATCH 039/125] reduce send/recv via uint8 --- src/output/render/renderer.cpp | 26 +++++++++++++++++--------- 1 file changed, 17 insertions(+), 9 deletions(-) diff --git a/src/output/render/renderer.cpp b/src/output/render/renderer.cpp index 051f991f0..5b66f6471 100644 --- a/src/output/render/renderer.cpp +++ b/src/output/render/renderer.cpp @@ -379,17 +379,20 @@ namespace out { constexpr int TAG_HDR = 7301; constexpr int TAG_DATA = 7302; + // Wire format is premultiplied uint8 RGBA (4x less bandwidth than float). + // Compositing stays in float; only the per-message quantization adds error + // (~1 LSB through the log(N)-deep tree), so fidelity is effectively that of + // the final 8-bit PNG. auto sendSub = [&](const SubImage& s, int dest) { int hdr[4] = { s.x0, s.y0, s.w, s.h }; MPI_Send(hdr, 4, MPI_INT, dest, TAG_HDR, MPI_COMM_WORLD); const int cnt = s.w * s.h * 4; if (cnt > 0) { - MPI_Send(s.rgba.data(), - cnt, - mpi::get_type(), - dest, - TAG_DATA, - MPI_COMM_WORLD); + std::vector bytes(static_cast(cnt)); + for (int i = 0; i < cnt; ++i) { + bytes[i] = quantize(s.rgba[i]); + } + MPI_Send(bytes.data(), cnt, MPI_UNSIGNED_CHAR, dest, TAG_DATA, MPI_COMM_WORLD); } }; auto recvSub = [&](int src) -> SubImage { @@ -402,14 +405,19 @@ namespace out { s.h = hdr[3]; const int cnt = s.w * s.h * 4; if (cnt > 0) { - s.rgba.resize(static_cast(cnt)); - MPI_Recv(s.rgba.data(), + std::vector bytes(static_cast(cnt)); + MPI_Recv(bytes.data(), cnt, - mpi::get_type(), + MPI_UNSIGNED_CHAR, src, TAG_DATA, MPI_COMM_WORLD, MPI_STATUS_IGNORE); + s.rgba.resize(static_cast(cnt)); + const real_t inv255 = ONE / static_cast(255); + for (int i = 0; i < cnt; ++i) { + s.rgba[i] = static_cast(bytes[i]) * inv255; + } } return s; }; From 6dd8117fcacd68eed5dee6b0578365a333741ecb Mon Sep 17 00:00:00 2001 From: Ludwig Boess Date: Tue, 30 Jun 2026 13:56:52 -0400 Subject: [PATCH 040/125] added option to define colorbar ticks as parameter --- src/output/render/colorbar.h | 62 ++++++++++++++++++++++++---------- src/output/render/renderer.cpp | 9 +++-- src/output/render/renderer.h | 9 ++--- 3 files changed, 56 insertions(+), 24 deletions(-) diff --git a/src/output/render/colorbar.h b/src/output/render/colorbar.h index 4c311bd90..3c208b850 100644 --- a/src/output/render/colorbar.h +++ b/src/output/render/colorbar.h @@ -26,6 +26,8 @@ #include #include #include +#include +#include namespace out { @@ -162,16 +164,19 @@ namespace out { * @param log_scale if true, ticks are spaced/labelled logarithmically * @param label title drawn above the bar (e.g. the field name) * @param bg background RGB (to auto-pick contrasting text color) + * @param ticks explicit tick values to label; if empty, 5 evenly-spaced + * ticks are generated. Values outside [vmin, vmax] are skipped. */ - inline void drawColorbar(uint8_t* rgba, - int W, - int H, - const std::string& colormap, - real_t vmin, - real_t vmax, - bool log_scale, - const std::string& label, - const real_t bg[3]) { + inline void drawColorbar(uint8_t* rgba, + int W, + int H, + const std::string& colormap, + real_t vmin, + real_t vmax, + bool log_scale, + const std::string& label, + const real_t bg[3], + const std::vector& ticks = {}) { using namespace cbar_hidden; const int s = scale(H); @@ -221,18 +226,39 @@ namespace out { } } - // ticks + labels + // ticks + labels: explicit values if given, else 5 evenly-spaced const bool can_log = log_scale and vmin > ZERO and vmax > ZERO; const real_t lvmin = can_log ? math::log10(vmin) : ZERO; const real_t lvmax = can_log ? math::log10(vmax) : ZERO; - const int nticks = 5; - for (int t = 0; t < nticks; ++t) { - const real_t u = static_cast(t) / - static_cast(nticks - 1); - const real_t val = can_log - ? math::pow(static_cast(10), - lvmin + (lvmax - lvmin) * u) - : (vmin + (vmax - vmin) * u); + std::vector> tk; // (u in [0,1], value) + if (ticks.empty()) { + const int nticks = 5; + for (int t = 0; t < nticks; ++t) { + const real_t u = static_cast(t) / + static_cast(nticks - 1); + const real_t val = can_log + ? math::pow(static_cast(10), + lvmin + (lvmax - lvmin) * u) + : (vmin + (vmax - vmin) * u); + tk.emplace_back(u, val); + } + } else { + const real_t span = can_log ? (lvmax - lvmin) : (vmax - vmin); + for (const real_t v : ticks) { + if (can_log and v <= ZERO) { + continue; + } + const real_t u = (span != ZERO) + ? ((can_log ? (math::log10(v) - lvmin) : (v - vmin)) / + span) + : ZERO; + if (u < static_cast(-1e-4) or u > ONE + static_cast(1e-4)) { + continue; // outside the colorbar range + } + tk.emplace_back(std::min(ONE, std::max(ZERO, u)), v); + } + } + for (const auto& [u, val] : tk) { const int ty = bar_y + static_cast((ONE - u) * static_cast(bar_h - 1)); diff --git a/src/output/render/renderer.cpp b/src/output/render/renderer.cpp index 5b66f6471..7c33fc2c5 100644 --- a/src/output/render/renderer.cpp +++ b/src/output/render/renderer.cpp @@ -228,6 +228,9 @@ namespace out { continue; } scene.label = toml::find_or(sc, "label", scene.field); + scene.ticks = toml::find_or>(sc, + "colorbar_ticks", + std::vector {}); scene.tf.vmin = toml::find_or(sc, "min", ZERO); scene.tf.vmax = toml::find_or(sc, "max", ONE); scene.tf.log_scale = toml::find_or(sc, "log", false); @@ -343,7 +346,8 @@ namespace out { scene.tf.vmax, scene.tf.log_scale, scene.label, - m_background); + m_background, + scene.ticks); ok = write_png(fname, CW, m_height, canvas.data()); } else { if (m_colorbar) { @@ -355,7 +359,8 @@ namespace out { scene.tf.vmax, scene.tf.log_scale, scene.label, - m_background); + m_background, + scene.ticks); } ok = write_png(fname, m_width, m_height, bytes.data()); } diff --git a/src/output/render/renderer.h b/src/output/render/renderer.h index 91f3d67dd..911847c47 100644 --- a/src/output/render/renderer.h +++ b/src/output/render/renderer.h @@ -69,10 +69,11 @@ namespace out { * @brief One rendered scalar field -> one PNG stream. */ struct Scene { - std::string field; // "N" | "Bmag" | "Jmag" | "smooth_xyz" - std::string prefix; // PNG filename prefix, e.g. "Bmag_" - std::string label; // colorbar title (defaults to field) - TransferFunction tf; + std::string field; // "N" | "Bmag" | "Jmag" | "smooth_xyz" + std::string prefix; // PNG filename prefix, e.g. "Bmag_" + std::string label; // colorbar title (defaults to field) + std::vector ticks; // explicit colorbar tick values (optional) + TransferFunction tf; }; /** From a84c821065af6547561e6b694de2ddb5efb9d4cf Mon Sep 17 00:00:00 2001 From: Ludwig Boess Date: Tue, 30 Jun 2026 14:05:41 -0400 Subject: [PATCH 041/125] generalize field rendering --- input.example.toml | 152 +++++++++++++++++++++ src/framework/domain/metadomain_render.cpp | 117 ++++++++++------ 2 files changed, 227 insertions(+), 42 deletions(-) diff --git a/input.example.toml b/input.example.toml index 8d6f25117..d92c3c212 100644 --- a/input.example.toml +++ b/input.example.toml @@ -681,6 +681,158 @@ # @default: [] custom = "" + # In-situ volume renderer. Ray-marches scalar fields on the GPU and writes + # PNG images directly to `/renders/` each cadence -- no field data is + # written to storage, and the result is seamless across MPI domain boundaries. + # @note: 3D Cartesian (Minkowski) only; a no-op for other dims/metrics + # @note: One PNG stream per scene (e.g. a density/|B|/|J| triptych) + [output.render] + # Toggle for the volume renderer + # @type: bool + # @default: false + enable = "" + # Number of timesteps between renders + # @type: uint + # @default: 0 + # @note: When `!= 0`, overrides `output.interval` + # @note: When `== 0`, `interval_time` (or `output.interval`) is used + interval = "" + # Physical (code) time interval between renders + # @type: float + # @default: -1.0 + # @note: When `< 0`, the output is controlled by `interval` + interval_time = "" + # Image width in pixels (the rendered region; the PNG is wider if a colorbar + # margin is added, see `colorbar_outside`) + # @type: int [> 0] + # @default: 1024 + width = "" + # Image height in pixels + # @type: int [> 0] + # @default: 1024 + height = "" + # Number of ray-march steps across the global box diagonal + # @type: int [> 0] + # @default: 400 + # @note: The world-space step is `box_diagonal / samples` unless `step_size` + # is set. Higher = better quality, slower. + samples = "" + # Fixed world-space step between ray samples + # @type: float [>= 0.0] + # @default: 0.0 + # @note: 0 derives the step from `samples`. The step is identical on all + # ranks, which is what makes the multi-domain composite seamless. + step_size = "" + # Stop marching a ray once its accumulated opacity reaches this value + # @type: float [0.0 -> 1.0] + # @default: 0.99 + # @note: Pure speed optimization; set to 1.0 to disable early termination + early_term_alpha = "" + # Number of entries in the color/opacity lookup table + # @type: int [> 1] + # @default: 256 + n_lut = "" + # Opaque background RGB (each channel 0..1) shown through transparent/low- + # opacity pixels; also fills the colorbar margin + # @type: array [size 3] + # @default: [0.0, 0.0, 0.0] + background = "" + # Draw a colorbar (gradient + value ticks + label) on each PNG + # @type: bool + # @default: true + colorbar = "" + # Draw the colorbar in an added right margin (the PNG becomes wider by a + # fixed strip) instead of overlaying it on the rendered volume + # @type: bool + # @default: true + colorbar_outside = "" + + # Camera. Defaults frame the whole global box from outside, looking down the + # (1,1,1) diagonal -- the production setup for which the structured composite + # is provably seamless. + [output.render.camera] + # Orthographic (true) or perspective (false) projection + # @type: bool + # @default: true + # @note: Orthographic is recommended; the seamless composite is always + # valid for it. Perspective is only seamless with the eye outside + # the box. + orthographic = "" + # Camera (eye) position in world (physical) coordinates + # @type: array [size 3] + # @default: box center pushed back ~1.7 box-diagonals along (1, 1, 1) + position = "" + # Point the camera looks at, in world coordinates + # @type: array [size 3] + # @default: box center + look_at = "" + # Camera up vector + # @type: array [size 3] + # @default: [0.0, 0.0, 1.0] + up = "" + # Vertical field of view in degrees (perspective only) + # @type: float [> 0.0] + # @default: 35.0 + fov = "" + # Vertical extent of the view in world units (orthographic only) + # @type: float [> 0.0] + # @default: the global box diagonal (the whole box fits from any angle) + ortho_height = "" + + # One scene per scalar field -> one PNG stream. Repeat the table for each. + [[output.render.scenes]] + # Scalar field to render (a volume render needs a scalar, so a vector is + # given as a magnitude or a single component) + # @required + # @type: string + # @enum: "N"; "Emag"/"Bmag"/"Jmag"; "{E,B,J}{1,2,3}" or "{E,B,J}{x,y,z}" + # @note: "N" = number density; "{E,B,J}mag" = vector magnitude |.| + # @note: "B1"/"Bx", "J3"/"Jz", ... = a single (signed) physical component + # @note: a bare vector ("E"/"B"/"J") is not renderable -- choose a + # component or the magnitude + # @note: components are signed; pair a symmetric `min`/`max` with a + # diverging colormap ("cool2warm") to center zero + field = "" + # PNG filename prefix; files are `.png` + # @type: string + # @default: "_" + prefix = "" + # Colorbar title + # @type: string + # @default: `field` + label = "" + # Lower bound of the value range mapped onto the colormap/opacity + # @type: float + # @default: 0.0 + min = "" + # Upper bound of the value range + # @type: float + # @default: 1.0 + max = "" + # Map the value range logarithmically + # @type: bool + # @default: false + # @note: Requires min > 0 and max > 0 + log = "" + # Colormap name + # @type: string + # @enum: "viridis", "inferno", "plasma", "cool2warm", "gray" + # @default: "viridis" + colormap = "" + # Opacity transfer function: [position, opacity] control points, both in + # [0, 1], piecewise-linear in the normalized value + # @type: array> + # @default: linear ramp (opacity = normalized value) + # @note: Keep the low end near 0 so empty regions stay transparent + # @example: [[0.0, 0.0], [0.3, 0.1], [1.0, 0.7]] + alpha = "" + # Explicit value(s) to label on the colorbar + # @type: array + # @default: 5 evenly-spaced ticks between min and max + # @note: Values outside [min, max] are skipped + # @example: [0.0, 0.5, 1.0] + colorbar_ticks = "" + [checkpoint] # Number of timesteps between checkpoints # @type: uint [> 0] diff --git a/src/framework/domain/metadomain_render.cpp b/src/framework/domain/metadomain_render.cpp index 013b855a4..ffdd51ba1 100644 --- a/src/framework/domain/metadomain_render.cpp +++ b/src/framework/domain/metadomain_render.cpp @@ -41,6 +41,7 @@ #include #include +#include #include #include #include @@ -174,9 +175,53 @@ namespace ntt { // sum boundary-crossing particle deposits back into active cells // (particles in neighbor domains deposit into our ghost zone) SynchronizeFields(*local_domain, Comm::Bckp, { 0, 1 }); - } else if (scene.field == "Bmag" or scene.field == "Jmag") { - const bool is_current = (scene.field == "Jmag"); - // raw vector components into bckp(:,0..2) + } else { + // Vector field as a scalar: "" with base in {E, B, J} + // and selector in {mag, 1/2/3, x/y/z}. A component (e.g. "B1"/"Bx") is + // signed; a magnitude (e.g. "Bmag") is non-negative. + const std::string& f = scene.field; + const char base = f.empty() + ? '?' + : static_cast(std::toupper(f[0])); + bool ok = true; + bool is_current = false; + uint8_t src_base = 0; // first component of the source field + PrepareOutputFlags interp = PrepareOutput::None; + if (base == 'B') { + src_base = em::bx1; + interp = PrepareOutput::InterpToCellCenterFromFaces; + } else if (base == 'E') { + src_base = em::ex1; + interp = PrepareOutput::InterpToCellCenterFromEdges; + } else if (base == 'J') { + is_current = true; + src_base = cur::jx1; + interp = PrepareOutput::InterpToCellCenterFromEdges; + } else { + ok = false; + } + // selector: -1 = magnitude, 0/1/2 = a single component + int comp = -2; + const std::string sel = (f.size() > 1) ? f.substr(1) : std::string {}; + if (sel == "mag") { + comp = -1; + } else if (sel == "1" or sel == "x") { + comp = 0; + } else if (sel == "2" or sel == "y") { + comp = 1; + } else if (sel == "3" or sel == "z") { + comp = 2; + } else { + ok = false; + } + if (not ok) { + raise::Warning("output.render: unknown field '" + scene.field + + "' (expected N, {E,B,J}mag, or " + "{E,B,J}{1,2,3}|{x,y,z}); skipping", + HERE); + continue; + } + // raw vector components into bckp(:, 0..2) if (is_current) { Kokkos::deep_copy( Kokkos::subview(bckp, Kokkos::ALL, Kokkos::ALL, Kokkos::ALL, @@ -188,17 +233,14 @@ namespace ntt { Kokkos::subview(bckp, Kokkos::ALL, Kokkos::ALL, Kokkos::ALL, cell_range_t(0, 3)), Kokkos::subview(local_domain->fields.em, Kokkos::ALL, Kokkos::ALL, - Kokkos::ALL, cell_range_t(em::bx1, em::bx3 + 1))); + Kokkos::ALL, cell_range_t(src_base, src_base + 3))); } // interpolate to cell centers + convert to physical basis -> (3,4,5) - PrepareOutputFlags interp = is_current - ? PrepareOutput::InterpToCellCenterFromEdges - : PrepareOutput::InterpToCellCenterFromFaces; - PrepareOutputFlags prepare = (S == SimEngine::SRPIC) - ? PrepareOutput::ConvertToHat - : PrepareOutput::ConvertToPhysCntrv; - list_t comp_from = { 0, 1, 2 }; - list_t comp_to = { 3, 4, 5 }; + const PrepareOutputFlags prepare = (S == SimEngine::SRPIC) + ? PrepareOutput::ConvertToHat + : PrepareOutput::ConvertToPhysCntrv; + list_t comp_from = { 0, 1, 2 }; + list_t comp_to = { 3, 4, 5 }; Kokkos::parallel_for( "RenderFieldsToPhys", local_domain->mesh.rangeActiveCells(), @@ -208,36 +250,27 @@ namespace ntt { comp_to, interp | prepare, metric)); - // magnitude -> bckp(:,0) - auto bckp_v = bckp; - Kokkos::parallel_for( - "RenderVectorMagnitude", - local_domain->mesh.rangeActiveCells(), - Lambda(cellidx_t i1, cellidx_t i2, cellidx_t i3) { - const real_t v1 = bckp_v(i1, i2, i3, 3); - const real_t v2 = bckp_v(i1, i2, i3, 4); - const real_t v3 = bckp_v(i1, i2, i3, 5); - bckp_v(i1, i2, i3, 0) = math::sqrt(v1 * v1 + v2 * v2 + v3 * v3); - }); - } else if (scene.field == "smooth_xyz") { - // continuous-by-construction regression field: x + y + z - auto bckp_v = bckp; - Kokkos::parallel_for( - "RenderSmoothXYZ", - local_domain->mesh.rangeActiveCells(), - Lambda(cellidx_t i1, cellidx_t i2, cellidx_t i3) { - coord_t x_Cd { ZERO }, x_Ph { ZERO }; - x_Cd[0] = COORD(i1) + HALF; - x_Cd[1] = COORD(i2) + HALF; - x_Cd[2] = COORD(i3) + HALF; - metric.template convert(x_Cd, x_Ph); - bckp_v(i1, i2, i3, 0) = x_Ph[0] + x_Ph[1] + x_Ph[2]; - }); - } else { - raise::Warning( - "output.render: unknown field '" + scene.field + "', skipping", - HERE); - continue; + // reduce to the scalar to render -> bckp(:, 0) + auto bckp_v = bckp; + const int cc = comp; + if (comp == -1) { + Kokkos::parallel_for( + "RenderVectorMagnitude", + local_domain->mesh.rangeActiveCells(), + Lambda(cellidx_t i1, cellidx_t i2, cellidx_t i3) { + const real_t v1 = bckp_v(i1, i2, i3, 3); + const real_t v2 = bckp_v(i1, i2, i3, 4); + const real_t v3 = bckp_v(i1, i2, i3, 5); + bckp_v(i1, i2, i3, 0) = math::sqrt(v1 * v1 + v2 * v2 + v3 * v3); + }); + } else { + Kokkos::parallel_for( + "RenderVectorComponent", + local_domain->mesh.rangeActiveCells(), + Lambda(cellidx_t i1, cellidx_t i2, cellidx_t i3) { + bckp_v(i1, i2, i3, 0) = bckp_v(i1, i2, i3, 3 + cc); + }); + } } // fill the ghost halo with neighbor active values so trilinear From 36ce55cb1971a7df24dabd0e8a54c607174f55d6 Mon Sep 17 00:00:00 2001 From: Ludwig Boess Date: Tue, 30 Jun 2026 14:20:42 -0400 Subject: [PATCH 042/125] generalized field rendering --- input.example.toml | 16 +- src/framework/domain/metadomain_render.cpp | 172 ++++++++++++++++++--- 2 files changed, 162 insertions(+), 26 deletions(-) diff --git a/input.example.toml b/input.example.toml index d92c3c212..b9c3c070a 100644 --- a/input.example.toml +++ b/input.example.toml @@ -781,15 +781,23 @@ # One scene per scalar field -> one PNG stream. Repeat the table for each. [[output.render.scenes]] - # Scalar field to render (a volume render needs a scalar, so a vector is + # Scalar field to render (a volume render needs a scalar, so vectors are # given as a magnitude or a single component) # @required # @type: string - # @enum: "N"; "Emag"/"Bmag"/"Jmag"; "{E,B,J}{1,2,3}" or "{E,B,J}{x,y,z}" - # @note: "N" = number density; "{E,B,J}mag" = vector magnitude |.| - # @note: "B1"/"Bx", "J3"/"Jz", ... = a single (signed) physical component + # @enum (fields): "{E,B,J}mag"; "{E,B,J}{1,2,3}" or "{E,B,J}{x,y,z}" + # @enum (moments): "N", "Nppc", "Rho", "Charge"; "T{i}{j}"; "V{i}" + # @note: "{E,B,J}mag" = vector magnitude |.|; "B1"/"Bx", "J3"/"Jz", ... = + # a single (signed) physical component # @note: a bare vector ("E"/"B"/"J") is not renderable -- choose a # component or the magnitude + # @note: "N"/"Nppc" = number / per-cell count, "Rho" = mass density, + # "Charge" = charge density + # @note: "T{i}{j}" = one stress-energy component, i,j in {t,x,y,z} or + # {0,1,2,3} (e.g. "Txx", "Ttt", "T0x"); "V{i}" = one bulk-velocity + # component, i in {x,y,z} or {1,2,3} (e.g. "Vx", "V1") + # @note: per-species selection with a "_" suffix on moments, e.g. + # "N_1", "Rho_2", "Txy_1_2", "V1_3"; default = all massive species # @note: components are signed; pair a symmetric `min`/`max` with a # diverging colormap ("cool2warm") to center zero field = "" diff --git a/src/framework/domain/metadomain_render.cpp b/src/framework/domain/metadomain_render.cpp index ffdd51ba1..b429a88a1 100644 --- a/src/framework/domain/metadomain_render.cpp +++ b/src/framework/domain/metadomain_render.cpp @@ -56,14 +56,24 @@ namespace ntt { void renderMoment(const SimulationParams& params, const Mesh& mesh, const std::vector>& prtl_species, - ndfield_t& buffer, - idx_t buff_idx) { - std::vector specs; - for (auto& sp : prtl_species) { - if (sp.mass() > 0) { - specs.push_back(sp.index()); + const std::vector& species, + const std::vector& components, + ndfield_t& buffer, + idx_t buff_idx) { + std::vector specs = species; + if (specs.empty()) { + // default: accumulate over all massive species + for (auto& sp : prtl_species) { + if (sp.mass() > 0) { + specs.push_back(sp.index()); + } } } + for (const auto& sp : specs) { + raise::ErrorIf((sp > prtl_species.size()) or (sp == 0), + "Invalid species index " + std::to_string(sp), + HERE); + } auto scatter_buff = Kokkos::Experimental::create_scatter_view(buffer); const auto use_weights = params.get("particles.use_weights"); const auto ni2 = mesh.n_active(in::x2); @@ -72,7 +82,6 @@ namespace ntt { "output.fields.smoothing.order"); const auto smooth_method = OutputSmoothingType::from_string( params.get("output.fields.smoothing.method")); - const std::vector components {}; for (const auto& sp : specs) { auto& prtl_spec = prtl_species[sp - 1]; Kokkos::parallel_for( @@ -166,34 +175,153 @@ namespace ntt { for (const auto& scene : g_renderer.scenes()) { Kokkos::deep_copy(bckp, ZERO); - if (scene.field == "N") { - renderMoment(params, - local_domain->mesh, - local_domain->species, - bckp, - 0u); + // Parse an optional trailing per-species suffix "__..."; + // species apply to particle moments only (N, Nppc, Rho, Charge, T, V). + std::string base = scene.field; + std::vector species; + { + const auto us = scene.field.find('_'); + if (us != std::string::npos) { + bool ok_sp = true; + std::size_t start = us + 1; + while (start <= scene.field.size()) { + const auto nx = scene.field.find('_', start); + const auto tok = scene.field.substr( + start, + (nx == std::string::npos) ? std::string::npos : nx - start); + if (tok.empty() or + tok.find_first_not_of("0123456789") != std::string::npos) { + ok_sp = false; + break; + } + species.push_back(static_cast(std::stoi(tok))); + if (nx == std::string::npos) { + break; + } + start = nx + 1; + } + if (ok_sp) { + base = scene.field.substr(0, us); + } else { + species.clear(); // not a species suffix; keep the full name + } + } + } + bool bad_species = false; + for (const auto sp : species) { + if (sp == 0 or sp > local_domain->species.size()) { + bad_species = true; + } + } + if (bad_species) { + raise::Warning("output.render: invalid species in '" + scene.field + + "', skipping", + HERE); + continue; + } + + // axis/index character -> {t,x,y,z} == {0,1,2,3}; -1 if invalid + auto axisIdx = [](char ch) -> int { + switch (ch) { + case 't': + case '0': + return 0; + case 'x': + case '1': + return 1; + case 'y': + case '2': + return 2; + case 'z': + case '3': + return 3; + default: + return -1; + } + }; + + bool handled = false; + if (base == "N" or base == "Nppc" or base == "Rho" or base == "Charge") { + // scalar particle moments + if (base == "N") { + renderMoment(params, local_domain->mesh, + local_domain->species, species, {}, + bckp, 0u); + } else if (base == "Nppc") { + renderMoment(params, local_domain->mesh, + local_domain->species, species, {}, + bckp, 0u); + } else if (base == "Rho") { + renderMoment(params, local_domain->mesh, + local_domain->species, species, {}, + bckp, 0u); + } else { + renderMoment(params, local_domain->mesh, + local_domain->species, species, {}, + bckp, 0u); + } // sum boundary-crossing particle deposits back into active cells - // (particles in neighbor domains deposit into our ghost zone) SynchronizeFields(*local_domain, Comm::Bckp, { 0, 1 }); - } else { + handled = true; + } else if (base.size() == 3 and base[0] == 'T') { + // a single stress-energy tensor component "T" + const int i = axisIdx(base[1]); + const int j = axisIdx(base[2]); + if (i >= 0 and j >= 0) { + const std::vector comps { static_cast(i), + static_cast(j) }; + renderMoment(params, local_domain->mesh, + local_domain->species, species, comps, + bckp, 0u); + SynchronizeFields(*local_domain, Comm::Bckp, { 0, 1 }); + handled = true; + } + } else if (base.size() == 2 and base[0] == 'V') { + // a single bulk-velocity component "V" (spatial); normalize by Rho + const int c = axisIdx(base[1]); + if (c >= 1 and c <= 3) { + const std::vector comps { static_cast(c) }; + renderMoment(params, local_domain->mesh, + local_domain->species, species, comps, + bckp, 0u); + renderMoment(params, local_domain->mesh, + local_domain->species, species, {}, + bckp, 1u); + SynchronizeFields(*local_domain, Comm::Bckp, { 0, 2 }); + // V_c = (mass-weighted bulk velocity) / Rho + auto bckp_v = bckp; + Kokkos::parallel_for( + "RenderNormalizeV", + local_domain->mesh.rangeActiveCells(), + Lambda(cellidx_t i1, cellidx_t i2, cellidx_t i3) { + const real_t rho = bckp_v(i1, i2, i3, 1); + bckp_v(i1, i2, i3, 0) = (rho != ZERO) + ? (bckp_v(i1, i2, i3, 0) / rho) + : ZERO; + }); + handled = true; + } + } + + if (not handled) { // Vector field as a scalar: "" with base in {E, B, J} // and selector in {mag, 1/2/3, x/y/z}. A component (e.g. "B1"/"Bx") is // signed; a magnitude (e.g. "Bmag") is non-negative. - const std::string& f = scene.field; - const char base = f.empty() - ? '?' - : static_cast(std::toupper(f[0])); + const std::string& f = base; + const char fbase = f.empty() + ? '?' + : static_cast(std::toupper(f[0])); bool ok = true; bool is_current = false; uint8_t src_base = 0; // first component of the source field PrepareOutputFlags interp = PrepareOutput::None; - if (base == 'B') { + if (fbase == 'B') { src_base = em::bx1; interp = PrepareOutput::InterpToCellCenterFromFaces; - } else if (base == 'E') { + } else if (fbase == 'E') { src_base = em::ex1; interp = PrepareOutput::InterpToCellCenterFromEdges; - } else if (base == 'J') { + } else if (fbase == 'J') { is_current = true; src_base = cur::jx1; interp = PrepareOutput::InterpToCellCenterFromEdges; From 866bf5041d53ab72eb23ad17f7bd0388ea1103aa Mon Sep 17 00:00:00 2001 From: Ludwig Boess Date: Tue, 30 Jun 2026 14:29:21 -0400 Subject: [PATCH 043/125] added Vmag rendering --- input.example.toml | 5 ++-- src/framework/domain/metadomain_render.cpp | 34 ++++++++++++++++++++++ 2 files changed, 37 insertions(+), 2 deletions(-) diff --git a/input.example.toml b/input.example.toml index b9c3c070a..08e13aa97 100644 --- a/input.example.toml +++ b/input.example.toml @@ -786,7 +786,7 @@ # @required # @type: string # @enum (fields): "{E,B,J}mag"; "{E,B,J}{1,2,3}" or "{E,B,J}{x,y,z}" - # @enum (moments): "N", "Nppc", "Rho", "Charge"; "T{i}{j}"; "V{i}" + # @enum (moments): "N", "Nppc", "Rho", "Charge"; "T{i}{j}"; "V{i}"; "Vmag" # @note: "{E,B,J}mag" = vector magnitude |.|; "B1"/"Bx", "J3"/"Jz", ... = # a single (signed) physical component # @note: a bare vector ("E"/"B"/"J") is not renderable -- choose a @@ -795,7 +795,8 @@ # "Charge" = charge density # @note: "T{i}{j}" = one stress-energy component, i,j in {t,x,y,z} or # {0,1,2,3} (e.g. "Txx", "Ttt", "T0x"); "V{i}" = one bulk-velocity - # component, i in {x,y,z} or {1,2,3} (e.g. "Vx", "V1") + # component, i in {x,y,z} or {1,2,3} (e.g. "Vx", "V1"); "Vmag" = + # bulk-velocity magnitude sqrt(V1^2+V2^2+V3^2) # @note: per-species selection with a "_" suffix on moments, e.g. # "N_1", "Rho_2", "Txy_1_2", "V1_3"; default = all massive species # @note: components are signed; pair a symmetric `min`/`max` with a diff --git a/src/framework/domain/metadomain_render.cpp b/src/framework/domain/metadomain_render.cpp index b429a88a1..03fa80dd9 100644 --- a/src/framework/domain/metadomain_render.cpp +++ b/src/framework/domain/metadomain_render.cpp @@ -276,6 +276,40 @@ namespace ntt { SynchronizeFields(*local_domain, Comm::Bckp, { 0, 1 }); handled = true; } + } else if (base == "Vmag") { + // bulk-velocity magnitude |V| = sqrt(V1^2 + V2^2 + V3^2). Each Vi is + // the mass-weighted flux normalized by Rho, so deposit the three + // spatial components into bckp(0..2) and Rho into bckp(3), sum-sync + // all four, divide, then reduce to the Euclidean norm in bckp(0). + renderMoment(params, local_domain->mesh, + local_domain->species, species, { 1u }, + bckp, 0u); + renderMoment(params, local_domain->mesh, + local_domain->species, species, { 2u }, + bckp, 1u); + renderMoment(params, local_domain->mesh, + local_domain->species, species, { 3u }, + bckp, 2u); + renderMoment(params, local_domain->mesh, + local_domain->species, species, {}, + bckp, 3u); + SynchronizeFields(*local_domain, Comm::Bckp, { 0, 4 }); + auto bckp_v = bckp; + Kokkos::parallel_for( + "RenderNormalizeVmag", + local_domain->mesh.rangeActiveCells(), + Lambda(cellidx_t i1, cellidx_t i2, cellidx_t i3) { + const real_t rho = bckp_v(i1, i2, i3, 3); + if (rho != ZERO) { + const real_t v1 = bckp_v(i1, i2, i3, 0) / rho; + const real_t v2 = bckp_v(i1, i2, i3, 1) / rho; + const real_t v3 = bckp_v(i1, i2, i3, 2) / rho; + bckp_v(i1, i2, i3, 0) = math::sqrt(v1 * v1 + v2 * v2 + v3 * v3); + } else { + bckp_v(i1, i2, i3, 0) = ZERO; + } + }); + handled = true; } else if (base.size() == 2 and base[0] == 'V') { // a single bulk-velocity component "V" (spatial); normalize by Rho const int c = axisIdx(base[1]); From afb45f540b5110c7f8cd3c8fdd19f6d0d1e747a5 Mon Sep 17 00:00:00 2001 From: Ludwig Boess Date: Tue, 30 Jun 2026 15:42:59 -0400 Subject: [PATCH 044/125] support for 2D rendering --- input.example.toml | 33 +- src/framework/domain/metadomain.h | 11 +- src/framework/domain/metadomain_render.cpp | 772 ++++++++++++++------- src/output/render/reduce.hpp | 178 +++++ src/output/render/renderer.cpp | 24 +- src/output/render/renderer.h | 16 +- src/output/render/slice2d.hpp | 210 ++++++ 7 files changed, 963 insertions(+), 281 deletions(-) create mode 100644 src/output/render/reduce.hpp create mode 100644 src/output/render/slice2d.hpp diff --git a/input.example.toml b/input.example.toml index 08e13aa97..b4d5db216 100644 --- a/input.example.toml +++ b/input.example.toml @@ -681,10 +681,19 @@ # @default: [] custom = "" - # In-situ volume renderer. Ray-marches scalar fields on the GPU and writes - # PNG images directly to `/renders/` each cadence -- no field data is - # written to storage, and the result is seamless across MPI domain boundaries. - # @note: 3D Cartesian (Minkowski) only; a no-op for other dims/metrics + # In-situ renderer. Renders scalar fields on the GPU and writes PNG images + # directly to `/renders/` each cadence -- no field data is written to + # storage, and the result is seamless across MPI domain boundaries. + # @note: two modes, selected automatically by the simulation dimension: + # - 3D Cartesian (Minkowski): volume ray-march (uses `samples`, + # `step_size`, `early_term_alpha`, and the [camera] table) + # - 2D (Minkowski, Spherical/QSpherical, and all GR Kerr-Schild): + # flat slice rasterizer. Cartesian shows the (x, y) plane; spherical + # shows the meridional (r, theta) half-plane mapped to Cartesian + # (X = r sin th, Z = r cos th), optionally mirrored (see `mirror`). + # The `samples`/`step_size`/`early_term_alpha`/[camera] keys are + # ignored in 2D (one opaque sample per pixel). + # @note: 1D (and 3D non-Cartesian, which does not exist) is a no-op # @note: One PNG stream per scene (e.g. a density/|B|/|J| triptych) [output.render] # Toggle for the volume renderer @@ -746,10 +755,17 @@ # @type: bool # @default: true colorbar_outside = "" + # 2D spherical slice only: mirror the meridional half-plane across the + # symmetry axis to render a full disk from one axisymmetric half. No effect + # on Cartesian or 3D rendering. + # @type: bool + # @default: true + mirror = "" - # Camera. Defaults frame the whole global box from outside, looking down the - # (1,1,1) diagonal -- the production setup for which the structured composite - # is provably seamless. + # Camera (3D volume mode only; ignored by the 2D slice rasterizer). Defaults + # frame the whole global box from outside, looking down the (1,1,1) diagonal + # -- the production setup for which the structured composite is provably + # seamless. [output.render.camera] # Orthographic (true) or perspective (false) projection # @type: bool @@ -797,6 +813,9 @@ # {0,1,2,3} (e.g. "Txx", "Ttt", "T0x"); "V{i}" = one bulk-velocity # component, i in {x,y,z} or {1,2,3} (e.g. "Vx", "V1"); "Vmag" = # bulk-velocity magnitude sqrt(V1^2+V2^2+V3^2) + # @note: moments follow the engine: SRPIC = tetrad-basis bulk 3-velocity + # and stress-energy; GRPIC = Eckart-frame 4-velocity (so "Vt"/"V0" + # = u^0 = Gamma/alpha is also valid) and contravariant T # @note: per-species selection with a "_" suffix on moments, e.g. # "N_1", "Rho_2", "Txy_1_2", "V1_3"; default = all massive species # @note: components are signed; pair a symmetric `min`/`max` with a diff --git a/src/framework/domain/metadomain.h b/src/framework/domain/metadomain.h index 7e5d9f420..899213369 100644 --- a/src/framework/domain/metadomain.h +++ b/src/framework/domain/metadomain.h @@ -178,13 +178,22 @@ namespace ntt { void redecomposeFromCheckpoint(const std::vector>&, const std::vector>&); - /* in-situ volume renderer (see metadomain_render.cpp) ------------------ */ + /* in-situ renderer (3D volume ray-march & 2D slice; metadomain_render.cpp) */ void InitRenderer(const SimulationParams&); auto Render(const SimulationParams&, timestep_t, timestep_t, simtime_t, simtime_t) -> bool; + // Prepare the scalar named by a scene's `field` into bckp(:, 0) (active + // cells synced; ghosts not yet halo-filled). Shared by the 2D and 3D render + // paths so the field grammar (moments, T/V components, |E,B,J|, species + // suffix) has a single source of truth. Returns false (and warns) for an + // unknown field or invalid species so the caller can skip the scene. + auto prepareRenderScalar(const SimulationParams&, + Domain&, + const std::string& field_name, + ndfield_t&) const -> bool; #endif void InitStatsWriter(const SimulationParams&, bool); diff --git a/src/framework/domain/metadomain_render.cpp b/src/framework/domain/metadomain_render.cpp index 03fa80dd9..c45ba7f23 100644 --- a/src/framework/domain/metadomain_render.cpp +++ b/src/framework/domain/metadomain_render.cpp @@ -37,11 +37,15 @@ #include "kernels/particle_moments.hpp" #include "output/render/composite.h" #include "output/render/raymarch.hpp" +#include "output/render/reduce.hpp" +#include "output/render/slice2d.hpp" #include #include +#include #include +#include #include #include #include @@ -102,8 +106,312 @@ namespace ntt { Kokkos::Experimental::contribute(buffer, scatter_buff); } + // Copy 3 contiguous components [from.first, from.first+3) of a source field + // into bckp(:, 0..2). Dimension-generic (the subview arity depends on D). + template + void copyVec3ToBckp(const ndfield_t& src, + const ndfield_t& dst, + const cell_range_t& from) { + const cell_range_t to { 0, 3 }; + if constexpr (D == Dim::_2D) { + Kokkos::deep_copy( + Kokkos::subview(dst, Kokkos::ALL, Kokkos::ALL, to), + Kokkos::subview(src, Kokkos::ALL, Kokkos::ALL, from)); + } else if constexpr (D == Dim::_3D) { + Kokkos::deep_copy( + Kokkos::subview(dst, Kokkos::ALL, Kokkos::ALL, Kokkos::ALL, to), + Kokkos::subview(src, Kokkos::ALL, Kokkos::ALL, Kokkos::ALL, from)); + } + } + } // namespace + template + auto Metadomain::prepareRenderScalar(const SimulationParams& params, + Domain& domain, + const std::string& field_name, + ndfield_t& bckp) const + -> bool { + // Parse an optional trailing per-species suffix "__..."; + // species apply to particle moments only (N, Nppc, Rho, Charge, T, V). + std::string base = field_name; + std::vector species; + { + const auto us = field_name.find('_'); + if (us != std::string::npos) { + bool ok_sp = true; + std::size_t start = us + 1; + while (start <= field_name.size()) { + const auto nx = field_name.find('_', start); + const auto tok = field_name.substr( + start, + (nx == std::string::npos) ? std::string::npos : nx - start); + if (tok.empty() or + tok.find_first_not_of("0123456789") != std::string::npos) { + ok_sp = false; + break; + } + species.push_back(static_cast(std::stoi(tok))); + if (nx == std::string::npos) { + break; + } + start = nx + 1; + } + if (ok_sp) { + base = field_name.substr(0, us); + } else { + species.clear(); // not a species suffix; keep the full name + } + } + } + bool bad_species = false; + for (const auto sp : species) { + if (sp == 0 or sp > domain.species.size()) { + bad_species = true; + } + } + if (bad_species) { + raise::Warning("output.render: invalid species in '" + field_name + + "', skipping", + HERE); + return false; + } + + // axis/index character -> {t,x,y,z} == {0,1,2,3}; -1 if invalid + auto axisIdx = [](char ch) -> int { + switch (ch) { + case 't': + case '0': + return 0; + case 'x': + case '1': + return 1; + case 'y': + case '2': + return 2; + case 'z': + case '3': + return 3; + default: + return -1; + } + }; + + const auto& mesh = domain.mesh; + const auto metric = mesh.metric; + + if (base == "N" or base == "Nppc" or base == "Rho" or base == "Charge") { + // scalar particle moments + if (base == "N") { + renderMoment(params, mesh, domain.species, species, {}, + bckp, 0u); + } else if (base == "Nppc") { + renderMoment(params, mesh, domain.species, species, + {}, bckp, 0u); + } else if (base == "Rho") { + renderMoment(params, mesh, domain.species, species, + {}, bckp, 0u); + } else { + renderMoment(params, mesh, domain.species, species, + {}, bckp, 0u); + } + // sum boundary-crossing particle deposits back into active cells + SynchronizeFields(domain, Comm::Bckp, { 0, 1 }); + return true; + } else if (base.size() == 3 and base[0] == 'T') { + // a single stress-energy tensor component "T" (same for SR & GR; + // the moment kernel branches on the engine internally) + const int i = axisIdx(base[1]); + const int j = axisIdx(base[2]); + if (i >= 0 and j >= 0) { + const std::vector comps { static_cast(i), + static_cast(j) }; + renderMoment(params, mesh, domain.species, species, + comps, bckp, 0u); + SynchronizeFields(domain, Comm::Bckp, { 0, 1 }); + return true; + } + } else if (base == "Vmag") { + // bulk-velocity magnitude |V| = sqrt(V1^2 + V2^2 + V3^2) + if constexpr (S == SimEngine::GRPIC) { + // GR: Eckart-frame 4-velocity; need all 4 components for the norm + renderMoment(params, mesh, domain.species, species, + { 0u }, bckp, 0u); + renderMoment(params, mesh, domain.species, species, + { 1u }, bckp, 1u); + renderMoment(params, mesh, domain.species, species, + { 2u }, bckp, 2u); + renderMoment(params, mesh, domain.species, species, + { 3u }, bckp, 3u); + SynchronizeFields(domain, Comm::Bckp, { 0, 4 }); + Kokkos::parallel_for( + "RenderNormalize4Vel", + mesh.rangeActiveCells(), + kernel::Normalize4VelocityByNorm_kernel( + bckp, bckp, 0, 1, 2, 3, metric)); + Kokkos::parallel_for( + "RenderTransform4Vel", + mesh.rangeActiveCells(), + kernel::Transform4VelocitySpatialToPhysical_kernel( + bckp, 1, 2, 3, metric)); + // |spatial physical 4-velocity| -> bckp(0) + Kokkos::parallel_for("RenderVmagGR", + mesh.rangeActiveCells(), + kernel::RenderMagnitude3_kernel(bckp, 1, + 2, 3, 0)); + } else { + // SR: mass-weighted bulk 3-velocity, normalized by Rho + renderMoment(params, mesh, domain.species, species, + { 1u }, bckp, 0u); + renderMoment(params, mesh, domain.species, species, + { 2u }, bckp, 1u); + renderMoment(params, mesh, domain.species, species, + { 3u }, bckp, 2u); + renderMoment(params, mesh, domain.species, species, + {}, bckp, 3u); + SynchronizeFields(domain, Comm::Bckp, { 0, 4 }); + Kokkos::parallel_for("RenderVmagSR", + mesh.rangeActiveCells(), + kernel::RenderVmagByRho_kernel(bckp, 0, 1, + 2, 3, 0)); + } + return true; + } else if (base.size() == 2 and base[0] == 'V') { + // a single bulk-velocity component "V" + const int c = axisIdx(base[1]); + if constexpr (S == SimEngine::GRPIC) { + // GR: 4-velocity component (t/0 = u^0 = Gamma/alpha; x,y,z spatial) + if (c >= 0 and c <= 3) { + renderMoment(params, mesh, domain.species, species, + { 0u }, bckp, 0u); + renderMoment(params, mesh, domain.species, species, + { 1u }, bckp, 1u); + renderMoment(params, mesh, domain.species, species, + { 2u }, bckp, 2u); + renderMoment(params, mesh, domain.species, species, + { 3u }, bckp, 3u); + SynchronizeFields(domain, Comm::Bckp, { 0, 4 }); + Kokkos::parallel_for( + "RenderNormalize4Vel", + mesh.rangeActiveCells(), + kernel::Normalize4VelocityByNorm_kernel( + bckp, bckp, 0, 1, 2, 3, metric)); + Kokkos::parallel_for( + "RenderTransform4Vel", + mesh.rangeActiveCells(), + kernel::Transform4VelocitySpatialToPhysical_kernel( + bckp, 1, 2, 3, metric)); + if (c != 0) { + Kokkos::parallel_for( + "RenderPickV", + mesh.rangeActiveCells(), + kernel::RenderPickComp_kernel( + bckp, static_cast(c), 0)); + } + return true; + } + } else { + // SR: spatial bulk velocity (x,y,z), normalized by Rho + if (c >= 1 and c <= 3) { + renderMoment(params, mesh, domain.species, species, + { static_cast(c) }, bckp, 0u); + renderMoment(params, mesh, domain.species, species, + {}, bckp, 1u); + SynchronizeFields(domain, Comm::Bckp, { 0, 2 }); + Kokkos::parallel_for("RenderNormalizeV", + mesh.rangeActiveCells(), + kernel::RenderDivideComp_kernel(bckp, 0, + 1)); + return true; + } + } + } else { + // Vector field as a scalar: "" with base in {E, B, J} + // and selector in {mag, 1/2/3, x/y/z}. A component (e.g. "B1"/"Bx") is + // signed; a magnitude (e.g. "Bmag") is non-negative. + const std::string& f = base; + const char fbase = f.empty() + ? '?' + : static_cast(std::toupper(f[0])); + bool ok = true; + bool is_current = false; + uint8_t src_base = 0; // first component of the source field + PrepareOutputFlags interp = PrepareOutput::None; + if (fbase == 'B') { + src_base = em::bx1; + interp = PrepareOutput::InterpToCellCenterFromFaces; + } else if (fbase == 'E') { + src_base = em::ex1; + interp = PrepareOutput::InterpToCellCenterFromEdges; + } else if (fbase == 'J') { + is_current = true; + src_base = cur::jx1; + interp = PrepareOutput::InterpToCellCenterFromEdges; + } else { + ok = false; + } + // selector: -1 = magnitude, 0/1/2 = a single component + int comp = -2; + const std::string sel = (f.size() > 1) ? f.substr(1) : std::string {}; + if (sel == "mag") { + comp = -1; + } else if (sel == "1" or sel == "x") { + comp = 0; + } else if (sel == "2" or sel == "y") { + comp = 1; + } else if (sel == "3" or sel == "z") { + comp = 2; + } else { + ok = false; + } + if (ok) { + // raw vector components into bckp(:, 0..2) + if (is_current) { + copyVec3ToBckp(domain.fields.cur, bckp, + cell_range_t(cur::jx1, cur::jx3 + 1)); + } else { + copyVec3ToBckp(domain.fields.em, bckp, + cell_range_t(src_base, src_base + 3)); + } + // interpolate to cell centers + convert to physical basis -> (3,4,5) + const PrepareOutputFlags prepare = (S == SimEngine::SRPIC) + ? PrepareOutput::ConvertToHat + : PrepareOutput::ConvertToPhysCntrv; + list_t comp_from = { 0, 1, 2 }; + list_t comp_to = { 3, 4, 5 }; + Kokkos::parallel_for( + "RenderFieldsToPhys", + mesh.rangeActiveCells(), + kernel::FieldsToPhys_kernel(bckp, + bckp, + comp_from, + comp_to, + interp | prepare, + metric)); + // reduce to the scalar to render -> bckp(:, 0) + if (comp == -1) { + Kokkos::parallel_for( + "RenderVectorMagnitude", + mesh.rangeActiveCells(), + kernel::RenderMagnitude3_kernel(bckp, 3, 4, 5, 0)); + } else { + Kokkos::parallel_for( + "RenderVectorComponent", + mesh.rangeActiveCells(), + kernel::RenderPickComp_kernel( + bckp, static_cast(3 + comp), 0)); + } + return true; + } + } + + raise::Warning("output.render: unknown field '" + field_name + + "' (expected N/Nppc/Rho/Charge, T{i}{j}, V{i}/Vmag, or " + "{E,B,J}{mag,1,2,3,x,y,z}); skipping", + HERE); + return false; + } + template void Metadomain::InitRenderer(const SimulationParams& params) { g_renderer.init(params, mesh().extent()); @@ -115,7 +423,9 @@ namespace ntt { timestep_t finished_step, simtime_t current_time, simtime_t finished_time) -> bool { + (void)current_time; if constexpr (M::Dim == Dim::_3D and M::CoordType == Coord::type::Cartesian) { + // ---- 3D volume ray-march (Minkowski only) ----------------------- // // structured-order composite assumes an axis-aligned, affine code<->world // map; only Cartesian (Minkowski) 3D qualifies. if (not g_renderer.enabled() or @@ -129,7 +439,7 @@ namespace ntt { raise::ErrorIf(local_domain->is_placeholder(), "local_domain is a placeholder", HERE); - logger::Checkpoint("Rendering output", HERE); + logger::Checkpoint("Rendering output (3D volume)", HERE); const auto& cam = g_renderer.camera(); const int W = g_renderer.width(); @@ -174,273 +484,14 @@ namespace ntt { bool rendered_any = false; for (const auto& scene : g_renderer.scenes()) { Kokkos::deep_copy(bckp, ZERO); - - // Parse an optional trailing per-species suffix "__..."; - // species apply to particle moments only (N, Nppc, Rho, Charge, T, V). - std::string base = scene.field; - std::vector species; - { - const auto us = scene.field.find('_'); - if (us != std::string::npos) { - bool ok_sp = true; - std::size_t start = us + 1; - while (start <= scene.field.size()) { - const auto nx = scene.field.find('_', start); - const auto tok = scene.field.substr( - start, - (nx == std::string::npos) ? std::string::npos : nx - start); - if (tok.empty() or - tok.find_first_not_of("0123456789") != std::string::npos) { - ok_sp = false; - break; - } - species.push_back(static_cast(std::stoi(tok))); - if (nx == std::string::npos) { - break; - } - start = nx + 1; - } - if (ok_sp) { - base = scene.field.substr(0, us); - } else { - species.clear(); // not a species suffix; keep the full name - } - } - } - bool bad_species = false; - for (const auto sp : species) { - if (sp == 0 or sp > local_domain->species.size()) { - bad_species = true; - } - } - if (bad_species) { - raise::Warning("output.render: invalid species in '" + scene.field + - "', skipping", - HERE); + if (not prepareRenderScalar(params, *local_domain, scene.field, bckp)) { continue; } - - // axis/index character -> {t,x,y,z} == {0,1,2,3}; -1 if invalid - auto axisIdx = [](char ch) -> int { - switch (ch) { - case 't': - case '0': - return 0; - case 'x': - case '1': - return 1; - case 'y': - case '2': - return 2; - case 'z': - case '3': - return 3; - default: - return -1; - } - }; - - bool handled = false; - if (base == "N" or base == "Nppc" or base == "Rho" or base == "Charge") { - // scalar particle moments - if (base == "N") { - renderMoment(params, local_domain->mesh, - local_domain->species, species, {}, - bckp, 0u); - } else if (base == "Nppc") { - renderMoment(params, local_domain->mesh, - local_domain->species, species, {}, - bckp, 0u); - } else if (base == "Rho") { - renderMoment(params, local_domain->mesh, - local_domain->species, species, {}, - bckp, 0u); - } else { - renderMoment(params, local_domain->mesh, - local_domain->species, species, {}, - bckp, 0u); - } - // sum boundary-crossing particle deposits back into active cells - SynchronizeFields(*local_domain, Comm::Bckp, { 0, 1 }); - handled = true; - } else if (base.size() == 3 and base[0] == 'T') { - // a single stress-energy tensor component "T" - const int i = axisIdx(base[1]); - const int j = axisIdx(base[2]); - if (i >= 0 and j >= 0) { - const std::vector comps { static_cast(i), - static_cast(j) }; - renderMoment(params, local_domain->mesh, - local_domain->species, species, comps, - bckp, 0u); - SynchronizeFields(*local_domain, Comm::Bckp, { 0, 1 }); - handled = true; - } - } else if (base == "Vmag") { - // bulk-velocity magnitude |V| = sqrt(V1^2 + V2^2 + V3^2). Each Vi is - // the mass-weighted flux normalized by Rho, so deposit the three - // spatial components into bckp(0..2) and Rho into bckp(3), sum-sync - // all four, divide, then reduce to the Euclidean norm in bckp(0). - renderMoment(params, local_domain->mesh, - local_domain->species, species, { 1u }, - bckp, 0u); - renderMoment(params, local_domain->mesh, - local_domain->species, species, { 2u }, - bckp, 1u); - renderMoment(params, local_domain->mesh, - local_domain->species, species, { 3u }, - bckp, 2u); - renderMoment(params, local_domain->mesh, - local_domain->species, species, {}, - bckp, 3u); - SynchronizeFields(*local_domain, Comm::Bckp, { 0, 4 }); - auto bckp_v = bckp; - Kokkos::parallel_for( - "RenderNormalizeVmag", - local_domain->mesh.rangeActiveCells(), - Lambda(cellidx_t i1, cellidx_t i2, cellidx_t i3) { - const real_t rho = bckp_v(i1, i2, i3, 3); - if (rho != ZERO) { - const real_t v1 = bckp_v(i1, i2, i3, 0) / rho; - const real_t v2 = bckp_v(i1, i2, i3, 1) / rho; - const real_t v3 = bckp_v(i1, i2, i3, 2) / rho; - bckp_v(i1, i2, i3, 0) = math::sqrt(v1 * v1 + v2 * v2 + v3 * v3); - } else { - bckp_v(i1, i2, i3, 0) = ZERO; - } - }); - handled = true; - } else if (base.size() == 2 and base[0] == 'V') { - // a single bulk-velocity component "V" (spatial); normalize by Rho - const int c = axisIdx(base[1]); - if (c >= 1 and c <= 3) { - const std::vector comps { static_cast(c) }; - renderMoment(params, local_domain->mesh, - local_domain->species, species, comps, - bckp, 0u); - renderMoment(params, local_domain->mesh, - local_domain->species, species, {}, - bckp, 1u); - SynchronizeFields(*local_domain, Comm::Bckp, { 0, 2 }); - // V_c = (mass-weighted bulk velocity) / Rho - auto bckp_v = bckp; - Kokkos::parallel_for( - "RenderNormalizeV", - local_domain->mesh.rangeActiveCells(), - Lambda(cellidx_t i1, cellidx_t i2, cellidx_t i3) { - const real_t rho = bckp_v(i1, i2, i3, 1); - bckp_v(i1, i2, i3, 0) = (rho != ZERO) - ? (bckp_v(i1, i2, i3, 0) / rho) - : ZERO; - }); - handled = true; - } - } - - if (not handled) { - // Vector field as a scalar: "" with base in {E, B, J} - // and selector in {mag, 1/2/3, x/y/z}. A component (e.g. "B1"/"Bx") is - // signed; a magnitude (e.g. "Bmag") is non-negative. - const std::string& f = base; - const char fbase = f.empty() - ? '?' - : static_cast(std::toupper(f[0])); - bool ok = true; - bool is_current = false; - uint8_t src_base = 0; // first component of the source field - PrepareOutputFlags interp = PrepareOutput::None; - if (fbase == 'B') { - src_base = em::bx1; - interp = PrepareOutput::InterpToCellCenterFromFaces; - } else if (fbase == 'E') { - src_base = em::ex1; - interp = PrepareOutput::InterpToCellCenterFromEdges; - } else if (fbase == 'J') { - is_current = true; - src_base = cur::jx1; - interp = PrepareOutput::InterpToCellCenterFromEdges; - } else { - ok = false; - } - // selector: -1 = magnitude, 0/1/2 = a single component - int comp = -2; - const std::string sel = (f.size() > 1) ? f.substr(1) : std::string {}; - if (sel == "mag") { - comp = -1; - } else if (sel == "1" or sel == "x") { - comp = 0; - } else if (sel == "2" or sel == "y") { - comp = 1; - } else if (sel == "3" or sel == "z") { - comp = 2; - } else { - ok = false; - } - if (not ok) { - raise::Warning("output.render: unknown field '" + scene.field + - "' (expected N, {E,B,J}mag, or " - "{E,B,J}{1,2,3}|{x,y,z}); skipping", - HERE); - continue; - } - // raw vector components into bckp(:, 0..2) - if (is_current) { - Kokkos::deep_copy( - Kokkos::subview(bckp, Kokkos::ALL, Kokkos::ALL, Kokkos::ALL, - cell_range_t(0, 3)), - Kokkos::subview(local_domain->fields.cur, Kokkos::ALL, Kokkos::ALL, - Kokkos::ALL, cell_range_t(cur::jx1, cur::jx3 + 1))); - } else { - Kokkos::deep_copy( - Kokkos::subview(bckp, Kokkos::ALL, Kokkos::ALL, Kokkos::ALL, - cell_range_t(0, 3)), - Kokkos::subview(local_domain->fields.em, Kokkos::ALL, Kokkos::ALL, - Kokkos::ALL, cell_range_t(src_base, src_base + 3))); - } - // interpolate to cell centers + convert to physical basis -> (3,4,5) - const PrepareOutputFlags prepare = (S == SimEngine::SRPIC) - ? PrepareOutput::ConvertToHat - : PrepareOutput::ConvertToPhysCntrv; - list_t comp_from = { 0, 1, 2 }; - list_t comp_to = { 3, 4, 5 }; - Kokkos::parallel_for( - "RenderFieldsToPhys", - local_domain->mesh.rangeActiveCells(), - kernel::FieldsToPhys_kernel(bckp, - bckp, - comp_from, - comp_to, - interp | prepare, - metric)); - // reduce to the scalar to render -> bckp(:, 0) - auto bckp_v = bckp; - const int cc = comp; - if (comp == -1) { - Kokkos::parallel_for( - "RenderVectorMagnitude", - local_domain->mesh.rangeActiveCells(), - Lambda(cellidx_t i1, cellidx_t i2, cellidx_t i3) { - const real_t v1 = bckp_v(i1, i2, i3, 3); - const real_t v2 = bckp_v(i1, i2, i3, 4); - const real_t v3 = bckp_v(i1, i2, i3, 5); - bckp_v(i1, i2, i3, 0) = math::sqrt(v1 * v1 + v2 * v2 + v3 * v3); - }); - } else { - Kokkos::parallel_for( - "RenderVectorComponent", - local_domain->mesh.rangeActiveCells(), - Lambda(cellidx_t i1, cellidx_t i2, cellidx_t i3) { - bckp_v(i1, i2, i3, 0) = bckp_v(i1, i2, i3, 3 + cc); - }); - } - } - // fill the ghost halo with neighbor active values so trilinear - // sampling is C0 across domain faces (this is a halo EXCHANGE, not the + // sampling is C0 across domain faces (a halo EXCHANGE, not the // sum-into-active that SynchronizeFields performs). CommunicateBckp(*local_domain, { 0, 1 }); - // ---- launch the ray-march kernel over the screen bbox ---------- // out::SubImage sub; if (on_screen) { sub.x0 = bx0; @@ -496,11 +547,199 @@ namespace ntt { rendered_any = true; } return rendered_any; + } else if constexpr (M::Dim == Dim::_2D) { + // ---- 2D slice rasterizer (Cartesian or spherical) --------------- // + // A 2D run has no depth to integrate: each pixel is one inverse-mapped + // sample, painted opaque. Domains tile the screen disjointly, so the + // sparse sub-images composite seamlessly regardless of order. + if (not g_renderer.enabled() or + not g_renderer.shouldRender(finished_step, finished_time)) { + return false; + } + raise::ErrorIf(l_subdomain_indices().size() != 1, + "Renderer supports one subdomain per rank only", + HERE); + auto local_domain = subdomain_ptr(l_subdomain_indices()[0]); + raise::ErrorIf(local_domain->is_placeholder(), + "local_domain is a placeholder", + HERE); + logger::Checkpoint("Rendering output (2D slice)", HERE); + + const int W = g_renderer.width(); + const int H = g_renderer.height(); + const bool mirror = g_renderer.mirror(); + + // global slice-plane world window (shared by all ranks -> seamless) + const auto gext = mesh().extent(); + real_t umin, umax, vmin, vmax; + if constexpr (M::CoordType == Coord::type::Cartesian) { + umin = gext[0].first; + umax = gext[0].second; + vmin = gext[1].first; + vmax = gext[1].second; + } else { + // meridional (X = r sin th, Z = r cos th) bounding box of the arc + const real_t rmax = gext[0].second; + const real_t th0 = gext[1].first; + const real_t th1 = gext[1].second; + umax = rmax; + umin = mirror ? -rmax : ZERO; + vmax = rmax * math::cos(th0); + vmin = rmax * math::cos(th1); + } + // expand the window to the image aspect (centered) so geometry is not + // stretched + { + const real_t waspect = (umax - umin) / (vmax - vmin); + const real_t iaspect = static_cast(W) / static_cast(H); + if (iaspect > waspect) { + const real_t cu = HALF * (umin + umax); + const real_t hu = HALF * (vmax - vmin) * iaspect; + umin = cu - hu; + umax = cu + hu; + } else { + const real_t cv = HALF * (vmin + vmax); + const real_t hv = HALF * (umax - umin) / iaspect; + vmin = cv - hv; + vmax = cv + hv; + } + } + + auto& bckp = local_domain->fields.bckp; + const int ext0 = static_cast(bckp.extent(0)); + const int ext1 = static_cast(bckp.extent(1)); + const auto metric = local_domain->mesh.metric; + const int n1 = static_cast(local_domain->mesh.n_active(in::x1)); + const int n2 = static_cast(local_domain->mesh.n_active(in::x2)); + + // screen-space bbox of this domain's footprint (host projection of the + // boundary; an arc for spherical, a box for Cartesian) + const auto le = local_domain->mesh.extent(); + auto toPix = [&](real_t u, real_t v, real_t& px, real_t& py) { + px = (u - umin) / (umax - umin) * static_cast(W) - HALF; + py = (vmax - v) / (vmax - vmin) * static_cast(H) - HALF; + }; + real_t minx = static_cast(1e30), miny = static_cast(1e30); + real_t maxx = static_cast(-1e30), maxy = static_cast(-1e30); + auto acc = [&](real_t u, real_t v) { + real_t px, py; + toPix(u, v, px, py); + minx = std::min(minx, px); + maxx = std::max(maxx, px); + miny = std::min(miny, py); + maxy = std::max(maxy, py); + }; + if constexpr (M::CoordType == Coord::type::Cartesian) { + acc(le[0].first, le[1].first); + acc(le[0].second, le[1].first); + acc(le[0].first, le[1].second); + acc(le[0].second, le[1].second); + } else { + const int NB = 33; + const real_t r0 = le[0].first, r1 = le[0].second; + const real_t a0 = le[1].first, a1 = le[1].second; + for (int k = 0; k < NB; ++k) { + const real_t t = static_cast(k) / static_cast(NB - 1); + const real_t rr = r0 + (r1 - r0) * t; + const real_t aa = a0 + (a1 - a0) * t; + // r-arcs at a0, a1 and theta-rays at r0, r1 + const real_t pts[4][2] = { { r0 * math::sin(aa), r0 * math::cos(aa) }, + { r1 * math::sin(aa), r1 * math::cos(aa) }, + { rr * math::sin(a0), rr * math::cos(a0) }, + { rr * math::sin(a1), rr * math::cos(a1) } }; + for (auto& p : pts) { + acc(p[0], p[1]); + if (mirror) { + acc(-p[0], p[1]); + } + } + } + } + const int pad = 2; + int x0 = static_cast(std::floor(minx)) - pad; + int x1 = static_cast(std::ceil(maxx)) + pad; + int y0 = static_cast(std::floor(miny)) - pad; + int y1 = static_cast(std::ceil(maxy)) + pad; + x0 = std::max(0, std::min(W, x0)); + x1 = std::max(0, std::min(W, x1)); + y0 = std::max(0, std::min(H, y0)); + y1 = std::max(0, std::min(H, y1)); + const int bx0 = x0, by0 = y0, bw = x1 - x0, bh = y1 - y0; + + // disjoint tiling -> any consistent total order composites correctly; + // a lexicographic key over the decomposition offsets is unique per rank. + const real_t fwd2d[3] = { ONE, ONE, ZERO }; + const uint64_t order_key = out::compositeOrderKey( + local_domain->offset_ndomains(), + ndomains_per_dim(), + fwd2d); + + bool rendered_any = false; + for (const auto& scene : g_renderer.scenes()) { + Kokkos::deep_copy(bckp, ZERO); + if (not prepareRenderScalar(params, *local_domain, scene.field, bckp)) { + continue; + } + CommunicateBckp(*local_domain, { 0, 1 }); + + out::SubImage sub; + if (bw > 0 and bh > 0) { + sub.x0 = bx0; + sub.y0 = by0; + sub.w = bw; + sub.h = bh; + const std::size_t bnpix = static_cast(bw) * + static_cast(bh); + array_t image { "render_img", bnpix }; + randacc_ndfield_t Fld { bckp }; + Kokkos::parallel_for( + "Slice2DRaster", + CreateRangePolicy({ 0, 0 }, + { static_cast(bw), + static_cast(bh) }), + kernel::SliceRaster_kernel(Fld, + 0u, + metric, + umin, + umax, + vmin, + vmax, + W, + H, + bx0, + by0, + bw, + mirror, + n1, + n2, + ext0, + ext1, + scene.tf.lut_opaque, + scene.tf.n_lut, + scene.tf.vmin, + scene.tf.vmax, + scene.tf.log_scale, + image)); + Kokkos::fence(); + + auto image_h = Kokkos::create_mirror_view(image); + Kokkos::deep_copy(image_h, image); + sub.rgba.resize(bnpix * 4); + for (std::size_t p = 0; p < bnpix; ++p) { + sub.rgba[p * 4 + 0] = image_h(p, 0); + sub.rgba[p * 4 + 1] = image_h(p, 1); + sub.rgba[p * 4 + 2] = image_h(p, 2); + sub.rgba[p * 4 + 3] = image_h(p, 3); + } + } + g_renderer.compositeAndWrite(sub, order_key, scene, current_step); + rendered_any = true; + } + return rendered_any; } else { (void)params; (void)current_step; (void)finished_step; - (void)current_time; (void)finished_time; return false; } @@ -513,7 +752,12 @@ namespace ntt { timestep_t, \ timestep_t, \ simtime_t, \ - simtime_t) -> bool; + simtime_t) -> bool; \ + template auto Metadomain>::prepareRenderScalar( \ + const SimulationParams&, \ + Domain>&, \ + const std::string&, \ + ndfield_t::Dim, 6>&) const -> bool; NTT_FOREACH_SPECIALIZATION(METADOMAIN_RENDER) diff --git a/src/output/render/reduce.hpp b/src/output/render/reduce.hpp new file mode 100644 index 000000000..18478fc6b --- /dev/null +++ b/src/output/render/reduce.hpp @@ -0,0 +1,178 @@ +/** + * @file output/render/reduce.hpp + * @brief Small dimension-generic cell reductions used by the in-situ renderer + * @implements + * - kernel::RenderMagnitude3_kernel + * - kernel::RenderPickComp_kernel + * - kernel::RenderDivideComp_kernel + * - kernel::RenderVmagByRho_kernel + * @namespaces: + * - kernel:: + * @macros: + * - OUTPUT_ENABLED + * @note + * Each functor provides 1D/2D/3D operator() overloads so the same object works + * with `mesh.rangeActiveCells()` of any dimension (the range policy selects the + * matching arity). They reduce the prepared (interpolated, synced) `bckp` + * scratch field down to the single scalar component the ray-march / slice + * kernel samples, so the field-grammar dispatch stays dimension-agnostic. + */ + +#ifndef OUTPUT_RENDER_REDUCE_HPP +#define OUTPUT_RENDER_REDUCE_HPP + +#include "global.h" + +#include "arch/kokkos_aliases.h" + +#include + +namespace kernel { + using namespace ntt; + + /** + * @brief F(.., co) = sqrt(F(.., c0)^2 + F(.., c1)^2 + F(.., c2)^2) + */ + template + class RenderMagnitude3_kernel { + ndfield_t F; + const std::uint8_t c0, c1, c2, co; + + public: + RenderMagnitude3_kernel(const ndfield_t& f, + std::uint8_t a, + std::uint8_t b, + std::uint8_t c, + std::uint8_t o) + : F { f } + , c0 { a } + , c1 { b } + , c2 { c } + , co { o } {} + + Inline void operator()(cellidx_t i1) const { + const real_t v0 = F(i1, c0), v1 = F(i1, c1), v2 = F(i1, c2); + F(i1, co) = math::sqrt(v0 * v0 + v1 * v1 + v2 * v2); + } + + Inline void operator()(cellidx_t i1, cellidx_t i2) const { + const real_t v0 = F(i1, i2, c0), v1 = F(i1, i2, c1), v2 = F(i1, i2, c2); + F(i1, i2, co) = math::sqrt(v0 * v0 + v1 * v1 + v2 * v2); + } + + Inline void operator()(cellidx_t i1, cellidx_t i2, cellidx_t i3) const { + const real_t v0 = F(i1, i2, i3, c0), v1 = F(i1, i2, i3, c1), + v2 = F(i1, i2, i3, c2); + F(i1, i2, i3, co) = math::sqrt(v0 * v0 + v1 * v1 + v2 * v2); + } + }; + + /** + * @brief F(.., co) = F(.., ci) (move one component into the render slot) + */ + template + class RenderPickComp_kernel { + ndfield_t F; + const std::uint8_t ci, co; + + public: + RenderPickComp_kernel(const ndfield_t& f, std::uint8_t i, std::uint8_t o) + : F { f } + , ci { i } + , co { o } {} + + Inline void operator()(cellidx_t i1) const { + F(i1, co) = F(i1, ci); + } + + Inline void operator()(cellidx_t i1, cellidx_t i2) const { + F(i1, i2, co) = F(i1, i2, ci); + } + + Inline void operator()(cellidx_t i1, cellidx_t i2, cellidx_t i3) const { + F(i1, i2, i3, co) = F(i1, i2, i3, ci); + } + }; + + /** + * @brief F(.., cnum) = (F(.., cden) != 0) ? F(.., cnum) / F(.., cden) : 0 + */ + template + class RenderDivideComp_kernel { + ndfield_t F; + const std::uint8_t cnum, cden; + + public: + RenderDivideComp_kernel(const ndfield_t& f, + std::uint8_t num, + std::uint8_t den) + : F { f } + , cnum { num } + , cden { den } {} + + Inline void operator()(cellidx_t i1) const { + const real_t d = F(i1, cden); + F(i1, cnum) = (d != ZERO) ? (F(i1, cnum) / d) : ZERO; + } + + Inline void operator()(cellidx_t i1, cellidx_t i2) const { + const real_t d = F(i1, i2, cden); + F(i1, i2, cnum) = (d != ZERO) ? (F(i1, i2, cnum) / d) : ZERO; + } + + Inline void operator()(cellidx_t i1, cellidx_t i2, cellidx_t i3) const { + const real_t d = F(i1, i2, i3, cden); + F(i1, i2, i3, cnum) = (d != ZERO) ? (F(i1, i2, i3, cnum) / d) : ZERO; + } + }; + + /** + * @brief F(.., co) = | (F(c0), F(c1), F(c2)) / F(crho) |, else 0. + * @note SR bulk-speed magnitude: the three mass-weighted flux components are + * each normalized by Rho before the Euclidean norm, in one pass. + */ + template + class RenderVmagByRho_kernel { + ndfield_t F; + const std::uint8_t c0, c1, c2, crho, co; + + public: + RenderVmagByRho_kernel(const ndfield_t& f, + std::uint8_t a, + std::uint8_t b, + std::uint8_t c, + std::uint8_t rho, + std::uint8_t o) + : F { f } + , c0 { a } + , c1 { b } + , c2 { c } + , crho { rho } + , co { o } {} + + Inline auto mag(real_t v0, real_t v1, real_t v2, real_t rho) const -> real_t { + if (rho == ZERO) { + return ZERO; + } + const real_t a = v0 / rho, b = v1 / rho, c = v2 / rho; + return math::sqrt(a * a + b * b + c * c); + } + + Inline void operator()(cellidx_t i1) const { + F(i1, co) = mag(F(i1, c0), F(i1, c1), F(i1, c2), F(i1, crho)); + } + + Inline void operator()(cellidx_t i1, cellidx_t i2) const { + F(i1, i2, co) = mag(F(i1, i2, c0), F(i1, i2, c1), F(i1, i2, c2), + F(i1, i2, crho)); + } + + Inline void operator()(cellidx_t i1, cellidx_t i2, cellidx_t i3) const { + F(i1, i2, i3, co) = mag(F(i1, i2, i3, c0), F(i1, i2, i3, c1), + F(i1, i2, i3, c2), F(i1, i2, i3, crho)); + } + }; + +} // namespace kernel + +#endif // OUTPUT_RENDER_REDUCE_HPP diff --git a/src/output/render/renderer.cpp b/src/output/render/renderer.cpp index 7c33fc2c5..128135895 100644 --- a/src/output/render/renderer.cpp +++ b/src/output/render/renderer.cpp @@ -69,10 +69,11 @@ namespace out { if (not enable) { return; } - // the renderer is a 3D feature; silently no-op otherwise. - if (global_extent.size() != 3) { - raise::Warning("output.render enabled but simulation is not 3D; " - "the volume renderer will be inactive", + // 2D (slice rasterizer) and 3D Cartesian (volume ray-march) are supported; + // 1D has nothing to render. + if (global_extent.size() != 2 and global_extent.size() != 3) { + raise::Warning("output.render enabled but simulation is 1D; " + "the renderer will be inactive", HERE); return; } @@ -108,6 +109,8 @@ namespace out { "render", "colorbar_outside", true); + // 2D slice mode (spherical only): mirror the half-plane into a full disk + m_mirror = toml::find_or(td, "output", "render", "mirror", true); // cadence: mirror output.* (interval in steps; interval_time in sim time) const auto interval = toml::find_or(td, @@ -122,10 +125,11 @@ namespace out { -1.0); m_tracker.init("render", interval, interval_time); - /* ---- camera --------------------------------------------------------- */ - real_t center[3], size[3]; + /* ---- camera (used by the 3D volume mode; the 2D slice path frames itself + * and ignores this, so a missing 3rd axis is zero-filled harmlessly) ---- */ + real_t center[3] = { ZERO, ZERO, ZERO }, size[3] = { ZERO, ZERO, ZERO }; real_t maxext = ZERO; - for (int d = 0; d < 3; ++d) { + for (std::size_t d = 0; d < global_extent.size() and d < 3; ++d) { center[d] = static_cast(0.5) * (global_extent[d].first + global_extent[d].second); size[d] = global_extent[d].second - global_extent[d].first; @@ -249,6 +253,10 @@ namespace out { } } scene.tf.lut = buildLUT(colormap, m_n_lut, alpha_pts); + // opaque companion LUT (alpha == 1) for the flat 2D slice rasterizer + scene.tf.lut_opaque = buildLUT(colormap, + m_n_lut, + { { ZERO, ONE }, { ONE, ONE } }); m_scenes.push_back(std::move(scene)); } @@ -258,7 +266,7 @@ namespace out { } m_enabled = true; - logger::Checkpoint("Volume renderer initialized", HERE); + logger::Checkpoint("In-situ renderer initialized", HERE); } void Renderer::compositeAndWrite(const SubImage& sub, diff --git a/src/output/render/renderer.h b/src/output/render/renderer.h index 911847c47..15b64b624 100644 --- a/src/output/render/renderer.h +++ b/src/output/render/renderer.h @@ -58,6 +58,10 @@ namespace out { */ struct TransferFunction { array_t lut; // device, (n_lut, 4), premultiplied RGBA + // opaque variant (alpha == 1 everywhere, so the premultiplied entries are + // straight RGB) used by the flat 2D slice rasterizer, where a single + // per-pixel sample should paint a solid heatmap rather than fade by opacity. + array_t lut_opaque; int n_lut { 256 }; real_t vmin { ZERO }; real_t vmax { ONE }; @@ -69,7 +73,7 @@ namespace out { * @brief One rendered scalar field -> one PNG stream. */ struct Scene { - std::string field; // "N" | "Bmag" | "Jmag" | "smooth_xyz" + std::string field; // "N" | "Bmag" | "Vmag" | "Txy" | "B1" ... std::string prefix; // PNG filename prefix, e.g. "Bmag_" std::string label; // colorbar title (defaults to field) std::vector ticks; // explicit colorbar tick values (optional) @@ -160,6 +164,13 @@ namespace out { return m_camera_dev; } + // 2D slice mode: mirror a spherical half-plane across the axis into a full + // disk (no effect on Cartesian or 3D rendering). + [[nodiscard]] + auto mirror() const -> bool { + return m_mirror; + } + [[nodiscard]] auto scenes() const -> const std::vector& { return m_scenes; @@ -182,6 +193,9 @@ namespace out { // draw the colorbar in an extended right margin (outside the render region) // rather than overlaying it on top of the rendered volume bool m_colorbar_outside { true }; + // 2D slice mode (spherical): mirror the meridional half-plane across the + // symmetry axis to render a full disk from one axisymmetric half + bool m_mirror { true }; CameraDevice m_camera_dev; std::vector m_scenes; diff --git a/src/output/render/slice2d.hpp b/src/output/render/slice2d.hpp new file mode 100644 index 000000000..335aab0aa --- /dev/null +++ b/src/output/render/slice2d.hpp @@ -0,0 +1,210 @@ +/** + * @file output/render/slice2d.hpp + * @brief Header-only Kokkos 2D slice rasterizer (one parallel_for over pixels) + * @implements + * - kernel::SliceRaster_kernel + * @namespaces: + * - kernel:: + * @macros: + * - OUTPUT_ENABLED + * @note + * The 2D counterpart of the volume ray-march: a 2D simulation has no depth to + * integrate, so each screen pixel is a single inverse-mapped sample of the + * prepared scalar, painted opaque. Two coordinate families are handled at + * compile time via M::CoordType: + * - Cartesian (Minkowski 2D): the screen window IS the (x, y) physical plane; + * the inverse map is the per-axis code conversion. + * - Spherical / Qspherical (2D SR & all 2D GR): the screen window is the + * meridional (X, Z) Cartesian plane; a pixel maps to physical + * r = sqrt(X^2 + Z^2), theta = atan2(|X|, Z), then per-axis code conversion + * (separable once in physical spherical coords). With `mirror`, X<0 is the + * theta-reflected half, yielding a full disk from one axisymmetric half. + * + * Seamlessness: every rank shares the same global screen window, so a pixel's + * world point is identical on all ranks. Each pixel's active-region membership + * (code index in [0, n]) selects exactly one domain in the interior (boundary + * pixels may be claimed by two, but the halo-filled value is continuous there), + * so the disjoint sub-images composite without seams regardless of order. + */ + +#ifndef OUTPUT_RENDER_SLICE2D_HPP +#define OUTPUT_RENDER_SLICE2D_HPP + +#include "enums.h" +#include "global.h" + +#include "arch/kokkos_aliases.h" +#include "traits/metric.h" + +#include "output/render/renderer.h" + +namespace kernel { + using namespace ntt; + + template + class SliceRaster_kernel { + static constexpr auto D = M::Dim; + static_assert(D == Dim::_2D, "SliceRaster_kernel is 2D only"); + + randacc_ndfield_t Fld; + const idx_t comp; + const M metric; + + // global orthographic window in slice-plane world coords (shared by ranks) + const real_t umin, umax, vmin, vmax; + const int W, H; // full frame size (ray generation / ndc) + const int bx0, by0, bw; // screen-bbox offset and width (output stride) + const bool mirror; // spherical: paint the X<0 reflected half too + + // local-domain active cell counts and View extents (membership + clamping) + const real_t n1, n2; + const int ext0, ext1; + + // transfer function (opaque LUT: premultiplied with alpha == 1) + array_t lut; + const int n_lut; + const real_t vlo, vhi; + const bool log_scale; + + array_t image; // output, (bw*bh, 4) premultiplied RGBA + + public: + SliceRaster_kernel(const randacc_ndfield_t& Fld_, + idx_t comp_, + const M& metric_, + real_t umin_, + real_t umax_, + real_t vmin_, + real_t vmax_, + int W_, + int H_, + int bx0_, + int by0_, + int bw_, + bool mirror_, + int n1_, + int n2_, + int ext0_, + int ext1_, + const array_t& lut_, + int n_lut_, + real_t vlo_, + real_t vhi_, + bool log_scale_, + const array_t& image_) + : Fld { Fld_ } + , comp { comp_ } + , metric { metric_ } + , umin { umin_ } + , umax { umax_ } + , vmin { vmin_ } + , vmax { vmax_ } + , W { W_ } + , H { H_ } + , bx0 { bx0_ } + , by0 { by0_ } + , bw { bw_ } + , mirror { mirror_ } + , n1 { static_cast(n1_) } + , n2 { static_cast(n2_) } + , ext0 { ext0_ } + , ext1 { ext1_ } + , lut { lut_ } + , n_lut { n_lut_ } + , vlo { vlo_ } + , vhi { vhi_ } + , log_scale { log_scale_ } + , image { image_ } {} + + // bilinear sample of the prepared scalar at continuous code coords + // (cc1, cc2), reading the ghost halo for corners just outside the box. + Inline auto sample(real_t cc1, real_t cc2) const -> real_t { + const real_t g0 = cc1 - HALF; // cell-center continuous index + const real_t g1 = cc2 - HALF; + const real_t f0 = math::floor(g0); + const real_t f1 = math::floor(g1); + const real_t t0 = g0 - f0; + const real_t t1 = g1 - f1; + int b0 = static_cast(f0) + static_cast(N_GHOSTS); + int b1 = static_cast(f1) + static_cast(N_GHOSTS); + b0 = (b0 < 0) ? 0 : ((b0 > ext0 - 2) ? ext0 - 2 : b0); + b1 = (b1 < 0) ? 0 : ((b1 > ext1 - 2) ? ext1 - 2 : b1); + const real_t c00 = Fld(b0, b1, comp); + const real_t c10 = Fld(b0 + 1, b1, comp); + const real_t c01 = Fld(b0, b1 + 1, comp); + const real_t c11 = Fld(b0 + 1, b1 + 1, comp); + const real_t c0 = c00 * (ONE - t0) + c10 * t0; + const real_t c1 = c01 * (ONE - t0) + c11 * t0; + return c0 * (ONE - t1) + c1 * t1; + } + + Inline void operator()(cellidx_t lpx, cellidx_t lpy) const { + const auto pix = static_cast(lpy) * + static_cast(bw) + + static_cast(lpx); + const int gpx = bx0 + static_cast(lpx); + const int gpy = by0 + static_cast(lpy); + // default transparent + image(pix, 0) = ZERO; + image(pix, 1) = ZERO; + image(pix, 2) = ZERO; + image(pix, 3) = ZERO; + + // pixel center -> slice-plane world coords (v flipped so +v is up) + const real_t u = umin + (static_cast(gpx) + HALF) / + static_cast(W) * (umax - umin); + const real_t v = vmax - (static_cast(gpy) + HALF) / + static_cast(H) * (vmax - vmin); + + // world -> continuous local code coords + real_t cc1, cc2; + if constexpr (M::CoordType == Coord::Cartesian) { + cc1 = metric.template convert<1, Crd::Ph, Crd::Cd>(u); + cc2 = metric.template convert<2, Crd::Ph, Crd::Cd>(v); + } else { + if (not mirror and u < ZERO) { + return; // only the X>=0 meridional half is physical + } + const real_t r = math::sqrt(u * u + v * v); + const real_t th = math::atan2(math::abs(u), v); // in [0, pi] + cc1 = metric.template convert<1, Crd::Ph, Crd::Cd>(r); + cc2 = metric.template convert<2, Crd::Ph, Crd::Cd>(th); + } + // active-region membership (inclusive so interiors gap-free, boundaries + // shared harmlessly); outside -> leave transparent + if (cc1 < ZERO or cc1 > n1 or cc2 < ZERO or cc2 > n2) { + return; + } + + const real_t s = sample(cc1, cc2); + // normalize through the transfer-function range + const real_t inv_range = (vhi > vlo) ? (ONE / (vhi - vlo)) : ZERO; + real_t uu; + if (log_scale) { + const real_t log_vlo = math::log10(vlo); + uu = (s > ZERO) ? (math::log10(s) - log_vlo) * inv_range : -ONE; + } else { + uu = (s - vlo) * inv_range; + } + if (uu < ZERO) { + uu = ZERO; + } else if (uu > ONE) { + uu = ONE; + } + int idx = static_cast(uu * static_cast(n_lut - 1) + HALF); + if (idx < 0) { + idx = 0; + } else if (idx > n_lut - 1) { + idx = n_lut - 1; + } + // opaque LUT: premultiplied with alpha == 1, so this is straight RGB + image(pix, 0) = lut(idx, 0); + image(pix, 1) = lut(idx, 1); + image(pix, 2) = lut(idx, 2); + image(pix, 3) = ONE; + } + }; + +} // namespace kernel + +#endif // OUTPUT_RENDER_SLICE2D_HPP From 02cf11c042d823fbba1f37dae6ef8f8f499d0b45 Mon Sep 17 00:00:00 2001 From: Ludwig Boess Date: Tue, 30 Jun 2026 17:24:13 -0400 Subject: [PATCH 045/125] rendering of axis spines, ticks and labels --- input.example.toml | 28 + src/framework/domain/metadomain_render.cpp | 59 ++ src/output/render/axes.h | 711 +++++++++++++++++++++ src/output/render/raymarch.hpp | 112 +++- src/output/render/renderer.cpp | 119 ++-- src/output/render/renderer.h | 80 +++ 6 files changed, 1045 insertions(+), 64 deletions(-) create mode 100644 src/output/render/axes.h diff --git a/input.example.toml b/input.example.toml index b4d5db216..527f5e9de 100644 --- a/input.example.toml +++ b/input.example.toml @@ -761,6 +761,34 @@ # @type: bool # @default: true mirror = "" + # Draw a spine (frame) + axis ticks + labels around the rendered region. + # The PNG gains left/bottom margins (background-filled) for the tick labels + # and axis names, so they never overlap the data. + # @type: bool + # @default: false + # @note: 2D Cartesian = a rectangular frame with linear spatial ticks; + # 2D spherical = polar axes (an "R" radial axis on the symmetry + # axis with R=0 centered, and a "Theta" axis along the curved + # outline / spine); + # 3D = the global box projected to a wireframe with ticks on the + # three silhouette edges (x bottom, y & z on the left) + axes = "" + # Axis names. 3D uses all three; the 2D slice uses the first two. When unset, + # the 2D slice defaults to "x","y" (Cartesian) or "X","Z" (spherical). + # @type: array [size <= 3] + # @default: ["x", "y", "z"] + axis_labels = "" + # Target number of ticks per axis (actual count is rounded to nice values) + # @type: int [>= 2] + # @default: 5 + axis_ticks = "" + # 3D only: target width (pixels) of the box wireframe "spine". The spine is + # drawn inside the ray-march (opaque, depth-occluded by the volume); its + # width is floored by the ray step, so for a crisper thin line raise + # `samples` as well. + # @type: float [> 0.0] + # @default: 2.0 + spine_width = "" # Camera (3D volume mode only; ignored by the 2D slice rasterizer). Defaults # frame the whole global box from outside, looking down the (1,1,1) diagonal diff --git a/src/framework/domain/metadomain_render.cpp b/src/framework/domain/metadomain_render.cpp index c45ba7f23..f9320a094 100644 --- a/src/framework/domain/metadomain_render.cpp +++ b/src/framework/domain/metadomain_render.cpp @@ -463,6 +463,28 @@ namespace ntt { : gdiag / static_cast(g_renderer.samples()); const int max_steps = 2 * g_renderer.samples() + 16; + // global box + depth-occluded spine (opaque box wireframe rendered inline + // in the march so the volume covers its far edges). The visual width is + // ~spine_width px; the 0.55*ds floor keeps the thin line gap-free at the + // current sampling (raise `samples` for a crisper, thinner line). + real_t glo[3] = { glob_ext[0].first, glob_ext[1].first, + glob_ext[2].first }; + real_t ghi[3] = { glob_ext[0].second, glob_ext[1].second, + glob_ext[2].second }; + const real_t px_w = (cam.half_h * static_cast(2)) / + static_cast(H); + const real_t spine_radius = + g_renderer.axes() + ? math::max(static_cast(0.55) * ds, + HALF * g_renderer.spineWidth() * px_w) + : ZERO; + // contrasting opaque spine color (white on dark bg, black on light) + const real_t bg_lum = static_cast(0.299) * g_renderer.background(0) + + static_cast(0.587) * g_renderer.background(1) + + static_cast(0.114) * g_renderer.background(2); + const real_t sc = (bg_lum < HALF) ? ONE : ZERO; + const real_t spine_rgb[3] = { sc, sc, sc }; + // composite order key (depends on the current decomposition offsets) const uint64_t order_key = out::compositeOrderKey( local_domain->offset_ndomains(), @@ -529,6 +551,10 @@ namespace ntt { scene.tf.vmax, scene.tf.log_scale, g_renderer.earlyAlpha(), + glo, + ghi, + spine_radius, + spine_rgb, image)); Kokkos::fence(); @@ -604,6 +630,39 @@ namespace ntt { vmax = cv + hv; } } + // spherical slices get a background border so the round outline and its + // R/theta labels are not clipped at the frame edges (Cartesian fills the + // frame and draws its ticks in dedicated margins, so it needs none). + if constexpr (M::CoordType != Coord::type::Cartesian) { + const real_t pad = static_cast(1.12); + const real_t cu = HALF * (umin + umax), hu = HALF * (umax - umin) * pad; + const real_t cv = HALF * (vmin + vmax), hv = HALF * (vmax - vmin) * pad; + umin = cu - hu; + umax = cu + hu; + vmin = cv - hv; + vmax = cv + hv; + } + + // hand the world window + axis names to the (host) axes overlay. Default + // names follow the coordinate family unless the toml set axis_labels. + { + const bool sph = (M::CoordType != Coord::type::Cartesian); + const std::string xl = g_renderer.axisLabelsSet() + ? g_renderer.axisLabel(0) + : (sph ? std::string("X") : std::string("x")); + const std::string yl = g_renderer.axisLabelsSet() + ? g_renderer.axisLabel(1) + : (sph ? std::string("Z") : std::string("y")); + g_renderer.setSliceFrame(umin, umax, vmin, vmax, xl, yl); + // curvilinear slices get polar axes (R radial + Theta arc); pass the + // global (r, theta) extent. + if (sph) { + g_renderer.setSlicePolar(true, gext[0].first, gext[0].second, + gext[1].first, gext[1].second, mirror); + } else { + g_renderer.setSlicePolar(false, ZERO, ONE, ZERO, ONE, mirror); + } + } auto& bckp = local_domain->fields.bckp; const int ext0 = static_cast(bckp.extent(0)); diff --git a/src/output/render/axes.h b/src/output/render/axes.h new file mode 100644 index 000000000..9807ec55b --- /dev/null +++ b/src/output/render/axes.h @@ -0,0 +1,711 @@ +/** + * @file output/render/axes.h + * @brief Draw a spine (frame), axis ticks and labels onto the opaque RGBA + * canvas, for both the 2D slice and the 3D volume renders. + * @implements + * - out::axesMargins + * - out::drawAxes2D + * - out::drawAxes3D + * @namespaces: + * - out:: + * @note + * Header-only, host-only, drawn on the MPI root rank after compositing (like the + * colorbar). Reuses the 5x7 bitmap font and helpers from colorbar.h. + * - 2D: the data region maps affinely to a world window [u0,u1]x[v0,v1]; a + * rectangular spine is drawn around it with linear ticks/labels in the + * surrounding margins (so they never overlap the data). + * - 3D: the global box is projected with the ray-march camera into a wireframe + * "spine"; ticks + labels are placed along the three edges emanating from the + * bottom-most projected corner, marks pushed outward from the box centroid. + */ + +#ifndef OUTPUT_RENDER_AXES_H +#define OUTPUT_RENDER_AXES_H + +#include "global.h" + +#include "output/render/colorbar.h" // glyph, scale, fmtNum +#include "output/render/composite.h" // projectToScreen, CameraDevice + +#include +#include +#include +#include +#include +#include + +namespace out { + + namespace axes_hidden { + + // set one opaque pixel, clipped to the full canvas [0,CW)x[0,CH) + inline void px(uint8_t* b, int CW, int CH, int x, int y, uint8_t c) { + if (x < 0 or x >= CW or y < 0 or y >= CH) { + return; + } + const std::size_t i = (static_cast(y) * CW + x) * 4; + b[i + 0] = c; + b[i + 1] = c; + b[i + 2] = c; + b[i + 3] = 255; + } + + inline void thickPx(uint8_t* b, int CW, int CH, int x, int y, int t, uint8_t c) { + for (int dy = -t; dy <= t; ++dy) { + for (int dx = -t; dx <= t; ++dx) { + px(b, CW, CH, x + dx, y + dy, c); + } + } + } + + // Bresenham line, thickness (2t+1) + inline void line(uint8_t* b, int CW, int CH, int x0, int y0, int x1, int y1, + int t, uint8_t c) { + int dx = std::abs(x1 - x0), sx = (x0 < x1) ? 1 : -1; + int dy = -std::abs(y1 - y0), sy = (y0 < y1) ? 1 : -1; + int err = dx + dy; + while (true) { + thickPx(b, CW, CH, x0, y0, t, c); + if (x0 == x1 and y0 == y1) { + break; + } + const int e2 = 2 * err; + if (e2 >= dy) { + err += dy; + x0 += sx; + } + if (e2 <= dx) { + err += dx; + y0 += sy; + } + } + } + + inline void text(uint8_t* b, int CW, int CH, int x, int y, + const std::string& str, int s, uint8_t c) { + int cx = x; + for (const char ch : str) { + const uint8_t* gl = cbar_hidden::glyph(ch); + for (int row = 0; row < 7; ++row) { + for (int col = 0; col < 5; ++col) { + if (gl[row] & (1u << (4 - col))) { + for (int dy = 0; dy < s; ++dy) { + for (int dx = 0; dx < s; ++dx) { + px(b, CW, CH, cx + col * s + dx, y + row * s + dy, c); + } + } + } + } + } + cx += 6 * s; + } + } + + // Rotated bitmap text: the baseline advances along unit (ax, ay); (ox, oy) + // is the text-local origin (top-left of the first glyph). Each glyph cell is + // oversampled 2x so rotation leaves no gaps. + inline void textRot(uint8_t* b, int CW, int CH, real_t ox, real_t oy, + const std::string& str, int s, real_t ax, real_t ay, + uint8_t c) { + const real_t dnx = -ay, dny = ax; // glyph "down" (perp. to baseline) + for (std::size_t ci = 0; ci < str.size(); ++ci) { + const uint8_t* gl = cbar_hidden::glyph(str[ci]); + const real_t base = static_cast(ci) * 6 * s; + for (int row = 0; row < 7; ++row) { + for (int col = 0; col < 5; ++col) { + if (not(gl[row] & (1u << (4 - col)))) { + continue; + } + for (int sy = 0; sy < 2 * s; ++sy) { + for (int sx = 0; sx < 2 * s; ++sx) { + const real_t u = base + col * s + static_cast(sx) * HALF; + const real_t v = row * s + static_cast(sy) * HALF; + px(b, CW, CH, + static_cast(std::lround(ox + u * ax + v * dnx)), + static_cast(std::lround(oy + u * ay + v * dny)), c); + } + } + } + } + } + } + + // rotated text centered on (cxp, cyp), baseline along unit (ax, ay) + inline void textRotCentered(uint8_t* b, int CW, int CH, real_t cxp, + real_t cyp, const std::string& str, int s, + real_t ax, real_t ay, uint8_t c) { + const real_t dnx = -ay, dny = ax; + const real_t w = static_cast(str.size()) * 6 * s; + const real_t h = 7 * s; + textRot(b, CW, CH, cxp - HALF * w * ax - HALF * h * dnx, + cyp - HALF * w * ay - HALF * h * dny, str, s, ax, ay, c); + } + + // orient a screen-space edge direction so text reads naturally (rightward + // for near-horizontal edges, upward for near-vertical ones) + inline void readableDir(real_t& ax, real_t& ay) { + const real_t n = std::sqrt(static_cast(ax * ax + ay * ay)); + if (n < static_cast(1e-9)) { + ax = ONE; + ay = ZERO; + return; + } + ax /= n; + ay /= n; + if (std::fabs(static_cast(ax)) >= std::fabs(static_cast(ay))) { + if (ax < ZERO) { + ax = -ax; + ay = -ay; + } + } else if (ay > ZERO) { + ax = -ax; + ay = -ay; + } + } + + // vertical stack of characters (top to bottom), used for the y-axis name + inline void textVert(uint8_t* b, int CW, int CH, int x, int y, + const std::string& str, int s, uint8_t c) { + int cy = y; + for (const char ch : str) { + const std::string one(1, ch); + text(b, CW, CH, x, cy, one, s, c); + cy += 8 * s; + } + } + + inline auto textW(const std::string& str, int s) -> int { + return static_cast(str.size()) * 6 * s; + } + + inline auto contrast(const real_t bg[3]) -> uint8_t { + const real_t lum = static_cast(0.299) * bg[0] + + static_cast(0.587) * bg[1] + + static_cast(0.114) * bg[2]; + return (lum < HALF) ? 255 : 0; + } + + // a "nice" number close to x (1/2/5 x 10^k) + inline auto niceNum(real_t x, bool round) -> real_t { + if (x <= ZERO) { + return ONE; + } + const real_t e = std::floor(std::log10(static_cast(x))); + const real_t f = x / static_cast(std::pow(10.0, e)); + real_t nf; + if (round) { + nf = (f < static_cast(1.5)) + ? ONE + : ((f < static_cast(3)) ? static_cast(2) + : (f < static_cast(7)) + ? static_cast(5) + : static_cast(10)); + } else { + nf = (f <= ONE) ? ONE + : (f <= static_cast(2)) + ? static_cast(2) + : (f <= static_cast(5)) ? static_cast(5) + : static_cast(10); + } + return nf * static_cast(std::pow(10.0, e)); + } + + // a tick at k*pi/D, carried with its reduced fraction k/D == n/d + struct PiTick { + real_t val; + int n, d; + }; + + // format a multiple of pi as "0", "PI", "PI", "PI/", "PI/" + // (the bitmap font has no greek glyph, so "PI" is spelled out) + inline auto fmtPi(int n, int d) -> std::string { + if (n == 0) { + return "0"; + } + std::string s; + if (n < 0) { + s += "-"; + n = -n; + } + if (n != 1) { + s += std::to_string(n); + } + s += "PI"; + if (d != 1) { + s += "/" + std::to_string(d); + } + return s; + } + + // ticks at nice fractions of pi spanning [lo, hi] + inline auto piTicks(real_t lo, real_t hi, int nticks) -> std::vector { + std::vector out; + const real_t PI = static_cast(constant::PI); + const real_t range = hi - lo; + if (not(range > ZERO) or nticks < 2) { + return out; + } + const real_t ideal = static_cast(nticks - 1) * PI / range; + const int Ds[] = { 1, 2, 3, 4, 6, 8, 12, 16, 24 }; + int D = 4; + real_t bestd = static_cast(1e30); + for (const int dd : Ds) { + const real_t df = std::fabs(static_cast(dd) - ideal); + if (df < bestd) { + bestd = df; + D = dd; + } + } + const real_t step = PI / static_cast(D); + const int k0 = static_cast(std::ceil(static_cast(lo / step) - + 1e-6)); + const int k1 = static_cast( + std::floor(static_cast(hi / step) + 1e-6)); + for (int k = k0; k <= k1; ++k) { + int n = k, d = D; + const int g = std::gcd(std::abs(n), d); + if (g > 0) { + n /= g; + d /= g; + } + out.push_back({ static_cast(k) * step, n, d }); + } + return out; + } + + inline auto niceTicks(real_t lo, real_t hi, int n) -> std::vector { + std::vector out; + if (not(hi > lo) or n < 2) { + return out; + } + const real_t step = niceNum((hi - lo) / static_cast(n - 1), true); + if (step <= ZERO) { + return out; + } + const real_t g0 = std::ceil(static_cast(lo / step)) * step; + const real_t eps = static_cast(1e-6) * step; + for (real_t v = g0; v <= hi + static_cast(0.5) * step; v += step) { + if (v >= lo - eps and v <= hi + eps) { + out.push_back((std::fabs(static_cast(v)) < eps) ? ZERO : v); + } + } + return out; + } + + } // namespace axes_hidden + + /** + * @brief Left/bottom margin (pixels) that the axes annotation needs. + * @note Zero when axes are disabled. Depends only on H, so it can size the + * canvas before drawing. + */ + inline void axesMargins(bool axes, int H, int& ml, int& mb) { + if (not axes) { + ml = 0; + mb = 0; + return; + } + const int s = cbar_hidden::scale(H); + const int cw = 6 * s; + const int ch = 8 * s; + const int tl = 5 * s; + const int gap = 2 * s; + ml = tl + gap + 7 * cw + gap + cw + gap; // ticks + numbers + y-axis name + mb = tl + gap + ch + gap + ch + gap; // ticks + numbers + x-axis name + } + + /** + * @brief Draw a 2D spine + ticks + labels around the data region. + * @param rgba canvas (CW*CH*4), opaque + * @param CW,CH canvas dimensions (includes margins) + * @param x0 left pixel of the data region (== left margin width) + * @param W,H data region dimensions + * @param u0,u1,v0,v1 world window mapped onto the data region (+v is up) + * @param xlabel,ylabel axis names + * @param bg background RGB (for contrasting text color) + * @param nticks target number of ticks per axis + */ + inline void drawAxes2D(uint8_t* rgba, + int CW, + int CH, + int x0, + int W, + int H, + real_t u0, + real_t u1, + real_t v0, + real_t v1, + const std::string& xlabel, + const std::string& ylabel, + const real_t bg[3], + int nticks) { + using namespace axes_hidden; + const uint8_t c = contrast(bg); + const int s = cbar_hidden::scale(H); + const int ch = 8 * s; + const int tl = 5 * s; + const int gap = 2 * s; + const int th = std::max(0, s / 2 - 1); // spine half-thickness + + const int xL = x0, xR = x0 + W - 1, yT = 0, yB = H - 1; + // spine + line(rgba, CW, CH, xL, yT, xR, yT, th, c); + line(rgba, CW, CH, xL, yB, xR, yB, th, c); + line(rgba, CW, CH, xL, yT, xL, yB, th, c); + line(rgba, CW, CH, xR, yT, xR, yB, th, c); + + auto X = [&](real_t u) -> int { + return static_cast(std::lround( + static_cast(xL) + + static_cast((u - u0) / (u1 - u0)) * (xR - xL))); + }; + auto Y = [&](real_t v) -> int { + return static_cast(std::lround( + static_cast(yB) - + static_cast((v - v0) / (v1 - v0)) * (yB - yT))); + }; + + // x ticks (bottom): marks + labels below the spine + for (const real_t tv : niceTicks(u0, u1, nticks)) { + const int x = X(tv); + if (x < xL or x > xR) { + continue; + } + line(rgba, CW, CH, x, yB, x, yB + tl, 0, c); + const std::string lab = cbar_hidden::fmtNum(tv); + text(rgba, CW, CH, x - textW(lab, s) / 2, yB + tl + gap, lab, s, c); + } + // y ticks (left): marks + right-aligned labels left of the spine + for (const real_t tv : niceTicks(v0, v1, nticks)) { + const int y = Y(tv); + if (y < yT or y > yB) { + continue; + } + line(rgba, CW, CH, xL, y, xL - tl, y, 0, c); + const std::string lab = cbar_hidden::fmtNum(tv); + text(rgba, CW, CH, xL - tl - gap - textW(lab, s), y - ch / 2, lab, s, c); + } + // axis names + if (not xlabel.empty()) { + text(rgba, CW, CH, xL + W / 2 - textW(xlabel, s) / 2, + yB + tl + gap + ch + gap, xlabel, s, c); + } + if (not ylabel.empty()) { + textVert(rgba, CW, CH, std::max(gap, x0 - tl - gap - 7 * (6 * s) - gap - 6 * s), + (yT + yB) / 2 - 4 * s * static_cast(ylabel.size()) / 2, + ylabel, s, c); + } + } + + /** + * @brief Draw polar (curvilinear) axes for a 2D spherical meridional slice. + * @param x0,W,H data region (the slice maps world (X = r sin th, Z = r cos th) + * onto it via the [u0,u1]x[v0,v1] window, aspect-matched) + * @param rmin,rmax,tmin,tmax global (r, theta) extent + * @param mirror whether the half-plane is mirrored into a full disk + * @param rlabel,tlabel names for the radial / angular axes (e.g. "R","Theta") + * @note Draws (1) a curvilinear spine: the outer & inner arcs plus the two + * radial edges (or full arcs when mirrored); (2) an "R" radial axis along the + * symmetry axis (X=0) with R=0 at the center, increasing outward; (3) a + * "Theta" angular axis with ticks along the outer arc. + */ + inline void drawAxesPolar(uint8_t* rgba, + int CW, + int CH, + int x0, + int W, + int H, + real_t u0, + real_t u1, + real_t v0, + real_t v1, + real_t rmin, + real_t rmax, + real_t tmin, + real_t tmax, + bool mirror, + const std::string& rlabel, + const std::string& tlabel, + const real_t bg[3], + int nticks) { + using namespace axes_hidden; + const uint8_t c = contrast(bg); + const int s = cbar_hidden::scale(H); + const int ch = 8 * s; + const int tl = 5 * s; + const int gap = 2 * s; + + auto WX = [&](real_t X) -> real_t { + return static_cast(x0) + (X - u0) / (u1 - u0) * W - HALF; + }; + auto WZ = [&](real_t Z) -> real_t { + return (v1 - Z) / (v1 - v0) * H - HALF; + }; + auto PX = [&](real_t X, real_t Z, int& qx, int& qy) { + qx = static_cast(std::lround(WX(X))); + qy = static_cast(std::lround(WZ(Z))); + }; + + const int NA = 160; + auto arc = [&](real_t r, real_t sgn) { + int qx, qy; + PX(sgn * r * std::sin(static_cast(tmin)), + r * std::cos(static_cast(tmin)), qx, qy); + for (int i = 1; i <= NA; ++i) { + const real_t th = tmin + (tmax - tmin) * i / NA; + int rx, ry; + PX(sgn * r * std::sin(static_cast(th)), + r * std::cos(static_cast(th)), rx, ry); + line(rgba, CW, CH, qx, qy, rx, ry, 0, c); + qx = rx; + qy = ry; + } + }; + auto ray = [&](real_t th) { + int ax, ay, bx, by; + PX(rmin * std::sin(static_cast(th)), + rmin * std::cos(static_cast(th)), ax, ay); + PX(rmax * std::sin(static_cast(th)), + rmax * std::cos(static_cast(th)), bx, by); + line(rgba, CW, CH, ax, ay, bx, by, 0, c); + }; + + // ---- curvilinear spine -------------------------------------------- // + arc(rmax, ONE); + arc(rmin, ONE); + if (mirror) { + arc(rmax, -ONE); + arc(rmin, -ONE); + } else { + ray(tmin); + ray(tmax); + } + + // ---- R axis: along the symmetry axis (X = 0), zero at the center --- // + const int rmaxlabW = textW(cbar_hidden::fmtNum(rmax), s); + for (const real_t Rv : niceTicks(ZERO, rmax, nticks)) { + for (int sg = 1; sg >= -1; sg -= 2) { + if (sg < 0 and Rv == ZERO) { + continue; // a single tick at the center + } + int ax, ay; + PX(ZERO, static_cast(sg) * Rv, ax, ay); + line(rgba, CW, CH, ax, ay, ax - tl, ay, 0, c); + const std::string lab = cbar_hidden::fmtNum(Rv); + text(rgba, CW, CH, ax - tl - gap - textW(lab, s), ay - ch / 2, lab, s, c); + } + } + if (not rlabel.empty()) { + int ax, ay; + PX(ZERO, ZERO, ax, ay); + const real_t off = tl + gap + rmaxlabW + gap + ch; + textRotCentered(rgba, CW, CH, ax - off, static_cast(ay), rlabel, s, + ZERO, -ONE, c); + } + + // ---- Theta axis: ticks + labels (fractions of pi) along the arc --- // + // widest tick label, so the "Theta" name can clear them all + real_t maxlw = static_cast(ch); + for (const auto& tk : piTicks(tmin, tmax, nticks)) { + maxlw = std::max(maxlw, + static_cast(textW(fmtPi(tk.n, tk.d), s))); + } + for (const auto& tk : piTicks(tmin, tmax, nticks)) { + const real_t ox = std::sin(static_cast(tk.val)); + const real_t oz = std::cos(static_cast(tk.val)); + int px0, py0; + PX(rmax * ox, rmax * oz, px0, py0); + const real_t dxp = ox, dyp = -oz; // outward pixel direction + line(rgba, CW, CH, px0, py0, + static_cast(std::lround(px0 + dxp * tl)), + static_cast(std::lround(py0 + dyp * tl)), 0, c); + const std::string lab = fmtPi(tk.n, tk.d); + // push the (horizontal) label box fully clear of the arc/tick at any + // angle: offset its center by its own support along the outward direction + const real_t inset = HALF * (static_cast(textW(lab, s)) * + std::fabs(static_cast(dxp)) + + static_cast(ch) * + std::fabs(static_cast(dyp))); + const real_t lo = tl + gap + inset; + text(rgba, CW, CH, + static_cast(std::lround(px0 + dxp * lo)) - textW(lab, s) / 2, + static_cast(std::lround(py0 + dyp * lo)) - ch / 2, lab, s, c); + } + if (not tlabel.empty()) { + const real_t tm = HALF * (tmin + tmax); + const real_t ox = std::sin(static_cast(tm)); + const real_t oz = std::cos(static_cast(tm)); + int px0, py0; + PX(rmax * ox, rmax * oz, px0, py0); + real_t adx = std::cos(static_cast(tm)); // arc tangent + real_t ady = std::sin(static_cast(tm)); + readableDir(adx, ady); + // beyond the tick labels (which reach ~tl+gap+maxlw from the arc) + const real_t off = tl + gap + maxlw + gap + ch; + textRotCentered(rgba, CW, CH, px0 + ox * off, py0 - oz * off, tlabel, s, + adx, ady, c); + } + } + + /** + * @brief Draw a 3D bounding-box wireframe spine + ticks + labels. + * @param rgba canvas (CW*CH*4), opaque + * @param CW,CH canvas dimensions + * @param x0 left pixel of the data region (left margin width) + * @param W,H data region dimensions (used for the camera projection) + * @param cam ray-march camera (inverted to project world -> screen) + * @param ext global box extent (3 axes) + * @param lab axis names [x, y, z] + * @param bg background RGB + * @param nticks target number of ticks per axis + */ + inline void drawAxes3D(uint8_t* rgba, + int CW, + int CH, + int x0, + int W, + int H, + const CameraDevice& cam, + const boundaries_t& ext, + const std::string lab[3], + const real_t bg[3], + int nticks) { + using namespace axes_hidden; + if (ext.size() < 3) { + return; + } + const uint8_t c = contrast(bg); + const int s = cbar_hidden::scale(H); + const int ch = 8 * s; + const int tl = 5 * s; + + auto corner = [&](int m, real_t p[3]) { + p[0] = (m & 1) ? ext[0].second : ext[0].first; + p[1] = (m & 2) ? ext[1].second : ext[1].first; + p[2] = (m & 4) ? ext[2].second : ext[2].first; + }; + real_t cx[8], cy[8]; + bool ok[8]; + for (int m = 0; m < 8; ++m) { + real_t p[3]; + corner(m, p); + real_t a, b; + ok[m] = projectToScreen(cam, W, H, p, a, b); + cx[m] = a + static_cast(x0); + cy[m] = b; + } + // box centroid (for outward tick/label direction) + real_t cen[3] = { HALF * (ext[0].first + ext[0].second), + HALF * (ext[1].first + ext[1].second), + HALF * (ext[2].first + ext[2].second) }; + real_t ccx = ZERO, ccy = ZERO; + { + real_t a, b; + projectToScreen(cam, W, H, cen, a, b); + ccx = a + static_cast(x0); + ccy = b; + } + + (void)ok; + // The wireframe "spine" is drawn in the ray-march (depth-occluded), so here + // we only annotate. For each axis, pick one *silhouette* edge (its two + // adjacent faces face opposite ways) and, among the two candidates, the one + // whose screen position matches the convention x=bottom, y & z on the left + // of the default diagonal view. + auto frontFace = [&](int axis, int side) -> bool { + const real_t nrm = (side != 0) ? ONE : -ONE; // outward normal sign + return (nrm * (-cam.forward[axis])) > ZERO; // points toward the camera? + }; + for (int d = 0; d < 3; ++d) { + const int e1 = (d == 0) ? 1 : 0; + const int e2 = (d == 2) ? 1 : 2; + int bs1 = 0, bs2 = 0; + bool found = false; + real_t best = ZERO; + for (int s1 = 0; s1 < 2; ++s1) { + for (int s2 = 0; s2 < 2; ++s2) { + const int m0 = (s1 << e1) | (s2 << e2); + const int m1 = m0 | (1 << d); + const real_t mx = HALF * (cx[m0] + cx[m1]); + const real_t my = HALF * (cy[m0] + cy[m1]); + // screen-position convention: x on the bottom, y & z on the left edges + real_t score = (d == 0) ? my : -mx; + if (frontFace(e1, s1) != frontFace(e2, s2)) { + score += static_cast(1e6); // strongly prefer silhouette edges + } + if (not found or score > best) { + best = score; + bs1 = s1; + bs2 = s2; + found = true; + } + } + } + const int m0 = (bs1 << e1) | (bs2 << e2); + const int m1 = m0 | (1 << d); + real_t o[3]; + corner(m0, o); // perpendicular coords fixed; axis d swept for ticks + const real_t lo = ext[d].first, hi = ext[d].second; + // screen-space perpendicular to the edge, flipped to point outward + real_t ex = cx[m1] - cx[m0], ey = cy[m1] - cy[m0]; + real_t el = std::sqrt(static_cast(ex * ex + ey * ey)); + if (el < ONE) { + el = ONE; + } + ex /= el; + ey /= el; + real_t pxd = -ey, pyd = ex; + { + const real_t mxv = HALF * (cx[m0] + cx[m1]) - ccx; + const real_t myv = HALF * (cy[m0] + cy[m1]) - ccy; + if (pxd * mxv + pyd * myv < ZERO) { + pxd = -pxd; + pyd = -pyd; + } + } + // readable text baseline aligned with the edge direction + real_t adx = ex, ady = ey; + readableDir(adx, ady); + const real_t numOff = static_cast(tl) + 5 * s; // number center + const real_t nameOff = static_cast(tl) + 14 * s; // axis-name center + // ticks + numeric labels (numbers rotated along the edge for x & y; the + // vertical z edge keeps horizontal numbers, which read more easily) + for (const real_t tv : niceTicks(lo, hi, nticks)) { + real_t p[3] = { o[0], o[1], o[2] }; + p[d] = tv; + real_t a, b; + if (not projectToScreen(cam, W, H, p, a, b)) { + continue; + } + a += static_cast(x0); + const int mx = static_cast(std::lround(a + pxd * tl)); + const int my = static_cast(std::lround(b + pyd * tl)); + line(rgba, CW, CH, static_cast(std::lround(a)), + static_cast(std::lround(b)), mx, my, 0, c); + const std::string l2 = cbar_hidden::fmtNum(tv); + if (d == 2) { + const int tx = (pxd < ZERO) ? (mx - textW(l2, s)) : mx; + text(rgba, CW, CH, tx, my - ch / 2, l2, s, c); + } else { + textRotCentered(rgba, CW, CH, a + pxd * numOff, b + pyd * numOff, l2, + s, adx, ady, c); + } + } + // axis name at the MIDDLE of the edge (near the central tick), aligned + // with the edge and pushed further outward than the numbers + if (not lab[d].empty()) { + real_t mid[3] = { o[0], o[1], o[2] }; + mid[d] = HALF * (lo + hi); + real_t a, b; + if (projectToScreen(cam, W, H, mid, a, b)) { + a += static_cast(x0); + textRotCentered(rgba, CW, CH, a + pxd * nameOff, b + pyd * nameOff, + lab[d], s, adx, ady, c); + } + } + } + } + +} // namespace out + +#endif // OUTPUT_RENDER_AXES_H diff --git a/src/output/render/raymarch.hpp b/src/output/render/raymarch.hpp index f10702b47..e0db322f3 100644 --- a/src/output/render/raymarch.hpp +++ b/src/output/render/raymarch.hpp @@ -61,6 +61,12 @@ namespace kernel { const bool log_scale; const real_t early_alpha; + // global box wireframe "spine": opaque (alpha 1) segments composited inline + // during the march, so the volume occludes the far edges. radius <= 0 off. + const real_t glo0, glo1, glo2, ghi0, ghi1, ghi2; + const real_t spine_radius; + const real_t spine_cr, spine_cg, spine_cb; + array_t image; // output, (bw*bh, 4) premultiplied RGBA public: @@ -86,6 +92,10 @@ namespace kernel { real_t vmax_, bool log_scale_, real_t early_alpha_, + const real_t glo[3], + const real_t ghi[3], + real_t spine_radius_, + const real_t spine_rgb[3], const array_t& image_) : Fld { Fld_ } , comp { comp_ } @@ -113,8 +123,54 @@ namespace kernel { , vmax { vmax_ } , log_scale { log_scale_ } , early_alpha { early_alpha_ } + , glo0 { glo[0] } + , glo1 { glo[1] } + , glo2 { glo[2] } + , ghi0 { ghi[0] } + , ghi1 { ghi[1] } + , ghi2 { ghi[2] } + , spine_radius { spine_radius_ } + , spine_cr { spine_rgb[0] } + , spine_cg { spine_rgb[1] } + , spine_cb { spine_rgb[2] } , image { image_ } {} + // distance test: is world point (px,py,pz) within `spine_radius` of any of + // the 12 global-box edges? (nearest parallel edge per axis == nearest + // perpendicular-plane corner) + Inline auto onSpine(real_t px, real_t py, real_t pz) const -> bool { + if (spine_radius <= ZERO) { + return false; + } + const real_t r2 = spine_radius * spine_radius; + const real_t pad = spine_radius; + // edges parallel to x (perp plane = y,z) + if (px >= glo0 - pad and px <= ghi0 + pad) { + const real_t dy = math::min(math::abs(py - glo1), math::abs(py - ghi1)); + const real_t dz = math::min(math::abs(pz - glo2), math::abs(pz - ghi2)); + if (dy * dy + dz * dz < r2) { + return true; + } + } + // edges parallel to y (perp plane = x,z) + if (py >= glo1 - pad and py <= ghi1 + pad) { + const real_t dx = math::min(math::abs(px - glo0), math::abs(px - ghi0)); + const real_t dz = math::min(math::abs(pz - glo2), math::abs(pz - ghi2)); + if (dx * dx + dz * dz < r2) { + return true; + } + } + // edges parallel to z (perp plane = x,y) + if (pz >= glo2 - pad and pz <= ghi2 + pad) { + const real_t dx = math::min(math::abs(px - glo0), math::abs(px - ghi0)); + const real_t dy = math::min(math::abs(py - glo1), math::abs(py - ghi1)); + if (dx * dx + dy * dy < r2) { + return true; + } + } + return false; + } + // trilinear sample of the prepared scalar at world point p, reading the // ghost halo for corners just outside the active box. Inline auto sample(real_t px, real_t py, real_t pz) const -> real_t { @@ -261,31 +317,41 @@ namespace kernel { real_t acc_r = ZERO, acc_g = ZERO, acc_b = ZERO, acc_a = ZERO; int steps = 0; while (t < t_exit and steps < max_steps) { - const real_t s = sample(ox + t * dx, oy + t * dy, oz + t * dz); - // normalize through the transfer function range - real_t u; - if (log_scale) { - u = (s > ZERO) ? (math::log10(s) - log_vmin) * inv_range - : -ONE; + const real_t px = ox + t * dx, py = oy + t * dy, pz = oz + t * dz; + real_t cr, cg, cb, ca; + if (onSpine(px, py, pz)) { + // opaque box edge (premultiplied; alpha == 1 -> color is straight RGB). + // Composited inline so accumulated foreground volume occludes it. + cr = spine_cr; + cg = spine_cg; + cb = spine_cb; + ca = ONE; } else { - u = (s - vmin) * inv_range; - } - if (u < ZERO) { - u = ZERO; - } else if (u > ONE) { - u = ONE; - } - int idx = static_cast(u * static_cast(n_lut - 1) + HALF); - if (idx < 0) { - idx = 0; - } else if (idx > n_lut - 1) { - idx = n_lut - 1; + const real_t s = sample(px, py, pz); + // normalize through the transfer function range + real_t u; + if (log_scale) { + u = (s > ZERO) ? (math::log10(s) - log_vmin) * inv_range : -ONE; + } else { + u = (s - vmin) * inv_range; + } + if (u < ZERO) { + u = ZERO; + } else if (u > ONE) { + u = ONE; + } + int idx = static_cast(u * static_cast(n_lut - 1) + HALF); + if (idx < 0) { + idx = 0; + } else if (idx > n_lut - 1) { + idx = n_lut - 1; + } + cr = lut(idx, 0); // premultiplied + cg = lut(idx, 1); + cb = lut(idx, 2); + ca = lut(idx, 3); } - const real_t cr = lut(idx, 0); // premultiplied - const real_t cg = lut(idx, 1); - const real_t cb = lut(idx, 2); - const real_t ca = lut(idx, 3); - const real_t w = ONE - acc_a; + const real_t w = ONE - acc_a; acc_r += w * cr; acc_g += w * cg; acc_b += w * cb; diff --git a/src/output/render/renderer.cpp b/src/output/render/renderer.cpp index 128135895..ebda438f5 100644 --- a/src/output/render/renderer.cpp +++ b/src/output/render/renderer.cpp @@ -8,6 +8,7 @@ #include "utils/log.h" #include "utils/numeric.h" +#include "output/render/axes.h" #include "output/render/colorbar.h" #include "output/render/composite.h" #include "output/render/png.h" @@ -112,6 +113,24 @@ namespace out { // 2D slice mode (spherical only): mirror the half-plane into a full disk m_mirror = toml::find_or(td, "output", "render", "mirror", true); + // axes: spine + ticks + labels around the rendered region + m_axes = toml::find_or(td, "output", "render", "axes", false); + m_axis_nticks = toml::find_or(td, "output", "render", "axis_ticks", 5); + m_spine_width = toml::find_or(td, "output", "render", "spine_width", + static_cast(2)); + m_global_extent = global_extent; + { + const auto al = toml::find_or>( + td, "output", "render", "axis_labels", std::vector {}); + m_axis_labels_set = not al.empty(); + for (std::size_t d = 0; d < al.size() and d < 3; ++d) { + m_axis_labels[d] = al[d]; + } + // default 2D slice names track the labels (overridden per-metric by Render) + m_slice_xlabel = m_axis_labels[0]; + m_slice_ylabel = m_axis_labels[1]; + } + // cadence: mirror output.* (interval in steps; interval_time in sim time) const auto interval = toml::find_or(td, "output", @@ -313,28 +332,50 @@ namespace out { } // composite the premultiplied image over the opaque background: // out = src_premult + (1 - src_alpha) * background, alpha = opaque. - std::vector bytes(n); + std::vector data(n); for (std::size_t p = 0; p < npix; ++p) { const real_t a = img[p * 4 + 3]; const real_t inv = ONE - a; - bytes[p * 4 + 0] = quantize(img[p * 4 + 0] + inv * m_background[0]); - bytes[p * 4 + 1] = quantize(img[p * 4 + 1] + inv * m_background[1]); - bytes[p * 4 + 2] = quantize(img[p * 4 + 2] + inv * m_background[2]); - bytes[p * 4 + 3] = 255; + data[p * 4 + 0] = quantize(img[p * 4 + 0] + inv * m_background[0]); + data[p * 4 + 1] = quantize(img[p * 4 + 1] + inv * m_background[1]); + data[p * 4 + 2] = quantize(img[p * 4 + 2] + inv * m_background[2]); + data[p * 4 + 3] = 255; } const auto fname = dir / fmt::format("%s%08lu.png", scene.prefix.c_str(), static_cast(step)); + + auto drawBar = [&](uint8_t* buf, int bw, int bh) { + if (m_colorbar) { + drawColorbar(buf, bw, bh, scene.tf.colormap, scene.tf.vmin, + scene.tf.vmax, scene.tf.log_scale, scene.label, + m_background, scene.ticks); + } + }; + + // canvas margins: axes (left + bottom) and the colorbar strip (right). + // The data region sits at (ml, 0); margins/strip are background-filled. + // The polar (curvilinear) overlay annotates inside the data region (the + // disk is centered with background around it), so it needs no margins. + const bool polar = (m_global_extent.size() == 2) and m_slice_polar; + int ml = 0, mb = 0; + out::axesMargins(m_axes and not polar, m_height, ml, mb); + const int strip = (m_colorbar and m_colorbar_outside) + ? colorbarBlockWidth(m_height) + : 0; + const int CW = ml + m_width + strip; + const int CH = m_height + mb; + bool ok = true; - if (m_colorbar and m_colorbar_outside) { - // extend the canvas to the right so the colorbar sits in its own margin, - // outside the rendered volume. - const int strip = colorbarBlockWidth(m_height); - const int CW = m_width + strip; - const uint8_t bR = quantize(m_background[0]); - const uint8_t bG = quantize(m_background[1]); - const uint8_t bB = quantize(m_background[2]); - std::vector canvas(static_cast(CW) * m_height * 4); + if (CW == m_width and CH == m_height and not m_axes) { + // no margins, no outside strip, no overlay: colorbar overlays the data + drawBar(data.data(), m_width, m_height); + ok = write_png(fname, m_width, m_height, data.data()); + } else { + const uint8_t bR = quantize(m_background[0]); + const uint8_t bG = quantize(m_background[1]); + const uint8_t bB = quantize(m_background[2]); + std::vector canvas(static_cast(CW) * CH * 4); for (std::size_t i = 0; i < canvas.size(); i += 4) { canvas[i + 0] = bR; canvas[i + 1] = bG; @@ -342,35 +383,31 @@ namespace out { canvas[i + 3] = 255; } for (int y = 0; y < m_height; ++y) { - std::copy_n(&bytes[static_cast(y) * m_width * 4], - static_cast(m_width) * 4, - &canvas[static_cast(y) * CW * 4]); + std::copy_n( + &data[static_cast(y) * m_width * 4], + static_cast(m_width) * 4, + &canvas[(static_cast(y) * CW + ml) * 4]); } - drawColorbar(canvas.data(), - CW, - m_height, - scene.tf.colormap, - scene.tf.vmin, - scene.tf.vmax, - scene.tf.log_scale, - scene.label, - m_background, - scene.ticks); - ok = write_png(fname, CW, m_height, canvas.data()); - } else { - if (m_colorbar) { - drawColorbar(bytes.data(), - m_width, - m_height, - scene.tf.colormap, - scene.tf.vmin, - scene.tf.vmax, - scene.tf.log_scale, - scene.label, - m_background, - scene.ticks); + if (m_axes) { + if (m_global_extent.size() == 3) { + out::drawAxes3D(canvas.data(), CW, CH, ml, m_width, m_height, + m_camera_dev, m_global_extent, m_axis_labels, + m_background, m_axis_nticks); + } else if (polar) { + out::drawAxesPolar(canvas.data(), CW, CH, ml, m_width, m_height, + m_slice_win[0], m_slice_win[1], m_slice_win[2], + m_slice_win[3], m_slice_rmin, m_slice_rmax, + m_slice_tmin, m_slice_tmax, m_slice_pmirror, "R", + "Theta", m_background, m_axis_nticks); + } else { + out::drawAxes2D(canvas.data(), CW, CH, ml, m_width, m_height, + m_slice_win[0], m_slice_win[1], m_slice_win[2], + m_slice_win[3], m_slice_xlabel, m_slice_ylabel, + m_background, m_axis_nticks); + } } - ok = write_png(fname, m_width, m_height, bytes.data()); + drawBar(canvas.data(), CW, CH); + ok = write_png(fname, CW, CH, canvas.data()); } if (not ok) { raise::Warning( diff --git a/src/output/render/renderer.h b/src/output/render/renderer.h index 15b64b624..15210f61e 100644 --- a/src/output/render/renderer.h +++ b/src/output/render/renderer.h @@ -171,6 +171,68 @@ namespace out { return m_mirror; } + [[nodiscard]] + auto axes() const -> bool { + return m_axes; + } + + [[nodiscard]] + auto background(int i) const -> real_t { + return m_background[(i < 0) ? 0 : ((i > 2) ? 2 : i)]; + } + + // target 3D spine line width in pixels + [[nodiscard]] + auto spineWidth() const -> real_t { + return m_spine_width; + } + + // whether `output.render.axis_labels` was set in the toml (so the 2D path + // honors it instead of substituting per-metric defaults) + [[nodiscard]] + auto axisLabelsSet() const -> bool { + return m_axis_labels_set; + } + + [[nodiscard]] + auto axisLabel(int d) const -> const std::string& { + return m_axis_labels[(d < 0) ? 0 : ((d > 2) ? 2 : d)]; + } + + // Set the world window + axis names the 2D slice path maps onto the image, + // so the (host) axes overlay can label spatial coordinates. Called by the + // templated 2D Render before compositing; the window is constant per run. + void setSliceFrame(real_t u0, + real_t u1, + real_t v0, + real_t v1, + const std::string& xlabel, + const std::string& ylabel) { + m_slice_win[0] = u0; + m_slice_win[1] = u1; + m_slice_win[2] = v0; + m_slice_win[3] = v1; + m_slice_xlabel = xlabel; + m_slice_ylabel = ylabel; + } + + // Mark the 2D slice as curvilinear (spherical) so the axes are drawn polar: + // a radial "R" axis on the symmetry axis + a "Theta" arc, with a curvilinear + // spine. Set per-frame by the templated 2D Render (constant per run). + void setSlicePolar(bool polar, + real_t rmin, + real_t rmax, + real_t tmin, + real_t tmax, + bool mir) { + m_slice_polar = polar; + m_slice_rmin = rmin; + m_slice_rmax = rmax; + m_slice_tmin = tmin; + m_slice_tmax = tmax; + m_slice_pmirror = mir; + } + [[nodiscard]] auto scenes() const -> const std::vector& { return m_scenes; @@ -196,6 +258,24 @@ namespace out { // 2D slice mode (spherical): mirror the meridional half-plane across the // symmetry axis to render a full disk from one axisymmetric half bool m_mirror { true }; + // draw a spine (frame) + axis ticks/labels around the rendered region + bool m_axes { false }; + bool m_axis_labels_set { false }; + int m_axis_nticks { 5 }; + real_t m_spine_width { static_cast(2) }; // 3D spine width (px) + std::string m_axis_labels[3] { "x", "y", "z" }; + // global world box (2 or 3 axes); used to project the 3D axes box and to + // know the render mode (size 2 => 2D slice, size 3 => 3D volume). + boundaries_t m_global_extent; + // 2D slice world window + axis names, set per-frame by the templated Render + real_t m_slice_win[4] { ZERO, ONE, ZERO, ONE }; + std::string m_slice_xlabel { "x" }; + std::string m_slice_ylabel { "y" }; + // 2D curvilinear (spherical) slice: draw polar axes instead of Cartesian + bool m_slice_polar { false }; + real_t m_slice_rmin { ZERO }, m_slice_rmax { ONE }; + real_t m_slice_tmin { ZERO }, m_slice_tmax { ONE }; + bool m_slice_pmirror { false }; CameraDevice m_camera_dev; std::vector m_scenes; From 5a0bffc6984234b2da64e9425663e674a8447921 Mon Sep 17 00:00:00 2001 From: Ludwig Boess Date: Tue, 30 Jun 2026 19:12:46 -0400 Subject: [PATCH 046/125] 3D field lines --- input.example.toml | 77 ++++ src/framework/domain/metadomain_render.cpp | 210 +++++++++- src/output/render/fieldlines.h | 457 +++++++++++++++++++++ src/output/render/raymarch.hpp | 137 +++++- src/output/render/renderer.cpp | 51 +++ src/output/render/renderer.h | 64 ++- 6 files changed, 979 insertions(+), 17 deletions(-) create mode 100644 src/output/render/fieldlines.h diff --git a/input.example.toml b/input.example.toml index 527f5e9de..78ff1fc45 100644 --- a/input.example.toml +++ b/input.example.toml @@ -848,6 +848,8 @@ # "N_1", "Rho_2", "Txy_1_2", "V1_3"; default = all massive species # @note: components are signed; pair a symmetric `min`/`max` with a # diverging colormap ("cool2warm") to center zero + # @note: "fieldlines" renders the magnetic field-line tubes on their own + # (no scalar volume sampled); see [output.render.fieldlines] below field = "" # PNG filename prefix; files are `.png` # @type: string @@ -888,6 +890,81 @@ # @note: Values outside [min, max] are skipped # @example: [0.0, 0.5, 1.0] colorbar_ticks = "" + # Overlay the magnetic field-line tubes inside this scene's volume + # @type: bool + # @default: false + # @note: requires the [output.render.fieldlines] table below (3D only). + # A scene with field = "fieldlines" instead renders them alone. + fieldlines = "" + + # Magnetic field-line tubes (3D volume mode only). Traces field lines through + # a coarse, MPI-replicated copy of the field and draws them as solid tubes, + # colored by |field|, composited inside the same ray-march as the volume so + # they are correctly occluded by it. Built once per frame and shared by every + # scene that opts in (per-scene `fieldlines = true`) and by any standalone + # `field = "fieldlines"` scene. The coarsening is what makes the lines cheap + # to trace (no parallel particle advection) and gives a smoothed morphology. + [output.render.fieldlines] + # Build the field-line geometry this run + # @type: bool + # @default: false + # @note: implied true if any scene sets `fieldlines = true` or uses + # `field = "fieldlines"` + enable = "" + # Vector field to trace + # @type: string + # @enum: "B", "E", "J" + # @default: "B" + field = "" + # Field coarsening factor (simulation cells per coarse cell, per axis) + # @type: int [1..16] + # @default: 4 + # @note: larger = smoother "morphology" lines + cheaper replication + # (the coarse field is ~ N_cells / bin^3 x 3 floats, on every rank) + bin = "" + # Seed-lattice spacing in screen pixels (sets line density) + # @type: float [> 0] + # @default: 8 + # @note: capped by `seed_max`; if seed_px asks for more seeds than that, + # the spacing grows to fit and seed_px no longer governs + seed_px = "" + # Hard cap on the seed count (the seed lattice is n^3, 2 lines per seed) + # @type: int [> 0] + # @default: 4096 + # @note: lower this for fewer / more widely spaced lines + seed_max = "" + # Tube radius in screen pixels + # @type: float [> 0] + # @default: 2 + tube_px = "" + # Tube colormap (mapped by |field| along the line) + # @type: string + # @enum: "viridis", "inferno", "plasma", "cool2warm", "gray" + # @default: "inferno" + colormap = "" + # Map the tube color range logarithmically + # @type: bool + # @default: false + # @note: requires min > 0 + log = "" + # Tube color range (lower / upper bound on |field|) + # @type: float + # @default: 0.0 + # @note: when min >= max, the range is auto-set from |field| along the lines + min = "" + max = "" + # RK4 integration step as a fraction of one coarse cell + # @type: float [> 0] + # @default: 0.5 + step_frac = "" + # Per-direction integration-step cap + # @type: int [> 0] + # @default: 4000 + max_steps = "" + # Maximum line length, in global box diagonals (per direction) + # @type: float [> 0] + # @default: 3.0 + max_length = "" [checkpoint] # Number of timesteps between checkpoints diff --git a/src/framework/domain/metadomain_render.cpp b/src/framework/domain/metadomain_render.cpp index f9320a094..da3127f53 100644 --- a/src/framework/domain/metadomain_render.cpp +++ b/src/framework/domain/metadomain_render.cpp @@ -36,6 +36,7 @@ #include "kernels/fields_to_phys.hpp" #include "kernels/particle_moments.hpp" #include "output/render/composite.h" +#include "output/render/fieldlines.h" #include "output/render/raymarch.hpp" #include "output/render/reduce.hpp" #include "output/render/slice2d.hpp" @@ -43,6 +44,12 @@ #include #include +#if defined(MPI_ENABLED) + #include "arch/mpi_aliases.h" + + #include +#endif + #include #include #include @@ -124,6 +131,120 @@ namespace ntt { } } + // Volume-average this domain's physical-basis vector field (B/E/J) onto a + // GLOBAL coarse grid, then MPI-replicate it so every rank holds the same + // field and can trace identical global field lines locally. `bckp` is used + // as scratch (overwritten). 3D only (the field-line renderer is Cartesian). + template + auto buildCoarseFieldVec(const Mesh& mesh, + const Fields& fields, + ndfield_t& bckp, + char fbase, + const real_t gorigin[3], + const int gnc[3], + const real_t gdx[3]) -> out::CoarseField { + const auto metric = mesh.metric; + // raw vector components -> bckp(0,1,2) + uint8_t src_base = em::bx1; + PrepareOutputFlags interp = PrepareOutput::InterpToCellCenterFromFaces; + bool is_current = false; + if (fbase == 'E') { + src_base = em::ex1; + interp = PrepareOutput::InterpToCellCenterFromEdges; + } else if (fbase == 'J') { + is_current = true; + src_base = cur::jx1; + interp = PrepareOutput::InterpToCellCenterFromEdges; + } + if (is_current) { + copyVec3ToBckp(fields.cur, bckp, + cell_range_t(cur::jx1, cur::jx3 + 1)); + } else { + copyVec3ToBckp(fields.em, bckp, + cell_range_t(src_base, src_base + 3)); + } + // interpolate to cell centers + convert to physical basis -> bckp(3,4,5) + const PrepareOutputFlags prepare = (S == SimEngine::SRPIC) + ? PrepareOutput::ConvertToHat + : PrepareOutput::ConvertToPhysCntrv; + list_t comp_from = { 0, 1, 2 }; + list_t comp_to = { 3, 4, 5 }; + Kokkos::parallel_for( + "RenderFLFieldsToPhys", + mesh.rangeActiveCells(), + kernel::FieldsToPhys_kernel(bckp, bckp, comp_from, comp_to, + interp | prepare, metric)); + Kokkos::fence(); + + // pull the physical components to host and bin into the coarse grid + auto bckp_h = Kokkos::create_mirror_view(bckp); + Kokkos::deep_copy(bckp_h, bckp); + + const std::size_t ncell = static_cast(gnc[0]) * + static_cast(gnc[1]) * + static_cast(gnc[2]); + std::vector sum(ncell * 3, ZERO); + std::vector cnt(ncell, ZERO); + + const auto le = mesh.extent(); + const real_t llo[3] = { le[0].first, le[1].first, le[2].first }; + const real_t lsz[3] = { le[0].second - le[0].first, + le[1].second - le[1].first, + le[2].second - le[2].first }; + const int nl[3] = { static_cast(mesh.n_active(in::x1)), + static_cast(mesh.n_active(in::x2)), + static_cast(mesh.n_active(in::x3)) }; + const int NG = static_cast(N_GHOSTS); + for (int k = 0; k < nl[2]; ++k) { + for (int j = 0; j < nl[1]; ++j) { + for (int i = 0; i < nl[0]; ++i) { + const real_t world[3] = { + llo[0] + (static_cast(i) + HALF) * lsz[0] / nl[0], + llo[1] + (static_cast(j) + HALF) * lsz[1] / nl[1], + llo[2] + (static_cast(k) + HALF) * lsz[2] / nl[2] + }; + int c[3]; + for (int d = 0; d < 3; ++d) { + int cc = static_cast( + std::floor((world[d] - gorigin[d]) / gdx[d])); + cc = (cc < 0) ? 0 : ((cc > gnc[d] - 1) ? gnc[d] - 1 : cc); + c[d] = cc; + } + const std::size_t lin = (static_cast(c[2]) * gnc[1] + + c[1]) * + gnc[0] + + c[0]; + sum[lin * 3 + 0] += bckp_h(i + NG, j + NG, k + NG, 3); + sum[lin * 3 + 1] += bckp_h(i + NG, j + NG, k + NG, 4); + sum[lin * 3 + 2] += bckp_h(i + NG, j + NG, k + NG, 5); + cnt[lin] += ONE; + } + } + } +#if defined(MPI_ENABLED) + MPI_Allreduce(MPI_IN_PLACE, sum.data(), static_cast(ncell * 3), + mpi::get_type(), MPI_SUM, MPI_COMM_WORLD); + MPI_Allreduce(MPI_IN_PLACE, cnt.data(), static_cast(ncell), + mpi::get_type(), MPI_SUM, MPI_COMM_WORLD); +#endif + out::CoarseField cf; + cf.B.assign(ncell * 3, ZERO); + for (int d = 0; d < 3; ++d) { + cf.n[d] = gnc[d]; + cf.origin[d] = gorigin[d]; + cf.dx[d] = gdx[d]; + } + for (std::size_t c = 0; c < ncell; ++c) { + if (cnt[c] > ZERO) { + const real_t inv = ONE / cnt[c]; + cf.B[c * 3 + 0] = sum[c * 3 + 0] * inv; + cf.B[c * 3 + 1] = sum[c * 3 + 1] * inv; + cf.B[c * 3 + 2] = sum[c * 3 + 2] * inv; + } + } + return cf; + } + } // namespace template @@ -503,16 +624,89 @@ namespace ntt { int bx0 = 0, by0 = 0, bw = 0, bh = 0; const bool on_screen = out::screenBBox(cam, W, H, lo, hi, bx0, by0, bw, bh); + // ---- magnetic-field-line tubes (built once, shared by every scene) --- // + // Every rank coarsens + replicates the field, traces the SAME global + // polylines, and keeps only the segments inside its own domain; the + // ordered cross-domain composite stitches them. Built before the scene + // loop so an overlay and a standalone tube scene share one geometry pass. + // NB: all ranks reach this together (cadence is collective), so the + // Allreduce inside buildCoarseFieldVec is safe. + const auto& flc = g_renderer.fieldlines(); + out::TubeSet tubes = out::emptyTubeSet(); + out::TubeSet empty = out::emptyTubeSet(); + bool have_tubes = false; + if (flc.enable) { + const int gN[3] = { static_cast(mesh().n_active(in::x1)), + static_cast(mesh().n_active(in::x2)), + static_cast(mesh().n_active(in::x3)) }; + int gnc[3]; + real_t gorigin[3], gdx[3]; + for (int d = 0; d < 3; ++d) { + gnc[d] = std::max(1, (gN[d] + flc.bin - 1) / flc.bin); + gorigin[d] = glob_ext[d].first; + gdx[d] = (glob_ext[d].second - glob_ext[d].first) / gnc[d]; + } + const char fb = static_cast( + std::toupper(flc.field.empty() ? 'B' : flc.field[0])); + out::CoarseField cf = buildCoarseFieldVec( + local_domain->mesh, local_domain->fields, bckp, fb, gorigin, gnc, gdx); + // seed/tube scale: world units per screen pixel (orthographic frame) + const real_t wpp = (cam.half_h * TWO) / static_cast(H); + real_t vlo, vhi; + auto lines = out::traceFieldLines(cf, flc, wpp, vlo, vhi); + if (flc.vmax > flc.vmin) { // explicit color range overrides auto + vlo = flc.vmin; + vhi = flc.vmax; + } + const real_t tube_world = math::max(flc.tube_px, ONE) * wpp; + const real_t eff_r = math::max(tube_world, + static_cast(0.55) * ds); + std::size_t n_kept = 0; + tubes = out::buildTubeSet(lines, eff_r, flc, vlo, vhi, lo, hi, cf, + n_kept); + have_tubes = true; + logger::Checkpoint("field lines: " + std::to_string(lines.size()) + + " global lines, " + std::to_string(n_kept) + + " local segments", + HERE); + } + bool rendered_any = false; for (const auto& scene : g_renderer.scenes()) { - Kokkos::deep_copy(bckp, ZERO); - if (not prepareRenderScalar(params, *local_domain, scene.field, bckp)) { + // a `field = "fieldlines"` scene renders the tubes standalone (no + // volume); any other scene may overlay them inside its volume. + const bool fl_only = (scene.field == "fieldlines"); + const bool volume_on = not fl_only; + const bool show_tubes = scene.show_fieldlines and have_tubes; + if (volume_on) { + Kokkos::deep_copy(bckp, ZERO); + if (not prepareRenderScalar(params, *local_domain, scene.field, bckp)) { + continue; + } + // fill the ghost halo with neighbor active values so trilinear + // sampling is C0 across domain faces (a halo EXCHANGE, not the + // sum-into-active that SynchronizeFields performs). + CommunicateBckp(*local_domain, { 0, 1 }); + } else if (not have_tubes) { + // standalone field-line scene but tracing produced nothing/disabled + raise::Warning("output.render: 'fieldlines' scene but no field-line " + "geometry; skipping", + HERE); continue; } - // fill the ghost halo with neighbor active values so trilinear - // sampling is C0 across domain faces (a halo EXCHANGE, not the - // sum-into-active that SynchronizeFields performs). - CommunicateBckp(*local_domain, { 0, 1 }); + const out::TubeSet& kt = show_tubes ? tubes : empty; + // a standalone tube scene colors its colorbar by |field|, not by the + // (unused) volume transfer function + out::Scene scene_cb = scene; + if (fl_only) { + scene_cb.tf.vmin = tubes.vmin; + scene_cb.tf.vmax = tubes.vmax; + scene_cb.tf.log_scale = tubes.log_scale; + scene_cb.tf.colormap = tubes.colormap; + if (scene_cb.label == "fieldlines") { + scene_cb.label = "|" + flc.field + "|"; + } + } out::SubImage sub; if (on_screen) { @@ -555,6 +749,8 @@ namespace ntt { ghi, spine_radius, spine_rgb, + kt, + volume_on, image)); Kokkos::fence(); @@ -569,7 +765,7 @@ namespace ntt { sub.rgba[p * 4 + 3] = image_h(p, 3); } } - g_renderer.compositeAndWrite(sub, order_key, scene, current_step); + g_renderer.compositeAndWrite(sub, order_key, scene_cb, current_step); rendered_any = true; } return rendered_any; diff --git a/src/output/render/fieldlines.h b/src/output/render/fieldlines.h new file mode 100644 index 000000000..b1f24f4db --- /dev/null +++ b/src/output/render/fieldlines.h @@ -0,0 +1,457 @@ +/** + * @file output/render/fieldlines.h + * @brief Host-side magnetic-field-line tracer + bucketed tube-geometry builder + * @implements + * - out::CoarseField + * - out::traceFieldLines + * - out::buildTubeSet + * - out::emptyTubeSet + * @namespaces: + * - out:: + * @macros: + * - OUTPUT_ENABLED + * @note + * Field lines are intrinsically non-local (a streamline wanders across MPI + * domains), which would normally demand parallel particle advection. We sidestep + * that entirely: the (physical-basis) field is volume-averaged onto a COARSE + * grid and MPI-replicated to every rank (see Metadomain::buildFieldLineTubes), + * so every rank traces the SAME global polylines locally and renders only the + * segments overlapping its own domain. The existing ordered cross-domain + * composite then stitches the pieces. Tracing/geometry here is metric-agnostic: + * the only supported 3D render mode is Cartesian (Minkowski), so the coarse grid + * is a plain uniform lattice in world coordinates. + * + * Performance: a ray sample must not test every segment. Segments are bucketed + * into the coarse grid (CSR), and since the tube radius is << one coarse cell, + * a sample only needs the segments registered in its own cell. The kernel + * (raymarch.hpp) walks that short bucket. + */ + +#ifndef OUTPUT_RENDER_FIELDLINES_H +#define OUTPUT_RENDER_FIELDLINES_H + +#include "global.h" + +#include "arch/kokkos_aliases.h" + +#include "output/render/renderer.h" +#include "output/render/transfer_fn.h" + +#include + +#include +#include +#include +#include +#include + +namespace out { + + /** + * @brief A coarse, MPI-replicated copy of the physical-basis vector field. + * @note B is laid out component-fastest: index (c0,c1,c2,comp) lives at + * ((c2*n1 + c1)*n0 + c0)*3 + comp, with c0 the fastest spatial axis. + */ + struct CoarseField { + std::vector B; // n0*n1*n2*3 + int n[3] { 0, 0, 0 }; + real_t origin[3] { ZERO, ZERO, ZERO }; + real_t dx[3] { ONE, ONE, ONE }; + }; + + /** @brief One traced field line: world-space vertices + per-vertex |field|. */ + struct Polyline { + std::vector> pts; + std::vector scal; + }; + + namespace fl_hidden { + + inline auto cellLinear(const CoarseField& cf, int c0, int c1, int c2) + -> std::size_t { + return (static_cast(c2) * cf.n[1] + c1) * cf.n[0] + c0; + } + + // trilinear sample of the coarse field at world point p -> B[3], |B|. + // Clamps to the grid (so a sample just outside a face still returns the edge + // value); membership in the global box is the caller's stop test. + inline auto sampleCoarse(const CoarseField& cf, const real_t p[3], real_t B[3]) + -> real_t { + int i0[3], i1[3]; + real_t fr[3]; + for (int d = 0; d < 3; ++d) { + if (cf.n[d] <= 1) { + i0[d] = 0; + i1[d] = 0; + fr[d] = ZERO; + continue; + } + const real_t g = (p[d] - cf.origin[d]) / cf.dx[d] - HALF; + real_t f = std::floor(g); + real_t t = g - f; + int b = static_cast(f); + if (b < 0) { + b = 0; + t = ZERO; + } else if (b > cf.n[d] - 2) { + b = cf.n[d] - 2; + t = ONE; + } + i0[d] = b; + i1[d] = b + 1; + fr[d] = t; + } + for (int comp = 0; comp < 3; ++comp) { + const real_t c000 = cf.B[cellLinear(cf, i0[0], i0[1], i0[2]) * 3 + comp]; + const real_t c100 = cf.B[cellLinear(cf, i1[0], i0[1], i0[2]) * 3 + comp]; + const real_t c010 = cf.B[cellLinear(cf, i0[0], i1[1], i0[2]) * 3 + comp]; + const real_t c110 = cf.B[cellLinear(cf, i1[0], i1[1], i0[2]) * 3 + comp]; + const real_t c001 = cf.B[cellLinear(cf, i0[0], i0[1], i1[2]) * 3 + comp]; + const real_t c101 = cf.B[cellLinear(cf, i1[0], i0[1], i1[2]) * 3 + comp]; + const real_t c011 = cf.B[cellLinear(cf, i0[0], i1[1], i1[2]) * 3 + comp]; + const real_t c111 = cf.B[cellLinear(cf, i1[0], i1[1], i1[2]) * 3 + comp]; + const real_t c00 = c000 * (ONE - fr[0]) + c100 * fr[0]; + const real_t c10 = c010 * (ONE - fr[0]) + c110 * fr[0]; + const real_t c01 = c001 * (ONE - fr[0]) + c101 * fr[0]; + const real_t c11 = c011 * (ONE - fr[0]) + c111 * fr[0]; + const real_t c0 = c00 * (ONE - fr[1]) + c10 * fr[1]; + const real_t c1 = c01 * (ONE - fr[1]) + c11 * fr[1]; + B[comp] = c0 * (ONE - fr[2]) + c1 * fr[2]; + } + return std::sqrt(B[0] * B[0] + B[1] * B[1] + B[2] * B[2]); + } + + inline auto insideBox(const CoarseField& cf, const real_t p[3]) -> bool { + for (int d = 0; d < 3; ++d) { + const real_t hi = cf.origin[d] + cf.n[d] * cf.dx[d]; + if (p[d] < cf.origin[d] or p[d] > hi) { + return false; + } + } + return true; + } + + } // namespace fl_hidden + + /** + * @brief Trace field lines through the coarse field by bidirectional RK4. + * @param cf coarse, replicated physical field + * @param cfg field-line configuration (seed density, step, caps) + * @param world_per_pixel world units per screen pixel (sets seed/tube scale) + * @param[out] out_vmin,out_vmax auto color range (min/max |field| along lines) + * @return global polylines (identical on every rank) + */ + inline auto traceFieldLines(const CoarseField& cf, + const FieldLineConfig& cfg, + real_t world_per_pixel, + real_t& out_vmin, + real_t& out_vmax) + -> std::vector { + using fl_hidden::insideBox; + using fl_hidden::sampleCoarse; + + std::vector lines; + if (cf.n[0] < 1 or cf.n[1] < 1 or cf.n[2] < 1) { + return lines; + } + + real_t size[3]; + real_t diag2 = ZERO; + real_t min_dx = static_cast(1e30); + for (int d = 0; d < 3; ++d) { + size[d] = cf.n[d] * cf.dx[d]; + diag2 += size[d] * size[d]; + min_dx = std::min(min_dx, cf.dx[d]); + } + const real_t box_diag = std::sqrt(diag2); + const real_t max_len = cfg.max_len_frac * box_diag; + const real_t h = std::max(cfg.step_frac, static_cast(1e-3)) * + min_dx; + const real_t eps = static_cast(1e-20); + + // seed lattice: spacing ~ seed_px screen pixels, grown to respect seed_max + real_t spacing = std::max(cfg.seed_px, ONE) * world_per_pixel; + int ns[3]; + auto countSeeds = [&](real_t s) -> long { + long tot = 1; + for (int d = 0; d < 3; ++d) { + ns[d] = std::max(1, static_cast(std::floor(size[d] / s))); + tot *= ns[d]; + } + return tot; + }; + long n_seed = countSeeds(spacing); + if (n_seed > cfg.seed_max and cfg.seed_max > 0) { + const real_t grow = std::cbrt(static_cast(n_seed) / + static_cast(cfg.seed_max)); + spacing *= grow; + countSeeds(spacing); // recompute ns[] for the grown spacing + } + + // unit-vector field derivative (×dir) used by RK4; false if |B| ~ 0 + auto deriv = [&](const real_t p[3], real_t dir, real_t out[3]) -> bool { + real_t B[3]; + const real_t m = sampleCoarse(cf, p, B); + if (m < eps) { + return false; + } + const real_t inv = dir / m; + out[0] = B[0] * inv; + out[1] = B[1] * inv; + out[2] = B[2] * inv; + return true; + }; + + out_vmin = static_cast(1e30); + out_vmax = static_cast(-1e30); + auto track = [&](real_t m) { + out_vmin = std::min(out_vmin, m); + out_vmax = std::max(out_vmax, m); + }; + + // integrate one direction (dir = +1 forward, -1 backward) from a seed + auto integrate = [&](const real_t seed[3], real_t dir) { + Polyline pl; + real_t p[3] = { seed[0], seed[1], seed[2] }; + real_t B0[3]; + real_t m0 = sampleCoarse(cf, p, B0); + if (m0 < eps) { + return; + } + pl.pts.push_back({ p[0], p[1], p[2] }); + pl.scal.push_back(m0); + track(m0); + real_t len = ZERO; + for (int step = 0; step < cfg.max_steps and len < max_len; ++step) { + real_t k1[3], k2[3], k3[3], k4[3], q[3]; + if (not deriv(p, dir, k1)) { + break; + } + for (int d = 0; d < 3; ++d) { + q[d] = p[d] + HALF * h * k1[d]; + } + if (not deriv(q, dir, k2)) { + break; + } + for (int d = 0; d < 3; ++d) { + q[d] = p[d] + HALF * h * k2[d]; + } + if (not deriv(q, dir, k3)) { + break; + } + for (int d = 0; d < 3; ++d) { + q[d] = p[d] + h * k3[d]; + } + if (not deriv(q, dir, k4)) { + break; + } + for (int d = 0; d < 3; ++d) { + p[d] += (h / static_cast(6)) * + (k1[d] + static_cast(2) * k2[d] + + static_cast(2) * k3[d] + k4[d]); + } + if (not insideBox(cf, p)) { + break; + } + real_t B[3]; + const real_t m = sampleCoarse(cf, p, B); + pl.pts.push_back({ p[0], p[1], p[2] }); + pl.scal.push_back(m); + track(m); + len += h; + } + if (pl.pts.size() >= 2) { + lines.push_back(std::move(pl)); + } + }; + + for (int k = 0; k < ns[2]; ++k) { + for (int j = 0; j < ns[1]; ++j) { + for (int i = 0; i < ns[0]; ++i) { + const real_t seed[3] = { + cf.origin[0] + (static_cast(i) + HALF) * size[0] / ns[0], + cf.origin[1] + (static_cast(j) + HALF) * size[1] / ns[1], + cf.origin[2] + (static_cast(k) + HALF) * size[2] / ns[2] + }; + integrate(seed, ONE); + integrate(seed, -ONE); + } + } + } + if (out_vmin > out_vmax) { // no lines traced + out_vmin = ZERO; + out_vmax = ONE; + } + return lines; + } + + /** + * @brief Bucket the field-line segments overlapping a domain AABB into a CSR + * grid index and pack them into device Views ready for the ray-march kernel. + * @param lines global polylines (every rank passes the same set) + * @param radius world-space tube radius (the ds floor is applied by the caller) + * @param cfg field-line configuration (colormap / log) + * @param vmin,vmax tube color range + * @param lo,hi this domain's world AABB (segments outside it are dropped) + * @param cf coarse grid geometry, reused as the bucket grid + * @param[out] n_kept number of segments kept for this domain (for logging) + */ + inline auto buildTubeSet(const std::vector& lines, + real_t radius, + const FieldLineConfig& cfg, + real_t vmin, + real_t vmax, + const real_t lo[3], + const real_t hi[3], + const CoarseField& cf, + std::size_t& n_kept) -> TubeSet { + TubeSet ts; + ts.radius = radius; + ts.vmin = vmin; + ts.vmax = (vmax > vmin) ? vmax : (vmin + ONE); + ts.log_scale = cfg.log_scale and (vmin > ZERO); + ts.colormap = cfg.colormap; + ts.n_lut = 256; + // Bucket grid: a LOCAL uniform lattice spanning only THIS domain's AABB + // (not the whole box), so cell_start stays O(local cells). A bucket cell is + // ~one coarse cell; the tube radius is far smaller, so a ray sample (always + // inside [lo,hi]) finds its nearest segment in its own bucket cell. + for (int d = 0; d < 3; ++d) { + ts.gdx[d] = (cf.dx[d] > ZERO) ? cf.dx[d] : ONE; + ts.gorigin[d] = lo[d]; + const real_t span = hi[d] - lo[d]; + ts.gnc[d] = std::max(1, static_cast(std::ceil(span / ts.gdx[d]))); + } + auto lin = [&](int c0, int c1, int c2) -> std::size_t { + return (static_cast(c2) * ts.gnc[1] + c1) * ts.gnc[0] + c0; + }; + // opaque LUT: a tube sample paints a solid color (alpha==1) + ts.lut = buildLUT(cfg.colormap, ts.n_lut, { { ZERO, ONE }, { ONE, ONE } }); + + // 1) keep segments whose radius-padded AABB overlaps this domain AABB. + // (We keep the whole segment, not a clipped piece: the kernel only ever + // samples within this domain's slab, so a shared segment shows only in + // the correct domain's depth range -- no double-draw.) + std::vector> kept; + for (const auto& pl : lines) { + for (std::size_t i = 0; i + 1 < pl.pts.size(); ++i) { + const auto& a = pl.pts[i]; + const auto& b = pl.pts[i + 1]; + bool overlap = true; + for (int d = 0; d < 3; ++d) { + const real_t smin = std::min(a[d], b[d]) - radius; + const real_t smax = std::max(a[d], b[d]) + radius; + if (smax < lo[d] or smin > hi[d]) { + overlap = false; + break; + } + } + if (overlap) { + kept.push_back({ a[0], a[1], a[2], b[0], b[1], b[2], pl.scal[i], + pl.scal[i + 1] }); + } + } + } + n_kept = kept.size(); + ts.n_seg = static_cast(kept.size()); + const std::size_t ncell = static_cast(ts.gnc[0]) * ts.gnc[1] * + ts.gnc[2]; + + // 2) CSR bucketing on the coarse grid: count, prefix-sum, scatter. Each + // segment is registered in every cell its radius-padded AABB overlaps. + auto cellOf = [&](real_t x, int d) -> int { + int c = static_cast(std::floor((x - ts.gorigin[d]) / ts.gdx[d])); + if (c < 0) { + c = 0; + } else if (c > ts.gnc[d] - 1) { + c = ts.gnc[d] - 1; + } + return c; + }; + auto cellRange = [&](const std::array& s, int d, int& c0, int& c1) { + const real_t smin = std::min(s[d], s[3 + d]) - radius; + const real_t smax = std::max(s[d], s[3 + d]) + radius; + c0 = cellOf(smin, d); + c1 = cellOf(smax, d); + }; + + std::vector count(ncell + 1, 0); + for (const auto& s : kept) { + int a0, a1, b0, b1, d0, d1; + cellRange(s, 0, a0, a1); + cellRange(s, 1, b0, b1); + cellRange(s, 2, d0, d1); + for (int c2 = d0; c2 <= d1; ++c2) { + for (int c1 = b0; c1 <= b1; ++c1) { + for (int c0 = a0; c0 <= a1; ++c0) { + ++count[lin(c0, c1, c2)]; + } + } + } + } + std::vector start(ncell + 1, 0); + for (std::size_t c = 0; c < ncell; ++c) { + start[c + 1] = start[c] + count[c]; + } + const std::size_t n_insert = static_cast(start[ncell]); + std::vector idx(n_insert, 0); + std::vector cursor(start.begin(), start.end()); // running write head + for (std::size_t si = 0; si < kept.size(); ++si) { + int a0, a1, b0, b1, d0, d1; + cellRange(kept[si], 0, a0, a1); + cellRange(kept[si], 1, b0, b1); + cellRange(kept[si], 2, d0, d1); + for (int c2 = d0; c2 <= d1; ++c2) { + for (int c1 = b0; c1 <= b1; ++c1) { + for (int c0 = a0; c0 <= a1; ++c0) { + const std::size_t cl = lin(c0, c1, c2); + idx[static_cast(cursor[cl]++)] = static_cast(si); + } + } + } + } + + // 3) upload to device + ts.seg = array_t("fl_seg", static_cast(ts.n_seg)); + if (ts.n_seg > 0) { + auto seg_h = Kokkos::create_mirror_view(ts.seg); + for (int s = 0; s < ts.n_seg; ++s) { + for (int c = 0; c < 8; ++c) { + seg_h(s, c) = kept[static_cast(s)][static_cast(c)]; + } + } + Kokkos::deep_copy(ts.seg, seg_h); + } + ts.cell_start = array_t("fl_cell_start", ncell + 1); + { + auto h = Kokkos::create_mirror_view(ts.cell_start); + for (std::size_t c = 0; c <= ncell; ++c) { + h(c) = start[c]; + } + Kokkos::deep_copy(ts.cell_start, h); + } + ts.seg_idx = array_t("fl_seg_idx", std::max(n_insert, 1)); + if (n_insert > 0) { + auto h = Kokkos::create_mirror_view(ts.seg_idx); + for (std::size_t k = 0; k < n_insert; ++k) { + h(k) = idx[k]; + } + Kokkos::deep_copy(ts.seg_idx, h); + } + return ts; + } + + /** @brief A valid but empty tube set (used when a scene shows no field lines). */ + inline auto emptyTubeSet() -> TubeSet { + TubeSet ts; + ts.n_seg = 0; + ts.seg = array_t("fl_seg_empty", 0); + ts.cell_start = array_t("fl_cell_start_empty", 1); + ts.seg_idx = array_t("fl_seg_idx_empty", 1); + ts.lut = buildLUT("inferno", 2, { { ZERO, ONE }, { ONE, ONE } }); + return ts; + } + +} // namespace out + +#endif // OUTPUT_RENDER_FIELDLINES_H diff --git a/src/output/render/raymarch.hpp b/src/output/render/raymarch.hpp index e0db322f3..a326d3f31 100644 --- a/src/output/render/raymarch.hpp +++ b/src/output/render/raymarch.hpp @@ -67,6 +67,24 @@ namespace kernel { const real_t spine_radius; const real_t spine_cr, spine_cg, spine_cb; + // magnetic-field-line tubes: opaque capsules colored by |field|, composited + // inline like the spine. Bucketed into the coarse grid (CSR) so a sample + // tests only the segments in its cell. tn_seg <= 0 disables them. + array_t tseg; // (tn_seg, 8): p0, p1, s0, s1 (world) + array_t tcell_start; // (ncell+1) CSR offsets + array_t tseg_idx; // segment indices grouped by cell + const int tn_seg; + const real_t tube_r2; // squared tube radius + const int tgnc0, tgnc1, tgnc2; + const real_t tg0, tg1, tg2; // bucket-grid origin + const real_t tdx0, tdx1, tdx2; // bucket-grid cell size + array_t tube_lut; + const int tube_n_lut; + const real_t tube_vmin, tube_vmax; + const bool tube_log; + // when false the scalar volume is not sampled (standalone field-line render) + const bool volume_enabled; + array_t image; // output, (bw*bh, 4) premultiplied RGBA public: @@ -96,6 +114,8 @@ namespace kernel { const real_t ghi[3], real_t spine_radius_, const real_t spine_rgb[3], + const out::TubeSet& tubes_, + bool volume_enabled_, const array_t& image_) : Fld { Fld_ } , comp { comp_ } @@ -133,6 +153,26 @@ namespace kernel { , spine_cr { spine_rgb[0] } , spine_cg { spine_rgb[1] } , spine_cb { spine_rgb[2] } + , tseg { tubes_.seg } + , tcell_start { tubes_.cell_start } + , tseg_idx { tubes_.seg_idx } + , tn_seg { tubes_.n_seg } + , tube_r2 { tubes_.radius * tubes_.radius } + , tgnc0 { tubes_.gnc[0] } + , tgnc1 { tubes_.gnc[1] } + , tgnc2 { tubes_.gnc[2] } + , tg0 { tubes_.gorigin[0] } + , tg1 { tubes_.gorigin[1] } + , tg2 { tubes_.gorigin[2] } + , tdx0 { tubes_.gdx[0] } + , tdx1 { tubes_.gdx[1] } + , tdx2 { tubes_.gdx[2] } + , tube_lut { tubes_.lut } + , tube_n_lut { tubes_.n_lut } + , tube_vmin { tubes_.vmin } + , tube_vmax { tubes_.vmax } + , tube_log { tubes_.log_scale } + , volume_enabled { volume_enabled_ } , image { image_ } {} // distance test: is world point (px,py,pz) within `spine_radius` of any of @@ -171,6 +211,48 @@ namespace kernel { return false; } + // is world point within `tube_radius` of any field-line segment? Walks only + // the bucket of the point's coarse cell (radius << one coarse cell, so a + // padded-AABB insertion guarantees the nearest segment is in this cell). + // On a hit, `scalar` is the |field| interpolated to the closest point. + Inline auto inTube(real_t px, real_t py, real_t pz, real_t& scalar) const + -> bool { + if (tn_seg <= 0) { + return false; + } + const int c0 = static_cast(math::floor((px - tg0) / tdx0)); + const int c1 = static_cast(math::floor((py - tg1) / tdx1)); + const int c2 = static_cast(math::floor((pz - tg2) / tdx2)); + if (c0 < 0 or c0 >= tgnc0 or c1 < 0 or c1 >= tgnc1 or c2 < 0 or + c2 >= tgnc2) { + return false; + } + const int lin = (c2 * tgnc1 + c1) * tgnc0 + c0; + const int kb = tcell_start(lin); + const int ke = tcell_start(lin + 1); + real_t best = tube_r2; + bool hit = false; + for (int k = kb; k < ke; ++k) { + const int s = tseg_idx(k); + const real_t ax = tseg(s, 0), ay = tseg(s, 1), az = tseg(s, 2); + const real_t bx = tseg(s, 3), by = tseg(s, 4), bz = tseg(s, 5); + const real_t ex = bx - ax, ey = by - ay, ez = bz - az; + const real_t wx = px - ax, wy = py - ay, wz = pz - az; + const real_t ee = ex * ex + ey * ey + ez * ez; + real_t tt = (ee > ZERO) ? (wx * ex + wy * ey + wz * ez) / ee : ZERO; + tt = (tt < ZERO) ? ZERO : ((tt > ONE) ? ONE : tt); + const real_t cx = ax + tt * ex, cy = ay + tt * ey, cz = az + tt * ez; + const real_t dx = px - cx, dy = py - cy, dz = pz - cz; + const real_t d2 = dx * dx + dy * dy + dz * dz; + if (d2 < best) { + best = d2; + scalar = tseg(s, 6) * (ONE - tt) + tseg(s, 7) * tt; + hit = true; + } + } + return hit; + } + // trilinear sample of the prepared scalar at world point p, reading the // ghost halo for corners just outside the active box. Inline auto sample(real_t px, real_t py, real_t pz) const -> real_t { @@ -311,6 +393,12 @@ namespace kernel { // ---- march at global sample positions t_k = k*ds ------------------- // const real_t inv_range = (vmax > vmin) ? (ONE / (vmax - vmin)) : ZERO; const real_t log_vmin = log_scale ? math::log10(vmin) : ZERO; + const real_t tube_inv_range = (tube_vmax > tube_vmin) + ? (ONE / (tube_vmax - tube_vmin)) + : ZERO; + const real_t tube_log_vmin = (tube_log and tube_vmin > ZERO) + ? math::log10(tube_vmin) + : ZERO; // first global sample index inside this segment: t_k >= t_enter const real_t k0 = math::ceil(t_enter / ds); real_t t = k0 * ds; @@ -318,7 +406,8 @@ namespace kernel { int steps = 0; while (t < t_exit and steps < max_steps) { const real_t px = ox + t * dx, py = oy + t * dy, pz = oz + t * dz; - real_t cr, cg, cb, ca; + real_t cr = ZERO, cg = ZERO, cb = ZERO, ca = ZERO; + real_t tube_s = ZERO; if (onSpine(px, py, pz)) { // opaque box edge (premultiplied; alpha == 1 -> color is straight RGB). // Composited inline so accumulated foreground volume occludes it. @@ -326,7 +415,33 @@ namespace kernel { cg = spine_cg; cb = spine_cb; ca = ONE; - } else { + } else if (inTube(px, py, pz, tube_s)) { + // opaque field-line tube colored by |field| through the tube LUT + real_t u; + if (tube_log) { + u = (tube_s > ZERO) + ? (math::log10(tube_s) - tube_log_vmin) * tube_inv_range + : -ONE; + } else { + u = (tube_s - tube_vmin) * tube_inv_range; + } + if (u < ZERO) { + u = ZERO; + } else if (u > ONE) { + u = ONE; + } + int idx = static_cast(u * static_cast(tube_n_lut - 1) + + HALF); + if (idx < 0) { + idx = 0; + } else if (idx > tube_n_lut - 1) { + idx = tube_n_lut - 1; + } + cr = tube_lut(idx, 0); // opaque LUT -> straight RGB, alpha 1 + cg = tube_lut(idx, 1); + cb = tube_lut(idx, 2); + ca = tube_lut(idx, 3); + } else if (volume_enabled) { const real_t s = sample(px, py, pz); // normalize through the transfer function range real_t u; @@ -351,13 +466,17 @@ namespace kernel { cb = lut(idx, 2); ca = lut(idx, 3); } - const real_t w = ONE - acc_a; - acc_r += w * cr; - acc_g += w * cg; - acc_b += w * cb; - acc_a += w * ca; - if (acc_a >= early_alpha) { - break; + // composite only a non-empty sample (volume-off rays contribute solely + // where they hit a tube or the spine) + if (ca > ZERO) { + const real_t w = ONE - acc_a; + acc_r += w * cr; + acc_g += w * cg; + acc_b += w * cb; + acc_a += w * ca; + if (acc_a >= early_alpha) { + break; + } } t += ds; ++steps; diff --git a/src/output/render/renderer.cpp b/src/output/render/renderer.cpp index ebda438f5..30ee7c578 100644 --- a/src/output/render/renderer.cpp +++ b/src/output/render/renderer.cpp @@ -254,6 +254,10 @@ namespace out { scene.ticks = toml::find_or>(sc, "colorbar_ticks", std::vector {}); + // overlay the B-field-line tubes inside this scene's volume; a dedicated + // `field = "fieldlines"` scene renders the tubes standalone (no volume). + scene.show_fieldlines = toml::find_or(sc, "fieldlines", false) or + (scene.field == "fieldlines"); scene.tf.vmin = toml::find_or(sc, "min", ZERO); scene.tf.vmax = toml::find_or(sc, "max", ONE); scene.tf.log_scale = toml::find_or(sc, "log", false); @@ -284,6 +288,53 @@ namespace out { return; } + /* ---- magnetic-field-line tube overlay ------------------------------- */ + // The tubes are built whenever the [output.render.fieldlines] section asks + // for them OR any scene requests the overlay (so a bare `field = + // "fieldlines"` scene works without a separate enable flag). + bool any_fl = false; + for (const auto& s : m_scenes) { + any_fl = any_fl or s.show_fieldlines; + } + m_fieldlines.enable = toml::find_or(td, "output", "render", + "fieldlines", "enable", false) or + any_fl; + if (m_fieldlines.enable) { + if (m_global_extent.size() != 3) { + raise::Warning("output.render.fieldlines is 3D-only; ignoring", HERE); + m_fieldlines.enable = false; + } else { + auto& fl = m_fieldlines; + fl.field = toml::find_or(td, "output", "render", + "fieldlines", "field", "B"); + fl.bin = toml::find_or(td, "output", "render", "fieldlines", + "bin", 4); + fl.bin = (fl.bin < 1) ? 1 : ((fl.bin > 16) ? 16 : fl.bin); + fl.seed_px = toml::find_or(td, "output", "render", "fieldlines", + "seed_px", static_cast(8)); + fl.tube_px = toml::find_or(td, "output", "render", "fieldlines", + "tube_px", static_cast(2)); + fl.colormap = toml::find_or(td, "output", "render", + "fieldlines", "colormap", + "inferno"); + fl.log_scale = toml::find_or(td, "output", "render", "fieldlines", + "log", false); + fl.vmin = toml::find_or(td, "output", "render", "fieldlines", + "min", ZERO); + fl.vmax = toml::find_or(td, "output", "render", "fieldlines", + "max", ZERO); + fl.step_frac = toml::find_or(td, "output", "render", "fieldlines", + "step_frac", static_cast(0.5)); + fl.max_steps = toml::find_or(td, "output", "render", "fieldlines", + "max_steps", 4000); + fl.max_len_frac = toml::find_or(td, "output", "render", + "fieldlines", "max_length", + static_cast(3)); + fl.seed_max = toml::find_or(td, "output", "render", "fieldlines", + "seed_max", 4096); + } + } + m_enabled = true; logger::Checkpoint("In-situ renderer initialized", HERE); } diff --git a/src/output/render/renderer.h b/src/output/render/renderer.h index 15210f61e..72f1c170d 100644 --- a/src/output/render/renderer.h +++ b/src/output/render/renderer.h @@ -69,14 +69,70 @@ namespace out { std::string colormap { "viridis" }; // for redrawing the colorbar }; + /** + * @brief Configuration for the magnetic-field-line tube overlay. + * @note The lines are traced once per frame through a coarse, MPI-replicated + * copy of the (physical-basis) field, so every rank produces the same global + * polylines and renders only the segments inside its own domain; the existing + * ordered cross-domain composite then stitches them. The coarsening (`bin`) + * is what makes the replicate-and-trace cheap and avoids parallel particle + * advection. See output/render/fieldlines.h. + */ + struct FieldLineConfig { + bool enable { false }; // build the geometry this run + std::string field { "B" }; // vector field to trace: "B" | "E" | "J" + int bin { 4 }; // coarsening factor (cells/coarse cell), 2..8 + real_t seed_px { 8 }; // seed lattice spacing in screen pixels + real_t tube_px { 2 }; // tube radius in screen pixels + std::string colormap { "inferno" }; + bool log_scale { false }; + real_t vmin { ZERO }; // tube color range; vmin>=vmax => auto |B| + real_t vmax { ZERO }; + real_t step_frac { static_cast(0.5) }; // RK4 step / coarse cell + int max_steps { 4000 }; // per-direction integration cap + real_t max_len_frac { static_cast(3) }; // x global box diagonal + int seed_max { 4096 }; // hard cap on seed count (spacing grows to fit) + }; + + /** + * @brief Device-side field-line geometry handed to the ray-march kernel. + * @note A flat capsule list: each row is (p0xyz, p1xyz, s0, s1) with s the + * per-vertex scalar (|field|) used to color the tube. Opaque (alpha==1) LUT, + * so a tube sample paints a solid color and is composited inline exactly like + * the box spine. Empty (n_seg==0) on ranks no line touches. + */ + struct TubeSet { + array_t seg; // (n_seg, 8): p0, p1, s0, s1 in world coords + int n_seg { 0 }; + real_t radius { ZERO }; // world-space tube radius (ds floor applied) + array_t lut; // premultiplied RGBA, opaque (alpha==1) + int n_lut { 256 }; + real_t vmin { ZERO }, vmax { ONE }; + bool log_scale { false }; + std::string colormap { "inferno" }; // for the standalone colorbar + // uniform-grid bucket index (CSR) so a ray sample tests only the few + // segments in its cell instead of all of them. Bucketing on the coarse + // grid is exact because the tube radius is << one coarse cell; a segment is + // registered in every cell its radius-padded AABB overlaps. + array_t cell_start; // (ncell+1) prefix offsets into seg_idx + array_t seg_idx; // segment indices, grouped by cell + int gnc[3] { 1, 1, 1 }; + real_t gorigin[3] { ZERO, ZERO, ZERO }; + real_t gdx[3] { ONE, ONE, ONE }; + }; + /** * @brief One rendered scalar field -> one PNG stream. + * @note `field == "fieldlines"` is a standalone tube scene: no scalar volume + * is sampled (the field lines render against the background alone). Any other + * field with `show_fieldlines` true overlays the tubes inside its volume. */ struct Scene { - std::string field; // "N" | "Bmag" | "Vmag" | "Txy" | "B1" ... + std::string field; // "N" | "Bmag" | "Vmag" | "Txy" | "B1" | "fieldlines" ... std::string prefix; // PNG filename prefix, e.g. "Bmag_" std::string label; // colorbar title (defaults to field) std::vector ticks; // explicit colorbar tick values (optional) + bool show_fieldlines { false }; // overlay B-field tubes in the volume TransferFunction tf; }; @@ -238,6 +294,11 @@ namespace out { return m_scenes; } + [[nodiscard]] + auto fieldlines() const -> const FieldLineConfig& { + return m_fieldlines; + } + private: bool m_enabled { false }; @@ -279,6 +340,7 @@ namespace out { CameraDevice m_camera_dev; std::vector m_scenes; + FieldLineConfig m_fieldlines; tools::Tracker m_tracker; path_t m_root; From 7f8e819c497092d6e6df91661049181734b50ce9 Mon Sep 17 00:00:00 2001 From: Ludwig Boess Date: Tue, 30 Jun 2026 20:25:32 -0400 Subject: [PATCH 047/125] 2D field lines --- input.example.toml | 47 ++- src/framework/domain/metadomain_render.cpp | 212 ++++++++++- src/output/render/fieldlines.h | 403 ++++++++++++++++++++- src/output/render/renderer.cpp | 18 +- src/output/render/renderer.h | 32 ++ src/output/render/slice2d.hpp | 288 +++++++++++++-- 6 files changed, 953 insertions(+), 47 deletions(-) diff --git a/input.example.toml b/input.example.toml index 78ff1fc45..a43ad92cd 100644 --- a/input.example.toml +++ b/input.example.toml @@ -897,13 +897,17 @@ # A scene with field = "fieldlines" instead renders them alone. fieldlines = "" - # Magnetic field-line tubes (3D volume mode only). Traces field lines through - # a coarse, MPI-replicated copy of the field and draws them as solid tubes, - # colored by |field|, composited inside the same ray-march as the volume so - # they are correctly occluded by it. Built once per frame and shared by every - # scene that opts in (per-scene `fieldlines = true`) and by any standalone - # `field = "fieldlines"` scene. The coarsening is what makes the lines cheap - # to trace (no parallel particle advection) and gives a smoothed morphology. + # Magnetic field lines, drawn from a coarse, MPI-replicated copy of the field + # so the geometry is global and seamless across domains (the coarsening is + # what makes this cheap -- no parallel particle advection / flux scan). + # - 3D (Cartesian): traced as solid tubes, colored by |field|, composited + # inside the volume ray-march so the volume correctly occludes them. + # - 2D (Cartesian): iso-contours of the flux function psi (Bx = d psi/dy, + # By = -d psi/dx), i.e. the in-plane field lines, colored by |B|. + # - 2D (spherical / Kerr-Schild): traced meridional streamlines of the + # poloidal (Br, Btheta) field (nt2py style). + # Built once per frame and shared by every scene that opts in (per-scene + # `fieldlines = true`) and by any standalone `field = "fieldlines"` scene. [output.render.fieldlines] # Build the field-line geometry this run # @type: bool @@ -919,29 +923,40 @@ # Field coarsening factor (simulation cells per coarse cell, per axis) # @type: int [1..16] # @default: 4 - # @note: larger = smoother "morphology" lines + cheaper replication - # (the coarse field is ~ N_cells / bin^3 x 3 floats, on every rank) + # @note: larger = smoother "morphology" lines + cheaper replication (the + # coarse field is ~ N_cells / bin^D floats/rank; D = sim dimension) bin = "" - # Seed-lattice spacing in screen pixels (sets line density) + # (3D tubes) Seed-lattice spacing in screen pixels (sets line density) # @type: float [> 0] # @default: 8 # @note: capped by `seed_max`; if seed_px asks for more seeds than that, # the spacing grows to fit and seed_px no longer governs seed_px = "" - # Hard cap on the seed count (the seed lattice is n^3, 2 lines per seed) + # (3D tubes) Hard cap on the seed count (lattice is n^3, 2 lines per seed) # @type: int [> 0] # @default: 4096 # @note: lower this for fewer / more widely spaced lines seed_max = "" - # Tube radius in screen pixels + # (2D contours) Number of evenly-spaced flux-function contour levels + # @type: int [> 0] + # @default: 16 + # @note: evenly-spaced psi levels => line density tracks |B| automatically + levels = "" + # Tube radius (3D) / contour line width (2D), in screen pixels # @type: float [> 0] # @default: 2 tube_px = "" - # Tube colormap (mapped by |field| along the line) + # Colormap for the field lines (mapped by |B| along each line) # @type: string # @enum: "viridis", "inferno", "plasma", "cool2warm", "gray" # @default: "inferno" colormap = "" + # Monochrome override: draw the lines in a single [r,g,b] color (each 0..1) + # instead of the |B| colormap -- reads well as an overlay on another volume + # @type: array [size 3] + # @default: [] (empty => color by |B|) + # @example: [1.0, 1.0, 1.0] # white field lines + color = "" # Map the tube color range logarithmically # @type: bool # @default: false @@ -953,15 +968,15 @@ # @note: when min >= max, the range is auto-set from |field| along the lines min = "" max = "" - # RK4 integration step as a fraction of one coarse cell + # (3D tubes) RK4 integration step as a fraction of one coarse cell # @type: float [> 0] # @default: 0.5 step_frac = "" - # Per-direction integration-step cap + # (3D tubes) Per-direction integration-step cap # @type: int [> 0] # @default: 4000 max_steps = "" - # Maximum line length, in global box diagonals (per direction) + # (3D tubes) Maximum line length, in global box diagonals (per direction) # @type: float [> 0] # @default: 3.0 max_length = "" diff --git a/src/framework/domain/metadomain_render.cpp b/src/framework/domain/metadomain_render.cpp index da3127f53..b6ba21c04 100644 --- a/src/framework/domain/metadomain_render.cpp +++ b/src/framework/domain/metadomain_render.cpp @@ -245,6 +245,103 @@ namespace ntt { return cf; } + // 2D analogue of buildCoarseFieldVec: volume-average the in-plane physical + // components (Bx, By) onto a coarse global 2D grid and MPI-replicate them, + // so every rank can integrate the SAME flux function for seamless contours. + template + auto buildCoarseField2D(const Mesh& mesh, + const Fields& fields, + ndfield_t& bckp, + char fbase, + const real_t gorigin[2], + const int gnc[2], + const real_t gdx[2]) -> out::CoarseField2D { + const auto metric = mesh.metric; + uint8_t src_base = em::bx1; + PrepareOutputFlags interp = PrepareOutput::InterpToCellCenterFromFaces; + bool is_current = false; + if (fbase == 'E') { + src_base = em::ex1; + interp = PrepareOutput::InterpToCellCenterFromEdges; + } else if (fbase == 'J') { + is_current = true; + src_base = cur::jx1; + interp = PrepareOutput::InterpToCellCenterFromEdges; + } + if (is_current) { + copyVec3ToBckp(fields.cur, bckp, + cell_range_t(cur::jx1, cur::jx3 + 1)); + } else { + copyVec3ToBckp(fields.em, bckp, + cell_range_t(src_base, src_base + 3)); + } + const PrepareOutputFlags prepare = (S == SimEngine::SRPIC) + ? PrepareOutput::ConvertToHat + : PrepareOutput::ConvertToPhysCntrv; + list_t comp_from = { 0, 1, 2 }; + list_t comp_to = { 3, 4, 5 }; + Kokkos::parallel_for( + "RenderFL2DFieldsToPhys", + mesh.rangeActiveCells(), + kernel::FieldsToPhys_kernel(bckp, bckp, comp_from, comp_to, + interp | prepare, metric)); + Kokkos::fence(); + + auto bckp_h = Kokkos::create_mirror_view(bckp); + Kokkos::deep_copy(bckp_h, bckp); + + const std::size_t ncell = static_cast(gnc[0]) * gnc[1]; + std::vector sum(ncell * 2, ZERO); + std::vector cnt(ncell, ZERO); + const auto le = mesh.extent(); + const real_t llo[2] = { le[0].first, le[1].first }; + const real_t lsz[2] = { le[0].second - le[0].first, + le[1].second - le[1].first }; + const int nl[2] = { static_cast(mesh.n_active(in::x1)), + static_cast(mesh.n_active(in::x2)) }; + const int NG = static_cast(N_GHOSTS); + for (int j = 0; j < nl[1]; ++j) { + for (int i = 0; i < nl[0]; ++i) { + const real_t world[2] = { + llo[0] + (static_cast(i) + HALF) * lsz[0] / nl[0], + llo[1] + (static_cast(j) + HALF) * lsz[1] / nl[1] + }; + int c[2]; + for (int d = 0; d < 2; ++d) { + int cc = static_cast( + std::floor((world[d] - gorigin[d]) / gdx[d])); + cc = (cc < 0) ? 0 : ((cc > gnc[d] - 1) ? gnc[d] - 1 : cc); + c[d] = cc; + } + const std::size_t lin = static_cast(c[1]) * gnc[0] + c[0]; + sum[lin * 2 + 0] += bckp_h(i + NG, j + NG, 3); // Bx + sum[lin * 2 + 1] += bckp_h(i + NG, j + NG, 4); // By + cnt[lin] += ONE; + } + } +#if defined(MPI_ENABLED) + MPI_Allreduce(MPI_IN_PLACE, sum.data(), static_cast(ncell * 2), + mpi::get_type(), MPI_SUM, MPI_COMM_WORLD); + MPI_Allreduce(MPI_IN_PLACE, cnt.data(), static_cast(ncell), + mpi::get_type(), MPI_SUM, MPI_COMM_WORLD); +#endif + out::CoarseField2D cf; + cf.B.assign(ncell * 2, ZERO); + for (int d = 0; d < 2; ++d) { + cf.n[d] = gnc[d]; + cf.origin[d] = gorigin[d]; + cf.dx[d] = gdx[d]; + } + for (std::size_t c = 0; c < ncell; ++c) { + if (cnt[c] > ZERO) { + const real_t inv = ONE / cnt[c]; + cf.B[c * 2 + 0] = sum[c * 2 + 0] * inv; + cf.B[c * 2 + 1] = sum[c * 2 + 1] * inv; + } + } + return cf; + } + } // namespace template @@ -929,13 +1026,117 @@ namespace ntt { ndomains_per_dim(), fwd2d); + // ---- 2D field lines (built once) ------------------------------------ // + // Cartesian: iso-contours of the flux function psi. Spherical/Kerr: traced + // meridional streamlines (nt2py style). Both come from a coarse, MPI- + // replicated copy of the in-plane field, so the geometry is global and + // seamless across the disjoint tiles. All ranks reach buildCoarseField2D + // together (collective Allreduce); M is fixed per run, so every rank takes + // the same Cartesian/spherical branch. + const auto& flc = g_renderer.fieldlines(); + out::ContourSet contours = out::emptyContourSet(); + out::ContourSet emptyc = out::emptyContourSet(); + out::TubeSet lines2d = out::emptyTubeSet(); + out::TubeSet emptyl = out::emptyTubeSet(); + bool have_fl = false; + real_t fl_vmin = ZERO, fl_vmax = ONE; + std::string fl_colormap = flc.colormap; + if (flc.enable) { + const int gN[2] = { static_cast(mesh().n_active(in::x1)), + static_cast(mesh().n_active(in::x2)) }; + int gnc[2]; + real_t gorigin[2], gdx[2]; + for (int d = 0; d < 2; ++d) { + gnc[d] = std::max(1, (gN[d] + flc.bin - 1) / flc.bin); + gorigin[d] = gext[d].first; + gdx[d] = (gext[d].second - gext[d].first) / gnc[d]; + } + const char fb = static_cast( + std::toupper(flc.field.empty() ? 'B' : flc.field[0])); + // coarse, replicated in-plane field: (Bx,By) for Cartesian, (Br,Bth) for + // spherical (FieldsToPhys writes the physical components in axis order) + out::CoarseField2D cf = buildCoarseField2D( + local_domain->mesh, local_domain->fields, bckp, fb, gorigin, gnc, gdx); + const real_t wpp = (umax - umin) / static_cast(W); + if constexpr (M::CoordType == Coord::type::Cartesian) { + std::vector psi; + real_t pmin, pmax, bmin, bmax; + out::computeFlux2D(cf, psi, pmin, pmax, bmin, bmax); + const real_t line_half = HALF * math::max(flc.tube_px, ONE); + contours = out::buildContourSet(cf, psi, pmin, pmax, bmin, bmax, flc, + line_half, wpp); + fl_vmin = contours.vmin; + fl_vmax = contours.vmax; + fl_colormap = contours.colormap; + logger::Checkpoint("field lines (2D): " + std::to_string(flc.levels) + + " psi contours on a " + std::to_string(gnc[0]) + + "x" + std::to_string(gnc[1]) + " grid", + HERE); + } else { + // traced meridional streamlines through the coarse (r, theta) field + real_t vlo, vhi; + auto poly = out::traceFieldLinesMeridional(cf, flc, wpp, mirror, vlo, + vhi); + if (flc.vmax > flc.vmin) { // explicit |B| range overrides auto + vlo = flc.vmin; + vhi = flc.vmax; + } + const real_t eff_r = math::max(flc.tube_px, ONE) * wpp; + // bucket grid for buildTubeSet: cell ~ a coarse dr (a length), AABB + // spans the (mirrored) meridional disk, z is a single thin slab at 0 + out::CoarseField bucket_cf; + bucket_cf.dx[0] = cf.dx[0]; + bucket_cf.dx[1] = cf.dx[0]; + bucket_cf.dx[2] = cf.dx[0]; + const real_t rmax = gext[0].second; + const real_t lo[3] = { mirror ? -rmax : ZERO, -rmax, -cf.dx[0] }; + const real_t hi[3] = { rmax, rmax, cf.dx[0] }; + std::size_t n_kept = 0; + lines2d = out::buildTubeSet(poly, eff_r, flc, vlo, vhi, lo, hi, + bucket_cf, n_kept); + fl_vmin = lines2d.vmin; + fl_vmax = lines2d.vmax; + fl_colormap = lines2d.colormap; + logger::Checkpoint("field lines (2D meridional): " + + std::to_string(poly.size()) + " lines, " + + std::to_string(n_kept) + " segments", + HERE); + } + have_fl = true; + } + bool rendered_any = false; for (const auto& scene : g_renderer.scenes()) { - Kokkos::deep_copy(bckp, ZERO); - if (not prepareRenderScalar(params, *local_domain, scene.field, bckp)) { + // standalone `field = "fieldlines"` -> lines only (no heatmap fill); any + // other scene with `fieldlines = true` overlays them on its heatmap. + const bool fl_only = (scene.field == "fieldlines"); + const bool heatmap_on = not fl_only; + const bool show_lines = scene.show_fieldlines and have_fl; + if (heatmap_on) { + Kokkos::deep_copy(bckp, ZERO); + if (not prepareRenderScalar(params, *local_domain, scene.field, bckp)) { + continue; + } + CommunicateBckp(*local_domain, { 0, 1 }); + } else if (not have_fl) { + raise::Warning("output.render: 'fieldlines' scene needs a 2D run with " + "[output.render.fieldlines]; skipping", + HERE); continue; } - CommunicateBckp(*local_domain, { 0, 1 }); + const out::ContourSet& kc = show_lines ? contours : emptyc; + const out::TubeSet& kt = show_lines ? lines2d : emptyl; + // a standalone field-line scene colors its colorbar by |B| + out::Scene scene_cb = scene; + if (fl_only) { + scene_cb.tf.vmin = fl_vmin; + scene_cb.tf.vmax = fl_vmax; + scene_cb.tf.log_scale = false; + scene_cb.tf.colormap = fl_colormap; + if (scene_cb.label == "fieldlines") { + scene_cb.label = "|" + flc.field + "|"; + } + } out::SubImage sub; if (bw > 0 and bh > 0) { @@ -974,6 +1175,9 @@ namespace ntt { scene.tf.vmin, scene.tf.vmax, scene.tf.log_scale, + kc, + kt, + heatmap_on, image)); Kokkos::fence(); @@ -987,7 +1191,7 @@ namespace ntt { sub.rgba[p * 4 + 3] = image_h(p, 3); } } - g_renderer.compositeAndWrite(sub, order_key, scene, current_step); + g_renderer.compositeAndWrite(sub, order_key, scene_cb, current_step); rendered_any = true; } return rendered_any; diff --git a/src/output/render/fieldlines.h b/src/output/render/fieldlines.h index b1f24f4db..a6adbdd7c 100644 --- a/src/output/render/fieldlines.h +++ b/src/output/render/fieldlines.h @@ -133,6 +133,30 @@ namespace out { } // namespace fl_hidden + /** @brief A constant-color opaque LUT (monochrome field lines). */ + inline auto buildSolidLUT(real_t r, real_t g, real_t b, int n_lut) + -> array_t { + array_t lut { "fl_solid_lut", static_cast(n_lut) }; + auto h = Kokkos::create_mirror_view(lut); + for (int i = 0; i < n_lut; ++i) { + h(i, 0) = r; // opaque -> premultiplied == straight RGB + h(i, 1) = g; + h(i, 2) = b; + h(i, 3) = ONE; + } + Kokkos::deep_copy(lut, h); + return lut; + } + + /** @brief The field-line LUT: a single color if cfg.color is set, else by |B|. */ + inline auto buildLineLUT(const FieldLineConfig& cfg, int n_lut) + -> array_t { + if (cfg.color.size() == 3) { + return buildSolidLUT(cfg.color[0], cfg.color[1], cfg.color[2], n_lut); + } + return buildLUT(cfg.colormap, n_lut, { { ZERO, ONE }, { ONE, ONE } }); + } + /** * @brief Trace field lines through the coarse field by bidirectional RK4. * @param cf coarse, replicated physical field @@ -325,8 +349,9 @@ namespace out { auto lin = [&](int c0, int c1, int c2) -> std::size_t { return (static_cast(c2) * ts.gnc[1] + c1) * ts.gnc[0] + c0; }; - // opaque LUT: a tube sample paints a solid color (alpha==1) - ts.lut = buildLUT(cfg.colormap, ts.n_lut, { { ZERO, ONE }, { ONE, ONE } }); + // opaque LUT: a tube sample paints a solid color (alpha==1), by |B| or a + // single monochrome color when cfg.color is set + ts.lut = buildLineLUT(cfg, ts.n_lut); // 1) keep segments whose radius-padded AABB overlaps this domain AABB. // (We keep the whole segment, not a clipped piece: the kernel only ever @@ -452,6 +477,380 @@ namespace out { return ts; } + // ====================================================================== // + // 2D field lines == contours of the flux function psi // + // ====================================================================== // + + /** + * @brief A coarse, MPI-replicated copy of the in-plane (Bx, By) field. + * @note Component-fastest: (c0,c1,comp) lives at (c1*n0 + c0)*2 + comp. + */ + struct CoarseField2D { + std::vector B; // n0*n1*2 + int n[2] { 0, 0 }; + real_t origin[2] { ZERO, ZERO }; + real_t dx[2] { ONE, ONE }; + }; + + /** + * @brief Integrate the flux function psi(x,y) from the coarse in-plane field. + * @note psi obeys Bx = d psi/dy, By = -d psi/dx; the trapezoidal cumulative + * integral (one pass along x at j=0, then up each column) is path-consistent + * up to div(B) = 0. Because cf is globally replicated, every rank gets the + * SAME psi -> contour levels are identical everywhere -> seamless lines. + * @param[out] psi (n0*n1) flux function, c0-fastest + * @param[out] psi_min,psi_max flux range (for level spacing) + * @param[out] bmin,bmax |B| range (for contour coloring) + */ + inline void computeFlux2D(const CoarseField2D& cf, + std::vector& psi, + real_t& psi_min, + real_t& psi_max, + real_t& bmin, + real_t& bmax) { + const int nx = cf.n[0], ny = cf.n[1]; + psi.assign(static_cast(nx) * ny, ZERO); + auto B = [&](int i, int j, int c) -> real_t { + return cf.B[(static_cast(j) * nx + i) * 2 + c]; + }; + auto P = [&](int i, int j) -> real_t& { + return psi[static_cast(j) * nx + i]; + }; + // bottom row (j = 0): d psi/dx = -By, trapezoidal in x + for (int i = 1; i < nx; ++i) { + P(i, 0) = P(i - 1, 0) - HALF * (B(i - 1, 0, 1) + B(i, 0, 1)) * cf.dx[0]; + } + // each column: d psi/dy = +Bx, trapezoidal in y + for (int i = 0; i < nx; ++i) { + for (int j = 1; j < ny; ++j) { + P(i, j) = P(i, j - 1) + HALF * (B(i, j - 1, 0) + B(i, j, 0)) * cf.dx[1]; + } + } + psi_min = static_cast(1e30); + psi_max = static_cast(-1e30); + bmin = static_cast(1e30); + bmax = static_cast(-1e30); + for (int j = 0; j < ny; ++j) { + for (int i = 0; i < nx; ++i) { + const real_t p = P(i, j); + psi_min = std::min(psi_min, p); + psi_max = std::max(psi_max, p); + const real_t bx = B(i, j, 0), by = B(i, j, 1); + const real_t b = std::sqrt(bx * bx + by * by); + bmin = std::min(bmin, b); + bmax = std::max(bmax, b); + } + } + if (psi_min > psi_max) { + psi_min = ZERO; + psi_max = ONE; + } + if (bmin > bmax) { + bmin = ZERO; + bmax = ONE; + } + } + + /** + * @brief Pack the flux function + contour parameters into a device ContourSet. + * @param line_half_px half the contour line width, in screen pixels + * @param wpp world units per screen pixel (sets the screen-space line width) + */ + inline auto buildContourSet(const CoarseField2D& cf, + const std::vector& psi, + real_t psi_min, + real_t psi_max, + real_t bmin, + real_t bmax, + const FieldLineConfig& cfg, + real_t line_half_px, + real_t wpp) -> ContourSet { + ContourSet cs; + cs.n0 = cf.n[0]; + cs.n1 = cf.n[1]; + cs.origin0 = cf.origin[0]; + cs.origin1 = cf.origin[1]; + cs.dx0 = cf.dx[0]; + cs.dx1 = cf.dx[1]; + const int nlev = std::max(1, cfg.levels); + cs.dlevel = (psi_max > psi_min) + ? (psi_max - psi_min) / static_cast(nlev) + : ONE; + cs.psi_ref = psi_min; + cs.line_half_px = line_half_px; + cs.wpp = wpp; + cs.colormap = cfg.colormap; + cs.n_lut = 256; + real_t vlo = bmin, vhi = bmax; + if (cfg.vmax > cfg.vmin) { // explicit |B| color range overrides auto + vlo = cfg.vmin; + vhi = cfg.vmax; + } + cs.vmin = vlo; + cs.vmax = (vhi > vlo) ? vhi : (vlo + ONE); + cs.lut = buildLineLUT(cfg, cs.n_lut); // by |B| or monochrome (cfg.color) + const std::size_t n = static_cast(cs.n0) * cs.n1; + cs.psi = array_t("fl_psi", std::max(n, 1)); + if (n > 0) { + auto h = Kokkos::create_mirror_view(cs.psi); + for (std::size_t k = 0; k < n; ++k) { + h(k) = psi[k]; + } + Kokkos::deep_copy(cs.psi, h); + } + cs.enabled = true; + return cs; + } + + /** @brief A valid but empty contour set (scene shows no 2D field lines). */ + inline auto emptyContourSet() -> ContourSet { + ContourSet cs; + cs.enabled = false; + cs.psi = array_t("fl_psi_empty", 1); + cs.lut = buildLUT("inferno", 2, { { ZERO, ONE }, { ONE, ONE } }); + return cs; + } + + // ====================================================================== // + // 2D spherical / Kerr-Schild == traced meridional streamlines (nt2py) // + // ====================================================================== // + + namespace fl_hidden { + // bilinear sample of the (Br, Btheta) coarse field at physical (r, theta); + // false if (r, theta) lies outside the grid (the integrator stops there). + inline auto sampleRTh(const CoarseField2D& cf, real_t r, real_t th, real_t B[2]) + -> bool { + const real_t rmin = cf.origin[0], thmin = cf.origin[1]; + const real_t rmax = rmin + cf.n[0] * cf.dx[0]; + const real_t thmax = thmin + cf.n[1] * cf.dx[1]; + const real_t tr = HALF * cf.dx[0], tt = HALF * cf.dx[1]; + if (r < rmin - tr or r > rmax + tr or th < thmin - tt or th > thmax + tt) { + return false; + } + int i0, i1, j0, j1; + real_t a0 = ZERO, a1 = ZERO; + if (cf.n[0] <= 1) { + i0 = 0; + i1 = 0; + } else { + const real_t g = (r - rmin) / cf.dx[0] - HALF; + const real_t f = std::floor(g); + int b = static_cast(f); + a0 = g - f; + if (b < 0) { + b = 0; + a0 = ZERO; + } else if (b > cf.n[0] - 2) { + b = cf.n[0] - 2; + a0 = ONE; + } + i0 = b; + i1 = b + 1; + } + if (cf.n[1] <= 1) { + j0 = 0; + j1 = 0; + } else { + const real_t g = (th - thmin) / cf.dx[1] - HALF; + const real_t f = std::floor(g); + int b = static_cast(f); + a1 = g - f; + if (b < 0) { + b = 0; + a1 = ZERO; + } else if (b > cf.n[1] - 2) { + b = cf.n[1] - 2; + a1 = ONE; + } + j0 = b; + j1 = b + 1; + } + for (int c = 0; c < 2; ++c) { + const real_t c00 = cf.B[(static_cast(j0) * cf.n[0] + i0) * 2 + c]; + const real_t c10 = cf.B[(static_cast(j0) * cf.n[0] + i1) * 2 + c]; + const real_t c01 = cf.B[(static_cast(j1) * cf.n[0] + i0) * 2 + c]; + const real_t c11 = cf.B[(static_cast(j1) * cf.n[0] + i1) * 2 + c]; + const real_t e0 = c00 * (ONE - a0) + c10 * a0; + const real_t e1 = c01 * (ONE - a0) + c11 * a0; + B[c] = e0 * (ONE - a1) + e1 * a1; + } + return true; + } + } // namespace fl_hidden + + /** + * @brief Trace poloidal field lines in the meridional (X, Z) plane (nt2py + * style): integrate (Fx, Fz) = (Br sin th + Bth cos th, Br cos th - Bth sin th) + * by bidirectional RK4 through the coarse (r, theta) field. Polylines are + * returned in meridional world coords (z = 0 so they reuse the 3D tube + * builder); with `mirror` the X<0 half is added as the theta-reflected copy. + * @param cf coarse field: component 0 = Br, 1 = Btheta, grid in (r, theta) + * @param world_per_pixel meridional world units per pixel (seed/length scale) + */ + inline auto traceFieldLinesMeridional(const CoarseField2D& cf, + const FieldLineConfig& cfg, + real_t world_per_pixel, + bool mirror, + real_t& out_vmin, + real_t& out_vmax) + -> std::vector { + using fl_hidden::sampleRTh; + std::vector lines; + if (cf.n[0] < 1 or cf.n[1] < 1) { + return lines; + } + const real_t rmin = cf.origin[0], thmin = cf.origin[1]; + const real_t rmax = rmin + cf.n[0] * cf.dx[0]; + const real_t thmax = thmin + cf.n[1] * cf.dx[1]; + const real_t h = std::max(cfg.step_frac, static_cast(1e-3)) * + cf.dx[0]; // step ~ a coarse dr (a length) + const real_t max_len = cfg.max_len_frac * rmax * static_cast(2); + const real_t eps = static_cast(1e-20); + + auto bmag = [&](real_t X, real_t Z) -> real_t { + const real_t r = std::sqrt(X * X + Z * Z); + const real_t th = std::atan2(std::abs(X), Z); + real_t B[2]; + if (not sampleRTh(cf, r, th, B)) { + return ZERO; + } + return std::sqrt(B[0] * B[0] + B[1] * B[1]); + }; + // unit meridional direction (x dir); false if |F| ~ 0 or outside the grid + auto deriv = [&](const real_t p[2], real_t dir, real_t out[2]) -> bool { + const real_t X = p[0], Z = p[1]; + const real_t r = std::sqrt(X * X + Z * Z); + const real_t th = std::atan2(std::abs(X), Z); + real_t B[2]; + if (not sampleRTh(cf, r, th, B)) { + return false; + } + const real_t st = std::sin(th), ct = std::cos(th); + // (Br, Bth) -> meridional Cartesian; sign of the X-component follows X so + // a line seeded in X>=0 stays in X>=0 (the X<0 half is the mirror image) + const real_t sgn = (X < ZERO) ? -ONE : ONE; + const real_t Fx = sgn * (B[0] * st + B[1] * ct); + const real_t Fz = B[0] * ct - B[1] * st; + const real_t m = std::sqrt(Fx * Fx + Fz * Fz); + if (m < eps) { + return false; + } + const real_t inv = dir / m; + out[0] = Fx * inv; + out[1] = Fz * inv; + return true; + }; + + out_vmin = static_cast(1e30); + out_vmax = static_cast(-1e30); + auto track = [&](real_t m) { + out_vmin = std::min(out_vmin, m); + out_vmax = std::max(out_vmax, m); + }; + auto inDomain = [&](real_t X, real_t Z) -> bool { + const real_t r = std::sqrt(X * X + Z * Z); + const real_t th = std::atan2(std::abs(X), Z); + return (r >= rmin and r <= rmax and th >= thmin and th <= thmax); + }; + + auto integrate = [&](const real_t seed[2], real_t dir) { + Polyline pl; + real_t p[2] = { seed[0], seed[1] }; + real_t m0 = bmag(p[0], p[1]); + if (m0 < eps) { + return; + } + pl.pts.push_back({ p[0], p[1], ZERO }); + pl.scal.push_back(m0); + track(m0); + real_t len = ZERO; + for (int step = 0; step < cfg.max_steps and len < max_len; ++step) { + real_t k1[2], k2[2], k3[2], k4[2], q[2]; + if (not deriv(p, dir, k1)) { + break; + } + q[0] = p[0] + HALF * h * k1[0]; + q[1] = p[1] + HALF * h * k1[1]; + if (not deriv(q, dir, k2)) { + break; + } + q[0] = p[0] + HALF * h * k2[0]; + q[1] = p[1] + HALF * h * k2[1]; + if (not deriv(q, dir, k3)) { + break; + } + q[0] = p[0] + h * k3[0]; + q[1] = p[1] + h * k3[1]; + if (not deriv(q, dir, k4)) { + break; + } + p[0] += (h / static_cast(6)) * + (k1[0] + static_cast(2) * k2[0] + + static_cast(2) * k3[0] + k4[0]); + p[1] += (h / static_cast(6)) * + (k1[1] + static_cast(2) * k2[1] + + static_cast(2) * k3[1] + k4[1]); + if (not inDomain(p[0], p[1])) { + break; + } + const real_t m = bmag(p[0], p[1]); + pl.pts.push_back({ p[0], p[1], ZERO }); + pl.scal.push_back(m); + track(m); + len += h; + } + if (pl.pts.size() >= 2) { + lines.push_back(std::move(pl)); + } + }; + + // seed lattice over the X>=0 meridional half, keeping in-domain seeds + const real_t Xhi = rmax, Zlo = -rmax, Zhi = rmax; + real_t spacing = std::max(cfg.seed_px, ONE) * world_per_pixel; + auto gridCount = [&](real_t s) -> long { + const long nx = std::max(1L, static_cast(std::floor(Xhi / s))); + const long nz = std::max(1L, static_cast(std::floor((Zhi - Zlo) / s))); + return nx * nz; + }; + if (gridCount(spacing) > cfg.seed_max and cfg.seed_max > 0) { + spacing *= std::sqrt(static_cast(gridCount(spacing)) / + static_cast(cfg.seed_max)); + } + const long nx = std::max(1L, static_cast(std::floor(Xhi / spacing))); + const long nz = std::max(1L, + static_cast(std::floor((Zhi - Zlo) / spacing))); + for (long iz = 0; iz < nz; ++iz) { + for (long ix = 0; ix < nx; ++ix) { + const real_t X = (static_cast(ix) + HALF) * Xhi / + static_cast(nx); + const real_t Z = Zlo + (static_cast(iz) + HALF) * (Zhi - Zlo) / + static_cast(nz); + if (not inDomain(X, Z)) { + continue; + } + const real_t seed[2] = { X, Z }; + integrate(seed, ONE); + integrate(seed, -ONE); + } + } + if (out_vmin > out_vmax) { + out_vmin = ZERO; + out_vmax = ONE; + } + // mirror the traced (X>=0) lines into the X<0 half for a full disk + if (mirror) { + const std::size_t n0 = lines.size(); + for (std::size_t i = 0; i < n0; ++i) { + Polyline m = lines[i]; + for (auto& q : m.pts) { + q[0] = -q[0]; + } + lines.push_back(std::move(m)); + } + } + return lines; + } + } // namespace out #endif // OUTPUT_RENDER_FIELDLINES_H diff --git a/src/output/render/renderer.cpp b/src/output/render/renderer.cpp index 30ee7c578..9b32a1e82 100644 --- a/src/output/render/renderer.cpp +++ b/src/output/render/renderer.cpp @@ -300,10 +300,12 @@ namespace out { "fieldlines", "enable", false) or any_fl; if (m_fieldlines.enable) { - if (m_global_extent.size() != 3) { - raise::Warning("output.render.fieldlines is 3D-only; ignoring", HERE); + if (m_global_extent.size() != 2 and m_global_extent.size() != 3) { + raise::Warning("output.render.fieldlines needs a 2D or 3D run; ignoring", + HERE); m_fieldlines.enable = false; } else { + // 3D -> traced tubes inside the volume; 2D -> flux-function contours auto& fl = m_fieldlines; fl.field = toml::find_or(td, "output", "render", "fieldlines", "field", "B"); @@ -317,6 +319,16 @@ namespace out { fl.colormap = toml::find_or(td, "output", "render", "fieldlines", "colormap", "inferno"); + // optional monochrome color [r,g,b]; overrides the colormap when set + fl.color = toml::find_or>(td, "output", "render", + "fieldlines", "color", + std::vector {}); + if (not fl.color.empty() and fl.color.size() != 3) { + raise::Warning("output.render.fieldlines.color must have 3 entries " + "[r,g,b]; ignoring", + HERE); + fl.color.clear(); + } fl.log_scale = toml::find_or(td, "output", "render", "fieldlines", "log", false); fl.vmin = toml::find_or(td, "output", "render", "fieldlines", @@ -332,6 +344,8 @@ namespace out { static_cast(3)); fl.seed_max = toml::find_or(td, "output", "render", "fieldlines", "seed_max", 4096); + fl.levels = toml::find_or(td, "output", "render", "fieldlines", + "levels", 16); } } diff --git a/src/output/render/renderer.h b/src/output/render/renderer.h index 72f1c170d..75e4a45c4 100644 --- a/src/output/render/renderer.h +++ b/src/output/render/renderer.h @@ -85,6 +85,10 @@ namespace out { real_t seed_px { 8 }; // seed lattice spacing in screen pixels real_t tube_px { 2 }; // tube radius in screen pixels std::string colormap { "inferno" }; + // monochrome override: when this holds 3 entries [r,g,b] in [0,1] the lines + // are drawn in that single color instead of the |B| colormap (reads well as + // an overlay on a density/other volume). Empty => color by |B|. + std::vector color {}; bool log_scale { false }; real_t vmin { ZERO }; // tube color range; vmin>=vmax => auto |B| real_t vmax { ZERO }; @@ -92,6 +96,34 @@ namespace out { int max_steps { 4000 }; // per-direction integration cap real_t max_len_frac { static_cast(3) }; // x global box diagonal int seed_max { 4096 }; // hard cap on seed count (spacing grows to fit) + // 2D only: number of evenly-spaced flux-function contour levels (field lines + // in 2D are iso-contours of the out-of-plane vector potential psi) + int levels { 16 }; + }; + + /** + * @brief Device-side 2D field-line geometry: the flux function psi on a coarse + * world grid, contoured per-pixel by the slice rasterizer. + * @note In 2D the in-plane field lines are the iso-contours of the flux + * function psi (Bx = d psi/dy, By = -d psi/dx). psi is integrated on a coarse, + * MPI-replicated copy of the field so the contour levels are global -> the + * lines are seamless across domains. The kernel draws a contour where psi is + * within a (screen-space) line width of a level, colored by |B| = |grad psi|. + */ + struct ContourSet { + array_t psi; // (n0*n1) flux function, c0-fastest + int n0 { 0 }, n1 { 0 }; + real_t origin0 { ZERO }, origin1 { ZERO }; + real_t dx0 { ONE }, dx1 { ONE }; + real_t dlevel { ONE }; // contour spacing in flux units + real_t psi_ref { ZERO }; // reference (zeroth) level + real_t line_half_px { ONE }; // half contour-line width, pixels + real_t wpp { ONE }; // world units per screen pixel + array_t lut; // opaque colormap, by |B| = |grad psi| + int n_lut { 256 }; + real_t vmin { ZERO }, vmax { ONE }; // |B| color range + bool enabled { false }; + std::string colormap { "inferno" }; // for the standalone colorbar }; /** diff --git a/src/output/render/slice2d.hpp b/src/output/render/slice2d.hpp index 335aab0aa..9f798b0b8 100644 --- a/src/output/render/slice2d.hpp +++ b/src/output/render/slice2d.hpp @@ -66,6 +66,36 @@ namespace kernel { const real_t vlo, vhi; const bool log_scale; + // 2D field-line contours: iso-levels of the flux function psi on a coarse + // world grid, colored by |B| = |grad psi|. Cartesian only; drawn where psi + // is within `cline_half_px` (screen px) of a level. `heatmap_on` false -> + // standalone contours (no scalar fill). + array_t cpsi; + const int cn0, cn1; + const real_t corigin0, corigin1, cdx0, cdx1; + const real_t cdlevel, cpsi_ref, cline_half_px, cwpp; + array_t clut; + const int cn_lut; + const real_t cvmin, cvmax; + const bool contour_on; + + // spherical/Kerr field lines: traced meridional streamlines drawn as lines, + // reusing the 3D tube segment-bucket geometry at z == 0 (queried at the + // pixel's (X, Z) world point). Cartesian uses the contours above instead. + array_t lseg; + array_t lcell_start; + array_t lseg_idx; + const int ln_seg; + const real_t line_r2; + const int lgnc0, lgnc1, lgnc2; + const real_t lg0, lg1, lg2, ldx0, ldx1, ldx2; + array_t line_lut; + const int line_n_lut; + const real_t line_vmin, line_vmax; + const bool line_on; + + const bool heatmap_on; + array_t image; // output, (bw*bh, 4) premultiplied RGBA public: @@ -91,6 +121,9 @@ namespace kernel { real_t vlo_, real_t vhi_, bool log_scale_, + const out::ContourSet& contours_, + const out::TubeSet& lines_, + bool heatmap_enabled_, const array_t& image_) : Fld { Fld_ } , comp { comp_ } @@ -114,6 +147,42 @@ namespace kernel { , vlo { vlo_ } , vhi { vhi_ } , log_scale { log_scale_ } + , cpsi { contours_.psi } + , cn0 { contours_.n0 } + , cn1 { contours_.n1 } + , corigin0 { contours_.origin0 } + , corigin1 { contours_.origin1 } + , cdx0 { contours_.dx0 } + , cdx1 { contours_.dx1 } + , cdlevel { contours_.dlevel } + , cpsi_ref { contours_.psi_ref } + , cline_half_px { contours_.line_half_px } + , cwpp { contours_.wpp } + , clut { contours_.lut } + , cn_lut { contours_.n_lut } + , cvmin { contours_.vmin } + , cvmax { contours_.vmax } + , contour_on { contours_.enabled } + , lseg { lines_.seg } + , lcell_start { lines_.cell_start } + , lseg_idx { lines_.seg_idx } + , ln_seg { lines_.n_seg } + , line_r2 { lines_.radius * lines_.radius } + , lgnc0 { lines_.gnc[0] } + , lgnc1 { lines_.gnc[1] } + , lgnc2 { lines_.gnc[2] } + , lg0 { lines_.gorigin[0] } + , lg1 { lines_.gorigin[1] } + , lg2 { lines_.gorigin[2] } + , ldx0 { lines_.gdx[0] } + , ldx1 { lines_.gdx[1] } + , ldx2 { lines_.gdx[2] } + , line_lut { lines_.lut } + , line_n_lut { lines_.n_lut } + , line_vmin { lines_.vmin } + , line_vmax { lines_.vmax } + , line_on { lines_.n_seg > 0 } + , heatmap_on { heatmap_enabled_ } , image { image_ } {} // bilinear sample of the prepared scalar at continuous code coords @@ -138,6 +207,98 @@ namespace kernel { return c0 * (ONE - t1) + c1 * t1; } + // bilinear sample of the coarse flux function psi at world point (x, y) + Inline auto sampleFlux(real_t x, real_t y) const -> real_t { + if (cn0 <= 0 or cn1 <= 0) { + return ZERO; + } + int i0, i1, j0, j1; + real_t t0 = ZERO, t1 = ZERO; + if (cn0 <= 1) { + i0 = 0; + i1 = 0; + } else { + const real_t g0 = (x - corigin0) / cdx0 - HALF; + const real_t f0 = math::floor(g0); + int b0 = static_cast(f0); + t0 = g0 - f0; + if (b0 < 0) { + b0 = 0; + t0 = ZERO; + } else if (b0 > cn0 - 2) { + b0 = cn0 - 2; + t0 = ONE; + } + i0 = b0; + i1 = b0 + 1; + } + if (cn1 <= 1) { + j0 = 0; + j1 = 0; + } else { + const real_t g1 = (y - corigin1) / cdx1 - HALF; + const real_t f1 = math::floor(g1); + int b1 = static_cast(f1); + t1 = g1 - f1; + if (b1 < 0) { + b1 = 0; + t1 = ZERO; + } else if (b1 > cn1 - 2) { + b1 = cn1 - 2; + t1 = ONE; + } + j0 = b1; + j1 = b1 + 1; + } + const real_t c00 = cpsi(j0 * cn0 + i0); + const real_t c10 = cpsi(j0 * cn0 + i1); + const real_t c01 = cpsi(j1 * cn0 + i0); + const real_t c11 = cpsi(j1 * cn0 + i1); + const real_t c0 = c00 * (ONE - t0) + c10 * t0; + const real_t c1 = c01 * (ONE - t0) + c11 * t0; + return c0 * (ONE - t1) + c1 * t1; + } + + // is meridional world point (x, z) within the line width of any traced + // streamline segment? Same bucketed distance-to-segment test as the 3D + // tube kernel, restricted to the z == 0 plane. On a hit, `scalar` is |B|. + Inline auto inLine(real_t x, real_t z, real_t& scalar) const -> bool { + if (ln_seg <= 0) { + return false; + } + const int c0 = static_cast(math::floor((x - lg0) / ldx0)); + const int c1 = static_cast(math::floor((z - lg1) / ldx1)); + const int c2 = static_cast(math::floor((ZERO - lg2) / ldx2)); + if (c0 < 0 or c0 >= lgnc0 or c1 < 0 or c1 >= lgnc1 or c2 < 0 or + c2 >= lgnc2) { + return false; + } + const int lin = (c2 * lgnc1 + c1) * lgnc0 + c0; + const int kb = lcell_start(lin); + const int ke = lcell_start(lin + 1); + real_t best = line_r2; + bool hit = false; + for (int k = kb; k < ke; ++k) { + const int s = lseg_idx(k); + const real_t ax = lseg(s, 0), az = lseg(s, 1); // (X, Z) in slots 0,1 + const real_t bx = lseg(s, 3), bz = lseg(s, 4); + const real_t ex = bx - ax, ez = bz - az; + const real_t wx = x - ax, wz = z - az; + const real_t ee = ex * ex + ez * ez; + real_t tt = (ee > ZERO) ? (wx * ex + wz * ez) / ee : ZERO; + tt = (tt < ZERO) ? ZERO : ((tt > ONE) ? ONE : tt); + const real_t cx = ax + tt * ex, cz = az + tt * ez; + const real_t dx = x - cx, dz = z - cz; + const real_t d2 = dx * dx + dz * dz; + if (d2 < best) { + best = d2; + scalar = lseg(s, 6) * (ONE - tt) + lseg(s, 7) * tt; + hit = true; + } + } + return hit; + } + Inline void operator()(cellidx_t lpx, cellidx_t lpy) const { const auto pix = static_cast(lpy) * static_cast(bw) + @@ -176,32 +337,113 @@ namespace kernel { return; } - const real_t s = sample(cc1, cc2); - // normalize through the transfer-function range - const real_t inv_range = (vhi > vlo) ? (ONE / (vhi - vlo)) : ZERO; - real_t uu; - if (log_scale) { - const real_t log_vlo = math::log10(vlo); - uu = (s > ZERO) ? (math::log10(s) - log_vlo) * inv_range : -ONE; - } else { - uu = (s - vlo) * inv_range; + real_t cr = ZERO, cg = ZERO, cb = ZERO; + bool painted = false; + if (heatmap_on) { + const real_t s = sample(cc1, cc2); + // normalize through the transfer-function range + const real_t inv_range = (vhi > vlo) ? (ONE / (vhi - vlo)) : ZERO; + real_t uu; + if (log_scale) { + const real_t log_vlo = math::log10(vlo); + uu = (s > ZERO) ? (math::log10(s) - log_vlo) * inv_range : -ONE; + } else { + uu = (s - vlo) * inv_range; + } + if (uu < ZERO) { + uu = ZERO; + } else if (uu > ONE) { + uu = ONE; + } + int idx = static_cast(uu * static_cast(n_lut - 1) + HALF); + if (idx < 0) { + idx = 0; + } else if (idx > n_lut - 1) { + idx = n_lut - 1; + } + // opaque LUT: premultiplied with alpha == 1, so this is straight RGB + cr = lut(idx, 0); + cg = lut(idx, 1); + cb = lut(idx, 2); + painted = true; } - if (uu < ZERO) { - uu = ZERO; - } else if (uu > ONE) { - uu = ONE; + // field-line contours: iso-levels of the flux function psi (Cartesian + // only, where (u, v) IS the (x, y) world plane). Drawn over the heatmap + // and colored by |B| = |grad psi|; a screen-space line width keeps the + // contours ~constant thickness regardless of local gradient. + if constexpr (M::CoordType == Coord::Cartesian) { + if (contour_on) { + const real_t psi0 = sampleFlux(u, v); + const real_t pl = sampleFlux(u - cwpp, v); + const real_t pr = sampleFlux(u + cwpp, v); + const real_t pd = sampleFlux(u, v - cwpp); + const real_t pup = sampleFlux(u, v + cwpp); + const real_t gx = (pr - pl) / (TWO * cwpp); + const real_t gy = (pup - pd) / (TWO * cwpp); + const real_t g = math::sqrt(gx * gx + gy * gy); // |B| + const real_t tlev = (cdlevel > ZERO) ? (psi0 - cpsi_ref) / cdlevel + : ZERO; + const real_t nlev = math::floor(tlev + HALF); // nearest level index + const real_t d_world = math::abs(psi0 - (cpsi_ref + nlev * cdlevel)); + const real_t eps = static_cast(1e-30); + const real_t d_screen = (g > eps) ? (d_world / (g * cwpp)) + : static_cast(1e30); + if (d_screen <= cline_half_px) { + const real_t invr = (cvmax > cvmin) ? (ONE / (cvmax - cvmin)) : ZERO; + real_t uu = (g - cvmin) * invr; + if (uu < ZERO) { + uu = ZERO; + } else if (uu > ONE) { + uu = ONE; + } + int idx = static_cast(uu * static_cast(cn_lut - 1) + + HALF); + if (idx < 0) { + idx = 0; + } else if (idx > cn_lut - 1) { + idx = cn_lut - 1; + } + cr = clut(idx, 0); + cg = clut(idx, 1); + cb = clut(idx, 2); + painted = true; + } + } + } else { + // spherical/Kerr: traced meridional streamlines. (u, v) is the (X, Z) + // world point; draw a line where it falls within a segment's width. + if (line_on) { + real_t sB; + if (inLine(u, v, sB)) { + const real_t invr = (line_vmax > line_vmin) + ? (ONE / (line_vmax - line_vmin)) + : ZERO; + real_t uu = (sB - line_vmin) * invr; + if (uu < ZERO) { + uu = ZERO; + } else if (uu > ONE) { + uu = ONE; + } + int idx = static_cast(uu * static_cast(line_n_lut - 1) + + HALF); + if (idx < 0) { + idx = 0; + } else if (idx > line_n_lut - 1) { + idx = line_n_lut - 1; + } + cr = line_lut(idx, 0); + cg = line_lut(idx, 1); + cb = line_lut(idx, 2); + painted = true; + } + } } - int idx = static_cast(uu * static_cast(n_lut - 1) + HALF); - if (idx < 0) { - idx = 0; - } else if (idx > n_lut - 1) { - idx = n_lut - 1; + if (painted) { + image(pix, 0) = cr; + image(pix, 1) = cg; + image(pix, 2) = cb; + image(pix, 3) = ONE; } - // opaque LUT: premultiplied with alpha == 1, so this is straight RGB - image(pix, 0) = lut(idx, 0); - image(pix, 1) = lut(idx, 1); - image(pix, 2) = lut(idx, 2); - image(pix, 3) = ONE; } }; From 4c6632517f1f7dd807ff4c9d4d727781fefebc6d Mon Sep 17 00:00:00 2001 From: Ludwig Boess Date: Tue, 30 Jun 2026 23:54:06 -0400 Subject: [PATCH 048/125] option to plot time label --- input.example.toml | 6 +++ src/framework/domain/metadomain_render.cpp | 6 ++- src/output/render/colorbar.h | 1 + src/output/render/renderer.cpp | 46 +++++++++++++++++++++- src/output/render/renderer.h | 6 ++- 5 files changed, 61 insertions(+), 4 deletions(-) diff --git a/input.example.toml b/input.example.toml index a43ad92cd..96b86884c 100644 --- a/input.example.toml +++ b/input.example.toml @@ -761,6 +761,12 @@ # @type: bool # @default: true mirror = "" + # Draw the current simulation time as a label ("T = ", fixed to 2 + # decimals) in the upper-right corner of the render region, in a contrasting + # color, vertically centered between the frame top and the colorbar. + # @type: bool + # @default: false + time_label = "" # Draw a spine (frame) + axis ticks + labels around the rendered region. # The PNG gains left/bottom margins (background-filled) for the tick labels # and axis names, so they never overlap the data. diff --git a/src/framework/domain/metadomain_render.cpp b/src/framework/domain/metadomain_render.cpp index b6ba21c04..2de1acd03 100644 --- a/src/framework/domain/metadomain_render.cpp +++ b/src/framework/domain/metadomain_render.cpp @@ -862,7 +862,8 @@ namespace ntt { sub.rgba[p * 4 + 3] = image_h(p, 3); } } - g_renderer.compositeAndWrite(sub, order_key, scene_cb, current_step); + g_renderer.compositeAndWrite(sub, order_key, scene_cb, current_step, + current_time); rendered_any = true; } return rendered_any; @@ -1191,7 +1192,8 @@ namespace ntt { sub.rgba[p * 4 + 3] = image_h(p, 3); } } - g_renderer.compositeAndWrite(sub, order_key, scene_cb, current_step); + g_renderer.compositeAndWrite(sub, order_key, scene_cb, current_step, + current_time); rendered_any = true; } return rendered_any; diff --git a/src/output/render/colorbar.h b/src/output/render/colorbar.h index 3c208b850..fbe52cb3b 100644 --- a/src/output/render/colorbar.h +++ b/src/output/render/colorbar.h @@ -53,6 +53,7 @@ namespace out { case '.': { static const uint8_t g[7] = { 0b00000, 0b00000, 0b00000, 0b00000, 0b00000, 0b00110, 0b00110 }; return g; } case '-': { static const uint8_t g[7] = { 0b00000, 0b00000, 0b00000, 0b11111, 0b00000, 0b00000, 0b00000 }; return g; } case '+': { static const uint8_t g[7] = { 0b00000, 0b00100, 0b00100, 0b11111, 0b00100, 0b00100, 0b00000 }; return g; } + case '=': { static const uint8_t g[7] = { 0b00000, 0b00000, 0b11111, 0b00000, 0b11111, 0b00000, 0b00000 }; return g; } case '_': { static const uint8_t g[7] = { 0b00000, 0b00000, 0b00000, 0b00000, 0b00000, 0b00000, 0b11111 }; return g; } case ':': { static const uint8_t g[7] = { 0b00000, 0b00110, 0b00110, 0b00000, 0b00110, 0b00110, 0b00000 }; return g; } case '/': { static const uint8_t g[7] = { 0b00001, 0b00010, 0b00010, 0b00100, 0b01000, 0b01000, 0b10000 }; return g; } diff --git a/src/output/render/renderer.cpp b/src/output/render/renderer.cpp index 9b32a1e82..8a4668ff4 100644 --- a/src/output/render/renderer.cpp +++ b/src/output/render/renderer.cpp @@ -26,6 +26,7 @@ #include #include #include +#include #include #include #include @@ -113,6 +114,10 @@ namespace out { // 2D slice mode (spherical only): mirror the half-plane into a full disk m_mirror = toml::find_or(td, "output", "render", "mirror", true); + // draw the current simulation time in the upper-right corner + m_time_label = toml::find_or(td, "output", "render", "time_label", + false); + // axes: spine + ticks + labels around the rendered region m_axes = toml::find_or(td, "output", "render", "axes", false); m_axis_nticks = toml::find_or(td, "output", "render", "axis_ticks", 5); @@ -356,7 +361,8 @@ namespace out { void Renderer::compositeAndWrite(const SubImage& sub, uint64_t order_key, const Scene& scene, - timestep_t step) const { + timestep_t step, + simtime_t time) const { const std::size_t npix = static_cast(m_width) * static_cast(m_height); const std::size_t n = npix * 4; @@ -418,6 +424,42 @@ namespace out { } }; + // upper-right corner label of the current simulation time. `data_left` is + // the x-offset of the render region inside the buffer (0 without axes, the + // left margin `ml` with axes), so the label sits in the render region's + // top-right, not over the colorbar strip. + auto drawTimeLabel = [&](uint8_t* buf, int cw, int ch, int data_left) { + if (not m_time_label) { + return; + } + const int s = cbar_hidden::scale(m_height); + char tbuf[48]; + // fixed-point so it reads e.g. "T = 12345.67" (up to 5 integer digits + // and 2 decimals; more integer digits still print, never truncated) + std::snprintf(tbuf, sizeof(tbuf), "T = %.2f", + static_cast(time)); + const std::string str(tbuf); + const int tw = static_cast(str.size()) * 6 * s; + const int pad = 3 * s; + const int tx = data_left + m_width - tw - pad; + // vertically center the label between the image top and the top of the + // colorbar bar. drawColorbar uses bar_h = ch/2, so the bar top is at + // bar_y = (ch - bar_h)/2 = ch/4; center the 7*s-tall glyphs in [0, bar_y]. + const int text_h = 7 * s; + const int bar_h = ch / 2; + const int cbar_top = m_colorbar ? (ch - bar_h) / 2 : (ch / 4); + int ty = (cbar_top - text_h) / 2; + if (ty < pad) { + ty = pad; + } + // contrasting text color (white on a dark background, black on light) + const real_t lum = static_cast(0.299) * m_background[0] + + static_cast(0.587) * m_background[1] + + static_cast(0.114) * m_background[2]; + const uint8_t tc = (lum < HALF) ? 255 : 0; + cbar_hidden::drawText(buf, cw, ch, tx, ty, str, s, tc, tc, tc); + }; + // canvas margins: axes (left + bottom) and the colorbar strip (right). // The data region sits at (ml, 0); margins/strip are background-filled. // The polar (curvilinear) overlay annotates inside the data region (the @@ -435,6 +477,7 @@ namespace out { if (CW == m_width and CH == m_height and not m_axes) { // no margins, no outside strip, no overlay: colorbar overlays the data drawBar(data.data(), m_width, m_height); + drawTimeLabel(data.data(), m_width, m_height, 0); ok = write_png(fname, m_width, m_height, data.data()); } else { const uint8_t bR = quantize(m_background[0]); @@ -472,6 +515,7 @@ namespace out { } } drawBar(canvas.data(), CW, CH); + drawTimeLabel(canvas.data(), CW, CH, ml); ok = write_png(fname, CW, CH, canvas.data()); } if (not ok) { diff --git a/src/output/render/renderer.h b/src/output/render/renderer.h index 75e4a45c4..d7aa7b462 100644 --- a/src/output/render/renderer.h +++ b/src/output/render/renderer.h @@ -208,13 +208,15 @@ namespace out { * @param order_key this rank's front-to-back sort key (see composite.h) * @param scene the scene being written (prefix, colorbar colormap/range/label) * @param step current timestep (for the filename cycle number) + * @param time current simulation time (drawn as a corner label if enabled) * @note Uses an order-preserving distributed tree reduce; only the MPI root * rank assembles the full frame and writes the file. */ void compositeAndWrite(const SubImage& sub, uint64_t order_key, const Scene& scene, - timestep_t step) const; + timestep_t step, + simtime_t time) const; /* getters -------------------------------------------------------------- */ [[nodiscard]] @@ -351,6 +353,8 @@ namespace out { // 2D slice mode (spherical): mirror the meridional half-plane across the // symmetry axis to render a full disk from one axisymmetric half bool m_mirror { true }; + // draw the current simulation time as a label in the upper-right corner + bool m_time_label { false }; // draw a spine (frame) + axis ticks/labels around the rendered region bool m_axes { false }; bool m_axis_labels_set { false }; From 95f08f2b180605de641bbde35f44e57c99bb9879 Mon Sep 17 00:00:00 2001 From: Ludwig Boess Date: Wed, 1 Jul 2026 10:27:41 -0400 Subject: [PATCH 049/125] added option to provide axis limits to rendering --- input.example.toml | 12 +++ src/framework/domain/metadomain_render.cpp | 101 +++++++++++++++------ src/output/render/renderer.cpp | 42 ++++++++- src/output/render/renderer.h | 33 +++++++ src/output/render/slice2d.hpp | 28 +++++- 5 files changed, 182 insertions(+), 34 deletions(-) diff --git a/input.example.toml b/input.example.toml index 96b86884c..ce0b917bf 100644 --- a/input.example.toml +++ b/input.example.toml @@ -720,6 +720,18 @@ # @type: int [> 0] # @default: 1024 height = "" + # Axis-aligned render region [lo, hi] in physical/world coords, per axis + # (x1/x2/x3). Any axis left unset spans the full domain. Clamped to the box. + # @type: array [size 2] + # @default: [] (full extent) + # @note: 3D -> the volume is depth-clipped to this box, the wireframe/axes + # frame it, and the default camera zooms to it; 2D -> the slice + # window is framed to it. For a spherical 2D slice, x1_lim crops the + # radius r and x2_lim crops the polar angle theta. + # @example: x1_lim = [-64.0, 64.0] + x1_lim = "" + x2_lim = "" + x3_lim = "" # Number of ray-march steps across the global box diagonal # @type: int [> 0] # @default: 400 diff --git a/src/framework/domain/metadomain_render.cpp b/src/framework/domain/metadomain_render.cpp index 2de1acd03..679077fb3 100644 --- a/src/framework/domain/metadomain_render.cpp +++ b/src/framework/domain/metadomain_render.cpp @@ -663,16 +663,32 @@ namespace ntt { const int W = g_renderer.width(); const int H = g_renderer.height(); - // per-domain world AABB + // optional axis-aligned render region (== full extent when uncropped) + const real_t rlo[3] = { g_renderer.regionLo(0), g_renderer.regionLo(1), + g_renderer.regionLo(2) }; + const real_t rhi[3] = { g_renderer.regionHi(0), g_renderer.regionHi(1), + g_renderer.regionHi(2) }; + // per-domain world AABB, clipped to the region const auto loc_ext = local_domain->mesh.extent(); - real_t lo[3] = { loc_ext[0].first, loc_ext[1].first, loc_ext[2].first }; - real_t hi[3] = { loc_ext[0].second, loc_ext[1].second, loc_ext[2].second }; + real_t lo[3] = { math::max(loc_ext[0].first, rlo[0]), + math::max(loc_ext[1].first, rlo[1]), + math::max(loc_ext[2].first, rlo[2]) }; + real_t hi[3] = { math::min(loc_ext[0].second, rhi[0]), + math::min(loc_ext[1].second, rhi[1]), + math::min(loc_ext[2].second, rhi[2]) }; + // does this domain intersect the region? if not, render nothing (but still + // join the collective composite / field-line reduce below). + const bool in_region = (lo[0] < hi[0]) and (lo[1] < hi[1]) and + (lo[2] < hi[2]); - // fixed global world step (identical on all ranks -> seamless) + // global extent (drives the field-line coarse grid, which spans the full + // field regardless of the crop) const auto glob_ext = mesh().extent(); - real_t gdiag = ZERO; + // fixed world step, identical on all ranks -> seamless. Sized to the region + // diagonal so `samples` spans the (possibly cropped) view. + real_t gdiag = ZERO; for (auto d { 0 }; d < 3; ++d) { - const real_t s = glob_ext[d].second - glob_ext[d].first; + const real_t s = rhi[d] - rlo[d]; gdiag += s * s; } gdiag = math::sqrt(gdiag); @@ -681,14 +697,12 @@ namespace ntt { : gdiag / static_cast(g_renderer.samples()); const int max_steps = 2 * g_renderer.samples() + 16; - // global box + depth-occluded spine (opaque box wireframe rendered inline + // region box + depth-occluded spine (opaque box wireframe rendered inline // in the march so the volume covers its far edges). The visual width is // ~spine_width px; the 0.55*ds floor keeps the thin line gap-free at the // current sampling (raise `samples` for a crisper, thinner line). - real_t glo[3] = { glob_ext[0].first, glob_ext[1].first, - glob_ext[2].first }; - real_t ghi[3] = { glob_ext[0].second, glob_ext[1].second, - glob_ext[2].second }; + real_t glo[3] = { rlo[0], rlo[1], rlo[2] }; + real_t ghi[3] = { rhi[0], rhi[1], rhi[2] }; const real_t px_w = (cam.half_h * static_cast(2)) / static_cast(H); const real_t spine_radius = @@ -719,7 +733,8 @@ namespace ntt { // screen-space bounding box of this domain's footprint (same for all // scenes); we only ray-march and composite within it. int bx0 = 0, by0 = 0, bw = 0, bh = 0; - const bool on_screen = out::screenBBox(cam, W, H, lo, hi, bx0, by0, bw, bh); + const bool on_screen = in_region and + out::screenBBox(cam, W, H, lo, hi, bx0, by0, bw, bh); // ---- magnetic-field-line tubes (built once, shared by every scene) --- // // Every rank coarsens + replicates the field, traces the SAME global @@ -889,23 +904,47 @@ namespace ntt { const int H = g_renderer.height(); const bool mirror = g_renderer.mirror(); - // global slice-plane world window (shared by all ranks -> seamless) - const auto gext = mesh().extent(); - real_t umin, umax, vmin, vmax; + // global slice-plane world window (shared by all ranks -> seamless), + // taken from the optional render region (== full extent when uncropped). + // gext (the full extent) is kept for the field-line coarse grid below. + const auto gext = mesh().extent(); + const real_t x1lo = g_renderer.regionLo(0), x1hi = g_renderer.regionHi(0); + const real_t x2lo = g_renderer.regionLo(1), x2hi = g_renderer.regionHi(1); + real_t umin, umax, vmin, vmax; if constexpr (M::CoordType == Coord::type::Cartesian) { - umin = gext[0].first; - umax = gext[0].second; - vmin = gext[1].first; - vmax = gext[1].second; + umin = x1lo; + umax = x1hi; + vmin = x2lo; + vmax = x2hi; } else { - // meridional (X = r sin th, Z = r cos th) bounding box of the arc - const real_t rmax = gext[0].second; - const real_t th0 = gext[1].first; - const real_t th1 = gext[1].second; - umax = rmax; - umin = mirror ? -rmax : ZERO; - vmax = rmax * math::cos(th0); - vmin = rmax * math::cos(th1); + // meridional (X = r sin th, Z = r cos th) bounding box of the cropped + // annular wedge r in [x1lo, x1hi], theta in [x2lo, x2hi]. Sample the + // boundary (arcs + rays) so the bbox is correct for any theta range. + umin = static_cast(1e30); + umax = static_cast(-1e30); + vmin = static_cast(1e30); + vmax = static_cast(-1e30); + const int NB = 65; + auto accXZ = [&](real_t r, real_t th) { + const real_t X = r * math::sin(th), Z = r * math::cos(th); + umin = std::min(umin, X); + umax = std::max(umax, X); + vmin = std::min(vmin, Z); + vmax = std::max(vmax, Z); + if (mirror) { + umin = std::min(umin, -X); + umax = std::max(umax, -X); + } + }; + for (int k = 0; k < NB; ++k) { + const real_t t = static_cast(k) / static_cast(NB - 1); + const real_t th = x2lo + (x2hi - x2lo) * t; + const real_t rr = x1lo + (x1hi - x1lo) * t; + accXZ(x1lo, th); + accXZ(x1hi, th); + accXZ(rr, x2lo); + accXZ(rr, x2hi); + } } // expand the window to the image aspect (centered) so geometry is not // stretched @@ -951,8 +990,7 @@ namespace ntt { // curvilinear slices get polar axes (R radial + Theta arc); pass the // global (r, theta) extent. if (sph) { - g_renderer.setSlicePolar(true, gext[0].first, gext[0].second, - gext[1].first, gext[1].second, mirror); + g_renderer.setSlicePolar(true, x1lo, x1hi, x2lo, x2hi, mirror); } else { g_renderer.setSlicePolar(false, ZERO, ONE, ZERO, ONE, mirror); } @@ -1167,6 +1205,11 @@ namespace ntt { by0, bw, mirror, + x1lo, + x1hi, + x2lo, + x2hi, + g_renderer.hasRegion(), n1, n2, ext0, diff --git a/src/output/render/renderer.cpp b/src/output/render/renderer.cpp index 8a4668ff4..7d38598ca 100644 --- a/src/output/render/renderer.cpp +++ b/src/output/render/renderer.cpp @@ -124,6 +124,39 @@ namespace out { m_spine_width = toml::find_or(td, "output", "render", "spine_width", static_cast(2)); m_global_extent = global_extent; + + // optional axis-aligned render region (physical coords). Unset axes default + // to the full extent; user limits are clamped to the box (nothing to render + // outside it). x{1,2,3}_lim -> axes {0,1,2} (r/theta for spherical 2D). + m_region = global_extent; + m_has_region = false; + { + const char* keys[3] = { "x1_lim", "x2_lim", "x3_lim" }; + for (std::size_t d = 0; d < global_extent.size() and d < 3; ++d) { + const auto lim = toml::find_or>( + td, "output", "render", keys[d], std::vector {}); + if (lim.empty()) { + continue; + } + if (lim.size() != 2 or lim[1] <= lim[0]) { + raise::Warning("output.render." + std::string(keys[d]) + + " must be [lo, hi] with hi > lo; ignoring", + HERE); + continue; + } + const real_t lo = std::max(lim[0], global_extent[d].first); + const real_t hi = std::min(lim[1], global_extent[d].second); + if (hi > lo) { + m_region[d] = { lo, hi }; + m_has_region = true; + } else { + raise::Warning("output.render." + std::string(keys[d]) + + " does not overlap the domain; ignoring", + HERE); + } + } + } + { const auto al = toml::find_or>( td, "output", "render", "axis_labels", std::vector {}); @@ -151,12 +184,13 @@ namespace out { /* ---- camera (used by the 3D volume mode; the 2D slice path frames itself * and ignores this, so a missing 3rd axis is zero-filled harmlessly) ---- */ + // frame the camera on the render region (== the full extent when uncropped) real_t center[3] = { ZERO, ZERO, ZERO }, size[3] = { ZERO, ZERO, ZERO }; real_t maxext = ZERO; - for (std::size_t d = 0; d < global_extent.size() and d < 3; ++d) { + for (std::size_t d = 0; d < m_region.size() and d < 3; ++d) { center[d] = static_cast(0.5) * - (global_extent[d].first + global_extent[d].second); - size[d] = global_extent[d].second - global_extent[d].first; + (m_region[d].first + m_region[d].second); + size[d] = m_region[d].second - m_region[d].first; maxext = (size[d] > maxext) ? size[d] : maxext; } const real_t diag = std::sqrt(size[0] * size[0] + size[1] * size[1] + @@ -499,7 +533,7 @@ namespace out { if (m_axes) { if (m_global_extent.size() == 3) { out::drawAxes3D(canvas.data(), CW, CH, ml, m_width, m_height, - m_camera_dev, m_global_extent, m_axis_labels, + m_camera_dev, m_region, m_axis_labels, m_background, m_axis_nticks); } else if (polar) { out::drawAxesPolar(canvas.data(), CW, CH, ml, m_width, m_height, diff --git a/src/output/render/renderer.h b/src/output/render/renderer.h index d7aa7b462..c71edc541 100644 --- a/src/output/render/renderer.h +++ b/src/output/render/renderer.h @@ -254,6 +254,34 @@ namespace out { return m_camera_dev; } + // Optional axis-aligned render region in physical/world coords. Always + // resolved (unset axes default to the full global extent), so the driver can + // use these unconditionally. `hasRegion()` reports whether any axis was + // overridden (e.g. to know a crop is active). `d` in {0,1,2} == {x1,x2,x3}. + [[nodiscard]] + auto hasRegion() const -> bool { + return m_has_region; + } + + [[nodiscard]] + auto regionLo(int d) const -> real_t { + const int k = (d < 0) ? 0 : ((d > 2) ? 2 : d); + return (static_cast(k) < m_region.size()) ? m_region[k].first + : ZERO; + } + + [[nodiscard]] + auto regionHi(int d) const -> real_t { + const int k = (d < 0) ? 0 : ((d > 2) ? 2 : d); + return (static_cast(k) < m_region.size()) ? m_region[k].second + : ZERO; + } + + [[nodiscard]] + auto region() const -> const boundaries_t& { + return m_region; + } + // 2D slice mode: mirror a spherical half-plane across the axis into a full // disk (no effect on Cartesian or 3D rendering). [[nodiscard]] @@ -364,6 +392,11 @@ namespace out { // global world box (2 or 3 axes); used to project the 3D axes box and to // know the render mode (size 2 => 2D slice, size 3 => 3D volume). boundaries_t m_global_extent; + // resolved render region [lo, hi] per axis (== global extent unless the + // user set x{1,2,3}_lim); the volume is clipped / the slice window is framed + // to this, and the default camera frames it. + boundaries_t m_region; + bool m_has_region { false }; // 2D slice world window + axis names, set per-frame by the templated Render real_t m_slice_win[4] { ZERO, ONE, ZERO, ONE }; std::string m_slice_xlabel { "x" }; diff --git a/src/output/render/slice2d.hpp b/src/output/render/slice2d.hpp index 9f798b0b8..7936cb3a1 100644 --- a/src/output/render/slice2d.hpp +++ b/src/output/render/slice2d.hpp @@ -56,6 +56,12 @@ namespace kernel { const int bx0, by0, bw; // screen-bbox offset and width (output stride) const bool mirror; // spherical: paint the X<0 reflected half too + // optional physical render-region clip: a pixel is drawn only if its + // coordinate is inside [rx1lo,rx1hi] x [rx2lo,rx2hi] (x1,x2 == x,y for + // Cartesian; r,theta for spherical). Off => the whole domain is drawn. + const real_t rx1lo, rx1hi, rx2lo, rx2hi; + const bool region_clip; + // local-domain active cell counts and View extents (membership + clamping) const real_t n1, n2; const int ext0, ext1; @@ -112,6 +118,11 @@ namespace kernel { int by0_, int bw_, bool mirror_, + real_t rx1lo_, + real_t rx1hi_, + real_t rx2lo_, + real_t rx2hi_, + bool region_clip_, int n1_, int n2_, int ext0_, @@ -138,6 +149,11 @@ namespace kernel { , by0 { by0_ } , bw { bw_ } , mirror { mirror_ } + , rx1lo { rx1lo_ } + , rx1hi { rx1hi_ } + , rx2lo { rx2lo_ } + , rx2hi { rx2hi_ } + , region_clip { region_clip_ } , n1 { static_cast(n1_) } , n2 { static_cast(n2_) } , ext0 { ext0_ } @@ -317,9 +333,15 @@ namespace kernel { const real_t v = vmax - (static_cast(gpy) + HALF) / static_cast(H) * (vmax - vmin); - // world -> continuous local code coords + // world -> continuous local code coords, with an optional physical + // render-region clip (so a crop hides domain data outside the region, not + // just reframes the view) real_t cc1, cc2; if constexpr (M::CoordType == Coord::Cartesian) { + if (region_clip and + (u < rx1lo or u > rx1hi or v < rx2lo or v > rx2hi)) { + return; + } cc1 = metric.template convert<1, Crd::Ph, Crd::Cd>(u); cc2 = metric.template convert<2, Crd::Ph, Crd::Cd>(v); } else { @@ -328,6 +350,10 @@ namespace kernel { } const real_t r = math::sqrt(u * u + v * v); const real_t th = math::atan2(math::abs(u), v); // in [0, pi] + if (region_clip and + (r < rx1lo or r > rx1hi or th < rx2lo or th > rx2hi)) { + return; + } cc1 = metric.template convert<1, Crd::Ph, Crd::Cd>(r); cc2 = metric.template convert<2, Crd::Ph, Crd::Cd>(th); } From 856d66ca7df0683d7ae688a0e47262422fee257f Mon Sep 17 00:00:00 2001 From: Ludwig Boess Date: Wed, 1 Jul 2026 11:50:36 -0400 Subject: [PATCH 050/125] added moving camera to track e.g. shock fronts --- input.example.toml | 13 +++++++ src/framework/domain/metadomain_render.cpp | 8 ++++ src/output/render/renderer.cpp | 44 ++++++++++++++++++++++ src/output/render/renderer.h | 22 ++++++++++- 4 files changed, 86 insertions(+), 1 deletion(-) diff --git a/input.example.toml b/input.example.toml index ce0b917bf..08b43370c 100644 --- a/input.example.toml +++ b/input.example.toml @@ -732,6 +732,19 @@ x1_lim = "" x2_lim = "" x3_lim = "" + # Moving view: translate the render region (and, in 3D, the camera) at this + # velocity in world-units-per-sim-time, to keep a propagating feature (e.g. a + # shock) in frame. Pair with x{1,2,3}_lim (the moving window). 2D uses the + # first two components; the motion starts at `camera_start_time`. + # @type: array [size 2 or 3] + # @default: [] (static view) + # @example: camera_velocity = [0.9, 0.0] # pan along +x1 at 0.9 c + camera_velocity = "" + # Sim time at which the view starts moving (static before it, e.g. to let an + # initial ramp-up finish) + # @type: float + # @default: 0.0 + camera_start_time = "" # Number of ray-march steps across the global box diagonal # @type: int [> 0] # @default: 400 diff --git a/src/framework/domain/metadomain_render.cpp b/src/framework/domain/metadomain_render.cpp index 679077fb3..23535acd9 100644 --- a/src/framework/domain/metadomain_render.cpp +++ b/src/framework/domain/metadomain_render.cpp @@ -659,6 +659,10 @@ namespace ntt { HERE); logger::Checkpoint("Rendering output (3D volume)", HERE); + // advance the moving view (region + camera) to this frame's time before + // reading camera()/region(); collective (same time on all ranks). + g_renderer.updateForTime(current_time); + const auto& cam = g_renderer.camera(); const int W = g_renderer.width(); const int H = g_renderer.height(); @@ -900,6 +904,10 @@ namespace ntt { HERE); logger::Checkpoint("Rendering output (2D slice)", HERE); + // advance the moving view (region window) to this frame's time before + // reading region(); collective (same time on all ranks). + g_renderer.updateForTime(current_time); + const int W = g_renderer.width(); const int H = g_renderer.height(); const bool mirror = g_renderer.mirror(); diff --git a/src/output/render/renderer.cpp b/src/output/render/renderer.cpp index 7d38598ca..003069f06 100644 --- a/src/output/render/renderer.cpp +++ b/src/output/render/renderer.cpp @@ -274,6 +274,30 @@ namespace out { m_camera_dev.half_h = static_cast(0.5) * ortho_height; m_camera_dev.half_w = m_camera_dev.half_h * m_camera_dev.aspect; + /* ---- moving view (pan the region/camera to track a feature) --------- */ + // remember the static region + camera eye; updateForTime() translates them. + m_region_base = m_region; + for (int d = 0; d < 3; ++d) { + m_eye_base[d] = m_camera_dev.eye[d]; + } + { + const auto vel = toml::find_or>( + td, "output", "render", "camera_velocity", std::vector {}); + for (std::size_t d = 0; d < vel.size() and d < 3; ++d) { + m_cam_vel[d] = vel[d]; + } + m_cam_moving = (m_cam_vel[0] != ZERO) or (m_cam_vel[1] != ZERO) or + (m_cam_vel[2] != ZERO); + m_cam_t0 = toml::find_or(td, "output", "render", + "camera_start_time", 0.0); + if (m_cam_moving and not m_has_region and global_extent.size() == 2) { + raise::Warning("output.render.camera_velocity set without x{1,2}_lim: " + "the 2D window will pan off the domain. Set a region to " + "track a feature within it.", + HERE); + } + } + /* ---- scenes --------------------------------------------------------- */ m_scenes.clear(); const auto scenes_arr = toml::find_or(td, @@ -392,6 +416,26 @@ namespace out { logger::Checkpoint("In-situ renderer initialized", HERE); } + void Renderer::updateForTime(simtime_t time) { + if (not m_cam_moving) { + return; + } + const real_t dt = static_cast( + (time > m_cam_t0) ? (time - m_cam_t0) : static_cast(0)); + const real_t shift[3] = { m_cam_vel[0] * dt, m_cam_vel[1] * dt, + m_cam_vel[2] * dt }; + // translate the render region (its width is preserved) + for (std::size_t d = 0; d < m_region.size() and d < 3; ++d) { + m_region[d] = { m_region_base[d].first + shift[d], + m_region_base[d].second + shift[d] }; + } + // translate the 3D camera by the same shift -- a pure pan: forward/right/up + // and the ortho height are unchanged, so only the eye moves. + for (int d = 0; d < 3; ++d) { + m_camera_dev.eye[d] = m_eye_base[d] + shift[d]; + } + } + void Renderer::compositeAndWrite(const SubImage& sub, uint64_t order_key, const Scene& scene, diff --git a/src/output/render/renderer.h b/src/output/render/renderer.h index c71edc541..5bbbb4257 100644 --- a/src/output/render/renderer.h +++ b/src/output/render/renderer.h @@ -202,6 +202,15 @@ namespace out { return m_enabled and m_tracker.shouldWrite(step, time); } + /** + * @brief Advance the moving view to `time`: translate the render region (and + * the 3D camera) by `camera_velocity * max(0, time - camera_start_time)`. + * @note A no-op unless `camera_velocity` was set. Call once per frame, before + * reading region()/camera(). All ranks pass the same time, so the + * shifted view is identical everywhere (the composite stays seamless). + */ + void updateForTime(simtime_t time); + /** * @brief Composite the per-rank sparse sub-image across MPI and write PNG. * @param sub this rank's sparse screen-space sub-image (premultiplied RGBA) @@ -394,9 +403,20 @@ namespace out { boundaries_t m_global_extent; // resolved render region [lo, hi] per axis (== global extent unless the // user set x{1,2,3}_lim); the volume is clipped / the slice window is framed - // to this, and the default camera frames it. + // to this, and the default camera frames it. `m_region` is the CURRENT region + // (shifted by the moving view below); `m_region_base` is the static toml one. boundaries_t m_region; + boundaries_t m_region_base; bool m_has_region { false }; + + // moving view: after `m_cam_t0`, the render region and the 3D camera + // translate at `m_cam_vel` (world units per unit sim-time) to keep a + // propagating feature (e.g. a shock) in frame. `m_eye_base` is the static + // camera eye. See updateForTime(). + real_t m_cam_vel[3] { ZERO, ZERO, ZERO }; + simtime_t m_cam_t0 { 0 }; + bool m_cam_moving { false }; + real_t m_eye_base[3] { ZERO, ZERO, ZERO }; // 2D slice world window + axis names, set per-frame by the templated Render real_t m_slice_win[4] { ZERO, ONE, ZERO, ONE }; std::string m_slice_xlabel { "x" }; From cc78232a91e8f7446979dc297c90f27a25919da9 Mon Sep 17 00:00:00 2001 From: Ludwig Boess Date: Wed, 1 Jul 2026 12:31:02 -0400 Subject: [PATCH 051/125] style updates --- src/output/render/axes.h | 57 ++++++++++++++--------- src/output/render/colorbar.h | 26 +++++++---- src/output/render/renderer.cpp | 85 +++++++++++++++++++++++++--------- 3 files changed, 114 insertions(+), 54 deletions(-) diff --git a/src/output/render/axes.h b/src/output/render/axes.h index 9807ec55b..96904b009 100644 --- a/src/output/render/axes.h +++ b/src/output/render/axes.h @@ -335,6 +335,10 @@ namespace out { real_t u1, real_t v0, real_t v1, + real_t du0, + real_t du1, + real_t dv0, + real_t dv1, const std::string& xlabel, const std::string& ylabel, const real_t bg[3], @@ -347,14 +351,12 @@ namespace out { const int gap = 2 * s; const int th = std::max(0, s / 2 - 1); // spine half-thickness + // [u0,u1]x[v0,v1] is the world window mapped onto the full data region; + // [du0,du1]x[dv0,dv1] is the actual data box (the domain/region), a sub-rect + // when the window was aspect-expanded. The spine + ticks clamp to the DATA + // box so the empty aspect pad stays outside the frame. const int xL = x0, xR = x0 + W - 1, yT = 0, yB = H - 1; - // spine - line(rgba, CW, CH, xL, yT, xR, yT, th, c); - line(rgba, CW, CH, xL, yB, xR, yB, th, c); - line(rgba, CW, CH, xL, yT, xL, yB, th, c); - line(rgba, CW, CH, xR, yT, xR, yB, th, c); - - auto X = [&](real_t u) -> int { + auto X = [&](real_t u) -> int { return static_cast(std::lround( static_cast(xL) + static_cast((u - u0) / (u1 - u0)) * (xR - xL))); @@ -364,35 +366,46 @@ namespace out { static_cast(yB) - static_cast((v - v0) / (v1 - v0)) * (yB - yT))); }; - - // x ticks (bottom): marks + labels below the spine - for (const real_t tv : niceTicks(u0, u1, nticks)) { + auto clampX = [&](int x) { return (x < xL) ? xL : ((x > xR) ? xR : x); }; + auto clampY = [&](int y) { return (y < yT) ? yT : ((y > yB) ? yB : y); }; + const int xLd = clampX(X(du0)), xRd = clampX(X(du1)); + const int yTd = clampY(Y(dv1)), yBd = clampY(Y(dv0)); // dv1 = top + + // spine around the data box + line(rgba, CW, CH, xLd, yTd, xRd, yTd, th, c); + line(rgba, CW, CH, xLd, yBd, xRd, yBd, th, c); + line(rgba, CW, CH, xLd, yTd, xLd, yBd, th, c); + line(rgba, CW, CH, xRd, yTd, xRd, yBd, th, c); + + // x ticks (below the data box): marks + labels + for (const real_t tv : niceTicks(du0, du1, nticks)) { const int x = X(tv); - if (x < xL or x > xR) { + if (x < xLd or x > xRd) { continue; } - line(rgba, CW, CH, x, yB, x, yB + tl, 0, c); + line(rgba, CW, CH, x, yBd, x, yBd + tl, 0, c); const std::string lab = cbar_hidden::fmtNum(tv); - text(rgba, CW, CH, x - textW(lab, s) / 2, yB + tl + gap, lab, s, c); + text(rgba, CW, CH, x - textW(lab, s) / 2, yBd + tl + gap, lab, s, c); } - // y ticks (left): marks + right-aligned labels left of the spine - for (const real_t tv : niceTicks(v0, v1, nticks)) { + // y ticks (left of the data box): marks + right-aligned labels + for (const real_t tv : niceTicks(dv0, dv1, nticks)) { const int y = Y(tv); - if (y < yT or y > yB) { + if (y < yTd or y > yBd) { continue; } - line(rgba, CW, CH, xL, y, xL - tl, y, 0, c); + line(rgba, CW, CH, xLd, y, xLd - tl, y, 0, c); const std::string lab = cbar_hidden::fmtNum(tv); - text(rgba, CW, CH, xL - tl - gap - textW(lab, s), y - ch / 2, lab, s, c); + text(rgba, CW, CH, xLd - tl - gap - textW(lab, s), y - ch / 2, lab, s, c); } // axis names if (not xlabel.empty()) { - text(rgba, CW, CH, xL + W / 2 - textW(xlabel, s) / 2, - yB + tl + gap + ch + gap, xlabel, s, c); + text(rgba, CW, CH, (xLd + xRd) / 2 - textW(xlabel, s) / 2, + yBd + tl + gap + ch + gap, xlabel, s, c); } if (not ylabel.empty()) { - textVert(rgba, CW, CH, std::max(gap, x0 - tl - gap - 7 * (6 * s) - gap - 6 * s), - (yT + yB) / 2 - 4 * s * static_cast(ylabel.size()) / 2, + textVert(rgba, CW, CH, + std::max(gap, xLd - tl - gap - 7 * (6 * s) - gap - 6 * s), + (yTd + yBd) / 2 - 4 * s * static_cast(ylabel.size()) / 2, ylabel, s, c); } } diff --git a/src/output/render/colorbar.h b/src/output/render/colorbar.h index fbe52cb3b..0ff9f085f 100644 --- a/src/output/render/colorbar.h +++ b/src/output/render/colorbar.h @@ -167,6 +167,10 @@ namespace out { * @param bg background RGB (to auto-pick contrasting text color) * @param ticks explicit tick values to label; if empty, 5 evenly-spaced * ticks are generated. Values outside [vmin, vmax] are skipped. + * @param span_top,span_bot vertical pixel span the bar should occupy (e.g. the + * data domain, so the bar is centered on the actual data). When + * span_top < 0 or the span is empty, the bar is half the canvas + * height, centered on the canvas (the default). */ inline void drawColorbar(uint8_t* rgba, int W, @@ -177,23 +181,27 @@ namespace out { bool log_scale, const std::string& label, const real_t bg[3], - const std::vector& ticks = {}) { + const std::vector& ticks = {}, + int span_top = -1, + int span_bot = -1) { using namespace cbar_hidden; - const int s = scale(H); - const int char_h = 8 * s; - const int bar_w = std::max(12, H / 50); - const int bar_h = H / 2; - const int gap = 3 * s; - const int pad = 4 * s; - const int block_w = colorbarBlockWidth(H); + const int s = scale(H); + const int char_h = 8 * s; + const int bar_w = std::max(12, H / 50); + const bool aligned = (span_top >= 0) and (span_bot > span_top); + const int bar_h = aligned ? (span_bot - span_top) : (H / 2); + const int gap = 3 * s; + const int pad = 4 * s; + const int block_w = colorbarBlockWidth(H); // place the bar near the right edge of the (possibly extended) canvas int bar_x = W - block_w + pad; if (bar_x < pad) { bar_x = pad; } - const int bar_y = (H - bar_h) / 2; + // vertical position: aligned to the data span if given, else canvas-centered + const int bar_y = aligned ? span_top : ((H - bar_h) / 2); // contrasting monochrome for text / frame / ticks const real_t lum = static_cast(0.299) * bg[0] + diff --git a/src/output/render/renderer.cpp b/src/output/render/renderer.cpp index 003069f06..54bbd3461 100644 --- a/src/output/render/renderer.cpp +++ b/src/output/render/renderer.cpp @@ -494,19 +494,20 @@ namespace out { scene.prefix.c_str(), static_cast(step)); - auto drawBar = [&](uint8_t* buf, int bw, int bh) { + auto drawBar = [&](uint8_t* buf, int bw, int bh, int span_top, int span_bot) { if (m_colorbar) { drawColorbar(buf, bw, bh, scene.tf.colormap, scene.tf.vmin, scene.tf.vmax, scene.tf.log_scale, scene.label, - m_background, scene.ticks); + m_background, scene.ticks, span_top, span_bot); } }; - // upper-right corner label of the current simulation time. `data_left` is - // the x-offset of the render region inside the buffer (0 without axes, the - // left margin `ml` with axes), so the label sits in the render region's - // top-right, not over the colorbar strip. - auto drawTimeLabel = [&](uint8_t* buf, int cw, int ch, int data_left) { + // Draw the sim-time label, right-aligned to `right_x` and vertically + // centered in the band [0, top_limit] -- i.e. OUTSIDE the plotted data: + // above the colorbar (3D / disk) or, for a 2D slice, in the aspect-pad + // above the data box (so it never sits inside the simulation axes). + auto drawTimeLabel = [&](uint8_t* buf, int cw, int ch, int right_x, + int top_limit) { if (not m_time_label) { return; } @@ -517,16 +518,11 @@ namespace out { std::snprintf(tbuf, sizeof(tbuf), "T = %.2f", static_cast(time)); const std::string str(tbuf); - const int tw = static_cast(str.size()) * 6 * s; - const int pad = 3 * s; - const int tx = data_left + m_width - tw - pad; - // vertically center the label between the image top and the top of the - // colorbar bar. drawColorbar uses bar_h = ch/2, so the bar top is at - // bar_y = (ch - bar_h)/2 = ch/4; center the 7*s-tall glyphs in [0, bar_y]. - const int text_h = 7 * s; - const int bar_h = ch / 2; - const int cbar_top = m_colorbar ? (ch - bar_h) / 2 : (ch / 4); - int ty = (cbar_top - text_h) / 2; + const int tw = static_cast(str.size()) * 6 * s; + const int pad = 3 * s; + const int tx = right_x - tw - pad; + const int text_h = 7 * s; + int ty = (top_limit - text_h) / 2; if (ty < pad) { ty = pad; } @@ -551,11 +547,50 @@ namespace out { const int CW = ml + m_width + strip; const int CH = m_height + mb; + // 2D-Cartesian data box (== the render region, before the aspect-expansion + // that pads the window with background): its top & right edges in + // data-region pixels. The axes/spine clamp to it and the time label sits + // in the pad above it, so neither includes the empty aspect padding. + const bool cart2d = (m_global_extent.size() == 2) and not polar; + int dbox_top = 0; // data box top edge (px from data top) + int dbox_bot = m_height; // data box bottom edge (px) + int dbox_right = m_width; // data box right edge (px from data left) + if (cart2d and m_region.size() >= 2) { + const real_t u0 = m_slice_win[0], u1 = m_slice_win[1]; + const real_t v0 = m_slice_win[2], v1 = m_slice_win[3]; + const real_t du1 = m_region[0].second; // data box right in world (x1) + const real_t dv0 = m_region[1].first; // data box bottom in world (x2) + const real_t dv1 = m_region[1].second; // data box top in world (x2) + if (u1 > u0) { + int r = static_cast(std::lround( + static_cast((du1 - u0) / (u1 - u0)) * (m_width - 1))); + dbox_right = (r < 0) ? 0 : ((r > m_width) ? m_width : r); + } + if (v1 > v0) { + int t = static_cast(std::lround( + static_cast((v1 - dv1) / (v1 - v0)) * (m_height - 1))); + int b = static_cast(std::lround( + static_cast((v1 - dv0) / (v1 - v0)) * (m_height - 1))); + dbox_top = (t < 0) ? 0 : ((t > m_height) ? m_height : t); + dbox_bot = (b < 0) ? 0 : ((b > m_height) ? m_height : b); + } + } + // time-label anchor: for a 2D slice, the top-right of the data box (label + // goes in the pad above it); otherwise the top-right above the colorbar. + const int cbar_top = m_colorbar ? (CH - CH / 2) / 2 : (CH / 4); + const int tl_right = cart2d ? (ml + dbox_right) : (ml + m_width); + const int tl_top = cart2d ? dbox_top : cbar_top; + // colorbar vertical span: aligned to the actual data domain for a 2D slice + // (so it's centered on the data, not the aspect-padded canvas); sentinel + // (-1) elsewhere -> drawColorbar centers it on the canvas as before. + const int cbar_span_top = cart2d ? dbox_top : -1; + const int cbar_span_bot = cart2d ? dbox_bot : -1; + bool ok = true; if (CW == m_width and CH == m_height and not m_axes) { // no margins, no outside strip, no overlay: colorbar overlays the data - drawBar(data.data(), m_width, m_height); - drawTimeLabel(data.data(), m_width, m_height, 0); + drawBar(data.data(), m_width, m_height, cbar_span_top, cbar_span_bot); + drawTimeLabel(data.data(), m_width, m_height, tl_right, tl_top); ok = write_png(fname, m_width, m_height, data.data()); } else { const uint8_t bR = quantize(m_background[0]); @@ -586,14 +621,18 @@ namespace out { m_slice_tmin, m_slice_tmax, m_slice_pmirror, "R", "Theta", m_background, m_axis_nticks); } else { + // data box (== region, un-expanded) so the spine hugs the domain, + // not the aspect-padded window + const real_t du0 = m_region[0].first, du1 = m_region[0].second; + const real_t dv0 = m_region[1].first, dv1 = m_region[1].second; out::drawAxes2D(canvas.data(), CW, CH, ml, m_width, m_height, m_slice_win[0], m_slice_win[1], m_slice_win[2], - m_slice_win[3], m_slice_xlabel, m_slice_ylabel, - m_background, m_axis_nticks); + m_slice_win[3], du0, du1, dv0, dv1, m_slice_xlabel, + m_slice_ylabel, m_background, m_axis_nticks); } } - drawBar(canvas.data(), CW, CH); - drawTimeLabel(canvas.data(), CW, CH, ml); + drawBar(canvas.data(), CW, CH, cbar_span_top, cbar_span_bot); + drawTimeLabel(canvas.data(), CW, CH, tl_right, tl_top); ok = write_png(fname, CW, CH, canvas.data()); } if (not ok) { From 0bf6e745779a129f1ad544a7bf173f25af19ae60 Mon Sep 17 00:00:00 2001 From: Ludwig Boess Date: Wed, 1 Jul 2026 12:58:54 -0400 Subject: [PATCH 052/125] added more colormaps --- input.example.toml | 10 +- src/output/render/transfer_fn.h | 432 ++++++++++++++++++++++++++++++++ 2 files changed, 440 insertions(+), 2 deletions(-) diff --git a/input.example.toml b/input.example.toml index 08b43370c..ddb991e87 100644 --- a/input.example.toml +++ b/input.example.toml @@ -905,7 +905,10 @@ log = "" # Colormap name # @type: string - # @enum: "viridis", "inferno", "plasma", "cool2warm", "gray" + # @enum: "viridis", "inferno", "plasma", "cool2warm", "gray", "RdBu_r", and the + # CMasher maps (BSD-3, https://cmasher.readthedocs.io): "dusk", + # "cosmic", "freeze", "apple", "gothic", "sunburst", "voltage", + # "ocean", "fusion", "prinsenvlag" (an optional "cmr." prefix is ok) # @default: "viridis" colormap = "" # Opacity transfer function: [position, opacity] control points, both in @@ -979,7 +982,10 @@ tube_px = "" # Colormap for the field lines (mapped by |B| along each line) # @type: string - # @enum: "viridis", "inferno", "plasma", "cool2warm", "gray" + # @enum: "viridis", "inferno", "plasma", "cool2warm", "gray", "RdBu_r", and the + # CMasher maps (BSD-3, https://cmasher.readthedocs.io): "dusk", + # "cosmic", "freeze", "apple", "gothic", "sunburst", "voltage", + # "ocean", "fusion", "prinsenvlag" (an optional "cmr." prefix is ok) # @default: "inferno" colormap = "" # Monochrome override: draw the lines in a single [r,g,b] color (each 0..1) diff --git a/src/output/render/transfer_fn.h b/src/output/render/transfer_fn.h index a7d682cb3..75759bf1d 100644 --- a/src/output/render/transfer_fn.h +++ b/src/output/render/transfer_fn.h @@ -84,6 +84,414 @@ namespace out { { 1.0f, 1.0f, 1.0f }, }; + // ----------------------------------------------------------------------- + // CMasher scientific colormaps (https://cmasher.readthedocs.io) + // + // The following anchor tables are uniform downsamplings (33 anchors) of the + // published CMasher colormap data, re-implemented from the source at + // https://github.com/1313e/CMasher (src/cmasher/colormaps//_norm.txt). + // At 33 anchors the linear-interpolation error versus the full 256/511-entry + // tables is < 4.5/255 for every map, i.e. visually indistinguishable. + // + // CMasher is distributed under the BSD 3-Clause License: + // Copyright (c) 2019-2021, Ellert van der Velden + // All rights reserved. + // Redistribution and use in source and binary forms, with or without + // modification, are permitted provided that the copyright notice, this list + // of conditions and the BSD-3-Clause disclaimer are retained. The name of the + // copyright holder may not be used to endorse products without permission. + // ----------------------------------------------------------------------- + + // cmasher::dusk (33 anchors sampled from the 256-entry table) + inline constexpr float dusk[33][3] = { + { 0.000000f, 0.000000f, 0.000000f }, + { 0.006238f, 0.007290f, 0.012708f }, + { 0.018622f, 0.026187f, 0.053076f }, + { 0.029900f, 0.055366f, 0.101787f }, + { 0.030149f, 0.086842f, 0.147576f }, + { 0.015542f, 0.120312f, 0.179819f }, + { 0.010319f, 0.152335f, 0.193847f }, + { 0.029326f, 0.181190f, 0.200163f }, + { 0.066363f, 0.207841f, 0.204068f }, + { 0.103996f, 0.233120f, 0.206493f }, + { 0.141503f, 0.257403f, 0.206888f }, + { 0.180382f, 0.280651f, 0.204334f }, + { 0.222140f, 0.302532f, 0.198293f }, + { 0.267566f, 0.322632f, 0.189012f }, + { 0.316579f, 0.340653f, 0.177441f }, + { 0.368580f, 0.356469f, 0.164931f }, + { 0.422824f, 0.370100f, 0.153037f }, + { 0.471588f, 0.380324f, 0.144475f }, + { 0.528296f, 0.390215f, 0.138408f }, + { 0.585639f, 0.398375f, 0.137901f }, + { 0.643321f, 0.404994f, 0.144156f }, + { 0.701148f, 0.410242f, 0.157700f }, + { 0.758937f, 0.414304f, 0.178666f }, + { 0.816278f, 0.417544f, 0.207638f }, + { 0.871717f, 0.421238f, 0.247431f }, + { 0.918409f, 0.431732f, 0.307531f }, + { 0.940305f, 0.464843f, 0.388730f }, + { 0.947103f, 0.510940f, 0.465808f }, + { 0.949229f, 0.559258f, 0.537097f }, + { 0.949161f, 0.607586f, 0.604625f }, + { 0.947863f, 0.655456f, 0.669259f }, + { 0.945994f, 0.702786f, 0.731248f }, + { 0.944208f, 0.749586f, 0.790456f }, + }; + + // cmasher::cosmic (33 anchors sampled from the 256-entry table) + inline constexpr float cosmic[33][3] = { + { 0.000000f, 0.000000f, 0.000000f }, + { 0.010239f, 0.006872f, 0.013172f }, + { 0.038809f, 0.022373f, 0.054399f }, + { 0.076237f, 0.043279f, 0.104947f }, + { 0.112700f, 0.063138f, 0.158384f }, + { 0.148869f, 0.079429f, 0.215952f }, + { 0.185079f, 0.091867f, 0.278775f }, + { 0.221470f, 0.099655f, 0.348020f }, + { 0.257982f, 0.101322f, 0.424930f }, + { 0.294209f, 0.094291f, 0.510709f }, + { 0.328977f, 0.074083f, 0.605931f }, + { 0.359190f, 0.034502f, 0.708287f }, + { 0.377711f, 0.016748f, 0.805820f }, + { 0.375894f, 0.098454f, 0.873588f }, + { 0.355826f, 0.192411f, 0.902268f }, + { 0.326497f, 0.271620f, 0.906235f }, + { 0.293925f, 0.337879f, 0.899030f }, + { 0.265164f, 0.388114f, 0.889366f }, + { 0.233446f, 0.439282f, 0.877747f }, + { 0.203862f, 0.485773f, 0.867189f }, + { 0.176988f, 0.529060f, 0.858475f }, + { 0.153054f, 0.570207f, 0.851855f }, + { 0.131837f, 0.609999f, 0.847280f }, + { 0.112515f, 0.649034f, 0.844504f }, + { 0.093584f, 0.687770f, 0.843135f }, + { 0.072938f, 0.726550f, 0.842663f }, + { 0.048159f, 0.765608f, 0.842480f }, + { 0.021220f, 0.805066f, 0.841902f }, + { 0.009956f, 0.844909f, 0.840192f }, + { 0.039372f, 0.884930f, 0.836603f }, + { 0.115052f, 0.924556f, 0.830460f }, + { 0.218056f, 0.962249f, 0.821634f }, + { 0.371763f, 0.992456f, 0.816521f }, + }; + + // cmasher::freeze (33 anchors sampled from the 256-entry table) + inline constexpr float freeze[33][3] = { + { 0.000000f, 0.000000f, 0.000000f }, + { 0.010910f, 0.009007f, 0.014906f }, + { 0.039247f, 0.030823f, 0.059160f }, + { 0.074148f, 0.060008f, 0.110444f }, + { 0.106606f, 0.087232f, 0.163635f }, + { 0.137212f, 0.112671f, 0.219689f }, + { 0.166145f, 0.136733f, 0.279301f }, + { 0.193344f, 0.159680f, 0.343042f }, + { 0.218514f, 0.181739f, 0.411377f }, + { 0.241052f, 0.203213f, 0.484587f }, + { 0.259859f, 0.224652f, 0.562539f }, + { 0.272953f, 0.247192f, 0.644102f }, + { 0.276779f, 0.273147f, 0.725721f }, + { 0.265827f, 0.306369f, 0.798716f }, + { 0.236083f, 0.349541f, 0.849696f }, + { 0.192331f, 0.398903f, 0.873582f }, + { 0.145422f, 0.448298f, 0.878894f }, + { 0.112009f, 0.489316f, 0.875932f }, + { 0.099232f, 0.533336f, 0.869029f }, + { 0.122966f, 0.574696f, 0.861171f }, + { 0.170084f, 0.613909f, 0.853779f }, + { 0.227234f, 0.651383f, 0.847515f }, + { 0.289300f, 0.687372f, 0.842698f }, + { 0.355048f, 0.721960f, 0.839586f }, + { 0.424505f, 0.755075f, 0.838630f }, + { 0.497701f, 0.786577f, 0.840759f }, + { 0.573685f, 0.816502f, 0.847460f }, + { 0.650307f, 0.845347f, 0.860140f }, + { 0.725396f, 0.873979f, 0.879132f }, + { 0.797886f, 0.903229f, 0.903669f }, + { 0.867683f, 0.933696f, 0.932619f }, + { 0.935038f, 0.965809f, 0.964980f }, + { 1.000000f, 1.000000f, 1.000000f }, + }; + + // cmasher::apple (33 anchors sampled from the 256-entry table) + inline constexpr float apple[33][3] = { + { 0.000000f, 0.000000f, 0.000000f }, + { 0.018449f, 0.006683f, 0.008999f }, + { 0.069836f, 0.019462f, 0.030130f }, + { 0.124822f, 0.033495f, 0.057048f }, + { 0.180264f, 0.045068f, 0.080127f }, + { 0.236726f, 0.050851f, 0.098639f }, + { 0.294404f, 0.049708f, 0.111860f }, + { 0.353148f, 0.039763f, 0.118281f }, + { 0.412085f, 0.021350f, 0.115115f }, + { 0.467944f, 0.008973f, 0.098099f }, + { 0.512805f, 0.042578f, 0.069239f }, + { 0.544883f, 0.107236f, 0.041311f }, + { 0.569454f, 0.165936f, 0.019988f }, + { 0.589142f, 0.220295f, 0.007127f }, + { 0.604821f, 0.272232f, 0.001856f }, + { 0.616738f, 0.322887f, 0.005791f }, + { 0.624881f, 0.372928f, 0.022422f }, + { 0.628812f, 0.416517f, 0.050591f }, + { 0.629507f, 0.466302f, 0.088693f }, + { 0.625985f, 0.516149f, 0.130045f }, + { 0.618051f, 0.566092f, 0.175071f }, + { 0.605455f, 0.616124f, 0.224400f }, + { 0.587974f, 0.666152f, 0.279195f }, + { 0.566011f, 0.715805f, 0.341631f }, + { 0.543179f, 0.763859f, 0.415364f }, + { 0.534328f, 0.806916f, 0.503615f }, + { 0.564391f, 0.840721f, 0.598311f }, + { 0.628027f, 0.867449f, 0.684639f }, + { 0.703665f, 0.891763f, 0.760351f }, + { 0.781104f, 0.916124f, 0.828098f }, + { 0.857052f, 0.941725f, 0.890058f }, + { 0.930451f, 0.969331f, 0.947434f }, + { 1.000000f, 1.000000f, 1.000000f }, + }; + + // cmasher::gothic (33 anchors sampled from the 256-entry table) + inline constexpr float gothic[33][3] = { + { 0.000000f, 0.000000f, 0.000000f }, + { 0.009103f, 0.009497f, 0.017125f }, + { 0.031259f, 0.032684f, 0.068646f }, + { 0.061552f, 0.062439f, 0.127729f }, + { 0.091294f, 0.088771f, 0.191076f }, + { 0.121658f, 0.111295f, 0.260048f }, + { 0.154308f, 0.129160f, 0.335784f }, + { 0.191149f, 0.140572f, 0.419137f }, + { 0.234428f, 0.142239f, 0.510129f }, + { 0.286418f, 0.128494f, 0.605972f }, + { 0.347449f, 0.092109f, 0.695963f }, + { 0.411712f, 0.037080f, 0.759733f }, + { 0.471546f, 0.020266f, 0.789424f }, + { 0.526274f, 0.060551f, 0.796780f }, + { 0.578066f, 0.108541f, 0.793004f }, + { 0.628368f, 0.151332f, 0.783641f }, + { 0.677621f, 0.190718f, 0.770763f }, + { 0.719390f, 0.224799f, 0.756764f }, + { 0.763054f, 0.267998f, 0.736645f }, + { 0.790627f, 0.328361f, 0.713928f }, + { 0.791142f, 0.402697f, 0.719584f }, + { 0.787649f, 0.467602f, 0.747951f }, + { 0.784959f, 0.526150f, 0.782754f }, + { 0.783559f, 0.580954f, 0.819616f }, + { 0.783736f, 0.633370f, 0.856997f }, + { 0.785586f, 0.684344f, 0.893939f }, + { 0.789112f, 0.734726f, 0.928646f }, + { 0.795666f, 0.785049f, 0.956164f }, + { 0.813132f, 0.833710f, 0.968235f }, + { 0.848938f, 0.877750f, 0.970393f }, + { 0.895684f, 0.918861f, 0.974579f }, + { 0.947082f, 0.959196f, 0.984058f }, + { 1.000000f, 1.000000f, 1.000000f }, + }; + + // cmasher::sunburst (33 anchors sampled from the 256-entry table) + inline constexpr float sunburst[33][3] = { + { 0.000000f, 0.000000f, 0.000000f }, + { 0.014374f, 0.007761f, 0.013830f }, + { 0.057181f, 0.023825f, 0.051569f }, + { 0.107503f, 0.043062f, 0.091435f }, + { 0.159109f, 0.059433f, 0.125622f }, + { 0.212012f, 0.071665f, 0.153805f }, + { 0.266014f, 0.080343f, 0.175895f }, + { 0.320933f, 0.085763f, 0.191981f }, + { 0.376633f, 0.087997f, 0.202182f }, + { 0.432991f, 0.086959f, 0.206564f }, + { 0.489851f, 0.082504f, 0.205087f }, + { 0.546970f, 0.074631f, 0.197554f }, + { 0.603921f, 0.064180f, 0.183549f }, + { 0.659892f, 0.055133f, 0.162324f }, + { 0.713225f, 0.059572f, 0.132732f }, + { 0.760503f, 0.093519f, 0.093993f }, + { 0.796917f, 0.154735f, 0.050648f }, + { 0.819328f, 0.216009f, 0.025564f }, + { 0.837747f, 0.284012f, 0.033882f }, + { 0.851299f, 0.348017f, 0.074896f }, + { 0.861316f, 0.408787f, 0.125585f }, + { 0.868446f, 0.467188f, 0.180749f }, + { 0.873123f, 0.523806f, 0.239715f }, + { 0.875769f, 0.578986f, 0.302576f }, + { 0.876914f, 0.632883f, 0.369564f }, + { 0.877316f, 0.685489f, 0.440900f }, + { 0.878101f, 0.736650f, 0.516675f }, + { 0.880900f, 0.786078f, 0.596685f }, + { 0.887890f, 0.833402f, 0.680181f }, + { 0.901543f, 0.878304f, 0.765633f }, + { 0.924079f, 0.920687f, 0.850654f }, + { 0.957291f, 0.960697f, 0.931527f }, + { 1.000000f, 1.000000f, 1.000000f }, + }; + + // cmasher::voltage (33 anchors sampled from the 256-entry table) + inline constexpr float voltage[33][3] = { + { 0.000000f, 0.000000f, 0.000000f }, + { 0.015829f, 0.007377f, 0.012045f }, + { 0.060346f, 0.022787f, 0.046824f }, + { 0.109165f, 0.041971f, 0.090162f }, + { 0.157514f, 0.059030f, 0.134981f }, + { 0.205921f, 0.071648f, 0.182831f }, + { 0.254512f, 0.079556f, 0.235168f }, + { 0.303074f, 0.082057f, 0.293555f }, + { 0.350914f, 0.078204f, 0.359600f }, + { 0.396550f, 0.067768f, 0.434356f }, + { 0.437427f, 0.055503f, 0.516639f }, + { 0.470517f, 0.059198f, 0.601227f }, + { 0.494085f, 0.093088f, 0.680818f }, + { 0.508421f, 0.145253f, 0.750719f }, + { 0.514803f, 0.203410f, 0.809898f }, + { 0.514558f, 0.262635f, 0.859195f }, + { 0.508787f, 0.321230f, 0.899892f }, + { 0.499930f, 0.371547f, 0.929319f }, + { 0.486263f, 0.427841f, 0.956532f }, + { 0.469915f, 0.482826f, 0.977159f }, + { 0.452770f, 0.536437f, 0.991010f }, + { 0.438275f, 0.588421f, 0.997509f }, + { 0.432385f, 0.638191f, 0.996018f }, + { 0.443151f, 0.684778f, 0.986745f }, + { 0.476274f, 0.727179f, 0.972136f }, + { 0.529358f, 0.765229f, 0.956912f }, + { 0.594140f, 0.799927f, 0.945473f }, + { 0.663403f, 0.832718f, 0.939967f }, + { 0.733296f, 0.864814f, 0.940775f }, + { 0.802260f, 0.897066f, 0.947563f }, + { 0.869795f, 0.930059f, 0.959867f }, + { 0.935781f, 0.964231f, 0.977340f }, + { 1.000000f, 1.000000f, 1.000000f }, + }; + + // cmasher::ocean (33 anchors sampled from the 256-entry table) + inline constexpr float ocean[33][3] = { + { 0.110363f, 0.001691f, 0.253026f }, + { 0.124516f, 0.043066f, 0.289045f }, + { 0.135665f, 0.084868f, 0.324305f }, + { 0.143805f, 0.121839f, 0.358190f }, + { 0.148976f, 0.156972f, 0.390248f }, + { 0.151327f, 0.191269f, 0.420093f }, + { 0.151211f, 0.225122f, 0.447430f }, + { 0.149271f, 0.258662f, 0.472116f }, + { 0.146471f, 0.291895f, 0.494196f }, + { 0.144043f, 0.324790f, 0.513894f }, + { 0.143346f, 0.357322f, 0.531555f }, + { 0.145639f, 0.389496f, 0.547565f }, + { 0.151848f, 0.421347f, 0.562289f }, + { 0.162417f, 0.452929f, 0.576030f }, + { 0.177344f, 0.484306f, 0.589017f }, + { 0.196368f, 0.515533f, 0.601409f }, + { 0.219188f, 0.546652f, 0.613299f }, + { 0.242129f, 0.573802f, 0.623330f }, + { 0.271793f, 0.604715f, 0.634385f }, + { 0.305461f, 0.635419f, 0.645036f }, + { 0.343819f, 0.665728f, 0.655393f }, + { 0.387893f, 0.695314f, 0.665825f }, + { 0.438718f, 0.723698f, 0.677325f }, + { 0.496053f, 0.750479f, 0.691913f }, + { 0.556972f, 0.775909f, 0.711830f }, + { 0.617772f, 0.800881f, 0.737580f }, + { 0.676718f, 0.826177f, 0.768091f }, + { 0.733666f, 0.852223f, 0.802113f }, + { 0.788975f, 0.879240f, 0.838730f }, + { 0.843051f, 0.907368f, 0.877312f }, + { 0.896212f, 0.936732f, 0.917381f }, + { 0.948620f, 0.967497f, 0.958467f }, + { 1.000000f, 1.000000f, 1.000000f }, + }; + + // cmasher::fusion (diverging; 33 anchors sampled from the 511-entry table) + inline constexpr float fusion[33][3] = { + { 0.152696f, 0.015942f, 0.069889f }, + { 0.243393f, 0.027996f, 0.138374f }, + { 0.339396f, 0.022662f, 0.187330f }, + { 0.434194f, 0.019908f, 0.202137f }, + { 0.518461f, 0.063682f, 0.191551f }, + { 0.591978f, 0.129217f, 0.171843f }, + { 0.656187f, 0.199931f, 0.150076f }, + { 0.710976f, 0.275318f, 0.131287f }, + { 0.754925f, 0.356047f, 0.126115f }, + { 0.784581f, 0.436574f, 0.150892f }, + { 0.803828f, 0.525723f, 0.220961f }, + { 0.815422f, 0.614033f, 0.328856f }, + { 0.828457f, 0.697985f, 0.458793f }, + { 0.850099f, 0.777148f, 0.598141f }, + { 0.884001f, 0.852908f, 0.739531f }, + { 0.932364f, 0.926807f, 0.878147f }, + { 1.000000f, 1.000000f, 1.000000f }, + { 0.882739f, 0.938948f, 0.943839f }, + { 0.759644f, 0.882704f, 0.900553f }, + { 0.630638f, 0.828950f, 0.872602f }, + { 0.499428f, 0.774306f, 0.861660f }, + { 0.381603f, 0.714439f, 0.863772f }, + { 0.301845f, 0.647196f, 0.868704f }, + { 0.272703f, 0.573492f, 0.869040f }, + { 0.281047f, 0.499433f, 0.863193f }, + { 0.307182f, 0.414842f, 0.849495f }, + { 0.335700f, 0.322756f, 0.825580f }, + { 0.356873f, 0.220225f, 0.784504f }, + { 0.360275f, 0.107524f, 0.709606f }, + { 0.327077f, 0.044725f, 0.576747f }, + { 0.256240f, 0.065580f, 0.425468f }, + { 0.175982f, 0.062679f, 0.300240f }, + { 0.095379f, 0.037917f, 0.194868f }, + }; + + // cmasher::prinsenvlag (diverging; 33 anchors sampled from the 511-entry table) + inline constexpr float prinsenvlag[33][3] = { + { 0.666523f, 0.321623f, 0.271748f }, + { 0.715454f, 0.343630f, 0.238454f }, + { 0.759975f, 0.370421f, 0.199167f }, + { 0.798794f, 0.402978f, 0.153663f }, + { 0.830257f, 0.442137f, 0.100784f }, + { 0.852097f, 0.488559f, 0.041408f }, + { 0.861821f, 0.542123f, 0.032681f }, + { 0.860262f, 0.599807f, 0.123019f }, + { 0.854920f, 0.655912f, 0.231638f }, + { 0.852701f, 0.704526f, 0.333943f }, + { 0.855639f, 0.752429f, 0.440786f }, + { 0.864494f, 0.797240f, 0.544831f }, + { 0.879227f, 0.839859f, 0.645897f }, + { 0.899719f, 0.880993f, 0.743689f }, + { 0.925990f, 0.921172f, 0.837642f }, + { 0.958977f, 0.960584f, 0.926102f }, + { 1.000000f, 1.000000f, 1.000000f }, + { 0.926250f, 0.969453f, 0.961500f }, + { 0.846431f, 0.941633f, 0.927771f }, + { 0.760433f, 0.915515f, 0.901576f }, + { 0.669026f, 0.889675f, 0.886082f }, + { 0.578083f, 0.861643f, 0.882953f }, + { 0.498336f, 0.829280f, 0.888643f }, + { 0.435785f, 0.792736f, 0.897592f }, + { 0.392859f, 0.755673f, 0.906500f }, + { 0.363227f, 0.713780f, 0.915753f }, + { 0.350577f, 0.669530f, 0.923924f }, + { 0.355484f, 0.622674f, 0.928631f }, + { 0.377364f, 0.573245f, 0.923046f }, + { 0.410189f, 0.524116f, 0.889701f }, + { 0.432426f, 0.483706f, 0.814621f }, + { 0.433280f, 0.452439f, 0.723666f }, + { 0.421591f, 0.424905f, 0.636181f }, + }; + + // ColorBrewer "RdBu" diverging map, reversed (blue -> white -> red), as + // exposed by matplotlib under the name "RdBu_r". matplotlib builds it by + // linearly interpolating these 11 control points, so the uniform-anchor + // scheme above reproduces it to < 1.6/255. + // Colors from ColorBrewer (https://colorbrewer2.org) by Cynthia A. Brewer, + // Geography, Pennsylvania State University -- Apache License 2.0. + inline constexpr float rdbu_r[11][3] = { + { 0.019608f, 0.188235f, 0.380392f }, + { 0.132026f, 0.403460f, 0.676278f }, + { 0.262745f, 0.576471f, 0.764706f }, + { 0.566474f, 0.768704f, 0.868512f }, + { 0.819608f, 0.898039f, 0.941176f }, + { 0.969089f, 0.966474f, 0.964937f }, + { 0.992157f, 0.858824f, 0.780392f }, + { 0.957555f, 0.651211f, 0.515110f }, + { 0.839216f, 0.376471f, 0.301961f }, + { 0.692272f, 0.092272f, 0.167705f }, + { 0.403922f, 0.000000f, 0.121569f }, + }; + inline auto lookup(const std::string& name) -> Anchors { if (name == "inferno") { return { inferno, 9 }; @@ -93,6 +501,30 @@ namespace out { return { cool2warm, 3 }; } else if (name == "gray" or name == "grey") { return { gray, 2 }; + // CMasher scientific colormaps (accept an optional "cmr." prefix so + // names can be copied straight from the CMasher documentation) + } else if (name == "dusk" or name == "cmr.dusk") { + return { dusk, 33 }; + } else if (name == "cosmic" or name == "cmr.cosmic") { + return { cosmic, 33 }; + } else if (name == "freeze" or name == "cmr.freeze") { + return { freeze, 33 }; + } else if (name == "apple" or name == "cmr.apple") { + return { apple, 33 }; + } else if (name == "gothic" or name == "cmr.gothic") { + return { gothic, 33 }; + } else if (name == "sunburst" or name == "cmr.sunburst") { + return { sunburst, 33 }; + } else if (name == "voltage" or name == "cmr.voltage") { + return { voltage, 33 }; + } else if (name == "ocean" or name == "cmr.ocean") { + return { ocean, 33 }; + } else if (name == "fusion" or name == "cmr.fusion") { + return { fusion, 33 }; + } else if (name == "prinsenvlag" or name == "cmr.prinsenvlag") { + return { prinsenvlag, 33 }; + } else if (name == "RdBu_r" or name == "rdbu_r") { + return { rdbu_r, 11 }; } else { // default / "viridis" return { viridis, 9 }; From 9d74601cfef6d5d9e4e72ca21b2cffb6aa049c35 Mon Sep 17 00:00:00 2001 From: LudwigBoess Date: Sun, 5 Jul 2026 21:35:15 +0000 Subject: [PATCH 053/125] always render axis ticks in the foreground --- src/output/render/axes.h | 16 ++++++++++------ 1 file changed, 10 insertions(+), 6 deletions(-) diff --git a/src/output/render/axes.h b/src/output/render/axes.h index 96904b009..72da984bc 100644 --- a/src/output/render/axes.h +++ b/src/output/render/axes.h @@ -621,10 +621,12 @@ namespace out { (void)ok; // The wireframe "spine" is drawn in the ray-march (depth-occluded), so here - // we only annotate. For each axis, pick one *silhouette* edge (its two - // adjacent faces face opposite ways) and, among the two candidates, the one - // whose screen position matches the convention x=bottom, y & z on the left - // of the default diagonal view. + // we only annotate. For each axis, pick one *silhouette* edge: its two + // adjacent faces point opposite ways relative to the camera (one toward it, + // one away), so the edge lies on the box OUTLINE and is always in the + // foreground -- never hidden behind the volume. Among the (usually two) + // silhouette candidates take the one nearest the bottom-left of the image, + // the conventional, view-independent place for axis annotation. auto frontFace = [&](int axis, int side) -> bool { const real_t nrm = (side != 0) ? ONE : -ONE; // outward normal sign return (nrm * (-cam.forward[axis])) > ZERO; // points toward the camera? @@ -641,8 +643,10 @@ namespace out { const int m1 = m0 | (1 << d); const real_t mx = HALF * (cx[m0] + cx[m1]); const real_t my = HALF * (cy[m0] + cy[m1]); - // screen-position convention: x on the bottom, y & z on the left edges - real_t score = (d == 0) ? my : -mx; + // prefer the foreground (bottom-left) edge: larger pixel-y is lower, + // smaller pixel-x is further left. Axis-independent, so it follows the + // camera instead of assuming the default diagonal view. + real_t score = my - mx; if (frontFace(e1, s1) != frontFace(e2, s2)) { score += static_cast(1e6); // strongly prefer silhouette edges } From 28f9dab2ff918d2c63491b6fc2f6a558fe0e9f88 Mon Sep 17 00:00:00 2001 From: LudwigBoess Date: Sun, 5 Jul 2026 21:36:07 +0000 Subject: [PATCH 054/125] python script to preview render orientation --- render_preview.py | 812 ++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 812 insertions(+) create mode 100644 render_preview.py diff --git a/render_preview.py b/render_preview.py new file mode 100644 index 000000000..77891cd21 --- /dev/null +++ b/render_preview.py @@ -0,0 +1,812 @@ +#!/usr/bin/env python3 +""" +render_preview.py -- fast, data-free preview of the entity in-situ renderer's +SCENE GEOMETRY. + +Purpose +------- +Reads a simulation `.toml` and draws the domain box / camera framing / axes / +region crop / field-line seed lattice, WITHOUT any simulation data or +ray-marching. It lets you iterate on camera orientation (e.g. the domain cube of +a 3D turbulence run) and framing without relaunching the simulation. + +It reproduces the SAME camera / projection the C++ renderer uses, so the preview +is trustworthy: the box you see here is the box the renderer will draw. + + IMPORTANT: the camera / projection / region / field-line-lattice math below is + a faithful port of the C++ renderer. If the C++ changes, THIS MUST BE UPDATED + IN SYNC. The controlling C++ sources (verified line-by-line while writing this) + are: + - src/output/render/renderer.cpp + Renderer::init -> toml parse, region resolution, default camera, + ortho/persp basis, ortho_height default = box diag, + eye = center + 1.7*diag*(1,1,1)/sqrt3, fov 35 deg + Renderer::updateForTime -> moving view (pure pan of region + eye) + - src/output/render/composite.h + projectToScreen (~L85-112) -> world -> pixel, ortho & perspective + screenBBox (~L121-163) + - src/output/render/raymarch.hpp + ray generation (~L308-335) -- the inverse of projectToScreen + - src/output/render/axes.h + drawAxes3D / drawAxes2D / drawAxesPolar, niceTicks / niceNum tick style + - src/framework/domain/metadomain_render.cpp + 2D window derivation (Cartesian window vs. spherical meridional wedge, + X = r sin th, Z = r cos th, aspect expansion, mirror), field-line setup + - src/framework/parameters/grid.cpp (~L470-497) + extent parse + theta,phi auto-fill for non-Cartesian metrics + - src/output/render/fieldlines.h + 3D seed lattice: spacing = max(seed_px,1)*wpp, grown by + cbrt(n_seed/seed_max) if over seed_max, ns[d]=floor(size[d]/spacing), + seeds at cell centers. + +Modes +----- + * 3D (Cartesian only -- the renderer's only 3D mode): projects the domain cube + (and region crop, if any) with the exact ray-march camera; optional field- + line SEED lattice scatter (schematic: seeds, not traced lines). + * 2D Cartesian: aspect-expanded slice window + domain/region box + ticks. + * 2D spherical/GR: meridional wedge (arcs at r in {rmin,rmax}, rays at + theta in {tmin,tmax}), mirrored into a full disk if `mirror`. + * 1D: nothing to render (warns). + +Usage +----- + module load python/3.13.0 + python render_preview.py [--out preview.png] [--time T] [--scene N] + +If --out is omitted, saves to /_preview.png. +""" + +import argparse +import math +import os +import sys + +try: + import tomllib # Python 3.11+ +except ModuleNotFoundError: # Python 3.10 and older (e.g. the miniforge3 module) + import tomli as tomllib # same load() API + +import numpy as np + +import matplotlib +matplotlib.use("Agg") # headless cluster: no interactive display +import matplotlib.pyplot as plt +from matplotlib.patches import Rectangle + + +# --------------------------------------------------------------------------- # +# toml helpers (mirror toml::find_or: return default if any key is missing) # +# --------------------------------------------------------------------------- # +def find_or(d, default, *keys): + cur = d + for k in keys: + if not isinstance(cur, dict) or k not in cur: + return default + cur = cur[k] + return cur + + +# --------------------------------------------------------------------------- # +# metric / extent handling (grid.cpp ~L416-497) # +# --------------------------------------------------------------------------- # +CARTESIAN_METRICS = {"minkowski"} +# everything else that entity supports is curvilinear (r-first extent): +# spherical, qspherical, kerr_schild, kerr_schild_0, qkerr_schild +def is_cartesian(metric_name): + return metric_name.strip().lower() in CARTESIAN_METRICS + + +def global_extent(td): + """Replicates grid.cpp: parse grid.extent, auto-fill theta/phi for + non-Cartesian metrics. Returns (extent_pairs, dim, cartesian, metric_name). + + extent_pairs is a list of (lo, hi) of length == dim (== len(resolution)). + """ + resolution = find_or(td, None, "grid", "resolution") + if resolution is None: + raise ValueError("grid.resolution missing") + dim = len(resolution) + metric_name = find_or(td, "minkowski", "grid", "metric", "metric") + cart = is_cartesian(metric_name) + + extent = find_or(td, None, "grid", "extent") + if extent is None: + raise ValueError("grid.extent missing") + # deep copy as list of [lo,hi] + ext = [list(pair) for pair in extent] + + # grid.cpp: if extent has more rows than dim, truncate to dim + if len(ext) > dim: + ext = ext[:dim] + + if not cart: + # non-Cartesian: extent gives only the r-range; append theta,(phi). + # (grid.cpp errors if >1 row is supplied for non-cartesian; we just + # keep the r-row and append.) + ext = [ext[0]] + ext.append([0.0, math.pi]) # theta in [0, pi] (2D and 3D) + if dim == 3: + ext.append([0.0, 2.0 * math.pi]) # phi in [0, 2pi] (3D only) + + if len(ext) != dim: + raise ValueError(f"inferred grid.extent has {len(ext)} rows, expected {dim}") + + pairs = [(float(p[0]), float(p[1])) for p in ext] + return pairs, dim, cart, metric_name + + +# --------------------------------------------------------------------------- # +# Camera (renderer.cpp Renderer::init, composite.h projectToScreen) # +# --------------------------------------------------------------------------- # +def _norm3(a): + n = math.sqrt(a[0] * a[0] + a[1] * a[1] + a[2] * a[2]) + if n > 1e-30: + return (a[0] / n, a[1] / n, a[2] / n) + return (a[0], a[1], a[2]) + + +def _cross3(a, b): + return (a[1] * b[2] - a[2] * b[1], + a[2] * b[0] - a[0] * b[2], + a[0] * b[1] - a[1] * b[0]) + + +def _dot3(a, b): + return a[0] * b[0] + a[1] * b[1] + a[2] * b[2] + + +class Camera: + """POD camera identical to out::CameraDevice, built exactly as in + Renderer::init. eye/forward/right/up in world coords; ortho/persp framing.""" + + def __init__(self, td, region, width, height): + # region is a list of (lo,hi); may be < 3 axes (zero-filled), mirroring + # the C++ which sizes center/size over m_region.size() and d<3. + center = [0.0, 0.0, 0.0] + size = [0.0, 0.0, 0.0] + for d in range(min(len(region), 3)): + center[d] = 0.5 * (region[d][0] + region[d][1]) + size[d] = region[d][1] - region[d][0] + diag = math.sqrt(size[0] ** 2 + size[1] ** 2 + size[2] ** 2) + + cam_ortho = find_or(td, True, "output", "render", "camera", "orthographic") + pos = find_or(td, [], "output", "render", "camera", "position") + look = find_or(td, [], "output", "render", "camera", "look_at") + up = find_or(td, [], "output", "render", "camera", "up") + fov = float(find_or(td, 35.0, "output", "render", "camera", "fov")) + # default ortho_height covers the box from any view -> == box diagonal + ortho_height = float( + find_or(td, diag, "output", "render", "camera", "ortho_height")) + + # default eye: box center pushed back along (1,1,1) by ~1.7 diagonals + eye = [0.0, 0.0, 0.0] + lookat = [0.0, 0.0, 0.0] + for d in range(3): + if len(pos) == 3: + eye[d] = float(pos[d]) + else: + eye[d] = center[d] + 1.7 * diag * 0.57735026919 + lookat[d] = float(look[d]) if len(look) == 3 else center[d] + if len(up) == 3: + upv = [float(up[0]), float(up[1]), float(up[2])] + else: + upv = [0.0, 0.0, 1.0] + + forward = _norm3([lookat[0] - eye[0], lookat[1] - eye[1], + lookat[2] - eye[2]]) + right = _norm3(_cross3(forward, upv)) + up_cam = _cross3(right, forward) # already unit (right,forward unit & perp) + + self.eye = list(eye) + self.forward = list(forward) + self.right = list(right) + self.up = list(up_cam) + self.aspect = float(width) / float(height) + self.tan_half_fov = math.tan(0.5 * fov * math.pi / 180.0) + self.orthographic = bool(cam_ortho) + self.half_h = 0.5 * ortho_height + self.half_w = self.half_h * self.aspect + # keep for the summary line + self.ortho_height = ortho_height + self.fov = fov + + def pan(self, shift): + """Moving view: pure pan of the eye (forward/right/up/half_h unchanged), + matching Renderer::updateForTime.""" + for d in range(3): + self.eye[d] += shift[d] + + def project(self, p, W, H): + """projectToScreen: world point -> (px, py) in pixels (py DOWN, + image origin top-left). Returns None if behind a perspective camera.""" + dx = p[0] - self.eye[0] + dy = p[1] - self.eye[1] + dz = p[2] - self.eye[2] + cx = dx * self.right[0] + dy * self.right[1] + dz * self.right[2] + cy = dx * self.up[0] + dy * self.up[1] + dz * self.up[2] + if self.orthographic: + fx = cx / self.half_w + fy = cy / self.half_h + else: + cz = dx * self.forward[0] + dy * self.forward[1] + dz * self.forward[2] + if cz <= 1e-6: + return None + fx = (cx / cz) / (self.aspect * self.tan_half_fov) + fy = (cy / cz) / self.tan_half_fov + px = (fx + 1.0) * 0.5 * W - 0.5 + py = (1.0 - fy) * 0.5 * H - 0.5 + return (px, py) + + +# --------------------------------------------------------------------------- # +# region resolution (renderer.cpp Renderer::init ~L128-158) # +# --------------------------------------------------------------------------- # +def resolve_region(td, ext): + """m_region starts as the global extent; x{d+1}_lim clamps axis d to the box + if it is a valid [lo,hi] with hi>lo that overlaps. Returns (region, has_region).""" + region = [list(p) for p in ext] + has_region = False + keys = ["x1_lim", "x2_lim", "x3_lim"] + for d in range(min(len(ext), 3)): + lim = find_or(td, [], "output", "render", keys[d]) + if not lim: + continue + if len(lim) != 2 or lim[1] <= lim[0]: + print(f" warning: output.render.{keys[d]} must be [lo,hi] with hi>lo; ignoring") + continue + lo = max(float(lim[0]), ext[d][0]) + hi = min(float(lim[1]), ext[d][1]) + if hi > lo: + region[d] = [lo, hi] + has_region = True + else: + print(f" warning: output.render.{keys[d]} does not overlap the domain; ignoring") + return [tuple(p) for p in region], has_region + + +def apply_moving_view(td, region, ext, cam, time, dim): + """Renderer::updateForTime: dt=max(0, t-t0); shift=vel*dt; pan region+eye.""" + vel = find_or(td, [], "output", "render", "camera_velocity") + if not vel: + return region + v = [0.0, 0.0, 0.0] + for d in range(min(len(vel), 3)): + v[d] = float(vel[d]) + if v == [0.0, 0.0, 0.0]: + return region + t0 = float(find_or(td, 0.0, "output", "render", "camera_start_time")) + dt = max(0.0, time - t0) + shift = [v[0] * dt, v[1] * dt, v[2] * dt] + new_region = [] + for d in range(len(region)): + s = shift[d] if d < 3 else 0.0 + new_region.append((region[d][0] + s, region[d][1] + s)) + cam.pan(shift) # cam only meaningful for the 3D mode; harmless otherwise + return new_region + + +# --------------------------------------------------------------------------- # +# "nice" ticks (axes.h niceNum / niceTicks) for annotation # +# --------------------------------------------------------------------------- # +def _nice_num(x, do_round): + if x <= 0.0: + return 1.0 + e = math.floor(math.log10(x)) + f = x / (10.0 ** e) + if do_round: + nf = 1.0 if f < 1.5 else (2.0 if f < 3.0 else (5.0 if f < 7.0 else 10.0)) + else: + nf = 1.0 if f <= 1.0 else (2.0 if f <= 2.0 else (5.0 if f <= 5.0 else 10.0)) + return nf * (10.0 ** e) + + +def nice_ticks(lo, hi, n): + out = [] + if not (hi > lo) or n < 2: + return out + step = _nice_num((hi - lo) / (n - 1), True) + if step <= 0.0: + return out + g0 = math.ceil(lo / step) * step + eps = 1e-6 * step + v = g0 + while v <= hi + 0.5 * step: + if lo - eps <= v <= hi + eps: + out.append(0.0 if abs(v) < eps else v) + v += step + return out + + +# --------------------------------------------------------------------------- # +# 3D field-line SEED lattice (fieldlines.h traceFieldLines seed setup) # +# --------------------------------------------------------------------------- # +def field_line_seeds_3d(td, region, cam, H): + """Reproduce the seed-lattice geometry (NOT the traced lines). + + world_per_pixel wpp = (cam.half_h*2)/H (metadomain_render.cpp) + spacing = max(seed_px,1)*wpp; if n_seed>seed_max: spacing *= cbrt(n_seed/seed_max) + ns[d] = max(1, floor(size[d]/spacing)); seeds at cell centers over the + field-line COARSE grid, which spans the FULL global extent -- BUT here we + seed over the render region's box (== extent when uncropped), matching the + coarse grid origin=extent.first in the uncropped case. We use the region box + the camera frames so the schematic overlays the drawn cube. + """ + fl_enable = find_or(td, False, "output", "render", "fieldlines", "enable") + # any scene may also request the overlay + scenes = find_or(td, [], "output", "render", "scenes") + any_fl = any( + (find_or(sc, False, "fieldlines") or find_or(sc, "", "field") == "fieldlines") + for sc in scenes + ) + if not (fl_enable or any_fl): + return None + + seed_px = float(find_or(td, 8.0, "output", "render", "fieldlines", "seed_px")) + seed_max = int(find_or(td, 4096, "output", "render", "fieldlines", "seed_max")) + + wpp = (cam.half_h * 2.0) / float(H) + size = [region[d][1] - region[d][0] for d in range(3)] + origin = [region[d][0] for d in range(3)] + + spacing = max(seed_px, 1.0) * wpp + + def count_seeds(sp): + ns = [] + tot = 1 + for d in range(3): + n = max(1, int(math.floor(size[d] / sp))) if sp > 0 else 1 + ns.append(n) + tot *= n + return tot, ns + + n_seed, ns = count_seeds(spacing) + if n_seed > seed_max and seed_max > 0: + grow = (float(n_seed) / float(seed_max)) ** (1.0 / 3.0) + spacing *= grow + n_seed, ns = count_seeds(spacing) + + seeds = [] + for k in range(ns[2]): + for j in range(ns[1]): + for i in range(ns[0]): + seeds.append(( + origin[0] + (i + 0.5) * size[0] / ns[0], + origin[1] + (j + 0.5) * size[1] / ns[1], + origin[2] + (k + 0.5) * size[2] / ns[2], + )) + return seeds, ns + + +# --------------------------------------------------------------------------- # +# cube edges: the 12 edges of an axis-aligned box given by (lo,hi) per axis # +# --------------------------------------------------------------------------- # +def cube_corners(box): + """box: list of (lo,hi) for 3 axes. Corner m: bit0->x, bit1->y, bit2->z + (matches the C++ corner() ordering in axes.h / screenBBox).""" + corners = [] + for m in range(8): + corners.append(( + box[0][1] if (m & 1) else box[0][0], + box[1][1] if (m & 2) else box[1][0], + box[2][1] if (m & 4) else box[2][0], + )) + return corners + + +# the 12 edges as (corner_i, corner_j) index pairs +CUBE_EDGES = [ + (0, 1), (2, 3), (4, 5), (6, 7), # x-parallel + (0, 2), (1, 3), (4, 6), (5, 7), # y-parallel + (0, 4), (1, 5), (2, 6), (3, 7), # z-parallel +] + + +# --------------------------------------------------------------------------- # +# 3D axis-annotation edge selection (axes.h drawAxes3D) # +# # +# A box axis has FOUR parallel edges; the ticks/labels go on exactly one. We # +# only ever pick a *silhouette* edge -- one whose two adjacent faces point in # +# opposite directions relative to the camera (one toward it, one away). A # +# silhouette edge always lies on the drawn OUTLINE of the box, so it is in the # +# foreground -- never occluded by the volume -- and its ticks, pushed outward # +# from the projected centroid, land in empty background. Among the (usually # +# two) silhouette candidates we take the one nearest the bottom-left of the # +# image (score = pixel_y - pixel_x), the conventional place for annotation. # +# This is a faithful port of out::drawAxes3D so the preview matches the # +# renderer's choice of which spine carries the axis. # +# --------------------------------------------------------------------------- # +def _front_face(cam, axis, side): + """True if the box face perpendicular to `axis` at `side` (1=high, 0=low) + faces the camera (its outward normal points back toward the eye).""" + nrm = 1.0 if side else -1.0 + return (nrm * (-cam.forward[axis])) > 0.0 + + +def select_axis_edge_3d(cam, d, cx, cy, ccx, ccy): + """Choose the edge parallel to axis d that carries the ticks + label. + + cx,cy are the 8 projected corner pixel coords (cube_corners order); ccx,ccy + the projected box centroid. Returns (m0, m1, pxd, pyd): the edge's two corner + indices (m0 has axis d at its low end) and the unit screen-space OUTWARD push + direction (perpendicular to the edge, pointing away from the centroid).""" + e1 = 1 if d == 0 else 0 # the two perpendicular axes + e2 = 1 if d == 2 else 2 + best = None + for s1 in (0, 1): + for s2 in (0, 1): + m0 = (s1 << e1) | (s2 << e2) + m1 = m0 | (1 << d) + mx = 0.5 * (cx[m0] + cx[m1]) + my = 0.5 * (cy[m0] + cy[m1]) + score = my - mx # prefer the foreground (bottom-left) edge + if _front_face(cam, e1, s1) != _front_face(cam, e2, s2): + score += 1e6 # strongly prefer silhouette (outline) edges + if best is None or score > best[0]: + best = (score, m0, m1) + _, m0, m1 = best + ex, ey = cx[m1] - cx[m0], cy[m1] - cy[m0] + el = math.hypot(ex, ey) or 1.0 + ex, ey = ex / el, ey / el + pxd, pyd = -ey, ex # screen-perpendicular to the edge + mxv = 0.5 * (cx[m0] + cx[m1]) - ccx + myv = 0.5 * (cy[m0] + cy[m1]) - ccy + if pxd * mxv + pyd * myv < 0.0: # flip to point away from the box centroid + pxd, pyd = -pxd, -pyd + return m0, m1, pxd, pyd + + +# --------------------------------------------------------------------------- # +# 3D preview # +# --------------------------------------------------------------------------- # +def draw_3d(td, ext, region, has_region, cam, W, H, out_path, sim_name): + fig, ax = plt.subplots(figsize=(W / 100.0, H / 100.0), dpi=100) + + def project_box(box, color, lw, label, ls="-"): + corners = cube_corners(box) + proj = [cam.project(c, W, H) for c in corners] + first = True + for (a, b) in CUBE_EDGES: + pa, pb = proj[a], proj[b] + if pa is None or pb is None: + continue # edge with a corner behind a perspective camera + ax.plot([pa[0], pb[0]], [pa[1], pb[1]], color=color, lw=lw, ls=ls, + label=(label if first else None), zorder=3) + first = False + + # full extent (light gray) + project_box([ext[0], ext[1], ext[2]], color="0.6", lw=1.2, + label="full extent") + # region crop, if distinct + if has_region: + project_box([region[0], region[1], region[2]], color="tab:blue", + lw=2.0, label="region crop") + + # axes tick labels: for each axis, pick the FOREGROUND (silhouette) edge of + # the framed box and annotate along it, exactly as out::drawAxes3D does, so + # the labels never end up on an edge hidden behind the volume. + axes_on = find_or(td, False, "output", "render", "axes") + nticks = int(find_or(td, 5, "output", "render", "axis_ticks")) + frame_box = region if has_region else ext + axis_names = find_or(td, [], "output", "render", "axis_labels") + default_names = ["x", "y", "z"] + if axes_on: + corners = cube_corners([frame_box[0], frame_box[1], frame_box[2]]) + cx = [0.0] * 8 + cy = [0.0] * 8 + for m in range(8): + pr = cam.project(corners[m], W, H) + cx[m] = pr[0] if pr is not None else 0.0 + cy[m] = pr[1] if pr is not None else 0.0 + cen = [0.5 * (frame_box[d][0] + frame_box[d][1]) for d in range(3)] + prc = cam.project(cen, W, H) + ccx = prc[0] if prc is not None else 0.0 + ccy = prc[1] if prc is not None else 0.0 + + tl = 8.0 # tick-mark length [px] + num_off = tl + 10.0 # numeric-label center offset from the edge [px] + name_off = tl + 30.0 # axis-name center offset from the edge [px] + for d in range(3): + name = axis_names[d] if d < len(axis_names) else default_names[d] + m0, m1, pxd, pyd = select_axis_edge_3d(cam, d, cx, cy, ccx, ccy) + # o = corner(m0): perpendicular coords fixed, axis d swept for ticks + o = list(corners[m0]) + lo_d, hi_d = frame_box[d][0], frame_box[d][1] + for tv in nice_ticks(lo_d, hi_d, nticks): + p = list(o) + p[d] = tv + pr = cam.project(p, W, H) + if pr is None: + continue + a, b = pr + ax.plot([a, a + pxd * tl], [b, b + pyd * tl], + color="0.35", lw=1.0, zorder=4) + ax.annotate(f"{tv:g}", (a + pxd * num_off, b + pyd * num_off), + fontsize=6, color="0.25", ha="center", va="center") + # axis name at the MIDDLE of the chosen edge, pushed further outward + mid = list(o) + mid[d] = 0.5 * (lo_d + hi_d) + pr = cam.project(mid, W, H) + if pr is not None: + a, b = pr + ax.annotate(name, (a + pxd * name_off, b + pyd * name_off), + fontsize=9, color="k", fontweight="bold", + ha="center", va="center") + + # field-line SEED lattice (schematic scatter, NOT traced lines) + fl = field_line_seeds_3d(td, frame_box, cam, H) + if fl is not None: + seeds, ns = fl + pxs, pys = [], [] + for s in seeds: + pr = cam.project(s, W, H) + if pr is not None: + pxs.append(pr[0]) + pys.append(pr[1]) + if pxs: + ax.scatter(pxs, pys, s=8, c="tab:red", marker="o", alpha=0.6, + edgecolors="none", zorder=2, + label=f"field-line seeds (schematic, {ns[0]}x{ns[1]}x{ns[2]})") + + ax.set_xlim(0, W) + ax.set_ylim(H, 0) # inverted y: origin upper-left, matches the PNG + ax.set_aspect("equal") # lock width:height 1:1 in pixel space + ax.set_xlabel("screen x [px]") + ax.set_ylabel("screen y [px]") + proj_kind = "orthographic" if cam.orthographic else f"perspective (fov {cam.fov:g})" + ax.set_title(f"{sim_name}: 3D scene preview ({proj_kind})", fontsize=10) + ax.legend(loc="upper right", fontsize=7, framealpha=0.85) + fig.tight_layout() + fig.savefig(out_path, dpi=100) + plt.close(fig) + + +# --------------------------------------------------------------------------- # +# 2D window derivation (metadomain_render.cpp 2D branch) # +# --------------------------------------------------------------------------- # +def derive_2d_window(td, ext, region, cartesian, mirror, W, H): + """Return (umin,umax,vmin,vmax) -- the aspect-expanded world window mapped + onto the WxH image, exactly as metadomain_render.cpp derives it.""" + x1lo, x1hi = region[0][0], region[0][1] + x2lo, x2hi = region[1][0], region[1][1] + if cartesian: + umin, umax, vmin, vmax = x1lo, x1hi, x2lo, x2hi + else: + # meridional (X = r sin th, Z = r cos th) bbox of the wedge, sampling the + # boundary (arcs at r={x1lo,x1hi}, rays at theta={x2lo,x2hi}). + umin, umax = 1e30, -1e30 + vmin, vmax = 1e30, -1e30 + NB = 65 + + def accXZ(r, th): + nonlocal umin, umax, vmin, vmax + X = r * math.sin(th) + Z = r * math.cos(th) + umin = min(umin, X); umax = max(umax, X) + vmin = min(vmin, Z); vmax = max(vmax, Z) + if mirror: + umin = min(umin, -X); umax = max(umax, -X) + + for k in range(NB): + t = k / (NB - 1) + th = x2lo + (x2hi - x2lo) * t + rr = x1lo + (x1hi - x1lo) * t + accXZ(x1lo, th); accXZ(x1hi, th) + accXZ(rr, x2lo); accXZ(rr, x2hi) + + # expand the window to the image aspect (centered) so geometry isn't stretched + waspect = (umax - umin) / (vmax - vmin) + iaspect = float(W) / float(H) + if iaspect > waspect: + cu = 0.5 * (umin + umax) + hu = 0.5 * (vmax - vmin) * iaspect + umin, umax = cu - hu, cu + hu + else: + cv = 0.5 * (vmin + vmax) + hv = 0.5 * (umax - umin) / iaspect + vmin, vmax = cv - hv, cv + hv + + # spherical slices get a 1.12x background border so the round outline + labels + # are not clipped (Cartesian fills the frame and needs none). + if not cartesian: + pad = 1.12 + cu = 0.5 * (umin + umax); hu = 0.5 * (umax - umin) * pad + cv = 0.5 * (vmin + vmax); hv = 0.5 * (vmax - vmin) * pad + umin, umax = cu - hu, cu + hu + vmin, vmax = cv - hv, cv + hv + + return umin, umax, vmin, vmax + + +def draw_2d_cartesian(td, ext, region, has_region, W, H, out_path, sim_name): + umin, umax, vmin, vmax = derive_2d_window(td, ext, region, True, False, W, H) + fig, ax = plt.subplots(figsize=(W / 100.0, H / 100.0), dpi=100) + + # aspect-expanded slice window (the background-padded frame) + ax.add_patch(Rectangle((umin, vmin), umax - umin, vmax - vmin, + fill=False, ec="0.7", lw=1.0, ls="--", + label="slice window (aspect-expanded)")) + # full domain box + ax.add_patch(Rectangle((ext[0][0], ext[1][0]), + ext[0][1] - ext[0][0], ext[1][1] - ext[1][0], + fill=False, ec="0.4", lw=1.5, label="domain")) + # region crop + if has_region: + ax.add_patch(Rectangle((region[0][0], region[1][0]), + region[0][1] - region[0][0], + region[1][1] - region[1][0], + fill=False, ec="tab:blue", lw=2.0, + label="region crop")) + + # ticks (nice numbers over the data box == region) + axes_on = find_or(td, False, "output", "render", "axes") + nticks = int(find_or(td, 5, "output", "render", "axis_ticks")) + if axes_on: + for tv in nice_ticks(region[0][0], region[0][1], nticks): + ax.axvline(tv, color="0.85", lw=0.5, zorder=0) + for tv in nice_ticks(region[1][0], region[1][1], nticks): + ax.axhline(tv, color="0.85", lw=0.5, zorder=0) + + ax.set_xlim(umin, umax) + ax.set_ylim(vmin, vmax) # +v up (Cartesian slice: y is up in world) + ax.set_aspect("equal") + ax.set_xlabel("x1") + ax.set_ylabel("x2") + ax.set_title(f"{sim_name}: 2D Cartesian slice preview", fontsize=10) + ax.legend(loc="upper right", fontsize=7, framealpha=0.85) + fig.tight_layout() + fig.savefig(out_path, dpi=100) + plt.close(fig) + + +def draw_2d_spherical(td, ext, region, has_region, mirror, W, H, out_path, sim_name): + umin, umax, vmin, vmax = derive_2d_window(td, ext, region, False, mirror, W, H) + # (r, theta) wedge of the render region + rmin, rmax = region[0][0], region[0][1] + tmin, tmax = region[1][0], region[1][1] + + fig, ax = plt.subplots(figsize=(W / 100.0, H / 100.0), dpi=100) + + def wedge_boundary(rmn, rmx, tmn, tmx, sign, color, lw, label=None): + # outer + inner arcs and two rays, in meridional (X=r sin th, Z=r cos th) + th = np.linspace(tmn, tmx, 200) + # outer arc + ax.plot(sign * rmx * np.sin(th), rmx * np.cos(th), color=color, lw=lw, + label=label) + # inner arc + ax.plot(sign * rmn * np.sin(th), rmn * np.cos(th), color=color, lw=lw) + # rays at tmin, tmax + for tt in (tmn, tmx): + ax.plot([sign * rmn * math.sin(tt), sign * rmx * math.sin(tt)], + [rmn * math.cos(tt), rmx * math.cos(tt)], color=color, lw=lw) + + # full extent wedge (light gray) + wedge_boundary(ext[0][0], ext[0][1], ext[1][0], ext[1][1], 1.0, "0.6", 1.2, + label="full extent") + if mirror: + wedge_boundary(ext[0][0], ext[0][1], ext[1][0], ext[1][1], -1.0, "0.6", 1.2) + + # region wedge (colored) if cropped + if has_region: + wedge_boundary(rmin, rmax, tmin, tmax, 1.0, "tab:blue", 2.0, + label="region crop") + if mirror: + wedge_boundary(rmin, rmax, tmin, tmax, -1.0, "tab:blue", 2.0) + + # radial ticks along the symmetry axis (X=0) + axes_on = find_or(td, False, "output", "render", "axes") + nticks = int(find_or(td, 5, "output", "render", "axis_ticks")) + if axes_on: + for Rv in nice_ticks(0.0, ext[0][1], nticks): + ax.plot(0.0, Rv, marker="+", color="0.3", ms=6) + ax.annotate(f"{Rv:g}", (0.0, Rv), fontsize=6, color="0.25", + xytext=(-8, 0), textcoords="offset points", ha="right", + va="center") + + ax.set_xlim(umin, umax) + ax.set_ylim(vmin, vmax) + ax.set_aspect("equal") + ax.set_xlabel("X = r sin(theta)") + ax.set_ylabel("Z = r cos(theta)") + m = "mirrored" if mirror else "half-plane" + ax.set_title(f"{sim_name}: 2D spherical meridional preview ({m})", fontsize=10) + ax.legend(loc="upper right", fontsize=7, framealpha=0.85) + fig.tight_layout() + fig.savefig(out_path, dpi=100) + plt.close(fig) + + +# --------------------------------------------------------------------------- # +# main # +# --------------------------------------------------------------------------- # +def main(): + ap = argparse.ArgumentParser( + description="Data-free preview of the entity in-situ renderer scene geometry.") + ap.add_argument("toml", help="simulation .toml file") + ap.add_argument("--out", default=None, + help="output PNG (default: /_preview.png)") + ap.add_argument("--time", type=float, default=0.0, + help="sim time T for the moving-view pan (default 0)") + ap.add_argument("--scene", type=int, default=None, + help="scene index (accepted for parity; geometry is scene-" + "independent, so it only affects the reported label)") + args = ap.parse_args() + + if not os.path.isfile(args.toml): + print(f"error: no such file: {args.toml}", file=sys.stderr) + return 2 + + with open(args.toml, "rb") as f: + td = tomllib.load(f) + + # renderer enabled? + if not find_or(td, False, "output", "render", "enable"): + print("note: [output.render].enable is false in this toml; previewing anyway.") + + sim_name = find_or(td, "sim", "simulation", "name") + width = int(find_or(td, 1024, "output", "render", "width")) + height = int(find_or(td, 1024, "output", "render", "height")) + mirror = bool(find_or(td, True, "output", "render", "mirror")) + + ext, dim, cartesian, metric_name = global_extent(td) + + # output path + if args.out: + out_path = args.out + else: + toml_dir = os.path.dirname(os.path.abspath(args.toml)) + out_path = os.path.join(toml_dir, f"{sim_name}_preview.png") + + # region + camera (region drives the default framing) + region, has_region = resolve_region(td, ext) + # camera is only meaningful in 3D, but building it is cheap & shares the pan + cam = Camera(td, region, width, height) + region = apply_moving_view(td, region, ext, cam, args.time, dim) + # NB: the C++ pans the eye but keeps forward/right/up/half_h; we already + # panned cam.eye, and the ortho_height / basis are region-independent after + # the initial framing, so cam is now consistent with the panned region. + + # ---- summary line ------------------------------------------------------- + def fmt_pairs(pairs): + return "[" + ", ".join(f"({p[0]:g},{p[1]:g})" for p in pairs) + "]" + + if dim == 3 and cartesian: + mode = "3D box (Cartesian volume)" + elif dim == 3: + mode = "3D (non-Cartesian: unsupported by renderer)" + elif dim == 2 and cartesian: + mode = "2D Cartesian slice" + elif dim == 2: + mode = "2D spherical meridional slice" + else: + mode = "1D (nothing to render)" + + eye_str = f"({cam.eye[0]:g},{cam.eye[1]:g},{cam.eye[2]:g})" + print(f"mode={mode} | metric={metric_name} | eye={eye_str} | " + f"ortho_height={cam.ortho_height:g} | region={fmt_pairs(region)}") + if args.scene is not None: + print(f" (scene index {args.scene} requested; geometry is scene-independent)") + + # ---- dispatch ----------------------------------------------------------- + if dim == 3 and cartesian: + draw_3d(td, ext, region, has_region, cam, width, height, out_path, sim_name) + elif dim == 3: + print("warning: 3D non-Cartesian is not a renderer mode (3D is Cartesian-" + "only); nothing drawn.") + return 1 + elif dim == 2 and cartesian: + draw_2d_cartesian(td, ext, region, has_region, width, height, out_path, + sim_name) + elif dim == 2: + draw_2d_spherical(td, ext, region, has_region, mirror, width, height, + out_path, sim_name) + else: + print("warning: 1D run -- the renderer is inactive; nothing to preview.") + return 1 + + print(f"wrote {out_path}") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) From e825a25827be68b367479faa5cadd33b7c84cdd4 Mon Sep 17 00:00:00 2001 From: LudwigBoess Date: Sat, 25 Jul 2026 21:09:09 +0000 Subject: [PATCH 055/125] planetarium dome rendering in 2D --- input.example.toml | 46 +++++++++++ src/framework/domain/metadomain_render.cpp | 95 ++++++++++++++++++---- src/output/render/renderer.cpp | 91 ++++++++++++++++++++- src/output/render/renderer.h | 40 +++++++++ src/output/render/slice2d.hpp | 63 ++++++++++++-- 5 files changed, 310 insertions(+), 25 deletions(-) diff --git a/input.example.toml b/input.example.toml index ddb991e87..7c7e8147b 100644 --- a/input.example.toml +++ b/input.example.toml @@ -854,6 +854,52 @@ # @default: the global box diagonal (the whole box fits from any angle) ortho_height = "" + # Fulldome fisheye ("planetarium dome master"). 2D only; a circular image is + # centered in the frame's inscribed circle with the corners left as the + # background (the dome master's black border). Set `width == height` (e.g. + # 4096) for a square master. When enabled, the axes and the outside colorbar + # strip are suppressed so the PNG stays exactly width x height. Seamless + # across MPI domains (the pixel->world map is a shared, deterministic function + # and the tiles stay disjoint). Ignored (with a warning) for 3D. + # @note: CARTESIAN -- the flat plane is warped radially into the disk; use + # `fov`/`radius`/`center`/`projection` below. + # @note: SPHERICAL / GR Kerr-Schild -- the meridional slice is ALREADY a + # disk, so dome mode only mirrors it to a full disk (see `mirror` + # above; keep it true) and fits it to the inscribed circle. The + # `fov`/`radius`/`center`/`projection` keys are ignored (the native + # (X, Z) meridional map is used, with image radius proportional to + # the physical radius r, r=0 at the disk center). + [output.render.dome] + # Build the fisheye dome master instead of the plain slice + # @type: bool + # @default: false + enable = "" + # (Cartesian only) Full dome field of view in degrees (image radius maps + # linearly to the dome zenith angle: the rim is at fov/2) + # @type: float [> 0.0, <= 180.0] + # @default: 180.0 # a full hemisphere + fov = "" + # (Cartesian only) World radius of the circular cutout mapped onto the dome + # @type: float [> 0.0] + # @default: half the shorter domain side (the largest centered disk that + # fits inside the box) + radius = "" + # (Cartesian only) World-space center of the cutout + # @type: array [size 2] + # @default: the domain center + center = "" + # (Cartesian only) How the dome zenith angle maps to a world radius on the + # flat slice + # @type: string + # @enum: "equidistant" (r proportional to angle; the fulldome image + # standard -- a straight radial scaling of the cutout) + # "gnomonic" (r ~ tan(angle); the slice as a flat "ceiling" + # tangent to the dome -- straight sim lines stay straight) + # "stereographic" (r ~ tan(angle/2); conformal, preserves shapes) + # "orthographic" (r ~ sin(angle); the slice as seen face-on) + # @default: "equidistant" + projection = "" + # One scene per scalar field -> one PNG stream. Repeat the table for each. [[output.render.scenes]] # Scalar field to render (a volume render needs a scalar, so vectors are diff --git a/src/framework/domain/metadomain_render.cpp b/src/framework/domain/metadomain_render.cpp index 23535acd9..58030006a 100644 --- a/src/framework/domain/metadomain_render.cpp +++ b/src/framework/domain/metadomain_render.cpp @@ -912,6 +912,18 @@ namespace ntt { const int H = g_renderer.height(); const bool mirror = g_renderer.mirror(); + // fulldome fisheye ("dome master"). Cartesian slices are a flat plane, so + // the kernel warps each pixel radially (fisheye). Curvilinear slices + // (spherical / GR Kerr-Schild) are ALREADY a meridional disk, so dome mode + // there is only a framing change: mirror to a full disk (the `mirror` + // default) and fit that disk to the frame's inscribed circle (the pad skip + // below), while the kernel keeps its native (X, Z) meridional map. Reported + // back so the (metric-agnostic) compositor keeps the frame a clean square. + // All ranks take the same branch (M is fixed per run), so it stays seamless + // across tiles. + const out::DomeMap dome = g_renderer.dome(); + g_renderer.setDomeActive(dome.enabled); + // global slice-plane world window (shared by all ranks -> seamless), // taken from the optional render region (== full extent when uncropped). // gext (the full extent) is kept for the field-line coarse grid below. @@ -973,15 +985,19 @@ namespace ntt { } // spherical slices get a background border so the round outline and its // R/theta labels are not clipped at the frame edges (Cartesian fills the - // frame and draws its ticks in dedicated margins, so it needs none). + // frame and draws its ticks in dedicated margins, so it needs none). A dome + // master skips it: the disk must reach the frame's inscribed circle (which + // the projector maps to the dome horizon), and it draws no axes. if constexpr (M::CoordType != Coord::type::Cartesian) { - const real_t pad = static_cast(1.12); - const real_t cu = HALF * (umin + umax), hu = HALF * (umax - umin) * pad; - const real_t cv = HALF * (vmin + vmax), hv = HALF * (vmax - vmin) * pad; - umin = cu - hu; - umax = cu + hu; - vmin = cv - hv; - vmax = cv + hv; + if (not dome.enabled) { + const real_t pad = static_cast(1.12); + const real_t cu = HALF * (umin + umax), hu = HALF * (umax - umin) * pad; + const real_t cv = HALF * (vmin + vmax), hv = HALF * (vmax - vmin) * pad; + umin = cu - hu; + umax = cu + hu; + vmin = cv - hv; + vmax = cv + hv; + } } // hand the world window + axis names to the (host) axes overlay. Default @@ -1015,8 +1031,40 @@ namespace ntt { // boundary; an arc for spherical, a box for Cartesian) const auto le = local_domain->mesh.extent(); auto toPix = [&](real_t u, real_t v, real_t& px, real_t& py) { - px = (u - umin) / (umax - umin) * static_cast(W) - HALF; - py = (vmax - v) / (vmax - vmin) * static_cast(H) - HALF; + if (dome.enabled and M::CoordType == Coord::type::Cartesian) { + // forward fisheye projection (inverse of the kernel's radial map), + // used to bound this domain's footprint on the dome disk. Cartesian + // only -- curvilinear dome uses the linear (X, Z) map below, matching + // the kernel's native meridional projection. + const real_t cxp = HALF * static_cast(W); + const real_t cyp = HALF * static_cast(H); + const real_t Rpx = HALF * static_cast(std::min(W, H)); + const real_t dx = u - dome.cx, dy = v - dome.cy; + const real_t rw = std::sqrt(dx * dx + dy * dy); + real_t fr = (dome.R > ZERO) ? (rw / dome.R) : ZERO; + if (fr > ONE) { + fr = ONE; // clamp onto the rim (conservative for the bbox) + } + real_t theta; + if (dome.law == out::DomeMap::Gnomonic) { + theta = std::atan(fr * std::tan(dome.theta_max)); + } else if (dome.law == out::DomeMap::Stereographic) { + theta = static_cast(2) * + std::atan(fr * std::tan(HALF * dome.theta_max)); + } else if (dome.law == out::DomeMap::Orthographic) { + theta = std::asin(fr * std::sin(dome.theta_max)); + } else { + theta = fr * dome.theta_max; + } + const real_t rho = (dome.theta_max > ZERO) ? (theta / dome.theta_max) + : fr; + const real_t phi = std::atan2(dy, dx); + px = cxp + rho * Rpx * std::cos(phi) - HALF; + py = cyp - rho * Rpx * std::sin(phi) - HALF; + } else { + px = (u - umin) / (umax - umin) * static_cast(W) - HALF; + py = (vmax - v) / (vmax - vmin) * static_cast(H) - HALF; + } }; real_t minx = static_cast(1e30), miny = static_cast(1e30); real_t maxx = static_cast(-1e30), maxy = static_cast(-1e30); @@ -1029,10 +1077,28 @@ namespace ntt { maxy = std::max(maxy, py); }; if constexpr (M::CoordType == Coord::type::Cartesian) { - acc(le[0].first, le[1].first); - acc(le[0].second, le[1].first); - acc(le[0].first, le[1].second); - acc(le[0].second, le[1].second); + if (dome.enabled) { + // the fisheye map is nonlinear (and a domain straddling the center + // wraps around the image center), so bound the footprint by sampling + // the whole domain-rectangle boundary, not just the 4 corners + const int NB = 65; + const real_t x0 = le[0].first, x1 = le[0].second; + const real_t y0 = le[1].first, y1 = le[1].second; + for (int k = 0; k < NB; ++k) { + const real_t t = static_cast(k) / static_cast(NB - 1); + const real_t xx = x0 + (x1 - x0) * t; + const real_t yy = y0 + (y1 - y0) * t; + acc(xx, y0); + acc(xx, y1); + acc(x0, yy); + acc(x1, yy); + } + } else { + acc(le[0].first, le[1].first); + acc(le[0].second, le[1].first); + acc(le[0].first, le[1].second); + acc(le[0].second, le[1].second); + } } else { const int NB = 33; const real_t r0 = le[0].first, r1 = le[0].second; @@ -1213,6 +1279,7 @@ namespace ntt { by0, bw, mirror, + dome, x1lo, x1hi, x2lo, diff --git a/src/output/render/renderer.cpp b/src/output/render/renderer.cpp index 54bbd3461..1f1a8feaf 100644 --- a/src/output/render/renderer.cpp +++ b/src/output/render/renderer.cpp @@ -157,6 +157,85 @@ namespace out { } } + /* ---- fulldome fisheye (planetarium dome master) --------------------- */ + // A 2D-only radial ("fisheye") projection of a circular cutout centered on + // the domain, drawn into the frame's inscribed circle (corners kept as the + // background border -> a valid dome master). The 3D dome is a separate + // workstream; warn if asked for here so it does not silently fall back to + // the volume camera. + { + const bool dome_enable = toml::find_or(td, "output", "render", "dome", + "enable", false); + if (dome_enable) { + if (global_extent.size() != 2) { + raise::Warning("output.render.dome is 2D-only for now; ignoring", HERE); + } else { + m_dome.enabled = true; + const real_t fov = toml::find_or(td, "output", "render", "dome", + "fov", static_cast(180)); + m_dome.theta_max = HALF * fov * static_cast(constant::PI) / + static_cast(180); + const auto proj = toml::find_or(td, "output", "render", + "dome", "projection", + "equidistant"); + if (proj == "gnomonic") { + m_dome.law = DomeMap::Gnomonic; + } else if (proj == "stereographic") { + m_dome.law = DomeMap::Stereographic; + } else if (proj == "orthographic") { + m_dome.law = DomeMap::Orthographic; + } else { + if (proj != "equidistant") { + raise::Warning("output.render.dome.projection '" + proj + + "' unknown; using 'equidistant'", + HERE); + } + m_dome.law = DomeMap::Equidistant; + } + // the gnomonic (flat-tangent) law diverges as the dome half-FOV -> 90 + // deg (a flat plane never reaches the horizon), so cap it below that. + if (m_dome.law == DomeMap::Gnomonic) { + const real_t cap = static_cast(89.0 * constant::PI / 180.0); + if (m_dome.theta_max >= cap) { + raise::Warning("output.render.dome: 'gnomonic' needs fov < 180 deg " + "(a flat plane cannot reach the dome horizon); " + "capping the half-FOV at 89 deg", + HERE); + m_dome.theta_max = cap; + } + } + // default center = domain center; default radius = the largest disk + // that fits inside the (rectangular) domain (half the shorter side). + const real_t Lx = global_extent[0].second - global_extent[0].first; + const real_t Ly = global_extent[1].second - global_extent[1].first; + m_dome.cx = HALF * (global_extent[0].first + global_extent[0].second); + m_dome.cy = HALF * (global_extent[1].first + global_extent[1].second); + const auto ctr = toml::find_or>( + td, "output", "render", "dome", "center", std::vector {}); + if (ctr.size() == 2) { + m_dome.cx = ctr[0]; + m_dome.cy = ctr[1]; + } else if (not ctr.empty()) { + raise::Warning("output.render.dome.center must have 2 entries " + "[x, y]; using the domain center", + HERE); + } + const real_t rdef = HALF * std::min(Lx, Ly); + m_dome.R = toml::find_or(td, "output", "render", "dome", + "radius", rdef); + if (m_dome.R <= ZERO) { + m_dome.R = rdef; + } + if (m_width != m_height) { + raise::Warning("output.render.dome: width != height; the fisheye " + "disk is centered on the shorter side and the frame " + "is not a square dome master", + HERE); + } + } + } + } + { const auto al = toml::find_or>( td, "output", "render", "axis_labels", std::vector {}); @@ -539,9 +618,12 @@ namespace out { // The polar (curvilinear) overlay annotates inside the data region (the // disk is centered with background around it), so it needs no margins. const bool polar = (m_global_extent.size() == 2) and m_slice_polar; + // a fisheye dome master must stay exactly W x H (its inscribed circle is + // the dome), so it takes no axes margins and no outside colorbar strip. + const bool dome = m_dome_active; int ml = 0, mb = 0; - out::axesMargins(m_axes and not polar, m_height, ml, mb); - const int strip = (m_colorbar and m_colorbar_outside) + out::axesMargins(m_axes and not polar and not dome, m_height, ml, mb); + const int strip = (m_colorbar and m_colorbar_outside and not dome) ? colorbarBlockWidth(m_height) : 0; const int CW = ml + m_width + strip; @@ -551,7 +633,8 @@ namespace out { // that pads the window with background): its top & right edges in // data-region pixels. The axes/spine clamp to it and the time label sits // in the pad above it, so neither includes the empty aspect padding. - const bool cart2d = (m_global_extent.size() == 2) and not polar; + const bool cart2d = (m_global_extent.size() == 2) and not polar and + not dome; int dbox_top = 0; // data box top edge (px from data top) int dbox_bot = m_height; // data box bottom edge (px) int dbox_right = m_width; // data box right edge (px from data left) @@ -609,7 +692,7 @@ namespace out { static_cast(m_width) * 4, &canvas[(static_cast(y) * CW + ml) * 4]); } - if (m_axes) { + if (m_axes and not dome) { if (m_global_extent.size() == 3) { out::drawAxes3D(canvas.data(), CW, CH, ml, m_width, m_height, m_camera_dev, m_region, m_axis_labels, diff --git a/src/output/render/renderer.h b/src/output/render/renderer.h index 5bbbb4257..8fe1fa196 100644 --- a/src/output/render/renderer.h +++ b/src/output/render/renderer.h @@ -168,6 +168,27 @@ namespace out { TransferFunction tf; }; + /** + * @brief Fulldome fisheye ("planetarium dome master") projection parameters. + * @note When `enabled`, the 2D slice rasterizer ignores the linear world + * window and instead maps each pixel radially: the frame's inscribed circle is + * the dome, a pixel at normalized image radius rho in [0,1] is the dome zenith + * angle theta = rho * theta_max (azimuthal-equidistant image law -- the + * fulldome standard), and `law` picks how theta maps to a world radius r in a + * disk of radius `R` centered at (cx, cy). Pixels outside the inscribed circle + * are left transparent, so the corners are the dome master's black border. + * Cartesian 2D only (see Metadomain::Render); a metric-agnostic POD so the + * (templated) Render can copy it by value into the device kernel. + */ + struct DomeMap { + enum Law { Equidistant = 0, Gnomonic = 1, Stereographic = 2, Orthographic = 3 }; + bool enabled { false }; + int law { Equidistant }; + real_t theta_max { static_cast(1.5707963267948966) }; // dome half-FOV (rad) + real_t cx { ZERO }, cy { ZERO }; // world center of the cutout + real_t R { ONE }; // world radius of the cutout + }; + /** * @brief A sparse screen-space sub-image: the bounding box of one domain's * projected footprint plus its premultiplied RGBA pixels. @@ -370,6 +391,20 @@ namespace out { return m_fieldlines; } + // Fulldome fisheye projection (planetarium dome master). `dome()` carries the + // parsed config + resolved center/radius/FOV; `enabled` there reflects the + // toml + a 2D run. The templated Render only activates it for Cartesian, so + // it reports the per-run truth back via setDomeActive(), which the (metric- + // agnostic) compositeAndWrite reads to emit a clean square frame. + [[nodiscard]] + auto dome() const -> const DomeMap& { + return m_dome; + } + + void setDomeActive(bool active) { + m_dome_active = active; + } + private: bool m_enabled { false }; @@ -431,6 +466,11 @@ namespace out { std::vector m_scenes; FieldLineConfig m_fieldlines; + // fulldome fisheye config (parsed in init); m_dome_active is set per-run by + // the templated Render (true only for a 2D Cartesian dome). + DomeMap m_dome; + bool m_dome_active { false }; + tools::Tracker m_tracker; path_t m_root; }; diff --git a/src/output/render/slice2d.hpp b/src/output/render/slice2d.hpp index 7936cb3a1..1949b5902 100644 --- a/src/output/render/slice2d.hpp +++ b/src/output/render/slice2d.hpp @@ -56,6 +56,12 @@ namespace kernel { const int bx0, by0, bw; // screen-bbox offset and width (output stride) const bool mirror; // spherical: paint the X<0 reflected half too + // fulldome fisheye ("dome master"): when enabled, a pixel maps radially + // (azimuthal-equidistant image law) to a world point in a disk of radius + // `R` centered at (cx, cy); pixels outside the inscribed circle stay + // transparent. Cartesian only (host disables it for spherical). + const out::DomeMap dome; + // optional physical render-region clip: a pixel is drawn only if its // coordinate is inside [rx1lo,rx1hi] x [rx2lo,rx2hi] (x1,x2 == x,y for // Cartesian; r,theta for spherical). Off => the whole domain is drawn. @@ -118,6 +124,7 @@ namespace kernel { int by0_, int bw_, bool mirror_, + const out::DomeMap& dome_, real_t rx1lo_, real_t rx1hi_, real_t rx2lo_, @@ -149,6 +156,7 @@ namespace kernel { , by0 { by0_ } , bw { bw_ } , mirror { mirror_ } + , dome { dome_ } , rx1lo { rx1lo_ } , rx1hi { rx1hi_ } , rx2lo { rx2lo_ } @@ -327,18 +335,59 @@ namespace kernel { image(pix, 2) = ZERO; image(pix, 3) = ZERO; - // pixel center -> slice-plane world coords (v flipped so +v is up) - const real_t u = umin + (static_cast(gpx) + HALF) / - static_cast(W) * (umax - umin); - const real_t v = vmax - (static_cast(gpy) + HALF) / - static_cast(H) * (vmax - vmin); + // pixel center -> slice-plane world coords (v flipped so +v is up). + // Cartesian dome mode maps the pixel radially (fisheye) instead, leaving + // the corners (outside the inscribed circle) transparent. Curvilinear + // slices are already a meridional disk, so they keep the linear window + // even in dome mode -> the fisheye is compile-time gated to Cartesian. + real_t u, v; + bool dome_fisheye = false; + if constexpr (M::CoordType == Coord::Cartesian) { + dome_fisheye = dome.enabled; + } + if (dome_fisheye) { + const real_t cxp = HALF * static_cast(W); + const real_t cyp = HALF * static_cast(H); + const real_t Rpx = HALF * static_cast((W < H) ? W : H); + const real_t dxp = (static_cast(gpx) + HALF) - cxp; + const real_t dyp = cyp - (static_cast(gpy) + HALF); // +y up + const real_t rho = math::sqrt(dxp * dxp + dyp * dyp) / Rpx; + if (rho > ONE) { + return; // outside the inscribed dome circle -> transparent border + } + const real_t phi = math::atan2(dyp, dxp); + const real_t theta = rho * dome.theta_max; // dome zenith angle + // normalized world radius fr = r / R for the chosen plane<->dome law + real_t fr; + if (dome.law == out::DomeMap::Gnomonic) { + const real_t tm = math::tan(dome.theta_max); + fr = (tm > ZERO) ? (math::tan(theta) / tm) : rho; + } else if (dome.law == out::DomeMap::Stereographic) { + const real_t tm = math::tan(HALF * dome.theta_max); + fr = (tm > ZERO) ? (math::tan(HALF * theta) / tm) : rho; + } else if (dome.law == out::DomeMap::Orthographic) { + const real_t sm = math::sin(dome.theta_max); + fr = (sm > ZERO) ? (math::sin(theta) / sm) : rho; + } else { // Equidistant (fulldome standard): r = R * theta/theta_max + fr = rho; + } + const real_t r = dome.R * fr; + u = dome.cx + r * math::cos(phi); + v = dome.cy + r * math::sin(phi); + } else { + u = umin + (static_cast(gpx) + HALF) / + static_cast(W) * (umax - umin); + v = vmax - (static_cast(gpy) + HALF) / + static_cast(H) * (vmax - vmin); + } // world -> continuous local code coords, with an optional physical // render-region clip (so a crop hides domain data outside the region, not - // just reframes the view) + // just reframes the view). The dome's own circular cutout replaces the + // rectangular clip, so it is skipped in dome mode. real_t cc1, cc2; if constexpr (M::CoordType == Coord::Cartesian) { - if (region_clip and + if (region_clip and not dome.enabled and (u < rx1lo or u > rx1hi or v < rx2lo or v > rx2hi)) { return; } From 01b526a03a4cad9fa0470bf4de76e6f737a9330e Mon Sep 17 00:00:00 2001 From: LudwigBoess Date: Mon, 27 Jul 2026 19:09:03 +0000 Subject: [PATCH 056/125] fix memory lead for intel sorting --- src/framework/containers/particles.h | 27 +++++ src/framework/containers/particles_sort.cpp | 105 +++++++++++++++----- src/global/utils/sort_dispatch.h | 7 +- 3 files changed, 114 insertions(+), 25 deletions(-) diff --git a/src/framework/containers/particles.h b/src/framework/containers/particles.h index 396b5aec8..3409b2f6f 100644 --- a/src/framework/containers/particles.h +++ b/src/framework/containers/particles.h @@ -99,6 +99,33 @@ namespace ntt { // vendor libraries detected by CMake. TileLayout m_tile_layout {}; +#if defined(TEAM_POLICY) && \ + ((defined(SYCL_ENABLED) && defined(ONEDPL_ENABLED)) || \ + (defined(CUDA_ENABLED) && defined(THRUST_ENABLED)) || \ + (defined(HIP_ENABLED) && defined(ROCTHRUST_ENABLED))) + // Persistent, grow-only scratch reused by every SortSpatially call so + // the sort makes (almost) no per-call device allocations. npart grows + // slowly and monotonically over a run, so these reallocate only a + // handful of times (grow-only, with headroom) instead of the sort + // churning ~6 transient device buffers every call. On the SYCL / + // Level-Zero USM pooling allocator (Aurora) that per-sort churn + // otherwise accumulates retained pool blocks until the device OOMs + // mid-run on a sort buffer -- see the notes in SortSpatially and + // apply_permutation_to_soa. `m_sort_keys` / `m_sort_perm` are reused + // only on the SYCL/oneDPL path, whose sort is in place; the CUDA/HIP + // double-buffer dispatch keeps using fresh transients (it may hand back + // a different buffer). The per-type gather scratch is reused on every + // vendor backend. + array_t m_sort_keys {}; + prtl_perm_t m_sort_perm {}; + array_t m_sort_scratch_int {}; + array_t m_sort_scratch_prtldx {}; + array_t m_sort_scratch_real {}; + array_t m_sort_scratch_tag {}; + array_t m_sort_scratch_pld_r {}; + array_t m_sort_scratch_pld_i {}; +#endif + public: // for empty allocation Particles() {} diff --git a/src/framework/containers/particles_sort.cpp b/src/framework/containers/particles_sort.cpp index 5e9ae4147..6795ab288 100644 --- a/src/framework/containers/particles_sort.cpp +++ b/src/framework/containers/particles_sort.cpp @@ -205,6 +205,35 @@ namespace ntt { m_is_sorted = true; } + namespace { + // Grow-only reserve for a persistent sort-scratch View: (re)allocate to + // `n` (plus 1/8 headroom) only when the current capacity is smaller, so a + // slowly growing npart triggers a handful of reallocations over a run + // instead of one device alloc/free per sort. The old buffer is released + // before the new one is allocated, keeping the transient peak at a single + // buffer during the (rare) grow. Callers always operate on the leading + // `[0, n)` prefix, so the extra headroom slots are never read. + template + inline void reserve_scratch_1d(V& v, const char* label, npart_t n) { + if (static_cast(v.extent(0)) < n) { + v = V {}; + v = V { label, n + n / 8u }; + } + } + + template + inline void reserve_scratch_2d(V& v, + const char* label, + npart_t n, + npart_t ncols) { + if (static_cast(v.extent(0)) < n or + static_cast(v.extent(1)) != ncols) { + v = V {}; + v = V { label, n + n / 8u, ncols }; + } + } + } // namespace + #if defined(TEAM_POLICY) template void Particles::compute_tile_offsets( @@ -296,7 +325,18 @@ namespace ntt { } // 2. Compute per-particle tile key (with min(i, i_prev)). +#if defined(TEAM_POLICY_USE_VENDOR_SORT) && \ + defined(SYCL_ENABLED) && defined(ONEDPL_ENABLED) + // oneDPL sorts the keys in place, so reuse a persistent, grow-only keys + // buffer instead of allocating a fresh one every sort. `tile_indices` + // aliases the (possibly over-capacity) persistent buffer; downstream + // consumers bound their work by the explicit `npart_local`, never by + // `tile_indices.extent(0)`. + reserve_scratch_1d(m_sort_keys, "tile_indices", npart_local); + array_t tile_indices = m_sort_keys; +#else array_t tile_indices { "tile_indices", npart_local }; +#endif Kokkos::parallel_for( "FillTileIndices", rangeActiveParticles(), @@ -321,31 +361,43 @@ namespace ntt { #if defined(TEAM_POLICY_USE_VENDOR_SORT) // Vendor path: produce an explicit permutation via sort_by_key, then // apply it to each SoA member by gathering the alive prefix through a - // reusable scratch buffer (one per member type, sized to the alive - // count, copied back in place). The *_prev arrays are skipped — see - // apply_permutation_to_soa. Peak transient = one - // `npart_partitioned × sizeof(member)` scratch at a time. - prtl_perm_t perm { "tile_perm", npart_local }; + // reusable scratch buffer (one per member type, copied back in place). + // The *_prev arrays are skipped — see apply_permutation_to_soa. On the + // SYCL/oneDPL path the keys, perm, and per-type gather scratch are all + // persistent, grow-only buffers (class members) reused across sorts, so + // steady state makes no per-sort device allocation; the fixed cost is a + // handful of `npart_partitioned`-sized buffers held resident. #if defined(SYCL_ENABLED) && defined(ONEDPL_ENABLED) + // Persistent, reused perm: oneDPL's sort_by_key is in place and never + // reassigns `perm`, so it can share a grow-only buffer across sorts (no + // per-sort device allocation). `npart_local` is passed explicitly since + // `tile_indices` is now an over-capacity persistent buffer whose extent + // is the reserved capacity, not the count to sort. + reserve_scratch_1d(m_sort_perm, "tile_perm", npart_local); + prtl_perm_t perm = m_sort_perm; sort_helpers::sort_by_key_dispatch(tile_indices, perm, n_bins, + npart_local, sort::backend::OneDPL {}); #elif defined(HIP_ENABLED) && defined(ROCTHRUST_ENABLED) + prtl_perm_t perm { "tile_perm", npart_local }; sort_helpers::sort_by_key_dispatch(tile_indices, perm, n_bins, sort::backend::Rocthrust {}); #else + prtl_perm_t perm { "tile_perm", npart_local }; sort_helpers::sort_by_key_dispatch(tile_indices, perm, n_bins, sort::backend::Thrust {}); #endif // `tile_indices` is sorted in place by sort_by_key. Build the tile - // offsets from it now, then drop it before the gather allocates its - // `maxnpart`-sized buffers — so the keys are not co-resident with - // them at the gather's peak (#2). + // offsets from it now, then release this handle. On the CUDA/HIP vendor + // path this frees the transient keys buffer before the gather; on SYCL + // `tile_indices` aliases the persistent `m_sort_keys`, so this only + // drops the local handle (the buffer is retained and reused next sort). compute_tile_offsets(tile_indices, total_tiles, npart_local); tile_indices = array_t {}; Kokkos::fence("SortSpatially: pre-gather drain"); @@ -589,14 +641,16 @@ namespace ntt { // Permuting prev would therefore reorder data that is overwritten // before it is ever observed. // - // Each block below allocates a single `n`-sized scratch buffer that - // is reused for every member of that type, then freed before the next - // block allocates its own. The whole gather thus makes one transient - // allocation per type group (int / prtldx_t / real_t / short / each - // payload) instead of one fresh maxnpart buffer per member, and peak - // transient is a single n-sized scratch at a time. + // Each block below reuses a single persistent, grow-only scratch buffer + // (one per member type) for every member of that type. `reserve_scratch_*` + // (re)allocates only when the live count outgrows capacity, so the gather + // makes no per-sort device allocation once warmed up (grow-only, with + // headroom) -- instead of one fresh buffer per type group every sort. + // Reused across members within a group; the fences in `permute_1d_into` + // keep the shared buffer from being clobbered before its copy-back drains. { - array_t scratch { "perm_scratch_int", n }; + reserve_scratch_1d(m_sort_scratch_int, "perm_scratch_int", n); + auto& scratch = m_sort_scratch_int; if constexpr (D == Dim::_1D or D == Dim::_2D or D == Dim::_3D) { permute_1d_into(i1, scratch, perm, n); } @@ -608,7 +662,8 @@ namespace ntt { } } { - array_t scratch { "perm_scratch_prtldx", n }; + reserve_scratch_1d(m_sort_scratch_prtldx, "perm_scratch_prtldx", n); + auto& scratch = m_sort_scratch_prtldx; if constexpr (D == Dim::_1D or D == Dim::_2D or D == Dim::_3D) { permute_1d_into(dx1, scratch, perm, n); } @@ -620,7 +675,8 @@ namespace ntt { } } { - array_t scratch { "perm_scratch_real", n }; + reserve_scratch_1d(m_sort_scratch_real, "perm_scratch_real", n); + auto& scratch = m_sort_scratch_real; permute_1d_into(ux1, scratch, perm, n); permute_1d_into(ux2, scratch, perm, n); permute_1d_into(ux3, scratch, perm, n); @@ -630,18 +686,19 @@ namespace ntt { } } { - array_t scratch { "perm_scratch_tag", n }; + reserve_scratch_1d(m_sort_scratch_tag, "perm_scratch_tag", n); + auto& scratch = m_sort_scratch_tag; permute_1d_into(tag, scratch, perm, n); } if (npld_r() > 0) { - const auto ncols = static_cast(npld_r()); - array_t scratch { "perm_scratch_pld_r", n, ncols }; - permute_2d_into(pld_r, scratch, perm, n, ncols); + const auto ncols = static_cast(npld_r()); + reserve_scratch_2d(m_sort_scratch_pld_r, "perm_scratch_pld_r", n, ncols); + permute_2d_into(pld_r, m_sort_scratch_pld_r, perm, n, ncols); } if (npld_i() > 0) { - const auto ncols = static_cast(npld_i()); - array_t scratch { "perm_scratch_pld_i", n, ncols }; - permute_2d_into(pld_i, scratch, perm, n, ncols); + const auto ncols = static_cast(npld_i()); + reserve_scratch_2d(m_sort_scratch_pld_i, "perm_scratch_pld_i", n, ncols); + permute_2d_into(pld_i, m_sort_scratch_pld_i, perm, n, ncols); } } #endif // TEAM_POLICY_USE_VENDOR_SORT diff --git a/src/global/utils/sort_dispatch.h b/src/global/utils/sort_dispatch.h index 432b1c1b9..075267539 100644 --- a/src/global/utils/sort_dispatch.h +++ b/src/global/utils/sort_dispatch.h @@ -115,11 +115,16 @@ namespace ntt::sort_helpers { } #if defined(SYCL_ENABLED) && defined(ONEDPL_ENABLED) + // `n` (the alive/local count) is passed explicitly rather than derived from + // `keys.extent(0)`, because the caller backs `keys` with a persistent, + // over-capacity buffer whose extent is the reserved capacity, not the count + // to sort. oneDPL's sort_by_key is in place, so no output/scratch buffers + // are allocated here and `perm` may safely be a reused persistent buffer. inline void sort_by_key_dispatch(const array_t& keys, prtl_perm_t& perm, ncells_t /*n_bins*/, + npart_t n, ::sort::backend::OneDPL) { - const auto n = static_cast(keys.extent(0)); if (n == 0u) { return; } From c1e851a09685ed579e3b7b30b4790c546c95785b Mon Sep 17 00:00:00 2001 From: LudwigBoess Date: Tue, 28 Jul 2026 15:58:26 +0000 Subject: [PATCH 057/125] 3D dome rendering --- input.example.toml | 33 +- src/framework/domain/metadomain_render.cpp | 87 ++- src/output/render/composite.h | 330 +++++++++++ src/output/render/raymarch.hpp | 48 +- src/output/render/renderer.cpp | 605 ++++++++++++++------- src/output/render/renderer.h | 52 ++ 6 files changed, 939 insertions(+), 216 deletions(-) diff --git a/input.example.toml b/input.example.toml index 7c7e8147b..8ccf454d2 100644 --- a/input.example.toml +++ b/input.example.toml @@ -720,6 +720,11 @@ # @type: int [> 0] # @default: 1024 height = "" + # Convenience: force a square frame (sets width == height == resolution), the + # natural shape for a dome master. Overrides `width`/`height` when > 0. + # @type: int [> 0] + # @default: 0 (use width/height) + resolution = "" # Axis-aligned render region [lo, hi] in physical/world coords, per axis # (x1/x2/x3). Any axis left unset spans the full domain. Clamped to the box. # @type: array [size 2] @@ -826,6 +831,18 @@ # -- the production setup for which the structured composite is provably # seamless. [output.render.camera] + # Projection mode. Overrides `orthographic` below when set. + # @type: string + # @enum: "orthographic", "perspective", "dome" + # @default: (unset -> use `orthographic`) + # @note: "dome" is a fulldome azimuthal-equidistant fisheye rendered from + # an INTERIOR eye (the domain center by default), i.e. a 3D + # planetarium dome master. It uses a depth-resolved (A-buffer) + # composite that is seamless across a full 3D domain decomposition + # (unlike ortho/perspective, which need the eye outside the box). + # Set a square frame (`resolution`, or width == height). `forward` + # is the dome ZENITH (screen-up defaults to +y for a +z zenith). + mode = "" # Orthographic (true) or perspective (false) projection # @type: bool # @default: true @@ -835,20 +852,26 @@ orthographic = "" # Camera (eye) position in world (physical) coordinates # @type: array [size 3] - # @default: box center pushed back ~1.7 box-diagonals along (1, 1, 1) + # @default: box center pushed back ~1.7 box-diagonals along (1, 1, 1); + # for `mode = "dome"`, the domain center (interior eye) position = "" - # Point the camera looks at, in world coordinates + # Point the camera looks at, in world coordinates (the dome ZENITH target) # @type: array [size 3] - # @default: box center + # @default: box center; for `mode = "dome"`, the zenith defaults to +z look_at = "" - # Camera up vector + # Camera up vector (dome: the disk's screen-up) # @type: array [size 3] - # @default: [0.0, 0.0, 1.0] + # @default: [0.0, 0.0, 1.0]; for `mode = "dome"`, [0.0, 1.0, 0.0] up = "" # Vertical field of view in degrees (perspective only) # @type: float [> 0.0] # @default: 35.0 fov = "" + # Full dome field of view in degrees (dome mode only): the image rim is at + # dome_fov/2 from the zenith (180 = a full hemisphere down to the horizon). + # @type: float [> 0.0, <= 360.0] + # @default: 180.0 + dome_fov = "" # Vertical extent of the view in world units (orthographic only) # @type: float [> 0.0] # @default: the global box diagonal (the whole box fits from any angle) diff --git a/src/framework/domain/metadomain_render.cpp b/src/framework/domain/metadomain_render.cpp index 58030006a..a43adcf9e 100644 --- a/src/framework/domain/metadomain_render.cpp +++ b/src/framework/domain/metadomain_render.cpp @@ -55,6 +55,7 @@ #include #include #include +#include #include namespace ntt { @@ -667,6 +668,12 @@ namespace ntt { const int W = g_renderer.width(); const int H = g_renderer.height(); + // fulldome fisheye from an interior eye: the composite is depth-resolved + // (A-buffer) instead of the ordered SubImage tree, and the frame is a + // clean square (setDomeActive -> writeFrame drops axes/colorbar margins). + const bool is_dome = (cam.projection == out::CameraDevice::Dome); + g_renderer.setDomeActive(is_dome); + // optional axis-aligned render region (== full extent when uncropped) const real_t rlo[3] = { g_renderer.regionLo(0), g_renderer.regionLo(1), g_renderer.regionLo(2) }; @@ -737,8 +744,11 @@ namespace ntt { // screen-space bounding box of this domain's footprint (same for all // scenes); we only ray-march and composite within it. int bx0 = 0, by0 = 0, bw = 0, bh = 0; - const bool on_screen = in_region and - out::screenBBox(cam, W, H, lo, hi, bx0, by0, bw, bh); + const bool on_screen = + in_region and (is_dome ? out::screenBBoxDome(cam, W, H, lo, hi, bx0, by0, + bw, bh) + : out::screenBBox(cam, W, H, lo, hi, bx0, by0, bw, + bh)); // ---- magnetic-field-line tubes (built once, shared by every scene) --- // // Every rank coarsens + replicates the field, traces the SAME global @@ -824,15 +834,17 @@ namespace ntt { } } - out::SubImage sub; + // external camera -> sparse SubImage (ordered composite); dome (interior + // eye) -> sparse FragImage (depth-resolved A-buffer composite). Both are + // built only when this domain is on-screen, but the composite call is + // collective, so every rank reaches it (with an empty image otherwise). + out::SubImage sub; + out::FragImage frag; if (on_screen) { - sub.x0 = bx0; - sub.y0 = by0; - sub.w = bw; - sub.h = bh; const std::size_t bnpix = static_cast(bw) * static_cast(bh); array_t image { "render_img", bnpix }; + array_t depth { "render_depth", bnpix }; randacc_ndfield_t Fld { bckp }; Kokkos::parallel_for( "VolumeRayMarch", @@ -867,22 +879,63 @@ namespace ntt { spine_rgb, kt, volume_on, - image)); + image, + depth)); Kokkos::fence(); - // device -> host, into a layout-agnostic pixel-major buffer + // device -> host, into layout-agnostic pixel-major buffers auto image_h = Kokkos::create_mirror_view(image); Kokkos::deep_copy(image_h, image); - sub.rgba.resize(bnpix * 4); - for (std::size_t p = 0; p < bnpix; ++p) { - sub.rgba[p * 4 + 0] = image_h(p, 0); - sub.rgba[p * 4 + 1] = image_h(p, 1); - sub.rgba[p * 4 + 2] = image_h(p, 2); - sub.rgba[p * 4 + 3] = image_h(p, 3); + if (is_dome) { + auto depth_h = Kokkos::create_mirror_view(depth); + Kokkos::deep_copy(depth_h, depth); + // leaf fragment image: one depth-tagged fragment per covered pixel + frag.x0 = bx0; + frag.y0 = by0; + frag.w = bw; + frag.h = bh; + frag.offs.assign(bnpix + 1, 0u); + for (std::size_t p = 0; p < bnpix; ++p) { + frag.offs[p + 1] = (image_h(p, 3) > ZERO) ? 1u : 0u; + } + for (std::size_t p = 0; p < bnpix; ++p) { + frag.offs[p + 1] += frag.offs[p]; + } + const std::size_t nfrag = frag.offs[bnpix]; + frag.depth.resize(nfrag); + frag.rgba.resize(nfrag * 4); + std::size_t o = 0; + for (std::size_t p = 0; p < bnpix; ++p) { + if (image_h(p, 3) > ZERO) { + frag.depth[o] = depth_h(p); + frag.rgba[o * 4 + 0] = image_h(p, 0); + frag.rgba[o * 4 + 1] = image_h(p, 1); + frag.rgba[o * 4 + 2] = image_h(p, 2); + frag.rgba[o * 4 + 3] = image_h(p, 3); + ++o; + } + } + } else { + sub.x0 = bx0; + sub.y0 = by0; + sub.w = bw; + sub.h = bh; + sub.rgba.resize(bnpix * 4); + for (std::size_t p = 0; p < bnpix; ++p) { + sub.rgba[p * 4 + 0] = image_h(p, 0); + sub.rgba[p * 4 + 1] = image_h(p, 1); + sub.rgba[p * 4 + 2] = image_h(p, 2); + sub.rgba[p * 4 + 3] = image_h(p, 3); + } } } - g_renderer.compositeAndWrite(sub, order_key, scene_cb, current_step, - current_time); + if (is_dome) { + g_renderer.compositeFragAndWrite(std::move(frag), scene_cb, + current_step, current_time); + } else { + g_renderer.compositeAndWrite(sub, order_key, scene_cb, current_step, + current_time); + } rendered_any = true; } return rendered_any; diff --git a/src/output/render/composite.h b/src/output/render/composite.h index 47bcdcc15..ecb14ced9 100644 --- a/src/output/render/composite.h +++ b/src/output/render/composite.h @@ -215,6 +215,336 @@ namespace out { return r; } + /* ====================================================================== */ + /* Interior-eye dome: fisheye projection + depth-resolved (A-buffer) */ + /* composite. Used when a single global front-to-back order does not */ + /* exist (camera inside the box). See renderer.h::FragImage. */ + /* ====================================================================== */ + + /** + * @brief Forward azimuthal-equidistant fisheye projection: world point -> the + * (fractional) dome pixel, inverting the dome ray generation. + * @return false if the point is outside the dome field of view. + * @note Uses the same ndc<->pixel convention as projectToScreen (so a square + * frame gives a centered disk of radius 1); the dome kernel forces aspect 1. + */ + inline auto projectToScreenDome(const CameraDevice& cam, + int W, + int H, + const real_t p[3], + real_t& outx, + real_t& outy) -> bool { + real_t v[3] = { p[0] - cam.eye[0], p[1] - cam.eye[1], p[2] - cam.eye[2] }; + const real_t n = std::sqrt(v[0] * v[0] + v[1] * v[1] + v[2] * v[2]); + if (n < static_cast(1e-20)) { + outx = HALF * static_cast(W) - HALF; // eye itself -> disk center + outy = HALF * static_cast(H) - HALF; + return true; + } + const real_t inv = ONE / n; + v[0] *= inv; + v[1] *= inv; + v[2] *= inv; + real_t cz = v[0] * cam.forward[0] + v[1] * cam.forward[1] + + v[2] * cam.forward[2]; + cz = (cz < -ONE) ? -ONE : ((cz > ONE) ? ONE : cz); + const real_t theta = std::acos(cz); + if (theta > cam.dome_half_fov) { + return false; // outside the dome FOV + } + const real_t r = theta / cam.dome_half_fov; // 0..1 image radius + const real_t cx = v[0] * cam.right[0] + v[1] * cam.right[1] + + v[2] * cam.right[2]; + const real_t cy = v[0] * cam.up[0] + v[1] * cam.up[1] + v[2] * cam.up[2]; + const real_t phi = std::atan2(cy, cx); + const real_t fx = r * std::cos(phi), fy = r * std::sin(phi); + outx = (fx + ONE) * HALF * static_cast(W) - HALF; + outy = (ONE - fy) * HALF * static_cast(H) - HALF; + return true; + } + + /** + * @brief Conservative screen-space bbox of a world AABB under the dome fisheye. + * @return false (empty) if the box is entirely outside the dome FOV. + * @note Never under-covers (that would drop fragments and reintroduce seams): + * - eye inside the (inclusive) AABB -> full frame + * - all edge samples outside the FOV -> empty + * - some in / some out (straddles the horizon)-> full frame + * - footprint contains the zenith or wraps the disk center (max angular gap + * between projected samples < pi) -> full frame + * - otherwise -> tight bbox of the samples + */ + inline auto screenBBoxDome(const CameraDevice& cam, + int W, + int H, + const real_t lo[3], + const real_t hi[3], + int& bx0, + int& by0, + int& bw, + int& bh) -> bool { + auto fullFrame = [&]() { + bx0 = 0; + by0 = 0; + bw = W; + bh = H; + return true; + }; + // eye inside the domain -> covers all azimuths + the zenith -> full frame + if (cam.eye[0] >= lo[0] and cam.eye[0] <= hi[0] and cam.eye[1] >= lo[1] and + cam.eye[1] <= hi[1] and cam.eye[2] >= lo[2] and cam.eye[2] <= hi[2]) { + return fullFrame(); + } + // the zenith ray (disk center) piercing this domain also means it covers the + // center -> full frame. A ray/AABB slab test from the eye along `forward` + // catches the case edge-sampling can miss (a slab pierced through a face, + // where the minr / azimuth-gap tests stay just under threshold -> a hole at + // the frame center). Only the forward half-line (t >= 0) is considered. + { + const real_t reps = static_cast(1e-12); + real_t te = ZERO, tx = static_cast(1e30); + bool hit = true; + for (int d = 0; d < 3; ++d) { + const real_t o = cam.eye[d], dd = cam.forward[d]; + if (dd > -reps and dd < reps) { + if (o < lo[d] or o > hi[d]) { + hit = false; + break; + } + } else { + real_t t1 = (lo[d] - o) / dd, t2 = (hi[d] - o) / dd; + if (t1 > t2) { + const real_t tmp = t1; + t1 = t2; + t2 = tmp; + } + te = (t1 > te) ? t1 : te; + tx = (t2 < tx) ? t2 : tx; + } + } + if (hit and te <= tx and tx >= ZERO) { + return fullFrame(); + } + } + real_t minx = static_cast(1e30), miny = static_cast(1e30); + real_t maxx = static_cast(-1e30), maxy = static_cast(-1e30); + real_t minr = static_cast(1e30); + int n_in = 0, n_out = 0; + // azimuths of in-FOV samples, for the "wraps the center" (largest-gap) test + std::vector phis; + phis.reserve(12 * 17); + const int NS = 48; // samples per AABB edge (a straight edge maps to a + // curved fisheye arc, so sample densely to bound it) + auto addPoint = [&](const real_t p[3]) { + real_t sx, sy; + if (not projectToScreenDome(cam, W, H, p, sx, sy)) { + ++n_out; + return; + } + ++n_in; + minx = std::min(minx, sx); + maxx = std::max(maxx, sx); + miny = std::min(miny, sy); + maxy = std::max(maxy, sy); + const real_t fx = TWO * (sx + HALF) / static_cast(W) - ONE; + const real_t fy = ONE - TWO * (sy + HALF) / static_cast(H); + minr = std::min(minr, std::sqrt(fx * fx + fy * fy)); + phis.push_back(std::atan2(fy, fx)); + }; + // sample all 12 edges of the AABB + for (int axis = 0; axis < 3; ++axis) { + for (int c = 0; c < 4; ++c) { + // the two AABB axes perpendicular to `axis` are fixed to lo/hi per `c` + const int a1 = (axis + 1) % 3, a2 = (axis + 2) % 3; + real_t p[3]; + p[a1] = (c & 1) ? hi[a1] : lo[a1]; + p[a2] = (c & 2) ? hi[a2] : lo[a2]; + for (int s = 0; s <= NS; ++s) { + const real_t t = static_cast(s) / static_cast(NS); + p[axis] = lo[axis] + (hi[axis] - lo[axis]) * t; + addPoint(p); + } + } + } + if (n_in == 0) { + bw = 0; + bh = 0; + return false; // entirely outside the FOV + } + if (n_out > 0) { + return fullFrame(); // straddles the FOV boundary -> conservative + } + if (minr < static_cast(1e-3)) { + return fullFrame(); // footprint reaches the zenith (disk center) + } + // largest cyclic gap between azimuths: if < pi the samples wrap the center, + // so the axis-aligned bbox of the boundary would miss the interior. + std::sort(phis.begin(), phis.end()); + real_t maxgap = ZERO; + const real_t twopi = static_cast(2.0 * 3.14159265358979323846); + for (std::size_t i = 0; i + 1 < phis.size(); ++i) { + maxgap = std::max(maxgap, phis[i + 1] - phis[i]); + } + if (not phis.empty()) { + maxgap = std::max(maxgap, (phis.front() + twopi) - phis.back()); + } + // a small tolerance past pi keeps borderline wraps conservative + if (maxgap < static_cast(3.14159265358979323846 + 0.05)) { + return fullFrame(); + } + // pad generously: the fisheye arc between edge samples can bulge a few px + const int pad = 4; + int x0 = static_cast(std::floor(minx)) - pad; + int x1 = static_cast(std::ceil(maxx)) + pad; + int y0 = static_cast(std::floor(miny)) - pad; + int y1 = static_cast(std::ceil(maxy)) + pad; + x0 = std::max(0, std::min(W, x0)); + x1 = std::max(0, std::min(W, x1)); + y0 = std::max(0, std::min(H, y0)); + y1 = std::max(0, std::min(H, y1)); + bx0 = x0; + by0 = y0; + bw = x1 - x0; + bh = y1 - y0; + return (bw > 0 and bh > 0); + } + + /** + * @brief Collapse one pixel's depth-sorted fragment list [k0, k1) into a + * single premultiplied RGBA via front-to-back "over". + */ + inline void fragOver(const std::vector& depth, + const std::vector& rgba, + uint32_t k0, + uint32_t k1, + real_t out[4]) { + (void)depth; // fragments are already ascending in depth + real_t acc[4] = { ZERO, ZERO, ZERO, ZERO }; + for (uint32_t k = k0; k < k1; ++k) { + overComposite(acc, &rgba[static_cast(k) * 4]); + if (acc[3] >= ONE) { + break; + } + } + out[0] = acc[0]; + out[1] = acc[1]; + out[2] = acc[2]; + out[3] = acc[3]; + } + + /** + * @brief Merge two depth-sorted fragment images: union the bboxes and, per + * pixel, merge the two ascending fragment lists by depth, then drop fragments + * once the accumulated alpha reaches `cull_alpha` (exact when cull_alpha == 1: + * only provably-occluded fragments are removed, so the result is independent + * of how the tree is grouped -> associative + commutative). + */ + inline auto mergeFrag(const FragImage& a, const FragImage& b, real_t cull_alpha) + -> FragImage { + if (a.w == 0 or a.h == 0) { + return b; + } + if (b.w == 0 or b.h == 0) { + return a; + } + const int ux0 = std::min(a.x0, b.x0); + const int uy0 = std::min(a.y0, b.y0); + const int ux1 = std::max(a.x0 + a.w, b.x0 + b.w); + const int uy1 = std::max(a.y0 + a.h, b.y0 + b.h); + FragImage r; + r.x0 = ux0; + r.y0 = uy0; + r.w = ux1 - ux0; + r.h = uy1 - uy0; + const std::size_t np = static_cast(r.w) * r.h; + r.offs.assign(np + 1, 0u); + + // fetch a source image's fragment range at global pixel (gx, gy) + auto range = [](const FragImage& s, int gx, int gy, uint32_t& k0, uint32_t& k1) { + const int lx = gx - s.x0, ly = gy - s.y0; + if (lx < 0 or ly < 0 or lx >= s.w or ly >= s.h) { + k0 = 0; + k1 = 0; + return; + } + const std::size_t p = static_cast(ly) * s.w + lx; + k0 = s.offs[p]; + k1 = s.offs[p + 1]; + }; + + // pass 1: per-pixel surviving-fragment count (merge + occlusion cull) + for (int gy = uy0; gy < uy1; ++gy) { + for (int gx = ux0; gx < ux1; ++gx) { + uint32_t ak0, ak1, bk0, bk1; + range(a, gx, gy, ak0, ak1); + range(b, gx, gy, bk0, bk1); + uint32_t ia = ak0, ib = bk0, cnt = 0; + real_t A = ZERO; + while ((ia < ak1 or ib < bk1) and A < cull_alpha) { + const bool takeA = (ib >= bk1) or + (ia < ak1 and a.depth[ia] <= b.depth[ib]); + const real_t al = takeA ? a.rgba[static_cast(ia) * 4 + 3] + : b.rgba[static_cast(ib) * 4 + 3]; + A += (ONE - A) * al; + ++cnt; + if (takeA) { + ++ia; + } else { + ++ib; + } + } + const std::size_t pix = static_cast(gy - uy0) * r.w + + (gx - ux0); + r.offs[pix + 1] = cnt; + } + } + // prefix-sum to offsets + for (std::size_t p = 0; p < np; ++p) { + r.offs[p + 1] += r.offs[p]; + } + const std::size_t nfrag = r.offs[np]; + r.depth.assign(nfrag, ZERO); + r.rgba.assign(nfrag * 4, ZERO); + + // pass 2: fill the merged, culled fragments + for (int gy = uy0; gy < uy1; ++gy) { + for (int gx = ux0; gx < ux1; ++gx) { + uint32_t ak0, ak1, bk0, bk1; + range(a, gx, gy, ak0, ak1); + range(b, gx, gy, bk0, bk1); + const std::size_t pix = static_cast(gy - uy0) * r.w + + (gx - ux0); + uint32_t ia = ak0, ib = bk0, o = r.offs[pix]; + const uint32_t oend = r.offs[pix + 1]; + while (o < oend) { + const bool takeA = (ib >= bk1) or + (ia < ak1 and a.depth[ia] <= b.depth[ib]); + if (takeA) { + r.depth[o] = a.depth[ia]; + const std::size_t s = static_cast(ia) * 4; + const std::size_t d = static_cast(o) * 4; + r.rgba[d + 0] = a.rgba[s + 0]; + r.rgba[d + 1] = a.rgba[s + 1]; + r.rgba[d + 2] = a.rgba[s + 2]; + r.rgba[d + 3] = a.rgba[s + 3]; + ++ia; + } else { + r.depth[o] = b.depth[ib]; + const std::size_t s = static_cast(ib) * 4; + const std::size_t d = static_cast(o) * 4; + r.rgba[d + 0] = b.rgba[s + 0]; + r.rgba[d + 1] = b.rgba[s + 1]; + r.rgba[d + 2] = b.rgba[s + 2]; + r.rgba[d + 3] = b.rgba[s + 3]; + ++ib; + } + ++o; + } + } + } + return r; + } + } // namespace out #endif // OUTPUT_RENDER_COMPOSITE_H diff --git a/src/output/render/raymarch.hpp b/src/output/render/raymarch.hpp index a326d3f31..4b4daebfb 100644 --- a/src/output/render/raymarch.hpp +++ b/src/output/render/raymarch.hpp @@ -86,6 +86,10 @@ namespace kernel { const bool volume_enabled; array_t image; // output, (bw*bh, 4) premultiplied RGBA + // per-pixel front depth (ray t at domain entry) for the interior-eye dome + // A-buffer composite; INF where the pixel produced no fragment. Written for + // every projection (the external composite simply ignores it). + array_t depth_img; public: VolumeRayMarch_kernel(const randacc_ndfield_t& Fld_, @@ -116,7 +120,8 @@ namespace kernel { const real_t spine_rgb[3], const out::TubeSet& tubes_, bool volume_enabled_, - const array_t& image_) + const array_t& image_, + const array_t& depth_img_) : Fld { Fld_ } , comp { comp_ } , metric { metric_ } @@ -173,7 +178,8 @@ namespace kernel { , tube_vmax { tubes_.vmax } , tube_log { tubes_.log_scale } , volume_enabled { volume_enabled_ } - , image { image_ } {} + , image { image_ } + , depth_img { depth_img_ } {} // distance test: is world point (px,py,pz) within `spine_radius` of any of // the 12 global-box edges? (nearest parallel edge per axis == nearest @@ -298,11 +304,13 @@ namespace kernel { static_cast(lpx); const int gpx = bx0 + static_cast(lpx); const int gpy = by0 + static_cast(lpy); - // default transparent - image(pix, 0) = ZERO; - image(pix, 1) = ZERO; - image(pix, 2) = ZERO; - image(pix, 3) = ZERO; + const real_t INF = static_cast(1e30); + // default transparent + no fragment + image(pix, 0) = ZERO; + image(pix, 1) = ZERO; + image(pix, 2) = ZERO; + image(pix, 3) = ZERO; + depth_img(pix) = INF; // ---- ray generation ------------------------------------------------ // const real_t fx = TWO * (static_cast(gpx) + HALF) / @@ -310,7 +318,25 @@ namespace kernel { const real_t fy = ONE - TWO * (static_cast(gpy) + HALF) / static_cast(H); real_t ox, oy, oz, dx, dy, dz; - if (cam.orthographic) { + if (cam.projection == out::CameraDevice::Dome) { + // fulldome azimuthal-equidistant fisheye from an interior eye: image + // radius rho in [0,1] -> zenith angle theta = rho * dome_half_fov, about + // the `forward` (zenith) axis; corners (rho > 1) are transparent. + const real_t rho = math::sqrt(fx * fx + fy * fy); + if (rho > ONE) { + return; // outside the dome disk + } + const real_t theta = rho * cam.dome_half_fov; + const real_t phi = math::atan2(fy, fx); + const real_t st = math::sin(theta), ct = math::cos(theta); + const real_t cp = math::cos(phi), sp = math::sin(phi); + dx = ct * cam.forward[0] + st * (cp * cam.right[0] + sp * cam.up[0]); + dy = ct * cam.forward[1] + st * (cp * cam.right[1] + sp * cam.up[1]); + dz = ct * cam.forward[2] + st * (cp * cam.right[2] + sp * cam.up[2]); + ox = cam.eye[0]; + oy = cam.eye[1]; + oz = cam.eye[2]; + } else if (cam.orthographic) { const real_t sx = fx * cam.half_w; const real_t sy = fy * cam.half_h; ox = cam.eye[0] + sx * cam.right[0] + sy * cam.up[0]; @@ -486,6 +512,12 @@ namespace kernel { image(pix, 1) = acc_g; image(pix, 2) = acc_b; image(pix, 3) = acc_a; + // record the domain's front depth (geometric slab entry) for a produced + // fragment; the A-buffer composite orders domains by this. Domain slabs + // are disjoint contiguous ray intervals, so t_enter is a correct key. + if (acc_a > ZERO) { + depth_img(pix) = t_enter; + } } }; diff --git a/src/output/render/renderer.cpp b/src/output/render/renderer.cpp index 1f1a8feaf..e007e8b78 100644 --- a/src/output/render/renderer.cpp +++ b/src/output/render/renderer.cpp @@ -30,6 +30,7 @@ #include #include #include +#include #include namespace out { @@ -84,6 +85,14 @@ namespace out { m_width = toml::find_or(td, "output", "render", "width", 1024); m_height = toml::find_or(td, "output", "render", "height", 1024); + // `resolution` is a convenience that forces a square frame (width == height), + // the natural shape for a dome master. + const int resolution = toml::find_or(td, "output", "render", + "resolution", 0); + if (resolution > 0) { + m_width = resolution; + m_height = resolution; + } m_samples = toml::find_or(td, "output", "render", "samples", 400); m_step_size = toml::find_or(td, "output", "render", "step_size", ZERO); m_early_alpha = toml::find_or(td, @@ -281,6 +290,27 @@ namespace out { "camera", "orthographic", true); + // `mode` overrides the `orthographic` flag: "orthographic" | "perspective" | + // "dome". The dome is a fulldome azimuthal-equidistant fisheye from an + // INTERIOR eye (the box center by default) -- see Metadomain::Render (3D). + const auto cam_mode = toml::find_or( + td, "output", "render", "camera", "mode", std::string {}); + int projection = ortho ? CameraDevice::Ortho : CameraDevice::Perspective; + if (cam_mode == "dome") { + projection = CameraDevice::Dome; + } else if (cam_mode == "perspective") { + projection = CameraDevice::Perspective; + } else if (cam_mode == "orthographic") { + projection = CameraDevice::Ortho; + } else if (not cam_mode.empty()) { + raise::Warning("output.render.camera.mode '" + cam_mode + + "' unknown (want orthographic/perspective/dome); using " + "the 'orthographic' flag", + HERE); + } + const bool is_dome = (projection == CameraDevice::Dome); + const real_t dome_fov = toml::find_or( + td, "output", "render", "camera", "dome_fov", static_cast(180)); auto pos = toml::find_or>(td, "output", "render", @@ -317,20 +347,44 @@ namespace out { real_t eye[3], lookat[3], upv[3]; for (int d = 0; d < 3; ++d) { - // default eye: box center pushed back along (1,1,1) by ~1.7 diagonals - eye[d] = (pos.size() == 3) - ? pos[d] - : center[d] + static_cast(1.7) * diag * - static_cast(0.57735026919); + if (pos.size() == 3) { + eye[d] = pos[d]; + } else if (is_dome) { + // interior eye: the domain center (looking outward at the sky) + eye[d] = center[d]; + } else { + // default eye: box center pushed back along (1,1,1) by ~1.7 diagonals + eye[d] = center[d] + static_cast(1.7) * diag * + static_cast(0.57735026919); + } lookat[d] = (look.size() == 3) ? look[d] : center[d]; } - upv[0] = (up.size() == 3) ? up[0] : ZERO; - upv[1] = (up.size() == 3) ? up[1] : ZERO; - upv[2] = (up.size() == 3) ? up[2] : ONE; + // up / screen-up: default +z, except the dome (whose default zenith is +z, + // so +z would be collinear with `forward`) uses +y as the disk's screen-up. + if (up.size() == 3) { + upv[0] = up[0]; + upv[1] = up[1]; + upv[2] = up[2]; + } else if (is_dome) { + upv[0] = ZERO; + upv[1] = ONE; + upv[2] = ZERO; + } else { + upv[0] = ZERO; + upv[1] = ZERO; + upv[2] = ONE; + } real_t forward[3] = { lookat[0] - eye[0], lookat[1] - eye[1], lookat[2] - eye[2] }; + // for the dome, `forward` is the ZENITH; with the default interior eye at the + // center (== lookat) the difference is zero, so default the zenith to +z. + if (is_dome and look.size() != 3) { + forward[0] = ZERO; + forward[1] = ZERO; + forward[2] = ONE; + } normalize3(forward); real_t right[3]; cross3(forward, upv, right); @@ -349,9 +403,35 @@ namespace out { m_camera_dev.tan_half_fov = std::tan(static_cast(0.5) * fov * static_cast(constant::PI) / static_cast(180.0)); - m_camera_dev.orthographic = ortho; + // keep `orthographic` consistent with the resolved projection so that + // `mode` actually overrides the flag: the kernel/screenBBox pick ortho vs + // perspective from `orthographic`, and only Dome is read off `projection`. + // (When `mode` is unset, `projection` was derived from `ortho`, so this + // round-trips to the original flag -- back-compatible.) + m_camera_dev.orthographic = (projection == CameraDevice::Ortho); m_camera_dev.half_h = static_cast(0.5) * ortho_height; m_camera_dev.half_w = m_camera_dev.half_h * m_camera_dev.aspect; + m_camera_dev.projection = projection; + if (is_dome) { + real_t hf = HALF * dome_fov * static_cast(constant::PI) / + static_cast(180); + if (hf <= ZERO) { + hf = HALF * static_cast(constant::PI); // fall back to a 180 dome + } + if (hf > static_cast(constant::PI)) { + hf = static_cast(constant::PI); // full sphere cap + } + m_camera_dev.dome_half_fov = hf; + // a fisheye disk needs a square frame; the kernel uses the full-frame ndc + m_camera_dev.aspect = ONE; + m_camera_dev.half_w = m_camera_dev.half_h; + if (m_width != m_height) { + raise::Warning("output.render.camera.mode='dome' wants width == height " + "for a circular dome master; the fisheye disk will be " + "elliptical otherwise", + HERE); + } + } /* ---- moving view (pan the region/camera to track a feature) --------- */ // remember the static region + camera eye; updateForTime() translates them. @@ -515,6 +595,189 @@ namespace out { } } + void Renderer::writeFrame(const std::vector& img, + const Scene& scene, + timestep_t step, + simtime_t time) const { + const std::size_t npix = static_cast(m_width) * + static_cast(m_height); + const std::size_t n = npix * 4; + // ensure /renders/ exists + const auto dir = m_root / path_t("renders"); + try { + if (not std::filesystem::exists(m_root)) { + std::filesystem::create_directory(m_root); + } + if (not std::filesystem::exists(dir)) { + std::filesystem::create_directory(dir); + } + } catch (const std::exception& e) { + raise::Warning(e.what(), HERE); + } + // composite the premultiplied image over the opaque background: + // out = src_premult + (1 - src_alpha) * background, alpha = opaque. + std::vector data(n); + for (std::size_t p = 0; p < npix; ++p) { + const real_t a = img[p * 4 + 3]; + const real_t inv = ONE - a; + data[p * 4 + 0] = quantize(img[p * 4 + 0] + inv * m_background[0]); + data[p * 4 + 1] = quantize(img[p * 4 + 1] + inv * m_background[1]); + data[p * 4 + 2] = quantize(img[p * 4 + 2] + inv * m_background[2]); + data[p * 4 + 3] = 255; + } + const auto fname = dir / fmt::format("%s%08lu.png", + scene.prefix.c_str(), + static_cast(step)); + + auto drawBar = [&](uint8_t* buf, int bw, int bh, int span_top, int span_bot) { + if (m_colorbar) { + drawColorbar(buf, bw, bh, scene.tf.colormap, scene.tf.vmin, + scene.tf.vmax, scene.tf.log_scale, scene.label, + m_background, scene.ticks, span_top, span_bot); + } + }; + + // Draw the sim-time label, right-aligned to `right_x` and vertically + // centered in the band [0, top_limit] -- i.e. OUTSIDE the plotted data: + // above the colorbar (3D / disk) or, for a 2D slice, in the aspect-pad + // above the data box (so it never sits inside the simulation axes). + auto drawTimeLabel = [&](uint8_t* buf, int cw, int ch, int right_x, + int top_limit) { + if (not m_time_label) { + return; + } + const int s = cbar_hidden::scale(m_height); + char tbuf[48]; + // fixed-point so it reads e.g. "T = 12345.67" (up to 5 integer digits + // and 2 decimals; more integer digits still print, never truncated) + std::snprintf(tbuf, sizeof(tbuf), "T = %.2f", static_cast(time)); + const std::string str(tbuf); + const int tw = static_cast(str.size()) * 6 * s; + const int pad = 3 * s; + const int tx = right_x - tw - pad; + const int text_h = 7 * s; + int ty = (top_limit - text_h) / 2; + if (ty < pad) { + ty = pad; + } + // contrasting text color (white on a dark background, black on light) + const real_t lum = static_cast(0.299) * m_background[0] + + static_cast(0.587) * m_background[1] + + static_cast(0.114) * m_background[2]; + const uint8_t tc = (lum < HALF) ? 255 : 0; + cbar_hidden::drawText(buf, cw, ch, tx, ty, str, s, tc, tc, tc); + }; + + // canvas margins: axes (left + bottom) and the colorbar strip (right). + // The data region sits at (ml, 0); margins/strip are background-filled. + // The polar (curvilinear) overlay annotates inside the data region (the + // disk is centered with background around it), so it needs no margins. + const bool polar = (m_global_extent.size() == 2) and m_slice_polar; + // a fisheye dome master (2D or 3D) must stay exactly W x H (its inscribed + // circle is the dome), so it takes no axes margins and no outside colorbar. + const bool dome = m_dome_active; + int ml = 0, mb = 0; + out::axesMargins(m_axes and not polar and not dome, m_height, ml, mb); + const int strip = (m_colorbar and m_colorbar_outside and not dome) + ? colorbarBlockWidth(m_height) + : 0; + const int CW = ml + m_width + strip; + const int CH = m_height + mb; + + // 2D-Cartesian data box (== the render region, before the aspect-expansion + // that pads the window with background): its top & right edges in + // data-region pixels. The axes/spine clamp to it and the time label sits + // in the pad above it, so neither includes the empty aspect padding. + const bool cart2d = (m_global_extent.size() == 2) and not polar and + not dome; + int dbox_top = 0; // data box top edge (px from data top) + int dbox_bot = m_height; // data box bottom edge (px) + int dbox_right = m_width; // data box right edge (px from data left) + if (cart2d and m_region.size() >= 2) { + const real_t u0 = m_slice_win[0], u1 = m_slice_win[1]; + const real_t v0 = m_slice_win[2], v1 = m_slice_win[3]; + const real_t du1 = m_region[0].second; // data box right in world (x1) + const real_t dv0 = m_region[1].first; // data box bottom in world (x2) + const real_t dv1 = m_region[1].second; // data box top in world (x2) + if (u1 > u0) { + int r = static_cast(std::lround( + static_cast((du1 - u0) / (u1 - u0)) * (m_width - 1))); + dbox_right = (r < 0) ? 0 : ((r > m_width) ? m_width : r); + } + if (v1 > v0) { + int t = static_cast(std::lround( + static_cast((v1 - dv1) / (v1 - v0)) * (m_height - 1))); + int b = static_cast(std::lround( + static_cast((v1 - dv0) / (v1 - v0)) * (m_height - 1))); + dbox_top = (t < 0) ? 0 : ((t > m_height) ? m_height : t); + dbox_bot = (b < 0) ? 0 : ((b > m_height) ? m_height : b); + } + } + // time-label anchor: for a 2D slice, the top-right of the data box (label + // goes in the pad above it); otherwise the top-right above the colorbar. + const int cbar_top = m_colorbar ? (CH - CH / 2) / 2 : (CH / 4); + const int tl_right = cart2d ? (ml + dbox_right) : (ml + m_width); + const int tl_top = cart2d ? dbox_top : cbar_top; + // colorbar vertical span: aligned to the actual data domain for a 2D slice + // (so it's centered on the data, not the aspect-padded canvas); sentinel + // (-1) elsewhere -> drawColorbar centers it on the canvas as before. + const int cbar_span_top = cart2d ? dbox_top : -1; + const int cbar_span_bot = cart2d ? dbox_bot : -1; + + bool ok = true; + if (CW == m_width and CH == m_height and not m_axes) { + // no margins, no outside strip, no overlay: colorbar overlays the data + drawBar(data.data(), m_width, m_height, cbar_span_top, cbar_span_bot); + drawTimeLabel(data.data(), m_width, m_height, tl_right, tl_top); + ok = write_png(fname, m_width, m_height, data.data()); + } else { + const uint8_t bR = quantize(m_background[0]); + const uint8_t bG = quantize(m_background[1]); + const uint8_t bB = quantize(m_background[2]); + std::vector canvas(static_cast(CW) * CH * 4); + for (std::size_t i = 0; i < canvas.size(); i += 4) { + canvas[i + 0] = bR; + canvas[i + 1] = bG; + canvas[i + 2] = bB; + canvas[i + 3] = 255; + } + for (int y = 0; y < m_height; ++y) { + std::copy_n(&data[static_cast(y) * m_width * 4], + static_cast(m_width) * 4, + &canvas[(static_cast(y) * CW + ml) * 4]); + } + if (m_axes and not dome) { + if (m_global_extent.size() == 3) { + out::drawAxes3D(canvas.data(), CW, CH, ml, m_width, m_height, + m_camera_dev, m_region, m_axis_labels, m_background, + m_axis_nticks); + } else if (polar) { + out::drawAxesPolar(canvas.data(), CW, CH, ml, m_width, m_height, + m_slice_win[0], m_slice_win[1], m_slice_win[2], + m_slice_win[3], m_slice_rmin, m_slice_rmax, + m_slice_tmin, m_slice_tmax, m_slice_pmirror, "R", + "Theta", m_background, m_axis_nticks); + } else { + // data box (== region, un-expanded) so the spine hugs the domain, + // not the aspect-padded window + const real_t du0 = m_region[0].first, du1 = m_region[0].second; + const real_t dv0 = m_region[1].first, dv1 = m_region[1].second; + out::drawAxes2D(canvas.data(), CW, CH, ml, m_width, m_height, + m_slice_win[0], m_slice_win[1], m_slice_win[2], + m_slice_win[3], du0, du1, dv0, dv1, m_slice_xlabel, + m_slice_ylabel, m_background, m_axis_nticks); + } + } + drawBar(canvas.data(), CW, CH, cbar_span_top, cbar_span_bot); + drawTimeLabel(canvas.data(), CW, CH, tl_right, tl_top); + ok = write_png(fname, CW, CH, canvas.data()); + } + if (not ok) { + raise::Warning(fmt::format("failed to write %s", fname.string().c_str()), + HERE); + } + } + void Renderer::compositeAndWrite(const SubImage& sub, uint64_t order_key, const Scene& scene, @@ -546,183 +809,7 @@ namespace out { }; auto write_image = [&](const std::vector& img) { - // ensure /renders/ exists - const auto dir = m_root / path_t("renders"); - try { - if (not std::filesystem::exists(m_root)) { - std::filesystem::create_directory(m_root); - } - if (not std::filesystem::exists(dir)) { - std::filesystem::create_directory(dir); - } - } catch (const std::exception& e) { - raise::Warning(e.what(), HERE); - } - // composite the premultiplied image over the opaque background: - // out = src_premult + (1 - src_alpha) * background, alpha = opaque. - std::vector data(n); - for (std::size_t p = 0; p < npix; ++p) { - const real_t a = img[p * 4 + 3]; - const real_t inv = ONE - a; - data[p * 4 + 0] = quantize(img[p * 4 + 0] + inv * m_background[0]); - data[p * 4 + 1] = quantize(img[p * 4 + 1] + inv * m_background[1]); - data[p * 4 + 2] = quantize(img[p * 4 + 2] + inv * m_background[2]); - data[p * 4 + 3] = 255; - } - const auto fname = dir / fmt::format("%s%08lu.png", - scene.prefix.c_str(), - static_cast(step)); - - auto drawBar = [&](uint8_t* buf, int bw, int bh, int span_top, int span_bot) { - if (m_colorbar) { - drawColorbar(buf, bw, bh, scene.tf.colormap, scene.tf.vmin, - scene.tf.vmax, scene.tf.log_scale, scene.label, - m_background, scene.ticks, span_top, span_bot); - } - }; - - // Draw the sim-time label, right-aligned to `right_x` and vertically - // centered in the band [0, top_limit] -- i.e. OUTSIDE the plotted data: - // above the colorbar (3D / disk) or, for a 2D slice, in the aspect-pad - // above the data box (so it never sits inside the simulation axes). - auto drawTimeLabel = [&](uint8_t* buf, int cw, int ch, int right_x, - int top_limit) { - if (not m_time_label) { - return; - } - const int s = cbar_hidden::scale(m_height); - char tbuf[48]; - // fixed-point so it reads e.g. "T = 12345.67" (up to 5 integer digits - // and 2 decimals; more integer digits still print, never truncated) - std::snprintf(tbuf, sizeof(tbuf), "T = %.2f", - static_cast(time)); - const std::string str(tbuf); - const int tw = static_cast(str.size()) * 6 * s; - const int pad = 3 * s; - const int tx = right_x - tw - pad; - const int text_h = 7 * s; - int ty = (top_limit - text_h) / 2; - if (ty < pad) { - ty = pad; - } - // contrasting text color (white on a dark background, black on light) - const real_t lum = static_cast(0.299) * m_background[0] + - static_cast(0.587) * m_background[1] + - static_cast(0.114) * m_background[2]; - const uint8_t tc = (lum < HALF) ? 255 : 0; - cbar_hidden::drawText(buf, cw, ch, tx, ty, str, s, tc, tc, tc); - }; - - // canvas margins: axes (left + bottom) and the colorbar strip (right). - // The data region sits at (ml, 0); margins/strip are background-filled. - // The polar (curvilinear) overlay annotates inside the data region (the - // disk is centered with background around it), so it needs no margins. - const bool polar = (m_global_extent.size() == 2) and m_slice_polar; - // a fisheye dome master must stay exactly W x H (its inscribed circle is - // the dome), so it takes no axes margins and no outside colorbar strip. - const bool dome = m_dome_active; - int ml = 0, mb = 0; - out::axesMargins(m_axes and not polar and not dome, m_height, ml, mb); - const int strip = (m_colorbar and m_colorbar_outside and not dome) - ? colorbarBlockWidth(m_height) - : 0; - const int CW = ml + m_width + strip; - const int CH = m_height + mb; - - // 2D-Cartesian data box (== the render region, before the aspect-expansion - // that pads the window with background): its top & right edges in - // data-region pixels. The axes/spine clamp to it and the time label sits - // in the pad above it, so neither includes the empty aspect padding. - const bool cart2d = (m_global_extent.size() == 2) and not polar and - not dome; - int dbox_top = 0; // data box top edge (px from data top) - int dbox_bot = m_height; // data box bottom edge (px) - int dbox_right = m_width; // data box right edge (px from data left) - if (cart2d and m_region.size() >= 2) { - const real_t u0 = m_slice_win[0], u1 = m_slice_win[1]; - const real_t v0 = m_slice_win[2], v1 = m_slice_win[3]; - const real_t du1 = m_region[0].second; // data box right in world (x1) - const real_t dv0 = m_region[1].first; // data box bottom in world (x2) - const real_t dv1 = m_region[1].second; // data box top in world (x2) - if (u1 > u0) { - int r = static_cast(std::lround( - static_cast((du1 - u0) / (u1 - u0)) * (m_width - 1))); - dbox_right = (r < 0) ? 0 : ((r > m_width) ? m_width : r); - } - if (v1 > v0) { - int t = static_cast(std::lround( - static_cast((v1 - dv1) / (v1 - v0)) * (m_height - 1))); - int b = static_cast(std::lround( - static_cast((v1 - dv0) / (v1 - v0)) * (m_height - 1))); - dbox_top = (t < 0) ? 0 : ((t > m_height) ? m_height : t); - dbox_bot = (b < 0) ? 0 : ((b > m_height) ? m_height : b); - } - } - // time-label anchor: for a 2D slice, the top-right of the data box (label - // goes in the pad above it); otherwise the top-right above the colorbar. - const int cbar_top = m_colorbar ? (CH - CH / 2) / 2 : (CH / 4); - const int tl_right = cart2d ? (ml + dbox_right) : (ml + m_width); - const int tl_top = cart2d ? dbox_top : cbar_top; - // colorbar vertical span: aligned to the actual data domain for a 2D slice - // (so it's centered on the data, not the aspect-padded canvas); sentinel - // (-1) elsewhere -> drawColorbar centers it on the canvas as before. - const int cbar_span_top = cart2d ? dbox_top : -1; - const int cbar_span_bot = cart2d ? dbox_bot : -1; - - bool ok = true; - if (CW == m_width and CH == m_height and not m_axes) { - // no margins, no outside strip, no overlay: colorbar overlays the data - drawBar(data.data(), m_width, m_height, cbar_span_top, cbar_span_bot); - drawTimeLabel(data.data(), m_width, m_height, tl_right, tl_top); - ok = write_png(fname, m_width, m_height, data.data()); - } else { - const uint8_t bR = quantize(m_background[0]); - const uint8_t bG = quantize(m_background[1]); - const uint8_t bB = quantize(m_background[2]); - std::vector canvas(static_cast(CW) * CH * 4); - for (std::size_t i = 0; i < canvas.size(); i += 4) { - canvas[i + 0] = bR; - canvas[i + 1] = bG; - canvas[i + 2] = bB; - canvas[i + 3] = 255; - } - for (int y = 0; y < m_height; ++y) { - std::copy_n( - &data[static_cast(y) * m_width * 4], - static_cast(m_width) * 4, - &canvas[(static_cast(y) * CW + ml) * 4]); - } - if (m_axes and not dome) { - if (m_global_extent.size() == 3) { - out::drawAxes3D(canvas.data(), CW, CH, ml, m_width, m_height, - m_camera_dev, m_region, m_axis_labels, - m_background, m_axis_nticks); - } else if (polar) { - out::drawAxesPolar(canvas.data(), CW, CH, ml, m_width, m_height, - m_slice_win[0], m_slice_win[1], m_slice_win[2], - m_slice_win[3], m_slice_rmin, m_slice_rmax, - m_slice_tmin, m_slice_tmax, m_slice_pmirror, "R", - "Theta", m_background, m_axis_nticks); - } else { - // data box (== region, un-expanded) so the spine hugs the domain, - // not the aspect-padded window - const real_t du0 = m_region[0].first, du1 = m_region[0].second; - const real_t dv0 = m_region[1].first, dv1 = m_region[1].second; - out::drawAxes2D(canvas.data(), CW, CH, ml, m_width, m_height, - m_slice_win[0], m_slice_win[1], m_slice_win[2], - m_slice_win[3], du0, du1, dv0, dv1, m_slice_xlabel, - m_slice_ylabel, m_background, m_axis_nticks); - } - } - drawBar(canvas.data(), CW, CH, cbar_span_top, cbar_span_bot); - drawTimeLabel(canvas.data(), CW, CH, tl_right, tl_top); - ok = write_png(fname, CW, CH, canvas.data()); - } - if (not ok) { - raise::Warning( - fmt::format("failed to write %s", fname.string().c_str()), - HERE); - } + writeFrame(img, scene, step, time); }; #if defined(MPI_ENABLED) @@ -839,4 +926,150 @@ namespace out { #endif } + void Renderer::compositeFragAndWrite(FragImage&& frag, + const Scene& scene, + timestep_t step, + simtime_t time) const { + // fully-opaque cull is exact: mergeFrag only drops provably-occluded + // fragments, so the result is independent of the tree's grouping. + const real_t cull = ONE; + + // collapse a merged fragment image into a full premultiplied float frame + auto fragToFull = [&](const FragImage& f) -> std::vector { + const std::size_t npix = static_cast(m_width) * + static_cast(m_height); + std::vector full(npix * 4, ZERO); + for (int y = 0; y < f.h; ++y) { + for (int x = 0; x < f.w; ++x) { + const int fx = f.x0 + x, fy = f.y0 + y; + if (fx < 0 or fx >= m_width or fy < 0 or fy >= m_height) { + continue; + } + const std::size_t p = static_cast(y) * f.w + x; + const uint32_t k0 = f.offs[p], k1 = f.offs[p + 1]; + if (k1 <= k0) { + continue; + } + real_t out[4]; + out::fragOver(f.depth, f.rgba, k0, k1, out); + const std::size_t fi = (static_cast(fy) * m_width + fx) * 4; + full[fi + 0] = out[0]; + full[fi + 1] = out[1]; + full[fi + 2] = out[2]; + full[fi + 3] = out[3]; + } + } + return full; + }; + +#if defined(MPI_ENABLED) + int rank = 0, size = 1; + MPI_Comm_rank(MPI_COMM_WORLD, &rank); + MPI_Comm_size(MPI_COMM_WORLD, &size); + + if (size == 1) { + writeFrame(fragToFull(frag), scene, step, time); + return; + } + + constexpr int TAG_HDR = 7401; + constexpr int TAG_OFFS = 7402; + constexpr int TAG_DEPTH = 7403; + constexpr int TAG_RGBA = 7404; + + // Wire format: header {x0,y0,w,h,n_frag}; per-pixel prefix offsets (uint32); + // per-fragment depth (full real_t, so the cross-rank ordering key is exact) + // and premultiplied RGBA (uint8; only this adds ~1 LSB through the tree). + auto sendFrag = [&](const FragImage& s, int dest) { + const uint32_t nfrag = s.offs.empty() + ? 0u + : s.offs.back(); + // MPI counts are `int`; the RGBA payload has nfrag*4 elements, so a single + // message overflows int once nfrag > INT_MAX/4. That regime (a 4096^2 + // near-opaque-free dome on many ranks) needs the band-tiling optimization; + // fail loudly here rather than send a negative count. + raise::ErrorIf(nfrag > 536870911u, + "dome A-buffer: per-message fragment count exceeds the MPI " + "int limit (nfrag*4 > INT_MAX). Lower render.resolution " + "(frame-band tiling is a pending optimization).", + HERE); + int hdr[5] = { s.x0, s.y0, s.w, s.h, static_cast(nfrag) }; + MPI_Send(hdr, 5, MPI_INT, dest, TAG_HDR, MPI_COMM_WORLD); + const int np = s.w * s.h; + if (np > 0) { + MPI_Send(s.offs.data(), np + 1, MPI_UINT32_T, dest, TAG_OFFS, + MPI_COMM_WORLD); + } + if (nfrag > 0) { + // depth is sent at full real_t precision (NOT downcast to float): the + // depth key orders fragments across ranks, and local fragments keep + // real_t, so a float round-trip would make cross-rank vs within-rank + // ordering disagree at close depths -> a seam at the domain boundary. + MPI_Send(s.depth.data(), static_cast(nfrag), + mpi::get_type(), dest, TAG_DEPTH, MPI_COMM_WORLD); + std::vector bytes(static_cast(nfrag) * 4); + for (std::size_t i = 0; i < bytes.size(); ++i) { + bytes[i] = quantize(s.rgba[i]); + } + MPI_Send(bytes.data(), static_cast(bytes.size()), + MPI_UNSIGNED_CHAR, dest, TAG_RGBA, MPI_COMM_WORLD); + } + }; + auto recvFrag = [&](int src) -> FragImage { + int hdr[5]; + MPI_Recv(hdr, 5, MPI_INT, src, TAG_HDR, MPI_COMM_WORLD, MPI_STATUS_IGNORE); + FragImage s; + s.x0 = hdr[0]; + s.y0 = hdr[1]; + s.w = hdr[2]; + s.h = hdr[3]; + const uint32_t nfrag = static_cast(hdr[4]); + const int np = s.w * s.h; + if (np > 0) { + s.offs.resize(static_cast(np) + 1); + MPI_Recv(s.offs.data(), np + 1, MPI_UINT32_T, src, TAG_OFFS, + MPI_COMM_WORLD, MPI_STATUS_IGNORE); + } + if (nfrag > 0) { + s.depth.resize(nfrag); + MPI_Recv(s.depth.data(), static_cast(nfrag), mpi::get_type(), + src, TAG_DEPTH, MPI_COMM_WORLD, MPI_STATUS_IGNORE); + std::vector bytes(static_cast(nfrag) * 4); + MPI_Recv(bytes.data(), static_cast(bytes.size()), + MPI_UNSIGNED_CHAR, src, TAG_RGBA, MPI_COMM_WORLD, + MPI_STATUS_IGNORE); + s.rgba.resize(static_cast(nfrag) * 4); + const real_t inv255 = ONE / static_cast(255); + for (std::size_t i = 0; i < s.rgba.size(); ++i) { + s.rgba[i] = static_cast(bytes[i]) * inv255; + } + } + return s; + }; + + // Order-INDEPENDENT binary tree reduction: mergeFrag (depth-sorted merge + + // occlusion cull) is associative + commutative, so no global order is needed + // -- reduce straight to rank 0 (== MPI_ROOT_RANK). Each round, the lower + // partner receives + merges, the upper sends and drops out. + FragImage cur = std::move(frag); + for (int s = 1; s < size; s <<= 1) { + if ((rank % (2 * s)) == 0) { + const int pp = rank + s; + if (pp < size) { + FragImage back = recvFrag(pp); + cur = mergeFrag(cur, back, cull); + } + } else if ((rank % (2 * s)) == s) { + sendFrag(cur, rank - s); + break; + } + } + if (rank == MPI_ROOT_RANK) { + writeFrame(fragToFull(cur), scene, step, time); + } +#else + writeFrame(fragToFull(frag), scene, step, time); +#endif + } + } // namespace out diff --git a/src/output/render/renderer.h b/src/output/render/renderer.h index 8fe1fa196..4e09dfa30 100644 --- a/src/output/render/renderer.h +++ b/src/output/render/renderer.h @@ -42,15 +42,22 @@ namespace out { * @note Trivially copyable; captured by value into the Kokkos kernel. */ struct CameraDevice { + // projection: 0 orthographic, 1 perspective (pinhole), 2 dome (fulldome + // azimuthal-equidistant fisheye from an interior eye). `orthographic` is + // kept for back-compat (== projection 0); the kernel branches on `projection`. + enum Projection { Ortho = 0, Perspective = 1, Dome = 2 }; real_t eye[3] { ZERO, ZERO, ZERO }; real_t right[3] { ONE, ZERO, ZERO }; real_t up[3] { ZERO, ONE, ZERO }; + // for the dome, `forward` is the ZENITH direction (center of the fisheye disk) real_t forward[3] { ZERO, ZERO, -ONE }; real_t tan_half_fov { ONE }; real_t aspect { ONE }; bool orthographic { true }; real_t half_w { ONE }; real_t half_h { ONE }; + int projection { Ortho }; + real_t dome_half_fov { static_cast(1.5707963267948966) }; // rad; 180 deg dome => PI/2 }; /** @@ -202,6 +209,27 @@ namespace out { std::vector rgba; // w*h*4 premultiplied, pixel-major }; + /** + * @brief A sparse screen-space *fragment* buffer for the interior-eye (dome) + * composite (A-buffer / deep image). + * @note With the camera inside the box there is no single global front-to-back + * domain order, so each domain contributes, per pixel, one depth-tagged + * premultiplied-RGBA fragment (its convex slab is one contiguous ray + * interval). Compositing is a per-pixel sort by `depth` (== ray t_enter) then + * front-to-back "over". The MPI reduce merges depth-sorted lists (associative + * + commutative), so no global order is needed. CSR layout because per-pixel + * fragment counts vary strongly across the frame. A leaf (one rank) holds 0/1 + * fragment per covered pixel. See composite.h::mergeFrag / fragOver. + */ + struct FragImage { + int x0 { 0 }, y0 { 0 }; // top-left pixel in the full frame + int w { 0 }, h { 0 }; // bbox size in pixels (0 => empty) + // per-pixel prefix offsets into depth/rgba, length w*h+1 (offs[0] == 0). + std::vector offs; + std::vector depth; // n_frag entries, ascending within each pixel + std::vector rgba; // n_frag*4 premultiplied, pixel-major + }; + class Renderer { public: Renderer() {} @@ -248,6 +276,22 @@ namespace out { timestep_t step, simtime_t time) const; + /** + * @brief Depth-resolved (A-buffer) composite for the interior-eye dome, then + * write the PNG. + * @param frag this rank's sparse screen-space fragment image (depth + RGBA) + * @param scene the scene being written + * @param step current timestep (for the filename cycle number) + * @param time current simulation time (drawn as a corner label if enabled) + * @note Order-independent: the tree reduce merges depth-sorted fragment lists + * (associative + commutative), so no global rank order is needed. The + * root collapses each pixel's list front-to-back and writes the file. + */ + void compositeFragAndWrite(FragImage&& frag, + const Scene& scene, + timestep_t step, + simtime_t time) const; + /* getters -------------------------------------------------------------- */ [[nodiscard]] auto enabled() const -> bool { @@ -406,6 +450,14 @@ namespace out { } private: + // Composite a full premultiplied float frame over the background and write + // the PNG (with the colorbar / axes / time-label overlays). Shared by the + // ordered (SubImage) composite and the depth-resolved (FragImage) composite. + void writeFrame(const std::vector& img, + const Scene& scene, + timestep_t step, + simtime_t time) const; + bool m_enabled { false }; int m_width { 1024 }; From 1c90d79d56d1d4077fa809c474e40463f72a7f21 Mon Sep 17 00:00:00 2001 From: LudwigBoess Date: Tue, 28 Jul 2026 23:22:50 +0000 Subject: [PATCH 058/125] extended external force to be able to access particle properties --- examples/external_fields/pgen.hpp | 3 ++- src/global/traits/archetypes.h | 27 ++++++++++++++++++++------- src/kernels/pushers/sr.hpp | 9 +++++---- tests/archetypes/pgen.cpp | 3 ++- tests/global/traits_archetypes.cpp | 6 +++--- tests/global/traits_policies.cpp | 2 +- tests/kernels/ext_force.cpp | 9 ++++++--- 7 files changed, 39 insertions(+), 20 deletions(-) diff --git a/examples/external_fields/pgen.hpp b/examples/external_fields/pgen.hpp index ac4f6faed..30fb2473d 100644 --- a/examples/external_fields/pgen.hpp +++ b/examples/external_fields/pgen.hpp @@ -40,7 +40,8 @@ namespace user { */ // f_ext: external force-field (acceleration): - Inline auto fx1(const coord_t&) const -> real_t { + Inline auto fx1(const coord_t&, const ntt::ParticleArrays&, prtlidx_t) const + -> real_t { return (sp % 2u == 0u) ? -HALF : HALF; } diff --git a/src/global/traits/archetypes.h b/src/global/traits/archetypes.h index e170170fb..c5582e74b 100644 --- a/src/global/traits/archetypes.h +++ b/src/global/traits/archetypes.h @@ -4,7 +4,7 @@ * @implements * - EnrgDistClass<> - checks if a class can be used as an energy distribution * - SpatialDistClass<> - checks if a class can be used as a spatial distribution - * - traits::fieldsetter::HasFx1, ::HasFx2, ::HasFx3 - checks for F functions in field setter class + * - traits::fieldsetter::HasFx1, ::HasFx2, ::HasFx3 - checks for particle-aware F functions fx*(x, particles, p) * - traits::fieldsetter::HasEx1, ::HasEx2, ::HasEx3 - checks for E functions in field setter class * - traits::fieldsetter::HasBx1, ::HasBx2, ::HasBx3 - checks for B functions in field setter class * - traits::fieldsetter::HasDx1, ::HasDx2, ::HasDx3 - checks for D functions in field setter class @@ -26,6 +26,10 @@ #include +namespace ntt { + struct ParticleArrays; +} // namespace ntt + template concept EnrgDistClass = requires(const ED& edist, const coord_t& x_Ph, @@ -50,18 +54,27 @@ concept SpatialDistClass = SimpleSpatialDistClass or namespace traits::fieldsetter { template - concept HasFx1 = requires(const T& t, const coord_t& x_Ph) { - { t.fx1(x_Ph) } -> std::convertible_to; + concept HasFx1 = requires(const T& t, + const coord_t& x_Ph, + const ntt::ParticleArrays& prtls, + prtlidx_t p) { + { t.fx1(x_Ph, prtls, p) } -> std::convertible_to; }; template - concept HasFx2 = requires(const T& t, const coord_t& x_Ph) { - { t.fx2(x_Ph) } -> std::convertible_to; + concept HasFx2 = requires(const T& t, + const coord_t& x_Ph, + const ntt::ParticleArrays& prtls, + prtlidx_t p) { + { t.fx2(x_Ph, prtls, p) } -> std::convertible_to; }; template - concept HasFx3 = requires(const T& t, const coord_t& x_Ph) { - { t.fx3(x_Ph) } -> std::convertible_to; + concept HasFx3 = requires(const T& t, + const coord_t& x_Ph, + const ntt::ParticleArrays& prtls, + prtlidx_t p) { + { t.fx3(x_Ph, prtls, p) } -> std::convertible_to; }; template diff --git a/src/kernels/pushers/sr.hpp b/src/kernels/pushers/sr.hpp index 78b5c4916..bc7d79f08 100644 --- a/src/kernels/pushers/sr.hpp +++ b/src/kernels/pushers/sr.hpp @@ -245,7 +245,7 @@ namespace kernel::sr { // compute the external force either user-provided or from the atmosphere model if constexpr (HasExtForce or Atm) { - getExternalForce(xp_Cd, xp_Ph, external_force_Cart); + getExternalForce(xp_Cd, xp_Ph, p, external_force_Cart); } if (ctx.pusher_flags & ParticlePusher::GCA) { @@ -1425,18 +1425,19 @@ namespace kernel::sr { Inline void getExternalForce(const coord_t& xp_Cd, const coord_t& xp_Ph, + [[maybe_unused]] prtlidx_t p, vec_t& external_force_Cart) const requires(Atm or HasExtForce) { real_t f_x1 = ZERO, f_x2 = ZERO, f_x3 = ZERO; if constexpr (HasExtFx1) { - f_x1 = policies.external_fields_policy.fx1(xp_Ph); + f_x1 = policies.external_fields_policy.fx1(xp_Ph, particles, p); } if constexpr (HasExtFx2) { - f_x2 = policies.external_fields_policy.fx2(xp_Ph); + f_x2 = policies.external_fields_policy.fx2(xp_Ph, particles, p); } if constexpr (HasExtFx3) { - f_x3 = policies.external_fields_policy.fx3(xp_Ph); + f_x3 = policies.external_fields_policy.fx3(xp_Ph, particles, p); } if constexpr (Atm) { if constexpr (D == Dim::_1D or D == Dim::_2D or D == Dim::_3D) { diff --git a/tests/archetypes/pgen.cpp b/tests/archetypes/pgen.cpp index d95e9cf9c..fb428493b 100644 --- a/tests/archetypes/pgen.cpp +++ b/tests/archetypes/pgen.cpp @@ -25,7 +25,8 @@ struct CustomFieldsetter { template struct ExtForce { - Inline auto fx1(const coord_t&) const -> real_t { + Inline auto fx1(const coord_t&, const ntt::ParticleArrays&, prtlidx_t) const + -> real_t { return ZERO; } diff --git a/tests/global/traits_archetypes.cpp b/tests/global/traits_archetypes.cpp index d72545102..575b7abe5 100644 --- a/tests/global/traits_archetypes.cpp +++ b/tests/global/traits_archetypes.cpp @@ -89,15 +89,15 @@ struct WithDx1Dx2Dx3 { }; struct WithFx1Fx2Fx3 { - real_t fx1(const coord_t&) const { + real_t fx1(const coord_t&, const ntt::ParticleArrays&, prtlidx_t) const { return ZERO; } - real_t fx2(const coord_t&) const { + real_t fx2(const coord_t&, const ntt::ParticleArrays&, prtlidx_t) const { return ZERO; } - real_t fx3(const coord_t&) const { + real_t fx3(const coord_t&, const ntt::ParticleArrays&, prtlidx_t) const { return ZERO; } }; diff --git a/tests/global/traits_policies.cpp b/tests/global/traits_policies.cpp index 5f5a83f9f..82ceb7e12 100644 --- a/tests/global/traits_policies.cpp +++ b/tests/global/traits_policies.cpp @@ -105,7 +105,7 @@ static_assert(not EmissionPolicyClass); // --- ExtFieldsPolicyClass --- struct WithFx1 { - real_t fx1(const coord_t&) const { + real_t fx1(const coord_t&, const ntt::ParticleArrays&, prtlidx_t) const { return ZERO; } }; diff --git a/tests/kernels/ext_force.cpp b/tests/kernels/ext_force.cpp index 69d773d10..d9e8203a5 100644 --- a/tests/kernels/ext_force.cpp +++ b/tests/kernels/ext_force.cpp @@ -54,15 +54,18 @@ void put_value(array_t& arr, T v, prtlidx_t p) { struct Force { Force(real_t force) : force { force } {} - Inline auto fx1(const coord_t&) const -> real_t { + Inline auto fx1(const coord_t&, const ntt::ParticleArrays&, prtlidx_t) const + -> real_t { return force * math::sin(ONE) * math::sin(ONE); } - Inline auto fx2(const coord_t&) const -> real_t { + Inline auto fx2(const coord_t&, const ntt::ParticleArrays&, prtlidx_t) const + -> real_t { return force * math::sin(ONE) * math::cos(ONE); } - Inline auto fx3(const coord_t&) const -> real_t { + Inline auto fx3(const coord_t&, const ntt::ParticleArrays&, prtlidx_t) const + -> real_t { return force * math::cos(ONE); } From b50fad676cbeef587d590b140a3dda1f8ee45ff7 Mon Sep 17 00:00:00 2001 From: LudwigBoess Date: Wed, 29 Jul 2026 01:28:07 +0000 Subject: [PATCH 059/125] only render half-sphere within `dome_radius` to avoid corner projection effects --- input.example.toml | 10 ++++++++++ src/framework/domain/metadomain_render.cpp | 8 +++++++- src/output/render/raymarch.hpp | 8 ++++++++ src/output/render/renderer.cpp | 17 +++++++++++++++++ src/output/render/renderer.h | 4 ++++ 5 files changed, 46 insertions(+), 1 deletion(-) diff --git a/input.example.toml b/input.example.toml index 8ccf454d2..9a8956596 100644 --- a/input.example.toml +++ b/input.example.toml @@ -872,6 +872,16 @@ # @type: float [> 0.0, <= 360.0] # @default: 180.0 dome_fov = "" + # Dome far-clip radius in world units (dome mode only): each ray stops this + # far from the eye, so the sampled region is a half-ball (hemisphere) of + # this radius rather than the whole box -> uniform path length and no box + # corner/edge projection artifacts. `samples` then counts steps across this + # radius. + # @type: float [>= 0.0] + # @default: the largest sphere centered in the box (half the shortest + # side), so it touches the face centers and never a corner + # @note: 0 disables the clip (rays march to the box boundary) + dome_radius = "" # Vertical extent of the view in world units (orthographic only) # @type: float [> 0.0] # @default: the global box diagonal (the whole box fits from any angle) diff --git a/src/framework/domain/metadomain_render.cpp b/src/framework/domain/metadomain_render.cpp index a43adcf9e..c119b8fc6 100644 --- a/src/framework/domain/metadomain_render.cpp +++ b/src/framework/domain/metadomain_render.cpp @@ -703,9 +703,15 @@ namespace ntt { gdiag += s * s; } gdiag = math::sqrt(gdiag); + // marched extent per ray: the dome clips each ray to `dome_radius`, so size + // the step by the radius (== `samples` steps across the hemisphere) rather + // than the box diagonal. Identical on all ranks -> seamless. + const real_t march_len = (is_dome and cam.dome_radius > ZERO) + ? cam.dome_radius + : gdiag; const real_t ds = (g_renderer.stepSize() > ZERO) ? g_renderer.stepSize() - : gdiag / static_cast(g_renderer.samples()); + : march_len / static_cast(g_renderer.samples()); const int max_steps = 2 * g_renderer.samples() + 16; // region box + depth-occluded spine (opaque box wireframe rendered inline diff --git a/src/output/render/raymarch.hpp b/src/output/render/raymarch.hpp index 4b4daebfb..f86be67f7 100644 --- a/src/output/render/raymarch.hpp +++ b/src/output/render/raymarch.hpp @@ -412,6 +412,14 @@ namespace kernel { t_enter = (t1 > t_enter) ? t1 : t_enter; t_exit = (t2 < t_exit) ? t2 : t_exit; } + // dome: clip each ray to a fixed radius around the interior eye, so the + // sampled region is a half-ball (hemisphere) of that radius rather than + // the whole box -> uniform path length, no box corner/edge projection + // artifacts. `t` is world distance from the shared eye (dir is unit), so + // this is a sphere clip and is identical on every rank (seamless). + if (cam.projection == out::CameraDevice::Dome and cam.dome_radius > ZERO) { + t_exit = (t_exit < cam.dome_radius) ? t_exit : cam.dome_radius; + } if (t_enter >= t_exit) { return; } diff --git a/src/output/render/renderer.cpp b/src/output/render/renderer.cpp index e007e8b78..d4f16f145 100644 --- a/src/output/render/renderer.cpp +++ b/src/output/render/renderer.cpp @@ -431,6 +431,23 @@ namespace out { "elliptical otherwise", HERE); } + // spherical far-clip: each ray stops at `dome_radius` from the eye, so the + // sampled region is a half-ball (hemisphere) instead of the whole box -> + // uniform path length, no box corner/edge projection artifacts. Default = + // the largest sphere centered in the box (half the shortest side). A value + // of 0 disables the clip (march to the box boundary); a negative value + // also selects the default. + real_t insc = static_cast(1e30); + for (std::size_t d = 0; d < m_region.size() and d < 3; ++d) { + insc = (size[d] < insc) ? size[d] : insc; + } + insc *= HALF; + real_t domeR = toml::find_or(td, "output", "render", "camera", + "dome_radius", insc); + if (domeR < ZERO) { + domeR = insc; + } + m_camera_dev.dome_radius = domeR; } /* ---- moving view (pan the region/camera to track a feature) --------- */ diff --git a/src/output/render/renderer.h b/src/output/render/renderer.h index 4e09dfa30..67a0a58ff 100644 --- a/src/output/render/renderer.h +++ b/src/output/render/renderer.h @@ -58,6 +58,10 @@ namespace out { real_t half_h { ONE }; int projection { Ortho }; real_t dome_half_fov { static_cast(1.5707963267948966) }; // rad; 180 deg dome => PI/2 + // dome far-clip: rays stop at this world distance from the eye, so the + // sampled region is a half-ball (hemisphere) of this radius instead of the + // whole box -> no box corner/edge path-length artifacts. 0 => no clip. + real_t dome_radius { ZERO }; }; /** From 2d0a157db9d0ccb36f90378765fee8740506d404 Mon Sep 17 00:00:00 2001 From: A Sullivan Date: Fri, 31 Jul 2026 10:28:19 -0400 Subject: [PATCH 060/125] added spectra3D components to entity_1.5rc --- src/framework/domain/metadomain_io.cpp | 249 ++++++++++++++++++++++++- src/framework/parameters/output.cpp | 45 ++++- src/framework/parameters/output.h | 8 + src/global/defaults.h | 7 + src/global/global.h | 1 + src/output/writer.cpp | 70 +++++++ src/output/writer.h | 1 + 7 files changed, 378 insertions(+), 3 deletions(-) diff --git a/src/framework/domain/metadomain_io.cpp b/src/framework/domain/metadomain_io.cpp index 3fc0f5f57..b42247634 100644 --- a/src/framework/domain/metadomain_io.cpp +++ b/src/framework/domain/metadomain_io.cpp @@ -109,7 +109,7 @@ namespace ntt { spectra_species.push_back(sp.index()); } g_writer.defineSpectraOutputs(spectra_species); - for (const auto& type : { "fields", "particles", "spectra" }) { + for (const auto& type : { "fields", "particles", "spectra", "spectra3D" }) { g_writer.addTracker(type, params.template get( "output." + std::string(type) + ".interval"), @@ -382,8 +382,13 @@ namespace ntt { g_writer.shouldWrite("spectra", finished_step, finished_time); + const auto write_spectra3D = params.template get( + "output.spectra3D.enable") and + g_writer.shouldWrite("spectra3D", + finished_step, + finished_time); const auto extension = params.template get("output.format"); - if (not(write_fields or write_particles or write_spectra) and + if (not(write_fields or write_particles or write_spectra or write_spectra3D) and extension != "disabled") { return false; } @@ -842,6 +847,246 @@ namespace ntt { g_writer.endWriting(WriteMode::Spectra); } // end shouldWrite("spectra", step, time) + if (write_spectra3D) { + g_writer.beginWriting(WriteMode::Spectra3D, current_step, current_time); + const auto log_bins = params.template get( + "output.spectra3D.log_bins"); + const auto n_bins = params.template get( + "output.spectra3D.n_bins"); + const auto& metric = local_domain->mesh.metric; + // extract the number of bins globally in each direction + const auto nx1_bins = params.template get("output.spectra3D.nx1"); + const auto nx2_bins = params.template get("output.spectra3D.nx2"); + const auto nx3_bins = params.template get("output.spectra3D.nx3"); + + // select the min and max energy for the spectra + auto e_min = params.template get("output.spectra3D.e_min"); + auto e_max = params.template get("output.spectra3D.e_max"); + + + const auto x1_size_local = local_domain->mesh.n_active()[0]; + const auto x1_offset_local = local_domain->offset_ncells()[0]; + const auto x1_size_global = mesh().n_active()[0]; + + // divide based on physical coordinates maybe rather than cells + + std::size_t x2_size_local = 0; + std::size_t x2_offset_local = 0; + std::size_t x2_size_global = 0; + + std::size_t x3_size_local = 0; + std::size_t x3_offset_local = 0; + std::size_t x3_size_global = 0; + + auto x1_min = mesh().extent(in::x1).first; // extent is in physical units, not code + decltype(x1_min) x2_min = 0; + decltype(x1_min) x3_min = 0; + + auto x1_max = mesh().extent(in::x1).second; // pairs in c++ are addressed by first and secocnd + decltype(x1_max) x2_max = 0; + decltype(x1_max) x3_max = 0; + + //auto x1_extent_local = mesh().extent(in::x1) + //const auto nx1 = local_domain->mesh.n_active(in::x1); + //const auto nx2 = local_domain->mesh.n_active(in::x2); + + //auto dx1 = (x1_extent_local.second - x1_extent_local.first) / (real_t)(nx1_bin - 1); + auto x1_extent_local = local_domain->mesh.extent(in::x1); + auto x1_min_local = x1_extent_local.first; + auto x1_max_local = x1_extent_local.second; + + auto x1_ind_rank_min = static_cast(static_cast(nx1_bins) * (x1_min_local - x1_min) / (x1_max - x1_min)); + auto x1_ind_rank_max = static_cast(static_cast(nx1_bins) * (x1_max_local - x1_min) / (x1_max - x1_min)); + + decltype(x1_min_local) x2_min_local=0; + decltype(x1_max_local) x2_max_local=0; + decltype(x1_ind_rank_min) x2_ind_rank_min=0; + decltype(x1_ind_rank_max) x2_ind_rank_max=0; + + decltype(x1_min_local) x3_min_local=0; + decltype(x1_max_local) x3_max_local=0; + decltype(x1_ind_rank_min) x3_ind_rank_min=0; + decltype(x1_ind_rank_max) x3_ind_rank_max=0; + + real_t dx1_bin = (x1_max - x1_min )/nx1_bins; + real_t dx2_bin = 1.; + real_t dx3_bin = 1.; + + + + if constexpr (M::PrtlDim == Dim::_2D or M::PrtlDim == Dim::_3D){ // only pick x2 if simulation is in 2D + + x2_min = mesh().extent(in::x2).first; + x2_max = mesh().extent(in::x2).second; + + auto x2_extent_local = local_domain->mesh.extent(in::x2); + x2_min_local = x2_extent_local.first; + x2_max_local = x2_extent_local.second; + + dx2_bin = (static_cast(x2_max) - static_cast(x2_min) )/static_cast(nx2_bins); + + x2_ind_rank_min = static_cast(static_cast(nx2_bins) * (x2_min_local - x1_min) / (x2_max - x2_min)); + x2_ind_rank_max = static_cast(static_cast(nx2_bins) * (x2_max_local - x1_min) / (x2_max - x2_min)); + } + + if constexpr (M::PrtlDim == Dim::_3D){ + + x3_min = mesh().extent(in::x3).first; + x3_max = mesh().extent(in::x3).second; + + auto x3_extent_local = local_domain->mesh.extent(in::x3); + x3_min_local = x3_extent_local.first; + x3_max_local = x3_extent_local.second; + + dx3_bin = (static_cast(x3_max) - static_cast(x3_min) )/static_cast(nx3_bins); + + x3_ind_rank_min = static_cast(static_cast(nx3_bins) * (x3_min_local - x3_min) / (x1_max - x1_min)); + x3_ind_rank_max = static_cast(static_cast(nx3_bins) * (x3_max_local - x3_min) / (x1_max - x1_min)); + } + + + + //auto dx_slice = mesh().dx1; + + + //auto dxslice = (x1_max - x1_min) / x_bins + + if (log_bins) { + e_min = math::log10(e_min); + e_max = math::log10(e_max); + } + array_t energy { "energy", n_bins + 1 }; + // generating the energy bins + Kokkos::parallel_for( + "GenerateEnergyBins", + n_bins + 1, + Lambda(index_t e) { + if (log_bins) { + energy(e) = math::pow(10.0, e_min + (e_max - e_min) * e / n_bins); + } else { + energy(e) = e_min + (e_max - e_min) * e / n_bins; + } + }); + + for (const auto& spec : g_writer.spectraWriters()) { + auto& species = local_domain->species[spec.species() - 1]; + array_t dn3d { "dn3d", nx1_bins, nx2_bins, nx3_bins, n_bins }; + auto dn3d_scatter = Kokkos::Experimental::create_scatter_view(dn3d); + auto ux1 = species.ux1; + auto ux2 = species.ux2; + auto ux3 = species.ux3; + auto i1 = species.i1; + auto dx1 = species.dx1; + //adeep_copy; + decltype(i1) i2; + decltype(i1) i3; + decltype(dx1) dx2; + decltype(dx1) dx3; + if constexpr (M::PrtlDim == Dim::_2D or M::PrtlDim == Dim::_3D){ // only pick x2 if simulation is in 2D + i2 = species.i2; + dx2 = species.dx2; + } + if constexpr (M::PrtlDim == Dim::_3D){ + i3 = species.i3; + dx3 = species.dx3; + } + auto weight = species.weight; + auto tag = species.tag; + const auto is_massive = species.mass() > 0.0f; + Kokkos::parallel_for( + "ComputeSpectra", + species.rangeActiveParticles(), + Lambda(index_t p) { + if (tag(p) != ParticleTag::alive) { + return; + } + + coord_t x_Cd {ZERO}; + if (D == Dim::_1D or D == Dim::_2D or D == Dim::_3D) { + x_Cd[0] = static_cast(i1(p)) + static_cast(dx1(p)); + } + if (D == Dim::_2D or D == Dim::_3D) { + x_Cd[1] = static_cast(i2(p)) + static_cast(dx2(p)); + } + if (D == Dim::_3D) { + x_Cd[2] = static_cast(i3(p)) + static_cast(dx3(p)); + } + coord_t x_Ph { ZERO }; + metric.template convert(x_Cd, x_Ph); + + real_t en; + if (is_massive) { + en = U2GAMMA(ux1(p), ux2(p), ux3(p)) - ONE; + } else { + en = NORM(ux1(p), ux2(p), ux3(p)); + } + if (log_bins) { + en = math::log10(en); + } + std::size_t e_ind = 0; + if (en <= e_min) { + e_ind = 0; + } else if (en >= e_max) { + e_ind = n_bins-1; + } else { + e_ind = static_cast( + static_cast(n_bins) * (en - e_min) / (e_max - e_min)); + } + + std::size_t x1_ind = 0; + if (x_Ph[0]<= x1_min) { + x1_ind = 0; + } else if (x_Ph[0] >= x1_max) { + x1_ind = nx1_bins-1; + } else { + x1_ind = static_cast( + static_cast(nx1_bins) * (x_Ph[0] - x1_min) / (x1_max - x1_min)); + } + + + std::size_t x2_ind = 0; + if (M::PrtlDim == Dim::_2D or M::PrtlDim == Dim::_3D){ // only pick x2 if simulation is in 2D + if (x_Ph[1] <= x2_min) { + x2_ind = 0; + } else if (x_Ph[1] >= x2_max) { + x2_ind = nx2_bins-1; + } else { + x2_ind = static_cast( + static_cast(nx2_bins) * (x_Ph[1] - x2_min) / (x2_max - x2_min)); + } + //real_t x2_center = x2_min + (x2_ind+0.5)*dx2_bin; + //owns_x2 = (x2_center >= x2_min_local) && (x2_center <= x2_max_local); + } + std::size_t x3_ind = 0; + if (M::PrtlDim == Dim::_3D){ // only pick x3 if simulation is in 3D + + if (x_Ph[2] <= x3_min) { + x3_ind = 0; + } else if (x_Ph[2] >= x3_max) { + x3_ind = nx3_bins-1; + } else { + x3_ind = static_cast( + static_cast(nx3_bins) * (x_Ph[2] - x3_min) / (x3_max - x3_min)); + } + + + } + + // now I want to save the nx_bins contained in each rank + // can save array of x1_inds which are saved? + // maybe I can ask what local x_min and x_max are for this rank, pass that to writeSpectrum3D + + + auto dn3d_acc = dn3d_scatter.access(); + dn3d_acc(x1_ind, x2_ind, x3_ind, e_ind) += weight(p); + }); + Kokkos::Experimental::contribute(dn3d, dn3d_scatter); + g_writer.writeSpectrum3D(dn3d, spec.name()); + } + g_writer.writeSpectrumBins(energy, "sEbn"); + g_writer.endWriting(WriteMode::Spectra3D); + } + return true; } diff --git a/src/framework/parameters/output.cpp b/src/framework/parameters/output.cpp index 6c1858e9d..13e1a5463 100644 --- a/src/framework/parameters/output.cpp +++ b/src/framework/parameters/output.cpp @@ -33,7 +33,7 @@ namespace ntt { HERE); categories.emplace(); - for (const auto& category : { "fields", "particles", "spectra", "stats" }) { + for (const auto& category : { "fields", "particles", "spectra", "spectra3D", "stats" }) { const auto q_int = toml::find_or(toml_data, "output", category, @@ -158,6 +158,41 @@ namespace ntt { "spectra", "n_bins", defaults::output::spec_nbins); + + /* Spectra3D ------------------------------------------------------------ */ + spectra3d_e_min = toml::find_or(toml_data, "output", "spectra3D", "e_min", defaults::output::spec3d_emin); + + spectra3d_e_max = toml::find_or(toml_data, "output", "spectra3D", "e_max", defaults::output::spec3d_emin); + + spectra3d_log_bins = toml::find_or(toml_data, + "output", + "spectra3D", + "log_bins", + defaults::output::spec3d_log); + + spectra3d_n_bins = toml::find_or(toml_data, + "output", + "spectra3D", + "n_bins", + defaults::output::spec3d_nbins); + + spectra3d_nx1 = toml::find_or(toml_data, + "output", + "spectra3D", + "nx1", + defaults::output::spec3d_nx1); + + spectra3d_nx2 = toml::find_or(toml_data, + "output", + "spectra3D", + "nx2", + defaults::output::spec3d_nx1); + + spectra3d_nx3 = toml::find_or(toml_data, + "output", + "spectra3D", + "nx3", + defaults::output::spec3d_nx1); /* Stats ---------------------------------------------------------------- */ stats_quantities = toml::find_or(toml_data, @@ -209,6 +244,14 @@ namespace ntt { params->set("output.spectra.log_bins", spectra_log_bins.value()); params->set("output.spectra.n_bins", spectra_n_bins.value()); + params->set("output.spectra3D.e_min", spectra3d_e_min.value()); + params->set("output.spectra3D.e_max", spectra3d_e_max.value()); + params->set("output.spectra3D.log_bins", spectra3d_log_bins.value()); + params->set("output.spectra3D.n_bins", spectra3d_n_bins.value()); + params->set("output.spectra3D.nx1", spectra3d_nx1.value()); + params->set("output.spectra3D.nx2", spectra3d_nx2.value()); + params->set("output.spectra3D.nx3", spectra3d_nx3.value()); + params->set("output.stats.quantities", stats_quantities.value()); params->set("output.stats.custom", stats_custom_quantities.value()); diff --git a/src/framework/parameters/output.h b/src/framework/parameters/output.h index 27b077bc5..680751514 100644 --- a/src/framework/parameters/output.h +++ b/src/framework/parameters/output.h @@ -55,6 +55,14 @@ namespace ntt { std::optional spectra_log_bins; std::optional spectra_n_bins; + std::optional spectra3d_e_min; + std::optional spectra3d_e_max; + std::optional spectra3d_log_bins; + std::optional spectra3d_n_bins; + std::optional spectra3d_nx1; + std::optional spectra3d_nx2; + std::optional spectra3d_nx3; + std::optional> stats_quantities; std::optional> stats_custom_quantities; diff --git a/src/global/defaults.h b/src/global/defaults.h index e1387677e..0b7e30355 100644 --- a/src/global/defaults.h +++ b/src/global/defaults.h @@ -74,6 +74,13 @@ namespace ntt::defaults { const real_t spec_emax = 1e3; const bool spec_log = true; const std::size_t spec_nbins = 200; + const real_t spec3d_emin = 1e-3; + const real_t spec3d_emax = 1e3; + const bool spec3d_log = true; + const std::size_t spec3d_nbins = 200; + const std::size_t spec3d_nx1 = 1; + const std::size_t spec3d_nx2 = 1; + const std::size_t spec3d_nx3 = 1; const std::vector stats_quantities = { "B^2", "E^2", "ExB", diff --git a/src/global/global.h b/src/global/global.h index c3fbc5061..00047359a 100644 --- a/src/global/global.h +++ b/src/global/global.h @@ -298,6 +298,7 @@ namespace WriteMode { Particles = 1 << 1, Spectra = 1 << 2, Stats = 1 << 3, + Spectra3D = 1 << 4, }; } // namespace WriteMode diff --git a/src/output/writer.cpp b/src/output/writer.cpp index a84745c5b..317d95af9 100644 --- a/src/output/writer.cpp +++ b/src/output/writer.cpp @@ -383,6 +383,74 @@ namespace out { #endif } +// spectrum3D, broken up into separate sub-domains + void Writer::writeSpectrum3D(const array_t& counts3D, + const std::string& varname) { + // need to include the rank specific coordinates contained here + std::string varname3d = varname + "_3D"; + //auto var = m_io.InquireVariable(varname); + auto counts3D_h = Kokkos::create_mirror_view(counts3D); + // copy to host + Kokkos::deep_copy(counts3D_h, counts3D); +#if defined(MPI_ENABLED) + array_t counts3D_all { "counts3D_all", counts3D.extent(0), counts3D.extent(1), counts3D.extent(2), counts3D.extent(3) }; + //auto counts_h_all = Kokkos::create_mirror_view(counts_all); + int rank; + int size; + MPI_Comm_rank(MPI_COMM_WORLD, &rank); + MPI_Comm_size(MPI_COMM_WORLD, &size); + + auto counts3D_h_all = Kokkos::create_mirror_view(counts3D_all); + + + MPI_Allreduce(counts3D_h.data(), + counts3D_h_all.data(), + counts3D_h.extent(0)*counts3D_h.extent(1)*counts3D_h.extent(2)*counts3D_h.extent(3),// size so probably need to multiply counts_h.extent(0)*counts_h.extent(1)*counts_h.extent(2) + mpi::get_type(), + MPI_SUM, + MPI_COMM_WORLD); // size + //Kokkos::deep_copy() +#else + int rank = 0; + int size = 1; +#endif + using Shape = std::vector; + + Shape shape = {counts3D.extent(0), counts3D.extent(1), counts3D.extent(2), counts3D.extent(3)}; // global size of counts3D_all + Shape start, count; // per-rank selection + + // split the domain along the x1 direction across the different ranks + auto split_x1 = [&](std::size_t n, int this_rank, int nranks) + { + // base number of cells to allocate to each rank + std::size_t base = n / nranks; + // remainder that do not fit neatly in one rank + std::size_t rem = n % nranks; + // number of cells to allocate to the rank including the remainder + std::size_t nloc = base + (this_rank <(int)rem ? 1 : 0); // allocate one more if this rank is less that the remainder + //offset to allocate + std::size_t off = base * this_rank + std::min(this_rank, rem); + + return std::pair{nloc, off}; + }; + + auto [nloc0, off0] = split_x1(counts3D.extent(0), rank, size); + start = {off0, 0, 0, 0}; + count = {nloc0, counts3D.extent(1), counts3D.extent(2), counts3D.extent(3)}; + + auto var = m_io.InquireVariable(varname3d); + if (!var) + { + var = m_io.DefineVariable(varname3d, shape, start, count, adios2::ConstantDims); + } + + //auto var = m_io.DefineVariable(varname3d, shape, start, count, adios2::ConstantDims); + + auto counts3D_h_all_slab = Kokkos::subview(counts3D_h_all, Kokkos::make_pair(off0, off0+nloc0), Kokkos::ALL(), Kokkos::ALL(), Kokkos::ALL()); + + m_writer.Put(var, counts3D_h_all_slab, adios2::Mode::Sync); + } + void Writer::writeSpectrumBins(const array_t& e_bins, const std::string& varname) { auto var = m_io.InquireVariable(varname); @@ -448,6 +516,8 @@ namespace out { mode_str = "particles"; } else if (write_mode == WriteMode::Spectra) { mode_str = "spectra"; + } else if (write_mode == WriteMode::Spectra3D) { + mode_str = "spectra3D"; } else { raise::Fatal("Unknown write mode", HERE); } diff --git a/src/output/writer.h b/src/output/writer.h index 055cb215e..b2516b33f 100644 --- a/src/output/writer.h +++ b/src/output/writer.h @@ -127,6 +127,7 @@ namespace out { npart_t, const std::string&); void writeSpectrum(const array_t&, const std::string&); + void writeSpectrum3D(const array_t&, const std::string&); void writeSpectrumBins(const array_t&, const std::string&); void beginWriting(WriteModeTags, timestep_t, simtime_t); From 7c828be35e4d6d8afd9b1bcab64fd75351927260 Mon Sep 17 00:00:00 2001 From: A Sullivan Date: Fri, 31 Jul 2026 12:56:13 -0400 Subject: [PATCH 061/125] fixed bug in metadomain_io.cpp and added template g_checkpoint_writer.io().template in metadomain_chckpt.cpp --- pgens/shock/shock.toml | 14 +++++++++++++- src/framework/domain/metadomain_chckpt.cpp | 6 +++--- src/framework/domain/metadomain_io.cpp | 4 ++-- 3 files changed, 18 insertions(+), 6 deletions(-) diff --git a/pgens/shock/shock.toml b/pgens/shock/shock.toml index 8cde86990..37515078f 100644 --- a/pgens/shock/shock.toml +++ b/pgens/shock/shock.toml @@ -63,7 +63,19 @@ stride = 10 [output.spectra] - enable = false + enable = true + e_min = 1e-2 + e_max = 1e1 + log_bins = true + + [output.spectra3D] + enable = true + e_min = 1e-2 + e_max = 1e1 + log_bins = true + nx1 = 40 + nx2 = 3 + nx3 = 1 [diagnostics] log_level = "WARNING" diff --git a/src/framework/domain/metadomain_chckpt.cpp b/src/framework/domain/metadomain_chckpt.cpp index 1116df85e..8ed0f7e99 100644 --- a/src/framework/domain/metadomain_chckpt.cpp +++ b/src/framework/domain/metadomain_chckpt.cpp @@ -80,17 +80,17 @@ namespace ntt { species.CheckpointDeclare(g_checkpoint_writer.io()); } for (auto d { 0u }; d < M::Dim; ++d) { - g_checkpoint_writer.io().DefineVariable( + g_checkpoint_writer.io().template DefineVariable( fmt::format("subdomain_x%d_min", d + 1), { adios2::UnknownDim }, { adios2::UnknownDim }, { adios2::UnknownDim }); - g_checkpoint_writer.io().DefineVariable( + g_checkpoint_writer.io().template DefineVariable( fmt::format("subdomain_x%d_max", d + 1), { adios2::UnknownDim }, { adios2::UnknownDim }, { adios2::UnknownDim }); - g_checkpoint_writer.io().DefineVariable( + g_checkpoint_writer.io().template DefineVariable( fmt::format("subdomain_nx%d", d + 1), { adios2::UnknownDim }, { adios2::UnknownDim }, diff --git a/src/framework/domain/metadomain_io.cpp b/src/framework/domain/metadomain_io.cpp index b42247634..8669efe10 100644 --- a/src/framework/domain/metadomain_io.cpp +++ b/src/framework/domain/metadomain_io.cpp @@ -960,7 +960,7 @@ namespace ntt { Kokkos::parallel_for( "GenerateEnergyBins", n_bins + 1, - Lambda(index_t e) { + Lambda(uint32_t e) { if (log_bins) { energy(e) = math::pow(10.0, e_min + (e_max - e_min) * e / n_bins); } else { @@ -996,7 +996,7 @@ namespace ntt { Kokkos::parallel_for( "ComputeSpectra", species.rangeActiveParticles(), - Lambda(index_t p) { + Lambda(prtlidx_t p) { if (tag(p) != ParticleTag::alive) { return; } From ad2f3db0fd8c14d14cd6f0d909f5ee623134e0f8 Mon Sep 17 00:00:00 2001 From: LudwigBoess Date: Fri, 7 Aug 2026 15:40:43 -0500 Subject: [PATCH 062/125] bugfix for loadbalancing when compiled with nvcc --- src/framework/domain/metadomain.h | 3 +- src/framework/domain/metadomain_loadbal.cpp | 8 +- src/kernels/pushers/sr_policies.h | 87 ++++++++++++++------- 3 files changed, 64 insertions(+), 34 deletions(-) diff --git a/src/framework/domain/metadomain.h b/src/framework/domain/metadomain.h index f090dade1..ed35a7140 100644 --- a/src/framework/domain/metadomain.h +++ b/src/framework/domain/metadomain.h @@ -146,8 +146,7 @@ namespace ntt { */ void Rebalance(unsigned int dim_mask, real_t tolerance, - ncells_t max_shift_cells) - requires(MetricClass); + ncells_t max_shift_cells); /* output-related ------------------------------------------------------- */ #if defined(OUTPUT_ENABLED) diff --git a/src/framework/domain/metadomain_loadbal.cpp b/src/framework/domain/metadomain_loadbal.cpp index 0768acad6..fdf167b67 100644 --- a/src/framework/domain/metadomain_loadbal.cpp +++ b/src/framework/domain/metadomain_loadbal.cpp @@ -110,9 +110,7 @@ namespace ntt { template void Metadomain::Rebalance(unsigned int dim_mask, real_t tolerance, - ncells_t max_shift_cells) - requires(MetricClass) - { + ncells_t max_shift_cells) { #if !defined(MPI_ENABLED) (void)dim_mask; (void)tolerance; @@ -377,6 +375,10 @@ namespace ntt { if (tag(p) != ParticleTag::alive) { return; } + // nvcc: force capture of all vars before any constexpr-if branch + (void)i1; (void)i1p; (void)i2; (void)i2p; (void)i3; (void)i3p; + (void)dx1; (void)dx2; (void)dx3; + (void)new_n1; (void)new_n2; (void)new_n3; if constexpr (M::Dim == Dim::_1D or M::Dim == Dim::_2D or M::Dim == Dim::_3D) { i1(p) += dx1; diff --git a/src/kernels/pushers/sr_policies.h b/src/kernels/pushers/sr_policies.h index b247b27b7..195fa380a 100644 --- a/src/kernels/pushers/sr_policies.h +++ b/src/kernels/pushers/sr_policies.h @@ -102,6 +102,57 @@ namespace kernel::sr { } } + // nvcc/nvc++ workaround: if constexpr inside generic lambdas does not reliably + // discard dead branches on some NVC++ versions. Factor the three optional-pgen + // dispatches into regular template functions so the discard happens at template + // instantiation, not inside a lambda operator(). + + template + void sr_dispatch_custom_emission(const PGen& pgen, + simtime_t time, + spidx_t sp, + DOM& dom, + Next&& next) { + if constexpr (::traits::pgen::HasEmissionPolicy) { + next(pgen.EmissionPolicy(time, sp, dom)); + } else { + raise::Error("Custom emission policy flag is set but problem " + "generator does not define an emission policy", + HERE); + } + } + + template + void sr_dispatch_cpu_policy(const PGen& pgen, + simtime_t time, + spidx_t sp, + DOM& dom, + Next&& next) { + if constexpr (::traits::pgen::HasCustomPrtlUpdate) { + next(pgen.CustomParticleUpdate(time, sp, dom)); + } else { + next(::traits::custom_prtl_update::NoPolicy_t {}); + } + } + + template + void sr_dispatch_extfields(const PGen& pgen, + simtime_t time, + spidx_t sp, + DOM& dom, + Next&& next) { + if constexpr (::traits::pgen::HasExternalFields) { + const auto [apply_extfields, external_fields] = pgen.ExternalFields(time, sp, dom); + if (apply_extfields) { + next(external_fields); + } else { + next(::traits::extfields::NoPolicy_t {}); + } + } else { + next(::traits::extfields::NoPolicy_t {}); + } + } + template void MakePusherPolicy(const PGen& pgen, DOM& domain, @@ -125,15 +176,11 @@ namespace kernel::sr { pusher_ctx)); break; case ntt::EmissionType::CUSTOM: - if constexpr (::traits::pgen::HasEmissionPolicy) { - next(pgen.EmissionPolicy(pusher_ctx.time, - pusher_ctx.species_index, - domain)); - } else { - raise::Error("Custom emission policy flag is set but problem " - "generator does not define an emission policy", - HERE); - } + sr_dispatch_custom_emission(pgen, + pusher_ctx.time, + pusher_ctx.species_index, + domain, + next); break; case ntt::EmissionType::NONE: default: @@ -143,29 +190,11 @@ namespace kernel::sr { }; auto with_custom_prtl_upd = [&](auto next) { - if constexpr (::traits::pgen::HasCustomPrtlUpdate) { - next(pgen.CustomParticleUpdate(pusher_ctx.time, - pusher_ctx.species_index, - domain)); - } else { - next(::traits::custom_prtl_update::NoPolicy_t {}); - } + sr_dispatch_cpu_policy(pgen, pusher_ctx.time, pusher_ctx.species_index, domain, next); }; auto with_ext_fields = [&](auto next) { - if constexpr (::traits::pgen::HasExternalFields) { - const auto [apply_extfields, external_fields] = pgen.ExternalFields( - pusher_ctx.time, - pusher_ctx.species_index, - domain); - if (apply_extfields) { - next(external_fields); - } else { - next(::traits::extfields::NoPolicy_t {}); - } - } else { - next(::traits::extfields::NoPolicy_t {}); - } + sr_dispatch_extfields(pgen, pusher_ctx.time, pusher_ctx.species_index, domain, next); }; with_emission([&](auto ep) { From 765784eb2410fbf768a853e396f6e0ce4efc6f8d Mon Sep 17 00:00:00 2001 From: LudwigBoess Date: Tue, 11 Aug 2026 21:23:00 +0000 Subject: [PATCH 063/125] replace the Rebalance particle-shift lambda with a ShiftPrtlIndices_kernel functor --- src/framework/domain/metadomain_loadbal.cpp | 135 +++++++++++++------- 1 file changed, 90 insertions(+), 45 deletions(-) diff --git a/src/framework/domain/metadomain_loadbal.cpp b/src/framework/domain/metadomain_loadbal.cpp index fdf167b67..a7bf97873 100644 --- a/src/framework/domain/metadomain_loadbal.cpp +++ b/src/framework/domain/metadomain_loadbal.cpp @@ -105,6 +105,81 @@ namespace ntt { } Kokkos::deep_copy(dst_dev, dst_h); } + + // Shift particle cell indices by dx_d after the local active-cell offset + // moved, and re-tag every particle whose new index falls outside the new + // active range [0, new_n_d) for the corresponding neighbor. + template + class ShiftPrtlIndices_kernel { + array_t i1, i1_prev, i2, i2_prev, i3, i3_prev; + array_t tag; + + const int dx1, dx2, dx3; + const int new_n1, new_n2, new_n3; + + public: + ShiftPrtlIndices_kernel(array_t& i1, + array_t& i1_prev, + array_t& i2, + array_t& i2_prev, + array_t& i3, + array_t& i3_prev, + array_t& tag, + int dx1, + int dx2, + int dx3, + int new_n1, + int new_n2, + int new_n3) + : i1 { i1 } + , i1_prev { i1_prev } + , i2 { i2 } + , i2_prev { i2_prev } + , i3 { i3 } + , i3_prev { i3_prev } + , tag { tag } + , dx1 { dx1 } + , dx2 { dx2 } + , dx3 { dx3 } + , new_n1 { new_n1 } + , new_n2 { new_n2 } + , new_n3 { new_n3 } {} + + Inline void operator()(prtlidx_t p) const { + if (tag(p) != ParticleTag::alive) { + return; + } + if constexpr (D == Dim::_1D or D == Dim::_2D or D == Dim::_3D) { + i1(p) += dx1; + i1_prev(p) += dx1; + } + if constexpr (D == Dim::_2D or D == Dim::_3D) { + i2(p) += dx2; + i2_prev(p) += dx2; + } + if constexpr (D == Dim::_3D) { + i3(p) += dx3; + i3_prev(p) += dx3; + } + if constexpr (D == Dim::_1D) { + tag(p) = mpi::SendTag(tag(p), i1(p) < 0, i1(p) >= new_n1); + } else if constexpr (D == Dim::_2D) { + tag(p) = mpi::SendTag(tag(p), + i1(p) < 0, + i1(p) >= new_n1, + i2(p) < 0, + i2(p) >= new_n2); + } else if constexpr (D == Dim::_3D) { + tag(p) = mpi::SendTag(tag(p), + i1(p) < 0, + i1(p) >= new_n1, + i2(p) < 0, + i2(p) >= new_n2, + i3(p) < 0, + i3(p) >= new_n3); + } + } + }; #endif // MPI_ENABLED template @@ -351,9 +426,6 @@ namespace ntt { if (sp.npart() == 0) { continue; } - auto i1 = sp.i1, i2 = sp.i2, i3 = sp.i3; - auto i1p = sp.i1_prev, i2p = sp.i2_prev, i3p = sp.i3_prev; - auto tag = sp.tag; const int dx1 = -delta[0]; int dx2 = 0; int dx3 = 0; @@ -368,48 +440,21 @@ namespace ntt { dx3 = -delta[2]; new_n3 = static_cast(new_local_ncells[2]); } - Kokkos::parallel_for( - "RebalanceShiftPrtls", - sp.rangeActiveParticles(), - Lambda(prtlidx_t p) { - if (tag(p) != ParticleTag::alive) { - return; - } - // nvcc: force capture of all vars before any constexpr-if branch - (void)i1; (void)i1p; (void)i2; (void)i2p; (void)i3; (void)i3p; - (void)dx1; (void)dx2; (void)dx3; - (void)new_n1; (void)new_n2; (void)new_n3; - if constexpr (M::Dim == Dim::_1D or M::Dim == Dim::_2D or - M::Dim == Dim::_3D) { - i1(p) += dx1; - i1p(p) += dx1; - } - if constexpr (M::Dim == Dim::_2D or M::Dim == Dim::_3D) { - i2(p) += dx2; - i2p(p) += dx2; - } - if constexpr (M::Dim == Dim::_3D) { - i3(p) += dx3; - i3p(p) += dx3; - } - if constexpr (M::Dim == Dim::_1D) { - tag(p) = mpi::SendTag(tag(p), i1(p) < 0, i1(p) >= new_n1); - } else if constexpr (M::Dim == Dim::_2D) { - tag(p) = mpi::SendTag(tag(p), - i1(p) < 0, - i1(p) >= new_n1, - i2(p) < 0, - i2(p) >= new_n2); - } else if constexpr (M::Dim == Dim::_3D) { - tag(p) = mpi::SendTag(tag(p), - i1(p) < 0, - i1(p) >= new_n1, - i2(p) < 0, - i2(p) >= new_n2, - i3(p) < 0, - i3(p) >= new_n3); - } - }); + Kokkos::parallel_for("RebalanceShiftPrtls", + sp.rangeActiveParticles(), + ShiftPrtlIndices_kernel { sp.i1, + sp.i1_prev, + sp.i2, + sp.i2_prev, + sp.i3, + sp.i3_prev, + sp.tag, + dx1, + dx2, + dx3, + new_n1, + new_n2, + new_n3 }); sp.set_unsorted(); } From 4be464a92689756b4acce734f2334c0021513663 Mon Sep 17 00:00:00 2001 From: Alisa Galishnikova <55898700+alisagk@users.noreply.github.com> Date: Mon, 17 Aug 2026 16:17:06 -0400 Subject: [PATCH 064/125] evaluate metric once for gr pusher all iterations --- src/kernels/pushers/gr.hpp | 65 +++++++++++++++++--------------------- 1 file changed, 29 insertions(+), 36 deletions(-) diff --git a/src/kernels/pushers/gr.hpp b/src/kernels/pushers/gr.hpp index ff3274bac..0a69b75a8 100644 --- a/src/kernels/pushers/gr.hpp +++ b/src/kernels/pushers/gr.hpp @@ -266,6 +266,21 @@ namespace kernel::gr { if constexpr (D == Dim::_1D) { raise::KernelError(HERE, "1D not applicable"); } else if constexpr (D == Dim::_2D) { + // evaluate the metric-derivative quantities once at `xp` + const real_t alpha_xp { metric.alpha(xp) }; + const real_t dr_alpha_xp { metric.dr_alpha(xp) }; + const real_t dt_alpha_xp { metric.dt_alpha(xp) }; + const real_t dr_beta1_xp { metric.dr_beta1(xp) }; + const real_t dt_beta1_xp { metric.dt_beta1(xp) }; + const real_t dr_h11_xp { metric.dr_h11(xp) }; + const real_t dr_h22_xp { metric.dr_h22(xp) }; + const real_t dr_h33_xp { metric.dr_h33(xp) }; + const real_t dr_h13_xp { metric.dr_h13(xp) }; + const real_t dt_h11_xp { metric.dt_h11(xp) }; + const real_t dt_h22_xp { metric.dt_h22(xp) }; + const real_t dt_h33_xp { metric.dt_h33(xp) }; + const real_t dt_h13_xp { metric.dt_h13(xp) }; + // initialize midpoint values & updated values vec_t vp_mid { ZERO }; vec_t vp_mid_cntrv { ZERO }; @@ -282,48 +297,26 @@ namespace kernel::gr { // find contravariant midpoint velocity metric.template transform(xp, vp_mid, vp_mid_cntrv); - // find Gamma / alpha at midpointы - real_t u0 { computeGamma(T {}, vp_mid, vp_mid_cntrv) / metric.alpha(xp) }; + // find Gamma / alpha at midpoint + real_t u0 { computeGamma(T {}, vp_mid, vp_mid_cntrv) / alpha_xp }; // find updated velocity - // vp_upd[0] = - // vp[0] + - // dt * - // (-metric.alpha(xp) * u0 * DERIVATIVE_IN_R(metric.alpha, xp) + - // vp_mid[0] * DERIVATIVE_IN_R(metric.beta1, xp) - - // (HALF / u0) * - // (DERIVATIVE_IN_R((metric.template h<1, 1>), xp) * SQR(vp_mid[0]) + - // DERIVATIVE_IN_R((metric.template h<2, 2>), xp) * SQR(vp_mid[1]) + - // DERIVATIVE_IN_R((metric.template h<3, 3>), xp) * SQR(vp_mid[2]) + - // TWO * DERIVATIVE_IN_R((metric.template h<1, 3>), xp) * - // vp_mid[0] * vp_mid[2])); - // vp_upd[1] = - // vp[1] + - // dt * - // (-metric.alpha(xp) * u0 * DERIVATIVE_IN_TH(metric.alpha, xp) + - // vp_mid[0] * DERIVATIVE_IN_TH(metric.beta1, xp) - - // (HALF / u0) * - // (DERIVATIVE_IN_TH((metric.template h<1, 1>), xp) * SQR(vp_mid[0]) + - // DERIVATIVE_IN_TH((metric.template h<2, 2>), xp) * SQR(vp_mid[1]) + - // DERIVATIVE_IN_TH((metric.template h<3, 3>), xp) * SQR(vp_mid[2]) + - // TWO * DERIVATIVE_IN_TH((metric.template h<1, 3>), xp) * - // vp_mid[0] * vp_mid[2])); vp_upd[0] = vp[0] + - ctx.dt * (-metric.alpha(xp) * u0 * metric.dr_alpha(xp) + - vp_mid[0] * metric.dr_beta1(xp) - + ctx.dt * (-alpha_xp * u0 * dr_alpha_xp + + vp_mid[0] * dr_beta1_xp - (HALF / u0) * - (metric.dr_h11(xp) * SQR(vp_mid[0]) + - metric.dr_h22(xp) * SQR(vp_mid[1]) + - metric.dr_h33(xp) * SQR(vp_mid[2]) + - TWO * metric.dr_h13(xp) * vp_mid[0] * vp_mid[2])); + (dr_h11_xp * SQR(vp_mid[0]) + + dr_h22_xp * SQR(vp_mid[1]) + + dr_h33_xp * SQR(vp_mid[2]) + + TWO * dr_h13_xp * vp_mid[0] * vp_mid[2])); vp_upd[1] = vp[1] + - ctx.dt * (-metric.alpha(xp) * u0 * metric.dt_alpha(xp) + - vp_mid[0] * metric.dt_beta1(xp) - + ctx.dt * (-alpha_xp * u0 * dt_alpha_xp + + vp_mid[0] * dt_beta1_xp - (HALF / u0) * - (metric.dt_h11(xp) * SQR(vp_mid[0]) + - metric.dt_h22(xp) * SQR(vp_mid[1]) + - metric.dt_h33(xp) * SQR(vp_mid[2]) + - TWO * metric.dt_h13(xp) * vp_mid[0] * vp_mid[2])); + (dt_h11_xp * SQR(vp_mid[0]) + + dt_h22_xp * SQR(vp_mid[1]) + + dt_h33_xp * SQR(vp_mid[2]) + + TWO * dt_h13_xp * vp_mid[0] * vp_mid[2])); } } else if constexpr (D == Dim::_3D) { raise::KernelNotImplementedError(HERE); From b932d8bbe7da7b431f71cb963a43719fbc818282 Mon Sep 17 00:00:00 2001 From: Alisa Galishnikova <55898700+alisagk@users.noreply.github.com> Date: Mon, 17 Aug 2026 18:40:53 -0400 Subject: [PATCH 065/125] explicit tetrad transform in gr pusher --- src/kernels/pushers/gr.hpp | 62 +++++++++++++++++++++++++++++++------- 1 file changed, 51 insertions(+), 11 deletions(-) diff --git a/src/kernels/pushers/gr.hpp b/src/kernels/pushers/gr.hpp index 0a69b75a8..f8447c125 100644 --- a/src/kernels/pushers/gr.hpp +++ b/src/kernels/pushers/gr.hpp @@ -168,13 +168,18 @@ namespace kernel::gr { /** * @brief EM pusher (Boris) substep. - * @param xp coordinate of the particle. + * @param alpha_xp lapse function at the particle position. + * @param h11_xp h^11(xp) (contravariant); h13_xp, h22_xp, h33_xp = h_13(xp), h_22(xp), h_33(xp) (covariant). * @param vp covariant velocity of the particle. * @param Dp_hat hatted electric field at the particle position. * @param Bp_hat hatted magnetic field at the particle position. * @param v_upd updated covarient velocity of the particle [return]. */ - Inline void EMHalfPush(const coord_t& xp, + Inline void EMHalfPush(real_t alpha_xp, + real_t h11_xp, + real_t h13_xp, + real_t h22_xp, + real_t h33_xp, const vec_t& vp, const vec_t& Dp_hat, const vec_t& Bp_hat, @@ -182,10 +187,16 @@ namespace kernel::gr { vec_t D0 { Dp_hat[0], Dp_hat[1], Dp_hat[2] }; vec_t B0 { Bp_hat[0], Bp_hat[1], Bp_hat[2] }; vec_t vp_hat { ZERO }, vp_upd_hat { ZERO }; - metric.template transform(xp, vp, vp_upd_hat); + // == transform(xp, vp, vp_upd_hat) + const real_t A0 { math::sqrt(h11_xp) }; + const real_t sqrt_h22_xp { math::sqrt(h22_xp) }; + const real_t sqrt_h33_xp { math::sqrt(h33_xp) }; + vp_upd_hat[0] = vp[0] * A0 - vp[2] * A0 * h13_xp / h33_xp; + vp_upd_hat[1] = vp[1] / sqrt_h22_xp; + vp_upd_hat[2] = vp[2] / sqrt_h33_xp; // this is a half-push - real_t COEFF { normalized_dt_half * HALF * metric.alpha(xp) }; + real_t COEFF { normalized_dt_half * HALF * alpha_xp }; D0[0] *= COEFF; D0[1] *= COEFF; @@ -213,7 +224,10 @@ namespace kernel::gr { vp_upd_hat[1] += vp_hat[2] * B0[0] - vp_hat[0] * B0[2] + D0[1]; vp_upd_hat[2] += vp_hat[0] * B0[1] - vp_hat[1] * B0[0] + D0[2]; - metric.template transform(xp, vp_upd_hat, vp_upd); + // == transform(xp, vp_upd_hat, vp_upd) + vp_upd[0] = vp_upd_hat[0] / A0 + vp_upd_hat[2] * h13_xp / sqrt_h33_xp; + vp_upd[1] = vp_upd_hat[1] * sqrt_h22_xp; + vp_upd[2] = vp_upd_hat[2] * sqrt_h33_xp; } // Helper functions @@ -267,6 +281,11 @@ namespace kernel::gr { raise::KernelError(HERE, "1D not applicable"); } else if constexpr (D == Dim::_2D) { // evaluate the metric-derivative quantities once at `xp` + const real_t h11_xp { metric.template h<1, 1>(xp) }; + const real_t h13_xp { metric.template h<1, 3>(xp) }; + const real_t h22_xp { metric.template h<2, 2>(xp) }; + const real_t h33_xp { metric.template h<3, 3>(xp) }; + const real_t alpha_xp { metric.alpha(xp) }; const real_t dr_alpha_xp { metric.dr_alpha(xp) }; const real_t dt_alpha_xp { metric.dt_alpha(xp) }; @@ -294,8 +313,10 @@ namespace kernel::gr { vp_mid[1] = HALF * (vp[1] + vp_upd[1]); vp_mid[2] = vp[2]; - // find contravariant midpoint velocity - metric.template transform(xp, vp_mid, vp_mid_cntrv); + // find contravariant midpoint velocity (== transform(xp, vp_mid, ...)) + vp_mid_cntrv[0] = vp_mid[0] * h11_xp + vp_mid[2] * h13_xp; + vp_mid_cntrv[1] = vp_mid[1] * h22_xp; + vp_mid_cntrv[2] = vp_mid[0] * h13_xp + vp_mid[2] * h33_xp; // find Gamma / alpha at midpoint real_t u0 { computeGamma(T {}, vp_mid, vp_mid_cntrv) / alpha_xp }; @@ -736,18 +757,37 @@ namespace kernel::gr { } xp_[1] = theta_Cd; + // h^11/h_13/h_22/h_33(xp) and alpha(xp) are fixed for this whole + // substep (only used at `xp`, never at `xp_`) -- evaluate once and + // reuse across interpolateFields' U->T transform and both EMHalfPush + // calls instead of re-deriving them from scratch at each of those 4 + // call sites + const real_t h11_xp { metric.template h<1, 1>(xp) }; + const real_t h13_xp { metric.template h_<1, 3>(xp) }; + const real_t h22_xp { metric.template h_<2, 2>(xp) }; + const real_t h33_xp { metric.template h_<3, 3>(xp) }; + const real_t alpha_xp { metric.alpha(xp) }; + const real_t sqrt_h11_xp { math::sqrt(h11_xp) }; + const real_t sqrt_h22_xp { math::sqrt(h22_xp) }; + const real_t sqrt_h33_xp { math::sqrt(h33_xp) }; + vec_t Dp_cntrv { ZERO }, Bp_cntrv { ZERO }, Dp_hat { ZERO }, Bp_hat { ZERO }; interpolateFields(p, Dp_cntrv, Bp_cntrv); - metric.template transform(xp, Dp_cntrv, Dp_hat); - metric.template transform(xp, Bp_cntrv, Bp_hat); + // == transform(xp, Dp_cntrv, Dp_hat) / (xp, Bp_cntrv, Bp_hat) + Dp_hat[0] = Dp_cntrv[0] / sqrt_h11_xp; + Dp_hat[1] = Dp_cntrv[1] * sqrt_h22_xp; + Dp_hat[2] = Dp_cntrv[2] * sqrt_h33_xp + Dp_cntrv[0] * h13_xp / sqrt_h33_xp; + Bp_hat[0] = Bp_cntrv[0] / sqrt_h11_xp; + Bp_hat[1] = Bp_cntrv[1] * sqrt_h22_xp; + Bp_hat[2] = Bp_cntrv[2] * sqrt_h33_xp + Bp_cntrv[0] * h13_xp / sqrt_h33_xp; vec_t vp { particles.ux1(p), particles.ux2(p), particles.ux3(p) }; /* -------------------------------- Leapfrog -------------------------------- */ /* u_i(n - 1/2) -> u*_i(n) */ vec_t vp_upd { ZERO }; - EMHalfPush(xp, vp, Dp_hat, Bp_hat, vp_upd); + EMHalfPush(alpha_xp, h11_xp, h13_xp, h22_xp, h33_xp, vp, Dp_hat, Bp_hat, vp_upd); /* u*_i(n) -> u**_i(n) */ vp[0] = vp_upd[0]; vp[1] = vp_upd[1]; @@ -757,7 +797,7 @@ namespace kernel::gr { vp[0] = vp_upd[0]; vp[1] = vp_upd[1]; vp[2] = vp_upd[2]; - EMHalfPush(xp, vp, Dp_hat, Bp_hat, vp_upd); + EMHalfPush(alpha_xp, h11_xp, h13_xp, h22_xp, h33_xp, vp, Dp_hat, Bp_hat, vp_upd); /* x^i(n) -> x^i(n + 1) */ coord_t xp_upd { ZERO }; GeodesicCoordinatePush(Massive_t {}, xp, vp_upd, xp_upd); From 3349f5c7053ca93b127234aa810a2871e1651e17 Mon Sep 17 00:00:00 2001 From: LudwigBoess Date: Fri, 21 Aug 2026 18:28:12 +0000 Subject: [PATCH 066/125] pass is_left through to CrossesPiston so a right-facing wall can reuse it --- src/archetypes/piston.h | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/src/archetypes/piston.h b/src/archetypes/piston.h index 66f7160f2..d9d81bb14 100644 --- a/src/archetypes/piston.h +++ b/src/archetypes/piston.h @@ -80,6 +80,7 @@ namespace arch { * @param piston_position Position of the piston at the start of timestep in global coordinates * @param piston_v Velocity of piston at current timestep * @param massive Whether the particle is massive or massless (e.g. photon) + * @param is_left Is piston on the left side of the box or right side of the box */ template Inline void Piston(prtlidx_t p, @@ -88,10 +89,11 @@ namespace arch { const M& metric, real_t piston_position, real_t piston_v, - bool massive) { + bool massive, + bool is_left = true) { // check if particle actually crosses the piston, if not return - if (!CrossesPiston(p, dt, particles, metric, piston_position, piston_v, true)) { + if (!CrossesPiston(p, dt, particles, metric, piston_position, piston_v, is_left)) { return; } // step 1: calculate the particle 3 velocity From 523c40135eead2e70fc322da8f7e021088b1a26a Mon Sep 17 00:00:00 2001 From: LudwigBoess Date: Fri, 21 Aug 2026 19:04:43 +0000 Subject: [PATCH 067/125] bugfix in conductor BCs --- src/kernels/fields_bcs.hpp | 48 ++++++++++++++------------------------ 1 file changed, 18 insertions(+), 30 deletions(-) diff --git a/src/kernels/fields_bcs.hpp b/src/kernels/fields_bcs.hpp index 729a09d10..c831fce4b 100644 --- a/src/kernels/fields_bcs.hpp +++ b/src/kernels/fields_bcs.hpp @@ -554,15 +554,13 @@ namespace kernel::bc { } if (tags & BC::B) { - if (i1 == 0) { - Fld(i_edge, em::bx1) = ZERO; - } else { + if (i1 != 0) { if constexpr (not P) { - Fld(i_edge - i1, em::bx1) = -Fld(i_edge + i1, em::bx1); + Fld(i_edge - i1, em::bx1) = Fld(i_edge + i1, em::bx1); Fld(i_edge - i1, em::bx2) = Fld(i_edge + i1 - 1, em::bx2); Fld(i_edge - i1, em::bx3) = Fld(i_edge + i1 - 1, em::bx3); } else { - Fld(i_edge + i1, em::bx1) = -Fld(i_edge - i1, em::bx1); + Fld(i_edge + i1, em::bx1) = Fld(i_edge - i1, em::bx1); Fld(i_edge + i1 - 1, em::bx2) = Fld(i_edge - i1, em::bx2); Fld(i_edge + i1 - 1, em::bx3) = Fld(i_edge - i1, em::bx3); } @@ -596,15 +594,13 @@ namespace kernel::bc { } if (tags & BC::B) { - if (i1 == 0) { - Fld(i_edge, i2, em::bx1) = ZERO; - } else { + if (i1 != 0) { if constexpr (not P) { - Fld(i_edge - i1, i2, em::bx1) = -Fld(i_edge + i1, i2, em::bx1); + Fld(i_edge - i1, i2, em::bx1) = Fld(i_edge + i1, i2, em::bx1); Fld(i_edge - i1, i2, em::bx2) = Fld(i_edge + i1 - 1, i2, em::bx2); Fld(i_edge - i1, i2, em::bx3) = Fld(i_edge + i1 - 1, i2, em::bx3); } else { - Fld(i_edge + i1, i2, em::bx1) = -Fld(i_edge - i1, i2, em::bx1); + Fld(i_edge + i1, i2, em::bx1) = Fld(i_edge - i1, i2, em::bx1); Fld(i_edge + i1 - 1, i2, em::bx2) = Fld(i_edge - i1, i2, em::bx2); Fld(i_edge + i1 - 1, i2, em::bx3) = Fld(i_edge - i1, i2, em::bx3); } @@ -629,16 +625,14 @@ namespace kernel::bc { } if (tags & BC::B) { - if (i2 == 0) { - Fld(i1, i_edge, em::bx2) = ZERO; - } else { + if (i2 != 0) { if constexpr (not P) { Fld(i1, i_edge - i2, em::bx1) = Fld(i1, i_edge + i2 - 1, em::bx1); - Fld(i1, i_edge - i2, em::bx2) = -Fld(i1, i_edge + i2, em::bx2); + Fld(i1, i_edge - i2, em::bx2) = Fld(i1, i_edge + i2, em::bx2); Fld(i1, i_edge - i2, em::bx3) = Fld(i1, i_edge + i2 - 1, em::bx3); } else { Fld(i1, i_edge + i2 - 1, em::bx1) = Fld(i1, i_edge - i2, em::bx1); - Fld(i1, i_edge + i2, em::bx2) = -Fld(i1, i_edge - i2, em::bx2); + Fld(i1, i_edge + i2, em::bx2) = Fld(i1, i_edge - i2, em::bx2); Fld(i1, i_edge + i2 - 1, em::bx3) = Fld(i1, i_edge - i2, em::bx3); } } @@ -678,11 +672,9 @@ namespace kernel::bc { } if (tags & BC::B) { - if (i1 == 0) { - Fld(i_edge, i2, i3, em::bx1) = ZERO; - } else { + if (i1 != 0) { if constexpr (not P) { - Fld(i_edge - i1, i2, i3, em::bx1) = -Fld(i_edge + i1, i2, i3, em::bx1); + Fld(i_edge - i1, i2, i3, em::bx1) = Fld(i_edge + i1, i2, i3, em::bx1); Fld(i_edge - i1, i2, i3, em::bx2) = Fld(i_edge + i1 - 1, i2, i3, @@ -692,7 +684,7 @@ namespace kernel::bc { i3, em::bx3); } else { - Fld(i_edge + i1, i2, i3, em::bx1) = -Fld(i_edge - i1, i2, i3, em::bx1); + Fld(i_edge + i1, i2, i3, em::bx1) = Fld(i_edge - i1, i2, i3, em::bx1); Fld(i_edge + i1 - 1, i2, i3, em::bx2) = Fld(i_edge - i1, i2, i3, @@ -729,15 +721,13 @@ namespace kernel::bc { } if (tags & BC::B) { - if (i2 == 0) { - Fld(i1, i_edge, i3, em::bx2) = ZERO; - } else { + if (i2 != 0) { if constexpr (not P) { Fld(i1, i_edge - i2, i3, em::bx1) = Fld(i1, i_edge + i2 - 1, i3, em::bx1); - Fld(i1, i_edge - i2, i3, em::bx2) = -Fld(i1, i_edge + i2, i3, em::bx2); + Fld(i1, i_edge - i2, i3, em::bx2) = Fld(i1, i_edge + i2, i3, em::bx2); Fld(i1, i_edge - i2, i3, em::bx3) = Fld(i1, i_edge + i2 - 1, i3, @@ -747,7 +737,7 @@ namespace kernel::bc { i_edge - i2, i3, em::bx1); - Fld(i1, i_edge + i2, i3, em::bx2) = -Fld(i1, i_edge - i2, i3, em::bx2); + Fld(i1, i_edge + i2, i3, em::bx2) = Fld(i1, i_edge - i2, i3, em::bx2); Fld(i1, i_edge + i2 - 1, i3, em::bx3) = Fld(i1, i_edge - i2, i3, @@ -780,9 +770,7 @@ namespace kernel::bc { } if (tags & BC::B) { - if (i3 == 0) { - Fld(i1, i2, i_edge, em::bx3) = ZERO; - } else { + if (i3 != 0) { if constexpr (not P) { Fld(i1, i2, i_edge - i3, em::bx1) = Fld(i1, i2, @@ -792,7 +780,7 @@ namespace kernel::bc { i2, i_edge + i3 - 1, em::bx2); - Fld(i1, i2, i_edge - i3, em::bx3) = -Fld(i1, i2, i_edge + i3, em::bx3); + Fld(i1, i2, i_edge - i3, em::bx3) = Fld(i1, i2, i_edge + i3, em::bx3); } else { Fld(i1, i2, i_edge + i3 - 1, em::bx1) = Fld(i1, i2, @@ -802,7 +790,7 @@ namespace kernel::bc { i2, i_edge - i3, em::bx2); - Fld(i1, i2, i_edge + i3, em::bx3) = -Fld(i1, i2, i_edge - i3, em::bx3); + Fld(i1, i2, i_edge + i3, em::bx3) = Fld(i1, i2, i_edge - i3, em::bx3); } } } From 1aa6ee9e9c9d74c3fe9602d1ce10de44126d40d6 Mon Sep 17 00:00:00 2001 From: LudwigBoess Date: Fri, 21 Aug 2026 19:28:06 +0000 Subject: [PATCH 068/125] Turn inline Lambda into proper kernel to avoid NVidia compilation error --- src/archetypes/moving_window.h | 83 +++++++++++++++++++++++----------- 1 file changed, 57 insertions(+), 26 deletions(-) diff --git a/src/archetypes/moving_window.h b/src/archetypes/moving_window.h index b8dc74fa2..3b9faca6f 100644 --- a/src/archetypes/moving_window.h +++ b/src/archetypes/moving_window.h @@ -134,6 +134,54 @@ namespace arch { } }; +#if defined(MPI_ENABLED) + /** + * @brief Assigns MPI send tags from the particle cell indices. + * @note Particles outside the active domain get the tag of the neighbor they belong to. + */ + template + struct PrtlRetag_kernel { + array_t i1, i2, i3; + array_t tag; + const int ni1, ni2, ni3; + + PrtlRetag_kernel(const array_t& i1, + const array_t& i2, + const array_t& i3, + const array_t& tag, + int ni1, + int ni2, + int ni3) + : i1 { i1 } + , i2 { i2 } + , i3 { i3 } + , tag { tag } + , ni1 { ni1 } + , ni2 { ni2 } + , ni3 { ni3 } {} + + Inline void operator()(prtlidx_t p) const { + if constexpr (D == Dim::_1D) { + tag(p) = mpi::SendTag(tag(p), i1(p) < 0, i1(p) >= ni1); + } else if constexpr (D == Dim::_2D) { + tag(p) = mpi::SendTag(tag(p), + i1(p) < 0, + i1(p) >= ni1, + i2(p) < 0, + i2(p) >= ni2); + } else if constexpr (D == Dim::_3D) { + tag(p) = mpi::SendTag(tag(p), + i1(p) < 0, + i1(p) >= ni1, + i2(p) < 0, + i2(p) >= ni2, + i3(p) < 0, + i3(p) >= ni3); + } + } + }; +#endif // MPI_ENABLED + /** * @brief Updates particle position and fields in the moving window. @@ -213,32 +261,15 @@ namespace arch { : 0; for (auto s { 0u }; s < nspec; ++s) { auto& species = domain.species[s]; - auto i1 = species.i1; - auto i2 = species.i2; - auto i3 = species.i3; - auto tag = species.tag; - Kokkos::parallel_for( - "RetagWindowParticles", - species.rangeActiveParticles(), - Lambda(prtlidx_t p) { - if constexpr (M::Dim == Dim::_1D) { - tag(p) = mpi::SendTag(tag(p), i1(p) < 0, i1(p) >= ni1); - } else if constexpr (M::Dim == Dim::_2D) { - tag(p) = mpi::SendTag(tag(p), - i1(p) < 0, - i1(p) >= ni1, - i2(p) < 0, - i2(p) >= ni2); - } else if constexpr (M::Dim == Dim::_3D) { - tag(p) = mpi::SendTag(tag(p), - i1(p) < 0, - i1(p) >= ni1, - i2(p) < 0, - i2(p) >= ni2, - i3(p) < 0, - i3(p) >= ni3); - } - }); + Kokkos::parallel_for("RetagWindowParticles", + species.rangeActiveParticles(), + PrtlRetag_kernel(species.i1, + species.i2, + species.i3, + species.tag, + ni1, + ni2, + ni3)); } #endif From 2b34ab1852458b24f687fbc21bde951a6bd45156 Mon Sep 17 00:00:00 2001 From: haykh Date: Tue, 8 Sep 2026 20:52:40 -0400 Subject: [PATCH 069/125] new more categorized report --- cmake/report.cmake | 163 ++++++++++++++++++++++++--------------------- 1 file changed, 86 insertions(+), 77 deletions(-) diff --git a/cmake/report.cmake b/cmake/report.cmake index d972fafc4..822d82650 100644 --- a/cmake/report.cmake +++ b/cmake/report.cmake @@ -83,7 +83,7 @@ printchoices( ${default_deposit} "${Blue}" DEPOSIT_REPORT - 46) + 44) printchoices( "Shape order" "shape_order" @@ -110,18 +110,16 @@ printchoices( OFF "${Green}" MPI_REPORT - 46) -if(${mpi} AND ${DEVICE_ENABLED}) - printchoices( - "GPU-aware MPI" - "gpu_aware_mpi" - "${ON_OFF_VALUES}" - ${gpu_aware_mpi} - OFF - "${Green}" - GPU_AWARE_MPI_REPORT - 46) -endif() + 44) +printchoices( + "GPU-aware MPI" + "gpu_aware_mpi" + "${ON_OFF_VALUES}" + ${gpu_aware_mpi} + OFF + "${Green}" + GPU_AWARE_MPI_REPORT + 44) printchoices( "Team Policy" "team_policy" @@ -130,36 +128,25 @@ printchoices( OFF "${Green}" TEAM_POLICY_REPORT - 46) -if(${team_policy}) - printchoices( - "Team Tile Size" - "team_policy_tile_size" - "${team_policy_tile_sizes}" - ${team_policy_tile_size} - ${default_team_policy_tile_size} - "${Blue}" - TEAM_POLICY_TILE_SIZE_REPORT - 46) - printchoices( - "Team Deposit Drift" - "team_policy_drift" - "${team_policy_drift}" - ${team_policy_drift} - 1 - "${Blue}" - TEAM_POLICY_DRIFT_REPORT - 46) - printchoices( - "Vendor sort" - "vendor_sort" - "${ON_OFF_VALUES}" - ${vendor_sort} - ON - "${Green}" - VENDOR_SORT_REPORT - 46) -endif() + 44) +printchoices( + "Tile Size" + "team_policy_tile_size" + "${team_policy_tile_sizes}" + ${team_policy_tile_size} + ${default_team_policy_tile_size} + "${Blue}" + TEAM_POLICY_TILE_SIZE_REPORT + 44) +printchoices( + "Vendor sort" + "vendor_sort" + "${ON_OFF_VALUES}" + ${vendor_sort} + ON + "${Green}" + VENDOR_SORT_REPORT + 44) printchoices( "Debug mode" "DEBUG" @@ -196,69 +183,91 @@ string(APPEND REPORT_TEXT ${DASHED_LINE_SYMBOL} "\n" "Configurations" "\n") if(${PGEN_FOUND}) string(APPEND REPORT_TEXT " " ${PGEN_REPORT} "\n") +else() + string( + APPEND + REPORT_TEXT + " - Problem generator [${Magenta}pgen${ColorReset}]: ${Dim}none${ColorReset}\n" + ) endif() -string(APPEND REPORT_TEXT " " ${TESTS_REPORT} "\n") + +string(REPLACE ";" "+" Kokkos_ARCH "${Kokkos_ARCH}") +string(REPLACE ";" "+" Kokkos_DEVICES "${Kokkos_DEVICES}") string( APPEND REPORT_TEXT " " - ${PRECISION_REPORT} + ${TESTS_REPORT} "\n" " " - ${DEPOSIT_REPORT} + ${OUTPUT_REPORT} + "\n" + " - Install prefix [${Magenta}CMAKE_INSTALL_PREFIX${ColorReset}]: " + "${CMAKE_INSTALL_PREFIX}" + "\n" + ${DASHED_LINE_SYMBOL} + "\n" + "Algorithmic specs" + "\n" + " " + ${PRECISION_REPORT} "\n" " " ${SHAPEFUNCTION_REPORT} "\n" + " > PIC-specific specs" + "\n" + " " + ${DEPOSIT_REPORT} + "\n" + ${DASHED_LINE_SYMBOL} + "\n" + "Performance specs" + "\n" " " - ${OUTPUT_REPORT} - "\n") - -string(REPLACE ";" "+" Kokkos_ARCH "${Kokkos_ARCH}") -string(REPLACE ";" "+" Kokkos_DEVICES "${Kokkos_DEVICES}") - -string( - APPEND - REPORT_TEXT + ${DEBUG_REPORT} + "\n" " - ARCH [${Magenta}Kokkos_ARCH_***${ColorReset}]: " "${Kokkos_ARCH}" "\n" " - DEVICES [${Magenta}Kokkos_ENABLE_***${ColorReset}]: " "${Kokkos_DEVICES}" "\n" - " " + " > Multi-node specs" + " ${Dim}[requires mpi=ON]${ColorReset}" + "\n" + " " ${MPI_REPORT} + "\n" + " " + ${GPU_AWARE_MPI_REPORT} + "\n" + " > Team-policy specs" + " ${Dim}[requires team_policy=ON]${ColorReset}" + "\n" + " " + ${TEAM_POLICY_REPORT} + "\n" + " " + ${TEAM_POLICY_TILE_SIZE_REPORT} + "\n" + " " + "- Deposit drift [${Magenta}team_policy_drift${ColorReset}]: " + ${team_policy_drift} + "\n" + " " + ${VENDOR_SORT_REPORT} "\n") -if(${mpi} AND ${DEVICE_ENABLED}) - string(APPEND REPORT_TEXT " " ${GPU_AWARE_MPI_REPORT} "\n") -endif() - -string(APPEND REPORT_TEXT " " ${TEAM_POLICY_REPORT} "\n") -if(${team_policy}) - string(APPEND REPORT_TEXT " " ${TEAM_POLICY_TILE_SIZE_REPORT} "\n") - string(APPEND REPORT_TEXT " " ${TEAM_POLICY_DRIFT_REPORT} "\n") - string(APPEND REPORT_TEXT " " ${VENDOR_SORT_REPORT} "\n") -endif() - string( APPEND REPORT_TEXT " " - ${DEBUG_REPORT} - "\n" - " - Install prefix [${Magenta}CMAKE_INSTALL_PREFIX${ColorReset}]: " - "${CMAKE_INSTALL_PREFIX}" - "\n" ${DASHED_LINE_SYMBOL} "\n" "Compilers & dependencies" - "\n") - -string( - APPEND - REPORT_TEXT + "\n" " - C compiler [${Magenta}CMAKE_C_COMPILER${ColorReset}]: v" ${CMAKE_C_COMPILER_VERSION} "\n" From 7dd38b7145d4b987fd3af3aafb95a768745f66e7 Mon Sep 17 00:00:00 2001 From: haykh Date: Tue, 8 Sep 2026 22:35:28 -0400 Subject: [PATCH 070/125] refactor deposit into separate files --- ideal_tile_size.py | 2 +- src/engines/grpic/currents.h | 45 +- src/engines/srpic/currents.h | 41 +- src/kernels/currents_deposit.hpp | 1246 ----------------- src/kernels/deposition/currents/global.hpp | 102 ++ .../deposition/currents/single-particle.hpp | 676 +++++++++ src/kernels/deposition/currents/tiled.hpp | 523 +++++++ tests/kernels/deposit.cpp | 2 +- tests/kernels/deposit_tiled.cpp | 210 +-- tests/kernels/particle_moments.cpp | 2 - 10 files changed, 1476 insertions(+), 1373 deletions(-) delete mode 100644 src/kernels/currents_deposit.hpp create mode 100644 src/kernels/deposition/currents/global.hpp create mode 100644 src/kernels/deposition/currents/single-particle.hpp create mode 100644 src/kernels/deposition/currents/tiled.hpp diff --git a/ideal_tile_size.py b/ideal_tile_size.py index e216f5981..d9e9d32ea 100644 --- a/ideal_tile_size.py +++ b/ideal_tile_size.py @@ -9,7 +9,7 @@ HALO = stencil_reach + drift (stencil_reach = shape_order for Esirkepov, 2 for the O==0 zigzag deposit; drift = the compile-time `team_policy_drift` CMake knob, NOT the runtime - spatial_sorting_interval -- see currents_deposit.hpp) + spatial_sorting_interval -- see kernels/deposition/currents/tiled.hpp) the tile size is squeezed by three competing pressures: diff --git a/src/engines/grpic/currents.h b/src/engines/grpic/currents.h index cb4032f54..15d4d94ca 100644 --- a/src/engines/grpic/currents.h +++ b/src/engines/grpic/currents.h @@ -25,7 +25,8 @@ #include "framework/domain/domain.h" #include "framework/domain/metadomain.h" #include "framework/parameters/parameters.h" -#include "kernels/currents_deposit.hpp" +#include "kernels/deposition/currents/global.hpp" +#include "kernels/deposition/currents/tiled.hpp" #include "kernels/digital_filter.hpp" namespace ntt { @@ -50,21 +51,21 @@ namespace ntt { /** * @brief Tiled deposit launcher (TeamPolicy + per-team scratch). * - * Identical in structure to the SRPIC launcher (`engines/srpic/currents.h`): - * iterates over `tile_layout.ntiles_total` teams; each team accumulates its - * tile's particle contributions in SLM scratch and atomically flushes to the - * global J (here `cur0`, the GRPIC half-step current). Requires the species - * to have been sorted with `team_policy` enabled (`tile_layout` populated by - * `SortSpatially`). + * Identical in structure to the SRPIC launcher + * (`engines/srpic/currents.h`): iterates over `tile_layout.ntiles_total` + * teams; each team accumulates its tile's particle contributions in SLM + * scratch and atomically flushes to the global J (here `cur0`, the GRPIC + * half-step current). Requires the species to have been sorted with + * `team_policy` enabled (`tile_layout` populated by `SortSpatially`). * - * The deposit body (`kernel::DepositOneParticle`) is - * the same shared math used by the flat path — it already carries the GR + * The deposit body (`kernel::DepositOneParticle`) + * is the same shared math used by the flat path — it already carries the GR * velocity-recovery branch — so the only engine-specific differences from * SRPIC are the `SimEngine::GRPIC` tag and the `cur0` target. * * Falls back to the flat kernel for the tail `[npart_partitioned, npart)` * exactly as SRPIC does; see the per-step coverage note in - * `kernels/currents_deposit.hpp`. + * `kernels/deposition/currents/tiled.hpp`. */ template void CallDepositKernelTiled(const Particles& species, @@ -87,8 +88,8 @@ namespace ntt { auto deposit_kernel = kernel::DepositCurrentsTiled_kernel { - cur, species, local_metric, (real_t)(species.charge()), - dt, layout, species.npart() + cur, species, local_metric, (real_t)(species.charge()), + dt, layout, species.npart() }; const auto scratch = Kokkos::PerTeam( @@ -111,20 +112,20 @@ namespace ntt { int ts = team_size_req; if (ts > ts_max) { raise::Warning( - fmt::format("algorithms.deposit.team_policy_team_size = %d exceeds " - "the tiled-deposit maximum %d on this backend; clamping " - "to %d", - team_size_req, - ts_max, - ts_max), + fmt::format( + "algorithms.deposit.team_policy_team_size = %d exceeds " + "the tiled-deposit maximum %d on this backend; clamping " + "to %d", + team_size_req, + ts_max, + ts_max), HERE); ts = ts_max; } policy = Kokkos::TeamPolicy<>(static_cast(layout.ntiles_total), ts); policy.set_scratch_size(0, scratch); - logger::Checkpoint( - fmt::format("Tiled deposit: explicit team size %d", ts), - HERE); + logger::Checkpoint(fmt::format("Tiled deposit: explicit team size %d", ts), + HERE); } Kokkos::parallel_for("CurrentsDepositTiled", policy, deposit_kernel); @@ -177,7 +178,7 @@ namespace ntt { // only case the tiled kernel cannot serve is the very first step, before // any SortSpatially has populated a layout; that species takes the flat // scatter-view path for that step alone. See engines/srpic/currents.h and - // kernels/currents_deposit.hpp for the full coverage argument. + // kernels/deposition/currents/tiled.hpp for the full coverage argument. for (auto& species : domain.species) { if ((species.pusher() == ParticlePusher::NONE) or (species.npart() == 0) or cmp::AlmostZero_host(species.charge())) { diff --git a/src/engines/srpic/currents.h b/src/engines/srpic/currents.h index 74946ed68..2a23fde69 100644 --- a/src/engines/srpic/currents.h +++ b/src/engines/srpic/currents.h @@ -24,7 +24,8 @@ #include "engines/srpic/utils.h" #include "framework/domain/domain.h" #include "framework/domain/metadomain.h" -#include "kernels/currents_deposit.hpp" +#include "kernels/deposition/currents/global.hpp" +#include "kernels/deposition/currents/tiled.hpp" #include "kernels/digital_filter.hpp" #include @@ -83,8 +84,8 @@ namespace ntt { auto deposit_kernel = kernel::DepositCurrentsTiled_kernel { - cur, species, local_metric, (real_t)(species.charge()), - dt, layout, species.npart() + cur, species, local_metric, (real_t)(species.charge()), + dt, layout, species.npart() }; const auto scratch = Kokkos::PerTeam( @@ -107,20 +108,20 @@ namespace ntt { int ts = team_size_req; if (ts > ts_max) { raise::Warning( - fmt::format("algorithms.deposit.team_policy_team_size = %d exceeds " - "the tiled-deposit maximum %d on this backend; clamping " - "to %d", - team_size_req, - ts_max, - ts_max), + fmt::format( + "algorithms.deposit.team_policy_team_size = %d exceeds " + "the tiled-deposit maximum %d on this backend; clamping " + "to %d", + team_size_req, + ts_max, + ts_max), HERE); ts = ts_max; } policy = Kokkos::TeamPolicy<>(static_cast(layout.ntiles_total), ts); policy.set_scratch_size(0, scratch); - logger::Checkpoint( - fmt::format("Tiled deposit: explicit team size %d", ts), - HERE); + logger::Checkpoint(fmt::format("Tiled deposit: explicit team size %d", ts), + HERE); } Kokkos::parallel_for("CurrentsDepositTiled", policy, deposit_kernel); @@ -170,7 +171,7 @@ namespace ntt { // - a particle whose full stencil has drifted out of its tile is // deposited straight to the global J view (the per-particle escape // valve); `team_policy_drift` sizes the scratch halo so the - // common in-tile case stays in fast SLM (see currents_deposit.hpp); + // common in-tile case stays in fast SLM (see kernels/deposition/currents/tiled.hpp); // - particles dead-tagged in place since the sort are clamped out by // the kernel and skipped by the dead-tag test; // - particles appended past the partition since the sort (injection / @@ -296,7 +297,7 @@ namespace ntt { // to the x2 upper bound when that side is AXIS — a physical boundary, so // never the shrinking comm margin. This folds the old RangeWithAxisBCs // fixup into make_range, letting the same loop serve every CoordType. - const int G = static_cast(N_GHOSTS); + const int G = static_cast(N_GHOSTS); const auto comm_side = [](FldsBC b) { return (b == FldsBC::PERIODIC) or (b == FldsBC::SYNC); }; @@ -335,8 +336,7 @@ namespace ntt { { domain.mesh.i_max(in::x1) + mh(0) }); } else if constexpr (M::Dim == Dim::_2D) { return CreateRangePolicy( - { domain.mesh.i_min(in::x1) - ml(0), - domain.mesh.i_min(in::x2) - ml(1) }, + { domain.mesh.i_min(in::x1) - ml(0), domain.mesh.i_min(in::x2) - ml(1) }, { domain.mesh.i_max(in::x1) + mh(0), domain.mesh.i_max(in::x2) + mh(1) }); } else { @@ -354,11 +354,10 @@ namespace ntt { Kokkos::parallel_for( "CurrentsFilter", make_range(m), - kernel::DigitalFilter_kernel( - domain.fields.buff, - domain.fields.cur, - size, - flds_bc)); + kernel::DigitalFilter_kernel(domain.fields.buff, + domain.fields.cur, + size, + flds_bc)); std::swap(domain.fields.cur, domain.fields.buff); --m; if (m < 0 or i == nfilter - 1u) { diff --git a/src/kernels/currents_deposit.hpp b/src/kernels/currents_deposit.hpp deleted file mode 100644 index 98fa0b661..000000000 --- a/src/kernels/currents_deposit.hpp +++ /dev/null @@ -1,1246 +0,0 @@ -/** - * @file kernels/currents_deposit.hpp - * @brief Covariant algorithms for the current deposition. - * - * Two kernels share the same per-particle body - * (`kernel::DepositOneParticle`): - * - `kernel::DepositCurrents_kernel` flat (RangePolicy over particles, - * writes into a `Kokkos::Experimental::ScatterView`). Always available. - * - `kernel::DepositCurrents_kernel_tiled` team-policy - * (one team per spatial tile, accumulates into team SLM scratch with - * atomic adds, then flushes to global J). Available when `team_policy=ON` - * (`#if defined(TEAM_POLICY)`). Stream 2 of the Pattern A plan. - * - * @implements - * - kernel::deposit::PrtlPack<> - * - kernel::DepositOneParticle<> - * - kernel::DepositCurrents_kernel<> - * - kernel::DepositCurrents_kernel_tiled<> (TEAM_POLICY only) - * @namespaces: - * - kernel:: - * - kernel::deposit:: - */ - -#ifndef KERNELS_CURRENTS_DEPOSIT_HPP -#define KERNELS_CURRENTS_DEPOSIT_HPP - -#include "enums.h" -#include "global.h" - -#include "arch/kokkos_aliases.h" -#include "traits/metric.h" -#include "utils/error.h" -#include "utils/numeric.h" - -#include "framework/containers/particles.h" -#include "kernels/particle_shapes.hpp" - -#include -#include - -#define i_di_to_Xi(I, DI) (static_cast((I)) + static_cast((DI))) - -namespace kernel { - using namespace ntt; - - /** - * @brief Per-particle deposit body, shared between the flat and tiled - * kernels. - * - * The caller supplies a `deposit_at(idx..., comp, val)` callback that - * applies the contribution `val` to the J component `comp` at the - * **global** J cell index `idx...` (already includes the `N_GHOSTS` - * offset). The flat kernel's callback simply does - * `J_acc(idx..., comp) += val` on its scatter-view accessor; the tiled - * kernel's callback translates `idx...` into per-tile scratch - * coordinates and uses `Kokkos::atomic_add` on SLM. Either way, this - * function is identical numerically and contains the only deposit math - * in the codebase. - * - * Dead particles return early. The callback is invoked once per cell - * write, with the dimension-appropriate signature: - * - 1D: `deposit_at(int g_i1, int comp, real_t val)` - * - 2D: `deposit_at(int g_i1, int g_i2, int comp, real_t val)` - * - 3D: `deposit_at(int g_i1, int g_i2, int g_i3, int comp, real_t val)` - */ - template - Inline void DepositOneParticle(prtlidx_t p, - const ParticleArrays& prtls, - const M& metric, - real_t charge, - real_t inv_dt, - DepositFn deposit_at) { - static_assert(O <= 11u, "Shape function order O must be <= 11"); - constexpr auto D = M::Dim; - - if (prtls.tag(p) == ParticleTag::dead) { - return; - } - - // recover particle velocity to deposit in unsimulated direction - [[maybe_unused]] - vec_t vp { ZERO }; - // `vp` only feeds the unsimulated-direction current in the 1D - // (jx2, jx3) and 2D (jx3) branches. In 3D every J component comes - // from the Esirkepov/zigzag charge motion and `vp` is never read, - // so the metric transform + 1/sqrt + NaN/Inf guard below is pure - // dead work there — skip it (also frees xp/inv_energy registers). - if constexpr (D != Dim::_3D) { - coord_t xp { ZERO }; - if constexpr (D == Dim::_1D) { - xp[0] = i_di_to_Xi(prtls.i1(p), prtls.dx1(p)); - } else if constexpr (D == Dim::_2D) { - if constexpr (M::PrtlDim == Dim::_3D) { - xp[0] = i_di_to_Xi(prtls.i1(p), prtls.dx1(p)); - xp[1] = i_di_to_Xi(prtls.i2(p), prtls.dx2(p)); - xp[2] = prtls.phi(p); - } else { - xp[0] = i_di_to_Xi(prtls.i1(p), prtls.dx1(p)); - xp[1] = i_di_to_Xi(prtls.i2(p), prtls.dx2(p)); - } - } else { - xp[0] = i_di_to_Xi(prtls.i1(p), prtls.dx1(p)); - xp[1] = i_di_to_Xi(prtls.i2(p), prtls.dx2(p)); - xp[2] = i_di_to_Xi(prtls.i3(p), prtls.dx3(p)); - } - auto inv_energy { ZERO }; - if constexpr (S == SimEngine::SRPIC) { - metric.template transform_xyz( - xp, - { prtls.ux1(p), prtls.ux2(p), prtls.ux3(p) }, - vp); - inv_energy = ONE / U2GAMMA(prtls.ux1(p), prtls.ux2(p), prtls.ux3(p)); - } else { - coord_t xp_ { ZERO }; - xp_[0] = xp[0]; - real_t theta_Cd { xp[1] }; - const auto theta_Ph { metric.template convert<2, Crd::Cd, Crd::Ph>( - theta_Cd) }; - const auto small_angle { static_cast(constant::SMALL_ANGLE_GR) }; - const auto large_angle { static_cast( - constant::PI - constant::SMALL_ANGLE_GR) }; - if (theta_Ph < small_angle) { - theta_Cd = metric.template convert<2, Crd::Ph, Crd::Cd>(small_angle); - } else if (theta_Ph >= large_angle) { - theta_Cd = metric.template convert<2, Crd::Ph, Crd::Cd>(large_angle); - } - xp_[1] = theta_Cd; - metric.template transform( - xp_, - { prtls.ux1(p), prtls.ux2(p), prtls.ux3(p) }, - vp); - inv_energy = metric.alpha(xp_) / - math::sqrt(ONE + prtls.ux1(p) * vp[0] + - prtls.ux2(p) * vp[1] + prtls.ux3(p) * vp[2]); - } - if (Kokkos::isnan(vp[2]) || Kokkos::isinf(vp[2])) { - vp[2] = ZERO; - } - vp[0] *= inv_energy; - vp[1] *= inv_energy; - vp[2] *= inv_energy; - } - - const real_t coeff { prtls.weight(p) * charge }; - - if constexpr (O == 0u) { - /* - Zig-zag deposit - */ - const auto dxp_r_1 { static_cast(prtls.i1(p) == prtls.i1_prev(p)) * - (prtls.dx1(p) + prtls.dx1_prev(p)) * - static_cast(INV_2) }; - - const real_t Wx1_1 { INV_2 * - (dxp_r_1 + prtls.dx1_prev(p) + - static_cast(prtls.i1(p) > prtls.i1_prev(p))) }; - const real_t Wx1_2 { INV_2 * - (prtls.dx1(p) + dxp_r_1 + - static_cast( - static_cast(prtls.i1(p) > prtls.i1_prev(p)) + - prtls.i1_prev(p) - prtls.i1(p))) }; - const real_t Fx1_1 { (static_cast(prtls.i1(p) > prtls.i1_prev(p)) + - dxp_r_1 - prtls.dx1_prev(p)) * - coeff * inv_dt }; - const real_t Fx1_2 { (static_cast( - prtls.i1(p) - prtls.i1_prev(p) - - static_cast(prtls.i1(p) > prtls.i1_prev(p))) + - prtls.dx1(p) - dxp_r_1) * - coeff * inv_dt }; - - if constexpr (D == Dim::_1D) { - const real_t Fx2_1 { HALF * vp[1] * coeff }; - const real_t Fx2_2 { HALF * vp[1] * coeff }; - - const real_t Fx3_1 { HALF * vp[2] * coeff }; - const real_t Fx3_2 { HALF * vp[2] * coeff }; - - deposit_at(prtls.i1_prev(p) + N_GHOSTS, cur::jx1, Fx1_1); - deposit_at(prtls.i1(p) + N_GHOSTS, cur::jx1, Fx1_2); - - deposit_at(prtls.i1_prev(p) + N_GHOSTS, cur::jx2, Fx2_1 * (ONE - Wx1_1)); - deposit_at(prtls.i1_prev(p) + N_GHOSTS + 1, cur::jx2, Fx2_1 * Wx1_1); - deposit_at(prtls.i1(p) + N_GHOSTS, cur::jx2, Fx2_2 * (ONE - Wx1_2)); - deposit_at(prtls.i1(p) + N_GHOSTS + 1, cur::jx2, Fx2_2 * Wx1_2); - - deposit_at(prtls.i1_prev(p) + N_GHOSTS, cur::jx3, Fx3_1 * (ONE - Wx1_1)); - deposit_at(prtls.i1_prev(p) + N_GHOSTS + 1, cur::jx3, Fx3_1 * Wx1_1); - deposit_at(prtls.i1(p) + N_GHOSTS, cur::jx3, Fx3_2 * (ONE - Wx1_2)); - deposit_at(prtls.i1(p) + N_GHOSTS + 1, cur::jx3, Fx3_2 * Wx1_2); - } else if constexpr (D == Dim::_2D || D == Dim::_3D) { - const auto dxp_r_2 { static_cast(prtls.i2(p) == prtls.i2_prev(p)) * - (prtls.dx2(p) + prtls.dx2_prev(p)) * - static_cast(INV_2) }; - - const real_t Wx2_1 { INV_2 * (dxp_r_2 + prtls.dx2_prev(p) + - static_cast(prtls.i2(p) > - prtls.i2_prev(p))) }; - const real_t Wx2_2 { INV_2 * - (prtls.dx2(p) + dxp_r_2 + - static_cast( - static_cast(prtls.i2(p) > prtls.i2_prev(p)) + - prtls.i2_prev(p) - prtls.i2(p))) }; - const real_t Fx2_1 { (static_cast(prtls.i2(p) > prtls.i2_prev(p)) + - dxp_r_2 - prtls.dx2_prev(p)) * - coeff * inv_dt }; - const real_t Fx2_2 { - (static_cast(prtls.i2(p) - prtls.i2_prev(p) - - static_cast(prtls.i2(p) > prtls.i2_prev(p))) + - prtls.dx2(p) - dxp_r_2) * - coeff * inv_dt - }; - - if constexpr (D == Dim::_2D) { - const real_t Fx3_1 { HALF * vp[2] * coeff }; - const real_t Fx3_2 { HALF * vp[2] * coeff }; - - deposit_at(prtls.i1_prev(p) + N_GHOSTS, - prtls.i2_prev(p) + N_GHOSTS, - cur::jx1, - Fx1_1 * (ONE - Wx2_1)); - deposit_at(prtls.i1_prev(p) + N_GHOSTS, - prtls.i2_prev(p) + N_GHOSTS + 1, - cur::jx1, - Fx1_1 * Wx2_1); - deposit_at(prtls.i1(p) + N_GHOSTS, - prtls.i2(p) + N_GHOSTS, - cur::jx1, - Fx1_2 * (ONE - Wx2_2)); - deposit_at(prtls.i1(p) + N_GHOSTS, - prtls.i2(p) + N_GHOSTS + 1, - cur::jx1, - Fx1_2 * Wx2_2); - - deposit_at(prtls.i1_prev(p) + N_GHOSTS, - prtls.i2_prev(p) + N_GHOSTS, - cur::jx2, - Fx2_1 * (ONE - Wx1_1)); - deposit_at(prtls.i1_prev(p) + N_GHOSTS + 1, - prtls.i2_prev(p) + N_GHOSTS, - cur::jx2, - Fx2_1 * Wx1_1); - deposit_at(prtls.i1(p) + N_GHOSTS, - prtls.i2(p) + N_GHOSTS, - cur::jx2, - Fx2_2 * (ONE - Wx1_2)); - deposit_at(prtls.i1(p) + N_GHOSTS + 1, - prtls.i2(p) + N_GHOSTS, - cur::jx2, - Fx2_2 * Wx1_2); - - deposit_at(prtls.i1_prev(p) + N_GHOSTS, - prtls.i2_prev(p) + N_GHOSTS, - cur::jx3, - Fx3_1 * (ONE - Wx1_1) * (ONE - Wx2_1)); - deposit_at(prtls.i1_prev(p) + N_GHOSTS + 1, - prtls.i2_prev(p) + N_GHOSTS, - cur::jx3, - Fx3_1 * Wx1_1 * (ONE - Wx2_1)); - deposit_at(prtls.i1_prev(p) + N_GHOSTS, - prtls.i2_prev(p) + N_GHOSTS + 1, - cur::jx3, - Fx3_1 * (ONE - Wx1_1) * Wx2_1); - deposit_at(prtls.i1_prev(p) + N_GHOSTS + 1, - prtls.i2_prev(p) + N_GHOSTS + 1, - cur::jx3, - Fx3_1 * Wx1_1 * Wx2_1); - - deposit_at(prtls.i1(p) + N_GHOSTS, - prtls.i2(p) + N_GHOSTS, - cur::jx3, - Fx3_2 * (ONE - Wx1_2) * (ONE - Wx2_2)); - deposit_at(prtls.i1(p) + N_GHOSTS + 1, - prtls.i2(p) + N_GHOSTS, - cur::jx3, - Fx3_2 * Wx1_2 * (ONE - Wx2_2)); - deposit_at(prtls.i1(p) + N_GHOSTS, - prtls.i2(p) + N_GHOSTS + 1, - cur::jx3, - Fx3_2 * (ONE - Wx1_2) * Wx2_2); - deposit_at(prtls.i1(p) + N_GHOSTS + 1, - prtls.i2(p) + N_GHOSTS + 1, - cur::jx3, - Fx3_2 * Wx1_2 * Wx2_2); - } else { - const auto dxp_r_3 { - static_cast(prtls.i3(p) == prtls.i3_prev(p)) * - (prtls.dx3(p) + prtls.dx3_prev(p)) * static_cast(INV_2) - }; - const real_t Wx3_1 { INV_2 * (dxp_r_3 + prtls.dx3_prev(p) + - static_cast( - prtls.i3(p) > prtls.i3_prev(p))) }; - const real_t Wx3_2 { - INV_2 * - (prtls.dx3(p) + dxp_r_3 + - static_cast(static_cast(prtls.i3(p) > prtls.i3_prev(p)) + - prtls.i3_prev(p) - prtls.i3(p))) - }; - const real_t Fx3_1 { (static_cast(prtls.i3(p) > prtls.i3_prev(p)) + - dxp_r_3 - prtls.dx3_prev(p)) * - coeff * inv_dt }; - const real_t Fx3_2 { - (static_cast(prtls.i3(p) - prtls.i3_prev(p) - - static_cast(prtls.i3(p) > prtls.i3_prev(p))) + - prtls.dx3(p) - dxp_r_3) * - coeff * inv_dt - }; - - deposit_at(prtls.i1_prev(p) + N_GHOSTS, - prtls.i2_prev(p) + N_GHOSTS, - prtls.i3_prev(p) + N_GHOSTS, - cur::jx1, - Fx1_1 * (ONE - Wx2_1) * (ONE - Wx3_1)); - deposit_at(prtls.i1_prev(p) + N_GHOSTS, - prtls.i2_prev(p) + N_GHOSTS + 1, - prtls.i3_prev(p) + N_GHOSTS, - cur::jx1, - Fx1_1 * Wx2_1 * (ONE - Wx3_1)); - deposit_at(prtls.i1_prev(p) + N_GHOSTS, - prtls.i2_prev(p) + N_GHOSTS, - prtls.i3_prev(p) + N_GHOSTS + 1, - cur::jx1, - Fx1_1 * (ONE - Wx2_1) * Wx3_1); - deposit_at(prtls.i1_prev(p) + N_GHOSTS, - prtls.i2_prev(p) + N_GHOSTS + 1, - prtls.i3_prev(p) + N_GHOSTS + 1, - cur::jx1, - Fx1_1 * Wx2_1 * Wx3_1); - - deposit_at(prtls.i1(p) + N_GHOSTS, - prtls.i2(p) + N_GHOSTS, - prtls.i3(p) + N_GHOSTS, - cur::jx1, - Fx1_2 * (ONE - Wx2_2) * (ONE - Wx3_2)); - deposit_at(prtls.i1(p) + N_GHOSTS, - prtls.i2(p) + N_GHOSTS + 1, - prtls.i3(p) + N_GHOSTS, - cur::jx1, - Fx1_2 * Wx2_2 * (ONE - Wx3_2)); - deposit_at(prtls.i1(p) + N_GHOSTS, - prtls.i2(p) + N_GHOSTS, - prtls.i3(p) + N_GHOSTS + 1, - cur::jx1, - Fx1_2 * (ONE - Wx2_2) * Wx3_2); - deposit_at(prtls.i1(p) + N_GHOSTS, - prtls.i2(p) + N_GHOSTS + 1, - prtls.i3(p) + N_GHOSTS + 1, - cur::jx1, - Fx1_2 * Wx2_2 * Wx3_2); - - deposit_at(prtls.i1_prev(p) + N_GHOSTS, - prtls.i2_prev(p) + N_GHOSTS, - prtls.i3_prev(p) + N_GHOSTS, - cur::jx2, - Fx2_1 * (ONE - Wx1_1) * (ONE - Wx3_1)); - deposit_at(prtls.i1_prev(p) + N_GHOSTS + 1, - prtls.i2_prev(p) + N_GHOSTS, - prtls.i3_prev(p) + N_GHOSTS, - cur::jx2, - Fx2_1 * Wx1_1 * (ONE - Wx3_1)); - deposit_at(prtls.i1_prev(p) + N_GHOSTS, - prtls.i2_prev(p) + N_GHOSTS, - prtls.i3_prev(p) + N_GHOSTS + 1, - cur::jx2, - Fx2_1 * (ONE - Wx1_1) * Wx3_1); - deposit_at(prtls.i1_prev(p) + N_GHOSTS + 1, - prtls.i2_prev(p) + N_GHOSTS, - prtls.i3_prev(p) + N_GHOSTS + 1, - cur::jx2, - Fx2_1 * Wx1_1 * Wx3_1); - - deposit_at(prtls.i1(p) + N_GHOSTS, - prtls.i2(p) + N_GHOSTS, - prtls.i3(p) + N_GHOSTS, - cur::jx2, - Fx2_2 * (ONE - Wx1_2) * (ONE - Wx3_2)); - deposit_at(prtls.i1(p) + N_GHOSTS + 1, - prtls.i2(p) + N_GHOSTS, - prtls.i3(p) + N_GHOSTS, - cur::jx2, - Fx2_2 * Wx1_2 * (ONE - Wx3_2)); - deposit_at(prtls.i1(p) + N_GHOSTS, - prtls.i2(p) + N_GHOSTS, - prtls.i3(p) + N_GHOSTS + 1, - cur::jx2, - Fx2_2 * (ONE - Wx1_2) * Wx3_2); - deposit_at(prtls.i1(p) + N_GHOSTS + 1, - prtls.i2(p) + N_GHOSTS, - prtls.i3(p) + N_GHOSTS + 1, - cur::jx2, - Fx2_2 * Wx1_2 * Wx3_2); - - deposit_at(prtls.i1_prev(p) + N_GHOSTS, - prtls.i2_prev(p) + N_GHOSTS, - prtls.i3_prev(p) + N_GHOSTS, - cur::jx3, - Fx3_1 * (ONE - Wx1_1) * (ONE - Wx2_1)); - deposit_at(prtls.i1_prev(p) + N_GHOSTS + 1, - prtls.i2_prev(p) + N_GHOSTS, - prtls.i3_prev(p) + N_GHOSTS, - cur::jx3, - Fx3_1 * Wx1_1 * (ONE - Wx2_1)); - deposit_at(prtls.i1_prev(p) + N_GHOSTS, - prtls.i2_prev(p) + N_GHOSTS + 1, - prtls.i3_prev(p) + N_GHOSTS, - cur::jx3, - Fx3_1 * (ONE - Wx1_1) * Wx2_1); - deposit_at(prtls.i1_prev(p) + N_GHOSTS + 1, - prtls.i2_prev(p) + N_GHOSTS + 1, - prtls.i3_prev(p) + N_GHOSTS, - cur::jx3, - Fx3_1 * Wx1_1 * Wx2_1); - - deposit_at(prtls.i1(p) + N_GHOSTS, - prtls.i2(p) + N_GHOSTS, - prtls.i3(p) + N_GHOSTS, - cur::jx3, - Fx3_2 * (ONE - Wx1_2) * (ONE - Wx2_2)); - deposit_at(prtls.i1(p) + N_GHOSTS + 1, - prtls.i2(p) + N_GHOSTS, - prtls.i3(p) + N_GHOSTS, - cur::jx3, - Fx3_2 * Wx1_2 * (ONE - Wx2_2)); - deposit_at(prtls.i1(p) + N_GHOSTS, - prtls.i2(p) + N_GHOSTS + 1, - prtls.i3(p) + N_GHOSTS, - cur::jx3, - Fx3_2 * (ONE - Wx1_2) * Wx2_2); - deposit_at(prtls.i1(p) + N_GHOSTS + 1, - prtls.i2(p) + N_GHOSTS + 1, - prtls.i3(p) + N_GHOSTS, - cur::jx3, - Fx3_2 * Wx1_2 * Wx2_2); - } - } - } else if constexpr ((O >= 1u) and (O <= 11u)) { - - // shape function in dim1 -> always required - real_t iS_x1[O + 2], fS_x1[O + 2]; - // indices of the shape function - int i1_min, i1_max; - - // call shape function - prtl_shape::for_deposit(prtls.i1_prev(p), - static_cast(prtls.dx1_prev(p)), - prtls.i1(p), - static_cast(prtls.dx1(p)), - i1_min, - i1_max, - iS_x1, - fS_x1); - - if constexpr (D == Dim::_1D) { - // (1D): fused Esirkepov, no [O+2] temporaries. - // jx1[i] = -Qdx1dt * sum_{i'=0}^{i} (fS_x1[i'] - iS_x1[i']) - // = -Qdx1dt * P1[i] (Eq. 38, 1D) - // Wx23[i] = HALF * (fS_x1[i] + iS_x1[i]) (computed inline) - const real_t Qdx1dt = coeff * inv_dt; - const real_t QVx2 = coeff * vp[1]; - const real_t QVx3 = coeff * vp[2]; - - // account for ghost cells - i1_min += N_GHOSTS; - i1_max += N_GHOSTS; - - // get number of update indices for asymmetric movement - const int di_x1 = i1_max - i1_min; - - // Current update — fused over the union line so the J cell - // stays L1-resident across the 3 component atomic_adds. - real_t P1 = ZERO; - for (int i = 0; i <= di_x1; ++i) { - P1 += fS_x1[i] - iS_x1[i]; - const int gi = i1_min + i; - const real_t Wx23 = HALF * (fS_x1[i] + iS_x1[i]); - if (i < di_x1) { - deposit_at(gi, cur::jx1, -Qdx1dt * P1); - } - deposit_at(gi, cur::jx2, QVx2 * Wx23); - deposit_at(gi, cur::jx3, QVx3 * Wx23); - } - - } else if constexpr (D == Dim::_2D) { - - // shape function in dim1 -> always required - real_t iS_x2[O + 2], fS_x2[O + 2]; - // indices of the shape function - int i2_min, i2_max; - - // call shape function - prtl_shape::for_deposit(prtls.i2_prev(p), - static_cast(prtls.dx2_prev(p)), - prtls.i2(p), - static_cast(prtls.dx2(p)), - i2_min, - i2_max, - iS_x2, - fS_x2); - - /** - * (2D): fused Esirkepov, no [O+2]^2 temporaries. - * - * Esirkepov 2001 Eq. 38 (simplified) is separable: with - * P1[i] = sum_{i'=0}^{i} (fS_x1[i'] - iS_x1[i']) and - * P2[j] = sum_{j'=0}^{j} (fS_x2[j'] - iS_x2[j']), - * jx1[i][j] = -Q*HALF * P1[i] * (fS_x2[j] + iS_x2[j]) - * jx2[i][j] = -Q*HALF * P2[j] * (fS_x1[i] + iS_x1[i]) - * Wx3[i][j] = THIRD*( fS_x2[j]*(HALF*iS_x1[i]+fS_x1[i]) - * + iS_x2[j]*(HALF*fS_x1[i]+iS_x1[i]) ) - * with Q = coeff*inv_dt (Qdx1dt == Qdx2dt). Same value as the - * old explicit Wx/jx tensors up to FP reassociation; - * charge-conserving by construction. Prefix sums carried as - * running scalars, so the only per-thread state is the - * existing 1D shape arrays. - */ - const real_t QVx3 = coeff * vp[2]; - // -Q*HALF prefactor (Qdx1dt == Qdx2dt == coeff*inv_dt) - const real_t cf = -(coeff * inv_dt) * HALF; - - // account for ghost cells - i1_min += N_GHOSTS; - i2_min += N_GHOSTS; - i1_max += N_GHOSTS; - i2_max += N_GHOSTS; - - // get number of update indices for asymmetric movement - const int di_x1 = i1_max - i1_min; - const int di_x2 = i2_max - i2_min; - - // Current update — fused over the union plane so the J cell - // line stays L1-resident across the 3 component atomic_adds. - real_t P1 = ZERO; - for (int i = 0; i <= di_x1; ++i) { - P1 += fS_x1[i] - iS_x1[i]; - const int gi = i1_min + i; - const real_t iSx1 = iS_x1[i]; - const real_t fSx1 = fS_x1[i]; - const real_t A1 = fSx1 + iSx1; // jx2 cross-factor - real_t P2 = ZERO; - for (int j = 0; j <= di_x2; ++j) { - P2 += fS_x2[j] - iS_x2[j]; - const int gj = i2_min + j; - const real_t iSx2 = iS_x2[j]; - const real_t fSx2 = fS_x2[j]; - if (i < di_x1) { - deposit_at(gi, gj, cur::jx1, cf * P1 * (fSx2 + iSx2)); - } - if (j < di_x2) { - deposit_at(gi, gj, cur::jx2, cf * P2 * A1); - } - const real_t Wx3 = THIRD * (fSx2 * (HALF * iSx1 + fSx1) + - iSx2 * (HALF * fSx1 + iSx1)); - deposit_at(gi, gj, cur::jx3, QVx3 * Wx3); - } - } - - } else if constexpr (D == Dim::_3D) { - // shape function in dim2 - real_t iS_x2[O + 2], fS_x2[O + 2]; - // indices of the shape function - int i2_min, i2_max; - // call shape function - prtl_shape::for_deposit(prtls.i2_prev(p), - static_cast(prtls.dx2_prev(p)), - prtls.i2(p), - static_cast(prtls.dx2(p)), - i2_min, - i2_max, - iS_x2, - fS_x2); - - // shape function in dim3 - real_t iS_x3[O + 2], fS_x3[O + 2]; - // indices of the shape function - int i3_min, i3_max; - - // call shape function - prtl_shape::for_deposit(prtls.i3_prev(p), - static_cast(prtls.dx3_prev(p)), - prtls.i3(p), - static_cast(prtls.dx3(p)), - i3_min, - i3_max, - iS_x3, - fS_x3); - - /** - * fused Esirkepov, no (O+2)^3 temporaries. - * - * The Esirkepov 3D current (2001, Eq. 31) is separable: with - * P1[i] = sum_{i'=0}^{i} (fS_x1[i'] - iS_x1[i']) (and likewise - * P2[j], P3[k]) the cumulative-sum currents collapse to - * - * jx1[i][j][k] = -Q*THIRD * P1[i] * G23(j,k) - * jx2[i][j][k] = -Q*THIRD * P2[j] * H13(i,k) - * jx3[i][j][k] = -Q*THIRD * P3[k] * F12(i,j) - * - * with the 1D-shape cross-factors - * - * G23(j,k) = iS_x2[j]*iS_x3[k] + fS_x2[j]*fS_x3[k] - * + HALF*(iS_x3[k]*fS_x2[j] + iS_x2[j]*fS_x3[k]) - * H13(i,k) = iS_x1[i]*iS_x3[k] + fS_x1[i]*fS_x3[k] - * + HALF*(iS_x3[k]*fS_x1[i] + iS_x1[i]*fS_x3[k]) - * F12(i,j) = iS_x1[i]*iS_x2[j] + fS_x1[i]*fS_x2[j] - * + HALF*(iS_x1[i]*fS_x2[j] + iS_x2[j]*fS_x1[i]) - * - * and Q = coeff*inv_dt (Qdxdt == Qdydt == Qdzdt). This is the - * same value as the old explicit Wx/jx tensors up to - * floating-point reassociation: charge-conserving by - * construction (the Esirkepov decomposition is exact). The - * prefix sums are carried as running scalars in the deposit - * loop, so the only per-thread state is the existing 1D shape - * arrays (no (O+2)^3 / (O+2)^2 locals, hence far fewer VGPRs - * and no private-memory tensor traffic). - */ - - // account for ghost cells - i1_min += N_GHOSTS; - i2_min += N_GHOSTS; - i3_min += N_GHOSTS; - i1_max += N_GHOSTS; - i2_max += N_GHOSTS; - i3_max += N_GHOSTS; - - // get number of update indices for asymmetric movement - const int di_x1 = i1_max - i1_min; - const int di_x2 = i2_max - i2_min; - const int di_x3 = i3_max - i3_min; - - // -Q*THIRD prefactor (Qdxdt == Qdydt == Qdzdt == coeff*inv_dt) - const real_t cf = -(coeff * inv_dt) * THIRD; - - /** - * Current update — fused over the union cube so the J cell - * line stays L1-resident across the 3 component atomic_adds. - * Per-cell branches on (i 11 not supported. Seriously. " - "What are you even doing here? Entity already goes to 11!"); - } - } - - /** - * @brief Flat current-deposition kernel. - * - * One thread per particle (RangePolicy). Writes are coalesced through a - * `Kokkos::Experimental::ScatterView` to avoid per-thread atomics on - * global J. Constructor signature is unchanged from prior versions — - * `engines/srpic/currents.h` continues to call it identically. - */ - template - class DepositCurrents_kernel { - static_assert(O <= 11u, "Shape function order O must be <= 11"); - static constexpr auto D = M::Dim; - - scatter_ndfield_t J; - const ParticleArrays prtls; - const M metric; - const real_t charge, inv_dt; - - public: - DepositCurrents_kernel(const scatter_ndfield_t& scatter_cur, - const ParticleArrays& prtls, - const M& metric, - real_t charge, - const real_t dt) - : J { scatter_cur } - , prtls { prtls } - , metric { metric } - , charge { charge } - , inv_dt { ONE / dt } { - raise::ErrorIf( - (O == 2u and N_GHOSTS < 2), - "Order of interpolation is 2, but number of ghost cells is < 2", - HERE); - } - - Inline auto operator()(prtlidx_t p) const -> void { - auto J_acc = J.access(); - if constexpr (D == Dim::_1D) { - DepositOneParticle(p, - prtls, - metric, - charge, - inv_dt, - [&](int g_i1, int comp, real_t v) { - J_acc(g_i1, comp) += v; - }); - } else if constexpr (D == Dim::_2D) { - DepositOneParticle(p, - prtls, - metric, - charge, - inv_dt, - [&](int g_i1, int g_i2, int comp, real_t v) { - J_acc(g_i1, g_i2, comp) += v; - }); - } else if constexpr (D == Dim::_3D) { - DepositOneParticle( - p, - prtls, - metric, - charge, - inv_dt, - [&](int g_i1, int g_i2, int g_i3, int comp, real_t v) { - J_acc(g_i1, g_i2, g_i3, comp) += v; - }); - } - } - }; - - /** - * @brief Tiled current-deposition kernel. - * - * One team per spatial tile (`league_size = ntiles_total`). Each team - * accumulates particle contributions into a per-team scratch buffer of - * shape `(T_TILE + 2*HALO)^D × 3` real_t, where `HALO = O + 1` cells per - * side. Scratch atomics live in SLM (PVC: ~5–10 cycles per - * `atomic_add`); the global J is touched only once per scratch cell at - * flush time. Compared with the flat scatter-view kernel: - * - global atomic pressure ~ (T_TILE + 2*HALO)^D × 3 per tile - * instead of (stencil writes per particle × particles) - * - per-particle stencil writes are tile-local (SLM) instead of - * scattering through global HBM - * - * Supports `O ∈ {0, ..., 11}`. `O == 0` (zigzag) is wired for - * A/B benchmarking against the flat scatter-view kernel — its narrow - * stencil typically makes scratch alloc/zero/flush overhead a - * regression there, but it's good to be able to measure the - * crossover. To revert and use flat for zigzag-only builds, change - * the dispatch in `engines/srpic/currents.h` from - * `#if defined(TEAM_POLICY)` to - * `#if defined(TEAM_POLICY) && (SHAPE_ORDER > 0)`. - * - * Particle iteration order is governed by `tile_offsets`: tile `t` - * owns particles `[tile_offsets(t), tile_offsets(t+1))`, post-sort. - * `SortSpatially` (`particles_sort.cpp`) is responsible for keeping - * the SoA arrays consistent with that. - * - * **Halo sizing and escape valve.** Sort runs at the end of a step - * (see `srpic.hpp`); a particle is pushed once per step thereafter, so - * its `min(i, i_prev)` may differ from the bin key by one cell of drift - * per step elapsed since the last sort. The scratch HALO is - * `STENCIL_REACH(O) + DRIFT`, where `STENCIL_REACH = 2` for zigzag - * (writes `{i_prev, i_prev+1, i, i+1}` ⇒ +2 above `min(i, i_prev)` with - * `|Δi|=1`) and `O` for Esirkepov. `DRIFT` is the `team_policy_drift` - * CMake knob (macro TEAM_POLICY_DRIFT) — the number of cells a particle - * may drift between two sorts that the halo is sized to absorb — and `1` - * by default (the every-step-sorted common case). It is independent of - * the sort cadence, which is set at runtime via `spatial_sorting_interval`; - * particles that drift past the halo take the escape valve below. - * - * Correctness does **not** depend on the halo size. Any particle whose - * full stencil escapes the scratch tile — because it drifted further - * than `DRIFT`, was reordered far from its tile by a no-sort-step - * `CommunicateParticles`, or because the halo is otherwise undersized — - * is deposited *as a whole* via a direct, bounds-clipped - * `Kokkos::atomic_add` on the global J view (the per-particle escape - * valve). Each particle's stencil is therefore deposited exactly once - * (entirely to SLM scratch when it fits, entirely to global J when it - * does not), so the path is charge-conserving; it is merely slower per - * write. Sizing `DRIFT` to the typical between-sort drift keeps the - * common case in fast SLM; sorting less often (or drifting past the - * halo) only costs escape-valve traffic, never accuracy. - * - * **Partition coverage.** The team iteration covers only the particles - * partitioned at the last sort, `[0, layout.npart_partitioned)`, clamped - * to the live `npart`. Particles appended past the partition since the - * sort are not seen here; the launcher (`engines/srpic/currents.h`) - * deposits that tail with the flat kernel so every active particle is - * covered exactly once regardless of sort cadence. - */ - template - class DepositCurrentsTiled_kernel { - static_assert(O <= 11u, "Shape order O must be <= 11"); - static_assert(T_TILE > 0u, "T_TILE must be positive"); - static constexpr auto D = M::Dim; - - /** - * Per-side scratch halo, derived from first principles. - * - * total halo = stencil_reach(O) + drift_between_sort_and_deposit - * - * stencil_reach(O) — maximum cells the deposit writes ABOVE - * min(i, i_prev) under CFL |v * dt/dx| <= 1/2: - * - O == 0 (zigzag): writes { i_prev, i_prev+1, i, i+1 } => +2 - * - O >= 1 Esirkepov: `for_deposit` returns an (O+2)-wide - * array but only O+1 entries are non-zero, and the union - * window satisfies `i_max - i_min <= O+1` (see - * particle_shapes.hpp::for_deposit). The genuine one-sided - * reach above min(i, i_prev) is therefore O, not O+1 — the - * old `O+1` carried one extra cell of conservative padding - * on top of the already-conservative drift term below. - * - * drift — sort runs at end-of-step (see srpic.hpp), so a particle is - * pushed once per step between its last sort and a given deposit. With - * a runtime sort interval of `K` (spatial_sorting_interval), a particle - * drifts at most `K` cells (CFL |v dt/dx| <= 1/2 ⇒ |Δi| <= 1 per step) - * before the next sort. The `team_policy_drift` CMake knob (macro - * TEAM_POLICY_DRIFT) sets DRIFT independently of `K`, sizing the halo so - * a particle that drifts up to DRIFT cells still deposits inside its - * tile scratch. DRIFT defaults to 1 (the sorted-every-step common case); - * any particle that drifts past the halo (e.g. a larger sort interval, - * or a CFL excursion) takes the per-particle global-J escape valve - * below — correct, only slower (see - * the class doc-comment for why this is charge-conserving). - */ - static constexpr int STENCIL_REACH = (O == 0u) ? 2 : static_cast(O); - // One-sided footprint reach for the per-particle escape valve: the - // deposit writes at most this many cells above max(i,i_prev) (and fewer - // below min), so [min - FOOTPRINT_REACH, max + FOOTPRINT_REACH] in cell - // coords conservatively bounds every deposited cell for any order - // (Esirkepov reaches max+O; O=0 zigzag reaches max+1). - static constexpr int FOOTPRINT_REACH = (O == 0u) ? 1 : static_cast(O); -#if defined(TEAM_POLICY_DRIFT) - static constexpr int DRIFT = static_cast(TEAM_POLICY_DRIFT); -#else - static constexpr int DRIFT = 1; -#endif - static constexpr int HALO = STENCIL_REACH + DRIFT; - static constexpr int TE = static_cast(T_TILE) + 2 * HALO; - - using exec_space = Kokkos::DefaultExecutionSpace; - using team_policy = Kokkos::TeamPolicy; - using member_t = typename team_policy::member_type; - - ndfield_t J; - ParticleArrays prtls; - const M metric; - const real_t charge, inv_dt; - - // Tile metadata produced by SortSpatially. - array_t tile_offsets; - ncells_t ntx1 { 1u }, ntx2 { 1u }, ntx3 { 1u }; - ncells_t total_tiles { 0u }; - - /** - * Current active-particle count. `tile_offsets` partitions only the - * particles that existed at the last sort ([0, layout.npart_partitioned)); - * `npart` may differ if the pusher dead-tagged particles in place since. - * Each team clamps its `[tile_offsets(t), tile_offsets(t+1))` slice to - * `npart` so stale slots past the live array are never read. Particles - * appended *beyond* the partition (npart > npart_partitioned) are not seen - * by any team here — the launcher deposits that tail separately. - */ - npart_t npart { 0u }; - - /** - * J's full storage extent including all ghost cells. Used to clip - * the cooperative flush so that a partial tile at the high end of - * the domain does not over-write past the J view. - */ - int j_ext1 { 0 }, j_ext2 { 0 }, j_ext3 { 0 }; - - public: - DepositCurrentsTiled_kernel(const ndfield_t& cur, - const ParticleArrays& prtls, - const M& metric, - real_t charge, - real_t dt, - const TileLayout& layout, - npart_t npart) - : J { cur } - , prtls { prtls } - , metric { metric } - , charge { charge } - , inv_dt { ONE / dt } - , tile_offsets { layout.tile_offsets } - , ntx1 { layout.ntiles_per_axis[0] } - , ntx2 { layout.ntiles_per_axis[1] } - , ntx3 { layout.ntiles_per_axis[2] } - , total_tiles { layout.ntiles_total } - , npart { npart } { - raise::ErrorIf( - layout.tile_size != T_TILE, - "Tiled deposit launched with mismatched T_TILE and runtime tile_size", - HERE); - /** - * @note: HALO is allowed to exceed N_GHOSTS. The cooperative - * scratch→J flush and the per-particle escape valve both bounds-clip - * their writes against `j_ext*` so writes that would land past J's - * ghost stripe are silently dropped (they only ever come from a - * particle whose stencil reaches into the domain ghost region, where - * CommunicateFields will re-supply the contribution). - */ - if constexpr (D == Dim::_1D or D == Dim::_2D or D == Dim::_3D) { - j_ext1 = static_cast(cur.extent(0)); - } - if constexpr (D == Dim::_2D or D == Dim::_3D) { - j_ext2 = static_cast(cur.extent(1)); - } - if constexpr (D == Dim::_3D) { - j_ext3 = static_cast(cur.extent(2)); - } - } - - /** - * @brief Per-team scratch size in bytes. Used by the launcher to set - * `team_policy.set_scratch_size(0, Kokkos::PerTeam(bytes))`. - */ - static constexpr size_t scratch_bytes() { - // The component count (3) is a *static* extent of scratch_ndfield_t - // (View / **[3] / ***[3]), so shmem_size() takes only the - // dynamic spatial extents — passing 3 as well trips Kokkos' - // `rank_dynamic != number of arguments` abort. This matches the - // scratch View construction below, which also omits the 3. - if constexpr (D == Dim::_1D) { - return scratch_ndfield_t::shmem_size(TE); - } else if constexpr (D == Dim::_2D) { - return scratch_ndfield_t::shmem_size(TE, TE); - } else { - return scratch_ndfield_t::shmem_size(TE, TE, TE); - } - } - - Inline void operator()(const member_t& team) const { - const auto tile_id = static_cast(team.league_rank()); - /** - * Tile coordinates (tile-grid indices) → tile origin in **active** - * cell coords (no ghost offset). Using ncells_t to match the linearised - * tile index produced by SortSpatially. - */ - ncells_t tx1 = 0, tx2 = 0, tx3 = 0; - if constexpr (D == Dim::_1D) { - tx1 = tile_id; - } else if constexpr (D == Dim::_2D) { - tx1 = tile_id / ntx2; - tx2 = tile_id - tx1 * ntx2; - } else { - const auto plane = ntx2 * ntx3; - tx1 = tile_id / plane; - const auto rem = tile_id - tx1 * plane; - tx2 = rem / ntx3; - tx3 = rem - tx2 * ntx3; - } - /** - * origin_active = lowest active-cell index in the tile (no ghost). - * origin_J = same value translated into J's storage coordinate - * (i.e. plus N_GHOSTS). - * origin_J_low = J coordinate of scratch index 0 (i.e. origin_J - HALO). - * local index `li` in scratch ↔ global J index `gi = li + origin_J_low`. - */ - const int origin_J1_low = static_cast(tx1 * T_TILE) + - static_cast(N_GHOSTS) - HALO; - const int origin_J2_low = static_cast(tx2 * T_TILE) + - static_cast(N_GHOSTS) - HALO; - const int origin_J3_low = static_cast(tx3 * T_TILE) + - static_cast(N_GHOSTS) - HALO; - - // Allocate scratch and cooperatively zero-fill it. - if constexpr (D == Dim::_1D) { - scratch_ndfield_t scr { team.team_scratch(0), TE }; - Kokkos::parallel_for(Kokkos::TeamThreadRange(team, TE * 3), - [&](ncells_t idx) { - const auto li = idx / 3; - const auto c = idx - li * 3; - scr(li, c) = ZERO; - }); - team.team_barrier(); - - // Clamp the tile's particle slice to the live array: slots past - // `npart` may hold stale (possibly alive-tagged) data from a prior - // step's compaction and must not be re-deposited. - const auto t_lo = tile_offsets(tile_id); - const auto t_hi = tile_offsets(tile_id + 1u); - const auto p_begin = (t_lo < npart) ? t_lo : npart; - const auto p_end = (t_hi < npart) ? t_hi : npart; - Kokkos::parallel_for( - Kokkos::TeamThreadRange(team, p_begin, p_end), - [&](prtlidx_t p) { - /** - * Per-particle escape valve: route the WHOLE particle to the - * global J view when its Esirkepov footprint does not fit - * inside this tile's scratch window [0,TE); only particles - * fully inside the tile touch SLM scratch. A particle drifts - * out of its tile when sorted less often than every step. - * - * The conservative footprint bound in cell coords, - * [min(i,i_prev) - O, max(i,i_prev) + O], covers - * prtl_shape::for_deposit for any order (i_min >= - * min-floor(O/2), i_max <= max+O), so when `to_scratch` is true - * every deposited cell is provably in [0,TE) and the scratch write - * needs no per-cell bounds test. The global path bounds-clips - * against J's storage extent (writes past the ghost stripe are - * re-supplied by SynchronizeFields(J)). - */ - const int i1c = prtls.i1(p), i1p = prtls.i1_prev(p); - const int lo1 = (i1c < i1p ? i1c : i1p) + static_cast(N_GHOSTS) - - FOOTPRINT_REACH - origin_J1_low; - const int hi1 = (i1c > i1p ? i1c : i1p) + static_cast(N_GHOSTS) + - FOOTPRINT_REACH - origin_J1_low; - const bool to_scratch = (lo1 >= 0 and hi1 < TE); - DepositOneParticle( - p, - prtls, - metric, - charge, - inv_dt, - [&](int g_i1, int comp, real_t v) { - if (to_scratch) { - Kokkos::atomic_add(&scr(g_i1 - origin_J1_low, comp), v); - } else if (g_i1 >= 0 and g_i1 < j_ext1) { - // Bounds-clip the escape-valve write against J's storage, - // exactly as the cooperative flush does. Cells past the - // ghost stripe are re-supplied by SynchronizeFields(J); an - // unclipped write here faults the GPU (an escaped boundary - // particle's stencil can reach past j_ext1). - Kokkos::atomic_add(&J(g_i1, comp), v); - } - }); - }); - team.team_barrier(); - - /** - * Cooperative flush of scratch to global J. Bounds-clip against - * the J view extent in case a partial high-end tile (or non-zero - * halo at domain edges) would otherwise write past J. - */ - Kokkos::parallel_for(Kokkos::TeamThreadRange(team, TE * 3), - [&](const int idx) { - const auto li = idx / 3; - const auto c = idx - li * 3; - const auto gi = li + origin_J1_low; - if (gi < 0 or gi >= j_ext1) { - return; - } - const real_t v = scr(li, c); - if (v != ZERO) { - Kokkos::atomic_add(&J(gi, c), v); - } - }); - } else if constexpr (D == Dim::_2D) { - scratch_ndfield_t scr { team.team_scratch(0), TE, TE }; - Kokkos::parallel_for(Kokkos::TeamThreadRange(team, SQR(TE) * 3), - [&](const int idx) { - const auto lij = idx / 3; - const auto c = idx - lij * 3; - const auto li = lij / TE; - const auto lj = lij - li * TE; - scr(li, lj, c) = ZERO; - }); - team.team_barrier(); - - // Clamp the tile's particle slice to the live array: slots past - // `npart` may hold stale (possibly alive-tagged) data from a prior - // step's compaction and must not be re-deposited. - const auto t_lo = tile_offsets(tile_id); - const auto t_hi = tile_offsets(tile_id + 1u); - const auto p_begin = (t_lo < npart) ? t_lo : npart; - const auto p_end = (t_hi < npart) ? t_hi : npart; - Kokkos::parallel_for( - Kokkos::TeamThreadRange(team, p_begin, p_end), - [&](prtlidx_t p) { - // See 1D branch for rationale: route the whole particle to the - // global escape valve unless its full footprint fits in scratch. - const int i1c = prtls.i1(p), i1p = prtls.i1_prev(p); - const int i2c = prtls.i2(p), i2p = prtls.i2_prev(p); - const int lo1 = (i1c < i1p ? i1c : i1p) + static_cast(N_GHOSTS) - - FOOTPRINT_REACH - origin_J1_low; - const int hi1 = (i1c > i1p ? i1c : i1p) + static_cast(N_GHOSTS) + - FOOTPRINT_REACH - origin_J1_low; - const int lo2 = (i2c < i2p ? i2c : i2p) + static_cast(N_GHOSTS) - - FOOTPRINT_REACH - origin_J2_low; - const int hi2 = (i2c > i2p ? i2c : i2p) + static_cast(N_GHOSTS) + - FOOTPRINT_REACH - origin_J2_low; - const bool to_scratch = (lo1 >= 0 and hi1 < TE and lo2 >= 0 and - hi2 < TE); - DepositOneParticle( - p, - prtls, - metric, - charge, - inv_dt, - [&](const int g_i1, const int g_i2, int comp, real_t v) { - if (to_scratch) { - Kokkos::atomic_add( - &scr(g_i1 - origin_J1_low, g_i2 - origin_J2_low, comp), - v); - } else if (g_i1 >= 0 and g_i1 < j_ext1 and g_i2 >= 0 and - g_i2 < j_ext2) { - // Bounds-clip as the cooperative flush does; an unclipped - // escape-valve write faults the GPU when an escaped boundary - // particle's stencil reaches past j_ext. - Kokkos::atomic_add(&J(g_i1, g_i2, comp), v); - } - }); - }); - team.team_barrier(); - - Kokkos::parallel_for(Kokkos::TeamThreadRange(team, SQR(TE) * 3), - [&](const int idx) { - const auto lij = idx / 3; - const auto c = idx - lij * 3; - const auto li = lij / TE; - const auto lj = lij - li * TE; - const auto gi = li + origin_J1_low; - const auto gj = lj + origin_J2_low; - if ((gi < 0 or gi >= j_ext1) or - (gj < 0 or gj >= j_ext2)) { - return; - } - const real_t v = scr(li, lj, c); - if (v != ZERO) { - Kokkos::atomic_add(&J(gi, gj, c), v); - } - }); - } else if constexpr (D == Dim::_3D) { - scratch_ndfield_t scr { team.team_scratch(0), TE, TE, TE }; - Kokkos::parallel_for(Kokkos::TeamThreadRange(team, CUBE(TE) * 3), - [&](const int idx) { - const auto lijk = idx / 3; - const auto c = idx - lijk * 3; - const auto li = lijk / (TE * TE); - const auto rem = lijk - li * TE * TE; - const auto lj = rem / TE; - const auto lk = rem - lj * TE; - scr(li, lj, lk, c) = ZERO; - }); - team.team_barrier(); - - // Clamp the tile's particle slice to the live array: slots past - // `npart` may hold stale (possibly alive-tagged) data from a prior - // step's compaction and must not be re-deposited. - const auto t_lo = tile_offsets(tile_id); - const auto t_hi = tile_offsets(tile_id + 1u); - const auto p_begin = (t_lo < npart) ? t_lo : npart; - const auto p_end = (t_hi < npart) ? t_hi : npart; - Kokkos::parallel_for( - Kokkos::TeamThreadRange(team, p_begin, p_end), - [&](prtlidx_t p) { - // See 1D branch for rationale: route the whole particle to the - // global escape valve unless its full footprint fits in scratch. - const int i1c = prtls.i1(p), i1p = prtls.i1_prev(p); - const int i2c = prtls.i2(p), i2p = prtls.i2_prev(p); - const int i3c = prtls.i3(p), i3p = prtls.i3_prev(p); - const int lo1 = (i1c < i1p ? i1c : i1p) + static_cast(N_GHOSTS) - - FOOTPRINT_REACH - origin_J1_low; - const int hi1 = (i1c > i1p ? i1c : i1p) + static_cast(N_GHOSTS) + - FOOTPRINT_REACH - origin_J1_low; - const int lo2 = (i2c < i2p ? i2c : i2p) + static_cast(N_GHOSTS) - - FOOTPRINT_REACH - origin_J2_low; - const int hi2 = (i2c > i2p ? i2c : i2p) + static_cast(N_GHOSTS) + - FOOTPRINT_REACH - origin_J2_low; - const int lo3 = (i3c < i3p ? i3c : i3p) + static_cast(N_GHOSTS) - - FOOTPRINT_REACH - origin_J3_low; - const int hi3 = (i3c > i3p ? i3c : i3p) + static_cast(N_GHOSTS) + - FOOTPRINT_REACH - origin_J3_low; - const bool to_scratch = (lo1 >= 0 and hi1 < TE and lo2 >= 0 and - hi2 < TE and lo3 >= 0 and hi3 < TE); - DepositOneParticle( - p, - prtls, - metric, - charge, - inv_dt, - [&](const int g_i1, const int g_i2, const int g_i3, int comp, real_t v) { - if (to_scratch) { - Kokkos::atomic_add(&scr(g_i1 - origin_J1_low, - g_i2 - origin_J2_low, - g_i3 - origin_J3_low, - comp), - v); - } else if (g_i1 >= 0 and g_i1 < j_ext1 and g_i2 >= 0 and - g_i2 < j_ext2 and g_i3 >= 0 and g_i3 < j_ext3) { - // Bounds-clip as the cooperative flush does; an unclipped - // escape-valve write faults the GPU when an escaped boundary - // particle's stencil reaches past j_ext. - Kokkos::atomic_add(&J(g_i1, g_i2, g_i3, comp), v); - } - }); - }); - team.team_barrier(); - - Kokkos::parallel_for(Kokkos::TeamThreadRange(team, CUBE(TE) * 3), - [&](const int idx) { - const int lijk = idx / 3; - const int c = idx - lijk * 3; - const int li = lijk / (TE * TE); - const int rem = lijk - li * TE * TE; - const int lj = rem / TE; - const int lk = rem - lj * TE; - const int gi = li + origin_J1_low; - const int gj = lj + origin_J2_low; - const int gk = lk + origin_J3_low; - if ((gi < 0 or gi >= j_ext1) or - (gj < 0 or gj >= j_ext2) or - (gk < 0 or gk >= j_ext3)) { - return; - } - const real_t v = scr(li, lj, lk, c); - if (v != ZERO) { - Kokkos::atomic_add(&J(gi, gj, gk, c), v); - } - }); - } - } - }; - -} // namespace kernel - -#undef i_di_to_Xi - -#endif // KERNELS_CURRENTS_DEPOSIT_HPP diff --git a/src/kernels/deposition/currents/global.hpp b/src/kernels/deposition/currents/global.hpp new file mode 100644 index 000000000..177bd16c8 --- /dev/null +++ b/src/kernels/deposition/currents/global.hpp @@ -0,0 +1,102 @@ +/** + * @file kernels/deposition/currents/global.hpp + * @brief Global current deposition kernel without tiling. + * + * @implements + * - kernel::DepositCurrents_kernel<> + * @namespaces: + * - kernel:: + */ + +#ifndef KERNELS_DEPOSITION_CURRENTS_GLOBAL_HPP +#define KERNELS_DEPOSITION_CURRENTS_GLOBAL_HPP + +#include "enums.h" +#include "global.h" + +#include "arch/kokkos_aliases.h" +#include "traits/metric.h" +#include "utils/error.h" +#include "utils/numeric.h" + +#include "framework/containers/particles.h" +#include "kernels/deposition/currents/single-particle.hpp" + +#include +#include + +namespace kernel { + using namespace ntt; + + /** + * @brief Flat current-deposition kernel. + * + * One thread per particle (RangePolicy). Writes are coalesced through a + * `Kokkos::Experimental::ScatterView` to avoid per-thread atomics on + * global J. Constructor signature is unchanged from prior versions — + * `engines/srpic/currents.h` continues to call it identically. + */ + template + class DepositCurrents_kernel { + static_assert(O <= 11u, "Shape function order O must be <= 11"); + static constexpr auto D = M::Dim; + + scatter_ndfield_t J; + const ParticleArrays prtls; + const M metric; + const real_t charge, inv_dt; + + public: + DepositCurrents_kernel(const scatter_ndfield_t& scatter_cur, + const ParticleArrays& prtls, + const M& metric, + real_t charge, + const real_t dt) + : J { scatter_cur } + , prtls { prtls } + , metric { metric } + , charge { charge } + , inv_dt { ONE / dt } { + raise::ErrorIf( + (O == 2u and N_GHOSTS < 2), + "Order of interpolation is 2, but number of ghost cells is < 2", + HERE); + } + + Inline auto operator()(prtlidx_t p) const -> void { + auto J_acc = J.access(); + if constexpr (D == Dim::_1D) { + DepositOneParticle(p, + prtls, + metric, + charge, + inv_dt, + [&](int g_i1, int comp, real_t v) { + J_acc(g_i1, comp) += v; + }); + } else if constexpr (D == Dim::_2D) { + DepositOneParticle(p, + prtls, + metric, + charge, + inv_dt, + [&](int g_i1, int g_i2, int comp, real_t v) { + J_acc(g_i1, g_i2, comp) += v; + }); + } else if constexpr (D == Dim::_3D) { + DepositOneParticle( + p, + prtls, + metric, + charge, + inv_dt, + [&](int g_i1, int g_i2, int g_i3, int comp, real_t v) { + J_acc(g_i1, g_i2, g_i3, comp) += v; + }); + } + } + }; + +} // namespace kernel + +#endif // KERNELS_DEPOSITION_CURRENTS_GLOBAL_HPP diff --git a/src/kernels/deposition/currents/single-particle.hpp b/src/kernels/deposition/currents/single-particle.hpp new file mode 100644 index 000000000..d5e12ed92 --- /dev/null +++ b/src/kernels/deposition/currents/single-particle.hpp @@ -0,0 +1,676 @@ +/** + * @file kernels/deposition/currents/single-particle.hpp + * @brief Per-particle current deposition math, shared between the flat and tiled kernels. + * + * @implements + * - kernel::DepositOneParticle<> + * @namespaces: + * - kernel:: + */ + +#ifndef KERNELS_DEPOSITION_CURRENTS_SINGLE_PARTICLE_HPP +#define KERNELS_DEPOSITION_CURRENTS_SINGLE_PARTICLE_HPP + +#include "enums.h" +#include "global.h" + +#include "arch/kokkos_aliases.h" +#include "traits/metric.h" +#include "utils/error.h" +#include "utils/numeric.h" + +#include "framework/containers/particles.h" +#include "kernels/particle_shapes.hpp" + +#include + +#define i_di_to_Xi(I, DI) (static_cast((I)) + static_cast((DI))) + +namespace kernel { + using namespace ntt; + + /** + * @brief Per-particle deposit body, shared between the flat and tiled + * kernels. + * + * The caller supplies a `deposit_at(idx..., comp, val)` callback that + * applies the contribution `val` to the J component `comp` at the + * **global** J cell index `idx...` (already includes the `N_GHOSTS` + * offset). The flat kernel's callback simply does + * `J_acc(idx..., comp) += val` on its scatter-view accessor; the tiled + * kernel's callback translates `idx...` into per-tile scratch + * coordinates and uses `Kokkos::atomic_add` on SLM. Either way, this + * function is identical numerically and contains the only deposit math + * in the codebase. + * + * Dead particles return early. The callback is invoked once per cell + * write, with the dimension-appropriate signature: + * - 1D: `deposit_at(int g_i1, int comp, real_t val)` + * - 2D: `deposit_at(int g_i1, int g_i2, int comp, real_t val)` + * - 3D: `deposit_at(int g_i1, int g_i2, int g_i3, int comp, real_t val)` + */ + template + Inline void DepositOneParticle(prtlidx_t p, + const ParticleArrays& prtls, + const M& metric, + real_t charge, + real_t inv_dt, + DepositFn deposit_at) { + static_assert(O <= 11u, "Shape function order O must be <= 11"); + constexpr auto D = M::Dim; + + if (prtls.tag(p) == ParticleTag::dead) { + return; + } + + // recover particle velocity to deposit in unsimulated direction + [[maybe_unused]] + vec_t vp { ZERO }; + // `vp` only feeds the unsimulated-direction current in the 1D + // (jx2, jx3) and 2D (jx3) branches. In 3D every J component comes + // from the Esirkepov/zigzag charge motion and `vp` is never read, + // so the metric transform + 1/sqrt + NaN/Inf guard below is pure + // dead work there — skip it (also frees xp/inv_energy registers). + if constexpr (D != Dim::_3D) { + coord_t xp { ZERO }; + if constexpr (D == Dim::_1D) { + xp[0] = i_di_to_Xi(prtls.i1(p), prtls.dx1(p)); + } else if constexpr (D == Dim::_2D) { + if constexpr (M::PrtlDim == Dim::_3D) { + xp[0] = i_di_to_Xi(prtls.i1(p), prtls.dx1(p)); + xp[1] = i_di_to_Xi(prtls.i2(p), prtls.dx2(p)); + xp[2] = prtls.phi(p); + } else { + xp[0] = i_di_to_Xi(prtls.i1(p), prtls.dx1(p)); + xp[1] = i_di_to_Xi(prtls.i2(p), prtls.dx2(p)); + } + } else { + xp[0] = i_di_to_Xi(prtls.i1(p), prtls.dx1(p)); + xp[1] = i_di_to_Xi(prtls.i2(p), prtls.dx2(p)); + xp[2] = i_di_to_Xi(prtls.i3(p), prtls.dx3(p)); + } + auto inv_energy { ZERO }; + if constexpr (S == SimEngine::SRPIC) { + metric.template transform_xyz( + xp, + { prtls.ux1(p), prtls.ux2(p), prtls.ux3(p) }, + vp); + inv_energy = ONE / U2GAMMA(prtls.ux1(p), prtls.ux2(p), prtls.ux3(p)); + } else { + coord_t xp_ { ZERO }; + xp_[0] = xp[0]; + real_t theta_Cd { xp[1] }; + const auto theta_Ph { metric.template convert<2, Crd::Cd, Crd::Ph>( + theta_Cd) }; + const auto small_angle { static_cast(constant::SMALL_ANGLE_GR) }; + const auto large_angle { static_cast( + constant::PI - constant::SMALL_ANGLE_GR) }; + if (theta_Ph < small_angle) { + theta_Cd = metric.template convert<2, Crd::Ph, Crd::Cd>(small_angle); + } else if (theta_Ph >= large_angle) { + theta_Cd = metric.template convert<2, Crd::Ph, Crd::Cd>(large_angle); + } + xp_[1] = theta_Cd; + metric.template transform( + xp_, + { prtls.ux1(p), prtls.ux2(p), prtls.ux3(p) }, + vp); + inv_energy = metric.alpha(xp_) / + math::sqrt(ONE + prtls.ux1(p) * vp[0] + + prtls.ux2(p) * vp[1] + prtls.ux3(p) * vp[2]); + } + if (Kokkos::isnan(vp[2]) || Kokkos::isinf(vp[2])) { + vp[2] = ZERO; + } + vp[0] *= inv_energy; + vp[1] *= inv_energy; + vp[2] *= inv_energy; + } + + const real_t coeff { prtls.weight(p) * charge }; + + if constexpr (O == 0u) { + /* + Zig-zag deposit + */ + const auto dxp_r_1 { static_cast(prtls.i1(p) == prtls.i1_prev(p)) * + (prtls.dx1(p) + prtls.dx1_prev(p)) * + static_cast(INV_2) }; + + const real_t Wx1_1 { INV_2 * + (dxp_r_1 + prtls.dx1_prev(p) + + static_cast(prtls.i1(p) > prtls.i1_prev(p))) }; + const real_t Wx1_2 { INV_2 * + (prtls.dx1(p) + dxp_r_1 + + static_cast( + static_cast(prtls.i1(p) > prtls.i1_prev(p)) + + prtls.i1_prev(p) - prtls.i1(p))) }; + const real_t Fx1_1 { (static_cast(prtls.i1(p) > prtls.i1_prev(p)) + + dxp_r_1 - prtls.dx1_prev(p)) * + coeff * inv_dt }; + const real_t Fx1_2 { (static_cast( + prtls.i1(p) - prtls.i1_prev(p) - + static_cast(prtls.i1(p) > prtls.i1_prev(p))) + + prtls.dx1(p) - dxp_r_1) * + coeff * inv_dt }; + + if constexpr (D == Dim::_1D) { + const real_t Fx2_1 { HALF * vp[1] * coeff }; + const real_t Fx2_2 { HALF * vp[1] * coeff }; + + const real_t Fx3_1 { HALF * vp[2] * coeff }; + const real_t Fx3_2 { HALF * vp[2] * coeff }; + + deposit_at(prtls.i1_prev(p) + N_GHOSTS, cur::jx1, Fx1_1); + deposit_at(prtls.i1(p) + N_GHOSTS, cur::jx1, Fx1_2); + + deposit_at(prtls.i1_prev(p) + N_GHOSTS, cur::jx2, Fx2_1 * (ONE - Wx1_1)); + deposit_at(prtls.i1_prev(p) + N_GHOSTS + 1, cur::jx2, Fx2_1 * Wx1_1); + deposit_at(prtls.i1(p) + N_GHOSTS, cur::jx2, Fx2_2 * (ONE - Wx1_2)); + deposit_at(prtls.i1(p) + N_GHOSTS + 1, cur::jx2, Fx2_2 * Wx1_2); + + deposit_at(prtls.i1_prev(p) + N_GHOSTS, cur::jx3, Fx3_1 * (ONE - Wx1_1)); + deposit_at(prtls.i1_prev(p) + N_GHOSTS + 1, cur::jx3, Fx3_1 * Wx1_1); + deposit_at(prtls.i1(p) + N_GHOSTS, cur::jx3, Fx3_2 * (ONE - Wx1_2)); + deposit_at(prtls.i1(p) + N_GHOSTS + 1, cur::jx3, Fx3_2 * Wx1_2); + } else if constexpr (D == Dim::_2D || D == Dim::_3D) { + const auto dxp_r_2 { static_cast(prtls.i2(p) == prtls.i2_prev(p)) * + (prtls.dx2(p) + prtls.dx2_prev(p)) * + static_cast(INV_2) }; + + const real_t Wx2_1 { INV_2 * (dxp_r_2 + prtls.dx2_prev(p) + + static_cast(prtls.i2(p) > + prtls.i2_prev(p))) }; + const real_t Wx2_2 { INV_2 * + (prtls.dx2(p) + dxp_r_2 + + static_cast( + static_cast(prtls.i2(p) > prtls.i2_prev(p)) + + prtls.i2_prev(p) - prtls.i2(p))) }; + const real_t Fx2_1 { (static_cast(prtls.i2(p) > prtls.i2_prev(p)) + + dxp_r_2 - prtls.dx2_prev(p)) * + coeff * inv_dt }; + const real_t Fx2_2 { + (static_cast(prtls.i2(p) - prtls.i2_prev(p) - + static_cast(prtls.i2(p) > prtls.i2_prev(p))) + + prtls.dx2(p) - dxp_r_2) * + coeff * inv_dt + }; + + if constexpr (D == Dim::_2D) { + const real_t Fx3_1 { HALF * vp[2] * coeff }; + const real_t Fx3_2 { HALF * vp[2] * coeff }; + + deposit_at(prtls.i1_prev(p) + N_GHOSTS, + prtls.i2_prev(p) + N_GHOSTS, + cur::jx1, + Fx1_1 * (ONE - Wx2_1)); + deposit_at(prtls.i1_prev(p) + N_GHOSTS, + prtls.i2_prev(p) + N_GHOSTS + 1, + cur::jx1, + Fx1_1 * Wx2_1); + deposit_at(prtls.i1(p) + N_GHOSTS, + prtls.i2(p) + N_GHOSTS, + cur::jx1, + Fx1_2 * (ONE - Wx2_2)); + deposit_at(prtls.i1(p) + N_GHOSTS, + prtls.i2(p) + N_GHOSTS + 1, + cur::jx1, + Fx1_2 * Wx2_2); + + deposit_at(prtls.i1_prev(p) + N_GHOSTS, + prtls.i2_prev(p) + N_GHOSTS, + cur::jx2, + Fx2_1 * (ONE - Wx1_1)); + deposit_at(prtls.i1_prev(p) + N_GHOSTS + 1, + prtls.i2_prev(p) + N_GHOSTS, + cur::jx2, + Fx2_1 * Wx1_1); + deposit_at(prtls.i1(p) + N_GHOSTS, + prtls.i2(p) + N_GHOSTS, + cur::jx2, + Fx2_2 * (ONE - Wx1_2)); + deposit_at(prtls.i1(p) + N_GHOSTS + 1, + prtls.i2(p) + N_GHOSTS, + cur::jx2, + Fx2_2 * Wx1_2); + + deposit_at(prtls.i1_prev(p) + N_GHOSTS, + prtls.i2_prev(p) + N_GHOSTS, + cur::jx3, + Fx3_1 * (ONE - Wx1_1) * (ONE - Wx2_1)); + deposit_at(prtls.i1_prev(p) + N_GHOSTS + 1, + prtls.i2_prev(p) + N_GHOSTS, + cur::jx3, + Fx3_1 * Wx1_1 * (ONE - Wx2_1)); + deposit_at(prtls.i1_prev(p) + N_GHOSTS, + prtls.i2_prev(p) + N_GHOSTS + 1, + cur::jx3, + Fx3_1 * (ONE - Wx1_1) * Wx2_1); + deposit_at(prtls.i1_prev(p) + N_GHOSTS + 1, + prtls.i2_prev(p) + N_GHOSTS + 1, + cur::jx3, + Fx3_1 * Wx1_1 * Wx2_1); + + deposit_at(prtls.i1(p) + N_GHOSTS, + prtls.i2(p) + N_GHOSTS, + cur::jx3, + Fx3_2 * (ONE - Wx1_2) * (ONE - Wx2_2)); + deposit_at(prtls.i1(p) + N_GHOSTS + 1, + prtls.i2(p) + N_GHOSTS, + cur::jx3, + Fx3_2 * Wx1_2 * (ONE - Wx2_2)); + deposit_at(prtls.i1(p) + N_GHOSTS, + prtls.i2(p) + N_GHOSTS + 1, + cur::jx3, + Fx3_2 * (ONE - Wx1_2) * Wx2_2); + deposit_at(prtls.i1(p) + N_GHOSTS + 1, + prtls.i2(p) + N_GHOSTS + 1, + cur::jx3, + Fx3_2 * Wx1_2 * Wx2_2); + } else { + const auto dxp_r_3 { + static_cast(prtls.i3(p) == prtls.i3_prev(p)) * + (prtls.dx3(p) + prtls.dx3_prev(p)) * static_cast(INV_2) + }; + const real_t Wx3_1 { INV_2 * (dxp_r_3 + prtls.dx3_prev(p) + + static_cast( + prtls.i3(p) > prtls.i3_prev(p))) }; + const real_t Wx3_2 { + INV_2 * + (prtls.dx3(p) + dxp_r_3 + + static_cast(static_cast(prtls.i3(p) > prtls.i3_prev(p)) + + prtls.i3_prev(p) - prtls.i3(p))) + }; + const real_t Fx3_1 { (static_cast(prtls.i3(p) > prtls.i3_prev(p)) + + dxp_r_3 - prtls.dx3_prev(p)) * + coeff * inv_dt }; + const real_t Fx3_2 { + (static_cast(prtls.i3(p) - prtls.i3_prev(p) - + static_cast(prtls.i3(p) > prtls.i3_prev(p))) + + prtls.dx3(p) - dxp_r_3) * + coeff * inv_dt + }; + + deposit_at(prtls.i1_prev(p) + N_GHOSTS, + prtls.i2_prev(p) + N_GHOSTS, + prtls.i3_prev(p) + N_GHOSTS, + cur::jx1, + Fx1_1 * (ONE - Wx2_1) * (ONE - Wx3_1)); + deposit_at(prtls.i1_prev(p) + N_GHOSTS, + prtls.i2_prev(p) + N_GHOSTS + 1, + prtls.i3_prev(p) + N_GHOSTS, + cur::jx1, + Fx1_1 * Wx2_1 * (ONE - Wx3_1)); + deposit_at(prtls.i1_prev(p) + N_GHOSTS, + prtls.i2_prev(p) + N_GHOSTS, + prtls.i3_prev(p) + N_GHOSTS + 1, + cur::jx1, + Fx1_1 * (ONE - Wx2_1) * Wx3_1); + deposit_at(prtls.i1_prev(p) + N_GHOSTS, + prtls.i2_prev(p) + N_GHOSTS + 1, + prtls.i3_prev(p) + N_GHOSTS + 1, + cur::jx1, + Fx1_1 * Wx2_1 * Wx3_1); + + deposit_at(prtls.i1(p) + N_GHOSTS, + prtls.i2(p) + N_GHOSTS, + prtls.i3(p) + N_GHOSTS, + cur::jx1, + Fx1_2 * (ONE - Wx2_2) * (ONE - Wx3_2)); + deposit_at(prtls.i1(p) + N_GHOSTS, + prtls.i2(p) + N_GHOSTS + 1, + prtls.i3(p) + N_GHOSTS, + cur::jx1, + Fx1_2 * Wx2_2 * (ONE - Wx3_2)); + deposit_at(prtls.i1(p) + N_GHOSTS, + prtls.i2(p) + N_GHOSTS, + prtls.i3(p) + N_GHOSTS + 1, + cur::jx1, + Fx1_2 * (ONE - Wx2_2) * Wx3_2); + deposit_at(prtls.i1(p) + N_GHOSTS, + prtls.i2(p) + N_GHOSTS + 1, + prtls.i3(p) + N_GHOSTS + 1, + cur::jx1, + Fx1_2 * Wx2_2 * Wx3_2); + + deposit_at(prtls.i1_prev(p) + N_GHOSTS, + prtls.i2_prev(p) + N_GHOSTS, + prtls.i3_prev(p) + N_GHOSTS, + cur::jx2, + Fx2_1 * (ONE - Wx1_1) * (ONE - Wx3_1)); + deposit_at(prtls.i1_prev(p) + N_GHOSTS + 1, + prtls.i2_prev(p) + N_GHOSTS, + prtls.i3_prev(p) + N_GHOSTS, + cur::jx2, + Fx2_1 * Wx1_1 * (ONE - Wx3_1)); + deposit_at(prtls.i1_prev(p) + N_GHOSTS, + prtls.i2_prev(p) + N_GHOSTS, + prtls.i3_prev(p) + N_GHOSTS + 1, + cur::jx2, + Fx2_1 * (ONE - Wx1_1) * Wx3_1); + deposit_at(prtls.i1_prev(p) + N_GHOSTS + 1, + prtls.i2_prev(p) + N_GHOSTS, + prtls.i3_prev(p) + N_GHOSTS + 1, + cur::jx2, + Fx2_1 * Wx1_1 * Wx3_1); + + deposit_at(prtls.i1(p) + N_GHOSTS, + prtls.i2(p) + N_GHOSTS, + prtls.i3(p) + N_GHOSTS, + cur::jx2, + Fx2_2 * (ONE - Wx1_2) * (ONE - Wx3_2)); + deposit_at(prtls.i1(p) + N_GHOSTS + 1, + prtls.i2(p) + N_GHOSTS, + prtls.i3(p) + N_GHOSTS, + cur::jx2, + Fx2_2 * Wx1_2 * (ONE - Wx3_2)); + deposit_at(prtls.i1(p) + N_GHOSTS, + prtls.i2(p) + N_GHOSTS, + prtls.i3(p) + N_GHOSTS + 1, + cur::jx2, + Fx2_2 * (ONE - Wx1_2) * Wx3_2); + deposit_at(prtls.i1(p) + N_GHOSTS + 1, + prtls.i2(p) + N_GHOSTS, + prtls.i3(p) + N_GHOSTS + 1, + cur::jx2, + Fx2_2 * Wx1_2 * Wx3_2); + + deposit_at(prtls.i1_prev(p) + N_GHOSTS, + prtls.i2_prev(p) + N_GHOSTS, + prtls.i3_prev(p) + N_GHOSTS, + cur::jx3, + Fx3_1 * (ONE - Wx1_1) * (ONE - Wx2_1)); + deposit_at(prtls.i1_prev(p) + N_GHOSTS + 1, + prtls.i2_prev(p) + N_GHOSTS, + prtls.i3_prev(p) + N_GHOSTS, + cur::jx3, + Fx3_1 * Wx1_1 * (ONE - Wx2_1)); + deposit_at(prtls.i1_prev(p) + N_GHOSTS, + prtls.i2_prev(p) + N_GHOSTS + 1, + prtls.i3_prev(p) + N_GHOSTS, + cur::jx3, + Fx3_1 * (ONE - Wx1_1) * Wx2_1); + deposit_at(prtls.i1_prev(p) + N_GHOSTS + 1, + prtls.i2_prev(p) + N_GHOSTS + 1, + prtls.i3_prev(p) + N_GHOSTS, + cur::jx3, + Fx3_1 * Wx1_1 * Wx2_1); + + deposit_at(prtls.i1(p) + N_GHOSTS, + prtls.i2(p) + N_GHOSTS, + prtls.i3(p) + N_GHOSTS, + cur::jx3, + Fx3_2 * (ONE - Wx1_2) * (ONE - Wx2_2)); + deposit_at(prtls.i1(p) + N_GHOSTS + 1, + prtls.i2(p) + N_GHOSTS, + prtls.i3(p) + N_GHOSTS, + cur::jx3, + Fx3_2 * Wx1_2 * (ONE - Wx2_2)); + deposit_at(prtls.i1(p) + N_GHOSTS, + prtls.i2(p) + N_GHOSTS + 1, + prtls.i3(p) + N_GHOSTS, + cur::jx3, + Fx3_2 * (ONE - Wx1_2) * Wx2_2); + deposit_at(prtls.i1(p) + N_GHOSTS + 1, + prtls.i2(p) + N_GHOSTS + 1, + prtls.i3(p) + N_GHOSTS, + cur::jx3, + Fx3_2 * Wx1_2 * Wx2_2); + } + } + } else if constexpr ((O >= 1u) and (O <= 11u)) { + + // shape function in dim1 -> always required + real_t iS_x1[O + 2], fS_x1[O + 2]; + // indices of the shape function + int i1_min, i1_max; + + // call shape function + prtl_shape::for_deposit(prtls.i1_prev(p), + static_cast(prtls.dx1_prev(p)), + prtls.i1(p), + static_cast(prtls.dx1(p)), + i1_min, + i1_max, + iS_x1, + fS_x1); + + if constexpr (D == Dim::_1D) { + // (1D): fused Esirkepov, no [O+2] temporaries. + // jx1[i] = -Qdx1dt * sum_{i'=0}^{i} (fS_x1[i'] - iS_x1[i']) + // = -Qdx1dt * P1[i] (Eq. 38, 1D) + // Wx23[i] = HALF * (fS_x1[i] + iS_x1[i]) (computed inline) + const real_t Qdx1dt = coeff * inv_dt; + const real_t QVx2 = coeff * vp[1]; + const real_t QVx3 = coeff * vp[2]; + + // account for ghost cells + i1_min += N_GHOSTS; + i1_max += N_GHOSTS; + + // get number of update indices for asymmetric movement + const int di_x1 = i1_max - i1_min; + + // Current update — fused over the union line so the J cell + // stays L1-resident across the 3 component atomic_adds. + real_t P1 = ZERO; + for (int i = 0; i <= di_x1; ++i) { + P1 += fS_x1[i] - iS_x1[i]; + const int gi = i1_min + i; + const real_t Wx23 = HALF * (fS_x1[i] + iS_x1[i]); + if (i < di_x1) { + deposit_at(gi, cur::jx1, -Qdx1dt * P1); + } + deposit_at(gi, cur::jx2, QVx2 * Wx23); + deposit_at(gi, cur::jx3, QVx3 * Wx23); + } + + } else if constexpr (D == Dim::_2D) { + + // shape function in dim1 -> always required + real_t iS_x2[O + 2], fS_x2[O + 2]; + // indices of the shape function + int i2_min, i2_max; + + // call shape function + prtl_shape::for_deposit(prtls.i2_prev(p), + static_cast(prtls.dx2_prev(p)), + prtls.i2(p), + static_cast(prtls.dx2(p)), + i2_min, + i2_max, + iS_x2, + fS_x2); + + /** + * (2D): fused Esirkepov, no [O+2]^2 temporaries. + * + * Esirkepov 2001 Eq. 38 (simplified) is separable: with + * P1[i] = sum_{i'=0}^{i} (fS_x1[i'] - iS_x1[i']) and + * P2[j] = sum_{j'=0}^{j} (fS_x2[j'] - iS_x2[j']), + * jx1[i][j] = -Q*HALF * P1[i] * (fS_x2[j] + iS_x2[j]) + * jx2[i][j] = -Q*HALF * P2[j] * (fS_x1[i] + iS_x1[i]) + * Wx3[i][j] = THIRD*( fS_x2[j]*(HALF*iS_x1[i]+fS_x1[i]) + * + iS_x2[j]*(HALF*fS_x1[i]+iS_x1[i]) ) + * with Q = coeff*inv_dt (Qdx1dt == Qdx2dt). Same value as the + * old explicit Wx/jx tensors up to FP reassociation; + * charge-conserving by construction. Prefix sums carried as + * running scalars, so the only per-thread state is the + * existing 1D shape arrays. + */ + const real_t QVx3 = coeff * vp[2]; + // -Q*HALF prefactor (Qdx1dt == Qdx2dt == coeff*inv_dt) + const real_t cf = -(coeff * inv_dt) * HALF; + + // account for ghost cells + i1_min += N_GHOSTS; + i2_min += N_GHOSTS; + i1_max += N_GHOSTS; + i2_max += N_GHOSTS; + + // get number of update indices for asymmetric movement + const int di_x1 = i1_max - i1_min; + const int di_x2 = i2_max - i2_min; + + // Current update — fused over the union plane so the J cell + // line stays L1-resident across the 3 component atomic_adds. + real_t P1 = ZERO; + for (int i = 0; i <= di_x1; ++i) { + P1 += fS_x1[i] - iS_x1[i]; + const int gi = i1_min + i; + const real_t iSx1 = iS_x1[i]; + const real_t fSx1 = fS_x1[i]; + const real_t A1 = fSx1 + iSx1; // jx2 cross-factor + real_t P2 = ZERO; + for (int j = 0; j <= di_x2; ++j) { + P2 += fS_x2[j] - iS_x2[j]; + const int gj = i2_min + j; + const real_t iSx2 = iS_x2[j]; + const real_t fSx2 = fS_x2[j]; + if (i < di_x1) { + deposit_at(gi, gj, cur::jx1, cf * P1 * (fSx2 + iSx2)); + } + if (j < di_x2) { + deposit_at(gi, gj, cur::jx2, cf * P2 * A1); + } + const real_t Wx3 = THIRD * (fSx2 * (HALF * iSx1 + fSx1) + + iSx2 * (HALF * fSx1 + iSx1)); + deposit_at(gi, gj, cur::jx3, QVx3 * Wx3); + } + } + + } else if constexpr (D == Dim::_3D) { + // shape function in dim2 + real_t iS_x2[O + 2], fS_x2[O + 2]; + // indices of the shape function + int i2_min, i2_max; + // call shape function + prtl_shape::for_deposit(prtls.i2_prev(p), + static_cast(prtls.dx2_prev(p)), + prtls.i2(p), + static_cast(prtls.dx2(p)), + i2_min, + i2_max, + iS_x2, + fS_x2); + + // shape function in dim3 + real_t iS_x3[O + 2], fS_x3[O + 2]; + // indices of the shape function + int i3_min, i3_max; + + // call shape function + prtl_shape::for_deposit(prtls.i3_prev(p), + static_cast(prtls.dx3_prev(p)), + prtls.i3(p), + static_cast(prtls.dx3(p)), + i3_min, + i3_max, + iS_x3, + fS_x3); + + /** + * fused Esirkepov, no (O+2)^3 temporaries. + * + * The Esirkepov 3D current (2001, Eq. 31) is separable: with + * P1[i] = sum_{i'=0}^{i} (fS_x1[i'] - iS_x1[i']) (and likewise + * P2[j], P3[k]) the cumulative-sum currents collapse to + * + * jx1[i][j][k] = -Q*THIRD * P1[i] * G23(j,k) + * jx2[i][j][k] = -Q*THIRD * P2[j] * H13(i,k) + * jx3[i][j][k] = -Q*THIRD * P3[k] * F12(i,j) + * + * with the 1D-shape cross-factors + * + * G23(j,k) = iS_x2[j]*iS_x3[k] + fS_x2[j]*fS_x3[k] + * + HALF*(iS_x3[k]*fS_x2[j] + iS_x2[j]*fS_x3[k]) + * H13(i,k) = iS_x1[i]*iS_x3[k] + fS_x1[i]*fS_x3[k] + * + HALF*(iS_x3[k]*fS_x1[i] + iS_x1[i]*fS_x3[k]) + * F12(i,j) = iS_x1[i]*iS_x2[j] + fS_x1[i]*fS_x2[j] + * + HALF*(iS_x1[i]*fS_x2[j] + iS_x2[j]*fS_x1[i]) + * + * and Q = coeff*inv_dt (Qdxdt == Qdydt == Qdzdt). This is the + * same value as the old explicit Wx/jx tensors up to + * floating-point reassociation: charge-conserving by + * construction (the Esirkepov decomposition is exact). The + * prefix sums are carried as running scalars in the deposit + * loop, so the only per-thread state is the existing 1D shape + * arrays (no (O+2)^3 / (O+2)^2 locals, hence far fewer VGPRs + * and no private-memory tensor traffic). + */ + + // account for ghost cells + i1_min += N_GHOSTS; + i2_min += N_GHOSTS; + i3_min += N_GHOSTS; + i1_max += N_GHOSTS; + i2_max += N_GHOSTS; + i3_max += N_GHOSTS; + + // get number of update indices for asymmetric movement + const int di_x1 = i1_max - i1_min; + const int di_x2 = i2_max - i2_min; + const int di_x3 = i3_max - i3_min; + + // -Q*THIRD prefactor (Qdxdt == Qdydt == Qdzdt == coeff*inv_dt) + const real_t cf = -(coeff * inv_dt) * THIRD; + + /** + * Current update — fused over the union cube so the J cell + * line stays L1-resident across the 3 component atomic_adds. + * Per-cell branches on (i 11 not supported. Seriously. " + "What are you even doing here? Entity already goes to 11!"); + } + } + +} // namespace kernel + +#undef i_di_to_Xi + +#endif // KERNELS_DEPOSITION_CURRENTS_SINGLE_PARTICLE_HPP diff --git a/src/kernels/deposition/currents/tiled.hpp b/src/kernels/deposition/currents/tiled.hpp new file mode 100644 index 000000000..6e211a89f --- /dev/null +++ b/src/kernels/deposition/currents/tiled.hpp @@ -0,0 +1,523 @@ +/** + * @file kernels/deposition/currents/tiled.hpp + * @brief Tiled current deposition kernel with per-team SLM scratch. + * + * @note Team-policy (one team per spatial tile, accumulates into team SLM scratch with + * atomic adds, then flushes to global J). Available when `team_policy=ON` + * (`#if defined(TEAM_POLICY)`). Stream 2 of the Pattern A plan. + * + * @implements + * - kernel::DepositCurrentsTiled_kernel<> + * @namespaces: + * - kernel:: + */ + +#ifndef KERNELS_DEPOSITION_CURRENTS_TILED_HPP +#define KERNELS_DEPOSITION_CURRENTS_TILED_HPP + +#include "enums.h" +#include "global.h" + +#include "arch/kokkos_aliases.h" +#include "traits/metric.h" +#include "utils/error.h" +#include "utils/numeric.h" + +#include "framework/containers/particles.h" +#include "kernels/deposition/currents/single-particle.hpp" + +#include + +namespace kernel { + using namespace ntt; + + /** + * @brief Tiled current-deposition kernel. + * + * One team per spatial tile (`league_size = ntiles_total`). Each team + * accumulates particle contributions into a per-team scratch buffer of + * shape `(T_TILE + 2*HALO)^D × 3` real_t, where `HALO = O + 1` cells per + * side. Scratch atomics live in SLM (PVC: ~5–10 cycles per + * `atomic_add`); the global J is touched only once per scratch cell at + * flush time. Compared with the flat scatter-view kernel: + * - global atomic pressure ~ (T_TILE + 2*HALO)^D × 3 per tile + * instead of (stencil writes per particle × particles) + * - per-particle stencil writes are tile-local (SLM) instead of + * scattering through global HBM + * + * Supports `O ∈ {0, ..., 11}`. `O == 0` (zigzag) is wired for + * A/B benchmarking against the flat scatter-view kernel — its narrow + * stencil typically makes scratch alloc/zero/flush overhead a + * regression there, but it's good to be able to measure the + * crossover. To revert and use flat for zigzag-only builds, change + * the dispatch in `engines/srpic/currents.h` from + * `#if defined(TEAM_POLICY)` to + * `#if defined(TEAM_POLICY) && (SHAPE_ORDER > 0)`. + * + * Particle iteration order is governed by `tile_offsets`: tile `t` + * owns particles `[tile_offsets(t), tile_offsets(t+1))`, post-sort. + * `SortSpatially` (`particles_sort.cpp`) is responsible for keeping + * the SoA arrays consistent with that. + * + * **Halo sizing and escape valve.** Sort runs at the end of a step + * (see `srpic.hpp`); a particle is pushed once per step thereafter, so + * its `min(i, i_prev)` may differ from the bin key by one cell of drift + * per step elapsed since the last sort. The scratch HALO is + * `STENCIL_REACH(O) + DRIFT`, where `STENCIL_REACH = 2` for zigzag + * (writes `{i_prev, i_prev+1, i, i+1}` ⇒ +2 above `min(i, i_prev)` with + * `|Δi|=1`) and `O` for Esirkepov. `DRIFT` is the `team_policy_drift` + * CMake knob (macro TEAM_POLICY_DRIFT) — the number of cells a particle + * may drift between two sorts that the halo is sized to absorb — and `1` + * by default (the every-step-sorted common case). It is independent of + * the sort cadence, which is set at runtime via `spatial_sorting_interval`; + * particles that drift past the halo take the escape valve below. + * + * Correctness does **not** depend on the halo size. Any particle whose + * full stencil escapes the scratch tile — because it drifted further + * than `DRIFT`, was reordered far from its tile by a no-sort-step + * `CommunicateParticles`, or because the halo is otherwise undersized — + * is deposited *as a whole* via a direct, bounds-clipped + * `Kokkos::atomic_add` on the global J view (the per-particle escape + * valve). Each particle's stencil is therefore deposited exactly once + * (entirely to SLM scratch when it fits, entirely to global J when it + * does not), so the path is charge-conserving; it is merely slower per + * write. Sizing `DRIFT` to the typical between-sort drift keeps the + * common case in fast SLM; sorting less often (or drifting past the + * halo) only costs escape-valve traffic, never accuracy. + * + * **Partition coverage.** The team iteration covers only the particles + * partitioned at the last sort, `[0, layout.npart_partitioned)`, clamped + * to the live `npart`. Particles appended past the partition since the + * sort are not seen here; the launcher (`engines/srpic/currents.h`) + * deposits that tail with the flat kernel so every active particle is + * covered exactly once regardless of sort cadence. + */ + template + class DepositCurrentsTiled_kernel { + static_assert(O <= 11u, "Shape order O must be <= 11"); + static_assert(T_TILE > 0u, "T_TILE must be positive"); + static constexpr auto D = M::Dim; + + /** + * Per-side scratch halo, derived from first principles. + * + * total halo = stencil_reach(O) + drift_between_sort_and_deposit + * + * stencil_reach(O) — maximum cells the deposit writes ABOVE + * min(i, i_prev) under CFL |v * dt/dx| <= 1/2: + * - O == 0 (zigzag): writes { i_prev, i_prev+1, i, i+1 } => +2 + * - O >= 1 Esirkepov: `for_deposit` returns an (O+2)-wide + * array but only O+1 entries are non-zero, and the union + * window satisfies `i_max - i_min <= O+1` (see + * particle_shapes.hpp::for_deposit). The genuine one-sided + * reach above min(i, i_prev) is therefore O, not O+1 — the + * old `O+1` carried one extra cell of conservative padding + * on top of the already-conservative drift term below. + * + * drift — sort runs at end-of-step (see srpic.hpp), so a particle is + * pushed once per step between its last sort and a given deposit. With + * a runtime sort interval of `K` (spatial_sorting_interval), a particle + * drifts at most `K` cells (CFL |v dt/dx| <= 1/2 ⇒ |Δi| <= 1 per step) + * before the next sort. The `team_policy_drift` CMake knob (macro + * TEAM_POLICY_DRIFT) sets DRIFT independently of `K`, sizing the halo so + * a particle that drifts up to DRIFT cells still deposits inside its + * tile scratch. DRIFT defaults to 1 (the sorted-every-step common case); + * any particle that drifts past the halo (e.g. a larger sort interval, + * or a CFL excursion) takes the per-particle global-J escape valve + * below — correct, only slower (see + * the class doc-comment for why this is charge-conserving). + */ + static constexpr int STENCIL_REACH = (O == 0u) ? 2 : static_cast(O); + // One-sided footprint reach for the per-particle escape valve: the + // deposit writes at most this many cells above max(i,i_prev) (and fewer + // below min), so [min - FOOTPRINT_REACH, max + FOOTPRINT_REACH] in cell + // coords conservatively bounds every deposited cell for any order + // (Esirkepov reaches max+O; O=0 zigzag reaches max+1). + static constexpr int FOOTPRINT_REACH = (O == 0u) ? 1 : static_cast(O); +#if defined(TEAM_POLICY_DRIFT) + static constexpr int DRIFT = static_cast(TEAM_POLICY_DRIFT); +#else + static constexpr int DRIFT = 1; +#endif + static constexpr int HALO = STENCIL_REACH + DRIFT; + static constexpr int TE = static_cast(T_TILE) + 2 * HALO; + + using exec_space = Kokkos::DefaultExecutionSpace; + using team_policy = Kokkos::TeamPolicy; + using member_t = typename team_policy::member_type; + + ndfield_t J; + ParticleArrays prtls; + const M metric; + const real_t charge, inv_dt; + + // Tile metadata produced by SortSpatially. + array_t tile_offsets; + ncells_t ntx1 { 1u }, ntx2 { 1u }, ntx3 { 1u }; + ncells_t total_tiles { 0u }; + + /** + * Current active-particle count. `tile_offsets` partitions only the + * particles that existed at the last sort ([0, layout.npart_partitioned)); + * `npart` may differ if the pusher dead-tagged particles in place since. + * Each team clamps its `[tile_offsets(t), tile_offsets(t+1))` slice to + * `npart` so stale slots past the live array are never read. Particles + * appended *beyond* the partition (npart > npart_partitioned) are not seen + * by any team here — the launcher deposits that tail separately. + */ + npart_t npart { 0u }; + + /** + * J's full storage extent including all ghost cells. Used to clip + * the cooperative flush so that a partial tile at the high end of + * the domain does not over-write past the J view. + */ + int j_ext1 { 0 }, j_ext2 { 0 }, j_ext3 { 0 }; + + public: + DepositCurrentsTiled_kernel(const ndfield_t& cur, + const ParticleArrays& prtls, + const M& metric, + real_t charge, + real_t dt, + const TileLayout& layout, + npart_t npart) + : J { cur } + , prtls { prtls } + , metric { metric } + , charge { charge } + , inv_dt { ONE / dt } + , tile_offsets { layout.tile_offsets } + , ntx1 { layout.ntiles_per_axis[0] } + , ntx2 { layout.ntiles_per_axis[1] } + , ntx3 { layout.ntiles_per_axis[2] } + , total_tiles { layout.ntiles_total } + , npart { npart } { + raise::ErrorIf( + layout.tile_size != T_TILE, + "Tiled deposit launched with mismatched T_TILE and runtime tile_size", + HERE); + /** + * @note: HALO is allowed to exceed N_GHOSTS. The cooperative + * scratch→J flush and the per-particle escape valve both bounds-clip + * their writes against `j_ext*` so writes that would land past J's + * ghost stripe are silently dropped (they only ever come from a + * particle whose stencil reaches into the domain ghost region, where + * CommunicateFields will re-supply the contribution). + */ + if constexpr (D == Dim::_1D or D == Dim::_2D or D == Dim::_3D) { + j_ext1 = static_cast(cur.extent(0)); + } + if constexpr (D == Dim::_2D or D == Dim::_3D) { + j_ext2 = static_cast(cur.extent(1)); + } + if constexpr (D == Dim::_3D) { + j_ext3 = static_cast(cur.extent(2)); + } + } + + /** + * @brief Per-team scratch size in bytes. Used by the launcher to set + * `team_policy.set_scratch_size(0, Kokkos::PerTeam(bytes))`. + */ + static constexpr size_t scratch_bytes() { + // The component count (3) is a *static* extent of scratch_ndfield_t + // (View / **[3] / ***[3]), so shmem_size() takes only the + // dynamic spatial extents — passing 3 as well trips Kokkos' + // `rank_dynamic != number of arguments` abort. This matches the + // scratch View construction below, which also omits the 3. + if constexpr (D == Dim::_1D) { + return scratch_ndfield_t::shmem_size(TE); + } else if constexpr (D == Dim::_2D) { + return scratch_ndfield_t::shmem_size(TE, TE); + } else { + return scratch_ndfield_t::shmem_size(TE, TE, TE); + } + } + + Inline void operator()(const member_t& team) const { + const auto tile_id = static_cast(team.league_rank()); + /** + * Tile coordinates (tile-grid indices) → tile origin in **active** + * cell coords (no ghost offset). Using ncells_t to match the linearised + * tile index produced by SortSpatially. + */ + ncells_t tx1 = 0, tx2 = 0, tx3 = 0; + if constexpr (D == Dim::_1D) { + tx1 = tile_id; + } else if constexpr (D == Dim::_2D) { + tx1 = tile_id / ntx2; + tx2 = tile_id - tx1 * ntx2; + } else { + const auto plane = ntx2 * ntx3; + tx1 = tile_id / plane; + const auto rem = tile_id - tx1 * plane; + tx2 = rem / ntx3; + tx3 = rem - tx2 * ntx3; + } + /** + * origin_active = lowest active-cell index in the tile (no ghost). + * origin_J = same value translated into J's storage coordinate + * (i.e. plus N_GHOSTS). + * origin_J_low = J coordinate of scratch index 0 (i.e. origin_J - HALO). + * local index `li` in scratch ↔ global J index `gi = li + origin_J_low`. + */ + const int origin_J1_low = static_cast(tx1 * T_TILE) + + static_cast(N_GHOSTS) - HALO; + const int origin_J2_low = static_cast(tx2 * T_TILE) + + static_cast(N_GHOSTS) - HALO; + const int origin_J3_low = static_cast(tx3 * T_TILE) + + static_cast(N_GHOSTS) - HALO; + + // Allocate scratch and cooperatively zero-fill it. + if constexpr (D == Dim::_1D) { + scratch_ndfield_t scr { team.team_scratch(0), TE }; + Kokkos::parallel_for(Kokkos::TeamThreadRange(team, TE * 3), + [&](ncells_t idx) { + const auto li = idx / 3; + const auto c = idx - li * 3; + scr(li, c) = ZERO; + }); + team.team_barrier(); + + // Clamp the tile's particle slice to the live array: slots past + // `npart` may hold stale (possibly alive-tagged) data from a prior + // step's compaction and must not be re-deposited. + const auto t_lo = tile_offsets(tile_id); + const auto t_hi = tile_offsets(tile_id + 1u); + const auto p_begin = (t_lo < npart) ? t_lo : npart; + const auto p_end = (t_hi < npart) ? t_hi : npart; + Kokkos::parallel_for( + Kokkos::TeamThreadRange(team, p_begin, p_end), + [&](prtlidx_t p) { + /** + * Per-particle escape valve: route the WHOLE particle to the + * global J view when its Esirkepov footprint does not fit + * inside this tile's scratch window [0,TE); only particles + * fully inside the tile touch SLM scratch. A particle drifts + * out of its tile when sorted less often than every step. + * + * The conservative footprint bound in cell coords, + * [min(i,i_prev) - O, max(i,i_prev) + O], covers + * prtl_shape::for_deposit for any order (i_min >= + * min-floor(O/2), i_max <= max+O), so when `to_scratch` is true + * every deposited cell is provably in [0,TE) and the scratch write + * needs no per-cell bounds test. The global path bounds-clips + * against J's storage extent (writes past the ghost stripe are + * re-supplied by SynchronizeFields(J)). + */ + const int i1c = prtls.i1(p), i1p = prtls.i1_prev(p); + const int lo1 = (i1c < i1p ? i1c : i1p) + static_cast(N_GHOSTS) - + FOOTPRINT_REACH - origin_J1_low; + const int hi1 = (i1c > i1p ? i1c : i1p) + static_cast(N_GHOSTS) + + FOOTPRINT_REACH - origin_J1_low; + const bool to_scratch = (lo1 >= 0 and hi1 < TE); + DepositOneParticle( + p, + prtls, + metric, + charge, + inv_dt, + [&](int g_i1, int comp, real_t v) { + if (to_scratch) { + Kokkos::atomic_add(&scr(g_i1 - origin_J1_low, comp), v); + } else if (g_i1 >= 0 and g_i1 < j_ext1) { + // Bounds-clip the escape-valve write against J's storage, + // exactly as the cooperative flush does. Cells past the + // ghost stripe are re-supplied by SynchronizeFields(J); an + // unclipped write here faults the GPU (an escaped boundary + // particle's stencil can reach past j_ext1). + Kokkos::atomic_add(&J(g_i1, comp), v); + } + }); + }); + team.team_barrier(); + + /** + * Cooperative flush of scratch to global J. Bounds-clip against + * the J view extent in case a partial high-end tile (or non-zero + * halo at domain edges) would otherwise write past J. + */ + Kokkos::parallel_for(Kokkos::TeamThreadRange(team, TE * 3), + [&](const int idx) { + const auto li = idx / 3; + const auto c = idx - li * 3; + const auto gi = li + origin_J1_low; + if (gi < 0 or gi >= j_ext1) { + return; + } + const real_t v = scr(li, c); + if (v != ZERO) { + Kokkos::atomic_add(&J(gi, c), v); + } + }); + } else if constexpr (D == Dim::_2D) { + scratch_ndfield_t scr { team.team_scratch(0), TE, TE }; + Kokkos::parallel_for(Kokkos::TeamThreadRange(team, SQR(TE) * 3), + [&](const int idx) { + const auto lij = idx / 3; + const auto c = idx - lij * 3; + const auto li = lij / TE; + const auto lj = lij - li * TE; + scr(li, lj, c) = ZERO; + }); + team.team_barrier(); + + // Clamp the tile's particle slice to the live array: slots past + // `npart` may hold stale (possibly alive-tagged) data from a prior + // step's compaction and must not be re-deposited. + const auto t_lo = tile_offsets(tile_id); + const auto t_hi = tile_offsets(tile_id + 1u); + const auto p_begin = (t_lo < npart) ? t_lo : npart; + const auto p_end = (t_hi < npart) ? t_hi : npart; + Kokkos::parallel_for( + Kokkos::TeamThreadRange(team, p_begin, p_end), + [&](prtlidx_t p) { + // See 1D branch for rationale: route the whole particle to the + // global escape valve unless its full footprint fits in scratch. + const int i1c = prtls.i1(p), i1p = prtls.i1_prev(p); + const int i2c = prtls.i2(p), i2p = prtls.i2_prev(p); + const int lo1 = (i1c < i1p ? i1c : i1p) + static_cast(N_GHOSTS) - + FOOTPRINT_REACH - origin_J1_low; + const int hi1 = (i1c > i1p ? i1c : i1p) + static_cast(N_GHOSTS) + + FOOTPRINT_REACH - origin_J1_low; + const int lo2 = (i2c < i2p ? i2c : i2p) + static_cast(N_GHOSTS) - + FOOTPRINT_REACH - origin_J2_low; + const int hi2 = (i2c > i2p ? i2c : i2p) + static_cast(N_GHOSTS) + + FOOTPRINT_REACH - origin_J2_low; + const bool to_scratch = (lo1 >= 0 and hi1 < TE and lo2 >= 0 and + hi2 < TE); + DepositOneParticle( + p, + prtls, + metric, + charge, + inv_dt, + [&](const int g_i1, const int g_i2, int comp, real_t v) { + if (to_scratch) { + Kokkos::atomic_add( + &scr(g_i1 - origin_J1_low, g_i2 - origin_J2_low, comp), + v); + } else if (g_i1 >= 0 and g_i1 < j_ext1 and g_i2 >= 0 and + g_i2 < j_ext2) { + // Bounds-clip as the cooperative flush does; an unclipped + // escape-valve write faults the GPU when an escaped boundary + // particle's stencil reaches past j_ext. + Kokkos::atomic_add(&J(g_i1, g_i2, comp), v); + } + }); + }); + team.team_barrier(); + + Kokkos::parallel_for(Kokkos::TeamThreadRange(team, SQR(TE) * 3), + [&](const int idx) { + const auto lij = idx / 3; + const auto c = idx - lij * 3; + const auto li = lij / TE; + const auto lj = lij - li * TE; + const auto gi = li + origin_J1_low; + const auto gj = lj + origin_J2_low; + if ((gi < 0 or gi >= j_ext1) or + (gj < 0 or gj >= j_ext2)) { + return; + } + const real_t v = scr(li, lj, c); + if (v != ZERO) { + Kokkos::atomic_add(&J(gi, gj, c), v); + } + }); + } else if constexpr (D == Dim::_3D) { + scratch_ndfield_t scr { team.team_scratch(0), TE, TE, TE }; + Kokkos::parallel_for(Kokkos::TeamThreadRange(team, CUBE(TE) * 3), + [&](const int idx) { + const auto lijk = idx / 3; + const auto c = idx - lijk * 3; + const auto li = lijk / (TE * TE); + const auto rem = lijk - li * TE * TE; + const auto lj = rem / TE; + const auto lk = rem - lj * TE; + scr(li, lj, lk, c) = ZERO; + }); + team.team_barrier(); + + // Clamp the tile's particle slice to the live array: slots past + // `npart` may hold stale (possibly alive-tagged) data from a prior + // step's compaction and must not be re-deposited. + const auto t_lo = tile_offsets(tile_id); + const auto t_hi = tile_offsets(tile_id + 1u); + const auto p_begin = (t_lo < npart) ? t_lo : npart; + const auto p_end = (t_hi < npart) ? t_hi : npart; + Kokkos::parallel_for( + Kokkos::TeamThreadRange(team, p_begin, p_end), + [&](prtlidx_t p) { + // See 1D branch for rationale: route the whole particle to the + // global escape valve unless its full footprint fits in scratch. + const int i1c = prtls.i1(p), i1p = prtls.i1_prev(p); + const int i2c = prtls.i2(p), i2p = prtls.i2_prev(p); + const int i3c = prtls.i3(p), i3p = prtls.i3_prev(p); + const int lo1 = (i1c < i1p ? i1c : i1p) + static_cast(N_GHOSTS) - + FOOTPRINT_REACH - origin_J1_low; + const int hi1 = (i1c > i1p ? i1c : i1p) + static_cast(N_GHOSTS) + + FOOTPRINT_REACH - origin_J1_low; + const int lo2 = (i2c < i2p ? i2c : i2p) + static_cast(N_GHOSTS) - + FOOTPRINT_REACH - origin_J2_low; + const int hi2 = (i2c > i2p ? i2c : i2p) + static_cast(N_GHOSTS) + + FOOTPRINT_REACH - origin_J2_low; + const int lo3 = (i3c < i3p ? i3c : i3p) + static_cast(N_GHOSTS) - + FOOTPRINT_REACH - origin_J3_low; + const int hi3 = (i3c > i3p ? i3c : i3p) + static_cast(N_GHOSTS) + + FOOTPRINT_REACH - origin_J3_low; + const bool to_scratch = (lo1 >= 0 and hi1 < TE and lo2 >= 0 and + hi2 < TE and lo3 >= 0 and hi3 < TE); + DepositOneParticle( + p, + prtls, + metric, + charge, + inv_dt, + [&](const int g_i1, const int g_i2, const int g_i3, int comp, real_t v) { + if (to_scratch) { + Kokkos::atomic_add(&scr(g_i1 - origin_J1_low, + g_i2 - origin_J2_low, + g_i3 - origin_J3_low, + comp), + v); + } else if (g_i1 >= 0 and g_i1 < j_ext1 and g_i2 >= 0 and + g_i2 < j_ext2 and g_i3 >= 0 and g_i3 < j_ext3) { + // Bounds-clip as the cooperative flush does; an unclipped + // escape-valve write faults the GPU when an escaped boundary + // particle's stencil reaches past j_ext. + Kokkos::atomic_add(&J(g_i1, g_i2, g_i3, comp), v); + } + }); + }); + team.team_barrier(); + + Kokkos::parallel_for(Kokkos::TeamThreadRange(team, CUBE(TE) * 3), + [&](const int idx) { + const int lijk = idx / 3; + const int c = idx - lijk * 3; + const int li = lijk / (TE * TE); + const int rem = lijk - li * TE * TE; + const int lj = rem / TE; + const int lk = rem - lj * TE; + const int gi = li + origin_J1_low; + const int gj = lj + origin_J2_low; + const int gk = lk + origin_J3_low; + if ((gi < 0 or gi >= j_ext1) or + (gj < 0 or gj >= j_ext2) or + (gk < 0 or gk >= j_ext3)) { + return; + } + const real_t v = scr(li, lj, lk, c); + if (v != ZERO) { + Kokkos::atomic_add(&J(gi, gj, gk, c), v); + } + }); + } + } + }; + +} // namespace kernel + +#endif // KERNELS_DEPOSITION_CURRENTS_TILED_HPP diff --git a/tests/kernels/deposit.cpp b/tests/kernels/deposit.cpp index 10095d304..9589b0f70 100644 --- a/tests/kernels/deposit.cpp +++ b/tests/kernels/deposit.cpp @@ -11,7 +11,7 @@ #include "metrics/qspherical.h" #include "metrics/spherical.h" -#include "kernels/currents_deposit.hpp" +#include "kernels/deposition/currents/global.hpp" #include #include diff --git a/tests/kernels/deposit_tiled.cpp b/tests/kernels/deposit_tiled.cpp index f058d6f30..90a3890bf 100644 --- a/tests/kernels/deposit_tiled.cpp +++ b/tests/kernels/deposit_tiled.cpp @@ -21,7 +21,8 @@ #include "metrics/minkowski.h" -#include "kernels/currents_deposit.hpp" +#include "kernels/deposition/currents/global.hpp" +#include "kernels/deposition/currents/tiled.hpp" #include #include @@ -35,12 +36,6 @@ namespace { using namespace ntt; - void errorIf(bool condition, const std::string& msg) { - if (condition) { - throw std::runtime_error(msg); - } - } - template void put_value(const array_t& arr, T value, int i) { auto h = Kokkos::create_mirror_view(arr); @@ -87,13 +82,12 @@ namespace { ncells_t ntx2, ncells_t tx1, ncells_t tx2) { - const ncells_t total_tiles = ntx1 * ntx2; - const ncells_t hot_tile = tx1 * ntx2 + tx2; + const ncells_t total_tiles = ntx1 * ntx2; + const ncells_t hot_tile = tx1 * ntx2 + tx2; array_t offsets("tile_offsets", total_tiles + 1u); - auto h = Kokkos::create_mirror_view(offsets); + auto h = Kokkos::create_mirror_view(offsets); for (ncells_t t = 0; t <= total_tiles; ++t) { - h(t) = (t <= hot_tile) ? static_cast(0) - : static_cast(1); + h(t) = (t <= hot_tile) ? static_cast(0) : static_cast(1); } Kokkos::deep_copy(offsets, h); return offsets; @@ -213,15 +207,17 @@ namespace { template void run_one_case() { - using metric_t = metric::Minkowski; + using metric_t = metric::Minkowski; constexpr unsigned short nx1 = 50u, nx2 = 50u; - metric_t metric { { nx1, nx2 }, + metric_t metric { + { nx1, nx2 }, { { 0.0, 55.0 }, { 0.0, 55.0 } }, - {} }; + {} + }; // Particle setup (mirrors deposit.cpp). const int i0 = 25, j0 = 21, i0f = 24, j0f = 20; - const real_t uz = 2.5; + const real_t uz = 2.5; const prtldx_t dxi = static_cast(0.65); const prtldx_t dxf = static_cast(0.99); const prtldx_t dyi = static_cast(0.65); @@ -272,13 +268,27 @@ namespace { 10, kernel::DepositCurrents_kernel( J_scat, - pack_arrays(i1, i2, i3, - i1_prev, i2_prev, i3_prev, - dx1, dx2, dx3, - dx1_prev, dx2_prev, dx3_prev, - ux1, ux2, ux3, - phi, weight, tag), - metric, charge, dt)); + pack_arrays(i1, + i2, + i3, + i1_prev, + i2_prev, + i3_prev, + dx1, + dx2, + dx3, + dx1_prev, + dx2_prev, + dx3_prev, + ux1, + ux2, + ux3, + phi, + weight, + tag), + metric, + charge, + dt)); Kokkos::Experimental::contribute(J_flat, J_scat); Kokkos::fence("flat deposit done"); } @@ -305,29 +315,40 @@ namespace { layout.ntiles_per_axis[2] = 1u; layout.ntiles_total = ntx1 * ntx2; layout.tile_size = T_TILE; - layout.tile_offsets = build_tile_offsets_single_particle(ntx1, - ntx2, - tx1, - tx2); + layout.tile_offsets = build_tile_offsets_single_particle(ntx1, ntx2, tx1, tx2); using kernel_t = kernel::DepositCurrentsTiled_kernel; // npart = full slot count (10): the lone alive particle sits in slot 0 // and the per-tile slice clamp keeps the (dead) tail out. kernel_t kern { J_tiled, - pack_arrays(i1, i2, i3, - i1_prev, i2_prev, i3_prev, - dx1, dx2, dx3, - dx1_prev, dx2_prev, dx3_prev, - ux1, ux2, ux3, - phi, weight, tag), - metric, charge, dt, layout, + pack_arrays(i1, + i2, + i3, + i1_prev, + i2_prev, + i3_prev, + dx1, + dx2, + dx3, + dx1_prev, + dx2_prev, + dx3_prev, + ux1, + ux2, + ux3, + phi, + weight, + tag), + metric, + charge, + dt, + layout, static_cast(10) }; Kokkos::TeamPolicy<> policy(static_cast(layout.ntiles_total), Kokkos::AUTO); - policy.set_scratch_size(0, - Kokkos::PerTeam(kernel_t::scratch_bytes())); + policy.set_scratch_size(0, Kokkos::PerTeam(kernel_t::scratch_bytes())); Kokkos::parallel_for("TiledDeposit", policy, kern); Kokkos::fence("tiled deposit done"); } @@ -338,8 +359,8 @@ namespace { Kokkos::deep_copy(h_flat, J_flat); Kokkos::deep_copy(h_tiled, J_tiled); - const real_t eps = static_cast(1.0e-5); - real_t max_diff = ZERO; + const real_t eps = static_cast(1.0e-5); + real_t max_diff = ZERO; int fail_count = 0; for (ncells_t i = 0; i < h_flat.extent(0); ++i) { for (ncells_t j = 0; j < h_flat.extent(1); ++j) { @@ -353,9 +374,8 @@ namespace { } if (diff > eps * math::max(mag, static_cast(1.0))) { if (fail_count < 5) { - std::cerr << " J(" << i << "," << j << ",c=" << c - << ") flat=" << a << " tiled=" << b - << " diff=" << diff << '\n'; + std::cerr << " J(" << i << "," << j << ",c=" << c << ") flat=" << a + << " tiled=" << b << " diff=" << diff << '\n'; } ++fail_count; } @@ -364,9 +384,8 @@ namespace { } if (fail_count > 0) { std::cerr << "X-1 deposit_tiled equivalence FAILED for O=" << O - << " T_TILE=" << T_TILE - << " : " << fail_count << " mismatches; max_diff=" << max_diff - << '\n'; + << " T_TILE=" << T_TILE << " : " << fail_count + << " mismatches; max_diff=" << max_diff << '\n'; throw std::logic_error("DepositCurrentsTiled_kernel mismatch"); } std::cerr << "X-1 deposit_tiled OK O=" << O << " T_TILE=" << T_TILE @@ -404,34 +423,37 @@ namespace { << ", build has " << N_GHOSTS << ")\n"; return; } - using metric_t = metric::Minkowski; + using metric_t = metric::Minkowski; constexpr unsigned short nx1 = 50u, nx2 = 50u; - metric_t metric { { nx1, nx2 }, { { 0.0, 55.0 }, { 0.0, 55.0 } }, {} }; + metric_t metric { + { nx1, nx2 }, + { { 0.0, 55.0 }, { 0.0, 55.0 } }, + {} + }; - constexpr int n_slots = 64; - constexpr int n_base = 5; + constexpr int n_slots = 64; + constexpr int n_base = 5; const int bases[n_base] = { 1, 13, 25, 37, 48 }; const int n_alive = n_base * n_base; // 25 - array_t i1 { "i1", n_slots }, i2 { "i2", n_slots }, - i3 { "i3", n_slots }; - array_t i1_prev { "i1_prev", n_slots }, + array_t i1 { "i1", n_slots }, i2 { "i2", n_slots }, i3 { "i3", n_slots }; + array_t i1_prev { "i1_prev", n_slots }, i2_prev { "i2_prev", n_slots }, i3_prev { "i3_prev", n_slots }; array_t dx1 { "dx1", n_slots }, dx2 { "dx2", n_slots }, dx3 { "dx3", n_slots }; array_t dx1_prev { "dx1_prev", n_slots }, dx2_prev { "dx2_prev", n_slots }, dx3_prev { "dx3_prev", n_slots }; - array_t ux1 { "ux1", n_slots }, ux2 { "ux2", n_slots }, + array_t ux1 { "ux1", n_slots }, ux2 { "ux2", n_slots }, ux3 { "ux3", n_slots }; - array_t phi { "phi", n_slots }, weight { "weight", n_slots }; - array_t tag { "tag", n_slots }; - const real_t charge = 1.0, dt = 1.0; + array_t phi { "phi", n_slots }, weight { "weight", n_slots }; + array_t tag { "tag", n_slots }; + const real_t charge = 1.0, dt = 1.0; // Fill alive particles on host (slots >= n_alive stay zero == dead). - auto h_i1 = Kokkos::create_mirror_view(i1); - auto h_i2 = Kokkos::create_mirror_view(i2); - auto h_i1p = Kokkos::create_mirror_view(i1_prev); - auto h_i2p = Kokkos::create_mirror_view(i2_prev); + auto h_i1 = Kokkos::create_mirror_view(i1); + auto h_i2 = Kokkos::create_mirror_view(i2); + auto h_i1p = Kokkos::create_mirror_view(i1_prev); + auto h_i2p = Kokkos::create_mirror_view(i2_prev); auto h_dx1 = Kokkos::create_mirror_view(dx1); auto h_dx2 = Kokkos::create_mirror_view(dx2); auto h_dx1p = Kokkos::create_mirror_view(dx1_prev); @@ -439,13 +461,13 @@ namespace { auto h_ux3 = Kokkos::create_mirror_view(ux3); auto h_w = Kokkos::create_mirror_view(weight); auto h_tag = Kokkos::create_mirror_view(tag); - int p = 0; + int p = 0; for (int a = 0; a < n_base; ++a) { for (int b = 0; b < n_base; ++b, ++p) { - h_i1p(p) = bases[a]; - h_i1(p) = bases[a] - 1; - h_i2p(p) = bases[b]; - h_i2(p) = bases[b] - 1; + h_i1p(p) = bases[a]; + h_i1(p) = bases[a] - 1; + h_i2p(p) = bases[b]; + h_i2(p) = bases[b] - 1; h_dx1p(p) = static_cast(0.65); h_dx1(p) = static_cast(0.99); h_dx2p(p) = static_cast(0.65); @@ -478,13 +500,27 @@ namespace { n_slots, kernel::DepositCurrents_kernel( J_scat, - pack_arrays(i1, i2, i3, - i1_prev, i2_prev, i3_prev, - dx1, dx2, dx3, - dx1_prev, dx2_prev, dx3_prev, - ux1, ux2, ux3, - phi, weight, tag), - metric, charge, dt)); + pack_arrays(i1, + i2, + i3, + i1_prev, + i2_prev, + i3_prev, + dx1, + dx2, + dx3, + dx1_prev, + dx2_prev, + dx3_prev, + ux1, + ux2, + ux3, + phi, + weight, + tag), + metric, + charge, + dt)); Kokkos::Experimental::contribute(J_flat, J_scat); Kokkos::fence("flat drift deposit done"); } @@ -515,19 +551,33 @@ namespace { // tile 0, so the team must walk [0, n_alive) and route the drifted // ones to the global-J escape valve. kernel_t kern { J_tiled, - pack_arrays(i1, i2, i3, - i1_prev, i2_prev, i3_prev, - dx1, dx2, dx3, - dx1_prev, dx2_prev, dx3_prev, - ux1, ux2, ux3, - phi, weight, tag), - metric, charge, dt, layout, + pack_arrays(i1, + i2, + i3, + i1_prev, + i2_prev, + i3_prev, + dx1, + dx2, + dx3, + dx1_prev, + dx2_prev, + dx3_prev, + ux1, + ux2, + ux3, + phi, + weight, + tag), + metric, + charge, + dt, + layout, static_cast(n_alive) }; Kokkos::TeamPolicy<> policy(static_cast(layout.ntiles_total), Kokkos::AUTO); - policy.set_scratch_size(0, - Kokkos::PerTeam(kernel_t::scratch_bytes())); + policy.set_scratch_size(0, Kokkos::PerTeam(kernel_t::scratch_bytes())); Kokkos::parallel_for("TiledDepositDrift", policy, kern); Kokkos::fence("tiled drift deposit done"); } diff --git a/tests/kernels/particle_moments.cpp b/tests/kernels/particle_moments.cpp index cbda7546f..d70fe713a 100644 --- a/tests/kernels/particle_moments.cpp +++ b/tests/kernels/particle_moments.cpp @@ -80,8 +80,6 @@ void testParticleMoments(const std::vector& res, 0, 0 }; - const float mass = 1.0; - const float charge = 1.0; const bool use_weights = false; const real_t inv_n0 = 1.0; From b2e2e39389446c807c45fd87ee56121048e5b8de Mon Sep 17 00:00:00 2001 From: haykh Date: Tue, 8 Sep 2026 22:36:35 -0400 Subject: [PATCH 071/125] formatting --- cmake/defaults.cmake | 10 +-- src/engines/engine.hpp | 15 +++-- src/engines/reporter.cpp | 13 ++-- src/framework/containers/fields_io.cpp | 15 +++-- src/framework/containers/particles.h | 1 + src/framework/containers/particles_sort.cpp | 59 ++++++++---------- src/framework/domain/metadomain.h | 4 +- src/framework/domain/metadomain_loadbal.cpp | 69 ++++++++++----------- src/framework/parameters/algorithms.cpp | 2 +- src/framework/parameters/parameters.cpp | 21 +++---- src/global/global.cpp | 7 +-- src/global/utils/diag.cpp | 2 +- src/global/utils/reporter.cpp | 4 +- src/global/utils/sort_dispatch.h | 58 ++++++++--------- src/global/utils/sorting.h | 12 ++-- src/kernels/pushers/sr_policies.h | 4 +- src/output/writer.cpp | 17 +++-- tests/framework/CMakeLists.txt | 4 +- tests/framework/particles_sort.cpp | 19 +++--- tests/framework/sort_by_key.cpp | 3 +- 20 files changed, 157 insertions(+), 182 deletions(-) diff --git a/cmake/defaults.cmake b/cmake/defaults.cmake index c03ebaf81..888be1b00 100644 --- a/cmake/defaults.cmake +++ b/cmake/defaults.cmake @@ -108,12 +108,12 @@ if(DEFINED ENV{Entity_ENABLE_VENDOR_SORT}) set(default_vendor_sort $ENV{Entity_ENABLE_VENDOR_SORT} CACHE INTERNAL - "Default flag for vendor sort_by_key (oneDPL/Thrust/rocThrust)") + "Default flag for vendor sort_by_key (oneDPL/Thrust/rocThrust)") else() set(default_vendor_sort ON CACHE INTERNAL - "Default flag for vendor sort_by_key (oneDPL/Thrust/rocThrust)") + "Default flag for vendor sort_by_key (oneDPL/Thrust/rocThrust)") endif() set_property(CACHE default_vendor_sort PROPERTY TYPE BOOL) @@ -123,5 +123,7 @@ set(default_team_policy_tile_size set(default_team_policy_drift 1 - CACHE INTERNAL - "Default tiled-deposit scratch halo drift for team_policy (cells between sorts)") + CACHE + INTERNAL + "Default tiled-deposit scratch halo drift for team_policy (cells between sorts)" +) diff --git a/src/engines/engine.hpp b/src/engines/engine.hpp index 161339a9d..63a5d1859 100644 --- a/src/engines/engine.hpp +++ b/src/engines/engine.hpp @@ -83,10 +83,10 @@ namespace ntt { const real_t dt; const std::size_t team_policy_team_size; const timestep_t max_steps; - const timestep_t start_step; - const simtime_t start_time; - simtime_t time; - timestep_t step; + const timestep_t start_step; + const simtime_t start_time; + simtime_t time; + timestep_t step; public: static constexpr Dimension D { M::Dim }; @@ -253,9 +253,8 @@ namespace ntt { "ParticlePusher", "FieldBoundaries", "ParticleBoundaries", "Communications", "Injector", "Custom", - "LoadBalance", - "ParticleSort", "Output", - "Checkpoint" }, + "LoadBalance", "ParticleSort", + "Output", "Checkpoint" }, []() { Kokkos::fence(); }, @@ -268,7 +267,7 @@ namespace ntt { const auto clear_interval = m_params.template get( "particles.clear_interval"); - const auto lb_enable = m_params.template get( + const auto lb_enable = m_params.template get( "simulation.domain.load_balance.enable"); const auto lb_interval = m_params.template get( "simulation.domain.load_balance.interval"); diff --git a/src/engines/reporter.cpp b/src/engines/reporter.cpp index fcd46fb81..2f19d500c 100644 --- a/src/engines/reporter.cpp +++ b/src/engines/reporter.cpp @@ -41,13 +41,12 @@ namespace ntt { "algorithms.deposit.team_policy_team_size") == 0u) { reporter::AddParam(report, 4, "Team size", "%s", "AUTO (Kokkos)"); } else { - reporter::AddParam( - report, - 4, - "Team size", - "%d (requested; clamped to backend max at launch)", - static_cast(params.template get( - "algorithms.deposit.team_policy_team_size"))); + reporter::AddParam(report, + 4, + "Team size", + "%d (requested; clamped to backend max at launch)", + static_cast(params.template get( + "algorithms.deposit.team_policy_team_size"))); } #endif reporter::AddParam(report, 4, "Metric", "%s", M.to_string()); diff --git a/src/framework/containers/fields_io.cpp b/src/framework/containers/fields_io.cpp index a17ab1430..85d302c57 100644 --- a/src/framework/containers/fields_io.cpp +++ b/src/framework/containers/fields_io.cpp @@ -62,11 +62,10 @@ namespace ntt { } template - void Fields::CheckpointWrite( - adios2::IO& io, - adios2::Engine& writer, - const std::vector& local_shape, - const std::vector& local_offset) const { + void Fields::CheckpointWrite(adios2::IO& io, + adios2::Engine& writer, + const std::vector& local_shape, + const std::vector& local_offset) const { logger::Checkpoint("Writing fields checkpoint", HERE); // Per-rank slab: re-set the variable selection to track the (possibly @@ -95,9 +94,9 @@ namespace ntt { template void Fields::CheckpointRead(adios2::IO&, \ adios2::Engine&, \ const adios2::Box&); \ - template void Fields::CheckpointWrite(adios2::IO&, \ - adios2::Engine&, \ - const std::vector&, \ + template void Fields::CheckpointWrite(adios2::IO&, \ + adios2::Engine&, \ + const std::vector&, \ const std::vector&) const; FIELDS_CHECKPOINTS(Dim::_1D, SimEngine::SRPIC) diff --git a/src/framework/containers/particles.h b/src/framework/containers/particles.h index 3409b2f6f..bdbd15a86 100644 --- a/src/framework/containers/particles.h +++ b/src/framework/containers/particles.h @@ -337,6 +337,7 @@ namespace ntt { ((defined(SYCL_ENABLED) && defined(ONEDPL_ENABLED)) || \ (defined(CUDA_ENABLED) && defined(THRUST_ENABLED)) || \ (defined(HIP_ENABLED) && defined(ROCTHRUST_ENABLED))) + private: /** * @brief Apply a particle-index permutation (built by oneDPL/Thrust diff --git a/src/framework/containers/particles_sort.cpp b/src/framework/containers/particles_sort.cpp index 6795ab288..4e45b8db9 100644 --- a/src/framework/containers/particles_sort.cpp +++ b/src/framework/containers/particles_sort.cpp @@ -10,8 +10,8 @@ #if defined(TEAM_POLICY) #if (defined(SYCL_ENABLED) && defined(ONEDPL_ENABLED)) || \ - (defined(CUDA_ENABLED) && defined(THRUST_ENABLED)) || \ - (defined(HIP_ENABLED) && defined(ROCTHRUST_ENABLED)) + (defined(CUDA_ENABLED) && defined(THRUST_ENABLED)) || \ + (defined(HIP_ENABLED) && defined(ROCTHRUST_ENABLED)) #define TEAM_POLICY_USE_VENDOR_SORT #include "utils/sort_dispatch.h" #endif @@ -222,10 +222,7 @@ namespace ntt { } template - inline void reserve_scratch_2d(V& v, - const char* label, - npart_t n, - npart_t ncols) { + inline void reserve_scratch_2d(V& v, const char* label, npart_t n, npart_t ncols) { if (static_cast(v.extent(0)) < n or static_cast(v.extent(1)) != ncols) { v = V {}; @@ -236,10 +233,9 @@ namespace ntt { #if defined(TEAM_POLICY) template - void Particles::compute_tile_offsets( - const array_t& tile_indices, - ncells_t total_tiles, - npart_t npart_local) { + void Particles::compute_tile_offsets(const array_t& tile_indices, + ncells_t total_tiles, + npart_t npart_local) { // Compute the per-tile prefix-sum `tile_offsets` for the tiled // pusher from the (already sorted) `tile_indices` — monotonically // non-decreasing for alive particles, with the dead sentinel @@ -280,7 +276,7 @@ namespace ntt { } Kokkos::deep_copy(tile_offsets, h_offsets); - m_tile_layout.tile_offsets = tile_offsets; + m_tile_layout.tile_offsets = tile_offsets; // tile_offsets(total_tiles) is the alive-particle count at sort time: // the tiles partition exactly [0, npart_partitioned). The deposit // launcher compares this against the live npart() to detect (and @@ -300,8 +296,7 @@ namespace ntt { return; } - constexpr unsigned short T = static_cast( - TEAM_POLICY_TILE_SIZE); + constexpr unsigned short T = static_cast(TEAM_POLICY_TILE_SIZE); static_assert(T > 0u, "TEAM_POLICY_TILE_SIZE must be > 0"); // 1. Compute per-axis tile counts and total_tiles. @@ -325,8 +320,8 @@ namespace ntt { } // 2. Compute per-particle tile key (with min(i, i_prev)). -#if defined(TEAM_POLICY_USE_VENDOR_SORT) && \ - defined(SYCL_ENABLED) && defined(ONEDPL_ENABLED) + #if defined(TEAM_POLICY_USE_VENDOR_SORT) && defined(SYCL_ENABLED) && \ + defined(ONEDPL_ENABLED) // oneDPL sorts the keys in place, so reuse a persistent, grow-only keys // buffer instead of allocating a fresh one every sort. `tile_indices` // aliases the (possibly over-capacity) persistent buffer; downstream @@ -334,23 +329,23 @@ namespace ntt { // `tile_indices.extent(0)`. reserve_scratch_1d(m_sort_keys, "tile_indices", npart_local); array_t tile_indices = m_sort_keys; -#else + #else array_t tile_indices { "tile_indices", npart_local }; -#endif + #endif Kokkos::parallel_for( "FillTileIndices", rangeActiveParticles(), sort::PositionToTileIndex { i1, - i2, - i3, - tag, - tile_indices, - ncells_active, - static_cast(T), - array_t {}, - i1_prev, - i2_prev, - i3_prev }); + i2, + i3, + tag, + tile_indices, + ncells_active, + static_cast(T), + array_t {}, + i1_prev, + i2_prev, + i3_prev }); // 3. Sort. Vendor library (oneDPL/Thrust) when compiled in; // Kokkos::BinSort otherwise. n_bins = total_tiles + 2 covers @@ -603,9 +598,7 @@ namespace ntt { }); Kokkos::fence("permute_2d_into: gather"); Kokkos::deep_copy( - Kokkos::subview(arr, - std::make_pair(static_cast(0), n), - Kokkos::ALL), + Kokkos::subview(arr, std::make_pair(static_cast(0), n), Kokkos::ALL), Kokkos::subview(scratch, std::make_pair(static_cast(0), n), Kokkos::ALL)); @@ -704,9 +697,9 @@ namespace ntt { #endif // TEAM_POLICY_USE_VENDOR_SORT #if defined(TEAM_POLICY_USE_VENDOR_SORT) - #define APPLY_PERM_INSTANTIATE(D, C) \ - template void Particles::apply_permutation_to_soa( \ - const prtl_perm_t&, npart_t); + #define APPLY_PERM_INSTANTIATE(D, C) \ + template void Particles::apply_permutation_to_soa(const prtl_perm_t&, \ + npart_t); #else #define APPLY_PERM_INSTANTIATE(D, C) #endif diff --git a/src/framework/domain/metadomain.h b/src/framework/domain/metadomain.h index ed35a7140..ede688cfb 100644 --- a/src/framework/domain/metadomain.h +++ b/src/framework/domain/metadomain.h @@ -144,9 +144,7 @@ namespace ntt { * @note Only neighbor communication is used (CommunicateFields ghosts + * CommunicateParticles). */ - void Rebalance(unsigned int dim_mask, - real_t tolerance, - ncells_t max_shift_cells); + void Rebalance(unsigned int dim_mask, real_t tolerance, ncells_t max_shift_cells); /* output-related ------------------------------------------------------- */ #if defined(OUTPUT_ENABLED) diff --git a/src/framework/domain/metadomain_loadbal.cpp b/src/framework/domain/metadomain_loadbal.cpp index a7bf97873..5f98f82b2 100644 --- a/src/framework/domain/metadomain_loadbal.cpp +++ b/src/framework/domain/metadomain_loadbal.cpp @@ -240,15 +240,15 @@ namespace ntt { const auto& off = g_domain_offsets[idx]; const auto ncells = g_subdomains[idx].mesh.n_active(); for (auto d { 0u }; d < M::Dim; ++d) { - ncells_per_pos[d][off[d]] = ncells[d]; - load_per_pos[d][off[d]] += static_cast(npart_per_dom[idx]); + ncells_per_pos[d][off[d]] = ncells[d]; + load_per_pos[d][off[d]] += static_cast(npart_per_dom[idx]); } } /* --- 3. Diffusion-style boundary shifts per balanced dim ------------- */ std::vector> new_ncells_per_pos = ncells_per_pos; const auto MIN_NCELLS = static_cast(2 * N_GHOSTS + 4); - bool any_shift = false; + bool any_shift = false; for (auto d { 0u }; d < M::Dim; ++d) { if ((dim_mask & (1u << d)) == 0u) { @@ -260,19 +260,18 @@ namespace ntt { } double total_load = 0.0; - double max_load = 0.0; - double min_load = std::numeric_limits::infinity(); + double max_load = 0.0; + double min_load = std::numeric_limits::infinity(); for (auto p { 0u }; p < N; ++p) { total_load += load_per_pos[d][p]; - max_load = std::max(max_load, load_per_pos[d][p]); - min_load = std::min(min_load, load_per_pos[d][p]); + max_load = std::max(max_load, load_per_pos[d][p]); + min_load = std::min(min_load, load_per_pos[d][p]); } if (total_load <= 0.0) { continue; } const auto mean = total_load / static_cast(N); - if (((max_load - min_load) / mean) < - static_cast(tolerance)) { + if (((max_load - min_load) / mean) < static_cast(tolerance)) { continue; } @@ -282,30 +281,28 @@ namespace ntt { std::vector bnd_shift(N + 1, 0); const int cap = static_cast(max_shift_cells); for (auto k { 1u }; k < N; ++k) { - const auto l_load = load_per_pos[d][k - 1]; - const auto r_load = load_per_pos[d][k]; - const auto l_density = (ncells_per_pos[d][k - 1] > 0) - ? l_load / static_cast( - ncells_per_pos[d][k - 1]) - : 0.0; - const auto r_density = (ncells_per_pos[d][k] > 0) - ? r_load / static_cast( - ncells_per_pos[d][k]) - : 0.0; + const auto l_load = load_per_pos[d][k - 1]; + const auto r_load = load_per_pos[d][k]; + const auto l_density = (ncells_per_pos[d][k - 1] > 0) + ? l_load / + static_cast(ncells_per_pos[d][k - 1]) + : 0.0; + const auto r_density = (ncells_per_pos[d][k] > 0) + ? r_load / + static_cast(ncells_per_pos[d][k]) + : 0.0; const auto avg_density = std::max(0.5 * (l_density + r_density), 1.0); // Move boundary towards the lighter side. Halve the gradient so that // a single sweep does roughly one diffusion step. - int shift = static_cast( + int shift = static_cast( std::round(0.5 * (l_load - r_load) / avg_density)); - shift = std::clamp(shift, -cap, cap); + shift = std::clamp(shift, -cap, cap); // Don't shrink either side below MIN_NCELLS. const int max_pos = static_cast(ncells_per_pos[d][k - 1]) - static_cast(MIN_NCELLS); const int max_neg = static_cast(ncells_per_pos[d][k]) - static_cast(MIN_NCELLS); - shift = std::clamp(shift, - -std::max(max_neg, 0), - std::max(max_pos, 0)); + shift = std::clamp(shift, -std::max(max_neg, 0), std::max(max_pos, 0)); bnd_shift[k] = shift; if (shift != 0) { any_shift = true; @@ -314,8 +311,7 @@ namespace ntt { // new_ncells[p] = ncells[p] + bnd_shift[p] - bnd_shift[p+1] for (auto p { 0u }; p < N; ++p) { new_ncells_per_pos[d][p] = static_cast( - static_cast(ncells_per_pos[d][p]) + bnd_shift[p] - - bnd_shift[p + 1]); + static_cast(ncells_per_pos[d][p]) + bnd_shift[p] - bnd_shift[p + 1]); } } @@ -351,11 +347,11 @@ namespace ntt { extent_per_pos[d].resize(N); ncells_t running { 0 }; for (auto p { 0u }; p < N; ++p) { - new_offset_per_pos[d][p] = running; - const auto x_lo = face_phys(d, static_cast(running)); - running += new_ncells_per_pos[d][p]; - const auto x_hi = face_phys(d, static_cast(running)); - extent_per_pos[d][p] = { x_lo, x_hi }; + new_offset_per_pos[d][p] = running; + const auto x_lo = face_phys(d, static_cast(running)); + running += new_ncells_per_pos[d][p]; + const auto x_hi = face_phys(d, static_cast(running)); + extent_per_pos[d][p] = { x_lo, x_hi }; } } @@ -365,7 +361,7 @@ namespace ntt { auto em_old_h = Kokkos::create_mirror_view(local_dom.fields.em); Kokkos::deep_copy(em_old_h, local_dom.fields.em); - auto em0_old_h = decltype(Kokkos::create_mirror_view(local_dom.fields.em0)) {}; + auto em0_old_h = decltype(Kokkos::create_mirror_view(local_dom.fields.em0)) {}; auto cur0_old_h = decltype(Kokkos::create_mirror_view(local_dom.fields.cur0)) {}; if constexpr (S == SimEngine::GRPIC) { em0_old_h = Kokkos::create_mirror_view(local_dom.fields.em0); @@ -378,8 +374,8 @@ namespace ntt { std::vector new_local_ncells(M::Dim); std::vector new_local_offset(M::Dim); for (unsigned int idx { 0 }; idx < g_ndomains; ++idx) { - auto& sub = g_subdomains[idx]; - const auto& off_ndoms = g_domain_offsets[idx]; + auto& sub = g_subdomains[idx]; + const auto& off_ndoms = g_domain_offsets[idx]; std::vector ncells_d(M::Dim); std::vector offset_d(M::Dim); boundaries_t ext_d; @@ -459,8 +455,9 @@ namespace ntt { } CommunicateParticles(local_dom); - logger::Checkpoint("Rebalance: domains shifted, fields and particles redistributed", - HERE); + logger::Checkpoint( + "Rebalance: domains shifted, fields and particles redistributed", + HERE); #endif // MPI_ENABLED } diff --git a/src/framework/parameters/algorithms.cpp b/src/framework/parameters/algorithms.cpp index 87c77fbd9..797cfb4f7 100644 --- a/src/framework/parameters/algorithms.cpp +++ b/src/framework/parameters/algorithms.cpp @@ -32,7 +32,7 @@ namespace ntt { defaults::current_filters); deposit_enable = toml::find_or(toml_data, "algorithms", "deposit", "enable", true); - deposit_order = static_cast(SHAPE_ORDER); + deposit_order = static_cast(SHAPE_ORDER); deposit_team_policy_team_size = toml::find_or(toml_data, "algorithms", "deposit", diff --git a/src/framework/parameters/parameters.cpp b/src/framework/parameters/parameters.cpp index 8b525d49a..f3d8d507e 100644 --- a/src/framework/parameters/parameters.cpp +++ b/src/framework/parameters/parameters.cpp @@ -265,21 +265,20 @@ namespace ntt { static_cast(N_GHOSTS))); { // dimensions: list of 1/2/3 mapped to a bitmask - const auto dim_ints = toml::find_or>( - toml_data, - "simulation", - "domain", - "load_balance", - "dimensions", - std::vector { 1 }); - unsigned int mask = 0u; + const auto dim_ints = toml::find_or>(toml_data, + "simulation", + "domain", + "load_balance", + "dimensions", + std::vector { 1 }); + unsigned int mask = 0u; for (const auto& d : dim_ints) { if (d == 1 or d == 2 or d == 3) { mask |= 1u << (d - 1); } else { - raise::Error( - "simulation.domain.load_balance.dimensions: unknown dim, expected 1/2/3", - HERE); + raise::Error("simulation.domain.load_balance.dimensions: unknown " + "dim, expected 1/2/3", + HERE); } } set("simulation.domain.load_balance.dim_mask", mask); diff --git a/src/global/global.cpp b/src/global/global.cpp index c3aa1bcdc..4ce7fd251 100644 --- a/src/global/global.cpp +++ b/src/global/global.cpp @@ -34,14 +34,11 @@ namespace { return; } hipMemPool_t pool = nullptr; - if (hipDeviceGetDefaultMemPool(&pool, device) != hipSuccess or - pool == nullptr) { + if (hipDeviceGetDefaultMemPool(&pool, device) != hipSuccess or pool == nullptr) { return; } uint64_t threshold = UINT64_MAX; - (void)hipMemPoolSetAttribute(pool, - hipMemPoolAttrReleaseThreshold, - &threshold); + (void)hipMemPoolSetAttribute(pool, hipMemPoolAttrReleaseThreshold, &threshold); } #endif // HIP_ENABLED } // namespace diff --git a/src/global/utils/diag.cpp b/src/global/utils/diag.cpp index 604a872ae..298225eda 100644 --- a/src/global/utils/diag.cpp +++ b/src/global/utils/diag.cpp @@ -63,7 +63,7 @@ namespace diag { const std::size_t tot_npart = std::accumulate(mpi_npart.begin(), mpi_npart.end(), static_cast(0)); - const npart_t max_idx = std::distance( + const npart_t max_idx = std::distance( mpi_npart.begin(), std::max_element(mpi_npart.begin(), mpi_npart.end())); const npart_t min_idx = std::distance( diff --git a/src/global/utils/reporter.cpp b/src/global/utils/reporter.cpp index 6113a89c7..d889ad560 100644 --- a/src/global/utils/reporter.cpp +++ b/src/global/utils/reporter.cpp @@ -253,8 +253,8 @@ namespace reporter { #if defined(TEAM_POLICY) AddParam(report, 4, "TEAM_POLICY", "%s", "ON"); - #if (defined(SYCL_ENABLED) && defined(ONEDPL_ENABLED)) || \ - (defined(CUDA_ENABLED) && defined(THRUST_ENABLED)) || \ + #if (defined(SYCL_ENABLED) && defined(ONEDPL_ENABLED)) || \ + (defined(CUDA_ENABLED) && defined(THRUST_ENABLED)) || \ (defined(HIP_ENABLED) && defined(ROCTHRUST_ENABLED)) AddParam(report, 4, "VENDOR_SORT", "%s", "ON"); #else diff --git a/src/global/utils/sort_dispatch.h b/src/global/utils/sort_dispatch.h index 075267539..e445a45ad 100644 --- a/src/global/utils/sort_dispatch.h +++ b/src/global/utils/sort_dispatch.h @@ -81,9 +81,8 @@ namespace ntt::sort_helpers { // bits when total_tiles ~ 176K). Returns at least 1. inline unsigned int significant_bits(ncells_t n_bins) { unsigned int bits = 0u; - while (bits < 32u && - (static_cast(1u) << bits) < - static_cast(n_bins)) { + while (bits < 32u && (static_cast(1u) << bits) < + static_cast(n_bins)) { ++bits; } return (bits == 0u) ? 1u : bits; @@ -123,15 +122,15 @@ namespace ntt::sort_helpers { inline void sort_by_key_dispatch(const array_t& keys, prtl_perm_t& perm, ncells_t /*n_bins*/, - npart_t n, + npart_t n, ::sort::backend::OneDPL) { if (n == 0u) { return; } - auto* keys_ptr = keys.data(); - auto* perm_ptr = perm.data(); - auto exec = Kokkos::DefaultExecutionSpace(); - auto perm_v = perm; + auto* keys_ptr = keys.data(); + auto* perm_ptr = perm.data(); + auto exec = Kokkos::DefaultExecutionSpace(); + auto perm_v = perm; Kokkos::parallel_for( "PermInitIota", n, @@ -181,14 +180,14 @@ namespace ntt::sort_helpers { cub::DoubleBuffer d_perm(perm.data(), perm_out.data()); std::size_t temp_bytes = 0; - auto err = cub::DeviceRadixSort::SortPairs(nullptr, - temp_bytes, - d_keys, - d_perm, - n, - 0, - end_bit, - stream); + auto err = cub::DeviceRadixSort::SortPairs(nullptr, + temp_bytes, + d_keys, + d_perm, + n, + 0, + end_bit, + stream); raise::ErrorIf(err != cudaSuccess, "cub::DeviceRadixSort::SortPairs (size query) failed", HERE); @@ -202,9 +201,7 @@ namespace ntt::sort_helpers { 0, end_bit, stream); - raise::ErrorIf(err != cudaSuccess, - "cub::DeviceRadixSort::SortPairs failed", - HERE); + raise::ErrorIf(err != cudaSuccess, "cub::DeviceRadixSort::SortPairs failed", HERE); exec.fence("sort_by_key_dispatch Thrust: post-sort"); // Publish results from whichever buffer cub left as Current() (depends on @@ -239,9 +236,9 @@ namespace ntt::sort_helpers { n, KOKKOS_LAMBDA(const npart_t i) { perm_v(i) = i; }); - array_t keys_out("tile_keys_sorted", n); - prtl_perm_t perm_out("tile_perm_sorted", n); - const unsigned int end_bit = significant_bits(n_bins); + array_t keys_out("tile_keys_sorted", n); + prtl_perm_t perm_out("tile_perm_sorted", n); + const unsigned int end_bit = significant_bits(n_bins); exec.fence("sort_by_key_dispatch Rocthrust: pre-sort"); auto stream = exec.hip_stream(); @@ -256,7 +253,7 @@ namespace ntt::sort_helpers { rocprim::double_buffer d_perm(perm.data(), perm_out.data()); std::size_t temp_bytes = 0; - auto err = rocprim::radix_sort_pairs(nullptr, + auto err = rocprim::radix_sort_pairs(nullptr, temp_bytes, d_keys, d_perm, @@ -277,9 +274,7 @@ namespace ntt::sort_helpers { 0u, end_bit, stream); - raise::ErrorIf(err != hipSuccess, - "rocprim::radix_sort_pairs failed", - HERE); + raise::ErrorIf(err != hipSuccess, "rocprim::radix_sort_pairs failed", HERE); exec.fence("sort_by_key_dispatch Rocthrust: post-sort"); // Publish results from whichever buffer rocprim left as `current()` @@ -305,15 +300,12 @@ namespace ntt::sort_helpers { if (n == 0u) { return; } - auto keys_h = Kokkos::create_mirror_view_and_copy(Kokkos::HostSpace(), - keys); + auto keys_h = Kokkos::create_mirror_view_and_copy(Kokkos::HostSpace(), keys); auto perm_h = Kokkos::create_mirror_view(perm); std::iota(perm_h.data(), perm_h.data() + n, npart_t { 0u }); - std::stable_sort(perm_h.data(), - perm_h.data() + n, - [&](npart_t a, npart_t b) { - return keys_h(a) < keys_h(b); - }); + std::stable_sort(perm_h.data(), perm_h.data() + n, [&](npart_t a, npart_t b) { + return keys_h(a) < keys_h(b); + }); Kokkos::deep_copy(perm, perm_h); } diff --git a/src/global/utils/sorting.h b/src/global/utils/sorting.h index 8442f5ddb..006c0bf27 100644 --- a/src/global/utils/sorting.h +++ b/src/global/utils/sorting.h @@ -99,7 +99,7 @@ namespace sort { // ~1.07e9, the linearised `tile_indices(p)` overflows past `n_bins`, // and BinSort's internal `atomic_add(&bin_count[wild_idx], 1)` // faults on an unmapped page. - int ncells1 { 1 }, ncells2 { 1 }, ncells3 { 1 }; + int ncells1 { 1 }, ncells2 { 1 }, ncells3 { 1 }; PositionToTileIndex(const array_t& i1_, const array_t& i2_, @@ -109,9 +109,9 @@ namespace sort { const std::vector& ncells, ncells_t tile_size_ = 1u, const array_t& num_ppt_ = { "num_ppt", 0u }, - const array_t& i1_prev_ = {}, - const array_t& i2_prev_ = {}, - const array_t& i3_prev_ = {}) + const array_t& i1_prev_ = {}, + const array_t& i2_prev_ = {}, + const array_t& i3_prev_ = {}) : i1 { i1_ } , i2 { i2_ } , i3 { i3_ } @@ -238,9 +238,13 @@ namespace sort { // availability of the corresponding vendor library. namespace backend { struct OneDPL {}; + struct Thrust {}; + struct Rocthrust {}; + struct StdSort {}; + // Always-available legacy fallback using Kokkos::BinSort. struct BinSort {}; } // namespace backend diff --git a/src/kernels/pushers/sr_policies.h b/src/kernels/pushers/sr_policies.h index efcef1c78..cc34698a5 100644 --- a/src/kernels/pushers/sr_policies.h +++ b/src/kernels/pushers/sr_policies.h @@ -166,8 +166,8 @@ namespace kernel::sr { ntt::EmissionTypeFlag emission_type, bool atm, F&& callback) { - constexpr bool has_emission = ::traits::pgen::HasEmissionPolicy; - constexpr bool has_cpu = ::traits::pgen::HasCustomPrtlUpdate; + constexpr bool has_emission = ::traits::pgen::HasEmissionPolicy; + constexpr bool has_cpu = ::traits::pgen::HasCustomPrtlUpdate; constexpr bool has_extfields = ::traits::pgen::HasExternalFields; auto with_emission = [&](auto next) { diff --git a/src/output/writer.cpp b/src/output/writer.cpp index 5b9ae89e2..50630bbcc 100644 --- a/src/output/writer.cpp +++ b/src/output/writer.cpp @@ -450,18 +450,15 @@ namespace out { // m_flds_l_corner_dwn / m_flds_l_shape_dwn are reversed for non-LayoutRight // (see defineMeshLayout / setLocalLayout); m_flds_l_corner / m_flds_l_shape // / m_flds_g_shape are not. Map the dim-order index to the dwn-array index. - constexpr bool layout_right = std::is_same< - typename ndfield_t::array_layout, - Kokkos::LayoutRight>::value; - const auto i_dwn = layout_right - ? static_cast(dim) - : (m_flds_g_shape.size() - 1u - - static_cast(dim)); + constexpr bool layout_right = std::is_same::array_layout, + Kokkos::LayoutRight>::value; + const auto i_dwn = layout_right ? static_cast(dim) + : (m_flds_g_shape.size() - 1u - + static_cast(dim)); const auto is_last = (m_flds_l_corner[dim] + m_flds_l_shape[dim] == m_flds_g_shape[dim]); - varc.SetSelection(adios2::Box( - { m_flds_l_corner_dwn[i_dwn] }, - { m_flds_l_shape_dwn[i_dwn] })); + varc.SetSelection(adios2::Box({ m_flds_l_corner_dwn[i_dwn] }, + { m_flds_l_shape_dwn[i_dwn] })); vare.SetSelection(adios2::Box( { m_flds_l_corner_dwn[i_dwn] }, { m_flds_l_shape_dwn[i_dwn] + (is_last ? 1ul : 0ul) })); diff --git a/tests/framework/CMakeLists.txt b/tests/framework/CMakeLists.txt index 9a2e2865e..818ea0f2c 100644 --- a/tests/framework/CMakeLists.txt +++ b/tests/framework/CMakeLists.txt @@ -44,8 +44,8 @@ else() gen_test(particles_sort false) endif() -# team_policy X-3: per-backend sort_by_key permutation test (only built -# when the compile-time team_policy toggle is on). +# team_policy X-3: per-backend sort_by_key permutation test (only built when the +# compile-time team_policy toggle is on). if(${team_policy}) gen_test(sort_by_key false) endif() diff --git a/tests/framework/particles_sort.cpp b/tests/framework/particles_sort.cpp index a2b82456e..591485220 100644 --- a/tests/framework/particles_sort.cpp +++ b/tests/framework/particles_sort.cpp @@ -102,8 +102,8 @@ auto main(int argc, char* argv[]) -> int { #else const ncells_t T = 1u; #endif - const auto na = grid.n_active(); - const ncells_t ntx2 = (na[1] + T - 1u) / T; + const auto na = grid.n_active(); + const ncells_t ntx2 = (na[1] + T - 1u) / T; const auto tile_of = [&](int a, int b) -> ncells_t { return (static_cast(a) / T) * ntx2 + (static_cast(b) / T); @@ -159,11 +159,10 @@ auto main(int argc, char* argv[]) -> int { raise::ErrorIf(pld_r_h(p, 1) != weight_h(p) + static_cast(10.5), "error in sorting particle real payload 1", HERE); - raise::ErrorIf( - pld_i_h(p, 0) != - static_cast(weight_h(p) + static_cast(10.0)), - "error in sorting particle integer payload 0", - HERE); + raise::ErrorIf(pld_i_h(p, 0) != static_cast( + weight_h(p) + static_cast(10.0)), + "error in sorting particle integer payload 0", + HERE); } raise::ErrorIf(n_alive_obs != 59u, "wrong number of alive particles after sort", @@ -267,9 +266,9 @@ auto main(int argc, char* argv[]) -> int { #else const ncells_t T = 1u; #endif - const auto na = grid.n_active(); - const ncells_t ntx2 = (na[1] + T - 1u) / T; - const ncells_t ntx3 = (na[2] + T - 1u) / T; + const auto na = grid.n_active(); + const ncells_t ntx2 = (na[1] + T - 1u) / T; + const ncells_t ntx3 = (na[2] + T - 1u) / T; const auto tile_of = [&](int a, int b, int c) -> ncells_t { return ((static_cast(a) / T) * ntx2 + (static_cast(b) / T)) * diff --git a/tests/framework/sort_by_key.cpp b/tests/framework/sort_by_key.cpp index 9ccecc732..64c4eb180 100644 --- a/tests/framework/sort_by_key.cpp +++ b/tests/framework/sort_by_key.cpp @@ -75,8 +75,7 @@ namespace { } for (npart_t i = 0u; i < n; ++i) { raise::ErrorIf(seen[i] != 1, - std::string("permutation not a bijection for backend ") + - label, + std::string("permutation not a bijection for backend ") + label, HERE); } From db342bc040db8b3595012bda4ba5dfadb780a26e Mon Sep 17 00:00:00 2001 From: haykh Date: Thu, 10 Sep 2026 08:04:37 -0400 Subject: [PATCH 072/125] moved mpi & team_policy cmake to separate files --- CMakeLists.txt | 176 ++++----------------------------------- cmake/dependencies.cmake | 2 +- cmake/mpi.cmake | 71 ++++++++++++++++ cmake/team_policy.cmake | 76 +++++++++++++++++ 4 files changed, 166 insertions(+), 159 deletions(-) create mode 100644 cmake/mpi.cmake create mode 100644 cmake/team_policy.cmake diff --git a/CMakeLists.txt b/CMakeLists.txt index 3b82fca9b..d9d1c0ce4 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -59,22 +59,26 @@ set(gpu_aware_mpi CACHE BOOL "Enable GPU-aware MPI") set(team_policy - ${default_team_policy} - CACHE BOOL "Enable team_policy tile-blocked deposit/pusher kernels") + ${default_team_policy} + CACHE BOOL "Enable team_policy tile-blocked deposit/pusher kernels") set(team_policy_tile_size - ${default_team_policy_tile_size} - CACHE STRING "team_policy tile edge length in cells") + ${default_team_policy_tile_size} + CACHE STRING "team_policy tile edge length in cells") set(team_policy_tile_sizes - "4;6;8;10;12;14;16" - CACHE STRING "team_policy tile-size choices") + "4;6;8;10;12;14;16" + CACHE STRING "team_policy tile-size choices") set(team_policy_drift - ${default_team_policy_drift} - CACHE STRING - "team_policy tiled-deposit scratch halo drift in cells (max cells a particle may move between two sorts). Sizes the deposit scratch halo only; the sort cadence is set at runtime via spatial_sorting_interval. Default 1.") + ${default_team_policy_drift} + CACHE + STRING + "team_policy tiled-deposit scratch halo drift in cells (max cells a particle may move between two sorts). Sizes the deposit scratch halo only; the sort cadence is set at runtime via spatial_sorting_interval. Default 1." +) set(vendor_sort - ${default_vendor_sort} - CACHE BOOL - "Use the vendor sort_by_key (oneDPL/Thrust/rocThrust) for the team_policy spatial sort when available. OFF forces the Kokkos::BinSort fallback, which sorts each SoA member in place (lower peak memory, no maxnpart gather buffer) at the cost of sort speed.") + ${default_vendor_sort} + CACHE + BOOL + "Use the vendor sort_by_key (oneDPL/Thrust/rocThrust) for the team_policy spatial sort when available. OFF forces the Kokkos::BinSort fallback, which sorts each SoA member in place (lower peak memory, no maxnpart gather buffer) at the cost of sort speed." +) # -------------------------- Compilation settings -------------------------- # set(CMAKE_CXX_STANDARD 20) @@ -156,156 +160,12 @@ endif() # ------------------------------ team_policy wiring ------------------------ # if(${team_policy}) - list(FIND team_policy_tile_sizes "${team_policy_tile_size}" _tps_idx) - if(_tps_idx EQUAL -1) - message(FATAL_ERROR - "${Red}team_policy_tile_size must be one of ${team_policy_tile_sizes}, " - "got '${team_policy_tile_size}'${ColorReset}") - endif() - add_compile_options("-D TEAM_POLICY") - add_compile_options("-D TEAM_POLICY_TILE_SIZE=${team_policy_tile_size}") - - # Compile-time tiled-deposit scratch halo drift. Sizes the halo so a - # particle that drifts up to DRIFT cells between two sorts still deposits - # inside its tile scratch; particles drifting further take the - # per-particle global-J escape valve (correct, only slower). This is - # independent of the sort cadence, which is set at runtime via - # `spatial_sorting_interval`. Defaults to 1 (the sorted-every-step case). - add_compile_options("-D TEAM_POLICY_DRIFT=${team_policy_drift}") - - # Vendor sort: oneDPL on SYCL, Thrust on CUDA, rocThrust/rocprim on HIP. - # When `vendor_sort` is ON (default) the available library is detected - # and used; the spatial sort then builds a single permutation that - # gathers all SoA members. When `vendor_sort` is OFF, or no library is - # found, the code falls back to Kokkos::BinSort, which sorts each member - # in place -- lower peak memory and no maxnpart gather buffer, at the - # cost of sort speed (negligible when sorting is a small fraction of the - # step). The `vendor_sort` knob lets you force the BinSort fallback even - # when a vendor library is present. - if(${vendor_sort}) - if("${Kokkos_DEVICES}" MATCHES "SYCL") - find_package(oneDPL QUIET) - if(oneDPL_FOUND) - message(STATUS "team_policy: oneDPL found, enabling SYCL sort_by_key") - add_compile_options("-D ONEDPL_ENABLED") - set(DEPENDENCIES ${DEPENDENCIES} oneDPL) - else() - message(STATUS "team_policy: oneDPL not found; using BinSort fallback " - "for SYCL sort_by_key") - endif() - endif() - - if("${Kokkos_DEVICES}" MATCHES "CUDA") - find_package(Thrust QUIET) - if(Thrust_FOUND) - message(STATUS "team_policy: Thrust enabled for CUDA sort_by_key") - add_compile_options("-D THRUST_ENABLED") - else() - message(STATUS "team_policy: Thrust not found; using BinSort fallback " - "for CUDA sort_by_key") - endif() - endif() - - if("${Kokkos_DEVICES}" MATCHES "HIP") - # rocThrust ships with ROCm. The HIP sort_by_key path uses rocprim's - # bounded-bit radix sort directly (rocprim is rocThrust's own - # dependency, so its headers come in transitively; we find it - # explicitly to keep the include path robust). This builds a single - # permutation that gathers all SoA members, instead of the legacy - # per-member Kokkos::BinSort path which allocates a fresh - # `sorted_values` buffer for every member every step (the dominant - # source of allocator churn / fragmentation on ROCm). - find_package(rocthrust QUIET) - if(rocthrust_FOUND) - message(STATUS "team_policy: rocThrust enabled for HIP sort_by_key") - add_compile_options("-D ROCTHRUST_ENABLED") - set(DEPENDENCIES ${DEPENDENCIES} roc::rocthrust) - find_package(rocprim QUIET) - if(rocprim_FOUND) - set(DEPENDENCIES ${DEPENDENCIES} roc::rocprim) - endif() - else() - message(STATUS "team_policy: rocThrust not found; using BinSort " - "fallback for HIP sort_by_key") - endif() - endif() - else() - message(STATUS "team_policy: vendor_sort=OFF; forcing Kokkos::BinSort " - "fallback for spatial sort_by_key") - endif() + include(${CMAKE_CURRENT_SOURCE_DIR}/cmake/team_policy.cmake) endif() # MPI if(${mpi}) - find_or_fetch_dependency(MPI FALSE REQUIRED) - include_directories(${MPI_CXX_INCLUDE_PATH}) - add_compile_options("-D MPI_ENABLED") - set(DEPENDENCIES ${DEPENDENCIES} MPI::MPI_CXX) - if(${DEVICE_ENABLED}) - if(${gpu_aware_mpi}) - add_compile_options("-D GPU_AWARE_MPI") - - # On Cray systems (e.g. Frontier) GPU-aware Cray MPICH can only - # handle device pointers if the GPU Transport Layer (GTL) library - # is linked. The Cray compiler wrappers (cc/CC) inject this - # automatically, but we build with hipcc/nvcc directly, so - # find_package(MPI) only finds base libmpi and the GTL is left - # out -> MPI_Sendrecv on a device pointer fails with - # "OFI ... Bad address". Add it explicitly here. - # - # Cray PE exports PE_MPICH_GTL_DIR_ / PE_MPICH_GTL_LIBS_ - # (e.g. amd_gfx90a -> -lmpi_gtl_hsa). Their absence means this is - # not a Cray MPICH build, in which case nothing extra is needed. - if("${Kokkos_DEVICES}" MATCHES "HIP") - set(_gtl_accels amd_gfx942 amd_gfx940 amd_gfx90a amd_gfx908 amd_gfx906) - elseif("${Kokkos_DEVICES}" MATCHES "CUDA") - set(_gtl_accels nvidia90 nvidia80 nvidia70) - elseif("${Kokkos_DEVICES}" MATCHES "SYCL") - set(_gtl_accels ponteVecchio) - else() - set(_gtl_accels "") - endif() - - set(_gtl_dir "") - set(_gtl_libflag "") - foreach(_accel ${_gtl_accels}) - if((NOT _gtl_dir) AND (DEFINED ENV{PE_MPICH_GTL_DIR_${_accel}})) - # strip the leading "-L" from the Cray-provided value - string(REGEX REPLACE "^-L" "" - _gtl_dir "$ENV{PE_MPICH_GTL_DIR_${_accel}}") - string(REGEX REPLACE "^-l" "" - _gtl_libflag "$ENV{PE_MPICH_GTL_LIBS_${_accel}}") - endif() - endforeach() - - if(_gtl_dir AND _gtl_libflag) - find_library(MPI_GTL_LIBRARY - NAMES ${_gtl_libflag} - HINTS "${_gtl_dir}" - NO_DEFAULT_PATH) - if(MPI_GTL_LIBRARY) - message(STATUS - "GPU-aware MPI: linking Cray GTL library ${MPI_GTL_LIBRARY}") - set(DEPENDENCIES ${DEPENDENCIES} ${MPI_GTL_LIBRARY}) - else() - message(FATAL_ERROR - "${Red}gpu_aware_mpi=ON: Cray MPICH detected but the GTL " - "library 'lib${_gtl_libflag}' was not found in '${_gtl_dir}'. " - "GPU-aware MPI will crash at runtime without it. Make sure the " - "craype-accel module is loaded, or build with gpu_aware_mpi=OFF." - "${ColorReset}") - endif() - else() - message(STATUS - "GPU-aware MPI: no Cray GTL environment found; assuming the MPI " - "implementation is GPU-aware without an extra transport library.") - endif() - endif() - else() - set(gpu_aware_mpi - OFF - CACHE BOOL "Use explicit copy when using MPI + GPU") - endif() + include(${CMAKE_CURRENT_SOURCE_DIR}/cmake/mpi.cmake) endif() # Output diff --git a/cmake/dependencies.cmake b/cmake/dependencies.cmake index 2024702b5..93a8a17da 100644 --- a/cmake/dependencies.cmake +++ b/cmake/dependencies.cmake @@ -4,7 +4,7 @@ set(Kokkos_REPOSITORY https://github.com/kokkos/kokkos.git CACHE STRING "Kokkos repository") set(Kokkos_TAG - 5.0.1 + 5.2.1 CACHE STRING "Kokkos tag") set(adios2_REPOSITORY https://github.com/ornladios/ADIOS2.git diff --git a/cmake/mpi.cmake b/cmake/mpi.cmake new file mode 100644 index 000000000..389faf4c8 --- /dev/null +++ b/cmake/mpi.cmake @@ -0,0 +1,71 @@ +find_or_fetch_dependency(MPI FALSE REQUIRED) +include_directories(${MPI_CXX_INCLUDE_PATH}) +add_compile_options("-D MPI_ENABLED") +set(DEPENDENCIES ${DEPENDENCIES} MPI::MPI_CXX) +if(${DEVICE_ENABLED}) + if(${gpu_aware_mpi}) + add_compile_options("-D GPU_AWARE_MPI") + + # On Cray systems (e.g. Frontier) GPU-aware Cray MPICH can only handle + # device pointers if the GPU Transport Layer (GTL) library is linked. The + # Cray compiler wrappers (cc/CC) inject this automatically, but we build + # with hipcc/nvcc directly, so find_package(MPI) only finds base libmpi and + # the GTL is left out -> MPI_Sendrecv on a device pointer fails with "OFI + # ... Bad address". Add it explicitly here. + # + # Cray PE exports PE_MPICH_GTL_DIR_ / PE_MPICH_GTL_LIBS_ (e.g. + # amd_gfx90a -> -lmpi_gtl_hsa). Their absence means this is not a Cray MPICH + # build, in which case nothing extra is needed. + if("${Kokkos_DEVICES}" MATCHES "HIP") + set(_gtl_accels amd_gfx942 amd_gfx940 amd_gfx90a amd_gfx908 amd_gfx906) + elseif("${Kokkos_DEVICES}" MATCHES "CUDA") + set(_gtl_accels nvidia90 nvidia80 nvidia70) + elseif("${Kokkos_DEVICES}" MATCHES "SYCL") + set(_gtl_accels ponteVecchio) + else() + set(_gtl_accels "") + endif() + + set(_gtl_dir "") + set(_gtl_libflag "") + foreach(_accel ${_gtl_accels}) + if((NOT _gtl_dir) AND (DEFINED ENV{PE_MPICH_GTL_DIR_${_accel}})) + # strip the leading "-L" from the Cray-provided value + string(REGEX REPLACE "^-L" "" _gtl_dir + "$ENV{PE_MPICH_GTL_DIR_${_accel}}") + string(REGEX REPLACE "^-l" "" _gtl_libflag + "$ENV{PE_MPICH_GTL_LIBS_${_accel}}") + endif() + endforeach() + + if(_gtl_dir AND _gtl_libflag) + find_library( + MPI_GTL_LIBRARY + NAMES ${_gtl_libflag} + HINTS "${_gtl_dir}" + NO_DEFAULT_PATH) + if(MPI_GTL_LIBRARY) + message( + STATUS "GPU-aware MPI: linking Cray GTL library ${MPI_GTL_LIBRARY}") + set(DEPENDENCIES ${DEPENDENCIES} ${MPI_GTL_LIBRARY}) + else() + message( + FATAL_ERROR + "${Red}gpu_aware_mpi=ON: Cray MPICH detected but the GTL " + "library 'lib${_gtl_libflag}' was not found in '${_gtl_dir}'. " + "GPU-aware MPI will crash at runtime without it. Make sure the " + "craype-accel module is loaded, or build with gpu_aware_mpi=OFF." + "${ColorReset}") + endif() + else() + message( + STATUS "GPU-aware MPI: no Cray GTL environment found; assuming the MPI " + "implementation is GPU-aware without an extra transport library." + ) + endif() + endif() +else() + set(gpu_aware_mpi + OFF + CACHE BOOL "Use explicit copy when using MPI + GPU") +endif() diff --git a/cmake/team_policy.cmake b/cmake/team_policy.cmake new file mode 100644 index 000000000..1217bc28c --- /dev/null +++ b/cmake/team_policy.cmake @@ -0,0 +1,76 @@ +list(FIND team_policy_tile_sizes "${team_policy_tile_size}" _tps_idx) +if(_tps_idx EQUAL -1) + message( + FATAL_ERROR + "${Red}team_policy_tile_size must be one of ${team_policy_tile_sizes}, " + "got '${team_policy_tile_size}'${ColorReset}") +endif() +add_compile_options("-D TEAM_POLICY") +add_compile_options("-D TEAM_POLICY_TILE_SIZE=${team_policy_tile_size}") + +# Compile-time tiled-deposit scratch halo drift. Sizes the halo so a particle +# that drifts up to DRIFT cells between two sorts still deposits inside its tile +# scratch; particles drifting further take the per-particle global-J escape +# valve (correct, only slower). This is independent of the sort cadence, which +# is set at runtime via `spatial_sorting_interval`. Defaults to 1 (the +# sorted-every-step case). +add_compile_options("-D TEAM_POLICY_DRIFT=${team_policy_drift}") + +# Vendor sort: oneDPL on SYCL, Thrust on CUDA, rocThrust/rocprim on HIP. When +# `vendor_sort` is ON (default) the available library is detected and used; the +# spatial sort then builds a single permutation that gathers all SoA members. +# When `vendor_sort` is OFF, or no library is found, the code falls back to +# Kokkos::BinSort, which sorts each member in place -- lower peak memory and no +# maxnpart gather buffer, at the cost of sort speed (negligible when sorting is +# a small fraction of the step). The `vendor_sort` knob lets you force the +# BinSort fallback even when a vendor library is present. +if(${vendor_sort}) + if("${Kokkos_DEVICES}" MATCHES "SYCL") + find_package(oneDPL QUIET) + if(oneDPL_FOUND) + message(STATUS "team_policy: oneDPL found, enabling SYCL sort_by_key") + add_compile_options("-D ONEDPL_ENABLED") + set(DEPENDENCIES ${DEPENDENCIES} oneDPL) + else() + message(STATUS "team_policy: oneDPL not found; using BinSort fallback " + "for SYCL sort_by_key") + endif() + endif() + + if("${Kokkos_DEVICES}" MATCHES "CUDA") + find_package(Thrust QUIET) + if(Thrust_FOUND) + message(STATUS "team_policy: Thrust enabled for CUDA sort_by_key") + add_compile_options("-D THRUST_ENABLED") + else() + message(STATUS "team_policy: Thrust not found; using BinSort fallback " + "for CUDA sort_by_key") + endif() + endif() + + if("${Kokkos_DEVICES}" MATCHES "HIP") + # rocThrust ships with ROCm. The HIP sort_by_key path uses rocprim's + # bounded-bit radix sort directly (rocprim is rocThrust's own dependency, so + # its headers come in transitively; we find it explicitly to keep the + # include path robust). This builds a single permutation that gathers all + # SoA members, instead of the legacy per-member Kokkos::BinSort path which + # allocates a fresh `sorted_values` buffer for every member every step (the + # dominant source of allocator churn / fragmentation on ROCm). + find_package(rocthrust QUIET) + if(rocthrust_FOUND) + message(STATUS "team_policy: rocThrust enabled for HIP sort_by_key") + add_compile_options("-D ROCTHRUST_ENABLED") + set(DEPENDENCIES ${DEPENDENCIES} roc::rocthrust) + find_package(rocprim QUIET) + if(rocprim_FOUND) + set(DEPENDENCIES ${DEPENDENCIES} roc::rocprim) + endif() + else() + message(STATUS "team_policy: rocThrust not found; using BinSort " + "fallback for HIP sort_by_key") + endif() + endif() +else() + message(STATUS "team_policy: vendor_sort=OFF; forcing Kokkos::BinSort " + "fallback for spatial sort_by_key") +endif() From f2e9f757cba68322833142d222568ef84d80ecc7 Mon Sep 17 00:00:00 2001 From: haykh Date: Thu, 10 Sep 2026 08:04:56 -0400 Subject: [PATCH 073/125] bump kokkos version to 5.2.1 --- dev/nix/adios2.nix | 2 +- dev/nix/kokkos.nix | 8 ++++---- extern/Kokkos | 2 +- extern/adios2 | 2 +- extern/entity-pgens | 2 +- 5 files changed, 8 insertions(+), 8 deletions(-) diff --git a/dev/nix/adios2.nix b/dev/nix/adios2.nix index 810c228dc..eb3130632 100644 --- a/dev/nix/adios2.nix +++ b/dev/nix/adios2.nix @@ -39,7 +39,7 @@ stdenv.mkDerivation { ]; propagatedBuildInputs = [ - pkgs.gcc13 + pkgs.gcc15 ] ++ ( if hdf5 then diff --git a/dev/nix/kokkos.nix b/dev/nix/kokkos.nix index f51d2c685..2af11c164 100644 --- a/dev/nix/kokkos.nix +++ b/dev/nix/kokkos.nix @@ -7,7 +7,7 @@ let name = "kokkos"; - pversion = "5.0.1"; + pversion = "5.2.1"; compilerPkgs = { "HIP" = with pkgs.rocmPackages; [ clang @@ -23,11 +23,11 @@ let pkgs.clang-tools cudatoolkit cuda_cudart - pkgs.gcc13 + pkgs.gcc15 ]; "NONE" = [ pkgs.clang-tools - pkgs.gcc13 + pkgs.gcc15 ]; }; getArch = @@ -57,7 +57,7 @@ pkgs.stdenv.mkDerivation rec { src = pkgs.fetchgit { url = "https://github.com/kokkos/kokkos/"; rev = "${pversion}"; - sha256 = "sha256-ChpwGBwE7sNovjdAM/iCeOqqwGufKxAh5vQ3qK6aFBU="; + sha256 = "sha256-9a0am5NR7WtZXA0Nn0nnCohlWlgFDbAzVcc4rvUbGZc="; }; nativeBuildInputs = with pkgs; [ diff --git a/extern/Kokkos b/extern/Kokkos index 37f70304d..bacb34d96 160000 --- a/extern/Kokkos +++ b/extern/Kokkos @@ -1 +1 @@ -Subproject commit 37f70304dc3676691af88d3ac3ba50cddbfa337f +Subproject commit bacb34d96809658652551b360595f14eb49c264e diff --git a/extern/adios2 b/extern/adios2 index 1ef0b5797..5eb1c9e46 160000 --- a/extern/adios2 +++ b/extern/adios2 @@ -1 +1 @@ -Subproject commit 1ef0b5797aeb8a1cc1bb36ec4089eaec19a2eea0 +Subproject commit 5eb1c9e46be94799bbbdbd379a07bf7d3e872278 diff --git a/extern/entity-pgens b/extern/entity-pgens index 386eefc80..0c9cfceac 160000 --- a/extern/entity-pgens +++ b/extern/entity-pgens @@ -1 +1 @@ -Subproject commit 386eefc80e2f63d0e29168869f881dd0b288952d +Subproject commit 0c9cfceaca9a0eaaec6c9d096656a94aecfe5335 From 6da13424bd878f2c0820c4761811af0308634b37 Mon Sep 17 00:00:00 2001 From: haykh Date: Thu, 10 Sep 2026 08:05:20 -0400 Subject: [PATCH 074/125] fixed quietly failing reduced stats test --- tests/kernels/reduced_stats.cpp | 188 +++++++++++++------------------- 1 file changed, 78 insertions(+), 110 deletions(-) diff --git a/tests/kernels/reduced_stats.cpp b/tests/kernels/reduced_stats.cpp index 79aa03637..ab8511f75 100644 --- a/tests/kernels/reduced_stats.cpp +++ b/tests/kernels/reduced_stats.cpp @@ -9,7 +9,10 @@ #include "metrics/minkowski.h" #include +#include +#include #include +#include using namespace ntt; using namespace metric; @@ -45,9 +48,7 @@ template void put_value(ndfield_t& arr, real_t v, unsigned short c) { range_t range; if constexpr (D == Dim::_1D) { - range = { - { 0u, arr.extent(0) } - }; + range = range_t({ 0u, arr.extent(0) }); } else if constexpr (D == Dim::_2D) { range = { { 0u, 0u }, @@ -75,15 +76,31 @@ auto compute_field_stat(const M& metric, return buff / metric.totVolume(); } -auto almost_equal(real_t a, real_t b, real_t acc) -> bool { - return (math::fabs(a - b) < acc * math::max(math::fabs(a), math::fabs(b))) + - (real_t)1e-10; +inline static constexpr auto epsilon = std::numeric_limits::epsilon(); + +/** + * Relative comparison with an absolute floor of `tol`, so that expected values + * of exactly zero are handled as well. + */ +auto almost_equal(real_t a, real_t b, real_t tol) -> bool { + return math::fabs(a - b) <= tol * math::max(math::fabs(b), ONE); +} + +void check(const std::string& name, real_t got, real_t expect, real_t tol) { + raise::ErrorIf(not almost_equal(got, expect, tol), + name + " does not match expected value: got " + + std::to_string(got) + ", expected " + std::to_string(expect) + + " (tolerance " + std::to_string(tol) + ")", + HERE); } +/** + * @param acc dimensionless safety factor on top of the expected round-off bound + */ template void testReducedStats(const std::vector& res, const boundaries_t& ext, - const real_t acc) { + const real_t acc = ONE) { raise::ErrorIf(res.size() != M::Dim, "Invalid resolution size", HERE); M metric { res, ext, {} }; @@ -177,105 +194,58 @@ void testReducedStats(const std::vector& res, put_value(J, values[11], cur::jx3); } - { - const auto Ex_Sq = compute_field_stat(metric, - EM, - J, - cell_range); - raise::ErrorIf(not almost_equal(Ex_Sq, (real_t)(1), acc), - "Ex_Sq does not match expected value", - HERE); - } - - { - const auto Ey_Sq = compute_field_stat(metric, - EM, - J, - cell_range); - raise::ErrorIf(not almost_equal(Ey_Sq, (real_t)(4), acc), - "Ey_Sq does not match expected value", - HERE); - } - - { - const auto Ez_Sq = compute_field_stat(metric, - EM, - J, - cell_range); - raise::ErrorIf(not almost_equal(Ez_Sq, (real_t)(9), acc), - "Ez_Sq does not match expected value", - HERE); - } - - { - const auto Bx_Sq = compute_field_stat(metric, - EM, - J, - cell_range); - raise::ErrorIf(not almost_equal(Bx_Sq, (real_t)(1), acc), - "Bx_Sq does not match expected value", - HERE); - } - - { - const auto By_Sq = compute_field_stat(metric, - EM, - J, - cell_range); - raise::ErrorIf(not almost_equal(By_Sq, (real_t)(4), acc), - "By_Sq does not match expected value", - HERE); - } - - { - const auto Bz_Sq = compute_field_stat(metric, - EM, - J, - cell_range); - raise::ErrorIf(not almost_equal(Bz_Sq, (real_t)(9), acc), - "Bz_Sq does not match expected value", - HERE); - } - - { - const auto ExB_x = compute_field_stat(metric, - EM, - J, - cell_range); - raise::ErrorIf(not almost_equal(ExB_x, (real_t)(12), acc), - "ExB_x does not match expected value", - HERE); - } - - { - const auto ExB_y = compute_field_stat(metric, - EM, - J, - cell_range); - raise::ErrorIf(not almost_equal(ExB_y, (real_t)(-6), acc), - "ExB_y does not match expected value", - HERE); - } - - { - const auto ExB_z = compute_field_stat(metric, - EM, - J, - cell_range); - raise::ErrorIf(not almost_equal(ExB_z, (real_t)(0), acc), - "ExB_z does not match expected value", - HERE); - } - - { - const auto JdotE = compute_field_stat(metric, - EM, - J, - cell_range); - raise::ErrorIf(not almost_equal(JdotE, (real_t)(11), acc), - "JdotE does not match expected value", - HERE); + // the stats kernels accumulate over all active cells with a plain (i.e. + // non-compensated) sum in `real_t`, so the round-off of the reduction itself + // grows linearly with the number of cells being reduced over + ncells_t ncells = 1; + for (const auto& r : res) { + ncells *= r; } + const auto tol = acc * epsilon * static_cast(ncells); + + check("Ex_Sq", + compute_field_stat(metric, EM, J, cell_range), + (real_t)(1), + tol); + check("Ey_Sq", + compute_field_stat(metric, EM, J, cell_range), + (real_t)(4), + tol); + check("Ez_Sq", + compute_field_stat(metric, EM, J, cell_range), + (real_t)(9), + tol); + + check("Bx_Sq", + compute_field_stat(metric, EM, J, cell_range), + (real_t)(1), + tol); + check("By_Sq", + compute_field_stat(metric, EM, J, cell_range), + (real_t)(4), + tol); + check("Bz_Sq", + compute_field_stat(metric, EM, J, cell_range), + (real_t)(9), + tol); + + check("ExB_x", + compute_field_stat(metric, EM, J, cell_range), + (real_t)(12), + tol); + check("ExB_y", + compute_field_stat(metric, EM, J, cell_range), + (real_t)(-6), + tol); + check("ExB_z", + compute_field_stat(metric, EM, J, cell_range), + (real_t)(0), + tol); + + check("JdotE", + compute_field_stat(metric, EM, J, cell_range), + (real_t)(11), + tol); } auto main(int argc, char* argv[]) -> int { @@ -289,13 +259,11 @@ auto main(int argc, char* argv[]) -> int { std::pair y_ext { 0.0, 4.92 }; std::pair z_ext { 0.0, 2.08 }; - testReducedStats>({ nx }, { x_ext }, 1e-6); + testReducedStats>({ nx }, { x_ext }); testReducedStats>({ nx, ny }, - { x_ext, y_ext }, - 1e-6); + { x_ext, y_ext }); testReducedStats>({ nx, ny, nz }, - { x_ext, y_ext, z_ext }, - 1e-6); + { x_ext, y_ext, z_ext }); } catch (std::exception& e) { std::cerr << e.what() << '\n'; From 69996027d63fc04a90365208d48c271e7860c6c8 Mon Sep 17 00:00:00 2001 From: haykh Date: Thu, 10 Sep 2026 08:05:38 -0400 Subject: [PATCH 075/125] RUNTESTS From 1fc642044dcf7810abd68b2921bba5c03ff1800e Mon Sep 17 00:00:00 2001 From: haykh Date: Thu, 10 Sep 2026 08:56:23 -0400 Subject: [PATCH 076/125] upd dependencies py for new kokkos v --- dependencies.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/dependencies.py b/dependencies.py index 114cfd145..8f39bde85 100755 --- a/dependencies.py +++ b/dependencies.py @@ -53,7 +53,7 @@ class Settings: ) # versions - kokkos_version: str = "5.0.1" + kokkos_version: str = "5.2.1" adios2_version: str = "2.11.0" # options From 23c3128d1e40b61273366473c4c9b63f4bde8624 Mon Sep 17 00:00:00 2001 From: A Sullivan Date: Mon, 14 Sep 2026 13:04:58 -0400 Subject: [PATCH 077/125] fixed some typos in the spectra3D additions (GPU still not fixed) --- src/framework/domain/metadomain_io.cpp | 77 +++++--------------------- src/framework/parameters/output.cpp | 6 +- src/output/writer.cpp | 1 + 3 files changed, 17 insertions(+), 67 deletions(-) diff --git a/src/framework/domain/metadomain_io.cpp b/src/framework/domain/metadomain_io.cpp index 8669efe10..78efe38c6 100644 --- a/src/framework/domain/metadomain_io.cpp +++ b/src/framework/domain/metadomain_io.cpp @@ -864,19 +864,6 @@ namespace ntt { auto e_max = params.template get("output.spectra3D.e_max"); - const auto x1_size_local = local_domain->mesh.n_active()[0]; - const auto x1_offset_local = local_domain->offset_ncells()[0]; - const auto x1_size_global = mesh().n_active()[0]; - - // divide based on physical coordinates maybe rather than cells - - std::size_t x2_size_local = 0; - std::size_t x2_offset_local = 0; - std::size_t x2_size_global = 0; - - std::size_t x3_size_local = 0; - std::size_t x3_offset_local = 0; - std::size_t x3_size_global = 0; auto x1_min = mesh().extent(in::x1).first; // extent is in physical units, not code decltype(x1_min) x2_min = 0; @@ -886,62 +873,25 @@ namespace ntt { decltype(x1_max) x2_max = 0; decltype(x1_max) x3_max = 0; - //auto x1_extent_local = mesh().extent(in::x1) - //const auto nx1 = local_domain->mesh.n_active(in::x1); - //const auto nx2 = local_domain->mesh.n_active(in::x2); - - //auto dx1 = (x1_extent_local.second - x1_extent_local.first) / (real_t)(nx1_bin - 1); - auto x1_extent_local = local_domain->mesh.extent(in::x1); - auto x1_min_local = x1_extent_local.first; - auto x1_max_local = x1_extent_local.second; - - auto x1_ind_rank_min = static_cast(static_cast(nx1_bins) * (x1_min_local - x1_min) / (x1_max - x1_min)); - auto x1_ind_rank_max = static_cast(static_cast(nx1_bins) * (x1_max_local - x1_min) / (x1_max - x1_min)); - - decltype(x1_min_local) x2_min_local=0; - decltype(x1_max_local) x2_max_local=0; - decltype(x1_ind_rank_min) x2_ind_rank_min=0; - decltype(x1_ind_rank_max) x2_ind_rank_max=0; - - decltype(x1_min_local) x3_min_local=0; - decltype(x1_max_local) x3_max_local=0; - decltype(x1_ind_rank_min) x3_ind_rank_min=0; - decltype(x1_ind_rank_max) x3_ind_rank_max=0; + - real_t dx1_bin = (x1_max - x1_min )/nx1_bins; - real_t dx2_bin = 1.; - real_t dx3_bin = 1.; - if constexpr (M::PrtlDim == Dim::_2D or M::PrtlDim == Dim::_3D){ // only pick x2 if simulation is in 2D + if constexpr (D == Dim::_2D or D == Dim::_3D){ // only pick x2 if simulation is in 2D x2_min = mesh().extent(in::x2).first; x2_max = mesh().extent(in::x2).second; - - auto x2_extent_local = local_domain->mesh.extent(in::x2); - x2_min_local = x2_extent_local.first; - x2_max_local = x2_extent_local.second; - - dx2_bin = (static_cast(x2_max) - static_cast(x2_min) )/static_cast(nx2_bins); - - x2_ind_rank_min = static_cast(static_cast(nx2_bins) * (x2_min_local - x1_min) / (x2_max - x2_min)); - x2_ind_rank_max = static_cast(static_cast(nx2_bins) * (x2_max_local - x1_min) / (x2_max - x2_min)); + } - if constexpr (M::PrtlDim == Dim::_3D){ + if constexpr (D == Dim::_3D){ x3_min = mesh().extent(in::x3).first; x3_max = mesh().extent(in::x3).second; - auto x3_extent_local = local_domain->mesh.extent(in::x3); - x3_min_local = x3_extent_local.first; - x3_max_local = x3_extent_local.second; - dx3_bin = (static_cast(x3_max) - static_cast(x3_min) )/static_cast(nx3_bins); - - x3_ind_rank_min = static_cast(static_cast(nx3_bins) * (x3_min_local - x3_min) / (x1_max - x1_min)); - x3_ind_rank_max = static_cast(static_cast(nx3_bins) * (x3_max_local - x3_min) / (x1_max - x1_min)); + } @@ -982,11 +932,11 @@ namespace ntt { decltype(i1) i3; decltype(dx1) dx2; decltype(dx1) dx3; - if constexpr (M::PrtlDim == Dim::_2D or M::PrtlDim == Dim::_3D){ // only pick x2 if simulation is in 2D + if constexpr (D == Dim::_2D or D == Dim::_3D){ // only pick x2 if simulation is in 2D i2 = species.i2; dx2 = species.dx2; } - if constexpr (M::PrtlDim == Dim::_3D){ + if constexpr (D == Dim::_3D){ i3 = species.i3; dx3 = species.dx3; } @@ -1002,13 +952,13 @@ namespace ntt { } coord_t x_Cd {ZERO}; - if (D == Dim::_1D or D == Dim::_2D or D == Dim::_3D) { + if constexpr (D == Dim::_1D or D == Dim::_2D or D == Dim::_3D) { x_Cd[0] = static_cast(i1(p)) + static_cast(dx1(p)); } - if (D == Dim::_2D or D == Dim::_3D) { + if constexpr (D == Dim::_2D or D == Dim::_3D) { x_Cd[1] = static_cast(i2(p)) + static_cast(dx2(p)); } - if (D == Dim::_3D) { + if constexpr (D == Dim::_3D) { x_Cd[2] = static_cast(i3(p)) + static_cast(dx3(p)); } coord_t x_Ph { ZERO }; @@ -1045,7 +995,7 @@ namespace ntt { std::size_t x2_ind = 0; - if (M::PrtlDim == Dim::_2D or M::PrtlDim == Dim::_3D){ // only pick x2 if simulation is in 2D + if constexpr (D == Dim::_2D or D == Dim::_3D){ // only pick x2 if simulation is in 2D if (x_Ph[1] <= x2_min) { x2_ind = 0; } else if (x_Ph[1] >= x2_max) { @@ -1054,11 +1004,10 @@ namespace ntt { x2_ind = static_cast( static_cast(nx2_bins) * (x_Ph[1] - x2_min) / (x2_max - x2_min)); } - //real_t x2_center = x2_min + (x2_ind+0.5)*dx2_bin; - //owns_x2 = (x2_center >= x2_min_local) && (x2_center <= x2_max_local); + } std::size_t x3_ind = 0; - if (M::PrtlDim == Dim::_3D){ // only pick x3 if simulation is in 3D + if constexpr (D == Dim::_3D){ // only pick x3 if simulation is in 3D if (x_Ph[2] <= x3_min) { x3_ind = 0; diff --git a/src/framework/parameters/output.cpp b/src/framework/parameters/output.cpp index 13e1a5463..570d73006 100644 --- a/src/framework/parameters/output.cpp +++ b/src/framework/parameters/output.cpp @@ -162,7 +162,7 @@ namespace ntt { /* Spectra3D ------------------------------------------------------------ */ spectra3d_e_min = toml::find_or(toml_data, "output", "spectra3D", "e_min", defaults::output::spec3d_emin); - spectra3d_e_max = toml::find_or(toml_data, "output", "spectra3D", "e_max", defaults::output::spec3d_emin); + spectra3d_e_max = toml::find_or(toml_data, "output", "spectra3D", "e_max", defaults::output::spec3d_emax); spectra3d_log_bins = toml::find_or(toml_data, "output", @@ -186,13 +186,13 @@ namespace ntt { "output", "spectra3D", "nx2", - defaults::output::spec3d_nx1); + defaults::output::spec3d_nx2); spectra3d_nx3 = toml::find_or(toml_data, "output", "spectra3D", "nx3", - defaults::output::spec3d_nx1); + defaults::output::spec3d_nx3); /* Stats ---------------------------------------------------------------- */ stats_quantities = toml::find_or(toml_data, diff --git a/src/output/writer.cpp b/src/output/writer.cpp index 317d95af9..80bfe360f 100644 --- a/src/output/writer.cpp +++ b/src/output/writer.cpp @@ -413,6 +413,7 @@ namespace out { #else int rank = 0; int size = 1; + auto counts3D_h_all = counts3D_h; #endif using Shape = std::vector; From 7bae89df3f7204789b215bcc82951e43dd5b27c2 Mon Sep 17 00:00:00 2001 From: haykh Date: Mon, 14 Sep 2026 14:15:36 -0400 Subject: [PATCH 078/125] gitignore devenv --- .gitignore | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/.gitignore b/.gitignore index e683ed8c5..cd508a596 100644 --- a/.gitignore +++ b/.gitignore @@ -64,4 +64,8 @@ action-token ignore-* tombi/ tidy/ -.claude \ No newline at end of file +.claude + +# devenv +.devenv* +devenv.local.nix From 50040dcbd3d69d99cac6179fd7c93d52027d3013 Mon Sep 17 00:00:00 2001 From: haykh Date: Mon, 14 Sep 2026 15:06:22 -0400 Subject: [PATCH 079/125] spectra3d merged with regular spectra output --- input.example.toml | 8 +- src/framework/domain/metadomain_io.cpp | 132 ++++++++++++------------- src/framework/parameters/output.cpp | 86 +++++++--------- src/framework/parameters/output.h | 17 +--- src/global/defaults.h | 9 +- src/global/global.h | 1 - 6 files changed, 109 insertions(+), 144 deletions(-) diff --git a/input.example.toml b/input.example.toml index 8d6f25117..8870b29b5 100644 --- a/input.example.toml +++ b/input.example.toml @@ -627,10 +627,14 @@ # @type: bool # @default: true log_bins = "" - # Number of bins for the spectra output + # Number of energy bins for the spectra output # @type: uint [> 0] # @default: 200 - n_bins = "" + num_energy_bins = "" + # Number of spatial bins for the spectra output + # @type: array [size 1 :->: 3] + # @default: [1, 1, 1] + num_spatial_bins = "" # Number of timesteps between spectra outputs # @type: uint # @default: 0 diff --git a/src/framework/domain/metadomain_io.cpp b/src/framework/domain/metadomain_io.cpp index a1bc13f00..793955959 100644 --- a/src/framework/domain/metadomain_io.cpp +++ b/src/framework/domain/metadomain_io.cpp @@ -383,10 +383,10 @@ namespace ntt { finished_step, finished_time); const auto write_spectra3D = params.template get( - "output.spectra3D.enable") and - g_writer.shouldWrite("spectra3D", - finished_step, - finished_time); + "output.spectra3D.enable") and + g_writer.shouldWrite("spectra3D", + finished_step, + finished_time); const auto extension = params.template get("output.format"); if (not(write_fields or write_particles or write_spectra or write_spectra3D) and extension != "disabled") { @@ -856,18 +856,19 @@ namespace ntt { "output.spectra3D.log_bins"); const auto n_bins = params.template get( "output.spectra3D.n_bins"); - const auto& metric = local_domain->mesh.metric; + const auto& metric = local_domain->mesh.metric; // extract the number of bins globally in each direction - const auto nx1_bins = params.template get("output.spectra3D.nx1"); - const auto nx2_bins = params.template get("output.spectra3D.nx2"); - const auto nx3_bins = params.template get("output.spectra3D.nx3"); + const auto nx1_bins = params.template get( + "output.spectra3D.nx1"); + const auto nx2_bins = params.template get( + "output.spectra3D.nx2"); + const auto nx3_bins = params.template get( + "output.spectra3D.nx3"); // select the min and max energy for the spectra auto e_min = params.template get("output.spectra3D.e_min"); auto e_max = params.template get("output.spectra3D.e_max"); - - auto x1_min = mesh().extent(in::x1).first; // extent is in physical units, not code decltype(x1_min) x2_min = 0; decltype(x1_min) x3_min = 0; @@ -875,34 +876,23 @@ namespace ntt { auto x1_max = mesh().extent(in::x1).second; // pairs in c++ are addressed by first and secocnd decltype(x1_max) x2_max = 0; decltype(x1_max) x3_max = 0; - - + if constexpr (D == Dim::_2D or + D == Dim::_3D) { // only pick x2 if simulation is in 2D - - - if constexpr (D == Dim::_2D or D == Dim::_3D){ // only pick x2 if simulation is in 2D - - x2_min = mesh().extent(in::x2).first; - x2_max = mesh().extent(in::x2).second; - + x2_min = mesh().extent(in::x2).first; + x2_max = mesh().extent(in::x2).second; } - if constexpr (D == Dim::_3D){ - - x3_min = mesh().extent(in::x3).first; - x3_max = mesh().extent(in::x3).second; + if constexpr (D == Dim::_3D) { - - + x3_min = mesh().extent(in::x3).first; + x3_max = mesh().extent(in::x3).second; } - - - //auto dx_slice = mesh().dx1; - + // auto dx_slice = mesh().dx1; - //auto dxslice = (x1_max - x1_min) / x_bins + // auto dxslice = (x1_max - x1_min) / x_bins if (log_bins) { e_min = math::log10(e_min); @@ -922,25 +912,26 @@ namespace ntt { }); for (const auto& spec : g_writer.spectraWriters()) { - auto& species = local_domain->species[spec.species() - 1]; + auto& species = local_domain->species[spec.species() - 1]; array_t dn3d { "dn3d", nx1_bins, nx2_bins, nx3_bins, n_bins }; - auto dn3d_scatter = Kokkos::Experimental::create_scatter_view(dn3d); - auto ux1 = species.ux1; - auto ux2 = species.ux2; - auto ux3 = species.ux3; - auto i1 = species.i1; - auto dx1 = species.dx1; - //adeep_copy; - decltype(i1) i2; - decltype(i1) i3; + auto dn3d_scatter = Kokkos::Experimental::create_scatter_view(dn3d); + auto ux1 = species.ux1; + auto ux2 = species.ux2; + auto ux3 = species.ux3; + auto i1 = species.i1; + auto dx1 = species.dx1; + // adeep_copy; + decltype(i1) i2; + decltype(i1) i3; decltype(dx1) dx2; decltype(dx1) dx3; - if constexpr (D == Dim::_2D or D == Dim::_3D){ // only pick x2 if simulation is in 2D - i2 = species.i2; + if constexpr (D == Dim::_2D or + D == Dim::_3D) { // only pick x2 if simulation is in 2D + i2 = species.i2; dx2 = species.dx2; } - if constexpr (D == Dim::_3D){ - i3 = species.i3; + if constexpr (D == Dim::_3D) { + i3 = species.i3; dx3 = species.dx3; } auto weight = species.weight; @@ -953,15 +944,15 @@ namespace ntt { if (tag(p) != ParticleTag::alive) { return; } - - coord_t x_Cd {ZERO}; - if constexpr (D == Dim::_1D or D == Dim::_2D or D == Dim::_3D) { + + coord_t x_Cd { ZERO }; + if constexpr (D == Dim::_1D or D == Dim::_2D or D == Dim::_3D) { x_Cd[0] = static_cast(i1(p)) + static_cast(dx1(p)); } - if constexpr (D == Dim::_2D or D == Dim::_3D) { + if constexpr (D == Dim::_2D or D == Dim::_3D) { x_Cd[1] = static_cast(i2(p)) + static_cast(dx2(p)); } - if constexpr (D == Dim::_3D) { + if constexpr (D == Dim::_3D) { x_Cd[2] = static_cast(i3(p)) + static_cast(dx3(p)); } coord_t x_Ph { ZERO }; @@ -980,56 +971,55 @@ namespace ntt { if (en <= e_min) { e_ind = 0; } else if (en >= e_max) { - e_ind = n_bins-1; + e_ind = n_bins - 1; } else { e_ind = static_cast( static_cast(n_bins) * (en - e_min) / (e_max - e_min)); } std::size_t x1_ind = 0; - if (x_Ph[0]<= x1_min) { + if (x_Ph[0] <= x1_min) { x1_ind = 0; } else if (x_Ph[0] >= x1_max) { - x1_ind = nx1_bins-1; + x1_ind = nx1_bins - 1; } else { - x1_ind = static_cast( - static_cast(nx1_bins) * (x_Ph[0] - x1_min) / (x1_max - x1_min)); + x1_ind = static_cast(static_cast(nx1_bins) * + (x_Ph[0] - x1_min) / + (x1_max - x1_min)); } - std::size_t x2_ind = 0; - if constexpr (D == Dim::_2D or D == Dim::_3D){ // only pick x2 if simulation is in 2D + if constexpr (D == Dim::_2D or + D == Dim::_3D) { // only pick x2 if simulation is in 2D if (x_Ph[1] <= x2_min) { x2_ind = 0; } else if (x_Ph[1] >= x2_max) { - x2_ind = nx2_bins-1; + x2_ind = nx2_bins - 1; } else { - x2_ind = static_cast( - static_cast(nx2_bins) * (x_Ph[1] - x2_min) / (x2_max - x2_min)); + x2_ind = static_cast(static_cast(nx2_bins) * + (x_Ph[1] - x2_min) / + (x2_max - x2_min)); } - } std::size_t x3_ind = 0; - if constexpr (D == Dim::_3D){ // only pick x3 if simulation is in 3D - + if constexpr (D == Dim::_3D) { // only pick x3 if simulation is in 3D + if (x_Ph[2] <= x3_min) { x3_ind = 0; } else if (x_Ph[2] >= x3_max) { - x3_ind = nx3_bins-1; + x3_ind = nx3_bins - 1; } else { - x3_ind = static_cast( - static_cast(nx3_bins) * (x_Ph[2] - x3_min) / (x3_max - x3_min)); + x3_ind = static_cast(static_cast(nx3_bins) * + (x_Ph[2] - x3_min) / + (x3_max - x3_min)); } - - - } + } // now I want to save the nx_bins contained in each rank // can save array of x1_inds which are saved? // maybe I can ask what local x_min and x_max are for this rank, pass that to writeSpectrum3D - - auto dn3d_acc = dn3d_scatter.access(); + auto dn3d_acc = dn3d_scatter.access(); dn3d_acc(x1_ind, x2_ind, x3_ind, e_ind) += weight(p); }); Kokkos::Experimental::contribute(dn3d, dn3d_scatter); @@ -1037,7 +1027,7 @@ namespace ntt { } g_writer.writeSpectrumBins(energy, "sEbn"); g_writer.endWriting(WriteMode::Spectra3D); - } + } return true; } diff --git a/src/framework/parameters/output.cpp b/src/framework/parameters/output.cpp index 570d73006..7591895a1 100644 --- a/src/framework/parameters/output.cpp +++ b/src/framework/parameters/output.cpp @@ -33,7 +33,7 @@ namespace ntt { HERE); categories.emplace(); - for (const auto& category : { "fields", "particles", "spectra", "spectra3D", "stats" }) { + for (const auto& category : { "fields", "particles", "spectra", "stats" }) { const auto q_int = toml::find_or(toml_data, "output", category, @@ -153,46 +153,37 @@ namespace ntt { "spectra", "log_bins", defaults::output::spec_log); - spectra_n_bins = toml::find_or(toml_data, - "output", - "spectra", - "n_bins", - defaults::output::spec_nbins); - - /* Spectra3D ------------------------------------------------------------ */ - spectra3d_e_min = toml::find_or(toml_data, "output", "spectra3D", "e_min", defaults::output::spec3d_emin); - - spectra3d_e_max = toml::find_or(toml_data, "output", "spectra3D", "e_max", defaults::output::spec3d_emax); - - spectra3d_log_bins = toml::find_or(toml_data, - "output", - "spectra3D", - "log_bins", - defaults::output::spec3d_log); - - spectra3d_n_bins = toml::find_or(toml_data, - "output", - "spectra3D", - "n_bins", - defaults::output::spec3d_nbins); - - spectra3d_nx1 = toml::find_or(toml_data, - "output", - "spectra3D", - "nx1", - defaults::output::spec3d_nx1); - - spectra3d_nx2 = toml::find_or(toml_data, - "output", - "spectra3D", - "nx2", - defaults::output::spec3d_nx2); - - spectra3d_nx3 = toml::find_or(toml_data, - "output", - "spectra3D", - "nx3", - defaults::output::spec3d_nx3); + if (toml_data.contains("output") and + toml_data.at("output").contains("spectra") and + toml_data.at("output").at("spectra").contains("n_bins")) { + spectra_num_energy_bins = toml::find(toml_data, + "output", + "spectra", + "n_bins"); + raise::Warning( + "`output.spectra.n_bins` is deprecated and will be removed in 1.6+ " + "versions, use `output.spectra.num_energy_bins` instead", + HERE); + } else { + spectra_num_energy_bins = toml::find_or(toml_data, + "output", + "spectra", + "num_energy_bins", + defaults::output::spec_num_e_bins); + } + spectra_num_spatial_bins = toml::find_or(toml_data, + "output", + "spectra", + "num_spatial_bins", + std::vector { 1, 1, 1 }); + if (spectra_num_spatial_bins->size() < static_cast(dim)) { + raise::Error("`output.spectra.num_spatial_bins` must have at least " + + std::to_string(static_cast(dim)) + " entries", + HERE); + } + spectra_num_spatial_bins->erase( + spectra_num_spatial_bins->begin() + static_cast(dim), + spectra_num_spatial_bins->end()); /* Stats ---------------------------------------------------------------- */ stats_quantities = toml::find_or(toml_data, @@ -242,15 +233,10 @@ namespace ntt { params->set("output.spectra.e_min", spectra_e_min.value()); params->set("output.spectra.e_max", spectra_e_max.value()); params->set("output.spectra.log_bins", spectra_log_bins.value()); - params->set("output.spectra.n_bins", spectra_n_bins.value()); - - params->set("output.spectra3D.e_min", spectra3d_e_min.value()); - params->set("output.spectra3D.e_max", spectra3d_e_max.value()); - params->set("output.spectra3D.log_bins", spectra3d_log_bins.value()); - params->set("output.spectra3D.n_bins", spectra3d_n_bins.value()); - params->set("output.spectra3D.nx1", spectra3d_nx1.value()); - params->set("output.spectra3D.nx2", spectra3d_nx2.value()); - params->set("output.spectra3D.nx3", spectra3d_nx3.value()); + params->set("output.spectra.num_energy_bins", + spectra_num_energy_bins.value()); + params->set("output.spectra.num_spatial_bins", + spectra_num_spatial_bins.value()); params->set("output.stats.quantities", stats_quantities.value()); params->set("output.stats.custom", stats_custom_quantities.value()); diff --git a/src/framework/parameters/output.h b/src/framework/parameters/output.h index 680751514..a51bf9290 100644 --- a/src/framework/parameters/output.h +++ b/src/framework/parameters/output.h @@ -50,18 +50,11 @@ namespace ntt { std::optional> particles_species; std::optional particles_stride; - std::optional spectra_e_min; - std::optional spectra_e_max; - std::optional spectra_log_bins; - std::optional spectra_n_bins; - - std::optional spectra3d_e_min; - std::optional spectra3d_e_max; - std::optional spectra3d_log_bins; - std::optional spectra3d_n_bins; - std::optional spectra3d_nx1; - std::optional spectra3d_nx2; - std::optional spectra3d_nx3; + std::optional spectra_e_min; + std::optional spectra_e_max; + std::optional spectra_log_bins; + std::optional spectra_num_energy_bins; + std::optional> spectra_num_spatial_bins; std::optional> stats_quantities; std::optional> stats_custom_quantities; diff --git a/src/global/defaults.h b/src/global/defaults.h index 919b11fde..82b4364cc 100644 --- a/src/global/defaults.h +++ b/src/global/defaults.h @@ -75,14 +75,7 @@ namespace ntt::defaults { const real_t spec_emin = 1e-3; const real_t spec_emax = 1e3; const bool spec_log = true; - const std::size_t spec_nbins = 200; - const real_t spec3d_emin = 1e-3; - const real_t spec3d_emax = 1e3; - const bool spec3d_log = true; - const std::size_t spec3d_nbins = 200; - const std::size_t spec3d_nx1 = 1; - const std::size_t spec3d_nx2 = 1; - const std::size_t spec3d_nx3 = 1; + const std::size_t spec_num_e_bins = 200; const std::vector stats_quantities = { "B^2", "E^2", "ExB", diff --git a/src/global/global.h b/src/global/global.h index 00047359a..c3fbc5061 100644 --- a/src/global/global.h +++ b/src/global/global.h @@ -298,7 +298,6 @@ namespace WriteMode { Particles = 1 << 1, Spectra = 1 << 2, Stats = 1 << 3, - Spectra3D = 1 << 4, }; } // namespace WriteMode From 3a753904e6281437bc151da64cce375ed61259d0 Mon Sep 17 00:00:00 2001 From: haykh Date: Mon, 14 Sep 2026 15:08:27 -0400 Subject: [PATCH 080/125] .gitignore for devenv --- .gitignore | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/.gitignore b/.gitignore index e683ed8c5..cd508a596 100644 --- a/.gitignore +++ b/.gitignore @@ -64,4 +64,8 @@ action-token ignore-* tombi/ tidy/ -.claude \ No newline at end of file +.claude + +# devenv +.devenv* +devenv.local.nix From 99a73390ab51a0598dcc8829ecb58d560152f629 Mon Sep 17 00:00:00 2001 From: haykh Date: Mon, 14 Sep 2026 17:20:10 -0400 Subject: [PATCH 081/125] refactor checkpoint/io/comms for metadomain + containers --- src/framework/CMakeLists.txt | 44 +- .../{fields_io.cpp => checkpoint/fields.cpp} | 0 .../particles.cpp} | 365 +------- .../particles.cpp} | 12 +- src/framework/containers/fields.h | 1 + src/framework/containers/io/particles.cpp | 371 ++++++++ src/framework/containers/particles.h | 6 +- src/framework/domain/checkpoint/init.cpp | 107 +++ .../resume.cpp} | 163 +--- src/framework/domain/checkpoint/write.cpp | 114 +++ src/framework/domain/comm/fields.cpp | 203 ++++ .../{comm_mpi.hpp => comm/fields_mpi.hpp} | 10 +- .../{comm_nompi.hpp => comm/fields_nompi.hpp} | 10 +- src/framework/domain/comm/fields_sync.cpp | 236 +++++ src/framework/domain/comm/particles.cpp | 126 +++ src/framework/domain/comm/utils.hpp | 209 +++++ .../domain/comm/vector_potential.cpp | 127 +++ src/framework/domain/io/fields.cpp | 586 ++++++++++++ src/framework/domain/io/init.cpp | 117 +++ src/framework/domain/io/spectra.cpp | 92 ++ src/framework/domain/io/write.cpp | 111 +++ src/framework/domain/metadomain.h | 35 +- src/framework/domain/metadomain_comm.cpp | 668 ------------- src/framework/domain/metadomain_io.cpp | 882 ------------------ src/framework/specialization_registry.h | 9 + tests/framework/comm-mpi.cpp | 2 +- tests/framework/comm-nompi.cpp | 2 +- 27 files changed, 2491 insertions(+), 2117 deletions(-) rename src/framework/containers/{fields_io.cpp => checkpoint/fields.cpp} (100%) rename src/framework/containers/{particles_io.cpp => checkpoint/particles.cpp} (59%) rename src/framework/containers/{particles_comm.cpp => comm/particles.cpp} (98%) create mode 100644 src/framework/containers/io/particles.cpp create mode 100644 src/framework/domain/checkpoint/init.cpp rename src/framework/domain/{metadomain_chckpt.cpp => checkpoint/resume.cpp} (52%) create mode 100644 src/framework/domain/checkpoint/write.cpp create mode 100644 src/framework/domain/comm/fields.cpp rename src/framework/domain/{comm_mpi.hpp => comm/fields_mpi.hpp} (98%) rename src/framework/domain/{comm_nompi.hpp => comm/fields_nompi.hpp} (95%) create mode 100644 src/framework/domain/comm/fields_sync.cpp create mode 100644 src/framework/domain/comm/particles.cpp create mode 100644 src/framework/domain/comm/utils.hpp create mode 100644 src/framework/domain/comm/vector_potential.cpp create mode 100644 src/framework/domain/io/fields.cpp create mode 100644 src/framework/domain/io/init.cpp create mode 100644 src/framework/domain/io/spectra.cpp create mode 100644 src/framework/domain/io/write.cpp delete mode 100644 src/framework/domain/metadomain_comm.cpp delete mode 100644 src/framework/domain/metadomain_io.cpp diff --git a/src/framework/CMakeLists.txt b/src/framework/CMakeLists.txt index 377c1f98a..4853e2935 100644 --- a/src/framework/CMakeLists.txt +++ b/src/framework/CMakeLists.txt @@ -14,17 +14,27 @@ # * domain/grid.cpp # * domain/metadomain.cpp # * domain/metadomain_sort.cpp -# * domain/metadomain_comm.cpp -# * domain/metadomain_chckpt.cpp # * domain/metadomain_stats.cpp -# * domain/metadomain_io.cpp # * domain/metadomain_reshape.cpp # * domain/metadomain_loadbal.cpp +# * domain/checkpoint/init.cpp +# * domain/checkpoint/write.cpp +# * domain/checkpoint/resume.cpp +# * domain/comm/fields.cpp +# * domain/comm/fields_sync.cpp +# * domain/comm/particles.cpp +# * domain/comm/vector_potential.cpp +# * domain/io/init.cpp +# * domain/io/write.cpp +# * domain/io/spectra.cpp +# * domain/io/fields.cpp # * containers/particles.cpp -# * containers/particles_comm.cpp -# * containers/particles_io.cpp # * containers/particles_sort.cpp # * containers/fields.cpp +# * containers/comm/particles.cpp +# * containers/io/particles.cpp +# * containers/checkpoint/particles.cpp +# * containers/checkpoint/fields.cpp # # @includes: # @@ -56,22 +66,34 @@ set(SOURCES ${SRC_DIR}/parameters/extra.cpp ${SRC_DIR}/domain/grid.cpp ${SRC_DIR}/domain/metadomain.cpp - ${SRC_DIR}/domain/metadomain_comm.cpp ${SRC_DIR}/domain/metadomain_sort.cpp ${SRC_DIR}/domain/metadomain_stats.cpp ${SRC_DIR}/domain/metadomain_reshape.cpp ${SRC_DIR}/domain/metadomain_loadbal.cpp + ${SRC_DIR}/domain/comm/fields.cpp + ${SRC_DIR}/domain/comm/fields_sync.cpp + ${SRC_DIR}/domain/comm/particles.cpp ${SRC_DIR}/containers/particles.cpp ${SRC_DIR}/containers/particles_sort.cpp ${SRC_DIR}/containers/fields.cpp) if(${output}) - list(APPEND SOURCES ${SRC_DIR}/domain/metadomain_io.cpp) - list(APPEND SOURCES ${SRC_DIR}/domain/metadomain_chckpt.cpp) - list(APPEND SOURCES ${SRC_DIR}/containers/fields_io.cpp) - list(APPEND SOURCES ${SRC_DIR}/containers/particles_io.cpp) + list( + APPEND + SOURCES + ${SRC_DIR}/domain/io/init.cpp + ${SRC_DIR}/domain/io/write.cpp + ${SRC_DIR}/domain/io/spectra.cpp + ${SRC_DIR}/domain/io/fields.cpp + ${SRC_DIR}/domain/checkpoint/init.cpp + ${SRC_DIR}/domain/checkpoint/write.cpp + ${SRC_DIR}/domain/checkpoint/resume.cpp + ${SRC_DIR}/containers/checkpoint/fields.cpp + ${SRC_DIR}/containers/checkpoint/particles.cpp + ${SRC_DIR}/containers/io/particles.cpp + ${SRC_DIR}/domain/comm/vector_potential.cpp) endif() if(${mpi}) - list(APPEND SOURCES ${SRC_DIR}/containers/particles_comm.cpp) + list(APPEND SOURCES ${SRC_DIR}/containers/comm/particles.cpp) endif() add_library(ntt_framework ${SOURCES}) diff --git a/src/framework/containers/fields_io.cpp b/src/framework/containers/checkpoint/fields.cpp similarity index 100% rename from src/framework/containers/fields_io.cpp rename to src/framework/containers/checkpoint/fields.cpp diff --git a/src/framework/containers/particles_io.cpp b/src/framework/containers/checkpoint/particles.cpp similarity index 59% rename from src/framework/containers/particles_io.cpp rename to src/framework/containers/checkpoint/particles.cpp index 484f1873e..4308b1fb8 100644 --- a/src/framework/containers/particles_io.cpp +++ b/src/framework/containers/checkpoint/particles.cpp @@ -1,15 +1,13 @@ +#include "framework/containers/particles.h" + #include "enums.h" #include "global.h" -#include "arch/kokkos_aliases.h" -#include "traits/metric.h" #include "utils/error.h" #include "utils/formatting.h" #include "utils/log.h" -#include "framework/containers/particles.h" #include "framework/specialization_registry.h" -#include "kernels/prtls_to_phys.hpp" #include "output/utils/readers.h" #include "output/utils/writers.h" @@ -27,332 +25,6 @@ #include namespace ntt { - /* * * * * * * * * - * Output - * * * * * * * * */ - template - void Particles::OutputDeclare(adios2::IO& io) const { - const auto n_addition_coords = ((D == Dim::_2D) and (C != Coord::Cartesian)) - ? 1 - : 0; - for (auto d { 0u }; d < D + n_addition_coords; ++d) { - io.DefineVariable(fmt::format("pX%d_%d", d + 1, index()), - { adios2::UnknownDim }, - { adios2::UnknownDim }, - { adios2::UnknownDim }); - } - for (auto d { 0u }; d < Dim::_3D; ++d) { - io.DefineVariable(fmt::format("pU%d_%d", d + 1, index()), - { adios2::UnknownDim }, - { adios2::UnknownDim }, - { adios2::UnknownDim }); - } - io.DefineVariable(fmt::format("pW_%d", index()), - { adios2::UnknownDim }, - { adios2::UnknownDim }, - { adios2::UnknownDim }); - if (npld_r() > 0) { - for (auto pr { 0 }; pr < npld_r(); ++pr) { - io.DefineVariable(fmt::format("pPLDR%d_%d", pr, index()), - { adios2::UnknownDim }, - { adios2::UnknownDim }, - { adios2::UnknownDim }); - } - } - auto num_track_plds = 0; - if (use_tracking()) { - io.DefineVariable(fmt::format("pIDX_%d", index()), - { adios2::UnknownDim }, - { adios2::UnknownDim }, - { adios2::UnknownDim }); -#if !defined(MPI_ENABLED) - num_track_plds = 1; -#else - num_track_plds = 2; - io.DefineVariable(fmt::format("pRNK_%d", index()), - { adios2::UnknownDim }, - { adios2::UnknownDim }, - { adios2::UnknownDim }); -#endif - } - if (npld_i() > num_track_plds) { - for (auto pr { num_track_plds }; pr < npld_i(); ++pr) { - io.DefineVariable( - fmt::format("pPLDI%d_%d", pr - num_track_plds, index()), - { adios2::UnknownDim }, - { adios2::UnknownDim }, - { adios2::UnknownDim }); - } - } - } - - template - template - void Particles::OutputWrite(adios2::IO& io, - adios2::Engine& writer, - npart_t prtl_stride, - std::size_t domains_total, - std::size_t domains_offset, - const M& metric) { - if (not is_sorted()) { - RemoveDead(); - } - npart_t nout; - array_t out_indices; - if (!use_tracking()) { - nout = npart() / prtl_stride; - } else { - nout = 0u; - const auto tag_d = this->tag; - const auto pld_i_d = this->pld_i; - Kokkos::parallel_reduce( - "CountOutputParticles", - rangeActiveParticles(), - Lambda(prtlidx_t p, npart_t & l_nout) { - if ((tag_d(p) == ParticleTag::alive) and - (pld_i_d(p, pldi::spcCtr) % prtl_stride == 0)) { - l_nout += 1; - } - }, - nout); - out_indices = array_t { "out_indices", nout }; - const array_t out_counter { "out_counter" }; - Kokkos::parallel_for( - "RecordOutputIndices", - rangeActiveParticles(), - Lambda(prtlidx_t p) { - if ((tag_d(p) == ParticleTag::alive) and - (pld_i_d(p, pldi::spcCtr) % prtl_stride == 0)) { - const auto p_out = Kokkos::atomic_fetch_add(&out_counter(), 1); - out_indices(p_out) = p; - } - }); - } - -#if !defined(MPI_ENABLED) - const std::size_t nout_offset = 0; - const std::size_t nout_total = nout; - (void)domains_total; - (void)domains_offset; -#else - // global totals/offsets are sums over all ranks and can exceed the - // per-rank `npart_t` (uint32_t) range at large problem sizes, so - // accumulate them in a 64-bit type - std::size_t nout_offset = 0; - std::size_t nout_total = 0; - auto nout_total_vec = std::vector(domains_total); - MPI_Allgather(&nout, - 1, - mpi::get_type(), - nout_total_vec.data(), - 1, - mpi::get_type(), - MPI_COMM_WORLD); - for (auto r = 0u; r < domains_total; ++r) { - if (r < domains_offset) { - nout_offset += nout_total_vec[r]; - } - nout_total += nout_total_vec[r]; - } -#endif // MPI_ENABLED - - array_t buff_x1, buff_x2, buff_x3; - array_t buff_ux1 { "ux1", nout }; - array_t buff_ux2 { "ux2", nout }; - array_t buff_ux3 { "ux3", nout }; - array_t buff_wei { "w", nout }; - if constexpr (D == Dim::_1D or D == Dim::_2D or D == Dim::_3D) { - buff_x1 = array_t { "x1", nout }; - } - if constexpr (D == Dim::_2D or D == Dim::_3D) { - buff_x2 = array_t { "x2", nout }; - } - if constexpr (D == Dim::_3D or ((D == Dim::_2D) and (C != Coord::Cartesian))) { - buff_x3 = array_t { "x3", nout }; - } - array_t buff_pldr; - array_t buff_pldi; - - if (npld_r() > 0) { - buff_pldr = array_t { "pldr", nout, npld_r() }; - } - if (npld_i() > 0) { - buff_pldi = array_t { "pldi", nout, npld_i() }; - } - - if (nout > 0) { - if (!use_tracking()) { - // clang-format off - Kokkos::parallel_for( - "PrtlToPhys", - nout, - kernel::PrtlToPhys_kernel(prtl_stride, out_indices, - buff_x1, buff_x2, buff_x3, - buff_ux1, buff_ux2, buff_ux3, - buff_wei, - buff_pldr, buff_pldi, - i1, i2, i3, - dx1, dx2, dx3, - ux1, ux2, ux3, - phi, weight, - pld_r, pld_i, - metric)); - // clang-format on - } else { - // clang-format off - Kokkos::parallel_for( - "PrtlToPhys", - nout, - kernel::PrtlToPhys_kernel(prtl_stride, out_indices, - buff_x1, buff_x2, buff_x3, - buff_ux1, buff_ux2, buff_ux3, - buff_wei, - buff_pldr, buff_pldi, - i1, i2, i3, - dx1, dx2, dx3, - ux1, ux2, ux3, - phi, weight, - pld_r, pld_i, - metric)); - // clang-format on - } - } - out::Write1DArray(io, - writer, - fmt::format("pW_%d", index()), - buff_wei, - nout, - nout_total, - nout_offset); - out::Write1DArray(io, - writer, - fmt::format("pU1_%d", index()), - buff_ux1, - nout, - nout_total, - nout_offset); - out::Write1DArray(io, - writer, - fmt::format("pU2_%d", index()), - buff_ux2, - nout, - nout_total, - nout_offset); - out::Write1DArray(io, - writer, - fmt::format("pU3_%d", index()), - buff_ux3, - nout, - nout_total, - nout_offset); - if constexpr (D == Dim::_1D or D == Dim::_2D or D == Dim::_3D) { - out::Write1DArray(io, - writer, - fmt::format("pX1_%d", index()), - buff_x1, - nout, - nout_total, - nout_offset); - } - if constexpr (D == Dim::_2D or D == Dim::_3D) { - out::Write1DArray(io, - writer, - fmt::format("pX2_%d", index()), - buff_x2, - nout, - nout_total, - nout_offset); - } - if constexpr (D == Dim::_3D or ((D == Dim::_2D) and (C != Coord::Cartesian))) { - out::Write1DArray(io, - writer, - fmt::format("pX3_%d", index()), - buff_x3, - nout, - nout_total, - nout_offset); - } - - if (npld_r() > 0) { - for (auto pr { 0 }; pr < npld_r(); ++pr) { - auto buff_sub = Kokkos::subview(buff_pldr, Kokkos::ALL, pr); - out::Write1DSubArray( - io, - writer, - fmt::format("pPLDR%d_%d", pr, index()), - buff_sub, - nout, - nout_total, - nout_offset); - } - } - auto num_track_plds = 0; - if (use_tracking()) { -#if !defined(MPI_ENABLED) - num_track_plds = 1; - { - auto buff_sub = Kokkos::subview(buff_pldi, - Kokkos::ALL, - static_cast(pldi::spcCtr)); - out::Write1DSubArray( - io, - writer, - fmt::format("pIDX_%d", index()), - buff_sub, - nout, - nout_total, - nout_offset); - } -#else - num_track_plds = 2; - { - auto buff_sub = Kokkos::subview(buff_pldi, - Kokkos::ALL, - static_cast(pldi::spcCtr)); - out::Write1DSubArray( - io, - writer, - fmt::format("pIDX_%d", index()), - buff_sub, - nout, - nout_total, - nout_offset); - } - { - auto buff_sub = Kokkos::subview(buff_pldi, - Kokkos::ALL, - static_cast(pldi::domIdx)); - out::Write1DSubArray( - io, - writer, - fmt::format("pRNK_%d", index()), - buff_sub, - nout, - nout_total, - nout_offset); - } -#endif - } - if (npld_i() > num_track_plds) { - for (auto pr { num_track_plds }; pr < npld_i(); ++pr) { - auto buff_sub = Kokkos::subview(buff_pldi, - Kokkos::ALL, - static_cast(pr)); - out::Write1DSubArray( - io, - writer, - fmt::format("pPLDI%d_%d", pr - num_track_plds, index()), - buff_sub, - nout, - nout_total, - nout_offset); - } - } - } - - /* * * * * * * * * - * Checkpoints - * * * * * * * * */ template void Particles::CheckpointDeclare(adios2::IO& io) const { @@ -823,31 +495,6 @@ namespace ntt { } } - // NOLINTBEGIN(bugprone-macro-parentheses) -#define PARTICLES_OUTPUT_DECLARE(D, C) \ - template void Particles::OutputDeclare(adios2::IO&) const; - - PARTICLES_OUTPUT_DECLARE(Dim::_1D, Coord::Cartesian) - PARTICLES_OUTPUT_DECLARE(Dim::_2D, Coord::Cartesian) - PARTICLES_OUTPUT_DECLARE(Dim::_3D, Coord::Cartesian) - PARTICLES_OUTPUT_DECLARE(Dim::_2D, Coord::Spherical) - PARTICLES_OUTPUT_DECLARE(Dim::_2D, Coord::Qspherical) - PARTICLES_OUTPUT_DECLARE(Dim::_3D, Coord::Spherical) - PARTICLES_OUTPUT_DECLARE(Dim::_3D, Coord::Qspherical) -#undef PARTICLES_OUTPUT_DECLARE - -#define PARTICLES_OUTPUT_WRITE(S, M, D) \ - template void Particles::Dim, M::CoordType>::OutputWrite>( \ - adios2::IO&, \ - adios2::Engine&, \ - npart_t, \ - std::size_t, \ - std::size_t, \ - const M&); - - NTT_FOREACH_SPECIALIZATION(PARTICLES_OUTPUT_WRITE) -#undef PARTICLES_OUTPUT_WRITE - #define PARTICLES_CHECKPOINTS(D, C) \ template void Particles::CheckpointDeclare(adios2::IO&) const; \ template void Particles::CheckpointRead(adios2::IO&, \ @@ -859,13 +506,7 @@ namespace ntt { std::size_t, \ std::size_t) const; - PARTICLES_CHECKPOINTS(Dim::_1D, Coord::Cartesian) - PARTICLES_CHECKPOINTS(Dim::_2D, Coord::Cartesian) - PARTICLES_CHECKPOINTS(Dim::_3D, Coord::Cartesian) - PARTICLES_CHECKPOINTS(Dim::_2D, Coord::Spherical) - PARTICLES_CHECKPOINTS(Dim::_2D, Coord::Qspherical) - PARTICLES_CHECKPOINTS(Dim::_3D, Coord::Spherical) - PARTICLES_CHECKPOINTS(Dim::_3D, Coord::Qspherical) + NTT_FOREACH_COORDINATE(PARTICLES_CHECKPOINTS) #undef PARTICLES_CHECKPOINTS // NOLINTEND(bugprone-macro-parentheses) diff --git a/src/framework/containers/particles_comm.cpp b/src/framework/containers/comm/particles.cpp similarity index 98% rename from src/framework/containers/particles_comm.cpp rename to src/framework/containers/comm/particles.cpp index 6ede3abf0..12a7fbad8 100644 --- a/src/framework/containers/particles_comm.cpp +++ b/src/framework/containers/comm/particles.cpp @@ -1,3 +1,5 @@ +#include "framework/containers/particles.h" + #include "enums.h" #include "global.h" @@ -9,7 +11,7 @@ #include "utils/formatting.h" #include "utils/log.h" -#include "framework/containers/particles.h" +#include "framework/specialization_registry.h" #include "kernels/comm.hpp" #include @@ -398,13 +400,7 @@ namespace ntt { const dir::map_t&, \ const dir::map_t&); - PARTICLES_COMM(Dim::_1D, Coord::Cartesian) - PARTICLES_COMM(Dim::_2D, Coord::Cartesian) - PARTICLES_COMM(Dim::_3D, Coord::Cartesian) - PARTICLES_COMM(Dim::_2D, Coord::Spherical) - PARTICLES_COMM(Dim::_2D, Coord::Qspherical) - PARTICLES_COMM(Dim::_3D, Coord::Spherical) - PARTICLES_COMM(Dim::_3D, Coord::Qspherical) + NTT_FOREACH_COORDINATE(PARTICLES_COMM) #undef PARTICLES_COMM } // namespace ntt diff --git a/src/framework/containers/fields.h b/src/framework/containers/fields.h index 2dcfd0a7e..96e75afab 100644 --- a/src/framework/containers/fields.h +++ b/src/framework/containers/fields.h @@ -5,6 +5,7 @@ * - ntt::Fields * @cpp: * - fields.cpp + * - checkpoint/fields.cpp * @namespaces: * - ntt:: * @note SRPIC engine allocates em(6), bckp(6), cur(3), buff(3) diff --git a/src/framework/containers/io/particles.cpp b/src/framework/containers/io/particles.cpp new file mode 100644 index 000000000..abd6f1c97 --- /dev/null +++ b/src/framework/containers/io/particles.cpp @@ -0,0 +1,371 @@ +#include "framework/containers/particles.h" + +#include "enums.h" +#include "global.h" + +#include "arch/kokkos_aliases.h" +#include "traits/metric.h" +#include "utils/formatting.h" + +#include "framework/specialization_registry.h" +#include "kernels/prtls_to_phys.hpp" +#include "output/utils/writers.h" + +#include +#include + +#if defined(MPI_ENABLED) + #include "arch/mpi_aliases.h" + + #include + + #include +#endif + +#include + +namespace ntt { + /* * * * * * * * * + * Output + * * * * * * * * */ + template + void Particles::OutputDeclare(adios2::IO& io) const { + const auto n_addition_coords = ((D == Dim::_2D) and (C != Coord::Cartesian)) + ? 1 + : 0; + for (auto d { 0u }; d < D + n_addition_coords; ++d) { + io.DefineVariable(fmt::format("pX%d_%d", d + 1, index()), + { adios2::UnknownDim }, + { adios2::UnknownDim }, + { adios2::UnknownDim }); + } + for (auto d { 0u }; d < Dim::_3D; ++d) { + io.DefineVariable(fmt::format("pU%d_%d", d + 1, index()), + { adios2::UnknownDim }, + { adios2::UnknownDim }, + { adios2::UnknownDim }); + } + io.DefineVariable(fmt::format("pW_%d", index()), + { adios2::UnknownDim }, + { adios2::UnknownDim }, + { adios2::UnknownDim }); + if (npld_r() > 0) { + for (auto pr { 0 }; pr < npld_r(); ++pr) { + io.DefineVariable(fmt::format("pPLDR%d_%d", pr, index()), + { adios2::UnknownDim }, + { adios2::UnknownDim }, + { adios2::UnknownDim }); + } + } + auto num_track_plds = 0; + if (use_tracking()) { + io.DefineVariable(fmt::format("pIDX_%d", index()), + { adios2::UnknownDim }, + { adios2::UnknownDim }, + { adios2::UnknownDim }); +#if !defined(MPI_ENABLED) + num_track_plds = 1; +#else + num_track_plds = 2; + io.DefineVariable(fmt::format("pRNK_%d", index()), + { adios2::UnknownDim }, + { adios2::UnknownDim }, + { adios2::UnknownDim }); +#endif + } + if (npld_i() > num_track_plds) { + for (auto pr { num_track_plds }; pr < npld_i(); ++pr) { + io.DefineVariable( + fmt::format("pPLDI%d_%d", pr - num_track_plds, index()), + { adios2::UnknownDim }, + { adios2::UnknownDim }, + { adios2::UnknownDim }); + } + } + } + + template + template + void Particles::OutputWrite(adios2::IO& io, + adios2::Engine& writer, + npart_t prtl_stride, + std::size_t domains_total, + std::size_t domains_offset, + const M& metric) { + if (not is_sorted()) { + RemoveDead(); + } + npart_t nout; + array_t out_indices; + if (!use_tracking()) { + nout = npart() / prtl_stride; + } else { + nout = 0u; + const auto tag_d = this->tag; + const auto pld_i_d = this->pld_i; + Kokkos::parallel_reduce( + "CountOutputParticles", + rangeActiveParticles(), + Lambda(prtlidx_t p, npart_t & l_nout) { + if ((tag_d(p) == ParticleTag::alive) and + (pld_i_d(p, pldi::spcCtr) % prtl_stride == 0)) { + l_nout += 1; + } + }, + nout); + out_indices = array_t { "out_indices", nout }; + const array_t out_counter { "out_counter" }; + Kokkos::parallel_for( + "RecordOutputIndices", + rangeActiveParticles(), + Lambda(prtlidx_t p) { + if ((tag_d(p) == ParticleTag::alive) and + (pld_i_d(p, pldi::spcCtr) % prtl_stride == 0)) { + const auto p_out = Kokkos::atomic_fetch_add(&out_counter(), 1); + out_indices(p_out) = p; + } + }); + } + +#if !defined(MPI_ENABLED) + const std::size_t nout_offset = 0; + const std::size_t nout_total = nout; + (void)domains_total; + (void)domains_offset; +#else + // global totals/offsets are sums over all ranks and can exceed the + // per-rank `npart_t` (uint32_t) range at large problem sizes, so + // accumulate them in a 64-bit type + std::size_t nout_offset = 0; + std::size_t nout_total = 0; + auto nout_total_vec = std::vector(domains_total); + MPI_Allgather(&nout, + 1, + mpi::get_type(), + nout_total_vec.data(), + 1, + mpi::get_type(), + MPI_COMM_WORLD); + for (auto r = 0u; r < domains_total; ++r) { + if (r < domains_offset) { + nout_offset += nout_total_vec[r]; + } + nout_total += nout_total_vec[r]; + } +#endif // MPI_ENABLED + + array_t buff_x1, buff_x2, buff_x3; + array_t buff_ux1 { "ux1", nout }; + array_t buff_ux2 { "ux2", nout }; + array_t buff_ux3 { "ux3", nout }; + array_t buff_wei { "w", nout }; + if constexpr (D == Dim::_1D or D == Dim::_2D or D == Dim::_3D) { + buff_x1 = array_t { "x1", nout }; + } + if constexpr (D == Dim::_2D or D == Dim::_3D) { + buff_x2 = array_t { "x2", nout }; + } + if constexpr (D == Dim::_3D or ((D == Dim::_2D) and (C != Coord::Cartesian))) { + buff_x3 = array_t { "x3", nout }; + } + array_t buff_pldr; + array_t buff_pldi; + + if (npld_r() > 0) { + buff_pldr = array_t { "pldr", nout, npld_r() }; + } + if (npld_i() > 0) { + buff_pldi = array_t { "pldi", nout, npld_i() }; + } + + if (nout > 0) { + if (!use_tracking()) { + // clang-format off + Kokkos::parallel_for( + "PrtlToPhys", + nout, + kernel::PrtlToPhys_kernel(prtl_stride, out_indices, + buff_x1, buff_x2, buff_x3, + buff_ux1, buff_ux2, buff_ux3, + buff_wei, + buff_pldr, buff_pldi, + i1, i2, i3, + dx1, dx2, dx3, + ux1, ux2, ux3, + phi, weight, + pld_r, pld_i, + metric)); + // clang-format on + } else { + // clang-format off + Kokkos::parallel_for( + "PrtlToPhys", + nout, + kernel::PrtlToPhys_kernel(prtl_stride, out_indices, + buff_x1, buff_x2, buff_x3, + buff_ux1, buff_ux2, buff_ux3, + buff_wei, + buff_pldr, buff_pldi, + i1, i2, i3, + dx1, dx2, dx3, + ux1, ux2, ux3, + phi, weight, + pld_r, pld_i, + metric)); + // clang-format on + } + } + out::Write1DArray(io, + writer, + fmt::format("pW_%d", index()), + buff_wei, + nout, + nout_total, + nout_offset); + out::Write1DArray(io, + writer, + fmt::format("pU1_%d", index()), + buff_ux1, + nout, + nout_total, + nout_offset); + out::Write1DArray(io, + writer, + fmt::format("pU2_%d", index()), + buff_ux2, + nout, + nout_total, + nout_offset); + out::Write1DArray(io, + writer, + fmt::format("pU3_%d", index()), + buff_ux3, + nout, + nout_total, + nout_offset); + if constexpr (D == Dim::_1D or D == Dim::_2D or D == Dim::_3D) { + out::Write1DArray(io, + writer, + fmt::format("pX1_%d", index()), + buff_x1, + nout, + nout_total, + nout_offset); + } + if constexpr (D == Dim::_2D or D == Dim::_3D) { + out::Write1DArray(io, + writer, + fmt::format("pX2_%d", index()), + buff_x2, + nout, + nout_total, + nout_offset); + } + if constexpr (D == Dim::_3D or ((D == Dim::_2D) and (C != Coord::Cartesian))) { + out::Write1DArray(io, + writer, + fmt::format("pX3_%d", index()), + buff_x3, + nout, + nout_total, + nout_offset); + } + + if (npld_r() > 0) { + for (auto pr { 0 }; pr < npld_r(); ++pr) { + auto buff_sub = Kokkos::subview(buff_pldr, Kokkos::ALL, pr); + out::Write1DSubArray( + io, + writer, + fmt::format("pPLDR%d_%d", pr, index()), + buff_sub, + nout, + nout_total, + nout_offset); + } + } + auto num_track_plds = 0; + if (use_tracking()) { +#if !defined(MPI_ENABLED) + num_track_plds = 1; + { + auto buff_sub = Kokkos::subview(buff_pldi, + Kokkos::ALL, + static_cast(pldi::spcCtr)); + out::Write1DSubArray( + io, + writer, + fmt::format("pIDX_%d", index()), + buff_sub, + nout, + nout_total, + nout_offset); + } +#else + num_track_plds = 2; + { + auto buff_sub = Kokkos::subview(buff_pldi, + Kokkos::ALL, + static_cast(pldi::spcCtr)); + out::Write1DSubArray( + io, + writer, + fmt::format("pIDX_%d", index()), + buff_sub, + nout, + nout_total, + nout_offset); + } + { + auto buff_sub = Kokkos::subview(buff_pldi, + Kokkos::ALL, + static_cast(pldi::domIdx)); + out::Write1DSubArray( + io, + writer, + fmt::format("pRNK_%d", index()), + buff_sub, + nout, + nout_total, + nout_offset); + } +#endif + } + if (npld_i() > num_track_plds) { + for (auto pr { num_track_plds }; pr < npld_i(); ++pr) { + auto buff_sub = Kokkos::subview(buff_pldi, + Kokkos::ALL, + static_cast(pr)); + out::Write1DSubArray( + io, + writer, + fmt::format("pPLDI%d_%d", pr - num_track_plds, index()), + buff_sub, + nout, + nout_total, + nout_offset); + } + } + } + + // NOLINTBEGIN(bugprone-macro-parentheses) +#define PARTICLES_OUTPUT_DECLARE(D, C) \ + template void Particles::OutputDeclare(adios2::IO&) const; + + NTT_FOREACH_COORDINATE(PARTICLES_OUTPUT_DECLARE) +#undef PARTICLES_OUTPUT_DECLARE + +#define PARTICLES_OUTPUT_WRITE(S, M, D) \ + template void Particles::Dim, M::CoordType>::OutputWrite>( \ + adios2::IO&, \ + adios2::Engine&, \ + npart_t, \ + std::size_t, \ + std::size_t, \ + const M&); + + NTT_FOREACH_SPECIALIZATION(PARTICLES_OUTPUT_WRITE) +#undef PARTICLES_OUTPUT_WRITE + // NOLINTEND(bugprone-macro-parentheses) + +} // namespace ntt diff --git a/src/framework/containers/particles.h b/src/framework/containers/particles.h index bdbd15a86..5f15b9621 100644 --- a/src/framework/containers/particles.h +++ b/src/framework/containers/particles.h @@ -6,9 +6,10 @@ * - ntt::Particles<> : ntt::ParticleSpecies, ntt::ParticleArrays * @cpp: * - particles.cpp - * - particles_io.cpp - * - particles_comm.cpp * - particles_sort.cpp + * - comm/particles.cpp + * - io/particles.cpp + * - checkpoint/particles.cpp * @namespaces: * - ntt:: * @macros: @@ -25,7 +26,6 @@ #include "traits/metric.h" #include "utils/error.h" #include "utils/formatting.h" -#include "utils/sorting.h" #include "framework/containers/species.h" #include "framework/domain/grid.h" diff --git a/src/framework/domain/checkpoint/init.cpp b/src/framework/domain/checkpoint/init.cpp new file mode 100644 index 000000000..1176d0f18 --- /dev/null +++ b/src/framework/domain/checkpoint/init.cpp @@ -0,0 +1,107 @@ +#include "defaults.h" +#include "enums.h" +#include "global.h" + +#include "traits/metric.h" +#include "utils/error.h" +#include "utils/formatting.h" + +#include "framework/domain/metadomain.h" +#include "framework/parameters/parameters.h" +#include "framework/specialization_registry.h" +#include "output/checkpoint.h" + +#if defined(MPI_ENABLED) + #include +#endif + +#include +#include +#include + +namespace ntt { + + template + void Metadomain::InitCheckpointWriter(adios2::ADIOS* ptr_adios, + const SimulationParams& params) { + raise::ErrorIf(ptr_adios == nullptr, "adios == nullptr", HERE); + raise::ErrorIf( + l_subdomain_indices().size() != 1, + "Checkpoint writing for now is only supported for one subdomain per rank", + HERE); + auto local_domain = subdomain_ptr(l_subdomain_indices()[0]); + raise::ErrorIf(local_domain->is_placeholder(), + "local_domain is a placeholder", + HERE); + + std::vector glob_shape_with_ghosts, off_ncells_with_ghosts; + for (auto d { 0u }; d < M::Dim; ++d) { + off_ncells_with_ghosts.push_back( + local_domain->offset_ncells()[d] + + 2 * N_GHOSTS * local_domain->offset_ndomains()[d]); + glob_shape_with_ghosts.push_back( + mesh().n_active()[d] + 2 * N_GHOSTS * ndomains_per_dim()[d]); + } + auto loc_shape_with_ghosts = local_domain->mesh.n_all(); + + std::vector npld_r, npld_i; + for (auto s { 0u }; s < local_domain->species.size(); ++s) { + npld_r.push_back(local_domain->species[s].npld_r()); + npld_i.push_back(local_domain->species[s].npld_i()); + } + + const path_t checkpoint_root = params.template get( + "checkpoint.write_path"); + + g_checkpoint_writer.init( + ptr_adios, + checkpoint_root, + params.template get("checkpoint.interval"), + params.template get("checkpoint.interval_time"), + params.template get("checkpoint.keep"), + params.template get("checkpoint.walltime"), + { params.template get("adios2.aggregators_per_node", + defaults::adios2::aggregators_per_node), + params.template get("adios2.max_shm_size", + defaults::adios2::max_shm_size), + params.template get("adios2.buffer_chunk_size", + defaults::adios2::buffer_chunk_size) }); + if (g_checkpoint_writer.enabled()) { + local_domain->fields.CheckpointDeclare(g_checkpoint_writer.io(), + loc_shape_with_ghosts, + glob_shape_with_ghosts, + off_ncells_with_ghosts); + for (const auto& species : local_domain->species) { + species.CheckpointDeclare(g_checkpoint_writer.io()); + } + for (auto d { 0u }; d < M::Dim; ++d) { + g_checkpoint_writer.io().DefineVariable( + fmt::format("subdomain_x%d_min", d + 1), + { adios2::UnknownDim }, + { adios2::UnknownDim }, + { adios2::UnknownDim }); + g_checkpoint_writer.io().DefineVariable( + fmt::format("subdomain_x%d_max", d + 1), + { adios2::UnknownDim }, + { adios2::UnknownDim }, + { adios2::UnknownDim }); + g_checkpoint_writer.io().DefineVariable( + fmt::format("subdomain_nx%d", d + 1), + { adios2::UnknownDim }, + { adios2::UnknownDim }, + { adios2::UnknownDim }); + } + } + } + + // NOLINTBEGIN(bugprone-macro-parentheses) +#define METADOMAIN_CHECKPOINTS(S, M, D) \ + template void Metadomain>::InitCheckpointWriter( \ + adios2::ADIOS*, \ + const SimulationParams&); + + NTT_FOREACH_SPECIALIZATION(METADOMAIN_CHECKPOINTS) +#undef METADOMAIN_CHECKPOINTS + // NOLINTEND(bugprone-macro-parentheses) + +} // namespace ntt diff --git a/src/framework/domain/metadomain_chckpt.cpp b/src/framework/domain/checkpoint/resume.cpp similarity index 52% rename from src/framework/domain/metadomain_chckpt.cpp rename to src/framework/domain/checkpoint/resume.cpp index fbf0f11cd..d0818c935 100644 --- a/src/framework/domain/metadomain_chckpt.cpp +++ b/src/framework/domain/checkpoint/resume.cpp @@ -1,4 +1,3 @@ -#include "defaults.h" #include "enums.h" #include "global.h" @@ -10,172 +9,19 @@ #include "framework/domain/metadomain.h" #include "framework/parameters/parameters.h" #include "framework/specialization_registry.h" -#include "output/checkpoint.h" #include "output/utils/readers.h" -#include "output/utils/writers.h" #if defined(MPI_ENABLED) #include #endif #include -#include #include #include #include namespace ntt { - template - void Metadomain::InitCheckpointWriter(adios2::ADIOS* ptr_adios, - const SimulationParams& params) { - raise::ErrorIf(ptr_adios == nullptr, "adios == nullptr", HERE); - raise::ErrorIf( - l_subdomain_indices().size() != 1, - "Checkpoint writing for now is only supported for one subdomain per rank", - HERE); - auto local_domain = subdomain_ptr(l_subdomain_indices()[0]); - raise::ErrorIf(local_domain->is_placeholder(), - "local_domain is a placeholder", - HERE); - - std::vector glob_shape_with_ghosts, off_ncells_with_ghosts; - for (auto d { 0u }; d < M::Dim; ++d) { - off_ncells_with_ghosts.push_back( - local_domain->offset_ncells()[d] + - 2 * N_GHOSTS * local_domain->offset_ndomains()[d]); - glob_shape_with_ghosts.push_back( - mesh().n_active()[d] + 2 * N_GHOSTS * ndomains_per_dim()[d]); - } - auto loc_shape_with_ghosts = local_domain->mesh.n_all(); - - std::vector npld_r, npld_i; - for (auto s { 0u }; s < local_domain->species.size(); ++s) { - npld_r.push_back(local_domain->species[s].npld_r()); - npld_i.push_back(local_domain->species[s].npld_i()); - } - - const path_t checkpoint_root = params.template get( - "checkpoint.write_path"); - - g_checkpoint_writer.init( - ptr_adios, - checkpoint_root, - params.template get("checkpoint.interval"), - params.template get("checkpoint.interval_time"), - params.template get("checkpoint.keep"), - params.template get("checkpoint.walltime"), - { params.template get("adios2.aggregators_per_node", - defaults::adios2::aggregators_per_node), - params.template get("adios2.max_shm_size", - defaults::adios2::max_shm_size), - params.template get("adios2.buffer_chunk_size", - defaults::adios2::buffer_chunk_size) }); - if (g_checkpoint_writer.enabled()) { - local_domain->fields.CheckpointDeclare(g_checkpoint_writer.io(), - loc_shape_with_ghosts, - glob_shape_with_ghosts, - off_ncells_with_ghosts); - for (const auto& species : local_domain->species) { - species.CheckpointDeclare(g_checkpoint_writer.io()); - } - for (auto d { 0u }; d < M::Dim; ++d) { - g_checkpoint_writer.io().DefineVariable( - fmt::format("subdomain_x%d_min", d + 1), - { adios2::UnknownDim }, - { adios2::UnknownDim }, - { adios2::UnknownDim }); - g_checkpoint_writer.io().DefineVariable( - fmt::format("subdomain_x%d_max", d + 1), - { adios2::UnknownDim }, - { adios2::UnknownDim }, - { adios2::UnknownDim }); - g_checkpoint_writer.io().DefineVariable( - fmt::format("subdomain_nx%d", d + 1), - { adios2::UnknownDim }, - { adios2::UnknownDim }, - { adios2::UnknownDim }); - } - } - } - - template - auto Metadomain::WriteCheckpoint(const SimulationParams& params, - timestep_t current_step, - timestep_t finished_step, - simtime_t current_time, - simtime_t finished_time) -> bool { - raise::ErrorIf( - l_subdomain_indices().size() != 1, - "Checkpointing for now is only supported for one subdomain per rank", - HERE); - if (not g_checkpoint_writer.shouldSave(finished_step, finished_time) or - finished_step <= 1) { - return false; - } - auto local_domain = subdomain_ptr(l_subdomain_indices()[0]); - raise::ErrorIf(local_domain->is_placeholder(), - "local_domain is a placeholder", - HERE); - logger::Checkpoint("Writing checkpoint", HERE); - g_checkpoint_writer.beginSaving(current_step, current_time); - { - if (g_checkpoint_writer.written().empty()) { - raise::Fatal("No checkpoint file to save metadata", HERE); - } - params.saveTOML(g_checkpoint_writer.written().back().second, current_time); - - // Recompute the local with-ghosts shape/offset every step so the - // ADIOS variable selection tracks any rebalance that has happened - // since InitCheckpointWriter. - std::vector loc_off_with_ghosts; - for (auto d { 0u }; d < M::Dim; ++d) { - loc_off_with_ghosts.push_back( - local_domain->offset_ncells()[d] + - 2 * N_GHOSTS * local_domain->offset_ndomains()[d]); - } - local_domain->fields.CheckpointWrite(g_checkpoint_writer.io(), - g_checkpoint_writer.writer(), - local_domain->mesh.n_all(), - loc_off_with_ghosts); -#if !defined(MPI_ENABLED) - const std::size_t dom_tot = 1, dom_offset = 0; -#else - const std::size_t dom_tot = g_mpi_size, dom_offset = g_mpi_rank; -#endif // MPI_ENABLED - - for (const auto& species : local_domain->species) { - species.CheckpointWrite(g_checkpoint_writer.io(), - g_checkpoint_writer.writer(), - dom_tot, - dom_offset); - } - for (auto d { 0u }; d < M::Dim; ++d) { - out::WriteVariable(g_checkpoint_writer.io(), - g_checkpoint_writer.writer(), - fmt::format("subdomain_x%d_min", d + 1), - local_domain->mesh.extent()[d].first, - dom_tot, - dom_offset); - out::WriteVariable(g_checkpoint_writer.io(), - g_checkpoint_writer.writer(), - fmt::format("subdomain_x%d_max", d + 1), - local_domain->mesh.extent()[d].second, - dom_tot, - dom_offset); - out::WriteVariable(g_checkpoint_writer.io(), - g_checkpoint_writer.writer(), - fmt::format("subdomain_nx%d", d + 1), - local_domain->mesh.n_active()[d], - dom_tot, - dom_offset); - } - } - g_checkpoint_writer.endSaving(); - logger::Checkpoint("Checkpoint written", HERE); - return true; - } - template void Metadomain::redecomposeFromCheckpoint( const std::vector>& dom_ncells, @@ -363,20 +209,13 @@ namespace ntt { // NOLINTBEGIN(bugprone-macro-parentheses) #define METADOMAIN_CHECKPOINTS(S, M, D) \ - template void Metadomain>::InitCheckpointWriter( \ - adios2::ADIOS*, \ - const SimulationParams&); \ - template auto Metadomain>::WriteCheckpoint(const SimulationParams&, \ - timestep_t, \ - timestep_t, \ - simtime_t, \ - simtime_t) -> bool; \ template void Metadomain>::ContinueFromCheckpoint( \ adios2::ADIOS*, \ const SimulationParams&); \ template void Metadomain>::redecomposeFromCheckpoint( \ const std::vector>&, \ const std::vector>&); + NTT_FOREACH_SPECIALIZATION(METADOMAIN_CHECKPOINTS) #undef METADOMAIN_CHECKPOINTS // NOLINTEND(bugprone-macro-parentheses) diff --git a/src/framework/domain/checkpoint/write.cpp b/src/framework/domain/checkpoint/write.cpp new file mode 100644 index 000000000..714795da5 --- /dev/null +++ b/src/framework/domain/checkpoint/write.cpp @@ -0,0 +1,114 @@ +#include "enums.h" +#include "global.h" + +#include "traits/metric.h" +#include "utils/error.h" +#include "utils/formatting.h" +#include "utils/log.h" + +#include "framework/domain/metadomain.h" +#include "framework/parameters/parameters.h" +#include "framework/specialization_registry.h" +#include "output/checkpoint.h" +#include "output/utils/writers.h" + +#if defined(MPI_ENABLED) + #include +#endif + +#include +#include +#include + +namespace ntt { + + template + auto Metadomain::WriteCheckpoint(const SimulationParams& params, + timestep_t current_step, + timestep_t finished_step, + simtime_t current_time, + simtime_t finished_time) -> bool { + raise::ErrorIf( + l_subdomain_indices().size() != 1, + "Checkpointing for now is only supported for one subdomain per rank", + HERE); + if (not g_checkpoint_writer.shouldSave(finished_step, finished_time) or + finished_step <= 1) { + return false; + } + auto local_domain = subdomain_ptr(l_subdomain_indices()[0]); + raise::ErrorIf(local_domain->is_placeholder(), + "local_domain is a placeholder", + HERE); + logger::Checkpoint("Writing checkpoint", HERE); + g_checkpoint_writer.beginSaving(current_step, current_time); + { + if (g_checkpoint_writer.written().empty()) { + raise::Fatal("No checkpoint file to save metadata", HERE); + } + params.saveTOML(g_checkpoint_writer.written().back().second, current_time); + + // Recompute the local with-ghosts shape/offset every step so the + // ADIOS variable selection tracks any rebalance that has happened + // since InitCheckpointWriter. + std::vector loc_off_with_ghosts; + for (auto d { 0u }; d < M::Dim; ++d) { + loc_off_with_ghosts.push_back( + local_domain->offset_ncells()[d] + + 2 * N_GHOSTS * local_domain->offset_ndomains()[d]); + } + local_domain->fields.CheckpointWrite(g_checkpoint_writer.io(), + g_checkpoint_writer.writer(), + local_domain->mesh.n_all(), + loc_off_with_ghosts); +#if !defined(MPI_ENABLED) + const std::size_t dom_tot = 1, dom_offset = 0; +#else + const std::size_t dom_tot = g_mpi_size, dom_offset = g_mpi_rank; +#endif // MPI_ENABLED + + for (const auto& species : local_domain->species) { + species.CheckpointWrite(g_checkpoint_writer.io(), + g_checkpoint_writer.writer(), + dom_tot, + dom_offset); + } + for (auto d { 0u }; d < M::Dim; ++d) { + out::WriteVariable(g_checkpoint_writer.io(), + g_checkpoint_writer.writer(), + fmt::format("subdomain_x%d_min", d + 1), + local_domain->mesh.extent()[d].first, + dom_tot, + dom_offset); + out::WriteVariable(g_checkpoint_writer.io(), + g_checkpoint_writer.writer(), + fmt::format("subdomain_x%d_max", d + 1), + local_domain->mesh.extent()[d].second, + dom_tot, + dom_offset); + out::WriteVariable(g_checkpoint_writer.io(), + g_checkpoint_writer.writer(), + fmt::format("subdomain_nx%d", d + 1), + local_domain->mesh.n_active()[d], + dom_tot, + dom_offset); + } + } + g_checkpoint_writer.endSaving(); + logger::Checkpoint("Checkpoint written", HERE); + return true; + } + + // NOLINTBEGIN(bugprone-macro-parentheses) +#define METADOMAIN_CHECKPOINTS(S, M, D) \ + template auto Metadomain>::WriteCheckpoint(const SimulationParams&, \ + timestep_t, \ + timestep_t, \ + simtime_t, \ + simtime_t) -> bool; + + NTT_FOREACH_SPECIALIZATION(METADOMAIN_CHECKPOINTS) +#undef METADOMAIN_CHECKPOINTS + // NOLINTEND(bugprone-macro-parentheses) + +} // namespace ntt diff --git a/src/framework/domain/comm/fields.cpp b/src/framework/domain/comm/fields.cpp new file mode 100644 index 000000000..121ad48e5 --- /dev/null +++ b/src/framework/domain/comm/fields.cpp @@ -0,0 +1,203 @@ +#include "enums.h" +#include "global.h" + +#include "arch/directions.h" +#include "traits/metric.h" +#include "utils/error.h" +#include "utils/formatting.h" +#include "utils/log.h" + +#include "framework/domain/domain.h" +#include "framework/domain/metadomain.h" +#include "framework/specialization_registry.h" + +#include "framework/domain/comm/utils.hpp" + +#if defined(MPI_ENABLED) + #include "framework/domain/comm/fields_mpi.hpp" +#else + #include "framework/domain/comm/fields_nompi.hpp" +#endif + +#include + +#include +#include + +namespace ntt { + + template + void Metadomain::CommunicateFields(Domain& domain, + CommTags tags) const { + const auto comm_em = ((S == SimEngine::SRPIC) and + ((tags & Comm::E) or (tags & Comm::B))) or + ((S == SimEngine::GRPIC) and + ((tags & Comm::D) or (tags & Comm::B))); + const bool comm_em0 = (S == SimEngine::GRPIC) and + ((tags & Comm::B0) or (tags & Comm::D0)); + const bool comm_j = (tags & Comm::J); + const bool comm_aux = (S == SimEngine::GRPIC) and + ((tags & Comm::E) or (tags & Comm::H)); + raise::ErrorIf(not(comm_em or comm_em0 or comm_j or comm_aux), + "CommunicateFields called with no task", + HERE); + + std::string comms; + if (tags & Comm::E) { + comms += "E "; + } + if (tags & Comm::B) { + comms += "B "; + } + if (tags & Comm::J) { + comms += "J "; + } + if (tags & Comm::D) { + comms += "D "; + } + if (tags & Comm::H) { + comms += "H "; + } + if (tags & Comm::D0) { + comms += "D0 "; + } + if (tags & Comm::B0) { + comms += "B0 "; + } + logger::Checkpoint(fmt::format("Communicating %s\n", comms.c_str()), HERE); + + /** + * @note this block is designed to support in the future multiple domains + * on a single rank, however that is not yet implemented + */ + // establish the last index ranges for fields (i.e., components) + auto comp_range_fld = cell_range_t {}; + auto comp_range_cur = cell_range_t {}; + if constexpr (S == SimEngine::GRPIC) { + if (((tags & Comm::D) and (tags & Comm::B)) or + ((tags & Comm::D0) and (tags & Comm::B0)) or + ((tags & Comm::E) and (tags & Comm::H))) { + comp_range_fld = cell_range_t(em::dx1, em::bx3 + 1); + } else if ((tags & Comm::D) or (tags & Comm::D0) or (tags & Comm::E)) { + comp_range_fld = cell_range_t(em::dx1, em::dx3 + 1); + } else if ((tags & Comm::B) or (tags & Comm::B0) or (tags & Comm::H)) { + comp_range_fld = cell_range_t(em::bx1, em::bx3 + 1); + } + } else if constexpr (S == SimEngine::SRPIC) { + if ((tags & Comm::E) and (tags & Comm::B)) { + comp_range_fld = cell_range_t(em::ex1, em::bx3 + 1); + } else if (tags & Comm::E) { + comp_range_fld = cell_range_t(em::ex1, em::ex3 + 1); + } else if (tags & Comm::B) { + comp_range_fld = cell_range_t(em::bx1, em::bx3 + 1); + } + } else { + raise::Error("Unknown simulation engine", HERE); + } + if (comm_j) { + comp_range_cur = cell_range_t(cur::jx1, cur::jx3 + 1); + } + // traverse in all directions and send/recv the fields + for (auto& direction : dir::Directions::all) { + const auto [send_params, + recv_params] = GetSendRecvParams(this, domain, direction, false); + const auto [send_indrank, send_slice] = send_params; + const auto [recv_indrank, recv_slice] = recv_params; + const auto [send_ind, send_rank] = send_indrank; + const auto [recv_ind, recv_rank] = recv_indrank; + if (send_rank < 0 and recv_rank < 0) { + continue; + } + if (comm_em) { + comm::CommunicateField(domain.index(), + domain.fields.em, + domain.fields.em, + send_ind, + recv_ind, + send_rank, + recv_rank, + send_slice, + recv_slice, + comp_range_fld, + false); + } + if constexpr (S == SimEngine::GRPIC) { + if (comm_aux) { + comm::CommunicateField(domain.index(), + domain.fields.aux, + domain.fields.aux, + send_ind, + recv_ind, + send_rank, + recv_rank, + send_slice, + recv_slice, + comp_range_fld, + false); + } + if (comm_em0) { + comm::CommunicateField(domain.index(), + domain.fields.em0, + domain.fields.em0, + send_ind, + recv_ind, + send_rank, + recv_rank, + send_slice, + recv_slice, + comp_range_fld, + false); + // @HACK_GR_1.2.0 -- this has to be done carefully + // comm::CommunicateField(domain.index(), + // domain.fields.aux, + // domain.fields.aux, + // send_ind, + // recv_ind, + // send_rank, + // recv_rank, + // send_slice, + // recv_slice, + // comp_range_fld, + // false); + } + if (comm_j) { + comm::CommunicateField(domain.index(), + domain.fields.cur0, + domain.fields.cur0, + send_ind, + recv_ind, + send_rank, + recv_rank, + send_slice, + recv_slice, + comp_range_cur, + false); + } + } else { + if (comm_j) { + comm::CommunicateField(domain.index(), + domain.fields.cur, + domain.fields.cur, + send_ind, + recv_ind, + send_rank, + recv_rank, + send_slice, + recv_slice, + comp_range_cur, + false); + } + } + } + } + + // NOLINTBEGIN(bugprone-macro-parentheses) +#define METADOMAIN_COMM(S, M, D) \ + template void Metadomain>::CommunicateFields(Domain>&, \ + CommTags) const; + + NTT_FOREACH_SPECIALIZATION(METADOMAIN_COMM) +#undef METADOMAIN_COMM + // NOLINTEND(bugprone-macro-parentheses) + +} // namespace ntt diff --git a/src/framework/domain/comm_mpi.hpp b/src/framework/domain/comm/fields_mpi.hpp similarity index 98% rename from src/framework/domain/comm_mpi.hpp rename to src/framework/domain/comm/fields_mpi.hpp index 52103c170..4ec60174b 100644 --- a/src/framework/domain/comm_mpi.hpp +++ b/src/framework/domain/comm/fields_mpi.hpp @@ -1,6 +1,6 @@ /** - * @file framework/domain/comm_mpi.hpp - * @brief MPI communication routines + * @file framework/domain/comm/fields_mpi.hpp + * @brief MPI communication routines for fields * @implements * - comm::CommunicateField<> -> void * @namespaces: @@ -8,8 +8,8 @@ * @note This should only be included if the MPI_ENABLED flag is set */ -#ifndef FRAMEWORK_DOMAIN_COMM_MPI_HPP -#define FRAMEWORK_DOMAIN_COMM_MPI_HPP +#ifndef FRAMEWORK_DOMAIN_COMM_FIELDS_MPI_HPP +#define FRAMEWORK_DOMAIN_COMM_FIELDS_MPI_HPP #include "global.h" @@ -367,4 +367,4 @@ namespace comm { } // namespace comm -#endif // FRAMEWORK_DOMAIN_COMM_MPI_HPP +#endif // FRAMEWORK_DOMAIN_COMM_FIELDS_MPI_HPP diff --git a/src/framework/domain/comm_nompi.hpp b/src/framework/domain/comm/fields_nompi.hpp similarity index 95% rename from src/framework/domain/comm_nompi.hpp rename to src/framework/domain/comm/fields_nompi.hpp index 16e20d261..05c12730e 100644 --- a/src/framework/domain/comm_nompi.hpp +++ b/src/framework/domain/comm/fields_nompi.hpp @@ -1,6 +1,6 @@ /** - * @file framework/domain/comm_nompi.hpp - * @brief Communication routines without mpi + * @file framework/domain/comm/fields_nompi.hpp + * @brief Communication routines for fields without mpi * @implements * - comm::CommunicateField<> -> void * @namespaces: @@ -8,8 +8,8 @@ * @note This should only be included if the MPI_ENABLED flag is not set */ -#ifndef FRAMEWORK_DOMAIN_COMM_NOMPI_HPP -#define FRAMEWORK_DOMAIN_COMM_NOMPI_HPP +#ifndef FRAMEWORK_DOMAIN_COMM_FIELDS_NOMPI_HPP +#define FRAMEWORK_DOMAIN_COMM_FIELDS_NOMPI_HPP #include "global.h" @@ -120,4 +120,4 @@ namespace comm { } // namespace comm -#endif // FRAMEWORK_DOMAIN_COMM_NOMPI_HPP +#endif // FRAMEWORK_DOMAIN_COMM_FIELDS_NOMPI_HPP diff --git a/src/framework/domain/comm/fields_sync.cpp b/src/framework/domain/comm/fields_sync.cpp new file mode 100644 index 000000000..eadd2c679 --- /dev/null +++ b/src/framework/domain/comm/fields_sync.cpp @@ -0,0 +1,236 @@ +#include "enums.h" +#include "global.h" + +#include "arch/directions.h" +#include "arch/kokkos_aliases.h" +#include "traits/metric.h" +#include "utils/error.h" +#include "utils/formatting.h" +#include "utils/log.h" +#include "utils/numeric.h" + +#include "framework/domain/domain.h" +#include "framework/domain/metadomain.h" +#include "framework/specialization_registry.h" + +#include "framework/domain/comm/utils.hpp" + +#if defined(MPI_ENABLED) + #include "framework/domain/comm/fields_mpi.hpp" +#else + #include "framework/domain/comm/fields_nompi.hpp" +#endif + +#include + +#include +#include + +namespace ntt { + + template + void AddBufferedFields(ndfield_t& field, + ndfield_t& buffer, + const range_t& range_policy, + const cell_range_t& components) { + const auto cmin = components.first; + const auto cmax = components.second; + if constexpr (D == Dim::_1D) { + Kokkos::parallel_for( + "AddBufferedFields", + range_policy, + Lambda(cellidx_t i1) { + for (auto c { cmin }; c < cmax; ++c) { + field(i1, c) += buffer(i1, c); + } + }); + } else if constexpr (D == Dim::_2D) { + Kokkos::parallel_for( + "AddBufferedFields", + range_policy, + Lambda(cellidx_t i1, cellidx_t i2) { + for (auto c { cmin }; c < cmax; ++c) { + field(i1, i2, c) += buffer(i1, i2, c); + } + }); + } else if constexpr (D == Dim::_3D) { + Kokkos::parallel_for( + "AddBuffers", + range_policy, + Lambda(cellidx_t i1, cellidx_t i2, cellidx_t i3) { + for (auto c { cmin }; c < cmax; ++c) { + field(i1, i2, i3, c) += buffer(i1, i2, i3, c); + } + }); + } else { + raise::Error("Wrong Dimension", HERE); + } + } + + template + void Metadomain::SynchronizeFields(Domain& domain, + CommTags tags, + const cell_range_t& components) const { + const bool comm_j = (tags & Comm::J); + const bool comm_bckp = (tags & Comm::Bckp); + const bool comm_buff = (tags & Comm::Buff); + raise::ErrorIf(not(comm_j || comm_bckp || comm_buff), + "SynchronizeFields called with no task or incorrect task", + HERE); + raise::ErrorIf(comm_j and comm_buff, + "SynchronizeFields cannot sync J and Buff at the same time", + HERE); + const auto synchronize = true; + + std::string comms; + if (comm_j) { + comms += "J "; + } + if (comm_bckp) { + comms += "Bckp "; + } + if (comm_buff) { + comms += "Buff "; + } + logger::Checkpoint(fmt::format("Synchronizing %s\n", comms.c_str()), HERE); + + auto comp_range_cur = cell_range_t {}; + if (comm_j) { + comp_range_cur = cell_range_t(cur::jx1, cur::jx3 + 1); + Kokkos::deep_copy(domain.fields.buff, ZERO); + } + ndfield_t bckp_recv; + ndfield_t buff_recv; + if (comm_bckp) { + if constexpr (M::Dim == Dim::_1D) { + bckp_recv = ndfield_t { "bckp_recv", + domain.fields.bckp.extent(0) }; + } else if constexpr (M::Dim == Dim::_2D) { + bckp_recv = ndfield_t { "bckp_recv", + domain.fields.bckp.extent(0), + domain.fields.bckp.extent(1) }; + } else if constexpr (M::Dim == Dim::_3D) { + bckp_recv = ndfield_t { "bckp_recv", + domain.fields.bckp.extent(0), + domain.fields.bckp.extent(1), + domain.fields.bckp.extent(2) }; + } + } + if (comm_buff) { + if constexpr (M::Dim == Dim::_1D) { + buff_recv = ndfield_t { "buff_recv", + domain.fields.buff.extent(0) }; + } else if constexpr (M::Dim == Dim::_2D) { + buff_recv = ndfield_t { "buff_recv", + domain.fields.buff.extent(0), + domain.fields.buff.extent(1) }; + } else if constexpr (M::Dim == Dim::_3D) { + buff_recv = ndfield_t { "buff_recv", + domain.fields.buff.extent(0), + domain.fields.buff.extent(1), + domain.fields.buff.extent(2) }; + } + } + // traverse in all directions and sync the fields + for (auto& direction : dir::Directions::all) { + const auto [send_params, + recv_params] = GetSendRecvParams(this, domain, direction, true); + const auto [send_indrank, send_slice] = send_params; + const auto [recv_indrank, recv_slice] = recv_params; + const auto [send_ind, send_rank] = send_indrank; + const auto [recv_ind, recv_rank] = recv_indrank; + if (send_rank < 0 and recv_rank < 0) { + continue; + } + if (comm_j) { + if constexpr (S == SimEngine::GRPIC) { + comm::CommunicateField(domain.index(), + domain.fields.cur0, + domain.fields.buff, + send_ind, + recv_ind, + send_rank, + recv_rank, + send_slice, + recv_slice, + comp_range_cur, + synchronize); + } else { + comm::CommunicateField(domain.index(), + domain.fields.cur, + domain.fields.buff, + send_ind, + recv_ind, + send_rank, + recv_rank, + send_slice, + recv_slice, + comp_range_cur, + synchronize); + } + } + if (comm_bckp) { + comm::CommunicateField(domain.index(), + domain.fields.bckp, + bckp_recv, + send_ind, + recv_ind, + send_rank, + recv_rank, + send_slice, + recv_slice, + components, + synchronize); + } + if (comm_buff) { + comm::CommunicateField(domain.index(), + domain.fields.buff, + buff_recv, + send_ind, + recv_ind, + send_rank, + recv_rank, + send_slice, + recv_slice, + components, + synchronize); + } + } + if (comm_j) { + if constexpr (S == SimEngine::GRPIC) { + AddBufferedFields(domain.fields.cur0, + domain.fields.buff, + domain.mesh.rangeActiveCells(), + comp_range_cur); + } else { + AddBufferedFields(domain.fields.cur, + domain.fields.buff, + domain.mesh.rangeActiveCells(), + comp_range_cur); + } + } + if (comm_bckp) { + AddBufferedFields(domain.fields.bckp, + bckp_recv, + domain.mesh.rangeActiveCells(), + components); + } + if (comm_buff) { + AddBufferedFields(domain.fields.buff, + buff_recv, + domain.mesh.rangeActiveCells(), + components); + } + } + + // NOLINTBEGIN(bugprone-macro-parentheses) +#define METADOMAIN_COMM(S, M, D) \ + template void Metadomain>::SynchronizeFields(Domain>&, \ + CommTags, \ + const cell_range_t&) const; + + NTT_FOREACH_SPECIALIZATION(METADOMAIN_COMM) +#undef METADOMAIN_COMM + // NOLINTEND(bugprone-macro-parentheses) + +} // namespace ntt diff --git a/src/framework/domain/comm/particles.cpp b/src/framework/domain/comm/particles.cpp new file mode 100644 index 000000000..1bb53181e --- /dev/null +++ b/src/framework/domain/comm/particles.cpp @@ -0,0 +1,126 @@ +#include "enums.h" +#include "global.h" + +#include "arch/directions.h" +#include "arch/kokkos_aliases.h" +#include "traits/metric.h" +#include "utils/log.h" + +#include "framework/domain/domain.h" +#include "framework/domain/metadomain.h" +#include "framework/specialization_registry.h" + +#include "framework/domain/comm/utils.hpp" + +#if defined(MPI_ENABLED) + #include "arch/mpi_tags.h" +#endif + +#include + +#include +#include +#include + +namespace ntt { + + template + void Metadomain::CommunicateParticles(Domain& domain) const { +#if defined(MPI_ENABLED) + logger::Checkpoint("Communicating particles\n", HERE); + for (auto& species : domain.species) { + const auto ntags = species.ntags(); + + // coordinate shifts per each direction + array_t shifts_in_x1 { "shifts_in_x1", ntags - 2 }; + array_t shifts_in_x2 { "shifts_in_x2", ntags - 2 }; + array_t shifts_in_x3 { "shifts_in_x3", ntags - 2 }; + auto shifts_in_x1_h = Kokkos::create_mirror_view(shifts_in_x1); + auto shifts_in_x2_h = Kokkos::create_mirror_view(shifts_in_x2); + auto shifts_in_x3_h = Kokkos::create_mirror_view(shifts_in_x3); + + // all directions requiring communication + dir::dirs_t dirs_to_comm; + + // ranks & indices of meshblock to send/recv from + dir::map_t send_ranks; + dir::map_t recv_ranks; + + for (const auto& direction : dir::Directions::all) { + // tags corresponding to the direction (both send & recv) + const auto tag_send = mpi::PrtlSendTag::dir2tag(direction); + + // get indices & ranks of send/recv meshblocks + const auto [send_params, + recv_params] = GetSendRecvRanks(this, domain, direction); + const auto [send_ind, send_rank] = send_params; + const auto [recv_ind, recv_rank] = recv_params; + + // skip if no communication is necessary + const auto is_sending = (send_rank >= 0); + const auto is_receiving = (recv_rank >= 0); + if (not is_sending and not is_receiving) { + continue; + } + dirs_to_comm.push_back(direction); + send_ranks[direction] = send_rank; + recv_ranks[direction] = recv_rank; + + // if sending, record displacements to apply before + // ... tag_send - 2: because we only shift tags > 2 (i.e. no dead/alive) + if (is_sending) { + if constexpr (D == Dim::_1D || D == Dim::_2D || D == Dim::_3D) { + if (direction[0] == -1) { + // sending backwards in x1 (add sx1 of target meshblock) + shifts_in_x1_h(tag_send - 2) = subdomain(send_ind).mesh.n_active( + in::x1); + } else if (direction[0] == 1) { + // sending forward in x1 (subtract sx1 of source meshblock) + shifts_in_x1_h(tag_send - 2) = -domain.mesh.n_active(in::x1); + } + } + if constexpr (D == Dim::_2D || D == Dim::_3D) { + if (direction[1] == -1) { + shifts_in_x2_h(tag_send - 2) = subdomain(send_ind).mesh.n_active( + in::x2); + } else if (direction[1] == 1) { + shifts_in_x2_h(tag_send - 2) = -domain.mesh.n_active(in::x2); + } + } + if constexpr (D == Dim::_3D) { + if (direction[2] == -1) { + shifts_in_x3_h(tag_send - 2) = subdomain(send_ind).mesh.n_active( + in::x3); + } else if (direction[2] == 1) { + shifts_in_x3_h(tag_send - 2) = -domain.mesh.n_active(in::x3); + } + } + } + } // end directions loop + + Kokkos::deep_copy(shifts_in_x1, shifts_in_x1_h); + Kokkos::deep_copy(shifts_in_x2, shifts_in_x2_h); + Kokkos::deep_copy(shifts_in_x3, shifts_in_x3_h); + + species.Communicate(dirs_to_comm, + shifts_in_x1, + shifts_in_x2, + shifts_in_x3, + send_ranks, + recv_ranks); + + } // end species loop +#else + (void)domain; +#endif + } + + // NOLINTBEGIN(bugprone-macro-parentheses) +#define METADOMAIN_COMM(S, M, D) \ + template void Metadomain>::CommunicateParticles(Domain>&) const; + + NTT_FOREACH_SPECIALIZATION(METADOMAIN_COMM) +#undef METADOMAIN_COMM + // NOLINTEND(bugprone-macro-parentheses) + +} // namespace ntt diff --git a/src/framework/domain/comm/utils.hpp b/src/framework/domain/comm/utils.hpp new file mode 100644 index 000000000..979d321c9 --- /dev/null +++ b/src/framework/domain/comm/utils.hpp @@ -0,0 +1,209 @@ +/** + * @file framework/domain/comm/utils.hpp + * @brief Utility functions for inter-domain communication + * @implements + * - ntt::GetSendRecvRanks<> -> std::pair + * - ntt::GetSendRecvParams<> -> std::pair + * @namespaces: + * - ntt:: + * @macros: + * - MPI_ENABLED + * - OUTPUT_ENABLED + */ + +#ifndef FRAMEWORK_DOMAIN_COMM_UTILS_HPP +#define FRAMEWORK_DOMAIN_COMM_UTILS_HPP + +#include "enums.h" +#include "global.h" + +#include "arch/directions.h" +#include "traits/metric.h" +#include "utils/error.h" +#include "utils/formatting.h" + +#include "framework/domain/domain.h" +#include "framework/domain/metadomain.h" + +#include + +#include +#include + +namespace ntt { + + using address_t = std::pair; + using comm_params_t = std::pair>; + + template + auto GetSendRecvRanks(const Metadomain* const metadomain, + Domain& domain, + const dir::direction_t& direction) + -> std::pair { + const Domain* send_to_nghbr_ptr = nullptr; + const Domain* recv_from_nghbr_ptr = nullptr; + // set pointers to the correct send/recv domains + // can coincide with the current domain if periodic + if (domain.mesh.flds_bc_in(direction) == FldsBC::PERIODIC) { + // sending / receiving from itself + raise::ErrorIf( + domain.neighbor_idx_in(direction) != domain.index(), + fmt::format( + "Periodic boundaries in `%s` imply communication within the " + "same domain, but %u != %u", + direction.to_string().c_str(), + domain.neighbor_idx_in(direction), + domain.index()), + HERE); + raise::ErrorIf( + domain.mesh.flds_bc_in(-direction) != FldsBC::PERIODIC, + "Periodic boundary conditions must be set in both directions", + HERE); + send_to_nghbr_ptr = &domain; + recv_from_nghbr_ptr = &domain; + } else if (domain.mesh.flds_bc_in(direction) == FldsBC::SYNC) { + // sending to other domain + raise::ErrorIf( + domain.neighbor_idx_in(direction) == domain.index(), + "Sync boundaries imply communication between separate domains", + HERE); + send_to_nghbr_ptr = metadomain->subdomain_ptr( + domain.neighbor_idx_in(direction)); + if (domain.mesh.flds_bc_in(-direction) == FldsBC::SYNC) { + // receiving from other domain + raise::ErrorIf( + domain.neighbor_idx_in(-direction) == domain.index(), + "Sync boundaries imply communication between separate domains", + HERE); + recv_from_nghbr_ptr = metadomain->subdomain_ptr( + domain.neighbor_idx_in(-direction)); + } + } else if (domain.mesh.flds_bc_in(-direction) == FldsBC::SYNC) { + // only receiving from other domain + raise::ErrorIf( + domain.neighbor_idx_in(-direction) == domain.index(), + "Sync boundaries imply communication between separate domains", + HERE); + recv_from_nghbr_ptr = metadomain->subdomain_ptr( + domain.neighbor_idx_in(-direction)); + } else { + // no communication necessary + return { + { 0, -1 }, + { 0, -1 } + }; + } +#if defined(MPI_ENABLED) + const auto send_rank = (send_to_nghbr_ptr != nullptr) + ? send_to_nghbr_ptr->mpi_rank() + : -1; + const auto recv_rank = (recv_from_nghbr_ptr != nullptr) + ? recv_from_nghbr_ptr->mpi_rank() + : -1; +#else + const auto send_rank = (send_to_nghbr_ptr != nullptr) ? 0 : -1; + const auto recv_rank = (recv_from_nghbr_ptr != nullptr) ? 0 : -1; +#endif + const auto send_ind = (send_to_nghbr_ptr != nullptr) + ? send_to_nghbr_ptr->index() + : 0; + const auto recv_ind = (recv_from_nghbr_ptr != nullptr) + ? recv_from_nghbr_ptr->index() + : 0; + (void)send_rank; + (void)recv_rank; + return { + { send_ind, send_rank }, + { recv_ind, recv_rank } + }; + } + + template + auto GetSendRecvParams(const Metadomain* const metadomain, + Domain& domain, + dir::direction_t direction, + bool synchronize) + -> std::pair { + const auto [send_indrank, + recv_indrank] = GetSendRecvRanks(metadomain, domain, direction); + const auto [send_ind, send_rank] = send_indrank; + const auto [recv_ind, recv_rank] = recv_indrank; + const auto is_sending = (send_rank >= 0); + const auto is_receiving = (recv_rank >= 0); + if (not(is_sending or is_receiving)) { + return { + { { 0, -1 }, {} }, + { { 0, -1 }, {} } + }; + } + auto send_slice = std::vector {}; + auto recv_slice = std::vector {}; + const in components[] = { in::x1, in::x2, in::x3 }; + // find the field components and indices to be sent/received + for (auto d { 0u }; d < direction.size(); ++d) { + const auto c = components[d]; + const auto dir = direction[d]; + if (not synchronize) { + // recv to: ghost zones + // send from: active zones + if (is_sending) { + if (dir == 0) { + send_slice.emplace_back(domain.mesh.i_min(c), domain.mesh.i_max(c)); + } else if (dir == 1) { + send_slice.emplace_back(domain.mesh.i_max(c) - N_GHOSTS, + domain.mesh.i_max(c)); + } else { + send_slice.emplace_back(domain.mesh.i_min(c), + domain.mesh.i_min(c) + N_GHOSTS); + } + } + if (is_receiving) { + if (-dir == 0) { + recv_slice.emplace_back(domain.mesh.i_min(c), domain.mesh.i_max(c)); + } else if (-dir == 1) { + recv_slice.emplace_back(domain.mesh.i_max(c), + domain.mesh.i_max(c) + N_GHOSTS); + } else { + recv_slice.emplace_back(domain.mesh.i_min(c) - N_GHOSTS, + domain.mesh.i_min(c)); + } + } + } else { + // recv to: active + ghost zones + // send from: active + ghost zones + if (is_sending) { + if (dir == 0) { + send_slice.emplace_back(domain.mesh.i_min(c) - N_GHOSTS, + domain.mesh.i_max(c) + N_GHOSTS); + } else if (dir == 1) { + send_slice.emplace_back(domain.mesh.i_max(c) - N_GHOSTS, + domain.mesh.i_max(c) + N_GHOSTS); + } else { + send_slice.emplace_back(domain.mesh.i_min(c) - N_GHOSTS, + domain.mesh.i_min(c) + N_GHOSTS); + } + } + if (is_receiving) { + if (-dir == 0) { + recv_slice.emplace_back(domain.mesh.i_min(c) - N_GHOSTS, + domain.mesh.i_max(c) + N_GHOSTS); + } else if (-dir == 1) { + recv_slice.emplace_back(domain.mesh.i_max(c) - N_GHOSTS, + domain.mesh.i_max(c) + N_GHOSTS); + } else { + recv_slice.emplace_back(domain.mesh.i_min(c) - N_GHOSTS, + domain.mesh.i_min(c) + N_GHOSTS); + } + } + } + } + + return { + { { send_ind, send_rank }, send_slice }, + { { recv_ind, recv_rank }, recv_slice }, + }; + } + +} // namespace ntt + +#endif // FRAMEWORK_DOMAIN_COMM_UTILS_HPP diff --git a/src/framework/domain/comm/vector_potential.cpp b/src/framework/domain/comm/vector_potential.cpp new file mode 100644 index 000000000..dfa749e46 --- /dev/null +++ b/src/framework/domain/comm/vector_potential.cpp @@ -0,0 +1,127 @@ +#include "enums.h" +#include "global.h" + +#include "arch/kokkos_aliases.h" +#include "traits/metric.h" +#include "utils/error.h" + +#include "framework/domain/domain.h" +#include "framework/domain/mesh.h" +#include "framework/domain/metadomain.h" +#include "framework/specialization_registry.h" + +#include +#include +#include + +#if defined(MPI_ENABLED) + #include "arch/mpi_aliases.h" + + #include +#endif // MPI_ENABLED + +#include +#include + +namespace ntt { + +#if defined(MPI_ENABLED) && defined(OUTPUT_ENABLED) + template + void ExtractVectorPotential(ndfield_t& buffer, + array_t& aphi_r, + unsigned short buff_idx, + const Mesh& mesh) { + Kokkos::parallel_for( + "AddVectorPotential", + mesh.rangeActiveCells(), + Lambda(cellidx_t i1, cellidx_t i2) { + buffer(i1, i2, buff_idx) += aphi_r(i1 - N_GHOSTS); + }); + } + + template + void Metadomain::CommunicateVectorPotential(unsigned short buff_idx) { + if constexpr (M::Dim == Dim::_2D) { + auto local_domain = subdomain_ptr(l_subdomain_indices()[0]); + const auto nx1 = local_domain->mesh.n_active(in::x1); + const auto nx2 = local_domain->mesh.n_active(in::x2); + + auto& buffer = local_domain->fields.bckp; + + const auto nranks_x1 = ndomains_per_dim()[0]; + const auto nranks_x2 = ndomains_per_dim()[1]; + + for (auto nr2 { 1u }; nr2 < nranks_x2; ++nr2) { + const auto rank_send_pre = (nr2 - 1u) * nranks_x1; + const auto rank_recv_pre = nr2 * nranks_x1; + for (auto nr1 { 0u }; nr1 < nranks_x1; ++nr1) { + const auto rank_send = rank_send_pre + nr1; + const auto rank_recv = rank_recv_pre + nr1; + if (static_cast(local_domain->mpi_rank()) == rank_send) { + array_t aphi_r { "Aphi_r", nx1 }; + Kokkos::deep_copy( + aphi_r, + Kokkos::subview(buffer, + std::make_pair(N_GHOSTS, N_GHOSTS + nx1), + N_GHOSTS + nx2 - 1, + buff_idx)); + #if !defined(DEVICE_ENABLED) || defined(GPU_AWARE_MPI) + MPI_Send(aphi_r.data(), + nx1, + mpi::get_type(), + rank_recv, + 0, + MPI_COMM_WORLD); + #else + auto aphi_r_h = Kokkos::create_mirror_view(aphi_r); + Kokkos::deep_copy(aphi_r_h, aphi_r); + MPI_Send(aphi_r_h.data(), + nx1, + mpi::get_type(), + rank_recv, + 0, + MPI_COMM_WORLD); + #endif + } else if (local_domain->mpi_rank() == rank_recv) { + array_t aphi_r { "Aphi_r", nx1 }; + #if !defined(DEVICE_ENABLED) || defined(GPU_AWARE_MPI) + MPI_Recv(aphi_r.data(), + nx1, + mpi::get_type(), + rank_send, + 0, + MPI_COMM_WORLD, + MPI_STATUS_IGNORE); + #else + auto aphi_r_h = Kokkos::create_mirror_view(aphi_r); + MPI_Recv(aphi_r_h.data(), + nx1, + mpi::get_type(), + rank_send, + 0, + MPI_COMM_WORLD, + MPI_STATUS_IGNORE); + Kokkos::deep_copy(aphi_r, aphi_r_h); + #endif + ExtractVectorPotential(buffer, aphi_r, buff_idx, local_domain->mesh); + } + } + } + } else { + raise::Error("CommunicateVectorPotential: comm vector potential only " + "possible for 2D", + HERE); + } + } + + // NOLINTBEGIN(bugprone-macro-parentheses) + #define COMMVECTORPOTENTIAL(S, M, D) \ + template void Metadomain>::CommunicateVectorPotential(unsigned short); + + NTT_FOREACH_SPECIALIZATION(COMMVECTORPOTENTIAL) + + #undef COMMVECTORPOTENTIAL +#endif + // NOLINTEND(bugprone-macro-parentheses) + +} // namespace ntt diff --git a/src/framework/domain/io/fields.cpp b/src/framework/domain/io/fields.cpp new file mode 100644 index 000000000..5582586d5 --- /dev/null +++ b/src/framework/domain/io/fields.cpp @@ -0,0 +1,586 @@ +#include "enums.h" +#include "global.h" + +#include "arch/kokkos_aliases.h" +#include "traits/metric.h" +#include "utils/error.h" +#include "utils/numeric.h" + +#include "framework/containers/particles.h" +#include "framework/domain/domain.h" +#include "framework/domain/mesh.h" +#include "framework/domain/metadomain.h" +#include "framework/parameters/parameters.h" +#include "framework/specialization_registry.h" +#include "kernels/divergences.hpp" +#include "kernels/fields_to_phys.hpp" +#include "kernels/particle_moments.hpp" + +#include +#include +#include + +#if defined(MPI_ENABLED) + #include +#endif // MPI_ENABLED + +#include +#include +#include +#include + +namespace ntt { + + template + void ComputeMoments(const SimulationParams& params, + const Mesh& mesh, + const std::vector>& prtl_species, + const std::vector& species, + const std::vector& components, + ndfield_t& buffer, + idx_t buff_idx) { + std::vector specs = species; + if (specs.empty()) { + // if no species specified, take all massive species + for (auto& sp : prtl_species) { + if (sp.mass() > 0) { + specs.push_back(sp.index()); + } + } + } + for (const auto& sp : specs) { + raise::ErrorIf((sp > prtl_species.size()) or (sp == 0), + "Invalid species index " + std::to_string(sp), + HERE); + } + auto scatter_buff = Kokkos::Experimental::create_scatter_view(buffer); + + // some parameters + const auto use_weights = params.template get("particles.use_weights"); + const auto ni2 = mesh.n_active(in::x2); + const auto inv_n0 = ONE / params.template get("scales.n0"); + const auto smooth_order = params.template get( + "output.fields.smoothing.order"); + const auto smooth_method = OutputSmoothingType::from_string( + params.template get("output.fields.smoothing.method")); + + for (const auto& sp : specs) { + auto& prtl_spec = prtl_species[sp - 1]; + Kokkos::parallel_for( + "ComputeMoments", + prtl_spec.rangeActiveParticles(), + kernel::ParticleMoments_kernel(components, + scatter_buff, + buff_idx, + prtl_spec, + use_weights, + mesh.metric, + mesh.flds_bc(), + ni2, + inv_n0, + smooth_order, + smooth_method)); + } + Kokkos::Experimental::contribute(buffer, scatter_buff); + } + + template + void ComputeVectorPotential(ndfield_t& buffer, + ndfield_t& EM, + unsigned short buff_idx, + const Mesh& mesh) { + if constexpr (M::Dim == Dim::_2D) { + const auto metric = mesh.metric; + Kokkos::parallel_for( + "ComputeVectorPotential", + mesh.rangeActiveCells(), + Lambda(cellidx_t i1, cellidx_t i2) { + const real_t i1_ { COORD(i1) }; + const ncells_t k_min = 0; + const ncells_t k_max = (i2 - (N_GHOSTS)); + real_t A3 = ZERO; + for (auto k { k_min }; k <= k_max; ++k) { + const real_t k_ = static_cast(k); + const real_t sqrt_detH_ij1 { metric.sqrt_det_h({ i1_, k_ - HALF }) }; + const real_t sqrt_detH_ij2 { metric.sqrt_det_h({ i1_, k_ + HALF }) }; + const auto k1 { k + N_GHOSTS }; + A3 += HALF * (sqrt_detH_ij1 * EM(i1, k1 - 1, em::bx1) + + sqrt_detH_ij2 * EM(i1, k1, em::bx1)); + } + buffer(i1, i2, buff_idx) = A3; + }); + + // @TODO: Implementation with team policies works on AMD, but not on NVIDIA GPUs + // + // using TeamPolicy = Kokkos::TeamPolicy; + // const auto nx1 = mesh.n_active(in::x1); + // const auto nx2 = mesh.n_active(in::x2); + // + // TeamPolicy policy(nx1, Kokkos::AUTO); + // + // Kokkos::parallel_for( + // "ComputeVectorPotential", + // policy, + // Lambda(const TeamPolicy::member_type& team_member) { + // cellidx_t i1 = team_member.league_rank(); + // Kokkos::parallel_scan( + // Kokkos::TeamThreadRange(team_member, nx2), + // [=](cellidx_t i2, real_t& update, const bool final_pass) { + // const auto i1_ { static_cast(i1) }; + // const auto i2_ { static_cast(i2) }; + // const real_t sqrt_detH_ijM { metric.sqrt_det_h({ i1_, i2_ - HALF }) }; + // const real_t sqrt_detH_ijP { metric.sqrt_det_h({ i1_, i2_ + HALF }) }; + // const auto input_val = + // HALF * + // (sqrt_detH_ijM * EM(i1 + N_GHOSTS, i2 + N_GHOSTS - 1, em::bx1) + + // sqrt_detH_ijP * EM(i1 + N_GHOSTS, i2 + N_GHOSTS, em::bx1)); + // if (final_pass) { + // buffer(i1 + N_GHOSTS, i2 + N_GHOSTS, buff_idx) = update; + // } + // update += input_val; + // }); + // }); + } else { + raise::KernelError( + HERE, + "ComputeVectorPotential: 2D implementation called for D != 2"); + } + } + + template + void DeepCopyFields(ndfield_t& fld_from, + ndfield_t& fld_to, + const cell_range_t& from, + const cell_range_t& to) { + for (auto d { 0u }; d < D; ++d) { + raise::ErrorIf(fld_from.extent(d) != fld_to.extent(d), + "Fields have different sizes " + + std::to_string(fld_from.extent(d)) + + " != " + std::to_string(fld_to.extent(d)), + HERE); + } + if constexpr (D == Dim::_1D) { + Kokkos::deep_copy(Kokkos::subview(fld_to, Kokkos::ALL, to), + Kokkos::subview(fld_from, Kokkos::ALL, from)); + } else if constexpr (D == Dim::_2D) { + Kokkos::deep_copy(Kokkos::subview(fld_to, Kokkos::ALL, Kokkos::ALL, to), + Kokkos::subview(fld_from, Kokkos::ALL, Kokkos::ALL, from)); + } else if constexpr (D == Dim::_3D) { + Kokkos::deep_copy( + Kokkos::subview(fld_to, Kokkos::ALL, Kokkos::ALL, Kokkos::ALL, to), + Kokkos::subview(fld_from, Kokkos::ALL, Kokkos::ALL, Kokkos::ALL, from)); + } + } + + template + void Metadomain::WriteFields(const SimulationParams& params, + Domain* local_domain, + timestep_t current_step, + timestep_t finished_step, + simtime_t current_time, + simtime_t finished_time, + const custom_field_output_t& CustomFieldOutput) { + g_writer.beginWriting(WriteMode::Fields, current_step, current_time); + const auto incl_ghosts = params.template get("output.debug.ghosts"); + const auto dwn = params.template get>( + "output.fields.downsampling"); + + auto off_ncells_with_ghosts = local_domain->offset_ncells(); + auto loc_shape_with_ghosts = local_domain->mesh.n_active(); + { // compute positions/sizes of meshblocks in cells in all dimensions + const auto off_ndomains = local_domain->offset_ndomains(); + if (incl_ghosts) { + for (auto d { 0 }; d <= M::Dim; ++d) { + off_ncells_with_ghosts[d] += 2 * N_GHOSTS * off_ndomains[d]; + loc_shape_with_ghosts[d] += 2 * N_GHOSTS; + } + } + } + // Refresh the writer's cached per-rank slab so that field/mesh writes + // pick up the (possibly rebalanced) current local layout. + g_writer.setLocalLayout(off_ncells_with_ghosts, loc_shape_with_ghosts); + for (auto dim { 0u }; dim < M::Dim; ++dim) { + const auto l_size = local_domain->mesh.n_active()[dim]; + const auto l_offset = local_domain->offset_ncells()[dim]; + const auto g_size = mesh().n_active()[dim]; + + const auto dwn_in_dim = dwn[dim]; + + const double n = l_size; + const double d = dwn_in_dim; + const double l = l_offset; + const double f = math::ceil(l / d) * d - l; + + const auto first_cell = static_cast(f); + const auto l_size_dwn = static_cast(math::ceil((n - f) / d)); + + const auto is_last = l_offset + l_size == g_size; + + const auto add_ghost = (incl_ghosts ? 2 * N_GHOSTS : 0); + const auto add_last = (is_last ? 1 : 0); + + const array_t xc { "Xc", l_size_dwn + add_ghost }; + const array_t xe { "Xe", l_size_dwn + add_ghost + add_last }; + + const auto offset = (incl_ghosts ? N_GHOSTS : 0); + const auto ncells = l_size_dwn; + + const auto& metric = local_domain->mesh.metric; + + Kokkos::parallel_for( + "GenerateMesh", + ncells, + Lambda(cellidx_t i_dwn) { + const auto i = first_cell + i_dwn * dwn_in_dim; + const auto i_ = static_cast(i); + coord_t x_Cd { ZERO }, x_Ph { ZERO }; + x_Cd[dim] = i_ + HALF; + // TODO : change to convert by component + metric.template convert(x_Cd, x_Ph); + xc(offset + i_dwn) = x_Ph[dim]; + x_Cd[dim] = i_; + metric.template convert(x_Cd, x_Ph); + xe(offset + i_dwn) = x_Ph[dim]; + if (is_last && i_dwn == ncells - 1) { + x_Cd[dim] = i_ + ONE; + metric.template convert(x_Cd, x_Ph); + xe(offset + i_dwn + 1) = x_Ph[dim]; + } + }); + g_writer.writeMesh( + dim, + xc, + xe, + { off_ncells_with_ghosts[dim], loc_shape_with_ghosts[dim] }); + } + const auto output_asis = params.template get("output.debug.as_is"); + // !TODO: this can probably be optimized to dump things at once + for (auto& fld : g_writer.fieldWriters()) { + Kokkos::deep_copy(local_domain->fields.bckp, ZERO); + std::vector names; + std::vector addresses; + if (fld.comp.empty() || fld.comp.size() == 1) { // scalar + names.push_back(fld.name()); + addresses.push_back(0); + if (fld.is_moment()) { + // output a particle distribution moment (single component) + // this includes T, Rho, Charge, N, Nppc + const auto c = static_cast(addresses.back()); + if (fld.id() == FldsID::T) { + raise::ErrorIf(fld.comp.size() != 1, + "Wrong # of components requested for T output", + HERE); + ComputeMoments(params, + local_domain->mesh, + local_domain->species, + fld.species, + fld.comp[0], + local_domain->fields.bckp, + c); + } else if (fld.id() == FldsID::Rho) { + ComputeMoments(params, + local_domain->mesh, + local_domain->species, + fld.species, + {}, + local_domain->fields.bckp, + c); + } else if (fld.id() == FldsID::Charge) { + ComputeMoments(params, + local_domain->mesh, + local_domain->species, + fld.species, + {}, + local_domain->fields.bckp, + c); + } else if (fld.id() == FldsID::N) { + ComputeMoments(params, + local_domain->mesh, + local_domain->species, + fld.species, + {}, + local_domain->fields.bckp, + c); + } else if (fld.id() == FldsID::Nppc) { + ComputeMoments(params, + local_domain->mesh, + local_domain->species, + fld.species, + {}, + local_domain->fields.bckp, + c); + } else { + raise::Error("Wrong moment requested for output", HERE); + } + } else if (fld.is_divergence()) { + // @TODO: is this correct for GR too? not em0? + const auto c = static_cast(addresses.back()); + Kokkos::parallel_for( + "ComputeDivergence", + local_domain->mesh.rangeActiveCells(), + kernel::ComputeDivergence_kernel(local_domain->mesh.metric, + local_domain->fields.em, + local_domain->fields.bckp, + c)); + } else if (fld.is_custom()) { + if (CustomFieldOutput) { + CustomFieldOutput(fld.name().substr(1), + local_domain->fields.bckp, + addresses.back(), + finished_step, + finished_time, + *local_domain); + } else { + raise::Error("Custom output requested but no function provided", HERE); + } + } else if (fld.is_vpotential()) { + if constexpr (S == SimEngine::GRPIC && M::Dim == Dim::_2D) { + const auto c = static_cast(addresses.back()); + ComputeVectorPotential(local_domain->fields.bckp, + local_domain->fields.em, + c, + local_domain->mesh); +#if defined(MPI_ENABLED) + CommunicateVectorPotential(c); +#endif + } else { + raise::Error( + "Vector potential can only be computed for GRPIC in 2D", + HERE); + } + } else { + raise::Error("Wrong # of components requested for " + "non-moment/non-custom output", + HERE); + } + SynchronizeFields(*local_domain, + Comm::Bckp, + { addresses.back(), addresses.back() + 1 }); + } else if (fld.comp.size() == 3) { // vector + for (auto i = 0; i < 3; ++i) { + names.push_back(fld.name(i)); + addresses.push_back(i + 3); + } + if (fld.is_moment()) { + for (auto i = 0; i < 3; ++i) { + const auto c = static_cast(addresses[i]); + if (fld.id() == FldsID::T) { + raise::ErrorIf(fld.comp[i].size() != 2, + "Wrong # of components requested for moment", + HERE); + ComputeMoments(params, + local_domain->mesh, + local_domain->species, + fld.species, + fld.comp[i], + local_domain->fields.bckp, + c); + } else if (fld.id() == FldsID::V) { + raise::ErrorIf(fld.comp[i].size() != 1, + "Wrong # of components requested for 3vel", + HERE); + ComputeMoments(params, + local_domain->mesh, + local_domain->species, + fld.species, + fld.comp[i], + local_domain->fields.bckp, + c); + } else { + raise::Error("Wrong moment requested for output", HERE); + } + } + raise::ErrorIf(addresses[1] - addresses[0] != addresses[2] - addresses[1], + "Indices for the backup are not contiguous", + HERE); + SynchronizeFields(*local_domain, + Comm::Bckp, + { addresses[0], addresses[2] + 1 }); + if constexpr (S == SimEngine::SRPIC) { + if (fld.id() == FldsID::V) { + // normalize 3vel * rho (combuted above) by rho + ComputeMoments(params, + local_domain->mesh, + local_domain->species, + fld.species, + {}, + local_domain->fields.bckp, + 0u); + SynchronizeFields(*local_domain, Comm::Bckp, { 0, 1 }); + Kokkos::parallel_for("NormalizeVectorByRho", + local_domain->mesh.rangeActiveCells(), + kernel::NormalizeVectorByRho_kernel( + local_domain->fields.bckp, + local_domain->fields.bckp, + 0, + addresses[0], + addresses[1], + addresses[2])); + } + } + } else { + // copy fields to bckp (:, 0, 1, 2) + // if as-is specified ==> copy directly to 3, 4, 5 + cell_range_t copy_to = { 0, 3 }; + if (output_asis) { + copy_to = { 3, 6 }; + } + if (fld.is_current()) { + DeepCopyFields(local_domain->fields.cur, + local_domain->fields.bckp, + { cur::jx1, cur::jx3 + 1 }, + copy_to); + } else if (fld.is_field()) { + if (S == SimEngine::GRPIC && fld.is_gr_aux_field()) { + if (fld.is_efield()) { + // GR: E + DeepCopyFields(local_domain->fields.aux, + local_domain->fields.bckp, + { em::ex1, em::ex3 + 1 }, + copy_to); + } else { + // GR: H + DeepCopyFields(local_domain->fields.aux, + local_domain->fields.bckp, + { em::hx1, em::hx3 + 1 }, + copy_to); + } + } else { + if (fld.is_efield()) { + // GR/SR: D/E + DeepCopyFields(local_domain->fields.em, + local_domain->fields.bckp, + { em::ex1, em::ex3 + 1 }, + copy_to); + } else { + // GR/SR: B + DeepCopyFields(local_domain->fields.em, + local_domain->fields.bckp, + { em::bx1, em::bx3 + 1 }, + copy_to); + } + } + } else { + raise::Error("Wrong field requested for output", HERE); + } + if (not output_asis) { + // copy fields from bckp(:, 0, 1, 2) -> bckp(:, 3, 4, 5) + // converting to proper basis and properly interpolating + list_t comp_from = { 0, 1, 2 }; + list_t comp_to = { 3, 4, 5 }; + DeepCopyFields(local_domain->fields.bckp, + local_domain->fields.bckp, + { 0, 3 }, + { 3, 6 }); + Kokkos::parallel_for("FieldsToPhys", + local_domain->mesh.rangeActiveCells(), + kernel::FieldsToPhys_kernel( + local_domain->fields.bckp, + local_domain->fields.bckp, + comp_from, + comp_to, + fld.interp_flag | fld.prepare_flag, + local_domain->mesh.metric)); + } + } + } else if (fld.comp.size() == 4) { // 4-vector + if constexpr (S == SimEngine::GRPIC) { + if (fld.is_moment() && fld.id() == FldsID::V) { + // Compute 4-velocity: V^μ (u^0, u^1, u^2, u^3) + for (auto i = 0; i < 4; ++i) { + names.push_back(fld.name(i)); + addresses.push_back(i); + const auto c = static_cast(addresses[i]); + raise::ErrorIf(fld.comp[i].size() != 1, + "Wrong # of components requested for 4-velocity", + HERE); + ComputeMoments(params, + local_domain->mesh, + local_domain->species, + fld.species, + fld.comp[i], + local_domain->fields.bckp, + c); + } + // Synchronize all 4 components + SynchronizeFields(*local_domain, + Comm::Bckp, + { addresses[0], addresses[3] + 1 }); + // Normalize 4-momentum flux: V^μ = N^μ / sqrt(-N_ν N^ν) + // (computed in coordinate contravariant basis) + Kokkos::parallel_for( + "Normalize4VelocityByNorm", + local_domain->mesh.rangeActiveCells(), + kernel::Normalize4VelocityByNorm_kernel( + local_domain->fields.bckp, + local_domain->fields.bckp, + addresses[0], // c_u0 (column for u^0) + addresses[1], // c_u1 (column for u^1) + addresses[2], // c_u2 (column for u^2) + addresses[3], // c_u3 (column for u^3) + local_domain->mesh.metric)); + // Transform spatial components to physical basis for output + // u^0 (Gamma/alpha) remains unitless, only u^i transform + Kokkos::parallel_for( + "Transform4VelocitySpatialToPhysical", + local_domain->mesh.rangeActiveCells(), + kernel::Transform4VelocitySpatialToPhysical_kernel( + local_domain->fields.bckp, + addresses[1], // c_u1 (column for u^1) + addresses[2], // c_u2 (column for u^2) + addresses[3], // c_u3 (column for u^3) + local_domain->mesh.metric)); + } else { + raise::Error("4-vector output only supported for V (bulk " + "velocity) moment in GRPIC", + HERE); + } + } else { + raise::Error("4-vector output only supported for GRPIC", HERE); + } + } else if (fld.comp.size() == 6) { // tensor + raise::ErrorIf(not fld.is_moment() or fld.id() != FldsID::T, + "Only T tensor has 6 components", + HERE); + for (auto i = 0; i < 6; ++i) { + names.push_back(fld.name(i)); + addresses.push_back(i); + const auto c = static_cast(addresses.back()); + raise::ErrorIf(fld.comp[i].size() != 2, + "Wrong # of components requested for moment", + HERE); + ComputeMoments(params, + local_domain->mesh, + local_domain->species, + fld.species, + fld.comp[i], + local_domain->fields.bckp, + c); + } + SynchronizeFields(*local_domain, + Comm::Bckp, + { addresses[0], addresses[5] + 1 }); + } else { + raise::Error("Wrong # of components requested for output", HERE); + } + g_writer.writeField(names, local_domain->fields.bckp, addresses); + } + g_writer.endWriting(WriteMode::Fields); + } + + // NOLINTBEGIN(bugprone-macro-parentheses) +#define METADOMAIN_OUTPUT(S, M, D) \ + template void Metadomain>::WriteFields(const SimulationParams&, \ + Domain>*, \ + timestep_t, \ + timestep_t, \ + simtime_t, \ + simtime_t, \ + const custom_field_output_t&); + + NTT_FOREACH_SPECIALIZATION(METADOMAIN_OUTPUT) + +#undef METADOMAIN_OUTPUT + // NOLINTEND(bugprone-macro-parentheses) + +} // namespace ntt diff --git a/src/framework/domain/io/init.cpp b/src/framework/domain/io/init.cpp new file mode 100644 index 000000000..3a965c4f9 --- /dev/null +++ b/src/framework/domain/io/init.cpp @@ -0,0 +1,117 @@ +#include "defaults.h" +#include "enums.h" +#include "global.h" + +#include "traits/metric.h" +#include "utils/error.h" + +#include "framework/domain/domain.h" +#include "framework/domain/mesh.h" +#include "framework/domain/metadomain.h" +#include "framework/parameters/parameters.h" +#include "framework/specialization_registry.h" + +#include +#include +#include + +#include +#include +#include +#include +#include + +namespace ntt { + + template + void Metadomain::InitWriter(adios2::ADIOS* ptr_adios, + const SimulationParams& params) { + raise::ErrorIf( + l_subdomain_indices().size() != 1, + "Output for now is only supported for one subdomain per rank", + HERE); + auto local_domain = subdomain_ptr(l_subdomain_indices()[0]); + raise::ErrorIf(local_domain->is_placeholder(), + "local_domain is a placeholder", + HERE); + const auto incl_ghosts = params.template get("output.debug.ghosts"); + + auto glob_shape_with_ghosts = mesh().n_active(); + auto off_ncells_with_ghosts = local_domain->offset_ncells(); + auto off_ndomains = local_domain->offset_ndomains(); + auto loc_shape_with_ghosts = local_domain->mesh.n_active(); + if (incl_ghosts) { + for (auto d { 0 }; d <= M::Dim; ++d) { + glob_shape_with_ghosts[d] += 2 * N_GHOSTS * ndomains_per_dim()[d]; + off_ncells_with_ghosts[d] += 2 * N_GHOSTS * off_ndomains[d]; + loc_shape_with_ghosts[d] += 2 * N_GHOSTS; + } + } + + g_writer.init( + ptr_adios, + params.template get("output.format"), + params.template get("simulation.name"), + { params.template get("adios2.aggregators_per_node", + defaults::adios2::aggregators_per_node), + params.template get("adios2.max_shm_size", + defaults::adios2::max_shm_size), + params.template get("adios2.buffer_chunk_size", + defaults::adios2::buffer_chunk_size) }); + g_writer.defineMeshLayout(glob_shape_with_ghosts, + off_ncells_with_ghosts, + loc_shape_with_ghosts, + { local_domain->index(), ndomains() }, + params.template get>( + "output.fields.downsampling"), + incl_ghosts, + M::CoordType); + const auto fields_to_write = params.template get>( + "output.fields.quantities"); + const auto custom_fields_to_write = params.template get>( + "output.fields.custom"); + std::vector all_fields_to_write; + std::merge(fields_to_write.begin(), + fields_to_write.end(), + custom_fields_to_write.begin(), + custom_fields_to_write.end(), + std::back_inserter(all_fields_to_write)); + const auto species_to_write = params.template get>( + "output.particles.species"); + g_writer.defineFieldOutputs(S, all_fields_to_write); + + g_writer.clearSpeciesIndex(); + for (const auto& s : species_to_write) { + g_writer.addSpeciesIndex(s); + } + for (const auto sp : g_writer.speciesIndices()) { + local_domain->species[sp - 1].OutputDeclare(g_writer.io()); + } + + // spectra write all particle species + std::vector spectra_species {}; + for (const auto& sp : species_params()) { + spectra_species.push_back(sp.index()); + } + g_writer.defineSpectraOutputs(spectra_species); + for (const auto& type : { "fields", "particles", "spectra" }) { + g_writer.addTracker(type, + params.template get( + "output." + std::string(type) + ".interval"), + params.template get( + "output." + std::string(type) + ".interval_time")); + } + g_writer.writeAttrs(params); + } + + // NOLINTBEGIN(bugprone-macro-parentheses) +#define METADOMAIN_OUTPUT(S, M, D) \ + template void Metadomain>::InitWriter(adios2::ADIOS*, \ + const SimulationParams&); + + NTT_FOREACH_SPECIALIZATION(METADOMAIN_OUTPUT) + +#undef METADOMAIN_OUTPUT + // NOLINTEND(bugprone-macro-parentheses) + +} // namespace ntt diff --git a/src/framework/domain/io/spectra.cpp b/src/framework/domain/io/spectra.cpp new file mode 100644 index 000000000..d063b9d3d --- /dev/null +++ b/src/framework/domain/io/spectra.cpp @@ -0,0 +1,92 @@ +#include "enums.h" +#include "global.h" + +#include "arch/kokkos_aliases.h" +#include "traits/metric.h" + +#include "framework/containers/particles.h" +#include "framework/domain/domain.h" +#include "framework/domain/mesh.h" +#include "framework/domain/metadomain.h" +#include "framework/parameters/parameters.h" +#include "framework/specialization_registry.h" +#include "kernels/particle_moments.hpp" + +#include +#include +#include + +#if defined(MPI_ENABLED) + #include +#endif // MPI_ENABLED + +#include +#include +#include +#include + +namespace ntt { + + template + void Metadomain::WriteSpectra(const SimulationParams& params, + Domain* local_domain, + timestep_t current_step, + simtime_t current_time) { + + g_writer.beginWriting(WriteMode::Spectra, current_step, current_time); + const auto log_bins = params.template get("output.spectra.log_bins"); + const auto n_bins = params.template get("output.spectra.n_bins"); + auto e_min = params.template get("output.spectra.e_min"); + auto e_max = params.template get("output.spectra.e_max"); + if (log_bins) { + e_min = math::log10(e_min); + e_max = math::log10(e_max); + } + const array_t energy { "energy", n_bins + 1 }; + Kokkos::parallel_for( + "GenerateEnergyBins", + n_bins + 1, + Lambda(uint32_t e) { + if (log_bins) { + energy(e) = math::pow(static_cast(10), + e_min + (e_max - e_min) * static_cast(e) / + static_cast(n_bins)); + } else { + energy(e) = e_min + (e_max - e_min) * static_cast(e) / + static_cast(n_bins); + } + }); + for (const auto& spec : g_writer.spectraWriters()) { + auto& species = local_domain->species[spec.species() - 1]; + array_t dn { "dn", n_bins }; + auto dn_scatter = Kokkos::Experimental::create_scatter_view(dn); + Kokkos::parallel_for( + "ComputeSpectra", + species.rangeActiveParticles(), + kernel::ParticleDistribution_kernel { species, + dn_scatter, + e_min, + e_max, + log_bins, + n_bins, + local_domain->mesh.metric }); + Kokkos::Experimental::contribute(dn, dn_scatter); + g_writer.writeSpectrum(dn, spec.name()); + } + g_writer.writeSpectrumBins(energy, "sEbn"); + g_writer.endWriting(WriteMode::Spectra); + } + + // NOLINTBEGIN(bugprone-macro-parentheses) +#define METADOMAIN_OUTPUT(S, M, D) \ + template void Metadomain>::WriteSpectra(const SimulationParams&, \ + Domain>*, \ + timestep_t, \ + simtime_t); + + NTT_FOREACH_SPECIALIZATION(METADOMAIN_OUTPUT) + +#undef METADOMAIN_OUTPUT + // NOLINTEND(bugprone-macro-parentheses) + +} // namespace ntt diff --git a/src/framework/domain/io/write.cpp b/src/framework/domain/io/write.cpp new file mode 100644 index 000000000..0c4deb13a --- /dev/null +++ b/src/framework/domain/io/write.cpp @@ -0,0 +1,111 @@ +#include "enums.h" +#include "global.h" + +#include "traits/metric.h" +#include "utils/error.h" +#include "utils/log.h" + +#include "framework/domain/domain.h" +#include "framework/domain/mesh.h" +#include "framework/domain/metadomain.h" +#include "framework/parameters/parameters.h" +#include "framework/specialization_registry.h" + +#include +#include +#include + +#if defined(MPI_ENABLED) + #include +#endif // MPI_ENABLED + +#include +#include + +namespace ntt { + + template + auto Metadomain::Write(const SimulationParams& params, + timestep_t current_step, + timestep_t finished_step, + simtime_t current_time, + simtime_t finished_time, + const custom_field_output_t& CustomFieldOutput) + -> bool { + raise::ErrorIf( + l_subdomain_indices().size() != 1, + "Output for now is only supported for one subdomain per rank", + HERE); + const auto write_fields = params.template get( + "output.fields.enable") and + g_writer.shouldWrite("fields", + finished_step, + finished_time); + const auto write_particles = params.template get( + "output.particles.enable") and + g_writer.shouldWrite("particles", + finished_step, + finished_time); + const auto write_spectra = params.template get( + "output.spectra.enable") and + g_writer.shouldWrite("spectra", + finished_step, + finished_time); + const auto extension = params.template get("output.format"); + if (not(write_fields or write_particles or write_spectra) and + extension != "disabled") { + return false; + } + auto local_domain = subdomain_ptr(l_subdomain_indices()[0]); + raise::ErrorIf(local_domain->is_placeholder(), + "local_domain is a placeholder", + HERE); + logger::Checkpoint("Writing output", HERE); + if (write_fields) { + WriteFields(params, + local_domain, + current_step, + finished_step, + current_time, + finished_time, + CustomFieldOutput); + } // end shouldWrite("fields", step, time) + + if (write_particles) { + g_writer.beginWriting(WriteMode::Particles, current_step, current_time); + const auto prtl_stride = params.template get( + "output.particles.stride"); + for (const auto spec : g_writer.speciesIndices()) { + local_domain->species[spec - 1].template OutputWrite( + g_writer.io(), + g_writer.writer(), + prtl_stride, + ndomains(), + local_domain->index(), + local_domain->mesh.metric); + } + g_writer.endWriting(WriteMode::Particles); + } // end shouldWrite("particles", step, time) + + if (write_spectra) { + WriteSpectra(params, local_domain, current_step, current_time); + } // end shouldWrite("spectra", step, time) + + return true; + } + + // NOLINTBEGIN(bugprone-macro-parentheses) +#define METADOMAIN_OUTPUT(S, M, D) \ + template auto Metadomain>::Write(const SimulationParams&, \ + timestep_t, \ + timestep_t, \ + simtime_t, \ + simtime_t, \ + const custom_field_output_t&) -> bool; + + NTT_FOREACH_SPECIALIZATION(METADOMAIN_OUTPUT) + +#undef METADOMAIN_OUTPUT + // NOLINTEND(bugprone-macro-parentheses) + +} // namespace ntt diff --git a/src/framework/domain/metadomain.h b/src/framework/domain/metadomain.h index ede688cfb..0330b5590 100644 --- a/src/framework/domain/metadomain.h +++ b/src/framework/domain/metadomain.h @@ -6,11 +6,19 @@ * - ntt::Metadomain<> * @cpp: * - metadomain.cpp - * - metadomain_comm.cpp - * - metadomain_chckpt.cpp - * - metadomain_io.cpp * - metadomain_stats.cpp * - metadomain_reshape.cpp + * - checkpoint/init.cpp + * - checkpoint/write.cpp + * - checkpoint/resume.cpp + * - comm/fields.cpp + * - comm/fields_sync.cpp + * - comm/particles.cpp + * - comm/vector_potential.cpp + * - io/init.cpp + * - io/write.cpp + * - io/fields.cpp + * - io/spectra.cpp * @namespaces: * - ntt:: * @macros: @@ -148,18 +156,27 @@ namespace ntt { /* output-related ------------------------------------------------------- */ #if defined(OUTPUT_ENABLED) + using custom_field_output_t = std::function&, + uint32_t, + timestep_t, + simtime_t, + const Domain&)>; void InitWriter(adios2::ADIOS*, const SimulationParams&); auto Write(const SimulationParams&, timestep_t, timestep_t, simtime_t, simtime_t, - const std::function&, - uint32_t, - timestep_t, - simtime_t, - const Domain&)>& = nullptr) -> bool; + const custom_field_output_t& = nullptr) -> bool; + void WriteFields(const SimulationParams&, + Domain*, + timestep_t, + timestep_t, + simtime_t, + simtime_t, + const custom_field_output_t&); + void WriteSpectra(const SimulationParams&, Domain*, timestep_t, simtime_t); void InitCheckpointWriter(adios2::ADIOS*, const SimulationParams&); auto WriteCheckpoint(const SimulationParams&, timestep_t, diff --git a/src/framework/domain/metadomain_comm.cpp b/src/framework/domain/metadomain_comm.cpp deleted file mode 100644 index f8c0f60d7..000000000 --- a/src/framework/domain/metadomain_comm.cpp +++ /dev/null @@ -1,668 +0,0 @@ -#include "enums.h" -#include "global.h" - -#include "arch/directions.h" -#include "arch/kokkos_aliases.h" -#include "traits/metric.h" -#include "utils/error.h" -#include "utils/formatting.h" -#include "utils/log.h" -#include "utils/numeric.h" - -#include "framework/domain/domain.h" -#include "framework/domain/metadomain.h" -#include "framework/specialization_registry.h" - -#if defined(MPI_ENABLED) - #include "arch/mpi_tags.h" - - #include "framework/domain/comm_mpi.hpp" -#else - #include "framework/domain/comm_nompi.hpp" -#endif - -#include - -#include -#include -#include - -namespace ntt { - - using address_t = std::pair; - using comm_params_t = std::pair>; - - template - auto GetSendRecvRanks(const Metadomain* const metadomain, - Domain& domain, - const dir::direction_t& direction) - -> std::pair { - const Domain* send_to_nghbr_ptr = nullptr; - const Domain* recv_from_nghbr_ptr = nullptr; - // set pointers to the correct send/recv domains - // can coincide with the current domain if periodic - if (domain.mesh.flds_bc_in(direction) == FldsBC::PERIODIC) { - // sending / receiving from itself - raise::ErrorIf( - domain.neighbor_idx_in(direction) != domain.index(), - fmt::format( - "Periodic boundaries in `%s` imply communication within the " - "same domain, but %u != %u", - direction.to_string().c_str(), - domain.neighbor_idx_in(direction), - domain.index()), - HERE); - raise::ErrorIf( - domain.mesh.flds_bc_in(-direction) != FldsBC::PERIODIC, - "Periodic boundary conditions must be set in both directions", - HERE); - send_to_nghbr_ptr = &domain; - recv_from_nghbr_ptr = &domain; - } else if (domain.mesh.flds_bc_in(direction) == FldsBC::SYNC) { - // sending to other domain - raise::ErrorIf( - domain.neighbor_idx_in(direction) == domain.index(), - "Sync boundaries imply communication between separate domains", - HERE); - send_to_nghbr_ptr = metadomain->subdomain_ptr( - domain.neighbor_idx_in(direction)); - if (domain.mesh.flds_bc_in(-direction) == FldsBC::SYNC) { - // receiving from other domain - raise::ErrorIf( - domain.neighbor_idx_in(-direction) == domain.index(), - "Sync boundaries imply communication between separate domains", - HERE); - recv_from_nghbr_ptr = metadomain->subdomain_ptr( - domain.neighbor_idx_in(-direction)); - } - } else if (domain.mesh.flds_bc_in(-direction) == FldsBC::SYNC) { - // only receiving from other domain - raise::ErrorIf( - domain.neighbor_idx_in(-direction) == domain.index(), - "Sync boundaries imply communication between separate domains", - HERE); - recv_from_nghbr_ptr = metadomain->subdomain_ptr( - domain.neighbor_idx_in(-direction)); - } else { - // no communication necessary - return { - { 0, -1 }, - { 0, -1 } - }; - } -#if defined(MPI_ENABLED) - const auto send_rank = (send_to_nghbr_ptr != nullptr) - ? send_to_nghbr_ptr->mpi_rank() - : -1; - const auto recv_rank = (recv_from_nghbr_ptr != nullptr) - ? recv_from_nghbr_ptr->mpi_rank() - : -1; -#else - const auto send_rank = (send_to_nghbr_ptr != nullptr) ? 0 : -1; - const auto recv_rank = (recv_from_nghbr_ptr != nullptr) ? 0 : -1; -#endif - const auto send_ind = (send_to_nghbr_ptr != nullptr) - ? send_to_nghbr_ptr->index() - : 0; - const auto recv_ind = (recv_from_nghbr_ptr != nullptr) - ? recv_from_nghbr_ptr->index() - : 0; - (void)send_rank; - (void)recv_rank; - return { - { send_ind, send_rank }, - { recv_ind, recv_rank } - }; - } - - template - auto GetSendRecvParams(const Metadomain* const metadomain, - Domain& domain, - dir::direction_t direction, - bool synchronize) - -> std::pair { - const auto [send_indrank, - recv_indrank] = GetSendRecvRanks(metadomain, domain, direction); - const auto [send_ind, send_rank] = send_indrank; - const auto [recv_ind, recv_rank] = recv_indrank; - const auto is_sending = (send_rank >= 0); - const auto is_receiving = (recv_rank >= 0); - if (not(is_sending or is_receiving)) { - return { - { { 0, -1 }, {} }, - { { 0, -1 }, {} } - }; - } - auto send_slice = std::vector {}; - auto recv_slice = std::vector {}; - const in components[] = { in::x1, in::x2, in::x3 }; - // find the field components and indices to be sent/received - for (auto d { 0u }; d < direction.size(); ++d) { - const auto c = components[d]; - const auto dir = direction[d]; - if (not synchronize) { - // recv to: ghost zones - // send from: active zones - if (is_sending) { - if (dir == 0) { - send_slice.emplace_back(domain.mesh.i_min(c), domain.mesh.i_max(c)); - } else if (dir == 1) { - send_slice.emplace_back(domain.mesh.i_max(c) - N_GHOSTS, - domain.mesh.i_max(c)); - } else { - send_slice.emplace_back(domain.mesh.i_min(c), - domain.mesh.i_min(c) + N_GHOSTS); - } - } - if (is_receiving) { - if (-dir == 0) { - recv_slice.emplace_back(domain.mesh.i_min(c), domain.mesh.i_max(c)); - } else if (-dir == 1) { - recv_slice.emplace_back(domain.mesh.i_max(c), - domain.mesh.i_max(c) + N_GHOSTS); - } else { - recv_slice.emplace_back(domain.mesh.i_min(c) - N_GHOSTS, - domain.mesh.i_min(c)); - } - } - } else { - // recv to: active + ghost zones - // send from: active + ghost zones - if (is_sending) { - if (dir == 0) { - send_slice.emplace_back(domain.mesh.i_min(c) - N_GHOSTS, - domain.mesh.i_max(c) + N_GHOSTS); - } else if (dir == 1) { - send_slice.emplace_back(domain.mesh.i_max(c) - N_GHOSTS, - domain.mesh.i_max(c) + N_GHOSTS); - } else { - send_slice.emplace_back(domain.mesh.i_min(c) - N_GHOSTS, - domain.mesh.i_min(c) + N_GHOSTS); - } - } - if (is_receiving) { - if (-dir == 0) { - recv_slice.emplace_back(domain.mesh.i_min(c) - N_GHOSTS, - domain.mesh.i_max(c) + N_GHOSTS); - } else if (-dir == 1) { - recv_slice.emplace_back(domain.mesh.i_max(c) - N_GHOSTS, - domain.mesh.i_max(c) + N_GHOSTS); - } else { - recv_slice.emplace_back(domain.mesh.i_min(c) - N_GHOSTS, - domain.mesh.i_min(c) + N_GHOSTS); - } - } - } - } - - return { - { { send_ind, send_rank }, send_slice }, - { { recv_ind, recv_rank }, recv_slice }, - }; - } - - template - void Metadomain::CommunicateFields(Domain& domain, - CommTags tags) const { - const auto comm_em = ((S == SimEngine::SRPIC) and - ((tags & Comm::E) or (tags & Comm::B))) or - ((S == SimEngine::GRPIC) and - ((tags & Comm::D) or (tags & Comm::B))); - const bool comm_em0 = (S == SimEngine::GRPIC) and - ((tags & Comm::B0) or (tags & Comm::D0)); - const bool comm_j = (tags & Comm::J); - const bool comm_aux = (S == SimEngine::GRPIC) and - ((tags & Comm::E) or (tags & Comm::H)); - raise::ErrorIf(not(comm_em or comm_em0 or comm_j or comm_aux), - "CommunicateFields called with no task", - HERE); - - std::string comms; - if (tags & Comm::E) { - comms += "E "; - } - if (tags & Comm::B) { - comms += "B "; - } - if (tags & Comm::J) { - comms += "J "; - } - if (tags & Comm::D) { - comms += "D "; - } - if (tags & Comm::H) { - comms += "H "; - } - if (tags & Comm::D0) { - comms += "D0 "; - } - if (tags & Comm::B0) { - comms += "B0 "; - } - logger::Checkpoint(fmt::format("Communicating %s\n", comms.c_str()), HERE); - - /** - * @note this block is designed to support in the future multiple domains - * on a single rank, however that is not yet implemented - */ - // establish the last index ranges for fields (i.e., components) - auto comp_range_fld = cell_range_t {}; - auto comp_range_cur = cell_range_t {}; - if constexpr (S == SimEngine::GRPIC) { - if (((tags & Comm::D) and (tags & Comm::B)) or - ((tags & Comm::D0) and (tags & Comm::B0)) or - ((tags & Comm::E) and (tags & Comm::H))) { - comp_range_fld = cell_range_t(em::dx1, em::bx3 + 1); - } else if ((tags & Comm::D) or (tags & Comm::D0) or (tags & Comm::E)) { - comp_range_fld = cell_range_t(em::dx1, em::dx3 + 1); - } else if ((tags & Comm::B) or (tags & Comm::B0) or (tags & Comm::H)) { - comp_range_fld = cell_range_t(em::bx1, em::bx3 + 1); - } - } else if constexpr (S == SimEngine::SRPIC) { - if ((tags & Comm::E) and (tags & Comm::B)) { - comp_range_fld = cell_range_t(em::ex1, em::bx3 + 1); - } else if (tags & Comm::E) { - comp_range_fld = cell_range_t(em::ex1, em::ex3 + 1); - } else if (tags & Comm::B) { - comp_range_fld = cell_range_t(em::bx1, em::bx3 + 1); - } - } else { - raise::Error("Unknown simulation engine", HERE); - } - if (comm_j) { - comp_range_cur = cell_range_t(cur::jx1, cur::jx3 + 1); - } - // traverse in all directions and send/recv the fields - for (auto& direction : dir::Directions::all) { - const auto [send_params, - recv_params] = GetSendRecvParams(this, domain, direction, false); - const auto [send_indrank, send_slice] = send_params; - const auto [recv_indrank, recv_slice] = recv_params; - const auto [send_ind, send_rank] = send_indrank; - const auto [recv_ind, recv_rank] = recv_indrank; - if (send_rank < 0 and recv_rank < 0) { - continue; - } - if (comm_em) { - comm::CommunicateField(domain.index(), - domain.fields.em, - domain.fields.em, - send_ind, - recv_ind, - send_rank, - recv_rank, - send_slice, - recv_slice, - comp_range_fld, - false); - } - if constexpr (S == SimEngine::GRPIC) { - if (comm_aux) { - comm::CommunicateField(domain.index(), - domain.fields.aux, - domain.fields.aux, - send_ind, - recv_ind, - send_rank, - recv_rank, - send_slice, - recv_slice, - comp_range_fld, - false); - } - if (comm_em0) { - comm::CommunicateField(domain.index(), - domain.fields.em0, - domain.fields.em0, - send_ind, - recv_ind, - send_rank, - recv_rank, - send_slice, - recv_slice, - comp_range_fld, - false); - // @HACK_GR_1.2.0 -- this has to be done carefully - // comm::CommunicateField(domain.index(), - // domain.fields.aux, - // domain.fields.aux, - // send_ind, - // recv_ind, - // send_rank, - // recv_rank, - // send_slice, - // recv_slice, - // comp_range_fld, - // false); - } - if (comm_j) { - comm::CommunicateField(domain.index(), - domain.fields.cur0, - domain.fields.cur0, - send_ind, - recv_ind, - send_rank, - recv_rank, - send_slice, - recv_slice, - comp_range_cur, - false); - } - } else { - if (comm_j) { - comm::CommunicateField(domain.index(), - domain.fields.cur, - domain.fields.cur, - send_ind, - recv_ind, - send_rank, - recv_rank, - send_slice, - recv_slice, - comp_range_cur, - false); - } - } - } - } - - template - void AddBufferedFields(ndfield_t& field, - ndfield_t& buffer, - const range_t& range_policy, - const cell_range_t& components) { - const auto cmin = components.first; - const auto cmax = components.second; - if constexpr (D == Dim::_1D) { - Kokkos::parallel_for( - "AddBufferedFields", - range_policy, - Lambda(cellidx_t i1) { - for (auto c { cmin }; c < cmax; ++c) { - field(i1, c) += buffer(i1, c); - } - }); - } else if constexpr (D == Dim::_2D) { - Kokkos::parallel_for( - "AddBufferedFields", - range_policy, - Lambda(cellidx_t i1, cellidx_t i2) { - for (auto c { cmin }; c < cmax; ++c) { - field(i1, i2, c) += buffer(i1, i2, c); - } - }); - } else if constexpr (D == Dim::_3D) { - Kokkos::parallel_for( - "AddBuffers", - range_policy, - Lambda(cellidx_t i1, cellidx_t i2, cellidx_t i3) { - for (auto c { cmin }; c < cmax; ++c) { - field(i1, i2, i3, c) += buffer(i1, i2, i3, c); - } - }); - } else { - raise::Error("Wrong Dimension", HERE); - } - } - - template - void Metadomain::SynchronizeFields(Domain& domain, - CommTags tags, - const cell_range_t& components) const { - const bool comm_j = (tags & Comm::J); - const bool comm_bckp = (tags & Comm::Bckp); - const bool comm_buff = (tags & Comm::Buff); - raise::ErrorIf(not(comm_j || comm_bckp || comm_buff), - "SynchronizeFields called with no task or incorrect task", - HERE); - raise::ErrorIf(comm_j and comm_buff, - "SynchronizeFields cannot sync J and Buff at the same time", - HERE); - const auto synchronize = true; - - std::string comms; - if (comm_j) { - comms += "J "; - } - if (comm_bckp) { - comms += "Bckp "; - } - if (comm_buff) { - comms += "Buff "; - } - logger::Checkpoint(fmt::format("Synchronizing %s\n", comms.c_str()), HERE); - - auto comp_range_cur = cell_range_t {}; - if (comm_j) { - comp_range_cur = cell_range_t(cur::jx1, cur::jx3 + 1); - Kokkos::deep_copy(domain.fields.buff, ZERO); - } - ndfield_t bckp_recv; - ndfield_t buff_recv; - if (comm_bckp) { - if constexpr (M::Dim == Dim::_1D) { - bckp_recv = ndfield_t { "bckp_recv", - domain.fields.bckp.extent(0) }; - } else if constexpr (M::Dim == Dim::_2D) { - bckp_recv = ndfield_t { "bckp_recv", - domain.fields.bckp.extent(0), - domain.fields.bckp.extent(1) }; - } else if constexpr (M::Dim == Dim::_3D) { - bckp_recv = ndfield_t { "bckp_recv", - domain.fields.bckp.extent(0), - domain.fields.bckp.extent(1), - domain.fields.bckp.extent(2) }; - } - } - if (comm_buff) { - if constexpr (M::Dim == Dim::_1D) { - buff_recv = ndfield_t { "buff_recv", - domain.fields.buff.extent(0) }; - } else if constexpr (M::Dim == Dim::_2D) { - buff_recv = ndfield_t { "buff_recv", - domain.fields.buff.extent(0), - domain.fields.buff.extent(1) }; - } else if constexpr (M::Dim == Dim::_3D) { - buff_recv = ndfield_t { "buff_recv", - domain.fields.buff.extent(0), - domain.fields.buff.extent(1), - domain.fields.buff.extent(2) }; - } - } - // traverse in all directions and sync the fields - for (auto& direction : dir::Directions::all) { - const auto [send_params, - recv_params] = GetSendRecvParams(this, domain, direction, true); - const auto [send_indrank, send_slice] = send_params; - const auto [recv_indrank, recv_slice] = recv_params; - const auto [send_ind, send_rank] = send_indrank; - const auto [recv_ind, recv_rank] = recv_indrank; - if (send_rank < 0 and recv_rank < 0) { - continue; - } - if (comm_j) { - if constexpr (S == SimEngine::GRPIC) { - comm::CommunicateField(domain.index(), - domain.fields.cur0, - domain.fields.buff, - send_ind, - recv_ind, - send_rank, - recv_rank, - send_slice, - recv_slice, - comp_range_cur, - synchronize); - } else { - comm::CommunicateField(domain.index(), - domain.fields.cur, - domain.fields.buff, - send_ind, - recv_ind, - send_rank, - recv_rank, - send_slice, - recv_slice, - comp_range_cur, - synchronize); - } - } - if (comm_bckp) { - comm::CommunicateField(domain.index(), - domain.fields.bckp, - bckp_recv, - send_ind, - recv_ind, - send_rank, - recv_rank, - send_slice, - recv_slice, - components, - synchronize); - } - if (comm_buff) { - comm::CommunicateField(domain.index(), - domain.fields.buff, - buff_recv, - send_ind, - recv_ind, - send_rank, - recv_rank, - send_slice, - recv_slice, - components, - synchronize); - } - } - if (comm_j) { - if constexpr (S == SimEngine::GRPIC) { - AddBufferedFields(domain.fields.cur0, - domain.fields.buff, - domain.mesh.rangeActiveCells(), - comp_range_cur); - } else { - AddBufferedFields(domain.fields.cur, - domain.fields.buff, - domain.mesh.rangeActiveCells(), - comp_range_cur); - } - } - if (comm_bckp) { - AddBufferedFields(domain.fields.bckp, - bckp_recv, - domain.mesh.rangeActiveCells(), - components); - } - if (comm_buff) { - AddBufferedFields(domain.fields.buff, - buff_recv, - domain.mesh.rangeActiveCells(), - components); - } - } - - template - void Metadomain::CommunicateParticles(Domain& domain) const { -#if defined(MPI_ENABLED) - logger::Checkpoint("Communicating particles\n", HERE); - for (auto& species : domain.species) { - const auto ntags = species.ntags(); - - // coordinate shifts per each direction - array_t shifts_in_x1 { "shifts_in_x1", ntags - 2 }; - array_t shifts_in_x2 { "shifts_in_x2", ntags - 2 }; - array_t shifts_in_x3 { "shifts_in_x3", ntags - 2 }; - auto shifts_in_x1_h = Kokkos::create_mirror_view(shifts_in_x1); - auto shifts_in_x2_h = Kokkos::create_mirror_view(shifts_in_x2); - auto shifts_in_x3_h = Kokkos::create_mirror_view(shifts_in_x3); - - // all directions requiring communication - dir::dirs_t dirs_to_comm; - - // ranks & indices of meshblock to send/recv from - dir::map_t send_ranks; - dir::map_t recv_ranks; - - for (const auto& direction : dir::Directions::all) { - // tags corresponding to the direction (both send & recv) - const auto tag_send = mpi::PrtlSendTag::dir2tag(direction); - - // get indices & ranks of send/recv meshblocks - const auto [send_params, - recv_params] = GetSendRecvRanks(this, domain, direction); - const auto [send_ind, send_rank] = send_params; - const auto [recv_ind, recv_rank] = recv_params; - - // skip if no communication is necessary - const auto is_sending = (send_rank >= 0); - const auto is_receiving = (recv_rank >= 0); - if (not is_sending and not is_receiving) { - continue; - } - dirs_to_comm.push_back(direction); - send_ranks[direction] = send_rank; - recv_ranks[direction] = recv_rank; - - // if sending, record displacements to apply before - // ... tag_send - 2: because we only shift tags > 2 (i.e. no dead/alive) - if (is_sending) { - if constexpr (D == Dim::_1D || D == Dim::_2D || D == Dim::_3D) { - if (direction[0] == -1) { - // sending backwards in x1 (add sx1 of target meshblock) - shifts_in_x1_h(tag_send - 2) = subdomain(send_ind).mesh.n_active( - in::x1); - } else if (direction[0] == 1) { - // sending forward in x1 (subtract sx1 of source meshblock) - shifts_in_x1_h(tag_send - 2) = -domain.mesh.n_active(in::x1); - } - } - if constexpr (D == Dim::_2D || D == Dim::_3D) { - if (direction[1] == -1) { - shifts_in_x2_h(tag_send - 2) = subdomain(send_ind).mesh.n_active( - in::x2); - } else if (direction[1] == 1) { - shifts_in_x2_h(tag_send - 2) = -domain.mesh.n_active(in::x2); - } - } - if constexpr (D == Dim::_3D) { - if (direction[2] == -1) { - shifts_in_x3_h(tag_send - 2) = subdomain(send_ind).mesh.n_active( - in::x3); - } else if (direction[2] == 1) { - shifts_in_x3_h(tag_send - 2) = -domain.mesh.n_active(in::x3); - } - } - } - } // end directions loop - - Kokkos::deep_copy(shifts_in_x1, shifts_in_x1_h); - Kokkos::deep_copy(shifts_in_x2, shifts_in_x2_h); - Kokkos::deep_copy(shifts_in_x3, shifts_in_x3_h); - - species.Communicate(dirs_to_comm, - shifts_in_x1, - shifts_in_x2, - shifts_in_x3, - send_ranks, - recv_ranks); - - } // end species loop -#else - (void)domain; -#endif - } - - // NOLINTBEGIN(bugprone-macro-parentheses) -#define METADOMAIN_COMM(S, M, D) \ - template void Metadomain>::CommunicateFields(Domain>&, \ - CommTags) const; \ - template void Metadomain>::SynchronizeFields(Domain>&, \ - CommTags, \ - const cell_range_t&) const; \ - template void Metadomain>::CommunicateParticles(Domain>&) const; - - NTT_FOREACH_SPECIALIZATION(METADOMAIN_COMM) -#undef METADOMAIN_COMM - // NOLINTEND(bugprone-macro-parentheses) - -} // namespace ntt diff --git a/src/framework/domain/metadomain_io.cpp b/src/framework/domain/metadomain_io.cpp deleted file mode 100644 index d53dc49a1..000000000 --- a/src/framework/domain/metadomain_io.cpp +++ /dev/null @@ -1,882 +0,0 @@ -#include "defaults.h" -#include "enums.h" -#include "global.h" - -#include "arch/kokkos_aliases.h" -#include "traits/metric.h" -#include "utils/error.h" -#include "utils/log.h" -#include "utils/numeric.h" - -#include "framework/containers/particles.h" -#include "framework/domain/domain.h" -#include "framework/domain/mesh.h" -#include "framework/domain/metadomain.h" -#include "framework/parameters/parameters.h" -#include "framework/specialization_registry.h" -#include "kernels/divergences.hpp" -#include "kernels/fields_to_phys.hpp" -#include "kernels/particle_moments.hpp" - -#include -#include -#include - -#if defined(MPI_ENABLED) - #include "arch/mpi_aliases.h" - - #include -#endif // MPI_ENABLED - -#include -#include -#include -#include -#include -#include -#include - -namespace ntt { - - template - void Metadomain::InitWriter(adios2::ADIOS* ptr_adios, - const SimulationParams& params) { - raise::ErrorIf( - l_subdomain_indices().size() != 1, - "Output for now is only supported for one subdomain per rank", - HERE); - auto local_domain = subdomain_ptr(l_subdomain_indices()[0]); - raise::ErrorIf(local_domain->is_placeholder(), - "local_domain is a placeholder", - HERE); - const auto incl_ghosts = params.template get("output.debug.ghosts"); - - auto glob_shape_with_ghosts = mesh().n_active(); - auto off_ncells_with_ghosts = local_domain->offset_ncells(); - auto off_ndomains = local_domain->offset_ndomains(); - auto loc_shape_with_ghosts = local_domain->mesh.n_active(); - if (incl_ghosts) { - for (auto d { 0 }; d <= M::Dim; ++d) { - glob_shape_with_ghosts[d] += 2 * N_GHOSTS * ndomains_per_dim()[d]; - off_ncells_with_ghosts[d] += 2 * N_GHOSTS * off_ndomains[d]; - loc_shape_with_ghosts[d] += 2 * N_GHOSTS; - } - } - - g_writer.init( - ptr_adios, - params.template get("output.format"), - params.template get("simulation.name"), - { params.template get("adios2.aggregators_per_node", - defaults::adios2::aggregators_per_node), - params.template get("adios2.max_shm_size", - defaults::adios2::max_shm_size), - params.template get("adios2.buffer_chunk_size", - defaults::adios2::buffer_chunk_size) }); - g_writer.defineMeshLayout(glob_shape_with_ghosts, - off_ncells_with_ghosts, - loc_shape_with_ghosts, - { local_domain->index(), ndomains() }, - params.template get>( - "output.fields.downsampling"), - incl_ghosts, - M::CoordType); - const auto fields_to_write = params.template get>( - "output.fields.quantities"); - const auto custom_fields_to_write = params.template get>( - "output.fields.custom"); - std::vector all_fields_to_write; - std::merge(fields_to_write.begin(), - fields_to_write.end(), - custom_fields_to_write.begin(), - custom_fields_to_write.end(), - std::back_inserter(all_fields_to_write)); - const auto species_to_write = params.template get>( - "output.particles.species"); - g_writer.defineFieldOutputs(S, all_fields_to_write); - - g_writer.clearSpeciesIndex(); - for (const auto& s : species_to_write) { - g_writer.addSpeciesIndex(s); - } - for (const auto sp : g_writer.speciesIndices()) { - local_domain->species[sp - 1].OutputDeclare(g_writer.io()); - } - - // spectra write all particle species - std::vector spectra_species {}; - for (const auto& sp : species_params()) { - spectra_species.push_back(sp.index()); - } - g_writer.defineSpectraOutputs(spectra_species); - for (const auto& type : { "fields", "particles", "spectra" }) { - g_writer.addTracker(type, - params.template get( - "output." + std::string(type) + ".interval"), - params.template get( - "output." + std::string(type) + ".interval_time")); - } - g_writer.writeAttrs(params); - } - - template - void ComputeMoments(const SimulationParams& params, - const Mesh& mesh, - const std::vector>& prtl_species, - const std::vector& species, - const std::vector& components, - ndfield_t& buffer, - idx_t buff_idx) { - std::vector specs = species; - if (specs.empty()) { - // if no species specified, take all massive species - for (auto& sp : prtl_species) { - if (sp.mass() > 0) { - specs.push_back(sp.index()); - } - } - } - for (const auto& sp : specs) { - raise::ErrorIf((sp > prtl_species.size()) or (sp == 0), - "Invalid species index " + std::to_string(sp), - HERE); - } - auto scatter_buff = Kokkos::Experimental::create_scatter_view(buffer); - - // some parameters - const auto use_weights = params.template get("particles.use_weights"); - const auto ni2 = mesh.n_active(in::x2); - const auto inv_n0 = ONE / params.template get("scales.n0"); - const auto smooth_order = params.template get( - "output.fields.smoothing.order"); - const auto smooth_method = OutputSmoothingType::from_string( - params.template get("output.fields.smoothing.method")); - - for (const auto& sp : specs) { - auto& prtl_spec = prtl_species[sp - 1]; - Kokkos::parallel_for( - "ComputeMoments", - prtl_spec.rangeActiveParticles(), - kernel::ParticleMoments_kernel(components, - scatter_buff, - buff_idx, - prtl_spec, - use_weights, - mesh.metric, - mesh.flds_bc(), - ni2, - inv_n0, - smooth_order, - smooth_method)); - } - Kokkos::Experimental::contribute(buffer, scatter_buff); - } - - template - void DeepCopyFields(ndfield_t& fld_from, - ndfield_t& fld_to, - const cell_range_t& from, - const cell_range_t& to) { - for (auto d { 0u }; d < D; ++d) { - raise::ErrorIf(fld_from.extent(d) != fld_to.extent(d), - "Fields have different sizes " + - std::to_string(fld_from.extent(d)) + - " != " + std::to_string(fld_to.extent(d)), - HERE); - } - if constexpr (D == Dim::_1D) { - Kokkos::deep_copy(Kokkos::subview(fld_to, Kokkos::ALL, to), - Kokkos::subview(fld_from, Kokkos::ALL, from)); - } else if constexpr (D == Dim::_2D) { - Kokkos::deep_copy(Kokkos::subview(fld_to, Kokkos::ALL, Kokkos::ALL, to), - Kokkos::subview(fld_from, Kokkos::ALL, Kokkos::ALL, from)); - } else if constexpr (D == Dim::_3D) { - Kokkos::deep_copy( - Kokkos::subview(fld_to, Kokkos::ALL, Kokkos::ALL, Kokkos::ALL, to), - Kokkos::subview(fld_from, Kokkos::ALL, Kokkos::ALL, Kokkos::ALL, from)); - } - } - - template - void ComputeVectorPotential(ndfield_t& buffer, - ndfield_t& EM, - unsigned short buff_idx, - const Mesh& mesh) { - if constexpr (M::Dim == Dim::_2D) { - const auto metric = mesh.metric; - Kokkos::parallel_for( - "ComputeVectorPotential", - mesh.rangeActiveCells(), - Lambda(cellidx_t i1, cellidx_t i2) { - const real_t i1_ { COORD(i1) }; - const ncells_t k_min = 0; - const ncells_t k_max = (i2 - (N_GHOSTS)); - real_t A3 = ZERO; - for (auto k { k_min }; k <= k_max; ++k) { - const real_t k_ = static_cast(k); - const real_t sqrt_detH_ij1 { metric.sqrt_det_h({ i1_, k_ - HALF }) }; - const real_t sqrt_detH_ij2 { metric.sqrt_det_h({ i1_, k_ + HALF }) }; - const auto k1 { k + N_GHOSTS }; - A3 += HALF * (sqrt_detH_ij1 * EM(i1, k1 - 1, em::bx1) + - sqrt_detH_ij2 * EM(i1, k1, em::bx1)); - } - buffer(i1, i2, buff_idx) = A3; - }); - - // @TODO: Implementation with team policies works on AMD, but not on NVIDIA GPUs - // - // using TeamPolicy = Kokkos::TeamPolicy; - // const auto nx1 = mesh.n_active(in::x1); - // const auto nx2 = mesh.n_active(in::x2); - // - // TeamPolicy policy(nx1, Kokkos::AUTO); - // - // Kokkos::parallel_for( - // "ComputeVectorPotential", - // policy, - // Lambda(const TeamPolicy::member_type& team_member) { - // cellidx_t i1 = team_member.league_rank(); - // Kokkos::parallel_scan( - // Kokkos::TeamThreadRange(team_member, nx2), - // [=](cellidx_t i2, real_t& update, const bool final_pass) { - // const auto i1_ { static_cast(i1) }; - // const auto i2_ { static_cast(i2) }; - // const real_t sqrt_detH_ijM { metric.sqrt_det_h({ i1_, i2_ - HALF }) }; - // const real_t sqrt_detH_ijP { metric.sqrt_det_h({ i1_, i2_ + HALF }) }; - // const auto input_val = - // HALF * - // (sqrt_detH_ijM * EM(i1 + N_GHOSTS, i2 + N_GHOSTS - 1, em::bx1) + - // sqrt_detH_ijP * EM(i1 + N_GHOSTS, i2 + N_GHOSTS, em::bx1)); - // if (final_pass) { - // buffer(i1 + N_GHOSTS, i2 + N_GHOSTS, buff_idx) = update; - // } - // update += input_val; - // }); - // }); - } else { - raise::KernelError( - HERE, - "ComputeVectorPotential: 2D implementation called for D != 2"); - } - } - -#if defined(MPI_ENABLED) && defined(OUTPUT_ENABLED) - template - void ExtractVectorPotential(ndfield_t& buffer, - array_t& aphi_r, - unsigned short buff_idx, - const Mesh& mesh) { - Kokkos::parallel_for( - "AddVectorPotential", - mesh.rangeActiveCells(), - Lambda(cellidx_t i1, cellidx_t i2) { - buffer(i1, i2, buff_idx) += aphi_r(i1 - N_GHOSTS); - }); - } - - template - void Metadomain::CommunicateVectorPotential(unsigned short buff_idx) { - if constexpr (M::Dim == Dim::_2D) { - auto local_domain = subdomain_ptr(l_subdomain_indices()[0]); - const auto nx1 = local_domain->mesh.n_active(in::x1); - const auto nx2 = local_domain->mesh.n_active(in::x2); - - auto& buffer = local_domain->fields.bckp; - - const auto nranks_x1 = ndomains_per_dim()[0]; - const auto nranks_x2 = ndomains_per_dim()[1]; - - for (auto nr2 { 1u }; nr2 < nranks_x2; ++nr2) { - const auto rank_send_pre = (nr2 - 1u) * nranks_x1; - const auto rank_recv_pre = nr2 * nranks_x1; - for (auto nr1 { 0u }; nr1 < nranks_x1; ++nr1) { - const auto rank_send = rank_send_pre + nr1; - const auto rank_recv = rank_recv_pre + nr1; - if (static_cast(local_domain->mpi_rank()) == rank_send) { - array_t aphi_r { "Aphi_r", nx1 }; - Kokkos::deep_copy( - aphi_r, - Kokkos::subview(buffer, - std::make_pair(N_GHOSTS, N_GHOSTS + nx1), - N_GHOSTS + nx2 - 1, - buff_idx)); - #if !defined(DEVICE_ENABLED) || defined(GPU_AWARE_MPI) - MPI_Send(aphi_r.data(), - nx1, - mpi::get_type(), - rank_recv, - 0, - MPI_COMM_WORLD); - #else - auto aphi_r_h = Kokkos::create_mirror_view(aphi_r); - Kokkos::deep_copy(aphi_r_h, aphi_r); - MPI_Send(aphi_r_h.data(), - nx1, - mpi::get_type(), - rank_recv, - 0, - MPI_COMM_WORLD); - #endif - } else if (local_domain->mpi_rank() == rank_recv) { - array_t aphi_r { "Aphi_r", nx1 }; - #if !defined(DEVICE_ENABLED) || defined(GPU_AWARE_MPI) - MPI_Recv(aphi_r.data(), - nx1, - mpi::get_type(), - rank_send, - 0, - MPI_COMM_WORLD, - MPI_STATUS_IGNORE); - #else - auto aphi_r_h = Kokkos::create_mirror_view(aphi_r); - MPI_Recv(aphi_r_h.data(), - nx1, - mpi::get_type(), - rank_send, - 0, - MPI_COMM_WORLD, - MPI_STATUS_IGNORE); - Kokkos::deep_copy(aphi_r, aphi_r_h); - #endif - ExtractVectorPotential(buffer, aphi_r, buff_idx, local_domain->mesh); - } - } - } - } else { - raise::Error("CommunicateVectorPotential: comm vector potential only " - "possible for 2D", - HERE); - } - } -#endif - - template - auto Metadomain::Write( - const SimulationParams& params, - timestep_t current_step, - timestep_t finished_step, - simtime_t current_time, - simtime_t finished_time, - const std::function&, - uint32_t, - timestep_t, - simtime_t, - const Domain&)>& CustomFieldOutput) -> bool { - raise::ErrorIf( - l_subdomain_indices().size() != 1, - "Output for now is only supported for one subdomain per rank", - HERE); - const auto write_fields = params.template get( - "output.fields.enable") and - g_writer.shouldWrite("fields", - finished_step, - finished_time); - const auto write_particles = params.template get( - "output.particles.enable") and - g_writer.shouldWrite("particles", - finished_step, - finished_time); - const auto write_spectra = params.template get( - "output.spectra.enable") and - g_writer.shouldWrite("spectra", - finished_step, - finished_time); - const auto extension = params.template get("output.format"); - if (not(write_fields or write_particles or write_spectra) and - extension != "disabled") { - return false; - } - auto local_domain = subdomain_ptr(l_subdomain_indices()[0]); - raise::ErrorIf(local_domain->is_placeholder(), - "local_domain is a placeholder", - HERE); - logger::Checkpoint("Writing output", HERE); - if (write_fields) { - g_writer.beginWriting(WriteMode::Fields, current_step, current_time); - const auto incl_ghosts = params.template get("output.debug.ghosts"); - const auto dwn = params.template get>( - "output.fields.downsampling"); - - auto off_ncells_with_ghosts = local_domain->offset_ncells(); - auto loc_shape_with_ghosts = local_domain->mesh.n_active(); - { // compute positions/sizes of meshblocks in cells in all dimensions - const auto off_ndomains = local_domain->offset_ndomains(); - if (incl_ghosts) { - for (auto d { 0 }; d <= M::Dim; ++d) { - off_ncells_with_ghosts[d] += 2 * N_GHOSTS * off_ndomains[d]; - loc_shape_with_ghosts[d] += 2 * N_GHOSTS; - } - } - } - // Refresh the writer's cached per-rank slab so that field/mesh writes - // pick up the (possibly rebalanced) current local layout. - g_writer.setLocalLayout(off_ncells_with_ghosts, loc_shape_with_ghosts); - for (auto dim { 0u }; dim < M::Dim; ++dim) { - const auto l_size = local_domain->mesh.n_active()[dim]; - const auto l_offset = local_domain->offset_ncells()[dim]; - const auto g_size = mesh().n_active()[dim]; - - const auto dwn_in_dim = dwn[dim]; - - const double n = l_size; - const double d = dwn_in_dim; - const double l = l_offset; - const double f = math::ceil(l / d) * d - l; - - const auto first_cell = static_cast(f); - const auto l_size_dwn = static_cast(math::ceil((n - f) / d)); - - const auto is_last = l_offset + l_size == g_size; - - const auto add_ghost = (incl_ghosts ? 2 * N_GHOSTS : 0); - const auto add_last = (is_last ? 1 : 0); - - const array_t xc { "Xc", l_size_dwn + add_ghost }; - const array_t xe { "Xe", l_size_dwn + add_ghost + add_last }; - - const auto offset = (incl_ghosts ? N_GHOSTS : 0); - const auto ncells = l_size_dwn; - - const auto& metric = local_domain->mesh.metric; - - Kokkos::parallel_for( - "GenerateMesh", - ncells, - Lambda(cellidx_t i_dwn) { - const auto i = first_cell + i_dwn * dwn_in_dim; - const auto i_ = static_cast(i); - coord_t x_Cd { ZERO }, x_Ph { ZERO }; - x_Cd[dim] = i_ + HALF; - // TODO : change to convert by component - metric.template convert(x_Cd, x_Ph); - xc(offset + i_dwn) = x_Ph[dim]; - x_Cd[dim] = i_; - metric.template convert(x_Cd, x_Ph); - xe(offset + i_dwn) = x_Ph[dim]; - if (is_last && i_dwn == ncells - 1) { - x_Cd[dim] = i_ + ONE; - metric.template convert(x_Cd, x_Ph); - xe(offset + i_dwn + 1) = x_Ph[dim]; - } - }); - g_writer.writeMesh( - dim, - xc, - xe, - { off_ncells_with_ghosts[dim], loc_shape_with_ghosts[dim] }); - } - const auto output_asis = params.template get("output.debug.as_is"); - // !TODO: this can probably be optimized to dump things at once - for (auto& fld : g_writer.fieldWriters()) { - Kokkos::deep_copy(local_domain->fields.bckp, ZERO); - std::vector names; - std::vector addresses; - if (fld.comp.empty() || fld.comp.size() == 1) { // scalar - names.push_back(fld.name()); - addresses.push_back(0); - if (fld.is_moment()) { - // output a particle distribution moment (single component) - // this includes T, Rho, Charge, N, Nppc - const auto c = static_cast(addresses.back()); - if (fld.id() == FldsID::T) { - raise::ErrorIf(fld.comp.size() != 1, - "Wrong # of components requested for T output", - HERE); - ComputeMoments(params, - local_domain->mesh, - local_domain->species, - fld.species, - fld.comp[0], - local_domain->fields.bckp, - c); - } else if (fld.id() == FldsID::Rho) { - ComputeMoments(params, - local_domain->mesh, - local_domain->species, - fld.species, - {}, - local_domain->fields.bckp, - c); - } else if (fld.id() == FldsID::Charge) { - ComputeMoments(params, - local_domain->mesh, - local_domain->species, - fld.species, - {}, - local_domain->fields.bckp, - c); - } else if (fld.id() == FldsID::N) { - ComputeMoments(params, - local_domain->mesh, - local_domain->species, - fld.species, - {}, - local_domain->fields.bckp, - c); - } else if (fld.id() == FldsID::Nppc) { - ComputeMoments(params, - local_domain->mesh, - local_domain->species, - fld.species, - {}, - local_domain->fields.bckp, - c); - } else { - raise::Error("Wrong moment requested for output", HERE); - } - } else if (fld.is_divergence()) { - // @TODO: is this correct for GR too? not em0? - const auto c = static_cast(addresses.back()); - Kokkos::parallel_for( - "ComputeDivergence", - local_domain->mesh.rangeActiveCells(), - kernel::ComputeDivergence_kernel(local_domain->mesh.metric, - local_domain->fields.em, - local_domain->fields.bckp, - c)); - } else if (fld.is_custom()) { - if (CustomFieldOutput) { - CustomFieldOutput(fld.name().substr(1), - local_domain->fields.bckp, - addresses.back(), - finished_step, - finished_time, - *local_domain); - } else { - raise::Error("Custom output requested but no function provided", - HERE); - } - } else if (fld.is_vpotential()) { - if constexpr (S == SimEngine::GRPIC && M::Dim == Dim::_2D) { - const auto c = static_cast(addresses.back()); - ComputeVectorPotential(local_domain->fields.bckp, - local_domain->fields.em, - c, - local_domain->mesh); -#if defined(MPI_ENABLED) - CommunicateVectorPotential(c); -#endif - } else { - raise::Error( - "Vector potential can only be computed for GRPIC in 2D", - HERE); - } - } else { - raise::Error("Wrong # of components requested for " - "non-moment/non-custom output", - HERE); - } - SynchronizeFields(*local_domain, - Comm::Bckp, - { addresses.back(), addresses.back() + 1 }); - } else if (fld.comp.size() == 3) { // vector - for (auto i = 0; i < 3; ++i) { - names.push_back(fld.name(i)); - addresses.push_back(i + 3); - } - if (fld.is_moment()) { - for (auto i = 0; i < 3; ++i) { - const auto c = static_cast(addresses[i]); - if (fld.id() == FldsID::T) { - raise::ErrorIf(fld.comp[i].size() != 2, - "Wrong # of components requested for moment", - HERE); - ComputeMoments(params, - local_domain->mesh, - local_domain->species, - fld.species, - fld.comp[i], - local_domain->fields.bckp, - c); - } else if (fld.id() == FldsID::V) { - raise::ErrorIf(fld.comp[i].size() != 1, - "Wrong # of components requested for 3vel", - HERE); - ComputeMoments(params, - local_domain->mesh, - local_domain->species, - fld.species, - fld.comp[i], - local_domain->fields.bckp, - c); - } else { - raise::Error("Wrong moment requested for output", HERE); - } - } - raise::ErrorIf(addresses[1] - addresses[0] != - addresses[2] - addresses[1], - "Indices for the backup are not contiguous", - HERE); - SynchronizeFields(*local_domain, - Comm::Bckp, - { addresses[0], addresses[2] + 1 }); - if constexpr (S == SimEngine::SRPIC) { - if (fld.id() == FldsID::V) { - // normalize 3vel * rho (combuted above) by rho - ComputeMoments(params, - local_domain->mesh, - local_domain->species, - fld.species, - {}, - local_domain->fields.bckp, - 0u); - SynchronizeFields(*local_domain, Comm::Bckp, { 0, 1 }); - Kokkos::parallel_for("NormalizeVectorByRho", - local_domain->mesh.rangeActiveCells(), - kernel::NormalizeVectorByRho_kernel( - local_domain->fields.bckp, - local_domain->fields.bckp, - 0, - addresses[0], - addresses[1], - addresses[2])); - } - } - } else { - // copy fields to bckp (:, 0, 1, 2) - // if as-is specified ==> copy directly to 3, 4, 5 - cell_range_t copy_to = { 0, 3 }; - if (output_asis) { - copy_to = { 3, 6 }; - } - if (fld.is_current()) { - DeepCopyFields(local_domain->fields.cur, - local_domain->fields.bckp, - { cur::jx1, cur::jx3 + 1 }, - copy_to); - } else if (fld.is_field()) { - if (S == SimEngine::GRPIC && fld.is_gr_aux_field()) { - if (fld.is_efield()) { - // GR: E - DeepCopyFields(local_domain->fields.aux, - local_domain->fields.bckp, - { em::ex1, em::ex3 + 1 }, - copy_to); - } else { - // GR: H - DeepCopyFields(local_domain->fields.aux, - local_domain->fields.bckp, - { em::hx1, em::hx3 + 1 }, - copy_to); - } - } else { - if (fld.is_efield()) { - // GR/SR: D/E - DeepCopyFields(local_domain->fields.em, - local_domain->fields.bckp, - { em::ex1, em::ex3 + 1 }, - copy_to); - } else { - // GR/SR: B - DeepCopyFields(local_domain->fields.em, - local_domain->fields.bckp, - { em::bx1, em::bx3 + 1 }, - copy_to); - } - } - } else { - raise::Error("Wrong field requested for output", HERE); - } - if (not output_asis) { - // copy fields from bckp(:, 0, 1, 2) -> bckp(:, 3, 4, 5) - // converting to proper basis and properly interpolating - list_t comp_from = { 0, 1, 2 }; - list_t comp_to = { 3, 4, 5 }; - DeepCopyFields(local_domain->fields.bckp, - local_domain->fields.bckp, - { 0, 3 }, - { 3, 6 }); - Kokkos::parallel_for("FieldsToPhys", - local_domain->mesh.rangeActiveCells(), - kernel::FieldsToPhys_kernel( - local_domain->fields.bckp, - local_domain->fields.bckp, - comp_from, - comp_to, - fld.interp_flag | fld.prepare_flag, - local_domain->mesh.metric)); - } - } - } else if (fld.comp.size() == 4) { // 4-vector - if constexpr (S == SimEngine::GRPIC) { - if (fld.is_moment() && fld.id() == FldsID::V) { - // Compute 4-velocity: V^μ (u^0, u^1, u^2, u^3) - for (auto i = 0; i < 4; ++i) { - names.push_back(fld.name(i)); - addresses.push_back(i); - const auto c = static_cast(addresses[i]); - raise::ErrorIf(fld.comp[i].size() != 1, - "Wrong # of components requested for 4-velocity", - HERE); - ComputeMoments(params, - local_domain->mesh, - local_domain->species, - fld.species, - fld.comp[i], - local_domain->fields.bckp, - c); - } - // Synchronize all 4 components - SynchronizeFields(*local_domain, - Comm::Bckp, - { addresses[0], addresses[3] + 1 }); - // Normalize 4-momentum flux: V^μ = N^μ / sqrt(-N_ν N^ν) - // (computed in coordinate contravariant basis) - Kokkos::parallel_for( - "Normalize4VelocityByNorm", - local_domain->mesh.rangeActiveCells(), - kernel::Normalize4VelocityByNorm_kernel( - local_domain->fields.bckp, - local_domain->fields.bckp, - addresses[0], // c_u0 (column for u^0) - addresses[1], // c_u1 (column for u^1) - addresses[2], // c_u2 (column for u^2) - addresses[3], // c_u3 (column for u^3) - local_domain->mesh.metric)); - // Transform spatial components to physical basis for output - // u^0 (Gamma/alpha) remains unitless, only u^i transform - Kokkos::parallel_for( - "Transform4VelocitySpatialToPhysical", - local_domain->mesh.rangeActiveCells(), - kernel::Transform4VelocitySpatialToPhysical_kernel( - local_domain->fields.bckp, - addresses[1], // c_u1 (column for u^1) - addresses[2], // c_u2 (column for u^2) - addresses[3], // c_u3 (column for u^3) - local_domain->mesh.metric)); - } else { - raise::Error("4-vector output only supported for V (bulk " - "velocity) moment in GRPIC", - HERE); - } - } else { - raise::Error("4-vector output only supported for GRPIC", HERE); - } - } else if (fld.comp.size() == 6) { // tensor - raise::ErrorIf(not fld.is_moment() or fld.id() != FldsID::T, - "Only T tensor has 6 components", - HERE); - for (auto i = 0; i < 6; ++i) { - names.push_back(fld.name(i)); - addresses.push_back(i); - const auto c = static_cast(addresses.back()); - raise::ErrorIf(fld.comp[i].size() != 2, - "Wrong # of components requested for moment", - HERE); - ComputeMoments(params, - local_domain->mesh, - local_domain->species, - fld.species, - fld.comp[i], - local_domain->fields.bckp, - c); - } - SynchronizeFields(*local_domain, - Comm::Bckp, - { addresses[0], addresses[5] + 1 }); - } else { - raise::Error("Wrong # of components requested for output", HERE); - } - g_writer.writeField(names, local_domain->fields.bckp, addresses); - } - g_writer.endWriting(WriteMode::Fields); - } // end shouldWrite("fields", step, time) - - if (write_particles) { - g_writer.beginWriting(WriteMode::Particles, current_step, current_time); - const auto prtl_stride = params.template get( - "output.particles.stride"); - for (const auto spec : g_writer.speciesIndices()) { - local_domain->species[spec - 1].template OutputWrite( - g_writer.io(), - g_writer.writer(), - prtl_stride, - ndomains(), - local_domain->index(), - local_domain->mesh.metric); - } - g_writer.endWriting(WriteMode::Particles); - } // end shouldWrite("particles", step, time) - - if (write_spectra) { - g_writer.beginWriting(WriteMode::Spectra, current_step, current_time); - const auto log_bins = params.template get( - "output.spectra.log_bins"); - const auto n_bins = params.template get("output.spectra.n_bins"); - auto e_min = params.template get("output.spectra.e_min"); - auto e_max = params.template get("output.spectra.e_max"); - if (log_bins) { - e_min = math::log10(e_min); - e_max = math::log10(e_max); - } - const array_t energy { "energy", n_bins + 1 }; - Kokkos::parallel_for( - "GenerateEnergyBins", - n_bins + 1, - Lambda(uint32_t e) { - if (log_bins) { - energy(e) = math::pow(static_cast(10), - e_min + (e_max - e_min) * static_cast(e) / - static_cast(n_bins)); - } else { - energy(e) = e_min + (e_max - e_min) * static_cast(e) / - static_cast(n_bins); - } - }); - for (const auto& spec : g_writer.spectraWriters()) { - auto& species = local_domain->species[spec.species() - 1]; - array_t dn { "dn", n_bins }; - auto dn_scatter = Kokkos::Experimental::create_scatter_view(dn); - Kokkos::parallel_for("ComputeSpectra", - species.rangeActiveParticles(), - kernel::ParticleDistribution_kernel { - species, - dn_scatter, - e_min, - e_max, - log_bins, - n_bins, - local_domain->mesh.metric }); - Kokkos::Experimental::contribute(dn, dn_scatter); - g_writer.writeSpectrum(dn, spec.name()); - } - g_writer.writeSpectrumBins(energy, "sEbn"); - g_writer.endWriting(WriteMode::Spectra); - } // end shouldWrite("spectra", step, time) - - return true; - } - - // NOLINTBEGIN(bugprone-macro-parentheses) -#define METADOMAIN_OUTPUT(S, M, D) \ - template void Metadomain>::InitWriter(adios2::ADIOS*, \ - const SimulationParams&); \ - template auto Metadomain>::Write( \ - const SimulationParams&, \ - timestep_t, \ - timestep_t, \ - simtime_t, \ - simtime_t, \ - const std::function::Dim, 6>&, \ - uint32_t, \ - timestep_t, \ - simtime_t, \ - const Domain>&)>&) -> bool; - - NTT_FOREACH_SPECIALIZATION(METADOMAIN_OUTPUT) - -#undef METADOMAIN_OUTPUT - -#if defined(MPI_ENABLED) - #define COMMVECTORPOTENTIAL(S, M, D) \ - template void Metadomain>::CommunicateVectorPotential(unsigned short); - - NTT_FOREACH_SPECIALIZATION(COMMVECTORPOTENTIAL) - - #undef COMMVECTORPOTENTIAL -#endif - // NOLINTEND(bugprone-macro-parentheses) - -} // namespace ntt diff --git a/src/framework/specialization_registry.h b/src/framework/specialization_registry.h index dea5772c6..60baaee8c 100644 --- a/src/framework/specialization_registry.h +++ b/src/framework/specialization_registry.h @@ -37,6 +37,15 @@ namespace ntt { static constexpr auto dimension = D; }; +#define NTT_FOREACH_COORDINATE(MACRO) \ + MACRO(Dim::_1D, Coord::Cartesian) \ + MACRO(Dim::_2D, Coord::Cartesian) \ + MACRO(Dim::_3D, Coord::Cartesian) \ + MACRO(Dim::_2D, Coord::Spherical) \ + MACRO(Dim::_2D, Coord::Qspherical) \ + MACRO(Dim::_3D, Coord::Spherical) \ + MACRO(Dim::_3D, Coord::Qspherical) + #define NTT_FOREACH_SPECIALIZATION(MACRO) \ MACRO(SimEngine::SRPIC, metric::Minkowski, Dim::_1D) \ MACRO(SimEngine::SRPIC, metric::Minkowski, Dim::_2D) \ diff --git a/tests/framework/comm-mpi.cpp b/tests/framework/comm-mpi.cpp index d9603c8c1..5f099285f 100644 --- a/tests/framework/comm-mpi.cpp +++ b/tests/framework/comm-mpi.cpp @@ -4,7 +4,7 @@ #include "utils/error.h" #include "utils/numeric.h" -#include "framework/domain/comm_mpi.hpp" +#include "framework/domain/comm/fields_mpi.hpp" #include #include diff --git a/tests/framework/comm-nompi.cpp b/tests/framework/comm-nompi.cpp index a6128a1f7..fdc2680a4 100644 --- a/tests/framework/comm-nompi.cpp +++ b/tests/framework/comm-nompi.cpp @@ -4,7 +4,7 @@ #include "arch/kokkos_aliases.h" #include "utils/numeric.h" -#include "framework/domain/comm_nompi.hpp" +#include "framework/domain/comm/fields_nompi.hpp" #include From c5bc9eb23bc59bc752bc56a7abdf0989aff50526 Mon Sep 17 00:00:00 2001 From: haykh Date: Mon, 14 Sep 2026 17:20:37 -0400 Subject: [PATCH 082/125] RUNPGENS + RUNTESTS From 1e41bd501261f00acafd9770af7c034d17f2c993 Mon Sep 17 00:00:00 2001 From: haykh Date: Mon, 14 Sep 2026 17:55:49 -0400 Subject: [PATCH 083/125] refactored stats output in the same spirit --- src/framework/CMakeLists.txt | 4 +- src/framework/domain/io/init.cpp | 44 ++++++++++- .../{metadomain_stats.cpp => io/stats.cpp} | 74 ++++--------------- src/framework/domain/metadomain.h | 17 ++--- 4 files changed, 66 insertions(+), 73 deletions(-) rename src/framework/domain/{metadomain_stats.cpp => io/stats.cpp} (76%) diff --git a/src/framework/CMakeLists.txt b/src/framework/CMakeLists.txt index 4853e2935..318e17c14 100644 --- a/src/framework/CMakeLists.txt +++ b/src/framework/CMakeLists.txt @@ -14,7 +14,6 @@ # * domain/grid.cpp # * domain/metadomain.cpp # * domain/metadomain_sort.cpp -# * domain/metadomain_stats.cpp # * domain/metadomain_reshape.cpp # * domain/metadomain_loadbal.cpp # * domain/checkpoint/init.cpp @@ -28,6 +27,7 @@ # * domain/io/write.cpp # * domain/io/spectra.cpp # * domain/io/fields.cpp +# * domain/io/stats.cpp # * containers/particles.cpp # * containers/particles_sort.cpp # * containers/fields.cpp @@ -67,12 +67,12 @@ set(SOURCES ${SRC_DIR}/domain/grid.cpp ${SRC_DIR}/domain/metadomain.cpp ${SRC_DIR}/domain/metadomain_sort.cpp - ${SRC_DIR}/domain/metadomain_stats.cpp ${SRC_DIR}/domain/metadomain_reshape.cpp ${SRC_DIR}/domain/metadomain_loadbal.cpp ${SRC_DIR}/domain/comm/fields.cpp ${SRC_DIR}/domain/comm/fields_sync.cpp ${SRC_DIR}/domain/comm/particles.cpp + ${SRC_DIR}/domain/io/stats.cpp ${SRC_DIR}/containers/particles.cpp ${SRC_DIR}/containers/particles_sort.cpp ${SRC_DIR}/containers/fields.cpp) diff --git a/src/framework/domain/io/init.cpp b/src/framework/domain/io/init.cpp index 3a965c4f9..c054dc9a5 100644 --- a/src/framework/domain/io/init.cpp +++ b/src/framework/domain/io/init.cpp @@ -104,10 +104,52 @@ namespace ntt { g_writer.writeAttrs(params); } + template + void Metadomain::InitStatsWriter(const SimulationParams& params, + bool is_resuming) { + raise::ErrorIf( + l_subdomain_indices().size() != 1, + "StatsWriter for now is only supported for one subdomain per rank", + HERE); + auto local_domain = subdomain_ptr(l_subdomain_indices()[0]); + raise::ErrorIf(local_domain->is_placeholder(), + "local_domain is a placeholder", + HERE); + const auto simname = params.template get("simulation.name"); + const auto filename = std::filesystem::path(simname) / + (simname + "_stats.csv"); + const auto enable_stats = params.template get("output.stats.enable"); + if (enable_stats and (not is_resuming)) { + CallOnce( + [](auto& filename) { + if (std::filesystem::exists(filename)) { + std::filesystem::remove(filename); + } + }, + filename); + } + const auto stats_to_write = params.template get>( + "output.stats.quantities"); + const auto custom_stats_to_write = params.template get>( + "output.stats.custom"); + g_stats_writer.init( + params.template get("output.stats.interval"), + params.template get("output.stats.interval_time")); + g_stats_writer.defineStatsFilename(filename); + g_stats_writer.defineStatsOutputs(stats_to_write, false); + g_stats_writer.defineStatsOutputs(custom_stats_to_write, true); + + if (not std::filesystem::exists(filename)) { + g_stats_writer.writeHeader(); + } + } + // NOLINTBEGIN(bugprone-macro-parentheses) #define METADOMAIN_OUTPUT(S, M, D) \ template void Metadomain>::InitWriter(adios2::ADIOS*, \ - const SimulationParams&); + const SimulationParams&); \ + template void Metadomain>::InitStatsWriter(const SimulationParams&, \ + bool); NTT_FOREACH_SPECIALIZATION(METADOMAIN_OUTPUT) diff --git a/src/framework/domain/metadomain_stats.cpp b/src/framework/domain/io/stats.cpp similarity index 76% rename from src/framework/domain/metadomain_stats.cpp rename to src/framework/domain/io/stats.cpp index 35ff26e75..b45c80f3b 100644 --- a/src/framework/domain/metadomain_stats.cpp +++ b/src/framework/domain/io/stats.cpp @@ -20,53 +20,11 @@ #include #include -#include -#include #include #include namespace ntt { - template - void Metadomain::InitStatsWriter(const SimulationParams& params, - bool is_resuming) { - raise::ErrorIf( - l_subdomain_indices().size() != 1, - "StatsWriter for now is only supported for one subdomain per rank", - HERE); - auto local_domain = subdomain_ptr(l_subdomain_indices()[0]); - raise::ErrorIf(local_domain->is_placeholder(), - "local_domain is a placeholder", - HERE); - const auto simname = params.template get("simulation.name"); - const auto filename = std::filesystem::path(simname) / - (simname + "_stats.csv"); - const auto enable_stats = params.template get("output.stats.enable"); - if (enable_stats and (not is_resuming)) { - CallOnce( - [](auto& filename) { - if (std::filesystem::exists(filename)) { - std::filesystem::remove(filename); - } - }, - filename); - } - const auto stats_to_write = params.template get>( - "output.stats.quantities"); - const auto custom_stats_to_write = params.template get>( - "output.stats.custom"); - g_stats_writer.init( - params.template get("output.stats.interval"), - params.template get("output.stats.interval_time")); - g_stats_writer.defineStatsFilename(filename); - g_stats_writer.defineStatsOutputs(stats_to_write, false); - g_stats_writer.defineStatsOutputs(custom_stats_to_write, true); - - if (not std::filesystem::exists(filename)) { - g_stats_writer.writeHeader(); - } - } - template auto ComputeMoments(const SimulationParams& params, const Mesh& mesh, @@ -182,14 +140,12 @@ namespace ntt { } template - auto Metadomain::WriteStats( - const SimulationParams& params, - timestep_t current_step, - timestep_t finished_step, - simtime_t current_time, - simtime_t finished_time, - const std::function< - real_t(const std::string&, timestep_t, simtime_t, const Domain&)>& CustomStat) + auto Metadomain::WriteStats(const SimulationParams& params, + timestep_t current_step, + timestep_t finished_step, + simtime_t current_time, + simtime_t finished_time, + const custom_stats_output_t& CustomStat) -> bool { if (not(params.template get("output.stats.enable") and g_stats_writer.shouldWrite(finished_step, finished_time))) { @@ -281,17 +237,13 @@ namespace ntt { } // NOLINTBEGIN(bugprone-macro-parentheses) -#define METADOMAIN_STATS(S, M, D) \ - template void Metadomain>::InitStatsWriter(const SimulationParams&, \ - bool); \ - template auto Metadomain>::WriteStats( \ - const SimulationParams&, \ - timestep_t, \ - timestep_t, \ - simtime_t, \ - simtime_t, \ - const std::function< \ - real_t(const std::string&, timestep_t, simtime_t, const Domain>&)>&) \ +#define METADOMAIN_STATS(S, M, D) \ + template auto Metadomain>::WriteStats(const SimulationParams&, \ + timestep_t, \ + timestep_t, \ + simtime_t, \ + simtime_t, \ + const custom_stats_output_t&) \ -> bool; NTT_FOREACH_SPECIALIZATION(METADOMAIN_STATS) diff --git a/src/framework/domain/metadomain.h b/src/framework/domain/metadomain.h index 0330b5590..bf7a562d4 100644 --- a/src/framework/domain/metadomain.h +++ b/src/framework/domain/metadomain.h @@ -189,16 +189,15 @@ namespace ntt { const std::vector>&); #endif + using custom_stats_output_t = std::function< + real_t(const std::string&, timestep_t, simtime_t, const Domain&)>; void InitStatsWriter(const SimulationParams&, bool); - auto WriteStats( - const SimulationParams&, - timestep_t, - timestep_t, - simtime_t, - simtime_t, - const std::function< - real_t(const std::string&, timestep_t, simtime_t, const Domain&)>& = nullptr) - -> bool; + auto WriteStats(const SimulationParams&, + timestep_t, + timestep_t, + simtime_t, + simtime_t, + const custom_stats_output_t& = nullptr) -> bool; /* setters -------------------------------------------------------------- */ void setFldsBC(const bc_in&, const FldsBC&); From 2076f0f6ccd16223ce9ac00dff152910e9b017e3 Mon Sep 17 00:00:00 2001 From: haykh Date: Tue, 15 Sep 2026 11:26:06 -0400 Subject: [PATCH 084/125] particle energy distribution calc refactored (incl. spatial) --- src/framework/domain/io/spectra.cpp | 157 ++++++++++--- src/global/arch/kokkos_aliases.h | 34 ++- src/kernels/particle_energy_distribution.hpp | 218 +++++++++++++++++++ src/kernels/particle_moments.hpp | 79 ------- 4 files changed, 375 insertions(+), 113 deletions(-) create mode 100644 src/kernels/particle_energy_distribution.hpp diff --git a/src/framework/domain/io/spectra.cpp b/src/framework/domain/io/spectra.cpp index 317c76b6d..ba54c6a65 100644 --- a/src/framework/domain/io/spectra.cpp +++ b/src/framework/domain/io/spectra.cpp @@ -10,7 +10,7 @@ #include "framework/domain/metadomain.h" #include "framework/parameters/parameters.h" #include "framework/specialization_registry.h" -#include "kernels/particle_moments.hpp" +#include "kernels/particle_energy_distribution.hpp" #include #include @@ -20,6 +20,7 @@ #include #endif // MPI_ENABLED +#include #include #include #include @@ -27,53 +28,143 @@ namespace ntt { + template + void GenerateEnergyDistribution(const Particles& species, + array_t& dn, + const kernel::EnergyBinning& energy_binning, + const M& metric) { + auto dn_scatter = Kokkos::Experimental::create_scatter_view(dn); + Kokkos::parallel_for("ComputeSpectraSpatial", + species.rangeActiveParticles(), + kernel::ParticleDistribution_kernel { species, + dn_scatter, + energy_binning, + metric }); + Kokkos::Experimental::contribute(dn, dn_scatter); + } + + template + void GenerateSpatialEnergyDistribution( + const Particles& species, + nddata_t(M::Dim) + 1, real_t>& dn, + const array_t& nmin_i, + const array_t& dncells_i, + const array_t& nbins_i, + const kernel::EnergyBinning& energy_binning, + const M& metric) { + auto dn_scatter = Kokkos::Experimental::create_scatter_view(dn); + Kokkos::parallel_for( + "ComputeSpectra", + species.rangeActiveParticles(), + kernel::ParticleDistributionSpatial_kernel { species, + dn_scatter, + nmin_i, + dncells_i, + nbins_i, + energy_binning, + metric }); + Kokkos::Experimental::contribute(dn, dn_scatter); + } + template void Metadomain::WriteSpectra(const SimulationParams& params, Domain* local_domain, timestep_t current_step, simtime_t current_time) { - + constexpr auto dim = static_cast(M::Dim); g_writer.beginWriting(WriteMode::Spectra, current_step, current_time); const auto log_bins = params.template get("output.spectra.log_bins"); - const auto n_bins = params.template get("output.spectra.n_bins"); - auto e_min = params.template get("output.spectra.e_min"); - auto e_max = params.template get("output.spectra.e_max"); + + const auto num_energy_bins = params.template get( + "output.spectra.num_energy_bins"); + auto e_min = params.template get("output.spectra.e_min"); + auto e_max = params.template get("output.spectra.e_max"); if (log_bins) { e_min = math::log10(e_min); e_max = math::log10(e_max); } - const array_t energy { "energy", n_bins + 1 }; - Kokkos::parallel_for( - "GenerateEnergyBins", - n_bins + 1, - Lambda(uint32_t e) { - if (log_bins) { - energy(e) = math::pow(static_cast(10), - e_min + (e_max - e_min) * static_cast(e) / - static_cast(n_bins)); + + const auto num_spatial_bins = params.template get>( + "output.spectra.num_spatial_bins"); + const auto spatial_binning_enabled = std::any_of(num_spatial_bins.begin(), + num_spatial_bins.end(), + [](const auto& n) { + return n != 1u; + }); + + // fractional number of cells per each direction in each bin + const array_t dncells_i { "dncells_i" }; + // left edge of the local domain in each direction in number of cells + const array_t nmin_i { "nmin_i" }; + const array_t nbins_i { "nbins_i" }; + if (spatial_binning_enabled) { + auto dncells_i_h = Kokkos::create_mirror_view(dncells_i); + auto nmin_i_h = Kokkos::create_mirror_view(nmin_i); + auto nbins_i_h = Kokkos::create_mirror_view(nbins_i); + for (auto d = 0u; d < dim; ++d) { + dncells_i_h(d) = static_cast(mesh().n_active(static_cast(d))) / + static_cast(num_spatial_bins[d]); + nmin_i_h(d) = static_cast(local_domain->offset_ncells()[d]); + nbins_i_h(d) = num_spatial_bins[d]; + } + Kokkos::deep_copy(dncells_i, dncells_i_h); + Kokkos::deep_copy(nmin_i, nmin_i_h); + Kokkos::deep_copy(nbins_i, nbins_i_h); + } + for (const auto& spec : g_writer.spectraWriters()) { + auto& species = local_domain->species[spec.species() - 1]; + if (not spatial_binning_enabled) { + array_t dn { "dn", num_energy_bins }; + GenerateEnergyDistribution(species, + dn, + { e_min, e_max, log_bins, num_energy_bins }, + local_domain->mesh.metric); + g_writer.writeSpectrum(dn, spec.name()); + } else { + nddata_t dn; + if constexpr (M::Dim == Dim::_1D) { + dn = { "dn", num_spatial_bins[0], num_energy_bins }; + } else if constexpr (M::Dim == Dim::_2D) { + dn = { "dn", num_spatial_bins[0], num_spatial_bins[1], num_energy_bins }; + } else if constexpr (M::Dim == Dim::_3D) { + dn = { "dn", + num_spatial_bins[0], + num_spatial_bins[1], + num_spatial_bins[2], + num_energy_bins }; } else { - energy(e) = e_min + (e_max - e_min) * static_cast(e) / - static_cast(n_bins); + raise::Error("invalid dimension", HERE); } - }); - for (const auto& spec : g_writer.spectraWriters()) { - auto& species = local_domain->species[spec.species() - 1]; - array_t dn { "dn", n_bins }; - auto dn_scatter = Kokkos::Experimental::create_scatter_view(dn); + GenerateSpatialEnergyDistribution( + species, + dn, + nmin_i, + dncells_i, + nbins_i, + { e_min, e_max, log_bins, num_energy_bins }, + local_domain->mesh.metric); + } + } + + { + // energy bins + const array_t energy { "energy", num_energy_bins + 1 }; Kokkos::parallel_for( - "ComputeSpectra", - species.rangeActiveParticles(), - kernel::ParticleDistribution_kernel { species, - dn_scatter, - e_min, - e_max, - log_bins, - n_bins, - local_domain->mesh.metric }); - Kokkos::Experimental::contribute(dn, dn_scatter); - g_writer.writeSpectrum(dn, spec.name()); + "GenerateEnergyBins", + num_energy_bins + 1, + Lambda(uint32_t e) { + if (log_bins) { + energy(e) = math::pow(static_cast(10), + e_min + (e_max - e_min) * static_cast(e) / + static_cast(num_energy_bins)); + } else { + energy(e) = e_min + (e_max - e_min) * static_cast(e) / + static_cast(num_energy_bins); + } + }); + + g_writer.writeSpectrumBins(energy, "sEbn"); } - g_writer.writeSpectrumBins(energy, "sEbn"); g_writer.endWriting(WriteMode::Spectra); } diff --git a/src/global/arch/kokkos_aliases.h b/src/global/arch/kokkos_aliases.h index 7212b554e..b0ed48773 100644 --- a/src/global/arch/kokkos_aliases.h +++ b/src/global/arch/kokkos_aliases.h @@ -4,7 +4,7 @@ * @implements * - ClassLambda, Lambda, Function, Inline macros * - array_t, array_mirror_t, scatter_array_t - * - ndarray_t, ndfield_t + * - nddata_t, scatter_nddata_t, ndarray_t, ndfield_t * - ndfield_mirror_t, scatter_ndfield_t * - range_t, range_h_t * - CreateRangePolicy, CreateRangePolicyOnHost @@ -80,6 +80,38 @@ namespace kokkos_aliases_hidden { template using nddata_t = typename kokkos_aliases_hidden::nddata_impl::type; +// Scatter view for nddata +namespace kokkos_aliases_hidden { + // c++ magic + template + struct scatter_nddata_impl { + using type = void; + }; + + template + struct scatter_nddata_impl<1, T> { + using type = scatter_array_t; + }; + + template + struct scatter_nddata_impl<2, T> { + using type = scatter_array_t; + }; + + template + struct scatter_nddata_impl<3, T> { + using type = scatter_array_t; + }; + + template + struct scatter_nddata_impl<4, T> { + using type = scatter_array_t; + }; +} // namespace kokkos_aliases_hidden + +template +using scatter_nddata_t = typename kokkos_aliases_hidden::scatter_nddata_impl::type; + template using ndarray_t = typename kokkos_aliases_hidden::nddata_impl::type; diff --git a/src/kernels/particle_energy_distribution.hpp b/src/kernels/particle_energy_distribution.hpp new file mode 100644 index 000000000..d1590480b --- /dev/null +++ b/src/kernels/particle_energy_distribution.hpp @@ -0,0 +1,218 @@ +/** + * @file kernels/particle_moments.hpp + * @brief Kernels for computing particle distribution functions + * @implements + * - kernel::EnergyBinning + * - kernel::ParticleDistributionBase_kernel<> + * - kernel::ParticleDistribution_kernel<> + * - kernel::ParticleDistributionSpatial_kernel<> + * @namespaces: + * - kernel:: + */ + +#ifndef KERNELS_PARTICLE_ENERGY_DISTRIBUTION_HPP +#define KERNELS_PARTICLE_ENERGY_DISTRIBUTION_HPP + +#include "enums.h" +#include "global.h" + +#include "arch/kokkos_aliases.h" +#include "traits/metric.h" +#include "utils/error.h" +#include "utils/numeric.h" + +#include "framework/containers/particles.h" + +namespace kernel { + using namespace ntt; + + struct EnergyBinning { + const real_t e_min, e_max; + const bool log_bins; + const size_t n_bins; + + EnergyBinning(real_t e_min, real_t e_max, bool log_bins, size_t n_bins) + : e_min { e_min } + , e_max { e_max } + , log_bins { log_bins } + , n_bins { n_bins } {} + }; + + template + struct ParticleDistributionBase_kernel { + const ParticleArrays particles; + const bool is_massive; + const M metric; + + const EnergyBinning energy_binning; + + ParticleDistributionBase_kernel(const Particles& particles, + const EnergyBinning& energy_binning, + const M& metric) + : particles { static_cast(particles) } + , is_massive { (particles.mass() != 0.0f) } + , energy_binning { energy_binning } + , metric { metric } {} + + Inline auto EnergyBinIndex(prtlidx_t p) const -> size_t { + real_t en; + if constexpr (S == SimEngine::SRPIC) { + if (is_massive) { + en = U2GAMMA(particles.ux1(p), particles.ux2(p), particles.ux3(p)) - ONE; + } else { + en = NORM(particles.ux1(p), particles.ux2(p), particles.ux3(p)); + } + } else if constexpr (S == SimEngine::GRPIC) { + coord_t x_Code { ZERO }; + x_Code[0] = static_cast(particles.i1(p)) + + static_cast(particles.dx1(p)); + x_Code[1] = static_cast(particles.i2(p)) + + static_cast(particles.dx2(p)); + + // raise full covariant 4-vector to get correct contravariant u^0 + // u^i != h^{ij} u_j + const real_t u_0_cov { metric.u_0( + x_Code, + { particles.ux1(p), particles.ux2(p), particles.ux3(p) }, + (is_massive) ? ONE : ZERO) }; + vec_t u_cntrv_4d { ZERO }; + metric.template transform_4d( + x_Code, + { u_0_cov, particles.ux1(p), particles.ux2(p), particles.ux3(p) }, + u_cntrv_4d); + // in GR: u^0 = Gamma/alpha + const real_t Gamma { metric.alpha(x_Code) * u_cntrv_4d[0] }; + en = is_massive ? (Gamma - ONE) : Gamma; + } + if (energy_binning.log_bins) { + en = math::log10(en); + } + if (en <= energy_binning.e_min) { + return 0u; + } else if (en >= energy_binning.e_max) { + return energy_binning.n_bins; + } else { + return static_cast(static_cast(energy_binning.n_bins) * + (en - energy_binning.e_min) / + (energy_binning.e_max - energy_binning.e_min)); + } + } + }; + + template + class ParticleDistribution_kernel + : public ParticleDistributionBase_kernel { + scatter_array_t dn_scatter; + using ParticleDistributionBase_kernel::particles; + using ParticleDistributionBase_kernel::EnergyBinIndex; + + public: + ParticleDistribution_kernel(const Particles& particles, + const scatter_array_t& dn_scatter, + const EnergyBinning& energy_binning, + const M& metric) + : ParticleDistributionBase_kernel(particles, energy_binning, metric) + , dn_scatter { dn_scatter } {} + + Inline void operator()(prtlidx_t p) const { + if (particles.tag(p) != ParticleTag::alive) { + return; + } + const auto e_ind = EnergyBinIndex(p); + auto dn_acc = dn_scatter.access(); + dn_acc(e_ind) += particles.weight(p); + } + }; + + template + class ParticleDistributionSpatial_kernel + : public ParticleDistributionBase_kernel { + static constexpr auto D = M::Dim; + static constexpr auto Dim = static_cast(D); + static constexpr auto DimP1 = Dim + 1u; + using ParticleDistributionBase_kernel::particles; + using ParticleDistributionBase_kernel::EnergyBinIndex; + + scatter_nddata_t dn_scatter; + const array_t nmin_i; + const array_t dncells_i; + const array_t nbins_i; + + public: + ParticleDistributionSpatial_kernel( + const Particles& particles, + const scatter_nddata_t& dn_scatter, + const array_t nmin_i, + const array_t dncells_i, + const array_t nbins_i, + const EnergyBinning& energy_binning, + const M& metric) + : ParticleDistributionBase_kernel(particles, energy_binning, metric) + , dn_scatter { dn_scatter } + , nmin_i { nmin_i } + , dncells_i { dncells_i } + , nbins_i { nbins_i } {} + + Inline auto SpatialBinIndex(real_t i, real_t nmin, real_t dncells, size_t nbins) const + -> size_t { + const auto ni = i + nmin; + if (ni <= ZERO) { + return 0u; + } else if (ni >= dncells * static_cast(nbins)) { + return nbins - 1u; + } else { + return static_cast(static_cast(nbins) * ni / dncells); + } + } + + Inline void operator()(prtlidx_t p) const { + if (particles.tag(p) != ParticleTag::alive) { + return; + } + const auto e_ind = EnergyBinIndex(p); + if constexpr (D == Dim::_1D) { + const auto i1_ind = SpatialBinIndex(static_cast(particles.i1(p)) + + static_cast(particles.dx1(p)), + nmin_i(0), + dncells_i(0), + nbins_i(0)); + auto dn_acc = dn_scatter.access(); + dn_acc(i1_ind, e_ind) += particles.weight(p); + } else if constexpr (D == Dim::_2D) { + const auto i1_ind = SpatialBinIndex(static_cast(particles.i1(p)) + + static_cast(particles.dx1(p)), + nmin_i(0), + dncells_i(0), + nbins_i(0)); + const auto i2_ind = SpatialBinIndex(static_cast(particles.i2(p)) + + static_cast(particles.dx2(p)), + nmin_i(1), + dncells_i(1), + nbins_i(1)); + dn_acc(i1_ind, i2_ind, e_ind) += particles.weight(p); + } else if constexpr (D == Dim::_3D) { + const auto i1_ind = SpatialBinIndex(static_cast(particles.i1(p)) + + static_cast(particles.dx1(p)), + nmin_i(0), + dncells_i(0), + nbins_i(0)); + const auto i2_ind = SpatialBinIndex(static_cast(particles.i2(p)) + + static_cast(particles.dx2(p)), + nmin_i(1), + dncells_i(1), + nbins_i(1)); + const auto i3_ind = SpatialBinIndex(static_cast(particles.i3(p)) + + static_cast(particles.dx3(p)), + nmin_i(2), + dncells_i(2), + nbins_i(2)); + dn_acc(i1_ind, i2_ind, i3_ind, e_ind) += particles.weight(p); + } else { + raise::KernelError(HERE, "invalid dimension"); + } + } + }; + +} // namespace kernel + +#endif // KERNELS_PARTICLE_ENERGY_DISTRIBUTION_HPP diff --git a/src/kernels/particle_moments.hpp b/src/kernels/particle_moments.hpp index 1244ba21d..bb5e837b0 100644 --- a/src/kernels/particle_moments.hpp +++ b/src/kernels/particle_moments.hpp @@ -710,85 +710,6 @@ namespace kernel { } }; - template - class ParticleDistribution_kernel { - const ParticleArrays particles; - const bool is_massive; - const M metric; - - const real_t e_min, e_max; - const bool log_bins; - const size_t n_bins; - - scatter_array_t dn_scatter; - - public: - ParticleDistribution_kernel(const Particles& particles, - const scatter_array_t& dn_scatter, - real_t e_min, - real_t e_max, - bool log_bins, - size_t n_bins, - const M& metric) - : particles { static_cast(particles) } - , is_massive { (particles.mass() != 0.0f) } - , dn_scatter { dn_scatter } - , e_min { e_min } - , e_max { e_max } - , log_bins { log_bins } - , n_bins { n_bins } - , metric { metric } {} - - Inline void operator()(prtlidx_t p) const { - if (particles.tag(p) != ParticleTag::alive) { - return; - } - real_t en; - if constexpr (S == SimEngine::SRPIC) { - if (is_massive) { - en = U2GAMMA(particles.ux1(p), particles.ux2(p), particles.ux3(p)) - ONE; - } else { - en = NORM(particles.ux1(p), particles.ux2(p), particles.ux3(p)); - } - } else if constexpr (S == SimEngine::GRPIC) { - coord_t x_Code { ZERO }; - x_Code[0] = static_cast(particles.i1(p)) + - static_cast(particles.dx1(p)); - x_Code[1] = static_cast(particles.i2(p)) + - static_cast(particles.dx2(p)); - - // raise full covariant 4-vector to get correct contravariant u^0 - // u^i != h^{ij} u_j - const real_t u_0_cov { metric.u_0( - x_Code, - { particles.ux1(p), particles.ux2(p), particles.ux3(p) }, - (is_massive) ? ONE : ZERO) }; - vec_t u_cntrv_4d { ZERO }; - metric.template transform_4d( - x_Code, - { u_0_cov, particles.ux1(p), particles.ux2(p), particles.ux3(p) }, - u_cntrv_4d); - // in GR: u^0 = Gamma/alpha - const real_t Gamma { metric.alpha(x_Code) * u_cntrv_4d[0] }; - en = is_massive ? (Gamma - ONE) : Gamma; - } - if (log_bins) { - en = math::log10(en); - } - size_t e_ind = 0; - if (en <= e_min) { - e_ind = 0; - } else if (en >= e_max) { - e_ind = n_bins; - } else { - e_ind = static_cast( - static_cast(n_bins) * (en - e_min) / (e_max - e_min)); - } - auto dn_acc = dn_scatter.access(); - dn_acc(e_ind) += particles.weight(p); - } - }; - } // namespace kernel #endif // KERNELS_PARTICLE_MOMENTS_HPP From 69899388b92ff2fc96010afc135e0722b8a0f4c3 Mon Sep 17 00:00:00 2001 From: haykh Date: Tue, 15 Sep 2026 11:47:39 -0400 Subject: [PATCH 085/125] declare and call spatial bins in writing --- src/framework/domain/io/init.cpp | 4 +++- src/framework/domain/io/spectra.cpp | 8 ++++++++ src/output/writer.h | 3 ++- 3 files changed, 13 insertions(+), 2 deletions(-) diff --git a/src/framework/domain/io/init.cpp b/src/framework/domain/io/init.cpp index c054dc9a5..472ba9618 100644 --- a/src/framework/domain/io/init.cpp +++ b/src/framework/domain/io/init.cpp @@ -93,7 +93,9 @@ namespace ntt { for (const auto& sp : species_params()) { spectra_species.push_back(sp.index()); } - g_writer.defineSpectraOutputs(spectra_species); + const auto num_spatial_bins = params.template get>( + "output.spectra.num_spatial_bins"); + g_writer.defineSpectraOutputs(spectra_species, num_spatial_bins); for (const auto& type : { "fields", "particles", "spectra" }) { g_writer.addTracker(type, params.template get( diff --git a/src/framework/domain/io/spectra.cpp b/src/framework/domain/io/spectra.cpp index ba54c6a65..b053fca54 100644 --- a/src/framework/domain/io/spectra.cpp +++ b/src/framework/domain/io/spectra.cpp @@ -143,6 +143,7 @@ namespace ntt { nbins_i, { e_min, e_max, log_bins, num_energy_bins }, local_domain->mesh.metric); + // @TODO: write } } @@ -165,6 +166,13 @@ namespace ntt { g_writer.writeSpectrumBins(energy, "sEbn"); } + if (spatial_binning_enabled) { + for (auto d = 0u; d < dim; ++d) { + array_t xi { "xi", num_spatial_bins[d] + 1 }; + // @TODO: fill spatial bins (in physical units) + g_writer.writeSpectrumBins(xi, "sX" + std::to_string(d + 1) + "bn"); + } + } g_writer.endWriting(WriteMode::Spectra); } diff --git a/src/output/writer.h b/src/output/writer.h index 9fee71d80..18193eaad 100644 --- a/src/output/writer.h +++ b/src/output/writer.h @@ -121,7 +121,8 @@ namespace out { const std::vector& loc_shape); void defineFieldOutputs(const SimEngine&, const std::vector&); - void defineSpectraOutputs(const std::vector&); + void defineSpectraOutputs(const std::vector&, + const std::vector&); void writeMesh(unsigned short, const array_t&, From 7c88d0fa4bb2580f0b127f051339af80959319ca Mon Sep 17 00:00:00 2001 From: haykh Date: Tue, 15 Sep 2026 13:23:19 -0400 Subject: [PATCH 086/125] spatially binned spectra done (probably) --- src/framework/domain/io/spectra.cpp | 215 ++++------------------- src/output/writer.cpp | 262 ++++++++++++++++++++-------- src/output/writer.h | 6 +- 3 files changed, 231 insertions(+), 252 deletions(-) diff --git a/src/framework/domain/io/spectra.cpp b/src/framework/domain/io/spectra.cpp index b053fca54..f628ff05b 100644 --- a/src/framework/domain/io/spectra.cpp +++ b/src/framework/domain/io/spectra.cpp @@ -66,6 +66,20 @@ namespace ntt { Kokkos::Experimental::contribute(dn, dn_scatter); } + template + void GenerateSpatialBins(array_t& bins, + const M& metric, + real_t dncells, + size_t nbins) { + Kokkos::parallel_for( + "GenerateSpatialBins", + nbins + 1, + Lambda(uint32_t b) { + bins(b) = metric.template convert( + static_cast(b) * dncells); + }); + } + template void Metadomain::WriteSpectra(const SimulationParams& params, Domain* local_domain, @@ -143,7 +157,8 @@ namespace ntt { nbins_i, { e_min, e_max, log_bins, num_energy_bins }, local_domain->mesh.metric); - // @TODO: write + g_writer.writeSpectrumSpatial(dim + 1u)>(dn, + spec.name()); } } @@ -167,9 +182,26 @@ namespace ntt { g_writer.writeSpectrumBins(energy, "sEbn"); } if (spatial_binning_enabled) { + const auto metric = mesh().metric; + const auto n_active = mesh().n_active(); for (auto d = 0u; d < dim; ++d) { array_t xi { "xi", num_spatial_bins[d] + 1 }; - // @TODO: fill spatial bins (in physical units) + // # of cells per spatial bin in direction `d` (fractional) + const auto dncells = static_cast(n_active[d]) / + static_cast(num_spatial_bins[d]); + if (d == 0) { + GenerateSpatialBins<1>(xi, metric, dncells, num_spatial_bins[d]); + } else if (d == 1) { + if constexpr (dim > 1) { + GenerateSpatialBins<2>(xi, metric, dncells, num_spatial_bins[d]); + } + } else if (d == 2) { + if constexpr (dim > 2) { + GenerateSpatialBins<3>(xi, metric, dncells, num_spatial_bins[d]); + } + } else { + raise::Error("invalid dimension", HERE); + } g_writer.writeSpectrumBins(xi, "sX" + std::to_string(d + 1) + "bn"); } } @@ -189,182 +221,3 @@ namespace ntt { // NOLINTEND(bugprone-macro-parentheses) } // namespace ntt - -// if (write_spectra3D) { -// g_writer.beginWriting(WriteMode::Spectra3D, current_step, current_time); -// const auto log_bins = params.template get( -// "output.spectra3D.log_bins"); -// const auto n_bins = params.template get( -// "output.spectra3D.n_bins"); -// const auto& metric = local_domain->mesh.metric; -// // extract the number of bins globally in each direction -// const auto nx1_bins = params.template get( -// "output.spectra3D.nx1"); -// const auto nx2_bins = params.template get( -// "output.spectra3D.nx2"); -// const auto nx3_bins = params.template get( -// "output.spectra3D.nx3"); -// -// // select the min and max energy for the spectra -// auto e_min = params.template get("output.spectra3D.e_min"); -// auto e_max = params.template get("output.spectra3D.e_max"); -// -// auto x1_min = mesh().extent(in::x1).first; // extent is in physical units, not code -// decltype(x1_min) x2_min = 0; -// decltype(x1_min) x3_min = 0; -// -// auto x1_max = mesh().extent(in::x1).second; // pairs in c++ are addressed by first and secocnd -// decltype(x1_max) x2_max = 0; -// decltype(x1_max) x3_max = 0; -// -// if constexpr (D == Dim::_2D or -// D == Dim::_3D) { // only pick x2 if simulation is in 2D -// -// x2_min = mesh().extent(in::x2).first; -// x2_max = mesh().extent(in::x2).second; -// } -// -// if constexpr (D == Dim::_3D) { -// -// x3_min = mesh().extent(in::x3).first; -// x3_max = mesh().extent(in::x3).second; -// } -// -// // auto dx_slice = mesh().dx1; -// -// // auto dxslice = (x1_max - x1_min) / x_bins -// -// if (log_bins) { -// e_min = math::log10(e_min); -// e_max = math::log10(e_max); -// } -// array_t energy { "energy", n_bins + 1 }; -// // generating the energy bins -// Kokkos::parallel_for( -// "GenerateEnergyBins", -// n_bins + 1, -// Lambda(uint32_t e) { -// if (log_bins) { -// energy(e) = math::pow(10.0, e_min + (e_max - e_min) * e / n_bins); -// } else { -// energy(e) = e_min + (e_max - e_min) * e / n_bins; -// } -// }); -// -// for (const auto& spec : g_writer.spectraWriters()) { -// auto& species = local_domain->species[spec.species() - 1]; -// array_t dn3d { "dn3d", nx1_bins, nx2_bins, nx3_bins, n_bins }; -// auto dn3d_scatter = Kokkos::Experimental::create_scatter_view(dn3d); -// auto ux1 = species.ux1; -// auto ux2 = species.ux2; -// auto ux3 = species.ux3; -// auto i1 = species.i1; -// auto dx1 = species.dx1; -// // adeep_copy; -// decltype(i1) i2; -// decltype(i1) i3; -// decltype(dx1) dx2; -// decltype(dx1) dx3; -// if constexpr (D == Dim::_2D or -// D == Dim::_3D) { // only pick x2 if simulation is in 2D -// i2 = species.i2; -// dx2 = species.dx2; -// } -// if constexpr (D == Dim::_3D) { -// i3 = species.i3; -// dx3 = species.dx3; -// } -// auto weight = species.weight; -// auto tag = species.tag; -// const auto is_massive = species.mass() > 0.0f; -// Kokkos::parallel_for( -// "ComputeSpectra", -// species.rangeActiveParticles(), -// Lambda(prtlidx_t p) { -// if (tag(p) != ParticleTag::alive) { -// return; -// } -// -// coord_t x_Cd { ZERO }; -// if constexpr (D == Dim::_1D or D == Dim::_2D or D == Dim::_3D) { -// x_Cd[0] = static_cast(i1(p)) + static_cast(dx1(p)); -// } -// if constexpr (D == Dim::_2D or D == Dim::_3D) { -// x_Cd[1] = static_cast(i2(p)) + static_cast(dx2(p)); -// } -// if constexpr (D == Dim::_3D) { -// x_Cd[2] = static_cast(i3(p)) + static_cast(dx3(p)); -// } -// coord_t x_Ph { ZERO }; -// metric.template convert(x_Cd, x_Ph); -// -// real_t en; -// if (is_massive) { -// en = U2GAMMA(ux1(p), ux2(p), ux3(p)) - ONE; -// } else { -// en = NORM(ux1(p), ux2(p), ux3(p)); -// } -// if (log_bins) { -// en = math::log10(en); -// } -// std::size_t e_ind = 0; -// if (en <= e_min) { -// e_ind = 0; -// } else if (en >= e_max) { -// e_ind = n_bins - 1; -// } else { -// e_ind = static_cast( -// static_cast(n_bins) * (en - e_min) / (e_max - e_min)); -// } -// -// std::size_t x1_ind = 0; -// if (x_Ph[0] <= x1_min) { -// x1_ind = 0; -// } else if (x_Ph[0] >= x1_max) { -// x1_ind = nx1_bins - 1; -// } else { -// x1_ind = static_cast(static_cast(nx1_bins) * -// (x_Ph[0] - x1_min) / -// (x1_max - x1_min)); -// } -// -// std::size_t x2_ind = 0; -// if constexpr (D == Dim::_2D or -// D == Dim::_3D) { // only pick x2 if simulation is in 2D -// if (x_Ph[1] <= x2_min) { -// x2_ind = 0; -// } else if (x_Ph[1] >= x2_max) { -// x2_ind = nx2_bins - 1; -// } else { -// x2_ind = static_cast(static_cast(nx2_bins) * -// (x_Ph[1] - x2_min) / -// (x2_max - x2_min)); -// } -// } -// std::size_t x3_ind = 0; -// if constexpr (D == Dim::_3D) { // only pick x3 if simulation is in 3D -// -// if (x_Ph[2] <= x3_min) { -// x3_ind = 0; -// } else if (x_Ph[2] >= x3_max) { -// x3_ind = nx3_bins - 1; -// } else { -// x3_ind = static_cast(static_cast(nx3_bins) * -// (x_Ph[2] - x3_min) / -// (x3_max - x3_min)); -// } -// } -// -// // now I want to save the nx_bins contained in each rank -// // can save array of x1_inds which are saved? -// // maybe I can ask what local x_min and x_max are for this rank, pass that to writeSpectrum3D -// -// auto dn3d_acc = dn3d_scatter.access(); -// dn3d_acc(x1_ind, x2_ind, x3_ind, e_ind) += weight(p); -// }); -// Kokkos::Experimental::contribute(dn3d, dn3d_scatter); -// g_writer.writeSpectrum3D(dn3d, spec.name()); -// } -// g_writer.writeSpectrumBins(energy, "sEbn"); -// g_writer.endWriting(WriteMode::Spectra3D); -// } diff --git a/src/output/writer.cpp b/src/output/writer.cpp index 69c509fb5..568143f9d 100644 --- a/src/output/writer.cpp +++ b/src/output/writer.cpp @@ -200,14 +200,34 @@ namespace out { } } - void Writer::defineSpectraOutputs(const std::vector& specs) { + void Writer::defineSpectraOutputs(const std::vector& specs, + const std::vector& num_spatial_bins) { m_spectra_writers.clear(); for (const auto& s : specs) { m_spectra_writers.emplace_back(s); } m_io.DefineVariable("sEbn", {}, {}, { adios2::UnknownDim }); + const auto spatial_binning_enabled = std::any_of(num_spatial_bins.begin(), + num_spatial_bins.end(), + [](const auto& n) { + return n != 1u; + }); + const auto nspec_dims = spatial_binning_enabled ? num_spatial_bins.size() + 1u + : 1u; for (const auto& sp : m_spectra_writers) { - m_io.DefineVariable(sp.name(), {}, {}, { adios2::UnknownDim }); + m_io.DefineVariable(sp.name(), + {}, + {}, + adios2::Dims(nspec_dims, adios2::UnknownDim)); + } + if (spatial_binning_enabled) { + const auto dim = num_spatial_bins.size(); + for (auto d { 0u }; d < dim; ++d) { + m_io.DefineVariable("sX" + std::to_string(d + 1) + "bn", + {}, + {}, + { adios2::UnknownDim }); + } } } @@ -383,6 +403,47 @@ namespace out { m_keepalive.emplace_back(array_h); } + template + void PutSpectrumSpatial(adios2::IO& io, + adios2::Engine& writer, + std::vector& keepalive, + const std::string& varname, + const HostView& counts_h) { + auto var = io.InquireVariable(varname); + + adios2::Dims start(N, 0u), count(N, 0u), zeros(N, 0u); + auto ntot { 1ul }; + for (auto d { 0u }; d < N; ++d) { + count[d] = counts_h.extent(d); + ntot *= counts_h.extent(d); + } + +#if defined(MPI_ENABLED) + HostView counts_h_all { "counts_h_all", counts_h.layout() }; + int rank; + MPI_Comm_rank(MPI_COMM_WORLD, &rank); + MPI_Reduce(counts_h.data(), + counts_h_all.data(), + static_cast(ntot), + mpi::get_type(), + MPI_SUM, + MPI_ROOT_RANK, + MPI_COMM_WORLD); + if (rank == MPI_ROOT_RANK) { + var.SetSelection(adios2::Box(start, count)); + writer.Put(var, counts_h_all.data(), adios2::Mode::Deferred); + keepalive.emplace_back(counts_h_all); + } else { + var.SetSelection(adios2::Box(start, zeros)); + writer.Put(var, nullptr, adios2::Mode::Sync); + } +#else + var.SetSelection(adios2::Box(start, count)); + writer.Put(var, counts_h.data(), adios2::Mode::Deferred); + keepalive.emplace_back(counts_h); +#endif + } + void Writer::writeSpectrum(const array_t& counts, const std::string& varname) { auto var = m_io.InquireVariable(varname); @@ -416,75 +477,130 @@ namespace out { #endif } -// spectrum3D, broken up into separate sub-domains - void Writer::writeSpectrum3D(const array_t& counts3D, - const std::string& varname) { - // need to include the rank specific coordinates contained here - std::string varname3d = varname + "_3D"; - //auto var = m_io.InquireVariable(varname); - auto counts3D_h = Kokkos::create_mirror_view(counts3D); - // copy to host - Kokkos::deep_copy(counts3D_h, counts3D); -#if defined(MPI_ENABLED) - array_t counts3D_all { "counts3D_all", counts3D.extent(0), counts3D.extent(1), counts3D.extent(2), counts3D.extent(3) }; - //auto counts_h_all = Kokkos::create_mirror_view(counts_all); - int rank; - int size; - MPI_Comm_rank(MPI_COMM_WORLD, &rank); - MPI_Comm_size(MPI_COMM_WORLD, &size); - - auto counts3D_h_all = Kokkos::create_mirror_view(counts3D_all); - - - MPI_Allreduce(counts3D_h.data(), - counts3D_h_all.data(), - counts3D_h.extent(0)*counts3D_h.extent(1)*counts3D_h.extent(2)*counts3D_h.extent(3),// size so probably need to multiply counts_h.extent(0)*counts_h.extent(1)*counts_h.extent(2) - mpi::get_type(), - MPI_SUM, - MPI_COMM_WORLD); // size - //Kokkos::deep_copy() -#else - int rank = 0; - int size = 1; - auto counts3D_h_all = counts3D_h; -#endif - using Shape = std::vector; - - Shape shape = {counts3D.extent(0), counts3D.extent(1), counts3D.extent(2), counts3D.extent(3)}; // global size of counts3D_all - Shape start, count; // per-rank selection - - // split the domain along the x1 direction across the different ranks - auto split_x1 = [&](std::size_t n, int this_rank, int nranks) - { - // base number of cells to allocate to each rank - std::size_t base = n / nranks; - // remainder that do not fit neatly in one rank - std::size_t rem = n % nranks; - // number of cells to allocate to the rank including the remainder - std::size_t nloc = base + (this_rank <(int)rem ? 1 : 0); // allocate one more if this rank is less that the remainder - //offset to allocate - std::size_t off = base * this_rank + std::min(this_rank, rem); - - return std::pair{nloc, off}; - }; - - auto [nloc0, off0] = split_x1(counts3D.extent(0), rank, size); - start = {off0, 0, 0, 0}; - count = {nloc0, counts3D.extent(1), counts3D.extent(2), counts3D.extent(3)}; - - auto var = m_io.InquireVariable(varname3d); - if (!var) - { - var = m_io.DefineVariable(varname3d, shape, start, count, adios2::ConstantDims); + template + void Writer::writeSpectrumSpatial(const nddata_t& counts, + const std::string& varname) { + static_assert(N >= 2 and N <= 4, "writeSpectrumSpatial: N must be 2, 3 or 4"); + // host-resident copy, layout inherited from `counts` + auto counts_h = Kokkos::create_mirror_view_and_copy(Kokkos::HostSpace(), + counts); + using counts_h_t = decltype(counts_h); + using layout_t = typename counts_h_t::array_layout; + if constexpr (std::is_same::value) { + PutSpectrumSpatial(m_io, m_writer, m_keepalive, varname, counts_h); + } else { + // ADIOS2 reads the raw buffer as row-major: remap on the host + using counts_rm_t = + Kokkos::View; + counts_rm_t counts_rm {}; + if constexpr (N == 2) { + counts_rm = counts_rm_t { "counts_rm", + counts_h.extent(0), + counts_h.extent(1) }; + } else if constexpr (N == 3) { + counts_rm = counts_rm_t { "counts_rm", + counts_h.extent(0), + counts_h.extent(1), + counts_h.extent(2) }; + } else { + counts_rm = counts_rm_t { "counts_rm", + counts_h.extent(0), + counts_h.extent(1), + counts_h.extent(2), + counts_h.extent(3) }; + } + Kokkos::deep_copy(counts_rm, counts_h); + PutSpectrumSpatial(m_io, m_writer, m_keepalive, varname, counts_rm); } - - //auto var = m_io.DefineVariable(varname3d, shape, start, count, adios2::ConstantDims); - - auto counts3D_h_all_slab = Kokkos::subview(counts3D_h_all, Kokkos::make_pair(off0, off0+nloc0), Kokkos::ALL(), Kokkos::ALL(), Kokkos::ALL()); - - m_writer.Put(var, counts3D_h_all_slab, adios2::Mode::Sync); } + // // spectrum3D, broken up into separate sub-domains + // void Writer::writeSpectrum3D(const array_t& counts3D, + // const std::string& varname) { + // // need to include the rank specific coordinates contained here + // std::string varname3d = varname + "_3D"; + // // auto var = m_io.InquireVariable(varname); + // auto counts3D_h = Kokkos::create_mirror_view(counts3D); + // // copy to host + // Kokkos::deep_copy(counts3D_h, counts3D); + // #if defined(MPI_ENABLED) + // array_t counts3D_all { "counts3D_all", + // counts3D.extent(0), + // counts3D.extent(1), + // counts3D.extent(2), + // counts3D.extent(3) }; + // // auto counts_h_all = Kokkos::create_mirror_view(counts_all); + // int rank; + // int size; + // MPI_Comm_rank(MPI_COMM_WORLD, &rank); + // MPI_Comm_size(MPI_COMM_WORLD, &size); + // + // auto counts3D_h_all = Kokkos::create_mirror_view(counts3D_all); + // + // MPI_Allreduce( + // counts3D_h.data(), + // counts3D_h_all.data(), + // counts3D_h.extent(0) * counts3D_h.extent(1) * counts3D_h.extent(2) * + // counts3D_h.extent( + // 3), // size so probably need to multiply counts_h.extent(0)*counts_h.extent(1)*counts_h.extent(2) + // mpi::get_type(), + // MPI_SUM, + // MPI_COMM_WORLD); // size + // // Kokkos::deep_copy() + // #else + // int rank = 0; + // int size = 1; + // auto counts3D_h_all = counts3D_h; + // #endif + // using Shape = std::vector; + // + // Shape shape = { counts3D.extent(0), + // counts3D.extent(1), + // counts3D.extent(2), + // counts3D.extent(3) }; // global size of counts3D_all + // Shape start, count; // per-rank selection + // + // // split the domain along the x1 direction across the different ranks + // auto split_x1 = [&](std::size_t n, int this_rank, int nranks) { + // // base number of cells to allocate to each rank + // std::size_t base = n / nranks; + // // remainder that do not fit neatly in one rank + // std::size_t rem = n % nranks; + // // number of cells to allocate to the rank including the remainder + // std::size_t nloc = base + + // (this_rank < (int)rem + // ? 1 + // : 0); // allocate one more if this rank is less that the remainder + // // offset to allocate + // std::size_t off = base * this_rank + std::min(this_rank, rem); + // + // return std::pair { nloc, off }; + // }; + // + // auto [nloc0, off0] = split_x1(counts3D.extent(0), rank, size); + // start = { off0, 0, 0, 0 }; + // count = { nloc0, counts3D.extent(1), counts3D.extent(2), counts3D.extent(3) }; + // + // auto var = m_io.InquireVariable(varname3d); + // if (!var) { + // var = m_io.DefineVariable(varname3d, + // shape, + // start, + // count, + // adios2::ConstantDims); + // } + // + // // auto var = m_io.DefineVariable(varname3d, shape, start, count, adios2::ConstantDims); + // + // auto counts3D_h_all_slab = Kokkos::subview(counts3D_h_all, + // Kokkos::make_pair(off0, off0 + nloc0), + // Kokkos::ALL(), + // Kokkos::ALL(), + // Kokkos::ALL()); + // + // m_writer.Put(var, counts3D_h_all_slab, adios2::Mode::Sync); + // } + void Writer::writeSpectrumBins(const array_t& e_bins, const std::string& varname) { auto var = m_io.InquireVariable(varname); @@ -567,8 +683,6 @@ namespace out { mode_str = "particles"; } else if (write_mode == WriteMode::Spectra) { mode_str = "spectra"; - } else if (write_mode == WriteMode::Spectra3D) { - mode_str = "spectra3D"; } else { raise::Fatal("Unknown write mode", HERE); } @@ -638,4 +752,12 @@ namespace out { WRITE_FIELD(Dim::_3D, 6) #undef WRITE_FIELD +#define WRITE_SPECTRUM_SPATIAL(N) \ + template void Writer::writeSpectrumSpatial(const nddata_t&, \ + const std::string&); + WRITE_SPECTRUM_SPATIAL(2) + WRITE_SPECTRUM_SPATIAL(3) + WRITE_SPECTRUM_SPATIAL(4) +#undef WRITE_SPECTRUM_SPATIAL + } // namespace out diff --git a/src/output/writer.h b/src/output/writer.h index 18193eaad..6e9a139d2 100644 --- a/src/output/writer.h +++ b/src/output/writer.h @@ -139,7 +139,11 @@ namespace out { npart_t, const std::string&); void writeSpectrum(const array_t&, const std::string&); - void writeSpectrum3D(const array_t&, const std::string&); + + template + void writeSpectrumSpatial(const nddata_t&, + const std::string&); // @TODO implement for N = 2, 3, 4 + void writeSpectrumBins(const array_t&, const std::string&); void beginWriting(WriteModeTags, timestep_t, simtime_t); From a94afa584600dbf69ec21caa9e7f5692426ab49a Mon Sep 17 00:00:00 2001 From: haykh Date: Tue, 15 Sep 2026 13:52:22 -0400 Subject: [PATCH 087/125] minor bugs --- src/framework/parameters/output.cpp | 11 ++++++----- src/kernels/particle_energy_distribution.hpp | 2 ++ 2 files changed, 8 insertions(+), 5 deletions(-) diff --git a/src/framework/parameters/output.cpp b/src/framework/parameters/output.cpp index 7591895a1..d2a6a9c44 100644 --- a/src/framework/parameters/output.cpp +++ b/src/framework/parameters/output.cpp @@ -171,11 +171,12 @@ namespace ntt { "num_energy_bins", defaults::output::spec_num_e_bins); } - spectra_num_spatial_bins = toml::find_or(toml_data, - "output", - "spectra", - "num_spatial_bins", - std::vector { 1, 1, 1 }); + spectra_num_spatial_bins = toml::find_or>( + toml_data, + "output", + "spectra", + "num_spatial_bins", + std::vector { 1, 1, 1 }); if (spectra_num_spatial_bins->size() < static_cast(dim)) { raise::Error("`output.spectra.num_spatial_bins` must have at least " + std::to_string(static_cast(dim)) + " entries", diff --git a/src/kernels/particle_energy_distribution.hpp b/src/kernels/particle_energy_distribution.hpp index d1590480b..29e53e60e 100644 --- a/src/kernels/particle_energy_distribution.hpp +++ b/src/kernels/particle_energy_distribution.hpp @@ -189,6 +189,7 @@ namespace kernel { nmin_i(1), dncells_i(1), nbins_i(1)); + auto dn_acc = dn_scatter.access(); dn_acc(i1_ind, i2_ind, e_ind) += particles.weight(p); } else if constexpr (D == Dim::_3D) { const auto i1_ind = SpatialBinIndex(static_cast(particles.i1(p)) + @@ -206,6 +207,7 @@ namespace kernel { nmin_i(2), dncells_i(2), nbins_i(2)); + auto dn_acc = dn_scatter.access(); dn_acc(i1_ind, i2_ind, i3_ind, e_ind) += particles.weight(p); } else { raise::KernelError(HERE, "invalid dimension"); From aebfdcc3e9c92aa7ab388dab7f3abb03f4611880 Mon Sep 17 00:00:00 2001 From: haykh Date: Tue, 15 Sep 2026 16:36:44 -0400 Subject: [PATCH 088/125] minor bugfix --- src/kernels/particle_energy_distribution.hpp | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/src/kernels/particle_energy_distribution.hpp b/src/kernels/particle_energy_distribution.hpp index 29e53e60e..eba88bac6 100644 --- a/src/kernels/particle_energy_distribution.hpp +++ b/src/kernels/particle_energy_distribution.hpp @@ -90,7 +90,7 @@ namespace kernel { if (en <= energy_binning.e_min) { return 0u; } else if (en >= energy_binning.e_max) { - return energy_binning.n_bins; + return energy_binning.n_bins - 1u; } else { return static_cast(static_cast(energy_binning.n_bins) * (en - energy_binning.e_min) / @@ -161,7 +161,7 @@ namespace kernel { } else if (ni >= dncells * static_cast(nbins)) { return nbins - 1u; } else { - return static_cast(static_cast(nbins) * ni / dncells); + return static_cast(ni / dncells); } } From 56b174c9ffe817ec26dd9a3cc8c728e4546db767 Mon Sep 17 00:00:00 2001 From: haykh Date: Tue, 15 Sep 2026 17:59:50 -0400 Subject: [PATCH 089/125] rm old spec3d --- src/output/writer.cpp | 87 ------------------------------------------- 1 file changed, 87 deletions(-) diff --git a/src/output/writer.cpp b/src/output/writer.cpp index 568143f9d..89410acc3 100644 --- a/src/output/writer.cpp +++ b/src/output/writer.cpp @@ -514,93 +514,6 @@ namespace out { } } - // // spectrum3D, broken up into separate sub-domains - // void Writer::writeSpectrum3D(const array_t& counts3D, - // const std::string& varname) { - // // need to include the rank specific coordinates contained here - // std::string varname3d = varname + "_3D"; - // // auto var = m_io.InquireVariable(varname); - // auto counts3D_h = Kokkos::create_mirror_view(counts3D); - // // copy to host - // Kokkos::deep_copy(counts3D_h, counts3D); - // #if defined(MPI_ENABLED) - // array_t counts3D_all { "counts3D_all", - // counts3D.extent(0), - // counts3D.extent(1), - // counts3D.extent(2), - // counts3D.extent(3) }; - // // auto counts_h_all = Kokkos::create_mirror_view(counts_all); - // int rank; - // int size; - // MPI_Comm_rank(MPI_COMM_WORLD, &rank); - // MPI_Comm_size(MPI_COMM_WORLD, &size); - // - // auto counts3D_h_all = Kokkos::create_mirror_view(counts3D_all); - // - // MPI_Allreduce( - // counts3D_h.data(), - // counts3D_h_all.data(), - // counts3D_h.extent(0) * counts3D_h.extent(1) * counts3D_h.extent(2) * - // counts3D_h.extent( - // 3), // size so probably need to multiply counts_h.extent(0)*counts_h.extent(1)*counts_h.extent(2) - // mpi::get_type(), - // MPI_SUM, - // MPI_COMM_WORLD); // size - // // Kokkos::deep_copy() - // #else - // int rank = 0; - // int size = 1; - // auto counts3D_h_all = counts3D_h; - // #endif - // using Shape = std::vector; - // - // Shape shape = { counts3D.extent(0), - // counts3D.extent(1), - // counts3D.extent(2), - // counts3D.extent(3) }; // global size of counts3D_all - // Shape start, count; // per-rank selection - // - // // split the domain along the x1 direction across the different ranks - // auto split_x1 = [&](std::size_t n, int this_rank, int nranks) { - // // base number of cells to allocate to each rank - // std::size_t base = n / nranks; - // // remainder that do not fit neatly in one rank - // std::size_t rem = n % nranks; - // // number of cells to allocate to the rank including the remainder - // std::size_t nloc = base + - // (this_rank < (int)rem - // ? 1 - // : 0); // allocate one more if this rank is less that the remainder - // // offset to allocate - // std::size_t off = base * this_rank + std::min(this_rank, rem); - // - // return std::pair { nloc, off }; - // }; - // - // auto [nloc0, off0] = split_x1(counts3D.extent(0), rank, size); - // start = { off0, 0, 0, 0 }; - // count = { nloc0, counts3D.extent(1), counts3D.extent(2), counts3D.extent(3) }; - // - // auto var = m_io.InquireVariable(varname3d); - // if (!var) { - // var = m_io.DefineVariable(varname3d, - // shape, - // start, - // count, - // adios2::ConstantDims); - // } - // - // // auto var = m_io.DefineVariable(varname3d, shape, start, count, adios2::ConstantDims); - // - // auto counts3D_h_all_slab = Kokkos::subview(counts3D_h_all, - // Kokkos::make_pair(off0, off0 + nloc0), - // Kokkos::ALL(), - // Kokkos::ALL(), - // Kokkos::ALL()); - // - // m_writer.Put(var, counts3D_h_all_slab, adios2::Mode::Sync); - // } - void Writer::writeSpectrumBins(const array_t& e_bins, const std::string& varname) { auto var = m_io.InquireVariable(varname); From 6697b95e6c3eba98dbeaedb933d696891db193e4 Mon Sep 17 00:00:00 2001 From: haykh Date: Tue, 15 Sep 2026 18:03:58 -0400 Subject: [PATCH 090/125] input for shock fixed back --- pgens/shock/shock.toml | 43 +++++++++++++++++------------------------- 1 file changed, 17 insertions(+), 26 deletions(-) diff --git a/pgens/shock/shock.toml b/pgens/shock/shock.toml index 37515078f..8d3270a74 100644 --- a/pgens/shock/shock.toml +++ b/pgens/shock/shock.toml @@ -40,47 +40,38 @@ maxnpart = 2e7 [setup] - drift_ux = 0.15 # speed towards the wall [c] - temperature = 0.168 # temperature of maxwell distribution [kB T / (m_i c^2)] - temperature_ratio = 1.0 # temperature ratio of electrons to protons - Bmag = 1.0 # magnetic field strength as fraction of magnetisation - Btheta = 63.0 # magnetic field angle in the plane - Bphi = 0.0 # magnetic field angle out of plane - filling_fraction = 0.99 # fraction of the shock piston filled with plasma - injector_velocity = 0.0 # speed of injector [c] - injection_start = 0.0 # start time of moving injector + drift_ux = 0.15 # speed towards the wall [c] + temperature = 0.168 # temperature of maxwell distribution [kB T / (m_i c^2)] + temperature_ratio = 1.0 # temperature ratio of electrons to protons + Bmag = 1.0 # magnetic field strength as fraction of magnetisation + Btheta = 63.0 # magnetic field angle in the plane + Bphi = 0.0 # magnetic field angle out of plane + filling_fraction = 0.99 # fraction of the shock piston filled with plasma + injector_velocity = 0.0 # speed of injector [c] + injection_start = 0.0 # start time of moving injector injection_frequency = 100 [output] interval_time = 1000.0 format = "BPFile" - + [output.fields] quantities = ["N", "B", "E"] [output.particles] enable = true - stride = 10 + stride = 10 [output.spectra] - enable = true - e_min = 1e-2 - e_max = 1e1 - log_bins = true - - [output.spectra3D] - enable = true - e_min = 1e-2 - e_max = 1e1 - log_bins = true - nx1 = 40 - nx2 = 3 - nx3 = 1 + enable = true + e_min = 1e-2 + e_max = 1e1 + log_bins = true + num_spatial_bins = [40, 3, 1] [diagnostics] log_level = "WARNING" - blocking_timers = true [checkpoint] interval = 10000 - keep = 1 \ No newline at end of file + keep = 2 From 87bf6ad608d0550a625b630f866f95cb39106b42 Mon Sep 17 00:00:00 2001 From: LudwigBoess Date: Wed, 16 Sep 2026 20:27:38 +0000 Subject: [PATCH 091/125] refactor particle-dependent external force via `HasExtFx*WithIndex` to keep backwards compatible `HasExtFx*` --- examples/external_fields/pgen.hpp | 3 +- src/global/traits/archetypes.h | 46 +++++++++------- src/global/traits/policies.h | 11 ++-- src/kernels/pushers/sr.hpp | 36 +++++++++---- tests/archetypes/pgen.cpp | 3 +- tests/global/traits_archetypes.cpp | 34 ++++++++++-- tests/global/traits_policies.cpp | 9 +++- tests/kernels/ext_force.cpp | 84 +++++++++++++++++++++++------- 8 files changed, 163 insertions(+), 63 deletions(-) diff --git a/examples/external_fields/pgen.hpp b/examples/external_fields/pgen.hpp index 30fb2473d..ac4f6faed 100644 --- a/examples/external_fields/pgen.hpp +++ b/examples/external_fields/pgen.hpp @@ -40,8 +40,7 @@ namespace user { */ // f_ext: external force-field (acceleration): - Inline auto fx1(const coord_t&, const ntt::ParticleArrays&, prtlidx_t) const - -> real_t { + Inline auto fx1(const coord_t&) const -> real_t { return (sp % 2u == 0u) ? -HALF : HALF; } diff --git a/src/global/traits/archetypes.h b/src/global/traits/archetypes.h index c5582e74b..0dfced45c 100644 --- a/src/global/traits/archetypes.h +++ b/src/global/traits/archetypes.h @@ -4,7 +4,9 @@ * @implements * - EnrgDistClass<> - checks if a class can be used as an energy distribution * - SpatialDistClass<> - checks if a class can be used as a spatial distribution - * - traits::fieldsetter::HasFx1, ::HasFx2, ::HasFx3 - checks for particle-aware F functions fx*(x, particles, p) + * - traits::fieldsetter::HasFx1, ::HasFx2, ::HasFx3 - checks for F functions in field setter class + * - traits::fieldsetter::HasFx1WithIndex, ::HasFx2WithIndex, ::HasFx3WithIndex + * - checks for F functions taking an additional particle index * - traits::fieldsetter::HasEx1, ::HasEx2, ::HasEx3 - checks for E functions in field setter class * - traits::fieldsetter::HasBx1, ::HasBx2, ::HasBx3 - checks for B functions in field setter class * - traits::fieldsetter::HasDx1, ::HasDx2, ::HasDx3 - checks for D functions in field setter class @@ -26,10 +28,6 @@ #include -namespace ntt { - struct ParticleArrays; -} // namespace ntt - template concept EnrgDistClass = requires(const ED& edist, const coord_t& x_Ph, @@ -54,27 +52,35 @@ concept SpatialDistClass = SimpleSpatialDistClass or namespace traits::fieldsetter { template - concept HasFx1 = requires(const T& t, - const coord_t& x_Ph, - const ntt::ParticleArrays& prtls, - prtlidx_t p) { - { t.fx1(x_Ph, prtls, p) } -> std::convertible_to; + concept HasFx1 = requires(const T& t, const coord_t& x_Ph) { + { t.fx1(x_Ph) } -> std::convertible_to; + }; + + template + concept HasFx2 = requires(const T& t, const coord_t& x_Ph) { + { t.fx2(x_Ph) } -> std::convertible_to; + }; + + template + concept HasFx3 = requires(const T& t, const coord_t& x_Ph) { + { t.fx3(x_Ph) } -> std::convertible_to; + }; + + // force components which additionally depend on the particle index + // (e.g., to access per-particle properties captured by the setter itself) + template + concept HasFx1WithIndex = requires(const T& t, const coord_t& x_Ph, prtlidx_t p) { + { t.fx1(x_Ph, p) } -> std::convertible_to; }; template - concept HasFx2 = requires(const T& t, - const coord_t& x_Ph, - const ntt::ParticleArrays& prtls, - prtlidx_t p) { - { t.fx2(x_Ph, prtls, p) } -> std::convertible_to; + concept HasFx2WithIndex = requires(const T& t, const coord_t& x_Ph, prtlidx_t p) { + { t.fx2(x_Ph, p) } -> std::convertible_to; }; template - concept HasFx3 = requires(const T& t, - const coord_t& x_Ph, - const ntt::ParticleArrays& prtls, - prtlidx_t p) { - { t.fx3(x_Ph, prtls, p) } -> std::convertible_to; + concept HasFx3WithIndex = requires(const T& t, const coord_t& x_Ph, prtlidx_t p) { + { t.fx3(x_Ph, p) } -> std::convertible_to; }; template diff --git a/src/global/traits/policies.h b/src/global/traits/policies.h index 7f6d98985..a317fd687 100644 --- a/src/global/traits/policies.h +++ b/src/global/traits/policies.h @@ -109,10 +109,13 @@ namespace traits::extfields { template concept ExtFieldsPolicyClass = (::traits::fieldsetter::HasFx1 or ::traits::fieldsetter::HasFx2 or - ::traits::fieldsetter::HasFx3 or ::traits::fieldsetter::HasEx1 or - ::traits::fieldsetter::HasEx2 or ::traits::fieldsetter::HasEx3 or - ::traits::fieldsetter::HasBx1 or ::traits::fieldsetter::HasBx2 or - ::traits::fieldsetter::HasBx3) or + ::traits::fieldsetter::HasFx3 or + ::traits::fieldsetter::HasFx1WithIndex or + ::traits::fieldsetter::HasFx2WithIndex or + ::traits::fieldsetter::HasFx3WithIndex or + ::traits::fieldsetter::HasEx1 or ::traits::fieldsetter::HasEx2 or + ::traits::fieldsetter::HasEx3 or ::traits::fieldsetter::HasBx1 or + ::traits::fieldsetter::HasBx2 or ::traits::fieldsetter::HasBx3) or ::traits::extfields::IsNoPolicy; namespace traits::custom_prtl_update { diff --git a/src/kernels/pushers/sr.hpp b/src/kernels/pushers/sr.hpp index bc7d79f08..6d691d3df 100644 --- a/src/kernels/pushers/sr.hpp +++ b/src/kernels/pushers/sr.hpp @@ -67,11 +67,19 @@ namespace kernel::sr { using F = typename P::ExternalFieldsPolicy; static constexpr auto Atm = P::ApplyAtmosphere; - static constexpr auto D = M::Dim; - static constexpr auto HasExtFx1 = ::traits::fieldsetter::HasFx1; - static constexpr auto HasExtFx2 = ::traits::fieldsetter::HasFx2; - static constexpr auto HasExtFx3 = ::traits::fieldsetter::HasFx3; - static constexpr auto HasExtForce = HasExtFx1 or HasExtFx2 or HasExtFx3; + static constexpr auto D = M::Dim; + static constexpr auto HasExtFx1 = ::traits::fieldsetter::HasFx1; + static constexpr auto HasExtFx2 = ::traits::fieldsetter::HasFx2; + static constexpr auto HasExtFx3 = ::traits::fieldsetter::HasFx3; + static constexpr auto HasExtFx1WithIndex = + ::traits::fieldsetter::HasFx1WithIndex; + static constexpr auto HasExtFx2WithIndex = + ::traits::fieldsetter::HasFx2WithIndex; + static constexpr auto HasExtFx3WithIndex = + ::traits::fieldsetter::HasFx3WithIndex; + static constexpr auto HasExtForce = HasExtFx1 or HasExtFx2 or HasExtFx3 or + HasExtFx1WithIndex or + HasExtFx2WithIndex or HasExtFx3WithIndex; static constexpr auto HasExtEx1 = ::traits::fieldsetter::HasEx1; static constexpr auto HasExtEx2 = ::traits::fieldsetter::HasEx2; static constexpr auto HasExtEx3 = ::traits::fieldsetter::HasEx3; @@ -1430,14 +1438,20 @@ namespace kernel::sr { requires(Atm or HasExtForce) { real_t f_x1 = ZERO, f_x2 = ZERO, f_x3 = ZERO; - if constexpr (HasExtFx1) { - f_x1 = policies.external_fields_policy.fx1(xp_Ph, particles, p); + if constexpr (HasExtFx1WithIndex) { + f_x1 = policies.external_fields_policy.fx1(xp_Ph, p); + } else if constexpr (HasExtFx1) { + f_x1 = policies.external_fields_policy.fx1(xp_Ph); } - if constexpr (HasExtFx2) { - f_x2 = policies.external_fields_policy.fx2(xp_Ph, particles, p); + if constexpr (HasExtFx2WithIndex) { + f_x2 = policies.external_fields_policy.fx2(xp_Ph, p); + } else if constexpr (HasExtFx2) { + f_x2 = policies.external_fields_policy.fx2(xp_Ph); } - if constexpr (HasExtFx3) { - f_x3 = policies.external_fields_policy.fx3(xp_Ph, particles, p); + if constexpr (HasExtFx3WithIndex) { + f_x3 = policies.external_fields_policy.fx3(xp_Ph, p); + } else if constexpr (HasExtFx3) { + f_x3 = policies.external_fields_policy.fx3(xp_Ph); } if constexpr (Atm) { if constexpr (D == Dim::_1D or D == Dim::_2D or D == Dim::_3D) { diff --git a/tests/archetypes/pgen.cpp b/tests/archetypes/pgen.cpp index fb428493b..d95e9cf9c 100644 --- a/tests/archetypes/pgen.cpp +++ b/tests/archetypes/pgen.cpp @@ -25,8 +25,7 @@ struct CustomFieldsetter { template struct ExtForce { - Inline auto fx1(const coord_t&, const ntt::ParticleArrays&, prtlidx_t) const - -> real_t { + Inline auto fx1(const coord_t&) const -> real_t { return ZERO; } diff --git a/tests/global/traits_archetypes.cpp b/tests/global/traits_archetypes.cpp index 575b7abe5..70ce3e9e2 100644 --- a/tests/global/traits_archetypes.cpp +++ b/tests/global/traits_archetypes.cpp @@ -89,15 +89,29 @@ struct WithDx1Dx2Dx3 { }; struct WithFx1Fx2Fx3 { - real_t fx1(const coord_t&, const ntt::ParticleArrays&, prtlidx_t) const { + real_t fx1(const coord_t&) const { return ZERO; } - real_t fx2(const coord_t&, const ntt::ParticleArrays&, prtlidx_t) const { + real_t fx2(const coord_t&) const { return ZERO; } - real_t fx3(const coord_t&, const ntt::ParticleArrays&, prtlidx_t) const { + real_t fx3(const coord_t&) const { + return ZERO; + } +}; + +struct WithIndexedFx1Fx2Fx3 { + real_t fx1(const coord_t&, prtlidx_t) const { + return ZERO; + } + + real_t fx2(const coord_t&, prtlidx_t) const { + return ZERO; + } + + real_t fx3(const coord_t&, prtlidx_t) const { return ZERO; } }; @@ -135,6 +149,20 @@ static_assert(traits::fieldsetter::HasFx1); static_assert(traits::fieldsetter::HasFx2); static_assert(traits::fieldsetter::HasFx3); +// the indexed variants are a separate, opt-in signature: neither form +// satisfies the other's trait +static_assert( + traits::fieldsetter::HasFx1WithIndex); +static_assert( + traits::fieldsetter::HasFx2WithIndex); +static_assert( + traits::fieldsetter::HasFx3WithIndex); +static_assert( + not traits::fieldsetter::HasFx1WithIndex); +static_assert( + not traits::fieldsetter::HasFx1); +static_assert(not traits::fieldsetter::HasFx1WithIndex); + // conditional variants require 3-arg signature returning Kokkos::pair static_assert( traits::fieldsetter::HasConditionalEx1); diff --git a/tests/global/traits_policies.cpp b/tests/global/traits_policies.cpp index 82ceb7e12..7f212dab3 100644 --- a/tests/global/traits_policies.cpp +++ b/tests/global/traits_policies.cpp @@ -105,7 +105,13 @@ static_assert(not EmissionPolicyClass); // --- ExtFieldsPolicyClass --- struct WithFx1 { - real_t fx1(const coord_t&, const ntt::ParticleArrays&, prtlidx_t) const { + real_t fx1(const coord_t&) const { + return ZERO; + } +}; + +struct WithIndexedFx1 { + real_t fx1(const coord_t&, prtlidx_t) const { return ZERO; } }; @@ -113,6 +119,7 @@ struct WithFx1 { struct Empty {}; static_assert(ExtFieldsPolicyClass); +static_assert(ExtFieldsPolicyClass); static_assert(not ExtFieldsPolicyClass); // --- CustomParticleUpdatePolicyClass with a real updater --- diff --git a/tests/kernels/ext_force.cpp b/tests/kernels/ext_force.cpp index d9e8203a5..0a705f19b 100644 --- a/tests/kernels/ext_force.cpp +++ b/tests/kernels/ext_force.cpp @@ -24,6 +24,7 @@ #include #include #include +#include #include using namespace ntt; @@ -51,21 +52,25 @@ void put_value(array_t& arr, T v, prtlidx_t p) { Kokkos::deep_copy(arr, h); } +// per-particle weight given to the two test particles; the indexed force +// setter reads it back from the particle arrays and scales the force by it +auto weight_of(prtlidx_t p) -> real_t { + return (p == 0) ? ONE : TWO; +} + +// force setter depending only on the position struct Force { Force(real_t force) : force { force } {} - Inline auto fx1(const coord_t&, const ntt::ParticleArrays&, prtlidx_t) const - -> real_t { + Inline auto fx1(const coord_t&) const -> real_t { return force * math::sin(ONE) * math::sin(ONE); } - Inline auto fx2(const coord_t&, const ntt::ParticleArrays&, prtlidx_t) const - -> real_t { + Inline auto fx2(const coord_t&) const -> real_t { return force * math::sin(ONE) * math::cos(ONE); } - Inline auto fx3(const coord_t&, const ntt::ParticleArrays&, prtlidx_t) const - -> real_t { + Inline auto fx3(const coord_t&) const -> real_t { return force * math::cos(ONE); } @@ -73,7 +78,31 @@ struct Force { const real_t force; }; -template +// force setter carrying its own copy of the particle arrays and reading +// per-particle properties through the particle index +struct IndexedForce { + IndexedForce(real_t force, const ParticleArrays& prtls) + : force { force } + , prtls { prtls } {} + + Inline auto fx1(const coord_t&, prtlidx_t p) const -> real_t { + return prtls.weight(p) * force * math::sin(ONE) * math::sin(ONE); + } + + Inline auto fx2(const coord_t&, prtlidx_t p) const -> real_t { + return prtls.weight(p) * force * math::sin(ONE) * math::cos(ONE); + } + + Inline auto fx3(const coord_t&, prtlidx_t p) const -> real_t { + return prtls.weight(p) * force * math::cos(ONE); + } + +private: + const real_t force; + const ParticleArrays prtls; +}; + +template void testPusher(const std::vector& res) { static_assert(M::Dim == 3); raise::ErrorIf(res.size() != M::Dim, "res.size() != M::Dim", HERE); @@ -144,6 +173,7 @@ void testPusher(const std::vector& res) { put_value(ux1, ux1_0, 0); put_value(ux2, ux2_0, 0); put_value(ux3, ux3_0, 0); + put_value(weight, weight_of(0), 0); put_value(tag, ParticleTag::alive, 0); put_value(i1, (int)(x1_0), 1); @@ -155,12 +185,11 @@ void testPusher(const std::vector& res) { put_value(ux1, -ux1_0, 1); put_value(ux2, -ux2_0, 1); put_value(ux3, -ux3_0, 1); + put_value(weight, weight_of(1), 1); put_value(tag, ParticleTag::alive, 1); const real_t eps = std::is_same_v ? 1e-4 : 1e-6; - const auto ext_force = Force { f_mag }; - static plog::RollingFileAppender file_appender( "pusher_log.csv"); plog::init(plog::verbose, &file_appender); @@ -189,8 +218,18 @@ void testPusher(const std::vector& res) { pusher_arrays.ux2 = ux2; pusher_arrays.ux3 = ux3; pusher_arrays.phi = phi; + pusher_arrays.weight = weight; pusher_arrays.tag = tag; + using ext_force_t = std::conditional_t; + const auto ext_force = [&]() -> ext_force_t { + if constexpr (Indexed) { + return { f_mag, pusher_arrays }; + } else { + return { f_mag }; + } + }(); + const auto pusher_policy = ::kernel::sr::PusherPolicy, ::traits::emission::NoPolicy_t, @@ -256,12 +295,16 @@ void testPusher(const std::vector& res) { ux2_(1), ux3_(1)); + // the indexed setter scales the force by the particle weight + const real_t scale_0 = Indexed ? weight_of(0) : ONE; + const real_t scale_1 = Indexed ? weight_of(1) : ONE; + { - const real_t ux1_expect = ux1_0 + (time + dt) * f_mag * std::sin(ONE) * - std::sin(ONE); - const real_t ux2_expect = ux2_0 + (time + dt) * f_mag * std::sin(ONE) * - std::cos(ONE); - const real_t ux3_expect = ux3_0 + (time + dt) * f_mag * std::cos(ONE); + const real_t f = (time + dt) * f_mag * scale_0; + + const real_t ux1_expect = ux1_0 + f * std::sin(ONE) * std::sin(ONE); + const real_t ux2_expect = ux2_0 + f * std::sin(ONE) * std::cos(ONE); + const real_t ux3_expect = ux3_0 + f * std::cos(ONE); check_value(t, ux1_(0), ux1_expect, eps, "Particle #1 ux1"); check_value(t, ux2_(0), ux2_expect, eps, "Particle #1 ux2"); @@ -269,11 +312,11 @@ void testPusher(const std::vector& res) { } { - const real_t ux1_expect = -ux1_0 + (time + dt) * f_mag * std::sin(ONE) * - std::sin(ONE); - const real_t ux2_expect = -ux2_0 + (time + dt) * f_mag * std::sin(ONE) * - std::cos(ONE); - const real_t ux3_expect = -ux3_0 + (time + dt) * f_mag * std::cos(ONE); + const real_t f = (time + dt) * f_mag * scale_1; + + const real_t ux1_expect = -ux1_0 + f * std::sin(ONE) * std::sin(ONE); + const real_t ux2_expect = -ux2_0 + f * std::sin(ONE) * std::cos(ONE); + const real_t ux3_expect = -ux3_0 + f * std::cos(ONE); check_value(t, ux1_(1), ux1_expect, eps, "Particle #2 ux1"); check_value(t, ux2_(1), ux2_expect, eps, "Particle #2 ux2"); @@ -288,7 +331,8 @@ auto main(int argc, char* argv[]) -> int { try { using namespace ntt; - testPusher>({ 10, 10, 10 }); + testPusher, false>({ 10, 10, 10 }); + testPusher, true>({ 10, 10, 10 }); } catch (std::exception& e) { std::cerr << e.what() << '\n'; From 7d6bb4590ae7dd86e69aa909a22fd403884eec8e Mon Sep 17 00:00:00 2001 From: haykh Date: Mon, 21 Sep 2026 10:33:38 -0400 Subject: [PATCH 092/125] vendor report moved --- cmake/report.cmake | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/cmake/report.cmake b/cmake/report.cmake index 822d82650..e6ad4aaf8 100644 --- a/cmake/report.cmake +++ b/cmake/report.cmake @@ -146,7 +146,7 @@ printchoices( ON "${Green}" VENDOR_SORT_REPORT - 44) + 46) printchoices( "Debug mode" "DEBUG" @@ -234,6 +234,9 @@ string( " - DEVICES [${Magenta}Kokkos_ENABLE_***${ColorReset}]: " "${Kokkos_DEVICES}" "\n" + " " + ${VENDOR_SORT_REPORT} + "\n" " > Multi-node specs" " ${Dim}[requires mpi=ON]${ColorReset}" "\n" @@ -255,9 +258,6 @@ string( " " "- Deposit drift [${Magenta}team_policy_drift${ColorReset}]: " ${team_policy_drift} - "\n" - " " - ${VENDOR_SORT_REPORT} "\n") string( From bdc3172053fcc7aff8e2f20a71e9286c8cbbd28d Mon Sep 17 00:00:00 2001 From: hayk Date: Mon, 21 Sep 2026 08:16:49 -0700 Subject: [PATCH 093/125] perlmutter dep fixed --- dependencies.py | 31 +++++++++++++++++++------------ 1 file changed, 19 insertions(+), 12 deletions(-) diff --git a/dependencies.py b/dependencies.py index 8f39bde85..54adf55b2 100755 --- a/dependencies.py +++ b/dependencies.py @@ -8,7 +8,6 @@ from dataclasses import dataclass, field from typing import Callable, List, Optional, Tuple - # ============================ # colors: edit these # ============================ @@ -54,7 +53,7 @@ class Settings: # versions kokkos_version: str = "5.2.1" - adios2_version: str = "2.11.0" + adios2_version: str = "2.12.1" # options kokkos_backend: str = "cpu" @@ -135,7 +134,9 @@ def InstallKokkosScriptModfile(settings: Settings) -> tuple[str, str]: [f"module load {module}" for module in settings.module_loads] ) src_path = os.path.join(prefix, "src", "kokkos") - install_path = os.path.join(prefix, "kokkos", version, backend, arch.lower() if arch else "") + install_path = os.path.join( + prefix, "kokkos", version, backend, arch.lower() if arch else "" + ) if os.path.exists(install_path) and not settings.overwrite: raise FileExistsError( f"Kokkos install path {install_path} already exists and overwrite is disabled" @@ -300,7 +301,7 @@ def InstallNt2pyScript(settings: Settings) -> str: }, "stellar": {"module_loads": []}, "perlmutter": { - "module_loads": ["gpu/1.0"], + "module_loads": ["gpu/1.0", "python/3.14-26.8.1"], "kokkos_backend": "cuda", "kokkos_arch": "AMPERE80", "extra_kokkos_flags": [ @@ -313,17 +314,19 @@ def InstallNt2pyScript(settings: Settings) -> str: ], }, "lumi": { - "module_loads": ["PrgEnv-cray", "cray-mpich", "craype-accel-amd-gfx90a", "rocm"], + "module_loads": [ + "PrgEnv-cray", + "cray-mpich", + "craype-accel-amd-gfx90a", + "rocm", + ], "kokkos_backend": "hip", "kokkos_arch": "AMD_GFX90A", "extra_kokkos_flags": [ "CMAKE_CXX_COMPILER=hipcc", "AMDGPU_TARGETS=gfx90a", ], - "extra_adios2_flags": [ - "CMAKE_CXX_COMPILER=CC", - "CMAKE_C_COMPILER=cc" - ] + "extra_adios2_flags": ["CMAKE_CXX_COMPILER=CC", "CMAKE_C_COMPILER=cc"], }, "frontier": {"module_loads": []}, "aurora": {"module_loads": []}, @@ -366,7 +369,7 @@ def on_install_confirmed(settings: Settings) -> None: "modules", "kokkos", settings.kokkos_version, - settings.kokkos_backend, + settings.kokkos_backend, settings.kokkos_arch.strip().lower(), ) os.makedirs(os.path.dirname(kokkos_modfile_file), exist_ok=True) @@ -1089,14 +1092,18 @@ def toggle(k: str): def menu_cluster(self) -> Tuple[str, str, List[MenuItem]]: def choose(name: str): - print ("CALLING:", name) + print("CALLING:", name) apply_preset(self.s, name) self.push("custom") return ( "cluster-specific", "pick a preset:", - [MenuItem(cluster, "apply preset", on_enter=lambda c=cluster: choose(c)) for cluster in list(PRESETS.keys())] + [MenuItem("back", "", on_enter=self.pop)], + [ + MenuItem(cluster, "apply preset", on_enter=lambda c=cluster: choose(c)) + for cluster in list(PRESETS.keys()) + ] + + [MenuItem("back", "", on_enter=self.pop)], ) def get_menu(self) -> Tuple[str, str, List[MenuItem]]: From 940019dd7d16924fcede9b8f5a3c21c75232949a Mon Sep 17 00:00:00 2001 From: haykh Date: Mon, 21 Sep 2026 12:17:05 -0400 Subject: [PATCH 094/125] spec test --- tests/framework/CMakeLists.txt | 2 + tests/framework/spectra.cpp | 303 +++++++++++++++++++++++++++++++++ 2 files changed, 305 insertions(+) create mode 100644 tests/framework/spectra.cpp diff --git a/tests/framework/CMakeLists.txt b/tests/framework/CMakeLists.txt index 818ea0f2c..3696c74a9 100644 --- a/tests/framework/CMakeLists.txt +++ b/tests/framework/CMakeLists.txt @@ -66,9 +66,11 @@ if(${output}) if(${mpi}) gen_test(checkpoint-extents true) gen_test(checkpoint-data true) + gen_test(spectra true) else() gen_test(checkpoint-extents false) gen_test(checkpoint-data false) + gen_test(spectra false) endif() endif() diff --git a/tests/framework/spectra.cpp b/tests/framework/spectra.cpp new file mode 100644 index 000000000..0b8c6a40a --- /dev/null +++ b/tests/framework/spectra.cpp @@ -0,0 +1,303 @@ +/** + * Tests the spatially binned spectra output. + * + * Particles with known cell positions and energies are placed on a 16x12 + * Minkowski grid, binned into 4x3 spatial bins x 4 energy bins, written out + * via `Metadomain::WriteSpectra`, and read back from the `.bp` file. + * + * In MPI mode the particles are distributed over the two subdomains, which + * exercises the reduction of the per-rank histograms in `writeSpectrumSpatial`. + */ + +#include "enums.h" +#include "global.h" + +#include "utils/comparators.h" +#include "utils/error.h" +#include "utils/formatting.h" + +#include "metrics/minkowski.h" + +#include "framework/domain/metadomain.h" +#include "framework/parameters/parameters.h" + +#include +#include +#include + +#if defined(MPI_ENABLED) + #include +#endif + +#include +#include +#include +#include +#include + +using namespace ntt; +using namespace metric; + +const std::string SIMNAME = "test_spectra"; + +namespace { + constexpr ncells_t NX1 = 16, NX2 = 12; + constexpr std::size_t NB1 = 4, NB2 = 3, NBE = 4; + constexpr real_t E_MIN = 0.0, E_MAX = 4.0; + + struct TestPrtl { + ncells_t gi1, gi2; // global cell index + real_t gamma; + real_t weight; + bool alive; + std::size_t b1, b2, be; // expected bin + }; + + // dx = 0.5 in both directions, so no particle sits on a bin edge + const std::vector PRTLS { + { 0, 0, 1.5, 1.0, true, 0, 0, 0 }, + { 2, 3, 1.5, 2.0, true, 0, 0, 0 }, // accumulates with the previous one + { 6, 1, 2.5, 1.0, true, 1, 0, 1 }, + { 9, 7, 3.5, 1.0, true, 2, 1, 2 }, + { 15, 11, 4.5, 1.0, true, 3, 2, 3 }, + { 5, 9, 1.5, 3.0, true, 1, 2, 0 }, + { 13, 5, 100.0, 1.0, true, 3, 1, 3 }, // above e_max -> last energy bin + { 11, 11, 2.5, 5.0, false, 0, 0, 0 }, // dead -> not counted + }; + + void cleanup() { + std::filesystem::remove_all(SIMNAME); + } + + auto read_block(adios2::Engine& reader, + adios2::Variable& var, + std::vector& data) -> adios2::Dims { + for (const auto& blk : reader.BlocksInfo(var, reader.CurrentStep())) { + std::size_t ntot { 1 }; + for (const auto& c : blk.Count) { + ntot *= c; + } + if (ntot == 0) { + continue; + } + var.SetBlockSelection(blk.BlockID); + data.resize(ntot); + reader.Get(var, data.data(), adios2::Mode::Sync); + return blk.Count; + } + raise::Error("no non-empty block written for " + var.Name(), HERE); + return {}; + } +} // namespace + +auto main(int argc, char* argv[]) -> int { + GlobalInitialize(argc, argv); + + try { + using M = Minkowski; + + const std::vector res { NX1, NX2 }; + const boundaries_t extent { + { static_cast(0.0), static_cast(NX1) }, + { static_cast(0.0), static_cast(NX2) } + }; + const boundaries_t fldsbc { + { FldsBC::PERIODIC, FldsBC::PERIODIC }, + { FldsBC::PERIODIC, FldsBC::PERIODIC } + }; + const boundaries_t prtlbc { + { PrtlBC::PERIODIC, PrtlBC::PERIODIC }, + { PrtlBC::PERIODIC, PrtlBC::PERIODIC } + }; + const std::vector decomp { -1, -1 }; + + const std::vector species_params { + ParticleSpecies { static_cast(1), + "e-", 1.0f, + -1.0f, + static_cast(100), + timestep_t { 0 }, + timestep_t { 0 }, + ParticlePusher::BORIS, + false, RadiativeDrag::NONE, + EmissionType::NONE, + static_cast(0), + static_cast(0) } + }; + +#if !defined(MPI_ENABLED) + const unsigned int ndomains { 1 }; + adios2::ADIOS adios; +#else + int mpi_size; + MPI_Comm_size(MPI_COMM_WORLD, &mpi_size); + raise::ErrorIf(mpi_size != 2, "this test requires exactly 2 MPI ranks", HERE); + const unsigned int ndomains { static_cast(mpi_size) }; + adios2::ADIOS adios { MPI_COMM_WORLD }; +#endif + + const auto out_step = timestep_t { 5 }; + const auto out_time = simtime_t { 1.25 }; + + Metadomain md { ndomains, decomp, res, extent, + fldsbc, prtlbc, {}, species_params }; + + auto* local = md.subdomain_ptr(md.l_subdomain_indices()[0]); + const auto off = local->offset_ncells(); + const auto nloc = local->mesh.n_active(); + + // place the particles that fall into this subdomain + { + auto& sp = local->species[0]; + + auto i1_h = Kokkos::create_mirror_view(sp.i1); + auto i2_h = Kokkos::create_mirror_view(sp.i2); + auto dx1_h = Kokkos::create_mirror_view(sp.dx1); + auto dx2_h = Kokkos::create_mirror_view(sp.dx2); + auto ux1_h = Kokkos::create_mirror_view(sp.ux1); + auto w_h = Kokkos::create_mirror_view(sp.weight); + auto tag_h = Kokkos::create_mirror_view(sp.tag); + + npart_t np { 0 }; + for (const auto& p : PRTLS) { + if (p.gi1 < off[0] or p.gi1 >= off[0] + nloc[0] or p.gi2 < off[1] or + p.gi2 >= off[1] + nloc[1]) { + continue; + } + i1_h(np) = static_cast(p.gi1 - off[0]); + i2_h(np) = static_cast(p.gi2 - off[1]); + dx1_h(np) = static_cast(0.5); + dx2_h(np) = static_cast(0.5); + ux1_h(np) = math::sqrt(p.gamma * p.gamma - ONE); + w_h(np) = p.weight; + tag_h(np) = p.alive ? ParticleTag::alive : ParticleTag::dead; + ++np; + } + sp.set_npart(np); + + Kokkos::deep_copy(sp.i1, i1_h); + Kokkos::deep_copy(sp.i2, i2_h); + Kokkos::deep_copy(sp.dx1, dx1_h); + Kokkos::deep_copy(sp.dx2, dx2_h); + Kokkos::deep_copy(sp.ux1, ux1_h); + Kokkos::deep_copy(sp.weight, w_h); + Kokkos::deep_copy(sp.tag, tag_h); + } + + SimulationParams params; + params.set("simulation.name", SIMNAME); + params.set("output.format", std::string { "BPFile" }); + params.set("output.debug.ghosts", false); + params.set("output.fields.quantities", std::vector {}); + params.set("output.fields.custom", std::vector {}); + params.set("output.fields.downsampling", std::vector { 1, 1 }); + params.set("output.particles.species", std::vector {}); + params.set("output.spectra.e_min", E_MIN); + params.set("output.spectra.e_max", E_MAX); + params.set("output.spectra.log_bins", false); + params.set("output.spectra.num_energy_bins", NBE); + params.set("output.spectra.num_spatial_bins", + std::vector { NB1, NB2 }); + for (const auto& type : { "fields", "particles", "spectra" }) { + params.set("output." + std::string(type) + ".interval", timestep_t { 1 }); + params.set("output." + std::string(type) + ".interval_time", + simtime_t { -1.0 }); + } + params.setRawData(toml::value { toml::table {} }); + + md.InitWriter(&adios, params); + md.WriteSpectra(params, local, out_step, out_time); + + adios.FlushAll(); + + // ── read back ───────────────────────────────────────────────────────────── + { + adios2::IO io = adios.DeclareIO("read-spectra"); + io.SetEngine("BPFile"); + + namespace fs = std::filesystem; + adios2::Engine reader = io.Open( + fs::path(SIMNAME) / fs::path("spectra") / + fs::path(fmt::format("spectra.%08lu.bp", out_step)), + adios2::Mode::Read); + raise::ErrorIf(reader.BeginStep() != adios2::StepStatus::OK, + "no step in the spectra file", + HERE); + + { + auto var = io.InquireVariable("sN_1"); + raise::ErrorIf(not var, "sN_1 not found", HERE); + std::vector dn; + const auto shape = read_block(reader, var, dn); + + raise::ErrorIf( + shape != adios2::Dims({ NB1, NB2, NBE }), + fmt::format( + "sN_1 shape is %s, expected %s", + fmt::formatVector(shape).c_str(), + fmt::formatVector(std::vector { NB1, NB2, NBE }).c_str()), + HERE); + + std::vector expected(NB1 * NB2 * NBE, ZERO); + for (const auto& p : PRTLS) { + if (p.alive) { + expected[(p.b1 * NB2 + p.b2) * NBE + p.be] += p.weight; + } + } + for (auto i { 0u }; i < expected.size(); ++i) { + raise::ErrorIf(not cmp::AlmostEqual(dn[i], expected[i]), + fmt::format("sN_1[%u] = %f, expected %f", + i, + static_cast(dn[i]), + static_cast(expected[i])), + HERE); + } + } + + // bin edges: energy in [e_min, e_max], spatial in physical units + const std::vector>> bins { + { "sEbn", { 0.0, 1.0, 2.0, 3.0, 4.0 } }, + { "sX1bn", { 0.0, 4.0, 8.0, 12.0, 16.0 } }, + { "sX2bn", { 0.0, 4.0, 8.0, 12.0 } } + }; + for (const auto& [name, expected] : bins) { + auto var = io.InquireVariable(name); + raise::ErrorIf(not var, name + " not found", HERE); + std::vector edges; + const auto shape = read_block(reader, var, edges); + raise::ErrorIf(shape != adios2::Dims({ expected.size() }), + fmt::format("%s has %lu edges, expected %lu", + name.c_str(), + shape.empty() ? 0ul : shape[0], + expected.size()), + HERE); + for (auto i { 0u }; i < expected.size(); ++i) { + raise::ErrorIf(not cmp::AlmostEqual(edges[i], expected[i]), + fmt::format("%s[%u] = %f, expected %f", + name.c_str(), + i, + static_cast(edges[i]), + static_cast(expected[i])), + HERE); + } + } + + reader.EndStep(); + reader.Close(); + } + + } catch (const std::exception& e) { + std::cerr << e.what() << '\n'; + CallOnce([&] { + cleanup(); + }); + GlobalFinalize(); + return 1; + } + + CallOnce([&] { + cleanup(); + }); + GlobalFinalize(); + return 0; +} From 7258dcae91fc9fb3c6caf4041cb281c87ab15507 Mon Sep 17 00:00:00 2001 From: haykh Date: Mon, 21 Sep 2026 12:17:11 -0400 Subject: [PATCH 095/125] RUNTESTS From 881a9e34e78b2596ae48862c72f25c2b0082137d Mon Sep 17 00:00:00 2001 From: haykh Date: Mon, 21 Sep 2026 12:18:02 -0400 Subject: [PATCH 096/125] RUNTESTS From c56f796e1597d282f6338e4dff460a1c37280d4c Mon Sep 17 00:00:00 2001 From: haykh Date: Mon, 21 Sep 2026 13:58:44 -0400 Subject: [PATCH 097/125] minor --- input.example.toml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/input.example.toml b/input.example.toml index 8d6f25117..901818928 100644 --- a/input.example.toml +++ b/input.example.toml @@ -623,7 +623,7 @@ # @type: float # @default: 1e3 e_max = "" - # Whether to use logarithmic bins + # Whether to use logarithmic bins for energy # @type: bool # @default: true log_bins = "" From 33ac2eb8617fa464caecfd03976180cc0f10a781 Mon Sep 17 00:00:00 2001 From: LudwigBoess Date: Mon, 21 Sep 2026 18:58:26 +0000 Subject: [PATCH 098/125] revert conductor BC normal-B fix, preserved on bug/conductor_bcs_normal_b --- src/kernels/fields_bcs.hpp | 48 ++++++++++++++++++++++++-------------- 1 file changed, 30 insertions(+), 18 deletions(-) diff --git a/src/kernels/fields_bcs.hpp b/src/kernels/fields_bcs.hpp index c831fce4b..729a09d10 100644 --- a/src/kernels/fields_bcs.hpp +++ b/src/kernels/fields_bcs.hpp @@ -554,13 +554,15 @@ namespace kernel::bc { } if (tags & BC::B) { - if (i1 != 0) { + if (i1 == 0) { + Fld(i_edge, em::bx1) = ZERO; + } else { if constexpr (not P) { - Fld(i_edge - i1, em::bx1) = Fld(i_edge + i1, em::bx1); + Fld(i_edge - i1, em::bx1) = -Fld(i_edge + i1, em::bx1); Fld(i_edge - i1, em::bx2) = Fld(i_edge + i1 - 1, em::bx2); Fld(i_edge - i1, em::bx3) = Fld(i_edge + i1 - 1, em::bx3); } else { - Fld(i_edge + i1, em::bx1) = Fld(i_edge - i1, em::bx1); + Fld(i_edge + i1, em::bx1) = -Fld(i_edge - i1, em::bx1); Fld(i_edge + i1 - 1, em::bx2) = Fld(i_edge - i1, em::bx2); Fld(i_edge + i1 - 1, em::bx3) = Fld(i_edge - i1, em::bx3); } @@ -594,13 +596,15 @@ namespace kernel::bc { } if (tags & BC::B) { - if (i1 != 0) { + if (i1 == 0) { + Fld(i_edge, i2, em::bx1) = ZERO; + } else { if constexpr (not P) { - Fld(i_edge - i1, i2, em::bx1) = Fld(i_edge + i1, i2, em::bx1); + Fld(i_edge - i1, i2, em::bx1) = -Fld(i_edge + i1, i2, em::bx1); Fld(i_edge - i1, i2, em::bx2) = Fld(i_edge + i1 - 1, i2, em::bx2); Fld(i_edge - i1, i2, em::bx3) = Fld(i_edge + i1 - 1, i2, em::bx3); } else { - Fld(i_edge + i1, i2, em::bx1) = Fld(i_edge - i1, i2, em::bx1); + Fld(i_edge + i1, i2, em::bx1) = -Fld(i_edge - i1, i2, em::bx1); Fld(i_edge + i1 - 1, i2, em::bx2) = Fld(i_edge - i1, i2, em::bx2); Fld(i_edge + i1 - 1, i2, em::bx3) = Fld(i_edge - i1, i2, em::bx3); } @@ -625,14 +629,16 @@ namespace kernel::bc { } if (tags & BC::B) { - if (i2 != 0) { + if (i2 == 0) { + Fld(i1, i_edge, em::bx2) = ZERO; + } else { if constexpr (not P) { Fld(i1, i_edge - i2, em::bx1) = Fld(i1, i_edge + i2 - 1, em::bx1); - Fld(i1, i_edge - i2, em::bx2) = Fld(i1, i_edge + i2, em::bx2); + Fld(i1, i_edge - i2, em::bx2) = -Fld(i1, i_edge + i2, em::bx2); Fld(i1, i_edge - i2, em::bx3) = Fld(i1, i_edge + i2 - 1, em::bx3); } else { Fld(i1, i_edge + i2 - 1, em::bx1) = Fld(i1, i_edge - i2, em::bx1); - Fld(i1, i_edge + i2, em::bx2) = Fld(i1, i_edge - i2, em::bx2); + Fld(i1, i_edge + i2, em::bx2) = -Fld(i1, i_edge - i2, em::bx2); Fld(i1, i_edge + i2 - 1, em::bx3) = Fld(i1, i_edge - i2, em::bx3); } } @@ -672,9 +678,11 @@ namespace kernel::bc { } if (tags & BC::B) { - if (i1 != 0) { + if (i1 == 0) { + Fld(i_edge, i2, i3, em::bx1) = ZERO; + } else { if constexpr (not P) { - Fld(i_edge - i1, i2, i3, em::bx1) = Fld(i_edge + i1, i2, i3, em::bx1); + Fld(i_edge - i1, i2, i3, em::bx1) = -Fld(i_edge + i1, i2, i3, em::bx1); Fld(i_edge - i1, i2, i3, em::bx2) = Fld(i_edge + i1 - 1, i2, i3, @@ -684,7 +692,7 @@ namespace kernel::bc { i3, em::bx3); } else { - Fld(i_edge + i1, i2, i3, em::bx1) = Fld(i_edge - i1, i2, i3, em::bx1); + Fld(i_edge + i1, i2, i3, em::bx1) = -Fld(i_edge - i1, i2, i3, em::bx1); Fld(i_edge + i1 - 1, i2, i3, em::bx2) = Fld(i_edge - i1, i2, i3, @@ -721,13 +729,15 @@ namespace kernel::bc { } if (tags & BC::B) { - if (i2 != 0) { + if (i2 == 0) { + Fld(i1, i_edge, i3, em::bx2) = ZERO; + } else { if constexpr (not P) { Fld(i1, i_edge - i2, i3, em::bx1) = Fld(i1, i_edge + i2 - 1, i3, em::bx1); - Fld(i1, i_edge - i2, i3, em::bx2) = Fld(i1, i_edge + i2, i3, em::bx2); + Fld(i1, i_edge - i2, i3, em::bx2) = -Fld(i1, i_edge + i2, i3, em::bx2); Fld(i1, i_edge - i2, i3, em::bx3) = Fld(i1, i_edge + i2 - 1, i3, @@ -737,7 +747,7 @@ namespace kernel::bc { i_edge - i2, i3, em::bx1); - Fld(i1, i_edge + i2, i3, em::bx2) = Fld(i1, i_edge - i2, i3, em::bx2); + Fld(i1, i_edge + i2, i3, em::bx2) = -Fld(i1, i_edge - i2, i3, em::bx2); Fld(i1, i_edge + i2 - 1, i3, em::bx3) = Fld(i1, i_edge - i2, i3, @@ -770,7 +780,9 @@ namespace kernel::bc { } if (tags & BC::B) { - if (i3 != 0) { + if (i3 == 0) { + Fld(i1, i2, i_edge, em::bx3) = ZERO; + } else { if constexpr (not P) { Fld(i1, i2, i_edge - i3, em::bx1) = Fld(i1, i2, @@ -780,7 +792,7 @@ namespace kernel::bc { i2, i_edge + i3 - 1, em::bx2); - Fld(i1, i2, i_edge - i3, em::bx3) = Fld(i1, i2, i_edge + i3, em::bx3); + Fld(i1, i2, i_edge - i3, em::bx3) = -Fld(i1, i2, i_edge + i3, em::bx3); } else { Fld(i1, i2, i_edge + i3 - 1, em::bx1) = Fld(i1, i2, @@ -790,7 +802,7 @@ namespace kernel::bc { i2, i_edge - i3, em::bx2); - Fld(i1, i2, i_edge + i3, em::bx3) = Fld(i1, i2, i_edge - i3, em::bx3); + Fld(i1, i2, i_edge + i3, em::bx3) = -Fld(i1, i2, i_edge - i3, em::bx3); } } } From 4aac8c92275f9f2c26e303a16a5049cfdb99e47d Mon Sep 17 00:00:00 2001 From: haykh Date: Mon, 21 Sep 2026 15:40:52 -0400 Subject: [PATCH 099/125] taplo -> tombi --- .taplo.toml | 6 - .tombi.toml | 16 + CODEGUIDE.md | 76 +- dev/nix/shell.nix | 6 +- entity.schema.json | 2920 +++++++++++++++++ input.example.toml => input.default.toml | 754 +++-- dependencies.py => scripts/dependencies.py | 0 scripts/generate_template.py | 505 +++ .../ideal_tile_size.py | 0 .../render_preview.py | 0 10 files changed, 3955 insertions(+), 328 deletions(-) delete mode 100644 .taplo.toml create mode 100644 .tombi.toml create mode 100644 entity.schema.json rename input.example.toml => input.default.toml (73%) rename dependencies.py => scripts/dependencies.py (100%) create mode 100755 scripts/generate_template.py rename ideal_tile_size.py => scripts/ideal_tile_size.py (100%) rename render_preview.py => scripts/render_preview.py (100%) diff --git a/.taplo.toml b/.taplo.toml deleted file mode 100644 index 423a47594..000000000 --- a/.taplo.toml +++ /dev/null @@ -1,6 +0,0 @@ -[formatting] - align_entries = true - indent_tables = true - indent_entries = true - trailing_newline = true - align_comments = true diff --git a/.tombi.toml b/.tombi.toml new file mode 100644 index 000000000..4124004e7 --- /dev/null +++ b/.tombi.toml @@ -0,0 +1,16 @@ +toml-version = "v1.0.0" + +[format] + [format.rules] + indent-sub-tables = true + indent-table-key-value-pairs = true + trailing-comment-alignment = true + +[schema] + enabled = true + strict = true + +[[schemas]] + path = "entity.schema.json" + include = ["*.toml"] + exclude = [".tombi.toml"] diff --git a/CODEGUIDE.md b/CODEGUIDE.md index 1e4dee68d..cfef82466 100644 --- a/CODEGUIDE.md +++ b/CODEGUIDE.md @@ -32,6 +32,11 @@ entity ├── pgens # problem generators ├── examples # example problem generators with standard use-cases ├── tutorials # problem generators from tutorials +├── scripts # user-facing helper scripts +│ ├── dependencies.py # deployment scripts on various machines +│ ├── generate_template.py # renders `input.default.toml` from `entity.schema.json` +│ ├── ideal_tile_size.py # recommends the team tile size for the tiled deposit +│ └── render_preview.py # previews the in-situ renderer geometry from an input file ├── src # main code containing all separate submodules │ ├── archetypes # archetypes which can be used by the user in problem generators │ ├── engines # simulation engines @@ -48,15 +53,15 @@ entity ├── .gitattributes ├── .gitignore ├── .gitmodules -├── .taplo.toml # formatting guidelines for toml files +├── .tombi.toml # formatting guidelines for toml files + schema association ├── CITATION ├── CODEGUIDE.md # this file ├── CMakeLists.txt # root cmake file ├── CODE_OF_CONDUCT.md ├── LICENSE ├── README.md -├── dependencies.py # deployment scripts on various machines -└── input.example.toml # most complete toml file with all possible input options +├── entity.schema.json # JSON Schema for the input file: the source of truth +└── input.default.toml # generated reference input with every option at its default ``` ## Testing @@ -75,13 +80,76 @@ You can also compile all the problem generators and run the ones from the `examp ./dev/scripts/tests.sh --build build_dir --flags "-D mpi=ON" --with_pgens --make_plots ``` +## Input configuration + +`entity.schema.json` is the single source of truth for the input file. It is a [JSON Schema](https://json-schema.org) (draft 2020-12) describing every table and key the code reads, and it serves two purposes at once: + +* editors validate and autocomplete input files against it as you type (see [Formatting](#formatting) below); +* `input.default.toml` -- the annotated reference input listing every option -- is *generated* from it, so the docs cannot drift from what is validated. + +Regenerate the reference input after any schema change: + +```sh +python scripts/generate_template.py -d -o input.default.toml +``` + +Dropping `-d` renders the same file with every value left as `""`, i.e. a blank form to fill in rather than a list of defaults. Writing to stdout (the default) is handy for reviewing a change: `diff <(python scripts/generate_template.py -d) input.default.toml`. + +### The `x-entity` annotations + +Standard JSON Schema keywords (`type`, `enum`, `minimum`, `items`, `prefixItems`, `required`, `default`, `deprecated`, ...) carry everything a validator can check. Everything else lives in an `x-entity` object on the node, and is what the generator turns into the `@`-annotations above each key: + +| field | meaning | +| --- | --- | +| `type` | the literal `@type:` string, e.g. `"array [size 1 :->: 3]"` -- richer than the JSON type | +| `default` | the literal `@default:` text, for defaults the code computes at runtime (`"N_GHOSTS"`, `"1% of the domain size"`) or that need a specific notation (`"1e-4"` rather than `0.0001`) | +| `notes` | ordered `@note:` lines; embedded newlines are kept as hard line breaks | +| `examples` | ordered `@example:` lines | +| `enum` | an *illustrative, non-exhaustive* value list, never validated (e.g. `output.fields.quantities`) | +| `deprecated` | the `@deprecated:` text, paired with the standard `"deprecated": true` | +| `inferred` | see below | + +`x-entity.inferred` sits on a **table** and lists quantities the code derives rather than reads -- `grid.dim`, `scales.sigma0`, `checkpoint.start_step`. They are deliberately *not* in `properties`, so `additionalProperties: false` rejects them as input keys, and the generator emits them as an `@inferred:` comment block after that table's own keys. + +### Adding a new input parameter + +1. Add the key to `entity.schema.json`, in the position you want it to appear in the reference input -- property order is emission order, and scalar keys are emitted before sub-tables regardless. +2. Give it a `description` (the brief line) and an `x-entity.type`; add real constraints (`minimum`, `enum`, `minItems`, ...) wherever they are checkable, and a `default` when it has a literal one. +3. Regenerate `input.default.toml`. +4. Parse it in `src/framework/parameters/`, and register any derived quantity under `x-entity.inferred`. + +Three things to keep in mind: + +* **String enums are matched case-insensitively by the code** (`fmt::toLower` is applied to `engine`, `metric`, the boundary lists, `pusher`, `log_level`, ...), so a bare `"enum"` would reject perfectly valid input. The convention is `anyOf: [{"enum": []}, {"type": "string", "pattern": "(?i)^(|...)$"}]` -- the enum branch drives completion and hover, the pattern branch keeps any casing legal. Note `(?i)` is a Rust/Python regex extension: tombi honours it, JS-based validators do not. +* **Every table is closed.** Set `additionalProperties: false` so typos are caught; tombi's `strict = true` closes objects that omit it anyway. `[setup]` is the one deliberate exception (`additionalProperties: true`), since its keys belong to the problem generator. +* **If a key's documented default is `[]`, the empty array must validate**, which `minItems` would otherwise forbid -- use `anyOf: [{"maxItems": 0}, {}]` (see `output.render.x1_lim`). + ## Code guidelines ### Formatting To maintain coherence throughout the source code, we use `clang-format` to enforce a uniform style. A corresponding `.clang-format` file with all the style-related settings can be found in the root directory of the code. To use this, one needs to have the `clang-format` executable (typically provided with the `llvm` package). After installing the `clang-format` itself (check by running `clang-format --version`), you can use it either manually by running `clang-format .` in the route directory of the code, or attach it to your favorite code editor to run on save. For VSCode, the recommended extension is [`xaver.clang-format`](https://github.com/xaverh/vscode-clang-format), for vim -- [`rhysd/vim-clang-format`](https://vimawesome.com/plugin/vim-clang-format), for nvim -- [`stevearc/conform.nvim`](https://github.com/stevearc/conform.nvim), for [emacs](https://www.vim.org/download.php). -You can run the formatting on all files with `./dev/scripts/format.sh`. +You can run the formatting on all files with `./dev/scripts/format.sh` (this covers C++ and CMake; TOML is handled separately, below). + +TOML files are formatted and validated with [`tombi`](https://tombi-toml.github.io/tombi/), which is a formatter, linter and language server in one. The settings live in `.tombi.toml` in the root directory, which also associates `entity.schema.json` with every `.toml` file in the tree -- so input files are checked against the schema as you edit them, with completion and hover documentation for every key. It is provided by the nix shell (`dev/nix`); otherwise install it with `uvx tombi`, `pip install tombi`, `npm i -g tombi` or `brew install tombi`. + +From the command line: + +```sh +tombi format # formats the whole project (or pass files/directories) +tombi format --check # verify only, for CI -- mirrors `format.sh --verify` +tombi lint # schema validation only +``` + +In the editor, point it at the `tombi lsp` language server. For VSCode, the extension is [`tombi-toml.tombi`](https://marketplace.visualstudio.com/items?itemName=tombi-toml.tombi); for nvim, `tombi` ships as a built-in `nvim-lspconfig` server, so `vim.lsp.enable('tombi')` is enough. Individual input files can opt into the schema explicitly -- useful outside the repo -- with a directive on the first line: + +```toml +#:schema ./entity.schema.json +``` + +> [!NOTE] +> `tombi` replaces `taplo`, which the project used previously and which is no longer maintained. Best practices are also enforced using `clang-tidy`; to generate recommendations for all the files, run `./dev/scripts/tidy.sh --build build_dir` where `build_dir` is the directory where the code was built, or for specific files: `./dev/scripts/tidy.sh --build build_dir --files "(file1|file2).cpp"` or only for the changed files: `./dev/scripts/tidy.sh --build build_dir --changed`. The recommendations will be in the `tidy/` directory. diff --git a/dev/nix/shell.nix b/dev/nix/shell.nix index addd470fa..6b6c950f2 100644 --- a/dev/nix/shell.nix +++ b/dev/nix/shell.nix @@ -55,14 +55,14 @@ pkgs.mkShell { neocmakelsp black pyright - taplo + tombi vscode-langservers-extracted ]; - LD_LIBRARY_PATH = pkgs.lib.makeLibraryPath ([ + LD_LIBRARY_PATH = pkgs.lib.makeLibraryPath [ pkgs.stdenv.cc.cc pkgs.zlib - ]); + ]; shellHook = '' BLUE='\033[0;34m' diff --git a/entity.schema.json b/entity.schema.json new file mode 100644 index 000000000..8f4528210 --- /dev/null +++ b/entity.schema.json @@ -0,0 +1,2920 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "https://entity-toolkit.github.io/schema/entity.schema.json", + "title": "entity", + "description": "Entity simulation input file", + "$comment": "GENERATOR CONVENTION -- this schema is the single source of truth for `input.template.toml`. Standard keywords carry the machine-checkable part (type/enum/minimum/maximum/items/prefixItems/minItems/maxItems/pattern/default/required/deprecated). The `x-entity` object carries the parts JSON Schema cannot express, all verbatim from the template comments: `type` = the literal `@type:` annotation (use it for the comment line whenever present, else derive from the standard keywords); `default` = the literal `@default:` annotation (use it whenever present, else format the standard `default`); `notes` = ordered `@note:` lines; `examples` = ordered `@example:` lines; `enum` = an illustrative, NON-exhaustive value list that must not be validated as a real `enum`; `deprecated` = the `@deprecated:` text. `x-entity.inferred` on an object lists quantities the code derives rather than reads; they are NOT valid input keys (so they are absent from `properties` and rejected by `additionalProperties: false`) and should be emitted as an `@inferred:` comment block after that table's own keys and before its sub-tables. Property order in `properties` is the emission order. Objects with `additionalProperties: true` are free-form. `$ref` is used once, for `$defs/colormap`.", + "type": "object", + "additionalProperties": false, + "required": [ + "simulation", + "grid", + "scales", + "particles" + ], + "$defs": { + "colormap": { + "type": "string", + "pattern": "(?i)^(cmr\\.)?(viridis|inferno|plasma|cool2warm|gray|RdBu_r|dusk|cosmic|freeze|apple|gothic|sunburst|voltage|ocean|fusion|prinsenvlag)$", + "x-entity": { + "type": "string", + "enum": [ + "viridis", + "inferno", + "plasma", + "cool2warm", + "gray", + "RdBu_r", + "and the CMasher maps (BSD-3, https://cmasher.readthedocs.io): \"dusk\", \"cosmic\", \"freeze\", \"apple\", \"gothic\", \"sunburst\", \"voltage\", \"ocean\", \"fusion\", \"prinsenvlag\" (an optional \"cmr.\" prefix is ok)" + ] + } + } + }, + "properties": { + "simulation": { + "description": "Global simulation parameters", + "type": "object", + "additionalProperties": false, + "required": [ + "name", + "engine", + "runtime" + ], + "properties": { + "name": { + "description": "Name of the simulation", + "type": "string", + "x-entity": { + "type": "string", + "notes": [ + "The name is used for the output files" + ] + } + }, + "engine": { + "description": "Simulation engine to use", + "anyOf": [ + { + "enum": [ + "SRPIC", + "GRPIC" + ] + }, + { + "type": "string", + "pattern": "(?i)^(SRPIC|GRPIC)$" + } + ], + "x-entity": { + "type": "string" + } + }, + "runtime": { + "description": "Max runtime in physical (code) units", + "type": "number", + "exclusiveMinimum": 0.0, + "x-entity": { + "type": "float [> 0]", + "examples": [ + "1e5" + ] + } + }, + "domain": { + "description": "Parameters specific to domain decomposition", + "type": "object", + "additionalProperties": false, + "properties": { + "number": { + "description": "Number of domains", + "type": "integer", + "minimum": 1, + "x-entity": { + "type": "int", + "default": "1 [no MPI]; MPI_SIZE [MPI]" + } + }, + "decomposition": { + "description": "Decomposition of the domain (for MPI) in each of the directions", + "type": "array", + "minItems": 1, + "maxItems": 3, + "items": { + "type": "integer" + }, + "default": [ + -1, + -1, + -1 + ], + "x-entity": { + "type": "array [size 1 :->: 3]", + "notes": [ + "-1 means the code will determine the decomposition in the specific direction automatically", + "Automatic detection is either done by inference from # of MPI tasks, or by balancing the grid size on each domain" + ], + "examples": [ + "[2, 2, 2] (total of 8 domains)" + ] + } + }, + "load_balance": { + "description": "Diffusion-style dynamic load balancing (Cartesian metrics only). Domain boundaries between MPI neighbors are nudged to equalize the active-particle count per rank. All inter-rank traffic uses only the existing nearest-neighbor field/particle communication paths.", + "type": "object", + "additionalProperties": false, + "properties": { + "enable": { + "description": "Enable dynamic load balancing", + "type": "boolean", + "default": false, + "x-entity": { + "type": "bool" + } + }, + "interval": { + "description": "Run the rebalancer every `interval` timesteps (0 disables)", + "type": "integer", + "minimum": 0, + "default": 0, + "x-entity": { + "type": "int" + } + }, + "dimensions": { + "description": "Dimensions along which load is redistributed (1 = x1, 2 = x2, 3 = x3)", + "type": "array", + "minItems": 1, + "maxItems": 3, + "uniqueItems": true, + "items": { + "type": "integer", + "enum": [ + 1, + 2, + 3 + ] + }, + "default": [ + 1 + ], + "x-entity": { + "type": "array of int, subset of [1, 2, 3]" + } + }, + "tolerance": { + "description": "Skip rebalancing along a dim when (max - min) / mean of the per-slice particle count is below this fraction", + "type": "number", + "minimum": 0.0, + "default": 0.1, + "x-entity": { + "type": "float" + } + }, + "max_shift": { + "description": "Maximum cell-shift per interior boundary per event; clamped at compile time to N_GHOSTS so the migrating field strip is already cached in the rank's ghost zone.", + "type": "integer", + "minimum": 0, + "x-entity": { + "type": "int", + "default": "N_GHOSTS" + } + } + } + } + } + } + } + }, + "grid": { + "description": "Parameters specific to grid geometry", + "type": "object", + "additionalProperties": false, + "required": [ + "resolution", + "extent", + "metric", + "boundaries" + ], + "x-entity": { + "inferred": [ + { + "name": "dim", + "brief": "Dimensionality of the grid", + "type": "short", + "enum": [ + 1, + 2, + 3 + ], + "from": "`grid.resolution`" + } + ] + }, + "properties": { + "resolution": { + "description": "Spatial resolution of the grid", + "type": "array", + "minItems": 1, + "maxItems": 3, + "items": { + "type": "integer", + "minimum": 1 + }, + "x-entity": { + "type": "array [size 1 :->: 3]", + "notes": [ + "Dimensionality is inferred from the size of this array" + ], + "examples": [ + "[1024, 1024, 1024]" + ] + } + }, + "extent": { + "description": "Physical extent of the grid", + "type": "array", + "minItems": 1, + "maxItems": 3, + "items": { + "type": "array", + "minItems": 2, + "maxItems": 2, + "prefixItems": [ + { + "type": "number" + }, + { + "type": "number" + } + ] + }, + "x-entity": { + "type": "array> [size 1 :->: 3]", + "notes": [ + "For spherical geometry, only specify `[[rmin, rmax]]`, other values are set automatically", + "For cartesian geometry, cell aspect ratio has to be 1: `dx=dy=dz`" + ], + "examples": [ + "[[0.0, 1.0], [-1.0, 1.0]]" + ] + } + }, + "metric": { + "description": "Metric-related parameters", + "type": "object", + "additionalProperties": false, + "required": [ + "metric" + ], + "x-entity": { + "inferred": [ + { + "name": "coord", + "brief": "Coordinate system on the grid", + "type": "string", + "enum": [ + "cartesian", + "spherical", + "qspherical" + ], + "from": "`grid.metric.metric`" + }, + { + "name": "ks_rh", + "brief": "Size of the horizon for GR Kerr Schild", + "type": "float", + "from": "`grid.metric.ks_a`" + }, + { + "name": "params", + "brief": "A map of all metric-specific parameters together (for easy access)", + "type": "map", + "from": "`grid.metric`" + } + ] + }, + "properties": { + "metric": { + "description": "Metric on the grid", + "anyOf": [ + { + "enum": [ + "Minkowski", + "Spherical", + "QSpherical", + "Kerr_Schild", + "QKerr_Schild", + "Kerr_Schild_0" + ] + }, + { + "type": "string", + "pattern": "(?i)^(Minkowski|Spherical|QSpherical|Kerr_Schild|QKerr_Schild|Kerr_Schild_0)$" + } + ], + "x-entity": { + "type": "string" + } + }, + "qsph_r0": { + "description": "`r0` paramter for the QSpherical metric `x1 = log(r-r0)`", + "type": "number", + "default": 0.0, + "x-entity": { + "type": "float [-inf -> rmin]", + "notes": [ + "Negative values produce almost uniform grid in r" + ] + } + }, + "qsph_h": { + "description": "`h` paramter for the QSpherical metric `th = x2 + 2*h x2 (pi-2*x2)*(pi-x2)/pi^2`", + "type": "number", + "minimum": -1.0, + "maximum": 1.0, + "default": 0.0, + "x-entity": { + "type": "float [-1 :->: 1]" + } + }, + "ks_a": { + "description": "Spin parameter for the Kerr Schild metric", + "type": "number", + "minimum": 0.0, + "maximum": 1.0, + "default": 0.0, + "x-entity": { + "type": "float [0 :-> 1]" + } + } + } + }, + "boundaries": { + "description": "Boundary-condition related parameters", + "type": "object", + "additionalProperties": false, + "required": [ + "fields", + "particles" + ], + "properties": { + "fields": { + "description": "Boundary conditions for fields", + "type": "array", + "minItems": 1, + "maxItems": 3, + "items": { + "type": "array", + "minItems": 1, + "maxItems": 2, + "items": { + "anyOf": [ + { + "enum": [ + "PERIODIC", + "MATCH", + "FIXED", + "ATMOSPHERE", + "CUSTOM", + "HORIZON", + "CONDUCTOR" + ] + }, + { + "type": "string", + "pattern": "(?i)^(PERIODIC|MATCH|FIXED|ATMOSPHERE|CUSTOM|HORIZON|CONDUCTOR)$" + } + ] + } + }, + "x-entity": { + "type": "array> [size 1 :->: 3]", + "notes": [ + "When periodic in any of the directions, you should only set one value: [..., [\"PERIODIC\"], ...]", + "In spherical, bondaries in theta/phi are set automatically (only specify bc @ `[rmin, rmax]`): [[\"ATMOSPHERE\", \"MATCH\"]]", + "In GR, the horizon boundary is set automatically (only specify bc @ rmax): [[\"MATCH\"]]" + ], + "examples": [ + "[[\"CUSTOM\", \"MATCH\"]] (for 2D spherical `[[rmin, rmax]]`)" + ] + } + }, + "particles": { + "description": "Boundary conditions for particles", + "type": "array", + "minItems": 1, + "maxItems": 3, + "items": { + "type": "array", + "minItems": 1, + "maxItems": 2, + "items": { + "anyOf": [ + { + "enum": [ + "PERIODIC", + "ABSORB", + "ATMOSPHERE", + "CUSTOM", + "REFLECT", + "HORIZON" + ] + }, + { + "type": "string", + "pattern": "(?i)^(PERIODIC|ABSORB|ATMOSPHERE|CUSTOM|REFLECT|HORIZON)$" + } + ] + } + }, + "x-entity": { + "type": "array> [size 1 :->: 3]", + "notes": [ + "When periodic in any of the directions, you should only set one value [..., [\"PERIODIC\"], ...]", + "In spherical, bondaries in theta/phi are set automatically (only specify bc @ `[rmin, rmax]`) [[\"ATMOSPHERE\", \"ABSORB\"]]", + "In GR, the horizon boundary is set automatically (only specify bc @ `rmax`): [[\"ABSORB\"]]" + ], + "examples": [ + "[[\"PERIODIC\"], [\"PERIODIC\"]]" + ] + } + }, + "match": { + "description": "Parameters specific to MATCH boundary conditions", + "type": "object", + "additionalProperties": false, + "properties": { + "ds": { + "description": "Size of the matching layer in each direction for fields in physical (code) units", + "anyOf": [ + { + "type": "number", + "exclusiveMinimum": 0.0 + }, + { + "type": "array", + "minItems": 1, + "maxItems": 3, + "items": { + "type": "array", + "maxItems": 2, + "items": { + "type": "number" + } + } + } + ], + "x-entity": { + "type": "float | array>", + "default": "1% of the domain size (in shortest dimension)", + "notes": [ + "In spherical, this is the size of the layer in `r` from the outer wall" + ], + "examples": [ + "`ds = 1.5` (will set the same for all directions)", + "`ds = [[1.5], [2.0, 1.0], [1.1]]` (will duplicate 1.5 for +/- `x1` and 1.1 for +/- `x3`)", + "`ds = [[], [1.5], []]` (will only set for x2)" + ] + } + } + } + }, + "absorb": { + "description": "Parameters specific to ABSORB boundary conditions", + "type": "object", + "additionalProperties": false, + "properties": { + "ds": { + "description": "Size of the absorption layer for particles in physical (code) units", + "type": "number", + "exclusiveMinimum": 0.0, + "x-entity": { + "type": "float", + "default": "1% of the domain size (in shortest dimension)", + "notes": [ + "In spherical, this is the size of the layer in `r` from the outer wall", + "In cartesian, this is the same for all dimensions where applicable" + ] + } + } + } + }, + "atmosphere": { + "description": "Parameters specific to ATMOSPHERE boundary conditions", + "type": "object", + "additionalProperties": false, + "x-entity": { + "inferred": [ + { + "name": "g", + "brief": "Acceleration due to imposed gravity", + "type": "float", + "from": "`grid.boundaries.atmosphere.temperature`, `grid.boundaries.atmosphere.height`", + "value": "`temperature / height`" + } + ] + }, + "properties": { + "temperature": { + "description": "Temperature of the atmosphere in units of `m0 c^2`", + "type": "number", + "minimum": 0.0, + "x-entity": { + "type": "float", + "notes": [ + "[required] if `ATMOSPHERE` is one of the boundaries" + ] + } + }, + "density": { + "description": "Peak number density of the atmosphere at base in units of `n0`", + "type": "number", + "minimum": 0.0, + "x-entity": { + "type": "float" + } + }, + "height": { + "description": "Pressure scale-height in physical units", + "type": "number", + "exclusiveMinimum": 0.0, + "x-entity": { + "type": "float" + } + }, + "species": { + "description": "Species indices of particles that populate the atmosphere", + "type": "array", + "minItems": 2, + "maxItems": 2, + "prefixItems": [ + { + "type": "integer", + "minimum": 1 + }, + { + "type": "integer", + "minimum": 1 + } + ], + "x-entity": { + "type": "array [size 2]" + } + }, + "ds": { + "description": "Distance from the edge to which the gravity is imposed in physical units", + "type": "number", + "minimum": 0.0, + "default": 0.0, + "x-entity": { + "type": "float", + "notes": [ + "0.0 means no limit" + ] + } + } + } + } + } + } + } + }, + "scales": { + "description": "Fiducial scales that fix the code unit system", + "type": "object", + "additionalProperties": false, + "required": [ + "larmor0", + "skindepth0" + ], + "x-entity": { + "inferred": [ + { + "name": "dx0", + "brief": "fiducial minimum size of the cell", + "type": "float", + "from": "`grid`" + }, + { + "name": "V0", + "brief": "fiducial elementary volume", + "type": "float", + "from": "`grid`" + }, + { + "name": "n0", + "brief": "Fiducial number density", + "type": "float", + "from": "`particles.ppc0`, `grid`", + "value": "`ppc0 / V0`" + }, + { + "name": "q0", + "brief": "Fiducial elementary charge", + "type": "float", + "from": "`scales.skindepth0`, `scales.n0`", + "value": "`1 / (n0 * skindepth0^2)`" + }, + { + "name": "sigma0", + "brief": "Fiducial magnetization parameter", + "type": "float", + "from": "`scales.larmor0`, `scales.skindepth0`", + "value": "`(skindepth0 / larmor0)^2`" + }, + { + "name": "B0", + "brief": "Fiducial magnetic field", + "type": "float", + "from": "`scales.larmor0`", + "value": "`1 / larmor0`" + }, + { + "name": "omegaB0", + "brief": "Fiducial cyclotron frequency", + "type": "float", + "from": "`scales.larmor0`", + "value": "`1 / larmor0`" + } + ] + }, + "properties": { + "larmor0": { + "description": "Fiducial larmor radius", + "type": "number", + "exclusiveMinimum": 0.0, + "x-entity": { + "type": "float [> 0.0]" + } + }, + "skindepth0": { + "description": "Fiducial plasma skin depth", + "type": "number", + "exclusiveMinimum": 0.0, + "x-entity": { + "type": "float [> 0.0]" + } + } + } + }, + "radiation": { + "description": "Radiative drag and photon emission parameters", + "type": "object", + "additionalProperties": false, + "properties": { + "drag": { + "description": "Radiation reaction (drag) parameters", + "type": "object", + "additionalProperties": false, + "properties": { + "synchrotron": { + "description": "Synchrotron drag parameters", + "type": "object", + "additionalProperties": false, + "properties": { + "gamma_rad": { + "description": "Radiation reaction limit gamma-factor for synchrotron", + "type": "number", + "exclusiveMinimum": 0.0, + "default": 1.0, + "x-entity": { + "type": "float [> 0.0]", + "notes": [ + "[required] if one of the species has `radiative_drag = \"synchrotron\"`" + ] + } + } + } + }, + "compton": { + "description": "Compton drag parameters", + "type": "object", + "additionalProperties": false, + "properties": { + "gamma_rad": { + "description": "Radiation reaction limit gamma-factor for Compton drag", + "type": "number", + "exclusiveMinimum": 0.0, + "default": 1.0, + "x-entity": { + "type": "float [> 0.0]", + "notes": [ + "[required] if one of the species has `radiative_drag = \"compton\"`" + ] + } + } + } + } + } + }, + "emission": { + "description": "Photon emission parameters", + "type": "object", + "additionalProperties": false, + "properties": { + "synchrotron": { + "description": "Synchrotron emission parameters", + "type": "object", + "additionalProperties": false, + "required": [ + "photon_species" + ], + "x-entity": { + "inferred": [ + { + "name": "nominal_probability", + "brief": "Nominal probability of the emission for a particle with `gamma * beta = 1`, charge-to-mass = `q0 / m0`", + "type": "float", + "from": "`.gamma_qed`, `.photon_weight`, `...drag.synchrotron.gamma_rad`, `scales.omegaB0`, `algorithms.timestep.dt`", + "value": "`0.1 * omegaB0 * dt * (gamma_qed / gamma_rad)^2 / photon_weight`" + }, + { + "name": "nominal_photon_energy", + "brief": "Nominal energy of the emitted photon for a particle with `gamma * beta = 1`, mass = `m0`", + "type": "float", + "from": "`.gamma_qed`", + "value": "`(1 / gamma_qed)^2`" + } + ] + }, + "properties": { + "gamma_qed": { + "description": "Gamma-factor of a particle emitting synchrotron photons at energy `m0 c^2` in fiducial magnetic field `B0`", + "type": "number", + "exclusiveMinimum": 1.0, + "default": 10.0, + "x-entity": { + "type": "float [> 1.0]" + } + }, + "photon_energy_min": { + "description": "Minimum photon energy for synchrotron emission (units of `m0 c^2`)", + "type": "number", + "exclusiveMinimum": 0.0, + "default": 0.0001, + "x-entity": { + "type": "float [> 0.0]", + "default": "1e-4" + } + }, + "photon_weight": { + "description": "Weights for the emitted synchrotron photons", + "type": "number", + "exclusiveMinimum": 0.0, + "default": 1.0, + "x-entity": { + "type": "float [> 0.0]" + } + }, + "photon_species": { + "description": "Index of species for the emitted photon", + "type": "integer", + "minimum": 1, + "x-entity": { + "type": "ushort [> 0]" + } + } + } + }, + "compton": { + "description": "Inverse Compton emission parameters", + "type": "object", + "additionalProperties": false, + "required": [ + "photon_species" + ], + "x-entity": { + "inferred": [ + { + "name": "nominal_probability", + "brief": "Nominal probability of the emission for a particle with `gamma * beta = 1`, charge-to-mass = `q0 / m0`", + "type": "float", + "from": "`.gamma_qed`, `.photon_weight`, `...drag.compton.gamma_rad`, `scales.omegaB0`, `algorithms.timestep.dt`", + "value": "`0.1 * omegaB0 * dt * (gamma_qed / gamma_rad)^2 / photon_weight`" + }, + { + "name": "nominal_photon_energy", + "brief": "Nominal energy of the emitted photon for a particle with `gamma * beta = 1`, mass = `m0`", + "type": "float", + "from": "`.gamma_qed`", + "value": "`(1 / gamma_qed)^2`" + } + ] + }, + "properties": { + "gamma_qed": { + "description": "Gamma-factor of a particle emitting inverse Compton photons at energy `m0 c^2` in fiducial magnetic field `B0`", + "type": "number", + "exclusiveMinimum": 1.0, + "default": 10.0, + "x-entity": { + "type": "float [> 1.0]" + } + }, + "photon_energy_min": { + "description": "Minimum photon energy for inverse Compton emission (units of `m0 c^2`)", + "type": "number", + "exclusiveMinimum": 0.0, + "default": 0.0001, + "x-entity": { + "type": "float [> 0.0]", + "default": "1e-4" + } + }, + "photon_weight": { + "description": "Weights for the emitted inverse Compton photons", + "type": "number", + "exclusiveMinimum": 0.0, + "default": 1.0, + "x-entity": { + "type": "float [> 0.0]" + } + }, + "photon_species": { + "description": "Index of species for the emitted photon", + "type": "integer", + "minimum": 1, + "x-entity": { + "type": "ushort [> 0]" + } + } + } + } + } + } + } + }, + "algorithms": { + "description": "Algorithm and solver tuning", + "type": "object", + "additionalProperties": false, + "properties": { + "current_filters": { + "description": "Number of current smoothing passes", + "type": "integer", + "minimum": 0, + "default": 0, + "x-entity": { + "type": "ushort [>= 0]" + } + }, + "timestep": { + "description": "Timestep parameters", + "type": "object", + "additionalProperties": false, + "x-entity": { + "inferred": [ + { + "name": "dt", + "brief": "timestep duration", + "type": "float", + "from": "`algorithms.timestep.CFL`, `scales.dx0`", + "value": "`CFL * dx0`" + } + ] + }, + "properties": { + "CFL": { + "description": "Courant-Friedrichs-Lewy number", + "type": "number", + "exclusiveMinimum": 0.0, + "maximum": 1.0, + "default": 0.95, + "x-entity": { + "type": "float [0.0 -> 1.0]", + "notes": [ + "CFL number determines the timestep duration" + ] + } + }, + "correction": { + "description": "Correction factor for the speed of light used in field solver", + "type": "number", + "default": 1.0, + "x-entity": { + "type": "float" + } + } + } + }, + "deposit": { + "description": "Current deposition parameters", + "type": "object", + "additionalProperties": false, + "x-entity": { + "inferred": [ + { + "name": "order", + "brief": "order of the particle shape function", + "type": "ushort [0 -> 10]", + "from": "compile-time definition `shape_order`" + } + ] + }, + "properties": { + "enable": { + "description": "Enable the current deposition", + "type": "boolean", + "default": true, + "x-entity": { + "type": "bool" + } + }, + "team_policy_team_size": { + "description": "team_policy tiled-deposit work-group (team) size", + "type": "integer", + "minimum": 0, + "default": 0, + "x-entity": { + "type": "uint [>= 0]", + "notes": [ + "0 keeps Kokkos::AUTO (backend occupancy heuristic); a positive value overrides it, clamped to the backend/scratch maximum at launch. Only used in `team_policy=ON` builds. Pick a multiple of the device subgroup width for best occupancy (see ideal_tile_size.py)" + ] + } + } + } + }, + "gr": { + "description": "GR pusher parameters", + "type": "object", + "additionalProperties": false, + "properties": { + "pusher_eps": { + "description": "Stepsize for numerical differentiation in GR pusher", + "type": "number", + "exclusiveMinimum": 0.0, + "default": 1e-06, + "x-entity": { + "type": "float [> 0.0]", + "default": "1e-6" + } + }, + "pusher_niter": { + "description": "Number of iterations for the Newton-Raphson method in GR pusher", + "type": "integer", + "minimum": 1, + "default": 10, + "x-entity": { + "type": "ushort [> 0]" + } + } + } + }, + "gca": { + "description": "Guiding-center approximation parameters", + "type": "object", + "additionalProperties": false, + "properties": { + "e_ovr_b_max": { + "description": "Maximum value for E/B allowed for GCA particles", + "type": "number", + "minimum": 0.0, + "maximum": 1.0, + "default": 0.9, + "x-entity": { + "type": "float [0.0 -> 1.0]" + } + }, + "larmor_max": { + "description": "Maximum Larmor radius allowed for GCA particles (in physical units)", + "type": "number", + "minimum": 0.0, + "default": 0.0, + "x-entity": { + "type": "float", + "notes": [ + "When `larmor_max` == 0, the limit is disabled" + ] + } + } + } + }, + "fieldsolver": { + "description": "Stencil coefficients for the field solver [notation as in Blinne+ (2018)]", + "type": "object", + "additionalProperties": false, + "x-entity": { + "notes": [ + "Standard Yee solver: `delta_i = beta_ij = 0.0`" + ] + }, + "properties": { + "enable": { + "description": "Enable the fieldsolver", + "type": "boolean", + "default": true, + "x-entity": { + "type": "bool" + } + }, + "delta_x": { + "description": "delta_x coefficient (for `F_{i +/- 3/2, j, k}`)", + "type": "number", + "default": 0.0, + "x-entity": { + "type": "float" + } + }, + "delta_y": { + "description": "delta_y coefficient (for `F_{i, j +/- 3/2, k}`)", + "type": "number", + "default": 0.0, + "x-entity": { + "type": "float", + "notes": [ + "Used only for 2D and 3D" + ] + } + }, + "delta_z": { + "description": "delta_z coefficient (for `F_{i, j, k +/- 3/2}`)", + "type": "number", + "default": 0.0, + "x-entity": { + "type": "float", + "notes": [ + "Used only for 3D" + ] + } + }, + "beta_xy": { + "description": "beta_xy coefficient (for `F_{i +/- 1/2, j +/- 1, k}`)", + "type": "number", + "default": 0.0, + "x-entity": { + "type": "float", + "notes": [ + "Used only for 2D and 3D" + ] + } + }, + "beta_yx": { + "description": "beta_yx coefficient (for `F_{i +/- 1, j +/- 1/2, k}`)", + "type": "number", + "default": 0.0, + "x-entity": { + "type": "float", + "notes": [ + "Used only for 2D and 3D" + ] + } + }, + "beta_xz": { + "description": "beta_xz coefficient (for `F_{i +/- 1/2, j, k +/- 1}`)", + "type": "number", + "default": 0.0, + "x-entity": { + "type": "float", + "notes": [ + "Used only for 3D" + ] + } + }, + "beta_zx": { + "description": "beta_zx coefficient (for `F_{i +/- 1, j, k +/- 1/2}`)", + "type": "number", + "default": 0.0, + "x-entity": { + "type": "float", + "notes": [ + "Used only for 3D" + ] + } + }, + "beta_yz": { + "description": "beta_yz coefficient (for `F_{i, j +/- 1/2, k +/- 1}`)", + "type": "number", + "default": 0.0, + "x-entity": { + "type": "float", + "notes": [ + "Used only for 3D" + ] + } + }, + "beta_zy": { + "description": "beta_zy coefficient (for `F_{i, j +/- 1, k +/- 1/2}`)", + "type": "number", + "default": 0.0, + "x-entity": { + "type": "float", + "notes": [ + "Used only for 3D" + ] + } + } + } + } + } + }, + "particles": { + "description": "Particle and species parameters", + "type": "object", + "additionalProperties": false, + "required": [ + "ppc0", + "species" + ], + "x-entity": { + "inferred": [ + { + "name": "nspec", + "brief": "Number of particle species", + "type": "uint", + "from": "`particles.species`" + } + ] + }, + "properties": { + "ppc0": { + "description": "Fiducial number of particles per cell", + "type": "number", + "exclusiveMinimum": 0.0, + "x-entity": { + "type": "float [> 0.0]" + } + }, + "use_weights": { + "description": "Toggle for using particle weights", + "type": "boolean", + "default": false, + "x-entity": { + "type": "bool" + } + }, + "clear_interval": { + "description": "Timesteps between particle re-sorting by tags (removing dead particles)", + "type": "integer", + "minimum": 0, + "default": 100, + "x-entity": { + "type": "uint", + "notes": [ + "Set to 0 to disable re-sorting" + ] + } + }, + "spatial_sorting_interval": { + "description": "Timesteps between spatial sorting of particles (for better cache performance)", + "type": "integer", + "minimum": 0, + "default": 0, + "x-entity": { + "type": "uint", + "notes": [ + "Set to 0 to disable spatial sorting" + ] + } + }, + "species": { + "description": "Particle species definitions", + "type": "array", + "minItems": 1, + "x-entity": { + "array_of_tables": true + }, + "items": { + "type": "object", + "additionalProperties": false, + "required": [ + "mass", + "charge", + "maxnpart" + ], + "properties": { + "label": { + "description": "Label of the species", + "type": "string", + "x-entity": { + "type": "string", + "default": "\"s\"", + "notes": [ + "`` is the index of the species in the list starting from 1" + ], + "examples": [ + "\"e-\"" + ] + } + }, + "mass": { + "description": "Mass of the species (in units of fiducial mass)", + "type": "number", + "minimum": 0.0, + "x-entity": { + "type": "float [>= 0.0]" + } + }, + "charge": { + "description": "Charge of the species (in units of fiducial charge)", + "type": "number", + "x-entity": { + "type": "float" + } + }, + "maxnpart": { + "description": "Maximum number of particles per task", + "type": "number", + "x-entity": { + "type": "uint [> 0]", + "notes": [ + "Read as a float, so exponential notation is fine (e.g. `1e8`)" + ] + }, + "exclusiveMinimum": 0 + }, + "pusher": { + "description": "Pusher algorithm for the species", + "anyOf": [ + { + "enum": [ + "Boris", + "Vay", + "Boris,GCA", + "Vay,GCA", + "Photon", + "None" + ] + }, + { + "type": "string", + "pattern": "(?i)^(Boris|Vay|Boris,GCA|Vay,GCA|Photon|None)$" + } + ], + "x-entity": { + "type": "string", + "default": "\"Boris\" [massive]; \"Photon\" [massless]" + } + }, + "n_payloads_real": { + "description": "Number of additional real-valued variables (payloads) for each particle of the given species", + "type": "integer", + "minimum": 0, + "default": 0, + "x-entity": { + "type": "ushort" + } + }, + "n_payloads_int": { + "description": "Number of additional integer-valued variables (payloads) for each particle of the given species", + "type": "integer", + "minimum": 0, + "default": 0, + "x-entity": { + "type": "ushort", + "notes": [ + "If tracking is enabled, one or two extra integer payloads are reserved (depending on whether MPI is enabled)" + ] + } + }, + "tracking": { + "description": "Enable tracking of particles using indices for the given species", + "type": "boolean", + "default": false, + "x-entity": { + "type": "bool" + } + }, + "radiative_drag": { + "description": "Radiation reaction to use for the species", + "type": "string", + "pattern": "(?i)^(None|(Synchrotron|Compton)(,(Synchrotron|Compton))*)$", + "default": "None", + "x-entity": { + "type": "string", + "enum": [ + "None", + "Synchrotron", + "Compton" + ], + "notes": [ + "Can also be coma-separated combination, e.g., \"Synchrotron,Compton\"", + "Relevant radiation.drag parameters should also be provided" + ] + } + }, + "emission": { + "description": "Particle emission policy for the species", + "anyOf": [ + { + "enum": [ + "None", + "Synchrotron", + "Compton" + ] + }, + { + "type": "string", + "pattern": "(?i)^(None|Synchrotron|Compton)$" + } + ], + "default": "None", + "x-entity": { + "type": "string", + "notes": [ + "Only one emission mechanism allowed", + "Appropriate radiation drag flag will be applied automatically (unless explicitly set to \"None\")" + ] + } + }, + "spatial_sorting_interval": { + "description": "Timesteps between spatial sorting of particles for given species", + "type": "integer", + "minimum": 0, + "default": 0, + "x-entity": { + "type": "uint", + "notes": [ + "Set to 0 to disable spatial sorting", + "Overrides `particles.spatial_sorting_interval` for the given species" + ] + } + }, + "clear_interval": { + "description": "Timesteps between particle re-sorting by tags (removing dead particles)", + "type": "integer", + "minimum": 0, + "default": 100, + "x-entity": { + "type": "uint", + "notes": [ + "Set to 0 to disable re-sorting", + "Overrides `particles.clear_interval` for the given species" + ] + } + } + } + } + } + } + }, + "setup": { + "description": "Parameters for specific problem generators and setups", + "type": "object", + "additionalProperties": true, + "x-entity": { + "notes": [ + "Free-form: keys are defined by the problem generator, so nothing here is validated" + ] + } + }, + "output": { + "description": "Output parameters", + "type": "object", + "additionalProperties": false, + "properties": { + "format": { + "description": "Output format", + "anyOf": [ + { + "enum": [ + "disabled", + "hdf5", + "BPFile" + ] + }, + { + "type": "string", + "pattern": "(?i)^(disabled|hdf5|BPFile)$" + } + ], + "default": "hdf5", + "x-entity": { + "type": "string" + } + }, + "interval": { + "description": "Number of timesteps between all outputs", + "type": "integer", + "minimum": 1, + "default": 1, + "x-entity": { + "type": "uint [> 0]", + "notes": [ + "Value is overriden by output intervals for specific outputs" + ] + } + }, + "interval_time": { + "description": "Physical (code) time interval between all outputs", + "type": "number", + "default": -1.0, + "x-entity": { + "type": "float", + "notes": [ + "When `interval_time` < 0, the output is controlled by `interval`, otherwise by `interval_time`", + "Value is overriden by output intervals for specific outputs" + ] + } + }, + "separate_files": { + "description": "Whether to output each timestep into separate files", + "type": "boolean", + "default": true, + "deprecated": true, + "x-entity": { + "type": "bool", + "deprecated": "starting v1.3.0" + } + }, + "fields": { + "description": "Field output parameters", + "type": "object", + "additionalProperties": false, + "properties": { + "enable": { + "description": "Toggle for the field output", + "type": "boolean", + "default": true, + "x-entity": { + "type": "bool" + } + }, + "quantities": { + "description": "Field quantities to output", + "type": "array", + "items": { + "type": "string" + }, + "default": [], + "x-entity": { + "type": "array", + "enum": [ + "E", + "B", + "J", + "divE", + "Rho", + "Charge", + "N", + "Nppc", + "T0i", + "Tij", + "Vi", + "D", + "H", + "divD", + "A" + ], + "notes": [ + "For `T`, you can use unspecified indices: `Tij`, `T0i`, or specific ones: `Ttt`, `T00`, `T02`, `T23`", + "For `T`, in cartesian can also use \"x\" \"y\" \"z\" instead of \"1\" \"2\" \"3\"", + "By default, we accumulate moments from all massive species, one can specify only specific species: `Ttt_1_2`, `Rho_1`, `Rho_3_4`" + ] + } + }, + "custom": { + "description": "Custom (user-defined) field quantities", + "type": "array", + "items": { + "type": "string" + }, + "default": [], + "x-entity": { + "type": "array" + } + }, + "interval": { + "description": "Number of timesteps between field outputs", + "type": "integer", + "minimum": 0, + "default": 0, + "x-entity": { + "type": "uint", + "notes": [ + "When `!= 0`, overrides `output.interval`", + "When `== 0`, `output.interval` is used" + ] + } + }, + "interval_time": { + "description": "Physical (code) time interval between field outputs", + "type": "number", + "default": -1.0, + "x-entity": { + "type": "float", + "notes": [ + "When `< 0`, the output is controlled by `interval`", + "When specified, overrides `output.interval_time`" + ] + } + }, + "downsampling": { + "description": "Downsample factor for the output of fields", + "anyOf": [ + { + "type": "integer", + "minimum": 1 + }, + { + "type": "array", + "minItems": 1, + "maxItems": 3, + "items": { + "type": "integer", + "minimum": 1 + } + } + ], + "default": [ + 1, + 1, + 1 + ], + "x-entity": { + "type": "uint | array [>= 1]", + "notes": [ + "The output is downsampled by the given factors in each direction", + "If a scalar is given, it is applied to all directions" + ] + } + }, + "smoothing": { + "description": "Smoothing of the output moments", + "type": "object", + "additionalProperties": false, + "properties": { + "order": { + "description": "Smoothing order for the output of moments (\"Rho\", \"Charge\", \"T\", ...)", + "type": "integer", + "minimum": 0, + "default": 0, + "x-entity": { + "type": "ushort" + } + }, + "method": { + "description": "Smoothing algorithm", + "anyOf": [ + { + "enum": [ + "const", + "spline" + ] + }, + { + "type": "string", + "pattern": "(?i)^(const|spline)$" + } + ], + "default": "spline", + "x-entity": { + "type": "string", + "notes": [ + "When using \"spline\", `order` corresponds to the order of the polynomial used", + "When using \"const\", the smoothing window is `ceil(order / 2)` in both directions" + ] + } + } + } + } + } + }, + "particles": { + "description": "Particle output parameters", + "type": "object", + "additionalProperties": false, + "properties": { + "enable": { + "description": "Toggle for the particles output", + "type": "boolean", + "default": true, + "x-entity": { + "type": "bool" + } + }, + "species": { + "description": "Particle species indices to output", + "type": "array", + "items": { + "type": "integer", + "minimum": 1 + }, + "default": [], + "x-entity": { + "type": "array", + "notes": [ + "If empty, all species are output" + ] + } + }, + "stride": { + "description": "Stride for the output of particles", + "type": "integer", + "exclusiveMinimum": 1, + "default": 100, + "x-entity": { + "type": "uint [> 1]" + } + }, + "interval": { + "description": "Number of timesteps between particle outputs", + "type": "integer", + "minimum": 0, + "default": 0, + "x-entity": { + "type": "uint", + "notes": [ + "When `!= 0`, overrides `output.interval`", + "When `== 0`, `output.interval` is used" + ] + } + }, + "interval_time": { + "description": "Physical (code) time interval between particle outputs", + "type": "number", + "default": -1.0, + "x-entity": { + "type": "float", + "notes": [ + "When `< 0`, the output is controlled by `interval`", + "When specified, overrides `output.interval_time`" + ] + } + } + } + }, + "spectra": { + "description": "Spectra output parameters", + "type": "object", + "additionalProperties": false, + "properties": { + "enable": { + "description": "Toggle for the spectra output", + "type": "boolean", + "default": true, + "x-entity": { + "type": "bool" + } + }, + "e_min": { + "description": "Minimum energy for the spectra output", + "type": "number", + "exclusiveMinimum": 0.0, + "default": 0.001, + "x-entity": { + "type": "float", + "default": "1e-3" + } + }, + "e_max": { + "description": "Maximum energy for the spectra output", + "type": "number", + "exclusiveMinimum": 0.0, + "default": 1000.0, + "x-entity": { + "type": "float", + "default": "1e3" + } + }, + "log_bins": { + "description": "Whether to use logarithmic bins for energy", + "type": "boolean", + "default": true, + "x-entity": { + "type": "bool" + } + }, + "num_bins": { + "description": "Number of energy bins for the spectra output", + "type": "integer", + "minimum": 1, + "default": 200, + "deprecated": true, + "x-entity": { + "type": "uint [> 0]", + "deprecated": "starting v1.5.0" + } + }, + "num_energy_bins": { + "description": "Number of energy bins for the spectra output", + "type": "integer", + "minimum": 1, + "default": 200, + "x-entity": { + "type": "uint [> 0]" + } + }, + "num_spatial_bins": { + "description": "Number of spatial bins for the spectra output", + "type": "array", + "minItems": 1, + "maxItems": 3, + "items": { + "type": "integer", + "minimum": 1 + }, + "default": [ + 1, + 1, + 1 + ], + "x-entity": { + "type": "array [size 1 :->: 3]" + } + }, + "interval": { + "description": "Number of timesteps between spectra outputs", + "type": "integer", + "minimum": 0, + "default": 0, + "x-entity": { + "type": "uint", + "notes": [ + "When `!= 0`, overrides `output.interval`", + "When `== 0`, `output.interval` is used" + ] + } + }, + "interval_time": { + "description": "Physical (code) time interval between spectra outputs", + "type": "number", + "default": -1.0, + "x-entity": { + "type": "float", + "notes": [ + "When `< 0`, the output is controlled by `interval`", + "When specified, overrides `output.interval_time`" + ] + } + } + } + }, + "debug": { + "description": "Debug output parameters", + "type": "object", + "additionalProperties": false, + "properties": { + "as_is": { + "description": "Output fields \"as is\" without conversions", + "type": "boolean", + "default": false, + "x-entity": { + "type": "bool" + } + }, + "ghosts": { + "description": "Output fields with values in ghost cells", + "type": "boolean", + "default": false, + "x-entity": { + "type": "bool" + } + } + } + }, + "stats": { + "description": "Integrated statistics output parameters", + "type": "object", + "additionalProperties": false, + "properties": { + "enable": { + "description": "Toggle for the stats output", + "type": "boolean", + "default": true, + "x-entity": { + "type": "bool" + } + }, + "interval": { + "description": "Number of timesteps between stat outputs", + "type": "integer", + "minimum": 1, + "default": 100, + "x-entity": { + "type": "uint [> 0]", + "notes": [ + "Overriden if `output.stats.interval_time != -1`" + ] + } + }, + "interval_time": { + "description": "Physical (code) time interval between stat outputs", + "type": "number", + "default": -1.0, + "x-entity": { + "type": "float", + "notes": [ + "When `< 0`, the output is controlled by `interval`" + ] + } + }, + "quantities": { + "description": "Field quantities to output", + "type": "array", + "items": { + "type": "string" + }, + "default": [ + "B^2", + "E^2", + "ExB", + "Rho", + "T00" + ], + "x-entity": { + "type": "array", + "enum": [ + "B^2", + "E^2", + "ExB", + "N", + "Npart", + "Charge", + "Rho", + "T00", + "T0i", + "Tij" + ], + "notes": [ + "For particle moments, ...", + "... same notation is used as for `output.fields.quantities`" + ] + } + }, + "custom": { + "description": "Custom (user-defined) stats", + "type": "array", + "items": { + "type": "string" + }, + "default": [], + "x-entity": { + "type": "array" + } + } + } + }, + "render": { + "description": "In-situ renderer. Renders scalar fields on the GPU and writes PNG images directly to `/renders/` each cadence -- no field data is written to storage, and the result is seamless across MPI domain boundaries.", + "type": "object", + "additionalProperties": false, + "x-entity": { + "notes": [ + "two modes, selected automatically by the simulation dimension:\n- 3D Cartesian (Minkowski): volume ray-march (uses `samples`, `step_size`, `early_term_alpha`, and the [camera] table)\n- 2D (Minkowski, Spherical/QSpherical, and all GR Kerr-Schild): flat slice rasterizer. Cartesian shows the (x, y) plane; spherical shows the meridional (r, theta) half-plane mapped to Cartesian (X = r sin th, Z = r cos th), optionally mirrored (see `mirror`). The `samples`/`step_size`/`early_term_alpha`/[camera] keys are ignored in 2D (one opaque sample per pixel).", + "1D (and 3D non-Cartesian, which does not exist) is a no-op", + "One PNG stream per scene (e.g. a density/|B|/|J| triptych)" + ] + }, + "properties": { + "enable": { + "description": "Toggle for the volume renderer", + "type": "boolean", + "default": false, + "x-entity": { + "type": "bool" + } + }, + "interval": { + "description": "Number of timesteps between renders", + "type": "integer", + "minimum": 0, + "default": 0, + "x-entity": { + "type": "uint", + "notes": [ + "When `!= 0`, overrides `output.interval`", + "When `== 0`, `interval_time` (or `output.interval`) is used" + ] + } + }, + "interval_time": { + "description": "Physical (code) time interval between renders", + "type": "number", + "default": -1.0, + "x-entity": { + "type": "float", + "notes": [ + "When `< 0`, the output is controlled by `interval`" + ] + } + }, + "width": { + "description": "Image width in pixels (the rendered region; the PNG is wider if a colorbar margin is added, see `colorbar_outside`)", + "type": "integer", + "minimum": 1, + "default": 1024, + "x-entity": { + "type": "int [> 0]" + } + }, + "height": { + "description": "Image height in pixels", + "type": "integer", + "minimum": 1, + "default": 1024, + "x-entity": { + "type": "int [> 0]" + } + }, + "resolution": { + "description": "Convenience: force a square frame (sets width == height == resolution), the natural shape for a dome master. Overrides `width`/`height` when > 0.", + "type": "integer", + "minimum": 0, + "default": 0, + "x-entity": { + "type": "int [> 0]", + "default": "0 (use width/height)" + } + }, + "x1_lim": { + "description": "Axis-aligned render region [lo, hi] along x1, in physical/world coords. Left unset it spans the full domain. Clamped to the box.", + "anyOf": [ + { + "type": "array", + "maxItems": 0 + }, + { + "type": "array", + "minItems": 2, + "maxItems": 2, + "prefixItems": [ + { + "type": "number" + }, + { + "type": "number" + } + ] + } + ], + "default": [], + "x-entity": { + "type": "array [size 2]", + "default": "[] (full extent)", + "notes": [ + "3D -> the volume is depth-clipped to this box, the wireframe/axes frame it, and the default camera zooms to it; 2D -> the slice window is framed to it. For a spherical 2D slice, x1_lim crops the radius r and x2_lim crops the polar angle theta." + ], + "examples": [ + "x1_lim = [-64.0, 64.0]" + ] + } + }, + "x2_lim": { + "description": "Axis-aligned render region [lo, hi] along x2, in physical/world coords. Left unset it spans the full domain. Clamped to the box.", + "anyOf": [ + { + "type": "array", + "maxItems": 0 + }, + { + "type": "array", + "minItems": 2, + "maxItems": 2, + "prefixItems": [ + { + "type": "number" + }, + { + "type": "number" + } + ] + } + ], + "default": [], + "x-entity": { + "type": "array [size 2]", + "default": "[] (full extent)", + "notes": [ + "For a spherical 2D slice, x2_lim crops the polar angle theta" + ] + } + }, + "x3_lim": { + "description": "Axis-aligned render region [lo, hi] along x3, in physical/world coords. Left unset it spans the full domain. Clamped to the box.", + "anyOf": [ + { + "type": "array", + "maxItems": 0 + }, + { + "type": "array", + "minItems": 2, + "maxItems": 2, + "prefixItems": [ + { + "type": "number" + }, + { + "type": "number" + } + ] + } + ], + "default": [], + "x-entity": { + "type": "array [size 2]", + "default": "[] (full extent)" + } + }, + "camera_velocity": { + "description": "Moving view: translate the render region (and, in 3D, the camera) at this velocity in world-units-per-sim-time, to keep a propagating feature (e.g. a shock) in frame. Pair with x{1,2,3}_lim (the moving window). 2D uses the first two components; the motion starts at `camera_start_time`.", + "anyOf": [ + { + "type": "array", + "maxItems": 0 + }, + { + "type": "array", + "minItems": 2, + "maxItems": 3, + "items": { + "type": "number" + } + } + ], + "default": [], + "x-entity": { + "type": "array [size 2 or 3]", + "default": "[] (static view)", + "examples": [ + "camera_velocity = [0.9, 0.0] # pan along +x1 at 0.9 c" + ] + } + }, + "camera_start_time": { + "description": "Sim time at which the view starts moving (static before it, e.g. to let an initial ramp-up finish)", + "type": "number", + "default": 0.0, + "x-entity": { + "type": "float" + } + }, + "samples": { + "description": "Number of ray-march steps across the global box diagonal", + "type": "integer", + "minimum": 1, + "default": 400, + "x-entity": { + "type": "int [> 0]", + "notes": [ + "The world-space step is `box_diagonal / samples` unless `step_size` is set. Higher = better quality, slower." + ] + } + }, + "step_size": { + "description": "Fixed world-space step between ray samples", + "type": "number", + "minimum": 0.0, + "default": 0.0, + "x-entity": { + "type": "float [>= 0.0]", + "notes": [ + "0 derives the step from `samples`. The step is identical on all ranks, which is what makes the multi-domain composite seamless." + ] + } + }, + "early_term_alpha": { + "description": "Stop marching a ray once its accumulated opacity reaches this value", + "type": "number", + "minimum": 0.0, + "maximum": 1.0, + "default": 0.99, + "x-entity": { + "type": "float [0.0 -> 1.0]", + "notes": [ + "Pure speed optimization; set to 1.0 to disable early termination" + ] + } + }, + "n_lut": { + "description": "Number of entries in the color/opacity lookup table", + "type": "integer", + "exclusiveMinimum": 1, + "default": 256, + "x-entity": { + "type": "int [> 1]" + } + }, + "background": { + "description": "Opaque background RGB (each channel 0..1) shown through transparent/low-opacity pixels; also fills the colorbar margin", + "type": "array", + "minItems": 3, + "maxItems": 3, + "items": { + "type": "number", + "minimum": 0.0, + "maximum": 1.0 + }, + "default": [ + 0.0, + 0.0, + 0.0 + ], + "x-entity": { + "type": "array [size 3]" + } + }, + "colorbar": { + "description": "Draw a colorbar (gradient + value ticks + label) on each PNG", + "type": "boolean", + "default": true, + "x-entity": { + "type": "bool" + } + }, + "colorbar_outside": { + "description": "Draw the colorbar in an added right margin (the PNG becomes wider by a fixed strip) instead of overlaying it on the rendered volume", + "type": "boolean", + "default": true, + "x-entity": { + "type": "bool" + } + }, + "mirror": { + "description": "2D spherical slice only: mirror the meridional half-plane across the symmetry axis to render a full disk from one axisymmetric half. No effect on Cartesian or 3D rendering.", + "type": "boolean", + "default": true, + "x-entity": { + "type": "bool" + } + }, + "time_label": { + "description": "Draw the current simulation time as a label (\"T = \", fixed to 2 decimals) in the upper-right corner of the render region, in a contrasting color, vertically centered between the frame top and the colorbar.", + "type": "boolean", + "default": false, + "x-entity": { + "type": "bool" + } + }, + "axes": { + "description": "Draw a spine (frame) + axis ticks + labels around the rendered region. The PNG gains left/bottom margins (background-filled) for the tick labels and axis names, so they never overlap the data.", + "type": "boolean", + "default": false, + "x-entity": { + "type": "bool", + "notes": [ + "2D Cartesian = a rectangular frame with linear spatial ticks;\n2D spherical = polar axes (an \"R\" radial axis on the symmetry axis with R=0 centered, and a \"Theta\" axis along the curved outline / spine);\n3D = the global box projected to a wireframe with ticks on the three silhouette edges (x bottom, y & z on the left)" + ] + } + }, + "axis_labels": { + "description": "Axis names. 3D uses all three; the 2D slice uses the first two. When unset, the 2D slice defaults to \"x\",\"y\" (Cartesian) or \"X\",\"Z\" (spherical).", + "type": "array", + "maxItems": 3, + "items": { + "type": "string" + }, + "default": [ + "x", + "y", + "z" + ], + "x-entity": { + "type": "array [size <= 3]" + } + }, + "axis_ticks": { + "description": "Target number of ticks per axis (actual count is rounded to nice values)", + "type": "integer", + "minimum": 2, + "default": 5, + "x-entity": { + "type": "int [>= 2]" + } + }, + "spine_width": { + "description": "3D only: target width (pixels) of the box wireframe \"spine\". The spine is drawn inside the ray-march (opaque, depth-occluded by the volume); its width is floored by the ray step, so for a crisper thin line raise `samples` as well.", + "type": "number", + "exclusiveMinimum": 0.0, + "default": 2.0, + "x-entity": { + "type": "float [> 0.0]" + } + }, + "camera": { + "description": "Camera (3D volume mode only; ignored by the 2D slice rasterizer). Defaults frame the whole global box from outside, looking down the (1,1,1) diagonal -- the production setup for which the structured composite is provably seamless.", + "type": "object", + "additionalProperties": false, + "properties": { + "mode": { + "description": "Projection mode. Overrides `orthographic` below when set.", + "anyOf": [ + { + "enum": [ + "orthographic", + "perspective", + "dome" + ] + }, + { + "type": "string", + "pattern": "(?i)^(orthographic|perspective|dome)$" + } + ], + "x-entity": { + "type": "string", + "default": "(unset -> use `orthographic`)", + "notes": [ + "\"dome\" is a fulldome azimuthal-equidistant fisheye rendered from an INTERIOR eye (the domain center by default), i.e. a 3D planetarium dome master. It uses a depth-resolved (A-buffer) composite that is seamless across a full 3D domain decomposition (unlike ortho/perspective, which need the eye outside the box). Set a square frame (`resolution`, or width == height). `forward` is the dome ZENITH (screen-up defaults to +y for a +z zenith)." + ] + } + }, + "orthographic": { + "description": "Orthographic (true) or perspective (false) projection", + "type": "boolean", + "default": true, + "x-entity": { + "type": "bool", + "notes": [ + "Orthographic is recommended; the seamless composite is always valid for it. Perspective is only seamless with the eye outside the box." + ] + } + }, + "position": { + "description": "Camera (eye) position in world (physical) coordinates", + "type": "array", + "minItems": 3, + "maxItems": 3, + "items": { + "type": "number" + }, + "x-entity": { + "type": "array [size 3]", + "default": "box center pushed back ~1.7 box-diagonals along (1, 1, 1);\nfor `mode = \"dome\"`, the domain center (interior eye)" + } + }, + "look_at": { + "description": "Point the camera looks at, in world coordinates (the dome ZENITH target)", + "type": "array", + "minItems": 3, + "maxItems": 3, + "items": { + "type": "number" + }, + "x-entity": { + "type": "array [size 3]", + "default": "box center; for `mode = \"dome\"`, the zenith defaults to +z" + } + }, + "up": { + "description": "Camera up vector (dome: the disk's screen-up)", + "type": "array", + "minItems": 3, + "maxItems": 3, + "items": { + "type": "number" + }, + "default": [ + 0.0, + 0.0, + 1.0 + ], + "x-entity": { + "type": "array [size 3]", + "default": "[0.0, 0.0, 1.0]; for `mode = \"dome\"`, [0.0, 1.0, 0.0]" + } + }, + "fov": { + "description": "Vertical field of view in degrees (perspective only)", + "type": "number", + "exclusiveMinimum": 0.0, + "default": 35.0, + "x-entity": { + "type": "float [> 0.0]" + } + }, + "dome_fov": { + "description": "Full dome field of view in degrees (dome mode only): the image rim is at dome_fov/2 from the zenith (180 = a full hemisphere down to the horizon).", + "type": "number", + "exclusiveMinimum": 0.0, + "maximum": 360.0, + "default": 180.0, + "x-entity": { + "type": "float [> 0.0, <= 360.0]" + } + }, + "dome_radius": { + "description": "Dome far-clip radius in world units (dome mode only): each ray stops this far from the eye, so the sampled region is a half-ball (hemisphere) of this radius rather than the whole box -> uniform path length and no box corner/edge projection artifacts. `samples` then counts steps across this radius.", + "type": "number", + "minimum": 0.0, + "x-entity": { + "type": "float [>= 0.0]", + "default": "the largest sphere centered in the box (half the shortest side), so it touches the face centers and never a corner", + "notes": [ + "0 disables the clip (rays march to the box boundary)" + ] + } + }, + "ortho_height": { + "description": "Vertical extent of the view in world units (orthographic only)", + "type": "number", + "exclusiveMinimum": 0.0, + "x-entity": { + "type": "float [> 0.0]", + "default": "the global box diagonal (the whole box fits from any angle)" + } + } + } + }, + "dome": { + "description": "Fulldome fisheye (\"planetarium dome master\"). 2D only; a circular image is centered in the frame's inscribed circle with the corners left as the background (the dome master's black border). Set `width == height` (e.g. 4096) for a square master. When enabled, the axes and the outside colorbar strip are suppressed so the PNG stays exactly width x height. Seamless across MPI domains (the pixel->world map is a shared, deterministic function and the tiles stay disjoint). Ignored (with a warning) for 3D.", + "type": "object", + "additionalProperties": false, + "x-entity": { + "notes": [ + "CARTESIAN -- the flat plane is warped radially into the disk; use `fov`/`radius`/`center`/`projection` below.", + "SPHERICAL / GR Kerr-Schild -- the meridional slice is ALREADY a disk, so dome mode only mirrors it to a full disk (see `mirror` above; keep it true) and fits it to the inscribed circle. The `fov`/`radius`/`center`/`projection` keys are ignored (the native (X, Z) meridional map is used, with image radius proportional to the physical radius r, r=0 at the disk center)." + ] + }, + "properties": { + "enable": { + "description": "Build the fisheye dome master instead of the plain slice", + "type": "boolean", + "default": false, + "x-entity": { + "type": "bool" + } + }, + "fov": { + "description": "(Cartesian only) Full dome field of view in degrees (image radius maps linearly to the dome zenith angle: the rim is at fov/2)", + "type": "number", + "exclusiveMinimum": 0.0, + "maximum": 180.0, + "default": 180.0, + "x-entity": { + "type": "float [> 0.0, <= 180.0]", + "default": "180.0 # a full hemisphere" + } + }, + "radius": { + "description": "(Cartesian only) World radius of the circular cutout mapped onto the dome", + "type": "number", + "exclusiveMinimum": 0.0, + "x-entity": { + "type": "float [> 0.0]", + "default": "half the shorter domain side (the largest centered disk that fits inside the box)" + } + }, + "center": { + "description": "(Cartesian only) World-space center of the cutout", + "type": "array", + "minItems": 2, + "maxItems": 2, + "items": { + "type": "number" + }, + "x-entity": { + "type": "array [size 2]", + "default": "the domain center" + } + }, + "projection": { + "description": "(Cartesian only) How the dome zenith angle maps to a world radius on the flat slice", + "anyOf": [ + { + "enum": [ + "equidistant", + "gnomonic", + "stereographic", + "orthographic" + ] + }, + { + "type": "string", + "pattern": "(?i)^(equidistant|gnomonic|stereographic|orthographic)$" + } + ], + "default": "equidistant", + "x-entity": { + "type": "string", + "enum": [ + "\"equidistant\" (r proportional to angle; the fulldome image standard -- a straight radial scaling of the cutout)", + "\"gnomonic\" (r ~ tan(angle); the slice as a flat \"ceiling\" tangent to the dome -- straight sim lines stay straight)", + "\"stereographic\" (r ~ tan(angle/2); conformal, preserves shapes)", + "\"orthographic\" (r ~ sin(angle); the slice as seen face-on)" + ] + } + } + } + }, + "scenes": { + "description": "One scene per scalar field -> one PNG stream. Repeat the table for each.", + "type": "array", + "x-entity": { + "array_of_tables": true + }, + "items": { + "type": "object", + "additionalProperties": false, + "required": [ + "field" + ], + "properties": { + "field": { + "description": "Scalar field to render (a volume render needs a scalar, so vectors are given as a magnitude or a single component)", + "type": "string", + "x-entity": { + "type": "string", + "enum": [ + "(fields): \"{E,B,J}mag\"; \"{E,B,J}{1,2,3}\" or \"{E,B,J}{x,y,z}\"", + "(moments): \"N\", \"Nppc\", \"Rho\", \"Charge\"; \"T{i}{j}\"; \"V{i}\"; \"Vmag\"" + ], + "notes": [ + "\"{E,B,J}mag\" = vector magnitude |.|; \"B1\"/\"Bx\", \"J3\"/\"Jz\", ... = a single (signed) physical component", + "a bare vector (\"E\"/\"B\"/\"J\") is not renderable -- choose a component or the magnitude", + "\"N\"/\"Nppc\" = number / per-cell count, \"Rho\" = mass density, \"Charge\" = charge density", + "\"T{i}{j}\" = one stress-energy component, i,j in {t,x,y,z} or {0,1,2,3} (e.g. \"Txx\", \"Ttt\", \"T0x\"); \"V{i}\" = one bulk-velocity component, i in {x,y,z} or {1,2,3} (e.g. \"Vx\", \"V1\"); \"Vmag\" = bulk-velocity magnitude sqrt(V1^2+V2^2+V3^2)", + "moments follow the engine: SRPIC = tetrad-basis bulk 3-velocity and stress-energy; GRPIC = Eckart-frame 4-velocity (so \"Vt\"/\"V0\" = u^0 = Gamma/alpha is also valid) and contravariant T", + "per-species selection with a \"_\" suffix on moments, e.g. \"N_1\", \"Rho_2\", \"Txy_1_2\", \"V1_3\"; default = all massive species", + "components are signed; pair a symmetric `min`/`max` with a diverging colormap (\"cool2warm\") to center zero", + "\"fieldlines\" renders the magnetic field-line tubes on their own (no scalar volume sampled); see [output.render.fieldlines] below" + ] + } + }, + "prefix": { + "description": "PNG filename prefix; files are `.png`", + "type": "string", + "x-entity": { + "type": "string", + "default": "\"_\"" + } + }, + "label": { + "description": "Colorbar title", + "type": "string", + "x-entity": { + "type": "string", + "default": "`field`" + } + }, + "min": { + "description": "Lower bound of the value range mapped onto the colormap/opacity", + "type": "number", + "default": 0.0, + "x-entity": { + "type": "float" + } + }, + "max": { + "description": "Upper bound of the value range", + "type": "number", + "default": 1.0, + "x-entity": { + "type": "float" + } + }, + "log": { + "description": "Map the value range logarithmically", + "type": "boolean", + "default": false, + "x-entity": { + "type": "bool", + "notes": [ + "Requires min > 0 and max > 0" + ] + } + }, + "colormap": { + "description": "Colormap name", + "$ref": "#/$defs/colormap", + "default": "viridis" + }, + "alpha": { + "description": "Opacity transfer function: [position, opacity] control points, both in [0, 1], piecewise-linear in the normalized value", + "type": "array", + "items": { + "type": "array", + "minItems": 2, + "maxItems": 2, + "prefixItems": [ + { + "type": "number", + "minimum": 0.0, + "maximum": 1.0 + }, + { + "type": "number", + "minimum": 0.0, + "maximum": 1.0 + } + ] + }, + "x-entity": { + "type": "array>", + "default": "linear ramp (opacity = normalized value)", + "notes": [ + "Keep the low end near 0 so empty regions stay transparent" + ], + "examples": [ + "[[0.0, 0.0], [0.3, 0.1], [1.0, 0.7]]" + ] + } + }, + "colorbar_ticks": { + "description": "Explicit value(s) to label on the colorbar", + "type": "array", + "items": { + "type": "number" + }, + "x-entity": { + "type": "array", + "default": "5 evenly-spaced ticks between min and max", + "notes": [ + "Values outside [min, max] are skipped" + ], + "examples": [ + "[0.0, 0.5, 1.0]" + ] + } + }, + "fieldlines": { + "description": "Overlay the magnetic field-line tubes inside this scene's volume", + "type": "boolean", + "default": false, + "x-entity": { + "type": "bool", + "notes": [ + "requires the [output.render.fieldlines] table below (3D only). A scene with field = \"fieldlines\" instead renders them alone." + ] + } + } + } + } + }, + "fieldlines": { + "description": "Magnetic field lines, drawn from a coarse, MPI-replicated copy of the field so the geometry is global and seamless across domains (the coarsening is what makes this cheap -- no parallel particle advection / flux scan).\n- 3D (Cartesian): traced as solid tubes, colored by |field|, composited inside the volume ray-march so the volume correctly occludes them.\n- 2D (Cartesian): iso-contours of the flux function psi (Bx = d psi/dy, By = -d psi/dx), i.e. the in-plane field lines, colored by |B|.\n- 2D (spherical / Kerr-Schild): traced meridional streamlines of the poloidal (Br, Btheta) field (nt2py style).\nBuilt once per frame and shared by every scene that opts in (per-scene `fieldlines = true`) and by any standalone `field = \"fieldlines\"` scene.", + "type": "object", + "additionalProperties": false, + "properties": { + "enable": { + "description": "Build the field-line geometry this run", + "type": "boolean", + "default": false, + "x-entity": { + "type": "bool", + "notes": [ + "implied true if any scene sets `fieldlines = true` or uses `field = \"fieldlines\"`" + ] + } + }, + "field": { + "description": "Vector field to trace", + "anyOf": [ + { + "enum": [ + "B", + "E", + "J" + ] + }, + { + "type": "string", + "pattern": "(?i)^(B|E|J)$" + } + ], + "default": "B", + "x-entity": { + "type": "string" + } + }, + "bin": { + "description": "Field coarsening factor (simulation cells per coarse cell, per axis)", + "type": "integer", + "minimum": 1, + "maximum": 16, + "default": 4, + "x-entity": { + "type": "int [1..16]", + "notes": [ + "larger = smoother \"morphology\" lines + cheaper replication (the coarse field is ~ N_cells / bin^D floats/rank; D = sim dimension)" + ] + } + }, + "seed_px": { + "description": "(3D tubes) Seed-lattice spacing in screen pixels (sets line density)", + "type": "number", + "exclusiveMinimum": 0.0, + "default": 8, + "x-entity": { + "type": "float [> 0]", + "notes": [ + "capped by `seed_max`; if seed_px asks for more seeds than that, the spacing grows to fit and seed_px no longer governs" + ] + } + }, + "seed_max": { + "description": "(3D tubes) Hard cap on the seed count (lattice is n^3, 2 lines per seed)", + "type": "integer", + "minimum": 1, + "default": 4096, + "x-entity": { + "type": "int [> 0]", + "notes": [ + "lower this for fewer / more widely spaced lines" + ] + } + }, + "levels": { + "description": "(2D contours) Number of evenly-spaced flux-function contour levels", + "type": "integer", + "minimum": 1, + "default": 16, + "x-entity": { + "type": "int [> 0]", + "notes": [ + "evenly-spaced psi levels => line density tracks |B| automatically" + ] + } + }, + "tube_px": { + "description": "Tube radius (3D) / contour line width (2D), in screen pixels", + "type": "number", + "exclusiveMinimum": 0.0, + "default": 2, + "x-entity": { + "type": "float [> 0]" + } + }, + "colormap": { + "description": "Colormap for the field lines (mapped by |B| along each line)", + "$ref": "#/$defs/colormap", + "default": "inferno" + }, + "color": { + "description": "Monochrome override: draw the lines in a single [r,g,b] color (each 0..1) instead of the |B| colormap -- reads well as an overlay on another volume", + "anyOf": [ + { + "type": "array", + "maxItems": 0 + }, + { + "type": "array", + "minItems": 3, + "maxItems": 3, + "items": { + "type": "number", + "minimum": 0.0, + "maximum": 1.0 + } + } + ], + "default": [], + "x-entity": { + "type": "array [size 3]", + "default": "[] (empty => color by |B|)", + "examples": [ + "[1.0, 1.0, 1.0] # white field lines" + ] + } + }, + "log": { + "description": "Map the tube color range logarithmically", + "type": "boolean", + "default": false, + "x-entity": { + "type": "bool", + "notes": [ + "requires min > 0" + ] + } + }, + "min": { + "description": "Tube color range: lower bound on |field|", + "type": "number", + "default": 0.0, + "x-entity": { + "type": "float", + "notes": [ + "when min >= max, the range is auto-set from |field| along the lines" + ] + } + }, + "max": { + "description": "Tube color range: upper bound on |field|", + "type": "number", + "default": 0.0, + "x-entity": { + "type": "float", + "notes": [ + "when min >= max, the range is auto-set from |field| along the lines" + ] + } + }, + "step_frac": { + "description": "(3D tubes) RK4 integration step as a fraction of one coarse cell", + "type": "number", + "exclusiveMinimum": 0.0, + "default": 0.5, + "x-entity": { + "type": "float [> 0]" + } + }, + "max_steps": { + "description": "(3D tubes) Per-direction integration-step cap", + "type": "integer", + "minimum": 1, + "default": 4000, + "x-entity": { + "type": "int [> 0]" + } + }, + "max_length": { + "description": "(3D tubes) Maximum line length, in global box diagonals (per direction)", + "type": "number", + "exclusiveMinimum": 0.0, + "default": 3.0, + "x-entity": { + "type": "float [> 0]" + } + } + } + } + } + } + } + }, + "checkpoint": { + "description": "Checkpointing parameters", + "type": "object", + "additionalProperties": false, + "x-entity": { + "inferred": [ + { + "name": "is_resuming", + "brief": "Whether the simulation is resuming from a checkpoint", + "type": "bool", + "from": "command-line flag" + }, + { + "name": "start_step", + "brief": "Timestep of the checkpoint used to resume", + "type": "uint", + "from": "automatically determined during restart" + }, + { + "name": "start_time", + "brief": "Time of the checkpoint used to resume", + "type": "float", + "from": "automatically determined during restart" + } + ] + }, + "properties": { + "interval": { + "description": "Number of timesteps between checkpoints", + "type": "integer", + "minimum": 1, + "default": 1000, + "x-entity": { + "type": "uint [> 0]" + } + }, + "interval_time": { + "description": "Physical (code) time interval between checkpoints", + "type": "number", + "default": -1.0, + "x-entity": { + "type": "float [> 0]", + "notes": [ + "When `< 0`, the output is controlled by `interval`" + ] + } + }, + "keep": { + "description": "Number of checkpoints to keep", + "type": "integer", + "minimum": -1, + "default": 2, + "x-entity": { + "type": "int", + "notes": [ + "0 = disable checkpointing", + "-1 = keep all checkpoints" + ] + } + }, + "walltime": { + "description": "Write a checkpoint once after a fixed walltime", + "type": "string", + "pattern": "^$|^[0-9]{2,}:[0-9]{2}:[0-9]{2}$", + "default": "00:00:00", + "x-entity": { + "type": "string", + "notes": [ + "The format is \"HH:MM:SS\"", + "Empty string or \"00:00:00\" disables this functionality", + "Writing checkpoint at walltime does not stop the simulation" + ] + } + }, + "write_path": { + "description": "Parent directory to write checkpoints to", + "type": "string", + "x-entity": { + "type": "string", + "default": "`.ckpt`", + "notes": [ + "The directory is created if it does not exist" + ] + } + }, + "read_path": { + "description": "Parent directory to use when resuming from a checkpoint", + "type": "string", + "x-entity": { + "type": "string", + "default": "inherit `write_path`" + } + } + } + }, + "adios2": { + "description": "ADIOS2 BP5 tuning, applied to both [output] and [checkpoint] writers", + "type": "object", + "additionalProperties": false, + "properties": { + "aggregators_per_node": { + "description": "Number of ADIOS2 aggregators per node", + "type": "integer", + "minimum": 0, + "default": 0, + "x-entity": { + "type": "uint", + "notes": [ + "Set to either MPI ranks/node or NICs/node for best performance\nIf set to 0, will use ADIOS2 default (one aggregator per node)" + ] + } + }, + "max_shm_size": { + "description": "Maximum shared-memory segment size per node, in bytes (BP5 MaxShmSize)", + "type": "integer", + "minimum": 0, + "default": 4294967296, + "x-entity": { + "type": "uint", + "notes": [ + "Lower this on memory-constrained nodes; matches ADIOS2's default" + ] + } + }, + "buffer_chunk_size": { + "description": "Internal serialization buffer chunk size, in bytes (BP5 BufferChunkSize)", + "type": "integer", + "minimum": 0, + "default": 16777216, + "x-entity": { + "type": "uint", + "notes": [ + "Scales with per-rank output volume; matches ADIOS2's default" + ] + } + } + } + }, + "diagnostics": { + "description": "Diagnostic logging parameters", + "type": "object", + "additionalProperties": false, + "properties": { + "interval": { + "description": "Number of timesteps between diagnostic logs", + "type": "integer", + "minimum": 1, + "default": 1, + "x-entity": { + "type": "int [> 0]" + } + }, + "blocking_timers": { + "description": "Blocking timers between successive algorithms", + "type": "boolean", + "default": false, + "x-entity": { + "type": "bool" + } + }, + "colored_stdout": { + "description": "Enable colored stdout", + "type": "boolean", + "default": true, + "x-entity": { + "type": "bool" + } + }, + "log_level": { + "description": "Specify the log level", + "anyOf": [ + { + "enum": [ + "VERBOSE", + "WARNING", + "ERROR" + ] + }, + { + "type": "string", + "pattern": "(?i)^(VERBOSE|WARNING|ERROR)$" + } + ], + "default": "VERBOSE", + "x-entity": { + "type": "string", + "notes": [ + "\"VERBOSE\" prints all messages, \"WARNING\" prints only warnings and errors, \"ERROR\" prints only errors" + ] + } + } + } + } + } +} diff --git a/input.example.toml b/input.default.toml similarity index 73% rename from input.example.toml rename to input.default.toml index 38ad8a7d0..c887b271d 100644 --- a/input.example.toml +++ b/input.default.toml @@ -1,3 +1,4 @@ +# Global simulation parameters [simulation] # Name of the simulation # @required @@ -8,69 +9,75 @@ # @required # @type: string # @enum: "SRPIC", "GRPIC" - engine = "" + engine = "SRPIC" # Max runtime in physical (code) units # @required # @type: float [> 0] # @example: 1e5 - runtime = "" + runtime = 1.0 + # Parameters specific to domain decomposition [simulation.domain] # Number of domains # @type: int # @default: 1 [no MPI]; MPI_SIZE [MPI] - number = "" + number = 1 # Decomposition of the domain (for MPI) in each of the directions # @type: array [size 1 :->: 3] # @default: [-1, -1, -1] - # @note: -1 means the code will determine the decomposition in the specific direction automatically - # @note: Automatic detection is either done by inference from # of MPI tasks, or by balancing the grid size on each domain + # @note: -1 means the code will determine the decomposition in the + # specific direction automatically + # @note: Automatic detection is either done by inference from # of MPI + # tasks, or by balancing the grid size on each domain # @example: [2, 2, 2] (total of 8 domains) - decomposition = "" + decomposition = [-1, -1, -1] - # Diffusion-style dynamic load balancing (Cartesian metrics only). - # Domain boundaries between MPI neighbors are nudged to equalize the + # Diffusion-style dynamic load balancing (Cartesian metrics only). Domain + # boundaries between MPI neighbors are nudged to equalize the # active-particle count per rank. All inter-rank traffic uses only the # existing nearest-neighbor field/particle communication paths. [simulation.domain.load_balance] # Enable dynamic load balancing # @type: bool # @default: false - enable = "" + enable = false # Run the rebalancer every `interval` timesteps (0 disables) # @type: int # @default: 0 - interval = "" + interval = 0 # Dimensions along which load is redistributed (1 = x1, 2 = x2, 3 = x3) # @type: array of int, subset of [1, 2, 3] # @default: [1] - dimensions = "" + # @enum: 1, 2, 3 + dimensions = [1] # Skip rebalancing along a dim when (max - min) / mean of the per-slice # particle count is below this fraction # @type: float # @default: 0.1 - tolerance = "" + tolerance = 0.1 # Maximum cell-shift per interior boundary per event; clamped at compile # time to N_GHOSTS so the migrating field strip is already cached in the # rank's ghost zone. # @type: int # @default: N_GHOSTS - max_shift = "" + max_shift = 0 +# Parameters specific to grid geometry [grid] # Spatial resolution of the grid # @required # @type: array [size 1 :->: 3] # @note: Dimensionality is inferred from the size of this array # @example: [1024, 1024, 1024] - resolution = "" + resolution = [1] # Physical extent of the grid # @required # @type: array> [size 1 :->: 3] - # @note: For spherical geometry, only specify `[[rmin, rmax]]`, other values are set automatically + # @note: For spherical geometry, only specify `[[rmin, rmax]]`, other values + # are set automatically # @note: For cartesian geometry, cell aspect ratio has to be 1: `dx=dy=dz` # @example: [[0.0, 1.0], [-1.0, 1.0]] - extent = "" + extent = [[0.0, 0.0]] # @inferred: # - dim @@ -79,25 +86,28 @@ # @enum: 1, 2, 3 # @from: `grid.resolution` + # Metric-related parameters [grid.metric] # Metric on the grid # @required # @type: string - # @enum: "Minkowski", "Spherical", "QSpherical", "Kerr_Schild", "QKerr_Schild", "Kerr_Schild_0" - metric = "" + # @enum: "Minkowski", "Spherical", "QSpherical", "Kerr_Schild", + # "QKerr_Schild", "Kerr_Schild_0" + metric = "Minkowski" # `r0` paramter for the QSpherical metric `x1 = log(r-r0)` # @type: float [-inf -> rmin] # @default: 0.0 # @note: Negative values produce almost uniform grid in r - qsph_r0 = "" - # `h` paramter for the QSpherical metric `th = x2 + 2*h x2 (pi-2*x2)*(pi-x2)/pi^2` + qsph_r0 = 0.0 + # `h` paramter for the QSpherical metric `th = x2 + 2*h x2 + # (pi-2*x2)*(pi-x2)/pi^2` # @type: float [-1 :->: 1] # @default: 0.0 - qsph_h = "" + qsph_h = 0.0 # Spin parameter for the Kerr Schild metric # @type: float [0 :-> 1] # @default: 0.0 - ks_a = "" + ks_a = 0.0 # @inferred: # - coord @@ -110,84 +120,104 @@ # @type: float # @from: `grid.metric.ks_a` # - params - # @brief: A map of all metric-specific parameters together (for easy access) + # @brief: A map of all metric-specific parameters together (for easy + # access) # @type: map # @from: `grid.metric` + # Boundary-condition related parameters [grid.boundaries] # Boundary conditions for fields # @required # @type: array> [size 1 :->: 3] - # @enum: "PERIODIC", "MATCH", "FIXED", "ATMOSPHERE", "CUSTOM", "HORIZON", "CONDUCTOR" - # @note: When periodic in any of the directions, you should only set one value: [..., ["PERIODIC"], ...] - # @note: In spherical, bondaries in theta/phi are set automatically (only specify bc @ `[rmin, rmax]`): [["ATMOSPHERE", "MATCH"]] - # @note: In GR, the horizon boundary is set automatically (only specify bc @ rmax): [["MATCH"]] + # @enum: "PERIODIC", "MATCH", "FIXED", "ATMOSPHERE", "CUSTOM", "HORIZON", + # "CONDUCTOR" + # @note: When periodic in any of the directions, you should only set one + # value: [..., ["PERIODIC"], ...] + # @note: In spherical, bondaries in theta/phi are set automatically (only + # specify bc @ `[rmin, rmax]`): [["ATMOSPHERE", "MATCH"]] + # @note: In GR, the horizon boundary is set automatically (only specify bc + # @ rmax): [["MATCH"]] # @example: [["CUSTOM", "MATCH"]] (for 2D spherical `[[rmin, rmax]]`) - fields = "" - # Boundary conditions for fields + fields = [["PERIODIC"]] + # Boundary conditions for particles # @required # @type: array> [size 1 :->: 3] - # @enum: "PERIODIC", "ABSORB", "ATMOSPHERE", "CUSTOM", "REFLECT", "HORIZON" - # @note: When periodic in any of the directions, you should only set one value [..., ["PERIODIC"], ...] - # @note: In spherical, bondaries in theta/phi are set automatically (only specify bc @ `[rmin, rmax]`) [["ATMOSPHERE", "ABSORB"]] - # @note: In GR, the horizon boundary is set automatically (only specify bc @ `rmax`): [["ABSORB"]] + # @enum: "PERIODIC", "ABSORB", "ATMOSPHERE", "CUSTOM", "REFLECT", + # "HORIZON" + # @note: When periodic in any of the directions, you should only set one + # value [..., ["PERIODIC"], ...] + # @note: In spherical, bondaries in theta/phi are set automatically (only + # specify bc @ `[rmin, rmax]`) [["ATMOSPHERE", "ABSORB"]] + # @note: In GR, the horizon boundary is set automatically (only specify bc + # @ `rmax`): [["ABSORB"]] # @example: [["PERIODIC"], ["PERIODIC"]] - particles = "" + particles = [["PERIODIC"]] + # Parameters specific to MATCH boundary conditions [grid.boundaries.match] - # Size of the matching layer in each direction for fields in physical (code) units + # Size of the matching layer in each direction for fields in physical + # (code) units # @type: float | array> # @default: 1% of the domain size (in shortest dimension) - # @note: In spherical, this is the size of the layer in `r` from the outer wall + # @note: In spherical, this is the size of the layer in `r` from the + # outer wall # @example: `ds = 1.5` (will set the same for all directions) - # @example: `ds = [[1.5], [2.0, 1.0], [1.1]]` (will duplicate 1.5 for +/- `x1` and 1.1 for +/- `x3`) + # @example: `ds = [[1.5], [2.0, 1.0], [1.1]]` (will duplicate 1.5 for + # +/- `x1` and 1.1 for +/- `x3`) # @example: `ds = [[], [1.5], []]` (will only set for x2) - ds = "" + ds = 1.0 + # Parameters specific to ABSORB boundary conditions [grid.boundaries.absorb] # Size of the absorption layer for particles in physical (code) units # @type: float # @default: 1% of the domain size (in shortest dimension) - # @note: In spherical, this is the size of the layer in `r` from the outer wall - # @note: In cartesian, this is the same for all dimensions where applicable - ds = "" + # @note: In spherical, this is the size of the layer in `r` from the + # outer wall + # @note: In cartesian, this is the same for all dimensions where + # applicable + ds = 1.0 + # Parameters specific to ATMOSPHERE boundary conditions [grid.boundaries.atmosphere] # Temperature of the atmosphere in units of `m0 c^2` # @type: float # @note: [required] if `ATMOSPHERE` is one of the boundaries - temperature = "" + temperature = 0.0 # Peak number density of the atmosphere at base in units of `n0` # @type: float - density = "" + density = 0.0 # Pressure scale-height in physical units # @type: float - height = "" + height = 1.0 # Species indices of particles that populate the atmosphere # @type: array [size 2] - species = "" + species = [1, 1] # Distance from the edge to which the gravity is imposed in physical units # @type: float # @default: 0.0 # @note: 0.0 means no limit - ds = "" + ds = 0.0 # @inferred: # - g # @brief: Acceleration due to imposed gravity # @type: float - # @from: `grid.boundaries.atmosphere.temperature`, `grid.boundaries.atmosphere.height` + # @from: `grid.boundaries.atmosphere.temperature`, + # `grid.boundaries.atmosphere.height` # @value: `temperature / height` +# Fiducial scales that fix the code unit system [scales] # Fiducial larmor radius # @required # @type: float [> 0.0] - larmor0 = "" + larmor0 = 1.0 # Fiducial plasma skin depth # @required # @type: float [> 0.0] - skindepth0 = "" + skindepth0 = 1.0 # @inferred: # - dx0 @@ -224,99 +254,124 @@ # @from: `scales.larmor0` # @value: `1 / larmor0` +# Radiative drag and photon emission parameters [radiation] + + # Radiation reaction (drag) parameters [radiation.drag] + + # Synchrotron drag parameters [radiation.drag.synchrotron] # Radiation reaction limit gamma-factor for synchrotron # @type: float [> 0.0] # @default: 1.0 - # @note: [required] if one of the species has `radiative_drag = "synchrotron"` - gamma_rad = "" + # @note: [required] if one of the species has `radiative_drag = + # "synchrotron"` + gamma_rad = 1.0 + # Compton drag parameters [radiation.drag.compton] # Radiation reaction limit gamma-factor for Compton drag # @type: float [> 0.0] # @default: 1.0 - # @note: [required] if one of the species has `radiative_drag = "compton"` - gamma_rad = "" + # @note: [required] if one of the species has `radiative_drag = + # "compton"` + gamma_rad = 1.0 + # Photon emission parameters [radiation.emission] + + # Synchrotron emission parameters [radiation.emission.synchrotron] - # Gamma-factor of a particle emitting synchrotron photons at energy `m0 c^2` in fiducial magnetic field `B0` + # Gamma-factor of a particle emitting synchrotron photons at energy `m0 + # c^2` in fiducial magnetic field `B0` # @type: float [> 1.0] # @default: 10.0 - gamma_qed = "" + gamma_qed = 10.0 # Minimum photon energy for synchrotron emission (units of `m0 c^2`) # @type: float [> 0.0] # @default: 1e-4 - photon_energy_min = "" + photon_energy_min = 1e-4 # Weights for the emitted synchrotron photons # @type: float [> 0.0] # @default: 1.0 - photon_weight = "" + photon_weight = 1.0 # Index of species for the emitted photon - # @type: ushort [> 0] # @required - photon_species = "" + # @type: ushort [> 0] + photon_species = 1 # @inferred: # - nominal_probability - # @brief: Nominal probability of the emission for a particle with `gamma * beta = 1`, charge-to-mass = `q0 / m0` + # @brief: Nominal probability of the emission for a particle with + # `gamma * beta = 1`, charge-to-mass = `q0 / m0` # @type: float - # @from: `.gamma_qed`, `.photon_weight`, `...drag.synchrotron.gamma_rad`, `scales.omegaB0`, `algorithms.timestep.dt` - # @value: `0.1 * omegaB0 * dt * (gamma_qed / gamma_rad)^2 / photon_weight` + # @from: `.gamma_qed`, `.photon_weight`, + # `...drag.synchrotron.gamma_rad`, `scales.omegaB0`, + # `algorithms.timestep.dt` + # @value: `0.1 * omegaB0 * dt * (gamma_qed / gamma_rad)^2 / + # photon_weight` # - nominal_photon_energy - # @brief: Nominal energy of the emitted photon for a particle with `gamma * beta = 1`, mass = `m0` + # @brief: Nominal energy of the emitted photon for a particle with + # `gamma * beta = 1`, mass = `m0` # @type: float # @from: `.gamma_qed` # @value: `(1 / gamma_qed)^2` + # Inverse Compton emission parameters [radiation.emission.compton] - # Gamma-factor of a particle emitting inverse Compton photons at energy `m0 c^2` in fiducial magnetic field `B0` + # Gamma-factor of a particle emitting inverse Compton photons at energy + # `m0 c^2` in fiducial magnetic field `B0` # @type: float [> 1.0] # @default: 10.0 - gamma_qed = "" + gamma_qed = 10.0 # Minimum photon energy for inverse Compton emission (units of `m0 c^2`) # @type: float [> 0.0] # @default: 1e-4 - photon_energy_min = "" + photon_energy_min = 1e-4 # Weights for the emitted inverse Compton photons # @type: float [> 0.0] # @default: 1.0 - photon_weight = "" + photon_weight = 1.0 # Index of species for the emitted photon - # @type: ushort [> 0] # @required - photon_species = "" + # @type: ushort [> 0] + photon_species = 1 # @inferred: # - nominal_probability - # @brief: Nominal probability of the emission for a particle with `gamma * beta = 1`, charge-to-mass = `q0 / m0` + # @brief: Nominal probability of the emission for a particle with + # `gamma * beta = 1`, charge-to-mass = `q0 / m0` # @type: float - # @from: `.gamma_qed`, `.photon_weight`, `...drag.compton.gamma_rad`, `scales.omegaB0`, `algorithms.timestep.dt` - # @value: `0.1 * omegaB0 * dt * (gamma_qed / gamma_rad)^2 / photon_weight` + # @from: `.gamma_qed`, `.photon_weight`, `...drag.compton.gamma_rad`, + # `scales.omegaB0`, `algorithms.timestep.dt` + # @value: `0.1 * omegaB0 * dt * (gamma_qed / gamma_rad)^2 / + # photon_weight` # - nominal_photon_energy - # @brief: Nominal energy of the emitted photon for a particle with `gamma * beta = 1`, mass = `m0` + # @brief: Nominal energy of the emitted photon for a particle with + # `gamma * beta = 1`, mass = `m0` # @type: float # @from: `.gamma_qed` # @value: `(1 / gamma_qed)^2` +# Algorithm and solver tuning [algorithms] # Number of current smoothing passes # @type: ushort [>= 0] # @default: 0 - current_filters = "" + current_filters = 0 + # Timestep parameters [algorithms.timestep] # Courant-Friedrichs-Lewy number # @type: float [0.0 -> 1.0] # @default: 0.95 # @note: CFL number determines the timestep duration - CFL = "" + CFL = 0.95 # Correction factor for the speed of light used in field solver # @type: float # @default: 1.0 - correction = "" + correction = 1.0 # @inferred: # - dt @@ -325,46 +380,50 @@ # @from: `algorithms.timestep.CFL`, `scales.dx0` # @value: `CFL * dx0` + # Current deposition parameters [algorithms.deposit] # Enable the current deposition # @type: bool # @default: true - enable = "" + enable = true # team_policy tiled-deposit work-group (team) size # @type: uint [>= 0] # @default: 0 # @note: 0 keeps Kokkos::AUTO (backend occupancy heuristic); a positive # value overrides it, clamped to the backend/scratch maximum at # launch. Only used in `team_policy=ON` builds. Pick a multiple of - # the device subgroup width for best occupancy (see ideal_tile_size.py) - team_policy_team_size = "" + # the device subgroup width for best occupancy (see + # ideal_tile_size.py) + team_policy_team_size = 0 # @inferred: # - order # @brief: order of the particle shape function - # @from: compile-time definition `shape_order` # @type: ushort [0 -> 10] + # @from: compile-time definition `shape_order` + # GR pusher parameters [algorithms.gr] # Stepsize for numerical differentiation in GR pusher # @type: float [> 0.0] # @default: 1e-6 - pusher_eps = "" + pusher_eps = 1e-6 # Number of iterations for the Newton-Raphson method in GR pusher # @type: ushort [> 0] # @default: 10 - pusher_niter = "" + pusher_niter = 10 + # Guiding-center approximation parameters [algorithms.gca] # Maximum value for E/B allowed for GCA particles # @type: float [0.0 -> 1.0] # @default: 0.9 - e_ovr_b_max = "" + e_ovr_b_max = 0.9 # Maximum Larmor radius allowed for GCA particles (in physical units) # @type: float # @default: 0.0 # @note: When `larmor_max` == 0, the limit is disabled - larmor_max = "" + larmor_max = 0.0 # Stencil coefficients for the field solver [notation as in Blinne+ (2018)] # @note: Standard Yee solver: `delta_i = beta_ij = 0.0` @@ -372,71 +431,73 @@ # Enable the fieldsolver # @type: bool # @default: true - enable = "" + enable = true # delta_x coefficient (for `F_{i +/- 3/2, j, k}`) # @type: float # @default: 0.0 - delta_x = "" + delta_x = 0.0 # delta_y coefficient (for `F_{i, j +/- 3/2, k}`) # @type: float # @default: 0.0 # @note: Used only for 2D and 3D - delta_y = "" + delta_y = 0.0 # delta_z coefficient (for `F_{i, j, k +/- 3/2}`) # @type: float # @default: 0.0 # @note: Used only for 3D - delta_z = "" + delta_z = 0.0 # beta_xy coefficient (for `F_{i +/- 1/2, j +/- 1, k}`) # @type: float # @default: 0.0 # @note: Used only for 2D and 3D - beta_xy = "" + beta_xy = 0.0 # beta_yx coefficient (for `F_{i +/- 1, j +/- 1/2, k}`) # @type: float # @default: 0.0 # @note: Used only for 2D and 3D - beta_yx = "" + beta_yx = 0.0 # beta_xz coefficient (for `F_{i +/- 1/2, j, k +/- 1}`) # @type: float # @default: 0.0 # @note: Used only for 3D - beta_xz = "" + beta_xz = 0.0 # beta_zx coefficient (for `F_{i +/- 1, j, k +/- 1/2}`) # @type: float # @default: 0.0 # @note: Used only for 3D - beta_zx = "" + beta_zx = 0.0 # beta_yz coefficient (for `F_{i, j +/- 1/2, k +/- 1}`) # @type: float # @default: 0.0 # @note: Used only for 3D - beta_yz = "" + beta_yz = 0.0 # beta_zy coefficient (for `F_{i, j +/- 1, k +/- 1/2}`) # @type: float # @default: 0.0 # @note: Used only for 3D - beta_zy = "" + beta_zy = 0.0 +# Particle and species parameters [particles] # Fiducial number of particles per cell # @required # @type: float [> 0.0] - ppc0 = "" + ppc0 = 1.0 # Toggle for using particle weights # @type: bool # @default: false - use_weights = "" + use_weights = false # Timesteps between particle re-sorting by tags (removing dead particles) # @type: uint # @default: 100 # @note: Set to 0 to disable re-sorting - clear_interval = "" - # Timesteps between spatial sorting of particles (for better cache performance) + clear_interval = 100 + # Timesteps between spatial sorting of particles (for better cache + # performance) # @type: uint # @default: 0 # @note: Set to 0 to disable spatial sorting - spatial_sorting_interval = "" + spatial_sorting_interval = 0 # @inferred: # - nspec @@ -444,366 +505,405 @@ # @type: uint # @from: `particles.species` + # Particle species definitions [[particles.species]] # Label of the species # @type: string # @default: "s" - # @example: "e-" # @note: `` is the index of the species in the list starting from 1 - label = "" + # @example: "e-" + label = "s" # Mass of the species (in units of fiducial mass) # @required # @type: float [>= 0.0] - mass = "" + mass = 0.0 # Charge of the species (in units of fiducial charge) # @required # @type: float - charge = "" + charge = 0.0 # Maximum number of particles per task # @required # @type: uint [> 0] - maxnpart = "" + # @note: Read as a float, so exponential notation is fine (e.g. `1e8`) + maxnpart = 1.0 # Pusher algorithm for the species # @type: string # @default: "Boris" [massive]; "Photon" [massless] # @enum: "Boris", "Vay", "Boris,GCA", "Vay,GCA", "Photon", "None" - pusher = "" - # Number of additional real-valued variables (payloads) for each particle of the given species + pusher = "Boris" + # Number of additional real-valued variables (payloads) for each particle of + # the given species # @type: ushort # @default: 0 - n_payloads_real = "" - # Number of additional integer-valued variables (payloads) for each particle of the given species + n_payloads_real = 0 + # Number of additional integer-valued variables (payloads) for each particle + # of the given species # @type: ushort # @default: 0 - # @note: If tracking is enabled, one or two extra integer payloads are reserved (depending on whether MPI is enabled) - n_payloads_int = "" + # @note: If tracking is enabled, one or two extra integer payloads are + # reserved (depending on whether MPI is enabled) + n_payloads_int = 0 # Enable tracking of particles using indices for the given species # @type: bool # @default: false - tracking = "" + tracking = false # Radiation reaction to use for the species # @type: string # @default: "None" # @enum: "None", "Synchrotron", "Compton" - # @note: Can also be coma-separated combination, e.g., "Synchrotron,Compton" + # @note: Can also be coma-separated combination, e.g., + # "Synchrotron,Compton" # @note: Relevant radiation.drag parameters should also be provided - radiative_drag = "" + radiative_drag = "None" # Particle emission policy for the species # @type: string # @default: "None" # @enum: "None", "Synchrotron", "Compton" # @note: Only one emission mechanism allowed - # @note: Appropriate radiation drag flag will be applied automatically (unless explicitly set to "None") - emission = "" + # @note: Appropriate radiation drag flag will be applied automatically + # (unless explicitly set to "None") + emission = "None" # Timesteps between spatial sorting of particles for given species # @type: uint # @default: 0 # @note: Set to 0 to disable spatial sorting - # @note: Overrides `particles.spatial_sorting_interval` for the given species - spatial_sorting_interval = "" + # @note: Overrides `particles.spatial_sorting_interval` for the given + # species + spatial_sorting_interval = 0 # Timesteps between particle re-sorting by tags (removing dead particles) # @type: uint # @default: 100 # @note: Set to 0 to disable re-sorting # @note: Overrides `particles.clear_interval` for the given species - clear_interval = "" + clear_interval = 100 # Parameters for specific problem generators and setups +# @note: Free-form: keys are defined by the problem generator, so nothing here +# is validated [setup] +# Output parameters [output] # Output format # @type: string # @default: "hdf5" # @enum: "disabled", "hdf5", "BPFile" - format = "" + format = "hdf5" # Number of timesteps between all outputs # @type: uint [> 0] # @default: 1 # @note: Value is overriden by output intervals for specific outputs - interval = "" + interval = 1 # Physical (code) time interval between all outputs # @type: float # @default: -1.0 - # @note: When `interval_time` < 0, the output is controlled by `interval`, otherwise by `interval_time` + # @note: When `interval_time` < 0, the output is controlled by `interval`, + # otherwise by `interval_time` # @note: Value is overriden by output intervals for specific outputs - interval_time = "" + interval_time = -1.0 # Whether to output each timestep into separate files # @type: bool # @default: true # @deprecated: starting v1.3.0 - separate_files = "" + separate_files = true + # Field output parameters [output.fields] # Toggle for the field output # @type: bool # @default: true - enable = "" + enable = true # Field quantities to output # @type: array # @default: [] - # @enum: "E", "B", "J", "divE", "Rho", "Charge", "N", "Nppc", "T0i", "Tij", "Vi", "D", "H", "divD", "A" - # @note: For `T`, you can use unspecified indices: `Tij`, `T0i`, or specific ones: `Ttt`, `T00`, `T02`, `T23` - # @note: For `T`, in cartesian can also use "x" "y" "z" instead of "1" "2" "3" - # @note: By default, we accumulate moments from all massive species, one can specify only specific species: `Ttt_1_2`, `Rho_1`, `Rho_3_4` - quantities = "" + # @enum: "E", "B", "J", "divE", "Rho", "Charge", "N", "Nppc", "T0i", + # "Tij", "Vi", "D", "H", "divD", "A" + # @note: For `T`, you can use unspecified indices: `Tij`, `T0i`, or + # specific ones: `Ttt`, `T00`, `T02`, `T23` + # @note: For `T`, in cartesian can also use "x" "y" "z" instead of "1" "2" + # "3" + # @note: By default, we accumulate moments from all massive species, one + # can specify only specific species: `Ttt_1_2`, `Rho_1`, `Rho_3_4` + quantities = [] # Custom (user-defined) field quantities # @type: array # @default: [] - custom = "" + custom = [] # Number of timesteps between field outputs # @type: uint # @default: 0 # @note: When `!= 0`, overrides `output.interval` # @note: When `== 0`, `output.interval` is used - interval = "" + interval = 0 # Physical (code) time interval between field outputs # @type: float # @default: -1.0 # @note: When `< 0`, the output is controlled by `interval` # @note: When specified, overrides `output.interval_time` - interval_time = "" + interval_time = -1.0 # Downsample factor for the output of fields # @type: uint | array [>= 1] # @default: [1, 1, 1] # @note: The output is downsampled by the given factors in each direction # @note: If a scalar is given, it is applied to all directions - downsampling = "" + downsampling = [1, 1, 1] + # Smoothing of the output moments [output.fields.smoothing] # Smoothing order for the output of moments ("Rho", "Charge", "T", ...) # @type: ushort # @default: 0 - order = "" - # Smoothing algorithm + order = 0 + # Smoothing algorithm # @type: string - # @enum: "const", "spline" # @default: "spline" - # @note: When using "spline", `order` corresponds to the order of the polynomial used - # @note: When using "const", the smoothing window is `ceil(order / 2)` in both directions - method = "" + # @enum: "const", "spline" + # @note: When using "spline", `order` corresponds to the order of the + # polynomial used + # @note: When using "const", the smoothing window is `ceil(order / 2)` + # in both directions + method = "spline" + # Particle output parameters [output.particles] # Toggle for the particles output # @type: bool # @default: true - enable = "" + enable = true # Particle species indices to output # @type: array # @default: [] # @note: If empty, all species are output - species = "" + species = [] # Stride for the output of particles # @type: uint [> 1] # @default: 100 - stride = "" + stride = 100 # Number of timesteps between particle outputs # @type: uint # @default: 0 # @note: When `!= 0`, overrides `output.interval` # @note: When `== 0`, `output.interval` is used - interval = "" + interval = 0 # Physical (code) time interval between particle outputs # @type: float # @default: -1.0 # @note: When `< 0`, the output is controlled by `interval` # @note: When specified, overrides `output.interval_time` - interval_time = "" + interval_time = -1.0 + # Spectra output parameters [output.spectra] # Toggle for the spectra output # @type: bool # @default: true - enable = "" + enable = true # Minimum energy for the spectra output # @type: float # @default: 1e-3 - e_min = "" + e_min = 1e-3 # Maximum energy for the spectra output # @type: float # @default: 1e3 - e_max = "" + e_max = 1e3 # Whether to use logarithmic bins for energy # @type: bool # @default: true - log_bins = "" + log_bins = true # Number of energy bins for the spectra output # @type: uint [> 0] # @default: 200 - num_energy_bins = "" + # @deprecated: starting v1.5.0 + num_bins = 200 + # Number of energy bins for the spectra output + # @type: uint [> 0] + # @default: 200 + num_energy_bins = 200 # Number of spatial bins for the spectra output - # @type: array [size 1 :->: 3] - # @default: [1, 1, 1] - num_spatial_bins = "" + # @type: array [size 1 :->: 3] + # @default: [1, 1, 1] + num_spatial_bins = [1, 1, 1] # Number of timesteps between spectra outputs # @type: uint # @default: 0 # @note: When `!= 0`, overrides `output.interval` # @note: When `== 0`, `output.interval` is used - interval = "" + interval = 0 # Physical (code) time interval between spectra outputs # @type: float # @default: -1.0 # @note: When `< 0`, the output is controlled by `interval` # @note: When specified, overrides `output.interval_time` - interval_time = "" + interval_time = -1.0 + # Debug output parameters [output.debug] # Output fields "as is" without conversions # @type: bool # @default: false - as_is = "" + as_is = false # Output fields with values in ghost cells # @type: bool # @default: false - ghosts = "" + ghosts = false + # Integrated statistics output parameters [output.stats] # Toggle for the stats output # @type: bool # @default: true - enable = "" + enable = true # Number of timesteps between stat outputs # @type: uint [> 0] # @default: 100 # @note: Overriden if `output.stats.interval_time != -1` - interval = "" + interval = 100 # Physical (code) time interval between stat outputs # @type: float # @default: -1.0 # @note: When `< 0`, the output is controlled by `interval` - interval_time = "" + interval_time = -1.0 # Field quantities to output # @type: array # @default: ["B^2", "E^2", "ExB", "Rho", "T00"] - # @enum: "B^2", "E^2", "ExB", "N", "Npart", "Charge", "Rho", "T00", "T0i", "Tij" + # @enum: "B^2", "E^2", "ExB", "N", "Npart", "Charge", "Rho", "T00", "T0i", + # "Tij" # @note: For particle moments, ... # @note: ... same notation is used as for `output.fields.quantities` - quantities = "" + quantities = ["B^2", "E^2", "ExB", "Rho", "T00"] # Custom (user-defined) stats # @type: array # @default: [] - custom = "" + custom = [] # In-situ renderer. Renders scalar fields on the GPU and writes PNG images # directly to `/renders/` each cadence -- no field data is written to # storage, and the result is seamless across MPI domain boundaries. # @note: two modes, selected automatically by the simulation dimension: # - 3D Cartesian (Minkowski): volume ray-march (uses `samples`, - # `step_size`, `early_term_alpha`, and the [camera] table) + # `step_size`, `early_term_alpha`, and the [camera] table) # - 2D (Minkowski, Spherical/QSpherical, and all GR Kerr-Schild): - # flat slice rasterizer. Cartesian shows the (x, y) plane; spherical - # shows the meridional (r, theta) half-plane mapped to Cartesian - # (X = r sin th, Z = r cos th), optionally mirrored (see `mirror`). - # The `samples`/`step_size`/`early_term_alpha`/[camera] keys are - # ignored in 2D (one opaque sample per pixel). + # flat slice rasterizer. Cartesian shows the (x, y) plane; spherical + # shows the meridional (r, theta) half-plane mapped to Cartesian (X = + # r sin th, Z = r cos th), optionally mirrored (see `mirror`). The + # `samples`/`step_size`/`early_term_alpha`/[camera] keys are ignored + # in 2D (one opaque sample per pixel). # @note: 1D (and 3D non-Cartesian, which does not exist) is a no-op # @note: One PNG stream per scene (e.g. a density/|B|/|J| triptych) [output.render] # Toggle for the volume renderer # @type: bool # @default: false - enable = "" + enable = false # Number of timesteps between renders # @type: uint # @default: 0 # @note: When `!= 0`, overrides `output.interval` # @note: When `== 0`, `interval_time` (or `output.interval`) is used - interval = "" + interval = 0 # Physical (code) time interval between renders # @type: float # @default: -1.0 # @note: When `< 0`, the output is controlled by `interval` - interval_time = "" + interval_time = -1.0 # Image width in pixels (the rendered region; the PNG is wider if a colorbar # margin is added, see `colorbar_outside`) # @type: int [> 0] # @default: 1024 - width = "" + width = 1024 # Image height in pixels # @type: int [> 0] # @default: 1024 - height = "" - # Convenience: force a square frame (sets width == height == resolution), the - # natural shape for a dome master. Overrides `width`/`height` when > 0. + height = 1024 + # Convenience: force a square frame (sets width == height == resolution), + # the natural shape for a dome master. Overrides `width`/`height` when > 0. # @type: int [> 0] # @default: 0 (use width/height) - resolution = "" - # Axis-aligned render region [lo, hi] in physical/world coords, per axis - # (x1/x2/x3). Any axis left unset spans the full domain. Clamped to the box. + resolution = 0 + # Axis-aligned render region [lo, hi] along x1, in physical/world coords. + # Left unset it spans the full domain. Clamped to the box. # @type: array [size 2] # @default: [] (full extent) # @note: 3D -> the volume is depth-clipped to this box, the wireframe/axes # frame it, and the default camera zooms to it; 2D -> the slice - # window is framed to it. For a spherical 2D slice, x1_lim crops the - # radius r and x2_lim crops the polar angle theta. + # window is framed to it. For a spherical 2D slice, x1_lim crops + # the radius r and x2_lim crops the polar angle theta. # @example: x1_lim = [-64.0, 64.0] - x1_lim = "" - x2_lim = "" - x3_lim = "" + x1_lim = [] + # Axis-aligned render region [lo, hi] along x2, in physical/world coords. + # Left unset it spans the full domain. Clamped to the box. + # @type: array [size 2] + # @default: [] (full extent) + # @note: For a spherical 2D slice, x2_lim crops the polar angle theta + x2_lim = [] + # Axis-aligned render region [lo, hi] along x3, in physical/world coords. + # Left unset it spans the full domain. Clamped to the box. + # @type: array [size 2] + # @default: [] (full extent) + x3_lim = [] # Moving view: translate the render region (and, in 3D, the camera) at this - # velocity in world-units-per-sim-time, to keep a propagating feature (e.g. a - # shock) in frame. Pair with x{1,2,3}_lim (the moving window). 2D uses the + # velocity in world-units-per-sim-time, to keep a propagating feature (e.g. + # a shock) in frame. Pair with x{1,2,3}_lim (the moving window). 2D uses the # first two components; the motion starts at `camera_start_time`. # @type: array [size 2 or 3] # @default: [] (static view) # @example: camera_velocity = [0.9, 0.0] # pan along +x1 at 0.9 c - camera_velocity = "" + camera_velocity = [] # Sim time at which the view starts moving (static before it, e.g. to let an # initial ramp-up finish) # @type: float # @default: 0.0 - camera_start_time = "" + camera_start_time = 0.0 # Number of ray-march steps across the global box diagonal # @type: int [> 0] # @default: 400 - # @note: The world-space step is `box_diagonal / samples` unless `step_size` - # is set. Higher = better quality, slower. - samples = "" + # @note: The world-space step is `box_diagonal / samples` unless + # `step_size` is set. Higher = better quality, slower. + samples = 400 # Fixed world-space step between ray samples # @type: float [>= 0.0] # @default: 0.0 # @note: 0 derives the step from `samples`. The step is identical on all # ranks, which is what makes the multi-domain composite seamless. - step_size = "" + step_size = 0.0 # Stop marching a ray once its accumulated opacity reaches this value # @type: float [0.0 -> 1.0] # @default: 0.99 # @note: Pure speed optimization; set to 1.0 to disable early termination - early_term_alpha = "" + early_term_alpha = 0.99 # Number of entries in the color/opacity lookup table # @type: int [> 1] # @default: 256 - n_lut = "" - # Opaque background RGB (each channel 0..1) shown through transparent/low- - # opacity pixels; also fills the colorbar margin + n_lut = 256 + # Opaque background RGB (each channel 0..1) shown through + # transparent/low-opacity pixels; also fills the colorbar margin # @type: array [size 3] # @default: [0.0, 0.0, 0.0] - background = "" + background = [0.0, 0.0, 0.0] # Draw a colorbar (gradient + value ticks + label) on each PNG # @type: bool # @default: true - colorbar = "" + colorbar = true # Draw the colorbar in an added right margin (the PNG becomes wider by a # fixed strip) instead of overlaying it on the rendered volume # @type: bool # @default: true - colorbar_outside = "" + colorbar_outside = true # 2D spherical slice only: mirror the meridional half-plane across the # symmetry axis to render a full disk from one axisymmetric half. No effect # on Cartesian or 3D rendering. # @type: bool # @default: true - mirror = "" + mirror = true # Draw the current simulation time as a label ("T = ", fixed to 2 # decimals) in the upper-right corner of the render region, in a contrasting # color, vertically centered between the frame top and the colorbar. # @type: bool # @default: false - time_label = "" - # Draw a spine (frame) + axis ticks + labels around the rendered region. - # The PNG gains left/bottom margins (background-filled) for the tick labels - # and axis names, so they never overlap the data. + time_label = false + # Draw a spine (frame) + axis ticks + labels around the rendered region. The + # PNG gains left/bottom margins (background-filled) for the tick labels and + # axis names, so they never overlap the data. # @type: bool # @default: false # @note: 2D Cartesian = a rectangular frame with linear spatial ticks; @@ -812,23 +912,24 @@ # outline / spine); # 3D = the global box projected to a wireframe with ticks on the # three silhouette edges (x bottom, y & z on the left) - axes = "" - # Axis names. 3D uses all three; the 2D slice uses the first two. When unset, - # the 2D slice defaults to "x","y" (Cartesian) or "X","Z" (spherical). + axes = false + # Axis names. 3D uses all three; the 2D slice uses the first two. When + # unset, the 2D slice defaults to "x","y" (Cartesian) or "X","Z" + # (spherical). # @type: array [size <= 3] # @default: ["x", "y", "z"] - axis_labels = "" + axis_labels = ["x", "y", "z"] # Target number of ticks per axis (actual count is rounded to nice values) # @type: int [>= 2] # @default: 5 - axis_ticks = "" + axis_ticks = 5 # 3D only: target width (pixels) of the box wireframe "spine". The spine is # drawn inside the ray-march (opaque, depth-occluded by the volume); its # width is floored by the ray step, so for a crisper thin line raise # `samples` as well. # @type: float [> 0.0] # @default: 2.0 - spine_width = "" + spine_width = 2.0 # Camera (3D volume mode only; ignored by the 2D slice rasterizer). Defaults # frame the whole global box from outside, looking down the (1,1,1) diagonal @@ -837,67 +938,69 @@ [output.render.camera] # Projection mode. Overrides `orthographic` below when set. # @type: string - # @enum: "orthographic", "perspective", "dome" # @default: (unset -> use `orthographic`) - # @note: "dome" is a fulldome azimuthal-equidistant fisheye rendered from - # an INTERIOR eye (the domain center by default), i.e. a 3D + # @enum: "orthographic", "perspective", "dome" + # @note: "dome" is a fulldome azimuthal-equidistant fisheye rendered + # from an INTERIOR eye (the domain center by default), i.e. a 3D # planetarium dome master. It uses a depth-resolved (A-buffer) - # composite that is seamless across a full 3D domain decomposition - # (unlike ortho/perspective, which need the eye outside the box). - # Set a square frame (`resolution`, or width == height). `forward` - # is the dome ZENITH (screen-up defaults to +y for a +z zenith). - mode = "" + # composite that is seamless across a full 3D domain + # decomposition (unlike ortho/perspective, which need the eye + # outside the box). Set a square frame (`resolution`, or width == + # height). `forward` is the dome ZENITH (screen-up defaults to +y + # for a +z zenith). + mode = "orthographic" # Orthographic (true) or perspective (false) projection # @type: bool # @default: true # @note: Orthographic is recommended; the seamless composite is always # valid for it. Perspective is only seamless with the eye outside # the box. - orthographic = "" + orthographic = true # Camera (eye) position in world (physical) coordinates # @type: array [size 3] # @default: box center pushed back ~1.7 box-diagonals along (1, 1, 1); # for `mode = "dome"`, the domain center (interior eye) - position = "" + position = [0.0, 0.0, 0.0] # Point the camera looks at, in world coordinates (the dome ZENITH target) # @type: array [size 3] # @default: box center; for `mode = "dome"`, the zenith defaults to +z - look_at = "" + look_at = [0.0, 0.0, 0.0] # Camera up vector (dome: the disk's screen-up) # @type: array [size 3] # @default: [0.0, 0.0, 1.0]; for `mode = "dome"`, [0.0, 1.0, 0.0] - up = "" + up = [0.0, 0.0, 1.0] # Vertical field of view in degrees (perspective only) # @type: float [> 0.0] # @default: 35.0 - fov = "" + fov = 35.0 # Full dome field of view in degrees (dome mode only): the image rim is at - # dome_fov/2 from the zenith (180 = a full hemisphere down to the horizon). + # dome_fov/2 from the zenith (180 = a full hemisphere down to the + # horizon). # @type: float [> 0.0, <= 360.0] # @default: 180.0 - dome_fov = "" - # Dome far-clip radius in world units (dome mode only): each ray stops this - # far from the eye, so the sampled region is a half-ball (hemisphere) of - # this radius rather than the whole box -> uniform path length and no box - # corner/edge projection artifacts. `samples` then counts steps across this - # radius. + dome_fov = 180.0 + # Dome far-clip radius in world units (dome mode only): each ray stops + # this far from the eye, so the sampled region is a half-ball (hemisphere) + # of this radius rather than the whole box -> uniform path length and no + # box corner/edge projection artifacts. `samples` then counts steps across + # this radius. # @type: float [>= 0.0] # @default: the largest sphere centered in the box (half the shortest # side), so it touches the face centers and never a corner # @note: 0 disables the clip (rays march to the box boundary) - dome_radius = "" + dome_radius = 0.0 # Vertical extent of the view in world units (orthographic only) # @type: float [> 0.0] # @default: the global box diagonal (the whole box fits from any angle) - ortho_height = "" + ortho_height = 1.0 # Fulldome fisheye ("planetarium dome master"). 2D only; a circular image is # centered in the frame's inscribed circle with the corners left as the # background (the dome master's black border). Set `width == height` (e.g. # 4096) for a square master. When enabled, the axes and the outside colorbar # strip are suppressed so the PNG stays exactly width x height. Seamless - # across MPI domains (the pixel->world map is a shared, deterministic function - # and the tiles stay disjoint). Ignored (with a warning) for 3D. + # across MPI domains (the pixel->world map is a shared, deterministic + # function and the tiles stay disjoint). Ignored (with a warning) for 3D. # @note: CARTESIAN -- the flat plane is warped radially into the disk; use # `fov`/`radius`/`center`/`projection` below. # @note: SPHERICAL / GR Kerr-Schild -- the meridional slice is ALREADY a @@ -910,32 +1013,34 @@ # Build the fisheye dome master instead of the plain slice # @type: bool # @default: false - enable = "" + enable = false # (Cartesian only) Full dome field of view in degrees (image radius maps # linearly to the dome zenith angle: the rim is at fov/2) # @type: float [> 0.0, <= 180.0] # @default: 180.0 # a full hemisphere - fov = "" - # (Cartesian only) World radius of the circular cutout mapped onto the dome + fov = 180.0 + # (Cartesian only) World radius of the circular cutout mapped onto the + # dome # @type: float [> 0.0] # @default: half the shorter domain side (the largest centered disk that # fits inside the box) - radius = "" + radius = 1.0 # (Cartesian only) World-space center of the cutout # @type: array [size 2] # @default: the domain center - center = "" + center = [0.0, 0.0] # (Cartesian only) How the dome zenith angle maps to a world radius on the # flat slice # @type: string + # @default: "equidistant" # @enum: "equidistant" (r proportional to angle; the fulldome image - # standard -- a straight radial scaling of the cutout) + # standard -- a straight radial scaling of the cutout), # "gnomonic" (r ~ tan(angle); the slice as a flat "ceiling" - # tangent to the dome -- straight sim lines stay straight) - # "stereographic" (r ~ tan(angle/2); conformal, preserves shapes) - # "orthographic" (r ~ sin(angle); the slice as seen face-on) - # @default: "equidistant" - projection = "" + # tangent to the dome -- straight sim lines stay straight), + # "stereographic" (r ~ tan(angle/2); conformal, preserves + # shapes), "orthographic" (r ~ sin(angle); the slice as seen + # face-on) + projection = "equidistant" # One scene per scalar field -> one PNG stream. Repeat the table for each. [[output.render.scenes]] @@ -943,32 +1048,36 @@ # given as a magnitude or a single component) # @required # @type: string - # @enum (fields): "{E,B,J}mag"; "{E,B,J}{1,2,3}" or "{E,B,J}{x,y,z}" - # @enum (moments): "N", "Nppc", "Rho", "Charge"; "T{i}{j}"; "V{i}"; "Vmag" - # @note: "{E,B,J}mag" = vector magnitude |.|; "B1"/"Bx", "J3"/"Jz", ... = - # a single (signed) physical component + # @enum: (fields): "{E,B,J}mag"; "{E,B,J}{1,2,3}" or "{E,B,J}{x,y,z}", + # (moments): "N", "Nppc", "Rho", "Charge"; "T{i}{j}"; "V{i}"; + # "Vmag" + # @note: "{E,B,J}mag" = vector magnitude |.|; "B1"/"Bx", "J3"/"Jz", ... + # = a single (signed) physical component # @note: a bare vector ("E"/"B"/"J") is not renderable -- choose a # component or the magnitude # @note: "N"/"Nppc" = number / per-cell count, "Rho" = mass density, # "Charge" = charge density # @note: "T{i}{j}" = one stress-energy component, i,j in {t,x,y,z} or - # {0,1,2,3} (e.g. "Txx", "Ttt", "T0x"); "V{i}" = one bulk-velocity - # component, i in {x,y,z} or {1,2,3} (e.g. "Vx", "V1"); "Vmag" = - # bulk-velocity magnitude sqrt(V1^2+V2^2+V3^2) + # {0,1,2,3} (e.g. "Txx", "Ttt", "T0x"); "V{i}" = one + # bulk-velocity component, i in {x,y,z} or {1,2,3} (e.g. "Vx", + # "V1"); "Vmag" = bulk-velocity magnitude sqrt(V1^2+V2^2+V3^2) # @note: moments follow the engine: SRPIC = tetrad-basis bulk 3-velocity - # and stress-energy; GRPIC = Eckart-frame 4-velocity (so "Vt"/"V0" - # = u^0 = Gamma/alpha is also valid) and contravariant T + # and stress-energy; GRPIC = Eckart-frame 4-velocity (so + # "Vt"/"V0" = u^0 = Gamma/alpha is also valid) and contravariant + # T # @note: per-species selection with a "_" suffix on moments, e.g. - # "N_1", "Rho_2", "Txy_1_2", "V1_3"; default = all massive species + # "N_1", "Rho_2", "Txy_1_2", "V1_3"; default = all massive + # species # @note: components are signed; pair a symmetric `min`/`max` with a # diverging colormap ("cool2warm") to center zero # @note: "fieldlines" renders the magnetic field-line tubes on their own - # (no scalar volume sampled); see [output.render.fieldlines] below + # (no scalar volume sampled); see [output.render.fieldlines] + # below field = "" # PNG filename prefix; files are `.png` # @type: string # @default: "_" - prefix = "" + prefix = "_" # Colorbar title # @type: string # @default: `field` @@ -976,53 +1085,55 @@ # Lower bound of the value range mapped onto the colormap/opacity # @type: float # @default: 0.0 - min = "" + min = 0.0 # Upper bound of the value range # @type: float # @default: 1.0 - max = "" + max = 1.0 # Map the value range logarithmically # @type: bool # @default: false # @note: Requires min > 0 and max > 0 - log = "" + log = false # Colormap name # @type: string - # @enum: "viridis", "inferno", "plasma", "cool2warm", "gray", "RdBu_r", and the - # CMasher maps (BSD-3, https://cmasher.readthedocs.io): "dusk", - # "cosmic", "freeze", "apple", "gothic", "sunburst", "voltage", - # "ocean", "fusion", "prinsenvlag" (an optional "cmr." prefix is ok) # @default: "viridis" - colormap = "" + # @enum: "viridis", "inferno", "plasma", "cool2warm", "gray", "RdBu_r", + # and the CMasher maps (BSD-3, https://cmasher.readthedocs.io): + # "dusk", "cosmic", "freeze", "apple", "gothic", "sunburst", + # "voltage", "ocean", "fusion", "prinsenvlag" (an optional "cmr." + # prefix is ok) + colormap = "viridis" # Opacity transfer function: [position, opacity] control points, both in # [0, 1], piecewise-linear in the normalized value # @type: array> # @default: linear ramp (opacity = normalized value) # @note: Keep the low end near 0 so empty regions stay transparent # @example: [[0.0, 0.0], [0.3, 0.1], [1.0, 0.7]] - alpha = "" + alpha = [] # Explicit value(s) to label on the colorbar # @type: array # @default: 5 evenly-spaced ticks between min and max # @note: Values outside [min, max] are skipped # @example: [0.0, 0.5, 1.0] - colorbar_ticks = "" + colorbar_ticks = [] # Overlay the magnetic field-line tubes inside this scene's volume # @type: bool # @default: false # @note: requires the [output.render.fieldlines] table below (3D only). # A scene with field = "fieldlines" instead renders them alone. - fieldlines = "" + fieldlines = false - # Magnetic field lines, drawn from a coarse, MPI-replicated copy of the field - # so the geometry is global and seamless across domains (the coarsening is - # what makes this cheap -- no parallel particle advection / flux scan). - # - 3D (Cartesian): traced as solid tubes, colored by |field|, composited - # inside the volume ray-march so the volume correctly occludes them. - # - 2D (Cartesian): iso-contours of the flux function psi (Bx = d psi/dy, - # By = -d psi/dx), i.e. the in-plane field lines, colored by |B|. - # - 2D (spherical / Kerr-Schild): traced meridional streamlines of the - # poloidal (Br, Btheta) field (nt2py style). + # Magnetic field lines, drawn from a coarse, MPI-replicated copy of the + # field so the geometry is global and seamless across domains (the + # coarsening is what makes this cheap -- no parallel particle advection / + # flux scan). + # - 3D (Cartesian): traced as solid tubes, colored by |field|, composited + # inside the volume ray-march so the volume correctly occludes them. + # - 2D (Cartesian): iso-contours of the flux function psi (Bx = d psi/dy, By + # = -d psi/dx), i.e. the in-plane field lines, colored by |B|. + # - 2D (spherical / Kerr-Schild): traced meridional streamlines of the + # poloidal (Br, Btheta) field (nt2py style). # Built once per frame and shared by every scene that opts in (per-scene # `fieldlines = true`) and by any standalone `field = "fieldlines"` scene. [output.render.fieldlines] @@ -1031,99 +1142,110 @@ # @default: false # @note: implied true if any scene sets `fieldlines = true` or uses # `field = "fieldlines"` - enable = "" + enable = false # Vector field to trace # @type: string - # @enum: "B", "E", "J" # @default: "B" - field = "" + # @enum: "B", "E", "J" + field = "B" # Field coarsening factor (simulation cells per coarse cell, per axis) # @type: int [1..16] # @default: 4 # @note: larger = smoother "morphology" lines + cheaper replication (the - # coarse field is ~ N_cells / bin^D floats/rank; D = sim dimension) - bin = "" + # coarse field is ~ N_cells / bin^D floats/rank; D = sim + # dimension) + bin = 4 # (3D tubes) Seed-lattice spacing in screen pixels (sets line density) # @type: float [> 0] # @default: 8 # @note: capped by `seed_max`; if seed_px asks for more seeds than that, # the spacing grows to fit and seed_px no longer governs - seed_px = "" + seed_px = 8 # (3D tubes) Hard cap on the seed count (lattice is n^3, 2 lines per seed) # @type: int [> 0] # @default: 4096 # @note: lower this for fewer / more widely spaced lines - seed_max = "" + seed_max = 4096 # (2D contours) Number of evenly-spaced flux-function contour levels # @type: int [> 0] # @default: 16 - # @note: evenly-spaced psi levels => line density tracks |B| automatically - levels = "" + # @note: evenly-spaced psi levels => line density tracks |B| + # automatically + levels = 16 # Tube radius (3D) / contour line width (2D), in screen pixels # @type: float [> 0] # @default: 2 - tube_px = "" + tube_px = 2 # Colormap for the field lines (mapped by |B| along each line) # @type: string - # @enum: "viridis", "inferno", "plasma", "cool2warm", "gray", "RdBu_r", and the - # CMasher maps (BSD-3, https://cmasher.readthedocs.io): "dusk", - # "cosmic", "freeze", "apple", "gothic", "sunburst", "voltage", - # "ocean", "fusion", "prinsenvlag" (an optional "cmr." prefix is ok) # @default: "inferno" - colormap = "" - # Monochrome override: draw the lines in a single [r,g,b] color (each 0..1) - # instead of the |B| colormap -- reads well as an overlay on another volume + # @enum: "viridis", "inferno", "plasma", "cool2warm", "gray", "RdBu_r", + # and the CMasher maps (BSD-3, https://cmasher.readthedocs.io): + # "dusk", "cosmic", "freeze", "apple", "gothic", "sunburst", + # "voltage", "ocean", "fusion", "prinsenvlag" (an optional "cmr." + # prefix is ok) + colormap = "inferno" + # Monochrome override: draw the lines in a single [r,g,b] color (each + # 0..1) instead of the |B| colormap -- reads well as an overlay on another + # volume # @type: array [size 3] # @default: [] (empty => color by |B|) # @example: [1.0, 1.0, 1.0] # white field lines - color = "" + color = [] # Map the tube color range logarithmically # @type: bool # @default: false # @note: requires min > 0 - log = "" - # Tube color range (lower / upper bound on |field|) + log = false + # Tube color range: lower bound on |field| + # @type: float + # @default: 0.0 + # @note: when min >= max, the range is auto-set from |field| along the + # lines + min = 0.0 + # Tube color range: upper bound on |field| # @type: float # @default: 0.0 - # @note: when min >= max, the range is auto-set from |field| along the lines - min = "" - max = "" + # @note: when min >= max, the range is auto-set from |field| along the + # lines + max = 0.0 # (3D tubes) RK4 integration step as a fraction of one coarse cell # @type: float [> 0] # @default: 0.5 - step_frac = "" + step_frac = 0.5 # (3D tubes) Per-direction integration-step cap # @type: int [> 0] # @default: 4000 - max_steps = "" + max_steps = 4000 # (3D tubes) Maximum line length, in global box diagonals (per direction) # @type: float [> 0] # @default: 3.0 - max_length = "" + max_length = 3.0 +# Checkpointing parameters [checkpoint] # Number of timesteps between checkpoints # @type: uint [> 0] # @default: 1000 - interval = "" + interval = 1000 # Physical (code) time interval between checkpoints # @type: float [> 0] # @default: -1.0 # @note: When `< 0`, the output is controlled by `interval` - interval_time = "" + interval_time = -1.0 # Number of checkpoints to keep # @type: int # @default: 2 # @note: 0 = disable checkpointing # @note: -1 = keep all checkpoints - keep = "" + keep = 2 # Write a checkpoint once after a fixed walltime # @type: string # @default: "00:00:00" # @note: The format is "HH:MM:SS" # @note: Empty string or "00:00:00" disables this functionality # @note: Writing checkpoint at walltime does not stop the simulation - walltime = "" + walltime = "00:00:00" # Parent directory to write checkpoints to # @type: string # @default: `.ckpt` @@ -1148,41 +1270,43 @@ # @type: float # @from: automatically determined during restart +# ADIOS2 BP5 tuning, applied to both [output] and [checkpoint] writers [adios2] - # ADIOS2 BP5 tuning, applied to both [output] and [checkpoint] writers # Number of ADIOS2 aggregators per node # @type: uint # @default: 0 # @note: Set to either MPI ranks/node or NICs/node for best performance # If set to 0, will use ADIOS2 default (one aggregator per node) - aggregators_per_node = "" + aggregators_per_node = 0 # Maximum shared-memory segment size per node, in bytes (BP5 MaxShmSize) # @type: uint # @default: 4294967296 # @note: Lower this on memory-constrained nodes; matches ADIOS2's default - max_shm_size = "" + max_shm_size = 4294967296 # Internal serialization buffer chunk size, in bytes (BP5 BufferChunkSize) # @type: uint # @default: 16777216 # @note: Scales with per-rank output volume; matches ADIOS2's default - buffer_chunk_size = "" + buffer_chunk_size = 16777216 +# Diagnostic logging parameters [diagnostics] # Number of timesteps between diagnostic logs # @type: int [> 0] # @default: 1 - interval = "" + interval = 1 # Blocking timers between successive algorithms # @type: bool # @default: false - blocking_timers = "" + blocking_timers = false # Enable colored stdout # @type: bool # @default: true - colored_stdout = "" + colored_stdout = true # Specify the log level # @type: string # @default: "VERBOSE" # @enum: "VERBOSE", "WARNING", "ERROR" - # @note: "VERBOSE" prints all messages, "WARNING" prints only warnings and errors, "ERROR" prints only errors - log_level = "" + # @note: "VERBOSE" prints all messages, "WARNING" prints only warnings and + # errors, "ERROR" prints only errors + log_level = "VERBOSE" diff --git a/dependencies.py b/scripts/dependencies.py similarity index 100% rename from dependencies.py rename to scripts/dependencies.py diff --git a/scripts/generate_template.py b/scripts/generate_template.py new file mode 100755 index 000000000..4695683a8 --- /dev/null +++ b/scripts/generate_template.py @@ -0,0 +1,505 @@ +#!/usr/bin/env python3 +"""Generate `input.default.toml` from `entity.schema.json`. + +The JSON Schema is the single source of truth for the input file: it drives editor +validation/completion (tombi) *and* the annotated reference input that ships with the +code. This script renders the second from the first, so the two can never drift. + +The schema splits into two halves: + + * Standard JSON Schema keywords -- `type`, `enum`, `minimum`, `items`, `required`, + `default`, ... -- carry everything a validator can check. + * An `x-entity` object per node carries what JSON Schema cannot express, verbatim from + the template's comment annotations: `type` (the literal `@type:` string, e.g. + "array [size 1 :->: 3]"), `default` (for non-JSON defaults such as + "1 [no MPI]; MPI_SIZE [MPI]"), `notes`, `examples`, `enum` (an illustrative, + NON-exhaustive list -- never validated), and `deprecated`. + +`x-entity.inferred` on a table lists quantities the code derives rather than reads. They +are deliberately absent from `properties` (so `additionalProperties: false` rejects them +as input keys) and are emitted here as an `@inferred:` comment block, after that table's +own keys and before its sub-tables. + +Layout rules, matching the hand-written template: + + * a table at depth d gets indent 2*d, its keys 2*(d+1) + * within a table: scalar keys, then the `@inferred:` block, then sub-tables + * a blank line precedes every sub-table (matching `tombi format`) + * every key gets a value: its default under `--defaults`, otherwise `""` -- a blank + form to fill in + +Usage: + + python scripts/generate_template.py -d -o input.default.toml # the reference input + python scripts/generate_template.py -d # ... to stdout + python scripts/generate_template.py # blank form, values "" + diff <(python scripts/generate_template.py -d) input.default.toml + +With `--defaults` every key carries a value instead of `""`: the literal `x-entity.default` +where it is one, else the JSON `default`, else a stand-in derived from the schema's own +constraints (first enum value, `minimum`, `minItems`, ...). That last case covers the +required keys, which the user must supply anyway, and the handful whose documented +default the code computes at runtime -- `N_GHOSTS`, "1% of the domain size", "box centre +pushed back ~1.7 box-diagonals". Those are listed on stderr, and their `@default:` or +`@required` annotation still spells out the real behaviour. +""" + +from __future__ import annotations + +import argparse +import json +import sys +import textwrap +import tomllib +from pathlib import Path +from typing import Any + +REPO = Path(__file__).resolve().parent.parent +DEFAULT_SCHEMA = REPO / "entity.schema.json" + +INDENT = " " + +# order of the `@`-annotations inside a key's comment block +ANNOTATION_ORDER = ("required", "type", "default", "deprecated", "enum", "note", "example") + + +# --------------------------------------------------------------------------- +# schema helpers +# --------------------------------------------------------------------------- + + +def resolve(node: dict, defs: dict) -> dict: + """Follow `$ref`, keeping any sibling keywords (2020-12 allows them).""" + while "$ref" in node: + target = defs[node["$ref"].rsplit("/", 1)[-1]] + merged = dict(target) + merged.update({k: v for k, v in node.items() if k != "$ref"}) + node = merged + return node + + +def kind(node: dict, defs: dict) -> str: + """'table' (TOML table), 'aot' (array of tables), or 'key' (scalar/array value).""" + if node.get("type") == "object" or "properties" in node: + return "table" + items = node.get("items") + if node.get("type") == "array" and isinstance(items, dict): + if resolve(items, defs).get("type") == "object": + return "aot" + return "key" + + +def schema_enum(node: dict, defs: dict) -> list | None: + """First real `enum` reachable through anyOf/oneOf/items (arrays of enums).""" + if "enum" in node: + return node["enum"] + for branch in ("anyOf", "oneOf"): + for sub in node.get(branch, []): + found = schema_enum(resolve(sub, defs), defs) + if found: + return found + items = node.get("items") + if isinstance(items, dict): + return schema_enum(resolve(items, defs), defs) + return None + + +def own_enum(node: dict, defs: dict) -> list | None: + """The node's own `enum`, including through anyOf/oneOf -- but NOT through `items`. + + Unlike `schema_enum`, this does not descend into array elements: it answers "what + values may THIS node take", which is what a placeholder needs. + """ + if "enum" in node: + return node["enum"] + for branch in ("anyOf", "oneOf"): + for sub in node.get(branch, []): + found = own_enum(resolve(sub, defs), defs) + if found: + return found + return None + + +def node_type(node: dict, defs: dict) -> str | None: + """First concrete JSON type of the node, looking into anyOf/oneOf branches.""" + t = node.get("type") + if isinstance(t, list): + return t[0] + if t: + return t + for branch in ("anyOf", "oneOf"): + for sub in node.get(branch, []): + found = node_type(resolve(sub, defs), defs) + if found: + return found + return None + + +def placeholder(node: dict, defs: dict) -> Any: + """A constraint-respecting stand-in for a key the schema gives no default for. + + Used for required keys (which the user must fill in anyway) and for keys whose + documented default is computed at runtime and so has no literal form -- N_GHOSTS, + "1% of the domain size", "box center pushed back ~1.7 box-diagonals", ... + """ + node = resolve(node, defs) + + values = own_enum(node, defs) + if values: + return values[0] + + # an anyOf/oneOf node keeps its constraints inside the branches, so pick the first + # branch that names a type and derive the stand-in from that + if "type" not in node: + for branch in ("anyOf", "oneOf"): + for sub in node.get(branch, []): + sub = resolve(sub, defs) + if node_type(sub, defs): + return placeholder(sub, defs) + + t = node_type(node, defs) + if t == "boolean": + return False + if t in ("integer", "number"): + lo = node.get("minimum") + if lo is None and "exclusiveMinimum" in node: + lo = node["exclusiveMinimum"] + 1 + value = lo if lo is not None else 0 + hi = node.get("maximum") + if hi is not None: + value = min(value, hi) + return int(value) if t == "integer" else float(value) + if t == "string": + return "" + if t == "array": + if "prefixItems" in node: + return [placeholder(i, defs) for i in node["prefixItems"]] + items = node.get("items") + count = node.get("minItems", 0) + if count and isinstance(items, dict): + return [placeholder(items, defs) for _ in range(count)] + return [] + if t == "object": + return {} + return "" + + +def derive_type(node: dict, defs: dict) -> str: + """Fallback `@type` when x-entity.type is absent (it should never be).""" + t = node.get("type") + if isinstance(t, list): + return " | ".join(t) + if t == "array": + items = node.get("items") + inner = derive_type(resolve(items, defs), defs) if isinstance(items, dict) else "any" + return f"array<{inner}>" + if t: + return t + for branch in ("anyOf", "oneOf"): + if branch in node: + return " | ".join(derive_type(resolve(s, defs), defs) for s in node[branch]) + return "any" + + +# --------------------------------------------------------------------------- +# value / annotation formatting +# --------------------------------------------------------------------------- + + +def toml_value(value: Any) -> str: + """Render a JSON default as the TOML literal it corresponds to.""" + if isinstance(value, bool): # before int -- bool is an int subclass + return "true" if value else "false" + if isinstance(value, str): + return json.dumps(value) + if isinstance(value, (int, float)): + return repr(value) + if isinstance(value, list): + return "[" + ", ".join(toml_value(v) for v in value) + "]" + if value is None: + return '""' + return str(value) + + +def as_toml_literal(text: str) -> str | None: + """Return `text` if it is already a standalone TOML value, else None. + + `x-entity.default` is prose more often than not ("N_GHOSTS", "1 [no MPI]; MPI_SIZE + [MPI]"), but when it *is* a literal it is the better source than the JSON `default`, + because it preserves the notation the docs use -- 1e-4 rather than 0.0001. Anything + carrying a comment marker is rejected so trailing asides do not leak into the value. + """ + if "#" in text: + return None + try: + tomllib.loads(f"x = {text}") + except (tomllib.TOMLDecodeError, ValueError): + return None + return text + + +def format_enum(values: list) -> str: + """Join enum values for an `@enum:` line. + + An `x-entity.enum` entry that already carries quotes or spaces is documentation prose + (e.g. the CMasher colormap aside) and is passed through untouched; a bare token is + quoted so it reads as the literal you would type. + """ + out = [] + for v in values: + if isinstance(v, str) and ('"' in v or " " in v): + out.append(v) + else: + out.append(toml_value(v)) + return ", ".join(out) + + +def wrap(text: str, initial: str, subsequent: str, width: int) -> list[str]: + """Wrap `text` to `width`, honouring embedded newlines as hard breaks.""" + lines: list[str] = [] + for i, chunk in enumerate(text.split("\n")): + prefix = initial if i == 0 else subsequent + chunk = chunk.rstrip() + if not chunk: + lines.append(prefix.rstrip()) + continue + lines.extend( + textwrap.wrap( + chunk, + width=width, + initial_indent=prefix, + subsequent_indent=subsequent, + break_long_words=False, + break_on_hyphens=False, + ) + or [prefix.rstrip()] + ) + return lines + + +def describe(text: str, indent: str, width: int) -> list[str]: + """A plain `# ...` description block.""" + return wrap(text, f"{indent}# ", f"{indent}# ", width) + + +def annotate(tag: str, text: str | None, indent: str, width: int) -> list[str]: + """A `# @tag: ...` line, continuations aligned under the text.""" + if text is None: + return [f"{indent}# @{tag}"] + initial = f"{indent}# @{tag}: " + return wrap(text, initial, f"{indent}# " + " " * (len(tag) + 3), width) + + +# --------------------------------------------------------------------------- +# rendering +# --------------------------------------------------------------------------- + + +class Renderer: + def __init__(self, schema: dict, width: int, defaults: bool = False) -> None: + self.defs = schema.get("$defs", {}) + self.width = width + self.defaults = defaults + self.out: list[str] = [] + # keys we had to invent a stand-in for, reported at the end + self.synthesized: list[str] = [] + + def render(self, schema: dict) -> str: + # the root behaves like a table at depth -1: no keys of its own, and its + # sub-tables land at depth 0 + self.render_body(schema, "", -1) + return "\n".join(self.out) + "\n" + + def render_body(self, node: dict, path: str, depth: int) -> None: + indent = INDENT * (depth + 1) + props = node.get("properties") or {} + required = set(node.get("required") or []) + + keys, subtables = [], [] + for name, raw in props.items(): + resolved = resolve(raw, self.defs) + entry = (name, raw, resolved) + (keys if kind(resolved, self.defs) == "key" else subtables).append(entry) + + wrote = False + for name, raw, resolved in keys: + self.emit_key( + name, raw, resolved, indent, name in required, f"{path}.{name}" if path else name + ) + wrote = True + + inferred = (node.get("x-entity") or {}).get("inferred") or [] + if inferred: + if wrote: + self.out.append("") + self.emit_inferred(inferred, indent) + wrote = True + + for name, raw, resolved in subtables: + # tombi puts a blank line before every sub-table, including the first one in + # a parent that has no keys of its own ([radiation] -> [radiation.drag]); + # `self.out` being non-empty is just "not the very first line of the file" + if wrote or self.out: + self.out.append("") + self.emit_table(name, raw, resolved, path, depth + 1) + wrote = True + + def key_value(self, name: str, node: dict, path: str, required: bool) -> str: + """The right-hand side of `name = ...`. + + In template mode every key is an empty string -- a form to fill in. In defaults + mode the precedence is: the literal `x-entity.default`, then the JSON `default`, + then a constraint-derived placeholder (recorded, since it is not a real default). + """ + if not self.defaults: + return '""' + + xe = node.get("x-entity") or {} + documented = xe.get("default") + if isinstance(documented, str): + literal = as_toml_literal(documented) + if literal is not None: + return literal + if "default" in node: + return toml_value(node["default"]) + + reason = "required" if required else (documented or "no documented default") + self.synthesized.append(f"{path} ({reason})") + return toml_value(placeholder(node, self.defs)) + + def emit_key( + self, name: str, raw: dict, node: dict, indent: str, required: bool, path: str = "" + ) -> None: + xe = node.get("x-entity") or {} + description = raw.get("description") or node.get("description") + if description: + self.out += describe(description, indent, self.width) + + if required: + self.out += annotate("required", None, indent, self.width) + + self.out += annotate( + "type", xe.get("type") or derive_type(node, self.defs), indent, self.width + ) + + default = xe.get("default") + if default is None and "default" in node: + default = toml_value(node["default"]) + if default is not None: + self.out += annotate("default", default, indent, self.width) + + if xe.get("deprecated"): + self.out += annotate("deprecated", xe["deprecated"], indent, self.width) + + values = xe.get("enum") or schema_enum(node, self.defs) + if values: + self.out += annotate("enum", format_enum(values), indent, self.width) + + for note in xe.get("notes", []): + self.out += annotate("note", note, indent, self.width) + for example in xe.get("examples", []): + self.out += annotate("example", example, indent, self.width) + + value = self.key_value(name, node, path, required) + self.out.append(f"{indent}{name} = {value}") + + def emit_table(self, name: str, raw: dict, node: dict, parent: str, depth: int) -> None: + indent = INDENT * depth + path = f"{parent}.{name}" if parent else name + is_aot = kind(node, self.defs) == "aot" + + description = raw.get("description") or node.get("description") + if description: + self.out += describe(description, indent, self.width) + for note in (node.get("x-entity") or {}).get("notes", []): + self.out += annotate("note", note, indent, self.width) + + self.out.append(f"{indent}[[{path}]]" if is_aot else f"{indent}[{path}]") + + body = resolve(node["items"], self.defs) if is_aot else node + self.render_body(body, path, depth) + + def emit_inferred(self, entries: list[dict], indent: str) -> None: + self.out.append(f"{indent}# @inferred:") + cont = f"{indent}# " + " " * 8 + for entry in entries: + self.out.append(f"{indent}# - {entry['name']}") + for tag in ("brief", "type", "enum", "from", "value"): + if tag not in entry: + continue + text = format_enum(entry[tag]) if tag == "enum" else str(entry[tag]) + self.out += wrap(text, f"{indent}# @{tag}: ", cont, self.width) + + +# --------------------------------------------------------------------------- + + +def main() -> int: + ap = argparse.ArgumentParser( + description="Generate the annotated input template from entity.schema.json.", + formatter_class=argparse.RawDescriptionHelpFormatter, + ) + ap.add_argument( + "schema", + nargs="?", + type=Path, + default=DEFAULT_SCHEMA, + help=f"JSON Schema to render (default: {DEFAULT_SCHEMA.name} at the repo root)", + ) + ap.add_argument( + "-o", + "--output", + type=Path, + default=None, + help="write here instead of stdout", + ) + ap.add_argument( + "-d", + "--defaults", + action="store_true", + help="fill each key with its default value instead of an empty string", + ) + ap.add_argument( + "-w", + "--width", + type=int, + default=80, + help="column at which comments wrap (default: 80)", + ) + args = ap.parse_args() + + try: + schema = json.loads(args.schema.read_text()) + except FileNotFoundError: + print(f"error: no such schema: {args.schema}", file=sys.stderr) + return 1 + except json.JSONDecodeError as err: + print(f"error: {args.schema} is not valid JSON: {err}", file=sys.stderr) + return 1 + + renderer = Renderer(schema, args.width, defaults=args.defaults) + text = renderer.render(schema) + + if args.output is None: + sys.stdout.write(text) + else: + args.output.write_text(text) + print( + f"wrote {args.output} ({text.count(chr(10))} lines) from {args.schema.name}", + file=sys.stderr, + ) + + if renderer.synthesized: + print( + f"note: {len(renderer.synthesized)} key(s) have no literal default; " + "a constraint-derived stand-in was used (the @default/@required annotation " + "above each one still documents the real behaviour):", + file=sys.stderr, + ) + for entry in renderer.synthesized: + print(f" {entry}", file=sys.stderr) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/ideal_tile_size.py b/scripts/ideal_tile_size.py similarity index 100% rename from ideal_tile_size.py rename to scripts/ideal_tile_size.py diff --git a/render_preview.py b/scripts/render_preview.py similarity index 100% rename from render_preview.py rename to scripts/render_preview.py From 3b7f863f074a89af7598d558f3f0c55e1d22fe52 Mon Sep 17 00:00:00 2001 From: haykh Date: Mon, 21 Sep 2026 15:41:22 -0400 Subject: [PATCH 100/125] scripts to separate folder --- scripts/ideal_tile_size.py | 0 scripts/render_preview.py | 0 2 files changed, 0 insertions(+), 0 deletions(-) mode change 100644 => 100755 scripts/ideal_tile_size.py mode change 100644 => 100755 scripts/render_preview.py diff --git a/scripts/ideal_tile_size.py b/scripts/ideal_tile_size.py old mode 100644 new mode 100755 diff --git a/scripts/render_preview.py b/scripts/render_preview.py old mode 100644 new mode 100755 From 3e1d9b80c4158ce423cd050f52067c107a9ab466 Mon Sep 17 00:00:00 2001 From: haykh Date: Mon, 21 Sep 2026 15:42:47 -0400 Subject: [PATCH 101/125] CITATION --- CITATION | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/CITATION b/CITATION index 8b1d4947d..a199dd8f8 100644 --- a/CITATION +++ b/CITATION @@ -26,7 +26,7 @@ For the general relativistic module, please cite the following paper: ```latex @ARTICLE{EntityGR_2025, author = {{Galishnikova}, Alisa and {Hakobyan}, Hayk and {Philippov}, Alexander and {Crinquand}, Benjamin}, - title = "{$\mathtt{Entity}$ -- Hardware-agnostic Particle-in-Cell Code for Plasma Astrophysics. II: General Relativistic Module}", + title = "{Entity -- Hardware-agnostic Particle-in-Cell Code for Plasma Astrophysics. II: General Relativistic Module}", journal = {arXiv e-prints}, keywords = {High Energy Astrophysical Phenomena}, year = 2025, From bd5773c7ba5b4126edc595c24808fec45274248b Mon Sep 17 00:00:00 2001 From: haykh Date: Mon, 21 Sep 2026 15:53:54 -0400 Subject: [PATCH 102/125] correct schema --- entity.schema.json | 12 ++++++------ input.default.toml | 16 ++++++++-------- 2 files changed, 14 insertions(+), 14 deletions(-) diff --git a/entity.schema.json b/entity.schema.json index 8f4528210..8204f3586 100644 --- a/entity.schema.json +++ b/entity.schema.json @@ -751,10 +751,10 @@ "description": "Minimum photon energy for synchrotron emission (units of `m0 c^2`)", "type": "number", "exclusiveMinimum": 0.0, - "default": 0.0001, + "default": 0.001, "x-entity": { "type": "float [> 0.0]", - "default": "1e-4" + "default": "1e-3" } }, "photon_weight": { @@ -815,10 +815,10 @@ "description": "Minimum photon energy for inverse Compton emission (units of `m0 c^2`)", "type": "number", "exclusiveMinimum": 0.0, - "default": 0.0001, + "default": 0.001, "x-entity": { "type": "float [> 0.0]", - "default": "1e-4" + "default": "1e-3" } }, "photon_weight": { @@ -1377,7 +1377,7 @@ "pattern": "(?i)^(disabled|hdf5|BPFile)$" } ], - "default": "hdf5", + "default": "bpfile", "x-entity": { "type": "string" } @@ -1386,7 +1386,7 @@ "description": "Number of timesteps between all outputs", "type": "integer", "minimum": 1, - "default": 1, + "default": 100, "x-entity": { "type": "uint [> 0]", "notes": [ diff --git a/input.default.toml b/input.default.toml index c887b271d..bb018f76f 100644 --- a/input.default.toml +++ b/input.default.toml @@ -290,8 +290,8 @@ gamma_qed = 10.0 # Minimum photon energy for synchrotron emission (units of `m0 c^2`) # @type: float [> 0.0] - # @default: 1e-4 - photon_energy_min = 1e-4 + # @default: 1e-3 + photon_energy_min = 1e-3 # Weights for the emitted synchrotron photons # @type: float [> 0.0] # @default: 1.0 @@ -327,8 +327,8 @@ gamma_qed = 10.0 # Minimum photon energy for inverse Compton emission (units of `m0 c^2`) # @type: float [> 0.0] - # @default: 1e-4 - photon_energy_min = 1e-4 + # @default: 1e-3 + photon_energy_min = 1e-3 # Weights for the emitted inverse Compton photons # @type: float [> 0.0] # @default: 1.0 @@ -586,14 +586,14 @@ [output] # Output format # @type: string - # @default: "hdf5" + # @default: "bpfile" # @enum: "disabled", "hdf5", "BPFile" - format = "hdf5" + format = "bpfile" # Number of timesteps between all outputs # @type: uint [> 0] - # @default: 1 + # @default: 100 # @note: Value is overriden by output intervals for specific outputs - interval = 1 + interval = 100 # Physical (code) time interval between all outputs # @type: float # @default: -1.0 From ec169053a12b6882c6632183dbf11c8b32d0c567 Mon Sep 17 00:00:00 2001 From: haykh Date: Mon, 21 Sep 2026 16:03:11 -0400 Subject: [PATCH 103/125] devenv setup Copied byte-identical from dev/qed (277eb74f) so a future merge of that branch resolves these as add/add with matching content. Co-Authored-By: Claude Opus 5 (1M context) --- dev/nix/devenv.lock | 45 ++++++++++++ dev/nix/devenv.nix | 173 ++++++++++++++++++++++++++++++++++++++++++++ dev/nix/devenv.yaml | 4 + 3 files changed, 222 insertions(+) create mode 100644 dev/nix/devenv.lock create mode 100644 dev/nix/devenv.nix create mode 100644 dev/nix/devenv.yaml diff --git a/dev/nix/devenv.lock b/dev/nix/devenv.lock new file mode 100644 index 000000000..25a1cc90d --- /dev/null +++ b/dev/nix/devenv.lock @@ -0,0 +1,45 @@ +{ + "nodes": { + "devenv": { + "locked": { + "dir": "src/modules", + "lastModified": 1789340509, + "narHash": "sha256-It2AD16uFwiAe9GvuLbk+j+qTDrQxT3XPclL71ZnJ3I=", + "owner": "cachix", + "repo": "devenv", + "rev": "6e830506b517d6a373f2dec6ee8c7f4683908856", + "type": "github" + }, + "original": { + "dir": "src/modules", + "owner": "cachix", + "repo": "devenv", + "type": "github" + } + }, + "nixpkgs": { + "locked": { + "lastModified": 1789286504, + "narHash": "sha256-eiEK7cKZORNEvX0GeF3RtNEF/JXhgf2RqSp3230q13E=", + "owner": "NixOS", + "repo": "nixpkgs", + "rev": "ef34387ddd751e1ab8857adf4676492d32eb24ec", + "type": "github" + }, + "original": { + "owner": "NixOS", + "ref": "nixos-unstable", + "repo": "nixpkgs", + "type": "github" + } + }, + "root": { + "inputs": { + "devenv": "devenv", + "nixpkgs": "nixpkgs" + } + } + }, + "root": "root", + "version": 7 +} \ No newline at end of file diff --git a/dev/nix/devenv.nix b/dev/nix/devenv.nix new file mode 100644 index 000000000..5da1b6a01 --- /dev/null +++ b/dev/nix/devenv.nix @@ -0,0 +1,173 @@ +# devenv counterpart of `shell.nix`; run from this directory: +# devenv shell # cpu-only +# devenv shell -P cuda -O entity.arch:string AMPERE80 # cuda +# devenv shell -P hip -O entity.arch:string AMD_GFX90A # hip +# devenv shell -P mpi -P hdf5 # adios2 with mpi + hdf5 +# persistent settings can be put into `devenv.local.nix` (gitignored). +{ + pkgs, + lib, + config, + inputs, + ... +}: + +let + cfg = config.entity; + + gpu = lib.toUpper cfg.gpu; + arch = lib.toUpper cfg.arch; + + # `shell.nix` imports nixpkgs with `allowUnfree`/`cudaSupport` decided by the + # requested backend. devenv instantiates its own `pkgs` before this module is + # evaluated, so it cannot be reconfigured from here -- import the same input + # ourselves and build everything from that instance. + nixpkgs = import inputs.nixpkgs { + inherit (pkgs.stdenv.hostPlatform) system; + config = { + allowUnfree = true; + cudaSupport = gpu == "CUDA"; + }; + }; + + adios2Pkg = nixpkgs.callPackage ./adios2.nix { + pkgs = nixpkgs; + inherit (cfg) hdf5 mpi; + }; + + kokkosPkg = nixpkgs.callPackage ./kokkos.nix { + pkgs = nixpkgs; + stdenv = nixpkgs.stdenv; + inherit arch gpu; + }; + + extraPkgs = map (name: nixpkgs.${name}) (lib.filter (s: s != "") (lib.splitString "," cfg.extra)); + + # compilers are picked by the backend; CUDA goes through kokkos' nvcc_wrapper + compilerEnv = + { + NONE = { + CXX = "g++"; + CC = "gcc"; + }; + HIP = { + CXX = "clang++"; + CC = "clang"; + }; + CUDA = { }; + } + .${gpu}; +in +{ + options.entity = { + gpu = lib.mkOption { + # case-insensitive, as in `shell.nix` + type = lib.types.enum [ + "NONE" + "none" + "CUDA" + "cuda" + "HIP" + "hip" + ]; + default = "NONE"; + description = "GPU backend to build Kokkos with."; + }; + + arch = lib.mkOption { + type = lib.types.str; + default = "NATIVE"; + example = "AMPERE80"; + description = '' + Kokkos architecture; mandatory when `gpu` is not `NONE`. See + https://kokkos.org/kokkos-core-wiki/get-started/configuration-guide.html#gpu-architectures + ''; + }; + + hdf5 = lib.mkOption { + type = lib.types.bool; + default = false; + description = "Build ADIOS2 with HDF5 support."; + }; + + mpi = lib.mkOption { + type = lib.types.bool; + default = false; + description = "Build ADIOS2 with MPI support."; + }; + + extra = lib.mkOption { + type = lib.types.str; + default = ""; + example = "gdb,valgrind"; + description = '' + Comma-separated nixpkgs attributes to add to the environment, kept for + parity with `shell.nix`. `-O packages:pkgs "gdb valgrind"` does the same + without going through this option. + ''; + }; + }; + + config = { + name = + "nt2" + (if gpu != "NONE" then "-${lib.toLower gpu}" else "") + (if cfg.mpi then "-mpi" else ""); + + profiles = { + cuda.module = { + entity.gpu = "CUDA"; + }; + hip.module = { + entity.gpu = "HIP"; + }; + mpi.module = { + entity.mpi = true; + }; + hdf5.module = { + entity.hdf5 = true; + }; + }; + + packages = + (with nixpkgs; [ + zlib + cmake + + adios2Pkg + kokkosPkg + + python314 + + cmake-format + cmake-lint + neocmakelsp + black + pyright + taplo + vscode-langservers-extracted + ]) + ++ extraPkgs; + + env = compilerEnv // { + LD_LIBRARY_PATH = lib.makeLibraryPath [ + nixpkgs.stdenv.cc.cc + nixpkgs.zlib + ]; + }; + + enterShell = '' + BLUE='\033[0;34m' + NC='\033[0m' + + echo "following environment variables are set:" + '' + + lib.concatStringsSep "" ( + lib.mapAttrsToList (name: value: '' + echo -e " ''${BLUE}${name}''${NC}=${value}" + '') compilerEnv + ) + + '' + echo "" + echo -e "${config.name} devenv activated" + ''; + }; +} diff --git a/dev/nix/devenv.yaml b/dev/nix/devenv.yaml new file mode 100644 index 000000000..b4c9128be --- /dev/null +++ b/dev/nix/devenv.yaml @@ -0,0 +1,4 @@ +# devenv inputs; update the lock with `devenv update --from path:./dev/nix` +inputs: + nixpkgs: + url: github:NixOS/nixpkgs/nixos-unstable From 6ea5dadc3339d37f24a6130886926b8a502096b0 Mon Sep 17 00:00:00 2001 From: haykh Date: Mon, 21 Sep 2026 16:13:42 -0400 Subject: [PATCH 104/125] adios2 version in nix --- dev/nix/adios2.nix | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/dev/nix/adios2.nix b/dev/nix/adios2.nix index eb3130632..f7114a017 100644 --- a/dev/nix/adios2.nix +++ b/dev/nix/adios2.nix @@ -6,7 +6,7 @@ let name = "adios2"; - version = "2.11.0"; + version = "2.12.1"; cmakeFlags = { CMAKE_CXX_STANDARD = "20"; CMAKE_CXX_EXTENSIONS = "OFF"; @@ -30,7 +30,7 @@ stdenv.mkDerivation { src = pkgs.fetchgit { url = "https://github.com/ornladios/ADIOS2/"; rev = "v${version}"; - sha256 = "sha256-yHPI///17poiCEb7Luu5qfqxTWm9Nh+o9r57mZT26U0="; + sha256 = "sha256-3jMvVYYO93/Pu7RW2x5mzTRMrZ3oC3IwGrUz2tSqJxQ="; }; nativeBuildInputs = with pkgs; [ From d732be0c9e33019486f29537a81da078e0df4b30 Mon Sep 17 00:00:00 2001 From: haykh Date: Mon, 21 Sep 2026 18:00:27 -0400 Subject: [PATCH 105/125] minor --- .tombi.toml | 10 ++++++++-- dev/nix/devenv.lock | 12 ++++++------ dev/nix/devenv.nix | 2 +- entity.schema.json | 20 ++++++++++---------- input.default.toml | 10 +++++----- 5 files changed, 30 insertions(+), 24 deletions(-) diff --git a/.tombi.toml b/.tombi.toml index 4124004e7..ea7e8c58b 100644 --- a/.tombi.toml +++ b/.tombi.toml @@ -12,5 +12,11 @@ toml-version = "v1.0.0" [[schemas]] path = "entity.schema.json" - include = ["*.toml"] - exclude = [".tombi.toml"] + include = ["**/*.toml"] + exclude = [ + ".tombi.toml", + "extern/**", # submodules: adios2's pyproject/REUSE, entity-pgens' own configs + ".venv/**", + "build/**", + "**/*.ckpt/**", # checkpoint metadata dumps carry a [metadata] table, not input + ] diff --git a/dev/nix/devenv.lock b/dev/nix/devenv.lock index 25a1cc90d..9e6f7f3b4 100644 --- a/dev/nix/devenv.lock +++ b/dev/nix/devenv.lock @@ -3,11 +3,11 @@ "devenv": { "locked": { "dir": "src/modules", - "lastModified": 1789340509, - "narHash": "sha256-It2AD16uFwiAe9GvuLbk+j+qTDrQxT3XPclL71ZnJ3I=", + "lastModified": 1789991752, + "narHash": "sha256-SiH+H00IutELZBg7BBNo6r5DU6E+DG7DjrbeX5+QLa0=", "owner": "cachix", "repo": "devenv", - "rev": "6e830506b517d6a373f2dec6ee8c7f4683908856", + "rev": "1c57b5dea0d400af97053fdd1a536fca17378f73", "type": "github" }, "original": { @@ -19,11 +19,11 @@ }, "nixpkgs": { "locked": { - "lastModified": 1789286504, - "narHash": "sha256-eiEK7cKZORNEvX0GeF3RtNEF/JXhgf2RqSp3230q13E=", + "lastModified": 1789921291, + "narHash": "sha256-Ft/BRnIqw1MywFoXydKobjjWmDFgDdYtSpJliE8+yUw=", "owner": "NixOS", "repo": "nixpkgs", - "rev": "ef34387ddd751e1ab8857adf4676492d32eb24ec", + "rev": "44a91898084f46797b5fac650c7e8c9ac38c43d4", "type": "github" }, "original": { diff --git a/dev/nix/devenv.nix b/dev/nix/devenv.nix index 5da1b6a01..8b78c6651 100644 --- a/dev/nix/devenv.nix +++ b/dev/nix/devenv.nix @@ -142,7 +142,7 @@ in neocmakelsp black pyright - taplo + tombi vscode-langservers-extracted ]) ++ extraPkgs; diff --git a/entity.schema.json b/entity.schema.json index 8204f3586..da44f30f9 100644 --- a/entity.schema.json +++ b/entity.schema.json @@ -158,7 +158,7 @@ 1 ], "x-entity": { - "type": "array of int, subset of [1, 2, 3]" + "type": "array [subset of {1, 2, 3}]" } }, "tolerance": { @@ -1112,8 +1112,7 @@ "type": "object", "additionalProperties": false, "required": [ - "ppc0", - "species" + "ppc0" ], "x-entity": { "inferred": [ @@ -1299,12 +1298,13 @@ "enum": [ "None", "Synchrotron", - "Compton" + "Compton", + "Custom" ] }, { "type": "string", - "pattern": "(?i)^(None|Synchrotron|Compton)$" + "pattern": "(?i)^(None|Synchrotron|Compton|Custom)$" } ], "default": "None", @@ -1600,11 +1600,11 @@ "stride": { "description": "Stride for the output of particles", "type": "integer", - "exclusiveMinimum": 1, "default": 100, "x-entity": { - "type": "uint [> 1]" - } + "type": "uint [>= 1]" + }, + "minimum": 1 }, "interval": { "description": "Number of timesteps between particle outputs", @@ -1674,7 +1674,7 @@ "type": "bool" } }, - "num_bins": { + "n_bins": { "description": "Number of energy bins for the spectra output", "type": "integer", "minimum": 1, @@ -1682,7 +1682,7 @@ "deprecated": true, "x-entity": { "type": "uint [> 0]", - "deprecated": "starting v1.5.0" + "deprecated": "removed in 1.6+, use `num_energy_bins` instead" } }, "num_energy_bins": { diff --git a/input.default.toml b/input.default.toml index bb018f76f..e6753ce69 100644 --- a/input.default.toml +++ b/input.default.toml @@ -46,7 +46,7 @@ # @default: 0 interval = 0 # Dimensions along which load is redistributed (1 = x1, 2 = x2, 3 = x3) - # @type: array of int, subset of [1, 2, 3] + # @type: array [subset of {1, 2, 3}] # @default: [1] # @enum: 1, 2, 3 dimensions = [1] @@ -558,7 +558,7 @@ # Particle emission policy for the species # @type: string # @default: "None" - # @enum: "None", "Synchrotron", "Compton" + # @enum: "None", "Synchrotron", "Compton", "Custom" # @note: Only one emission mechanism allowed # @note: Appropriate radiation drag flag will be applied automatically # (unless explicitly set to "None") @@ -676,7 +676,7 @@ # @note: If empty, all species are output species = [] # Stride for the output of particles - # @type: uint [> 1] + # @type: uint [>= 1] # @default: 100 stride = 100 # Number of timesteps between particle outputs @@ -713,8 +713,8 @@ # Number of energy bins for the spectra output # @type: uint [> 0] # @default: 200 - # @deprecated: starting v1.5.0 - num_bins = 200 + # @deprecated: removed in 1.6+, use `num_energy_bins` instead + n_bins = 200 # Number of energy bins for the spectra output # @type: uint [> 0] # @default: 200 From 6e530615b966c5603e0efdd2a102fdc065f6b834 Mon Sep 17 00:00:00 2001 From: haykh Date: Thu, 24 Sep 2026 19:14:45 -0400 Subject: [PATCH 106/125] cmake v bump, team_policy/vendor_sort moved to separate file --- CMakeLists.txt | 75 +++++++++++++------ ...{team_policy.cmake => tiled_deposit.cmake} | 0 2 files changed, 53 insertions(+), 22 deletions(-) rename cmake/{team_policy.cmake => tiled_deposit.cmake} (100%) diff --git a/CMakeLists.txt b/CMakeLists.txt index d9d1c0ce4..1e971071d 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -1,6 +1,6 @@ # cmake-lint: disable=C0103,C0111,E1120,R0913,R0915 -cmake_minimum_required(VERSION 3.16) +cmake_minimum_required(VERSION 3.22) cmake_policy(SET CMP0110 NEW) set(PROJECT_NAME entity) @@ -58,27 +58,42 @@ set(gpu_aware_mpi ${default_gpu_aware_mpi} CACHE BOOL "Enable GPU-aware MPI") -set(team_policy - ${default_team_policy} - CACHE BOOL "Enable team_policy tile-blocked deposit/pusher kernels") -set(team_policy_tile_size - ${default_team_policy_tile_size} - CACHE STRING "team_policy tile edge length in cells") -set(team_policy_tile_sizes +# deprecated `team_policy*` options (renamed to `tiled_deposit*`); forward them +# unless the new name was given explicitly, and drop the stale cache entries +foreach(_suffix "" "_tile_size" "_drift") + if(DEFINED team_policy${_suffix}) + if(NOT DEFINED tiled_deposit${_suffix}) + message(WARNING "`team_policy${_suffix}` is deprecated, " + "use `tiled_deposit${_suffix}` instead") + if("${_suffix}" STREQUAL "") + set(_type BOOL) + else() + set(_type STRING) + endif() + set(tiled_deposit${_suffix} + ${team_policy${_suffix}} + CACHE ${_type} "") + endif() + unset(team_policy${_suffix} CACHE) + endif() +endforeach() +unset(team_policy_tile_sizes CACHE) + +set(tiled_deposit + ${default_tiled_deposit} + CACHE BOOL "Enable tile-blocked deposit/pusher kernels") +set(tiled_deposit_tile_size + ${default_tiled_deposit_tile_size} + CACHE STRING "tiled deposit tile edge length in cells") +set(tiled_deposit_tile_sizes "4;6;8;10;12;14;16" - CACHE STRING "team_policy tile-size choices") -set(team_policy_drift - ${default_team_policy_drift} - CACHE - STRING - "team_policy tiled-deposit scratch halo drift in cells (max cells a particle may move between two sorts). Sizes the deposit scratch halo only; the sort cadence is set at runtime via spatial_sorting_interval. Default 1." -) + CACHE STRING "tiled deposit tile-size choices") +set(tiled_deposit_drift + ${default_tiled_deposit_drift} + CACHE STRING "tiled deposit scratch halo drift in cells") set(vendor_sort ${default_vendor_sort} - CACHE - BOOL - "Use the vendor sort_by_key (oneDPL/Thrust/rocThrust) for the team_policy spatial sort when available. OFF forces the Kokkos::BinSort fallback, which sorts each SoA member in place (lower peak memory, no maxnpart gather buffer) at the cost of sort speed." -) + CACHE BOOL "Use the vendor sort_by_key") # -------------------------- Compilation settings -------------------------- # set(CMAKE_CXX_STANDARD 20) @@ -158,9 +173,25 @@ else() set(DEVICE_ENABLED OFF) endif() -# ------------------------------ team_policy wiring ------------------------ # -if(${team_policy}) - include(${CMAKE_CURRENT_SOURCE_DIR}/cmake/team_policy.cmake) +if(NOT ${DEVICE_ENABLED}) + set(vendor_sort OFF) + set(gpu_aware_mpi OFF) +endif() + +# tiled deposit +if(${tiled_deposit}) + include(${CMAKE_CURRENT_SOURCE_DIR}/cmake/tiled_deposit.cmake) +else() + message( + STATUS "tiled_deposit=OFF; using global deposit scheme with ScatterViews") +endif() + +# vendor-specific sorting routines +if(${vendor_sort}) + include(${CMAKE_CURRENT_SOURCE_DIR}/cmake/vendor_sort.cmake) +else() + message(STATUS "vendor_sort=OFF; forcing Kokkos::BinSort " + "fallback for spatial sort_by_key") endif() # MPI diff --git a/cmake/team_policy.cmake b/cmake/tiled_deposit.cmake similarity index 100% rename from cmake/team_policy.cmake rename to cmake/tiled_deposit.cmake From ed57031dd6e348de42cb04e4ed583974aff51805 Mon Sep 17 00:00:00 2001 From: haykh Date: Thu, 24 Sep 2026 19:15:00 -0400 Subject: [PATCH 107/125] bump adios2 v --- cmake/dependencies.cmake | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/cmake/dependencies.cmake b/cmake/dependencies.cmake index 93a8a17da..07575ab34 100644 --- a/cmake/dependencies.cmake +++ b/cmake/dependencies.cmake @@ -10,7 +10,7 @@ set(adios2_REPOSITORY https://github.com/ornladios/ADIOS2.git CACHE STRING "ADIOS2 repository") set(adios2_TAG - v2.11.0 + v2.12.1 CACHE STRING "ADIOS2 tag") set(CONNECTION_CHECKED From 6ac7f48b0ee16497907023fc742bfb57b9ed5218 Mon Sep 17 00:00:00 2001 From: haykh Date: Thu, 24 Sep 2026 19:15:12 -0400 Subject: [PATCH 108/125] tiled_deposit and vendor_sort specific cmake --- cmake/tiled_deposit.cmake | 71 ++++----------------------------------- cmake/vendor_sort.cmake | 51 ++++++++++++++++++++++++++++ 2 files changed, 57 insertions(+), 65 deletions(-) create mode 100644 cmake/vendor_sort.cmake diff --git a/cmake/tiled_deposit.cmake b/cmake/tiled_deposit.cmake index 1217bc28c..608e8cf80 100644 --- a/cmake/tiled_deposit.cmake +++ b/cmake/tiled_deposit.cmake @@ -1,12 +1,12 @@ -list(FIND team_policy_tile_sizes "${team_policy_tile_size}" _tps_idx) +list(FIND tiled_deposit_tile_sizes "${tiled_deposit_tile_size}" _tps_idx) if(_tps_idx EQUAL -1) message( FATAL_ERROR - "${Red}team_policy_tile_size must be one of ${team_policy_tile_sizes}, " - "got '${team_policy_tile_size}'${ColorReset}") + "${Red}tiled_deposit_tile_size must be one of ${tiled_deposit_tile_sizes}, " + "got '${tiled_deposit_tile_size}'${ColorReset}") endif() -add_compile_options("-D TEAM_POLICY") -add_compile_options("-D TEAM_POLICY_TILE_SIZE=${team_policy_tile_size}") +add_compile_options("-D TILED_DEPOSIT") +add_compile_options("-D TILED_DEPOSIT_TILE_SIZE=${tiled_deposit_tile_size}") # Compile-time tiled-deposit scratch halo drift. Sizes the halo so a particle # that drifts up to DRIFT cells between two sorts still deposits inside its tile @@ -14,63 +14,4 @@ add_compile_options("-D TEAM_POLICY_TILE_SIZE=${team_policy_tile_size}") # valve (correct, only slower). This is independent of the sort cadence, which # is set at runtime via `spatial_sorting_interval`. Defaults to 1 (the # sorted-every-step case). -add_compile_options("-D TEAM_POLICY_DRIFT=${team_policy_drift}") - -# Vendor sort: oneDPL on SYCL, Thrust on CUDA, rocThrust/rocprim on HIP. When -# `vendor_sort` is ON (default) the available library is detected and used; the -# spatial sort then builds a single permutation that gathers all SoA members. -# When `vendor_sort` is OFF, or no library is found, the code falls back to -# Kokkos::BinSort, which sorts each member in place -- lower peak memory and no -# maxnpart gather buffer, at the cost of sort speed (negligible when sorting is -# a small fraction of the step). The `vendor_sort` knob lets you force the -# BinSort fallback even when a vendor library is present. -if(${vendor_sort}) - if("${Kokkos_DEVICES}" MATCHES "SYCL") - find_package(oneDPL QUIET) - if(oneDPL_FOUND) - message(STATUS "team_policy: oneDPL found, enabling SYCL sort_by_key") - add_compile_options("-D ONEDPL_ENABLED") - set(DEPENDENCIES ${DEPENDENCIES} oneDPL) - else() - message(STATUS "team_policy: oneDPL not found; using BinSort fallback " - "for SYCL sort_by_key") - endif() - endif() - - if("${Kokkos_DEVICES}" MATCHES "CUDA") - find_package(Thrust QUIET) - if(Thrust_FOUND) - message(STATUS "team_policy: Thrust enabled for CUDA sort_by_key") - add_compile_options("-D THRUST_ENABLED") - else() - message(STATUS "team_policy: Thrust not found; using BinSort fallback " - "for CUDA sort_by_key") - endif() - endif() - - if("${Kokkos_DEVICES}" MATCHES "HIP") - # rocThrust ships with ROCm. The HIP sort_by_key path uses rocprim's - # bounded-bit radix sort directly (rocprim is rocThrust's own dependency, so - # its headers come in transitively; we find it explicitly to keep the - # include path robust). This builds a single permutation that gathers all - # SoA members, instead of the legacy per-member Kokkos::BinSort path which - # allocates a fresh `sorted_values` buffer for every member every step (the - # dominant source of allocator churn / fragmentation on ROCm). - find_package(rocthrust QUIET) - if(rocthrust_FOUND) - message(STATUS "team_policy: rocThrust enabled for HIP sort_by_key") - add_compile_options("-D ROCTHRUST_ENABLED") - set(DEPENDENCIES ${DEPENDENCIES} roc::rocthrust) - find_package(rocprim QUIET) - if(rocprim_FOUND) - set(DEPENDENCIES ${DEPENDENCIES} roc::rocprim) - endif() - else() - message(STATUS "team_policy: rocThrust not found; using BinSort " - "fallback for HIP sort_by_key") - endif() - endif() -else() - message(STATUS "team_policy: vendor_sort=OFF; forcing Kokkos::BinSort " - "fallback for spatial sort_by_key") -endif() +add_compile_options("-D TILED_DEPOSIT_DRIFT=${tiled_deposit_drift}") diff --git a/cmake/vendor_sort.cmake b/cmake/vendor_sort.cmake new file mode 100644 index 000000000..16176604e --- /dev/null +++ b/cmake/vendor_sort.cmake @@ -0,0 +1,51 @@ +# Vendor sort: oneDPL on SYCL, Thrust on CUDA, rocThrust/rocprim on HIP. When +# `vendor_sort` is ON (default) the available library is detected and used; the +# spatial sort then builds a single permutation that gathers all SoA members. +# When `vendor_sort` is OFF, or no library is found, the code falls back to +# Kokkos::BinSort, which sorts each member in place -- lower peak memory and no +# maxnpart gather buffer, at the cost of sort speed (negligible when sorting is +# a small fraction of the step). The `vendor_sort` knob lets you force the +# BinSort fallback even when a vendor library is present. +if("${Kokkos_DEVICES}" MATCHES "SYCL") + find_package(oneDPL QUIET) + if(oneDPL_FOUND) + message(STATUS "oneDPL found, enabling SYCL sort_by_key") + add_compile_options("-D ONEDPL_ENABLED") + set(DEPENDENCIES ${DEPENDENCIES} oneDPL) + else() + message(STATUS "oneDPL not found; using BinSort fallback " + "for SYCL sort_by_key") + endif() +elseif("${Kokkos_DEVICES}" MATCHES "CUDA") + find_package(Thrust QUIET) + if(Thrust_FOUND) + message(STATUS "Thrust enabled for CUDA sort_by_key") + add_compile_options("-D THRUST_ENABLED") + else() + message(STATUS "Thrust not found; using BinSort fallback " + "for CUDA sort_by_key") + endif() +elseif("${Kokkos_DEVICES}" MATCHES "HIP") + # rocThrust ships with ROCm. The HIP sort_by_key path uses rocprim's + # bounded-bit radix sort directly (rocprim is rocThrust's own dependency, so + # its headers come in transitively; we find it explicitly to keep the include + # path robust). This builds a single permutation that gathers all SoA members, + # instead of the legacy per-member Kokkos::BinSort path which allocates a + # fresh `sorted_values` buffer for every member every step (the dominant + # source of allocator churn / fragmentation on ROCm). + find_package(rocthrust QUIET) + if(rocthrust_FOUND) + message(STATUS "rocThrust enabled for HIP sort_by_key") + add_compile_options("-D ROCTHRUST_ENABLED") + set(DEPENDENCIES ${DEPENDENCIES} roc::rocthrust) + find_package(rocprim QUIET) + if(rocprim_FOUND) + set(DEPENDENCIES ${DEPENDENCIES} roc::rocprim) + endif() + else() + message(STATUS "rocThrust not found; using BinSort " + "fallback for HIP sort_by_key") + endif() +else() + message(FATAL_ERROR "vendor_sort enabled, but device not recognized") +endif() From 7e22bb620c41e3408f83f3fac7a0daf20de79d45 Mon Sep 17 00:00:00 2001 From: haykh Date: Thu, 24 Sep 2026 19:15:21 -0400 Subject: [PATCH 109/125] update on defaults and report --- cmake/defaults.cmake | 30 +++++++++++++++++------------- cmake/report.cmake | 30 +++++++++++++++--------------- 2 files changed, 32 insertions(+), 28 deletions(-) diff --git a/cmake/defaults.cmake b/cmake/defaults.cmake index 888be1b00..60b97e0d7 100644 --- a/cmake/defaults.cmake +++ b/cmake/defaults.cmake @@ -93,16 +93,22 @@ endif() set_property(CACHE default_gpu_aware_mpi PROPERTY TYPE BOOL) -if(DEFINED ENV{Entity_ENABLE_TEAM_POLICY}) - set(default_team_policy +if(DEFINED ENV{Entity_ENABLE_TILED_DEPOSIT}) + set(default_tiled_deposit + $ENV{Entity_ENABLE_TILED_DEPOSIT} + CACHE INTERNAL "Default flag for tiled_deposit tile-blocked kernels") +elseif(DEFINED ENV{Entity_ENABLE_TEAM_POLICY}) + message(WARNING "`Entity_ENABLE_TEAM_POLICY` is deprecated, " + "use `Entity_ENABLE_TILED_DEPOSIT` instead") + set(default_tiled_deposit $ENV{Entity_ENABLE_TEAM_POLICY} - CACHE INTERNAL "Default flag for team_policy tile-blocked kernels") + CACHE INTERNAL "Default flag for tiled_deposit tile-blocked kernels") else() - set(default_team_policy + set(default_tiled_deposit OFF - CACHE INTERNAL "Default flag for team_policy tile-blocked kernels") + CACHE INTERNAL "Default flag for tiled_deposit tile-blocked kernels") endif() -set_property(CACHE default_team_policy PROPERTY TYPE BOOL) +set_property(CACHE default_tiled_deposit PROPERTY TYPE BOOL) if(DEFINED ENV{Entity_ENABLE_VENDOR_SORT}) set(default_vendor_sort @@ -117,13 +123,11 @@ else() endif() set_property(CACHE default_vendor_sort PROPERTY TYPE BOOL) -set(default_team_policy_tile_size +set(default_tiled_deposit_tile_size 8 - CACHE INTERNAL "Default tile edge length in cells for team_policy") + CACHE INTERNAL "Default tile edge length in cells for tiled_deposit") -set(default_team_policy_drift +set(default_tiled_deposit_drift 1 - CACHE - INTERNAL - "Default tiled-deposit scratch halo drift for team_policy (cells between sorts)" -) + CACHE INTERNAL + "Default tiled-deposit scratch halo drift (cells between sorts)") diff --git a/cmake/report.cmake b/cmake/report.cmake index e6ad4aaf8..f9470d602 100644 --- a/cmake/report.cmake +++ b/cmake/report.cmake @@ -121,22 +121,22 @@ printchoices( GPU_AWARE_MPI_REPORT 44) printchoices( - "Team Policy" - "team_policy" + "Tiled Deposit" + "tiled_deposit" "${ON_OFF_VALUES}" - ${team_policy} + ${tiled_deposit} OFF "${Green}" - TEAM_POLICY_REPORT + TILED_DEPOSIT_REPORT 44) printchoices( "Tile Size" - "team_policy_tile_size" - "${team_policy_tile_sizes}" - ${team_policy_tile_size} - ${default_team_policy_tile_size} + "tiled_deposit_tile_size" + "${tiled_deposit_tile_sizes}" + ${tiled_deposit_tile_size} + ${default_tiled_deposit_tile_size} "${Blue}" - TEAM_POLICY_TILE_SIZE_REPORT + TILED_DEPOSIT_TILE_SIZE_REPORT 44) printchoices( "Vendor sort" @@ -246,18 +246,18 @@ string( " " ${GPU_AWARE_MPI_REPORT} "\n" - " > Team-policy specs" - " ${Dim}[requires team_policy=ON]${ColorReset}" + " > Tiled-deposit specs" + " ${Dim}[requires tiled_deposit=ON]${ColorReset}" "\n" " " - ${TEAM_POLICY_REPORT} + ${TILED_DEPOSIT_REPORT} "\n" " " - ${TEAM_POLICY_TILE_SIZE_REPORT} + ${TILED_DEPOSIT_TILE_SIZE_REPORT} "\n" " " - "- Deposit drift [${Magenta}team_policy_drift${ColorReset}]: " - ${team_policy_drift} + "- Deposit drift [${Magenta}tiled_deposit_drift${ColorReset}]: " + ${tiled_deposit_drift} "\n") string( From e4303a60f2eb6d5f6b3b445e278dd2ed8ea12801 Mon Sep 17 00:00:00 2001 From: haykh Date: Thu, 24 Sep 2026 19:15:55 -0400 Subject: [PATCH 110/125] [BUG!!!] FixFieldsConst was silently ignored --- examples/custom_particle_update/pgen.hpp | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/examples/custom_particle_update/pgen.hpp b/examples/custom_particle_update/pgen.hpp index 30e464099..95033de44 100644 --- a/examples/custom_particle_update/pgen.hpp +++ b/examples/custom_particle_update/pgen.hpp @@ -116,7 +116,8 @@ namespace user { arch::InjectGlobally(metadomain, local_domain, (spidx_t)2, data_i); } - auto FixFieldsConst(const bc_in&, const em&) const -> std::pair { + auto FixFieldsConst(simtime_t, const bc_in&, const em&) const + -> std::pair { return { ZERO, false }; } From 83da0a44c5c80bad3a7f9a24b49766667d67e22d Mon Sep 17 00:00:00 2001 From: haykh Date: Thu, 24 Sep 2026 19:16:12 -0400 Subject: [PATCH 111/125] [BUGv2!!!] FixFieldsConst was silently ignored in shock --- pgens/shock/pgen.hpp | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/pgens/shock/pgen.hpp b/pgens/shock/pgen.hpp index 7a7aaac21..223e05696 100644 --- a/pgens/shock/pgen.hpp +++ b/pgens/shock/pgen.hpp @@ -120,7 +120,7 @@ namespace user { return init_flds; } - auto FixFieldsConst(const bc_in&, const em& comp) const + auto FixFieldsConst(simtime_t, const bc_in&, const em& comp) const -> std::pair { if (comp == em::ex1) { return { init_flds.ex1({ ZERO }), true }; From 6b7e0054e190c9453a1d9d829bf7a7faf046b2c4 Mon Sep 17 00:00:00 2001 From: haykh Date: Thu, 24 Sep 2026 19:16:26 -0400 Subject: [PATCH 112/125] render_preview -> render + movie capability --- scripts/{render_preview.py => render.py} | 515 ++++++++++++++++------- 1 file changed, 365 insertions(+), 150 deletions(-) rename scripts/{render_preview.py => render.py} (70%) diff --git a/scripts/render_preview.py b/scripts/render.py similarity index 70% rename from scripts/render_preview.py rename to scripts/render.py index 77891cd21..05bb3635b 100755 --- a/scripts/render_preview.py +++ b/scripts/render.py @@ -1,44 +1,10 @@ #!/usr/bin/env python3 """ -render_preview.py -- fast, data-free preview of the entity in-situ renderer's -SCENE GEOMETRY. - -Purpose -------- Reads a simulation `.toml` and draws the domain box / camera framing / axes / region crop / field-line seed lattice, WITHOUT any simulation data or ray-marching. It lets you iterate on camera orientation (e.g. the domain cube of a 3D turbulence run) and framing without relaunching the simulation. -It reproduces the SAME camera / projection the C++ renderer uses, so the preview -is trustworthy: the box you see here is the box the renderer will draw. - - IMPORTANT: the camera / projection / region / field-line-lattice math below is - a faithful port of the C++ renderer. If the C++ changes, THIS MUST BE UPDATED - IN SYNC. The controlling C++ sources (verified line-by-line while writing this) - are: - - src/output/render/renderer.cpp - Renderer::init -> toml parse, region resolution, default camera, - ortho/persp basis, ortho_height default = box diag, - eye = center + 1.7*diag*(1,1,1)/sqrt3, fov 35 deg - Renderer::updateForTime -> moving view (pure pan of region + eye) - - src/output/render/composite.h - projectToScreen (~L85-112) -> world -> pixel, ortho & perspective - screenBBox (~L121-163) - - src/output/render/raymarch.hpp - ray generation (~L308-335) -- the inverse of projectToScreen - - src/output/render/axes.h - drawAxes3D / drawAxes2D / drawAxesPolar, niceTicks / niceNum tick style - - src/framework/domain/metadomain_render.cpp - 2D window derivation (Cartesian window vs. spherical meridional wedge, - X = r sin th, Z = r cos th, aspect expansion, mirror), field-line setup - - src/framework/parameters/grid.cpp (~L470-497) - extent parse + theta,phi auto-fill for non-Cartesian metrics - - src/output/render/fieldlines.h - 3D seed lattice: spacing = max(seed_px,1)*wpp, grown by - cbrt(n_seed/seed_max) if over seed_max, ns[d]=floor(size[d]/spacing), - seeds at cell centers. - Modes ----- * 3D (Cartesian only -- the renderer's only 3D mode): projects the domain cube @@ -48,18 +14,12 @@ * 2D spherical/GR: meridional wedge (arcs at r in {rmin,rmax}, rays at theta in {tmin,tmax}), mirrored into a full disk if `mirror`. * 1D: nothing to render (warns). - -Usage ------ - module load python/3.13.0 - python render_preview.py [--out preview.png] [--time T] [--scene N] - -If --out is omitted, saves to /_preview.png. """ import argparse import math import os +import subprocess import sys try: @@ -67,9 +27,9 @@ except ModuleNotFoundError: # Python 3.10 and older (e.g. the miniforge3 module) import tomli as tomllib # same load() API +import matplotlib import numpy as np -import matplotlib matplotlib.use("Agg") # headless cluster: no interactive display import matplotlib.pyplot as plt from matplotlib.patches import Rectangle @@ -91,6 +51,8 @@ def find_or(d, default, *keys): # metric / extent handling (grid.cpp ~L416-497) # # --------------------------------------------------------------------------- # CARTESIAN_METRICS = {"minkowski"} + + # everything else that entity supports is curvilinear (r-first extent): # spherical, qspherical, kerr_schild, kerr_schild_0, qkerr_schild def is_cartesian(metric_name): @@ -125,7 +87,7 @@ def global_extent(td): # (grid.cpp errors if >1 row is supplied for non-cartesian; we just # keep the r-row and append.) ext = [ext[0]] - ext.append([0.0, math.pi]) # theta in [0, pi] (2D and 3D) + ext.append([0.0, math.pi]) # theta in [0, pi] (2D and 3D) if dim == 3: ext.append([0.0, 2.0 * math.pi]) # phi in [0, 2pi] (3D only) @@ -147,9 +109,11 @@ def _norm3(a): def _cross3(a, b): - return (a[1] * b[2] - a[2] * b[1], - a[2] * b[0] - a[0] * b[2], - a[0] * b[1] - a[1] * b[0]) + return ( + a[1] * b[2] - a[2] * b[1], + a[2] * b[0] - a[0] * b[2], + a[0] * b[1] - a[1] * b[0], + ) def _dot3(a, b): @@ -177,7 +141,8 @@ def __init__(self, td, region, width, height): fov = float(find_or(td, 35.0, "output", "render", "camera", "fov")) # default ortho_height covers the box from any view -> == box diagonal ortho_height = float( - find_or(td, diag, "output", "render", "camera", "ortho_height")) + find_or(td, diag, "output", "render", "camera", "ortho_height") + ) # default eye: box center pushed back along (1,1,1) by ~1.7 diagonals eye = [0.0, 0.0, 0.0] @@ -193,8 +158,7 @@ def __init__(self, td, region, width, height): else: upv = [0.0, 0.0, 1.0] - forward = _norm3([lookat[0] - eye[0], lookat[1] - eye[1], - lookat[2] - eye[2]]) + forward = _norm3([lookat[0] - eye[0], lookat[1] - eye[1], lookat[2] - eye[2]]) right = _norm3(_cross3(forward, upv)) up_cam = _cross3(right, forward) # already unit (right,forward unit & perp) @@ -253,7 +217,9 @@ def resolve_region(td, ext): if not lim: continue if len(lim) != 2 or lim[1] <= lim[0]: - print(f" warning: output.render.{keys[d]} must be [lo,hi] with hi>lo; ignoring") + print( + f" warning: output.render.{keys[d]} must be [lo,hi] with hi>lo; ignoring" + ) continue lo = max(float(lim[0]), ext[d][0]) hi = min(float(lim[1]), ext[d][1]) @@ -261,7 +227,9 @@ def resolve_region(td, ext): region[d] = [lo, hi] has_region = True else: - print(f" warning: output.render.{keys[d]} does not overlap the domain; ignoring") + print( + f" warning: output.render.{keys[d]} does not overlap the domain; ignoring" + ) return [tuple(p) for p in region], has_region @@ -293,12 +261,12 @@ def _nice_num(x, do_round): if x <= 0.0: return 1.0 e = math.floor(math.log10(x)) - f = x / (10.0 ** e) + f = x / (10.0**e) if do_round: nf = 1.0 if f < 1.5 else (2.0 if f < 3.0 else (5.0 if f < 7.0 else 10.0)) else: nf = 1.0 if f <= 1.0 else (2.0 if f <= 2.0 else (5.0 if f <= 5.0 else 10.0)) - return nf * (10.0 ** e) + return nf * (10.0**e) def nice_ticks(lo, hi, n): @@ -355,13 +323,13 @@ def count_seeds(sp): ns = [] tot = 1 for d in range(3): - n = max(1, int(math.floor(size[d] / sp))) if sp > 0 else 1 + n = max(1, math.floor(size[d] / sp)) if sp > 0 else 1 ns.append(n) tot *= n return tot, ns n_seed, ns = count_seeds(spacing) - if n_seed > seed_max and seed_max > 0: + if n_seed > seed_max > 0: grow = (float(n_seed) / float(seed_max)) ** (1.0 / 3.0) spacing *= grow n_seed, ns = count_seeds(spacing) @@ -370,11 +338,13 @@ def count_seeds(sp): for k in range(ns[2]): for j in range(ns[1]): for i in range(ns[0]): - seeds.append(( - origin[0] + (i + 0.5) * size[0] / ns[0], - origin[1] + (j + 0.5) * size[1] / ns[1], - origin[2] + (k + 0.5) * size[2] / ns[2], - )) + seeds.append( + ( + origin[0] + (i + 0.5) * size[0] / ns[0], + origin[1] + (j + 0.5) * size[1] / ns[1], + origin[2] + (k + 0.5) * size[2] / ns[2], + ) + ) return seeds, ns @@ -386,19 +356,30 @@ def cube_corners(box): (matches the C++ corner() ordering in axes.h / screenBBox).""" corners = [] for m in range(8): - corners.append(( - box[0][1] if (m & 1) else box[0][0], - box[1][1] if (m & 2) else box[1][0], - box[2][1] if (m & 4) else box[2][0], - )) + corners.append( + ( + box[0][1] if (m & 1) else box[0][0], + box[1][1] if (m & 2) else box[1][0], + box[2][1] if (m & 4) else box[2][0], + ) + ) return corners # the 12 edges as (corner_i, corner_j) index pairs CUBE_EDGES = [ - (0, 1), (2, 3), (4, 5), (6, 7), # x-parallel - (0, 2), (1, 3), (4, 6), (5, 7), # y-parallel - (0, 4), (1, 5), (2, 6), (3, 7), # z-parallel + (0, 1), + (2, 3), + (4, 5), + (6, 7), # x-parallel + (0, 2), + (1, 3), + (4, 6), + (5, 7), # y-parallel + (0, 4), + (1, 5), + (2, 6), + (3, 7), # z-parallel ] @@ -430,7 +411,7 @@ def select_axis_edge_3d(cam, d, cx, cy, ccx, ccy): the projected box centroid. Returns (m0, m1, pxd, pyd): the edge's two corner indices (m0 has axis d at its low end) and the unit screen-space OUTWARD push direction (perpendicular to the edge, pointing away from the centroid).""" - e1 = 1 if d == 0 else 0 # the two perpendicular axes + e1 = 1 if d == 0 else 0 # the two perpendicular axes e2 = 1 if d == 2 else 2 best = None for s1 in (0, 1): @@ -448,17 +429,14 @@ def select_axis_edge_3d(cam, d, cx, cy, ccx, ccy): ex, ey = cx[m1] - cx[m0], cy[m1] - cy[m0] el = math.hypot(ex, ey) or 1.0 ex, ey = ex / el, ey / el - pxd, pyd = -ey, ex # screen-perpendicular to the edge + pxd, pyd = -ey, ex # screen-perpendicular to the edge mxv = 0.5 * (cx[m0] + cx[m1]) - ccx myv = 0.5 * (cy[m0] + cy[m1]) - ccy - if pxd * mxv + pyd * myv < 0.0: # flip to point away from the box centroid + if pxd * mxv + pyd * myv < 0.0: # flip to point away from the box centroid pxd, pyd = -pxd, -pyd return m0, m1, pxd, pyd -# --------------------------------------------------------------------------- # -# 3D preview # -# --------------------------------------------------------------------------- # def draw_3d(td, ext, region, has_region, cam, W, H, out_path, sim_name): fig, ax = plt.subplots(figsize=(W / 100.0, H / 100.0), dpi=100) @@ -466,21 +444,31 @@ def project_box(box, color, lw, label, ls="-"): corners = cube_corners(box) proj = [cam.project(c, W, H) for c in corners] first = True - for (a, b) in CUBE_EDGES: + for a, b in CUBE_EDGES: pa, pb = proj[a], proj[b] if pa is None or pb is None: continue # edge with a corner behind a perspective camera - ax.plot([pa[0], pb[0]], [pa[1], pb[1]], color=color, lw=lw, ls=ls, - label=(label if first else None), zorder=3) + ax.plot( + [pa[0], pb[0]], + [pa[1], pb[1]], + color=color, + lw=lw, + ls=ls, + label=(label if first else None), + zorder=3, + ) first = False # full extent (light gray) - project_box([ext[0], ext[1], ext[2]], color="0.6", lw=1.2, - label="full extent") + project_box([ext[0], ext[1], ext[2]], color="0.6", lw=1.2, label="full extent") # region crop, if distinct if has_region: - project_box([region[0], region[1], region[2]], color="tab:blue", - lw=2.0, label="region crop") + project_box( + [region[0], region[1], region[2]], + color="tab:blue", + lw=2.0, + label="region crop", + ) # axes tick labels: for each axis, pick the FOREGROUND (silhouette) edge of # the framed box and annotate along it, exactly as out::drawAxes3D does, so @@ -503,12 +491,12 @@ def project_box(box, color, lw, label, ls="-"): ccx = prc[0] if prc is not None else 0.0 ccy = prc[1] if prc is not None else 0.0 - tl = 8.0 # tick-mark length [px] - num_off = tl + 10.0 # numeric-label center offset from the edge [px] + tl = 8.0 # tick-mark length [px] + num_off = tl + 10.0 # numeric-label center offset from the edge [px] name_off = tl + 30.0 # axis-name center offset from the edge [px] for d in range(3): name = axis_names[d] if d < len(axis_names) else default_names[d] - m0, m1, pxd, pyd = select_axis_edge_3d(cam, d, cx, cy, ccx, ccy) + m0, _, pxd, pyd = select_axis_edge_3d(cam, d, cx, cy, ccx, ccy) # o = corner(m0): perpendicular coords fixed, axis d swept for ticks o = list(corners[m0]) lo_d, hi_d = frame_box[d][0], frame_box[d][1] @@ -519,19 +507,32 @@ def project_box(box, color, lw, label, ls="-"): if pr is None: continue a, b = pr - ax.plot([a, a + pxd * tl], [b, b + pyd * tl], - color="0.35", lw=1.0, zorder=4) - ax.annotate(f"{tv:g}", (a + pxd * num_off, b + pyd * num_off), - fontsize=6, color="0.25", ha="center", va="center") + ax.plot( + [a, a + pxd * tl], [b, b + pyd * tl], color="0.35", lw=1.0, zorder=4 + ) + ax.annotate( + f"{tv:g}", + (a + pxd * num_off, b + pyd * num_off), + fontsize=6, + color="0.25", + ha="center", + va="center", + ) # axis name at the MIDDLE of the chosen edge, pushed further outward mid = list(o) mid[d] = 0.5 * (lo_d + hi_d) pr = cam.project(mid, W, H) if pr is not None: a, b = pr - ax.annotate(name, (a + pxd * name_off, b + pyd * name_off), - fontsize=9, color="k", fontweight="bold", - ha="center", va="center") + ax.annotate( + name, + (a + pxd * name_off, b + pyd * name_off), + fontsize=9, + color="k", + fontweight="bold", + ha="center", + va="center", + ) # field-line SEED lattice (schematic scatter, NOT traced lines) fl = field_line_seeds_3d(td, frame_box, cam, H) @@ -544,9 +545,17 @@ def project_box(box, color, lw, label, ls="-"): pxs.append(pr[0]) pys.append(pr[1]) if pxs: - ax.scatter(pxs, pys, s=8, c="tab:red", marker="o", alpha=0.6, - edgecolors="none", zorder=2, - label=f"field-line seeds (schematic, {ns[0]}x{ns[1]}x{ns[2]})") + ax.scatter( + pxs, + pys, + s=8, + c="tab:red", + marker="o", + alpha=0.6, + edgecolors="none", + zorder=2, + label=f"field-line seeds (schematic, {ns[0]}x{ns[1]}x{ns[2]})", + ) ax.set_xlim(0, W) ax.set_ylim(H, 0) # inverted y: origin upper-left, matches the PNG @@ -561,9 +570,6 @@ def project_box(box, color, lw, label, ls="-"): plt.close(fig) -# --------------------------------------------------------------------------- # -# 2D window derivation (metadomain_render.cpp 2D branch) # -# --------------------------------------------------------------------------- # def derive_2d_window(td, ext, region, cartesian, mirror, W, H): """Return (umin,umax,vmin,vmax) -- the aspect-expanded world window mapped onto the WxH image, exactly as metadomain_render.cpp derives it.""" @@ -582,17 +588,22 @@ def accXZ(r, th): nonlocal umin, umax, vmin, vmax X = r * math.sin(th) Z = r * math.cos(th) - umin = min(umin, X); umax = max(umax, X) - vmin = min(vmin, Z); vmax = max(vmax, Z) + umin = min(umin, X) + umax = max(umax, X) + vmin = min(vmin, Z) + vmax = max(vmax, Z) if mirror: - umin = min(umin, -X); umax = max(umax, -X) + umin = min(umin, -X) + umax = max(umax, -X) for k in range(NB): t = k / (NB - 1) th = x2lo + (x2hi - x2lo) * t rr = x1lo + (x1hi - x1lo) * t - accXZ(x1lo, th); accXZ(x1hi, th) - accXZ(rr, x2lo); accXZ(rr, x2hi) + accXZ(x1lo, th) + accXZ(x1hi, th) + accXZ(rr, x2lo) + accXZ(rr, x2hi) # expand the window to the image aspect (centered) so geometry isn't stretched waspect = (umax - umin) / (vmax - vmin) @@ -610,8 +621,10 @@ def accXZ(r, th): # are not clipped (Cartesian fills the frame and needs none). if not cartesian: pad = 1.12 - cu = 0.5 * (umin + umax); hu = 0.5 * (umax - umin) * pad - cv = 0.5 * (vmin + vmax); hv = 0.5 * (vmax - vmin) * pad + cu = 0.5 * (umin + umax) + hu = 0.5 * (umax - umin) * pad + cv = 0.5 * (vmin + vmax) + hv = 0.5 * (vmax - vmin) * pad umin, umax = cu - hu, cu + hu vmin, vmax = cv - hv, cv + hv @@ -623,20 +636,43 @@ def draw_2d_cartesian(td, ext, region, has_region, W, H, out_path, sim_name): fig, ax = plt.subplots(figsize=(W / 100.0, H / 100.0), dpi=100) # aspect-expanded slice window (the background-padded frame) - ax.add_patch(Rectangle((umin, vmin), umax - umin, vmax - vmin, - fill=False, ec="0.7", lw=1.0, ls="--", - label="slice window (aspect-expanded)")) + ax.add_patch( + Rectangle( + (umin, vmin), + umax - umin, + vmax - vmin, + fill=False, + ec="0.7", + lw=1.0, + ls="--", + label="slice window (aspect-expanded)", + ) + ) # full domain box - ax.add_patch(Rectangle((ext[0][0], ext[1][0]), - ext[0][1] - ext[0][0], ext[1][1] - ext[1][0], - fill=False, ec="0.4", lw=1.5, label="domain")) + ax.add_patch( + Rectangle( + (ext[0][0], ext[1][0]), + ext[0][1] - ext[0][0], + ext[1][1] - ext[1][0], + fill=False, + ec="0.4", + lw=1.5, + label="domain", + ) + ) # region crop if has_region: - ax.add_patch(Rectangle((region[0][0], region[1][0]), - region[0][1] - region[0][0], - region[1][1] - region[1][0], - fill=False, ec="tab:blue", lw=2.0, - label="region crop")) + ax.add_patch( + Rectangle( + (region[0][0], region[1][0]), + region[0][1] - region[0][0], + region[1][1] - region[1][0], + fill=False, + ec="tab:blue", + lw=2.0, + label="region crop", + ) + ) # ticks (nice numbers over the data box == region) axes_on = find_or(td, False, "output", "render", "axes") @@ -671,25 +707,32 @@ def wedge_boundary(rmn, rmx, tmn, tmx, sign, color, lw, label=None): # outer + inner arcs and two rays, in meridional (X=r sin th, Z=r cos th) th = np.linspace(tmn, tmx, 200) # outer arc - ax.plot(sign * rmx * np.sin(th), rmx * np.cos(th), color=color, lw=lw, - label=label) + ax.plot( + sign * rmx * np.sin(th), rmx * np.cos(th), color=color, lw=lw, label=label + ) # inner arc ax.plot(sign * rmn * np.sin(th), rmn * np.cos(th), color=color, lw=lw) # rays at tmin, tmax for tt in (tmn, tmx): - ax.plot([sign * rmn * math.sin(tt), sign * rmx * math.sin(tt)], - [rmn * math.cos(tt), rmx * math.cos(tt)], color=color, lw=lw) + ax.plot( + [sign * rmn * math.sin(tt), sign * rmx * math.sin(tt)], + [rmn * math.cos(tt), rmx * math.cos(tt)], + color=color, + lw=lw, + ) # full extent wedge (light gray) - wedge_boundary(ext[0][0], ext[0][1], ext[1][0], ext[1][1], 1.0, "0.6", 1.2, - label="full extent") + wedge_boundary( + ext[0][0], ext[0][1], ext[1][0], ext[1][1], 1.0, "0.6", 1.2, label="full extent" + ) if mirror: wedge_boundary(ext[0][0], ext[0][1], ext[1][0], ext[1][1], -1.0, "0.6", 1.2) # region wedge (colored) if cropped if has_region: - wedge_boundary(rmin, rmax, tmin, tmax, 1.0, "tab:blue", 2.0, - label="region crop") + wedge_boundary( + rmin, rmax, tmin, tmax, 1.0, "tab:blue", 2.0, label="region crop" + ) if mirror: wedge_boundary(rmin, rmax, tmin, tmax, -1.0, "tab:blue", 2.0) @@ -699,9 +742,16 @@ def wedge_boundary(rmn, rmx, tmn, tmx, sign, color, lw, label=None): if axes_on: for Rv in nice_ticks(0.0, ext[0][1], nticks): ax.plot(0.0, Rv, marker="+", color="0.3", ms=6) - ax.annotate(f"{Rv:g}", (0.0, Rv), fontsize=6, color="0.25", - xytext=(-8, 0), textcoords="offset points", ha="right", - va="center") + ax.annotate( + f"{Rv:g}", + (0.0, Rv), + fontsize=6, + color="0.25", + xytext=(-8, 0), + textcoords="offset points", + ha="right", + va="center", + ) ax.set_xlim(umin, umax) ax.set_ylim(vmin, vmax) @@ -716,22 +766,7 @@ def wedge_boundary(rmn, rmx, tmn, tmx, sign, color, lw, label=None): plt.close(fig) -# --------------------------------------------------------------------------- # -# main # -# --------------------------------------------------------------------------- # -def main(): - ap = argparse.ArgumentParser( - description="Data-free preview of the entity in-situ renderer scene geometry.") - ap.add_argument("toml", help="simulation .toml file") - ap.add_argument("--out", default=None, - help="output PNG (default: /_preview.png)") - ap.add_argument("--time", type=float, default=0.0, - help="sim time T for the moving-view pan (default 0)") - ap.add_argument("--scene", type=int, default=None, - help="scene index (accepted for parity; geometry is scene-" - "independent, so it only affects the reported label)") - args = ap.parse_args() - +def preview(args): if not os.path.isfile(args.toml): print(f"error: no such file: {args.toml}", file=sys.stderr) return 2 @@ -782,8 +817,10 @@ def fmt_pairs(pairs): mode = "1D (nothing to render)" eye_str = f"({cam.eye[0]:g},{cam.eye[1]:g},{cam.eye[2]:g})" - print(f"mode={mode} | metric={metric_name} | eye={eye_str} | " - f"ortho_height={cam.ortho_height:g} | region={fmt_pairs(region)}") + print( + f"mode={mode} | metric={metric_name} | eye={eye_str} | " + f"ortho_height={cam.ortho_height:g} | region={fmt_pairs(region)}" + ) if args.scene is not None: print(f" (scene index {args.scene} requested; geometry is scene-independent)") @@ -791,15 +828,19 @@ def fmt_pairs(pairs): if dim == 3 and cartesian: draw_3d(td, ext, region, has_region, cam, width, height, out_path, sim_name) elif dim == 3: - print("warning: 3D non-Cartesian is not a renderer mode (3D is Cartesian-" - "only); nothing drawn.") + print( + "warning: 3D non-Cartesian is not a renderer mode (3D is Cartesian-" + "only); nothing drawn." + ) return 1 elif dim == 2 and cartesian: - draw_2d_cartesian(td, ext, region, has_region, width, height, out_path, - sim_name) + draw_2d_cartesian( + td, ext, region, has_region, width, height, out_path, sim_name + ) elif dim == 2: - draw_2d_spherical(td, ext, region, has_region, mirror, width, height, - out_path, sim_name) + draw_2d_spherical( + td, ext, region, has_region, mirror, width, height, out_path, sim_name + ) else: print("warning: 1D run -- the renderer is inactive; nothing to preview.") return 1 @@ -808,5 +849,179 @@ def fmt_pairs(pairs): return 0 +# ffmpeg -nostdin -framerate $framerate $inputspec -c:v libx264 -crf $compression -filter_complex \"[0:v]format=yuv420p,pad=ceil(iw/2)*2:ceil(ih/2)*2\" $output" + + +def merge(args): + try: + subprocess.run(["ffmpeg", "-version"], check=True, stdout=subprocess.DEVNULL) + except (subprocess.CalledProcessError, FileNotFoundError): + print( + "error: ffmpeg not found or not executable; cannot merge PNGs into a movie" + ) + return 1 + png_dir = os.path.abspath(args.path) + png_files = {f.split("_")[0] for f in os.listdir(png_dir) if f.endswith(".png")} + print(png_files, png_dir) + + ffmpeg_prekwargs = [ + "ffmpeg", + "-nostdin", + "-framerate", + str(args.framerate), + ] + ffmpeg_postkwargs = [ + "-c:v", + "libx264", + "-crf", + str(args.compression), + "-filter_complex", + "[0:v]format=yuv420p,pad=ceil(iw/2)*2:ceil(ih/2)*2", + ] + + if args.prefix: + if args.prefix not in png_files: + print(f"error: prefix '{args.prefix}' not found in {png_dir}") + return 1 + prefixes = [args.prefix] + else: + prefixes = sorted(png_files) + if args.merge and len(prefixes) > 1: + # arrange prefixes on a grid with args.cols columns (empty cells are black) + n = len(prefixes) + cols = max(1, min(args.cols, n)) + rows = (n + cols - 1) // cols + inputs = [] + for prefix in prefixes: + # -framerate/-pattern_type are per-input options: repeat before every -i + inputs += [ + "-framerate", + str(args.framerate), + "-pattern_type", + "glob", + "-i", + os.path.join(png_dir, f"{prefix}_*.png"), + ] + layout = [] + for i in range(n): + r, c = divmod(i, cols) + x = "+".join(f"w{j}" for j in range(c)) or "0" + y = "+".join(f"h{k * cols}" for k in range(r)) or "0" + layout.append(f"{x}_{y}") + filter_complex = ( + "".join(f"[{i}:v]" for i in range(n)) + + f"xstack=inputs={n}:layout={'|'.join(layout)}:fill=black:shortest=1," + + "format=yuv420p,pad=ceil(iw/2)*2:ceil(ih/2)*2" + ) + print(f"merging {n} scenes into a {rows}x{cols} grid") + kwargs = ( + ["ffmpeg", "-nostdin"] + + inputs + + ["-c:v", "libx264", "-crf", str(args.compression)] + + ["-filter_complex", filter_complex] + + ["merged_render.mp4"] + ) + subprocess.run(kwargs, check=True) + + else: + extra_kwargs = ["-pattern_type", "glob"] + for prefix in prefixes: + kwargs = ( + ffmpeg_prekwargs + + extra_kwargs + + ["-i", os.path.join(png_dir, f"{prefix}_*.png")] + + ffmpeg_postkwargs + + [prefix + "_render.mp4"] + ) + subprocess.run(kwargs, check=True) + return 0 + + +def main(): + ap = argparse.ArgumentParser( + description="Helper tools for the Entity on-the-fly renderer" + ) + sp = ap.add_subparsers(help="commands", required=True) + preview_sp = sp.add_parser( + "preview", + help="draw a preview of the simulation domain", + ) + movie_sp = sp.add_parser( + "movie", + help="merge the rendered .png into a movie", + ) + + preview_sp.add_argument("toml", help="simulation .toml file") + preview_sp.add_argument( + "-o", + "--out", + default=None, + help="output PNG (default: /_preview.png)", + ) + preview_sp.add_argument( + "-t", + "--time", + type=float, + default=0.0, + help="sim time T for the moving-view pan (default 0)", + ) + preview_sp.add_argument( + "-s", + "--scene", + type=int, + default=None, + help="scene index (accepted for parity; geometry is scene-" + "independent, so it only affects the reported label)", + ) + + movie_sp.add_argument( + "path", + help="path to the rendered PNGs", + ) + movie_sp.add_argument( + "-p", + "--prefix", + type=str, + help="prefix for the scene (when rendering only one scene without -m | --merge flag)", + ) + movie_sp.add_argument( + "-c", + "--cols", + type=int, + default=1, + help="number of columns for combining multiple scenes (default: 1)", + ) + movie_sp.add_argument( + "-m", + "--merge", + action="store_true", + help="merge multiple scenes into a single movie (default: false)", + ) + movie_sp.add_argument( + "-r", + "--framerate", + type=int, + default=30, + help="framerate for the output movie (default: 30)", + ) + movie_sp.add_argument( + "-z", + "--compression", + type=int, + default=30, + help="compression level (default: 1)", + ) + + preview_sp.set_defaults(func=preview) + movie_sp.set_defaults(func=merge) + + args = ap.parse_args() + args.func(args) + + if len(sys.argv) == 1: + ap.print_help() + return 1 + + if __name__ == "__main__": sys.exit(main()) From b78af715825b9d45bec653e8e24340dba40ccfc8 Mon Sep 17 00:00:00 2001 From: haykh Date: Thu, 24 Sep 2026 19:22:32 -0400 Subject: [PATCH 113/125] team_policy -> tiled_deposit --- entity.schema.json | 17 ++++++++-- input.default.toml | 13 +++++--- scripts/ideal_tile_size.py | 28 ++++++++--------- src/engines/engine.hpp | 14 +++++---- src/engines/grpic/currents.h | 18 +++++------ src/engines/reporter.cpp | 12 +++---- src/engines/srpic/currents.h | 20 ++++++------ src/framework/containers/particles.h | 10 +++--- src/framework/containers/particles_sort.cpp | 35 +++++++++++---------- src/framework/parameters/algorithms.h | 2 +- src/global/arch/kokkos_aliases.h | 2 +- src/global/defaults.h | 2 +- src/global/utils/reporter.cpp | 6 ++-- src/global/utils/sort_dispatch.h | 8 ++--- src/kernels/deposition/currents/tiled.hpp | 20 ++++++------ tests/framework/CMakeLists.txt | 6 ++-- tests/framework/particles_sort.cpp | 26 +++++++-------- tests/framework/sort_by_key.cpp | 4 +-- tests/kernels/CMakeLists.txt | 2 +- tests/kernels/deposit_tiled.cpp | 2 +- 20 files changed, 133 insertions(+), 114 deletions(-) diff --git a/entity.schema.json b/entity.schema.json index da44f30f9..603c13559 100644 --- a/entity.schema.json +++ b/entity.schema.json @@ -119,7 +119,7 @@ } }, "load_balance": { - "description": "Diffusion-style dynamic load balancing (Cartesian metrics only). Domain boundaries between MPI neighbors are nudged to equalize the active-particle count per rank. All inter-rank traffic uses only the existing nearest-neighbor field/particle communication paths.", + "description": "Diffusion-style dynamic load balancing. Domain boundaries between MPI neighbors are nudged to equalize the active-particle count per rank. All inter-rank traffic uses only the existing nearest-neighbor field/particle communication paths.", "type": "object", "additionalProperties": false, "properties": { @@ -921,14 +921,25 @@ } }, "team_policy_team_size": { - "description": "team_policy tiled-deposit work-group (team) size", + "description": "Tiled-deposit work-group (team) size", + "type": "integer", + "minimum": 0, + "default": 0, + "deprecated": true, + "x-entity": { + "type": "uint [>= 0]", + "deprecated": "removed in 1.6+, use `tiled_deposit_team_size` instead" + } + }, + "tiled_deposit_team_size": { + "description": "Tiled-deposit work-group (team) size", "type": "integer", "minimum": 0, "default": 0, "x-entity": { "type": "uint [>= 0]", "notes": [ - "0 keeps Kokkos::AUTO (backend occupancy heuristic); a positive value overrides it, clamped to the backend/scratch maximum at launch. Only used in `team_policy=ON` builds. Pick a multiple of the device subgroup width for best occupancy (see ideal_tile_size.py)" + "0 keeps Kokkos::AUTO (backend occupancy heuristic); a positive value overrides it, clamped to the backend/scratch maximum at launch. Only used in `tiled_deposit=ON` builds. Pick a multiple of the device subgroup width for best occupancy (see ideal_tile_size.py)" ] } } diff --git a/input.default.toml b/input.default.toml index e6753ce69..3e9724dbc 100644 --- a/input.default.toml +++ b/input.default.toml @@ -386,15 +386,20 @@ # @type: bool # @default: true enable = true - # team_policy tiled-deposit work-group (team) size + # Tiled-deposit work-group (team) size + # @type: uint [>= 0] + # @default: 0 + # @deprecated: removed in 1.6+, use `tiled_deposit_team_size` instead + team_policy_team_size = 0 + # Tiled-deposit work-group (team) size # @type: uint [>= 0] # @default: 0 # @note: 0 keeps Kokkos::AUTO (backend occupancy heuristic); a positive # value overrides it, clamped to the backend/scratch maximum at - # launch. Only used in `team_policy=ON` builds. Pick a multiple of - # the device subgroup width for best occupancy (see + # launch. Only used in `tiled_deposit=ON` builds. Pick a multiple + # of the device subgroup width for best occupancy (see # ideal_tile_size.py) - team_policy_team_size = 0 + tiled_deposit_team_size = 0 # @inferred: # - order diff --git a/scripts/ideal_tile_size.py b/scripts/ideal_tile_size.py index d9e9d32ea..784065161 100755 --- a/scripts/ideal_tile_size.py +++ b/scripts/ideal_tile_size.py @@ -8,7 +8,7 @@ TE = T_TILE + 2*HALO HALO = stencil_reach + drift (stencil_reach = shape_order for Esirkepov, 2 for the O==0 zigzag deposit; drift = the compile-time - `team_policy_drift` CMake knob, NOT the runtime + `tiled_deposit_drift` CMake knob, NOT the runtime spatial_sorting_interval -- see kernels/deposition/currents/tiled.hpp) the tile size is squeezed by three competing pressures: @@ -25,9 +25,9 @@ Recommendation = the largest tile that respects the particle budget and shared-memory residency; if that tile would be mostly halo, it is grown (toward lower halo) up to the shared-memory limit. This is a first-order model -- confirm by sweeping the entity knobs - -D team_policy_tile_size= -D team_policy_drift= and re-profiling (see roofline/). + -D tiled_deposit_tile_size= -D tiled_deposit_drift= and re-profiling (see roofline/). The team (work-group) size defaults to Kokkos::AUTO; override it at runtime with the - [algorithms.deposit] team_policy_team_size = (0 = AUTO) + [algorithms.deposit] tiled_deposit_team_size = (0 = AUTO) toml knob -- clamped to the backend maximum at launch (engines/srpic/currents.h). Two ways to drive it: @@ -99,14 +99,14 @@ def __init__(self): self.shape_order = 2 # entity shape_order self.precision = "single" # single / double self.components = 3 # current-field components (J has 3) - self.drift = 1 # team_policy_drift: cells of drift the scratch halo absorbs + self.drift = 1 # tiled_deposit_drift: cells of drift the scratch halo absorbs # (compile-time CMake knob, independent of spatial_sorting_interval) self.target_resident = 2 # work-groups resident per compute unit self.npart_cap = 1600.0 # particle-per-tile budget (contention / load-balance proxy) self.halo_max = 0.70 # halo fraction above which the tile is grown self.grid = 0 # cells per dim (0 disables the GPU-fill check) self.balance_factor = 4 # min tiles per compute unit - self.min_tile = 4 # entity's team_policy_tile_sizes list starts at 4 + self.min_tile = 4 # entity's tiled_deposit_tile_sizes list starts at 4 self.max_tile = 64 @@ -121,7 +121,7 @@ def resolve_arch(name): def recommend(hw, p): """p: Settings or argparse namespace. Returns dict with rows, chosen row, binding.""" # Matches DepositCurrentsTiled_kernel: STENCIL_REACH = O for Esirkepov (O>=1), - # 2 for the O==0 zigzag deposit; HALO = STENCIL_REACH + TEAM_POLICY_DRIFT. + # 2 for the O==0 zigzag deposit; HALO = STENCIL_REACH + TILED_DEPOSIT_DRIFT. stencil_reach = 2 if p.shape_order == 0 else p.shape_order halo = stencil_reach + p.drift real = PRECISION[p.precision] @@ -198,7 +198,7 @@ def report_lines(name, key, hw, p, res): reach = 2 if p.shape_order == 0 else p.shape_order reach_kind = "zigzag" if p.shape_order == 0 else "Esirkepov O" L.append(" HALO = stencil_reach + drift = %d + %d = %d -> TE = T_TILE + %d" - " (reach %d = %s; drift = team_policy_drift)" + " (reach %d = %s; drift = tiled_deposit_drift)" % (reach, p.drift, res["halo"], 2 * res["halo"], reach, reach_kind)) L.append(" shared mem %s KiB/%s (budget %s KiB for %d resident WGs); subgroup=%d, n_cu=%d" % (kib(hw["smem_cu"]), hw["cu"], kib(hw["smem_cu"] / p.target_resident), @@ -228,17 +228,17 @@ def report_lines(name, key, hw, p, res): L.append(" %.1f KiB scratch/team, %d work-groups resident/%s, %.0f particles/team, %.0f%% halo" % (c["scratch"] / 1024.0, c["resident"], hw["cu"], c["npart"], 100 * c["halo_frac"])) team = min(hw["max_wg"], 256 - 256 % hw["subgroup"]) - extra = "" if c["T"] <= 16 else " (entity's team_policy_tile_sizes list stops at 16; extend it)" - L.append(" entity build: -D team_policy=ON -D team_policy_tile_size=%d -D team_policy_drift=%d%s" + extra = "" if c["T"] <= 16 else " (entity's tiled_deposit_tile_sizes list stops at 16; extend it)" + L.append(" entity build: -D tiled_deposit=ON -D tiled_deposit_tile_size=%d -D tiled_deposit_drift=%d%s" % (min(c["T"], 16), p.drift, extra)) L.append(" team (work-group) size: Kokkos::AUTO by default; to override, set in the toml") - L.append(" [algorithms.deposit] team_policy_team_size = %d (0 = AUTO; keep a multiple of" + L.append(" [algorithms.deposit] tiled_deposit_team_size = %d (0 = AUTO; keep a multiple of" % team) L.append(" subgroup=%d), then sweep around it and re-profile" % hw["subgroup"]) # contextual guidance if c["halo_frac"] > p.halo_max: if p.drift > 1: - L.append(" !! %.0f%% of the tile is halo, inflated by team_policy_drift=%d; lower it " + L.append(" !! %.0f%% of the tile is halo, inflated by tiled_deposit_drift=%d; lower it " "(and sort at least that often via spatial_sorting_interval)" % (100 * c["halo_frac"], p.drift)) else: @@ -775,7 +775,7 @@ def cyc_prec(): ), MenuItem( "drift", - "team_policy_drift CMake knob: cells the scratch halo absorbs (>= spatial_sorting_interval)", + "tiled_deposit_drift CMake knob: cells the scratch halo absorbs (>= spatial_sorting_interval)", right=lambda: str(self.s.drift), on_enter=lambda: self.edit_int("drift", "drift", minv=0), ), @@ -926,7 +926,7 @@ def run_cli(argv) -> int: ap.add_argument("--precision", choices=("single", "double"), default="single") ap.add_argument("--components", type=int, default=3, help="current-field components (J has 3)") ap.add_argument("--drift", type=int, default=1, - help="team_policy_drift CMake knob (compile-time): cells of drift the scratch " + help="tiled_deposit_drift CMake knob (compile-time): cells of drift the scratch " "halo absorbs; size it >= spatial_sorting_interval") ap.add_argument("--target-resident", type=int, default=2, help="work-groups resident per compute unit") ap.add_argument("--npart-cap", type=float, default=1600, @@ -935,7 +935,7 @@ def run_cli(argv) -> int: ap.add_argument("--grid", type=int, default=0, help="cells per dim (optional; enables a GPU-fill check)") ap.add_argument("--balance-factor", type=int, default=4, help="min tiles per compute unit") ap.add_argument("--min-tile", type=int, default=4, - help="smallest T_TILE to consider (entity's team_policy_tile_sizes starts at 4)") + help="smallest T_TILE to consider (entity's tiled_deposit_tile_sizes starts at 4)") ap.add_argument("--max-tile", type=int, default=64) p = ap.parse_args(argv) diff --git a/src/engines/engine.hpp b/src/engines/engine.hpp index f2f70fb10..13b5e8e28 100644 --- a/src/engines/engine.hpp +++ b/src/engines/engine.hpp @@ -81,7 +81,7 @@ namespace ntt { const bool is_resuming; const simtime_t runtime; const real_t dt; - const std::size_t team_policy_team_size; + const std::size_t tiled_deposit_team_size; const timestep_t max_steps; const timestep_t start_step; const simtime_t start_time; @@ -110,8 +110,8 @@ namespace ntt { , is_resuming { m_params.get("checkpoint.is_resuming") } , runtime { m_params.get("simulation.runtime") } , dt { m_params.get("algorithms.timestep.dt") } - , team_policy_team_size { m_params.get( - "algorithms.deposit.team_policy_team_size") } + , tiled_deposit_team_size { m_params.get( + "algorithms.deposit.tiled_deposit_team_size") } , max_steps { static_cast(runtime / dt) } , start_step { m_params.get("checkpoint.start_step") } , start_time { m_params.get("checkpoint.start_time") } @@ -130,8 +130,8 @@ namespace ntt { auto parameters = prm::Parameters {}; parameters.set("dt", static_cast(dt)); parameters.set("time", static_cast(time)); - parameters.set("team_policy_team_size", - static_cast(team_policy_team_size)); + parameters.set("tiled_deposit_team_size", + static_cast(tiled_deposit_team_size)); return parameters; } }; @@ -142,8 +142,8 @@ namespace ntt { #if defined(OUTPUT_ENABLED) m_metadomain.InitWriter(&m_adios, m_params); m_metadomain.InitCheckpointWriter(&m_adios, m_params); - m_metadomain.InitRenderer(m_params); #endif + m_metadomain.InitRenderer(m_params); logger::Checkpoint("Initializing Engine", HERE); if (not is_resuming) { // start a new simulation with initial conditions @@ -369,11 +369,13 @@ namespace ntt { time - dt); } timers.stop("Output"); +#endif timers.start("Render"); print_render = m_metadomain.Render(m_params, step, step - 1, time, time - dt); timers.stop("Render"); +#if defined(OUTPUT_ENABLED) timers.start("Checkpoint"); print_checkpoint = m_metadomain.WriteCheckpoint(m_params, step, diff --git a/src/engines/grpic/currents.h b/src/engines/grpic/currents.h index 15d4d94ca..9194dea7a 100644 --- a/src/engines/grpic/currents.h +++ b/src/engines/grpic/currents.h @@ -3,7 +3,7 @@ * @brief Current deposition and filtering routines for the GRPIC engine * @implements * - ntt::grpic::CallDepositKernel<> -> void (flat path) - * - ntt::grpic::CallDepositKernelTiled<> -> void (TEAM_POLICY) + * - ntt::grpic::CallDepositKernelTiled<> -> void (TILED_DEPOSIT) * - ntt::grpic::CurrentsDeposit<> -> void * - ntt::grpic::CurrentsFilter<> -> void * @namespaces: @@ -47,7 +47,7 @@ namespace ntt { dt)); } -#if defined(TEAM_POLICY) +#if defined(TILED_DEPOSIT) /** * @brief Tiled deposit launcher (TeamPolicy + per-team scratch). * @@ -56,7 +56,7 @@ namespace ntt { * teams; each team accumulates its tile's particle contributions in SLM * scratch and atomically flushes to the global J (here `cur0`, the GRPIC * half-step current). Requires the species to have been sorted with - * `team_policy` enabled (`tile_layout` populated by `SortSpatially`). + * `tiled_deposit` enabled (`tile_layout` populated by `SortSpatially`). * * The deposit body (`kernel::DepositOneParticle`) * is the same shared math used by the flat path — it already carries the GR @@ -75,7 +75,7 @@ namespace ntt { int team_size_req) { static_assert(O <= 11u, "Shape order must be <= 11"); constexpr unsigned short T = static_cast( - TEAM_POLICY_TILE_SIZE); + TILED_DEPOSIT_TILE_SIZE); const auto& layout = species.tile_layout(); raise::ErrorIf(layout.ntiles_total == 0u, "CallDepositKernelTiled: tile_layout has 0 tiles — call " @@ -97,7 +97,7 @@ namespace ntt { // Team (work-group) size. The default (team_size_req == 0) leaves // Kokkos::AUTO, which sizes the team from the backend occupancy - // heuristic. A positive `algorithms.deposit.team_policy_team_size` + // heuristic. A positive `algorithms.deposit.tiled_deposit_team_size` // overrides it, clamped to the scratch/backend-feasible maximum so an // over-large request cannot abort the launch (Kokkos errors when // team_size > team_size_max). No portable subgroup rounding is applied; @@ -113,7 +113,7 @@ namespace ntt { if (ts > ts_max) { raise::Warning( fmt::format( - "algorithms.deposit.team_policy_team_size = %d exceeds " + "algorithms.deposit.tiled_deposit_team_size = %d exceeds " "the tiled-deposit maximum %d on this backend; clamping " "to %d", team_size_req, @@ -153,7 +153,7 @@ namespace ntt { Kokkos::Experimental::contribute(cur_nc, scatter_cur); } } -#endif // TEAM_POLICY +#endif // TILED_DEPOSIT template void CurrentsDeposit(Domain& domain, @@ -163,12 +163,12 @@ namespace ntt { // pre-zeros it — this is the single source of truth, matching SRPIC). Kokkos::deep_copy(domain.fields.cur0, ZERO); -#if defined(TEAM_POLICY) +#if defined(TILED_DEPOSIT) // Optional runtime override for the tiled-deposit team (work-group) size; // 0 (default) keeps Kokkos::AUTO. Clamped to the backend max in the // launcher (see CallDepositKernelTiled). const auto team_size_req = static_cast( - engine_params.get("team_policy_team_size", + engine_params.get("tiled_deposit_team_size", std::optional { 0u })); // Tiled deposit. Correctness no longer depends on the SoA being in a diff --git a/src/engines/reporter.cpp b/src/engines/reporter.cpp index 2f19d500c..7752840f1 100644 --- a/src/engines/reporter.cpp +++ b/src/engines/reporter.cpp @@ -32,13 +32,13 @@ namespace ntt { "%s", params.template get("simulation.name").c_str()); reporter::AddParam(report, 4, "Engine", "%s", SimEngine(S).to_string()); -#if defined(TEAM_POLICY) - reporter::AddParam(report, 4, "Tile size", "%d", TEAM_POLICY_TILE_SIZE); - #if defined(TEAM_POLICY_DRIFT) - reporter::AddParam(report, 4, "Halo drift", "%d", TEAM_POLICY_DRIFT); +#if defined(TILED_DEPOSIT) + reporter::AddParam(report, 4, "Tile size", "%d", TILED_DEPOSIT_TILE_SIZE); + #if defined(TILED_DEPOSIT_DRIFT) + reporter::AddParam(report, 4, "Halo drift", "%d", TILED_DEPOSIT_DRIFT); #endif if (params.template get( - "algorithms.deposit.team_policy_team_size") == 0u) { + "algorithms.deposit.tiled_deposit_team_size") == 0u) { reporter::AddParam(report, 4, "Team size", "%s", "AUTO (Kokkos)"); } else { reporter::AddParam(report, @@ -46,7 +46,7 @@ namespace ntt { "Team size", "%d (requested; clamped to backend max at launch)", static_cast(params.template get( - "algorithms.deposit.team_policy_team_size"))); + "algorithms.deposit.tiled_deposit_team_size"))); } #endif reporter::AddParam(report, 4, "Metric", "%s", M.to_string()); diff --git a/src/engines/srpic/currents.h b/src/engines/srpic/currents.h index 2a23fde69..239a8a00d 100644 --- a/src/engines/srpic/currents.h +++ b/src/engines/srpic/currents.h @@ -3,7 +3,7 @@ * @brief Current deposition and filtering routines for the SRPIC engine * @implements * - ntt::srpic::CallDepositKernel<> -> void (flat path) - * - ntt::srpic::CallDepositKernelTiled<> -> void (TEAM_POLICY) + * - ntt::srpic::CallDepositKernelTiled<> -> void (TILED_DEPOSIT) * - ntt::srpic::CurrentsDeposit<> -> void * - ntt::srpic::CurrentsFilter<> -> void * @namespaces: @@ -48,14 +48,14 @@ namespace ntt { dt)); } -#if defined(TEAM_POLICY) +#if defined(TILED_DEPOSIT) /** * @brief Tiled deposit launcher (TeamPolicy + per-team scratch). * * Iterates over `tile_layout.ntiles_total` teams; each team accumulates * its tile's particle contributions in SLM scratch and atomically * flushes to the global J. Requires the species to have been sorted - * with `team_policy` enabled (`tile_layout` populated by + * with `tiled_deposit` enabled (`tile_layout` populated by * `SortSpatially`). * * Falls back to the flat kernel if `tile_offsets` is empty — this @@ -71,7 +71,7 @@ namespace ntt { int team_size_req) { static_assert(O <= 11u, "Shape order must be <= 11"); constexpr unsigned short T = static_cast( - TEAM_POLICY_TILE_SIZE); + TILED_DEPOSIT_TILE_SIZE); const auto& layout = species.tile_layout(); raise::ErrorIf(layout.ntiles_total == 0u, "CallDepositKernelTiled: tile_layout has 0 tiles — call " @@ -93,7 +93,7 @@ namespace ntt { // Team (work-group) size. The default (team_size_req == 0) leaves // Kokkos::AUTO, which sizes the team from the backend occupancy - // heuristic. A positive `algorithms.deposit.team_policy_team_size` + // heuristic. A positive `algorithms.deposit.tiled_deposit_team_size` // overrides it, clamped to the scratch/backend-feasible maximum so an // over-large request cannot abort the launch (Kokkos errors when // team_size > team_size_max). No portable subgroup rounding is applied; @@ -109,7 +109,7 @@ namespace ntt { if (ts > ts_max) { raise::Warning( fmt::format( - "algorithms.deposit.team_policy_team_size = %d exceeds " + "algorithms.deposit.tiled_deposit_team_size = %d exceeds " "the tiled-deposit maximum %d on this backend; clamping " "to %d", team_size_req, @@ -149,7 +149,7 @@ namespace ntt { Kokkos::Experimental::contribute(cur_nc, scatter_cur); } } -#endif // TEAM_POLICY +#endif // TILED_DEPOSIT template void CurrentsDeposit(Domain& domain, @@ -157,12 +157,12 @@ namespace ntt { const auto dt = engine_params.get("dt"); Kokkos::deep_copy(domain.fields.cur, ZERO); -#if defined(TEAM_POLICY) +#if defined(TILED_DEPOSIT) // Optional runtime override for the tiled-deposit team (work-group) size; // 0 (default) keeps Kokkos::AUTO. Clamped to the backend max in the // launcher (see CallDepositKernelTiled). const auto team_size_req = static_cast( - engine_params.get("team_policy_team_size", + engine_params.get("tiled_deposit_team_size", std::optional { 0u })); // Tiled deposit. Correctness no longer depends on the SoA being in a @@ -170,7 +170,7 @@ namespace ntt { // partition per-particle: // - a particle whose full stencil has drifted out of its tile is // deposited straight to the global J view (the per-particle escape - // valve); `team_policy_drift` sizes the scratch halo so the + // valve); `tiled_deposit_drift` sizes the scratch halo so the // common in-tile case stays in fast SLM (see kernels/deposition/currents/tiled.hpp); // - particles dead-tagged in place since the sort are clamped out by // the kernel and skipped by the dead-tag test; diff --git a/src/framework/containers/particles.h b/src/framework/containers/particles.h index 5f15b9621..bdca0858a 100644 --- a/src/framework/containers/particles.h +++ b/src/framework/containers/particles.h @@ -91,7 +91,7 @@ namespace ntt { const uint8_t m_ntags { (uint8_t)(2 + math::pow(3, (int)D) - 1) }; #endif - // team_policy: tile metadata produced by SortSpatially + // tiled_deposit: tile metadata produced by SortSpatially // and consumed by the tiled deposit / pusher kernels. Lazily // allocated on first sort. The sort backend itself (oneDPL on SYCL, // Thrust on CUDA, std::sort on Host, Kokkos::BinSort otherwise) is @@ -99,7 +99,7 @@ namespace ntt { // vendor libraries detected by CMake. TileLayout m_tile_layout {}; -#if defined(TEAM_POLICY) && \ +#if defined(TILED_DEPOSIT) && \ ((defined(SYCL_ENABLED) && defined(ONEDPL_ENABLED)) || \ (defined(CUDA_ENABLED) && defined(THRUST_ENABLED)) || \ (defined(HIP_ENABLED) && defined(ROCTHRUST_ENABLED))) @@ -234,7 +234,7 @@ namespace ntt { return m_ntags; } -#if defined(TEAM_POLICY) +#if defined(TILED_DEPOSIT) // Build m_tile_layout.tile_offsets / npart_partitioned from the // already-sorted tile-index keys. A separate member function (not a // lambda local to SortSpatially) so the inner device kernel is not an @@ -326,14 +326,14 @@ namespace ntt { /** * @brief Sort particles spatially by their cell indices * @param grid The grid object to get the cell information for sorting - * @note In team_policy mode (compile-time `team_policy=ON`), also + * @note In tiled_deposit mode (compile-time `tiled_deposit=ON`), also * populates `m_tile_layout` with tile-offset and per-tile * permutation metadata that the tiled deposit/pusher kernels * consume. */ void SortSpatially(const Grid&); -#if defined(TEAM_POLICY) && \ +#if defined(TILED_DEPOSIT) && \ ((defined(SYCL_ENABLED) && defined(ONEDPL_ENABLED)) || \ (defined(CUDA_ENABLED) && defined(THRUST_ENABLED)) || \ (defined(HIP_ENABLED) && defined(ROCTHRUST_ENABLED))) diff --git a/src/framework/containers/particles_sort.cpp b/src/framework/containers/particles_sort.cpp index 4e45b8db9..02a6584f4 100644 --- a/src/framework/containers/particles_sort.cpp +++ b/src/framework/containers/particles_sort.cpp @@ -8,11 +8,11 @@ #include "framework/containers/particles.h" #include "framework/domain/grid.h" -#if defined(TEAM_POLICY) +#if defined(TILED_DEPOSIT) #if (defined(SYCL_ENABLED) && defined(ONEDPL_ENABLED)) || \ (defined(CUDA_ENABLED) && defined(THRUST_ENABLED)) || \ (defined(HIP_ENABLED) && defined(ROCTHRUST_ENABLED)) - #define TEAM_POLICY_USE_VENDOR_SORT + #define TILED_DEPOSIT_USE_VENDOR_SORT #include "utils/sort_dispatch.h" #endif #endif @@ -231,7 +231,7 @@ namespace ntt { } } // namespace -#if defined(TEAM_POLICY) +#if defined(TILED_DEPOSIT) template void Particles::compute_tile_offsets(const array_t& tile_indices, ncells_t total_tiles, @@ -283,12 +283,12 @@ namespace ntt { // separately deposit) particles appended since this sort. m_tile_layout.npart_partitioned = h_offsets(total_tiles); } -#endif // TEAM_POLICY +#endif // TILED_DEPOSIT template void Particles::SortSpatially(const Grid& grid) { -#if defined(TEAM_POLICY) - // ---------------------- team_policy: tile-based sort ------------------ // +#if defined(TILED_DEPOSIT) + // ---------------------- tiled_deposit: tile-based sort ------------------ // const auto npart_local = npart(); if (npart_local == 0u) { m_tile_layout = TileLayout {}; @@ -296,8 +296,9 @@ namespace ntt { return; } - constexpr unsigned short T = static_cast(TEAM_POLICY_TILE_SIZE); - static_assert(T > 0u, "TEAM_POLICY_TILE_SIZE must be > 0"); + constexpr unsigned short T = static_cast( + TILED_DEPOSIT_TILE_SIZE); + static_assert(T > 0u, "TILED_DEPOSIT_TILE_SIZE must be > 0"); // 1. Compute per-axis tile counts and total_tiles. const auto ncells_active = grid.n_active(); @@ -320,7 +321,7 @@ namespace ntt { } // 2. Compute per-particle tile key (with min(i, i_prev)). - #if defined(TEAM_POLICY_USE_VENDOR_SORT) && defined(SYCL_ENABLED) && \ + #if defined(TILED_DEPOSIT_USE_VENDOR_SORT) && defined(SYCL_ENABLED) && \ defined(ONEDPL_ENABLED) // oneDPL sorts the keys in place, so reuse a persistent, grow-only keys // buffer instead of allocating a fresh one every sort. `tile_indices` @@ -353,7 +354,7 @@ namespace ntt { const ncells_t n_bins = total_tiles + 2u; const auto slice = prtl_slice_t(0, npart_local); - #if defined(TEAM_POLICY_USE_VENDOR_SORT) + #if defined(TILED_DEPOSIT_USE_VENDOR_SORT) // Vendor path: produce an explicit permutation via sort_by_key, then // apply it to each SoA member by gathering the alive prefix through a // reusable scratch buffer (one per member type, copied back in place). @@ -451,7 +452,7 @@ namespace ntt { // gather to hoist this ahead of). sorter.sort(tile_indices); compute_tile_offsets(tile_indices, total_tiles, npart_local); - #endif // TEAM_POLICY_USE_VENDOR_SORT + #endif // TILED_DEPOSIT_USE_VENDOR_SORT // Populate `m_tile_layout` size/shape. `tile_perm` is not used in the // current design — the SoA arrays are physically permuted into tile @@ -475,8 +476,8 @@ namespace ntt { // (RemoveDead remains the compactor when spatial sorting is disabled.) set_npart(m_tile_layout.npart_partitioned); - Kokkos::fence("SortSpatially: end of team_policy path"); -#else // !TEAM_POLICY — legacy in-place BinSort by global cell index + Kokkos::fence("SortSpatially: end of tiled_deposit path"); +#else // !TILED_DEPOSIT — legacy in-place BinSort by global cell index const auto nx2 = grid.n_active(in::x2); const auto nx3 = grid.n_active(in::x3); const auto total_cells = grid.num_active(); @@ -532,10 +533,10 @@ namespace ntt { for (auto pldi { 0u }; pldi < npld_i(); ++pldi) { sorter.sort(Kokkos::subview(pld_i, slice, pldi)); } -#endif // TEAM_POLICY +#endif // TILED_DEPOSIT } -#if defined(TEAM_POLICY_USE_VENDOR_SORT) +#if defined(TILED_DEPOSIT_USE_VENDOR_SORT) namespace permute_helpers { // Permute a 1D SoA member `arr` by `perm` in place, using a @@ -694,9 +695,9 @@ namespace ntt { permute_2d_into(pld_i, m_sort_scratch_pld_i, perm, n, ncols); } } -#endif // TEAM_POLICY_USE_VENDOR_SORT +#endif // TILED_DEPOSIT_USE_VENDOR_SORT -#if defined(TEAM_POLICY_USE_VENDOR_SORT) +#if defined(TILED_DEPOSIT_USE_VENDOR_SORT) #define APPLY_PERM_INSTANTIATE(D, C) \ template void Particles::apply_permutation_to_soa(const prtl_perm_t&, \ npart_t); diff --git a/src/framework/parameters/algorithms.h b/src/framework/parameters/algorithms.h index c46f480fe..f253448c5 100644 --- a/src/framework/parameters/algorithms.h +++ b/src/framework/parameters/algorithms.h @@ -34,7 +34,7 @@ namespace ntt { std::optional deposit_enable; std::optional deposit_order; - std::optional deposit_team_policy_team_size; + std::optional deposit_tiled_team_size; std::optional fieldsolver_enable; std::optional> fieldsolver_stencil_coeffs; diff --git a/src/global/arch/kokkos_aliases.h b/src/global/arch/kokkos_aliases.h index b0ed48773..5ca287032 100644 --- a/src/global/arch/kokkos_aliases.h +++ b/src/global/arch/kokkos_aliases.h @@ -364,7 +364,7 @@ template auto CreateRangePolicyOnHost(const tuple_t&, const tuple_t&) -> range_h_t; -// --------------------------- team_policy types ---------------------------- // +// ------------------------- tiled_deposit types --------------------------- // // Particle permutation index: maps a sorted-position p in [0, npart) to a // pre-sort particle index. Produced by SortSpatially, consumed by tiled // pusher and deposit kernels to walk particles tile-by-tile without diff --git a/src/global/defaults.h b/src/global/defaults.h index 82b4364cc..101e54659 100644 --- a/src/global/defaults.h +++ b/src/global/defaults.h @@ -22,7 +22,7 @@ namespace ntt::defaults { const unsigned short current_filters = 0; - const std::size_t team_policy_team_size = 0; + const std::size_t tiled_deposit_team_size = 0; const std::string em_pusher = "Boris"; const std::string ph_pusher = "Photon"; diff --git a/src/global/utils/reporter.cpp b/src/global/utils/reporter.cpp index d889ad560..21d33ee7d 100644 --- a/src/global/utils/reporter.cpp +++ b/src/global/utils/reporter.cpp @@ -251,8 +251,8 @@ namespace reporter { AddParam(report, 4, "GPU_AWARE_MPI", "%s", "OFF"); #endif -#if defined(TEAM_POLICY) - AddParam(report, 4, "TEAM_POLICY", "%s", "ON"); +#if defined(TILED_DEPOSIT) + AddParam(report, 4, "TILED_DEPOSIT", "%s", "ON"); #if (defined(SYCL_ENABLED) && defined(ONEDPL_ENABLED)) || \ (defined(CUDA_ENABLED) && defined(THRUST_ENABLED)) || \ (defined(HIP_ENABLED) && defined(ROCTHRUST_ENABLED)) @@ -261,7 +261,7 @@ namespace reporter { AddParam(report, 4, "VENDOR_SORT", "%s", "OFF (BinSort)"); #endif #else - AddParam(report, 4, "TEAM_POLICY", "%s", "OFF"); + AddParam(report, 4, "TILED_DEPOSIT", "%s", "OFF"); #endif report += "\n"; return report; diff --git a/src/global/utils/sort_dispatch.h b/src/global/utils/sort_dispatch.h index e445a45ad..70674d840 100644 --- a/src/global/utils/sort_dispatch.h +++ b/src/global/utils/sort_dispatch.h @@ -1,12 +1,12 @@ /** * @file utils/sort_dispatch.h - * @brief Backend-dispatched sort_by_key for team_policy SortSpatially. + * @brief Backend-dispatched sort_by_key for tiled_deposit SortSpatially. * @implements * - sort_helpers::sort_by_key_dispatch -> void (BinSort, OneDPL, Thrust, StdSort) * @namespaces: * - ntt::sort_helpers:: * @macros: - * - TEAM_POLICY + * - TILED_DEPOSIT * - SYCL_ENABLED, ONEDPL_ENABLED (oneDPL overload) * - CUDA_ENABLED, THRUST_ENABLED (Thrust overload) * @@ -26,8 +26,8 @@ #ifndef GLOBAL_UTILS_SORT_DISPATCH_H #define GLOBAL_UTILS_SORT_DISPATCH_H -#if !defined(TEAM_POLICY) - #error "sort_dispatch.h is only meaningful when TEAM_POLICY is defined" +#if !defined(TILED_DEPOSIT) + #error "sort_dispatch.h is only meaningful when TILED_DEPOSIT is defined" #endif #include "global.h" diff --git a/src/kernels/deposition/currents/tiled.hpp b/src/kernels/deposition/currents/tiled.hpp index 6e211a89f..1a1bd94ee 100644 --- a/src/kernels/deposition/currents/tiled.hpp +++ b/src/kernels/deposition/currents/tiled.hpp @@ -3,8 +3,8 @@ * @brief Tiled current deposition kernel with per-team SLM scratch. * * @note Team-policy (one team per spatial tile, accumulates into team SLM scratch with - * atomic adds, then flushes to global J). Available when `team_policy=ON` - * (`#if defined(TEAM_POLICY)`). Stream 2 of the Pattern A plan. + * atomic adds, then flushes to global J). Available when `tiled_deposit=ON` + * (`#if defined(TILED_DEPOSIT)`). Stream 2 of the Pattern A plan. * * @implements * - kernel::DepositCurrentsTiled_kernel<> @@ -51,8 +51,8 @@ namespace kernel { * regression there, but it's good to be able to measure the * crossover. To revert and use flat for zigzag-only builds, change * the dispatch in `engines/srpic/currents.h` from - * `#if defined(TEAM_POLICY)` to - * `#if defined(TEAM_POLICY) && (SHAPE_ORDER > 0)`. + * `#if defined(TILED_DEPOSIT)` to + * `#if defined(TILED_DEPOSIT) && (SHAPE_ORDER > 0)`. * * Particle iteration order is governed by `tile_offsets`: tile `t` * owns particles `[tile_offsets(t), tile_offsets(t+1))`, post-sort. @@ -65,8 +65,8 @@ namespace kernel { * per step elapsed since the last sort. The scratch HALO is * `STENCIL_REACH(O) + DRIFT`, where `STENCIL_REACH = 2` for zigzag * (writes `{i_prev, i_prev+1, i, i+1}` ⇒ +2 above `min(i, i_prev)` with - * `|Δi|=1`) and `O` for Esirkepov. `DRIFT` is the `team_policy_drift` - * CMake knob (macro TEAM_POLICY_DRIFT) — the number of cells a particle + * `|Δi|=1`) and `O` for Esirkepov. `DRIFT` is the `tiled_deposit_drift` + * CMake knob (macro TILED_DEPOSIT_DRIFT) — the number of cells a particle * may drift between two sorts that the halo is sized to absorb — and `1` * by default (the every-step-sorted common case). It is independent of * the sort cadence, which is set at runtime via `spatial_sorting_interval`; @@ -118,8 +118,8 @@ namespace kernel { * pushed once per step between its last sort and a given deposit. With * a runtime sort interval of `K` (spatial_sorting_interval), a particle * drifts at most `K` cells (CFL |v dt/dx| <= 1/2 ⇒ |Δi| <= 1 per step) - * before the next sort. The `team_policy_drift` CMake knob (macro - * TEAM_POLICY_DRIFT) sets DRIFT independently of `K`, sizing the halo so + * before the next sort. The `tiled_deposit_drift` CMake knob (macro + * TILED_DEPOSIT_DRIFT) sets DRIFT independently of `K`, sizing the halo so * a particle that drifts up to DRIFT cells still deposits inside its * tile scratch. DRIFT defaults to 1 (the sorted-every-step common case); * any particle that drifts past the halo (e.g. a larger sort interval, @@ -134,8 +134,8 @@ namespace kernel { // coords conservatively bounds every deposited cell for any order // (Esirkepov reaches max+O; O=0 zigzag reaches max+1). static constexpr int FOOTPRINT_REACH = (O == 0u) ? 1 : static_cast(O); -#if defined(TEAM_POLICY_DRIFT) - static constexpr int DRIFT = static_cast(TEAM_POLICY_DRIFT); +#if defined(TILED_DEPOSIT_DRIFT) + static constexpr int DRIFT = static_cast(TILED_DEPOSIT_DRIFT); #else static constexpr int DRIFT = 1; #endif diff --git a/tests/framework/CMakeLists.txt b/tests/framework/CMakeLists.txt index 3696c74a9..24e84d1eb 100644 --- a/tests/framework/CMakeLists.txt +++ b/tests/framework/CMakeLists.txt @@ -44,9 +44,9 @@ else() gen_test(particles_sort false) endif() -# team_policy X-3: per-backend sort_by_key permutation test (only built when the -# compile-time team_policy toggle is on). -if(${team_policy}) +# tiled_deposit X-3: per-backend sort_by_key permutation test (only built when +# the compile-time tiled_deposit toggle is on). +if(${tiled_deposit}) gen_test(sort_by_key false) endif() diff --git a/tests/framework/particles_sort.cpp b/tests/framework/particles_sort.cpp index 591485220..9f1a08ef7 100644 --- a/tests/framework/particles_sort.cpp +++ b/tests/framework/particles_sort.cpp @@ -63,7 +63,7 @@ auto main(int argc, char* argv[]) -> int { i2_p(p) = 23u; weight_p(p) = 3.0; } - // team_policy keys on min(i, i_prev); without a meaningful + // tiled_deposit keys on min(i, i_prev); without a meaningful // i_prev every key would collapse to 0. Set i_prev = i so the // tile key reduces to the particle's current cell. i1_prev_p(p) = i1_p(p); @@ -96,9 +96,9 @@ auto main(int argc, char* argv[]) -> int { Kokkos::deep_copy(pld_i_h, prtls.pld_i); // Tile geometry, mirroring sort::PositionToTileIndex. T = 1 (no - // team_policy) reproduces the legacy per-cell ordering. -#if defined(TEAM_POLICY) - const ncells_t T = static_cast(TEAM_POLICY_TILE_SIZE); + // tiled_deposit) reproduces the legacy per-cell ordering. +#if defined(TILED_DEPOSIT) + const ncells_t T = static_cast(TILED_DEPOSIT_TILE_SIZE); #else const ncells_t T = 1u; #endif @@ -115,15 +115,15 @@ auto main(int argc, char* argv[]) -> int { // non-decreasing tile index; (2) every SoA member is permuted by the // *same* permutation, so each alive slot still satisfies // pld == f(weight); (3) no alive particle is lost. Only [0, npart()) - // is defined after a sort. The team_policy path compacts — it drops + // is defined after a sort. The tiled_deposit path compacts — it drops // the dead, so npart() equals the alive count and [0, npart()) is // entirely alive; the legacy (non-team) path keeps the dead as a // weight == -1 suffix, leaving npart() unchanged. Iterating // [0, npart()) exercises both: the prefix-sorted / no-alive-after-dead // checks below hold either way. -#if defined(TEAM_POLICY) +#if defined(TILED_DEPOSIT) raise::ErrorIf(prtls.npart() != 59u, - "team_policy sort must compact: npart() should equal " + "tiled_deposit sort must compact: npart() should equal " "the alive count", HERE); #else @@ -229,7 +229,7 @@ auto main(int argc, char* argv[]) -> int { i3_p(p) = 7u; weight_p(p) = 4.0; } - // see 2D block: i_prev = i so the team_policy tile key reduces + // see 2D block: i_prev = i so the tiled_deposit tile key reduces // to the particle's current cell. i1_prev_p(p) = i1_p(p); i2_prev_p(p) = i2_p(p); @@ -258,11 +258,11 @@ auto main(int argc, char* argv[]) -> int { // Same invariants as the 2D block (no payloads here): alive prefix // sorted by non-decreasing tile index, alive count preserved. The - // team_policy path compacts the dead away (npart() == alive count); + // tiled_deposit path compacts the dead away (npart() == alive count); // the legacy path keeps them as a weight == -1 suffix. T = 1 // reproduces the legacy per-cell order. -#if defined(TEAM_POLICY) - const ncells_t T = static_cast(TEAM_POLICY_TILE_SIZE); +#if defined(TILED_DEPOSIT) + const ncells_t T = static_cast(TILED_DEPOSIT_TILE_SIZE); #else const ncells_t T = 1u; #endif @@ -276,9 +276,9 @@ auto main(int argc, char* argv[]) -> int { (static_cast(c) / T); }; -#if defined(TEAM_POLICY) +#if defined(TILED_DEPOSIT) raise::ErrorIf(prtls.npart() != 59u, - "team_policy sort must compact: npart() should equal " + "tiled_deposit sort must compact: npart() should equal " "the alive count", HERE); #else diff --git a/tests/framework/sort_by_key.cpp b/tests/framework/sort_by_key.cpp index 64c4eb180..31a1bc097 100644 --- a/tests/framework/sort_by_key.cpp +++ b/tests/framework/sort_by_key.cpp @@ -1,5 +1,5 @@ /** - * @brief X-3 (team_policy) — sort_by_key permutation test. + * @brief X-3 (tiled_deposit) — sort_by_key permutation test. * * Exercises every backend overload of `ntt::sort_helpers::sort_by_key_dispatch` * that is compiled in for the current Kokkos device. For each backend: @@ -11,7 +11,7 @@ * promise stability per their documentation but we don't bake that into * the test). * - * Built only when `team_policy=ON` at CMake time. + * Built only when `tiled_deposit=ON` at CMake time. */ #include "enums.h" #include "global.h" diff --git a/tests/kernels/CMakeLists.txt b/tests/kernels/CMakeLists.txt index 0fe23f578..7e6b6b280 100644 --- a/tests/kernels/CMakeLists.txt +++ b/tests/kernels/CMakeLists.txt @@ -26,7 +26,7 @@ endfunction() gen_test(faraday_mink) gen_test(ampere_mink) gen_test(deposit) -if(${team_policy}) +if(${tiled_deposit}) gen_test(deposit_tiled) endif() gen_test(digital_filter) diff --git a/tests/kernels/deposit_tiled.cpp b/tests/kernels/deposit_tiled.cpp index 90a3890bf..693181334 100644 --- a/tests/kernels/deposit_tiled.cpp +++ b/tests/kernels/deposit_tiled.cpp @@ -7,7 +7,7 @@ * for shape orders O = 1..11 and asserts that the resulting J array is * identical cell-by-cell within a small floating-point tolerance. * - * Built only when `team_policy=ON` (`-D TEAM_POLICY` defined). The test + * Built only when `tiled_deposit=ON` (`-D TILED_DEPOSIT` defined). The test * matches the per-particle setup used in `deposit.cpp` so that any * regression in the shared `kernel::deposit::deposit_one_particle` body * is caught by both tests. From 54ca1886a732b02ac261b5452406102db6e2e41e Mon Sep 17 00:00:00 2001 From: haykh Date: Thu, 24 Sep 2026 19:26:51 -0400 Subject: [PATCH 114/125] rm deprecated team_policy flags --- CMakeLists.txt | 21 --------------------- 1 file changed, 21 deletions(-) diff --git a/CMakeLists.txt b/CMakeLists.txt index 1e971071d..c904ccab2 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -58,27 +58,6 @@ set(gpu_aware_mpi ${default_gpu_aware_mpi} CACHE BOOL "Enable GPU-aware MPI") -# deprecated `team_policy*` options (renamed to `tiled_deposit*`); forward them -# unless the new name was given explicitly, and drop the stale cache entries -foreach(_suffix "" "_tile_size" "_drift") - if(DEFINED team_policy${_suffix}) - if(NOT DEFINED tiled_deposit${_suffix}) - message(WARNING "`team_policy${_suffix}` is deprecated, " - "use `tiled_deposit${_suffix}` instead") - if("${_suffix}" STREQUAL "") - set(_type BOOL) - else() - set(_type STRING) - endif() - set(tiled_deposit${_suffix} - ${team_policy${_suffix}} - CACHE ${_type} "") - endif() - unset(team_policy${_suffix} CACHE) - endif() -endforeach() -unset(team_policy_tile_sizes CACHE) - set(tiled_deposit ${default_tiled_deposit} CACHE BOOL "Enable tile-blocked deposit/pusher kernels") From 2a30aaf1a5f1825d324b378fb2343fdcd186d6bf Mon Sep 17 00:00:00 2001 From: haykh Date: Thu, 24 Sep 2026 19:27:34 -0400 Subject: [PATCH 115/125] render now independent of output --- src/framework/CMakeLists.txt | 8 +- src/framework/domain/io/init.cpp | 46 +- .../{metadomain_render.cpp => io/render.cpp} | 746 +++++++++++------- src/framework/domain/metadomain.h | 15 +- src/output/CMakeLists.txt | 23 +- src/output/render/axes.h | 390 ++++++--- src/output/render/colorbar.h | 335 ++++++-- src/output/render/composite.h | 208 +++-- src/output/render/fieldlines.h | 208 ++--- src/output/render/png.h | 40 +- src/output/render/raymarch.hpp | 137 ++-- src/output/render/reduce.hpp | 78 +- src/output/render/renderer.cpp | 619 +++++++++------ src/output/render/renderer.h | 178 +++-- src/output/render/slice2d.hpp | 94 +-- src/output/render/transfer_fn.h | 12 +- 16 files changed, 1884 insertions(+), 1253 deletions(-) rename src/framework/domain/{metadomain_render.cpp => io/render.cpp} (67%) diff --git a/src/framework/CMakeLists.txt b/src/framework/CMakeLists.txt index 63da7084d..e735890ab 100644 --- a/src/framework/CMakeLists.txt +++ b/src/framework/CMakeLists.txt @@ -16,7 +16,6 @@ # * domain/metadomain_sort.cpp # * domain/metadomain_reshape.cpp # * domain/metadomain_loadbal.cpp -# * domain/metadomain_render.cpp # * domain/checkpoint/init.cpp # * domain/checkpoint/write.cpp # * domain/checkpoint/resume.cpp @@ -29,6 +28,7 @@ # * domain/io/spectra.cpp # * domain/io/fields.cpp # * domain/io/stats.cpp +# * domain/io/render.cpp # * containers/particles.cpp # * containers/particles_sort.cpp # * containers/fields.cpp @@ -70,11 +70,12 @@ set(SOURCES ${SRC_DIR}/domain/metadomain_sort.cpp ${SRC_DIR}/domain/metadomain_reshape.cpp ${SRC_DIR}/domain/metadomain_loadbal.cpp - ${SRC_DIR}/domain/metadomain_render.cpp + ${SRC_DIR}/domain/io/init.cpp + ${SRC_DIR}/domain/io/render.cpp + ${SRC_DIR}/domain/io/stats.cpp ${SRC_DIR}/domain/comm/fields.cpp ${SRC_DIR}/domain/comm/fields_sync.cpp ${SRC_DIR}/domain/comm/particles.cpp - ${SRC_DIR}/domain/io/stats.cpp ${SRC_DIR}/containers/particles.cpp ${SRC_DIR}/containers/particles_sort.cpp ${SRC_DIR}/containers/fields.cpp) @@ -82,7 +83,6 @@ if(${output}) list( APPEND SOURCES - ${SRC_DIR}/domain/io/init.cpp ${SRC_DIR}/domain/io/write.cpp ${SRC_DIR}/domain/io/spectra.cpp ${SRC_DIR}/domain/io/fields.cpp diff --git a/src/framework/domain/io/init.cpp b/src/framework/domain/io/init.cpp index 472ba9618..2f563f89a 100644 --- a/src/framework/domain/io/init.cpp +++ b/src/framework/domain/io/init.cpp @@ -1,4 +1,3 @@ -#include "defaults.h" #include "enums.h" #include "global.h" @@ -6,7 +5,6 @@ #include "utils/error.h" #include "framework/domain/domain.h" -#include "framework/domain/mesh.h" #include "framework/domain/metadomain.h" #include "framework/parameters/parameters.h" #include "framework/specialization_registry.h" @@ -15,14 +13,23 @@ #include #include -#include -#include -#include +#if defined(OUTPUT_ENABLED) + #include "defaults.h" + + #include + #include + + #include + #include + #include +#endif // OUTPUT_ENABLED + #include #include namespace ntt { +#if defined(OUTPUT_ENABLED) template void Metadomain::InitWriter(adios2::ADIOS* ptr_adios, const SimulationParams& params) { @@ -105,6 +112,7 @@ namespace ntt { } g_writer.writeAttrs(params); } +#endif template void Metadomain::InitStatsWriter(const SimulationParams& params, @@ -146,16 +154,30 @@ namespace ntt { } } + template + void Metadomain::InitRenderer(const SimulationParams& params) { + g_renderer.init(params, mesh().extent()); + } + +#if defined(OUTPUT_ENABLED) // NOLINTBEGIN(bugprone-macro-parentheses) -#define METADOMAIN_OUTPUT(S, M, D) \ - template void Metadomain>::InitWriter(adios2::ADIOS*, \ - const SimulationParams&); \ - template void Metadomain>::InitStatsWriter(const SimulationParams&, \ - bool); + #define METADOMAIN_OUTPUT_INIT(S, M, D) \ + template void Metadomain>::InitWriter(adios2::ADIOS*, \ + const SimulationParams&); + + NTT_FOREACH_SPECIALIZATION(METADOMAIN_OUTPUT_INIT) - NTT_FOREACH_SPECIALIZATION(METADOMAIN_OUTPUT) + #undef METADOMAIN_OUTPUT_INIT + // NOLINTEND(bugprone-macro-parentheses) +#endif -#undef METADOMAIN_OUTPUT + // NOLINTBEGIN(bugprone-macro-parentheses) +#define METADOMAIN_OUTPUT_INIT(S, M, D) \ + template void Metadomain>::InitStatsWriter(const SimulationParams&, \ + bool); \ + template void Metadomain>::InitRenderer(const SimulationParams&); + NTT_FOREACH_SPECIALIZATION(METADOMAIN_OUTPUT_INIT) +#undef METADOMAIN_OUTPUT_INIT // NOLINTEND(bugprone-macro-parentheses) } // namespace ntt diff --git a/src/framework/domain/metadomain_render.cpp b/src/framework/domain/io/render.cpp similarity index 67% rename from src/framework/domain/metadomain_render.cpp rename to src/framework/domain/io/render.cpp index c119b8fc6..cbf11edf0 100644 --- a/src/framework/domain/metadomain_render.cpp +++ b/src/framework/domain/io/render.cpp @@ -1,5 +1,5 @@ /** - * @file framework/domain/metadomain_render.cpp + * @file framework/domain/io/render.cpp * @brief Metadomain driver for the in-situ volume renderer * @implements * - ntt::Metadomain::InitRenderer @@ -8,7 +8,6 @@ * - ntt:: * @macros: * - MPI_ENABLED - * - OUTPUT_ENABLED * @note * This is the templated counterpart of the (plain) out::Renderer: it owns the * per-(engine, metric, dim) field preparation, the device ray-march kernel @@ -37,6 +36,7 @@ #include "kernels/particle_moments.hpp" #include "output/render/composite.h" #include "output/render/fieldlines.h" + #include "output/render/raymarch.hpp" #include "output/render/reduce.hpp" #include "output/render/slice2d.hpp" @@ -87,9 +87,9 @@ namespace ntt { HERE); } auto scatter_buff = Kokkos::Experimental::create_scatter_view(buffer); - const auto use_weights = params.get("particles.use_weights"); - const auto ni2 = mesh.n_active(in::x2); - const auto inv_n0 = ONE / params.get("scales.n0"); + const auto use_weights = params.get("particles.use_weights"); + const auto ni2 = mesh.n_active(in::x2); + const auto inv_n0 = ONE / params.get("scales.n0"); const auto smooth_order = params.get( "output.fields.smoothing.order"); const auto smooth_method = OutputSmoothingType::from_string( @@ -122,9 +122,8 @@ namespace ntt { const cell_range_t& from) { const cell_range_t to { 0, 3 }; if constexpr (D == Dim::_2D) { - Kokkos::deep_copy( - Kokkos::subview(dst, Kokkos::ALL, Kokkos::ALL, to), - Kokkos::subview(src, Kokkos::ALL, Kokkos::ALL, from)); + Kokkos::deep_copy(Kokkos::subview(dst, Kokkos::ALL, Kokkos::ALL, to), + Kokkos::subview(src, Kokkos::ALL, Kokkos::ALL, from)); } else if constexpr (D == Dim::_3D) { Kokkos::deep_copy( Kokkos::subview(dst, Kokkos::ALL, Kokkos::ALL, Kokkos::ALL, to), @@ -137,17 +136,17 @@ namespace ntt { // field and can trace identical global field lines locally. `bckp` is used // as scratch (overwritten). 3D only (the field-line renderer is Cartesian). template - auto buildCoarseFieldVec(const Mesh& mesh, - const Fields& fields, - ndfield_t& bckp, - char fbase, - const real_t gorigin[3], - const int gnc[3], - const real_t gdx[3]) -> out::CoarseField { - const auto metric = mesh.metric; + auto buildCoarseFieldVec(const Mesh& mesh, + const Fields& fields, + ndfield_t& bckp, + char fbase, + const real_t gorigin[3], + const int gnc[3], + const real_t gdx[3]) -> out::CoarseField { + const auto metric = mesh.metric; // raw vector components -> bckp(0,1,2) uint8_t src_base = em::bx1; - PrepareOutputFlags interp = PrepareOutput::InterpToCellCenterFromFaces; + PrepareOutputFlags interp = PrepareOutput::InterpToCellCenterFromFaces; bool is_current = false; if (fbase == 'E') { src_base = em::ex1; @@ -158,23 +157,28 @@ namespace ntt { interp = PrepareOutput::InterpToCellCenterFromEdges; } if (is_current) { - copyVec3ToBckp(fields.cur, bckp, + copyVec3ToBckp(fields.cur, + bckp, cell_range_t(cur::jx1, cur::jx3 + 1)); } else { - copyVec3ToBckp(fields.em, bckp, + copyVec3ToBckp(fields.em, + bckp, cell_range_t(src_base, src_base + 3)); } // interpolate to cell centers + convert to physical basis -> bckp(3,4,5) - const PrepareOutputFlags prepare = (S == SimEngine::SRPIC) - ? PrepareOutput::ConvertToHat - : PrepareOutput::ConvertToPhysCntrv; + const PrepareOutputFlags prepare = (S == SimEngine::SRPIC) + ? PrepareOutput::ConvertToHat + : PrepareOutput::ConvertToPhysCntrv; list_t comp_from = { 0, 1, 2 }; list_t comp_to = { 3, 4, 5 }; - Kokkos::parallel_for( - "RenderFLFieldsToPhys", - mesh.rangeActiveCells(), - kernel::FieldsToPhys_kernel(bckp, bckp, comp_from, comp_to, - interp | prepare, metric)); + Kokkos::parallel_for("RenderFLFieldsToPhys", + mesh.rangeActiveCells(), + kernel::FieldsToPhys_kernel(bckp, + bckp, + comp_from, + comp_to, + interp | prepare, + metric)); Kokkos::fence(); // pull the physical components to host and bin into the coarse grid @@ -187,7 +191,7 @@ namespace ntt { std::vector sum(ncell * 3, ZERO); std::vector cnt(ncell, ZERO); - const auto le = mesh.extent(); + const auto le = mesh.extent(); const real_t llo[3] = { le[0].first, le[1].first, le[2].first }; const real_t lsz[3] = { le[0].second - le[0].first, le[1].second - le[1].first, @@ -200,9 +204,12 @@ namespace ntt { for (int j = 0; j < nl[1]; ++j) { for (int i = 0; i < nl[0]; ++i) { const real_t world[3] = { - llo[0] + (static_cast(i) + HALF) * lsz[0] / nl[0], - llo[1] + (static_cast(j) + HALF) * lsz[1] / nl[1], - llo[2] + (static_cast(k) + HALF) * lsz[2] / nl[2] + llo[0] + (static_cast(i) + HALF) * lsz[0] / + static_cast(nl[0]), + llo[1] + (static_cast(j) + HALF) * lsz[1] / + static_cast(nl[1]), + llo[2] + (static_cast(k) + HALF) * lsz[2] / + static_cast(nl[2]) }; int c[3]; for (int d = 0; d < 3; ++d) { @@ -223,10 +230,18 @@ namespace ntt { } } #if defined(MPI_ENABLED) - MPI_Allreduce(MPI_IN_PLACE, sum.data(), static_cast(ncell * 3), - mpi::get_type(), MPI_SUM, MPI_COMM_WORLD); - MPI_Allreduce(MPI_IN_PLACE, cnt.data(), static_cast(ncell), - mpi::get_type(), MPI_SUM, MPI_COMM_WORLD); + MPI_Allreduce(MPI_IN_PLACE, + sum.data(), + static_cast(ncell * 3), + mpi::get_type(), + MPI_SUM, + MPI_COMM_WORLD); + MPI_Allreduce(MPI_IN_PLACE, + cnt.data(), + static_cast(ncell), + mpi::get_type(), + MPI_SUM, + MPI_COMM_WORLD); #endif out::CoarseField cf; cf.B.assign(ncell * 3, ZERO); @@ -237,10 +252,10 @@ namespace ntt { } for (std::size_t c = 0; c < ncell; ++c) { if (cnt[c] > ZERO) { - const real_t inv = ONE / cnt[c]; - cf.B[c * 3 + 0] = sum[c * 3 + 0] * inv; - cf.B[c * 3 + 1] = sum[c * 3 + 1] * inv; - cf.B[c * 3 + 2] = sum[c * 3 + 2] * inv; + const real_t inv = ONE / cnt[c]; + cf.B[c * 3 + 0] = sum[c * 3 + 0] * inv; + cf.B[c * 3 + 1] = sum[c * 3 + 1] * inv; + cf.B[c * 3 + 2] = sum[c * 3 + 2] * inv; } } return cf; @@ -256,10 +271,10 @@ namespace ntt { char fbase, const real_t gorigin[2], const int gnc[2], - const real_t gdx[2]) -> out::CoarseField2D { - const auto metric = mesh.metric; + const real_t gdx[2]) -> out::CoarseField2D { + const auto metric = mesh.metric; uint8_t src_base = em::bx1; - PrepareOutputFlags interp = PrepareOutput::InterpToCellCenterFromFaces; + PrepareOutputFlags interp = PrepareOutput::InterpToCellCenterFromFaces; bool is_current = false; if (fbase == 'E') { src_base = em::ex1; @@ -270,22 +285,27 @@ namespace ntt { interp = PrepareOutput::InterpToCellCenterFromEdges; } if (is_current) { - copyVec3ToBckp(fields.cur, bckp, + copyVec3ToBckp(fields.cur, + bckp, cell_range_t(cur::jx1, cur::jx3 + 1)); } else { - copyVec3ToBckp(fields.em, bckp, + copyVec3ToBckp(fields.em, + bckp, cell_range_t(src_base, src_base + 3)); } - const PrepareOutputFlags prepare = (S == SimEngine::SRPIC) - ? PrepareOutput::ConvertToHat - : PrepareOutput::ConvertToPhysCntrv; + const PrepareOutputFlags prepare = (S == SimEngine::SRPIC) + ? PrepareOutput::ConvertToHat + : PrepareOutput::ConvertToPhysCntrv; list_t comp_from = { 0, 1, 2 }; list_t comp_to = { 3, 4, 5 }; - Kokkos::parallel_for( - "RenderFL2DFieldsToPhys", - mesh.rangeActiveCells(), - kernel::FieldsToPhys_kernel(bckp, bckp, comp_from, comp_to, - interp | prepare, metric)); + Kokkos::parallel_for("RenderFL2DFieldsToPhys", + mesh.rangeActiveCells(), + kernel::FieldsToPhys_kernel(bckp, + bckp, + comp_from, + comp_to, + interp | prepare, + metric)); Kokkos::fence(); auto bckp_h = Kokkos::create_mirror_view(bckp); @@ -294,7 +314,7 @@ namespace ntt { const std::size_t ncell = static_cast(gnc[0]) * gnc[1]; std::vector sum(ncell * 2, ZERO); std::vector cnt(ncell, ZERO); - const auto le = mesh.extent(); + const auto le = mesh.extent(); const real_t llo[2] = { le[0].first, le[1].first }; const real_t lsz[2] = { le[0].second - le[0].first, le[1].second - le[1].first }; @@ -303,14 +323,14 @@ namespace ntt { const int NG = static_cast(N_GHOSTS); for (int j = 0; j < nl[1]; ++j) { for (int i = 0; i < nl[0]; ++i) { - const real_t world[2] = { - llo[0] + (static_cast(i) + HALF) * lsz[0] / nl[0], - llo[1] + (static_cast(j) + HALF) * lsz[1] / nl[1] - }; - int c[2]; + const real_t world[2] = { llo[0] + (static_cast(i) + HALF) * + lsz[0] / static_cast(nl[0]), + llo[1] + (static_cast(j) + HALF) * + lsz[1] / + static_cast(nl[1]) }; + int c[2]; for (int d = 0; d < 2; ++d) { - int cc = static_cast( - std::floor((world[d] - gorigin[d]) / gdx[d])); + int cc = static_cast(std::floor((world[d] - gorigin[d]) / gdx[d])); cc = (cc < 0) ? 0 : ((cc > gnc[d] - 1) ? gnc[d] - 1 : cc); c[d] = cc; } @@ -321,10 +341,18 @@ namespace ntt { } } #if defined(MPI_ENABLED) - MPI_Allreduce(MPI_IN_PLACE, sum.data(), static_cast(ncell * 2), - mpi::get_type(), MPI_SUM, MPI_COMM_WORLD); - MPI_Allreduce(MPI_IN_PLACE, cnt.data(), static_cast(ncell), - mpi::get_type(), MPI_SUM, MPI_COMM_WORLD); + MPI_Allreduce(MPI_IN_PLACE, + sum.data(), + static_cast(ncell * 2), + mpi::get_type(), + MPI_SUM, + MPI_COMM_WORLD); + MPI_Allreduce(MPI_IN_PLACE, + cnt.data(), + static_cast(ncell), + mpi::get_type(), + MPI_SUM, + MPI_COMM_WORLD); #endif out::CoarseField2D cf; cf.B.assign(ncell * 2, ZERO); @@ -349,7 +377,7 @@ namespace ntt { auto Metadomain::prepareRenderScalar(const SimulationParams& params, Domain& domain, const std::string& field_name, - ndfield_t& bckp) const + ndfield_t& bckp) const -> bool { // Parse an optional trailing per-species suffix "__..."; // species apply to particle moments only (N, Nppc, Rho, Charge, T, V). @@ -390,9 +418,9 @@ namespace ntt { } } if (bad_species) { - raise::Warning("output.render: invalid species in '" + field_name + - "', skipping", - HERE); + raise::Warning( + "output.render: invalid species in '" + field_name + "', skipping", + HERE); return false; } @@ -422,17 +450,31 @@ namespace ntt { if (base == "N" or base == "Nppc" or base == "Rho" or base == "Charge") { // scalar particle moments if (base == "N") { - renderMoment(params, mesh, domain.species, species, {}, - bckp, 0u); + renderMoment(params, mesh, domain.species, species, {}, bckp, 0u); } else if (base == "Nppc") { - renderMoment(params, mesh, domain.species, species, - {}, bckp, 0u); + renderMoment(params, + mesh, + domain.species, + species, + {}, + bckp, + 0u); } else if (base == "Rho") { - renderMoment(params, mesh, domain.species, species, - {}, bckp, 0u); + renderMoment(params, + mesh, + domain.species, + species, + {}, + bckp, + 0u); } else { - renderMoment(params, mesh, domain.species, species, - {}, bckp, 0u); + renderMoment(params, + mesh, + domain.species, + species, + {}, + bckp, + 0u); } // sum boundary-crossing particle deposits back into active cells SynchronizeFields(domain, Comm::Bckp, { 0, 1 }); @@ -445,8 +487,13 @@ namespace ntt { if (i >= 0 and j >= 0) { const std::vector comps { static_cast(i), static_cast(j) }; - renderMoment(params, mesh, domain.species, species, - comps, bckp, 0u); + renderMoment(params, + mesh, + domain.species, + species, + comps, + bckp, + 0u); SynchronizeFields(domain, Comm::Bckp, { 0, 1 }); return true; } @@ -454,45 +501,93 @@ namespace ntt { // bulk-velocity magnitude |V| = sqrt(V1^2 + V2^2 + V3^2) if constexpr (S == SimEngine::GRPIC) { // GR: Eckart-frame 4-velocity; need all 4 components for the norm - renderMoment(params, mesh, domain.species, species, - { 0u }, bckp, 0u); - renderMoment(params, mesh, domain.species, species, - { 1u }, bckp, 1u); - renderMoment(params, mesh, domain.species, species, - { 2u }, bckp, 2u); - renderMoment(params, mesh, domain.species, species, - { 3u }, bckp, 3u); + renderMoment(params, + mesh, + domain.species, + species, + { 0u }, + bckp, + 0u); + renderMoment(params, + mesh, + domain.species, + species, + { 1u }, + bckp, + 1u); + renderMoment(params, + mesh, + domain.species, + species, + { 2u }, + bckp, + 2u); + renderMoment(params, + mesh, + domain.species, + species, + { 3u }, + bckp, + 3u); SynchronizeFields(domain, Comm::Bckp, { 0, 4 }); Kokkos::parallel_for( "RenderNormalize4Vel", mesh.rangeActiveCells(), - kernel::Normalize4VelocityByNorm_kernel( - bckp, bckp, 0, 1, 2, 3, metric)); + kernel::Normalize4VelocityByNorm_kernel(bckp, + bckp, + 0, + 1, + 2, + 3, + metric)); Kokkos::parallel_for( "RenderTransform4Vel", mesh.rangeActiveCells(), - kernel::Transform4VelocitySpatialToPhysical_kernel( - bckp, 1, 2, 3, metric)); + kernel::Transform4VelocitySpatialToPhysical_kernel(bckp, + 1, + 2, + 3, + metric)); // |spatial physical 4-velocity| -> bckp(0) - Kokkos::parallel_for("RenderVmagGR", - mesh.rangeActiveCells(), - kernel::RenderMagnitude3_kernel(bckp, 1, - 2, 3, 0)); + Kokkos::parallel_for( + "RenderVmagGR", + mesh.rangeActiveCells(), + render::RenderMagnitude3_kernel(bckp, 1, 2, 3, 0)); } else { // SR: mass-weighted bulk 3-velocity, normalized by Rho - renderMoment(params, mesh, domain.species, species, - { 1u }, bckp, 0u); - renderMoment(params, mesh, domain.species, species, - { 2u }, bckp, 1u); - renderMoment(params, mesh, domain.species, species, - { 3u }, bckp, 2u); - renderMoment(params, mesh, domain.species, species, - {}, bckp, 3u); + renderMoment(params, + mesh, + domain.species, + species, + { 1u }, + bckp, + 0u); + renderMoment(params, + mesh, + domain.species, + species, + { 2u }, + bckp, + 1u); + renderMoment(params, + mesh, + domain.species, + species, + { 3u }, + bckp, + 2u); + renderMoment(params, + mesh, + domain.species, + species, + {}, + bckp, + 3u); SynchronizeFields(domain, Comm::Bckp, { 0, 4 }); - Kokkos::parallel_for("RenderVmagSR", - mesh.rangeActiveCells(), - kernel::RenderVmagByRho_kernel(bckp, 0, 1, - 2, 3, 0)); + Kokkos::parallel_for( + "RenderVmagSR", + mesh.rangeActiveCells(), + render::RenderVmagByRho_kernel(bckp, 0, 1, 2, 3, 0)); } return true; } else if (base.size() == 2 and base[0] == 'V') { @@ -501,46 +596,86 @@ namespace ntt { if constexpr (S == SimEngine::GRPIC) { // GR: 4-velocity component (t/0 = u^0 = Gamma/alpha; x,y,z spatial) if (c >= 0 and c <= 3) { - renderMoment(params, mesh, domain.species, species, - { 0u }, bckp, 0u); - renderMoment(params, mesh, domain.species, species, - { 1u }, bckp, 1u); - renderMoment(params, mesh, domain.species, species, - { 2u }, bckp, 2u); - renderMoment(params, mesh, domain.species, species, - { 3u }, bckp, 3u); + renderMoment(params, + mesh, + domain.species, + species, + { 0u }, + bckp, + 0u); + renderMoment(params, + mesh, + domain.species, + species, + { 1u }, + bckp, + 1u); + renderMoment(params, + mesh, + domain.species, + species, + { 2u }, + bckp, + 2u); + renderMoment(params, + mesh, + domain.species, + species, + { 3u }, + bckp, + 3u); SynchronizeFields(domain, Comm::Bckp, { 0, 4 }); Kokkos::parallel_for( "RenderNormalize4Vel", mesh.rangeActiveCells(), - kernel::Normalize4VelocityByNorm_kernel( - bckp, bckp, 0, 1, 2, 3, metric)); + kernel::Normalize4VelocityByNorm_kernel(bckp, + bckp, + 0, + 1, + 2, + 3, + metric)); Kokkos::parallel_for( "RenderTransform4Vel", mesh.rangeActiveCells(), kernel::Transform4VelocitySpatialToPhysical_kernel( - bckp, 1, 2, 3, metric)); + bckp, + 1, + 2, + 3, + metric)); if (c != 0) { Kokkos::parallel_for( "RenderPickV", mesh.rangeActiveCells(), - kernel::RenderPickComp_kernel( - bckp, static_cast(c), 0)); + render::RenderPickComp_kernel(bckp, + static_cast(c), + 0)); } return true; } } else { // SR: spatial bulk velocity (x,y,z), normalized by Rho if (c >= 1 and c <= 3) { - renderMoment(params, mesh, domain.species, species, - { static_cast(c) }, bckp, 0u); - renderMoment(params, mesh, domain.species, species, - {}, bckp, 1u); + renderMoment(params, + mesh, + domain.species, + species, + { static_cast(c) }, + bckp, + 0u); + renderMoment(params, + mesh, + domain.species, + species, + {}, + bckp, + 1u); SynchronizeFields(domain, Comm::Bckp, { 0, 2 }); - Kokkos::parallel_for("RenderNormalizeV", - mesh.rangeActiveCells(), - kernel::RenderDivideComp_kernel(bckp, 0, - 1)); + Kokkos::parallel_for( + "RenderNormalizeV", + mesh.rangeActiveCells(), + render::RenderDivideComp_kernel(bckp, 0, 1)); return true; } } @@ -548,10 +683,8 @@ namespace ntt { // Vector field as a scalar: "" with base in {E, B, J} // and selector in {mag, 1/2/3, x/y/z}. A component (e.g. "B1"/"Bx") is // signed; a magnitude (e.g. "Bmag") is non-negative. - const std::string& f = base; - const char fbase = f.empty() - ? '?' - : static_cast(std::toupper(f[0])); + const std::string& f = base; + const char fbase = f.empty() ? '?' : static_cast(std::toupper(f[0])); bool ok = true; bool is_current = false; uint8_t src_base = 0; // first component of the source field @@ -586,39 +719,41 @@ namespace ntt { if (ok) { // raw vector components into bckp(:, 0..2) if (is_current) { - copyVec3ToBckp(domain.fields.cur, bckp, + copyVec3ToBckp(domain.fields.cur, + bckp, cell_range_t(cur::jx1, cur::jx3 + 1)); } else { - copyVec3ToBckp(domain.fields.em, bckp, + copyVec3ToBckp(domain.fields.em, + bckp, cell_range_t(src_base, src_base + 3)); } // interpolate to cell centers + convert to physical basis -> (3,4,5) - const PrepareOutputFlags prepare = (S == SimEngine::SRPIC) - ? PrepareOutput::ConvertToHat - : PrepareOutput::ConvertToPhysCntrv; + const PrepareOutputFlags prepare = (S == SimEngine::SRPIC) + ? PrepareOutput::ConvertToHat + : PrepareOutput::ConvertToPhysCntrv; list_t comp_from = { 0, 1, 2 }; list_t comp_to = { 3, 4, 5 }; - Kokkos::parallel_for( - "RenderFieldsToPhys", - mesh.rangeActiveCells(), - kernel::FieldsToPhys_kernel(bckp, - bckp, - comp_from, - comp_to, - interp | prepare, - metric)); + Kokkos::parallel_for("RenderFieldsToPhys", + mesh.rangeActiveCells(), + kernel::FieldsToPhys_kernel(bckp, + bckp, + comp_from, + comp_to, + interp | prepare, + metric)); // reduce to the scalar to render -> bckp(:, 0) if (comp == -1) { Kokkos::parallel_for( "RenderVectorMagnitude", mesh.rangeActiveCells(), - kernel::RenderMagnitude3_kernel(bckp, 3, 4, 5, 0)); + render::RenderMagnitude3_kernel(bckp, 3, 4, 5, 0)); } else { - Kokkos::parallel_for( - "RenderVectorComponent", - mesh.rangeActiveCells(), - kernel::RenderPickComp_kernel( - bckp, static_cast(3 + comp), 0)); + Kokkos::parallel_for("RenderVectorComponent", + mesh.rangeActiveCells(), + render::RenderPickComp_kernel( + bckp, + static_cast(3 + comp), + 0)); } return true; } @@ -631,11 +766,6 @@ namespace ntt { return false; } - template - void Metadomain::InitRenderer(const SimulationParams& params) { - g_renderer.init(params, mesh().extent()); - } - template auto Metadomain::Render(const SimulationParams& params, timestep_t current_step, @@ -675,44 +805,46 @@ namespace ntt { g_renderer.setDomeActive(is_dome); // optional axis-aligned render region (== full extent when uncropped) - const real_t rlo[3] = { g_renderer.regionLo(0), g_renderer.regionLo(1), - g_renderer.regionLo(2) }; - const real_t rhi[3] = { g_renderer.regionHi(0), g_renderer.regionHi(1), - g_renderer.regionHi(2) }; + const real_t rlo[3] = { g_renderer.regionLo(0), + g_renderer.regionLo(1), + g_renderer.regionLo(2) }; + const real_t rhi[3] = { g_renderer.regionHi(0), + g_renderer.regionHi(1), + g_renderer.regionHi(2) }; // per-domain world AABB, clipped to the region - const auto loc_ext = local_domain->mesh.extent(); - real_t lo[3] = { math::max(loc_ext[0].first, rlo[0]), - math::max(loc_ext[1].first, rlo[1]), - math::max(loc_ext[2].first, rlo[2]) }; - real_t hi[3] = { math::min(loc_ext[0].second, rhi[0]), - math::min(loc_ext[1].second, rhi[1]), - math::min(loc_ext[2].second, rhi[2]) }; - // does this domain intersect the region? if not, render nothing (but still - // join the collective composite / field-line reduce below). - const bool in_region = (lo[0] < hi[0]) and (lo[1] < hi[1]) and + const auto loc_ext = local_domain->mesh.extent(); + real_t lo[3] = { math::max(loc_ext[0].first, rlo[0]), + math::max(loc_ext[1].first, rlo[1]), + math::max(loc_ext[2].first, rlo[2]) }; + real_t hi[3] = { math::min(loc_ext[0].second, rhi[0]), + math::min(loc_ext[1].second, rhi[1]), + math::min(loc_ext[2].second, rhi[2]) }; + // does this domain intersect the region? if not, render nothing (but + // still join the collective composite / field-line reduce below). + const bool in_region = (lo[0] < hi[0]) and (lo[1] < hi[1]) and (lo[2] < hi[2]); // global extent (drives the field-line coarse grid, which spans the full // field regardless of the crop) const auto glob_ext = mesh().extent(); - // fixed world step, identical on all ranks -> seamless. Sized to the region - // diagonal so `samples` spans the (possibly cropped) view. - real_t gdiag = ZERO; + // fixed world step, identical on all ranks -> seamless. Sized to the + // region diagonal so `samples` spans the (possibly cropped) view. + real_t gdiag = ZERO; for (auto d { 0 }; d < 3; ++d) { - const real_t s = rhi[d] - rlo[d]; + const real_t s = rhi[d] - rlo[d]; gdiag += s * s; } - gdiag = math::sqrt(gdiag); - // marched extent per ray: the dome clips each ray to `dome_radius`, so size - // the step by the radius (== `samples` steps across the hemisphere) rather - // than the box diagonal. Identical on all ranks -> seamless. + gdiag = math::sqrt(gdiag); + // marched extent per ray: the dome clips each ray to `dome_radius`, so + // size the step by the radius (== `samples` steps across the hemisphere) + // rather than the box diagonal. Identical on all ranks -> seamless. const real_t march_len = (is_dome and cam.dome_radius > ZERO) ? cam.dome_radius : gdiag; - const real_t ds = (g_renderer.stepSize() > ZERO) - ? g_renderer.stepSize() - : march_len / static_cast(g_renderer.samples()); - const int max_steps = 2 * g_renderer.samples() + 16; + const real_t ds = (g_renderer.stepSize() > ZERO) + ? g_renderer.stepSize() + : march_len / static_cast(g_renderer.samples()); + const int max_steps = 2 * g_renderer.samples() + 16; // region box + depth-occluded spine (opaque box wireframe rendered inline // in the march so the volume covers its far edges). The visual width is @@ -722,16 +854,15 @@ namespace ntt { real_t ghi[3] = { rhi[0], rhi[1], rhi[2] }; const real_t px_w = (cam.half_h * static_cast(2)) / static_cast(H); - const real_t spine_radius = - g_renderer.axes() - ? math::max(static_cast(0.55) * ds, - HALF * g_renderer.spineWidth() * px_w) - : ZERO; + const real_t spine_radius = g_renderer.axes() + ? math::max(static_cast(0.55) * ds, + HALF * g_renderer.spineWidth() * px_w) + : ZERO; // contrasting opaque spine color (white on dark bg, black on light) const real_t bg_lum = static_cast(0.299) * g_renderer.background(0) + static_cast(0.587) * g_renderer.background(1) + static_cast(0.114) * g_renderer.background(2); - const real_t sc = (bg_lum < HALF) ? ONE : ZERO; + const real_t sc = (bg_lum < HALF) ? ONE : ZERO; const real_t spine_rgb[3] = { sc, sc, sc }; // composite order key (depends on the current decomposition offsets) @@ -751,13 +882,12 @@ namespace ntt { // scenes); we only ray-march and composite within it. int bx0 = 0, by0 = 0, bw = 0, bh = 0; const bool on_screen = - in_region and (is_dome ? out::screenBBoxDome(cam, W, H, lo, hi, bx0, by0, - bw, bh) - : out::screenBBox(cam, W, H, lo, hi, bx0, by0, bw, - bh)); + in_region and + (is_dome ? out::screenBBoxDome(cam, W, H, lo, hi, bx0, by0, bw, bh) + : out::screenBBox(cam, W, H, lo, hi, bx0, by0, bw, bh)); - // ---- magnetic-field-line tubes (built once, shared by every scene) --- // - // Every rank coarsens + replicates the field, traces the SAME global + // ---- magnetic-field-line tubes (built once, shared by every scene) --- + // // Every rank coarsens + replicates the field, traces the SAME global // polylines, and keeps only the segments inside its own domain; the // ordered cross-domain composite stitches them. Built before the scene // loop so an overlay and a standalone tube scene share one geometry pass. @@ -780,22 +910,25 @@ namespace ntt { } const char fb = static_cast( std::toupper(flc.field.empty() ? 'B' : flc.field[0])); - out::CoarseField cf = buildCoarseFieldVec( - local_domain->mesh, local_domain->fields, bckp, fb, gorigin, gnc, gdx); + out::CoarseField cf = buildCoarseFieldVec(local_domain->mesh, + local_domain->fields, + bckp, + fb, + gorigin, + gnc, + gdx); // seed/tube scale: world units per screen pixel (orthographic frame) - const real_t wpp = (cam.half_h * TWO) / static_cast(H); - real_t vlo, vhi; - auto lines = out::traceFieldLines(cf, flc, wpp, vlo, vhi); + const real_t wpp = (cam.half_h * TWO) / static_cast(H); + real_t vlo, vhi; + auto lines = out::traceFieldLines(cf, flc, wpp, vlo, vhi); if (flc.vmax > flc.vmin) { // explicit color range overrides auto vlo = flc.vmin; vhi = flc.vmax; } const real_t tube_world = math::max(flc.tube_px, ONE) * wpp; - const real_t eff_r = math::max(tube_world, - static_cast(0.55) * ds); + const real_t eff_r = math::max(tube_world, static_cast(0.55) * ds); std::size_t n_kept = 0; - tubes = out::buildTubeSet(lines, eff_r, flc, vlo, vhi, lo, hi, cf, - n_kept); + tubes = out::buildTubeSet(lines, eff_r, flc, vlo, vhi, lo, hi, cf, n_kept); have_tubes = true; logger::Checkpoint("field lines: " + std::to_string(lines.size()) + " global lines, " + std::to_string(n_kept) + @@ -826,10 +959,10 @@ namespace ntt { HERE); continue; } - const out::TubeSet& kt = show_tubes ? tubes : empty; + const out::TubeSet& kt = show_tubes ? tubes : empty; // a standalone tube scene colors its colorbar by |field|, not by the // (unused) volume transfer function - out::Scene scene_cb = scene; + out::Scene scene_cb = scene; if (fl_only) { scene_cb.tf.vmin = tubes.vmin; scene_cb.tf.vmax = tubes.vmax; @@ -854,10 +987,10 @@ namespace ntt { randacc_ndfield_t Fld { bckp }; Kokkos::parallel_for( "VolumeRayMarch", - CreateRangePolicy({ 0, 0 }, - { static_cast(bw), - static_cast(bh) }), - kernel::VolumeRayMarch_kernel(Fld, + CreateRangePolicy( + { 0, 0 }, + { static_cast(bw), static_cast(bh) }), + render::VolumeRayMarch_kernel(Fld, 0u, metric, cam, @@ -913,7 +1046,7 @@ namespace ntt { std::size_t o = 0; for (std::size_t p = 0; p < bnpix; ++p) { if (image_h(p, 3) > ZERO) { - frag.depth[o] = depth_h(p); + frag.depth[o] = depth_h(p); frag.rgba[o * 4 + 0] = image_h(p, 0); frag.rgba[o * 4 + 1] = image_h(p, 1); frag.rgba[o * 4 + 2] = image_h(p, 2); @@ -936,10 +1069,15 @@ namespace ntt { } } if (is_dome) { - g_renderer.compositeFragAndWrite(std::move(frag), scene_cb, - current_step, current_time); + g_renderer.compositeFragAndWrite(std::move(frag), + scene_cb, + current_step, + current_time); } else { - g_renderer.compositeAndWrite(sub, order_key, scene_cb, current_step, + g_renderer.compositeAndWrite(sub, + order_key, + scene_cb, + current_step, current_time); } rendered_any = true; @@ -973,13 +1111,13 @@ namespace ntt { // fulldome fisheye ("dome master"). Cartesian slices are a flat plane, so // the kernel warps each pixel radially (fisheye). Curvilinear slices - // (spherical / GR Kerr-Schild) are ALREADY a meridional disk, so dome mode - // there is only a framing change: mirror to a full disk (the `mirror` - // default) and fit that disk to the frame's inscribed circle (the pad skip - // below), while the kernel keeps its native (X, Z) meridional map. Reported - // back so the (metric-agnostic) compositor keeps the frame a clean square. - // All ranks take the same branch (M is fixed per run), so it stays seamless - // across tiles. + // (spherical / GR Kerr-Schild) are ALREADY a meridional disk, so dome + // mode there is only a framing change: mirror to a full disk (the + // `mirror` default) and fit that disk to the frame's inscribed circle + // (the pad skip below), while the kernel keeps its native (X, Z) + // meridional map. Reported back so the (metric-agnostic) compositor keeps + // the frame a clean square. All ranks take the same branch (M is fixed + // per run), so it stays seamless across tiles. const out::DomeMap dome = g_renderer.dome(); g_renderer.setDomeActive(dome.enabled); @@ -999,11 +1137,11 @@ namespace ntt { // meridional (X = r sin th, Z = r cos th) bounding box of the cropped // annular wedge r in [x1lo, x1hi], theta in [x2lo, x2hi]. Sample the // boundary (arcs + rays) so the bbox is correct for any theta range. - umin = static_cast(1e30); - umax = static_cast(-1e30); - vmin = static_cast(1e30); - vmax = static_cast(-1e30); - const int NB = 65; + umin = static_cast(1e30); + umax = static_cast(-1e30); + vmin = static_cast(1e30); + vmax = static_cast(-1e30); + const int NB = 65; auto accXZ = [&](real_t r, real_t th) { const real_t X = r * math::sin(th), Z = r * math::cos(th); umin = std::min(umin, X); @@ -1016,7 +1154,7 @@ namespace ntt { } }; for (int k = 0; k < NB; ++k) { - const real_t t = static_cast(k) / static_cast(NB - 1); + const real_t t = static_cast(k) / static_cast(NB - 1); const real_t th = x2lo + (x2hi - x2lo) * t; const real_t rr = x1lo + (x1hi - x1lo) * t; accXZ(x1lo, th); @@ -1033,20 +1171,20 @@ namespace ntt { if (iaspect > waspect) { const real_t cu = HALF * (umin + umax); const real_t hu = HALF * (vmax - vmin) * iaspect; - umin = cu - hu; - umax = cu + hu; + umin = cu - hu; + umax = cu + hu; } else { const real_t cv = HALF * (vmin + vmax); const real_t hv = HALF * (umax - umin) / iaspect; - vmin = cv - hv; - vmax = cv + hv; + vmin = cv - hv; + vmax = cv + hv; } } // spherical slices get a background border so the round outline and its // R/theta labels are not clipped at the frame edges (Cartesian fills the - // frame and draws its ticks in dedicated margins, so it needs none). A dome - // master skips it: the disk must reach the frame's inscribed circle (which - // the projector maps to the dome horizon), and it draws no axes. + // frame and draws its ticks in dedicated margins, so it needs none). A + // dome master skips it: the disk must reach the frame's inscribed circle + // (which the projector maps to the dome horizon), and it draws no axes. if constexpr (M::CoordType != Coord::type::Cartesian) { if (not dome.enabled) { const real_t pad = static_cast(1.12); @@ -1083,13 +1221,13 @@ namespace ntt { const int ext0 = static_cast(bckp.extent(0)); const int ext1 = static_cast(bckp.extent(1)); const auto metric = local_domain->mesh.metric; - const int n1 = static_cast(local_domain->mesh.n_active(in::x1)); - const int n2 = static_cast(local_domain->mesh.n_active(in::x2)); + const int n1 = static_cast(local_domain->mesh.n_active(in::x1)); + const int n2 = static_cast(local_domain->mesh.n_active(in::x2)); // screen-space bbox of this domain's footprint (host projection of the // boundary; an arc for spherical, a box for Cartesian) - const auto le = local_domain->mesh.extent(); - auto toPix = [&](real_t u, real_t v, real_t& px, real_t& py) { + const auto le = local_domain->mesh.extent(); + auto toPix = [&](real_t u, real_t v, real_t& px, real_t& py) { if (dome.enabled and M::CoordType == Coord::type::Cartesian) { // forward fisheye projection (inverse of the kernel's radial map), // used to bound this domain's footprint on the dome disk. Cartesian @@ -1098,9 +1236,9 @@ namespace ntt { const real_t cxp = HALF * static_cast(W); const real_t cyp = HALF * static_cast(H); const real_t Rpx = HALF * static_cast(std::min(W, H)); - const real_t dx = u - dome.cx, dy = v - dome.cy; - const real_t rw = std::sqrt(dx * dx + dy * dy); - real_t fr = (dome.R > ZERO) ? (rw / dome.R) : ZERO; + const real_t dx = u - dome.cx, dy = v - dome.cy; + const real_t rw = std::sqrt(dx * dx + dy * dy); + real_t fr = (dome.R > ZERO) ? (rw / dome.R) : ZERO; if (fr > ONE) { fr = ONE; // clamp onto the rim (conservative for the bbox) } @@ -1116,10 +1254,10 @@ namespace ntt { theta = fr * dome.theta_max; } const real_t rho = (dome.theta_max > ZERO) ? (theta / dome.theta_max) - : fr; + : fr; const real_t phi = std::atan2(dy, dx); - px = cxp + rho * Rpx * std::cos(phi) - HALF; - py = cyp - rho * Rpx * std::sin(phi) - HALF; + px = cxp + rho * Rpx * std::cos(phi) - HALF; + py = cyp - rho * Rpx * std::sin(phi) - HALF; } else { px = (u - umin) / (umax - umin) * static_cast(W) - HALF; py = (vmax - v) / (vmax - vmin) * static_cast(H) - HALF; @@ -1127,7 +1265,7 @@ namespace ntt { }; real_t minx = static_cast(1e30), miny = static_cast(1e30); real_t maxx = static_cast(-1e30), maxy = static_cast(-1e30); - auto acc = [&](real_t u, real_t v) { + auto acc = [&](real_t u, real_t v) { real_t px, py; toPix(u, v, px, py); minx = std::min(minx, px); @@ -1144,7 +1282,7 @@ namespace ntt { const real_t x0 = le[0].first, x1 = le[0].second; const real_t y0 = le[1].first, y1 = le[1].second; for (int k = 0; k < NB; ++k) { - const real_t t = static_cast(k) / static_cast(NB - 1); + const real_t t = static_cast(k) / static_cast(NB - 1); const real_t xx = x0 + (x1 - x0) * t; const real_t yy = y0 + (y1 - y0) * t; acc(xx, y0); @@ -1164,13 +1302,15 @@ namespace ntt { const real_t a0 = le[1].first, a1 = le[1].second; for (int k = 0; k < NB; ++k) { const real_t t = static_cast(k) / static_cast(NB - 1); - const real_t rr = r0 + (r1 - r0) * t; - const real_t aa = a0 + (a1 - a0) * t; + const real_t rr = r0 + (r1 - r0) * t; + const real_t aa = a0 + (a1 - a0) * t; // r-arcs at a0, a1 and theta-rays at r0, r1 - const real_t pts[4][2] = { { r0 * math::sin(aa), r0 * math::cos(aa) }, - { r1 * math::sin(aa), r1 * math::cos(aa) }, - { rr * math::sin(a0), rr * math::cos(a0) }, - { rr * math::sin(a1), rr * math::cos(a1) } }; + const real_t pts[4][2] = { + { r0 * math::sin(aa), r0 * math::cos(aa) }, + { r1 * math::sin(aa), r1 * math::cos(aa) }, + { rr * math::sin(a0), rr * math::cos(a0) }, + { rr * math::sin(a1), rr * math::cos(a1) } + }; for (auto& p : pts) { acc(p[0], p[1]); if (mirror) { @@ -1198,20 +1338,20 @@ namespace ntt { ndomains_per_dim(), fwd2d); - // ---- 2D field lines (built once) ------------------------------------ // - // Cartesian: iso-contours of the flux function psi. Spherical/Kerr: traced - // meridional streamlines (nt2py style). Both come from a coarse, MPI- - // replicated copy of the in-plane field, so the geometry is global and - // seamless across the disjoint tiles. All ranks reach buildCoarseField2D - // together (collective Allreduce); M is fixed per run, so every rank takes - // the same Cartesian/spherical branch. - const auto& flc = g_renderer.fieldlines(); - out::ContourSet contours = out::emptyContourSet(); - out::ContourSet emptyc = out::emptyContourSet(); - out::TubeSet lines2d = out::emptyTubeSet(); - out::TubeSet emptyl = out::emptyTubeSet(); - bool have_fl = false; - real_t fl_vmin = ZERO, fl_vmax = ONE; + // ---- 2D field lines (built once) ------------------------------------ + // // Cartesian: iso-contours of the flux function psi. Spherical/Kerr: + // traced meridional streamlines (nt2py style). Both come from a coarse, + // MPI- replicated copy of the in-plane field, so the geometry is global + // and seamless across the disjoint tiles. All ranks reach + // buildCoarseField2D together (collective Allreduce); M is fixed per run, + // so every rank takes the same Cartesian/spherical branch. + const auto& flc = g_renderer.fieldlines(); + out::ContourSet contours = out::emptyContourSet(); + out::ContourSet emptyc = out::emptyContourSet(); + out::TubeSet lines2d = out::emptyTubeSet(); + out::TubeSet emptyl = out::emptyTubeSet(); + bool have_fl = false; + real_t fl_vmin = ZERO, fl_vmax = ONE; std::string fl_colormap = flc.colormap; if (flc.enable) { const int gN[2] = { static_cast(mesh().n_active(in::x1)), @@ -1227,16 +1367,21 @@ namespace ntt { std::toupper(flc.field.empty() ? 'B' : flc.field[0])); // coarse, replicated in-plane field: (Bx,By) for Cartesian, (Br,Bth) for // spherical (FieldsToPhys writes the physical components in axis order) - out::CoarseField2D cf = buildCoarseField2D( - local_domain->mesh, local_domain->fields, bckp, fb, gorigin, gnc, gdx); - const real_t wpp = (umax - umin) / static_cast(W); + out::CoarseField2D cf = buildCoarseField2D(local_domain->mesh, + local_domain->fields, + bckp, + fb, + gorigin, + gnc, + gdx); + const real_t wpp = (umax - umin) / static_cast(W); if constexpr (M::CoordType == Coord::type::Cartesian) { std::vector psi; real_t pmin, pmax, bmin, bmax; out::computeFlux2D(cf, psi, pmin, pmax, bmin, bmax); const real_t line_half = HALF * math::max(flc.tube_px, ONE); - contours = out::buildContourSet(cf, psi, pmin, pmax, bmin, bmax, flc, - line_half, wpp); + contours = + out::buildContourSet(cf, psi, pmin, pmax, bmin, bmax, flc, line_half, wpp); fl_vmin = contours.vmin; fl_vmax = contours.vmax; fl_colormap = contours.colormap; @@ -1247,32 +1392,30 @@ namespace ntt { } else { // traced meridional streamlines through the coarse (r, theta) field real_t vlo, vhi; - auto poly = out::traceFieldLinesMeridional(cf, flc, wpp, mirror, vlo, - vhi); + auto poly = out::traceFieldLinesMeridional(cf, flc, wpp, mirror, vlo, vhi); if (flc.vmax > flc.vmin) { // explicit |B| range overrides auto vlo = flc.vmin; vhi = flc.vmax; } - const real_t eff_r = math::max(flc.tube_px, ONE) * wpp; + const real_t eff_r = math::max(flc.tube_px, ONE) * wpp; // bucket grid for buildTubeSet: cell ~ a coarse dr (a length), AABB // spans the (mirrored) meridional disk, z is a single thin slab at 0 out::CoarseField bucket_cf; - bucket_cf.dx[0] = cf.dx[0]; - bucket_cf.dx[1] = cf.dx[0]; - bucket_cf.dx[2] = cf.dx[0]; - const real_t rmax = gext[0].second; - const real_t lo[3] = { mirror ? -rmax : ZERO, -rmax, -cf.dx[0] }; - const real_t hi[3] = { rmax, rmax, cf.dx[0] }; + bucket_cf.dx[0] = cf.dx[0]; + bucket_cf.dx[1] = cf.dx[0]; + bucket_cf.dx[2] = cf.dx[0]; + const real_t rmax = gext[0].second; + const real_t lo[3] = { mirror ? -rmax : ZERO, -rmax, -cf.dx[0] }; + const real_t hi[3] = { rmax, rmax, cf.dx[0] }; std::size_t n_kept = 0; - lines2d = out::buildTubeSet(poly, eff_r, flc, vlo, vhi, lo, hi, - bucket_cf, n_kept); + lines2d = out::buildTubeSet(poly, eff_r, flc, vlo, vhi, lo, hi, bucket_cf, n_kept); fl_vmin = lines2d.vmin; fl_vmax = lines2d.vmax; fl_colormap = lines2d.colormap; - logger::Checkpoint("field lines (2D meridional): " + - std::to_string(poly.size()) + " lines, " + - std::to_string(n_kept) + " segments", - HERE); + logger::Checkpoint( + "field lines (2D meridional): " + std::to_string(poly.size()) + + " lines, " + std::to_string(n_kept) + " segments", + HERE); } have_fl = true; } @@ -1291,15 +1434,16 @@ namespace ntt { } CommunicateBckp(*local_domain, { 0, 1 }); } else if (not have_fl) { - raise::Warning("output.render: 'fieldlines' scene needs a 2D run with " - "[output.render.fieldlines]; skipping", - HERE); + raise::Warning( + "output.render: 'fieldlines' scene needs a 2D run with " + "[output.render.fieldlines]; skipping", + HERE); continue; } - const out::ContourSet& kc = show_lines ? contours : emptyc; - const out::TubeSet& kt = show_lines ? lines2d : emptyl; + const out::ContourSet& kc = show_lines ? contours : emptyc; + const out::TubeSet& kt = show_lines ? lines2d : emptyl; // a standalone field-line scene colors its colorbar by |B| - out::Scene scene_cb = scene; + out::Scene scene_cb = scene; if (fl_only) { scene_cb.tf.vmin = fl_vmin; scene_cb.tf.vmax = fl_vmax; @@ -1312,20 +1456,20 @@ namespace ntt { out::SubImage sub; if (bw > 0 and bh > 0) { - sub.x0 = bx0; - sub.y0 = by0; - sub.w = bw; - sub.h = bh; + sub.x0 = bx0; + sub.y0 = by0; + sub.w = bw; + sub.h = bh; const std::size_t bnpix = static_cast(bw) * static_cast(bh); array_t image { "render_img", bnpix }; randacc_ndfield_t Fld { bckp }; Kokkos::parallel_for( "Slice2DRaster", - CreateRangePolicy({ 0, 0 }, - { static_cast(bw), - static_cast(bh) }), - kernel::SliceRaster_kernel(Fld, + CreateRangePolicy( + { 0, 0 }, + { static_cast(bw), static_cast(bh) }), + render::SliceRaster_kernel(Fld, 0u, metric, umin, @@ -1369,8 +1513,7 @@ namespace ntt { sub.rgba[p * 4 + 3] = image_h(p, 3); } } - g_renderer.compositeAndWrite(sub, order_key, scene_cb, current_step, - current_time); + g_renderer.compositeAndWrite(sub, order_key, scene_cb, current_step, current_time); rendered_any = true; } return rendered_any; @@ -1385,7 +1528,6 @@ namespace ntt { // NOLINTBEGIN(bugprone-macro-parentheses) #define METADOMAIN_RENDER(S, M, D) \ - template void Metadomain>::InitRenderer(const SimulationParams&); \ template auto Metadomain>::Render(const SimulationParams&, \ timestep_t, \ timestep_t, \ diff --git a/src/framework/domain/metadomain.h b/src/framework/domain/metadomain.h index cae8553a0..dd0eb3380 100644 --- a/src/framework/domain/metadomain.h +++ b/src/framework/domain/metadomain.h @@ -39,6 +39,7 @@ #include "framework/domain/domain.h" #include "framework/domain/mesh.h" #include "framework/parameters/parameters.h" +#include "output/render/renderer.h" #include "output/stats.h" #if defined(MPI_ENABLED) @@ -47,7 +48,6 @@ #if defined(OUTPUT_ENABLED) #include "output/checkpoint.h" - #include "output/render/renderer.h" #include "output/writer.h" #include @@ -191,14 +191,12 @@ namespace ntt { void ContinueFromCheckpoint(adios2::ADIOS*, const SimulationParams&); void redecomposeFromCheckpoint(const std::vector>&, const std::vector>&); +#endif - /* in-situ renderer (3D volume ray-march & 2D slice; metadomain_render.cpp) */ + /* in-situ renderer (3D volume ray-march & 2D slice) */ void InitRenderer(const SimulationParams&); - auto Render(const SimulationParams&, - timestep_t, - timestep_t, - simtime_t, - simtime_t) -> bool; + auto Render(const SimulationParams&, timestep_t, timestep_t, simtime_t, simtime_t) + -> bool; // Prepare the scalar named by a scene's `field` into bckp(:, 0) (active // cells synced; ghosts not yet halo-filled). Shared by the 2D and 3D render // paths so the field grammar (moments, T/V components, |E,B,J|, species @@ -208,7 +206,6 @@ namespace ntt { Domain&, const std::string& field_name, ndfield_t&) const -> bool; -#endif using custom_stats_output_t = std::function< real_t(const std::string&, timestep_t, simtime_t, const Domain&)>; @@ -339,8 +336,8 @@ namespace ntt { #if defined(OUTPUT_ENABLED) out::Writer g_writer; checkpoint::Writer g_checkpoint_writer; - out::Renderer g_renderer; #endif + out::Renderer g_renderer; #if defined(MPI_ENABLED) int g_mpi_rank { -1 }, g_mpi_size { -1 }; diff --git a/src/output/CMakeLists.txt b/src/output/CMakeLists.txt index f0a490d5e..96af8abb4 100644 --- a/src/output/CMakeLists.txt +++ b/src/output/CMakeLists.txt @@ -8,6 +8,10 @@ # * fields.cpp # * stats.cpp # * utils/interpret_prompt.cpp +# * utils/writers.cpp +# * utils/readers.cpp +# * utils/tuning.cpp +# * render/renderer.cpp # # @includes: # @@ -26,15 +30,18 @@ set(SRC_DIR ${CMAKE_CURRENT_SOURCE_DIR}) -set(SOURCES ${SRC_DIR}/stats.cpp ${SRC_DIR}/fields.cpp - ${SRC_DIR}/utils/interpret_prompt.cpp) +set(SOURCES + ${SRC_DIR}/stats.cpp ${SRC_DIR}/fields.cpp + ${SRC_DIR}/utils/interpret_prompt.cpp ${SRC_DIR}/render/renderer.cpp) if(${output}) - list(APPEND SOURCES ${SRC_DIR}/writer.cpp) - list(APPEND SOURCES ${SRC_DIR}/checkpoint.cpp) - list(APPEND SOURCES ${SRC_DIR}/render/renderer.cpp) - list(APPEND SOURCES ${SRC_DIR}/utils/writers.cpp) - list(APPEND SOURCES ${SRC_DIR}/utils/readers.cpp) - list(APPEND SOURCES ${SRC_DIR}/utils/tuning.cpp) + list( + APPEND + SOURCES + ${SRC_DIR}/writer.cpp + ${SRC_DIR}/checkpoint.cpp + ${SRC_DIR}/utils/writers.cpp + ${SRC_DIR}/utils/readers.cpp + ${SRC_DIR}/utils/tuning.cpp) endif() add_library(ntt_output ${SOURCES}) diff --git a/src/output/render/axes.h b/src/output/render/axes.h index 72da984bc..1142c81be 100644 --- a/src/output/render/axes.h +++ b/src/output/render/axes.h @@ -44,10 +44,10 @@ namespace out { return; } const std::size_t i = (static_cast(y) * CW + x) * 4; - b[i + 0] = c; - b[i + 1] = c; - b[i + 2] = c; - b[i + 3] = 255; + b[i + 0] = c; + b[i + 1] = c; + b[i + 2] = c; + b[i + 3] = 255; } inline void thickPx(uint8_t* b, int CW, int CH, int x, int y, int t, uint8_t c) { @@ -59,8 +59,15 @@ namespace out { } // Bresenham line, thickness (2t+1) - inline void line(uint8_t* b, int CW, int CH, int x0, int y0, int x1, int y1, - int t, uint8_t c) { + inline void line(uint8_t* b, + int CW, + int CH, + int x0, + int y0, + int x1, + int y1, + int t, + uint8_t c) { int dx = std::abs(x1 - x0), sx = (x0 < x1) ? 1 : -1; int dy = -std::abs(y1 - y0), sy = (y0 < y1) ? 1 : -1; int err = dx + dy; @@ -81,8 +88,14 @@ namespace out { } } - inline void text(uint8_t* b, int CW, int CH, int x, int y, - const std::string& str, int s, uint8_t c) { + inline void text(uint8_t* b, + int CW, + int CH, + int x, + int y, + const std::string& str, + int s, + uint8_t c) { int cx = x; for (const char ch : str) { const uint8_t* gl = cbar_hidden::glyph(ch); @@ -102,15 +115,22 @@ namespace out { } // Rotated bitmap text: the baseline advances along unit (ax, ay); (ox, oy) - // is the text-local origin (top-left of the first glyph). Each glyph cell is - // oversampled 2x so rotation leaves no gaps. - inline void textRot(uint8_t* b, int CW, int CH, real_t ox, real_t oy, - const std::string& str, int s, real_t ax, real_t ay, - uint8_t c) { + // is the text-local origin (top-left of the first glyph). Each glyph cell + // is oversampled 2x so rotation leaves no gaps. + inline void textRot(uint8_t* b, + int CW, + int CH, + real_t ox, + real_t oy, + const std::string& str, + int s, + real_t ax, + real_t ay, + uint8_t c) { const real_t dnx = -ay, dny = ax; // glyph "down" (perp. to baseline) for (std::size_t ci = 0; ci < str.size(); ++ci) { - const uint8_t* gl = cbar_hidden::glyph(str[ci]); - const real_t base = static_cast(ci) * 6 * s; + const uint8_t* gl = cbar_hidden::glyph(str[ci]); + const real_t base = static_cast(ci) * static_cast(6 * s); for (int row = 0; row < 7; ++row) { for (int col = 0; col < 5; ++col) { if (not(gl[row] & (1u << (4 - col)))) { @@ -118,11 +138,16 @@ namespace out { } for (int sy = 0; sy < 2 * s; ++sy) { for (int sx = 0; sx < 2 * s; ++sx) { - const real_t u = base + col * s + static_cast(sx) * HALF; - const real_t v = row * s + static_cast(sy) * HALF; - px(b, CW, CH, + const real_t u = base + static_cast(col * s) + + static_cast(sx) * HALF; + const real_t v = static_cast(row * s) + + static_cast(sy) * HALF; + px(b, + CW, + CH, static_cast(std::lround(ox + u * ax + v * dnx)), - static_cast(std::lround(oy + u * ay + v * dny)), c); + static_cast(std::lround(oy + u * ay + v * dny)), + c); } } } @@ -131,20 +156,36 @@ namespace out { } // rotated text centered on (cxp, cyp), baseline along unit (ax, ay) - inline void textRotCentered(uint8_t* b, int CW, int CH, real_t cxp, - real_t cyp, const std::string& str, int s, - real_t ax, real_t ay, uint8_t c) { + inline void textRotCentered(uint8_t* b, + int CW, + int CH, + real_t cxp, + real_t cyp, + const std::string& str, + int s, + real_t ax, + real_t ay, + uint8_t c) { const real_t dnx = -ay, dny = ax; - const real_t w = static_cast(str.size()) * 6 * s; - const real_t h = 7 * s; - textRot(b, CW, CH, cxp - HALF * w * ax - HALF * h * dnx, - cyp - HALF * w * ay - HALF * h * dny, str, s, ax, ay, c); + const real_t w = static_cast(str.size() * 6 * s); + const real_t h = static_cast(7 * s); + textRot(b, + CW, + CH, + cxp - HALF * w * ax - HALF * h * dnx, + cyp - HALF * w * ay - HALF * h * dny, + str, + s, + ax, + ay, + c); } // orient a screen-space edge direction so text reads naturally (rightward // for near-horizontal edges, upward for near-vertical ones) inline void readableDir(real_t& ax, real_t& ay) { - const real_t n = std::sqrt(static_cast(ax * ax + ay * ay)); + const real_t n = static_cast( + std::sqrt(static_cast(ax * ax + ay * ay))); if (n < static_cast(1e-9)) { ax = ONE; ay = ZERO; @@ -164,8 +205,14 @@ namespace out { } // vertical stack of characters (top to bottom), used for the y-axis name - inline void textVert(uint8_t* b, int CW, int CH, int x, int y, - const std::string& str, int s, uint8_t c) { + inline void textVert(uint8_t* b, + int CW, + int CH, + int x, + int y, + const std::string& str, + int s, + uint8_t c) { int cy = y; for (const char ch : str) { const std::string one(1, ch); @@ -196,16 +243,14 @@ namespace out { if (round) { nf = (f < static_cast(1.5)) ? ONE - : ((f < static_cast(3)) ? static_cast(2) - : (f < static_cast(7)) - ? static_cast(5) - : static_cast(10)); + : ((f < static_cast(3)) ? static_cast(2) + : (f < static_cast(7)) ? static_cast(5) + : static_cast(10)); } else { - nf = (f <= ONE) ? ONE - : (f <= static_cast(2)) - ? static_cast(2) - : (f <= static_cast(5)) ? static_cast(5) - : static_cast(10); + nf = (f <= ONE) ? ONE + : (f <= static_cast(2)) ? static_cast(2) + : (f <= static_cast(5)) ? static_cast(5) + : static_cast(10); } return nf * static_cast(std::pow(10.0, e)); } @@ -225,7 +270,7 @@ namespace out { std::string s; if (n < 0) { s += "-"; - n = -n; + n = -n; } if (n != 1) { s += std::to_string(n); @@ -245,21 +290,22 @@ namespace out { if (not(range > ZERO) or nticks < 2) { return out; } - const real_t ideal = static_cast(nticks - 1) * PI / range; - const int Ds[] = { 1, 2, 3, 4, 6, 8, 12, 16, 24 }; - int D = 4; - real_t bestd = static_cast(1e30); + const real_t ideal = static_cast(nticks - 1) * PI / range; + const int Ds[] = { 1, 2, 3, 4, 6, 8, 12, 16, 24 }; + int D = 4; + real_t bestd = static_cast(1e30); for (const int dd : Ds) { - const real_t df = std::fabs(static_cast(dd) - ideal); + const real_t df = static_cast( + std::fabs(static_cast(dd) - ideal)); if (df < bestd) { bestd = df; D = dd; } } const real_t step = PI / static_cast(D); - const int k0 = static_cast(std::ceil(static_cast(lo / step) - - 1e-6)); - const int k1 = static_cast( + const int k0 = static_cast( + std::ceil(static_cast(lo / step) - 1e-6)); + const int k1 = static_cast( std::floor(static_cast(hi / step) + 1e-6)); for (int k = k0; k <= k1; ++k) { int n = k, d = D; @@ -282,7 +328,8 @@ namespace out { if (step <= ZERO) { return out; } - const real_t g0 = std::ceil(static_cast(lo / step)) * step; + const real_t g0 = static_cast( + std::ceil(static_cast(lo / step)) * step); const real_t eps = static_cast(1e-6) * step; for (real_t v = g0; v <= hi + static_cast(0.5) * step; v += step) { if (v >= lo - eps and v <= hi + eps) { @@ -352,22 +399,26 @@ namespace out { const int th = std::max(0, s / 2 - 1); // spine half-thickness // [u0,u1]x[v0,v1] is the world window mapped onto the full data region; - // [du0,du1]x[dv0,dv1] is the actual data box (the domain/region), a sub-rect - // when the window was aspect-expanded. The spine + ticks clamp to the DATA - // box so the empty aspect pad stays outside the frame. + // [du0,du1]x[dv0,dv1] is the actual data box (the domain/region), a + // sub-rect when the window was aspect-expanded. The spine + ticks clamp to + // the DATA box so the empty aspect pad stays outside the frame. const int xL = x0, xR = x0 + W - 1, yT = 0, yB = H - 1; - auto X = [&](real_t u) -> int { - return static_cast(std::lround( - static_cast(xL) + - static_cast((u - u0) / (u1 - u0)) * (xR - xL))); + auto X = [&](real_t u) -> int { + return static_cast( + std::lround(static_cast(xL) + + static_cast((u - u0) / (u1 - u0)) * (xR - xL))); }; auto Y = [&](real_t v) -> int { - return static_cast(std::lround( - static_cast(yB) - - static_cast((v - v0) / (v1 - v0)) * (yB - yT))); + return static_cast( + std::lround(static_cast(yB) - + static_cast((v - v0) / (v1 - v0)) * (yB - yT))); + }; + auto clampX = [&](int x) { + return (x < xL) ? xL : ((x > xR) ? xR : x); + }; + auto clampY = [&](int y) { + return (y < yT) ? yT : ((y > yB) ? yB : y); }; - auto clampX = [&](int x) { return (x < xL) ? xL : ((x > xR) ? xR : x); }; - auto clampY = [&](int y) { return (y < yT) ? yT : ((y > yB) ? yB : y); }; const int xLd = clampX(X(du0)), xRd = clampX(X(du1)); const int yTd = clampY(Y(dv1)), yBd = clampY(Y(dv0)); // dv1 = top @@ -399,21 +450,31 @@ namespace out { } // axis names if (not xlabel.empty()) { - text(rgba, CW, CH, (xLd + xRd) / 2 - textW(xlabel, s) / 2, - yBd + tl + gap + ch + gap, xlabel, s, c); + text(rgba, + CW, + CH, + (xLd + xRd) / 2 - textW(xlabel, s) / 2, + yBd + tl + gap + ch + gap, + xlabel, + s, + c); } if (not ylabel.empty()) { - textVert(rgba, CW, CH, + textVert(rgba, + CW, + CH, std::max(gap, xLd - tl - gap - 7 * (6 * s) - gap - 6 * s), (yTd + yBd) / 2 - 4 * s * static_cast(ylabel.size()) / 2, - ylabel, s, c); + ylabel, + s, + c); } } /** * @brief Draw polar (curvilinear) axes for a 2D spherical meridional slice. - * @param x0,W,H data region (the slice maps world (X = r sin th, Z = r cos th) - * onto it via the [u0,u1]x[v0,v1] window, aspect-matched) + * @param x0,W,H data region (the slice maps world (X = r sin th, Z = r cos + * th) onto it via the [u0,u1]x[v0,v1] window, aspect-matched) * @param rmin,rmax,tmin,tmax global (r, theta) extent * @param mirror whether the half-plane is mirrored into a full disk * @param rlabel,tlabel names for the radial / angular axes (e.g. "R","Theta") @@ -449,26 +510,31 @@ namespace out { const int gap = 2 * s; auto WX = [&](real_t X) -> real_t { - return static_cast(x0) + (X - u0) / (u1 - u0) * W - HALF; + return static_cast(x0) + + (X - u0) / (u1 - u0) * static_cast(W) - HALF; }; auto WZ = [&](real_t Z) -> real_t { - return (v1 - Z) / (v1 - v0) * H - HALF; + return (v1 - Z) / (v1 - v0) * static_cast(H) - HALF; }; auto PX = [&](real_t X, real_t Z, int& qx, int& qy) { qx = static_cast(std::lround(WX(X))); qy = static_cast(std::lround(WZ(Z))); }; - const int NA = 160; + const int NA = 160; auto arc = [&](real_t r, real_t sgn) { int qx, qy; - PX(sgn * r * std::sin(static_cast(tmin)), - r * std::cos(static_cast(tmin)), qx, qy); + PX(static_cast(sgn * r * std::sin(static_cast(tmin))), + static_cast(r * std::cos(static_cast(tmin))), + qx, + qy); for (int i = 1; i <= NA; ++i) { - const real_t th = tmin + (tmax - tmin) * i / NA; + const real_t th = tmin + (tmax - tmin) * static_cast(i) / NA; int rx, ry; - PX(sgn * r * std::sin(static_cast(th)), - r * std::cos(static_cast(th)), rx, ry); + PX(static_cast(sgn * r * std::sin(static_cast(th))), + static_cast(r * std::cos(static_cast(th))), + rx, + ry); line(rgba, CW, CH, qx, qy, rx, ry, 0, c); qx = rx; qy = ry; @@ -476,10 +542,14 @@ namespace out { }; auto ray = [&](real_t th) { int ax, ay, bx, by; - PX(rmin * std::sin(static_cast(th)), - rmin * std::cos(static_cast(th)), ax, ay); - PX(rmax * std::sin(static_cast(th)), - rmax * std::cos(static_cast(th)), bx, by); + PX(static_cast(rmin * std::sin(static_cast(th))), + static_cast(rmin * std::cos(static_cast(th))), + ax, + ay); + PX(static_cast(rmax * std::sin(static_cast(th))), + static_cast(rmax * std::cos(static_cast(th))), + bx, + by); line(rgba, CW, CH, ax, ay, bx, by, 0, c); }; @@ -511,52 +581,86 @@ namespace out { if (not rlabel.empty()) { int ax, ay; PX(ZERO, ZERO, ax, ay); - const real_t off = tl + gap + rmaxlabW + gap + ch; - textRotCentered(rgba, CW, CH, ax - off, static_cast(ay), rlabel, s, - ZERO, -ONE, c); + const real_t off = static_cast(tl + gap + rmaxlabW + gap + ch); + textRotCentered(rgba, + CW, + CH, + static_cast(ax) - off, + static_cast(ay), + rlabel, + s, + ZERO, + -ONE, + c); } // ---- Theta axis: ticks + labels (fractions of pi) along the arc --- // // widest tick label, so the "Theta" name can clear them all real_t maxlw = static_cast(ch); for (const auto& tk : piTicks(tmin, tmax, nticks)) { - maxlw = std::max(maxlw, - static_cast(textW(fmtPi(tk.n, tk.d), s))); + maxlw = std::max(maxlw, static_cast(textW(fmtPi(tk.n, tk.d), s))); } for (const auto& tk : piTicks(tmin, tmax, nticks)) { - const real_t ox = std::sin(static_cast(tk.val)); - const real_t oz = std::cos(static_cast(tk.val)); - int px0, py0; + const real_t ox = static_cast(std::sin(static_cast(tk.val))); + const real_t oz = static_cast(std::cos(static_cast(tk.val))); + int px0, py0; PX(rmax * ox, rmax * oz, px0, py0); const real_t dxp = ox, dyp = -oz; // outward pixel direction - line(rgba, CW, CH, px0, py0, - static_cast(std::lround(px0 + dxp * tl)), - static_cast(std::lround(py0 + dyp * tl)), 0, c); - const std::string lab = fmtPi(tk.n, tk.d); + line(rgba, + CW, + CH, + px0, + py0, + static_cast(std::lround( + static_cast(px0) + dxp * static_cast(tl))), + static_cast(std::lround( + static_cast(py0) + dyp * static_cast(tl))), + 0, + c); + const std::string lab = fmtPi(tk.n, tk.d); // push the (horizontal) label box fully clear of the arc/tick at any // angle: offset its center by its own support along the outward direction - const real_t inset = HALF * (static_cast(textW(lab, s)) * - std::fabs(static_cast(dxp)) + + const real_t inset = HALF * (static_cast(textW(lab, s)) * + static_cast( + std::fabs(static_cast(dxp))) + static_cast(ch) * - std::fabs(static_cast(dyp))); - const real_t lo = tl + gap + inset; - text(rgba, CW, CH, - static_cast(std::lround(px0 + dxp * lo)) - textW(lab, s) / 2, - static_cast(std::lround(py0 + dyp * lo)) - ch / 2, lab, s, c); + static_cast( + std::fabs(static_cast(dyp)))); + const real_t lo = static_cast(tl + gap) + inset; + text(rgba, + CW, + CH, + static_cast(std::lround(static_cast(px0) + dxp * lo)) - + textW(lab, s) / 2, + static_cast(std::lround(static_cast(py0) + dyp * lo)) - + ch / 2, + lab, + s, + c); } if (not tlabel.empty()) { const real_t tm = HALF * (tmin + tmax); - const real_t ox = std::sin(static_cast(tm)); - const real_t oz = std::cos(static_cast(tm)); + const real_t ox = static_cast(std::sin(static_cast(tm))); + const real_t oz = static_cast(std::cos(static_cast(tm))); int px0, py0; PX(rmax * ox, rmax * oz, px0, py0); - real_t adx = std::cos(static_cast(tm)); // arc tangent - real_t ady = std::sin(static_cast(tm)); + real_t adx = static_cast( + std::cos(static_cast(tm))); // arc tangent + real_t ady = static_cast(std::sin(static_cast(tm))); readableDir(adx, ady); // beyond the tick labels (which reach ~tl+gap+maxlw from the arc) - const real_t off = tl + gap + maxlw + gap + ch; - textRotCentered(rgba, CW, CH, px0 + ox * off, py0 - oz * off, tlabel, s, - adx, ady, c); + const real_t off = static_cast(tl + gap) + maxlw + + static_cast(gap + ch); + textRotCentered(rgba, + CW, + CH, + static_cast(px0) + ox * off, + static_cast(py0) - oz * off, + tlabel, + s, + adx, + ady, + c); } } @@ -629,23 +733,23 @@ namespace out { // the conventional, view-independent place for axis annotation. auto frontFace = [&](int axis, int side) -> bool { const real_t nrm = (side != 0) ? ONE : -ONE; // outward normal sign - return (nrm * (-cam.forward[axis])) > ZERO; // points toward the camera? + return (nrm * (-cam.forward[axis])) > ZERO; // points toward the camera? }; for (int d = 0; d < 3; ++d) { - const int e1 = (d == 0) ? 1 : 0; - const int e2 = (d == 2) ? 1 : 2; + const int e1 = (d == 0) ? 1 : 0; + const int e2 = (d == 2) ? 1 : 2; int bs1 = 0, bs2 = 0; bool found = false; - real_t best = ZERO; + real_t best = ZERO; for (int s1 = 0; s1 < 2; ++s1) { for (int s2 = 0; s2 < 2; ++s2) { - const int m0 = (s1 << e1) | (s2 << e2); - const int m1 = m0 | (1 << d); - const real_t mx = HALF * (cx[m0] + cx[m1]); - const real_t my = HALF * (cy[m0] + cy[m1]); + const int m0 = (s1 << e1) | (s2 << e2); + const int m1 = m0 | (1 << d); + const real_t mx = HALF * (cx[m0] + cx[m1]); + const real_t my = HALF * (cy[m0] + cy[m1]); // prefer the foreground (bottom-left) edge: larger pixel-y is lower, - // smaller pixel-x is further left. Axis-independent, so it follows the - // camera instead of assuming the default diagonal view. + // smaller pixel-x is further left. Axis-independent, so it follows + // the camera instead of assuming the default diagonal view. real_t score = my - mx; if (frontFace(e1, s1) != frontFace(e2, s2)) { score += static_cast(1e6); // strongly prefer silhouette edges @@ -664,13 +768,14 @@ namespace out { corner(m0, o); // perpendicular coords fixed; axis d swept for ticks const real_t lo = ext[d].first, hi = ext[d].second; // screen-space perpendicular to the edge, flipped to point outward - real_t ex = cx[m1] - cx[m0], ey = cy[m1] - cy[m0]; - real_t el = std::sqrt(static_cast(ex * ex + ey * ey)); + real_t ex = cx[m1] - cx[m0], ey = cy[m1] - cy[m0]; + real_t el = static_cast( + std::sqrt(static_cast(ex * ex + ey * ey))); if (el < ONE) { el = ONE; } - ex /= el; - ey /= el; + ex /= el; + ey /= el; real_t pxd = -ey, pyd = ex; { const real_t mxv = HALF * (cx[m0] + cx[m1]) - ccx; @@ -683,8 +788,10 @@ namespace out { // readable text baseline aligned with the edge direction real_t adx = ex, ady = ey; readableDir(adx, ady); - const real_t numOff = static_cast(tl) + 5 * s; // number center - const real_t nameOff = static_cast(tl) + 14 * s; // axis-name center + const real_t numOff = static_cast(tl) + + static_cast(5 * s); // number center + const real_t nameOff = static_cast(tl) + + static_cast(14 * s); // axis-name center // ticks + numeric labels (numbers rotated along the edge for x & y; the // vertical z edge keeps horizontal numbers, which read more easily) for (const real_t tv : niceTicks(lo, hi, nticks)) { @@ -694,18 +801,35 @@ namespace out { if (not projectToScreen(cam, W, H, p, a, b)) { continue; } - a += static_cast(x0); - const int mx = static_cast(std::lround(a + pxd * tl)); - const int my = static_cast(std::lround(b + pyd * tl)); - line(rgba, CW, CH, static_cast(std::lround(a)), - static_cast(std::lround(b)), mx, my, 0, c); + a += static_cast(x0); + const int mx = static_cast( + std::lround(a + pxd * static_cast(tl))); + const int my = static_cast( + std::lround(b + pyd * static_cast(tl))); + line(rgba, + CW, + CH, + static_cast(std::lround(a)), + static_cast(std::lround(b)), + mx, + my, + 0, + c); const std::string l2 = cbar_hidden::fmtNum(tv); if (d == 2) { const int tx = (pxd < ZERO) ? (mx - textW(l2, s)) : mx; text(rgba, CW, CH, tx, my - ch / 2, l2, s, c); } else { - textRotCentered(rgba, CW, CH, a + pxd * numOff, b + pyd * numOff, l2, - s, adx, ady, c); + textRotCentered(rgba, + CW, + CH, + a + pxd * numOff, + b + pyd * numOff, + l2, + s, + adx, + ady, + c); } } // axis name at the MIDDLE of the edge (near the central tick), aligned @@ -716,8 +840,16 @@ namespace out { real_t a, b; if (projectToScreen(cam, W, H, mid, a, b)) { a += static_cast(x0); - textRotCentered(rgba, CW, CH, a + pxd * nameOff, b + pyd * nameOff, - lab[d], s, adx, ady, c); + textRotCentered(rgba, + CW, + CH, + a + pxd * nameOff, + b + pyd * nameOff, + lab[d], + s, + adx, + ady, + c); } } } diff --git a/src/output/render/colorbar.h b/src/output/render/colorbar.h index 0ff9f085f..7c01742e0 100644 --- a/src/output/render/colorbar.h +++ b/src/output/render/colorbar.h @@ -40,67 +40,256 @@ namespace out { c = static_cast(c - 'a' + 'A'); } switch (c) { - case '0': { static const uint8_t g[7] = { 0b01110, 0b10001, 0b10011, 0b10101, 0b11001, 0b10001, 0b01110 }; return g; } - case '1': { static const uint8_t g[7] = { 0b00100, 0b01100, 0b00100, 0b00100, 0b00100, 0b00100, 0b01110 }; return g; } - case '2': { static const uint8_t g[7] = { 0b01110, 0b10001, 0b00001, 0b00010, 0b00100, 0b01000, 0b11111 }; return g; } - case '3': { static const uint8_t g[7] = { 0b11111, 0b00010, 0b00100, 0b00010, 0b00001, 0b10001, 0b01110 }; return g; } - case '4': { static const uint8_t g[7] = { 0b00010, 0b00110, 0b01010, 0b10010, 0b11111, 0b00010, 0b00010 }; return g; } - case '5': { static const uint8_t g[7] = { 0b11111, 0b10000, 0b11110, 0b00001, 0b00001, 0b10001, 0b01110 }; return g; } - case '6': { static const uint8_t g[7] = { 0b00110, 0b01000, 0b10000, 0b11110, 0b10001, 0b10001, 0b01110 }; return g; } - case '7': { static const uint8_t g[7] = { 0b11111, 0b00001, 0b00010, 0b00100, 0b01000, 0b01000, 0b01000 }; return g; } - case '8': { static const uint8_t g[7] = { 0b01110, 0b10001, 0b10001, 0b01110, 0b10001, 0b10001, 0b01110 }; return g; } - case '9': { static const uint8_t g[7] = { 0b01110, 0b10001, 0b10001, 0b01111, 0b00001, 0b00010, 0b01100 }; return g; } - case '.': { static const uint8_t g[7] = { 0b00000, 0b00000, 0b00000, 0b00000, 0b00000, 0b00110, 0b00110 }; return g; } - case '-': { static const uint8_t g[7] = { 0b00000, 0b00000, 0b00000, 0b11111, 0b00000, 0b00000, 0b00000 }; return g; } - case '+': { static const uint8_t g[7] = { 0b00000, 0b00100, 0b00100, 0b11111, 0b00100, 0b00100, 0b00000 }; return g; } - case '=': { static const uint8_t g[7] = { 0b00000, 0b00000, 0b11111, 0b00000, 0b11111, 0b00000, 0b00000 }; return g; } - case '_': { static const uint8_t g[7] = { 0b00000, 0b00000, 0b00000, 0b00000, 0b00000, 0b00000, 0b11111 }; return g; } - case ':': { static const uint8_t g[7] = { 0b00000, 0b00110, 0b00110, 0b00000, 0b00110, 0b00110, 0b00000 }; return g; } - case '/': { static const uint8_t g[7] = { 0b00001, 0b00010, 0b00010, 0b00100, 0b01000, 0b01000, 0b10000 }; return g; } - case 'A': { static const uint8_t g[7] = { 0b01110, 0b10001, 0b10001, 0b11111, 0b10001, 0b10001, 0b10001 }; return g; } - case 'B': { static const uint8_t g[7] = { 0b11110, 0b10001, 0b10001, 0b11110, 0b10001, 0b10001, 0b11110 }; return g; } - case 'C': { static const uint8_t g[7] = { 0b01110, 0b10001, 0b10000, 0b10000, 0b10000, 0b10001, 0b01110 }; return g; } - case 'D': { static const uint8_t g[7] = { 0b11100, 0b10010, 0b10001, 0b10001, 0b10001, 0b10010, 0b11100 }; return g; } - case 'E': { static const uint8_t g[7] = { 0b11111, 0b10000, 0b10000, 0b11110, 0b10000, 0b10000, 0b11111 }; return g; } - case 'F': { static const uint8_t g[7] = { 0b11111, 0b10000, 0b10000, 0b11110, 0b10000, 0b10000, 0b10000 }; return g; } - case 'G': { static const uint8_t g[7] = { 0b01110, 0b10001, 0b10000, 0b10111, 0b10001, 0b10001, 0b01111 }; return g; } - case 'H': { static const uint8_t g[7] = { 0b10001, 0b10001, 0b10001, 0b11111, 0b10001, 0b10001, 0b10001 }; return g; } - case 'I': { static const uint8_t g[7] = { 0b01110, 0b00100, 0b00100, 0b00100, 0b00100, 0b00100, 0b01110 }; return g; } - case 'J': { static const uint8_t g[7] = { 0b00111, 0b00010, 0b00010, 0b00010, 0b10010, 0b10010, 0b01100 }; return g; } - case 'K': { static const uint8_t g[7] = { 0b10001, 0b10010, 0b10100, 0b11000, 0b10100, 0b10010, 0b10001 }; return g; } - case 'L': { static const uint8_t g[7] = { 0b10000, 0b10000, 0b10000, 0b10000, 0b10000, 0b10000, 0b11111 }; return g; } - case 'M': { static const uint8_t g[7] = { 0b10001, 0b11011, 0b10101, 0b10101, 0b10001, 0b10001, 0b10001 }; return g; } - case 'N': { static const uint8_t g[7] = { 0b10001, 0b11001, 0b10101, 0b10011, 0b10001, 0b10001, 0b10001 }; return g; } - case 'O': { static const uint8_t g[7] = { 0b01110, 0b10001, 0b10001, 0b10001, 0b10001, 0b10001, 0b01110 }; return g; } - case 'P': { static const uint8_t g[7] = { 0b11110, 0b10001, 0b10001, 0b11110, 0b10000, 0b10000, 0b10000 }; return g; } - case 'Q': { static const uint8_t g[7] = { 0b01110, 0b10001, 0b10001, 0b10001, 0b10101, 0b10010, 0b01101 }; return g; } - case 'R': { static const uint8_t g[7] = { 0b11110, 0b10001, 0b10001, 0b11110, 0b10100, 0b10010, 0b10001 }; return g; } - case 'S': { static const uint8_t g[7] = { 0b01111, 0b10000, 0b10000, 0b01110, 0b00001, 0b00001, 0b11110 }; return g; } - case 'T': { static const uint8_t g[7] = { 0b11111, 0b00100, 0b00100, 0b00100, 0b00100, 0b00100, 0b00100 }; return g; } - case 'U': { static const uint8_t g[7] = { 0b10001, 0b10001, 0b10001, 0b10001, 0b10001, 0b10001, 0b01110 }; return g; } - case 'V': { static const uint8_t g[7] = { 0b10001, 0b10001, 0b10001, 0b10001, 0b10001, 0b01010, 0b00100 }; return g; } - case 'W': { static const uint8_t g[7] = { 0b10001, 0b10001, 0b10001, 0b10101, 0b10101, 0b11011, 0b10001 }; return g; } - case 'X': { static const uint8_t g[7] = { 0b10001, 0b10001, 0b01010, 0b00100, 0b01010, 0b10001, 0b10001 }; return g; } - case 'Y': { static const uint8_t g[7] = { 0b10001, 0b10001, 0b01010, 0b00100, 0b00100, 0b00100, 0b00100 }; return g; } - case 'Z': { static const uint8_t g[7] = { 0b11111, 0b00001, 0b00010, 0b00100, 0b01000, 0b10000, 0b11111 }; return g; } - default: { static const uint8_t g[7] = { 0, 0, 0, 0, 0, 0, 0 }; return g; } // blank + case '0': { + static const uint8_t g[7] = { 0b01110, 0b10001, 0b10011, 0b10101, + 0b11001, 0b10001, 0b01110 }; + return g; + } + case '1': { + static const uint8_t g[7] = { 0b00100, 0b01100, 0b00100, 0b00100, + 0b00100, 0b00100, 0b01110 }; + return g; + } + case '2': { + static const uint8_t g[7] = { 0b01110, 0b10001, 0b00001, 0b00010, + 0b00100, 0b01000, 0b11111 }; + return g; + } + case '3': { + static const uint8_t g[7] = { 0b11111, 0b00010, 0b00100, 0b00010, + 0b00001, 0b10001, 0b01110 }; + return g; + } + case '4': { + static const uint8_t g[7] = { 0b00010, 0b00110, 0b01010, 0b10010, + 0b11111, 0b00010, 0b00010 }; + return g; + } + case '5': { + static const uint8_t g[7] = { 0b11111, 0b10000, 0b11110, 0b00001, + 0b00001, 0b10001, 0b01110 }; + return g; + } + case '6': { + static const uint8_t g[7] = { 0b00110, 0b01000, 0b10000, 0b11110, + 0b10001, 0b10001, 0b01110 }; + return g; + } + case '7': { + static const uint8_t g[7] = { 0b11111, 0b00001, 0b00010, 0b00100, + 0b01000, 0b01000, 0b01000 }; + return g; + } + case '8': { + static const uint8_t g[7] = { 0b01110, 0b10001, 0b10001, 0b01110, + 0b10001, 0b10001, 0b01110 }; + return g; + } + case '9': { + static const uint8_t g[7] = { 0b01110, 0b10001, 0b10001, 0b01111, + 0b00001, 0b00010, 0b01100 }; + return g; + } + case '.': { + static const uint8_t g[7] = { 0b00000, 0b00000, 0b00000, 0b00000, + 0b00000, 0b00110, 0b00110 }; + return g; + } + case '-': { + static const uint8_t g[7] = { 0b00000, 0b00000, 0b00000, 0b11111, + 0b00000, 0b00000, 0b00000 }; + return g; + } + case '+': { + static const uint8_t g[7] = { 0b00000, 0b00100, 0b00100, 0b11111, + 0b00100, 0b00100, 0b00000 }; + return g; + } + case '=': { + static const uint8_t g[7] = { 0b00000, 0b00000, 0b11111, 0b00000, + 0b11111, 0b00000, 0b00000 }; + return g; + } + case '_': { + static const uint8_t g[7] = { 0b00000, 0b00000, 0b00000, 0b00000, + 0b00000, 0b00000, 0b11111 }; + return g; + } + case ':': { + static const uint8_t g[7] = { 0b00000, 0b00110, 0b00110, 0b00000, + 0b00110, 0b00110, 0b00000 }; + return g; + } + case '/': { + static const uint8_t g[7] = { 0b00001, 0b00010, 0b00010, 0b00100, + 0b01000, 0b01000, 0b10000 }; + return g; + } + case 'A': { + static const uint8_t g[7] = { 0b01110, 0b10001, 0b10001, 0b11111, + 0b10001, 0b10001, 0b10001 }; + return g; + } + case 'B': { + static const uint8_t g[7] = { 0b11110, 0b10001, 0b10001, 0b11110, + 0b10001, 0b10001, 0b11110 }; + return g; + } + case 'C': { + static const uint8_t g[7] = { 0b01110, 0b10001, 0b10000, 0b10000, + 0b10000, 0b10001, 0b01110 }; + return g; + } + case 'D': { + static const uint8_t g[7] = { 0b11100, 0b10010, 0b10001, 0b10001, + 0b10001, 0b10010, 0b11100 }; + return g; + } + case 'E': { + static const uint8_t g[7] = { 0b11111, 0b10000, 0b10000, 0b11110, + 0b10000, 0b10000, 0b11111 }; + return g; + } + case 'F': { + static const uint8_t g[7] = { 0b11111, 0b10000, 0b10000, 0b11110, + 0b10000, 0b10000, 0b10000 }; + return g; + } + case 'G': { + static const uint8_t g[7] = { 0b01110, 0b10001, 0b10000, 0b10111, + 0b10001, 0b10001, 0b01111 }; + return g; + } + case 'H': { + static const uint8_t g[7] = { 0b10001, 0b10001, 0b10001, 0b11111, + 0b10001, 0b10001, 0b10001 }; + return g; + } + case 'I': { + static const uint8_t g[7] = { 0b01110, 0b00100, 0b00100, 0b00100, + 0b00100, 0b00100, 0b01110 }; + return g; + } + case 'J': { + static const uint8_t g[7] = { 0b00111, 0b00010, 0b00010, 0b00010, + 0b10010, 0b10010, 0b01100 }; + return g; + } + case 'K': { + static const uint8_t g[7] = { 0b10001, 0b10010, 0b10100, 0b11000, + 0b10100, 0b10010, 0b10001 }; + return g; + } + case 'L': { + static const uint8_t g[7] = { 0b10000, 0b10000, 0b10000, 0b10000, + 0b10000, 0b10000, 0b11111 }; + return g; + } + case 'M': { + static const uint8_t g[7] = { 0b10001, 0b11011, 0b10101, 0b10101, + 0b10001, 0b10001, 0b10001 }; + return g; + } + case 'N': { + static const uint8_t g[7] = { 0b10001, 0b11001, 0b10101, 0b10011, + 0b10001, 0b10001, 0b10001 }; + return g; + } + case 'O': { + static const uint8_t g[7] = { 0b01110, 0b10001, 0b10001, 0b10001, + 0b10001, 0b10001, 0b01110 }; + return g; + } + case 'P': { + static const uint8_t g[7] = { 0b11110, 0b10001, 0b10001, 0b11110, + 0b10000, 0b10000, 0b10000 }; + return g; + } + case 'Q': { + static const uint8_t g[7] = { 0b01110, 0b10001, 0b10001, 0b10001, + 0b10101, 0b10010, 0b01101 }; + return g; + } + case 'R': { + static const uint8_t g[7] = { 0b11110, 0b10001, 0b10001, 0b11110, + 0b10100, 0b10010, 0b10001 }; + return g; + } + case 'S': { + static const uint8_t g[7] = { 0b01111, 0b10000, 0b10000, 0b01110, + 0b00001, 0b00001, 0b11110 }; + return g; + } + case 'T': { + static const uint8_t g[7] = { 0b11111, 0b00100, 0b00100, 0b00100, + 0b00100, 0b00100, 0b00100 }; + return g; + } + case 'U': { + static const uint8_t g[7] = { 0b10001, 0b10001, 0b10001, 0b10001, + 0b10001, 0b10001, 0b01110 }; + return g; + } + case 'V': { + static const uint8_t g[7] = { 0b10001, 0b10001, 0b10001, 0b10001, + 0b10001, 0b01010, 0b00100 }; + return g; + } + case 'W': { + static const uint8_t g[7] = { 0b10001, 0b10001, 0b10001, 0b10101, + 0b10101, 0b11011, 0b10001 }; + return g; + } + case 'X': { + static const uint8_t g[7] = { 0b10001, 0b10001, 0b01010, 0b00100, + 0b01010, 0b10001, 0b10001 }; + return g; + } + case 'Y': { + static const uint8_t g[7] = { 0b10001, 0b10001, 0b01010, 0b00100, + 0b00100, 0b00100, 0b00100 }; + return g; + } + case 'Z': { + static const uint8_t g[7] = { 0b11111, 0b00001, 0b00010, 0b00100, + 0b01000, 0b10000, 0b11111 }; + return g; + } + default: { + static const uint8_t g[7] = { 0, 0, 0, 0, 0, 0, 0 }; + return g; + } // blank } } - inline void setPx(uint8_t* rgba, int W, int H, int x, int y, uint8_t r, - uint8_t g, uint8_t b) { + inline void setPx(uint8_t* rgba, + int W, + int H, + int x, + int y, + uint8_t r, + uint8_t g, + uint8_t b) { if (x < 0 or x >= W or y < 0 or y >= H) { return; } const std::size_t i = (static_cast(y) * W + x) * 4; - rgba[i + 0] = r; - rgba[i + 1] = g; - rgba[i + 2] = b; - rgba[i + 3] = 255; + rgba[i + 0] = r; + rgba[i + 1] = g; + rgba[i + 2] = b; + rgba[i + 3] = 255; } - inline void drawChar(uint8_t* rgba, int W, int H, int x, int y, char c, - int s, uint8_t r, uint8_t g, uint8_t b) { + inline void drawChar(uint8_t* rgba, + int W, + int H, + int x, + int y, + char c, + int s, + uint8_t r, + uint8_t g, + uint8_t b) { const uint8_t* gl = glyph(c); for (int row = 0; row < 7; ++row) { for (int col = 0; col < 5; ++col) { @@ -115,9 +304,16 @@ namespace out { } } - inline void drawText(uint8_t* rgba, int W, int H, int x, int y, - const std::string& str, int s, uint8_t r, uint8_t g, - uint8_t b) { + inline void drawText(uint8_t* rgba, + int W, + int H, + int x, + int y, + const std::string& str, + int s, + uint8_t r, + uint8_t g, + uint8_t b) { int cx = x; for (const char c : str) { drawChar(rgba, W, H, cx, y, c, s, r, g, b); @@ -204,18 +400,17 @@ namespace out { const int bar_y = aligned ? span_top : ((H - bar_h) / 2); // contrasting monochrome for text / frame / ticks - const real_t lum = static_cast(0.299) * bg[0] + + const real_t lum = static_cast(0.299) * bg[0] + static_cast(0.587) * bg[1] + static_cast(0.114) * bg[2]; - const uint8_t tc = (lum < HALF) ? 255 : 0; + const uint8_t tc = (lum < HALF) ? 255 : 0; // gradient strip (top = vmax, bottom = vmin) for (int j = 0; j < bar_h; ++j) { - const real_t u = (bar_h > 1) - ? ONE - static_cast(j) / - static_cast(bar_h - 1) - : ZERO; - real_t cr, cg, cb; + const real_t u = (bar_h > 1) ? ONE - static_cast(j) / + static_cast(bar_h - 1) + : ZERO; + real_t cr, cg, cb; colormapRGB(colormap, u, cr, cg, cb); const uint8_t R = quant(cr), G = quant(cg), B = quant(cb); for (int i = 0; i < bar_w; ++i) { @@ -243,12 +438,10 @@ namespace out { if (ticks.empty()) { const int nticks = 5; for (int t = 0; t < nticks; ++t) { - const real_t u = static_cast(t) / - static_cast(nticks - 1); - const real_t val = can_log - ? math::pow(static_cast(10), - lvmin + (lvmax - lvmin) * u) - : (vmin + (vmax - vmin) * u); + const real_t u = static_cast(t) / static_cast(nticks - 1); + const real_t val = can_log ? math::pow(static_cast(10), + lvmin + (lvmax - lvmin) * u) + : (vmin + (vmax - vmin) * u); tk.emplace_back(u, val); } } else { @@ -258,8 +451,7 @@ namespace out { continue; } const real_t u = (span != ZERO) - ? ((can_log ? (math::log10(v) - lvmin) : (v - vmin)) / - span) + ? ((can_log ? (math::log10(v) - lvmin) : (v - vmin)) / span) : ZERO; if (u < static_cast(-1e-4) or u > ONE + static_cast(1e-4)) { continue; // outside the colorbar range @@ -269,8 +461,7 @@ namespace out { } for (const auto& [u, val] : tk) { const int ty = bar_y + - static_cast((ONE - u) * - static_cast(bar_h - 1)); + static_cast((ONE - u) * static_cast(bar_h - 1)); // tick line for (int i = 0; i < gap; ++i) { for (int w = 0; w < std::max(1, s / 2); ++w) { diff --git a/src/output/render/composite.h b/src/output/render/composite.h index ecb14ced9..22145c9ac 100644 --- a/src/output/render/composite.h +++ b/src/output/render/composite.h @@ -47,15 +47,13 @@ namespace out { */ inline auto compositeOrderKey(const std::vector& offset, const std::vector& ndoms, - const real_t forward[3]) - -> uint64_t { + const real_t forward[3]) -> uint64_t { uint64_t key = 0; - for (std::size_t d = 0; d < ndoms.size(); ++d) { - const unsigned int Dd = ndoms[d]; - const unsigned int od = offset[d]; - const unsigned int kd = (forward[d] >= ZERO) ? od : (Dd - 1u - od); - key = key * static_cast(Dd) + - static_cast(kd); + for (size_t d = 0; d < ndoms.size(); ++d) { + const unsigned int Dd = ndoms[d]; + const unsigned int od = offset[d]; + const unsigned int kd = (forward[d] >= ZERO) ? od : (Dd - 1u - od); + key = key * static_cast(Dd) + static_cast(kd); } return key; } @@ -70,11 +68,11 @@ namespace out { * Associative with identity (0,0,0,0); segments must be supplied front first. */ inline void overComposite(real_t acc[4], const real_t seg[4]) { - const real_t one_minus_a = ONE - acc[3]; - acc[0] += one_minus_a * seg[0]; - acc[1] += one_minus_a * seg[1]; - acc[2] += one_minus_a * seg[2]; - acc[3] += one_minus_a * seg[3]; + const real_t one_minus_a = ONE - acc[3]; + acc[0] += one_minus_a * seg[0]; + acc[1] += one_minus_a * seg[1]; + acc[2] += one_minus_a * seg[2]; + acc[3] += one_minus_a * seg[3]; } /** @@ -133,7 +131,7 @@ namespace out { const real_t p[3] = { (c & 1) ? hi[0] : lo[0], (c & 2) ? hi[1] : lo[1], (c & 4) ? hi[2] : lo[2] }; - real_t sx, sy; + real_t sx, sy; if (not projectToScreen(cam, W, H, p, sx, sy)) { bx0 = 0; by0 = 0; @@ -151,14 +149,14 @@ namespace out { int x1 = static_cast(std::ceil(maxx)) + pad; int y0 = static_cast(std::floor(miny)) - pad; int y1 = static_cast(std::ceil(maxy)) + pad; - x0 = std::max(0, std::min(W, x0)); - x1 = std::max(0, std::min(W, x1)); - y0 = std::max(0, std::min(H, y0)); - y1 = std::max(0, std::min(H, y1)); - bx0 = x0; - by0 = y0; - bw = x1 - x0; - bh = y1 - y0; + x0 = std::max(0, std::min(W, x0)); + x1 = std::max(0, std::min(W, x1)); + y0 = std::max(0, std::min(H, y0)); + y1 = std::max(0, std::min(H, y1)); + bx0 = x0; + by0 = y0; + bw = x1 - x0; + bh = y1 - y0; return (bw > 0 and bh > 0); } @@ -184,32 +182,32 @@ namespace out { r.y0 = uy0; r.w = ux1 - ux0; r.h = uy1 - uy0; - r.rgba.assign(static_cast(r.w) * r.h * 4, ZERO); + r.rgba.assign(static_cast(r.w) * r.h * 4, ZERO); // place `back` for (int y = 0; y < b.h; ++y) { for (int x = 0; x < b.w; ++x) { - const std::size_t ri = (static_cast(b.y0 + y - uy0) * r.w + - (b.x0 + x - ux0)) * - 4; - const std::size_t bi = (static_cast(y) * b.w + x) * 4; - r.rgba[ri + 0] = b.rgba[bi + 0]; - r.rgba[ri + 1] = b.rgba[bi + 1]; - r.rgba[ri + 2] = b.rgba[bi + 2]; - r.rgba[ri + 3] = b.rgba[bi + 3]; + const size_t ri = (static_cast(b.y0 + y - uy0) * r.w + + (b.x0 + x - ux0)) * + 4; + const size_t bi = (static_cast(y) * b.w + x) * 4; + r.rgba[ri + 0] = b.rgba[bi + 0]; + r.rgba[ri + 1] = b.rgba[bi + 1]; + r.rgba[ri + 2] = b.rgba[bi + 2]; + r.rgba[ri + 3] = b.rgba[bi + 3]; } } // `front` OVER the (back-filled) result for (int y = 0; y < f.h; ++y) { for (int x = 0; x < f.w; ++x) { - const std::size_t ri = (static_cast(f.y0 + y - uy0) * r.w + - (f.x0 + x - ux0)) * - 4; - const std::size_t fi = (static_cast(y) * f.w + x) * 4; - const real_t inv = ONE - f.rgba[fi + 3]; - r.rgba[ri + 0] = f.rgba[fi + 0] + inv * r.rgba[ri + 0]; - r.rgba[ri + 1] = f.rgba[fi + 1] + inv * r.rgba[ri + 1]; - r.rgba[ri + 2] = f.rgba[fi + 2] + inv * r.rgba[ri + 2]; - r.rgba[ri + 3] = f.rgba[fi + 3] + inv * r.rgba[ri + 3]; + const size_t ri = (static_cast(f.y0 + y - uy0) * r.w + + (f.x0 + x - ux0)) * + 4; + const size_t fi = (static_cast(y) * f.w + x) * 4; + const real_t inv = ONE - f.rgba[fi + 3]; + r.rgba[ri + 0] = f.rgba[fi + 0] + inv * r.rgba[ri + 0]; + r.rgba[ri + 1] = f.rgba[fi + 1] + inv * r.rgba[ri + 1]; + r.rgba[ri + 2] = f.rgba[fi + 2] + inv * r.rgba[ri + 2]; + r.rgba[ri + 3] = f.rgba[fi + 3] + inv * r.rgba[ri + 3]; } } return r; @@ -234,18 +232,18 @@ namespace out { const real_t p[3], real_t& outx, real_t& outy) -> bool { - real_t v[3] = { p[0] - cam.eye[0], p[1] - cam.eye[1], p[2] - cam.eye[2] }; + real_t v[3] = { p[0] - cam.eye[0], p[1] - cam.eye[1], p[2] - cam.eye[2] }; const real_t n = std::sqrt(v[0] * v[0] + v[1] * v[1] + v[2] * v[2]); if (n < static_cast(1e-20)) { outx = HALF * static_cast(W) - HALF; // eye itself -> disk center outy = HALF * static_cast(H) - HALF; return true; } - const real_t inv = ONE / n; - v[0] *= inv; - v[1] *= inv; - v[2] *= inv; - real_t cz = v[0] * cam.forward[0] + v[1] * cam.forward[1] + + const real_t inv = ONE / n; + v[0] *= inv; + v[1] *= inv; + v[2] *= inv; + real_t cz = v[0] * cam.forward[0] + v[1] * cam.forward[1] + v[2] * cam.forward[2]; cz = (cz < -ONE) ? -ONE : ((cz > ONE) ? ONE : cz); const real_t theta = std::acos(cz); @@ -255,9 +253,9 @@ namespace out { const real_t r = theta / cam.dome_half_fov; // 0..1 image radius const real_t cx = v[0] * cam.right[0] + v[1] * cam.right[1] + v[2] * cam.right[2]; - const real_t cy = v[0] * cam.up[0] + v[1] * cam.up[1] + v[2] * cam.up[2]; + const real_t cy = v[0] * cam.up[0] + v[1] * cam.up[1] + v[2] * cam.up[2]; const real_t phi = std::atan2(cy, cx); - const real_t fx = r * std::cos(phi), fy = r * std::sin(phi); + const real_t fx = r * std::cos(phi), fy = r * std::sin(phi); outx = (fx + ONE) * HALF * static_cast(W) - HALF; outy = (ONE - fy) * HALF * static_cast(H) - HALF; return true; @@ -326,14 +324,14 @@ namespace out { return fullFrame(); } } - real_t minx = static_cast(1e30), miny = static_cast(1e30); - real_t maxx = static_cast(-1e30), maxy = static_cast(-1e30); - real_t minr = static_cast(1e30); - int n_in = 0, n_out = 0; + real_t minx = static_cast(1e30), miny = static_cast(1e30); + real_t maxx = static_cast(-1e30), maxy = static_cast(-1e30); + real_t minr = static_cast(1e30); + int n_in = 0, n_out = 0; // azimuths of in-FOV samples, for the "wraps the center" (largest-gap) test std::vector phis; - phis.reserve(12 * 17); - const int NS = 48; // samples per AABB edge (a straight edge maps to a + phis.reserve(12ul * 17ul); + const int NS = 48; // samples per AABB edge (a straight edge maps to a // curved fisheye arc, so sample densely to bound it) auto addPoint = [&](const real_t p[3]) { real_t sx, sy; @@ -342,13 +340,13 @@ namespace out { return; } ++n_in; - minx = std::min(minx, sx); - maxx = std::max(maxx, sx); - miny = std::min(miny, sy); - maxy = std::max(maxy, sy); + minx = std::min(minx, sx); + maxx = std::max(maxx, sx); + miny = std::min(miny, sy); + maxy = std::max(maxy, sy); const real_t fx = TWO * (sx + HALF) / static_cast(W) - ONE; const real_t fy = ONE - TWO * (sy + HALF) / static_cast(H); - minr = std::min(minr, std::sqrt(fx * fx + fy * fy)); + minr = std::min(minr, std::sqrt(fx * fx + fy * fy)); phis.push_back(std::atan2(fy, fx)); }; // sample all 12 edges of the AABB @@ -382,7 +380,7 @@ namespace out { std::sort(phis.begin(), phis.end()); real_t maxgap = ZERO; const real_t twopi = static_cast(2.0 * 3.14159265358979323846); - for (std::size_t i = 0; i + 1 < phis.size(); ++i) { + for (size_t i = 0; i + 1 < phis.size(); ++i) { maxgap = std::max(maxgap, phis[i + 1] - phis[i]); } if (not phis.empty()) { @@ -398,14 +396,14 @@ namespace out { int x1 = static_cast(std::ceil(maxx)) + pad; int y0 = static_cast(std::floor(miny)) - pad; int y1 = static_cast(std::ceil(maxy)) + pad; - x0 = std::max(0, std::min(W, x0)); - x1 = std::max(0, std::min(W, x1)); - y0 = std::max(0, std::min(H, y0)); - y1 = std::max(0, std::min(H, y1)); - bx0 = x0; - by0 = y0; - bw = x1 - x0; - bh = y1 - y0; + x0 = std::max(0, std::min(W, x0)); + x1 = std::max(0, std::min(W, x1)); + y0 = std::max(0, std::min(H, y0)); + y1 = std::max(0, std::min(H, y1)); + bx0 = x0; + by0 = y0; + bw = x1 - x0; + bh = y1 - y0; return (bw > 0 and bh > 0); } @@ -421,7 +419,7 @@ namespace out { (void)depth; // fragments are already ascending in depth real_t acc[4] = { ZERO, ZERO, ZERO, ZERO }; for (uint32_t k = k0; k < k1; ++k) { - overComposite(acc, &rgba[static_cast(k) * 4]); + overComposite(acc, &rgba[static_cast(k) * 4]); if (acc[3] >= ONE) { break; } @@ -435,9 +433,9 @@ namespace out { /** * @brief Merge two depth-sorted fragment images: union the bboxes and, per * pixel, merge the two ascending fragment lists by depth, then drop fragments - * once the accumulated alpha reaches `cull_alpha` (exact when cull_alpha == 1: - * only provably-occluded fragments are removed, so the result is independent - * of how the tree is grouped -> associative + commutative). + * once the accumulated alpha reaches `cull_alpha` (exact when cull_alpha == + * 1: only provably-occluded fragments are removed, so the result is + * independent of how the tree is grouped -> associative + commutative). */ inline auto mergeFrag(const FragImage& a, const FragImage& b, real_t cull_alpha) -> FragImage { @@ -452,11 +450,11 @@ namespace out { const int ux1 = std::max(a.x0 + a.w, b.x0 + b.w); const int uy1 = std::max(a.y0 + a.h, b.y0 + b.h); FragImage r; - r.x0 = ux0; - r.y0 = uy0; - r.w = ux1 - ux0; - r.h = uy1 - uy0; - const std::size_t np = static_cast(r.w) * r.h; + r.x0 = ux0; + r.y0 = uy0; + r.w = ux1 - ux0; + r.h = uy1 - uy0; + const size_t np = static_cast(r.w) * r.h; r.offs.assign(np + 1, 0u); // fetch a source image's fragment range at global pixel (gx, gy) @@ -467,9 +465,9 @@ namespace out { k1 = 0; return; } - const std::size_t p = static_cast(ly) * s.w + lx; - k0 = s.offs[p]; - k1 = s.offs[p + 1]; + const size_t p = static_cast(ly) * s.w + lx; + k0 = s.offs[p]; + k1 = s.offs[p + 1]; }; // pass 1: per-pixel surviving-fragment count (merge + occlusion cull) @@ -479,13 +477,13 @@ namespace out { range(a, gx, gy, ak0, ak1); range(b, gx, gy, bk0, bk1); uint32_t ia = ak0, ib = bk0, cnt = 0; - real_t A = ZERO; + real_t A = ZERO; while ((ia < ak1 or ib < bk1) and A < cull_alpha) { const bool takeA = (ib >= bk1) or (ia < ak1 and a.depth[ia] <= b.depth[ib]); - const real_t al = takeA ? a.rgba[static_cast(ia) * 4 + 3] - : b.rgba[static_cast(ib) * 4 + 3]; - A += (ONE - A) * al; + const real_t al = takeA ? a.rgba[static_cast(ia) * 4 + 3] + : b.rgba[static_cast(ib) * 4 + 3]; + A += (ONE - A) * al; ++cnt; if (takeA) { ++ia; @@ -493,16 +491,15 @@ namespace out { ++ib; } } - const std::size_t pix = static_cast(gy - uy0) * r.w + - (gx - ux0); - r.offs[pix + 1] = cnt; + const size_t pix = static_cast(gy - uy0) * r.w + (gx - ux0); + r.offs[pix + 1] = cnt; } } // prefix-sum to offsets - for (std::size_t p = 0; p < np; ++p) { + for (size_t p = 0; p < np; ++p) { r.offs[p + 1] += r.offs[p]; } - const std::size_t nfrag = r.offs[np]; + const size_t nfrag = r.offs[np]; r.depth.assign(nfrag, ZERO); r.rgba.assign(nfrag * 4, ZERO); @@ -512,30 +509,29 @@ namespace out { uint32_t ak0, ak1, bk0, bk1; range(a, gx, gy, ak0, ak1); range(b, gx, gy, bk0, bk1); - const std::size_t pix = static_cast(gy - uy0) * r.w + - (gx - ux0); - uint32_t ia = ak0, ib = bk0, o = r.offs[pix]; + const size_t pix = static_cast(gy - uy0) * r.w + (gx - ux0); + uint32_t ia = ak0, ib = bk0, o = r.offs[pix]; const uint32_t oend = r.offs[pix + 1]; while (o < oend) { const bool takeA = (ib >= bk1) or (ia < ak1 and a.depth[ia] <= b.depth[ib]); if (takeA) { - r.depth[o] = a.depth[ia]; - const std::size_t s = static_cast(ia) * 4; - const std::size_t d = static_cast(o) * 4; - r.rgba[d + 0] = a.rgba[s + 0]; - r.rgba[d + 1] = a.rgba[s + 1]; - r.rgba[d + 2] = a.rgba[s + 2]; - r.rgba[d + 3] = a.rgba[s + 3]; + r.depth[o] = a.depth[ia]; + const size_t s = static_cast(ia) * 4; + const size_t d = static_cast(o) * 4; + r.rgba[d + 0] = a.rgba[s + 0]; + r.rgba[d + 1] = a.rgba[s + 1]; + r.rgba[d + 2] = a.rgba[s + 2]; + r.rgba[d + 3] = a.rgba[s + 3]; ++ia; } else { - r.depth[o] = b.depth[ib]; - const std::size_t s = static_cast(ib) * 4; - const std::size_t d = static_cast(o) * 4; - r.rgba[d + 0] = b.rgba[s + 0]; - r.rgba[d + 1] = b.rgba[s + 1]; - r.rgba[d + 2] = b.rgba[s + 2]; - r.rgba[d + 3] = b.rgba[s + 3]; + r.depth[o] = b.depth[ib]; + const size_t s = static_cast(ib) * 4; + const size_t d = static_cast(o) * 4; + r.rgba[d + 0] = b.rgba[s + 0]; + r.rgba[d + 1] = b.rgba[s + 1]; + r.rgba[d + 2] = b.rgba[s + 2]; + r.rgba[d + 3] = b.rgba[s + 3]; ++ib; } ++o; diff --git a/src/output/render/fieldlines.h b/src/output/render/fieldlines.h index a6adbdd7c..4b9ce8fe3 100644 --- a/src/output/render/fieldlines.h +++ b/src/output/render/fieldlines.h @@ -8,8 +8,6 @@ * - out::emptyTubeSet * @namespaces: * - out:: - * @macros: - * - OUTPUT_ENABLED * @note * Field lines are intrinsically non-local (a streamline wanders across MPI * domains), which would normally demand parallel particle advection. We sidestep @@ -53,7 +51,7 @@ namespace out { * ((c2*n1 + c1)*n0 + c0)*3 + comp, with c0 the fastest spatial axis. */ struct CoarseField { - std::vector B; // n0*n1*n2*3 + std::vector B; // n0*n1*n2*3 int n[3] { 0, 0, 0 }; real_t origin[3] { ZERO, ZERO, ZERO }; real_t dx[3] { ONE, ONE, ONE }; @@ -68,13 +66,13 @@ namespace out { namespace fl_hidden { inline auto cellLinear(const CoarseField& cf, int c0, int c1, int c2) - -> std::size_t { - return (static_cast(c2) * cf.n[1] + c1) * cf.n[0] + c0; + -> size_t { + return (static_cast(c2) * cf.n[1] + c1) * cf.n[0] + c0; } // trilinear sample of the coarse field at world point p -> B[3], |B|. - // Clamps to the grid (so a sample just outside a face still returns the edge - // value); membership in the global box is the caller's stop test. + // Clamps to the grid (so a sample just outside a face still returns the + // edge value); membership in the global box is the caller's stop test. inline auto sampleCoarse(const CoarseField& cf, const real_t p[3], real_t B[3]) -> real_t { int i0[3], i1[3]; @@ -110,20 +108,20 @@ namespace out { const real_t c101 = cf.B[cellLinear(cf, i1[0], i0[1], i1[2]) * 3 + comp]; const real_t c011 = cf.B[cellLinear(cf, i0[0], i1[1], i1[2]) * 3 + comp]; const real_t c111 = cf.B[cellLinear(cf, i1[0], i1[1], i1[2]) * 3 + comp]; - const real_t c00 = c000 * (ONE - fr[0]) + c100 * fr[0]; - const real_t c10 = c010 * (ONE - fr[0]) + c110 * fr[0]; - const real_t c01 = c001 * (ONE - fr[0]) + c101 * fr[0]; - const real_t c11 = c011 * (ONE - fr[0]) + c111 * fr[0]; - const real_t c0 = c00 * (ONE - fr[1]) + c10 * fr[1]; - const real_t c1 = c01 * (ONE - fr[1]) + c11 * fr[1]; - B[comp] = c0 * (ONE - fr[2]) + c1 * fr[2]; + const real_t c00 = c000 * (ONE - fr[0]) + c100 * fr[0]; + const real_t c10 = c010 * (ONE - fr[0]) + c110 * fr[0]; + const real_t c01 = c001 * (ONE - fr[0]) + c101 * fr[0]; + const real_t c11 = c011 * (ONE - fr[0]) + c111 * fr[0]; + const real_t c0 = c00 * (ONE - fr[1]) + c10 * fr[1]; + const real_t c1 = c01 * (ONE - fr[1]) + c11 * fr[1]; + B[comp] = c0 * (ONE - fr[2]) + c1 * fr[2]; } return std::sqrt(B[0] * B[0] + B[1] * B[1] + B[2] * B[2]); } inline auto insideBox(const CoarseField& cf, const real_t p[3]) -> bool { for (int d = 0; d < 3; ++d) { - const real_t hi = cf.origin[d] + cf.n[d] * cf.dx[d]; + const real_t hi = cf.origin[d] + static_cast(cf.n[d]) * cf.dx[d]; if (p[d] < cf.origin[d] or p[d] > hi) { return false; } @@ -136,7 +134,7 @@ namespace out { /** @brief A constant-color opaque LUT (monochrome field lines). */ inline auto buildSolidLUT(real_t r, real_t g, real_t b, int n_lut) -> array_t { - array_t lut { "fl_solid_lut", static_cast(n_lut) }; + array_t lut { "fl_solid_lut", static_cast(n_lut) }; auto h = Kokkos::create_mirror_view(lut); for (int i = 0; i < n_lut; ++i) { h(i, 0) = r; // opaque -> premultiplied == straight RGB @@ -154,7 +152,12 @@ namespace out { if (cfg.color.size() == 3) { return buildSolidLUT(cfg.color[0], cfg.color[1], cfg.color[2], n_lut); } - return buildLUT(cfg.colormap, n_lut, { { ZERO, ONE }, { ONE, ONE } }); + return buildLUT(cfg.colormap, + n_lut, + { + { ZERO, ONE }, + { ONE, ONE } + }); } /** @@ -169,8 +172,7 @@ namespace out { const FieldLineConfig& cfg, real_t world_per_pixel, real_t& out_vmin, - real_t& out_vmax) - -> std::vector { + real_t& out_vmax) -> std::vector { using fl_hidden::insideBox; using fl_hidden::sampleCoarse; @@ -180,17 +182,16 @@ namespace out { } real_t size[3]; - real_t diag2 = ZERO; + real_t diag2 = ZERO; real_t min_dx = static_cast(1e30); for (int d = 0; d < 3; ++d) { - size[d] = cf.n[d] * cf.dx[d]; + size[d] = static_cast(cf.n[d]) * cf.dx[d]; diag2 += size[d] * size[d]; - min_dx = std::min(min_dx, cf.dx[d]); + min_dx = std::min(min_dx, cf.dx[d]); } const real_t box_diag = std::sqrt(diag2); const real_t max_len = cfg.max_len_frac * box_diag; - const real_t h = std::max(cfg.step_frac, static_cast(1e-3)) * - min_dx; + const real_t h = std::max(cfg.step_frac, static_cast(1e-3)) * min_dx; const real_t eps = static_cast(1e-20); // seed lattice: spacing ~ seed_px screen pixels, grown to respect seed_max @@ -206,8 +207,8 @@ namespace out { }; long n_seed = countSeeds(spacing); if (n_seed > cfg.seed_max and cfg.seed_max > 0) { - const real_t grow = std::cbrt(static_cast(n_seed) / - static_cast(cfg.seed_max)); + const real_t grow = std::cbrt( + static_cast(n_seed) / static_cast(cfg.seed_max)); spacing *= grow; countSeeds(spacing); // recompute ns[] for the grown spacing } @@ -226,8 +227,8 @@ namespace out { return true; }; - out_vmin = static_cast(1e30); - out_vmax = static_cast(-1e30); + out_vmin = static_cast(1e30); + out_vmax = static_cast(-1e30); auto track = [&](real_t m) { out_vmin = std::min(out_vmin, m); out_vmax = std::max(out_vmax, m); @@ -236,7 +237,7 @@ namespace out { // integrate one direction (dir = +1 forward, -1 backward) from a seed auto integrate = [&](const real_t seed[3], real_t dir) { Polyline pl; - real_t p[3] = { seed[0], seed[1], seed[2] }; + real_t p[3] = { seed[0], seed[1], seed[2] }; real_t B0[3]; real_t m0 = sampleCoarse(cf, p, B0); if (m0 < eps) { @@ -293,9 +294,12 @@ namespace out { for (int j = 0; j < ns[1]; ++j) { for (int i = 0; i < ns[0]; ++i) { const real_t seed[3] = { - cf.origin[0] + (static_cast(i) + HALF) * size[0] / ns[0], - cf.origin[1] + (static_cast(j) + HALF) * size[1] / ns[1], - cf.origin[2] + (static_cast(k) + HALF) * size[2] / ns[2] + cf.origin[0] + (static_cast(i) + HALF) * size[0] / + static_cast(ns[0]), + cf.origin[1] + (static_cast(j) + HALF) * size[1] / + static_cast(ns[1]), + cf.origin[2] + (static_cast(k) + HALF) * size[2] / + static_cast(ns[2]) }; integrate(seed, ONE); integrate(seed, -ONE); @@ -328,7 +332,7 @@ namespace out { const real_t lo[3], const real_t hi[3], const CoarseField& cf, - std::size_t& n_kept) -> TubeSet { + size_t& n_kept) -> TubeSet { TubeSet ts; ts.radius = radius; ts.vmin = vmin; @@ -346,8 +350,8 @@ namespace out { const real_t span = hi[d] - lo[d]; ts.gnc[d] = std::max(1, static_cast(std::ceil(span / ts.gdx[d]))); } - auto lin = [&](int c0, int c1, int c2) -> std::size_t { - return (static_cast(c2) * ts.gnc[1] + c1) * ts.gnc[0] + c0; + auto lin = [&](int c0, int c1, int c2) -> size_t { + return (static_cast(c2) * ts.gnc[1] + c1) * ts.gnc[0] + c0; }; // opaque LUT: a tube sample paints a solid color (alpha==1), by |B| or a // single monochrome color when cfg.color is set @@ -359,10 +363,10 @@ namespace out { // the correct domain's depth range -- no double-draw.) std::vector> kept; for (const auto& pl : lines) { - for (std::size_t i = 0; i + 1 < pl.pts.size(); ++i) { - const auto& a = pl.pts[i]; - const auto& b = pl.pts[i + 1]; - bool overlap = true; + for (size_t i = 0; i + 1 < pl.pts.size(); ++i) { + const auto& a = pl.pts[i]; + const auto& b = pl.pts[i + 1]; + bool overlap = true; for (int d = 0; d < 3; ++d) { const real_t smin = std::min(a[d], b[d]) - radius; const real_t smax = std::max(a[d], b[d]) + radius; @@ -372,15 +376,14 @@ namespace out { } } if (overlap) { - kept.push_back({ a[0], a[1], a[2], b[0], b[1], b[2], pl.scal[i], - pl.scal[i + 1] }); + kept.push_back( + { a[0], a[1], a[2], b[0], b[1], b[2], pl.scal[i], pl.scal[i + 1] }); } } } - n_kept = kept.size(); - ts.n_seg = static_cast(kept.size()); - const std::size_t ncell = static_cast(ts.gnc[0]) * ts.gnc[1] * - ts.gnc[2]; + n_kept = kept.size(); + ts.n_seg = static_cast(kept.size()); + const size_t ncell = static_cast(ts.gnc[0]) * ts.gnc[1] * ts.gnc[2]; // 2) CSR bucketing on the coarse grid: count, prefix-sum, scatter. Each // segment is registered in every cell its radius-padded AABB overlaps. @@ -396,8 +399,8 @@ namespace out { auto cellRange = [&](const std::array& s, int d, int& c0, int& c1) { const real_t smin = std::min(s[d], s[3 + d]) - radius; const real_t smax = std::max(s[d], s[3 + d]) + radius; - c0 = cellOf(smin, d); - c1 = cellOf(smax, d); + c0 = cellOf(smin, d); + c1 = cellOf(smax, d); }; std::vector count(ncell + 1, 0); @@ -415,13 +418,13 @@ namespace out { } } std::vector start(ncell + 1, 0); - for (std::size_t c = 0; c < ncell; ++c) { + for (size_t c = 0; c < ncell; ++c) { start[c + 1] = start[c] + count[c]; } - const std::size_t n_insert = static_cast(start[ncell]); - std::vector idx(n_insert, 0); - std::vector cursor(start.begin(), start.end()); // running write head - for (std::size_t si = 0; si < kept.size(); ++si) { + const size_t n_insert = static_cast(start[ncell]); + std::vector idx(n_insert, 0); + std::vector cursor(start.begin(), start.end()); // running write head + for (size_t si = 0; si < kept.size(); ++si) { int a0, a1, b0, b1, d0, d1; cellRange(kept[si], 0, a0, a1); cellRange(kept[si], 1, b0, b1); @@ -429,20 +432,20 @@ namespace out { for (int c2 = d0; c2 <= d1; ++c2) { for (int c1 = b0; c1 <= b1; ++c1) { for (int c0 = a0; c0 <= a1; ++c0) { - const std::size_t cl = lin(c0, c1, c2); - idx[static_cast(cursor[cl]++)] = static_cast(si); + const size_t cl = lin(c0, c1, c2); + idx[static_cast(cursor[cl]++)] = static_cast(si); } } } } // 3) upload to device - ts.seg = array_t("fl_seg", static_cast(ts.n_seg)); + ts.seg = array_t("fl_seg", static_cast(ts.n_seg)); if (ts.n_seg > 0) { auto seg_h = Kokkos::create_mirror_view(ts.seg); for (int s = 0; s < ts.n_seg; ++s) { for (int c = 0; c < 8; ++c) { - seg_h(s, c) = kept[static_cast(s)][static_cast(c)]; + seg_h(s, c) = kept[static_cast(s)][static_cast(c)]; } } Kokkos::deep_copy(ts.seg, seg_h); @@ -450,15 +453,15 @@ namespace out { ts.cell_start = array_t("fl_cell_start", ncell + 1); { auto h = Kokkos::create_mirror_view(ts.cell_start); - for (std::size_t c = 0; c <= ncell; ++c) { + for (size_t c = 0; c <= ncell; ++c) { h(c) = start[c]; } Kokkos::deep_copy(ts.cell_start, h); } - ts.seg_idx = array_t("fl_seg_idx", std::max(n_insert, 1)); + ts.seg_idx = array_t("fl_seg_idx", std::max(n_insert, 1)); if (n_insert > 0) { auto h = Kokkos::create_mirror_view(ts.seg_idx); - for (std::size_t k = 0; k < n_insert; ++k) { + for (size_t k = 0; k < n_insert; ++k) { h(k) = idx[k]; } Kokkos::deep_copy(ts.seg_idx, h); @@ -473,7 +476,12 @@ namespace out { ts.seg = array_t("fl_seg_empty", 0); ts.cell_start = array_t("fl_cell_start_empty", 1); ts.seg_idx = array_t("fl_seg_idx_empty", 1); - ts.lut = buildLUT("inferno", 2, { { ZERO, ONE }, { ONE, ONE } }); + ts.lut = buildLUT("inferno", + 2, + { + { ZERO, ONE }, + { ONE, ONE } + }); return ts; } @@ -486,7 +494,7 @@ namespace out { * @note Component-fastest: (c0,c1,comp) lives at (c1*n0 + c0)*2 + comp. */ struct CoarseField2D { - std::vector B; // n0*n1*2 + std::vector B; // n0*n1*2 int n[2] { 0, 0 }; real_t origin[2] { ZERO, ZERO }; real_t dx[2] { ONE, ONE }; @@ -509,12 +517,12 @@ namespace out { real_t& bmin, real_t& bmax) { const int nx = cf.n[0], ny = cf.n[1]; - psi.assign(static_cast(nx) * ny, ZERO); + psi.assign(static_cast(nx) * ny, ZERO); auto B = [&](int i, int j, int c) -> real_t { - return cf.B[(static_cast(j) * nx + i) * 2 + c]; + return cf.B[(static_cast(j) * nx + i) * 2 + c]; }; auto P = [&](int i, int j) -> real_t& { - return psi[static_cast(j) * nx + i]; + return psi[static_cast(j) * nx + i]; }; // bottom row (j = 0): d psi/dx = -By, trapezoidal in x for (int i = 1; i < nx; ++i) { @@ -532,13 +540,13 @@ namespace out { bmax = static_cast(-1e30); for (int j = 0; j < ny; ++j) { for (int i = 0; i < nx; ++i) { - const real_t p = P(i, j); - psi_min = std::min(psi_min, p); - psi_max = std::max(psi_max, p); + const real_t p = P(i, j); + psi_min = std::min(psi_min, p); + psi_max = std::max(psi_max, p); const real_t bx = B(i, j, 0), by = B(i, j, 1); - const real_t b = std::sqrt(bx * bx + by * by); - bmin = std::min(bmin, b); - bmax = std::max(bmax, b); + const real_t b = std::sqrt(bx * bx + by * by); + bmin = std::min(bmin, b); + bmax = std::max(bmax, b); } } if (psi_min > psi_max) { @@ -589,11 +597,11 @@ namespace out { cs.vmin = vlo; cs.vmax = (vhi > vlo) ? vhi : (vlo + ONE); cs.lut = buildLineLUT(cfg, cs.n_lut); // by |B| or monochrome (cfg.color) - const std::size_t n = static_cast(cs.n0) * cs.n1; - cs.psi = array_t("fl_psi", std::max(n, 1)); + const size_t n = static_cast(cs.n0) * cs.n1; + cs.psi = array_t("fl_psi", std::max(n, 1)); if (n > 0) { auto h = Kokkos::create_mirror_view(cs.psi); - for (std::size_t k = 0; k < n; ++k) { + for (size_t k = 0; k < n; ++k) { h(k) = psi[k]; } Kokkos::deep_copy(cs.psi, h); @@ -607,7 +615,12 @@ namespace out { ContourSet cs; cs.enabled = false; cs.psi = array_t("fl_psi_empty", 1); - cs.lut = buildLUT("inferno", 2, { { ZERO, ONE }, { ONE, ONE } }); + cs.lut = buildLUT("inferno", + 2, + { + { ZERO, ONE }, + { ONE, ONE } + }); return cs; } @@ -621,8 +634,8 @@ namespace out { inline auto sampleRTh(const CoarseField2D& cf, real_t r, real_t th, real_t B[2]) -> bool { const real_t rmin = cf.origin[0], thmin = cf.origin[1]; - const real_t rmax = rmin + cf.n[0] * cf.dx[0]; - const real_t thmax = thmin + cf.n[1] * cf.dx[1]; + const real_t rmax = rmin + static_cast(cf.n[0]) * cf.dx[0]; + const real_t thmax = thmin + static_cast(cf.n[1]) * cf.dx[1]; const real_t tr = HALF * cf.dx[0], tt = HALF * cf.dx[1]; if (r < rmin - tr or r > rmax + tr or th < thmin - tt or th > thmax + tt) { return false; @@ -666,13 +679,13 @@ namespace out { j1 = b + 1; } for (int c = 0; c < 2; ++c) { - const real_t c00 = cf.B[(static_cast(j0) * cf.n[0] + i0) * 2 + c]; - const real_t c10 = cf.B[(static_cast(j0) * cf.n[0] + i1) * 2 + c]; - const real_t c01 = cf.B[(static_cast(j1) * cf.n[0] + i0) * 2 + c]; - const real_t c11 = cf.B[(static_cast(j1) * cf.n[0] + i1) * 2 + c]; - const real_t e0 = c00 * (ONE - a0) + c10 * a0; - const real_t e1 = c01 * (ONE - a0) + c11 * a0; - B[c] = e0 * (ONE - a1) + e1 * a1; + const real_t c00 = cf.B[(static_cast(j0) * cf.n[0] + i0) * 2 + c]; + const real_t c10 = cf.B[(static_cast(j0) * cf.n[0] + i1) * 2 + c]; + const real_t c01 = cf.B[(static_cast(j1) * cf.n[0] + i0) * 2 + c]; + const real_t c11 = cf.B[(static_cast(j1) * cf.n[0] + i1) * 2 + c]; + const real_t e0 = c00 * (ONE - a0) + c10 * a0; + const real_t e1 = c01 * (ONE - a0) + c11 * a0; + B[c] = e0 * (ONE - a1) + e1 * a1; } return true; } @@ -680,8 +693,8 @@ namespace out { /** * @brief Trace poloidal field lines in the meridional (X, Z) plane (nt2py - * style): integrate (Fx, Fz) = (Br sin th + Bth cos th, Br cos th - Bth sin th) - * by bidirectional RK4 through the coarse (r, theta) field. Polylines are + * style): integrate (Fx, Fz) = (Br sin th + Bth cos th, Br cos th - Bth sin + * th) by bidirectional RK4 through the coarse (r, theta) field. Polylines are * returned in meridional world coords (z = 0 so they reuse the 3D tube * builder); with `mirror` the X<0 half is added as the theta-reflected copy. * @param cf coarse field: component 0 = Br, 1 = Btheta, grid in (r, theta) @@ -692,23 +705,22 @@ namespace out { real_t world_per_pixel, bool mirror, real_t& out_vmin, - real_t& out_vmax) - -> std::vector { + real_t& out_vmax) -> std::vector { using fl_hidden::sampleRTh; std::vector lines; if (cf.n[0] < 1 or cf.n[1] < 1) { return lines; } const real_t rmin = cf.origin[0], thmin = cf.origin[1]; - const real_t rmax = rmin + cf.n[0] * cf.dx[0]; - const real_t thmax = thmin + cf.n[1] * cf.dx[1]; - const real_t h = std::max(cfg.step_frac, static_cast(1e-3)) * + const real_t rmax = rmin + static_cast(cf.n[0]) * cf.dx[0]; + const real_t thmax = thmin + static_cast(cf.n[1]) * cf.dx[1]; + const real_t h = std::max(cfg.step_frac, static_cast(1e-3)) * cf.dx[0]; // step ~ a coarse dr (a length) const real_t max_len = cfg.max_len_frac * rmax * static_cast(2); const real_t eps = static_cast(1e-20); auto bmag = [&](real_t X, real_t Z) -> real_t { - const real_t r = std::sqrt(X * X + Z * Z); + const real_t r = std::sqrt(X * X + Z * Z); const real_t th = std::atan2(std::abs(X), Z); real_t B[2]; if (not sampleRTh(cf, r, th, B)) { @@ -719,7 +731,7 @@ namespace out { // unit meridional direction (x dir); false if |F| ~ 0 or outside the grid auto deriv = [&](const real_t p[2], real_t dir, real_t out[2]) -> bool { const real_t X = p[0], Z = p[1]; - const real_t r = std::sqrt(X * X + Z * Z); + const real_t r = std::sqrt(X * X + Z * Z); const real_t th = std::atan2(std::abs(X), Z); real_t B[2]; if (not sampleRTh(cf, r, th, B)) { @@ -741,14 +753,14 @@ namespace out { return true; }; - out_vmin = static_cast(1e30); - out_vmax = static_cast(-1e30); + out_vmin = static_cast(1e30); + out_vmax = static_cast(-1e30); auto track = [&](real_t m) { out_vmin = std::min(out_vmin, m); out_vmax = std::max(out_vmax, m); }; auto inDomain = [&](real_t X, real_t Z) -> bool { - const real_t r = std::sqrt(X * X + Z * Z); + const real_t r = std::sqrt(X * X + Z * Z); const real_t th = std::atan2(std::abs(X), Z); return (r >= rmin and r <= rmax and th >= thmin and th <= thmax); }; @@ -806,7 +818,7 @@ namespace out { // seed lattice over the X>=0 meridional half, keeping in-domain seeds const real_t Xhi = rmax, Zlo = -rmax, Zhi = rmax; - real_t spacing = std::max(cfg.seed_px, ONE) * world_per_pixel; + real_t spacing = std::max(cfg.seed_px, ONE) * world_per_pixel; auto gridCount = [&](real_t s) -> long { const long nx = std::max(1L, static_cast(std::floor(Xhi / s))); const long nz = std::max(1L, static_cast(std::floor((Zhi - Zlo) / s))); @@ -839,8 +851,8 @@ namespace out { } // mirror the traced (X>=0) lines into the X<0 half for a full disk if (mirror) { - const std::size_t n0 = lines.size(); - for (std::size_t i = 0; i < n0; ++i) { + const size_t n0 = lines.size(); + for (size_t i = 0; i < n0; ++i) { Polyline m = lines[i]; for (auto& q : m.pts) { q[0] = -q[0]; diff --git a/src/output/render/png.h b/src/output/render/png.h index 3f22cd5f5..7eba41a44 100644 --- a/src/output/render/png.h +++ b/src/output/render/png.h @@ -22,14 +22,13 @@ #include #include -#include #include namespace out { namespace png_hidden { - inline auto crc32(const uint8_t* data, std::size_t len) -> uint32_t { + inline auto crc32(const uint8_t* data, size_t len) -> uint32_t { static uint32_t table[256]; static bool ready = false; if (not ready) { @@ -43,16 +42,16 @@ namespace out { ready = true; } uint32_t c = 0xFFFFFFFFu; - for (std::size_t i = 0; i < len; ++i) { + for (size_t i = 0; i < len; ++i) { c = table[(c ^ data[i]) & 0xFFu] ^ (c >> 8); } return c ^ 0xFFFFFFFFu; } - inline auto adler32(const uint8_t* data, std::size_t len) -> uint32_t { + inline auto adler32(const uint8_t* data, size_t len) -> uint32_t { constexpr uint32_t MOD = 65521u; uint32_t a = 1u, b = 0u; - for (std::size_t i = 0; i < len; ++i) { + for (size_t i = 0; i < len; ++i) { a = (a + data[i]) % MOD; b = (b + a) % MOD; } @@ -81,13 +80,14 @@ namespace out { } // zlib stream wrapping `raw` in stored (BTYPE=00) DEFLATE blocks - inline auto zlib_store(const std::vector& raw) -> std::vector { + inline auto zlib_store(const std::vector& raw) + -> std::vector { std::vector z; z.push_back(0x78); // CMF: CM=8, CINFO=7 z.push_back(0x01); // FLG: makes (CMF<<8 | FLG) % 31 == 0, no dict, level 0 - std::size_t off = 0; - const std::size_t n = raw.size(); - constexpr std::size_t BLOCK = 65535u; + size_t off = 0; + const size_t n = raw.size(); + constexpr size_t BLOCK = 65535u; if (n == 0) { z.push_back(0x01); // final, stored z.push_back(0x00); @@ -96,8 +96,8 @@ namespace out { z.push_back(0xFF); } while (off < n) { - const std::size_t len = (n - off > BLOCK) ? BLOCK : (n - off); - const bool final = (off + len >= n); + const size_t len = (n - off > BLOCK) ? BLOCK : (n - off); + const bool final = (off + len >= n); z.push_back(final ? 0x01 : 0x00); const uint16_t l = static_cast(len); const uint16_t nl = static_cast(~l); @@ -105,7 +105,9 @@ namespace out { z.push_back(static_cast((l >> 8) & 0xFFu)); z.push_back(static_cast(nl & 0xFFu)); z.push_back(static_cast((nl >> 8) & 0xFFu)); - z.insert(z.end(), raw.begin() + off, raw.begin() + off + len); + z.insert(z.end(), + raw.begin() + static_cast(off), + raw.begin() + static_cast(off + len)); off += len; } put_u32_be(z, adler32(raw.data(), raw.size())); @@ -122,17 +124,15 @@ namespace out { * @param rgba pointer to width*height*4 bytes, row-major, top-left origin * @return true on success */ - inline auto write_png(const path_t& path, - int width, - int height, - const uint8_t* rgba) -> bool { + inline auto write_png(const path_t& path, int width, int height, const uint8_t* rgba) + -> bool { using namespace png_hidden; - const std::size_t w = static_cast(width); - const std::size_t h = static_cast(height); + const size_t w = static_cast(width); + const size_t h = static_cast(height); // build filtered raw scanlines: each row prefixed with filter byte 0 (None) std::vector raw; raw.reserve(h * (1 + w * 4)); - for (std::size_t y = 0; y < h; ++y) { + for (size_t y = 0u; y < h; ++y) { raw.push_back(0x00); const uint8_t* row = rgba + y * w * 4; raw.insert(raw.end(), row, row + w * 4); @@ -140,7 +140,7 @@ namespace out { std::vector file; // PNG signature - const uint8_t sig[8] = { 137, 80, 78, 71, 13, 10, 26, 10 }; + const uint8_t sig[8] = { 137, 80, 78, 71, 13, 10, 26, 10 }; file.insert(file.end(), sig, sig + 8); // IHDR diff --git a/src/output/render/raymarch.hpp b/src/output/render/raymarch.hpp index f86be67f7..de09ff4f5 100644 --- a/src/output/render/raymarch.hpp +++ b/src/output/render/raymarch.hpp @@ -2,11 +2,9 @@ * @file output/render/raymarch.hpp * @brief Header-only Kokkos volume ray-march kernel (one parallel_for over pixels) * @implements - * - kernel::VolumeRayMarch_kernel + * - render::VolumeRayMarch_kernel * @namespaces: - * - kernel:: - * @macros: - * - OUTPUT_ENABLED + * - render:: * @note * Pure Kokkos: the only device entities are Views, the (trivially-copyable) * metric, and the POD camera. Runs in Kokkos::DefaultExecutionSpace, inheriting @@ -25,15 +23,13 @@ #ifndef OUTPUT_RENDER_RAYMARCH_HPP #define OUTPUT_RENDER_RAYMARCH_HPP -#include "enums.h" #include "global.h" #include "arch/kokkos_aliases.h" -#include "traits/metric.h" #include "output/render/renderer.h" -namespace kernel { +namespace render { using namespace ntt; template @@ -49,10 +45,10 @@ namespace kernel { const real_t lo0, lo1, lo2, hi0, hi1, hi2; const int ext0, ext1, ext2; - const int W, H; // full frame size (for ray generation / ndc) + const int W, H; // full frame size (for ray generation / ndc) const int bx0, by0, bw; // screen-bbox offset and width (output stride) - const real_t ds; // fixed world step (global, identical on all ranks) - const int max_steps; // safety cap on the marching loop + const real_t ds; // fixed world step (global, identical on all ranks) + const int max_steps; // safety cap on the marching loop // transfer function array_t lut; @@ -74,10 +70,10 @@ namespace kernel { array_t tcell_start; // (ncell+1) CSR offsets array_t tseg_idx; // segment indices grouped by cell const int tn_seg; - const real_t tube_r2; // squared tube radius + const real_t tube_r2; // squared tube radius const int tgnc0, tgnc1, tgnc2; - const real_t tg0, tg1, tg2; // bucket-grid origin - const real_t tdx0, tdx1, tdx2; // bucket-grid cell size + const real_t tg0, tg1, tg2; // bucket-grid origin + const real_t tdx0, tdx1, tdx2; // bucket-grid cell size array_t tube_lut; const int tube_n_lut; const real_t tube_vmin, tube_vmax; @@ -188,7 +184,7 @@ namespace kernel { if (spine_radius <= ZERO) { return false; } - const real_t r2 = spine_radius * spine_radius; + const real_t r2 = spine_radius * spine_radius; const real_t pad = spine_radius; // edges parallel to x (perp plane = y,z) if (px >= glo0 - pad and px <= ghi0 + pad) { @@ -229,15 +225,14 @@ namespace kernel { const int c0 = static_cast(math::floor((px - tg0) / tdx0)); const int c1 = static_cast(math::floor((py - tg1) / tdx1)); const int c2 = static_cast(math::floor((pz - tg2) / tdx2)); - if (c0 < 0 or c0 >= tgnc0 or c1 < 0 or c1 >= tgnc1 or c2 < 0 or - c2 >= tgnc2) { + if (c0 < 0 or c0 >= tgnc0 or c1 < 0 or c1 >= tgnc1 or c2 < 0 or c2 >= tgnc2) { return false; } - const int lin = (c2 * tgnc1 + c1) * tgnc0 + c0; - const int kb = tcell_start(lin); - const int ke = tcell_start(lin + 1); - real_t best = tube_r2; - bool hit = false; + const int lin = (c2 * tgnc1 + c1) * tgnc0 + c0; + const int kb = tcell_start(lin); + const int ke = tcell_start(lin + 1); + real_t best = tube_r2; + bool hit = false; for (int k = kb; k < ke; ++k) { const int s = tseg_idx(k); const real_t ax = tseg(s, 0), ay = tseg(s, 1), az = tseg(s, 2); @@ -245,8 +240,8 @@ namespace kernel { const real_t ex = bx - ax, ey = by - ay, ez = bz - az; const real_t wx = px - ax, wy = py - ay, wz = pz - az; const real_t ee = ex * ex + ey * ey + ez * ez; - real_t tt = (ee > ZERO) ? (wx * ex + wy * ey + wz * ez) / ee : ZERO; - tt = (tt < ZERO) ? ZERO : ((tt > ONE) ? ONE : tt); + real_t tt = (ee > ZERO) ? (wx * ex + wy * ey + wz * ez) / ee : ZERO; + tt = (tt < ZERO) ? ZERO : ((tt > ONE) ? ONE : tt); const real_t cx = ax + tt * ex, cy = ay + tt * ey, cz = az + tt * ez; const real_t dx = px - cx, dy = py - cy, dz = pz - cz; const real_t d2 = dx * dx + dy * dy + dz * dz; @@ -273,13 +268,13 @@ namespace kernel { const real_t t1 = g1 - f1; const real_t t2 = g2 - f2; // base View index of the lower corner (active cell i -> View i + N_GHOSTS) - int b0 = static_cast(f0) + static_cast(N_GHOSTS); - int b1 = static_cast(f1) + static_cast(N_GHOSTS); - int b2 = static_cast(f2) + static_cast(N_GHOSTS); + int b0 = static_cast(f0) + static_cast(N_GHOSTS); + int b1 = static_cast(f1) + static_cast(N_GHOSTS); + int b2 = static_cast(f2) + static_cast(N_GHOSTS); // clamp so both corners (b, b+1) stay in range [0, ext-1] - b0 = (b0 < 0) ? 0 : ((b0 > ext0 - 2) ? ext0 - 2 : b0); - b1 = (b1 < 0) ? 0 : ((b1 > ext1 - 2) ? ext1 - 2 : b1); - b2 = (b2 < 0) ? 0 : ((b2 > ext2 - 2) ? ext2 - 2 : b2); + b0 = (b0 < 0) ? 0 : ((b0 > ext0 - 2) ? ext0 - 2 : b0); + b1 = (b1 < 0) ? 0 : ((b1 > ext1 - 2) ? ext1 - 2 : b1); + b2 = (b2 < 0) ? 0 : ((b2 > ext2 - 2) ? ext2 - 2 : b2); const real_t c000 = Fld(b0, b1, b2, comp); const real_t c100 = Fld(b0 + 1, b1, b2, comp); const real_t c010 = Fld(b0, b1 + 1, b2, comp); @@ -302,26 +297,27 @@ namespace kernel { const auto pix = static_cast(lpy) * static_cast(bw) + static_cast(lpx); - const int gpx = bx0 + static_cast(lpx); - const int gpy = by0 + static_cast(lpy); + const int gpx = bx0 + static_cast(lpx); + const int gpy = by0 + static_cast(lpy); const real_t INF = static_cast(1e30); // default transparent + no fragment - image(pix, 0) = ZERO; - image(pix, 1) = ZERO; - image(pix, 2) = ZERO; - image(pix, 3) = ZERO; - depth_img(pix) = INF; + image(pix, 0) = ZERO; + image(pix, 1) = ZERO; + image(pix, 2) = ZERO; + image(pix, 3) = ZERO; + depth_img(pix) = INF; // ---- ray generation ------------------------------------------------ // const real_t fx = TWO * (static_cast(gpx) + HALF) / - static_cast(W) - ONE; + static_cast(W) - + ONE; const real_t fy = ONE - TWO * (static_cast(gpy) + HALF) / static_cast(H); real_t ox, oy, oz, dx, dy, dz; if (cam.projection == out::CameraDevice::Dome) { // fulldome azimuthal-equidistant fisheye from an interior eye: image - // radius rho in [0,1] -> zenith angle theta = rho * dome_half_fov, about - // the `forward` (zenith) axis; corners (rho > 1) are transparent. + // radius rho in [0,1] -> zenith angle theta = rho * dome_half_fov, + // about the `forward` (zenith) axis; corners (rho > 1) are transparent. const real_t rho = math::sqrt(fx * fx + fy * fy); if (rho > ONE) { return; // outside the dome disk @@ -339,25 +335,25 @@ namespace kernel { } else if (cam.orthographic) { const real_t sx = fx * cam.half_w; const real_t sy = fy * cam.half_h; - ox = cam.eye[0] + sx * cam.right[0] + sy * cam.up[0]; - oy = cam.eye[1] + sx * cam.right[1] + sy * cam.up[1]; - oz = cam.eye[2] + sx * cam.right[2] + sy * cam.up[2]; - dx = cam.forward[0]; - dy = cam.forward[1]; - dz = cam.forward[2]; + ox = cam.eye[0] + sx * cam.right[0] + sy * cam.up[0]; + oy = cam.eye[1] + sx * cam.right[1] + sy * cam.up[1]; + oz = cam.eye[2] + sx * cam.right[2] + sy * cam.up[2]; + dx = cam.forward[0]; + dy = cam.forward[1]; + dz = cam.forward[2]; } else { - const real_t nx = fx * cam.aspect * cam.tan_half_fov; - const real_t ny = fy * cam.tan_half_fov; - dx = cam.forward[0] + nx * cam.right[0] + ny * cam.up[0]; - dy = cam.forward[1] + nx * cam.right[1] + ny * cam.up[1]; - dz = cam.forward[2] + nx * cam.right[2] + ny * cam.up[2]; - const real_t inv = ONE / math::sqrt(dx * dx + dy * dy + dz * dz); - dx *= inv; - dy *= inv; - dz *= inv; - ox = cam.eye[0]; - oy = cam.eye[1]; - oz = cam.eye[2]; + const real_t nx = fx * cam.aspect * cam.tan_half_fov; + const real_t ny = fy * cam.tan_half_fov; + dx = cam.forward[0] + nx * cam.right[0] + ny * cam.up[0]; + dy = cam.forward[1] + nx * cam.right[1] + ny * cam.up[1]; + dz = cam.forward[2] + nx * cam.right[2] + ny * cam.up[2]; + const real_t inv = ONE / math::sqrt(dx * dx + dy * dy + dz * dz); + dx *= inv; + dy *= inv; + dz *= inv; + ox = cam.eye[0]; + oy = cam.eye[1]; + oz = cam.eye[2]; } // ---- ray-AABB slab test against [lo, hi] --------------------------- // @@ -430,12 +426,12 @@ namespace kernel { const real_t tube_inv_range = (tube_vmax > tube_vmin) ? (ONE / (tube_vmax - tube_vmin)) : ZERO; - const real_t tube_log_vmin = (tube_log and tube_vmin > ZERO) - ? math::log10(tube_vmin) - : ZERO; + const real_t tube_log_vmin = (tube_log and tube_vmin > ZERO) + ? math::log10(tube_vmin) + : ZERO; // first global sample index inside this segment: t_k >= t_enter - const real_t k0 = math::ceil(t_enter / ds); - real_t t = k0 * ds; + const real_t k0 = math::ceil(t_enter / ds); + real_t t = k0 * ds; real_t acc_r = ZERO, acc_g = ZERO, acc_b = ZERO, acc_a = ZERO; int steps = 0; while (t < t_exit and steps < max_steps) { @@ -464,8 +460,7 @@ namespace kernel { } else if (u > ONE) { u = ONE; } - int idx = static_cast(u * static_cast(tube_n_lut - 1) + - HALF); + int idx = static_cast(u * static_cast(tube_n_lut - 1) + HALF); if (idx < 0) { idx = 0; } else if (idx > tube_n_lut - 1) { @@ -478,7 +473,7 @@ namespace kernel { } else if (volume_enabled) { const real_t s = sample(px, py, pz); // normalize through the transfer function range - real_t u; + real_t u; if (log_scale) { u = (s > ZERO) ? (math::log10(s) - log_vmin) * inv_range : -ONE; } else { @@ -503,11 +498,11 @@ namespace kernel { // composite only a non-empty sample (volume-off rays contribute solely // where they hit a tube or the spine) if (ca > ZERO) { - const real_t w = ONE - acc_a; - acc_r += w * cr; - acc_g += w * cg; - acc_b += w * cb; - acc_a += w * ca; + const real_t w = ONE - acc_a; + acc_r += w * cr; + acc_g += w * cg; + acc_b += w * cb; + acc_a += w * ca; if (acc_a >= early_alpha) { break; } @@ -529,6 +524,6 @@ namespace kernel { } }; -} // namespace kernel +} // namespace render #endif // OUTPUT_RENDER_RAYMARCH_HPP diff --git a/src/output/render/reduce.hpp b/src/output/render/reduce.hpp index 18478fc6b..b5a590251 100644 --- a/src/output/render/reduce.hpp +++ b/src/output/render/reduce.hpp @@ -2,14 +2,12 @@ * @file output/render/reduce.hpp * @brief Small dimension-generic cell reductions used by the in-situ renderer * @implements - * - kernel::RenderMagnitude3_kernel - * - kernel::RenderPickComp_kernel - * - kernel::RenderDivideComp_kernel - * - kernel::RenderVmagByRho_kernel + * - render::RenderMagnitude3_kernel + * - render::RenderPickComp_kernel + * - render::RenderDivideComp_kernel + * - render::RenderVmagByRho_kernel * @namespaces: - * - kernel:: - * @macros: - * - OUTPUT_ENABLED + * - render:: * @note * Each functor provides 1D/2D/3D operator() overloads so the same object works * with `mesh.rangeActiveCells()` of any dimension (the range policy selects the @@ -24,26 +22,27 @@ #include "global.h" #include "arch/kokkos_aliases.h" +#include "utils/numeric.h" #include -namespace kernel { +namespace render { using namespace ntt; /** * @brief F(.., co) = sqrt(F(.., c0)^2 + F(.., c1)^2 + F(.., c2)^2) */ - template + template class RenderMagnitude3_kernel { - ndfield_t F; - const std::uint8_t c0, c1, c2, co; + ndfield_t F; + const uint8_t c0, c1, c2, co; public: RenderMagnitude3_kernel(const ndfield_t& f, - std::uint8_t a, - std::uint8_t b, - std::uint8_t c, - std::uint8_t o) + uint8_t a, + uint8_t b, + uint8_t c, + uint8_t o) : F { f } , c0 { a } , c1 { b } @@ -62,7 +61,7 @@ namespace kernel { Inline void operator()(cellidx_t i1, cellidx_t i2, cellidx_t i3) const { const real_t v0 = F(i1, i2, i3, c0), v1 = F(i1, i2, i3, c1), - v2 = F(i1, i2, i3, c2); + v2 = F(i1, i2, i3, c2); F(i1, i2, i3, co) = math::sqrt(v0 * v0 + v1 * v1 + v2 * v2); } }; @@ -70,13 +69,13 @@ namespace kernel { /** * @brief F(.., co) = F(.., ci) (move one component into the render slot) */ - template + template class RenderPickComp_kernel { - ndfield_t F; - const std::uint8_t ci, co; + ndfield_t F; + const uint8_t ci, co; public: - RenderPickComp_kernel(const ndfield_t& f, std::uint8_t i, std::uint8_t o) + RenderPickComp_kernel(const ndfield_t& f, uint8_t i, uint8_t o) : F { f } , ci { i } , co { o } {} @@ -97,15 +96,13 @@ namespace kernel { /** * @brief F(.., cnum) = (F(.., cden) != 0) ? F(.., cnum) / F(.., cden) : 0 */ - template + template class RenderDivideComp_kernel { - ndfield_t F; - const std::uint8_t cnum, cden; + ndfield_t F; + const uint8_t cnum, cden; public: - RenderDivideComp_kernel(const ndfield_t& f, - std::uint8_t num, - std::uint8_t den) + RenderDivideComp_kernel(const ndfield_t& f, uint8_t num, uint8_t den) : F { f } , cnum { num } , cden { den } {} @@ -131,18 +128,18 @@ namespace kernel { * @note SR bulk-speed magnitude: the three mass-weighted flux components are * each normalized by Rho before the Euclidean norm, in one pass. */ - template + template class RenderVmagByRho_kernel { - ndfield_t F; - const std::uint8_t c0, c1, c2, crho, co; + ndfield_t F; + const uint8_t c0, c1, c2, crho, co; public: RenderVmagByRho_kernel(const ndfield_t& f, - std::uint8_t a, - std::uint8_t b, - std::uint8_t c, - std::uint8_t rho, - std::uint8_t o) + uint8_t a, + uint8_t b, + uint8_t c, + uint8_t rho, + uint8_t o) : F { f } , c0 { a } , c1 { b } @@ -163,16 +160,19 @@ namespace kernel { } Inline void operator()(cellidx_t i1, cellidx_t i2) const { - F(i1, i2, co) = mag(F(i1, i2, c0), F(i1, i2, c1), F(i1, i2, c2), - F(i1, i2, crho)); + F(i1, + i2, + co) = mag(F(i1, i2, c0), F(i1, i2, c1), F(i1, i2, c2), F(i1, i2, crho)); } Inline void operator()(cellidx_t i1, cellidx_t i2, cellidx_t i3) const { - F(i1, i2, i3, co) = mag(F(i1, i2, i3, c0), F(i1, i2, i3, c1), - F(i1, i2, i3, c2), F(i1, i2, i3, crho)); + F(i1, i2, i3, co) = mag(F(i1, i2, i3, c0), + F(i1, i2, i3, c1), + F(i1, i2, i3, c2), + F(i1, i2, i3, crho)); } }; -} // namespace kernel +} // namespace render #endif // OUTPUT_RENDER_REDUCE_HPP diff --git a/src/output/render/renderer.cpp b/src/output/render/renderer.cpp index d4f16f145..272bedd3b 100644 --- a/src/output/render/renderer.cpp +++ b/src/output/render/renderer.cpp @@ -65,7 +65,7 @@ namespace out { void Renderer::init(const ntt::SimulationParams& params, const boundaries_t& global_extent) { - m_enabled = false; + m_enabled = false; const auto& td = params.data(); const bool enable = toml::find_or(td, "output", "render", "enable", false); @@ -85,10 +85,9 @@ namespace out { m_width = toml::find_or(td, "output", "render", "width", 1024); m_height = toml::find_or(td, "output", "render", "height", 1024); - // `resolution` is a convenience that forces a square frame (width == height), - // the natural shape for a dome master. - const int resolution = toml::find_or(td, "output", "render", - "resolution", 0); + // `resolution` is a convenience that forces a square frame (width == + // height), the natural shape for a dome master. + const int resolution = toml::find_or(td, "output", "render", "resolution", 0); if (resolution > 0) { m_width = resolution; m_height = resolution; @@ -100,7 +99,7 @@ namespace out { "render", "early_term_alpha", static_cast(0.99)); - m_n_lut = toml::find_or(td, "output", "render", "n_lut", 256); + m_n_lut = toml::find_or(td, "output", "render", "n_lut", 256); // opaque background color (shows through low-alpha pixels); default black const auto bg = toml::find_or>(td, @@ -124,13 +123,15 @@ namespace out { m_mirror = toml::find_or(td, "output", "render", "mirror", true); // draw the current simulation time in the upper-right corner - m_time_label = toml::find_or(td, "output", "render", "time_label", - false); + m_time_label = toml::find_or(td, "output", "render", "time_label", false); // axes: spine + ticks + labels around the rendered region - m_axes = toml::find_or(td, "output", "render", "axes", false); + m_axes = toml::find_or(td, "output", "render", "axes", false); m_axis_nticks = toml::find_or(td, "output", "render", "axis_ticks", 5); - m_spine_width = toml::find_or(td, "output", "render", "spine_width", + m_spine_width = toml::find_or(td, + "output", + "render", + "spine_width", static_cast(2)); m_global_extent = global_extent; @@ -141,9 +142,12 @@ namespace out { m_has_region = false; { const char* keys[3] = { "x1_lim", "x2_lim", "x3_lim" }; - for (std::size_t d = 0; d < global_extent.size() and d < 3; ++d) { - const auto lim = toml::find_or>( - td, "output", "render", keys[d], std::vector {}); + for (size_t d = 0; d < global_extent.size() and d < 3; ++d) { + const auto lim = toml::find_or>(td, + "output", + "render", + keys[d], + std::vector {}); if (lim.empty()) { continue; } @@ -173,19 +177,26 @@ namespace out { // workstream; warn if asked for here so it does not silently fall back to // the volume camera. { - const bool dome_enable = toml::find_or(td, "output", "render", "dome", - "enable", false); + const bool dome_enable = + toml::find_or(td, "output", "render", "dome", "enable", false); if (dome_enable) { if (global_extent.size() != 2) { raise::Warning("output.render.dome is 2D-only for now; ignoring", HERE); } else { - m_dome.enabled = true; - const real_t fov = toml::find_or(td, "output", "render", "dome", - "fov", static_cast(180)); + m_dome.enabled = true; + const real_t fov = toml::find_or(td, + "output", + "render", + "dome", + "fov", + static_cast(180)); m_dome.theta_max = HALF * fov * static_cast(constant::PI) / static_cast(180); - const auto proj = toml::find_or(td, "output", "render", - "dome", "projection", + const auto proj = toml::find_or(td, + "output", + "render", + "dome", + "projection", "equidistant"); if (proj == "gnomonic") { m_dome.law = DomeMap::Gnomonic; @@ -206,10 +217,11 @@ namespace out { if (m_dome.law == DomeMap::Gnomonic) { const real_t cap = static_cast(89.0 * constant::PI / 180.0); if (m_dome.theta_max >= cap) { - raise::Warning("output.render.dome: 'gnomonic' needs fov < 180 deg " - "(a flat plane cannot reach the dome horizon); " - "capping the half-FOV at 89 deg", - HERE); + raise::Warning( + "output.render.dome: 'gnomonic' needs fov < 180 deg " + "(a flat plane cannot reach the dome horizon); " + "capping the half-FOV at 89 deg", + HERE); m_dome.theta_max = cap; } } @@ -220,7 +232,12 @@ namespace out { m_dome.cx = HALF * (global_extent[0].first + global_extent[0].second); m_dome.cy = HALF * (global_extent[1].first + global_extent[1].second); const auto ctr = toml::find_or>( - td, "output", "render", "dome", "center", std::vector {}); + td, + "output", + "render", + "dome", + "center", + std::vector {}); if (ctr.size() == 2) { m_dome.cx = ctr[0]; m_dome.cy = ctr[1]; @@ -230,8 +247,7 @@ namespace out { HERE); } const real_t rdef = HALF * std::min(Lx, Ly); - m_dome.R = toml::find_or(td, "output", "render", "dome", - "radius", rdef); + m_dome.R = toml::find_or(td, "output", "render", "dome", "radius", rdef); if (m_dome.R <= ZERO) { m_dome.R = rdef; } @@ -247,9 +263,13 @@ namespace out { { const auto al = toml::find_or>( - td, "output", "render", "axis_labels", std::vector {}); + td, + "output", + "render", + "axis_labels", + std::vector {}); m_axis_labels_set = not al.empty(); - for (std::size_t d = 0; d < al.size() and d < 3; ++d) { + for (size_t d = 0; d < al.size() and d < 3; ++d) { m_axis_labels[d] = al[d]; } // default 2D slice names track the labels (overridden per-metric by Render) @@ -258,16 +278,16 @@ namespace out { } // cadence: mirror output.* (interval in steps; interval_time in sim time) - const auto interval = toml::find_or(td, - "output", - "render", - "interval", - 0u); - const auto interval_time = toml::find_or(td, - "output", - "render", - "interval_time", - -1.0); + auto interval = toml::find_or(td, "output", "render", "interval", 0u); + auto interval_time = toml::find_or(td, + "output", + "render", + "interval_time", + -1.0); + if ((interval == 0) and (interval_time == -1.0)) { + interval = params.template get("output.interval"); + interval_time = params.template get("output.interval_time"); + } m_tracker.init("render", interval, interval_time); /* ---- camera (used by the 3D volume mode; the 2D slice path frames itself @@ -275,26 +295,26 @@ namespace out { // frame the camera on the render region (== the full extent when uncropped) real_t center[3] = { ZERO, ZERO, ZERO }, size[3] = { ZERO, ZERO, ZERO }; real_t maxext = ZERO; - for (std::size_t d = 0; d < m_region.size() and d < 3; ++d) { + for (size_t d = 0; d < m_region.size() and d < 3; ++d) { center[d] = static_cast(0.5) * (m_region[d].first + m_region[d].second); size[d] = m_region[d].second - m_region[d].first; maxext = (size[d] > maxext) ? size[d] : maxext; } - const real_t diag = std::sqrt(size[0] * size[0] + size[1] * size[1] + - size[2] * size[2]); + const real_t diag = std::sqrt( + size[0] * size[0] + size[1] * size[1] + size[2] * size[2]); - const bool ortho = toml::find_or(td, - "output", - "render", - "camera", - "orthographic", - true); + const bool ortho = + toml::find_or(td, "output", "render", "camera", "orthographic", true); // `mode` overrides the `orthographic` flag: "orthographic" | "perspective" | // "dome". The dome is a fulldome azimuthal-equidistant fisheye from an // INTERIOR eye (the box center by default) -- see Metadomain::Render (3D). - const auto cam_mode = toml::find_or( - td, "output", "render", "camera", "mode", std::string {}); + const auto cam_mode = toml::find_or(td, + "output", + "render", + "camera", + "mode", + std::string {}); int projection = ortho ? CameraDevice::Ortho : CameraDevice::Perspective; if (cam_mode == "dome") { projection = CameraDevice::Dome; @@ -309,27 +329,31 @@ namespace out { HERE); } const bool is_dome = (projection == CameraDevice::Dome); - const real_t dome_fov = toml::find_or( - td, "output", "render", "camera", "dome_fov", static_cast(180)); - auto pos = toml::find_or>(td, + const real_t dome_fov = toml::find_or(td, + "output", + "render", + "camera", + "dome_fov", + static_cast(180)); + auto pos = toml::find_or>(td, "output", "render", "camera", "position", std::vector {}); - auto look = toml::find_or>(td, + auto look = toml::find_or>(td, "output", "render", "camera", "look_at", std::vector {}); - auto up = toml::find_or>(td, + auto up = toml::find_or>(td, "output", "render", "camera", "up", std::vector {}); - const real_t fov = toml::find_or(td, + const real_t fov = toml::find_or(td, "output", "render", "camera", @@ -423,27 +447,31 @@ namespace out { } m_camera_dev.dome_half_fov = hf; // a fisheye disk needs a square frame; the kernel uses the full-frame ndc - m_camera_dev.aspect = ONE; - m_camera_dev.half_w = m_camera_dev.half_h; + m_camera_dev.aspect = ONE; + m_camera_dev.half_w = m_camera_dev.half_h; if (m_width != m_height) { raise::Warning("output.render.camera.mode='dome' wants width == height " "for a circular dome master; the fisheye disk will be " "elliptical otherwise", HERE); } - // spherical far-clip: each ray stops at `dome_radius` from the eye, so the - // sampled region is a half-ball (hemisphere) instead of the whole box -> - // uniform path length, no box corner/edge projection artifacts. Default = - // the largest sphere centered in the box (half the shortest side). A value - // of 0 disables the clip (march to the box boundary); a negative value - // also selects the default. + // spherical far-clip: each ray stops at `dome_radius` from the eye, so + // the sampled region is a half-ball (hemisphere) instead of the whole box + // -> uniform path length, no box corner/edge projection artifacts. + // Default = the largest sphere centered in the box (half the shortest + // side). A value of 0 disables the clip (march to the box boundary); a + // negative value also selects the default. real_t insc = static_cast(1e30); - for (std::size_t d = 0; d < m_region.size() and d < 3; ++d) { + for (size_t d = 0; d < m_region.size() and d < 3; ++d) { insc = (size[d] < insc) ? size[d] : insc; } - insc *= HALF; - real_t domeR = toml::find_or(td, "output", "render", "camera", - "dome_radius", insc); + insc *= HALF; + real_t domeR = toml::find_or(td, + "output", + "render", + "camera", + "dome_radius", + insc); if (domeR < ZERO) { domeR = insc; } @@ -457,15 +485,21 @@ namespace out { m_eye_base[d] = m_camera_dev.eye[d]; } { - const auto vel = toml::find_or>( - td, "output", "render", "camera_velocity", std::vector {}); - for (std::size_t d = 0; d < vel.size() and d < 3; ++d) { + const auto vel = toml::find_or>(td, + "output", + "render", + "camera_velocity", + std::vector {}); + for (size_t d = 0; d < vel.size() and d < 3; ++d) { m_cam_vel[d] = vel[d]; } m_cam_moving = (m_cam_vel[0] != ZERO) or (m_cam_vel[1] != ZERO) or (m_cam_vel[2] != ZERO); - m_cam_t0 = toml::find_or(td, "output", "render", - "camera_start_time", 0.0); + m_cam_t0 = toml::find_or(td, + "output", + "render", + "camera_start_time", + 0.0); if (m_cam_moving and not m_has_region and global_extent.size() == 2) { raise::Warning("output.render.camera_velocity set without x{1,2}_lim: " "the 2D window will pan off the domain. Set a region to " @@ -483,26 +517,26 @@ namespace out { toml::array {}); for (const auto& sc : scenes_arr) { Scene scene; - scene.field = toml::find_or(sc, "field", ""); + scene.field = toml::find_or(sc, "field", ""); scene.prefix = toml::find_or(sc, "prefix", scene.field + "_"); if (scene.field.empty()) { raise::Warning("output.render scene with no field; skipping", HERE); continue; } - scene.label = toml::find_or(sc, "label", scene.field); - scene.ticks = toml::find_or>(sc, - "colorbar_ticks", - std::vector {}); + scene.label = toml::find_or(sc, "label", scene.field); + scene.ticks = toml::find_or>(sc, + "colorbar_ticks", + std::vector {}); // overlay the B-field-line tubes inside this scene's volume; a dedicated // `field = "fieldlines"` scene renders the tubes standalone (no volume). scene.show_fieldlines = toml::find_or(sc, "fieldlines", false) or (scene.field == "fieldlines"); - scene.tf.vmin = toml::find_or(sc, "min", ZERO); - scene.tf.vmax = toml::find_or(sc, "max", ONE); + scene.tf.vmin = toml::find_or(sc, "min", ZERO); + scene.tf.vmax = toml::find_or(sc, "max", ONE); scene.tf.log_scale = toml::find_or(sc, "log", false); scene.tf.n_lut = m_n_lut; const auto colormap = toml::find_or(sc, "colormap", "viridis"); - scene.tf.colormap = colormap; + scene.tf.colormap = colormap; // alpha control points: array of [position, alpha] pairs const auto alpha_raw = toml::find_or>>( sc, @@ -514,11 +548,14 @@ namespace out { alpha_pts.push_back({ p[0], p[1] }); } } - scene.tf.lut = buildLUT(colormap, m_n_lut, alpha_pts); + scene.tf.lut = buildLUT(colormap, m_n_lut, alpha_pts); // opaque companion LUT (alpha == 1) for the flat 2D slice rasterizer scene.tf.lut_opaque = buildLUT(colormap, m_n_lut, - { { ZERO, ONE }, { ONE, ONE } }); + { + { ZERO, ONE }, + { ONE, ONE } + }); m_scenes.push_back(std::move(scene)); } @@ -535,32 +572,50 @@ namespace out { for (const auto& s : m_scenes) { any_fl = any_fl or s.show_fieldlines; } - m_fieldlines.enable = toml::find_or(td, "output", "render", - "fieldlines", "enable", false) or - any_fl; + m_fieldlines.enable = + toml::find_or(td, "output", "render", "fieldlines", "enable", false) or + any_fl; if (m_fieldlines.enable) { if (m_global_extent.size() != 2 and m_global_extent.size() != 3) { - raise::Warning("output.render.fieldlines needs a 2D or 3D run; ignoring", - HERE); + raise::Warning( + "output.render.fieldlines needs a 2D or 3D run; ignoring", + HERE); m_fieldlines.enable = false; } else { // 3D -> traced tubes inside the volume; 2D -> flux-function contours - auto& fl = m_fieldlines; - fl.field = toml::find_or(td, "output", "render", - "fieldlines", "field", "B"); - fl.bin = toml::find_or(td, "output", "render", "fieldlines", - "bin", 4); - fl.bin = (fl.bin < 1) ? 1 : ((fl.bin > 16) ? 16 : fl.bin); - fl.seed_px = toml::find_or(td, "output", "render", "fieldlines", - "seed_px", static_cast(8)); - fl.tube_px = toml::find_or(td, "output", "render", "fieldlines", - "tube_px", static_cast(2)); - fl.colormap = toml::find_or(td, "output", "render", - "fieldlines", "colormap", + auto& fl = m_fieldlines; + fl.field = toml::find_or(td, + "output", + "render", + "fieldlines", + "field", + "B"); + fl.bin = toml::find_or(td, "output", "render", "fieldlines", "bin", 4); + fl.bin = (fl.bin < 1) ? 1 : ((fl.bin > 16) ? 16 : fl.bin); + fl.seed_px = toml::find_or(td, + "output", + "render", + "fieldlines", + "seed_px", + static_cast(8)); + fl.tube_px = toml::find_or(td, + "output", + "render", + "fieldlines", + "tube_px", + static_cast(2)); + fl.colormap = toml::find_or(td, + "output", + "render", + "fieldlines", + "colormap", "inferno"); // optional monochrome color [r,g,b]; overrides the colormap when set - fl.color = toml::find_or>(td, "output", "render", - "fieldlines", "color", + fl.color = toml::find_or>(td, + "output", + "render", + "fieldlines", + "color", std::vector {}); if (not fl.color.empty() and fl.color.size() != 3) { raise::Warning("output.render.fieldlines.color must have 3 entries " @@ -568,23 +623,36 @@ namespace out { HERE); fl.color.clear(); } - fl.log_scale = toml::find_or(td, "output", "render", "fieldlines", - "log", false); - fl.vmin = toml::find_or(td, "output", "render", "fieldlines", - "min", ZERO); - fl.vmax = toml::find_or(td, "output", "render", "fieldlines", - "max", ZERO); - fl.step_frac = toml::find_or(td, "output", "render", "fieldlines", - "step_frac", static_cast(0.5)); - fl.max_steps = toml::find_or(td, "output", "render", "fieldlines", - "max_steps", 4000); - fl.max_len_frac = toml::find_or(td, "output", "render", - "fieldlines", "max_length", + fl.log_scale = + toml::find_or(td, "output", "render", "fieldlines", "log", false); + fl.vmin = toml::find_or(td, "output", "render", "fieldlines", "min", ZERO); + fl.vmax = toml::find_or(td, "output", "render", "fieldlines", "max", ZERO); + fl.step_frac = toml::find_or(td, + "output", + "render", + "fieldlines", + "step_frac", + static_cast(0.5)); + fl.max_steps = toml::find_or(td, + "output", + "render", + "fieldlines", + "max_steps", + 4000); + fl.max_len_frac = toml::find_or(td, + "output", + "render", + "fieldlines", + "max_length", static_cast(3)); - fl.seed_max = toml::find_or(td, "output", "render", "fieldlines", - "seed_max", 4096); - fl.levels = toml::find_or(td, "output", "render", "fieldlines", - "levels", 16); + fl.seed_max = toml::find_or(td, + "output", + "render", + "fieldlines", + "seed_max", + 4096); + fl.levels = + toml::find_or(td, "output", "render", "fieldlines", "levels", 16); } } @@ -598,10 +666,11 @@ namespace out { } const real_t dt = static_cast( (time > m_cam_t0) ? (time - m_cam_t0) : static_cast(0)); - const real_t shift[3] = { m_cam_vel[0] * dt, m_cam_vel[1] * dt, + const real_t shift[3] = { m_cam_vel[0] * dt, + m_cam_vel[1] * dt, m_cam_vel[2] * dt }; // translate the render region (its width is preserved) - for (std::size_t d = 0; d < m_region.size() and d < 3; ++d) { + for (size_t d = 0; d < m_region.size() and d < 3; ++d) { m_region[d] = { m_region_base[d].first + shift[d], m_region_base[d].second + shift[d] }; } @@ -616,11 +685,11 @@ namespace out { const Scene& scene, timestep_t step, simtime_t time) const { - const std::size_t npix = static_cast(m_width) * - static_cast(m_height); - const std::size_t n = npix * 4; + const size_t npix = static_cast(m_width) * + static_cast(m_height); + const size_t n = npix * 4; // ensure /renders/ exists - const auto dir = m_root / path_t("renders"); + const auto dir = m_root / path_t("renders"); try { if (not std::filesystem::exists(m_root)) { std::filesystem::create_directory(m_root); @@ -634,13 +703,13 @@ namespace out { // composite the premultiplied image over the opaque background: // out = src_premult + (1 - src_alpha) * background, alpha = opaque. std::vector data(n); - for (std::size_t p = 0; p < npix; ++p) { + for (size_t p = 0; p < npix; ++p) { const real_t a = img[p * 4 + 3]; const real_t inv = ONE - a; - data[p * 4 + 0] = quantize(img[p * 4 + 0] + inv * m_background[0]); - data[p * 4 + 1] = quantize(img[p * 4 + 1] + inv * m_background[1]); - data[p * 4 + 2] = quantize(img[p * 4 + 2] + inv * m_background[2]); - data[p * 4 + 3] = 255; + data[p * 4 + 0] = quantize(img[p * 4 + 0] + inv * m_background[0]); + data[p * 4 + 1] = quantize(img[p * 4 + 1] + inv * m_background[1]); + data[p * 4 + 2] = quantize(img[p * 4 + 2] + inv * m_background[2]); + data[p * 4 + 3] = 255; } const auto fname = dir / fmt::format("%s%08lu.png", scene.prefix.c_str(), @@ -648,9 +717,18 @@ namespace out { auto drawBar = [&](uint8_t* buf, int bw, int bh, int span_top, int span_bot) { if (m_colorbar) { - drawColorbar(buf, bw, bh, scene.tf.colormap, scene.tf.vmin, - scene.tf.vmax, scene.tf.log_scale, scene.label, - m_background, scene.ticks, span_top, span_bot); + drawColorbar(buf, + bw, + bh, + scene.tf.colormap, + scene.tf.vmin, + scene.tf.vmax, + scene.tf.log_scale, + scene.label, + m_background, + scene.ticks, + span_top, + span_bot); } }; @@ -658,32 +736,32 @@ namespace out { // centered in the band [0, top_limit] -- i.e. OUTSIDE the plotted data: // above the colorbar (3D / disk) or, for a 2D slice, in the aspect-pad // above the data box (so it never sits inside the simulation axes). - auto drawTimeLabel = [&](uint8_t* buf, int cw, int ch, int right_x, - int top_limit) { - if (not m_time_label) { - return; - } - const int s = cbar_hidden::scale(m_height); - char tbuf[48]; - // fixed-point so it reads e.g. "T = 12345.67" (up to 5 integer digits - // and 2 decimals; more integer digits still print, never truncated) - std::snprintf(tbuf, sizeof(tbuf), "T = %.2f", static_cast(time)); - const std::string str(tbuf); - const int tw = static_cast(str.size()) * 6 * s; - const int pad = 3 * s; - const int tx = right_x - tw - pad; - const int text_h = 7 * s; - int ty = (top_limit - text_h) / 2; - if (ty < pad) { - ty = pad; - } - // contrasting text color (white on a dark background, black on light) - const real_t lum = static_cast(0.299) * m_background[0] + - static_cast(0.587) * m_background[1] + - static_cast(0.114) * m_background[2]; - const uint8_t tc = (lum < HALF) ? 255 : 0; - cbar_hidden::drawText(buf, cw, ch, tx, ty, str, s, tc, tc, tc); - }; + auto drawTimeLabel = + [&](uint8_t* buf, int cw, int ch, int right_x, int top_limit) { + if (not m_time_label) { + return; + } + const int s = cbar_hidden::scale(m_height); + char tbuf[48]; + // fixed-point so it reads e.g. "T = 12345.67" (up to 5 integer digits + // and 2 decimals; more integer digits still print, never truncated) + std::snprintf(tbuf, sizeof(tbuf), "T = %.2f", static_cast(time)); + const std::string str(tbuf); + const int tw = static_cast(str.size()) * 6 * s; + const int pad = 3 * s; + const int tx = right_x - tw - pad; + const int text_h = 7 * s; + int ty = (top_limit - text_h) / 2; + if (ty < pad) { + ty = pad; + } + // contrasting text color (white on a dark background, black on light) + const real_t lum = static_cast(0.299) * m_background[0] + + static_cast(0.587) * m_background[1] + + static_cast(0.114) * m_background[2]; + const uint8_t tc = (lum < HALF) ? 255 : 0; + cbar_hidden::drawText(buf, cw, ch, tx, ty, str, s, tc, tc, tc); + }; // canvas margins: axes (left + bottom) and the colorbar strip (right). // The data region sits at (ml, 0); margins/strip are background-filled. @@ -692,24 +770,23 @@ namespace out { const bool polar = (m_global_extent.size() == 2) and m_slice_polar; // a fisheye dome master (2D or 3D) must stay exactly W x H (its inscribed // circle is the dome), so it takes no axes margins and no outside colorbar. - const bool dome = m_dome_active; + const bool dome = m_dome_active; int ml = 0, mb = 0; out::axesMargins(m_axes and not polar and not dome, m_height, ml, mb); const int strip = (m_colorbar and m_colorbar_outside and not dome) ? colorbarBlockWidth(m_height) : 0; - const int CW = ml + m_width + strip; - const int CH = m_height + mb; + const int CW = ml + m_width + strip; + const int CH = m_height + mb; // 2D-Cartesian data box (== the render region, before the aspect-expansion // that pads the window with background): its top & right edges in // data-region pixels. The axes/spine clamp to it and the time label sits // in the pad above it, so neither includes the empty aspect padding. - const bool cart2d = (m_global_extent.size() == 2) and not polar and - not dome; - int dbox_top = 0; // data box top edge (px from data top) - int dbox_bot = m_height; // data box bottom edge (px) - int dbox_right = m_width; // data box right edge (px from data left) + const bool cart2d = (m_global_extent.size() == 2) and not polar and not dome; + int dbox_top = 0; // data box top edge (px from data top) + int dbox_bot = m_height; // data box bottom edge (px) + int dbox_right = m_width; // data box right edge (px from data left) if (cart2d and m_region.size() >= 2) { const real_t u0 = m_slice_win[0], u1 = m_slice_win[1]; const real_t v0 = m_slice_win[2], v1 = m_slice_win[3]; @@ -717,14 +794,14 @@ namespace out { const real_t dv0 = m_region[1].first; // data box bottom in world (x2) const real_t dv1 = m_region[1].second; // data box top in world (x2) if (u1 > u0) { - int r = static_cast(std::lround( + int r = static_cast(std::lround( static_cast((du1 - u0) / (u1 - u0)) * (m_width - 1))); dbox_right = (r < 0) ? 0 : ((r > m_width) ? m_width : r); } if (v1 > v0) { - int t = static_cast(std::lround( + int t = static_cast(std::lround( static_cast((v1 - dv1) / (v1 - v0)) * (m_height - 1))); - int b = static_cast(std::lround( + int b = static_cast(std::lround( static_cast((v1 - dv0) / (v1 - v0)) * (m_height - 1))); dbox_top = (t < 0) ? 0 : ((t > m_height) ? m_height : t); dbox_bot = (b < 0) ? 0 : ((b > m_height) ? m_height : b); @@ -732,9 +809,9 @@ namespace out { } // time-label anchor: for a 2D slice, the top-right of the data box (label // goes in the pad above it); otherwise the top-right above the colorbar. - const int cbar_top = m_colorbar ? (CH - CH / 2) / 2 : (CH / 4); - const int tl_right = cart2d ? (ml + dbox_right) : (ml + m_width); - const int tl_top = cart2d ? dbox_top : cbar_top; + const int cbar_top = m_colorbar ? (CH - CH / 2) / 2 : (CH / 4); + const int tl_right = cart2d ? (ml + dbox_right) : (ml + m_width); + const int tl_top = cart2d ? dbox_top : cbar_top; // colorbar vertical span: aligned to the actual data domain for a 2D slice // (so it's centered on the data, not the aspect-padded canvas); sentinel // (-1) elsewhere -> drawColorbar centers it on the canvas as before. @@ -748,41 +825,77 @@ namespace out { drawTimeLabel(data.data(), m_width, m_height, tl_right, tl_top); ok = write_png(fname, m_width, m_height, data.data()); } else { - const uint8_t bR = quantize(m_background[0]); - const uint8_t bG = quantize(m_background[1]); - const uint8_t bB = quantize(m_background[2]); - std::vector canvas(static_cast(CW) * CH * 4); - for (std::size_t i = 0; i < canvas.size(); i += 4) { + const uint8_t bR = quantize(m_background[0]); + const uint8_t bG = quantize(m_background[1]); + const uint8_t bB = quantize(m_background[2]); + std::vector canvas(static_cast(CW) * CH * 4); + for (size_t i = 0; i < canvas.size(); i += 4) { canvas[i + 0] = bR; canvas[i + 1] = bG; canvas[i + 2] = bB; canvas[i + 3] = 255; } for (int y = 0; y < m_height; ++y) { - std::copy_n(&data[static_cast(y) * m_width * 4], - static_cast(m_width) * 4, - &canvas[(static_cast(y) * CW + ml) * 4]); + std::copy_n(&data[static_cast(y) * m_width * 4], + static_cast(m_width) * 4, + &canvas[(static_cast(y) * CW + ml) * 4]); } if (m_axes and not dome) { if (m_global_extent.size() == 3) { - out::drawAxes3D(canvas.data(), CW, CH, ml, m_width, m_height, - m_camera_dev, m_region, m_axis_labels, m_background, + out::drawAxes3D(canvas.data(), + CW, + CH, + ml, + m_width, + m_height, + m_camera_dev, + m_region, + m_axis_labels, + m_background, m_axis_nticks); } else if (polar) { - out::drawAxesPolar(canvas.data(), CW, CH, ml, m_width, m_height, - m_slice_win[0], m_slice_win[1], m_slice_win[2], - m_slice_win[3], m_slice_rmin, m_slice_rmax, - m_slice_tmin, m_slice_tmax, m_slice_pmirror, "R", - "Theta", m_background, m_axis_nticks); + out::drawAxesPolar(canvas.data(), + CW, + CH, + ml, + m_width, + m_height, + m_slice_win[0], + m_slice_win[1], + m_slice_win[2], + m_slice_win[3], + m_slice_rmin, + m_slice_rmax, + m_slice_tmin, + m_slice_tmax, + m_slice_pmirror, + "R", + "Theta", + m_background, + m_axis_nticks); } else { // data box (== region, un-expanded) so the spine hugs the domain, // not the aspect-padded window const real_t du0 = m_region[0].first, du1 = m_region[0].second; const real_t dv0 = m_region[1].first, dv1 = m_region[1].second; - out::drawAxes2D(canvas.data(), CW, CH, ml, m_width, m_height, - m_slice_win[0], m_slice_win[1], m_slice_win[2], - m_slice_win[3], du0, du1, dv0, dv1, m_slice_xlabel, - m_slice_ylabel, m_background, m_axis_nticks); + out::drawAxes2D(canvas.data(), + CW, + CH, + ml, + m_width, + m_height, + m_slice_win[0], + m_slice_win[1], + m_slice_win[2], + m_slice_win[3], + du0, + du1, + dv0, + dv1, + m_slice_xlabel, + m_slice_ylabel, + m_background, + m_axis_nticks); } } drawBar(canvas.data(), CW, CH, cbar_span_top, cbar_span_bot); @@ -800,9 +913,9 @@ namespace out { const Scene& scene, timestep_t step, simtime_t time) const { - const std::size_t npix = static_cast(m_width) * - static_cast(m_height); - const std::size_t n = npix * 4; + const size_t npix = static_cast(m_width) * + static_cast(m_height); + const size_t n = npix * 4; // expand a sparse sub-image into a full transparent frame (premultiplied) auto subToFull = [&](const SubImage& s) -> std::vector { @@ -814,12 +927,12 @@ namespace out { if (fx < 0 or fx >= m_width or fy < 0 or fy >= m_height) { continue; } - const std::size_t fi = (static_cast(fy) * m_width + fx) * 4; - const std::size_t si = (static_cast(y) * s.w + x) * 4; - full[fi + 0] = s.rgba[si + 0]; - full[fi + 1] = s.rgba[si + 1]; - full[fi + 2] = s.rgba[si + 2]; - full[fi + 3] = s.rgba[si + 3]; + const size_t fi = (static_cast(fy) * m_width + fx) * 4; + const size_t si = (static_cast(y) * s.w + x) * 4; + full[fi + 0] = s.rgba[si + 0]; + full[fi + 1] = s.rgba[si + 1]; + full[fi + 2] = s.rgba[si + 2]; + full[fi + 3] = s.rgba[si + 3]; } } return full; @@ -851,7 +964,7 @@ namespace out { MPI_Send(hdr, 4, MPI_INT, dest, TAG_HDR, MPI_COMM_WORLD); const int cnt = s.w * s.h * 4; if (cnt > 0) { - std::vector bytes(static_cast(cnt)); + std::vector bytes(static_cast(cnt)); for (int i = 0; i < cnt; ++i) { bytes[i] = quantize(s.rgba[i]); } @@ -868,7 +981,7 @@ namespace out { s.h = hdr[3]; const int cnt = s.w * s.h * 4; if (cnt > 0) { - std::vector bytes(static_cast(cnt)); + std::vector bytes(static_cast(cnt)); MPI_Recv(bytes.data(), cnt, MPI_UNSIGNED_CHAR, @@ -876,7 +989,7 @@ namespace out { TAG_DATA, MPI_COMM_WORLD, MPI_STATUS_IGNORE); - s.rgba.resize(static_cast(cnt)); + s.rgba.resize(static_cast(cnt)); const real_t inv255 = ONE / static_cast(255); for (int i = 0; i < cnt; ++i) { s.rgba[i] = static_cast(bytes[i]) * inv255; @@ -889,7 +1002,7 @@ namespace out { // derives the same global front-to-back order, so no rank needs the others' // images to agree on the composite order. const unsigned long long my_key = static_cast(order_key); - std::vector keys(static_cast(size)); + std::vector keys(static_cast(size)); MPI_Allgather(&my_key, 1, MPI_UNSIGNED_LONG_LONG, @@ -953,8 +1066,8 @@ namespace out { // collapse a merged fragment image into a full premultiplied float frame auto fragToFull = [&](const FragImage& f) -> std::vector { - const std::size_t npix = static_cast(m_width) * - static_cast(m_height); + const size_t npix = static_cast(m_width) * + static_cast(m_height); std::vector full(npix * 4, ZERO); for (int y = 0; y < f.h; ++y) { for (int x = 0; x < f.w; ++x) { @@ -962,18 +1075,18 @@ namespace out { if (fx < 0 or fx >= m_width or fy < 0 or fy >= m_height) { continue; } - const std::size_t p = static_cast(y) * f.w + x; - const uint32_t k0 = f.offs[p], k1 = f.offs[p + 1]; + const size_t p = static_cast(y) * f.w + x; + const uint32_t k0 = f.offs[p], k1 = f.offs[p + 1]; if (k1 <= k0) { continue; } real_t out[4]; out::fragOver(f.depth, f.rgba, k0, k1, out); - const std::size_t fi = (static_cast(fy) * m_width + fx) * 4; - full[fi + 0] = out[0]; - full[fi + 1] = out[1]; - full[fi + 2] = out[2]; - full[fi + 3] = out[3]; + const size_t fi = (static_cast(fy) * m_width + fx) * 4; + full[fi + 0] = out[0]; + full[fi + 1] = out[1]; + full[fi + 2] = out[2]; + full[fi + 3] = out[3]; } } return full; @@ -998,38 +1111,44 @@ namespace out { // per-fragment depth (full real_t, so the cross-rank ordering key is exact) // and premultiplied RGBA (uint8; only this adds ~1 LSB through the tree). auto sendFrag = [&](const FragImage& s, int dest) { - const uint32_t nfrag = s.offs.empty() - ? 0u - : s.offs.back(); - // MPI counts are `int`; the RGBA payload has nfrag*4 elements, so a single - // message overflows int once nfrag > INT_MAX/4. That regime (a 4096^2 - // near-opaque-free dome on many ranks) needs the band-tiling optimization; - // fail loudly here rather than send a negative count. - raise::ErrorIf(nfrag > 536870911u, - "dome A-buffer: per-message fragment count exceeds the MPI " - "int limit (nfrag*4 > INT_MAX). Lower render.resolution " - "(frame-band tiling is a pending optimization).", - HERE); + const uint32_t nfrag = s.offs.empty() ? 0u : s.offs.back(); + // MPI counts are `int`; the RGBA payload has nfrag*4 elements, so a + // single message overflows int once nfrag > INT_MAX/4. That regime (a + // 4096^2 near-opaque-free dome on many ranks) needs the band-tiling + // optimization; fail loudly here rather than send a negative count. + raise::ErrorIf( + nfrag > 536870911u, + "dome A-buffer: per-message fragment count exceeds the MPI " + "int limit (nfrag*4 > INT_MAX). Lower render.resolution " + "(frame-band tiling is a pending optimization).", + HERE); int hdr[5] = { s.x0, s.y0, s.w, s.h, static_cast(nfrag) }; MPI_Send(hdr, 5, MPI_INT, dest, TAG_HDR, MPI_COMM_WORLD); const int np = s.w * s.h; if (np > 0) { - MPI_Send(s.offs.data(), np + 1, MPI_UINT32_T, dest, TAG_OFFS, - MPI_COMM_WORLD); + MPI_Send(s.offs.data(), np + 1, MPI_UINT32_T, dest, TAG_OFFS, MPI_COMM_WORLD); } if (nfrag > 0) { // depth is sent at full real_t precision (NOT downcast to float): the // depth key orders fragments across ranks, and local fragments keep // real_t, so a float round-trip would make cross-rank vs within-rank // ordering disagree at close depths -> a seam at the domain boundary. - MPI_Send(s.depth.data(), static_cast(nfrag), - mpi::get_type(), dest, TAG_DEPTH, MPI_COMM_WORLD); - std::vector bytes(static_cast(nfrag) * 4); - for (std::size_t i = 0; i < bytes.size(); ++i) { + MPI_Send(s.depth.data(), + static_cast(nfrag), + mpi::get_type(), + dest, + TAG_DEPTH, + MPI_COMM_WORLD); + std::vector bytes(static_cast(nfrag) * 4); + for (size_t i = 0; i < bytes.size(); ++i) { bytes[i] = quantize(s.rgba[i]); } - MPI_Send(bytes.data(), static_cast(bytes.size()), - MPI_UNSIGNED_CHAR, dest, TAG_RGBA, MPI_COMM_WORLD); + MPI_Send(bytes.data(), + static_cast(bytes.size()), + MPI_UNSIGNED_CHAR, + dest, + TAG_RGBA, + MPI_COMM_WORLD); } }; auto recvFrag = [&](int src) -> FragImage { @@ -1043,21 +1162,35 @@ namespace out { const uint32_t nfrag = static_cast(hdr[4]); const int np = s.w * s.h; if (np > 0) { - s.offs.resize(static_cast(np) + 1); - MPI_Recv(s.offs.data(), np + 1, MPI_UINT32_T, src, TAG_OFFS, - MPI_COMM_WORLD, MPI_STATUS_IGNORE); + s.offs.resize(static_cast(np) + 1); + MPI_Recv(s.offs.data(), + np + 1, + MPI_UINT32_T, + src, + TAG_OFFS, + MPI_COMM_WORLD, + MPI_STATUS_IGNORE); } if (nfrag > 0) { s.depth.resize(nfrag); - MPI_Recv(s.depth.data(), static_cast(nfrag), mpi::get_type(), - src, TAG_DEPTH, MPI_COMM_WORLD, MPI_STATUS_IGNORE); - std::vector bytes(static_cast(nfrag) * 4); - MPI_Recv(bytes.data(), static_cast(bytes.size()), - MPI_UNSIGNED_CHAR, src, TAG_RGBA, MPI_COMM_WORLD, + MPI_Recv(s.depth.data(), + static_cast(nfrag), + mpi::get_type(), + src, + TAG_DEPTH, + MPI_COMM_WORLD, + MPI_STATUS_IGNORE); + std::vector bytes(static_cast(nfrag) * 4); + MPI_Recv(bytes.data(), + static_cast(bytes.size()), + MPI_UNSIGNED_CHAR, + src, + TAG_RGBA, + MPI_COMM_WORLD, MPI_STATUS_IGNORE); - s.rgba.resize(static_cast(nfrag) * 4); + s.rgba.resize(static_cast(nfrag) * 4); const real_t inv255 = ONE / static_cast(255); - for (std::size_t i = 0; i < s.rgba.size(); ++i) { + for (size_t i = 0; i < s.rgba.size(); ++i) { s.rgba[i] = static_cast(bytes[i]) * inv255; } } diff --git a/src/output/render/renderer.h b/src/output/render/renderer.h index 67a0a58ff..a529ee4e9 100644 --- a/src/output/render/renderer.h +++ b/src/output/render/renderer.h @@ -12,7 +12,6 @@ * - out:: * @macros: * - MPI_ENABLED - * - OUTPUT_ENABLED * @note * The Renderer is intentionally NOT templated on the engine/metric: it owns * only metric-agnostic, host-side work (config parsing, cadence tracking, the @@ -45,7 +44,12 @@ namespace out { // projection: 0 orthographic, 1 perspective (pinhole), 2 dome (fulldome // azimuthal-equidistant fisheye from an interior eye). `orthographic` is // kept for back-compat (== projection 0); the kernel branches on `projection`. - enum Projection { Ortho = 0, Perspective = 1, Dome = 2 }; + enum Projection : uint8_t { + Ortho = 0, + Perspective = 1, + Dome = 2 + }; + real_t eye[3] { ZERO, ZERO, ZERO }; real_t right[3] { ONE, ZERO, ZERO }; real_t up[3] { ZERO, ONE, ZERO }; @@ -57,7 +61,8 @@ namespace out { real_t half_w { ONE }; real_t half_h { ONE }; int projection { Ortho }; - real_t dome_half_fov { static_cast(1.5707963267948966) }; // rad; 180 deg dome => PI/2 + real_t dome_half_fov { static_cast( + 1.5707963267948966) }; // rad; 180 deg dome => PI/2 // dome far-clip: rays stop at this world distance from the eye, so the // sampled region is a half-ball (hemisphere) of this radius instead of the // whole box -> no box corner/edge path-length artifacts. 0 => no clip. @@ -68,7 +73,7 @@ namespace out { * @brief Per-scene transfer function: premultiplied RGBA device LUT + range. */ struct TransferFunction { - array_t lut; // device, (n_lut, 4), premultiplied RGBA + array_t lut; // device, (n_lut, 4), premultiplied RGBA // opaque variant (alpha == 1 everywhere, so the premultiplied entries are // straight RGB) used by the flat 2D slice rasterizer, where a single // per-pixel sample should paint a solid heatmap rather than fade by opacity. @@ -90,26 +95,26 @@ namespace out { * advection. See output/render/fieldlines.h. */ struct FieldLineConfig { - bool enable { false }; // build the geometry this run - std::string field { "B" }; // vector field to trace: "B" | "E" | "J" - int bin { 4 }; // coarsening factor (cells/coarse cell), 2..8 - real_t seed_px { 8 }; // seed lattice spacing in screen pixels - real_t tube_px { 2 }; // tube radius in screen pixels + bool enable { false }; // build the geometry this run + std::string field { "B" }; // vector field to trace: "B" | "E" | "J" + int bin { 4 }; // coarsening factor (cells/coarse cell), 2..8 + real_t seed_px { 8 }; // seed lattice spacing in screen pixels + real_t tube_px { 2 }; // tube radius in screen pixels std::string colormap { "inferno" }; // monochrome override: when this holds 3 entries [r,g,b] in [0,1] the lines // are drawn in that single color instead of the |B| colormap (reads well as // an overlay on a density/other volume). Empty => color by |B|. - std::vector color {}; - bool log_scale { false }; - real_t vmin { ZERO }; // tube color range; vmin>=vmax => auto |B| - real_t vmax { ZERO }; - real_t step_frac { static_cast(0.5) }; // RK4 step / coarse cell - int max_steps { 4000 }; // per-direction integration cap - real_t max_len_frac { static_cast(3) }; // x global box diagonal - int seed_max { 4096 }; // hard cap on seed count (spacing grows to fit) - // 2D only: number of evenly-spaced flux-function contour levels (field lines - // in 2D are iso-contours of the out-of-plane vector potential psi) - int levels { 16 }; + std::vector color; + bool log_scale { false }; + real_t vmin { ZERO }; // tube color range; vmin>=vmax => auto |B| + real_t vmax { ZERO }; + real_t step_frac { static_cast(0.5) }; // RK4 step / coarse cell + int max_steps { 4000 }; // per-direction integration cap + real_t max_len_frac { static_cast(3) }; // x global box diagonal + int seed_max { 4096 }; // hard cap on seed count (spacing grows to fit) + // 2D only: number of evenly-spaced flux-function contour levels (field + // lines in 2D are iso-contours of the out-of-plane vector potential psi) + int levels { 16 }; }; /** @@ -122,15 +127,15 @@ namespace out { * within a (screen-space) line width of a level, colored by |B| = |grad psi|. */ struct ContourSet { - array_t psi; // (n0*n1) flux function, c0-fastest - int n0 { 0 }, n1 { 0 }; - real_t origin0 { ZERO }, origin1 { ZERO }; - real_t dx0 { ONE }, dx1 { ONE }; - real_t dlevel { ONE }; // contour spacing in flux units - real_t psi_ref { ZERO }; // reference (zeroth) level - real_t line_half_px { ONE }; // half contour-line width, pixels - real_t wpp { ONE }; // world units per screen pixel - array_t lut; // opaque colormap, by |B| = |grad psi| + array_t psi; // (n0*n1) flux function, c0-fastest + int n0 { 0 }, n1 { 0 }; + real_t origin0 { ZERO }, origin1 { ZERO }; + real_t dx0 { ONE }, dx1 { ONE }; + real_t dlevel { ONE }; // contour spacing in flux units + real_t psi_ref { ZERO }; // reference (zeroth) level + real_t line_half_px { ONE }; // half contour-line width, pixels + real_t wpp { ONE }; // world units per screen pixel + array_t lut; // opaque colormap, by |B| = |grad psi| int n_lut { 256 }; real_t vmin { ZERO }, vmax { ONE }; // |B| color range bool enabled { false }; @@ -145,10 +150,10 @@ namespace out { * the box spine. Empty (n_seg==0) on ranks no line touches. */ struct TubeSet { - array_t seg; // (n_seg, 8): p0, p1, s0, s1 in world coords + array_t seg; // (n_seg, 8): p0, p1, s0, s1 in world coords int n_seg { 0 }; - real_t radius { ZERO }; // world-space tube radius (ds floor applied) - array_t lut; // premultiplied RGBA, opaque (alpha==1) + real_t radius { ZERO }; // world-space tube radius (ds floor applied) + array_t lut; // premultiplied RGBA, opaque (alpha==1) int n_lut { 256 }; real_t vmin { ZERO }, vmax { ONE }; bool log_scale { false }; @@ -157,8 +162,8 @@ namespace out { // segments in its cell instead of all of them. Bucketing on the coarse // grid is exact because the tube radius is << one coarse cell; a segment is // registered in every cell its radius-padded AABB overlaps. - array_t cell_start; // (ncell+1) prefix offsets into seg_idx - array_t seg_idx; // segment indices, grouped by cell + array_t cell_start; // (ncell+1) prefix offsets into seg_idx + array_t seg_idx; // segment indices, grouped by cell int gnc[3] { 1, 1, 1 }; real_t gorigin[3] { ZERO, ZERO, ZERO }; real_t gdx[3] { ONE, ONE, ONE }; @@ -171,31 +176,38 @@ namespace out { * field with `show_fieldlines` true overlays the tubes inside its volume. */ struct Scene { - std::string field; // "N" | "Bmag" | "Vmag" | "Txy" | "B1" | "fieldlines" ... - std::string prefix; // PNG filename prefix, e.g. "Bmag_" - std::string label; // colorbar title (defaults to field) - std::vector ticks; // explicit colorbar tick values (optional) - bool show_fieldlines { false }; // overlay B-field tubes in the volume - TransferFunction tf; + std::string field; // "N" | "Bmag" | "Vmag" | "Txy" | "B1" | "fieldlines" ... + std::string prefix; // PNG filename prefix, e.g. "Bmag_" + std::string label; // colorbar title (defaults to field) + std::vector ticks; // explicit colorbar tick values (optional) + bool show_fieldlines { false }; // overlay B-field tubes in the volume + TransferFunction tf; }; /** * @brief Fulldome fisheye ("planetarium dome master") projection parameters. * @note When `enabled`, the 2D slice rasterizer ignores the linear world - * window and instead maps each pixel radially: the frame's inscribed circle is - * the dome, a pixel at normalized image radius rho in [0,1] is the dome zenith - * angle theta = rho * theta_max (azimuthal-equidistant image law -- the - * fulldome standard), and `law` picks how theta maps to a world radius r in a - * disk of radius `R` centered at (cx, cy). Pixels outside the inscribed circle - * are left transparent, so the corners are the dome master's black border. - * Cartesian 2D only (see Metadomain::Render); a metric-agnostic POD so the - * (templated) Render can copy it by value into the device kernel. + * window and instead maps each pixel radially: the frame's inscribed circle + * is the dome, a pixel at normalized image radius rho in [0,1] is the dome + * zenith angle theta = rho * theta_max (azimuthal-equidistant image law -- + * the fulldome standard), and `law` picks how theta maps to a world radius r + * in a disk of radius `R` centered at (cx, cy). Pixels outside the inscribed + * circle are left transparent, so the corners are the dome master's black + * border. Cartesian 2D only (see Metadomain::Render); a metric-agnostic POD + * so the (templated) Render can copy it by value into the device kernel. */ struct DomeMap { - enum Law { Equidistant = 0, Gnomonic = 1, Stereographic = 2, Orthographic = 3 }; + enum Law : uint8_t { + Equidistant = 0, + Gnomonic = 1, + Stereographic = 2, + Orthographic = 3 + }; + bool enabled { false }; int law { Equidistant }; - real_t theta_max { static_cast(1.5707963267948966) }; // dome half-FOV (rad) + real_t theta_max { static_cast( + 1.5707963267948966) }; // dome half-FOV (rad) real_t cx { ZERO }, cy { ZERO }; // world center of the cutout real_t R { ONE }; // world radius of the cutout }; @@ -210,7 +222,7 @@ namespace out { struct SubImage { int x0 { 0 }, y0 { 0 }; // top-left pixel in the full frame int w { 0 }, h { 0 }; // bbox size in pixels (0 => empty) - std::vector rgba; // w*h*4 premultiplied, pixel-major + std::vector rgba; // w*h*4 premultiplied, pixel-major }; /** @@ -226,8 +238,8 @@ namespace out { * fragment per covered pixel. See composite.h::mergeFrag / fragOver. */ struct FragImage { - int x0 { 0 }, y0 { 0 }; // top-left pixel in the full frame - int w { 0 }, h { 0 }; // bbox size in pixels (0 => empty) + int x0 { 0 }, y0 { 0 }; // top-left pixel in the full frame + int w { 0 }, h { 0 }; // bbox size in pixels (0 => empty) // per-pixel prefix offsets into depth/rgba, length w*h+1 (offs[0] == 0). std::vector offs; std::vector depth; // n_frag entries, ascending within each pixel @@ -281,14 +293,14 @@ namespace out { simtime_t time) const; /** - * @brief Depth-resolved (A-buffer) composite for the interior-eye dome, then - * write the PNG. + * @brief Depth-resolved (A-buffer) composite for the interior-eye dome, + * then write the PNG. * @param frag this rank's sparse screen-space fragment image (depth + RGBA) * @param scene the scene being written * @param step current timestep (for the filename cycle number) * @param time current simulation time (drawn as a corner label if enabled) - * @note Order-independent: the tree reduce merges depth-sorted fragment lists - * (associative + commutative), so no global rank order is needed. The + * @note Order-independent: the tree reduce merges depth-sorted fragment + * lists (associative + commutative), so no global rank order is needed. The * root collapses each pixel's list front-to-back and writes the file. */ void compositeFragAndWrite(FragImage&& frag, @@ -333,8 +345,8 @@ namespace out { } // Optional axis-aligned render region in physical/world coords. Always - // resolved (unset axes default to the full global extent), so the driver can - // use these unconditionally. `hasRegion()` reports whether any axis was + // resolved (unset axes default to the full global extent), so the driver + // can use these unconditionally. `hasRegion()` reports whether any axis was // overridden (e.g. to know a crop is active). `d` in {0,1,2} == {x1,x2,x3}. [[nodiscard]] auto hasRegion() const -> bool { @@ -404,12 +416,12 @@ namespace out { real_t v1, const std::string& xlabel, const std::string& ylabel) { - m_slice_win[0] = u0; - m_slice_win[1] = u1; - m_slice_win[2] = v0; - m_slice_win[3] = v1; - m_slice_xlabel = xlabel; - m_slice_ylabel = ylabel; + m_slice_win[0] = u0; + m_slice_win[1] = u1; + m_slice_win[2] = v0; + m_slice_win[3] = v1; + m_slice_xlabel = xlabel; + m_slice_ylabel = ylabel; } // Mark the 2D slice as curvilinear (spherical) so the axes are drawn polar: @@ -467,28 +479,28 @@ namespace out { int m_width { 1024 }; int m_height { 1024 }; int m_samples { 400 }; - real_t m_step_size { ZERO }; // world units/step; 0 => derive from samples + real_t m_step_size { ZERO }; // world units/step; 0 => derive from samples real_t m_early_alpha { static_cast(0.99) }; int m_n_lut { 256 }; // opaque background composited under the final image (shows through // low-alpha pixels); defaults to black. real_t m_background[3] { ZERO, ZERO, ZERO }; // draw a colorbar (gradient + value ticks + label) on each PNG - bool m_colorbar { true }; + bool m_colorbar { true }; // draw the colorbar in an extended right margin (outside the render region) // rather than overlaying it on top of the rendered volume - bool m_colorbar_outside { true }; + bool m_colorbar_outside { true }; // 2D slice mode (spherical): mirror the meridional half-plane across the // symmetry axis to render a full disk from one axisymmetric half - bool m_mirror { true }; + bool m_mirror { true }; // draw the current simulation time as a label in the upper-right corner - bool m_time_label { false }; + bool m_time_label { false }; // draw a spine (frame) + axis ticks/labels around the rendered region - bool m_axes { false }; - bool m_axis_labels_set { false }; - int m_axis_nticks { 5 }; - real_t m_spine_width { static_cast(2) }; // 3D spine width (px) - std::string m_axis_labels[3] { "x", "y", "z" }; + bool m_axes { false }; + bool m_axis_labels_set { false }; + int m_axis_nticks { 5 }; + real_t m_spine_width { static_cast(2) }; // 3D spine width (px) + std::string m_axis_labels[3] { "x", "y", "z" }; // global world box (2 or 3 axes); used to project the 3D axes box and to // know the render mode (size 2 => 2D slice, size 3 => 3D volume). boundaries_t m_global_extent; @@ -504,19 +516,19 @@ namespace out { // translate at `m_cam_vel` (world units per unit sim-time) to keep a // propagating feature (e.g. a shock) in frame. `m_eye_base` is the static // camera eye. See updateForTime(). - real_t m_cam_vel[3] { ZERO, ZERO, ZERO }; - simtime_t m_cam_t0 { 0 }; - bool m_cam_moving { false }; - real_t m_eye_base[3] { ZERO, ZERO, ZERO }; + real_t m_cam_vel[3] { ZERO, ZERO, ZERO }; + simtime_t m_cam_t0 { 0 }; + bool m_cam_moving { false }; + real_t m_eye_base[3] { ZERO, ZERO, ZERO }; // 2D slice world window + axis names, set per-frame by the templated Render real_t m_slice_win[4] { ZERO, ONE, ZERO, ONE }; std::string m_slice_xlabel { "x" }; std::string m_slice_ylabel { "y" }; // 2D curvilinear (spherical) slice: draw polar axes instead of Cartesian - bool m_slice_polar { false }; - real_t m_slice_rmin { ZERO }, m_slice_rmax { ONE }; - real_t m_slice_tmin { ZERO }, m_slice_tmax { ONE }; - bool m_slice_pmirror { false }; + bool m_slice_polar { false }; + real_t m_slice_rmin { ZERO }, m_slice_rmax { ONE }; + real_t m_slice_tmin { ZERO }, m_slice_tmax { ONE }; + bool m_slice_pmirror { false }; CameraDevice m_camera_dev; std::vector m_scenes; diff --git a/src/output/render/slice2d.hpp b/src/output/render/slice2d.hpp index 1949b5902..198958746 100644 --- a/src/output/render/slice2d.hpp +++ b/src/output/render/slice2d.hpp @@ -2,11 +2,9 @@ * @file output/render/slice2d.hpp * @brief Header-only Kokkos 2D slice rasterizer (one parallel_for over pixels) * @implements - * - kernel::SliceRaster_kernel + * - render::SliceRaster_kernel * @namespaces: - * - kernel:: - * @macros: - * - OUTPUT_ENABLED + * - render:: * @note * The 2D counterpart of the volume ray-march: a 2D simulation has no depth to * integrate, so each screen pixel is a single inverse-mapped sample of the @@ -34,11 +32,10 @@ #include "global.h" #include "arch/kokkos_aliases.h" -#include "traits/metric.h" #include "output/render/renderer.h" -namespace kernel { +namespace render { using namespace ntt; template @@ -106,7 +103,7 @@ namespace kernel { const real_t line_vmin, line_vmax; const bool line_on; - const bool heatmap_on; + const bool heatmap_on; array_t image; // output, (bw*bh, 4) premultiplied RGBA @@ -212,16 +209,16 @@ namespace kernel { // bilinear sample of the prepared scalar at continuous code coords // (cc1, cc2), reading the ghost halo for corners just outside the box. Inline auto sample(real_t cc1, real_t cc2) const -> real_t { - const real_t g0 = cc1 - HALF; // cell-center continuous index - const real_t g1 = cc2 - HALF; - const real_t f0 = math::floor(g0); - const real_t f1 = math::floor(g1); - const real_t t0 = g0 - f0; - const real_t t1 = g1 - f1; - int b0 = static_cast(f0) + static_cast(N_GHOSTS); - int b1 = static_cast(f1) + static_cast(N_GHOSTS); - b0 = (b0 < 0) ? 0 : ((b0 > ext0 - 2) ? ext0 - 2 : b0); - b1 = (b1 < 0) ? 0 : ((b1 > ext1 - 2) ? ext1 - 2 : b1); + const real_t g0 = cc1 - HALF; // cell-center continuous index + const real_t g1 = cc2 - HALF; + const real_t f0 = math::floor(g0); + const real_t f1 = math::floor(g1); + const real_t t0 = g0 - f0; + const real_t t1 = g1 - f1; + int b0 = static_cast(f0) + static_cast(N_GHOSTS); + int b1 = static_cast(f1) + static_cast(N_GHOSTS); + b0 = (b0 < 0) ? 0 : ((b0 > ext0 - 2) ? ext0 - 2 : b0); + b1 = (b1 < 0) ? 0 : ((b1 > ext1 - 2) ? ext1 - 2 : b1); const real_t c00 = Fld(b0, b1, comp); const real_t c10 = Fld(b0 + 1, b1, comp); const real_t c01 = Fld(b0, b1 + 1, comp); @@ -293,15 +290,14 @@ namespace kernel { const int c0 = static_cast(math::floor((x - lg0) / ldx0)); const int c1 = static_cast(math::floor((z - lg1) / ldx1)); const int c2 = static_cast(math::floor((ZERO - lg2) / ldx2)); - if (c0 < 0 or c0 >= lgnc0 or c1 < 0 or c1 >= lgnc1 or c2 < 0 or - c2 >= lgnc2) { + if (c0 < 0 or c0 >= lgnc0 or c1 < 0 or c1 >= lgnc1 or c2 < 0 or c2 >= lgnc2) { return false; } - const int lin = (c2 * lgnc1 + c1) * lgnc0 + c0; - const int kb = lcell_start(lin); - const int ke = lcell_start(lin + 1); - real_t best = line_r2; - bool hit = false; + const int lin = (c2 * lgnc1 + c1) * lgnc0 + c0; + const int kb = lcell_start(lin); + const int ke = lcell_start(lin + 1); + real_t best = line_r2; + bool hit = false; for (int k = kb; k < ke; ++k) { const int s = lseg_idx(k); const real_t ax = lseg(s, 0), az = lseg(s, 1); // (X, Z) in slots 0,1 @@ -324,9 +320,8 @@ namespace kernel { } Inline void operator()(cellidx_t lpx, cellidx_t lpy) const { - const auto pix = static_cast(lpy) * - static_cast(bw) + - static_cast(lpx); + const auto pix = static_cast(lpy) * static_cast(bw) + + static_cast(lpx); const int gpx = bx0 + static_cast(lpx); const int gpy = by0 + static_cast(lpy); // default transparent @@ -358,27 +353,27 @@ namespace kernel { const real_t phi = math::atan2(dyp, dxp); const real_t theta = rho * dome.theta_max; // dome zenith angle // normalized world radius fr = r / R for the chosen plane<->dome law - real_t fr; + real_t fr; if (dome.law == out::DomeMap::Gnomonic) { const real_t tm = math::tan(dome.theta_max); - fr = (tm > ZERO) ? (math::tan(theta) / tm) : rho; + fr = (tm > ZERO) ? (math::tan(theta) / tm) : rho; } else if (dome.law == out::DomeMap::Stereographic) { const real_t tm = math::tan(HALF * dome.theta_max); - fr = (tm > ZERO) ? (math::tan(HALF * theta) / tm) : rho; + fr = (tm > ZERO) ? (math::tan(HALF * theta) / tm) : rho; } else if (dome.law == out::DomeMap::Orthographic) { const real_t sm = math::sin(dome.theta_max); - fr = (sm > ZERO) ? (math::sin(theta) / sm) : rho; + fr = (sm > ZERO) ? (math::sin(theta) / sm) : rho; } else { // Equidistant (fulldome standard): r = R * theta/theta_max fr = rho; } const real_t r = dome.R * fr; - u = dome.cx + r * math::cos(phi); - v = dome.cy + r * math::sin(phi); + u = dome.cx + r * math::cos(phi); + v = dome.cy + r * math::sin(phi); } else { - u = umin + (static_cast(gpx) + HALF) / - static_cast(W) * (umax - umin); - v = vmax - (static_cast(gpy) + HALF) / - static_cast(H) * (vmax - vmin); + u = umin + (static_cast(gpx) + HALF) / static_cast(W) * + (umax - umin); + v = vmax - (static_cast(gpy) + HALF) / static_cast(H) * + (vmax - vmin); } // world -> continuous local code coords, with an optional physical @@ -399,8 +394,7 @@ namespace kernel { } const real_t r = math::sqrt(u * u + v * v); const real_t th = math::atan2(math::abs(u), v); // in [0, pi] - if (region_clip and - (r < rx1lo or r > rx1hi or th < rx2lo or th > rx2hi)) { + if (region_clip and (r < rx1lo or r > rx1hi or th < rx2lo or th > rx2hi)) { return; } cc1 = metric.template convert<1, Crd::Ph, Crd::Cd>(r); @@ -415,7 +409,7 @@ namespace kernel { real_t cr = ZERO, cg = ZERO, cb = ZERO; bool painted = false; if (heatmap_on) { - const real_t s = sample(cc1, cc2); + const real_t s = sample(cc1, cc2); // normalize through the transfer-function range const real_t inv_range = (vhi > vlo) ? (ONE / (vhi - vlo)) : ZERO; real_t uu; @@ -456,23 +450,21 @@ namespace kernel { const real_t gx = (pr - pl) / (TWO * cwpp); const real_t gy = (pup - pd) / (TWO * cwpp); const real_t g = math::sqrt(gx * gx + gy * gy); // |B| - const real_t tlev = (cdlevel > ZERO) ? (psi0 - cpsi_ref) / cdlevel - : ZERO; + const real_t tlev = (cdlevel > ZERO) ? (psi0 - cpsi_ref) / cdlevel : ZERO; const real_t nlev = math::floor(tlev + HALF); // nearest level index - const real_t d_world = math::abs(psi0 - (cpsi_ref + nlev * cdlevel)); - const real_t eps = static_cast(1e-30); + const real_t d_world = math::abs(psi0 - (cpsi_ref + nlev * cdlevel)); + const real_t eps = static_cast(1e-30); const real_t d_screen = (g > eps) ? (d_world / (g * cwpp)) : static_cast(1e30); if (d_screen <= cline_half_px) { const real_t invr = (cvmax > cvmin) ? (ONE / (cvmax - cvmin)) : ZERO; - real_t uu = (g - cvmin) * invr; + real_t uu = (g - cvmin) * invr; if (uu < ZERO) { uu = ZERO; } else if (uu > ONE) { uu = ONE; } - int idx = static_cast(uu * static_cast(cn_lut - 1) + - HALF); + int idx = static_cast(uu * static_cast(cn_lut - 1) + HALF); if (idx < 0) { idx = 0; } else if (idx > cn_lut - 1) { @@ -493,14 +485,14 @@ namespace kernel { const real_t invr = (line_vmax > line_vmin) ? (ONE / (line_vmax - line_vmin)) : ZERO; - real_t uu = (sB - line_vmin) * invr; + real_t uu = (sB - line_vmin) * invr; if (uu < ZERO) { uu = ZERO; } else if (uu > ONE) { uu = ONE; } - int idx = static_cast(uu * static_cast(line_n_lut - 1) + - HALF); + int idx = static_cast( + uu * static_cast(line_n_lut - 1) + HALF); if (idx < 0) { idx = 0; } else if (idx > line_n_lut - 1) { @@ -522,6 +514,6 @@ namespace kernel { } }; -} // namespace kernel +} // namespace render #endif // OUTPUT_RENDER_SLICE2D_HPP diff --git a/src/output/render/transfer_fn.h b/src/output/render/transfer_fn.h index 75759bf1d..3b99a3a74 100644 --- a/src/output/render/transfer_fn.h +++ b/src/output/render/transfer_fn.h @@ -554,11 +554,11 @@ namespace out { b = anchors.rgb[anchors.n - 1][2]; return; } - const real_t x = u * static_cast(anchors.n - 1); - const int i0 = static_cast(x); - const int i1 = (i0 + 1 < anchors.n) ? (i0 + 1) : i0; - const real_t t = x - static_cast(i0); - r = static_cast(anchors.rgb[i0][0]) * (ONE - t) + + const real_t x = u * static_cast(anchors.n - 1); + const int i0 = static_cast(x); + const int i1 = (i0 + 1 < anchors.n) ? (i0 + 1) : i0; + const real_t t = x - static_cast(i0); + r = static_cast(anchors.rgb[i0][0]) * (ONE - t) + static_cast(anchors.rgb[i1][0]) * t; g = static_cast(anchors.rgb[i0][1]) * (ONE - t) + static_cast(anchors.rgb[i1][1]) * t; @@ -607,7 +607,7 @@ namespace out { const real_t u = (n_lut > 1) ? static_cast(i) / static_cast(n_lut - 1) : ZERO; - real_t r, g, b; + real_t r, g, b; colormapRGB(colormap, u, r, g, b); const real_t a = alphaAt(alpha_pts, u); lut_h(i, 0) = r * a; // premultiplied From e5db7cc66ccb956cc79f4d1176532642e3619e74 Mon Sep 17 00:00:00 2001 From: haykh Date: Thu, 24 Sep 2026 19:27:40 -0400 Subject: [PATCH 116/125] minor --- CODEGUIDE.md | 4 ++-- dev/nix/devenv.nix | 22 ++++++++++++++++++++-- 2 files changed, 22 insertions(+), 4 deletions(-) diff --git a/CODEGUIDE.md b/CODEGUIDE.md index cfef82466..fffe53e61 100644 --- a/CODEGUIDE.md +++ b/CODEGUIDE.md @@ -36,7 +36,7 @@ entity │ ├── dependencies.py # deployment scripts on various machines │ ├── generate_template.py # renders `input.default.toml` from `entity.schema.json` │ ├── ideal_tile_size.py # recommends the team tile size for the tiled deposit -│ └── render_preview.py # previews the in-situ renderer geometry from an input file +│ └── render.py # helper tools for the on-the-fly rendering routine ├── src # main code containing all separate submodules │ ├── archetypes # archetypes which can be used by the user in problem generators │ ├── engines # simulation engines @@ -188,4 +188,4 @@ Best practices are also enforced using `clang-tidy`; to generate recommendations * There is no difference between `.h` and `.hpp` files as both indicate C++ header files. As a consistency convention, we use `.h` for common headers which may be included from multiple `.cpp` files (e.g., metrics), while `.hpp` are very specific headers for only a single (or a couple of) .cpp file (e.g. kernels). -* Do assertions on parameters and quantities whenever possible. Outside the kernels, use `raise::Error(message, HERE)` and `raise::ErrorIf(condition, message, HERE)` to throw exceptions. Inside the kernels, use `raise::KernelError(HERE, message, **args)`. To enable compile-time errors, use `static_assert(condition, message)`. The `HERE` keyword is macro that includes the filename and line number in the error message. +* Do assertions on parameters and quantities whenever possible. Outside the kernels, use `raise::Error(message, HERE)` and `raise::ErrorIf(condition, message, HERE)` to throw exceptions. Inside the kernels, use `raise::KernelError(HERE, message)`. To enable compile-time errors, use `static_assert(condition, message)`. The `HERE` keyword is macro that includes the filename and line number in the error message. diff --git a/dev/nix/devenv.nix b/dev/nix/devenv.nix index 8b78c6651..91206dfac 100644 --- a/dev/nix/devenv.nix +++ b/dev/nix/devenv.nix @@ -18,6 +18,10 @@ let gpu = lib.toUpper cfg.gpu; arch = lib.toUpper cfg.arch; + # override with + # devenv shell -O languages.python.package:pkg python312 + py = "314"; + # `shell.nix` imports nixpkgs with `allowUnfree`/`cudaSupport` decided by the # requested backend. devenv instantiates its own `pkgs` before this module is # evaluated, so it cannot be reconfigured from here -- import the same input @@ -127,6 +131,22 @@ in }; }; + languages.python = { + enable = true; + package = pkgs."python${py}"; + + venv = { + enable = true; + requirements = '' + ipykernel + jupyterlab + nt2py + ruff + pyright + ''; + }; + }; + packages = (with nixpkgs; [ zlib @@ -135,8 +155,6 @@ in adios2Pkg kokkosPkg - python314 - cmake-format cmake-lint neocmakelsp From cefb0354ce9522d7a1ae995268598c85062e71b5 Mon Sep 17 00:00:00 2001 From: haykh Date: Thu, 24 Sep 2026 19:27:48 -0400 Subject: [PATCH 117/125] deprecation on team_policy flag --- src/framework/parameters/algorithms.cpp | 33 +++++++++++++++++++------ 1 file changed, 25 insertions(+), 8 deletions(-) diff --git a/src/framework/parameters/algorithms.cpp b/src/framework/parameters/algorithms.cpp index 797cfb4f7..db34af852 100644 --- a/src/framework/parameters/algorithms.cpp +++ b/src/framework/parameters/algorithms.cpp @@ -3,6 +3,7 @@ #include "defaults.h" #include "global.h" +#include "utils/log.h" #include "utils/numeric.h" #include "framework/parameters/parameters.h" @@ -32,12 +33,28 @@ namespace ntt { defaults::current_filters); deposit_enable = toml::find_or(toml_data, "algorithms", "deposit", "enable", true); - deposit_order = static_cast(SHAPE_ORDER); - deposit_team_policy_team_size = toml::find_or(toml_data, - "algorithms", - "deposit", - "team_policy_team_size", - defaults::team_policy_team_size); + deposit_order = static_cast(SHAPE_ORDER); + if ( + toml_data.contains("algorithms") and + toml_data.at("algorithms").contains("deposit") and + toml_data.at("algorithms").at("deposit").contains("team_policy_team_size") and + not toml_data.at("algorithms").at("deposit").contains("tiled_deposit_team_size")) { + deposit_tiled_team_size = toml::find( + toml_data, + "algorithms", + "deposit", + "team_policy_team_size"); + raise::Warning("`algorithms.deposit.team_policy_team_size` is " + "deprecated and will be removed in 1.6+ versions, use " + "`algorithms.deposit.tiled_deposit_team_size` instead", + HERE); + } else { + deposit_tiled_team_size = toml::find_or(toml_data, + "algorithms", + "deposit", + "tiled_deposit_team_size", + defaults::tiled_deposit_team_size); + } fieldsolver_enable = toml::find_or(toml_data, "algorithms", @@ -145,8 +162,8 @@ namespace ntt { params->set("algorithms.deposit.enable", deposit_enable.value()); params->set("algorithms.deposit.order", deposit_order.value()); - params->set("algorithms.deposit.team_policy_team_size", - deposit_team_policy_team_size.value()); + params->set("algorithms.deposit.tiled_deposit_team_size", + deposit_tiled_team_size.value()); params->set("algorithms.fieldsolver.enable", fieldsolver_enable.value()); for (const auto& [key, value] : fieldsolver_stencil_coeffs.value()) { From c8a728560ac56c07c730c7090baa6323e163a646 Mon Sep 17 00:00:00 2001 From: haykh Date: Sun, 27 Sep 2026 18:45:08 -0400 Subject: [PATCH 118/125] changed render params --- CODEGUIDE.md | 1 - entity.schema.json | 1641 ++++++++++++++------------- input.default.toml | 893 +++++++-------- src/framework/parameters/output.cpp | 4 - src/framework/parameters/render.cpp | 0 src/framework/parameters/render.h | 37 + src/output/render/renderer.cpp | 11 +- 7 files changed, 1301 insertions(+), 1286 deletions(-) create mode 100644 src/framework/parameters/render.cpp create mode 100644 src/framework/parameters/render.h diff --git a/CODEGUIDE.md b/CODEGUIDE.md index fffe53e61..7824195c0 100644 --- a/CODEGUIDE.md +++ b/CODEGUIDE.md @@ -148,7 +148,6 @@ In the editor, point it at the `tombi lsp` language server. For VSCode, the exte #:schema ./entity.schema.json ``` -> [!NOTE] > `tombi` replaces `taplo`, which the project used previously and which is no longer maintained. Best practices are also enforced using `clang-tidy`; to generate recommendations for all the files, run `./dev/scripts/tidy.sh --build build_dir` where `build_dir` is the directory where the code was built, or for specific files: `./dev/scripts/tidy.sh --build build_dir --files "(file1|file2).cpp"` or only for the changed files: `./dev/scripts/tidy.sh --build build_dir --changed`. The recommendations will be in the `tidy/` directory. diff --git a/entity.schema.json b/entity.schema.json index 603c13559..6ca79d81d 100644 --- a/entity.schema.json +++ b/entity.schema.json @@ -1417,16 +1417,6 @@ ] } }, - "separate_files": { - "description": "Whether to output each timestep into separate files", - "type": "boolean", - "default": true, - "deprecated": true, - "x-entity": { - "type": "bool", - "deprecated": "starting v1.3.0" - } - }, "fields": { "description": "Field output parameters", "type": "object", @@ -1854,80 +1844,230 @@ } } } + } + } + }, + "checkpoint": { + "description": "Checkpointing parameters", + "type": "object", + "additionalProperties": false, + "x-entity": { + "inferred": [ + { + "name": "is_resuming", + "brief": "Whether the simulation is resuming from a checkpoint", + "type": "bool", + "from": "command-line flag" + }, + { + "name": "start_step", + "brief": "Timestep of the checkpoint used to resume", + "type": "uint", + "from": "automatically determined during restart" + }, + { + "name": "start_time", + "brief": "Time of the checkpoint used to resume", + "type": "float", + "from": "automatically determined during restart" + } + ] + }, + "properties": { + "interval": { + "description": "Number of timesteps between checkpoints", + "type": "integer", + "minimum": 1, + "default": 1000, + "x-entity": { + "type": "uint [> 0]" + } + }, + "interval_time": { + "description": "Physical (code) time interval between checkpoints", + "type": "number", + "default": -1.0, + "x-entity": { + "type": "float [> 0]", + "notes": [ + "When `< 0`, the output is controlled by `interval`" + ] + } + }, + "keep": { + "description": "Number of checkpoints to keep", + "type": "integer", + "minimum": -1, + "default": 2, + "x-entity": { + "type": "int", + "notes": [ + "0 = disable checkpointing", + "-1 = keep all checkpoints" + ] + } + }, + "walltime": { + "description": "Write a checkpoint once after a fixed walltime", + "type": "string", + "pattern": "^$|^[0-9]{2,}:[0-9]{2}:[0-9]{2}$", + "default": "00:00:00", + "x-entity": { + "type": "string", + "notes": [ + "The format is \"HH:MM:SS\"", + "Empty string or \"00:00:00\" disables this functionality", + "Writing checkpoint at walltime does not stop the simulation" + ] + } + }, + "write_path": { + "description": "Parent directory to write checkpoints to", + "type": "string", + "x-entity": { + "type": "string", + "default": "`.ckpt`", + "notes": [ + "The directory is created if it does not exist" + ] + } + }, + "read_path": { + "description": "Parent directory to use when resuming from a checkpoint", + "type": "string", + "x-entity": { + "type": "string", + "default": "inherit `write_path`" + } + } + } + }, + "adios2": { + "description": "ADIOS2 BP5 tuning, applied to both [output] and [checkpoint] writers", + "type": "object", + "additionalProperties": false, + "properties": { + "aggregators_per_node": { + "description": "Number of ADIOS2 aggregators per node", + "type": "integer", + "minimum": 0, + "default": 0, + "x-entity": { + "type": "uint", + "notes": [ + "Set to either MPI ranks/node or NICs/node for best performance\nIf set to 0, will use ADIOS2 default (one aggregator per node)" + ] + } + }, + "max_shm_size": { + "description": "Maximum shared-memory segment size per node, in bytes (BP5 MaxShmSize)", + "type": "integer", + "minimum": 0, + "default": 4294967296, + "x-entity": { + "type": "uint", + "notes": [ + "Lower this on memory-constrained nodes; matches ADIOS2's default" + ] + } + }, + "buffer_chunk_size": { + "description": "Internal serialization buffer chunk size, in bytes (BP5 BufferChunkSize)", + "type": "integer", + "minimum": 0, + "default": 16777216, + "x-entity": { + "type": "uint", + "notes": [ + "Scales with per-rank output volume; matches ADIOS2's default" + ] + } + } + } + }, + "render": { + "description": "In-situ renderer. Renders scalar fields on the GPU and writes PNG images directly to `/renders/` each cadence -- no field data is written to storage, and the result is seamless across MPI domain boundaries.", + "type": "object", + "additionalProperties": false, + "x-entity": { + "notes": [ + "two modes, selected automatically by the simulation dimension:\n- 3D Cartesian (Minkowski): volume ray-march (uses `samples`, `step_size`, `early_term_alpha`, and the [camera] table)\n- 2D (Minkowski, Spherical/QSpherical, and all GR Kerr-Schild): flat slice rasterizer. Cartesian shows the (x, y) plane; spherical shows the meridional (r, theta) half-plane mapped to Cartesian (X = r sin th, Z = r cos th), optionally mirrored (see `mirror`). The `samples`/`step_size`/`early_term_alpha`/[camera] keys are ignored in 2D (one opaque sample per pixel).", + "1D (and 3D non-Cartesian, which does not exist) is a no-op", + "One PNG stream per scene (e.g. a density/|B|/|J| triptych)" + ] + }, + "properties": { + "enable": { + "description": "Toggle for the on-the-fly renderer", + "type": "boolean", + "default": false, + "x-entity": { + "type": "bool" + } + }, + "interval": { + "description": "Number of timesteps between renders", + "type": "integer", + "minimum": 0, + "default": 0, + "x-entity": { + "type": "uint", + "notes": [ + "When `!= 0`, overrides `output.interval`", + "When `== 0`, `interval_time` (or `output.interval`) is used" + ] + } + }, + "interval_time": { + "description": "Physical (code) time interval between renders", + "type": "number", + "default": -1.0, + "x-entity": { + "type": "float", + "notes": [ + "When `< 0`, the output is controlled by `interval`" + ] + } + }, + "width": { + "description": "Image width in pixels (the rendered region; the PNG is wider if a colorbar margin is added, see `colorbar_outside`)", + "type": "integer", + "minimum": 1, + "default": 1024, + "x-entity": { + "type": "int [> 0]" + } + }, + "height": { + "description": "Image height in pixels", + "type": "integer", + "minimum": 1, + "default": 1024, + "x-entity": { + "type": "int [> 0]" + } + }, + "resolution": { + "description": "Convenience: force a square frame (sets width == height == resolution), the natural shape for a dome master. Overrides `width`/`height` when > 0.", + "type": "integer", + "minimum": 0, + "default": 0, + "x-entity": { + "type": "int [> 0]", + "default": "0 (use width/height)" + } }, - "render": { - "description": "In-situ renderer. Renders scalar fields on the GPU and writes PNG images directly to `/renders/` each cadence -- no field data is written to storage, and the result is seamless across MPI domain boundaries.", + "extent": { + "description": "Limit the render region to axis-aligned box in physical/world coordinates. Left unset it spans the full domain. Clamped to the box.", "type": "object", "additionalProperties": false, "x-entity": { "notes": [ - "two modes, selected automatically by the simulation dimension:\n- 3D Cartesian (Minkowski): volume ray-march (uses `samples`, `step_size`, `early_term_alpha`, and the [camera] table)\n- 2D (Minkowski, Spherical/QSpherical, and all GR Kerr-Schild): flat slice rasterizer. Cartesian shows the (x, y) plane; spherical shows the meridional (r, theta) half-plane mapped to Cartesian (X = r sin th, Z = r cos th), optionally mirrored (see `mirror`). The `samples`/`step_size`/`early_term_alpha`/[camera] keys are ignored in 2D (one opaque sample per pixel).", - "1D (and 3D non-Cartesian, which does not exist) is a no-op", - "One PNG stream per scene (e.g. a density/|B|/|J| triptych)" + "3D -> the volume is depth-clipped to this box, the wireframe/axes frame it, and the default camera zooms to it; 2D -> the slice window is framed to it" ] }, "properties": { - "enable": { - "description": "Toggle for the volume renderer", - "type": "boolean", - "default": false, - "x-entity": { - "type": "bool" - } - }, - "interval": { - "description": "Number of timesteps between renders", - "type": "integer", - "minimum": 0, - "default": 0, - "x-entity": { - "type": "uint", - "notes": [ - "When `!= 0`, overrides `output.interval`", - "When `== 0`, `interval_time` (or `output.interval`) is used" - ] - } - }, - "interval_time": { - "description": "Physical (code) time interval between renders", - "type": "number", - "default": -1.0, - "x-entity": { - "type": "float", - "notes": [ - "When `< 0`, the output is controlled by `interval`" - ] - } - }, - "width": { - "description": "Image width in pixels (the rendered region; the PNG is wider if a colorbar margin is added, see `colorbar_outside`)", - "type": "integer", - "minimum": 1, - "default": 1024, - "x-entity": { - "type": "int [> 0]" - } - }, - "height": { - "description": "Image height in pixels", - "type": "integer", - "minimum": 1, - "default": 1024, - "x-entity": { - "type": "int [> 0]" - } - }, - "resolution": { - "description": "Convenience: force a square frame (sets width == height == resolution), the natural shape for a dome master. Overrides `width`/`height` when > 0.", - "type": "integer", - "minimum": 0, - "default": 0, - "x-entity": { - "type": "int [> 0]", - "default": "0 (use width/height)" - } - }, - "x1_lim": { + "x1": { "description": "Axis-aligned render region [lo, hi] along x1, in physical/world coords. Left unset it spans the full domain. Clamped to the box.", "anyOf": [ { @@ -1953,15 +2093,15 @@ "type": "array [size 2]", "default": "[] (full extent)", "notes": [ - "3D -> the volume is depth-clipped to this box, the wireframe/axes frame it, and the default camera zooms to it; 2D -> the slice window is framed to it. For a spherical 2D slice, x1_lim crops the radius r and x2_lim crops the polar angle theta." + "For a spherical 2D slice, x1 crops the radius r." ], "examples": [ - "x1_lim = [-64.0, 64.0]" + "x1 = [-64.0, 64.0]" ] } }, - "x2_lim": { - "description": "Axis-aligned render region [lo, hi] along x2, in physical/world coords. Left unset it spans the full domain. Clamped to the box.", + "x2": { + "description": "Render region [lo, hi] along x2.", "anyOf": [ { "type": "array", @@ -1986,12 +2126,12 @@ "type": "array [size 2]", "default": "[] (full extent)", "notes": [ - "For a spherical 2D slice, x2_lim crops the polar angle theta" + "For a spherical 2D slice, x2 crops the polar angle theta." ] } }, - "x3_lim": { - "description": "Axis-aligned render region [lo, hi] along x3, in physical/world coords. Left unset it spans the full domain. Clamped to the box.", + "x3": { + "description": "Render region [lo, hi] along x3.", "anyOf": [ { "type": "array", @@ -2016,40 +2156,14 @@ "type": "array [size 2]", "default": "[] (full extent)" } - }, - "camera_velocity": { - "description": "Moving view: translate the render region (and, in 3D, the camera) at this velocity in world-units-per-sim-time, to keep a propagating feature (e.g. a shock) in frame. Pair with x{1,2,3}_lim (the moving window). 2D uses the first two components; the motion starts at `camera_start_time`.", - "anyOf": [ - { - "type": "array", - "maxItems": 0 - }, - { - "type": "array", - "minItems": 2, - "maxItems": 3, - "items": { - "type": "number" - } - } - ], - "default": [], - "x-entity": { - "type": "array [size 2 or 3]", - "default": "[] (static view)", - "examples": [ - "camera_velocity = [0.9, 0.0] # pan along +x1 at 0.9 c" - ] - } - }, - "camera_start_time": { - "description": "Sim time at which the view starts moving (static before it, e.g. to let an initial ramp-up finish)", - "type": "number", - "default": 0.0, - "x-entity": { - "type": "float" - } - }, + } + } + }, + "volume": { + "description": "Volume rendering parameters (3D Cartesian only)", + "type": "object", + "additionalProperties": false, + "properties": { "samples": { "description": "Number of ray-march steps across the global box diagonal", "type": "integer", @@ -2086,647 +2200,676 @@ "Pure speed optimization; set to 1.0 to disable early termination" ] } + } + } + }, + "n_lut": { + "description": "Number of entries in the color/opacity lookup table", + "type": "integer", + "exclusiveMinimum": 1, + "default": 256, + "x-entity": { + "type": "int [> 1]" + } + }, + "background": { + "description": "Opaque background RGB (each channel 0..1) shown through transparent/low-opacity pixels; also fills the colorbar margin", + "type": "array", + "minItems": 3, + "maxItems": 3, + "items": { + "type": "number", + "minimum": 0.0, + "maximum": 1.0 + }, + "default": [ + 0.0, + 0.0, + 0.0 + ], + "x-entity": { + "type": "array [size 3]" + } + }, + "colorbar": { + "description": "Draw a colorbar (gradient + value ticks + label) on each PNG", + "type": "boolean", + "default": true, + "x-entity": { + "type": "bool" + } + }, + "colorbar_outside": { + "description": "Draw the colorbar in an added right margin (the PNG becomes wider by a fixed strip) instead of overlaying it on the rendered volume", + "type": "boolean", + "default": true, + "x-entity": { + "type": "bool" + } + }, + "mirror": { + "description": "2D spherical slice only: mirror the meridional half-plane across the symmetry axis to render a full disk from one axisymmetric half. No effect on Cartesian or 3D rendering.", + "type": "boolean", + "default": true, + "x-entity": { + "type": "bool" + } + }, + "time_label": { + "description": "Draw the current simulation time as a label (\"T = \", fixed to 2 decimals) in the upper-right corner of the render region, in a contrasting color, vertically centered between the frame top and the colorbar.", + "type": "boolean", + "default": false, + "x-entity": { + "type": "bool" + } + }, + "axes": { + "description": "Draw a spine (frame) + axis ticks + labels around the rendered region. The PNG gains left/bottom margins (background-filled) for the tick labels and axis names, so they never overlap the data.", + "type": "boolean", + "default": false, + "x-entity": { + "type": "bool", + "notes": [ + "2D Cartesian = a rectangular frame with linear spatial ticks;\n2D spherical = polar axes (an \"R\" radial axis on the symmetry axis with R=0 centered, and a \"Theta\" axis along the curved outline / spine);\n3D = the global box projected to a wireframe with ticks on the three silhouette edges (x bottom, y & z on the left)" + ] + } + }, + "axis_labels": { + "description": "Axis names. 3D uses all three; the 2D slice uses the first two. When unset, the 2D slice defaults to \"x\",\"y\" (Cartesian) or \"X\",\"Z\" (spherical).", + "type": "array", + "maxItems": 3, + "items": { + "type": "string" + }, + "default": [ + "x", + "y", + "z" + ], + "x-entity": { + "type": "array [size <= 3]" + } + }, + "axis_ticks": { + "description": "Target number of ticks per axis (actual count is rounded to nice values)", + "type": "integer", + "minimum": 2, + "default": 5, + "x-entity": { + "type": "int [>= 2]" + } + }, + "spine_width": { + "description": "3D only: target width (pixels) of the box wireframe \"spine\". The spine is drawn inside the ray-march (opaque, depth-occluded by the volume); its width is floored by the ray step, so for a crisper thin line raise `samples` as well.", + "type": "number", + "exclusiveMinimum": 0.0, + "default": 2.0, + "x-entity": { + "type": "float [> 0.0]" + } + }, + "moving_view": { + "description": "Translate the render region (and, in 3D, the camera), to keep a propagating feature (e.g. a shock) in frame. Pair with x{1,2,3} extent to crop the moving window.", + "type": "object", + "additionalProperties": false, + "properties": { + "velocity": { + "description": "Velocity of the moving camera in world units.", + "anyOf": [ + { + "type": "array", + "maxItems": 0 + }, + { + "type": "array", + "minItems": 2, + "maxItems": 3, + "items": { + "type": "number" + } + } + ], + "default": [], + "x-entity": { + "type": "array [size 2 or 3]", + "default": "[] (static view)", + "examples": [ + "velocity = [0.9, 0.0] # pan along +x1 at 0.9 c" + ] + } }, - "n_lut": { - "description": "Number of entries in the color/opacity lookup table", - "type": "integer", - "exclusiveMinimum": 1, - "default": 256, + "start_time": { + "description": "Sim time at which the view starts moving (static before it, e.g. to let an initial ramp-up finish)", + "type": "number", + "default": 0.0, + "x-entity": { + "type": "float" + } + } + } + }, + "camera": { + "description": "Camera (3D volume mode only; ignored by the 2D slice rasterizer). Defaults frame the whole global box from outside, looking down the (1,1,1) diagonal -- the production setup for which the structured composite is provably seamless.", + "type": "object", + "additionalProperties": false, + "properties": { + "mode": { + "description": "Projection mode. Overrides `orthographic` below when set.", + "anyOf": [ + { + "enum": [ + "orthographic", + "perspective", + "dome" + ] + }, + { + "type": "string", + "pattern": "(?i)^(orthographic|perspective|dome)$" + } + ], "x-entity": { - "type": "int [> 1]" + "type": "string", + "default": "(unset -> use `orthographic`)", + "notes": [ + "\"dome\" is a fulldome azimuthal-equidistant fisheye rendered from an INTERIOR eye (the domain center by default), i.e. a 3D planetarium dome master. It uses a depth-resolved (A-buffer) composite that is seamless across a full 3D domain decomposition (unlike ortho/perspective, which need the eye outside the box). Set a square frame (`resolution`, or width == height). `forward` is the dome ZENITH (screen-up defaults to +y for a +z zenith)." + ] } }, - "background": { - "description": "Opaque background RGB (each channel 0..1) shown through transparent/low-opacity pixels; also fills the colorbar margin", + "position": { + "description": "Camera (eye) position in world (physical) coordinates", "type": "array", "minItems": 3, "maxItems": 3, "items": { - "type": "number", - "minimum": 0.0, - "maximum": 1.0 + "type": "number" + }, + "x-entity": { + "type": "array [size 3]", + "default": "box center pushed back ~1.7 box-diagonals along (1, 1, 1);\nfor `mode = \"dome\"`, the domain center (interior eye)" + } + }, + "look_at": { + "description": "Point the camera looks at, in world coordinates (the dome ZENITH target)", + "type": "array", + "minItems": 3, + "maxItems": 3, + "items": { + "type": "number" + }, + "x-entity": { + "type": "array [size 3]", + "default": "box center; for `mode = \"dome\"`, the zenith defaults to +z" + } + }, + "up": { + "description": "Camera up vector (dome: the disk's screen-up)", + "type": "array", + "minItems": 3, + "maxItems": 3, + "items": { + "type": "number" }, "default": [ 0.0, 0.0, - 0.0 + 1.0 ], "x-entity": { - "type": "array [size 3]" + "type": "array [size 3]", + "default": "[0.0, 0.0, 1.0]; for `mode = \"dome\"`, [0.0, 1.0, 0.0]" } }, - "colorbar": { - "description": "Draw a colorbar (gradient + value ticks + label) on each PNG", - "type": "boolean", - "default": true, + "fov": { + "description": "Vertical field of view in degrees (perspective only)", + "type": "number", + "exclusiveMinimum": 0.0, + "default": 35.0, "x-entity": { - "type": "bool" + "type": "float [> 0.0]" } }, - "colorbar_outside": { - "description": "Draw the colorbar in an added right margin (the PNG becomes wider by a fixed strip) instead of overlaying it on the rendered volume", - "type": "boolean", - "default": true, + "dome_fov": { + "description": "Full dome field of view in degrees (dome mode only): the image rim is at dome_fov/2 from the zenith (180 = a full hemisphere down to the horizon).", + "type": "number", + "exclusiveMinimum": 0.0, + "maximum": 360.0, + "default": 180.0, "x-entity": { - "type": "bool" + "type": "float [> 0.0, <= 360.0]" } }, - "mirror": { - "description": "2D spherical slice only: mirror the meridional half-plane across the symmetry axis to render a full disk from one axisymmetric half. No effect on Cartesian or 3D rendering.", - "type": "boolean", - "default": true, + "dome_radius": { + "description": "Dome far-clip radius in world units (dome mode only): each ray stops this far from the eye, so the sampled region is a half-ball (hemisphere) of this radius rather than the whole box -> uniform path length and no box corner/edge projection artifacts. `samples` then counts steps across this radius.", + "type": "number", + "minimum": 0.0, "x-entity": { - "type": "bool" + "type": "float [>= 0.0]", + "default": "the largest sphere centered in the box (half the shortest side), so it touches the face centers and never a corner", + "notes": [ + "0 disables the clip (rays march to the box boundary)" + ] } }, - "time_label": { - "description": "Draw the current simulation time as a label (\"T = \", fixed to 2 decimals) in the upper-right corner of the render region, in a contrasting color, vertically centered between the frame top and the colorbar.", + "ortho_height": { + "description": "Vertical extent of the view in world units (orthographic only)", + "type": "number", + "exclusiveMinimum": 0.0, + "x-entity": { + "type": "float [> 0.0]", + "default": "the global box diagonal (the whole box fits from any angle)" + } + } + } + }, + "dome": { + "description": "Fulldome fisheye (\"planetarium dome master\"). 2D only; a circular image is centered in the frame's inscribed circle with the corners left as the background (the dome master's black border). Set `width == height` (e.g. 4096) for a square master. When enabled, the axes and the outside colorbar strip are suppressed so the PNG stays exactly width x height. Seamless across MPI domains (the pixel->world map is a shared, deterministic function and the tiles stay disjoint). Ignored (with a warning) for 3D.", + "type": "object", + "additionalProperties": false, + "x-entity": { + "notes": [ + "CARTESIAN -- the flat plane is warped radially into the disk; use `fov`/`radius`/`center`/`projection` below.", + "SPHERICAL / GR Kerr-Schild -- the meridional slice is ALREADY a disk, so dome mode only mirrors it to a full disk (see `mirror` above; keep it true) and fits it to the inscribed circle. The `fov`/`radius`/`center`/`projection` keys are ignored (the native (X, Z) meridional map is used, with image radius proportional to the physical radius r, r=0 at the disk center)." + ] + }, + "properties": { + "enable": { + "description": "Build the fisheye dome master instead of the plain slice", "type": "boolean", "default": false, "x-entity": { "type": "bool" } }, - "axes": { - "description": "Draw a spine (frame) + axis ticks + labels around the rendered region. The PNG gains left/bottom margins (background-filled) for the tick labels and axis names, so they never overlap the data.", + "fov": { + "description": "(Cartesian only) Full dome field of view in degrees (image radius maps linearly to the dome zenith angle: the rim is at fov/2)", + "type": "number", + "exclusiveMinimum": 0.0, + "maximum": 180.0, + "default": 180.0, + "x-entity": { + "type": "float [> 0.0, <= 180.0]", + "default": "180.0 # a full hemisphere" + } + }, + "radius": { + "description": "(Cartesian only) World radius of the circular cutout mapped onto the dome", + "type": "number", + "exclusiveMinimum": 0.0, + "x-entity": { + "type": "float [> 0.0]", + "default": "half the shorter domain side (the largest centered disk that fits inside the box)" + } + }, + "center": { + "description": "(Cartesian only) World-space center of the cutout", + "type": "array", + "minItems": 2, + "maxItems": 2, + "items": { + "type": "number" + }, + "x-entity": { + "type": "array [size 2]", + "default": "the domain center" + } + }, + "projection": { + "description": "(Cartesian only) How the dome zenith angle maps to a world radius on the flat slice", + "anyOf": [ + { + "enum": [ + "equidistant", + "gnomonic", + "stereographic", + "orthographic" + ] + }, + { + "type": "string", + "pattern": "(?i)^(equidistant|gnomonic|stereographic|orthographic)$" + } + ], + "default": "equidistant", + "x-entity": { + "type": "string", + "enum": [ + "\"equidistant\" (r proportional to angle; the fulldome image standard -- a straight radial scaling of the cutout)", + "\"gnomonic\" (r ~ tan(angle); the slice as a flat \"ceiling\" tangent to the dome -- straight sim lines stay straight)", + "\"stereographic\" (r ~ tan(angle/2); conformal, preserves shapes)", + "\"orthographic\" (r ~ sin(angle); the slice as seen face-on)" + ] + } + } + } + }, + "fieldlines": { + "description": "Magnetic field lines, drawn from a coarse, MPI-replicated copy of the field so the geometry is global and seamless across domains (the coarsening is what makes this cheap -- no parallel particle advection / flux scan).\n- 3D (Cartesian): traced as solid tubes, colored by |field|, composited inside the volume ray-march so the volume correctly occludes them.\n- 2D (Cartesian): iso-contours of the flux function psi (Bx = d psi/dy, By = -d psi/dx), i.e. the in-plane field lines, colored by |B|.\n- 2D (spherical / Kerr-Schild): traced meridional streamlines of the poloidal (Br, Btheta) field (nt2py style).\nBuilt once per frame and shared by every scene that opts in (per-scene `fieldlines = true`) and by any standalone `field = \"fieldlines\"` scene.", + "type": "object", + "additionalProperties": false, + "properties": { + "enable": { + "description": "Build the field-line geometry this run", "type": "boolean", "default": false, "x-entity": { "type": "bool", "notes": [ - "2D Cartesian = a rectangular frame with linear spatial ticks;\n2D spherical = polar axes (an \"R\" radial axis on the symmetry axis with R=0 centered, and a \"Theta\" axis along the curved outline / spine);\n3D = the global box projected to a wireframe with ticks on the three silhouette edges (x bottom, y & z on the left)" + "implied true if any scene sets `fieldlines = true` or uses `field = \"fieldlines\"`" ] } }, - "axis_labels": { - "description": "Axis names. 3D uses all three; the 2D slice uses the first two. When unset, the 2D slice defaults to \"x\",\"y\" (Cartesian) or \"X\",\"Z\" (spherical).", - "type": "array", - "maxItems": 3, - "items": { - "type": "string" - }, - "default": [ - "x", - "y", - "z" + "field": { + "description": "Vector field to trace", + "anyOf": [ + { + "enum": [ + "B", + "E", + "J" + ] + }, + { + "type": "string", + "pattern": "(?i)^(B|E|J)$" + } ], + "default": "B", "x-entity": { - "type": "array [size <= 3]" + "type": "string" } }, - "axis_ticks": { - "description": "Target number of ticks per axis (actual count is rounded to nice values)", + "bin": { + "description": "Field coarsening factor (simulation cells per coarse cell, per axis)", "type": "integer", - "minimum": 2, - "default": 5, + "minimum": 1, + "maximum": 16, + "default": 4, "x-entity": { - "type": "int [>= 2]" + "type": "int [1..16]", + "notes": [ + "larger = smoother \"morphology\" lines + cheaper replication (the coarse field is ~ N_cells / bin^D floats/rank; D = sim dimension)" + ] } }, - "spine_width": { - "description": "3D only: target width (pixels) of the box wireframe \"spine\". The spine is drawn inside the ray-march (opaque, depth-occluded by the volume); its width is floored by the ray step, so for a crisper thin line raise `samples` as well.", + "seed_px": { + "description": "(3D tubes) Seed-lattice spacing in screen pixels (sets line density)", "type": "number", "exclusiveMinimum": 0.0, - "default": 2.0, + "default": 8, "x-entity": { - "type": "float [> 0.0]" + "type": "float [> 0]", + "notes": [ + "capped by `seed_max`; if seed_px asks for more seeds than that, the spacing grows to fit and seed_px no longer governs" + ] } }, - "camera": { - "description": "Camera (3D volume mode only; ignored by the 2D slice rasterizer). Defaults frame the whole global box from outside, looking down the (1,1,1) diagonal -- the production setup for which the structured composite is provably seamless.", - "type": "object", - "additionalProperties": false, - "properties": { - "mode": { - "description": "Projection mode. Overrides `orthographic` below when set.", - "anyOf": [ - { - "enum": [ - "orthographic", - "perspective", - "dome" - ] - }, - { - "type": "string", - "pattern": "(?i)^(orthographic|perspective|dome)$" - } - ], - "x-entity": { - "type": "string", - "default": "(unset -> use `orthographic`)", - "notes": [ - "\"dome\" is a fulldome azimuthal-equidistant fisheye rendered from an INTERIOR eye (the domain center by default), i.e. a 3D planetarium dome master. It uses a depth-resolved (A-buffer) composite that is seamless across a full 3D domain decomposition (unlike ortho/perspective, which need the eye outside the box). Set a square frame (`resolution`, or width == height). `forward` is the dome ZENITH (screen-up defaults to +y for a +z zenith)." - ] - } - }, - "orthographic": { - "description": "Orthographic (true) or perspective (false) projection", - "type": "boolean", - "default": true, - "x-entity": { - "type": "bool", - "notes": [ - "Orthographic is recommended; the seamless composite is always valid for it. Perspective is only seamless with the eye outside the box." - ] - } - }, - "position": { - "description": "Camera (eye) position in world (physical) coordinates", - "type": "array", - "minItems": 3, - "maxItems": 3, - "items": { - "type": "number" - }, - "x-entity": { - "type": "array [size 3]", - "default": "box center pushed back ~1.7 box-diagonals along (1, 1, 1);\nfor `mode = \"dome\"`, the domain center (interior eye)" - } - }, - "look_at": { - "description": "Point the camera looks at, in world coordinates (the dome ZENITH target)", + "seed_max": { + "description": "(3D tubes) Hard cap on the seed count (lattice is n^3, 2 lines per seed)", + "type": "integer", + "minimum": 1, + "default": 4096, + "x-entity": { + "type": "int [> 0]", + "notes": [ + "lower this for fewer / more widely spaced lines" + ] + } + }, + "levels": { + "description": "(2D contours) Number of evenly-spaced flux-function contour levels", + "type": "integer", + "minimum": 1, + "default": 16, + "x-entity": { + "type": "int [> 0]", + "notes": [ + "evenly-spaced psi levels => line density tracks |B| automatically" + ] + } + }, + "tube_px": { + "description": "Tube radius (3D) / contour line width (2D), in screen pixels", + "type": "number", + "exclusiveMinimum": 0.0, + "default": 2, + "x-entity": { + "type": "float [> 0]" + } + }, + "colormap": { + "description": "Colormap for the field lines (mapped by |B| along each line)", + "$ref": "#/$defs/colormap", + "default": "inferno" + }, + "color": { + "description": "Monochrome override: draw the lines in a single [r,g,b] color (each 0..1) instead of the |B| colormap -- reads well as an overlay on another volume", + "anyOf": [ + { "type": "array", - "minItems": 3, - "maxItems": 3, - "items": { - "type": "number" - }, - "x-entity": { - "type": "array [size 3]", - "default": "box center; for `mode = \"dome\"`, the zenith defaults to +z" - } + "maxItems": 0 }, - "up": { - "description": "Camera up vector (dome: the disk's screen-up)", + { "type": "array", "minItems": 3, "maxItems": 3, "items": { - "type": "number" - }, - "default": [ - 0.0, - 0.0, - 1.0 - ], - "x-entity": { - "type": "array [size 3]", - "default": "[0.0, 0.0, 1.0]; for `mode = \"dome\"`, [0.0, 1.0, 0.0]" - } - }, - "fov": { - "description": "Vertical field of view in degrees (perspective only)", - "type": "number", - "exclusiveMinimum": 0.0, - "default": 35.0, - "x-entity": { - "type": "float [> 0.0]" - } - }, - "dome_fov": { - "description": "Full dome field of view in degrees (dome mode only): the image rim is at dome_fov/2 from the zenith (180 = a full hemisphere down to the horizon).", - "type": "number", - "exclusiveMinimum": 0.0, - "maximum": 360.0, - "default": 180.0, - "x-entity": { - "type": "float [> 0.0, <= 360.0]" - } - }, - "dome_radius": { - "description": "Dome far-clip radius in world units (dome mode only): each ray stops this far from the eye, so the sampled region is a half-ball (hemisphere) of this radius rather than the whole box -> uniform path length and no box corner/edge projection artifacts. `samples` then counts steps across this radius.", - "type": "number", - "minimum": 0.0, - "x-entity": { - "type": "float [>= 0.0]", - "default": "the largest sphere centered in the box (half the shortest side), so it touches the face centers and never a corner", - "notes": [ - "0 disables the clip (rays march to the box boundary)" - ] - } - }, - "ortho_height": { - "description": "Vertical extent of the view in world units (orthographic only)", - "type": "number", - "exclusiveMinimum": 0.0, - "x-entity": { - "type": "float [> 0.0]", - "default": "the global box diagonal (the whole box fits from any angle)" + "type": "number", + "minimum": 0.0, + "maximum": 1.0 } } + ], + "default": [], + "x-entity": { + "type": "array [size 3]", + "default": "[] (empty => color by |B|)", + "examples": [ + "[1.0, 1.0, 1.0] # white field lines" + ] } }, - "dome": { - "description": "Fulldome fisheye (\"planetarium dome master\"). 2D only; a circular image is centered in the frame's inscribed circle with the corners left as the background (the dome master's black border). Set `width == height` (e.g. 4096) for a square master. When enabled, the axes and the outside colorbar strip are suppressed so the PNG stays exactly width x height. Seamless across MPI domains (the pixel->world map is a shared, deterministic function and the tiles stay disjoint). Ignored (with a warning) for 3D.", - "type": "object", - "additionalProperties": false, + "log": { + "description": "Map the tube color range logarithmically", + "type": "boolean", + "default": false, + "x-entity": { + "type": "bool", + "notes": [ + "requires min > 0" + ] + } + }, + "min": { + "description": "Tube color range: lower bound on |field|", + "type": "number", + "default": 0.0, + "x-entity": { + "type": "float", + "notes": [ + "when min >= max, the range is auto-set from |field| along the lines" + ] + } + }, + "max": { + "description": "Tube color range: upper bound on |field|", + "type": "number", + "default": 0.0, "x-entity": { + "type": "float", "notes": [ - "CARTESIAN -- the flat plane is warped radially into the disk; use `fov`/`radius`/`center`/`projection` below.", - "SPHERICAL / GR Kerr-Schild -- the meridional slice is ALREADY a disk, so dome mode only mirrors it to a full disk (see `mirror` above; keep it true) and fits it to the inscribed circle. The `fov`/`radius`/`center`/`projection` keys are ignored (the native (X, Z) meridional map is used, with image radius proportional to the physical radius r, r=0 at the disk center)." + "when min >= max, the range is auto-set from |field| along the lines" ] + } + }, + "step_frac": { + "description": "(3D tubes) RK4 integration step as a fraction of one coarse cell", + "type": "number", + "exclusiveMinimum": 0.0, + "default": 0.5, + "x-entity": { + "type": "float [> 0]" + } + }, + "max_steps": { + "description": "(3D tubes) Per-direction integration-step cap", + "type": "integer", + "minimum": 1, + "default": 4000, + "x-entity": { + "type": "int [> 0]" + } + }, + "max_length": { + "description": "(3D tubes) Maximum line length, in global box diagonals (per direction)", + "type": "number", + "exclusiveMinimum": 0.0, + "default": 3.0, + "x-entity": { + "type": "float [> 0]" + } + } + } + }, + "scene": { + "description": "One scene per scalar field -> one PNG stream. Repeat the table for each.", + "type": "array", + "x-entity": { + "array_of_tables": true + }, + "items": { + "type": "object", + "additionalProperties": false, + "required": [ + "field" + ], + "properties": { + "field": { + "description": "Scalar field to render (a volume render needs a scalar, so vectors are given as a magnitude or a single component)", + "type": "string", + "x-entity": { + "type": "string", + "enum": [ + "(fields): \"{E,B,J}mag\"; \"{E,B,J}{1,2,3}\" or \"{E,B,J}{x,y,z}\"", + "(moments): \"N\", \"Nppc\", \"Rho\", \"Charge\"; \"T{i}{j}\"; \"V{i}\"; \"Vmag\"" + ], + "notes": [ + "\"{E,B,J}mag\" = vector magnitude |.|; \"B1\"/\"Bx\", \"J3\"/\"Jz\", ... = a single (signed) physical component", + "a bare vector (\"E\"/\"B\"/\"J\") is not renderable -- choose a component or the magnitude", + "\"N\"/\"Nppc\" = number / per-cell count, \"Rho\" = mass density, \"Charge\" = charge density", + "\"T{i}{j}\" = one stress-energy component, i,j in {t,x,y,z} or {0,1,2,3} (e.g. \"Txx\", \"Ttt\", \"T0x\"); \"V{i}\" = one bulk-velocity component, i in {x,y,z} or {1,2,3} (e.g. \"Vx\", \"V1\"); \"Vmag\" = bulk-velocity magnitude sqrt(V1^2+V2^2+V3^2)", + "moments follow the engine: SRPIC = tetrad-basis bulk 3-velocity and stress-energy; GRPIC = Eckart-frame 4-velocity (so \"Vt\"/\"V0\" = u^0 = Gamma/alpha is also valid) and contravariant T", + "per-species selection with a \"_\" suffix on moments, e.g. \"N_1\", \"Rho_2\", \"Txy_1_2\", \"V1_3\"; default = all massive species", + "components are signed; pair a symmetric `min`/`max` with a diverging colormap (\"cool2warm\") to center zero", + "\"fieldlines\" renders the magnetic field-line tubes on their own (no scalar volume sampled); see [render.fieldlines] below" + ] + } }, - "properties": { - "enable": { - "description": "Build the fisheye dome master instead of the plain slice", - "type": "boolean", - "default": false, - "x-entity": { - "type": "bool" - } - }, - "fov": { - "description": "(Cartesian only) Full dome field of view in degrees (image radius maps linearly to the dome zenith angle: the rim is at fov/2)", - "type": "number", - "exclusiveMinimum": 0.0, - "maximum": 180.0, - "default": 180.0, - "x-entity": { - "type": "float [> 0.0, <= 180.0]", - "default": "180.0 # a full hemisphere" - } - }, - "radius": { - "description": "(Cartesian only) World radius of the circular cutout mapped onto the dome", - "type": "number", - "exclusiveMinimum": 0.0, - "x-entity": { - "type": "float [> 0.0]", - "default": "half the shorter domain side (the largest centered disk that fits inside the box)" - } - }, - "center": { - "description": "(Cartesian only) World-space center of the cutout", + "prefix": { + "description": "PNG filename prefix; files are `.png`", + "type": "string", + "x-entity": { + "type": "string", + "default": "\"_\"" + } + }, + "label": { + "description": "Colorbar title", + "type": "string", + "x-entity": { + "type": "string", + "default": "`field`" + } + }, + "min": { + "description": "Lower bound of the value range mapped onto the colormap/opacity", + "type": "number", + "default": 0.0, + "x-entity": { + "type": "float" + } + }, + "max": { + "description": "Upper bound of the value range", + "type": "number", + "default": 1.0, + "x-entity": { + "type": "float" + } + }, + "log": { + "description": "Map the value range logarithmically", + "type": "boolean", + "default": false, + "x-entity": { + "type": "bool", + "notes": [ + "Requires min > 0 and max > 0" + ] + } + }, + "colormap": { + "description": "Colormap name", + "$ref": "#/$defs/colormap", + "default": "viridis" + }, + "alpha": { + "description": "Opacity transfer function: [position, opacity] control points, both in [0, 1], piecewise-linear in the normalized value", + "type": "array", + "items": { "type": "array", "minItems": 2, "maxItems": 2, - "items": { - "type": "number" - }, - "x-entity": { - "type": "array [size 2]", - "default": "the domain center" - } - }, - "projection": { - "description": "(Cartesian only) How the dome zenith angle maps to a world radius on the flat slice", - "anyOf": [ + "prefixItems": [ { - "enum": [ - "equidistant", - "gnomonic", - "stereographic", - "orthographic" - ] + "type": "number", + "minimum": 0.0, + "maximum": 1.0 }, { - "type": "string", - "pattern": "(?i)^(equidistant|gnomonic|stereographic|orthographic)$" + "type": "number", + "minimum": 0.0, + "maximum": 1.0 } + ] + }, + "x-entity": { + "type": "array>", + "default": "linear ramp (opacity = normalized value)", + "notes": [ + "Keep the low end near 0 so empty regions stay transparent" ], - "default": "equidistant", - "x-entity": { - "type": "string", - "enum": [ - "\"equidistant\" (r proportional to angle; the fulldome image standard -- a straight radial scaling of the cutout)", - "\"gnomonic\" (r ~ tan(angle); the slice as a flat \"ceiling\" tangent to the dome -- straight sim lines stay straight)", - "\"stereographic\" (r ~ tan(angle/2); conformal, preserves shapes)", - "\"orthographic\" (r ~ sin(angle); the slice as seen face-on)" - ] - } + "examples": [ + "[[0.0, 0.0], [0.3, 0.1], [1.0, 0.7]]" + ] } - } - }, - "scenes": { - "description": "One scene per scalar field -> one PNG stream. Repeat the table for each.", - "type": "array", - "x-entity": { - "array_of_tables": true }, - "items": { - "type": "object", - "additionalProperties": false, - "required": [ - "field" - ], - "properties": { - "field": { - "description": "Scalar field to render (a volume render needs a scalar, so vectors are given as a magnitude or a single component)", - "type": "string", - "x-entity": { - "type": "string", - "enum": [ - "(fields): \"{E,B,J}mag\"; \"{E,B,J}{1,2,3}\" or \"{E,B,J}{x,y,z}\"", - "(moments): \"N\", \"Nppc\", \"Rho\", \"Charge\"; \"T{i}{j}\"; \"V{i}\"; \"Vmag\"" - ], - "notes": [ - "\"{E,B,J}mag\" = vector magnitude |.|; \"B1\"/\"Bx\", \"J3\"/\"Jz\", ... = a single (signed) physical component", - "a bare vector (\"E\"/\"B\"/\"J\") is not renderable -- choose a component or the magnitude", - "\"N\"/\"Nppc\" = number / per-cell count, \"Rho\" = mass density, \"Charge\" = charge density", - "\"T{i}{j}\" = one stress-energy component, i,j in {t,x,y,z} or {0,1,2,3} (e.g. \"Txx\", \"Ttt\", \"T0x\"); \"V{i}\" = one bulk-velocity component, i in {x,y,z} or {1,2,3} (e.g. \"Vx\", \"V1\"); \"Vmag\" = bulk-velocity magnitude sqrt(V1^2+V2^2+V3^2)", - "moments follow the engine: SRPIC = tetrad-basis bulk 3-velocity and stress-energy; GRPIC = Eckart-frame 4-velocity (so \"Vt\"/\"V0\" = u^0 = Gamma/alpha is also valid) and contravariant T", - "per-species selection with a \"_\" suffix on moments, e.g. \"N_1\", \"Rho_2\", \"Txy_1_2\", \"V1_3\"; default = all massive species", - "components are signed; pair a symmetric `min`/`max` with a diverging colormap (\"cool2warm\") to center zero", - "\"fieldlines\" renders the magnetic field-line tubes on their own (no scalar volume sampled); see [output.render.fieldlines] below" - ] - } - }, - "prefix": { - "description": "PNG filename prefix; files are `.png`", - "type": "string", - "x-entity": { - "type": "string", - "default": "\"_\"" - } - }, - "label": { - "description": "Colorbar title", - "type": "string", - "x-entity": { - "type": "string", - "default": "`field`" - } - }, - "min": { - "description": "Lower bound of the value range mapped onto the colormap/opacity", - "type": "number", - "default": 0.0, - "x-entity": { - "type": "float" - } - }, - "max": { - "description": "Upper bound of the value range", - "type": "number", - "default": 1.0, - "x-entity": { - "type": "float" - } - }, - "log": { - "description": "Map the value range logarithmically", - "type": "boolean", - "default": false, - "x-entity": { - "type": "bool", - "notes": [ - "Requires min > 0 and max > 0" - ] - } - }, - "colormap": { - "description": "Colormap name", - "$ref": "#/$defs/colormap", - "default": "viridis" - }, - "alpha": { - "description": "Opacity transfer function: [position, opacity] control points, both in [0, 1], piecewise-linear in the normalized value", - "type": "array", - "items": { - "type": "array", - "minItems": 2, - "maxItems": 2, - "prefixItems": [ - { - "type": "number", - "minimum": 0.0, - "maximum": 1.0 - }, - { - "type": "number", - "minimum": 0.0, - "maximum": 1.0 - } - ] - }, - "x-entity": { - "type": "array>", - "default": "linear ramp (opacity = normalized value)", - "notes": [ - "Keep the low end near 0 so empty regions stay transparent" - ], - "examples": [ - "[[0.0, 0.0], [0.3, 0.1], [1.0, 0.7]]" - ] - } - }, - "colorbar_ticks": { - "description": "Explicit value(s) to label on the colorbar", - "type": "array", - "items": { - "type": "number" - }, - "x-entity": { - "type": "array", - "default": "5 evenly-spaced ticks between min and max", - "notes": [ - "Values outside [min, max] are skipped" - ], - "examples": [ - "[0.0, 0.5, 1.0]" - ] - } - }, - "fieldlines": { - "description": "Overlay the magnetic field-line tubes inside this scene's volume", - "type": "boolean", - "default": false, - "x-entity": { - "type": "bool", - "notes": [ - "requires the [output.render.fieldlines] table below (3D only). A scene with field = \"fieldlines\" instead renders them alone." - ] - } - } - } - } - }, - "fieldlines": { - "description": "Magnetic field lines, drawn from a coarse, MPI-replicated copy of the field so the geometry is global and seamless across domains (the coarsening is what makes this cheap -- no parallel particle advection / flux scan).\n- 3D (Cartesian): traced as solid tubes, colored by |field|, composited inside the volume ray-march so the volume correctly occludes them.\n- 2D (Cartesian): iso-contours of the flux function psi (Bx = d psi/dy, By = -d psi/dx), i.e. the in-plane field lines, colored by |B|.\n- 2D (spherical / Kerr-Schild): traced meridional streamlines of the poloidal (Br, Btheta) field (nt2py style).\nBuilt once per frame and shared by every scene that opts in (per-scene `fieldlines = true`) and by any standalone `field = \"fieldlines\"` scene.", - "type": "object", - "additionalProperties": false, - "properties": { - "enable": { - "description": "Build the field-line geometry this run", - "type": "boolean", - "default": false, - "x-entity": { - "type": "bool", - "notes": [ - "implied true if any scene sets `fieldlines = true` or uses `field = \"fieldlines\"`" - ] - } - }, - "field": { - "description": "Vector field to trace", - "anyOf": [ - { - "enum": [ - "B", - "E", - "J" - ] - }, - { - "type": "string", - "pattern": "(?i)^(B|E|J)$" - } - ], - "default": "B", - "x-entity": { - "type": "string" - } - }, - "bin": { - "description": "Field coarsening factor (simulation cells per coarse cell, per axis)", - "type": "integer", - "minimum": 1, - "maximum": 16, - "default": 4, - "x-entity": { - "type": "int [1..16]", - "notes": [ - "larger = smoother \"morphology\" lines + cheaper replication (the coarse field is ~ N_cells / bin^D floats/rank; D = sim dimension)" - ] - } - }, - "seed_px": { - "description": "(3D tubes) Seed-lattice spacing in screen pixels (sets line density)", - "type": "number", - "exclusiveMinimum": 0.0, - "default": 8, - "x-entity": { - "type": "float [> 0]", - "notes": [ - "capped by `seed_max`; if seed_px asks for more seeds than that, the spacing grows to fit and seed_px no longer governs" - ] - } - }, - "seed_max": { - "description": "(3D tubes) Hard cap on the seed count (lattice is n^3, 2 lines per seed)", - "type": "integer", - "minimum": 1, - "default": 4096, - "x-entity": { - "type": "int [> 0]", - "notes": [ - "lower this for fewer / more widely spaced lines" - ] - } - }, - "levels": { - "description": "(2D contours) Number of evenly-spaced flux-function contour levels", - "type": "integer", - "minimum": 1, - "default": 16, - "x-entity": { - "type": "int [> 0]", - "notes": [ - "evenly-spaced psi levels => line density tracks |B| automatically" - ] - } - }, - "tube_px": { - "description": "Tube radius (3D) / contour line width (2D), in screen pixels", - "type": "number", - "exclusiveMinimum": 0.0, - "default": 2, - "x-entity": { - "type": "float [> 0]" - } - }, - "colormap": { - "description": "Colormap for the field lines (mapped by |B| along each line)", - "$ref": "#/$defs/colormap", - "default": "inferno" + "colorbar_ticks": { + "description": "Explicit value(s) to label on the colorbar", + "type": "array", + "items": { + "type": "number" }, - "color": { - "description": "Monochrome override: draw the lines in a single [r,g,b] color (each 0..1) instead of the |B| colormap -- reads well as an overlay on another volume", - "anyOf": [ - { - "type": "array", - "maxItems": 0 - }, - { - "type": "array", - "minItems": 3, - "maxItems": 3, - "items": { - "type": "number", - "minimum": 0.0, - "maximum": 1.0 - } - } + "x-entity": { + "type": "array", + "default": "5 evenly-spaced ticks between min and max", + "notes": [ + "Values outside [min, max] are skipped" ], - "default": [], - "x-entity": { - "type": "array [size 3]", - "default": "[] (empty => color by |B|)", - "examples": [ - "[1.0, 1.0, 1.0] # white field lines" - ] - } - }, - "log": { - "description": "Map the tube color range logarithmically", - "type": "boolean", - "default": false, - "x-entity": { - "type": "bool", - "notes": [ - "requires min > 0" - ] - } - }, - "min": { - "description": "Tube color range: lower bound on |field|", - "type": "number", - "default": 0.0, - "x-entity": { - "type": "float", - "notes": [ - "when min >= max, the range is auto-set from |field| along the lines" - ] - } - }, - "max": { - "description": "Tube color range: upper bound on |field|", - "type": "number", - "default": 0.0, - "x-entity": { - "type": "float", - "notes": [ - "when min >= max, the range is auto-set from |field| along the lines" - ] - } - }, - "step_frac": { - "description": "(3D tubes) RK4 integration step as a fraction of one coarse cell", - "type": "number", - "exclusiveMinimum": 0.0, - "default": 0.5, - "x-entity": { - "type": "float [> 0]" - } - }, - "max_steps": { - "description": "(3D tubes) Per-direction integration-step cap", - "type": "integer", - "minimum": 1, - "default": 4000, - "x-entity": { - "type": "int [> 0]" - } - }, - "max_length": { - "description": "(3D tubes) Maximum line length, in global box diagonals (per direction)", - "type": "number", - "exclusiveMinimum": 0.0, - "default": 3.0, - "x-entity": { - "type": "float [> 0]" - } + "examples": [ + "[0.0, 0.5, 1.0]" + ] + } + }, + "fieldlines": { + "description": "Overlay the magnetic field-line tubes inside this scene's volume", + "type": "boolean", + "default": false, + "x-entity": { + "type": "bool", + "notes": [ + "requires the [render.fieldlines] enabled (3D only). A scene with field = \"fieldlines\" instead renders them alone." + ] } } } @@ -2734,144 +2877,6 @@ } } }, - "checkpoint": { - "description": "Checkpointing parameters", - "type": "object", - "additionalProperties": false, - "x-entity": { - "inferred": [ - { - "name": "is_resuming", - "brief": "Whether the simulation is resuming from a checkpoint", - "type": "bool", - "from": "command-line flag" - }, - { - "name": "start_step", - "brief": "Timestep of the checkpoint used to resume", - "type": "uint", - "from": "automatically determined during restart" - }, - { - "name": "start_time", - "brief": "Time of the checkpoint used to resume", - "type": "float", - "from": "automatically determined during restart" - } - ] - }, - "properties": { - "interval": { - "description": "Number of timesteps between checkpoints", - "type": "integer", - "minimum": 1, - "default": 1000, - "x-entity": { - "type": "uint [> 0]" - } - }, - "interval_time": { - "description": "Physical (code) time interval between checkpoints", - "type": "number", - "default": -1.0, - "x-entity": { - "type": "float [> 0]", - "notes": [ - "When `< 0`, the output is controlled by `interval`" - ] - } - }, - "keep": { - "description": "Number of checkpoints to keep", - "type": "integer", - "minimum": -1, - "default": 2, - "x-entity": { - "type": "int", - "notes": [ - "0 = disable checkpointing", - "-1 = keep all checkpoints" - ] - } - }, - "walltime": { - "description": "Write a checkpoint once after a fixed walltime", - "type": "string", - "pattern": "^$|^[0-9]{2,}:[0-9]{2}:[0-9]{2}$", - "default": "00:00:00", - "x-entity": { - "type": "string", - "notes": [ - "The format is \"HH:MM:SS\"", - "Empty string or \"00:00:00\" disables this functionality", - "Writing checkpoint at walltime does not stop the simulation" - ] - } - }, - "write_path": { - "description": "Parent directory to write checkpoints to", - "type": "string", - "x-entity": { - "type": "string", - "default": "`.ckpt`", - "notes": [ - "The directory is created if it does not exist" - ] - } - }, - "read_path": { - "description": "Parent directory to use when resuming from a checkpoint", - "type": "string", - "x-entity": { - "type": "string", - "default": "inherit `write_path`" - } - } - } - }, - "adios2": { - "description": "ADIOS2 BP5 tuning, applied to both [output] and [checkpoint] writers", - "type": "object", - "additionalProperties": false, - "properties": { - "aggregators_per_node": { - "description": "Number of ADIOS2 aggregators per node", - "type": "integer", - "minimum": 0, - "default": 0, - "x-entity": { - "type": "uint", - "notes": [ - "Set to either MPI ranks/node or NICs/node for best performance\nIf set to 0, will use ADIOS2 default (one aggregator per node)" - ] - } - }, - "max_shm_size": { - "description": "Maximum shared-memory segment size per node, in bytes (BP5 MaxShmSize)", - "type": "integer", - "minimum": 0, - "default": 4294967296, - "x-entity": { - "type": "uint", - "notes": [ - "Lower this on memory-constrained nodes; matches ADIOS2's default" - ] - } - }, - "buffer_chunk_size": { - "description": "Internal serialization buffer chunk size, in bytes (BP5 BufferChunkSize)", - "type": "integer", - "minimum": 0, - "default": 16777216, - "x-entity": { - "type": "uint", - "notes": [ - "Scales with per-rank output volume; matches ADIOS2's default" - ] - } - } - } - }, "diagnostics": { "description": "Diagnostic logging parameters", "type": "object", diff --git a/input.default.toml b/input.default.toml index 3e9724dbc..3377cc6f7 100644 --- a/input.default.toml +++ b/input.default.toml @@ -32,10 +32,10 @@ # @example: [2, 2, 2] (total of 8 domains) decomposition = [-1, -1, -1] - # Diffusion-style dynamic load balancing (Cartesian metrics only). Domain - # boundaries between MPI neighbors are nudged to equalize the - # active-particle count per rank. All inter-rank traffic uses only the - # existing nearest-neighbor field/particle communication paths. + # Diffusion-style dynamic load balancing. Domain boundaries between MPI + # neighbors are nudged to equalize the active-particle count per rank. All + # inter-rank traffic uses only the existing nearest-neighbor field/particle + # communication paths. [simulation.domain.load_balance] # Enable dynamic load balancing # @type: bool @@ -606,11 +606,6 @@ # otherwise by `interval_time` # @note: Value is overriden by output intervals for specific outputs interval_time = -1.0 - # Whether to output each timestep into separate files - # @type: bool - # @default: true - # @deprecated: starting v1.3.0 - separate_files = true # Field output parameters [output.fields] @@ -781,452 +776,6 @@ # @default: [] custom = [] - # In-situ renderer. Renders scalar fields on the GPU and writes PNG images - # directly to `/renders/` each cadence -- no field data is written to - # storage, and the result is seamless across MPI domain boundaries. - # @note: two modes, selected automatically by the simulation dimension: - # - 3D Cartesian (Minkowski): volume ray-march (uses `samples`, - # `step_size`, `early_term_alpha`, and the [camera] table) - # - 2D (Minkowski, Spherical/QSpherical, and all GR Kerr-Schild): - # flat slice rasterizer. Cartesian shows the (x, y) plane; spherical - # shows the meridional (r, theta) half-plane mapped to Cartesian (X = - # r sin th, Z = r cos th), optionally mirrored (see `mirror`). The - # `samples`/`step_size`/`early_term_alpha`/[camera] keys are ignored - # in 2D (one opaque sample per pixel). - # @note: 1D (and 3D non-Cartesian, which does not exist) is a no-op - # @note: One PNG stream per scene (e.g. a density/|B|/|J| triptych) - [output.render] - # Toggle for the volume renderer - # @type: bool - # @default: false - enable = false - # Number of timesteps between renders - # @type: uint - # @default: 0 - # @note: When `!= 0`, overrides `output.interval` - # @note: When `== 0`, `interval_time` (or `output.interval`) is used - interval = 0 - # Physical (code) time interval between renders - # @type: float - # @default: -1.0 - # @note: When `< 0`, the output is controlled by `interval` - interval_time = -1.0 - # Image width in pixels (the rendered region; the PNG is wider if a colorbar - # margin is added, see `colorbar_outside`) - # @type: int [> 0] - # @default: 1024 - width = 1024 - # Image height in pixels - # @type: int [> 0] - # @default: 1024 - height = 1024 - # Convenience: force a square frame (sets width == height == resolution), - # the natural shape for a dome master. Overrides `width`/`height` when > 0. - # @type: int [> 0] - # @default: 0 (use width/height) - resolution = 0 - # Axis-aligned render region [lo, hi] along x1, in physical/world coords. - # Left unset it spans the full domain. Clamped to the box. - # @type: array [size 2] - # @default: [] (full extent) - # @note: 3D -> the volume is depth-clipped to this box, the wireframe/axes - # frame it, and the default camera zooms to it; 2D -> the slice - # window is framed to it. For a spherical 2D slice, x1_lim crops - # the radius r and x2_lim crops the polar angle theta. - # @example: x1_lim = [-64.0, 64.0] - x1_lim = [] - # Axis-aligned render region [lo, hi] along x2, in physical/world coords. - # Left unset it spans the full domain. Clamped to the box. - # @type: array [size 2] - # @default: [] (full extent) - # @note: For a spherical 2D slice, x2_lim crops the polar angle theta - x2_lim = [] - # Axis-aligned render region [lo, hi] along x3, in physical/world coords. - # Left unset it spans the full domain. Clamped to the box. - # @type: array [size 2] - # @default: [] (full extent) - x3_lim = [] - # Moving view: translate the render region (and, in 3D, the camera) at this - # velocity in world-units-per-sim-time, to keep a propagating feature (e.g. - # a shock) in frame. Pair with x{1,2,3}_lim (the moving window). 2D uses the - # first two components; the motion starts at `camera_start_time`. - # @type: array [size 2 or 3] - # @default: [] (static view) - # @example: camera_velocity = [0.9, 0.0] # pan along +x1 at 0.9 c - camera_velocity = [] - # Sim time at which the view starts moving (static before it, e.g. to let an - # initial ramp-up finish) - # @type: float - # @default: 0.0 - camera_start_time = 0.0 - # Number of ray-march steps across the global box diagonal - # @type: int [> 0] - # @default: 400 - # @note: The world-space step is `box_diagonal / samples` unless - # `step_size` is set. Higher = better quality, slower. - samples = 400 - # Fixed world-space step between ray samples - # @type: float [>= 0.0] - # @default: 0.0 - # @note: 0 derives the step from `samples`. The step is identical on all - # ranks, which is what makes the multi-domain composite seamless. - step_size = 0.0 - # Stop marching a ray once its accumulated opacity reaches this value - # @type: float [0.0 -> 1.0] - # @default: 0.99 - # @note: Pure speed optimization; set to 1.0 to disable early termination - early_term_alpha = 0.99 - # Number of entries in the color/opacity lookup table - # @type: int [> 1] - # @default: 256 - n_lut = 256 - # Opaque background RGB (each channel 0..1) shown through - # transparent/low-opacity pixels; also fills the colorbar margin - # @type: array [size 3] - # @default: [0.0, 0.0, 0.0] - background = [0.0, 0.0, 0.0] - # Draw a colorbar (gradient + value ticks + label) on each PNG - # @type: bool - # @default: true - colorbar = true - # Draw the colorbar in an added right margin (the PNG becomes wider by a - # fixed strip) instead of overlaying it on the rendered volume - # @type: bool - # @default: true - colorbar_outside = true - # 2D spherical slice only: mirror the meridional half-plane across the - # symmetry axis to render a full disk from one axisymmetric half. No effect - # on Cartesian or 3D rendering. - # @type: bool - # @default: true - mirror = true - # Draw the current simulation time as a label ("T = ", fixed to 2 - # decimals) in the upper-right corner of the render region, in a contrasting - # color, vertically centered between the frame top and the colorbar. - # @type: bool - # @default: false - time_label = false - # Draw a spine (frame) + axis ticks + labels around the rendered region. The - # PNG gains left/bottom margins (background-filled) for the tick labels and - # axis names, so they never overlap the data. - # @type: bool - # @default: false - # @note: 2D Cartesian = a rectangular frame with linear spatial ticks; - # 2D spherical = polar axes (an "R" radial axis on the symmetry - # axis with R=0 centered, and a "Theta" axis along the curved - # outline / spine); - # 3D = the global box projected to a wireframe with ticks on the - # three silhouette edges (x bottom, y & z on the left) - axes = false - # Axis names. 3D uses all three; the 2D slice uses the first two. When - # unset, the 2D slice defaults to "x","y" (Cartesian) or "X","Z" - # (spherical). - # @type: array [size <= 3] - # @default: ["x", "y", "z"] - axis_labels = ["x", "y", "z"] - # Target number of ticks per axis (actual count is rounded to nice values) - # @type: int [>= 2] - # @default: 5 - axis_ticks = 5 - # 3D only: target width (pixels) of the box wireframe "spine". The spine is - # drawn inside the ray-march (opaque, depth-occluded by the volume); its - # width is floored by the ray step, so for a crisper thin line raise - # `samples` as well. - # @type: float [> 0.0] - # @default: 2.0 - spine_width = 2.0 - - # Camera (3D volume mode only; ignored by the 2D slice rasterizer). Defaults - # frame the whole global box from outside, looking down the (1,1,1) diagonal - # -- the production setup for which the structured composite is provably - # seamless. - [output.render.camera] - # Projection mode. Overrides `orthographic` below when set. - # @type: string - # @default: (unset -> use `orthographic`) - # @enum: "orthographic", "perspective", "dome" - # @note: "dome" is a fulldome azimuthal-equidistant fisheye rendered - # from an INTERIOR eye (the domain center by default), i.e. a 3D - # planetarium dome master. It uses a depth-resolved (A-buffer) - # composite that is seamless across a full 3D domain - # decomposition (unlike ortho/perspective, which need the eye - # outside the box). Set a square frame (`resolution`, or width == - # height). `forward` is the dome ZENITH (screen-up defaults to +y - # for a +z zenith). - mode = "orthographic" - # Orthographic (true) or perspective (false) projection - # @type: bool - # @default: true - # @note: Orthographic is recommended; the seamless composite is always - # valid for it. Perspective is only seamless with the eye outside - # the box. - orthographic = true - # Camera (eye) position in world (physical) coordinates - # @type: array [size 3] - # @default: box center pushed back ~1.7 box-diagonals along (1, 1, 1); - # for `mode = "dome"`, the domain center (interior eye) - position = [0.0, 0.0, 0.0] - # Point the camera looks at, in world coordinates (the dome ZENITH target) - # @type: array [size 3] - # @default: box center; for `mode = "dome"`, the zenith defaults to +z - look_at = [0.0, 0.0, 0.0] - # Camera up vector (dome: the disk's screen-up) - # @type: array [size 3] - # @default: [0.0, 0.0, 1.0]; for `mode = "dome"`, [0.0, 1.0, 0.0] - up = [0.0, 0.0, 1.0] - # Vertical field of view in degrees (perspective only) - # @type: float [> 0.0] - # @default: 35.0 - fov = 35.0 - # Full dome field of view in degrees (dome mode only): the image rim is at - # dome_fov/2 from the zenith (180 = a full hemisphere down to the - # horizon). - # @type: float [> 0.0, <= 360.0] - # @default: 180.0 - dome_fov = 180.0 - # Dome far-clip radius in world units (dome mode only): each ray stops - # this far from the eye, so the sampled region is a half-ball (hemisphere) - # of this radius rather than the whole box -> uniform path length and no - # box corner/edge projection artifacts. `samples` then counts steps across - # this radius. - # @type: float [>= 0.0] - # @default: the largest sphere centered in the box (half the shortest - # side), so it touches the face centers and never a corner - # @note: 0 disables the clip (rays march to the box boundary) - dome_radius = 0.0 - # Vertical extent of the view in world units (orthographic only) - # @type: float [> 0.0] - # @default: the global box diagonal (the whole box fits from any angle) - ortho_height = 1.0 - - # Fulldome fisheye ("planetarium dome master"). 2D only; a circular image is - # centered in the frame's inscribed circle with the corners left as the - # background (the dome master's black border). Set `width == height` (e.g. - # 4096) for a square master. When enabled, the axes and the outside colorbar - # strip are suppressed so the PNG stays exactly width x height. Seamless - # across MPI domains (the pixel->world map is a shared, deterministic - # function and the tiles stay disjoint). Ignored (with a warning) for 3D. - # @note: CARTESIAN -- the flat plane is warped radially into the disk; use - # `fov`/`radius`/`center`/`projection` below. - # @note: SPHERICAL / GR Kerr-Schild -- the meridional slice is ALREADY a - # disk, so dome mode only mirrors it to a full disk (see `mirror` - # above; keep it true) and fits it to the inscribed circle. The - # `fov`/`radius`/`center`/`projection` keys are ignored (the native - # (X, Z) meridional map is used, with image radius proportional to - # the physical radius r, r=0 at the disk center). - [output.render.dome] - # Build the fisheye dome master instead of the plain slice - # @type: bool - # @default: false - enable = false - # (Cartesian only) Full dome field of view in degrees (image radius maps - # linearly to the dome zenith angle: the rim is at fov/2) - # @type: float [> 0.0, <= 180.0] - # @default: 180.0 # a full hemisphere - fov = 180.0 - # (Cartesian only) World radius of the circular cutout mapped onto the - # dome - # @type: float [> 0.0] - # @default: half the shorter domain side (the largest centered disk that - # fits inside the box) - radius = 1.0 - # (Cartesian only) World-space center of the cutout - # @type: array [size 2] - # @default: the domain center - center = [0.0, 0.0] - # (Cartesian only) How the dome zenith angle maps to a world radius on the - # flat slice - # @type: string - # @default: "equidistant" - # @enum: "equidistant" (r proportional to angle; the fulldome image - # standard -- a straight radial scaling of the cutout), - # "gnomonic" (r ~ tan(angle); the slice as a flat "ceiling" - # tangent to the dome -- straight sim lines stay straight), - # "stereographic" (r ~ tan(angle/2); conformal, preserves - # shapes), "orthographic" (r ~ sin(angle); the slice as seen - # face-on) - projection = "equidistant" - - # One scene per scalar field -> one PNG stream. Repeat the table for each. - [[output.render.scenes]] - # Scalar field to render (a volume render needs a scalar, so vectors are - # given as a magnitude or a single component) - # @required - # @type: string - # @enum: (fields): "{E,B,J}mag"; "{E,B,J}{1,2,3}" or "{E,B,J}{x,y,z}", - # (moments): "N", "Nppc", "Rho", "Charge"; "T{i}{j}"; "V{i}"; - # "Vmag" - # @note: "{E,B,J}mag" = vector magnitude |.|; "B1"/"Bx", "J3"/"Jz", ... - # = a single (signed) physical component - # @note: a bare vector ("E"/"B"/"J") is not renderable -- choose a - # component or the magnitude - # @note: "N"/"Nppc" = number / per-cell count, "Rho" = mass density, - # "Charge" = charge density - # @note: "T{i}{j}" = one stress-energy component, i,j in {t,x,y,z} or - # {0,1,2,3} (e.g. "Txx", "Ttt", "T0x"); "V{i}" = one - # bulk-velocity component, i in {x,y,z} or {1,2,3} (e.g. "Vx", - # "V1"); "Vmag" = bulk-velocity magnitude sqrt(V1^2+V2^2+V3^2) - # @note: moments follow the engine: SRPIC = tetrad-basis bulk 3-velocity - # and stress-energy; GRPIC = Eckart-frame 4-velocity (so - # "Vt"/"V0" = u^0 = Gamma/alpha is also valid) and contravariant - # T - # @note: per-species selection with a "_" suffix on moments, e.g. - # "N_1", "Rho_2", "Txy_1_2", "V1_3"; default = all massive - # species - # @note: components are signed; pair a symmetric `min`/`max` with a - # diverging colormap ("cool2warm") to center zero - # @note: "fieldlines" renders the magnetic field-line tubes on their own - # (no scalar volume sampled); see [output.render.fieldlines] - # below - field = "" - # PNG filename prefix; files are `.png` - # @type: string - # @default: "_" - prefix = "_" - # Colorbar title - # @type: string - # @default: `field` - label = "" - # Lower bound of the value range mapped onto the colormap/opacity - # @type: float - # @default: 0.0 - min = 0.0 - # Upper bound of the value range - # @type: float - # @default: 1.0 - max = 1.0 - # Map the value range logarithmically - # @type: bool - # @default: false - # @note: Requires min > 0 and max > 0 - log = false - # Colormap name - # @type: string - # @default: "viridis" - # @enum: "viridis", "inferno", "plasma", "cool2warm", "gray", "RdBu_r", - # and the CMasher maps (BSD-3, https://cmasher.readthedocs.io): - # "dusk", "cosmic", "freeze", "apple", "gothic", "sunburst", - # "voltage", "ocean", "fusion", "prinsenvlag" (an optional "cmr." - # prefix is ok) - colormap = "viridis" - # Opacity transfer function: [position, opacity] control points, both in - # [0, 1], piecewise-linear in the normalized value - # @type: array> - # @default: linear ramp (opacity = normalized value) - # @note: Keep the low end near 0 so empty regions stay transparent - # @example: [[0.0, 0.0], [0.3, 0.1], [1.0, 0.7]] - alpha = [] - # Explicit value(s) to label on the colorbar - # @type: array - # @default: 5 evenly-spaced ticks between min and max - # @note: Values outside [min, max] are skipped - # @example: [0.0, 0.5, 1.0] - colorbar_ticks = [] - # Overlay the magnetic field-line tubes inside this scene's volume - # @type: bool - # @default: false - # @note: requires the [output.render.fieldlines] table below (3D only). - # A scene with field = "fieldlines" instead renders them alone. - fieldlines = false - - # Magnetic field lines, drawn from a coarse, MPI-replicated copy of the - # field so the geometry is global and seamless across domains (the - # coarsening is what makes this cheap -- no parallel particle advection / - # flux scan). - # - 3D (Cartesian): traced as solid tubes, colored by |field|, composited - # inside the volume ray-march so the volume correctly occludes them. - # - 2D (Cartesian): iso-contours of the flux function psi (Bx = d psi/dy, By - # = -d psi/dx), i.e. the in-plane field lines, colored by |B|. - # - 2D (spherical / Kerr-Schild): traced meridional streamlines of the - # poloidal (Br, Btheta) field (nt2py style). - # Built once per frame and shared by every scene that opts in (per-scene - # `fieldlines = true`) and by any standalone `field = "fieldlines"` scene. - [output.render.fieldlines] - # Build the field-line geometry this run - # @type: bool - # @default: false - # @note: implied true if any scene sets `fieldlines = true` or uses - # `field = "fieldlines"` - enable = false - # Vector field to trace - # @type: string - # @default: "B" - # @enum: "B", "E", "J" - field = "B" - # Field coarsening factor (simulation cells per coarse cell, per axis) - # @type: int [1..16] - # @default: 4 - # @note: larger = smoother "morphology" lines + cheaper replication (the - # coarse field is ~ N_cells / bin^D floats/rank; D = sim - # dimension) - bin = 4 - # (3D tubes) Seed-lattice spacing in screen pixels (sets line density) - # @type: float [> 0] - # @default: 8 - # @note: capped by `seed_max`; if seed_px asks for more seeds than that, - # the spacing grows to fit and seed_px no longer governs - seed_px = 8 - # (3D tubes) Hard cap on the seed count (lattice is n^3, 2 lines per seed) - # @type: int [> 0] - # @default: 4096 - # @note: lower this for fewer / more widely spaced lines - seed_max = 4096 - # (2D contours) Number of evenly-spaced flux-function contour levels - # @type: int [> 0] - # @default: 16 - # @note: evenly-spaced psi levels => line density tracks |B| - # automatically - levels = 16 - # Tube radius (3D) / contour line width (2D), in screen pixels - # @type: float [> 0] - # @default: 2 - tube_px = 2 - # Colormap for the field lines (mapped by |B| along each line) - # @type: string - # @default: "inferno" - # @enum: "viridis", "inferno", "plasma", "cool2warm", "gray", "RdBu_r", - # and the CMasher maps (BSD-3, https://cmasher.readthedocs.io): - # "dusk", "cosmic", "freeze", "apple", "gothic", "sunburst", - # "voltage", "ocean", "fusion", "prinsenvlag" (an optional "cmr." - # prefix is ok) - colormap = "inferno" - # Monochrome override: draw the lines in a single [r,g,b] color (each - # 0..1) instead of the |B| colormap -- reads well as an overlay on another - # volume - # @type: array [size 3] - # @default: [] (empty => color by |B|) - # @example: [1.0, 1.0, 1.0] # white field lines - color = [] - # Map the tube color range logarithmically - # @type: bool - # @default: false - # @note: requires min > 0 - log = false - # Tube color range: lower bound on |field| - # @type: float - # @default: 0.0 - # @note: when min >= max, the range is auto-set from |field| along the - # lines - min = 0.0 - # Tube color range: upper bound on |field| - # @type: float - # @default: 0.0 - # @note: when min >= max, the range is auto-set from |field| along the - # lines - max = 0.0 - # (3D tubes) RK4 integration step as a fraction of one coarse cell - # @type: float [> 0] - # @default: 0.5 - step_frac = 0.5 - # (3D tubes) Per-direction integration-step cap - # @type: int [> 0] - # @default: 4000 - max_steps = 4000 - # (3D tubes) Maximum line length, in global box diagonals (per direction) - # @type: float [> 0] - # @default: 3.0 - max_length = 3.0 - # Checkpointing parameters [checkpoint] # Number of timesteps between checkpoints @@ -1294,6 +843,440 @@ # @note: Scales with per-rank output volume; matches ADIOS2's default buffer_chunk_size = 16777216 +# In-situ renderer. Renders scalar fields on the GPU and writes PNG images +# directly to `/renders/` each cadence -- no field data is written to +# storage, and the result is seamless across MPI domain boundaries. +# @note: two modes, selected automatically by the simulation dimension: +# - 3D Cartesian (Minkowski): volume ray-march (uses `samples`, +# `step_size`, `early_term_alpha`, and the [camera] table) +# - 2D (Minkowski, Spherical/QSpherical, and all GR Kerr-Schild): flat +# slice rasterizer. Cartesian shows the (x, y) plane; spherical shows +# the meridional (r, theta) half-plane mapped to Cartesian (X = r sin +# th, Z = r cos th), optionally mirrored (see `mirror`). The +# `samples`/`step_size`/`early_term_alpha`/[camera] keys are ignored in +# 2D (one opaque sample per pixel). +# @note: 1D (and 3D non-Cartesian, which does not exist) is a no-op +# @note: One PNG stream per scene (e.g. a density/|B|/|J| triptych) +[render] + # Toggle for the on-the-fly renderer + # @type: bool + # @default: false + enable = false + # Number of timesteps between renders + # @type: uint + # @default: 0 + # @note: When `!= 0`, overrides `output.interval` + # @note: When `== 0`, `interval_time` (or `output.interval`) is used + interval = 0 + # Physical (code) time interval between renders + # @type: float + # @default: -1.0 + # @note: When `< 0`, the output is controlled by `interval` + interval_time = -1.0 + # Image width in pixels (the rendered region; the PNG is wider if a colorbar + # margin is added, see `colorbar_outside`) + # @type: int [> 0] + # @default: 1024 + width = 1024 + # Image height in pixels + # @type: int [> 0] + # @default: 1024 + height = 1024 + # Convenience: force a square frame (sets width == height == resolution), the + # natural shape for a dome master. Overrides `width`/`height` when > 0. + # @type: int [> 0] + # @default: 0 (use width/height) + resolution = 0 + # Number of entries in the color/opacity lookup table + # @type: int [> 1] + # @default: 256 + n_lut = 256 + # Opaque background RGB (each channel 0..1) shown through + # transparent/low-opacity pixels; also fills the colorbar margin + # @type: array [size 3] + # @default: [0.0, 0.0, 0.0] + background = [0.0, 0.0, 0.0] + # Draw a colorbar (gradient + value ticks + label) on each PNG + # @type: bool + # @default: true + colorbar = true + # Draw the colorbar in an added right margin (the PNG becomes wider by a fixed + # strip) instead of overlaying it on the rendered volume + # @type: bool + # @default: true + colorbar_outside = true + # 2D spherical slice only: mirror the meridional half-plane across the + # symmetry axis to render a full disk from one axisymmetric half. No effect on + # Cartesian or 3D rendering. + # @type: bool + # @default: true + mirror = true + # Draw the current simulation time as a label ("T = ", fixed to 2 + # decimals) in the upper-right corner of the render region, in a contrasting + # color, vertically centered between the frame top and the colorbar. + # @type: bool + # @default: false + time_label = false + # Draw a spine (frame) + axis ticks + labels around the rendered region. The + # PNG gains left/bottom margins (background-filled) for the tick labels and + # axis names, so they never overlap the data. + # @type: bool + # @default: false + # @note: 2D Cartesian = a rectangular frame with linear spatial ticks; + # 2D spherical = polar axes (an "R" radial axis on the symmetry axis + # with R=0 centered, and a "Theta" axis along the curved outline / + # spine); + # 3D = the global box projected to a wireframe with ticks on the + # three silhouette edges (x bottom, y & z on the left) + axes = false + # Axis names. 3D uses all three; the 2D slice uses the first two. When unset, + # the 2D slice defaults to "x","y" (Cartesian) or "X","Z" (spherical). + # @type: array [size <= 3] + # @default: ["x", "y", "z"] + axis_labels = ["x", "y", "z"] + # Target number of ticks per axis (actual count is rounded to nice values) + # @type: int [>= 2] + # @default: 5 + axis_ticks = 5 + # 3D only: target width (pixels) of the box wireframe "spine". The spine is + # drawn inside the ray-march (opaque, depth-occluded by the volume); its width + # is floored by the ray step, so for a crisper thin line raise `samples` as + # well. + # @type: float [> 0.0] + # @default: 2.0 + spine_width = 2.0 + + # Limit the render region to axis-aligned box in physical/world coordinates. + # Left unset it spans the full domain. Clamped to the box. + # @note: 3D -> the volume is depth-clipped to this box, the wireframe/axes + # frame it, and the default camera zooms to it; 2D -> the slice + # window is framed to it + [render.extent] + # Axis-aligned render region [lo, hi] along x1, in physical/world coords. + # Left unset it spans the full domain. Clamped to the box. + # @type: array [size 2] + # @default: [] (full extent) + # @note: For a spherical 2D slice, x1 crops the radius r. + # @example: x1 = [-64.0, 64.0] + x1 = [] + # Render region [lo, hi] along x2. + # @type: array [size 2] + # @default: [] (full extent) + # @note: For a spherical 2D slice, x2 crops the polar angle theta. + x2 = [] + # Render region [lo, hi] along x3. + # @type: array [size 2] + # @default: [] (full extent) + x3 = [] + + # Volume rendering parameters (3D Cartesian only) + [render.volume] + # Number of ray-march steps across the global box diagonal + # @type: int [> 0] + # @default: 400 + # @note: The world-space step is `box_diagonal / samples` unless + # `step_size` is set. Higher = better quality, slower. + samples = 400 + # Fixed world-space step between ray samples + # @type: float [>= 0.0] + # @default: 0.0 + # @note: 0 derives the step from `samples`. The step is identical on all + # ranks, which is what makes the multi-domain composite seamless. + step_size = 0.0 + # Stop marching a ray once its accumulated opacity reaches this value + # @type: float [0.0 -> 1.0] + # @default: 0.99 + # @note: Pure speed optimization; set to 1.0 to disable early termination + early_term_alpha = 0.99 + + # Translate the render region (and, in 3D, the camera), to keep a propagating + # feature (e.g. a shock) in frame. Pair with x{1,2,3} extent to crop the + # moving window. + [render.moving_view] + # Velocity of the moving camera in world units. + # @type: array [size 2 or 3] + # @default: [] (static view) + # @example: velocity = [0.9, 0.0] # pan along +x1 at 0.9 c + velocity = [] + # Sim time at which the view starts moving (static before it, e.g. to let an + # initial ramp-up finish) + # @type: float + # @default: 0.0 + start_time = 0.0 + + # Camera (3D volume mode only; ignored by the 2D slice rasterizer). Defaults + # frame the whole global box from outside, looking down the (1,1,1) diagonal + # -- the production setup for which the structured composite is provably + # seamless. + [render.camera] + # Projection mode. Overrides `orthographic` below when set. + # @type: string + # @default: (unset -> use `orthographic`) + # @enum: "orthographic", "perspective", "dome" + # @note: "dome" is a fulldome azimuthal-equidistant fisheye rendered from + # an INTERIOR eye (the domain center by default), i.e. a 3D + # planetarium dome master. It uses a depth-resolved (A-buffer) + # composite that is seamless across a full 3D domain decomposition + # (unlike ortho/perspective, which need the eye outside the box). + # Set a square frame (`resolution`, or width == height). `forward` + # is the dome ZENITH (screen-up defaults to +y for a +z zenith). + mode = "orthographic" + # Camera (eye) position in world (physical) coordinates + # @type: array [size 3] + # @default: box center pushed back ~1.7 box-diagonals along (1, 1, 1); + # for `mode = "dome"`, the domain center (interior eye) + position = [0.0, 0.0, 0.0] + # Point the camera looks at, in world coordinates (the dome ZENITH target) + # @type: array [size 3] + # @default: box center; for `mode = "dome"`, the zenith defaults to +z + look_at = [0.0, 0.0, 0.0] + # Camera up vector (dome: the disk's screen-up) + # @type: array [size 3] + # @default: [0.0, 0.0, 1.0]; for `mode = "dome"`, [0.0, 1.0, 0.0] + up = [0.0, 0.0, 1.0] + # Vertical field of view in degrees (perspective only) + # @type: float [> 0.0] + # @default: 35.0 + fov = 35.0 + # Full dome field of view in degrees (dome mode only): the image rim is at + # dome_fov/2 from the zenith (180 = a full hemisphere down to the horizon). + # @type: float [> 0.0, <= 360.0] + # @default: 180.0 + dome_fov = 180.0 + # Dome far-clip radius in world units (dome mode only): each ray stops this + # far from the eye, so the sampled region is a half-ball (hemisphere) of + # this radius rather than the whole box -> uniform path length and no box + # corner/edge projection artifacts. `samples` then counts steps across this + # radius. + # @type: float [>= 0.0] + # @default: the largest sphere centered in the box (half the shortest + # side), so it touches the face centers and never a corner + # @note: 0 disables the clip (rays march to the box boundary) + dome_radius = 0.0 + # Vertical extent of the view in world units (orthographic only) + # @type: float [> 0.0] + # @default: the global box diagonal (the whole box fits from any angle) + ortho_height = 1.0 + + # Fulldome fisheye ("planetarium dome master"). 2D only; a circular image is + # centered in the frame's inscribed circle with the corners left as the + # background (the dome master's black border). Set `width == height` (e.g. + # 4096) for a square master. When enabled, the axes and the outside colorbar + # strip are suppressed so the PNG stays exactly width x height. Seamless + # across MPI domains (the pixel->world map is a shared, deterministic function + # and the tiles stay disjoint). Ignored (with a warning) for 3D. + # @note: CARTESIAN -- the flat plane is warped radially into the disk; use + # `fov`/`radius`/`center`/`projection` below. + # @note: SPHERICAL / GR Kerr-Schild -- the meridional slice is ALREADY a + # disk, so dome mode only mirrors it to a full disk (see `mirror` + # above; keep it true) and fits it to the inscribed circle. The + # `fov`/`radius`/`center`/`projection` keys are ignored (the native + # (X, Z) meridional map is used, with image radius proportional to + # the physical radius r, r=0 at the disk center). + [render.dome] + # Build the fisheye dome master instead of the plain slice + # @type: bool + # @default: false + enable = false + # (Cartesian only) Full dome field of view in degrees (image radius maps + # linearly to the dome zenith angle: the rim is at fov/2) + # @type: float [> 0.0, <= 180.0] + # @default: 180.0 # a full hemisphere + fov = 180.0 + # (Cartesian only) World radius of the circular cutout mapped onto the dome + # @type: float [> 0.0] + # @default: half the shorter domain side (the largest centered disk that + # fits inside the box) + radius = 1.0 + # (Cartesian only) World-space center of the cutout + # @type: array [size 2] + # @default: the domain center + center = [0.0, 0.0] + # (Cartesian only) How the dome zenith angle maps to a world radius on the + # flat slice + # @type: string + # @default: "equidistant" + # @enum: "equidistant" (r proportional to angle; the fulldome image + # standard -- a straight radial scaling of the cutout), "gnomonic" + # (r ~ tan(angle); the slice as a flat "ceiling" tangent to the + # dome -- straight sim lines stay straight), "stereographic" (r ~ + # tan(angle/2); conformal, preserves shapes), "orthographic" (r ~ + # sin(angle); the slice as seen face-on) + projection = "equidistant" + + # Magnetic field lines, drawn from a coarse, MPI-replicated copy of the field + # so the geometry is global and seamless across domains (the coarsening is + # what makes this cheap -- no parallel particle advection / flux scan). + # - 3D (Cartesian): traced as solid tubes, colored by |field|, composited + # inside the volume ray-march so the volume correctly occludes them. + # - 2D (Cartesian): iso-contours of the flux function psi (Bx = d psi/dy, By = + # -d psi/dx), i.e. the in-plane field lines, colored by |B|. + # - 2D (spherical / Kerr-Schild): traced meridional streamlines of the + # poloidal (Br, Btheta) field (nt2py style). + # Built once per frame and shared by every scene that opts in (per-scene + # `fieldlines = true`) and by any standalone `field = "fieldlines"` scene. + [render.fieldlines] + # Build the field-line geometry this run + # @type: bool + # @default: false + # @note: implied true if any scene sets `fieldlines = true` or uses `field + # = "fieldlines"` + enable = false + # Vector field to trace + # @type: string + # @default: "B" + # @enum: "B", "E", "J" + field = "B" + # Field coarsening factor (simulation cells per coarse cell, per axis) + # @type: int [1..16] + # @default: 4 + # @note: larger = smoother "morphology" lines + cheaper replication (the + # coarse field is ~ N_cells / bin^D floats/rank; D = sim dimension) + bin = 4 + # (3D tubes) Seed-lattice spacing in screen pixels (sets line density) + # @type: float [> 0] + # @default: 8 + # @note: capped by `seed_max`; if seed_px asks for more seeds than that, + # the spacing grows to fit and seed_px no longer governs + seed_px = 8 + # (3D tubes) Hard cap on the seed count (lattice is n^3, 2 lines per seed) + # @type: int [> 0] + # @default: 4096 + # @note: lower this for fewer / more widely spaced lines + seed_max = 4096 + # (2D contours) Number of evenly-spaced flux-function contour levels + # @type: int [> 0] + # @default: 16 + # @note: evenly-spaced psi levels => line density tracks |B| automatically + levels = 16 + # Tube radius (3D) / contour line width (2D), in screen pixels + # @type: float [> 0] + # @default: 2 + tube_px = 2 + # Colormap for the field lines (mapped by |B| along each line) + # @type: string + # @default: "inferno" + # @enum: "viridis", "inferno", "plasma", "cool2warm", "gray", "RdBu_r", + # and the CMasher maps (BSD-3, https://cmasher.readthedocs.io): + # "dusk", "cosmic", "freeze", "apple", "gothic", "sunburst", + # "voltage", "ocean", "fusion", "prinsenvlag" (an optional "cmr." + # prefix is ok) + colormap = "inferno" + # Monochrome override: draw the lines in a single [r,g,b] color (each 0..1) + # instead of the |B| colormap -- reads well as an overlay on another volume + # @type: array [size 3] + # @default: [] (empty => color by |B|) + # @example: [1.0, 1.0, 1.0] # white field lines + color = [] + # Map the tube color range logarithmically + # @type: bool + # @default: false + # @note: requires min > 0 + log = false + # Tube color range: lower bound on |field| + # @type: float + # @default: 0.0 + # @note: when min >= max, the range is auto-set from |field| along the + # lines + min = 0.0 + # Tube color range: upper bound on |field| + # @type: float + # @default: 0.0 + # @note: when min >= max, the range is auto-set from |field| along the + # lines + max = 0.0 + # (3D tubes) RK4 integration step as a fraction of one coarse cell + # @type: float [> 0] + # @default: 0.5 + step_frac = 0.5 + # (3D tubes) Per-direction integration-step cap + # @type: int [> 0] + # @default: 4000 + max_steps = 4000 + # (3D tubes) Maximum line length, in global box diagonals (per direction) + # @type: float [> 0] + # @default: 3.0 + max_length = 3.0 + + # One scene per scalar field -> one PNG stream. Repeat the table for each. + [[render.scene]] + # Scalar field to render (a volume render needs a scalar, so vectors are + # given as a magnitude or a single component) + # @required + # @type: string + # @enum: (fields): "{E,B,J}mag"; "{E,B,J}{1,2,3}" or "{E,B,J}{x,y,z}", + # (moments): "N", "Nppc", "Rho", "Charge"; "T{i}{j}"; "V{i}"; + # "Vmag" + # @note: "{E,B,J}mag" = vector magnitude |.|; "B1"/"Bx", "J3"/"Jz", ... = + # a single (signed) physical component + # @note: a bare vector ("E"/"B"/"J") is not renderable -- choose a + # component or the magnitude + # @note: "N"/"Nppc" = number / per-cell count, "Rho" = mass density, + # "Charge" = charge density + # @note: "T{i}{j}" = one stress-energy component, i,j in {t,x,y,z} or + # {0,1,2,3} (e.g. "Txx", "Ttt", "T0x"); "V{i}" = one bulk-velocity + # component, i in {x,y,z} or {1,2,3} (e.g. "Vx", "V1"); "Vmag" = + # bulk-velocity magnitude sqrt(V1^2+V2^2+V3^2) + # @note: moments follow the engine: SRPIC = tetrad-basis bulk 3-velocity + # and stress-energy; GRPIC = Eckart-frame 4-velocity (so "Vt"/"V0" + # = u^0 = Gamma/alpha is also valid) and contravariant T + # @note: per-species selection with a "_" suffix on moments, e.g. + # "N_1", "Rho_2", "Txy_1_2", "V1_3"; default = all massive species + # @note: components are signed; pair a symmetric `min`/`max` with a + # diverging colormap ("cool2warm") to center zero + # @note: "fieldlines" renders the magnetic field-line tubes on their own + # (no scalar volume sampled); see [render.fieldlines] below + field = "" + # PNG filename prefix; files are `.png` + # @type: string + # @default: "_" + prefix = "_" + # Colorbar title + # @type: string + # @default: `field` + label = "" + # Lower bound of the value range mapped onto the colormap/opacity + # @type: float + # @default: 0.0 + min = 0.0 + # Upper bound of the value range + # @type: float + # @default: 1.0 + max = 1.0 + # Map the value range logarithmically + # @type: bool + # @default: false + # @note: Requires min > 0 and max > 0 + log = false + # Colormap name + # @type: string + # @default: "viridis" + # @enum: "viridis", "inferno", "plasma", "cool2warm", "gray", "RdBu_r", + # and the CMasher maps (BSD-3, https://cmasher.readthedocs.io): + # "dusk", "cosmic", "freeze", "apple", "gothic", "sunburst", + # "voltage", "ocean", "fusion", "prinsenvlag" (an optional "cmr." + # prefix is ok) + colormap = "viridis" + # Opacity transfer function: [position, opacity] control points, both in [0, + # 1], piecewise-linear in the normalized value + # @type: array> + # @default: linear ramp (opacity = normalized value) + # @note: Keep the low end near 0 so empty regions stay transparent + # @example: [[0.0, 0.0], [0.3, 0.1], [1.0, 0.7]] + alpha = [] + # Explicit value(s) to label on the colorbar + # @type: array + # @default: 5 evenly-spaced ticks between min and max + # @note: Values outside [min, max] are skipped + # @example: [0.0, 0.5, 1.0] + colorbar_ticks = [] + # Overlay the magnetic field-line tubes inside this scene's volume + # @type: bool + # @default: false + # @note: requires the [render.fieldlines] enabled (3D only). A scene with + # field = "fieldlines" instead renders them alone. + fieldlines = false + # Diagnostic logging parameters [diagnostics] # Number of timesteps between diagnostic logs diff --git a/src/framework/parameters/output.cpp b/src/framework/parameters/output.cpp index d2a6a9c44..03087f08a 100644 --- a/src/framework/parameters/output.cpp +++ b/src/framework/parameters/output.cpp @@ -27,10 +27,6 @@ namespace ntt { "output", "interval_time", -1.0); - raise::ErrorIf( - not toml::find_or(toml_data, "output", "separate_files", true), - "separate_files=false is deprecated", - HERE); categories.emplace(); for (const auto& category : { "fields", "particles", "spectra", "stats" }) { diff --git a/src/framework/parameters/render.cpp b/src/framework/parameters/render.cpp new file mode 100644 index 000000000..e69de29bb diff --git a/src/framework/parameters/render.h b/src/framework/parameters/render.h new file mode 100644 index 000000000..4162e2965 --- /dev/null +++ b/src/framework/parameters/render.h @@ -0,0 +1,37 @@ +/** + * @file framework/parameters/render.h + * @brief Auxiliary functions for reading in on-the-fly render parameters + * @implements + * - ntt::params::Render + * @cpp: + * - render.cpp + * @namespaces: + * - ntt::params:: + */ +#ifndef FRAMEWORK_PARAMETERS_RENDER_H +#define FRAMEWORK_PARAMETERS_RENDER_H + +#include "global.h" + +#include "framework/parameters/parameters.h" + +#include + +#include +#include + +namespace ntt { + namespace params { + + struct Render { + + void read(const std::map&, + const toml::value&, + const SimulationParams* const); + void setParams(const std::map&, SimulationParams*) const; + }; + + } // namespace params +} // namespace ntt + +#endif diff --git a/src/output/render/renderer.cpp b/src/output/render/renderer.cpp index 272bedd3b..433315199 100644 --- a/src/output/render/renderer.cpp +++ b/src/output/render/renderer.cpp @@ -304,18 +304,13 @@ namespace out { const real_t diag = std::sqrt( size[0] * size[0] + size[1] * size[1] + size[2] * size[2]); - const bool ortho = - toml::find_or(td, "output", "render", "camera", "orthographic", true); - // `mode` overrides the `orthographic` flag: "orthographic" | "perspective" | - // "dome". The dome is a fulldome azimuthal-equidistant fisheye from an - // INTERIOR eye (the box center by default) -- see Metadomain::Render (3D). - const auto cam_mode = toml::find_or(td, + const auto cam_mode = toml::find_or(td, "output", "render", "camera", "mode", std::string {}); - int projection = ortho ? CameraDevice::Ortho : CameraDevice::Perspective; + auto projection = CameraDevice::Ortho; if (cam_mode == "dome") { projection = CameraDevice::Dome; } else if (cam_mode == "perspective") { @@ -325,7 +320,7 @@ namespace out { } else if (not cam_mode.empty()) { raise::Warning("output.render.camera.mode '" + cam_mode + "' unknown (want orthographic/perspective/dome); using " - "the 'orthographic' flag", + "orthographic projection", HERE); } const bool is_dome = (projection == CameraDevice::Dome); From eae626fcdba94a3a85a3b7bea9f6d265e173cfc1 Mon Sep 17 00:00:00 2001 From: haykh Date: Sun, 27 Sep 2026 18:59:06 -0400 Subject: [PATCH 119/125] [output.render]->[render] --- CODEGUIDE.md | 2 +- scripts/render.py | 60 +-- src/framework/CMakeLists.txt | 2 + src/framework/domain/io/render.cpp | 16 +- src/framework/parameters/parameters.cpp | 6 + src/framework/parameters/render.cpp | 434 ++++++++++++++++++++ src/framework/parameters/render.h | 95 ++++- src/output/render/renderer.cpp | 503 +++++++----------------- src/output/render/renderer.h | 12 +- 9 files changed, 719 insertions(+), 411 deletions(-) diff --git a/CODEGUIDE.md b/CODEGUIDE.md index 7824195c0..d69cf03b2 100644 --- a/CODEGUIDE.md +++ b/CODEGUIDE.md @@ -122,7 +122,7 @@ Three things to keep in mind: * **String enums are matched case-insensitively by the code** (`fmt::toLower` is applied to `engine`, `metric`, the boundary lists, `pusher`, `log_level`, ...), so a bare `"enum"` would reject perfectly valid input. The convention is `anyOf: [{"enum": []}, {"type": "string", "pattern": "(?i)^(|...)$"}]` -- the enum branch drives completion and hover, the pattern branch keeps any casing legal. Note `(?i)` is a Rust/Python regex extension: tombi honours it, JS-based validators do not. * **Every table is closed.** Set `additionalProperties: false` so typos are caught; tombi's `strict = true` closes objects that omit it anyway. `[setup]` is the one deliberate exception (`additionalProperties: true`), since its keys belong to the problem generator. -* **If a key's documented default is `[]`, the empty array must validate**, which `minItems` would otherwise forbid -- use `anyOf: [{"maxItems": 0}, {}]` (see `output.render.x1_lim`). +* **If a key's documented default is `[]`, the empty array must validate**, which `minItems` would otherwise forbid -- use `anyOf: [{"maxItems": 0}, {}]` (see `render.extent.x1`). ## Code guidelines diff --git a/scripts/render.py b/scripts/render.py index 05bb3635b..ffe11606e 100755 --- a/scripts/render.py +++ b/scripts/render.py @@ -134,14 +134,16 @@ def __init__(self, td, region, width, height): size[d] = region[d][1] - region[d][0] diag = math.sqrt(size[0] ** 2 + size[1] ** 2 + size[2] ** 2) - cam_ortho = find_or(td, True, "output", "render", "camera", "orthographic") - pos = find_or(td, [], "output", "render", "camera", "position") - look = find_or(td, [], "output", "render", "camera", "look_at") - up = find_or(td, [], "output", "render", "camera", "up") - fov = float(find_or(td, 35.0, "output", "render", "camera", "fov")) + cam_ortho = ( + find_or(td, "orthographic", "render", "camera", "mode") != "perspective" + ) + pos = find_or(td, [], "render", "camera", "position") + look = find_or(td, [], "render", "camera", "look_at") + up = find_or(td, [], "render", "camera", "up") + fov = float(find_or(td, 35.0, "render", "camera", "fov")) # default ortho_height covers the box from any view -> == box diagonal ortho_height = float( - find_or(td, diag, "output", "render", "camera", "ortho_height") + find_or(td, diag, "render", "camera", "ortho_height") ) # default eye: box center pushed back along (1,1,1) by ~1.7 diagonals @@ -207,18 +209,18 @@ def project(self, p, W, H): # region resolution (renderer.cpp Renderer::init ~L128-158) # # --------------------------------------------------------------------------- # def resolve_region(td, ext): - """m_region starts as the global extent; x{d+1}_lim clamps axis d to the box + """m_region starts as the global extent; extent.x{d+1} clamps axis d to the box if it is a valid [lo,hi] with hi>lo that overlaps. Returns (region, has_region).""" region = [list(p) for p in ext] has_region = False - keys = ["x1_lim", "x2_lim", "x3_lim"] + keys = ["x1", "x2", "x3"] for d in range(min(len(ext), 3)): - lim = find_or(td, [], "output", "render", keys[d]) + lim = find_or(td, [], "render", "extent", keys[d]) if not lim: continue if len(lim) != 2 or lim[1] <= lim[0]: print( - f" warning: output.render.{keys[d]} must be [lo,hi] with hi>lo; ignoring" + f" warning: render.extent.{keys[d]} must be [lo,hi] with hi>lo; ignoring" ) continue lo = max(float(lim[0]), ext[d][0]) @@ -228,14 +230,14 @@ def resolve_region(td, ext): has_region = True else: print( - f" warning: output.render.{keys[d]} does not overlap the domain; ignoring" + f" warning: render.extent.{keys[d]} does not overlap the domain; ignoring" ) return [tuple(p) for p in region], has_region def apply_moving_view(td, region, ext, cam, time, dim): """Renderer::updateForTime: dt=max(0, t-t0); shift=vel*dt; pan region+eye.""" - vel = find_or(td, [], "output", "render", "camera_velocity") + vel = find_or(td, [], "render", "moving_view", "velocity") if not vel: return region v = [0.0, 0.0, 0.0] @@ -243,7 +245,7 @@ def apply_moving_view(td, region, ext, cam, time, dim): v[d] = float(vel[d]) if v == [0.0, 0.0, 0.0]: return region - t0 = float(find_or(td, 0.0, "output", "render", "camera_start_time")) + t0 = float(find_or(td, 0.0, "render", "moving_view", "start_time")) dt = max(0.0, time - t0) shift = [v[0] * dt, v[1] * dt, v[2] * dt] new_region = [] @@ -300,9 +302,9 @@ def field_line_seeds_3d(td, region, cam, H): coarse grid origin=extent.first in the uncropped case. We use the region box the camera frames so the schematic overlays the drawn cube. """ - fl_enable = find_or(td, False, "output", "render", "fieldlines", "enable") + fl_enable = find_or(td, False, "render", "fieldlines", "enable") # any scene may also request the overlay - scenes = find_or(td, [], "output", "render", "scenes") + scenes = find_or(td, [], "render", "scene") any_fl = any( (find_or(sc, False, "fieldlines") or find_or(sc, "", "field") == "fieldlines") for sc in scenes @@ -310,8 +312,8 @@ def field_line_seeds_3d(td, region, cam, H): if not (fl_enable or any_fl): return None - seed_px = float(find_or(td, 8.0, "output", "render", "fieldlines", "seed_px")) - seed_max = int(find_or(td, 4096, "output", "render", "fieldlines", "seed_max")) + seed_px = float(find_or(td, 8.0, "render", "fieldlines", "seed_px")) + seed_max = int(find_or(td, 4096, "render", "fieldlines", "seed_max")) wpp = (cam.half_h * 2.0) / float(H) size = [region[d][1] - region[d][0] for d in range(3)] @@ -473,10 +475,10 @@ def project_box(box, color, lw, label, ls="-"): # axes tick labels: for each axis, pick the FOREGROUND (silhouette) edge of # the framed box and annotate along it, exactly as out::drawAxes3D does, so # the labels never end up on an edge hidden behind the volume. - axes_on = find_or(td, False, "output", "render", "axes") - nticks = int(find_or(td, 5, "output", "render", "axis_ticks")) + axes_on = find_or(td, False, "render", "axes") + nticks = int(find_or(td, 5, "render", "axis_ticks")) frame_box = region if has_region else ext - axis_names = find_or(td, [], "output", "render", "axis_labels") + axis_names = find_or(td, [], "render", "axis_labels") default_names = ["x", "y", "z"] if axes_on: corners = cube_corners([frame_box[0], frame_box[1], frame_box[2]]) @@ -675,8 +677,8 @@ def draw_2d_cartesian(td, ext, region, has_region, W, H, out_path, sim_name): ) # ticks (nice numbers over the data box == region) - axes_on = find_or(td, False, "output", "render", "axes") - nticks = int(find_or(td, 5, "output", "render", "axis_ticks")) + axes_on = find_or(td, False, "render", "axes") + nticks = int(find_or(td, 5, "render", "axis_ticks")) if axes_on: for tv in nice_ticks(region[0][0], region[0][1], nticks): ax.axvline(tv, color="0.85", lw=0.5, zorder=0) @@ -737,8 +739,8 @@ def wedge_boundary(rmn, rmx, tmn, tmx, sign, color, lw, label=None): wedge_boundary(rmin, rmax, tmin, tmax, -1.0, "tab:blue", 2.0) # radial ticks along the symmetry axis (X=0) - axes_on = find_or(td, False, "output", "render", "axes") - nticks = int(find_or(td, 5, "output", "render", "axis_ticks")) + axes_on = find_or(td, False, "render", "axes") + nticks = int(find_or(td, 5, "render", "axis_ticks")) if axes_on: for Rv in nice_ticks(0.0, ext[0][1], nticks): ax.plot(0.0, Rv, marker="+", color="0.3", ms=6) @@ -775,13 +777,13 @@ def preview(args): td = tomllib.load(f) # renderer enabled? - if not find_or(td, False, "output", "render", "enable"): - print("note: [output.render].enable is false in this toml; previewing anyway.") + if not find_or(td, False, "render", "enable"): + print("note: [render].enable is false in this toml; previewing anyway.") sim_name = find_or(td, "sim", "simulation", "name") - width = int(find_or(td, 1024, "output", "render", "width")) - height = int(find_or(td, 1024, "output", "render", "height")) - mirror = bool(find_or(td, True, "output", "render", "mirror")) + width = int(find_or(td, 1024, "render", "width")) + height = int(find_or(td, 1024, "render", "height")) + mirror = bool(find_or(td, True, "render", "mirror")) ext, dim, cartesian, metric_name = global_extent(td) diff --git a/src/framework/CMakeLists.txt b/src/framework/CMakeLists.txt index e735890ab..7494c11b2 100644 --- a/src/framework/CMakeLists.txt +++ b/src/framework/CMakeLists.txt @@ -10,6 +10,7 @@ # * parameters/output.cpp # * parameters/algorithms.cpp # * parameters/extra.cpp +# * parameters/render.cpp # * simulation.cpp # * domain/grid.cpp # * domain/metadomain.cpp @@ -65,6 +66,7 @@ set(SOURCES ${SRC_DIR}/parameters/output.cpp ${SRC_DIR}/parameters/algorithms.cpp ${SRC_DIR}/parameters/extra.cpp + ${SRC_DIR}/parameters/render.cpp ${SRC_DIR}/domain/grid.cpp ${SRC_DIR}/domain/metadomain.cpp ${SRC_DIR}/domain/metadomain_sort.cpp diff --git a/src/framework/domain/io/render.cpp b/src/framework/domain/io/render.cpp index cbf11edf0..c67d2f44e 100644 --- a/src/framework/domain/io/render.cpp +++ b/src/framework/domain/io/render.cpp @@ -418,9 +418,8 @@ namespace ntt { } } if (bad_species) { - raise::Warning( - "output.render: invalid species in '" + field_name + "', skipping", - HERE); + raise::Warning("render: invalid species in '" + field_name + "', skipping", + HERE); return false; } @@ -759,7 +758,7 @@ namespace ntt { } } - raise::Warning("output.render: unknown field '" + field_name + + raise::Warning("render: unknown field '" + field_name + "' (expected N/Nppc/Rho/Charge, T{i}{j}, V{i}/Vmag, or " "{E,B,J}{mag,1,2,3,x,y,z}); skipping", HERE); @@ -954,7 +953,7 @@ namespace ntt { CommunicateBckp(*local_domain, { 0, 1 }); } else if (not have_tubes) { // standalone field-line scene but tracing produced nothing/disabled - raise::Warning("output.render: 'fieldlines' scene but no field-line " + raise::Warning("render: 'fieldlines' scene but no field-line " "geometry; skipping", HERE); continue; @@ -1434,10 +1433,9 @@ namespace ntt { } CommunicateBckp(*local_domain, { 0, 1 }); } else if (not have_fl) { - raise::Warning( - "output.render: 'fieldlines' scene needs a 2D run with " - "[output.render.fieldlines]; skipping", - HERE); + raise::Warning("render: 'fieldlines' scene needs a 2D run with " + "[render.fieldlines]; skipping", + HERE); continue; } const out::ContourSet& kc = show_lines ? contours : emptyc; diff --git a/src/framework/parameters/parameters.cpp b/src/framework/parameters/parameters.cpp index f3d8d507e..080944b4c 100644 --- a/src/framework/parameters/parameters.cpp +++ b/src/framework/parameters/parameters.cpp @@ -15,6 +15,7 @@ #include "framework/parameters/grid.h" #include "framework/parameters/output.h" #include "framework/parameters/particles.h" +#include "framework/parameters/render.h" #include @@ -165,6 +166,11 @@ namespace ntt { output_params.read(dim, get("particles.nspec"), toml_data); output_params.setParams(this); + /* [render] ------------------------------------------------------------- */ + params::Render render_params; + render_params.read(toml_data, this); + render_params.setParams(this); + /* [checkpoint] --------------------------------------------------------- */ set("checkpoint.interval", toml::find_or(toml_data, diff --git a/src/framework/parameters/render.cpp b/src/framework/parameters/render.cpp index e69de29bb..401c8f44d 100644 --- a/src/framework/parameters/render.cpp +++ b/src/framework/parameters/render.cpp @@ -0,0 +1,434 @@ +#include "framework/parameters/render.h" + +#include "global.h" + +#include "utils/error.h" +#include "utils/numeric.h" + +#include "framework/parameters/parameters.h" + +#include + +#include +#include +#include +#include + +namespace ntt { + namespace params { + + namespace { + // `render..` if present; otherwise nullopt (used for keys whose + // defaults depend on the domain geometry and are resolved by the renderer) + template + auto findOpt(const toml::value& toml_data, + const std::string& table, + const std::string& key) -> std::optional { + if (toml_data.contains("render") and + toml_data.at("render").contains(table) and + toml_data.at("render").at(table).contains(key)) { + return toml::find(toml_data, "render", table, key); + } + return std::nullopt; + } + } // namespace + + void Render::read(const toml::value& toml_data, + const SimulationParams* const params) { + enable = toml::find_or(toml_data, "render", "enable", false); + if (not enable) { + return; + } + + /* cadence -------------------------------------------------------------- */ + interval = toml::find_or(toml_data, "render", "interval", 0u); + interval_time = toml::find_or(toml_data, + "render", + "interval_time", + -1.0); + if ((interval.value() == 0) and (interval_time.value() == -1.0)) { + interval = params->template get("output.interval"); + interval_time = params->template get("output.interval_time"); + } + + /* image ---------------------------------------------------------------- */ + width = toml::find_or(toml_data, "render", "width", 1024); + height = toml::find_or(toml_data, "render", "height", 1024); + // `resolution` is a convenience that forces a square frame (width == + // height), the natural shape for a dome master. + const auto resolution = toml::find_or(toml_data, "render", "resolution", 0); + if (resolution > 0) { + width = resolution; + height = resolution; + } + raise::ErrorIf(width.value() <= 0 or height.value() <= 0, + "render.width and render.height must be > 0", + HERE); + n_lut = toml::find_or(toml_data, "render", "n_lut", 256); + background = toml::find_or>( + toml_data, + "render", + "background", + std::vector { ZERO, ZERO, ZERO }); + if (background->size() != 3) { + raise::Warning("render.background must have 3 entries [r, g, b]; " + "using black", + HERE); + background = std::vector { ZERO, ZERO, ZERO }; + } + colorbar = toml::find_or(toml_data, "render", "colorbar", true); + colorbar_outside = toml::find_or(toml_data, "render", "colorbar_outside", true); + mirror = toml::find_or(toml_data, "render", "mirror", true); + time_label = toml::find_or(toml_data, "render", "time_label", false); + axes = toml::find_or(toml_data, "render", "axes", false); + // empty => unset (the 2D slice then picks per-metric default names) + axis_labels = toml::find_or>( + toml_data, + "render", + "axis_labels", + std::vector {}); + axis_ticks = toml::find_or(toml_data, "render", "axis_ticks", 5); + spine_width = toml::find_or(toml_data, + "render", + "spine_width", + static_cast(2)); + + /* [render.extent] ------------------------------------------------------ */ + extent.emplace(); + for (const auto& key : { "x1", "x2", "x3" }) { + auto lim = toml::find_or>(toml_data, + "render", + "extent", + key, + std::vector {}); + if (not lim.empty() and (lim.size() != 2 or lim[1] <= lim[0])) { + raise::Warning("render.extent." + std::string(key) + + " must be [lo, hi] with hi > lo; ignoring", + HERE); + lim.clear(); + } + extent->push_back(lim); + } + + /* [render.volume] ------------------------------------------------------ */ + volume_samples = toml::find_or(toml_data, "render", "volume", "samples", 400); + volume_step_size = toml::find_or(toml_data, + "render", + "volume", + "step_size", + ZERO); + volume_early_term_alpha = toml::find_or(toml_data, + "render", + "volume", + "early_term_alpha", + static_cast(0.99)); + + /* [render.moving_view] ------------------------------------------------- */ + moving_view_velocity = toml::find_or>( + toml_data, + "render", + "moving_view", + "velocity", + std::vector {}); + moving_view_start_time = toml::find_or(toml_data, + "render", + "moving_view", + "start_time", + 0.0); + + /* [render.camera] ------------------------------------------------------ */ + camera_mode = toml::find_or(toml_data, + "render", + "camera", + "mode", + "orthographic"); + if (camera_mode.value() != "orthographic" and + camera_mode.value() != "perspective" and camera_mode.value() != "dome") { + raise::Warning( + "render.camera.mode '" + camera_mode.value() + + "' unknown (want orthographic/perspective/dome); using " + "orthographic projection", + HERE); + camera_mode = "orthographic"; + } + camera_position = toml::find_or>(toml_data, + "render", + "camera", + "position", + std::vector {}); + camera_look_at = toml::find_or>(toml_data, + "render", + "camera", + "look_at", + std::vector {}); + camera_up = toml::find_or>(toml_data, + "render", + "camera", + "up", + std::vector {}); + camera_fov = toml::find_or(toml_data, + "render", + "camera", + "fov", + static_cast(35)); + camera_dome_fov = toml::find_or(toml_data, + "render", + "camera", + "dome_fov", + static_cast(180)); + camera_dome_radius = findOpt(toml_data, "camera", "dome_radius"); + camera_ortho_height = findOpt(toml_data, "camera", "ortho_height"); + + /* [render.dome] -------------------------------------------------------- */ + dome_enable = toml::find_or(toml_data, "render", "dome", "enable", false); + dome_fov = toml::find_or(toml_data, + "render", + "dome", + "fov", + static_cast(180)); + dome_radius = findOpt(toml_data, "dome", "radius"); + dome_center = toml::find_or>(toml_data, + "render", + "dome", + "center", + std::vector {}); + if (not dome_center->empty() and dome_center->size() != 2) { + raise::Warning("render.dome.center must have 2 entries [x, y]; using " + "the domain center", + HERE); + dome_center->clear(); + } + dome_projection = toml::find_or(toml_data, + "render", + "dome", + "projection", + "equidistant"); + if (dome_projection.value() != "equidistant" and + dome_projection.value() != "gnomonic" and + dome_projection.value() != "stereographic" and + dome_projection.value() != "orthographic") { + raise::Warning("render.dome.projection '" + dome_projection.value() + + "' unknown; using 'equidistant'", + HERE); + dome_projection = "equidistant"; + } + + /* [[render.scene]] ----------------------------------------------------- */ + scenes.emplace(); + bool any_fieldlines = false; + const auto scenes_arr = toml::find_or(toml_data, + "render", + "scene", + toml::array {}); + for (const auto& sc : scenes_arr) { + RenderScene scene; + scene.field = toml::find_or(sc, "field", ""); + if (scene.field.empty()) { + raise::Warning("render.scene with no field; skipping", HERE); + continue; + } + scene.prefix = toml::find_or(sc, "prefix", scene.field + "_"); + scene.label = toml::find_or(sc, "label", scene.field); + scene.min = toml::find_or(sc, "min", ZERO); + scene.max = toml::find_or(sc, "max", ONE); + scene.log = toml::find_or(sc, "log", false); + scene.colormap = toml::find_or(sc, "colormap", "viridis"); + // alpha control points: array of [position, alpha] pairs + scene.alpha = toml::find_or>>( + sc, + "alpha", + std::vector> {}); + scene.colorbar_ticks = toml::find_or>( + sc, + "colorbar_ticks", + std::vector {}); + // overlay the field-line tubes inside this scene's volume; a dedicated + // `field = "fieldlines"` scene renders the tubes standalone (no volume). + scene.fieldlines = toml::find_or(sc, "fieldlines", false) or + (scene.field == "fieldlines"); + any_fieldlines = any_fieldlines or scene.fieldlines; + scenes->push_back(scene); + } + if (scenes->empty()) { + raise::Warning("render enabled but no valid scenes; disabling", HERE); + enable = false; + return; + } + + /* [render.fieldlines] -------------------------------------------------- */ + // the field lines are built whenever the section asks for them OR any + // scene requests the overlay (so a bare `field = "fieldlines"` scene + // works without a separate enable flag). + fieldlines_enable = toml::find_or(toml_data, + "render", + "fieldlines", + "enable", + false) or + any_fieldlines; + fieldlines_field = toml::find_or(toml_data, + "render", + "fieldlines", + "field", + "B"); + fieldlines_bin = toml::find_or(toml_data, "render", "fieldlines", "bin", 4); + fieldlines_bin = (fieldlines_bin.value() < 1) + ? 1 + : ((fieldlines_bin.value() > 16) + ? 16 + : fieldlines_bin.value()); + fieldlines_seed_px = toml::find_or(toml_data, + "render", + "fieldlines", + "seed_px", + static_cast(8)); + fieldlines_seed_max = toml::find_or(toml_data, + "render", + "fieldlines", + "seed_max", + 4096); + fieldlines_levels = toml::find_or(toml_data, + "render", + "fieldlines", + "levels", + 16); + fieldlines_tube_px = toml::find_or(toml_data, + "render", + "fieldlines", + "tube_px", + static_cast(2)); + fieldlines_colormap = toml::find_or(toml_data, + "render", + "fieldlines", + "colormap", + "inferno"); + // optional monochrome color [r,g,b]; overrides the colormap when set + fieldlines_color = toml::find_or>(toml_data, + "render", + "fieldlines", + "color", + std::vector {}); + if (not fieldlines_color->empty() and fieldlines_color->size() != 3) { + raise::Warning("render.fieldlines.color must have 3 entries [r, g, b]; " + "ignoring", + HERE); + fieldlines_color->clear(); + } + fieldlines_log = toml::find_or(toml_data, "render", "fieldlines", "log", false); + fieldlines_min = toml::find_or(toml_data, + "render", + "fieldlines", + "min", + ZERO); + fieldlines_max = toml::find_or(toml_data, + "render", + "fieldlines", + "max", + ZERO); + fieldlines_step_frac = toml::find_or(toml_data, + "render", + "fieldlines", + "step_frac", + static_cast(0.5)); + fieldlines_max_steps = toml::find_or(toml_data, + "render", + "fieldlines", + "max_steps", + 4000); + fieldlines_max_length = toml::find_or(toml_data, + "render", + "fieldlines", + "max_length", + static_cast(3)); + } + + void Render::setParams(SimulationParams* params) const { + params->set("render.enable", enable); + if (not enable) { + return; + } + params->set("render.interval", interval.value()); + params->set("render.interval_time", interval_time.value()); + + params->set("render.width", width.value()); + params->set("render.height", height.value()); + params->set("render.n_lut", n_lut.value()); + params->set("render.background", background.value()); + params->set("render.colorbar", colorbar.value()); + params->set("render.colorbar_outside", colorbar_outside.value()); + params->set("render.mirror", mirror.value()); + params->set("render.time_label", time_label.value()); + params->set("render.axes", axes.value()); + params->set("render.axis_labels", axis_labels.value()); + params->set("render.axis_ticks", axis_ticks.value()); + params->set("render.spine_width", spine_width.value()); + + params->set("render.extent.x1", extent.value()[0]); + params->set("render.extent.x2", extent.value()[1]); + params->set("render.extent.x3", extent.value()[2]); + + params->set("render.volume.samples", volume_samples.value()); + params->set("render.volume.step_size", volume_step_size.value()); + params->set("render.volume.early_term_alpha", + volume_early_term_alpha.value()); + + params->set("render.moving_view.velocity", moving_view_velocity.value()); + params->set("render.moving_view.start_time", moving_view_start_time.value()); + + params->set("render.camera.mode", camera_mode.value()); + params->set("render.camera.position", camera_position.value()); + params->set("render.camera.look_at", camera_look_at.value()); + params->set("render.camera.up", camera_up.value()); + params->set("render.camera.fov", camera_fov.value()); + params->set("render.camera.dome_fov", camera_dome_fov.value()); + if (camera_dome_radius.has_value()) { + params->set("render.camera.dome_radius", camera_dome_radius.value()); + } + if (camera_ortho_height.has_value()) { + params->set("render.camera.ortho_height", camera_ortho_height.value()); + } + + params->set("render.dome.enable", dome_enable.value()); + params->set("render.dome.fov", dome_fov.value()); + if (dome_radius.has_value()) { + params->set("render.dome.radius", dome_radius.value()); + } + params->set("render.dome.center", dome_center.value()); + params->set("render.dome.projection", dome_projection.value()); + + params->set("render.fieldlines.enable", fieldlines_enable.value()); + params->set("render.fieldlines.field", fieldlines_field.value()); + params->set("render.fieldlines.bin", fieldlines_bin.value()); + params->set("render.fieldlines.seed_px", fieldlines_seed_px.value()); + params->set("render.fieldlines.seed_max", fieldlines_seed_max.value()); + params->set("render.fieldlines.levels", fieldlines_levels.value()); + params->set("render.fieldlines.tube_px", fieldlines_tube_px.value()); + params->set("render.fieldlines.colormap", fieldlines_colormap.value()); + params->set("render.fieldlines.color", fieldlines_color.value()); + params->set("render.fieldlines.log", fieldlines_log.value()); + params->set("render.fieldlines.min", fieldlines_min.value()); + params->set("render.fieldlines.max", fieldlines_max.value()); + params->set("render.fieldlines.step_frac", fieldlines_step_frac.value()); + params->set("render.fieldlines.max_steps", fieldlines_max_steps.value()); + params->set("render.fieldlines.max_length", fieldlines_max_length.value()); + + // scenes are flattened into indexed keys (`render.scene..`) so + // that every entry stays a plain (serializable) parameter type + params->set("render.nscenes", scenes->size()); + for (std::size_t i = 0; i < scenes->size(); ++i) { + const auto& sc = scenes.value()[i]; + const auto pfx = "render.scene." + std::to_string(i) + "."; + params->set(pfx + "field", sc.field); + params->set(pfx + "prefix", sc.prefix); + params->set(pfx + "label", sc.label); + params->set(pfx + "min", sc.min); + params->set(pfx + "max", sc.max); + params->set(pfx + "log", sc.log); + params->set(pfx + "colormap", sc.colormap); + params->set(pfx + "alpha", sc.alpha); + params->set(pfx + "colorbar_ticks", sc.colorbar_ticks); + params->set(pfx + "fieldlines", sc.fieldlines); + } + } + + } // namespace params +} // namespace ntt diff --git a/src/framework/parameters/render.h b/src/framework/parameters/render.h index 4162e2965..bde7e93af 100644 --- a/src/framework/parameters/render.h +++ b/src/framework/parameters/render.h @@ -2,11 +2,16 @@ * @file framework/parameters/render.h * @brief Auxiliary functions for reading in on-the-fly render parameters * @implements + * - ntt::params::RenderScene * - ntt::params::Render * @cpp: * - render.cpp * @namespaces: * - ntt::params:: + * @note Only the raw (geometry-independent) configuration is resolved here; + * defaults that depend on the domain extent (camera framing, dome radius, + * clamping of the render region to the box) are resolved by out::Renderer. + * Those keys are only set in SimulationParams when given in the input. */ #ifndef FRAMEWORK_PARAMETERS_RENDER_H #define FRAMEWORK_PARAMETERS_RENDER_H @@ -17,21 +22,99 @@ #include -#include #include +#include +#include namespace ntt { namespace params { + struct RenderScene { + std::string field; + std::string prefix; + std::string label; + real_t min; + real_t max; + bool log; + std::string colormap; + std::vector> alpha; + std::vector colorbar_ticks; + bool fieldlines; + }; + struct Render { + bool enable { false }; + + std::optional interval; + std::optional interval_time; + + std::optional width; + std::optional height; + std::optional n_lut; + std::optional> background; + std::optional colorbar; + std::optional colorbar_outside; + std::optional mirror; + std::optional time_label; + std::optional axes; + std::optional> axis_labels; + std::optional axis_ticks; + std::optional spine_width; + + // [render.extent]: x{1,2,3} -> [lo, hi] or empty (full extent) + std::optional>> extent; + + // [render.volume] + std::optional volume_samples; + std::optional volume_step_size; + std::optional volume_early_term_alpha; + + // [render.moving_view] + std::optional> moving_view_velocity; + std::optional moving_view_start_time; + + // [render.camera] + std::optional camera_mode; + std::optional> camera_position; + std::optional> camera_look_at; + std::optional> camera_up; + std::optional camera_fov; + std::optional camera_dome_fov; + std::optional camera_dome_radius; + std::optional camera_ortho_height; + + // [render.dome] + std::optional dome_enable; + std::optional dome_fov; + std::optional dome_radius; + std::optional> dome_center; + std::optional dome_projection; + + // [render.fieldlines] + std::optional fieldlines_enable; + std::optional fieldlines_field; + std::optional fieldlines_bin; + std::optional fieldlines_seed_px; + std::optional fieldlines_seed_max; + std::optional fieldlines_levels; + std::optional fieldlines_tube_px; + std::optional fieldlines_colormap; + std::optional> fieldlines_color; + std::optional fieldlines_log; + std::optional fieldlines_min; + std::optional fieldlines_max; + std::optional fieldlines_step_frac; + std::optional fieldlines_max_steps; + std::optional fieldlines_max_length; + + // [[render.scene]] + std::optional> scenes; - void read(const std::map&, - const toml::value&, - const SimulationParams* const); - void setParams(const std::map&, SimulationParams*) const; + void read(const toml::value&, const SimulationParams* const); + void setParams(SimulationParams*) const; }; } // namespace params } // namespace ntt -#endif +#endif // FRAMEWORK_PARAMETERS_RENDER_H diff --git a/src/output/render/renderer.cpp b/src/output/render/renderer.cpp index 433315199..312b0683d 100644 --- a/src/output/render/renderer.cpp +++ b/src/output/render/renderer.cpp @@ -14,8 +14,6 @@ #include "output/render/png.h" #include "output/render/transfer_fn.h" -#include - #if defined(MPI_ENABLED) #include "arch/mpi_aliases.h" @@ -65,17 +63,15 @@ namespace out { void Renderer::init(const ntt::SimulationParams& params, const boundaries_t& global_extent) { - m_enabled = false; - const auto& td = params.data(); + m_enabled = false; - const bool enable = toml::find_or(td, "output", "render", "enable", false); - if (not enable) { + if (not params.get("render.enable")) { return; } // 2D (slice rasterizer) and 3D Cartesian (volume ray-march) are supported; // 1D has nothing to render. if (global_extent.size() != 2 and global_extent.size() != 3) { - raise::Warning("output.render enabled but simulation is 1D; " + raise::Warning("render enabled but simulation is 1D; " "the renderer will be inactive", HERE); return; @@ -83,87 +79,53 @@ namespace out { m_root = path_t(params.get("simulation.name")); - m_width = toml::find_or(td, "output", "render", "width", 1024); - m_height = toml::find_or(td, "output", "render", "height", 1024); - // `resolution` is a convenience that forces a square frame (width == - // height), the natural shape for a dome master. - const int resolution = toml::find_or(td, "output", "render", "resolution", 0); - if (resolution > 0) { - m_width = resolution; - m_height = resolution; - } - m_samples = toml::find_or(td, "output", "render", "samples", 400); - m_step_size = toml::find_or(td, "output", "render", "step_size", ZERO); - m_early_alpha = toml::find_or(td, - "output", - "render", - "early_term_alpha", - static_cast(0.99)); - m_n_lut = toml::find_or(td, "output", "render", "n_lut", 256); + m_width = params.get("render.width"); + m_height = params.get("render.height"); + m_samples = params.get("render.volume.samples"); + m_step_size = params.get("render.volume.step_size"); + m_early_alpha = params.get("render.volume.early_term_alpha"); + m_n_lut = params.get("render.n_lut"); - // opaque background color (shows through low-alpha pixels); default black - const auto bg = toml::find_or>(td, - "output", - "render", - "background", - std::vector {}); - if (bg.size() == 3) { - m_background[0] = bg[0]; - m_background[1] = bg[1]; - m_background[2] = bg[2]; + // opaque background color (shows through low-alpha pixels) + const auto bg = params.get>("render.background"); + for (auto i = 0u; i < 3; ++i) { + m_background[i] = bg[i]; } - m_colorbar = toml::find_or(td, "output", "render", "colorbar", true); - m_colorbar_outside = toml::find_or(td, - "output", - "render", - "colorbar_outside", - true); + m_colorbar = params.get("render.colorbar"); + m_colorbar_outside = params.get("render.colorbar_outside"); // 2D slice mode (spherical only): mirror the half-plane into a full disk - m_mirror = toml::find_or(td, "output", "render", "mirror", true); + m_mirror = params.get("render.mirror"); // draw the current simulation time in the upper-right corner - m_time_label = toml::find_or(td, "output", "render", "time_label", false); + m_time_label = params.get("render.time_label"); // axes: spine + ticks + labels around the rendered region - m_axes = toml::find_or(td, "output", "render", "axes", false); - m_axis_nticks = toml::find_or(td, "output", "render", "axis_ticks", 5); - m_spine_width = toml::find_or(td, - "output", - "render", - "spine_width", - static_cast(2)); + m_axes = params.get("render.axes"); + m_axis_nticks = params.get("render.axis_ticks"); + m_spine_width = params.get("render.spine_width"); m_global_extent = global_extent; // optional axis-aligned render region (physical coords). Unset axes default // to the full extent; user limits are clamped to the box (nothing to render - // outside it). x{1,2,3}_lim -> axes {0,1,2} (r/theta for spherical 2D). + // outside it). extent.x{1,2,3} -> axes {0,1,2} (r/theta for spherical 2D). m_region = global_extent; m_has_region = false; { - const char* keys[3] = { "x1_lim", "x2_lim", "x3_lim" }; + const char* keys[3] = { "x1", "x2", "x3" }; for (size_t d = 0; d < global_extent.size() and d < 3; ++d) { - const auto lim = toml::find_or>(td, - "output", - "render", - keys[d], - std::vector {}); + const auto lim = params.get>( + "render.extent." + std::string(keys[d])); if (lim.empty()) { continue; } - if (lim.size() != 2 or lim[1] <= lim[0]) { - raise::Warning("output.render." + std::string(keys[d]) + - " must be [lo, hi] with hi > lo; ignoring", - HERE); - continue; - } const real_t lo = std::max(lim[0], global_extent[d].first); const real_t hi = std::min(lim[1], global_extent[d].second); if (hi > lo) { m_region[d] = { lo, hi }; m_has_region = true; } else { - raise::Warning("output.render." + std::string(keys[d]) + + raise::Warning("render.extent." + std::string(keys[d]) + " does not overlap the domain; ignoring", HERE); } @@ -176,98 +138,66 @@ namespace out { // background border -> a valid dome master). The 3D dome is a separate // workstream; warn if asked for here so it does not silently fall back to // the volume camera. - { - const bool dome_enable = - toml::find_or(td, "output", "render", "dome", "enable", false); - if (dome_enable) { - if (global_extent.size() != 2) { - raise::Warning("output.render.dome is 2D-only for now; ignoring", HERE); + if (params.get("render.dome.enable")) { + if (global_extent.size() != 2) { + raise::Warning("render.dome is 2D-only for now; ignoring", HERE); + } else { + m_dome.enabled = true; + const real_t fov = params.get("render.dome.fov"); + m_dome.theta_max = HALF * fov * static_cast(constant::PI) / + static_cast(180); + const auto proj = params.get("render.dome.projection"); + if (proj == "gnomonic") { + m_dome.law = DomeMap::Gnomonic; + } else if (proj == "stereographic") { + m_dome.law = DomeMap::Stereographic; + } else if (proj == "orthographic") { + m_dome.law = DomeMap::Orthographic; } else { - m_dome.enabled = true; - const real_t fov = toml::find_or(td, - "output", - "render", - "dome", - "fov", - static_cast(180)); - m_dome.theta_max = HALF * fov * static_cast(constant::PI) / - static_cast(180); - const auto proj = toml::find_or(td, - "output", - "render", - "dome", - "projection", - "equidistant"); - if (proj == "gnomonic") { - m_dome.law = DomeMap::Gnomonic; - } else if (proj == "stereographic") { - m_dome.law = DomeMap::Stereographic; - } else if (proj == "orthographic") { - m_dome.law = DomeMap::Orthographic; - } else { - if (proj != "equidistant") { - raise::Warning("output.render.dome.projection '" + proj + - "' unknown; using 'equidistant'", - HERE); - } - m_dome.law = DomeMap::Equidistant; - } - // the gnomonic (flat-tangent) law diverges as the dome half-FOV -> 90 - // deg (a flat plane never reaches the horizon), so cap it below that. - if (m_dome.law == DomeMap::Gnomonic) { - const real_t cap = static_cast(89.0 * constant::PI / 180.0); - if (m_dome.theta_max >= cap) { - raise::Warning( - "output.render.dome: 'gnomonic' needs fov < 180 deg " - "(a flat plane cannot reach the dome horizon); " - "capping the half-FOV at 89 deg", - HERE); - m_dome.theta_max = cap; - } - } - // default center = domain center; default radius = the largest disk - // that fits inside the (rectangular) domain (half the shorter side). - const real_t Lx = global_extent[0].second - global_extent[0].first; - const real_t Ly = global_extent[1].second - global_extent[1].first; - m_dome.cx = HALF * (global_extent[0].first + global_extent[0].second); - m_dome.cy = HALF * (global_extent[1].first + global_extent[1].second); - const auto ctr = toml::find_or>( - td, - "output", - "render", - "dome", - "center", - std::vector {}); - if (ctr.size() == 2) { - m_dome.cx = ctr[0]; - m_dome.cy = ctr[1]; - } else if (not ctr.empty()) { - raise::Warning("output.render.dome.center must have 2 entries " - "[x, y]; using the domain center", - HERE); - } - const real_t rdef = HALF * std::min(Lx, Ly); - m_dome.R = toml::find_or(td, "output", "render", "dome", "radius", rdef); - if (m_dome.R <= ZERO) { - m_dome.R = rdef; - } - if (m_width != m_height) { - raise::Warning("output.render.dome: width != height; the fisheye " - "disk is centered on the shorter side and the frame " - "is not a square dome master", + m_dome.law = DomeMap::Equidistant; + } + // the gnomonic (flat-tangent) law diverges as the dome half-FOV -> 90 + // deg (a flat plane never reaches the horizon), so cap it below that. + if (m_dome.law == DomeMap::Gnomonic) { + const real_t cap = static_cast(89.0 * constant::PI / 180.0); + if (m_dome.theta_max >= cap) { + raise::Warning("render.dome: 'gnomonic' needs fov < 180 deg " + "(a flat plane cannot reach the dome horizon); " + "capping the half-FOV at 89 deg", HERE); + m_dome.theta_max = cap; } } + // default center = domain center; default radius = the largest disk + // that fits inside the (rectangular) domain (half the shorter side). + const real_t Lx = global_extent[0].second - global_extent[0].first; + const real_t Ly = global_extent[1].second - global_extent[1].first; + m_dome.cx = HALF * (global_extent[0].first + global_extent[0].second); + m_dome.cy = HALF * (global_extent[1].first + global_extent[1].second); + const auto ctr = params.get>("render.dome.center"); + if (ctr.size() == 2) { + m_dome.cx = ctr[0]; + m_dome.cy = ctr[1]; + } + const real_t rdef = HALF * std::min(Lx, Ly); + m_dome.R = params.contains("render.dome.radius") + ? params.get("render.dome.radius") + : rdef; + if (m_dome.R <= ZERO) { + m_dome.R = rdef; + } + if (m_width != m_height) { + raise::Warning("render.dome: width != height; the fisheye " + "disk is centered on the shorter side and the frame " + "is not a square dome master", + HERE); + } } } { - const auto al = toml::find_or>( - td, - "output", - "render", - "axis_labels", - std::vector {}); + const auto al = params.get>( + "render.axis_labels"); m_axis_labels_set = not al.empty(); for (size_t d = 0; d < al.size() and d < 3; ++d) { m_axis_labels[d] = al[d]; @@ -277,92 +207,45 @@ namespace out { m_slice_ylabel = m_axis_labels[1]; } - // cadence: mirror output.* (interval in steps; interval_time in sim time) - auto interval = toml::find_or(td, "output", "render", "interval", 0u); - auto interval_time = toml::find_or(td, - "output", - "render", - "interval_time", - -1.0); - if ((interval == 0) and (interval_time == -1.0)) { - interval = params.template get("output.interval"); - interval_time = params.template get("output.interval_time"); - } - m_tracker.init("render", interval, interval_time); + // cadence (falls back to output.interval{,_time}; see params::Render) + m_tracker.init("render", + params.get("render.interval"), + params.get("render.interval_time")); /* ---- camera (used by the 3D volume mode; the 2D slice path frames itself * and ignores this, so a missing 3rd axis is zero-filled harmlessly) ---- */ // frame the camera on the render region (== the full extent when uncropped) real_t center[3] = { ZERO, ZERO, ZERO }, size[3] = { ZERO, ZERO, ZERO }; - real_t maxext = ZERO; for (size_t d = 0; d < m_region.size() and d < 3; ++d) { center[d] = static_cast(0.5) * (m_region[d].first + m_region[d].second); size[d] = m_region[d].second - m_region[d].first; - maxext = (size[d] > maxext) ? size[d] : maxext; } const real_t diag = std::sqrt( size[0] * size[0] + size[1] * size[1] + size[2] * size[2]); - const auto cam_mode = toml::find_or(td, - "output", - "render", - "camera", - "mode", - std::string {}); + // "orthographic" | "perspective" | "dome". The dome is a fulldome + // azimuthal-equidistant fisheye from an INTERIOR eye (the box center by + // default) -- see Metadomain::Render (3D). + const auto cam_mode = params.get("render.camera.mode"); auto projection = CameraDevice::Ortho; if (cam_mode == "dome") { projection = CameraDevice::Dome; } else if (cam_mode == "perspective") { projection = CameraDevice::Perspective; - } else if (cam_mode == "orthographic") { - projection = CameraDevice::Ortho; - } else if (not cam_mode.empty()) { - raise::Warning("output.render.camera.mode '" + cam_mode + - "' unknown (want orthographic/perspective/dome); using " - "orthographic projection", - HERE); } const bool is_dome = (projection == CameraDevice::Dome); - const real_t dome_fov = toml::find_or(td, - "output", - "render", - "camera", - "dome_fov", - static_cast(180)); - auto pos = toml::find_or>(td, - "output", - "render", - "camera", - "position", - std::vector {}); - auto look = toml::find_or>(td, - "output", - "render", - "camera", - "look_at", - std::vector {}); - auto up = toml::find_or>(td, - "output", - "render", - "camera", - "up", - std::vector {}); - const real_t fov = toml::find_or(td, - "output", - "render", - "camera", - "fov", - static_cast(35.0)); + const real_t dome_fov = params.get("render.camera.dome_fov"); + const auto pos = params.get>("render.camera.position"); + const auto look = params.get>("render.camera.look_at"); + const auto up = params.get>("render.camera.up"); + const real_t fov = params.get("render.camera.fov"); // default covers the box from any view direction (default camera looks // down the diagonal), so nothing is clipped without explicit framing. - (void)maxext; - const real_t ortho_height = toml::find_or(td, - "output", - "render", - "camera", - "ortho_height", - diag); + const real_t ortho_height = params.contains("render.camera.ortho_height") + ? params.get( + "render.camera.ortho_height") + : diag; real_t eye[3], lookat[3], upv[3]; for (int d = 0; d < 3; ++d) { @@ -422,11 +305,8 @@ namespace out { m_camera_dev.tan_half_fov = std::tan(static_cast(0.5) * fov * static_cast(constant::PI) / static_cast(180.0)); - // keep `orthographic` consistent with the resolved projection so that - // `mode` actually overrides the flag: the kernel/screenBBox pick ortho vs - // perspective from `orthographic`, and only Dome is read off `projection`. - // (When `mode` is unset, `projection` was derived from `ortho`, so this - // round-trips to the original flag -- back-compatible.) + // the kernel/screenBBox pick ortho vs perspective from `orthographic`, and + // only Dome is read off `projection`. m_camera_dev.orthographic = (projection == CameraDevice::Ortho); m_camera_dev.half_h = static_cast(0.5) * ortho_height; m_camera_dev.half_w = m_camera_dev.half_h * m_camera_dev.aspect; @@ -445,7 +325,7 @@ namespace out { m_camera_dev.aspect = ONE; m_camera_dev.half_w = m_camera_dev.half_h; if (m_width != m_height) { - raise::Warning("output.render.camera.mode='dome' wants width == height " + raise::Warning("render.camera.mode='dome' wants width == height " "for a circular dome master; the fisheye disk will be " "elliptical otherwise", HERE); @@ -461,12 +341,9 @@ namespace out { insc = (size[d] < insc) ? size[d] : insc; } insc *= HALF; - real_t domeR = toml::find_or(td, - "output", - "render", - "camera", - "dome_radius", - insc); + real_t domeR = params.contains("render.camera.dome_radius") + ? params.get("render.camera.dome_radius") + : insc; if (domeR < ZERO) { domeR = insc; } @@ -480,63 +357,42 @@ namespace out { m_eye_base[d] = m_camera_dev.eye[d]; } { - const auto vel = toml::find_or>(td, - "output", - "render", - "camera_velocity", - std::vector {}); + const auto vel = params.get>( + "render.moving_view.velocity"); for (size_t d = 0; d < vel.size() and d < 3; ++d) { m_cam_vel[d] = vel[d]; } m_cam_moving = (m_cam_vel[0] != ZERO) or (m_cam_vel[1] != ZERO) or (m_cam_vel[2] != ZERO); - m_cam_t0 = toml::find_or(td, - "output", - "render", - "camera_start_time", - 0.0); + m_cam_t0 = params.get("render.moving_view.start_time"); if (m_cam_moving and not m_has_region and global_extent.size() == 2) { - raise::Warning("output.render.camera_velocity set without x{1,2}_lim: " - "the 2D window will pan off the domain. Set a region to " - "track a feature within it.", + raise::Warning("render.moving_view.velocity set without " + "render.extent.x{1,2}: the 2D window will pan off the " + "domain. Set a region to track a feature within it.", HERE); } } /* ---- scenes --------------------------------------------------------- */ m_scenes.clear(); - const auto scenes_arr = toml::find_or(td, - "output", - "render", - "scenes", - toml::array {}); - for (const auto& sc : scenes_arr) { - Scene scene; - scene.field = toml::find_or(sc, "field", ""); - scene.prefix = toml::find_or(sc, "prefix", scene.field + "_"); - if (scene.field.empty()) { - raise::Warning("output.render scene with no field; skipping", HERE); - continue; - } - scene.label = toml::find_or(sc, "label", scene.field); - scene.ticks = toml::find_or>(sc, - "colorbar_ticks", - std::vector {}); - // overlay the B-field-line tubes inside this scene's volume; a dedicated - // `field = "fieldlines"` scene renders the tubes standalone (no volume). - scene.show_fieldlines = toml::find_or(sc, "fieldlines", false) or - (scene.field == "fieldlines"); - scene.tf.vmin = toml::find_or(sc, "min", ZERO); - scene.tf.vmax = toml::find_or(sc, "max", ONE); - scene.tf.log_scale = toml::find_or(sc, "log", false); - scene.tf.n_lut = m_n_lut; - const auto colormap = toml::find_or(sc, "colormap", "viridis"); - scene.tf.colormap = colormap; + const auto nscenes = params.get("render.nscenes"); + for (std::size_t i = 0; i < nscenes; ++i) { + const auto pfx = "render.scene." + std::to_string(i) + "."; + Scene scene; + scene.field = params.get(pfx + "field"); + scene.prefix = params.get(pfx + "prefix"); + scene.label = params.get(pfx + "label"); + scene.ticks = params.get>(pfx + "colorbar_ticks"); + scene.show_fieldlines = params.get(pfx + "fieldlines"); + scene.tf.vmin = params.get(pfx + "min"); + scene.tf.vmax = params.get(pfx + "max"); + scene.tf.log_scale = params.get(pfx + "log"); + scene.tf.n_lut = m_n_lut; + const auto colormap = params.get(pfx + "colormap"); + scene.tf.colormap = colormap; // alpha control points: array of [position, alpha] pairs - const auto alpha_raw = toml::find_or>>( - sc, - "alpha", - std::vector> {}); + const auto alpha_raw = params.get>>( + pfx + "alpha"); std::vector> alpha_pts; for (const auto& p : alpha_raw) { if (p.size() >= 2) { @@ -554,101 +410,28 @@ namespace out { m_scenes.push_back(std::move(scene)); } - if (m_scenes.empty()) { - raise::Warning("output.render enabled but no valid scenes; disabling", HERE); - return; - } - /* ---- magnetic-field-line tube overlay ------------------------------- */ - // The tubes are built whenever the [output.render.fieldlines] section asks - // for them OR any scene requests the overlay (so a bare `field = - // "fieldlines"` scene works without a separate enable flag). - bool any_fl = false; - for (const auto& s : m_scenes) { - any_fl = any_fl or s.show_fieldlines; - } - m_fieldlines.enable = - toml::find_or(td, "output", "render", "fieldlines", "enable", false) or - any_fl; + // enabled by [render.fieldlines] OR by any scene requesting the overlay + // (resolved in params::Render) + m_fieldlines.enable = params.get("render.fieldlines.enable"); if (m_fieldlines.enable) { - if (m_global_extent.size() != 2 and m_global_extent.size() != 3) { - raise::Warning( - "output.render.fieldlines needs a 2D or 3D run; ignoring", - HERE); - m_fieldlines.enable = false; - } else { - // 3D -> traced tubes inside the volume; 2D -> flux-function contours - auto& fl = m_fieldlines; - fl.field = toml::find_or(td, - "output", - "render", - "fieldlines", - "field", - "B"); - fl.bin = toml::find_or(td, "output", "render", "fieldlines", "bin", 4); - fl.bin = (fl.bin < 1) ? 1 : ((fl.bin > 16) ? 16 : fl.bin); - fl.seed_px = toml::find_or(td, - "output", - "render", - "fieldlines", - "seed_px", - static_cast(8)); - fl.tube_px = toml::find_or(td, - "output", - "render", - "fieldlines", - "tube_px", - static_cast(2)); - fl.colormap = toml::find_or(td, - "output", - "render", - "fieldlines", - "colormap", - "inferno"); - // optional monochrome color [r,g,b]; overrides the colormap when set - fl.color = toml::find_or>(td, - "output", - "render", - "fieldlines", - "color", - std::vector {}); - if (not fl.color.empty() and fl.color.size() != 3) { - raise::Warning("output.render.fieldlines.color must have 3 entries " - "[r,g,b]; ignoring", - HERE); - fl.color.clear(); - } - fl.log_scale = - toml::find_or(td, "output", "render", "fieldlines", "log", false); - fl.vmin = toml::find_or(td, "output", "render", "fieldlines", "min", ZERO); - fl.vmax = toml::find_or(td, "output", "render", "fieldlines", "max", ZERO); - fl.step_frac = toml::find_or(td, - "output", - "render", - "fieldlines", - "step_frac", - static_cast(0.5)); - fl.max_steps = toml::find_or(td, - "output", - "render", - "fieldlines", - "max_steps", - 4000); - fl.max_len_frac = toml::find_or(td, - "output", - "render", - "fieldlines", - "max_length", - static_cast(3)); - fl.seed_max = toml::find_or(td, - "output", - "render", - "fieldlines", - "seed_max", - 4096); - fl.levels = - toml::find_or(td, "output", "render", "fieldlines", "levels", 16); - } + // 3D -> traced tubes inside the volume; 2D -> flux-function contours + auto& fl = m_fieldlines; + fl.field = params.get("render.fieldlines.field"); + fl.bin = params.get("render.fieldlines.bin"); + fl.seed_px = params.get("render.fieldlines.seed_px"); + fl.tube_px = params.get("render.fieldlines.tube_px"); + fl.colormap = params.get("render.fieldlines.colormap"); + // optional monochrome color [r,g,b]; overrides the colormap when set + fl.color = params.get>("render.fieldlines.color"); + fl.log_scale = params.get("render.fieldlines.log"); + fl.vmin = params.get("render.fieldlines.min"); + fl.vmax = params.get("render.fieldlines.max"); + fl.step_frac = params.get("render.fieldlines.step_frac"); + fl.max_steps = params.get("render.fieldlines.max_steps"); + fl.max_len_frac = params.get("render.fieldlines.max_length"); + fl.seed_max = params.get("render.fieldlines.seed_max"); + fl.levels = params.get("render.fieldlines.levels"); } m_enabled = true; diff --git a/src/output/render/renderer.h b/src/output/render/renderer.h index a529ee4e9..213ab04ef 100644 --- a/src/output/render/renderer.h +++ b/src/output/render/renderer.h @@ -255,8 +255,8 @@ namespace out { Renderer(Renderer&&) = default; /** - * @brief Parse `[output.render.*]` and build the camera + per-scene LUTs. - * @param params simulation parameters (raw toml read via params.data()) + * @brief Build the camera + per-scene LUTs from the `render.*` parameters. + * @param params simulation parameters (see ntt::params::Render) * @param global_extent global physical box, for default camera framing */ void init(const ntt::SimulationParams& params, @@ -269,8 +269,8 @@ namespace out { /** * @brief Advance the moving view to `time`: translate the render region (and - * the 3D camera) by `camera_velocity * max(0, time - camera_start_time)`. - * @note A no-op unless `camera_velocity` was set. Call once per frame, before + * the 3D camera) by `moving_view.velocity * max(0, time - moving_view.start_time)`. + * @note A no-op unless `moving_view.velocity` was set. Call once per frame, before * reading region()/camera(). All ranks pass the same time, so the * shifted view is identical everywhere (the composite stays seamless). */ @@ -395,7 +395,7 @@ namespace out { return m_spine_width; } - // whether `output.render.axis_labels` was set in the toml (so the 2D path + // whether `render.axis_labels` was set (so the 2D path // honors it instead of substituting per-metric defaults) [[nodiscard]] auto axisLabelsSet() const -> bool { @@ -505,7 +505,7 @@ namespace out { // know the render mode (size 2 => 2D slice, size 3 => 3D volume). boundaries_t m_global_extent; // resolved render region [lo, hi] per axis (== global extent unless the - // user set x{1,2,3}_lim); the volume is clipped / the slice window is framed + // user set render.extent.x{1,2,3}); the volume is clipped / the slice window is framed // to this, and the default camera frames it. `m_region` is the CURRENT region // (shifted by the moving view below); `m_region_base` is the static toml one. boundaries_t m_region; From 23a3eae55c55a446c8329c071a01ed0a7b240ded Mon Sep 17 00:00:00 2001 From: haykh Date: Mon, 28 Sep 2026 09:56:13 -0400 Subject: [PATCH 120/125] disable certain flags if reqs not met --- CMakeLists.txt | 3 +++ 1 file changed, 3 insertions(+) diff --git a/CMakeLists.txt b/CMakeLists.txt index c904ccab2..d94a5b988 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -154,6 +154,9 @@ endif() if(NOT ${DEVICE_ENABLED}) set(vendor_sort OFF) +endif() + +if(NOT ${DEVICE_ENABLED} OR NOT ${mpi}) set(gpu_aware_mpi OFF) endif() From 84c89d386e0dafa7e21e13fe1ae1e7ae9032ca43 Mon Sep 17 00:00:00 2001 From: haykh Date: Mon, 28 Sep 2026 10:19:48 -0400 Subject: [PATCH 121/125] formatting with tombi --- .tombi.toml | 5 +- input.default.toml | 260 ++++++++++++------------- pgens/magnetosphere/magnetosphere.toml | 2 +- pgens/shock/shock.toml | 18 +- pgens/streaming/twostream.toml | 8 +- pgens/streaming/weibel.toml | 8 +- pgens/turbulence/turbulence.toml | 2 +- pgens/wald/wald.toml | 2 +- scripts/dependencies.py | 66 +++---- 9 files changed, 181 insertions(+), 190 deletions(-) diff --git a/.tombi.toml b/.tombi.toml index ea7e8c58b..977830c46 100644 --- a/.tombi.toml +++ b/.tombi.toml @@ -4,6 +4,7 @@ toml-version = "v1.0.0" [format.rules] indent-sub-tables = true indent-table-key-value-pairs = true + key-value-equals-sign-alignment = true trailing-comment-alignment = true [schema] @@ -15,8 +16,8 @@ toml-version = "v1.0.0" include = ["**/*.toml"] exclude = [ ".tombi.toml", - "extern/**", # submodules: adios2's pyproject/REUSE, entity-pgens' own configs + "extern/**", ".venv/**", "build/**", - "**/*.ckpt/**", # checkpoint metadata dumps carry a [metadata] table, not input + "**/*.ckpt/**", ] diff --git a/input.default.toml b/input.default.toml index 3377cc6f7..17daf3819 100644 --- a/input.default.toml +++ b/input.default.toml @@ -4,12 +4,12 @@ # @required # @type: string # @note: The name is used for the output files - name = "" + name = "" # Simulation engine to use # @required # @type: string # @enum: "SRPIC", "GRPIC" - engine = "SRPIC" + engine = "SRPIC" # Max runtime in physical (code) units # @required # @type: float [> 0] @@ -21,7 +21,7 @@ # Number of domains # @type: int # @default: 1 [no MPI]; MPI_SIZE [MPI] - number = 1 + number = 1 # Decomposition of the domain (for MPI) in each of the directions # @type: array [size 1 :->: 3] # @default: [-1, -1, -1] @@ -40,11 +40,11 @@ # Enable dynamic load balancing # @type: bool # @default: false - enable = false + enable = false # Run the rebalancer every `interval` timesteps (0 disables) # @type: int # @default: 0 - interval = 0 + interval = 0 # Dimensions along which load is redistributed (1 = x1, 2 = x2, 3 = x3) # @type: array [subset of {1, 2, 3}] # @default: [1] @@ -54,13 +54,13 @@ # particle count is below this fraction # @type: float # @default: 0.1 - tolerance = 0.1 + tolerance = 0.1 # Maximum cell-shift per interior boundary per event; clamped at compile # time to N_GHOSTS so the migrating field strip is already cached in the # rank's ghost zone. # @type: int # @default: N_GHOSTS - max_shift = 0 + max_shift = 0 # Parameters specific to grid geometry [grid] @@ -77,7 +77,7 @@ # are set automatically # @note: For cartesian geometry, cell aspect ratio has to be 1: `dx=dy=dz` # @example: [[0.0, 1.0], [-1.0, 1.0]] - extent = [[0.0, 0.0]] + extent = [[0.0, 0.0]] # @inferred: # - dim @@ -93,7 +93,7 @@ # @type: string # @enum: "Minkowski", "Spherical", "QSpherical", "Kerr_Schild", # "QKerr_Schild", "Kerr_Schild_0" - metric = "Minkowski" + metric = "Minkowski" # `r0` paramter for the QSpherical metric `x1 = log(r-r0)` # @type: float [-inf -> rmin] # @default: 0.0 @@ -103,11 +103,11 @@ # (pi-2*x2)*(pi-x2)/pi^2` # @type: float [-1 :->: 1] # @default: 0.0 - qsph_h = 0.0 + qsph_h = 0.0 # Spin parameter for the Kerr Schild metric # @type: float [0 :-> 1] # @default: 0.0 - ks_a = 0.0 + ks_a = 0.0 # @inferred: # - coord @@ -139,7 +139,7 @@ # @note: In GR, the horizon boundary is set automatically (only specify bc # @ rmax): [["MATCH"]] # @example: [["CUSTOM", "MATCH"]] (for 2D spherical `[[rmin, rmax]]`) - fields = [["PERIODIC"]] + fields = [["PERIODIC"]] # Boundary conditions for particles # @required # @type: array> [size 1 :->: 3] @@ -187,18 +187,18 @@ temperature = 0.0 # Peak number density of the atmosphere at base in units of `n0` # @type: float - density = 0.0 + density = 0.0 # Pressure scale-height in physical units # @type: float - height = 1.0 + height = 1.0 # Species indices of particles that populate the atmosphere # @type: array [size 2] - species = [1, 1] + species = [1, 1] # Distance from the edge to which the gravity is imposed in physical units # @type: float # @default: 0.0 # @note: 0.0 means no limit - ds = 0.0 + ds = 0.0 # @inferred: # - g @@ -213,7 +213,7 @@ # Fiducial larmor radius # @required # @type: float [> 0.0] - larmor0 = 1.0 + larmor0 = 1.0 # Fiducial plasma skin depth # @required # @type: float [> 0.0] @@ -287,7 +287,7 @@ # c^2` in fiducial magnetic field `B0` # @type: float [> 1.0] # @default: 10.0 - gamma_qed = 10.0 + gamma_qed = 10.0 # Minimum photon energy for synchrotron emission (units of `m0 c^2`) # @type: float [> 0.0] # @default: 1e-3 @@ -295,11 +295,11 @@ # Weights for the emitted synchrotron photons # @type: float [> 0.0] # @default: 1.0 - photon_weight = 1.0 + photon_weight = 1.0 # Index of species for the emitted photon # @required # @type: ushort [> 0] - photon_species = 1 + photon_species = 1 # @inferred: # - nominal_probability @@ -324,7 +324,7 @@ # `m0 c^2` in fiducial magnetic field `B0` # @type: float [> 1.0] # @default: 10.0 - gamma_qed = 10.0 + gamma_qed = 10.0 # Minimum photon energy for inverse Compton emission (units of `m0 c^2`) # @type: float [> 0.0] # @default: 1e-3 @@ -332,11 +332,11 @@ # Weights for the emitted inverse Compton photons # @type: float [> 0.0] # @default: 1.0 - photon_weight = 1.0 + photon_weight = 1.0 # Index of species for the emitted photon # @required # @type: ushort [> 0] - photon_species = 1 + photon_species = 1 # @inferred: # - nominal_probability @@ -367,7 +367,7 @@ # @type: float [0.0 -> 1.0] # @default: 0.95 # @note: CFL number determines the timestep duration - CFL = 0.95 + CFL = 0.95 # Correction factor for the speed of light used in field solver # @type: float # @default: 1.0 @@ -385,12 +385,12 @@ # Enable the current deposition # @type: bool # @default: true - enable = true + enable = true # Tiled-deposit work-group (team) size # @type: uint [>= 0] # @default: 0 # @deprecated: removed in 1.6+, use `tiled_deposit_team_size` instead - team_policy_team_size = 0 + team_policy_team_size = 0 # Tiled-deposit work-group (team) size # @type: uint [>= 0] # @default: 0 @@ -412,7 +412,7 @@ # Stepsize for numerical differentiation in GR pusher # @type: float [> 0.0] # @default: 1e-6 - pusher_eps = 1e-6 + pusher_eps = 1e-6 # Number of iterations for the Newton-Raphson method in GR pusher # @type: ushort [> 0] # @default: 10 @@ -428,7 +428,7 @@ # @type: float # @default: 0.0 # @note: When `larmor_max` == 0, the limit is disabled - larmor_max = 0.0 + larmor_max = 0.0 # Stencil coefficients for the field solver [notation as in Blinne+ (2018)] # @note: Standard Yee solver: `delta_i = beta_ij = 0.0` @@ -436,7 +436,7 @@ # Enable the fieldsolver # @type: bool # @default: true - enable = true + enable = true # delta_x coefficient (for `F_{i +/- 3/2, j, k}`) # @type: float # @default: 0.0 @@ -487,16 +487,16 @@ # Fiducial number of particles per cell # @required # @type: float [> 0.0] - ppc0 = 1.0 + ppc0 = 1.0 # Toggle for using particle weights # @type: bool # @default: false - use_weights = false + use_weights = false # Timesteps between particle re-sorting by tags (removing dead particles) # @type: uint # @default: 100 # @note: Set to 0 to disable re-sorting - clear_interval = 100 + clear_interval = 100 # Timesteps between spatial sorting of particles (for better cache # performance) # @type: uint @@ -517,41 +517,41 @@ # @default: "s" # @note: `` is the index of the species in the list starting from 1 # @example: "e-" - label = "s" + label = "s" # Mass of the species (in units of fiducial mass) # @required # @type: float [>= 0.0] - mass = 0.0 + mass = 0.0 # Charge of the species (in units of fiducial charge) # @required # @type: float - charge = 0.0 + charge = 0.0 # Maximum number of particles per task # @required # @type: uint [> 0] # @note: Read as a float, so exponential notation is fine (e.g. `1e8`) - maxnpart = 1.0 + maxnpart = 1.0 # Pusher algorithm for the species # @type: string # @default: "Boris" [massive]; "Photon" [massless] # @enum: "Boris", "Vay", "Boris,GCA", "Vay,GCA", "Photon", "None" - pusher = "Boris" + pusher = "Boris" # Number of additional real-valued variables (payloads) for each particle of # the given species # @type: ushort # @default: 0 - n_payloads_real = 0 + n_payloads_real = 0 # Number of additional integer-valued variables (payloads) for each particle # of the given species # @type: ushort # @default: 0 # @note: If tracking is enabled, one or two extra integer payloads are # reserved (depending on whether MPI is enabled) - n_payloads_int = 0 + n_payloads_int = 0 # Enable tracking of particles using indices for the given species # @type: bool # @default: false - tracking = false + tracking = false # Radiation reaction to use for the species # @type: string # @default: "None" @@ -559,7 +559,7 @@ # @note: Can also be coma-separated combination, e.g., # "Synchrotron,Compton" # @note: Relevant radiation.drag parameters should also be provided - radiative_drag = "None" + radiative_drag = "None" # Particle emission policy for the species # @type: string # @default: "None" @@ -567,7 +567,7 @@ # @note: Only one emission mechanism allowed # @note: Appropriate radiation drag flag will be applied automatically # (unless explicitly set to "None") - emission = "None" + emission = "None" # Timesteps between spatial sorting of particles for given species # @type: uint # @default: 0 @@ -580,7 +580,7 @@ # @default: 100 # @note: Set to 0 to disable re-sorting # @note: Overrides `particles.clear_interval` for the given species - clear_interval = 100 + clear_interval = 100 # Parameters for specific problem generators and setups # @note: Free-form: keys are defined by the problem generator, so nothing here @@ -593,12 +593,12 @@ # @type: string # @default: "bpfile" # @enum: "disabled", "hdf5", "BPFile" - format = "bpfile" + format = "bpfile" # Number of timesteps between all outputs # @type: uint [> 0] # @default: 100 # @note: Value is overriden by output intervals for specific outputs - interval = 100 + interval = 100 # Physical (code) time interval between all outputs # @type: float # @default: -1.0 @@ -612,7 +612,7 @@ # Toggle for the field output # @type: bool # @default: true - enable = true + enable = true # Field quantities to output # @type: array # @default: [] @@ -624,17 +624,17 @@ # "3" # @note: By default, we accumulate moments from all massive species, one # can specify only specific species: `Ttt_1_2`, `Rho_1`, `Rho_3_4` - quantities = [] + quantities = [] # Custom (user-defined) field quantities # @type: array # @default: [] - custom = [] + custom = [] # Number of timesteps between field outputs # @type: uint # @default: 0 # @note: When `!= 0`, overrides `output.interval` # @note: When `== 0`, `output.interval` is used - interval = 0 + interval = 0 # Physical (code) time interval between field outputs # @type: float # @default: -1.0 @@ -646,14 +646,14 @@ # @default: [1, 1, 1] # @note: The output is downsampled by the given factors in each direction # @note: If a scalar is given, it is applied to all directions - downsampling = [1, 1, 1] + downsampling = [1, 1, 1] # Smoothing of the output moments [output.fields.smoothing] # Smoothing order for the output of moments ("Rho", "Charge", "T", ...) # @type: ushort # @default: 0 - order = 0 + order = 0 # Smoothing algorithm # @type: string # @default: "spline" @@ -669,22 +669,22 @@ # Toggle for the particles output # @type: bool # @default: true - enable = true + enable = true # Particle species indices to output # @type: array # @default: [] # @note: If empty, all species are output - species = [] + species = [] # Stride for the output of particles # @type: uint [>= 1] # @default: 100 - stride = 100 + stride = 100 # Number of timesteps between particle outputs # @type: uint # @default: 0 # @note: When `!= 0`, overrides `output.interval` # @note: When `== 0`, `output.interval` is used - interval = 0 + interval = 0 # Physical (code) time interval between particle outputs # @type: float # @default: -1.0 @@ -697,28 +697,28 @@ # Toggle for the spectra output # @type: bool # @default: true - enable = true + enable = true # Minimum energy for the spectra output # @type: float # @default: 1e-3 - e_min = 1e-3 + e_min = 1e-3 # Maximum energy for the spectra output # @type: float # @default: 1e3 - e_max = 1e3 + e_max = 1e3 # Whether to use logarithmic bins for energy # @type: bool # @default: true - log_bins = true + log_bins = true # Number of energy bins for the spectra output # @type: uint [> 0] # @default: 200 # @deprecated: removed in 1.6+, use `num_energy_bins` instead - n_bins = 200 + n_bins = 200 # Number of energy bins for the spectra output # @type: uint [> 0] # @default: 200 - num_energy_bins = 200 + num_energy_bins = 200 # Number of spatial bins for the spectra output # @type: array [size 1 :->: 3] # @default: [1, 1, 1] @@ -728,20 +728,20 @@ # @default: 0 # @note: When `!= 0`, overrides `output.interval` # @note: When `== 0`, `output.interval` is used - interval = 0 + interval = 0 # Physical (code) time interval between spectra outputs # @type: float # @default: -1.0 # @note: When `< 0`, the output is controlled by `interval` # @note: When specified, overrides `output.interval_time` - interval_time = -1.0 + interval_time = -1.0 # Debug output parameters [output.debug] # Output fields "as is" without conversions # @type: bool # @default: false - as_is = false + as_is = false # Output fields with values in ghost cells # @type: bool # @default: false @@ -752,12 +752,12 @@ # Toggle for the stats output # @type: bool # @default: true - enable = true + enable = true # Number of timesteps between stat outputs # @type: uint [> 0] # @default: 100 # @note: Overriden if `output.stats.interval_time != -1` - interval = 100 + interval = 100 # Physical (code) time interval between stat outputs # @type: float # @default: -1.0 @@ -770,18 +770,18 @@ # "Tij" # @note: For particle moments, ... # @note: ... same notation is used as for `output.fields.quantities` - quantities = ["B^2", "E^2", "ExB", "Rho", "T00"] + quantities = ["B^2", "E^2", "ExB", "Rho", "T00"] # Custom (user-defined) stats # @type: array # @default: [] - custom = [] + custom = [] # Checkpointing parameters [checkpoint] # Number of timesteps between checkpoints # @type: uint [> 0] # @default: 1000 - interval = 1000 + interval = 1000 # Physical (code) time interval between checkpoints # @type: float [> 0] # @default: -1.0 @@ -792,23 +792,23 @@ # @default: 2 # @note: 0 = disable checkpointing # @note: -1 = keep all checkpoints - keep = 2 + keep = 2 # Write a checkpoint once after a fixed walltime # @type: string # @default: "00:00:00" # @note: The format is "HH:MM:SS" # @note: Empty string or "00:00:00" disables this functionality # @note: Writing checkpoint at walltime does not stop the simulation - walltime = "00:00:00" + walltime = "00:00:00" # Parent directory to write checkpoints to # @type: string # @default: `.ckpt` # @note: The directory is created if it does not exist - write_path = "" + write_path = "" # Parent directory to use when resuming from a checkpoint # @type: string # @default: inherit `write_path` - read_path = "" + read_path = "" # @inferred: # - is_resuming @@ -836,12 +836,12 @@ # @type: uint # @default: 4294967296 # @note: Lower this on memory-constrained nodes; matches ADIOS2's default - max_shm_size = 4294967296 + max_shm_size = 4294967296 # Internal serialization buffer chunk size, in bytes (BP5 BufferChunkSize) # @type: uint # @default: 16777216 # @note: Scales with per-rank output volume; matches ADIOS2's default - buffer_chunk_size = 16777216 + buffer_chunk_size = 16777216 # In-situ renderer. Renders scalar fields on the GPU and writes PNG images # directly to `/renders/` each cadence -- no field data is written to @@ -861,45 +861,45 @@ # Toggle for the on-the-fly renderer # @type: bool # @default: false - enable = false + enable = false # Number of timesteps between renders # @type: uint # @default: 0 # @note: When `!= 0`, overrides `output.interval` # @note: When `== 0`, `interval_time` (or `output.interval`) is used - interval = 0 + interval = 0 # Physical (code) time interval between renders # @type: float # @default: -1.0 # @note: When `< 0`, the output is controlled by `interval` - interval_time = -1.0 + interval_time = -1.0 # Image width in pixels (the rendered region; the PNG is wider if a colorbar # margin is added, see `colorbar_outside`) # @type: int [> 0] # @default: 1024 - width = 1024 + width = 1024 # Image height in pixels # @type: int [> 0] # @default: 1024 - height = 1024 + height = 1024 # Convenience: force a square frame (sets width == height == resolution), the # natural shape for a dome master. Overrides `width`/`height` when > 0. # @type: int [> 0] # @default: 0 (use width/height) - resolution = 0 + resolution = 0 # Number of entries in the color/opacity lookup table # @type: int [> 1] # @default: 256 - n_lut = 256 + n_lut = 256 # Opaque background RGB (each channel 0..1) shown through # transparent/low-opacity pixels; also fills the colorbar margin # @type: array [size 3] # @default: [0.0, 0.0, 0.0] - background = [0.0, 0.0, 0.0] + background = [0.0, 0.0, 0.0] # Draw a colorbar (gradient + value ticks + label) on each PNG # @type: bool # @default: true - colorbar = true + colorbar = true # Draw the colorbar in an added right margin (the PNG becomes wider by a fixed # strip) instead of overlaying it on the rendered volume # @type: bool @@ -910,13 +910,13 @@ # Cartesian or 3D rendering. # @type: bool # @default: true - mirror = true + mirror = true # Draw the current simulation time as a label ("T = ", fixed to 2 # decimals) in the upper-right corner of the render region, in a contrasting # color, vertically centered between the frame top and the colorbar. # @type: bool # @default: false - time_label = false + time_label = false # Draw a spine (frame) + axis ticks + labels around the rendered region. The # PNG gains left/bottom margins (background-filled) for the tick labels and # axis names, so they never overlap the data. @@ -928,23 +928,23 @@ # spine); # 3D = the global box projected to a wireframe with ticks on the # three silhouette edges (x bottom, y & z on the left) - axes = false + axes = false # Axis names. 3D uses all three; the 2D slice uses the first two. When unset, # the 2D slice defaults to "x","y" (Cartesian) or "X","Z" (spherical). # @type: array [size <= 3] # @default: ["x", "y", "z"] - axis_labels = ["x", "y", "z"] + axis_labels = ["x", "y", "z"] # Target number of ticks per axis (actual count is rounded to nice values) # @type: int [>= 2] # @default: 5 - axis_ticks = 5 + axis_ticks = 5 # 3D only: target width (pixels) of the box wireframe "spine". The spine is # drawn inside the ray-march (opaque, depth-occluded by the volume); its width # is floored by the ray step, so for a crisper thin line raise `samples` as # well. # @type: float [> 0.0] # @default: 2.0 - spine_width = 2.0 + spine_width = 2.0 # Limit the render region to axis-aligned box in physical/world coordinates. # Left unset it spans the full domain. Clamped to the box. @@ -976,13 +976,13 @@ # @default: 400 # @note: The world-space step is `box_diagonal / samples` unless # `step_size` is set. Higher = better quality, slower. - samples = 400 + samples = 400 # Fixed world-space step between ray samples # @type: float [>= 0.0] # @default: 0.0 # @note: 0 derives the step from `samples`. The step is identical on all # ranks, which is what makes the multi-domain composite seamless. - step_size = 0.0 + step_size = 0.0 # Stop marching a ray once its accumulated opacity reaches this value # @type: float [0.0 -> 1.0] # @default: 0.99 @@ -997,7 +997,7 @@ # @type: array [size 2 or 3] # @default: [] (static view) # @example: velocity = [0.9, 0.0] # pan along +x1 at 0.9 c - velocity = [] + velocity = [] # Sim time at which the view starts moving (static before it, e.g. to let an # initial ramp-up finish) # @type: float @@ -1020,29 +1020,29 @@ # (unlike ortho/perspective, which need the eye outside the box). # Set a square frame (`resolution`, or width == height). `forward` # is the dome ZENITH (screen-up defaults to +y for a +z zenith). - mode = "orthographic" + mode = "orthographic" # Camera (eye) position in world (physical) coordinates # @type: array [size 3] # @default: box center pushed back ~1.7 box-diagonals along (1, 1, 1); # for `mode = "dome"`, the domain center (interior eye) - position = [0.0, 0.0, 0.0] + position = [0.0, 0.0, 0.0] # Point the camera looks at, in world coordinates (the dome ZENITH target) # @type: array [size 3] # @default: box center; for `mode = "dome"`, the zenith defaults to +z - look_at = [0.0, 0.0, 0.0] + look_at = [0.0, 0.0, 0.0] # Camera up vector (dome: the disk's screen-up) # @type: array [size 3] # @default: [0.0, 0.0, 1.0]; for `mode = "dome"`, [0.0, 1.0, 0.0] - up = [0.0, 0.0, 1.0] + up = [0.0, 0.0, 1.0] # Vertical field of view in degrees (perspective only) # @type: float [> 0.0] # @default: 35.0 - fov = 35.0 + fov = 35.0 # Full dome field of view in degrees (dome mode only): the image rim is at # dome_fov/2 from the zenith (180 = a full hemisphere down to the horizon). # @type: float [> 0.0, <= 360.0] # @default: 180.0 - dome_fov = 180.0 + dome_fov = 180.0 # Dome far-clip radius in world units (dome mode only): each ray stops this # far from the eye, so the sampled region is a half-ball (hemisphere) of # this radius rather than the whole box -> uniform path length and no box @@ -1052,7 +1052,7 @@ # @default: the largest sphere centered in the box (half the shortest # side), so it touches the face centers and never a corner # @note: 0 disables the clip (rays march to the box boundary) - dome_radius = 0.0 + dome_radius = 0.0 # Vertical extent of the view in world units (orthographic only) # @type: float [> 0.0] # @default: the global box diagonal (the whole box fits from any angle) @@ -1077,21 +1077,21 @@ # Build the fisheye dome master instead of the plain slice # @type: bool # @default: false - enable = false + enable = false # (Cartesian only) Full dome field of view in degrees (image radius maps # linearly to the dome zenith angle: the rim is at fov/2) # @type: float [> 0.0, <= 180.0] # @default: 180.0 # a full hemisphere - fov = 180.0 + fov = 180.0 # (Cartesian only) World radius of the circular cutout mapped onto the dome # @type: float [> 0.0] # @default: half the shorter domain side (the largest centered disk that # fits inside the box) - radius = 1.0 + radius = 1.0 # (Cartesian only) World-space center of the cutout # @type: array [size 2] # @default: the domain center - center = [0.0, 0.0] + center = [0.0, 0.0] # (Cartesian only) How the dome zenith angle maps to a world radius on the # flat slice # @type: string @@ -1121,38 +1121,38 @@ # @default: false # @note: implied true if any scene sets `fieldlines = true` or uses `field # = "fieldlines"` - enable = false + enable = false # Vector field to trace # @type: string # @default: "B" # @enum: "B", "E", "J" - field = "B" + field = "B" # Field coarsening factor (simulation cells per coarse cell, per axis) # @type: int [1..16] # @default: 4 # @note: larger = smoother "morphology" lines + cheaper replication (the # coarse field is ~ N_cells / bin^D floats/rank; D = sim dimension) - bin = 4 + bin = 4 # (3D tubes) Seed-lattice spacing in screen pixels (sets line density) # @type: float [> 0] # @default: 8 # @note: capped by `seed_max`; if seed_px asks for more seeds than that, # the spacing grows to fit and seed_px no longer governs - seed_px = 8 + seed_px = 8 # (3D tubes) Hard cap on the seed count (lattice is n^3, 2 lines per seed) # @type: int [> 0] # @default: 4096 # @note: lower this for fewer / more widely spaced lines - seed_max = 4096 + seed_max = 4096 # (2D contours) Number of evenly-spaced flux-function contour levels # @type: int [> 0] # @default: 16 # @note: evenly-spaced psi levels => line density tracks |B| automatically - levels = 16 + levels = 16 # Tube radius (3D) / contour line width (2D), in screen pixels # @type: float [> 0] # @default: 2 - tube_px = 2 + tube_px = 2 # Colormap for the field lines (mapped by |B| along each line) # @type: string # @default: "inferno" @@ -1161,38 +1161,38 @@ # "dusk", "cosmic", "freeze", "apple", "gothic", "sunburst", # "voltage", "ocean", "fusion", "prinsenvlag" (an optional "cmr." # prefix is ok) - colormap = "inferno" + colormap = "inferno" # Monochrome override: draw the lines in a single [r,g,b] color (each 0..1) # instead of the |B| colormap -- reads well as an overlay on another volume # @type: array [size 3] # @default: [] (empty => color by |B|) # @example: [1.0, 1.0, 1.0] # white field lines - color = [] + color = [] # Map the tube color range logarithmically # @type: bool # @default: false # @note: requires min > 0 - log = false + log = false # Tube color range: lower bound on |field| # @type: float # @default: 0.0 # @note: when min >= max, the range is auto-set from |field| along the # lines - min = 0.0 + min = 0.0 # Tube color range: upper bound on |field| # @type: float # @default: 0.0 # @note: when min >= max, the range is auto-set from |field| along the # lines - max = 0.0 + max = 0.0 # (3D tubes) RK4 integration step as a fraction of one coarse cell # @type: float [> 0] # @default: 0.5 - step_frac = 0.5 + step_frac = 0.5 # (3D tubes) Per-direction integration-step cap # @type: int [> 0] # @default: 4000 - max_steps = 4000 + max_steps = 4000 # (3D tubes) Maximum line length, in global box diagonals (per direction) # @type: float [> 0] # @default: 3.0 @@ -1226,28 +1226,28 @@ # diverging colormap ("cool2warm") to center zero # @note: "fieldlines" renders the magnetic field-line tubes on their own # (no scalar volume sampled); see [render.fieldlines] below - field = "" + field = "" # PNG filename prefix; files are `.png` # @type: string # @default: "_" - prefix = "_" + prefix = "_" # Colorbar title # @type: string # @default: `field` - label = "" + label = "" # Lower bound of the value range mapped onto the colormap/opacity # @type: float # @default: 0.0 - min = 0.0 + min = 0.0 # Upper bound of the value range # @type: float # @default: 1.0 - max = 1.0 + max = 1.0 # Map the value range logarithmically # @type: bool # @default: false # @note: Requires min > 0 and max > 0 - log = false + log = false # Colormap name # @type: string # @default: "viridis" @@ -1256,14 +1256,14 @@ # "dusk", "cosmic", "freeze", "apple", "gothic", "sunburst", # "voltage", "ocean", "fusion", "prinsenvlag" (an optional "cmr." # prefix is ok) - colormap = "viridis" + colormap = "viridis" # Opacity transfer function: [position, opacity] control points, both in [0, # 1], piecewise-linear in the normalized value # @type: array> # @default: linear ramp (opacity = normalized value) # @note: Keep the low end near 0 so empty regions stay transparent # @example: [[0.0, 0.0], [0.3, 0.1], [1.0, 0.7]] - alpha = [] + alpha = [] # Explicit value(s) to label on the colorbar # @type: array # @default: 5 evenly-spaced ticks between min and max @@ -1275,14 +1275,14 @@ # @default: false # @note: requires the [render.fieldlines] enabled (3D only). A scene with # field = "fieldlines" instead renders them alone. - fieldlines = false + fieldlines = false # Diagnostic logging parameters [diagnostics] # Number of timesteps between diagnostic logs # @type: int [> 0] # @default: 1 - interval = 1 + interval = 1 # Blocking timers between successive algorithms # @type: bool # @default: false @@ -1290,11 +1290,11 @@ # Enable colored stdout # @type: bool # @default: true - colored_stdout = true + colored_stdout = true # Specify the log level # @type: string # @default: "VERBOSE" # @enum: "VERBOSE", "WARNING", "ERROR" # @note: "VERBOSE" prints all messages, "WARNING" prints only warnings and # errors, "ERROR" prints only errors - log_level = "VERBOSE" + log_level = "VERBOSE" diff --git a/pgens/magnetosphere/magnetosphere.toml b/pgens/magnetosphere/magnetosphere.toml index 4a8eca87e..4b564a295 100644 --- a/pgens/magnetosphere/magnetosphere.toml +++ b/pgens/magnetosphere/magnetosphere.toml @@ -59,7 +59,7 @@ [setup] Bsurf = 1.0 - field_geometry = "dipole" # can be "dipole" or "monopole" (default: dipole) + field_geometry = "dipole" # can be "dipole" or "monopole" (default: dipole) period = 60.0 [output] diff --git a/pgens/shock/shock.toml b/pgens/shock/shock.toml index 8d3270a74..267346580 100644 --- a/pgens/shock/shock.toml +++ b/pgens/shock/shock.toml @@ -40,15 +40,15 @@ maxnpart = 2e7 [setup] - drift_ux = 0.15 # speed towards the wall [c] - temperature = 0.168 # temperature of maxwell distribution [kB T / (m_i c^2)] - temperature_ratio = 1.0 # temperature ratio of electrons to protons - Bmag = 1.0 # magnetic field strength as fraction of magnetisation - Btheta = 63.0 # magnetic field angle in the plane - Bphi = 0.0 # magnetic field angle out of plane - filling_fraction = 0.99 # fraction of the shock piston filled with plasma - injector_velocity = 0.0 # speed of injector [c] - injection_start = 0.0 # start time of moving injector + drift_ux = 0.15 # speed towards the wall [c] + temperature = 0.168 # temperature of maxwell distribution [kB T / (m_i c^2)] + temperature_ratio = 1.0 # temperature ratio of electrons to protons + Bmag = 1.0 # magnetic field strength as fraction of magnetisation + Btheta = 63.0 # magnetic field angle in the plane + Bphi = 0.0 # magnetic field angle out of plane + filling_fraction = 0.99 # fraction of the shock piston filled with plasma + injector_velocity = 0.0 # speed of injector [c] + injection_start = 0.0 # start time of moving injector injection_frequency = 100 [output] diff --git a/pgens/streaming/twostream.toml b/pgens/streaming/twostream.toml index 1b2334777..3df077c05 100644 --- a/pgens/streaming/twostream.toml +++ b/pgens/streaming/twostream.toml @@ -57,13 +57,13 @@ # Drift 4-velocities for each species in all 3 directions # @type: array of floats (length = nspec) # @default: [ 0.0, ... ] - drifts_in_x = [0.1, 0.0, -0.1, 0.0] - drifts_in_y = [0.0, 0.0, 0.0, 0.0] - drifts_in_z = [0.0, 0.0, 0.0, 0.0] + drifts_in_x = [0.1, 0.0, -0.1, 0.0] + drifts_in_y = [0.0, 0.0, 0.0, 0.0] + drifts_in_z = [0.0, 0.0, 0.0, 0.0] # Pair-wise species densities in units of n0 # @type: array of floats (length = nspec/2) # @default: [ 2 / nspec, ... ] - densities = [0.5, 0.5] + densities = [0.5, 0.5] # Species temperatures in units of m0 (c^2) # @type: array of floats (length = nspec) # @default: [ 0.0, ... ] diff --git a/pgens/streaming/weibel.toml b/pgens/streaming/weibel.toml index 0d1f15bca..1e3b5911d 100644 --- a/pgens/streaming/weibel.toml +++ b/pgens/streaming/weibel.toml @@ -55,13 +55,13 @@ # Drift 4-velocities for each species in all 3 directions # @type: array of floats (length = nspec) # @default: [ 0.0, ... ] - drifts_in_x = [0.0, 0.0, 0.0, 0.0] - drifts_in_y = [0.0, 0.0, 0.0, 0.0] - drifts_in_z = [0.3, 0.3, -0.3, -0.3] + drifts_in_x = [0.0, 0.0, 0.0, 0.0] + drifts_in_y = [0.0, 0.0, 0.0, 0.0] + drifts_in_z = [0.3, 0.3, -0.3, -0.3] # Pair-wise species densities in units of n0 # @type: array of floats (length = nspec/2) # @default: [ 2 / nspec, ... ] - densities = [0.5, 0.5] + densities = [0.5, 0.5] # Species temperatures in units of m0 (c^2) # @type: array of floats (length = nspec) # @default: [ 0.0, ... ] diff --git a/pgens/turbulence/turbulence.toml b/pgens/turbulence/turbulence.toml index a79bd07ad..25dc24bff 100644 --- a/pgens/turbulence/turbulence.toml +++ b/pgens/turbulence/turbulence.toml @@ -45,7 +45,6 @@ omega_0 = 0.0156 gamma_0 = 0.0078 - [output] format = "BPFile" interval_time = 12.0 @@ -58,6 +57,7 @@ [output.spectra] enable = false + [output.stats] enable = false diff --git a/pgens/wald/wald.toml b/pgens/wald/wald.toml index b062c9ca1..8ef46dc46 100644 --- a/pgens/wald/wald.toml +++ b/pgens/wald/wald.toml @@ -37,7 +37,7 @@ ppc0 = 2.0 [setup] - init_field = "wald" # or "vertical" + init_field = "wald" # or "vertical" [output] format = "BPFile" diff --git a/scripts/dependencies.py b/scripts/dependencies.py index 54adf55b2..d00c4d109 100755 --- a/scripts/dependencies.py +++ b/scripts/dependencies.py @@ -5,12 +5,8 @@ import curses import json import os +from collections.abc import Callable from dataclasses import dataclass, field -from typing import Callable, List, Optional, Tuple - -# ============================ -# colors: edit these -# ============================ # foreground colors (use curses.COLOR_* or -1 for default) COLOR_TITLE_FG = curses.COLOR_BLUE @@ -58,11 +54,11 @@ class Settings: # options kokkos_backend: str = "cpu" kokkos_arch: str = "" - extra_kokkos_flags: List[str] = field(default_factory=list) + extra_kokkos_flags: list[str] = field(default_factory=list) adios2_mpi: str = "non-mpi" - extra_adios2_flags: List[str] = field(default_factory=list) + extra_adios2_flags: list[str] = field(default_factory=list) - module_loads: List[str] = field(default_factory=list) + module_loads: list[str] = field(default_factory=list) def from_json(self, json_str: str) -> None: data = json.loads(json_str) @@ -160,7 +156,7 @@ def InstallKokkosScriptModfile(settings: Settings) -> tuple[str, str]: -D CMAKE_CXX_STANDARD={cxx_standard} \\ -D CMAKE_CXX_EXTENSIONS=OFF \\ -D CMAKE_POSITION_INDEPENDENT_CODE=TRUE \\ - -D Kokkos_ARCH_{arch}=ON {f'-D Kokkos_ENABLE_{backend.upper()}=ON' if backend != 'cpu' else ''} \\ + -D Kokkos_ARCH_{arch}=ON {f"-D Kokkos_ENABLE_{backend.upper()}=ON" if backend != "cpu" else ""} \\ -D CMAKE_INSTALL_PREFIX={install_path} {extra_flags} && \\ cmake --build build -j $(nproc) && \\ cmake --install build""" @@ -185,7 +181,7 @@ def InstallKokkosScriptModfile(settings: Settings) -> tuple[str, str]: setenv Kokkos_DIR $basedir setenv Kokkos_ARCH_{arch} ON -{f'setenv Kokkos_ENABLE_{backend.upper()} ON' if backend != 'cpu' else ''}""" +{f"setenv Kokkos_ENABLE_{backend.upper()} ON" if backend != "cpu" else ""}""" return (unindent(script), unindent(modfile)) @@ -414,17 +410,16 @@ def on_install_confirmed(settings: Settings) -> None: settings_json = os.path.join(settings.install_prefix, "settings.json") with open(settings_json, "w") as f: f.write(settings.to_json()) - return @dataclass class MenuItem: label: str hint: str = "" - right: Optional[Callable[[], str]] = None - on_enter: Optional[Callable[[], None]] = None - on_space: Optional[Callable[[], None]] = None - disabled: Optional[Callable[[], bool]] = None + right: Callable[[], str] | None = None + on_enter: Callable[[], None] | None = None + on_space: Callable[[], None] | None = None + disabled: Callable[[], bool] | None = None class TuiExitInstall(Exception): @@ -441,7 +436,7 @@ def __init__(self, stdscr): self.s.from_json(json.dumps(data)) self.state = "mainmenu" - self.stack: List[Tuple[str, int]] = [] + self.stack: list[tuple[str, int]] = [] self.selected = 0 self.scroll = 0 self.message = "use arrows or j/k" @@ -519,7 +514,7 @@ def hline(self, y: int) -> None: except curses.error: pass - def draw_keybar(self, y: int, x: int, pairs: List[Tuple[str, str]]) -> None: + def draw_keybar(self, y: int, x: int, pairs: list[tuple[str, str]]) -> None: cur_x = x for key, action in pairs: self.add(y, cur_x, key, self.cp(PAIR_KEY) | curses.A_BOLD) @@ -548,7 +543,7 @@ def breadcrumb(self) -> str: return f"mainmenu › cluster-specific › {self.s.cluster}" return "mainmenu" - def draw_menu(self, title: str, prompt: str, items: List[MenuItem]) -> None: + def draw_menu(self, title: str, prompt: str, items: list[MenuItem]) -> None: self.stdscr.erase() h, w = self.stdscr.getmaxyx() @@ -589,8 +584,7 @@ def draw_menu(self, title: str, prompt: str, items: List[MenuItem]) -> None: else: self.selected = max(0, min(self.selected, n - 1)) - if self.selected < self.scroll: - self.scroll = self.selected + self.scroll = min(self.scroll, self.selected) if self.selected >= self.scroll + view_h: self.scroll = self.selected - view_h + 1 self.scroll = max(0, min(self.scroll, max(0, n - view_h))) @@ -639,7 +633,7 @@ def draw_menu(self, title: str, prompt: str, items: List[MenuItem]) -> None: # ----- modals ----- - def input_box(self, title: str, prompt: str, initial: str) -> Optional[str]: + def input_box(self, title: str, prompt: str, initial: str) -> str | None: h, w = self.stdscr.getmaxyx() win_h, win_w = 9, min(86, max(46, w - 6)) top, left = max(0, (h - win_h) // 2), max(0, (w - win_w) // 2) @@ -717,7 +711,7 @@ def confirm_install(self) -> bool: # ----- helpers ----- - def cycle(self, current: str, options: List[str]) -> str: + def cycle(self, current: str, options: list[str]) -> str: if current not in options: return options[0] i = options.index(current) @@ -757,8 +751,7 @@ def module_editor(self) -> None: self.add(list_y, 2, "(empty) press a to add", self.cp(PAIR_HINT)) else: self.mod_sel = max(0, min(self.mod_sel, n - 1)) - if self.mod_sel < self.mod_scroll: - self.mod_scroll = self.mod_sel + self.mod_scroll = min(self.mod_scroll, self.mod_sel) if self.mod_sel >= self.mod_scroll + view_h: self.mod_scroll = self.mod_sel - view_h + 1 self.mod_scroll = max(0, min(self.mod_scroll, max(0, n - view_h))) @@ -846,7 +839,7 @@ def module_editor(self) -> None: # ----- menus ----- - def versions_menu(self) -> Tuple[str, str, List[MenuItem]]: + def versions_menu(self) -> tuple[str, str, list[MenuItem]]: def edit_kokkos(): val = self.input_box( "kokkos version", "enter version/tag:", self.s.kokkos_version @@ -881,7 +874,7 @@ def edit_adios2(): ], ) - def options_menu(self) -> Tuple[str, str, List[MenuItem]]: + def options_menu(self) -> tuple[str, str, list[MenuItem]]: def cycle_kokkos(): self.s.kokkos_backend = self.cycle(self.s.kokkos_backend, KOKKOS_BACKENDS) @@ -910,7 +903,7 @@ def cycle_adios2(): MenuItem( "kokkos arch", "enter to edit (optional)", - right=lambda: (self.s.kokkos_arch.strip() or "-"), + right=lambda: self.s.kokkos_arch.strip() or "-", on_enter=edit_kokkos_arch, disabled=lambda: not self.s.apps.get("Kokkos", False), ), @@ -926,7 +919,7 @@ def cycle_adios2(): ], ) - def menu_main(self) -> Tuple[str, str, List[MenuItem]]: + def menu_main(self) -> tuple[str, str, list[MenuItem]]: return ( "entity deps", "main menu:", @@ -945,7 +938,7 @@ def menu_main(self) -> Tuple[str, str, List[MenuItem]]: ], ) - def menu_custom(self) -> Tuple[str, str, List[MenuItem]]: + def menu_custom(self) -> tuple[str, str, list[MenuItem]]: def toggle_write_modulefiles(): self.s.write_modulefiles = not self.s.write_modulefiles @@ -1058,7 +1051,7 @@ def do_install(): ], ) - def menu_apps(self) -> Tuple[str, str, List[MenuItem]]: + def menu_apps(self) -> tuple[str, str, list[MenuItem]]: def toggle(k: str): self.s.apps[k] = not self.s.apps.get(k, False) @@ -1090,7 +1083,7 @@ def toggle(k: str): ], ) - def menu_cluster(self) -> Tuple[str, str, List[MenuItem]]: + def menu_cluster(self) -> tuple[str, str, list[MenuItem]]: def choose(name: str): print("CALLING:", name) apply_preset(self.s, name) @@ -1106,7 +1099,7 @@ def choose(name: str): + [MenuItem("back", "", on_enter=self.pop)], ) - def get_menu(self) -> Tuple[str, str, List[MenuItem]]: + def get_menu(self) -> tuple[str, str, list[MenuItem]]: if self.state == "mainmenu": return self.menu_main() if self.state == "custom": @@ -1127,7 +1120,7 @@ def get_menu(self) -> Tuple[str, str, List[MenuItem]]: def is_disabled(self, it: MenuItem) -> bool: return bool(it.disabled and it.disabled()) - def move_sel(self, items: List[MenuItem], delta: int) -> None: + def move_sel(self, items: list[MenuItem], delta: int) -> None: if not items: return n = len(items) @@ -1138,7 +1131,7 @@ def move_sel(self, items: List[MenuItem], delta: int) -> None: return self.selected = start - def activate(self, items: List[MenuItem], enter: bool) -> None: + def activate(self, items: list[MenuItem], enter: bool) -> None: if not items: return it = items[self.selected] @@ -1188,10 +1181,7 @@ def run(self) -> None: def _wrapper_capture(stdscr) -> None: app = App(stdscr) - try: - app.run() - except TuiExitInstall: - raise + app.run() if __name__ == "__main__": From 427cfdfd92593f8bba316a3bbca607b5c1041719 Mon Sep 17 00:00:00 2001 From: alcauchy Date: Mon, 28 Sep 2026 15:29:43 -0400 Subject: [PATCH 122/125] Fix slow parallel checkpoint read (COMM_SELF + O(N^2) metadata loop) The restart reader opened the BP file on MPI_COMM_SELF, so every rank independently parsed the full global metadata instead of a single rank-0 read + broadcast. Phase 1 also had every rank loop over all g_ndomains reading each subdomain's extent/ncells via per-element Mode::Sync Gets. Both costs scale ~O(N_ranks^2) and made restart reads dominate wall time at large rank counts (~3 h at 65536 ranks vs a 2-3 min write). - Open the reader on MPI_COMM_WORLD so ADIOS2 reads/broadcasts metadata collectively and can aggregate reads. - In Phase 1, each rank now reads only its own subdomain entry and MPI_Allgathers the ncells/xmin/xmax arrays; global_extent and the reconstruction check are computed from the gathered layout. Non-MPI path unchanged. Verified ~50x faster on Frontier (read 209 s vs ~3 h) in the sibling entity_bh tree. Co-Authored-By: Claude Opus 4.8 --- src/framework/domain/metadomain_chckpt.cpp | 94 +++++++++++++++++++--- 1 file changed, 83 insertions(+), 11 deletions(-) diff --git a/src/framework/domain/metadomain_chckpt.cpp b/src/framework/domain/metadomain_chckpt.cpp index 1116df85e..71ab0180a 100644 --- a/src/framework/domain/metadomain_chckpt.cpp +++ b/src/framework/domain/metadomain_chckpt.cpp @@ -15,6 +15,8 @@ #include "output/utils/writers.h" #if defined(MPI_ENABLED) + #include "arch/mpi_aliases.h" + #include #endif @@ -268,22 +270,76 @@ namespace ntt { #if !defined(MPI_ENABLED) adios2::Engine reader = io.Open(fname, adios2::Mode::Read); #else - adios2::Engine reader = io.Open(fname, adios2::Mode::Read, MPI_COMM_SELF); + adios2::Engine reader = io.Open(fname, adios2::Mode::Read, MPI_COMM_WORLD); #endif reader.BeginStep(); - // Phase 1: read all subdomain metadata to detect size changes + // Phase 1: read the saved subdomain metadata (extent + ncells per domain). std::vector> saved_ncells(g_ndomains, std::vector(M::Dim)); std::vector> saved_extents(g_ndomains); - boundaries_t global_extent; - for (auto d { 0u }; d < M::Dim; ++d) { - global_extent.emplace_back(std::numeric_limits::max(), - std::numeric_limits::lowest()); - } - bool needs_reconstruction = false; +#if defined(MPI_ENABLED) + // Each rank reads only its own entry and all-gathers, instead of every rank + // looping over all g_ndomains. The all-domains loop is an O(g_ndomains^2) + // storm of tiny synchronous reads at large rank counts. + { + std::vector loc_ncells(M::Dim); + std::vector loc_xmin(M::Dim), loc_xmax(M::Dim); + const auto local_off = static_cast(g_mpi_rank); + for (auto d { 0u }; d < M::Dim; ++d) { + out::ReadVariable(io, + reader, + fmt::format("subdomain_x%d_min", d + 1), + loc_xmin[d], + local_off); + out::ReadVariable(io, + reader, + fmt::format("subdomain_x%d_max", d + 1), + loc_xmax[d], + local_off); + out::ReadVariable(io, + reader, + fmt::format("subdomain_nx%d", d + 1), + loc_ncells[d], + local_off); + } + + std::vector all_ncells(g_ndomains * M::Dim); + std::vector all_xmin(g_ndomains * M::Dim); + std::vector all_xmax(g_ndomains * M::Dim); + MPI_Allgather(loc_ncells.data(), + static_cast(M::Dim), + mpi::get_type(), + all_ncells.data(), + static_cast(M::Dim), + mpi::get_type(), + MPI_COMM_WORLD); + MPI_Allgather(loc_xmin.data(), + static_cast(M::Dim), + mpi::get_type(), + all_xmin.data(), + static_cast(M::Dim), + mpi::get_type(), + MPI_COMM_WORLD); + MPI_Allgather(loc_xmax.data(), + static_cast(M::Dim), + mpi::get_type(), + all_xmax.data(), + static_cast(M::Dim), + mpi::get_type(), + MPI_COMM_WORLD); + + for (unsigned int dom_idx { 0 }; dom_idx < g_ndomains; ++dom_idx) { + for (auto d { 0u }; d < M::Dim; ++d) { + saved_ncells[dom_idx][d] = all_ncells[dom_idx * M::Dim + d]; + saved_extents[dom_idx].emplace_back(all_xmin[dom_idx * M::Dim + d], + all_xmax[dom_idx * M::Dim + d]); + } + } + } +#else for (unsigned int dom_idx { 0 }; dom_idx < g_ndomains; ++dom_idx) { for (auto d { 0u }; d < M::Dim; ++d) { real_t x_min, x_max; @@ -298,8 +354,6 @@ namespace ntt { x_max, dom_idx); saved_extents[dom_idx].emplace_back(x_min, x_max); - global_extent[d].first = std::min(global_extent[d].first, x_min); - global_extent[d].second = std::max(global_extent[d].second, x_max); ncells_t nx; out::ReadVariable(io, @@ -308,8 +362,26 @@ namespace ntt { nx, dom_idx); saved_ncells[dom_idx][d] = nx; + } + } +#endif - if (nx != subdomain_ptr(dom_idx)->mesh.n_active()[d]) { + // Reduce the gathered layout into the global extent and detect whether the + // domain decomposition changed since the checkpoint was written. + boundaries_t global_extent; + for (auto d { 0u }; d < M::Dim; ++d) { + global_extent.emplace_back(std::numeric_limits::max(), + std::numeric_limits::lowest()); + } + + bool needs_reconstruction = false; + for (unsigned int dom_idx { 0 }; dom_idx < g_ndomains; ++dom_idx) { + for (auto d { 0u }; d < M::Dim; ++d) { + global_extent[d].first = std::min(global_extent[d].first, + saved_extents[dom_idx][d].first); + global_extent[d].second = std::max(global_extent[d].second, + saved_extents[dom_idx][d].second); + if (saved_ncells[dom_idx][d] != subdomain_ptr(dom_idx)->mesh.n_active()[d]) { needs_reconstruction = true; } } From 7b1249d86a3867ede9fbaa71a49922930bb522c5 Mon Sep 17 00:00:00 2001 From: haykh Date: Mon, 5 Oct 2026 11:39:11 -0400 Subject: [PATCH 123/125] 1.5.0rc merge cleanup --- src/framework/domain/checkpoint/resume.cpp | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/framework/domain/checkpoint/resume.cpp b/src/framework/domain/checkpoint/resume.cpp index 11053c742..2e071810b 100644 --- a/src/framework/domain/checkpoint/resume.cpp +++ b/src/framework/domain/checkpoint/resume.cpp @@ -144,7 +144,7 @@ namespace ntt { { std::vector loc_ncells(M::Dim); std::vector loc_xmin(M::Dim), loc_xmax(M::Dim); - const auto local_off = static_cast(g_mpi_rank); + const auto local_off = static_cast(g_mpi_rank); for (auto d { 0u }; d < M::Dim; ++d) { out::ReadVariable(io, reader, From 3d276bcf858a95a9f7fe996aee2f6a6972a82290 Mon Sep 17 00:00:00 2001 From: haykh Date: Mon, 5 Oct 2026 11:55:12 -0400 Subject: [PATCH 124/125] upd readme --- README.md | 3 +++ 1 file changed, 3 insertions(+) diff --git a/README.md b/README.md index 41ee4c7bc..45ebcfbb8 100644 --- a/README.md +++ b/README.md @@ -12,6 +12,9 @@ Our [detailed documentation](https://entity-toolkit.github.io/) includes everyth ## News +- [Oct 2026]: spectra output now support **spatial binning** [PR #226](https://github.com/entity-toolkit/entity/pull/226) +- [Oct 2026]: native [**on-the-fly plotter**](https://entity-toolkit.github.io/wiki/content/3-advanced/3-rendering/) is now available [PR #219](https://github.com/entity-toolkit/entity/pull/219) +- [Oct 2026]: added [**dynamic load balancing**](https://entity-toolkit.github.io/wiki/content/3-advanced/4-load_balancing/) for multi-node runs [PR #216](https://github.com/entity-toolkit/entity/pull/216) - [May 2026]: our **method paper** for higher-order shape functions and generalized field stencils is [online](https://ui.adsabs.harvard.edu/abs/2026arXiv260515260B/abstract) - [Apr 2026]: user-defined **custom particle update** functionality [PR #198](https://github.com/entity-toolkit/entity/pull/198) - [Apr 2026]: **moving window** [PR #196](https://github.com/entity-toolkit/entity/pull/196) From e60fbb78992f59fd7ece77db8c441ff3b3c06864 Mon Sep 17 00:00:00 2001 From: haykh Date: Tue, 6 Oct 2026 10:53:22 -0400 Subject: [PATCH 125/125] cartesian_sr merged from dev/qed --- tutorials/cartesian_sr/cartesian_sr.py | 70 +++++++++++++++++++ ...al_cartesian_sr.toml => cartesian_sr.toml} | 2 +- 2 files changed, 71 insertions(+), 1 deletion(-) create mode 100644 tutorials/cartesian_sr/cartesian_sr.py rename tutorials/cartesian_sr/{tutorial_cartesian_sr.toml => cartesian_sr.toml} (96%) diff --git a/tutorials/cartesian_sr/cartesian_sr.py b/tutorials/cartesian_sr/cartesian_sr.py new file mode 100644 index 000000000..289d8c4e1 --- /dev/null +++ b/tutorials/cartesian_sr/cartesian_sr.py @@ -0,0 +1,70 @@ +import matplotlib.pyplot as plt +import nt2 +import numpy as np + + +def get_dipole(xs, ys): + xx, yy = np.meshgrid(xs, ys) + rr = np.sqrt(xx**2 + yy**2) + bx = 3 * xx * yy / rr**5 + by = (3 * yy**2 - rr**2) / rr**5 + return bx, by + + +def plot(t, data): + fig = plt.figure(figsize=(6, 5), dpi=150) + gs = fig.add_gridspec(1, 2, width_ratios=[1, 0.05], wspace=0.05) + ax = fig.add_subplot(gs[0, 0]) + ax_cbar = fig.add_subplot(gs[0, 1]) + + d = data.fields.sel(t=t, method="nearest") + (d.N_1 + d.N_2).plot(ax=ax, vmin=0, vmax=20, cmap="inferno", add_colorbar=False) + cbar_ticks = np.linspace(0, 20, 50) + ax_cbar.pcolormesh( + [0, 1], + cbar_ticks, + np.array([cbar_ticks] * 2).T, + cmap="inferno", + vmin=0, + vmax=20, + rasterized=True, + ) + ax_cbar.yaxis.tick_right() + ax_cbar.yaxis.set_label_position("right") + ax_cbar.set(xticks=[], ylabel=r"$n_\pm$") + for spine in ax_cbar.spines.values(): + spine.set_visible(False) + + ys = np.linspace(-2.9, 2.9, 50) + xs = -0.5 * np.ones_like(ys) + + bx, by = get_dipole(data.fields.x, data.fields.y) + ax.streamplot( + data.fields.x.values, + data.fields.y.values, + d.Bx.values + bx, + d.By.values + by, + color="#ffffff90", + linewidth=0.25, + zorder=10, + arrowstyle="->", + arrowsize=0.5, + density=30, + start_points=np.array([xs, ys]).T, + ) + ax.add_artist( + plt.Circle((0, 0), data.attrs["setup.r_plummet"], color="C0", zorder=20) + ) + + ax.set( + xlim=(-4, 2), + ylim=(-3, 3), + aspect=1, + xlabel="$x$", + ylabel="$y$", + title=f"$t = {t:.2f}$", + ) + + +data = nt2.Data("cartesian_sr") +data.makeMovie() diff --git a/tutorials/cartesian_sr/tutorial_cartesian_sr.toml b/tutorials/cartesian_sr/cartesian_sr.toml similarity index 96% rename from tutorials/cartesian_sr/tutorial_cartesian_sr.toml rename to tutorials/cartesian_sr/cartesian_sr.toml index 8a85767b2..e9b4231c7 100644 --- a/tutorials/cartesian_sr/tutorial_cartesian_sr.toml +++ b/tutorials/cartesian_sr/cartesian_sr.toml @@ -1,5 +1,5 @@ [simulation] - name = "tutorial_cartesian_sr" + name = "cartesian_sr" engine = "srpic" runtime = 20.0