diff --git a/.taplo.toml b/.taplo.toml deleted file mode 100644 index 423a47594..000000000 --- a/.taplo.toml +++ /dev/null @@ -1,6 +0,0 @@ -[formatting] - align_entries = true - indent_tables = true - indent_entries = true - trailing_newline = true - align_comments = true diff --git a/.tombi.toml b/.tombi.toml new file mode 100644 index 000000000..ea7e8c58b --- /dev/null +++ b/.tombi.toml @@ -0,0 +1,22 @@ +toml-version = "v1.0.0" + +[format] + [format.rules] + indent-sub-tables = true + indent-table-key-value-pairs = true + trailing-comment-alignment = true + +[schema] + enabled = true + strict = true + +[[schemas]] + path = "entity.schema.json" + include = ["**/*.toml"] + exclude = [ + ".tombi.toml", + "extern/**", # submodules: adios2's pyproject/REUSE, entity-pgens' own configs + ".venv/**", + "build/**", + "**/*.ckpt/**", # checkpoint metadata dumps carry a [metadata] table, not input + ] diff --git a/CITATION b/CITATION index 8b1d4947d..a199dd8f8 100644 --- a/CITATION +++ b/CITATION @@ -26,7 +26,7 @@ For the general relativistic module, please cite the following paper: ```latex @ARTICLE{EntityGR_2025, author = {{Galishnikova}, Alisa and {Hakobyan}, Hayk and {Philippov}, Alexander and {Crinquand}, Benjamin}, - title = "{$\mathtt{Entity}$ -- Hardware-agnostic Particle-in-Cell Code for Plasma Astrophysics. II: General Relativistic Module}", + title = "{Entity -- Hardware-agnostic Particle-in-Cell Code for Plasma Astrophysics. II: General Relativistic Module}", journal = {arXiv e-prints}, keywords = {High Energy Astrophysical Phenomena}, year = 2025, diff --git a/CMakeLists.txt b/CMakeLists.txt index d9d1c0ce4..c904ccab2 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -1,6 +1,6 @@ # cmake-lint: disable=C0103,C0111,E1120,R0913,R0915 -cmake_minimum_required(VERSION 3.16) +cmake_minimum_required(VERSION 3.22) cmake_policy(SET CMP0110 NEW) set(PROJECT_NAME entity) @@ -58,27 +58,21 @@ set(gpu_aware_mpi ${default_gpu_aware_mpi} CACHE BOOL "Enable GPU-aware MPI") -set(team_policy - ${default_team_policy} - CACHE BOOL "Enable team_policy tile-blocked deposit/pusher kernels") -set(team_policy_tile_size - ${default_team_policy_tile_size} - CACHE STRING "team_policy tile edge length in cells") -set(team_policy_tile_sizes +set(tiled_deposit + ${default_tiled_deposit} + CACHE BOOL "Enable tile-blocked deposit/pusher kernels") +set(tiled_deposit_tile_size + ${default_tiled_deposit_tile_size} + CACHE STRING "tiled deposit tile edge length in cells") +set(tiled_deposit_tile_sizes "4;6;8;10;12;14;16" - CACHE STRING "team_policy tile-size choices") -set(team_policy_drift - ${default_team_policy_drift} - CACHE - STRING - "team_policy tiled-deposit scratch halo drift in cells (max cells a particle may move between two sorts). Sizes the deposit scratch halo only; the sort cadence is set at runtime via spatial_sorting_interval. Default 1." -) + CACHE STRING "tiled deposit tile-size choices") +set(tiled_deposit_drift + ${default_tiled_deposit_drift} + CACHE STRING "tiled deposit scratch halo drift in cells") set(vendor_sort ${default_vendor_sort} - CACHE - BOOL - "Use the vendor sort_by_key (oneDPL/Thrust/rocThrust) for the team_policy spatial sort when available. OFF forces the Kokkos::BinSort fallback, which sorts each SoA member in place (lower peak memory, no maxnpart gather buffer) at the cost of sort speed." -) + CACHE BOOL "Use the vendor sort_by_key") # -------------------------- Compilation settings -------------------------- # set(CMAKE_CXX_STANDARD 20) @@ -158,9 +152,25 @@ else() set(DEVICE_ENABLED OFF) endif() -# ------------------------------ team_policy wiring ------------------------ # -if(${team_policy}) - include(${CMAKE_CURRENT_SOURCE_DIR}/cmake/team_policy.cmake) +if(NOT ${DEVICE_ENABLED}) + set(vendor_sort OFF) + set(gpu_aware_mpi OFF) +endif() + +# tiled deposit +if(${tiled_deposit}) + include(${CMAKE_CURRENT_SOURCE_DIR}/cmake/tiled_deposit.cmake) +else() + message( + STATUS "tiled_deposit=OFF; using global deposit scheme with ScatterViews") +endif() + +# vendor-specific sorting routines +if(${vendor_sort}) + include(${CMAKE_CURRENT_SOURCE_DIR}/cmake/vendor_sort.cmake) +else() + message(STATUS "vendor_sort=OFF; forcing Kokkos::BinSort " + "fallback for spatial sort_by_key") endif() # MPI diff --git a/CODEGUIDE.md b/CODEGUIDE.md index 1e4dee68d..d69cf03b2 100644 --- a/CODEGUIDE.md +++ b/CODEGUIDE.md @@ -32,6 +32,11 @@ entity ├── pgens # problem generators ├── examples # example problem generators with standard use-cases ├── tutorials # problem generators from tutorials +├── scripts # user-facing helper scripts +│ ├── dependencies.py # deployment scripts on various machines +│ ├── generate_template.py # renders `input.default.toml` from `entity.schema.json` +│ ├── ideal_tile_size.py # recommends the team tile size for the tiled deposit +│ └── render.py # helper tools for the on-the-fly rendering routine ├── src # main code containing all separate submodules │ ├── archetypes # archetypes which can be used by the user in problem generators │ ├── engines # simulation engines @@ -48,15 +53,15 @@ entity ├── .gitattributes ├── .gitignore ├── .gitmodules -├── .taplo.toml # formatting guidelines for toml files +├── .tombi.toml # formatting guidelines for toml files + schema association ├── CITATION ├── CODEGUIDE.md # this file ├── CMakeLists.txt # root cmake file ├── CODE_OF_CONDUCT.md ├── LICENSE ├── README.md -├── dependencies.py # deployment scripts on various machines -└── input.example.toml # most complete toml file with all possible input options +├── entity.schema.json # JSON Schema for the input file: the source of truth +└── input.default.toml # generated reference input with every option at its default ``` ## Testing @@ -75,13 +80,75 @@ You can also compile all the problem generators and run the ones from the `examp ./dev/scripts/tests.sh --build build_dir --flags "-D mpi=ON" --with_pgens --make_plots ``` +## Input configuration + +`entity.schema.json` is the single source of truth for the input file. It is a [JSON Schema](https://json-schema.org) (draft 2020-12) describing every table and key the code reads, and it serves two purposes at once: + +* editors validate and autocomplete input files against it as you type (see [Formatting](#formatting) below); +* `input.default.toml` -- the annotated reference input listing every option -- is *generated* from it, so the docs cannot drift from what is validated. + +Regenerate the reference input after any schema change: + +```sh +python scripts/generate_template.py -d -o input.default.toml +``` + +Dropping `-d` renders the same file with every value left as `""`, i.e. a blank form to fill in rather than a list of defaults. Writing to stdout (the default) is handy for reviewing a change: `diff <(python scripts/generate_template.py -d) input.default.toml`. + +### The `x-entity` annotations + +Standard JSON Schema keywords (`type`, `enum`, `minimum`, `items`, `prefixItems`, `required`, `default`, `deprecated`, ...) carry everything a validator can check. Everything else lives in an `x-entity` object on the node, and is what the generator turns into the `@`-annotations above each key: + +| field | meaning | +| --- | --- | +| `type` | the literal `@type:` string, e.g. `"array [size 1 :->: 3]"` -- richer than the JSON type | +| `default` | the literal `@default:` text, for defaults the code computes at runtime (`"N_GHOSTS"`, `"1% of the domain size"`) or that need a specific notation (`"1e-4"` rather than `0.0001`) | +| `notes` | ordered `@note:` lines; embedded newlines are kept as hard line breaks | +| `examples` | ordered `@example:` lines | +| `enum` | an *illustrative, non-exhaustive* value list, never validated (e.g. `output.fields.quantities`) | +| `deprecated` | the `@deprecated:` text, paired with the standard `"deprecated": true` | +| `inferred` | see below | + +`x-entity.inferred` sits on a **table** and lists quantities the code derives rather than reads -- `grid.dim`, `scales.sigma0`, `checkpoint.start_step`. They are deliberately *not* in `properties`, so `additionalProperties: false` rejects them as input keys, and the generator emits them as an `@inferred:` comment block after that table's own keys. + +### Adding a new input parameter + +1. Add the key to `entity.schema.json`, in the position you want it to appear in the reference input -- property order is emission order, and scalar keys are emitted before sub-tables regardless. +2. Give it a `description` (the brief line) and an `x-entity.type`; add real constraints (`minimum`, `enum`, `minItems`, ...) wherever they are checkable, and a `default` when it has a literal one. +3. Regenerate `input.default.toml`. +4. Parse it in `src/framework/parameters/`, and register any derived quantity under `x-entity.inferred`. + +Three things to keep in mind: + +* **String enums are matched case-insensitively by the code** (`fmt::toLower` is applied to `engine`, `metric`, the boundary lists, `pusher`, `log_level`, ...), so a bare `"enum"` would reject perfectly valid input. The convention is `anyOf: [{"enum": []}, {"type": "string", "pattern": "(?i)^(|...)$"}]` -- the enum branch drives completion and hover, the pattern branch keeps any casing legal. Note `(?i)` is a Rust/Python regex extension: tombi honours it, JS-based validators do not. +* **Every table is closed.** Set `additionalProperties: false` so typos are caught; tombi's `strict = true` closes objects that omit it anyway. `[setup]` is the one deliberate exception (`additionalProperties: true`), since its keys belong to the problem generator. +* **If a key's documented default is `[]`, the empty array must validate**, which `minItems` would otherwise forbid -- use `anyOf: [{"maxItems": 0}, {}]` (see `render.extent.x1`). + ## Code guidelines ### Formatting To maintain coherence throughout the source code, we use `clang-format` to enforce a uniform style. A corresponding `.clang-format` file with all the style-related settings can be found in the root directory of the code. To use this, one needs to have the `clang-format` executable (typically provided with the `llvm` package). After installing the `clang-format` itself (check by running `clang-format --version`), you can use it either manually by running `clang-format .` in the route directory of the code, or attach it to your favorite code editor to run on save. For VSCode, the recommended extension is [`xaver.clang-format`](https://github.com/xaverh/vscode-clang-format), for vim -- [`rhysd/vim-clang-format`](https://vimawesome.com/plugin/vim-clang-format), for nvim -- [`stevearc/conform.nvim`](https://github.com/stevearc/conform.nvim), for [emacs](https://www.vim.org/download.php). -You can run the formatting on all files with `./dev/scripts/format.sh`. +You can run the formatting on all files with `./dev/scripts/format.sh` (this covers C++ and CMake; TOML is handled separately, below). + +TOML files are formatted and validated with [`tombi`](https://tombi-toml.github.io/tombi/), which is a formatter, linter and language server in one. The settings live in `.tombi.toml` in the root directory, which also associates `entity.schema.json` with every `.toml` file in the tree -- so input files are checked against the schema as you edit them, with completion and hover documentation for every key. It is provided by the nix shell (`dev/nix`); otherwise install it with `uvx tombi`, `pip install tombi`, `npm i -g tombi` or `brew install tombi`. + +From the command line: + +```sh +tombi format # formats the whole project (or pass files/directories) +tombi format --check # verify only, for CI -- mirrors `format.sh --verify` +tombi lint # schema validation only +``` + +In the editor, point it at the `tombi lsp` language server. For VSCode, the extension is [`tombi-toml.tombi`](https://marketplace.visualstudio.com/items?itemName=tombi-toml.tombi); for nvim, `tombi` ships as a built-in `nvim-lspconfig` server, so `vim.lsp.enable('tombi')` is enough. Individual input files can opt into the schema explicitly -- useful outside the repo -- with a directive on the first line: + +```toml +#:schema ./entity.schema.json +``` + +> `tombi` replaces `taplo`, which the project used previously and which is no longer maintained. Best practices are also enforced using `clang-tidy`; to generate recommendations for all the files, run `./dev/scripts/tidy.sh --build build_dir` where `build_dir` is the directory where the code was built, or for specific files: `./dev/scripts/tidy.sh --build build_dir --files "(file1|file2).cpp"` or only for the changed files: `./dev/scripts/tidy.sh --build build_dir --changed`. The recommendations will be in the `tidy/` directory. @@ -120,4 +187,4 @@ Best practices are also enforced using `clang-tidy`; to generate recommendations * There is no difference between `.h` and `.hpp` files as both indicate C++ header files. As a consistency convention, we use `.h` for common headers which may be included from multiple `.cpp` files (e.g., metrics), while `.hpp` are very specific headers for only a single (or a couple of) .cpp file (e.g. kernels). -* Do assertions on parameters and quantities whenever possible. Outside the kernels, use `raise::Error(message, HERE)` and `raise::ErrorIf(condition, message, HERE)` to throw exceptions. Inside the kernels, use `raise::KernelError(HERE, message, **args)`. To enable compile-time errors, use `static_assert(condition, message)`. The `HERE` keyword is macro that includes the filename and line number in the error message. +* Do assertions on parameters and quantities whenever possible. Outside the kernels, use `raise::Error(message, HERE)` and `raise::ErrorIf(condition, message, HERE)` to throw exceptions. Inside the kernels, use `raise::KernelError(HERE, message)`. To enable compile-time errors, use `static_assert(condition, message)`. The `HERE` keyword is macro that includes the filename and line number in the error message. diff --git a/cmake/defaults.cmake b/cmake/defaults.cmake index 888be1b00..60b97e0d7 100644 --- a/cmake/defaults.cmake +++ b/cmake/defaults.cmake @@ -93,16 +93,22 @@ endif() set_property(CACHE default_gpu_aware_mpi PROPERTY TYPE BOOL) -if(DEFINED ENV{Entity_ENABLE_TEAM_POLICY}) - set(default_team_policy +if(DEFINED ENV{Entity_ENABLE_TILED_DEPOSIT}) + set(default_tiled_deposit + $ENV{Entity_ENABLE_TILED_DEPOSIT} + CACHE INTERNAL "Default flag for tiled_deposit tile-blocked kernels") +elseif(DEFINED ENV{Entity_ENABLE_TEAM_POLICY}) + message(WARNING "`Entity_ENABLE_TEAM_POLICY` is deprecated, " + "use `Entity_ENABLE_TILED_DEPOSIT` instead") + set(default_tiled_deposit $ENV{Entity_ENABLE_TEAM_POLICY} - CACHE INTERNAL "Default flag for team_policy tile-blocked kernels") + CACHE INTERNAL "Default flag for tiled_deposit tile-blocked kernels") else() - set(default_team_policy + set(default_tiled_deposit OFF - CACHE INTERNAL "Default flag for team_policy tile-blocked kernels") + CACHE INTERNAL "Default flag for tiled_deposit tile-blocked kernels") endif() -set_property(CACHE default_team_policy PROPERTY TYPE BOOL) +set_property(CACHE default_tiled_deposit PROPERTY TYPE BOOL) if(DEFINED ENV{Entity_ENABLE_VENDOR_SORT}) set(default_vendor_sort @@ -117,13 +123,11 @@ else() endif() set_property(CACHE default_vendor_sort PROPERTY TYPE BOOL) -set(default_team_policy_tile_size +set(default_tiled_deposit_tile_size 8 - CACHE INTERNAL "Default tile edge length in cells for team_policy") + CACHE INTERNAL "Default tile edge length in cells for tiled_deposit") -set(default_team_policy_drift +set(default_tiled_deposit_drift 1 - CACHE - INTERNAL - "Default tiled-deposit scratch halo drift for team_policy (cells between sorts)" -) + CACHE INTERNAL + "Default tiled-deposit scratch halo drift (cells between sorts)") diff --git a/cmake/dependencies.cmake b/cmake/dependencies.cmake index 93a8a17da..07575ab34 100644 --- a/cmake/dependencies.cmake +++ b/cmake/dependencies.cmake @@ -10,7 +10,7 @@ set(adios2_REPOSITORY https://github.com/ornladios/ADIOS2.git CACHE STRING "ADIOS2 repository") set(adios2_TAG - v2.11.0 + v2.12.1 CACHE STRING "ADIOS2 tag") set(CONNECTION_CHECKED diff --git a/cmake/report.cmake b/cmake/report.cmake index e6ad4aaf8..f9470d602 100644 --- a/cmake/report.cmake +++ b/cmake/report.cmake @@ -121,22 +121,22 @@ printchoices( GPU_AWARE_MPI_REPORT 44) printchoices( - "Team Policy" - "team_policy" + "Tiled Deposit" + "tiled_deposit" "${ON_OFF_VALUES}" - ${team_policy} + ${tiled_deposit} OFF "${Green}" - TEAM_POLICY_REPORT + TILED_DEPOSIT_REPORT 44) printchoices( "Tile Size" - "team_policy_tile_size" - "${team_policy_tile_sizes}" - ${team_policy_tile_size} - ${default_team_policy_tile_size} + "tiled_deposit_tile_size" + "${tiled_deposit_tile_sizes}" + ${tiled_deposit_tile_size} + ${default_tiled_deposit_tile_size} "${Blue}" - TEAM_POLICY_TILE_SIZE_REPORT + TILED_DEPOSIT_TILE_SIZE_REPORT 44) printchoices( "Vendor sort" @@ -246,18 +246,18 @@ string( " " ${GPU_AWARE_MPI_REPORT} "\n" - " > Team-policy specs" - " ${Dim}[requires team_policy=ON]${ColorReset}" + " > Tiled-deposit specs" + " ${Dim}[requires tiled_deposit=ON]${ColorReset}" "\n" " " - ${TEAM_POLICY_REPORT} + ${TILED_DEPOSIT_REPORT} "\n" " " - ${TEAM_POLICY_TILE_SIZE_REPORT} + ${TILED_DEPOSIT_TILE_SIZE_REPORT} "\n" " " - "- Deposit drift [${Magenta}team_policy_drift${ColorReset}]: " - ${team_policy_drift} + "- Deposit drift [${Magenta}tiled_deposit_drift${ColorReset}]: " + ${tiled_deposit_drift} "\n") string( diff --git a/cmake/team_policy.cmake b/cmake/team_policy.cmake deleted file mode 100644 index 1217bc28c..000000000 --- a/cmake/team_policy.cmake +++ /dev/null @@ -1,76 +0,0 @@ -list(FIND team_policy_tile_sizes "${team_policy_tile_size}" _tps_idx) -if(_tps_idx EQUAL -1) - message( - FATAL_ERROR - "${Red}team_policy_tile_size must be one of ${team_policy_tile_sizes}, " - "got '${team_policy_tile_size}'${ColorReset}") -endif() -add_compile_options("-D TEAM_POLICY") -add_compile_options("-D TEAM_POLICY_TILE_SIZE=${team_policy_tile_size}") - -# Compile-time tiled-deposit scratch halo drift. Sizes the halo so a particle -# that drifts up to DRIFT cells between two sorts still deposits inside its tile -# scratch; particles drifting further take the per-particle global-J escape -# valve (correct, only slower). This is independent of the sort cadence, which -# is set at runtime via `spatial_sorting_interval`. Defaults to 1 (the -# sorted-every-step case). -add_compile_options("-D TEAM_POLICY_DRIFT=${team_policy_drift}") - -# Vendor sort: oneDPL on SYCL, Thrust on CUDA, rocThrust/rocprim on HIP. When -# `vendor_sort` is ON (default) the available library is detected and used; the -# spatial sort then builds a single permutation that gathers all SoA members. -# When `vendor_sort` is OFF, or no library is found, the code falls back to -# Kokkos::BinSort, which sorts each member in place -- lower peak memory and no -# maxnpart gather buffer, at the cost of sort speed (negligible when sorting is -# a small fraction of the step). The `vendor_sort` knob lets you force the -# BinSort fallback even when a vendor library is present. -if(${vendor_sort}) - if("${Kokkos_DEVICES}" MATCHES "SYCL") - find_package(oneDPL QUIET) - if(oneDPL_FOUND) - message(STATUS "team_policy: oneDPL found, enabling SYCL sort_by_key") - add_compile_options("-D ONEDPL_ENABLED") - set(DEPENDENCIES ${DEPENDENCIES} oneDPL) - else() - message(STATUS "team_policy: oneDPL not found; using BinSort fallback " - "for SYCL sort_by_key") - endif() - endif() - - if("${Kokkos_DEVICES}" MATCHES "CUDA") - find_package(Thrust QUIET) - if(Thrust_FOUND) - message(STATUS "team_policy: Thrust enabled for CUDA sort_by_key") - add_compile_options("-D THRUST_ENABLED") - else() - message(STATUS "team_policy: Thrust not found; using BinSort fallback " - "for CUDA sort_by_key") - endif() - endif() - - if("${Kokkos_DEVICES}" MATCHES "HIP") - # rocThrust ships with ROCm. The HIP sort_by_key path uses rocprim's - # bounded-bit radix sort directly (rocprim is rocThrust's own dependency, so - # its headers come in transitively; we find it explicitly to keep the - # include path robust). This builds a single permutation that gathers all - # SoA members, instead of the legacy per-member Kokkos::BinSort path which - # allocates a fresh `sorted_values` buffer for every member every step (the - # dominant source of allocator churn / fragmentation on ROCm). - find_package(rocthrust QUIET) - if(rocthrust_FOUND) - message(STATUS "team_policy: rocThrust enabled for HIP sort_by_key") - add_compile_options("-D ROCTHRUST_ENABLED") - set(DEPENDENCIES ${DEPENDENCIES} roc::rocthrust) - find_package(rocprim QUIET) - if(rocprim_FOUND) - set(DEPENDENCIES ${DEPENDENCIES} roc::rocprim) - endif() - else() - message(STATUS "team_policy: rocThrust not found; using BinSort " - "fallback for HIP sort_by_key") - endif() - endif() -else() - message(STATUS "team_policy: vendor_sort=OFF; forcing Kokkos::BinSort " - "fallback for spatial sort_by_key") -endif() diff --git a/cmake/tiled_deposit.cmake b/cmake/tiled_deposit.cmake new file mode 100644 index 000000000..608e8cf80 --- /dev/null +++ b/cmake/tiled_deposit.cmake @@ -0,0 +1,17 @@ +list(FIND tiled_deposit_tile_sizes "${tiled_deposit_tile_size}" _tps_idx) +if(_tps_idx EQUAL -1) + message( + FATAL_ERROR + "${Red}tiled_deposit_tile_size must be one of ${tiled_deposit_tile_sizes}, " + "got '${tiled_deposit_tile_size}'${ColorReset}") +endif() +add_compile_options("-D TILED_DEPOSIT") +add_compile_options("-D TILED_DEPOSIT_TILE_SIZE=${tiled_deposit_tile_size}") + +# Compile-time tiled-deposit scratch halo drift. Sizes the halo so a particle +# that drifts up to DRIFT cells between two sorts still deposits inside its tile +# scratch; particles drifting further take the per-particle global-J escape +# valve (correct, only slower). This is independent of the sort cadence, which +# is set at runtime via `spatial_sorting_interval`. Defaults to 1 (the +# sorted-every-step case). +add_compile_options("-D TILED_DEPOSIT_DRIFT=${tiled_deposit_drift}") diff --git a/cmake/vendor_sort.cmake b/cmake/vendor_sort.cmake new file mode 100644 index 000000000..16176604e --- /dev/null +++ b/cmake/vendor_sort.cmake @@ -0,0 +1,51 @@ +# Vendor sort: oneDPL on SYCL, Thrust on CUDA, rocThrust/rocprim on HIP. When +# `vendor_sort` is ON (default) the available library is detected and used; the +# spatial sort then builds a single permutation that gathers all SoA members. +# When `vendor_sort` is OFF, or no library is found, the code falls back to +# Kokkos::BinSort, which sorts each member in place -- lower peak memory and no +# maxnpart gather buffer, at the cost of sort speed (negligible when sorting is +# a small fraction of the step). The `vendor_sort` knob lets you force the +# BinSort fallback even when a vendor library is present. +if("${Kokkos_DEVICES}" MATCHES "SYCL") + find_package(oneDPL QUIET) + if(oneDPL_FOUND) + message(STATUS "oneDPL found, enabling SYCL sort_by_key") + add_compile_options("-D ONEDPL_ENABLED") + set(DEPENDENCIES ${DEPENDENCIES} oneDPL) + else() + message(STATUS "oneDPL not found; using BinSort fallback " + "for SYCL sort_by_key") + endif() +elseif("${Kokkos_DEVICES}" MATCHES "CUDA") + find_package(Thrust QUIET) + if(Thrust_FOUND) + message(STATUS "Thrust enabled for CUDA sort_by_key") + add_compile_options("-D THRUST_ENABLED") + else() + message(STATUS "Thrust not found; using BinSort fallback " + "for CUDA sort_by_key") + endif() +elseif("${Kokkos_DEVICES}" MATCHES "HIP") + # rocThrust ships with ROCm. The HIP sort_by_key path uses rocprim's + # bounded-bit radix sort directly (rocprim is rocThrust's own dependency, so + # its headers come in transitively; we find it explicitly to keep the include + # path robust). This builds a single permutation that gathers all SoA members, + # instead of the legacy per-member Kokkos::BinSort path which allocates a + # fresh `sorted_values` buffer for every member every step (the dominant + # source of allocator churn / fragmentation on ROCm). + find_package(rocthrust QUIET) + if(rocthrust_FOUND) + message(STATUS "rocThrust enabled for HIP sort_by_key") + add_compile_options("-D ROCTHRUST_ENABLED") + set(DEPENDENCIES ${DEPENDENCIES} roc::rocthrust) + find_package(rocprim QUIET) + if(rocprim_FOUND) + set(DEPENDENCIES ${DEPENDENCIES} roc::rocprim) + endif() + else() + message(STATUS "rocThrust not found; using BinSort " + "fallback for HIP sort_by_key") + endif() +else() + message(FATAL_ERROR "vendor_sort enabled, but device not recognized") +endif() diff --git a/dev/nix/adios2.nix b/dev/nix/adios2.nix index eb3130632..f7114a017 100644 --- a/dev/nix/adios2.nix +++ b/dev/nix/adios2.nix @@ -6,7 +6,7 @@ let name = "adios2"; - version = "2.11.0"; + version = "2.12.1"; cmakeFlags = { CMAKE_CXX_STANDARD = "20"; CMAKE_CXX_EXTENSIONS = "OFF"; @@ -30,7 +30,7 @@ stdenv.mkDerivation { src = pkgs.fetchgit { url = "https://github.com/ornladios/ADIOS2/"; rev = "v${version}"; - sha256 = "sha256-yHPI///17poiCEb7Luu5qfqxTWm9Nh+o9r57mZT26U0="; + sha256 = "sha256-3jMvVYYO93/Pu7RW2x5mzTRMrZ3oC3IwGrUz2tSqJxQ="; }; nativeBuildInputs = with pkgs; [ diff --git a/dev/nix/devenv.lock b/dev/nix/devenv.lock new file mode 100644 index 000000000..9e6f7f3b4 --- /dev/null +++ b/dev/nix/devenv.lock @@ -0,0 +1,45 @@ +{ + "nodes": { + "devenv": { + "locked": { + "dir": "src/modules", + "lastModified": 1789991752, + "narHash": "sha256-SiH+H00IutELZBg7BBNo6r5DU6E+DG7DjrbeX5+QLa0=", + "owner": "cachix", + "repo": "devenv", + "rev": "1c57b5dea0d400af97053fdd1a536fca17378f73", + "type": "github" + }, + "original": { + "dir": "src/modules", + "owner": "cachix", + "repo": "devenv", + "type": "github" + } + }, + "nixpkgs": { + "locked": { + "lastModified": 1789921291, + "narHash": "sha256-Ft/BRnIqw1MywFoXydKobjjWmDFgDdYtSpJliE8+yUw=", + "owner": "NixOS", + "repo": "nixpkgs", + "rev": "44a91898084f46797b5fac650c7e8c9ac38c43d4", + "type": "github" + }, + "original": { + "owner": "NixOS", + "ref": "nixos-unstable", + "repo": "nixpkgs", + "type": "github" + } + }, + "root": { + "inputs": { + "devenv": "devenv", + "nixpkgs": "nixpkgs" + } + } + }, + "root": "root", + "version": 7 +} \ No newline at end of file diff --git a/dev/nix/devenv.nix b/dev/nix/devenv.nix new file mode 100644 index 000000000..91206dfac --- /dev/null +++ b/dev/nix/devenv.nix @@ -0,0 +1,191 @@ +# devenv counterpart of `shell.nix`; run from this directory: +# devenv shell # cpu-only +# devenv shell -P cuda -O entity.arch:string AMPERE80 # cuda +# devenv shell -P hip -O entity.arch:string AMD_GFX90A # hip +# devenv shell -P mpi -P hdf5 # adios2 with mpi + hdf5 +# persistent settings can be put into `devenv.local.nix` (gitignored). +{ + pkgs, + lib, + config, + inputs, + ... +}: + +let + cfg = config.entity; + + gpu = lib.toUpper cfg.gpu; + arch = lib.toUpper cfg.arch; + + # override with + # devenv shell -O languages.python.package:pkg python312 + py = "314"; + + # `shell.nix` imports nixpkgs with `allowUnfree`/`cudaSupport` decided by the + # requested backend. devenv instantiates its own `pkgs` before this module is + # evaluated, so it cannot be reconfigured from here -- import the same input + # ourselves and build everything from that instance. + nixpkgs = import inputs.nixpkgs { + inherit (pkgs.stdenv.hostPlatform) system; + config = { + allowUnfree = true; + cudaSupport = gpu == "CUDA"; + }; + }; + + adios2Pkg = nixpkgs.callPackage ./adios2.nix { + pkgs = nixpkgs; + inherit (cfg) hdf5 mpi; + }; + + kokkosPkg = nixpkgs.callPackage ./kokkos.nix { + pkgs = nixpkgs; + stdenv = nixpkgs.stdenv; + inherit arch gpu; + }; + + extraPkgs = map (name: nixpkgs.${name}) (lib.filter (s: s != "") (lib.splitString "," cfg.extra)); + + # compilers are picked by the backend; CUDA goes through kokkos' nvcc_wrapper + compilerEnv = + { + NONE = { + CXX = "g++"; + CC = "gcc"; + }; + HIP = { + CXX = "clang++"; + CC = "clang"; + }; + CUDA = { }; + } + .${gpu}; +in +{ + options.entity = { + gpu = lib.mkOption { + # case-insensitive, as in `shell.nix` + type = lib.types.enum [ + "NONE" + "none" + "CUDA" + "cuda" + "HIP" + "hip" + ]; + default = "NONE"; + description = "GPU backend to build Kokkos with."; + }; + + arch = lib.mkOption { + type = lib.types.str; + default = "NATIVE"; + example = "AMPERE80"; + description = '' + Kokkos architecture; mandatory when `gpu` is not `NONE`. See + https://kokkos.org/kokkos-core-wiki/get-started/configuration-guide.html#gpu-architectures + ''; + }; + + hdf5 = lib.mkOption { + type = lib.types.bool; + default = false; + description = "Build ADIOS2 with HDF5 support."; + }; + + mpi = lib.mkOption { + type = lib.types.bool; + default = false; + description = "Build ADIOS2 with MPI support."; + }; + + extra = lib.mkOption { + type = lib.types.str; + default = ""; + example = "gdb,valgrind"; + description = '' + Comma-separated nixpkgs attributes to add to the environment, kept for + parity with `shell.nix`. `-O packages:pkgs "gdb valgrind"` does the same + without going through this option. + ''; + }; + }; + + config = { + name = + "nt2" + (if gpu != "NONE" then "-${lib.toLower gpu}" else "") + (if cfg.mpi then "-mpi" else ""); + + profiles = { + cuda.module = { + entity.gpu = "CUDA"; + }; + hip.module = { + entity.gpu = "HIP"; + }; + mpi.module = { + entity.mpi = true; + }; + hdf5.module = { + entity.hdf5 = true; + }; + }; + + languages.python = { + enable = true; + package = pkgs."python${py}"; + + venv = { + enable = true; + requirements = '' + ipykernel + jupyterlab + nt2py + ruff + pyright + ''; + }; + }; + + packages = + (with nixpkgs; [ + zlib + cmake + + adios2Pkg + kokkosPkg + + cmake-format + cmake-lint + neocmakelsp + black + pyright + tombi + vscode-langservers-extracted + ]) + ++ extraPkgs; + + env = compilerEnv // { + LD_LIBRARY_PATH = lib.makeLibraryPath [ + nixpkgs.stdenv.cc.cc + nixpkgs.zlib + ]; + }; + + enterShell = '' + BLUE='\033[0;34m' + NC='\033[0m' + + echo "following environment variables are set:" + '' + + lib.concatStringsSep "" ( + lib.mapAttrsToList (name: value: '' + echo -e " ''${BLUE}${name}''${NC}=${value}" + '') compilerEnv + ) + + '' + echo "" + echo -e "${config.name} devenv activated" + ''; + }; +} diff --git a/dev/nix/devenv.yaml b/dev/nix/devenv.yaml new file mode 100644 index 000000000..b4c9128be --- /dev/null +++ b/dev/nix/devenv.yaml @@ -0,0 +1,4 @@ +# devenv inputs; update the lock with `devenv update --from path:./dev/nix` +inputs: + nixpkgs: + url: github:NixOS/nixpkgs/nixos-unstable diff --git a/dev/nix/shell.nix b/dev/nix/shell.nix index addd470fa..6b6c950f2 100644 --- a/dev/nix/shell.nix +++ b/dev/nix/shell.nix @@ -55,14 +55,14 @@ pkgs.mkShell { neocmakelsp black pyright - taplo + tombi vscode-langservers-extracted ]; - LD_LIBRARY_PATH = pkgs.lib.makeLibraryPath ([ + LD_LIBRARY_PATH = pkgs.lib.makeLibraryPath [ pkgs.stdenv.cc.cc pkgs.zlib - ]); + ]; shellHook = '' BLUE='\033[0;34m' diff --git a/entity.schema.json b/entity.schema.json new file mode 100644 index 000000000..6ca79d81d --- /dev/null +++ b/entity.schema.json @@ -0,0 +1,2936 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "https://entity-toolkit.github.io/schema/entity.schema.json", + "title": "entity", + "description": "Entity simulation input file", + "$comment": "GENERATOR CONVENTION -- this schema is the single source of truth for `input.template.toml`. Standard keywords carry the machine-checkable part (type/enum/minimum/maximum/items/prefixItems/minItems/maxItems/pattern/default/required/deprecated). The `x-entity` object carries the parts JSON Schema cannot express, all verbatim from the template comments: `type` = the literal `@type:` annotation (use it for the comment line whenever present, else derive from the standard keywords); `default` = the literal `@default:` annotation (use it whenever present, else format the standard `default`); `notes` = ordered `@note:` lines; `examples` = ordered `@example:` lines; `enum` = an illustrative, NON-exhaustive value list that must not be validated as a real `enum`; `deprecated` = the `@deprecated:` text. `x-entity.inferred` on an object lists quantities the code derives rather than reads; they are NOT valid input keys (so they are absent from `properties` and rejected by `additionalProperties: false`) and should be emitted as an `@inferred:` comment block after that table's own keys and before its sub-tables. Property order in `properties` is the emission order. Objects with `additionalProperties: true` are free-form. `$ref` is used once, for `$defs/colormap`.", + "type": "object", + "additionalProperties": false, + "required": [ + "simulation", + "grid", + "scales", + "particles" + ], + "$defs": { + "colormap": { + "type": "string", + "pattern": "(?i)^(cmr\\.)?(viridis|inferno|plasma|cool2warm|gray|RdBu_r|dusk|cosmic|freeze|apple|gothic|sunburst|voltage|ocean|fusion|prinsenvlag)$", + "x-entity": { + "type": "string", + "enum": [ + "viridis", + "inferno", + "plasma", + "cool2warm", + "gray", + "RdBu_r", + "and the CMasher maps (BSD-3, https://cmasher.readthedocs.io): \"dusk\", \"cosmic\", \"freeze\", \"apple\", \"gothic\", \"sunburst\", \"voltage\", \"ocean\", \"fusion\", \"prinsenvlag\" (an optional \"cmr.\" prefix is ok)" + ] + } + } + }, + "properties": { + "simulation": { + "description": "Global simulation parameters", + "type": "object", + "additionalProperties": false, + "required": [ + "name", + "engine", + "runtime" + ], + "properties": { + "name": { + "description": "Name of the simulation", + "type": "string", + "x-entity": { + "type": "string", + "notes": [ + "The name is used for the output files" + ] + } + }, + "engine": { + "description": "Simulation engine to use", + "anyOf": [ + { + "enum": [ + "SRPIC", + "GRPIC" + ] + }, + { + "type": "string", + "pattern": "(?i)^(SRPIC|GRPIC)$" + } + ], + "x-entity": { + "type": "string" + } + }, + "runtime": { + "description": "Max runtime in physical (code) units", + "type": "number", + "exclusiveMinimum": 0.0, + "x-entity": { + "type": "float [> 0]", + "examples": [ + "1e5" + ] + } + }, + "domain": { + "description": "Parameters specific to domain decomposition", + "type": "object", + "additionalProperties": false, + "properties": { + "number": { + "description": "Number of domains", + "type": "integer", + "minimum": 1, + "x-entity": { + "type": "int", + "default": "1 [no MPI]; MPI_SIZE [MPI]" + } + }, + "decomposition": { + "description": "Decomposition of the domain (for MPI) in each of the directions", + "type": "array", + "minItems": 1, + "maxItems": 3, + "items": { + "type": "integer" + }, + "default": [ + -1, + -1, + -1 + ], + "x-entity": { + "type": "array [size 1 :->: 3]", + "notes": [ + "-1 means the code will determine the decomposition in the specific direction automatically", + "Automatic detection is either done by inference from # of MPI tasks, or by balancing the grid size on each domain" + ], + "examples": [ + "[2, 2, 2] (total of 8 domains)" + ] + } + }, + "load_balance": { + "description": "Diffusion-style dynamic load balancing. Domain boundaries between MPI neighbors are nudged to equalize the active-particle count per rank. All inter-rank traffic uses only the existing nearest-neighbor field/particle communication paths.", + "type": "object", + "additionalProperties": false, + "properties": { + "enable": { + "description": "Enable dynamic load balancing", + "type": "boolean", + "default": false, + "x-entity": { + "type": "bool" + } + }, + "interval": { + "description": "Run the rebalancer every `interval` timesteps (0 disables)", + "type": "integer", + "minimum": 0, + "default": 0, + "x-entity": { + "type": "int" + } + }, + "dimensions": { + "description": "Dimensions along which load is redistributed (1 = x1, 2 = x2, 3 = x3)", + "type": "array", + "minItems": 1, + "maxItems": 3, + "uniqueItems": true, + "items": { + "type": "integer", + "enum": [ + 1, + 2, + 3 + ] + }, + "default": [ + 1 + ], + "x-entity": { + "type": "array [subset of {1, 2, 3}]" + } + }, + "tolerance": { + "description": "Skip rebalancing along a dim when (max - min) / mean of the per-slice particle count is below this fraction", + "type": "number", + "minimum": 0.0, + "default": 0.1, + "x-entity": { + "type": "float" + } + }, + "max_shift": { + "description": "Maximum cell-shift per interior boundary per event; clamped at compile time to N_GHOSTS so the migrating field strip is already cached in the rank's ghost zone.", + "type": "integer", + "minimum": 0, + "x-entity": { + "type": "int", + "default": "N_GHOSTS" + } + } + } + } + } + } + } + }, + "grid": { + "description": "Parameters specific to grid geometry", + "type": "object", + "additionalProperties": false, + "required": [ + "resolution", + "extent", + "metric", + "boundaries" + ], + "x-entity": { + "inferred": [ + { + "name": "dim", + "brief": "Dimensionality of the grid", + "type": "short", + "enum": [ + 1, + 2, + 3 + ], + "from": "`grid.resolution`" + } + ] + }, + "properties": { + "resolution": { + "description": "Spatial resolution of the grid", + "type": "array", + "minItems": 1, + "maxItems": 3, + "items": { + "type": "integer", + "minimum": 1 + }, + "x-entity": { + "type": "array [size 1 :->: 3]", + "notes": [ + "Dimensionality is inferred from the size of this array" + ], + "examples": [ + "[1024, 1024, 1024]" + ] + } + }, + "extent": { + "description": "Physical extent of the grid", + "type": "array", + "minItems": 1, + "maxItems": 3, + "items": { + "type": "array", + "minItems": 2, + "maxItems": 2, + "prefixItems": [ + { + "type": "number" + }, + { + "type": "number" + } + ] + }, + "x-entity": { + "type": "array> [size 1 :->: 3]", + "notes": [ + "For spherical geometry, only specify `[[rmin, rmax]]`, other values are set automatically", + "For cartesian geometry, cell aspect ratio has to be 1: `dx=dy=dz`" + ], + "examples": [ + "[[0.0, 1.0], [-1.0, 1.0]]" + ] + } + }, + "metric": { + "description": "Metric-related parameters", + "type": "object", + "additionalProperties": false, + "required": [ + "metric" + ], + "x-entity": { + "inferred": [ + { + "name": "coord", + "brief": "Coordinate system on the grid", + "type": "string", + "enum": [ + "cartesian", + "spherical", + "qspherical" + ], + "from": "`grid.metric.metric`" + }, + { + "name": "ks_rh", + "brief": "Size of the horizon for GR Kerr Schild", + "type": "float", + "from": "`grid.metric.ks_a`" + }, + { + "name": "params", + "brief": "A map of all metric-specific parameters together (for easy access)", + "type": "map", + "from": "`grid.metric`" + } + ] + }, + "properties": { + "metric": { + "description": "Metric on the grid", + "anyOf": [ + { + "enum": [ + "Minkowski", + "Spherical", + "QSpherical", + "Kerr_Schild", + "QKerr_Schild", + "Kerr_Schild_0" + ] + }, + { + "type": "string", + "pattern": "(?i)^(Minkowski|Spherical|QSpherical|Kerr_Schild|QKerr_Schild|Kerr_Schild_0)$" + } + ], + "x-entity": { + "type": "string" + } + }, + "qsph_r0": { + "description": "`r0` paramter for the QSpherical metric `x1 = log(r-r0)`", + "type": "number", + "default": 0.0, + "x-entity": { + "type": "float [-inf -> rmin]", + "notes": [ + "Negative values produce almost uniform grid in r" + ] + } + }, + "qsph_h": { + "description": "`h` paramter for the QSpherical metric `th = x2 + 2*h x2 (pi-2*x2)*(pi-x2)/pi^2`", + "type": "number", + "minimum": -1.0, + "maximum": 1.0, + "default": 0.0, + "x-entity": { + "type": "float [-1 :->: 1]" + } + }, + "ks_a": { + "description": "Spin parameter for the Kerr Schild metric", + "type": "number", + "minimum": 0.0, + "maximum": 1.0, + "default": 0.0, + "x-entity": { + "type": "float [0 :-> 1]" + } + } + } + }, + "boundaries": { + "description": "Boundary-condition related parameters", + "type": "object", + "additionalProperties": false, + "required": [ + "fields", + "particles" + ], + "properties": { + "fields": { + "description": "Boundary conditions for fields", + "type": "array", + "minItems": 1, + "maxItems": 3, + "items": { + "type": "array", + "minItems": 1, + "maxItems": 2, + "items": { + "anyOf": [ + { + "enum": [ + "PERIODIC", + "MATCH", + "FIXED", + "ATMOSPHERE", + "CUSTOM", + "HORIZON", + "CONDUCTOR" + ] + }, + { + "type": "string", + "pattern": "(?i)^(PERIODIC|MATCH|FIXED|ATMOSPHERE|CUSTOM|HORIZON|CONDUCTOR)$" + } + ] + } + }, + "x-entity": { + "type": "array> [size 1 :->: 3]", + "notes": [ + "When periodic in any of the directions, you should only set one value: [..., [\"PERIODIC\"], ...]", + "In spherical, bondaries in theta/phi are set automatically (only specify bc @ `[rmin, rmax]`): [[\"ATMOSPHERE\", \"MATCH\"]]", + "In GR, the horizon boundary is set automatically (only specify bc @ rmax): [[\"MATCH\"]]" + ], + "examples": [ + "[[\"CUSTOM\", \"MATCH\"]] (for 2D spherical `[[rmin, rmax]]`)" + ] + } + }, + "particles": { + "description": "Boundary conditions for particles", + "type": "array", + "minItems": 1, + "maxItems": 3, + "items": { + "type": "array", + "minItems": 1, + "maxItems": 2, + "items": { + "anyOf": [ + { + "enum": [ + "PERIODIC", + "ABSORB", + "ATMOSPHERE", + "CUSTOM", + "REFLECT", + "HORIZON" + ] + }, + { + "type": "string", + "pattern": "(?i)^(PERIODIC|ABSORB|ATMOSPHERE|CUSTOM|REFLECT|HORIZON)$" + } + ] + } + }, + "x-entity": { + "type": "array> [size 1 :->: 3]", + "notes": [ + "When periodic in any of the directions, you should only set one value [..., [\"PERIODIC\"], ...]", + "In spherical, bondaries in theta/phi are set automatically (only specify bc @ `[rmin, rmax]`) [[\"ATMOSPHERE\", \"ABSORB\"]]", + "In GR, the horizon boundary is set automatically (only specify bc @ `rmax`): [[\"ABSORB\"]]" + ], + "examples": [ + "[[\"PERIODIC\"], [\"PERIODIC\"]]" + ] + } + }, + "match": { + "description": "Parameters specific to MATCH boundary conditions", + "type": "object", + "additionalProperties": false, + "properties": { + "ds": { + "description": "Size of the matching layer in each direction for fields in physical (code) units", + "anyOf": [ + { + "type": "number", + "exclusiveMinimum": 0.0 + }, + { + "type": "array", + "minItems": 1, + "maxItems": 3, + "items": { + "type": "array", + "maxItems": 2, + "items": { + "type": "number" + } + } + } + ], + "x-entity": { + "type": "float | array>", + "default": "1% of the domain size (in shortest dimension)", + "notes": [ + "In spherical, this is the size of the layer in `r` from the outer wall" + ], + "examples": [ + "`ds = 1.5` (will set the same for all directions)", + "`ds = [[1.5], [2.0, 1.0], [1.1]]` (will duplicate 1.5 for +/- `x1` and 1.1 for +/- `x3`)", + "`ds = [[], [1.5], []]` (will only set for x2)" + ] + } + } + } + }, + "absorb": { + "description": "Parameters specific to ABSORB boundary conditions", + "type": "object", + "additionalProperties": false, + "properties": { + "ds": { + "description": "Size of the absorption layer for particles in physical (code) units", + "type": "number", + "exclusiveMinimum": 0.0, + "x-entity": { + "type": "float", + "default": "1% of the domain size (in shortest dimension)", + "notes": [ + "In spherical, this is the size of the layer in `r` from the outer wall", + "In cartesian, this is the same for all dimensions where applicable" + ] + } + } + } + }, + "atmosphere": { + "description": "Parameters specific to ATMOSPHERE boundary conditions", + "type": "object", + "additionalProperties": false, + "x-entity": { + "inferred": [ + { + "name": "g", + "brief": "Acceleration due to imposed gravity", + "type": "float", + "from": "`grid.boundaries.atmosphere.temperature`, `grid.boundaries.atmosphere.height`", + "value": "`temperature / height`" + } + ] + }, + "properties": { + "temperature": { + "description": "Temperature of the atmosphere in units of `m0 c^2`", + "type": "number", + "minimum": 0.0, + "x-entity": { + "type": "float", + "notes": [ + "[required] if `ATMOSPHERE` is one of the boundaries" + ] + } + }, + "density": { + "description": "Peak number density of the atmosphere at base in units of `n0`", + "type": "number", + "minimum": 0.0, + "x-entity": { + "type": "float" + } + }, + "height": { + "description": "Pressure scale-height in physical units", + "type": "number", + "exclusiveMinimum": 0.0, + "x-entity": { + "type": "float" + } + }, + "species": { + "description": "Species indices of particles that populate the atmosphere", + "type": "array", + "minItems": 2, + "maxItems": 2, + "prefixItems": [ + { + "type": "integer", + "minimum": 1 + }, + { + "type": "integer", + "minimum": 1 + } + ], + "x-entity": { + "type": "array [size 2]" + } + }, + "ds": { + "description": "Distance from the edge to which the gravity is imposed in physical units", + "type": "number", + "minimum": 0.0, + "default": 0.0, + "x-entity": { + "type": "float", + "notes": [ + "0.0 means no limit" + ] + } + } + } + } + } + } + } + }, + "scales": { + "description": "Fiducial scales that fix the code unit system", + "type": "object", + "additionalProperties": false, + "required": [ + "larmor0", + "skindepth0" + ], + "x-entity": { + "inferred": [ + { + "name": "dx0", + "brief": "fiducial minimum size of the cell", + "type": "float", + "from": "`grid`" + }, + { + "name": "V0", + "brief": "fiducial elementary volume", + "type": "float", + "from": "`grid`" + }, + { + "name": "n0", + "brief": "Fiducial number density", + "type": "float", + "from": "`particles.ppc0`, `grid`", + "value": "`ppc0 / V0`" + }, + { + "name": "q0", + "brief": "Fiducial elementary charge", + "type": "float", + "from": "`scales.skindepth0`, `scales.n0`", + "value": "`1 / (n0 * skindepth0^2)`" + }, + { + "name": "sigma0", + "brief": "Fiducial magnetization parameter", + "type": "float", + "from": "`scales.larmor0`, `scales.skindepth0`", + "value": "`(skindepth0 / larmor0)^2`" + }, + { + "name": "B0", + "brief": "Fiducial magnetic field", + "type": "float", + "from": "`scales.larmor0`", + "value": "`1 / larmor0`" + }, + { + "name": "omegaB0", + "brief": "Fiducial cyclotron frequency", + "type": "float", + "from": "`scales.larmor0`", + "value": "`1 / larmor0`" + } + ] + }, + "properties": { + "larmor0": { + "description": "Fiducial larmor radius", + "type": "number", + "exclusiveMinimum": 0.0, + "x-entity": { + "type": "float [> 0.0]" + } + }, + "skindepth0": { + "description": "Fiducial plasma skin depth", + "type": "number", + "exclusiveMinimum": 0.0, + "x-entity": { + "type": "float [> 0.0]" + } + } + } + }, + "radiation": { + "description": "Radiative drag and photon emission parameters", + "type": "object", + "additionalProperties": false, + "properties": { + "drag": { + "description": "Radiation reaction (drag) parameters", + "type": "object", + "additionalProperties": false, + "properties": { + "synchrotron": { + "description": "Synchrotron drag parameters", + "type": "object", + "additionalProperties": false, + "properties": { + "gamma_rad": { + "description": "Radiation reaction limit gamma-factor for synchrotron", + "type": "number", + "exclusiveMinimum": 0.0, + "default": 1.0, + "x-entity": { + "type": "float [> 0.0]", + "notes": [ + "[required] if one of the species has `radiative_drag = \"synchrotron\"`" + ] + } + } + } + }, + "compton": { + "description": "Compton drag parameters", + "type": "object", + "additionalProperties": false, + "properties": { + "gamma_rad": { + "description": "Radiation reaction limit gamma-factor for Compton drag", + "type": "number", + "exclusiveMinimum": 0.0, + "default": 1.0, + "x-entity": { + "type": "float [> 0.0]", + "notes": [ + "[required] if one of the species has `radiative_drag = \"compton\"`" + ] + } + } + } + } + } + }, + "emission": { + "description": "Photon emission parameters", + "type": "object", + "additionalProperties": false, + "properties": { + "synchrotron": { + "description": "Synchrotron emission parameters", + "type": "object", + "additionalProperties": false, + "required": [ + "photon_species" + ], + "x-entity": { + "inferred": [ + { + "name": "nominal_probability", + "brief": "Nominal probability of the emission for a particle with `gamma * beta = 1`, charge-to-mass = `q0 / m0`", + "type": "float", + "from": "`.gamma_qed`, `.photon_weight`, `...drag.synchrotron.gamma_rad`, `scales.omegaB0`, `algorithms.timestep.dt`", + "value": "`0.1 * omegaB0 * dt * (gamma_qed / gamma_rad)^2 / photon_weight`" + }, + { + "name": "nominal_photon_energy", + "brief": "Nominal energy of the emitted photon for a particle with `gamma * beta = 1`, mass = `m0`", + "type": "float", + "from": "`.gamma_qed`", + "value": "`(1 / gamma_qed)^2`" + } + ] + }, + "properties": { + "gamma_qed": { + "description": "Gamma-factor of a particle emitting synchrotron photons at energy `m0 c^2` in fiducial magnetic field `B0`", + "type": "number", + "exclusiveMinimum": 1.0, + "default": 10.0, + "x-entity": { + "type": "float [> 1.0]" + } + }, + "photon_energy_min": { + "description": "Minimum photon energy for synchrotron emission (units of `m0 c^2`)", + "type": "number", + "exclusiveMinimum": 0.0, + "default": 0.001, + "x-entity": { + "type": "float [> 0.0]", + "default": "1e-3" + } + }, + "photon_weight": { + "description": "Weights for the emitted synchrotron photons", + "type": "number", + "exclusiveMinimum": 0.0, + "default": 1.0, + "x-entity": { + "type": "float [> 0.0]" + } + }, + "photon_species": { + "description": "Index of species for the emitted photon", + "type": "integer", + "minimum": 1, + "x-entity": { + "type": "ushort [> 0]" + } + } + } + }, + "compton": { + "description": "Inverse Compton emission parameters", + "type": "object", + "additionalProperties": false, + "required": [ + "photon_species" + ], + "x-entity": { + "inferred": [ + { + "name": "nominal_probability", + "brief": "Nominal probability of the emission for a particle with `gamma * beta = 1`, charge-to-mass = `q0 / m0`", + "type": "float", + "from": "`.gamma_qed`, `.photon_weight`, `...drag.compton.gamma_rad`, `scales.omegaB0`, `algorithms.timestep.dt`", + "value": "`0.1 * omegaB0 * dt * (gamma_qed / gamma_rad)^2 / photon_weight`" + }, + { + "name": "nominal_photon_energy", + "brief": "Nominal energy of the emitted photon for a particle with `gamma * beta = 1`, mass = `m0`", + "type": "float", + "from": "`.gamma_qed`", + "value": "`(1 / gamma_qed)^2`" + } + ] + }, + "properties": { + "gamma_qed": { + "description": "Gamma-factor of a particle emitting inverse Compton photons at energy `m0 c^2` in fiducial magnetic field `B0`", + "type": "number", + "exclusiveMinimum": 1.0, + "default": 10.0, + "x-entity": { + "type": "float [> 1.0]" + } + }, + "photon_energy_min": { + "description": "Minimum photon energy for inverse Compton emission (units of `m0 c^2`)", + "type": "number", + "exclusiveMinimum": 0.0, + "default": 0.001, + "x-entity": { + "type": "float [> 0.0]", + "default": "1e-3" + } + }, + "photon_weight": { + "description": "Weights for the emitted inverse Compton photons", + "type": "number", + "exclusiveMinimum": 0.0, + "default": 1.0, + "x-entity": { + "type": "float [> 0.0]" + } + }, + "photon_species": { + "description": "Index of species for the emitted photon", + "type": "integer", + "minimum": 1, + "x-entity": { + "type": "ushort [> 0]" + } + } + } + } + } + } + } + }, + "algorithms": { + "description": "Algorithm and solver tuning", + "type": "object", + "additionalProperties": false, + "properties": { + "current_filters": { + "description": "Number of current smoothing passes", + "type": "integer", + "minimum": 0, + "default": 0, + "x-entity": { + "type": "ushort [>= 0]" + } + }, + "timestep": { + "description": "Timestep parameters", + "type": "object", + "additionalProperties": false, + "x-entity": { + "inferred": [ + { + "name": "dt", + "brief": "timestep duration", + "type": "float", + "from": "`algorithms.timestep.CFL`, `scales.dx0`", + "value": "`CFL * dx0`" + } + ] + }, + "properties": { + "CFL": { + "description": "Courant-Friedrichs-Lewy number", + "type": "number", + "exclusiveMinimum": 0.0, + "maximum": 1.0, + "default": 0.95, + "x-entity": { + "type": "float [0.0 -> 1.0]", + "notes": [ + "CFL number determines the timestep duration" + ] + } + }, + "correction": { + "description": "Correction factor for the speed of light used in field solver", + "type": "number", + "default": 1.0, + "x-entity": { + "type": "float" + } + } + } + }, + "deposit": { + "description": "Current deposition parameters", + "type": "object", + "additionalProperties": false, + "x-entity": { + "inferred": [ + { + "name": "order", + "brief": "order of the particle shape function", + "type": "ushort [0 -> 10]", + "from": "compile-time definition `shape_order`" + } + ] + }, + "properties": { + "enable": { + "description": "Enable the current deposition", + "type": "boolean", + "default": true, + "x-entity": { + "type": "bool" + } + }, + "team_policy_team_size": { + "description": "Tiled-deposit work-group (team) size", + "type": "integer", + "minimum": 0, + "default": 0, + "deprecated": true, + "x-entity": { + "type": "uint [>= 0]", + "deprecated": "removed in 1.6+, use `tiled_deposit_team_size` instead" + } + }, + "tiled_deposit_team_size": { + "description": "Tiled-deposit work-group (team) size", + "type": "integer", + "minimum": 0, + "default": 0, + "x-entity": { + "type": "uint [>= 0]", + "notes": [ + "0 keeps Kokkos::AUTO (backend occupancy heuristic); a positive value overrides it, clamped to the backend/scratch maximum at launch. Only used in `tiled_deposit=ON` builds. Pick a multiple of the device subgroup width for best occupancy (see ideal_tile_size.py)" + ] + } + } + } + }, + "gr": { + "description": "GR pusher parameters", + "type": "object", + "additionalProperties": false, + "properties": { + "pusher_eps": { + "description": "Stepsize for numerical differentiation in GR pusher", + "type": "number", + "exclusiveMinimum": 0.0, + "default": 1e-06, + "x-entity": { + "type": "float [> 0.0]", + "default": "1e-6" + } + }, + "pusher_niter": { + "description": "Number of iterations for the Newton-Raphson method in GR pusher", + "type": "integer", + "minimum": 1, + "default": 10, + "x-entity": { + "type": "ushort [> 0]" + } + } + } + }, + "gca": { + "description": "Guiding-center approximation parameters", + "type": "object", + "additionalProperties": false, + "properties": { + "e_ovr_b_max": { + "description": "Maximum value for E/B allowed for GCA particles", + "type": "number", + "minimum": 0.0, + "maximum": 1.0, + "default": 0.9, + "x-entity": { + "type": "float [0.0 -> 1.0]" + } + }, + "larmor_max": { + "description": "Maximum Larmor radius allowed for GCA particles (in physical units)", + "type": "number", + "minimum": 0.0, + "default": 0.0, + "x-entity": { + "type": "float", + "notes": [ + "When `larmor_max` == 0, the limit is disabled" + ] + } + } + } + }, + "fieldsolver": { + "description": "Stencil coefficients for the field solver [notation as in Blinne+ (2018)]", + "type": "object", + "additionalProperties": false, + "x-entity": { + "notes": [ + "Standard Yee solver: `delta_i = beta_ij = 0.0`" + ] + }, + "properties": { + "enable": { + "description": "Enable the fieldsolver", + "type": "boolean", + "default": true, + "x-entity": { + "type": "bool" + } + }, + "delta_x": { + "description": "delta_x coefficient (for `F_{i +/- 3/2, j, k}`)", + "type": "number", + "default": 0.0, + "x-entity": { + "type": "float" + } + }, + "delta_y": { + "description": "delta_y coefficient (for `F_{i, j +/- 3/2, k}`)", + "type": "number", + "default": 0.0, + "x-entity": { + "type": "float", + "notes": [ + "Used only for 2D and 3D" + ] + } + }, + "delta_z": { + "description": "delta_z coefficient (for `F_{i, j, k +/- 3/2}`)", + "type": "number", + "default": 0.0, + "x-entity": { + "type": "float", + "notes": [ + "Used only for 3D" + ] + } + }, + "beta_xy": { + "description": "beta_xy coefficient (for `F_{i +/- 1/2, j +/- 1, k}`)", + "type": "number", + "default": 0.0, + "x-entity": { + "type": "float", + "notes": [ + "Used only for 2D and 3D" + ] + } + }, + "beta_yx": { + "description": "beta_yx coefficient (for `F_{i +/- 1, j +/- 1/2, k}`)", + "type": "number", + "default": 0.0, + "x-entity": { + "type": "float", + "notes": [ + "Used only for 2D and 3D" + ] + } + }, + "beta_xz": { + "description": "beta_xz coefficient (for `F_{i +/- 1/2, j, k +/- 1}`)", + "type": "number", + "default": 0.0, + "x-entity": { + "type": "float", + "notes": [ + "Used only for 3D" + ] + } + }, + "beta_zx": { + "description": "beta_zx coefficient (for `F_{i +/- 1, j, k +/- 1/2}`)", + "type": "number", + "default": 0.0, + "x-entity": { + "type": "float", + "notes": [ + "Used only for 3D" + ] + } + }, + "beta_yz": { + "description": "beta_yz coefficient (for `F_{i, j +/- 1/2, k +/- 1}`)", + "type": "number", + "default": 0.0, + "x-entity": { + "type": "float", + "notes": [ + "Used only for 3D" + ] + } + }, + "beta_zy": { + "description": "beta_zy coefficient (for `F_{i, j +/- 1, k +/- 1/2}`)", + "type": "number", + "default": 0.0, + "x-entity": { + "type": "float", + "notes": [ + "Used only for 3D" + ] + } + } + } + } + } + }, + "particles": { + "description": "Particle and species parameters", + "type": "object", + "additionalProperties": false, + "required": [ + "ppc0" + ], + "x-entity": { + "inferred": [ + { + "name": "nspec", + "brief": "Number of particle species", + "type": "uint", + "from": "`particles.species`" + } + ] + }, + "properties": { + "ppc0": { + "description": "Fiducial number of particles per cell", + "type": "number", + "exclusiveMinimum": 0.0, + "x-entity": { + "type": "float [> 0.0]" + } + }, + "use_weights": { + "description": "Toggle for using particle weights", + "type": "boolean", + "default": false, + "x-entity": { + "type": "bool" + } + }, + "clear_interval": { + "description": "Timesteps between particle re-sorting by tags (removing dead particles)", + "type": "integer", + "minimum": 0, + "default": 100, + "x-entity": { + "type": "uint", + "notes": [ + "Set to 0 to disable re-sorting" + ] + } + }, + "spatial_sorting_interval": { + "description": "Timesteps between spatial sorting of particles (for better cache performance)", + "type": "integer", + "minimum": 0, + "default": 0, + "x-entity": { + "type": "uint", + "notes": [ + "Set to 0 to disable spatial sorting" + ] + } + }, + "species": { + "description": "Particle species definitions", + "type": "array", + "minItems": 1, + "x-entity": { + "array_of_tables": true + }, + "items": { + "type": "object", + "additionalProperties": false, + "required": [ + "mass", + "charge", + "maxnpart" + ], + "properties": { + "label": { + "description": "Label of the species", + "type": "string", + "x-entity": { + "type": "string", + "default": "\"s\"", + "notes": [ + "`` is the index of the species in the list starting from 1" + ], + "examples": [ + "\"e-\"" + ] + } + }, + "mass": { + "description": "Mass of the species (in units of fiducial mass)", + "type": "number", + "minimum": 0.0, + "x-entity": { + "type": "float [>= 0.0]" + } + }, + "charge": { + "description": "Charge of the species (in units of fiducial charge)", + "type": "number", + "x-entity": { + "type": "float" + } + }, + "maxnpart": { + "description": "Maximum number of particles per task", + "type": "number", + "x-entity": { + "type": "uint [> 0]", + "notes": [ + "Read as a float, so exponential notation is fine (e.g. `1e8`)" + ] + }, + "exclusiveMinimum": 0 + }, + "pusher": { + "description": "Pusher algorithm for the species", + "anyOf": [ + { + "enum": [ + "Boris", + "Vay", + "Boris,GCA", + "Vay,GCA", + "Photon", + "None" + ] + }, + { + "type": "string", + "pattern": "(?i)^(Boris|Vay|Boris,GCA|Vay,GCA|Photon|None)$" + } + ], + "x-entity": { + "type": "string", + "default": "\"Boris\" [massive]; \"Photon\" [massless]" + } + }, + "n_payloads_real": { + "description": "Number of additional real-valued variables (payloads) for each particle of the given species", + "type": "integer", + "minimum": 0, + "default": 0, + "x-entity": { + "type": "ushort" + } + }, + "n_payloads_int": { + "description": "Number of additional integer-valued variables (payloads) for each particle of the given species", + "type": "integer", + "minimum": 0, + "default": 0, + "x-entity": { + "type": "ushort", + "notes": [ + "If tracking is enabled, one or two extra integer payloads are reserved (depending on whether MPI is enabled)" + ] + } + }, + "tracking": { + "description": "Enable tracking of particles using indices for the given species", + "type": "boolean", + "default": false, + "x-entity": { + "type": "bool" + } + }, + "radiative_drag": { + "description": "Radiation reaction to use for the species", + "type": "string", + "pattern": "(?i)^(None|(Synchrotron|Compton)(,(Synchrotron|Compton))*)$", + "default": "None", + "x-entity": { + "type": "string", + "enum": [ + "None", + "Synchrotron", + "Compton" + ], + "notes": [ + "Can also be coma-separated combination, e.g., \"Synchrotron,Compton\"", + "Relevant radiation.drag parameters should also be provided" + ] + } + }, + "emission": { + "description": "Particle emission policy for the species", + "anyOf": [ + { + "enum": [ + "None", + "Synchrotron", + "Compton", + "Custom" + ] + }, + { + "type": "string", + "pattern": "(?i)^(None|Synchrotron|Compton|Custom)$" + } + ], + "default": "None", + "x-entity": { + "type": "string", + "notes": [ + "Only one emission mechanism allowed", + "Appropriate radiation drag flag will be applied automatically (unless explicitly set to \"None\")" + ] + } + }, + "spatial_sorting_interval": { + "description": "Timesteps between spatial sorting of particles for given species", + "type": "integer", + "minimum": 0, + "default": 0, + "x-entity": { + "type": "uint", + "notes": [ + "Set to 0 to disable spatial sorting", + "Overrides `particles.spatial_sorting_interval` for the given species" + ] + } + }, + "clear_interval": { + "description": "Timesteps between particle re-sorting by tags (removing dead particles)", + "type": "integer", + "minimum": 0, + "default": 100, + "x-entity": { + "type": "uint", + "notes": [ + "Set to 0 to disable re-sorting", + "Overrides `particles.clear_interval` for the given species" + ] + } + } + } + } + } + } + }, + "setup": { + "description": "Parameters for specific problem generators and setups", + "type": "object", + "additionalProperties": true, + "x-entity": { + "notes": [ + "Free-form: keys are defined by the problem generator, so nothing here is validated" + ] + } + }, + "output": { + "description": "Output parameters", + "type": "object", + "additionalProperties": false, + "properties": { + "format": { + "description": "Output format", + "anyOf": [ + { + "enum": [ + "disabled", + "hdf5", + "BPFile" + ] + }, + { + "type": "string", + "pattern": "(?i)^(disabled|hdf5|BPFile)$" + } + ], + "default": "bpfile", + "x-entity": { + "type": "string" + } + }, + "interval": { + "description": "Number of timesteps between all outputs", + "type": "integer", + "minimum": 1, + "default": 100, + "x-entity": { + "type": "uint [> 0]", + "notes": [ + "Value is overriden by output intervals for specific outputs" + ] + } + }, + "interval_time": { + "description": "Physical (code) time interval between all outputs", + "type": "number", + "default": -1.0, + "x-entity": { + "type": "float", + "notes": [ + "When `interval_time` < 0, the output is controlled by `interval`, otherwise by `interval_time`", + "Value is overriden by output intervals for specific outputs" + ] + } + }, + "fields": { + "description": "Field output parameters", + "type": "object", + "additionalProperties": false, + "properties": { + "enable": { + "description": "Toggle for the field output", + "type": "boolean", + "default": true, + "x-entity": { + "type": "bool" + } + }, + "quantities": { + "description": "Field quantities to output", + "type": "array", + "items": { + "type": "string" + }, + "default": [], + "x-entity": { + "type": "array", + "enum": [ + "E", + "B", + "J", + "divE", + "Rho", + "Charge", + "N", + "Nppc", + "T0i", + "Tij", + "Vi", + "D", + "H", + "divD", + "A" + ], + "notes": [ + "For `T`, you can use unspecified indices: `Tij`, `T0i`, or specific ones: `Ttt`, `T00`, `T02`, `T23`", + "For `T`, in cartesian can also use \"x\" \"y\" \"z\" instead of \"1\" \"2\" \"3\"", + "By default, we accumulate moments from all massive species, one can specify only specific species: `Ttt_1_2`, `Rho_1`, `Rho_3_4`" + ] + } + }, + "custom": { + "description": "Custom (user-defined) field quantities", + "type": "array", + "items": { + "type": "string" + }, + "default": [], + "x-entity": { + "type": "array" + } + }, + "interval": { + "description": "Number of timesteps between field outputs", + "type": "integer", + "minimum": 0, + "default": 0, + "x-entity": { + "type": "uint", + "notes": [ + "When `!= 0`, overrides `output.interval`", + "When `== 0`, `output.interval` is used" + ] + } + }, + "interval_time": { + "description": "Physical (code) time interval between field outputs", + "type": "number", + "default": -1.0, + "x-entity": { + "type": "float", + "notes": [ + "When `< 0`, the output is controlled by `interval`", + "When specified, overrides `output.interval_time`" + ] + } + }, + "downsampling": { + "description": "Downsample factor for the output of fields", + "anyOf": [ + { + "type": "integer", + "minimum": 1 + }, + { + "type": "array", + "minItems": 1, + "maxItems": 3, + "items": { + "type": "integer", + "minimum": 1 + } + } + ], + "default": [ + 1, + 1, + 1 + ], + "x-entity": { + "type": "uint | array [>= 1]", + "notes": [ + "The output is downsampled by the given factors in each direction", + "If a scalar is given, it is applied to all directions" + ] + } + }, + "smoothing": { + "description": "Smoothing of the output moments", + "type": "object", + "additionalProperties": false, + "properties": { + "order": { + "description": "Smoothing order for the output of moments (\"Rho\", \"Charge\", \"T\", ...)", + "type": "integer", + "minimum": 0, + "default": 0, + "x-entity": { + "type": "ushort" + } + }, + "method": { + "description": "Smoothing algorithm", + "anyOf": [ + { + "enum": [ + "const", + "spline" + ] + }, + { + "type": "string", + "pattern": "(?i)^(const|spline)$" + } + ], + "default": "spline", + "x-entity": { + "type": "string", + "notes": [ + "When using \"spline\", `order` corresponds to the order of the polynomial used", + "When using \"const\", the smoothing window is `ceil(order / 2)` in both directions" + ] + } + } + } + } + } + }, + "particles": { + "description": "Particle output parameters", + "type": "object", + "additionalProperties": false, + "properties": { + "enable": { + "description": "Toggle for the particles output", + "type": "boolean", + "default": true, + "x-entity": { + "type": "bool" + } + }, + "species": { + "description": "Particle species indices to output", + "type": "array", + "items": { + "type": "integer", + "minimum": 1 + }, + "default": [], + "x-entity": { + "type": "array", + "notes": [ + "If empty, all species are output" + ] + } + }, + "stride": { + "description": "Stride for the output of particles", + "type": "integer", + "default": 100, + "x-entity": { + "type": "uint [>= 1]" + }, + "minimum": 1 + }, + "interval": { + "description": "Number of timesteps between particle outputs", + "type": "integer", + "minimum": 0, + "default": 0, + "x-entity": { + "type": "uint", + "notes": [ + "When `!= 0`, overrides `output.interval`", + "When `== 0`, `output.interval` is used" + ] + } + }, + "interval_time": { + "description": "Physical (code) time interval between particle outputs", + "type": "number", + "default": -1.0, + "x-entity": { + "type": "float", + "notes": [ + "When `< 0`, the output is controlled by `interval`", + "When specified, overrides `output.interval_time`" + ] + } + } + } + }, + "spectra": { + "description": "Spectra output parameters", + "type": "object", + "additionalProperties": false, + "properties": { + "enable": { + "description": "Toggle for the spectra output", + "type": "boolean", + "default": true, + "x-entity": { + "type": "bool" + } + }, + "e_min": { + "description": "Minimum energy for the spectra output", + "type": "number", + "exclusiveMinimum": 0.0, + "default": 0.001, + "x-entity": { + "type": "float", + "default": "1e-3" + } + }, + "e_max": { + "description": "Maximum energy for the spectra output", + "type": "number", + "exclusiveMinimum": 0.0, + "default": 1000.0, + "x-entity": { + "type": "float", + "default": "1e3" + } + }, + "log_bins": { + "description": "Whether to use logarithmic bins for energy", + "type": "boolean", + "default": true, + "x-entity": { + "type": "bool" + } + }, + "n_bins": { + "description": "Number of energy bins for the spectra output", + "type": "integer", + "minimum": 1, + "default": 200, + "deprecated": true, + "x-entity": { + "type": "uint [> 0]", + "deprecated": "removed in 1.6+, use `num_energy_bins` instead" + } + }, + "num_energy_bins": { + "description": "Number of energy bins for the spectra output", + "type": "integer", + "minimum": 1, + "default": 200, + "x-entity": { + "type": "uint [> 0]" + } + }, + "num_spatial_bins": { + "description": "Number of spatial bins for the spectra output", + "type": "array", + "minItems": 1, + "maxItems": 3, + "items": { + "type": "integer", + "minimum": 1 + }, + "default": [ + 1, + 1, + 1 + ], + "x-entity": { + "type": "array [size 1 :->: 3]" + } + }, + "interval": { + "description": "Number of timesteps between spectra outputs", + "type": "integer", + "minimum": 0, + "default": 0, + "x-entity": { + "type": "uint", + "notes": [ + "When `!= 0`, overrides `output.interval`", + "When `== 0`, `output.interval` is used" + ] + } + }, + "interval_time": { + "description": "Physical (code) time interval between spectra outputs", + "type": "number", + "default": -1.0, + "x-entity": { + "type": "float", + "notes": [ + "When `< 0`, the output is controlled by `interval`", + "When specified, overrides `output.interval_time`" + ] + } + } + } + }, + "debug": { + "description": "Debug output parameters", + "type": "object", + "additionalProperties": false, + "properties": { + "as_is": { + "description": "Output fields \"as is\" without conversions", + "type": "boolean", + "default": false, + "x-entity": { + "type": "bool" + } + }, + "ghosts": { + "description": "Output fields with values in ghost cells", + "type": "boolean", + "default": false, + "x-entity": { + "type": "bool" + } + } + } + }, + "stats": { + "description": "Integrated statistics output parameters", + "type": "object", + "additionalProperties": false, + "properties": { + "enable": { + "description": "Toggle for the stats output", + "type": "boolean", + "default": true, + "x-entity": { + "type": "bool" + } + }, + "interval": { + "description": "Number of timesteps between stat outputs", + "type": "integer", + "minimum": 1, + "default": 100, + "x-entity": { + "type": "uint [> 0]", + "notes": [ + "Overriden if `output.stats.interval_time != -1`" + ] + } + }, + "interval_time": { + "description": "Physical (code) time interval between stat outputs", + "type": "number", + "default": -1.0, + "x-entity": { + "type": "float", + "notes": [ + "When `< 0`, the output is controlled by `interval`" + ] + } + }, + "quantities": { + "description": "Field quantities to output", + "type": "array", + "items": { + "type": "string" + }, + "default": [ + "B^2", + "E^2", + "ExB", + "Rho", + "T00" + ], + "x-entity": { + "type": "array", + "enum": [ + "B^2", + "E^2", + "ExB", + "N", + "Npart", + "Charge", + "Rho", + "T00", + "T0i", + "Tij" + ], + "notes": [ + "For particle moments, ...", + "... same notation is used as for `output.fields.quantities`" + ] + } + }, + "custom": { + "description": "Custom (user-defined) stats", + "type": "array", + "items": { + "type": "string" + }, + "default": [], + "x-entity": { + "type": "array" + } + } + } + } + } + }, + "checkpoint": { + "description": "Checkpointing parameters", + "type": "object", + "additionalProperties": false, + "x-entity": { + "inferred": [ + { + "name": "is_resuming", + "brief": "Whether the simulation is resuming from a checkpoint", + "type": "bool", + "from": "command-line flag" + }, + { + "name": "start_step", + "brief": "Timestep of the checkpoint used to resume", + "type": "uint", + "from": "automatically determined during restart" + }, + { + "name": "start_time", + "brief": "Time of the checkpoint used to resume", + "type": "float", + "from": "automatically determined during restart" + } + ] + }, + "properties": { + "interval": { + "description": "Number of timesteps between checkpoints", + "type": "integer", + "minimum": 1, + "default": 1000, + "x-entity": { + "type": "uint [> 0]" + } + }, + "interval_time": { + "description": "Physical (code) time interval between checkpoints", + "type": "number", + "default": -1.0, + "x-entity": { + "type": "float [> 0]", + "notes": [ + "When `< 0`, the output is controlled by `interval`" + ] + } + }, + "keep": { + "description": "Number of checkpoints to keep", + "type": "integer", + "minimum": -1, + "default": 2, + "x-entity": { + "type": "int", + "notes": [ + "0 = disable checkpointing", + "-1 = keep all checkpoints" + ] + } + }, + "walltime": { + "description": "Write a checkpoint once after a fixed walltime", + "type": "string", + "pattern": "^$|^[0-9]{2,}:[0-9]{2}:[0-9]{2}$", + "default": "00:00:00", + "x-entity": { + "type": "string", + "notes": [ + "The format is \"HH:MM:SS\"", + "Empty string or \"00:00:00\" disables this functionality", + "Writing checkpoint at walltime does not stop the simulation" + ] + } + }, + "write_path": { + "description": "Parent directory to write checkpoints to", + "type": "string", + "x-entity": { + "type": "string", + "default": "`.ckpt`", + "notes": [ + "The directory is created if it does not exist" + ] + } + }, + "read_path": { + "description": "Parent directory to use when resuming from a checkpoint", + "type": "string", + "x-entity": { + "type": "string", + "default": "inherit `write_path`" + } + } + } + }, + "adios2": { + "description": "ADIOS2 BP5 tuning, applied to both [output] and [checkpoint] writers", + "type": "object", + "additionalProperties": false, + "properties": { + "aggregators_per_node": { + "description": "Number of ADIOS2 aggregators per node", + "type": "integer", + "minimum": 0, + "default": 0, + "x-entity": { + "type": "uint", + "notes": [ + "Set to either MPI ranks/node or NICs/node for best performance\nIf set to 0, will use ADIOS2 default (one aggregator per node)" + ] + } + }, + "max_shm_size": { + "description": "Maximum shared-memory segment size per node, in bytes (BP5 MaxShmSize)", + "type": "integer", + "minimum": 0, + "default": 4294967296, + "x-entity": { + "type": "uint", + "notes": [ + "Lower this on memory-constrained nodes; matches ADIOS2's default" + ] + } + }, + "buffer_chunk_size": { + "description": "Internal serialization buffer chunk size, in bytes (BP5 BufferChunkSize)", + "type": "integer", + "minimum": 0, + "default": 16777216, + "x-entity": { + "type": "uint", + "notes": [ + "Scales with per-rank output volume; matches ADIOS2's default" + ] + } + } + } + }, + "render": { + "description": "In-situ renderer. Renders scalar fields on the GPU and writes PNG images directly to `/renders/` each cadence -- no field data is written to storage, and the result is seamless across MPI domain boundaries.", + "type": "object", + "additionalProperties": false, + "x-entity": { + "notes": [ + "two modes, selected automatically by the simulation dimension:\n- 3D Cartesian (Minkowski): volume ray-march (uses `samples`, `step_size`, `early_term_alpha`, and the [camera] table)\n- 2D (Minkowski, Spherical/QSpherical, and all GR Kerr-Schild): flat slice rasterizer. Cartesian shows the (x, y) plane; spherical shows the meridional (r, theta) half-plane mapped to Cartesian (X = r sin th, Z = r cos th), optionally mirrored (see `mirror`). The `samples`/`step_size`/`early_term_alpha`/[camera] keys are ignored in 2D (one opaque sample per pixel).", + "1D (and 3D non-Cartesian, which does not exist) is a no-op", + "One PNG stream per scene (e.g. a density/|B|/|J| triptych)" + ] + }, + "properties": { + "enable": { + "description": "Toggle for the on-the-fly renderer", + "type": "boolean", + "default": false, + "x-entity": { + "type": "bool" + } + }, + "interval": { + "description": "Number of timesteps between renders", + "type": "integer", + "minimum": 0, + "default": 0, + "x-entity": { + "type": "uint", + "notes": [ + "When `!= 0`, overrides `output.interval`", + "When `== 0`, `interval_time` (or `output.interval`) is used" + ] + } + }, + "interval_time": { + "description": "Physical (code) time interval between renders", + "type": "number", + "default": -1.0, + "x-entity": { + "type": "float", + "notes": [ + "When `< 0`, the output is controlled by `interval`" + ] + } + }, + "width": { + "description": "Image width in pixels (the rendered region; the PNG is wider if a colorbar margin is added, see `colorbar_outside`)", + "type": "integer", + "minimum": 1, + "default": 1024, + "x-entity": { + "type": "int [> 0]" + } + }, + "height": { + "description": "Image height in pixels", + "type": "integer", + "minimum": 1, + "default": 1024, + "x-entity": { + "type": "int [> 0]" + } + }, + "resolution": { + "description": "Convenience: force a square frame (sets width == height == resolution), the natural shape for a dome master. Overrides `width`/`height` when > 0.", + "type": "integer", + "minimum": 0, + "default": 0, + "x-entity": { + "type": "int [> 0]", + "default": "0 (use width/height)" + } + }, + "extent": { + "description": "Limit the render region to axis-aligned box in physical/world coordinates. Left unset it spans the full domain. Clamped to the box.", + "type": "object", + "additionalProperties": false, + "x-entity": { + "notes": [ + "3D -> the volume is depth-clipped to this box, the wireframe/axes frame it, and the default camera zooms to it; 2D -> the slice window is framed to it" + ] + }, + "properties": { + "x1": { + "description": "Axis-aligned render region [lo, hi] along x1, in physical/world coords. Left unset it spans the full domain. Clamped to the box.", + "anyOf": [ + { + "type": "array", + "maxItems": 0 + }, + { + "type": "array", + "minItems": 2, + "maxItems": 2, + "prefixItems": [ + { + "type": "number" + }, + { + "type": "number" + } + ] + } + ], + "default": [], + "x-entity": { + "type": "array [size 2]", + "default": "[] (full extent)", + "notes": [ + "For a spherical 2D slice, x1 crops the radius r." + ], + "examples": [ + "x1 = [-64.0, 64.0]" + ] + } + }, + "x2": { + "description": "Render region [lo, hi] along x2.", + "anyOf": [ + { + "type": "array", + "maxItems": 0 + }, + { + "type": "array", + "minItems": 2, + "maxItems": 2, + "prefixItems": [ + { + "type": "number" + }, + { + "type": "number" + } + ] + } + ], + "default": [], + "x-entity": { + "type": "array [size 2]", + "default": "[] (full extent)", + "notes": [ + "For a spherical 2D slice, x2 crops the polar angle theta." + ] + } + }, + "x3": { + "description": "Render region [lo, hi] along x3.", + "anyOf": [ + { + "type": "array", + "maxItems": 0 + }, + { + "type": "array", + "minItems": 2, + "maxItems": 2, + "prefixItems": [ + { + "type": "number" + }, + { + "type": "number" + } + ] + } + ], + "default": [], + "x-entity": { + "type": "array [size 2]", + "default": "[] (full extent)" + } + } + } + }, + "volume": { + "description": "Volume rendering parameters (3D Cartesian only)", + "type": "object", + "additionalProperties": false, + "properties": { + "samples": { + "description": "Number of ray-march steps across the global box diagonal", + "type": "integer", + "minimum": 1, + "default": 400, + "x-entity": { + "type": "int [> 0]", + "notes": [ + "The world-space step is `box_diagonal / samples` unless `step_size` is set. Higher = better quality, slower." + ] + } + }, + "step_size": { + "description": "Fixed world-space step between ray samples", + "type": "number", + "minimum": 0.0, + "default": 0.0, + "x-entity": { + "type": "float [>= 0.0]", + "notes": [ + "0 derives the step from `samples`. The step is identical on all ranks, which is what makes the multi-domain composite seamless." + ] + } + }, + "early_term_alpha": { + "description": "Stop marching a ray once its accumulated opacity reaches this value", + "type": "number", + "minimum": 0.0, + "maximum": 1.0, + "default": 0.99, + "x-entity": { + "type": "float [0.0 -> 1.0]", + "notes": [ + "Pure speed optimization; set to 1.0 to disable early termination" + ] + } + } + } + }, + "n_lut": { + "description": "Number of entries in the color/opacity lookup table", + "type": "integer", + "exclusiveMinimum": 1, + "default": 256, + "x-entity": { + "type": "int [> 1]" + } + }, + "background": { + "description": "Opaque background RGB (each channel 0..1) shown through transparent/low-opacity pixels; also fills the colorbar margin", + "type": "array", + "minItems": 3, + "maxItems": 3, + "items": { + "type": "number", + "minimum": 0.0, + "maximum": 1.0 + }, + "default": [ + 0.0, + 0.0, + 0.0 + ], + "x-entity": { + "type": "array [size 3]" + } + }, + "colorbar": { + "description": "Draw a colorbar (gradient + value ticks + label) on each PNG", + "type": "boolean", + "default": true, + "x-entity": { + "type": "bool" + } + }, + "colorbar_outside": { + "description": "Draw the colorbar in an added right margin (the PNG becomes wider by a fixed strip) instead of overlaying it on the rendered volume", + "type": "boolean", + "default": true, + "x-entity": { + "type": "bool" + } + }, + "mirror": { + "description": "2D spherical slice only: mirror the meridional half-plane across the symmetry axis to render a full disk from one axisymmetric half. No effect on Cartesian or 3D rendering.", + "type": "boolean", + "default": true, + "x-entity": { + "type": "bool" + } + }, + "time_label": { + "description": "Draw the current simulation time as a label (\"T = \", fixed to 2 decimals) in the upper-right corner of the render region, in a contrasting color, vertically centered between the frame top and the colorbar.", + "type": "boolean", + "default": false, + "x-entity": { + "type": "bool" + } + }, + "axes": { + "description": "Draw a spine (frame) + axis ticks + labels around the rendered region. The PNG gains left/bottom margins (background-filled) for the tick labels and axis names, so they never overlap the data.", + "type": "boolean", + "default": false, + "x-entity": { + "type": "bool", + "notes": [ + "2D Cartesian = a rectangular frame with linear spatial ticks;\n2D spherical = polar axes (an \"R\" radial axis on the symmetry axis with R=0 centered, and a \"Theta\" axis along the curved outline / spine);\n3D = the global box projected to a wireframe with ticks on the three silhouette edges (x bottom, y & z on the left)" + ] + } + }, + "axis_labels": { + "description": "Axis names. 3D uses all three; the 2D slice uses the first two. When unset, the 2D slice defaults to \"x\",\"y\" (Cartesian) or \"X\",\"Z\" (spherical).", + "type": "array", + "maxItems": 3, + "items": { + "type": "string" + }, + "default": [ + "x", + "y", + "z" + ], + "x-entity": { + "type": "array [size <= 3]" + } + }, + "axis_ticks": { + "description": "Target number of ticks per axis (actual count is rounded to nice values)", + "type": "integer", + "minimum": 2, + "default": 5, + "x-entity": { + "type": "int [>= 2]" + } + }, + "spine_width": { + "description": "3D only: target width (pixels) of the box wireframe \"spine\". The spine is drawn inside the ray-march (opaque, depth-occluded by the volume); its width is floored by the ray step, so for a crisper thin line raise `samples` as well.", + "type": "number", + "exclusiveMinimum": 0.0, + "default": 2.0, + "x-entity": { + "type": "float [> 0.0]" + } + }, + "moving_view": { + "description": "Translate the render region (and, in 3D, the camera), to keep a propagating feature (e.g. a shock) in frame. Pair with x{1,2,3} extent to crop the moving window.", + "type": "object", + "additionalProperties": false, + "properties": { + "velocity": { + "description": "Velocity of the moving camera in world units.", + "anyOf": [ + { + "type": "array", + "maxItems": 0 + }, + { + "type": "array", + "minItems": 2, + "maxItems": 3, + "items": { + "type": "number" + } + } + ], + "default": [], + "x-entity": { + "type": "array [size 2 or 3]", + "default": "[] (static view)", + "examples": [ + "velocity = [0.9, 0.0] # pan along +x1 at 0.9 c" + ] + } + }, + "start_time": { + "description": "Sim time at which the view starts moving (static before it, e.g. to let an initial ramp-up finish)", + "type": "number", + "default": 0.0, + "x-entity": { + "type": "float" + } + } + } + }, + "camera": { + "description": "Camera (3D volume mode only; ignored by the 2D slice rasterizer). Defaults frame the whole global box from outside, looking down the (1,1,1) diagonal -- the production setup for which the structured composite is provably seamless.", + "type": "object", + "additionalProperties": false, + "properties": { + "mode": { + "description": "Projection mode. Overrides `orthographic` below when set.", + "anyOf": [ + { + "enum": [ + "orthographic", + "perspective", + "dome" + ] + }, + { + "type": "string", + "pattern": "(?i)^(orthographic|perspective|dome)$" + } + ], + "x-entity": { + "type": "string", + "default": "(unset -> use `orthographic`)", + "notes": [ + "\"dome\" is a fulldome azimuthal-equidistant fisheye rendered from an INTERIOR eye (the domain center by default), i.e. a 3D planetarium dome master. It uses a depth-resolved (A-buffer) composite that is seamless across a full 3D domain decomposition (unlike ortho/perspective, which need the eye outside the box). Set a square frame (`resolution`, or width == height). `forward` is the dome ZENITH (screen-up defaults to +y for a +z zenith)." + ] + } + }, + "position": { + "description": "Camera (eye) position in world (physical) coordinates", + "type": "array", + "minItems": 3, + "maxItems": 3, + "items": { + "type": "number" + }, + "x-entity": { + "type": "array [size 3]", + "default": "box center pushed back ~1.7 box-diagonals along (1, 1, 1);\nfor `mode = \"dome\"`, the domain center (interior eye)" + } + }, + "look_at": { + "description": "Point the camera looks at, in world coordinates (the dome ZENITH target)", + "type": "array", + "minItems": 3, + "maxItems": 3, + "items": { + "type": "number" + }, + "x-entity": { + "type": "array [size 3]", + "default": "box center; for `mode = \"dome\"`, the zenith defaults to +z" + } + }, + "up": { + "description": "Camera up vector (dome: the disk's screen-up)", + "type": "array", + "minItems": 3, + "maxItems": 3, + "items": { + "type": "number" + }, + "default": [ + 0.0, + 0.0, + 1.0 + ], + "x-entity": { + "type": "array [size 3]", + "default": "[0.0, 0.0, 1.0]; for `mode = \"dome\"`, [0.0, 1.0, 0.0]" + } + }, + "fov": { + "description": "Vertical field of view in degrees (perspective only)", + "type": "number", + "exclusiveMinimum": 0.0, + "default": 35.0, + "x-entity": { + "type": "float [> 0.0]" + } + }, + "dome_fov": { + "description": "Full dome field of view in degrees (dome mode only): the image rim is at dome_fov/2 from the zenith (180 = a full hemisphere down to the horizon).", + "type": "number", + "exclusiveMinimum": 0.0, + "maximum": 360.0, + "default": 180.0, + "x-entity": { + "type": "float [> 0.0, <= 360.0]" + } + }, + "dome_radius": { + "description": "Dome far-clip radius in world units (dome mode only): each ray stops this far from the eye, so the sampled region is a half-ball (hemisphere) of this radius rather than the whole box -> uniform path length and no box corner/edge projection artifacts. `samples` then counts steps across this radius.", + "type": "number", + "minimum": 0.0, + "x-entity": { + "type": "float [>= 0.0]", + "default": "the largest sphere centered in the box (half the shortest side), so it touches the face centers and never a corner", + "notes": [ + "0 disables the clip (rays march to the box boundary)" + ] + } + }, + "ortho_height": { + "description": "Vertical extent of the view in world units (orthographic only)", + "type": "number", + "exclusiveMinimum": 0.0, + "x-entity": { + "type": "float [> 0.0]", + "default": "the global box diagonal (the whole box fits from any angle)" + } + } + } + }, + "dome": { + "description": "Fulldome fisheye (\"planetarium dome master\"). 2D only; a circular image is centered in the frame's inscribed circle with the corners left as the background (the dome master's black border). Set `width == height` (e.g. 4096) for a square master. When enabled, the axes and the outside colorbar strip are suppressed so the PNG stays exactly width x height. Seamless across MPI domains (the pixel->world map is a shared, deterministic function and the tiles stay disjoint). Ignored (with a warning) for 3D.", + "type": "object", + "additionalProperties": false, + "x-entity": { + "notes": [ + "CARTESIAN -- the flat plane is warped radially into the disk; use `fov`/`radius`/`center`/`projection` below.", + "SPHERICAL / GR Kerr-Schild -- the meridional slice is ALREADY a disk, so dome mode only mirrors it to a full disk (see `mirror` above; keep it true) and fits it to the inscribed circle. The `fov`/`radius`/`center`/`projection` keys are ignored (the native (X, Z) meridional map is used, with image radius proportional to the physical radius r, r=0 at the disk center)." + ] + }, + "properties": { + "enable": { + "description": "Build the fisheye dome master instead of the plain slice", + "type": "boolean", + "default": false, + "x-entity": { + "type": "bool" + } + }, + "fov": { + "description": "(Cartesian only) Full dome field of view in degrees (image radius maps linearly to the dome zenith angle: the rim is at fov/2)", + "type": "number", + "exclusiveMinimum": 0.0, + "maximum": 180.0, + "default": 180.0, + "x-entity": { + "type": "float [> 0.0, <= 180.0]", + "default": "180.0 # a full hemisphere" + } + }, + "radius": { + "description": "(Cartesian only) World radius of the circular cutout mapped onto the dome", + "type": "number", + "exclusiveMinimum": 0.0, + "x-entity": { + "type": "float [> 0.0]", + "default": "half the shorter domain side (the largest centered disk that fits inside the box)" + } + }, + "center": { + "description": "(Cartesian only) World-space center of the cutout", + "type": "array", + "minItems": 2, + "maxItems": 2, + "items": { + "type": "number" + }, + "x-entity": { + "type": "array [size 2]", + "default": "the domain center" + } + }, + "projection": { + "description": "(Cartesian only) How the dome zenith angle maps to a world radius on the flat slice", + "anyOf": [ + { + "enum": [ + "equidistant", + "gnomonic", + "stereographic", + "orthographic" + ] + }, + { + "type": "string", + "pattern": "(?i)^(equidistant|gnomonic|stereographic|orthographic)$" + } + ], + "default": "equidistant", + "x-entity": { + "type": "string", + "enum": [ + "\"equidistant\" (r proportional to angle; the fulldome image standard -- a straight radial scaling of the cutout)", + "\"gnomonic\" (r ~ tan(angle); the slice as a flat \"ceiling\" tangent to the dome -- straight sim lines stay straight)", + "\"stereographic\" (r ~ tan(angle/2); conformal, preserves shapes)", + "\"orthographic\" (r ~ sin(angle); the slice as seen face-on)" + ] + } + } + } + }, + "fieldlines": { + "description": "Magnetic field lines, drawn from a coarse, MPI-replicated copy of the field so the geometry is global and seamless across domains (the coarsening is what makes this cheap -- no parallel particle advection / flux scan).\n- 3D (Cartesian): traced as solid tubes, colored by |field|, composited inside the volume ray-march so the volume correctly occludes them.\n- 2D (Cartesian): iso-contours of the flux function psi (Bx = d psi/dy, By = -d psi/dx), i.e. the in-plane field lines, colored by |B|.\n- 2D (spherical / Kerr-Schild): traced meridional streamlines of the poloidal (Br, Btheta) field (nt2py style).\nBuilt once per frame and shared by every scene that opts in (per-scene `fieldlines = true`) and by any standalone `field = \"fieldlines\"` scene.", + "type": "object", + "additionalProperties": false, + "properties": { + "enable": { + "description": "Build the field-line geometry this run", + "type": "boolean", + "default": false, + "x-entity": { + "type": "bool", + "notes": [ + "implied true if any scene sets `fieldlines = true` or uses `field = \"fieldlines\"`" + ] + } + }, + "field": { + "description": "Vector field to trace", + "anyOf": [ + { + "enum": [ + "B", + "E", + "J" + ] + }, + { + "type": "string", + "pattern": "(?i)^(B|E|J)$" + } + ], + "default": "B", + "x-entity": { + "type": "string" + } + }, + "bin": { + "description": "Field coarsening factor (simulation cells per coarse cell, per axis)", + "type": "integer", + "minimum": 1, + "maximum": 16, + "default": 4, + "x-entity": { + "type": "int [1..16]", + "notes": [ + "larger = smoother \"morphology\" lines + cheaper replication (the coarse field is ~ N_cells / bin^D floats/rank; D = sim dimension)" + ] + } + }, + "seed_px": { + "description": "(3D tubes) Seed-lattice spacing in screen pixels (sets line density)", + "type": "number", + "exclusiveMinimum": 0.0, + "default": 8, + "x-entity": { + "type": "float [> 0]", + "notes": [ + "capped by `seed_max`; if seed_px asks for more seeds than that, the spacing grows to fit and seed_px no longer governs" + ] + } + }, + "seed_max": { + "description": "(3D tubes) Hard cap on the seed count (lattice is n^3, 2 lines per seed)", + "type": "integer", + "minimum": 1, + "default": 4096, + "x-entity": { + "type": "int [> 0]", + "notes": [ + "lower this for fewer / more widely spaced lines" + ] + } + }, + "levels": { + "description": "(2D contours) Number of evenly-spaced flux-function contour levels", + "type": "integer", + "minimum": 1, + "default": 16, + "x-entity": { + "type": "int [> 0]", + "notes": [ + "evenly-spaced psi levels => line density tracks |B| automatically" + ] + } + }, + "tube_px": { + "description": "Tube radius (3D) / contour line width (2D), in screen pixels", + "type": "number", + "exclusiveMinimum": 0.0, + "default": 2, + "x-entity": { + "type": "float [> 0]" + } + }, + "colormap": { + "description": "Colormap for the field lines (mapped by |B| along each line)", + "$ref": "#/$defs/colormap", + "default": "inferno" + }, + "color": { + "description": "Monochrome override: draw the lines in a single [r,g,b] color (each 0..1) instead of the |B| colormap -- reads well as an overlay on another volume", + "anyOf": [ + { + "type": "array", + "maxItems": 0 + }, + { + "type": "array", + "minItems": 3, + "maxItems": 3, + "items": { + "type": "number", + "minimum": 0.0, + "maximum": 1.0 + } + } + ], + "default": [], + "x-entity": { + "type": "array [size 3]", + "default": "[] (empty => color by |B|)", + "examples": [ + "[1.0, 1.0, 1.0] # white field lines" + ] + } + }, + "log": { + "description": "Map the tube color range logarithmically", + "type": "boolean", + "default": false, + "x-entity": { + "type": "bool", + "notes": [ + "requires min > 0" + ] + } + }, + "min": { + "description": "Tube color range: lower bound on |field|", + "type": "number", + "default": 0.0, + "x-entity": { + "type": "float", + "notes": [ + "when min >= max, the range is auto-set from |field| along the lines" + ] + } + }, + "max": { + "description": "Tube color range: upper bound on |field|", + "type": "number", + "default": 0.0, + "x-entity": { + "type": "float", + "notes": [ + "when min >= max, the range is auto-set from |field| along the lines" + ] + } + }, + "step_frac": { + "description": "(3D tubes) RK4 integration step as a fraction of one coarse cell", + "type": "number", + "exclusiveMinimum": 0.0, + "default": 0.5, + "x-entity": { + "type": "float [> 0]" + } + }, + "max_steps": { + "description": "(3D tubes) Per-direction integration-step cap", + "type": "integer", + "minimum": 1, + "default": 4000, + "x-entity": { + "type": "int [> 0]" + } + }, + "max_length": { + "description": "(3D tubes) Maximum line length, in global box diagonals (per direction)", + "type": "number", + "exclusiveMinimum": 0.0, + "default": 3.0, + "x-entity": { + "type": "float [> 0]" + } + } + } + }, + "scene": { + "description": "One scene per scalar field -> one PNG stream. Repeat the table for each.", + "type": "array", + "x-entity": { + "array_of_tables": true + }, + "items": { + "type": "object", + "additionalProperties": false, + "required": [ + "field" + ], + "properties": { + "field": { + "description": "Scalar field to render (a volume render needs a scalar, so vectors are given as a magnitude or a single component)", + "type": "string", + "x-entity": { + "type": "string", + "enum": [ + "(fields): \"{E,B,J}mag\"; \"{E,B,J}{1,2,3}\" or \"{E,B,J}{x,y,z}\"", + "(moments): \"N\", \"Nppc\", \"Rho\", \"Charge\"; \"T{i}{j}\"; \"V{i}\"; \"Vmag\"" + ], + "notes": [ + "\"{E,B,J}mag\" = vector magnitude |.|; \"B1\"/\"Bx\", \"J3\"/\"Jz\", ... = a single (signed) physical component", + "a bare vector (\"E\"/\"B\"/\"J\") is not renderable -- choose a component or the magnitude", + "\"N\"/\"Nppc\" = number / per-cell count, \"Rho\" = mass density, \"Charge\" = charge density", + "\"T{i}{j}\" = one stress-energy component, i,j in {t,x,y,z} or {0,1,2,3} (e.g. \"Txx\", \"Ttt\", \"T0x\"); \"V{i}\" = one bulk-velocity component, i in {x,y,z} or {1,2,3} (e.g. \"Vx\", \"V1\"); \"Vmag\" = bulk-velocity magnitude sqrt(V1^2+V2^2+V3^2)", + "moments follow the engine: SRPIC = tetrad-basis bulk 3-velocity and stress-energy; GRPIC = Eckart-frame 4-velocity (so \"Vt\"/\"V0\" = u^0 = Gamma/alpha is also valid) and contravariant T", + "per-species selection with a \"_\" suffix on moments, e.g. \"N_1\", \"Rho_2\", \"Txy_1_2\", \"V1_3\"; default = all massive species", + "components are signed; pair a symmetric `min`/`max` with a diverging colormap (\"cool2warm\") to center zero", + "\"fieldlines\" renders the magnetic field-line tubes on their own (no scalar volume sampled); see [render.fieldlines] below" + ] + } + }, + "prefix": { + "description": "PNG filename prefix; files are `.png`", + "type": "string", + "x-entity": { + "type": "string", + "default": "\"_\"" + } + }, + "label": { + "description": "Colorbar title", + "type": "string", + "x-entity": { + "type": "string", + "default": "`field`" + } + }, + "min": { + "description": "Lower bound of the value range mapped onto the colormap/opacity", + "type": "number", + "default": 0.0, + "x-entity": { + "type": "float" + } + }, + "max": { + "description": "Upper bound of the value range", + "type": "number", + "default": 1.0, + "x-entity": { + "type": "float" + } + }, + "log": { + "description": "Map the value range logarithmically", + "type": "boolean", + "default": false, + "x-entity": { + "type": "bool", + "notes": [ + "Requires min > 0 and max > 0" + ] + } + }, + "colormap": { + "description": "Colormap name", + "$ref": "#/$defs/colormap", + "default": "viridis" + }, + "alpha": { + "description": "Opacity transfer function: [position, opacity] control points, both in [0, 1], piecewise-linear in the normalized value", + "type": "array", + "items": { + "type": "array", + "minItems": 2, + "maxItems": 2, + "prefixItems": [ + { + "type": "number", + "minimum": 0.0, + "maximum": 1.0 + }, + { + "type": "number", + "minimum": 0.0, + "maximum": 1.0 + } + ] + }, + "x-entity": { + "type": "array>", + "default": "linear ramp (opacity = normalized value)", + "notes": [ + "Keep the low end near 0 so empty regions stay transparent" + ], + "examples": [ + "[[0.0, 0.0], [0.3, 0.1], [1.0, 0.7]]" + ] + } + }, + "colorbar_ticks": { + "description": "Explicit value(s) to label on the colorbar", + "type": "array", + "items": { + "type": "number" + }, + "x-entity": { + "type": "array", + "default": "5 evenly-spaced ticks between min and max", + "notes": [ + "Values outside [min, max] are skipped" + ], + "examples": [ + "[0.0, 0.5, 1.0]" + ] + } + }, + "fieldlines": { + "description": "Overlay the magnetic field-line tubes inside this scene's volume", + "type": "boolean", + "default": false, + "x-entity": { + "type": "bool", + "notes": [ + "requires the [render.fieldlines] enabled (3D only). A scene with field = \"fieldlines\" instead renders them alone." + ] + } + } + } + } + } + } + }, + "diagnostics": { + "description": "Diagnostic logging parameters", + "type": "object", + "additionalProperties": false, + "properties": { + "interval": { + "description": "Number of timesteps between diagnostic logs", + "type": "integer", + "minimum": 1, + "default": 1, + "x-entity": { + "type": "int [> 0]" + } + }, + "blocking_timers": { + "description": "Blocking timers between successive algorithms", + "type": "boolean", + "default": false, + "x-entity": { + "type": "bool" + } + }, + "colored_stdout": { + "description": "Enable colored stdout", + "type": "boolean", + "default": true, + "x-entity": { + "type": "bool" + } + }, + "log_level": { + "description": "Specify the log level", + "anyOf": [ + { + "enum": [ + "VERBOSE", + "WARNING", + "ERROR" + ] + }, + { + "type": "string", + "pattern": "(?i)^(VERBOSE|WARNING|ERROR)$" + } + ], + "default": "VERBOSE", + "x-entity": { + "type": "string", + "notes": [ + "\"VERBOSE\" prints all messages, \"WARNING\" prints only warnings and errors, \"ERROR\" prints only errors" + ] + } + } + } + } + } +} diff --git a/examples/custom_particle_update/pgen.hpp b/examples/custom_particle_update/pgen.hpp index 30e464099..95033de44 100644 --- a/examples/custom_particle_update/pgen.hpp +++ b/examples/custom_particle_update/pgen.hpp @@ -116,7 +116,8 @@ namespace user { arch::InjectGlobally(metadomain, local_domain, (spidx_t)2, data_i); } - auto FixFieldsConst(const bc_in&, const em&) const -> std::pair { + auto FixFieldsConst(simtime_t, const bc_in&, const em&) const + -> std::pair { return { ZERO, false }; } diff --git a/input.default.toml b/input.default.toml new file mode 100644 index 000000000..3377cc6f7 --- /dev/null +++ b/input.default.toml @@ -0,0 +1,1300 @@ +# Global simulation parameters +[simulation] + # Name of the simulation + # @required + # @type: string + # @note: The name is used for the output files + name = "" + # Simulation engine to use + # @required + # @type: string + # @enum: "SRPIC", "GRPIC" + engine = "SRPIC" + # Max runtime in physical (code) units + # @required + # @type: float [> 0] + # @example: 1e5 + runtime = 1.0 + + # Parameters specific to domain decomposition + [simulation.domain] + # Number of domains + # @type: int + # @default: 1 [no MPI]; MPI_SIZE [MPI] + number = 1 + # Decomposition of the domain (for MPI) in each of the directions + # @type: array [size 1 :->: 3] + # @default: [-1, -1, -1] + # @note: -1 means the code will determine the decomposition in the + # specific direction automatically + # @note: Automatic detection is either done by inference from # of MPI + # tasks, or by balancing the grid size on each domain + # @example: [2, 2, 2] (total of 8 domains) + decomposition = [-1, -1, -1] + + # Diffusion-style dynamic load balancing. Domain boundaries between MPI + # neighbors are nudged to equalize the active-particle count per rank. All + # inter-rank traffic uses only the existing nearest-neighbor field/particle + # communication paths. + [simulation.domain.load_balance] + # Enable dynamic load balancing + # @type: bool + # @default: false + enable = false + # Run the rebalancer every `interval` timesteps (0 disables) + # @type: int + # @default: 0 + interval = 0 + # Dimensions along which load is redistributed (1 = x1, 2 = x2, 3 = x3) + # @type: array [subset of {1, 2, 3}] + # @default: [1] + # @enum: 1, 2, 3 + dimensions = [1] + # Skip rebalancing along a dim when (max - min) / mean of the per-slice + # particle count is below this fraction + # @type: float + # @default: 0.1 + tolerance = 0.1 + # Maximum cell-shift per interior boundary per event; clamped at compile + # time to N_GHOSTS so the migrating field strip is already cached in the + # rank's ghost zone. + # @type: int + # @default: N_GHOSTS + max_shift = 0 + +# Parameters specific to grid geometry +[grid] + # Spatial resolution of the grid + # @required + # @type: array [size 1 :->: 3] + # @note: Dimensionality is inferred from the size of this array + # @example: [1024, 1024, 1024] + resolution = [1] + # Physical extent of the grid + # @required + # @type: array> [size 1 :->: 3] + # @note: For spherical geometry, only specify `[[rmin, rmax]]`, other values + # are set automatically + # @note: For cartesian geometry, cell aspect ratio has to be 1: `dx=dy=dz` + # @example: [[0.0, 1.0], [-1.0, 1.0]] + extent = [[0.0, 0.0]] + + # @inferred: + # - dim + # @brief: Dimensionality of the grid + # @type: short + # @enum: 1, 2, 3 + # @from: `grid.resolution` + + # Metric-related parameters + [grid.metric] + # Metric on the grid + # @required + # @type: string + # @enum: "Minkowski", "Spherical", "QSpherical", "Kerr_Schild", + # "QKerr_Schild", "Kerr_Schild_0" + metric = "Minkowski" + # `r0` paramter for the QSpherical metric `x1 = log(r-r0)` + # @type: float [-inf -> rmin] + # @default: 0.0 + # @note: Negative values produce almost uniform grid in r + qsph_r0 = 0.0 + # `h` paramter for the QSpherical metric `th = x2 + 2*h x2 + # (pi-2*x2)*(pi-x2)/pi^2` + # @type: float [-1 :->: 1] + # @default: 0.0 + qsph_h = 0.0 + # Spin parameter for the Kerr Schild metric + # @type: float [0 :-> 1] + # @default: 0.0 + ks_a = 0.0 + + # @inferred: + # - coord + # @brief: Coordinate system on the grid + # @type: string + # @enum: "cartesian", "spherical", "qspherical" + # @from: `grid.metric.metric` + # - ks_rh + # @brief: Size of the horizon for GR Kerr Schild + # @type: float + # @from: `grid.metric.ks_a` + # - params + # @brief: A map of all metric-specific parameters together (for easy + # access) + # @type: map + # @from: `grid.metric` + + # Boundary-condition related parameters + [grid.boundaries] + # Boundary conditions for fields + # @required + # @type: array> [size 1 :->: 3] + # @enum: "PERIODIC", "MATCH", "FIXED", "ATMOSPHERE", "CUSTOM", "HORIZON", + # "CONDUCTOR" + # @note: When periodic in any of the directions, you should only set one + # value: [..., ["PERIODIC"], ...] + # @note: In spherical, bondaries in theta/phi are set automatically (only + # specify bc @ `[rmin, rmax]`): [["ATMOSPHERE", "MATCH"]] + # @note: In GR, the horizon boundary is set automatically (only specify bc + # @ rmax): [["MATCH"]] + # @example: [["CUSTOM", "MATCH"]] (for 2D spherical `[[rmin, rmax]]`) + fields = [["PERIODIC"]] + # Boundary conditions for particles + # @required + # @type: array> [size 1 :->: 3] + # @enum: "PERIODIC", "ABSORB", "ATMOSPHERE", "CUSTOM", "REFLECT", + # "HORIZON" + # @note: When periodic in any of the directions, you should only set one + # value [..., ["PERIODIC"], ...] + # @note: In spherical, bondaries in theta/phi are set automatically (only + # specify bc @ `[rmin, rmax]`) [["ATMOSPHERE", "ABSORB"]] + # @note: In GR, the horizon boundary is set automatically (only specify bc + # @ `rmax`): [["ABSORB"]] + # @example: [["PERIODIC"], ["PERIODIC"]] + particles = [["PERIODIC"]] + + # Parameters specific to MATCH boundary conditions + [grid.boundaries.match] + # Size of the matching layer in each direction for fields in physical + # (code) units + # @type: float | array> + # @default: 1% of the domain size (in shortest dimension) + # @note: In spherical, this is the size of the layer in `r` from the + # outer wall + # @example: `ds = 1.5` (will set the same for all directions) + # @example: `ds = [[1.5], [2.0, 1.0], [1.1]]` (will duplicate 1.5 for + # +/- `x1` and 1.1 for +/- `x3`) + # @example: `ds = [[], [1.5], []]` (will only set for x2) + ds = 1.0 + + # Parameters specific to ABSORB boundary conditions + [grid.boundaries.absorb] + # Size of the absorption layer for particles in physical (code) units + # @type: float + # @default: 1% of the domain size (in shortest dimension) + # @note: In spherical, this is the size of the layer in `r` from the + # outer wall + # @note: In cartesian, this is the same for all dimensions where + # applicable + ds = 1.0 + + # Parameters specific to ATMOSPHERE boundary conditions + [grid.boundaries.atmosphere] + # Temperature of the atmosphere in units of `m0 c^2` + # @type: float + # @note: [required] if `ATMOSPHERE` is one of the boundaries + temperature = 0.0 + # Peak number density of the atmosphere at base in units of `n0` + # @type: float + density = 0.0 + # Pressure scale-height in physical units + # @type: float + height = 1.0 + # Species indices of particles that populate the atmosphere + # @type: array [size 2] + species = [1, 1] + # Distance from the edge to which the gravity is imposed in physical units + # @type: float + # @default: 0.0 + # @note: 0.0 means no limit + ds = 0.0 + + # @inferred: + # - g + # @brief: Acceleration due to imposed gravity + # @type: float + # @from: `grid.boundaries.atmosphere.temperature`, + # `grid.boundaries.atmosphere.height` + # @value: `temperature / height` + +# Fiducial scales that fix the code unit system +[scales] + # Fiducial larmor radius + # @required + # @type: float [> 0.0] + larmor0 = 1.0 + # Fiducial plasma skin depth + # @required + # @type: float [> 0.0] + skindepth0 = 1.0 + + # @inferred: + # - dx0 + # @brief: fiducial minimum size of the cell + # @type: float + # @from: `grid` + # - V0 + # @brief: fiducial elementary volume + # @type: float + # @from: `grid` + # - n0 + # @brief: Fiducial number density + # @type: float + # @from: `particles.ppc0`, `grid` + # @value: `ppc0 / V0` + # - q0 + # @brief: Fiducial elementary charge + # @type: float + # @from: `scales.skindepth0`, `scales.n0` + # @value: `1 / (n0 * skindepth0^2)` + # - sigma0 + # @brief: Fiducial magnetization parameter + # @type: float + # @from: `scales.larmor0`, `scales.skindepth0` + # @value: `(skindepth0 / larmor0)^2` + # - B0 + # @brief: Fiducial magnetic field + # @type: float + # @from: `scales.larmor0` + # @value: `1 / larmor0` + # - omegaB0 + # @brief: Fiducial cyclotron frequency + # @type: float + # @from: `scales.larmor0` + # @value: `1 / larmor0` + +# Radiative drag and photon emission parameters +[radiation] + + # Radiation reaction (drag) parameters + [radiation.drag] + + # Synchrotron drag parameters + [radiation.drag.synchrotron] + # Radiation reaction limit gamma-factor for synchrotron + # @type: float [> 0.0] + # @default: 1.0 + # @note: [required] if one of the species has `radiative_drag = + # "synchrotron"` + gamma_rad = 1.0 + + # Compton drag parameters + [radiation.drag.compton] + # Radiation reaction limit gamma-factor for Compton drag + # @type: float [> 0.0] + # @default: 1.0 + # @note: [required] if one of the species has `radiative_drag = + # "compton"` + gamma_rad = 1.0 + + # Photon emission parameters + [radiation.emission] + + # Synchrotron emission parameters + [radiation.emission.synchrotron] + # Gamma-factor of a particle emitting synchrotron photons at energy `m0 + # c^2` in fiducial magnetic field `B0` + # @type: float [> 1.0] + # @default: 10.0 + gamma_qed = 10.0 + # Minimum photon energy for synchrotron emission (units of `m0 c^2`) + # @type: float [> 0.0] + # @default: 1e-3 + photon_energy_min = 1e-3 + # Weights for the emitted synchrotron photons + # @type: float [> 0.0] + # @default: 1.0 + photon_weight = 1.0 + # Index of species for the emitted photon + # @required + # @type: ushort [> 0] + photon_species = 1 + + # @inferred: + # - nominal_probability + # @brief: Nominal probability of the emission for a particle with + # `gamma * beta = 1`, charge-to-mass = `q0 / m0` + # @type: float + # @from: `.gamma_qed`, `.photon_weight`, + # `...drag.synchrotron.gamma_rad`, `scales.omegaB0`, + # `algorithms.timestep.dt` + # @value: `0.1 * omegaB0 * dt * (gamma_qed / gamma_rad)^2 / + # photon_weight` + # - nominal_photon_energy + # @brief: Nominal energy of the emitted photon for a particle with + # `gamma * beta = 1`, mass = `m0` + # @type: float + # @from: `.gamma_qed` + # @value: `(1 / gamma_qed)^2` + + # Inverse Compton emission parameters + [radiation.emission.compton] + # Gamma-factor of a particle emitting inverse Compton photons at energy + # `m0 c^2` in fiducial magnetic field `B0` + # @type: float [> 1.0] + # @default: 10.0 + gamma_qed = 10.0 + # Minimum photon energy for inverse Compton emission (units of `m0 c^2`) + # @type: float [> 0.0] + # @default: 1e-3 + photon_energy_min = 1e-3 + # Weights for the emitted inverse Compton photons + # @type: float [> 0.0] + # @default: 1.0 + photon_weight = 1.0 + # Index of species for the emitted photon + # @required + # @type: ushort [> 0] + photon_species = 1 + + # @inferred: + # - nominal_probability + # @brief: Nominal probability of the emission for a particle with + # `gamma * beta = 1`, charge-to-mass = `q0 / m0` + # @type: float + # @from: `.gamma_qed`, `.photon_weight`, `...drag.compton.gamma_rad`, + # `scales.omegaB0`, `algorithms.timestep.dt` + # @value: `0.1 * omegaB0 * dt * (gamma_qed / gamma_rad)^2 / + # photon_weight` + # - nominal_photon_energy + # @brief: Nominal energy of the emitted photon for a particle with + # `gamma * beta = 1`, mass = `m0` + # @type: float + # @from: `.gamma_qed` + # @value: `(1 / gamma_qed)^2` + +# Algorithm and solver tuning +[algorithms] + # Number of current smoothing passes + # @type: ushort [>= 0] + # @default: 0 + current_filters = 0 + + # Timestep parameters + [algorithms.timestep] + # Courant-Friedrichs-Lewy number + # @type: float [0.0 -> 1.0] + # @default: 0.95 + # @note: CFL number determines the timestep duration + CFL = 0.95 + # Correction factor for the speed of light used in field solver + # @type: float + # @default: 1.0 + correction = 1.0 + + # @inferred: + # - dt + # @brief: timestep duration + # @type: float + # @from: `algorithms.timestep.CFL`, `scales.dx0` + # @value: `CFL * dx0` + + # Current deposition parameters + [algorithms.deposit] + # Enable the current deposition + # @type: bool + # @default: true + enable = true + # Tiled-deposit work-group (team) size + # @type: uint [>= 0] + # @default: 0 + # @deprecated: removed in 1.6+, use `tiled_deposit_team_size` instead + team_policy_team_size = 0 + # Tiled-deposit work-group (team) size + # @type: uint [>= 0] + # @default: 0 + # @note: 0 keeps Kokkos::AUTO (backend occupancy heuristic); a positive + # value overrides it, clamped to the backend/scratch maximum at + # launch. Only used in `tiled_deposit=ON` builds. Pick a multiple + # of the device subgroup width for best occupancy (see + # ideal_tile_size.py) + tiled_deposit_team_size = 0 + + # @inferred: + # - order + # @brief: order of the particle shape function + # @type: ushort [0 -> 10] + # @from: compile-time definition `shape_order` + + # GR pusher parameters + [algorithms.gr] + # Stepsize for numerical differentiation in GR pusher + # @type: float [> 0.0] + # @default: 1e-6 + pusher_eps = 1e-6 + # Number of iterations for the Newton-Raphson method in GR pusher + # @type: ushort [> 0] + # @default: 10 + pusher_niter = 10 + + # Guiding-center approximation parameters + [algorithms.gca] + # Maximum value for E/B allowed for GCA particles + # @type: float [0.0 -> 1.0] + # @default: 0.9 + e_ovr_b_max = 0.9 + # Maximum Larmor radius allowed for GCA particles (in physical units) + # @type: float + # @default: 0.0 + # @note: When `larmor_max` == 0, the limit is disabled + larmor_max = 0.0 + + # Stencil coefficients for the field solver [notation as in Blinne+ (2018)] + # @note: Standard Yee solver: `delta_i = beta_ij = 0.0` + [algorithms.fieldsolver] + # Enable the fieldsolver + # @type: bool + # @default: true + enable = true + # delta_x coefficient (for `F_{i +/- 3/2, j, k}`) + # @type: float + # @default: 0.0 + delta_x = 0.0 + # delta_y coefficient (for `F_{i, j +/- 3/2, k}`) + # @type: float + # @default: 0.0 + # @note: Used only for 2D and 3D + delta_y = 0.0 + # delta_z coefficient (for `F_{i, j, k +/- 3/2}`) + # @type: float + # @default: 0.0 + # @note: Used only for 3D + delta_z = 0.0 + # beta_xy coefficient (for `F_{i +/- 1/2, j +/- 1, k}`) + # @type: float + # @default: 0.0 + # @note: Used only for 2D and 3D + beta_xy = 0.0 + # beta_yx coefficient (for `F_{i +/- 1, j +/- 1/2, k}`) + # @type: float + # @default: 0.0 + # @note: Used only for 2D and 3D + beta_yx = 0.0 + # beta_xz coefficient (for `F_{i +/- 1/2, j, k +/- 1}`) + # @type: float + # @default: 0.0 + # @note: Used only for 3D + beta_xz = 0.0 + # beta_zx coefficient (for `F_{i +/- 1, j, k +/- 1/2}`) + # @type: float + # @default: 0.0 + # @note: Used only for 3D + beta_zx = 0.0 + # beta_yz coefficient (for `F_{i, j +/- 1/2, k +/- 1}`) + # @type: float + # @default: 0.0 + # @note: Used only for 3D + beta_yz = 0.0 + # beta_zy coefficient (for `F_{i, j +/- 1, k +/- 1/2}`) + # @type: float + # @default: 0.0 + # @note: Used only for 3D + beta_zy = 0.0 + +# Particle and species parameters +[particles] + # Fiducial number of particles per cell + # @required + # @type: float [> 0.0] + ppc0 = 1.0 + # Toggle for using particle weights + # @type: bool + # @default: false + use_weights = false + # Timesteps between particle re-sorting by tags (removing dead particles) + # @type: uint + # @default: 100 + # @note: Set to 0 to disable re-sorting + clear_interval = 100 + # Timesteps between spatial sorting of particles (for better cache + # performance) + # @type: uint + # @default: 0 + # @note: Set to 0 to disable spatial sorting + spatial_sorting_interval = 0 + + # @inferred: + # - nspec + # @brief: Number of particle species + # @type: uint + # @from: `particles.species` + + # Particle species definitions + [[particles.species]] + # Label of the species + # @type: string + # @default: "s" + # @note: `` is the index of the species in the list starting from 1 + # @example: "e-" + label = "s" + # Mass of the species (in units of fiducial mass) + # @required + # @type: float [>= 0.0] + mass = 0.0 + # Charge of the species (in units of fiducial charge) + # @required + # @type: float + charge = 0.0 + # Maximum number of particles per task + # @required + # @type: uint [> 0] + # @note: Read as a float, so exponential notation is fine (e.g. `1e8`) + maxnpart = 1.0 + # Pusher algorithm for the species + # @type: string + # @default: "Boris" [massive]; "Photon" [massless] + # @enum: "Boris", "Vay", "Boris,GCA", "Vay,GCA", "Photon", "None" + pusher = "Boris" + # Number of additional real-valued variables (payloads) for each particle of + # the given species + # @type: ushort + # @default: 0 + n_payloads_real = 0 + # Number of additional integer-valued variables (payloads) for each particle + # of the given species + # @type: ushort + # @default: 0 + # @note: If tracking is enabled, one or two extra integer payloads are + # reserved (depending on whether MPI is enabled) + n_payloads_int = 0 + # Enable tracking of particles using indices for the given species + # @type: bool + # @default: false + tracking = false + # Radiation reaction to use for the species + # @type: string + # @default: "None" + # @enum: "None", "Synchrotron", "Compton" + # @note: Can also be coma-separated combination, e.g., + # "Synchrotron,Compton" + # @note: Relevant radiation.drag parameters should also be provided + radiative_drag = "None" + # Particle emission policy for the species + # @type: string + # @default: "None" + # @enum: "None", "Synchrotron", "Compton", "Custom" + # @note: Only one emission mechanism allowed + # @note: Appropriate radiation drag flag will be applied automatically + # (unless explicitly set to "None") + emission = "None" + # Timesteps between spatial sorting of particles for given species + # @type: uint + # @default: 0 + # @note: Set to 0 to disable spatial sorting + # @note: Overrides `particles.spatial_sorting_interval` for the given + # species + spatial_sorting_interval = 0 + # Timesteps between particle re-sorting by tags (removing dead particles) + # @type: uint + # @default: 100 + # @note: Set to 0 to disable re-sorting + # @note: Overrides `particles.clear_interval` for the given species + clear_interval = 100 + +# Parameters for specific problem generators and setups +# @note: Free-form: keys are defined by the problem generator, so nothing here +# is validated +[setup] + +# Output parameters +[output] + # Output format + # @type: string + # @default: "bpfile" + # @enum: "disabled", "hdf5", "BPFile" + format = "bpfile" + # Number of timesteps between all outputs + # @type: uint [> 0] + # @default: 100 + # @note: Value is overriden by output intervals for specific outputs + interval = 100 + # Physical (code) time interval between all outputs + # @type: float + # @default: -1.0 + # @note: When `interval_time` < 0, the output is controlled by `interval`, + # otherwise by `interval_time` + # @note: Value is overriden by output intervals for specific outputs + interval_time = -1.0 + + # Field output parameters + [output.fields] + # Toggle for the field output + # @type: bool + # @default: true + enable = true + # Field quantities to output + # @type: array + # @default: [] + # @enum: "E", "B", "J", "divE", "Rho", "Charge", "N", "Nppc", "T0i", + # "Tij", "Vi", "D", "H", "divD", "A" + # @note: For `T`, you can use unspecified indices: `Tij`, `T0i`, or + # specific ones: `Ttt`, `T00`, `T02`, `T23` + # @note: For `T`, in cartesian can also use "x" "y" "z" instead of "1" "2" + # "3" + # @note: By default, we accumulate moments from all massive species, one + # can specify only specific species: `Ttt_1_2`, `Rho_1`, `Rho_3_4` + quantities = [] + # Custom (user-defined) field quantities + # @type: array + # @default: [] + custom = [] + # Number of timesteps between field outputs + # @type: uint + # @default: 0 + # @note: When `!= 0`, overrides `output.interval` + # @note: When `== 0`, `output.interval` is used + interval = 0 + # Physical (code) time interval between field outputs + # @type: float + # @default: -1.0 + # @note: When `< 0`, the output is controlled by `interval` + # @note: When specified, overrides `output.interval_time` + interval_time = -1.0 + # Downsample factor for the output of fields + # @type: uint | array [>= 1] + # @default: [1, 1, 1] + # @note: The output is downsampled by the given factors in each direction + # @note: If a scalar is given, it is applied to all directions + downsampling = [1, 1, 1] + + # Smoothing of the output moments + [output.fields.smoothing] + # Smoothing order for the output of moments ("Rho", "Charge", "T", ...) + # @type: ushort + # @default: 0 + order = 0 + # Smoothing algorithm + # @type: string + # @default: "spline" + # @enum: "const", "spline" + # @note: When using "spline", `order` corresponds to the order of the + # polynomial used + # @note: When using "const", the smoothing window is `ceil(order / 2)` + # in both directions + method = "spline" + + # Particle output parameters + [output.particles] + # Toggle for the particles output + # @type: bool + # @default: true + enable = true + # Particle species indices to output + # @type: array + # @default: [] + # @note: If empty, all species are output + species = [] + # Stride for the output of particles + # @type: uint [>= 1] + # @default: 100 + stride = 100 + # Number of timesteps between particle outputs + # @type: uint + # @default: 0 + # @note: When `!= 0`, overrides `output.interval` + # @note: When `== 0`, `output.interval` is used + interval = 0 + # Physical (code) time interval between particle outputs + # @type: float + # @default: -1.0 + # @note: When `< 0`, the output is controlled by `interval` + # @note: When specified, overrides `output.interval_time` + interval_time = -1.0 + + # Spectra output parameters + [output.spectra] + # Toggle for the spectra output + # @type: bool + # @default: true + enable = true + # Minimum energy for the spectra output + # @type: float + # @default: 1e-3 + e_min = 1e-3 + # Maximum energy for the spectra output + # @type: float + # @default: 1e3 + e_max = 1e3 + # Whether to use logarithmic bins for energy + # @type: bool + # @default: true + log_bins = true + # Number of energy bins for the spectra output + # @type: uint [> 0] + # @default: 200 + # @deprecated: removed in 1.6+, use `num_energy_bins` instead + n_bins = 200 + # Number of energy bins for the spectra output + # @type: uint [> 0] + # @default: 200 + num_energy_bins = 200 + # Number of spatial bins for the spectra output + # @type: array [size 1 :->: 3] + # @default: [1, 1, 1] + num_spatial_bins = [1, 1, 1] + # Number of timesteps between spectra outputs + # @type: uint + # @default: 0 + # @note: When `!= 0`, overrides `output.interval` + # @note: When `== 0`, `output.interval` is used + interval = 0 + # Physical (code) time interval between spectra outputs + # @type: float + # @default: -1.0 + # @note: When `< 0`, the output is controlled by `interval` + # @note: When specified, overrides `output.interval_time` + interval_time = -1.0 + + # Debug output parameters + [output.debug] + # Output fields "as is" without conversions + # @type: bool + # @default: false + as_is = false + # Output fields with values in ghost cells + # @type: bool + # @default: false + ghosts = false + + # Integrated statistics output parameters + [output.stats] + # Toggle for the stats output + # @type: bool + # @default: true + enable = true + # Number of timesteps between stat outputs + # @type: uint [> 0] + # @default: 100 + # @note: Overriden if `output.stats.interval_time != -1` + interval = 100 + # Physical (code) time interval between stat outputs + # @type: float + # @default: -1.0 + # @note: When `< 0`, the output is controlled by `interval` + interval_time = -1.0 + # Field quantities to output + # @type: array + # @default: ["B^2", "E^2", "ExB", "Rho", "T00"] + # @enum: "B^2", "E^2", "ExB", "N", "Npart", "Charge", "Rho", "T00", "T0i", + # "Tij" + # @note: For particle moments, ... + # @note: ... same notation is used as for `output.fields.quantities` + quantities = ["B^2", "E^2", "ExB", "Rho", "T00"] + # Custom (user-defined) stats + # @type: array + # @default: [] + custom = [] + +# Checkpointing parameters +[checkpoint] + # Number of timesteps between checkpoints + # @type: uint [> 0] + # @default: 1000 + interval = 1000 + # Physical (code) time interval between checkpoints + # @type: float [> 0] + # @default: -1.0 + # @note: When `< 0`, the output is controlled by `interval` + interval_time = -1.0 + # Number of checkpoints to keep + # @type: int + # @default: 2 + # @note: 0 = disable checkpointing + # @note: -1 = keep all checkpoints + keep = 2 + # Write a checkpoint once after a fixed walltime + # @type: string + # @default: "00:00:00" + # @note: The format is "HH:MM:SS" + # @note: Empty string or "00:00:00" disables this functionality + # @note: Writing checkpoint at walltime does not stop the simulation + walltime = "00:00:00" + # Parent directory to write checkpoints to + # @type: string + # @default: `.ckpt` + # @note: The directory is created if it does not exist + write_path = "" + # Parent directory to use when resuming from a checkpoint + # @type: string + # @default: inherit `write_path` + read_path = "" + + # @inferred: + # - is_resuming + # @brief: Whether the simulation is resuming from a checkpoint + # @type: bool + # @from: command-line flag + # - start_step + # @brief: Timestep of the checkpoint used to resume + # @type: uint + # @from: automatically determined during restart + # - start_time + # @brief: Time of the checkpoint used to resume + # @type: float + # @from: automatically determined during restart + +# ADIOS2 BP5 tuning, applied to both [output] and [checkpoint] writers +[adios2] + # Number of ADIOS2 aggregators per node + # @type: uint + # @default: 0 + # @note: Set to either MPI ranks/node or NICs/node for best performance + # If set to 0, will use ADIOS2 default (one aggregator per node) + aggregators_per_node = 0 + # Maximum shared-memory segment size per node, in bytes (BP5 MaxShmSize) + # @type: uint + # @default: 4294967296 + # @note: Lower this on memory-constrained nodes; matches ADIOS2's default + max_shm_size = 4294967296 + # Internal serialization buffer chunk size, in bytes (BP5 BufferChunkSize) + # @type: uint + # @default: 16777216 + # @note: Scales with per-rank output volume; matches ADIOS2's default + buffer_chunk_size = 16777216 + +# In-situ renderer. Renders scalar fields on the GPU and writes PNG images +# directly to `/renders/` each cadence -- no field data is written to +# storage, and the result is seamless across MPI domain boundaries. +# @note: two modes, selected automatically by the simulation dimension: +# - 3D Cartesian (Minkowski): volume ray-march (uses `samples`, +# `step_size`, `early_term_alpha`, and the [camera] table) +# - 2D (Minkowski, Spherical/QSpherical, and all GR Kerr-Schild): flat +# slice rasterizer. Cartesian shows the (x, y) plane; spherical shows +# the meridional (r, theta) half-plane mapped to Cartesian (X = r sin +# th, Z = r cos th), optionally mirrored (see `mirror`). The +# `samples`/`step_size`/`early_term_alpha`/[camera] keys are ignored in +# 2D (one opaque sample per pixel). +# @note: 1D (and 3D non-Cartesian, which does not exist) is a no-op +# @note: One PNG stream per scene (e.g. a density/|B|/|J| triptych) +[render] + # Toggle for the on-the-fly renderer + # @type: bool + # @default: false + enable = false + # Number of timesteps between renders + # @type: uint + # @default: 0 + # @note: When `!= 0`, overrides `output.interval` + # @note: When `== 0`, `interval_time` (or `output.interval`) is used + interval = 0 + # Physical (code) time interval between renders + # @type: float + # @default: -1.0 + # @note: When `< 0`, the output is controlled by `interval` + interval_time = -1.0 + # Image width in pixels (the rendered region; the PNG is wider if a colorbar + # margin is added, see `colorbar_outside`) + # @type: int [> 0] + # @default: 1024 + width = 1024 + # Image height in pixels + # @type: int [> 0] + # @default: 1024 + height = 1024 + # Convenience: force a square frame (sets width == height == resolution), the + # natural shape for a dome master. Overrides `width`/`height` when > 0. + # @type: int [> 0] + # @default: 0 (use width/height) + resolution = 0 + # Number of entries in the color/opacity lookup table + # @type: int [> 1] + # @default: 256 + n_lut = 256 + # Opaque background RGB (each channel 0..1) shown through + # transparent/low-opacity pixels; also fills the colorbar margin + # @type: array [size 3] + # @default: [0.0, 0.0, 0.0] + background = [0.0, 0.0, 0.0] + # Draw a colorbar (gradient + value ticks + label) on each PNG + # @type: bool + # @default: true + colorbar = true + # Draw the colorbar in an added right margin (the PNG becomes wider by a fixed + # strip) instead of overlaying it on the rendered volume + # @type: bool + # @default: true + colorbar_outside = true + # 2D spherical slice only: mirror the meridional half-plane across the + # symmetry axis to render a full disk from one axisymmetric half. No effect on + # Cartesian or 3D rendering. + # @type: bool + # @default: true + mirror = true + # Draw the current simulation time as a label ("T = ", fixed to 2 + # decimals) in the upper-right corner of the render region, in a contrasting + # color, vertically centered between the frame top and the colorbar. + # @type: bool + # @default: false + time_label = false + # Draw a spine (frame) + axis ticks + labels around the rendered region. The + # PNG gains left/bottom margins (background-filled) for the tick labels and + # axis names, so they never overlap the data. + # @type: bool + # @default: false + # @note: 2D Cartesian = a rectangular frame with linear spatial ticks; + # 2D spherical = polar axes (an "R" radial axis on the symmetry axis + # with R=0 centered, and a "Theta" axis along the curved outline / + # spine); + # 3D = the global box projected to a wireframe with ticks on the + # three silhouette edges (x bottom, y & z on the left) + axes = false + # Axis names. 3D uses all three; the 2D slice uses the first two. When unset, + # the 2D slice defaults to "x","y" (Cartesian) or "X","Z" (spherical). + # @type: array [size <= 3] + # @default: ["x", "y", "z"] + axis_labels = ["x", "y", "z"] + # Target number of ticks per axis (actual count is rounded to nice values) + # @type: int [>= 2] + # @default: 5 + axis_ticks = 5 + # 3D only: target width (pixels) of the box wireframe "spine". The spine is + # drawn inside the ray-march (opaque, depth-occluded by the volume); its width + # is floored by the ray step, so for a crisper thin line raise `samples` as + # well. + # @type: float [> 0.0] + # @default: 2.0 + spine_width = 2.0 + + # Limit the render region to axis-aligned box in physical/world coordinates. + # Left unset it spans the full domain. Clamped to the box. + # @note: 3D -> the volume is depth-clipped to this box, the wireframe/axes + # frame it, and the default camera zooms to it; 2D -> the slice + # window is framed to it + [render.extent] + # Axis-aligned render region [lo, hi] along x1, in physical/world coords. + # Left unset it spans the full domain. Clamped to the box. + # @type: array [size 2] + # @default: [] (full extent) + # @note: For a spherical 2D slice, x1 crops the radius r. + # @example: x1 = [-64.0, 64.0] + x1 = [] + # Render region [lo, hi] along x2. + # @type: array [size 2] + # @default: [] (full extent) + # @note: For a spherical 2D slice, x2 crops the polar angle theta. + x2 = [] + # Render region [lo, hi] along x3. + # @type: array [size 2] + # @default: [] (full extent) + x3 = [] + + # Volume rendering parameters (3D Cartesian only) + [render.volume] + # Number of ray-march steps across the global box diagonal + # @type: int [> 0] + # @default: 400 + # @note: The world-space step is `box_diagonal / samples` unless + # `step_size` is set. Higher = better quality, slower. + samples = 400 + # Fixed world-space step between ray samples + # @type: float [>= 0.0] + # @default: 0.0 + # @note: 0 derives the step from `samples`. The step is identical on all + # ranks, which is what makes the multi-domain composite seamless. + step_size = 0.0 + # Stop marching a ray once its accumulated opacity reaches this value + # @type: float [0.0 -> 1.0] + # @default: 0.99 + # @note: Pure speed optimization; set to 1.0 to disable early termination + early_term_alpha = 0.99 + + # Translate the render region (and, in 3D, the camera), to keep a propagating + # feature (e.g. a shock) in frame. Pair with x{1,2,3} extent to crop the + # moving window. + [render.moving_view] + # Velocity of the moving camera in world units. + # @type: array [size 2 or 3] + # @default: [] (static view) + # @example: velocity = [0.9, 0.0] # pan along +x1 at 0.9 c + velocity = [] + # Sim time at which the view starts moving (static before it, e.g. to let an + # initial ramp-up finish) + # @type: float + # @default: 0.0 + start_time = 0.0 + + # Camera (3D volume mode only; ignored by the 2D slice rasterizer). Defaults + # frame the whole global box from outside, looking down the (1,1,1) diagonal + # -- the production setup for which the structured composite is provably + # seamless. + [render.camera] + # Projection mode. Overrides `orthographic` below when set. + # @type: string + # @default: (unset -> use `orthographic`) + # @enum: "orthographic", "perspective", "dome" + # @note: "dome" is a fulldome azimuthal-equidistant fisheye rendered from + # an INTERIOR eye (the domain center by default), i.e. a 3D + # planetarium dome master. It uses a depth-resolved (A-buffer) + # composite that is seamless across a full 3D domain decomposition + # (unlike ortho/perspective, which need the eye outside the box). + # Set a square frame (`resolution`, or width == height). `forward` + # is the dome ZENITH (screen-up defaults to +y for a +z zenith). + mode = "orthographic" + # Camera (eye) position in world (physical) coordinates + # @type: array [size 3] + # @default: box center pushed back ~1.7 box-diagonals along (1, 1, 1); + # for `mode = "dome"`, the domain center (interior eye) + position = [0.0, 0.0, 0.0] + # Point the camera looks at, in world coordinates (the dome ZENITH target) + # @type: array [size 3] + # @default: box center; for `mode = "dome"`, the zenith defaults to +z + look_at = [0.0, 0.0, 0.0] + # Camera up vector (dome: the disk's screen-up) + # @type: array [size 3] + # @default: [0.0, 0.0, 1.0]; for `mode = "dome"`, [0.0, 1.0, 0.0] + up = [0.0, 0.0, 1.0] + # Vertical field of view in degrees (perspective only) + # @type: float [> 0.0] + # @default: 35.0 + fov = 35.0 + # Full dome field of view in degrees (dome mode only): the image rim is at + # dome_fov/2 from the zenith (180 = a full hemisphere down to the horizon). + # @type: float [> 0.0, <= 360.0] + # @default: 180.0 + dome_fov = 180.0 + # Dome far-clip radius in world units (dome mode only): each ray stops this + # far from the eye, so the sampled region is a half-ball (hemisphere) of + # this radius rather than the whole box -> uniform path length and no box + # corner/edge projection artifacts. `samples` then counts steps across this + # radius. + # @type: float [>= 0.0] + # @default: the largest sphere centered in the box (half the shortest + # side), so it touches the face centers and never a corner + # @note: 0 disables the clip (rays march to the box boundary) + dome_radius = 0.0 + # Vertical extent of the view in world units (orthographic only) + # @type: float [> 0.0] + # @default: the global box diagonal (the whole box fits from any angle) + ortho_height = 1.0 + + # Fulldome fisheye ("planetarium dome master"). 2D only; a circular image is + # centered in the frame's inscribed circle with the corners left as the + # background (the dome master's black border). Set `width == height` (e.g. + # 4096) for a square master. When enabled, the axes and the outside colorbar + # strip are suppressed so the PNG stays exactly width x height. Seamless + # across MPI domains (the pixel->world map is a shared, deterministic function + # and the tiles stay disjoint). Ignored (with a warning) for 3D. + # @note: CARTESIAN -- the flat plane is warped radially into the disk; use + # `fov`/`radius`/`center`/`projection` below. + # @note: SPHERICAL / GR Kerr-Schild -- the meridional slice is ALREADY a + # disk, so dome mode only mirrors it to a full disk (see `mirror` + # above; keep it true) and fits it to the inscribed circle. The + # `fov`/`radius`/`center`/`projection` keys are ignored (the native + # (X, Z) meridional map is used, with image radius proportional to + # the physical radius r, r=0 at the disk center). + [render.dome] + # Build the fisheye dome master instead of the plain slice + # @type: bool + # @default: false + enable = false + # (Cartesian only) Full dome field of view in degrees (image radius maps + # linearly to the dome zenith angle: the rim is at fov/2) + # @type: float [> 0.0, <= 180.0] + # @default: 180.0 # a full hemisphere + fov = 180.0 + # (Cartesian only) World radius of the circular cutout mapped onto the dome + # @type: float [> 0.0] + # @default: half the shorter domain side (the largest centered disk that + # fits inside the box) + radius = 1.0 + # (Cartesian only) World-space center of the cutout + # @type: array [size 2] + # @default: the domain center + center = [0.0, 0.0] + # (Cartesian only) How the dome zenith angle maps to a world radius on the + # flat slice + # @type: string + # @default: "equidistant" + # @enum: "equidistant" (r proportional to angle; the fulldome image + # standard -- a straight radial scaling of the cutout), "gnomonic" + # (r ~ tan(angle); the slice as a flat "ceiling" tangent to the + # dome -- straight sim lines stay straight), "stereographic" (r ~ + # tan(angle/2); conformal, preserves shapes), "orthographic" (r ~ + # sin(angle); the slice as seen face-on) + projection = "equidistant" + + # Magnetic field lines, drawn from a coarse, MPI-replicated copy of the field + # so the geometry is global and seamless across domains (the coarsening is + # what makes this cheap -- no parallel particle advection / flux scan). + # - 3D (Cartesian): traced as solid tubes, colored by |field|, composited + # inside the volume ray-march so the volume correctly occludes them. + # - 2D (Cartesian): iso-contours of the flux function psi (Bx = d psi/dy, By = + # -d psi/dx), i.e. the in-plane field lines, colored by |B|. + # - 2D (spherical / Kerr-Schild): traced meridional streamlines of the + # poloidal (Br, Btheta) field (nt2py style). + # Built once per frame and shared by every scene that opts in (per-scene + # `fieldlines = true`) and by any standalone `field = "fieldlines"` scene. + [render.fieldlines] + # Build the field-line geometry this run + # @type: bool + # @default: false + # @note: implied true if any scene sets `fieldlines = true` or uses `field + # = "fieldlines"` + enable = false + # Vector field to trace + # @type: string + # @default: "B" + # @enum: "B", "E", "J" + field = "B" + # Field coarsening factor (simulation cells per coarse cell, per axis) + # @type: int [1..16] + # @default: 4 + # @note: larger = smoother "morphology" lines + cheaper replication (the + # coarse field is ~ N_cells / bin^D floats/rank; D = sim dimension) + bin = 4 + # (3D tubes) Seed-lattice spacing in screen pixels (sets line density) + # @type: float [> 0] + # @default: 8 + # @note: capped by `seed_max`; if seed_px asks for more seeds than that, + # the spacing grows to fit and seed_px no longer governs + seed_px = 8 + # (3D tubes) Hard cap on the seed count (lattice is n^3, 2 lines per seed) + # @type: int [> 0] + # @default: 4096 + # @note: lower this for fewer / more widely spaced lines + seed_max = 4096 + # (2D contours) Number of evenly-spaced flux-function contour levels + # @type: int [> 0] + # @default: 16 + # @note: evenly-spaced psi levels => line density tracks |B| automatically + levels = 16 + # Tube radius (3D) / contour line width (2D), in screen pixels + # @type: float [> 0] + # @default: 2 + tube_px = 2 + # Colormap for the field lines (mapped by |B| along each line) + # @type: string + # @default: "inferno" + # @enum: "viridis", "inferno", "plasma", "cool2warm", "gray", "RdBu_r", + # and the CMasher maps (BSD-3, https://cmasher.readthedocs.io): + # "dusk", "cosmic", "freeze", "apple", "gothic", "sunburst", + # "voltage", "ocean", "fusion", "prinsenvlag" (an optional "cmr." + # prefix is ok) + colormap = "inferno" + # Monochrome override: draw the lines in a single [r,g,b] color (each 0..1) + # instead of the |B| colormap -- reads well as an overlay on another volume + # @type: array [size 3] + # @default: [] (empty => color by |B|) + # @example: [1.0, 1.0, 1.0] # white field lines + color = [] + # Map the tube color range logarithmically + # @type: bool + # @default: false + # @note: requires min > 0 + log = false + # Tube color range: lower bound on |field| + # @type: float + # @default: 0.0 + # @note: when min >= max, the range is auto-set from |field| along the + # lines + min = 0.0 + # Tube color range: upper bound on |field| + # @type: float + # @default: 0.0 + # @note: when min >= max, the range is auto-set from |field| along the + # lines + max = 0.0 + # (3D tubes) RK4 integration step as a fraction of one coarse cell + # @type: float [> 0] + # @default: 0.5 + step_frac = 0.5 + # (3D tubes) Per-direction integration-step cap + # @type: int [> 0] + # @default: 4000 + max_steps = 4000 + # (3D tubes) Maximum line length, in global box diagonals (per direction) + # @type: float [> 0] + # @default: 3.0 + max_length = 3.0 + + # One scene per scalar field -> one PNG stream. Repeat the table for each. + [[render.scene]] + # Scalar field to render (a volume render needs a scalar, so vectors are + # given as a magnitude or a single component) + # @required + # @type: string + # @enum: (fields): "{E,B,J}mag"; "{E,B,J}{1,2,3}" or "{E,B,J}{x,y,z}", + # (moments): "N", "Nppc", "Rho", "Charge"; "T{i}{j}"; "V{i}"; + # "Vmag" + # @note: "{E,B,J}mag" = vector magnitude |.|; "B1"/"Bx", "J3"/"Jz", ... = + # a single (signed) physical component + # @note: a bare vector ("E"/"B"/"J") is not renderable -- choose a + # component or the magnitude + # @note: "N"/"Nppc" = number / per-cell count, "Rho" = mass density, + # "Charge" = charge density + # @note: "T{i}{j}" = one stress-energy component, i,j in {t,x,y,z} or + # {0,1,2,3} (e.g. "Txx", "Ttt", "T0x"); "V{i}" = one bulk-velocity + # component, i in {x,y,z} or {1,2,3} (e.g. "Vx", "V1"); "Vmag" = + # bulk-velocity magnitude sqrt(V1^2+V2^2+V3^2) + # @note: moments follow the engine: SRPIC = tetrad-basis bulk 3-velocity + # and stress-energy; GRPIC = Eckart-frame 4-velocity (so "Vt"/"V0" + # = u^0 = Gamma/alpha is also valid) and contravariant T + # @note: per-species selection with a "_" suffix on moments, e.g. + # "N_1", "Rho_2", "Txy_1_2", "V1_3"; default = all massive species + # @note: components are signed; pair a symmetric `min`/`max` with a + # diverging colormap ("cool2warm") to center zero + # @note: "fieldlines" renders the magnetic field-line tubes on their own + # (no scalar volume sampled); see [render.fieldlines] below + field = "" + # PNG filename prefix; files are `.png` + # @type: string + # @default: "_" + prefix = "_" + # Colorbar title + # @type: string + # @default: `field` + label = "" + # Lower bound of the value range mapped onto the colormap/opacity + # @type: float + # @default: 0.0 + min = 0.0 + # Upper bound of the value range + # @type: float + # @default: 1.0 + max = 1.0 + # Map the value range logarithmically + # @type: bool + # @default: false + # @note: Requires min > 0 and max > 0 + log = false + # Colormap name + # @type: string + # @default: "viridis" + # @enum: "viridis", "inferno", "plasma", "cool2warm", "gray", "RdBu_r", + # and the CMasher maps (BSD-3, https://cmasher.readthedocs.io): + # "dusk", "cosmic", "freeze", "apple", "gothic", "sunburst", + # "voltage", "ocean", "fusion", "prinsenvlag" (an optional "cmr." + # prefix is ok) + colormap = "viridis" + # Opacity transfer function: [position, opacity] control points, both in [0, + # 1], piecewise-linear in the normalized value + # @type: array> + # @default: linear ramp (opacity = normalized value) + # @note: Keep the low end near 0 so empty regions stay transparent + # @example: [[0.0, 0.0], [0.3, 0.1], [1.0, 0.7]] + alpha = [] + # Explicit value(s) to label on the colorbar + # @type: array + # @default: 5 evenly-spaced ticks between min and max + # @note: Values outside [min, max] are skipped + # @example: [0.0, 0.5, 1.0] + colorbar_ticks = [] + # Overlay the magnetic field-line tubes inside this scene's volume + # @type: bool + # @default: false + # @note: requires the [render.fieldlines] enabled (3D only). A scene with + # field = "fieldlines" instead renders them alone. + fieldlines = false + +# Diagnostic logging parameters +[diagnostics] + # Number of timesteps between diagnostic logs + # @type: int [> 0] + # @default: 1 + interval = 1 + # Blocking timers between successive algorithms + # @type: bool + # @default: false + blocking_timers = false + # Enable colored stdout + # @type: bool + # @default: true + colored_stdout = true + # Specify the log level + # @type: string + # @default: "VERBOSE" + # @enum: "VERBOSE", "WARNING", "ERROR" + # @note: "VERBOSE" prints all messages, "WARNING" prints only warnings and + # errors, "ERROR" prints only errors + log_level = "VERBOSE" diff --git a/input.example.toml b/input.example.toml deleted file mode 100644 index b85c5f020..000000000 --- a/input.example.toml +++ /dev/null @@ -1,772 +0,0 @@ -[simulation] - # Name of the simulation - # @required - # @type: string - # @note: The name is used for the output files - name = "" - # Simulation engine to use - # @required - # @type: string - # @enum: "SRPIC", "GRPIC" - engine = "" - # Max runtime in physical (code) units - # @required - # @type: float [> 0] - # @example: 1e5 - runtime = "" - - [simulation.domain] - # Number of domains - # @type: int - # @default: 1 [no MPI]; MPI_SIZE [MPI] - number = "" - # Decomposition of the domain (for MPI) in each of the directions - # @type: array [size 1 :->: 3] - # @default: [-1, -1, -1] - # @note: -1 means the code will determine the decomposition in the specific direction automatically - # @note: Automatic detection is either done by inference from # of MPI tasks, or by balancing the grid size on each domain - # @example: [2, 2, 2] (total of 8 domains) - decomposition = "" - - # Diffusion-style dynamic load balancing (Cartesian metrics only). - # Domain boundaries between MPI neighbors are nudged to equalize the - # active-particle count per rank. All inter-rank traffic uses only the - # existing nearest-neighbor field/particle communication paths. - [simulation.domain.load_balance] - # Enable dynamic load balancing - # @type: bool - # @default: false - enable = "" - # Run the rebalancer every `interval` timesteps (0 disables) - # @type: int - # @default: 0 - interval = "" - # Dimensions along which load is redistributed (1 = x1, 2 = x2, 3 = x3) - # @type: array of int, subset of [1, 2, 3] - # @default: [1] - dimensions = "" - # Skip rebalancing along a dim when (max - min) / mean of the per-slice - # particle count is below this fraction - # @type: float - # @default: 0.1 - tolerance = "" - # Maximum cell-shift per interior boundary per event; clamped at compile - # time to N_GHOSTS so the migrating field strip is already cached in the - # rank's ghost zone. - # @type: int - # @default: N_GHOSTS - max_shift = "" - -[grid] - # Spatial resolution of the grid - # @required - # @type: array [size 1 :->: 3] - # @note: Dimensionality is inferred from the size of this array - # @example: [1024, 1024, 1024] - resolution = "" - # Physical extent of the grid - # @required - # @type: array> [size 1 :->: 3] - # @note: For spherical geometry, only specify `[[rmin, rmax]]`, other values are set automatically - # @note: For cartesian geometry, cell aspect ratio has to be 1: `dx=dy=dz` - # @example: [[0.0, 1.0], [-1.0, 1.0]] - extent = "" - - # @inferred: - # - dim - # @brief: Dimensionality of the grid - # @type: short - # @enum: 1, 2, 3 - # @from: `grid.resolution` - - [grid.metric] - # Metric on the grid - # @required - # @type: string - # @enum: "Minkowski", "Spherical", "QSpherical", "Kerr_Schild", "QKerr_Schild", "Kerr_Schild_0" - metric = "" - # `r0` paramter for the QSpherical metric `x1 = log(r-r0)` - # @type: float [-inf -> rmin] - # @default: 0.0 - # @note: Negative values produce almost uniform grid in r - qsph_r0 = "" - # `h` paramter for the QSpherical metric `th = x2 + 2*h x2 (pi-2*x2)*(pi-x2)/pi^2` - # @type: float [-1 :->: 1] - # @default: 0.0 - qsph_h = "" - # Spin parameter for the Kerr Schild metric - # @type: float [0 :-> 1] - # @default: 0.0 - ks_a = "" - - # @inferred: - # - coord - # @brief: Coordinate system on the grid - # @type: string - # @enum: "cartesian", "spherical", "qspherical" - # @from: `grid.metric.metric` - # - ks_rh - # @brief: Size of the horizon for GR Kerr Schild - # @type: float - # @from: `grid.metric.ks_a` - # - params - # @brief: A map of all metric-specific parameters together (for easy access) - # @type: map - # @from: `grid.metric` - - [grid.boundaries] - # Boundary conditions for fields - # @required - # @type: array> [size 1 :->: 3] - # @enum: "PERIODIC", "MATCH", "FIXED", "ATMOSPHERE", "CUSTOM", "HORIZON", "CONDUCTOR" - # @note: When periodic in any of the directions, you should only set one value: [..., ["PERIODIC"], ...] - # @note: In spherical, bondaries in theta/phi are set automatically (only specify bc @ `[rmin, rmax]`): [["ATMOSPHERE", "MATCH"]] - # @note: In GR, the horizon boundary is set automatically (only specify bc @ rmax): [["MATCH"]] - # @example: [["CUSTOM", "MATCH"]] (for 2D spherical `[[rmin, rmax]]`) - fields = "" - # Boundary conditions for fields - # @required - # @type: array> [size 1 :->: 3] - # @enum: "PERIODIC", "ABSORB", "ATMOSPHERE", "CUSTOM", "REFLECT", "HORIZON" - # @note: When periodic in any of the directions, you should only set one value [..., ["PERIODIC"], ...] - # @note: In spherical, bondaries in theta/phi are set automatically (only specify bc @ `[rmin, rmax]`) [["ATMOSPHERE", "ABSORB"]] - # @note: In GR, the horizon boundary is set automatically (only specify bc @ `rmax`): [["ABSORB"]] - # @example: [["PERIODIC"], ["PERIODIC"]] - particles = "" - - [grid.boundaries.match] - # Size of the matching layer in each direction for fields in physical (code) units - # @type: float | array> - # @default: 1% of the domain size (in shortest dimension) - # @note: In spherical, this is the size of the layer in `r` from the outer wall - # @example: `ds = 1.5` (will set the same for all directions) - # @example: `ds = [[1.5], [2.0, 1.0], [1.1]]` (will duplicate 1.5 for +/- `x1` and 1.1 for +/- `x3`) - # @example: `ds = [[], [1.5], []]` (will only set for x2) - ds = "" - - [grid.boundaries.absorb] - # Size of the absorption layer for particles in physical (code) units - # @type: float - # @default: 1% of the domain size (in shortest dimension) - # @note: In spherical, this is the size of the layer in `r` from the outer wall - # @note: In cartesian, this is the same for all dimensions where applicable - ds = "" - - [grid.boundaries.atmosphere] - # Temperature of the atmosphere in units of `m0 c^2` - # @type: float - # @note: [required] if `ATMOSPHERE` is one of the boundaries - temperature = "" - # Peak number density of the atmosphere at base in units of `n0` - # @type: float - density = "" - # Pressure scale-height in physical units - # @type: float - height = "" - # Species indices of particles that populate the atmosphere - # @type: array [size 2] - species = "" - # Distance from the edge to which the gravity is imposed in physical units - # @type: float - # @default: 0.0 - # @note: 0.0 means no limit - ds = "" - - # @inferred: - # - g - # @brief: Acceleration due to imposed gravity - # @type: float - # @from: `grid.boundaries.atmosphere.temperature`, `grid.boundaries.atmosphere.height` - # @value: `temperature / height` - -[scales] - # Fiducial larmor radius - # @required - # @type: float [> 0.0] - larmor0 = "" - # Fiducial plasma skin depth - # @required - # @type: float [> 0.0] - skindepth0 = "" - - # @inferred: - # - dx0 - # @brief: fiducial minimum size of the cell - # @type: float - # @from: `grid` - # - V0 - # @brief: fiducial elementary volume - # @type: float - # @from: `grid` - # - n0 - # @brief: Fiducial number density - # @type: float - # @from: `particles.ppc0`, `grid` - # @value: `ppc0 / V0` - # - q0 - # @brief: Fiducial elementary charge - # @type: float - # @from: `scales.skindepth0`, `scales.n0` - # @value: `1 / (n0 * skindepth0^2)` - # - sigma0 - # @brief: Fiducial magnetization parameter - # @type: float - # @from: `scales.larmor0`, `scales.skindepth0` - # @value: `(skindepth0 / larmor0)^2` - # - B0 - # @brief: Fiducial magnetic field - # @type: float - # @from: `scales.larmor0` - # @value: `1 / larmor0` - # - omegaB0 - # @brief: Fiducial cyclotron frequency - # @type: float - # @from: `scales.larmor0` - # @value: `1 / larmor0` - -[radiation] - [radiation.drag] - [radiation.drag.synchrotron] - # Radiation reaction limit gamma-factor for synchrotron - # @type: float [> 0.0] - # @default: 1.0 - # @note: [required] if one of the species has `radiative_drag = "synchrotron"` - gamma_rad = "" - - [radiation.drag.compton] - # Radiation reaction limit gamma-factor for Compton drag - # @type: float [> 0.0] - # @default: 1.0 - # @note: [required] if one of the species has `radiative_drag = "compton"` - gamma_rad = "" - - [radiation.emission] - [radiation.emission.synchrotron] - # Gamma-factor of a particle emitting synchrotron photons at energy `m0 c^2` in fiducial magnetic field `B0` - # @type: float [> 1.0] - # @default: 10.0 - gamma_qed = "" - # Minimum photon energy for synchrotron emission (units of `m0 c^2`) - # @type: float [> 0.0] - # @default: 1e-4 - photon_energy_min = "" - # Weights for the emitted synchrotron photons - # @type: float [> 0.0] - # @default: 1.0 - photon_weight = "" - # Index of species for the emitted photon - # @type: ushort [> 0] - # @required - photon_species = "" - - # @inferred: - # - nominal_probability - # @brief: Nominal probability of the emission for a particle with `gamma * beta = 1`, charge-to-mass = `q0 / m0` - # @type: float - # @from: `.gamma_qed`, `.photon_weight`, `...drag.synchrotron.gamma_rad`, `scales.omegaB0`, `algorithms.timestep.dt` - # @value: `0.1 * omegaB0 * dt * (gamma_qed / gamma_rad)^2 / photon_weight` - # - nominal_photon_energy - # @brief: Nominal energy of the emitted photon for a particle with `gamma * beta = 1`, mass = `m0` - # @type: float - # @from: `.gamma_qed` - # @value: `(1 / gamma_qed)^2` - - [radiation.emission.compton] - # Gamma-factor of a particle emitting inverse Compton photons at energy `m0 c^2` in fiducial magnetic field `B0` - # @type: float [> 1.0] - # @default: 10.0 - gamma_qed = "" - # Minimum photon energy for inverse Compton emission (units of `m0 c^2`) - # @type: float [> 0.0] - # @default: 1e-4 - photon_energy_min = "" - # Weights for the emitted inverse Compton photons - # @type: float [> 0.0] - # @default: 1.0 - photon_weight = "" - # Index of species for the emitted photon - # @type: ushort [> 0] - # @required - photon_species = "" - - # @inferred: - # - nominal_probability - # @brief: Nominal probability of the emission for a particle with `gamma * beta = 1`, charge-to-mass = `q0 / m0` - # @type: float - # @from: `.gamma_qed`, `.photon_weight`, `...drag.compton.gamma_rad`, `scales.omegaB0`, `algorithms.timestep.dt` - # @value: `0.1 * omegaB0 * dt * (gamma_qed / gamma_rad)^2 / photon_weight` - # - nominal_photon_energy - # @brief: Nominal energy of the emitted photon for a particle with `gamma * beta = 1`, mass = `m0` - # @type: float - # @from: `.gamma_qed` - # @value: `(1 / gamma_qed)^2` - -[algorithms] - # Number of current smoothing passes - # @type: ushort [>= 0] - # @default: 0 - current_filters = "" - - [algorithms.timestep] - # Courant-Friedrichs-Lewy number - # @type: float [0.0 -> 1.0] - # @default: 0.95 - # @note: CFL number determines the timestep duration - CFL = "" - # Correction factor for the speed of light used in field solver - # @type: float - # @default: 1.0 - correction = "" - - # @inferred: - # - dt - # @brief: timestep duration - # @type: float - # @from: `algorithms.timestep.CFL`, `scales.dx0` - # @value: `CFL * dx0` - - [algorithms.deposit] - # Enable the current deposition - # @type: bool - # @default: true - enable = "" - # team_policy tiled-deposit work-group (team) size - # @type: uint [>= 0] - # @default: 0 - # @note: 0 keeps Kokkos::AUTO (backend occupancy heuristic); a positive - # value overrides it, clamped to the backend/scratch maximum at - # launch. Only used in `team_policy=ON` builds. Pick a multiple of - # the device subgroup width for best occupancy (see ideal_tile_size.py) - team_policy_team_size = "" - - # @inferred: - # - order - # @brief: order of the particle shape function - # @from: compile-time definition `shape_order` - # @type: ushort [0 -> 10] - - [algorithms.gr] - # Stepsize for numerical differentiation in GR pusher - # @type: float [> 0.0] - # @default: 1e-6 - pusher_eps = "" - # Number of iterations for the Newton-Raphson method in GR pusher - # @type: ushort [> 0] - # @default: 10 - pusher_niter = "" - - [algorithms.gca] - # Maximum value for E/B allowed for GCA particles - # @type: float [0.0 -> 1.0] - # @default: 0.9 - e_ovr_b_max = "" - # Maximum Larmor radius allowed for GCA particles (in physical units) - # @type: float - # @default: 0.0 - # @note: When `larmor_max` == 0, the limit is disabled - larmor_max = "" - - # Stencil coefficients for the field solver [notation as in Blinne+ (2018)] - # @note: Standard Yee solver: `delta_i = beta_ij = 0.0` - [algorithms.fieldsolver] - # Enable the fieldsolver - # @type: bool - # @default: true - enable = "" - # delta_x coefficient (for `F_{i +/- 3/2, j, k}`) - # @type: float - # @default: 0.0 - delta_x = "" - # delta_y coefficient (for `F_{i, j +/- 3/2, k}`) - # @type: float - # @default: 0.0 - # @note: Used only for 2D and 3D - delta_y = "" - # delta_z coefficient (for `F_{i, j, k +/- 3/2}`) - # @type: float - # @default: 0.0 - # @note: Used only for 3D - delta_z = "" - # beta_xy coefficient (for `F_{i +/- 1/2, j +/- 1, k}`) - # @type: float - # @default: 0.0 - # @note: Used only for 2D and 3D - beta_xy = "" - # beta_yx coefficient (for `F_{i +/- 1, j +/- 1/2, k}`) - # @type: float - # @default: 0.0 - # @note: Used only for 2D and 3D - beta_yx = "" - # beta_xz coefficient (for `F_{i +/- 1/2, j, k +/- 1}`) - # @type: float - # @default: 0.0 - # @note: Used only for 3D - beta_xz = "" - # beta_zx coefficient (for `F_{i +/- 1, j, k +/- 1/2}`) - # @type: float - # @default: 0.0 - # @note: Used only for 3D - beta_zx = "" - # beta_yz coefficient (for `F_{i, j +/- 1/2, k +/- 1}`) - # @type: float - # @default: 0.0 - # @note: Used only for 3D - beta_yz = "" - # beta_zy coefficient (for `F_{i, j +/- 1, k +/- 1/2}`) - # @type: float - # @default: 0.0 - # @note: Used only for 3D - beta_zy = "" - -[particles] - # Fiducial number of particles per cell - # @required - # @type: float [> 0.0] - ppc0 = "" - # Toggle for using particle weights - # @type: bool - # @default: false - use_weights = "" - # Timesteps between particle re-sorting by tags (removing dead particles) - # @type: uint - # @default: 100 - # @note: Set to 0 to disable re-sorting - clear_interval = "" - # Timesteps between spatial sorting of particles (for better cache performance) - # @type: uint - # @default: 0 - # @note: Set to 0 to disable spatial sorting - spatial_sorting_interval = "" - - # @inferred: - # - nspec - # @brief: Number of particle species - # @type: uint - # @from: `particles.species` - - [[particles.species]] - # Label of the species - # @type: string - # @default: "s" - # @example: "e-" - # @note: `` is the index of the species in the list starting from 1 - label = "" - # Mass of the species (in units of fiducial mass) - # @required - # @type: float [>= 0.0] - mass = "" - # Charge of the species (in units of fiducial charge) - # @required - # @type: float - charge = "" - # Maximum number of particles per task - # @required - # @type: uint [> 0] - maxnpart = "" - # Pusher algorithm for the species - # @type: string - # @default: "Boris" [massive]; "Photon" [massless] - # @enum: "Boris", "Vay", "Boris,GCA", "Vay,GCA", "Photon", "None" - pusher = "" - # Number of additional real-valued variables (payloads) for each particle of the given species - # @type: ushort - # @default: 0 - n_payloads_real = "" - # Number of additional integer-valued variables (payloads) for each particle of the given species - # @type: ushort - # @default: 0 - # @note: If tracking is enabled, one or two extra integer payloads are reserved (depending on whether MPI is enabled) - n_payloads_int = "" - # Enable tracking of particles using indices for the given species - # @type: bool - # @default: false - tracking = "" - # Radiation reaction to use for the species - # @type: string - # @default: "None" - # @enum: "None", "Synchrotron", "Compton" - # @note: Can also be coma-separated combination, e.g., "Synchrotron,Compton" - # @note: Relevant radiation.drag parameters should also be provided - radiative_drag = "" - # Particle emission policy for the species - # @type: string - # @default: "None" - # @enum: "None", "Synchrotron", "Compton" - # @note: Only one emission mechanism allowed - # @note: Appropriate radiation drag flag will be applied automatically (unless explicitly set to "None") - emission = "" - # Timesteps between spatial sorting of particles for given species - # @type: uint - # @default: 0 - # @note: Set to 0 to disable spatial sorting - # @note: Overrides `particles.spatial_sorting_interval` for the given species - spatial_sorting_interval = "" - # Timesteps between particle re-sorting by tags (removing dead particles) - # @type: uint - # @default: 100 - # @note: Set to 0 to disable re-sorting - # @note: Overrides `particles.clear_interval` for the given species - clear_interval = "" - -# Parameters for specific problem generators and setups -[setup] - -[output] - # Output format - # @type: string - # @default: "hdf5" - # @enum: "disabled", "hdf5", "BPFile" - format = "" - # Number of timesteps between all outputs - # @type: uint [> 0] - # @default: 1 - # @note: Value is overriden by output intervals for specific outputs - interval = "" - # Physical (code) time interval between all outputs - # @type: float - # @default: -1.0 - # @note: When `interval_time` < 0, the output is controlled by `interval`, otherwise by `interval_time` - # @note: Value is overriden by output intervals for specific outputs - interval_time = "" - # Whether to output each timestep into separate files - # @type: bool - # @default: true - # @deprecated: starting v1.3.0 - separate_files = "" - - [output.fields] - # Toggle for the field output - # @type: bool - # @default: true - enable = "" - # Field quantities to output - # @type: array - # @default: [] - # @enum: "E", "B", "J", "divE", "Rho", "Charge", "N", "Nppc", "T0i", "Tij", "Vi", "D", "H", "divD", "A" - # @note: For `T`, you can use unspecified indices: `Tij`, `T0i`, or specific ones: `Ttt`, `T00`, `T02`, `T23` - # @note: For `T`, in cartesian can also use "x" "y" "z" instead of "1" "2" "3" - # @note: By default, we accumulate moments from all massive species, one can specify only specific species: `Ttt_1_2`, `Rho_1`, `Rho_3_4` - quantities = "" - # Custom (user-defined) field quantities - # @type: array - # @default: [] - custom = "" - # Number of timesteps between field outputs - # @type: uint - # @default: 0 - # @note: When `!= 0`, overrides `output.interval` - # @note: When `== 0`, `output.interval` is used - interval = "" - # Physical (code) time interval between field outputs - # @type: float - # @default: -1.0 - # @note: When `< 0`, the output is controlled by `interval` - # @note: When specified, overrides `output.interval_time` - interval_time = "" - # Downsample factor for the output of fields - # @type: uint | array [>= 1] - # @default: [1, 1, 1] - # @note: The output is downsampled by the given factors in each direction - # @note: If a scalar is given, it is applied to all directions - downsampling = "" - - [output.fields.smoothing] - # Smoothing order for the output of moments ("Rho", "Charge", "T", ...) - # @type: ushort - # @default: 0 - order = "" - # Smoothing algorithm - # @type: string - # @enum: "const", "spline" - # @default: "spline" - # @note: When using "spline", `order` corresponds to the order of the polynomial used - # @note: When using "const", the smoothing window is `ceil(order / 2)` in both directions - method = "" - - [output.particles] - # Toggle for the particles output - # @type: bool - # @default: true - enable = "" - # Particle species indices to output - # @type: array - # @default: [] - # @note: If empty, all species are output - species = "" - # Stride for the output of particles - # @type: uint [> 1] - # @default: 100 - stride = "" - # Number of timesteps between particle outputs - # @type: uint - # @default: 0 - # @note: When `!= 0`, overrides `output.interval` - # @note: When `== 0`, `output.interval` is used - interval = "" - # Physical (code) time interval between particle outputs - # @type: float - # @default: -1.0 - # @note: When `< 0`, the output is controlled by `interval` - # @note: When specified, overrides `output.interval_time` - interval_time = "" - - [output.spectra] - # Toggle for the spectra output - # @type: bool - # @default: true - enable = "" - # Minimum energy for the spectra output - # @type: float - # @default: 1e-3 - e_min = "" - # Maximum energy for the spectra output - # @type: float - # @default: 1e3 - e_max = "" - # Whether to use logarithmic bins for energy - # @type: bool - # @default: true - log_bins = "" - # Number of energy bins for the spectra output - # @type: uint [> 0] - # @default: 200 - num_energy_bins = "" - # Number of spatial bins for the spectra output - # @type: array [size 1 :->: 3] - # @default: [1, 1, 1] - num_spatial_bins = "" - # Number of timesteps between spectra outputs - # @type: uint - # @default: 0 - # @note: When `!= 0`, overrides `output.interval` - # @note: When `== 0`, `output.interval` is used - interval = "" - # Physical (code) time interval between spectra outputs - # @type: float - # @default: -1.0 - # @note: When `< 0`, the output is controlled by `interval` - # @note: When specified, overrides `output.interval_time` - interval_time = "" - - [output.debug] - # Output fields "as is" without conversions - # @type: bool - # @default: false - as_is = "" - # Output fields with values in ghost cells - # @type: bool - # @default: false - ghosts = "" - - [output.stats] - # Toggle for the stats output - # @type: bool - # @default: true - enable = "" - # Number of timesteps between stat outputs - # @type: uint [> 0] - # @default: 100 - # @note: Overriden if `output.stats.interval_time != -1` - interval = "" - # Physical (code) time interval between stat outputs - # @type: float - # @default: -1.0 - # @note: When `< 0`, the output is controlled by `interval` - interval_time = "" - # Field quantities to output - # @type: array - # @default: ["B^2", "E^2", "ExB", "Rho", "T00"] - # @enum: "B^2", "E^2", "ExB", "N", "Npart", "Charge", "Rho", "T00", "T0i", "Tij" - # @note: For particle moments, ... - # @note: ... same notation is used as for `output.fields.quantities` - quantities = "" - # Custom (user-defined) stats - # @type: array - # @default: [] - custom = "" - -[checkpoint] - # Number of timesteps between checkpoints - # @type: uint [> 0] - # @default: 1000 - interval = "" - # Physical (code) time interval between checkpoints - # @type: float [> 0] - # @default: -1.0 - # @note: When `< 0`, the output is controlled by `interval` - interval_time = "" - # Number of checkpoints to keep - # @type: int - # @default: 2 - # @note: 0 = disable checkpointing - # @note: -1 = keep all checkpoints - keep = "" - # Write a checkpoint once after a fixed walltime - # @type: string - # @default: "00:00:00" - # @note: The format is "HH:MM:SS" - # @note: Empty string or "00:00:00" disables this functionality - # @note: Writing checkpoint at walltime does not stop the simulation - walltime = "" - # Parent directory to write checkpoints to - # @type: string - # @default: `.ckpt` - # @note: The directory is created if it does not exist - write_path = "" - # Parent directory to use when resuming from a checkpoint - # @type: string - # @default: inherit `write_path` - read_path = "" - - # @inferred: - # - is_resuming - # @brief: Whether the simulation is resuming from a checkpoint - # @type: bool - # @from: command-line flag - # - start_step - # @brief: Timestep of the checkpoint used to resume - # @type: uint - # @from: automatically determined during restart - # - start_time - # @brief: Time of the checkpoint used to resume - # @type: float - # @from: automatically determined during restart - -[adios2] - # ADIOS2 BP5 tuning, applied to both [output] and [checkpoint] writers - # Number of ADIOS2 aggregators per node - # @type: uint - # @default: 0 - # @note: Set to either MPI ranks/node or NICs/node for best performance - # If set to 0, will use ADIOS2 default (one aggregator per node) - aggregators_per_node = "" - # Maximum shared-memory segment size per node, in bytes (BP5 MaxShmSize) - # @type: uint - # @default: 4294967296 - # @note: Lower this on memory-constrained nodes; matches ADIOS2's default - max_shm_size = "" - # Internal serialization buffer chunk size, in bytes (BP5 BufferChunkSize) - # @type: uint - # @default: 16777216 - # @note: Scales with per-rank output volume; matches ADIOS2's default - buffer_chunk_size = "" - -[diagnostics] - # Number of timesteps between diagnostic logs - # @type: int [> 0] - # @default: 1 - interval = "" - # Blocking timers between successive algorithms - # @type: bool - # @default: false - blocking_timers = "" - # Enable colored stdout - # @type: bool - # @default: true - colored_stdout = "" - # Specify the log level - # @type: string - # @default: "VERBOSE" - # @enum: "VERBOSE", "WARNING", "ERROR" - # @note: "VERBOSE" prints all messages, "WARNING" prints only warnings and errors, "ERROR" prints only errors - log_level = "" diff --git a/pgens/shock/pgen.hpp b/pgens/shock/pgen.hpp index 7a7aaac21..223e05696 100644 --- a/pgens/shock/pgen.hpp +++ b/pgens/shock/pgen.hpp @@ -120,7 +120,7 @@ namespace user { return init_flds; } - auto FixFieldsConst(const bc_in&, const em& comp) const + auto FixFieldsConst(simtime_t, const bc_in&, const em& comp) const -> std::pair { if (comp == em::ex1) { return { init_flds.ex1({ ZERO }), true }; diff --git a/dependencies.py b/scripts/dependencies.py similarity index 100% rename from dependencies.py rename to scripts/dependencies.py diff --git a/scripts/generate_template.py b/scripts/generate_template.py new file mode 100755 index 000000000..4695683a8 --- /dev/null +++ b/scripts/generate_template.py @@ -0,0 +1,505 @@ +#!/usr/bin/env python3 +"""Generate `input.default.toml` from `entity.schema.json`. + +The JSON Schema is the single source of truth for the input file: it drives editor +validation/completion (tombi) *and* the annotated reference input that ships with the +code. This script renders the second from the first, so the two can never drift. + +The schema splits into two halves: + + * Standard JSON Schema keywords -- `type`, `enum`, `minimum`, `items`, `required`, + `default`, ... -- carry everything a validator can check. + * An `x-entity` object per node carries what JSON Schema cannot express, verbatim from + the template's comment annotations: `type` (the literal `@type:` string, e.g. + "array [size 1 :->: 3]"), `default` (for non-JSON defaults such as + "1 [no MPI]; MPI_SIZE [MPI]"), `notes`, `examples`, `enum` (an illustrative, + NON-exhaustive list -- never validated), and `deprecated`. + +`x-entity.inferred` on a table lists quantities the code derives rather than reads. They +are deliberately absent from `properties` (so `additionalProperties: false` rejects them +as input keys) and are emitted here as an `@inferred:` comment block, after that table's +own keys and before its sub-tables. + +Layout rules, matching the hand-written template: + + * a table at depth d gets indent 2*d, its keys 2*(d+1) + * within a table: scalar keys, then the `@inferred:` block, then sub-tables + * a blank line precedes every sub-table (matching `tombi format`) + * every key gets a value: its default under `--defaults`, otherwise `""` -- a blank + form to fill in + +Usage: + + python scripts/generate_template.py -d -o input.default.toml # the reference input + python scripts/generate_template.py -d # ... to stdout + python scripts/generate_template.py # blank form, values "" + diff <(python scripts/generate_template.py -d) input.default.toml + +With `--defaults` every key carries a value instead of `""`: the literal `x-entity.default` +where it is one, else the JSON `default`, else a stand-in derived from the schema's own +constraints (first enum value, `minimum`, `minItems`, ...). That last case covers the +required keys, which the user must supply anyway, and the handful whose documented +default the code computes at runtime -- `N_GHOSTS`, "1% of the domain size", "box centre +pushed back ~1.7 box-diagonals". Those are listed on stderr, and their `@default:` or +`@required` annotation still spells out the real behaviour. +""" + +from __future__ import annotations + +import argparse +import json +import sys +import textwrap +import tomllib +from pathlib import Path +from typing import Any + +REPO = Path(__file__).resolve().parent.parent +DEFAULT_SCHEMA = REPO / "entity.schema.json" + +INDENT = " " + +# order of the `@`-annotations inside a key's comment block +ANNOTATION_ORDER = ("required", "type", "default", "deprecated", "enum", "note", "example") + + +# --------------------------------------------------------------------------- +# schema helpers +# --------------------------------------------------------------------------- + + +def resolve(node: dict, defs: dict) -> dict: + """Follow `$ref`, keeping any sibling keywords (2020-12 allows them).""" + while "$ref" in node: + target = defs[node["$ref"].rsplit("/", 1)[-1]] + merged = dict(target) + merged.update({k: v for k, v in node.items() if k != "$ref"}) + node = merged + return node + + +def kind(node: dict, defs: dict) -> str: + """'table' (TOML table), 'aot' (array of tables), or 'key' (scalar/array value).""" + if node.get("type") == "object" or "properties" in node: + return "table" + items = node.get("items") + if node.get("type") == "array" and isinstance(items, dict): + if resolve(items, defs).get("type") == "object": + return "aot" + return "key" + + +def schema_enum(node: dict, defs: dict) -> list | None: + """First real `enum` reachable through anyOf/oneOf/items (arrays of enums).""" + if "enum" in node: + return node["enum"] + for branch in ("anyOf", "oneOf"): + for sub in node.get(branch, []): + found = schema_enum(resolve(sub, defs), defs) + if found: + return found + items = node.get("items") + if isinstance(items, dict): + return schema_enum(resolve(items, defs), defs) + return None + + +def own_enum(node: dict, defs: dict) -> list | None: + """The node's own `enum`, including through anyOf/oneOf -- but NOT through `items`. + + Unlike `schema_enum`, this does not descend into array elements: it answers "what + values may THIS node take", which is what a placeholder needs. + """ + if "enum" in node: + return node["enum"] + for branch in ("anyOf", "oneOf"): + for sub in node.get(branch, []): + found = own_enum(resolve(sub, defs), defs) + if found: + return found + return None + + +def node_type(node: dict, defs: dict) -> str | None: + """First concrete JSON type of the node, looking into anyOf/oneOf branches.""" + t = node.get("type") + if isinstance(t, list): + return t[0] + if t: + return t + for branch in ("anyOf", "oneOf"): + for sub in node.get(branch, []): + found = node_type(resolve(sub, defs), defs) + if found: + return found + return None + + +def placeholder(node: dict, defs: dict) -> Any: + """A constraint-respecting stand-in for a key the schema gives no default for. + + Used for required keys (which the user must fill in anyway) and for keys whose + documented default is computed at runtime and so has no literal form -- N_GHOSTS, + "1% of the domain size", "box center pushed back ~1.7 box-diagonals", ... + """ + node = resolve(node, defs) + + values = own_enum(node, defs) + if values: + return values[0] + + # an anyOf/oneOf node keeps its constraints inside the branches, so pick the first + # branch that names a type and derive the stand-in from that + if "type" not in node: + for branch in ("anyOf", "oneOf"): + for sub in node.get(branch, []): + sub = resolve(sub, defs) + if node_type(sub, defs): + return placeholder(sub, defs) + + t = node_type(node, defs) + if t == "boolean": + return False + if t in ("integer", "number"): + lo = node.get("minimum") + if lo is None and "exclusiveMinimum" in node: + lo = node["exclusiveMinimum"] + 1 + value = lo if lo is not None else 0 + hi = node.get("maximum") + if hi is not None: + value = min(value, hi) + return int(value) if t == "integer" else float(value) + if t == "string": + return "" + if t == "array": + if "prefixItems" in node: + return [placeholder(i, defs) for i in node["prefixItems"]] + items = node.get("items") + count = node.get("minItems", 0) + if count and isinstance(items, dict): + return [placeholder(items, defs) for _ in range(count)] + return [] + if t == "object": + return {} + return "" + + +def derive_type(node: dict, defs: dict) -> str: + """Fallback `@type` when x-entity.type is absent (it should never be).""" + t = node.get("type") + if isinstance(t, list): + return " | ".join(t) + if t == "array": + items = node.get("items") + inner = derive_type(resolve(items, defs), defs) if isinstance(items, dict) else "any" + return f"array<{inner}>" + if t: + return t + for branch in ("anyOf", "oneOf"): + if branch in node: + return " | ".join(derive_type(resolve(s, defs), defs) for s in node[branch]) + return "any" + + +# --------------------------------------------------------------------------- +# value / annotation formatting +# --------------------------------------------------------------------------- + + +def toml_value(value: Any) -> str: + """Render a JSON default as the TOML literal it corresponds to.""" + if isinstance(value, bool): # before int -- bool is an int subclass + return "true" if value else "false" + if isinstance(value, str): + return json.dumps(value) + if isinstance(value, (int, float)): + return repr(value) + if isinstance(value, list): + return "[" + ", ".join(toml_value(v) for v in value) + "]" + if value is None: + return '""' + return str(value) + + +def as_toml_literal(text: str) -> str | None: + """Return `text` if it is already a standalone TOML value, else None. + + `x-entity.default` is prose more often than not ("N_GHOSTS", "1 [no MPI]; MPI_SIZE + [MPI]"), but when it *is* a literal it is the better source than the JSON `default`, + because it preserves the notation the docs use -- 1e-4 rather than 0.0001. Anything + carrying a comment marker is rejected so trailing asides do not leak into the value. + """ + if "#" in text: + return None + try: + tomllib.loads(f"x = {text}") + except (tomllib.TOMLDecodeError, ValueError): + return None + return text + + +def format_enum(values: list) -> str: + """Join enum values for an `@enum:` line. + + An `x-entity.enum` entry that already carries quotes or spaces is documentation prose + (e.g. the CMasher colormap aside) and is passed through untouched; a bare token is + quoted so it reads as the literal you would type. + """ + out = [] + for v in values: + if isinstance(v, str) and ('"' in v or " " in v): + out.append(v) + else: + out.append(toml_value(v)) + return ", ".join(out) + + +def wrap(text: str, initial: str, subsequent: str, width: int) -> list[str]: + """Wrap `text` to `width`, honouring embedded newlines as hard breaks.""" + lines: list[str] = [] + for i, chunk in enumerate(text.split("\n")): + prefix = initial if i == 0 else subsequent + chunk = chunk.rstrip() + if not chunk: + lines.append(prefix.rstrip()) + continue + lines.extend( + textwrap.wrap( + chunk, + width=width, + initial_indent=prefix, + subsequent_indent=subsequent, + break_long_words=False, + break_on_hyphens=False, + ) + or [prefix.rstrip()] + ) + return lines + + +def describe(text: str, indent: str, width: int) -> list[str]: + """A plain `# ...` description block.""" + return wrap(text, f"{indent}# ", f"{indent}# ", width) + + +def annotate(tag: str, text: str | None, indent: str, width: int) -> list[str]: + """A `# @tag: ...` line, continuations aligned under the text.""" + if text is None: + return [f"{indent}# @{tag}"] + initial = f"{indent}# @{tag}: " + return wrap(text, initial, f"{indent}# " + " " * (len(tag) + 3), width) + + +# --------------------------------------------------------------------------- +# rendering +# --------------------------------------------------------------------------- + + +class Renderer: + def __init__(self, schema: dict, width: int, defaults: bool = False) -> None: + self.defs = schema.get("$defs", {}) + self.width = width + self.defaults = defaults + self.out: list[str] = [] + # keys we had to invent a stand-in for, reported at the end + self.synthesized: list[str] = [] + + def render(self, schema: dict) -> str: + # the root behaves like a table at depth -1: no keys of its own, and its + # sub-tables land at depth 0 + self.render_body(schema, "", -1) + return "\n".join(self.out) + "\n" + + def render_body(self, node: dict, path: str, depth: int) -> None: + indent = INDENT * (depth + 1) + props = node.get("properties") or {} + required = set(node.get("required") or []) + + keys, subtables = [], [] + for name, raw in props.items(): + resolved = resolve(raw, self.defs) + entry = (name, raw, resolved) + (keys if kind(resolved, self.defs) == "key" else subtables).append(entry) + + wrote = False + for name, raw, resolved in keys: + self.emit_key( + name, raw, resolved, indent, name in required, f"{path}.{name}" if path else name + ) + wrote = True + + inferred = (node.get("x-entity") or {}).get("inferred") or [] + if inferred: + if wrote: + self.out.append("") + self.emit_inferred(inferred, indent) + wrote = True + + for name, raw, resolved in subtables: + # tombi puts a blank line before every sub-table, including the first one in + # a parent that has no keys of its own ([radiation] -> [radiation.drag]); + # `self.out` being non-empty is just "not the very first line of the file" + if wrote or self.out: + self.out.append("") + self.emit_table(name, raw, resolved, path, depth + 1) + wrote = True + + def key_value(self, name: str, node: dict, path: str, required: bool) -> str: + """The right-hand side of `name = ...`. + + In template mode every key is an empty string -- a form to fill in. In defaults + mode the precedence is: the literal `x-entity.default`, then the JSON `default`, + then a constraint-derived placeholder (recorded, since it is not a real default). + """ + if not self.defaults: + return '""' + + xe = node.get("x-entity") or {} + documented = xe.get("default") + if isinstance(documented, str): + literal = as_toml_literal(documented) + if literal is not None: + return literal + if "default" in node: + return toml_value(node["default"]) + + reason = "required" if required else (documented or "no documented default") + self.synthesized.append(f"{path} ({reason})") + return toml_value(placeholder(node, self.defs)) + + def emit_key( + self, name: str, raw: dict, node: dict, indent: str, required: bool, path: str = "" + ) -> None: + xe = node.get("x-entity") or {} + description = raw.get("description") or node.get("description") + if description: + self.out += describe(description, indent, self.width) + + if required: + self.out += annotate("required", None, indent, self.width) + + self.out += annotate( + "type", xe.get("type") or derive_type(node, self.defs), indent, self.width + ) + + default = xe.get("default") + if default is None and "default" in node: + default = toml_value(node["default"]) + if default is not None: + self.out += annotate("default", default, indent, self.width) + + if xe.get("deprecated"): + self.out += annotate("deprecated", xe["deprecated"], indent, self.width) + + values = xe.get("enum") or schema_enum(node, self.defs) + if values: + self.out += annotate("enum", format_enum(values), indent, self.width) + + for note in xe.get("notes", []): + self.out += annotate("note", note, indent, self.width) + for example in xe.get("examples", []): + self.out += annotate("example", example, indent, self.width) + + value = self.key_value(name, node, path, required) + self.out.append(f"{indent}{name} = {value}") + + def emit_table(self, name: str, raw: dict, node: dict, parent: str, depth: int) -> None: + indent = INDENT * depth + path = f"{parent}.{name}" if parent else name + is_aot = kind(node, self.defs) == "aot" + + description = raw.get("description") or node.get("description") + if description: + self.out += describe(description, indent, self.width) + for note in (node.get("x-entity") or {}).get("notes", []): + self.out += annotate("note", note, indent, self.width) + + self.out.append(f"{indent}[[{path}]]" if is_aot else f"{indent}[{path}]") + + body = resolve(node["items"], self.defs) if is_aot else node + self.render_body(body, path, depth) + + def emit_inferred(self, entries: list[dict], indent: str) -> None: + self.out.append(f"{indent}# @inferred:") + cont = f"{indent}# " + " " * 8 + for entry in entries: + self.out.append(f"{indent}# - {entry['name']}") + for tag in ("brief", "type", "enum", "from", "value"): + if tag not in entry: + continue + text = format_enum(entry[tag]) if tag == "enum" else str(entry[tag]) + self.out += wrap(text, f"{indent}# @{tag}: ", cont, self.width) + + +# --------------------------------------------------------------------------- + + +def main() -> int: + ap = argparse.ArgumentParser( + description="Generate the annotated input template from entity.schema.json.", + formatter_class=argparse.RawDescriptionHelpFormatter, + ) + ap.add_argument( + "schema", + nargs="?", + type=Path, + default=DEFAULT_SCHEMA, + help=f"JSON Schema to render (default: {DEFAULT_SCHEMA.name} at the repo root)", + ) + ap.add_argument( + "-o", + "--output", + type=Path, + default=None, + help="write here instead of stdout", + ) + ap.add_argument( + "-d", + "--defaults", + action="store_true", + help="fill each key with its default value instead of an empty string", + ) + ap.add_argument( + "-w", + "--width", + type=int, + default=80, + help="column at which comments wrap (default: 80)", + ) + args = ap.parse_args() + + try: + schema = json.loads(args.schema.read_text()) + except FileNotFoundError: + print(f"error: no such schema: {args.schema}", file=sys.stderr) + return 1 + except json.JSONDecodeError as err: + print(f"error: {args.schema} is not valid JSON: {err}", file=sys.stderr) + return 1 + + renderer = Renderer(schema, args.width, defaults=args.defaults) + text = renderer.render(schema) + + if args.output is None: + sys.stdout.write(text) + else: + args.output.write_text(text) + print( + f"wrote {args.output} ({text.count(chr(10))} lines) from {args.schema.name}", + file=sys.stderr, + ) + + if renderer.synthesized: + print( + f"note: {len(renderer.synthesized)} key(s) have no literal default; " + "a constraint-derived stand-in was used (the @default/@required annotation " + "above each one still documents the real behaviour):", + file=sys.stderr, + ) + for entry in renderer.synthesized: + print(f" {entry}", file=sys.stderr) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/ideal_tile_size.py b/scripts/ideal_tile_size.py old mode 100644 new mode 100755 similarity index 96% rename from ideal_tile_size.py rename to scripts/ideal_tile_size.py index d9e9d32ea..784065161 --- a/ideal_tile_size.py +++ b/scripts/ideal_tile_size.py @@ -8,7 +8,7 @@ TE = T_TILE + 2*HALO HALO = stencil_reach + drift (stencil_reach = shape_order for Esirkepov, 2 for the O==0 zigzag deposit; drift = the compile-time - `team_policy_drift` CMake knob, NOT the runtime + `tiled_deposit_drift` CMake knob, NOT the runtime spatial_sorting_interval -- see kernels/deposition/currents/tiled.hpp) the tile size is squeezed by three competing pressures: @@ -25,9 +25,9 @@ Recommendation = the largest tile that respects the particle budget and shared-memory residency; if that tile would be mostly halo, it is grown (toward lower halo) up to the shared-memory limit. This is a first-order model -- confirm by sweeping the entity knobs - -D team_policy_tile_size= -D team_policy_drift= and re-profiling (see roofline/). + -D tiled_deposit_tile_size= -D tiled_deposit_drift= and re-profiling (see roofline/). The team (work-group) size defaults to Kokkos::AUTO; override it at runtime with the - [algorithms.deposit] team_policy_team_size = (0 = AUTO) + [algorithms.deposit] tiled_deposit_team_size = (0 = AUTO) toml knob -- clamped to the backend maximum at launch (engines/srpic/currents.h). Two ways to drive it: @@ -99,14 +99,14 @@ def __init__(self): self.shape_order = 2 # entity shape_order self.precision = "single" # single / double self.components = 3 # current-field components (J has 3) - self.drift = 1 # team_policy_drift: cells of drift the scratch halo absorbs + self.drift = 1 # tiled_deposit_drift: cells of drift the scratch halo absorbs # (compile-time CMake knob, independent of spatial_sorting_interval) self.target_resident = 2 # work-groups resident per compute unit self.npart_cap = 1600.0 # particle-per-tile budget (contention / load-balance proxy) self.halo_max = 0.70 # halo fraction above which the tile is grown self.grid = 0 # cells per dim (0 disables the GPU-fill check) self.balance_factor = 4 # min tiles per compute unit - self.min_tile = 4 # entity's team_policy_tile_sizes list starts at 4 + self.min_tile = 4 # entity's tiled_deposit_tile_sizes list starts at 4 self.max_tile = 64 @@ -121,7 +121,7 @@ def resolve_arch(name): def recommend(hw, p): """p: Settings or argparse namespace. Returns dict with rows, chosen row, binding.""" # Matches DepositCurrentsTiled_kernel: STENCIL_REACH = O for Esirkepov (O>=1), - # 2 for the O==0 zigzag deposit; HALO = STENCIL_REACH + TEAM_POLICY_DRIFT. + # 2 for the O==0 zigzag deposit; HALO = STENCIL_REACH + TILED_DEPOSIT_DRIFT. stencil_reach = 2 if p.shape_order == 0 else p.shape_order halo = stencil_reach + p.drift real = PRECISION[p.precision] @@ -198,7 +198,7 @@ def report_lines(name, key, hw, p, res): reach = 2 if p.shape_order == 0 else p.shape_order reach_kind = "zigzag" if p.shape_order == 0 else "Esirkepov O" L.append(" HALO = stencil_reach + drift = %d + %d = %d -> TE = T_TILE + %d" - " (reach %d = %s; drift = team_policy_drift)" + " (reach %d = %s; drift = tiled_deposit_drift)" % (reach, p.drift, res["halo"], 2 * res["halo"], reach, reach_kind)) L.append(" shared mem %s KiB/%s (budget %s KiB for %d resident WGs); subgroup=%d, n_cu=%d" % (kib(hw["smem_cu"]), hw["cu"], kib(hw["smem_cu"] / p.target_resident), @@ -228,17 +228,17 @@ def report_lines(name, key, hw, p, res): L.append(" %.1f KiB scratch/team, %d work-groups resident/%s, %.0f particles/team, %.0f%% halo" % (c["scratch"] / 1024.0, c["resident"], hw["cu"], c["npart"], 100 * c["halo_frac"])) team = min(hw["max_wg"], 256 - 256 % hw["subgroup"]) - extra = "" if c["T"] <= 16 else " (entity's team_policy_tile_sizes list stops at 16; extend it)" - L.append(" entity build: -D team_policy=ON -D team_policy_tile_size=%d -D team_policy_drift=%d%s" + extra = "" if c["T"] <= 16 else " (entity's tiled_deposit_tile_sizes list stops at 16; extend it)" + L.append(" entity build: -D tiled_deposit=ON -D tiled_deposit_tile_size=%d -D tiled_deposit_drift=%d%s" % (min(c["T"], 16), p.drift, extra)) L.append(" team (work-group) size: Kokkos::AUTO by default; to override, set in the toml") - L.append(" [algorithms.deposit] team_policy_team_size = %d (0 = AUTO; keep a multiple of" + L.append(" [algorithms.deposit] tiled_deposit_team_size = %d (0 = AUTO; keep a multiple of" % team) L.append(" subgroup=%d), then sweep around it and re-profile" % hw["subgroup"]) # contextual guidance if c["halo_frac"] > p.halo_max: if p.drift > 1: - L.append(" !! %.0f%% of the tile is halo, inflated by team_policy_drift=%d; lower it " + L.append(" !! %.0f%% of the tile is halo, inflated by tiled_deposit_drift=%d; lower it " "(and sort at least that often via spatial_sorting_interval)" % (100 * c["halo_frac"], p.drift)) else: @@ -775,7 +775,7 @@ def cyc_prec(): ), MenuItem( "drift", - "team_policy_drift CMake knob: cells the scratch halo absorbs (>= spatial_sorting_interval)", + "tiled_deposit_drift CMake knob: cells the scratch halo absorbs (>= spatial_sorting_interval)", right=lambda: str(self.s.drift), on_enter=lambda: self.edit_int("drift", "drift", minv=0), ), @@ -926,7 +926,7 @@ def run_cli(argv) -> int: ap.add_argument("--precision", choices=("single", "double"), default="single") ap.add_argument("--components", type=int, default=3, help="current-field components (J has 3)") ap.add_argument("--drift", type=int, default=1, - help="team_policy_drift CMake knob (compile-time): cells of drift the scratch " + help="tiled_deposit_drift CMake knob (compile-time): cells of drift the scratch " "halo absorbs; size it >= spatial_sorting_interval") ap.add_argument("--target-resident", type=int, default=2, help="work-groups resident per compute unit") ap.add_argument("--npart-cap", type=float, default=1600, @@ -935,7 +935,7 @@ def run_cli(argv) -> int: ap.add_argument("--grid", type=int, default=0, help="cells per dim (optional; enables a GPU-fill check)") ap.add_argument("--balance-factor", type=int, default=4, help="min tiles per compute unit") ap.add_argument("--min-tile", type=int, default=4, - help="smallest T_TILE to consider (entity's team_policy_tile_sizes starts at 4)") + help="smallest T_TILE to consider (entity's tiled_deposit_tile_sizes starts at 4)") ap.add_argument("--max-tile", type=int, default=64) p = ap.parse_args(argv) diff --git a/scripts/render.py b/scripts/render.py new file mode 100755 index 000000000..ffe11606e --- /dev/null +++ b/scripts/render.py @@ -0,0 +1,1029 @@ +#!/usr/bin/env python3 +""" +Reads a simulation `.toml` and draws the domain box / camera framing / axes / +region crop / field-line seed lattice, WITHOUT any simulation data or +ray-marching. It lets you iterate on camera orientation (e.g. the domain cube of +a 3D turbulence run) and framing without relaunching the simulation. + +Modes +----- + * 3D (Cartesian only -- the renderer's only 3D mode): projects the domain cube + (and region crop, if any) with the exact ray-march camera; optional field- + line SEED lattice scatter (schematic: seeds, not traced lines). + * 2D Cartesian: aspect-expanded slice window + domain/region box + ticks. + * 2D spherical/GR: meridional wedge (arcs at r in {rmin,rmax}, rays at + theta in {tmin,tmax}), mirrored into a full disk if `mirror`. + * 1D: nothing to render (warns). +""" + +import argparse +import math +import os +import subprocess +import sys + +try: + import tomllib # Python 3.11+ +except ModuleNotFoundError: # Python 3.10 and older (e.g. the miniforge3 module) + import tomli as tomllib # same load() API + +import matplotlib +import numpy as np + +matplotlib.use("Agg") # headless cluster: no interactive display +import matplotlib.pyplot as plt +from matplotlib.patches import Rectangle + + +# --------------------------------------------------------------------------- # +# toml helpers (mirror toml::find_or: return default if any key is missing) # +# --------------------------------------------------------------------------- # +def find_or(d, default, *keys): + cur = d + for k in keys: + if not isinstance(cur, dict) or k not in cur: + return default + cur = cur[k] + return cur + + +# --------------------------------------------------------------------------- # +# metric / extent handling (grid.cpp ~L416-497) # +# --------------------------------------------------------------------------- # +CARTESIAN_METRICS = {"minkowski"} + + +# everything else that entity supports is curvilinear (r-first extent): +# spherical, qspherical, kerr_schild, kerr_schild_0, qkerr_schild +def is_cartesian(metric_name): + return metric_name.strip().lower() in CARTESIAN_METRICS + + +def global_extent(td): + """Replicates grid.cpp: parse grid.extent, auto-fill theta/phi for + non-Cartesian metrics. Returns (extent_pairs, dim, cartesian, metric_name). + + extent_pairs is a list of (lo, hi) of length == dim (== len(resolution)). + """ + resolution = find_or(td, None, "grid", "resolution") + if resolution is None: + raise ValueError("grid.resolution missing") + dim = len(resolution) + metric_name = find_or(td, "minkowski", "grid", "metric", "metric") + cart = is_cartesian(metric_name) + + extent = find_or(td, None, "grid", "extent") + if extent is None: + raise ValueError("grid.extent missing") + # deep copy as list of [lo,hi] + ext = [list(pair) for pair in extent] + + # grid.cpp: if extent has more rows than dim, truncate to dim + if len(ext) > dim: + ext = ext[:dim] + + if not cart: + # non-Cartesian: extent gives only the r-range; append theta,(phi). + # (grid.cpp errors if >1 row is supplied for non-cartesian; we just + # keep the r-row and append.) + ext = [ext[0]] + ext.append([0.0, math.pi]) # theta in [0, pi] (2D and 3D) + if dim == 3: + ext.append([0.0, 2.0 * math.pi]) # phi in [0, 2pi] (3D only) + + if len(ext) != dim: + raise ValueError(f"inferred grid.extent has {len(ext)} rows, expected {dim}") + + pairs = [(float(p[0]), float(p[1])) for p in ext] + return pairs, dim, cart, metric_name + + +# --------------------------------------------------------------------------- # +# Camera (renderer.cpp Renderer::init, composite.h projectToScreen) # +# --------------------------------------------------------------------------- # +def _norm3(a): + n = math.sqrt(a[0] * a[0] + a[1] * a[1] + a[2] * a[2]) + if n > 1e-30: + return (a[0] / n, a[1] / n, a[2] / n) + return (a[0], a[1], a[2]) + + +def _cross3(a, b): + return ( + a[1] * b[2] - a[2] * b[1], + a[2] * b[0] - a[0] * b[2], + a[0] * b[1] - a[1] * b[0], + ) + + +def _dot3(a, b): + return a[0] * b[0] + a[1] * b[1] + a[2] * b[2] + + +class Camera: + """POD camera identical to out::CameraDevice, built exactly as in + Renderer::init. eye/forward/right/up in world coords; ortho/persp framing.""" + + def __init__(self, td, region, width, height): + # region is a list of (lo,hi); may be < 3 axes (zero-filled), mirroring + # the C++ which sizes center/size over m_region.size() and d<3. + center = [0.0, 0.0, 0.0] + size = [0.0, 0.0, 0.0] + for d in range(min(len(region), 3)): + center[d] = 0.5 * (region[d][0] + region[d][1]) + size[d] = region[d][1] - region[d][0] + diag = math.sqrt(size[0] ** 2 + size[1] ** 2 + size[2] ** 2) + + cam_ortho = ( + find_or(td, "orthographic", "render", "camera", "mode") != "perspective" + ) + pos = find_or(td, [], "render", "camera", "position") + look = find_or(td, [], "render", "camera", "look_at") + up = find_or(td, [], "render", "camera", "up") + fov = float(find_or(td, 35.0, "render", "camera", "fov")) + # default ortho_height covers the box from any view -> == box diagonal + ortho_height = float( + find_or(td, diag, "render", "camera", "ortho_height") + ) + + # default eye: box center pushed back along (1,1,1) by ~1.7 diagonals + eye = [0.0, 0.0, 0.0] + lookat = [0.0, 0.0, 0.0] + for d in range(3): + if len(pos) == 3: + eye[d] = float(pos[d]) + else: + eye[d] = center[d] + 1.7 * diag * 0.57735026919 + lookat[d] = float(look[d]) if len(look) == 3 else center[d] + if len(up) == 3: + upv = [float(up[0]), float(up[1]), float(up[2])] + else: + upv = [0.0, 0.0, 1.0] + + forward = _norm3([lookat[0] - eye[0], lookat[1] - eye[1], lookat[2] - eye[2]]) + right = _norm3(_cross3(forward, upv)) + up_cam = _cross3(right, forward) # already unit (right,forward unit & perp) + + self.eye = list(eye) + self.forward = list(forward) + self.right = list(right) + self.up = list(up_cam) + self.aspect = float(width) / float(height) + self.tan_half_fov = math.tan(0.5 * fov * math.pi / 180.0) + self.orthographic = bool(cam_ortho) + self.half_h = 0.5 * ortho_height + self.half_w = self.half_h * self.aspect + # keep for the summary line + self.ortho_height = ortho_height + self.fov = fov + + def pan(self, shift): + """Moving view: pure pan of the eye (forward/right/up/half_h unchanged), + matching Renderer::updateForTime.""" + for d in range(3): + self.eye[d] += shift[d] + + def project(self, p, W, H): + """projectToScreen: world point -> (px, py) in pixels (py DOWN, + image origin top-left). Returns None if behind a perspective camera.""" + dx = p[0] - self.eye[0] + dy = p[1] - self.eye[1] + dz = p[2] - self.eye[2] + cx = dx * self.right[0] + dy * self.right[1] + dz * self.right[2] + cy = dx * self.up[0] + dy * self.up[1] + dz * self.up[2] + if self.orthographic: + fx = cx / self.half_w + fy = cy / self.half_h + else: + cz = dx * self.forward[0] + dy * self.forward[1] + dz * self.forward[2] + if cz <= 1e-6: + return None + fx = (cx / cz) / (self.aspect * self.tan_half_fov) + fy = (cy / cz) / self.tan_half_fov + px = (fx + 1.0) * 0.5 * W - 0.5 + py = (1.0 - fy) * 0.5 * H - 0.5 + return (px, py) + + +# --------------------------------------------------------------------------- # +# region resolution (renderer.cpp Renderer::init ~L128-158) # +# --------------------------------------------------------------------------- # +def resolve_region(td, ext): + """m_region starts as the global extent; extent.x{d+1} clamps axis d to the box + if it is a valid [lo,hi] with hi>lo that overlaps. Returns (region, has_region).""" + region = [list(p) for p in ext] + has_region = False + keys = ["x1", "x2", "x3"] + for d in range(min(len(ext), 3)): + lim = find_or(td, [], "render", "extent", keys[d]) + if not lim: + continue + if len(lim) != 2 or lim[1] <= lim[0]: + print( + f" warning: render.extent.{keys[d]} must be [lo,hi] with hi>lo; ignoring" + ) + continue + lo = max(float(lim[0]), ext[d][0]) + hi = min(float(lim[1]), ext[d][1]) + if hi > lo: + region[d] = [lo, hi] + has_region = True + else: + print( + f" warning: render.extent.{keys[d]} does not overlap the domain; ignoring" + ) + return [tuple(p) for p in region], has_region + + +def apply_moving_view(td, region, ext, cam, time, dim): + """Renderer::updateForTime: dt=max(0, t-t0); shift=vel*dt; pan region+eye.""" + vel = find_or(td, [], "render", "moving_view", "velocity") + if not vel: + return region + v = [0.0, 0.0, 0.0] + for d in range(min(len(vel), 3)): + v[d] = float(vel[d]) + if v == [0.0, 0.0, 0.0]: + return region + t0 = float(find_or(td, 0.0, "render", "moving_view", "start_time")) + dt = max(0.0, time - t0) + shift = [v[0] * dt, v[1] * dt, v[2] * dt] + new_region = [] + for d in range(len(region)): + s = shift[d] if d < 3 else 0.0 + new_region.append((region[d][0] + s, region[d][1] + s)) + cam.pan(shift) # cam only meaningful for the 3D mode; harmless otherwise + return new_region + + +# --------------------------------------------------------------------------- # +# "nice" ticks (axes.h niceNum / niceTicks) for annotation # +# --------------------------------------------------------------------------- # +def _nice_num(x, do_round): + if x <= 0.0: + return 1.0 + e = math.floor(math.log10(x)) + f = x / (10.0**e) + if do_round: + nf = 1.0 if f < 1.5 else (2.0 if f < 3.0 else (5.0 if f < 7.0 else 10.0)) + else: + nf = 1.0 if f <= 1.0 else (2.0 if f <= 2.0 else (5.0 if f <= 5.0 else 10.0)) + return nf * (10.0**e) + + +def nice_ticks(lo, hi, n): + out = [] + if not (hi > lo) or n < 2: + return out + step = _nice_num((hi - lo) / (n - 1), True) + if step <= 0.0: + return out + g0 = math.ceil(lo / step) * step + eps = 1e-6 * step + v = g0 + while v <= hi + 0.5 * step: + if lo - eps <= v <= hi + eps: + out.append(0.0 if abs(v) < eps else v) + v += step + return out + + +# --------------------------------------------------------------------------- # +# 3D field-line SEED lattice (fieldlines.h traceFieldLines seed setup) # +# --------------------------------------------------------------------------- # +def field_line_seeds_3d(td, region, cam, H): + """Reproduce the seed-lattice geometry (NOT the traced lines). + + world_per_pixel wpp = (cam.half_h*2)/H (metadomain_render.cpp) + spacing = max(seed_px,1)*wpp; if n_seed>seed_max: spacing *= cbrt(n_seed/seed_max) + ns[d] = max(1, floor(size[d]/spacing)); seeds at cell centers over the + field-line COARSE grid, which spans the FULL global extent -- BUT here we + seed over the render region's box (== extent when uncropped), matching the + coarse grid origin=extent.first in the uncropped case. We use the region box + the camera frames so the schematic overlays the drawn cube. + """ + fl_enable = find_or(td, False, "render", "fieldlines", "enable") + # any scene may also request the overlay + scenes = find_or(td, [], "render", "scene") + any_fl = any( + (find_or(sc, False, "fieldlines") or find_or(sc, "", "field") == "fieldlines") + for sc in scenes + ) + if not (fl_enable or any_fl): + return None + + seed_px = float(find_or(td, 8.0, "render", "fieldlines", "seed_px")) + seed_max = int(find_or(td, 4096, "render", "fieldlines", "seed_max")) + + wpp = (cam.half_h * 2.0) / float(H) + size = [region[d][1] - region[d][0] for d in range(3)] + origin = [region[d][0] for d in range(3)] + + spacing = max(seed_px, 1.0) * wpp + + def count_seeds(sp): + ns = [] + tot = 1 + for d in range(3): + n = max(1, math.floor(size[d] / sp)) if sp > 0 else 1 + ns.append(n) + tot *= n + return tot, ns + + n_seed, ns = count_seeds(spacing) + if n_seed > seed_max > 0: + grow = (float(n_seed) / float(seed_max)) ** (1.0 / 3.0) + spacing *= grow + n_seed, ns = count_seeds(spacing) + + seeds = [] + for k in range(ns[2]): + for j in range(ns[1]): + for i in range(ns[0]): + seeds.append( + ( + origin[0] + (i + 0.5) * size[0] / ns[0], + origin[1] + (j + 0.5) * size[1] / ns[1], + origin[2] + (k + 0.5) * size[2] / ns[2], + ) + ) + return seeds, ns + + +# --------------------------------------------------------------------------- # +# cube edges: the 12 edges of an axis-aligned box given by (lo,hi) per axis # +# --------------------------------------------------------------------------- # +def cube_corners(box): + """box: list of (lo,hi) for 3 axes. Corner m: bit0->x, bit1->y, bit2->z + (matches the C++ corner() ordering in axes.h / screenBBox).""" + corners = [] + for m in range(8): + corners.append( + ( + box[0][1] if (m & 1) else box[0][0], + box[1][1] if (m & 2) else box[1][0], + box[2][1] if (m & 4) else box[2][0], + ) + ) + return corners + + +# the 12 edges as (corner_i, corner_j) index pairs +CUBE_EDGES = [ + (0, 1), + (2, 3), + (4, 5), + (6, 7), # x-parallel + (0, 2), + (1, 3), + (4, 6), + (5, 7), # y-parallel + (0, 4), + (1, 5), + (2, 6), + (3, 7), # z-parallel +] + + +# --------------------------------------------------------------------------- # +# 3D axis-annotation edge selection (axes.h drawAxes3D) # +# # +# A box axis has FOUR parallel edges; the ticks/labels go on exactly one. We # +# only ever pick a *silhouette* edge -- one whose two adjacent faces point in # +# opposite directions relative to the camera (one toward it, one away). A # +# silhouette edge always lies on the drawn OUTLINE of the box, so it is in the # +# foreground -- never occluded by the volume -- and its ticks, pushed outward # +# from the projected centroid, land in empty background. Among the (usually # +# two) silhouette candidates we take the one nearest the bottom-left of the # +# image (score = pixel_y - pixel_x), the conventional place for annotation. # +# This is a faithful port of out::drawAxes3D so the preview matches the # +# renderer's choice of which spine carries the axis. # +# --------------------------------------------------------------------------- # +def _front_face(cam, axis, side): + """True if the box face perpendicular to `axis` at `side` (1=high, 0=low) + faces the camera (its outward normal points back toward the eye).""" + nrm = 1.0 if side else -1.0 + return (nrm * (-cam.forward[axis])) > 0.0 + + +def select_axis_edge_3d(cam, d, cx, cy, ccx, ccy): + """Choose the edge parallel to axis d that carries the ticks + label. + + cx,cy are the 8 projected corner pixel coords (cube_corners order); ccx,ccy + the projected box centroid. Returns (m0, m1, pxd, pyd): the edge's two corner + indices (m0 has axis d at its low end) and the unit screen-space OUTWARD push + direction (perpendicular to the edge, pointing away from the centroid).""" + e1 = 1 if d == 0 else 0 # the two perpendicular axes + e2 = 1 if d == 2 else 2 + best = None + for s1 in (0, 1): + for s2 in (0, 1): + m0 = (s1 << e1) | (s2 << e2) + m1 = m0 | (1 << d) + mx = 0.5 * (cx[m0] + cx[m1]) + my = 0.5 * (cy[m0] + cy[m1]) + score = my - mx # prefer the foreground (bottom-left) edge + if _front_face(cam, e1, s1) != _front_face(cam, e2, s2): + score += 1e6 # strongly prefer silhouette (outline) edges + if best is None or score > best[0]: + best = (score, m0, m1) + _, m0, m1 = best + ex, ey = cx[m1] - cx[m0], cy[m1] - cy[m0] + el = math.hypot(ex, ey) or 1.0 + ex, ey = ex / el, ey / el + pxd, pyd = -ey, ex # screen-perpendicular to the edge + mxv = 0.5 * (cx[m0] + cx[m1]) - ccx + myv = 0.5 * (cy[m0] + cy[m1]) - ccy + if pxd * mxv + pyd * myv < 0.0: # flip to point away from the box centroid + pxd, pyd = -pxd, -pyd + return m0, m1, pxd, pyd + + +def draw_3d(td, ext, region, has_region, cam, W, H, out_path, sim_name): + fig, ax = plt.subplots(figsize=(W / 100.0, H / 100.0), dpi=100) + + def project_box(box, color, lw, label, ls="-"): + corners = cube_corners(box) + proj = [cam.project(c, W, H) for c in corners] + first = True + for a, b in CUBE_EDGES: + pa, pb = proj[a], proj[b] + if pa is None or pb is None: + continue # edge with a corner behind a perspective camera + ax.plot( + [pa[0], pb[0]], + [pa[1], pb[1]], + color=color, + lw=lw, + ls=ls, + label=(label if first else None), + zorder=3, + ) + first = False + + # full extent (light gray) + project_box([ext[0], ext[1], ext[2]], color="0.6", lw=1.2, label="full extent") + # region crop, if distinct + if has_region: + project_box( + [region[0], region[1], region[2]], + color="tab:blue", + lw=2.0, + label="region crop", + ) + + # axes tick labels: for each axis, pick the FOREGROUND (silhouette) edge of + # the framed box and annotate along it, exactly as out::drawAxes3D does, so + # the labels never end up on an edge hidden behind the volume. + axes_on = find_or(td, False, "render", "axes") + nticks = int(find_or(td, 5, "render", "axis_ticks")) + frame_box = region if has_region else ext + axis_names = find_or(td, [], "render", "axis_labels") + default_names = ["x", "y", "z"] + if axes_on: + corners = cube_corners([frame_box[0], frame_box[1], frame_box[2]]) + cx = [0.0] * 8 + cy = [0.0] * 8 + for m in range(8): + pr = cam.project(corners[m], W, H) + cx[m] = pr[0] if pr is not None else 0.0 + cy[m] = pr[1] if pr is not None else 0.0 + cen = [0.5 * (frame_box[d][0] + frame_box[d][1]) for d in range(3)] + prc = cam.project(cen, W, H) + ccx = prc[0] if prc is not None else 0.0 + ccy = prc[1] if prc is not None else 0.0 + + tl = 8.0 # tick-mark length [px] + num_off = tl + 10.0 # numeric-label center offset from the edge [px] + name_off = tl + 30.0 # axis-name center offset from the edge [px] + for d in range(3): + name = axis_names[d] if d < len(axis_names) else default_names[d] + m0, _, pxd, pyd = select_axis_edge_3d(cam, d, cx, cy, ccx, ccy) + # o = corner(m0): perpendicular coords fixed, axis d swept for ticks + o = list(corners[m0]) + lo_d, hi_d = frame_box[d][0], frame_box[d][1] + for tv in nice_ticks(lo_d, hi_d, nticks): + p = list(o) + p[d] = tv + pr = cam.project(p, W, H) + if pr is None: + continue + a, b = pr + ax.plot( + [a, a + pxd * tl], [b, b + pyd * tl], color="0.35", lw=1.0, zorder=4 + ) + ax.annotate( + f"{tv:g}", + (a + pxd * num_off, b + pyd * num_off), + fontsize=6, + color="0.25", + ha="center", + va="center", + ) + # axis name at the MIDDLE of the chosen edge, pushed further outward + mid = list(o) + mid[d] = 0.5 * (lo_d + hi_d) + pr = cam.project(mid, W, H) + if pr is not None: + a, b = pr + ax.annotate( + name, + (a + pxd * name_off, b + pyd * name_off), + fontsize=9, + color="k", + fontweight="bold", + ha="center", + va="center", + ) + + # field-line SEED lattice (schematic scatter, NOT traced lines) + fl = field_line_seeds_3d(td, frame_box, cam, H) + if fl is not None: + seeds, ns = fl + pxs, pys = [], [] + for s in seeds: + pr = cam.project(s, W, H) + if pr is not None: + pxs.append(pr[0]) + pys.append(pr[1]) + if pxs: + ax.scatter( + pxs, + pys, + s=8, + c="tab:red", + marker="o", + alpha=0.6, + edgecolors="none", + zorder=2, + label=f"field-line seeds (schematic, {ns[0]}x{ns[1]}x{ns[2]})", + ) + + ax.set_xlim(0, W) + ax.set_ylim(H, 0) # inverted y: origin upper-left, matches the PNG + ax.set_aspect("equal") # lock width:height 1:1 in pixel space + ax.set_xlabel("screen x [px]") + ax.set_ylabel("screen y [px]") + proj_kind = "orthographic" if cam.orthographic else f"perspective (fov {cam.fov:g})" + ax.set_title(f"{sim_name}: 3D scene preview ({proj_kind})", fontsize=10) + ax.legend(loc="upper right", fontsize=7, framealpha=0.85) + fig.tight_layout() + fig.savefig(out_path, dpi=100) + plt.close(fig) + + +def derive_2d_window(td, ext, region, cartesian, mirror, W, H): + """Return (umin,umax,vmin,vmax) -- the aspect-expanded world window mapped + onto the WxH image, exactly as metadomain_render.cpp derives it.""" + x1lo, x1hi = region[0][0], region[0][1] + x2lo, x2hi = region[1][0], region[1][1] + if cartesian: + umin, umax, vmin, vmax = x1lo, x1hi, x2lo, x2hi + else: + # meridional (X = r sin th, Z = r cos th) bbox of the wedge, sampling the + # boundary (arcs at r={x1lo,x1hi}, rays at theta={x2lo,x2hi}). + umin, umax = 1e30, -1e30 + vmin, vmax = 1e30, -1e30 + NB = 65 + + def accXZ(r, th): + nonlocal umin, umax, vmin, vmax + X = r * math.sin(th) + Z = r * math.cos(th) + umin = min(umin, X) + umax = max(umax, X) + vmin = min(vmin, Z) + vmax = max(vmax, Z) + if mirror: + umin = min(umin, -X) + umax = max(umax, -X) + + for k in range(NB): + t = k / (NB - 1) + th = x2lo + (x2hi - x2lo) * t + rr = x1lo + (x1hi - x1lo) * t + accXZ(x1lo, th) + accXZ(x1hi, th) + accXZ(rr, x2lo) + accXZ(rr, x2hi) + + # expand the window to the image aspect (centered) so geometry isn't stretched + waspect = (umax - umin) / (vmax - vmin) + iaspect = float(W) / float(H) + if iaspect > waspect: + cu = 0.5 * (umin + umax) + hu = 0.5 * (vmax - vmin) * iaspect + umin, umax = cu - hu, cu + hu + else: + cv = 0.5 * (vmin + vmax) + hv = 0.5 * (umax - umin) / iaspect + vmin, vmax = cv - hv, cv + hv + + # spherical slices get a 1.12x background border so the round outline + labels + # are not clipped (Cartesian fills the frame and needs none). + if not cartesian: + pad = 1.12 + cu = 0.5 * (umin + umax) + hu = 0.5 * (umax - umin) * pad + cv = 0.5 * (vmin + vmax) + hv = 0.5 * (vmax - vmin) * pad + umin, umax = cu - hu, cu + hu + vmin, vmax = cv - hv, cv + hv + + return umin, umax, vmin, vmax + + +def draw_2d_cartesian(td, ext, region, has_region, W, H, out_path, sim_name): + umin, umax, vmin, vmax = derive_2d_window(td, ext, region, True, False, W, H) + fig, ax = plt.subplots(figsize=(W / 100.0, H / 100.0), dpi=100) + + # aspect-expanded slice window (the background-padded frame) + ax.add_patch( + Rectangle( + (umin, vmin), + umax - umin, + vmax - vmin, + fill=False, + ec="0.7", + lw=1.0, + ls="--", + label="slice window (aspect-expanded)", + ) + ) + # full domain box + ax.add_patch( + Rectangle( + (ext[0][0], ext[1][0]), + ext[0][1] - ext[0][0], + ext[1][1] - ext[1][0], + fill=False, + ec="0.4", + lw=1.5, + label="domain", + ) + ) + # region crop + if has_region: + ax.add_patch( + Rectangle( + (region[0][0], region[1][0]), + region[0][1] - region[0][0], + region[1][1] - region[1][0], + fill=False, + ec="tab:blue", + lw=2.0, + label="region crop", + ) + ) + + # ticks (nice numbers over the data box == region) + axes_on = find_or(td, False, "render", "axes") + nticks = int(find_or(td, 5, "render", "axis_ticks")) + if axes_on: + for tv in nice_ticks(region[0][0], region[0][1], nticks): + ax.axvline(tv, color="0.85", lw=0.5, zorder=0) + for tv in nice_ticks(region[1][0], region[1][1], nticks): + ax.axhline(tv, color="0.85", lw=0.5, zorder=0) + + ax.set_xlim(umin, umax) + ax.set_ylim(vmin, vmax) # +v up (Cartesian slice: y is up in world) + ax.set_aspect("equal") + ax.set_xlabel("x1") + ax.set_ylabel("x2") + ax.set_title(f"{sim_name}: 2D Cartesian slice preview", fontsize=10) + ax.legend(loc="upper right", fontsize=7, framealpha=0.85) + fig.tight_layout() + fig.savefig(out_path, dpi=100) + plt.close(fig) + + +def draw_2d_spherical(td, ext, region, has_region, mirror, W, H, out_path, sim_name): + umin, umax, vmin, vmax = derive_2d_window(td, ext, region, False, mirror, W, H) + # (r, theta) wedge of the render region + rmin, rmax = region[0][0], region[0][1] + tmin, tmax = region[1][0], region[1][1] + + fig, ax = plt.subplots(figsize=(W / 100.0, H / 100.0), dpi=100) + + def wedge_boundary(rmn, rmx, tmn, tmx, sign, color, lw, label=None): + # outer + inner arcs and two rays, in meridional (X=r sin th, Z=r cos th) + th = np.linspace(tmn, tmx, 200) + # outer arc + ax.plot( + sign * rmx * np.sin(th), rmx * np.cos(th), color=color, lw=lw, label=label + ) + # inner arc + ax.plot(sign * rmn * np.sin(th), rmn * np.cos(th), color=color, lw=lw) + # rays at tmin, tmax + for tt in (tmn, tmx): + ax.plot( + [sign * rmn * math.sin(tt), sign * rmx * math.sin(tt)], + [rmn * math.cos(tt), rmx * math.cos(tt)], + color=color, + lw=lw, + ) + + # full extent wedge (light gray) + wedge_boundary( + ext[0][0], ext[0][1], ext[1][0], ext[1][1], 1.0, "0.6", 1.2, label="full extent" + ) + if mirror: + wedge_boundary(ext[0][0], ext[0][1], ext[1][0], ext[1][1], -1.0, "0.6", 1.2) + + # region wedge (colored) if cropped + if has_region: + wedge_boundary( + rmin, rmax, tmin, tmax, 1.0, "tab:blue", 2.0, label="region crop" + ) + if mirror: + wedge_boundary(rmin, rmax, tmin, tmax, -1.0, "tab:blue", 2.0) + + # radial ticks along the symmetry axis (X=0) + axes_on = find_or(td, False, "render", "axes") + nticks = int(find_or(td, 5, "render", "axis_ticks")) + if axes_on: + for Rv in nice_ticks(0.0, ext[0][1], nticks): + ax.plot(0.0, Rv, marker="+", color="0.3", ms=6) + ax.annotate( + f"{Rv:g}", + (0.0, Rv), + fontsize=6, + color="0.25", + xytext=(-8, 0), + textcoords="offset points", + ha="right", + va="center", + ) + + ax.set_xlim(umin, umax) + ax.set_ylim(vmin, vmax) + ax.set_aspect("equal") + ax.set_xlabel("X = r sin(theta)") + ax.set_ylabel("Z = r cos(theta)") + m = "mirrored" if mirror else "half-plane" + ax.set_title(f"{sim_name}: 2D spherical meridional preview ({m})", fontsize=10) + ax.legend(loc="upper right", fontsize=7, framealpha=0.85) + fig.tight_layout() + fig.savefig(out_path, dpi=100) + plt.close(fig) + + +def preview(args): + if not os.path.isfile(args.toml): + print(f"error: no such file: {args.toml}", file=sys.stderr) + return 2 + + with open(args.toml, "rb") as f: + td = tomllib.load(f) + + # renderer enabled? + if not find_or(td, False, "render", "enable"): + print("note: [render].enable is false in this toml; previewing anyway.") + + sim_name = find_or(td, "sim", "simulation", "name") + width = int(find_or(td, 1024, "render", "width")) + height = int(find_or(td, 1024, "render", "height")) + mirror = bool(find_or(td, True, "render", "mirror")) + + ext, dim, cartesian, metric_name = global_extent(td) + + # output path + if args.out: + out_path = args.out + else: + toml_dir = os.path.dirname(os.path.abspath(args.toml)) + out_path = os.path.join(toml_dir, f"{sim_name}_preview.png") + + # region + camera (region drives the default framing) + region, has_region = resolve_region(td, ext) + # camera is only meaningful in 3D, but building it is cheap & shares the pan + cam = Camera(td, region, width, height) + region = apply_moving_view(td, region, ext, cam, args.time, dim) + # NB: the C++ pans the eye but keeps forward/right/up/half_h; we already + # panned cam.eye, and the ortho_height / basis are region-independent after + # the initial framing, so cam is now consistent with the panned region. + + # ---- summary line ------------------------------------------------------- + def fmt_pairs(pairs): + return "[" + ", ".join(f"({p[0]:g},{p[1]:g})" for p in pairs) + "]" + + if dim == 3 and cartesian: + mode = "3D box (Cartesian volume)" + elif dim == 3: + mode = "3D (non-Cartesian: unsupported by renderer)" + elif dim == 2 and cartesian: + mode = "2D Cartesian slice" + elif dim == 2: + mode = "2D spherical meridional slice" + else: + mode = "1D (nothing to render)" + + eye_str = f"({cam.eye[0]:g},{cam.eye[1]:g},{cam.eye[2]:g})" + print( + f"mode={mode} | metric={metric_name} | eye={eye_str} | " + f"ortho_height={cam.ortho_height:g} | region={fmt_pairs(region)}" + ) + if args.scene is not None: + print(f" (scene index {args.scene} requested; geometry is scene-independent)") + + # ---- dispatch ----------------------------------------------------------- + if dim == 3 and cartesian: + draw_3d(td, ext, region, has_region, cam, width, height, out_path, sim_name) + elif dim == 3: + print( + "warning: 3D non-Cartesian is not a renderer mode (3D is Cartesian-" + "only); nothing drawn." + ) + return 1 + elif dim == 2 and cartesian: + draw_2d_cartesian( + td, ext, region, has_region, width, height, out_path, sim_name + ) + elif dim == 2: + draw_2d_spherical( + td, ext, region, has_region, mirror, width, height, out_path, sim_name + ) + else: + print("warning: 1D run -- the renderer is inactive; nothing to preview.") + return 1 + + print(f"wrote {out_path}") + return 0 + + +# ffmpeg -nostdin -framerate $framerate $inputspec -c:v libx264 -crf $compression -filter_complex \"[0:v]format=yuv420p,pad=ceil(iw/2)*2:ceil(ih/2)*2\" $output" + + +def merge(args): + try: + subprocess.run(["ffmpeg", "-version"], check=True, stdout=subprocess.DEVNULL) + except (subprocess.CalledProcessError, FileNotFoundError): + print( + "error: ffmpeg not found or not executable; cannot merge PNGs into a movie" + ) + return 1 + png_dir = os.path.abspath(args.path) + png_files = {f.split("_")[0] for f in os.listdir(png_dir) if f.endswith(".png")} + print(png_files, png_dir) + + ffmpeg_prekwargs = [ + "ffmpeg", + "-nostdin", + "-framerate", + str(args.framerate), + ] + ffmpeg_postkwargs = [ + "-c:v", + "libx264", + "-crf", + str(args.compression), + "-filter_complex", + "[0:v]format=yuv420p,pad=ceil(iw/2)*2:ceil(ih/2)*2", + ] + + if args.prefix: + if args.prefix not in png_files: + print(f"error: prefix '{args.prefix}' not found in {png_dir}") + return 1 + prefixes = [args.prefix] + else: + prefixes = sorted(png_files) + if args.merge and len(prefixes) > 1: + # arrange prefixes on a grid with args.cols columns (empty cells are black) + n = len(prefixes) + cols = max(1, min(args.cols, n)) + rows = (n + cols - 1) // cols + inputs = [] + for prefix in prefixes: + # -framerate/-pattern_type are per-input options: repeat before every -i + inputs += [ + "-framerate", + str(args.framerate), + "-pattern_type", + "glob", + "-i", + os.path.join(png_dir, f"{prefix}_*.png"), + ] + layout = [] + for i in range(n): + r, c = divmod(i, cols) + x = "+".join(f"w{j}" for j in range(c)) or "0" + y = "+".join(f"h{k * cols}" for k in range(r)) or "0" + layout.append(f"{x}_{y}") + filter_complex = ( + "".join(f"[{i}:v]" for i in range(n)) + + f"xstack=inputs={n}:layout={'|'.join(layout)}:fill=black:shortest=1," + + "format=yuv420p,pad=ceil(iw/2)*2:ceil(ih/2)*2" + ) + print(f"merging {n} scenes into a {rows}x{cols} grid") + kwargs = ( + ["ffmpeg", "-nostdin"] + + inputs + + ["-c:v", "libx264", "-crf", str(args.compression)] + + ["-filter_complex", filter_complex] + + ["merged_render.mp4"] + ) + subprocess.run(kwargs, check=True) + + else: + extra_kwargs = ["-pattern_type", "glob"] + for prefix in prefixes: + kwargs = ( + ffmpeg_prekwargs + + extra_kwargs + + ["-i", os.path.join(png_dir, f"{prefix}_*.png")] + + ffmpeg_postkwargs + + [prefix + "_render.mp4"] + ) + subprocess.run(kwargs, check=True) + return 0 + + +def main(): + ap = argparse.ArgumentParser( + description="Helper tools for the Entity on-the-fly renderer" + ) + sp = ap.add_subparsers(help="commands", required=True) + preview_sp = sp.add_parser( + "preview", + help="draw a preview of the simulation domain", + ) + movie_sp = sp.add_parser( + "movie", + help="merge the rendered .png into a movie", + ) + + preview_sp.add_argument("toml", help="simulation .toml file") + preview_sp.add_argument( + "-o", + "--out", + default=None, + help="output PNG (default: /_preview.png)", + ) + preview_sp.add_argument( + "-t", + "--time", + type=float, + default=0.0, + help="sim time T for the moving-view pan (default 0)", + ) + preview_sp.add_argument( + "-s", + "--scene", + type=int, + default=None, + help="scene index (accepted for parity; geometry is scene-" + "independent, so it only affects the reported label)", + ) + + movie_sp.add_argument( + "path", + help="path to the rendered PNGs", + ) + movie_sp.add_argument( + "-p", + "--prefix", + type=str, + help="prefix for the scene (when rendering only one scene without -m | --merge flag)", + ) + movie_sp.add_argument( + "-c", + "--cols", + type=int, + default=1, + help="number of columns for combining multiple scenes (default: 1)", + ) + movie_sp.add_argument( + "-m", + "--merge", + action="store_true", + help="merge multiple scenes into a single movie (default: false)", + ) + movie_sp.add_argument( + "-r", + "--framerate", + type=int, + default=30, + help="framerate for the output movie (default: 30)", + ) + movie_sp.add_argument( + "-z", + "--compression", + type=int, + default=30, + help="compression level (default: 1)", + ) + + preview_sp.set_defaults(func=preview) + movie_sp.set_defaults(func=merge) + + args = ap.parse_args() + args.func(args) + + if len(sys.argv) == 1: + ap.print_help() + return 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/src/engines/engine.hpp b/src/engines/engine.hpp index 63a5d1859..13b5e8e28 100644 --- a/src/engines/engine.hpp +++ b/src/engines/engine.hpp @@ -81,7 +81,7 @@ namespace ntt { const bool is_resuming; const simtime_t runtime; const real_t dt; - const std::size_t team_policy_team_size; + const std::size_t tiled_deposit_team_size; const timestep_t max_steps; const timestep_t start_step; const simtime_t start_time; @@ -110,8 +110,8 @@ namespace ntt { , is_resuming { m_params.get("checkpoint.is_resuming") } , runtime { m_params.get("simulation.runtime") } , dt { m_params.get("algorithms.timestep.dt") } - , team_policy_team_size { m_params.get( - "algorithms.deposit.team_policy_team_size") } + , tiled_deposit_team_size { m_params.get( + "algorithms.deposit.tiled_deposit_team_size") } , max_steps { static_cast(runtime / dt) } , start_step { m_params.get("checkpoint.start_step") } , start_time { m_params.get("checkpoint.start_time") } @@ -130,8 +130,8 @@ namespace ntt { auto parameters = prm::Parameters {}; parameters.set("dt", static_cast(dt)); parameters.set("time", static_cast(time)); - parameters.set("team_policy_team_size", - static_cast(team_policy_team_size)); + parameters.set("tiled_deposit_team_size", + static_cast(tiled_deposit_team_size)); return parameters; } }; @@ -143,6 +143,7 @@ namespace ntt { m_metadomain.InitWriter(&m_adios, m_params); m_metadomain.InitCheckpointWriter(&m_adios, m_params); #endif + m_metadomain.InitRenderer(m_params); logger::Checkpoint("Initializing Engine", HERE); if (not is_resuming) { // start a new simulation with initial conditions @@ -254,7 +255,8 @@ namespace ntt { "ParticleBoundaries", "Communications", "Injector", "Custom", "LoadBalance", "ParticleSort", - "Output", "Checkpoint" }, + "Output", "Render", + "Checkpoint" }, []() { Kokkos::fence(); }, @@ -322,6 +324,7 @@ namespace ntt { ++step; auto print_output = false; + auto print_render = false; auto print_checkpoint = false; #if defined(OUTPUT_ENABLED) timers.start("Output"); @@ -366,7 +369,13 @@ namespace ntt { time - dt); } timers.stop("Output"); +#endif + + timers.start("Render"); + print_render = m_metadomain.Render(m_params, step, step - 1, time, time - dt); + timers.stop("Render"); +#if defined(OUTPUT_ENABLED) timers.start("Checkpoint"); print_checkpoint = m_metadomain.WriteCheckpoint(m_params, step, @@ -393,6 +402,7 @@ namespace ntt { m_metadomain.l_maxnpart_perspec(), print_prtl_clear, print_output, + print_render, print_checkpoint, m_params.get("diagnostics.colored_stdout")); } diff --git a/src/engines/grpic/currents.h b/src/engines/grpic/currents.h index 15d4d94ca..9194dea7a 100644 --- a/src/engines/grpic/currents.h +++ b/src/engines/grpic/currents.h @@ -3,7 +3,7 @@ * @brief Current deposition and filtering routines for the GRPIC engine * @implements * - ntt::grpic::CallDepositKernel<> -> void (flat path) - * - ntt::grpic::CallDepositKernelTiled<> -> void (TEAM_POLICY) + * - ntt::grpic::CallDepositKernelTiled<> -> void (TILED_DEPOSIT) * - ntt::grpic::CurrentsDeposit<> -> void * - ntt::grpic::CurrentsFilter<> -> void * @namespaces: @@ -47,7 +47,7 @@ namespace ntt { dt)); } -#if defined(TEAM_POLICY) +#if defined(TILED_DEPOSIT) /** * @brief Tiled deposit launcher (TeamPolicy + per-team scratch). * @@ -56,7 +56,7 @@ namespace ntt { * teams; each team accumulates its tile's particle contributions in SLM * scratch and atomically flushes to the global J (here `cur0`, the GRPIC * half-step current). Requires the species to have been sorted with - * `team_policy` enabled (`tile_layout` populated by `SortSpatially`). + * `tiled_deposit` enabled (`tile_layout` populated by `SortSpatially`). * * The deposit body (`kernel::DepositOneParticle`) * is the same shared math used by the flat path — it already carries the GR @@ -75,7 +75,7 @@ namespace ntt { int team_size_req) { static_assert(O <= 11u, "Shape order must be <= 11"); constexpr unsigned short T = static_cast( - TEAM_POLICY_TILE_SIZE); + TILED_DEPOSIT_TILE_SIZE); const auto& layout = species.tile_layout(); raise::ErrorIf(layout.ntiles_total == 0u, "CallDepositKernelTiled: tile_layout has 0 tiles — call " @@ -97,7 +97,7 @@ namespace ntt { // Team (work-group) size. The default (team_size_req == 0) leaves // Kokkos::AUTO, which sizes the team from the backend occupancy - // heuristic. A positive `algorithms.deposit.team_policy_team_size` + // heuristic. A positive `algorithms.deposit.tiled_deposit_team_size` // overrides it, clamped to the scratch/backend-feasible maximum so an // over-large request cannot abort the launch (Kokkos errors when // team_size > team_size_max). No portable subgroup rounding is applied; @@ -113,7 +113,7 @@ namespace ntt { if (ts > ts_max) { raise::Warning( fmt::format( - "algorithms.deposit.team_policy_team_size = %d exceeds " + "algorithms.deposit.tiled_deposit_team_size = %d exceeds " "the tiled-deposit maximum %d on this backend; clamping " "to %d", team_size_req, @@ -153,7 +153,7 @@ namespace ntt { Kokkos::Experimental::contribute(cur_nc, scatter_cur); } } -#endif // TEAM_POLICY +#endif // TILED_DEPOSIT template void CurrentsDeposit(Domain& domain, @@ -163,12 +163,12 @@ namespace ntt { // pre-zeros it — this is the single source of truth, matching SRPIC). Kokkos::deep_copy(domain.fields.cur0, ZERO); -#if defined(TEAM_POLICY) +#if defined(TILED_DEPOSIT) // Optional runtime override for the tiled-deposit team (work-group) size; // 0 (default) keeps Kokkos::AUTO. Clamped to the backend max in the // launcher (see CallDepositKernelTiled). const auto team_size_req = static_cast( - engine_params.get("team_policy_team_size", + engine_params.get("tiled_deposit_team_size", std::optional { 0u })); // Tiled deposit. Correctness no longer depends on the SoA being in a diff --git a/src/engines/reporter.cpp b/src/engines/reporter.cpp index 2f19d500c..7752840f1 100644 --- a/src/engines/reporter.cpp +++ b/src/engines/reporter.cpp @@ -32,13 +32,13 @@ namespace ntt { "%s", params.template get("simulation.name").c_str()); reporter::AddParam(report, 4, "Engine", "%s", SimEngine(S).to_string()); -#if defined(TEAM_POLICY) - reporter::AddParam(report, 4, "Tile size", "%d", TEAM_POLICY_TILE_SIZE); - #if defined(TEAM_POLICY_DRIFT) - reporter::AddParam(report, 4, "Halo drift", "%d", TEAM_POLICY_DRIFT); +#if defined(TILED_DEPOSIT) + reporter::AddParam(report, 4, "Tile size", "%d", TILED_DEPOSIT_TILE_SIZE); + #if defined(TILED_DEPOSIT_DRIFT) + reporter::AddParam(report, 4, "Halo drift", "%d", TILED_DEPOSIT_DRIFT); #endif if (params.template get( - "algorithms.deposit.team_policy_team_size") == 0u) { + "algorithms.deposit.tiled_deposit_team_size") == 0u) { reporter::AddParam(report, 4, "Team size", "%s", "AUTO (Kokkos)"); } else { reporter::AddParam(report, @@ -46,7 +46,7 @@ namespace ntt { "Team size", "%d (requested; clamped to backend max at launch)", static_cast(params.template get( - "algorithms.deposit.team_policy_team_size"))); + "algorithms.deposit.tiled_deposit_team_size"))); } #endif reporter::AddParam(report, 4, "Metric", "%s", M.to_string()); diff --git a/src/engines/srpic/currents.h b/src/engines/srpic/currents.h index 2a23fde69..239a8a00d 100644 --- a/src/engines/srpic/currents.h +++ b/src/engines/srpic/currents.h @@ -3,7 +3,7 @@ * @brief Current deposition and filtering routines for the SRPIC engine * @implements * - ntt::srpic::CallDepositKernel<> -> void (flat path) - * - ntt::srpic::CallDepositKernelTiled<> -> void (TEAM_POLICY) + * - ntt::srpic::CallDepositKernelTiled<> -> void (TILED_DEPOSIT) * - ntt::srpic::CurrentsDeposit<> -> void * - ntt::srpic::CurrentsFilter<> -> void * @namespaces: @@ -48,14 +48,14 @@ namespace ntt { dt)); } -#if defined(TEAM_POLICY) +#if defined(TILED_DEPOSIT) /** * @brief Tiled deposit launcher (TeamPolicy + per-team scratch). * * Iterates over `tile_layout.ntiles_total` teams; each team accumulates * its tile's particle contributions in SLM scratch and atomically * flushes to the global J. Requires the species to have been sorted - * with `team_policy` enabled (`tile_layout` populated by + * with `tiled_deposit` enabled (`tile_layout` populated by * `SortSpatially`). * * Falls back to the flat kernel if `tile_offsets` is empty — this @@ -71,7 +71,7 @@ namespace ntt { int team_size_req) { static_assert(O <= 11u, "Shape order must be <= 11"); constexpr unsigned short T = static_cast( - TEAM_POLICY_TILE_SIZE); + TILED_DEPOSIT_TILE_SIZE); const auto& layout = species.tile_layout(); raise::ErrorIf(layout.ntiles_total == 0u, "CallDepositKernelTiled: tile_layout has 0 tiles — call " @@ -93,7 +93,7 @@ namespace ntt { // Team (work-group) size. The default (team_size_req == 0) leaves // Kokkos::AUTO, which sizes the team from the backend occupancy - // heuristic. A positive `algorithms.deposit.team_policy_team_size` + // heuristic. A positive `algorithms.deposit.tiled_deposit_team_size` // overrides it, clamped to the scratch/backend-feasible maximum so an // over-large request cannot abort the launch (Kokkos errors when // team_size > team_size_max). No portable subgroup rounding is applied; @@ -109,7 +109,7 @@ namespace ntt { if (ts > ts_max) { raise::Warning( fmt::format( - "algorithms.deposit.team_policy_team_size = %d exceeds " + "algorithms.deposit.tiled_deposit_team_size = %d exceeds " "the tiled-deposit maximum %d on this backend; clamping " "to %d", team_size_req, @@ -149,7 +149,7 @@ namespace ntt { Kokkos::Experimental::contribute(cur_nc, scatter_cur); } } -#endif // TEAM_POLICY +#endif // TILED_DEPOSIT template void CurrentsDeposit(Domain& domain, @@ -157,12 +157,12 @@ namespace ntt { const auto dt = engine_params.get("dt"); Kokkos::deep_copy(domain.fields.cur, ZERO); -#if defined(TEAM_POLICY) +#if defined(TILED_DEPOSIT) // Optional runtime override for the tiled-deposit team (work-group) size; // 0 (default) keeps Kokkos::AUTO. Clamped to the backend max in the // launcher (see CallDepositKernelTiled). const auto team_size_req = static_cast( - engine_params.get("team_policy_team_size", + engine_params.get("tiled_deposit_team_size", std::optional { 0u })); // Tiled deposit. Correctness no longer depends on the SoA being in a @@ -170,7 +170,7 @@ namespace ntt { // partition per-particle: // - a particle whose full stencil has drifted out of its tile is // deposited straight to the global J view (the per-particle escape - // valve); `team_policy_drift` sizes the scratch halo so the + // valve); `tiled_deposit_drift` sizes the scratch halo so the // common in-tile case stays in fast SLM (see kernels/deposition/currents/tiled.hpp); // - particles dead-tagged in place since the sort are clamped out by // the kernel and skipped by the dead-tag test; diff --git a/src/framework/CMakeLists.txt b/src/framework/CMakeLists.txt index 318e17c14..7494c11b2 100644 --- a/src/framework/CMakeLists.txt +++ b/src/framework/CMakeLists.txt @@ -10,6 +10,7 @@ # * parameters/output.cpp # * parameters/algorithms.cpp # * parameters/extra.cpp +# * parameters/render.cpp # * simulation.cpp # * domain/grid.cpp # * domain/metadomain.cpp @@ -28,6 +29,7 @@ # * domain/io/spectra.cpp # * domain/io/fields.cpp # * domain/io/stats.cpp +# * domain/io/render.cpp # * containers/particles.cpp # * containers/particles_sort.cpp # * containers/fields.cpp @@ -64,15 +66,18 @@ set(SOURCES ${SRC_DIR}/parameters/output.cpp ${SRC_DIR}/parameters/algorithms.cpp ${SRC_DIR}/parameters/extra.cpp + ${SRC_DIR}/parameters/render.cpp ${SRC_DIR}/domain/grid.cpp ${SRC_DIR}/domain/metadomain.cpp ${SRC_DIR}/domain/metadomain_sort.cpp ${SRC_DIR}/domain/metadomain_reshape.cpp ${SRC_DIR}/domain/metadomain_loadbal.cpp + ${SRC_DIR}/domain/io/init.cpp + ${SRC_DIR}/domain/io/render.cpp + ${SRC_DIR}/domain/io/stats.cpp ${SRC_DIR}/domain/comm/fields.cpp ${SRC_DIR}/domain/comm/fields_sync.cpp ${SRC_DIR}/domain/comm/particles.cpp - ${SRC_DIR}/domain/io/stats.cpp ${SRC_DIR}/containers/particles.cpp ${SRC_DIR}/containers/particles_sort.cpp ${SRC_DIR}/containers/fields.cpp) @@ -80,7 +85,6 @@ if(${output}) list( APPEND SOURCES - ${SRC_DIR}/domain/io/init.cpp ${SRC_DIR}/domain/io/write.cpp ${SRC_DIR}/domain/io/spectra.cpp ${SRC_DIR}/domain/io/fields.cpp diff --git a/src/framework/containers/particles.h b/src/framework/containers/particles.h index 5f15b9621..bdca0858a 100644 --- a/src/framework/containers/particles.h +++ b/src/framework/containers/particles.h @@ -91,7 +91,7 @@ namespace ntt { const uint8_t m_ntags { (uint8_t)(2 + math::pow(3, (int)D) - 1) }; #endif - // team_policy: tile metadata produced by SortSpatially + // tiled_deposit: tile metadata produced by SortSpatially // and consumed by the tiled deposit / pusher kernels. Lazily // allocated on first sort. The sort backend itself (oneDPL on SYCL, // Thrust on CUDA, std::sort on Host, Kokkos::BinSort otherwise) is @@ -99,7 +99,7 @@ namespace ntt { // vendor libraries detected by CMake. TileLayout m_tile_layout {}; -#if defined(TEAM_POLICY) && \ +#if defined(TILED_DEPOSIT) && \ ((defined(SYCL_ENABLED) && defined(ONEDPL_ENABLED)) || \ (defined(CUDA_ENABLED) && defined(THRUST_ENABLED)) || \ (defined(HIP_ENABLED) && defined(ROCTHRUST_ENABLED))) @@ -234,7 +234,7 @@ namespace ntt { return m_ntags; } -#if defined(TEAM_POLICY) +#if defined(TILED_DEPOSIT) // Build m_tile_layout.tile_offsets / npart_partitioned from the // already-sorted tile-index keys. A separate member function (not a // lambda local to SortSpatially) so the inner device kernel is not an @@ -326,14 +326,14 @@ namespace ntt { /** * @brief Sort particles spatially by their cell indices * @param grid The grid object to get the cell information for sorting - * @note In team_policy mode (compile-time `team_policy=ON`), also + * @note In tiled_deposit mode (compile-time `tiled_deposit=ON`), also * populates `m_tile_layout` with tile-offset and per-tile * permutation metadata that the tiled deposit/pusher kernels * consume. */ void SortSpatially(const Grid&); -#if defined(TEAM_POLICY) && \ +#if defined(TILED_DEPOSIT) && \ ((defined(SYCL_ENABLED) && defined(ONEDPL_ENABLED)) || \ (defined(CUDA_ENABLED) && defined(THRUST_ENABLED)) || \ (defined(HIP_ENABLED) && defined(ROCTHRUST_ENABLED))) diff --git a/src/framework/containers/particles_sort.cpp b/src/framework/containers/particles_sort.cpp index 4e45b8db9..02a6584f4 100644 --- a/src/framework/containers/particles_sort.cpp +++ b/src/framework/containers/particles_sort.cpp @@ -8,11 +8,11 @@ #include "framework/containers/particles.h" #include "framework/domain/grid.h" -#if defined(TEAM_POLICY) +#if defined(TILED_DEPOSIT) #if (defined(SYCL_ENABLED) && defined(ONEDPL_ENABLED)) || \ (defined(CUDA_ENABLED) && defined(THRUST_ENABLED)) || \ (defined(HIP_ENABLED) && defined(ROCTHRUST_ENABLED)) - #define TEAM_POLICY_USE_VENDOR_SORT + #define TILED_DEPOSIT_USE_VENDOR_SORT #include "utils/sort_dispatch.h" #endif #endif @@ -231,7 +231,7 @@ namespace ntt { } } // namespace -#if defined(TEAM_POLICY) +#if defined(TILED_DEPOSIT) template void Particles::compute_tile_offsets(const array_t& tile_indices, ncells_t total_tiles, @@ -283,12 +283,12 @@ namespace ntt { // separately deposit) particles appended since this sort. m_tile_layout.npart_partitioned = h_offsets(total_tiles); } -#endif // TEAM_POLICY +#endif // TILED_DEPOSIT template void Particles::SortSpatially(const Grid& grid) { -#if defined(TEAM_POLICY) - // ---------------------- team_policy: tile-based sort ------------------ // +#if defined(TILED_DEPOSIT) + // ---------------------- tiled_deposit: tile-based sort ------------------ // const auto npart_local = npart(); if (npart_local == 0u) { m_tile_layout = TileLayout {}; @@ -296,8 +296,9 @@ namespace ntt { return; } - constexpr unsigned short T = static_cast(TEAM_POLICY_TILE_SIZE); - static_assert(T > 0u, "TEAM_POLICY_TILE_SIZE must be > 0"); + constexpr unsigned short T = static_cast( + TILED_DEPOSIT_TILE_SIZE); + static_assert(T > 0u, "TILED_DEPOSIT_TILE_SIZE must be > 0"); // 1. Compute per-axis tile counts and total_tiles. const auto ncells_active = grid.n_active(); @@ -320,7 +321,7 @@ namespace ntt { } // 2. Compute per-particle tile key (with min(i, i_prev)). - #if defined(TEAM_POLICY_USE_VENDOR_SORT) && defined(SYCL_ENABLED) && \ + #if defined(TILED_DEPOSIT_USE_VENDOR_SORT) && defined(SYCL_ENABLED) && \ defined(ONEDPL_ENABLED) // oneDPL sorts the keys in place, so reuse a persistent, grow-only keys // buffer instead of allocating a fresh one every sort. `tile_indices` @@ -353,7 +354,7 @@ namespace ntt { const ncells_t n_bins = total_tiles + 2u; const auto slice = prtl_slice_t(0, npart_local); - #if defined(TEAM_POLICY_USE_VENDOR_SORT) + #if defined(TILED_DEPOSIT_USE_VENDOR_SORT) // Vendor path: produce an explicit permutation via sort_by_key, then // apply it to each SoA member by gathering the alive prefix through a // reusable scratch buffer (one per member type, copied back in place). @@ -451,7 +452,7 @@ namespace ntt { // gather to hoist this ahead of). sorter.sort(tile_indices); compute_tile_offsets(tile_indices, total_tiles, npart_local); - #endif // TEAM_POLICY_USE_VENDOR_SORT + #endif // TILED_DEPOSIT_USE_VENDOR_SORT // Populate `m_tile_layout` size/shape. `tile_perm` is not used in the // current design — the SoA arrays are physically permuted into tile @@ -475,8 +476,8 @@ namespace ntt { // (RemoveDead remains the compactor when spatial sorting is disabled.) set_npart(m_tile_layout.npart_partitioned); - Kokkos::fence("SortSpatially: end of team_policy path"); -#else // !TEAM_POLICY — legacy in-place BinSort by global cell index + Kokkos::fence("SortSpatially: end of tiled_deposit path"); +#else // !TILED_DEPOSIT — legacy in-place BinSort by global cell index const auto nx2 = grid.n_active(in::x2); const auto nx3 = grid.n_active(in::x3); const auto total_cells = grid.num_active(); @@ -532,10 +533,10 @@ namespace ntt { for (auto pldi { 0u }; pldi < npld_i(); ++pldi) { sorter.sort(Kokkos::subview(pld_i, slice, pldi)); } -#endif // TEAM_POLICY +#endif // TILED_DEPOSIT } -#if defined(TEAM_POLICY_USE_VENDOR_SORT) +#if defined(TILED_DEPOSIT_USE_VENDOR_SORT) namespace permute_helpers { // Permute a 1D SoA member `arr` by `perm` in place, using a @@ -694,9 +695,9 @@ namespace ntt { permute_2d_into(pld_i, m_sort_scratch_pld_i, perm, n, ncols); } } -#endif // TEAM_POLICY_USE_VENDOR_SORT +#endif // TILED_DEPOSIT_USE_VENDOR_SORT -#if defined(TEAM_POLICY_USE_VENDOR_SORT) +#if defined(TILED_DEPOSIT_USE_VENDOR_SORT) #define APPLY_PERM_INSTANTIATE(D, C) \ template void Particles::apply_permutation_to_soa(const prtl_perm_t&, \ npart_t); diff --git a/src/framework/domain/comm/fields.cpp b/src/framework/domain/comm/fields.cpp index 121ad48e5..e5d58431d 100644 --- a/src/framework/domain/comm/fields.cpp +++ b/src/framework/domain/comm/fields.cpp @@ -191,10 +191,44 @@ namespace ntt { } } + template + void Metadomain::CommunicateBckp(Domain& domain, + const cell_range_t& components) const { + // Halo FILL of the bckp buffer: copy each neighbor's active boundary cells + // into this domain's ghost zones (additive=false). This is distinct from + // SynchronizeFields, which sums ghost-deposited values back into active + // cells. The renderer needs the ghost halo populated so trilinear sampling + // near a domain face reads valid neighbor values (C0 across the face). + for (auto& direction : dir::Directions::all) { + const auto [send_params, + recv_params] = GetSendRecvParams(this, domain, direction, false); + const auto [send_indrank, send_slice] = send_params; + const auto [recv_indrank, recv_slice] = recv_params; + const auto [send_ind, send_rank] = send_indrank; + const auto [recv_ind, recv_rank] = recv_indrank; + if (send_rank < 0 and recv_rank < 0) { + continue; + } + comm::CommunicateField(domain.index(), + domain.fields.bckp, + domain.fields.bckp, + send_ind, + recv_ind, + send_rank, + recv_rank, + send_slice, + recv_slice, + components, + false); + } + } + // NOLINTBEGIN(bugprone-macro-parentheses) #define METADOMAIN_COMM(S, M, D) \ template void Metadomain>::CommunicateFields(Domain>&, \ - CommTags) const; + CommTags) const; \ + template void Metadomain>::CommunicateBckp(Domain>&, \ + const cell_range_t&) const; NTT_FOREACH_SPECIALIZATION(METADOMAIN_COMM) #undef METADOMAIN_COMM diff --git a/src/framework/domain/io/init.cpp b/src/framework/domain/io/init.cpp index 472ba9618..2f563f89a 100644 --- a/src/framework/domain/io/init.cpp +++ b/src/framework/domain/io/init.cpp @@ -1,4 +1,3 @@ -#include "defaults.h" #include "enums.h" #include "global.h" @@ -6,7 +5,6 @@ #include "utils/error.h" #include "framework/domain/domain.h" -#include "framework/domain/mesh.h" #include "framework/domain/metadomain.h" #include "framework/parameters/parameters.h" #include "framework/specialization_registry.h" @@ -15,14 +13,23 @@ #include #include -#include -#include -#include +#if defined(OUTPUT_ENABLED) + #include "defaults.h" + + #include + #include + + #include + #include + #include +#endif // OUTPUT_ENABLED + #include #include namespace ntt { +#if defined(OUTPUT_ENABLED) template void Metadomain::InitWriter(adios2::ADIOS* ptr_adios, const SimulationParams& params) { @@ -105,6 +112,7 @@ namespace ntt { } g_writer.writeAttrs(params); } +#endif template void Metadomain::InitStatsWriter(const SimulationParams& params, @@ -146,16 +154,30 @@ namespace ntt { } } + template + void Metadomain::InitRenderer(const SimulationParams& params) { + g_renderer.init(params, mesh().extent()); + } + +#if defined(OUTPUT_ENABLED) // NOLINTBEGIN(bugprone-macro-parentheses) -#define METADOMAIN_OUTPUT(S, M, D) \ - template void Metadomain>::InitWriter(adios2::ADIOS*, \ - const SimulationParams&); \ - template void Metadomain>::InitStatsWriter(const SimulationParams&, \ - bool); + #define METADOMAIN_OUTPUT_INIT(S, M, D) \ + template void Metadomain>::InitWriter(adios2::ADIOS*, \ + const SimulationParams&); + + NTT_FOREACH_SPECIALIZATION(METADOMAIN_OUTPUT_INIT) - NTT_FOREACH_SPECIALIZATION(METADOMAIN_OUTPUT) + #undef METADOMAIN_OUTPUT_INIT + // NOLINTEND(bugprone-macro-parentheses) +#endif -#undef METADOMAIN_OUTPUT + // NOLINTBEGIN(bugprone-macro-parentheses) +#define METADOMAIN_OUTPUT_INIT(S, M, D) \ + template void Metadomain>::InitStatsWriter(const SimulationParams&, \ + bool); \ + template void Metadomain>::InitRenderer(const SimulationParams&); + NTT_FOREACH_SPECIALIZATION(METADOMAIN_OUTPUT_INIT) +#undef METADOMAIN_OUTPUT_INIT // NOLINTEND(bugprone-macro-parentheses) } // namespace ntt diff --git a/src/framework/domain/io/render.cpp b/src/framework/domain/io/render.cpp new file mode 100644 index 000000000..c67d2f44e --- /dev/null +++ b/src/framework/domain/io/render.cpp @@ -0,0 +1,1545 @@ +/** + * @file framework/domain/io/render.cpp + * @brief Metadomain driver for the in-situ volume renderer + * @implements + * - ntt::Metadomain::InitRenderer + * - ntt::Metadomain::Render + * @namespaces: + * - ntt:: + * @macros: + * - MPI_ENABLED + * @note + * This is the templated counterpart of the (plain) out::Renderer: it owns the + * per-(engine, metric, dim) field preparation, the device ray-march kernel + * launch, and the device->host copy. It reuses the exact field-prep code paths + * as Metadomain::Write (ComputeMoments / FieldsToPhys), then hands the per-rank + * host image to out::Renderer for the MPI ordered composite and PNG write. + * Active only for 3D Minkowski; a no-op otherwise (the structured-order + * composite assumes an axis-aligned, affine code<->world map). + */ + +#include "enums.h" +#include "global.h" + +#include "arch/kokkos_aliases.h" +#include "traits/metric.h" +#include "utils/error.h" +#include "utils/log.h" + +#include "framework/containers/particles.h" +#include "framework/domain/domain.h" +#include "framework/domain/mesh.h" +#include "framework/domain/metadomain.h" +#include "framework/parameters/parameters.h" +#include "framework/specialization_registry.h" +#include "kernels/fields_to_phys.hpp" +#include "kernels/particle_moments.hpp" +#include "output/render/composite.h" +#include "output/render/fieldlines.h" + +#include "output/render/raymarch.hpp" +#include "output/render/reduce.hpp" +#include "output/render/slice2d.hpp" + +#include +#include + +#if defined(MPI_ENABLED) + #include "arch/mpi_aliases.h" + + #include +#endif + +#include +#include +#include +#include +#include +#include +#include + +namespace ntt { + + namespace { + + // Mirror of Metadomain::Write's ComputeMoments, kept with internal linkage + // so it does not collide with the (identically-named) one in metadomain_io. + template + void renderMoment(const SimulationParams& params, + const Mesh& mesh, + const std::vector>& prtl_species, + const std::vector& species, + const std::vector& components, + ndfield_t& buffer, + idx_t buff_idx) { + std::vector specs = species; + if (specs.empty()) { + // default: accumulate over all massive species + for (auto& sp : prtl_species) { + if (sp.mass() > 0) { + specs.push_back(sp.index()); + } + } + } + for (const auto& sp : specs) { + raise::ErrorIf((sp > prtl_species.size()) or (sp == 0), + "Invalid species index " + std::to_string(sp), + HERE); + } + auto scatter_buff = Kokkos::Experimental::create_scatter_view(buffer); + const auto use_weights = params.get("particles.use_weights"); + const auto ni2 = mesh.n_active(in::x2); + const auto inv_n0 = ONE / params.get("scales.n0"); + const auto smooth_order = params.get( + "output.fields.smoothing.order"); + const auto smooth_method = OutputSmoothingType::from_string( + params.get("output.fields.smoothing.method")); + for (const auto& sp : specs) { + auto& prtl_spec = prtl_species[sp - 1]; + Kokkos::parallel_for( + "RenderComputeMoments", + prtl_spec.rangeActiveParticles(), + kernel::ParticleMoments_kernel(components, + scatter_buff, + buff_idx, + prtl_spec, + use_weights, + mesh.metric, + mesh.flds_bc(), + ni2, + inv_n0, + smooth_order, + smooth_method)); + } + Kokkos::Experimental::contribute(buffer, scatter_buff); + } + + // Copy 3 contiguous components [from.first, from.first+3) of a source field + // into bckp(:, 0..2). Dimension-generic (the subview arity depends on D). + template + void copyVec3ToBckp(const ndfield_t& src, + const ndfield_t& dst, + const cell_range_t& from) { + const cell_range_t to { 0, 3 }; + if constexpr (D == Dim::_2D) { + Kokkos::deep_copy(Kokkos::subview(dst, Kokkos::ALL, Kokkos::ALL, to), + Kokkos::subview(src, Kokkos::ALL, Kokkos::ALL, from)); + } else if constexpr (D == Dim::_3D) { + Kokkos::deep_copy( + Kokkos::subview(dst, Kokkos::ALL, Kokkos::ALL, Kokkos::ALL, to), + Kokkos::subview(src, Kokkos::ALL, Kokkos::ALL, Kokkos::ALL, from)); + } + } + + // Volume-average this domain's physical-basis vector field (B/E/J) onto a + // GLOBAL coarse grid, then MPI-replicate it so every rank holds the same + // field and can trace identical global field lines locally. `bckp` is used + // as scratch (overwritten). 3D only (the field-line renderer is Cartesian). + template + auto buildCoarseFieldVec(const Mesh& mesh, + const Fields& fields, + ndfield_t& bckp, + char fbase, + const real_t gorigin[3], + const int gnc[3], + const real_t gdx[3]) -> out::CoarseField { + const auto metric = mesh.metric; + // raw vector components -> bckp(0,1,2) + uint8_t src_base = em::bx1; + PrepareOutputFlags interp = PrepareOutput::InterpToCellCenterFromFaces; + bool is_current = false; + if (fbase == 'E') { + src_base = em::ex1; + interp = PrepareOutput::InterpToCellCenterFromEdges; + } else if (fbase == 'J') { + is_current = true; + src_base = cur::jx1; + interp = PrepareOutput::InterpToCellCenterFromEdges; + } + if (is_current) { + copyVec3ToBckp(fields.cur, + bckp, + cell_range_t(cur::jx1, cur::jx3 + 1)); + } else { + copyVec3ToBckp(fields.em, + bckp, + cell_range_t(src_base, src_base + 3)); + } + // interpolate to cell centers + convert to physical basis -> bckp(3,4,5) + const PrepareOutputFlags prepare = (S == SimEngine::SRPIC) + ? PrepareOutput::ConvertToHat + : PrepareOutput::ConvertToPhysCntrv; + list_t comp_from = { 0, 1, 2 }; + list_t comp_to = { 3, 4, 5 }; + Kokkos::parallel_for("RenderFLFieldsToPhys", + mesh.rangeActiveCells(), + kernel::FieldsToPhys_kernel(bckp, + bckp, + comp_from, + comp_to, + interp | prepare, + metric)); + Kokkos::fence(); + + // pull the physical components to host and bin into the coarse grid + auto bckp_h = Kokkos::create_mirror_view(bckp); + Kokkos::deep_copy(bckp_h, bckp); + + const std::size_t ncell = static_cast(gnc[0]) * + static_cast(gnc[1]) * + static_cast(gnc[2]); + std::vector sum(ncell * 3, ZERO); + std::vector cnt(ncell, ZERO); + + const auto le = mesh.extent(); + const real_t llo[3] = { le[0].first, le[1].first, le[2].first }; + const real_t lsz[3] = { le[0].second - le[0].first, + le[1].second - le[1].first, + le[2].second - le[2].first }; + const int nl[3] = { static_cast(mesh.n_active(in::x1)), + static_cast(mesh.n_active(in::x2)), + static_cast(mesh.n_active(in::x3)) }; + const int NG = static_cast(N_GHOSTS); + for (int k = 0; k < nl[2]; ++k) { + for (int j = 0; j < nl[1]; ++j) { + for (int i = 0; i < nl[0]; ++i) { + const real_t world[3] = { + llo[0] + (static_cast(i) + HALF) * lsz[0] / + static_cast(nl[0]), + llo[1] + (static_cast(j) + HALF) * lsz[1] / + static_cast(nl[1]), + llo[2] + (static_cast(k) + HALF) * lsz[2] / + static_cast(nl[2]) + }; + int c[3]; + for (int d = 0; d < 3; ++d) { + int cc = static_cast( + std::floor((world[d] - gorigin[d]) / gdx[d])); + cc = (cc < 0) ? 0 : ((cc > gnc[d] - 1) ? gnc[d] - 1 : cc); + c[d] = cc; + } + const std::size_t lin = (static_cast(c[2]) * gnc[1] + + c[1]) * + gnc[0] + + c[0]; + sum[lin * 3 + 0] += bckp_h(i + NG, j + NG, k + NG, 3); + sum[lin * 3 + 1] += bckp_h(i + NG, j + NG, k + NG, 4); + sum[lin * 3 + 2] += bckp_h(i + NG, j + NG, k + NG, 5); + cnt[lin] += ONE; + } + } + } +#if defined(MPI_ENABLED) + MPI_Allreduce(MPI_IN_PLACE, + sum.data(), + static_cast(ncell * 3), + mpi::get_type(), + MPI_SUM, + MPI_COMM_WORLD); + MPI_Allreduce(MPI_IN_PLACE, + cnt.data(), + static_cast(ncell), + mpi::get_type(), + MPI_SUM, + MPI_COMM_WORLD); +#endif + out::CoarseField cf; + cf.B.assign(ncell * 3, ZERO); + for (int d = 0; d < 3; ++d) { + cf.n[d] = gnc[d]; + cf.origin[d] = gorigin[d]; + cf.dx[d] = gdx[d]; + } + for (std::size_t c = 0; c < ncell; ++c) { + if (cnt[c] > ZERO) { + const real_t inv = ONE / cnt[c]; + cf.B[c * 3 + 0] = sum[c * 3 + 0] * inv; + cf.B[c * 3 + 1] = sum[c * 3 + 1] * inv; + cf.B[c * 3 + 2] = sum[c * 3 + 2] * inv; + } + } + return cf; + } + + // 2D analogue of buildCoarseFieldVec: volume-average the in-plane physical + // components (Bx, By) onto a coarse global 2D grid and MPI-replicate them, + // so every rank can integrate the SAME flux function for seamless contours. + template + auto buildCoarseField2D(const Mesh& mesh, + const Fields& fields, + ndfield_t& bckp, + char fbase, + const real_t gorigin[2], + const int gnc[2], + const real_t gdx[2]) -> out::CoarseField2D { + const auto metric = mesh.metric; + uint8_t src_base = em::bx1; + PrepareOutputFlags interp = PrepareOutput::InterpToCellCenterFromFaces; + bool is_current = false; + if (fbase == 'E') { + src_base = em::ex1; + interp = PrepareOutput::InterpToCellCenterFromEdges; + } else if (fbase == 'J') { + is_current = true; + src_base = cur::jx1; + interp = PrepareOutput::InterpToCellCenterFromEdges; + } + if (is_current) { + copyVec3ToBckp(fields.cur, + bckp, + cell_range_t(cur::jx1, cur::jx3 + 1)); + } else { + copyVec3ToBckp(fields.em, + bckp, + cell_range_t(src_base, src_base + 3)); + } + const PrepareOutputFlags prepare = (S == SimEngine::SRPIC) + ? PrepareOutput::ConvertToHat + : PrepareOutput::ConvertToPhysCntrv; + list_t comp_from = { 0, 1, 2 }; + list_t comp_to = { 3, 4, 5 }; + Kokkos::parallel_for("RenderFL2DFieldsToPhys", + mesh.rangeActiveCells(), + kernel::FieldsToPhys_kernel(bckp, + bckp, + comp_from, + comp_to, + interp | prepare, + metric)); + Kokkos::fence(); + + auto bckp_h = Kokkos::create_mirror_view(bckp); + Kokkos::deep_copy(bckp_h, bckp); + + const std::size_t ncell = static_cast(gnc[0]) * gnc[1]; + std::vector sum(ncell * 2, ZERO); + std::vector cnt(ncell, ZERO); + const auto le = mesh.extent(); + const real_t llo[2] = { le[0].first, le[1].first }; + const real_t lsz[2] = { le[0].second - le[0].first, + le[1].second - le[1].first }; + const int nl[2] = { static_cast(mesh.n_active(in::x1)), + static_cast(mesh.n_active(in::x2)) }; + const int NG = static_cast(N_GHOSTS); + for (int j = 0; j < nl[1]; ++j) { + for (int i = 0; i < nl[0]; ++i) { + const real_t world[2] = { llo[0] + (static_cast(i) + HALF) * + lsz[0] / static_cast(nl[0]), + llo[1] + (static_cast(j) + HALF) * + lsz[1] / + static_cast(nl[1]) }; + int c[2]; + for (int d = 0; d < 2; ++d) { + int cc = static_cast(std::floor((world[d] - gorigin[d]) / gdx[d])); + cc = (cc < 0) ? 0 : ((cc > gnc[d] - 1) ? gnc[d] - 1 : cc); + c[d] = cc; + } + const std::size_t lin = static_cast(c[1]) * gnc[0] + c[0]; + sum[lin * 2 + 0] += bckp_h(i + NG, j + NG, 3); // Bx + sum[lin * 2 + 1] += bckp_h(i + NG, j + NG, 4); // By + cnt[lin] += ONE; + } + } +#if defined(MPI_ENABLED) + MPI_Allreduce(MPI_IN_PLACE, + sum.data(), + static_cast(ncell * 2), + mpi::get_type(), + MPI_SUM, + MPI_COMM_WORLD); + MPI_Allreduce(MPI_IN_PLACE, + cnt.data(), + static_cast(ncell), + mpi::get_type(), + MPI_SUM, + MPI_COMM_WORLD); +#endif + out::CoarseField2D cf; + cf.B.assign(ncell * 2, ZERO); + for (int d = 0; d < 2; ++d) { + cf.n[d] = gnc[d]; + cf.origin[d] = gorigin[d]; + cf.dx[d] = gdx[d]; + } + for (std::size_t c = 0; c < ncell; ++c) { + if (cnt[c] > ZERO) { + const real_t inv = ONE / cnt[c]; + cf.B[c * 2 + 0] = sum[c * 2 + 0] * inv; + cf.B[c * 2 + 1] = sum[c * 2 + 1] * inv; + } + } + return cf; + } + + } // namespace + + template + auto Metadomain::prepareRenderScalar(const SimulationParams& params, + Domain& domain, + const std::string& field_name, + ndfield_t& bckp) const + -> bool { + // Parse an optional trailing per-species suffix "__..."; + // species apply to particle moments only (N, Nppc, Rho, Charge, T, V). + std::string base = field_name; + std::vector species; + { + const auto us = field_name.find('_'); + if (us != std::string::npos) { + bool ok_sp = true; + std::size_t start = us + 1; + while (start <= field_name.size()) { + const auto nx = field_name.find('_', start); + const auto tok = field_name.substr( + start, + (nx == std::string::npos) ? std::string::npos : nx - start); + if (tok.empty() or + tok.find_first_not_of("0123456789") != std::string::npos) { + ok_sp = false; + break; + } + species.push_back(static_cast(std::stoi(tok))); + if (nx == std::string::npos) { + break; + } + start = nx + 1; + } + if (ok_sp) { + base = field_name.substr(0, us); + } else { + species.clear(); // not a species suffix; keep the full name + } + } + } + bool bad_species = false; + for (const auto sp : species) { + if (sp == 0 or sp > domain.species.size()) { + bad_species = true; + } + } + if (bad_species) { + raise::Warning("render: invalid species in '" + field_name + "', skipping", + HERE); + return false; + } + + // axis/index character -> {t,x,y,z} == {0,1,2,3}; -1 if invalid + auto axisIdx = [](char ch) -> int { + switch (ch) { + case 't': + case '0': + return 0; + case 'x': + case '1': + return 1; + case 'y': + case '2': + return 2; + case 'z': + case '3': + return 3; + default: + return -1; + } + }; + + const auto& mesh = domain.mesh; + const auto metric = mesh.metric; + + if (base == "N" or base == "Nppc" or base == "Rho" or base == "Charge") { + // scalar particle moments + if (base == "N") { + renderMoment(params, mesh, domain.species, species, {}, bckp, 0u); + } else if (base == "Nppc") { + renderMoment(params, + mesh, + domain.species, + species, + {}, + bckp, + 0u); + } else if (base == "Rho") { + renderMoment(params, + mesh, + domain.species, + species, + {}, + bckp, + 0u); + } else { + renderMoment(params, + mesh, + domain.species, + species, + {}, + bckp, + 0u); + } + // sum boundary-crossing particle deposits back into active cells + SynchronizeFields(domain, Comm::Bckp, { 0, 1 }); + return true; + } else if (base.size() == 3 and base[0] == 'T') { + // a single stress-energy tensor component "T" (same for SR & GR; + // the moment kernel branches on the engine internally) + const int i = axisIdx(base[1]); + const int j = axisIdx(base[2]); + if (i >= 0 and j >= 0) { + const std::vector comps { static_cast(i), + static_cast(j) }; + renderMoment(params, + mesh, + domain.species, + species, + comps, + bckp, + 0u); + SynchronizeFields(domain, Comm::Bckp, { 0, 1 }); + return true; + } + } else if (base == "Vmag") { + // bulk-velocity magnitude |V| = sqrt(V1^2 + V2^2 + V3^2) + if constexpr (S == SimEngine::GRPIC) { + // GR: Eckart-frame 4-velocity; need all 4 components for the norm + renderMoment(params, + mesh, + domain.species, + species, + { 0u }, + bckp, + 0u); + renderMoment(params, + mesh, + domain.species, + species, + { 1u }, + bckp, + 1u); + renderMoment(params, + mesh, + domain.species, + species, + { 2u }, + bckp, + 2u); + renderMoment(params, + mesh, + domain.species, + species, + { 3u }, + bckp, + 3u); + SynchronizeFields(domain, Comm::Bckp, { 0, 4 }); + Kokkos::parallel_for( + "RenderNormalize4Vel", + mesh.rangeActiveCells(), + kernel::Normalize4VelocityByNorm_kernel(bckp, + bckp, + 0, + 1, + 2, + 3, + metric)); + Kokkos::parallel_for( + "RenderTransform4Vel", + mesh.rangeActiveCells(), + kernel::Transform4VelocitySpatialToPhysical_kernel(bckp, + 1, + 2, + 3, + metric)); + // |spatial physical 4-velocity| -> bckp(0) + Kokkos::parallel_for( + "RenderVmagGR", + mesh.rangeActiveCells(), + render::RenderMagnitude3_kernel(bckp, 1, 2, 3, 0)); + } else { + // SR: mass-weighted bulk 3-velocity, normalized by Rho + renderMoment(params, + mesh, + domain.species, + species, + { 1u }, + bckp, + 0u); + renderMoment(params, + mesh, + domain.species, + species, + { 2u }, + bckp, + 1u); + renderMoment(params, + mesh, + domain.species, + species, + { 3u }, + bckp, + 2u); + renderMoment(params, + mesh, + domain.species, + species, + {}, + bckp, + 3u); + SynchronizeFields(domain, Comm::Bckp, { 0, 4 }); + Kokkos::parallel_for( + "RenderVmagSR", + mesh.rangeActiveCells(), + render::RenderVmagByRho_kernel(bckp, 0, 1, 2, 3, 0)); + } + return true; + } else if (base.size() == 2 and base[0] == 'V') { + // a single bulk-velocity component "V" + const int c = axisIdx(base[1]); + if constexpr (S == SimEngine::GRPIC) { + // GR: 4-velocity component (t/0 = u^0 = Gamma/alpha; x,y,z spatial) + if (c >= 0 and c <= 3) { + renderMoment(params, + mesh, + domain.species, + species, + { 0u }, + bckp, + 0u); + renderMoment(params, + mesh, + domain.species, + species, + { 1u }, + bckp, + 1u); + renderMoment(params, + mesh, + domain.species, + species, + { 2u }, + bckp, + 2u); + renderMoment(params, + mesh, + domain.species, + species, + { 3u }, + bckp, + 3u); + SynchronizeFields(domain, Comm::Bckp, { 0, 4 }); + Kokkos::parallel_for( + "RenderNormalize4Vel", + mesh.rangeActiveCells(), + kernel::Normalize4VelocityByNorm_kernel(bckp, + bckp, + 0, + 1, + 2, + 3, + metric)); + Kokkos::parallel_for( + "RenderTransform4Vel", + mesh.rangeActiveCells(), + kernel::Transform4VelocitySpatialToPhysical_kernel( + bckp, + 1, + 2, + 3, + metric)); + if (c != 0) { + Kokkos::parallel_for( + "RenderPickV", + mesh.rangeActiveCells(), + render::RenderPickComp_kernel(bckp, + static_cast(c), + 0)); + } + return true; + } + } else { + // SR: spatial bulk velocity (x,y,z), normalized by Rho + if (c >= 1 and c <= 3) { + renderMoment(params, + mesh, + domain.species, + species, + { static_cast(c) }, + bckp, + 0u); + renderMoment(params, + mesh, + domain.species, + species, + {}, + bckp, + 1u); + SynchronizeFields(domain, Comm::Bckp, { 0, 2 }); + Kokkos::parallel_for( + "RenderNormalizeV", + mesh.rangeActiveCells(), + render::RenderDivideComp_kernel(bckp, 0, 1)); + return true; + } + } + } else { + // Vector field as a scalar: "" with base in {E, B, J} + // and selector in {mag, 1/2/3, x/y/z}. A component (e.g. "B1"/"Bx") is + // signed; a magnitude (e.g. "Bmag") is non-negative. + const std::string& f = base; + const char fbase = f.empty() ? '?' : static_cast(std::toupper(f[0])); + bool ok = true; + bool is_current = false; + uint8_t src_base = 0; // first component of the source field + PrepareOutputFlags interp = PrepareOutput::None; + if (fbase == 'B') { + src_base = em::bx1; + interp = PrepareOutput::InterpToCellCenterFromFaces; + } else if (fbase == 'E') { + src_base = em::ex1; + interp = PrepareOutput::InterpToCellCenterFromEdges; + } else if (fbase == 'J') { + is_current = true; + src_base = cur::jx1; + interp = PrepareOutput::InterpToCellCenterFromEdges; + } else { + ok = false; + } + // selector: -1 = magnitude, 0/1/2 = a single component + int comp = -2; + const std::string sel = (f.size() > 1) ? f.substr(1) : std::string {}; + if (sel == "mag") { + comp = -1; + } else if (sel == "1" or sel == "x") { + comp = 0; + } else if (sel == "2" or sel == "y") { + comp = 1; + } else if (sel == "3" or sel == "z") { + comp = 2; + } else { + ok = false; + } + if (ok) { + // raw vector components into bckp(:, 0..2) + if (is_current) { + copyVec3ToBckp(domain.fields.cur, + bckp, + cell_range_t(cur::jx1, cur::jx3 + 1)); + } else { + copyVec3ToBckp(domain.fields.em, + bckp, + cell_range_t(src_base, src_base + 3)); + } + // interpolate to cell centers + convert to physical basis -> (3,4,5) + const PrepareOutputFlags prepare = (S == SimEngine::SRPIC) + ? PrepareOutput::ConvertToHat + : PrepareOutput::ConvertToPhysCntrv; + list_t comp_from = { 0, 1, 2 }; + list_t comp_to = { 3, 4, 5 }; + Kokkos::parallel_for("RenderFieldsToPhys", + mesh.rangeActiveCells(), + kernel::FieldsToPhys_kernel(bckp, + bckp, + comp_from, + comp_to, + interp | prepare, + metric)); + // reduce to the scalar to render -> bckp(:, 0) + if (comp == -1) { + Kokkos::parallel_for( + "RenderVectorMagnitude", + mesh.rangeActiveCells(), + render::RenderMagnitude3_kernel(bckp, 3, 4, 5, 0)); + } else { + Kokkos::parallel_for("RenderVectorComponent", + mesh.rangeActiveCells(), + render::RenderPickComp_kernel( + bckp, + static_cast(3 + comp), + 0)); + } + return true; + } + } + + raise::Warning("render: unknown field '" + field_name + + "' (expected N/Nppc/Rho/Charge, T{i}{j}, V{i}/Vmag, or " + "{E,B,J}{mag,1,2,3,x,y,z}); skipping", + HERE); + return false; + } + + template + auto Metadomain::Render(const SimulationParams& params, + timestep_t current_step, + timestep_t finished_step, + simtime_t current_time, + simtime_t finished_time) -> bool { + (void)current_time; + if constexpr (M::Dim == Dim::_3D and M::CoordType == Coord::type::Cartesian) { + // ---- 3D volume ray-march (Minkowski only) ----------------------- // + // structured-order composite assumes an axis-aligned, affine code<->world + // map; only Cartesian (Minkowski) 3D qualifies. + if (not g_renderer.enabled() or + not g_renderer.shouldRender(finished_step, finished_time)) { + return false; + } + raise::ErrorIf(l_subdomain_indices().size() != 1, + "Renderer supports one subdomain per rank only", + HERE); + auto local_domain = subdomain_ptr(l_subdomain_indices()[0]); + raise::ErrorIf(local_domain->is_placeholder(), + "local_domain is a placeholder", + HERE); + logger::Checkpoint("Rendering output (3D volume)", HERE); + + // advance the moving view (region + camera) to this frame's time before + // reading camera()/region(); collective (same time on all ranks). + g_renderer.updateForTime(current_time); + + const auto& cam = g_renderer.camera(); + const int W = g_renderer.width(); + const int H = g_renderer.height(); + + // fulldome fisheye from an interior eye: the composite is depth-resolved + // (A-buffer) instead of the ordered SubImage tree, and the frame is a + // clean square (setDomeActive -> writeFrame drops axes/colorbar margins). + const bool is_dome = (cam.projection == out::CameraDevice::Dome); + g_renderer.setDomeActive(is_dome); + + // optional axis-aligned render region (== full extent when uncropped) + const real_t rlo[3] = { g_renderer.regionLo(0), + g_renderer.regionLo(1), + g_renderer.regionLo(2) }; + const real_t rhi[3] = { g_renderer.regionHi(0), + g_renderer.regionHi(1), + g_renderer.regionHi(2) }; + // per-domain world AABB, clipped to the region + const auto loc_ext = local_domain->mesh.extent(); + real_t lo[3] = { math::max(loc_ext[0].first, rlo[0]), + math::max(loc_ext[1].first, rlo[1]), + math::max(loc_ext[2].first, rlo[2]) }; + real_t hi[3] = { math::min(loc_ext[0].second, rhi[0]), + math::min(loc_ext[1].second, rhi[1]), + math::min(loc_ext[2].second, rhi[2]) }; + // does this domain intersect the region? if not, render nothing (but + // still join the collective composite / field-line reduce below). + const bool in_region = (lo[0] < hi[0]) and (lo[1] < hi[1]) and + (lo[2] < hi[2]); + + // global extent (drives the field-line coarse grid, which spans the full + // field regardless of the crop) + const auto glob_ext = mesh().extent(); + // fixed world step, identical on all ranks -> seamless. Sized to the + // region diagonal so `samples` spans the (possibly cropped) view. + real_t gdiag = ZERO; + for (auto d { 0 }; d < 3; ++d) { + const real_t s = rhi[d] - rlo[d]; + gdiag += s * s; + } + gdiag = math::sqrt(gdiag); + // marched extent per ray: the dome clips each ray to `dome_radius`, so + // size the step by the radius (== `samples` steps across the hemisphere) + // rather than the box diagonal. Identical on all ranks -> seamless. + const real_t march_len = (is_dome and cam.dome_radius > ZERO) + ? cam.dome_radius + : gdiag; + const real_t ds = (g_renderer.stepSize() > ZERO) + ? g_renderer.stepSize() + : march_len / static_cast(g_renderer.samples()); + const int max_steps = 2 * g_renderer.samples() + 16; + + // region box + depth-occluded spine (opaque box wireframe rendered inline + // in the march so the volume covers its far edges). The visual width is + // ~spine_width px; the 0.55*ds floor keeps the thin line gap-free at the + // current sampling (raise `samples` for a crisper, thinner line). + real_t glo[3] = { rlo[0], rlo[1], rlo[2] }; + real_t ghi[3] = { rhi[0], rhi[1], rhi[2] }; + const real_t px_w = (cam.half_h * static_cast(2)) / + static_cast(H); + const real_t spine_radius = g_renderer.axes() + ? math::max(static_cast(0.55) * ds, + HALF * g_renderer.spineWidth() * px_w) + : ZERO; + // contrasting opaque spine color (white on dark bg, black on light) + const real_t bg_lum = static_cast(0.299) * g_renderer.background(0) + + static_cast(0.587) * g_renderer.background(1) + + static_cast(0.114) * g_renderer.background(2); + const real_t sc = (bg_lum < HALF) ? ONE : ZERO; + const real_t spine_rgb[3] = { sc, sc, sc }; + + // composite order key (depends on the current decomposition offsets) + const uint64_t order_key = out::compositeOrderKey( + local_domain->offset_ndomains(), + ndomains_per_dim(), + cam.forward); + + auto& bckp = local_domain->fields.bckp; + const int ext0 = static_cast(bckp.extent(0)); + const int ext1 = static_cast(bckp.extent(1)); + const int ext2 = static_cast(bckp.extent(2)); + + const auto metric = local_domain->mesh.metric; + + // screen-space bounding box of this domain's footprint (same for all + // scenes); we only ray-march and composite within it. + int bx0 = 0, by0 = 0, bw = 0, bh = 0; + const bool on_screen = + in_region and + (is_dome ? out::screenBBoxDome(cam, W, H, lo, hi, bx0, by0, bw, bh) + : out::screenBBox(cam, W, H, lo, hi, bx0, by0, bw, bh)); + + // ---- magnetic-field-line tubes (built once, shared by every scene) --- + // // Every rank coarsens + replicates the field, traces the SAME global + // polylines, and keeps only the segments inside its own domain; the + // ordered cross-domain composite stitches them. Built before the scene + // loop so an overlay and a standalone tube scene share one geometry pass. + // NB: all ranks reach this together (cadence is collective), so the + // Allreduce inside buildCoarseFieldVec is safe. + const auto& flc = g_renderer.fieldlines(); + out::TubeSet tubes = out::emptyTubeSet(); + out::TubeSet empty = out::emptyTubeSet(); + bool have_tubes = false; + if (flc.enable) { + const int gN[3] = { static_cast(mesh().n_active(in::x1)), + static_cast(mesh().n_active(in::x2)), + static_cast(mesh().n_active(in::x3)) }; + int gnc[3]; + real_t gorigin[3], gdx[3]; + for (int d = 0; d < 3; ++d) { + gnc[d] = std::max(1, (gN[d] + flc.bin - 1) / flc.bin); + gorigin[d] = glob_ext[d].first; + gdx[d] = (glob_ext[d].second - glob_ext[d].first) / gnc[d]; + } + const char fb = static_cast( + std::toupper(flc.field.empty() ? 'B' : flc.field[0])); + out::CoarseField cf = buildCoarseFieldVec(local_domain->mesh, + local_domain->fields, + bckp, + fb, + gorigin, + gnc, + gdx); + // seed/tube scale: world units per screen pixel (orthographic frame) + const real_t wpp = (cam.half_h * TWO) / static_cast(H); + real_t vlo, vhi; + auto lines = out::traceFieldLines(cf, flc, wpp, vlo, vhi); + if (flc.vmax > flc.vmin) { // explicit color range overrides auto + vlo = flc.vmin; + vhi = flc.vmax; + } + const real_t tube_world = math::max(flc.tube_px, ONE) * wpp; + const real_t eff_r = math::max(tube_world, static_cast(0.55) * ds); + std::size_t n_kept = 0; + tubes = out::buildTubeSet(lines, eff_r, flc, vlo, vhi, lo, hi, cf, n_kept); + have_tubes = true; + logger::Checkpoint("field lines: " + std::to_string(lines.size()) + + " global lines, " + std::to_string(n_kept) + + " local segments", + HERE); + } + + bool rendered_any = false; + for (const auto& scene : g_renderer.scenes()) { + // a `field = "fieldlines"` scene renders the tubes standalone (no + // volume); any other scene may overlay them inside its volume. + const bool fl_only = (scene.field == "fieldlines"); + const bool volume_on = not fl_only; + const bool show_tubes = scene.show_fieldlines and have_tubes; + if (volume_on) { + Kokkos::deep_copy(bckp, ZERO); + if (not prepareRenderScalar(params, *local_domain, scene.field, bckp)) { + continue; + } + // fill the ghost halo with neighbor active values so trilinear + // sampling is C0 across domain faces (a halo EXCHANGE, not the + // sum-into-active that SynchronizeFields performs). + CommunicateBckp(*local_domain, { 0, 1 }); + } else if (not have_tubes) { + // standalone field-line scene but tracing produced nothing/disabled + raise::Warning("render: 'fieldlines' scene but no field-line " + "geometry; skipping", + HERE); + continue; + } + const out::TubeSet& kt = show_tubes ? tubes : empty; + // a standalone tube scene colors its colorbar by |field|, not by the + // (unused) volume transfer function + out::Scene scene_cb = scene; + if (fl_only) { + scene_cb.tf.vmin = tubes.vmin; + scene_cb.tf.vmax = tubes.vmax; + scene_cb.tf.log_scale = tubes.log_scale; + scene_cb.tf.colormap = tubes.colormap; + if (scene_cb.label == "fieldlines") { + scene_cb.label = "|" + flc.field + "|"; + } + } + + // external camera -> sparse SubImage (ordered composite); dome (interior + // eye) -> sparse FragImage (depth-resolved A-buffer composite). Both are + // built only when this domain is on-screen, but the composite call is + // collective, so every rank reaches it (with an empty image otherwise). + out::SubImage sub; + out::FragImage frag; + if (on_screen) { + const std::size_t bnpix = static_cast(bw) * + static_cast(bh); + array_t image { "render_img", bnpix }; + array_t depth { "render_depth", bnpix }; + randacc_ndfield_t Fld { bckp }; + Kokkos::parallel_for( + "VolumeRayMarch", + CreateRangePolicy( + { 0, 0 }, + { static_cast(bw), static_cast(bh) }), + render::VolumeRayMarch_kernel(Fld, + 0u, + metric, + cam, + lo, + hi, + ext0, + ext1, + ext2, + W, + H, + bx0, + by0, + bw, + ds, + max_steps, + scene.tf.lut, + scene.tf.n_lut, + scene.tf.vmin, + scene.tf.vmax, + scene.tf.log_scale, + g_renderer.earlyAlpha(), + glo, + ghi, + spine_radius, + spine_rgb, + kt, + volume_on, + image, + depth)); + Kokkos::fence(); + + // device -> host, into layout-agnostic pixel-major buffers + auto image_h = Kokkos::create_mirror_view(image); + Kokkos::deep_copy(image_h, image); + if (is_dome) { + auto depth_h = Kokkos::create_mirror_view(depth); + Kokkos::deep_copy(depth_h, depth); + // leaf fragment image: one depth-tagged fragment per covered pixel + frag.x0 = bx0; + frag.y0 = by0; + frag.w = bw; + frag.h = bh; + frag.offs.assign(bnpix + 1, 0u); + for (std::size_t p = 0; p < bnpix; ++p) { + frag.offs[p + 1] = (image_h(p, 3) > ZERO) ? 1u : 0u; + } + for (std::size_t p = 0; p < bnpix; ++p) { + frag.offs[p + 1] += frag.offs[p]; + } + const std::size_t nfrag = frag.offs[bnpix]; + frag.depth.resize(nfrag); + frag.rgba.resize(nfrag * 4); + std::size_t o = 0; + for (std::size_t p = 0; p < bnpix; ++p) { + if (image_h(p, 3) > ZERO) { + frag.depth[o] = depth_h(p); + frag.rgba[o * 4 + 0] = image_h(p, 0); + frag.rgba[o * 4 + 1] = image_h(p, 1); + frag.rgba[o * 4 + 2] = image_h(p, 2); + frag.rgba[o * 4 + 3] = image_h(p, 3); + ++o; + } + } + } else { + sub.x0 = bx0; + sub.y0 = by0; + sub.w = bw; + sub.h = bh; + sub.rgba.resize(bnpix * 4); + for (std::size_t p = 0; p < bnpix; ++p) { + sub.rgba[p * 4 + 0] = image_h(p, 0); + sub.rgba[p * 4 + 1] = image_h(p, 1); + sub.rgba[p * 4 + 2] = image_h(p, 2); + sub.rgba[p * 4 + 3] = image_h(p, 3); + } + } + } + if (is_dome) { + g_renderer.compositeFragAndWrite(std::move(frag), + scene_cb, + current_step, + current_time); + } else { + g_renderer.compositeAndWrite(sub, + order_key, + scene_cb, + current_step, + current_time); + } + rendered_any = true; + } + return rendered_any; + } else if constexpr (M::Dim == Dim::_2D) { + // ---- 2D slice rasterizer (Cartesian or spherical) --------------- // + // A 2D run has no depth to integrate: each pixel is one inverse-mapped + // sample, painted opaque. Domains tile the screen disjointly, so the + // sparse sub-images composite seamlessly regardless of order. + if (not g_renderer.enabled() or + not g_renderer.shouldRender(finished_step, finished_time)) { + return false; + } + raise::ErrorIf(l_subdomain_indices().size() != 1, + "Renderer supports one subdomain per rank only", + HERE); + auto local_domain = subdomain_ptr(l_subdomain_indices()[0]); + raise::ErrorIf(local_domain->is_placeholder(), + "local_domain is a placeholder", + HERE); + logger::Checkpoint("Rendering output (2D slice)", HERE); + + // advance the moving view (region window) to this frame's time before + // reading region(); collective (same time on all ranks). + g_renderer.updateForTime(current_time); + + const int W = g_renderer.width(); + const int H = g_renderer.height(); + const bool mirror = g_renderer.mirror(); + + // fulldome fisheye ("dome master"). Cartesian slices are a flat plane, so + // the kernel warps each pixel radially (fisheye). Curvilinear slices + // (spherical / GR Kerr-Schild) are ALREADY a meridional disk, so dome + // mode there is only a framing change: mirror to a full disk (the + // `mirror` default) and fit that disk to the frame's inscribed circle + // (the pad skip below), while the kernel keeps its native (X, Z) + // meridional map. Reported back so the (metric-agnostic) compositor keeps + // the frame a clean square. All ranks take the same branch (M is fixed + // per run), so it stays seamless across tiles. + const out::DomeMap dome = g_renderer.dome(); + g_renderer.setDomeActive(dome.enabled); + + // global slice-plane world window (shared by all ranks -> seamless), + // taken from the optional render region (== full extent when uncropped). + // gext (the full extent) is kept for the field-line coarse grid below. + const auto gext = mesh().extent(); + const real_t x1lo = g_renderer.regionLo(0), x1hi = g_renderer.regionHi(0); + const real_t x2lo = g_renderer.regionLo(1), x2hi = g_renderer.regionHi(1); + real_t umin, umax, vmin, vmax; + if constexpr (M::CoordType == Coord::type::Cartesian) { + umin = x1lo; + umax = x1hi; + vmin = x2lo; + vmax = x2hi; + } else { + // meridional (X = r sin th, Z = r cos th) bounding box of the cropped + // annular wedge r in [x1lo, x1hi], theta in [x2lo, x2hi]. Sample the + // boundary (arcs + rays) so the bbox is correct for any theta range. + umin = static_cast(1e30); + umax = static_cast(-1e30); + vmin = static_cast(1e30); + vmax = static_cast(-1e30); + const int NB = 65; + auto accXZ = [&](real_t r, real_t th) { + const real_t X = r * math::sin(th), Z = r * math::cos(th); + umin = std::min(umin, X); + umax = std::max(umax, X); + vmin = std::min(vmin, Z); + vmax = std::max(vmax, Z); + if (mirror) { + umin = std::min(umin, -X); + umax = std::max(umax, -X); + } + }; + for (int k = 0; k < NB; ++k) { + const real_t t = static_cast(k) / static_cast(NB - 1); + const real_t th = x2lo + (x2hi - x2lo) * t; + const real_t rr = x1lo + (x1hi - x1lo) * t; + accXZ(x1lo, th); + accXZ(x1hi, th); + accXZ(rr, x2lo); + accXZ(rr, x2hi); + } + } + // expand the window to the image aspect (centered) so geometry is not + // stretched + { + const real_t waspect = (umax - umin) / (vmax - vmin); + const real_t iaspect = static_cast(W) / static_cast(H); + if (iaspect > waspect) { + const real_t cu = HALF * (umin + umax); + const real_t hu = HALF * (vmax - vmin) * iaspect; + umin = cu - hu; + umax = cu + hu; + } else { + const real_t cv = HALF * (vmin + vmax); + const real_t hv = HALF * (umax - umin) / iaspect; + vmin = cv - hv; + vmax = cv + hv; + } + } + // spherical slices get a background border so the round outline and its + // R/theta labels are not clipped at the frame edges (Cartesian fills the + // frame and draws its ticks in dedicated margins, so it needs none). A + // dome master skips it: the disk must reach the frame's inscribed circle + // (which the projector maps to the dome horizon), and it draws no axes. + if constexpr (M::CoordType != Coord::type::Cartesian) { + if (not dome.enabled) { + const real_t pad = static_cast(1.12); + const real_t cu = HALF * (umin + umax), hu = HALF * (umax - umin) * pad; + const real_t cv = HALF * (vmin + vmax), hv = HALF * (vmax - vmin) * pad; + umin = cu - hu; + umax = cu + hu; + vmin = cv - hv; + vmax = cv + hv; + } + } + + // hand the world window + axis names to the (host) axes overlay. Default + // names follow the coordinate family unless the toml set axis_labels. + { + const bool sph = (M::CoordType != Coord::type::Cartesian); + const std::string xl = g_renderer.axisLabelsSet() + ? g_renderer.axisLabel(0) + : (sph ? std::string("X") : std::string("x")); + const std::string yl = g_renderer.axisLabelsSet() + ? g_renderer.axisLabel(1) + : (sph ? std::string("Z") : std::string("y")); + g_renderer.setSliceFrame(umin, umax, vmin, vmax, xl, yl); + // curvilinear slices get polar axes (R radial + Theta arc); pass the + // global (r, theta) extent. + if (sph) { + g_renderer.setSlicePolar(true, x1lo, x1hi, x2lo, x2hi, mirror); + } else { + g_renderer.setSlicePolar(false, ZERO, ONE, ZERO, ONE, mirror); + } + } + + auto& bckp = local_domain->fields.bckp; + const int ext0 = static_cast(bckp.extent(0)); + const int ext1 = static_cast(bckp.extent(1)); + const auto metric = local_domain->mesh.metric; + const int n1 = static_cast(local_domain->mesh.n_active(in::x1)); + const int n2 = static_cast(local_domain->mesh.n_active(in::x2)); + + // screen-space bbox of this domain's footprint (host projection of the + // boundary; an arc for spherical, a box for Cartesian) + const auto le = local_domain->mesh.extent(); + auto toPix = [&](real_t u, real_t v, real_t& px, real_t& py) { + if (dome.enabled and M::CoordType == Coord::type::Cartesian) { + // forward fisheye projection (inverse of the kernel's radial map), + // used to bound this domain's footprint on the dome disk. Cartesian + // only -- curvilinear dome uses the linear (X, Z) map below, matching + // the kernel's native meridional projection. + const real_t cxp = HALF * static_cast(W); + const real_t cyp = HALF * static_cast(H); + const real_t Rpx = HALF * static_cast(std::min(W, H)); + const real_t dx = u - dome.cx, dy = v - dome.cy; + const real_t rw = std::sqrt(dx * dx + dy * dy); + real_t fr = (dome.R > ZERO) ? (rw / dome.R) : ZERO; + if (fr > ONE) { + fr = ONE; // clamp onto the rim (conservative for the bbox) + } + real_t theta; + if (dome.law == out::DomeMap::Gnomonic) { + theta = std::atan(fr * std::tan(dome.theta_max)); + } else if (dome.law == out::DomeMap::Stereographic) { + theta = static_cast(2) * + std::atan(fr * std::tan(HALF * dome.theta_max)); + } else if (dome.law == out::DomeMap::Orthographic) { + theta = std::asin(fr * std::sin(dome.theta_max)); + } else { + theta = fr * dome.theta_max; + } + const real_t rho = (dome.theta_max > ZERO) ? (theta / dome.theta_max) + : fr; + const real_t phi = std::atan2(dy, dx); + px = cxp + rho * Rpx * std::cos(phi) - HALF; + py = cyp - rho * Rpx * std::sin(phi) - HALF; + } else { + px = (u - umin) / (umax - umin) * static_cast(W) - HALF; + py = (vmax - v) / (vmax - vmin) * static_cast(H) - HALF; + } + }; + real_t minx = static_cast(1e30), miny = static_cast(1e30); + real_t maxx = static_cast(-1e30), maxy = static_cast(-1e30); + auto acc = [&](real_t u, real_t v) { + real_t px, py; + toPix(u, v, px, py); + minx = std::min(minx, px); + maxx = std::max(maxx, px); + miny = std::min(miny, py); + maxy = std::max(maxy, py); + }; + if constexpr (M::CoordType == Coord::type::Cartesian) { + if (dome.enabled) { + // the fisheye map is nonlinear (and a domain straddling the center + // wraps around the image center), so bound the footprint by sampling + // the whole domain-rectangle boundary, not just the 4 corners + const int NB = 65; + const real_t x0 = le[0].first, x1 = le[0].second; + const real_t y0 = le[1].first, y1 = le[1].second; + for (int k = 0; k < NB; ++k) { + const real_t t = static_cast(k) / static_cast(NB - 1); + const real_t xx = x0 + (x1 - x0) * t; + const real_t yy = y0 + (y1 - y0) * t; + acc(xx, y0); + acc(xx, y1); + acc(x0, yy); + acc(x1, yy); + } + } else { + acc(le[0].first, le[1].first); + acc(le[0].second, le[1].first); + acc(le[0].first, le[1].second); + acc(le[0].second, le[1].second); + } + } else { + const int NB = 33; + const real_t r0 = le[0].first, r1 = le[0].second; + const real_t a0 = le[1].first, a1 = le[1].second; + for (int k = 0; k < NB; ++k) { + const real_t t = static_cast(k) / static_cast(NB - 1); + const real_t rr = r0 + (r1 - r0) * t; + const real_t aa = a0 + (a1 - a0) * t; + // r-arcs at a0, a1 and theta-rays at r0, r1 + const real_t pts[4][2] = { + { r0 * math::sin(aa), r0 * math::cos(aa) }, + { r1 * math::sin(aa), r1 * math::cos(aa) }, + { rr * math::sin(a0), rr * math::cos(a0) }, + { rr * math::sin(a1), rr * math::cos(a1) } + }; + for (auto& p : pts) { + acc(p[0], p[1]); + if (mirror) { + acc(-p[0], p[1]); + } + } + } + } + const int pad = 2; + int x0 = static_cast(std::floor(minx)) - pad; + int x1 = static_cast(std::ceil(maxx)) + pad; + int y0 = static_cast(std::floor(miny)) - pad; + int y1 = static_cast(std::ceil(maxy)) + pad; + x0 = std::max(0, std::min(W, x0)); + x1 = std::max(0, std::min(W, x1)); + y0 = std::max(0, std::min(H, y0)); + y1 = std::max(0, std::min(H, y1)); + const int bx0 = x0, by0 = y0, bw = x1 - x0, bh = y1 - y0; + + // disjoint tiling -> any consistent total order composites correctly; + // a lexicographic key over the decomposition offsets is unique per rank. + const real_t fwd2d[3] = { ONE, ONE, ZERO }; + const uint64_t order_key = out::compositeOrderKey( + local_domain->offset_ndomains(), + ndomains_per_dim(), + fwd2d); + + // ---- 2D field lines (built once) ------------------------------------ + // // Cartesian: iso-contours of the flux function psi. Spherical/Kerr: + // traced meridional streamlines (nt2py style). Both come from a coarse, + // MPI- replicated copy of the in-plane field, so the geometry is global + // and seamless across the disjoint tiles. All ranks reach + // buildCoarseField2D together (collective Allreduce); M is fixed per run, + // so every rank takes the same Cartesian/spherical branch. + const auto& flc = g_renderer.fieldlines(); + out::ContourSet contours = out::emptyContourSet(); + out::ContourSet emptyc = out::emptyContourSet(); + out::TubeSet lines2d = out::emptyTubeSet(); + out::TubeSet emptyl = out::emptyTubeSet(); + bool have_fl = false; + real_t fl_vmin = ZERO, fl_vmax = ONE; + std::string fl_colormap = flc.colormap; + if (flc.enable) { + const int gN[2] = { static_cast(mesh().n_active(in::x1)), + static_cast(mesh().n_active(in::x2)) }; + int gnc[2]; + real_t gorigin[2], gdx[2]; + for (int d = 0; d < 2; ++d) { + gnc[d] = std::max(1, (gN[d] + flc.bin - 1) / flc.bin); + gorigin[d] = gext[d].first; + gdx[d] = (gext[d].second - gext[d].first) / gnc[d]; + } + const char fb = static_cast( + std::toupper(flc.field.empty() ? 'B' : flc.field[0])); + // coarse, replicated in-plane field: (Bx,By) for Cartesian, (Br,Bth) for + // spherical (FieldsToPhys writes the physical components in axis order) + out::CoarseField2D cf = buildCoarseField2D(local_domain->mesh, + local_domain->fields, + bckp, + fb, + gorigin, + gnc, + gdx); + const real_t wpp = (umax - umin) / static_cast(W); + if constexpr (M::CoordType == Coord::type::Cartesian) { + std::vector psi; + real_t pmin, pmax, bmin, bmax; + out::computeFlux2D(cf, psi, pmin, pmax, bmin, bmax); + const real_t line_half = HALF * math::max(flc.tube_px, ONE); + contours = + out::buildContourSet(cf, psi, pmin, pmax, bmin, bmax, flc, line_half, wpp); + fl_vmin = contours.vmin; + fl_vmax = contours.vmax; + fl_colormap = contours.colormap; + logger::Checkpoint("field lines (2D): " + std::to_string(flc.levels) + + " psi contours on a " + std::to_string(gnc[0]) + + "x" + std::to_string(gnc[1]) + " grid", + HERE); + } else { + // traced meridional streamlines through the coarse (r, theta) field + real_t vlo, vhi; + auto poly = out::traceFieldLinesMeridional(cf, flc, wpp, mirror, vlo, vhi); + if (flc.vmax > flc.vmin) { // explicit |B| range overrides auto + vlo = flc.vmin; + vhi = flc.vmax; + } + const real_t eff_r = math::max(flc.tube_px, ONE) * wpp; + // bucket grid for buildTubeSet: cell ~ a coarse dr (a length), AABB + // spans the (mirrored) meridional disk, z is a single thin slab at 0 + out::CoarseField bucket_cf; + bucket_cf.dx[0] = cf.dx[0]; + bucket_cf.dx[1] = cf.dx[0]; + bucket_cf.dx[2] = cf.dx[0]; + const real_t rmax = gext[0].second; + const real_t lo[3] = { mirror ? -rmax : ZERO, -rmax, -cf.dx[0] }; + const real_t hi[3] = { rmax, rmax, cf.dx[0] }; + std::size_t n_kept = 0; + lines2d = out::buildTubeSet(poly, eff_r, flc, vlo, vhi, lo, hi, bucket_cf, n_kept); + fl_vmin = lines2d.vmin; + fl_vmax = lines2d.vmax; + fl_colormap = lines2d.colormap; + logger::Checkpoint( + "field lines (2D meridional): " + std::to_string(poly.size()) + + " lines, " + std::to_string(n_kept) + " segments", + HERE); + } + have_fl = true; + } + + bool rendered_any = false; + for (const auto& scene : g_renderer.scenes()) { + // standalone `field = "fieldlines"` -> lines only (no heatmap fill); any + // other scene with `fieldlines = true` overlays them on its heatmap. + const bool fl_only = (scene.field == "fieldlines"); + const bool heatmap_on = not fl_only; + const bool show_lines = scene.show_fieldlines and have_fl; + if (heatmap_on) { + Kokkos::deep_copy(bckp, ZERO); + if (not prepareRenderScalar(params, *local_domain, scene.field, bckp)) { + continue; + } + CommunicateBckp(*local_domain, { 0, 1 }); + } else if (not have_fl) { + raise::Warning("render: 'fieldlines' scene needs a 2D run with " + "[render.fieldlines]; skipping", + HERE); + continue; + } + const out::ContourSet& kc = show_lines ? contours : emptyc; + const out::TubeSet& kt = show_lines ? lines2d : emptyl; + // a standalone field-line scene colors its colorbar by |B| + out::Scene scene_cb = scene; + if (fl_only) { + scene_cb.tf.vmin = fl_vmin; + scene_cb.tf.vmax = fl_vmax; + scene_cb.tf.log_scale = false; + scene_cb.tf.colormap = fl_colormap; + if (scene_cb.label == "fieldlines") { + scene_cb.label = "|" + flc.field + "|"; + } + } + + out::SubImage sub; + if (bw > 0 and bh > 0) { + sub.x0 = bx0; + sub.y0 = by0; + sub.w = bw; + sub.h = bh; + const std::size_t bnpix = static_cast(bw) * + static_cast(bh); + array_t image { "render_img", bnpix }; + randacc_ndfield_t Fld { bckp }; + Kokkos::parallel_for( + "Slice2DRaster", + CreateRangePolicy( + { 0, 0 }, + { static_cast(bw), static_cast(bh) }), + render::SliceRaster_kernel(Fld, + 0u, + metric, + umin, + umax, + vmin, + vmax, + W, + H, + bx0, + by0, + bw, + mirror, + dome, + x1lo, + x1hi, + x2lo, + x2hi, + g_renderer.hasRegion(), + n1, + n2, + ext0, + ext1, + scene.tf.lut_opaque, + scene.tf.n_lut, + scene.tf.vmin, + scene.tf.vmax, + scene.tf.log_scale, + kc, + kt, + heatmap_on, + image)); + Kokkos::fence(); + + auto image_h = Kokkos::create_mirror_view(image); + Kokkos::deep_copy(image_h, image); + sub.rgba.resize(bnpix * 4); + for (std::size_t p = 0; p < bnpix; ++p) { + sub.rgba[p * 4 + 0] = image_h(p, 0); + sub.rgba[p * 4 + 1] = image_h(p, 1); + sub.rgba[p * 4 + 2] = image_h(p, 2); + sub.rgba[p * 4 + 3] = image_h(p, 3); + } + } + g_renderer.compositeAndWrite(sub, order_key, scene_cb, current_step, current_time); + rendered_any = true; + } + return rendered_any; + } else { + (void)params; + (void)current_step; + (void)finished_step; + (void)finished_time; + return false; + } + } + + // NOLINTBEGIN(bugprone-macro-parentheses) +#define METADOMAIN_RENDER(S, M, D) \ + template auto Metadomain>::Render(const SimulationParams&, \ + timestep_t, \ + timestep_t, \ + simtime_t, \ + simtime_t) -> bool; \ + template auto Metadomain>::prepareRenderScalar( \ + const SimulationParams&, \ + Domain>&, \ + const std::string&, \ + ndfield_t::Dim, 6>&) const -> bool; + + NTT_FOREACH_SPECIALIZATION(METADOMAIN_RENDER) + +#undef METADOMAIN_RENDER + // NOLINTEND(bugprone-macro-parentheses) + +} // namespace ntt diff --git a/src/framework/domain/metadomain.h b/src/framework/domain/metadomain.h index bf7a562d4..dd0eb3380 100644 --- a/src/framework/domain/metadomain.h +++ b/src/framework/domain/metadomain.h @@ -39,6 +39,7 @@ #include "framework/domain/domain.h" #include "framework/domain/mesh.h" #include "framework/parameters/parameters.h" +#include "output/render/renderer.h" #include "output/stats.h" #if defined(MPI_ENABLED) @@ -104,6 +105,9 @@ namespace ntt { void SynchronizeFields(Domain&, CommTags, const cell_range_t& = { 0, 0 }) const; + // Halo-fill of the bckp buffer (neighbor active cells -> local ghosts), + // used by the in-situ renderer for seamless trilinear sampling. + void CommunicateBckp(Domain&, const cell_range_t&) const; #if defined(MPI_ENABLED) && defined(OUTPUT_ENABLED) void CommunicateVectorPotential(unsigned short); #endif @@ -189,6 +193,20 @@ namespace ntt { const std::vector>&); #endif + /* in-situ renderer (3D volume ray-march & 2D slice) */ + void InitRenderer(const SimulationParams&); + auto Render(const SimulationParams&, timestep_t, timestep_t, simtime_t, simtime_t) + -> bool; + // Prepare the scalar named by a scene's `field` into bckp(:, 0) (active + // cells synced; ghosts not yet halo-filled). Shared by the 2D and 3D render + // paths so the field grammar (moments, T/V components, |E,B,J|, species + // suffix) has a single source of truth. Returns false (and warns) for an + // unknown field or invalid species so the caller can skip the scene. + auto prepareRenderScalar(const SimulationParams&, + Domain&, + const std::string& field_name, + ndfield_t&) const -> bool; + using custom_stats_output_t = std::function< real_t(const std::string&, timestep_t, simtime_t, const Domain&)>; void InitStatsWriter(const SimulationParams&, bool); @@ -319,6 +337,7 @@ namespace ntt { out::Writer g_writer; checkpoint::Writer g_checkpoint_writer; #endif + out::Renderer g_renderer; #if defined(MPI_ENABLED) int g_mpi_rank { -1 }, g_mpi_size { -1 }; diff --git a/src/framework/parameters/algorithms.cpp b/src/framework/parameters/algorithms.cpp index 797cfb4f7..db34af852 100644 --- a/src/framework/parameters/algorithms.cpp +++ b/src/framework/parameters/algorithms.cpp @@ -3,6 +3,7 @@ #include "defaults.h" #include "global.h" +#include "utils/log.h" #include "utils/numeric.h" #include "framework/parameters/parameters.h" @@ -32,12 +33,28 @@ namespace ntt { defaults::current_filters); deposit_enable = toml::find_or(toml_data, "algorithms", "deposit", "enable", true); - deposit_order = static_cast(SHAPE_ORDER); - deposit_team_policy_team_size = toml::find_or(toml_data, - "algorithms", - "deposit", - "team_policy_team_size", - defaults::team_policy_team_size); + deposit_order = static_cast(SHAPE_ORDER); + if ( + toml_data.contains("algorithms") and + toml_data.at("algorithms").contains("deposit") and + toml_data.at("algorithms").at("deposit").contains("team_policy_team_size") and + not toml_data.at("algorithms").at("deposit").contains("tiled_deposit_team_size")) { + deposit_tiled_team_size = toml::find( + toml_data, + "algorithms", + "deposit", + "team_policy_team_size"); + raise::Warning("`algorithms.deposit.team_policy_team_size` is " + "deprecated and will be removed in 1.6+ versions, use " + "`algorithms.deposit.tiled_deposit_team_size` instead", + HERE); + } else { + deposit_tiled_team_size = toml::find_or(toml_data, + "algorithms", + "deposit", + "tiled_deposit_team_size", + defaults::tiled_deposit_team_size); + } fieldsolver_enable = toml::find_or(toml_data, "algorithms", @@ -145,8 +162,8 @@ namespace ntt { params->set("algorithms.deposit.enable", deposit_enable.value()); params->set("algorithms.deposit.order", deposit_order.value()); - params->set("algorithms.deposit.team_policy_team_size", - deposit_team_policy_team_size.value()); + params->set("algorithms.deposit.tiled_deposit_team_size", + deposit_tiled_team_size.value()); params->set("algorithms.fieldsolver.enable", fieldsolver_enable.value()); for (const auto& [key, value] : fieldsolver_stencil_coeffs.value()) { diff --git a/src/framework/parameters/algorithms.h b/src/framework/parameters/algorithms.h index c46f480fe..f253448c5 100644 --- a/src/framework/parameters/algorithms.h +++ b/src/framework/parameters/algorithms.h @@ -34,7 +34,7 @@ namespace ntt { std::optional deposit_enable; std::optional deposit_order; - std::optional deposit_team_policy_team_size; + std::optional deposit_tiled_team_size; std::optional fieldsolver_enable; std::optional> fieldsolver_stencil_coeffs; diff --git a/src/framework/parameters/output.cpp b/src/framework/parameters/output.cpp index d2a6a9c44..03087f08a 100644 --- a/src/framework/parameters/output.cpp +++ b/src/framework/parameters/output.cpp @@ -27,10 +27,6 @@ namespace ntt { "output", "interval_time", -1.0); - raise::ErrorIf( - not toml::find_or(toml_data, "output", "separate_files", true), - "separate_files=false is deprecated", - HERE); categories.emplace(); for (const auto& category : { "fields", "particles", "spectra", "stats" }) { diff --git a/src/framework/parameters/parameters.cpp b/src/framework/parameters/parameters.cpp index f3d8d507e..080944b4c 100644 --- a/src/framework/parameters/parameters.cpp +++ b/src/framework/parameters/parameters.cpp @@ -15,6 +15,7 @@ #include "framework/parameters/grid.h" #include "framework/parameters/output.h" #include "framework/parameters/particles.h" +#include "framework/parameters/render.h" #include @@ -165,6 +166,11 @@ namespace ntt { output_params.read(dim, get("particles.nspec"), toml_data); output_params.setParams(this); + /* [render] ------------------------------------------------------------- */ + params::Render render_params; + render_params.read(toml_data, this); + render_params.setParams(this); + /* [checkpoint] --------------------------------------------------------- */ set("checkpoint.interval", toml::find_or(toml_data, diff --git a/src/framework/parameters/render.cpp b/src/framework/parameters/render.cpp new file mode 100644 index 000000000..401c8f44d --- /dev/null +++ b/src/framework/parameters/render.cpp @@ -0,0 +1,434 @@ +#include "framework/parameters/render.h" + +#include "global.h" + +#include "utils/error.h" +#include "utils/numeric.h" + +#include "framework/parameters/parameters.h" + +#include + +#include +#include +#include +#include + +namespace ntt { + namespace params { + + namespace { + // `render..` if present; otherwise nullopt (used for keys whose + // defaults depend on the domain geometry and are resolved by the renderer) + template + auto findOpt(const toml::value& toml_data, + const std::string& table, + const std::string& key) -> std::optional { + if (toml_data.contains("render") and + toml_data.at("render").contains(table) and + toml_data.at("render").at(table).contains(key)) { + return toml::find(toml_data, "render", table, key); + } + return std::nullopt; + } + } // namespace + + void Render::read(const toml::value& toml_data, + const SimulationParams* const params) { + enable = toml::find_or(toml_data, "render", "enable", false); + if (not enable) { + return; + } + + /* cadence -------------------------------------------------------------- */ + interval = toml::find_or(toml_data, "render", "interval", 0u); + interval_time = toml::find_or(toml_data, + "render", + "interval_time", + -1.0); + if ((interval.value() == 0) and (interval_time.value() == -1.0)) { + interval = params->template get("output.interval"); + interval_time = params->template get("output.interval_time"); + } + + /* image ---------------------------------------------------------------- */ + width = toml::find_or(toml_data, "render", "width", 1024); + height = toml::find_or(toml_data, "render", "height", 1024); + // `resolution` is a convenience that forces a square frame (width == + // height), the natural shape for a dome master. + const auto resolution = toml::find_or(toml_data, "render", "resolution", 0); + if (resolution > 0) { + width = resolution; + height = resolution; + } + raise::ErrorIf(width.value() <= 0 or height.value() <= 0, + "render.width and render.height must be > 0", + HERE); + n_lut = toml::find_or(toml_data, "render", "n_lut", 256); + background = toml::find_or>( + toml_data, + "render", + "background", + std::vector { ZERO, ZERO, ZERO }); + if (background->size() != 3) { + raise::Warning("render.background must have 3 entries [r, g, b]; " + "using black", + HERE); + background = std::vector { ZERO, ZERO, ZERO }; + } + colorbar = toml::find_or(toml_data, "render", "colorbar", true); + colorbar_outside = toml::find_or(toml_data, "render", "colorbar_outside", true); + mirror = toml::find_or(toml_data, "render", "mirror", true); + time_label = toml::find_or(toml_data, "render", "time_label", false); + axes = toml::find_or(toml_data, "render", "axes", false); + // empty => unset (the 2D slice then picks per-metric default names) + axis_labels = toml::find_or>( + toml_data, + "render", + "axis_labels", + std::vector {}); + axis_ticks = toml::find_or(toml_data, "render", "axis_ticks", 5); + spine_width = toml::find_or(toml_data, + "render", + "spine_width", + static_cast(2)); + + /* [render.extent] ------------------------------------------------------ */ + extent.emplace(); + for (const auto& key : { "x1", "x2", "x3" }) { + auto lim = toml::find_or>(toml_data, + "render", + "extent", + key, + std::vector {}); + if (not lim.empty() and (lim.size() != 2 or lim[1] <= lim[0])) { + raise::Warning("render.extent." + std::string(key) + + " must be [lo, hi] with hi > lo; ignoring", + HERE); + lim.clear(); + } + extent->push_back(lim); + } + + /* [render.volume] ------------------------------------------------------ */ + volume_samples = toml::find_or(toml_data, "render", "volume", "samples", 400); + volume_step_size = toml::find_or(toml_data, + "render", + "volume", + "step_size", + ZERO); + volume_early_term_alpha = toml::find_or(toml_data, + "render", + "volume", + "early_term_alpha", + static_cast(0.99)); + + /* [render.moving_view] ------------------------------------------------- */ + moving_view_velocity = toml::find_or>( + toml_data, + "render", + "moving_view", + "velocity", + std::vector {}); + moving_view_start_time = toml::find_or(toml_data, + "render", + "moving_view", + "start_time", + 0.0); + + /* [render.camera] ------------------------------------------------------ */ + camera_mode = toml::find_or(toml_data, + "render", + "camera", + "mode", + "orthographic"); + if (camera_mode.value() != "orthographic" and + camera_mode.value() != "perspective" and camera_mode.value() != "dome") { + raise::Warning( + "render.camera.mode '" + camera_mode.value() + + "' unknown (want orthographic/perspective/dome); using " + "orthographic projection", + HERE); + camera_mode = "orthographic"; + } + camera_position = toml::find_or>(toml_data, + "render", + "camera", + "position", + std::vector {}); + camera_look_at = toml::find_or>(toml_data, + "render", + "camera", + "look_at", + std::vector {}); + camera_up = toml::find_or>(toml_data, + "render", + "camera", + "up", + std::vector {}); + camera_fov = toml::find_or(toml_data, + "render", + "camera", + "fov", + static_cast(35)); + camera_dome_fov = toml::find_or(toml_data, + "render", + "camera", + "dome_fov", + static_cast(180)); + camera_dome_radius = findOpt(toml_data, "camera", "dome_radius"); + camera_ortho_height = findOpt(toml_data, "camera", "ortho_height"); + + /* [render.dome] -------------------------------------------------------- */ + dome_enable = toml::find_or(toml_data, "render", "dome", "enable", false); + dome_fov = toml::find_or(toml_data, + "render", + "dome", + "fov", + static_cast(180)); + dome_radius = findOpt(toml_data, "dome", "radius"); + dome_center = toml::find_or>(toml_data, + "render", + "dome", + "center", + std::vector {}); + if (not dome_center->empty() and dome_center->size() != 2) { + raise::Warning("render.dome.center must have 2 entries [x, y]; using " + "the domain center", + HERE); + dome_center->clear(); + } + dome_projection = toml::find_or(toml_data, + "render", + "dome", + "projection", + "equidistant"); + if (dome_projection.value() != "equidistant" and + dome_projection.value() != "gnomonic" and + dome_projection.value() != "stereographic" and + dome_projection.value() != "orthographic") { + raise::Warning("render.dome.projection '" + dome_projection.value() + + "' unknown; using 'equidistant'", + HERE); + dome_projection = "equidistant"; + } + + /* [[render.scene]] ----------------------------------------------------- */ + scenes.emplace(); + bool any_fieldlines = false; + const auto scenes_arr = toml::find_or(toml_data, + "render", + "scene", + toml::array {}); + for (const auto& sc : scenes_arr) { + RenderScene scene; + scene.field = toml::find_or(sc, "field", ""); + if (scene.field.empty()) { + raise::Warning("render.scene with no field; skipping", HERE); + continue; + } + scene.prefix = toml::find_or(sc, "prefix", scene.field + "_"); + scene.label = toml::find_or(sc, "label", scene.field); + scene.min = toml::find_or(sc, "min", ZERO); + scene.max = toml::find_or(sc, "max", ONE); + scene.log = toml::find_or(sc, "log", false); + scene.colormap = toml::find_or(sc, "colormap", "viridis"); + // alpha control points: array of [position, alpha] pairs + scene.alpha = toml::find_or>>( + sc, + "alpha", + std::vector> {}); + scene.colorbar_ticks = toml::find_or>( + sc, + "colorbar_ticks", + std::vector {}); + // overlay the field-line tubes inside this scene's volume; a dedicated + // `field = "fieldlines"` scene renders the tubes standalone (no volume). + scene.fieldlines = toml::find_or(sc, "fieldlines", false) or + (scene.field == "fieldlines"); + any_fieldlines = any_fieldlines or scene.fieldlines; + scenes->push_back(scene); + } + if (scenes->empty()) { + raise::Warning("render enabled but no valid scenes; disabling", HERE); + enable = false; + return; + } + + /* [render.fieldlines] -------------------------------------------------- */ + // the field lines are built whenever the section asks for them OR any + // scene requests the overlay (so a bare `field = "fieldlines"` scene + // works without a separate enable flag). + fieldlines_enable = toml::find_or(toml_data, + "render", + "fieldlines", + "enable", + false) or + any_fieldlines; + fieldlines_field = toml::find_or(toml_data, + "render", + "fieldlines", + "field", + "B"); + fieldlines_bin = toml::find_or(toml_data, "render", "fieldlines", "bin", 4); + fieldlines_bin = (fieldlines_bin.value() < 1) + ? 1 + : ((fieldlines_bin.value() > 16) + ? 16 + : fieldlines_bin.value()); + fieldlines_seed_px = toml::find_or(toml_data, + "render", + "fieldlines", + "seed_px", + static_cast(8)); + fieldlines_seed_max = toml::find_or(toml_data, + "render", + "fieldlines", + "seed_max", + 4096); + fieldlines_levels = toml::find_or(toml_data, + "render", + "fieldlines", + "levels", + 16); + fieldlines_tube_px = toml::find_or(toml_data, + "render", + "fieldlines", + "tube_px", + static_cast(2)); + fieldlines_colormap = toml::find_or(toml_data, + "render", + "fieldlines", + "colormap", + "inferno"); + // optional monochrome color [r,g,b]; overrides the colormap when set + fieldlines_color = toml::find_or>(toml_data, + "render", + "fieldlines", + "color", + std::vector {}); + if (not fieldlines_color->empty() and fieldlines_color->size() != 3) { + raise::Warning("render.fieldlines.color must have 3 entries [r, g, b]; " + "ignoring", + HERE); + fieldlines_color->clear(); + } + fieldlines_log = toml::find_or(toml_data, "render", "fieldlines", "log", false); + fieldlines_min = toml::find_or(toml_data, + "render", + "fieldlines", + "min", + ZERO); + fieldlines_max = toml::find_or(toml_data, + "render", + "fieldlines", + "max", + ZERO); + fieldlines_step_frac = toml::find_or(toml_data, + "render", + "fieldlines", + "step_frac", + static_cast(0.5)); + fieldlines_max_steps = toml::find_or(toml_data, + "render", + "fieldlines", + "max_steps", + 4000); + fieldlines_max_length = toml::find_or(toml_data, + "render", + "fieldlines", + "max_length", + static_cast(3)); + } + + void Render::setParams(SimulationParams* params) const { + params->set("render.enable", enable); + if (not enable) { + return; + } + params->set("render.interval", interval.value()); + params->set("render.interval_time", interval_time.value()); + + params->set("render.width", width.value()); + params->set("render.height", height.value()); + params->set("render.n_lut", n_lut.value()); + params->set("render.background", background.value()); + params->set("render.colorbar", colorbar.value()); + params->set("render.colorbar_outside", colorbar_outside.value()); + params->set("render.mirror", mirror.value()); + params->set("render.time_label", time_label.value()); + params->set("render.axes", axes.value()); + params->set("render.axis_labels", axis_labels.value()); + params->set("render.axis_ticks", axis_ticks.value()); + params->set("render.spine_width", spine_width.value()); + + params->set("render.extent.x1", extent.value()[0]); + params->set("render.extent.x2", extent.value()[1]); + params->set("render.extent.x3", extent.value()[2]); + + params->set("render.volume.samples", volume_samples.value()); + params->set("render.volume.step_size", volume_step_size.value()); + params->set("render.volume.early_term_alpha", + volume_early_term_alpha.value()); + + params->set("render.moving_view.velocity", moving_view_velocity.value()); + params->set("render.moving_view.start_time", moving_view_start_time.value()); + + params->set("render.camera.mode", camera_mode.value()); + params->set("render.camera.position", camera_position.value()); + params->set("render.camera.look_at", camera_look_at.value()); + params->set("render.camera.up", camera_up.value()); + params->set("render.camera.fov", camera_fov.value()); + params->set("render.camera.dome_fov", camera_dome_fov.value()); + if (camera_dome_radius.has_value()) { + params->set("render.camera.dome_radius", camera_dome_radius.value()); + } + if (camera_ortho_height.has_value()) { + params->set("render.camera.ortho_height", camera_ortho_height.value()); + } + + params->set("render.dome.enable", dome_enable.value()); + params->set("render.dome.fov", dome_fov.value()); + if (dome_radius.has_value()) { + params->set("render.dome.radius", dome_radius.value()); + } + params->set("render.dome.center", dome_center.value()); + params->set("render.dome.projection", dome_projection.value()); + + params->set("render.fieldlines.enable", fieldlines_enable.value()); + params->set("render.fieldlines.field", fieldlines_field.value()); + params->set("render.fieldlines.bin", fieldlines_bin.value()); + params->set("render.fieldlines.seed_px", fieldlines_seed_px.value()); + params->set("render.fieldlines.seed_max", fieldlines_seed_max.value()); + params->set("render.fieldlines.levels", fieldlines_levels.value()); + params->set("render.fieldlines.tube_px", fieldlines_tube_px.value()); + params->set("render.fieldlines.colormap", fieldlines_colormap.value()); + params->set("render.fieldlines.color", fieldlines_color.value()); + params->set("render.fieldlines.log", fieldlines_log.value()); + params->set("render.fieldlines.min", fieldlines_min.value()); + params->set("render.fieldlines.max", fieldlines_max.value()); + params->set("render.fieldlines.step_frac", fieldlines_step_frac.value()); + params->set("render.fieldlines.max_steps", fieldlines_max_steps.value()); + params->set("render.fieldlines.max_length", fieldlines_max_length.value()); + + // scenes are flattened into indexed keys (`render.scene..`) so + // that every entry stays a plain (serializable) parameter type + params->set("render.nscenes", scenes->size()); + for (std::size_t i = 0; i < scenes->size(); ++i) { + const auto& sc = scenes.value()[i]; + const auto pfx = "render.scene." + std::to_string(i) + "."; + params->set(pfx + "field", sc.field); + params->set(pfx + "prefix", sc.prefix); + params->set(pfx + "label", sc.label); + params->set(pfx + "min", sc.min); + params->set(pfx + "max", sc.max); + params->set(pfx + "log", sc.log); + params->set(pfx + "colormap", sc.colormap); + params->set(pfx + "alpha", sc.alpha); + params->set(pfx + "colorbar_ticks", sc.colorbar_ticks); + params->set(pfx + "fieldlines", sc.fieldlines); + } + } + + } // namespace params +} // namespace ntt diff --git a/src/framework/parameters/render.h b/src/framework/parameters/render.h new file mode 100644 index 000000000..bde7e93af --- /dev/null +++ b/src/framework/parameters/render.h @@ -0,0 +1,120 @@ +/** + * @file framework/parameters/render.h + * @brief Auxiliary functions for reading in on-the-fly render parameters + * @implements + * - ntt::params::RenderScene + * - ntt::params::Render + * @cpp: + * - render.cpp + * @namespaces: + * - ntt::params:: + * @note Only the raw (geometry-independent) configuration is resolved here; + * defaults that depend on the domain extent (camera framing, dome radius, + * clamping of the render region to the box) are resolved by out::Renderer. + * Those keys are only set in SimulationParams when given in the input. + */ +#ifndef FRAMEWORK_PARAMETERS_RENDER_H +#define FRAMEWORK_PARAMETERS_RENDER_H + +#include "global.h" + +#include "framework/parameters/parameters.h" + +#include + +#include +#include +#include + +namespace ntt { + namespace params { + + struct RenderScene { + std::string field; + std::string prefix; + std::string label; + real_t min; + real_t max; + bool log; + std::string colormap; + std::vector> alpha; + std::vector colorbar_ticks; + bool fieldlines; + }; + + struct Render { + bool enable { false }; + + std::optional interval; + std::optional interval_time; + + std::optional width; + std::optional height; + std::optional n_lut; + std::optional> background; + std::optional colorbar; + std::optional colorbar_outside; + std::optional mirror; + std::optional time_label; + std::optional axes; + std::optional> axis_labels; + std::optional axis_ticks; + std::optional spine_width; + + // [render.extent]: x{1,2,3} -> [lo, hi] or empty (full extent) + std::optional>> extent; + + // [render.volume] + std::optional volume_samples; + std::optional volume_step_size; + std::optional volume_early_term_alpha; + + // [render.moving_view] + std::optional> moving_view_velocity; + std::optional moving_view_start_time; + + // [render.camera] + std::optional camera_mode; + std::optional> camera_position; + std::optional> camera_look_at; + std::optional> camera_up; + std::optional camera_fov; + std::optional camera_dome_fov; + std::optional camera_dome_radius; + std::optional camera_ortho_height; + + // [render.dome] + std::optional dome_enable; + std::optional dome_fov; + std::optional dome_radius; + std::optional> dome_center; + std::optional dome_projection; + + // [render.fieldlines] + std::optional fieldlines_enable; + std::optional fieldlines_field; + std::optional fieldlines_bin; + std::optional fieldlines_seed_px; + std::optional fieldlines_seed_max; + std::optional fieldlines_levels; + std::optional fieldlines_tube_px; + std::optional fieldlines_colormap; + std::optional> fieldlines_color; + std::optional fieldlines_log; + std::optional fieldlines_min; + std::optional fieldlines_max; + std::optional fieldlines_step_frac; + std::optional fieldlines_max_steps; + std::optional fieldlines_max_length; + + // [[render.scene]] + std::optional> scenes; + + void read(const toml::value&, const SimulationParams* const); + void setParams(SimulationParams*) const; + }; + + } // namespace params +} // namespace ntt + +#endif // FRAMEWORK_PARAMETERS_RENDER_H diff --git a/src/global/arch/kokkos_aliases.h b/src/global/arch/kokkos_aliases.h index b0ed48773..5ca287032 100644 --- a/src/global/arch/kokkos_aliases.h +++ b/src/global/arch/kokkos_aliases.h @@ -364,7 +364,7 @@ template auto CreateRangePolicyOnHost(const tuple_t&, const tuple_t&) -> range_h_t; -// --------------------------- team_policy types ---------------------------- // +// ------------------------- tiled_deposit types --------------------------- // // Particle permutation index: maps a sorted-position p in [0, npart) to a // pre-sort particle index. Produced by SortSpatially, consumed by tiled // pusher and deposit kernels to walk particles tile-by-tile without diff --git a/src/global/defaults.h b/src/global/defaults.h index 82b4364cc..101e54659 100644 --- a/src/global/defaults.h +++ b/src/global/defaults.h @@ -22,7 +22,7 @@ namespace ntt::defaults { const unsigned short current_filters = 0; - const std::size_t team_policy_team_size = 0; + const std::size_t tiled_deposit_team_size = 0; const std::string em_pusher = "Boris"; const std::string ph_pusher = "Photon"; diff --git a/src/global/global.h b/src/global/global.h index c3fbc5061..dafd8663c 100644 --- a/src/global/global.h +++ b/src/global/global.h @@ -254,6 +254,7 @@ namespace Timer { PrintParticleSort = 1 << 4, PrintCheckpoint = 1 << 5, PrintNormed = 1 << 6, + PrintRender = 1 << 7, Default = PrintNormed | PrintTotal | PrintTitle | AutoConvert, }; } // namespace Timer diff --git a/src/global/utils/diag.cpp b/src/global/utils/diag.cpp index 298225eda..2b124c4b3 100644 --- a/src/global/utils/diag.cpp +++ b/src/global/utils/diag.cpp @@ -90,6 +90,7 @@ namespace diag { const std::vector& species_maxnpart, bool print_prtl_clear, bool print_output, + bool print_render, bool print_checkpoint, bool print_colors) { DiagFlags diag_flags = Diag::Default; @@ -106,6 +107,9 @@ namespace diag { if (print_output) { timer_flags |= Timer::PrintOutput; } + if (print_render) { + timer_flags |= Timer::PrintRender; + } if (print_checkpoint) { timer_flags |= Timer::PrintCheckpoint; } diff --git a/src/global/utils/diag.h b/src/global/utils/diag.h index 18669c1f1..61186aa1c 100644 --- a/src/global/utils/diag.h +++ b/src/global/utils/diag.h @@ -36,6 +36,7 @@ namespace diag { * @param maxnpart (per each species) * @param particlesort (if true, dead particles were removed) * @param output (if true, output was written) + * @param render (if true, a volume render was produced) * @param checkpoint (if true, checkpoint was written) * @param colorful_print (if true, print with colors) */ @@ -52,6 +53,7 @@ namespace diag { bool, bool, bool, + bool, bool); } // namespace diag diff --git a/src/global/utils/reporter.cpp b/src/global/utils/reporter.cpp index d889ad560..21d33ee7d 100644 --- a/src/global/utils/reporter.cpp +++ b/src/global/utils/reporter.cpp @@ -251,8 +251,8 @@ namespace reporter { AddParam(report, 4, "GPU_AWARE_MPI", "%s", "OFF"); #endif -#if defined(TEAM_POLICY) - AddParam(report, 4, "TEAM_POLICY", "%s", "ON"); +#if defined(TILED_DEPOSIT) + AddParam(report, 4, "TILED_DEPOSIT", "%s", "ON"); #if (defined(SYCL_ENABLED) && defined(ONEDPL_ENABLED)) || \ (defined(CUDA_ENABLED) && defined(THRUST_ENABLED)) || \ (defined(HIP_ENABLED) && defined(ROCTHRUST_ENABLED)) @@ -261,7 +261,7 @@ namespace reporter { AddParam(report, 4, "VENDOR_SORT", "%s", "OFF (BinSort)"); #endif #else - AddParam(report, 4, "TEAM_POLICY", "%s", "OFF"); + AddParam(report, 4, "TILED_DEPOSIT", "%s", "OFF"); #endif report += "\n"; return report; diff --git a/src/global/utils/sort_dispatch.h b/src/global/utils/sort_dispatch.h index e445a45ad..70674d840 100644 --- a/src/global/utils/sort_dispatch.h +++ b/src/global/utils/sort_dispatch.h @@ -1,12 +1,12 @@ /** * @file utils/sort_dispatch.h - * @brief Backend-dispatched sort_by_key for team_policy SortSpatially. + * @brief Backend-dispatched sort_by_key for tiled_deposit SortSpatially. * @implements * - sort_helpers::sort_by_key_dispatch -> void (BinSort, OneDPL, Thrust, StdSort) * @namespaces: * - ntt::sort_helpers:: * @macros: - * - TEAM_POLICY + * - TILED_DEPOSIT * - SYCL_ENABLED, ONEDPL_ENABLED (oneDPL overload) * - CUDA_ENABLED, THRUST_ENABLED (Thrust overload) * @@ -26,8 +26,8 @@ #ifndef GLOBAL_UTILS_SORT_DISPATCH_H #define GLOBAL_UTILS_SORT_DISPATCH_H -#if !defined(TEAM_POLICY) - #error "sort_dispatch.h is only meaningful when TEAM_POLICY is defined" +#if !defined(TILED_DEPOSIT) + #error "sort_dispatch.h is only meaningful when TILED_DEPOSIT is defined" #endif #include "global.h" diff --git a/src/global/utils/timer.cpp b/src/global/utils/timer.cpp index 6f2045be6..85647d899 100644 --- a/src/global/utils/timer.cpp +++ b/src/global/utils/timer.cpp @@ -136,7 +136,10 @@ namespace timer { auto Timers::printAll(TimerFlags flags, npart_t npart, ncells_t ncells) const -> std::string { - const std::vector extras { "ParticleSort", "Output", "Checkpoint" }; + const std::vector extras { "ParticleSort", + "Output", + "Render", + "Checkpoint" }; const auto stats = gather(extras, npart, ncells); if (stats.empty()) { return ""; @@ -262,6 +265,7 @@ namespace timer { // print extra timers for output/checkpoint/particleSort const std::vector extras_f { Timer::PrintParticleSort, Timer::PrintOutput, + Timer::PrintRender, Timer::PrintCheckpoint }; for (auto i { 0u }; i < extras.size(); ++i) { const auto& name = extras[i]; diff --git a/src/kernels/deposition/currents/tiled.hpp b/src/kernels/deposition/currents/tiled.hpp index 6e211a89f..1a1bd94ee 100644 --- a/src/kernels/deposition/currents/tiled.hpp +++ b/src/kernels/deposition/currents/tiled.hpp @@ -3,8 +3,8 @@ * @brief Tiled current deposition kernel with per-team SLM scratch. * * @note Team-policy (one team per spatial tile, accumulates into team SLM scratch with - * atomic adds, then flushes to global J). Available when `team_policy=ON` - * (`#if defined(TEAM_POLICY)`). Stream 2 of the Pattern A plan. + * atomic adds, then flushes to global J). Available when `tiled_deposit=ON` + * (`#if defined(TILED_DEPOSIT)`). Stream 2 of the Pattern A plan. * * @implements * - kernel::DepositCurrentsTiled_kernel<> @@ -51,8 +51,8 @@ namespace kernel { * regression there, but it's good to be able to measure the * crossover. To revert and use flat for zigzag-only builds, change * the dispatch in `engines/srpic/currents.h` from - * `#if defined(TEAM_POLICY)` to - * `#if defined(TEAM_POLICY) && (SHAPE_ORDER > 0)`. + * `#if defined(TILED_DEPOSIT)` to + * `#if defined(TILED_DEPOSIT) && (SHAPE_ORDER > 0)`. * * Particle iteration order is governed by `tile_offsets`: tile `t` * owns particles `[tile_offsets(t), tile_offsets(t+1))`, post-sort. @@ -65,8 +65,8 @@ namespace kernel { * per step elapsed since the last sort. The scratch HALO is * `STENCIL_REACH(O) + DRIFT`, where `STENCIL_REACH = 2` for zigzag * (writes `{i_prev, i_prev+1, i, i+1}` ⇒ +2 above `min(i, i_prev)` with - * `|Δi|=1`) and `O` for Esirkepov. `DRIFT` is the `team_policy_drift` - * CMake knob (macro TEAM_POLICY_DRIFT) — the number of cells a particle + * `|Δi|=1`) and `O` for Esirkepov. `DRIFT` is the `tiled_deposit_drift` + * CMake knob (macro TILED_DEPOSIT_DRIFT) — the number of cells a particle * may drift between two sorts that the halo is sized to absorb — and `1` * by default (the every-step-sorted common case). It is independent of * the sort cadence, which is set at runtime via `spatial_sorting_interval`; @@ -118,8 +118,8 @@ namespace kernel { * pushed once per step between its last sort and a given deposit. With * a runtime sort interval of `K` (spatial_sorting_interval), a particle * drifts at most `K` cells (CFL |v dt/dx| <= 1/2 ⇒ |Δi| <= 1 per step) - * before the next sort. The `team_policy_drift` CMake knob (macro - * TEAM_POLICY_DRIFT) sets DRIFT independently of `K`, sizing the halo so + * before the next sort. The `tiled_deposit_drift` CMake knob (macro + * TILED_DEPOSIT_DRIFT) sets DRIFT independently of `K`, sizing the halo so * a particle that drifts up to DRIFT cells still deposits inside its * tile scratch. DRIFT defaults to 1 (the sorted-every-step common case); * any particle that drifts past the halo (e.g. a larger sort interval, @@ -134,8 +134,8 @@ namespace kernel { // coords conservatively bounds every deposited cell for any order // (Esirkepov reaches max+O; O=0 zigzag reaches max+1). static constexpr int FOOTPRINT_REACH = (O == 0u) ? 1 : static_cast(O); -#if defined(TEAM_POLICY_DRIFT) - static constexpr int DRIFT = static_cast(TEAM_POLICY_DRIFT); +#if defined(TILED_DEPOSIT_DRIFT) + static constexpr int DRIFT = static_cast(TILED_DEPOSIT_DRIFT); #else static constexpr int DRIFT = 1; #endif diff --git a/src/output/CMakeLists.txt b/src/output/CMakeLists.txt index 4cf1bf410..96af8abb4 100644 --- a/src/output/CMakeLists.txt +++ b/src/output/CMakeLists.txt @@ -8,6 +8,10 @@ # * fields.cpp # * stats.cpp # * utils/interpret_prompt.cpp +# * utils/writers.cpp +# * utils/readers.cpp +# * utils/tuning.cpp +# * render/renderer.cpp # # @includes: # @@ -26,14 +30,18 @@ set(SRC_DIR ${CMAKE_CURRENT_SOURCE_DIR}) -set(SOURCES ${SRC_DIR}/stats.cpp ${SRC_DIR}/fields.cpp - ${SRC_DIR}/utils/interpret_prompt.cpp) +set(SOURCES + ${SRC_DIR}/stats.cpp ${SRC_DIR}/fields.cpp + ${SRC_DIR}/utils/interpret_prompt.cpp ${SRC_DIR}/render/renderer.cpp) if(${output}) - list(APPEND SOURCES ${SRC_DIR}/writer.cpp) - list(APPEND SOURCES ${SRC_DIR}/checkpoint.cpp) - list(APPEND SOURCES ${SRC_DIR}/utils/writers.cpp) - list(APPEND SOURCES ${SRC_DIR}/utils/readers.cpp) - list(APPEND SOURCES ${SRC_DIR}/utils/tuning.cpp) + list( + APPEND + SOURCES + ${SRC_DIR}/writer.cpp + ${SRC_DIR}/checkpoint.cpp + ${SRC_DIR}/utils/writers.cpp + ${SRC_DIR}/utils/readers.cpp + ${SRC_DIR}/utils/tuning.cpp) endif() add_library(ntt_output ${SOURCES}) diff --git a/src/output/render/axes.h b/src/output/render/axes.h new file mode 100644 index 000000000..1142c81be --- /dev/null +++ b/src/output/render/axes.h @@ -0,0 +1,860 @@ +/** + * @file output/render/axes.h + * @brief Draw a spine (frame), axis ticks and labels onto the opaque RGBA + * canvas, for both the 2D slice and the 3D volume renders. + * @implements + * - out::axesMargins + * - out::drawAxes2D + * - out::drawAxes3D + * @namespaces: + * - out:: + * @note + * Header-only, host-only, drawn on the MPI root rank after compositing (like the + * colorbar). Reuses the 5x7 bitmap font and helpers from colorbar.h. + * - 2D: the data region maps affinely to a world window [u0,u1]x[v0,v1]; a + * rectangular spine is drawn around it with linear ticks/labels in the + * surrounding margins (so they never overlap the data). + * - 3D: the global box is projected with the ray-march camera into a wireframe + * "spine"; ticks + labels are placed along the three edges emanating from the + * bottom-most projected corner, marks pushed outward from the box centroid. + */ + +#ifndef OUTPUT_RENDER_AXES_H +#define OUTPUT_RENDER_AXES_H + +#include "global.h" + +#include "output/render/colorbar.h" // glyph, scale, fmtNum +#include "output/render/composite.h" // projectToScreen, CameraDevice + +#include +#include +#include +#include +#include +#include + +namespace out { + + namespace axes_hidden { + + // set one opaque pixel, clipped to the full canvas [0,CW)x[0,CH) + inline void px(uint8_t* b, int CW, int CH, int x, int y, uint8_t c) { + if (x < 0 or x >= CW or y < 0 or y >= CH) { + return; + } + const std::size_t i = (static_cast(y) * CW + x) * 4; + b[i + 0] = c; + b[i + 1] = c; + b[i + 2] = c; + b[i + 3] = 255; + } + + inline void thickPx(uint8_t* b, int CW, int CH, int x, int y, int t, uint8_t c) { + for (int dy = -t; dy <= t; ++dy) { + for (int dx = -t; dx <= t; ++dx) { + px(b, CW, CH, x + dx, y + dy, c); + } + } + } + + // Bresenham line, thickness (2t+1) + inline void line(uint8_t* b, + int CW, + int CH, + int x0, + int y0, + int x1, + int y1, + int t, + uint8_t c) { + int dx = std::abs(x1 - x0), sx = (x0 < x1) ? 1 : -1; + int dy = -std::abs(y1 - y0), sy = (y0 < y1) ? 1 : -1; + int err = dx + dy; + while (true) { + thickPx(b, CW, CH, x0, y0, t, c); + if (x0 == x1 and y0 == y1) { + break; + } + const int e2 = 2 * err; + if (e2 >= dy) { + err += dy; + x0 += sx; + } + if (e2 <= dx) { + err += dx; + y0 += sy; + } + } + } + + inline void text(uint8_t* b, + int CW, + int CH, + int x, + int y, + const std::string& str, + int s, + uint8_t c) { + int cx = x; + for (const char ch : str) { + const uint8_t* gl = cbar_hidden::glyph(ch); + for (int row = 0; row < 7; ++row) { + for (int col = 0; col < 5; ++col) { + if (gl[row] & (1u << (4 - col))) { + for (int dy = 0; dy < s; ++dy) { + for (int dx = 0; dx < s; ++dx) { + px(b, CW, CH, cx + col * s + dx, y + row * s + dy, c); + } + } + } + } + } + cx += 6 * s; + } + } + + // Rotated bitmap text: the baseline advances along unit (ax, ay); (ox, oy) + // is the text-local origin (top-left of the first glyph). Each glyph cell + // is oversampled 2x so rotation leaves no gaps. + inline void textRot(uint8_t* b, + int CW, + int CH, + real_t ox, + real_t oy, + const std::string& str, + int s, + real_t ax, + real_t ay, + uint8_t c) { + const real_t dnx = -ay, dny = ax; // glyph "down" (perp. to baseline) + for (std::size_t ci = 0; ci < str.size(); ++ci) { + const uint8_t* gl = cbar_hidden::glyph(str[ci]); + const real_t base = static_cast(ci) * static_cast(6 * s); + for (int row = 0; row < 7; ++row) { + for (int col = 0; col < 5; ++col) { + if (not(gl[row] & (1u << (4 - col)))) { + continue; + } + for (int sy = 0; sy < 2 * s; ++sy) { + for (int sx = 0; sx < 2 * s; ++sx) { + const real_t u = base + static_cast(col * s) + + static_cast(sx) * HALF; + const real_t v = static_cast(row * s) + + static_cast(sy) * HALF; + px(b, + CW, + CH, + static_cast(std::lround(ox + u * ax + v * dnx)), + static_cast(std::lround(oy + u * ay + v * dny)), + c); + } + } + } + } + } + } + + // rotated text centered on (cxp, cyp), baseline along unit (ax, ay) + inline void textRotCentered(uint8_t* b, + int CW, + int CH, + real_t cxp, + real_t cyp, + const std::string& str, + int s, + real_t ax, + real_t ay, + uint8_t c) { + const real_t dnx = -ay, dny = ax; + const real_t w = static_cast(str.size() * 6 * s); + const real_t h = static_cast(7 * s); + textRot(b, + CW, + CH, + cxp - HALF * w * ax - HALF * h * dnx, + cyp - HALF * w * ay - HALF * h * dny, + str, + s, + ax, + ay, + c); + } + + // orient a screen-space edge direction so text reads naturally (rightward + // for near-horizontal edges, upward for near-vertical ones) + inline void readableDir(real_t& ax, real_t& ay) { + const real_t n = static_cast( + std::sqrt(static_cast(ax * ax + ay * ay))); + if (n < static_cast(1e-9)) { + ax = ONE; + ay = ZERO; + return; + } + ax /= n; + ay /= n; + if (std::fabs(static_cast(ax)) >= std::fabs(static_cast(ay))) { + if (ax < ZERO) { + ax = -ax; + ay = -ay; + } + } else if (ay > ZERO) { + ax = -ax; + ay = -ay; + } + } + + // vertical stack of characters (top to bottom), used for the y-axis name + inline void textVert(uint8_t* b, + int CW, + int CH, + int x, + int y, + const std::string& str, + int s, + uint8_t c) { + int cy = y; + for (const char ch : str) { + const std::string one(1, ch); + text(b, CW, CH, x, cy, one, s, c); + cy += 8 * s; + } + } + + inline auto textW(const std::string& str, int s) -> int { + return static_cast(str.size()) * 6 * s; + } + + inline auto contrast(const real_t bg[3]) -> uint8_t { + const real_t lum = static_cast(0.299) * bg[0] + + static_cast(0.587) * bg[1] + + static_cast(0.114) * bg[2]; + return (lum < HALF) ? 255 : 0; + } + + // a "nice" number close to x (1/2/5 x 10^k) + inline auto niceNum(real_t x, bool round) -> real_t { + if (x <= ZERO) { + return ONE; + } + const real_t e = std::floor(std::log10(static_cast(x))); + const real_t f = x / static_cast(std::pow(10.0, e)); + real_t nf; + if (round) { + nf = (f < static_cast(1.5)) + ? ONE + : ((f < static_cast(3)) ? static_cast(2) + : (f < static_cast(7)) ? static_cast(5) + : static_cast(10)); + } else { + nf = (f <= ONE) ? ONE + : (f <= static_cast(2)) ? static_cast(2) + : (f <= static_cast(5)) ? static_cast(5) + : static_cast(10); + } + return nf * static_cast(std::pow(10.0, e)); + } + + // a tick at k*pi/D, carried with its reduced fraction k/D == n/d + struct PiTick { + real_t val; + int n, d; + }; + + // format a multiple of pi as "0", "PI", "PI", "PI/", "PI/" + // (the bitmap font has no greek glyph, so "PI" is spelled out) + inline auto fmtPi(int n, int d) -> std::string { + if (n == 0) { + return "0"; + } + std::string s; + if (n < 0) { + s += "-"; + n = -n; + } + if (n != 1) { + s += std::to_string(n); + } + s += "PI"; + if (d != 1) { + s += "/" + std::to_string(d); + } + return s; + } + + // ticks at nice fractions of pi spanning [lo, hi] + inline auto piTicks(real_t lo, real_t hi, int nticks) -> std::vector { + std::vector out; + const real_t PI = static_cast(constant::PI); + const real_t range = hi - lo; + if (not(range > ZERO) or nticks < 2) { + return out; + } + const real_t ideal = static_cast(nticks - 1) * PI / range; + const int Ds[] = { 1, 2, 3, 4, 6, 8, 12, 16, 24 }; + int D = 4; + real_t bestd = static_cast(1e30); + for (const int dd : Ds) { + const real_t df = static_cast( + std::fabs(static_cast(dd) - ideal)); + if (df < bestd) { + bestd = df; + D = dd; + } + } + const real_t step = PI / static_cast(D); + const int k0 = static_cast( + std::ceil(static_cast(lo / step) - 1e-6)); + const int k1 = static_cast( + std::floor(static_cast(hi / step) + 1e-6)); + for (int k = k0; k <= k1; ++k) { + int n = k, d = D; + const int g = std::gcd(std::abs(n), d); + if (g > 0) { + n /= g; + d /= g; + } + out.push_back({ static_cast(k) * step, n, d }); + } + return out; + } + + inline auto niceTicks(real_t lo, real_t hi, int n) -> std::vector { + std::vector out; + if (not(hi > lo) or n < 2) { + return out; + } + const real_t step = niceNum((hi - lo) / static_cast(n - 1), true); + if (step <= ZERO) { + return out; + } + const real_t g0 = static_cast( + std::ceil(static_cast(lo / step)) * step); + const real_t eps = static_cast(1e-6) * step; + for (real_t v = g0; v <= hi + static_cast(0.5) * step; v += step) { + if (v >= lo - eps and v <= hi + eps) { + out.push_back((std::fabs(static_cast(v)) < eps) ? ZERO : v); + } + } + return out; + } + + } // namespace axes_hidden + + /** + * @brief Left/bottom margin (pixels) that the axes annotation needs. + * @note Zero when axes are disabled. Depends only on H, so it can size the + * canvas before drawing. + */ + inline void axesMargins(bool axes, int H, int& ml, int& mb) { + if (not axes) { + ml = 0; + mb = 0; + return; + } + const int s = cbar_hidden::scale(H); + const int cw = 6 * s; + const int ch = 8 * s; + const int tl = 5 * s; + const int gap = 2 * s; + ml = tl + gap + 7 * cw + gap + cw + gap; // ticks + numbers + y-axis name + mb = tl + gap + ch + gap + ch + gap; // ticks + numbers + x-axis name + } + + /** + * @brief Draw a 2D spine + ticks + labels around the data region. + * @param rgba canvas (CW*CH*4), opaque + * @param CW,CH canvas dimensions (includes margins) + * @param x0 left pixel of the data region (== left margin width) + * @param W,H data region dimensions + * @param u0,u1,v0,v1 world window mapped onto the data region (+v is up) + * @param xlabel,ylabel axis names + * @param bg background RGB (for contrasting text color) + * @param nticks target number of ticks per axis + */ + inline void drawAxes2D(uint8_t* rgba, + int CW, + int CH, + int x0, + int W, + int H, + real_t u0, + real_t u1, + real_t v0, + real_t v1, + real_t du0, + real_t du1, + real_t dv0, + real_t dv1, + const std::string& xlabel, + const std::string& ylabel, + const real_t bg[3], + int nticks) { + using namespace axes_hidden; + const uint8_t c = contrast(bg); + const int s = cbar_hidden::scale(H); + const int ch = 8 * s; + const int tl = 5 * s; + const int gap = 2 * s; + const int th = std::max(0, s / 2 - 1); // spine half-thickness + + // [u0,u1]x[v0,v1] is the world window mapped onto the full data region; + // [du0,du1]x[dv0,dv1] is the actual data box (the domain/region), a + // sub-rect when the window was aspect-expanded. The spine + ticks clamp to + // the DATA box so the empty aspect pad stays outside the frame. + const int xL = x0, xR = x0 + W - 1, yT = 0, yB = H - 1; + auto X = [&](real_t u) -> int { + return static_cast( + std::lround(static_cast(xL) + + static_cast((u - u0) / (u1 - u0)) * (xR - xL))); + }; + auto Y = [&](real_t v) -> int { + return static_cast( + std::lround(static_cast(yB) - + static_cast((v - v0) / (v1 - v0)) * (yB - yT))); + }; + auto clampX = [&](int x) { + return (x < xL) ? xL : ((x > xR) ? xR : x); + }; + auto clampY = [&](int y) { + return (y < yT) ? yT : ((y > yB) ? yB : y); + }; + const int xLd = clampX(X(du0)), xRd = clampX(X(du1)); + const int yTd = clampY(Y(dv1)), yBd = clampY(Y(dv0)); // dv1 = top + + // spine around the data box + line(rgba, CW, CH, xLd, yTd, xRd, yTd, th, c); + line(rgba, CW, CH, xLd, yBd, xRd, yBd, th, c); + line(rgba, CW, CH, xLd, yTd, xLd, yBd, th, c); + line(rgba, CW, CH, xRd, yTd, xRd, yBd, th, c); + + // x ticks (below the data box): marks + labels + for (const real_t tv : niceTicks(du0, du1, nticks)) { + const int x = X(tv); + if (x < xLd or x > xRd) { + continue; + } + line(rgba, CW, CH, x, yBd, x, yBd + tl, 0, c); + const std::string lab = cbar_hidden::fmtNum(tv); + text(rgba, CW, CH, x - textW(lab, s) / 2, yBd + tl + gap, lab, s, c); + } + // y ticks (left of the data box): marks + right-aligned labels + for (const real_t tv : niceTicks(dv0, dv1, nticks)) { + const int y = Y(tv); + if (y < yTd or y > yBd) { + continue; + } + line(rgba, CW, CH, xLd, y, xLd - tl, y, 0, c); + const std::string lab = cbar_hidden::fmtNum(tv); + text(rgba, CW, CH, xLd - tl - gap - textW(lab, s), y - ch / 2, lab, s, c); + } + // axis names + if (not xlabel.empty()) { + text(rgba, + CW, + CH, + (xLd + xRd) / 2 - textW(xlabel, s) / 2, + yBd + tl + gap + ch + gap, + xlabel, + s, + c); + } + if (not ylabel.empty()) { + textVert(rgba, + CW, + CH, + std::max(gap, xLd - tl - gap - 7 * (6 * s) - gap - 6 * s), + (yTd + yBd) / 2 - 4 * s * static_cast(ylabel.size()) / 2, + ylabel, + s, + c); + } + } + + /** + * @brief Draw polar (curvilinear) axes for a 2D spherical meridional slice. + * @param x0,W,H data region (the slice maps world (X = r sin th, Z = r cos + * th) onto it via the [u0,u1]x[v0,v1] window, aspect-matched) + * @param rmin,rmax,tmin,tmax global (r, theta) extent + * @param mirror whether the half-plane is mirrored into a full disk + * @param rlabel,tlabel names for the radial / angular axes (e.g. "R","Theta") + * @note Draws (1) a curvilinear spine: the outer & inner arcs plus the two + * radial edges (or full arcs when mirrored); (2) an "R" radial axis along the + * symmetry axis (X=0) with R=0 at the center, increasing outward; (3) a + * "Theta" angular axis with ticks along the outer arc. + */ + inline void drawAxesPolar(uint8_t* rgba, + int CW, + int CH, + int x0, + int W, + int H, + real_t u0, + real_t u1, + real_t v0, + real_t v1, + real_t rmin, + real_t rmax, + real_t tmin, + real_t tmax, + bool mirror, + const std::string& rlabel, + const std::string& tlabel, + const real_t bg[3], + int nticks) { + using namespace axes_hidden; + const uint8_t c = contrast(bg); + const int s = cbar_hidden::scale(H); + const int ch = 8 * s; + const int tl = 5 * s; + const int gap = 2 * s; + + auto WX = [&](real_t X) -> real_t { + return static_cast(x0) + + (X - u0) / (u1 - u0) * static_cast(W) - HALF; + }; + auto WZ = [&](real_t Z) -> real_t { + return (v1 - Z) / (v1 - v0) * static_cast(H) - HALF; + }; + auto PX = [&](real_t X, real_t Z, int& qx, int& qy) { + qx = static_cast(std::lround(WX(X))); + qy = static_cast(std::lround(WZ(Z))); + }; + + const int NA = 160; + auto arc = [&](real_t r, real_t sgn) { + int qx, qy; + PX(static_cast(sgn * r * std::sin(static_cast(tmin))), + static_cast(r * std::cos(static_cast(tmin))), + qx, + qy); + for (int i = 1; i <= NA; ++i) { + const real_t th = tmin + (tmax - tmin) * static_cast(i) / NA; + int rx, ry; + PX(static_cast(sgn * r * std::sin(static_cast(th))), + static_cast(r * std::cos(static_cast(th))), + rx, + ry); + line(rgba, CW, CH, qx, qy, rx, ry, 0, c); + qx = rx; + qy = ry; + } + }; + auto ray = [&](real_t th) { + int ax, ay, bx, by; + PX(static_cast(rmin * std::sin(static_cast(th))), + static_cast(rmin * std::cos(static_cast(th))), + ax, + ay); + PX(static_cast(rmax * std::sin(static_cast(th))), + static_cast(rmax * std::cos(static_cast(th))), + bx, + by); + line(rgba, CW, CH, ax, ay, bx, by, 0, c); + }; + + // ---- curvilinear spine -------------------------------------------- // + arc(rmax, ONE); + arc(rmin, ONE); + if (mirror) { + arc(rmax, -ONE); + arc(rmin, -ONE); + } else { + ray(tmin); + ray(tmax); + } + + // ---- R axis: along the symmetry axis (X = 0), zero at the center --- // + const int rmaxlabW = textW(cbar_hidden::fmtNum(rmax), s); + for (const real_t Rv : niceTicks(ZERO, rmax, nticks)) { + for (int sg = 1; sg >= -1; sg -= 2) { + if (sg < 0 and Rv == ZERO) { + continue; // a single tick at the center + } + int ax, ay; + PX(ZERO, static_cast(sg) * Rv, ax, ay); + line(rgba, CW, CH, ax, ay, ax - tl, ay, 0, c); + const std::string lab = cbar_hidden::fmtNum(Rv); + text(rgba, CW, CH, ax - tl - gap - textW(lab, s), ay - ch / 2, lab, s, c); + } + } + if (not rlabel.empty()) { + int ax, ay; + PX(ZERO, ZERO, ax, ay); + const real_t off = static_cast(tl + gap + rmaxlabW + gap + ch); + textRotCentered(rgba, + CW, + CH, + static_cast(ax) - off, + static_cast(ay), + rlabel, + s, + ZERO, + -ONE, + c); + } + + // ---- Theta axis: ticks + labels (fractions of pi) along the arc --- // + // widest tick label, so the "Theta" name can clear them all + real_t maxlw = static_cast(ch); + for (const auto& tk : piTicks(tmin, tmax, nticks)) { + maxlw = std::max(maxlw, static_cast(textW(fmtPi(tk.n, tk.d), s))); + } + for (const auto& tk : piTicks(tmin, tmax, nticks)) { + const real_t ox = static_cast(std::sin(static_cast(tk.val))); + const real_t oz = static_cast(std::cos(static_cast(tk.val))); + int px0, py0; + PX(rmax * ox, rmax * oz, px0, py0); + const real_t dxp = ox, dyp = -oz; // outward pixel direction + line(rgba, + CW, + CH, + px0, + py0, + static_cast(std::lround( + static_cast(px0) + dxp * static_cast(tl))), + static_cast(std::lround( + static_cast(py0) + dyp * static_cast(tl))), + 0, + c); + const std::string lab = fmtPi(tk.n, tk.d); + // push the (horizontal) label box fully clear of the arc/tick at any + // angle: offset its center by its own support along the outward direction + const real_t inset = HALF * (static_cast(textW(lab, s)) * + static_cast( + std::fabs(static_cast(dxp))) + + static_cast(ch) * + static_cast( + std::fabs(static_cast(dyp)))); + const real_t lo = static_cast(tl + gap) + inset; + text(rgba, + CW, + CH, + static_cast(std::lround(static_cast(px0) + dxp * lo)) - + textW(lab, s) / 2, + static_cast(std::lround(static_cast(py0) + dyp * lo)) - + ch / 2, + lab, + s, + c); + } + if (not tlabel.empty()) { + const real_t tm = HALF * (tmin + tmax); + const real_t ox = static_cast(std::sin(static_cast(tm))); + const real_t oz = static_cast(std::cos(static_cast(tm))); + int px0, py0; + PX(rmax * ox, rmax * oz, px0, py0); + real_t adx = static_cast( + std::cos(static_cast(tm))); // arc tangent + real_t ady = static_cast(std::sin(static_cast(tm))); + readableDir(adx, ady); + // beyond the tick labels (which reach ~tl+gap+maxlw from the arc) + const real_t off = static_cast(tl + gap) + maxlw + + static_cast(gap + ch); + textRotCentered(rgba, + CW, + CH, + static_cast(px0) + ox * off, + static_cast(py0) - oz * off, + tlabel, + s, + adx, + ady, + c); + } + } + + /** + * @brief Draw a 3D bounding-box wireframe spine + ticks + labels. + * @param rgba canvas (CW*CH*4), opaque + * @param CW,CH canvas dimensions + * @param x0 left pixel of the data region (left margin width) + * @param W,H data region dimensions (used for the camera projection) + * @param cam ray-march camera (inverted to project world -> screen) + * @param ext global box extent (3 axes) + * @param lab axis names [x, y, z] + * @param bg background RGB + * @param nticks target number of ticks per axis + */ + inline void drawAxes3D(uint8_t* rgba, + int CW, + int CH, + int x0, + int W, + int H, + const CameraDevice& cam, + const boundaries_t& ext, + const std::string lab[3], + const real_t bg[3], + int nticks) { + using namespace axes_hidden; + if (ext.size() < 3) { + return; + } + const uint8_t c = contrast(bg); + const int s = cbar_hidden::scale(H); + const int ch = 8 * s; + const int tl = 5 * s; + + auto corner = [&](int m, real_t p[3]) { + p[0] = (m & 1) ? ext[0].second : ext[0].first; + p[1] = (m & 2) ? ext[1].second : ext[1].first; + p[2] = (m & 4) ? ext[2].second : ext[2].first; + }; + real_t cx[8], cy[8]; + bool ok[8]; + for (int m = 0; m < 8; ++m) { + real_t p[3]; + corner(m, p); + real_t a, b; + ok[m] = projectToScreen(cam, W, H, p, a, b); + cx[m] = a + static_cast(x0); + cy[m] = b; + } + // box centroid (for outward tick/label direction) + real_t cen[3] = { HALF * (ext[0].first + ext[0].second), + HALF * (ext[1].first + ext[1].second), + HALF * (ext[2].first + ext[2].second) }; + real_t ccx = ZERO, ccy = ZERO; + { + real_t a, b; + projectToScreen(cam, W, H, cen, a, b); + ccx = a + static_cast(x0); + ccy = b; + } + + (void)ok; + // The wireframe "spine" is drawn in the ray-march (depth-occluded), so here + // we only annotate. For each axis, pick one *silhouette* edge: its two + // adjacent faces point opposite ways relative to the camera (one toward it, + // one away), so the edge lies on the box OUTLINE and is always in the + // foreground -- never hidden behind the volume. Among the (usually two) + // silhouette candidates take the one nearest the bottom-left of the image, + // the conventional, view-independent place for axis annotation. + auto frontFace = [&](int axis, int side) -> bool { + const real_t nrm = (side != 0) ? ONE : -ONE; // outward normal sign + return (nrm * (-cam.forward[axis])) > ZERO; // points toward the camera? + }; + for (int d = 0; d < 3; ++d) { + const int e1 = (d == 0) ? 1 : 0; + const int e2 = (d == 2) ? 1 : 2; + int bs1 = 0, bs2 = 0; + bool found = false; + real_t best = ZERO; + for (int s1 = 0; s1 < 2; ++s1) { + for (int s2 = 0; s2 < 2; ++s2) { + const int m0 = (s1 << e1) | (s2 << e2); + const int m1 = m0 | (1 << d); + const real_t mx = HALF * (cx[m0] + cx[m1]); + const real_t my = HALF * (cy[m0] + cy[m1]); + // prefer the foreground (bottom-left) edge: larger pixel-y is lower, + // smaller pixel-x is further left. Axis-independent, so it follows + // the camera instead of assuming the default diagonal view. + real_t score = my - mx; + if (frontFace(e1, s1) != frontFace(e2, s2)) { + score += static_cast(1e6); // strongly prefer silhouette edges + } + if (not found or score > best) { + best = score; + bs1 = s1; + bs2 = s2; + found = true; + } + } + } + const int m0 = (bs1 << e1) | (bs2 << e2); + const int m1 = m0 | (1 << d); + real_t o[3]; + corner(m0, o); // perpendicular coords fixed; axis d swept for ticks + const real_t lo = ext[d].first, hi = ext[d].second; + // screen-space perpendicular to the edge, flipped to point outward + real_t ex = cx[m1] - cx[m0], ey = cy[m1] - cy[m0]; + real_t el = static_cast( + std::sqrt(static_cast(ex * ex + ey * ey))); + if (el < ONE) { + el = ONE; + } + ex /= el; + ey /= el; + real_t pxd = -ey, pyd = ex; + { + const real_t mxv = HALF * (cx[m0] + cx[m1]) - ccx; + const real_t myv = HALF * (cy[m0] + cy[m1]) - ccy; + if (pxd * mxv + pyd * myv < ZERO) { + pxd = -pxd; + pyd = -pyd; + } + } + // readable text baseline aligned with the edge direction + real_t adx = ex, ady = ey; + readableDir(adx, ady); + const real_t numOff = static_cast(tl) + + static_cast(5 * s); // number center + const real_t nameOff = static_cast(tl) + + static_cast(14 * s); // axis-name center + // ticks + numeric labels (numbers rotated along the edge for x & y; the + // vertical z edge keeps horizontal numbers, which read more easily) + for (const real_t tv : niceTicks(lo, hi, nticks)) { + real_t p[3] = { o[0], o[1], o[2] }; + p[d] = tv; + real_t a, b; + if (not projectToScreen(cam, W, H, p, a, b)) { + continue; + } + a += static_cast(x0); + const int mx = static_cast( + std::lround(a + pxd * static_cast(tl))); + const int my = static_cast( + std::lround(b + pyd * static_cast(tl))); + line(rgba, + CW, + CH, + static_cast(std::lround(a)), + static_cast(std::lround(b)), + mx, + my, + 0, + c); + const std::string l2 = cbar_hidden::fmtNum(tv); + if (d == 2) { + const int tx = (pxd < ZERO) ? (mx - textW(l2, s)) : mx; + text(rgba, CW, CH, tx, my - ch / 2, l2, s, c); + } else { + textRotCentered(rgba, + CW, + CH, + a + pxd * numOff, + b + pyd * numOff, + l2, + s, + adx, + ady, + c); + } + } + // axis name at the MIDDLE of the edge (near the central tick), aligned + // with the edge and pushed further outward than the numbers + if (not lab[d].empty()) { + real_t mid[3] = { o[0], o[1], o[2] }; + mid[d] = HALF * (lo + hi); + real_t a, b; + if (projectToScreen(cam, W, H, mid, a, b)) { + a += static_cast(x0); + textRotCentered(rgba, + CW, + CH, + a + pxd * nameOff, + b + pyd * nameOff, + lab[d], + s, + adx, + ady, + c); + } + } + } + } + +} // namespace out + +#endif // OUTPUT_RENDER_AXES_H diff --git a/src/output/render/colorbar.h b/src/output/render/colorbar.h new file mode 100644 index 000000000..7c01742e0 --- /dev/null +++ b/src/output/render/colorbar.h @@ -0,0 +1,496 @@ +/** + * @file output/render/colorbar.h + * @brief Draw a colorbar (gradient strip + value ticks + label) onto an + * opaque 8-bit RGBA image, using a self-contained 5x7 bitmap font. + * @implements + * - out::drawColorbar + * @namespaces: + * - out:: + * @note + * Header-only, host-only. No font dependency: a compact 5x7 ASCII font (digits, + * sign/exponent symbols, and A-Z) is embedded. Lowercase is mapped to + * uppercase; unknown glyphs render as blank. Drawn on the MPI root rank after + * the final composite, so it only ever touches the output buffer. + */ + +#ifndef OUTPUT_RENDER_COLORBAR_H +#define OUTPUT_RENDER_COLORBAR_H + +#include "global.h" + +#include "utils/numeric.h" + +#include "output/render/transfer_fn.h" + +#include +#include +#include +#include +#include +#include + +namespace out { + + namespace cbar_hidden { + + // 5x7 glyph: 7 rows, low 5 bits per row, bit 4 = leftmost column. + inline auto glyph(char c) -> const uint8_t* { + // map lowercase to uppercase + if (c >= 'a' and c <= 'z') { + c = static_cast(c - 'a' + 'A'); + } + switch (c) { + case '0': { + static const uint8_t g[7] = { 0b01110, 0b10001, 0b10011, 0b10101, + 0b11001, 0b10001, 0b01110 }; + return g; + } + case '1': { + static const uint8_t g[7] = { 0b00100, 0b01100, 0b00100, 0b00100, + 0b00100, 0b00100, 0b01110 }; + return g; + } + case '2': { + static const uint8_t g[7] = { 0b01110, 0b10001, 0b00001, 0b00010, + 0b00100, 0b01000, 0b11111 }; + return g; + } + case '3': { + static const uint8_t g[7] = { 0b11111, 0b00010, 0b00100, 0b00010, + 0b00001, 0b10001, 0b01110 }; + return g; + } + case '4': { + static const uint8_t g[7] = { 0b00010, 0b00110, 0b01010, 0b10010, + 0b11111, 0b00010, 0b00010 }; + return g; + } + case '5': { + static const uint8_t g[7] = { 0b11111, 0b10000, 0b11110, 0b00001, + 0b00001, 0b10001, 0b01110 }; + return g; + } + case '6': { + static const uint8_t g[7] = { 0b00110, 0b01000, 0b10000, 0b11110, + 0b10001, 0b10001, 0b01110 }; + return g; + } + case '7': { + static const uint8_t g[7] = { 0b11111, 0b00001, 0b00010, 0b00100, + 0b01000, 0b01000, 0b01000 }; + return g; + } + case '8': { + static const uint8_t g[7] = { 0b01110, 0b10001, 0b10001, 0b01110, + 0b10001, 0b10001, 0b01110 }; + return g; + } + case '9': { + static const uint8_t g[7] = { 0b01110, 0b10001, 0b10001, 0b01111, + 0b00001, 0b00010, 0b01100 }; + return g; + } + case '.': { + static const uint8_t g[7] = { 0b00000, 0b00000, 0b00000, 0b00000, + 0b00000, 0b00110, 0b00110 }; + return g; + } + case '-': { + static const uint8_t g[7] = { 0b00000, 0b00000, 0b00000, 0b11111, + 0b00000, 0b00000, 0b00000 }; + return g; + } + case '+': { + static const uint8_t g[7] = { 0b00000, 0b00100, 0b00100, 0b11111, + 0b00100, 0b00100, 0b00000 }; + return g; + } + case '=': { + static const uint8_t g[7] = { 0b00000, 0b00000, 0b11111, 0b00000, + 0b11111, 0b00000, 0b00000 }; + return g; + } + case '_': { + static const uint8_t g[7] = { 0b00000, 0b00000, 0b00000, 0b00000, + 0b00000, 0b00000, 0b11111 }; + return g; + } + case ':': { + static const uint8_t g[7] = { 0b00000, 0b00110, 0b00110, 0b00000, + 0b00110, 0b00110, 0b00000 }; + return g; + } + case '/': { + static const uint8_t g[7] = { 0b00001, 0b00010, 0b00010, 0b00100, + 0b01000, 0b01000, 0b10000 }; + return g; + } + case 'A': { + static const uint8_t g[7] = { 0b01110, 0b10001, 0b10001, 0b11111, + 0b10001, 0b10001, 0b10001 }; + return g; + } + case 'B': { + static const uint8_t g[7] = { 0b11110, 0b10001, 0b10001, 0b11110, + 0b10001, 0b10001, 0b11110 }; + return g; + } + case 'C': { + static const uint8_t g[7] = { 0b01110, 0b10001, 0b10000, 0b10000, + 0b10000, 0b10001, 0b01110 }; + return g; + } + case 'D': { + static const uint8_t g[7] = { 0b11100, 0b10010, 0b10001, 0b10001, + 0b10001, 0b10010, 0b11100 }; + return g; + } + case 'E': { + static const uint8_t g[7] = { 0b11111, 0b10000, 0b10000, 0b11110, + 0b10000, 0b10000, 0b11111 }; + return g; + } + case 'F': { + static const uint8_t g[7] = { 0b11111, 0b10000, 0b10000, 0b11110, + 0b10000, 0b10000, 0b10000 }; + return g; + } + case 'G': { + static const uint8_t g[7] = { 0b01110, 0b10001, 0b10000, 0b10111, + 0b10001, 0b10001, 0b01111 }; + return g; + } + case 'H': { + static const uint8_t g[7] = { 0b10001, 0b10001, 0b10001, 0b11111, + 0b10001, 0b10001, 0b10001 }; + return g; + } + case 'I': { + static const uint8_t g[7] = { 0b01110, 0b00100, 0b00100, 0b00100, + 0b00100, 0b00100, 0b01110 }; + return g; + } + case 'J': { + static const uint8_t g[7] = { 0b00111, 0b00010, 0b00010, 0b00010, + 0b10010, 0b10010, 0b01100 }; + return g; + } + case 'K': { + static const uint8_t g[7] = { 0b10001, 0b10010, 0b10100, 0b11000, + 0b10100, 0b10010, 0b10001 }; + return g; + } + case 'L': { + static const uint8_t g[7] = { 0b10000, 0b10000, 0b10000, 0b10000, + 0b10000, 0b10000, 0b11111 }; + return g; + } + case 'M': { + static const uint8_t g[7] = { 0b10001, 0b11011, 0b10101, 0b10101, + 0b10001, 0b10001, 0b10001 }; + return g; + } + case 'N': { + static const uint8_t g[7] = { 0b10001, 0b11001, 0b10101, 0b10011, + 0b10001, 0b10001, 0b10001 }; + return g; + } + case 'O': { + static const uint8_t g[7] = { 0b01110, 0b10001, 0b10001, 0b10001, + 0b10001, 0b10001, 0b01110 }; + return g; + } + case 'P': { + static const uint8_t g[7] = { 0b11110, 0b10001, 0b10001, 0b11110, + 0b10000, 0b10000, 0b10000 }; + return g; + } + case 'Q': { + static const uint8_t g[7] = { 0b01110, 0b10001, 0b10001, 0b10001, + 0b10101, 0b10010, 0b01101 }; + return g; + } + case 'R': { + static const uint8_t g[7] = { 0b11110, 0b10001, 0b10001, 0b11110, + 0b10100, 0b10010, 0b10001 }; + return g; + } + case 'S': { + static const uint8_t g[7] = { 0b01111, 0b10000, 0b10000, 0b01110, + 0b00001, 0b00001, 0b11110 }; + return g; + } + case 'T': { + static const uint8_t g[7] = { 0b11111, 0b00100, 0b00100, 0b00100, + 0b00100, 0b00100, 0b00100 }; + return g; + } + case 'U': { + static const uint8_t g[7] = { 0b10001, 0b10001, 0b10001, 0b10001, + 0b10001, 0b10001, 0b01110 }; + return g; + } + case 'V': { + static const uint8_t g[7] = { 0b10001, 0b10001, 0b10001, 0b10001, + 0b10001, 0b01010, 0b00100 }; + return g; + } + case 'W': { + static const uint8_t g[7] = { 0b10001, 0b10001, 0b10001, 0b10101, + 0b10101, 0b11011, 0b10001 }; + return g; + } + case 'X': { + static const uint8_t g[7] = { 0b10001, 0b10001, 0b01010, 0b00100, + 0b01010, 0b10001, 0b10001 }; + return g; + } + case 'Y': { + static const uint8_t g[7] = { 0b10001, 0b10001, 0b01010, 0b00100, + 0b00100, 0b00100, 0b00100 }; + return g; + } + case 'Z': { + static const uint8_t g[7] = { 0b11111, 0b00001, 0b00010, 0b00100, + 0b01000, 0b10000, 0b11111 }; + return g; + } + default: { + static const uint8_t g[7] = { 0, 0, 0, 0, 0, 0, 0 }; + return g; + } // blank + } + } + + inline void setPx(uint8_t* rgba, + int W, + int H, + int x, + int y, + uint8_t r, + uint8_t g, + uint8_t b) { + if (x < 0 or x >= W or y < 0 or y >= H) { + return; + } + const std::size_t i = (static_cast(y) * W + x) * 4; + rgba[i + 0] = r; + rgba[i + 1] = g; + rgba[i + 2] = b; + rgba[i + 3] = 255; + } + + inline void drawChar(uint8_t* rgba, + int W, + int H, + int x, + int y, + char c, + int s, + uint8_t r, + uint8_t g, + uint8_t b) { + const uint8_t* gl = glyph(c); + for (int row = 0; row < 7; ++row) { + for (int col = 0; col < 5; ++col) { + if (gl[row] & (1u << (4 - col))) { + for (int dy = 0; dy < s; ++dy) { + for (int dx = 0; dx < s; ++dx) { + setPx(rgba, W, H, x + col * s + dx, y + row * s + dy, r, g, b); + } + } + } + } + } + } + + inline void drawText(uint8_t* rgba, + int W, + int H, + int x, + int y, + const std::string& str, + int s, + uint8_t r, + uint8_t g, + uint8_t b) { + int cx = x; + for (const char c : str) { + drawChar(rgba, W, H, cx, y, c, s, r, g, b); + cx += 6 * s; // 5px glyph + 1px spacing + } + } + + inline auto quant(real_t v) -> uint8_t { + const real_t c = (v < ZERO) ? ZERO : ((v > ONE) ? ONE : v); + return static_cast(c * static_cast(255.0) + HALF); + } + + inline auto fmtNum(real_t v) -> std::string { + char buf[32]; + std::snprintf(buf, sizeof(buf), "%.3g", static_cast(v)); + return std::string(buf); + } + + inline auto scale(int H) -> int { + return std::max(2, H / 400); + } + + } // namespace cbar_hidden + + /** + * @brief Pixel width of the colorbar block (bar + gap + labels + padding). + * @note Depends only on H, so it can size a canvas margin before drawing. + */ + inline auto colorbarBlockWidth(int H) -> int { + const int s = cbar_hidden::scale(H); + const int char_w = 6 * s; + const int bar_w = std::max(12, H / 50); + const int gap = 3 * s; + const int label_w = 9 * char_w; // room for e.g. "-1.23e+04" + const int pad = 4 * s; + return pad + bar_w + gap + label_w + pad; + } + + /** + * @brief Draw a vertical colorbar onto an opaque RGBA buffer. + * @param rgba width*height*4 bytes, opaque (alpha forced to 255 on drawn px) + * @param W,H image dimensions + * @param colormap name of the colormap to redraw the gradient + * @param vmin,vmax value range mapped onto the bar + * @param log_scale if true, ticks are spaced/labelled logarithmically + * @param label title drawn above the bar (e.g. the field name) + * @param bg background RGB (to auto-pick contrasting text color) + * @param ticks explicit tick values to label; if empty, 5 evenly-spaced + * ticks are generated. Values outside [vmin, vmax] are skipped. + * @param span_top,span_bot vertical pixel span the bar should occupy (e.g. the + * data domain, so the bar is centered on the actual data). When + * span_top < 0 or the span is empty, the bar is half the canvas + * height, centered on the canvas (the default). + */ + inline void drawColorbar(uint8_t* rgba, + int W, + int H, + const std::string& colormap, + real_t vmin, + real_t vmax, + bool log_scale, + const std::string& label, + const real_t bg[3], + const std::vector& ticks = {}, + int span_top = -1, + int span_bot = -1) { + using namespace cbar_hidden; + + const int s = scale(H); + const int char_h = 8 * s; + const int bar_w = std::max(12, H / 50); + const bool aligned = (span_top >= 0) and (span_bot > span_top); + const int bar_h = aligned ? (span_bot - span_top) : (H / 2); + const int gap = 3 * s; + const int pad = 4 * s; + const int block_w = colorbarBlockWidth(H); + + // place the bar near the right edge of the (possibly extended) canvas + int bar_x = W - block_w + pad; + if (bar_x < pad) { + bar_x = pad; + } + // vertical position: aligned to the data span if given, else canvas-centered + const int bar_y = aligned ? span_top : ((H - bar_h) / 2); + + // contrasting monochrome for text / frame / ticks + const real_t lum = static_cast(0.299) * bg[0] + + static_cast(0.587) * bg[1] + + static_cast(0.114) * bg[2]; + const uint8_t tc = (lum < HALF) ? 255 : 0; + + // gradient strip (top = vmax, bottom = vmin) + for (int j = 0; j < bar_h; ++j) { + const real_t u = (bar_h > 1) ? ONE - static_cast(j) / + static_cast(bar_h - 1) + : ZERO; + real_t cr, cg, cb; + colormapRGB(colormap, u, cr, cg, cb); + const uint8_t R = quant(cr), G = quant(cg), B = quant(cb); + for (int i = 0; i < bar_w; ++i) { + setPx(rgba, W, H, bar_x + i, bar_y + j, R, G, B); + } + } + + // frame (thickness s) + for (int t = 0; t < s; ++t) { + for (int i = -t; i < bar_w + t; ++i) { + setPx(rgba, W, H, bar_x + i, bar_y - t, tc, tc, tc); + setPx(rgba, W, H, bar_x + i, bar_y + bar_h - 1 + t, tc, tc, tc); + } + for (int j = -t; j < bar_h + t; ++j) { + setPx(rgba, W, H, bar_x - t, bar_y + j, tc, tc, tc); + setPx(rgba, W, H, bar_x + bar_w - 1 + t, bar_y + j, tc, tc, tc); + } + } + + // ticks + labels: explicit values if given, else 5 evenly-spaced + const bool can_log = log_scale and vmin > ZERO and vmax > ZERO; + const real_t lvmin = can_log ? math::log10(vmin) : ZERO; + const real_t lvmax = can_log ? math::log10(vmax) : ZERO; + std::vector> tk; // (u in [0,1], value) + if (ticks.empty()) { + const int nticks = 5; + for (int t = 0; t < nticks; ++t) { + const real_t u = static_cast(t) / static_cast(nticks - 1); + const real_t val = can_log ? math::pow(static_cast(10), + lvmin + (lvmax - lvmin) * u) + : (vmin + (vmax - vmin) * u); + tk.emplace_back(u, val); + } + } else { + const real_t span = can_log ? (lvmax - lvmin) : (vmax - vmin); + for (const real_t v : ticks) { + if (can_log and v <= ZERO) { + continue; + } + const real_t u = (span != ZERO) + ? ((can_log ? (math::log10(v) - lvmin) : (v - vmin)) / span) + : ZERO; + if (u < static_cast(-1e-4) or u > ONE + static_cast(1e-4)) { + continue; // outside the colorbar range + } + tk.emplace_back(std::min(ONE, std::max(ZERO, u)), v); + } + } + for (const auto& [u, val] : tk) { + const int ty = bar_y + + static_cast((ONE - u) * static_cast(bar_h - 1)); + // tick line + for (int i = 0; i < gap; ++i) { + for (int w = 0; w < std::max(1, s / 2); ++w) { + setPx(rgba, W, H, bar_x + bar_w + i, ty + w, tc, tc, tc); + } + } + // label, vertically centered on the tick + drawText(rgba, + W, + H, + bar_x + bar_w + gap + 2 * s, + ty - char_h / 2, + fmtNum(val), + s, + tc, + tc, + tc); + } + + // title above the bar, left-aligned to the bar so it stays in the strip + if (not label.empty()) { + int ty = bar_y - char_h - 2 * gap; + if (ty < 0) { + ty = 0; + } + drawText(rgba, W, H, bar_x, ty, label, s, tc, tc, tc); + } + } + +} // namespace out + +#endif // OUTPUT_RENDER_COLORBAR_H diff --git a/src/output/render/composite.h b/src/output/render/composite.h new file mode 100644 index 000000000..22145c9ac --- /dev/null +++ b/src/output/render/composite.h @@ -0,0 +1,546 @@ +/** + * @file output/render/composite.h + * @brief Front-to-back visibility ordering for the structured decomposition + * and the premultiplied "over" compositing operator. + * @implements + * - out::compositeOrderKey + * - out::overComposite + * @namespaces: + * - out:: + * @note + * entity decomposes the global box into a regular Dx x Dy x Dz grid of domains + * (domain index == MPI rank). For a camera viewing the box from outside, the + * correct global front-to-back order is a deterministic per-axis ordering by + * which side of each split plane the camera sits on -- no general depth sort, + * no cyclic overlap. Ordered premultiplied "over" of the non-overlapping, + * correctly-ordered per-domain segments reconstructs the single-image ray + * integral, hence is seamless. + */ + +#ifndef OUTPUT_RENDER_COMPOSITE_H +#define OUTPUT_RENDER_COMPOSITE_H + +#include "global.h" + +#include "utils/numeric.h" + +#include "output/render/renderer.h" + +#include +#include +#include +#include +#include + +namespace out { + + /** + * @brief Total-order sort key placing nearer domains first (front-to-back). + * @param offset integer grid coordinate of the domain (offset_ndomains) + * @param ndoms number of domains per axis (ndomains_per_dim) + * @param forward camera view direction (world == code axes for Minkowski) + * @return a single key; ascending key == front-to-back. Smaller is nearer. + * + * For axis d: if the camera looks toward +d (forward[d] >= 0), the smaller + * grid index is nearer, so key_d = offset_d. Otherwise key_d is reversed. + * The per-axis keys are packed lexicographically (axis 0 most significant). + */ + inline auto compositeOrderKey(const std::vector& offset, + const std::vector& ndoms, + const real_t forward[3]) -> uint64_t { + uint64_t key = 0; + for (size_t d = 0; d < ndoms.size(); ++d) { + const unsigned int Dd = ndoms[d]; + const unsigned int od = offset[d]; + const unsigned int kd = (forward[d] >= ZERO) ? od : (Dd - 1u - od); + key = key * static_cast(Dd) + static_cast(kd); + } + return key; + } + + /** + * @brief Accumulate one segment into a front-to-back running composite. + * @param acc 4-element premultiplied RGBA accumulator (modified in place) + * @param seg 4-element premultiplied RGBA of the next (further) segment + * + * acc holds everything in front of seg. The "over" operator: + * C_acc += (1 - A_acc) * C_seg ; A_acc += (1 - A_acc) * A_seg + * Associative with identity (0,0,0,0); segments must be supplied front first. + */ + inline void overComposite(real_t acc[4], const real_t seg[4]) { + const real_t one_minus_a = ONE - acc[3]; + acc[0] += one_minus_a * seg[0]; + acc[1] += one_minus_a * seg[1]; + acc[2] += one_minus_a * seg[2]; + acc[3] += one_minus_a * seg[3]; + } + + /** + * @brief Project a world point to a (fractional) screen pixel, inverting the + * ray-march kernel's ray generation. + * @return false if the point is behind a perspective camera (no projection) + */ + inline auto projectToScreen(const CameraDevice& cam, + int W, + int H, + const real_t p[3], + real_t& outx, + real_t& outy) -> bool { + const real_t dx = p[0] - cam.eye[0]; + const real_t dy = p[1] - cam.eye[1]; + const real_t dz = p[2] - cam.eye[2]; + const real_t cx = dx * cam.right[0] + dy * cam.right[1] + dz * cam.right[2]; + const real_t cy = dx * cam.up[0] + dy * cam.up[1] + dz * cam.up[2]; + real_t fx, fy; + if (cam.orthographic) { + fx = cx / cam.half_w; + fy = cy / cam.half_h; + } else { + const real_t cz = dx * cam.forward[0] + dy * cam.forward[1] + + dz * cam.forward[2]; + if (cz <= static_cast(1e-6)) { + return false; + } + fx = (cx / cz) / (cam.aspect * cam.tan_half_fov); + fy = (cy / cz) / cam.tan_half_fov; + } + outx = (fx + ONE) * HALF * static_cast(W) - HALF; + outy = (ONE - fy) * HALF * static_cast(H) - HALF; + return true; + } + + /** + * @brief Screen-space bounding box (in pixels) of a world-space AABB. + * @param lo,hi world AABB corners + * @param[out] bx0,by0,bw,bh clamped pixel bbox (top-left + size) + * @return false if the box projects to an empty on-screen region + * @note Falls back to the full frame if any corner is behind the camera. + */ + inline auto screenBBox(const CameraDevice& cam, + int W, + int H, + const real_t lo[3], + const real_t hi[3], + int& bx0, + int& by0, + int& bw, + int& bh) -> bool { + real_t minx = static_cast(1e30), miny = static_cast(1e30); + real_t maxx = static_cast(-1e30), maxy = static_cast(-1e30); + for (int c = 0; c < 8; ++c) { + const real_t p[3] = { (c & 1) ? hi[0] : lo[0], + (c & 2) ? hi[1] : lo[1], + (c & 4) ? hi[2] : lo[2] }; + real_t sx, sy; + if (not projectToScreen(cam, W, H, p, sx, sy)) { + bx0 = 0; + by0 = 0; + bw = W; + bh = H; + return true; // conservative fallback + } + minx = std::min(minx, sx); + maxx = std::max(maxx, sx); + miny = std::min(miny, sy); + maxy = std::max(maxy, sy); + } + const int pad = 2; + int x0 = static_cast(std::floor(minx)) - pad; + int x1 = static_cast(std::ceil(maxx)) + pad; + int y0 = static_cast(std::floor(miny)) - pad; + int y1 = static_cast(std::ceil(maxy)) + pad; + x0 = std::max(0, std::min(W, x0)); + x1 = std::max(0, std::min(W, x1)); + y0 = std::max(0, std::min(H, y0)); + y1 = std::max(0, std::min(H, y1)); + bx0 = x0; + by0 = y0; + bw = x1 - x0; + bh = y1 - y0; + return (bw > 0 and bh > 0); + } + + /** + * @brief Composite two sparse sub-images: `front` OVER `back`. + * @return a sub-image spanning the union of the two bounding boxes + * @note premultiplied "over": out = front + (1 - front.a) * back. Associative, + * so a tree of these reproduces the sequential front-to-back composite. + */ + inline auto overSub(const SubImage& f, const SubImage& b) -> SubImage { + if (f.w == 0 or f.h == 0) { + return b; + } + if (b.w == 0 or b.h == 0) { + return f; + } + const int ux0 = std::min(f.x0, b.x0); + const int uy0 = std::min(f.y0, b.y0); + const int ux1 = std::max(f.x0 + f.w, b.x0 + b.w); + const int uy1 = std::max(f.y0 + f.h, b.y0 + b.h); + SubImage r; + r.x0 = ux0; + r.y0 = uy0; + r.w = ux1 - ux0; + r.h = uy1 - uy0; + r.rgba.assign(static_cast(r.w) * r.h * 4, ZERO); + // place `back` + for (int y = 0; y < b.h; ++y) { + for (int x = 0; x < b.w; ++x) { + const size_t ri = (static_cast(b.y0 + y - uy0) * r.w + + (b.x0 + x - ux0)) * + 4; + const size_t bi = (static_cast(y) * b.w + x) * 4; + r.rgba[ri + 0] = b.rgba[bi + 0]; + r.rgba[ri + 1] = b.rgba[bi + 1]; + r.rgba[ri + 2] = b.rgba[bi + 2]; + r.rgba[ri + 3] = b.rgba[bi + 3]; + } + } + // `front` OVER the (back-filled) result + for (int y = 0; y < f.h; ++y) { + for (int x = 0; x < f.w; ++x) { + const size_t ri = (static_cast(f.y0 + y - uy0) * r.w + + (f.x0 + x - ux0)) * + 4; + const size_t fi = (static_cast(y) * f.w + x) * 4; + const real_t inv = ONE - f.rgba[fi + 3]; + r.rgba[ri + 0] = f.rgba[fi + 0] + inv * r.rgba[ri + 0]; + r.rgba[ri + 1] = f.rgba[fi + 1] + inv * r.rgba[ri + 1]; + r.rgba[ri + 2] = f.rgba[fi + 2] + inv * r.rgba[ri + 2]; + r.rgba[ri + 3] = f.rgba[fi + 3] + inv * r.rgba[ri + 3]; + } + } + return r; + } + + /* ====================================================================== */ + /* Interior-eye dome: fisheye projection + depth-resolved (A-buffer) */ + /* composite. Used when a single global front-to-back order does not */ + /* exist (camera inside the box). See renderer.h::FragImage. */ + /* ====================================================================== */ + + /** + * @brief Forward azimuthal-equidistant fisheye projection: world point -> the + * (fractional) dome pixel, inverting the dome ray generation. + * @return false if the point is outside the dome field of view. + * @note Uses the same ndc<->pixel convention as projectToScreen (so a square + * frame gives a centered disk of radius 1); the dome kernel forces aspect 1. + */ + inline auto projectToScreenDome(const CameraDevice& cam, + int W, + int H, + const real_t p[3], + real_t& outx, + real_t& outy) -> bool { + real_t v[3] = { p[0] - cam.eye[0], p[1] - cam.eye[1], p[2] - cam.eye[2] }; + const real_t n = std::sqrt(v[0] * v[0] + v[1] * v[1] + v[2] * v[2]); + if (n < static_cast(1e-20)) { + outx = HALF * static_cast(W) - HALF; // eye itself -> disk center + outy = HALF * static_cast(H) - HALF; + return true; + } + const real_t inv = ONE / n; + v[0] *= inv; + v[1] *= inv; + v[2] *= inv; + real_t cz = v[0] * cam.forward[0] + v[1] * cam.forward[1] + + v[2] * cam.forward[2]; + cz = (cz < -ONE) ? -ONE : ((cz > ONE) ? ONE : cz); + const real_t theta = std::acos(cz); + if (theta > cam.dome_half_fov) { + return false; // outside the dome FOV + } + const real_t r = theta / cam.dome_half_fov; // 0..1 image radius + const real_t cx = v[0] * cam.right[0] + v[1] * cam.right[1] + + v[2] * cam.right[2]; + const real_t cy = v[0] * cam.up[0] + v[1] * cam.up[1] + v[2] * cam.up[2]; + const real_t phi = std::atan2(cy, cx); + const real_t fx = r * std::cos(phi), fy = r * std::sin(phi); + outx = (fx + ONE) * HALF * static_cast(W) - HALF; + outy = (ONE - fy) * HALF * static_cast(H) - HALF; + return true; + } + + /** + * @brief Conservative screen-space bbox of a world AABB under the dome fisheye. + * @return false (empty) if the box is entirely outside the dome FOV. + * @note Never under-covers (that would drop fragments and reintroduce seams): + * - eye inside the (inclusive) AABB -> full frame + * - all edge samples outside the FOV -> empty + * - some in / some out (straddles the horizon)-> full frame + * - footprint contains the zenith or wraps the disk center (max angular gap + * between projected samples < pi) -> full frame + * - otherwise -> tight bbox of the samples + */ + inline auto screenBBoxDome(const CameraDevice& cam, + int W, + int H, + const real_t lo[3], + const real_t hi[3], + int& bx0, + int& by0, + int& bw, + int& bh) -> bool { + auto fullFrame = [&]() { + bx0 = 0; + by0 = 0; + bw = W; + bh = H; + return true; + }; + // eye inside the domain -> covers all azimuths + the zenith -> full frame + if (cam.eye[0] >= lo[0] and cam.eye[0] <= hi[0] and cam.eye[1] >= lo[1] and + cam.eye[1] <= hi[1] and cam.eye[2] >= lo[2] and cam.eye[2] <= hi[2]) { + return fullFrame(); + } + // the zenith ray (disk center) piercing this domain also means it covers the + // center -> full frame. A ray/AABB slab test from the eye along `forward` + // catches the case edge-sampling can miss (a slab pierced through a face, + // where the minr / azimuth-gap tests stay just under threshold -> a hole at + // the frame center). Only the forward half-line (t >= 0) is considered. + { + const real_t reps = static_cast(1e-12); + real_t te = ZERO, tx = static_cast(1e30); + bool hit = true; + for (int d = 0; d < 3; ++d) { + const real_t o = cam.eye[d], dd = cam.forward[d]; + if (dd > -reps and dd < reps) { + if (o < lo[d] or o > hi[d]) { + hit = false; + break; + } + } else { + real_t t1 = (lo[d] - o) / dd, t2 = (hi[d] - o) / dd; + if (t1 > t2) { + const real_t tmp = t1; + t1 = t2; + t2 = tmp; + } + te = (t1 > te) ? t1 : te; + tx = (t2 < tx) ? t2 : tx; + } + } + if (hit and te <= tx and tx >= ZERO) { + return fullFrame(); + } + } + real_t minx = static_cast(1e30), miny = static_cast(1e30); + real_t maxx = static_cast(-1e30), maxy = static_cast(-1e30); + real_t minr = static_cast(1e30); + int n_in = 0, n_out = 0; + // azimuths of in-FOV samples, for the "wraps the center" (largest-gap) test + std::vector phis; + phis.reserve(12ul * 17ul); + const int NS = 48; // samples per AABB edge (a straight edge maps to a + // curved fisheye arc, so sample densely to bound it) + auto addPoint = [&](const real_t p[3]) { + real_t sx, sy; + if (not projectToScreenDome(cam, W, H, p, sx, sy)) { + ++n_out; + return; + } + ++n_in; + minx = std::min(minx, sx); + maxx = std::max(maxx, sx); + miny = std::min(miny, sy); + maxy = std::max(maxy, sy); + const real_t fx = TWO * (sx + HALF) / static_cast(W) - ONE; + const real_t fy = ONE - TWO * (sy + HALF) / static_cast(H); + minr = std::min(minr, std::sqrt(fx * fx + fy * fy)); + phis.push_back(std::atan2(fy, fx)); + }; + // sample all 12 edges of the AABB + for (int axis = 0; axis < 3; ++axis) { + for (int c = 0; c < 4; ++c) { + // the two AABB axes perpendicular to `axis` are fixed to lo/hi per `c` + const int a1 = (axis + 1) % 3, a2 = (axis + 2) % 3; + real_t p[3]; + p[a1] = (c & 1) ? hi[a1] : lo[a1]; + p[a2] = (c & 2) ? hi[a2] : lo[a2]; + for (int s = 0; s <= NS; ++s) { + const real_t t = static_cast(s) / static_cast(NS); + p[axis] = lo[axis] + (hi[axis] - lo[axis]) * t; + addPoint(p); + } + } + } + if (n_in == 0) { + bw = 0; + bh = 0; + return false; // entirely outside the FOV + } + if (n_out > 0) { + return fullFrame(); // straddles the FOV boundary -> conservative + } + if (minr < static_cast(1e-3)) { + return fullFrame(); // footprint reaches the zenith (disk center) + } + // largest cyclic gap between azimuths: if < pi the samples wrap the center, + // so the axis-aligned bbox of the boundary would miss the interior. + std::sort(phis.begin(), phis.end()); + real_t maxgap = ZERO; + const real_t twopi = static_cast(2.0 * 3.14159265358979323846); + for (size_t i = 0; i + 1 < phis.size(); ++i) { + maxgap = std::max(maxgap, phis[i + 1] - phis[i]); + } + if (not phis.empty()) { + maxgap = std::max(maxgap, (phis.front() + twopi) - phis.back()); + } + // a small tolerance past pi keeps borderline wraps conservative + if (maxgap < static_cast(3.14159265358979323846 + 0.05)) { + return fullFrame(); + } + // pad generously: the fisheye arc between edge samples can bulge a few px + const int pad = 4; + int x0 = static_cast(std::floor(minx)) - pad; + int x1 = static_cast(std::ceil(maxx)) + pad; + int y0 = static_cast(std::floor(miny)) - pad; + int y1 = static_cast(std::ceil(maxy)) + pad; + x0 = std::max(0, std::min(W, x0)); + x1 = std::max(0, std::min(W, x1)); + y0 = std::max(0, std::min(H, y0)); + y1 = std::max(0, std::min(H, y1)); + bx0 = x0; + by0 = y0; + bw = x1 - x0; + bh = y1 - y0; + return (bw > 0 and bh > 0); + } + + /** + * @brief Collapse one pixel's depth-sorted fragment list [k0, k1) into a + * single premultiplied RGBA via front-to-back "over". + */ + inline void fragOver(const std::vector& depth, + const std::vector& rgba, + uint32_t k0, + uint32_t k1, + real_t out[4]) { + (void)depth; // fragments are already ascending in depth + real_t acc[4] = { ZERO, ZERO, ZERO, ZERO }; + for (uint32_t k = k0; k < k1; ++k) { + overComposite(acc, &rgba[static_cast(k) * 4]); + if (acc[3] >= ONE) { + break; + } + } + out[0] = acc[0]; + out[1] = acc[1]; + out[2] = acc[2]; + out[3] = acc[3]; + } + + /** + * @brief Merge two depth-sorted fragment images: union the bboxes and, per + * pixel, merge the two ascending fragment lists by depth, then drop fragments + * once the accumulated alpha reaches `cull_alpha` (exact when cull_alpha == + * 1: only provably-occluded fragments are removed, so the result is + * independent of how the tree is grouped -> associative + commutative). + */ + inline auto mergeFrag(const FragImage& a, const FragImage& b, real_t cull_alpha) + -> FragImage { + if (a.w == 0 or a.h == 0) { + return b; + } + if (b.w == 0 or b.h == 0) { + return a; + } + const int ux0 = std::min(a.x0, b.x0); + const int uy0 = std::min(a.y0, b.y0); + const int ux1 = std::max(a.x0 + a.w, b.x0 + b.w); + const int uy1 = std::max(a.y0 + a.h, b.y0 + b.h); + FragImage r; + r.x0 = ux0; + r.y0 = uy0; + r.w = ux1 - ux0; + r.h = uy1 - uy0; + const size_t np = static_cast(r.w) * r.h; + r.offs.assign(np + 1, 0u); + + // fetch a source image's fragment range at global pixel (gx, gy) + auto range = [](const FragImage& s, int gx, int gy, uint32_t& k0, uint32_t& k1) { + const int lx = gx - s.x0, ly = gy - s.y0; + if (lx < 0 or ly < 0 or lx >= s.w or ly >= s.h) { + k0 = 0; + k1 = 0; + return; + } + const size_t p = static_cast(ly) * s.w + lx; + k0 = s.offs[p]; + k1 = s.offs[p + 1]; + }; + + // pass 1: per-pixel surviving-fragment count (merge + occlusion cull) + for (int gy = uy0; gy < uy1; ++gy) { + for (int gx = ux0; gx < ux1; ++gx) { + uint32_t ak0, ak1, bk0, bk1; + range(a, gx, gy, ak0, ak1); + range(b, gx, gy, bk0, bk1); + uint32_t ia = ak0, ib = bk0, cnt = 0; + real_t A = ZERO; + while ((ia < ak1 or ib < bk1) and A < cull_alpha) { + const bool takeA = (ib >= bk1) or + (ia < ak1 and a.depth[ia] <= b.depth[ib]); + const real_t al = takeA ? a.rgba[static_cast(ia) * 4 + 3] + : b.rgba[static_cast(ib) * 4 + 3]; + A += (ONE - A) * al; + ++cnt; + if (takeA) { + ++ia; + } else { + ++ib; + } + } + const size_t pix = static_cast(gy - uy0) * r.w + (gx - ux0); + r.offs[pix + 1] = cnt; + } + } + // prefix-sum to offsets + for (size_t p = 0; p < np; ++p) { + r.offs[p + 1] += r.offs[p]; + } + const size_t nfrag = r.offs[np]; + r.depth.assign(nfrag, ZERO); + r.rgba.assign(nfrag * 4, ZERO); + + // pass 2: fill the merged, culled fragments + for (int gy = uy0; gy < uy1; ++gy) { + for (int gx = ux0; gx < ux1; ++gx) { + uint32_t ak0, ak1, bk0, bk1; + range(a, gx, gy, ak0, ak1); + range(b, gx, gy, bk0, bk1); + const size_t pix = static_cast(gy - uy0) * r.w + (gx - ux0); + uint32_t ia = ak0, ib = bk0, o = r.offs[pix]; + const uint32_t oend = r.offs[pix + 1]; + while (o < oend) { + const bool takeA = (ib >= bk1) or + (ia < ak1 and a.depth[ia] <= b.depth[ib]); + if (takeA) { + r.depth[o] = a.depth[ia]; + const size_t s = static_cast(ia) * 4; + const size_t d = static_cast(o) * 4; + r.rgba[d + 0] = a.rgba[s + 0]; + r.rgba[d + 1] = a.rgba[s + 1]; + r.rgba[d + 2] = a.rgba[s + 2]; + r.rgba[d + 3] = a.rgba[s + 3]; + ++ia; + } else { + r.depth[o] = b.depth[ib]; + const size_t s = static_cast(ib) * 4; + const size_t d = static_cast(o) * 4; + r.rgba[d + 0] = b.rgba[s + 0]; + r.rgba[d + 1] = b.rgba[s + 1]; + r.rgba[d + 2] = b.rgba[s + 2]; + r.rgba[d + 3] = b.rgba[s + 3]; + ++ib; + } + ++o; + } + } + } + return r; + } + +} // namespace out + +#endif // OUTPUT_RENDER_COMPOSITE_H diff --git a/src/output/render/fieldlines.h b/src/output/render/fieldlines.h new file mode 100644 index 000000000..4b9ce8fe3 --- /dev/null +++ b/src/output/render/fieldlines.h @@ -0,0 +1,868 @@ +/** + * @file output/render/fieldlines.h + * @brief Host-side magnetic-field-line tracer + bucketed tube-geometry builder + * @implements + * - out::CoarseField + * - out::traceFieldLines + * - out::buildTubeSet + * - out::emptyTubeSet + * @namespaces: + * - out:: + * @note + * Field lines are intrinsically non-local (a streamline wanders across MPI + * domains), which would normally demand parallel particle advection. We sidestep + * that entirely: the (physical-basis) field is volume-averaged onto a COARSE + * grid and MPI-replicated to every rank (see Metadomain::buildFieldLineTubes), + * so every rank traces the SAME global polylines locally and renders only the + * segments overlapping its own domain. The existing ordered cross-domain + * composite then stitches the pieces. Tracing/geometry here is metric-agnostic: + * the only supported 3D render mode is Cartesian (Minkowski), so the coarse grid + * is a plain uniform lattice in world coordinates. + * + * Performance: a ray sample must not test every segment. Segments are bucketed + * into the coarse grid (CSR), and since the tube radius is << one coarse cell, + * a sample only needs the segments registered in its own cell. The kernel + * (raymarch.hpp) walks that short bucket. + */ + +#ifndef OUTPUT_RENDER_FIELDLINES_H +#define OUTPUT_RENDER_FIELDLINES_H + +#include "global.h" + +#include "arch/kokkos_aliases.h" + +#include "output/render/renderer.h" +#include "output/render/transfer_fn.h" + +#include + +#include +#include +#include +#include +#include + +namespace out { + + /** + * @brief A coarse, MPI-replicated copy of the physical-basis vector field. + * @note B is laid out component-fastest: index (c0,c1,c2,comp) lives at + * ((c2*n1 + c1)*n0 + c0)*3 + comp, with c0 the fastest spatial axis. + */ + struct CoarseField { + std::vector B; // n0*n1*n2*3 + int n[3] { 0, 0, 0 }; + real_t origin[3] { ZERO, ZERO, ZERO }; + real_t dx[3] { ONE, ONE, ONE }; + }; + + /** @brief One traced field line: world-space vertices + per-vertex |field|. */ + struct Polyline { + std::vector> pts; + std::vector scal; + }; + + namespace fl_hidden { + + inline auto cellLinear(const CoarseField& cf, int c0, int c1, int c2) + -> size_t { + return (static_cast(c2) * cf.n[1] + c1) * cf.n[0] + c0; + } + + // trilinear sample of the coarse field at world point p -> B[3], |B|. + // Clamps to the grid (so a sample just outside a face still returns the + // edge value); membership in the global box is the caller's stop test. + inline auto sampleCoarse(const CoarseField& cf, const real_t p[3], real_t B[3]) + -> real_t { + int i0[3], i1[3]; + real_t fr[3]; + for (int d = 0; d < 3; ++d) { + if (cf.n[d] <= 1) { + i0[d] = 0; + i1[d] = 0; + fr[d] = ZERO; + continue; + } + const real_t g = (p[d] - cf.origin[d]) / cf.dx[d] - HALF; + real_t f = std::floor(g); + real_t t = g - f; + int b = static_cast(f); + if (b < 0) { + b = 0; + t = ZERO; + } else if (b > cf.n[d] - 2) { + b = cf.n[d] - 2; + t = ONE; + } + i0[d] = b; + i1[d] = b + 1; + fr[d] = t; + } + for (int comp = 0; comp < 3; ++comp) { + const real_t c000 = cf.B[cellLinear(cf, i0[0], i0[1], i0[2]) * 3 + comp]; + const real_t c100 = cf.B[cellLinear(cf, i1[0], i0[1], i0[2]) * 3 + comp]; + const real_t c010 = cf.B[cellLinear(cf, i0[0], i1[1], i0[2]) * 3 + comp]; + const real_t c110 = cf.B[cellLinear(cf, i1[0], i1[1], i0[2]) * 3 + comp]; + const real_t c001 = cf.B[cellLinear(cf, i0[0], i0[1], i1[2]) * 3 + comp]; + const real_t c101 = cf.B[cellLinear(cf, i1[0], i0[1], i1[2]) * 3 + comp]; + const real_t c011 = cf.B[cellLinear(cf, i0[0], i1[1], i1[2]) * 3 + comp]; + const real_t c111 = cf.B[cellLinear(cf, i1[0], i1[1], i1[2]) * 3 + comp]; + const real_t c00 = c000 * (ONE - fr[0]) + c100 * fr[0]; + const real_t c10 = c010 * (ONE - fr[0]) + c110 * fr[0]; + const real_t c01 = c001 * (ONE - fr[0]) + c101 * fr[0]; + const real_t c11 = c011 * (ONE - fr[0]) + c111 * fr[0]; + const real_t c0 = c00 * (ONE - fr[1]) + c10 * fr[1]; + const real_t c1 = c01 * (ONE - fr[1]) + c11 * fr[1]; + B[comp] = c0 * (ONE - fr[2]) + c1 * fr[2]; + } + return std::sqrt(B[0] * B[0] + B[1] * B[1] + B[2] * B[2]); + } + + inline auto insideBox(const CoarseField& cf, const real_t p[3]) -> bool { + for (int d = 0; d < 3; ++d) { + const real_t hi = cf.origin[d] + static_cast(cf.n[d]) * cf.dx[d]; + if (p[d] < cf.origin[d] or p[d] > hi) { + return false; + } + } + return true; + } + + } // namespace fl_hidden + + /** @brief A constant-color opaque LUT (monochrome field lines). */ + inline auto buildSolidLUT(real_t r, real_t g, real_t b, int n_lut) + -> array_t { + array_t lut { "fl_solid_lut", static_cast(n_lut) }; + auto h = Kokkos::create_mirror_view(lut); + for (int i = 0; i < n_lut; ++i) { + h(i, 0) = r; // opaque -> premultiplied == straight RGB + h(i, 1) = g; + h(i, 2) = b; + h(i, 3) = ONE; + } + Kokkos::deep_copy(lut, h); + return lut; + } + + /** @brief The field-line LUT: a single color if cfg.color is set, else by |B|. */ + inline auto buildLineLUT(const FieldLineConfig& cfg, int n_lut) + -> array_t { + if (cfg.color.size() == 3) { + return buildSolidLUT(cfg.color[0], cfg.color[1], cfg.color[2], n_lut); + } + return buildLUT(cfg.colormap, + n_lut, + { + { ZERO, ONE }, + { ONE, ONE } + }); + } + + /** + * @brief Trace field lines through the coarse field by bidirectional RK4. + * @param cf coarse, replicated physical field + * @param cfg field-line configuration (seed density, step, caps) + * @param world_per_pixel world units per screen pixel (sets seed/tube scale) + * @param[out] out_vmin,out_vmax auto color range (min/max |field| along lines) + * @return global polylines (identical on every rank) + */ + inline auto traceFieldLines(const CoarseField& cf, + const FieldLineConfig& cfg, + real_t world_per_pixel, + real_t& out_vmin, + real_t& out_vmax) -> std::vector { + using fl_hidden::insideBox; + using fl_hidden::sampleCoarse; + + std::vector lines; + if (cf.n[0] < 1 or cf.n[1] < 1 or cf.n[2] < 1) { + return lines; + } + + real_t size[3]; + real_t diag2 = ZERO; + real_t min_dx = static_cast(1e30); + for (int d = 0; d < 3; ++d) { + size[d] = static_cast(cf.n[d]) * cf.dx[d]; + diag2 += size[d] * size[d]; + min_dx = std::min(min_dx, cf.dx[d]); + } + const real_t box_diag = std::sqrt(diag2); + const real_t max_len = cfg.max_len_frac * box_diag; + const real_t h = std::max(cfg.step_frac, static_cast(1e-3)) * min_dx; + const real_t eps = static_cast(1e-20); + + // seed lattice: spacing ~ seed_px screen pixels, grown to respect seed_max + real_t spacing = std::max(cfg.seed_px, ONE) * world_per_pixel; + int ns[3]; + auto countSeeds = [&](real_t s) -> long { + long tot = 1; + for (int d = 0; d < 3; ++d) { + ns[d] = std::max(1, static_cast(std::floor(size[d] / s))); + tot *= ns[d]; + } + return tot; + }; + long n_seed = countSeeds(spacing); + if (n_seed > cfg.seed_max and cfg.seed_max > 0) { + const real_t grow = std::cbrt( + static_cast(n_seed) / static_cast(cfg.seed_max)); + spacing *= grow; + countSeeds(spacing); // recompute ns[] for the grown spacing + } + + // unit-vector field derivative (×dir) used by RK4; false if |B| ~ 0 + auto deriv = [&](const real_t p[3], real_t dir, real_t out[3]) -> bool { + real_t B[3]; + const real_t m = sampleCoarse(cf, p, B); + if (m < eps) { + return false; + } + const real_t inv = dir / m; + out[0] = B[0] * inv; + out[1] = B[1] * inv; + out[2] = B[2] * inv; + return true; + }; + + out_vmin = static_cast(1e30); + out_vmax = static_cast(-1e30); + auto track = [&](real_t m) { + out_vmin = std::min(out_vmin, m); + out_vmax = std::max(out_vmax, m); + }; + + // integrate one direction (dir = +1 forward, -1 backward) from a seed + auto integrate = [&](const real_t seed[3], real_t dir) { + Polyline pl; + real_t p[3] = { seed[0], seed[1], seed[2] }; + real_t B0[3]; + real_t m0 = sampleCoarse(cf, p, B0); + if (m0 < eps) { + return; + } + pl.pts.push_back({ p[0], p[1], p[2] }); + pl.scal.push_back(m0); + track(m0); + real_t len = ZERO; + for (int step = 0; step < cfg.max_steps and len < max_len; ++step) { + real_t k1[3], k2[3], k3[3], k4[3], q[3]; + if (not deriv(p, dir, k1)) { + break; + } + for (int d = 0; d < 3; ++d) { + q[d] = p[d] + HALF * h * k1[d]; + } + if (not deriv(q, dir, k2)) { + break; + } + for (int d = 0; d < 3; ++d) { + q[d] = p[d] + HALF * h * k2[d]; + } + if (not deriv(q, dir, k3)) { + break; + } + for (int d = 0; d < 3; ++d) { + q[d] = p[d] + h * k3[d]; + } + if (not deriv(q, dir, k4)) { + break; + } + for (int d = 0; d < 3; ++d) { + p[d] += (h / static_cast(6)) * + (k1[d] + static_cast(2) * k2[d] + + static_cast(2) * k3[d] + k4[d]); + } + if (not insideBox(cf, p)) { + break; + } + real_t B[3]; + const real_t m = sampleCoarse(cf, p, B); + pl.pts.push_back({ p[0], p[1], p[2] }); + pl.scal.push_back(m); + track(m); + len += h; + } + if (pl.pts.size() >= 2) { + lines.push_back(std::move(pl)); + } + }; + + for (int k = 0; k < ns[2]; ++k) { + for (int j = 0; j < ns[1]; ++j) { + for (int i = 0; i < ns[0]; ++i) { + const real_t seed[3] = { + cf.origin[0] + (static_cast(i) + HALF) * size[0] / + static_cast(ns[0]), + cf.origin[1] + (static_cast(j) + HALF) * size[1] / + static_cast(ns[1]), + cf.origin[2] + (static_cast(k) + HALF) * size[2] / + static_cast(ns[2]) + }; + integrate(seed, ONE); + integrate(seed, -ONE); + } + } + } + if (out_vmin > out_vmax) { // no lines traced + out_vmin = ZERO; + out_vmax = ONE; + } + return lines; + } + + /** + * @brief Bucket the field-line segments overlapping a domain AABB into a CSR + * grid index and pack them into device Views ready for the ray-march kernel. + * @param lines global polylines (every rank passes the same set) + * @param radius world-space tube radius (the ds floor is applied by the caller) + * @param cfg field-line configuration (colormap / log) + * @param vmin,vmax tube color range + * @param lo,hi this domain's world AABB (segments outside it are dropped) + * @param cf coarse grid geometry, reused as the bucket grid + * @param[out] n_kept number of segments kept for this domain (for logging) + */ + inline auto buildTubeSet(const std::vector& lines, + real_t radius, + const FieldLineConfig& cfg, + real_t vmin, + real_t vmax, + const real_t lo[3], + const real_t hi[3], + const CoarseField& cf, + size_t& n_kept) -> TubeSet { + TubeSet ts; + ts.radius = radius; + ts.vmin = vmin; + ts.vmax = (vmax > vmin) ? vmax : (vmin + ONE); + ts.log_scale = cfg.log_scale and (vmin > ZERO); + ts.colormap = cfg.colormap; + ts.n_lut = 256; + // Bucket grid: a LOCAL uniform lattice spanning only THIS domain's AABB + // (not the whole box), so cell_start stays O(local cells). A bucket cell is + // ~one coarse cell; the tube radius is far smaller, so a ray sample (always + // inside [lo,hi]) finds its nearest segment in its own bucket cell. + for (int d = 0; d < 3; ++d) { + ts.gdx[d] = (cf.dx[d] > ZERO) ? cf.dx[d] : ONE; + ts.gorigin[d] = lo[d]; + const real_t span = hi[d] - lo[d]; + ts.gnc[d] = std::max(1, static_cast(std::ceil(span / ts.gdx[d]))); + } + auto lin = [&](int c0, int c1, int c2) -> size_t { + return (static_cast(c2) * ts.gnc[1] + c1) * ts.gnc[0] + c0; + }; + // opaque LUT: a tube sample paints a solid color (alpha==1), by |B| or a + // single monochrome color when cfg.color is set + ts.lut = buildLineLUT(cfg, ts.n_lut); + + // 1) keep segments whose radius-padded AABB overlaps this domain AABB. + // (We keep the whole segment, not a clipped piece: the kernel only ever + // samples within this domain's slab, so a shared segment shows only in + // the correct domain's depth range -- no double-draw.) + std::vector> kept; + for (const auto& pl : lines) { + for (size_t i = 0; i + 1 < pl.pts.size(); ++i) { + const auto& a = pl.pts[i]; + const auto& b = pl.pts[i + 1]; + bool overlap = true; + for (int d = 0; d < 3; ++d) { + const real_t smin = std::min(a[d], b[d]) - radius; + const real_t smax = std::max(a[d], b[d]) + radius; + if (smax < lo[d] or smin > hi[d]) { + overlap = false; + break; + } + } + if (overlap) { + kept.push_back( + { a[0], a[1], a[2], b[0], b[1], b[2], pl.scal[i], pl.scal[i + 1] }); + } + } + } + n_kept = kept.size(); + ts.n_seg = static_cast(kept.size()); + const size_t ncell = static_cast(ts.gnc[0]) * ts.gnc[1] * ts.gnc[2]; + + // 2) CSR bucketing on the coarse grid: count, prefix-sum, scatter. Each + // segment is registered in every cell its radius-padded AABB overlaps. + auto cellOf = [&](real_t x, int d) -> int { + int c = static_cast(std::floor((x - ts.gorigin[d]) / ts.gdx[d])); + if (c < 0) { + c = 0; + } else if (c > ts.gnc[d] - 1) { + c = ts.gnc[d] - 1; + } + return c; + }; + auto cellRange = [&](const std::array& s, int d, int& c0, int& c1) { + const real_t smin = std::min(s[d], s[3 + d]) - radius; + const real_t smax = std::max(s[d], s[3 + d]) + radius; + c0 = cellOf(smin, d); + c1 = cellOf(smax, d); + }; + + std::vector count(ncell + 1, 0); + for (const auto& s : kept) { + int a0, a1, b0, b1, d0, d1; + cellRange(s, 0, a0, a1); + cellRange(s, 1, b0, b1); + cellRange(s, 2, d0, d1); + for (int c2 = d0; c2 <= d1; ++c2) { + for (int c1 = b0; c1 <= b1; ++c1) { + for (int c0 = a0; c0 <= a1; ++c0) { + ++count[lin(c0, c1, c2)]; + } + } + } + } + std::vector start(ncell + 1, 0); + for (size_t c = 0; c < ncell; ++c) { + start[c + 1] = start[c] + count[c]; + } + const size_t n_insert = static_cast(start[ncell]); + std::vector idx(n_insert, 0); + std::vector cursor(start.begin(), start.end()); // running write head + for (size_t si = 0; si < kept.size(); ++si) { + int a0, a1, b0, b1, d0, d1; + cellRange(kept[si], 0, a0, a1); + cellRange(kept[si], 1, b0, b1); + cellRange(kept[si], 2, d0, d1); + for (int c2 = d0; c2 <= d1; ++c2) { + for (int c1 = b0; c1 <= b1; ++c1) { + for (int c0 = a0; c0 <= a1; ++c0) { + const size_t cl = lin(c0, c1, c2); + idx[static_cast(cursor[cl]++)] = static_cast(si); + } + } + } + } + + // 3) upload to device + ts.seg = array_t("fl_seg", static_cast(ts.n_seg)); + if (ts.n_seg > 0) { + auto seg_h = Kokkos::create_mirror_view(ts.seg); + for (int s = 0; s < ts.n_seg; ++s) { + for (int c = 0; c < 8; ++c) { + seg_h(s, c) = kept[static_cast(s)][static_cast(c)]; + } + } + Kokkos::deep_copy(ts.seg, seg_h); + } + ts.cell_start = array_t("fl_cell_start", ncell + 1); + { + auto h = Kokkos::create_mirror_view(ts.cell_start); + for (size_t c = 0; c <= ncell; ++c) { + h(c) = start[c]; + } + Kokkos::deep_copy(ts.cell_start, h); + } + ts.seg_idx = array_t("fl_seg_idx", std::max(n_insert, 1)); + if (n_insert > 0) { + auto h = Kokkos::create_mirror_view(ts.seg_idx); + for (size_t k = 0; k < n_insert; ++k) { + h(k) = idx[k]; + } + Kokkos::deep_copy(ts.seg_idx, h); + } + return ts; + } + + /** @brief A valid but empty tube set (used when a scene shows no field lines). */ + inline auto emptyTubeSet() -> TubeSet { + TubeSet ts; + ts.n_seg = 0; + ts.seg = array_t("fl_seg_empty", 0); + ts.cell_start = array_t("fl_cell_start_empty", 1); + ts.seg_idx = array_t("fl_seg_idx_empty", 1); + ts.lut = buildLUT("inferno", + 2, + { + { ZERO, ONE }, + { ONE, ONE } + }); + return ts; + } + + // ====================================================================== // + // 2D field lines == contours of the flux function psi // + // ====================================================================== // + + /** + * @brief A coarse, MPI-replicated copy of the in-plane (Bx, By) field. + * @note Component-fastest: (c0,c1,comp) lives at (c1*n0 + c0)*2 + comp. + */ + struct CoarseField2D { + std::vector B; // n0*n1*2 + int n[2] { 0, 0 }; + real_t origin[2] { ZERO, ZERO }; + real_t dx[2] { ONE, ONE }; + }; + + /** + * @brief Integrate the flux function psi(x,y) from the coarse in-plane field. + * @note psi obeys Bx = d psi/dy, By = -d psi/dx; the trapezoidal cumulative + * integral (one pass along x at j=0, then up each column) is path-consistent + * up to div(B) = 0. Because cf is globally replicated, every rank gets the + * SAME psi -> contour levels are identical everywhere -> seamless lines. + * @param[out] psi (n0*n1) flux function, c0-fastest + * @param[out] psi_min,psi_max flux range (for level spacing) + * @param[out] bmin,bmax |B| range (for contour coloring) + */ + inline void computeFlux2D(const CoarseField2D& cf, + std::vector& psi, + real_t& psi_min, + real_t& psi_max, + real_t& bmin, + real_t& bmax) { + const int nx = cf.n[0], ny = cf.n[1]; + psi.assign(static_cast(nx) * ny, ZERO); + auto B = [&](int i, int j, int c) -> real_t { + return cf.B[(static_cast(j) * nx + i) * 2 + c]; + }; + auto P = [&](int i, int j) -> real_t& { + return psi[static_cast(j) * nx + i]; + }; + // bottom row (j = 0): d psi/dx = -By, trapezoidal in x + for (int i = 1; i < nx; ++i) { + P(i, 0) = P(i - 1, 0) - HALF * (B(i - 1, 0, 1) + B(i, 0, 1)) * cf.dx[0]; + } + // each column: d psi/dy = +Bx, trapezoidal in y + for (int i = 0; i < nx; ++i) { + for (int j = 1; j < ny; ++j) { + P(i, j) = P(i, j - 1) + HALF * (B(i, j - 1, 0) + B(i, j, 0)) * cf.dx[1]; + } + } + psi_min = static_cast(1e30); + psi_max = static_cast(-1e30); + bmin = static_cast(1e30); + bmax = static_cast(-1e30); + for (int j = 0; j < ny; ++j) { + for (int i = 0; i < nx; ++i) { + const real_t p = P(i, j); + psi_min = std::min(psi_min, p); + psi_max = std::max(psi_max, p); + const real_t bx = B(i, j, 0), by = B(i, j, 1); + const real_t b = std::sqrt(bx * bx + by * by); + bmin = std::min(bmin, b); + bmax = std::max(bmax, b); + } + } + if (psi_min > psi_max) { + psi_min = ZERO; + psi_max = ONE; + } + if (bmin > bmax) { + bmin = ZERO; + bmax = ONE; + } + } + + /** + * @brief Pack the flux function + contour parameters into a device ContourSet. + * @param line_half_px half the contour line width, in screen pixels + * @param wpp world units per screen pixel (sets the screen-space line width) + */ + inline auto buildContourSet(const CoarseField2D& cf, + const std::vector& psi, + real_t psi_min, + real_t psi_max, + real_t bmin, + real_t bmax, + const FieldLineConfig& cfg, + real_t line_half_px, + real_t wpp) -> ContourSet { + ContourSet cs; + cs.n0 = cf.n[0]; + cs.n1 = cf.n[1]; + cs.origin0 = cf.origin[0]; + cs.origin1 = cf.origin[1]; + cs.dx0 = cf.dx[0]; + cs.dx1 = cf.dx[1]; + const int nlev = std::max(1, cfg.levels); + cs.dlevel = (psi_max > psi_min) + ? (psi_max - psi_min) / static_cast(nlev) + : ONE; + cs.psi_ref = psi_min; + cs.line_half_px = line_half_px; + cs.wpp = wpp; + cs.colormap = cfg.colormap; + cs.n_lut = 256; + real_t vlo = bmin, vhi = bmax; + if (cfg.vmax > cfg.vmin) { // explicit |B| color range overrides auto + vlo = cfg.vmin; + vhi = cfg.vmax; + } + cs.vmin = vlo; + cs.vmax = (vhi > vlo) ? vhi : (vlo + ONE); + cs.lut = buildLineLUT(cfg, cs.n_lut); // by |B| or monochrome (cfg.color) + const size_t n = static_cast(cs.n0) * cs.n1; + cs.psi = array_t("fl_psi", std::max(n, 1)); + if (n > 0) { + auto h = Kokkos::create_mirror_view(cs.psi); + for (size_t k = 0; k < n; ++k) { + h(k) = psi[k]; + } + Kokkos::deep_copy(cs.psi, h); + } + cs.enabled = true; + return cs; + } + + /** @brief A valid but empty contour set (scene shows no 2D field lines). */ + inline auto emptyContourSet() -> ContourSet { + ContourSet cs; + cs.enabled = false; + cs.psi = array_t("fl_psi_empty", 1); + cs.lut = buildLUT("inferno", + 2, + { + { ZERO, ONE }, + { ONE, ONE } + }); + return cs; + } + + // ====================================================================== // + // 2D spherical / Kerr-Schild == traced meridional streamlines (nt2py) // + // ====================================================================== // + + namespace fl_hidden { + // bilinear sample of the (Br, Btheta) coarse field at physical (r, theta); + // false if (r, theta) lies outside the grid (the integrator stops there). + inline auto sampleRTh(const CoarseField2D& cf, real_t r, real_t th, real_t B[2]) + -> bool { + const real_t rmin = cf.origin[0], thmin = cf.origin[1]; + const real_t rmax = rmin + static_cast(cf.n[0]) * cf.dx[0]; + const real_t thmax = thmin + static_cast(cf.n[1]) * cf.dx[1]; + const real_t tr = HALF * cf.dx[0], tt = HALF * cf.dx[1]; + if (r < rmin - tr or r > rmax + tr or th < thmin - tt or th > thmax + tt) { + return false; + } + int i0, i1, j0, j1; + real_t a0 = ZERO, a1 = ZERO; + if (cf.n[0] <= 1) { + i0 = 0; + i1 = 0; + } else { + const real_t g = (r - rmin) / cf.dx[0] - HALF; + const real_t f = std::floor(g); + int b = static_cast(f); + a0 = g - f; + if (b < 0) { + b = 0; + a0 = ZERO; + } else if (b > cf.n[0] - 2) { + b = cf.n[0] - 2; + a0 = ONE; + } + i0 = b; + i1 = b + 1; + } + if (cf.n[1] <= 1) { + j0 = 0; + j1 = 0; + } else { + const real_t g = (th - thmin) / cf.dx[1] - HALF; + const real_t f = std::floor(g); + int b = static_cast(f); + a1 = g - f; + if (b < 0) { + b = 0; + a1 = ZERO; + } else if (b > cf.n[1] - 2) { + b = cf.n[1] - 2; + a1 = ONE; + } + j0 = b; + j1 = b + 1; + } + for (int c = 0; c < 2; ++c) { + const real_t c00 = cf.B[(static_cast(j0) * cf.n[0] + i0) * 2 + c]; + const real_t c10 = cf.B[(static_cast(j0) * cf.n[0] + i1) * 2 + c]; + const real_t c01 = cf.B[(static_cast(j1) * cf.n[0] + i0) * 2 + c]; + const real_t c11 = cf.B[(static_cast(j1) * cf.n[0] + i1) * 2 + c]; + const real_t e0 = c00 * (ONE - a0) + c10 * a0; + const real_t e1 = c01 * (ONE - a0) + c11 * a0; + B[c] = e0 * (ONE - a1) + e1 * a1; + } + return true; + } + } // namespace fl_hidden + + /** + * @brief Trace poloidal field lines in the meridional (X, Z) plane (nt2py + * style): integrate (Fx, Fz) = (Br sin th + Bth cos th, Br cos th - Bth sin + * th) by bidirectional RK4 through the coarse (r, theta) field. Polylines are + * returned in meridional world coords (z = 0 so they reuse the 3D tube + * builder); with `mirror` the X<0 half is added as the theta-reflected copy. + * @param cf coarse field: component 0 = Br, 1 = Btheta, grid in (r, theta) + * @param world_per_pixel meridional world units per pixel (seed/length scale) + */ + inline auto traceFieldLinesMeridional(const CoarseField2D& cf, + const FieldLineConfig& cfg, + real_t world_per_pixel, + bool mirror, + real_t& out_vmin, + real_t& out_vmax) -> std::vector { + using fl_hidden::sampleRTh; + std::vector lines; + if (cf.n[0] < 1 or cf.n[1] < 1) { + return lines; + } + const real_t rmin = cf.origin[0], thmin = cf.origin[1]; + const real_t rmax = rmin + static_cast(cf.n[0]) * cf.dx[0]; + const real_t thmax = thmin + static_cast(cf.n[1]) * cf.dx[1]; + const real_t h = std::max(cfg.step_frac, static_cast(1e-3)) * + cf.dx[0]; // step ~ a coarse dr (a length) + const real_t max_len = cfg.max_len_frac * rmax * static_cast(2); + const real_t eps = static_cast(1e-20); + + auto bmag = [&](real_t X, real_t Z) -> real_t { + const real_t r = std::sqrt(X * X + Z * Z); + const real_t th = std::atan2(std::abs(X), Z); + real_t B[2]; + if (not sampleRTh(cf, r, th, B)) { + return ZERO; + } + return std::sqrt(B[0] * B[0] + B[1] * B[1]); + }; + // unit meridional direction (x dir); false if |F| ~ 0 or outside the grid + auto deriv = [&](const real_t p[2], real_t dir, real_t out[2]) -> bool { + const real_t X = p[0], Z = p[1]; + const real_t r = std::sqrt(X * X + Z * Z); + const real_t th = std::atan2(std::abs(X), Z); + real_t B[2]; + if (not sampleRTh(cf, r, th, B)) { + return false; + } + const real_t st = std::sin(th), ct = std::cos(th); + // (Br, Bth) -> meridional Cartesian; sign of the X-component follows X so + // a line seeded in X>=0 stays in X>=0 (the X<0 half is the mirror image) + const real_t sgn = (X < ZERO) ? -ONE : ONE; + const real_t Fx = sgn * (B[0] * st + B[1] * ct); + const real_t Fz = B[0] * ct - B[1] * st; + const real_t m = std::sqrt(Fx * Fx + Fz * Fz); + if (m < eps) { + return false; + } + const real_t inv = dir / m; + out[0] = Fx * inv; + out[1] = Fz * inv; + return true; + }; + + out_vmin = static_cast(1e30); + out_vmax = static_cast(-1e30); + auto track = [&](real_t m) { + out_vmin = std::min(out_vmin, m); + out_vmax = std::max(out_vmax, m); + }; + auto inDomain = [&](real_t X, real_t Z) -> bool { + const real_t r = std::sqrt(X * X + Z * Z); + const real_t th = std::atan2(std::abs(X), Z); + return (r >= rmin and r <= rmax and th >= thmin and th <= thmax); + }; + + auto integrate = [&](const real_t seed[2], real_t dir) { + Polyline pl; + real_t p[2] = { seed[0], seed[1] }; + real_t m0 = bmag(p[0], p[1]); + if (m0 < eps) { + return; + } + pl.pts.push_back({ p[0], p[1], ZERO }); + pl.scal.push_back(m0); + track(m0); + real_t len = ZERO; + for (int step = 0; step < cfg.max_steps and len < max_len; ++step) { + real_t k1[2], k2[2], k3[2], k4[2], q[2]; + if (not deriv(p, dir, k1)) { + break; + } + q[0] = p[0] + HALF * h * k1[0]; + q[1] = p[1] + HALF * h * k1[1]; + if (not deriv(q, dir, k2)) { + break; + } + q[0] = p[0] + HALF * h * k2[0]; + q[1] = p[1] + HALF * h * k2[1]; + if (not deriv(q, dir, k3)) { + break; + } + q[0] = p[0] + h * k3[0]; + q[1] = p[1] + h * k3[1]; + if (not deriv(q, dir, k4)) { + break; + } + p[0] += (h / static_cast(6)) * + (k1[0] + static_cast(2) * k2[0] + + static_cast(2) * k3[0] + k4[0]); + p[1] += (h / static_cast(6)) * + (k1[1] + static_cast(2) * k2[1] + + static_cast(2) * k3[1] + k4[1]); + if (not inDomain(p[0], p[1])) { + break; + } + const real_t m = bmag(p[0], p[1]); + pl.pts.push_back({ p[0], p[1], ZERO }); + pl.scal.push_back(m); + track(m); + len += h; + } + if (pl.pts.size() >= 2) { + lines.push_back(std::move(pl)); + } + }; + + // seed lattice over the X>=0 meridional half, keeping in-domain seeds + const real_t Xhi = rmax, Zlo = -rmax, Zhi = rmax; + real_t spacing = std::max(cfg.seed_px, ONE) * world_per_pixel; + auto gridCount = [&](real_t s) -> long { + const long nx = std::max(1L, static_cast(std::floor(Xhi / s))); + const long nz = std::max(1L, static_cast(std::floor((Zhi - Zlo) / s))); + return nx * nz; + }; + if (gridCount(spacing) > cfg.seed_max and cfg.seed_max > 0) { + spacing *= std::sqrt(static_cast(gridCount(spacing)) / + static_cast(cfg.seed_max)); + } + const long nx = std::max(1L, static_cast(std::floor(Xhi / spacing))); + const long nz = std::max(1L, + static_cast(std::floor((Zhi - Zlo) / spacing))); + for (long iz = 0; iz < nz; ++iz) { + for (long ix = 0; ix < nx; ++ix) { + const real_t X = (static_cast(ix) + HALF) * Xhi / + static_cast(nx); + const real_t Z = Zlo + (static_cast(iz) + HALF) * (Zhi - Zlo) / + static_cast(nz); + if (not inDomain(X, Z)) { + continue; + } + const real_t seed[2] = { X, Z }; + integrate(seed, ONE); + integrate(seed, -ONE); + } + } + if (out_vmin > out_vmax) { + out_vmin = ZERO; + out_vmax = ONE; + } + // mirror the traced (X>=0) lines into the X<0 half for a full disk + if (mirror) { + const size_t n0 = lines.size(); + for (size_t i = 0; i < n0; ++i) { + Polyline m = lines[i]; + for (auto& q : m.pts) { + q[0] = -q[0]; + } + lines.push_back(std::move(m)); + } + } + return lines; + } + +} // namespace out + +#endif // OUTPUT_RENDER_FIELDLINES_H diff --git a/src/output/render/png.h b/src/output/render/png.h new file mode 100644 index 000000000..7eba41a44 --- /dev/null +++ b/src/output/render/png.h @@ -0,0 +1,174 @@ +/** + * @file output/render/png.h + * @brief Self-contained, dependency-free PNG (8-bit RGBA) encoder + * @implements + * - out::write_png + * @namespaces: + * - out:: + * @note + * Header-only. Emits a valid PNG using stored (uncompressed) DEFLATE blocks + * wrapped in a zlib stream, with per-scanline filter type 0 (None). This keeps + * the encoder tiny and provably correct at the cost of compression ratio; the + * resulting files are still orders of magnitude smaller than the full-field + * dumps the renderer is meant to replace. A drop-in stronger encoder (e.g. a + * fixed-Huffman DEFLATE, or vendored stb_image_write) can replace the IDAT + * producer without touching callers. + */ + +#ifndef OUTPUT_RENDER_PNG_H +#define OUTPUT_RENDER_PNG_H + +#include "global.h" + +#include +#include +#include + +namespace out { + + namespace png_hidden { + + inline auto crc32(const uint8_t* data, size_t len) -> uint32_t { + static uint32_t table[256]; + static bool ready = false; + if (not ready) { + for (uint32_t n = 0; n < 256; ++n) { + uint32_t c = n; + for (int k = 0; k < 8; ++k) { + c = (c & 1u) ? (0xEDB88320u ^ (c >> 1)) : (c >> 1); + } + table[n] = c; + } + ready = true; + } + uint32_t c = 0xFFFFFFFFu; + for (size_t i = 0; i < len; ++i) { + c = table[(c ^ data[i]) & 0xFFu] ^ (c >> 8); + } + return c ^ 0xFFFFFFFFu; + } + + inline auto adler32(const uint8_t* data, size_t len) -> uint32_t { + constexpr uint32_t MOD = 65521u; + uint32_t a = 1u, b = 0u; + for (size_t i = 0; i < len; ++i) { + a = (a + data[i]) % MOD; + b = (b + a) % MOD; + } + return (b << 16) | a; + } + + inline void put_u32_be(std::vector& v, uint32_t x) { + v.push_back(static_cast((x >> 24) & 0xFFu)); + v.push_back(static_cast((x >> 16) & 0xFFu)); + v.push_back(static_cast((x >> 8) & 0xFFu)); + v.push_back(static_cast(x & 0xFFu)); + } + + inline void write_chunk(std::vector& out, + const char type[4], + const std::vector& data) { + put_u32_be(out, static_cast(data.size())); + std::vector typed_data; + typed_data.reserve(4 + data.size()); + for (int i = 0; i < 4; ++i) { + typed_data.push_back(static_cast(type[i])); + } + typed_data.insert(typed_data.end(), data.begin(), data.end()); + out.insert(out.end(), typed_data.begin(), typed_data.end()); + put_u32_be(out, crc32(typed_data.data(), typed_data.size())); + } + + // zlib stream wrapping `raw` in stored (BTYPE=00) DEFLATE blocks + inline auto zlib_store(const std::vector& raw) + -> std::vector { + std::vector z; + z.push_back(0x78); // CMF: CM=8, CINFO=7 + z.push_back(0x01); // FLG: makes (CMF<<8 | FLG) % 31 == 0, no dict, level 0 + size_t off = 0; + const size_t n = raw.size(); + constexpr size_t BLOCK = 65535u; + if (n == 0) { + z.push_back(0x01); // final, stored + z.push_back(0x00); + z.push_back(0x00); + z.push_back(0xFF); + z.push_back(0xFF); + } + while (off < n) { + const size_t len = (n - off > BLOCK) ? BLOCK : (n - off); + const bool final = (off + len >= n); + z.push_back(final ? 0x01 : 0x00); + const uint16_t l = static_cast(len); + const uint16_t nl = static_cast(~l); + z.push_back(static_cast(l & 0xFFu)); + z.push_back(static_cast((l >> 8) & 0xFFu)); + z.push_back(static_cast(nl & 0xFFu)); + z.push_back(static_cast((nl >> 8) & 0xFFu)); + z.insert(z.end(), + raw.begin() + static_cast(off), + raw.begin() + static_cast(off + len)); + off += len; + } + put_u32_be(z, adler32(raw.data(), raw.size())); + return z; + } + + } // namespace png_hidden + + /** + * @brief Write an 8-bit RGBA buffer to a PNG file. + * @param path output file path + * @param width image width in pixels + * @param height image height in pixels + * @param rgba pointer to width*height*4 bytes, row-major, top-left origin + * @return true on success + */ + inline auto write_png(const path_t& path, int width, int height, const uint8_t* rgba) + -> bool { + using namespace png_hidden; + const size_t w = static_cast(width); + const size_t h = static_cast(height); + // build filtered raw scanlines: each row prefixed with filter byte 0 (None) + std::vector raw; + raw.reserve(h * (1 + w * 4)); + for (size_t y = 0u; y < h; ++y) { + raw.push_back(0x00); + const uint8_t* row = rgba + y * w * 4; + raw.insert(raw.end(), row, row + w * 4); + } + + std::vector file; + // PNG signature + const uint8_t sig[8] = { 137, 80, 78, 71, 13, 10, 26, 10 }; + file.insert(file.end(), sig, sig + 8); + + // IHDR + std::vector ihdr; + put_u32_be(ihdr, static_cast(width)); + put_u32_be(ihdr, static_cast(height)); + ihdr.push_back(8); // bit depth + ihdr.push_back(6); // color type: RGBA + ihdr.push_back(0); // compression + ihdr.push_back(0); // filter + ihdr.push_back(0); // interlace + write_chunk(file, "IHDR", ihdr); + + // IDAT + write_chunk(file, "IDAT", zlib_store(raw)); + + // IEND + write_chunk(file, "IEND", {}); + + std::ofstream f(path, std::ios::binary); + if (not f.good()) { + return false; + } + f.write(reinterpret_cast(file.data()), + static_cast(file.size())); + return f.good(); + } + +} // namespace out + +#endif // OUTPUT_RENDER_PNG_H diff --git a/src/output/render/raymarch.hpp b/src/output/render/raymarch.hpp new file mode 100644 index 000000000..de09ff4f5 --- /dev/null +++ b/src/output/render/raymarch.hpp @@ -0,0 +1,529 @@ +/** + * @file output/render/raymarch.hpp + * @brief Header-only Kokkos volume ray-march kernel (one parallel_for over pixels) + * @implements + * - render::VolumeRayMarch_kernel + * @namespaces: + * - render:: + * @note + * Pure Kokkos: the only device entities are Views, the (trivially-copyable) + * metric, and the POD camera. Runs in Kokkos::DefaultExecutionSpace, inheriting + * whatever backend entity was built with (HIP / CUDA / SYCL / OpenMP). + * + * Seamlessness: every rank marches at GLOBAL sample positions t_k = k*ds + * measured from the shared camera eye, with a FIXED world-space step `ds` + * identical on all ranks. Each global sample therefore lands in exactly one + * domain (half-open membership via the per-domain slab interval [t_enter, + * t_exit)), so the ordered cross-domain "over" composite reproduces the single + * full-ray integral exactly. Trilinear sampling reads into the 1-cell ghost + * halo entity already exchanges, so the per-rank field is C0-continuous up to + * the shared face. + */ + +#ifndef OUTPUT_RENDER_RAYMARCH_HPP +#define OUTPUT_RENDER_RAYMARCH_HPP + +#include "global.h" + +#include "arch/kokkos_aliases.h" + +#include "output/render/renderer.h" + +namespace render { + using namespace ntt; + + template + class VolumeRayMarch_kernel { + static constexpr auto D = M::Dim; + + randacc_ndfield_t Fld; + const idx_t comp; + const M metric; + const out::CameraDevice cam; + + // local-domain world AABB and View extents (for index clamping) + const real_t lo0, lo1, lo2, hi0, hi1, hi2; + const int ext0, ext1, ext2; + + const int W, H; // full frame size (for ray generation / ndc) + const int bx0, by0, bw; // screen-bbox offset and width (output stride) + const real_t ds; // fixed world step (global, identical on all ranks) + const int max_steps; // safety cap on the marching loop + + // transfer function + array_t lut; + const int n_lut; + const real_t vmin, vmax; + const bool log_scale; + const real_t early_alpha; + + // global box wireframe "spine": opaque (alpha 1) segments composited inline + // during the march, so the volume occludes the far edges. radius <= 0 off. + const real_t glo0, glo1, glo2, ghi0, ghi1, ghi2; + const real_t spine_radius; + const real_t spine_cr, spine_cg, spine_cb; + + // magnetic-field-line tubes: opaque capsules colored by |field|, composited + // inline like the spine. Bucketed into the coarse grid (CSR) so a sample + // tests only the segments in its cell. tn_seg <= 0 disables them. + array_t tseg; // (tn_seg, 8): p0, p1, s0, s1 (world) + array_t tcell_start; // (ncell+1) CSR offsets + array_t tseg_idx; // segment indices grouped by cell + const int tn_seg; + const real_t tube_r2; // squared tube radius + const int tgnc0, tgnc1, tgnc2; + const real_t tg0, tg1, tg2; // bucket-grid origin + const real_t tdx0, tdx1, tdx2; // bucket-grid cell size + array_t tube_lut; + const int tube_n_lut; + const real_t tube_vmin, tube_vmax; + const bool tube_log; + // when false the scalar volume is not sampled (standalone field-line render) + const bool volume_enabled; + + array_t image; // output, (bw*bh, 4) premultiplied RGBA + // per-pixel front depth (ray t at domain entry) for the interior-eye dome + // A-buffer composite; INF where the pixel produced no fragment. Written for + // every projection (the external composite simply ignores it). + array_t depth_img; + + public: + VolumeRayMarch_kernel(const randacc_ndfield_t& Fld_, + idx_t comp_, + const M& metric_, + const out::CameraDevice& cam_, + const real_t lo[3], + const real_t hi[3], + int ext0_, + int ext1_, + int ext2_, + int W_, + int H_, + int bx0_, + int by0_, + int bw_, + real_t ds_, + int max_steps_, + const array_t& lut_, + int n_lut_, + real_t vmin_, + real_t vmax_, + bool log_scale_, + real_t early_alpha_, + const real_t glo[3], + const real_t ghi[3], + real_t spine_radius_, + const real_t spine_rgb[3], + const out::TubeSet& tubes_, + bool volume_enabled_, + const array_t& image_, + const array_t& depth_img_) + : Fld { Fld_ } + , comp { comp_ } + , metric { metric_ } + , cam { cam_ } + , lo0 { lo[0] } + , lo1 { lo[1] } + , lo2 { lo[2] } + , hi0 { hi[0] } + , hi1 { hi[1] } + , hi2 { hi[2] } + , ext0 { ext0_ } + , ext1 { ext1_ } + , ext2 { ext2_ } + , W { W_ } + , H { H_ } + , bx0 { bx0_ } + , by0 { by0_ } + , bw { bw_ } + , ds { ds_ } + , max_steps { max_steps_ } + , lut { lut_ } + , n_lut { n_lut_ } + , vmin { vmin_ } + , vmax { vmax_ } + , log_scale { log_scale_ } + , early_alpha { early_alpha_ } + , glo0 { glo[0] } + , glo1 { glo[1] } + , glo2 { glo[2] } + , ghi0 { ghi[0] } + , ghi1 { ghi[1] } + , ghi2 { ghi[2] } + , spine_radius { spine_radius_ } + , spine_cr { spine_rgb[0] } + , spine_cg { spine_rgb[1] } + , spine_cb { spine_rgb[2] } + , tseg { tubes_.seg } + , tcell_start { tubes_.cell_start } + , tseg_idx { tubes_.seg_idx } + , tn_seg { tubes_.n_seg } + , tube_r2 { tubes_.radius * tubes_.radius } + , tgnc0 { tubes_.gnc[0] } + , tgnc1 { tubes_.gnc[1] } + , tgnc2 { tubes_.gnc[2] } + , tg0 { tubes_.gorigin[0] } + , tg1 { tubes_.gorigin[1] } + , tg2 { tubes_.gorigin[2] } + , tdx0 { tubes_.gdx[0] } + , tdx1 { tubes_.gdx[1] } + , tdx2 { tubes_.gdx[2] } + , tube_lut { tubes_.lut } + , tube_n_lut { tubes_.n_lut } + , tube_vmin { tubes_.vmin } + , tube_vmax { tubes_.vmax } + , tube_log { tubes_.log_scale } + , volume_enabled { volume_enabled_ } + , image { image_ } + , depth_img { depth_img_ } {} + + // distance test: is world point (px,py,pz) within `spine_radius` of any of + // the 12 global-box edges? (nearest parallel edge per axis == nearest + // perpendicular-plane corner) + Inline auto onSpine(real_t px, real_t py, real_t pz) const -> bool { + if (spine_radius <= ZERO) { + return false; + } + const real_t r2 = spine_radius * spine_radius; + const real_t pad = spine_radius; + // edges parallel to x (perp plane = y,z) + if (px >= glo0 - pad and px <= ghi0 + pad) { + const real_t dy = math::min(math::abs(py - glo1), math::abs(py - ghi1)); + const real_t dz = math::min(math::abs(pz - glo2), math::abs(pz - ghi2)); + if (dy * dy + dz * dz < r2) { + return true; + } + } + // edges parallel to y (perp plane = x,z) + if (py >= glo1 - pad and py <= ghi1 + pad) { + const real_t dx = math::min(math::abs(px - glo0), math::abs(px - ghi0)); + const real_t dz = math::min(math::abs(pz - glo2), math::abs(pz - ghi2)); + if (dx * dx + dz * dz < r2) { + return true; + } + } + // edges parallel to z (perp plane = x,y) + if (pz >= glo2 - pad and pz <= ghi2 + pad) { + const real_t dx = math::min(math::abs(px - glo0), math::abs(px - ghi0)); + const real_t dy = math::min(math::abs(py - glo1), math::abs(py - ghi1)); + if (dx * dx + dy * dy < r2) { + return true; + } + } + return false; + } + + // is world point within `tube_radius` of any field-line segment? Walks only + // the bucket of the point's coarse cell (radius << one coarse cell, so a + // padded-AABB insertion guarantees the nearest segment is in this cell). + // On a hit, `scalar` is the |field| interpolated to the closest point. + Inline auto inTube(real_t px, real_t py, real_t pz, real_t& scalar) const + -> bool { + if (tn_seg <= 0) { + return false; + } + const int c0 = static_cast(math::floor((px - tg0) / tdx0)); + const int c1 = static_cast(math::floor((py - tg1) / tdx1)); + const int c2 = static_cast(math::floor((pz - tg2) / tdx2)); + if (c0 < 0 or c0 >= tgnc0 or c1 < 0 or c1 >= tgnc1 or c2 < 0 or c2 >= tgnc2) { + return false; + } + const int lin = (c2 * tgnc1 + c1) * tgnc0 + c0; + const int kb = tcell_start(lin); + const int ke = tcell_start(lin + 1); + real_t best = tube_r2; + bool hit = false; + for (int k = kb; k < ke; ++k) { + const int s = tseg_idx(k); + const real_t ax = tseg(s, 0), ay = tseg(s, 1), az = tseg(s, 2); + const real_t bx = tseg(s, 3), by = tseg(s, 4), bz = tseg(s, 5); + const real_t ex = bx - ax, ey = by - ay, ez = bz - az; + const real_t wx = px - ax, wy = py - ay, wz = pz - az; + const real_t ee = ex * ex + ey * ey + ez * ez; + real_t tt = (ee > ZERO) ? (wx * ex + wy * ey + wz * ez) / ee : ZERO; + tt = (tt < ZERO) ? ZERO : ((tt > ONE) ? ONE : tt); + const real_t cx = ax + tt * ex, cy = ay + tt * ey, cz = az + tt * ez; + const real_t dx = px - cx, dy = py - cy, dz = pz - cz; + const real_t d2 = dx * dx + dy * dy + dz * dz; + if (d2 < best) { + best = d2; + scalar = tseg(s, 6) * (ONE - tt) + tseg(s, 7) * tt; + hit = true; + } + } + return hit; + } + + // trilinear sample of the prepared scalar at world point p, reading the + // ghost halo for corners just outside the active box. + Inline auto sample(real_t px, real_t py, real_t pz) const -> real_t { + // world -> code (cell-center continuous index = code index - 1/2) + const real_t g0 = metric.template convert<1, Crd::Ph, Crd::Cd>(px) - HALF; + const real_t g1 = metric.template convert<2, Crd::Ph, Crd::Cd>(py) - HALF; + const real_t g2 = metric.template convert<3, Crd::Ph, Crd::Cd>(pz) - HALF; + const real_t f0 = math::floor(g0); + const real_t f1 = math::floor(g1); + const real_t f2 = math::floor(g2); + const real_t t0 = g0 - f0; + const real_t t1 = g1 - f1; + const real_t t2 = g2 - f2; + // base View index of the lower corner (active cell i -> View i + N_GHOSTS) + int b0 = static_cast(f0) + static_cast(N_GHOSTS); + int b1 = static_cast(f1) + static_cast(N_GHOSTS); + int b2 = static_cast(f2) + static_cast(N_GHOSTS); + // clamp so both corners (b, b+1) stay in range [0, ext-1] + b0 = (b0 < 0) ? 0 : ((b0 > ext0 - 2) ? ext0 - 2 : b0); + b1 = (b1 < 0) ? 0 : ((b1 > ext1 - 2) ? ext1 - 2 : b1); + b2 = (b2 < 0) ? 0 : ((b2 > ext2 - 2) ? ext2 - 2 : b2); + const real_t c000 = Fld(b0, b1, b2, comp); + const real_t c100 = Fld(b0 + 1, b1, b2, comp); + const real_t c010 = Fld(b0, b1 + 1, b2, comp); + const real_t c110 = Fld(b0 + 1, b1 + 1, b2, comp); + const real_t c001 = Fld(b0, b1, b2 + 1, comp); + const real_t c101 = Fld(b0 + 1, b1, b2 + 1, comp); + const real_t c011 = Fld(b0, b1 + 1, b2 + 1, comp); + const real_t c111 = Fld(b0 + 1, b1 + 1, b2 + 1, comp); + const real_t c00 = c000 * (ONE - t0) + c100 * t0; + const real_t c10 = c010 * (ONE - t0) + c110 * t0; + const real_t c01 = c001 * (ONE - t0) + c101 * t0; + const real_t c11 = c011 * (ONE - t0) + c111 * t0; + const real_t c0 = c00 * (ONE - t1) + c10 * t1; + const real_t c1 = c01 * (ONE - t1) + c11 * t1; + return c0 * (ONE - t2) + c1 * t2; + } + + Inline void operator()(cellidx_t lpx, cellidx_t lpy) const { + // local bbox index -> output pixel; global pixel -> ray generation + const auto pix = static_cast(lpy) * + static_cast(bw) + + static_cast(lpx); + const int gpx = bx0 + static_cast(lpx); + const int gpy = by0 + static_cast(lpy); + const real_t INF = static_cast(1e30); + // default transparent + no fragment + image(pix, 0) = ZERO; + image(pix, 1) = ZERO; + image(pix, 2) = ZERO; + image(pix, 3) = ZERO; + depth_img(pix) = INF; + + // ---- ray generation ------------------------------------------------ // + const real_t fx = TWO * (static_cast(gpx) + HALF) / + static_cast(W) - + ONE; + const real_t fy = ONE - TWO * (static_cast(gpy) + HALF) / + static_cast(H); + real_t ox, oy, oz, dx, dy, dz; + if (cam.projection == out::CameraDevice::Dome) { + // fulldome azimuthal-equidistant fisheye from an interior eye: image + // radius rho in [0,1] -> zenith angle theta = rho * dome_half_fov, + // about the `forward` (zenith) axis; corners (rho > 1) are transparent. + const real_t rho = math::sqrt(fx * fx + fy * fy); + if (rho > ONE) { + return; // outside the dome disk + } + const real_t theta = rho * cam.dome_half_fov; + const real_t phi = math::atan2(fy, fx); + const real_t st = math::sin(theta), ct = math::cos(theta); + const real_t cp = math::cos(phi), sp = math::sin(phi); + dx = ct * cam.forward[0] + st * (cp * cam.right[0] + sp * cam.up[0]); + dy = ct * cam.forward[1] + st * (cp * cam.right[1] + sp * cam.up[1]); + dz = ct * cam.forward[2] + st * (cp * cam.right[2] + sp * cam.up[2]); + ox = cam.eye[0]; + oy = cam.eye[1]; + oz = cam.eye[2]; + } else if (cam.orthographic) { + const real_t sx = fx * cam.half_w; + const real_t sy = fy * cam.half_h; + ox = cam.eye[0] + sx * cam.right[0] + sy * cam.up[0]; + oy = cam.eye[1] + sx * cam.right[1] + sy * cam.up[1]; + oz = cam.eye[2] + sx * cam.right[2] + sy * cam.up[2]; + dx = cam.forward[0]; + dy = cam.forward[1]; + dz = cam.forward[2]; + } else { + const real_t nx = fx * cam.aspect * cam.tan_half_fov; + const real_t ny = fy * cam.tan_half_fov; + dx = cam.forward[0] + nx * cam.right[0] + ny * cam.up[0]; + dy = cam.forward[1] + nx * cam.right[1] + ny * cam.up[1]; + dz = cam.forward[2] + nx * cam.right[2] + ny * cam.up[2]; + const real_t inv = ONE / math::sqrt(dx * dx + dy * dy + dz * dz); + dx *= inv; + dy *= inv; + dz *= inv; + ox = cam.eye[0]; + oy = cam.eye[1]; + oz = cam.eye[2]; + } + + // ---- ray-AABB slab test against [lo, hi] --------------------------- // + real_t t_enter = ZERO; + real_t t_exit = static_cast(1e30); + const real_t eps = static_cast(1e-12); + // axis 0 + if (math::abs(dx) < eps) { + if (ox < lo0 or ox > hi0) { + return; + } + } else { + real_t t1 = (lo0 - ox) / dx; + real_t t2 = (hi0 - ox) / dx; + if (t1 > t2) { + const real_t tmp = t1; + t1 = t2; + t2 = tmp; + } + t_enter = (t1 > t_enter) ? t1 : t_enter; + t_exit = (t2 < t_exit) ? t2 : t_exit; + } + // axis 1 + if (math::abs(dy) < eps) { + if (oy < lo1 or oy > hi1) { + return; + } + } else { + real_t t1 = (lo1 - oy) / dy; + real_t t2 = (hi1 - oy) / dy; + if (t1 > t2) { + const real_t tmp = t1; + t1 = t2; + t2 = tmp; + } + t_enter = (t1 > t_enter) ? t1 : t_enter; + t_exit = (t2 < t_exit) ? t2 : t_exit; + } + // axis 2 + if (math::abs(dz) < eps) { + if (oz < lo2 or oz > hi2) { + return; + } + } else { + real_t t1 = (lo2 - oz) / dz; + real_t t2 = (hi2 - oz) / dz; + if (t1 > t2) { + const real_t tmp = t1; + t1 = t2; + t2 = tmp; + } + t_enter = (t1 > t_enter) ? t1 : t_enter; + t_exit = (t2 < t_exit) ? t2 : t_exit; + } + // dome: clip each ray to a fixed radius around the interior eye, so the + // sampled region is a half-ball (hemisphere) of that radius rather than + // the whole box -> uniform path length, no box corner/edge projection + // artifacts. `t` is world distance from the shared eye (dir is unit), so + // this is a sphere clip and is identical on every rank (seamless). + if (cam.projection == out::CameraDevice::Dome and cam.dome_radius > ZERO) { + t_exit = (t_exit < cam.dome_radius) ? t_exit : cam.dome_radius; + } + if (t_enter >= t_exit) { + return; + } + + // ---- march at global sample positions t_k = k*ds ------------------- // + const real_t inv_range = (vmax > vmin) ? (ONE / (vmax - vmin)) : ZERO; + const real_t log_vmin = log_scale ? math::log10(vmin) : ZERO; + const real_t tube_inv_range = (tube_vmax > tube_vmin) + ? (ONE / (tube_vmax - tube_vmin)) + : ZERO; + const real_t tube_log_vmin = (tube_log and tube_vmin > ZERO) + ? math::log10(tube_vmin) + : ZERO; + // first global sample index inside this segment: t_k >= t_enter + const real_t k0 = math::ceil(t_enter / ds); + real_t t = k0 * ds; + real_t acc_r = ZERO, acc_g = ZERO, acc_b = ZERO, acc_a = ZERO; + int steps = 0; + while (t < t_exit and steps < max_steps) { + const real_t px = ox + t * dx, py = oy + t * dy, pz = oz + t * dz; + real_t cr = ZERO, cg = ZERO, cb = ZERO, ca = ZERO; + real_t tube_s = ZERO; + if (onSpine(px, py, pz)) { + // opaque box edge (premultiplied; alpha == 1 -> color is straight RGB). + // Composited inline so accumulated foreground volume occludes it. + cr = spine_cr; + cg = spine_cg; + cb = spine_cb; + ca = ONE; + } else if (inTube(px, py, pz, tube_s)) { + // opaque field-line tube colored by |field| through the tube LUT + real_t u; + if (tube_log) { + u = (tube_s > ZERO) + ? (math::log10(tube_s) - tube_log_vmin) * tube_inv_range + : -ONE; + } else { + u = (tube_s - tube_vmin) * tube_inv_range; + } + if (u < ZERO) { + u = ZERO; + } else if (u > ONE) { + u = ONE; + } + int idx = static_cast(u * static_cast(tube_n_lut - 1) + HALF); + if (idx < 0) { + idx = 0; + } else if (idx > tube_n_lut - 1) { + idx = tube_n_lut - 1; + } + cr = tube_lut(idx, 0); // opaque LUT -> straight RGB, alpha 1 + cg = tube_lut(idx, 1); + cb = tube_lut(idx, 2); + ca = tube_lut(idx, 3); + } else if (volume_enabled) { + const real_t s = sample(px, py, pz); + // normalize through the transfer function range + real_t u; + if (log_scale) { + u = (s > ZERO) ? (math::log10(s) - log_vmin) * inv_range : -ONE; + } else { + u = (s - vmin) * inv_range; + } + if (u < ZERO) { + u = ZERO; + } else if (u > ONE) { + u = ONE; + } + int idx = static_cast(u * static_cast(n_lut - 1) + HALF); + if (idx < 0) { + idx = 0; + } else if (idx > n_lut - 1) { + idx = n_lut - 1; + } + cr = lut(idx, 0); // premultiplied + cg = lut(idx, 1); + cb = lut(idx, 2); + ca = lut(idx, 3); + } + // composite only a non-empty sample (volume-off rays contribute solely + // where they hit a tube or the spine) + if (ca > ZERO) { + const real_t w = ONE - acc_a; + acc_r += w * cr; + acc_g += w * cg; + acc_b += w * cb; + acc_a += w * ca; + if (acc_a >= early_alpha) { + break; + } + } + t += ds; + ++steps; + } + + image(pix, 0) = acc_r; + image(pix, 1) = acc_g; + image(pix, 2) = acc_b; + image(pix, 3) = acc_a; + // record the domain's front depth (geometric slab entry) for a produced + // fragment; the A-buffer composite orders domains by this. Domain slabs + // are disjoint contiguous ray intervals, so t_enter is a correct key. + if (acc_a > ZERO) { + depth_img(pix) = t_enter; + } + } + }; + +} // namespace render + +#endif // OUTPUT_RENDER_RAYMARCH_HPP diff --git a/src/output/render/reduce.hpp b/src/output/render/reduce.hpp new file mode 100644 index 000000000..b5a590251 --- /dev/null +++ b/src/output/render/reduce.hpp @@ -0,0 +1,178 @@ +/** + * @file output/render/reduce.hpp + * @brief Small dimension-generic cell reductions used by the in-situ renderer + * @implements + * - render::RenderMagnitude3_kernel + * - render::RenderPickComp_kernel + * - render::RenderDivideComp_kernel + * - render::RenderVmagByRho_kernel + * @namespaces: + * - render:: + * @note + * Each functor provides 1D/2D/3D operator() overloads so the same object works + * with `mesh.rangeActiveCells()` of any dimension (the range policy selects the + * matching arity). They reduce the prepared (interpolated, synced) `bckp` + * scratch field down to the single scalar component the ray-march / slice + * kernel samples, so the field-grammar dispatch stays dimension-agnostic. + */ + +#ifndef OUTPUT_RENDER_REDUCE_HPP +#define OUTPUT_RENDER_REDUCE_HPP + +#include "global.h" + +#include "arch/kokkos_aliases.h" +#include "utils/numeric.h" + +#include + +namespace render { + using namespace ntt; + + /** + * @brief F(.., co) = sqrt(F(.., c0)^2 + F(.., c1)^2 + F(.., c2)^2) + */ + template + class RenderMagnitude3_kernel { + ndfield_t F; + const uint8_t c0, c1, c2, co; + + public: + RenderMagnitude3_kernel(const ndfield_t& f, + uint8_t a, + uint8_t b, + uint8_t c, + uint8_t o) + : F { f } + , c0 { a } + , c1 { b } + , c2 { c } + , co { o } {} + + Inline void operator()(cellidx_t i1) const { + const real_t v0 = F(i1, c0), v1 = F(i1, c1), v2 = F(i1, c2); + F(i1, co) = math::sqrt(v0 * v0 + v1 * v1 + v2 * v2); + } + + Inline void operator()(cellidx_t i1, cellidx_t i2) const { + const real_t v0 = F(i1, i2, c0), v1 = F(i1, i2, c1), v2 = F(i1, i2, c2); + F(i1, i2, co) = math::sqrt(v0 * v0 + v1 * v1 + v2 * v2); + } + + Inline void operator()(cellidx_t i1, cellidx_t i2, cellidx_t i3) const { + const real_t v0 = F(i1, i2, i3, c0), v1 = F(i1, i2, i3, c1), + v2 = F(i1, i2, i3, c2); + F(i1, i2, i3, co) = math::sqrt(v0 * v0 + v1 * v1 + v2 * v2); + } + }; + + /** + * @brief F(.., co) = F(.., ci) (move one component into the render slot) + */ + template + class RenderPickComp_kernel { + ndfield_t F; + const uint8_t ci, co; + + public: + RenderPickComp_kernel(const ndfield_t& f, uint8_t i, uint8_t o) + : F { f } + , ci { i } + , co { o } {} + + Inline void operator()(cellidx_t i1) const { + F(i1, co) = F(i1, ci); + } + + Inline void operator()(cellidx_t i1, cellidx_t i2) const { + F(i1, i2, co) = F(i1, i2, ci); + } + + Inline void operator()(cellidx_t i1, cellidx_t i2, cellidx_t i3) const { + F(i1, i2, i3, co) = F(i1, i2, i3, ci); + } + }; + + /** + * @brief F(.., cnum) = (F(.., cden) != 0) ? F(.., cnum) / F(.., cden) : 0 + */ + template + class RenderDivideComp_kernel { + ndfield_t F; + const uint8_t cnum, cden; + + public: + RenderDivideComp_kernel(const ndfield_t& f, uint8_t num, uint8_t den) + : F { f } + , cnum { num } + , cden { den } {} + + Inline void operator()(cellidx_t i1) const { + const real_t d = F(i1, cden); + F(i1, cnum) = (d != ZERO) ? (F(i1, cnum) / d) : ZERO; + } + + Inline void operator()(cellidx_t i1, cellidx_t i2) const { + const real_t d = F(i1, i2, cden); + F(i1, i2, cnum) = (d != ZERO) ? (F(i1, i2, cnum) / d) : ZERO; + } + + Inline void operator()(cellidx_t i1, cellidx_t i2, cellidx_t i3) const { + const real_t d = F(i1, i2, i3, cden); + F(i1, i2, i3, cnum) = (d != ZERO) ? (F(i1, i2, i3, cnum) / d) : ZERO; + } + }; + + /** + * @brief F(.., co) = | (F(c0), F(c1), F(c2)) / F(crho) |, else 0. + * @note SR bulk-speed magnitude: the three mass-weighted flux components are + * each normalized by Rho before the Euclidean norm, in one pass. + */ + template + class RenderVmagByRho_kernel { + ndfield_t F; + const uint8_t c0, c1, c2, crho, co; + + public: + RenderVmagByRho_kernel(const ndfield_t& f, + uint8_t a, + uint8_t b, + uint8_t c, + uint8_t rho, + uint8_t o) + : F { f } + , c0 { a } + , c1 { b } + , c2 { c } + , crho { rho } + , co { o } {} + + Inline auto mag(real_t v0, real_t v1, real_t v2, real_t rho) const -> real_t { + if (rho == ZERO) { + return ZERO; + } + const real_t a = v0 / rho, b = v1 / rho, c = v2 / rho; + return math::sqrt(a * a + b * b + c * c); + } + + Inline void operator()(cellidx_t i1) const { + F(i1, co) = mag(F(i1, c0), F(i1, c1), F(i1, c2), F(i1, crho)); + } + + Inline void operator()(cellidx_t i1, cellidx_t i2) const { + F(i1, + i2, + co) = mag(F(i1, i2, c0), F(i1, i2, c1), F(i1, i2, c2), F(i1, i2, crho)); + } + + Inline void operator()(cellidx_t i1, cellidx_t i2, cellidx_t i3) const { + F(i1, i2, i3, co) = mag(F(i1, i2, i3, c0), + F(i1, i2, i3, c1), + F(i1, i2, i3, c2), + F(i1, i2, i3, crho)); + } + }; + +} // namespace render + +#endif // OUTPUT_RENDER_REDUCE_HPP diff --git a/src/output/render/renderer.cpp b/src/output/render/renderer.cpp new file mode 100644 index 000000000..312b0683d --- /dev/null +++ b/src/output/render/renderer.cpp @@ -0,0 +1,1003 @@ +#include "output/render/renderer.h" + +#include "global.h" + +#include "arch/kokkos_aliases.h" +#include "utils/error.h" +#include "utils/formatting.h" +#include "utils/log.h" +#include "utils/numeric.h" + +#include "output/render/axes.h" +#include "output/render/colorbar.h" +#include "output/render/composite.h" +#include "output/render/png.h" +#include "output/render/transfer_fn.h" + +#if defined(MPI_ENABLED) + #include "arch/mpi_aliases.h" + + #include +#endif + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +namespace out { + + namespace { + + inline void cross3(const real_t a[3], const real_t b[3], real_t out[3]) { + out[0] = a[1] * b[2] - a[2] * b[1]; + out[1] = a[2] * b[0] - a[0] * b[2]; + out[2] = a[0] * b[1] - a[1] * b[0]; + } + + inline auto norm3(const real_t a[3]) -> real_t { + return std::sqrt(a[0] * a[0] + a[1] * a[1] + a[2] * a[2]); + } + + inline void normalize3(real_t a[3]) { + const real_t n = norm3(a); + if (n > static_cast(1e-30)) { + a[0] /= n; + a[1] /= n; + a[2] /= n; + } + } + + inline auto quantize(real_t v) -> uint8_t { + const real_t c = (v < ZERO) ? ZERO : ((v > ONE) ? ONE : v); + return static_cast(c * static_cast(255.0) + HALF); + } + + } // namespace + + void Renderer::init(const ntt::SimulationParams& params, + const boundaries_t& global_extent) { + m_enabled = false; + + if (not params.get("render.enable")) { + return; + } + // 2D (slice rasterizer) and 3D Cartesian (volume ray-march) are supported; + // 1D has nothing to render. + if (global_extent.size() != 2 and global_extent.size() != 3) { + raise::Warning("render enabled but simulation is 1D; " + "the renderer will be inactive", + HERE); + return; + } + + m_root = path_t(params.get("simulation.name")); + + m_width = params.get("render.width"); + m_height = params.get("render.height"); + m_samples = params.get("render.volume.samples"); + m_step_size = params.get("render.volume.step_size"); + m_early_alpha = params.get("render.volume.early_term_alpha"); + m_n_lut = params.get("render.n_lut"); + + // opaque background color (shows through low-alpha pixels) + const auto bg = params.get>("render.background"); + for (auto i = 0u; i < 3; ++i) { + m_background[i] = bg[i]; + } + + m_colorbar = params.get("render.colorbar"); + m_colorbar_outside = params.get("render.colorbar_outside"); + // 2D slice mode (spherical only): mirror the half-plane into a full disk + m_mirror = params.get("render.mirror"); + + // draw the current simulation time in the upper-right corner + m_time_label = params.get("render.time_label"); + + // axes: spine + ticks + labels around the rendered region + m_axes = params.get("render.axes"); + m_axis_nticks = params.get("render.axis_ticks"); + m_spine_width = params.get("render.spine_width"); + m_global_extent = global_extent; + + // optional axis-aligned render region (physical coords). Unset axes default + // to the full extent; user limits are clamped to the box (nothing to render + // outside it). extent.x{1,2,3} -> axes {0,1,2} (r/theta for spherical 2D). + m_region = global_extent; + m_has_region = false; + { + const char* keys[3] = { "x1", "x2", "x3" }; + for (size_t d = 0; d < global_extent.size() and d < 3; ++d) { + const auto lim = params.get>( + "render.extent." + std::string(keys[d])); + if (lim.empty()) { + continue; + } + const real_t lo = std::max(lim[0], global_extent[d].first); + const real_t hi = std::min(lim[1], global_extent[d].second); + if (hi > lo) { + m_region[d] = { lo, hi }; + m_has_region = true; + } else { + raise::Warning("render.extent." + std::string(keys[d]) + + " does not overlap the domain; ignoring", + HERE); + } + } + } + + /* ---- fulldome fisheye (planetarium dome master) --------------------- */ + // A 2D-only radial ("fisheye") projection of a circular cutout centered on + // the domain, drawn into the frame's inscribed circle (corners kept as the + // background border -> a valid dome master). The 3D dome is a separate + // workstream; warn if asked for here so it does not silently fall back to + // the volume camera. + if (params.get("render.dome.enable")) { + if (global_extent.size() != 2) { + raise::Warning("render.dome is 2D-only for now; ignoring", HERE); + } else { + m_dome.enabled = true; + const real_t fov = params.get("render.dome.fov"); + m_dome.theta_max = HALF * fov * static_cast(constant::PI) / + static_cast(180); + const auto proj = params.get("render.dome.projection"); + if (proj == "gnomonic") { + m_dome.law = DomeMap::Gnomonic; + } else if (proj == "stereographic") { + m_dome.law = DomeMap::Stereographic; + } else if (proj == "orthographic") { + m_dome.law = DomeMap::Orthographic; + } else { + m_dome.law = DomeMap::Equidistant; + } + // the gnomonic (flat-tangent) law diverges as the dome half-FOV -> 90 + // deg (a flat plane never reaches the horizon), so cap it below that. + if (m_dome.law == DomeMap::Gnomonic) { + const real_t cap = static_cast(89.0 * constant::PI / 180.0); + if (m_dome.theta_max >= cap) { + raise::Warning("render.dome: 'gnomonic' needs fov < 180 deg " + "(a flat plane cannot reach the dome horizon); " + "capping the half-FOV at 89 deg", + HERE); + m_dome.theta_max = cap; + } + } + // default center = domain center; default radius = the largest disk + // that fits inside the (rectangular) domain (half the shorter side). + const real_t Lx = global_extent[0].second - global_extent[0].first; + const real_t Ly = global_extent[1].second - global_extent[1].first; + m_dome.cx = HALF * (global_extent[0].first + global_extent[0].second); + m_dome.cy = HALF * (global_extent[1].first + global_extent[1].second); + const auto ctr = params.get>("render.dome.center"); + if (ctr.size() == 2) { + m_dome.cx = ctr[0]; + m_dome.cy = ctr[1]; + } + const real_t rdef = HALF * std::min(Lx, Ly); + m_dome.R = params.contains("render.dome.radius") + ? params.get("render.dome.radius") + : rdef; + if (m_dome.R <= ZERO) { + m_dome.R = rdef; + } + if (m_width != m_height) { + raise::Warning("render.dome: width != height; the fisheye " + "disk is centered on the shorter side and the frame " + "is not a square dome master", + HERE); + } + } + } + + { + const auto al = params.get>( + "render.axis_labels"); + m_axis_labels_set = not al.empty(); + for (size_t d = 0; d < al.size() and d < 3; ++d) { + m_axis_labels[d] = al[d]; + } + // default 2D slice names track the labels (overridden per-metric by Render) + m_slice_xlabel = m_axis_labels[0]; + m_slice_ylabel = m_axis_labels[1]; + } + + // cadence (falls back to output.interval{,_time}; see params::Render) + m_tracker.init("render", + params.get("render.interval"), + params.get("render.interval_time")); + + /* ---- camera (used by the 3D volume mode; the 2D slice path frames itself + * and ignores this, so a missing 3rd axis is zero-filled harmlessly) ---- */ + // frame the camera on the render region (== the full extent when uncropped) + real_t center[3] = { ZERO, ZERO, ZERO }, size[3] = { ZERO, ZERO, ZERO }; + for (size_t d = 0; d < m_region.size() and d < 3; ++d) { + center[d] = static_cast(0.5) * + (m_region[d].first + m_region[d].second); + size[d] = m_region[d].second - m_region[d].first; + } + const real_t diag = std::sqrt( + size[0] * size[0] + size[1] * size[1] + size[2] * size[2]); + + // "orthographic" | "perspective" | "dome". The dome is a fulldome + // azimuthal-equidistant fisheye from an INTERIOR eye (the box center by + // default) -- see Metadomain::Render (3D). + const auto cam_mode = params.get("render.camera.mode"); + auto projection = CameraDevice::Ortho; + if (cam_mode == "dome") { + projection = CameraDevice::Dome; + } else if (cam_mode == "perspective") { + projection = CameraDevice::Perspective; + } + const bool is_dome = (projection == CameraDevice::Dome); + const real_t dome_fov = params.get("render.camera.dome_fov"); + const auto pos = params.get>("render.camera.position"); + const auto look = params.get>("render.camera.look_at"); + const auto up = params.get>("render.camera.up"); + const real_t fov = params.get("render.camera.fov"); + // default covers the box from any view direction (default camera looks + // down the diagonal), so nothing is clipped without explicit framing. + const real_t ortho_height = params.contains("render.camera.ortho_height") + ? params.get( + "render.camera.ortho_height") + : diag; + + real_t eye[3], lookat[3], upv[3]; + for (int d = 0; d < 3; ++d) { + if (pos.size() == 3) { + eye[d] = pos[d]; + } else if (is_dome) { + // interior eye: the domain center (looking outward at the sky) + eye[d] = center[d]; + } else { + // default eye: box center pushed back along (1,1,1) by ~1.7 diagonals + eye[d] = center[d] + static_cast(1.7) * diag * + static_cast(0.57735026919); + } + lookat[d] = (look.size() == 3) ? look[d] : center[d]; + } + // up / screen-up: default +z, except the dome (whose default zenith is +z, + // so +z would be collinear with `forward`) uses +y as the disk's screen-up. + if (up.size() == 3) { + upv[0] = up[0]; + upv[1] = up[1]; + upv[2] = up[2]; + } else if (is_dome) { + upv[0] = ZERO; + upv[1] = ONE; + upv[2] = ZERO; + } else { + upv[0] = ZERO; + upv[1] = ZERO; + upv[2] = ONE; + } + + real_t forward[3] = { lookat[0] - eye[0], + lookat[1] - eye[1], + lookat[2] - eye[2] }; + // for the dome, `forward` is the ZENITH; with the default interior eye at the + // center (== lookat) the difference is zero, so default the zenith to +z. + if (is_dome and look.size() != 3) { + forward[0] = ZERO; + forward[1] = ZERO; + forward[2] = ONE; + } + normalize3(forward); + real_t right[3]; + cross3(forward, upv, right); + normalize3(right); + real_t up_cam[3]; + cross3(right, forward, up_cam); + + for (int d = 0; d < 3; ++d) { + m_camera_dev.eye[d] = eye[d]; + m_camera_dev.forward[d] = forward[d]; + m_camera_dev.right[d] = right[d]; + m_camera_dev.up[d] = up_cam[d]; + } + m_camera_dev.aspect = static_cast(m_width) / + static_cast(m_height); + m_camera_dev.tan_half_fov = std::tan(static_cast(0.5) * fov * + static_cast(constant::PI) / + static_cast(180.0)); + // the kernel/screenBBox pick ortho vs perspective from `orthographic`, and + // only Dome is read off `projection`. + m_camera_dev.orthographic = (projection == CameraDevice::Ortho); + m_camera_dev.half_h = static_cast(0.5) * ortho_height; + m_camera_dev.half_w = m_camera_dev.half_h * m_camera_dev.aspect; + m_camera_dev.projection = projection; + if (is_dome) { + real_t hf = HALF * dome_fov * static_cast(constant::PI) / + static_cast(180); + if (hf <= ZERO) { + hf = HALF * static_cast(constant::PI); // fall back to a 180 dome + } + if (hf > static_cast(constant::PI)) { + hf = static_cast(constant::PI); // full sphere cap + } + m_camera_dev.dome_half_fov = hf; + // a fisheye disk needs a square frame; the kernel uses the full-frame ndc + m_camera_dev.aspect = ONE; + m_camera_dev.half_w = m_camera_dev.half_h; + if (m_width != m_height) { + raise::Warning("render.camera.mode='dome' wants width == height " + "for a circular dome master; the fisheye disk will be " + "elliptical otherwise", + HERE); + } + // spherical far-clip: each ray stops at `dome_radius` from the eye, so + // the sampled region is a half-ball (hemisphere) instead of the whole box + // -> uniform path length, no box corner/edge projection artifacts. + // Default = the largest sphere centered in the box (half the shortest + // side). A value of 0 disables the clip (march to the box boundary); a + // negative value also selects the default. + real_t insc = static_cast(1e30); + for (size_t d = 0; d < m_region.size() and d < 3; ++d) { + insc = (size[d] < insc) ? size[d] : insc; + } + insc *= HALF; + real_t domeR = params.contains("render.camera.dome_radius") + ? params.get("render.camera.dome_radius") + : insc; + if (domeR < ZERO) { + domeR = insc; + } + m_camera_dev.dome_radius = domeR; + } + + /* ---- moving view (pan the region/camera to track a feature) --------- */ + // remember the static region + camera eye; updateForTime() translates them. + m_region_base = m_region; + for (int d = 0; d < 3; ++d) { + m_eye_base[d] = m_camera_dev.eye[d]; + } + { + const auto vel = params.get>( + "render.moving_view.velocity"); + for (size_t d = 0; d < vel.size() and d < 3; ++d) { + m_cam_vel[d] = vel[d]; + } + m_cam_moving = (m_cam_vel[0] != ZERO) or (m_cam_vel[1] != ZERO) or + (m_cam_vel[2] != ZERO); + m_cam_t0 = params.get("render.moving_view.start_time"); + if (m_cam_moving and not m_has_region and global_extent.size() == 2) { + raise::Warning("render.moving_view.velocity set without " + "render.extent.x{1,2}: the 2D window will pan off the " + "domain. Set a region to track a feature within it.", + HERE); + } + } + + /* ---- scenes --------------------------------------------------------- */ + m_scenes.clear(); + const auto nscenes = params.get("render.nscenes"); + for (std::size_t i = 0; i < nscenes; ++i) { + const auto pfx = "render.scene." + std::to_string(i) + "."; + Scene scene; + scene.field = params.get(pfx + "field"); + scene.prefix = params.get(pfx + "prefix"); + scene.label = params.get(pfx + "label"); + scene.ticks = params.get>(pfx + "colorbar_ticks"); + scene.show_fieldlines = params.get(pfx + "fieldlines"); + scene.tf.vmin = params.get(pfx + "min"); + scene.tf.vmax = params.get(pfx + "max"); + scene.tf.log_scale = params.get(pfx + "log"); + scene.tf.n_lut = m_n_lut; + const auto colormap = params.get(pfx + "colormap"); + scene.tf.colormap = colormap; + // alpha control points: array of [position, alpha] pairs + const auto alpha_raw = params.get>>( + pfx + "alpha"); + std::vector> alpha_pts; + for (const auto& p : alpha_raw) { + if (p.size() >= 2) { + alpha_pts.push_back({ p[0], p[1] }); + } + } + scene.tf.lut = buildLUT(colormap, m_n_lut, alpha_pts); + // opaque companion LUT (alpha == 1) for the flat 2D slice rasterizer + scene.tf.lut_opaque = buildLUT(colormap, + m_n_lut, + { + { ZERO, ONE }, + { ONE, ONE } + }); + m_scenes.push_back(std::move(scene)); + } + + /* ---- magnetic-field-line tube overlay ------------------------------- */ + // enabled by [render.fieldlines] OR by any scene requesting the overlay + // (resolved in params::Render) + m_fieldlines.enable = params.get("render.fieldlines.enable"); + if (m_fieldlines.enable) { + // 3D -> traced tubes inside the volume; 2D -> flux-function contours + auto& fl = m_fieldlines; + fl.field = params.get("render.fieldlines.field"); + fl.bin = params.get("render.fieldlines.bin"); + fl.seed_px = params.get("render.fieldlines.seed_px"); + fl.tube_px = params.get("render.fieldlines.tube_px"); + fl.colormap = params.get("render.fieldlines.colormap"); + // optional monochrome color [r,g,b]; overrides the colormap when set + fl.color = params.get>("render.fieldlines.color"); + fl.log_scale = params.get("render.fieldlines.log"); + fl.vmin = params.get("render.fieldlines.min"); + fl.vmax = params.get("render.fieldlines.max"); + fl.step_frac = params.get("render.fieldlines.step_frac"); + fl.max_steps = params.get("render.fieldlines.max_steps"); + fl.max_len_frac = params.get("render.fieldlines.max_length"); + fl.seed_max = params.get("render.fieldlines.seed_max"); + fl.levels = params.get("render.fieldlines.levels"); + } + + m_enabled = true; + logger::Checkpoint("In-situ renderer initialized", HERE); + } + + void Renderer::updateForTime(simtime_t time) { + if (not m_cam_moving) { + return; + } + const real_t dt = static_cast( + (time > m_cam_t0) ? (time - m_cam_t0) : static_cast(0)); + const real_t shift[3] = { m_cam_vel[0] * dt, + m_cam_vel[1] * dt, + m_cam_vel[2] * dt }; + // translate the render region (its width is preserved) + for (size_t d = 0; d < m_region.size() and d < 3; ++d) { + m_region[d] = { m_region_base[d].first + shift[d], + m_region_base[d].second + shift[d] }; + } + // translate the 3D camera by the same shift -- a pure pan: forward/right/up + // and the ortho height are unchanged, so only the eye moves. + for (int d = 0; d < 3; ++d) { + m_camera_dev.eye[d] = m_eye_base[d] + shift[d]; + } + } + + void Renderer::writeFrame(const std::vector& img, + const Scene& scene, + timestep_t step, + simtime_t time) const { + const size_t npix = static_cast(m_width) * + static_cast(m_height); + const size_t n = npix * 4; + // ensure /renders/ exists + const auto dir = m_root / path_t("renders"); + try { + if (not std::filesystem::exists(m_root)) { + std::filesystem::create_directory(m_root); + } + if (not std::filesystem::exists(dir)) { + std::filesystem::create_directory(dir); + } + } catch (const std::exception& e) { + raise::Warning(e.what(), HERE); + } + // composite the premultiplied image over the opaque background: + // out = src_premult + (1 - src_alpha) * background, alpha = opaque. + std::vector data(n); + for (size_t p = 0; p < npix; ++p) { + const real_t a = img[p * 4 + 3]; + const real_t inv = ONE - a; + data[p * 4 + 0] = quantize(img[p * 4 + 0] + inv * m_background[0]); + data[p * 4 + 1] = quantize(img[p * 4 + 1] + inv * m_background[1]); + data[p * 4 + 2] = quantize(img[p * 4 + 2] + inv * m_background[2]); + data[p * 4 + 3] = 255; + } + const auto fname = dir / fmt::format("%s%08lu.png", + scene.prefix.c_str(), + static_cast(step)); + + auto drawBar = [&](uint8_t* buf, int bw, int bh, int span_top, int span_bot) { + if (m_colorbar) { + drawColorbar(buf, + bw, + bh, + scene.tf.colormap, + scene.tf.vmin, + scene.tf.vmax, + scene.tf.log_scale, + scene.label, + m_background, + scene.ticks, + span_top, + span_bot); + } + }; + + // Draw the sim-time label, right-aligned to `right_x` and vertically + // centered in the band [0, top_limit] -- i.e. OUTSIDE the plotted data: + // above the colorbar (3D / disk) or, for a 2D slice, in the aspect-pad + // above the data box (so it never sits inside the simulation axes). + auto drawTimeLabel = + [&](uint8_t* buf, int cw, int ch, int right_x, int top_limit) { + if (not m_time_label) { + return; + } + const int s = cbar_hidden::scale(m_height); + char tbuf[48]; + // fixed-point so it reads e.g. "T = 12345.67" (up to 5 integer digits + // and 2 decimals; more integer digits still print, never truncated) + std::snprintf(tbuf, sizeof(tbuf), "T = %.2f", static_cast(time)); + const std::string str(tbuf); + const int tw = static_cast(str.size()) * 6 * s; + const int pad = 3 * s; + const int tx = right_x - tw - pad; + const int text_h = 7 * s; + int ty = (top_limit - text_h) / 2; + if (ty < pad) { + ty = pad; + } + // contrasting text color (white on a dark background, black on light) + const real_t lum = static_cast(0.299) * m_background[0] + + static_cast(0.587) * m_background[1] + + static_cast(0.114) * m_background[2]; + const uint8_t tc = (lum < HALF) ? 255 : 0; + cbar_hidden::drawText(buf, cw, ch, tx, ty, str, s, tc, tc, tc); + }; + + // canvas margins: axes (left + bottom) and the colorbar strip (right). + // The data region sits at (ml, 0); margins/strip are background-filled. + // The polar (curvilinear) overlay annotates inside the data region (the + // disk is centered with background around it), so it needs no margins. + const bool polar = (m_global_extent.size() == 2) and m_slice_polar; + // a fisheye dome master (2D or 3D) must stay exactly W x H (its inscribed + // circle is the dome), so it takes no axes margins and no outside colorbar. + const bool dome = m_dome_active; + int ml = 0, mb = 0; + out::axesMargins(m_axes and not polar and not dome, m_height, ml, mb); + const int strip = (m_colorbar and m_colorbar_outside and not dome) + ? colorbarBlockWidth(m_height) + : 0; + const int CW = ml + m_width + strip; + const int CH = m_height + mb; + + // 2D-Cartesian data box (== the render region, before the aspect-expansion + // that pads the window with background): its top & right edges in + // data-region pixels. The axes/spine clamp to it and the time label sits + // in the pad above it, so neither includes the empty aspect padding. + const bool cart2d = (m_global_extent.size() == 2) and not polar and not dome; + int dbox_top = 0; // data box top edge (px from data top) + int dbox_bot = m_height; // data box bottom edge (px) + int dbox_right = m_width; // data box right edge (px from data left) + if (cart2d and m_region.size() >= 2) { + const real_t u0 = m_slice_win[0], u1 = m_slice_win[1]; + const real_t v0 = m_slice_win[2], v1 = m_slice_win[3]; + const real_t du1 = m_region[0].second; // data box right in world (x1) + const real_t dv0 = m_region[1].first; // data box bottom in world (x2) + const real_t dv1 = m_region[1].second; // data box top in world (x2) + if (u1 > u0) { + int r = static_cast(std::lround( + static_cast((du1 - u0) / (u1 - u0)) * (m_width - 1))); + dbox_right = (r < 0) ? 0 : ((r > m_width) ? m_width : r); + } + if (v1 > v0) { + int t = static_cast(std::lround( + static_cast((v1 - dv1) / (v1 - v0)) * (m_height - 1))); + int b = static_cast(std::lround( + static_cast((v1 - dv0) / (v1 - v0)) * (m_height - 1))); + dbox_top = (t < 0) ? 0 : ((t > m_height) ? m_height : t); + dbox_bot = (b < 0) ? 0 : ((b > m_height) ? m_height : b); + } + } + // time-label anchor: for a 2D slice, the top-right of the data box (label + // goes in the pad above it); otherwise the top-right above the colorbar. + const int cbar_top = m_colorbar ? (CH - CH / 2) / 2 : (CH / 4); + const int tl_right = cart2d ? (ml + dbox_right) : (ml + m_width); + const int tl_top = cart2d ? dbox_top : cbar_top; + // colorbar vertical span: aligned to the actual data domain for a 2D slice + // (so it's centered on the data, not the aspect-padded canvas); sentinel + // (-1) elsewhere -> drawColorbar centers it on the canvas as before. + const int cbar_span_top = cart2d ? dbox_top : -1; + const int cbar_span_bot = cart2d ? dbox_bot : -1; + + bool ok = true; + if (CW == m_width and CH == m_height and not m_axes) { + // no margins, no outside strip, no overlay: colorbar overlays the data + drawBar(data.data(), m_width, m_height, cbar_span_top, cbar_span_bot); + drawTimeLabel(data.data(), m_width, m_height, tl_right, tl_top); + ok = write_png(fname, m_width, m_height, data.data()); + } else { + const uint8_t bR = quantize(m_background[0]); + const uint8_t bG = quantize(m_background[1]); + const uint8_t bB = quantize(m_background[2]); + std::vector canvas(static_cast(CW) * CH * 4); + for (size_t i = 0; i < canvas.size(); i += 4) { + canvas[i + 0] = bR; + canvas[i + 1] = bG; + canvas[i + 2] = bB; + canvas[i + 3] = 255; + } + for (int y = 0; y < m_height; ++y) { + std::copy_n(&data[static_cast(y) * m_width * 4], + static_cast(m_width) * 4, + &canvas[(static_cast(y) * CW + ml) * 4]); + } + if (m_axes and not dome) { + if (m_global_extent.size() == 3) { + out::drawAxes3D(canvas.data(), + CW, + CH, + ml, + m_width, + m_height, + m_camera_dev, + m_region, + m_axis_labels, + m_background, + m_axis_nticks); + } else if (polar) { + out::drawAxesPolar(canvas.data(), + CW, + CH, + ml, + m_width, + m_height, + m_slice_win[0], + m_slice_win[1], + m_slice_win[2], + m_slice_win[3], + m_slice_rmin, + m_slice_rmax, + m_slice_tmin, + m_slice_tmax, + m_slice_pmirror, + "R", + "Theta", + m_background, + m_axis_nticks); + } else { + // data box (== region, un-expanded) so the spine hugs the domain, + // not the aspect-padded window + const real_t du0 = m_region[0].first, du1 = m_region[0].second; + const real_t dv0 = m_region[1].first, dv1 = m_region[1].second; + out::drawAxes2D(canvas.data(), + CW, + CH, + ml, + m_width, + m_height, + m_slice_win[0], + m_slice_win[1], + m_slice_win[2], + m_slice_win[3], + du0, + du1, + dv0, + dv1, + m_slice_xlabel, + m_slice_ylabel, + m_background, + m_axis_nticks); + } + } + drawBar(canvas.data(), CW, CH, cbar_span_top, cbar_span_bot); + drawTimeLabel(canvas.data(), CW, CH, tl_right, tl_top); + ok = write_png(fname, CW, CH, canvas.data()); + } + if (not ok) { + raise::Warning(fmt::format("failed to write %s", fname.string().c_str()), + HERE); + } + } + + void Renderer::compositeAndWrite(const SubImage& sub, + uint64_t order_key, + const Scene& scene, + timestep_t step, + simtime_t time) const { + const size_t npix = static_cast(m_width) * + static_cast(m_height); + const size_t n = npix * 4; + + // expand a sparse sub-image into a full transparent frame (premultiplied) + auto subToFull = [&](const SubImage& s) -> std::vector { + std::vector full(n, ZERO); + for (int y = 0; y < s.h; ++y) { + for (int x = 0; x < s.w; ++x) { + const int fx = s.x0 + x; + const int fy = s.y0 + y; + if (fx < 0 or fx >= m_width or fy < 0 or fy >= m_height) { + continue; + } + const size_t fi = (static_cast(fy) * m_width + fx) * 4; + const size_t si = (static_cast(y) * s.w + x) * 4; + full[fi + 0] = s.rgba[si + 0]; + full[fi + 1] = s.rgba[si + 1]; + full[fi + 2] = s.rgba[si + 2]; + full[fi + 3] = s.rgba[si + 3]; + } + } + return full; + }; + + auto write_image = [&](const std::vector& img) { + writeFrame(img, scene, step, time); + }; + +#if defined(MPI_ENABLED) + int rank = 0, size = 1; + MPI_Comm_rank(MPI_COMM_WORLD, &rank); + MPI_Comm_size(MPI_COMM_WORLD, &size); + + if (size == 1) { + write_image(subToFull(sub)); + return; + } + + constexpr int TAG_HDR = 7301; + constexpr int TAG_DATA = 7302; + + // Wire format is premultiplied uint8 RGBA (4x less bandwidth than float). + // Compositing stays in float; only the per-message quantization adds error + // (~1 LSB through the log(N)-deep tree), so fidelity is effectively that of + // the final 8-bit PNG. + auto sendSub = [&](const SubImage& s, int dest) { + int hdr[4] = { s.x0, s.y0, s.w, s.h }; + MPI_Send(hdr, 4, MPI_INT, dest, TAG_HDR, MPI_COMM_WORLD); + const int cnt = s.w * s.h * 4; + if (cnt > 0) { + std::vector bytes(static_cast(cnt)); + for (int i = 0; i < cnt; ++i) { + bytes[i] = quantize(s.rgba[i]); + } + MPI_Send(bytes.data(), cnt, MPI_UNSIGNED_CHAR, dest, TAG_DATA, MPI_COMM_WORLD); + } + }; + auto recvSub = [&](int src) -> SubImage { + int hdr[4]; + MPI_Recv(hdr, 4, MPI_INT, src, TAG_HDR, MPI_COMM_WORLD, MPI_STATUS_IGNORE); + SubImage s; + s.x0 = hdr[0]; + s.y0 = hdr[1]; + s.w = hdr[2]; + s.h = hdr[3]; + const int cnt = s.w * s.h * 4; + if (cnt > 0) { + std::vector bytes(static_cast(cnt)); + MPI_Recv(bytes.data(), + cnt, + MPI_UNSIGNED_CHAR, + src, + TAG_DATA, + MPI_COMM_WORLD, + MPI_STATUS_IGNORE); + s.rgba.resize(static_cast(cnt)); + const real_t inv255 = ONE / static_cast(255); + for (int i = 0; i < cnt; ++i) { + s.rgba[i] = static_cast(bytes[i]) * inv255; + } + } + return s; + }; + + // Every rank learns the full key vector (one uint64 each: ~tiny) and + // derives the same global front-to-back order, so no rank needs the others' + // images to agree on the composite order. + const unsigned long long my_key = static_cast(order_key); + std::vector keys(static_cast(size)); + MPI_Allgather(&my_key, + 1, + MPI_UNSIGNED_LONG_LONG, + keys.data(), + 1, + MPI_UNSIGNED_LONG_LONG, + MPI_COMM_WORLD); + std::vector order(size); // order[position] = world rank, front-to-back + std::iota(order.begin(), order.end(), 0); + std::stable_sort(order.begin(), order.end(), [&](int a, int b) { + return keys[a] < keys[b]; + }); + std::vector pos(size); // pos[world rank] = front-to-back position + for (int i = 0; i < size; ++i) { + pos[order[i]] = i; + } + + // Order-preserving binary tree reduction over positions. At level `s`, the + // front of each pair (lower position) receives the back partner's image and + // composites front OVER back; the back partner sends and drops out. "over" + // is associative, so this reproduces the sequential front-to-back composite + // in O(log nranks) rounds with no single-rank bottleneck. + SubImage cur = sub; + const int P = pos[rank]; + for (int s = 1; s < size; s <<= 1) { + if ((P % (2 * s)) == 0) { + const int pp = P + s; + if (pp < size) { + const SubImage back = recvSub(order[pp]); + cur = overSub(cur, back); // cur is the front + } + } else if ((P % (2 * s)) == s) { + sendSub(cur, order[P - s]); + break; // absorbed into the front partner + } + } + + // The fully composited image now lives at position 0; deliver it to root. + if (rank == order[0]) { + if (rank == MPI_ROOT_RANK) { + write_image(subToFull(cur)); + } else { + sendSub(cur, MPI_ROOT_RANK); + } + } else if (rank == MPI_ROOT_RANK) { + write_image(subToFull(recvSub(order[0]))); + } +#else + (void)order_key; + write_image(subToFull(sub)); +#endif + } + + void Renderer::compositeFragAndWrite(FragImage&& frag, + const Scene& scene, + timestep_t step, + simtime_t time) const { + // fully-opaque cull is exact: mergeFrag only drops provably-occluded + // fragments, so the result is independent of the tree's grouping. + const real_t cull = ONE; + + // collapse a merged fragment image into a full premultiplied float frame + auto fragToFull = [&](const FragImage& f) -> std::vector { + const size_t npix = static_cast(m_width) * + static_cast(m_height); + std::vector full(npix * 4, ZERO); + for (int y = 0; y < f.h; ++y) { + for (int x = 0; x < f.w; ++x) { + const int fx = f.x0 + x, fy = f.y0 + y; + if (fx < 0 or fx >= m_width or fy < 0 or fy >= m_height) { + continue; + } + const size_t p = static_cast(y) * f.w + x; + const uint32_t k0 = f.offs[p], k1 = f.offs[p + 1]; + if (k1 <= k0) { + continue; + } + real_t out[4]; + out::fragOver(f.depth, f.rgba, k0, k1, out); + const size_t fi = (static_cast(fy) * m_width + fx) * 4; + full[fi + 0] = out[0]; + full[fi + 1] = out[1]; + full[fi + 2] = out[2]; + full[fi + 3] = out[3]; + } + } + return full; + }; + +#if defined(MPI_ENABLED) + int rank = 0, size = 1; + MPI_Comm_rank(MPI_COMM_WORLD, &rank); + MPI_Comm_size(MPI_COMM_WORLD, &size); + + if (size == 1) { + writeFrame(fragToFull(frag), scene, step, time); + return; + } + + constexpr int TAG_HDR = 7401; + constexpr int TAG_OFFS = 7402; + constexpr int TAG_DEPTH = 7403; + constexpr int TAG_RGBA = 7404; + + // Wire format: header {x0,y0,w,h,n_frag}; per-pixel prefix offsets (uint32); + // per-fragment depth (full real_t, so the cross-rank ordering key is exact) + // and premultiplied RGBA (uint8; only this adds ~1 LSB through the tree). + auto sendFrag = [&](const FragImage& s, int dest) { + const uint32_t nfrag = s.offs.empty() ? 0u : s.offs.back(); + // MPI counts are `int`; the RGBA payload has nfrag*4 elements, so a + // single message overflows int once nfrag > INT_MAX/4. That regime (a + // 4096^2 near-opaque-free dome on many ranks) needs the band-tiling + // optimization; fail loudly here rather than send a negative count. + raise::ErrorIf( + nfrag > 536870911u, + "dome A-buffer: per-message fragment count exceeds the MPI " + "int limit (nfrag*4 > INT_MAX). Lower render.resolution " + "(frame-band tiling is a pending optimization).", + HERE); + int hdr[5] = { s.x0, s.y0, s.w, s.h, static_cast(nfrag) }; + MPI_Send(hdr, 5, MPI_INT, dest, TAG_HDR, MPI_COMM_WORLD); + const int np = s.w * s.h; + if (np > 0) { + MPI_Send(s.offs.data(), np + 1, MPI_UINT32_T, dest, TAG_OFFS, MPI_COMM_WORLD); + } + if (nfrag > 0) { + // depth is sent at full real_t precision (NOT downcast to float): the + // depth key orders fragments across ranks, and local fragments keep + // real_t, so a float round-trip would make cross-rank vs within-rank + // ordering disagree at close depths -> a seam at the domain boundary. + MPI_Send(s.depth.data(), + static_cast(nfrag), + mpi::get_type(), + dest, + TAG_DEPTH, + MPI_COMM_WORLD); + std::vector bytes(static_cast(nfrag) * 4); + for (size_t i = 0; i < bytes.size(); ++i) { + bytes[i] = quantize(s.rgba[i]); + } + MPI_Send(bytes.data(), + static_cast(bytes.size()), + MPI_UNSIGNED_CHAR, + dest, + TAG_RGBA, + MPI_COMM_WORLD); + } + }; + auto recvFrag = [&](int src) -> FragImage { + int hdr[5]; + MPI_Recv(hdr, 5, MPI_INT, src, TAG_HDR, MPI_COMM_WORLD, MPI_STATUS_IGNORE); + FragImage s; + s.x0 = hdr[0]; + s.y0 = hdr[1]; + s.w = hdr[2]; + s.h = hdr[3]; + const uint32_t nfrag = static_cast(hdr[4]); + const int np = s.w * s.h; + if (np > 0) { + s.offs.resize(static_cast(np) + 1); + MPI_Recv(s.offs.data(), + np + 1, + MPI_UINT32_T, + src, + TAG_OFFS, + MPI_COMM_WORLD, + MPI_STATUS_IGNORE); + } + if (nfrag > 0) { + s.depth.resize(nfrag); + MPI_Recv(s.depth.data(), + static_cast(nfrag), + mpi::get_type(), + src, + TAG_DEPTH, + MPI_COMM_WORLD, + MPI_STATUS_IGNORE); + std::vector bytes(static_cast(nfrag) * 4); + MPI_Recv(bytes.data(), + static_cast(bytes.size()), + MPI_UNSIGNED_CHAR, + src, + TAG_RGBA, + MPI_COMM_WORLD, + MPI_STATUS_IGNORE); + s.rgba.resize(static_cast(nfrag) * 4); + const real_t inv255 = ONE / static_cast(255); + for (size_t i = 0; i < s.rgba.size(); ++i) { + s.rgba[i] = static_cast(bytes[i]) * inv255; + } + } + return s; + }; + + // Order-INDEPENDENT binary tree reduction: mergeFrag (depth-sorted merge + + // occlusion cull) is associative + commutative, so no global order is needed + // -- reduce straight to rank 0 (== MPI_ROOT_RANK). Each round, the lower + // partner receives + merges, the upper sends and drops out. + FragImage cur = std::move(frag); + for (int s = 1; s < size; s <<= 1) { + if ((rank % (2 * s)) == 0) { + const int pp = rank + s; + if (pp < size) { + FragImage back = recvFrag(pp); + cur = mergeFrag(cur, back, cull); + } + } else if ((rank % (2 * s)) == s) { + sendFrag(cur, rank - s); + break; + } + } + if (rank == MPI_ROOT_RANK) { + writeFrame(fragToFull(cur), scene, step, time); + } +#else + writeFrame(fragToFull(frag), scene, step, time); +#endif + } + +} // namespace out diff --git a/src/output/render/renderer.h b/src/output/render/renderer.h new file mode 100644 index 000000000..213ab04ef --- /dev/null +++ b/src/output/render/renderer.h @@ -0,0 +1,548 @@ +/** + * @file output/render/renderer.h + * @brief In-situ volume renderer: configuration, cadence, host composite & PNG + * @implements + * - out::Renderer + * - out::CameraDevice + * - out::TransferFunction + * - out::Scene + * @cpp: + * - render/renderer.cpp + * @namespaces: + * - out:: + * @macros: + * - MPI_ENABLED + * @note + * The Renderer is intentionally NOT templated on the engine/metric: it owns + * only metric-agnostic, host-side work (config parsing, cadence tracking, the + * MPI ordered composite and PNG encode). The device ray-march kernel and the + * field preparation live in the templated `Metadomain::Render`, mirroring + * the `out::Writer` (plain) / `Metadomain::Write` (templated) split. + */ + +#ifndef OUTPUT_RENDER_RENDERER_H +#define OUTPUT_RENDER_RENDERER_H + +#include "global.h" + +#include "arch/kokkos_aliases.h" +#include "utils/tools.h" + +#include "framework/parameters/parameters.h" + +#include +#include +#include + +namespace out { + + /** + * @brief Device-friendly POD camera + precomputed per-pixel ray basis. + * @note Trivially copyable; captured by value into the Kokkos kernel. + */ + struct CameraDevice { + // projection: 0 orthographic, 1 perspective (pinhole), 2 dome (fulldome + // azimuthal-equidistant fisheye from an interior eye). `orthographic` is + // kept for back-compat (== projection 0); the kernel branches on `projection`. + enum Projection : uint8_t { + Ortho = 0, + Perspective = 1, + Dome = 2 + }; + + real_t eye[3] { ZERO, ZERO, ZERO }; + real_t right[3] { ONE, ZERO, ZERO }; + real_t up[3] { ZERO, ONE, ZERO }; + // for the dome, `forward` is the ZENITH direction (center of the fisheye disk) + real_t forward[3] { ZERO, ZERO, -ONE }; + real_t tan_half_fov { ONE }; + real_t aspect { ONE }; + bool orthographic { true }; + real_t half_w { ONE }; + real_t half_h { ONE }; + int projection { Ortho }; + real_t dome_half_fov { static_cast( + 1.5707963267948966) }; // rad; 180 deg dome => PI/2 + // dome far-clip: rays stop at this world distance from the eye, so the + // sampled region is a half-ball (hemisphere) of this radius instead of the + // whole box -> no box corner/edge path-length artifacts. 0 => no clip. + real_t dome_radius { ZERO }; + }; + + /** + * @brief Per-scene transfer function: premultiplied RGBA device LUT + range. + */ + struct TransferFunction { + array_t lut; // device, (n_lut, 4), premultiplied RGBA + // opaque variant (alpha == 1 everywhere, so the premultiplied entries are + // straight RGB) used by the flat 2D slice rasterizer, where a single + // per-pixel sample should paint a solid heatmap rather than fade by opacity. + array_t lut_opaque; + int n_lut { 256 }; + real_t vmin { ZERO }; + real_t vmax { ONE }; + bool log_scale { false }; + std::string colormap { "viridis" }; // for redrawing the colorbar + }; + + /** + * @brief Configuration for the magnetic-field-line tube overlay. + * @note The lines are traced once per frame through a coarse, MPI-replicated + * copy of the (physical-basis) field, so every rank produces the same global + * polylines and renders only the segments inside its own domain; the existing + * ordered cross-domain composite then stitches them. The coarsening (`bin`) + * is what makes the replicate-and-trace cheap and avoids parallel particle + * advection. See output/render/fieldlines.h. + */ + struct FieldLineConfig { + bool enable { false }; // build the geometry this run + std::string field { "B" }; // vector field to trace: "B" | "E" | "J" + int bin { 4 }; // coarsening factor (cells/coarse cell), 2..8 + real_t seed_px { 8 }; // seed lattice spacing in screen pixels + real_t tube_px { 2 }; // tube radius in screen pixels + std::string colormap { "inferno" }; + // monochrome override: when this holds 3 entries [r,g,b] in [0,1] the lines + // are drawn in that single color instead of the |B| colormap (reads well as + // an overlay on a density/other volume). Empty => color by |B|. + std::vector color; + bool log_scale { false }; + real_t vmin { ZERO }; // tube color range; vmin>=vmax => auto |B| + real_t vmax { ZERO }; + real_t step_frac { static_cast(0.5) }; // RK4 step / coarse cell + int max_steps { 4000 }; // per-direction integration cap + real_t max_len_frac { static_cast(3) }; // x global box diagonal + int seed_max { 4096 }; // hard cap on seed count (spacing grows to fit) + // 2D only: number of evenly-spaced flux-function contour levels (field + // lines in 2D are iso-contours of the out-of-plane vector potential psi) + int levels { 16 }; + }; + + /** + * @brief Device-side 2D field-line geometry: the flux function psi on a coarse + * world grid, contoured per-pixel by the slice rasterizer. + * @note In 2D the in-plane field lines are the iso-contours of the flux + * function psi (Bx = d psi/dy, By = -d psi/dx). psi is integrated on a coarse, + * MPI-replicated copy of the field so the contour levels are global -> the + * lines are seamless across domains. The kernel draws a contour where psi is + * within a (screen-space) line width of a level, colored by |B| = |grad psi|. + */ + struct ContourSet { + array_t psi; // (n0*n1) flux function, c0-fastest + int n0 { 0 }, n1 { 0 }; + real_t origin0 { ZERO }, origin1 { ZERO }; + real_t dx0 { ONE }, dx1 { ONE }; + real_t dlevel { ONE }; // contour spacing in flux units + real_t psi_ref { ZERO }; // reference (zeroth) level + real_t line_half_px { ONE }; // half contour-line width, pixels + real_t wpp { ONE }; // world units per screen pixel + array_t lut; // opaque colormap, by |B| = |grad psi| + int n_lut { 256 }; + real_t vmin { ZERO }, vmax { ONE }; // |B| color range + bool enabled { false }; + std::string colormap { "inferno" }; // for the standalone colorbar + }; + + /** + * @brief Device-side field-line geometry handed to the ray-march kernel. + * @note A flat capsule list: each row is (p0xyz, p1xyz, s0, s1) with s the + * per-vertex scalar (|field|) used to color the tube. Opaque (alpha==1) LUT, + * so a tube sample paints a solid color and is composited inline exactly like + * the box spine. Empty (n_seg==0) on ranks no line touches. + */ + struct TubeSet { + array_t seg; // (n_seg, 8): p0, p1, s0, s1 in world coords + int n_seg { 0 }; + real_t radius { ZERO }; // world-space tube radius (ds floor applied) + array_t lut; // premultiplied RGBA, opaque (alpha==1) + int n_lut { 256 }; + real_t vmin { ZERO }, vmax { ONE }; + bool log_scale { false }; + std::string colormap { "inferno" }; // for the standalone colorbar + // uniform-grid bucket index (CSR) so a ray sample tests only the few + // segments in its cell instead of all of them. Bucketing on the coarse + // grid is exact because the tube radius is << one coarse cell; a segment is + // registered in every cell its radius-padded AABB overlaps. + array_t cell_start; // (ncell+1) prefix offsets into seg_idx + array_t seg_idx; // segment indices, grouped by cell + int gnc[3] { 1, 1, 1 }; + real_t gorigin[3] { ZERO, ZERO, ZERO }; + real_t gdx[3] { ONE, ONE, ONE }; + }; + + /** + * @brief One rendered scalar field -> one PNG stream. + * @note `field == "fieldlines"` is a standalone tube scene: no scalar volume + * is sampled (the field lines render against the background alone). Any other + * field with `show_fieldlines` true overlays the tubes inside its volume. + */ + struct Scene { + std::string field; // "N" | "Bmag" | "Vmag" | "Txy" | "B1" | "fieldlines" ... + std::string prefix; // PNG filename prefix, e.g. "Bmag_" + std::string label; // colorbar title (defaults to field) + std::vector ticks; // explicit colorbar tick values (optional) + bool show_fieldlines { false }; // overlay B-field tubes in the volume + TransferFunction tf; + }; + + /** + * @brief Fulldome fisheye ("planetarium dome master") projection parameters. + * @note When `enabled`, the 2D slice rasterizer ignores the linear world + * window and instead maps each pixel radially: the frame's inscribed circle + * is the dome, a pixel at normalized image radius rho in [0,1] is the dome + * zenith angle theta = rho * theta_max (azimuthal-equidistant image law -- + * the fulldome standard), and `law` picks how theta maps to a world radius r + * in a disk of radius `R` centered at (cx, cy). Pixels outside the inscribed + * circle are left transparent, so the corners are the dome master's black + * border. Cartesian 2D only (see Metadomain::Render); a metric-agnostic POD + * so the (templated) Render can copy it by value into the device kernel. + */ + struct DomeMap { + enum Law : uint8_t { + Equidistant = 0, + Gnomonic = 1, + Stereographic = 2, + Orthographic = 3 + }; + + bool enabled { false }; + int law { Equidistant }; + real_t theta_max { static_cast( + 1.5707963267948966) }; // dome half-FOV (rad) + real_t cx { ZERO }, cy { ZERO }; // world center of the cutout + real_t R { ONE }; // world radius of the cutout + }; + + /** + * @brief A sparse screen-space sub-image: the bounding box of one domain's + * projected footprint plus its premultiplied RGBA pixels. + * @note Each domain covers only a small part of the screen, so compositing + * these sparse boxes (not full frames) is what lets the renderer scale to + * thousands of ranks. + */ + struct SubImage { + int x0 { 0 }, y0 { 0 }; // top-left pixel in the full frame + int w { 0 }, h { 0 }; // bbox size in pixels (0 => empty) + std::vector rgba; // w*h*4 premultiplied, pixel-major + }; + + /** + * @brief A sparse screen-space *fragment* buffer for the interior-eye (dome) + * composite (A-buffer / deep image). + * @note With the camera inside the box there is no single global front-to-back + * domain order, so each domain contributes, per pixel, one depth-tagged + * premultiplied-RGBA fragment (its convex slab is one contiguous ray + * interval). Compositing is a per-pixel sort by `depth` (== ray t_enter) then + * front-to-back "over". The MPI reduce merges depth-sorted lists (associative + * + commutative), so no global order is needed. CSR layout because per-pixel + * fragment counts vary strongly across the frame. A leaf (one rank) holds 0/1 + * fragment per covered pixel. See composite.h::mergeFrag / fragOver. + */ + struct FragImage { + int x0 { 0 }, y0 { 0 }; // top-left pixel in the full frame + int w { 0 }, h { 0 }; // bbox size in pixels (0 => empty) + // per-pixel prefix offsets into depth/rgba, length w*h+1 (offs[0] == 0). + std::vector offs; + std::vector depth; // n_frag entries, ascending within each pixel + std::vector rgba; // n_frag*4 premultiplied, pixel-major + }; + + class Renderer { + public: + Renderer() {} + + ~Renderer() = default; + + Renderer(Renderer&&) = default; + + /** + * @brief Build the camera + per-scene LUTs from the `render.*` parameters. + * @param params simulation parameters (see ntt::params::Render) + * @param global_extent global physical box, for default camera framing + */ + void init(const ntt::SimulationParams& params, + const boundaries_t& global_extent); + + [[nodiscard]] + auto shouldRender(timestep_t step, simtime_t time) -> bool { + return m_enabled and m_tracker.shouldWrite(step, time); + } + + /** + * @brief Advance the moving view to `time`: translate the render region (and + * the 3D camera) by `moving_view.velocity * max(0, time - moving_view.start_time)`. + * @note A no-op unless `moving_view.velocity` was set. Call once per frame, before + * reading region()/camera(). All ranks pass the same time, so the + * shifted view is identical everywhere (the composite stays seamless). + */ + void updateForTime(simtime_t time); + + /** + * @brief Composite the per-rank sparse sub-image across MPI and write PNG. + * @param sub this rank's sparse screen-space sub-image (premultiplied RGBA) + * @param order_key this rank's front-to-back sort key (see composite.h) + * @param scene the scene being written (prefix, colorbar colormap/range/label) + * @param step current timestep (for the filename cycle number) + * @param time current simulation time (drawn as a corner label if enabled) + * @note Uses an order-preserving distributed tree reduce; only the MPI root + * rank assembles the full frame and writes the file. + */ + void compositeAndWrite(const SubImage& sub, + uint64_t order_key, + const Scene& scene, + timestep_t step, + simtime_t time) const; + + /** + * @brief Depth-resolved (A-buffer) composite for the interior-eye dome, + * then write the PNG. + * @param frag this rank's sparse screen-space fragment image (depth + RGBA) + * @param scene the scene being written + * @param step current timestep (for the filename cycle number) + * @param time current simulation time (drawn as a corner label if enabled) + * @note Order-independent: the tree reduce merges depth-sorted fragment + * lists (associative + commutative), so no global rank order is needed. The + * root collapses each pixel's list front-to-back and writes the file. + */ + void compositeFragAndWrite(FragImage&& frag, + const Scene& scene, + timestep_t step, + simtime_t time) const; + + /* getters -------------------------------------------------------------- */ + [[nodiscard]] + auto enabled() const -> bool { + return m_enabled; + } + + [[nodiscard]] + auto width() const -> int { + return m_width; + } + + [[nodiscard]] + auto height() const -> int { + return m_height; + } + + [[nodiscard]] + auto samples() const -> int { + return m_samples; + } + + [[nodiscard]] + auto stepSize() const -> real_t { + return m_step_size; + } + + [[nodiscard]] + auto earlyAlpha() const -> real_t { + return m_early_alpha; + } + + [[nodiscard]] + auto camera() const -> const CameraDevice& { + return m_camera_dev; + } + + // Optional axis-aligned render region in physical/world coords. Always + // resolved (unset axes default to the full global extent), so the driver + // can use these unconditionally. `hasRegion()` reports whether any axis was + // overridden (e.g. to know a crop is active). `d` in {0,1,2} == {x1,x2,x3}. + [[nodiscard]] + auto hasRegion() const -> bool { + return m_has_region; + } + + [[nodiscard]] + auto regionLo(int d) const -> real_t { + const int k = (d < 0) ? 0 : ((d > 2) ? 2 : d); + return (static_cast(k) < m_region.size()) ? m_region[k].first + : ZERO; + } + + [[nodiscard]] + auto regionHi(int d) const -> real_t { + const int k = (d < 0) ? 0 : ((d > 2) ? 2 : d); + return (static_cast(k) < m_region.size()) ? m_region[k].second + : ZERO; + } + + [[nodiscard]] + auto region() const -> const boundaries_t& { + return m_region; + } + + // 2D slice mode: mirror a spherical half-plane across the axis into a full + // disk (no effect on Cartesian or 3D rendering). + [[nodiscard]] + auto mirror() const -> bool { + return m_mirror; + } + + [[nodiscard]] + auto axes() const -> bool { + return m_axes; + } + + [[nodiscard]] + auto background(int i) const -> real_t { + return m_background[(i < 0) ? 0 : ((i > 2) ? 2 : i)]; + } + + // target 3D spine line width in pixels + [[nodiscard]] + auto spineWidth() const -> real_t { + return m_spine_width; + } + + // whether `render.axis_labels` was set (so the 2D path + // honors it instead of substituting per-metric defaults) + [[nodiscard]] + auto axisLabelsSet() const -> bool { + return m_axis_labels_set; + } + + [[nodiscard]] + auto axisLabel(int d) const -> const std::string& { + return m_axis_labels[(d < 0) ? 0 : ((d > 2) ? 2 : d)]; + } + + // Set the world window + axis names the 2D slice path maps onto the image, + // so the (host) axes overlay can label spatial coordinates. Called by the + // templated 2D Render before compositing; the window is constant per run. + void setSliceFrame(real_t u0, + real_t u1, + real_t v0, + real_t v1, + const std::string& xlabel, + const std::string& ylabel) { + m_slice_win[0] = u0; + m_slice_win[1] = u1; + m_slice_win[2] = v0; + m_slice_win[3] = v1; + m_slice_xlabel = xlabel; + m_slice_ylabel = ylabel; + } + + // Mark the 2D slice as curvilinear (spherical) so the axes are drawn polar: + // a radial "R" axis on the symmetry axis + a "Theta" arc, with a curvilinear + // spine. Set per-frame by the templated 2D Render (constant per run). + void setSlicePolar(bool polar, + real_t rmin, + real_t rmax, + real_t tmin, + real_t tmax, + bool mir) { + m_slice_polar = polar; + m_slice_rmin = rmin; + m_slice_rmax = rmax; + m_slice_tmin = tmin; + m_slice_tmax = tmax; + m_slice_pmirror = mir; + } + + [[nodiscard]] + auto scenes() const -> const std::vector& { + return m_scenes; + } + + [[nodiscard]] + auto fieldlines() const -> const FieldLineConfig& { + return m_fieldlines; + } + + // Fulldome fisheye projection (planetarium dome master). `dome()` carries the + // parsed config + resolved center/radius/FOV; `enabled` there reflects the + // toml + a 2D run. The templated Render only activates it for Cartesian, so + // it reports the per-run truth back via setDomeActive(), which the (metric- + // agnostic) compositeAndWrite reads to emit a clean square frame. + [[nodiscard]] + auto dome() const -> const DomeMap& { + return m_dome; + } + + void setDomeActive(bool active) { + m_dome_active = active; + } + + private: + // Composite a full premultiplied float frame over the background and write + // the PNG (with the colorbar / axes / time-label overlays). Shared by the + // ordered (SubImage) composite and the depth-resolved (FragImage) composite. + void writeFrame(const std::vector& img, + const Scene& scene, + timestep_t step, + simtime_t time) const; + + bool m_enabled { false }; + + int m_width { 1024 }; + int m_height { 1024 }; + int m_samples { 400 }; + real_t m_step_size { ZERO }; // world units/step; 0 => derive from samples + real_t m_early_alpha { static_cast(0.99) }; + int m_n_lut { 256 }; + // opaque background composited under the final image (shows through + // low-alpha pixels); defaults to black. + real_t m_background[3] { ZERO, ZERO, ZERO }; + // draw a colorbar (gradient + value ticks + label) on each PNG + bool m_colorbar { true }; + // draw the colorbar in an extended right margin (outside the render region) + // rather than overlaying it on top of the rendered volume + bool m_colorbar_outside { true }; + // 2D slice mode (spherical): mirror the meridional half-plane across the + // symmetry axis to render a full disk from one axisymmetric half + bool m_mirror { true }; + // draw the current simulation time as a label in the upper-right corner + bool m_time_label { false }; + // draw a spine (frame) + axis ticks/labels around the rendered region + bool m_axes { false }; + bool m_axis_labels_set { false }; + int m_axis_nticks { 5 }; + real_t m_spine_width { static_cast(2) }; // 3D spine width (px) + std::string m_axis_labels[3] { "x", "y", "z" }; + // global world box (2 or 3 axes); used to project the 3D axes box and to + // know the render mode (size 2 => 2D slice, size 3 => 3D volume). + boundaries_t m_global_extent; + // resolved render region [lo, hi] per axis (== global extent unless the + // user set render.extent.x{1,2,3}); the volume is clipped / the slice window is framed + // to this, and the default camera frames it. `m_region` is the CURRENT region + // (shifted by the moving view below); `m_region_base` is the static toml one. + boundaries_t m_region; + boundaries_t m_region_base; + bool m_has_region { false }; + + // moving view: after `m_cam_t0`, the render region and the 3D camera + // translate at `m_cam_vel` (world units per unit sim-time) to keep a + // propagating feature (e.g. a shock) in frame. `m_eye_base` is the static + // camera eye. See updateForTime(). + real_t m_cam_vel[3] { ZERO, ZERO, ZERO }; + simtime_t m_cam_t0 { 0 }; + bool m_cam_moving { false }; + real_t m_eye_base[3] { ZERO, ZERO, ZERO }; + // 2D slice world window + axis names, set per-frame by the templated Render + real_t m_slice_win[4] { ZERO, ONE, ZERO, ONE }; + std::string m_slice_xlabel { "x" }; + std::string m_slice_ylabel { "y" }; + // 2D curvilinear (spherical) slice: draw polar axes instead of Cartesian + bool m_slice_polar { false }; + real_t m_slice_rmin { ZERO }, m_slice_rmax { ONE }; + real_t m_slice_tmin { ZERO }, m_slice_tmax { ONE }; + bool m_slice_pmirror { false }; + + CameraDevice m_camera_dev; + std::vector m_scenes; + FieldLineConfig m_fieldlines; + + // fulldome fisheye config (parsed in init); m_dome_active is set per-run by + // the templated Render (true only for a 2D Cartesian dome). + DomeMap m_dome; + bool m_dome_active { false }; + + tools::Tracker m_tracker; + path_t m_root; + }; + +} // namespace out + +#endif // OUTPUT_RENDER_RENDERER_H diff --git a/src/output/render/slice2d.hpp b/src/output/render/slice2d.hpp new file mode 100644 index 000000000..198958746 --- /dev/null +++ b/src/output/render/slice2d.hpp @@ -0,0 +1,519 @@ +/** + * @file output/render/slice2d.hpp + * @brief Header-only Kokkos 2D slice rasterizer (one parallel_for over pixels) + * @implements + * - render::SliceRaster_kernel + * @namespaces: + * - render:: + * @note + * The 2D counterpart of the volume ray-march: a 2D simulation has no depth to + * integrate, so each screen pixel is a single inverse-mapped sample of the + * prepared scalar, painted opaque. Two coordinate families are handled at + * compile time via M::CoordType: + * - Cartesian (Minkowski 2D): the screen window IS the (x, y) physical plane; + * the inverse map is the per-axis code conversion. + * - Spherical / Qspherical (2D SR & all 2D GR): the screen window is the + * meridional (X, Z) Cartesian plane; a pixel maps to physical + * r = sqrt(X^2 + Z^2), theta = atan2(|X|, Z), then per-axis code conversion + * (separable once in physical spherical coords). With `mirror`, X<0 is the + * theta-reflected half, yielding a full disk from one axisymmetric half. + * + * Seamlessness: every rank shares the same global screen window, so a pixel's + * world point is identical on all ranks. Each pixel's active-region membership + * (code index in [0, n]) selects exactly one domain in the interior (boundary + * pixels may be claimed by two, but the halo-filled value is continuous there), + * so the disjoint sub-images composite without seams regardless of order. + */ + +#ifndef OUTPUT_RENDER_SLICE2D_HPP +#define OUTPUT_RENDER_SLICE2D_HPP + +#include "enums.h" +#include "global.h" + +#include "arch/kokkos_aliases.h" + +#include "output/render/renderer.h" + +namespace render { + using namespace ntt; + + template + class SliceRaster_kernel { + static constexpr auto D = M::Dim; + static_assert(D == Dim::_2D, "SliceRaster_kernel is 2D only"); + + randacc_ndfield_t Fld; + const idx_t comp; + const M metric; + + // global orthographic window in slice-plane world coords (shared by ranks) + const real_t umin, umax, vmin, vmax; + const int W, H; // full frame size (ray generation / ndc) + const int bx0, by0, bw; // screen-bbox offset and width (output stride) + const bool mirror; // spherical: paint the X<0 reflected half too + + // fulldome fisheye ("dome master"): when enabled, a pixel maps radially + // (azimuthal-equidistant image law) to a world point in a disk of radius + // `R` centered at (cx, cy); pixels outside the inscribed circle stay + // transparent. Cartesian only (host disables it for spherical). + const out::DomeMap dome; + + // optional physical render-region clip: a pixel is drawn only if its + // coordinate is inside [rx1lo,rx1hi] x [rx2lo,rx2hi] (x1,x2 == x,y for + // Cartesian; r,theta for spherical). Off => the whole domain is drawn. + const real_t rx1lo, rx1hi, rx2lo, rx2hi; + const bool region_clip; + + // local-domain active cell counts and View extents (membership + clamping) + const real_t n1, n2; + const int ext0, ext1; + + // transfer function (opaque LUT: premultiplied with alpha == 1) + array_t lut; + const int n_lut; + const real_t vlo, vhi; + const bool log_scale; + + // 2D field-line contours: iso-levels of the flux function psi on a coarse + // world grid, colored by |B| = |grad psi|. Cartesian only; drawn where psi + // is within `cline_half_px` (screen px) of a level. `heatmap_on` false -> + // standalone contours (no scalar fill). + array_t cpsi; + const int cn0, cn1; + const real_t corigin0, corigin1, cdx0, cdx1; + const real_t cdlevel, cpsi_ref, cline_half_px, cwpp; + array_t clut; + const int cn_lut; + const real_t cvmin, cvmax; + const bool contour_on; + + // spherical/Kerr field lines: traced meridional streamlines drawn as lines, + // reusing the 3D tube segment-bucket geometry at z == 0 (queried at the + // pixel's (X, Z) world point). Cartesian uses the contours above instead. + array_t lseg; + array_t lcell_start; + array_t lseg_idx; + const int ln_seg; + const real_t line_r2; + const int lgnc0, lgnc1, lgnc2; + const real_t lg0, lg1, lg2, ldx0, ldx1, ldx2; + array_t line_lut; + const int line_n_lut; + const real_t line_vmin, line_vmax; + const bool line_on; + + const bool heatmap_on; + + array_t image; // output, (bw*bh, 4) premultiplied RGBA + + public: + SliceRaster_kernel(const randacc_ndfield_t& Fld_, + idx_t comp_, + const M& metric_, + real_t umin_, + real_t umax_, + real_t vmin_, + real_t vmax_, + int W_, + int H_, + int bx0_, + int by0_, + int bw_, + bool mirror_, + const out::DomeMap& dome_, + real_t rx1lo_, + real_t rx1hi_, + real_t rx2lo_, + real_t rx2hi_, + bool region_clip_, + int n1_, + int n2_, + int ext0_, + int ext1_, + const array_t& lut_, + int n_lut_, + real_t vlo_, + real_t vhi_, + bool log_scale_, + const out::ContourSet& contours_, + const out::TubeSet& lines_, + bool heatmap_enabled_, + const array_t& image_) + : Fld { Fld_ } + , comp { comp_ } + , metric { metric_ } + , umin { umin_ } + , umax { umax_ } + , vmin { vmin_ } + , vmax { vmax_ } + , W { W_ } + , H { H_ } + , bx0 { bx0_ } + , by0 { by0_ } + , bw { bw_ } + , mirror { mirror_ } + , dome { dome_ } + , rx1lo { rx1lo_ } + , rx1hi { rx1hi_ } + , rx2lo { rx2lo_ } + , rx2hi { rx2hi_ } + , region_clip { region_clip_ } + , n1 { static_cast(n1_) } + , n2 { static_cast(n2_) } + , ext0 { ext0_ } + , ext1 { ext1_ } + , lut { lut_ } + , n_lut { n_lut_ } + , vlo { vlo_ } + , vhi { vhi_ } + , log_scale { log_scale_ } + , cpsi { contours_.psi } + , cn0 { contours_.n0 } + , cn1 { contours_.n1 } + , corigin0 { contours_.origin0 } + , corigin1 { contours_.origin1 } + , cdx0 { contours_.dx0 } + , cdx1 { contours_.dx1 } + , cdlevel { contours_.dlevel } + , cpsi_ref { contours_.psi_ref } + , cline_half_px { contours_.line_half_px } + , cwpp { contours_.wpp } + , clut { contours_.lut } + , cn_lut { contours_.n_lut } + , cvmin { contours_.vmin } + , cvmax { contours_.vmax } + , contour_on { contours_.enabled } + , lseg { lines_.seg } + , lcell_start { lines_.cell_start } + , lseg_idx { lines_.seg_idx } + , ln_seg { lines_.n_seg } + , line_r2 { lines_.radius * lines_.radius } + , lgnc0 { lines_.gnc[0] } + , lgnc1 { lines_.gnc[1] } + , lgnc2 { lines_.gnc[2] } + , lg0 { lines_.gorigin[0] } + , lg1 { lines_.gorigin[1] } + , lg2 { lines_.gorigin[2] } + , ldx0 { lines_.gdx[0] } + , ldx1 { lines_.gdx[1] } + , ldx2 { lines_.gdx[2] } + , line_lut { lines_.lut } + , line_n_lut { lines_.n_lut } + , line_vmin { lines_.vmin } + , line_vmax { lines_.vmax } + , line_on { lines_.n_seg > 0 } + , heatmap_on { heatmap_enabled_ } + , image { image_ } {} + + // bilinear sample of the prepared scalar at continuous code coords + // (cc1, cc2), reading the ghost halo for corners just outside the box. + Inline auto sample(real_t cc1, real_t cc2) const -> real_t { + const real_t g0 = cc1 - HALF; // cell-center continuous index + const real_t g1 = cc2 - HALF; + const real_t f0 = math::floor(g0); + const real_t f1 = math::floor(g1); + const real_t t0 = g0 - f0; + const real_t t1 = g1 - f1; + int b0 = static_cast(f0) + static_cast(N_GHOSTS); + int b1 = static_cast(f1) + static_cast(N_GHOSTS); + b0 = (b0 < 0) ? 0 : ((b0 > ext0 - 2) ? ext0 - 2 : b0); + b1 = (b1 < 0) ? 0 : ((b1 > ext1 - 2) ? ext1 - 2 : b1); + const real_t c00 = Fld(b0, b1, comp); + const real_t c10 = Fld(b0 + 1, b1, comp); + const real_t c01 = Fld(b0, b1 + 1, comp); + const real_t c11 = Fld(b0 + 1, b1 + 1, comp); + const real_t c0 = c00 * (ONE - t0) + c10 * t0; + const real_t c1 = c01 * (ONE - t0) + c11 * t0; + return c0 * (ONE - t1) + c1 * t1; + } + + // bilinear sample of the coarse flux function psi at world point (x, y) + Inline auto sampleFlux(real_t x, real_t y) const -> real_t { + if (cn0 <= 0 or cn1 <= 0) { + return ZERO; + } + int i0, i1, j0, j1; + real_t t0 = ZERO, t1 = ZERO; + if (cn0 <= 1) { + i0 = 0; + i1 = 0; + } else { + const real_t g0 = (x - corigin0) / cdx0 - HALF; + const real_t f0 = math::floor(g0); + int b0 = static_cast(f0); + t0 = g0 - f0; + if (b0 < 0) { + b0 = 0; + t0 = ZERO; + } else if (b0 > cn0 - 2) { + b0 = cn0 - 2; + t0 = ONE; + } + i0 = b0; + i1 = b0 + 1; + } + if (cn1 <= 1) { + j0 = 0; + j1 = 0; + } else { + const real_t g1 = (y - corigin1) / cdx1 - HALF; + const real_t f1 = math::floor(g1); + int b1 = static_cast(f1); + t1 = g1 - f1; + if (b1 < 0) { + b1 = 0; + t1 = ZERO; + } else if (b1 > cn1 - 2) { + b1 = cn1 - 2; + t1 = ONE; + } + j0 = b1; + j1 = b1 + 1; + } + const real_t c00 = cpsi(j0 * cn0 + i0); + const real_t c10 = cpsi(j0 * cn0 + i1); + const real_t c01 = cpsi(j1 * cn0 + i0); + const real_t c11 = cpsi(j1 * cn0 + i1); + const real_t c0 = c00 * (ONE - t0) + c10 * t0; + const real_t c1 = c01 * (ONE - t0) + c11 * t0; + return c0 * (ONE - t1) + c1 * t1; + } + + // is meridional world point (x, z) within the line width of any traced + // streamline segment? Same bucketed distance-to-segment test as the 3D + // tube kernel, restricted to the z == 0 plane. On a hit, `scalar` is |B|. + Inline auto inLine(real_t x, real_t z, real_t& scalar) const -> bool { + if (ln_seg <= 0) { + return false; + } + const int c0 = static_cast(math::floor((x - lg0) / ldx0)); + const int c1 = static_cast(math::floor((z - lg1) / ldx1)); + const int c2 = static_cast(math::floor((ZERO - lg2) / ldx2)); + if (c0 < 0 or c0 >= lgnc0 or c1 < 0 or c1 >= lgnc1 or c2 < 0 or c2 >= lgnc2) { + return false; + } + const int lin = (c2 * lgnc1 + c1) * lgnc0 + c0; + const int kb = lcell_start(lin); + const int ke = lcell_start(lin + 1); + real_t best = line_r2; + bool hit = false; + for (int k = kb; k < ke; ++k) { + const int s = lseg_idx(k); + const real_t ax = lseg(s, 0), az = lseg(s, 1); // (X, Z) in slots 0,1 + const real_t bx = lseg(s, 3), bz = lseg(s, 4); + const real_t ex = bx - ax, ez = bz - az; + const real_t wx = x - ax, wz = z - az; + const real_t ee = ex * ex + ez * ez; + real_t tt = (ee > ZERO) ? (wx * ex + wz * ez) / ee : ZERO; + tt = (tt < ZERO) ? ZERO : ((tt > ONE) ? ONE : tt); + const real_t cx = ax + tt * ex, cz = az + tt * ez; + const real_t dx = x - cx, dz = z - cz; + const real_t d2 = dx * dx + dz * dz; + if (d2 < best) { + best = d2; + scalar = lseg(s, 6) * (ONE - tt) + lseg(s, 7) * tt; + hit = true; + } + } + return hit; + } + + Inline void operator()(cellidx_t lpx, cellidx_t lpy) const { + const auto pix = static_cast(lpy) * static_cast(bw) + + static_cast(lpx); + const int gpx = bx0 + static_cast(lpx); + const int gpy = by0 + static_cast(lpy); + // default transparent + image(pix, 0) = ZERO; + image(pix, 1) = ZERO; + image(pix, 2) = ZERO; + image(pix, 3) = ZERO; + + // pixel center -> slice-plane world coords (v flipped so +v is up). + // Cartesian dome mode maps the pixel radially (fisheye) instead, leaving + // the corners (outside the inscribed circle) transparent. Curvilinear + // slices are already a meridional disk, so they keep the linear window + // even in dome mode -> the fisheye is compile-time gated to Cartesian. + real_t u, v; + bool dome_fisheye = false; + if constexpr (M::CoordType == Coord::Cartesian) { + dome_fisheye = dome.enabled; + } + if (dome_fisheye) { + const real_t cxp = HALF * static_cast(W); + const real_t cyp = HALF * static_cast(H); + const real_t Rpx = HALF * static_cast((W < H) ? W : H); + const real_t dxp = (static_cast(gpx) + HALF) - cxp; + const real_t dyp = cyp - (static_cast(gpy) + HALF); // +y up + const real_t rho = math::sqrt(dxp * dxp + dyp * dyp) / Rpx; + if (rho > ONE) { + return; // outside the inscribed dome circle -> transparent border + } + const real_t phi = math::atan2(dyp, dxp); + const real_t theta = rho * dome.theta_max; // dome zenith angle + // normalized world radius fr = r / R for the chosen plane<->dome law + real_t fr; + if (dome.law == out::DomeMap::Gnomonic) { + const real_t tm = math::tan(dome.theta_max); + fr = (tm > ZERO) ? (math::tan(theta) / tm) : rho; + } else if (dome.law == out::DomeMap::Stereographic) { + const real_t tm = math::tan(HALF * dome.theta_max); + fr = (tm > ZERO) ? (math::tan(HALF * theta) / tm) : rho; + } else if (dome.law == out::DomeMap::Orthographic) { + const real_t sm = math::sin(dome.theta_max); + fr = (sm > ZERO) ? (math::sin(theta) / sm) : rho; + } else { // Equidistant (fulldome standard): r = R * theta/theta_max + fr = rho; + } + const real_t r = dome.R * fr; + u = dome.cx + r * math::cos(phi); + v = dome.cy + r * math::sin(phi); + } else { + u = umin + (static_cast(gpx) + HALF) / static_cast(W) * + (umax - umin); + v = vmax - (static_cast(gpy) + HALF) / static_cast(H) * + (vmax - vmin); + } + + // world -> continuous local code coords, with an optional physical + // render-region clip (so a crop hides domain data outside the region, not + // just reframes the view). The dome's own circular cutout replaces the + // rectangular clip, so it is skipped in dome mode. + real_t cc1, cc2; + if constexpr (M::CoordType == Coord::Cartesian) { + if (region_clip and not dome.enabled and + (u < rx1lo or u > rx1hi or v < rx2lo or v > rx2hi)) { + return; + } + cc1 = metric.template convert<1, Crd::Ph, Crd::Cd>(u); + cc2 = metric.template convert<2, Crd::Ph, Crd::Cd>(v); + } else { + if (not mirror and u < ZERO) { + return; // only the X>=0 meridional half is physical + } + const real_t r = math::sqrt(u * u + v * v); + const real_t th = math::atan2(math::abs(u), v); // in [0, pi] + if (region_clip and (r < rx1lo or r > rx1hi or th < rx2lo or th > rx2hi)) { + return; + } + cc1 = metric.template convert<1, Crd::Ph, Crd::Cd>(r); + cc2 = metric.template convert<2, Crd::Ph, Crd::Cd>(th); + } + // active-region membership (inclusive so interiors gap-free, boundaries + // shared harmlessly); outside -> leave transparent + if (cc1 < ZERO or cc1 > n1 or cc2 < ZERO or cc2 > n2) { + return; + } + + real_t cr = ZERO, cg = ZERO, cb = ZERO; + bool painted = false; + if (heatmap_on) { + const real_t s = sample(cc1, cc2); + // normalize through the transfer-function range + const real_t inv_range = (vhi > vlo) ? (ONE / (vhi - vlo)) : ZERO; + real_t uu; + if (log_scale) { + const real_t log_vlo = math::log10(vlo); + uu = (s > ZERO) ? (math::log10(s) - log_vlo) * inv_range : -ONE; + } else { + uu = (s - vlo) * inv_range; + } + if (uu < ZERO) { + uu = ZERO; + } else if (uu > ONE) { + uu = ONE; + } + int idx = static_cast(uu * static_cast(n_lut - 1) + HALF); + if (idx < 0) { + idx = 0; + } else if (idx > n_lut - 1) { + idx = n_lut - 1; + } + // opaque LUT: premultiplied with alpha == 1, so this is straight RGB + cr = lut(idx, 0); + cg = lut(idx, 1); + cb = lut(idx, 2); + painted = true; + } + // field-line contours: iso-levels of the flux function psi (Cartesian + // only, where (u, v) IS the (x, y) world plane). Drawn over the heatmap + // and colored by |B| = |grad psi|; a screen-space line width keeps the + // contours ~constant thickness regardless of local gradient. + if constexpr (M::CoordType == Coord::Cartesian) { + if (contour_on) { + const real_t psi0 = sampleFlux(u, v); + const real_t pl = sampleFlux(u - cwpp, v); + const real_t pr = sampleFlux(u + cwpp, v); + const real_t pd = sampleFlux(u, v - cwpp); + const real_t pup = sampleFlux(u, v + cwpp); + const real_t gx = (pr - pl) / (TWO * cwpp); + const real_t gy = (pup - pd) / (TWO * cwpp); + const real_t g = math::sqrt(gx * gx + gy * gy); // |B| + const real_t tlev = (cdlevel > ZERO) ? (psi0 - cpsi_ref) / cdlevel : ZERO; + const real_t nlev = math::floor(tlev + HALF); // nearest level index + const real_t d_world = math::abs(psi0 - (cpsi_ref + nlev * cdlevel)); + const real_t eps = static_cast(1e-30); + const real_t d_screen = (g > eps) ? (d_world / (g * cwpp)) + : static_cast(1e30); + if (d_screen <= cline_half_px) { + const real_t invr = (cvmax > cvmin) ? (ONE / (cvmax - cvmin)) : ZERO; + real_t uu = (g - cvmin) * invr; + if (uu < ZERO) { + uu = ZERO; + } else if (uu > ONE) { + uu = ONE; + } + int idx = static_cast(uu * static_cast(cn_lut - 1) + HALF); + if (idx < 0) { + idx = 0; + } else if (idx > cn_lut - 1) { + idx = cn_lut - 1; + } + cr = clut(idx, 0); + cg = clut(idx, 1); + cb = clut(idx, 2); + painted = true; + } + } + } else { + // spherical/Kerr: traced meridional streamlines. (u, v) is the (X, Z) + // world point; draw a line where it falls within a segment's width. + if (line_on) { + real_t sB; + if (inLine(u, v, sB)) { + const real_t invr = (line_vmax > line_vmin) + ? (ONE / (line_vmax - line_vmin)) + : ZERO; + real_t uu = (sB - line_vmin) * invr; + if (uu < ZERO) { + uu = ZERO; + } else if (uu > ONE) { + uu = ONE; + } + int idx = static_cast( + uu * static_cast(line_n_lut - 1) + HALF); + if (idx < 0) { + idx = 0; + } else if (idx > line_n_lut - 1) { + idx = line_n_lut - 1; + } + cr = line_lut(idx, 0); + cg = line_lut(idx, 1); + cb = line_lut(idx, 2); + painted = true; + } + } + } + if (painted) { + image(pix, 0) = cr; + image(pix, 1) = cg; + image(pix, 2) = cb; + image(pix, 3) = ONE; + } + } + }; + +} // namespace render + +#endif // OUTPUT_RENDER_SLICE2D_HPP diff --git a/src/output/render/transfer_fn.h b/src/output/render/transfer_fn.h new file mode 100644 index 000000000..3b99a3a74 --- /dev/null +++ b/src/output/render/transfer_fn.h @@ -0,0 +1,624 @@ +/** + * @file output/render/transfer_fn.h + * @brief Colormap tables and premultiplied RGBA look-up-table builder + * @implements + * - out::buildLUT + * - out::colormapRGB + * @namespaces: + * - out:: + * @note + * Colormaps are stored as a handful of anchor colors and linearly + * interpolated; this is visually indistinguishable from the full 256-entry + * matplotlib tables for volume rendering while keeping the header compact. + * The LUT is built on the host and deep-copied to a device View of shape + * (N_LUT, 4) holding premultiplied RGBA (R=r*a, G=g*a, B=b*a, A=a). + */ + +#ifndef OUTPUT_RENDER_TRANSFER_FN_H +#define OUTPUT_RENDER_TRANSFER_FN_H + +#include "global.h" + +#include "arch/kokkos_aliases.h" +#include "utils/numeric.h" + +#include +#include +#include + +namespace out { + + namespace cmap_hidden { + + // anchor colors sampled at uniform positions in [0, 1] + struct Anchors { + const float (*rgb)[3]; + int n; + }; + + inline constexpr float viridis[9][3] = { + { 0.267004f, 0.004874f, 0.329415f }, + { 0.282623f, 0.140926f, 0.457517f }, + { 0.253935f, 0.265254f, 0.529983f }, + { 0.206756f, 0.371758f, 0.553117f }, + { 0.163625f, 0.471133f, 0.558148f }, + { 0.127568f, 0.566949f, 0.550556f }, + { 0.134692f, 0.658636f, 0.517649f }, + { 0.477504f, 0.821444f, 0.318195f }, + { 0.993248f, 0.906157f, 0.143936f }, + }; + + inline constexpr float inferno[9][3] = { + { 0.001462f, 0.000466f, 0.013866f }, + { 0.087411f, 0.044556f, 0.224813f }, + { 0.258234f, 0.038571f, 0.406485f }, + { 0.416331f, 0.090203f, 0.432943f }, + { 0.578304f, 0.148039f, 0.404411f }, + { 0.735683f, 0.215906f, 0.330245f }, + { 0.865006f, 0.316822f, 0.226055f }, + { 0.954506f, 0.468744f, 0.099874f }, + { 0.988362f, 0.998364f, 0.644924f }, + }; + + inline constexpr float plasma[9][3] = { + { 0.050383f, 0.029803f, 0.527975f }, + { 0.287076f, 0.010855f, 0.627295f }, + { 0.417642f, 0.000564f, 0.658390f }, + { 0.562738f, 0.051545f, 0.641509f }, + { 0.692840f, 0.165141f, 0.564522f }, + { 0.798216f, 0.280197f, 0.469538f }, + { 0.881443f, 0.392529f, 0.383229f }, + { 0.949217f, 0.517763f, 0.295662f }, + { 0.940015f, 0.975158f, 0.131326f }, + }; + + // Moreland cool-to-warm diverging + inline constexpr float cool2warm[3][3] = { + { 0.230f, 0.299f, 0.754f }, + { 0.865f, 0.865f, 0.865f }, + { 0.706f, 0.016f, 0.150f }, + }; + + inline constexpr float gray[2][3] = { + { 0.0f, 0.0f, 0.0f }, + { 1.0f, 1.0f, 1.0f }, + }; + + // ----------------------------------------------------------------------- + // CMasher scientific colormaps (https://cmasher.readthedocs.io) + // + // The following anchor tables are uniform downsamplings (33 anchors) of the + // published CMasher colormap data, re-implemented from the source at + // https://github.com/1313e/CMasher (src/cmasher/colormaps//_norm.txt). + // At 33 anchors the linear-interpolation error versus the full 256/511-entry + // tables is < 4.5/255 for every map, i.e. visually indistinguishable. + // + // CMasher is distributed under the BSD 3-Clause License: + // Copyright (c) 2019-2021, Ellert van der Velden + // All rights reserved. + // Redistribution and use in source and binary forms, with or without + // modification, are permitted provided that the copyright notice, this list + // of conditions and the BSD-3-Clause disclaimer are retained. The name of the + // copyright holder may not be used to endorse products without permission. + // ----------------------------------------------------------------------- + + // cmasher::dusk (33 anchors sampled from the 256-entry table) + inline constexpr float dusk[33][3] = { + { 0.000000f, 0.000000f, 0.000000f }, + { 0.006238f, 0.007290f, 0.012708f }, + { 0.018622f, 0.026187f, 0.053076f }, + { 0.029900f, 0.055366f, 0.101787f }, + { 0.030149f, 0.086842f, 0.147576f }, + { 0.015542f, 0.120312f, 0.179819f }, + { 0.010319f, 0.152335f, 0.193847f }, + { 0.029326f, 0.181190f, 0.200163f }, + { 0.066363f, 0.207841f, 0.204068f }, + { 0.103996f, 0.233120f, 0.206493f }, + { 0.141503f, 0.257403f, 0.206888f }, + { 0.180382f, 0.280651f, 0.204334f }, + { 0.222140f, 0.302532f, 0.198293f }, + { 0.267566f, 0.322632f, 0.189012f }, + { 0.316579f, 0.340653f, 0.177441f }, + { 0.368580f, 0.356469f, 0.164931f }, + { 0.422824f, 0.370100f, 0.153037f }, + { 0.471588f, 0.380324f, 0.144475f }, + { 0.528296f, 0.390215f, 0.138408f }, + { 0.585639f, 0.398375f, 0.137901f }, + { 0.643321f, 0.404994f, 0.144156f }, + { 0.701148f, 0.410242f, 0.157700f }, + { 0.758937f, 0.414304f, 0.178666f }, + { 0.816278f, 0.417544f, 0.207638f }, + { 0.871717f, 0.421238f, 0.247431f }, + { 0.918409f, 0.431732f, 0.307531f }, + { 0.940305f, 0.464843f, 0.388730f }, + { 0.947103f, 0.510940f, 0.465808f }, + { 0.949229f, 0.559258f, 0.537097f }, + { 0.949161f, 0.607586f, 0.604625f }, + { 0.947863f, 0.655456f, 0.669259f }, + { 0.945994f, 0.702786f, 0.731248f }, + { 0.944208f, 0.749586f, 0.790456f }, + }; + + // cmasher::cosmic (33 anchors sampled from the 256-entry table) + inline constexpr float cosmic[33][3] = { + { 0.000000f, 0.000000f, 0.000000f }, + { 0.010239f, 0.006872f, 0.013172f }, + { 0.038809f, 0.022373f, 0.054399f }, + { 0.076237f, 0.043279f, 0.104947f }, + { 0.112700f, 0.063138f, 0.158384f }, + { 0.148869f, 0.079429f, 0.215952f }, + { 0.185079f, 0.091867f, 0.278775f }, + { 0.221470f, 0.099655f, 0.348020f }, + { 0.257982f, 0.101322f, 0.424930f }, + { 0.294209f, 0.094291f, 0.510709f }, + { 0.328977f, 0.074083f, 0.605931f }, + { 0.359190f, 0.034502f, 0.708287f }, + { 0.377711f, 0.016748f, 0.805820f }, + { 0.375894f, 0.098454f, 0.873588f }, + { 0.355826f, 0.192411f, 0.902268f }, + { 0.326497f, 0.271620f, 0.906235f }, + { 0.293925f, 0.337879f, 0.899030f }, + { 0.265164f, 0.388114f, 0.889366f }, + { 0.233446f, 0.439282f, 0.877747f }, + { 0.203862f, 0.485773f, 0.867189f }, + { 0.176988f, 0.529060f, 0.858475f }, + { 0.153054f, 0.570207f, 0.851855f }, + { 0.131837f, 0.609999f, 0.847280f }, + { 0.112515f, 0.649034f, 0.844504f }, + { 0.093584f, 0.687770f, 0.843135f }, + { 0.072938f, 0.726550f, 0.842663f }, + { 0.048159f, 0.765608f, 0.842480f }, + { 0.021220f, 0.805066f, 0.841902f }, + { 0.009956f, 0.844909f, 0.840192f }, + { 0.039372f, 0.884930f, 0.836603f }, + { 0.115052f, 0.924556f, 0.830460f }, + { 0.218056f, 0.962249f, 0.821634f }, + { 0.371763f, 0.992456f, 0.816521f }, + }; + + // cmasher::freeze (33 anchors sampled from the 256-entry table) + inline constexpr float freeze[33][3] = { + { 0.000000f, 0.000000f, 0.000000f }, + { 0.010910f, 0.009007f, 0.014906f }, + { 0.039247f, 0.030823f, 0.059160f }, + { 0.074148f, 0.060008f, 0.110444f }, + { 0.106606f, 0.087232f, 0.163635f }, + { 0.137212f, 0.112671f, 0.219689f }, + { 0.166145f, 0.136733f, 0.279301f }, + { 0.193344f, 0.159680f, 0.343042f }, + { 0.218514f, 0.181739f, 0.411377f }, + { 0.241052f, 0.203213f, 0.484587f }, + { 0.259859f, 0.224652f, 0.562539f }, + { 0.272953f, 0.247192f, 0.644102f }, + { 0.276779f, 0.273147f, 0.725721f }, + { 0.265827f, 0.306369f, 0.798716f }, + { 0.236083f, 0.349541f, 0.849696f }, + { 0.192331f, 0.398903f, 0.873582f }, + { 0.145422f, 0.448298f, 0.878894f }, + { 0.112009f, 0.489316f, 0.875932f }, + { 0.099232f, 0.533336f, 0.869029f }, + { 0.122966f, 0.574696f, 0.861171f }, + { 0.170084f, 0.613909f, 0.853779f }, + { 0.227234f, 0.651383f, 0.847515f }, + { 0.289300f, 0.687372f, 0.842698f }, + { 0.355048f, 0.721960f, 0.839586f }, + { 0.424505f, 0.755075f, 0.838630f }, + { 0.497701f, 0.786577f, 0.840759f }, + { 0.573685f, 0.816502f, 0.847460f }, + { 0.650307f, 0.845347f, 0.860140f }, + { 0.725396f, 0.873979f, 0.879132f }, + { 0.797886f, 0.903229f, 0.903669f }, + { 0.867683f, 0.933696f, 0.932619f }, + { 0.935038f, 0.965809f, 0.964980f }, + { 1.000000f, 1.000000f, 1.000000f }, + }; + + // cmasher::apple (33 anchors sampled from the 256-entry table) + inline constexpr float apple[33][3] = { + { 0.000000f, 0.000000f, 0.000000f }, + { 0.018449f, 0.006683f, 0.008999f }, + { 0.069836f, 0.019462f, 0.030130f }, + { 0.124822f, 0.033495f, 0.057048f }, + { 0.180264f, 0.045068f, 0.080127f }, + { 0.236726f, 0.050851f, 0.098639f }, + { 0.294404f, 0.049708f, 0.111860f }, + { 0.353148f, 0.039763f, 0.118281f }, + { 0.412085f, 0.021350f, 0.115115f }, + { 0.467944f, 0.008973f, 0.098099f }, + { 0.512805f, 0.042578f, 0.069239f }, + { 0.544883f, 0.107236f, 0.041311f }, + { 0.569454f, 0.165936f, 0.019988f }, + { 0.589142f, 0.220295f, 0.007127f }, + { 0.604821f, 0.272232f, 0.001856f }, + { 0.616738f, 0.322887f, 0.005791f }, + { 0.624881f, 0.372928f, 0.022422f }, + { 0.628812f, 0.416517f, 0.050591f }, + { 0.629507f, 0.466302f, 0.088693f }, + { 0.625985f, 0.516149f, 0.130045f }, + { 0.618051f, 0.566092f, 0.175071f }, + { 0.605455f, 0.616124f, 0.224400f }, + { 0.587974f, 0.666152f, 0.279195f }, + { 0.566011f, 0.715805f, 0.341631f }, + { 0.543179f, 0.763859f, 0.415364f }, + { 0.534328f, 0.806916f, 0.503615f }, + { 0.564391f, 0.840721f, 0.598311f }, + { 0.628027f, 0.867449f, 0.684639f }, + { 0.703665f, 0.891763f, 0.760351f }, + { 0.781104f, 0.916124f, 0.828098f }, + { 0.857052f, 0.941725f, 0.890058f }, + { 0.930451f, 0.969331f, 0.947434f }, + { 1.000000f, 1.000000f, 1.000000f }, + }; + + // cmasher::gothic (33 anchors sampled from the 256-entry table) + inline constexpr float gothic[33][3] = { + { 0.000000f, 0.000000f, 0.000000f }, + { 0.009103f, 0.009497f, 0.017125f }, + { 0.031259f, 0.032684f, 0.068646f }, + { 0.061552f, 0.062439f, 0.127729f }, + { 0.091294f, 0.088771f, 0.191076f }, + { 0.121658f, 0.111295f, 0.260048f }, + { 0.154308f, 0.129160f, 0.335784f }, + { 0.191149f, 0.140572f, 0.419137f }, + { 0.234428f, 0.142239f, 0.510129f }, + { 0.286418f, 0.128494f, 0.605972f }, + { 0.347449f, 0.092109f, 0.695963f }, + { 0.411712f, 0.037080f, 0.759733f }, + { 0.471546f, 0.020266f, 0.789424f }, + { 0.526274f, 0.060551f, 0.796780f }, + { 0.578066f, 0.108541f, 0.793004f }, + { 0.628368f, 0.151332f, 0.783641f }, + { 0.677621f, 0.190718f, 0.770763f }, + { 0.719390f, 0.224799f, 0.756764f }, + { 0.763054f, 0.267998f, 0.736645f }, + { 0.790627f, 0.328361f, 0.713928f }, + { 0.791142f, 0.402697f, 0.719584f }, + { 0.787649f, 0.467602f, 0.747951f }, + { 0.784959f, 0.526150f, 0.782754f }, + { 0.783559f, 0.580954f, 0.819616f }, + { 0.783736f, 0.633370f, 0.856997f }, + { 0.785586f, 0.684344f, 0.893939f }, + { 0.789112f, 0.734726f, 0.928646f }, + { 0.795666f, 0.785049f, 0.956164f }, + { 0.813132f, 0.833710f, 0.968235f }, + { 0.848938f, 0.877750f, 0.970393f }, + { 0.895684f, 0.918861f, 0.974579f }, + { 0.947082f, 0.959196f, 0.984058f }, + { 1.000000f, 1.000000f, 1.000000f }, + }; + + // cmasher::sunburst (33 anchors sampled from the 256-entry table) + inline constexpr float sunburst[33][3] = { + { 0.000000f, 0.000000f, 0.000000f }, + { 0.014374f, 0.007761f, 0.013830f }, + { 0.057181f, 0.023825f, 0.051569f }, + { 0.107503f, 0.043062f, 0.091435f }, + { 0.159109f, 0.059433f, 0.125622f }, + { 0.212012f, 0.071665f, 0.153805f }, + { 0.266014f, 0.080343f, 0.175895f }, + { 0.320933f, 0.085763f, 0.191981f }, + { 0.376633f, 0.087997f, 0.202182f }, + { 0.432991f, 0.086959f, 0.206564f }, + { 0.489851f, 0.082504f, 0.205087f }, + { 0.546970f, 0.074631f, 0.197554f }, + { 0.603921f, 0.064180f, 0.183549f }, + { 0.659892f, 0.055133f, 0.162324f }, + { 0.713225f, 0.059572f, 0.132732f }, + { 0.760503f, 0.093519f, 0.093993f }, + { 0.796917f, 0.154735f, 0.050648f }, + { 0.819328f, 0.216009f, 0.025564f }, + { 0.837747f, 0.284012f, 0.033882f }, + { 0.851299f, 0.348017f, 0.074896f }, + { 0.861316f, 0.408787f, 0.125585f }, + { 0.868446f, 0.467188f, 0.180749f }, + { 0.873123f, 0.523806f, 0.239715f }, + { 0.875769f, 0.578986f, 0.302576f }, + { 0.876914f, 0.632883f, 0.369564f }, + { 0.877316f, 0.685489f, 0.440900f }, + { 0.878101f, 0.736650f, 0.516675f }, + { 0.880900f, 0.786078f, 0.596685f }, + { 0.887890f, 0.833402f, 0.680181f }, + { 0.901543f, 0.878304f, 0.765633f }, + { 0.924079f, 0.920687f, 0.850654f }, + { 0.957291f, 0.960697f, 0.931527f }, + { 1.000000f, 1.000000f, 1.000000f }, + }; + + // cmasher::voltage (33 anchors sampled from the 256-entry table) + inline constexpr float voltage[33][3] = { + { 0.000000f, 0.000000f, 0.000000f }, + { 0.015829f, 0.007377f, 0.012045f }, + { 0.060346f, 0.022787f, 0.046824f }, + { 0.109165f, 0.041971f, 0.090162f }, + { 0.157514f, 0.059030f, 0.134981f }, + { 0.205921f, 0.071648f, 0.182831f }, + { 0.254512f, 0.079556f, 0.235168f }, + { 0.303074f, 0.082057f, 0.293555f }, + { 0.350914f, 0.078204f, 0.359600f }, + { 0.396550f, 0.067768f, 0.434356f }, + { 0.437427f, 0.055503f, 0.516639f }, + { 0.470517f, 0.059198f, 0.601227f }, + { 0.494085f, 0.093088f, 0.680818f }, + { 0.508421f, 0.145253f, 0.750719f }, + { 0.514803f, 0.203410f, 0.809898f }, + { 0.514558f, 0.262635f, 0.859195f }, + { 0.508787f, 0.321230f, 0.899892f }, + { 0.499930f, 0.371547f, 0.929319f }, + { 0.486263f, 0.427841f, 0.956532f }, + { 0.469915f, 0.482826f, 0.977159f }, + { 0.452770f, 0.536437f, 0.991010f }, + { 0.438275f, 0.588421f, 0.997509f }, + { 0.432385f, 0.638191f, 0.996018f }, + { 0.443151f, 0.684778f, 0.986745f }, + { 0.476274f, 0.727179f, 0.972136f }, + { 0.529358f, 0.765229f, 0.956912f }, + { 0.594140f, 0.799927f, 0.945473f }, + { 0.663403f, 0.832718f, 0.939967f }, + { 0.733296f, 0.864814f, 0.940775f }, + { 0.802260f, 0.897066f, 0.947563f }, + { 0.869795f, 0.930059f, 0.959867f }, + { 0.935781f, 0.964231f, 0.977340f }, + { 1.000000f, 1.000000f, 1.000000f }, + }; + + // cmasher::ocean (33 anchors sampled from the 256-entry table) + inline constexpr float ocean[33][3] = { + { 0.110363f, 0.001691f, 0.253026f }, + { 0.124516f, 0.043066f, 0.289045f }, + { 0.135665f, 0.084868f, 0.324305f }, + { 0.143805f, 0.121839f, 0.358190f }, + { 0.148976f, 0.156972f, 0.390248f }, + { 0.151327f, 0.191269f, 0.420093f }, + { 0.151211f, 0.225122f, 0.447430f }, + { 0.149271f, 0.258662f, 0.472116f }, + { 0.146471f, 0.291895f, 0.494196f }, + { 0.144043f, 0.324790f, 0.513894f }, + { 0.143346f, 0.357322f, 0.531555f }, + { 0.145639f, 0.389496f, 0.547565f }, + { 0.151848f, 0.421347f, 0.562289f }, + { 0.162417f, 0.452929f, 0.576030f }, + { 0.177344f, 0.484306f, 0.589017f }, + { 0.196368f, 0.515533f, 0.601409f }, + { 0.219188f, 0.546652f, 0.613299f }, + { 0.242129f, 0.573802f, 0.623330f }, + { 0.271793f, 0.604715f, 0.634385f }, + { 0.305461f, 0.635419f, 0.645036f }, + { 0.343819f, 0.665728f, 0.655393f }, + { 0.387893f, 0.695314f, 0.665825f }, + { 0.438718f, 0.723698f, 0.677325f }, + { 0.496053f, 0.750479f, 0.691913f }, + { 0.556972f, 0.775909f, 0.711830f }, + { 0.617772f, 0.800881f, 0.737580f }, + { 0.676718f, 0.826177f, 0.768091f }, + { 0.733666f, 0.852223f, 0.802113f }, + { 0.788975f, 0.879240f, 0.838730f }, + { 0.843051f, 0.907368f, 0.877312f }, + { 0.896212f, 0.936732f, 0.917381f }, + { 0.948620f, 0.967497f, 0.958467f }, + { 1.000000f, 1.000000f, 1.000000f }, + }; + + // cmasher::fusion (diverging; 33 anchors sampled from the 511-entry table) + inline constexpr float fusion[33][3] = { + { 0.152696f, 0.015942f, 0.069889f }, + { 0.243393f, 0.027996f, 0.138374f }, + { 0.339396f, 0.022662f, 0.187330f }, + { 0.434194f, 0.019908f, 0.202137f }, + { 0.518461f, 0.063682f, 0.191551f }, + { 0.591978f, 0.129217f, 0.171843f }, + { 0.656187f, 0.199931f, 0.150076f }, + { 0.710976f, 0.275318f, 0.131287f }, + { 0.754925f, 0.356047f, 0.126115f }, + { 0.784581f, 0.436574f, 0.150892f }, + { 0.803828f, 0.525723f, 0.220961f }, + { 0.815422f, 0.614033f, 0.328856f }, + { 0.828457f, 0.697985f, 0.458793f }, + { 0.850099f, 0.777148f, 0.598141f }, + { 0.884001f, 0.852908f, 0.739531f }, + { 0.932364f, 0.926807f, 0.878147f }, + { 1.000000f, 1.000000f, 1.000000f }, + { 0.882739f, 0.938948f, 0.943839f }, + { 0.759644f, 0.882704f, 0.900553f }, + { 0.630638f, 0.828950f, 0.872602f }, + { 0.499428f, 0.774306f, 0.861660f }, + { 0.381603f, 0.714439f, 0.863772f }, + { 0.301845f, 0.647196f, 0.868704f }, + { 0.272703f, 0.573492f, 0.869040f }, + { 0.281047f, 0.499433f, 0.863193f }, + { 0.307182f, 0.414842f, 0.849495f }, + { 0.335700f, 0.322756f, 0.825580f }, + { 0.356873f, 0.220225f, 0.784504f }, + { 0.360275f, 0.107524f, 0.709606f }, + { 0.327077f, 0.044725f, 0.576747f }, + { 0.256240f, 0.065580f, 0.425468f }, + { 0.175982f, 0.062679f, 0.300240f }, + { 0.095379f, 0.037917f, 0.194868f }, + }; + + // cmasher::prinsenvlag (diverging; 33 anchors sampled from the 511-entry table) + inline constexpr float prinsenvlag[33][3] = { + { 0.666523f, 0.321623f, 0.271748f }, + { 0.715454f, 0.343630f, 0.238454f }, + { 0.759975f, 0.370421f, 0.199167f }, + { 0.798794f, 0.402978f, 0.153663f }, + { 0.830257f, 0.442137f, 0.100784f }, + { 0.852097f, 0.488559f, 0.041408f }, + { 0.861821f, 0.542123f, 0.032681f }, + { 0.860262f, 0.599807f, 0.123019f }, + { 0.854920f, 0.655912f, 0.231638f }, + { 0.852701f, 0.704526f, 0.333943f }, + { 0.855639f, 0.752429f, 0.440786f }, + { 0.864494f, 0.797240f, 0.544831f }, + { 0.879227f, 0.839859f, 0.645897f }, + { 0.899719f, 0.880993f, 0.743689f }, + { 0.925990f, 0.921172f, 0.837642f }, + { 0.958977f, 0.960584f, 0.926102f }, + { 1.000000f, 1.000000f, 1.000000f }, + { 0.926250f, 0.969453f, 0.961500f }, + { 0.846431f, 0.941633f, 0.927771f }, + { 0.760433f, 0.915515f, 0.901576f }, + { 0.669026f, 0.889675f, 0.886082f }, + { 0.578083f, 0.861643f, 0.882953f }, + { 0.498336f, 0.829280f, 0.888643f }, + { 0.435785f, 0.792736f, 0.897592f }, + { 0.392859f, 0.755673f, 0.906500f }, + { 0.363227f, 0.713780f, 0.915753f }, + { 0.350577f, 0.669530f, 0.923924f }, + { 0.355484f, 0.622674f, 0.928631f }, + { 0.377364f, 0.573245f, 0.923046f }, + { 0.410189f, 0.524116f, 0.889701f }, + { 0.432426f, 0.483706f, 0.814621f }, + { 0.433280f, 0.452439f, 0.723666f }, + { 0.421591f, 0.424905f, 0.636181f }, + }; + + // ColorBrewer "RdBu" diverging map, reversed (blue -> white -> red), as + // exposed by matplotlib under the name "RdBu_r". matplotlib builds it by + // linearly interpolating these 11 control points, so the uniform-anchor + // scheme above reproduces it to < 1.6/255. + // Colors from ColorBrewer (https://colorbrewer2.org) by Cynthia A. Brewer, + // Geography, Pennsylvania State University -- Apache License 2.0. + inline constexpr float rdbu_r[11][3] = { + { 0.019608f, 0.188235f, 0.380392f }, + { 0.132026f, 0.403460f, 0.676278f }, + { 0.262745f, 0.576471f, 0.764706f }, + { 0.566474f, 0.768704f, 0.868512f }, + { 0.819608f, 0.898039f, 0.941176f }, + { 0.969089f, 0.966474f, 0.964937f }, + { 0.992157f, 0.858824f, 0.780392f }, + { 0.957555f, 0.651211f, 0.515110f }, + { 0.839216f, 0.376471f, 0.301961f }, + { 0.692272f, 0.092272f, 0.167705f }, + { 0.403922f, 0.000000f, 0.121569f }, + }; + + inline auto lookup(const std::string& name) -> Anchors { + if (name == "inferno") { + return { inferno, 9 }; + } else if (name == "plasma") { + return { plasma, 9 }; + } else if (name == "cool2warm" or name == "coolwarm") { + return { cool2warm, 3 }; + } else if (name == "gray" or name == "grey") { + return { gray, 2 }; + // CMasher scientific colormaps (accept an optional "cmr." prefix so + // names can be copied straight from the CMasher documentation) + } else if (name == "dusk" or name == "cmr.dusk") { + return { dusk, 33 }; + } else if (name == "cosmic" or name == "cmr.cosmic") { + return { cosmic, 33 }; + } else if (name == "freeze" or name == "cmr.freeze") { + return { freeze, 33 }; + } else if (name == "apple" or name == "cmr.apple") { + return { apple, 33 }; + } else if (name == "gothic" or name == "cmr.gothic") { + return { gothic, 33 }; + } else if (name == "sunburst" or name == "cmr.sunburst") { + return { sunburst, 33 }; + } else if (name == "voltage" or name == "cmr.voltage") { + return { voltage, 33 }; + } else if (name == "ocean" or name == "cmr.ocean") { + return { ocean, 33 }; + } else if (name == "fusion" or name == "cmr.fusion") { + return { fusion, 33 }; + } else if (name == "prinsenvlag" or name == "cmr.prinsenvlag") { + return { prinsenvlag, 33 }; + } else if (name == "RdBu_r" or name == "rdbu_r") { + return { rdbu_r, 11 }; + } else { + // default / "viridis" + return { viridis, 9 }; + } + } + + } // namespace cmap_hidden + + /** + * @brief Sample a named colormap at u in [0, 1], returning RGB in [0, 1]. + */ + inline void colormapRGB(const std::string& name, + real_t u, + real_t& r, + real_t& g, + real_t& b) { + const auto anchors = cmap_hidden::lookup(name); + if (u <= ZERO) { + r = anchors.rgb[0][0]; + g = anchors.rgb[0][1]; + b = anchors.rgb[0][2]; + return; + } + if (u >= ONE) { + r = anchors.rgb[anchors.n - 1][0]; + g = anchors.rgb[anchors.n - 1][1]; + b = anchors.rgb[anchors.n - 1][2]; + return; + } + const real_t x = u * static_cast(anchors.n - 1); + const int i0 = static_cast(x); + const int i1 = (i0 + 1 < anchors.n) ? (i0 + 1) : i0; + const real_t t = x - static_cast(i0); + r = static_cast(anchors.rgb[i0][0]) * (ONE - t) + + static_cast(anchors.rgb[i1][0]) * t; + g = static_cast(anchors.rgb[i0][1]) * (ONE - t) + + static_cast(anchors.rgb[i1][1]) * t; + b = static_cast(anchors.rgb[i0][2]) * (ONE - t) + + static_cast(anchors.rgb[i1][2]) * t; + } + + /** + * @brief Piecewise-linear opacity from sorted (position, alpha) control points. + */ + inline auto alphaAt(const std::vector>& pts, real_t u) + -> real_t { + if (pts.empty()) { + return u; // sensible default: linear ramp + } + if (u <= pts.front()[0]) { + return pts.front()[1]; + } + if (u >= pts.back()[0]) { + return pts.back()[1]; + } + for (std::size_t i = 0; i + 1 < pts.size(); ++i) { + if (u >= pts[i][0] and u <= pts[i + 1][0]) { + const real_t span = pts[i + 1][0] - pts[i][0]; + const real_t t = (span > ZERO) ? (u - pts[i][0]) / span : ZERO; + return pts[i][1] * (ONE - t) + pts[i + 1][1] * t; + } + } + return pts.back()[1]; + } + + /** + * @brief Build a premultiplied RGBA device LUT from a colormap + alpha points. + * @param colormap name of the colormap + * @param n_lut number of entries + * @param alpha_pts sorted (position, alpha) control points in [0,1]x[0,1] + * @return device View of shape (n_lut, 4), premultiplied RGBA + */ + inline auto buildLUT(const std::string& colormap, + int n_lut, + const std::vector>& alpha_pts) + -> array_t { + array_t lut { "render_lut", static_cast(n_lut) }; + auto lut_h = Kokkos::create_mirror_view(lut); + for (int i = 0; i < n_lut; ++i) { + const real_t u = (n_lut > 1) + ? static_cast(i) / static_cast(n_lut - 1) + : ZERO; + real_t r, g, b; + colormapRGB(colormap, u, r, g, b); + const real_t a = alphaAt(alpha_pts, u); + lut_h(i, 0) = r * a; // premultiplied + lut_h(i, 1) = g * a; + lut_h(i, 2) = b * a; + lut_h(i, 3) = a; + } + Kokkos::deep_copy(lut, lut_h); + return lut; + } + +} // namespace out + +#endif // OUTPUT_RENDER_TRANSFER_FN_H diff --git a/tests/framework/CMakeLists.txt b/tests/framework/CMakeLists.txt index 3696c74a9..24e84d1eb 100644 --- a/tests/framework/CMakeLists.txt +++ b/tests/framework/CMakeLists.txt @@ -44,9 +44,9 @@ else() gen_test(particles_sort false) endif() -# team_policy X-3: per-backend sort_by_key permutation test (only built when the -# compile-time team_policy toggle is on). -if(${team_policy}) +# tiled_deposit X-3: per-backend sort_by_key permutation test (only built when +# the compile-time tiled_deposit toggle is on). +if(${tiled_deposit}) gen_test(sort_by_key false) endif() diff --git a/tests/framework/particles_sort.cpp b/tests/framework/particles_sort.cpp index 591485220..9f1a08ef7 100644 --- a/tests/framework/particles_sort.cpp +++ b/tests/framework/particles_sort.cpp @@ -63,7 +63,7 @@ auto main(int argc, char* argv[]) -> int { i2_p(p) = 23u; weight_p(p) = 3.0; } - // team_policy keys on min(i, i_prev); without a meaningful + // tiled_deposit keys on min(i, i_prev); without a meaningful // i_prev every key would collapse to 0. Set i_prev = i so the // tile key reduces to the particle's current cell. i1_prev_p(p) = i1_p(p); @@ -96,9 +96,9 @@ auto main(int argc, char* argv[]) -> int { Kokkos::deep_copy(pld_i_h, prtls.pld_i); // Tile geometry, mirroring sort::PositionToTileIndex. T = 1 (no - // team_policy) reproduces the legacy per-cell ordering. -#if defined(TEAM_POLICY) - const ncells_t T = static_cast(TEAM_POLICY_TILE_SIZE); + // tiled_deposit) reproduces the legacy per-cell ordering. +#if defined(TILED_DEPOSIT) + const ncells_t T = static_cast(TILED_DEPOSIT_TILE_SIZE); #else const ncells_t T = 1u; #endif @@ -115,15 +115,15 @@ auto main(int argc, char* argv[]) -> int { // non-decreasing tile index; (2) every SoA member is permuted by the // *same* permutation, so each alive slot still satisfies // pld == f(weight); (3) no alive particle is lost. Only [0, npart()) - // is defined after a sort. The team_policy path compacts — it drops + // is defined after a sort. The tiled_deposit path compacts — it drops // the dead, so npart() equals the alive count and [0, npart()) is // entirely alive; the legacy (non-team) path keeps the dead as a // weight == -1 suffix, leaving npart() unchanged. Iterating // [0, npart()) exercises both: the prefix-sorted / no-alive-after-dead // checks below hold either way. -#if defined(TEAM_POLICY) +#if defined(TILED_DEPOSIT) raise::ErrorIf(prtls.npart() != 59u, - "team_policy sort must compact: npart() should equal " + "tiled_deposit sort must compact: npart() should equal " "the alive count", HERE); #else @@ -229,7 +229,7 @@ auto main(int argc, char* argv[]) -> int { i3_p(p) = 7u; weight_p(p) = 4.0; } - // see 2D block: i_prev = i so the team_policy tile key reduces + // see 2D block: i_prev = i so the tiled_deposit tile key reduces // to the particle's current cell. i1_prev_p(p) = i1_p(p); i2_prev_p(p) = i2_p(p); @@ -258,11 +258,11 @@ auto main(int argc, char* argv[]) -> int { // Same invariants as the 2D block (no payloads here): alive prefix // sorted by non-decreasing tile index, alive count preserved. The - // team_policy path compacts the dead away (npart() == alive count); + // tiled_deposit path compacts the dead away (npart() == alive count); // the legacy path keeps them as a weight == -1 suffix. T = 1 // reproduces the legacy per-cell order. -#if defined(TEAM_POLICY) - const ncells_t T = static_cast(TEAM_POLICY_TILE_SIZE); +#if defined(TILED_DEPOSIT) + const ncells_t T = static_cast(TILED_DEPOSIT_TILE_SIZE); #else const ncells_t T = 1u; #endif @@ -276,9 +276,9 @@ auto main(int argc, char* argv[]) -> int { (static_cast(c) / T); }; -#if defined(TEAM_POLICY) +#if defined(TILED_DEPOSIT) raise::ErrorIf(prtls.npart() != 59u, - "team_policy sort must compact: npart() should equal " + "tiled_deposit sort must compact: npart() should equal " "the alive count", HERE); #else diff --git a/tests/framework/sort_by_key.cpp b/tests/framework/sort_by_key.cpp index 64c4eb180..31a1bc097 100644 --- a/tests/framework/sort_by_key.cpp +++ b/tests/framework/sort_by_key.cpp @@ -1,5 +1,5 @@ /** - * @brief X-3 (team_policy) — sort_by_key permutation test. + * @brief X-3 (tiled_deposit) — sort_by_key permutation test. * * Exercises every backend overload of `ntt::sort_helpers::sort_by_key_dispatch` * that is compiled in for the current Kokkos device. For each backend: @@ -11,7 +11,7 @@ * promise stability per their documentation but we don't bake that into * the test). * - * Built only when `team_policy=ON` at CMake time. + * Built only when `tiled_deposit=ON` at CMake time. */ #include "enums.h" #include "global.h" diff --git a/tests/kernels/CMakeLists.txt b/tests/kernels/CMakeLists.txt index 0fe23f578..7e6b6b280 100644 --- a/tests/kernels/CMakeLists.txt +++ b/tests/kernels/CMakeLists.txt @@ -26,7 +26,7 @@ endfunction() gen_test(faraday_mink) gen_test(ampere_mink) gen_test(deposit) -if(${team_policy}) +if(${tiled_deposit}) gen_test(deposit_tiled) endif() gen_test(digital_filter) diff --git a/tests/kernels/deposit_tiled.cpp b/tests/kernels/deposit_tiled.cpp index 90a3890bf..693181334 100644 --- a/tests/kernels/deposit_tiled.cpp +++ b/tests/kernels/deposit_tiled.cpp @@ -7,7 +7,7 @@ * for shape orders O = 1..11 and asserts that the resulting J array is * identical cell-by-cell within a small floating-point tolerance. * - * Built only when `team_policy=ON` (`-D TEAM_POLICY` defined). The test + * Built only when `tiled_deposit=ON` (`-D TILED_DEPOSIT` defined). The test * matches the per-particle setup used in `deposit.cpp` so that any * regression in the shared `kernel::deposit::deposit_one_particle` body * is caught by both tests.