From a2726817e1c9c8d13a4dfe43f6f700646a151cf5 Mon Sep 17 00:00:00 2001 From: said Date: Sun, 27 Sep 2026 01:45:03 +0100 Subject: [PATCH 01/11] codex: test Open MPI and matching mpi4py on Linux and macOS --- .github/workflows/openmpi-integration.yml | 48 ++++++-- CHANGELOG.md | 3 + docs/developer/workflows/ci.md | 2 +- docs/user/tutorials/openmpi-f08.md | 106 +++++++++--------- .../end_to_end/test_openmpi_f08.py | 21 +++- 5 files changed, 116 insertions(+), 64 deletions(-) diff --git a/.github/workflows/openmpi-integration.yml b/.github/workflows/openmpi-integration.yml index b7fd0565b..30d51475e 100644 --- a/.github/workflows/openmpi-integration.yml +++ b/.github/workflows/openmpi-integration.yml @@ -13,8 +13,8 @@ concurrency: jobs: openmpi: - name: Open MPI mpi_f08 · ${{ matrix.version }} · Ubuntu 24.04 - runs-on: ubuntu-24.04 + name: Open MPI mpi_f08 · ${{ matrix.version }} · ${{ matrix.target }} + runs-on: ${{ matrix.runner }} timeout-minutes: 90 strategy: fail-fast: false @@ -24,13 +24,24 @@ jobs: # mpi_f08_types, 5.0 re-exports them from a configured mpi_types. - version: "4.1.8" series: "v4.1" + target: Ubuntu 24.04 + runner: ubuntu-24.04 - version: "5.0.11" series: "v5.0" + target: Ubuntu 24.04 + runner: ubuntu-24.04 + - version: "4.1.8" + series: "v4.1" + target: macOS 15 ARM64 + runner: macos-15 + - version: "5.0.11" + series: "v5.0" + target: macOS 15 ARM64 + runner: macos-15 env: OPENMPI_VERSION: ${{ matrix.version }} OPENMPI_SERIES: ${{ matrix.series }} PRIK_GFORTRAN_BINARY: gfortran-13 - PRIK_GFORTRAN_PACKAGE: gfortran-13 PYTHONPATH: . steps: - name: Checkout repository @@ -43,12 +54,25 @@ jobs: run: | python -m pip install --upgrade pip python -m pip install -e ".[qa]" - - name: Install pinned GFortran + - name: Install pinned GFortran on Ubuntu + if: runner.os == 'Linux' shell: bash run: | if ! command -v "$PRIK_GFORTRAN_BINARY" >/dev/null 2>&1; then sudo apt-get update - sudo apt-get install --yes "$PRIK_GFORTRAN_PACKAGE" + sudo apt-get install --yes gfortran-13 + fi + compiler_dir="$RUNNER_TEMP/prik-gfortran" + mkdir -p "$compiler_dir" + ln -sf "$(command -v "$PRIK_GFORTRAN_BINARY")" "$compiler_dir/gfortran" + echo "$compiler_dir" >> "$GITHUB_PATH" + "$compiler_dir/gfortran" --version + - name: Install pinned GFortran on macOS + if: runner.os == 'macOS' + shell: bash + run: | + if ! command -v "$PRIK_GFORTRAN_BINARY" >/dev/null 2>&1; then + brew install gcc@13 fi compiler_dir="$RUNNER_TEMP/prik-gfortran" mkdir -p "$compiler_dir" @@ -60,7 +84,7 @@ jobs: uses: actions/cache@v4 with: path: ~/prik-openmpi/${{ matrix.version }} - key: openmpi-${{ matrix.version }}-ubuntu-24.04-gfortran-13-v1 + key: openmpi-${{ matrix.version }}-${{ matrix.runner }}-gfortran-13-v1 - name: Build and install Open MPI if: steps.openmpi-cache.outputs.cache-hit != 'true' shell: bash @@ -72,12 +96,21 @@ jobs: mv "$root/openmpi-$OPENMPI_VERSION" "$root/source" cd "$root/build" ../source/configure --prefix="$root/install" --enable-mpi-fortran=usempif08 FC=gfortran - make -j"$(nproc)" + make -j2 make install # The test reads sources and generated headers from these trees and # links the installation, so build objects are not cached. find . \( -name '*.o' -o -name '*.lo' -o -name '*.a' -o -name '*.la' \) -delete find . -type d -name .libs -prune -exec rm -rf {} + + - name: Build mpi4py against this Open MPI installation + shell: bash + run: | + root="$HOME/prik-openmpi/$OPENMPI_VERSION" + export PATH="$root/install/bin:$PATH" + export LD_LIBRARY_PATH="$root/install/lib${LD_LIBRARY_PATH:+:$LD_LIBRARY_PATH}" + export DYLD_LIBRARY_PATH="$root/install/lib${DYLD_LIBRARY_PATH:+:$DYLD_LIBRARY_PATH}" + export MPI4PY_BUILD_MPICC="$root/install/bin/mpicc" + python -m pip install --no-cache-dir --no-binary=mpi4py mpi4py==4.1.2 - name: Run the Open MPI mpi_f08 workflow test shell: bash env: @@ -86,6 +119,7 @@ jobs: root="$HOME/prik-openmpi/$OPENMPI_VERSION" export PATH="$root/install/bin:$PATH" export LD_LIBRARY_PATH="$root/install/lib${LD_LIBRARY_PATH:+:$LD_LIBRARY_PATH}" + export DYLD_LIBRARY_PATH="$root/install/lib${DYLD_LIBRARY_PATH:+:$DYLD_LIBRARY_PATH}" export PRIK_OPENMPI_SOURCE="$root/source" export PRIK_OPENMPI_BUILD="$root/build" export PRIK_OPENMPI_MPIFORT="$root/install/bin/mpifort" diff --git a/CHANGELOG.md b/CHANGELOG.md index 7e58d915c..d39b44468 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,6 +7,9 @@ release tags add a leading `v` to the package version. ## Unreleased +- Open MPI integration CI now runs the `mpi_f08` tutorial on Linux and macOS + against Open MPI 4.1 and 5.0, and compares its two-rank result with mpi4py + built from the same installation. - The test suite consolidates overlapping checks around compiled workflows and retains focused parser, semantic, diagnostic, and ABI boundary coverage; contributor guidance now favors observable behavior over implementation shape. diff --git a/docs/developer/workflows/ci.md b/docs/developer/workflows/ci.md index 77926e7fa..20e8ffda9 100644 --- a/docs/developer/workflows/ci.md +++ b/docs/developer/workflows/ci.md @@ -17,7 +17,7 @@ contributors need to administer. | --- | --- | | Static analysis | Linting, formatting, security, dead code, and changed-code complexity policy. | | Compiler and platform tests | Supported Python versions, Linux and macOS, GNU Fortran, IFX, and Flang. | -| Open MPI Integration | The Open MPI `mpi_f08` workflow on Ubuntu for one Open MPI 4.1 and one 5.0 release: each is built from source, a restricted contract is generated from `mpi-f08.F90` with module discovery, and a two-rank program runs against the built wrapper. | +| Open MPI Integration | The Open MPI `mpi_f08` workflow on Linux and macOS for one Open MPI 4.1 and one 5.0 release: each is built from source, a restricted contract is generated from `mpi-f08.F90` with module discovery, and a two-rank program runs against the built wrapper and mpi4py compiled with that installation. | | Real Libraries Portability | Maintained real-library examples across the hosted Linux and macOS architecture/compiler matrix, with deep BLAS and LAPACK audits on Linux x86-64. | | Documentation and benchmarks | Required performance benchmark and generated snapshot, documentation tests, and a strict site build. | diff --git a/docs/user/tutorials/openmpi-f08.md b/docs/user/tutorials/openmpi-f08.md index 5e9f8949d..74295fb42 100644 --- a/docs/user/tutorials/openmpi-f08.md +++ b/docs/user/tutorials/openmpi-f08.md @@ -77,6 +77,46 @@ Two layers make this work: kept small on purpose: an illustration of the approach, not a complete MPI binding. +## Install Open MPI and mpi4py + +This tutorial uses Open MPI 5.0.11. CI also tests Open MPI 4.1.8, on Linux and +macOS. For each version, CI builds the PRIK wrapper and compares this page's +two-rank program with mpi4py. Keep the source and configured build trees: PRIK +reads them to generate the contract, then links the wrapper to that same +installation. You need a C compiler, GNU Fortran 13, and Python with PRIK +installed. Run these commands in one shell, from a directory where you want +the Open MPI source archive: + +```bash +OMPI_ROOT="$HOME/openmpi-5.0.11" +TUTORIAL_DIR="$PWD" +mkdir -p "$OMPI_ROOT/source" "$OMPI_ROOT/build" +curl -fsSLO https://download.open-mpi.org/release/open-mpi/v5.0/openmpi-5.0.11.tar.bz2 +tar -xjf openmpi-5.0.11.tar.bz2 -C "$OMPI_ROOT/source" --strip-components=1 +cd "$OMPI_ROOT/build" +../source/configure --prefix="$OMPI_ROOT/install" --enable-mpi-fortran=usempif08 FC=gfortran-13 +make -j2 && make install +cd "$TUTORIAL_DIR" +export PATH="$OMPI_ROOT/install/bin:$PATH" +export LD_LIBRARY_PATH="$OMPI_ROOT/install/lib${LD_LIBRARY_PATH:+:$LD_LIBRARY_PATH}" +export DYLD_LIBRARY_PATH="$OMPI_ROOT/install/lib${DYLD_LIBRARY_PATH:+:$DYLD_LIBRARY_PATH}" +OMPI_SRC="$OMPI_ROOT/source" +OMPI_BUILD="$OMPI_ROOT/build" +``` + +Build mpi4py against this Open MPI, rather than using a prebuilt wheel: + +```bash +MPI4PY_BUILD_MPICC="$OMPI_ROOT/install/bin/mpicc" \ + python3 -m pip install --no-cache-dir --no-binary=mpi4py mpi4py==4.1.2 +mpirun --version +python3 -c 'from mpi4py import MPI; print(MPI.Get_library_version())' +``` + +Both version checks should report Open MPI 5.0.11. Use this installation's +`mpifort`, `mpicc`, and `mpirun` throughout. To test 4.1.8 instead, change +`5.0.11` to `4.1.8` and `v5.0` to `v4.1` in the commands above. + ## 1. See what PRIK reads `mpi_f08` is an ordinary Fortran module that gathers other modules: @@ -172,15 +212,9 @@ their signatures name, without publishing the rest of the API. ## 3. Generate the contract -PRIK reads the Fortran declarations from a configured Open MPI source tree and -its build tree, which must be the ones your installed Open MPI was built from; -[why it must match](#why-the-configured-tree-must-match-the-installation) -comes later. Point two shell variables at them: - -```bash -OMPI_SRC=/path/to/openmpi-5.0.11 -OMPI_BUILD=/path/to/openmpi-5.0.11/build -``` +PRIK reads the Fortran declarations from the Open MPI source and configured +build trees you kept during installation. Keep `OMPI_SRC` and `OMPI_BUILD` set +to those trees. Generate the contract from the one entry source, letting PRIK discover the modules it uses under both trees: @@ -623,8 +657,15 @@ the boundary as native buffers: `Recv` wrote into rank 1's `data`, `Bcast` into every rank's `data`, and each reduction into its receive array directly. No part of Open MPI was rebuilt. -With `from mpi4py import MPI` in place of `import prik_mpi as MPI`, mpi4py -runs the same program and prints the same lines. +To compare with the mpi4py installation from the setup above, run the same +program with only its import changed: + +```bash +sed 's/^import prik_mpi as MPI$/from mpi4py import MPI/' mpi_example.py > mpi4py_example.py +mpirun -n 2 python3 mpi4py_example.py +``` + +The two programs print the same lines. ## How fast it is @@ -648,49 +689,6 @@ mpi4py's `MPI.Wtime` for all three: an AMD Ryzen 5 5600H laptop (x86-64, with Python 3.10, NumPy 2.2, GCC and gfortran 11.4, Open MPI 5.0.11, and mpi4py 4.1.2. -## Why the configured tree must match the installation - -`mpi_f08` is not the same text in every Open MPI build. `configure` decides -details of the Fortran interface for the compiler and options it is given, and -writes some of the Fortran sources and headers PRIK reads -- the -`MPI_IN_PLACE` declaration shown earlier is one of them. Two builds of the -same Open MPI version with the same `gfortran` but different Fortran flags can -therefore declare different interfaces. PRIK has to read the declarations of -the build it links against, so the source and build trees must be the ones -that installation was configured from; the version number alone does not -guarantee that. - -Open MPI records each configure run in the installation and in the build tree, -so you can compare them. The installation reports when, where, by whom, and -with which command line it was configured: - -```bash -ompi_info --parsable | grep '^config:\(timestamp\|host\|user\|cli\)' -``` - -and the build tree records the same run: - -```bash -grep '^OPAL_CONFIGURE_\(DATE\|HOST\|USER\)' "$OMPI_BUILD/Makefile" -grep OPAL_CONFIGURE_CLI "$OMPI_BUILD/opal/include/opal/version.h" -``` - -The simplest way to have a matching pair is to build Open MPI yourself and -keep its trees: - -```bash -tar -xjf openmpi-5.0.11.tar.bz2 -mkdir openmpi-5.0.11/build && cd openmpi-5.0.11/build -../configure --prefix="$HOME/openmpi-5.0.11" --enable-mpi-fortran=usempif08 FC=gfortran -make -j4 && make install -``` - -PRIK's Open MPI integration test runs the commands in this tutorial and checks -this relationship before it starts: it requires the configured tree and the -installation to record the same configure run. Continuous integration runs it -against Open MPI 4.1.8 and 5.0.11 built this way. Those are the configurations -exercised in CI, not the only ones that can work. - ## Limitations This tutorial selected eighteen names; the rest of `mpi_f08` works the same diff --git a/tests/fortran/assumed_types/end_to_end/test_openmpi_f08.py b/tests/fortran/assumed_types/end_to_end/test_openmpi_f08.py index 35163f2ae..54d1df945 100644 --- a/tests/fortran/assumed_types/end_to_end/test_openmpi_f08.py +++ b/tests/fortran/assumed_types/end_to_end/test_openmpi_f08.py @@ -5,7 +5,8 @@ configured Open MPI sources, replace its facade with the tutorial's edited one, build it against the installation without compiling any Open MPI source, save the tutorial's Python files beside the extension, and run its -mpi4py-style program there under the Open MPI launcher. +mpi4py-style program there under the Open MPI launcher. The same program then +runs through mpi4py built against that installation. """ from __future__ import annotations @@ -270,10 +271,26 @@ def run(script: str) -> list[str]: ) return sorted(completed.stdout.splitlines()) - assert run("mpi_example.py") == [ + prik_result = run("mpi_example.py") + assert prik_result == [ "rank 0 max [2, 3]", "rank 0 of 2: bcast [0, 1, 2], sum [3, 5], in place [3, 5]", "rank 1 of 2: bcast [0, 1, 2], sum [3, 5], in place [3, 5]", "rank 1 received [0, 1, 2, 3]", ] + mpi4py_library = subprocess.run( + [sys.executable, "-c", "from mpi4py import MPI; print(MPI.Get_library_version().splitlines()[0])"], + check=True, + capture_output=True, + text=True, + timeout=60, + cwd=tmp_path, + ).stdout + assert f"Open MPI v{_tree_configuration(source, build)['version']}" in mpi4py_library + example = (tmp_path / "mpi_example.py").read_text(encoding="utf-8") + assert example.count("import prik_mpi as MPI") == 1 + (tmp_path / "mpi4py_example.py").write_text( + example.replace("import prik_mpi as MPI", "from mpi4py import MPI"), encoding="utf-8" + ) + assert run("mpi4py_example.py") == prik_result assert run(STATUS_IGNORE_CHECK.name) == ["Mpi_Status: status tag 21, ignored status unchanged True"] From 3cf53ac5204769634c4191bb943190b588595558 Mon Sep 17 00:00:00 2001 From: said Date: Sun, 27 Sep 2026 05:44:59 +0100 Subject: [PATCH 02/11] codex: pair GNU compilers in Open MPI CI --- .github/workflows/openmpi-integration.yml | 17 ++++++++++++----- CHANGELOG.md | 4 ++-- docs/user/tutorials/openmpi-f08.md | 9 ++++++--- .../end_to_end/test_openmpi_f08.py | 4 +++- 4 files changed, 23 insertions(+), 11 deletions(-) diff --git a/.github/workflows/openmpi-integration.yml b/.github/workflows/openmpi-integration.yml index 30d51475e..e3cf91716 100644 --- a/.github/workflows/openmpi-integration.yml +++ b/.github/workflows/openmpi-integration.yml @@ -42,6 +42,7 @@ jobs: OPENMPI_VERSION: ${{ matrix.version }} OPENMPI_SERIES: ${{ matrix.series }} PRIK_GFORTRAN_BINARY: gfortran-13 + PRIK_GCC_BINARY: gcc-13 PYTHONPATH: . steps: - name: Checkout repository @@ -58,33 +59,39 @@ jobs: if: runner.os == 'Linux' shell: bash run: | - if ! command -v "$PRIK_GFORTRAN_BINARY" >/dev/null 2>&1; then + if ! command -v "$PRIK_GFORTRAN_BINARY" >/dev/null 2>&1 || \ + ! command -v "$PRIK_GCC_BINARY" >/dev/null 2>&1; then sudo apt-get update - sudo apt-get install --yes gfortran-13 + sudo apt-get install --yes gfortran-13 gcc-13 fi compiler_dir="$RUNNER_TEMP/prik-gfortran" mkdir -p "$compiler_dir" ln -sf "$(command -v "$PRIK_GFORTRAN_BINARY")" "$compiler_dir/gfortran" + ln -sf "$(command -v "$PRIK_GCC_BINARY")" "$compiler_dir/gcc" echo "$compiler_dir" >> "$GITHUB_PATH" "$compiler_dir/gfortran" --version + "$compiler_dir/gcc" --version - name: Install pinned GFortran on macOS if: runner.os == 'macOS' shell: bash run: | - if ! command -v "$PRIK_GFORTRAN_BINARY" >/dev/null 2>&1; then + if ! command -v "$PRIK_GFORTRAN_BINARY" >/dev/null 2>&1 || \ + ! command -v "$PRIK_GCC_BINARY" >/dev/null 2>&1; then brew install gcc@13 fi compiler_dir="$RUNNER_TEMP/prik-gfortran" mkdir -p "$compiler_dir" ln -sf "$(command -v "$PRIK_GFORTRAN_BINARY")" "$compiler_dir/gfortran" + ln -sf "$(command -v "$PRIK_GCC_BINARY")" "$compiler_dir/gcc" echo "$compiler_dir" >> "$GITHUB_PATH" "$compiler_dir/gfortran" --version + "$compiler_dir/gcc" --version - name: Restore the Open MPI source, configured build, and installation id: openmpi-cache uses: actions/cache@v4 with: path: ~/prik-openmpi/${{ matrix.version }} - key: openmpi-${{ matrix.version }}-${{ matrix.runner }}-gfortran-13-v1 + key: openmpi-${{ matrix.version }}-${{ matrix.runner }}-gcc-gfortran-13-v2 - name: Build and install Open MPI if: steps.openmpi-cache.outputs.cache-hit != 'true' shell: bash @@ -95,7 +102,7 @@ jobs: | tar -xj -C "$root" mv "$root/openmpi-$OPENMPI_VERSION" "$root/source" cd "$root/build" - ../source/configure --prefix="$root/install" --enable-mpi-fortran=usempif08 FC=gfortran + ../source/configure --prefix="$root/install" --enable-mpi-fortran=usempif08 CC=gcc FC=gfortran make -j2 make install # The test reads sources and generated headers from these trees and diff --git a/CHANGELOG.md b/CHANGELOG.md index d39b44468..7a2744f05 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -8,8 +8,8 @@ release tags add a leading `v` to the package version. ## Unreleased - Open MPI integration CI now runs the `mpi_f08` tutorial on Linux and macOS - against Open MPI 4.1 and 5.0, and compares its two-rank result with mpi4py - built from the same installation. + against Open MPI 4.1 and 5.0 with paired GNU C/Fortran compilers, and + compares its two-rank result with mpi4py built from the same installation. - The test suite consolidates overlapping checks around compiled workflows and retains focused parser, semantic, diagnostic, and ABI boundary coverage; contributor guidance now favors observable behavior over implementation shape. diff --git a/docs/user/tutorials/openmpi-f08.md b/docs/user/tutorials/openmpi-f08.md index 74295fb42..34199222e 100644 --- a/docs/user/tutorials/openmpi-f08.md +++ b/docs/user/tutorials/openmpi-f08.md @@ -83,18 +83,21 @@ This tutorial uses Open MPI 5.0.11. CI also tests Open MPI 4.1.8, on Linux and macOS. For each version, CI builds the PRIK wrapper and compares this page's two-rank program with mpi4py. Keep the source and configured build trees: PRIK reads them to generate the contract, then links the wrapper to that same -installation. You need a C compiler, GNU Fortran 13, and Python with PRIK +installation. You need GNU Fortran 13 and GCC 13, plus Python with PRIK installed. Run these commands in one shell, from a directory where you want the Open MPI source archive: ```bash OMPI_ROOT="$HOME/openmpi-5.0.11" TUTORIAL_DIR="$PWD" -mkdir -p "$OMPI_ROOT/source" "$OMPI_ROOT/build" +mkdir -p "$OMPI_ROOT/source" "$OMPI_ROOT/build" "$OMPI_ROOT/toolchain" +ln -sf "$(command -v gfortran-13)" "$OMPI_ROOT/toolchain/gfortran" +ln -sf "$(command -v gcc-13)" "$OMPI_ROOT/toolchain/gcc" +export PATH="$OMPI_ROOT/toolchain:$PATH" curl -fsSLO https://download.open-mpi.org/release/open-mpi/v5.0/openmpi-5.0.11.tar.bz2 tar -xjf openmpi-5.0.11.tar.bz2 -C "$OMPI_ROOT/source" --strip-components=1 cd "$OMPI_ROOT/build" -../source/configure --prefix="$OMPI_ROOT/install" --enable-mpi-fortran=usempif08 FC=gfortran-13 +../source/configure --prefix="$OMPI_ROOT/install" --enable-mpi-fortran=usempif08 CC=gcc FC=gfortran make -j2 && make install cd "$TUTORIAL_DIR" export PATH="$OMPI_ROOT/install/bin:$PATH" diff --git a/tests/fortran/assumed_types/end_to_end/test_openmpi_f08.py b/tests/fortran/assumed_types/end_to_end/test_openmpi_f08.py index 54d1df945..b14dff336 100644 --- a/tests/fortran/assumed_types/end_to_end/test_openmpi_f08.py +++ b/tests/fortran/assumed_types/end_to_end/test_openmpi_f08.py @@ -237,12 +237,14 @@ def showme(flag: str) -> list[str]: "2", "--json", ], - check=True, + check=False, capture_output=True, text=True, timeout=300, cwd=tmp_path, ) + if built.returncode: + pytest.fail(f"Open MPI contract replay build failed:\n{built.stdout}\n{built.stderr}") payload = json.loads(built.stdout) # Only PRIK's bridge and binding compile; Open MPI's own sources do not. assert payload["native_build_plan"]["compilation_units"] == [] From 9dd9fdfd9fba13dccb91300f2669674db0046765 Mon Sep 17 00:00:00 2001 From: said Date: Sun, 27 Sep 2026 09:33:26 +0100 Subject: [PATCH 03/11] codex: measure Open MPI bindings against matched mpi4py --- .github/workflows/openmpi-integration.yml | 11 +- CHANGELOG.md | 2 + benchmarks/openmpi_f08.py | 114 ++++++++++++++++++ docs/developer/workflows/ci.md | 2 +- docs/user/tutorials/openmpi-f08.md | 38 +++--- .../end_to_end/test_openmpi_f08.py | 20 +++ 6 files changed, 164 insertions(+), 23 deletions(-) create mode 100644 benchmarks/openmpi_f08.py diff --git a/.github/workflows/openmpi-integration.yml b/.github/workflows/openmpi-integration.yml index e3cf91716..706d4d620 100644 --- a/.github/workflows/openmpi-integration.yml +++ b/.github/workflows/openmpi-integration.yml @@ -122,6 +122,8 @@ jobs: shell: bash env: PRIK_OPENMPI_REQUIRED: "1" + PRIK_OPENMPI_BENCHMARK: "1" + PRIK_OPENMPI_BENCHMARK_DIR: ${{ runner.temp }}/openmpi-benchmark run: | root="$HOME/prik-openmpi/$OPENMPI_VERSION" export PATH="$root/install/bin:$PATH" @@ -131,4 +133,11 @@ jobs: export PRIK_OPENMPI_BUILD="$root/build" export PRIK_OPENMPI_MPIFORT="$root/install/bin/mpifort" export PRIK_OPENMPI_LAUNCHER="$root/install/bin/mpirun" - python -m pytest -q -rs tests/fortran/assumed_types/end_to_end/test_openmpi_f08.py + python -m pytest -q -rs -s tests/fortran/assumed_types/end_to_end/test_openmpi_f08.py + - name: Upload matched Open MPI benchmark results + if: always() + uses: actions/upload-artifact@v4 + with: + name: openmpi-benchmark-${{ matrix.version }}-${{ matrix.runner }} + path: ${{ runner.temp }}/openmpi-benchmark/*.json + if-no-files-found: ignore diff --git a/CHANGELOG.md b/CHANGELOG.md index 7a2744f05..07c2ec682 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -10,6 +10,8 @@ release tags add a leading `v` to the package version. - Open MPI integration CI now runs the `mpi_f08` tutorial on Linux and macOS against Open MPI 4.1 and 5.0 with paired GNU C/Fortran compilers, and compares its two-rank result with mpi4py built from the same installation. + The tutorial provides a repeatable matched-installation benchmark for the + generated calls, Python facade, and mpi4py in place of unverified timings. - The test suite consolidates overlapping checks around compiled workflows and retains focused parser, semantic, diagnostic, and ABI boundary coverage; contributor guidance now favors observable behavior over implementation shape. diff --git a/benchmarks/openmpi_f08.py b/benchmarks/openmpi_f08.py new file mode 100644 index 000000000..4161b4215 --- /dev/null +++ b/benchmarks/openmpi_f08.py @@ -0,0 +1,114 @@ +"""Compare the tutorial's three APIs against one Open MPI installation. + +Run each backend separately with two ranks from the directory containing the +generated extension and ``prik_mpi.py``. Rank zero prints one JSON record. +""" + +from __future__ import annotations + +import argparse +import atexit +import json +import os +import timeit + +import mpi4py +import numpy as np + + +def main() -> None: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("backend", choices=("direct", "facade", "mpi4py")) + backend = parser.parse_args().backend + + if backend != "mpi4py": + mpi4py.rc.initialize = False + mpi4py.rc.finalize = False + from mpi4py import MPI as timer_mpi + + if backend == "direct": + from prik_openmpi_f08 import mpi_f08 as mpi + + mpi.init() + atexit.register(mpi.finalize) + comm = mpi.mpi_comm_world + + def barrier() -> None: + mpi.barrier(comm) + + rank = int(mpi.comm_rank(comm)) + ranks = int(mpi.comm_size(comm)) + datatype = mpi.mpi_int + op = mpi.mpi_sum + rank_call = "mpi.comm_rank(comm)" + barrier_call = "mpi.barrier(comm)" + allreduce_call = "mpi.allreduce(send, recv, datatype, op, comm)" + elif backend == "facade": + import prik_mpi as mpi + + comm = mpi.COMM_WORLD + barrier = comm.Barrier + rank = int(comm.Get_rank()) + ranks = int(comm.Get_size()) + datatype = None + op = mpi.SUM + rank_call = "comm.Get_rank()" + barrier_call = "comm.Barrier()" + allreduce_call = "comm.Allreduce(send, recv, op=op)" + else: + mpi = timer_mpi + comm = mpi.COMM_WORLD + barrier = comm.Barrier + rank = int(comm.Get_rank()) + ranks = int(comm.Get_size()) + datatype = None + op = mpi.SUM + rank_call = "comm.Get_rank()" + barrier_call = "comm.Barrier()" + allreduce_call = "comm.Allreduce(send, recv, op=op)" + + results: dict[str, float] = {} + + def measure( + statement: str, iterations: int, *, send: np.ndarray | None = None, recv: np.ndarray | None = None + ) -> float: + timer = timeit.Timer( + statement, + timer=timer_mpi.Wtime, + globals={"mpi": mpi, "comm": comm, "send": send, "recv": recv, "datatype": datatype, "op": op}, + ) + barrier() + timer.timeit(number=min(iterations, 100)) + samples = [] + for _ in range(5): + barrier() + samples.append(timer.timeit(number=iterations) * 1e9 / iterations) + barrier() + return min(samples) + + for size, iterations in ((1, 20_000), (1_024, 20_000), (1_048_576, 8)): + send = np.full(size, rank + 1, dtype=np.int32) + recv = np.empty_like(send) + results[f"allreduce_{size}"] = measure(allreduce_call, iterations, send=send, recv=recv) + expected = ranks * (ranks + 1) // 2 + if int(recv[0]) != expected or int(recv[-1]) != expected: + raise AssertionError(f"Allreduce produced an incorrect result for {size} values") + + results["barrier"] = measure(barrier_call, 20_000) + results["get_rank"] = measure(rank_call, 200_000) + if rank == 0: + print( + json.dumps( + { + "backend": backend, + "mpi_version": os.environ.get("OPENMPI_VERSION"), + "timer": "mpi4py.MPI.Wtime", + "ranks": ranks, + "ns_per_call": results, + } + ) + ) + + +if __name__ == "__main__": + main() diff --git a/docs/developer/workflows/ci.md b/docs/developer/workflows/ci.md index 20e8ffda9..6971c62a0 100644 --- a/docs/developer/workflows/ci.md +++ b/docs/developer/workflows/ci.md @@ -17,7 +17,7 @@ contributors need to administer. | --- | --- | | Static analysis | Linting, formatting, security, dead code, and changed-code complexity policy. | | Compiler and platform tests | Supported Python versions, Linux and macOS, GNU Fortran, IFX, and Flang. | -| Open MPI Integration | The Open MPI `mpi_f08` workflow on Linux and macOS for one Open MPI 4.1 and one 5.0 release: each is built from source, a restricted contract is generated from `mpi-f08.F90` with module discovery, and a two-rank program runs against the built wrapper and mpi4py compiled with that installation. | +| Open MPI Integration | The Open MPI `mpi_f08` workflow on Linux and macOS for one Open MPI 4.1 and one 5.0 release: each is built from source, a restricted contract is generated from `mpi-f08.F90` with module discovery, and a two-rank program runs against the built wrapper and mpi4py compiled with that installation. Matched-installation call timings are uploaded as benchmark artifacts. | | Real Libraries Portability | Maintained real-library examples across the hosted Linux and macOS architecture/compiler matrix, with deep BLAS and LAPACK audits on Linux x86-64. | | Documentation and benchmarks | Required performance benchmark and generated snapshot, documentation tests, and a strict site build. | diff --git a/docs/user/tutorials/openmpi-f08.md b/docs/user/tutorials/openmpi-f08.md index 34199222e..7c785085b 100644 --- a/docs/user/tutorials/openmpi-f08.md +++ b/docs/user/tutorials/openmpi-f08.md @@ -670,27 +670,23 @@ mpirun -n 2 python3 mpi4py_example.py The two programs print the same lines. -## How fast it is - -Time per call, compared with mpi4py: - -| Call | mpi4py | Generated functions | `prik_mpi.py` | -| --- | ---: | ---: | ---: | -| `Allreduce`, 1 integer | 1.02 µs | 0.67 µs (35% faster) | 0.92 µs (10% faster) | -| `Allreduce`, 1,024 integers | 2.33 µs | 1.78 µs (23% faster) | 2.15 µs (7% faster) | -| `Allreduce`, 1,048,576 integers | 2.33 ms | 2.12 ms (about the same) | 2.37 ms (about the same) | -| `Barrier` | 0.32 µs | 0.32 µs (the same) | 0.50 µs (57% slower) | -| `Get_rank` | 19 ns | 134 ns (about 7× slower) | 297 ns (about 15× slower) | - -`Get_rank` is the cheapest call, so its time is almost all the overhead of -making a call. That overhead is small, but higher through PRIK than through -mpi4py, and can be optimized later. - -These were measured on a local machine, with two ranks on it, and timed with -mpi4py's `MPI.Wtime` for all three: an AMD Ryzen 5 5600H laptop (x86-64, -6 cores and 12 threads, up to 4.28 GHz, 7 GB of memory) running Ubuntu 22.04, -with Python 3.10, NumPy 2.2, GCC and gfortran 11.4, Open MPI 5.0.11, and -mpi4py 4.1.2. +## Compare call times + +After building the wrapper, copy `benchmarks/openmpi_f08.py` from the PRIK +checkout beside `prik_mpi.py` and `prik_openmpi_f08.so`. Run the same five +operations through each API with the Open MPI launcher from the setup above: + +```bash +for backend in direct facade mpi4py; do + mpirun -n 2 python3 openmpi_f08.py "$backend" +done +``` + +The script reports nanoseconds per call for `Allreduce` with 1, 1,024, and +1,048,576 `np.int32` values, `Barrier`, and `Get_rank`. It uses the same two +ranks, arrays, prepared MPI handles, mpi4py `MPI.Wtime` timer, and repeated +batches for each backend. Compare results from one machine and one Open MPI +installation; timings vary by host. ## Limitations diff --git a/tests/fortran/assumed_types/end_to_end/test_openmpi_f08.py b/tests/fortran/assumed_types/end_to_end/test_openmpi_f08.py index b14dff336..5b7503e72 100644 --- a/tests/fortran/assumed_types/end_to_end/test_openmpi_f08.py +++ b/tests/fortran/assumed_types/end_to_end/test_openmpi_f08.py @@ -33,6 +33,7 @@ PROGRAM = (FIXTURES / "runtime" / "prik_mpi.py", FIXTURES / "runtime" / "mpi_example.py") # Not shown in the tutorial: shows MPI_STATUS_IGNORE reaches Open MPI as itself. STATUS_IGNORE_CHECK = FIXTURES / "runtime" / "mpi_status_ignore_check.py" +BENCHMARK = Path(__file__).resolve().parents[4] / "benchmarks" / "openmpi_f08.py" # ``ompi_info`` reports these for the configure run that built the # installation, and a configured tree records the same values, so they # identify that run: its date, host, user, and exact command line. @@ -296,3 +297,22 @@ def run(script: str) -> list[str]: ) assert run("mpi4py_example.py") == prik_result assert run(STATUS_IGNORE_CHECK.name) == ["Mpi_Status: status tag 21, ignored status unchanged True"] + + if os.environ.get("PRIK_OPENMPI_BENCHMARK") == "1": + shutil.copyfile(BENCHMARK, tmp_path / BENCHMARK.name) + report_dir = Path(os.environ["PRIK_OPENMPI_BENCHMARK_DIR"]) + report_dir.mkdir(parents=True, exist_ok=True) + for backend in ("direct", "facade", "mpi4py"): + result = subprocess.run( + [launcher, "-n", "2", sys.executable, BENCHMARK.name, backend], + capture_output=True, + text=True, + timeout=300, + cwd=tmp_path, + ) + if result.returncode: + pytest.fail(f"Open MPI {backend} benchmark failed:\n{result.stdout}\n{result.stderr}") + report = json.loads(result.stdout) + assert report["backend"] == backend + (report_dir / f"openmpi-{backend}.json").write_text(json.dumps(report, indent=2) + "\n", encoding="utf-8") + print(f"Open MPI {backend} benchmark: {result.stdout.strip()}", flush=True) From d8fc6c2f00322f141c271f2585d7d765dbbe82ff Mon Sep 17 00:00:00 2001 From: said Date: Sun, 27 Sep 2026 09:54:29 +0100 Subject: [PATCH 04/11] codex: document matched Open MPI benchmark results --- CHANGELOG.md | 5 +++-- docs/user/tutorials/openmpi-f08.md | 16 ++++++++++++++++ 2 files changed, 19 insertions(+), 2 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 07c2ec682..24473b34b 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -10,8 +10,9 @@ release tags add a leading `v` to the package version. - Open MPI integration CI now runs the `mpi_f08` tutorial on Linux and macOS against Open MPI 4.1 and 5.0 with paired GNU C/Fortran compilers, and compares its two-rank result with mpi4py built from the same installation. - The tutorial provides a repeatable matched-installation benchmark for the - generated calls, Python facade, and mpi4py in place of unverified timings. + The tutorial provides a repeatable matched-installation benchmark and a + labeled local results table for the generated calls, Python facade, and + mpi4py; CI uploads separate results for each platform and Open MPI version. - The test suite consolidates overlapping checks around compiled workflows and retains focused parser, semantic, diagnostic, and ABI boundary coverage; contributor guidance now favors observable behavior over implementation shape. diff --git a/docs/user/tutorials/openmpi-f08.md b/docs/user/tutorials/openmpi-f08.md index 7c785085b..6a6194769 100644 --- a/docs/user/tutorials/openmpi-f08.md +++ b/docs/user/tutorials/openmpi-f08.md @@ -688,6 +688,22 @@ ranks, arrays, prepared MPI handles, mpi4py `MPI.Wtime` timer, and repeated batches for each backend. Compare results from one machine and one Open MPI installation; timings vary by host. +These results were measured on an AMD Ryzen 5 5600H running Ubuntu 22.04.5, +with Open MPI 5.0.11 built using GNU Fortran 11.4 and mpi4py 4.1.2 built +against that same installation. Each value is the median of three runs; each +run takes the fastest of five timed batches with two ranks. + +| Operation | Generated API | `prik_mpi.py` | mpi4py | +| --- | ---: | ---: | ---: | +| `Allreduce`, 1 `int32` | 0.851 µs | 0.987 µs | 1.106 µs | +| `Allreduce`, 1,024 `int32` values | 3.262 µs | 3.494 µs | 3.652 µs | +| `Allreduce`, 1,048,576 `int32` values | 2.279 ms | 2.379 ms | 2.424 ms | +| `Barrier` | 0.443 µs | 0.507 µs | 0.342 µs | +| `Get_rank` | 247 ns | 310 ns | 31 ns | + +CI uploads JSON results for each Linux and macOS Open MPI job so you can +compare its measurements with this local run. + ## Limitations This tutorial selected eighteen names; the rest of `mpi_f08` works the same From 834235338f7addb458ba0e00279c9d2ca5074123 Mon Sep 17 00:00:00 2001 From: said Date: Sun, 27 Sep 2026 10:10:16 +0100 Subject: [PATCH 05/11] codex: show Open MPI benchmark comparisons against mpi4py --- CHANGELOG.md | 5 +++-- docs/user/tutorials/openmpi-f08.md | 17 +++++++++++------ 2 files changed, 14 insertions(+), 8 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 24473b34b..625232336 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -11,8 +11,9 @@ release tags add a leading `v` to the package version. against Open MPI 4.1 and 5.0 with paired GNU C/Fortran compilers, and compares its two-rank result with mpi4py built from the same installation. The tutorial provides a repeatable matched-installation benchmark and a - labeled local results table for the generated calls, Python facade, and - mpi4py; CI uploads separate results for each platform and Open MPI version. + labeled local results table comparing the generated calls and Python facade + with mpi4py, including relative timings; CI uploads separate results for + each platform and Open MPI version. - The test suite consolidates overlapping checks around compiled workflows and retains focused parser, semantic, diagnostic, and ABI boundary coverage; contributor guidance now favors observable behavior over implementation shape. diff --git a/docs/user/tutorials/openmpi-f08.md b/docs/user/tutorials/openmpi-f08.md index 6a6194769..7194d24d2 100644 --- a/docs/user/tutorials/openmpi-f08.md +++ b/docs/user/tutorials/openmpi-f08.md @@ -693,13 +693,18 @@ with Open MPI 5.0.11 built using GNU Fortran 11.4 and mpi4py 4.1.2 built against that same installation. Each value is the median of three runs; each run takes the fastest of five timed batches with two ranks. -| Operation | Generated API | `prik_mpi.py` | mpi4py | +Time per call, compared with mpi4py: + +| Operation | mpi4py | Generated API | `prik_mpi.py` | | --- | ---: | ---: | ---: | -| `Allreduce`, 1 `int32` | 0.851 µs | 0.987 µs | 1.106 µs | -| `Allreduce`, 1,024 `int32` values | 3.262 µs | 3.494 µs | 3.652 µs | -| `Allreduce`, 1,048,576 `int32` values | 2.279 ms | 2.379 ms | 2.424 ms | -| `Barrier` | 0.443 µs | 0.507 µs | 0.342 µs | -| `Get_rank` | 247 ns | 310 ns | 31 ns | +| `Allreduce`, 1 `int32` | 1.106 µs | 0.851 µs (23% faster) | 0.987 µs (11% faster) | +| `Allreduce`, 1,024 `int32` values | 3.652 µs | 3.262 µs (11% faster) | 3.494 µs (4% faster) | +| `Allreduce`, 1,048,576 `int32` values | 2.424 ms | 2.279 ms (6% faster) | 2.379 ms (2% faster) | +| `Barrier` | 0.342 µs | 0.443 µs (30% slower) | 0.507 µs (48% slower) | +| `Get_rank` | 31 ns | 247 ns (about 8× slower) | 310 ns (10× slower) | + +The relative figures describe this local run; the differences for the largest +`Allreduce` are small compared with its variation between runs. CI uploads JSON results for each Linux and macOS Open MPI job so you can compare its measurements with this local run. From 382d648c2b696606e0cf54beb9a2022a08e08a9b Mon Sep 17 00:00:00 2001 From: said Date: Sun, 27 Sep 2026 10:12:13 +0100 Subject: [PATCH 06/11] codex: explain Get_rank benchmark overhead --- docs/user/tutorials/openmpi-f08.md | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/docs/user/tutorials/openmpi-f08.md b/docs/user/tutorials/openmpi-f08.md index 7194d24d2..d71075549 100644 --- a/docs/user/tutorials/openmpi-f08.md +++ b/docs/user/tutorials/openmpi-f08.md @@ -703,6 +703,10 @@ Time per call, compared with mpi4py: | `Barrier` | 0.342 µs | 0.443 µs (30% slower) | 0.507 µs (48% slower) | | `Get_rank` | 31 ns | 247 ns (about 8× slower) | 310 ns (10× slower) | +`Get_rank` is the cheapest call, so its time is almost all the overhead of +making a call. That overhead is small, but higher through PRIK than through +mpi4py, and can be optimized later. + The relative figures describe this local run; the differences for the largest `Allreduce` are small compared with its variation between runs. From ab99f0068477d592df60b74dc6a07a3822ea4335 Mon Sep 17 00:00:00 2001 From: said Date: Sun, 27 Sep 2026 11:02:08 +0100 Subject: [PATCH 07/11] codex: serve module variables through descriptors on the module type A generated module looked every attribute up through its own getattro, which compared the name with each module variable before falling back to the ordinary module lookup, so each module.function(...) call paid about 100 ns before the call started. Module variables are now get/set descriptors on the module's heap type, which CPython finds through the type's attribute cache, and functions resolve as on a plain module. The read-only and deletion messages are unchanged. The Open MPI tutorial's comparison is re-measured with its benchmark script: Get_rank through the generated API drops from 245 to 164 ns. Co-Authored-By: Claude Opus 5.5 --- CHANGELOG.md | 5 + docs/user/tutorials/openmpi-f08.md | 10 +- prik/printers/c.py | 143 +++++++++++++---------------- 3 files changed, 74 insertions(+), 84 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 625232336..de50cb96b 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,6 +7,11 @@ release tags add a leading `v` to the package version. ## Unreleased +- Generated extension modules serve their module variables through + descriptors on the module type, so looking up a function or any other + ordinary attribute costs what it costs on a plain module instead of first + being compared with every module variable name. `mpi.comm_rank(comm)` on + the Open MPI tutorial's extension drops from 245 to 164 ns. - Open MPI integration CI now runs the `mpi_f08` tutorial on Linux and macOS against Open MPI 4.1 and 5.0 with paired GNU C/Fortran compilers, and compares its two-rank result with mpi4py built from the same installation. diff --git a/docs/user/tutorials/openmpi-f08.md b/docs/user/tutorials/openmpi-f08.md index d71075549..57b25b483 100644 --- a/docs/user/tutorials/openmpi-f08.md +++ b/docs/user/tutorials/openmpi-f08.md @@ -697,11 +697,11 @@ Time per call, compared with mpi4py: | Operation | mpi4py | Generated API | `prik_mpi.py` | | --- | ---: | ---: | ---: | -| `Allreduce`, 1 `int32` | 1.106 µs | 0.851 µs (23% faster) | 0.987 µs (11% faster) | -| `Allreduce`, 1,024 `int32` values | 3.652 µs | 3.262 µs (11% faster) | 3.494 µs (4% faster) | -| `Allreduce`, 1,048,576 `int32` values | 2.424 ms | 2.279 ms (6% faster) | 2.379 ms (2% faster) | -| `Barrier` | 0.342 µs | 0.443 µs (30% slower) | 0.507 µs (48% slower) | -| `Get_rank` | 31 ns | 247 ns (about 8× slower) | 310 ns (10× slower) | +| `Allreduce`, 1 `int32` | 1.101 µs | 0.738 µs (33% faster) | 0.887 µs (19% faster) | +| `Allreduce`, 1,024 `int32` values | 3.675 µs | 3.439 µs (6% faster) | 3.387 µs (8% faster) | +| `Allreduce`, 1,048,576 `int32` values | 2.247 ms | 2.272 ms (1% slower) | 2.106 ms (6% faster) | +| `Barrier` | 0.342 µs | 0.390 µs (14% slower) | 0.457 µs (34% slower) | +| `Get_rank` | 32 ns | 164 ns (about 5× slower) | 233 ns (about 7× slower) | `Get_rank` is the cheapest call, so its time is almost all the overhead of making a call. That overhead is small, but higher through PRIK than through diff --git a/prik/printers/c.py b/prik/printers/c.py index 189d5e1de..cbcc22348 100644 --- a/prik/printers/c.py +++ b/prik/printers/c.py @@ -149,109 +149,94 @@ def _visit_CModuleDef(self, node: CModuleDef) -> str: ) def _visit_CModulePropertySupport(self, node: CModulePropertySupport) -> str: - """Render all generated module-property routing support in stable order. + """Render module variables as get/set descriptors on a heap module type. - The node supplies getter and setter entries plus the heap subtype name. - This method returns the three dependent C definitions: attribute getter, - attribute setter, and module-type installer. + CPython finds a descriptor through the type's cached attribute lookup, + so reading a function or any other ordinary attribute costs what it + costs on a plain module, while each variable still reaches its + generated getter and setter ahead of the module dictionary. """ return "\n\n".join( ( - self._module_getattro_source(node), - self._module_setattro_source(node), + *( + self._module_property_accessors_source(node, index, entry) + for index, entry in enumerate(node.entries) + ), + self._module_property_table_source(node), self._module_property_type_source(node), ) ) - def _module_getattro_source(self, node: CModulePropertySupport) -> str: - """Build the module attribute getter for every declared property entry. - - The returned function compares only Unicode attribute names, delegates - matching names to generated getters, and preserves the base module - fallback for all other attributes. - """ - lines = [f"static PyObject *{node.name}_getattro(PyObject *self, PyObject *name)", "{"] - lines.append(" if (PyUnicode_Check(name)) {") - for entry in node.entries: - lines.extend(self._module_getter_entry_source(entry)) - lines.extend((" }", " return PyModule_Type.tp_getattro(self, name);", "}")) - return "\n".join(lines) - - def _module_getter_entry_source(self, node: CModulePropertyEntry) -> tuple[str, ...]: - """Build one getter dispatch branch from a property entry. - - The tuple is inserted into the enclosing Unicode-name guard. It returns - NULL on comparison failure and calls exactly the getter named by the - supplied entry when its Python name matches. - """ - name = self._c_string_literal(node.python_name) - return ( - " {", - f" int comparison = PyUnicode_CompareWithASCIIString(name, {name});", - " if (comparison == -1 && PyErr_Occurred()) return NULL;", - f" if (comparison == 0) return {node.getter_name}();", - " }", - ) - - def _module_setattro_source(self, node: CModulePropertySupport) -> str: - """Build the module attribute setter for every declared property entry. - - The returned function dispatches writable properties to their generated - setters and keeps the base module setter as the nonmatching fallback. + def _module_property_accessors_source( + self, + node: CModulePropertySupport, + index: int, + entry: CModulePropertyEntry, + ) -> str: + """Adapt one entry's generated getter and setter to descriptor signatures. + + A read-only entry rejects replacement and deletion with its own message, + and a writable one rejects deletion before calling its setter; an entry + with neither has no setter, so CPython reports it as not writable. """ - lines = [f"static int {node.name}_setattro(PyObject *self, PyObject *name, PyObject *value)", "{"] - lines.append(" if (PyUnicode_Check(name)) {") - for entry in node.entries: - lines.extend(self._module_setter_entry_source(entry)) - lines.extend((" }", " return PyModule_Type.tp_setattro(self, name, value);", "}")) - return "\n".join(lines) - - def _module_setter_entry_source(self, node: CModulePropertyEntry) -> tuple[str, ...]: - """Build one setter dispatch branch and its node-selected error path. - - The tuple rejects replacement for read-only entries. Writable entries - reject deletion before calling their generated setter with the supplied - value; those rules are already encoded by the backend node. - """ - name = self._c_string_literal(node.python_name) lines = [ - " {", - f" int comparison = PyUnicode_CompareWithASCIIString(name, {name});", - " if (comparison == -1 && PyErr_Occurred()) return -1;", - " if (comparison == 0) {", + f"static PyObject *{node.name}_get_{index}(PyObject *self, void *closure)", + "{", + " (void)self;", + " (void)closure;", + f" return {entry.getter_name}();", + "}", ] - if node.reject_replacement: + if entry.reject_replacement or entry.setter_name is not None: lines.extend( ( - f' PyErr_SetString(PyExc_AttributeError, "module variable {node.python_name} is read-only");', - " return -1;", + f"static int {node.name}_set_{index}(PyObject *self, PyObject *value, void *closure)", + "{", + " (void)self;", + " (void)closure;", ) ) - else: - lines.extend( - ( - " if (value == NULL) {", - f' PyErr_SetString(PyExc_AttributeError, "module variable {node.python_name} cannot be deleted");', - " return -1;", - " }", - f" return {node.setter_name}(value);", + if entry.reject_replacement: + lines.extend( + ( + f' PyErr_SetString(PyExc_AttributeError, "module variable {entry.python_name} is read-only");', + " return -1;", + ) + ) + else: + lines.extend( + ( + " if (value == NULL) {", + f' PyErr_SetString(PyExc_AttributeError, "module variable {entry.python_name} cannot be deleted");', + " return -1;", + " }", + f" return {entry.setter_name}(value);", + ) ) + lines.append("}") + return "\n".join(lines) + + def _module_property_table_source(self, node: CModulePropertySupport) -> str: + """Render the descriptor table naming each entry's accessors.""" + lines = [f"static PyGetSetDef {node.name}_getset[] = {{"] + for index, entry in enumerate(node.entries): + setter = f"{node.name}_set_{index}" if entry.reject_replacement or entry.setter_name is not None else "NULL" + lines.append( + f" {{{self._c_string_literal(entry.python_name)}, {node.name}_get_{index}, {setter}, NULL, NULL}}," ) - lines.extend((" }", " }")) - return tuple(lines) + lines.extend((" {NULL, NULL, NULL, NULL, NULL}", "};")) + return "\n".join(lines) def _module_property_type_source(self, node: CModulePropertySupport) -> str: - """Build C slots, type spec, and installer for module property support. + """Build the heap module type carrying the descriptors, and its installer. - The returned definitions are ordered so the installer can reference the - generated slots and type spec without forward declarations. The node's - name is reused consistently for all emitted symbols. + The installer gives the module this type, so every later attribute + lookup on the module sees the descriptors. """ return "\n".join( ( f"static PyType_Slot {node.name}_slots[] = {{", - f" {{Py_tp_getattro, (void *){node.name}_getattro}},", - f" {{Py_tp_setattro, (void *){node.name}_setattro}},", + f" {{Py_tp_getset, (void *){node.name}_getset}},", " {0, NULL}", "};", f"static PyType_Spec {node.name}_spec = {{", From 1e6d4f8ec59d48e684795ecf348474f2f26c5d8c Mon Sep 17 00:00:00 2001 From: said Date: Sun, 27 Sep 2026 11:39:15 +0100 Subject: [PATCH 08/11] codex: streamline the Open MPI tutorial around its wrapped and Python APIs The tutorial now opens with what PRIK produces from Open MPI's sources, names the generated extension the wrapped API and prik_mpi.py the Python API, links and downloads every file it uses, and condenses its walkthrough into seven steps. The benchmark binds each callable and fixes buffer placement before timing, and its backends are named wrapped, python, and mpi4py in the script, the integration test, and the results table. Co-Authored-By: Claude Opus 5.5 --- CHANGELOG.md | 7 +- benchmarks/openmpi_f08.py | 58 ++- docs/user/tutorials/openmpi-f08.md | 462 +++++++----------- .../end_to_end/test_openmpi_f08.py | 2 +- 4 files changed, 214 insertions(+), 315 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index de50cb96b..cadc008f8 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -16,9 +16,10 @@ release tags add a leading `v` to the package version. against Open MPI 4.1 and 5.0 with paired GNU C/Fortran compilers, and compares its two-rank result with mpi4py built from the same installation. The tutorial provides a repeatable matched-installation benchmark and a - labeled local results table comparing the generated calls and Python facade - with mpi4py, including relative timings; CI uploads separate results for - each platform and Open MPI version. + labeled local results table comparing its wrapped API and mpi4py-style + Python API with mpi4py, including relative timings. The benchmark binds + each callable and fixes buffer placement before timing; CI uploads separate + results for each platform and Open MPI version. - The test suite consolidates overlapping checks around compiled workflows and retains focused parser, semantic, diagnostic, and ABI boundary coverage; contributor guidance now favors observable behavior over implementation shape. diff --git a/benchmarks/openmpi_f08.py b/benchmarks/openmpi_f08.py index 4161b4215..bb5b3ae3a 100644 --- a/benchmarks/openmpi_f08.py +++ b/benchmarks/openmpi_f08.py @@ -16,9 +16,18 @@ import numpy as np +def aligned_int32(size: int, byte_offset: int) -> np.ndarray: + """Give each backend the same buffer placement modulo 4 KiB.""" + page_bytes = 4_096 + item_bytes = np.dtype(np.int32).itemsize + backing = np.empty(size + page_bytes // item_bytes, dtype=np.int32) + start = ((byte_offset - backing.ctypes.data % page_bytes) % page_bytes) // item_bytes + return backing[start : start + size] + + def main() -> None: parser = argparse.ArgumentParser(description=__doc__) - parser.add_argument("backend", choices=("direct", "facade", "mpi4py")) + parser.add_argument("backend", choices=("wrapped", "python", "mpi4py")) backend = parser.parse_args().backend if backend != "mpi4py": @@ -26,7 +35,7 @@ def main() -> None: mpi4py.rc.finalize = False from mpi4py import MPI as timer_mpi - if backend == "direct": + if backend == "wrapped": from prik_openmpi_f08 import mpi_f08 as mpi mpi.init() @@ -40,10 +49,13 @@ def barrier() -> None: ranks = int(mpi.comm_size(comm)) datatype = mpi.mpi_int op = mpi.mpi_sum - rank_call = "mpi.comm_rank(comm)" - barrier_call = "mpi.barrier(comm)" - allreduce_call = "mpi.allreduce(send, recv, datatype, op, comm)" - elif backend == "facade": + rank_fn = mpi.comm_rank + barrier_fn = mpi.barrier + allreduce_fn = mpi.allreduce + rank_call = "rank_fn(comm)" + barrier_call = "barrier_fn(comm)" + allreduce_call = "allreduce_fn(send, recv, datatype, op, comm)" + elif backend == "python": import prik_mpi as mpi comm = mpi.COMM_WORLD @@ -52,9 +64,12 @@ def barrier() -> None: ranks = int(comm.Get_size()) datatype = None op = mpi.SUM - rank_call = "comm.Get_rank()" - barrier_call = "comm.Barrier()" - allreduce_call = "comm.Allreduce(send, recv, op=op)" + rank_fn = comm.Get_rank + barrier_fn = comm.Barrier + allreduce_fn = comm.Allreduce + rank_call = "rank_fn()" + barrier_call = "barrier_fn()" + allreduce_call = "allreduce_fn(send, recv, op=op)" else: mpi = timer_mpi comm = mpi.COMM_WORLD @@ -63,9 +78,12 @@ def barrier() -> None: ranks = int(comm.Get_size()) datatype = None op = mpi.SUM - rank_call = "comm.Get_rank()" - barrier_call = "comm.Barrier()" - allreduce_call = "comm.Allreduce(send, recv, op=op)" + rank_fn = comm.Get_rank + barrier_fn = comm.Barrier + allreduce_fn = comm.Allreduce + rank_call = "rank_fn()" + barrier_call = "barrier_fn()" + allreduce_call = "allreduce_fn(send, recv, op=op)" results: dict[str, float] = {} @@ -75,7 +93,16 @@ def measure( timer = timeit.Timer( statement, timer=timer_mpi.Wtime, - globals={"mpi": mpi, "comm": comm, "send": send, "recv": recv, "datatype": datatype, "op": op}, + globals={ + "comm": comm, + "send": send, + "recv": recv, + "datatype": datatype, + "op": op, + "rank_fn": rank_fn, + "barrier_fn": barrier_fn, + "allreduce_fn": allreduce_fn, + }, ) barrier() timer.timeit(number=min(iterations, 100)) @@ -87,8 +114,9 @@ def measure( return min(samples) for size, iterations in ((1, 20_000), (1_024, 20_000), (1_048_576, 8)): - send = np.full(size, rank + 1, dtype=np.int32) - recv = np.empty_like(send) + send = aligned_int32(size, 0) + send.fill(rank + 1) + recv = aligned_int32(size, 2_048) results[f"allreduce_{size}"] = measure(allreduce_call, iterations, send=send, recv=recv) expected = ranks * (ranks + 1) // 2 if int(recv[0]) != expected or int(recv[-1]) != expected: diff --git a/docs/user/tutorials/openmpi-f08.md b/docs/user/tutorials/openmpi-f08.md index 57b25b483..df438e93d 100644 --- a/docs/user/tutorials/openmpi-f08.md +++ b/docs/user/tutorials/openmpi-f08.md @@ -1,8 +1,8 @@ --- title: Wrap Open MPI mpi_f08 for Python -description: Turn a reviewed part of Open MPI's Fortran mpi_f08 interface into an mpi4py-style Python MPI API +description: Generate a Python extension from Open MPI's Fortran mpi_f08 sources and give it an mpi4py-style Python API audience: users -prerequisites: an Open MPI installation with mpi_f08, and the configured Open MPI source and build trees it was built from +prerequisites: Python with PRIK and NumPy, a GNU Fortran and GCC pair, and Open MPI built from source as shown below related: ../guide/wrapping-modules.md, ../guide/callbacks.md, ../reference/cli-commands.md, ../reference/pyi-format.md status: maintained publication: reviewed @@ -10,9 +10,11 @@ publication: reviewed # Wrap Open MPI `mpi_f08` for Python -This tutorial turns a reviewed part of Open MPI's real Fortran `mpi_f08` -interface into a Python MPI API modeled on [mpi4py](https://mpi4py.readthedocs.io), -the standard Python binding of MPI. At the end, this program runs under +PRIK reads Open MPI's Fortran `mpi_f08` sources and generates a Python +extension that calls the installed Open MPI library. You write no C, Cython, +or `ctypes`. You edit the generated `.pyi` contract into the Python signatures +you want, add a short Python module in the style of +[mpi4py](https://mpi4py.readthedocs.io), and this program runs under `mpirun`: @@ -57,35 +59,53 @@ if rank == 0: print(f"rank 0 max {largest.tolist()}") ``` -If you know mpi4py, you know this program: `COMM_WORLD`, `Get_rank`, -`Send`/`Recv`, `Bcast`, `Reduce`, `Allreduce`, `MPI.SUM`, and `MPI.IN_PLACE` -are all spelled as mpi4py spells them. Replace `import prik_mpi as MPI` with -`from mpi4py import MPI` and the same program runs under mpi4py and prints -the same lines. - -Two layers make this work: - -- **The extension PRIK generates.** Every MPI routine, handle type, and - constant the program reaches comes from the declarations in Open MPI's - Fortran sources, and every call enters the installed Open MPI library through - generated code. You shape that native API by editing a generated `.pyi` - contract: hiding counts that follow from the buffers, turning error codes - into exceptions, returning results instead of filling output arguments. -- **A short Python module, `prik_mpi.py`.** It gives the native API mpi4py's - shape: a `Comm` class with methods and keyword defaults. It is ordinary - Python over the generated functions, with no C and no `ctypes`, and it is - kept small on purpose: an illustration of the approach, not a complete MPI - binding. +It is also a valid mpi4py program: replace `import prik_mpi as MPI` with +`from mpi4py import MPI` and mpi4py prints the same lines. + +You build two APIs on the way: + +- **The wrapped API**, `prik_openmpi_f08`: the extension PRIK generates from + Open MPI's declarations and your edited contract. Every call goes straight + into Open MPI. +- **The Python API**, `prik_mpi.py`: about fifty lines of Python that give the + wrapped API mpi4py's `Comm` object and keyword defaults. + +On small messages both are faster than mpi4py; see +[Compare call times](#compare-call-times). + +## What you need + +- Linux or macOS, with Python 3.10 or later, NumPy, and PRIK installed; see + [Installation](../getting-started/installation.md). +- GNU Fortran and GCC of the same version; CI uses version 13. +- Open MPI and mpi4py, built as shown in the next section. +- The files this tutorial uses, from the PRIK repository: + [`mpi_exports.txt`](https://github.com/PyNumLab/prik/blob/main/tests/fortran/assumed_types/end_to_end/fixtures/contracts/openmpi/mpi_exports.txt), + [`mpi_f08.pyi`](https://github.com/PyNumLab/prik/blob/main/tests/fortran/assumed_types/end_to_end/fixtures/contracts/openmpi/mpi_f08.pyi), + [`prik_mpi.py`](https://github.com/PyNumLab/prik/blob/main/tests/fortran/assumed_types/end_to_end/fixtures/runtime/prik_mpi.py), + [`mpi_example.py`](https://github.com/PyNumLab/prik/blob/main/tests/fortran/assumed_types/end_to_end/fixtures/runtime/mpi_example.py), and + [`openmpi_f08.py`](https://github.com/PyNumLab/prik/blob/main/benchmarks/openmpi_f08.py). Each is also shown in + full below. Download them into an empty working directory: + +```bash +PRIK_RAW=https://raw.githubusercontent.com/PyNumLab/prik/main +for file in \ + tests/fortran/assumed_types/end_to_end/fixtures/contracts/openmpi/mpi_exports.txt \ + tests/fortran/assumed_types/end_to_end/fixtures/contracts/openmpi/mpi_f08.pyi \ + tests/fortran/assumed_types/end_to_end/fixtures/runtime/prik_mpi.py \ + tests/fortran/assumed_types/end_to_end/fixtures/runtime/mpi_example.py \ + benchmarks/openmpi_f08.py; do + curl -fsSLO "$PRIK_RAW/$file" +done +``` ## Install Open MPI and mpi4py -This tutorial uses Open MPI 5.0.11. CI also tests Open MPI 4.1.8, on Linux and -macOS. For each version, CI builds the PRIK wrapper and compares this page's -two-rank program with mpi4py. Keep the source and configured build trees: PRIK -reads them to generate the contract, then links the wrapper to that same -installation. You need GNU Fortran 13 and GCC 13, plus Python with PRIK -installed. Run these commands in one shell, from a directory where you want -the Open MPI source archive: +This tutorial uses [Open MPI 5.0.11](https://www.open-mpi.org/software/ompi/v5.0/); +CI also tests Open MPI 4.1.8, on Linux and macOS. Build it from source and +keep its source and configured build trees: PRIK reads them to generate the +contract, then links the extension to the same installation. Run these +commands in one shell, from your working directory: ```bash OMPI_ROOT="$HOME/openmpi-5.0.11" @@ -107,7 +127,8 @@ OMPI_SRC="$OMPI_ROOT/source" OMPI_BUILD="$OMPI_ROOT/build" ``` -Build mpi4py against this Open MPI, rather than using a prebuilt wheel: +Then build mpi4py against this Open MPI rather than installing a prebuilt +wheel, and check that both report Open MPI 5.0.11: ```bash MPI4PY_BUILD_MPICC="$OMPI_ROOT/install/bin/mpicc" \ @@ -116,39 +137,24 @@ mpirun --version python3 -c 'from mpi4py import MPI; print(MPI.Get_library_version())' ``` -Both version checks should report Open MPI 5.0.11. Use this installation's -`mpifort`, `mpicc`, and `mpirun` throughout. To test 4.1.8 instead, change -`5.0.11` to `4.1.8` and `v5.0` to `v4.1` in the commands above. +Use this installation's `mpifort` and `mpirun` throughout. For Open MPI 4.1.8, +change `5.0.11` to `4.1.8` and `v5.0` to `v4.1` in these commands. -## 1. See what PRIK reads +## 1. Choose what to wrap -`mpi_f08` is an ordinary Fortran module that gathers other modules: - -```fortran -module mpi_f08 - use mpi_f08_types - use mpi_f08_interfaces - use pmpi_f08_interfaces - use mpi_f08_callbacks - use mpi_f08_interfaces_callbacks -end module mpi_f08 -``` - -The handle types and predefined objects are declared in those modules: +`mpi_f08` is an ordinary Fortran module that gathers others. Its handle types +and predefined objects are Fortran declarations: ```fortran type(MPI_Comm), parameter :: MPI_COMM_WORLD = MPI_Comm(OMPI_MPI_COMM_WORLD) type(MPI_Op), parameter :: MPI_SUM = MPI_Op(OMPI_MPI_SUM) type(MPI_Datatype), parameter :: MPI_INT = MPI_Datatype(OMPI_MPI_INT) -integer MPI_ANY_SOURCE -parameter (MPI_ANY_SOURCE=-1) - integer, bind(C, name="mpi_fortran_in_place_") :: MPI_IN_PLACE type(MPI_Status), bind(C, name="mpi_fortran_status_ignore_") :: MPI_STATUS_IGNORE ``` -and each MPI routine is a generic interface over its specific procedures: +and each MPI routine is a generic interface: ```fortran interface MPI_Allreduce @@ -165,20 +171,8 @@ interface MPI_Allreduce end interface MPI_Allreduce ``` -These excerpts are simplified from Open MPI 5.0. PRIK does not need to be -told where each declaration lives. It starts from the entry source -`mpi-f08.F90`, follows every `use` to the source that defines that module, and -continues through the modules those use. File names play no part: Open MPI -declares its types in `mpi_f08_types` in one release series and in a -configure-generated `mpi_types` module in another, and PRIK finds whichever -the sources declare. Some of these declarations, such as `MPI_IN_PLACE` above, -are in headers that Open MPI's `configure` writes into its build tree. - -## 2. Choose a small native surface - -`mpi_f08` publishes hundreds of routines and constants. Start with a reviewed -subset instead of all of them. Save the Fortran identities to publish in -`mpi_exports.txt`: +`mpi_f08` publishes hundreds of names. `mpi_exports.txt` selects the ones +the program needs, each qualified by the module that publishes it: ```text @@ -202,25 +196,14 @@ mpi_f08::MPI_ANY_SOURCE mpi_f08::MPI_ANY_TAG ``` -These are the routines and objects behind the mpi4py names the program uses. -The program does not name `MPI_STATUS_IGNORE`; `Recv` passes it, since this -small API reports no status. - -Selecting symbols this way is not an MPI feature. `--export-symbols` accepts -module-qualified public symbols from any Fortran project -- procedures, -generics, and module variables -- and here every name is qualified by -`mpi_f08`, the facade that re-exports it. PRIK publishes exactly those names -and keeps whatever supporting declarations they need, such as the handle types -their signatures name, without publishing the rest of the API. - -## 3. Generate the contract +PRIK publishes exactly these names and keeps the declarations they depend on, +such as the handle types in their signatures. `--export-symbols` works this +way for any Fortran project; nothing here is specific to MPI. -PRIK reads the Fortran declarations from the Open MPI source and configured -build trees you kept during installation. Keep `OMPI_SRC` and `OMPI_BUILD` set -to those trees. +## 2. Generate the contract -Generate the contract from the one entry source, letting PRIK discover the -modules it uses under both trees: +Point PRIK at the one entry source. It follows each `use` to the source that +defines that module, under both Open MPI trees: ```bash python3 -m prik generate --pyi \ @@ -238,29 +221,14 @@ python3 -m prik generate --pyi \ -I "$OMPI_SRC/ompi/include" ``` -The Open MPI sources are preprocessed Fortran that includes headers from both -trees: the hand-written ones stay in the source tree, and the configured ones -are generated into the build tree beside them. The `-I` options name those -directories so PRIK reads each source exactly as the Fortran compiler did. +The `-I` directories are where Open MPI's Fortran sources find their headers, +some of which `configure` wrote into the build tree. PRIK reads the sources +to learn their declarations; it does not compile them. -PRIK reads these sources to learn their declarations. It does not compile -them. - -## 4. Read the generated contract - -The `contract/` directory holds one editable `.pyi` file per Fortran module the -selection needs. `contract/mpi_f08.pyi` publishes the selected names: - -```python -from .mpi_f08_types import mpi_any_source, mpi_any_tag, mpi_comm_world, mpi_in_place, mpi_int, mpi_max, mpi_status_ignore, mpi_sum -from .mpi_f08_interfaces import mpi_allreduce, mpi_barrier, mpi_bcast, mpi_comm_rank, mpi_comm_size, mpi_finalize, mpi_init, mpi_recv, mpi_reduce, mpi_send -from .mpi_types import Mpi_Comm, Mpi_Datatype, Mpi_Op, Mpi_Status -``` +## 3. Read the contract -These excerpts come from Open MPI 5.0, which declares the handle types in -`mpi_types`. Open MPI 4.1 declares them in `mpi_f08_types`, so there they are -imported from that module and written as `mpi_f08_types.Mpi_Datatype` and so -on. The predefined objects become typed module attributes in +`contract/` holds one editable `.pyi` file per Fortran module the selection +needs. The predefined objects keep their Fortran types, in `contract/mpi_f08_types.pyi`: ```python @@ -277,7 +245,7 @@ mpi_in_place: Int32[()] mpi_status_ignore: Mpi_Status ``` -and each selected routine keeps its Fortran interface in +and each routine keeps its Fortran interface, in `contract/mpi_f08_interfaces.pyi`: ```python @@ -293,63 +261,31 @@ def mpi_allreduce( ) -> Returns["ierror", Int32[()]] | None: ... ``` -`MPI_Allreduce` is a generic whose one specific is `MPI_Allreduce_f08`, so -`mpi_allreduce` is an overload of the specific declared as `mpi_allreduce_f08` -in the same file, and a call to it calls that specific. - -Four mappings are worth a closer look. - -**Handles are concrete types.** `Mpi_Comm`, `Mpi_Datatype`, `Mpi_Op`, and -`Mpi_Status` are the derived types Open MPI declares, with their real -components. For example, `Mpi_Comm` holds the one integer handle Open MPI -stores, and `Mpi_Status` the source, tag, and error of a message: - -```python -class Mpi_Status: - def __init__( - self, - *, - mpi_source: Int32 = ..., - mpi_tag: Int32 = ..., - mpi_error: Int32 = ... - ) -> None: ... - - mpi_source: Int32 - mpi_tag: Int32 - mpi_error: Int32 -``` - -A handle argument accepts only an object of its own type, and the predefined -handles are `Final` module constants. - -**`MPI_IN_PLACE` is native storage.** Its declaration is a C-bound integer -module variable, so it becomes rank-zero native integer storage, `Int32[()]`. -Python sees `mpi_in_place` as a live NumPy scalar view of Open MPI's own -`MPI_IN_PLACE` variable. Passing that view as a buffer passes the address of -that variable, which is how Open MPI recognizes an in-place operation. -Nothing here knows about MPI: it is the same mapping any Fortran module -variable declared this way receives. +These excerpts come from Open MPI 5.0, which declares the handle types in +`mpi_types`; Open MPI 4.1 declares them in `mpi_f08_types`. Three mappings +matter for MPI: -**`MPI_STATUS_IGNORE` is an `Mpi_Status` object.** It is declared as a -C-bound module variable of type `MPI_Status`, so it keeps that type: it is not -a generic buffer, and it is accepted wherever an `Mpi_Status` is. Passing a -module variable passes the variable itself, not a copy, so Open MPI receives -its own `MPI_STATUS_IGNORE` and recognizes it by address. +- **Handles are typed objects.** `Mpi_Comm`, `Mpi_Datatype`, `Mpi_Op`, and + `Mpi_Status` are Open MPI's derived types with their real fields, and a + handle argument accepts only its own type. +- **Buffers are `AnyNative`.** MPI's `type(*)` buffers accept any contiguous + NumPy array, passed by address. +- **Special objects are Open MPI's own.** `mpi_in_place` and + `mpi_status_ignore` are live views of Open MPI's `MPI_IN_PLACE` and + `MPI_STATUS_IGNORE` variables. Passing one passes that variable's address, + which is how Open MPI recognizes it. -**Choice buffers are `AnyNative`.** The send and receive buffers of MPI -routines are `type(*)` dummies, which accept data of any type, so they become -`AnyNative[Flat]`: any NumPy array, passed as a raw address. `AnyNative` -appears only for these assumed-type dummies; `MPI_IN_PLACE` is a concrete -module object with a concrete type. +This contract already builds. Its functions are still Fortran's, though: every +count is an argument, and every call returns an error code. -This contract already builds, and its functions are Fortran's: every count -is an argument, every routine takes and returns the optional `ierror`, and -`MPI_Comm_rank` returns the rank together with that error code. +## 4. Edit the contract into a Python API -## 5. Edit the facade into a Python API +Replace `contract/mpi_f08.pyi`, the module the extension publishes, with the +edited file you downloaded: -The contract is yours to edit. Replace `contract/mpi_f08.pyi`, the facade the -extension publishes, with this one: +```bash +cp mpi_f08.pyi contract/mpi_f08.pyi +``` ```python @@ -471,45 +407,23 @@ __all__ = [ ] ``` -Each function still calls the Fortran routine it names; only its Python face -changes. `@native_call` lists the native arguments in Fortran order and says -where each one comes from: +Each function still calls the Fortran routine it names; only its Python +signature changes. `@native_call` lists the Fortran arguments in order and +says where each comes from: | Edit | Example | Effect in Python | | --- | --- | --- | | `@bind("MPI_Send")` on `def send` | every function | The Python name differs from the Fortran name it calls. | -| `Int32(Arg(0).size)` | `count` of `MPI_Send` | The count is computed from the buffer, so the caller does not pass it. | -| `Return("rank", 0)` | `rank` of `MPI_Comm_rank` | The output argument becomes the return value: `comm_rank` returns the rank. | +| `Int32(Arg(0).size)` | `count` of `MPI_Send` | The count comes from the buffer, so the caller does not pass it. | +| `Return("rank", 0)` | `rank` of `MPI_Comm_rank` | The output argument becomes the return value. | | `Hidden("ierror", Int32)` with `@raises(status="ierror", success=0)` | every function | The error code is not an argument; a nonzero code raises an exception. | -`recv` keeps its `status` as an argument. A caller that wants the status -passes an `Mpi_Status` for Open MPI to fill in; one that does not passes -`mpi_status_ignore`, which tells Open MPI not to. +The same file works with Open MPI 4.1 and 5.0. -The facade imports its handle types and constants from -`mpi_f08_types` in both Open MPI 4.1 and 5.0: in 5.0 that module re-exports -them from `mpi_types`, and PRIK follows the re-export to the declaration. So -the same edited file serves both release series. +## 5. Build the wrapped API -## 6. Build from the contract - -The contract, not the Open MPI sources, is what the extension is built from: - -```text -Open MPI Fortran sources - | -PRIK semantic analysis - | -restricted .pyi contract, edited into the API you want - | -PRIK wrapper generation - | -Python extension linked to the installed Open MPI -``` - -Build `contract/__init__.pyi`, asking the installed `mpifort` wrapper for the -compiler it wraps, the flags that find Open MPI's Fortran modules, and the -libraries to link: +Build the edited contract, asking `mpifort` for the compiler, flags, and +libraries of your Open MPI: ```bash python3 -m prik contract/__init__.pyi \ @@ -521,16 +435,12 @@ python3 -m prik contract/__init__.pyi \ --out-dir build ``` -This compiles in `build/` and writes `prik_openmpi_f08.so`, a stable copy of -the extension, in the current directory; the rest of this tutorial works in -that directory and imports the extension from there. The build compiles only -the bridge and binding PRIK generates; no Open MPI source is compiled. The -installed Open MPI already provides the implementation, and the extension -links against its libraries. `--compiler` takes the compiler `mpifort` runs -rather than `mpifort` itself because PRIK identifies a Fortran compiler's -family from its executable name. +This writes `prik_openmpi_f08.so` in the working directory. Only the code +PRIK generates is compiled; the extension links to the installed Open MPI. +`--compiler` takes the compiler `mpifort` runs, because PRIK recognizes a +compiler by its executable name. -The extension can already run MPI: +The wrapped API already runs MPI: ```python import numpy as np @@ -544,12 +454,11 @@ mpi_f08.allreduce(values, total, mpi_f08.mpi_int, mpi_f08.mpi_sum, mpi_f08.mpi_c mpi_f08.finalize() ``` -## 7. Add the mpi4py-style layer +## 6. Add the Python API -A contract describes native calls. What mpi4py adds on top of MPI is a Python -object model, and that belongs in Python. Save this module as `prik_mpi.py` -in the same directory as `prik_openmpi_f08.so`. It is deliberately small: one -datatype and no status, enough to show the shape. +mpi4py's object model is Python, so the Python API is plain Python over the +wrapped API. `prik_mpi.py`, downloaded beside `prik_openmpi_f08.so`, is +deliberately small: ```python @@ -615,36 +524,24 @@ _mpi.init() atexit.register(_mpi.finalize) ``` -Everything in it calls the generated functions of step 5: - -- **Objects and methods.** `Comm` wraps an `Mpi_Comm` handle and spells - mpi4py's methods; `COMM_WORLD` wraps `mpi_comm_world`. Each method is one - call to a generated function. -- **One datatype.** Every buffer is an `np.int32` array, so every call passes - `MPI_INT`. -- **No status.** `Recv` always passes `MPI_STATUS_IGNORE`, as mpi4py does when - it is given no status. -- **Defaults and `np.int32` values.** `tag`, `source=ANY_SOURCE`, `root`, - and `op=SUM` are keyword defaults. Ranks and tags are the `np.int32` values - the contract's `Int32` arguments take from the start -- `Get_rank` returns - one, `ANY_SOURCE` and the defaults are, and `rank + 1` stays one -- so they - pass straight through. Converting a plain integer on every call would cost - more than a small MPI call. -- **Lifetime.** As with mpi4py, importing the module initializes MPI, and MPI - is finalized when the interpreter exits. - -## 8. Run it under Open MPI - -Save the program from the top of this page as `mpi_example.py` in the same -directory, beside `prik_mpi.py` and `prik_openmpi_f08.so`, and start two ranks -there with the installed Open MPI launcher: +- `Comm` wraps an `Mpi_Comm` handle and spells mpi4py's methods; each method + is one call to the wrapped API. +- Every buffer is an `np.int32` array sent as `MPI_INT`, and `Recv` passes + `MPI_STATUS_IGNORE`, as mpi4py does when given no status. +- Ranks and tags are `np.int32`, the type the contract takes, from the start: + `Get_rank` returns one, and `rank + 1` stays one. They pass straight + through without conversion. +- As with mpi4py, importing the module starts MPI, and exiting stops it. + +## 7. Run it + +Run the program from the top of this page with two ranks: ```bash mpirun -n 2 python3 mpi_example.py ``` -The two ranks print these lines, each rank's lines in order but the ranks in -whichever order they finish: +The ranks print these lines, in whichever order they finish: ```text rank 1 received [0, 1, 2, 3] @@ -653,94 +550,67 @@ rank 0 max [2, 3] rank 1 of 2: bcast [0, 1, 2], sum [3, 5], in place [3, 5] ``` -Here is what happened. `mpirun` started two Python processes as MPI ranks. -Each call went through `prik_mpi.py`, then PRIK's generated binding and -bridge, straight into the installed Open MPI library. The NumPy arrays crossed -the boundary as native buffers: `Recv` wrote into rank 1's `data`, `Bcast` -into every rank's `data`, and each reduction into its receive array directly. -No part of Open MPI was rebuilt. +Each call went from `prik_mpi.py` through the wrapped API into Open MPI, and +the NumPy arrays crossed as native buffers, written in place by `Recv`, +`Bcast`, and the reductions. -To compare with the mpi4py installation from the setup above, run the same -program with only its import changed: +Run the same program with mpi4py to compare: ```bash sed 's/^import prik_mpi as MPI$/from mpi4py import MPI/' mpi_example.py > mpi4py_example.py mpirun -n 2 python3 mpi4py_example.py ``` -The two programs print the same lines. +It prints the same lines. ## Compare call times -After building the wrapper, copy `benchmarks/openmpi_f08.py` from the PRIK -checkout beside `prik_mpi.py` and `prik_openmpi_f08.so`. Run the same five -operations through each API with the Open MPI launcher from the setup above: +`openmpi_f08.py` times the same operations through each API: ```bash -for backend in direct facade mpi4py; do - mpirun -n 2 python3 openmpi_f08.py "$backend" +for api in wrapped python mpi4py; do + mpirun -n 2 python3 openmpi_f08.py "$api" done ``` -The script reports nanoseconds per call for `Allreduce` with 1, 1,024, and -1,048,576 `np.int32` values, `Barrier`, and `Get_rank`. It uses the same two -ranks, arrays, prepared MPI handles, mpi4py `MPI.Wtime` timer, and repeated -batches for each backend. Compare results from one machine and one Open MPI -installation; timings vary by host. - -These results were measured on an AMD Ryzen 5 5600H running Ubuntu 22.04.5, -with Open MPI 5.0.11 built using GNU Fortran 11.4 and mpi4py 4.1.2 built -against that same installation. Each value is the median of three runs; each -run takes the fastest of five timed batches with two ranks. - -Time per call, compared with mpi4py: +It reports nanoseconds per call for `Allreduce` with 1, 1,024, and 1,048,576 +`np.int32` values, `Barrier`, and `Get_rank`, timing every API the same way: +one prepared call per loop iteration, the same buffer placement, and mpi4py's +`MPI.Wtime`. On an AMD Ryzen 5 5600H with Ubuntu 22.04.5, Open MPI 5.0.11 built +with GNU Fortran 11.4, and mpi4py 4.1.2 built against it, the median of five +runs was: -| Operation | mpi4py | Generated API | `prik_mpi.py` | +| Operation | mpi4py | Wrapped API | Python API | | --- | ---: | ---: | ---: | -| `Allreduce`, 1 `int32` | 1.101 µs | 0.738 µs (33% faster) | 0.887 µs (19% faster) | -| `Allreduce`, 1,024 `int32` values | 3.675 µs | 3.439 µs (6% faster) | 3.387 µs (8% faster) | -| `Allreduce`, 1,048,576 `int32` values | 2.247 ms | 2.272 ms (1% slower) | 2.106 ms (6% faster) | -| `Barrier` | 0.342 µs | 0.390 µs (14% slower) | 0.457 µs (34% slower) | -| `Get_rank` | 32 ns | 164 ns (about 5× slower) | 233 ns (about 7× slower) | +| `Allreduce`, 1 `int32` | 1.037 µs | 0.697 µs (33% faster) | 0.841 µs (19% faster) | +| `Allreduce`, 1,024 `int32` values | 2.314 µs | 1.779 µs (23% faster) | 2.027 µs (12% faster) | +| `Allreduce`, 1,048,576 `int32` values | 2.719 ms | 2.752 ms (about the same) | 2.827 ms (about the same) | +| `Barrier` | 0.330 µs | 0.328 µs (about the same) | 0.428 µs (30% slower) | +| `Get_rank` | 30 ns | 145 ns (about 5× slower) | 236 ns (about 8× slower) | `Get_rank` is the cheapest call, so its time is almost all the overhead of making a call. That overhead is small, but higher through PRIK than through -mpi4py, and can be optimized later. - -The relative figures describe this local run; the differences for the largest -`Allreduce` are small compared with its variation between runs. - -CI uploads JSON results for each Linux and macOS Open MPI job so you can -compare its measurements with this local run. +mpi4py, and can be optimized later. Timings vary by machine; compare APIs on +one machine and one Open MPI installation. ## Limitations -This tutorial selected eighteen names; the rest of `mpi_f08` works the same -way when you select it, within these limits of what PRIK supports today: - -- **Arrays of handles.** Routines taking an array of derived-type values, such - as the request and status arrays of `MPI_Waitall`, are not supported: - building a contract that selects one stops with an error naming the - argument. -- **Stored callbacks.** PRIK passes a Python callable as a callback that is - valid only during the call it is passed to. MPI keeps some callbacks for - later -- the copy and delete functions of `MPI_Comm_create_keyval`, or the - function given to `MPI_Op_create` -- and calls them after that call has - returned, which is not supported. See [Callbacks](../guide/callbacks.md). -- **Nonblocking buffers.** Routines such as `MPI_Isend` wrap, but the - operation keeps using its buffer after the call returns. PRIK does not hold - on to that NumPy array, so your program must keep it alive and unchanged - until the operation completes. This is also why `prik_mpi.py` offers no - `Isend` or `Irecv`: mpi4py's request objects keep their buffers alive, and - a faithful imitation would need arrays of requests, the first limitation. - -`prik_mpi.py` is an illustration, not a complete binding. Its buffers are -contiguous `np.int32` arrays only -- a strided view is refused with a -`TypeError` -- and it reports no status. Ranks and tags must be `np.int32`, -as in the program above; a plain Python `int` is refused with a `TypeError`, -where mpi4py accepts one. It has none of mpi4py's other datatypes, -communicators, pickled-object methods, or `MPI.Exception`: under Open MPI's -default error handler an MPI error aborts the job, and otherwise a nonzero -`ierror` raises the exception the contract's `@raises` produces. Each of these -is more Python over the same kind of generated calls, or more names in the -export list. +The eighteen names selected here are a start; the rest of `mpi_f08` wraps the +same way when you select it, within these limits: + +- **Arrays of handles**, such as the request and status arrays of + `MPI_Waitall`, are not supported yet; selecting one stops the build with an + error naming the argument. +- **Stored callbacks** are not supported: a Python callback is valid only + during the call it is passed to, and MPI keeps some for later, such as the + functions given to `MPI_Comm_create_keyval` or `MPI_Op_create`. See + [Callbacks](../guide/callbacks.md). +- **Nonblocking buffers** must stay alive: routines such as `MPI_Isend` wrap, + but your program must keep the array alive and unchanged until the operation + completes. + +`prik_mpi.py` illustrates the approach rather than covering mpi4py. It takes +contiguous `np.int32` buffers and `np.int32` ranks and tags only, where mpi4py +also accepts other datatypes and plain Python integers, and it reports no +status. More of mpi4py is more Python over the same kind of wrapped calls, or +more names in `mpi_exports.txt`. diff --git a/tests/fortran/assumed_types/end_to_end/test_openmpi_f08.py b/tests/fortran/assumed_types/end_to_end/test_openmpi_f08.py index 5b7503e72..503f17180 100644 --- a/tests/fortran/assumed_types/end_to_end/test_openmpi_f08.py +++ b/tests/fortran/assumed_types/end_to_end/test_openmpi_f08.py @@ -302,7 +302,7 @@ def run(script: str) -> list[str]: shutil.copyfile(BENCHMARK, tmp_path / BENCHMARK.name) report_dir = Path(os.environ["PRIK_OPENMPI_BENCHMARK_DIR"]) report_dir.mkdir(parents=True, exist_ok=True) - for backend in ("direct", "facade", "mpi4py"): + for backend in ("wrapped", "python", "mpi4py"): result = subprocess.run( [launcher, "-n", "2", sys.executable, BENCHMARK.name, backend], capture_output=True, From 90c88a262ff7609285586939fa4a2f8653e80e4a Mon Sep 17 00:00:00 2001 From: said Date: Sun, 27 Sep 2026 11:42:52 +0100 Subject: [PATCH 09/11] codex: drop the np.int32 explanations from the Open MPI tutorial Co-Authored-By: Claude Opus 5.5 --- docs/user/tutorials/openmpi-f08.md | 7 ------- .../end_to_end/fixtures/runtime/mpi_example.py | 2 -- .../assumed_types/end_to_end/fixtures/runtime/prik_mpi.py | 2 -- 3 files changed, 11 deletions(-) diff --git a/docs/user/tutorials/openmpi-f08.md b/docs/user/tutorials/openmpi-f08.md index df438e93d..31cc30090 100644 --- a/docs/user/tutorials/openmpi-f08.md +++ b/docs/user/tutorials/openmpi-f08.md @@ -27,8 +27,6 @@ comm = MPI.COMM_WORLD rank = comm.Get_rank() size = comm.Get_size() -# Buffers are np.int32 arrays, and ranks and tags are np.int32 too: -# Get_rank returns one, and rank + 1 stays one. ROOT = np.int32(0) TAG = np.int32(77) @@ -474,8 +472,6 @@ import numpy as np from prik_openmpi_f08 import mpi_f08 as _mpi -# Ranks and tags are np.int32, the type the contract takes: the extension -# returns them as np.int32, and so are these constants and defaults. ANY_SOURCE = _mpi.mpi_any_source ANY_TAG = _mpi.mpi_any_tag IN_PLACE = _mpi.mpi_in_place @@ -528,9 +524,6 @@ atexit.register(_mpi.finalize) is one call to the wrapped API. - Every buffer is an `np.int32` array sent as `MPI_INT`, and `Recv` passes `MPI_STATUS_IGNORE`, as mpi4py does when given no status. -- Ranks and tags are `np.int32`, the type the contract takes, from the start: - `Get_rank` returns one, and `rank + 1` stays one. They pass straight - through without conversion. - As with mpi4py, importing the module starts MPI, and exiting stops it. ## 7. Run it diff --git a/tests/fortran/assumed_types/end_to_end/fixtures/runtime/mpi_example.py b/tests/fortran/assumed_types/end_to_end/fixtures/runtime/mpi_example.py index 3ebdfd48a..367fe2b30 100644 --- a/tests/fortran/assumed_types/end_to_end/fixtures/runtime/mpi_example.py +++ b/tests/fortran/assumed_types/end_to_end/fixtures/runtime/mpi_example.py @@ -6,8 +6,6 @@ rank = comm.Get_rank() size = comm.Get_size() -# Buffers are np.int32 arrays, and ranks and tags are np.int32 too: -# Get_rank returns one, and rank + 1 stays one. ROOT = np.int32(0) TAG = np.int32(77) diff --git a/tests/fortran/assumed_types/end_to_end/fixtures/runtime/prik_mpi.py b/tests/fortran/assumed_types/end_to_end/fixtures/runtime/prik_mpi.py index 97e2187ff..52554a313 100644 --- a/tests/fortran/assumed_types/end_to_end/fixtures/runtime/prik_mpi.py +++ b/tests/fortran/assumed_types/end_to_end/fixtures/runtime/prik_mpi.py @@ -10,8 +10,6 @@ from prik_openmpi_f08 import mpi_f08 as _mpi -# Ranks and tags are np.int32, the type the contract takes: the extension -# returns them as np.int32, and so are these constants and defaults. ANY_SOURCE = _mpi.mpi_any_source ANY_TAG = _mpi.mpi_any_tag IN_PLACE = _mpi.mpi_in_place From 1175a4c71e0da80d6348c2e41a5dc332840e950b Mon Sep 17 00:00:00 2001 From: said Date: Sun, 27 Sep 2026 11:44:23 +0100 Subject: [PATCH 10/11] codex: cut the Open MPI tutorial to the steps and the results Co-Authored-By: Claude Opus 5.5 --- docs/user/tutorials/openmpi-f08.md | 257 +++++++---------------------- 1 file changed, 58 insertions(+), 199 deletions(-) diff --git a/docs/user/tutorials/openmpi-f08.md b/docs/user/tutorials/openmpi-f08.md index 31cc30090..6738f8f34 100644 --- a/docs/user/tutorials/openmpi-f08.md +++ b/docs/user/tutorials/openmpi-f08.md @@ -11,11 +11,11 @@ publication: reviewed # Wrap Open MPI `mpi_f08` for Python PRIK reads Open MPI's Fortran `mpi_f08` sources and generates a Python -extension that calls the installed Open MPI library. You write no C, Cython, -or `ctypes`. You edit the generated `.pyi` contract into the Python signatures -you want, add a short Python module in the style of -[mpi4py](https://mpi4py.readthedocs.io), and this program runs under -`mpirun`: +extension that calls the installed Open MPI library, with no hand-written C, +Cython, or `ctypes`. You edit the generated `.pyi` contract into the Python +signatures you want and add a short [mpi4py](https://mpi4py.readthedocs.io)-style +module. The result runs this program, which mpi4py also runs unchanged apart +from its import: ```python @@ -57,33 +57,23 @@ if rank == 0: print(f"rank 0 max {largest.tolist()}") ``` -It is also a valid mpi4py program: replace `import prik_mpi as MPI` with -`from mpi4py import MPI` and mpi4py prints the same lines. - -You build two APIs on the way: - -- **The wrapped API**, `prik_openmpi_f08`: the extension PRIK generates from - Open MPI's declarations and your edited contract. Every call goes straight - into Open MPI. -- **The Python API**, `prik_mpi.py`: about fifty lines of Python that give the - wrapped API mpi4py's `Comm` object and keyword defaults. - -On small messages both are faster than mpi4py; see +You build two APIs: the **wrapped API** (`prik_openmpi_f08`), generated by +PRIK, and the **Python API** (`prik_mpi.py`), a few lines of Python on top of +it. Both are faster than mpi4py on small messages; see [Compare call times](#compare-call-times). ## What you need -- Linux or macOS, with Python 3.10 or later, NumPy, and PRIK installed; see +- Linux or macOS, Python 3.10 or later, NumPy, and PRIK; see [Installation](../getting-started/installation.md). - GNU Fortran and GCC of the same version; CI uses version 13. -- Open MPI and mpi4py, built as shown in the next section. -- The files this tutorial uses, from the PRIK repository: +- The tutorial's files, from the PRIK repository: [`mpi_exports.txt`](https://github.com/PyNumLab/prik/blob/main/tests/fortran/assumed_types/end_to_end/fixtures/contracts/openmpi/mpi_exports.txt), [`mpi_f08.pyi`](https://github.com/PyNumLab/prik/blob/main/tests/fortran/assumed_types/end_to_end/fixtures/contracts/openmpi/mpi_f08.pyi), [`prik_mpi.py`](https://github.com/PyNumLab/prik/blob/main/tests/fortran/assumed_types/end_to_end/fixtures/runtime/prik_mpi.py), [`mpi_example.py`](https://github.com/PyNumLab/prik/blob/main/tests/fortran/assumed_types/end_to_end/fixtures/runtime/mpi_example.py), and - [`openmpi_f08.py`](https://github.com/PyNumLab/prik/blob/main/benchmarks/openmpi_f08.py). Each is also shown in - full below. Download them into an empty working directory: + [`openmpi_f08.py`](https://github.com/PyNumLab/prik/blob/main/benchmarks/openmpi_f08.py). Download them into an + empty working directory: ```bash PRIK_RAW=https://raw.githubusercontent.com/PyNumLab/prik/main @@ -99,11 +89,9 @@ done ## Install Open MPI and mpi4py -This tutorial uses [Open MPI 5.0.11](https://www.open-mpi.org/software/ompi/v5.0/); -CI also tests Open MPI 4.1.8, on Linux and macOS. Build it from source and -keep its source and configured build trees: PRIK reads them to generate the -contract, then links the extension to the same installation. Run these -commands in one shell, from your working directory: +Build [Open MPI 5.0.11](https://www.open-mpi.org/software/ompi/v5.0/) from +source and keep its source and build trees, which PRIK reads. From your +working directory: ```bash OMPI_ROOT="$HOME/openmpi-5.0.11" @@ -125,8 +113,7 @@ OMPI_SRC="$OMPI_ROOT/source" OMPI_BUILD="$OMPI_ROOT/build" ``` -Then build mpi4py against this Open MPI rather than installing a prebuilt -wheel, and check that both report Open MPI 5.0.11: +Build mpi4py against the same Open MPI; both checks should report 5.0.11: ```bash MPI4PY_BUILD_MPICC="$OMPI_ROOT/install/bin/mpicc" \ @@ -135,42 +122,13 @@ mpirun --version python3 -c 'from mpi4py import MPI; print(MPI.Get_library_version())' ``` -Use this installation's `mpifort` and `mpirun` throughout. For Open MPI 4.1.8, -change `5.0.11` to `4.1.8` and `v5.0` to `v4.1` in these commands. +For Open MPI 4.1.8, which CI also tests, change `5.0.11` to `4.1.8` and `v5.0` +to `v4.1`. ## 1. Choose what to wrap -`mpi_f08` is an ordinary Fortran module that gathers others. Its handle types -and predefined objects are Fortran declarations: - -```fortran -type(MPI_Comm), parameter :: MPI_COMM_WORLD = MPI_Comm(OMPI_MPI_COMM_WORLD) -type(MPI_Op), parameter :: MPI_SUM = MPI_Op(OMPI_MPI_SUM) -type(MPI_Datatype), parameter :: MPI_INT = MPI_Datatype(OMPI_MPI_INT) - -integer, bind(C, name="mpi_fortran_in_place_") :: MPI_IN_PLACE -type(MPI_Status), bind(C, name="mpi_fortran_status_ignore_") :: MPI_STATUS_IGNORE -``` - -and each MPI routine is a generic interface: - -```fortran -interface MPI_Allreduce - subroutine MPI_Allreduce_f08(sendbuf, recvbuf, count, datatype, op, comm, ierror) - use :: mpi_f08_types, only : MPI_Datatype, MPI_Op, MPI_Comm - type(*), dimension(*), intent(in) :: sendbuf - type(*), dimension(*) :: recvbuf - integer, intent(in) :: count - type(MPI_Datatype), intent(in) :: datatype - type(MPI_Op), intent(in) :: op - type(MPI_Comm), intent(in) :: comm - integer, optional, intent(out) :: ierror - end subroutine MPI_Allreduce_f08 -end interface MPI_Allreduce -``` - -`mpi_f08` publishes hundreds of names. `mpi_exports.txt` selects the ones -the program needs, each qualified by the module that publishes it: +`mpi_exports.txt` selects the names the program needs from `mpi_f08`'s +hundreds: ```text @@ -194,14 +152,10 @@ mpi_f08::MPI_ANY_SOURCE mpi_f08::MPI_ANY_TAG ``` -PRIK publishes exactly these names and keeps the declarations they depend on, -such as the handle types in their signatures. `--export-symbols` works this -way for any Fortran project; nothing here is specific to MPI. - ## 2. Generate the contract -Point PRIK at the one entry source. It follows each `use` to the source that -defines that module, under both Open MPI trees: +PRIK starts from the entry source and follows each `use` through both Open MPI +trees. The `-I` directories are where those sources find their headers: ```bash python3 -m prik generate --pyi \ @@ -219,32 +173,9 @@ python3 -m prik generate --pyi \ -I "$OMPI_SRC/ompi/include" ``` -The `-I` directories are where Open MPI's Fortran sources find their headers, -some of which `configure` wrote into the build tree. PRIK reads the sources -to learn their declarations; it does not compile them. - -## 3. Read the contract - -`contract/` holds one editable `.pyi` file per Fortran module the selection -needs. The predefined objects keep their Fortran types, in -`contract/mpi_f08_types.pyi`: - -```python -mpi_any_source: Final[Int32] = -1 - -mpi_comm_world: Final[Mpi_Comm] - -mpi_sum: Final[Mpi_Op] - -mpi_int: Final[Mpi_Datatype] - -mpi_in_place: Int32[()] - -mpi_status_ignore: Mpi_Status -``` - -and each routine keeps its Fortran interface, in -`contract/mpi_f08_interfaces.pyi`: +`contract/` now holds one `.pyi` file per Fortran module, with Open MPI's +handle types, constants, and routines as declared in Fortran. For example, +`contract/mpi_f08_interfaces.pyi` has: ```python @overload("mpi_allreduce_f08") @@ -259,27 +190,9 @@ def mpi_allreduce( ) -> Returns["ierror", Int32[()]] | None: ... ``` -These excerpts come from Open MPI 5.0, which declares the handle types in -`mpi_types`; Open MPI 4.1 declares them in `mpi_f08_types`. Three mappings -matter for MPI: - -- **Handles are typed objects.** `Mpi_Comm`, `Mpi_Datatype`, `Mpi_Op`, and - `Mpi_Status` are Open MPI's derived types with their real fields, and a - handle argument accepts only its own type. -- **Buffers are `AnyNative`.** MPI's `type(*)` buffers accept any contiguous - NumPy array, passed by address. -- **Special objects are Open MPI's own.** `mpi_in_place` and - `mpi_status_ignore` are live views of Open MPI's `MPI_IN_PLACE` and - `MPI_STATUS_IGNORE` variables. Passing one passes that variable's address, - which is how Open MPI recognizes it. - -This contract already builds. Its functions are still Fortran's, though: every -count is an argument, and every call returns an error code. +## 3. Edit the contract -## 4. Edit the contract into a Python API - -Replace `contract/mpi_f08.pyi`, the module the extension publishes, with the -edited file you downloaded: +Replace the published module with the edited one you downloaded: ```bash cp mpi_f08.pyi contract/mpi_f08.pyi @@ -405,23 +318,17 @@ __all__ = [ ] ``` -Each function still calls the Fortran routine it names; only its Python -signature changes. `@native_call` lists the Fortran arguments in order and -says where each comes from: - -| Edit | Example | Effect in Python | -| --- | --- | --- | -| `@bind("MPI_Send")` on `def send` | every function | The Python name differs from the Fortran name it calls. | -| `Int32(Arg(0).size)` | `count` of `MPI_Send` | The count comes from the buffer, so the caller does not pass it. | -| `Return("rank", 0)` | `rank` of `MPI_Comm_rank` | The output argument becomes the return value. | -| `Hidden("ierror", Int32)` with `@raises(status="ierror", success=0)` | every function | The error code is not an argument; a nonzero code raises an exception. | +Each function still calls the Fortran routine it names; the edits change only +its Python signature: -The same file works with Open MPI 4.1 and 5.0. +| Edit | Effect in Python | +| --- | --- | +| `@bind("MPI_Send")` on `def send` | The Python name differs from the Fortran name. | +| `Int32(Arg(0).size)` | The count comes from the buffer, so the caller does not pass it. | +| `Return("rank", 0)` | The output argument becomes the return value. | +| `Hidden("ierror", Int32)` with `@raises(status="ierror", success=0)` | A nonzero error code raises an exception. | -## 5. Build the wrapped API - -Build the edited contract, asking `mpifort` for the compiler, flags, and -libraries of your Open MPI: +## 4. Build the wrapped API ```bash python3 -m prik contract/__init__.pyi \ @@ -433,30 +340,13 @@ python3 -m prik contract/__init__.pyi \ --out-dir build ``` -This writes `prik_openmpi_f08.so` in the working directory. Only the code -PRIK generates is compiled; the extension links to the installed Open MPI. -`--compiler` takes the compiler `mpifort` runs, because PRIK recognizes a -compiler by its executable name. +This writes `prik_openmpi_f08.so` in the working directory. Only PRIK's +generated code is compiled; the extension links to the installed Open MPI. -The wrapped API already runs MPI: +## 5. Add the Python API -```python -import numpy as np - -from prik_openmpi_f08 import mpi_f08 - -mpi_f08.init() -values = np.array([1, 2], dtype=np.int32) -total = np.empty_like(values) -mpi_f08.allreduce(values, total, mpi_f08.mpi_int, mpi_f08.mpi_sum, mpi_f08.mpi_comm_world) -mpi_f08.finalize() -``` - -## 6. Add the Python API - -mpi4py's object model is Python, so the Python API is plain Python over the -wrapped API. `prik_mpi.py`, downloaded beside `prik_openmpi_f08.so`, is -deliberately small: +`prik_mpi.py` gives the wrapped API mpi4py's `Comm` object. It is kept small: +one datatype, `MPI_INT`, and no receive status. ```python @@ -520,21 +410,13 @@ _mpi.init() atexit.register(_mpi.finalize) ``` -- `Comm` wraps an `Mpi_Comm` handle and spells mpi4py's methods; each method - is one call to the wrapped API. -- Every buffer is an `np.int32` array sent as `MPI_INT`, and `Recv` passes - `MPI_STATUS_IGNORE`, as mpi4py does when given no status. -- As with mpi4py, importing the module starts MPI, and exiting stops it. - -## 7. Run it - -Run the program from the top of this page with two ranks: +## 6. Run it ```bash mpirun -n 2 python3 mpi_example.py ``` -The ranks print these lines, in whichever order they finish: +The ranks print, in either order: ```text rank 1 received [0, 1, 2, 3] @@ -543,35 +425,24 @@ rank 0 max [2, 3] rank 1 of 2: bcast [0, 1, 2], sum [3, 5], in place [3, 5] ``` -Each call went from `prik_mpi.py` through the wrapped API into Open MPI, and -the NumPy arrays crossed as native buffers, written in place by `Recv`, -`Bcast`, and the reductions. - -Run the same program with mpi4py to compare: +The same program under mpi4py prints the same lines: ```bash sed 's/^import prik_mpi as MPI$/from mpi4py import MPI/' mpi_example.py > mpi4py_example.py mpirun -n 2 python3 mpi4py_example.py ``` -It prints the same lines. - ## Compare call times -`openmpi_f08.py` times the same operations through each API: - ```bash for api in wrapped python mpi4py; do mpirun -n 2 python3 openmpi_f08.py "$api" done ``` -It reports nanoseconds per call for `Allreduce` with 1, 1,024, and 1,048,576 -`np.int32` values, `Barrier`, and `Get_rank`, timing every API the same way: -one prepared call per loop iteration, the same buffer placement, and mpi4py's -`MPI.Wtime`. On an AMD Ryzen 5 5600H with Ubuntu 22.04.5, Open MPI 5.0.11 built -with GNU Fortran 11.4, and mpi4py 4.1.2 built against it, the median of five -runs was: +`openmpi_f08.py` times every API the same way with mpi4py's `MPI.Wtime`. On an +AMD Ryzen 5 5600H with Ubuntu 22.04.5, Open MPI 5.0.11, and mpi4py 4.1.2, the +median of five runs was: | Operation | mpi4py | Wrapped API | Python API | | --- | ---: | ---: | ---: | @@ -581,29 +452,17 @@ runs was: | `Barrier` | 0.330 µs | 0.328 µs (about the same) | 0.428 µs (30% slower) | | `Get_rank` | 30 ns | 145 ns (about 5× slower) | 236 ns (about 8× slower) | -`Get_rank` is the cheapest call, so its time is almost all the overhead of -making a call. That overhead is small, but higher through PRIK than through -mpi4py, and can be optimized later. Timings vary by machine; compare APIs on -one machine and one Open MPI installation. +`Get_rank` does almost nothing, so its time is the overhead of a call, which +PRIK can still reduce. ## Limitations -The eighteen names selected here are a start; the rest of `mpi_f08` wraps the -same way when you select it, within these limits: - -- **Arrays of handles**, such as the request and status arrays of - `MPI_Waitall`, are not supported yet; selecting one stops the build with an - error naming the argument. -- **Stored callbacks** are not supported: a Python callback is valid only - during the call it is passed to, and MPI keeps some for later, such as the - functions given to `MPI_Comm_create_keyval` or `MPI_Op_create`. See - [Callbacks](../guide/callbacks.md). -- **Nonblocking buffers** must stay alive: routines such as `MPI_Isend` wrap, - but your program must keep the array alive and unchanged until the operation - completes. - -`prik_mpi.py` illustrates the approach rather than covering mpi4py. It takes -contiguous `np.int32` buffers and `np.int32` ranks and tags only, where mpi4py -also accepts other datatypes and plain Python integers, and it reports no -status. More of mpi4py is more Python over the same kind of wrapped calls, or -more names in `mpi_exports.txt`. +- Routines taking arrays of handles, such as `MPI_Waitall`, are not supported + yet. +- A Python callback is valid only during the call it is passed to, so MPI + callbacks kept for later, such as those of `MPI_Op_create`, are not + supported. See [Callbacks](../guide/callbacks.md). +- Nonblocking routines such as `MPI_Isend` work, but your program must keep + the buffer alive until the operation completes. +- `prik_mpi.py` accepts `np.int32` buffers, ranks, and tags only; a plain + Python `int` is refused with a `TypeError`. From 95ba1d403c5cb1117b583c1bb70759995a5e4c5a Mon Sep 17 00:00:00 2001 From: said Date: Sun, 27 Sep 2026 11:49:45 +0100 Subject: [PATCH 11/11] codex: remove the Open MPI tutorial's limitations section Co-Authored-By: Claude Opus 5.5 --- docs/user/tutorials/openmpi-f08.md | 12 ------------ 1 file changed, 12 deletions(-) diff --git a/docs/user/tutorials/openmpi-f08.md b/docs/user/tutorials/openmpi-f08.md index 6738f8f34..70a435855 100644 --- a/docs/user/tutorials/openmpi-f08.md +++ b/docs/user/tutorials/openmpi-f08.md @@ -454,15 +454,3 @@ median of five runs was: `Get_rank` does almost nothing, so its time is the overhead of a call, which PRIK can still reduce. - -## Limitations - -- Routines taking arrays of handles, such as `MPI_Waitall`, are not supported - yet. -- A Python callback is valid only during the call it is passed to, so MPI - callbacks kept for later, such as those of `MPI_Op_create`, are not - supported. See [Callbacks](../guide/callbacks.md). -- Nonblocking routines such as `MPI_Isend` work, but your program must keep - the buffer alive until the operation completes. -- `prik_mpi.py` accepts `np.int32` buffers, ranks, and tags only; a plain - Python `int` is refused with a `TypeError`.