Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
23 changes: 23 additions & 0 deletions .github/precomputed-examples.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,23 @@
{
"_comment": "Examples too expensive to simulate in CI. run_example.py downloads each archived run (a tar.gz of docs/public/examples/struphy_gallery_runs/<sim_folder>, made by scripts/pack_precomputed_run.py), checks its sha256, unpacks it and only post-processes it. Fill in url and sha256 after uploading the archive to a release. Each job is the Slurm allocation scripts/submit_precomputed_run.py requests for the simulation (estimates; adjust after the first run).",
"itpa-tae-linear-mhd": {
"sim_folder": "itpa_tae_linear_mhd",
"url": null,
"sha256": null,
"job": {
"mpi_ranks": 32,
"time": "08:00:00",
"mem": null
}
},
"itpa-tae-shear-alfven": {
"sim_folder": "itpa_tae_shear_alfven",
"url": null,
"sha256": null,
"job": {
"mpi_ranks": 32,
"time": "24:00:00",
"mem": null
}
}
}
5 changes: 3 additions & 2 deletions .github/scripts/example-cache.cjs
Original file line number Diff line number Diff line change
Expand Up @@ -2,13 +2,14 @@
async function exampleCache(example, prefix, suffix, glob) {
const patterns = [
`docs/src/examples/${example}.py`,
'submodules/plasma-plots/src/plasma_plots/**/*.py',
'requirements.txt',
'requirements-examples.txt',
'catalogue_docs.py',
'generate_examples.py',
'run_example.py',
'.github/precomputed-examples.json',
'generate_example_domain.py',
'generate_domains.py',
'submodules/plasma-plots/struphy/src/struphy/**/*.py',
];
return {
key: `${prefix}${suffix}-${example}-${await glob.hashFiles(patterns.join('\n'))}`,
Expand Down
4 changes: 3 additions & 1 deletion .github/scripts/list-examples.cjs
Original file line number Diff line number Diff line change
@@ -1,8 +1,10 @@
const fs = require('node:fs/promises');
const { exampleCache } = require('./example-cache.cjs');

const CI_EXCLUDED = ['itpa-tae-linear-mhd', 'itpa-tae-shear-alfven'];

module.exports = async function listExamples({ cache, glob, core, exclude = ['poisson-source'] }) {
const excluded = new Set(exclude);
const excluded = new Set([...CI_EXCLUDED, ...exclude]);
const examples = (await fs.readdir('docs/src/examples'))
.filter(name => name.endsWith('.py') && !name.startsWith('_') && !excluded.has(name.slice(0, -3)))
.map(name => name.slice(0, -3)).sort();
Expand Down
4 changes: 2 additions & 2 deletions .github/scripts/list-examples.test.cjs
Original file line number Diff line number Diff line change
Expand Up @@ -80,6 +80,6 @@ test('shared keys retain prefix, suffix, example and source hash', async () => {
});
assert.equal(result.key, 'pitagora-example-v1-custom-alpha-hash');
assert.equal(patterns[0], 'docs/src/examples/alpha.py');
assert.ok(patterns.includes('submodules/plasma-plots/struphy/src/struphy/**/*.py'));
assert.ok(patterns.includes('submodules/plasma-plots/src/plasma_plots/**/*.py'));
assert.ok(patterns.includes('requirements.txt'));
assert.ok(patterns.includes('requirements-examples.txt'));
});
5 changes: 5 additions & 0 deletions .github/scripts/list-pitagora-examples.cjs
Original file line number Diff line number Diff line change
Expand Up @@ -2,6 +2,7 @@ const fs = require('node:fs/promises');
const { exampleCache } = require('./example-cache.cjs');

const selectionPath = '.github/pitagora-examples.txt';
const CI_EXCLUDED = new Set(['itpa-tae-linear-mhd', 'itpa-tae-shear-alfven']);

async function selectedPitagoraExamples() {
const text = await fs.readFile(selectionPath, 'utf8');
Expand All @@ -10,6 +11,10 @@ async function selectedPitagoraExamples() {
.map(line => line.split('#', 1)[0].trim())
.filter(Boolean);
if (!examples.length) throw new Error(`${selectionPath} contains no examples`);
const excluded = examples.filter(example => CI_EXCLUDED.has(example));
if (excluded.length) {
throw new Error(`${selectionPath} includes examples disabled in CI: ${excluded.join(', ')}`);
}
if (new Set(examples).size !== examples.length) {
throw new Error(`${selectionPath} contains duplicate examples`);
}
Expand Down
64 changes: 10 additions & 54 deletions .github/scripts/run-pitagora-example.py
Original file line number Diff line number Diff line change
Expand Up @@ -6,19 +6,11 @@
import argparse
import os
import shlex
import sys
from pathlib import Path

from slurm_script_generator.slurm_script import SlurmScript
from slurm_script_generator.squeue import SQueue

MODULES = [
"gcc/12.3.0",
"python/3.11.7",
"hdf5/1.14.3--gcc--12.3.0",
"cmake/3.27.9",
"netcdf-fortran/4.6.1--gcc--12.3.0",
"netlib-scalapack/2.2.0--openmpi--4.1.6--gcc--12.3.0-ucx1.20",
]
sys.path.insert(0, str(Path(__file__).resolve().parents[2] / "scripts"))
from slurm_jobs import environment_commands, example_job, submit_and_wait # noqa: E402

# Orszag--Tang writes 401 field snapshots and post-processing loads them all
# before evaluating the output grid. It exceeds the debug partition's default
Expand All @@ -27,15 +19,6 @@
MPI_RANKS_BY_EXAMPLE = {"orszag-tang-vortex": 4}


def tail_log(path: Path, lines: int = 200) -> None:
"""Print a bounded batch log in a collapsible GitHub Actions group."""
if not path.is_file():
return
print(f"::group::{path.name}")
print("\n".join(path.read_text(errors="replace").splitlines()[-lines:]))
print("::endgroup::")


def main() -> int:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("example", help="Gallery example stem")
Expand All @@ -49,48 +32,21 @@ def main() -> int:
run_id = os.environ["GITHUB_RUN_ID"]
run_attempt = os.environ["GITHUB_RUN_ATTEMPT"]
job_name = f"struphy-{args.example}-{run_id}-{run_attempt}"
script_path = runner_temp / f"{job_name}.sbatch"
stdout = workspace / f"slurm-{job_name}-%j.out"
stderr = workspace / f"slurm-{job_name}-%j.err"
mpi_ranks = MPI_RANKS_BY_EXAMPLE.get(args.example, 1)
commands = [
"set -euo pipefail",
f"source {shlex.quote(str(virtual_env / 'bin' / 'activate'))}",
]
if mpi_ranks == 1:
# A one-task Slurm allocation is not an MPI launch.
commands.append("export STRUPHY_MPI=0")
else:
# mpi4py must see the Open MPI environment that `cli.py --mpi` creates.
commands.append("unset STRUPHY_MPI")
commands = environment_commands(virtual_env, mpi_ranks)
commands.append(f"python cli.py run {shlex.quote(args.example)} --mpi {mpi_ranks}")

script = SlurmScript(
script = example_job(
job_name=job_name,
account=args.account,
partition=args.partition,
nodes=1,
ntasks=mpi_ranks,
cpus_per_task=1,
time="00:30:00",
workspace=workspace,
log_dir=workspace,
commands=commands,
mpi_ranks=mpi_ranks,
mem=MEMORY_BY_EXAMPLE.get(args.example),
chdir=str(workspace),
output=str(stdout),
error=str(stderr),
modules=MODULES,
custom_commands=commands,
)
job_id = script.submit_job(path=str(script_path), verbose=True)
print(f"Submitted {job_name} as Slurm job {job_id}")

try:
state = SQueue().wait_until_done(job_id=job_id, poll_interval=15, check=True)[
job_id
]
print(f"Slurm job {job_id} finished with state {state or 'unknown'}.")
finally:
tail_log(workspace / f"slurm-{job_name}-{job_id}.out")
tail_log(workspace / f"slurm-{job_name}-{job_id}.err")
submit_and_wait(script, job_name, runner_temp / f"{job_name}.sbatch", workspace)
return 0


Expand Down
12 changes: 7 additions & 5 deletions .github/workflows/build-site.yml
Original file line number Diff line number Diff line change
Expand Up @@ -27,8 +27,6 @@ jobs:
steps:
- name: Check out repository
uses: actions/checkout@v4
with:
submodules: recursive

- name: Install cache client
run: npm install --prefix "$RUNNER_TEMP/example-cache" --no-audit --no-fund --ignore-scripts @actions/cache@4.0.5
Expand Down Expand Up @@ -88,16 +86,17 @@ jobs:
steps:
- name: Check out repository
uses: actions/checkout@v4
with:
submodules: recursive

- name: Set up Python
uses: actions/setup-python@v5
with:
python-version: '3.12'

- name: Install Struphy
run: python -m pip install --upgrade pip && python -m pip install ./submodules/plasma-plots/struphy ./submodules/plasma-plots
run: python -m pip install --upgrade pip && python -m pip install -r requirements.txt

- name: Check domain discovery
run: python -m pytest -q tests/test_generate_domains.py

- name: Generate equilibrium slices
run: python generate_equilibrium_slices.py
Expand Down Expand Up @@ -139,6 +138,9 @@ jobs:
cache: npm
cache-dependency-path: docs/package-lock.json

- name: Check example build validation
run: python -m pytest -q tests/test_example_build.py

- name: Install dependencies
working-directory: docs
run: npm ci
Expand Down
10 changes: 4 additions & 6 deletions .github/workflows/pitagora-examples.yml
Original file line number Diff line number Diff line change
Expand Up @@ -27,10 +27,8 @@ jobs:
SLURM_PARTITION: dcgp_fua_dbg
EXAMPLES: ${{ inputs.examples }}
steps:
- name: Check out repository and Struphy
- name: Check out repository
uses: actions/checkout@v4
with:
submodules: recursive

- name: Install Struphy and run example
# A login shell initializes the cluster's module command. Keep setup and
Expand All @@ -53,13 +51,13 @@ jobs:
export CXX="$(command -v g++)"
source /pitagora_scratch/userexternal/mlindqvi/github_runner_builds/.venv/bin/activate

git -C submodules/plasma-plots/struphy rev-parse HEAD
python -m pip install './submodules/plasma-plots/struphy[phys,mpi]' './submodules/plasma-plots[gallery]'
python -m pip uninstall -y struphy
python -m pip install --upgrade -r requirements-examples.txt
python -m pip install 'scope-profiler[pproc]>=0.6.1,<=0.7.0' kaleido 'slurm-script-generator==0.4.0'
# Install Chrome into Kaleido's user-accessible location for PNG export.
kaleido_get_chrome
struphy compile --status
# struphy compile -y --language fortran
struphy compile -y --language fortran
# The Python API generates a self-contained serial batch script,
# submits it, waits for its exact job ID, verifies its final Slurm
# state, and prints its batch logs. Submit exactly one job at a time
Expand Down
73 changes: 24 additions & 49 deletions .github/workflows/run-example.yml
Original file line number Diff line number Diff line change
Expand Up @@ -29,17 +29,9 @@ jobs:
steps:
- name: Check out repository
uses: actions/checkout@v4
with:
submodules: recursive

- name: Trust the checked-out worktree
# Actions containers run as root; git refuses to touch a worktree owned by another
# UID ("detected dubious ownership") unless told to trust it -- struphy's own
# editable install below shells out to git (setuptools-scm) and would hit this.
run: git config --global --add safe.directory '*'

# Each run is deterministic (fixed seeds), so cache on this example script's and
# Struphy's own source, and skip the expensive rebuild when neither changed. Every
# Python requirements, and skip the expensive rebuild when neither changed. Every
# file an example writes is named after its script stem (the metadata JSON, the
# figures, their thumbnails); no example's stem is a prefix of another's, and the
# glob also covers two-stream-instability's extra "-phasespace" pair.
Expand All @@ -50,40 +42,26 @@ jobs:
example: ${{ inputs.example }}
cache-prefix: struphy-example-serial-v1

- name: Install Struphy from this repo's pinned submodule commit
# Same approach as struphy's own CI
# (submodules/plasma-plots/struphy/.github/actions/install/struphy_in_container/action.yml):
# the image bakes in whatever struphy commit was HEAD when it was last built, so
# move /struphy_fortran_'s own checkout to the exact commit our submodule pins
# (rather than reinstalling editable from a second copy of the source) and
# reinstall there, reusing the image's already-installed venv and already-built
# gvec/desc-opt.
#
# scope-profiler[pproc] and kaleido aren't pulled in by the phys/mpi extras (only by
# struphy's own dev/doc extras) but the example scripts need both: scope-profiler's
# pproc submodule for the durations/gantt/flame plot-data export, kaleido for
# figure.write_image(...)'s static PNG export.
#
# plasma-plots registers the .plasma accessors that the example scripts use, and its
# gallery extra brings plasma_plots.gallery, the helpers that write every example's
# figures, profiling and metadata. It has no compiled parts, so it is installed straight
# from our pinned submodule.
- name: Install Python dependencies from PyPI
id: install-python
# Replace the image's editable Struphy install with a released wheel.
# The gallery extra provides Plotly exports and profiling helpers.
if: steps.cache-example.outputs.cache-hit != 'true'
shell: bash
run: |
STRUPHY_PLOTS="$PWD/submodules/plasma-plots"
STRUPHY_SHA=$(git -C submodules/plasma-plots/struphy rev-parse HEAD)
cd /struphy_fortran_
git fetch origin --tags
git checkout "$STRUPHY_SHA"
source env_fortran_/bin/activate
pip install -U --upgrade-strategy eager -e ".[phys,mpi]"
# Struphy pins scope-profiler <=0.5.0 as a hard dependency, but the example
# scripts use plot_durations(metrics=(...)), a plural-metrics API that only
# exists from 0.6.1 onward -- --upgrade forces past struphy's conservative pin
# to the version this was actually developed and tested against.
pip install --upgrade "scope-profiler[pproc]>=0.6.1" kaleido
pip install "$STRUPHY_PLOTS[gallery]"
source /struphy_fortran_/env_fortran_/bin/activate
python -m pip uninstall -y struphy
python -m pip install --upgrade -r requirements-examples.txt
python - <<'PY'
import os
import sysconfig
from importlib.metadata import distribution, version
with open(os.environ["GITHUB_OUTPUT"], "a") as output:
print(f"struphy-path={distribution('struphy').locate_file('struphy')}", file=output)
print(f"struphy-version={version('struphy')}", file=output)
print(f"python-abi={sysconfig.get_config_var('SOABI')}", file=output)
print(f"pyccel-version={version('pyccel')}", file=output)
PY

- name: Rebuild GVEC for this runner's CPU
# The published gvec wheel is built with -march=native. GitHub-hosted
Expand All @@ -97,22 +75,19 @@ jobs:
lscpu | grep -E 'Model name|Flags'
python -m pip install --no-binary=gvec --no-cache-dir --no-deps --force-reinstall 'gvec==1.5.0'

# Compiled kernels depend only on struphy's own source, not on which example this
# shard runs -- cache them under their own key (shared across all 10 shards) so that
# when only an example script changes, the 1-2 min recompile happens once instead of
# once per shard. Keyed independently of cache-example above; checked out to the
# right commit by the (fast, still-unconditional) install step regardless of hit/miss.
# Cache kernels in the installed package, using the resolved release and
# Python/compiler versions so an upgraded wheel gets a fresh compilation.
- name: Cache compiled Struphy kernels
id: cache-kernels
if: steps.cache-example.outputs.cache-hit != 'true'
uses: actions/cache@v4
with:
path: |
/struphy_fortran_/src/struphy/**/__pyccel__
/struphy_fortran_/src/struphy/**/*.so
key: struphy-kernels-${{ hashFiles('submodules/plasma-plots/struphy/src/struphy/**/*.py') }}
${{ steps.install-python.outputs.struphy-path }}/**/__pyccel__
${{ steps.install-python.outputs.struphy-path }}/**/*.so
key: struphy-kernels-pypi-${{ runner.os }}-${{ runner.arch }}-${{ steps.install-python.outputs.struphy-version }}-${{ steps.install-python.outputs.python-abi }}-${{ steps.install-python.outputs.pyccel-version }}-${{ hashFiles('requirements*.txt') }}

# Chrome doesn't depend on the struphy commit or example scripts, so cache it under
# Chrome doesn't depend on the Struphy release or example scripts, so cache it under
# its own static key instead of the cache-example key above -- otherwise every
# cache-miss shard redundantly re-downloads ~150-200MB from Google plus its apt
# dependencies for no reason.
Expand Down
6 changes: 6 additions & 0 deletions .gitignore
Original file line number Diff line number Diff line change
Expand Up @@ -36,3 +36,9 @@ docs/public/examples/struphy.log.*
.idea/
.vscode/
.DS_Store

# Cluster runs of precomputed examples (scripts/submit_precomputed_run.py)
/struphy-precomputed-*.sbatch
/slurm-*.out
/slurm-*.err
/*-run.tar.gz
3 changes: 0 additions & 3 deletions .gitmodules

This file was deleted.

Loading
Loading