diff --git a/.github/workflows/asv-benchmarking-pr.yml b/.github/workflows/asv-benchmarking-pr.yml index 2d143f365..b7968d7b6 100644 --- a/.github/workflows/asv-benchmarking-pr.yml +++ b/.github/workflows/asv-benchmarking-pr.yml @@ -12,67 +12,241 @@ on: pull_request: types: [opened, reopened, synchronize, labeled] workflow_dispatch: + inputs: + base_ref: + description: >- + Branch, tag or commit to compare against, via its merge-base with the + dispatched ref (so name the branch a topic branch will merge into). + required: false + default: main + shards: + description: >- + Runners to split the suite over. Each pays a fixed ~1 min (env + restore, install, numba warmup), so past about four gains little. + required: false + default: "4" env: PR_HEAD_LABEL: ${{ github.event.pull_request.head.label }} + ASV_DIR: "./benchmarks" + CONDA_ENV_FILE: ci/environment.yml + # Shared by every shard so their results merge under one machine. + ASV_MACHINE: gh-linux-x64 + SHARDS: ${{ github.event.inputs.shards || '4' }} jobs: - benchmark: + setup: + name: Setup if: ${{ contains(github.event.pull_request.labels.*.name, 'run-benchmark') && github.event_name == 'pull_request' || github.event_name == 'workflow_dispatch' }} - name: Linux runs-on: ubuntu-latest - env: - ASV_DIR: "./benchmarks" - CONDA_ENV_FILE: ci/environment.yml - + outputs: + base: ${{ steps.base.outputs.sha }} + shards: ${{ steps.plan.outputs.shards }} steps: - - uses: actions/checkout@v7 + - &checkout + uses: actions/checkout@v7 with: fetch-depth: 0 - - name: Record CPU topology + # A diagnostic; never fail the job over it. run: | - # A diagnostic; never fail the job over it. lscpu | grep -E 'Model name|^CPU\(s\):|Thread\(s\) per core|Core\(s\) per socket|Socket\(s\)|CPU max MHz' || lscpu || true - - name: Set up Conda environment + - &conda + name: Set up Conda environment uses: mamba-org/setup-micromamba@v3 with: environment-file: ${{env.CONDA_ENV_FILE}} cache-environment: true environment-name: uxarray_build - cache-environment-key: "${{runner.os}}-${{runner.arch}}-py${{env.PYTHON_VERSION}}-${{env.TODAY}}-${{hashFiles(env.CONDA_ENV_FILE)}}-benchmark" + cache-environment-key: "${{runner.os}}-${{runner.arch}}-${{hashFiles(env.CONDA_ENV_FILE)}}-benchmark" create-args: >- asv python-build mamba + - name: Resolve the baseline commit + id: base + # pull_request.* expressions are empty on workflow_dispatch. + env: + BASE: ${{ github.event.pull_request.base.sha }} + BASE_REF: ${{ github.event.inputs.base_ref }} + run: | + set -ex + if [ -z "$BASE" ]; then + BASE=$(git merge-base HEAD "origin/$BASE_REF" 2>/dev/null || git merge-base HEAD "$BASE_REF") + fi + # Dispatched from the base itself: compare against the commit before. + [ "$BASE" != "$GITHUB_SHA" ] || BASE=$(git rev-parse HEAD^) + echo "sha=$BASE" >> "$GITHUB_OUTPUT" + # asv rebuilds its own env under benchmarks/env/ each run; caching skips the # solve. No restore-keys: asv reuses a restored env without checking it. - - name: Cache asv environment + - &asv-env + name: Cache asv's benchmark environment uses: actions/cache@v6 with: path: ${{ env.ASV_DIR }}/env key: "asv-env-${{runner.os}}-${{runner.arch}}-${{hashFiles(env.CONDA_ENV_FILE, 'benchmarks/asv.conf.json')}}" - - name: Run Benchmarks + - &fixtures + name: Cache the benchmark fixtures + uses: actions/cache@v6 + with: + path: | + ${{ env.ASV_DIR }}/oQU*.nc + ${{ env.ASV_DIR }}/_io_cache + key: asv-fixtures-${{ runner.os }}-${{ hashFiles('benchmarks/helpers/_fixtures.py') }} + restore-keys: | + asv-fixtures-${{ runner.os }}- + + # The env cache key has no commit in it, so on a hit nothing is saved and the + # wheels setup builds would be lost; cache them (a few MB) for the shards. + - &wheels + name: Cache the built wheels + uses: actions/cache@v6 + with: + path: ${{ env.ASV_DIR }}/env/*/asv-build-cache + key: asv-wheels-${{ runner.os }}-${{ github.run_id }} + + # _partition balances shards by asv's recorded durations; with none it + # splits by count. Restore-only: the merge job saves the updated tree. A + # PR's first run falls back to main's, saved by asv-benchmarking.yml. + - name: Restore recorded durations + uses: actions/cache/restore@v6 + with: + path: ${{ env.ASV_DIR }}/results + key: asv-results-${{ runner.os }}-${{ github.run_id }} + restore-keys: | + asv-results-${{ runner.os }}- + + - name: Pre-build, discover and plan + id: plan shell: bash -l {0} - id: benchmark + working-directory: ${{ env.ASV_DIR }} + env: + BASE: ${{ steps.base.outputs.sha }} + # --bench just-discover builds the env and the commit's wheel and writes + # results/benchmarks.json without running anything. Done once here so a + # cold cache costs one conda solve, not one per shard. run: | - set -x + set -ex # Fill the fixture cache before asv preimports the suite, which would otherwise build it serially in the forkserver parent (cd .. && python -m benchmarks.helpers._fixtures) - # ID this runner - asv machine --yes - echo "Baseline: ${{ github.event.pull_request.base.sha }} (${{ github.event.pull_request.base.label }})" - echo "Contender: ${GITHUB_SHA} ($PR_HEAD_LABEL)" - # Run benchmarks for current commit against base - ASV_OPTIONS="--split --show-stderr" - asv continuous $ASV_OPTIONS ${{ github.event.pull_request.base.sha }} ${GITHUB_SHA} - # Save compare results - asv compare --split ${{ github.event.pull_request.base.sha }} ${GITHUB_SHA} > asv_compare_results.txt + (cd .. && python -m benchmarks.helpers._machine) + asv run --bench just-discover "${BASE}^!" + asv run --bench just-discover "${GITHUB_SHA}^!" + # Logs the split so a lopsided one is visible without opening each shard. + PYTHONPATH=.. python -m benchmarks.helpers._partition --shards "$SHARDS" + echo "shards=[$(seq -s, 0 $((SHARDS - 1)))]" >> "$GITHUB_OUTPUT" + + # The whole tree, not just benchmarks.json, so shards weigh the same durations. + - name: Upload the discovered suite + uses: actions/upload-artifact@v7 + with: + name: asv-plan + path: ${{ env.ASV_DIR }}/results + + benchmark: + name: Shard ${{ matrix.shard }} + needs: setup + runs-on: ubuntu-latest + strategy: + # A failed shard still leaves the rest worth merging. + fail-fast: false + matrix: + shard: ${{ fromJSON(needs.setup.outputs.shards) }} + steps: + - *checkout + - *conda + - *asv-env + - *fixtures + - *wheels + + - name: Download the discovered suite + uses: actions/download-artifact@v8 + with: + name: asv-plan + path: plan + + - name: Run shard + shell: bash -l {0} + working-directory: ${{ env.ASV_DIR }} + env: + BASE: ${{ needs.setup.outputs.base }} + SHARD: ${{ matrix.shard }} + # Partition from setup's plan, not this runner's cache, so all shards split the + # same suite. Nothing checks they tile it; a mismatch silently skips benchmarks. + run: | + set -ex + (cd .. && python -m benchmarks.helpers._fixtures) + (cd .. && python -m benchmarks.helpers._machine) + ASV_ARGS=$(PYTHONPATH=.. python -m benchmarks.helpers._partition \ + --shards "$SHARDS" --shard "$SHARD" --results ../plan --config asv.conf.json) + echo "Baseline: $BASE" + echo "Contender: ${GITHUB_SHA} (${PR_HEAD_LABEL:-$GITHUB_REF_NAME})" + # asv run, not continuous: continuous exits 1 on a regression, same as a + # broken run. The merge job does the comparison. + printf '%s\n' "${GITHUB_SHA}" "$BASE" > "$RUNNER_TEMP/commits.txt" + # Exit 2 means a benchmark failed, not the run; keep the shard's other + # results and warn rather than fail. + status=0 + asv run --show-stderr -m "$ASV_MACHINE" $ASV_ARGS \ + "HASHFILE:$RUNNER_TEMP/commits.txt" || status=$? + if [ "$status" -eq 2 ]; then + echo "::warning title=Benchmark failures in shard ${SHARD}::asv exited 2; see the failed entries above" + elif [ "$status" -ne 0 ]; then + exit "$status" + fi + + - name: Upload shard results + if: always() + uses: actions/upload-artifact@v7 + with: + name: asv-shard-${{ matrix.shard }} + path: ${{ env.ASV_DIR }}/results.shard${{ matrix.shard }} + if-no-files-found: warn + + merge: + name: Merge and compare + needs: [setup, benchmark] + # Run even if a shard failed: the surviving shards' results are still worth merging. + if: ${{ always() && needs.setup.result == 'success' }} + runs-on: ubuntu-latest + steps: + - *checkout + - *conda + + # No merge-multiple: the shards' results files share names. + - name: Download the shards + uses: actions/download-artifact@v8 + with: + pattern: asv-shard-* + path: shards + + - name: Merge and compare + shell: bash -l {0} working-directory: ${{ env.ASV_DIR }} + env: + BASE: ${{ needs.setup.outputs.base }} + run: | + set -ex + (cd .. && python -m benchmarks.helpers._machine) + (cd .. && python -m benchmarks.helpers._merge --out benchmarks/results shards/asv-shard-*) + asv compare --split --machine "$ASV_MACHINE" "$BASE" "${GITHUB_SHA}" \ + > asv_compare_results.txt + cat asv_compare_results.txt + # Where the time went, and how the next run will split it. + PYTHONPATH=.. python -m benchmarks.helpers._partition --shards "$SHARDS" || true + + # Keyed per run so setup's prefix restore picks up the newest. + - name: Save recorded durations for the next run + if: always() + uses: actions/cache/save@v6 + with: + path: ${{ env.ASV_DIR }}/results + key: asv-results-${{ runner.os }}-${{ github.run_id }} - name: Save PR number if: always() @@ -81,7 +255,7 @@ jobs: - uses: actions/upload-artifact@v7 if: always() with: - name: asv-benchmark-results-${{ runner.os }} + name: asv-benchmark-results-Linux path: | ${{ env.ASV_DIR }}/results ${{ env.ASV_DIR }}/asv_compare_results.txt diff --git a/.github/workflows/asv-benchmarking.yml b/.github/workflows/asv-benchmarking.yml index f5f751cf0..aaf367ae6 100644 --- a/.github/workflows/asv-benchmarking.yml +++ b/.github/workflows/asv-benchmarking.yml @@ -9,6 +9,8 @@ on: jobs: benchmark: runs-on: ubuntu-latest + # Otherwise only the 6h platform cap stops a runaway range build. + timeout-minutes: 180 defaults: run: shell: bash -el {0} @@ -73,6 +75,24 @@ jobs: cp -r uxarray-asv/results benchmarks/ fi + # asv builds benchmarks/env from ci/environment.yml (per asv.conf.json), not + # ci/asv.yml. No restore-keys: asv reuses a restored env without checking it. + - name: Cache asv's benchmark environment + uses: actions/cache@v6 + with: + path: ${{ env.ASV_DIR }}/env + key: "asv-env-${{runner.os}}-${{runner.arch}}-${{hashFiles('ci/environment.yml', 'benchmarks/asv.conf.json')}}" + + - name: Cache the benchmark fixtures + uses: actions/cache@v6 + with: + path: | + ${{ env.ASV_DIR }}/oQU*.nc + ${{ env.ASV_DIR }}/_io_cache + key: asv-fixtures-${{ runner.os }}-${{ hashFiles('benchmarks/helpers/_fixtures.py') }} + restore-keys: | + asv-fixtures-${{ runner.os }}- + - name: Run benchmarks shell: bash -l {0} id: benchmark @@ -81,7 +101,9 @@ jobs: python -m benchmarks.helpers._fixtures cd benchmarks asv machine --machine GH-Actions --os ubuntu-latest --arch x64 --cpu "2-core unknown" --ram 7GB - asv run v2024.02.0..main --skip-existing --parallel || true + # --skip-existing-commits, not --skip-existing: the latter still builds + # every commit in the range before skipping benchmarks it already has. + asv run main^! --skip-existing-commits --parallel || true - name: Commit and push benchmark results run: | @@ -104,3 +126,28 @@ jobs: force: true repository: UXARRAY/uxarray-asv directory: uxarray-asv + + # The PR workflow plans its shards from asv's recorded durations, and a + # PR's own caches are invisible to other PRs, so without this every PR's + # first run splits by count. Only this commit's results: _partition + # averages every file it finds, and the history has since-redefined + # benchmarks. After the push, so the pruning never reaches uxarray-asv. + - name: Keep only this commit's durations + id: durations + working-directory: ${{ env.ASV_DIR }} + run: | + sha=$(git rev-parse HEAD) + find results -name '*.json' ! -name benchmarks.json ! -name machine.json \ + ! -name "${sha:0:8}-*" -delete + # A failed run leaves nothing; don't shadow an older cache with that. + if compgen -G "results/*/${sha:0:8}-*.json" > /dev/null; then + echo "found=true" >> "$GITHUB_OUTPUT" + fi + + # Same path and key prefix the PR workflow restores. + - name: Save recorded durations for PR runs + if: steps.durations.outputs.found == 'true' + uses: actions/cache/save@v6 + with: + path: ${{ env.ASV_DIR }}/results + key: asv-results-${{ runner.os }}-${{ github.run_id }} diff --git a/.gitignore b/.gitignore index 5c52cd586..54c92b40f 100644 --- a/.gitignore +++ b/.gitignore @@ -164,3 +164,6 @@ benchmarks/env benchmarks/results benchmarks/html benchmarks/_io_cache +# Generated per shard by benchmarks/helpers/_partition.py --shard +benchmarks/asv.conf*.shard*.json +benchmarks/results.shard* diff --git a/benchmarks/asv.conf.json b/benchmarks/asv.conf.json index da57d9917..18133e54e 100644 --- a/benchmarks/asv.conf.json +++ b/benchmarks/asv.conf.json @@ -128,12 +128,18 @@ // Belt to that brace: if TBB is ever unavailable, pick the other // fork-safe layer rather. ``forksafe`` raises if no fork-safe layer // exists at all. - "env_nobuild": {"NUMBA_THREADING_LAYER": ["forksafe"]} + // BLAS is pinned so numpy does not compete with numba for cores. + "env_nobuild": { + "NUMBA_THREADING_LAYER": ["forksafe"], + "OMP_NUM_THREADS": ["1"], + "MKL_NUM_THREADS": ["1"], + "OPENBLAS_NUM_THREADS": ["1"] + } }, + // No ``python -m build``: its output went to {build_dir}/dist, unused. "build_command": [ - "python -m build", "python -mpip wheel --no-deps --no-build-isolation --no-index -w {build_cache_dir} {build_dir}" ], diff --git a/benchmarks/helpers/_machine.py b/benchmarks/helpers/_machine.py new file mode 100644 index 000000000..85a707c22 --- /dev/null +++ b/benchmarks/helpers/_machine.py @@ -0,0 +1,38 @@ +"""Records this host in asv's machine file under a stable name, and prints it. + +asv names the machine after the hostname, which changes every hosted-runner +job and every cluster node, so results cannot be compared across runs or +merged across shards (``results//...``). ``asv machine --machine +NAME`` alone drops the detected ``cpu``/``num_cpu``/``ram``, hence this. + +The name is ``$ASV_MACHINE``, then ``$NCAR_HOST`` (the cluster, not the node), +then the node name minus trailing digits. Login and compute nodes do not share +a stem, so set ``ASV_MACHINE`` if you use both without ``NCAR_HOST``. + +Usage:: + + python -m benchmarks.helpers._machine # record, print the name + python -m benchmarks.helpers._machine --print # just print it +""" + +import os +import platform +import re +import sys + + +def default_name(): + for variable in ("ASV_MACHINE", "NCAR_HOST"): + if os.environ.get(variable): + return os.environ[variable] + node = platform.node().split(".")[0] + return re.sub(r"[-_]?\d+$", "", node) or node + + +if __name__ == "__main__": + name = default_name() + if "--print" not in sys.argv[1:]: + from asv.machine import Machine, MachineCollection + + MachineCollection.save(name, {**Machine.get_defaults(), "machine": name}) + print(name) diff --git a/benchmarks/helpers/_merge.py b/benchmarks/helpers/_merge.py new file mode 100644 index 000000000..7f2a11f30 --- /dev/null +++ b/benchmarks/helpers/_merge.py @@ -0,0 +1,71 @@ +"""Merges a sharded run's results directories back into one tree. + +asv rewrites the whole ``/-.json`` at the end of a run, +so each shard writes its own directory and they are combined here. Shards +split by whole benchmark, so every row belongs to one shard and the merge is a +union. + +Usage:: + + python -m benchmarks.helpers._merge --out results results.shard* +""" + +import argparse +import json +import sys +from pathlib import Path + +BENCHMARK_DIR = Path(__file__).resolve().parents[1] + +# Every shard's benchmarks.json lists the whole suite (asv saves all it +# discovers, not just what ``--bench`` selects), and machine.json is pinned to +# one name, so any shard's copy will do. +_WHOLE_FILES = frozenset({"benchmarks.json", "machine.json"}) + + +def merge(shard_dirs, out_dir): + """Merges ``shard_dirs`` into ``out_dir``; returns ``{relative path: data}``.""" + merged = {} + for shard_dir in map(Path, shard_dirs): + for path in sorted(shard_dir.rglob("*.json")): + rel = path.relative_to(shard_dir) + data = json.loads(path.read_text()) + if rel not in merged or path.name in _WHOLE_FILES: + merged[rel] = data + continue + into = merged[rel] + into["results"].update(data["results"]) + # ```` and ````, which every shard pays; the + # max is what a single run would report. + for key, value in data.get("durations", {}).items(): + into["durations"][key] = max(into["durations"].get(key, 0.0), value) + for rel, data in merged.items(): + target = Path(out_dir) / rel + target.parent.mkdir(parents=True, exist_ok=True) + target.write_text(json.dumps(data)) + return merged + + +def main(argv=None): + parser = argparse.ArgumentParser( + prog="python -m benchmarks.helpers._merge", + description="Merge a sharded run's results directories into one tree.", + ) + parser.add_argument("shards", nargs="+", help="Shard results directories to merge.") + parser.add_argument( + "--out", + default=str(BENCHMARK_DIR / "results"), + help="Directory to write the merged tree to (default benchmarks/results).", + ) + args = parser.parse_args(argv) + merged = merge(args.shards, args.out) + if not merged: + parser.error(f"no results under {', '.join(args.shards)}") + for rel, data in sorted(merged.items()): + if "results" in data: + print(f"{rel}: {len(data['results'])} benchmarks") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/benchmarks/helpers/_partition.py b/benchmarks/helpers/_partition.py new file mode 100644 index 000000000..99f4e81ad --- /dev/null +++ b/benchmarks/helpers/_partition.py @@ -0,0 +1,176 @@ +"""Splits the suite into shards of roughly equal cost, one per runner. + +asv's ``--parallel`` only parallelizes environment builds, and ``time_*`` +results need an uncontended machine, so each shard gets its own runner. + +Shards split by whole class: a class's benchmarks share the kernels its first +one compiles, so splitting it makes every shard recompile (``FaceBounds``: 59s +in one shard, 221s across four). Numba's ``cache=True`` does not help, as asv +reinstalls the wheel per commit. Every ``setup_cache`` but the suite-wide one in +``_fixtures`` (cheap to repeat) lives on a single class, so stays in one shard. + +Each shard writes to its own ``results_dir``, since asv rewrites the whole +per-commit results file at the end of a run. Weights are the ``duration`` asv +records per benchmark; benchmarks without one get the median. + +Usage:: + + python -m benchmarks.helpers._partition --shards 4 + asv run $(python -m benchmarks.helpers._partition --shards 4 --shard 0) + + # Thread sweep over the whole suite (add ``--bench`` to narrow it). + for n in 1 2 4 8; do + asv run $(python -m benchmarks.helpers._partition --shards 1 --shard 0 \ + --env NUMBA_NUM_THREADS=$n) + done +""" + +import argparse +import json +import re +import statistics +import sys +from pathlib import Path + +BENCHMARK_DIR = Path(__file__).resolve().parents[1] + + +def load_benchmarks(results_dir): + """The discovered benchmarks, as asv wrote them to ``benchmarks.json``.""" + discovered = json.loads((Path(results_dir) / "benchmarks.json").read_text()) + # asv stores its own format version alongside the benchmarks. + discovered.pop("version", None) + return discovered + + +def load_weights(results_dirs): + """Mean recorded duration per benchmark, in seconds. + + Mean, not latest: a cold numba cache can make one run hundreds of times + slower than warm, and a single outlier would skew the plan. + """ + samples = {} + for results_dir in results_dirs: + for path in sorted(Path(results_dir).glob("*/*.json")): + data = json.loads(path.read_text()) + columns = data.get("result_columns") or [] + if "duration" not in columns: + continue + index = columns.index("duration") + for name, row in data["results"].items(): + if len(row) > index and row[index] is not None: + samples.setdefault(name, []).append(float(row[index])) + return {name: statistics.fmean(values) for name, values in samples.items()} + + +def costs(benchmarks, weights): + """Each benchmark's weight, the median for those without one.""" + known = [value for value in weights.values() if value > 0] + default = statistics.median(known) if known else 1.0 + return {name: weights.get(name, default) for name in benchmarks} + + +def plan(benchmarks, n_shards, weights=None): + """Partitions ``benchmarks`` into ``n_shards`` lists of names. + + Greedy longest-first (LPT): within about 1% of optimal on this suite and + easy to debug. Ties break by name, so every shard computes the same plan. + """ + cost = costs(benchmarks, weights or {}) + classes = {} + for name in sorted(benchmarks): + # The class -- or the module, for a bare function. + classes.setdefault(name.rsplit(".", 1)[0], []).append(name) + class_cost = {owner: sum(cost[name] for name in names) for owner, names in classes.items()} + + shards = [[] for _ in range(n_shards)] + loads = [0.0] * n_shards + for owner in sorted(classes, key=lambda owner: (-class_cost[owner], owner)): + target = loads.index(min(loads)) + shards[target].extend(classes[owner]) + loads[target] += class_cost[owner] + return shards + + +def write_shard_config(base_config, shard, env): + """Copies ``base_config`` beside itself with the shard's ``results_dir``. + + ``env`` overrides ``env_nobuild`` variables, which asv folds into the + environment name, so a sweep's runs do not overwrite one another. + """ + # asv configs are JSON with JS comments; lazy so the module runs without asv. + from asv import util + + base = Path(base_config) + config = util.load_json(str(base), js_comments=True) + config["results_dir"] = f"{config.get('results_dir', 'results')}.shard{shard}" + # Several values would each be an environment running the whole suite. + config["matrix"]["env_nobuild"].update({key: [value] for key, value in env.items()}) + out = base.with_name(f"{base.stem}.shard{shard}{base.suffix}") + out.write_text(json.dumps(config, indent=4)) + return out + + +def _report(benchmarks, shards, weights): + cost = costs(benchmarks, weights) + total = sum(cost.values()) + loads = [sum(cost[name] for name in shard) for shard in shards] + ideal = total / len(shards) + print(f"{len(benchmarks)} benchmarks, {len(weights)} with recorded durations, " + f"{total / 60:.1f} min of work") + for index, (shard, load) in enumerate(zip(shards, loads)): + print(f" shard {index}: {len(shard):3} benchmarks {load / 60:5.1f} min " + f"{100 * (load - ideal) / ideal:+5.1f}%") + print(f" speedup {total / max(loads):.2f}x of a possible {len(shards)}x") + for name in sorted(cost, key=cost.get, reverse=True)[:15]: + print(f" {cost[name]:8.1f}s {name}") + + +def main(argv=None): + parser = argparse.ArgumentParser( + prog="python -m benchmarks.helpers._partition", + description="Split the benchmark suite into shards of roughly equal cost.", + ) + parser.add_argument("--shards", type=int, default=4, help="Number of shards (default 4).") + parser.add_argument( + "--shard", type=int, help="Write this shard's config and print its asv run arguments." + ) + parser.add_argument( + "--results", + action="append", + help="Results directory to read (repeatable; default benchmarks/results).", + ) + parser.add_argument( + "--config", + default=str(BENCHMARK_DIR / "asv.conf.json"), + help="Base asv config for --shard (default benchmarks/asv.conf.json).", + ) + parser.add_argument( + "--env", + action="append", + default=[], + metavar="KEY=VALUE", + help="Override an env_nobuild variable, e.g. NUMBA_NUM_THREADS=4 (repeatable).", + ) + args = parser.parse_args(argv) + + results_dirs = args.results or [str(BENCHMARK_DIR / "results")] + benchmarks = load_benchmarks(results_dirs[0]) + weights = load_weights(results_dirs) + shards = plan(benchmarks, args.shards, weights) + if args.shard is None: + _report(benchmarks, shards, weights) + return 0 + + env = dict(entry.split("=", 1) for entry in args.env) + # ``--bench`` alongside ``--config``, so a shard can't pair its benchmarks + # with another shard's results dir. asv matches parameterized benchmarks as + # ``name(p0, p1)``, hence ``($|\()`` rather than ``$``. + print("--config", write_shard_config(args.config, args.shard, env)) + for name in shards[args.shard]: + print("--bench", f"^{re.escape(name)}($|\\()") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/benchmarks/mpas_ocean.py b/benchmarks/mpas_ocean.py index 7364451a6..8d8687e64 100644 --- a/benchmarks/mpas_ocean.py +++ b/benchmarks/mpas_ocean.py @@ -386,17 +386,37 @@ def track_peakmem_const_lat(self, resolution, lat_step): track_peakmem_const_lat.unit = "bytes" +# Great-circle radii in degrees. Mean element spacing is roughly 4.3 degrees at +# 480km and 1.1 at 120km, so 5.0 is a few neighbors per face on the coarse mesh +# and dozens on the fine one. +NEIGHBORHOOD_RADIUS = 5.0 + +# asv re-runs ``setup`` before every sample, about six times per benchmark +# process, and the neighborhood setups (fixture, warmup reduction, radius +# query) cost far more than the calls they time: ``NeighborhoodReduce. +# time_reduce`` timed a 5ms kernel at 120km but took 27s, almost all setup. The +# timed calls only read what setup builds, so it is built once per process. +_prepared = {} + + +def _once_per_process(key, build): + if key not in _prepared: + _prepared[key] = build() + return _prepared[key] + + class NeighborhoodBuild(DatasetBenchmark): """Construction cost of a ``Neighborhood``, split into its three stages. ``Neighborhood`` claims the neighbor query costs more than any reduction run - on it. ``r`` is a great-circle radius in degrees, against - a mean element spacing of roughly 4.3 degrees at 480km and 1.1 at 120km, so - the smallest radius here is near self-only on the coarser mesh. + on it. The timings skip r=15 (over a second per call at 120km); the + ``track_*`` benchmarks skip r=1, where they barely move. """ param_names = DatasetBenchmark.param_names + ['r'] - params = DatasetBenchmark.params + [[1.0, 5.0, 15.0]] + params = DatasetBenchmark.params + [[1.0, NEIGHBORHOOD_RADIUS]] + # asv reads ``params`` from the method before the class. + _track_params = DatasetBenchmark.params + [[NEIGHBORHOOD_RADIUS, 15.0]] def setup(self, resolution, r): super().setup(resolution) @@ -420,6 +440,7 @@ def track_nbytes_neighbors(self, resolution, r): nb = Neighborhood(self.uxgrid, r=r, on="face centers") return nb._flat.nbytes + nb._starts.nbytes + nb._counts.nbytes + track_nbytes_neighbors.params = _track_params track_nbytes_neighbors.unit = "bytes" def track_peakmem_build(self, resolution, r): @@ -427,6 +448,7 @@ def track_peakmem_build(self, resolution, r): return peak_allocated( lambda: Neighborhood(self.uxgrid, r=r, on="face centers")) + track_peakmem_build.params = _track_params track_peakmem_build.unit = "bytes" def track_mean_neighbors(self, resolution, r): @@ -434,6 +456,7 @@ def track_mean_neighbors(self, resolution, r): nb = Neighborhood(self.uxgrid, r=r, on="face centers") return round(float(nb.n_neighbors.mean()), 2) + track_mean_neighbors.params = _track_params track_mean_neighbors.unit = "elements" @@ -453,7 +476,7 @@ class NeighborhoodReduce(DatasetBenchmark): param_names = DatasetBenchmark.param_names + ['reduction'] params = DatasetBenchmark.params + [['mean', 'median']] - radius = 15.0 + radius = NEIGHBORHOOD_RADIUS @staticmethod def _run(neighborhood, reduction): @@ -465,6 +488,17 @@ def _run(neighborhood, reduction): return getattr(neighborhood, reduction)() def setup(self, resolution, reduction): + self.uxds, self.nb = _once_per_process( + ('NeighborhoodReduce', resolution, reduction), + lambda: self._prepare(resolution, reduction)) + # The grid caches one ball tree, and ``time_dataset_reduce`` leaves it + # on edges. Put back the face tree a fresh setup leaves, so every sample + # rebuilds the same trees. + self.uxds.uxgrid.get_ball_tree(coordinates="face centers", + coordinate_system="spherical", + distance_metric="haversine") + + def _prepare(self, resolution, reduction): super().setup(resolution) uxgrid = self.uxds.uxgrid @@ -488,7 +522,7 @@ def setup(self, resolution, reduction): # locations now so the first timed call is not the one that pays. _, _, _ = uxgrid.node_lon, uxgrid.edge_lon, uxgrid.face_lon - self.nb = self.uxds[data_var].neighborhood(r=self.radius) + return self.uxds, self.uxds[data_var].neighborhood(r=self.radius) def time_reduce(self, resolution, reduction): """The kernel alone: the query was paid for in ``setup``.""" @@ -521,9 +555,14 @@ class NeighborhoodDask(DatasetBenchmark): params = DatasetBenchmark.params + [['numpy', 'time_chunks', 'grid_chunks']] n_time = 12 - radius = 5.0 + radius = NEIGHBORHOOD_RADIUS def setup(self, resolution, chunking): + self.uxds, self.nb = _once_per_process( + ('NeighborhoodDask', resolution, chunking), + lambda: self._prepare(resolution, chunking)) + + def _prepare(self, resolution, chunking): super().setup(resolution) grid, data = file_path_dict[self.params[0][0]] _ = ux.open_dataset(grid, data)[data_var].neighborhood(r=1.0).mean() @@ -540,15 +579,16 @@ def setup(self, resolution, chunking): # Built here, so these measure the reduction and the graph it runs # through rather than the query. - self.nb = uxda.neighborhood(r=self.radius) + nb = uxda.neighborhood(r=self.radius) # One reduction here too, to warm the dask graph path -- and, for # 'grid_chunks', to let the rechunk warning through exactly once... - _ = self.nb.mean().compute() + _ = nb.mean().compute() # ...then silence the repeats. warnings.filterwarnings('ignore', category=UserWarning, message='Rechunking') + return self.uxds, nb def time_mean(self, resolution, chunking): _ = self.nb.mean().compute()