GitHub Viewer
#!/usr/bin/env python3
"""Run the upstream pyperformance benchmark suite (https://github.com/python/pyperformance)
against a Python executable -- RustPython, a real CPython, or both -- and
catalog which benchmarks pass, fail, or time out under each.
Background
----------
pyperformance cannot be installed as-is on RustPython: its runtime dependency
`pyperf` hard-depends on `psutil`, a C-extension package. RustPython has no
CPython C-API / extension-module loading support (no `_imp.create_dynamic` /
`_imp.exec_dynamic`, no compiler config vars such as LDCXXSHARED in
`_sysconfigdata`), so building or loading any C extension fails.
Workaround: `pyperf` itself already disables psutil usage on interpreters
that report `Py_GIL_DISABLED=1` (see `pyperf._utils.USE_PSUTIL`), which is
exactly what RustPython reports (it has no GIL). So a functionality-free,
pure-Python "psutil" stub (see stub_psutil/) is enough to satisfy pip's
dependency resolution -- pyperf never actually calls into it at runtime on
RustPython. This unblocks any *pure-Python* pyperformance benchmark; any
benchmark whose own workload (not just pyperf) requires a real C extension
(e.g. lxml, numpy, greenlet) will still fail, and that failure is a genuine
finding: it marks a real C-extension gap in RustPython, not a tooling
artifact of this script. A real CPython target doesn't need this workaround
at all -- pass --no-psutil-stub for it, so it installs and uses the genuine
psutil.
This script:
1. Ensures a *host* CPython venv with pyperformance installed (pyperformance
itself only runs under a real CPython; the --python target is just the
interpreter it benchmarks, which may be that same CPython or RustPython).
2. Builds the pure-Python psutil stub wheel once (skipped with --no-psutil-stub).
3. Runs every benchmark pyperformance knows about, one at a time, against
the given --python target, with a timeout per benchmark.
4. Writes a JSON + Markdown catalog of the results under results//.
Usage
-----
python3 scripts/pyperformance/run_all.py \\
--python target/release/rustpython --label rustpython
python3 scripts/pyperformance/run_all.py \\
--python "$(command -v python3.13)" --label cpython3.13 --no-psutil-stub
Then compare two labels' catalogs with compare.py (see its docstring).
Re-running resumes from results//catalog.json by default (skips
benchmarks already recorded); pass --force to redo everything.
"""
from __future__ import annotations
import argparse
import hashlib
import json
import os
import re
import shutil
import signal
import subprocess
from pathlib import Path
SCRIPT_DIR = Path(__file__).resolve().parent
REPO_ROOT = SCRIPT_DIR.parent.parent
STUB_PSUTIL_DIR = SCRIPT_DIR / "stub_psutil"
DEFAULT_CACHE_DIR = REPO_ROOT / "target" / "pyperformance"
DEFAULT_OUT_DIR = SCRIPT_DIR / "results"
DEFAULT_TIMEOUT = 180
def log(msg: str) -> None:
print(f"[pyperformance-runner] {msg}", flush=True)
def executable_fingerprint(path: Path) -> str:
"""Hash of the target executable's contents -- and, when RUSTPYTHONPATH
points it at a stdlib copy, that stdlib's contents too, since that's the
other half of what pyperf actually runs. Used to invalidate a resumed
catalog when a local rebuild (or a different target under the same
--label) replaces either one from under the earlier results.
"""
digest = hashlib.sha256()
digest.update(path.read_bytes())
rustpythonpath = os.environ.get("RUSTPYTHONPATH")
if rustpythonpath:
for file in sorted(Path(rustpythonpath).rglob("*")):
if file.is_file():
digest.update(str(file.relative_to(rustpythonpath)).encode())
digest.update(file.read_bytes())
return digest.hexdigest()
def find_host_python() -> str:
for candidate in (
"python3.13",
"python3.12",
"python3.11",
"python3.10",
"python3",
):
path = shutil.which(candidate)
if path:
return path
raise SystemExit("No usable host CPython found on PATH (need python3.10+)")
def ensure_host_venv(cache_dir: Path) -> Path:
venv_dir = cache_dir / "host-venv"
pip_marker = venv_dir / "pyvenv.cfg"
pyperf_installed = False
if pip_marker.exists():
pip_bin = venv_dir / "bin" / "pip"
result = subprocess.run(
[str(pip_bin), "show", "pyperformance"],
capture_output=True,
text=True,
)
pyperf_installed = result.returncode == 0
if not pyperf_installed:
log(
f"setting up host venv at {venv_dir} (this runs pyperformance's CLI; "
"RustPython is only the --python target it benchmarks)"
)
shutil.rmtree(venv_dir, ignore_errors=True)
venv_dir.parent.mkdir(parents=True, exist_ok=True)
host_python = find_host_python()
subprocess.run([host_python, "-m", "venv", str(venv_dir)], check=True)
pip_bin = venv_dir / "bin" / "pip"
subprocess.run([str(pip_bin), "install", "-q", "-U", "pip"], check=True)
subprocess.run([str(pip_bin), "install", "-q", "pyperformance"], check=True)
return venv_dir
def ensure_stub_psutil_wheel(host_venv: Path, cache_dir: Path) -> Path:
stub_pkgs_dir = cache_dir / "stub-pkgs"
existing = list(stub_pkgs_dir.glob("psutil-*.whl"))
if existing:
return stub_pkgs_dir
log(
"building pure-Python psutil stub wheel (see scripts/pyperformance/stub_psutil/)"
)
stub_pkgs_dir.mkdir(parents=True, exist_ok=True)
pip_bin = host_venv / "bin" / "pip"
subprocess.run(
[
str(pip_bin),
"wheel",
str(STUB_PSUTIL_DIR),
"-w",
str(stub_pkgs_dir),
"--no-deps",
"-q",
],
check=True,
)
return stub_pkgs_dir
def list_benchmarks(host_venv: Path) -> list[str]:
pyperformance_bin = host_venv / "bin" / "pyperformance"
result = subprocess.run(
[str(pyperformance_bin), "list"],
capture_output=True,
text=True,
check=True,
)
names = []
for line in result.stdout.splitlines():
line = line.strip()
if line.startswith("- "):
names.append(line[2:].strip())
return names
MEAN_RE = re.compile(r"Mean \+- std dev:\s*(.+)")
FAILED_BUILD_RE = re.compile(r"Failed to build (\S+)")
FAILED_WHEEL_RE = re.compile(r"Failed building wheel for (\S+)")
LDCXXSHARED_RE = re.compile(
r"Unexpected None in config vars: \[('LDCXXSHARED'[^\]]*)\]"
)
# The root-cause exception raised by the *benchmark script itself* (running under
# RustPython), as opposed to pyperformance's own wrapper exceptions
# (e.g. "RuntimeError: Benchmark died", "RuntimeError: ... failed with exit code ...")
# which are always the LAST exception in the combined output, not the first.
PYPERFORMANCE_WRAPPER_ERRORS = {
"Benchmark died",
}
EXCEPTION_LINE_RE = re.compile(
r"^(?:\w+\.)*(\w*(?:Error|Exception)): (.+)$", re.MULTILINE
)
def classify_failure(combined_output: str) -> str:
if LDCXXSHARED_RE.search(combined_output):
pkg_match = FAILED_WHEEL_RE.search(combined_output) or FAILED_BUILD_RE.search(
combined_output
)
pkg = pkg_match.group(1) if pkg_match else "unknown package"
return f"C-extension build blocked (no compiler config / no CPython C-API) building {pkg}"
pkg_match = FAILED_WHEEL_RE.search(combined_output) or FAILED_BUILD_RE.search(
combined_output
)
if pkg_match:
return f"failed to build/install {pkg_match.group(1)}"
# Prefer the first real exception in the log: it's raised by the benchmark
# script running under RustPython, before pyperformance's own wrapper
# exceptions (always last) obscure it with a generic "Benchmark died".
for exc_type, exc_msg in EXCEPTION_LINE_RE.findall(combined_output):
if exc_msg.strip() in PYPERFORMANCE_WRAPPER_ERRORS:
continue
if "failed with exit code" in exc_msg:
continue
return f"{exc_type}: {exc_msg.strip()}"[:300]
m = re.search(r"^(ERROR: .*)$", combined_output, re.MULTILINE)
if m:
return m.group(1)[:200]
tail = "\n".join(combined_output.strip().splitlines()[-5:])
return tail[:400] or "unknown failure"
def run_one_benchmark(
pyperformance_bin: Path,
target_python: Path,
bench: str,
work_dir: Path,
out_dir: Path,
stub_pkgs_dir: Path | None,
timeout: int,
extra_args: list[str],
) -> dict:
result_json = out_dir / "raw" / f"{bench}.json"
result_json.parent.mkdir(parents=True, exist_ok=True)
if result_json.exists():
result_json.unlink()
env = os.environ.copy()
cmd = [
str(pyperformance_bin),
"run",
"--python",
str(target_python),
"-b",
bench,
"-o",
str(result_json),
*extra_args,
]
inherited = []
if stub_pkgs_dir is not None:
# Work around the target interpreter's lack of a C-extension psutil (see
# module docstring). Real CPython doesn't need this -- pass
# stub_pkgs_dir=None for it so it installs and uses the real psutil.
env["PIP_FIND_LINKS"] = str(stub_pkgs_dir)
inherited.append("PIP_FIND_LINKS")
if "RUSTPYTHONPATH" in env:
# A RustPython binary copied away from its checkout (as CI does when it
# keeps one build of each commit around) can only find the stdlib
# through this variable, and pyperf re-executes the target interpreter
# in a venv of its own, so it has to survive that hop too.
inherited.append("RUSTPYTHONPATH")
if inherited:
# One comma-separated flag, not one flag per variable: pyperf's
# `--inherit-environ` is a plain (non-appending) option, so repeating it
# keeps only the last name.
cmd += ["--inherit-environ", ",".join(inherited)]
log(f"running {bench} ...")
# A new session makes this process the leader of its own process group, so
# a timeout can kill the whole group -- pyperf re-execs the target
# interpreter as a child, and `subprocess.run`'s own timeout handling only
# ever reaches the direct child, leaving that descendant free to keep
# burning CPU into the next benchmark's measurement.
proc = subprocess.Popen(
cmd,
cwd=work_dir,
env=env,
stdout=subprocess.PIPE,
stderr=subprocess.PIPE,
text=True,
start_new_session=True,
)
try:
stdout, stderr = proc.communicate(timeout=timeout)
except subprocess.TimeoutExpired:
os.killpg(proc.pid, signal.SIGKILL)
proc.communicate() # reap the process, discard its output
return {
"benchmark": bench,
"status": "timeout",
"detail": f"exceeded {timeout}s timeout",
"mean": None,
}
combined = (stdout or "") + "\n" + (stderr or "")
if proc.returncode == 0 and result_json.exists():
mean_match = MEAN_RE.search(combined)
return {
"benchmark": bench,
"status": "ok",
"detail": None,
"mean": mean_match.group(1).strip() if mean_match else None,
}
return {
"benchmark": bench,
"status": "fail",
"detail": classify_failure(combined),
"mean": None,
}
def write_catalog(results: list[dict], out_dir: Path, label: str) -> None:
out_dir.mkdir(parents=True, exist_ok=True)
catalog_json = out_dir / "catalog.json"
# Write-then-rename rather than truncate-in-place: this is called after
# every single benchmark, so an interruption mid-write must not leave
# catalog.json half-written -- that would both lose every earlier result
# in the file and make the next run's json.loads() fail outright.
tmp_path = out_dir / "catalog.json.tmp"
tmp_path.write_text(json.dumps(results, indent=2, sort_keys=False) + "\n")
tmp_path.replace(catalog_json)
ok = [r for r in results if r["status"] == "ok"]
fail = [r for r in results if r["status"] == "fail"]
timeout = [r for r in results if r["status"] == "timeout"]
lines = []
lines.append(f"# pyperformance on {label} -- benchmark catalog")
lines.append("")
lines.append(
f"Generated by `scripts/pyperformance/run_all.py --label {label}`. "
f"`{label}` was benchmarked as the `--python` target of upstream "
"`pyperformance`. See the script docstring for the psutil-stub workaround "
"used for interpreters without CPython C-API / C-extension support."
)
lines.append("")
lines.append(
f"**{len(ok)} passed, {len(fail)} failed, {len(timeout)} timed out** "
f"out of {len(results)} benchmarks."
)
lines.append("")
lines.append(f"| Benchmark | Status | Mean ({label}) | Notes |")
lines.append("|---|---|---|---|")
for r in sorted(results, key=lambda r: (r["status"] != "ok", r["benchmark"])):
status_label = {"ok": "✅ pass", "fail": "❌ fail", "timeout": "⏱ timeout"}[
r["status"]
]
mean = r["mean"] or ""
detail = (r["detail"] or "").replace("|", "\\|").replace("\n", " ")
lines.append(f"| {r['benchmark']} | {status_label} | {mean} | {detail} |")
lines.append("")
catalog_md = out_dir / "CATALOG.md"
catalog_md.write_text("\n".join(lines) + "\n")
log(f"wrote {catalog_json} and {catalog_md}")
def parse_args() -> argparse.Namespace:
parser = argparse.ArgumentParser(
description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter
)
parser.add_argument(
"--python",
"--rustpython",
dest="python",
type=Path,
default=REPO_ROOT / "target" / "release" / "rustpython",
help="path to the Python (or RustPython) executable to benchmark "
"(default: target/release/rustpython)",
)
parser.add_argument(
"--label",
type=str,
default=None,
help="name for this target, used for the results subdirectory and report "
"headings (default: the executable's basename, e.g. 'rustpython' or "
"'python3.13')",
)
parser.add_argument(
"--no-psutil-stub",
action="store_true",
help="don't inject the pure-Python psutil stub -- use this for a real "
"CPython target, which can install and use the genuine psutil",
)
parser.add_argument(
"--timeout",
type=int,
default=DEFAULT_TIMEOUT,
help="per-benchmark timeout in seconds (default: %(default)s)",
)
parser.add_argument(
"--out",
type=Path,
default=DEFAULT_OUT_DIR,
help="parent directory to write /catalog.json and /CATALOG.md into",
)
parser.add_argument(
"--cache-dir",
type=Path,
default=DEFAULT_CACHE_DIR,
help="directory for the host venv and stub wheel cache (default: target/pyperformance)",
)
parser.add_argument(
"--benchmarks",
type=str,
default=None,
help="comma-separated list of benchmarks to run (default: everything pyperformance knows)",
)
parser.add_argument(
"--fast",
action="store_true",
default=True,
help="pass --fast to pyperformance run (default: on, for a quicker catalog run)",
)
parser.add_argument(
"--rigorous",
action="store_true",
help="pass --rigorous to pyperformance run instead of --fast",
)
parser.add_argument(
"--force",
action="store_true",
help="re-run benchmarks even if already present in an existing catalog.json",
)
return parser.parse_args()
def resolve_target_python(python_arg: Path) -> Path:
target_python = shutil.which(str(python_arg)) or str(python_arg)
target_python = Path(target_python)
if not target_python.exists():
raise SystemExit(
f"Python executable not found: {python_arg} "
"(for RustPython, build it first, e.g. `cargo build --release --features ssl`)"
)
return target_python.resolve()
def load_resumable_catalog(
out_dir: Path, target_python: Path, label: str
) -> dict[str, dict]:
"""Load catalog.json to resume from, keyed by benchmark name -- unless
the target executable (or the stdlib RUSTPYTHONPATH points it at) has
changed since that catalog was written, in which case it is invalidated
(and every benchmark re-run) instead of trusted.
"""
out_dir.mkdir(parents=True, exist_ok=True)
fingerprint_path = out_dir / ".executable-fingerprint"
catalog_json = out_dir / "catalog.json"
current_fingerprint = executable_fingerprint(target_python)
stale = (
fingerprint_path.exists()
and fingerprint_path.read_text().strip() != current_fingerprint
)
all_results: dict[str, dict] = {}
if stale:
log(
f"target executable for label {label!r} changed since the last run; "
"ignoring the existing catalog and re-running all benchmarks"
)
# Persist the invalidated (empty) catalog before recording the new
# fingerprint below -- otherwise a crash between the two writes would
# leave a fingerprint that matches the *new* executable pointing at a
# catalog.json still full of results measured against the old one.
write_catalog([], out_dir, label)
elif catalog_json.exists():
for r in json.loads(catalog_json.read_text()):
all_results[r["benchmark"]] = r
fingerprint_path.write_text(current_fingerprint + "\n")
return all_results
def main() -> None:
args = parse_args()
target_python = resolve_target_python(args.python)
label = args.label or target_python.name
# Every benchmark subprocess runs with cwd=work_dir (below); a relative
# --out, --cache-dir, or RUSTPYTHONPATH would then resolve against that
# directory instead of the one this script was invoked from.
args.out = args.out.resolve()
args.cache_dir = args.cache_dir.resolve()
if os.environ.get("RUSTPYTHONPATH"):
os.environ["RUSTPYTHONPATH"] = str(Path(os.environ["RUSTPYTHONPATH"]).resolve())
out_dir = args.out / label
extra_args = ["--rigorous"] if args.rigorous else ["--fast"]
host_venv = ensure_host_venv(args.cache_dir)
stub_pkgs_dir = (
None
if args.no_psutil_stub
else ensure_stub_psutil_wheel(host_venv, args.cache_dir)
)
pyperformance_bin = host_venv / "bin" / "pyperformance"
benchmarks = (
[b.strip() for b in args.benchmarks.split(",") if b.strip()]
if args.benchmarks
else list_benchmarks(host_venv)
)
log(f"{len(benchmarks)} benchmarks to run against {label} ({target_python})")
work_dir = args.cache_dir / "work" / label
work_dir.mkdir(parents=True, exist_ok=True)
# Starts as *every* benchmark previously recorded (not just this
# invocation's --benchmarks subset), so a partial/targeted re-run never
# drops earlier results from catalog.json/CATALOG.md -- it only updates
# the entries it actually re-ran.
all_results = load_resumable_catalog(out_dir, target_python, label)
for bench in benchmarks:
if bench in all_results and not args.force:
log(f"skipping {bench} (already in catalog; use --force to redo)")
continue
r = run_one_benchmark(
pyperformance_bin,
target_python,
bench,
work_dir,
out_dir,
stub_pkgs_dir,
args.timeout,
extra_args,
)
log(
f" -> {bench}: {r['status']}"
+ (f" ({r['detail']})" if r["detail"] else "")
)
all_results[bench] = r
write_catalog(
list(all_results.values()), out_dir, label
) # incremental, so a crash keeps progress
write_catalog(list(all_results.values()), out_dir, label)
ran = [all_results[b] for b in benchmarks]
ok = sum(1 for r in ran if r["status"] == "ok")
log(
f"done: {ok}/{len(ran)} requested benchmarks passed "
f"({len(all_results)} total in catalog for {label})"
)
if __name__ == "__main__":
main()