Repository navigation
Add scripts for long-term monitoring of eval performance #671
New issue
Have a question about this project? Sign up for a free GitHub account to open an issue and contact its maintainers and the community.
By clicking “Sign up for GitHub”, you agree to our terms of service and privacy statement. We’ll occasionally send you account related emails.
Already on GitHub? Sign in to your account
base: main
Are you sure you want to change the base?
Changes from all commits
File filter
Filter by extension
Conversations
Jump to
Diff view
Diff view
There are no files selected for viewing
| Original file line number | Diff line number | Diff line change |
|---|---|---|
| @@ -0,0 +1,134 @@ | ||
| #!/usr/bin/env nix | ||
| #!nix shell --inputs-from .. nixpkgs#python3 --command python3 | ||
|
|
||
| """ | ||
| Benchmark Nix evaluation performance across a set of Nix releases. | ||
|
|
||
| For each tag (e.g. `v3.23.1` for Determinate Nix, `2.35.2` for upstream Nix, | ||
| or a full Git revision of DeterminateSystems/nix-src), each test and each run | ||
| number, this runs the test and appends a row to the CSV file. Runs that are | ||
| already recorded in the CSV file are skipped, and Nix versions are only built | ||
| when needed. | ||
| """ | ||
|
|
||
| import argparse | ||
| import csv | ||
| import os | ||
| import re | ||
| import subprocess | ||
| import sys | ||
| import time | ||
| from pathlib import Path | ||
|
|
||
| NIXPKGS = "github:NixOS/nixpkgs/531670d871c0e29724a02f3cbcac170adc65b58c" | ||
|
|
||
| TESTS = { | ||
| "search": ["search", NIXPKGS, "fizzbuzz", "--no-eval-cache"], | ||
| "firefox": ["eval", "--json", f"{NIXPKGS}#firefox", "--read-only"], | ||
| "plasma": [ | ||
| "eval", | ||
| "--json", | ||
| "-I", | ||
| f"nixpkgs=flake:{NIXPKGS}", | ||
| "--file", | ||
| "<nixpkgs/nixos/release-combined.nix>", | ||
| "nixos.tests.plasma6.x86_64-linux", | ||
| "--read-only", | ||
| ], | ||
| } | ||
|
|
||
| EXTRA_ARGS = ["--option", "eval-cores", "0"] | ||
|
|
||
| HEADER = ["nix-tag", "test-name", "test-run", "elapsed-time", "cpu-time", "kernel-time", "max-rss-kib"] | ||
|
|
||
|
|
||
| def flake_ref(tag: str) -> str: | ||
| # Determinate Nix tags and revisions come from nix-src, upstream tags from NixOS/nix. | ||
| if tag.startswith("v") or re.fullmatch(r"[0-9a-f]{40}", tag): | ||
| return f"github:DeterminateSystems/nix-src/{tag}" | ||
| return f"github:NixOS/nix/{tag}" | ||
|
|
||
|
|
||
| def build_nix(tag: str) -> Path | None: | ||
| ref = flake_ref(tag) | ||
| print(f"building {ref}...", file=sys.stderr) | ||
| res = subprocess.run(["nix", "build", "--no-link", "--print-out-paths", ref], stdout=subprocess.PIPE, text=True) | ||
| if res.returncode != 0: | ||
| print(f"failed to build {ref}, skipping", file=sys.stderr) | ||
| return None | ||
| # The package has multiple outputs (e.g. `man`), so find the one with the binary. | ||
| for out in res.stdout.split(): | ||
| nix_bin = Path(out) / "bin" / "nix" | ||
| if os.access(nix_bin, os.X_OK): | ||
| return nix_bin | ||
| print(f"{ref} does not provide bin/nix, skipping", file=sys.stderr) | ||
| return None | ||
|
|
||
|
|
||
| def run_test(nix_bin: Path, test: str) -> list[str] | None: | ||
| """Run a test, returning the measurements, or None if the test failed.""" | ||
| start = time.monotonic() | ||
| proc = subprocess.Popen([nix_bin, *TESTS[test], *EXTRA_ARGS], stdout=subprocess.DEVNULL) | ||
| _, status, rusage = os.wait4(proc.pid, 0) | ||
| elapsed = time.monotonic() - start | ||
| proc.returncode = os.waitstatus_to_exitcode(status) | ||
| if proc.returncode != 0: | ||
| return None | ||
| return [ | ||
| f"{elapsed:.3f}", | ||
| f"{rusage.ru_utime:.3f}", | ||
| f"{rusage.ru_stime:.3f}", | ||
| str(rusage.ru_maxrss), # KiB on Linux | ||
|
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. 🗄️ Data Integrity & Integration | 🟠 Major | ⚡ Quick win Normalize RSS before writing the KiB column. On macOS, 🤖 Prompt for AI Agents |
||
| ] | ||
|
|
||
|
|
||
| def main() -> None: | ||
| parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) | ||
| parser.add_argument("tags", metavar="TAG", nargs="+", help="Nix release tag or nix-src revision") | ||
| parser.add_argument("-o", "--output", type=Path, default=Path("eval-benchmark.csv"), help="CSV file to append to") | ||
| parser.add_argument("-n", "--runs", type=int, default=5, help="number of runs per test/tag pair") | ||
| parser.add_argument("-t", "--test", dest="tests", action="append", choices=TESTS.keys(), help="test to run (default: all)") | ||
| args = parser.parse_args() | ||
|
|
||
| tests = args.tests or list(TESTS.keys()) | ||
|
|
||
| done = set() | ||
| if args.output.exists(): | ||
| with args.output.open(newline="") as f: | ||
| for row in csv.DictReader(f): | ||
| done.add((row["nix-tag"], row["test-name"], row["test-run"])) | ||
| else: | ||
| with args.output.open("w", newline="") as f: | ||
| csv.writer(f).writerow(HEADER) | ||
|
Comment on lines
+96
to
+102
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. 🗄️ Data Integrity & Integration | 🟡 Minor | ⚡ Quick win Write a header when the output file is empty. If 🤖 Prompt for AI Agents |
||
|
|
||
| for tag in args.tags: | ||
| nix_bin = None | ||
| build_failed = False | ||
|
|
||
| for test in tests: | ||
| if build_failed: | ||
| break | ||
|
|
||
| for run in range(1, args.runs + 1): | ||
| if (tag, test, str(run)) in done: | ||
| continue | ||
|
|
||
| if nix_bin is None: | ||
| nix_bin = build_nix(tag) | ||
| if nix_bin is None: | ||
| build_failed = True | ||
| break | ||
|
|
||
| measurements = run_test(nix_bin, test) | ||
| if measurements is None: | ||
| print(f"{tag} {test} {run}: failed", file=sys.stderr) | ||
| break | ||
|
|
||
| # Append each row immediately so that an interrupted run loses at most one measurement. | ||
| with args.output.open("a", newline="") as f: | ||
| csv.writer(f).writerow([tag, test, run, *measurements]) | ||
|
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. 🗄️ Data Integrity & Integration | 🟡 Minor | ⚡ Quick win Mark each appended run as completed. If a tag or 🤖 Prompt for AI Agents |
||
| print(f"{tag} {test} {run}: {measurements[0]} s", file=sys.stderr) | ||
|
|
||
|
|
||
| if __name__ == "__main__": | ||
| main() | ||
| Original file line number | Diff line number | Diff line change |
|---|---|---|
| @@ -0,0 +1,124 @@ | ||
| #!/usr/bin/env nix | ||
| #! nix shell --impure --expr `` | ||
| #! nix with (builtins.getFlake (toString ./..)).inputs.nixpkgs.legacyPackages.${builtins.currentSystem}; | ||
| #! nix python3.withPackages (ps: [ ps.matplotlib ]) | ||
| #! nix `` | ||
| #! nix --command python3 | ||
|
|
||
| """Plot the minimum elapsed time and max RSS per Nix release for each test in an eval-benchmark.py CSV file.""" | ||
|
|
||
| import argparse | ||
| import csv | ||
| import re | ||
| from collections import defaultdict | ||
|
|
||
| import matplotlib | ||
|
|
||
| matplotlib.use("Agg") | ||
| import matplotlib.pyplot as plt | ||
|
|
||
| parser = argparse.ArgumentParser(description=__doc__) | ||
| parser.add_argument("csv", nargs="?", default="eval-benchmark.csv", metavar="FILE.csv") | ||
| parser.add_argument( | ||
| "-o", "--output", help="output file (default: eval-benchmark.png, or .svg with --svg)" | ||
| ) | ||
| parser.add_argument( | ||
| "--svg", | ||
| action="store_true", | ||
| help="write a transparent SVG styled for a dark background", | ||
| ) | ||
| parser.add_argument("--log", action="store_true", help="use a logarithmic y-axis") | ||
| args = parser.parse_args() | ||
|
|
||
| if args.output is None: | ||
| args.output = "eval-benchmark.svg" if args.svg else "eval-benchmark.png" | ||
|
|
||
| if args.svg: | ||
| fg = "#dddddd" | ||
| plt.rcParams.update({ | ||
| "svg.fonttype": "none", | ||
| "text.color": fg, | ||
| "axes.labelcolor": fg, | ||
| "axes.edgecolor": fg, | ||
| "axes.titlecolor": fg, | ||
| "xtick.color": fg, | ||
| "ytick.color": fg, | ||
| "grid.color": fg, | ||
| "legend.edgecolor": fg, | ||
| "legend.facecolor": "none", | ||
| "legend.labelcolor": fg, | ||
| "figure.facecolor": "none", | ||
| "axes.facecolor": "none", | ||
| "savefig.facecolor": "none", | ||
| "savefig.edgecolor": "none", | ||
| }) | ||
|
|
||
|
|
||
| def version_key(tag): | ||
| """Sort key for tags such that v3.1.0 < v3.2.0 < v3.10.0.""" | ||
| return [int(p) if p.isdigit() else p for p in re.split(r"(\d+)", tag.removeprefix("v"))] | ||
|
|
||
|
|
||
| def is_revision(tag): | ||
| return re.fullmatch(r"[0-9a-f]{40}", tag) is not None | ||
|
|
||
|
|
||
| def is_determinate(tag): | ||
| # Revisions are of DeterminateSystems/nix-src. | ||
| return tag.startswith("v") or is_revision(tag) | ||
|
|
||
|
|
||
| # (CSV column, y-axis label, scale factor), plotted as columns from left to right. | ||
| METRICS = [ | ||
| ("elapsed-time", "Elapsed time (s)", 1), | ||
| ("max-rss-kib", "Max RSS (MiB)", 1 / 1024), | ||
| ] | ||
|
|
||
| # data[metric][test][tag] = list of measurements | ||
| data = {metric: defaultdict(lambda: defaultdict(list)) for metric, _, _ in METRICS} | ||
| with open(args.csv, newline="") as f: | ||
| for row in csv.DictReader(f): | ||
| for metric, _, scale in METRICS: | ||
| data[metric][row["test-name"]][row["nix-tag"]].append(float(row[metric]) * scale) | ||
|
|
||
| times = data["elapsed-time"] | ||
| tests = list(times) | ||
| # Release tags sorted by version, followed by revisions in the order in which they appear in the CSV file. | ||
| all_tags = list(dict.fromkeys(tag for per_tag in times.values() for tag in per_tag)) | ||
| tags = sorted([t for t in all_tags if not is_revision(t)], key=version_key) + [t for t in all_tags if is_revision(t)] | ||
|
Comment on lines
+87
to
+88
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. 🎯 Functional Correctness | 🟡 Minor | ⚡ Quick win Preserve CSV order for revisions. Line 87 collects tags by test, not by CSV row. If a revision first appears under a later test, the plot can place it after a revision recorded later under the first test. Collect unique tags while reading the CSV, then sort release tags and retain revisions in their first-seen order. This preserves the stated benchmark order. 🤖 Prompt for AI Agents |
||
| x = {tag: i for i, tag in enumerate(tags)} | ||
|
|
||
| fig, axes = plt.subplots( | ||
| len(tests), | ||
| len(METRICS), | ||
| figsize=(max(10, 0.4 * len(tags)) * len(METRICS), 4 * len(tests)), | ||
| sharex=True, | ||
| squeeze=False, | ||
| ) | ||
|
|
||
| for row_axes, test in zip(axes, tests): | ||
| for ax, (metric, ylabel, _) in zip(row_axes, METRICS): | ||
| values = data[metric][test] | ||
| for name, pred in [("Nix", lambda t: not is_determinate(t)), ("Determinate Nix", is_determinate)]: | ||
| series = [tag for tag in tags if tag in values and pred(tag)] | ||
| if series: | ||
| ax.plot([x[t] for t in series], [min(values[t]) for t in series], marker=".", label=name) | ||
| ax.set_title(test) | ||
| ax.set_ylabel(ylabel) | ||
| if args.log: | ||
| ax.set_yscale("log") | ||
| else: | ||
| ax.set_ylim(bottom=0) | ||
| ax.grid(True, axis="y", linestyle=":", alpha=0.5) | ||
| ax.legend(title="minimum values") | ||
|
|
||
| for ax in axes[-1]: | ||
| ax.set_xticks(range(len(tags)), [t[:10] if is_revision(t) else t for t in tags], rotation=45, ha="right") | ||
| ax.set_xlabel("Nix release") | ||
| fig.suptitle("Nix evaluation performance") | ||
| fig.tight_layout() | ||
| if args.svg: | ||
| fig.savefig(args.output, format="svg", transparent=True) | ||
| else: | ||
| fig.savefig(args.output, dpi=150) | ||
| print(f"wrote {args.output}") | ||
There was a problem hiding this comment.
Choose a reason for hiding this comment
The reason will be displayed to describe this comment to others. Learn more.
🎯 Functional Correctness | 🟠 Major | ⚡ Quick win
Disable the evaluation cache for the Firefox benchmark.
When the same flake reference is evaluated again, Nix can reuse its flake evaluation cache. The
searchcase disables this cache, butfirefoxdoes not. Later Firefox runs can therefore measure a cache lookup instead of the intended evaluation. Add--no-eval-cacheto this test. (nix.dev)🤖 Prompt for AI Agents