From d2795d17111edfa86312cd8570ba9b2a2b754e38 Mon Sep 17 00:00:00 2001 From: Ramesh Padmanabhaiah <22363102+codeforester@users.noreply.github.com> Date: Wed, 2 Sep 2026 14:43:56 +0530 Subject: [PATCH] ci: calibrate WSL benchmark budget --- .github/workflows/tests.yml | 2 +- docs/performance.md | 11 +++++-- scripts/benchmark_runtime.py | 55 ++++++++++++++++++++++++++++++--- tests/test_benchmark_runtime.py | 6 ++++ 4 files changed, 65 insertions(+), 9 deletions(-) diff --git a/.github/workflows/tests.yml b/.github/workflows/tests.yml index 68e7db0..842d9c0 100644 --- a/.github/workflows/tests.yml +++ b/.github/workflows/tests.yml @@ -138,4 +138,4 @@ jobs: $drive = $env:GITHUB_WORKSPACE.Substring(0, 1).ToLowerInvariant() $path = $env:GITHUB_WORKSPACE.Substring(2).Replace('\', '/') $linuxWorkspace = "/mnt/$drive$path" - wsl --distribution Ubuntu --user root -- bash -lc "set -eu; cd '$linuxWorkspace'; sed -i 's/\r$//' tests/full_validate.sh tests/validate.sh; apt-get update -qq; apt-get install -y -qq python3-venv python3.14-venv; python3 -m venv /tmp/base-cli-venv; . /tmp/base-cli-venv/bin/activate; python -m pip install '.[dev,typer,quality]'; bash tests/full_validate.sh" + wsl --distribution Ubuntu --user root -- bash -lc "set -eu; cd '$linuxWorkspace'; sed -i 's/\r$//' tests/full_validate.sh tests/validate.sh; apt-get update -qq; apt-get install -y -qq python3-venv python3.14-venv; python3 -m venv /tmp/base-cli-venv; . /tmp/base-cli-venv/bin/activate; python -m pip install '.[dev,typer,quality]'; export BASE_CLI_BENCHMARK_PLATFORM=wsl; bash tests/full_validate.sh" diff --git a/docs/performance.md b/docs/performance.md index 7e1c188..4cd70df 100644 --- a/docs/performance.md +++ b/docs/performance.md @@ -17,12 +17,17 @@ Cyclopts. Install the optional benchmark extra to include Cyclopts: python -m pip install 'base-cli[benchmark]' ``` -The CI quality job checks the base-cli sample p95 against these budgets: +The CI quality job checks the base-cli sample p95 against these budgets. The +benchmark records the selected platform profile in both text and JSON output; +set `BASE_CLI_BENCHMARK_PLATFORM` when a runner's filesystem or virtualization +boundary is not represented by the host operating system. Supported profiles +are `unix`, `macos`, `windows`, and `wsl`. | Measurement | Budget | | --- | ---: | -| Fresh `import base_cli` (Unix) | 750 ms | -| Fresh `import base_cli` (Windows) | 1,000 ms | +| Fresh `import base_cli` (native Unix/macOS) | 750 ms | +| Fresh `import base_cli` (native Windows) | 1,000 ms | +| Fresh `import base_cli` (WSL2 on a Windows-mounted checkout) | 1,000 ms | | Isolated invocation and runtime filesystem setup | 1,500 ms | The benchmark reports the median, p95, and maximum for seven samples. Pass diff --git a/scripts/benchmark_runtime.py b/scripts/benchmark_runtime.py index 13064b4..a2b5c2d 100644 --- a/scripts/benchmark_runtime.py +++ b/scripts/benchmark_runtime.py @@ -16,10 +16,44 @@ from pathlib import Path from typing import Any, TypedDict, cast -# Windows hosted runners have a materially slower fresh Python process start -# (the import probe includes that process startup by design). Keep the tighter -# budget on Unix while allowing the documented Windows baseline headroom. -IMPORT_P95_BUDGET_MS = 1_000.0 if os.name == "nt" else 750.0 +# Fresh-process startup is materially slower on native Windows and on WSL +# when the checkout is on the Windows-mounted filesystem. CI can select the +# platform explicitly with BASE_CLI_BENCHMARK_PLATFORM; the fallback keeps +# local runs useful without requiring a setting. +IMPORT_P95_BUDGETS_MS = { + "unix": 750.0, + "macos": 750.0, + "windows": 1_000.0, + "wsl": 1_000.0, +} + + +def _is_wsl() -> bool: + try: + proc_version = Path("/proc/version").read_text(encoding="utf-8").lower() + except OSError: + return False + return "microsoft" in proc_version or "wsl" in proc_version + + +def _benchmark_platform() -> str: + configured = os.environ.get("BASE_CLI_BENCHMARK_PLATFORM", "").strip().lower() + if configured: + if configured not in IMPORT_P95_BUDGETS_MS: + supported = ", ".join(sorted(IMPORT_P95_BUDGETS_MS)) + raise ValueError(f"BASE_CLI_BENCHMARK_PLATFORM must be one of {supported}; got {configured!r}") + return configured + if os.name == "nt": + return "windows" + if sys.platform == "darwin": + return "macos" + if _is_wsl(): + return "wsl" + return "unix" + + +BENCHMARK_PLATFORM = _benchmark_platform() +IMPORT_P95_BUDGET_MS = IMPORT_P95_BUDGETS_MS[BENCHMARK_PLATFORM] INVOCATION_P95_BUDGET_MS = 1_500.0 DEFAULT_ITERATIONS = 7 FRAMEWORKS = ("base-cli", "click", "typer", "cyclopts") @@ -69,8 +103,19 @@ def main() -> int: "invocation_ms": _summary(_measure_invocations(args.iterations, framework)), } if args.json: - print(json.dumps({"iterations": args.iterations, "frameworks": metrics}, sort_keys=True)) + print( + json.dumps( + { + "iterations": args.iterations, + "platform": BENCHMARK_PLATFORM, + "import_p95_budget_ms": IMPORT_P95_BUDGET_MS, + "frameworks": metrics, + }, + sort_keys=True, + ) + ) else: + print(f"benchmark platform: {BENCHMARK_PLATFORM} (import p95 budget {IMPORT_P95_BUDGET_MS:.0f} ms)") for framework, result in metrics.items(): if result.get("status") == "unavailable": print(f"{framework}: unavailable (install it to include this comparison)") diff --git a/tests/test_benchmark_runtime.py b/tests/test_benchmark_runtime.py index 92bed09..4ba4032 100644 --- a/tests/test_benchmark_runtime.py +++ b/tests/test_benchmark_runtime.py @@ -25,3 +25,9 @@ def test_framework_comparison_has_stable_public_set(self) -> None: benchmark_runtime.FRAMEWORKS, ("base-cli", "click", "typer", "cyclopts"), ) + + def test_platform_profiles_have_explicit_import_budgets(self) -> None: + self.assertEqual(benchmark_runtime.IMPORT_P95_BUDGETS_MS["unix"], 750.0) + self.assertEqual(benchmark_runtime.IMPORT_P95_BUDGETS_MS["macos"], 750.0) + self.assertEqual(benchmark_runtime.IMPORT_P95_BUDGETS_MS["windows"], 1_000.0) + self.assertEqual(benchmark_runtime.IMPORT_P95_BUDGETS_MS["wsl"], 1_000.0)