#!/usr/bin/env python3 """Corpus overhead harness for nash (docs/NASH.md §11, M2/M3 gate: <3%). Replays the corpus under bash and nash in alternating rounds and compares total wall-clock **and CPU time** per shell. Only the shell subprocess is measured (fixture seeding and filesystem snapshots are outside both clocks). Rounds alternate shell order so cache/thermal drift cancels; the reported figures use the median round total. Two measurements, because they answer different questions (docs/NASH_STREAM_PERF_PLAN.md §2): * **wall** is what a single command waits for — the M3 gate. * **cpu** (user+sys of the shell and everything it spawned, via `rusage`) is what nash actually spends. The pipe tee runs on its own thread, so on idle hardware it can add CPU while *costing no wall time at all* — and a wall-only harness would report that as free. It is not: the deployed box runs many agents at once, where that CPU comes out of everyone's clock. Beyond the corpus, `--stream-mb` runs a **throughput** case — hundreds of MB across one and two pipe links — which is the shape the tee is optimized for and the one the corpus (a few KB per command) cannot see. Usage: overhead.py [--nash PATH] [--bash PATH] [--corpus PATH] [--rounds N] [--observe-spool DIR] [--gate PCT] [--stream-mb MB] [--stream-gate PCT] [--json PATH] [--allow-nash-baseline] --observe-spool enables nash observation (spool transport) so the measured configuration is the deployed one; the env is set identically for bash, where it is inert. The baseline shell must be a *real* bash: on a Nucleic-managed box `/bin/bash` IS nash (docs/NASH.md §7), so the default baseline is `/usr/bin/bash.real` and a baseline that reports `NUCLEIC_NASH=1` is refused outright — benchmarking nash against itself reports ~0% overhead and means nothing. """ import argparse import json import os import resource import shutil import statistics import subprocess import sys import tempfile import time from replay import FIXTURES, TIMEOUT_S, seed #: Where a Nucleic-managed image keeps the real bash after the nash divert. DEFAULT_BASH = "/usr/bin/bash.real" def child_cpu(): """User+sys seconds of every child this process has reaped.""" usage = resource.getrusage(resource.RUSAGE_CHILDREN) return usage.ru_utime + usage.ru_stime def run_timed(shell, cmd, extra_env, workdir=None): """Run one command, returning (wall_seconds, cpu_seconds). CPU comes from the delta of `RUSAGE_CHILDREN`, which only counts children already reaped — `subprocess.run` waits, and the harness runs one child at a time, so the delta is exactly this command's shell and its descendants (including nash's tee and flusher threads). """ parent = None if workdir is None: parent = tempfile.mkdtemp(prefix="nash-overhead-") workdir = os.path.join(parent, "workspace") os.makedirs(workdir) seed(workdir) env = { "PATH": "/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin", "HOME": workdir, "LC_ALL": "C", "LANG": "C", "TERM": "dumb", "SHELL": shell, } if extra_env: env.update(extra_env) cpu_before = child_cpu() start = time.monotonic() try: subprocess.run( [shell, "-c", cmd], cwd=workdir, env=env, capture_output=True, timeout=TIMEOUT_S, ) except subprocess.TimeoutExpired: pass elapsed = time.monotonic() - start cpu = child_cpu() - cpu_before if parent: shutil.rmtree(parent, ignore_errors=True) return elapsed, cpu def is_nash(shell): """Whether `shell` is nash wearing another name (docs/NASH.md §2). Probed with a scrubbed environment on purpose: the harness itself is very likely running *under* nash, which exports `NUCLEIC_NASH=1` to everything it spawns — inheriting that would make every shell look like nash. Only a shell that sets the variable for itself answers 1 here. """ try: out = subprocess.run( [shell, "-c", 'printf %s "${NUCLEIC_NASH-}"'], capture_output=True, timeout=30, text=True, env={"PATH": "/usr/bin:/bin"}, ) except (OSError, subprocess.SubprocessError): return False return out.stdout.strip() == "1" def stream_cases(megabytes): """Throughput scripts: the same payload across one link and across two. `/dev/zero → /dev/null` deliberately: the point is the cost of *carrying* bytes across a tapped link, so neither end should be doing work of its own. """ count = megabytes * 1024 * 1024 return [ (f"1 link ({megabytes} MB)", f"head -c {count} /dev/zero | cat > /dev/null"), ( f"2 links ({megabytes} MB)", f"head -c {count} /dev/zero | cat | cat > /dev/null", ), ] def measure(shell, cmds, extra_env): """Total (wall, cpu) for one pass over `cmds`.""" wall = cpu = 0.0 for cmd in cmds: w, c = run_timed(shell, cmd, extra_env) wall += w cpu += c return wall, cpu def percent(nash, bash): """nash's cost over bash's, in percent. Infinite-safe for a zero baseline.""" return 100.0 * (nash / bash - 1.0) if bash > 0 else float("nan") def report(label, bash_vals, nash_vals, gate=None): """Print (and return) one comparison's medians and percentages.""" bash_wall = statistics.median(w for w, _ in bash_vals) bash_cpu = statistics.median(c for _, c in bash_vals) nash_wall = statistics.median(w for w, _ in nash_vals) nash_cpu = statistics.median(c for _, c in nash_vals) result = { "bash_wall_s": bash_wall, "bash_cpu_s": bash_cpu, "nash_wall_s": nash_wall, "nash_cpu_s": nash_cpu, "wall_percent": percent(nash_wall, bash_wall), "cpu_percent": percent(nash_cpu, bash_cpu), } print(f"\n{label}") print(f" bash: {bash_wall:.3f}s wall / {bash_cpu:.3f}s cpu") print(f" nash: {nash_wall:.3f}s wall / {nash_cpu:.3f}s cpu") suffix = f" (gate: <{gate:.1f}%)" if gate is not None else "" print( f" overhead: {result['wall_percent']:+.2f}% wall / " f"{result['cpu_percent']:+.2f}% cpu{suffix}" ) return result def main(): ap = argparse.ArgumentParser() ap.add_argument("--nash", default=os.environ.get("NASH_BIN", "nash")) ap.add_argument( "--bash", default=DEFAULT_BASH, help=f"baseline shell — must not be nash (default {DEFAULT_BASH})", ) ap.add_argument( "--corpus", default=os.path.join(os.path.dirname(os.path.abspath(__file__)), "corpus.jsonl"), ) ap.add_argument("--rounds", type=int, default=5) ap.add_argument("--observe-spool", help="enable nash observation, spooling to this dir") ap.add_argument("--gate", type=float, default=3.0, help="max corpus wall overhead percent") ap.add_argument( "--stream-mb", type=int, default=256, help="payload per throughput case (0 disables the stream cases)", ) ap.add_argument( "--stream-gate", type=float, default=50.0, help="max stream CPU overhead percent (the tee's own cost)", ) ap.add_argument( "--allow-nash-baseline", action="store_true", help="benchmark against a nash baseline anyway (produces meaningless numbers)", ) ap.add_argument("--json", help="also write results to this path") args = ap.parse_args() # A baseline that is itself nash makes every number here ~0% and hides whatever # regressed. On a Nucleic box that is the *default* state of /bin/bash, so this is # a refusal rather than a warning (docs/NASH_STREAM_PERF_PLAN.md §regression guard). if is_nash(args.bash) and not args.allow_nash_baseline: print( f"error: baseline shell {args.bash} reports NUCLEIC_NASH=1 — it IS nash.\n" f" Point --bash at the real bash ({DEFAULT_BASH} on a Nucleic image),\n" " or pass --allow-nash-baseline if you really mean to compare nash to nash.", file=sys.stderr, ) return 2 observe_env = None if args.observe_spool: os.makedirs(args.observe_spool, exist_ok=True) observe_env = { "NUCLEIC_SHELL_SPOOL": args.observe_spool, "NUCLEIC_SESSION_ID": "corpus-overhead", } with open(args.corpus) as f: cmds = [json.loads(line)["cmd"] for line in f if line.strip()] streams = stream_cases(args.stream_mb) if args.stream_mb > 0 else [] # Warm-up: one untimed pass per shell (page cache, binary load). for shell in (args.bash, args.nash): for cmd in cmds[:10]: run_timed(shell, cmd, observe_env) totals = {"bash": [], "nash": []} stream_totals = {name: {"bash": [], "nash": []} for name, _ in streams} for round_no in range(args.rounds): order = [("bash", args.bash), ("nash", args.nash)] if round_no % 2: order.reverse() for name, shell in order: wall, cpu = measure(shell, cmds, observe_env) totals[name].append((wall, cpu)) print(f"round {round_no + 1} {name}: {wall:.3f}s wall / {cpu:.3f}s cpu", flush=True) for case, script in streams: stream_totals[case][name].append(run_timed(shell, script, observe_env)) corpus = report( f"corpus ({len(cmds)} cmds x {args.rounds} rounds)", totals["bash"], totals["nash"], gate=args.gate, ) stream_results = { case: report(f"stream {case}", vals["bash"], vals["nash"], gate=args.stream_gate) for case, vals in stream_totals.items() } corpus_ok = corpus["wall_percent"] < args.gate # The stream cases are gated on CPU: the tee's cost is a copier thread, which idle # hardware hides from wall-clock entirely. stream_ok = all(r["cpu_percent"] < args.stream_gate for r in stream_results.values()) if args.json: with open(args.json, "w") as f: json.dump( { "cmds": len(cmds), "rounds": args.rounds, "totals": {k: [{"wall_s": w, "cpu_s": c} for w, c in v] for k, v in totals.items()}, "corpus": corpus, "streams": stream_results, "stream_mb": args.stream_mb, "gate_percent": args.gate, "stream_gate_percent": args.stream_gate, "observed": bool(args.observe_spool), "bash": args.bash, "nash": args.nash, # Kept for readers of the pre-CPU schema. "bash_median_s": corpus["bash_wall_s"], "nash_median_s": corpus["nash_wall_s"], "overhead_percent": corpus["wall_percent"], }, f, indent=1, ) print(f"\nM3 overhead gate (corpus wall <{args.gate:.1f}%): {'PASS' if corpus_ok else 'FAIL'}") if streams: print( f"stream gate (cpu <{args.stream_gate:.1f}%): {'PASS' if stream_ok else 'FAIL'}" ) return 0 if corpus_ok and stream_ok else 1 if __name__ == "__main__": sys.exit(main())