Files

308 lines
11 KiB
Python
Raw Permalink Normal View History

#!/usr/bin/env python3
"""Corpus overhead harness for nash (docs/NASH.md §11, M2/M3 gate: <3%).
Replays the corpus under bash and nash in alternating rounds and compares total
wall-clock **and CPU time** per shell. Only the shell subprocess is measured
(fixture seeding and filesystem snapshots are outside both clocks). Rounds
alternate shell order so cache/thermal drift cancels; the reported figures use
the median round total.
Two measurements, because they answer different questions
(docs/NASH_STREAM_PERF_PLAN.md §2):
* **wall** is what a single command waits for — the M3 gate.
* **cpu** (user+sys of the shell and everything it spawned, via `rusage`) is
what nash actually spends. The pipe tee runs on its own thread, so on idle
hardware it can add CPU while *costing no wall time at all* — and a
wall-only harness would report that as free. It is not: the deployed box
runs many agents at once, where that CPU comes out of everyone's clock.
Beyond the corpus, `--stream-mb` runs a **throughput** case — hundreds of MB
across one and two pipe links — which is the shape the tee is optimized for and
the one the corpus (a few KB per command) cannot see.
Usage: overhead.py [--nash PATH] [--bash PATH] [--corpus PATH] [--rounds N]
[--observe-spool DIR] [--gate PCT] [--stream-mb MB]
[--stream-gate PCT] [--json PATH] [--allow-nash-baseline]
--observe-spool enables nash observation (spool transport) so the measured
configuration is the deployed one; the env is set identically for bash, where
it is inert.
The baseline shell must be a *real* bash: on a Nucleic-managed box `/bin/bash`
IS nash (docs/NASH.md §7), so the default baseline is `/usr/bin/bash.real` and
a baseline that reports `NUCLEIC_NASH=1` is refused outright — benchmarking
nash against itself reports ~0% overhead and means nothing.
"""
import argparse
import json
import os
import resource
import shutil
import statistics
import subprocess
import sys
import tempfile
import time
from replay import FIXTURES, TIMEOUT_S, seed
#: Where a Nucleic-managed image keeps the real bash after the nash divert.
DEFAULT_BASH = "/usr/bin/bash.real"
def child_cpu():
"""User+sys seconds of every child this process has reaped."""
usage = resource.getrusage(resource.RUSAGE_CHILDREN)
return usage.ru_utime + usage.ru_stime
def run_timed(shell, cmd, extra_env, workdir=None):
"""Run one command, returning (wall_seconds, cpu_seconds).
CPU comes from the delta of `RUSAGE_CHILDREN`, which only counts children
already reaped — `subprocess.run` waits, and the harness runs one child at a
time, so the delta is exactly this command's shell and its descendants
(including nash's tee and flusher threads).
"""
parent = None
if workdir is None:
parent = tempfile.mkdtemp(prefix="nash-overhead-")
workdir = os.path.join(parent, "workspace")
os.makedirs(workdir)
seed(workdir)
env = {
"PATH": "/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin",
"HOME": workdir,
"LC_ALL": "C",
"LANG": "C",
"TERM": "dumb",
"SHELL": shell,
}
if extra_env:
env.update(extra_env)
cpu_before = child_cpu()
start = time.monotonic()
try:
subprocess.run(
[shell, "-c", cmd],
cwd=workdir,
env=env,
capture_output=True,
timeout=TIMEOUT_S,
)
except subprocess.TimeoutExpired:
pass
elapsed = time.monotonic() - start
cpu = child_cpu() - cpu_before
if parent:
shutil.rmtree(parent, ignore_errors=True)
return elapsed, cpu
def is_nash(shell):
"""Whether `shell` is nash wearing another name (docs/NASH.md §2).
Probed with a scrubbed environment on purpose: the harness itself is very
likely running *under* nash, which exports `NUCLEIC_NASH=1` to everything it
spawns — inheriting that would make every shell look like nash. Only a shell
that sets the variable for itself answers 1 here.
"""
try:
out = subprocess.run(
[shell, "-c", 'printf %s "${NUCLEIC_NASH-}"'],
capture_output=True,
timeout=30,
text=True,
env={"PATH": "/usr/bin:/bin"},
)
except (OSError, subprocess.SubprocessError):
return False
return out.stdout.strip() == "1"
def stream_cases(megabytes):
"""Throughput scripts: the same payload across one link and across two.
`/dev/zero → /dev/null` deliberately: the point is the cost of *carrying*
bytes across a tapped link, so neither end should be doing work of its own.
"""
count = megabytes * 1024 * 1024
return [
(f"1 link ({megabytes} MB)", f"head -c {count} /dev/zero | cat > /dev/null"),
(
f"2 links ({megabytes} MB)",
f"head -c {count} /dev/zero | cat | cat > /dev/null",
),
]
def measure(shell, cmds, extra_env):
"""Total (wall, cpu) for one pass over `cmds`."""
wall = cpu = 0.0
for cmd in cmds:
w, c = run_timed(shell, cmd, extra_env)
wall += w
cpu += c
return wall, cpu
def percent(nash, bash):
"""nash's cost over bash's, in percent. Infinite-safe for a zero baseline."""
return 100.0 * (nash / bash - 1.0) if bash > 0 else float("nan")
def report(label, bash_vals, nash_vals, gate=None):
"""Print (and return) one comparison's medians and percentages."""
bash_wall = statistics.median(w for w, _ in bash_vals)
bash_cpu = statistics.median(c for _, c in bash_vals)
nash_wall = statistics.median(w for w, _ in nash_vals)
nash_cpu = statistics.median(c for _, c in nash_vals)
result = {
"bash_wall_s": bash_wall,
"bash_cpu_s": bash_cpu,
"nash_wall_s": nash_wall,
"nash_cpu_s": nash_cpu,
"wall_percent": percent(nash_wall, bash_wall),
"cpu_percent": percent(nash_cpu, bash_cpu),
}
print(f"\n{label}")
print(f" bash: {bash_wall:.3f}s wall / {bash_cpu:.3f}s cpu")
print(f" nash: {nash_wall:.3f}s wall / {nash_cpu:.3f}s cpu")
suffix = f" (gate: <{gate:.1f}%)" if gate is not None else ""
print(
f" overhead: {result['wall_percent']:+.2f}% wall / "
f"{result['cpu_percent']:+.2f}% cpu{suffix}"
)
return result
def main():
ap = argparse.ArgumentParser()
ap.add_argument("--nash", default=os.environ.get("NASH_BIN", "nash"))
ap.add_argument(
"--bash",
default=DEFAULT_BASH,
help=f"baseline shell — must not be nash (default {DEFAULT_BASH})",
)
ap.add_argument(
"--corpus",
default=os.path.join(os.path.dirname(os.path.abspath(__file__)), "corpus.jsonl"),
)
ap.add_argument("--rounds", type=int, default=5)
ap.add_argument("--observe-spool", help="enable nash observation, spooling to this dir")
ap.add_argument("--gate", type=float, default=3.0, help="max corpus wall overhead percent")
ap.add_argument(
"--stream-mb",
type=int,
default=256,
help="payload per throughput case (0 disables the stream cases)",
)
ap.add_argument(
"--stream-gate",
type=float,
default=50.0,
help="max stream CPU overhead percent (the tee's own cost)",
)
ap.add_argument(
"--allow-nash-baseline",
action="store_true",
help="benchmark against a nash baseline anyway (produces meaningless numbers)",
)
ap.add_argument("--json", help="also write results to this path")
args = ap.parse_args()
# A baseline that is itself nash makes every number here ~0% and hides whatever
# regressed. On a Nucleic box that is the *default* state of /bin/bash, so this is
# a refusal rather than a warning (docs/NASH_STREAM_PERF_PLAN.md §regression guard).
if is_nash(args.bash) and not args.allow_nash_baseline:
print(
f"error: baseline shell {args.bash} reports NUCLEIC_NASH=1 — it IS nash.\n"
f" Point --bash at the real bash ({DEFAULT_BASH} on a Nucleic image),\n"
" or pass --allow-nash-baseline if you really mean to compare nash to nash.",
file=sys.stderr,
)
return 2
observe_env = None
if args.observe_spool:
os.makedirs(args.observe_spool, exist_ok=True)
observe_env = {
"NUCLEIC_SHELL_SPOOL": args.observe_spool,
"NUCLEIC_SESSION_ID": "corpus-overhead",
}
with open(args.corpus) as f:
cmds = [json.loads(line)["cmd"] for line in f if line.strip()]
streams = stream_cases(args.stream_mb) if args.stream_mb > 0 else []
# Warm-up: one untimed pass per shell (page cache, binary load).
for shell in (args.bash, args.nash):
for cmd in cmds[:10]:
run_timed(shell, cmd, observe_env)
totals = {"bash": [], "nash": []}
stream_totals = {name: {"bash": [], "nash": []} for name, _ in streams}
for round_no in range(args.rounds):
order = [("bash", args.bash), ("nash", args.nash)]
if round_no % 2:
order.reverse()
for name, shell in order:
wall, cpu = measure(shell, cmds, observe_env)
totals[name].append((wall, cpu))
print(f"round {round_no + 1} {name}: {wall:.3f}s wall / {cpu:.3f}s cpu", flush=True)
for case, script in streams:
stream_totals[case][name].append(run_timed(shell, script, observe_env))
corpus = report(
f"corpus ({len(cmds)} cmds x {args.rounds} rounds)",
totals["bash"],
totals["nash"],
gate=args.gate,
)
stream_results = {
case: report(f"stream {case}", vals["bash"], vals["nash"], gate=args.stream_gate)
for case, vals in stream_totals.items()
}
corpus_ok = corpus["wall_percent"] < args.gate
# The stream cases are gated on CPU: the tee's cost is a copier thread, which idle
# hardware hides from wall-clock entirely.
stream_ok = all(r["cpu_percent"] < args.stream_gate for r in stream_results.values())
if args.json:
with open(args.json, "w") as f:
json.dump(
{
"cmds": len(cmds),
"rounds": args.rounds,
"totals": {k: [{"wall_s": w, "cpu_s": c} for w, c in v] for k, v in totals.items()},
"corpus": corpus,
"streams": stream_results,
"stream_mb": args.stream_mb,
"gate_percent": args.gate,
"stream_gate_percent": args.stream_gate,
"observed": bool(args.observe_spool),
"bash": args.bash,
"nash": args.nash,
# Kept for readers of the pre-CPU schema.
"bash_median_s": corpus["bash_wall_s"],
"nash_median_s": corpus["nash_wall_s"],
"overhead_percent": corpus["wall_percent"],
},
f,
indent=1,
)
print(f"\nM3 overhead gate (corpus wall <{args.gate:.1f}%): {'PASS' if corpus_ok else 'FAIL'}")
if streams:
print(
f"stream gate (cpu <{args.stream_gate:.1f}%): {'PASS' if stream_ok else 'FAIL'}"
)
return 0 if corpus_ok and stream_ok else 1
if __name__ == "__main__":
sys.exit(main())