2026-07-19 02:14:32 +00:00
|
|
|
#!/usr/bin/env python3
|
2026-07-29 04:03:22 -07:00
|
|
|
"""Corpus overhead harness for nash (docs/NASH.md §11, M2/M3 gate: <3%).
|
2026-07-19 02:14:32 +00:00
|
|
|
|
|
|
|
|
Replays the corpus under bash and nash in alternating rounds and compares total
|
2026-07-29 04:03:22 -07:00
|
|
|
wall-clock **and CPU time** per shell. Only the shell subprocess is measured
|
|
|
|
|
(fixture seeding and filesystem snapshots are outside both clocks). Rounds
|
|
|
|
|
alternate shell order so cache/thermal drift cancels; the reported figures use
|
|
|
|
|
the median round total.
|
|
|
|
|
|
|
|
|
|
Two measurements, because they answer different questions
|
|
|
|
|
(docs/NASH_STREAM_PERF_PLAN.md §2):
|
|
|
|
|
|
|
|
|
|
* **wall** is what a single command waits for — the M3 gate.
|
|
|
|
|
* **cpu** (user+sys of the shell and everything it spawned, via `rusage`) is
|
|
|
|
|
what nash actually spends. The pipe tee runs on its own thread, so on idle
|
|
|
|
|
hardware it can add CPU while *costing no wall time at all* — and a
|
|
|
|
|
wall-only harness would report that as free. It is not: the deployed box
|
|
|
|
|
runs many agents at once, where that CPU comes out of everyone's clock.
|
|
|
|
|
|
|
|
|
|
Beyond the corpus, `--stream-mb` runs a **throughput** case — hundreds of MB
|
|
|
|
|
across one and two pipe links — which is the shape the tee is optimized for and
|
|
|
|
|
the one the corpus (a few KB per command) cannot see.
|
2026-07-19 02:14:32 +00:00
|
|
|
|
|
|
|
|
Usage: overhead.py [--nash PATH] [--bash PATH] [--corpus PATH] [--rounds N]
|
2026-07-29 04:03:22 -07:00
|
|
|
[--observe-spool DIR] [--gate PCT] [--stream-mb MB]
|
|
|
|
|
[--stream-gate PCT] [--json PATH] [--allow-nash-baseline]
|
2026-07-19 02:14:32 +00:00
|
|
|
|
|
|
|
|
--observe-spool enables nash observation (spool transport) so the measured
|
|
|
|
|
configuration is the deployed one; the env is set identically for bash, where
|
|
|
|
|
it is inert.
|
2026-07-29 04:03:22 -07:00
|
|
|
|
|
|
|
|
The baseline shell must be a *real* bash: on a Nucleic-managed box `/bin/bash`
|
|
|
|
|
IS nash (docs/NASH.md §7), so the default baseline is `/usr/bin/bash.real` and
|
|
|
|
|
a baseline that reports `NUCLEIC_NASH=1` is refused outright — benchmarking
|
|
|
|
|
nash against itself reports ~0% overhead and means nothing.
|
2026-07-19 02:14:32 +00:00
|
|
|
"""
|
|
|
|
|
|
|
|
|
|
import argparse
|
|
|
|
|
import json
|
|
|
|
|
import os
|
2026-07-29 04:03:22 -07:00
|
|
|
import resource
|
2026-07-19 02:14:32 +00:00
|
|
|
import shutil
|
|
|
|
|
import statistics
|
|
|
|
|
import subprocess
|
|
|
|
|
import sys
|
|
|
|
|
import tempfile
|
|
|
|
|
import time
|
|
|
|
|
|
|
|
|
|
from replay import FIXTURES, TIMEOUT_S, seed
|
|
|
|
|
|
2026-07-29 04:03:22 -07:00
|
|
|
#: Where a Nucleic-managed image keeps the real bash after the nash divert.
|
|
|
|
|
DEFAULT_BASH = "/usr/bin/bash.real"
|
|
|
|
|
|
2026-07-19 02:14:32 +00:00
|
|
|
|
2026-07-29 04:03:22 -07:00
|
|
|
def child_cpu():
|
|
|
|
|
"""User+sys seconds of every child this process has reaped."""
|
|
|
|
|
usage = resource.getrusage(resource.RUSAGE_CHILDREN)
|
|
|
|
|
return usage.ru_utime + usage.ru_stime
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def run_timed(shell, cmd, extra_env, workdir=None):
|
|
|
|
|
"""Run one command, returning (wall_seconds, cpu_seconds).
|
|
|
|
|
|
|
|
|
|
CPU comes from the delta of `RUSAGE_CHILDREN`, which only counts children
|
|
|
|
|
already reaped — `subprocess.run` waits, and the harness runs one child at a
|
|
|
|
|
time, so the delta is exactly this command's shell and its descendants
|
|
|
|
|
(including nash's tee and flusher threads).
|
|
|
|
|
"""
|
|
|
|
|
parent = None
|
|
|
|
|
if workdir is None:
|
|
|
|
|
parent = tempfile.mkdtemp(prefix="nash-overhead-")
|
|
|
|
|
workdir = os.path.join(parent, "workspace")
|
|
|
|
|
os.makedirs(workdir)
|
|
|
|
|
seed(workdir)
|
2026-07-19 02:14:32 +00:00
|
|
|
env = {
|
|
|
|
|
"PATH": "/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin",
|
|
|
|
|
"HOME": workdir,
|
|
|
|
|
"LC_ALL": "C",
|
|
|
|
|
"LANG": "C",
|
|
|
|
|
"TERM": "dumb",
|
|
|
|
|
"SHELL": shell,
|
|
|
|
|
}
|
|
|
|
|
if extra_env:
|
|
|
|
|
env.update(extra_env)
|
2026-07-29 04:03:22 -07:00
|
|
|
cpu_before = child_cpu()
|
2026-07-19 02:14:32 +00:00
|
|
|
start = time.monotonic()
|
|
|
|
|
try:
|
|
|
|
|
subprocess.run(
|
|
|
|
|
[shell, "-c", cmd],
|
|
|
|
|
cwd=workdir,
|
|
|
|
|
env=env,
|
|
|
|
|
capture_output=True,
|
|
|
|
|
timeout=TIMEOUT_S,
|
|
|
|
|
)
|
|
|
|
|
except subprocess.TimeoutExpired:
|
|
|
|
|
pass
|
|
|
|
|
elapsed = time.monotonic() - start
|
2026-07-29 04:03:22 -07:00
|
|
|
cpu = child_cpu() - cpu_before
|
|
|
|
|
if parent:
|
|
|
|
|
shutil.rmtree(parent, ignore_errors=True)
|
|
|
|
|
return elapsed, cpu
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def is_nash(shell):
|
|
|
|
|
"""Whether `shell` is nash wearing another name (docs/NASH.md §2).
|
|
|
|
|
|
|
|
|
|
Probed with a scrubbed environment on purpose: the harness itself is very
|
|
|
|
|
likely running *under* nash, which exports `NUCLEIC_NASH=1` to everything it
|
|
|
|
|
spawns — inheriting that would make every shell look like nash. Only a shell
|
|
|
|
|
that sets the variable for itself answers 1 here.
|
|
|
|
|
"""
|
|
|
|
|
try:
|
|
|
|
|
out = subprocess.run(
|
|
|
|
|
[shell, "-c", 'printf %s "${NUCLEIC_NASH-}"'],
|
|
|
|
|
capture_output=True,
|
|
|
|
|
timeout=30,
|
|
|
|
|
text=True,
|
|
|
|
|
env={"PATH": "/usr/bin:/bin"},
|
|
|
|
|
)
|
|
|
|
|
except (OSError, subprocess.SubprocessError):
|
|
|
|
|
return False
|
|
|
|
|
return out.stdout.strip() == "1"
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def stream_cases(megabytes):
|
|
|
|
|
"""Throughput scripts: the same payload across one link and across two.
|
|
|
|
|
|
|
|
|
|
`/dev/zero → /dev/null` deliberately: the point is the cost of *carrying*
|
|
|
|
|
bytes across a tapped link, so neither end should be doing work of its own.
|
|
|
|
|
"""
|
|
|
|
|
count = megabytes * 1024 * 1024
|
|
|
|
|
return [
|
|
|
|
|
(f"1 link ({megabytes} MB)", f"head -c {count} /dev/zero | cat > /dev/null"),
|
|
|
|
|
(
|
|
|
|
|
f"2 links ({megabytes} MB)",
|
|
|
|
|
f"head -c {count} /dev/zero | cat | cat > /dev/null",
|
|
|
|
|
),
|
|
|
|
|
]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def measure(shell, cmds, extra_env):
|
|
|
|
|
"""Total (wall, cpu) for one pass over `cmds`."""
|
|
|
|
|
wall = cpu = 0.0
|
|
|
|
|
for cmd in cmds:
|
|
|
|
|
w, c = run_timed(shell, cmd, extra_env)
|
|
|
|
|
wall += w
|
|
|
|
|
cpu += c
|
|
|
|
|
return wall, cpu
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def percent(nash, bash):
|
|
|
|
|
"""nash's cost over bash's, in percent. Infinite-safe for a zero baseline."""
|
|
|
|
|
return 100.0 * (nash / bash - 1.0) if bash > 0 else float("nan")
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def report(label, bash_vals, nash_vals, gate=None):
|
|
|
|
|
"""Print (and return) one comparison's medians and percentages."""
|
|
|
|
|
bash_wall = statistics.median(w for w, _ in bash_vals)
|
|
|
|
|
bash_cpu = statistics.median(c for _, c in bash_vals)
|
|
|
|
|
nash_wall = statistics.median(w for w, _ in nash_vals)
|
|
|
|
|
nash_cpu = statistics.median(c for _, c in nash_vals)
|
|
|
|
|
result = {
|
|
|
|
|
"bash_wall_s": bash_wall,
|
|
|
|
|
"bash_cpu_s": bash_cpu,
|
|
|
|
|
"nash_wall_s": nash_wall,
|
|
|
|
|
"nash_cpu_s": nash_cpu,
|
|
|
|
|
"wall_percent": percent(nash_wall, bash_wall),
|
|
|
|
|
"cpu_percent": percent(nash_cpu, bash_cpu),
|
|
|
|
|
}
|
|
|
|
|
print(f"\n{label}")
|
|
|
|
|
print(f" bash: {bash_wall:.3f}s wall / {bash_cpu:.3f}s cpu")
|
|
|
|
|
print(f" nash: {nash_wall:.3f}s wall / {nash_cpu:.3f}s cpu")
|
|
|
|
|
suffix = f" (gate: <{gate:.1f}%)" if gate is not None else ""
|
|
|
|
|
print(
|
|
|
|
|
f" overhead: {result['wall_percent']:+.2f}% wall / "
|
|
|
|
|
f"{result['cpu_percent']:+.2f}% cpu{suffix}"
|
|
|
|
|
)
|
|
|
|
|
return result
|
2026-07-19 02:14:32 +00:00
|
|
|
|
|
|
|
|
|
|
|
|
|
def main():
|
|
|
|
|
ap = argparse.ArgumentParser()
|
|
|
|
|
ap.add_argument("--nash", default=os.environ.get("NASH_BIN", "nash"))
|
2026-07-29 04:03:22 -07:00
|
|
|
ap.add_argument(
|
|
|
|
|
"--bash",
|
|
|
|
|
default=DEFAULT_BASH,
|
|
|
|
|
help=f"baseline shell — must not be nash (default {DEFAULT_BASH})",
|
|
|
|
|
)
|
2026-07-19 02:14:32 +00:00
|
|
|
ap.add_argument(
|
|
|
|
|
"--corpus",
|
|
|
|
|
default=os.path.join(os.path.dirname(os.path.abspath(__file__)), "corpus.jsonl"),
|
|
|
|
|
)
|
|
|
|
|
ap.add_argument("--rounds", type=int, default=5)
|
|
|
|
|
ap.add_argument("--observe-spool", help="enable nash observation, spooling to this dir")
|
2026-07-29 04:03:22 -07:00
|
|
|
ap.add_argument("--gate", type=float, default=3.0, help="max corpus wall overhead percent")
|
|
|
|
|
ap.add_argument(
|
|
|
|
|
"--stream-mb",
|
|
|
|
|
type=int,
|
|
|
|
|
default=256,
|
|
|
|
|
help="payload per throughput case (0 disables the stream cases)",
|
|
|
|
|
)
|
|
|
|
|
ap.add_argument(
|
|
|
|
|
"--stream-gate",
|
|
|
|
|
type=float,
|
|
|
|
|
default=50.0,
|
|
|
|
|
help="max stream CPU overhead percent (the tee's own cost)",
|
|
|
|
|
)
|
|
|
|
|
ap.add_argument(
|
|
|
|
|
"--allow-nash-baseline",
|
|
|
|
|
action="store_true",
|
|
|
|
|
help="benchmark against a nash baseline anyway (produces meaningless numbers)",
|
|
|
|
|
)
|
2026-07-19 02:14:32 +00:00
|
|
|
ap.add_argument("--json", help="also write results to this path")
|
|
|
|
|
args = ap.parse_args()
|
|
|
|
|
|
2026-07-29 04:03:22 -07:00
|
|
|
# A baseline that is itself nash makes every number here ~0% and hides whatever
|
|
|
|
|
# regressed. On a Nucleic box that is the *default* state of /bin/bash, so this is
|
|
|
|
|
# a refusal rather than a warning (docs/NASH_STREAM_PERF_PLAN.md §regression guard).
|
|
|
|
|
if is_nash(args.bash) and not args.allow_nash_baseline:
|
|
|
|
|
print(
|
|
|
|
|
f"error: baseline shell {args.bash} reports NUCLEIC_NASH=1 — it IS nash.\n"
|
|
|
|
|
f" Point --bash at the real bash ({DEFAULT_BASH} on a Nucleic image),\n"
|
|
|
|
|
" or pass --allow-nash-baseline if you really mean to compare nash to nash.",
|
|
|
|
|
file=sys.stderr,
|
|
|
|
|
)
|
|
|
|
|
return 2
|
|
|
|
|
|
2026-07-19 02:14:32 +00:00
|
|
|
observe_env = None
|
|
|
|
|
if args.observe_spool:
|
|
|
|
|
os.makedirs(args.observe_spool, exist_ok=True)
|
|
|
|
|
observe_env = {
|
|
|
|
|
"NUCLEIC_SHELL_SPOOL": args.observe_spool,
|
|
|
|
|
"NUCLEIC_SESSION_ID": "corpus-overhead",
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
with open(args.corpus) as f:
|
|
|
|
|
cmds = [json.loads(line)["cmd"] for line in f if line.strip()]
|
2026-07-29 04:03:22 -07:00
|
|
|
streams = stream_cases(args.stream_mb) if args.stream_mb > 0 else []
|
2026-07-19 02:14:32 +00:00
|
|
|
|
|
|
|
|
# Warm-up: one untimed pass per shell (page cache, binary load).
|
|
|
|
|
for shell in (args.bash, args.nash):
|
|
|
|
|
for cmd in cmds[:10]:
|
|
|
|
|
run_timed(shell, cmd, observe_env)
|
|
|
|
|
|
|
|
|
|
totals = {"bash": [], "nash": []}
|
2026-07-29 04:03:22 -07:00
|
|
|
stream_totals = {name: {"bash": [], "nash": []} for name, _ in streams}
|
2026-07-19 02:14:32 +00:00
|
|
|
for round_no in range(args.rounds):
|
|
|
|
|
order = [("bash", args.bash), ("nash", args.nash)]
|
|
|
|
|
if round_no % 2:
|
|
|
|
|
order.reverse()
|
|
|
|
|
for name, shell in order:
|
2026-07-29 04:03:22 -07:00
|
|
|
wall, cpu = measure(shell, cmds, observe_env)
|
|
|
|
|
totals[name].append((wall, cpu))
|
|
|
|
|
print(f"round {round_no + 1} {name}: {wall:.3f}s wall / {cpu:.3f}s cpu", flush=True)
|
|
|
|
|
for case, script in streams:
|
|
|
|
|
stream_totals[case][name].append(run_timed(shell, script, observe_env))
|
2026-07-19 02:14:32 +00:00
|
|
|
|
2026-07-29 04:03:22 -07:00
|
|
|
corpus = report(
|
|
|
|
|
f"corpus ({len(cmds)} cmds x {args.rounds} rounds)",
|
|
|
|
|
totals["bash"],
|
|
|
|
|
totals["nash"],
|
|
|
|
|
gate=args.gate,
|
|
|
|
|
)
|
|
|
|
|
stream_results = {
|
|
|
|
|
case: report(f"stream {case}", vals["bash"], vals["nash"], gate=args.stream_gate)
|
|
|
|
|
for case, vals in stream_totals.items()
|
|
|
|
|
}
|
2026-07-19 02:14:32 +00:00
|
|
|
|
2026-07-29 04:03:22 -07:00
|
|
|
corpus_ok = corpus["wall_percent"] < args.gate
|
|
|
|
|
# The stream cases are gated on CPU: the tee's cost is a copier thread, which idle
|
|
|
|
|
# hardware hides from wall-clock entirely.
|
|
|
|
|
stream_ok = all(r["cpu_percent"] < args.stream_gate for r in stream_results.values())
|
2026-07-19 02:14:32 +00:00
|
|
|
|
|
|
|
|
if args.json:
|
|
|
|
|
with open(args.json, "w") as f:
|
|
|
|
|
json.dump(
|
|
|
|
|
{
|
|
|
|
|
"cmds": len(cmds),
|
|
|
|
|
"rounds": args.rounds,
|
2026-07-29 04:03:22 -07:00
|
|
|
"totals": {k: [{"wall_s": w, "cpu_s": c} for w, c in v] for k, v in totals.items()},
|
|
|
|
|
"corpus": corpus,
|
|
|
|
|
"streams": stream_results,
|
|
|
|
|
"stream_mb": args.stream_mb,
|
2026-07-19 02:14:32 +00:00
|
|
|
"gate_percent": args.gate,
|
2026-07-29 04:03:22 -07:00
|
|
|
"stream_gate_percent": args.stream_gate,
|
2026-07-19 02:14:32 +00:00
|
|
|
"observed": bool(args.observe_spool),
|
2026-07-29 04:03:22 -07:00
|
|
|
"bash": args.bash,
|
|
|
|
|
"nash": args.nash,
|
|
|
|
|
# Kept for readers of the pre-CPU schema.
|
|
|
|
|
"bash_median_s": corpus["bash_wall_s"],
|
|
|
|
|
"nash_median_s": corpus["nash_wall_s"],
|
|
|
|
|
"overhead_percent": corpus["wall_percent"],
|
2026-07-19 02:14:32 +00:00
|
|
|
},
|
|
|
|
|
f,
|
|
|
|
|
indent=1,
|
|
|
|
|
)
|
|
|
|
|
|
2026-07-29 04:03:22 -07:00
|
|
|
print(f"\nM3 overhead gate (corpus wall <{args.gate:.1f}%): {'PASS' if corpus_ok else 'FAIL'}")
|
|
|
|
|
if streams:
|
|
|
|
|
print(
|
|
|
|
|
f"stream gate (cpu <{args.stream_gate:.1f}%): {'PASS' if stream_ok else 'FAIL'}"
|
|
|
|
|
)
|
|
|
|
|
return 0 if corpus_ok and stream_ok else 1
|
2026-07-19 02:14:32 +00:00
|
|
|
|
|
|
|
|
|
|
|
|
|
if __name__ == "__main__":
|
|
|
|
|
sys.exit(main())
|