Merge nucleic/dapper-feather-lynx-w4wa into dev

This commit is contained in:
2026-07-22 04:47:03 -07:00
parent b13b39e729
commit 7443a7f431
3 changed files with 367 additions and 0 deletions
+135
View File
@@ -0,0 +1,135 @@
#!/usr/bin/env python3
"""Extract the real `host_exec` command corpus from session transcripts (docs/NASH.md §10.6).
Sweeps a directory tree of transcript captures (JSON or JSONL — the Claude stream-json capture
files and session transcript stores both work) for `host_exec` tool calls, dedupes the command
strings, auto-classifies each as replayable or `excluded:<reason>` (network / host-mutating /
signing / toolchain — exclusions are explicit, never silent), and appends NEW entries to the
host corpus, preserving anything already there (including the self-test entries the harness
requires).
Usage: extract_host_corpus.py TRANSCRIPTS_DIR [--corpus PATH] [--dry-run]
"""
import argparse
import json
import os
import re
import sys
EXCLUSION_PATTERNS = [
# (reason, regex over the whole command) — first hit wins; order: most-specific first.
("network", r"\b(curl|wget|git\s+(push|pull|fetch|clone)|gh\s|ssh\s|scp\s|rsync\s.*:)"),
("mutates-host", r"\b(brew\s+(install|upgrade|uninstall)|sudo\s|defaults\s+write|killall\s"
r"|launchctl\s|xcode-select\s+--install)"),
("signing-identity", r"\b(codesign|notarytool|productsign|security\s+import)"),
# Toolchain builds replay only against a real tree (--worktree), not the synthetic fixtures.
("toolchain — replay with --worktree against a real package",
r"\b(xcodebuild|swift\s+(build|test|run)|xcrun\s|make\b)"),
]
def classify(cmd):
for reason, pattern in EXCLUSION_PATTERNS:
if re.search(pattern, cmd):
return f"excluded:{reason}"
return "replayable"
def walk_json(node, found):
"""Recursively collect host_exec tool-call commands from any JSON shape."""
if isinstance(node, dict):
name = node.get("name", "")
if isinstance(name, str) and name.endswith("host_exec"):
command = (node.get("input") or {}).get("command")
if isinstance(command, str) and command.strip():
found.append(command.strip())
for value in node.values():
walk_json(value, found)
elif isinstance(node, list):
for value in node:
walk_json(value, found)
def commands_in_file(path):
found = []
try:
with open(path, encoding="utf-8", errors="replace") as f:
text = f.read()
except OSError:
return found
# Whole-file JSON first; else JSONL line by line.
try:
walk_json(json.loads(text), found)
return found
except ValueError:
pass
for line in text.splitlines():
line = line.strip()
if not line or "host_exec" not in line:
continue
try:
walk_json(json.loads(line), found)
except ValueError:
continue
return found
def main():
ap = argparse.ArgumentParser()
ap.add_argument("transcripts_dir")
ap.add_argument(
"--corpus",
default=os.path.join(
os.path.dirname(os.path.abspath(__file__)), "host-corpus.jsonl"),
)
ap.add_argument("--dry-run", action="store_true")
args = ap.parse_args()
commands = []
for root, _dirs, files in os.walk(args.transcripts_dir):
for name in files:
if name.endswith((".json", ".jsonl")):
commands.extend(commands_in_file(os.path.join(root, name)))
existing_cmds = set()
next_index = 1
if os.path.exists(args.corpus):
with open(args.corpus) as f:
for line in f:
line = line.strip()
if line:
entry = json.loads(line)
existing_cmds.add(entry["cmd"])
match = re.match(r"hx-(\d+)$", entry["id"])
if match:
next_index = max(next_index, int(match.group(1)) + 1)
new_entries = []
seen = set()
for cmd in commands:
if cmd in existing_cmds or cmd in seen:
continue
seen.add(cmd)
entry = {"id": f"hx-{next_index}", "cmd": cmd}
next_index += 1
klass = classify(cmd)
if klass != "replayable":
entry["class"] = klass
new_entries.append(entry)
print(f"transcripts: {args.transcripts_dir}")
print(f"host_exec commands found: {len(commands)} unique-new: {len(new_entries)}")
for entry in new_entries:
tag = entry.get("class", "replayable")
print(f" [{tag}] {entry['id']}: {entry['cmd']!r}")
if args.dry_run or not new_entries:
return
with open(args.corpus, "a") as f:
for entry in new_entries:
f.write(json.dumps(entry) + "\n")
print(f"appended {len(new_entries)} entries -> {args.corpus}")
if __name__ == "__main__":
main()
+19
View File
@@ -0,0 +1,19 @@
{"id": "st-word-split", "cmd": "x=\"a b\"; printf '[%s]\\n' $x", "expect": "diverge"}
{"id": "st-noglob-nomatch", "cmd": "echo *.doesnotexist", "expect": "diverge"}
{"id": "st-echo-escapes", "cmd": "echo 'a\\nb'", "expect": "diverge"}
{"id": "pipeline-grep-sort", "cmd": "grep -c ERROR logs/app.log && sort -u b.txt | head -2"}
{"id": "redirect-append", "cmd": "wc -l < a.txt > out.txt; echo done >> out.txt; cat out.txt"}
{"id": "cmdsub-quoted", "cmd": "n=$(wc -l < a.txt | tr -d ' '); echo \"lines=$n\""}
{"id": "find-sed", "cmd": "find src -name '*.py' | sort; sed -n '1p' src/util.py"}
{"id": "make-target", "cmd": "make check"}
{"id": "quoted-vars-match", "cmd": "x=\"a b\"; printf '[%s]\\n' \"$x\""}
{"id": "heredoc-write", "cmd": "cat > note.txt <<'EOF'\nhello host\nEOF\ncat note.txt"}
{"id": "exit-code-propagation", "cmd": "false; echo \"code=$?\""}
{"id": "cd-and-list", "cmd": "cd src && ls | sort"}
{"id": "arithmetic", "cmd": "i=$((2+3)); echo $i"}
{"id": "conditional-test", "cmd": "if [ -f data.csv ]; then cut -d, -f1 data.csv | tail -n +2 | sort; fi"}
{"id": "git-push", "cmd": "git push origin main", "class": "excluded:network"}
{"id": "curl-fetch", "cmd": "curl -fsSL https://example.com/artifact.tgz | tar xz", "class": "excluded:network"}
{"id": "brew-install", "cmd": "brew install jq", "class": "excluded:mutates-host"}
{"id": "swift-build-test", "cmd": "swift build && swift test", "class": "excluded:toolchain — replay with --worktree against a real package"}
{"id": "codesign-app", "cmd": "codesign --force --sign - dist/My.app", "class": "excluded:signing-identity"}
+213
View File
@@ -0,0 +1,213 @@
#!/usr/bin/env python3
"""Differential replay harness for the macOS HOST surface (docs/NASH.md §10.6 — the M5.5 gate).
The host risk is *semantic divergence*, not parse failure: a command that parses under nash but
means something different under zsh (unquoted word splitting, no-match globbing) runs to
completion with different results and no fallback event. So this harness replays each corpus
command under BOTH `zsh -lc` (today's host shell) and `nash -c` (the M5.5 flip: login sourcing is
replaced by the LoginShellEnv snapshot, hence `-c`) in identical disposable workspaces, then
diffs exit code, stdout, AND the resulting file-system tree. stderr is compared separately and
never counts against parity (error wording legitimately differs).
Corpus entries (host-corpus.jsonl, one JSON object per line):
{"id": "...", "cmd": "...", # a replayable command
"class": "replayable" | "excluded:<reason>", # exclusions are REPORTED, never silent
"expect": "match" | "diverge"} # 'diverge' = harness SELF-TEST entry: a known
# zsh≠bash-family semantic difference the
# diff engine MUST catch (else the engine is
# blind and the gate is meaningless)
Workspaces: a synthetic fixture tree by default; pass --worktree PATH to clone a real project
tree instead (cp -R into the scratch dir) so toolchain commands (`swift build`, `make`) replay
against real inputs — §10.6's "disposable clone".
Verifying the harness before a darwin nash exists: `--nash /bin/bash` is a faithful stand-in for
the divergence semantics under test (bash word-splits and passes unmatched globs through exactly
as nash's brush core does; zsh does neither), so the self-test entries must report DIVERGE and
real entries' verdicts are meaningful. The real gate run uses the built nash.
Exit status: 0 iff every replayable expect-match entry matched AND every expect-diverge
self-test diverged AND at least one self-test ran.
"""
import argparse
import hashlib
import json
import os
import shutil
import subprocess
import sys
import tempfile
FIXTURES = {
"a.txt": "first line\nsecond line\nthird line\n",
"b.txt": "banana\napple\nbanana\ncherry\napple\n",
"data.csv": "name,team,score\nalice,red,10\nbob,blue,7\ncarol,red,9\ndave,blue,7\n",
"logs/app.log": (
"2026-01-01 INFO boot ok\n"
"2026-01-01 ERROR disk full\n"
"2026-01-01 INFO retry\n"
"2026-01-01 WARN slow\n"
"2026-01-01 ERROR net down\n"
"2026-01-01 INFO done\n"
),
"src/main.py": "print('main')\n",
"src/util.py": "def add(a, b):\n return a + b\n",
"README.md": "# Sample\n",
"Makefile": "check:\n\t@echo make-ran\n",
}
TIMEOUT_S = 60
def seed(workdir, worktree=None):
if worktree:
shutil.copytree(worktree, workdir, symlinks=True, dirs_exist_ok=True)
return
for rel, content in FIXTURES.items():
path = os.path.join(workdir, rel)
os.makedirs(os.path.dirname(path), exist_ok=True)
with open(path, "w") as f:
f.write(content)
def snapshot(workdir):
"""Relative path -> sha256 of contents, for every file in the tree."""
out = {}
for root, _dirs, files in os.walk(workdir):
for name in files:
path = os.path.join(root, name)
rel = os.path.relpath(path, workdir)
try:
with open(path, "rb") as f:
out[rel] = hashlib.sha256(f.read()).hexdigest()
except OSError:
out[rel] = "<unreadable>"
return out
def run_one(shell, mode, cmd, worktree=None):
parent = tempfile.mkdtemp(prefix="nash-hostdiff-")
workdir = os.path.join(parent, "workspace")
os.makedirs(workdir, exist_ok=True)
seed(workdir, worktree)
# A pinned, minimal env on purpose: the point of §7.7.2 is that the flip must NOT depend on
# what login sourcing would add — a command that only works under `zsh -lc`'s rc-sourced env
# shows up here as a divergence to investigate, which is the gate doing its job.
env = {
"PATH": "/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin",
"HOME": workdir,
"LC_ALL": "C",
"LANG": "C",
"TERM": "dumb",
"SHELL": shell,
}
try:
proc = subprocess.run(
[shell, mode, cmd], cwd=workdir, env=env, capture_output=True, timeout=TIMEOUT_S
)
code, out, err = proc.returncode, proc.stdout, proc.stderr
except subprocess.TimeoutExpired:
code, out, err = "TIMEOUT", b"", b""
fs = snapshot(workdir)
shutil.rmtree(parent, ignore_errors=True)
def norm(b):
text = b.decode("utf-8", "replace")
return text.replace(workdir, "__WORK__").replace(parent, "__TMP__")
return {"code": code, "stdout": norm(out), "stderr": norm(err), "fs": fs}
def classify_outcome(z, n):
core_equal = (
z["code"] == n["code"] and z["stdout"] == n["stdout"] and z["fs"] == n["fs"]
)
if core_equal and z["stderr"] == n["stderr"]:
return "MATCH"
if core_equal:
return "STDERR_ONLY"
return "DIVERGE"
def main():
ap = argparse.ArgumentParser()
ap.add_argument("--nash", default=os.environ.get("NASH_BIN", "nash"),
help="nash binary (or /bin/bash as a semantics stand-in pre-nash)")
ap.add_argument("--zsh", default="/bin/zsh")
ap.add_argument(
"--corpus",
default=os.path.join(
os.path.dirname(os.path.abspath(__file__)), "host-corpus.jsonl"),
)
ap.add_argument("--worktree", help="clone this tree as the workspace (disposable copy)")
ap.add_argument("--json", help="also write the full report to this path")
args = ap.parse_args()
entries = []
with open(args.corpus) as f:
for line in f:
line = line.strip()
if line:
entries.append(json.loads(line))
excluded = [e for e in entries if e.get("class", "replayable").startswith("excluded:")]
replayable = [e for e in entries if e not in excluded]
results = []
for entry in replayable:
z = run_one(args.zsh, "-lc", entry["cmd"], args.worktree)
n = run_one(args.nash, "-c", entry["cmd"], args.worktree)
outcome = classify_outcome(z, n)
expect = entry.get("expect", "match")
if expect == "diverge":
verdict = "SELFTEST_OK" if outcome == "DIVERGE" else "SELFTEST_BLIND"
else:
verdict = {"MATCH": "PASS", "STDERR_ONLY": "STDERR_ONLY", "DIVERGE": "DIVERGE"}[outcome]
results.append({
"id": entry["id"], "cmd": entry["cmd"], "expect": expect,
"verdict": verdict, "zsh": z, "nash": n,
})
marker = {"PASS": ".", "STDERR_ONLY": "s", "DIVERGE": "X",
"SELFTEST_OK": "+", "SELFTEST_BLIND": "!"}[verdict]
print(marker, end="", flush=True)
print()
diverges = [r for r in results if r["verdict"] == "DIVERGE"]
blind = [r for r in results if r["verdict"] == "SELFTEST_BLIND"]
selftests = [r for r in results if r["expect"] == "diverge"]
gated = [r for r in results if r["expect"] == "match"]
passes = sum(1 for r in gated if r["verdict"] in ("PASS", "STDERR_ONLY"))
parity = 100.0 * passes / len(gated) if gated else 0.0
print(f"\nreplayable: {len(gated)} parity: {passes}/{len(gated)} = {parity:.1f}%"
f" self-tests: {len(selftests) - len(blind)}/{len(selftests)} caught")
# Exclusions are part of the report, never silent (§10.6).
for e in excluded:
print(f" excluded [{e['class'].split(':', 1)[1]}]: {e['id']}: {e['cmd']!r}")
for r in diverges + blind:
print(f"\n{r['verdict']} {r['id']}: {r['cmd']!r}")
print(f" zsh : code={r['zsh']['code']} stdout={r['zsh']['stdout']!r}")
print(f" nash: code={r['nash']['code']} stdout={r['nash']['stdout']!r}")
if r["zsh"]["fs"] != r["nash"]["fs"]:
only = set(r["zsh"]["fs"]) ^ set(r["nash"]["fs"])
differs = {
k for k in set(r["zsh"]["fs"]) & set(r["nash"]["fs"])
if r["zsh"]["fs"][k] != r["nash"]["fs"][k]
}
print(f" fs delta: only-one-side={sorted(only)} content-differs={sorted(differs)}")
if args.json:
with open(args.json, "w") as f:
json.dump({"parity_percent": parity, "results": results,
"excluded": excluded}, f, indent=1)
ok = not diverges and not blind and selftests
if not selftests:
print("no self-test entries ran — the corpus must keep them (they prove the diff "
"engine can see semantic divergence)")
sys.exit(0 if ok else 1)
if __name__ == "__main__":
main()