Merge nucleic/dapper-feather-lynx-w4wa into dev
This commit is contained in:
Executable
+135
@@ -0,0 +1,135 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""Extract the real `host_exec` command corpus from session transcripts (docs/NASH.md §10.6).
|
||||||
|
|
||||||
|
Sweeps a directory tree of transcript captures (JSON or JSONL — the Claude stream-json capture
|
||||||
|
files and session transcript stores both work) for `host_exec` tool calls, dedupes the command
|
||||||
|
strings, auto-classifies each as replayable or `excluded:<reason>` (network / host-mutating /
|
||||||
|
signing / toolchain — exclusions are explicit, never silent), and appends NEW entries to the
|
||||||
|
host corpus, preserving anything already there (including the self-test entries the harness
|
||||||
|
requires).
|
||||||
|
|
||||||
|
Usage: extract_host_corpus.py TRANSCRIPTS_DIR [--corpus PATH] [--dry-run]
|
||||||
|
"""
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import re
|
||||||
|
import sys
|
||||||
|
|
||||||
|
EXCLUSION_PATTERNS = [
|
||||||
|
# (reason, regex over the whole command) — first hit wins; order: most-specific first.
|
||||||
|
("network", r"\b(curl|wget|git\s+(push|pull|fetch|clone)|gh\s|ssh\s|scp\s|rsync\s.*:)"),
|
||||||
|
("mutates-host", r"\b(brew\s+(install|upgrade|uninstall)|sudo\s|defaults\s+write|killall\s"
|
||||||
|
r"|launchctl\s|xcode-select\s+--install)"),
|
||||||
|
("signing-identity", r"\b(codesign|notarytool|productsign|security\s+import)"),
|
||||||
|
# Toolchain builds replay only against a real tree (--worktree), not the synthetic fixtures.
|
||||||
|
("toolchain — replay with --worktree against a real package",
|
||||||
|
r"\b(xcodebuild|swift\s+(build|test|run)|xcrun\s|make\b)"),
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
|
def classify(cmd):
|
||||||
|
for reason, pattern in EXCLUSION_PATTERNS:
|
||||||
|
if re.search(pattern, cmd):
|
||||||
|
return f"excluded:{reason}"
|
||||||
|
return "replayable"
|
||||||
|
|
||||||
|
|
||||||
|
def walk_json(node, found):
|
||||||
|
"""Recursively collect host_exec tool-call commands from any JSON shape."""
|
||||||
|
if isinstance(node, dict):
|
||||||
|
name = node.get("name", "")
|
||||||
|
if isinstance(name, str) and name.endswith("host_exec"):
|
||||||
|
command = (node.get("input") or {}).get("command")
|
||||||
|
if isinstance(command, str) and command.strip():
|
||||||
|
found.append(command.strip())
|
||||||
|
for value in node.values():
|
||||||
|
walk_json(value, found)
|
||||||
|
elif isinstance(node, list):
|
||||||
|
for value in node:
|
||||||
|
walk_json(value, found)
|
||||||
|
|
||||||
|
|
||||||
|
def commands_in_file(path):
|
||||||
|
found = []
|
||||||
|
try:
|
||||||
|
with open(path, encoding="utf-8", errors="replace") as f:
|
||||||
|
text = f.read()
|
||||||
|
except OSError:
|
||||||
|
return found
|
||||||
|
# Whole-file JSON first; else JSONL line by line.
|
||||||
|
try:
|
||||||
|
walk_json(json.loads(text), found)
|
||||||
|
return found
|
||||||
|
except ValueError:
|
||||||
|
pass
|
||||||
|
for line in text.splitlines():
|
||||||
|
line = line.strip()
|
||||||
|
if not line or "host_exec" not in line:
|
||||||
|
continue
|
||||||
|
try:
|
||||||
|
walk_json(json.loads(line), found)
|
||||||
|
except ValueError:
|
||||||
|
continue
|
||||||
|
return found
|
||||||
|
|
||||||
|
|
||||||
|
def main():
|
||||||
|
ap = argparse.ArgumentParser()
|
||||||
|
ap.add_argument("transcripts_dir")
|
||||||
|
ap.add_argument(
|
||||||
|
"--corpus",
|
||||||
|
default=os.path.join(
|
||||||
|
os.path.dirname(os.path.abspath(__file__)), "host-corpus.jsonl"),
|
||||||
|
)
|
||||||
|
ap.add_argument("--dry-run", action="store_true")
|
||||||
|
args = ap.parse_args()
|
||||||
|
|
||||||
|
commands = []
|
||||||
|
for root, _dirs, files in os.walk(args.transcripts_dir):
|
||||||
|
for name in files:
|
||||||
|
if name.endswith((".json", ".jsonl")):
|
||||||
|
commands.extend(commands_in_file(os.path.join(root, name)))
|
||||||
|
|
||||||
|
existing_cmds = set()
|
||||||
|
next_index = 1
|
||||||
|
if os.path.exists(args.corpus):
|
||||||
|
with open(args.corpus) as f:
|
||||||
|
for line in f:
|
||||||
|
line = line.strip()
|
||||||
|
if line:
|
||||||
|
entry = json.loads(line)
|
||||||
|
existing_cmds.add(entry["cmd"])
|
||||||
|
match = re.match(r"hx-(\d+)$", entry["id"])
|
||||||
|
if match:
|
||||||
|
next_index = max(next_index, int(match.group(1)) + 1)
|
||||||
|
|
||||||
|
new_entries = []
|
||||||
|
seen = set()
|
||||||
|
for cmd in commands:
|
||||||
|
if cmd in existing_cmds or cmd in seen:
|
||||||
|
continue
|
||||||
|
seen.add(cmd)
|
||||||
|
entry = {"id": f"hx-{next_index}", "cmd": cmd}
|
||||||
|
next_index += 1
|
||||||
|
klass = classify(cmd)
|
||||||
|
if klass != "replayable":
|
||||||
|
entry["class"] = klass
|
||||||
|
new_entries.append(entry)
|
||||||
|
|
||||||
|
print(f"transcripts: {args.transcripts_dir}")
|
||||||
|
print(f"host_exec commands found: {len(commands)} unique-new: {len(new_entries)}")
|
||||||
|
for entry in new_entries:
|
||||||
|
tag = entry.get("class", "replayable")
|
||||||
|
print(f" [{tag}] {entry['id']}: {entry['cmd']!r}")
|
||||||
|
if args.dry_run or not new_entries:
|
||||||
|
return
|
||||||
|
with open(args.corpus, "a") as f:
|
||||||
|
for entry in new_entries:
|
||||||
|
f.write(json.dumps(entry) + "\n")
|
||||||
|
print(f"appended {len(new_entries)} entries -> {args.corpus}")
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
@@ -0,0 +1,19 @@
|
|||||||
|
{"id": "st-word-split", "cmd": "x=\"a b\"; printf '[%s]\\n' $x", "expect": "diverge"}
|
||||||
|
{"id": "st-noglob-nomatch", "cmd": "echo *.doesnotexist", "expect": "diverge"}
|
||||||
|
{"id": "st-echo-escapes", "cmd": "echo 'a\\nb'", "expect": "diverge"}
|
||||||
|
{"id": "pipeline-grep-sort", "cmd": "grep -c ERROR logs/app.log && sort -u b.txt | head -2"}
|
||||||
|
{"id": "redirect-append", "cmd": "wc -l < a.txt > out.txt; echo done >> out.txt; cat out.txt"}
|
||||||
|
{"id": "cmdsub-quoted", "cmd": "n=$(wc -l < a.txt | tr -d ' '); echo \"lines=$n\""}
|
||||||
|
{"id": "find-sed", "cmd": "find src -name '*.py' | sort; sed -n '1p' src/util.py"}
|
||||||
|
{"id": "make-target", "cmd": "make check"}
|
||||||
|
{"id": "quoted-vars-match", "cmd": "x=\"a b\"; printf '[%s]\\n' \"$x\""}
|
||||||
|
{"id": "heredoc-write", "cmd": "cat > note.txt <<'EOF'\nhello host\nEOF\ncat note.txt"}
|
||||||
|
{"id": "exit-code-propagation", "cmd": "false; echo \"code=$?\""}
|
||||||
|
{"id": "cd-and-list", "cmd": "cd src && ls | sort"}
|
||||||
|
{"id": "arithmetic", "cmd": "i=$((2+3)); echo $i"}
|
||||||
|
{"id": "conditional-test", "cmd": "if [ -f data.csv ]; then cut -d, -f1 data.csv | tail -n +2 | sort; fi"}
|
||||||
|
{"id": "git-push", "cmd": "git push origin main", "class": "excluded:network"}
|
||||||
|
{"id": "curl-fetch", "cmd": "curl -fsSL https://example.com/artifact.tgz | tar xz", "class": "excluded:network"}
|
||||||
|
{"id": "brew-install", "cmd": "brew install jq", "class": "excluded:mutates-host"}
|
||||||
|
{"id": "swift-build-test", "cmd": "swift build && swift test", "class": "excluded:toolchain — replay with --worktree against a real package"}
|
||||||
|
{"id": "codesign-app", "cmd": "codesign --force --sign - dist/My.app", "class": "excluded:signing-identity"}
|
||||||
Executable
+213
@@ -0,0 +1,213 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""Differential replay harness for the macOS HOST surface (docs/NASH.md §10.6 — the M5.5 gate).
|
||||||
|
|
||||||
|
The host risk is *semantic divergence*, not parse failure: a command that parses under nash but
|
||||||
|
means something different under zsh (unquoted word splitting, no-match globbing) runs to
|
||||||
|
completion with different results and no fallback event. So this harness replays each corpus
|
||||||
|
command under BOTH `zsh -lc` (today's host shell) and `nash -c` (the M5.5 flip: login sourcing is
|
||||||
|
replaced by the LoginShellEnv snapshot, hence `-c`) in identical disposable workspaces, then
|
||||||
|
diffs exit code, stdout, AND the resulting file-system tree. stderr is compared separately and
|
||||||
|
never counts against parity (error wording legitimately differs).
|
||||||
|
|
||||||
|
Corpus entries (host-corpus.jsonl, one JSON object per line):
|
||||||
|
{"id": "...", "cmd": "...", # a replayable command
|
||||||
|
"class": "replayable" | "excluded:<reason>", # exclusions are REPORTED, never silent
|
||||||
|
"expect": "match" | "diverge"} # 'diverge' = harness SELF-TEST entry: a known
|
||||||
|
# zsh≠bash-family semantic difference the
|
||||||
|
# diff engine MUST catch (else the engine is
|
||||||
|
# blind and the gate is meaningless)
|
||||||
|
|
||||||
|
Workspaces: a synthetic fixture tree by default; pass --worktree PATH to clone a real project
|
||||||
|
tree instead (cp -R into the scratch dir) so toolchain commands (`swift build`, `make`) replay
|
||||||
|
against real inputs — §10.6's "disposable clone".
|
||||||
|
|
||||||
|
Verifying the harness before a darwin nash exists: `--nash /bin/bash` is a faithful stand-in for
|
||||||
|
the divergence semantics under test (bash word-splits and passes unmatched globs through exactly
|
||||||
|
as nash's brush core does; zsh does neither), so the self-test entries must report DIVERGE and
|
||||||
|
real entries' verdicts are meaningful. The real gate run uses the built nash.
|
||||||
|
|
||||||
|
Exit status: 0 iff every replayable expect-match entry matched AND every expect-diverge
|
||||||
|
self-test diverged AND at least one self-test ran.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import hashlib
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import shutil
|
||||||
|
import subprocess
|
||||||
|
import sys
|
||||||
|
import tempfile
|
||||||
|
|
||||||
|
FIXTURES = {
|
||||||
|
"a.txt": "first line\nsecond line\nthird line\n",
|
||||||
|
"b.txt": "banana\napple\nbanana\ncherry\napple\n",
|
||||||
|
"data.csv": "name,team,score\nalice,red,10\nbob,blue,7\ncarol,red,9\ndave,blue,7\n",
|
||||||
|
"logs/app.log": (
|
||||||
|
"2026-01-01 INFO boot ok\n"
|
||||||
|
"2026-01-01 ERROR disk full\n"
|
||||||
|
"2026-01-01 INFO retry\n"
|
||||||
|
"2026-01-01 WARN slow\n"
|
||||||
|
"2026-01-01 ERROR net down\n"
|
||||||
|
"2026-01-01 INFO done\n"
|
||||||
|
),
|
||||||
|
"src/main.py": "print('main')\n",
|
||||||
|
"src/util.py": "def add(a, b):\n return a + b\n",
|
||||||
|
"README.md": "# Sample\n",
|
||||||
|
"Makefile": "check:\n\t@echo make-ran\n",
|
||||||
|
}
|
||||||
|
|
||||||
|
TIMEOUT_S = 60
|
||||||
|
|
||||||
|
|
||||||
|
def seed(workdir, worktree=None):
|
||||||
|
if worktree:
|
||||||
|
shutil.copytree(worktree, workdir, symlinks=True, dirs_exist_ok=True)
|
||||||
|
return
|
||||||
|
for rel, content in FIXTURES.items():
|
||||||
|
path = os.path.join(workdir, rel)
|
||||||
|
os.makedirs(os.path.dirname(path), exist_ok=True)
|
||||||
|
with open(path, "w") as f:
|
||||||
|
f.write(content)
|
||||||
|
|
||||||
|
|
||||||
|
def snapshot(workdir):
|
||||||
|
"""Relative path -> sha256 of contents, for every file in the tree."""
|
||||||
|
out = {}
|
||||||
|
for root, _dirs, files in os.walk(workdir):
|
||||||
|
for name in files:
|
||||||
|
path = os.path.join(root, name)
|
||||||
|
rel = os.path.relpath(path, workdir)
|
||||||
|
try:
|
||||||
|
with open(path, "rb") as f:
|
||||||
|
out[rel] = hashlib.sha256(f.read()).hexdigest()
|
||||||
|
except OSError:
|
||||||
|
out[rel] = "<unreadable>"
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def run_one(shell, mode, cmd, worktree=None):
|
||||||
|
parent = tempfile.mkdtemp(prefix="nash-hostdiff-")
|
||||||
|
workdir = os.path.join(parent, "workspace")
|
||||||
|
os.makedirs(workdir, exist_ok=True)
|
||||||
|
seed(workdir, worktree)
|
||||||
|
# A pinned, minimal env on purpose: the point of §7.7.2 is that the flip must NOT depend on
|
||||||
|
# what login sourcing would add — a command that only works under `zsh -lc`'s rc-sourced env
|
||||||
|
# shows up here as a divergence to investigate, which is the gate doing its job.
|
||||||
|
env = {
|
||||||
|
"PATH": "/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin",
|
||||||
|
"HOME": workdir,
|
||||||
|
"LC_ALL": "C",
|
||||||
|
"LANG": "C",
|
||||||
|
"TERM": "dumb",
|
||||||
|
"SHELL": shell,
|
||||||
|
}
|
||||||
|
try:
|
||||||
|
proc = subprocess.run(
|
||||||
|
[shell, mode, cmd], cwd=workdir, env=env, capture_output=True, timeout=TIMEOUT_S
|
||||||
|
)
|
||||||
|
code, out, err = proc.returncode, proc.stdout, proc.stderr
|
||||||
|
except subprocess.TimeoutExpired:
|
||||||
|
code, out, err = "TIMEOUT", b"", b""
|
||||||
|
fs = snapshot(workdir)
|
||||||
|
shutil.rmtree(parent, ignore_errors=True)
|
||||||
|
|
||||||
|
def norm(b):
|
||||||
|
text = b.decode("utf-8", "replace")
|
||||||
|
return text.replace(workdir, "__WORK__").replace(parent, "__TMP__")
|
||||||
|
|
||||||
|
return {"code": code, "stdout": norm(out), "stderr": norm(err), "fs": fs}
|
||||||
|
|
||||||
|
|
||||||
|
def classify_outcome(z, n):
|
||||||
|
core_equal = (
|
||||||
|
z["code"] == n["code"] and z["stdout"] == n["stdout"] and z["fs"] == n["fs"]
|
||||||
|
)
|
||||||
|
if core_equal and z["stderr"] == n["stderr"]:
|
||||||
|
return "MATCH"
|
||||||
|
if core_equal:
|
||||||
|
return "STDERR_ONLY"
|
||||||
|
return "DIVERGE"
|
||||||
|
|
||||||
|
|
||||||
|
def main():
|
||||||
|
ap = argparse.ArgumentParser()
|
||||||
|
ap.add_argument("--nash", default=os.environ.get("NASH_BIN", "nash"),
|
||||||
|
help="nash binary (or /bin/bash as a semantics stand-in pre-nash)")
|
||||||
|
ap.add_argument("--zsh", default="/bin/zsh")
|
||||||
|
ap.add_argument(
|
||||||
|
"--corpus",
|
||||||
|
default=os.path.join(
|
||||||
|
os.path.dirname(os.path.abspath(__file__)), "host-corpus.jsonl"),
|
||||||
|
)
|
||||||
|
ap.add_argument("--worktree", help="clone this tree as the workspace (disposable copy)")
|
||||||
|
ap.add_argument("--json", help="also write the full report to this path")
|
||||||
|
args = ap.parse_args()
|
||||||
|
|
||||||
|
entries = []
|
||||||
|
with open(args.corpus) as f:
|
||||||
|
for line in f:
|
||||||
|
line = line.strip()
|
||||||
|
if line:
|
||||||
|
entries.append(json.loads(line))
|
||||||
|
|
||||||
|
excluded = [e for e in entries if e.get("class", "replayable").startswith("excluded:")]
|
||||||
|
replayable = [e for e in entries if e not in excluded]
|
||||||
|
|
||||||
|
results = []
|
||||||
|
for entry in replayable:
|
||||||
|
z = run_one(args.zsh, "-lc", entry["cmd"], args.worktree)
|
||||||
|
n = run_one(args.nash, "-c", entry["cmd"], args.worktree)
|
||||||
|
outcome = classify_outcome(z, n)
|
||||||
|
expect = entry.get("expect", "match")
|
||||||
|
if expect == "diverge":
|
||||||
|
verdict = "SELFTEST_OK" if outcome == "DIVERGE" else "SELFTEST_BLIND"
|
||||||
|
else:
|
||||||
|
verdict = {"MATCH": "PASS", "STDERR_ONLY": "STDERR_ONLY", "DIVERGE": "DIVERGE"}[outcome]
|
||||||
|
results.append({
|
||||||
|
"id": entry["id"], "cmd": entry["cmd"], "expect": expect,
|
||||||
|
"verdict": verdict, "zsh": z, "nash": n,
|
||||||
|
})
|
||||||
|
marker = {"PASS": ".", "STDERR_ONLY": "s", "DIVERGE": "X",
|
||||||
|
"SELFTEST_OK": "+", "SELFTEST_BLIND": "!"}[verdict]
|
||||||
|
print(marker, end="", flush=True)
|
||||||
|
print()
|
||||||
|
|
||||||
|
diverges = [r for r in results if r["verdict"] == "DIVERGE"]
|
||||||
|
blind = [r for r in results if r["verdict"] == "SELFTEST_BLIND"]
|
||||||
|
selftests = [r for r in results if r["expect"] == "diverge"]
|
||||||
|
gated = [r for r in results if r["expect"] == "match"]
|
||||||
|
passes = sum(1 for r in gated if r["verdict"] in ("PASS", "STDERR_ONLY"))
|
||||||
|
parity = 100.0 * passes / len(gated) if gated else 0.0
|
||||||
|
|
||||||
|
print(f"\nreplayable: {len(gated)} parity: {passes}/{len(gated)} = {parity:.1f}%"
|
||||||
|
f" self-tests: {len(selftests) - len(blind)}/{len(selftests)} caught")
|
||||||
|
# Exclusions are part of the report, never silent (§10.6).
|
||||||
|
for e in excluded:
|
||||||
|
print(f" excluded [{e['class'].split(':', 1)[1]}]: {e['id']}: {e['cmd']!r}")
|
||||||
|
for r in diverges + blind:
|
||||||
|
print(f"\n{r['verdict']} {r['id']}: {r['cmd']!r}")
|
||||||
|
print(f" zsh : code={r['zsh']['code']} stdout={r['zsh']['stdout']!r}")
|
||||||
|
print(f" nash: code={r['nash']['code']} stdout={r['nash']['stdout']!r}")
|
||||||
|
if r["zsh"]["fs"] != r["nash"]["fs"]:
|
||||||
|
only = set(r["zsh"]["fs"]) ^ set(r["nash"]["fs"])
|
||||||
|
differs = {
|
||||||
|
k for k in set(r["zsh"]["fs"]) & set(r["nash"]["fs"])
|
||||||
|
if r["zsh"]["fs"][k] != r["nash"]["fs"][k]
|
||||||
|
}
|
||||||
|
print(f" fs delta: only-one-side={sorted(only)} content-differs={sorted(differs)}")
|
||||||
|
|
||||||
|
if args.json:
|
||||||
|
with open(args.json, "w") as f:
|
||||||
|
json.dump({"parity_percent": parity, "results": results,
|
||||||
|
"excluded": excluded}, f, indent=1)
|
||||||
|
|
||||||
|
ok = not diverges and not blind and selftests
|
||||||
|
if not selftests:
|
||||||
|
print("no self-test entries ran — the corpus must keep them (they prove the diff "
|
||||||
|
"engine can see semantic divergence)")
|
||||||
|
sys.exit(0 if ok else 1)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
Reference in New Issue
Block a user