Merge nucleic/dapper-feather-lynx-w4wa into dev
This commit is contained in:
Executable
+135
@@ -0,0 +1,135 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Extract the real `host_exec` command corpus from session transcripts (docs/NASH.md §10.6).
|
||||
|
||||
Sweeps a directory tree of transcript captures (JSON or JSONL — the Claude stream-json capture
|
||||
files and session transcript stores both work) for `host_exec` tool calls, dedupes the command
|
||||
strings, auto-classifies each as replayable or `excluded:<reason>` (network / host-mutating /
|
||||
signing / toolchain — exclusions are explicit, never silent), and appends NEW entries to the
|
||||
host corpus, preserving anything already there (including the self-test entries the harness
|
||||
requires).
|
||||
|
||||
Usage: extract_host_corpus.py TRANSCRIPTS_DIR [--corpus PATH] [--dry-run]
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
|
||||
EXCLUSION_PATTERNS = [
|
||||
# (reason, regex over the whole command) — first hit wins; order: most-specific first.
|
||||
("network", r"\b(curl|wget|git\s+(push|pull|fetch|clone)|gh\s|ssh\s|scp\s|rsync\s.*:)"),
|
||||
("mutates-host", r"\b(brew\s+(install|upgrade|uninstall)|sudo\s|defaults\s+write|killall\s"
|
||||
r"|launchctl\s|xcode-select\s+--install)"),
|
||||
("signing-identity", r"\b(codesign|notarytool|productsign|security\s+import)"),
|
||||
# Toolchain builds replay only against a real tree (--worktree), not the synthetic fixtures.
|
||||
("toolchain — replay with --worktree against a real package",
|
||||
r"\b(xcodebuild|swift\s+(build|test|run)|xcrun\s|make\b)"),
|
||||
]
|
||||
|
||||
|
||||
def classify(cmd):
|
||||
for reason, pattern in EXCLUSION_PATTERNS:
|
||||
if re.search(pattern, cmd):
|
||||
return f"excluded:{reason}"
|
||||
return "replayable"
|
||||
|
||||
|
||||
def walk_json(node, found):
|
||||
"""Recursively collect host_exec tool-call commands from any JSON shape."""
|
||||
if isinstance(node, dict):
|
||||
name = node.get("name", "")
|
||||
if isinstance(name, str) and name.endswith("host_exec"):
|
||||
command = (node.get("input") or {}).get("command")
|
||||
if isinstance(command, str) and command.strip():
|
||||
found.append(command.strip())
|
||||
for value in node.values():
|
||||
walk_json(value, found)
|
||||
elif isinstance(node, list):
|
||||
for value in node:
|
||||
walk_json(value, found)
|
||||
|
||||
|
||||
def commands_in_file(path):
|
||||
found = []
|
||||
try:
|
||||
with open(path, encoding="utf-8", errors="replace") as f:
|
||||
text = f.read()
|
||||
except OSError:
|
||||
return found
|
||||
# Whole-file JSON first; else JSONL line by line.
|
||||
try:
|
||||
walk_json(json.loads(text), found)
|
||||
return found
|
||||
except ValueError:
|
||||
pass
|
||||
for line in text.splitlines():
|
||||
line = line.strip()
|
||||
if not line or "host_exec" not in line:
|
||||
continue
|
||||
try:
|
||||
walk_json(json.loads(line), found)
|
||||
except ValueError:
|
||||
continue
|
||||
return found
|
||||
|
||||
|
||||
def main():
|
||||
ap = argparse.ArgumentParser()
|
||||
ap.add_argument("transcripts_dir")
|
||||
ap.add_argument(
|
||||
"--corpus",
|
||||
default=os.path.join(
|
||||
os.path.dirname(os.path.abspath(__file__)), "host-corpus.jsonl"),
|
||||
)
|
||||
ap.add_argument("--dry-run", action="store_true")
|
||||
args = ap.parse_args()
|
||||
|
||||
commands = []
|
||||
for root, _dirs, files in os.walk(args.transcripts_dir):
|
||||
for name in files:
|
||||
if name.endswith((".json", ".jsonl")):
|
||||
commands.extend(commands_in_file(os.path.join(root, name)))
|
||||
|
||||
existing_cmds = set()
|
||||
next_index = 1
|
||||
if os.path.exists(args.corpus):
|
||||
with open(args.corpus) as f:
|
||||
for line in f:
|
||||
line = line.strip()
|
||||
if line:
|
||||
entry = json.loads(line)
|
||||
existing_cmds.add(entry["cmd"])
|
||||
match = re.match(r"hx-(\d+)$", entry["id"])
|
||||
if match:
|
||||
next_index = max(next_index, int(match.group(1)) + 1)
|
||||
|
||||
new_entries = []
|
||||
seen = set()
|
||||
for cmd in commands:
|
||||
if cmd in existing_cmds or cmd in seen:
|
||||
continue
|
||||
seen.add(cmd)
|
||||
entry = {"id": f"hx-{next_index}", "cmd": cmd}
|
||||
next_index += 1
|
||||
klass = classify(cmd)
|
||||
if klass != "replayable":
|
||||
entry["class"] = klass
|
||||
new_entries.append(entry)
|
||||
|
||||
print(f"transcripts: {args.transcripts_dir}")
|
||||
print(f"host_exec commands found: {len(commands)} unique-new: {len(new_entries)}")
|
||||
for entry in new_entries:
|
||||
tag = entry.get("class", "replayable")
|
||||
print(f" [{tag}] {entry['id']}: {entry['cmd']!r}")
|
||||
if args.dry_run or not new_entries:
|
||||
return
|
||||
with open(args.corpus, "a") as f:
|
||||
for entry in new_entries:
|
||||
f.write(json.dumps(entry) + "\n")
|
||||
print(f"appended {len(new_entries)} entries -> {args.corpus}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,19 @@
|
||||
{"id": "st-word-split", "cmd": "x=\"a b\"; printf '[%s]\\n' $x", "expect": "diverge"}
|
||||
{"id": "st-noglob-nomatch", "cmd": "echo *.doesnotexist", "expect": "diverge"}
|
||||
{"id": "st-echo-escapes", "cmd": "echo 'a\\nb'", "expect": "diverge"}
|
||||
{"id": "pipeline-grep-sort", "cmd": "grep -c ERROR logs/app.log && sort -u b.txt | head -2"}
|
||||
{"id": "redirect-append", "cmd": "wc -l < a.txt > out.txt; echo done >> out.txt; cat out.txt"}
|
||||
{"id": "cmdsub-quoted", "cmd": "n=$(wc -l < a.txt | tr -d ' '); echo \"lines=$n\""}
|
||||
{"id": "find-sed", "cmd": "find src -name '*.py' | sort; sed -n '1p' src/util.py"}
|
||||
{"id": "make-target", "cmd": "make check"}
|
||||
{"id": "quoted-vars-match", "cmd": "x=\"a b\"; printf '[%s]\\n' \"$x\""}
|
||||
{"id": "heredoc-write", "cmd": "cat > note.txt <<'EOF'\nhello host\nEOF\ncat note.txt"}
|
||||
{"id": "exit-code-propagation", "cmd": "false; echo \"code=$?\""}
|
||||
{"id": "cd-and-list", "cmd": "cd src && ls | sort"}
|
||||
{"id": "arithmetic", "cmd": "i=$((2+3)); echo $i"}
|
||||
{"id": "conditional-test", "cmd": "if [ -f data.csv ]; then cut -d, -f1 data.csv | tail -n +2 | sort; fi"}
|
||||
{"id": "git-push", "cmd": "git push origin main", "class": "excluded:network"}
|
||||
{"id": "curl-fetch", "cmd": "curl -fsSL https://example.com/artifact.tgz | tar xz", "class": "excluded:network"}
|
||||
{"id": "brew-install", "cmd": "brew install jq", "class": "excluded:mutates-host"}
|
||||
{"id": "swift-build-test", "cmd": "swift build && swift test", "class": "excluded:toolchain — replay with --worktree against a real package"}
|
||||
{"id": "codesign-app", "cmd": "codesign --force --sign - dist/My.app", "class": "excluded:signing-identity"}
|
||||
Executable
+213
@@ -0,0 +1,213 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Differential replay harness for the macOS HOST surface (docs/NASH.md §10.6 — the M5.5 gate).
|
||||
|
||||
The host risk is *semantic divergence*, not parse failure: a command that parses under nash but
|
||||
means something different under zsh (unquoted word splitting, no-match globbing) runs to
|
||||
completion with different results and no fallback event. So this harness replays each corpus
|
||||
command under BOTH `zsh -lc` (today's host shell) and `nash -c` (the M5.5 flip: login sourcing is
|
||||
replaced by the LoginShellEnv snapshot, hence `-c`) in identical disposable workspaces, then
|
||||
diffs exit code, stdout, AND the resulting file-system tree. stderr is compared separately and
|
||||
never counts against parity (error wording legitimately differs).
|
||||
|
||||
Corpus entries (host-corpus.jsonl, one JSON object per line):
|
||||
{"id": "...", "cmd": "...", # a replayable command
|
||||
"class": "replayable" | "excluded:<reason>", # exclusions are REPORTED, never silent
|
||||
"expect": "match" | "diverge"} # 'diverge' = harness SELF-TEST entry: a known
|
||||
# zsh≠bash-family semantic difference the
|
||||
# diff engine MUST catch (else the engine is
|
||||
# blind and the gate is meaningless)
|
||||
|
||||
Workspaces: a synthetic fixture tree by default; pass --worktree PATH to clone a real project
|
||||
tree instead (cp -R into the scratch dir) so toolchain commands (`swift build`, `make`) replay
|
||||
against real inputs — §10.6's "disposable clone".
|
||||
|
||||
Verifying the harness before a darwin nash exists: `--nash /bin/bash` is a faithful stand-in for
|
||||
the divergence semantics under test (bash word-splits and passes unmatched globs through exactly
|
||||
as nash's brush core does; zsh does neither), so the self-test entries must report DIVERGE and
|
||||
real entries' verdicts are meaningful. The real gate run uses the built nash.
|
||||
|
||||
Exit status: 0 iff every replayable expect-match entry matched AND every expect-diverge
|
||||
self-test diverged AND at least one self-test ran.
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import hashlib
|
||||
import json
|
||||
import os
|
||||
import shutil
|
||||
import subprocess
|
||||
import sys
|
||||
import tempfile
|
||||
|
||||
FIXTURES = {
|
||||
"a.txt": "first line\nsecond line\nthird line\n",
|
||||
"b.txt": "banana\napple\nbanana\ncherry\napple\n",
|
||||
"data.csv": "name,team,score\nalice,red,10\nbob,blue,7\ncarol,red,9\ndave,blue,7\n",
|
||||
"logs/app.log": (
|
||||
"2026-01-01 INFO boot ok\n"
|
||||
"2026-01-01 ERROR disk full\n"
|
||||
"2026-01-01 INFO retry\n"
|
||||
"2026-01-01 WARN slow\n"
|
||||
"2026-01-01 ERROR net down\n"
|
||||
"2026-01-01 INFO done\n"
|
||||
),
|
||||
"src/main.py": "print('main')\n",
|
||||
"src/util.py": "def add(a, b):\n return a + b\n",
|
||||
"README.md": "# Sample\n",
|
||||
"Makefile": "check:\n\t@echo make-ran\n",
|
||||
}
|
||||
|
||||
TIMEOUT_S = 60
|
||||
|
||||
|
||||
def seed(workdir, worktree=None):
|
||||
if worktree:
|
||||
shutil.copytree(worktree, workdir, symlinks=True, dirs_exist_ok=True)
|
||||
return
|
||||
for rel, content in FIXTURES.items():
|
||||
path = os.path.join(workdir, rel)
|
||||
os.makedirs(os.path.dirname(path), exist_ok=True)
|
||||
with open(path, "w") as f:
|
||||
f.write(content)
|
||||
|
||||
|
||||
def snapshot(workdir):
|
||||
"""Relative path -> sha256 of contents, for every file in the tree."""
|
||||
out = {}
|
||||
for root, _dirs, files in os.walk(workdir):
|
||||
for name in files:
|
||||
path = os.path.join(root, name)
|
||||
rel = os.path.relpath(path, workdir)
|
||||
try:
|
||||
with open(path, "rb") as f:
|
||||
out[rel] = hashlib.sha256(f.read()).hexdigest()
|
||||
except OSError:
|
||||
out[rel] = "<unreadable>"
|
||||
return out
|
||||
|
||||
|
||||
def run_one(shell, mode, cmd, worktree=None):
|
||||
parent = tempfile.mkdtemp(prefix="nash-hostdiff-")
|
||||
workdir = os.path.join(parent, "workspace")
|
||||
os.makedirs(workdir, exist_ok=True)
|
||||
seed(workdir, worktree)
|
||||
# A pinned, minimal env on purpose: the point of §7.7.2 is that the flip must NOT depend on
|
||||
# what login sourcing would add — a command that only works under `zsh -lc`'s rc-sourced env
|
||||
# shows up here as a divergence to investigate, which is the gate doing its job.
|
||||
env = {
|
||||
"PATH": "/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin",
|
||||
"HOME": workdir,
|
||||
"LC_ALL": "C",
|
||||
"LANG": "C",
|
||||
"TERM": "dumb",
|
||||
"SHELL": shell,
|
||||
}
|
||||
try:
|
||||
proc = subprocess.run(
|
||||
[shell, mode, cmd], cwd=workdir, env=env, capture_output=True, timeout=TIMEOUT_S
|
||||
)
|
||||
code, out, err = proc.returncode, proc.stdout, proc.stderr
|
||||
except subprocess.TimeoutExpired:
|
||||
code, out, err = "TIMEOUT", b"", b""
|
||||
fs = snapshot(workdir)
|
||||
shutil.rmtree(parent, ignore_errors=True)
|
||||
|
||||
def norm(b):
|
||||
text = b.decode("utf-8", "replace")
|
||||
return text.replace(workdir, "__WORK__").replace(parent, "__TMP__")
|
||||
|
||||
return {"code": code, "stdout": norm(out), "stderr": norm(err), "fs": fs}
|
||||
|
||||
|
||||
def classify_outcome(z, n):
|
||||
core_equal = (
|
||||
z["code"] == n["code"] and z["stdout"] == n["stdout"] and z["fs"] == n["fs"]
|
||||
)
|
||||
if core_equal and z["stderr"] == n["stderr"]:
|
||||
return "MATCH"
|
||||
if core_equal:
|
||||
return "STDERR_ONLY"
|
||||
return "DIVERGE"
|
||||
|
||||
|
||||
def main():
|
||||
ap = argparse.ArgumentParser()
|
||||
ap.add_argument("--nash", default=os.environ.get("NASH_BIN", "nash"),
|
||||
help="nash binary (or /bin/bash as a semantics stand-in pre-nash)")
|
||||
ap.add_argument("--zsh", default="/bin/zsh")
|
||||
ap.add_argument(
|
||||
"--corpus",
|
||||
default=os.path.join(
|
||||
os.path.dirname(os.path.abspath(__file__)), "host-corpus.jsonl"),
|
||||
)
|
||||
ap.add_argument("--worktree", help="clone this tree as the workspace (disposable copy)")
|
||||
ap.add_argument("--json", help="also write the full report to this path")
|
||||
args = ap.parse_args()
|
||||
|
||||
entries = []
|
||||
with open(args.corpus) as f:
|
||||
for line in f:
|
||||
line = line.strip()
|
||||
if line:
|
||||
entries.append(json.loads(line))
|
||||
|
||||
excluded = [e for e in entries if e.get("class", "replayable").startswith("excluded:")]
|
||||
replayable = [e for e in entries if e not in excluded]
|
||||
|
||||
results = []
|
||||
for entry in replayable:
|
||||
z = run_one(args.zsh, "-lc", entry["cmd"], args.worktree)
|
||||
n = run_one(args.nash, "-c", entry["cmd"], args.worktree)
|
||||
outcome = classify_outcome(z, n)
|
||||
expect = entry.get("expect", "match")
|
||||
if expect == "diverge":
|
||||
verdict = "SELFTEST_OK" if outcome == "DIVERGE" else "SELFTEST_BLIND"
|
||||
else:
|
||||
verdict = {"MATCH": "PASS", "STDERR_ONLY": "STDERR_ONLY", "DIVERGE": "DIVERGE"}[outcome]
|
||||
results.append({
|
||||
"id": entry["id"], "cmd": entry["cmd"], "expect": expect,
|
||||
"verdict": verdict, "zsh": z, "nash": n,
|
||||
})
|
||||
marker = {"PASS": ".", "STDERR_ONLY": "s", "DIVERGE": "X",
|
||||
"SELFTEST_OK": "+", "SELFTEST_BLIND": "!"}[verdict]
|
||||
print(marker, end="", flush=True)
|
||||
print()
|
||||
|
||||
diverges = [r for r in results if r["verdict"] == "DIVERGE"]
|
||||
blind = [r for r in results if r["verdict"] == "SELFTEST_BLIND"]
|
||||
selftests = [r for r in results if r["expect"] == "diverge"]
|
||||
gated = [r for r in results if r["expect"] == "match"]
|
||||
passes = sum(1 for r in gated if r["verdict"] in ("PASS", "STDERR_ONLY"))
|
||||
parity = 100.0 * passes / len(gated) if gated else 0.0
|
||||
|
||||
print(f"\nreplayable: {len(gated)} parity: {passes}/{len(gated)} = {parity:.1f}%"
|
||||
f" self-tests: {len(selftests) - len(blind)}/{len(selftests)} caught")
|
||||
# Exclusions are part of the report, never silent (§10.6).
|
||||
for e in excluded:
|
||||
print(f" excluded [{e['class'].split(':', 1)[1]}]: {e['id']}: {e['cmd']!r}")
|
||||
for r in diverges + blind:
|
||||
print(f"\n{r['verdict']} {r['id']}: {r['cmd']!r}")
|
||||
print(f" zsh : code={r['zsh']['code']} stdout={r['zsh']['stdout']!r}")
|
||||
print(f" nash: code={r['nash']['code']} stdout={r['nash']['stdout']!r}")
|
||||
if r["zsh"]["fs"] != r["nash"]["fs"]:
|
||||
only = set(r["zsh"]["fs"]) ^ set(r["nash"]["fs"])
|
||||
differs = {
|
||||
k for k in set(r["zsh"]["fs"]) & set(r["nash"]["fs"])
|
||||
if r["zsh"]["fs"][k] != r["nash"]["fs"][k]
|
||||
}
|
||||
print(f" fs delta: only-one-side={sorted(only)} content-differs={sorted(differs)}")
|
||||
|
||||
if args.json:
|
||||
with open(args.json, "w") as f:
|
||||
json.dump({"parity_percent": parity, "results": results,
|
||||
"excluded": excluded}, f, indent=1)
|
||||
|
||||
ok = not diverges and not blind and selftests
|
||||
if not selftests:
|
||||
print("no self-test entries ran — the corpus must keep them (they prove the diff "
|
||||
"engine can see semantic divergence)")
|
||||
sys.exit(0 if ok else 1)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user