Merge nucleic/sleek-thistle-egret-fyej into dev

This commit is contained in:
2026-07-18 05:19:31 -07:00
commit 1cce71f10b
11 changed files with 3371 additions and 0 deletions
+29
View File
@@ -0,0 +1,29 @@
# nash ↔ bash divergence log
Live record of behavioral divergences found by the replay harness
(`replay.py`, corpus in `corpus.jsonl`). Per docs/NASH.md §10.3/§11, each entry
stays open until fixed in the fork (or upstream) and re-verified by the corpus.
## Open
(none)
## Closed
### D1 — IFS word-splitting applied to literal words (fixed in fork)
- **Found**: M0 corpus run (`ifs-split`), brush-shell-v0.4.0.
- **Repro**: `IFS=,; set -- a,b,c; echo $1` → bash `a b c`, nash `a`.
- **Root cause**: brush applied IFS splitting to *literal* words, not just
expansion results: under `IFS=,`, `echo a,b,c` printed `a b c` (bash:
`a,b,c`) and `set -- a,b,c` received 3 args (bash: 1). POSIX/bash split only
the results of parameter/command/arithmetic expansion. Mechanically,
`brush-core/src/expansion.rs` had a two-state `ExpansionPiece`
(`Splittable`/`Unsplittable`) conflating "field-splittable" with
"glob-active", and literal text was marked `Splittable`.
- **Fix**: added a third piece state `LiteralText` (never field-split, glob
chars active) and produce it for unquoted literal text, the no-expansion-chars
fast path, and retained-backslash escapes (`// nash:` markers in
`expansion.rs`). Corpus back to 100% (94/94); candidate for upstreaming.
- **Regression tests**: corpus `ifs-split` (+ `glob-all`, `glob-txt`,
`cond-pattern`, `case` guard the glob/pattern side); brush compat suite.
+94
View File
@@ -0,0 +1,94 @@
{"id": "echo-basic", "cmd": "echo hello"}
{"id": "printf-args", "cmd": "printf '%s\\n' one two three"}
{"id": "pipe-sort", "cmd": "ls | sort"}
{"id": "grep-n", "cmd": "cat a.txt | grep -n line"}
{"id": "grep-count", "cmd": "grep -c ERROR logs/app.log"}
{"id": "redir-in", "cmd": "wc -l < a.txt"}
{"id": "redir-out-chain", "cmd": "head -2 data.csv > head.csv && cat head.csv"}
{"id": "csv-pipeline", "cmd": "tail -n +2 data.csv | cut -d, -f2 | sort | uniq -c | sort -rn"}
{"id": "arith", "cmd": "x=5; y=7; echo $((x*y+1))"}
{"id": "for-loop", "cmd": "for i in 1 2 3; do echo \"i=$i\"; done"}
{"id": "while-loop", "cmd": "i=0; while [ $i -lt 3 ]; do echo loop $i; i=$((i+1)); done"}
{"id": "until-loop", "cmd": "i=3; until [ $i -le 0 ]; do echo down $i; i=$((i-1)); done"}
{"id": "if-file", "cmd": "if [ -f a.txt ]; then echo yes; else echo no; fi"}
{"id": "case", "cmd": "case \"abc\" in a*) echo starts-a;; *) echo other;; esac"}
{"id": "func-basic", "cmd": "f() { echo \"fn:$1\"; }; f world"}
{"id": "func-local", "cmd": "greet() { local n=\"$1\"; echo \"hi $n\"; }; greet nash"}
{"id": "and-or", "cmd": "echo one && echo two || echo three"}
{"id": "or-fallback", "cmd": "false || echo fallback"}
{"id": "and-ok", "cmd": "true && echo ok"}
{"id": "subshell-cd", "cmd": "(cd src && pwd | xargs basename)"}
{"id": "param-default", "cmd": "VAR=hello; echo \"${VAR:-default} ${UNSET:-default}\""}
{"id": "param-strip", "cmd": "s=\"filename.tar.gz\"; echo \"${s%.gz} ${s%%.*} ${s#file}\""}
{"id": "param-replace-len", "cmd": "s=\"hello world\"; echo \"${s/world/nash} ${#s}\""}
{"id": "brace-expand", "cmd": "echo {a,b,c}{1,2}"}
{"id": "glob-txt", "cmd": "ls *.txt | sort"}
{"id": "array-basic", "cmd": "arr=(x y z); echo \"${arr[1]} ${#arr[@]} ${arr[@]}\""}
{"id": "awk-sum", "cmd": "seq 1 5 | awk '{s+=$1} END {print s}'"}
{"id": "sed-print", "cmd": "sed -n '2p' a.txt"}
{"id": "sed-file-out", "cmd": "sed 's/line/LINE/g' a.txt > upper.txt; wc -l upper.txt"}
{"id": "sort-uniq", "cmd": "sort b.txt | uniq | head -3"}
{"id": "cut-tr", "cmd": "cut -d, -f1 data.csv | tr 'a-z' 'A-Z' | tail -2"}
{"id": "find-name", "cmd": "find . -name '*.py' | sort"}
{"id": "find-exec", "cmd": "find src -type f -name '*.py' -exec wc -l {} + | sort -k2"}
{"id": "tr-wc", "cmd": "echo \"a b c\" | tr ' ' '\\n' | wc -l"}
{"id": "heredoc-expand", "cmd": "cat <<EOF\nheredoc line 1\nvalue: $((2+3))\nEOF"}
{"id": "heredoc-literal", "cmd": "cat <<'EOF'\nliteral $HOME\nEOF"}
{"id": "herestring-read", "cmd": "read first rest <<< \"alpha beta gamma\"; echo \"$first|$rest\""}
{"id": "append", "cmd": "echo hi > out.txt; echo again >> out.txt; cat out.txt"}
{"id": "redir-stderr-file", "cmd": "ls missing-dir 2> err.txt; test -s err.txt && echo captured"}
{"id": "stderr-to-pipe", "cmd": "ls missing-dir 2>&1 | grep -o 'No such file or directory'"}
{"id": "brace-group-redir", "cmd": "{ echo brace1; echo brace2; } > group.txt; cat group.txt"}
{"id": "subshell-simple", "cmd": "(echo subshell) && echo after"}
{"id": "cmdsub-nested", "cmd": "echo $(echo nested $(echo deeper))"}
{"id": "cmdsub-multiline", "cmd": "files=$(ls *.txt | sort | head -2); echo \"$files\" | wc -l"}
{"id": "cmdsub-grep", "cmd": "n=$(grep -c INFO logs/app.log); echo \"count=$n\""}
{"id": "basename-pwd", "cmd": "echo \"$(basename \"$PWD\")\""}
{"id": "set-e-false", "cmd": "set -e; false; echo unreachable"}
{"id": "set-e-true", "cmd": "set -e; true; echo reached"}
{"id": "pipefail", "cmd": "set -o pipefail; false | true; echo \"code=$?\""}
{"id": "yes-head", "cmd": "yes | head -3"}
{"id": "read-loop", "cmd": "printf 'a\\nb\\n' | while read l; do echo \"got:$l\"; done"}
{"id": "exit-code", "cmd": "echo start; exit 3"}
{"id": "child-bash", "cmd": "bash -c 'echo child'"}
{"id": "command-v", "cmd": "command -v ls > /dev/null && echo has-ls"}
{"id": "type-t", "cmd": "type -t ls"}
{"id": "cond-pattern", "cmd": "[[ \"abc\" == a?c ]] && echo pattern-match"}
{"id": "cond-numeric", "cmd": "[[ \"5\" -gt 3 ]] && echo gt"}
{"id": "test-dir", "cmd": "test -d src && echo isdir"}
{"id": "echo-n", "cmd": "echo -n \"no-newline\"; echo \"|end\""}
{"id": "printf-pad", "cmd": "printf '%05d\\n' 42"}
{"id": "ifs-split", "cmd": "IFS=,; set -- a,b,c; echo $1"}
{"id": "glob-all", "cmd": "echo *"}
{"id": "ls-dir", "cmd": "ls -1 src/"}
{"id": "mkdir-touch-find", "cmd": "mkdir -p deep/nested/dir && touch deep/nested/dir/f.txt && find deep -type f"}
{"id": "cp-diff", "cmd": "cp a.txt copy.txt && diff a.txt copy.txt && echo identical"}
{"id": "mv-ls", "cmd": "mv b.txt moved.txt && ls moved.txt"}
{"id": "rm-check", "cmd": "cp a.txt c2.txt && rm c2.txt && ls c2.txt 2>/dev/null; echo \"gone rc=$?\""}
{"id": "background-wait", "cmd": "(sleep 0.1; echo bg) & wait; echo done"}
{"id": "procsub-diff", "cmd": "diff <(echo a) <(echo a) && echo same"}
{"id": "procsub-comm", "cmd": "comm -12 <(printf '1\\n2\\n3\\n') <(printf '2\\n3\\n4\\n')"}
{"id": "export-child", "cmd": "export FOO=bar; bash -c 'echo $FOO'"}
{"id": "xargs-n1", "cmd": "printf 'a b c' | xargs -n1 echo"}
{"id": "awk-begin", "cmd": "awk 'BEGIN{printf \"%.2f\\n\", 10/3}'"}
{"id": "python-child", "cmd": "python3 -c 'print(6*7)'"}
{"id": "node-child", "cmd": "node -e 'console.log(\"js\")'"}
{"id": "tar-roundtrip", "cmd": "tar -czf arch.tgz src && tar -tzf arch.tgz | sort"}
{"id": "tee-multi", "cmd": "echo 'x' | tee t1.txt t2.txt && cat t1.txt t2.txt"}
{"id": "trap-exit", "cmd": "trap 'echo trapped' EXIT; echo body"}
{"id": "quoting-mixed", "cmd": "printf '%s\\n' \"double \\\"quoted\\\" and 'single'\""}
{"id": "escapes", "cmd": "echo \"backslash: \\\\ dollar: \\$HOME\""}
{"id": "line-continuation", "cmd": "echo one \\\ntwo"}
{"id": "func-return", "cmd": "f(){ return 4; }; f; echo \"rc=$?\""}
{"id": "shift-args", "cmd": "set -- a b c; shift; echo \"$1 $#\""}
{"id": "getopts", "cmd": "f() { while getopts 'x:y' o; do echo \"o=$o a=$OPTARG\"; done; }; f -x val -y"}
{"id": "declare-i", "cmd": "declare -i n=5; n+=2; echo $n"}
{"id": "dirname-basename", "cmd": "dirname src/main.py; basename src/main.py .py"}
{"id": "realpath", "cmd": "realpath a.txt"}
{"id": "stat-size", "cmd": "stat -c %s a.txt"}
{"id": "od-herestring", "cmd": "od -c <<< 'hi' | head -1"}
{"id": "dev-null", "cmd": "echo discarded > /dev/null; echo kept"}
{"id": "fd-dup-order", "cmd": "{ echo to-out; echo to-err >&2; } 2>/dev/null"}
{"id": "curl-sh-shape", "cmd": "printf 'echo piped-script\\n' | bash"}
{"id": "seq-paste", "cmd": "seq 1 4 | paste -sd+ | bc 2>/dev/null || seq 1 4 | paste -sd,"}
{"id": "nested-quotes-cmdsub", "cmd": "echo \"outer $(echo \"inner $(echo deepest)\")\""}
+182
View File
@@ -0,0 +1,182 @@
#!/usr/bin/env python3
"""Transcript-replay compat harness for nash (docs/NASH.md §10.3, M0 gate).
Runs every corpus command under bash and nash in identical fresh workspaces and
compares exit code, stdout, and resulting filesystem state. stderr is compared
separately and does NOT count against parity (error-message wording legitimately
differs between shells); a stderr-only difference is reported for review.
Usage: replay.py [--nash PATH] [--bash PATH] [--corpus PATH] [--json PATH]
Parity target (M0/M3 gate): >= 99%.
"""
import argparse
import hashlib
import json
import os
import shutil
import subprocess
import sys
import tempfile
FIXTURES = {
"a.txt": "first line\nsecond line\nthird line\n",
"b.txt": "banana\napple\nbanana\ncherry\napple\n",
"data.csv": "name,team,score\nalice,red,10\nbob,blue,7\ncarol,red,9\ndave,blue,7\n",
"logs/app.log": (
"2026-01-01 INFO boot ok\n"
"2026-01-01 ERROR disk full\n"
"2026-01-01 INFO retry\n"
"2026-01-01 WARN slow\n"
"2026-01-01 ERROR net down\n"
"2026-01-01 INFO done\n"
),
"src/main.py": "print('main')\n",
"src/util.py": "def add(a, b):\n return a + b\n",
"README.md": "# Sample\n",
}
TIMEOUT_S = 20
def seed(workdir):
for rel, content in FIXTURES.items():
path = os.path.join(workdir, rel)
os.makedirs(os.path.dirname(path), exist_ok=True)
with open(path, "w") as f:
f.write(content)
def snapshot(workdir):
"""Relative path -> sha256 of contents, for every file in the tree."""
out = {}
for root, _dirs, files in os.walk(workdir):
for name in files:
path = os.path.join(root, name)
rel = os.path.relpath(path, workdir)
try:
with open(path, "rb") as f:
out[rel] = hashlib.sha256(f.read()).hexdigest()
except OSError:
out[rel] = "<unreadable>"
return out
def run_one(shell, cmd, mode="-c", extra_env=None):
parent = tempfile.mkdtemp(prefix="nash-corpus-")
workdir = os.path.join(parent, "workspace")
os.makedirs(workdir)
seed(workdir)
env = {
"PATH": "/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin",
"HOME": workdir,
"LC_ALL": "C",
"LANG": "C",
"TERM": "dumb",
"SHELL": shell,
}
if extra_env:
env.update(extra_env)
argv = [shell, mode, cmd] if mode else [shell, cmd]
try:
proc = subprocess.run(
argv, cwd=workdir, env=env, capture_output=True, timeout=TIMEOUT_S
)
code, out, err = proc.returncode, proc.stdout, proc.stderr
except subprocess.TimeoutExpired:
code, out, err = "TIMEOUT", b"", b""
fs = snapshot(workdir)
shutil.rmtree(parent, ignore_errors=True)
def norm(b):
text = b.decode("utf-8", "replace")
return text.replace(workdir, "__WORK__").replace(parent, "__TMP__")
return {"code": code, "stdout": norm(out), "stderr": norm(err), "fs": fs}
def classify(b, n):
core_equal = (
b["code"] == n["code"] and b["stdout"] == n["stdout"] and b["fs"] == n["fs"]
)
if core_equal and b["stderr"] == n["stderr"]:
return "PASS"
if core_equal:
return "STDERR_ONLY"
return "DIVERGE"
def main():
ap = argparse.ArgumentParser()
ap.add_argument("--nash", default=os.environ.get("NASH_BIN", "nash"))
ap.add_argument("--bash", default="/bin/bash")
ap.add_argument(
"--corpus",
default=os.path.join(os.path.dirname(os.path.abspath(__file__)), "corpus.jsonl"),
)
ap.add_argument("--json", help="also write the full report to this path")
args = ap.parse_args()
entries = []
with open(args.corpus) as f:
for line in f:
line = line.strip()
if line:
entries.append(json.loads(line))
results = []
for entry in entries:
b = run_one(args.bash, entry["cmd"])
n = run_one(args.nash, entry["cmd"])
verdict = classify(b, n)
results.append({"id": entry["id"], "cmd": entry["cmd"], "verdict": verdict,
"bash": b, "nash": n})
marker = {"PASS": ".", "STDERR_ONLY": "s", "DIVERGE": "X"}[verdict]
print(marker, end="", flush=True)
print()
# Invocation-mode smoke tests (Nucleic exec paths use -lc, scripts, stdin).
modes = []
for mode_id, mode, cmd in [
("mode--lc", "-lc", "echo login-mode"),
("mode--c-args", "-c", "echo argv0-test"),
]:
b = run_one(args.bash, cmd, mode=mode)
n = run_one(args.nash, cmd, mode=mode)
modes.append({"id": mode_id, "verdict": classify(b, n), "bash": b, "nash": n})
passes = sum(1 for r in results if r["verdict"] in ("PASS", "STDERR_ONLY"))
stderr_only = [r for r in results if r["verdict"] == "STDERR_ONLY"]
diverges = [r for r in results if r["verdict"] == "DIVERGE"]
parity = 100.0 * passes / len(results) if results else 0.0
print(f"\ncorpus: {len(results)} parity: {passes}/{len(results)} = {parity:.1f}%")
print(f" clean pass: {len(results) - len(stderr_only) - len(diverges)}")
print(f" stderr-only: {len(stderr_only)} ({', '.join(r['id'] for r in stderr_only) or '-'})")
print(f" diverge: {len(diverges)}")
for r in diverges:
print(f"\nDIVERGE {r['id']}: {r['cmd']!r}")
print(f" bash: code={r['bash']['code']} stdout={r['bash']['stdout']!r}")
print(f" nash: code={r['nash']['code']} stdout={r['nash']['stdout']!r}")
if r["bash"]["fs"] != r["nash"]["fs"]:
keys = set(r["bash"]["fs"]) ^ set(r["nash"]["fs"])
same = {
k for k in set(r["bash"]["fs"]) & set(r["nash"]["fs"])
if r["bash"]["fs"][k] != r["nash"]["fs"][k]
}
print(f" fs delta: only-one-side={sorted(keys)} content-differs={sorted(same)}")
for m in modes:
status = "ok" if m["verdict"] in ("PASS", "STDERR_ONLY") else "DIVERGE"
print(f"mode {m['id']}: {status}")
if args.json:
with open(args.json, "w") as f:
json.dump({"parity_percent": parity, "results": results, "modes": modes}, f, indent=1)
gate = parity >= 99.0 and all(m["verdict"] != "DIVERGE" for m in modes)
print(f"\nM0 gate (>=99% parity, modes ok): {'PASS' if gate else 'FAIL'}")
return 0 if gate else 1
if __name__ == "__main__":
sys.exit(main())