Merge nucleic/plucky-north-vole-sdna into dev
This commit is contained in:
@@ -27,7 +27,8 @@ from purpose_data import DataError, SOURCE_FIELDS, canonical_json, prompt_hash,
|
||||
SCRIPT_DIR = Path(__file__).resolve().parent
|
||||
DEFAULT_INPUT = SCRIPT_DIR / ".artifacts" / "swe-chat" / "candidates.jsonl"
|
||||
DEFAULT_OUTPUT = SCRIPT_DIR / ".artifacts" / "swe-chat" / "labeled-source.jsonl"
|
||||
MODEL = "gpt-5.6-luna"
|
||||
DEFAULT_MODEL = "gpt-5.6-luna"
|
||||
DEFAULT_REASONING_EFFORT = "low"
|
||||
STATE_SCHEMA_VERSION = 1
|
||||
DEFAULT_MAX_RESPONSE_CHARS = 48_000
|
||||
FENCED_CODE_RE = re.compile(r"(?ms)^[ \t]*```[^\n]*\n.*?^[ \t]*```[ \t]*$")
|
||||
@@ -252,7 +253,7 @@ def invoke_codex(args: argparse.Namespace, batch: Sequence[Candidate]) -> list[t
|
||||
command += ["exec", "--ephemeral", "--ignore-user-config", "--ignore-rules", "--skip-git-repo-check"]
|
||||
if isolation != "external":
|
||||
command += ["--sandbox", "read-only"]
|
||||
command += ["--model", MODEL, "--config", 'model_reasoning_effort="low"', "--output-schema", str(schema_path), "--output-last-message", str(response_path), "--color", "never", "-"]
|
||||
command += ["--model", args.model, "--config", f'model_reasoning_effort="{args.reasoning_effort}"', "--output-schema", str(schema_path), "--output-last-message", str(response_path), "--color", "never", "-"]
|
||||
try:
|
||||
completed = subprocess.run(command, input=prompt, text=True, stdout=subprocess.PIPE, stderr=subprocess.PIPE, cwd=root, timeout=args.timeout_seconds, check=False)
|
||||
if completed.returncode != 0:
|
||||
@@ -345,7 +346,7 @@ def run(args: argparse.Namespace) -> dict[str, int]:
|
||||
pending, batch_size=args.batch_size, batch_chars=args.batch_chars,
|
||||
max_prompt_chars=args.max_prompt_chars, max_response_chars=args.max_response_chars,
|
||||
)
|
||||
print(f"input={len(source)} resumed={len(states)} pending={len(pending)} batches={len(batch_list)} model={MODEL}", flush=True)
|
||||
print(f"input={len(source)} resumed={len(states)} pending={len(pending)} batches={len(batch_list)} model={args.model} reasoning={args.reasoning_effort}", flush=True)
|
||||
for number, batch in enumerate(batch_list, 1):
|
||||
decisions = invoke_codex(args, batch) # type: ignore[arg-type]
|
||||
newly: list[dict[str, Any]] = []
|
||||
@@ -373,6 +374,8 @@ def build_parser() -> argparse.ArgumentParser:
|
||||
parser.add_argument("--state", type=Path)
|
||||
parser.add_argument("--audit", type=Path)
|
||||
parser.add_argument("--codex", default="codex")
|
||||
parser.add_argument("--model", default=DEFAULT_MODEL)
|
||||
parser.add_argument("--reasoning-effort", default=DEFAULT_REASONING_EFFORT)
|
||||
parser.add_argument("--codex-isolation", choices=base.CODEX_ISOLATION_CHOICES, default="auto")
|
||||
parser.add_argument("--batch-size", type=int, default=20)
|
||||
parser.add_argument("--batch-chars", type=int, default=80_000)
|
||||
@@ -393,6 +396,8 @@ def main(argv: Sequence[str] | None = None) -> int:
|
||||
args.audit = args.audit.expanduser().resolve() if args.audit else args.output.with_name(f"{args.output.stem}.audit.jsonl")
|
||||
if args.batch_size <= 0 or args.batch_chars <= 0 or args.max_prompt_chars < 1_000 or args.max_response_chars < 1_000 or args.timeout_seconds <= 0 or args.max_attempts <= 0 or (args.limit_sessions is not None and args.limit_sessions <= 0):
|
||||
parser.error("batch sizes, timeout, attempts, and --limit-sessions must be positive; prompt/response limits must be at least 1000")
|
||||
if not args.model.strip() or not args.reasoning_effort.strip():
|
||||
parser.error("--model and --reasoning-effort must be non-empty")
|
||||
try:
|
||||
metrics = run(args)
|
||||
except (DataError, OSError, ValueError, subprocess.SubprocessError) as error:
|
||||
|
||||
Reference in New Issue
Block a user