diff --git a/benchmarks/twin2silicon/README.md b/benchmarks/twin2silicon/README.md new file mode 100644 index 0000000..c4479ca --- /dev/null +++ b/benchmarks/twin2silicon/README.md @@ -0,0 +1,56 @@ +# Cross-runtime HIL smoke comparison + +This is a runtime smoke comparison, not a leaderboard. It runs OpenCode, Codex +CLI, and Claude Code against the same public firmware task, then sends each +completed candidate to the existing physical HIL oracle. + +## Interpretation + +Native models are intentionally not overridden: each runtime uses its installed +default. A raw-model comparison belongs in the separate OpenCode model matrix, +not this cross-runtime smoke test. A runtime may report token usage but +subscription cost is unknown unless that runtime explicitly provides a cost; +the harness never estimates subscription cost. + +The hardware command flashes the connected board. Treat it as a destructive, +opt-in operation and identify the UART and JTAG device deliberately. One smoke +run is useful only as operational evidence. Any publishable result needs +multiple tasks and repeated fresh trials before drawing a conclusion. + +## Commands + +Run the offline contract suite: + +```bash +npm run test:runtime-smoke:offline +``` + +Run a connected-board comparison only with explicit hardware and output +locations. Existing authenticated runtime sessions are used as-is. +The trial instructions are noninteractive: this command authorizes inspection, +the smallest in-scope repair, and compilation without a confirmation prompt. +Before any flash, the harness checks that the selected UART's PlatformIO device +entry reports the supplied JTAG serial. + +```bash +LABWIRED_HIL=1 \ +LABWIRED_UART_DEVICE=/dev/cu.usbmodem11101 \ +LABWIRED_JTAG_SERIAL=3C:0F:02:DF:EC:F8 \ +LABWIRED_OPENOCD="$HOME/.platformio/packages/tool-openocd-esp32/bin/openocd" \ +LABWIRED_MATRIX_OUTPUT=/Volumes/LabWired/hil-runs/runtime-smoke-$(date +%Y%m%d-%H%M%S) \ +npm run test:runtime-smoke:hardware +``` + +The entry point checks OpenCode, Codex CLI, Claude Code, PlatformIO, and +OpenOCD and prints their versions before running; operators may capture stdout +with the smoke evidence. Temporary data is placed on +`/Volumes/LabWired` when that volume is available. It does not configure +runtime accounts; authenticate each native runtime before invoking it. + +## Evidence + +`LABWIRED_MATRIX_OUTPUT` must be a new directory. The matrix writes +`matrix.json` after each runtime, plus per-runtime trial directories under +`trials/`. Completed candidates contain runtime logs, `agent-result.json`, +and `usage.json`; HIL evidence, including `run.json`, is stored in that +trial's `hil/` directory. Preserve these records when reporting a smoke run. diff --git a/benchmarks/twin2silicon/hil/__init__.py b/benchmarks/twin2silicon/hil/__init__.py new file mode 100644 index 0000000..7411ae8 --- /dev/null +++ b/benchmarks/twin2silicon/hil/__init__.py @@ -0,0 +1 @@ +"""Reusable hardware-in-the-loop benchmark primitives.""" diff --git a/benchmarks/twin2silicon/hil/esp32s3.py b/benchmarks/twin2silicon/hil/esp32s3.py new file mode 100644 index 0000000..7ec2378 --- /dev/null +++ b/benchmarks/twin2silicon/hil/esp32s3.py @@ -0,0 +1,456 @@ +"""Offline-testable ESP32-S3 HIL evidence collection primitives.""" + +from dataclasses import dataclass +import fcntl +import hashlib +import json +import math +import os +from pathlib import Path +import re +import select +import termios +import time +import tty +from typing import Callable, Literal, Mapping, Optional, Sequence + +from .process import run_command +from .results import CommandResult, PathLike + + +_NAME = re.compile(r"^[A-Za-z_][A-Za-z0-9_]*$") +_SERIAL = re.compile(r"^[A-Za-z0-9_.:/+-]+$") +_MARKER = re.compile(r"^@@REG ([A-Za-z_][A-Za-z0-9_]*) (0x[0-9A-Fa-f]{8})$") +_VALUE = re.compile(r"^(0x[0-9A-Fa-f]{8}): ((?:0x)?[0-9A-Fa-f]{8})$") + + +def _uint32(value: object, field: str) -> int: + if isinstance(value, bool): + raise TypeError(f"{field} must be an integer or hexadecimal string") + if isinstance(value, str): + if re.fullmatch(r"0x[0-9A-Fa-f]+", value) is None: + raise ValueError(f"invalid hexadecimal {field}") + parsed = int(value, 16) + elif isinstance(value, int): + parsed = value + else: + raise TypeError(f"{field} must be an integer or hexadecimal string") + if parsed < 0 or parsed > 0xFFFFFFFF: + raise ValueError(f"{field} is outside uint32 bounds") + return parsed + + +@dataclass(frozen=True) +class RegisterAssertion: + name: str + address: int + mask: int + expected: int + + def __post_init__(self) -> None: + if not _NAME.fullmatch(self.name): + raise ValueError("register assertion name is unsafe") + for field in ("address", "mask", "expected"): + value = getattr(self, field) + if isinstance(value, bool) or not isinstance(value, int) or not 0 <= value <= 0xFFFFFFFF: + raise ValueError(f"{field} is outside uint32 bounds") + if self.address % 4: + raise ValueError("register address must be 32-bit aligned") + if self.expected & ~self.mask: + raise ValueError("expected value contains bits outside mask") + + @classmethod + def from_json(cls, record: Mapping[str, object]) -> "RegisterAssertion": + name = record.get("name") + if not isinstance(name, str): + raise TypeError("register assertion name must be a string") + return cls(name, _uint32(record.get("address"), "address"), + _uint32(record.get("mask"), "mask"), + _uint32(record.get("expected"), "expected")) + + +@dataclass(frozen=True) +class Esp32S3Config: + uart_baud: int + uart_ready_prefix: str + uart_timeout_seconds: float + identity_command: tuple[str, ...] + identity_expected_board: str + identity_timeout_seconds: float + flash_target: str + flash_artifact: str + flash_timeout_seconds: float + openocd_board_config: str + openocd_startup_timeout_seconds: float + openocd_command_timeout_seconds: float + platformio_project_dir: Optional[str] + platformio_environment: Optional[str] + assertions: tuple[RegisterAssertion, ...] + + @classmethod + def from_oracle(cls, oracle: Mapping[str, object]) -> "Esp32S3Config": + uart = oracle.get("uart") + identity = oracle.get("identity") + flash = oracle.get("flash") + openocd = oracle.get("openocd") + platformio = oracle.get("platformio") + records = oracle.get("register_assertions") + if (not isinstance(uart, Mapping) or not isinstance(identity, Mapping) + or not isinstance(flash, Mapping) or not isinstance(openocd, Mapping) + or not isinstance(records, list)): + raise TypeError("oracle phase mappings and register_assertions are required") + if platformio is not None and not isinstance(platformio, Mapping): + raise TypeError("platformio must be a mapping") + assertions = tuple(RegisterAssertion.from_json(record) for record in records) + if not assertions: + raise ValueError("at least one register assertion is required") + names = [item.name for item in assertions] + addresses = [item.address for item in assertions] + if len(names) != len(set(names)) or len(addresses) != len(set(addresses)): + raise ValueError("register assertion names and addresses must be unique") + baud = _positive_int(uart.get("baud"), "uart baud") + command = identity.get("command") + if not isinstance(command, list) or not command: + raise ValueError("identity command must be a nonempty list") + normalized_command = tuple(_safe_string(item, "identity command item") for item in command) + project_dir = environment = None + if platformio is not None: + project_dir = _safe_string(platformio.get("project_dir"), "platformio project_dir") + environment = _safe_string(platformio.get("environment"), "platformio environment") + return cls( + baud, + _safe_string(uart.get("ready_prefix"), "uart ready_prefix"), + _nonnegative_float(uart.get("timeout_seconds"), "uart timeout"), + normalized_command, + _safe_string(identity.get("expected_board"), "identity expected_board"), + _nonnegative_float(identity.get("timeout_seconds"), "identity timeout"), + _safe_string(flash.get("target"), "flash target"), + _safe_string(flash.get("artifact"), "flash artifact"), + _nonnegative_float(flash.get("timeout_seconds"), "flash timeout"), + _safe_string(openocd.get("board_config"), "openocd board_config"), + _nonnegative_float(openocd.get("startup_timeout_seconds"), "openocd startup timeout"), + _nonnegative_float(openocd.get("command_timeout_seconds"), "openocd command timeout"), + project_dir, + environment, + assertions, + ) + + +def _positive_int(value: object, field: str) -> int: + if isinstance(value, bool) or not isinstance(value, int) or value <= 0: + raise ValueError(f"{field} must be positive") + return value + + +def _nonnegative_float(value: object, field: str) -> float: + if (isinstance(value, bool) or not isinstance(value, (int, float)) + or not math.isfinite(value) or value < 0): + raise ValueError(f"{field} must be finite and nonnegative") + return float(value) + + +def _safe_string(value: object, field: str) -> str: + if not isinstance(value, str): + raise TypeError(f"{field} must be a string") + if not value or any(character in value for character in "\r\n\x00"): + raise ValueError(f"{field} must be a nonempty safe string") + return value + + +class BoardLockTimeout(TimeoutError): + pass + + +class BoardLock: + def __init__(self, directory: PathLike, identity: str, *, timeout_seconds: float, + poll_interval_seconds: float = 0.01) -> None: + if (not identity or not math.isfinite(timeout_seconds) or timeout_seconds < 0 + or not math.isfinite(poll_interval_seconds) or poll_interval_seconds <= 0): + raise ValueError("invalid board lock parameters") + safe = re.sub(r"[^A-Za-z0-9_.-]", "_", identity)[:48] or "board" + digest = hashlib.sha256(identity.encode()).hexdigest()[:12] + self.path = Path(directory) / f"{safe}-{digest}.lock" + self.identity = identity + self.timeout_seconds = timeout_seconds + self.poll_interval_seconds = poll_interval_seconds + self._file = None + + def acquire(self) -> "BoardLock": + self.path.parent.mkdir(parents=True, exist_ok=True) + held = open(self.path, "a+", encoding="utf-8") + deadline = time.monotonic() + self.timeout_seconds + while True: + try: + fcntl.flock(held.fileno(), fcntl.LOCK_EX | fcntl.LOCK_NB) + break + except BlockingIOError: + remaining = deadline - time.monotonic() + if remaining <= 0: + held.close() + raise BoardLockTimeout(f"board {self.identity!r} is locked") + time.sleep(min(self.poll_interval_seconds, remaining)) + self._file = held + try: + held.seek(0) + held.truncate() + json.dump({"identity": self.identity, "pid": os.getpid(), "acquired_monotonic": time.monotonic()}, held) + held.flush() + os.fsync(held.fileno()) + except BaseException: + self.release() + raise + return self + + def release(self) -> None: + if self._file is not None: + held, self._file = self._file, None + try: + fcntl.flock(held.fileno(), fcntl.LOCK_UN) + finally: + held.close() + + def __enter__(self) -> "BoardLock": + return self.acquire() + + def __exit__(self, exc_type, exc, traceback) -> None: + self.release() + + +@dataclass(frozen=True) +class PhaseResult: + status: Literal["pass", "hardware_fail", "infrastructure_error"] + category: Optional[str] = None + detail: Optional[str] = None + command_result: Optional[CommandResult] = None + + +Runner = Callable[..., CommandResult] + + +def validate_identity(command: Sequence[os.PathLike[str] | str], expected_serial: str, *, + cwd: PathLike, evidence_dir: PathLike, timeout_seconds: float, + runner: Runner = run_command) -> PhaseResult: + evidence = Path(evidence_dir) + try: + result = runner(command, cwd=cwd, stdout_path=evidence / "identity.stdout.log", + stderr_path=evidence / "identity.stderr.log", timeout_seconds=timeout_seconds) + except OSError as error: + return PhaseResult("infrastructure_error", "board_identity", str(error)) + if result.cleanup_error or result.timed_out or result.returncode != 0: + return PhaseResult("infrastructure_error", "board_identity", "identity command failed", result) + try: + lines = [line.strip() for line in Path(result.stdout_path).read_text(encoding="utf-8").splitlines() + if line.strip()] + except (OSError, UnicodeError) as error: + return PhaseResult("infrastructure_error", "board_identity", str(error), result) + if lines != [expected_serial]: + return PhaseResult("infrastructure_error", "board_identity", + f"expected exactly [{expected_serial!r}], observed {lines!r}", result) + return PhaseResult("pass", command_result=result) + + +def flash_firmware(command: Sequence[os.PathLike[str] | str], *, cwd: PathLike, + evidence_dir: PathLike, timeout_seconds: float, identity_validated: bool, + runner: Runner = run_command) -> PhaseResult: + if not identity_validated: + raise ValueError("board identity must be validated before flashing") + evidence = Path(evidence_dir) + try: + result = runner(command, cwd=cwd, stdout_path=evidence / "flash.stdout.log", + stderr_path=evidence / "flash.stderr.log", timeout_seconds=timeout_seconds) + except OSError as error: + return PhaseResult("infrastructure_error", "flash_infrastructure", str(error)) + if result.cleanup_error or result.timed_out: + return PhaseResult("infrastructure_error", "flash_infrastructure", "flash did not exit cleanly", result) + if result.returncode: + return PhaseResult("hardware_fail", "flash", "candidate firmware flash failed", result) + return PhaseResult("pass", command_result=result) + + +@dataclass(frozen=True) +class UartResult: + matched: bool + bytes_captured: int + timed_out: bool + termination_reason: Literal["matched", "timeout", "max_bytes"] + + +def capture_uart_nonce(device: PathLike, baud: int, nonce: str, timeout_seconds: float, + log: PathLike, *, max_bytes: int = 65536) -> UartResult: + fd = os.open(device, os.O_RDWR | os.O_NOCTTY | os.O_NONBLOCK) + try: + speeds = {9600: termios.B9600, 115200: termios.B115200} + if baud not in speeds: + raise ValueError(f"unsupported UART baud: {baud}") + if not math.isfinite(timeout_seconds) or timeout_seconds < 0 or max_bytes <= 0: + raise ValueError("invalid UART bounds") + attrs = termios.tcgetattr(fd) + tty.setraw(fd, termios.TCSANOW) + attrs = termios.tcgetattr(fd) + attrs[4] = attrs[5] = speeds[baud] + attrs[2] = (attrs[2] & ~(termios.CSIZE | termios.PARENB | termios.CSTOPB)) | termios.CS8 | termios.CLOCAL | termios.CREAD + termios.tcsetattr(fd, termios.TCSANOW, attrs) + deadline = time.monotonic() + timeout_seconds + captured = bytearray() + pending = bytearray() + matched = False + expected = f"LABWIRED_READY:{nonce}".encode() + while not matched and len(captured) < max_bytes: + remaining = deadline - time.monotonic() + if remaining <= 0: + break + readable, _, _ = select.select([fd], [], [], remaining) + if not readable: + break + try: + chunk = os.read(fd, min(4096, max_bytes - len(captured))) + except BlockingIOError: + continue + if not chunk: + continue + captured.extend(chunk) + pending.extend(chunk) + while b"\n" in pending: + line, _, remainder = pending.partition(b"\n") + pending = bytearray(remainder) + if line.endswith(b"\r"): + line = line[:-1] + if line == expected: + matched = True + break + Path(log).parent.mkdir(parents=True, exist_ok=True) + Path(log).write_bytes(captured) + reason: Literal["matched", "timeout", "max_bytes"] + if matched: + reason = "matched" + elif len(captured) >= max_bytes: + reason = "max_bytes" + else: + reason = "timeout" + return UartResult(matched, len(captured), reason == "timeout", reason) + finally: + os.close(fd) + + +def build_openocd_command(executable: str, config: str, adapter_serial: str, + assertions: Sequence[RegisterAssertion]) -> list[str]: + if not assertions: + raise ValueError("at least one register assertion is required") + if not _SERIAL.fullmatch(adapter_serial): + raise ValueError("unsafe adapter serial") + if any(character in config for character in "\r\n\x00"): + raise ValueError("unsafe OpenOCD config") + commands = [f"adapter serial {adapter_serial}", "adapter speed 4000", "init", "reset run", + "sleep 750", "halt"] + for assertion in assertions: + commands.extend((f'echo "@@REG {assertion.name} 0x{assertion.address:08x}"', + f'echo [capture "mdw 0x{assertion.address:08x} 1"]')) + commands.append("exit") + return [executable, "-f", config, "-c", "; ".join(commands)] + + +def parse_openocd_registers(text: str, requested: Sequence[RegisterAssertion]) -> dict[str, int]: + if not requested: + raise ValueError("at least one register assertion is required") + by_name = {item.name: item for item in requested} + if len(by_name) != len(requested): + raise ValueError("requested assertion names are not unique") + observed: dict[str, int] = {} + lines = text.splitlines() + index = 0 + while index < len(lines): + stripped = lines[index].strip() + marker = _MARKER.fullmatch(stripped) + if marker is None: + if (stripped.startswith("@@REG") or _VALUE.fullmatch(stripped) + or stripped.lower().startswith("error:")): + raise ValueError("unpaired or malformed register observation") + index += 1 + continue + name, address_text = marker.groups() + assertion = by_name.get(name) + if assertion is None or name in observed or int(address_text, 16) != assertion.address: + raise ValueError("invalid or duplicate register marker") + if index + 1 >= len(lines): + raise ValueError("register marker has no observation") + value_line = _VALUE.fullmatch(lines[index + 1].strip()) + if value_line is None or int(value_line.group(1), 16) != assertion.address: + raise ValueError("register observation is malformed or has wrong address") + observed[name] = int(value_line.group(2), 16) + index += 2 + if set(observed) != set(by_name): + raise ValueError("missing requested register observations") + return observed + + +@dataclass(frozen=True) +class RegisterObservation: + name: str + address: int + value: int + mask: int + expected: int + passed: bool + + +@dataclass(frozen=True) +class RegisterEvaluation: + status: Literal["pass", "hardware_fail"] + observations: tuple[RegisterObservation, ...] + + +def evaluate_registers(observed: Mapping[str, int], assertions: Sequence[RegisterAssertion]) -> RegisterEvaluation: + if not assertions: + raise ValueError("at least one register assertion is required") + if set(observed) != {item.name for item in assertions}: + raise ValueError("observed registers do not exactly match assertions") + for name, value in observed.items(): + if isinstance(value, bool) or not isinstance(value, int): + raise TypeError(f"observed register {name} must be an integer") + if value < 0 or value > 0xFFFFFFFF: + raise ValueError(f"observed register {name} is outside uint32 bounds") + results = tuple(RegisterObservation(item.name, item.address, observed[item.name], item.mask, + item.expected, (observed[item.name] & item.mask) == item.expected) + for item in assertions) + return RegisterEvaluation("pass" if all(item.passed for item in results) else "hardware_fail", results) + + +@dataclass(frozen=True) +class OpenOcdResult: + status: Literal["pass", "hardware_fail", "infrastructure_error"] + category: Optional[Literal["openocd"]] + detail: Optional[str] + observed: Optional[dict[str, int]] + evaluation: Optional[RegisterEvaluation] + command_result: Optional[CommandResult] + + +def read_registers(executable: str, config: str, adapter_serial: str, + assertions: Sequence[RegisterAssertion], *, cwd: PathLike, + evidence_dir: PathLike, timeout_seconds: float, + runner: Runner = run_command) -> OpenOcdResult: + if not math.isfinite(timeout_seconds) or timeout_seconds < 0: + raise ValueError("OpenOCD timeout must be finite and nonnegative") + command = build_openocd_command(executable, config, adapter_serial, assertions) + evidence = Path(evidence_dir) + try: + result = runner(command, cwd=cwd, stdout_path=evidence / "openocd.stdout.log", + stderr_path=evidence / "openocd.stderr.log", timeout_seconds=timeout_seconds) + except OSError as error: + return OpenOcdResult("infrastructure_error", "openocd", str(error), None, None, None) + if result.cleanup_error: + return OpenOcdResult("infrastructure_error", "openocd", result.cleanup_error, + None, None, result) + if result.timed_out: + return OpenOcdResult("infrastructure_error", "openocd", "OpenOCD timed out", + None, None, result) + if result.returncode: + return OpenOcdResult("infrastructure_error", "openocd", + f"OpenOCD exited {result.returncode}", None, None, result) + try: + transcript = Path(result.stderr_path).read_text(encoding="utf-8") + observed = parse_openocd_registers(transcript, assertions) + evaluation = evaluate_registers(observed, assertions) + except (OSError, UnicodeError, ValueError) as error: + return OpenOcdResult("infrastructure_error", "openocd", str(error), None, None, result) + return OpenOcdResult(evaluation.status, None, None, observed, evaluation, result) diff --git a/benchmarks/twin2silicon/hil/process.py b/benchmarks/twin2silicon/hil/process.py new file mode 100644 index 0000000..63a1c2a --- /dev/null +++ b/benchmarks/twin2silicon/hil/process.py @@ -0,0 +1,105 @@ +"""Bounded subprocess execution with persistent evidence.""" + +from datetime import datetime, timezone +import os +from pathlib import Path +import signal +import subprocess +import time +from typing import Sequence, Union + +from .results import CommandResult, PathLike + + +def _utc_now() -> str: + return datetime.now(timezone.utc).isoformat().replace("+00:00", "Z") + + +def _process_group_exists(process_group_id: int) -> bool: + try: + os.killpg(process_group_id, 0) + except ProcessLookupError: + return False + except PermissionError: + return True + return True + + +def _wait_for_process_group_exit(process_group_id: int, timeout_seconds: float) -> bool: + deadline = time.monotonic() + timeout_seconds + while _process_group_exists(process_group_id): + remaining = deadline - time.monotonic() + if remaining <= 0: + return False + time.sleep(min(0.01, remaining)) + return True + + +def run_command( + command: Sequence[Union[str, os.PathLike[str]]], + *, + cwd: PathLike, + stdout_path: PathLike, + stderr_path: PathLike, + timeout_seconds: float, +) -> CommandResult: + normalized_command = tuple(os.fspath(part) for part in command) + normalized_cwd = str(Path(cwd).resolve()) + normalized_stdout = str(Path(stdout_path).resolve()) + normalized_stderr = str(Path(stderr_path).resolve()) + Path(normalized_stdout).parent.mkdir(parents=True, exist_ok=True) + Path(normalized_stderr).parent.mkdir(parents=True, exist_ok=True) + + started_at = _utc_now() + started_monotonic = time.monotonic() + timed_out = False + cleanup_error = None + with open(normalized_stdout, "wb") as stdout, open(normalized_stderr, "wb") as stderr: + process = subprocess.Popen( + normalized_command, + cwd=normalized_cwd, + stdout=stdout, + stderr=stderr, + start_new_session=True, + ) + try: + process.communicate(timeout=timeout_seconds) + except subprocess.TimeoutExpired: + timed_out = True + try: + os.killpg(process.pid, signal.SIGTERM) + except (PermissionError, ProcessLookupError): + pass + if not _wait_for_process_group_exit(process.pid, 0.5): + try: + os.killpg(process.pid, signal.SIGKILL) + except (PermissionError, ProcessLookupError): + pass + try: + process.wait(timeout=0.5) + except subprocess.TimeoutExpired: + try: + os.killpg(process.pid, signal.SIGKILL) + except (PermissionError, ProcessLookupError): + pass + try: + process.wait(timeout=0.5) + except subprocess.TimeoutExpired: + cleanup_error = "process_group_did_not_exit" + group_exited = _wait_for_process_group_exit(process.pid, 0.5) + if not group_exited and cleanup_error is None: + cleanup_error = "process_group_did_not_exit" + + ended_at = _utc_now() + return CommandResult( + command=normalized_command, + cwd=normalized_cwd, + returncode=process.returncode if process.returncode is not None else -signal.SIGKILL, + timed_out=timed_out, + started_at_utc=started_at, + ended_at_utc=ended_at, + duration_seconds=time.monotonic() - started_monotonic, + stdout_path=normalized_stdout, + stderr_path=normalized_stderr, + cleanup_error=cleanup_error, + ) diff --git a/benchmarks/twin2silicon/hil/results.py b/benchmarks/twin2silicon/hil/results.py new file mode 100644 index 0000000..f5fe117 --- /dev/null +++ b/benchmarks/twin2silicon/hil/results.py @@ -0,0 +1,90 @@ +"""Result contracts and durable result-file helpers.""" + +from dataclasses import dataclass +import hashlib +import json +import os +from pathlib import Path +import tempfile +from typing import Any, Literal, Optional, Union + + +PathLike = Union[str, os.PathLike[str]] +ModelStatus = Literal["pass", "fail", "not_run"] +CompileStatus = Literal["pass", "fail", "not_run"] +SimulatorStatus = Literal["pass", "fail", "not_run", "not_supported"] +HardwareStatus = Literal["pass", "fail", "not_run"] +InfrastructureStatus = Literal["ok", "error"] + + +@dataclass(frozen=True) +class CommandResult: + command: tuple[str, ...] + cwd: str + returncode: int + timed_out: bool + started_at_utc: str + ended_at_utc: str + duration_seconds: float + stdout_path: str + stderr_path: str + cleanup_error: Optional[str] = None + + +@dataclass(frozen=True) +class RunResult: + model_status: ModelStatus + compile_status: CompileStatus + simulator_status: SimulatorStatus + hardware_status: HardwareStatus + infrastructure_status: InfrastructureStatus + failure_category: Optional[str] = None + detail: Optional[str] = None + + @classmethod + def infrastructure_error(cls, category: str, detail: str) -> "RunResult": + return cls( + model_status="not_run", + compile_status="not_run", + simulator_status="not_supported", + hardware_status="not_run", + infrastructure_status="error", + failure_category=category, + detail=detail, + ) + + +def sha256_file(path: PathLike) -> str: + digest = hashlib.sha256() + with open(path, "rb") as source: + for chunk in iter(lambda: source.read(1024 * 1024), b""): + digest.update(chunk) + return digest.hexdigest() + + +def write_json_atomic(path: PathLike, value: Any) -> None: + destination = Path(path) + destination.parent.mkdir(parents=True, exist_ok=True) + temporary_path: Optional[str] = None + try: + with tempfile.NamedTemporaryFile( + mode="w", + encoding="utf-8", + dir=destination.parent, + prefix=f".{destination.name}.", + suffix=".tmp", + delete=False, + ) as temporary: + temporary_path = temporary.name + json.dump(value, temporary, indent=2, sort_keys=True) + temporary.write("\n") + temporary.flush() + os.fsync(temporary.fileno()) + os.replace(temporary_path, destination) + temporary_path = None + finally: + if temporary_path is not None: + try: + os.unlink(temporary_path) + except FileNotFoundError: + pass diff --git a/benchmarks/twin2silicon/identify_pio_device.py b/benchmarks/twin2silicon/identify_pio_device.py new file mode 100644 index 0000000..b38120a --- /dev/null +++ b/benchmarks/twin2silicon/identify_pio_device.py @@ -0,0 +1,76 @@ +#!/usr/bin/env python3 +"""Confirm that a selected UART is reported by PlatformIO for one JTAG serial.""" + +from __future__ import annotations + +import argparse +import json +import re +import subprocess +import sys +from typing import Any + + +_DEFAULT_TIMEOUT_SECONDS = 5.0 + + +def _positive_seconds(value: str) -> float: + try: + seconds = float(value) + except ValueError as error: + raise argparse.ArgumentTypeError("timeout must be a number") from error + if seconds <= 0: + raise argparse.ArgumentTypeError("timeout must be greater than zero") + return seconds + + +def _parser() -> argparse.ArgumentParser: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--uart-device", required=True) + parser.add_argument("--jtag-serial", required=True) + parser.add_argument("--timeout-seconds", type=_positive_seconds, default=_DEFAULT_TIMEOUT_SECONDS, + help=argparse.SUPPRESS) + return parser + + +def _has_exact_serial(hwid: str, serial: str) -> bool: + return re.search(r"(?:^|\s)SER=" + re.escape(serial) + r"(?=\s|$)", hwid) is not None + + +def _devices(timeout_seconds: float) -> list[dict[str, Any]] | None: + try: + completed = subprocess.run( + ["pio", "device", "list", "--json-output"], + stdin=subprocess.DEVNULL, + stdout=subprocess.PIPE, + stderr=subprocess.DEVNULL, + text=True, + timeout=timeout_seconds, + check=False, + ) + except (OSError, subprocess.TimeoutExpired): + return None + if completed.returncode != 0: + return None + try: + value = json.loads(completed.stdout) + except json.JSONDecodeError: + return None + return value if isinstance(value, list) and all(isinstance(item, dict) for item in value) else None + + +def main(argv: list[str] | None = None) -> int: + args = _parser().parse_args(argv) + devices = _devices(args.timeout_seconds) + if devices is None: + return 2 + for device in devices: + if device.get("port") == args.uart_device and isinstance(device.get("hwid"), str): + if _has_exact_serial(device["hwid"], args.jtag_serial): + print(args.jtag_serial) + return 0 + return 2 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/benchmarks/twin2silicon/prepare-oracle.py b/benchmarks/twin2silicon/prepare-oracle.py new file mode 100644 index 0000000..5e74d40 --- /dev/null +++ b/benchmarks/twin2silicon/prepare-oracle.py @@ -0,0 +1,10 @@ +from pathlib import Path +import sys + +template, system, firmware, output = map(Path, sys.argv[1:]) +text = template.read_text() +text = text.replace('"__SYSTEM__"', f'"{system.resolve()}"') +text = text.replace('"__FIRMWARE__"', f'"{firmware.resolve()}"') +if "__SYSTEM__" in text or "__FIRMWARE__" in text: + raise SystemExit("unresolved oracle placeholder") +output.write_text(text) diff --git a/benchmarks/twin2silicon/run_agent.py b/benchmarks/twin2silicon/run_agent.py new file mode 100644 index 0000000..8fadc84 --- /dev/null +++ b/benchmarks/twin2silicon/run_agent.py @@ -0,0 +1,293 @@ +#!/usr/bin/env python3 +"""Run one bounded native runtime trial against a public firmware task.""" + +from __future__ import annotations + +import argparse +from contextlib import contextmanager +from dataclasses import asdict +import json +import math +import os +from pathlib import Path +import shutil +import stat +import subprocess +import sys +import tempfile +import time +from typing import Iterator + +if __package__ in (None, ""): + sys.path.insert(0, str(Path(__file__).resolve().parents[2])) + +from benchmarks.twin2silicon.hil.process import run_command +from benchmarks.twin2silicon.hil.results import write_json_atomic +from benchmarks.twin2silicon.runtime_adapters import ( + AdapterContext, + NormalizedUsage, + build_runtime_command, + extract_native_model, + normalize_usage, + write_codex_mcp_config, +) + + +RUNTIMES = ("opencode", "codex", "claude") +ROOT = Path(__file__).resolve().parent +TASKS = ROOT / "tasks" +INSTRUCTIONS = ROOT / "shared-agent-instructions.md" +RUNTIME_CONFIG = ROOT / "runtime-config" + + +def _positive_seconds(value: str) -> float: + try: + seconds = float(value) + except ValueError as error: + raise argparse.ArgumentTypeError("timeout must be a number") from error + if seconds <= 0: + raise argparse.ArgumentTypeError("timeout must be greater than zero") + return seconds + + +def _parser() -> argparse.ArgumentParser: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("runtime", choices=RUNTIMES) + parser.add_argument("--task", required=True) + parser.add_argument("--output", required=True, type=Path) + parser.add_argument("--executable") + parser.add_argument("--timeout-seconds", type=_positive_seconds) + return parser + + +def _task_root(value: str) -> Path: + candidate = Path(value) + if len(candidate.parts) == 1: + candidate = TASKS / candidate + return candidate.resolve() + + +def _public_inputs(task_root: Path) -> tuple[Path, float, int]: + task = json.loads((task_root / "task.json").read_text(encoding="utf-8")) + public_dir = task["public_dir"] + budget = task["budgets"]["wall_time_seconds"] + repair_iterations = task["budgets"]["repair_iterations"] + if isinstance(public_dir, Path) or not isinstance(public_dir, str): + raise ValueError("task public_dir must be a path string") + if ( + isinstance(budget, bool) + or not isinstance(budget, (int, float)) + or not math.isfinite(float(budget)) + or budget <= 0 + ): + raise ValueError("task wall_time_seconds must be positive") + if ( + isinstance(repair_iterations, bool) + or not isinstance(repair_iterations, int) + or repair_iterations <= 0 + ): + raise ValueError("task repair_iterations must be a positive integer") + source_root = task_root / public_dir + public_root = source_root.resolve() + if task_root not in public_root.parents or not source_root.is_dir(): + raise ValueError("task public_dir must name a directory below the task root") + _reject_public_symlinks(source_root) + return public_root, float(budget), repair_iterations + + +def _reject_public_symlinks(public_root: Path) -> None: + for directory, directories, files in os.walk(public_root, followlinks=False): + for path in (Path(directory), *(Path(directory) / name for name in directories + files)): + if stat.S_ISLNK(os.lstat(path).st_mode): + raise ValueError("public inputs must not contain symlinks") + + +def _prepare_runtime_config(runtime: str, config_dir: Path) -> None: + config_dir.mkdir(parents=True, exist_ok=True) + if runtime == "codex": + write_codex_mcp_config(config_dir) + return + source_name = "opencode.json" if runtime == "opencode" else "claude-mcp.json" + shutil.copyfile(RUNTIME_CONFIG / source_name, config_dir / source_name) + + +@contextmanager +def _runtime_environment(runtime: str, config_dir: Path) -> Iterator[None]: + if runtime == "codex": + with _isolated_codex_home(): + yield + return + values = { + "opencode": {"OPENCODE_CONFIG": str(config_dir / "opencode.json")}, + "claude": {}, + }[runtime] + previous = {key: os.environ.get(key) for key in values} + os.environ.update(values) + try: + yield + finally: + for key, value in previous.items(): + if value is None: + os.environ.pop(key, None) + else: + os.environ[key] = value + + +@contextmanager +def _isolated_codex_home() -> Iterator[None]: + configured_home = os.environ.get("CODEX_HOME") + source_home = ( + Path(configured_home).expanduser() + if configured_home + else Path.home() / ".codex" + ) + with tempfile.TemporaryDirectory(prefix="twin2silicon-codex-") as directory: + isolated_home = Path(directory) + config_path = write_codex_mcp_config(isolated_home) + config_path.chmod(0o600) + source_auth = source_home / "auth.json" + if source_auth.is_file(): + destination_auth = isolated_home / "auth.json" + shutil.copyfile(source_auth, destination_auth) + destination_auth.chmod(0o600) + previous = os.environ.get("CODEX_HOME") + os.environ["CODEX_HOME"] = str(isolated_home) + try: + yield + finally: + if previous is None: + os.environ.pop("CODEX_HOME", None) + else: + os.environ["CODEX_HOME"] = previous + + +def _version(executable: str, cwd: Path, timeout_seconds: float) -> str | None: + try: + completed = subprocess.run( + [executable, "--version"], + cwd=cwd, + text=True, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + timeout=min(timeout_seconds, 5), + check=False, + ) + except (OSError, subprocess.TimeoutExpired): + return None + output = (completed.stdout or completed.stderr).strip() + return output.splitlines()[0][:4096] if output else None + + +def _prompt(candidate: Path, instructions: str, repair_iterations: int) -> str: + readme = candidate / "README.md" + task_prompt = readme.read_text(encoding="utf-8") if readme.is_file() else "Repair the public firmware task." + return ( + f"{instructions}\n\n# Trial limit\n\n" + f"Maximum repair attempts: {repair_iterations}\n\n# Public task\n\n{task_prompt}" + ) + + +def _unavailable_usage() -> NormalizedUsage: + return NormalizedUsage(None, None, None, None, None, None, "runtime did not expose usage") + + +def _write_trial_result(trial: Path, result: dict[str, object], usage: NormalizedUsage) -> None: + write_json_atomic(trial / "agent-result.json", result) + write_json_atomic(trial / "usage.json", asdict(usage)) + + +def main(argv: list[str] | None = None) -> int: + args = _parser().parse_args(argv) + trial = args.output.resolve() + if os.path.lexists(trial): + print("output path already exists", file=sys.stderr) + return 2 + + trial.mkdir(parents=True) + usage = _unavailable_usage() + result: dict[str, object] = { + "schema_version": "1.0", + "runtime": args.runtime, + "model_override": None, + "native_model": None, + "native_model_unavailable_reason": "runtime did not expose model", + "status": "infrastructure_error", + "returncode": None, + "timed_out": False, + "elapsed_seconds": 0.0, + "executable_version": None, + "stdout_path": "agent.stdout.log", + "stderr_path": "agent.stderr.log", + } + + try: + public_root, budget_seconds, repair_iterations = _public_inputs(_task_root(args.task)) + timeout_seconds = min(args.timeout_seconds or budget_seconds, budget_seconds) + candidate = trial / "candidate" + shutil.copytree(public_root, candidate) + instructions = INSTRUCTIONS.read_text(encoding="utf-8") + (candidate / ("CLAUDE.md" if args.runtime == "claude" else "AGENTS.md")).write_text( + instructions, encoding="utf-8" + ) + config_dir = trial / "runtime-config" + _prepare_runtime_config(args.runtime, config_dir) + executable = args.executable or args.runtime + context = AdapterContext( + runtime=args.runtime, + executable=executable, + workspace=candidate, + prompt=_prompt(candidate, instructions, repair_iterations), + config_dir=config_dir, + stdout_path=trial / "agent.stdout.log", + stderr_path=trial / "agent.stderr.log", + ) + command = build_runtime_command(args.runtime, context) + with _runtime_environment(args.runtime, config_dir): + result["executable_version"] = _version(executable, candidate, timeout_seconds) + started = time.monotonic() + try: + completed = run_command( + command, + cwd=candidate, + stdout_path=context.stdout_path, + stderr_path=context.stderr_path, + timeout_seconds=timeout_seconds, + ) + except OSError as error: + result["elapsed_seconds"] = time.monotonic() - started + result["error"] = str(error) + else: + result.update( + returncode=completed.returncode, + timed_out=completed.timed_out, + elapsed_seconds=completed.duration_seconds, + ) + if completed.timed_out: + result["status"] = "timeout" + elif completed.cleanup_error: + result["status"] = "infrastructure_error" + result["error"] = completed.cleanup_error + elif completed.returncode: + result["status"] = "failed" + else: + result["status"] = "completed" + try: + stdout_lines = context.stdout_path.read_text( + encoding="utf-8", errors="replace" + ).splitlines() + usage = normalize_usage(args.runtime, stdout_lines) + result["native_model"] = extract_native_model(args.runtime, stdout_lines) + result["native_model_unavailable_reason"] = ( + None if result["native_model"] is not None else "runtime did not expose model" + ) + except OSError: + usage = _unavailable_usage() + except Exception as error: + result["error"] = str(error) + + _write_trial_result(trial, result, usage) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/benchmarks/twin2silicon/run_hil.py b/benchmarks/twin2silicon/run_hil.py new file mode 100644 index 0000000..f9268d6 --- /dev/null +++ b/benchmarks/twin2silicon/run_hil.py @@ -0,0 +1,181 @@ +#!/usr/bin/env python3 +"""Run one simple ESP32-S3 build, flash, UART, and JTAG evaluation.""" + +from __future__ import annotations + +import argparse +import json +import math +import os +from pathlib import Path +import secrets +import shutil +import sys + +if __package__ in (None, ""): + sys.path.insert(0, str(Path(__file__).resolve().parents[2])) + +from benchmarks.twin2silicon.hil.esp32s3 import ( + BoardLock, + Esp32S3Config, + capture_uart_nonce, + flash_firmware, + read_registers, + validate_identity, +) +from benchmarks.twin2silicon.hil.process import run_command +from benchmarks.twin2silicon.hil.results import sha256_file, write_json_atomic + + +TASKS = Path(__file__).resolve().parent / "tasks" + + +def _usage(path: Path) -> dict: + data = json.loads(path.read_text()) + tokens = data["tokens"] + rates = data["rates_usd_per_million"] + for value in (*tokens.values(), *rates.values(), data["requests"]): + if isinstance(value, bool) or not isinstance(value, (int, float)) or not math.isfinite(value) or value < 0: + raise ValueError("usage values must be finite and nonnegative") + data["estimated_cost_usd"] = ( + tokens["fresh_input"] * rates["fresh_input"] + + tokens["cached_input"] * rates["cached_input"] + + tokens["output"] * rates["output"] + ) / 1_000_000 + return data + + +def _parser() -> argparse.ArgumentParser: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("task") + parser.add_argument("--run-dir", required=True, type=Path) + parser.add_argument("--candidate", required=True, type=Path) + parser.add_argument("--jtag-serial", required=True) + parser.add_argument("--uart-device", required=True) + parser.add_argument("--platformio", default="pio") + parser.add_argument("--openocd", required=True) + parser.add_argument("--identity-command-json") + parser.add_argument("--usage-json", type=Path) + return parser + + +def main(argv: list[str] | None = None) -> int: + args = _parser().parse_args(argv) + run_dir = args.run_dir.resolve() + if run_dir.exists(): + print("run directory already exists", file=sys.stderr) + return 2 + run_dir.mkdir(parents=True) + result = { + "schema_version": "1.0", + "status": "running", + "compile_status": "not_run", + "hardware_status": "not_run", + "infrastructure_status": "ok", + "uart": None, + "registers": [], + "cost": None, + "hashes": {}, + } + + def save() -> None: + write_json_atomic(run_dir / "run.json", result) + + save() + try: + task_root = (TASKS / args.task).resolve() if len(Path(args.task).parts) == 1 else Path(args.task).resolve() + task = json.loads((task_root / "task.json").read_text()) + oracle_path = task_root / task["hidden_oracle"] + config = Esp32S3Config.from_oracle(json.loads(oracle_path.read_text())) + workspace = run_dir / "workspace" + shutil.copytree(args.candidate.resolve(strict=True), workspace) + nonce = secrets.token_hex(16) + header = workspace / "firmware/include/run_nonce.h" + header.parent.mkdir(parents=True, exist_ok=True) + header.write_text(f'#pragma once\n#define LABWIRED_RUN_NONCE "{nonce}"\n') + result["task_id"] = task["id"] + result["hashes"]["oracle"] = sha256_file(oracle_path) + if args.usage_json: + cost = _usage(args.usage_json) + result["cost"] = cost + write_json_atomic(run_dir / "cost.json", cost) + save() + + firmware = workspace / "firmware" + build = [args.platformio, "run", "--project-dir", str(firmware), + "--environment", config.platformio_environment or "esp32s3"] + clean = run_command(build + ["--target", "clean"], cwd=workspace, + stdout_path=run_dir / "clean.stdout.log", + stderr_path=run_dir / "clean.stderr.log", + timeout_seconds=task["budgets"]["wall_time_seconds"]) + if clean.timed_out or clean.cleanup_error or clean.returncode: + raise RuntimeError("clean failed") + compiled = run_command(build, cwd=workspace, stdout_path=run_dir / "build.stdout.log", + stderr_path=run_dir / "build.stderr.log", + timeout_seconds=task["budgets"]["wall_time_seconds"]) + if compiled.timed_out or compiled.cleanup_error: + raise RuntimeError("build infrastructure failed") + if compiled.returncode: + result.update(status="fail", compile_status="fail") + save() + return 0 + result["compile_status"] = "pass" + artifact = firmware / config.flash_artifact + result["hashes"]["firmware"] = sha256_file(artifact) + save() + + identity_command = (json.loads(args.identity_command_json) + if args.identity_command_json else list(config.identity_command)) + with BoardLock(run_dir.parent / ".board-locks", args.jtag_serial, + timeout_seconds=config.identity_timeout_seconds): + identity = validate_identity(identity_command, args.jtag_serial, cwd=workspace, + evidence_dir=run_dir, + timeout_seconds=config.identity_timeout_seconds) + if identity.status != "pass": + raise RuntimeError(identity.detail or "board identity failed") + + flash_command = build + ["--upload-port", args.uart_device, "--target", config.flash_target] + flashed = flash_firmware(flash_command, cwd=workspace, evidence_dir=run_dir, + timeout_seconds=config.flash_timeout_seconds, + identity_validated=True) + if flashed.status == "infrastructure_error": + raise RuntimeError(flashed.detail or "flash infrastructure failed") + if flashed.status == "hardware_fail": + result.update(status="fail", hardware_status="fail") + save() + return 0 + + uart = capture_uart_nonce( + args.uart_device, config.uart_baud, nonce, + config.uart_timeout_seconds, run_dir / "uart.log") + result["uart"] = {"matched": uart.matched, "termination": uart.termination_reason, + "bytes": uart.bytes_captured} + if not uart.matched: + result.update(status="fail", hardware_status="fail") + save() + return 0 + + registers = read_registers(args.openocd, config.openocd_board_config, + args.jtag_serial, config.assertions, cwd=workspace, + evidence_dir=run_dir, + timeout_seconds=config.openocd_command_timeout_seconds) + if registers.status == "infrastructure_error": + raise RuntimeError(registers.detail or "OpenOCD failed") + result["registers"] = [ + {"name": item.name, "passed": item.passed, + "observed_masked": f"0x{item.value & item.mask:08x}"} + for item in registers.evaluation.observations + ] + result["hardware_status"] = "pass" if registers.status == "pass" else "fail" + result["status"] = "pass" if registers.status == "pass" else "fail" + save() + return 0 + except Exception as error: + result.update(status="invalid", infrastructure_status="error", error=str(error)) + save() + print(error, file=sys.stderr) + return 2 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/benchmarks/twin2silicon/run_matrix.py b/benchmarks/twin2silicon/run_matrix.py new file mode 100644 index 0000000..6b73258 --- /dev/null +++ b/benchmarks/twin2silicon/run_matrix.py @@ -0,0 +1,368 @@ +#!/usr/bin/env python3 +"""Run fresh native-runtime trials sequentially and retain one HIL oracle result per candidate.""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import math +import os +from pathlib import Path +import subprocess +import sys +import time +from typing import Any + +if __package__ in (None, ""): + sys.path.insert(0, str(Path(__file__).resolve().parents[2])) + +from benchmarks.twin2silicon.hil.results import write_json_atomic + + +RUNTIMES = ("opencode", "codex", "claude") +ROOT = Path(__file__).resolve().parent +TASKS = ROOT / "tasks" +USAGE_FIELDS = ( + "requests", + "fresh_input", + "cached_input", + "reasoning", + "output", + "estimated_cost_usd", +) + + +def _parser() -> argparse.ArgumentParser: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--task", required=True) + parser.add_argument("--output", required=True, type=Path) + parser.add_argument("--jtag-serial", required=True) + parser.add_argument("--uart-device", required=True) + parser.add_argument("--openocd", required=True) + parser.add_argument("--identity-command-json") + parser.add_argument("--runtime", choices=RUNTIMES, action="append") + parser.add_argument("--agent-only", action="store_true") + # Test-only seams. They are deliberately omitted from operator help. + parser.add_argument("--agent-script", type=Path, default=ROOT / "run_agent.py", help=argparse.SUPPRESS) + parser.add_argument("--hil-script", type=Path, default=ROOT / "run_hil.py", help=argparse.SUPPRESS) + parser.add_argument("--agent-executable", action="append", default=[], help=argparse.SUPPRESS) + return parser + + +def _task_root(value: str) -> Path: + requested = Path(value) + return (TASKS / requested if len(requested.parts) == 1 else requested).resolve() + + +def _task_details(task_root: Path) -> tuple[str, Path, float]: + task = json.loads((task_root / "task.json").read_text(encoding="utf-8")) + task_id = task["id"] + public_dir = task["public_dir"] + budget = task["budgets"]["wall_time_seconds"] + if not isinstance(task_id, str) or not isinstance(public_dir, str): + raise ValueError("task id and public_dir must be strings") + if isinstance(budget, bool) or not isinstance(budget, (int, float)) or not math.isfinite(budget) or budget <= 0: + raise ValueError("task wall_time_seconds must be positive") + public_root = (task_root / public_dir).resolve() + if task_root not in public_root.parents or not public_root.is_dir(): + raise ValueError("task public_dir must name a directory below the task root") + return task_id, public_root, float(budget) + + +def _tree_sha256(root: Path) -> str: + digest = hashlib.sha256() + for path in sorted(root.rglob("*"), key=lambda item: item.relative_to(root).as_posix()): + if path.is_symlink(): + raise ValueError("public inputs must not contain symlinks") + if not path.is_file(): + continue + digest.update(path.relative_to(root).as_posix().encode("utf-8")) + digest.update(b"\0") + with path.open("rb") as source: + for chunk in iter(lambda: source.read(1024 * 1024), b""): + digest.update(chunk) + digest.update(b"\0") + return digest.hexdigest() + + +def _read_json(path: Path) -> dict[str, Any] | None: + try: + value = json.loads(path.read_text(encoding="utf-8")) + except (OSError, json.JSONDecodeError): + return None + return value if isinstance(value, dict) else None + + +def _number(value: object) -> int | float | None: + if isinstance(value, bool) or not isinstance(value, (int, float)) or not math.isfinite(value) or value < 0: + return None + return value + + +def _safe_text(value: object) -> str | None: + if not isinstance(value, str): + return None + text = " ".join(value.split()) + return text[:4096] if text else None + + +def _result_number(result: dict[str, Any] | None, name: str) -> int | float | None: + return _number(result.get(name)) if result else None + + +def _result_bool(result: dict[str, Any] | None, name: str) -> bool | None: + value = result.get(name) if result else None + return value if isinstance(value, bool) else None + + +def _normalized_usage(path: Path) -> dict[str, object]: + source = _read_json(path) or {} + normalized = {field: _number(source.get(field)) for field in USAGE_FIELDS} + reason = source.get("unavailable_reason") + if not isinstance(reason, str) or not reason: + reason = None + if reason is None: + values = tuple(normalized.values()) + if all(value is None for value in values): + reason = "runtime did not expose usage" + elif any(value is None for value in values): + reason = "one or more usage fields unavailable" + normalized["unavailable_reason"] = reason + return normalized + + +def _has_hil_usage_schema(path: Path) -> bool: + """Only forward evidence that run_hil.py can validate without invented rates.""" + source = _read_json(path) + if source is None: + return False + tokens = source.get("tokens") + rates = source.get("rates_usd_per_million") + if not isinstance(tokens, dict) or not isinstance(rates, dict): + return False + required = ("fresh_input", "cached_input", "output") + if "requests" not in source or any(field not in tokens or field not in rates for field in required): + return False + values = [source["requests"], *tokens.values(), *rates.values()] + return all(_number(value) is not None for value in values) + + +def _parse_executables(values: list[str], parser: argparse.ArgumentParser) -> dict[str, str]: + executables: dict[str, str] = {} + for value in values: + runtime, separator, executable = value.partition("=") + if separator != "=" or runtime not in RUNTIMES or not executable: + parser.error("--agent-executable must be RUNTIME=PATH") + executables[runtime] = executable + return executables + + +def _run_child(command: list[str], cwd: Path, stdout_path: Path, stderr_path: Path) -> tuple[int | None, str | None]: + try: + with stdout_path.open("wb") as stdout, stderr_path.open("wb") as stderr: + completed = subprocess.run(command, cwd=cwd, stdout=stdout, stderr=stderr, check=False) + except OSError as error: + return None, str(error) + return completed.returncode, None + + +def _agent_row( + runtime: str, + task_root: Path, + trial: Path, + initial_public_sha256: str, + agent_script: Path, + executable: str | None, +) -> dict[str, object]: + command = [sys.executable, str(agent_script), runtime, "--task", str(task_root), "--output", str(trial)] + if executable is not None: + command.extend(("--executable", executable)) + returncode, child_error = _run_child( + command, ROOT, + trial.parent / f"{trial.name}.matrix-agent.stdout.log", + trial.parent / f"{trial.name}.matrix-agent.stderr.log", + ) + result = _read_json(trial / "agent-result.json") + agent_status = result.get("status") if result and isinstance(result.get("status"), str) else "infrastructure_error" + infrastructure_category: str | None = None + infrastructure_error: str | None = None + if child_error is not None or returncode != 0: + agent_status = "infrastructure_error" + infrastructure_category = "agent_runner" + infrastructure_error = _safe_text( + child_error if child_error is not None else f"agent runner exited with status {returncode}" + ) + elif result is None: + infrastructure_category = "agent_runner" + infrastructure_error = "agent runner did not produce agent-result.json" + elif agent_status == "infrastructure_error": + infrastructure_category = "agent_runtime" + infrastructure_error = _safe_text(result.get("error")) or "agent runtime infrastructure error" + native_model = _safe_text(result.get("native_model")) if result else None + native_model_unavailable_reason = ( + None + if native_model is not None + else _safe_text(result.get("native_model_unavailable_reason")) if result else None + ) + if native_model is None and native_model_unavailable_reason is None: + native_model_unavailable_reason = "runtime did not expose model" + row: dict[str, object] = { + "runtime": runtime, + "native_model": native_model, + "native_model_unavailable_reason": native_model_unavailable_reason, + "initial_public_sha256": initial_public_sha256, + "agent_status": agent_status, + "agent_returncode": _result_number(result, "returncode"), + "agent_timed_out": _result_bool(result, "timed_out"), + "elapsed_agent_seconds": _result_number(result, "elapsed_seconds"), + "agent_result": result, + "agent_child_returncode": returncode, + "usage": _normalized_usage(trial / "usage.json"), + "repair_count": None, + "tool_call_count": None, + "invalid_call_count": None, + "observability_reason": "runtime did not expose repair/tool-call counts", + "compile_status": "not_run", + "hil_status": "not_run", + "hil_run": None, + "elapsed_hil_seconds": None, + "final_success": False, + "infrastructure_category": infrastructure_category, + "infrastructure_error": infrastructure_error, + } + if child_error is not None: + row["agent_error"] = child_error + elif returncode != 0: + row["agent_error"] = f"agent runner exited with status {returncode}" + elif result is None: + row["agent_error"] = "agent runner did not produce agent-result.json" + return row + + +def _run_hil(row: dict[str, object], task_root: Path, trial: Path, args: argparse.Namespace) -> None: + if row["agent_status"] != "completed": + return + candidate = trial / "candidate" + if not candidate.is_dir(): + row["hil_status"] = "invalid" + row["hil_error"] = "completed agent did not produce a candidate directory" + row["compile_status"] = "invalid" + row["infrastructure_category"] = "candidate" + row["infrastructure_error"] = row["hil_error"] + return + run_dir = trial / "hil" + command = [ + sys.executable, str(args.hil_script), str(task_root), "--run-dir", str(run_dir), + "--candidate", str(candidate), "--jtag-serial", args.jtag_serial, + "--uart-device", args.uart_device, "--openocd", args.openocd, + "--identity-command-json", args.identity_command_json, + ] + usage_path = trial / "usage.json" + if _has_hil_usage_schema(usage_path): + command.extend(("--usage-json", str(usage_path))) + started = time.monotonic() + returncode, child_error = _run_child( + command, ROOT, trial / "matrix-hil.stdout.log", trial / "matrix-hil.stderr.log", + ) + row["elapsed_hil_seconds"] = time.monotonic() - started + run = _read_json(run_dir / "run.json") + row["hil_run"] = run + row["hil_child_returncode"] = returncode + if child_error is not None: + row["hil_status"] = "invalid" + row["hil_error"] = child_error + row["compile_status"] = "invalid" + row["infrastructure_category"] = "hil_runner" + row["infrastructure_error"] = _safe_text(child_error) + elif run is None: + row["hil_status"] = "invalid" + row["hil_error"] = "HIL runner did not produce run.json" + row["compile_status"] = "invalid" + row["infrastructure_category"] = "hil_runner" + row["infrastructure_error"] = row["hil_error"] + else: + status = run.get("status") + row["hil_status"] = status if isinstance(status, str) else "invalid" + compile_status = run.get("compile_status") + row["compile_status"] = compile_status if isinstance(compile_status, str) else None + row["final_success"] = row["hil_status"] == "pass" + + +def _display(value: object, width: int) -> str: + text = "-" if value is None else str(value) + return text[:width] + + +def _seconds(value: object) -> str: + return f"{value:.3f}" if isinstance(value, (int, float)) else "-" + + +def _compact_usage(usage: object) -> str: + if not isinstance(usage, dict): + return "-" + fresh = usage.get("fresh_input") + cached = usage.get("cached_input") + output = usage.get("output") + reasoning = usage.get("reasoning") + cost = usage.get("estimated_cost_usd") + tokens = f"i={fresh if fresh is not None else '-'}+{cached if cached is not None else '-'}" + tokens += f" o={output if output is not None else '-'} r={reasoning if reasoning is not None else '-'}" + return f"{tokens} ${cost:.6g}" if isinstance(cost, (int, float)) else tokens + + +def _print_summary(rows: list[dict[str, object]]) -> None: + print( + f"{'RUNTIME':<10} {'MODEL':<20} {'AGENT':<18} {'RETURN':<7} {'TIMEOUT':<7} " + f"{'REPAIR':<7} {'TOOLS':<7} {'INVALID':<7} {'COMPILE':<12} {'HIL':<12} " + f"{'SUCCESS':<7} {'INFRA':<14} {'A_SEC':>8} {'H_SEC':>8} TOKENS/COST" + ) + for row in rows: + print( + f"{_display(row['runtime'], 10):<10} {_display(row['native_model'], 20):<20} " + f"{_display(row['agent_status'], 18):<18} {_display(row['agent_returncode'], 7):<7} " + f"{_display(row['agent_timed_out'], 7):<7} {_display(row['repair_count'], 7):<7} " + f"{_display(row['tool_call_count'], 7):<7} {_display(row['invalid_call_count'], 7):<7} " + f"{_display(row['compile_status'], 12):<12} {_display(row['hil_status'], 12):<12} " + f"{_display(row['final_success'], 7):<7} {_display(row['infrastructure_category'], 14):<14} " + f"{_seconds(row['elapsed_agent_seconds']):>8} {_seconds(row['elapsed_hil_seconds']):>8} " + f"{_compact_usage(row['usage'])}" + ) + + +def main(argv: list[str] | None = None) -> int: + parser = _parser() + args = parser.parse_args(argv) + if not args.agent_only and not args.identity_command_json: + parser.error("--identity-command-json is required unless --agent-only") + output = args.output.resolve() + if os.path.lexists(output): + print("output path already exists", file=sys.stderr) + return 2 + try: + task_root = _task_root(args.task) + task_id, public_root, _budget_seconds = _task_details(task_root) + executables = _parse_executables(args.agent_executable, parser) + initial_public_sha256 = _tree_sha256(public_root) + except (OSError, ValueError, KeyError, TypeError, json.JSONDecodeError) as error: + print(str(error), file=sys.stderr) + return 2 + + output.mkdir(parents=True) + trials_root = output / "trials" + trials_root.mkdir() + rows: list[dict[str, object]] = [] + matrix: dict[str, object] = {"schema_version": "1.0", "task_id": task_id, "trials": rows} + for runtime in args.runtime or list(RUNTIMES): + trial = trials_root / runtime + row = _agent_row(runtime, task_root, trial, initial_public_sha256, args.agent_script, executables.get(runtime)) + if not args.agent_only: + _run_hil(row, task_root, trial, args) + rows.append(row) + write_json_atomic(output / "matrix.json", matrix) + _print_summary(rows) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/benchmarks/twin2silicon/runtime-config/claude-mcp.json b/benchmarks/twin2silicon/runtime-config/claude-mcp.json new file mode 100644 index 0000000..26a35c5 --- /dev/null +++ b/benchmarks/twin2silicon/runtime-config/claude-mcp.json @@ -0,0 +1,11 @@ +{ + "mcpServers": { + "labwired": { + "command": "npx", + "args": [ + "-y", + "@labwired/mcp" + ] + } + } +} diff --git a/benchmarks/twin2silicon/runtime-config/opencode.json b/benchmarks/twin2silicon/runtime-config/opencode.json new file mode 100644 index 0000000..65ada90 --- /dev/null +++ b/benchmarks/twin2silicon/runtime-config/opencode.json @@ -0,0 +1,46 @@ +{ + "$schema": "https://opencode.ai/config.json", + "mcp": { + "labwired": { + "type": "local", + "command": [ + "npx", + "-y", + "@labwired/mcp" + ], + "enabled": true, + "timeout": 120000, + "environment": { + "LABWIRED_CLI": "{env:LABWIRED_CLI}", + "LABWIRED_BUILDER_URL": "{env:LABWIRED_BUILDER_URL}" + } + } + }, + "permission": { + "skill": { + "brainstorming": "allow", + "develop": "allow", + "bringup": "allow", + "desk-hw": "allow", + "dispatching-parallel-agents": "allow", + "executing-plans": "allow", + "finishing-a-development-branch": "allow", + "golden-path": "allow", + "observe": "allow", + "prove": "allow", + "receiving-code-review": "allow", + "requesting-code-review": "allow", + "subagent-driven-development": "allow", + "systematic-debugging": "allow", + "test-driven-development": "allow", + "using-git-worktrees": "allow", + "using-superpowers": "allow", + "verification-before-completion": "allow", + "writing-plans": "allow", + "writing-skills": "allow", + "customize-labwired-agent": "allow", + "customize-opencode": "deny", + "import-circuit": "allow" + } + } +} diff --git a/benchmarks/twin2silicon/runtime_adapters.py b/benchmarks/twin2silicon/runtime_adapters.py new file mode 100644 index 0000000..79466e9 --- /dev/null +++ b/benchmarks/twin2silicon/runtime_adapters.py @@ -0,0 +1,358 @@ +"""Native CLI contracts and streaming usage normalization for benchmark runtimes. + +The caller runs each command with ``AdapterContext.workspace`` as its working +directory. Claude does not expose a workspace command-line option, so its MCP +configuration is deliberately placed beneath ``config_dir``; the command uses +``config_dir / 'claude-mcp.json'`` without creating that file. + +``NormalizedUsage.output`` consistently excludes separately reported reasoning +tokens. Codex reports ``output_tokens`` as its total output token count, so +its ``reasoning_output_tokens`` is subtracted when both values are valid. +Claude's ``output_tokens_details.thinking_tokens`` is also a subset of the +reported output total; for a non-overlapping comparison output field we retain +thinking as ``reasoning`` and subtract it from ``output``. +""" + +from __future__ import annotations + +from collections.abc import Iterable, Iterator +from dataclasses import dataclass +import json +import math +from pathlib import Path +from typing import Literal + + +RuntimeName = Literal["opencode", "codex", "claude"] +_MAX_USAGE_COUNT = 1_000_000_000_000 +_MAX_COST_USD = 1_000_000_000.0 +_CLAUDE_MCP_CONFIG = "claude-mcp.json" + + +def codex_mcp_toml() -> str: + """Return the isolated local LabWired MCP registration for Codex trials.""" + + return '[mcp_servers.labwired]\ncommand = "npx"\nargs = ["-y", "@labwired/mcp"]\n' + + +def write_codex_mcp_config(config_dir: Path) -> Path: + """Write the isolated Codex MCP configuration and return its path.""" + + config_dir.mkdir(parents=True, exist_ok=True) + config_path = config_dir / "config.toml" + config_path.write_text(codex_mcp_toml(), encoding="utf-8") + return config_path + + +@dataclass(frozen=True) +class AdapterContext: + """Execution inputs shared by the native runtime adapters.""" + + runtime: RuntimeName + executable: str + workspace: Path + prompt: str + config_dir: Path + stdout_path: Path + stderr_path: Path + + +@dataclass(frozen=True) +class NormalizedUsage: + """Usage available directly from a runtime's structured output.""" + + requests: int | None + fresh_input: int | None + cached_input: int | None + reasoning: int | None + output: int | None + estimated_cost_usd: float | None + unavailable_reason: str | None + + +def build_runtime_command(runtime: RuntimeName, context: AdapterContext) -> list[str]: + """Build the native, model-default command for ``runtime``.""" + + if runtime != context.runtime: + raise ValueError("runtime does not match context") + if runtime == "codex": + return [ + context.executable, + "exec", + "--json", + "--ephemeral", + "--skip-git-repo-check", + "-s", + "workspace-write", + "-C", + str(context.workspace), + context.prompt, + ] + if runtime == "claude": + return [ + context.executable, + "--print", + "--verbose", + "--output-format", + "stream-json", + "--no-session-persistence", + "--permission-mode", + "acceptEdits", + "--mcp-config", + str(context.config_dir / _CLAUDE_MCP_CONFIG), + "--strict-mcp-config", + context.prompt, + ] + if runtime == "opencode": + return [ + context.executable, + "run", + "--format", + "json", + "--dir", + str(context.workspace), + context.prompt, + ] + raise ValueError(f"unsupported runtime: {runtime}") + + +def normalize_usage(runtime: RuntimeName, lines: Iterable[str]) -> NormalizedUsage: + """Normalize newline-delimited structured runtime output without pricing inference.""" + + records = _json_records(lines) + if runtime == "opencode": + return _normalize_opencode(records) + if runtime == "codex": + return _normalize_codex(records) + if runtime == "claude": + return _normalize_claude(records) + raise ValueError(f"unsupported runtime: {runtime}") + + +def extract_native_model(runtime: RuntimeName, lines: Iterable[str]) -> str | None: + """Return a model only when a runtime's structured event explicitly names it. + + Runtime versions and prompts are not model evidence. The narrow event + locations below are deliberately conservative so an arbitrary JSON field + cannot be reported as a native model selection. + """ + + for record in _json_records(lines): + if runtime == "claude": + if record.get("type") == "system" and record.get("subtype") == "init": + model = _model_text(record.get("model")) + if model is not None: + return model + elif runtime == "codex": + if record.get("type") == "turn.started": + model = _model_text(record.get("model")) + if model is not None: + return model + elif runtime == "opencode": + if record.get("type") == "step_start": + part = _mapping(record.get("part")) + model = _model_text(part.get("model")) if part else None + if model is not None: + return model + else: + raise ValueError(f"unsupported runtime: {runtime}") + return None + + +def _json_records(lines: Iterable[str]) -> Iterator[dict[str, object]]: + source = lines.splitlines() if isinstance(lines, str) else lines + for line in source: + try: + record = json.loads(line) + except (TypeError, ValueError, json.JSONDecodeError): + continue + if isinstance(record, dict): + yield record + + +def _normalize_opencode(records: Iterable[dict[str, object]]) -> NormalizedUsage: + totals: dict[str, int] = {} + overflowed_totals: set[str] = set() + request_count = 0 + cost_total = 0.0 + has_cost = False + cost_overflowed = False + + for record in records: + if record.get("type") != "step_finish": + continue + part = _mapping(record.get("part")) + tokens = _mapping(part.get("tokens")) if part else None + values = { + "fresh_input": _bounded_int(tokens.get("input")) if tokens else None, + "cached_input": _bounded_int(_mapping(tokens.get("cache")).get("read")) + if tokens and _mapping(tokens.get("cache")) + else None, + "reasoning": _bounded_int(tokens.get("reasoning")) if tokens else None, + "output": _bounded_int(tokens.get("output")) if tokens else None, + } + cost = _bounded_float(part.get("cost")) if part else None + if all(value is None for value in values.values()) and cost is None: + continue + if request_count >= _MAX_USAGE_COUNT: + return _usage_result(0, {}, None) + request_count += 1 + for name, value in values.items(): + if value is None or name in overflowed_totals: + continue + total = totals.get(name, 0) + value + if total > _MAX_USAGE_COUNT: + totals.pop(name, None) + overflowed_totals.add(name) + else: + totals[name] = total + if cost is not None and not cost_overflowed: + total_cost = cost_total + cost + if total_cost > _MAX_COST_USD: + cost_overflowed = True + has_cost = False + else: + cost_total = total_cost + has_cost = True + + return _usage_result(request_count, totals, cost_total if has_cost else None) + + +def _normalize_codex(records: Iterable[dict[str, object]]) -> NormalizedUsage: + final_usage: dict[str, int | None] | None = None + for record in records: + if record.get("type") != "turn.completed": + continue + usage = _mapping(record.get("usage")) + final_usage = _codex_token_values(usage) + + if final_usage is None or not any( + value is not None for value in final_usage.values() + ): + return _usage_result(0, {}, None) + return _usage_result( + 1, {name: value for name, value in final_usage.items() if value is not None}, None + ) + + +def _normalize_claude(records: Iterable[dict[str, object]]) -> NormalizedUsage: + final_usage: dict[str, int | None] | None = None + final_cost: float | None = None + for record in records: + if record.get("type") != "result": + continue + usage = _mapping(record.get("usage")) + details = _mapping(usage.get("output_tokens_details")) if usage else None + total_output = _bounded_int(usage.get("output_tokens")) if usage else None + thinking = _bounded_int(details.get("thinking_tokens")) if details else None + if ( + total_output is not None + and thinking is not None + and thinking > total_output + ): + thinking = None + output = None + else: + output = total_output - thinking if total_output is not None and thinking is not None else total_output + values = { + "fresh_input": _bounded_int(usage.get("input_tokens")) if usage else None, + "cached_input": _bounded_int(usage.get("cache_read_input_tokens")) if usage else None, + "reasoning": thinking, + "output": output, + } + cost = _bounded_float(record.get("total_cost_usd")) + final_usage = values + final_cost = cost + + if final_usage is None or ( + not any(value is not None for value in final_usage.values()) + and final_cost is None + ): + return _usage_result(0, {}, None) + return _usage_result( + 1, + {name: value for name, value in final_usage.items() if value is not None}, + final_cost, + ) + + +def _codex_token_values(usage: dict[str, object] | None) -> dict[str, int | None]: + if usage is None: + return { + "fresh_input": None, + "cached_input": None, + "reasoning": None, + "output": None, + } + total_input = _bounded_int(usage.get("input_tokens")) + cached_input = _bounded_int(usage.get("cached_input_tokens")) + if total_input is not None and cached_input is None: + cached_input = 0 + if total_input is None or cached_input is None or cached_input > total_input: + fresh_input = None + if total_input is not None and cached_input is not None and cached_input > total_input: + cached_input = None + else: + fresh_input = total_input - cached_input + reasoning = _bounded_int(usage.get("reasoning_output_tokens")) + output = _bounded_int(usage.get("output_tokens")) + if reasoning is not None and output is not None: + if reasoning > output: + reasoning = None + output = None + else: + output -= reasoning + return { + "fresh_input": fresh_input, + "cached_input": cached_input, + "reasoning": reasoning, + "output": output, + } + + +def _usage_result( + requests: int, values: dict[str, int], cost: float | None +) -> NormalizedUsage: + if requests == 0: + return NormalizedUsage( + None, None, None, None, None, None, "runtime did not expose usage" + ) + if not values and cost is None: + return NormalizedUsage( + requests, None, None, None, None, None, "runtime did not expose usage" + ) + return NormalizedUsage( + requests, + values.get("fresh_input"), + values.get("cached_input"), + values.get("reasoning"), + values.get("output"), + cost, + None, + ) + + +def _mapping(value: object) -> dict[str, object] | None: + return value if isinstance(value, dict) else None + + +def _bounded_int(value: object) -> int | None: + if isinstance(value, bool) or not isinstance(value, int): + return None + return value if 0 <= value <= _MAX_USAGE_COUNT else None + + +def _bounded_float(value: object) -> float | None: + if isinstance(value, bool) or not isinstance(value, (int, float)): + return None + number = float(value) + if not math.isfinite(number) or not 0 <= number <= _MAX_COST_USD: + return None + return number + + +def _model_text(value: object) -> str | None: + if not isinstance(value, str): + return None + text = value.strip() + return text[:4096] if text else None diff --git a/benchmarks/twin2silicon/shared-agent-instructions.md b/benchmarks/twin2silicon/shared-agent-instructions.md new file mode 100644 index 0000000..14e4b1d --- /dev/null +++ b/benchmarks/twin2silicon/shared-agent-instructions.md @@ -0,0 +1,31 @@ +# Firmware repair trial + +This trial is noninteractive. This prompt authorizes you to inspect, make the +smallest in-scope repair, and compile. Do not pause or ask for confirmation. + +Work only inside the public workspace provided for this trial. Do not read, +search for, copy, modify, or infer from hidden task files, hidden oracle files, +or HIL result directories. Do not self-grade the repair or claim that it passes +hardware or a hidden oracle. Independent evaluation is performed outside this +workspace. + +Inspect the public task and make the smallest firmware repair that addresses +the stated problem. Preserve the project structure. Do not make unrelated +refactors, change test or evaluation files, add a workaround that bypasses the +requested behavior, or alter the task budget. + +The supplied task budget is binding. Count each edit-and-test cycle, including +the first attempt, and stop when `budgets.repair_iterations` is reached. Use +failures to make only focused repairs. + +Compile the public firmware with its existing build command. Your final report +must include compile evidence: the command, target, exit status, and relevant +diagnostics. If compilation cannot run, say why and report the evidence that is +available. Do not substitute source inspection for compile evidence. + +LabWired MCP tools are optional context and compile aids, not the final oracle. +Use them only when they help ground public hardware facts or compile the public +firmware. Their output does not prove hidden-oracle or hardware success. + +Report changed files, the repair rationale, compile evidence, and remaining +limits plainly. Do not claim more than the public evidence supports. diff --git a/benchmarks/twin2silicon/tasks/esp32s3-gpio-hil-001/hidden/hil-oracle.json b/benchmarks/twin2silicon/tasks/esp32s3-gpio-hil-001/hidden/hil-oracle.json new file mode 100644 index 0000000..6de002a --- /dev/null +++ b/benchmarks/twin2silicon/tasks/esp32s3-gpio-hil-001/hidden/hil-oracle.json @@ -0,0 +1,41 @@ +{ + "schema_version": "1.0", + "platformio": { + "project_dir": "public/firmware", + "environment": "esp32s3" + }, + "flash": { + "target": "upload", + "artifact": ".pio/build/esp32s3/firmware.bin", + "timeout_seconds": 120 + }, + "identity": { + "command": ["__LABWIRED_IDENTITY_RUNNER__"], + "expected_board": "esp32-s3-devkitc-1", + "timeout_seconds": 10 + }, + "uart": { + "ready_prefix": "LABWIRED_READY:", + "baud": 115200, + "timeout_seconds": 30 + }, + "openocd": { + "board_config": "board/esp32s3-builtin.cfg", + "startup_timeout_seconds": 20, + "command_timeout_seconds": 10 + }, + "register_assertions": [ + { + "name": "gpio2_output_enabled", + "address": "0x60004020", + "mask": "0x00000004", + "expected": "0x00000004" + }, + { + "name": "gpio2_output_high", + "address": "0x60004004", + "mask": "0x00000004", + "expected": "0x00000004" + } + ] +} diff --git a/benchmarks/twin2silicon/tasks/esp32s3-gpio-hil-001/public/README.md b/benchmarks/twin2silicon/tasks/esp32s3-gpio-hil-001/public/README.md new file mode 100644 index 0000000..4f42d8a --- /dev/null +++ b/benchmarks/twin2silicon/tasks/esp32s3-gpio-hil-001/public/README.md @@ -0,0 +1,11 @@ +# ESP32-S3 GPIO repair + +Repair the ESP-IDF firmware in `firmware/` so GPIO 2 is driven high when the +firmware reports readiness. Keep the readiness message and nonce behavior +intact. The evaluator builds the PlatformIO project, flashes an ESP32-S3 +DevKitC-1, observes its console output, and validates the resulting hardware +state. + +The checked-in nonce header makes the project build standalone. During an +evaluation run, the harness replaces its placeholder value with a unique run +nonce. diff --git a/benchmarks/twin2silicon/tasks/esp32s3-gpio-hil-001/public/firmware/include/run_nonce.h b/benchmarks/twin2silicon/tasks/esp32s3-gpio-hil-001/public/firmware/include/run_nonce.h new file mode 100644 index 0000000..e877c7e --- /dev/null +++ b/benchmarks/twin2silicon/tasks/esp32s3-gpio-hil-001/public/firmware/include/run_nonce.h @@ -0,0 +1,3 @@ +#pragma once + +#define LABWIRED_RUN_NONCE "standalone-placeholder" diff --git a/benchmarks/twin2silicon/tasks/esp32s3-gpio-hil-001/public/firmware/platformio.ini b/benchmarks/twin2silicon/tasks/esp32s3-gpio-hil-001/public/firmware/platformio.ini new file mode 100644 index 0000000..473b385 --- /dev/null +++ b/benchmarks/twin2silicon/tasks/esp32s3-gpio-hil-001/public/firmware/platformio.ini @@ -0,0 +1,7 @@ +[env:esp32s3] +platform = platformio/espressif32@7.0.1 +board = esp32-s3-devkitc-1 +framework = espidf +monitor_speed = 115200 +board_build.flash_size = 4MB +board_upload.flash_size = 4MB diff --git a/benchmarks/twin2silicon/tasks/esp32s3-gpio-hil-001/public/firmware/sdkconfig.defaults b/benchmarks/twin2silicon/tasks/esp32s3-gpio-hil-001/public/firmware/sdkconfig.defaults new file mode 100644 index 0000000..4f55897 --- /dev/null +++ b/benchmarks/twin2silicon/tasks/esp32s3-gpio-hil-001/public/firmware/sdkconfig.defaults @@ -0,0 +1,8 @@ +CONFIG_ESP_CONSOLE_USB_SERIAL_JTAG=y +# CONFIG_ESP_CONSOLE_UART_DEFAULT is not set +# CONFIG_ESP_CONSOLE_UART_CUSTOM is not set +# CONFIG_ESP_CONSOLE_NONE is not set +# CONFIG_ESP_CONSOLE_UART is not set +CONFIG_ESP_CONSOLE_SECONDARY_NONE=y +# CONFIG_ESP_CONSOLE_SECONDARY_USB_SERIAL_JTAG is not set +CONFIG_ESPTOOLPY_FLASHSIZE_4MB=y diff --git a/benchmarks/twin2silicon/tasks/esp32s3-gpio-hil-001/public/firmware/src/main.c b/benchmarks/twin2silicon/tasks/esp32s3-gpio-hil-001/public/firmware/src/main.c new file mode 100644 index 0000000..6218356 --- /dev/null +++ b/benchmarks/twin2silicon/tasks/esp32s3-gpio-hil-001/public/firmware/src/main.c @@ -0,0 +1,21 @@ +#include + +#include "driver/gpio.h" +#include "esp_err.h" +#include "freertos/FreeRTOS.h" +#include "freertos/task.h" +#include "run_nonce.h" + +#define TEST_GPIO GPIO_NUM_2 + +void app_main(void) +{ + ESP_ERROR_CHECK(gpio_set_direction(TEST_GPIO, GPIO_MODE_INPUT)); + ESP_ERROR_CHECK(gpio_set_level(TEST_GPIO, 1)); + + for (;;) { + printf("LABWIRED_READY:%s\n", LABWIRED_RUN_NONCE); + fflush(stdout); + vTaskDelay(pdMS_TO_TICKS(1000)); + } +} diff --git a/benchmarks/twin2silicon/tasks/esp32s3-gpio-hil-001/task.json b/benchmarks/twin2silicon/tasks/esp32s3-gpio-hil-001/task.json new file mode 100644 index 0000000..cd9f888 --- /dev/null +++ b/benchmarks/twin2silicon/tasks/esp32s3-gpio-hil-001/task.json @@ -0,0 +1,17 @@ +{ + "schema_version": "1.0", + "id": "esp32s3-gpio-hil-001", + "board": "esp32-s3-devkitc-1", + "framework": "espidf", + "model": "labwired/labwired-fast", + "budgets": { + "wall_time_seconds": 1200, + "model_tokens": 50000, + "repair_iterations": 6, + "simulator_runs": 0, + "diagnostic_hil_runs": 0 + }, + "public_dir": "public", + "hidden_oracle": "hidden/hil-oracle.json", + "firmware_elf": "firmware/.pio/build/esp32s3/firmware.elf" +} diff --git a/benchmarks/twin2silicon/tasks/f103-gpio-clock-001/hidden/oracle.yaml b/benchmarks/twin2silicon/tasks/f103-gpio-clock-001/hidden/oracle.yaml new file mode 100644 index 0000000..d331f14 --- /dev/null +++ b/benchmarks/twin2silicon/tasks/f103-gpio-clock-001/hidden/oracle.yaml @@ -0,0 +1,8 @@ +schema_version: "1.0" +inputs: + system: "__SYSTEM__" + firmware: "__FIRMWARE__" +limits: + max_steps: 200000 +assertions: + - uart_contains: "READY" diff --git a/benchmarks/twin2silicon/tasks/f103-gpio-clock-001/hidden/system.yaml b/benchmarks/twin2silicon/tasks/f103-gpio-clock-001/hidden/system.yaml new file mode 100644 index 0000000..df89161 --- /dev/null +++ b/benchmarks/twin2silicon/tasks/f103-gpio-clock-001/hidden/system.yaml @@ -0,0 +1,4 @@ +name: "f103-gpio-clock-001" +chip: "/Volumes/LabWired/worktrees/twin2silicon-core/configs/chips/stm32f103.yaml" +external_devices: [] +board_io: [] diff --git a/benchmarks/twin2silicon/tasks/f103-gpio-clock-001/public/README.md b/benchmarks/twin2silicon/tasks/f103-gpio-clock-001/public/README.md new file mode 100644 index 0000000..b0c09cf --- /dev/null +++ b/benchmarks/twin2silicon/tasks/f103-gpio-clock-001/public/README.md @@ -0,0 +1,6 @@ +# GPIO startup fault + +The STM32F103 firmware must configure PA0 as a push-pull output, drive it high, +and print `READY` on USART1. It builds, but the GPIO requirement does not pass +on the target. Diagnose and repair the firmware. Do not change the linker +script, startup code, or required output string. diff --git a/benchmarks/twin2silicon/tasks/f103-gpio-clock-001/public/firmware/Makefile b/benchmarks/twin2silicon/tasks/f103-gpio-clock-001/public/firmware/Makefile new file mode 100644 index 0000000..bb23966 --- /dev/null +++ b/benchmarks/twin2silicon/tasks/f103-gpio-clock-001/public/firmware/Makefile @@ -0,0 +1,13 @@ +BUILD_DIR := build +CROSS ?= arm-none-eabi- +CC := $(CROSS)gcc +CFLAGS := -mcpu=cortex-m3 -mthumb -ffreestanding -fno-builtin -ffunction-sections -fdata-sections -Os -g -Wall -Wextra -Werror +LDFLAGS := -nostdlib -T bench.ld -Wl,--gc-sections +all: $(BUILD_DIR)/firmware.elf +$(BUILD_DIR): + mkdir -p $@ +$(BUILD_DIR)/firmware.elf: startup.c main.c bench.ld | $(BUILD_DIR) + $(CC) $(CFLAGS) $(LDFLAGS) startup.c main.c -o $@ +clean: + rm -rf $(BUILD_DIR) +.PHONY: all clean diff --git a/benchmarks/twin2silicon/tasks/f103-gpio-clock-001/public/firmware/bench.ld b/benchmarks/twin2silicon/tasks/f103-gpio-clock-001/public/firmware/bench.ld new file mode 100644 index 0000000..686bc00 --- /dev/null +++ b/benchmarks/twin2silicon/tasks/f103-gpio-clock-001/public/firmware/bench.ld @@ -0,0 +1,12 @@ +ENTRY(Reset) +MEMORY { FLASH (rx) : ORIGIN = 0x08000000, LENGTH = 64K + RAM (rwx) : ORIGIN = 0x20000000, LENGTH = 20K } +_estack = ORIGIN(RAM) + LENGTH(RAM); +SECTIONS { + .isr_vector : { KEEP(*(.isr_vector)) } > FLASH + .text : { *(.text*) *(.rodata*) } > FLASH + _sidata = LOADADDR(.data); + .data : { . = ALIGN(4); _sdata = .; *(.data*) . = ALIGN(4); _edata = .; } > RAM AT > FLASH + .bss : { . = ALIGN(4); _sbss = .; *(.bss*) *(COMMON) . = ALIGN(4); _ebss = .; } > RAM + /DISCARD/ : { *(.ARM.exidx*) *(.note.gnu.build-id*) } +} diff --git a/benchmarks/twin2silicon/tasks/f103-gpio-clock-001/public/firmware/main.c b/benchmarks/twin2silicon/tasks/f103-gpio-clock-001/public/firmware/main.c new file mode 100644 index 0000000..32c0d0d --- /dev/null +++ b/benchmarks/twin2silicon/tasks/f103-gpio-clock-001/public/firmware/main.c @@ -0,0 +1,25 @@ +#include +#define REG32(a) (*(volatile uint32_t *)(a)) +#define RCC_APB2ENR REG32(0x40021018u) +#define GPIOA_CRL REG32(0x40010800u) +#define GPIOA_ODR REG32(0x4001080cu) +#define USART1_SR REG32(0x40013800u) +#define USART1_DR REG32(0x40013804u) +#define USART1_CR1 REG32(0x4001380cu) + +static void putc(char c) { + while ((USART1_SR & (1u << 7)) == 0u) {} + USART1_DR = (uint32_t)(uint8_t)c; +} + +int main(void) { + RCC_APB2ENR |= (1u << 14); + USART1_CR1 = (1u << 13) | (1u << 3); + GPIOA_CRL = (GPIOA_CRL & ~0xfu) | 0x3u; + GPIOA_ODR |= 1u; + if ((GPIOA_CRL & 0xfu) == 0x3u && (GPIOA_ODR & 1u) != 0u) { + const char *s = "READY\n"; + while (*s) putc(*s++); + } + for (;;) {} +} diff --git a/benchmarks/twin2silicon/tasks/f103-gpio-clock-001/public/firmware/startup.c b/benchmarks/twin2silicon/tasks/f103-gpio-clock-001/public/firmware/startup.c new file mode 100644 index 0000000..75c1094 --- /dev/null +++ b/benchmarks/twin2silicon/tasks/f103-gpio-clock-001/public/firmware/startup.c @@ -0,0 +1,19 @@ +#include +extern uint32_t _sidata, _sdata, _edata, _sbss, _ebss, _estack; +extern int main(void); +void Default_Handler(void) { for (;;) {} } +__attribute__((used, noreturn)) static void Reset_C(void) { + uint32_t *src = &_sidata, *dst = &_sdata; + while (dst < &_edata) *dst++ = *src++; + for (dst = &_sbss; dst < &_ebss;) *dst++ = 0u; + (void)main(); + for (;;) {} +} +__attribute__((naked, used, noreturn)) void Reset(void) { + __asm volatile("ldr r0, =_estack\nmov sp, r0\nbl Reset_C\nb .\n"); +} +__attribute__((section(".isr_vector"), used)) void (*const g_vectors[16])(void) = { + (void (*)(void))&_estack, Reset, Default_Handler, Default_Handler, + Default_Handler, Default_Handler, Default_Handler, 0, 0, 0, 0, + Default_Handler, Default_Handler, 0, Default_Handler, Default_Handler, +}; diff --git a/benchmarks/twin2silicon/tasks/f103-gpio-clock-001/task.json b/benchmarks/twin2silicon/tasks/f103-gpio-clock-001/task.json new file mode 100644 index 0000000..818e083 --- /dev/null +++ b/benchmarks/twin2silicon/tasks/f103-gpio-clock-001/task.json @@ -0,0 +1 @@ +{"schema_version":"1.0","id":"f103-gpio-clock-001","board":"stm32f103-bluepill","model":"labwired/labwired-fast","budgets":{"wall_time_seconds":1200,"model_tokens":50000,"repair_iterations":6,"simulator_runs":8,"diagnostic_hil_runs":0},"public_dir":"public","hidden_oracle":"hidden/oracle.yaml","hidden_system":"hidden/system.yaml","firmware_elf":"firmware/build/firmware.elf"} diff --git a/config/opencode.deepinfra.json b/config/opencode.deepinfra.json index 9828f5e..ec3efa4 100644 --- a/config/opencode.deepinfra.json +++ b/config/opencode.deepinfra.json @@ -48,7 +48,7 @@ "npm": "@ai-sdk/openai-compatible", "name": "DeepInfra", "options": { - "baseURL": "https://api.deepinfra.com/v1/openai", + "baseURL": "https://api.deepinfra.com/v1", "apiKey": "{env:DEEPINFRA_API_KEY}" }, "models": { diff --git a/config/opencode.hosted.json b/config/opencode.hosted.json index 01e6892..dbea864 100644 --- a/config/opencode.hosted.json +++ b/config/opencode.hosted.json @@ -3,7 +3,7 @@ "mcp": { "labwired": { "type": "remote", - "url": "https://api.labwired.com/mcp", + "url": "https://api.labwired.com/mcp?toolNames=unprefixed", "enabled": true, "oauth": false, "headers": { diff --git a/docs/superpowers/plans/2026-08-15-cross-runtime-hil-smoke.md b/docs/superpowers/plans/2026-08-15-cross-runtime-hil-smoke.md new file mode 100644 index 0000000..45b1a92 --- /dev/null +++ b/docs/superpowers/plans/2026-08-15-cross-runtime-hil-smoke.md @@ -0,0 +1,437 @@ +# Cross-Runtime HIL Smoke Test Implementation Plan + +> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking. + +**Goal:** Run OpenCode, Codex CLI, and Claude Code with their native/default models against identical LabWired firmware tasks, then score every candidate with the existing physical HIL oracle and emit one normalized comparison. + +**Architecture:** A stdlib-only adapter module builds native commands and normalizes runtime output. A single-trial controller copies public inputs, writes runtime-native instruction/MCP files, runs one bounded agent, and emits `agent-result.json` plus `usage.json`. A matrix controller runs independent trials and invokes the existing `run_hil.py`; it never interprets or replaces the hidden oracle. + +**Tech Stack:** Python 3 standard library, existing LabWired skills/MCP package, OpenCode 1.18.7, Codex CLI, Claude Code, PlatformIO, ESP-IDF, OpenOCD, UART/JTAG HIL. + +--- + +## File Structure + +- Create `benchmarks/twin2silicon/runtime_adapters.py`: runtime command construction and output normalization only. +- Create `benchmarks/twin2silicon/run_agent.py`: one fresh candidate-generation trial and its evidence. +- Create `benchmarks/twin2silicon/run_matrix.py`: sequential cross-runtime execution, HIL invocation, and summary output. +- Create `benchmarks/twin2silicon/shared-agent-instructions.md`: model-neutral firmware repair instructions mapped into each runtime. +- Create `benchmarks/twin2silicon/runtime-config/opencode.json`: existing local LabWired MCP profile with no model override. +- Create `benchmarks/twin2silicon/runtime-config/claude-mcp.json`: local LabWired MCP registration for Claude Code. +- Modify `tests/twin2silicon-hil.py`: offline adapter, trial, normalization, and matrix tests using fake executables. +- Create `tests/twin2silicon-runtime-smoke.sh`: opt-in connected-board entry point. +- Modify `package.json`: expose offline and connected smoke commands. +- Modify `benchmarks/twin2silicon/README.md` or create it if absent: document the matrix contract and usage. + +### Task 1: Define the normalized runtime contract + +**Files:** +- Create: `benchmarks/twin2silicon/runtime_adapters.py` +- Modify: `tests/twin2silicon-hil.py` + +- [ ] **Step 1: Write failing command-construction tests** + +Add `RuntimeAdapterTests` that creates an `AdapterContext` and asserts: + +```python +self.assertEqual( + build_runtime_command("codex", context)[:2], + ["codex", "exec"], +) +self.assertNotIn("--model", build_runtime_command("codex", context)) +self.assertEqual(build_runtime_command("claude", context)[:2], ["claude", "--print"]) +self.assertNotIn("--model", build_runtime_command("claude", context)) +self.assertEqual(build_runtime_command("opencode", context)[:2], ["opencode", "run"]) +self.assertNotIn("--model", build_runtime_command("opencode", context)) +``` + +Also assert each command selects structured output and the supplied workspace without embedding a hidden-oracle path. + +- [ ] **Step 2: Run the focused test and verify RED** + +Run: + +```bash +python3 tests/twin2silicon-hil.py -k RuntimeAdapterTests +``` + +Expected: import failure for `benchmarks.twin2silicon.runtime_adapters`. + +- [ ] **Step 3: Implement the minimal adapter types and commands** + +Define frozen dataclasses: + +```python +@dataclass(frozen=True) +class AdapterContext: + runtime: Literal["opencode", "codex", "claude"] + executable: str + workspace: Path + prompt: str + config_dir: Path + stdout_path: Path + stderr_path: Path + +@dataclass(frozen=True) +class NormalizedUsage: + requests: int | None + fresh_input: int | None + cached_input: int | None + reasoning: int | None + output: int | None + estimated_cost_usd: float | None + unavailable_reason: str | None +``` + +Construct native/default commands without model flags: + +```python +codex exec --json --ephemeral --skip-git-repo-check -s workspace-write -C WORKSPACE PROMPT +claude --print --output-format stream-json --no-session-persistence --permission-mode acceptEdits --mcp-config CONFIG PROMPT +opencode run --format json --dir WORKSPACE PROMPT +``` + +Use runtime-specific environment variables only for config discovery; do not copy credentials into evidence. + +- [ ] **Step 4: Write failing usage-normalization tests** + +Use compact fixtures for: + +- OpenCode `step_finish.part.tokens` and `part.cost` events; +- Codex JSONL token-usage events; +- Claude stream-json `result` usage and `total_cost_usd`; +- malformed output and successful output with no accounting. + +Expected normalized behavior: + +```python +self.assertEqual(usage.output, 1076) +self.assertEqual(usage.estimated_cost_usd, 0.007476282) +self.assertIsNone(missing.estimated_cost_usd) +self.assertEqual(missing.unavailable_reason, "runtime did not expose usage") +``` + +- [ ] **Step 5: Implement streaming parsers and GREEN the tests** + +Parse line-by-line with bounded integer/float validation. Sum OpenCode step costs, use Codex totals from the final usage event, and use Claude's final result object. Never infer subscription cost from token counts. + +Run: + +```bash +python3 tests/twin2silicon-hil.py -k RuntimeAdapterTests +``` + +Expected: all adapter tests pass. + +- [ ] **Step 6: Commit Task 1** + +```bash +git add benchmarks/twin2silicon/runtime_adapters.py tests/twin2silicon-hil.py +git commit -m "feat(bench): define native runtime adapters" +``` + +### Task 2: Add shared instructions and runtime-native MCP configuration + +**Files:** +- Create: `benchmarks/twin2silicon/shared-agent-instructions.md` +- Create: `benchmarks/twin2silicon/runtime-config/opencode.json` +- Create: `benchmarks/twin2silicon/runtime-config/claude-mcp.json` +- Modify: `tests/twin2silicon-hil.py` + +- [ ] **Step 1: Write failing configuration-contract tests** + +Assert that shared instructions: + +- require the smallest firmware repair; +- permit only public workspace access; +- prohibit hidden-oracle access and self-grading; +- require compile evidence; +- cap repair attempts at the task budget; +- name LabWired MCP tools as optional context/compile aids, not as the final oracle. + +Parse both JSON configs and assert they launch exactly: + +```json +{"command": "npx", "args": ["-y", "@labwired/mcp"]} +``` + +Assert the OpenCode config declares no `model` key. Codex MCP is supplied through isolated TOML generated by `run_agent.py`, so the contract test checks the generated TOML rather than a third static config. + +- [ ] **Step 2: Run focused tests and verify RED** + +```bash +python3 tests/twin2silicon-hil.py -k RuntimeConfigurationTests +``` + +Expected: missing instruction/config files. + +- [ ] **Step 3: Add the minimal shared instruction file** + +Keep it below 700 words and reuse the behavioral substance of `skills/develop/SKILL.md`: inspect, ground, edit, compile, report evidence, do not claim hardware success. + +- [ ] **Step 4: Add native MCP configs** + +OpenCode config uses a local `labwired` MCP entry and existing permission allowlist, but omits provider/model declarations so the installed native default remains authoritative. Claude config uses `mcpServers.labwired.command = "npx"` and `args = ["-y", "@labwired/mcp"]`. + +Generate Codex TOML in the trial config directory: + +```toml +[mcp_servers.labwired] +command = "npx" +args = ["-y", "@labwired/mcp"] +``` + +- [ ] **Step 5: Run tests and commit Task 2** + +```bash +python3 tests/twin2silicon-hil.py -k RuntimeConfigurationTests +git add benchmarks/twin2silicon/shared-agent-instructions.md benchmarks/twin2silicon/runtime-config tests/twin2silicon-hil.py +git commit -m "feat(bench): share LabWired instructions across runtimes" +``` + +### Task 3: Implement one bounded agent trial + +**Files:** +- Create: `benchmarks/twin2silicon/run_agent.py` +- Modify: `tests/twin2silicon-hil.py` + +- [ ] **Step 1: Write failing end-to-end fake-runtime tests** + +Create one fake executable per runtime. Each fake: + +- validates its expected native argv; +- confirms the candidate starts with `GPIO_MODE_INPUT`; +- changes only that token to `GPIO_MODE_OUTPUT`; +- emits representative native JSON usage; +- exits zero. + +Invoke `run_agent.py` and assert: + +```python +self.assertEqual(result["status"], "completed") +self.assertEqual(result["runtime"], runtime) +self.assertIsNone(result["model_override"]) +self.assertIn("GPIO_MODE_OUTPUT", candidate_source) +self.assertTrue((trial / "agent.stdout.log").is_file()) +self.assertTrue((trial / "agent.stderr.log").is_file()) +self.assertTrue((trial / "usage.json").is_file()) +``` + +Add cases for nonzero exit, timeout, missing executable, malformed JSON, and an existing output directory. Assert none exposes or copies the hidden oracle. + +- [ ] **Step 2: Run focused tests and verify RED** + +```bash +python3 tests/twin2silicon-hil.py -k RunAgentTests +``` + +Expected: `run_agent.py` missing. + +- [ ] **Step 3: Implement the single-trial CLI** + +CLI: + +```text +run_agent.py RUNTIME --task TASK --output TRIAL_DIR + [--executable PATH] [--timeout-seconds N] +``` + +Behavior: + +1. reject an existing output path; +2. read `task.json` only to locate `public_dir` and budgets; +3. copy only `public_dir` to `TRIAL_DIR/candidate`; +4. combine the model-neutral task prompt with shared instructions; +5. write runtime-native config under `TRIAL_DIR/runtime-config`, copy the same + shared instruction text to `candidate/AGENTS.md` for Codex/OpenCode and + `candidate/CLAUDE.md` for Claude Code; +6. invoke `run_command` with the adapter's argv and bounded timeout; +7. parse usage without failing the trial when accounting is unavailable; +8. atomically write `agent-result.json` and `usage.json`. + +Set `agent-result.status` to `completed`, `failed`, `timeout`, or `infrastructure_error`. Record monotonic elapsed time, exit code, and executable version. Do not run HIL here. + +- [ ] **Step 4: Run focused and full offline tests** + +```bash +python3 tests/twin2silicon-hil.py -k RunAgentTests +python3 tests/twin2silicon-hil.py +``` + +Expected: all tests pass. + +- [ ] **Step 5: Commit Task 3** + +```bash +git add benchmarks/twin2silicon/run_agent.py tests/twin2silicon-hil.py +git commit -m "feat(bench): run one native agent trial" +``` + +### Task 4: Add the sequential runtime matrix and normalized summary + +**Files:** +- Create: `benchmarks/twin2silicon/run_matrix.py` +- Modify: `tests/twin2silicon-hil.py` + +- [ ] **Step 1: Write failing matrix tests** + +Use fake adapters and a fake HIL executable to assert: + +- runtime order is `opencode`, `codex`, `claude`; +- each receives byte-identical public input hashes; +- one adapter failure does not prevent later trials; +- HIL runs only for completed candidates; +- the hidden-oracle path appears only in HIL argv, never adapter argv/logs; +- summary rows retain unknown usage as `null` plus a reason. + +Expected summary skeleton: + +```json +{ + "schema_version": "1.0", + "task_id": "esp32s3-gpio-hil-001", + "trials": [ + {"runtime": "opencode", "agent_status": "completed", "hil_status": "pass"}, + {"runtime": "codex", "agent_status": "failed", "hil_status": "not_run"}, + {"runtime": "claude", "agent_status": "completed", "hil_status": "fail"} + ] +} +``` + +- [ ] **Step 2: Run focused tests and verify RED** + +```bash +python3 tests/twin2silicon-hil.py -k RuntimeMatrixTests +``` + +Expected: `run_matrix.py` missing. + +- [ ] **Step 3: Implement the matrix CLI** + +CLI: + +```text +run_matrix.py --task TASK --output MATRIX_DIR + --jtag-serial SERIAL --uart-device DEVICE --openocd PATH + [--runtime opencode --runtime codex --runtime claude] + [--agent-only] +``` + +Run trials sequentially because one physical board is shared. Invoke the existing `run_hil.py` as a child process with each completed candidate. Pass `--usage-json` only when the adapter produced all numeric fields required by the existing HIL cost schema; otherwise let HIL record `cost: null` and retain the explicit unavailable reason in the matrix row. Read, do not reinterpret, each HIL `run.json`. Atomically write `matrix.json` after every trial and print a fixed-width summary at completion. + +- [ ] **Step 4: Verify matrix failure isolation and full suite** + +```bash +python3 tests/twin2silicon-hil.py -k RuntimeMatrixTests +python3 tests/twin2silicon-hil.py +``` + +Expected: all tests pass; fake matrix contains all three rows after an injected middle failure. + +- [ ] **Step 5: Commit Task 4** + +```bash +git add benchmarks/twin2silicon/run_matrix.py tests/twin2silicon-hil.py +git commit -m "feat(bench): compare native runtimes with one HIL oracle" +``` + +### Task 5: Add operator entry points and documentation + +**Files:** +- Create: `tests/twin2silicon-runtime-smoke.sh` +- Create: `benchmarks/twin2silicon/README.md` +- Modify: `package.json` +- Modify: `tests/twin2silicon-hil.py` + +- [ ] **Step 1: Write failing packaging/entry-point assertions** + +Extend the existing contract tests to require: + +```json +{ + "test:runtime-smoke:offline": "python3 tests/twin2silicon-hil.py", + "test:runtime-smoke:hardware": "bash tests/twin2silicon-runtime-smoke.sh" +} +``` + +Assert the shell entry point refuses to run unless `LABWIRED_HIL=1`, requires explicit UART/JTAG/OpenOCD values, places temporary data on `/Volumes/LabWired` when available, and never contains credentials. + +- [ ] **Step 2: Run tests and verify RED** + +```bash +python3 tests/twin2silicon-hil.py -k RuntimePackagingTests +``` + +Expected: scripts and package commands missing. + +- [ ] **Step 3: Implement the opt-in shell entry point** + +The script validates installed `opencode`, `codex`, `claude`, `pio`, and OpenOCD; prints their versions; then invokes `run_matrix.py` with explicit hardware arguments. It never reads API secrets itself—each native runtime uses its existing authenticated session. + +- [ ] **Step 4: Document experiment interpretation** + +Document: + +- this is a runtime smoke comparison, not a leaderboard claim; +- native models are intentionally not overridden; +- raw model comparisons require the separate OpenCode model matrix; +- missing subscription cost remains unknown; +- connected hardware is modified by flash operations; +- every publishable result needs multiple tasks and repeated fresh trials. + +- [ ] **Step 5: Run offline verification and commit Task 5** + +```bash +npm run test:runtime-smoke:offline +bash tests/twin2silicon-runtime-smoke.sh +``` + +Expected: offline suite passes; hardware script exits with a clear `LABWIRED_HIL=1 required` message when not opted in. + +```bash +git add benchmarks/twin2silicon/README.md tests/twin2silicon-runtime-smoke.sh package.json tests/twin2silicon-hil.py +git commit -m "docs(bench): add cross-runtime smoke entry points" +``` + +### Task 6: Run the connected-board smoke matrix + +**Files:** +- Runtime evidence only under an explicit matrix output directory outside tracked source. + +- [ ] **Step 1: Verify authentication without exposing credentials** + +Run native read-only status commands for OpenCode/LabWired, Codex, and Claude Code. Record only runtime versions and authenticated/not-authenticated status. + +- [ ] **Step 2: Confirm the target S3 identity** + +Use the existing serial/JTAG identification command to confirm the selected device is ESP32-S3 and record its serial. Do not select by port order alone. + +- [ ] **Step 3: Run one fresh native trial per runtime** + +Example: + +```bash +LABWIRED_HIL=1 \ +LABWIRED_UART_DEVICE=/dev/cu.usbmodem11101 \ +LABWIRED_JTAG_SERIAL=3C:0F:02:DF:EC:F8 \ +LABWIRED_OPENOCD="$HOME/.platformio/packages/tool-openocd-esp32/bin/openocd" \ +LABWIRED_MATRIX_OUTPUT=/Volumes/LabWired/hil-runs/runtime-smoke-$(date +%Y%m%d-%H%M%S) \ +bash tests/twin2silicon-runtime-smoke.sh +``` + +Expected: three trial directories and a parseable `matrix.json`; individual failures remain rows rather than aborting the matrix. + +- [ ] **Step 4: Verify evidence and summarize honestly** + +Check every completed candidate's `run.json`, artifact hashes, UART nonce, and register observations. Report success, time, tokens, cost availability, and infrastructure issues without claiming a general winner. + +- [ ] **Step 5: Run final repository verification** + +```bash +npm run test:runtime-smoke:offline +python3 tests/twin2silicon-hil.py +git diff --check +git status --short +``` + +Expected: all offline tests pass, diff check is clean, and only pre-existing `out/` remains untracked. diff --git a/docs/superpowers/plans/2026-08-15-esp32s3-hil-benchmark.md b/docs/superpowers/plans/2026-08-15-esp32s3-hil-benchmark.md new file mode 100644 index 0000000..8c86cda --- /dev/null +++ b/docs/superpowers/plans/2026-08-15-esp32s3-hil-benchmark.md @@ -0,0 +1,402 @@ +# ESP32-S3 HIL Benchmark Implementation Plan + +> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking. + +**Goal:** Add one reproducible ESP-IDF repair task and an automated evaluator that scores a connected ESP32-S3 using both a nonce-tagged USB-serial observation and hidden GPIO register assertions read through USB-JTAG. + +**Architecture:** Keep model execution separate from physical evaluation. A Python standard-library benchmark package owns typed results, bounded subprocesses, board locking, UART/JTAG evidence, and manifest generation; a small CLI composes those units. PlatformIO supplies the pinned ESP-IDF build/flash path, while fixture executables make every orchestration branch testable without hardware. + +**Tech Stack:** Python 3 standard library, PlatformIO with Espressif32/ESP-IDF, Espressif OpenOCD, macOS USB serial, Bash test registration, JSON task/oracle manifests. + +--- + +## File Map + +- Create `benchmarks/twin2silicon/hil/__init__.py`: package marker and public result types. +- Create `benchmarks/twin2silicon/hil/process.py`: timeout-bounded subprocess execution and captured command evidence. +- Create `benchmarks/twin2silicon/hil/esp32s3.py`: board identity validation, lock lifecycle, UART capture, flash, OpenOCD reads, and register evaluation. +- Create `benchmarks/twin2silicon/hil/results.py`: canonical status/result dataclasses, hashing, and atomic JSON output. +- Create `benchmarks/twin2silicon/run_hil.py`: CLI that loads the task, prepares a run, optionally invokes the Agent, and evaluates the candidate. +- Create `benchmarks/twin2silicon/tasks/esp32s3-gpio-hil-001/task.json`: public metadata and budgets. +- Create `benchmarks/twin2silicon/tasks/esp32s3-gpio-hil-001/hidden/hil-oracle.json`: GPIO register assertions and transport settings. +- Create `benchmarks/twin2silicon/tasks/esp32s3-gpio-hil-001/public/README.md`: repair prompt. +- Create `benchmarks/twin2silicon/tasks/esp32s3-gpio-hil-001/public/firmware/platformio.ini`: pinned ESP-IDF project configuration. +- Create `benchmarks/twin2silicon/tasks/esp32s3-gpio-hil-001/public/firmware/sdkconfig.defaults`: USB Serial/JTAG console configuration. +- Create `benchmarks/twin2silicon/tasks/esp32s3-gpio-hil-001/public/firmware/src/main.c`: deliberately faulty GPIO firmware. +- Create `tests/twin2silicon-hil.py`: offline unit/integration tests with fixture executables and pseudo-terminals. +- Create `tests/twin2silicon-hil.sh`: shell entrypoint for the offline suite. +- Create `tests/twin2silicon-hil-live.sh`: explicit destructive live acceptance test. +- Modify `tests/all.sh`: register only the offline HIL suite. +- Modify `package.json`: add `test:twin2silicon-hil` and `test:twin2silicon-hil:live` commands. +- Modify `docs/TESTING.md`: document offline and destructive live commands. + +### Task 1: Define result and process contracts + +**Files:** +- Create: `benchmarks/twin2silicon/hil/__init__.py` +- Create: `benchmarks/twin2silicon/hil/results.py` +- Create: `benchmarks/twin2silicon/hil/process.py` +- Create: `tests/twin2silicon-hil.py` + +- [ ] **Step 1: Write failing tests for statuses, atomic evidence, hashing, timeout, and process-group cleanup** + +Add tests that import `CommandResult`, `RunResult`, `sha256_file`, `write_json_atomic`, and `run_command`. Assert that: + +```python +def test_run_result_keeps_infrastructure_separate_from_candidate_failure(self): + result = RunResult.infrastructure_error("board_identity", "wrong adapter") + self.assertEqual(result.hardware_status, "not_run") + self.assertEqual(result.infrastructure_status, "error") + self.assertEqual(result.failure_category, "board_identity") + +def test_run_command_times_out_and_captures_evidence(self): + result = run_command( + [sys.executable, "-c", "import time; time.sleep(10)"], + cwd=self.tmp, + timeout_seconds=0.1, + stdout_path=self.tmp / "stdout.log", + stderr_path=self.tmp / "stderr.log", + ) + self.assertTrue(result.timed_out) + self.assertIsNotNone(result.duration_seconds) + self.assertNotEqual(result.returncode, 0) +``` + +- [ ] **Step 2: Run the tests and verify RED** + +Run: `python3 tests/twin2silicon-hil.py -k 'run_result or run_command'` + +Expected: import failure because `benchmarks.twin2silicon.hil` does not exist. + +- [ ] **Step 3: Implement minimal typed results and bounded command execution** + +Use frozen dataclasses and literal string statuses. `run_command()` must use `subprocess.Popen(..., start_new_session=True)`, `communicate(timeout=...)`, then `os.killpg(process.pid, signal.SIGTERM)` followed by a bounded SIGKILL fallback. It returns command, sanitized cwd, return code, timeout flag, start/end UTC timestamps, duration, and log paths. `write_json_atomic()` writes a sibling temporary file, fsyncs it, and replaces the destination. + +The constructors must encode these exact distinctions: + +```python +@classmethod +def infrastructure_error(cls, category: str, detail: str) -> "RunResult": + return cls( + model_status="not_run", + compile_status="not_run", + simulator_status="not_supported", + hardware_status="not_run", + infrastructure_status="error", + failure_category=category, + detail=detail, + ) +``` + +- [ ] **Step 4: Run focused tests and verify GREEN** + +Run: `python3 tests/twin2silicon-hil.py -k 'run_result or run_command or atomic or sha256'` + +Expected: all selected tests pass and the timeout test completes in under two seconds. + +- [ ] **Step 5: Commit** + +```bash +git add benchmarks/twin2silicon/hil tests/twin2silicon-hil.py +git commit -m "feat(bench): add bounded HIL result primitives" +``` + +### Task 2: Add the ESP32-S3 task fixture + +**Files:** +- Create: `benchmarks/twin2silicon/tasks/esp32s3-gpio-hil-001/task.json` +- Create: `benchmarks/twin2silicon/tasks/esp32s3-gpio-hil-001/hidden/hil-oracle.json` +- Create: `benchmarks/twin2silicon/tasks/esp32s3-gpio-hil-001/public/README.md` +- Create: `benchmarks/twin2silicon/tasks/esp32s3-gpio-hil-001/public/firmware/platformio.ini` +- Create: `benchmarks/twin2silicon/tasks/esp32s3-gpio-hil-001/public/firmware/sdkconfig.defaults` +- Create: `benchmarks/twin2silicon/tasks/esp32s3-gpio-hil-001/public/firmware/src/main.c` +- Test: `tests/twin2silicon-hil.py` + +- [ ] **Step 1: Write a failing fixture-contract test** + +Load both JSON files and assert schema `1.0`, task id `esp32s3-gpio-hil-001`, board `esp32-s3-devkitc-1`, framework `espidf`, a 50,000-token cap, zero diagnostic HIL calls for the model, UART prefix `LABWIRED_READY:`, and exactly these GPIO2 assertions: + +```json +[ + {"name":"gpio2_output_enabled","address":"0x60004020","mask":"0x00000004","expected":"0x00000004"}, + {"name":"gpio2_output_high","address":"0x60004004","mask":"0x00000004","expected":"0x00000004"} +] +``` + +Also assert that the public tree contains no expected register values or JTAG commands, and that `main.c` initially calls `gpio_set_direction(TEST_GPIO, GPIO_MODE_INPUT)`. + +- [ ] **Step 2: Run the fixture test and verify RED** + +Run: `python3 tests/twin2silicon-hil.py -k fixture_contract` + +Expected: failure because the task directory is absent. + +- [ ] **Step 3: Create the minimal ESP-IDF fixture** + +Pin PlatformIO's `espressif32` platform to the installed, lockable version selected by `pio pkg list`; set `board = esp32-s3-devkitc-1`, `framework = espidf`, `monitor_speed = 115200`, and `board_upload.flash_size = 4MB`. Configure USB Serial/JTAG as the primary console in `sdkconfig.defaults`. + +In `main.c`, define `TEST_GPIO GPIO_NUM_2`, include generated `run_nonce.h`, set the initial deliberate defect to input mode, call `gpio_set_level(TEST_GPIO, 1)`, print `LABWIRED_READY:%s\n`, flush stdout, and remain alive. The intended one-line repair is `GPIO_MODE_OUTPUT`. + +- [ ] **Step 4: Verify the fixture starts red semantically and builds** + +Run: + +```bash +python3 tests/twin2silicon-hil.py -k fixture_contract +pio run -d benchmarks/twin2silicon/tasks/esp32s3-gpio-hil-001/public/firmware +``` + +Expected: contract passes and PlatformIO reports `SUCCESS`; the source still configures GPIO2 as input. + +- [ ] **Step 5: Commit** + +```bash +git add benchmarks/twin2silicon/tasks/esp32s3-gpio-hil-001 tests/twin2silicon-hil.py +git commit -m "test(bench): add ESP32-S3 GPIO repair fixture" +``` + +### Task 3: Implement board identity, locking, UART, and JTAG evaluation + +**Files:** +- Create: `benchmarks/twin2silicon/hil/esp32s3.py` +- Modify: `tests/twin2silicon-hil.py` + +- [ ] **Step 1: Write failing offline hardware-adapter tests** + +Use temporary executable fixtures and `pty.openpty()` to cover: + +- exactly one configured JTAG serial succeeds; +- absent, wrong, or duplicate serials return `board_identity` infrastructure errors before flash; +- a second evaluator cannot acquire the same board lock; +- UART accepts only `LABWIRED_READY:` and rejects absent, wrong, and stale nonces; +- flash nonzero exit is a candidate failure only after identity succeeds; +- OpenOCD text is parsed only for explicitly named `@@REG
` records; +- masked register mismatch returns hardware `fail`; +- all assertions matching returns hardware `pass`; +- timeouts close the pseudo-terminal, terminate children, and release the lock. + +- [ ] **Step 2: Run the adapter tests and verify RED** + +Run: `python3 tests/twin2silicon-hil.py -k 'identity or lock or uart or jtag or register or flash'` + +Expected: import failure for `hil.esp32s3`. + +- [ ] **Step 3: Implement the adapter with dependency-injected commands** + +Define `Esp32S3Config`, `RegisterAssertion`, `BoardLock`, `validate_identity()`, `capture_uart_nonce()`, `flash_firmware()`, `read_registers()`, and `evaluate_registers()`. Commands come from the hidden descriptor or explicit CLI overrides; no shell interpolation is permitted. + +Identity validation consumes a command returning one serial per line and requires `matches == [expected_serial]`. OpenOCD receives: + +```text +adapter serial ; adapter speed 4000; init; reset run; sleep 750; +halt; echo "@@REG gpio2_output_enabled 0x60004020"; mdw 0x60004020 1; +echo "@@REG gpio2_output_high 0x60004004"; mdw 0x60004004 1; exit +``` + +The parser pairs each marker only with the immediately following OpenOCD memory word line and rejects missing, duplicate, or unrequested observations. + +- [ ] **Step 4: Run adapter tests and verify GREEN** + +Run: `python3 tests/twin2silicon-hil.py -k 'identity or lock or uart or jtag or register or flash'` + +Expected: all selected tests pass without accessing `/dev` or a network. + +- [ ] **Step 5: Commit** + +```bash +git add benchmarks/twin2silicon/hil/esp32s3.py tests/twin2silicon-hil.py +git commit -m "feat(bench): evaluate ESP32-S3 UART and JTAG evidence" +``` + +### Task 4: Compose the benchmark CLI and evidence manifest + +**Files:** +- Create: `benchmarks/twin2silicon/run_hil.py` +- Modify: `tests/twin2silicon-hil.py` + +- [ ] **Step 1: Write failing end-to-end fixture tests** + +Run the CLI against fake identity, PlatformIO, UART, and OpenOCD endpoints. Assert: + +- `--evaluate-only` copies a candidate, injects a 128-bit nonce header, and never exposes `hidden/` beneath the workspace; +- complete evidence yields exit 0 with hardware `pass`; +- a valid model failure also yields exit 0 with hardware `fail`; +- infrastructure failure yields exit 2 and is excluded from valid aggregate counts; +- manifest paths are relative, environment values are allowlisted, and secret-looking values never appear; +- every raw artifact listed in `run.json` has a matching SHA-256; +- rerunning into an existing run directory is refused; +- model budgets and provider usage supplied through `--usage-json` are preserved and cost arithmetic is exact. + +- [ ] **Step 2: Run CLI tests and verify RED** + +Run: `python3 tests/twin2silicon-hil.py -k 'cli or manifest or secret or cost'` + +Expected: failure because `run_hil.py` is absent. + +- [ ] **Step 3: Implement the CLI** + +Support these explicit modes and arguments: + +```text +run_hil.py TASK --run-dir DIR --evaluate-only --candidate DIR +run_hil.py TASK --run-dir DIR --agent-bin PATH --model MODEL + --jtag-serial SERIAL --uart-device DEVICE --openocd PATH + --usage-json FILE +``` + +The Agent mode copies `public/`, injects `include/run_nonce.h`, invokes `labwired-agent agent run` in the workspace with the public README as prompt, and then evaluates the resulting candidate. `--evaluate-only` skips model invocation but uses the same build/HIL/evidence path. Require explicit `--jtag-serial` and `--uart-device` for live operation; tests may inject fixture commands. + +Write `run.json` atomically after every phase so interrupted runs remain diagnosable, then finalize hashes only after all logs close. Use exit 0 for completed pass/fail and exit 2 for invalid infrastructure. + +- [ ] **Step 4: Run all offline Python tests and verify GREEN** + +Run: `python3 tests/twin2silicon-hil.py` + +Expected: all tests pass; no test opens a real serial device, invokes real PlatformIO, or contacts a model API. + +- [ ] **Step 5: Commit** + +```bash +git add benchmarks/twin2silicon/run_hil.py tests/twin2silicon-hil.py +git commit -m "feat(bench): orchestrate reproducible ESP32-S3 HIL runs" +``` + +### Task 5: Register tests and operator documentation + +**Files:** +- Create: `tests/twin2silicon-hil.sh` +- Create: `tests/twin2silicon-hil-live.sh` +- Modify: `tests/all.sh` +- Modify: `package.json` +- Modify: `docs/TESTING.md` + +- [ ] **Step 1: Write the shell entrypoints and registration assertions first** + +`tests/twin2silicon-hil.sh` runs `python3 tests/twin2silicon-hil.py`. `tests/twin2silicon-hil-live.sh` refuses to run unless `LABWIRED_HIL_DESTRUCTIVE=1`, `ESP32S3_SERIAL`, and `ESP32S3_UART` are set, then calls `run_hil.py --evaluate-only` with the known-good repaired fixture copy. + +Add static assertions that ordinary `tests/all.sh` includes only the offline script and never the live script. + +- [ ] **Step 2: Run registration assertions and verify RED** + +Run: `bash tests/twin2silicon-hil.sh` + +Expected: failure until package/test registration assertions are satisfied. + +- [ ] **Step 3: Register the offline lane and document the destructive lane** + +Add to `tests/all.sh`: + +```bash +run "twin2silicon-hil" "$ROOT/tests/twin2silicon-hil.sh" +``` + +Add package scripts: + +```json +"test:twin2silicon-hil": "bash tests/twin2silicon-hil.sh", +"test:twin2silicon-hil:live": "bash tests/twin2silicon-hil-live.sh" +``` + +Document the exact live command, destructive warning, required board serial/UART variables, result semantics, and evidence directory. + +- [ ] **Step 4: Verify offline registration and syntax** + +Run: + +```bash +bash -n tests/twin2silicon-hil.sh tests/twin2silicon-hil-live.sh +npm run test:twin2silicon-hil +``` + +Expected: syntax clean and all offline HIL tests pass. + +- [ ] **Step 5: Commit** + +```bash +git add tests/twin2silicon-hil.sh tests/twin2silicon-hil-live.sh tests/all.sh package.json docs/TESTING.md +git commit -m "test(bench): register ESP32-S3 HIL benchmark lanes" +``` + +### Task 6: Run the destructive connected-board acceptance test + +**Files:** +- Runtime artifacts only: `out/twin2silicon/esp32s3-gpio-hil-001-/` + +- [ ] **Step 1: Verify prerequisites without modifying the board** + +Run: + +```bash +pio --version +/private/tmp/openocd-esp32/bin/openocd --version +system_profiler SPUSBDataType | grep -A20 'USB JTAG/serial debug unit' +ls -l /dev/cu.usbmodem* +``` + +Expected: PlatformIO and Espressif OpenOCD are available, JTAG serial `9C:CC:01:D0:98:E0` is present exactly once, and the chosen S3 UART device exists. + +- [ ] **Step 2: Prove the live gate refuses implicit destructive execution** + +Run: `bash tests/twin2silicon-hil-live.sh` + +Expected: nonzero exit explaining `LABWIRED_HIL_DESTRUCTIVE=1` is required; no flash command runs. + +- [ ] **Step 3: Run a fresh known-good physical acceptance** + +Copy the public fixture to a temporary candidate, change only `GPIO_MODE_INPUT` to `GPIO_MODE_OUTPUT`, then run: + +```bash +LABWIRED_HIL_DESTRUCTIVE=1 \ +ESP32S3_SERIAL='9C:CC:01:D0:98:E0' \ +ESP32S3_UART='/dev/cu.usbmodem11101' \ +bash tests/twin2silicon-hil-live.sh +``` + +Expected: clean build and flash succeed, current nonce is observed, both GPIO2 masked assertions pass, `hardware_status` is `pass`, and evidence hashes validate. + +- [ ] **Step 4: Run the complete offline regression suite** + +Run: + +```bash +npm run test:twin2silicon-hil +bash tests/twin2silicon-fixture.sh +git diff --check +``` + +Expected: both benchmark suites pass and the diff check is clean. If the legacy fixture requires `LABWIRED_CLI`, provide the same simulator binary used by its existing documented invocation. + +- [ ] **Step 5: Record acceptance evidence without committing runtime output** + +Confirm `out/` remains untracked. Report the run directory, manifest link, firmware hash, nonce assertion, JTAG observations, elapsed time, and whether any infrastructure retry occurred. + +### Task 7: Final review and branch handoff + +**Files:** +- Review all files changed since design commit `4cbf6eb`. + +- [ ] **Step 1: Review requirements against the approved design** + +Check that model execution and evaluation are separated, hidden values never enter the workspace, missing hardware is not a model failure, board identity is explicit, subprocesses and locks are bounded, evidence is hash-addressed, and the live test is absent from ordinary CI. + +- [ ] **Step 2: Run final verification from a clean process** + +Run: + +```bash +npm run test:twin2silicon-hil +python3 -m py_compile benchmarks/twin2silicon/run_hil.py benchmarks/twin2silicon/hil/*.py tests/twin2silicon-hil.py +git diff --check 4cbf6eb..HEAD +git status --short +``` + +Expected: tests and compilation pass, diff is clean, and only the pre-existing untracked `out/` remains. + +- [ ] **Step 3: Request code review and address only verified findings** + +Review the implementation for specification compliance first and code quality second. Reproduce every blocking finding with a focused failing test before changing production behavior. + +- [ ] **Step 4: Prepare PR-only handoff** + +Push the feature branch and open a draft PR. Do not merge or deploy; the user previously requested PR-only delivery for repository changes. + diff --git a/docs/superpowers/specs/2026-08-15-cross-runtime-hil-smoke-design.md b/docs/superpowers/specs/2026-08-15-cross-runtime-hil-smoke-design.md new file mode 100644 index 0000000..eb483fe --- /dev/null +++ b/docs/superpowers/specs/2026-08-15-cross-runtime-hil-smoke-design.md @@ -0,0 +1,159 @@ +# Cross-Runtime HIL Smoke Test Design + +## Goal + +Measure how effectively OpenCode, Codex CLI, and Claude Code use the same +LabWired hardware context to repair one ESP32-S3 firmware defect. Each runtime +uses its native/default model. The existing physical HIL oracle, not the agent, +decides success. + +## Scope + +The first smoke test runs one fresh trial for each runtime on +`esp32s3-gpio-hil-001`. It proves adapter portability and produces comparable +evidence before the benchmark grows to more tasks or repeated trials. + +This work does not add another orchestration framework, sandbox, scheduler, +leaderboard service, or generalized experiment engine. It does not claim model +or runtime superiority from three single trials. + +## Experimental Controls + +All three trials receive the same: + +- public task directory and seeded defect; +- model-neutral repair prompt; +- LabWired firmware skill instructions; +- LabWired MCP server where the runtime supports MCP; +- wall-time and repair-attempt budgets; +- physical ESP32-S3, UART port, JTAG identity, nonce policy, and hidden oracle; +- build, flash, UART, and register scoring implementation. + +The runtime and its native/default model are the independent variable. Built-in +runtime tools, context management, system prompts, and token accounting are +part of the runtime being measured and must be disclosed rather than hidden. + +## Architecture + +The implementation adds three thin candidate-generation adapters behind one +small command-line interface: + +```text +fresh public task + | + +-- OpenCode adapter ----+ + +-- Codex adapter -------+--> candidate workspace --> existing run_hil.py + +-- Claude adapter ------+ | + +--> run.json + +--> cost.json +``` + +An adapter may prepare runtime-native skill or MCP configuration, launch the +runtime, and normalize its usage output. It must not compile, flash, inspect +hidden files, score hardware, or reinterpret the oracle result in controller +code. Runtime-native compilation during repair is allowed, but final scoring +always uses `benchmarks/twin2silicon/run_hil.py`. + +## Adapter Contract + +Each adapter accepts: + +- runtime name; +- public task directory; +- fresh output directory; +- shared prompt file; +- wall-time limit. + +Each adapter produces: + +- `candidate/`: the runtime's final public workspace; +- `agent.stdout.log` and `agent.stderr.log`; +- `agent-result.json`: normalized runtime status and timing; +- `usage.json`: normalized token rates and estimated cost when the runtime + exposes enough information; otherwise explicit `null` fields and a reason. + +`agent-result.json` contains the runtime, reported native model when available, +exit status, timeout flag, elapsed seconds, repair status, and paths to raw +logs. Raw runtime output remains available for later auditing. + +## Skills and MCP + +The canonical LabWired firmware instructions remain in the repository. Each +adapter maps them into the runtime's supported instruction mechanism without +rewriting their substance: + +- OpenCode uses the existing LabWired skills and MCP configuration. +- Codex receives repository instructions plus the LabWired MCP registration. +- Claude Code receives repository instructions plus the LabWired MCP + registration. + +Tool names may differ because runtimes namespace MCP tools differently. The +adapter may map names, but the underlying LabWired MCP server and tool behavior +must be identical. + +## Evaluation Flow + +For each runtime, the controller: + +1. copies only the public task into a fresh workspace; +2. installs the runtime-specific instruction and MCP adapter; +3. invokes the native/default model with the shared prompt and time limit; +4. records raw output, normalized timing, usage, and cost; +5. passes the final candidate to the existing HIL runner; +6. records the authoritative compile, flash, UART, and register result; +7. emits one comparison row without changing or repairing the candidate. + +An adapter failure is recorded as an infrastructure failure. A clean agent exit +with an incorrect candidate is a benchmark failure. Missing token accounting is +reported as unknown rather than estimated from unrelated data. + +## Comparison Output + +The smoke test emits a JSON summary and a concise table with: + +- runtime and reported native model; +- agent exit/timeout status; +- compile and physical HIL status; +- final success; +- elapsed agent and HIL time; +- repair/tool-call counts when observable; +- fresh, cached, reasoning, and output tokens when observable; +- API or subscription cost when observable; +- invalid tool calls and infrastructure error category. + +Values unavailable from a native runtime are `null` with a reason. Subscription +access is not converted into a fictitious per-run API price. + +## Error Handling + +Every trial uses a new output directory and bounded subprocess execution. A +failed adapter cannot prevent the remaining runtime trials from running. The +controller never retries a model silently; every extra model attempt would be a +new recorded trial. It refuses to overwrite an existing trial directory and +never modifies the public fixture or hidden oracle. + +## Testing + +Offline tests use fake runtime executables to verify: + +- identical fresh candidate inputs; +- exact native/default-model behavior (no model override); +- prompt and instruction delivery; +- bounded timeout and nonzero-exit classification; +- normalized output with explicit unknown usage; +- continued execution after one adapter fails; +- invocation of the existing HIL boundary without exposing hidden files to an + adapter. + +One opt-in connected-board smoke test then runs the three installed runtimes on +the ESP32-S3 fixture. Its results are evidence, not a unit-test prerequisite. + +## Acceptance Criteria + +- OpenCode, Codex CLI, and Claude Code each run through a thin native adapter. +- No adapter specifies a non-native model override. +- All candidates are created from identical public bytes. +- All final candidates are scored by the unchanged HIL entry point. +- Results use one normalized schema while retaining raw logs. +- Unknown tokens or costs are explicit and never fabricated. +- Existing HIL tests remain green and `out/` remains untouched. diff --git a/docs/superpowers/specs/2026-08-15-esp32s3-hil-benchmark-design.md b/docs/superpowers/specs/2026-08-15-esp32s3-hil-benchmark-design.md new file mode 100644 index 0000000..518d3b0 --- /dev/null +++ b/docs/superpowers/specs/2026-08-15-esp32s3-hil-benchmark-design.md @@ -0,0 +1,105 @@ +# ESP32-S3 HIL Benchmark Design + +**Date:** 2026-08-15 + +## Objective + +Build the first end-to-end Twin2Silicon hardware-in-the-loop repair benchmark on the connected ESP32-S3. A model run is successful only when its candidate firmware builds, flashes to the intended board, emits a run-specific UART nonce, and leaves the physical GPIO peripheral in the hidden oracle's expected state as observed through USB-JTAG. + +This specification covers one reusable evaluator and one task, `esp32s3-gpio-hil-001`. A larger task suite, scheduled lab service, and multi-model leaderboard are separate follow-on work. + +## Task + +The public task is an ESP-IDF C firmware project built through PlatformIO. It contains one realistic GPIO configuration defect and a concise repair prompt. The model may modify only the copied public workspace. The task's hidden descriptor contains: + +- the board profile and PlatformIO environment; +- the expected USB-JTAG serial identity, supplied by evaluator configuration rather than model context; +- the UART device-selection rule and baud rate; +- the flash artifact location; +- a GPIO register read plan with address, mask, and expected masked value; +- build, flash, UART, JTAG, wall-time, token, and iteration limits. + +The evaluator injects a cryptographically random run nonce into a generated header before the model starts. Correct firmware prints `LABWIRED_READY:`. A fixed string or output from an earlier run cannot satisfy the UART oracle. + +## Architecture + +The implementation consists of four focused units: + +1. **Task runner.** Creates a run directory and isolated public workspace, injects the nonce, invokes the existing public LabWired Agent, enforces model budgets, and records provider usage. +2. **Build and flash adapter.** Runs pinned PlatformIO commands, identifies the produced firmware, flashes only after board identity validation, and preserves command logs and exit metadata. +3. **UART/JTAG HIL evaluator.** Acquires an exclusive board lock, validates the expected Espressif USB-JTAG adapter, captures UART until the nonce or timeout, halts the target with Espressif OpenOCD, reads the hidden GPIO register plan, and evaluates masked values. +4. **Evidence writer.** Produces one canonical result manifest plus immutable raw logs and SHA-256 hashes. + +The runner orchestrates these units in this order: + +```text +prepare workspace + -> model repair + -> clean build + -> acquire board lock + -> validate board identity + -> start UART capture + -> flash and reset + -> observe nonce + -> halt and read GPIO registers + -> evaluate oracle + -> release board lock + -> write evidence +``` + +Simulator scoring remains a separate field. It may be `not_supported` for this first ESP-IDF task and cannot substitute for the physical HIL result. + +## Hardware Safety and Identity + +The evaluator is destructive to the board's installed firmware, as explicitly authorized. It must not operate on an ambiguous target. + +- The expected USB-JTAG serial is an explicit evaluator input. The currently connected S3 reports `9C:CC:01:D0:98:E0`, but the repository does not treat that machine-specific value as a universal default. +- Zero or multiple matching adapters produces `infrastructure_error`; the evaluator does not flash. +- UART selection must resolve to exactly one device associated with the target profile. A CLI override is allowed for laboratory setup and is recorded in redacted form in the manifest. +- A filesystem lock keyed by board identity prevents concurrent flash/JTAG runs. Lock acquisition has a timeout and records contention as infrastructure failure. +- Every subprocess has a hard timeout and process-group cleanup. Signal handling releases ports, OpenOCD, and the board lock. + +## Result Contract + +The canonical `run.json` records: + +- schema, run, task, harness, and model identifiers; +- model request count, fresh/cached/output/reasoning tokens, final context size, latency, and estimated provider cost; +- configured budgets and whether each was respected; +- `model_status`, `compile_status`, `simulator_status`, `hardware_status`, and `infrastructure_status` as independent fields; +- UART nonce result and JTAG register assertions without exposing hidden expectations to the model; +- termination reason and normalized failure category; +- hashes for the candidate source tree, firmware, oracle descriptor, raw logs, and evaluator result. + +`hardware_status` is `pass` only when the clean build, flash, nonce, and every required register assertion pass. Missing tools, missing hardware, ambiguous identity, port contention, and transport failure are `infrastructure_error`, never model failures. A candidate that builds and flashes but misses the nonce or register oracle is `fail`. + +The runner exits zero only for a valid completed evaluation, whether the model passes or fails. It uses a distinct nonzero exit for invalid or incomplete infrastructure runs so batch aggregation cannot silently count them. + +## Evidence and Reproducibility + +Each run directory contains the public prompt, initial and final source hashes, exact harness revision, sanitized tool versions, model/provider identity, budgets, command timing, raw build/flash/UART/OpenOCD logs, parsed register observations, `run.json`, and `cost.json`. + +Secrets, bearer tokens, full environment dumps, and hidden oracle values are not copied into model-visible files. Published results may include the oracle after the evaluation set is retired; active hidden tasks publish only schema and aggregate assertion outcomes. + +Pricing is versioned with source URL, effective date, and fresh-input, cached-input, and output rates. Cost remains an estimate unless reconciled against a provider invoice. + +## Testing + +Offline tests drive the real orchestration boundary with fixture executables for PlatformIO, serial capture, and OpenOCD. They cover: + +- a complete pass; +- compilation and flash failures; +- absent, incorrect, and stale UART nonces; +- incorrect GPIO masked values; +- absent, wrong, and ambiguous board identity; +- UART, JTAG, and lock timeouts; +- subprocess interruption and cleanup; +- correct separation of model failure and infrastructure error; +- stable JSON schema, evidence hashes, and cost arithmetic. + +Tests are written before production behavior and must demonstrate the expected red failure before implementation. The live acceptance test is opt-in, names the target serial and UART explicitly, overwrites the board firmware, and is excluded from ordinary CI. Acceptance requires one fresh physical run that produces the nonce and passing GPIO register evidence. + +## Non-Goals + +This first increment does not add external voltage or waveform instrumentation, continuous unattended lab scheduling, automatic firmware restoration, more than one task, or GPT/Claude/Grok comparison runs. Those additions build on the evaluator contract after the connected ESP32-S3 path is proven. + diff --git a/fixtures/twin2silicon/native-002/claude.stdout.jsonl b/fixtures/twin2silicon/native-002/claude.stdout.jsonl new file mode 100644 index 0000000..0afd5db --- /dev/null +++ b/fixtures/twin2silicon/native-002/claude.stdout.jsonl @@ -0,0 +1,2 @@ +{"type":"system","subtype":"init","model":"claude-sonnet-4-20250514"} +{"type":"result","subtype":"success","usage":{"input_tokens":5732,"cache_read_input_tokens":2048,"output_tokens":961,"output_tokens_details":{"thinking_tokens":417}},"total_cost_usd":0.042} diff --git a/fixtures/twin2silicon/native-002/codex.stdout.jsonl b/fixtures/twin2silicon/native-002/codex.stdout.jsonl new file mode 100644 index 0000000..46c7f98 --- /dev/null +++ b/fixtures/twin2silicon/native-002/codex.stdout.jsonl @@ -0,0 +1,2 @@ +{"type":"thread.started","thread_id":"thread-native-002"} +{"type":"turn.completed","usage":{"input_tokens":16431,"cached_input_tokens":12288,"reasoning_output_tokens":621,"output_tokens":2892}} diff --git a/package.json b/package.json index 2b9a648..165d1e6 100644 --- a/package.json +++ b/package.json @@ -148,9 +148,12 @@ "test:develop:release": "LABWIRED_ACCEPTANCE_REQUIRE_COMPLETE=1 bash tests/develop-acceptance-smoke.sh", "test:dispatcher": "bash tests/dispatcher.sh", "test:agent-lifecycle": "bash tests/agent-lifecycle.sh", + "test:tool-names": "bash tests/public-tool-names.sh", "test:public-install-safety": "bash tests/public-install-safety.sh", "test:install": "bash tests/install-smoke.sh", - "test:llm": "bash tests/llm-deepinfra.sh" + "test:llm": "bash tests/llm-deepinfra.sh", + "test:runtime-smoke:offline": "python3 tests/twin2silicon-hil.py", + "test:runtime-smoke:hardware": "bash tests/twin2silicon-runtime-smoke.sh" }, "publishConfig": { "access": "public" diff --git a/tests/all.sh b/tests/all.sh index ca01fe4..bb85f83 100755 --- a/tests/all.sh +++ b/tests/all.sh @@ -26,6 +26,7 @@ run "skills-verify-all" "$ROOT/tests/skills-verify-all.sh" run "develop-skill" "$ROOT/tests/develop-skill.sh" run "develop-acceptance-smoke" "$ROOT/tests/develop-acceptance-smoke.sh" run "hosted-config" "$ROOT/tests/hosted-config.sh" +run "public-tool-names" "$ROOT/tests/public-tool-names.sh" run "hosted-auth-probe" "$ROOT/tests/hosted-auth-probe.sh" run "agents-tool-search" "$ROOT/tests/agents-tool-search.sh" run "desktop-session" "$ROOT/tests/desktop-session.sh" diff --git a/tests/hosted-config.sh b/tests/hosted-config.sh index b00e0d2..72c5082 100644 --- a/tests/hosted-config.sh +++ b/tests/hosted-config.sh @@ -18,10 +18,21 @@ test -f "$ROOT/config/opencode.json" || bad "missing opencode.json" python3 - <_. The +# hosted profile strips the raw server prefix, so model-facing names remain +# canonical instead of becoming labwired_labwired_*. +raw_name = "labwired_context" +wire_name = raw_name.removeprefix("labwired_") +model_name = f"labwired_{wire_name}" +assert model_name == raw_name, model_name +assert not model_name.startswith("labwired_labwired_"), model_name # Bearer header is the product auth path — do not open OpenCode MCP OAuth /connect. assert mcp.get("oauth") is False, mcp auth = (mcp.get("headers") or {}).get("Authorization", "") diff --git a/tests/public-tool-names.sh b/tests/public-tool-names.sh new file mode 100644 index 0000000..4b7e4e7 --- /dev/null +++ b/tests/public-tool-names.sh @@ -0,0 +1,152 @@ +#!/usr/bin/env bash +# Resolve the real OpenCode registry through the public LabWired launcher, with +# both MCP and model traffic confined to one stdlib-only localhost fixture. +set -euo pipefail +ROOT="$(cd "$(dirname "$0")/.." && pwd)" +TMP="$(mktemp -d)" +cleanup() { + if [[ -n "${SERVER_PID:-}" ]]; then + kill "$SERVER_PID" 2>/dev/null || true + wait "$SERVER_PID" 2>/dev/null || true + fi + rm -rf "$TMP" +} +trap cleanup EXIT + +PORT_FILE="$TMP/port" +CAPTURE_FILE="$TMP/model-request.json" +python3 - "$PORT_FILE" "$CAPTURE_FILE" <<'PY' & +import json, sys +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer +from pathlib import Path +from urllib.parse import urlparse + +port_file, capture_file = map(Path, sys.argv[1:]) +class Handler(BaseHTTPRequestHandler): + def log_message(self, *_): + pass + + def do_GET(self): + if urlparse(self.path).path != "/v1/models": + self.send_response(404) + self.end_headers() + return + data = json.dumps({"object": "list", "data": [{"id": "labwired-default", "object": "model"}]}).encode() + self.send_response(200) + self.send_header("content-type", "application/json") + self.send_header("content-length", str(len(data))) + self.end_headers() + self.wfile.write(data) + + def do_POST(self): + length = int(self.headers.get("content-length", "0")) + body = json.loads(self.rfile.read(length) or b"{}") + parsed = urlparse(self.path) + if parsed.path == "/v1/chat/completions": + capture_file.write_text(json.dumps(body)) + payload = { + "id": "offline", "object": "chat.completion", "created": 0, "model": "offline", + "choices": [{"index": 0, "message": {"role": "assistant", "content": "done"}, "finish_reason": "stop"}], + "usage": {"prompt_tokens": 1, "completion_tokens": 1, "total_tokens": 2}, + } + else: + self.send_response(404) + self.end_headers() + return + data = json.dumps(payload).encode() + self.send_response(200) + self.send_header("content-type", "application/json") + self.send_header("content-length", str(len(data))) + self.end_headers() + self.wfile.write(data) + +server = ThreadingHTTPServer(("127.0.0.1", 0), Handler) +port_file.write_text(str(server.server_port)) +server.serve_forever() +PY +SERVER_PID=$! +for _ in $(seq 1 100); do + [[ -s "$PORT_FILE" ]] && break + sleep 0.05 +done +[[ -s "$PORT_FILE" ]] || { echo "FAIL offline fixture did not start" >&2; exit 1; } +PORT="$(cat "$PORT_FILE")" + +mkdir -p "$TMP/home" "$TMP/labwired" "$TMP/config" "$TMP/project" "$TMP/bin" +PORT="$PORT" TMP="$TMP" python3 - <<'PY' +import json, os +from pathlib import Path + +tmp = Path(os.environ["TMP"]) +port = os.environ["PORT"] +(tmp / "project/opencode.json").write_text(json.dumps({ + "$schema": "https://opencode.ai/config.json", + "provider": {"offline": { + "npm": "@ai-sdk/openai-compatible", + "name": "Offline", + "options": {"baseURL": f"http://127.0.0.1:{port}/v1", "apiKey": "offline"}, + "models": {"default": {"name": "Offline"}}, + }}, + "model": "offline/default", +})) +(tmp / "bin/npx").write_text("#!/bin/sh\nexec python3 \"$FAKE_MCP\"\n") +(tmp / "bin/npx").chmod(0o755) +(tmp / "fake-mcp.py").write_text(r'''import json, sys + +names = ["context", "compile", "run", "verify"] +for line in sys.stdin: + try: + request = json.loads(line) + except Exception: + continue + method = request.get("method") + if method == "initialize": + result = { + "protocolVersion": "2025-06-18", + "capabilities": {"tools": {}}, + "serverInfo": {"name": "offline-labwired", "version": "1"}, + } + elif method == "tools/list": + result = {"tools": [ + {"name": name, "description": name, "inputSchema": {"type": "object", "properties": {}}} + for name in names + ]} + elif method == "ping": + result = {} + else: + continue + print(json.dumps({"jsonrpc": "2.0", "id": request.get("id"), "result": result}), flush=True) +''') +PY + +if ! ( + cd "$TMP/project" + HOME="$TMP/home" \ + LABWIRED_HOME="$TMP/labwired" \ + OPENCODE_CONFIG_DIR="$TMP/config" \ + OPENCODE_CONFIG="$TMP/project/opencode.json" \ + FAKE_MCP="$TMP/fake-mcp.py" \ + PATH="$TMP/bin:$PATH" \ + LABWIRED_SKIP_LOGIN=1 \ + "$ROOT/bin/labwired-agent" agent run --model offline/default --format json "Reply done without calling tools." \ + >"$TMP/run.out" 2>"$TMP/run.err" +); then + cat "$TMP/run.out" >&2 + cat "$TMP/run.err" >&2 + echo "FAIL public launcher exited before model capture" >&2 + exit 1 +fi + +[[ -s "$CAPTURE_FILE" ]] || { cat "$TMP/run.err" >&2; echo "FAIL model request not captured" >&2; exit 1; } +CAPTURE_FILE="$CAPTURE_FILE" python3 - <<'PY' +import json, os +body = json.load(open(os.environ["CAPTURE_FILE"])) +names = [tool["function"]["name"] for tool in body.get("tools", []) if tool.get("type") == "function"] +expected = ["labwired_context", "labwired_compile", "labwired_run", "labwired_verify"] +for name in expected: + assert names.count(name) == 1, (name, names) +assert not any(name.startswith("labwired_labwired_") for name in names), names +print("ok public launcher sends canonical MCP tool names to the model provider") +PY + +echo "public-tool-names PASS" diff --git a/tests/skills-verify-all.sh b/tests/skills-verify-all.sh index ccde157..876b46a 100755 --- a/tests/skills-verify-all.sh +++ b/tests/skills-verify-all.sh @@ -70,6 +70,10 @@ for cfg in "$ROOT/config/opencode.json" "$ROOT/config/opencode.hosted.json" \ done done +deepinfra_base="$(python3 -c 'import json,sys; print(json.load(open(sys.argv[1]))["provider"]["deepinfra"]["options"]["baseURL"])' "$ROOT/config/opencode.deepinfra.json")" +[[ "$deepinfra_base" == "https://api.deepinfra.com/v1" ]] \ + && pass "DeepInfra OpenCode base URL" || bad "DeepInfra OpenCode base URL: $deepinfra_base" + n=$(find "$ROOT/skills" -mindepth 1 -maxdepth 1 -type d | wc -l | tr -d ' ') # 7 domain packs + customize-labwired-agent + 14 superpowers = 22 if [[ "$n" -eq 22 ]]; then diff --git a/tests/twin2silicon-fixture.sh b/tests/twin2silicon-fixture.sh new file mode 100755 index 0000000..0f49197 --- /dev/null +++ b/tests/twin2silicon-fixture.sh @@ -0,0 +1,22 @@ +#!/usr/bin/env bash +set -euo pipefail + +ROOT="$(cd "$(dirname "$0")/.." && pwd)" +TASK="$ROOT/benchmarks/twin2silicon/tasks/f103-gpio-clock-001" +TMP="$(mktemp -d)" +trap 'rm -rf "$TMP"' EXIT + +cp -R "$TASK/public/." "$TMP/" +make -C "$TMP/firmware" +python3 "$ROOT/benchmarks/twin2silicon/prepare-oracle.py" \ + "$TASK/hidden/oracle.yaml" "$TASK/hidden/system.yaml" \ + "$TMP/firmware/build/firmware.elf" "$TMP/oracle.yaml" + +if "${LABWIRED_CLI:?set LABWIRED_CLI}" test \ + --script "$TMP/oracle.yaml" --output-dir "$TMP/result"; then + echo "FAIL: buggy fixture unexpectedly passed" + exit 1 +fi + +grep -q '"status": "fail"' "$TMP/result/result.json" +echo "ok twin2silicon fixture starts red" diff --git a/tests/twin2silicon-hil.py b/tests/twin2silicon-hil.py new file mode 100644 index 0000000..dd9db45 --- /dev/null +++ b/tests/twin2silicon-hil.py @@ -0,0 +1,2122 @@ +#!/usr/bin/env python3 +import hashlib +import json +import os +from pathlib import Path +import pty +import signal +import subprocess +import sys +import tempfile +import textwrap +import threading +import time +import tty +from typing import get_args, get_origin, get_type_hints, Literal +import unittest +from unittest import mock + + +REPOSITORY_ROOT = Path(__file__).resolve().parents[1] +sys.path.insert(0, str(REPOSITORY_ROOT)) + +from benchmarks.twin2silicon.hil.process import run_command +from benchmarks.twin2silicon.hil import process as process_module +from benchmarks.twin2silicon.hil.results import ( + CommandResult, + RunResult, + sha256_file, + write_json_atomic, +) +from benchmarks.twin2silicon.hil.esp32s3 import ( + BoardLock, + BoardLockTimeout, + Esp32S3Config, + RegisterAssertion, + build_openocd_command, + capture_uart_nonce, + evaluate_registers, + flash_firmware, + parse_openocd_registers, + read_registers, + validate_identity, +) +from benchmarks.twin2silicon.runtime_adapters import ( + AdapterContext, + NormalizedUsage, + build_runtime_command, + codex_mcp_toml, + extract_native_model, + normalize_usage, + write_codex_mcp_config, +) + + +def executable_fixture(directory, body): + Path(directory).mkdir(parents=True, exist_ok=True) + path = Path(directory) / "fixture.py" + path.write_text("#!/usr/bin/env python3\n" + body, encoding="utf-8") + path.chmod(0o755) + return path + + +class RuntimeAdapterTests(unittest.TestCase): + def adapter_context(self, runtime, executable=None): + directory = Path("/tmp/runtime-adapter-test") + return AdapterContext( + runtime=runtime, + executable=executable or runtime, + workspace=directory / "workspace", + prompt="Complete the public firmware task.", + config_dir=directory / "config", + stdout_path=directory / "stdout.log", + stderr_path=directory / "stderr.log", + ) + + def test_build_runtime_commands_use_native_structured_modes(self): + contexts = { + "codex": self.adapter_context("codex"), + "claude": self.adapter_context("claude"), + "opencode": self.adapter_context("opencode"), + } + commands = { + runtime: build_runtime_command(runtime, context) + for runtime, context in contexts.items() + } + + self.assertEqual(commands["codex"][:2], ["codex", "exec"]) + self.assertEqual(commands["claude"][:2], ["claude", "--print"]) + self.assertEqual(commands["opencode"][:2], ["opencode", "run"]) + self.assertEqual( + commands["codex"], + [ + "codex", "exec", "--json", "--ephemeral", "--skip-git-repo-check", + "-s", "workspace-write", "-C", str(contexts["codex"].workspace), + contexts["codex"].prompt, + ], + ) + self.assertEqual( + commands["claude"], + [ + "claude", "--print", "--verbose", "--output-format", "stream-json", + "--no-session-persistence", "--permission-mode", "acceptEdits", + "--mcp-config", str(contexts["claude"].config_dir / "claude-mcp.json"), + "--strict-mcp-config", + contexts["claude"].prompt, + ], + ) + self.assertEqual( + commands["opencode"], + [ + "opencode", "run", "--format", "json", "--dir", + str(contexts["opencode"].workspace), contexts["opencode"].prompt, + ], + ) + self.assertEqual( + Path(commands["claude"][commands["claude"].index("--mcp-config") + 1]).parent, + contexts["claude"].config_dir, + ) + hidden_oracle = "benchmarks/twin2silicon/tasks/esp32s3-gpio-hil-001/hidden/hil-oracle.json" + for command in commands.values(): + with self.subTest(command=command[:2]): + self.assertNotIn("--model", command) + self.assertNotIn(hidden_oracle, " ".join(command)) + + def test_build_runtime_command_rejects_context_runtime_mismatch(self): + context = self.adapter_context("codex") + + with self.assertRaisesRegex(ValueError, "runtime does not match context"): + build_runtime_command("claude", context) + + def test_adapter_dataclasses_are_frozen(self): + context = self.adapter_context("codex") + usage = NormalizedUsage(None, None, None, None, None, None, None) + + with self.assertRaises((AttributeError, TypeError)): + context.prompt = "another prompt" + with self.assertRaises((AttributeError, TypeError)): + usage.output = 1 + + def test_normalize_opencode_step_finish_events(self): + lines = [ + json.dumps({ + "type": "step_finish", + "part": { + "tokens": {"input": 800, "cache": {"read": 200}, "reasoning": 20, "output": 500}, + "cost": 0.002476282, + }, + }), + json.dumps({ + "type": "step_finish", + "part": { + "tokens": {"input": 600, "cache": {"read": 300}, "reasoning": 28, "output": 576}, + "cost": 0.005, + }, + }), + ] + + usage = normalize_usage("opencode", lines) + + self.assertEqual(usage.requests, 2) + self.assertEqual(usage.fresh_input, 1400) + self.assertEqual(usage.cached_input, 500) + self.assertEqual(usage.reasoning, 48) + self.assertEqual(usage.output, 1076) + self.assertAlmostEqual(usage.estimated_cost_usd, 0.007476282) + self.assertIsNone(usage.unavailable_reason) + + def test_normalize_codex_uses_final_cumulative_usage(self): + lines = [ + json.dumps({"type": "turn.completed", "usage": { + "input_tokens": 1000, "cached_input_tokens": 200, + "reasoning_output_tokens": 10, "output_tokens": 500, + }}), + json.dumps({"type": "turn.completed", "usage": { + "input_tokens": 1400, "cached_input_tokens": 300, + "reasoning_output_tokens": 48, "output_tokens": 1076, + }}), + ] + + usage = normalize_usage("codex", lines) + + self.assertEqual(usage.requests, 1) + self.assertEqual(usage.fresh_input, 1100) + self.assertEqual(usage.cached_input, 300) + self.assertEqual(usage.reasoning, 48) + self.assertEqual(usage.output, 1028) + self.assertIsNone(usage.estimated_cost_usd) + self.assertIsNone(usage.unavailable_reason) + + def test_normalize_codex_native_002_usage_excludes_reasoning_from_output(self): + lines = (REPOSITORY_ROOT / "fixtures/twin2silicon/native-002/codex.stdout.jsonl").read_text().splitlines() + + usage = normalize_usage("codex", lines) + + self.assertEqual( + usage, + NormalizedUsage(1, 4143, 12288, 621, 2271, None, None), + ) + + def test_normalize_codex_rejects_cached_input_larger_than_total(self): + usage = normalize_usage("codex", [json.dumps({ + "type": "turn.completed", + "usage": { + "input_tokens": 10, + "cached_input_tokens": 11, + "reasoning_output_tokens": 2, + "output_tokens": 5, + }, + })]) + + self.assertEqual(usage.requests, 1) + self.assertIsNone(usage.fresh_input) + self.assertIsNone(usage.cached_input) + self.assertEqual(usage.reasoning, 2) + self.assertEqual(usage.output, 3) + + def test_normalize_codex_keeps_output_total_when_reasoning_is_absent(self): + usage = normalize_usage("codex", [json.dumps({ + "type": "turn.completed", + "usage": {"input_tokens": 10, "cached_input_tokens": 0, "output_tokens": 5}, + })]) + + self.assertIsNone(usage.reasoning) + self.assertEqual(usage.output, 5) + + def test_normalize_codex_rejects_reasoning_larger_than_total_output(self): + usage = normalize_usage("codex", [json.dumps({ + "type": "turn.completed", + "usage": { + "input_tokens": 10, + "cached_input_tokens": 0, + "reasoning_output_tokens": 6, + "output_tokens": 5, + }, + })]) + + self.assertIsNone(usage.reasoning) + self.assertIsNone(usage.output) + + def test_normalize_codex_uses_the_final_terminal_event(self): + lines = [ + json.dumps({"type": "turn.completed", "usage": { + "input_tokens": 1400, "output_tokens": 1076, + }}), + json.dumps({"type": "turn.completed"}), + ] + + usage = normalize_usage("codex", lines) + + self.assertEqual( + usage, + NormalizedUsage(None, None, None, None, None, None, "runtime did not expose usage"), + ) + + def test_normalize_codex_accepts_zero_token_usage(self): + usage = normalize_usage( + "codex", + [json.dumps({ + "type": "turn.completed", + "usage": {"input_tokens": 0, "output_tokens": 0}, + })], + ) + + self.assertEqual(usage.requests, 1) + self.assertEqual(usage.fresh_input, 0) + self.assertEqual(usage.output, 0) + self.assertIsNone(usage.unavailable_reason) + + def test_normalize_claude_final_result_usage_and_cost(self): + lines = [ + json.dumps({"type": "assistant", "message": {"content": []}}), + json.dumps({ + "type": "result", + "subtype": "success", + "usage": { + "input_tokens": 1400, + "cache_read_input_tokens": 300, + "output_tokens": 1076, + }, + "total_cost_usd": 0.007476282, + }), + ] + + usage = normalize_usage("claude", lines) + + self.assertEqual(usage.requests, 1) + self.assertEqual(usage.fresh_input, 1400) + self.assertEqual(usage.cached_input, 300) + self.assertIsNone(usage.reasoning) + self.assertEqual(usage.output, 1076) + self.assertAlmostEqual(usage.estimated_cost_usd, 0.007476282) + self.assertIsNone(usage.unavailable_reason) + + def test_normalize_claude_native_002_usage_excludes_thinking_from_output(self): + lines = (REPOSITORY_ROOT / "fixtures/twin2silicon/native-002/claude.stdout.jsonl").read_text().splitlines() + + usage = normalize_usage("claude", lines) + + self.assertEqual( + usage, + NormalizedUsage(1, 5732, 2048, 417, 544, 0.042, None), + ) + + def test_normalize_claude_rejects_thinking_larger_than_total_output(self): + usage = normalize_usage("claude", [json.dumps({ + "type": "result", + "usage": { + "input_tokens": 10, + "output_tokens": 3, + "output_tokens_details": {"thinking_tokens": 4}, + }, + })]) + + self.assertEqual(usage.requests, 1) + self.assertEqual(usage.fresh_input, 10) + self.assertIsNone(usage.reasoning) + self.assertIsNone(usage.output) + + def test_extract_native_model_accepts_only_runtime_event_model_fields(self): + self.assertEqual( + extract_native_model("claude", [json.dumps({ + "type": "system", "subtype": "init", "model": "claude-sonnet-4-20250514", + })]), + "claude-sonnet-4-20250514", + ) + self.assertEqual( + extract_native_model("codex", [json.dumps({ + "type": "turn.started", "model": "gpt-5-codex", + })]), + "gpt-5-codex", + ) + self.assertEqual( + extract_native_model("opencode", [json.dumps({ + "type": "step_start", "part": {"model": "deepseek-r1"}, + })]), + "deepseek-r1", + ) + self.assertIsNone(extract_native_model("codex", [json.dumps({ + "type": "turn.completed", "model": "untrusted", "usage": {}, + })])) + + def test_normalize_claude_uses_the_final_result_event(self): + lines = [ + json.dumps({ + "type": "result", + "usage": {"input_tokens": 1400, "output_tokens": 1076}, + "total_cost_usd": 0.007476282, + }), + json.dumps({"type": "result", "subtype": "success", "result": "complete"}), + ] + + usage = normalize_usage("claude", lines) + + self.assertEqual( + usage, + NormalizedUsage(None, None, None, None, None, None, "runtime did not expose usage"), + ) + + def test_normalize_claude_accepts_zero_token_usage(self): + usage = normalize_usage( + "claude", + [json.dumps({ + "type": "result", + "usage": {"input_tokens": 0, "output_tokens": 0}, + "total_cost_usd": 0.0, + })], + ) + + self.assertEqual(usage.requests, 1) + self.assertEqual(usage.fresh_input, 0) + self.assertEqual(usage.output, 0) + self.assertEqual(usage.estimated_cost_usd, 0.0) + self.assertIsNone(usage.unavailable_reason) + + def test_normalize_usage_ignores_malformed_lines(self): + usage = normalize_usage("opencode", ["not json", "{", "[]"]) + + self.assertEqual( + usage, + NormalizedUsage(None, None, None, None, None, None, "runtime did not expose usage"), + ) + + def test_normalize_opencode_rejects_aggregate_overflow(self): + lines = [ + json.dumps({ + "type": "step_finish", + "part": { + "tokens": {"output": 1_000_000_000_000}, + "cost": 1_000_000_000.0, + }, + }), + json.dumps({ + "type": "step_finish", + "part": {"tokens": {"output": 1}, "cost": 0.01}, + }), + ] + + usage = normalize_usage("opencode", lines) + + self.assertEqual(usage.requests, 2) + self.assertIsNone(usage.output) + self.assertIsNone(usage.estimated_cost_usd) + self.assertEqual(usage.unavailable_reason, "runtime did not expose usage") + + def test_normalize_usage_marks_success_without_accounting_unavailable(self): + usage = normalize_usage( + "claude", + [json.dumps({"type": "result", "subtype": "success", "result": "complete"})], + ) + + self.assertEqual(usage.estimated_cost_usd, None) + self.assertEqual(usage.unavailable_reason, "runtime did not expose usage") + + +class RuntimeConfigurationTests(unittest.TestCase): + def test_shared_instructions_bound_repairs_and_evidence_to_public_workspace(self): + instructions_path = ( + REPOSITORY_ROOT + / "benchmarks" + / "twin2silicon" + / "shared-agent-instructions.md" + ) + instructions = instructions_path.read_text(encoding="utf-8") + + self.assertLess(len(instructions.split()), 700) + for required_text in ( + "smallest firmware repair", + "public workspace", + "hidden oracle", + "self-grade", + "compile evidence", + "repair_iterations", + "optional context and compile aids", + "not the final oracle", + ): + with self.subTest(required_text=required_text): + self.assertIn(required_text, instructions) + + def test_runtime_mcp_configs_use_only_the_local_labwired_server(self): + config_root = REPOSITORY_ROOT / "benchmarks" / "twin2silicon" / "runtime-config" + opencode = json.loads((config_root / "opencode.json").read_text(encoding="utf-8")) + claude = json.loads((config_root / "claude-mcp.json").read_text(encoding="utf-8")) + + self.assertNotIn("model", opencode) + self.assertNotIn("provider", opencode) + self.assertEqual(set(opencode["mcp"]), {"labwired"}) + self.assertEqual(set(claude["mcpServers"]), {"labwired"}) + self.assertEqual(opencode["mcp"]["labwired"]["type"], "local") + self.assertEqual( + opencode["mcp"]["labwired"]["command"], + ["npx", "-y", "@labwired/mcp"], + ) + self.assertTrue(opencode["mcp"]["labwired"]["enabled"]) + installed_profile = json.loads( + (REPOSITORY_ROOT / "config" / "opencode.json").read_text(encoding="utf-8") + ) + self.assertEqual( + opencode["permission"]["skill"], + installed_profile["permission"]["skill"], + ) + self.assertEqual(claude["mcpServers"]["labwired"], { + "command": "npx", + "args": ["-y", "@labwired/mcp"], + }) + + def test_codex_mcp_toml_uses_the_local_labwired_server(self): + self.assertEqual( + codex_mcp_toml(), + '[mcp_servers.labwired]\ncommand = "npx"\nargs = ["-y", "@labwired/mcp"]\n', + ) + + def test_write_codex_mcp_config_creates_an_isolated_config_file(self): + with tempfile.TemporaryDirectory() as directory: + config_dir = Path(directory) / "runtime-config" + + config_path = write_codex_mcp_config(config_dir) + + self.assertEqual(config_path, config_dir / "config.toml") + self.assertEqual(config_path.read_text(encoding="utf-8"), codex_mcp_toml()) + + +class RuntimeMatrixTests(unittest.TestCase): + script = REPOSITORY_ROOT / "benchmarks" / "twin2silicon" / "run_matrix.py" + task = REPOSITORY_ROOT / "benchmarks" / "twin2silicon" / "tasks" / "esp32s3-gpio-hil-001" + + def _fake_agent(self, directory, usage_schema="normalized", child_returncode=0): + body = textwrap.dedent(f""" + import argparse + import json + from pathlib import Path + import shutil + + parser = argparse.ArgumentParser() + parser.add_argument("runtime") + parser.add_argument("--task", required=True) + parser.add_argument("--output", required=True) + parser.add_argument("--executable") + parser.add_argument("--timeout-seconds") + args = parser.parse_args() + output = Path(args.output) + task = Path(args.task) + output.mkdir(parents=True) + shutil.copytree(task / "public", output / "candidate") + status = {{"opencode": "completed", "codex": "failed", "claude": "completed"}}[args.runtime] + (output / "agent-result.json").write_text(json.dumps({{ + "schema_version": "1.0", "runtime": args.runtime, + "native_model": f"fake-{{args.runtime}}-model", + "native_model_unavailable_reason": None, + "status": status, "returncode": 0 if status == "completed" else 9, + "timed_out": False, "elapsed_seconds": 0.25, + }})) + usage = {{ + "requests": 1 if args.runtime != "codex" else None, + "fresh_input": 10 if args.runtime != "codex" else None, + "cached_input": 0 if args.runtime != "codex" else None, + "reasoning": None, + "output": 5 if args.runtime != "codex" else None, + "estimated_cost_usd": None, + "unavailable_reason": "runtime did not expose usage" if args.runtime == "codex" else None, + }} + if {usage_schema!r} == "hil-valid": + usage.update({{ + "tokens": {{"fresh_input": 10, "cached_input": 0, "output": 5}}, + "rates_usd_per_million": {{"fresh_input": 1, "cached_input": 1, "output": 1}}, + }}) + elif {usage_schema!r} == "hil-extra-invalid": + usage.update({{ + "tokens": {{"fresh_input": 10, "cached_input": 0, "output": 5, "reasoning": None}}, + "rates_usd_per_million": {{"fresh_input": 1, "cached_input": 1, "output": 1}}, + }}) + elif {usage_schema!r} == "partial": + usage["estimated_cost_usd"] = 0.000015 + (output / "usage.json").write_text(json.dumps(usage)) + (output / "agent-invocation.json").write_text(json.dumps(vars(args), sort_keys=True)) + raise SystemExit({child_returncode!r}) + """) + return executable_fixture(directory, body) + + def _fake_hil(self, directory): + body = textwrap.dedent(""" + import argparse + import json + from pathlib import Path + + parser = argparse.ArgumentParser() + parser.add_argument("task") + parser.add_argument("--run-dir", required=True) + parser.add_argument("--candidate", required=True) + parser.add_argument("--jtag-serial", required=True) + parser.add_argument("--uart-device", required=True) + parser.add_argument("--openocd", required=True) + parser.add_argument("--platformio") + parser.add_argument("--identity-command-json") + parser.add_argument("--usage-json") + args = parser.parse_args() + run_dir = Path(args.run_dir) + run_dir.mkdir(parents=True) + status = "pass" if Path(args.candidate).parent.name == "opencode" else "fail" + (run_dir / "run.json").write_text(json.dumps({ + "schema_version": "1.0", "status": status, + "compile_status": status, "cost": None, + })) + (run_dir / "hil-invocation.json").write_text(json.dumps(vars(args), sort_keys=True)) + """) + return executable_fixture(directory, body) + + def _run_cli(self, output, agent, hil, *extra, identity_command_json='["fake-identity"]'): + command = [ + sys.executable, str(self.script), "--task", str(self.task), + "--output", str(output), "--jtag-serial", "fake-jtag", + "--uart-device", "/dev/fake-uart", "--openocd", "/fake/openocd", + "--agent-script", str(agent), "--hil-script", str(hil), *extra, + ] + if identity_command_json is not None and "--identity-command-json" not in extra: + command.extend(("--identity-command-json", identity_command_json)) + return subprocess.run( + command, + cwd=REPOSITORY_ROOT, + text=True, + capture_output=True, + timeout=10, + ) + + def test_matrix_preserves_runtime_order_hashes_and_failure_isolation(self): + with tempfile.TemporaryDirectory() as directory: + root = Path(directory) + completed = self._run_cli( + root / "matrix", self._fake_agent(root / "agent"), self._fake_hil(root / "hil"), + ) + + self.assertEqual(completed.returncode, 0, completed.stderr) + matrix = json.loads((root / "matrix" / "matrix.json").read_text()) + self.assertEqual(matrix["schema_version"], "1.0") + self.assertEqual(matrix["task_id"], "esp32s3-gpio-hil-001") + rows = matrix["trials"] + self.assertEqual([row["runtime"] for row in rows], ["opencode", "codex", "claude"]) + self.assertEqual([row["agent_status"] for row in rows], ["completed", "failed", "completed"]) + self.assertEqual([row["hil_status"] for row in rows], ["pass", "not_run", "fail"]) + self.assertEqual(len({row["initial_public_sha256"] for row in rows}), 1) + self.assertIsNone(rows[1]["usage"]["requests"]) + self.assertEqual(rows[1]["usage"]["unavailable_reason"], "runtime did not expose usage") + self.assertIsNone(rows[0]["hil_run"]["cost"]) + self.assertIn("RUNTIME", completed.stdout) + + hidden_name = "hil-oracle.json" + for trial in (path for path in (root / "matrix" / "trials").iterdir() if path.is_dir()): + with self.subTest(trial=trial.name): + agent_invocation = (trial / "agent-invocation.json").read_text(encoding="utf-8") + self.assertNotIn(hidden_name, agent_invocation) + if trial.name != "codex": + hil_invocation = (trial / "hil" / "hil-invocation.json").read_text(encoding="utf-8") + self.assertNotIn("--usage-json", hil_invocation) + + def test_matrix_emits_complete_comparison_evidence(self): + with tempfile.TemporaryDirectory() as directory: + root = Path(directory) + completed = self._run_cli( + root / "matrix", self._fake_agent(root / "agent"), self._fake_hil(root / "hil"), + ) + + self.assertEqual(completed.returncode, 0, completed.stderr) + rows = json.loads((root / "matrix" / "matrix.json").read_text())["trials"] + passed, failed, agent_only = rows + + self.assertEqual(passed["native_model"], "fake-opencode-model") + self.assertIsNone(passed["native_model_unavailable_reason"]) + self.assertEqual(passed["agent_returncode"], 0) + self.assertFalse(passed["agent_timed_out"]) + self.assertEqual(passed["compile_status"], "pass") + self.assertEqual(passed["hil_status"], "pass") + self.assertTrue(passed["final_success"]) + self.assertEqual(passed["elapsed_agent_seconds"], 0.25) + self.assertGreaterEqual(passed["elapsed_hil_seconds"], 0) + self.assertIsNone(passed["repair_count"]) + self.assertIsNone(passed["tool_call_count"]) + self.assertIsNone(passed["invalid_call_count"]) + self.assertEqual(passed["observability_reason"], "runtime did not expose repair/tool-call counts") + self.assertIsNone(passed["infrastructure_category"]) + self.assertIsNone(passed["infrastructure_error"]) + + self.assertEqual(failed["agent_status"], "failed") + self.assertEqual(failed["compile_status"], "not_run") + self.assertFalse(failed["final_success"]) + self.assertEqual(agent_only["compile_status"], "fail") + self.assertFalse(agent_only["final_success"]) + self.assertIn("MODEL", completed.stdout) + self.assertIn("COMPILE", completed.stdout) + self.assertIn("RETURN", completed.stdout) + self.assertIn("TIMEOUT", completed.stdout) + self.assertIn("REPAIR", completed.stdout) + self.assertIn("TOOLS", completed.stdout) + self.assertIn("INVALID", completed.stdout) + self.assertIn("INFRA", completed.stdout) + self.assertIn("A_SEC", completed.stdout) + self.assertIn("H_SEC", completed.stdout) + self.assertIn("TOKENS/COST", completed.stdout) + self.assertIn("fake-opencode-model", completed.stdout) + + def test_agent_only_skips_hil_and_repeated_runtime_selects_trials(self): + with tempfile.TemporaryDirectory() as directory: + root = Path(directory) + completed = self._run_cli( + root / "matrix", self._fake_agent(root / "agent"), self._fake_hil(root / "hil"), + "--agent-only", "--runtime", "claude", "--runtime", "opencode", + ) + + self.assertEqual(completed.returncode, 0, completed.stderr) + rows = json.loads((root / "matrix" / "matrix.json").read_text())["trials"] + self.assertEqual([row["runtime"] for row in rows], ["claude", "opencode"]) + self.assertEqual([row["hil_status"] for row in rows], ["not_run", "not_run"]) + self.assertTrue(all(row["hil_run"] is None for row in rows)) + self.assertFalse(any((root / "matrix" / "trials").glob("*/hil"))) + self.assertTrue(all(row["compile_status"] == "not_run" for row in rows)) + self.assertTrue(all(not row["final_success"] for row in rows)) + + def test_matrix_requires_identity_unless_agent_only(self): + with tempfile.TemporaryDirectory() as directory: + root = Path(directory) + agent = self._fake_agent(root / "agent") + hil = self._fake_hil(root / "hil") + + missing = self._run_cli(root / "missing", agent, hil, identity_command_json=None) + self.assertEqual(missing.returncode, 2) + self.assertIn("--identity-command-json is required unless --agent-only", missing.stderr) + self.assertFalse((root / "missing").exists()) + + empty = self._run_cli(root / "empty", agent, hil, identity_command_json="") + self.assertEqual(empty.returncode, 2) + self.assertIn("--identity-command-json is required unless --agent-only", empty.stderr) + self.assertFalse((root / "empty").exists()) + + agent_only = self._run_cli( + root / "agent-only", agent, hil, "--agent-only", "--runtime", "opencode", + identity_command_json=None, + ) + self.assertEqual(agent_only.returncode, 0, agent_only.stderr) + self.assertFalse((root / "agent-only" / "trials" / "opencode" / "hil").exists()) + + def test_matrix_forwards_identity_only_to_hil(self): + with tempfile.TemporaryDirectory() as directory: + root = Path(directory) + identity_command_json = json.dumps([ + sys.executable, "identify_pio_device.py", "--uart-device", "/dev/fake-uart", + "--jtag-serial", "fake-jtag", + ]) + completed = self._run_cli( + root / "matrix", self._fake_agent(root / "agent"), self._fake_hil(root / "hil"), + "--runtime", "opencode", identity_command_json=identity_command_json, + ) + + self.assertEqual(completed.returncode, 0, completed.stderr) + trial = root / "matrix" / "trials" / "opencode" + invocation = json.loads((trial / "hil" / "hil-invocation.json").read_text()) + self.assertEqual(invocation["identity_command_json"], identity_command_json) + agent_invocation = (trial / "agent-invocation.json").read_text(encoding="utf-8") + self.assertNotIn("identity_command_json", agent_invocation) + self.assertNotIn("identify_pio_device.py", agent_invocation) + + def test_matrix_forwards_only_usage_that_the_hil_cost_schema_accepts(self): + with tempfile.TemporaryDirectory() as directory: + root = Path(directory) + hil = self._fake_hil(root / "hil") + for schema, expected_usage_json in ( + ("hil-valid", True), + ("hil-extra-invalid", False), + ): + with self.subTest(schema=schema): + output = root / schema + completed = self._run_cli( + output, self._fake_agent(root / f"{schema}-agent", schema), hil, + "--runtime", "opencode", + ) + + self.assertEqual(completed.returncode, 0, completed.stderr) + invocation = json.loads( + (output / "trials" / "opencode" / "hil" / "hil-invocation.json").read_text() + ) + self.assertEqual(invocation["usage_json"] is not None, expected_usage_json) + + def test_matrix_preserves_known_partial_usage_with_an_accurate_reason(self): + with tempfile.TemporaryDirectory() as directory: + root = Path(directory) + completed = self._run_cli( + root / "matrix", self._fake_agent(root / "agent", "partial"), self._fake_hil(root / "hil"), + "--agent-only", "--runtime", "opencode", + ) + + self.assertEqual(completed.returncode, 0, completed.stderr) + usage = json.loads((root / "matrix" / "matrix.json").read_text())["trials"][0]["usage"] + self.assertEqual(usage["fresh_input"], 10) + self.assertEqual(usage["output"], 5) + self.assertEqual(usage["estimated_cost_usd"], 0.000015) + self.assertIsNone(usage["reasoning"]) + self.assertEqual(usage["unavailable_reason"], "one or more usage fields unavailable") + + def test_matrix_does_not_run_hil_after_a_nonzero_agent_child_exit(self): + with tempfile.TemporaryDirectory() as directory: + root = Path(directory) + completed = self._run_cli( + root / "matrix", self._fake_agent(root / "agent", child_returncode=7), self._fake_hil(root / "hil"), + "--runtime", "opencode", + ) + + self.assertEqual(completed.returncode, 0, completed.stderr) + row = json.loads((root / "matrix" / "matrix.json").read_text())["trials"][0] + self.assertEqual(row["agent_status"], "infrastructure_error") + self.assertEqual(row["agent_result"]["status"], "completed") + self.assertEqual(row["agent_child_returncode"], 7) + self.assertIn("exited with status 7", row["agent_error"]) + self.assertEqual(row["hil_status"], "not_run") + self.assertFalse((root / "matrix" / "trials" / "opencode" / "hil").exists()) + + def test_matrix_rejects_existing_output_and_invalid_runtime(self): + with tempfile.TemporaryDirectory() as directory: + root = Path(directory) + agent = self._fake_agent(root / "agent") + hil = self._fake_hil(root / "hil") + output = root / "matrix" + output.mkdir() + marker = output / "keep" + marker.write_text("existing", encoding="utf-8") + + existing = self._run_cli(output, agent, hil) + self.assertEqual(existing.returncode, 2) + self.assertIn("output path already exists", existing.stderr) + self.assertEqual(marker.read_text(encoding="utf-8"), "existing") + + invalid = self._run_cli(root / "new", agent, hil, "--runtime", "unknown") + self.assertEqual(invalid.returncode, 2) + self.assertIn("invalid choice", invalid.stderr) + + +class RuntimePackagingTests(unittest.TestCase): + script = REPOSITORY_ROOT / "tests" / "twin2silicon-runtime-smoke.sh" + + def test_runtime_smoke_entry_points_are_opt_in_and_documented(self): + package = json.loads((REPOSITORY_ROOT / "package.json").read_text(encoding="utf-8")) + self.assertEqual( + package["scripts"]["test:runtime-smoke:offline"], + "python3 tests/twin2silicon-hil.py", + ) + self.assertEqual( + package["scripts"]["test:runtime-smoke:hardware"], + "bash tests/twin2silicon-runtime-smoke.sh", + ) + + source = self.script.read_text(encoding="utf-8") + for variable in ( + "LABWIRED_HIL", + "LABWIRED_UART_DEVICE", + "LABWIRED_JTAG_SERIAL", + "LABWIRED_OPENOCD", + "LABWIRED_MATRIX_OUTPUT", + "/Volumes/LabWired", + "identify_pio_device.py", + "--identity-command-json", + ): + with self.subTest(variable=variable): + self.assertIn(variable, source) + self.assertNotRegex(source.lower(), r"api[_-]?key|credential|secret|token") + + refused = subprocess.run( + ["bash", str(self.script)], + cwd=REPOSITORY_ROOT, + text=True, + capture_output=True, + timeout=5, + ) + self.assertNotEqual(refused.returncode, 0) + self.assertIn("LABWIRED_HIL=1 required", refused.stderr) + + def test_runtime_smoke_readme_states_the_comparison_limits_and_hardware_effect(self): + readme = ( + REPOSITORY_ROOT / "benchmarks" / "twin2silicon" / "README.md" + ).read_text(encoding="utf-8").lower() + for wording in ( + "smoke comparison", + "not a leaderboard", + "not overridden", + "opencode model matrix", + "unknown", + "flash", + "multiple tasks", + "repeated fresh trials", + "prints their versions", + ): + with self.subTest(wording=wording): + self.assertIn(wording, readme) + self.assertNotIn("records their versions", readme) + + def test_runtime_smoke_constructs_a_pio_identity_command_for_the_selected_uart_and_jtag(self): + source = self.script.read_text(encoding="utf-8") + + self.assertIn("json.dumps([sys.executable, sys.argv[1], \"--uart-device\", sys.argv[2],", source) + self.assertIn('"--jtag-serial", sys.argv[3]])', source) + self.assertIn('"$LABWIRED_UART_DEVICE"', source) + self.assertIn('"$LABWIRED_JTAG_SERIAL"', source) + + +class PioIdentityDeviceTests(unittest.TestCase): + script = REPOSITORY_ROOT / "benchmarks" / "twin2silicon" / "identify_pio_device.py" + + def _fake_pio(self, directory, body): + path = Path(directory) / "pio" + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text("#!/usr/bin/env python3\n" + body, encoding="utf-8") + path.chmod(0o755) + return path + + def _run_cli(self, pio, *extra): + environment = os.environ.copy() + environment["PATH"] = str(pio.parent) + os.pathsep + environment["PATH"] + return subprocess.run( + [ + sys.executable, str(self.script), "--uart-device", "/dev/fake-uart", + "--jtag-serial", "JTAG-1", *extra, + ], + cwd=REPOSITORY_ROOT, + env=environment, + text=True, + capture_output=True, + timeout=5, + ) + + def test_exact_port_and_serial_match_prints_only_the_requested_serial(self): + with tempfile.TemporaryDirectory() as directory: + pio = self._fake_pio(Path(directory), textwrap.dedent(""" + import json, sys + assert sys.argv[1:] == ["device", "list", "--json-output"] + print(json.dumps([ + {"port": "/dev/other", "hwid": "SER=JTAG-1"}, + {"port": "/dev/fake-uart", "hwid": "USB VID:PID=10C4:EA60 SER=JTAG-1 LOCATION=1-1"}, + ])) + """)) + + completed = self._run_cli(pio) + + self.assertEqual(completed.returncode, 0, completed.stderr) + self.assertEqual(completed.stdout, "JTAG-1\n") + + def test_wrong_mapping_and_malformed_output_fail_without_stdout(self): + cases = { + "wrong": "import json; print(json.dumps([{\"port\": \"/dev/fake-uart\", \"hwid\": \"SER=OTHER\"}]))\n", + "malformed": "print('not json')\n", + } + for name, body in cases.items(): + with self.subTest(name=name), tempfile.TemporaryDirectory() as directory: + completed = self._run_cli(self._fake_pio(Path(directory), body)) + + self.assertNotEqual(completed.returncode, 0) + self.assertEqual(completed.stdout, "") + + def test_timeout_fails_without_stdout(self): + with tempfile.TemporaryDirectory() as directory: + pio = self._fake_pio(Path(directory), "import time; time.sleep(30)\n") + + completed = self._run_cli(pio, "--timeout-seconds", "0.05") + + self.assertNotEqual(completed.returncode, 0) + self.assertEqual(completed.stdout, "") + + +class FixtureContractTests(unittest.TestCase): + def test_esp32s3_gpio_hil_fixture_contract(self): + task_root = ( + REPOSITORY_ROOT + / "benchmarks" + / "twin2silicon" + / "tasks" + / "esp32s3-gpio-hil-001" + ) + self.assertTrue(task_root.is_dir(), f"missing fixture: {task_root}") + task = json.loads((task_root / "task.json").read_text(encoding="utf-8")) + oracle = json.loads( + (task_root / task["hidden_oracle"]).read_text(encoding="utf-8") + ) + + self.assertEqual(task["schema_version"], "1.0") + self.assertEqual(task["id"], "esp32s3-gpio-hil-001") + self.assertEqual(task["board"], "esp32-s3-devkitc-1") + self.assertEqual(task["framework"], "espidf") + self.assertEqual(task["budgets"]["model_tokens"], 50000) + self.assertEqual(task["budgets"]["diagnostic_hil_runs"], 0) + self.assertEqual(oracle["schema_version"], "1.0") + self.assertEqual(oracle["uart"]["ready_prefix"], "LABWIRED_READY:") + self.assertEqual( + oracle["register_assertions"], + [ + { + "name": "gpio2_output_enabled", + "address": "0x60004020", + "mask": "0x00000004", + "expected": "0x00000004", + }, + { + "name": "gpio2_output_high", + "address": "0x60004004", + "mask": "0x00000004", + "expected": "0x00000004", + }, + ], + ) + + public_files = sorted( + path for path in (task_root / task["public_dir"]).rglob("*") if path.is_file() + ) + expected_public_files = { + "README.md", + "firmware/include/run_nonce.h", + "firmware/platformio.ini", + "firmware/sdkconfig.defaults", + "firmware/src/main.c", + } + self.assertEqual( + {str(path.relative_to(task_root / task["public_dir"])) for path in public_files}, + expected_public_files, + ) + public_text = "\n".join(path.read_text(encoding="utf-8") for path in public_files) + for hidden_detail in ( + "0x60004020", + "0x60004004", + "0x00000004", + "esp32s3-builtin.cfg", + "openocd", + "mdw", + ): + with self.subTest(hidden_detail=hidden_detail): + self.assertNotIn(hidden_detail, public_text.lower()) + main_source = (task_root / "public" / "firmware" / "src" / "main.c").read_text( + encoding="utf-8" + ) + self.assertIn("gpio_set_direction(TEST_GPIO, GPIO_MODE_INPUT)", main_source) + self.assertLess( + main_source.index("for (;;)"), + main_source.index('printf("LABWIRED_READY:'), + ) + sdkconfig_defaults = ( + task_root / "public" / "firmware" / "sdkconfig.defaults" + ).read_text(encoding="utf-8") + self.assertIn("# CONFIG_ESP_CONSOLE_NONE is not set", sdkconfig_defaults) + self.assertNotIn("CONFIG_ESP_CONSOLE_UART_NONE", sdkconfig_defaults) + platformio_ini = ( + task_root / "public" / "firmware" / "platformio.ini" + ).read_text(encoding="utf-8") + self.assertIn("board_upload.flash_size = 4MB", platformio_ini) + + +class ResultContractTests(unittest.TestCase): + def test_run_result_infrastructure_error_marks_execution_not_run(self): + result = RunResult.infrastructure_error("board_identity", "wrong adapter") + + self.assertEqual(result.model_status, "not_run") + self.assertEqual(result.compile_status, "not_run") + self.assertEqual(result.simulator_status, "not_supported") + self.assertEqual(result.hardware_status, "not_run") + self.assertEqual(result.infrastructure_status, "error") + self.assertEqual(result.failure_category, "board_identity") + self.assertEqual(result.detail, "wrong adapter") + + def test_run_result_status_fields_use_explicit_literal_contracts(self): + hints = get_type_hints(RunResult) + expected_choices = { + "model_status": ("pass", "fail", "not_run"), + "compile_status": ("pass", "fail", "not_run"), + "simulator_status": ("pass", "fail", "not_run", "not_supported"), + "hardware_status": ("pass", "fail", "not_run"), + "infrastructure_status": ("ok", "error"), + } + + for field, choices in expected_choices.items(): + with self.subTest(field=field): + self.assertIs(get_origin(hints[field]), Literal) + self.assertEqual(get_args(hints[field]), choices) + + def test_sha256_file_streams_file_contents(self): + with tempfile.TemporaryDirectory() as directory: + source = Path(directory) / "firmware.bin" + contents = (b"LabWired\x00" * 10000) + b"tail" + source.write_bytes(contents) + + self.assertEqual(sha256_file(source), hashlib.sha256(contents).hexdigest()) + + def test_write_json_atomic_replaces_destination_with_json(self): + with tempfile.TemporaryDirectory() as directory: + destination = Path(directory) / "result.json" + destination.write_text("stale", encoding="utf-8") + + write_json_atomic(destination, {"status": "ok", "count": 2}) + + self.assertEqual( + json.loads(destination.read_text(encoding="utf-8")), + {"status": "ok", "count": 2}, + ) + self.assertEqual(list(Path(directory).iterdir()), [destination]) + + +class ProcessContractTests(unittest.TestCase): + def test_run_command_timeout_captures_evidence_and_is_bounded(self): + with tempfile.TemporaryDirectory() as directory: + evidence = Path(directory) + started = time.monotonic() + result = run_command( + [ + sys.executable, + "-c", + "import sys,time; print('stdout evidence', flush=True); " + "print('stderr evidence', file=sys.stderr, flush=True); time.sleep(30)", + ], + cwd=evidence, + stdout_path=evidence / "stdout.log", + stderr_path=evidence / "stderr.log", + timeout_seconds=0.1, + ) + elapsed = time.monotonic() - started + + self.assertIsInstance(result, CommandResult) + self.assertTrue(result.timed_out) + self.assertNotEqual(result.returncode, 0) + self.assertGreater(result.duration_seconds, 0) + self.assertLess(elapsed, 2) + self.assertEqual((evidence / "stdout.log").read_text(), "stdout evidence\n") + self.assertEqual((evidence / "stderr.log").read_text(), "stderr evidence\n") + self.assertEqual(result.cwd, str(evidence.resolve())) + self.assertTrue(result.started_at_utc.endswith("Z")) + self.assertTrue(result.ended_at_utc.endswith("Z")) + self.assertIsNone(result.cleanup_error) + + def test_run_command_timeout_terminates_process_group(self): + with tempfile.TemporaryDirectory() as directory: + evidence = Path(directory) + child_ready = evidence / "child-ready" + child_terminated = evidence / "child-terminated" + script = textwrap.dedent( + f""" + import pathlib, signal, subprocess, sys, time + child = ''' + import pathlib, signal, time + ready = pathlib.Path({str(child_ready)!r}) + terminated = pathlib.Path({str(child_terminated)!r}) + def stop(signum, frame): + terminated.write_text("terminated") + raise SystemExit(0) + signal.signal(signal.SIGTERM, stop) + ready.write_text("ready") + while True: time.sleep(1) + ''' + subprocess.Popen([sys.executable, "-c", child]) + while not pathlib.Path({str(child_ready)!r}).exists(): + pass + def stop(signum, frame): + while not pathlib.Path({str(child_terminated)!r}).exists(): + pass + raise SystemExit(0) + signal.signal(signal.SIGTERM, stop) + print("child synchronized", flush=True) + while True: time.sleep(1) + """ + ) + + result = run_command( + [sys.executable, "-c", script], + cwd=evidence, + stdout_path=evidence / "group.stdout.log", + stderr_path=evidence / "group.stderr.log", + timeout_seconds=0.2, + ) + + self.assertTrue(result.timed_out) + self.assertEqual((evidence / "group.stdout.log").read_text(), "child synchronized\n") + self.assertEqual(child_terminated.read_text(), "terminated") + + def test_run_command_timeout_kills_descendant_that_ignores_sigterm(self): + with tempfile.TemporaryDirectory() as directory: + evidence = Path(directory) + leader_pid_path = evidence / "leader-pid" + child_pid_path = evidence / "ignoring-child-pid" + script = textwrap.dedent( + f""" + import os, pathlib, signal, subprocess, sys, time + pathlib.Path({str(leader_pid_path)!r}).write_text(str(os.getpid())) + child = ''' + import os, pathlib, signal, time + signal.signal(signal.SIGTERM, signal.SIG_IGN) + pathlib.Path({str(child_pid_path)!r}).write_text(str(os.getpid())) + while True: time.sleep(1) + ''' + subprocess.Popen([sys.executable, "-c", child]) + while not pathlib.Path({str(child_pid_path)!r}).exists(): + pass + print("ignoring child synchronized", flush=True) + while True: time.sleep(1) + """ + ) + + try: + result = run_command( + [sys.executable, "-c", script], + cwd=evidence, + stdout_path=evidence / "ignoring.stdout.log", + stderr_path=evidence / "ignoring.stderr.log", + timeout_seconds=0.2, + ) + leader_pid = int(leader_pid_path.read_text()) + child_pid = int(child_pid_path.read_text()) + + self.assertTrue(result.timed_out) + with self.assertRaises(ProcessLookupError): + os.killpg(leader_pid, 0) + with self.assertRaises(ProcessLookupError): + os.kill(child_pid, 0) + finally: + if leader_pid_path.exists(): + try: + os.killpg(int(leader_pid_path.read_text()), signal.SIGKILL) + except (PermissionError, ProcessLookupError): + pass + + def test_run_command_reports_cleanup_error_when_leader_cannot_be_reaped(self): + class UnreapableProcess: + pid = 424242 + returncode = None + + def __init__(self): + self.wait_timeouts = [] + + def communicate(self, timeout): + raise subprocess.TimeoutExpired(("stuck-tool",), timeout) + + def wait(self, timeout=None): + self.wait_timeouts.append(timeout) + if timeout is None: + raise AssertionError("run_command used an unbounded wait") + raise subprocess.TimeoutExpired(("stuck-tool",), timeout) + + def fake_killpg(process_group_id, signal_number): + if signal_number == 0: + raise ProcessLookupError + + with tempfile.TemporaryDirectory() as directory: + evidence = Path(directory) + fake_process = UnreapableProcess() + started = time.monotonic() + with mock.patch.object(process_module.subprocess, "Popen", return_value=fake_process), mock.patch.object( + process_module.os, "killpg", side_effect=fake_killpg + ): + result = run_command( + ["stuck-tool"], + cwd=evidence, + stdout_path=evidence / "stuck.stdout.log", + stderr_path=evidence / "stuck.stderr.log", + timeout_seconds=0.01, + ) + + self.assertLess(time.monotonic() - started, 1) + self.assertEqual(result.cleanup_error, "process_group_did_not_exit") + self.assertTrue(fake_process.wait_timeouts) + self.assertNotIn(None, fake_process.wait_timeouts) + + def test_run_command_reports_cleanup_error_when_descendant_remains_after_reap(self): + class ReapedLeader: + pid = 424243 + returncode = -signal.SIGTERM + + def communicate(self, timeout): + raise subprocess.TimeoutExpired(("stuck-descendant",), timeout) + + def wait(self, timeout): + return self.returncode + + with tempfile.TemporaryDirectory() as directory: + evidence = Path(directory) + started = time.monotonic() + with mock.patch.object(process_module.subprocess, "Popen", return_value=ReapedLeader()), mock.patch.object( + process_module.os, "killpg" + ), mock.patch.object(process_module, "_process_group_exists", return_value=True): + result = run_command( + ["stuck-descendant"], + cwd=evidence, + stdout_path=evidence / "descendant.stdout.log", + stderr_path=evidence / "descendant.stderr.log", + timeout_seconds=0.01, + ) + + self.assertLess(time.monotonic() - started, 2) + self.assertEqual(result.cleanup_error, "process_group_did_not_exit") + + +class Esp32S3ConfigTests(unittest.TestCase): + def test_parses_shipped_oracle_exactly(self): + oracle_path = (REPOSITORY_ROOT / "benchmarks/twin2silicon/tasks/esp32s3-gpio-hil-001/hidden/hil-oracle.json") + config = Esp32S3Config.from_oracle(json.loads(oracle_path.read_text())) + self.assertEqual(config.uart_ready_prefix, "LABWIRED_READY:") + self.assertEqual((config.uart_baud, config.uart_timeout_seconds), (115200, 30)) + self.assertEqual(config.identity_command, ("__LABWIRED_IDENTITY_RUNNER__",)) + self.assertEqual((config.identity_expected_board, config.identity_timeout_seconds), + ("esp32-s3-devkitc-1", 10)) + self.assertEqual((config.flash_target, config.flash_artifact, config.flash_timeout_seconds), + ("upload", ".pio/build/esp32s3/firmware.bin", 120)) + self.assertEqual((config.openocd_board_config, config.openocd_startup_timeout_seconds, + config.openocd_command_timeout_seconds), ("board/esp32s3-builtin.cfg", 20, 10)) + self.assertEqual((config.platformio_project_dir, config.platformio_environment), + ("public/firmware", "esp32s3")) + self.assertEqual(len(config.assertions), 2) + + def test_parses_valid_oracle_and_hex_register_values(self): + config = Esp32S3Config.from_oracle( + { + "uart": {"ready_prefix": "READY:", "baud": 115200, "timeout_seconds": 1}, + "identity": {"command": ["identity"], "expected_board": "board", "timeout_seconds": 0}, + "flash": {"target": "upload", "artifact": "firmware.bin", "timeout_seconds": 0}, + "openocd": {"board_config": "board.cfg", "startup_timeout_seconds": 0, + "command_timeout_seconds": 0}, + "register_assertions": [ + {"name": "gpio", "address": "0x60004020", "mask": "0x4", "expected": "0x4"} + ], + } + ) + self.assertEqual(config.assertions[0].address, 0x60004020) + self.assertEqual(config.assertions[0].mask, 4) + + def test_rejects_invalid_bounds_alignment_duplicates_and_names(self): + good = {"name": "gpio", "address": "0x60004020", "mask": "0x4", "expected": "0x4"} + bad = [ + {**good, "address": "-1"}, + {**good, "address": "0x60004021"}, + {**good, "mask": "0x100000000"}, + {**good, "expected": "xyz"}, + {**good, "expected": "4"}, + {**good, "mask": 1.5}, + {**good, "name": "gpio; shutdown"}, + ] + for record in bad: + with self.subTest(record=record), self.assertRaises((TypeError, ValueError)): + RegisterAssertion.from_json(record) + with self.assertRaises(ValueError): + Esp32S3Config.from_oracle({ + "uart": {"ready_prefix": "READY:", "baud": 115200, "timeout_seconds": 1}, + "identity": {"command": ["id"], "expected_board": "board", "timeout_seconds": 0}, + "flash": {"target": "upload", "artifact": "fw", "timeout_seconds": 0}, + "openocd": {"board_config": "cfg", "startup_timeout_seconds": 0, "command_timeout_seconds": 0}, + "register_assertions": [good, {**good, "address": "0x60004024"}], + }) + with self.assertRaises(ValueError): + Esp32S3Config.from_oracle({ + "uart": {"ready_prefix": "READY:", "baud": 115200, "timeout_seconds": 1}, + "identity": {"command": ["id"], "expected_board": "board", "timeout_seconds": 0}, + "flash": {"target": "upload", "artifact": "fw", "timeout_seconds": 0}, + "openocd": {"board_config": "cfg", "startup_timeout_seconds": 0, "command_timeout_seconds": 0}, + "register_assertions": [good, {**good, "name": "gpio2"}], + }) + + def test_rejects_empty_configuration_assertions_and_nonfinite_timeout(self): + for timeout in (float("nan"), float("inf")): + with self.subTest(timeout=timeout), self.assertRaises(ValueError): + Esp32S3Config.from_oracle({ + "uart": {"ready_prefix": "READY:", "baud": 115200, "timeout_seconds": timeout}, + "identity": {"command": ["id"], "expected_board": "board", "timeout_seconds": 0}, + "flash": {"target": "upload", "artifact": "fw", "timeout_seconds": 0}, + "openocd": {"board_config": "cfg", "startup_timeout_seconds": 0, "command_timeout_seconds": 0}, + "register_assertions": [{"name": "gpio", "address": "0x4", "mask": "0x4", "expected": "0x4"}], + }) + with self.assertRaises(ValueError): + Esp32S3Config.from_oracle({ + "uart": {"ready_prefix": "", "baud": 115200, "timeout_seconds": 1}, + "identity": {"command": [], "expected_board": "", "timeout_seconds": 0}, + "flash": {"target": "", "artifact": "", "timeout_seconds": 0}, + "openocd": {"board_config": "", "startup_timeout_seconds": 0, "command_timeout_seconds": 0}, + "register_assertions": [], + }) + + +class BoardIdentityAndFlashTests(unittest.TestCase): + def test_exactly_one_configured_serial_succeeds(self): + with tempfile.TemporaryDirectory() as directory: + tool = executable_fixture(directory, "print(' JTAG-1 ')\n") + result = validate_identity([tool], "JTAG-1", cwd=directory, evidence_dir=directory, timeout_seconds=1) + self.assertEqual((result.status, result.category), ("pass", None)) + + def test_absent_wrong_and_duplicate_serials_are_infrastructure_before_flash(self): + for output in ("", "OTHER\\n", "JTAG-1\\nJTAG-1\\n"): + with self.subTest(output=output), tempfile.TemporaryDirectory() as directory: + marker = Path(directory) / "flashed" + tool = executable_fixture(directory, f"print({output!r}, end='')\n") + result = validate_identity([tool], "JTAG-1", cwd=directory, evidence_dir=directory, timeout_seconds=1) + self.assertEqual((result.status, result.category), ("infrastructure_error", "board_identity")) + flash = executable_fixture(directory, f"from pathlib import Path; Path({str(marker)!r}).write_text('flashed')\n") + with self.assertRaises(ValueError): + flash_firmware([flash], cwd=directory, evidence_dir=directory, timeout_seconds=1, + identity_validated=False) + self.assertFalse(marker.exists()) + + def test_identity_success_then_flash_executes_in_order(self): + with tempfile.TemporaryDirectory() as directory: + marker = Path(directory) / "flashed" + identity_tool = executable_fixture(directory, "print('JTAG-1')\n") + identity = validate_identity([identity_tool], "JTAG-1", cwd=directory, + evidence_dir=directory, timeout_seconds=1) + flash_tool = Path(directory) / "flash.py" + flash_tool.write_text("#!/usr/bin/env python3\nfrom pathlib import Path\nPath(%r).write_text('flashed')\n" % str(marker)) + flash_tool.chmod(0o755) + flashed = flash_firmware([flash_tool], cwd=directory, evidence_dir=directory, + timeout_seconds=1, identity_validated=identity.status == "pass") + self.assertEqual(flashed.status, "pass") + self.assertEqual(marker.read_text(), "flashed") + + def test_identity_tool_timeout_nonzero_and_cleanup_are_infrastructure(self): + with tempfile.TemporaryDirectory() as directory: + slow = executable_fixture(directory, "import time; time.sleep(30)\n") + result = validate_identity([slow], "JTAG-1", cwd=directory, evidence_dir=directory, timeout_seconds=.05) + self.assertEqual(result.status, "infrastructure_error") + with tempfile.TemporaryDirectory() as directory: + failed = executable_fixture(directory, "raise SystemExit(3)\n") + result = validate_identity([failed], "JTAG-1", cwd=directory, evidence_dir=directory, timeout_seconds=1) + self.assertEqual((result.status, result.category), ("infrastructure_error", "board_identity")) + fake = CommandResult(("x",), "/", 0, False, "", "", 0, "/tmp/o", "/tmp/e", "cleanup") + with tempfile.TemporaryDirectory() as directory: + result = validate_identity(["x"], "JTAG-1", cwd=directory, evidence_dir=directory, + timeout_seconds=1, runner=lambda *a, **k: fake) + self.assertEqual((result.status, result.category), ("infrastructure_error", "board_identity")) + + def test_non_utf8_identity_output_is_infrastructure(self): + with tempfile.TemporaryDirectory() as directory: + tool = executable_fixture(directory, "import sys; sys.stdout.buffer.write(b'\\xff\\xfe')\n") + result = validate_identity([tool], "JTAG-1", cwd=directory, evidence_dir=directory, + timeout_seconds=1) + self.assertEqual((result.status, result.category), ("infrastructure_error", "board_identity")) + + def test_command_launch_errors_are_infrastructure(self): + def unavailable(*args, **kwargs): + raise FileNotFoundError("tool missing") + with tempfile.TemporaryDirectory() as directory: + identity = validate_identity(["missing"], "JTAG-1", cwd=directory, evidence_dir=directory, + timeout_seconds=1, runner=unavailable) + flash = flash_firmware(["missing"], cwd=directory, evidence_dir=directory, + timeout_seconds=1, identity_validated=True, runner=unavailable) + self.assertEqual(identity.status, "infrastructure_error") + self.assertEqual(flash.status, "infrastructure_error") + + def test_flash_classifies_nonzero_as_hardware_and_timeout_cleanup_as_infrastructure(self): + def result(code=0, timed_out=False, cleanup=None): + return CommandResult(("flash",), "/", code, timed_out, "", "", 0, "/tmp/o", "/tmp/e", cleanup) + with tempfile.TemporaryDirectory() as directory: + failed = flash_firmware(["flash"], cwd=directory, evidence_dir=directory, timeout_seconds=1, + identity_validated=True, runner=lambda *a, **k: result(2)) + timeout = flash_firmware(["flash"], cwd=directory, evidence_dir=directory, timeout_seconds=1, + identity_validated=True, runner=lambda *a, **k: result(-15, True)) + cleanup = flash_firmware(["flash"], cwd=directory, evidence_dir=directory, timeout_seconds=1, + identity_validated=True, runner=lambda *a, **k: result(0, False, "stuck")) + self.assertEqual((failed.status, failed.category), ("hardware_fail", "flash")) + self.assertEqual(timeout.status, "infrastructure_error") + self.assertEqual(cleanup.status, "infrastructure_error") + with tempfile.TemporaryDirectory() as directory, self.assertRaises(ValueError): + flash_firmware(["flash"], cwd=directory, evidence_dir=directory, timeout_seconds=1, + identity_validated=False) + + +class BoardLockTests(unittest.TestCase): + def test_lock_is_identity_keyed_bounded_and_released_normally(self): + with tempfile.TemporaryDirectory() as directory: + first = BoardLock(directory, "usb/serial:one", timeout_seconds=.1) + with first: + self.assertIn("usb_serial_one", first.path.name) + started = time.monotonic() + with self.assertRaises(BoardLockTimeout): + with BoardLock(directory, "usb/serial:one", timeout_seconds=.05): + pass + self.assertLess(time.monotonic() - started, .5) + with BoardLock(directory, "usb/serial:one", timeout_seconds=.1): + pass + + def test_lock_releases_after_exception_and_stale_metadata_does_not_claim_lock(self): + with tempfile.TemporaryDirectory() as directory: + lock = BoardLock(directory, "JTAG-1", timeout_seconds=.1) + lock.path.parent.mkdir(parents=True, exist_ok=True) + lock.path.write_text('{"pid": 999999, "identity": "stale"}') + with self.assertRaises(RuntimeError): + with lock: + metadata = json.loads(lock.path.read_text()) + self.assertEqual(metadata["identity"], "JTAG-1") + raise RuntimeError("candidate failed") + with BoardLock(directory, "JTAG-1", timeout_seconds=.1): + pass + + def test_lock_releases_if_metadata_persistence_fails_and_rejects_nonfinite_timeout(self): + with tempfile.TemporaryDirectory() as directory: + with mock.patch("benchmarks.twin2silicon.hil.esp32s3.os.fsync", side_effect=OSError("disk")): + with self.assertRaises(OSError): + with BoardLock(directory, "JTAG-1", timeout_seconds=.1): + pass + with BoardLock(directory, "JTAG-1", timeout_seconds=.1): + pass + for timeout in (float("nan"), float("inf")): + with self.subTest(timeout=timeout), self.assertRaises(ValueError): + BoardLock(directory, "JTAG-1", timeout_seconds=timeout) + + +class UartNonceTests(unittest.TestCase): + def _capture(self, chunks, nonce="current", timeout=.2, max_bytes=64): + master, slave = pty.openpty() + device = os.ttyname(slave) + tty.setraw(slave) + with tempfile.TemporaryDirectory() as directory: + log = Path(directory) / "uart.log" + started = threading.Event() + def writer(): + started.wait() + for chunk in chunks: + os.write(master, chunk) + thread = threading.Thread(target=writer) + thread.start() + started.set() + try: + result = capture_uart_nonce(device, 115200, nonce, timeout, log, max_bytes=max_bytes) + finally: + thread.join(1) + os.close(master) + os.close(slave) + return result, log.read_bytes() + + def test_accepts_only_exact_current_nonce_as_complete_line_and_logs_raw_bytes(self): + result, raw = self._capture([b"boot\r\nLABWIRED_READY:current\r", b"\n"]) + self.assertTrue(result.matched) + self.assertEqual(raw, b"boot\r\nLABWIRED_READY:current\r\n") + + def test_rejects_absent_wrong_stale_or_incomplete_nonce_with_bounded_evidence(self): + for chunks in ([b"boot\n"], [b"LABWIRED_READY:wrong\n"], + [b"LABWIRED_READY:stale\n"], [b"LABWIRED_READY:current"]): + with self.subTest(chunks=chunks): + started = time.monotonic() + result, raw = self._capture(chunks, timeout=.05) + self.assertFalse(result.matched) + self.assertLess(time.monotonic() - started, .5) + self.assertLessEqual(len(raw), 64) + + def test_rejects_unsupported_baud_and_closes_opened_fd(self): + master, slave = pty.openpty() + device = os.ttyname(slave) + os.close(slave) + real_close = os.close + closed = [] + with tempfile.TemporaryDirectory() as directory, mock.patch("benchmarks.twin2silicon.hil.esp32s3.os.close", side_effect=lambda fd: (closed.append(fd), real_close(fd))[1]): + with self.assertRaises(ValueError): + capture_uart_nonce(device, 12345, "n", .01, Path(directory) / "log") + real_close(master) + self.assertTrue(closed) + + def test_timeout_closes_the_opened_uart_fd(self): + master, slave = pty.openpty() + device = os.ttyname(slave) + opened = [] + real_open = os.open + with tempfile.TemporaryDirectory() as directory, mock.patch( + "benchmarks.twin2silicon.hil.esp32s3.os.open", + side_effect=lambda *args, **kwargs: (lambda fd: (opened.append(fd), fd)[1])(real_open(*args, **kwargs)), + ): + result = capture_uart_nonce(device, 115200, "never", .01, Path(directory) / "uart.log") + os.close(master) + os.close(slave) + self.assertFalse(result.matched) + self.assertTrue(result.timed_out) + self.assertEqual(result.termination_reason, "timeout") + self.assertEqual(len(opened), 1) + with self.assertRaises(OSError) as error: + os.fstat(opened[0]) + self.assertEqual(error.exception.errno, 9) + + def test_max_bytes_exhaustion_is_not_reported_as_timeout(self): + result, raw = self._capture([b"1234567890"], timeout=1, max_bytes=10) + self.assertFalse(result.matched) + self.assertFalse(result.timed_out) + self.assertEqual(result.termination_reason, "max_bytes") + self.assertEqual(raw, b"1234567890") + + +class OpenOcdEvidenceTests(unittest.TestCase): + def setUp(self): + self.assertions = ( + RegisterAssertion("enable", 0x60004020, 4, 4), + RegisterAssertion("high", 0x60004004, 4, 4), + ) + + def test_command_is_argv_and_requests_marked_records_at_fixed_speed(self): + command = build_openocd_command("openocd", "board.cfg", "JTAG-1", self.assertions) + self.assertEqual(command, ["openocd", "-f", "board.cfg", "-c", + 'adapter serial JTAG-1; adapter speed 4000; init; reset run; sleep 750; halt; ' + 'echo "@@REG enable 0x60004020"; echo [capture "mdw 0x60004020 1"]; ' + 'echo "@@REG high 0x60004004"; echo [capture "mdw 0x60004004 1"]; exit']) + + def test_empty_assertions_are_rejected_by_all_register_paths(self): + with self.assertRaises(ValueError): + build_openocd_command("openocd", "board.cfg", "serial", ()) + with self.assertRaises(ValueError): + parse_openocd_registers("", ()) + with self.assertRaises(ValueError): + evaluate_registers({}, ()) + + def test_parser_accepts_only_immediately_paired_canonical_requested_records(self): + text = "noise\n@@REG enable 0x60004020\n0x60004020: 0x00000004\n@@REG high 0x60004004\n0x60004004: 0x00000004\n" + self.assertEqual(parse_openocd_registers(text, self.assertions), {"enable": 4, "high": 4}) + invalid = [ + text.replace("0x60004020: 0x00000004", "noise\n0x60004020: 0x00000004"), + text + "@@REG enable 0x60004020\n0x60004020: 0x00000004\n", + text.replace("enable", "other"), + text.replace("@@REG enable 0x60004020", "@@REG enable 0x60004024"), + text.replace("0x60004020: 0x00000004", "60004020 = 4"), + text.replace("@@REG enable 0x60004020", "@@REG enable not-an-address"), + text + "@@REG malformed\n", + text + "Error: target not halted\n", + text.split("@@REG high")[0], + ] + for evidence in invalid: + with self.subTest(evidence=evidence), self.assertRaises(ValueError): + parse_openocd_registers(evidence, self.assertions) + + def test_parser_accepts_real_espressif_bare_eight_digit_mdw_value(self): + text = "@@REG enable 0x60004020\n0x60004020: 00000004\n@@REG high 0x60004004\n0x60004004: 00000004\n" + self.assertEqual(parse_openocd_registers(text, self.assertions), {"enable": 4, "high": 4}) + + def test_masked_mismatch_fails_and_all_assertions_pass(self): + passing = evaluate_registers({"enable": 0x104, "high": 4}, self.assertions) + failing = evaluate_registers({"enable": 0, "high": 4}, self.assertions) + self.assertEqual(passing.status, "pass") + self.assertEqual(failing.status, "hardware_fail") + self.assertFalse(failing.observations[0].passed) + + def test_evaluation_rejects_non_uint32_observed_values(self): + for value in (True, -1, 0x100000000, 1.5, "4"): + with self.subTest(value=value), self.assertRaises((TypeError, ValueError)): + evaluate_registers({"enable": value, "high": 4}, self.assertions) + + +class OpenOcdExecutionTests(unittest.TestCase): + def setUp(self): + self.assertions = (RegisterAssertion("gpio", 0x60004020, 4, 4),) + + def _run(self, directory, body, timeout=1): + tool = executable_fixture(directory, body) + return read_registers(tool, "board.cfg", "JTAG-1", self.assertions, cwd=directory, + evidence_dir=directory, timeout_seconds=timeout) + + def test_reads_openocd_stderr_and_returns_typed_evaluation(self): + with tempfile.TemporaryDirectory() as directory: + result = self._run(directory, "import sys\nprint('@@REG gpio 0x60004020', file=sys.stderr)\nprint('0x60004020: 0x00000004', file=sys.stderr)\n") + self.assertEqual((result.status, result.category), ("pass", None)) + self.assertEqual(result.observed, {"gpio": 4}) + self.assertEqual(result.evaluation.status, "pass") + self.assertEqual(result.command_result.returncode, 0) + + def test_classifies_nonzero_timeout_cleanup_launch_and_parse_as_infrastructure(self): + with tempfile.TemporaryDirectory() as directory: + nonzero = self._run(directory, "raise SystemExit(2)\n") + timeout = self._run(directory, "import time; time.sleep(30)\n", timeout=.05) + malformed = self._run(directory, "import sys; print('@@REG gpio bad', file=sys.stderr)\n") + fake = CommandResult(("x",), "/", 0, False, "", "", 0, "/tmp/o", "/tmp/e", "stuck") + cleanup = read_registers("x", "c", "s", self.assertions, cwd=directory, + evidence_dir=directory, timeout_seconds=1, runner=lambda *a, **k: fake) + launch = read_registers("x", "c", "s", self.assertions, cwd=directory, + evidence_dir=directory, timeout_seconds=1, + runner=lambda *a, **k: (_ for _ in ()).throw(FileNotFoundError("missing"))) + for result in (nonzero, timeout, malformed, cleanup, launch): + with self.subTest(result=result): + self.assertEqual((result.status, result.category), ("infrastructure_error", "openocd")) + + def test_timeout_terminates_openocd_process_group(self): + with tempfile.TemporaryDirectory() as directory: + terminated = Path(directory) / "child-terminated" + ready = Path(directory) / "child-ready" + body = textwrap.dedent(f""" + import pathlib, signal, subprocess, sys, time + child = ''' + import pathlib, signal, time + terminated = pathlib.Path({str(terminated)!r}) + def stop(signum, frame): + terminated.write_text("terminated") + raise SystemExit(0) + signal.signal(signal.SIGTERM, stop) + pathlib.Path({str(ready)!r}).write_text("ready") + while True: time.sleep(1) + ''' + subprocess.Popen([sys.executable, "-c", child]) + while not pathlib.Path({str(ready)!r}).exists(): pass + def stop(signum, frame): + while not pathlib.Path({str(terminated)!r}).exists(): pass + raise SystemExit(0) + signal.signal(signal.SIGTERM, stop) + print("ready", flush=True) + while True: time.sleep(1) + """) + result = self._run(directory, body, timeout=.5) + self.assertEqual(result.status, "infrastructure_error") + self.assertEqual(terminated.read_text(), "terminated") + + +class SimpleHilRunnerTests(unittest.TestCase): + def _run_cli(self, *arguments): + return subprocess.run( + [sys.executable, str(REPOSITORY_ROOT / "benchmarks/twin2silicon/run_hil.py"), + *map(str, arguments)], + cwd=REPOSITORY_ROOT, + text=True, + capture_output=True, + timeout=15, + ) + + def test_complete_fake_hil_pass(self): + task = REPOSITORY_ROOT / "benchmarks/twin2silicon/tasks/esp32s3-gpio-hil-001" + with tempfile.TemporaryDirectory() as directory: + root = Path(directory) + candidate = root / "candidate" + import shutil + shutil.copytree(task / "public", candidate) + (candidate / "firmware/src/main.c").write_text( + (candidate / "firmware/src/main.c").read_text().replace( + "GPIO_MODE_INPUT", "GPIO_MODE_OUTPUT" + ) + ) + flash_marker = root / "flashed" + run_dir = root / "run" + master, slave = pty.openpty() + uart = os.ttyname(slave) + wrong_uart = "/dev/cu.usbmodem11201" + pio_dir = root / "pio" + pio_dir.mkdir() + pio = executable_fixture(pio_dir, textwrap.dedent(f""" + import pathlib, sys, time + args = sys.argv[1:] + project = pathlib.Path(args[args.index('--project-dir') + 1]) + if 'clean' in args: + raise SystemExit(0) + if 'upload' in args: + assert args[args.index('--upload-port') + 1] == {uart!r} + assert {wrong_uart!r} not in args + uart_log = pathlib.Path({str(root / "run/uart.log")!r}) + deadline = time.monotonic() + 1 + while time.monotonic() < deadline and not uart_log.exists(): + time.sleep(.01) + if uart_log.exists(): + print('UART capture started before flash', file=sys.stderr) + raise SystemExit(2) + pathlib.Path({str(flash_marker)!r}).write_text('flashed') + raise SystemExit(0) + artifact = project / '.pio/build/esp32s3/firmware.bin' + artifact.parent.mkdir(parents=True, exist_ok=True) + artifact.write_bytes(b'firmware') + """)) + identity_dir = root / "identity" + identity_dir.mkdir() + identity = executable_fixture(identity_dir, "print('JTAG-1')\n") + openocd_dir = root / "openocd" + openocd_dir.mkdir() + openocd = executable_fixture(openocd_dir, textwrap.dedent(""" + import sys + print('@@REG gpio2_output_enabled 0x60004020', file=sys.stderr) + print('0x60004020: 00000004', file=sys.stderr) + print('@@REG gpio2_output_high 0x60004004', file=sys.stderr) + print('0x60004004: 00000004', file=sys.stderr) + """)) + def write_uart(): + deadline = time.monotonic() + 10 + header = run_dir / "workspace/firmware/include/run_nonce.h" + while time.monotonic() < deadline and not (flash_marker.exists() and header.exists()): + time.sleep(.01) + if flash_marker.exists() and header.exists(): + nonce = header.read_text().split('"')[1] + os.write(master, f"LABWIRED_READY:{nonce}\n".encode()) + writer = threading.Thread(target=write_uart) + writer.start() + result = self._run_cli( + task, "--run-dir", run_dir, "--candidate", candidate, + "--jtag-serial", "JTAG-1", "--uart-device", uart, + "--platformio", pio, "--openocd", openocd, + "--identity-command-json", json.dumps([str(identity)]), + ) + writer.join(10) + os.close(master) + os.close(slave) + self.assertEqual(result.returncode, 0, result.stderr) + manifest = json.loads((run_dir / "run.json").read_text()) + self.assertEqual(manifest["status"], "pass") + self.assertEqual(manifest["compile_status"], "pass") + self.assertEqual(manifest["hardware_status"], "pass") + self.assertTrue(manifest["uart"]["matched"]) + self.assertTrue(all(item["passed"] for item in manifest["registers"])) + + def test_compile_failure_never_touches_hardware(self): + task = REPOSITORY_ROOT / "benchmarks/twin2silicon/tasks/esp32s3-gpio-hil-001" + with tempfile.TemporaryDirectory() as directory: + root = Path(directory) + candidate = root / "candidate" + import shutil + shutil.copytree(task / "public", candidate) + marker = root / "hardware-ran" + pio_dir = root / "pio" + pio_dir.mkdir() + pio = executable_fixture(pio_dir, "import sys; raise SystemExit(0 if 'clean' in sys.argv else 1)\n") + hardware_dir = root / "hardware" + hardware_dir.mkdir() + hardware = executable_fixture( + hardware_dir, + f"from pathlib import Path\nPath({str(marker)!r}).write_text('ran')\n", + ) + result = self._run_cli( + task, "--run-dir", root / "run", "--candidate", candidate, + "--jtag-serial", "JTAG-1", "--uart-device", "/dev/null", + "--platformio", pio, "--openocd", hardware, + "--identity-command-json", json.dumps([str(hardware)]), + ) + self.assertEqual(result.returncode, 0, result.stderr) + manifest = json.loads((root / "run/run.json").read_text()) + self.assertEqual(manifest["status"], "fail") + self.assertEqual(manifest["compile_status"], "fail") + self.assertEqual(manifest["hardware_status"], "not_run") + self.assertFalse(marker.exists()) + + +class RunAgentTests(unittest.TestCase): + task = REPOSITORY_ROOT / "benchmarks/twin2silicon/tasks/esp32s3-gpio-hil-001" + script = REPOSITORY_ROOT / "benchmarks/twin2silicon/run_agent.py" + + def _fake_runtime(self, directory, runtime, mode="success", repair_iterations=6, expects_auth=True): + codex_auth_assertion = ( + "assert (codex_home / 'auth.json').read_text(encoding='utf-8') == os.environ['EXPECTED_AUTH']" + if expects_auth + else "assert not (codex_home / 'auth.json').exists()" + ) + codex_auth_mode_assertion = ( + "assert (codex_home / 'auth.json').stat().st_mode & 0o777 == 0o600" + if expects_auth + else "" + ) + workspace_code = { + "codex": f"workspace = Path(args[args.index('-C') + 1])\nassert args[:2] == ['exec', '--json']\nassert '--ephemeral' in args and '--skip-git-repo-check' in args\nassert '-c' not in args\ncodex_home = Path(os.environ['CODEX_HOME'])\nassert codex_home != Path(os.environ['SOURCE_CODEX_HOME'])\nassert codex_home != Path(os.environ['EXPECTED_TRIAL'])\nassert codex_home != Path(os.environ['EXPECTED_CONFIG'])\nassert (codex_home / 'config.toml').read_text(encoding='utf-8') == '[mcp_servers.labwired]\\ncommand = \"npx\"\\nargs = [\"-y\", \"@labwired/mcp\"]\\n'\n{codex_auth_assertion}\n{codex_auth_mode_assertion}\n(workspace / 'effective-codex-home').write_text(str(codex_home), encoding='utf-8')", + "claude": "workspace = Path.cwd()\nassert args[:3] == ['--print', '--verbose', '--output-format']\nassert args[args.index('--output-format') + 1] == 'stream-json'\nassert args[args.index('--mcp-config') + 1] == str(Path(os.environ['EXPECTED_CONFIG']) / 'claude-mcp.json')\nassert '--strict-mcp-config' in args", + "opencode": "workspace = Path(args[args.index('--dir') + 1])\nassert args[:2] == ['run', '--format']\nassert args[args.index('--format') + 1] == 'json'\nassert os.environ['OPENCODE_CONFIG'] == str(Path(os.environ['EXPECTED_CONFIG']) / 'opencode.json')", + }[runtime] + output = { + "codex": "print(json.dumps({'type': 'turn.started', 'model': 'fake-codex-model'})); print(json.dumps({'type': 'turn.completed', 'usage': {'input_tokens': 12, 'cached_input_tokens': 0, 'output_tokens': 3}}))", + "claude": "print(json.dumps({'type': 'system', 'subtype': 'init', 'model': 'fake-claude-model'})); print(json.dumps({'type': 'result', 'usage': {'input_tokens': 12, 'output_tokens': 3}, 'total_cost_usd': 0.01}))", + "opencode": "print(json.dumps({'type': 'step_start', 'part': {'model': 'fake-opencode-model'}})); print(json.dumps({'type': 'step_finish', 'part': {'tokens': {'input': 12, 'output': 3}, 'cost': 0.01}}))", + }[runtime] + body = textwrap.dedent(f""" + import json + import os + from pathlib import Path + import sys + import time + + args = sys.argv[1:] + if args == ['--version']: + print('fake-{runtime} 1.0') + raise SystemExit(0) + assert '--model' not in args + assert Path(os.environ['EXPECTED_INSTRUCTIONS']).read_text(encoding='utf-8') in args[-1] + assert 'This trial is noninteractive.' in args[-1] + assert 'GPIO 2 is driven high' in args[-1] + assert 'Maximum repair attempts: {repair_iterations}' in args[-1] + assert 'hil-oracle.json' not in args[-1] + # workspace checks + source = workspace / 'firmware/src/main.c' + contents = source.read_text(encoding='utf-8') + assert 'GPIO_MODE_INPUT' in contents + assert 'GPIO_MODE_OUTPUT' not in contents + assert not (workspace / 'hidden').exists() + instruction = workspace / {'CLAUDE.md' if runtime == 'claude' else 'AGENTS.md'!r} + assert instruction.read_text(encoding='utf-8') == Path(os.environ['EXPECTED_INSTRUCTIONS']).read_text(encoding='utf-8') + assert 'This trial is noninteractive.' in instruction.read_text(encoding='utf-8') + if {mode!r} == 'timeout': + time.sleep(30) + source.write_text(contents.replace('GPIO_MODE_INPUT', 'GPIO_MODE_OUTPUT'), encoding='utf-8') + if {mode!r} == 'nonzero': + raise SystemExit(9) + if {mode!r} == 'malformed': + print('not json') + elif {mode!r} == 'missing': + pass + else: + {output} + """).replace("# workspace checks", workspace_code) + return executable_fixture(directory, body) + + def _run_cli(self, runtime, executable, trial, timeout_seconds=2, task=None, source_auth=True): + source_codex_home = (trial.parent / "source-codex-home").resolve() + source_codex_home.mkdir(parents=True) + auth_contents = "sentinel-codex-auth-credential" + auth_path = source_codex_home / "auth.json" + if source_auth: + auth_path.write_text(auth_contents, encoding="utf-8") + auth_path.chmod(0o600) + environment = os.environ.copy() + environment.update({ + "EXPECTED_CONFIG": str((trial / "runtime-config").resolve()), + "CODEX_HOME": str(source_codex_home), + "SOURCE_CODEX_HOME": str(source_codex_home), + "EXPECTED_TRIAL": str(trial.resolve()), + "EXPECTED_AUTH": auth_contents, + "EXPECTED_INSTRUCTIONS": str( + REPOSITORY_ROOT / "benchmarks/twin2silicon/shared-agent-instructions.md" + ), + }) + return subprocess.run( + [ + sys.executable, str(self.script), runtime, + "--task", str(task or self.task), "--output", str(trial), + "--executable", str(executable), + "--timeout-seconds", str(timeout_seconds), + ], + cwd=REPOSITORY_ROOT, + env=environment, + text=True, + capture_output=True, + timeout=10, + ) + + def _assert_trial_does_not_expose_hidden_oracle(self, trial): + hidden_name = "hil-oracle.json" + for path in trial.rglob("*"): + with self.subTest(path=path): + self.assertNotIn(hidden_name, str(path)) + if path.is_file(): + self.assertNotIn(hidden_name, path.read_text(encoding="utf-8", errors="replace")) + self.assertFalse((trial / "candidate/hidden").exists()) + + def test_native_runtimes_create_completed_public_candidates(self): + for runtime in ("opencode", "codex", "claude"): + with self.subTest(runtime=runtime), tempfile.TemporaryDirectory() as directory: + root = Path(directory) + executable = self._fake_runtime(root / runtime, runtime) + trial = root / "trial" + + completed = self._run_cli(runtime, executable, trial) + + self.assertEqual(completed.returncode, 0, completed.stderr) + result = json.loads((trial / "agent-result.json").read_text()) + usage = json.loads((trial / "usage.json").read_text()) + candidate_source = (trial / "candidate/firmware/src/main.c").read_text() + self.assertEqual(result["status"], "completed") + self.assertEqual(result["runtime"], runtime) + self.assertIsNone(result["model_override"]) + self.assertEqual(result["native_model"], f"fake-{runtime}-model") + self.assertIsNone(result["native_model_unavailable_reason"]) + self.assertEqual(result["returncode"], 0) + self.assertFalse(result["timed_out"]) + self.assertGreaterEqual(result["elapsed_seconds"], 0) + self.assertEqual(result["executable_version"], f"fake-{runtime} 1.0") + self.assertIn("GPIO_MODE_OUTPUT", candidate_source) + self.assertTrue((trial / "agent.stdout.log").is_file()) + self.assertTrue((trial / "agent.stderr.log").is_file()) + self.assertTrue((trial / "runtime-config").is_dir()) + self.assertFalse((trial / "candidate/task.json").exists()) + self.assertEqual(usage["requests"], 1) + self.assertIsNone(usage["unavailable_reason"]) + self._assert_trial_does_not_expose_hidden_oracle(trial) + if runtime == "codex": + isolated_home = Path((trial / "candidate/effective-codex-home").read_text()) + self.assertFalse(isolated_home.exists()) + for path in trial.rglob("*"): + if path.is_file(): + self.assertNotIn( + "sentinel-codex-auth-credential", + path.read_text(encoding="utf-8", errors="replace"), + ) + + def test_codex_without_source_auth_keeps_an_isolated_config(self): + with tempfile.TemporaryDirectory() as directory: + root = Path(directory) + executable = self._fake_runtime( + root / "runtime", "codex", expects_auth=False, + ) + trial = root / "trial" + + completed = self._run_cli("codex", executable, trial, source_auth=False) + + self.assertEqual(completed.returncode, 0, completed.stderr) + result = json.loads((trial / "agent-result.json").read_text()) + self.assertEqual(result["status"], "completed") + isolated_home = Path((trial / "candidate/effective-codex-home").read_text()) + self.assertFalse(isolated_home.exists()) + + def test_nonzero_runtime_is_failed_but_retains_evidence(self): + with tempfile.TemporaryDirectory() as directory: + root = Path(directory) + executable = self._fake_runtime(root / "runtime", "codex", "nonzero") + trial = root / "trial" + + completed = self._run_cli("codex", executable, trial) + + self.assertEqual(completed.returncode, 0, completed.stderr) + result = json.loads((trial / "agent-result.json").read_text()) + self.assertEqual(result["status"], "failed") + self.assertEqual(result["returncode"], 9) + self.assertFalse(result["timed_out"]) + self.assertTrue((trial / "agent.stdout.log").is_file()) + self._assert_trial_does_not_expose_hidden_oracle(trial) + + def test_timeout_runtime_is_recorded(self): + with tempfile.TemporaryDirectory() as directory: + root = Path(directory) + executable = self._fake_runtime(root / "runtime", "opencode", "timeout") + trial = root / "trial" + + completed = self._run_cli("opencode", executable, trial, timeout_seconds=0.1) + + self.assertEqual(completed.returncode, 0, completed.stderr) + result = json.loads((trial / "agent-result.json").read_text()) + self.assertEqual(result["status"], "timeout") + self.assertTrue(result["timed_out"]) + self.assertNotEqual(result["returncode"], 0) + self._assert_trial_does_not_expose_hidden_oracle(trial) + + def test_missing_executable_is_an_infrastructure_error(self): + with tempfile.TemporaryDirectory() as directory: + trial = Path(directory) / "trial" + + completed = self._run_cli("claude", Path(directory) / "missing", trial) + + self.assertEqual(completed.returncode, 0, completed.stderr) + result = json.loads((trial / "agent-result.json").read_text()) + self.assertEqual(result["status"], "infrastructure_error") + self.assertIsNone(result["returncode"]) + self._assert_trial_does_not_expose_hidden_oracle(trial) + + def test_missing_or_malformed_usage_does_not_fail_a_completed_trial(self): + for mode in ("malformed", "missing"): + with self.subTest(mode=mode), tempfile.TemporaryDirectory() as directory: + root = Path(directory) + executable = self._fake_runtime(root / "runtime", "claude", mode) + trial = root / "trial" + + completed = self._run_cli("claude", executable, trial) + + self.assertEqual(completed.returncode, 0, completed.stderr) + result = json.loads((trial / "agent-result.json").read_text()) + usage = json.loads((trial / "usage.json").read_text()) + self.assertEqual(result["status"], "completed") + self.assertEqual(usage["unavailable_reason"], "runtime did not expose usage") + self._assert_trial_does_not_expose_hidden_oracle(trial) + + def test_missing_native_model_has_an_explicit_unavailable_reason(self): + with tempfile.TemporaryDirectory() as directory: + root = Path(directory) + executable = executable_fixture(root / "runtime", textwrap.dedent(""" + import json + import sys + + if sys.argv[1:] == ["--version"]: + print("fake-claude 1.0") + raise SystemExit(0) + print(json.dumps({ + "type": "result", + "usage": {"input_tokens": 1, "output_tokens": 1}, + })) + """)) + trial = root / "trial" + + completed = self._run_cli("claude", executable, trial) + + self.assertEqual(completed.returncode, 0, completed.stderr) + result = json.loads((trial / "agent-result.json").read_text()) + self.assertIsNone(result["native_model"]) + self.assertEqual( + result["native_model_unavailable_reason"], + "runtime did not expose model", + ) + + def test_existing_output_is_rejected_without_overwrite(self): + with tempfile.TemporaryDirectory() as directory: + root = Path(directory) + trial = root / "trial" + trial.mkdir() + marker = trial / "keep" + marker.write_text("existing") + + completed = self._run_cli("codex", root / "missing", trial) + + self.assertEqual(completed.returncode, 2) + self.assertIn("output path already exists", completed.stderr) + self.assertEqual(marker.read_text(), "existing") + self.assertFalse((trial / "agent-result.json").exists()) + + def test_public_symlink_is_rejected_before_candidate_copy(self): + with tempfile.TemporaryDirectory() as directory: + root = Path(directory) + task = root / "task" + public = task / "public" + hidden = task / "hidden" + public.mkdir(parents=True) + hidden.mkdir() + oracle = hidden / "hil-oracle.json" + oracle.write_text("hidden oracle evidence", encoding="utf-8") + (public / "leaked-oracle").symlink_to(oracle) + (task / "task.json").write_text(json.dumps({ + "public_dir": "public", + "budgets": {"wall_time_seconds": 1, "repair_iterations": 1}, + }), encoding="utf-8") + trial = root / "trial" + + completed = self._run_cli( + "codex", root / "missing", trial, task=task, + ) + + self.assertEqual(completed.returncode, 0, completed.stderr) + result = json.loads((trial / "agent-result.json").read_text()) + self.assertEqual(result["status"], "infrastructure_error") + self.assertIn("symlink", result["error"]) + self.assertFalse((trial / "candidate").exists()) + self.assertNotIn("hidden oracle evidence", (trial / "agent-result.json").read_text()) + + def test_prompt_uses_the_task_repair_iteration_budget(self): + with tempfile.TemporaryDirectory() as directory: + root = Path(directory) + task = root / "task" + import shutil + shutil.copytree(self.task / "public", task / "public") + (task / "task.json").write_text(json.dumps({ + "public_dir": "public", + "budgets": {"wall_time_seconds": 2, "repair_iterations": 2}, + }), encoding="utf-8") + executable = self._fake_runtime( + root / "runtime", "opencode", repair_iterations=2, + ) + trial = root / "trial" + + completed = self._run_cli("opencode", executable, trial, task=task) + + self.assertEqual(completed.returncode, 0, completed.stderr) + result = json.loads((trial / "agent-result.json").read_text()) + self.assertEqual(result["status"], "completed") + self.assertFalse((trial / "candidate/task.json").exists()) + + def test_invalid_trial_budgets_are_rejected_before_candidate_copy(self): + cases = ( + ({"wall_time_seconds": 1, "repair_iterations": 0}, "repair_iterations"), + ({"wall_time_seconds": float("nan"), "repair_iterations": 1}, "wall_time_seconds"), + ) + for budgets, expected_error in cases: + with self.subTest(budgets=budgets), tempfile.TemporaryDirectory() as directory: + root = Path(directory) + task = root / "task" + (task / "public").mkdir(parents=True) + (task / "task.json").write_text(json.dumps({ + "public_dir": "public", "budgets": budgets, + }), encoding="utf-8") + trial = root / "trial" + + completed = self._run_cli("codex", root / "missing", trial, task=task) + + self.assertEqual(completed.returncode, 0, completed.stderr) + result = json.loads((trial / "agent-result.json").read_text()) + self.assertEqual(result["status"], "infrastructure_error") + self.assertIn(expected_error, result["error"]) + self.assertFalse((trial / "candidate").exists()) + + +if __name__ == "__main__": + if "-k" in sys.argv: + pattern_index = sys.argv.index("-k") + 1 + if pattern_index < len(sys.argv) and " or " in sys.argv[pattern_index]: + patterns = sys.argv.pop(pattern_index).split(" or ") + sys.argv.pop(pattern_index - 1) + for pattern in patterns: + sys.argv.extend(("-k", pattern)) + unittest.main() diff --git a/tests/twin2silicon-runtime-smoke.sh b/tests/twin2silicon-runtime-smoke.sh new file mode 100755 index 0000000..35199af --- /dev/null +++ b/tests/twin2silicon-runtime-smoke.sh @@ -0,0 +1,82 @@ +#!/usr/bin/env bash +# Run the connected-board runtime smoke matrix only after an explicit opt-in. +set -euo pipefail + +if [[ "${LABWIRED_HIL:-}" != "1" ]]; then + echo "LABWIRED_HIL=1 required; this command flashes connected hardware." >&2 + exit 2 +fi + +require_variable() { + local name="$1" + if [[ -z "${!name:-}" ]]; then + echo "$name required" >&2 + exit 2 + fi +} + +require_command() { + local name="$1" + if ! command -v "$name" >/dev/null 2>&1; then + echo "$name is required but was not found on PATH" >&2 + exit 2 + fi +} + +print_version() { + local label="$1" + shift + local version + if ! version="$("$@" 2>&1)"; then + echo "unable to determine $label version" >&2 + exit 2 + fi + printf '%s version: %s\n' "$label" "${version%%$'\n'*}" +} + +require_variable LABWIRED_UART_DEVICE +require_variable LABWIRED_JTAG_SERIAL +require_variable LABWIRED_OPENOCD +require_variable LABWIRED_MATRIX_OUTPUT +require_command opencode +require_command codex +require_command claude +require_command pio + +if [[ ! -x "$LABWIRED_OPENOCD" ]]; then + echo "LABWIRED_OPENOCD must name an executable OpenOCD binary" >&2 + exit 2 +fi + +if [[ -d /Volumes/LabWired && -w /Volumes/LabWired ]]; then + temporary_root="$(mktemp -d /Volumes/LabWired/twin2silicon-runtime-smoke.XXXXXX)" +else + temporary_root="$(mktemp -d)" +fi +trap 'rm -rf "$temporary_root"' EXIT +export TMPDIR="$temporary_root" + +print_version opencode opencode --version +print_version codex codex --version +print_version claude claude --version +print_version pio pio --version +print_version openocd "$LABWIRED_OPENOCD" --version + +repository_root="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" +task="${LABWIRED_TASK:-esp32s3-gpio-hil-001}" +identity_command_json="$(python3 - "$repository_root/benchmarks/twin2silicon/identify_pio_device.py" "$LABWIRED_UART_DEVICE" "$LABWIRED_JTAG_SERIAL" <<'PY' +import json +import sys + +print(json.dumps([sys.executable, sys.argv[1], "--uart-device", sys.argv[2], + "--jtag-serial", sys.argv[3]])) +PY +)" + +python3 "$repository_root/benchmarks/twin2silicon/run_matrix.py" \ + --task "$task" \ + --output "$LABWIRED_MATRIX_OUTPUT" \ + --jtag-serial "$LABWIRED_JTAG_SERIAL" \ + --uart-device "$LABWIRED_UART_DEVICE" \ + --openocd "$LABWIRED_OPENOCD" \ + --identity-command-json "$identity_command_json"