iop/scripts/agent_comparison_benchmark.py
toki de4d8f4ff8 feat(benchmark): IOP Agent 연결 경로를 추가한다
세 Agent의 direct route를 동일한 fail-closed preflight와 격리 실행 경계에서 비교하고, 관측되지 않은 preset 셀이 실행되는 것을 막기 위해 연결 계약과 증거 수집 흐름을 고정한다.
2026-08-10 08:11:04 +09:00

299 lines
10 KiB
Python

#!/usr/bin/env python3
"""
Public CLI for the agent comparison benchmark manifest.
Usage:
python3 scripts/agent_comparison_benchmark.py validate --manifest PATH
python3 scripts/agent_comparison_benchmark.py preflight --manifest PATH
python3 scripts/agent_comparison_benchmark.py run --manifest PATH
python3 scripts/agent_comparison_benchmark.py resume --manifest PATH --run-id RUN_ID
python3 scripts/agent_comparison_benchmark.py status --manifest PATH --run-id RUN_ID
Exits:
0 - manifest is valid or every direct preflight cell is ready
64 - usage error (missing args, bad flags)
69 - validation/state failed or preflight is blocked
"""
from __future__ import annotations
import argparse
import hashlib
import sys
from pathlib import Path
from typing import Callable
# Ensure the repo root is on sys.path for imports.
_REPO_ROOT = Path(__file__).resolve().parent.parent
if str(_REPO_ROOT) not in sys.path:
sys.path.insert(0, str(_REPO_ROOT))
from scripts.agent_benchmark.manifest import (
CALLER_ENUM,
MatrixCell,
ManifestError,
Timeout,
load_manifest,
)
from scripts.agent_benchmark.attempts import (
Attempt,
CapabilityUnavailable,
ExecutionAdapter,
PreflightObservation,
RunStore,
preflight_manifest,
run_slots,
)
from scripts.agent_benchmark.connectivity import (
ISSUE_RESUME_CODES,
CallerCapability,
ConnectivityIssue,
RequestedEffectiveBinding,
make_result,
)
from scripts.agent_benchmark.claude_iop import claude_capability
from scripts.agent_benchmark.agy_iop import AGY_CALLER
from scripts.agent_benchmark.codex_iop import codex_capability
from scripts.agent_benchmark.lifecycle import InvocationResult, SupervisorLocator
from scripts.agent_benchmark.workspace import PreparedWorkspace, prepare_workspace
EXIT_VALID = 0
EXIT_USAGE = 64
EXIT_INVALID = 69
class _RegisteredExecutionAdapter:
"""Typed execution registration with an explicit live-observation gap.
A later authorized-live adapter can replace these registrations without
changing the CLI, evidence schema, or run writer. Until then a direct cell
is never reported ready from requested values alone, so invoke is unreachable.
"""
def __init__(self, capability: CallerCapability) -> None:
self.capability = capability
@staticmethod
def _identity(caller: str, kind: str) -> str:
raw = f"iop-benchmark-unobserved-v1:{caller}:{kind}".encode("ascii")
return "sha256:" + hashlib.sha256(raw).hexdigest()
def preflight(self, cell: MatrixCell) -> PreflightObservation:
binding = RequestedEffectiveBinding(
cell.id,
cell.caller,
cell.iop.route_kind,
cell.iop.route_id,
cell.iop.request_model,
cell.iop.requested_effort,
)
issue = ConnectivityIssue(
"stream_incompatible", ISSUE_RESUME_CODES["stream_incompatible"]
)
result = make_result(cell, self.capability, binding, (issue,))
return PreflightObservation(
result,
self._identity(cell.caller, "endpoint"),
self._identity(cell.caller, "config"),
)
def invoke(
self,
cell: MatrixCell,
prepared: PreparedWorkspace,
attempt: Attempt,
task_payload: bytes,
timeout: Timeout,
on_started: Callable[[SupervisorLocator, str], None],
) -> InvocationResult:
"""Remain unreachable until a live observer replaces this registration."""
raise CapabilityUnavailable("capability-unavailable: caller-adapter")
def build_adapter_registry() -> dict[str, ExecutionAdapter]:
"""Build the exact three-caller registry from completed adapter modules."""
registry: dict[str, ExecutionAdapter] = {
"claude": _RegisteredExecutionAdapter(claude_capability()),
# agy's completed module exposes its documented caller constant while
# the same closed capability tuple is enforced by its preflight parser.
"agy": _RegisteredExecutionAdapter(
CallerCapability(
AGY_CALLER,
("direct", "execution_preset"),
("high", "low", "medium"),
)
),
"codex": _RegisteredExecutionAdapter(codex_capability()),
}
if tuple(registry) != CALLER_ENUM:
raise RuntimeError("caller adapter registry is invalid")
return registry
class _SanitizedArgumentParser(argparse.ArgumentParser):
def error(self, message: str) -> None:
print("error: invalid usage", file=sys.stderr)
sys.exit(EXIT_USAGE)
def _build_parser() -> argparse.ArgumentParser:
parser = _SanitizedArgumentParser(
prog="agent_comparison_benchmark",
description="Agent comparison benchmark manifest tools.",
)
sub = parser.add_subparsers(dest="command", required=True)
p_validate = sub.add_parser(
"validate",
help="Validate a benchmark manifest JSON file.",
)
p_validate.add_argument(
"--manifest",
required=True,
help="Path to the manifest JSON file.",
)
for command in ("preflight", "run", "resume", "status"):
entry = sub.add_parser(command, help=f"Safely {command} benchmark state.")
entry.add_argument("--manifest", required=True, help="Path to the manifest JSON file.")
if command in {"resume", "status"}:
entry.add_argument("--run-id", required=True, help="Harness-generated run id.")
if command == "resume":
entry.add_argument("--retry-failed", action="store_true")
return parser
def _cmd_validate(args: argparse.Namespace) -> int:
manifest_path = Path(args.manifest)
if not manifest_path.is_file():
print("error: manifest is unavailable", file=sys.stderr)
return EXIT_INVALID
try:
load_manifest(manifest_path)
except ManifestError as exc:
print(f"error: {exc}", file=sys.stderr)
return EXIT_INVALID
except Exception:
print("error: manifest validation failed", file=sys.stderr)
return EXIT_INVALID
print("ok: manifest is valid")
return EXIT_VALID
def _cmd_state(args: argparse.Namespace) -> int:
try:
manifest_path = Path(args.manifest)
manifest = load_manifest(manifest_path, repo_root=_REPO_ROOT)
raw = manifest_path.read_bytes()
store = RunStore(_REPO_ROOT)
if args.command == "run":
run = store.create(manifest, raw)
else:
run = store.open(manifest, args.run_id, raw)
if args.command == "status":
print("ok: " + str(store.status(run, manifest)["attempts"]))
return EXIT_VALID
completed = run_slots(
store,
run,
manifest,
adapters=build_adapter_registry(),
prepare=lambda bound_manifest, attempt: prepare_workspace(
bound_manifest,
attempt.root,
attempt.identity,
repo_root=_REPO_ROOT,
),
retry_failed=bool(getattr(args, "retry_failed", False)),
)
status = store.status(run, manifest)
preflight = status["preflight"]
preflight_summary = (
f"run_id={run.run_id} status={preflight['latest_status']} "
f"ready={preflight['ready']} "
f"registration_required={preflight['registration_required']} "
f"implementation_gap={preflight['implementation_gap']}"
)
if preflight["latest_status"] != "ready":
print("error: preflight blocked " + preflight_summary, file=sys.stderr)
return EXIT_INVALID
attempts = status["attempts"]
attempt_summary = " ".join(
f"{state}={attempts[state]}"
for state in (
"success", "failed", "timed_out", "cancelled", "interrupted", "running"
)
)
unresolved = 0
for slot in store.slots(manifest):
retained = store.attempts(run, slot)
if not retained or retained[-1].state != "success":
unresolved += 1
summary = (
f"run_id={run.run_id} completed={len(completed)} "
f"unresolved={unresolved} {attempt_summary}"
)
if unresolved:
print("error: benchmark execution failed " + summary, file=sys.stderr)
return EXIT_INVALID
print(f"ok: {args.command} " + summary)
return EXIT_VALID
except CapabilityUnavailable:
print("error: capability unavailable", file=sys.stderr)
except Exception:
print("error: benchmark state is unavailable", file=sys.stderr)
return EXIT_INVALID
def _cmd_preflight(args: argparse.Namespace) -> int:
try:
manifest_path = Path(args.manifest)
manifest = load_manifest(manifest_path, repo_root=_REPO_ROOT)
raw = manifest_path.read_bytes()
run, record = preflight_manifest(
RunStore(_REPO_ROOT),
manifest,
raw,
adapters=build_adapter_registry(),
)
counts = {status: 0 for status in ("ready", "registration_required", "implementation_gap")}
for result in record["results"]:
counts[result["status"]] += 1
summary = (
f"run_id={run.run_id} status={record['status']} "
f"ready={counts['ready']} "
f"registration_required={counts['registration_required']} "
f"implementation_gap={counts['implementation_gap']}"
)
if record["status"] == "ready":
print("ok: preflight " + summary)
return EXIT_VALID
print("error: preflight blocked " + summary, file=sys.stderr)
except CapabilityUnavailable:
print("error: capability unavailable", file=sys.stderr)
except Exception:
print("error: benchmark preflight is unavailable", file=sys.stderr)
return EXIT_INVALID
def main(argv: list[str] | None = None) -> int:
parser = _build_parser()
try:
args = parser.parse_args(argv)
except SystemExit as exc:
return exc.code if isinstance(exc.code, int) else EXIT_USAGE
if args.command == "validate":
return _cmd_validate(args)
if args.command == "preflight":
return _cmd_preflight(args)
if args.command in {"run", "resume", "status"}:
return _cmd_state(args)
return EXIT_USAGE
if __name__ == "__main__":
sys.exit(main())