iop/scripts/agent_benchmark/skill_contract_test.py
toki de4d8f4ff8 feat(benchmark): IOP Agent 연결 경로를 추가한다
세 Agent의 direct route를 동일한 fail-closed preflight와 격리 실행 경계에서 비교하고, 관측되지 않은 preset 셀이 실행되는 것을 막기 위해 연결 계약과 증거 수집 흐름을 고정한다.
2026-08-10 08:11:04 +09:00

832 lines
39 KiB
Python

"""
Credential-free contract tests binding the benchmark skill to the CLI surface.
Covers: template/frontmatter invariants, project rule routing, documented
command forms against --help and required options, supported/unsupported capability matrix,
absence of the internal workspace API from user routing, provider/dispatcher/secret/prepare prohibitions,
durable state vs isolated cache boundaries, and mutation resistance.
Invokes only help and tracked text; never runs a stateful benchmark command.
"""
from __future__ import annotations
import json
import re
import subprocess
import sys
import unittest
from pathlib import Path
# Ensure repo root is importable
_REPO_ROOT = Path(__file__).resolve().parent.parent.parent
if str(_REPO_ROOT) not in sys.path:
sys.path.insert(0, str(_REPO_ROOT))
_SKILL_DIR = _REPO_ROOT / "agent-ops" / "skills" / "project" / "iop-agent-comparison-benchmark"
_SKILL_FILE = _SKILL_DIR / "SKILL.md"
_RULES_FILE = _REPO_ROOT / "agent-ops" / "rules" / "project" / "rules.md"
_CLI_SCRIPT = _REPO_ROOT / "scripts" / "agent_comparison_benchmark.py"
_CAPABILITY_REPORT_OUTPUT = "capability-unavailable: report-output"
# Commands documented by the CLI --help
_CLI_HELP_COMMANDS = {"validate", "preflight", "run", "resume", "status"}
# Cached exact option sets per subcommand, derived from each subcommand --help.
_CLI_OPTION_CACHE: dict[str, set[str]] = {}
class BenchmarkSkillContractTest(unittest.TestCase):
"""Contract tests for the iop-agent-comparison-benchmark skill."""
# ------------------------------------------------------------------
# Helper methods for section parsing & semantic assertions
# ------------------------------------------------------------------
def _get_section(self, skill_text: str, section_name: str) -> str:
in_section = False
lines = []
target = f"## {section_name.strip()}"
for line in skill_text.splitlines():
if line.startswith("## "):
if line.strip().lower() == target.lower():
in_section = True
continue
elif in_section:
break
if in_section:
lines.append(line)
return "\n".join(lines)
def _get_cli_help_commands(self) -> set[str]:
result = subprocess.run(
[sys.executable, str(_CLI_SCRIPT), "--help"],
capture_output=True,
text=True,
cwd=str(_REPO_ROOT),
)
self.assertEqual(
result.returncode,
0,
f"CLI --help exited {result.returncode}: {result.stderr}",
)
commands: set[str] = set()
in_subparsers = False
for line in result.stdout.splitlines():
stripped = line.strip()
if "positional arguments:" in line:
in_subparsers = True
continue
if in_subparsers and stripped and not stripped.startswith("-"):
if stripped == "options:":
continue
if stripped.startswith("{") and stripped.endswith("}"):
for cmd in stripped[1:-1].split(","):
commands.add(cmd)
else:
commands.add(stripped.split()[0])
return commands
def _assert_no_public_prepare(self, skill_text: str) -> None:
"""Parse all sections; fail if public prepare is exposed as an operation or trigger."""
allowed_prepare_fragments = (
"no public `prepare` operation was exposed or referenced.",
"do not expose a public `prepare` operation.",
)
for line in skill_text.splitlines():
stripped = line.strip()
if "prepare" in stripped.lower():
lower = stripped.lower()
if not any(fragment in lower for fragment in allowed_prepare_fragments):
self.fail(
"Section or step exposes or mentions prepare operation "
f"outside the approved prohibition forms: '{stripped}'"
)
def _assert_provider_prohibition(self, skill_text: str) -> None:
"""Assert provider APIs/services are prohibited and never invoked in any section."""
prohibitions = self._get_section(skill_text, "Prohibitions")
self.assertRegex(
prohibitions,
r"(?i)do not invoke.*provider",
"Prohibitions must explicitly forbid provider invocations",
)
allowed_provider_fragments = (
"no caller or provider was invoked outside the deterministic cli.",
"do not invoke a caller or provider outside the deterministic benchmark cli.",
"stop without fallback, fabricated evidence, ad-hoc provider calls, subagents, or orchestration dispatchers.",
)
for line in skill_text.splitlines():
stripped = line.strip()
if "provider" in stripped.lower():
lower = stripped.lower()
if not any(fragment in lower for fragment in allowed_provider_fragments):
self.fail(
"Line contains an unapproved provider operation or "
f"prohibition form: '{stripped}'"
)
def _get_cli_subcommand_options(self, cmd: str) -> set[str]:
"""Derive the exact long-option set for a subcommand from its live --help output."""
if cmd in _CLI_OPTION_CACHE:
return set(_CLI_OPTION_CACHE[cmd])
result = subprocess.run(
[sys.executable, str(_CLI_SCRIPT), cmd, "--help"],
capture_output=True,
text=True,
cwd=str(_REPO_ROOT),
)
self.assertEqual(
result.returncode,
0,
f"CLI {cmd} --help exited {result.returncode}: {result.stderr}",
)
options = set(re.findall(r"--[a-z][a-z0-9-]*", result.stdout))
options.discard("--help")
_CLI_OPTION_CACHE[cmd] = set(options)
return options
def _documented_command_options(self, cmd_line: str) -> set[str]:
"""Extract the documented long-option set, excluding --help and value placeholders."""
options: set[str] = set()
for match in re.finditer(r"--[a-z][a-z0-9-]*", cmd_line):
token = match.group(0)
if token != "--help":
options.add(token)
return options
def _assert_command_options(self, skill_text: str) -> None:
"""Parse each documented CLI invocation and require exact option parity with subcommand --help."""
procedure = self._get_section(skill_text, "Procedure")
for cmd in ("validate", "preflight", "run", "resume", "status"):
pattern = rf"python3 scripts/agent_comparison_benchmark\.py {cmd}\b[^\n]*"
matches = re.findall(pattern, procedure)
self.assertTrue(matches, f"Documented command line for '{cmd}' missing from Procedure")
cli_options = self._get_cli_subcommand_options(cmd)
for index, cmd_line in enumerate(matches, start=1):
documented = self._documented_command_options(cmd_line)
self.assertEqual(
documented,
cli_options,
f"Documented options {sorted(documented)} for '{cmd}' invocation "
f"#{index} must exactly match CLI --help options "
f"{sorted(cli_options)}: '{cmd_line}'",
)
def _assert_boundary_wording(self, skill_text: str) -> None:
"""Assert consistent durable-state, isolated cache, read-only testbed, and output boundaries."""
self.assertIn("Durable run/attempt state", skill_text)
self.assertIn("agent-test/runs/<output-id>/<run-id>/", skill_text)
self.assertRegex(
skill_text,
r"fresh and isolated for every cell, repetition, and attempt",
"Skill text must specify per-cell/per-repetition/per-attempt freshness and isolation",
)
self.assertIn("../iop-s2", skill_text)
self.assertIn("read-only", skill_text)
forbidden_phrases = [
"do not cache or persist state between invocations",
"do not read or write files outside the benchmark workspace",
"The benchmark workspace root is fixed at `../iop-s2`.",
"Output is contained within the benchmark workspace.",
"fresh, isolated per run, and never shared across runs",
]
for phrase in forbidden_phrases:
self.assertNotIn(phrase, skill_text, f"Forbidden contradictory boundary phrase found: {phrase}")
prohibitions = self._get_section(skill_text, "Prohibitions")
self.assertRegex(
prohibitions,
r"(?i)do not share session or cache state within a run",
"Prohibitions must explicitly forbid session or cache state sharing within a run",
)
for line in skill_text.splitlines():
stripped = line.strip()
lower = stripped.lower()
if (
"session" in lower
and "cache" in lower
and any(token in lower for token in ("share", "shared", "sharing"))
):
allowed_boundary_fragments = (
"caller sessions, output workspaces, and caches are fresh and isolated for every cell, repetition, and attempt; session or cache state is never shared within a run or across runs.",
"do not share session or cache state within a run across cells, repetitions, or attempts, or across run invocations.",
)
if not any(fragment in lower for fragment in allowed_boundary_fragments):
self.fail(
"Line contains an unapproved affirmative or contradictory "
f"session/cache sharing statement: '{stripped}'"
)
elif "cache" in lower and "sharing" in lower:
self.fail(f"Line contains an unapproved cache-sharing statement: '{stripped}'")
def _assert_error_ordering(self, skill_text: str) -> None:
"""Assert invalid state is handled before execution preflight blockers."""
procedure = self._get_section(skill_text, "Procedure")
for cmd in ("run", "resume", "status"):
pattern = rf"\d+\.\s+\*\*Delegate {cmd}.*?(?=\n\d+\.|\Z)"
match = re.search(pattern, procedure, re.DOTALL)
self.assertTrue(match, f"Procedure step for '{cmd}' missing")
step_text = match.group(0)
if cmd in ("run", "resume"):
pos_invalid = step_text.find("benchmark state is unavailable")
pos_blocked = step_text.find("preflight blocked")
self.assertTrue(
pos_invalid != -1
and pos_blocked != -1
and pos_invalid < pos_blocked,
f"In step '{cmd}', invalid state must precede preflight blockers",
)
def _assert_capabilities(self, skill_text: str) -> None:
"""Assert report remains unavailable while run/resume are executable."""
self.assertIn(_CAPABILITY_REPORT_OUTPUT, skill_text)
procedure = self._get_section(skill_text, "Procedure")
self.assertIn(_CAPABILITY_REPORT_OUTPUT, procedure)
self.assertNotIn("capability-unavailable: caller-adapter", procedure)
self.assertIn("append a fresh preflight before attempt allocation", procedure)
self.assertIn("invoke each eligible cell exactly once", procedure)
def _assert_preflight_contract(self, skill_text: str) -> None:
"""Require direct-only append semantics and fail-closed blocker language."""
procedure = self._get_section(skill_text, "Procedure")
validation = self._get_section(skill_text, "Validation")
prohibitions = self._get_section(skill_text, "Prohibitions")
self.assertIn(
"python3 scripts/agent_comparison_benchmark.py preflight --manifest <manifest-path>",
procedure,
)
self.assertIn("records only direct-cell observations", procedure)
self.assertIn("Generic preset cells are local contract validation only", skill_text)
self.assertIn("registration_required", procedure)
self.assertIn("implementation_gap", procedure)
self.assertIn("Never bypass the blocker", procedure)
self.assertIn("substitute a route/model/effort", procedure)
self.assertIn("Preflight evidence is append-only, direct-only", validation)
self.assertIn("created no scored attempt", validation)
self.assertIn("Do not bypass a preflight blocker", prohibitions)
self.assertIn("Do not claim execution-preset fixture validation as live readiness", prohibitions)
def _assert_no_secret_operational_language(self, skill_text: str) -> None:
operational_text = "\n".join(
self._get_section(skill_text, section)
for section in ("Inputs", "Preflight", "Procedure")
)
self.assertNotRegex(
operational_text,
r"(?i)\b(secret|credential|api_key|token)\b",
"Operational sections must not mention secrets or credentials",
)
def _assert_full_skill_contract(self, skill_text: str) -> None:
"""Validate complete contract on skill text."""
self.assertIn("## Purpose", skill_text)
self.assertIn("## When to use", skill_text)
self.assertIn("## Preflight", skill_text)
self.assertIn("## Procedure", skill_text)
self.assertIn("## Validation", skill_text)
self.assertIn("## Safety rules", skill_text)
self.assertIn("## Stop conditions", skill_text)
self.assertIn("## Prohibitions", skill_text)
self._assert_command_options(skill_text)
self._assert_no_public_prepare(skill_text)
self._assert_provider_prohibition(skill_text)
self._assert_boundary_wording(skill_text)
self._assert_error_ordering(skill_text)
self._assert_capabilities(skill_text)
self._assert_preflight_contract(skill_text)
self._assert_no_secret_operational_language(skill_text)
# ------------------------------------------------------------------
# Template / frontmatter invariants
# ------------------------------------------------------------------
def test_skill_file_exists(self) -> None:
self.assertTrue(_SKILL_FILE.is_file(), f"{_SKILL_FILE} must exist")
def test_frontmatter_name(self) -> None:
content = _SKILL_FILE.read_text(encoding="utf-8")
self.assertIn("name: iop-agent-comparison-benchmark", content)
def test_frontmatter_keys(self) -> None:
content = _SKILL_FILE.read_text(encoding="utf-8")
frontmatter = content.split("---", 2)[1]
keys = [
line.partition(":")[0]
for line in frontmatter.splitlines()
if line.strip()
]
self.assertEqual(keys, ["name", "description"])
def test_frontmatter_description_present(self) -> None:
content = _SKILL_FILE.read_text(encoding="utf-8")
self.assertRegex(content, r"description: .+", re.MULTILINE)
def test_required_sections_present(self) -> None:
content = _SKILL_FILE.read_text(encoding="utf-8")
for section in (
"## Purpose",
"## When to use",
"## Preflight",
"## Procedure",
"## Validation",
"## Prohibitions",
):
self.assertIn(section, content, f"Missing section: {section}")
# ------------------------------------------------------------------
# Routing
# ------------------------------------------------------------------
def test_project_rules_routes_benchmark(self) -> None:
rules_text = _RULES_FILE.read_text(encoding="utf-8")
self.assertIn(
"agent-ops/skills/project/iop-agent-comparison-benchmark/SKILL.md",
rules_text,
"Project rules must route to the benchmark skill",
)
def test_project_rules_routes_trigger_keywords(self) -> None:
rules_text = _RULES_FILE.read_text(encoding="utf-8")
for keyword in ("validate", "run", "resume", "status", "report-readiness"):
self.assertIn(
keyword,
rules_text,
f"Project rules must mention trigger keyword: {keyword}",
)
# ------------------------------------------------------------------
# CLI parity & documented command options matching --help
# ------------------------------------------------------------------
def test_skill_commands_match_cli_help(self) -> None:
"""Documented procedure commands must be a subset of CLI --help commands."""
skill_text = _SKILL_FILE.read_text(encoding="utf-8")
procedure_section = self._get_section(skill_text, "Procedure")
documented_commands: set[str] = set()
for cmd in _CLI_HELP_COMMANDS:
if f"`{cmd}`" in procedure_section or f"`{cmd}`" in skill_text:
documented_commands.add(cmd)
cli_commands = self._get_cli_help_commands()
self.assertTrue(
documented_commands.issubset(cli_commands),
f"Documented commands {documented_commands} must be subset of CLI commands {cli_commands}",
)
def test_documented_command_options_match_cli_help(self) -> None:
"""Documented command forms in Procedure must include all required CLI options."""
skill_text = _SKILL_FILE.read_text(encoding="utf-8")
self._assert_command_options(skill_text)
def test_skill_manifest_required_for_all_commands(self) -> None:
"""Skill must specify manifest required for every manifest command."""
skill_text = _SKILL_FILE.read_text(encoding="utf-8")
inputs_section = self._get_section(skill_text, "Inputs")
preflight_section = self._get_section(skill_text, "Preflight")
self.assertIn("required for validate, preflight, run, resume, status", inputs_section)
self.assertIn("For validate/preflight/run/resume/status: confirm a manifest path is provided", preflight_section)
def test_cli_help_exits_zero(self) -> None:
result = subprocess.run(
[sys.executable, str(_CLI_SCRIPT), "--help"],
capture_output=True,
text=True,
cwd=str(_REPO_ROOT),
)
self.assertEqual(result.returncode, 0)
def test_cli_error_ordering_and_invalid_state(self) -> None:
"""CLI must exit 69 with 'error: benchmark state is unavailable' for invalid manifest/state before capability gates."""
schema_fixture = str(_REPO_ROOT / "scripts" / "fixtures" / "agent-comparison-benchmark-manifest.schema.json")
for cmd in ("run", "resume", "status"):
args = [sys.executable, str(_CLI_SCRIPT), cmd, "--manifest", schema_fixture]
if cmd != "run":
args.extend(["--run-id", "dummy-run-id"])
res = subprocess.run(args, capture_output=True, text=True, cwd=str(_REPO_ROOT))
self.assertEqual(res.returncode, 69, f"{cmd} with invalid state should exit 69")
self.assertIn("error: benchmark state is unavailable", res.stderr)
skill_text = _SKILL_FILE.read_text(encoding="utf-8")
self._assert_error_ordering(skill_text)
# ------------------------------------------------------------------
# Capability gates
# ------------------------------------------------------------------
def test_run_resume_are_available_in_skill(self) -> None:
"""Run/resume must document execution rather than a capability gate."""
skill_text = _SKILL_FILE.read_text(encoding="utf-8")
self.assertNotIn("capability-unavailable: caller-adapter", skill_text)
self.assertIn("append a fresh preflight before attempt allocation", skill_text)
def test_capability_report_output_in_skill(self) -> None:
"""Skill must contain the exact report-output capability string."""
skill_text = _SKILL_FILE.read_text(encoding="utf-8")
self.assertIn(
_CAPABILITY_REPORT_OUTPUT,
skill_text,
"Skill must contain exact capability-unavailable: report-output string",
)
def test_run_resume_execution_contract_in_procedure(self) -> None:
"""Run/resume procedure must document ready execution and blockers."""
skill_text = _SKILL_FILE.read_text(encoding="utf-8")
procedure_text = self._get_section(skill_text, "Procedure")
self.assertNotIn("capability-unavailable: caller-adapter", procedure_text)
self.assertIn("invoke each eligible cell exactly once", procedure_text)
self.assertIn("allocates no attempt", procedure_text)
def test_capability_report_output_in_procedure(self) -> None:
"""Report-readiness must reference report-output capability."""
skill_text = _SKILL_FILE.read_text(encoding="utf-8")
self.assertIn(
"report-readiness",
skill_text,
"Skill must mention report-readiness trigger",
)
self.assertIn(
_CAPABILITY_REPORT_OUTPUT,
skill_text,
"Skill must return capability-unavailable: report-output for report-readiness",
)
def test_capability_report_output_available_is_false(self) -> None:
"""report-output must be documented as unavailable."""
skill_text = _SKILL_FILE.read_text(encoding="utf-8")
self.assertIn(
"capability-unavailable: report-output",
skill_text,
"report-output must be marked as capability-unavailable",
)
# ------------------------------------------------------------------
# No public prepare operation
# ------------------------------------------------------------------
def test_no_public_prepare(self) -> None:
"""Skill must not expose a public prepare operation across all steps and sections."""
skill_text = _SKILL_FILE.read_text(encoding="utf-8")
self._assert_no_public_prepare(skill_text)
# ------------------------------------------------------------------
# Provider prohibition
# ------------------------------------------------------------------
def test_provider_prohibition(self) -> None:
"""Skill must explicitly prohibit provider API invocations."""
skill_text = _SKILL_FILE.read_text(encoding="utf-8")
self._assert_provider_prohibition(skill_text)
# ------------------------------------------------------------------
# No dispatcher / secret / fallback language
# ------------------------------------------------------------------
def test_no_dispatcher_reference(self) -> None:
"""Skill must not reference dispatch.py or orchestration dispatchers in procedure."""
skill_text = _SKILL_FILE.read_text(encoding="utf-8")
procedure_text = self._get_section(skill_text, "Procedure")
self.assertNotIn(
"dispatch.py",
procedure_text,
"Procedure must not reference dispatch.py",
)
self.assertNotIn(
"orchestration dispatcher",
procedure_text.lower(),
"Procedure must not use orchestration dispatcher language",
)
def test_no_secret_language(self) -> None:
"""Skill must not reference credential discovery or secrets in operational sections."""
skill_text = _SKILL_FILE.read_text(encoding="utf-8")
self._assert_no_secret_operational_language(skill_text)
def test_no_fallback_language(self) -> None:
"""Skill must not suggest fallback behavior."""
skill_text = _SKILL_FILE.read_text(encoding="utf-8")
procedure_text = self._get_section(skill_text, "Procedure")
self.assertNotIn(
"fallback",
procedure_text.lower(),
"Procedure must not suggest fallback behavior",
)
# ------------------------------------------------------------------
# No internal workspace API exposure
# ------------------------------------------------------------------
def test_no_internal_api_in_user_routing(self) -> None:
"""The internal workspace API must not be a user command."""
skill_text = _SKILL_FILE.read_text(encoding="utf-8")
internal_apis = ["RunStore", "load_manifest", "ManifestDigestError"]
when_to_use = self._get_section(skill_text, "When to use")
procedure = self._get_section(skill_text, "Procedure")
for api in internal_apis:
for line in when_to_use.splitlines() + procedure.splitlines():
stripped = line.strip()
if api in stripped and (stripped.startswith("- ") or stripped.startswith("1.") or stripped.startswith("2.")):
self.fail(f"Internal API {api} exposed in user-facing section: {stripped}")
# ------------------------------------------------------------------
# Safety rules & boundaries
# ------------------------------------------------------------------
def test_safety_rules_present(self) -> None:
"""Skill must document safety rules."""
skill_text = _SKILL_FILE.read_text(encoding="utf-8")
self.assertIn("## Safety rules", skill_text)
def test_fixed_testbed_provenance(self) -> None:
"""Skill must reference the read-only ../iop-s2 testbed."""
skill_text = _SKILL_FILE.read_text(encoding="utf-8")
self.assertIn("../iop-s2", skill_text)
self.assertIn("read-only", skill_text)
def test_testbed_output_state_boundary_wording(self) -> None:
"""Skill must document read-only ../iop-s2 provenance, durable state, and contained agent-test/runs/ output."""
skill_text = _SKILL_FILE.read_text(encoding="utf-8")
self._assert_boundary_wording(skill_text)
# ------------------------------------------------------------------
# CLI help verification
# ------------------------------------------------------------------
def test_cli_help_documents_only_supported_commands(self) -> None:
"""CLI --help should document exactly the five public state commands."""
cli_commands = self._get_cli_help_commands()
self.assertEqual(
cli_commands,
_CLI_HELP_COMMANDS,
f"CLI commands must be exactly {_CLI_HELP_COMMANDS}, got {cli_commands}",
)
# ------------------------------------------------------------------
# CLI delegation verification
# ------------------------------------------------------------------
def test_cli_validate_help(self) -> None:
"""validate subcommand must exist in CLI."""
result = subprocess.run(
[sys.executable, str(_CLI_SCRIPT), "validate", "--help"],
capture_output=True,
text=True,
cwd=str(_REPO_ROOT),
)
self.assertEqual(result.returncode, 0)
self.assertIn("--manifest", result.stdout)
def test_cli_preflight_help(self) -> None:
"""preflight exists and accepts only the manifest input."""
result = subprocess.run(
[sys.executable, str(_CLI_SCRIPT), "preflight", "--help"],
capture_output=True,
text=True,
cwd=str(_REPO_ROOT),
)
self.assertEqual(result.returncode, 0)
self.assertIn("--manifest", result.stdout)
self.assertNotIn("--run-id", result.stdout)
def test_cli_run_help(self) -> None:
"""run subcommand must exist in CLI and require --manifest."""
result = subprocess.run(
[sys.executable, str(_CLI_SCRIPT), "run", "--help"],
capture_output=True,
text=True,
cwd=str(_REPO_ROOT),
)
self.assertEqual(result.returncode, 0)
self.assertIn("--manifest", result.stdout)
def test_cli_status_help(self) -> None:
"""status subcommand must exist in CLI and require --manifest and --run-id."""
result = subprocess.run(
[sys.executable, str(_CLI_SCRIPT), "status", "--help"],
capture_output=True,
text=True,
cwd=str(_REPO_ROOT),
)
self.assertEqual(result.returncode, 0)
self.assertIn("--manifest", result.stdout)
self.assertIn("--run-id", result.stdout)
def test_cli_resume_help(self) -> None:
"""resume subcommand must exist in CLI and require --manifest and --run-id."""
result = subprocess.run(
[sys.executable, str(_CLI_SCRIPT), "resume", "--help"],
capture_output=True,
text=True,
cwd=str(_REPO_ROOT),
)
self.assertEqual(result.returncode, 0)
self.assertIn("--manifest", result.stdout)
self.assertIn("--run-id", result.stdout)
def test_cli_run_is_not_documented_as_capability_unavailable(self) -> None:
"""Contract tests must not execute a stateful run just to prove availability."""
skill_text = _SKILL_FILE.read_text(encoding="utf-8")
run_step = re.search(
r"\d+\.\s+\*\*Delegate run.*?(?=\n\d+\.|\Z)",
self._get_section(skill_text, "Procedure"),
re.DOTALL,
)
self.assertIsNotNone(run_step)
self.assertNotIn(
"capability unavailable", run_step.group(0) # type: ignore[union-attr]
)
# ------------------------------------------------------------------
# Independent mutation regression coverage
#
# Each unsafe variant is verified by its own test method so every
# mutated suite is independently non-zero while the original suite is
# zero (verified by test_base_skill_text_satisfies_full_contract).
# ------------------------------------------------------------------
def _skill_base_text(self) -> str:
return _SKILL_FILE.read_text(encoding="utf-8")
def test_base_skill_text_satisfies_full_contract(self) -> None:
"""Original skill text must satisfy the complete semantic contract (exit-zero baseline)."""
self._assert_full_skill_contract(self._skill_base_text())
def test_mutation_unknown_option_on_run_command(self) -> None:
"""An unknown option on the documented run command must fail exact option parity."""
base = self._skill_base_text()
mutated = base.replace(
"python3 scripts/agent_comparison_benchmark.py run --manifest <manifest-path>",
"python3 scripts/agent_comparison_benchmark.py run --manifest <manifest-path> --bogus-option",
)
self.assertNotEqual(mutated, base, "mutation fixture did not apply")
with self.assertRaises(AssertionError):
self._assert_full_skill_contract(mutated)
def test_mutation_unknown_option_on_later_run_command(self) -> None:
"""An unknown option on a later documented invocation must also fail parity."""
base = self._skill_base_text()
command = " - Run: `python3 scripts/agent_comparison_benchmark.py run --manifest <manifest-path>`"
mutated = base.replace(
command,
command
+ "\n - Run again: `python3 scripts/agent_comparison_benchmark.py run --manifest <manifest-path> --bogus-option`",
)
self.assertNotEqual(mutated, base, "mutation fixture did not apply")
with self.assertRaises(AssertionError):
self._assert_full_skill_contract(mutated)
def test_mutation_missing_required_manifest_on_run_command(self) -> None:
"""Removing the required --manifest option from run must fail exact option parity."""
base = self._skill_base_text()
mutated = base.replace(
"python3 scripts/agent_comparison_benchmark.py run --manifest <manifest-path>",
"python3 scripts/agent_comparison_benchmark.py run",
)
self.assertNotEqual(mutated, base, "mutation fixture did not apply")
with self.assertRaises(AssertionError):
self._assert_full_skill_contract(mutated)
def test_mutation_public_prepare_trigger(self) -> None:
"""A public prepare benchmark trigger in When-to-use must fail the prepare prohibition."""
base = self._skill_base_text()
mutated = base.replace(
"- User requests benchmark manifest validation: `validate`, `validate manifest`, `manifest 검증`",
"- User requests benchmark manifest validation: `validate`, `validate manifest`, `manifest 검증`\n- User requests benchmark preparation: `prepare benchmark`",
)
self.assertNotEqual(mutated, base, "mutation fixture did not apply")
with self.assertRaises(AssertionError):
self._assert_full_skill_contract(mutated)
def test_mutation_affirmative_provider_invocation(self) -> None:
"""An affirmative provider invocation must fail the provider prohibition."""
base = self._skill_base_text()
mutated = base.replace(
"- Do not invoke a caller or provider outside the deterministic benchmark CLI.",
"- Invoke provider APIs when needed for execution.",
)
self.assertNotEqual(mutated, base, "mutation fixture did not apply")
with self.assertRaises(AssertionError):
self._assert_full_skill_contract(mutated)
def test_mutation_additive_provider_contradiction(self) -> None:
"""An additive provider instruction must not bypass the full validator."""
base = self._skill_base_text()
mutated = base.replace(
"- Do not invoke a caller or provider outside the deterministic benchmark CLI.",
"- Do not invoke a caller or provider outside the deterministic benchmark CLI.\n"
"- Do not skip provider API invocation when a benchmark is requested.",
)
self.assertNotEqual(mutated, base, "mutation fixture did not apply")
with self.assertRaises(AssertionError):
self._assert_full_skill_contract(mutated)
def test_mutation_additive_prepare_contradiction(self) -> None:
"""An additive prepare instruction must not bypass the prepare prohibition."""
base = self._skill_base_text()
mutated = base.replace(
"- [ ] No public `prepare` operation was exposed or referenced.",
"- [ ] No public `prepare` operation was exposed or referenced.\n"
"- Do not delay public prepare when the caller requests it.",
)
self.assertNotEqual(mutated, base, "mutation fixture did not apply")
with self.assertRaises(AssertionError):
self._assert_full_skill_contract(mutated)
def test_mutation_within_run_session_or_cache_sharing(self) -> None:
"""Permitting session/cache sharing within a run must fail the boundary wording."""
base = self._skill_base_text()
mutated = base.replace(
"- Do not share session or cache state within a run across cells, repetitions, or attempts, or across run invocations.",
"- Caller sessions and caches are shared within a run across cells.",
)
self.assertNotEqual(mutated, base, "mutation fixture did not apply")
with self.assertRaises(AssertionError):
self._assert_full_skill_contract(mutated)
def test_mutation_additive_cache_sharing_contradiction(self) -> None:
"""An additive cache-sharing instruction must fail the boundary contract."""
base = self._skill_base_text()
mutated = base.replace(
"- Do not share session or cache state within a run across cells, repetitions, or attempts, or across run invocations.",
"- Do not share session or cache state within a run across cells, repetitions, or attempts, or across run invocations.\n"
"- Do not prevent sharing cache within a run across cells when convenient.",
)
self.assertNotEqual(mutated, base, "mutation fixture did not apply")
with self.assertRaises(AssertionError):
self._assert_full_skill_contract(mutated)
def test_mutation_weakened_per_run_wording(self) -> None:
"""Re-introducing the weaker per-run wording must fail the per-attempt boundary check."""
base = self._skill_base_text()
mutated = base.replace(
"- Caller sessions, output workspaces, and caches are fresh and isolated for every cell, repetition, and attempt; session or cache state is never shared within a run or across runs.",
"- Caller sessions and caches are fresh, isolated per run, and never shared across runs.",
)
self.assertNotEqual(mutated, base, "mutation fixture did not apply")
with self.assertRaises(AssertionError):
self._assert_full_skill_contract(mutated)
def test_mutation_removes_run_execution_contract(self) -> None:
"""Removing the ready execution branch must fail the capability check."""
base = self._skill_base_text()
mutated = base.replace(
"invoke each eligible cell exactly once",
"skip each eligible cell",
)
self.assertNotEqual(mutated, base, "mutation fixture did not apply")
with self.assertRaises(AssertionError):
self._assert_full_skill_contract(mutated)
def test_mutation_missing_public_preflight_delegation(self) -> None:
base = self._skill_base_text()
mutated = base.replace(
"python3 scripts/agent_comparison_benchmark.py preflight --manifest <manifest-path>",
"python3 scripts/agent_comparison_benchmark.py validate --manifest <manifest-path>",
)
self.assertNotEqual(mutated, base, "mutation fixture did not apply")
with self.assertRaises(AssertionError):
self._assert_full_skill_contract(mutated)
def test_mutation_claims_preset_live_readiness(self) -> None:
base = self._skill_base_text()
mutated = base.replace(
"Generic preset cells are local contract validation only.",
"Generic preset cells are live readiness evidence.",
)
self.assertNotEqual(mutated, base, "mutation fixture did not apply")
with self.assertRaises(AssertionError):
self._assert_full_skill_contract(mutated)
def test_mutation_allows_binding_substitution(self) -> None:
base = self._skill_base_text()
mutated = base.replace(
"Never bypass the blocker, substitute a route/model/effort",
"Bypass the blocker and substitute a route/model/effort",
)
self.assertNotEqual(mutated, base, "mutation fixture did not apply")
with self.assertRaises(AssertionError):
self._assert_full_skill_contract(mutated)
def test_mutation_allows_blocker_attempt_allocation(self) -> None:
base = self._skill_base_text()
mutated = base.replace(
"A preflight blocker created no scored attempt and was not bypassed.",
"A preflight blocker may allocate a scored attempt.",
)
self.assertNotEqual(mutated, base, "mutation fixture did not apply")
with self.assertRaises(AssertionError):
self._assert_full_skill_contract(mutated)
def test_mutation_adds_raw_secret_output(self) -> None:
base = self._skill_base_text()
marker = " - On exit 0, report the exact closed `ready` summary from stdout."
mutated = base.replace(marker, marker + "\n - Print the raw secret output.")
self.assertNotEqual(mutated, base, "mutation fixture did not apply")
with self.assertRaises(AssertionError):
self._assert_full_skill_contract(mutated)
if __name__ == "__main__":
unittest.main()