3858 lines
146 KiB
Python
3858 lines
146 KiB
Python
#!/usr/bin/env python3
|
|
"""Dispatch every independently-ready agent-task pair until the task group completes."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import asyncio
|
|
import fcntl
|
|
import hashlib
|
|
import json
|
|
import os
|
|
import re
|
|
import signal
|
|
import shutil
|
|
import subprocess
|
|
import sys
|
|
import uuid
|
|
from dataclasses import dataclass, field
|
|
from datetime import datetime, timedelta, timezone
|
|
from pathlib import Path
|
|
from typing import Any
|
|
|
|
|
|
SEP = "-" * 42
|
|
PLAN_RE = re.compile(r"^PLAN-(local|cloud)-G(0[1-9]|10)\.md$")
|
|
REVIEW_RE = re.compile(r"^CODE_REVIEW-(local|cloud)-G(0[1-9]|10)\.md$")
|
|
SUBTASK_RE = re.compile(r"^(?P<index>\d{2})(?:\+(?P<deps>\d{2}(?:,\d{2})*))?_[a-z0-9_]+$")
|
|
VERDICT_HEADING_RE = re.compile(r"^## 코드리뷰 결과[ \t]*$", re.MULTILINE)
|
|
VERDICT_LINE_RE = re.compile(
|
|
r"^(?:-\s*)?(?:\*\*)?종합 판정(?:\*\*)?\s*:\s*(PASS|WARN|FAIL)[ \t]*$",
|
|
re.MULTILINE,
|
|
)
|
|
VERDICT_BLOCK_RE = re.compile(
|
|
r"^###\s+종합 판정[ \t]*$\s*^(?:\*\*)?(PASS|WARN|FAIL)(?:\*\*)?[ \t]*$",
|
|
re.MULTILINE,
|
|
)
|
|
PLAN_IDENTITY_RE = re.compile(
|
|
r"<!--\s+task=(?P<task>\S+)\s+plan=(?P<plan>\d+)\s+tag=(?P<tag>\S+)\s+-->"
|
|
)
|
|
FINAL_CHECK_RE = re.compile(r"^-\s+\[[xX]\]\s+CODE_REVIEW-\*-G\?\?\.md", re.MULTILINE)
|
|
WORK_LOG_NAME = "WORK_LOG.md"
|
|
WORK_LOG_ARCHIVE_RE = re.compile(r"^work_log_(\d+)\.log$")
|
|
AGENT_PROCESS_MARKER_ENV = "IOP_AGENT_TASK_EXECUTION_ID"
|
|
KST = timezone(timedelta(hours=9), name="KST")
|
|
STREAM_HEARTBEAT_SECONDS = 30
|
|
PI_MODEL_RESPONSE_STALL_SECONDS = 3 * 60
|
|
PI_SESSION_SCHEMA_VERSION = 3
|
|
RECOVERY_FAILURE_LIMIT = 10
|
|
SELF_CHECK_INCOMPLETE_LIMIT = 10
|
|
REVIEW_NO_PROGRESS_LIMIT = 10
|
|
PROVIDER_TRANSPORT_FAILURES = frozenset(
|
|
{"provider-connection", "provider-stream-disconnect"}
|
|
)
|
|
FAILURE_EVIDENCE_LIMIT = 2000
|
|
# Used only to reject a stale locator whose dispatcher and agent PIDs are both
|
|
# gone. A live process is inspected after silence; it is never killed solely by
|
|
# this fallback clock.
|
|
CODEX_STREAM_STALL_SECONDS = 5 * 60
|
|
PROMOTABLE_PATTERNS = {
|
|
"context-limit": [
|
|
r"context (?:length|window)", r"maximum context", r"prompt is too long",
|
|
r"too many tokens", r"token limit", r"exceeded.{0,40}token",
|
|
r"output (?:token )?limit", r"maximum output", r"\bmax_tokens\b",
|
|
r"response (?:is )?too long",
|
|
],
|
|
"provider-quota": [
|
|
r"rate.?limit", r"\bquota\b", r"resource_exhausted", r"\b429\b",
|
|
r"usage limit", r"capacity limit",
|
|
],
|
|
"model-unavailable": [
|
|
r"model.{0,40}(?:not found|unavailable)", r"overloaded",
|
|
r"temporarily unavailable",
|
|
],
|
|
"provider-connection": [
|
|
r"\bprovider[_ -]?tunnel[_ -]?error\b",
|
|
(
|
|
r"(?:provider|backend|/v1/chat/completions|/v1/responses)"
|
|
r".{0,160}(?:connection refused|dial tcp)"
|
|
),
|
|
],
|
|
"provider-stream-disconnect": [
|
|
r"backend connection failed during streaming request",
|
|
r"sse stream before done",
|
|
r"llama-server was unresponsive",
|
|
r"backend watchdog",
|
|
r"model will be reloaded automatically on retry",
|
|
(
|
|
r"(?:provider|backend|sse).{0,160}"
|
|
r"curl error: failure when receiving data from the peer"
|
|
),
|
|
],
|
|
}
|
|
|
|
|
|
class DispatcherAlreadyRunning(RuntimeError):
|
|
"""A live dispatcher owns the workspace; this is non-terminal tracking state."""
|
|
|
|
|
|
class DispatcherTerminalStateError(RuntimeError):
|
|
"""Persistent workspace state prevents safe dispatch before work can start."""
|
|
|
|
|
|
class DispatcherInterruptedWithActiveWork(RuntimeError):
|
|
"""A control-plane error occurred after one or more agent tasks had started."""
|
|
|
|
|
|
def now_iso() -> str:
|
|
return datetime.now(timezone.utc).isoformat()
|
|
|
|
|
|
def work_log_now_iso() -> str:
|
|
return datetime.now(KST).isoformat()
|
|
|
|
|
|
def banner(event: str, task: str, lines: list[str] | None = None) -> None:
|
|
display_task = task.rsplit("/", 1)[-1]
|
|
print(SEP, flush=True)
|
|
print(f"{event}: {display_task}", flush=True)
|
|
print(SEP, flush=True)
|
|
if display_task != task:
|
|
print(f"task={task}", flush=True)
|
|
for line in lines or []:
|
|
print(line, flush=True)
|
|
|
|
|
|
def sha256_file(path: Path | None) -> str:
|
|
if path is None or not path.exists():
|
|
return "none"
|
|
digest = hashlib.sha256()
|
|
with path.open("rb") as stream:
|
|
for chunk in iter(lambda: stream.read(65536), b""):
|
|
digest.update(chunk)
|
|
return digest.hexdigest()
|
|
|
|
|
|
DISPATCHER_SOURCE_PATH = Path(__file__).resolve()
|
|
DISPATCHER_SOURCE_SHA256 = sha256_file(DISPATCHER_SOURCE_PATH)
|
|
DISPATCHER_PROCESS_STARTED_AT = now_iso()
|
|
|
|
|
|
def dispatcher_source_provenance() -> dict[str, Any]:
|
|
current_sha256 = sha256_file(DISPATCHER_SOURCE_PATH)
|
|
return {
|
|
"dispatcher_pid": os.getpid(),
|
|
"dispatcher_process_start_token": process_start_token(os.getpid()),
|
|
"dispatcher_process_started_at": DISPATCHER_PROCESS_STARTED_AT,
|
|
"dispatcher_source_path": str(DISPATCHER_SOURCE_PATH),
|
|
"dispatcher_source_sha256": DISPATCHER_SOURCE_SHA256,
|
|
"dispatcher_source_current_sha256": current_sha256,
|
|
"dispatcher_source_matches_loaded": current_sha256 == DISPATCHER_SOURCE_SHA256,
|
|
}
|
|
|
|
|
|
def plan_identity(path: Path | None) -> str:
|
|
if path is None or not path.exists():
|
|
return "none"
|
|
text = path.read_text(encoding="utf-8", errors="replace")[:1024]
|
|
match = PLAN_IDENTITY_RE.search(text)
|
|
if not match:
|
|
return sha256_file(path)
|
|
identity = "\0".join(match.group(name) for name in ("task", "plan", "tag"))
|
|
return "meta:" + hashlib.sha256(identity.encode()).hexdigest()
|
|
|
|
|
|
def write_json(path: Path, value: dict[str, Any]) -> None:
|
|
path.parent.mkdir(parents=True, exist_ok=True)
|
|
temporary = path.with_suffix(path.suffix + ".tmp")
|
|
temporary.write_text(json.dumps(value, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
|
|
temporary.replace(path)
|
|
|
|
|
|
def milestone_work_log_path(task: Task) -> Path:
|
|
return (
|
|
task.directory.parent / WORK_LOG_NAME
|
|
if "/" in task.name
|
|
else task.directory / WORK_LOG_NAME
|
|
)
|
|
|
|
|
|
def append_work_log_event(
|
|
path: Path,
|
|
*,
|
|
task_name: str,
|
|
event: str,
|
|
execution_id: str,
|
|
role: str,
|
|
attempt: int,
|
|
model: str,
|
|
result: str,
|
|
locator: Path,
|
|
) -> Path:
|
|
path.parent.mkdir(parents=True, exist_ok=True)
|
|
with path.open("a+", encoding="utf-8") as stream:
|
|
fcntl.flock(stream.fileno(), fcntl.LOCK_EX)
|
|
try:
|
|
stream.seek(0)
|
|
text = stream.read()
|
|
if not text:
|
|
stream.write(
|
|
"# Milestone Work Log\n\n"
|
|
"> Dispatcher-owned execution timeline. Workers and reviewers do not edit this file.\n\n"
|
|
"| seq | time | event | task | role | attempt | model | result | locator |\n"
|
|
"|---:|---|---|---|---|---:|---|---|---|\n"
|
|
)
|
|
sequence = 1
|
|
else:
|
|
stream.seek(0, os.SEEK_END)
|
|
if "| seq | time | event | task | role | attempt | model | result | locator |" not in text:
|
|
if not text.endswith("\n"):
|
|
stream.write("\n")
|
|
stream.write(
|
|
"\n## Dispatcher Timeline\n\n"
|
|
"> Dispatcher-owned. Workers and reviewers do not edit this section.\n\n"
|
|
"| seq | time | event | task | role | attempt | model | result | locator |\n"
|
|
"|---:|---|---|---|---|---:|---|---|---|\n"
|
|
)
|
|
sequence = 1 + max(
|
|
(
|
|
int(match.group(1))
|
|
for match in re.finditer(r"^\|\s*(\d+)\s*\|", text, re.MULTILINE)
|
|
),
|
|
default=0,
|
|
)
|
|
if not text.endswith("\n"):
|
|
stream.write("\n")
|
|
|
|
def cell(value: Any) -> str:
|
|
return str(value).replace("|", r"\|").replace("\n", " ")
|
|
|
|
stream.write(
|
|
f"| {sequence} | {work_log_now_iso()} | {cell(event)} | "
|
|
f"{cell(task_name)} | "
|
|
f"{cell(role)} | {attempt} | {cell(model)} | {cell(result)} | "
|
|
f"{cell(locator.resolve())} |\n"
|
|
)
|
|
stream.flush()
|
|
finally:
|
|
fcntl.flock(stream.fileno(), fcntl.LOCK_UN)
|
|
return path
|
|
|
|
|
|
def append_milestone_event(
|
|
task: Task,
|
|
*,
|
|
event: str,
|
|
execution_id: str,
|
|
role: str,
|
|
attempt: int,
|
|
model: str,
|
|
result: str,
|
|
locator: Path,
|
|
) -> Path:
|
|
return append_work_log_event(
|
|
milestone_work_log_path(task),
|
|
task_name=task.name,
|
|
event=event,
|
|
execution_id=execution_id,
|
|
role=role,
|
|
attempt=attempt,
|
|
model=model,
|
|
result=result,
|
|
locator=locator,
|
|
)
|
|
|
|
|
|
def safe_name(value: str) -> str:
|
|
return re.sub(r"[^A-Za-z0-9_.-]+", "__", value).strip("_") or "task"
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class AgentSpec:
|
|
cli: str
|
|
model: str
|
|
display: str
|
|
local_pi: bool = False
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class PiSessionState:
|
|
phase: str
|
|
expected_tool_call_ids: tuple[str, ...] = ()
|
|
completed_tool_call_ids: tuple[str, ...] = ()
|
|
pending_tool_call_ids: tuple[str, ...] = ()
|
|
reason: str = ""
|
|
|
|
|
|
@dataclass
|
|
class Task:
|
|
name: str
|
|
directory: Path
|
|
plan: Path | None
|
|
review: Path | None
|
|
user_review: Path | None
|
|
recovery: bool
|
|
errors: list[str] = field(default_factory=list)
|
|
index: int = 0
|
|
deps: tuple[str, ...] = ()
|
|
write_set: set[str] = field(default_factory=set)
|
|
write_set_known: bool = False
|
|
plan_hash: str = "none"
|
|
lane: str | None = None
|
|
grade: int | None = None
|
|
|
|
|
|
def next_execution_identity(
|
|
store: StateStore,
|
|
task: Task,
|
|
role: str,
|
|
) -> tuple[int, str]:
|
|
attempt = store.next_attempt(task, role)
|
|
identity = (
|
|
f"{safe_name(task.name)}__p{plan_number(task)}__{role}__a{attempt:02d}"
|
|
)
|
|
return attempt, identity
|
|
|
|
|
|
class StateStore:
|
|
def __init__(self, workspace: Path):
|
|
workspace_id = hashlib.sha256(str(workspace).encode()).hexdigest()[:16]
|
|
git_marker = workspace / ".git"
|
|
git_directory: Path | None = None
|
|
if git_marker.is_dir():
|
|
git_directory = git_marker
|
|
elif git_marker.is_file():
|
|
marker = git_marker.read_text(encoding="utf-8", errors="replace").strip()
|
|
if marker.startswith("gitdir:"):
|
|
candidate = Path(marker.split(":", 1)[1].strip())
|
|
git_directory = candidate if candidate.is_absolute() else (workspace / candidate).resolve()
|
|
candidates = []
|
|
if git_directory is not None:
|
|
candidates.append(git_directory / "agent-task-dispatcher")
|
|
state_base = Path(os.environ.get("XDG_STATE_HOME", str(Path.home() / ".local" / "state")))
|
|
candidates.append(state_base / "agent-task-dispatcher" / workspace_id)
|
|
self.root = candidates[-1]
|
|
last_error: OSError | None = None
|
|
for candidate in candidates:
|
|
try:
|
|
candidate.mkdir(parents=True, exist_ok=True)
|
|
self.root = candidate
|
|
last_error = None
|
|
break
|
|
except OSError as exc:
|
|
last_error = exc
|
|
if last_error is not None:
|
|
raise DispatcherTerminalStateError(
|
|
f"dispatcher state 디렉터리를 만들 수 없다: {candidates}"
|
|
) from last_error
|
|
self.path = self.root / "state.json"
|
|
self.runs = self.root / "runs"
|
|
self.runs.mkdir(exist_ok=True)
|
|
self.lock_stream = (self.root / "dispatcher.lock").open("a+", encoding="utf-8")
|
|
try:
|
|
fcntl.flock(self.lock_stream.fileno(), fcntl.LOCK_EX | fcntl.LOCK_NB)
|
|
except BlockingIOError as exc:
|
|
self.lock_stream.seek(0)
|
|
owner = self.lock_stream.read().strip() or "owner metadata unavailable"
|
|
self.lock_stream.close()
|
|
raise DispatcherAlreadyRunning(
|
|
f"같은 workspace의 dispatcher가 이미 실행 중이다: "
|
|
f"{self.root}; owner={owner}"
|
|
) from exc
|
|
try:
|
|
self.lock_stream.seek(0)
|
|
self.lock_stream.truncate()
|
|
self.lock_stream.write(
|
|
json.dumps(dispatcher_source_provenance(), ensure_ascii=False) + "\n"
|
|
)
|
|
self.lock_stream.flush()
|
|
except OSError as exc:
|
|
self.lock_stream.close()
|
|
raise DispatcherTerminalStateError(
|
|
f"dispatcher lock owner metadata를 기록할 수 없다: {self.root}"
|
|
) from exc
|
|
if self.path.exists():
|
|
try:
|
|
self.data = json.loads(self.path.read_text(encoding="utf-8"))
|
|
except (json.JSONDecodeError, OSError) as exc:
|
|
self.lock_stream.close()
|
|
raise DispatcherTerminalStateError(
|
|
f"dispatcher state를 읽을 수 없다: {self.path}"
|
|
) from exc
|
|
if not isinstance(self.data, dict):
|
|
self.lock_stream.close()
|
|
raise DispatcherTerminalStateError(
|
|
f"dispatcher state가 object가 아니다: {self.path}"
|
|
)
|
|
else:
|
|
self.data = {"tasks": {}, "attempt_counters": {}}
|
|
|
|
def save(self) -> None:
|
|
write_json(self.path, self.data)
|
|
|
|
def close(self) -> None:
|
|
if not self.lock_stream.closed:
|
|
self.lock_stream.close()
|
|
|
|
def task_state(self, task: Task) -> dict[str, Any]:
|
|
tasks = self.data.setdefault("tasks", {})
|
|
current = tasks.get(task.name)
|
|
if not current or current.get("plan_hash") != task.plan_hash:
|
|
current = {
|
|
"plan_hash": task.plan_hash,
|
|
"worker_done": False,
|
|
"worker_cli": None,
|
|
"worker_model": None,
|
|
"selfcheck_done": False,
|
|
"blocked": None,
|
|
"active_stage": None,
|
|
"active_locator": None,
|
|
"review_no_progress": 0,
|
|
"selfcheck_incomplete": 0,
|
|
"recovery_failures": {},
|
|
}
|
|
tasks[task.name] = current
|
|
self.save()
|
|
return current
|
|
|
|
def peek_task_state(self, task: Task) -> dict[str, Any]:
|
|
current = self.data.get("tasks", {}).get(task.name)
|
|
if current and current.get("plan_hash") == task.plan_hash:
|
|
return dict(current)
|
|
return {
|
|
"plan_hash": task.plan_hash,
|
|
"worker_done": False,
|
|
"worker_cli": None,
|
|
"worker_model": None,
|
|
"selfcheck_done": False,
|
|
"blocked": None,
|
|
"active_stage": None,
|
|
"active_locator": None,
|
|
"review_no_progress": 0,
|
|
"selfcheck_incomplete": 0,
|
|
"recovery_failures": {},
|
|
}
|
|
|
|
def update_task(self, task: Task, **values: Any) -> None:
|
|
state = self.task_state(task)
|
|
state.update(values)
|
|
self.save()
|
|
|
|
def mark_active(self, task: Task, stage: str, locator: Path | None = None) -> None:
|
|
self.update_task(
|
|
task,
|
|
active_stage=stage,
|
|
active_locator=str(locator) if locator else None,
|
|
active_started_at=now_iso(),
|
|
)
|
|
|
|
def clear_active(self, task: Task) -> None:
|
|
self.update_task(
|
|
task,
|
|
active_stage=None,
|
|
active_locator=None,
|
|
active_started_at=None,
|
|
)
|
|
|
|
def next_attempt(self, task: Task, role: str) -> int:
|
|
key = f"{task.name}|{task.plan_hash}|{role}"
|
|
counters = self.data.setdefault("attempt_counters", {})
|
|
number = int(counters.get(key, 0))
|
|
counters[key] = number + 1
|
|
self.save()
|
|
return number
|
|
|
|
def clear_blocked(self, task_group: str | None = None) -> None:
|
|
prefix = f"{task_group}/" if task_group else None
|
|
for task_name, value in self.data.get("tasks", {}).items():
|
|
if (
|
|
task_group is not None
|
|
and task_name != task_group
|
|
and not task_name.startswith(prefix)
|
|
):
|
|
continue
|
|
value["blocked"] = None
|
|
value["review_no_progress"] = 0
|
|
value["selfcheck_incomplete"] = 0
|
|
value["recovery_failures"] = {}
|
|
self.save()
|
|
|
|
def prepare_orchestration(
|
|
self,
|
|
scope: str,
|
|
tasks: list[Task],
|
|
workspace: Path,
|
|
) -> None:
|
|
orchestrations = self.data.setdefault("orchestrations", {})
|
|
current = orchestrations.get(scope)
|
|
if current is None or (current.get("status") == "complete" and tasks):
|
|
current = {"status": "running", "tasks": {}}
|
|
orchestrations[scope] = current
|
|
changed = False
|
|
tracked = current.setdefault("tasks", {})
|
|
for task in tasks:
|
|
record = tracked.get(task.name)
|
|
if record is None:
|
|
tracked[task.name] = {
|
|
"status": "active",
|
|
"archive": None,
|
|
"archive_baseline": [
|
|
str(path.resolve())
|
|
for path in matching_archive_directories_by_name(
|
|
workspace,
|
|
task.name,
|
|
require_complete=False,
|
|
)
|
|
],
|
|
}
|
|
changed = True
|
|
continue
|
|
if record.get("status") != "complete" and (
|
|
record.get("status") != "active" or "reason" in record
|
|
):
|
|
record["status"] = "active"
|
|
record.pop("reason", None)
|
|
changed = True
|
|
if changed or current.get("status") != "running":
|
|
current["status"] = "running"
|
|
self.save()
|
|
|
|
def mark_orchestration_task_complete(
|
|
self,
|
|
scope: str,
|
|
task_name: str,
|
|
archive: str | Path,
|
|
) -> None:
|
|
archive_path = Path(archive).resolve()
|
|
if not archive_path.is_dir() or not (archive_path / "complete.log").is_file():
|
|
raise RuntimeError(
|
|
f"완료 archive에 complete.log가 없다: task={task_name} archive={archive_path}"
|
|
)
|
|
current = self.data.setdefault("orchestrations", {}).setdefault(
|
|
scope, {"status": "running", "tasks": {}}
|
|
)
|
|
tracked = current.setdefault("tasks", {})
|
|
record = tracked.setdefault(
|
|
task_name,
|
|
{"status": "active", "archive": None, "archive_baseline": []},
|
|
)
|
|
record.update(status="complete", archive=str(archive_path))
|
|
record.pop("reason", None)
|
|
self.save()
|
|
cleanup_completed_task_attempt_logs(self.runs, task_name)
|
|
|
|
def mark_orchestration_blocked(
|
|
self,
|
|
scope: str,
|
|
outcomes: dict[str, tuple[str, str]],
|
|
) -> None:
|
|
current = self.data.setdefault("orchestrations", {}).setdefault(
|
|
scope, {"status": "running", "tasks": {}}
|
|
)
|
|
current["status"] = "blocked"
|
|
tracked = current.setdefault("tasks", {})
|
|
for task_name, (status, reason) in outcomes.items():
|
|
record = tracked.setdefault(
|
|
task_name,
|
|
{
|
|
"status": "active",
|
|
"archive": None,
|
|
"archive_baseline": [],
|
|
},
|
|
)
|
|
if record.get("status") == "complete":
|
|
continue
|
|
record.update(status=status, reason=reason)
|
|
self.save()
|
|
|
|
def reconcile_orchestration(
|
|
self,
|
|
scope: str,
|
|
workspace: Path,
|
|
active_or_running: set[str],
|
|
) -> tuple[dict[str, str], dict[str, str]]:
|
|
current = self.data.setdefault("orchestrations", {}).setdefault(
|
|
scope, {"status": "running", "tasks": {}}
|
|
)
|
|
completed: dict[str, str] = {}
|
|
errors: dict[str, str] = {}
|
|
changed = False
|
|
for task_name, record in current.setdefault("tasks", {}).items():
|
|
if record.get("status") == "complete":
|
|
archive = str(record.get("archive") or "")
|
|
if archive and (Path(archive) / "complete.log").is_file():
|
|
completed[task_name] = archive
|
|
else:
|
|
errors[task_name] = "persisted complete archive가 유효하지 않다"
|
|
continue
|
|
if task_name in active_or_running:
|
|
continue
|
|
baseline = set(str(path) for path in record.get("archive_baseline", []))
|
|
candidates = [
|
|
path
|
|
for path in matching_archive_directories_by_name(workspace, task_name)
|
|
if str(path.resolve()) not in baseline
|
|
]
|
|
if len(candidates) == 1:
|
|
archive = str(candidates[0].resolve())
|
|
record.update(status="complete", archive=archive)
|
|
completed[task_name] = archive
|
|
changed = True
|
|
elif not candidates:
|
|
errors[task_name] = (
|
|
"관찰된 task가 active와 새 complete.log archive 모두에서 사라졌다"
|
|
)
|
|
else:
|
|
errors[task_name] = (
|
|
"새 complete.log archive가 여러 개라 완료 경로를 확정할 수 없다: "
|
|
+ ",".join(str(path) for path in candidates)
|
|
)
|
|
if changed:
|
|
self.save()
|
|
for task_name in completed:
|
|
if task_name not in active_or_running:
|
|
cleanup_completed_task_attempt_logs(self.runs, task_name)
|
|
return completed, errors
|
|
|
|
def orchestration_tasks(self, scope: str) -> set[str]:
|
|
current = self.data.get("orchestrations", {}).get(scope, {})
|
|
return set(current.get("tasks", {}))
|
|
|
|
def mark_orchestration_complete(self, scope: str) -> None:
|
|
current = self.data.setdefault("orchestrations", {}).setdefault(
|
|
scope, {"status": "running", "tasks": {}}
|
|
)
|
|
current["status"] = "complete"
|
|
self.save()
|
|
|
|
|
|
def orchestration_live_agent_processes(
|
|
store: StateStore,
|
|
scope: str,
|
|
) -> dict[str, str]:
|
|
"""Return observed tasks with live or conservatively active evidence."""
|
|
task_states = store.data.get("tasks", {})
|
|
live: dict[str, str] = {}
|
|
for task_name in store.orchestration_tasks(scope):
|
|
state = task_states.get(task_name)
|
|
if not isinstance(state, dict):
|
|
continue
|
|
is_live, detail = external_active_is_live(state)
|
|
if is_live:
|
|
live[task_name] = detail
|
|
return live
|
|
|
|
|
|
def parse_route(plan: Path | None) -> tuple[str | None, int | None]:
|
|
if plan is None:
|
|
return None, None
|
|
match = PLAN_RE.match(plan.name)
|
|
if not match:
|
|
return None, None
|
|
return match.group(1), int(match.group(2))
|
|
|
|
|
|
def parse_task_name(task_root: Path, directory: Path) -> str:
|
|
return directory.relative_to(task_root).as_posix()
|
|
|
|
|
|
def extract_write_set(plan: Path | None, workspace: Path) -> tuple[set[str], bool]:
|
|
if plan is None or not plan.exists():
|
|
return set(), False
|
|
text = plan.read_text(encoding="utf-8", errors="replace")
|
|
match = re.search(r"^## 수정 파일 요약\s*$([\s\S]*?)(?=^##\s|\Z)", text, re.MULTILINE)
|
|
if not match:
|
|
return set(), False
|
|
result: set[str] = set()
|
|
invalid = False
|
|
for line in match.group(1).splitlines():
|
|
if not line.lstrip().startswith("|"):
|
|
continue
|
|
cells = [cell.strip() for cell in line.strip().strip("|").split("|")]
|
|
if not cells:
|
|
continue
|
|
for value in re.findall(r"`([^`]+)`", cells[0]):
|
|
normalized = re.sub(r":\d+(?::\d+)?$", "", value.strip())
|
|
if normalized and not normalized.startswith(("http://", "https://")):
|
|
if any(character in normalized for character in "*?[]"):
|
|
invalid = True
|
|
continue
|
|
candidate = Path(normalized)
|
|
resolved = (
|
|
candidate.resolve()
|
|
if candidate.is_absolute()
|
|
else (workspace / candidate).resolve()
|
|
)
|
|
try:
|
|
resolved.relative_to(workspace)
|
|
except ValueError:
|
|
invalid = True
|
|
continue
|
|
if resolved == workspace or resolved.is_dir():
|
|
invalid = True
|
|
continue
|
|
result.add(str(resolved))
|
|
return result, bool(result) and not invalid
|
|
|
|
|
|
def latest_verdict_log(directory: Path) -> Path | None:
|
|
logs = sorted(directory.glob("code_review_*.log"), key=lambda p: p.stat().st_mtime)
|
|
if not logs:
|
|
return None
|
|
newest = logs[-1]
|
|
return newest if read_verdict(newest) else None
|
|
|
|
|
|
def matching_plan_log(directory: Path, review_log: Path | None) -> Path | None:
|
|
if review_log is None:
|
|
return None
|
|
review_identity = plan_identity(review_log)
|
|
matches = [
|
|
path
|
|
for path in directory.glob("plan_*.log")
|
|
if plan_identity(path) == review_identity
|
|
]
|
|
return max(matches, key=lambda path: path.stat().st_mtime_ns) if matches else None
|
|
|
|
|
|
def read_task_directory(workspace: Path, directory: Path) -> Task | None:
|
|
"""Read one already-known task directory without scanning the task group."""
|
|
task_root = workspace / "agent-task"
|
|
if not directory.is_dir():
|
|
return None
|
|
plans = sorted(p for p in directory.iterdir() if p.is_file() and PLAN_RE.match(p.name))
|
|
reviews = sorted(p for p in directory.iterdir() if p.is_file() and REVIEW_RE.match(p.name))
|
|
users = sorted(directory.glob("USER_REVIEW.md"))
|
|
complete = directory / "complete.log"
|
|
recovery_log = latest_verdict_log(directory)
|
|
if not plans and not reviews and not users and not complete.exists() and recovery_log is None:
|
|
return None
|
|
name = parse_task_name(task_root, directory)
|
|
errors: list[str] = []
|
|
if len(plans) > 1:
|
|
errors.append(f"active PLAN이 {len(plans)}개다")
|
|
if len(reviews) > 1:
|
|
errors.append(f"active CODE_REVIEW가 {len(reviews)}개다")
|
|
if len(users) > 1:
|
|
errors.append(f"USER_REVIEW가 {len(users)}개다")
|
|
if users and (plans or reviews):
|
|
errors.append("USER_REVIEW stop state와 active PLAN/CODE_REVIEW가 공존한다")
|
|
plan = plans[0] if len(plans) == 1 else None
|
|
review = reviews[0] if len(reviews) == 1 else None
|
|
recovery = complete.exists() or recovery_log is not None
|
|
if bool(plan) != bool(review) and not recovery:
|
|
errors.append("active PLAN/CODE_REVIEW pair가 불완전하다")
|
|
relative = directory.relative_to(task_root)
|
|
subtask = relative.parts[1] if len(relative.parts) == 2 else None
|
|
index = 0
|
|
deps: tuple[str, ...] = ()
|
|
if subtask:
|
|
match = SUBTASK_RE.match(subtask)
|
|
if match:
|
|
index = int(match.group("index"))
|
|
deps = tuple((match.group("deps") or "").split(",")) if match.group("deps") else ()
|
|
else:
|
|
errors.append(f"split subtask 이름이 계약과 다르다: {subtask}")
|
|
lane, grade = parse_route(plan)
|
|
recovery_plan = matching_plan_log(directory, recovery_log)
|
|
write_set_source = plan or recovery_plan
|
|
write_set, write_set_known = extract_write_set(write_set_source, workspace)
|
|
if plan is not None:
|
|
metadata = PLAN_IDENTITY_RE.search(
|
|
plan.read_text(encoding="utf-8", errors="replace")[:1024]
|
|
)
|
|
if metadata is None:
|
|
errors.append("PLAN task/plan/tag metadata를 판별할 수 없다")
|
|
elif metadata.group("task") != name:
|
|
errors.append(
|
|
f"PLAN task metadata가 디렉터리와 다르다: {metadata.group('task')}"
|
|
)
|
|
return Task(
|
|
name=name,
|
|
directory=directory,
|
|
plan=plan,
|
|
review=review,
|
|
user_review=users[0] if len(users) == 1 else None,
|
|
recovery=recovery,
|
|
errors=errors,
|
|
index=index,
|
|
deps=deps,
|
|
write_set=write_set,
|
|
write_set_known=write_set_known,
|
|
plan_hash=(
|
|
plan_identity(plan)
|
|
if plan
|
|
else sha256_file(
|
|
recovery_log
|
|
or (users[0] if len(users) == 1 else complete)
|
|
)
|
|
),
|
|
lane=lane,
|
|
grade=grade,
|
|
)
|
|
|
|
|
|
def scan_tasks(
|
|
workspace: Path,
|
|
task_group: str | None,
|
|
*,
|
|
exclude_names: set[str] | None = None,
|
|
) -> list[Task]:
|
|
task_root = workspace / "agent-task"
|
|
if not task_root.is_dir():
|
|
raise DispatcherTerminalStateError(
|
|
f"agent-task 디렉터리가 없다: {task_root}"
|
|
)
|
|
directories: list[Path] = []
|
|
try:
|
|
groups = [task_root / task_group] if task_group else sorted(
|
|
p for p in task_root.iterdir() if p.is_dir() and p.name != "archive"
|
|
)
|
|
except FileNotFoundError:
|
|
return []
|
|
for group in groups:
|
|
if not group.is_dir():
|
|
continue
|
|
directories.append(group)
|
|
try:
|
|
directories.extend(sorted(p for p in group.iterdir() if p.is_dir()))
|
|
except FileNotFoundError:
|
|
continue
|
|
tasks = [
|
|
task
|
|
for directory in directories
|
|
if (
|
|
exclude_names is None
|
|
or parse_task_name(task_root, directory) not in exclude_names
|
|
)
|
|
if (task := read_task_directory(workspace, directory)) is not None
|
|
]
|
|
return sorted(tasks, key=lambda task: (task.index, task.name))
|
|
|
|
|
|
def dependency_candidates(workspace: Path, task: Task, predecessor: str) -> list[Path]:
|
|
parts = task.name.split("/")
|
|
if len(parts) != 2:
|
|
return []
|
|
group = parts[0]
|
|
task_root = workspace / "agent-task"
|
|
found: list[Path] = []
|
|
active_group = task_root / group
|
|
for pattern in (f"{predecessor}_*/complete.log", f"{predecessor}+*/complete.log"):
|
|
found.extend(active_group.glob(pattern))
|
|
archive = task_root / "archive"
|
|
if archive.is_dir():
|
|
try:
|
|
years = list(archive.iterdir())
|
|
except FileNotFoundError:
|
|
years = []
|
|
for year in years:
|
|
if not year.is_dir():
|
|
continue
|
|
try:
|
|
months = list(year.iterdir())
|
|
except FileNotFoundError:
|
|
continue
|
|
for month in months:
|
|
archived_group = month / group
|
|
if not archived_group.is_dir():
|
|
continue
|
|
for pattern in (f"{predecessor}_*/complete.log", f"{predecessor}+*/complete.log"):
|
|
found.extend(archived_group.glob(pattern))
|
|
return sorted(set(path.resolve() for path in found))
|
|
|
|
|
|
def dependency_state(workspace: Path, task: Task) -> tuple[bool, str]:
|
|
missing: list[str] = []
|
|
ambiguous: list[str] = []
|
|
for predecessor in task.deps:
|
|
candidates = dependency_candidates(workspace, task, predecessor)
|
|
if not candidates:
|
|
missing.append(predecessor)
|
|
elif len(candidates) > 1:
|
|
ambiguous.append(f"{predecessor}={','.join(str(p) for p in candidates)}")
|
|
if ambiguous:
|
|
return False, "dependency ambiguity: " + "; ".join(ambiguous)
|
|
if missing:
|
|
return False, "predecessor complete.log 대기: " + ",".join(missing)
|
|
return True, "ready"
|
|
|
|
|
|
def live_predecessors(
|
|
task: Task,
|
|
active_task_names: set[str],
|
|
) -> list[str]:
|
|
parts = task.name.split("/")
|
|
if len(parts) != 2 or not task.deps:
|
|
return []
|
|
group = parts[0]
|
|
live: list[str] = []
|
|
for predecessor in task.deps:
|
|
prefix = re.compile(rf"^{re.escape(predecessor)}(?:[+_])")
|
|
if any(
|
|
name.startswith(f"{group}/")
|
|
and prefix.match(name.split("/", 1)[1])
|
|
for name in active_task_names
|
|
):
|
|
live.append(predecessor)
|
|
return live
|
|
|
|
|
|
def route_agent(task: Task) -> AgentSpec:
|
|
if task.lane is None or task.grade is None:
|
|
raise RuntimeError("PLAN route를 판별할 수 없다")
|
|
grade = task.grade
|
|
if task.lane == "local":
|
|
if grade <= 6:
|
|
return AgentSpec("pi", "ornith-fast", "pi/iop/ornith-fast", local_pi=True)
|
|
if grade <= 8:
|
|
return AgentSpec("pi", "laguna-s:2.1", "pi/iop/laguna-s:2.1", local_pi=True)
|
|
return AgentSpec("claude", "claude-opus-4-8", "claude/claude-opus-4-8 xhigh")
|
|
if grade <= 2:
|
|
return AgentSpec("agy", "Gemini 3.5 Flash (Low)", "agy/Gemini 3.5 Flash (Low)")
|
|
if grade <= 4:
|
|
return AgentSpec("agy", "Gemini 3.5 Flash (Medium)", "agy/Gemini 3.5 Flash (Medium)")
|
|
if grade <= 6:
|
|
return AgentSpec("agy", "Gemini 3.5 Flash (High)", "agy/Gemini 3.5 Flash (High)")
|
|
if grade <= 8:
|
|
return AgentSpec("claude", "claude-opus-4-8", "claude/claude-opus-4-8 xhigh")
|
|
return AgentSpec("codex", "gpt-5.6-sol", "codex/gpt-5.6-sol xhigh")
|
|
|
|
|
|
def plan_number(task: Task) -> int:
|
|
if task.plan and task.plan.exists():
|
|
match = PLAN_IDENTITY_RE.search(
|
|
task.plan.read_text(encoding="utf-8", errors="replace")[:1024]
|
|
)
|
|
if match:
|
|
return int(match.group("plan"))
|
|
return 0
|
|
|
|
|
|
def task_requires_selfcheck(task: Task) -> bool:
|
|
return task.lane == "local" and task.grade is not None and task.grade <= 8
|
|
|
|
|
|
def concrete_user_review_value(value: str) -> bool:
|
|
normalized = value.strip().strip("`").strip()
|
|
normalized = re.sub(r"^-\s*", "", normalized).strip()
|
|
if not normalized or re.search(r"\{[^}]+\}|<[^>]+>", normalized):
|
|
return False
|
|
return normalized.casefold() not in {
|
|
"-",
|
|
"n/a",
|
|
"na",
|
|
"none",
|
|
"unknown",
|
|
"미정",
|
|
"없음",
|
|
"해당 없음",
|
|
}
|
|
|
|
|
|
def user_review_blocker_state(path: Path) -> tuple[bool, str]:
|
|
if not path.is_file():
|
|
return False, "파일이 없다"
|
|
try:
|
|
text = path.read_text(encoding="utf-8", errors="replace")
|
|
except OSError as exc:
|
|
return False, f"파일을 읽을 수 없다: {exc}"
|
|
status = markdown_section(text, "상태").strip().strip("`")
|
|
if status != "USER_REVIEW":
|
|
return False, "상태가 USER_REVIEW가 아니다"
|
|
reason = markdown_section(text, "사유")
|
|
if not re.search(r"(?m)^-\s*유형:\s*milestone-lock\s*$", reason):
|
|
return False, "milestone-lock 유형이 아니다"
|
|
target = re.search(r"(?m)^-\s*연결 대상:\s*(.+?)\s*$", reason)
|
|
target_value = target.group(1) if target else ""
|
|
if (
|
|
not concrete_user_review_value(target_value)
|
|
or "agent-roadmap/" not in target_value
|
|
or "/milestones/" not in target_value
|
|
or ".md" not in target_value
|
|
):
|
|
return False, "구체적인 Milestone 연결 대상이 없다"
|
|
evidence = markdown_section(text, "차단 근거")
|
|
evidence_line = re.search(r"(?m)^-\s*차단 판단 근거:\s*(.+?)\s*$", evidence)
|
|
if evidence_line is None or not concrete_user_review_value(
|
|
evidence_line.group(1)
|
|
):
|
|
return False, "구체적인 차단 판단 근거가 없다"
|
|
decision = markdown_section(text, "연결 결정 필요")
|
|
unresolved = [
|
|
value
|
|
for value in re.findall(r"(?m)^-\s*\[\s\]\s+(.+?)\s*$", decision)
|
|
if concrete_user_review_value(value)
|
|
]
|
|
if not unresolved:
|
|
return False, "미해결 연결 결정 항목이 없다"
|
|
resume = markdown_section(text, "재개 조건")
|
|
resume_conditions = [
|
|
line
|
|
for line in resume.splitlines()
|
|
if concrete_user_review_value(line)
|
|
]
|
|
if not resume_conditions:
|
|
return False, "구체적인 재개 조건이 없다"
|
|
return True, "unresolved milestone-lock decision"
|
|
|
|
|
|
def task_stage(task: Task, state: dict[str, Any]) -> str:
|
|
if task.errors:
|
|
return "blocked"
|
|
if task.user_review:
|
|
if task.plan is not None or task.review is not None:
|
|
return "blocked"
|
|
blocking, _ = user_review_blocker_state(task.user_review)
|
|
return "user-review" if blocking else "blocked"
|
|
if task.recovery and (task.plan is None or task.review is None):
|
|
return "review"
|
|
if task.review and task.review.exists():
|
|
text = task.review.read_text(encoding="utf-8", errors="replace")
|
|
if verdict_from_text(text):
|
|
return "review"
|
|
if state.get("worker_done"):
|
|
if task_requires_selfcheck(task) and not state.get("selfcheck_done"):
|
|
return "selfcheck"
|
|
return "review"
|
|
if task.review and task.review.exists() and FINAL_CHECK_RE.search(text):
|
|
return "review"
|
|
return "worker"
|
|
|
|
|
|
def markdown_section(text: str, heading: str) -> str:
|
|
match = re.search(rf"^## {re.escape(heading)}[ \t]*$", text, re.MULTILINE)
|
|
if match is None:
|
|
return ""
|
|
next_heading = re.search(r"^##\s+", text[match.end():], re.MULTILINE)
|
|
end = match.end() + next_heading.start() if next_heading else len(text)
|
|
return text[match.end():end].strip()
|
|
|
|
|
|
def implementation_review_errors(task: Task) -> list[str]:
|
|
if task.review is None or not task.review.is_file():
|
|
return ["CODE_REVIEW 파일 없음"]
|
|
text = task.review.read_text(encoding="utf-8", errors="replace")
|
|
errors: list[str] = []
|
|
checklist = markdown_section(text, "구현 체크리스트")
|
|
if not checklist or re.search(r"^-\s+\[\s\]", checklist, re.MULTILINE):
|
|
errors.append("구현 체크리스트 미완료")
|
|
completion = markdown_section(text, "구현 항목별 완료 여부")
|
|
if not completion or re.search(r"\|\s*\[\s\]\s*\|", completion):
|
|
errors.append("구현 항목 완료 여부 미작성")
|
|
for heading in ("계획 대비 변경 사항", "주요 설계 결정"):
|
|
body = markdown_section(text, heading)
|
|
if not body or re.fullmatch(r"_[^_]*_", body, re.DOTALL):
|
|
errors.append(f"{heading} 미작성")
|
|
verification = markdown_section(text, "검증 결과")
|
|
if not verification or re.search(r"_미실행(?:[^_]*)_", verification):
|
|
errors.append("검증 결과 미작성")
|
|
if not FINAL_CHECK_RE.search(text):
|
|
errors.append("CODE_REVIEW 동기화 체크 미완료")
|
|
return errors
|
|
|
|
|
|
def classify_failure_with_evidence(output: str) -> tuple[str, str | None]:
|
|
lines = output.splitlines()
|
|
for category, patterns in PROMOTABLE_PATTERNS.items():
|
|
for line in reversed(lines):
|
|
lowered = line.lower()
|
|
if any(re.search(pattern, lowered, re.DOTALL) for pattern in patterns):
|
|
return category, line
|
|
return "generic-error", None
|
|
|
|
|
|
def classify_failure(output: str) -> str:
|
|
return classify_failure_with_evidence(output)[0]
|
|
|
|
|
|
def termination_signal(return_code: int) -> tuple[str, bool] | None:
|
|
signal_number: int | None = None
|
|
inferred = False
|
|
if return_code < 0:
|
|
signal_number = -return_code
|
|
elif return_code > 128:
|
|
signal_number = return_code - 128
|
|
inferred = True
|
|
if signal_number is None:
|
|
return None
|
|
try:
|
|
return signal.Signals(signal_number).name, inferred
|
|
except ValueError:
|
|
return None
|
|
|
|
|
|
def failure_report_lines(failure: str, locator: Path) -> list[str]:
|
|
record: dict[str, Any] = {}
|
|
try:
|
|
record = json.loads(locator.read_text(encoding="utf-8"))
|
|
except (OSError, json.JSONDecodeError):
|
|
pass
|
|
failure_class = str(record.get("failure_class") or failure)
|
|
source = str(record.get("failure_source") or "unverified")
|
|
provider_confirmed = bool(
|
|
record.get("provider_transport_failure_confirmed", False)
|
|
)
|
|
lines = [
|
|
f"failure_class={failure_class}",
|
|
f"failure_source={source}",
|
|
"provider_transport_failure_confirmed="
|
|
f"{str(provider_confirmed).lower()}",
|
|
]
|
|
if record.get("dispatcher_pid") is not None:
|
|
lines.append(f"dispatcher_pid={record['dispatcher_pid']}")
|
|
if record.get("agent_pid") is not None:
|
|
lines.append(f"agent_pid={record['agent_pid']}")
|
|
if record.get("dispatcher_source_sha256"):
|
|
lines.append(
|
|
f"dispatcher_source_sha256={record['dispatcher_source_sha256']}"
|
|
)
|
|
source_matches_loaded = record.get("dispatcher_source_matches_loaded")
|
|
if source_matches_loaded is not None:
|
|
lines.append(
|
|
"dispatcher_source_matches_loaded="
|
|
f"{str(bool(source_matches_loaded)).lower()}"
|
|
)
|
|
if (
|
|
source_matches_loaded is False
|
|
and record.get("dispatcher_source_current_sha256")
|
|
):
|
|
lines.append(
|
|
"dispatcher_source_current_sha256="
|
|
f"{record['dispatcher_source_current_sha256']}"
|
|
)
|
|
if provider_confirmed:
|
|
evidence_source = record.get("failure_evidence_source")
|
|
evidence = record.get("failure_evidence_excerpt")
|
|
if evidence_source:
|
|
lines.append(f"provider_evidence_source={evidence_source}")
|
|
if evidence:
|
|
rendered = str(evidence).replace("\r", r"\r").replace("\n", r"\n")
|
|
lines.append(f"provider_evidence={rendered}")
|
|
if failure_class == "session-stall":
|
|
lines.extend(
|
|
[
|
|
f"timeout_phase={record.get('pi_session_phase') or 'unknown'}",
|
|
f"timeout_seconds={record.get('session_stall_seconds') or 'unknown'}",
|
|
"termination_initiator="
|
|
f"{record.get('termination_initiator') or 'dispatcher'}",
|
|
]
|
|
)
|
|
elif failure_class == "process-terminated":
|
|
lines.extend(
|
|
[
|
|
f"termination_signal={record.get('termination_signal') or 'unknown'}",
|
|
"termination_initiator="
|
|
f"{record.get('termination_initiator') or 'unknown'}",
|
|
]
|
|
)
|
|
lines.append(f"locator={locator}")
|
|
return lines
|
|
|
|
|
|
def terminal_diagnostic(cli: str, channel: str, line: str) -> str | None:
|
|
if channel == "stderr":
|
|
return line
|
|
try:
|
|
value = json.loads(line)
|
|
except json.JSONDecodeError:
|
|
if cli == "agy" and re.match(
|
|
r"^\s*(?:error|fatal|provider error|model error)\b", line, re.IGNORECASE
|
|
):
|
|
return line
|
|
return None
|
|
event_type = str(value.get("type", ""))
|
|
if cli == "codex" and event_type in {"turn.failed", "error"}:
|
|
return json.dumps(value.get("error", value), ensure_ascii=False)
|
|
if cli == "claude":
|
|
subtype = str(value.get("subtype", ""))
|
|
if event_type == "result" and (value.get("is_error") or subtype.startswith("error")):
|
|
return str(value.get("result") or value)
|
|
if event_type == "system" and subtype.startswith("error"):
|
|
return json.dumps(value, ensure_ascii=False)
|
|
return None
|
|
|
|
|
|
def codex_collaboration_tool(line: str) -> str | None:
|
|
try:
|
|
value = json.loads(line)
|
|
except json.JSONDecodeError:
|
|
return None
|
|
item = value.get("item") or {}
|
|
if (
|
|
value.get("type") == "item.started"
|
|
and item.get("type") == "collab_tool_call"
|
|
and item.get("tool")
|
|
):
|
|
return str(item["tool"])
|
|
return None
|
|
|
|
|
|
async def terminate_process_group(
|
|
process: asyncio.subprocess.Process,
|
|
grace_seconds: float = 5,
|
|
) -> None:
|
|
"""Terminate the exact subprocess group and escalate if descendants remain."""
|
|
try:
|
|
os.killpg(process.pid, signal.SIGTERM)
|
|
except ProcessLookupError:
|
|
if process.returncode is None:
|
|
await process.wait()
|
|
return
|
|
|
|
if process.returncode is None:
|
|
try:
|
|
await asyncio.wait_for(process.wait(), timeout=grace_seconds)
|
|
except TimeoutError:
|
|
try:
|
|
os.killpg(process.pid, signal.SIGKILL)
|
|
except ProcessLookupError:
|
|
pass
|
|
await process.wait()
|
|
return
|
|
|
|
try:
|
|
os.killpg(process.pid, 0)
|
|
except ProcessLookupError:
|
|
return
|
|
try:
|
|
os.killpg(process.pid, signal.SIGKILL)
|
|
except ProcessLookupError:
|
|
pass
|
|
|
|
|
|
def agy_log_diagnostics(path: Path) -> list[str]:
|
|
if not path.exists():
|
|
return []
|
|
diagnostics: list[str] = []
|
|
for line in path.read_text(encoding="utf-8", errors="replace").splitlines()[-200:]:
|
|
if re.search(r"\b(?:error|fatal)\b", line, re.IGNORECASE):
|
|
diagnostics.append(line)
|
|
elif re.search(r"RESOURCE_EXHAUSTED|rate.?limit|model.{0,40}unavailable", line, re.IGNORECASE):
|
|
diagnostics.append(line)
|
|
return diagnostics
|
|
|
|
|
|
def promoted_spec(spec: AgentSpec, recovery_count: int) -> AgentSpec | None:
|
|
if spec.cli == "agy":
|
|
return AgentSpec("claude", "claude-opus-4-8", "claude/claude-opus-4-8 xhigh")
|
|
if spec.cli == "claude":
|
|
return AgentSpec("codex", "gpt-5.6-sol", "codex/gpt-5.6-sol xhigh")
|
|
if spec.cli == "codex" and recovery_count < 1:
|
|
return AgentSpec("codex", "gpt-5.6-sol", "codex/gpt-5.6-sol xhigh")
|
|
return None
|
|
|
|
|
|
def render_json_line(cli: str, line: str) -> tuple[list[str], str | None]:
|
|
try:
|
|
value = json.loads(line)
|
|
except json.JSONDecodeError:
|
|
return [line.rstrip()], None
|
|
session_id = value.get("thread_id") or value.get("session_id")
|
|
rendered: list[str] = []
|
|
if cli == "codex":
|
|
if value.get("type") == "thread.started" and session_id:
|
|
rendered.append(f"session={session_id}")
|
|
item = value.get("item") or {}
|
|
item_type = item.get("type")
|
|
if item_type == "agent_message" and item.get("text"):
|
|
rendered.extend(str(item["text"]).splitlines())
|
|
elif item_type == "command_execution":
|
|
rendered.append(f"$ {item.get('command', '')} (exit={item.get('exit_code', '?')})")
|
|
elif item_type in {"mcp_tool_call", "web_search"}:
|
|
rendered.append(f"{item_type}: {item.get('server', '')} {item.get('tool', item.get('query', ''))}")
|
|
elif value.get("type") == "turn.failed":
|
|
rendered.append(str(value.get("error", value)))
|
|
elif cli == "claude":
|
|
message = value.get("message") or {}
|
|
for block in message.get("content") or []:
|
|
if block.get("type") == "text":
|
|
rendered.extend(str(block.get("text", "")).splitlines())
|
|
elif block.get("type") == "tool_use":
|
|
rendered.append(f"tool={block.get('name', '')}")
|
|
if value.get("type") == "result" and value.get("result"):
|
|
rendered.extend(str(value["result"]).splitlines())
|
|
session_id = session_id or value.get("session_id")
|
|
return rendered, str(session_id) if session_id else None
|
|
|
|
|
|
def native_session_path(cli: str, workspace: Path, session_id: str | None, attempt_dir: Path) -> str | None:
|
|
if cli == "claude" and session_id:
|
|
encoded = str(workspace).replace("/", "-")
|
|
return str(Path.home() / ".claude" / "projects" / encoded / f"{session_id}.jsonl")
|
|
if cli == "pi" and session_id:
|
|
matches = list((attempt_dir / "pi-sessions").glob(f"*{session_id}*.jsonl"))
|
|
return str(matches[0]) if matches else str(attempt_dir / "pi-sessions")
|
|
if cli == "codex" and session_id:
|
|
matches = list((Path.home() / ".codex" / "sessions").glob(f"**/*{session_id}*.jsonl"))
|
|
return str(matches[0]) if matches else str(Path.home() / ".codex" / "sessions")
|
|
return None
|
|
|
|
|
|
def native_session_mtime_ns(path: str | None) -> int | None:
|
|
if not path:
|
|
return None
|
|
candidate = Path(path)
|
|
return candidate.stat().st_mtime_ns if candidate.is_file() else None
|
|
|
|
|
|
def reverse_jsonl_lines(path: Path):
|
|
with path.open("rb") as stream:
|
|
stream.seek(0, os.SEEK_END)
|
|
position = stream.tell()
|
|
buffer = b""
|
|
while position > 0:
|
|
read_size = min(8192, position)
|
|
position -= read_size
|
|
stream.seek(position)
|
|
buffer = stream.read(read_size) + buffer
|
|
lines = buffer.split(b"\n")
|
|
buffer = lines[0]
|
|
for line in reversed(lines[1:]):
|
|
if line.strip():
|
|
yield line
|
|
if buffer.strip():
|
|
yield buffer
|
|
|
|
|
|
def pi_session_header_version(path: Path) -> int | None:
|
|
with path.open("rb") as stream:
|
|
first_line = stream.readline()
|
|
if not first_line.strip():
|
|
return None
|
|
header = json.loads(first_line)
|
|
if not isinstance(header, dict) or header.get("type") != "session":
|
|
return None
|
|
version = header.get("version")
|
|
return version if isinstance(version, int) else None
|
|
|
|
|
|
def pi_native_session_state(path: str | None) -> PiSessionState:
|
|
if not path:
|
|
return PiSessionState("starting", reason="native-session-path-missing")
|
|
candidate = Path(path)
|
|
if not candidate.is_file():
|
|
return PiSessionState("starting", reason="native-session-file-missing")
|
|
completed_ids: list[str] = []
|
|
expected_entry_id: str | None = None
|
|
active_leaf_found = False
|
|
try:
|
|
version = pi_session_header_version(candidate)
|
|
if version != PI_SESSION_SCHEMA_VERSION:
|
|
return PiSessionState(
|
|
"unknown",
|
|
reason=(
|
|
f"unsupported-session-version:{version}"
|
|
if version is not None
|
|
else "session-header-invalid"
|
|
),
|
|
)
|
|
for raw_line in reverse_jsonl_lines(candidate):
|
|
value = json.loads(raw_line)
|
|
if not isinstance(value, dict):
|
|
return PiSessionState("unknown", reason="invalid-entry-schema")
|
|
if value.get("type") == "session":
|
|
break
|
|
entry_id = value.get("id")
|
|
parent_id = value.get("parentId")
|
|
if (
|
|
not isinstance(entry_id, str)
|
|
or not entry_id
|
|
or "parentId" not in value
|
|
or (parent_id is not None and not isinstance(parent_id, str))
|
|
):
|
|
return PiSessionState("unknown", reason="invalid-entry-identity")
|
|
if active_leaf_found and entry_id != expected_entry_id:
|
|
continue
|
|
active_leaf_found = True
|
|
expected_entry_id = parent_id
|
|
if value.get("type") != "message":
|
|
continue
|
|
message = value.get("message")
|
|
if not isinstance(message, dict):
|
|
return PiSessionState("unknown", reason="invalid-message-schema")
|
|
role = message.get("role")
|
|
if role == "toolResult":
|
|
tool_call_id = message.get("toolCallId")
|
|
if not isinstance(tool_call_id, str) or not tool_call_id:
|
|
return PiSessionState(
|
|
"unknown", reason="tool-result-id-missing"
|
|
)
|
|
if tool_call_id in completed_ids:
|
|
return PiSessionState(
|
|
"unknown", reason="duplicate-tool-result-id"
|
|
)
|
|
completed_ids.append(tool_call_id)
|
|
continue
|
|
if role == "user":
|
|
if completed_ids:
|
|
return PiSessionState(
|
|
"unknown", reason="tool-results-without-assistant"
|
|
)
|
|
return PiSessionState("awaiting-model", reason="user-message")
|
|
if role != "assistant":
|
|
return PiSessionState(
|
|
"unknown", reason=f"unsupported-message-role:{role}"
|
|
)
|
|
|
|
content = message.get("content")
|
|
if not isinstance(content, list):
|
|
return PiSessionState(
|
|
"unknown", reason="assistant-content-not-list"
|
|
)
|
|
if any(
|
|
not isinstance(block, dict)
|
|
or block.get("type") not in {"text", "thinking", "toolCall"}
|
|
for block in content
|
|
):
|
|
return PiSessionState(
|
|
"unknown", reason="unsupported-assistant-content"
|
|
)
|
|
tool_calls = [
|
|
block
|
|
for block in content
|
|
if isinstance(block, dict) and block.get("type") == "toolCall"
|
|
]
|
|
if not tool_calls:
|
|
if completed_ids:
|
|
return PiSessionState(
|
|
"unknown", reason="tool-results-without-tool-calls"
|
|
)
|
|
return PiSessionState("finishing", reason="assistant-final")
|
|
|
|
expected_ids: list[str] = []
|
|
for tool_call in tool_calls:
|
|
tool_call_id = tool_call.get("id")
|
|
if not isinstance(tool_call_id, str) or not tool_call_id:
|
|
return PiSessionState(
|
|
"unknown", reason="tool-call-id-missing"
|
|
)
|
|
if tool_call_id in expected_ids:
|
|
return PiSessionState(
|
|
"unknown", reason="duplicate-tool-call-id"
|
|
)
|
|
expected_ids.append(tool_call_id)
|
|
|
|
unexpected_ids = [
|
|
tool_call_id
|
|
for tool_call_id in completed_ids
|
|
if tool_call_id not in expected_ids
|
|
]
|
|
if unexpected_ids:
|
|
return PiSessionState(
|
|
"unknown", reason="tool-result-id-not-in-latest-batch"
|
|
)
|
|
completed_set = set(completed_ids)
|
|
completed = tuple(
|
|
tool_call_id
|
|
for tool_call_id in expected_ids
|
|
if tool_call_id in completed_set
|
|
)
|
|
pending = tuple(
|
|
tool_call_id
|
|
for tool_call_id in expected_ids
|
|
if tool_call_id not in completed_set
|
|
)
|
|
return PiSessionState(
|
|
"tool-running" if pending else "awaiting-model",
|
|
expected_tool_call_ids=tuple(expected_ids),
|
|
completed_tool_call_ids=completed,
|
|
pending_tool_call_ids=pending,
|
|
reason=(
|
|
"pending-tool-results"
|
|
if pending
|
|
else "all-tool-results-recorded"
|
|
),
|
|
)
|
|
if completed_ids:
|
|
return PiSessionState(
|
|
"unknown", reason="tool-results-without-assistant"
|
|
)
|
|
if active_leaf_found and expected_entry_id is not None:
|
|
return PiSessionState("unknown", reason="active-branch-parent-missing")
|
|
except (OSError, UnicodeDecodeError, json.JSONDecodeError):
|
|
return PiSessionState("unknown", reason="unreadable-jsonl")
|
|
return PiSessionState("starting", reason="no-message-events")
|
|
|
|
|
|
def pi_native_session_phase(path: str | None) -> str:
|
|
return pi_native_session_state(path).phase
|
|
|
|
|
|
def log_tail_excerpt(path: Path, *, byte_limit: int = 8192, char_limit: int = 2000) -> str:
|
|
"""Return a bounded recent log excerpt without loading a long reasoning stream."""
|
|
try:
|
|
with path.open("rb") as stream:
|
|
stream.seek(max(0, path.stat().st_size - byte_limit))
|
|
text = stream.read().decode("utf-8", errors="replace")
|
|
except OSError as exc:
|
|
return f"<stream log unavailable: {exc}>"
|
|
return text[-char_limit:]
|
|
|
|
|
|
def process_start_token(value: Any) -> str | None:
|
|
"""Read Linux process start ticks so PID reuse is not treated as liveness."""
|
|
try:
|
|
pid = int(value)
|
|
text = Path(f"/proc/{pid}/stat").read_text(encoding="utf-8")
|
|
close = text.rfind(")")
|
|
fields = text[close + 2 :].split()
|
|
return fields[19] if close >= 0 and len(fields) > 19 else None
|
|
except (TypeError, ValueError, OSError):
|
|
return None
|
|
|
|
|
|
def process_is_alive(value: Any, expected_start_token: Any = None) -> bool:
|
|
"""Return whether the same attempt/dispatcher process still exists."""
|
|
try:
|
|
pid = int(value)
|
|
if pid <= 0:
|
|
return False
|
|
os.kill(pid, 0)
|
|
except (TypeError, ValueError, OSError):
|
|
return False
|
|
current_token = process_start_token(pid)
|
|
if (
|
|
expected_start_token is not None
|
|
and current_token is not None
|
|
and str(expected_start_token) != current_token
|
|
):
|
|
return False
|
|
return True
|
|
|
|
|
|
def marked_agent_process_pids(marker: str) -> list[int]:
|
|
"""Find live processes carrying the per-attempt environment marker."""
|
|
expected = f"{AGENT_PROCESS_MARKER_ENV}={marker}".encode()
|
|
matches: list[int] = []
|
|
for environ in Path("/proc").glob("[0-9]*/environ"):
|
|
try:
|
|
values = environ.read_bytes().split(b"\0")
|
|
pid = int(environ.parent.name)
|
|
except (OSError, ValueError):
|
|
continue
|
|
if expected in values:
|
|
matches.append(pid)
|
|
return sorted(matches)
|
|
|
|
|
|
def external_active_is_live(state: dict[str, Any]) -> tuple[bool, str]:
|
|
raw_locator = state.get("active_locator")
|
|
if not raw_locator:
|
|
return False, "active locator 없음"
|
|
target = Path(str(raw_locator))
|
|
locator_path = target if target.name == "locator.json" else target / "locator.json"
|
|
locator: dict[str, Any] = {}
|
|
if locator_path.is_file():
|
|
try:
|
|
locator = json.loads(locator_path.read_text(encoding="utf-8"))
|
|
except (OSError, json.JSONDecodeError):
|
|
return False, f"locator 판독 실패: {locator_path}"
|
|
status = str(locator.get("status") or "")
|
|
if status and status != "running":
|
|
return False, f"locator status={status}"
|
|
|
|
# The stream may legitimately remain quiet during long reasoning. A live
|
|
# process is stronger evidence than a locator or dispatcher heartbeat, and
|
|
# prevents a second dispatcher from duplicating an active attempt.
|
|
agent_pid_recorded = locator.get("agent_pid") not in (None, "")
|
|
for field, token_field in (
|
|
("agent_pid", "agent_process_start_token"),
|
|
("dispatcher_pid", "dispatcher_process_start_token"),
|
|
):
|
|
if process_is_alive(locator.get(field), locator.get(token_field)):
|
|
return True, f"{field}={locator[field]} alive; output stream is monitored"
|
|
process_marker = str(locator.get("agent_process_marker") or "")
|
|
if process_marker:
|
|
marker_pids = marked_agent_process_pids(process_marker)
|
|
if marker_pids:
|
|
return (
|
|
True,
|
|
"agent process marker alive: "
|
|
+ ",".join(str(pid) for pid in marker_pids),
|
|
)
|
|
return (
|
|
False,
|
|
"agent process marker is absent from the process table",
|
|
)
|
|
if agent_pid_recorded:
|
|
return (
|
|
False,
|
|
"recorded agent process identity is no longer alive",
|
|
)
|
|
|
|
native_raw = locator.get("native_session_path")
|
|
native = Path(str(native_raw)) if native_raw else None
|
|
if native and native.is_dir():
|
|
sessions = list(native.glob("*.jsonl"))
|
|
native = max(sessions, key=lambda path: path.stat().st_mtime_ns) if sessions else None
|
|
if native is None or not native.is_file():
|
|
roots = [target] if target.is_dir() else [target.parent]
|
|
sessions = [
|
|
path
|
|
for root in roots
|
|
for path in (*root.glob("*.jsonl"), *root.glob("pi-sessions/*.jsonl"))
|
|
]
|
|
native = max(sessions, key=lambda path: path.stat().st_mtime_ns) if sessions else None
|
|
now = datetime.now(timezone.utc).timestamp()
|
|
cli = str(locator.get("cli") or "")
|
|
stream_progress_at: float | None = None
|
|
stream_raw = locator.get("stream_log")
|
|
stream = Path(str(stream_raw)) if stream_raw else None
|
|
if stream and stream.is_file():
|
|
stream_progress_at = stream.stat().st_mtime
|
|
if native and native.is_file():
|
|
native_progress_at = native.stat().st_mtime
|
|
progress_at = max(native_progress_at, stream_progress_at or 0.0)
|
|
inactive = max(0.0, now - progress_at)
|
|
if cli == "pi":
|
|
phase = pi_native_session_phase(str(native))
|
|
# Only an exact incomplete toolCall -> toolResult batch is a tool
|
|
# execution interval. Unknown/starting/model-reasoning states
|
|
# must never be treated as a stalled tool merely because their
|
|
# native event file is quiet.
|
|
if phase == "tool-running":
|
|
return (
|
|
True,
|
|
"phase=tool-running with no agent PID evidence; "
|
|
"time-based duplicate recovery is disabled",
|
|
)
|
|
return (
|
|
True,
|
|
f"phase={phase} native+stream inactive={inactive:.1f}s "
|
|
"with no agent PID evidence; time-based duplicate recovery is disabled",
|
|
)
|
|
return (
|
|
True,
|
|
"native+stream inactive="
|
|
f"{inactive:.1f}s with no agent PID evidence; "
|
|
"time-based duplicate recovery is disabled",
|
|
)
|
|
|
|
if stream_progress_at is not None:
|
|
inactive = max(0.0, now - stream_progress_at)
|
|
return (
|
|
True,
|
|
f"stream inactive={inactive:.1f}s with no agent PID evidence; "
|
|
"time-based duplicate recovery is disabled",
|
|
)
|
|
|
|
return False, f"active 증거 없음: {raw_locator}"
|
|
|
|
|
|
def laguna_resume_locator(state: dict[str, Any]) -> Path | None:
|
|
raw_locator = state.get("active_locator")
|
|
if not raw_locator:
|
|
return None
|
|
locator = Path(str(raw_locator))
|
|
try:
|
|
record = json.loads(locator.read_text(encoding="utf-8"))
|
|
except (OSError, json.JSONDecodeError):
|
|
return None
|
|
if (
|
|
record.get("cli") != "pi"
|
|
or not str(record.get("model", "")).startswith("laguna-s")
|
|
or record.get("failure_class") not in {"context-limit", "session-stall"}
|
|
or record.get("status") != "failed"
|
|
):
|
|
return None
|
|
native_raw = record.get("native_session_path")
|
|
native = Path(str(native_raw)) if native_raw else None
|
|
if native is None or not native.exists():
|
|
return None
|
|
return locator
|
|
|
|
|
|
def agy_conversations() -> dict[Path, int]:
|
|
root = Path.home() / ".gemini" / "antigravity-cli" / "conversations"
|
|
if not root.is_dir():
|
|
return {}
|
|
return {path: path.stat().st_mtime_ns for path in root.glob("*.db")}
|
|
|
|
|
|
def build_command(
|
|
spec: AgentSpec,
|
|
prompt: str,
|
|
workspace: Path,
|
|
session_id: str,
|
|
attempt_dir: Path,
|
|
pi_resume_session: Path | None = None,
|
|
) -> list[str]:
|
|
if spec.cli == "codex":
|
|
return [
|
|
"codex", "exec", "--json", "-C", str(workspace), "-m", spec.model,
|
|
"-c", 'model_reasoning_effort="xhigh"',
|
|
"--dangerously-bypass-approvals-and-sandbox", prompt,
|
|
]
|
|
if spec.cli == "claude":
|
|
return [
|
|
"claude", "-p", "--output-format", "stream-json", "--verbose",
|
|
"--session-id", session_id, "--model", spec.model, "--effort", "xhigh",
|
|
"--dangerously-skip-permissions", prompt,
|
|
]
|
|
if spec.cli == "agy":
|
|
return [
|
|
"agy", "--print", "--print-timeout", "8h", "--model", spec.model,
|
|
"--dangerously-skip-permissions", "--log-file", str(attempt_dir / "agy-cli.log"), prompt,
|
|
]
|
|
if spec.cli == "pi":
|
|
command = [
|
|
"pi", "-p", "--mode", "json", "--approve", "--provider", "iop", "--model", spec.model,
|
|
"--thinking", "high",
|
|
]
|
|
if pi_resume_session is not None:
|
|
command.extend(
|
|
[
|
|
"--session", str(pi_resume_session),
|
|
"--session-dir", str(pi_resume_session.parent),
|
|
]
|
|
)
|
|
else:
|
|
command.extend(
|
|
[
|
|
"--session-id", session_id,
|
|
"--session-dir", str(attempt_dir / "pi-sessions"),
|
|
]
|
|
)
|
|
command.append(prompt)
|
|
return command
|
|
raise RuntimeError(f"지원하지 않는 CLI: {spec.cli}")
|
|
|
|
|
|
async def invoke(
|
|
workspace: Path,
|
|
store: StateStore,
|
|
task: Task,
|
|
role: str,
|
|
spec: AgentSpec,
|
|
prompt: str,
|
|
resume_locator: Path | None = None,
|
|
) -> tuple[int, str | None, Path]:
|
|
attempt, identity = next_execution_identity(store, task, role)
|
|
attempt_dir = store.runs / f"{datetime.now(timezone.utc).strftime('%Y%m%dT%H%M%SZ')}__{identity}"
|
|
attempt_dir.mkdir(parents=True, exist_ok=False)
|
|
locator_path = attempt_dir / "locator.json"
|
|
stream_path = attempt_dir / "stream.log"
|
|
heartbeat_path = attempt_dir / "heartbeat.log"
|
|
stream_path.touch()
|
|
heartbeat_path.touch()
|
|
session_id = str(uuid.uuid4())
|
|
process_marker = f"{identity}__{uuid.uuid4()}"
|
|
pi_resume_session: Path | None = None
|
|
if spec.local_pi and resume_locator and resume_locator.is_file():
|
|
try:
|
|
prior = json.loads(resume_locator.read_text(encoding="utf-8"))
|
|
except (OSError, json.JSONDecodeError):
|
|
prior = {}
|
|
prior_native = prior.get("native_session_path")
|
|
candidate = Path(str(prior_native)) if prior_native else None
|
|
if candidate and candidate.is_dir():
|
|
sessions = list(candidate.glob("*.jsonl"))
|
|
candidate = max(sessions, key=lambda path: path.stat().st_mtime_ns) if sessions else None
|
|
if candidate and candidate.is_file():
|
|
pi_resume_session = candidate
|
|
session_id = str(prior.get("session_id") or candidate.stem)
|
|
started_at = now_iso()
|
|
work_log_path = milestone_work_log_path(task)
|
|
record: dict[str, Any] = {
|
|
"execution_id": identity,
|
|
"task": task.name,
|
|
"plan_number": plan_number(task),
|
|
"role": role,
|
|
"attempt": attempt,
|
|
"workspace": str(workspace),
|
|
**dispatcher_source_provenance(),
|
|
"cli": spec.cli,
|
|
"model": spec.model,
|
|
"agent_process_marker": process_marker,
|
|
"plan_path": str(task.plan) if task.plan else None,
|
|
"review_path": str(task.review) if task.review else None,
|
|
"session_id": session_id if spec.cli in {"claude", "pi"} else None,
|
|
"native_session_path": (
|
|
str(pi_resume_session)
|
|
if pi_resume_session is not None
|
|
else native_session_path(spec.cli, workspace, session_id, attempt_dir)
|
|
),
|
|
"output_log": str(stream_path),
|
|
"stream_log": str(stream_path),
|
|
"heartbeat_log": str(heartbeat_path),
|
|
"cli_log": str(attempt_dir / "agy-cli.log") if spec.cli == "agy" else None,
|
|
"work_log": str(work_log_path.resolve()),
|
|
"started_at": started_at,
|
|
"status": "running",
|
|
"resumed_from_locator": str(resume_locator) if pi_resume_session else None,
|
|
}
|
|
write_json(locator_path, record)
|
|
store.update_task(task, active_locator=str(locator_path))
|
|
prefix = f"[{task.directory.name}][{role}][a{attempt:02d}]"
|
|
|
|
def persist_locator_record() -> None:
|
|
"""Do not abort a live model solely because a locator refresh failed."""
|
|
try:
|
|
write_json(locator_path, record)
|
|
except OSError as exc:
|
|
record["locator_write_error"] = str(exc)
|
|
print(
|
|
f"{prefix} locator 기록 경고: locator={locator_path} error={exc}",
|
|
flush=True,
|
|
)
|
|
|
|
print(f"{prefix} locator={locator_path}", flush=True)
|
|
try:
|
|
append_milestone_event(
|
|
task,
|
|
event="START",
|
|
execution_id=identity,
|
|
role=role,
|
|
attempt=attempt,
|
|
model=spec.display,
|
|
result="running",
|
|
locator=locator_path,
|
|
)
|
|
except OSError as exc:
|
|
line = f"milestone work log setup failed: {exc}"
|
|
heartbeat_path.write_text(line + "\n", encoding="utf-8")
|
|
record.update(
|
|
status="failed",
|
|
finished_at=now_iso(),
|
|
exit_code=1,
|
|
failure_class="work-log-setup",
|
|
failure_source="work-log",
|
|
provider_transport_failure_confirmed=False,
|
|
work_log_error=str(exc),
|
|
)
|
|
persist_locator_record()
|
|
print(f"{prefix} {line}", flush=True)
|
|
return 1, "work-log-setup", locator_path
|
|
command = build_command(
|
|
spec,
|
|
prompt,
|
|
workspace,
|
|
session_id,
|
|
attempt_dir,
|
|
pi_resume_session=pi_resume_session,
|
|
)
|
|
before_agy = agy_conversations() if spec.cli == "agy" else {}
|
|
diagnostics: list[str] = []
|
|
diagnostic_origins: list[str] = []
|
|
control_violation: str | None = None
|
|
try:
|
|
process = await asyncio.create_subprocess_exec(
|
|
*command,
|
|
cwd=workspace,
|
|
env={
|
|
**os.environ,
|
|
AGENT_PROCESS_MARKER_ENV: process_marker,
|
|
},
|
|
stdout=asyncio.subprocess.PIPE,
|
|
stderr=asyncio.subprocess.PIPE,
|
|
limit=10 * 1024 * 1024,
|
|
start_new_session=True,
|
|
)
|
|
# Keep the child PID in the locator before monitoring output. If this
|
|
# dispatcher is interrupted, a later dispatcher can distinguish a
|
|
# genuinely live, silent model from a stale locator and must not launch
|
|
# a duplicate continuation.
|
|
record["agent_pid"] = process.pid
|
|
record["agent_process_start_token"] = process_start_token(process.pid)
|
|
persist_locator_record()
|
|
except FileNotFoundError:
|
|
line = f"command not found: {command[0]}"
|
|
heartbeat_path.write_text(line + "\n", encoding="utf-8")
|
|
failure_class = "generic-error"
|
|
try:
|
|
append_milestone_event(
|
|
task,
|
|
event="FINISH",
|
|
execution_id=identity,
|
|
role=role,
|
|
attempt=attempt,
|
|
model=spec.display,
|
|
result=f"failed:{failure_class}:127",
|
|
locator=locator_path,
|
|
)
|
|
except OSError as exc:
|
|
record["work_log_runtime_error"] = str(exc)
|
|
failure_class = "work-log-runtime-write"
|
|
record.update(
|
|
status="failed",
|
|
finished_at=now_iso(),
|
|
exit_code=127,
|
|
failure_class=failure_class,
|
|
failure_source=(
|
|
"work-log" if failure_class == "work-log-runtime-write" else "cli-launch"
|
|
),
|
|
provider_transport_failure_confirmed=False,
|
|
)
|
|
persist_locator_record()
|
|
print(f"{prefix} {line}", flush=True)
|
|
return 127, failure_class, locator_path
|
|
|
|
readers: list[asyncio.Task[None]] = []
|
|
try:
|
|
assert process.stdout is not None and process.stderr is not None
|
|
queue: asyncio.Queue[tuple[str, bytes | None]] = asyncio.Queue()
|
|
|
|
async def pump(channel: str, stream: asyncio.StreamReader) -> None:
|
|
try:
|
|
while True:
|
|
raw = await stream.readline()
|
|
if not raw:
|
|
break
|
|
await queue.put((channel, raw))
|
|
finally:
|
|
await queue.put((channel, None))
|
|
|
|
readers = [
|
|
asyncio.create_task(pump("stdout", process.stdout)),
|
|
asyncio.create_task(pump("stderr", process.stderr)),
|
|
]
|
|
finished_streams = 0
|
|
loop = asyncio.get_running_loop()
|
|
last_native_mtime: int | None = None
|
|
last_stream_mtime: int | None = None
|
|
last_native_progress_at = loop.time()
|
|
last_stream_progress_at = loop.time()
|
|
with (
|
|
stream_path.open("w", encoding="utf-8") as stream_log,
|
|
heartbeat_path.open("a", encoding="utf-8") as heartbeat_log,
|
|
):
|
|
while finished_streams < len(readers):
|
|
try:
|
|
channel, raw = await asyncio.wait_for(
|
|
queue.get(), timeout=STREAM_HEARTBEAT_SECONDS
|
|
)
|
|
except asyncio.TimeoutError:
|
|
try:
|
|
stream_mtime = stream_path.stat().st_mtime_ns
|
|
except OSError:
|
|
stream_mtime = None
|
|
if stream_mtime is not None:
|
|
record["stream_log_mtime_ns"] = stream_mtime
|
|
if stream_mtime != last_stream_mtime:
|
|
last_stream_mtime = stream_mtime
|
|
last_stream_progress_at = loop.time()
|
|
record.pop("pi_silence_inspection", None)
|
|
native_path = (
|
|
str(pi_resume_session)
|
|
if pi_resume_session is not None
|
|
else native_session_path(
|
|
spec.cli,
|
|
workspace,
|
|
record.get("session_id"),
|
|
attempt_dir,
|
|
)
|
|
)
|
|
if native_path:
|
|
record["native_session_path"] = native_path
|
|
native_mtime = native_session_mtime_ns(
|
|
record.get("native_session_path")
|
|
)
|
|
if native_mtime is not None:
|
|
record["native_session_mtime_ns"] = native_mtime
|
|
if native_mtime != last_native_mtime:
|
|
last_native_mtime = native_mtime
|
|
last_native_progress_at = loop.time()
|
|
# Native events and the separately flushed stream log
|
|
# are peer progress signals. A trailing toolResult only
|
|
# selects the timeout budget; it never overrides later
|
|
# reasoning/text output.
|
|
record["pi_activity_state"] = "working"
|
|
pi_session_state = pi_native_session_state(
|
|
record.get("native_session_path")
|
|
)
|
|
pi_phase = pi_session_state.phase
|
|
is_pi_tool_execution = pi_phase == "tool-running"
|
|
# Outside a toolCall->toolResult interval, model stdout/stderr
|
|
# is the liveness signal. A completed tool result changes phase
|
|
# but must not reset the model-response silence clock.
|
|
pi_inactive_seconds = loop.time() - (
|
|
max(last_native_progress_at, last_stream_progress_at)
|
|
if is_pi_tool_execution
|
|
else last_stream_progress_at
|
|
)
|
|
if spec.local_pi:
|
|
record["pi_session_phase"] = pi_phase
|
|
record["pi_session_phase_reason"] = (
|
|
pi_session_state.reason
|
|
)
|
|
record["pi_expected_tool_call_ids"] = list(
|
|
pi_session_state.expected_tool_call_ids
|
|
)
|
|
record["pi_completed_tool_call_ids"] = list(
|
|
pi_session_state.completed_tool_call_ids
|
|
)
|
|
record["pi_pending_tool_call_ids"] = list(
|
|
pi_session_state.pending_tool_call_ids
|
|
)
|
|
record["pi_stall_timeout_seconds"] = None
|
|
record.setdefault("pi_activity_state", "starting")
|
|
if (
|
|
spec.local_pi
|
|
and not is_pi_tool_execution
|
|
and pi_inactive_seconds >= PI_MODEL_RESPONSE_STALL_SECONDS
|
|
and "pi_silence_inspection" not in record
|
|
):
|
|
inspection = {
|
|
"at": now_iso(),
|
|
"silence_seconds": round(pi_inactive_seconds, 3),
|
|
"stream_tail": log_tail_excerpt(stream_path),
|
|
}
|
|
record["pi_silence_inspection"] = inspection
|
|
diagnostic = (
|
|
f"Pi {pi_phase} stream produced no update for "
|
|
f"{pi_inactive_seconds:.1f}s; recorded stream tail for inspection "
|
|
"without terminating the model process"
|
|
)
|
|
heartbeat_log.write(f"[silence-inspection] {diagnostic}\n")
|
|
heartbeat_log.flush()
|
|
persist_locator_record()
|
|
print(f"{prefix} 모델응답점검: {diagnostic}", flush=True)
|
|
non_pi_inactive_seconds = loop.time() - max(
|
|
last_native_progress_at, last_stream_progress_at
|
|
)
|
|
if (
|
|
not spec.local_pi
|
|
and non_pi_inactive_seconds
|
|
>= PI_MODEL_RESPONSE_STALL_SECONDS
|
|
and "stream_silence_inspection" not in record
|
|
):
|
|
inspection = {
|
|
"at": now_iso(),
|
|
"silence_seconds": round(non_pi_inactive_seconds, 3),
|
|
"stream_tail": log_tail_excerpt(stream_path),
|
|
}
|
|
record["stream_silence_inspection"] = inspection
|
|
diagnostic = (
|
|
f"{spec.cli} emitted no stream output or native-session event for "
|
|
f"{non_pi_inactive_seconds:.1f}s; recorded stream tail for inspection "
|
|
"without terminating the model process"
|
|
)
|
|
heartbeat_log.write(f"[silence-inspection] {diagnostic}\n")
|
|
heartbeat_log.flush()
|
|
persist_locator_record()
|
|
print(f"{prefix} 모델응답점검: {diagnostic}", flush=True)
|
|
heartbeat = (
|
|
f"작업중... locator={locator_path} "
|
|
f"native_session={record.get('native_session_path') or 'none'} "
|
|
f"native_mtime_ns={record.get('native_session_mtime_ns', 'none')}"
|
|
)
|
|
if spec.local_pi:
|
|
heartbeat += (
|
|
f" pi_activity={record.get('pi_activity_state')}"
|
|
f" pi_phase={pi_phase}"
|
|
)
|
|
heartbeat_log.write(f"[heartbeat] {heartbeat}\n")
|
|
heartbeat_log.flush()
|
|
persist_locator_record()
|
|
print(f"{prefix} {heartbeat}", flush=True)
|
|
continue
|
|
if raw is None:
|
|
finished_streams += 1
|
|
continue
|
|
record.pop("pi_silence_inspection", None)
|
|
record.pop("stream_silence_inspection", None)
|
|
if spec.local_pi and channel == "stdout":
|
|
record["pi_activity_state"] = "streaming"
|
|
line = raw.decode("utf-8", errors="replace").rstrip("\n")
|
|
stream_log.write(f"[{channel}] {line}\n")
|
|
stream_log.flush()
|
|
diagnostic = terminal_diagnostic(spec.cli, channel, line)
|
|
if diagnostic:
|
|
diagnostics.append(diagnostic)
|
|
diagnostic_origins.append(f"{spec.cli}:{channel}")
|
|
if spec.cli == "codex" and role == "review" and channel == "stdout":
|
|
collaboration_tool = codex_collaboration_tool(line)
|
|
if collaboration_tool and control_violation is None:
|
|
control_violation = collaboration_tool
|
|
diagnostics.append(
|
|
f"official review invoked forbidden collaboration tool: "
|
|
f"{collaboration_tool}"
|
|
)
|
|
diagnostic_origins.append("dispatcher:review-control")
|
|
print(
|
|
f"{prefix} 리뷰 제어 계약 위반: collaboration-tool="
|
|
f"{collaboration_tool}",
|
|
flush=True,
|
|
)
|
|
await terminate_process_group(process)
|
|
rendered, discovered = (
|
|
render_json_line(spec.cli, line) if channel == "stdout" else ([line], None)
|
|
)
|
|
if discovered and record.get("session_id") != discovered:
|
|
record["session_id"] = discovered
|
|
if pi_resume_session is None:
|
|
record["native_session_path"] = native_session_path(
|
|
spec.cli, workspace, discovered, attempt_dir
|
|
)
|
|
persist_locator_record()
|
|
for display_line in rendered:
|
|
if display_line:
|
|
print(f"{prefix} {display_line}", flush=True)
|
|
await asyncio.gather(*readers)
|
|
return_code = await process.wait()
|
|
except asyncio.CancelledError:
|
|
for reader in readers:
|
|
reader.cancel()
|
|
if readers:
|
|
await asyncio.gather(*readers, return_exceptions=True)
|
|
await terminate_process_group(process)
|
|
runtime_error: OSError | None = None
|
|
try:
|
|
append_milestone_event(
|
|
task,
|
|
event="FINISH",
|
|
execution_id=identity,
|
|
role=role,
|
|
attempt=attempt,
|
|
model=spec.display,
|
|
result="failed:cancelled",
|
|
locator=locator_path,
|
|
)
|
|
except OSError as exc:
|
|
runtime_error = exc
|
|
record.update(
|
|
status="failed",
|
|
finished_at=now_iso(),
|
|
exit_code="cancelled",
|
|
failure_class="cancelled",
|
|
failure_source="caller-cancel",
|
|
provider_transport_failure_confirmed=False,
|
|
)
|
|
if runtime_error is not None:
|
|
record["work_log_runtime_error"] = str(runtime_error)
|
|
persist_locator_record()
|
|
raise
|
|
|
|
if spec.cli == "agy":
|
|
after_agy = agy_conversations()
|
|
changed = [
|
|
path for path, mtime in after_agy.items()
|
|
if path not in before_agy or before_agy[path] != mtime
|
|
]
|
|
if changed:
|
|
selected = max(changed, key=lambda path: after_agy[path])
|
|
record["session_id"] = selected.stem
|
|
record["native_session_path"] = str(selected)
|
|
agy_diagnostics = agy_log_diagnostics(attempt_dir / "agy-cli.log")
|
|
diagnostics.extend(agy_diagnostics)
|
|
diagnostic_origins.extend("agy:cli-log" for _ in agy_diagnostics)
|
|
native_path = (
|
|
str(pi_resume_session)
|
|
if pi_resume_session is not None
|
|
else native_session_path(
|
|
spec.cli, workspace, record.get("session_id"), attempt_dir
|
|
)
|
|
)
|
|
if native_path:
|
|
record["native_session_path"] = native_path
|
|
native_mtime = native_session_mtime_ns(record.get("native_session_path"))
|
|
if native_mtime is not None:
|
|
record["native_session_mtime_ns"] = native_mtime
|
|
failure_source: str | None = None
|
|
failure_evidence: str | None = None
|
|
failure_evidence_source: str | None = None
|
|
provider_transport_failure_confirmed = False
|
|
termination = termination_signal(return_code)
|
|
if termination is not None:
|
|
record["termination_signal"] = termination[0]
|
|
record["termination_signal_inferred"] = termination[1]
|
|
if control_violation:
|
|
failure_class = "review-control-violation"
|
|
failure_source = "dispatcher-control"
|
|
failure_evidence_source = "dispatcher:review-control"
|
|
for index in range(len(diagnostics) - 1, -1, -1):
|
|
if diagnostic_origins[index] == failure_evidence_source:
|
|
failure_evidence = diagnostics[index]
|
|
break
|
|
elif return_code != 0 and termination is not None:
|
|
failure_class = "process-terminated"
|
|
failure_source = "process-termination"
|
|
record["termination_initiator"] = "unknown"
|
|
elif return_code != 0:
|
|
failure_class, failure_evidence = classify_failure_with_evidence(
|
|
"\n".join(diagnostics[-50:])
|
|
)
|
|
if failure_evidence is not None:
|
|
for index in range(len(diagnostics) - 1, -1, -1):
|
|
if diagnostics[index] == failure_evidence:
|
|
failure_evidence_source = diagnostic_origins[index]
|
|
break
|
|
if failure_class in PROVIDER_TRANSPORT_FAILURES:
|
|
failure_source = "provider-terminal-diagnostic"
|
|
provider_transport_failure_confirmed = failure_evidence is not None
|
|
elif failure_evidence is not None:
|
|
failure_source = "cli-terminal-diagnostic"
|
|
else:
|
|
failure_source = "cli-exit"
|
|
else:
|
|
failure_class = None
|
|
try:
|
|
append_milestone_event(
|
|
task,
|
|
event="FINISH",
|
|
execution_id=identity,
|
|
role=role,
|
|
attempt=attempt,
|
|
model=spec.display,
|
|
result=(
|
|
f"succeeded:0"
|
|
if return_code == 0 and failure_class is None
|
|
else f"failed:{failure_class or 'generic-error'}:{return_code}"
|
|
),
|
|
locator=locator_path,
|
|
)
|
|
except OSError as exc:
|
|
if failure_class is not None:
|
|
record["prior_failure_class"] = failure_class
|
|
record["work_log_runtime_error"] = str(exc)
|
|
failure_class = "work-log-runtime-write"
|
|
failure_source = "work-log"
|
|
failure_evidence = None
|
|
failure_evidence_source = None
|
|
provider_transport_failure_confirmed = False
|
|
if failure_evidence is not None:
|
|
record["failure_evidence_excerpt"] = failure_evidence[:FAILURE_EVIDENCE_LIMIT]
|
|
record["failure_evidence_truncated"] = (
|
|
len(failure_evidence) > FAILURE_EVIDENCE_LIMIT
|
|
)
|
|
if failure_evidence_source is not None:
|
|
record["failure_evidence_source"] = failure_evidence_source
|
|
record.update(
|
|
status="succeeded" if return_code == 0 and failure_class is None else "failed",
|
|
finished_at=now_iso(),
|
|
exit_code=return_code,
|
|
failure_class=failure_class,
|
|
failure_source=failure_source,
|
|
provider_transport_failure_confirmed=provider_transport_failure_confirmed,
|
|
)
|
|
persist_locator_record()
|
|
return return_code, failure_class, locator_path
|
|
|
|
|
|
def base_prompt(task: Task, role: str, spec: AgentSpec) -> str:
|
|
if role == "review":
|
|
target = task.review or task.directory
|
|
if task.review:
|
|
return f"Read {target.resolve()} and start the review. Final in Korean."
|
|
return f"Continue the review for {target.resolve()}. Final in Korean."
|
|
if task.plan is None:
|
|
raise RuntimeError("worker PLAN이 없다")
|
|
target = task.plan.resolve()
|
|
if role == "selfcheck":
|
|
if task.review is None:
|
|
raise RuntimeError("selfcheck CODE_REVIEW 파일이 없다")
|
|
return (
|
|
f"Think in English. Final in Korean. Read {task.review.resolve()} and fill "
|
|
"every missing implementation field. Do not finish until all implementation "
|
|
"fields are complete. This is a self-check of completed work, not a review. "
|
|
f"Read {target} and finish any missing work. Recheck and fix your work."
|
|
)
|
|
if spec.local_pi:
|
|
return f"Think in English. Final in Korean. Read {target} and complete the task."
|
|
return f"Read {target} and complete the task. Final in Korean."
|
|
|
|
|
|
def continuation_prompt(
|
|
task: Task,
|
|
role: str,
|
|
locator: Path,
|
|
*,
|
|
local_pi: bool = False,
|
|
resume_same_pi_session: bool = False,
|
|
) -> str:
|
|
if local_pi:
|
|
if resume_same_pi_session:
|
|
return (
|
|
"Think in English. Final in Korean. Continue this session and complete "
|
|
"the current task."
|
|
)
|
|
if role == "selfcheck" and task.plan and task.review:
|
|
return (
|
|
f"Think in English. Final in Korean. Read {task.review.resolve()} and fill "
|
|
"every missing implementation field. Do not finish until all implementation "
|
|
"fields are complete. This is a self-check of completed work, not a review. "
|
|
f"Read {task.plan.resolve()} and finish any missing work. Recheck and fix "
|
|
"your work."
|
|
)
|
|
target = task.plan or task.directory
|
|
return f"Think in English. Final in Korean. Read {target.resolve()} and complete the task."
|
|
if role == "review":
|
|
return f"Continue the review for {task.directory.resolve()}. Final in Korean."
|
|
return (
|
|
f"Continue from {locator.resolve()}. Check the saved context and current "
|
|
"workspace. Final in Korean."
|
|
)
|
|
|
|
|
|
async def run_escalating(
|
|
workspace: Path,
|
|
store: StateStore,
|
|
task: Task,
|
|
role: str,
|
|
initial: AgentSpec,
|
|
invocation_semaphore: asyncio.Semaphore | None = None,
|
|
initial_resume_locator: Path | None = None,
|
|
) -> tuple[bool, Path | None]:
|
|
spec = initial
|
|
previous_locator = initial_resume_locator
|
|
codex_recovery_count = 0
|
|
codex_session_stall_retries = 0
|
|
review_control_retries = 0
|
|
pi_recovery_retries = 0
|
|
generic_retries = 0
|
|
terminal_recovery_retries = 0
|
|
pi_resume_locator = initial_resume_locator
|
|
recovery_failures = 0
|
|
if isinstance(store, StateStore):
|
|
persisted = store.task_state(task).get("recovery_failures", {})
|
|
if isinstance(persisted, dict):
|
|
recovery_failures = int(persisted.get(role, 0))
|
|
if recovery_failures >= RECOVERY_FAILURE_LIMIT:
|
|
locator = initial_resume_locator
|
|
reason = (
|
|
f"{role} recovery failure limit already exhausted: "
|
|
f"{recovery_failures}/{RECOVERY_FAILURE_LIMIT}"
|
|
)
|
|
if isinstance(store, StateStore):
|
|
store.update_task(task, blocked=f"{reason} locator={locator}")
|
|
banner(
|
|
"작업차단",
|
|
task.name,
|
|
[
|
|
"reason=recovery-failure-limit",
|
|
f"retry={recovery_failures}/{RECOVERY_FAILURE_LIMIT}",
|
|
f"locator={locator}",
|
|
],
|
|
)
|
|
return False, locator
|
|
while True:
|
|
prompt = (
|
|
base_prompt(task, role, spec)
|
|
if previous_locator is None
|
|
else continuation_prompt(
|
|
task,
|
|
role,
|
|
previous_locator,
|
|
local_pi=spec.local_pi,
|
|
resume_same_pi_session=pi_resume_locator is not None,
|
|
)
|
|
)
|
|
if invocation_semaphore is None:
|
|
rc, failure, locator = await invoke(
|
|
workspace,
|
|
store,
|
|
task,
|
|
role,
|
|
spec,
|
|
prompt,
|
|
resume_locator=pi_resume_locator,
|
|
)
|
|
else:
|
|
async with invocation_semaphore:
|
|
rc, failure, locator = await invoke(
|
|
workspace,
|
|
store,
|
|
task,
|
|
role,
|
|
spec,
|
|
prompt,
|
|
resume_locator=pi_resume_locator,
|
|
)
|
|
pi_resume_locator = None
|
|
if rc == 0 and failure is None:
|
|
if isinstance(store, StateStore):
|
|
state = store.task_state(task)
|
|
persisted = dict(state.get("recovery_failures", {}))
|
|
persisted.pop(role, None)
|
|
store.update_task(task, recovery_failures=persisted)
|
|
return True, locator
|
|
failure = failure or "generic-error"
|
|
if failure in {
|
|
"work-log-blocked",
|
|
"work-log-incomplete",
|
|
"work-log-setup",
|
|
"work-log-runtime-write",
|
|
}:
|
|
banner("작업차단", task.name, failure_report_lines(failure, locator))
|
|
return False, locator
|
|
recovery_failures += 1
|
|
if isinstance(store, StateStore):
|
|
state = store.task_state(task)
|
|
persisted = dict(state.get("recovery_failures", {}))
|
|
persisted[role] = recovery_failures
|
|
store.update_task(task, recovery_failures=persisted)
|
|
if recovery_failures >= RECOVERY_FAILURE_LIMIT:
|
|
reason = (
|
|
f"{role} recovery failure limit exhausted: "
|
|
f"{recovery_failures}/{RECOVERY_FAILURE_LIMIT}"
|
|
)
|
|
if isinstance(store, StateStore):
|
|
store.update_task(task, blocked=f"{reason} locator={locator}")
|
|
banner(
|
|
"작업차단",
|
|
task.name,
|
|
[
|
|
"reason=recovery-failure-limit",
|
|
*failure_report_lines(failure, locator),
|
|
f"retry={recovery_failures}/{RECOVERY_FAILURE_LIMIT}",
|
|
],
|
|
)
|
|
return False, locator
|
|
if role == "review" and failure == "review-control-violation":
|
|
review_control_retries += 1
|
|
banner(
|
|
"리뷰재시도",
|
|
task.name,
|
|
[
|
|
*failure_report_lines(failure, locator),
|
|
f"retry={recovery_failures}/{RECOVERY_FAILURE_LIMIT}",
|
|
],
|
|
)
|
|
previous_locator = None
|
|
await asyncio.sleep(min(30, 2 ** min(review_control_retries, 5)))
|
|
continue
|
|
if spec.local_pi:
|
|
if (
|
|
spec.model.startswith("laguna-s")
|
|
and failure in {"context-limit", "session-stall"}
|
|
):
|
|
pi_recovery_retries += 1
|
|
banner(
|
|
"Pi세션연속재시작",
|
|
task.name,
|
|
[
|
|
f"model={spec.display}",
|
|
*failure_report_lines(failure, locator),
|
|
f"retry={recovery_failures}/{RECOVERY_FAILURE_LIMIT}",
|
|
],
|
|
)
|
|
previous_locator = locator
|
|
pi_resume_locator = locator
|
|
await asyncio.sleep(min(30, 2 ** min(pi_recovery_retries, 5)))
|
|
continue
|
|
pi_recovery_retries += 1
|
|
if failure == "session-stall":
|
|
event = "세션응답복구재시도"
|
|
elif failure in {
|
|
"provider-connection",
|
|
"provider-stream-disconnect",
|
|
}:
|
|
event = "세션연결재시도"
|
|
else:
|
|
event = "Pi복구재시도"
|
|
banner(
|
|
event,
|
|
task.name,
|
|
[
|
|
f"model={spec.display}",
|
|
*failure_report_lines(failure, locator),
|
|
f"retry={recovery_failures}/{RECOVERY_FAILURE_LIMIT}",
|
|
],
|
|
)
|
|
previous_locator = locator
|
|
await asyncio.sleep(min(30, 2 ** min(pi_recovery_retries, 5)))
|
|
continue
|
|
if spec.cli == "codex" and failure == "session-stall":
|
|
codex_session_stall_retries += 1
|
|
banner(
|
|
"세션응답복구재시도",
|
|
task.name,
|
|
[
|
|
f"model={spec.display}",
|
|
*failure_report_lines(failure, locator),
|
|
f"retry={recovery_failures}/{RECOVERY_FAILURE_LIMIT}",
|
|
],
|
|
)
|
|
previous_locator = locator
|
|
await asyncio.sleep(min(30, 2 ** min(codex_session_stall_retries, 5)))
|
|
continue
|
|
if failure == "generic-error":
|
|
generic_retries += 1
|
|
banner(
|
|
"작업복구재시도",
|
|
task.name,
|
|
[
|
|
f"model={spec.display}",
|
|
*failure_report_lines(failure, locator),
|
|
f"retry={recovery_failures}/{RECOVERY_FAILURE_LIMIT}",
|
|
],
|
|
)
|
|
previous_locator = locator
|
|
await asyncio.sleep(min(30, 2 ** min(generic_retries, 5)))
|
|
continue
|
|
next_spec = promoted_spec(spec, codex_recovery_count)
|
|
if next_spec is None:
|
|
terminal_recovery_retries += 1
|
|
banner(
|
|
"모델복구재시도",
|
|
task.name,
|
|
[
|
|
f"model={spec.display}",
|
|
*failure_report_lines(failure, locator),
|
|
f"retry={recovery_failures}/{RECOVERY_FAILURE_LIMIT}",
|
|
],
|
|
)
|
|
previous_locator = locator
|
|
await asyncio.sleep(min(30, 2 ** min(terminal_recovery_retries, 5)))
|
|
continue
|
|
if spec.cli == "codex":
|
|
codex_recovery_count += 1
|
|
banner(
|
|
"모델승격",
|
|
task.name,
|
|
[
|
|
f"from={spec.display}",
|
|
f"to={next_spec.display}",
|
|
*failure_report_lines(failure, locator),
|
|
],
|
|
)
|
|
spec = next_spec
|
|
previous_locator = locator
|
|
|
|
|
|
def task_signature(workspace: Path, task: Task) -> str:
|
|
digest = hashlib.sha256()
|
|
if not task.directory.exists():
|
|
return "moved"
|
|
for path in sorted(p for p in task.directory.iterdir() if p.is_file()):
|
|
if (
|
|
PLAN_RE.match(path.name)
|
|
or REVIEW_RE.match(path.name)
|
|
or path.name.endswith(".log")
|
|
):
|
|
digest.update(path.name.encode())
|
|
digest.update(sha256_file(path).encode())
|
|
for raw_path in sorted(task.write_set):
|
|
path = Path(raw_path)
|
|
path = path if path.is_absolute() else workspace / path
|
|
digest.update(raw_path.encode())
|
|
if path.is_file():
|
|
digest.update(str(path.stat().st_mode).encode())
|
|
digest.update(sha256_file(path).encode())
|
|
elif path.exists():
|
|
digest.update(b"non-file")
|
|
else:
|
|
digest.update(b"missing")
|
|
return digest.hexdigest()
|
|
|
|
|
|
def read_verdict(path: Path) -> str | None:
|
|
if not path.exists():
|
|
return None
|
|
return verdict_from_text(path.read_text(encoding="utf-8", errors="replace"))
|
|
|
|
|
|
def verdict_from_text(text: str) -> str | None:
|
|
headings = list(VERDICT_HEADING_RE.finditer(text))
|
|
if not headings:
|
|
return None
|
|
heading = headings[-1]
|
|
next_heading = re.search(r"^##\s+", text[heading.end():], re.MULTILINE)
|
|
end = heading.end() + next_heading.start() if next_heading else len(text)
|
|
section = text[heading.end():end]
|
|
inline_matches = list(VERDICT_LINE_RE.finditer(section))
|
|
block_matches = list(VERDICT_BLOCK_RE.finditer(section))
|
|
matches = inline_matches + block_matches
|
|
return matches[0].group(1) if len(matches) == 1 else None
|
|
|
|
|
|
def matching_archive_directories_by_name(
|
|
workspace: Path,
|
|
task_name: str,
|
|
*,
|
|
require_complete: bool = True,
|
|
) -> list[Path]:
|
|
archive = workspace / "agent-task" / "archive"
|
|
parts = task_name.split("/")
|
|
if not archive.is_dir() or len(parts) not in {1, 2}:
|
|
return []
|
|
group = parts[0]
|
|
final_name = parts[-1]
|
|
suffix_re = re.compile(rf"^{re.escape(final_name)}(?:_\d+)?$")
|
|
matches: list[Path] = []
|
|
try:
|
|
years = list(archive.iterdir())
|
|
except FileNotFoundError:
|
|
return []
|
|
for year in years:
|
|
if not year.is_dir():
|
|
continue
|
|
try:
|
|
months = list(year.iterdir())
|
|
except FileNotFoundError:
|
|
continue
|
|
for month in months:
|
|
if not month.is_dir():
|
|
continue
|
|
parent = month if len(parts) == 1 else month / group
|
|
if not parent.is_dir():
|
|
continue
|
|
try:
|
|
candidates = list(parent.iterdir())
|
|
except FileNotFoundError:
|
|
continue
|
|
for candidate in candidates:
|
|
if (
|
|
candidate.is_dir()
|
|
and suffix_re.match(candidate.name)
|
|
and (
|
|
not require_complete
|
|
or (candidate / "complete.log").is_file()
|
|
)
|
|
):
|
|
matches.append(candidate)
|
|
return sorted(matches)
|
|
|
|
|
|
def matching_archive_directories(workspace: Path, task: Task) -> list[Path]:
|
|
return matching_archive_directories_by_name(workspace, task.name)
|
|
|
|
|
|
def task_group_name(task_name: str) -> str:
|
|
return task_name.split("/", 1)[0]
|
|
|
|
|
|
def work_log_event_cells(line: str) -> list[str] | None:
|
|
stripped = line.strip()
|
|
if not stripped.startswith("|") or not stripped.endswith("|"):
|
|
return None
|
|
cells = [
|
|
cell.strip().replace(r"\|", "|")
|
|
for cell in re.split(r"(?<!\\)\|", stripped[1:-1])
|
|
]
|
|
return cells if len(cells) == 9 else None
|
|
|
|
|
|
def unfinished_work_log_attempts(path: Path) -> list[dict[str, Any]]:
|
|
"""Return START rows that have no matching FINISH row."""
|
|
try:
|
|
lines = path.read_text(
|
|
encoding="utf-8",
|
|
errors="replace",
|
|
).splitlines()
|
|
except OSError:
|
|
raise
|
|
open_attempts: dict[str, dict[str, Any]] = {}
|
|
for line in lines:
|
|
cells = work_log_event_cells(line)
|
|
if cells is None or cells[2] not in {"START", "FINISH"}:
|
|
continue
|
|
try:
|
|
sequence = int(cells[0])
|
|
attempt = int(cells[5])
|
|
except ValueError:
|
|
continue
|
|
locator = cells[8]
|
|
key = locator or "\0".join(
|
|
(cells[3], cells[4], cells[5], cells[6])
|
|
)
|
|
if cells[2] == "START":
|
|
open_attempts[key] = {
|
|
"sequence": sequence,
|
|
"task_name": cells[3],
|
|
"role": cells[4],
|
|
"attempt": attempt,
|
|
"model": cells[6],
|
|
"locator": locator,
|
|
}
|
|
else:
|
|
open_attempts.pop(key, None)
|
|
return sorted(
|
|
open_attempts.values(),
|
|
key=lambda record: int(record["sequence"]),
|
|
)
|
|
|
|
|
|
def close_unfinished_work_log_attempts(path: Path) -> int:
|
|
"""Close orphaned START rows after verified group completion."""
|
|
unfinished = unfinished_work_log_attempts(path)
|
|
for record in unfinished:
|
|
locator = Path(str(record["locator"]))
|
|
append_work_log_event(
|
|
path,
|
|
task_name=str(record["task_name"]),
|
|
event="FINISH",
|
|
execution_id=f"reconciled-{record['sequence']}",
|
|
role=str(record["role"]),
|
|
attempt=int(record["attempt"]),
|
|
model=str(record["model"]),
|
|
result="reconciled:verified-complete-archive",
|
|
locator=locator,
|
|
)
|
|
return len(unfinished)
|
|
|
|
|
|
def archived_task_group_directories(
|
|
workspace: Path,
|
|
task_group: str,
|
|
) -> list[Path]:
|
|
"""Return month-local archive directories for one logical task group."""
|
|
archive_root = workspace / "agent-task" / "archive"
|
|
if not archive_root.is_dir():
|
|
return []
|
|
suffix_re = re.compile(rf"^{re.escape(task_group)}(?:_\d+)?$")
|
|
matches: list[Path] = []
|
|
try:
|
|
years = list(archive_root.iterdir())
|
|
except FileNotFoundError:
|
|
return []
|
|
for year in years:
|
|
if not year.is_dir():
|
|
continue
|
|
try:
|
|
months = list(year.iterdir())
|
|
except FileNotFoundError:
|
|
continue
|
|
for month in months:
|
|
if not month.is_dir():
|
|
continue
|
|
try:
|
|
candidates = list(month.iterdir())
|
|
except FileNotFoundError:
|
|
continue
|
|
matches.extend(
|
|
candidate
|
|
for candidate in candidates
|
|
if candidate.is_dir() and suffix_re.fullmatch(candidate.name)
|
|
)
|
|
return sorted(matches)
|
|
|
|
|
|
def next_work_log_archive_number(
|
|
workspace: Path,
|
|
task_group: str,
|
|
) -> int:
|
|
numbers = [
|
|
int(match.group(1))
|
|
for directory in archived_task_group_directories(workspace, task_group)
|
|
for path in directory.glob("work_log_*.log")
|
|
if (match := WORK_LOG_ARCHIVE_RE.fullmatch(path.name))
|
|
]
|
|
return max(numbers, default=-1) + 1
|
|
|
|
|
|
def completed_group_archive_directory(
|
|
task_group: str,
|
|
task_names: set[str],
|
|
completed_tasks: dict[str, str],
|
|
) -> Path | None:
|
|
candidates: list[tuple[int, str, Path]] = []
|
|
for task_name in task_names:
|
|
archive_raw = completed_tasks.get(task_name)
|
|
if not archive_raw:
|
|
continue
|
|
archive = Path(archive_raw)
|
|
complete_log = archive / "complete.log"
|
|
if not archive.is_dir() or not complete_log.is_file():
|
|
continue
|
|
target = archive if task_name == task_group else archive.parent
|
|
try:
|
|
completed_at = complete_log.stat().st_mtime_ns
|
|
except OSError:
|
|
continue
|
|
candidates.append((completed_at, str(target), target))
|
|
return (
|
|
max(candidates, key=lambda item: (item[0], item[1]))[2]
|
|
if candidates
|
|
else None
|
|
)
|
|
|
|
|
|
def archive_completed_group_work_logs(
|
|
workspace: Path,
|
|
observed_tasks: set[str],
|
|
completed_tasks: dict[str, str],
|
|
active_or_running: set[str],
|
|
) -> tuple[dict[str, str], dict[str, str]]:
|
|
"""Archive each completed task-group timeline after its last writer exits."""
|
|
observed_by_group: dict[str, set[str]] = {}
|
|
for task_name in observed_tasks:
|
|
observed_by_group.setdefault(task_group_name(task_name), set()).add(
|
|
task_name
|
|
)
|
|
active_groups = {
|
|
task_group_name(task_name)
|
|
for task_name in active_or_running
|
|
}
|
|
archived: dict[str, str] = {}
|
|
errors: dict[str, str] = {}
|
|
for task_group, task_names in sorted(observed_by_group.items()):
|
|
if task_group in active_groups or not task_names <= set(completed_tasks):
|
|
continue
|
|
active_source = (
|
|
workspace / "agent-task" / task_group / WORK_LOG_NAME
|
|
)
|
|
legacy_sources = {
|
|
Path(completed_tasks[task_name]) / WORK_LOG_NAME
|
|
for task_name in task_names
|
|
if (Path(completed_tasks[task_name]) / WORK_LOG_NAME).is_file()
|
|
}
|
|
sources = {
|
|
*legacy_sources,
|
|
*([active_source] if active_source.is_file() else []),
|
|
}
|
|
if not sources:
|
|
continue
|
|
if len(sources) != 1:
|
|
errors[task_group] = (
|
|
"active/archive WORK_LOG source가 여러 개라 자동 병합할 수 없다: "
|
|
+ ",".join(str(path) for path in sorted(sources))
|
|
)
|
|
continue
|
|
target_directory = completed_group_archive_directory(
|
|
task_group,
|
|
task_names,
|
|
completed_tasks,
|
|
)
|
|
if target_directory is None:
|
|
errors[task_group] = (
|
|
"검증된 task archive에서 WORK_LOG 대상 디렉터리를 정할 수 없다"
|
|
)
|
|
continue
|
|
archive_number = next_work_log_archive_number(
|
|
workspace,
|
|
task_group,
|
|
)
|
|
destination = target_directory / f"work_log_{archive_number}.log"
|
|
source = next(iter(sources))
|
|
if destination.exists():
|
|
errors[task_group] = (
|
|
"WORK_LOG archive destination이 이미 존재한다: "
|
|
f"source={source} destination={destination}"
|
|
)
|
|
continue
|
|
try:
|
|
close_unfinished_work_log_attempts(source)
|
|
source.replace(destination)
|
|
except OSError as exc:
|
|
errors[task_group] = (
|
|
f"WORK_LOG archive 실패: source={source} "
|
|
f"destination={destination} error={exc}"
|
|
)
|
|
continue
|
|
if source == active_source:
|
|
for task_name in sorted(task_names, reverse=True):
|
|
if "/" not in task_name:
|
|
continue
|
|
try:
|
|
(workspace / "agent-task" / task_name).rmdir()
|
|
except OSError:
|
|
pass
|
|
try:
|
|
active_source.parent.rmdir()
|
|
except OSError:
|
|
pass
|
|
archived[task_group] = str(destination.resolve())
|
|
return archived, errors
|
|
|
|
|
|
def task_attempt_log_directories(runs: Path, task_name: str) -> list[Path]:
|
|
"""Return dispatcher-owned attempt directories whose locator names the task."""
|
|
matches: list[Path] = []
|
|
if not runs.is_dir():
|
|
return matches
|
|
for attempt_dir in runs.iterdir():
|
|
if not attempt_dir.is_dir():
|
|
continue
|
|
locator = attempt_dir / "locator.json"
|
|
try:
|
|
record = json.loads(locator.read_text(encoding="utf-8"))
|
|
except (OSError, json.JSONDecodeError):
|
|
continue
|
|
if record.get("task") != task_name:
|
|
continue
|
|
matches.append(attempt_dir)
|
|
return matches
|
|
|
|
|
|
def cleanup_completed_task_attempt_logs(runs: Path, task_name: str) -> int:
|
|
"""Remove only dispatcher-owned logs for a task after its complete archive exists."""
|
|
removed = 0
|
|
for attempt_dir in task_attempt_log_directories(runs, task_name):
|
|
try:
|
|
shutil.rmtree(attempt_dir)
|
|
except OSError as exc:
|
|
print(
|
|
f"[attempt-log-cleanup-warning] task={task_name} "
|
|
f"path={attempt_dir} error={exc}",
|
|
flush=True,
|
|
)
|
|
continue
|
|
removed += 1
|
|
return removed
|
|
|
|
|
|
def review_fingerprints(workspace: Path, task: Task) -> set[tuple[str, str]]:
|
|
directories = [task.directory] if task.directory.is_dir() else []
|
|
directories.extend(matching_archive_directories(workspace, task))
|
|
fingerprints: set[tuple[str, str]] = set()
|
|
for directory in directories:
|
|
for path in directory.iterdir():
|
|
if path.is_file() and (path.name.startswith("code_review_") or REVIEW_RE.match(path.name)):
|
|
fingerprints.add((str(path.resolve()), sha256_file(path)))
|
|
return fingerprints
|
|
|
|
|
|
def review_outcome(
|
|
workspace: Path, task: Task, prior_fingerprints: set[tuple[str, str]]
|
|
) -> dict[str, str]:
|
|
if task.directory.is_dir():
|
|
directories = [task.directory]
|
|
archives: list[Path] = []
|
|
else:
|
|
archives = matching_archive_directories(workspace, task)
|
|
directories = list(archives)
|
|
newest_directory: Path | None = None
|
|
newest_log: Path | None = None
|
|
newest_mtime = -1
|
|
for directory in directories:
|
|
logs = list(directory.glob("code_review_*.log"))
|
|
logs.extend(path for path in directory.iterdir() if path.is_file() and REVIEW_RE.match(path.name))
|
|
for log in logs:
|
|
mtime = log.stat().st_mtime_ns
|
|
fingerprint = (str(log.resolve()), sha256_file(log))
|
|
if fingerprint not in prior_fingerprints and mtime > newest_mtime and read_verdict(log):
|
|
newest_directory = directory
|
|
newest_log = log
|
|
newest_mtime = mtime
|
|
verdict = read_verdict(newest_log) if newest_log else "UNKNOWN"
|
|
if newest_directory in archives:
|
|
state = "archived"
|
|
elif newest_directory and (newest_directory / "USER_REVIEW.md").exists():
|
|
state = "user-review"
|
|
elif newest_directory and (newest_directory / "complete.log").exists():
|
|
state = "complete-finalization"
|
|
elif newest_log and REVIEW_RE.match(newest_log.name):
|
|
state = "finalization-pending"
|
|
elif newest_directory and any(REVIEW_RE.match(path.name) for path in newest_directory.iterdir() if path.is_file()):
|
|
state = "follow-up"
|
|
else:
|
|
state = "changed"
|
|
return {
|
|
"verdict": verdict or "UNKNOWN",
|
|
"state": state,
|
|
"path": str(newest_directory or task.directory),
|
|
"review_log": str(newest_log) if newest_log else "unknown",
|
|
}
|
|
|
|
|
|
async def run_worker(
|
|
workspace: Path,
|
|
store: StateStore,
|
|
task: Task,
|
|
semaphores: dict[str, asyncio.Semaphore],
|
|
resume_locator: Path | None = None,
|
|
) -> None:
|
|
spec = route_agent(task)
|
|
resource = f"pi:{spec.model}" if spec.cli == "pi" else spec.cli
|
|
work_log = milestone_work_log_path(task)
|
|
banner(
|
|
"작업시작",
|
|
task.name,
|
|
[
|
|
f"model={spec.display}",
|
|
f"plan={task.plan.resolve()}",
|
|
f"work_log={work_log.resolve()}",
|
|
],
|
|
)
|
|
success, locator = await run_escalating(
|
|
workspace,
|
|
store,
|
|
task,
|
|
"worker",
|
|
spec,
|
|
invocation_semaphore=semaphores[resource],
|
|
initial_resume_locator=resume_locator,
|
|
)
|
|
if not success:
|
|
current = store.task_state(task).get("blocked")
|
|
store.update_task(
|
|
task, blocked=current or f"worker failure locator={locator}"
|
|
)
|
|
return
|
|
store.update_task(
|
|
task,
|
|
worker_done=True,
|
|
worker_cli=spec.cli,
|
|
worker_model=spec.model,
|
|
selfcheck_done=not spec.local_pi,
|
|
blocked=None,
|
|
)
|
|
|
|
|
|
async def run_selfcheck(
|
|
workspace: Path,
|
|
store: StateStore,
|
|
task: Task,
|
|
semaphores: dict[str, asyncio.Semaphore],
|
|
resume_locator: Path | None = None,
|
|
) -> None:
|
|
spec = route_agent(task)
|
|
if not spec.local_pi:
|
|
raise RuntimeError("Pi가 아닌 route에 selfcheck stage가 배정됐다")
|
|
resource = f"pi:{spec.model}"
|
|
work_log = milestone_work_log_path(task)
|
|
banner(
|
|
"자가검증시작",
|
|
task.name,
|
|
[
|
|
f"model={spec.display}",
|
|
f"plan={task.plan.resolve()}",
|
|
f"work_log={work_log.resolve()}",
|
|
],
|
|
)
|
|
retry = 0
|
|
if isinstance(store, StateStore):
|
|
retry = int(store.task_state(task).get("selfcheck_incomplete", 0))
|
|
if retry >= SELF_CHECK_INCOMPLETE_LIMIT:
|
|
locator = resume_locator
|
|
reason = (
|
|
"selfcheck implementation field limit already exhausted: "
|
|
f"{retry}/{SELF_CHECK_INCOMPLETE_LIMIT}"
|
|
)
|
|
store.update_task(task, blocked=f"{reason} locator={locator}")
|
|
banner(
|
|
"작업차단",
|
|
task.name,
|
|
[
|
|
"reason=selfcheck-incomplete-limit",
|
|
f"retry={retry}/{SELF_CHECK_INCOMPLETE_LIMIT}",
|
|
f"locator={locator}",
|
|
],
|
|
)
|
|
return
|
|
while True:
|
|
success, locator = await run_escalating(
|
|
workspace,
|
|
store,
|
|
task,
|
|
"selfcheck",
|
|
spec,
|
|
invocation_semaphore=semaphores[resource],
|
|
initial_resume_locator=resume_locator,
|
|
)
|
|
resume_locator = None
|
|
if not success:
|
|
current = store.task_state(task).get("blocked")
|
|
store.update_task(
|
|
task, blocked=current or f"selfcheck failure locator={locator}"
|
|
)
|
|
return
|
|
errors = implementation_review_errors(task)
|
|
if not errors:
|
|
break
|
|
retry += 1
|
|
if retry >= SELF_CHECK_INCOMPLETE_LIMIT:
|
|
reason = (
|
|
"selfcheck implementation fields remain incomplete: "
|
|
f"{retry}/{SELF_CHECK_INCOMPLETE_LIMIT}"
|
|
)
|
|
store.update_task(
|
|
task,
|
|
blocked=f"{reason} locator={locator}",
|
|
selfcheck_incomplete=retry,
|
|
)
|
|
banner(
|
|
"작업차단",
|
|
task.name,
|
|
[
|
|
"reason=selfcheck-incomplete-limit",
|
|
f"detail={'; '.join(errors)}",
|
|
f"retry={retry}/{SELF_CHECK_INCOMPLETE_LIMIT}",
|
|
f"locator={locator}",
|
|
],
|
|
)
|
|
return
|
|
store.update_task(task, selfcheck_incomplete=retry)
|
|
banner(
|
|
"자가검증재시도",
|
|
task.name,
|
|
[
|
|
f"reason={'; '.join(errors)}",
|
|
f"retry={retry}/{SELF_CHECK_INCOMPLETE_LIMIT}",
|
|
f"locator={locator}",
|
|
],
|
|
)
|
|
store.update_task(
|
|
task,
|
|
selfcheck_done=True,
|
|
selfcheck_incomplete=0,
|
|
blocked=None,
|
|
)
|
|
|
|
|
|
async def run_review(
|
|
workspace: Path,
|
|
store: StateStore,
|
|
task: Task,
|
|
) -> str | None:
|
|
spec = AgentSpec("codex", "gpt-5.6-sol", "codex/gpt-5.6-sol xhigh")
|
|
state = store.task_state(task)
|
|
prior_no_progress = int(state.get("review_no_progress", 0))
|
|
if prior_no_progress >= REVIEW_NO_PROGRESS_LIMIT:
|
|
locator = state.get("active_locator")
|
|
reason = (
|
|
"review no-progress limit already exhausted: "
|
|
f"{prior_no_progress}/{REVIEW_NO_PROGRESS_LIMIT}"
|
|
)
|
|
store.update_task(task, blocked=f"{reason} locator={locator}")
|
|
banner(
|
|
"작업차단",
|
|
task.name,
|
|
[
|
|
"reason=review-no-progress-limit",
|
|
f"unchanged_review_attempts={prior_no_progress}/{REVIEW_NO_PROGRESS_LIMIT}",
|
|
f"locator={locator}",
|
|
],
|
|
)
|
|
return None
|
|
before = task_signature(workspace, task)
|
|
prior_review_fingerprints = review_fingerprints(workspace, task)
|
|
target = task.review.resolve() if task.review else task.directory.resolve()
|
|
banner("리뷰시작", task.name, [f"model={spec.display}", f"review={target}"])
|
|
success, locator = await run_escalating(workspace, store, task, "review", spec)
|
|
if not success:
|
|
current = store.task_state(task).get("blocked")
|
|
store.update_task(
|
|
task, blocked=current or f"review failure locator={locator}"
|
|
)
|
|
return None
|
|
after = task_signature(workspace, task)
|
|
if before == after:
|
|
state = store.task_state(task)
|
|
count = int(state.get("review_no_progress", 0)) + 1
|
|
if count >= REVIEW_NO_PROGRESS_LIMIT:
|
|
reason = (
|
|
"review made no progress: "
|
|
f"{count}/{REVIEW_NO_PROGRESS_LIMIT} locator={locator}"
|
|
)
|
|
store.update_task(
|
|
task,
|
|
review_no_progress=count,
|
|
blocked=reason,
|
|
)
|
|
banner(
|
|
"작업차단",
|
|
task.name,
|
|
[
|
|
"reason=review-no-progress-limit",
|
|
f"unchanged_review_attempts={count}/{REVIEW_NO_PROGRESS_LIMIT}",
|
|
f"locator={locator}",
|
|
],
|
|
)
|
|
return None
|
|
store.update_task(task, review_no_progress=count)
|
|
banner(
|
|
"루프정체경고",
|
|
task.name,
|
|
[
|
|
f"unchanged_review_attempts={count}/{REVIEW_NO_PROGRESS_LIMIT}",
|
|
f"locator={locator}",
|
|
],
|
|
)
|
|
await asyncio.sleep(min(30, count * 5))
|
|
else:
|
|
store.update_task(task, review_no_progress=0, blocked=None)
|
|
outcome = review_outcome(workspace, task, prior_review_fingerprints)
|
|
banner(
|
|
"리뷰결과",
|
|
task.name,
|
|
[
|
|
f"verdict={outcome['verdict']}",
|
|
f"state={outcome['state']}",
|
|
f"path={outcome['path']}",
|
|
f"review_log={outcome['review_log']}",
|
|
f"locator={locator}",
|
|
],
|
|
)
|
|
if outcome["verdict"] == "PASS" and outcome["state"] == "archived":
|
|
banner("작업완료", task.name, [f"archive={outcome['path']}", f"locator={locator}"])
|
|
return outcome["path"]
|
|
if outcome["verdict"] == "UNKNOWN" or outcome["state"] == "changed":
|
|
raise RuntimeError(
|
|
"official review가 판정과 다음 파일 상태를 materialize하지 않았다; "
|
|
f"locator={locator}"
|
|
)
|
|
if outcome["state"] == "archived":
|
|
raise RuntimeError(
|
|
f"PASS가 아닌 review가 완료 archive로 이동했다: "
|
|
f"verdict={outcome['verdict']} locator={locator}"
|
|
)
|
|
return None
|
|
|
|
|
|
def status_lines(task: Task, stage: str, dependency: str) -> list[str]:
|
|
route = f"{task.lane}-G{task.grade:02d}" if task.lane and task.grade else "recovery"
|
|
return [f"stage={stage}", f"route={route}", f"dependency={dependency}"]
|
|
|
|
|
|
def select_dispatch_candidates(
|
|
ready: list[tuple[Task, str]],
|
|
) -> tuple[list[tuple[Task, str]], list[tuple[Task, str]], str]:
|
|
ready_reviews = [(task, stage) for task, stage in ready if stage == "review"]
|
|
ready_workers = [(task, stage) for task, stage in ready if stage in {"worker", "selfcheck"}]
|
|
# Each scheduler pass admits every dependency-ready task. The running map
|
|
# prevents only duplicate attempts for the same task; phases and write-sets
|
|
# never impose a cross-task barrier.
|
|
return ready_reviews + ready_workers, [], ""
|
|
|
|
|
|
def ensure_review_shared_state(workspace: Path) -> None:
|
|
helper = workspace / "agent-ops" / "bin" / "ai-ignore.sh"
|
|
if not helper.is_file():
|
|
raise RuntimeError(f"review shared-state helper가 없다: {helper}")
|
|
command = [
|
|
"bash",
|
|
"-c",
|
|
'source "$1" && agent_ops_ensure_gitignore_task_artifact_block "$2"',
|
|
"agent-task-review-preflight",
|
|
str(helper),
|
|
str(workspace / ".gitignore"),
|
|
]
|
|
completed = subprocess.run(
|
|
command,
|
|
cwd=workspace,
|
|
capture_output=True,
|
|
text=True,
|
|
check=False,
|
|
)
|
|
if completed.returncode != 0:
|
|
diagnostic = (completed.stderr or completed.stdout).strip()
|
|
raise RuntimeError(
|
|
f"review shared-state preflight 실패: {diagnostic or completed.returncode}"
|
|
)
|
|
|
|
|
|
async def dispatch(args: argparse.Namespace) -> int:
|
|
workspace = Path(args.workspace).resolve()
|
|
store = StateStore(workspace)
|
|
try:
|
|
try:
|
|
return await dispatch_with_store(args, workspace, store)
|
|
except Exception as exc:
|
|
# A scheduler/control-plane exception must not make asyncio.run()
|
|
# cancel already-running agent attempts. Keep this loop alive until
|
|
# every owned background task finishes naturally; the next
|
|
# dispatcher run reconciles their file/state results.
|
|
current = asyncio.current_task()
|
|
active = [
|
|
task
|
|
for task in asyncio.all_tasks()
|
|
if task is not current and not task.done()
|
|
]
|
|
if active:
|
|
banner(
|
|
"디스패처복구대기",
|
|
args.task_group or "agent-task",
|
|
[
|
|
f"running_async_tasks={len(active)}",
|
|
"scheduler 예외와 무관하게 실행 중 agent를 자연 종료까지 추적",
|
|
],
|
|
)
|
|
await asyncio.gather(*active, return_exceptions=True)
|
|
raise DispatcherInterruptedWithActiveWork(
|
|
f"running agent가 있던 scheduler 예외: {exc}"
|
|
) from exc
|
|
raise
|
|
finally:
|
|
store.close()
|
|
|
|
|
|
async def dispatch_with_store(
|
|
args: argparse.Namespace,
|
|
workspace: Path,
|
|
store: StateStore,
|
|
) -> int:
|
|
orchestration_scope = args.task_group or "__all__"
|
|
if args.retry_blocked and not args.dry_run:
|
|
store.clear_blocked(args.task_group)
|
|
semaphores = {
|
|
"pi:ornith-fast": asyncio.Semaphore(3),
|
|
"pi:laguna-s:2.1": asyncio.Semaphore(2),
|
|
"agy": asyncio.Semaphore(1),
|
|
"claude": asyncio.Semaphore(64),
|
|
"codex": asyncio.Semaphore(64),
|
|
}
|
|
running: dict[str, asyncio.Task[str | None]] = {}
|
|
last_wait: dict[str, str] = {}
|
|
completed_tasks: dict[str, str] = {}
|
|
fatal_errors: dict[str, str] = {}
|
|
control_plane_errors: dict[str, str] = {}
|
|
work_log_archive_errors: dict[str, str] = {}
|
|
review_shared_state_ready = False
|
|
candidate_scope: set[str] | None = None
|
|
task_cache: dict[str, Task] | None = None
|
|
resume_locators: dict[str, Path] = {}
|
|
live_external_processes: dict[str, str] = {}
|
|
|
|
while True:
|
|
if task_cache is None:
|
|
tasks = scan_tasks(workspace, args.task_group)
|
|
task_cache = {task.name: task for task in tasks}
|
|
else:
|
|
tasks = sorted(task_cache.values(), key=lambda task: (task.index, task.name))
|
|
if args.dry_run:
|
|
persistent_errors: dict[str, str] = {}
|
|
observed_tasks: set[str] = set()
|
|
live_external_processes = {}
|
|
else:
|
|
store.prepare_orchestration(orchestration_scope, tasks, workspace)
|
|
live_external_processes = orchestration_live_agent_processes(
|
|
store,
|
|
orchestration_scope,
|
|
)
|
|
active_or_running = (
|
|
{task.name for task in tasks}
|
|
| set(running)
|
|
| set(live_external_processes)
|
|
)
|
|
reconciled_completed, persistent_errors = store.reconcile_orchestration(
|
|
orchestration_scope,
|
|
workspace,
|
|
active_or_running,
|
|
)
|
|
completed_tasks.update(reconciled_completed)
|
|
for task_name in persistent_errors:
|
|
completed_tasks.pop(task_name, None)
|
|
observed_tasks = store.orchestration_tasks(orchestration_scope)
|
|
work_log_archives, work_log_archive_errors = (
|
|
archive_completed_group_work_logs(
|
|
workspace,
|
|
observed_tasks,
|
|
completed_tasks,
|
|
active_or_running,
|
|
)
|
|
)
|
|
for task_group, archive in sorted(work_log_archives.items()):
|
|
banner(
|
|
"작업로그아카이브",
|
|
task_group,
|
|
[f"archive={archive}"],
|
|
)
|
|
if not tasks and not running:
|
|
if live_external_processes:
|
|
for task_name, detail in sorted(
|
|
live_external_processes.items()
|
|
):
|
|
banner(
|
|
"작업수행중",
|
|
task_name,
|
|
[
|
|
"이전 dispatcher의 model process를 종료시키지 않고 추적",
|
|
detail,
|
|
],
|
|
)
|
|
await asyncio.sleep(STREAM_HEARTBEAT_SECONDS)
|
|
continue
|
|
if control_plane_errors:
|
|
banner(
|
|
"디스패치추적대기",
|
|
args.task_group or "agent-task",
|
|
[
|
|
"예상하지 못한 dispatcher 중단 결과를 재조정해야 함",
|
|
*(
|
|
f"interrupted[{name}]={reason}"
|
|
for name, reason in sorted(control_plane_errors.items())
|
|
),
|
|
],
|
|
)
|
|
return 3
|
|
if work_log_archive_errors:
|
|
banner(
|
|
"디스패치추적대기",
|
|
args.task_group or "agent-task",
|
|
[
|
|
"완료 task group의 WORK_LOG archive를 재시도해야 함",
|
|
*(
|
|
f"work-log-archive[{group}]={reason}"
|
|
for group, reason in sorted(
|
|
work_log_archive_errors.items()
|
|
)
|
|
),
|
|
],
|
|
)
|
|
return 3
|
|
if args.task_group and not observed_tasks and not completed_tasks:
|
|
reason = (
|
|
"명시한 task group에서 관찰된 active task나 "
|
|
"검증된 complete.log 이력이 없다"
|
|
)
|
|
if not args.dry_run:
|
|
store.mark_orchestration_blocked(orchestration_scope, {})
|
|
banner(
|
|
"디스패치차단",
|
|
args.task_group,
|
|
[f"reason=unobserved-task-group", reason],
|
|
)
|
|
return 2
|
|
pending_attempt_logs = {
|
|
name: paths
|
|
for name in completed_tasks
|
|
if (
|
|
paths := task_attempt_log_directories(
|
|
store.runs,
|
|
name,
|
|
)
|
|
)
|
|
}
|
|
if pending_attempt_logs:
|
|
banner(
|
|
"디스패치추적대기",
|
|
args.task_group or "agent-task",
|
|
[
|
|
"완료 task의 attempt 로그 정리가 아직 끝나지 않음",
|
|
*(
|
|
f"attempt-log-cleanup-pending[{name}]="
|
|
+ ",".join(str(path) for path in paths)
|
|
for name, paths in sorted(pending_attempt_logs.items())
|
|
),
|
|
],
|
|
)
|
|
return 3
|
|
incomplete = sorted(observed_tasks - completed_tasks.keys())
|
|
if incomplete or fatal_errors or persistent_errors:
|
|
details = [
|
|
*(f"incomplete={name}" for name in incomplete),
|
|
*(
|
|
f"persistent[{name}]={reason}"
|
|
for name, reason in sorted(persistent_errors.items())
|
|
),
|
|
*(f"error[{name}]={reason}" for name, reason in sorted(fatal_errors.items())),
|
|
]
|
|
if not args.dry_run:
|
|
store.mark_orchestration_blocked(
|
|
orchestration_scope,
|
|
{
|
|
name: (
|
|
"blocked",
|
|
persistent_errors.get(name)
|
|
or fatal_errors.get(name)
|
|
or "관찰된 task가 완료되지 않았다",
|
|
)
|
|
for name in (
|
|
set(incomplete)
|
|
| set(fatal_errors)
|
|
| set(persistent_errors)
|
|
)
|
|
},
|
|
)
|
|
banner("디스패치차단", args.task_group or "agent-task", details)
|
|
return 2
|
|
if not args.dry_run:
|
|
store.mark_orchestration_complete(orchestration_scope)
|
|
banner(
|
|
"작업완료",
|
|
args.task_group or "agent-task",
|
|
[
|
|
"active task 없음",
|
|
f"verified_complete_tasks={len(completed_tasks)}",
|
|
*(f"complete[{name}]={path}" for name, path in sorted(completed_tasks.items())),
|
|
],
|
|
)
|
|
return 0
|
|
|
|
task_by_name = {task.name: task for task in tasks}
|
|
finished_names: set[str] = set()
|
|
complete_log_created = False
|
|
for name, future in list(running.items()):
|
|
if not future.done():
|
|
continue
|
|
finished_names.add(name)
|
|
completed_archive: str | None = None
|
|
try:
|
|
completed_archive = future.result()
|
|
if completed_archive:
|
|
complete_log_created = True
|
|
store.mark_orchestration_task_complete(
|
|
orchestration_scope, name, completed_archive
|
|
)
|
|
completed_tasks[name] = completed_archive
|
|
except Exception as exc: # keep other independent tasks alive
|
|
banner(
|
|
"디스패치추적대기",
|
|
name,
|
|
[f"agent coroutine exception={exc}"],
|
|
)
|
|
control_plane_errors[name] = str(exc)
|
|
task = task_by_name.get(name)
|
|
if task:
|
|
store.clear_active(task)
|
|
if not completed_archive:
|
|
refreshed = read_task_directory(workspace, task.directory)
|
|
if refreshed is None:
|
|
task_cache.pop(name, None)
|
|
else:
|
|
task_cache[name] = refreshed
|
|
if completed_archive:
|
|
task_cache.pop(name, None)
|
|
del running[name]
|
|
if finished_names:
|
|
if complete_log_created:
|
|
# Running reviewers may be archiving their own active directory
|
|
# while this completion-triggered scan runs. Preserve their
|
|
# already-loaded Task snapshots and do not reread those mutable
|
|
# directories until their futures finish.
|
|
running_snapshots = {
|
|
name: task_cache[name]
|
|
for name in running
|
|
if name in task_cache
|
|
}
|
|
tasks = scan_tasks(
|
|
workspace,
|
|
args.task_group,
|
|
exclude_names=set(running),
|
|
)
|
|
task_cache = {
|
|
**{task.name: task for task in tasks},
|
|
**running_snapshots,
|
|
}
|
|
tasks = sorted(
|
|
task_cache.values(),
|
|
key=lambda task: (task.index, task.name),
|
|
)
|
|
store.prepare_orchestration(orchestration_scope, tasks, workspace)
|
|
live_external_processes = orchestration_live_agent_processes(
|
|
store,
|
|
orchestration_scope,
|
|
)
|
|
active_or_running = (
|
|
{task.name for task in tasks}
|
|
| set(running)
|
|
| set(live_external_processes)
|
|
)
|
|
reconciled_completed, persistent_errors = store.reconcile_orchestration(
|
|
orchestration_scope,
|
|
workspace,
|
|
active_or_running,
|
|
)
|
|
completed_tasks.update(reconciled_completed)
|
|
for task_name in persistent_errors:
|
|
completed_tasks.pop(task_name, None)
|
|
observed_tasks = store.orchestration_tasks(orchestration_scope)
|
|
work_log_archives, work_log_archive_errors = (
|
|
archive_completed_group_work_logs(
|
|
workspace,
|
|
observed_tasks,
|
|
completed_tasks,
|
|
active_or_running,
|
|
)
|
|
)
|
|
for task_group, archive in sorted(
|
|
work_log_archives.items()
|
|
):
|
|
banner(
|
|
"작업로그아카이브",
|
|
task_group,
|
|
[f"archive={archive}"],
|
|
)
|
|
candidate_scope = None
|
|
else:
|
|
tasks = sorted(task_cache.values(), key=lambda task: (task.index, task.name))
|
|
candidate_scope = finished_names
|
|
|
|
if not tasks and not running:
|
|
# A completion-triggered full scan may have removed the last active task.
|
|
continue
|
|
|
|
ready: list[tuple[Task, str]] = []
|
|
waiting_tasks: list[str] = []
|
|
externally_active: list[tuple[Task, str]] = []
|
|
blocked_details: dict[str, tuple[str, str, str]] = {}
|
|
for task in tasks:
|
|
# A running future owns this task directory. Re-reading its review
|
|
# or dependency files can race with review finalization/archive and
|
|
# must never interrupt unrelated tasks.
|
|
if task.name in running:
|
|
continue
|
|
if task.name in control_plane_errors:
|
|
reason = (
|
|
"예상하지 못한 agent coroutine 중단 결과를 "
|
|
"다음 dispatcher가 재조정해야 함: "
|
|
f"{control_plane_errors[task.name]}"
|
|
)
|
|
blocked_details[task.name] = (
|
|
"디스패치추적대기",
|
|
"interrupted",
|
|
reason,
|
|
)
|
|
waiting_tasks.append(task.name)
|
|
continue
|
|
state = store.peek_task_state(task) if args.dry_run else store.task_state(task)
|
|
active_predecessors = live_predecessors(
|
|
task,
|
|
set(running) | set(live_external_processes),
|
|
)
|
|
if active_predecessors:
|
|
dependency_ready = False
|
|
dependency = (
|
|
"predecessor FINISH 대기: "
|
|
+ ",".join(active_predecessors)
|
|
)
|
|
else:
|
|
dependency_ready, dependency = dependency_state(
|
|
workspace,
|
|
task,
|
|
)
|
|
stage = task_stage(task, state)
|
|
if state.get("active_stage"):
|
|
active_stage = str(state["active_stage"])
|
|
active_live, active_detail = external_active_is_live(state)
|
|
if active_live:
|
|
reason = f"외부 실행중: stage={active_stage}; {active_detail}"
|
|
active_key = (
|
|
f"active|{active_stage}|{state.get('active_locator') or 'unknown'}"
|
|
)
|
|
externally_active.append((task, active_stage))
|
|
blocked_details[task.name] = ("작업중", stage, reason)
|
|
if not args.dry_run and last_wait.get(task.name) != active_key:
|
|
banner("작업중", task.name, status_lines(task, stage, reason))
|
|
last_wait[task.name] = active_key
|
|
continue
|
|
if not args.dry_run:
|
|
resume_locator = laguna_resume_locator(state)
|
|
if resume_locator is not None:
|
|
resume_locators[task.name] = resume_locator
|
|
banner(
|
|
"작업복구",
|
|
task.name,
|
|
status_lines(task, stage, f"stale active 제외: {active_detail}"),
|
|
)
|
|
store.clear_active(task)
|
|
state = store.task_state(task)
|
|
reason = ""
|
|
if task.errors:
|
|
reason = "; ".join(task.errors)
|
|
elif task.user_review:
|
|
blocking, detail = user_review_blocker_state(task.user_review)
|
|
if task.plan is not None or task.review is not None:
|
|
blocking = False
|
|
detail = "active PLAN/CODE_REVIEW와 공존한다"
|
|
reason = (
|
|
f"USER_REVIEW 대기: {task.user_review}; {detail}"
|
|
if blocking
|
|
else (
|
|
"USER_REVIEW stop 계약 불충족: "
|
|
f"{task.user_review}; {detail}"
|
|
)
|
|
)
|
|
elif state.get("blocked"):
|
|
reason = str(state["blocked"])
|
|
elif not dependency_ready:
|
|
reason = dependency
|
|
if reason:
|
|
wait_key = f"{stage}|{reason}"
|
|
event = (
|
|
"작업차단"
|
|
if stage in {"blocked", "user-review"} or state.get("blocked")
|
|
else "작업대기"
|
|
)
|
|
blocked_details[task.name] = (event, stage, reason)
|
|
if not args.dry_run and last_wait.get(task.name) != wait_key:
|
|
banner(event, task.name, status_lines(task, stage, reason))
|
|
last_wait[task.name] = wait_key
|
|
waiting_tasks.append(task.name)
|
|
continue
|
|
if task.name not in running and (
|
|
candidate_scope is None or task.name in candidate_scope
|
|
):
|
|
ready.append((task, stage))
|
|
|
|
if args.dry_run:
|
|
ready_by_name = {task.name: stage for task, stage in ready}
|
|
for task in tasks:
|
|
if task.name in ready_by_name:
|
|
stage = ready_by_name[task.name]
|
|
spec = (
|
|
AgentSpec("codex", "gpt-5.6-sol", "codex/gpt-5.6-sol xhigh")
|
|
if stage == "review"
|
|
else route_agent(task)
|
|
)
|
|
banner("작업대기", task.name, status_lines(task, stage, "ready") + [f"model={spec.display}"])
|
|
else:
|
|
event, stage, reason = blocked_details[task.name]
|
|
banner(event, task.name, status_lines(task, stage, reason))
|
|
return 2 if waiting_tasks and not ready else 0
|
|
|
|
candidates, deferred, phase_wait_reason = select_dispatch_candidates(ready)
|
|
for task, stage in deferred:
|
|
wait_key = f"{stage}|{phase_wait_reason}"
|
|
if last_wait.get(task.name) != wait_key:
|
|
banner("작업대기", task.name, status_lines(task, stage, phase_wait_reason))
|
|
last_wait[task.name] = wait_key
|
|
|
|
if (
|
|
not review_shared_state_ready
|
|
and any(stage == "review" for _, stage in candidates)
|
|
):
|
|
try:
|
|
ensure_review_shared_state(workspace)
|
|
except (OSError, RuntimeError) as exc:
|
|
# Shared review setup is a blocker only for reviews. It must
|
|
# not prevent dependency-independent workers/selfchecks from
|
|
# starting and draining in the same scheduler pass.
|
|
remaining_candidates: list[tuple[Task, str]] = []
|
|
for task, stage in candidates:
|
|
if stage != "review":
|
|
remaining_candidates.append((task, stage))
|
|
continue
|
|
reason = f"review shared-state preflight failed: {exc}"
|
|
store.update_task(task, blocked=reason)
|
|
fatal_errors[task.name] = reason
|
|
waiting_tasks.append(task.name)
|
|
blocked_details[task.name] = (
|
|
"작업차단",
|
|
stage,
|
|
reason,
|
|
)
|
|
banner(
|
|
"작업차단",
|
|
task.name,
|
|
status_lines(task, stage, reason),
|
|
)
|
|
candidates = remaining_candidates
|
|
else:
|
|
review_shared_state_ready = True
|
|
|
|
scheduled = False
|
|
for task, stage in candidates:
|
|
store.mark_active(task, stage)
|
|
resume_locator = resume_locators.pop(task.name, None)
|
|
if stage == "review":
|
|
future = asyncio.create_task(run_review(workspace, store, task))
|
|
elif stage == "selfcheck":
|
|
future = asyncio.create_task(
|
|
run_selfcheck(
|
|
workspace,
|
|
store,
|
|
task,
|
|
semaphores,
|
|
**(
|
|
{"resume_locator": resume_locator}
|
|
if resume_locator is not None
|
|
else {}
|
|
),
|
|
)
|
|
)
|
|
else:
|
|
future = asyncio.create_task(
|
|
run_worker(
|
|
workspace,
|
|
store,
|
|
task,
|
|
semaphores,
|
|
**(
|
|
{"resume_locator": resume_locator}
|
|
if resume_locator is not None
|
|
else {}
|
|
),
|
|
)
|
|
)
|
|
running[task.name] = future
|
|
last_wait.pop(task.name, None)
|
|
scheduled = True
|
|
|
|
if running:
|
|
await asyncio.wait(running.values(), return_when=asyncio.FIRST_COMPLETED)
|
|
continue
|
|
if work_log_archive_errors:
|
|
banner(
|
|
"디스패치추적대기",
|
|
args.task_group or "agent-task",
|
|
[
|
|
"완료 task group의 WORK_LOG archive를 재시도해야 함",
|
|
*(
|
|
f"work-log-archive[{group}]={reason}"
|
|
for group, reason in sorted(
|
|
work_log_archive_errors.items()
|
|
)
|
|
),
|
|
],
|
|
)
|
|
return 3
|
|
if externally_active:
|
|
banner(
|
|
"디스패치추적대기",
|
|
args.task_group or "agent-task",
|
|
["새 실행 후보 없음", "active task는 caller가 계속 추적"],
|
|
)
|
|
return 3
|
|
if control_plane_errors:
|
|
banner(
|
|
"디스패치추적대기",
|
|
args.task_group or "agent-task",
|
|
[
|
|
"실행 중이던 독립 작업을 모두 소진했고 재조정이 필요함",
|
|
*(
|
|
f"interrupted[{name}]={reason}"
|
|
for name, reason in sorted(control_plane_errors.items())
|
|
),
|
|
],
|
|
)
|
|
return 3
|
|
if not scheduled:
|
|
store.mark_orchestration_blocked(
|
|
orchestration_scope,
|
|
{
|
|
name: (
|
|
"blocked" if event == "작업차단" else "waiting",
|
|
detail,
|
|
)
|
|
for name, (event, _, detail) in blocked_details.items()
|
|
},
|
|
)
|
|
reason = f"waiting={','.join(sorted(waiting_tasks))}"
|
|
banner(
|
|
"디스패치차단",
|
|
args.task_group or "agent-task",
|
|
[
|
|
"실행 가능한 독립 작업을 모두 소진함",
|
|
reason,
|
|
f"verified_complete_tasks={len(completed_tasks)}",
|
|
*(
|
|
f"complete[{name}]={path}"
|
|
for name, path in sorted(completed_tasks.items())
|
|
),
|
|
*(
|
|
f"{name}: stage={stage}; reason={detail}"
|
|
for name, (_, stage, detail) in sorted(
|
|
blocked_details.items()
|
|
)
|
|
),
|
|
],
|
|
)
|
|
return 2
|
|
|
|
|
|
def parse_args() -> argparse.Namespace:
|
|
parser = argparse.ArgumentParser(description=__doc__)
|
|
parser.add_argument("--workspace", default=".", help="repository root (default: current directory)")
|
|
parser.add_argument("--task-group", help="run only agent-task/<task_group>")
|
|
parser.add_argument("--dry-run", action="store_true", help="classify and print without launching CLIs")
|
|
parser.add_argument("--retry-blocked", action="store_true", help="clear dispatcher-local blocked state")
|
|
return parser.parse_args()
|
|
|
|
|
|
def main() -> int:
|
|
args = parse_args()
|
|
try:
|
|
return asyncio.run(dispatch(args))
|
|
except KeyboardInterrupt:
|
|
print("\n중단됨", file=sys.stderr)
|
|
return 130
|
|
except DispatcherAlreadyRunning as exc:
|
|
print(f"dispatcher active: {exc}", file=sys.stderr)
|
|
return 3
|
|
except DispatcherTerminalStateError as exc:
|
|
print(f"dispatcher error: {exc}", file=sys.stderr)
|
|
return 2
|
|
except Exception as exc:
|
|
# An unexpected dispatcher failure is not proof that the task group is
|
|
# drained or terminal. The caller must inspect active PIDs/locators and
|
|
# recover instead of treating it like exit 2.
|
|
print(f"dispatcher interrupted: {exc}", file=sys.stderr)
|
|
return 3
|
|
|
|
|
|
if __name__ == "__main__":
|
|
raise SystemExit(main())
|