Files
codex-mmo/libexec/mmo_diagnostics.py
T

978 lines
40 KiB
Python
Raw Normal View History

2026-08-24 08:11:59 -07:00
#!/usr/bin/env python3
"""Profile validation, environment diagnostics, and live smoke workflows."""
from __future__ import annotations
import contextlib
import json
import os
import select
import shutil
import subprocess
import sys
import textwrap
import time
from collections.abc import Callable
from pathlib import Path
from typing import Any
from mmo_app_server import _flat_mcp_dynamic_tool_name, app_server_protocol_status
from mmo_gateway import dry_run_gateway, ensure_gateway, gateway_models
from mmo_profiles import (
builtin_auth_link_mode,
load_settings,
profile_summary,
resolve_profile,
)
from mmo_runtime import (
cancel_job,
cancel_session,
create_session,
finish_session,
list_jobs,
mark_session_running,
run_root_exec,
spawn_job,
stop_session,
wait_for_jobs,
)
from mmo_snapshot import compile_profile
from mmo_state import TERMINAL_SESSION_STATUSES, root_mcp_capability_environment
from mmo_tool_mcp import (
load_tool_mcp_registry_with_sources,
tool_mcp_readiness,
tool_mcp_registry_root,
)
from mmo_util import (
config_root,
filtered_environment,
install_root,
package_version,
parse_env_file,
read_toml,
state_root,
strict_json_loads,
)
def _root_harness_prompt(prompt: str, execution_mode: str) -> str:
"""Make a bounded harness task terminal under the root's real lifecycle."""
prompt = prompt.strip()
if execution_mode != "goal":
return prompt
return (
prompt + "\n\nHarness lifecycle requirement: this root uses a durable Codex goal. "
"After every requested action and validation is complete, call `update_goal` "
'with `status="complete"` exactly once before the final assistant message. '
"Do not mark the goal complete while required work or descendant integration "
"remains; a final message alone does not terminate an active goal."
)
def tool_mcp_status(resolved: dict[str, Any] | None = None) -> dict[str, Any]:
registry, sources = load_tool_mcp_registry_with_sources()
selected = set(registry)
agents: dict[str, Any] = {}
required_by: dict[str, list[str]] = {server_id: [] for server_id in registry}
if resolved is not None:
selected = set(resolved.get("tool_mcp_servers", {}))
registry = {
server_id: server for server_id, server in resolved.get("tool_mcp_servers", {}).items()
}
for agent_id, agent in resolved["agents"].items():
grants = agent.get("tool_mcp_servers", {})
agents[agent_id] = grants
for server_id, grant in grants.items():
if grant["required"]:
required_by.setdefault(server_id, []).append(agent_id)
servers: dict[str, Any] = {}
for server_id in sorted(selected):
status = tool_mcp_readiness(server_id, registry[server_id], sources=sources)
status["required_by"] = sorted(required_by.get(server_id, []))
servers[server_id] = status
required = (
list(servers.values())
if resolved is None
else [item for item in servers.values() if item["required_by"]]
)
return {
"registry_root": str(tool_mcp_registry_root()),
"servers": servers,
"agents": agents,
"transport_ok": all(bool(item["transport_ready"]) for item in required),
"credentials_ok": all(bool(item["environment_ready"]) for item in required),
"passed": all(bool(item["ready"]) for item in required),
}
def profile_validation_report(profile: str, bindings: dict[str, str]) -> dict[str, Any]:
resolved = resolve_profile(profile, bindings=bindings)
snapshot = compile_profile(profile, bindings=bindings)
routes_valid = True
routes_error = None
routes_path = Path(snapshot["directory"]) / "routes.toml"
if routes_path.is_file():
try:
read_toml(routes_path)
except Exception as exc:
routes_valid = False
routes_error = str(exc)
return {
"valid": routes_valid,
"profile": profile_summary(profile, bindings=bindings),
"snapshot": snapshot["manifest"],
"generated_routes_toml_valid": routes_valid,
"generated_routes_error": routes_error,
"tool_mcp": tool_mcp_status(resolved),
"modalities": {
key: {
"requires": agent["requires_modalities"],
"model": resolved["models"][agent["model"]]["modalities"],
"transport": resolved["routes"][agent["route"]]["transport_modalities"],
}
for key, agent in resolved["agents"].items()
},
}
def _credentials_status(resolved: dict[str, Any]) -> dict[str, Any]:
values = parse_env_file(config_root() / "credentials.env")
values.update({key: value for key, value in os.environ.items() if value})
required: dict[str, bool] = {}
builtin_auth: dict[str, Any] = {}
settings = load_settings()
base_home = Path(str(settings.get("base_codex_home", "~/.codex"))).expanduser()
for key, route in resolved["routes"].items():
credential_envs = route.get("credential_envs", [])
if credential_envs:
label = "/".join(str(item) for item in credential_envs)
required[label] = any(values.get(str(item)) for item in credential_envs)
if route["driver"] == "codex_builtin" and route.get("auth") == "chatgpt":
mode = builtin_auth_link_mode(route, settings)
builtin_auth[key] = {
"base_codex_home": str(base_home),
"auth_json": (base_home / "auth.json").is_file(),
"auth_link_mode": mode,
"file_auth_transferable": mode in {"shared", "copy"}
and (base_home / "auth.json").is_file(),
"note": (
"Codex 0.149 keyring entries are scoped to canonical CODEX_HOME; "
"an isolated generated home requires file-backed auth.json."
),
}
return {"environment_credentials": required, "builtin_auth": builtin_auth}
def _codex_auth_status(binary: str, home: Path) -> dict[str, Any]:
command = [binary, "login", "status"]
try:
result = subprocess.run(
command,
env=filtered_environment(extra={"CODEX_HOME": str(home)}),
text=True,
capture_output=True,
timeout=30,
check=False,
)
except (OSError, subprocess.SubprocessError) as exc:
return {
"passed": False,
"command": command,
"error": f"{type(exc).__name__}: {exc}",
}
return {
"passed": result.returncode == 0,
"command": command,
"exit_code": result.returncode,
"stdout": result.stdout[-2000:],
"stderr": result.stderr[-2000:],
}
def _mcp_handshake(session: dict[str, Any]) -> dict[str, Any]:
env = filtered_environment(
extra={
"MMO_INSTALL_ROOT": str(install_root()),
"MMO_CONFIG_ROOT": str(config_root()),
"MMO_STATE_ROOT": str(state_root()),
**root_mcp_capability_environment(session),
"MMO_PROFILE_SNAPSHOT": session["snapshot_hash"],
"MMO_ALLOWED_ROOT": session["allowed_root"],
}
)
initialize = {
"jsonrpc": "2.0",
"id": 1,
"method": "initialize",
"params": {
"protocolVersion": "2025-06-18",
"capabilities": {},
"clientInfo": {"name": "doctor", "version": package_version()},
},
}
process = subprocess.Popen(
[sys.executable, str(install_root() / "libexec" / "mmo_mcp.py")],
stdin=subprocess.PIPE,
stdout=subprocess.PIPE,
stderr=subprocess.PIPE,
text=True,
env=env,
start_new_session=True,
)
responses: list[Any] = []
stderr = ""
handshake_error = None
try:
if process.stdin is None or process.stdout is None:
raise RuntimeError("MCP doctor probe did not receive its requested pipes")
process.stdin.write(json.dumps(initialize, allow_nan=False) + "\n")
process.stdin.flush()
readable, _writable, _exceptional = select.select([process.stdout], [], [], 15)
if not readable:
raise subprocess.TimeoutExpired(process.args, 15)
initialize_line = process.stdout.readline()
if not initialize_line:
raise RuntimeError("MCP server closed stdout before initialize response")
initialize_response = strict_json_loads(initialize_line)
responses.append(initialize_response)
initialize_result = (
initialize_response.get("result")
if isinstance(initialize_response, dict)
and initialize_response.get("jsonrpc") == "2.0"
and initialize_response.get("id") == 1
and "error" not in initialize_response
else None
)
if (
not isinstance(initialize_result, dict)
or initialize_result.get("protocolVersion") != "2025-06-18"
):
raise RuntimeError("MCP initialize response is invalid")
for message in (
{"jsonrpc": "2.0", "method": "notifications/initialized", "params": {}},
{"jsonrpc": "2.0", "id": 2, "method": "tools/list", "params": {}},
):
process.stdin.write(json.dumps(message, allow_nan=False) + "\n")
process.stdin.flush()
process.stdin.close()
process.stdin = None
stdout, stderr = process.communicate(timeout=15)
for line in stdout.splitlines():
with contextlib.suppress(json.JSONDecodeError, ValueError):
responses.append(strict_json_loads(line))
except Exception as exc:
handshake_error = f"{type(exc).__name__}: {exc}"
with contextlib.suppress(OSError):
if process.stdin is not None:
process.stdin.close()
process.stdin = None
if process.poll() is None:
process.kill()
with contextlib.suppress(subprocess.SubprocessError, OSError):
_stdout, stderr = process.communicate(timeout=2)
tools = []
for response in responses:
if not isinstance(response, dict):
continue
if response.get("id") == 2:
result_value = response.get("result", {})
if isinstance(result_value, dict):
tools = [
item["name"]
for item in result_value.get("tools", [])
if isinstance(item, dict) and isinstance(item.get("name"), str)
]
return {
"passed": handshake_error is None and process.returncode == 0 and "agent_status" in tools,
"exit_code": process.returncode,
"tools": tools,
"initialize": responses[0] if responses else None,
"error": handshake_error,
"stderr": stderr[-4000:],
}
def doctor(
profile: str,
*,
live: bool,
probe: bool,
bindings: dict[str, str],
progress: Callable[[str], None] | None = None,
) -> dict[str, Any]:
if probe and not live:
raise ValueError("--probe requires --live because it performs a live model request")
if progress:
progress(f"Resolving and validating profile {profile}...")
resolved = resolve_profile(profile, bindings=bindings)
validation = profile_validation_report(profile, bindings)
checks: dict[str, Any] = {
"package_version": package_version(),
"python": {"version": sys.version.split()[0], "supported": sys.version_info >= (3, 11)},
"profile": validation,
"paths": {
"install_root": str(install_root()),
"config_root": str(config_root()),
"state_root": str(state_root()),
},
"binaries": {
"codex": shutil.which(
os.environ.get("MMO_CODEX_BIN") or str(load_settings().get("codex_bin", "codex"))
),
"switchyard-server": shutil.which(
str(load_settings().get("switchyard_bin", "switchyard-server"))
),
"git": shutil.which("git"),
},
"credentials": _credentials_status(resolved),
"tool_mcp": tool_mcp_status(resolved),
"live": None,
}
checks["app_server_protocol"] = app_server_protocol_status(checks["binaries"]["codex"])
required_env = checks["credentials"]["environment_credentials"]
builtin_auth_required = bool(checks["credentials"]["builtin_auth"])
if builtin_auth_required:
codex_binary = checks["binaries"]["codex"]
base_home = Path(str(load_settings().get("base_codex_home", "~/.codex"))).expanduser()
checks["credentials"]["codex_login_status"] = (
_codex_auth_status(str(codex_binary), base_home)
if codex_binary
else {"passed": False, "error": "Codex binary not found"}
)
login_ok = bool(checks["credentials"]["codex_login_status"].get("passed"))
for auth in checks["credentials"]["builtin_auth"].values():
transferable = bool(auth["file_auth_transferable"])
auth["usable_by_generated_home"] = login_ok and transferable
if not login_ok:
auth["reason"] = "the configured base Codex home is not logged in"
elif auth["auth_link_mode"] == "none":
auth["reason"] = "settings disable authentication propagation"
elif not auth["auth_json"]:
auth["reason"] = (
"login is keyring-only; configure file-backed Codex auth for isolated homes"
)
else:
auth["reason"] = None
if progress:
progress(f"Compiling diagnostic snapshot for {profile}...")
snapshot = compile_profile(profile, bindings=bindings)
gateway_required = bool(snapshot["manifest"]["gateway_required"])
offline_ok = (
checks["python"]["supported"]
and validation["valid"]
and bool(checks["binaries"]["git"])
and bool(checks["binaries"]["codex"])
and bool(checks["app_server_protocol"]["passed"])
and (not gateway_required or bool(checks["binaries"]["switchyard-server"]))
and bool(checks["tool_mcp"]["transport_ok"])
)
checks["offline_ok"] = offline_ok
checks["credentials_ok"] = (
all(required_env.values())
and (
not builtin_auth_required
or all(
bool(auth.get("usable_by_generated_home"))
for auth in checks["credentials"]["builtin_auth"].values()
)
)
and bool(checks["tool_mcp"]["credentials_ok"])
)
if live:
live_result: dict[str, Any] = {}
if snapshot["manifest"]["gateway_required"]:
if progress:
progress("Checking the live Switchyard gateway and advertised routes...")
try:
dry = dry_run_gateway(snapshot["manifest"]["snapshot_hash"])
live_result["switchyard_dry_run"] = {
"passed": dry is not None and dry.returncode == 0,
"exit_code": dry.returncode if dry else None,
"stdout": dry.stdout[-4000:] if dry else "",
"stderr": dry.stderr[-4000:] if dry else "",
}
gateway = ensure_gateway(snapshot["manifest"]["snapshot_hash"])
live_result["gateway"] = gateway
models = gateway_models(snapshot["manifest"]["snapshot_hash"])
live_result["gateway_models"] = models
advertised = (
{
item["id"]
for item in models.get("data", [])
if isinstance(item, dict) and isinstance(item.get("id"), str)
}
if isinstance(models, dict)
else set()
)
expected = set(snapshot["manifest"]["route_ids"].values())
live_result["routes_advertised"] = {
"passed": expected == advertised,
"expected": sorted(expected),
"advertised": sorted(advertised),
"missing": sorted(expected - advertised),
"unexpected": sorted(advertised - expected),
}
except Exception as exc:
live_result["gateway_error"] = f"{type(exc).__name__}: {exc}"
if resolved["capabilities"]["mcp_agents"]:
if progress:
progress("Checking the internal Agent MCP handshake...")
session = None
try:
session = create_session(profile=profile, cwd=os.getcwd(), bindings=bindings)
mark_session_running(
session["session_id"],
os.getpid(),
expected_run_id=str(session["current_run_id"]),
)
live_result["mcp"] = _mcp_handshake(session)
except Exception as exc:
live_result["mcp_error"] = f"{type(exc).__name__}: {exc}"
finally:
if session is not None:
with contextlib.suppress(Exception):
finish_session(
session["session_id"],
exit_code=0 if "mcp_error" not in live_result else 1,
error=live_result.get("mcp_error"),
expected_run_id=str(session["current_run_id"]),
)
else:
live_result["mcp"] = {
"passed": True,
"skipped": True,
"reason": "profile has no MCP participants",
}
if probe:
if progress:
progress("Sending the live root-model probe...")
try:
root = resolved["agents"][resolved["profile"]["root"]]
result = run_root_exec(
profile=profile,
cwd=os.getcwd(),
prompt=_root_harness_prompt(
"Return exactly MMO_ROOT_OK and nothing else as the final assistant "
"message. Do not spawn agents or call tools other than the required "
"goal-lifecycle update.",
root["execution_mode"],
),
bindings=bindings,
wall_timeout_seconds=300,
sandbox_mode="read-only",
label="doctor-root-probe",
)
live_result["root_model_probe"] = {
"passed": result.get("exit_code") == 0
and result.get("result", "").strip() == "MMO_ROOT_OK",
"result": result,
}
if result.get("status") not in TERMINAL_SESSION_STATUSES:
live_result["root_model_probe"]["cleanup"] = _cleanup_detached_harness_session(
result["session"]["session_id"]
)
except Exception as exc:
live_result["root_model_probe"] = {
"passed": False,
"error": f"{type(exc).__name__}: {exc}",
}
failed_session_id = getattr(exc, "mmo_session_id", None)
if (
isinstance(failed_session_id, str)
and getattr(exc, "mmo_session_status", None) not in TERMINAL_SESSION_STATUSES
):
try:
live_result["root_model_probe"]["cleanup"] = (
_cleanup_detached_harness_session(failed_session_id)
)
except Exception as cleanup_exc:
live_result["root_model_probe"]["cleanup_error"] = (
f"{type(cleanup_exc).__name__}: {cleanup_exc}"
)
checks["live"] = live_result
live_checks = [
value.get("passed")
for value in live_result.values()
if isinstance(value, dict) and "passed" in value
]
checks["live_ok"] = not any(key.endswith("_error") for key in live_result) and all(
live_checks
)
checks["passed"] = (
bool(checks["offline_ok"])
and bool(checks["credentials_ok"])
and (not live or bool(checks.get("live_ok")))
)
return checks
def _smoke_backend(task: dict[str, Any], agent: dict[str, Any]) -> str:
"""Select the declared execution backend for a smoke task.
Ambiguous hybrid tasks default to MCP because that path has enforceable
scope, resource, lineage, and result-contract semantics. Profiles that need
to exercise Codex native subagents must say ``backend = "native"``.
"""
explicit = task.get("backend")
if explicit:
return str(explicit)
backends = list(agent.get("backends", []))
if len(backends) == 1:
return str(backends[0])
if "mcp" in backends:
return "mcp"
if "native" in backends:
return "native"
raise ValueError(f"agent {task.get('agent')!r} has no executable smoke backend")
def _native_smoke_prompt(
*,
agent_id: str,
native_name: str,
task_kind: str,
task: str,
) -> str:
return textwrap.dedent(
f"""
This is an automated Codex native-subagent acceptance test.
You MUST delegate the work below through Codex's native subagent tool to
the configured custom role named `{native_name}` (profile agent
`{agent_id}`). Do not call the `mmo_mesh` Agent MCP server for this task,
and do not perform the delegated investigation yourself.
Task kind: {task_kind}
Delegated task:
{task}
Wait for that native subagent to complete, inspect its returned result,
and then summarize it. If and only if the named native subagent was
actually used and returned successfully, include this exact line at the
end of your answer:
MMO_NATIVE_SMOKE_OK
"""
).strip()
def _dynamic_tool_aliases(required_tools: list[str]) -> dict[str, str]:
"""Map every valid server/tool split to one required smoke-tool name.
Tool MCP server IDs may contain dots, as may tool names. Profile validation
has already proved that exactly one split is granted to the smoke role. The
compatibility bridge exposes the selected split as a flat dynamic function;
accepting every syntactic split here lets the evidence reader recognize that
function without weakening the earlier grant/ambiguity validation.
"""
aliases: dict[str, str] = {}
for qualified in required_tools:
for index, character in enumerate(qualified):
if character != ".":
continue
server = qualified[:index]
tool = qualified[index + 1 :]
if server and tool:
aliases[_flat_mcp_dynamic_tool_name(server, tool)] = qualified
return aliases
def _successful_mcp_tools(
events_path: str | Path | None,
*,
dynamic_tool_aliases: dict[str, str] | None = None,
) -> set[str]:
"""Return MCP tools whose Codex event reached a successful terminal state."""
if not events_path:
return set()
path = Path(events_path)
if not path.is_file():
return set()
successful: set[str] = set()
for line in path.read_text(encoding="utf-8", errors="replace").splitlines():
try:
event = json.loads(line)
except json.JSONDecodeError:
continue
pending: list[Any] = [event]
while pending:
value = pending.pop()
if isinstance(value, dict):
item_type = value.get("type")
if (
isinstance(item_type, str)
and item_type in {"mcp_tool_call", "mcpToolCall"}
and value.get("status") == "completed"
and value.get("error") is None
):
server = value.get("server") or value.get("serverName")
tool = value.get("tool") or value.get("toolName")
if isinstance(server, str) and server and isinstance(tool, str) and tool:
successful.add(f"{server}.{tool}")
elif (
item_type == "dynamicToolCall"
and value.get("status") == "completed"
and value.get("success") is True
):
tool = value.get("tool")
if isinstance(tool, str) and dynamic_tool_aliases:
qualified = dynamic_tool_aliases.get(tool)
if qualified is not None:
successful.add(qualified)
pending.extend(value.values())
elif isinstance(value, list):
pending.extend(value)
return successful
def _smoke_tool_evidence(
task: dict[str, Any], events_path: str | Path | None
) -> tuple[list[str], list[str], list[str]]:
required = list(task.get("required_mcp_tools", []))
observed = sorted(
_successful_mcp_tools(
events_path,
dynamic_tool_aliases=_dynamic_tool_aliases(required),
)
)
missing = sorted(set(required) - set(observed))
return required, observed, missing
def _wait_for_smoke_job(
job_id: str,
*,
session_id: str,
wall_timeout_seconds: int,
wait_seconds: int,
) -> dict[str, Any]:
"""Wait in MCP-sized slices without turning the harness wall into a role timeout."""
started = time.monotonic()
deadline = started + wall_timeout_seconds
waited: dict[str, Any] | None = None
while True:
remaining = deadline - time.monotonic()
timeout = (
0
if wait_seconds == 0 or remaining <= 0
else min(120, wait_seconds, max(1, int(remaining + 0.999)))
)
waited = wait_for_jobs(
[job_id],
session_id=session_id,
timeout_seconds=timeout,
include_results=True,
)
if not waited["unfinished"] or wait_seconds == 0 or time.monotonic() >= deadline:
break
waited["harness_wall_timeout_seconds"] = wall_timeout_seconds
waited["harness_wall_exhausted"] = bool(waited["unfinished"])
waited["harness_elapsed_seconds"] = time.monotonic() - started
return waited
def _cleanup_detached_harness_session(session_id: str) -> dict[str, Any]:
"""Guarantee that a smoke harness leaves no recoverable run mutating its fixture."""
try:
return {"mode": "graceful_stop", "result": stop_session(session_id, grace_seconds=0)}
except Exception as stop_error:
try:
return {"mode": "immediate_cancel", "result": cancel_session(session_id)}
except Exception as cancel_error:
raise RuntimeError(
"smoke harness could not retire detached session: "
f"{type(stop_error).__name__}: {stop_error}; "
f"{type(cancel_error).__name__}: {cancel_error}"
) from cancel_error
def smoke_profile(
profile: str,
*,
cwd: str,
bindings: dict[str, str],
root_only: bool,
workers_only: bool,
progress: Callable[[str], None] | None = None,
) -> dict[str, Any]:
if progress:
progress(f"Resolving smoke tasks for profile {profile}...")
resolved = resolve_profile(profile, bindings=bindings)
smoke = resolved.get("smoke") or {"tasks": []}
root_id = resolved["profile"]["root"]
tasks = smoke.get("tasks", [])
results: list[dict[str, Any]] = []
root_tasks = [task for task in tasks if task["agent"] == root_id]
worker_tasks = [task for task in tasks if task["agent"] != root_id]
root_execution_mode = resolved["agents"][root_id]["execution_mode"]
if not workers_only:
for task in root_tasks:
if progress:
progress(f"Running root smoke task for {task['agent']}...")
try:
result = run_root_exec(
profile=profile,
cwd=cwd,
prompt=_root_harness_prompt(task["task"], root_execution_mode),
bindings=bindings,
wall_timeout_seconds=int(task.get("wall_timeout_seconds", 600)),
sandbox_mode=task.get("mode", "read-only"),
label=f"smoke-{task['agent']}",
)
required_tools, observed_tools, missing_tools = _smoke_tool_evidence(
task, result.get("events_path")
)
results.append(
{
"agent": task["agent"],
"backend": "root",
"passed": result["exit_code"] == 0 and not missing_tools,
"required_mcp_tools": required_tools,
"observed_mcp_tools": observed_tools,
"missing_mcp_tools": missing_tools,
"result": result,
}
)
if result.get("status") not in TERMINAL_SESSION_STATUSES:
try:
results[-1]["cleanup"] = _cleanup_detached_harness_session(
result["session"]["session_id"]
)
except Exception as cleanup_exc:
results[-1]["passed"] = False
results[-1]["cleanup_error"] = (
f"{type(cleanup_exc).__name__}: {cleanup_exc}"
)
except Exception as exc:
required_tools, observed_tools, missing_tools = _smoke_tool_evidence(task, None)
failure = {
"agent": task["agent"],
"backend": "root",
"passed": False,
"required_mcp_tools": required_tools,
"observed_mcp_tools": observed_tools,
"missing_mcp_tools": missing_tools,
"error": f"{type(exc).__name__}: {exc}",
}
failed_session_id = getattr(exc, "mmo_session_id", None)
if (
isinstance(failed_session_id, str)
and getattr(exc, "mmo_session_status", None) not in TERMINAL_SESSION_STATUSES
):
try:
failure["cleanup"] = _cleanup_detached_harness_session(failed_session_id)
except Exception as cleanup_exc:
failure["cleanup_error"] = f"{type(cleanup_exc).__name__}: {cleanup_exc}"
results.append(failure)
if not root_only:
native_tasks: list[dict[str, Any]] = []
mcp_tasks: list[dict[str, Any]] = []
for task in worker_tasks:
backend = _smoke_backend(task, resolved["agents"][task["agent"]])
if backend == "native":
native_tasks.append(task)
elif backend == "mcp":
mcp_tasks.append(task)
else: # profile validation should make this unreachable
results.append(
{
"agent": task["agent"],
"backend": backend,
"passed": False,
"error": f"unsupported smoke backend: {backend}",
}
)
# Native subagents are owned by a Codex root thread, so the live smoke
# test must exercise the actual root -> native-agent path. A successful
# direct model call would not prove that the custom role is discoverable
# or that Codex can spawn it.
for task in native_tasks:
if progress:
progress(f"Running native-agent smoke task for {task['agent']}...")
agent = resolved["agents"][task["agent"]]
try:
result = run_root_exec(
profile=profile,
cwd=cwd,
prompt=_root_harness_prompt(
_native_smoke_prompt(
agent_id=task["agent"],
native_name=agent["native_name"],
task_kind=task["task_kind"],
task=task["task"],
),
root_execution_mode,
),
bindings=bindings,
wall_timeout_seconds=int(task.get("wall_timeout_seconds", 900)),
sandbox_mode=task.get("mode", "read-only"),
label=f"smoke-native-{task['agent']}",
)
marker_present = "MMO_NATIVE_SMOKE_OK" in result.get("result", "")
required_tools, observed_tools, missing_tools = _smoke_tool_evidence(
task, result.get("events_path")
)
results.append(
{
"agent": task["agent"],
"native_name": agent["native_name"],
"backend": "native",
"passed": (
result["exit_code"] == 0 and marker_present and not missing_tools
),
"marker_present": marker_present,
"required_mcp_tools": required_tools,
"observed_mcp_tools": observed_tools,
"missing_mcp_tools": missing_tools,
"result": result,
}
)
if result.get("status") not in TERMINAL_SESSION_STATUSES:
try:
results[-1]["cleanup"] = _cleanup_detached_harness_session(
result["session"]["session_id"]
)
except Exception as cleanup_exc:
results[-1]["passed"] = False
results[-1]["cleanup_error"] = (
f"{type(cleanup_exc).__name__}: {cleanup_exc}"
)
except Exception as exc:
required_tools, observed_tools, missing_tools = _smoke_tool_evidence(task, None)
failure = {
"agent": task["agent"],
"native_name": agent["native_name"],
"backend": "native",
"passed": False,
"required_mcp_tools": required_tools,
"observed_mcp_tools": observed_tools,
"missing_mcp_tools": missing_tools,
"error": f"{type(exc).__name__}: {exc}",
}
failed_session_id = getattr(exc, "mmo_session_id", None)
if (
isinstance(failed_session_id, str)
and getattr(exc, "mmo_session_status", None) not in TERMINAL_SESSION_STATUSES
):
try:
failure["cleanup"] = _cleanup_detached_harness_session(failed_session_id)
except Exception as cleanup_exc:
failure["cleanup_error"] = f"{type(cleanup_exc).__name__}: {cleanup_exc}"
results.append(failure)
# Agent MCP tasks share one execution run so active capacity, scope
# leases, resource groups, and result contracts are tested against
# the same durable supervisor state.
if mcp_tasks:
session = None
try:
session = create_session(profile=profile, cwd=cwd, bindings=bindings)
mark_session_running(
session["session_id"],
os.getpid(),
expected_run_id=str(session["current_run_id"]),
)
for task in mcp_tasks:
if progress:
progress(f"Running Agent MCP smoke task for {task['agent']}...")
try:
job = spawn_job(
session_id=session["session_id"],
caller_agent=root_id,
caller_job_id=None,
agent_id=task["agent"],
task_kind=task.get("task_kind"),
task=task.get("task"),
literal_task=task.get("literal_task"),
mode=task.get("mode", "read-only"),
write_scope_values=task.get("write_scope", []),
attachments=task.get("attachments", []),
label=f"smoke-{task['agent']}",
)
waited = _wait_for_smoke_job(
job["job_id"],
session_id=session["session_id"],
wall_timeout_seconds=int(task.get("wall_timeout_seconds", 900)),
wait_seconds=int(task.get("wait_seconds", 120)),
)
if waited["unfinished"]:
waited["cleanup"] = cancel_job(
job["job_id"],
session_id=session["session_id"],
cascade=True,
reason="smoke harness wall limit expired",
)
waited["cleanup_wait"] = wait_for_jobs(
[job["job_id"]],
session_id=session["session_id"],
timeout_seconds=30,
include_results=True,
)
final = load_job_public(job["job_id"])
required_tools, observed_tools, missing_tools = _smoke_tool_evidence(
task, final.get("events_path")
)
passed = (
final["status"] in {"completed", "completed_with_warnings"}
and not missing_tools
)
results.append(
{
"agent": task["agent"],
"backend": "mcp",
"passed": passed,
"required_mcp_tools": required_tools,
"observed_mcp_tools": observed_tools,
"missing_mcp_tools": missing_tools,
"job": final,
"wait": waited,
}
)
except Exception as exc:
required_tools, observed_tools, missing_tools = _smoke_tool_evidence(
task, None
)
results.append(
{
"agent": task["agent"],
"backend": "mcp",
"passed": False,
"required_mcp_tools": required_tools,
"observed_mcp_tools": observed_tools,
"missing_mcp_tools": missing_tools,
"error": f"{type(exc).__name__}: {exc}",
}
)
finally:
if session:
finished = finish_session(
session["session_id"],
exit_code=0 if all(item["passed"] for item in results) else 1,
expected_run_id=str(session["current_run_id"]),
)
if finished.get("status") not in TERMINAL_SESSION_STATUSES:
cleanup = _cleanup_detached_harness_session(session["session_id"])
for item in results:
if item.get("backend") == "mcp":
item.setdefault("session_cleanup", cleanup)
return {
"profile": resolved["profile"]["id"],
"passed": bool(results) and all(item["passed"] for item in results),
"results": results,
}
def load_job_public(job_id: str) -> dict[str, Any]:
for item in list_jobs(job_ids=[job_id], limit=1):
return item
raise FileNotFoundError(job_id)