2261 lines
94 KiB
Python
2261 lines
94 KiB
Python
|
|
#!/usr/bin/env python3
|
||
|
|
"""Validated, deterministic source snapshots for bundled model inventories.
|
||
|
|
|
||
|
|
Every externally maintained bundled inventory uses the same envelope. The
|
||
|
|
catalog generator consumes only the reviewed ``catalog`` records in these
|
||
|
|
snapshots; route-specific source evidence remains alongside each record so
|
||
|
|
refreshes can be audited without making catalog generation depend on a network.
|
||
|
|
"""
|
||
|
|
|
||
|
|
from __future__ import annotations
|
||
|
|
|
||
|
|
import math
|
||
|
|
import re
|
||
|
|
from collections.abc import Mapping, Sequence
|
||
|
|
from datetime import date, datetime
|
||
|
|
from decimal import Decimal, InvalidOperation
|
||
|
|
from pathlib import Path
|
||
|
|
from typing import Any
|
||
|
|
from urllib.parse import urlsplit
|
||
|
|
|
||
|
|
from mmo_util import read_json, stable_hash, valid_absolute_uri, validate_id
|
||
|
|
from mmo_version import APP_SERVER_PROTOCOL_CODEX_VERSION, MMO_SCHEMA_VERSION
|
||
|
|
|
||
|
|
SNAPSHOT_FIELDS = {
|
||
|
|
"schema_version",
|
||
|
|
"inventory",
|
||
|
|
"adapter",
|
||
|
|
"fingerprint_fields",
|
||
|
|
"as_of",
|
||
|
|
"dynamic",
|
||
|
|
"sources",
|
||
|
|
"discovery",
|
||
|
|
"captures",
|
||
|
|
"models",
|
||
|
|
"models_sha256",
|
||
|
|
}
|
||
|
|
MODEL_RECORD_FIELDS = {"catalog", "evidence"}
|
||
|
|
CAPTURE_FIELDS = {"source", "retrieved_at", "response_sha256"}
|
||
|
|
DISCOVERY_FIELDS = {
|
||
|
|
"endpoint",
|
||
|
|
"endpoint_selections",
|
||
|
|
"ignored_model_prefixes",
|
||
|
|
"known_deprecated_models",
|
||
|
|
}
|
||
|
|
SOURCE_REFERENCE_FIELDS = {
|
||
|
|
"source",
|
||
|
|
"availability_source",
|
||
|
|
"capability_source",
|
||
|
|
"pricing_source",
|
||
|
|
}
|
||
|
|
OPENROUTER_REASONING_ORDER = ("none", "minimal", "low", "medium", "high", "xhigh", "max")
|
||
|
|
OPENROUTER_SOURCE_IDS = {
|
||
|
|
"openrouter-models-api": "https://openrouter.ai/api/v1/models",
|
||
|
|
"openrouter-zdr-endpoints": "https://openrouter.ai/api/v1/endpoints/zdr",
|
||
|
|
"openrouter-openapi": "https://openrouter.ai/openapi.json",
|
||
|
|
"openrouter-reasoning": "https://openrouter.ai/docs/guides/best-practices/reasoning-tokens",
|
||
|
|
"openrouter-tool-calling": "https://openrouter.ai/docs/guides/features/tool-calling",
|
||
|
|
"openrouter-usage": "https://openrouter.ai/docs/cookbook/administration/usage-accounting",
|
||
|
|
}
|
||
|
|
OPENCODE_ZEN_SOURCE_IDS = {
|
||
|
|
"models-dev-opencode-zen": "https://models.dev/api.json",
|
||
|
|
"opencode-zen-docs": "https://opencode.ai/docs/zen",
|
||
|
|
"opencode-zen-docs-source": (
|
||
|
|
"https://raw.githubusercontent.com/anomalyco/opencode/"
|
||
|
|
"dev/packages/web/src/content/docs/zen.mdx"
|
||
|
|
),
|
||
|
|
"opencode-zen-models": "https://opencode.ai/zen/v1/models",
|
||
|
|
}
|
||
|
|
OPENCODE_ZEN_ROUTE_BY_NPM = {
|
||
|
|
"@ai-sdk/anthropic": "opencode_zen_anthropic_messages",
|
||
|
|
"@ai-sdk/google": "opencode_zen_google_catalog",
|
||
|
|
"@ai-sdk/openai": "opencode_zen_responses",
|
||
|
|
"@ai-sdk/openai-compatible": "opencode_zen_openai_chat",
|
||
|
|
}
|
||
|
|
OPENCODE_GO_SOURCE_IDS = {
|
||
|
|
"models-dev-opencode-go": "https://models.dev/api.json",
|
||
|
|
"opencode-go-docs": "https://opencode.ai/docs/go/",
|
||
|
|
"opencode-go-docs-source": (
|
||
|
|
"https://raw.githubusercontent.com/anomalyco/opencode/"
|
||
|
|
"dev/packages/web/src/content/docs/go.mdx"
|
||
|
|
),
|
||
|
|
"opencode-go-models": "https://opencode.ai/zen/go/v1/models",
|
||
|
|
}
|
||
|
|
OPENCODE_GO_ROUTE_BY_NPM = {
|
||
|
|
"@ai-sdk/anthropic": "opencode_go_anthropic_messages",
|
||
|
|
"@ai-sdk/openai": "opencode_go_responses",
|
||
|
|
"@ai-sdk/openai-compatible": "opencode_go_openai_chat",
|
||
|
|
}
|
||
|
|
CODEX_REASONING_ORDER = ("none", "minimal", "low", "medium", "high", "xhigh", "max", "ultra")
|
||
|
|
FULL_FINGERPRINT_FIELDS = (
|
||
|
|
"upstream_id",
|
||
|
|
"canonical_slug",
|
||
|
|
"pricing",
|
||
|
|
"limits",
|
||
|
|
"modalities",
|
||
|
|
"tools",
|
||
|
|
"reasoning",
|
||
|
|
"structured_output",
|
||
|
|
"supported_parameters",
|
||
|
|
"deprecation",
|
||
|
|
"endpoint_metadata",
|
||
|
|
)
|
||
|
|
|
||
|
|
|
||
|
|
def infer_model_maker(upstream_id: str) -> str:
|
||
|
|
"""Return a stable model-maker identity, distinct from access operator."""
|
||
|
|
|
||
|
|
slug = _nonempty_string(upstream_id, "upstream model id").casefold().removeprefix("~")
|
||
|
|
if "/" in slug:
|
||
|
|
return slug.split("/", 1)[0]
|
||
|
|
prefixes = {
|
||
|
|
"claude": "anthropic",
|
||
|
|
"deepseek": "deepseek",
|
||
|
|
"gemini": "google",
|
||
|
|
"glm": "zai",
|
||
|
|
"gpt": "openai",
|
||
|
|
"o1": "openai",
|
||
|
|
"o3": "openai",
|
||
|
|
"o4": "openai",
|
||
|
|
"kimi": "moonshotai",
|
||
|
|
"minimax": "minimax",
|
||
|
|
"nemotron": "nvidia",
|
||
|
|
"qwen": "qwen",
|
||
|
|
}
|
||
|
|
return next((maker for prefix, maker in prefixes.items() if slug.startswith(prefix)), "unknown")
|
||
|
|
|
||
|
|
|
||
|
|
def _object(value: Any, label: str) -> dict[str, Any]:
|
||
|
|
if not isinstance(value, Mapping):
|
||
|
|
raise ValueError(f"{label} must be an object")
|
||
|
|
if not all(isinstance(key, str) for key in value):
|
||
|
|
raise ValueError(f"{label} keys must be strings")
|
||
|
|
return dict(value)
|
||
|
|
|
||
|
|
|
||
|
|
def _nonempty_string(value: Any, label: str) -> str:
|
||
|
|
if not isinstance(value, str) or not value.strip():
|
||
|
|
raise ValueError(f"{label} must be a non-empty string")
|
||
|
|
return value
|
||
|
|
|
||
|
|
|
||
|
|
def _string_list(value: Any, label: str, *, allow_empty: bool = True) -> list[str]:
|
||
|
|
if not isinstance(value, list) or not all(isinstance(item, str) and item for item in value):
|
||
|
|
raise ValueError(f"{label} must be a list of non-empty strings")
|
||
|
|
if not allow_empty and not value:
|
||
|
|
raise ValueError(f"{label} must not be empty")
|
||
|
|
if len(value) != len(set(value)):
|
||
|
|
raise ValueError(f"{label} must not contain duplicates")
|
||
|
|
return list(value)
|
||
|
|
|
||
|
|
|
||
|
|
def _iso_date(value: Any, label: str) -> str:
|
||
|
|
text = _nonempty_string(value, label)
|
||
|
|
try:
|
||
|
|
parsed = date.fromisoformat(text)
|
||
|
|
except ValueError as exc:
|
||
|
|
raise ValueError(f"{label} must use YYYY-MM-DD") from exc
|
||
|
|
if parsed.isoformat() != text:
|
||
|
|
raise ValueError(f"{label} must use canonical YYYY-MM-DD")
|
||
|
|
return text
|
||
|
|
|
||
|
|
|
||
|
|
def _sha256(value: Any, label: str) -> str:
|
||
|
|
text = _nonempty_string(value, label)
|
||
|
|
if len(text) != 64 or any(char not in "0123456789abcdef" for char in text):
|
||
|
|
raise ValueError(f"{label} must be a lowercase SHA-256 digest")
|
||
|
|
return text
|
||
|
|
|
||
|
|
|
||
|
|
def _utc_timestamp(value: Any, label: str) -> str:
|
||
|
|
text = _nonempty_string(value, label)
|
||
|
|
try:
|
||
|
|
parsed = datetime.strptime(text, "%Y-%m-%dT%H:%M:%SZ")
|
||
|
|
except ValueError as exc:
|
||
|
|
raise ValueError(f"{label} must use YYYY-MM-DDTHH:MM:SSZ") from exc
|
||
|
|
if parsed.strftime("%Y-%m-%dT%H:%M:%SZ") != text:
|
||
|
|
raise ValueError(f"{label} must use canonical UTC timestamp form")
|
||
|
|
return text
|
||
|
|
|
||
|
|
|
||
|
|
def _https_url(value: Any, label: str) -> str:
|
||
|
|
text = _nonempty_string(value, label)
|
||
|
|
if not valid_absolute_uri(text):
|
||
|
|
raise ValueError(f"{label} must be an absolute HTTPS URL without userinfo")
|
||
|
|
try:
|
||
|
|
parsed = urlsplit(text)
|
||
|
|
_ = parsed.port
|
||
|
|
except ValueError as exc:
|
||
|
|
raise ValueError(f"{label} must be an absolute HTTPS URL without userinfo") from exc
|
||
|
|
if (
|
||
|
|
parsed.scheme.lower() != "https"
|
||
|
|
or not parsed.hostname
|
||
|
|
or parsed.username is not None
|
||
|
|
or parsed.password is not None
|
||
|
|
):
|
||
|
|
raise ValueError(f"{label} must be an absolute HTTPS URL without userinfo")
|
||
|
|
return text
|
||
|
|
|
||
|
|
|
||
|
|
def _positive_int(value: Any, label: str, *, allow_zero: bool = False) -> int:
|
||
|
|
minimum = 0 if allow_zero else 1
|
||
|
|
if not isinstance(value, int) or isinstance(value, bool) or value < minimum:
|
||
|
|
qualifier = "non-negative" if allow_zero else "positive"
|
||
|
|
raise ValueError(f"{label} must be a {qualifier} integer")
|
||
|
|
return value
|
||
|
|
|
||
|
|
|
||
|
|
def _validate_catalog_record(
|
||
|
|
inventory: str,
|
||
|
|
key: str,
|
||
|
|
value: Any,
|
||
|
|
sources: Mapping[str, str],
|
||
|
|
) -> dict[str, Any]:
|
||
|
|
validate_id(key, "inventory snapshot model key")
|
||
|
|
catalog = _object(value, f"inventory {inventory} model {key}.catalog")
|
||
|
|
if catalog.get("inventory") != inventory:
|
||
|
|
raise ValueError(f"inventory {inventory} model {key} has mismatched inventory membership")
|
||
|
|
for field in ("maker", "route", "upstream_id", "display_name", "description", "source"):
|
||
|
|
_nonempty_string(catalog.get(field), f"inventory {inventory} model {key}.{field}")
|
||
|
|
route_key = validate_id(str(catalog["route"]), f"inventory {inventory} route id")
|
||
|
|
if not key.startswith(f"{route_key}__"):
|
||
|
|
raise ValueError(
|
||
|
|
f"inventory {inventory} model {key}: catalog key must start with exact route "
|
||
|
|
f"namespace {route_key!r} followed by '__'"
|
||
|
|
)
|
||
|
|
_positive_int(
|
||
|
|
catalog.get("context_window"),
|
||
|
|
f"inventory {inventory} model {key}.context_window",
|
||
|
|
allow_zero=True,
|
||
|
|
)
|
||
|
|
for field in SOURCE_REFERENCE_FIELDS:
|
||
|
|
source = catalog.get(field)
|
||
|
|
if source is not None and source not in sources:
|
||
|
|
raise ValueError(
|
||
|
|
f"inventory {inventory} model {key}.{field} references unknown source {source!r}"
|
||
|
|
)
|
||
|
|
return catalog
|
||
|
|
|
||
|
|
|
||
|
|
def validate_inventory_snapshot(
|
||
|
|
value: Any,
|
||
|
|
*,
|
||
|
|
label: str = "inventory snapshot",
|
||
|
|
expected_inventory: str | None = None,
|
||
|
|
) -> dict[str, Any]:
|
||
|
|
"""Validate and copy one common inventory snapshot envelope."""
|
||
|
|
|
||
|
|
snapshot = _object(value, label)
|
||
|
|
unknown = sorted(set(snapshot) - SNAPSHOT_FIELDS)
|
||
|
|
missing = sorted(SNAPSHOT_FIELDS - set(snapshot))
|
||
|
|
if unknown or missing:
|
||
|
|
raise ValueError(f"{label} fields are invalid: missing={missing}, unknown={unknown}")
|
||
|
|
schema = snapshot["schema_version"]
|
||
|
|
if not isinstance(schema, int) or isinstance(schema, bool) or schema != MMO_SCHEMA_VERSION:
|
||
|
|
raise ValueError(f"unsupported {label} schema_version")
|
||
|
|
inventory = validate_id(_nonempty_string(snapshot["inventory"], f"{label}.inventory"))
|
||
|
|
adapter = validate_id(_nonempty_string(snapshot["adapter"], f"{label}.adapter"))
|
||
|
|
fingerprint_fields = _string_list(
|
||
|
|
snapshot["fingerprint_fields"], f"{label}.fingerprint_fields", allow_empty=False
|
||
|
|
)
|
||
|
|
if expected_inventory is not None and inventory != expected_inventory:
|
||
|
|
raise ValueError(
|
||
|
|
f"{label} inventory {inventory!r} does not match filename {expected_inventory!r}"
|
||
|
|
)
|
||
|
|
_iso_date(snapshot["as_of"], f"{label}.as_of")
|
||
|
|
if not isinstance(snapshot["dynamic"], bool):
|
||
|
|
raise ValueError(f"{label}.dynamic must be boolean")
|
||
|
|
|
||
|
|
sources_raw = _object(snapshot["sources"], f"{label}.sources")
|
||
|
|
if not sources_raw:
|
||
|
|
raise ValueError(f"{label}.sources must not be empty")
|
||
|
|
sources = {
|
||
|
|
validate_id(key, f"{label} source id"): _https_url(url, f"{label}.sources.{key}")
|
||
|
|
for key, url in sources_raw.items()
|
||
|
|
}
|
||
|
|
|
||
|
|
discovery = _object(snapshot["discovery"], f"{label}.discovery")
|
||
|
|
unknown_discovery = sorted(set(discovery) - DISCOVERY_FIELDS)
|
||
|
|
if unknown_discovery:
|
||
|
|
raise ValueError(f"{label}.discovery has unknown fields: {unknown_discovery}")
|
||
|
|
if "endpoint" in discovery:
|
||
|
|
_https_url(discovery["endpoint"], f"{label}.discovery.endpoint")
|
||
|
|
if "endpoint_selections" in discovery:
|
||
|
|
selections = _object(
|
||
|
|
discovery["endpoint_selections"],
|
||
|
|
f"{label}.discovery.endpoint_selections",
|
||
|
|
)
|
||
|
|
if not selections:
|
||
|
|
raise ValueError(f"{label}.discovery.endpoint_selections must not be empty")
|
||
|
|
discovery["endpoint_selections"] = dict(
|
||
|
|
sorted(
|
||
|
|
(
|
||
|
|
_nonempty_string(model_id, f"{label} endpoint-selection model"),
|
||
|
|
_nonempty_string(tag, f"{label} endpoint-selection tag"),
|
||
|
|
)
|
||
|
|
for model_id, tag in selections.items()
|
||
|
|
)
|
||
|
|
)
|
||
|
|
for field in ("ignored_model_prefixes", "known_deprecated_models"):
|
||
|
|
if field in discovery:
|
||
|
|
_string_list(discovery[field], f"{label}.discovery.{field}")
|
||
|
|
|
||
|
|
captures = snapshot["captures"]
|
||
|
|
if not isinstance(captures, list):
|
||
|
|
raise ValueError(f"{label}.captures must be a list")
|
||
|
|
normalized_captures: list[dict[str, Any]] = []
|
||
|
|
for index, raw_capture in enumerate(captures):
|
||
|
|
capture = _object(raw_capture, f"{label}.captures[{index}]")
|
||
|
|
unknown_capture = sorted(set(capture) - CAPTURE_FIELDS)
|
||
|
|
missing_capture = sorted(CAPTURE_FIELDS - set(capture))
|
||
|
|
if unknown_capture or missing_capture:
|
||
|
|
raise ValueError(
|
||
|
|
f"{label}.captures[{index}] fields are invalid: "
|
||
|
|
f"missing={missing_capture}, unknown={unknown_capture}"
|
||
|
|
)
|
||
|
|
source = _nonempty_string(capture["source"], f"{label}.captures[{index}].source")
|
||
|
|
if source not in sources:
|
||
|
|
raise ValueError(f"{label}.captures[{index}] references unknown source {source!r}")
|
||
|
|
_utc_timestamp(capture["retrieved_at"], f"{label}.captures[{index}].retrieved_at")
|
||
|
|
_sha256(capture["response_sha256"], f"{label}.captures[{index}].response_sha256")
|
||
|
|
normalized_captures.append(capture)
|
||
|
|
|
||
|
|
records = _object(snapshot["models"], f"{label}.models")
|
||
|
|
if not records:
|
||
|
|
raise ValueError(f"{label}.models must not be empty")
|
||
|
|
expected_hash = _sha256(snapshot["models_sha256"], f"{label}.models_sha256")
|
||
|
|
actual_hash = stable_hash(records)
|
||
|
|
if actual_hash != expected_hash:
|
||
|
|
raise ValueError(
|
||
|
|
f"{label}.models integrity mismatch: expected {expected_hash}, calculated {actual_hash}"
|
||
|
|
)
|
||
|
|
normalized_records: dict[str, dict[str, Any]] = {}
|
||
|
|
binding_keys: dict[tuple[str, str], str] = {}
|
||
|
|
for key, raw_record in records.items():
|
||
|
|
record = _object(raw_record, f"{label}.models.{key}")
|
||
|
|
unknown_record = sorted(set(record) - MODEL_RECORD_FIELDS)
|
||
|
|
missing_record = sorted({"catalog"} - set(record))
|
||
|
|
if unknown_record or missing_record:
|
||
|
|
raise ValueError(
|
||
|
|
f"{label}.models.{key} fields are invalid: "
|
||
|
|
f"missing={missing_record}, unknown={unknown_record}"
|
||
|
|
)
|
||
|
|
catalog = _validate_catalog_record(inventory, key, record["catalog"], sources)
|
||
|
|
binding = (str(catalog["route"]), str(catalog["upstream_id"]))
|
||
|
|
previous = binding_keys.get(binding)
|
||
|
|
if previous is not None:
|
||
|
|
raise ValueError(
|
||
|
|
f"{label}.models.{previous} and {key} duplicate route/upstream binding {binding!r}"
|
||
|
|
)
|
||
|
|
binding_keys[binding] = key
|
||
|
|
normalized = {"catalog": catalog}
|
||
|
|
if "evidence" in record:
|
||
|
|
normalized["evidence"] = _object(record["evidence"], f"{label}.models.{key}.evidence")
|
||
|
|
normalized_records[key] = normalized
|
||
|
|
|
||
|
|
return {
|
||
|
|
"schema_version": MMO_SCHEMA_VERSION,
|
||
|
|
"inventory": inventory,
|
||
|
|
"adapter": adapter,
|
||
|
|
"fingerprint_fields": fingerprint_fields,
|
||
|
|
"as_of": snapshot["as_of"],
|
||
|
|
"dynamic": snapshot["dynamic"],
|
||
|
|
"sources": dict(sorted(sources.items())),
|
||
|
|
"discovery": discovery,
|
||
|
|
"captures": normalized_captures,
|
||
|
|
"models": dict(sorted(normalized_records.items())),
|
||
|
|
"models_sha256": expected_hash,
|
||
|
|
}
|
||
|
|
|
||
|
|
|
||
|
|
def load_inventory_snapshots(directory: Path) -> list[dict[str, Any]]:
|
||
|
|
"""Load every ``*.json`` snapshot in a directory in deterministic order."""
|
||
|
|
|
||
|
|
if not directory.is_dir():
|
||
|
|
raise FileNotFoundError(f"inventory snapshot directory not found: {directory}")
|
||
|
|
snapshots: list[dict[str, Any]] = []
|
||
|
|
seen: set[str] = set()
|
||
|
|
paths = sorted(directory.glob("*.json"), key=lambda item: item.name)
|
||
|
|
if not paths:
|
||
|
|
raise ValueError(f"inventory snapshot directory is empty: {directory}")
|
||
|
|
for path in paths:
|
||
|
|
if path.is_symlink() or not path.is_file():
|
||
|
|
raise ValueError(f"inventory snapshot must be a regular non-symlink file: {path}")
|
||
|
|
snapshot = validate_inventory_snapshot(
|
||
|
|
read_json(path), label=str(path), expected_inventory=path.stem
|
||
|
|
)
|
||
|
|
inventory = str(snapshot["inventory"])
|
||
|
|
if inventory in seen:
|
||
|
|
raise ValueError(f"duplicate inventory snapshot: {inventory}")
|
||
|
|
seen.add(inventory)
|
||
|
|
snapshots.append(snapshot)
|
||
|
|
return snapshots
|
||
|
|
|
||
|
|
|
||
|
|
def build_inventory_snapshot(
|
||
|
|
*,
|
||
|
|
inventory: str,
|
||
|
|
adapter: str,
|
||
|
|
fingerprint_fields: Sequence[str],
|
||
|
|
as_of: str,
|
||
|
|
dynamic: bool,
|
||
|
|
sources: Mapping[str, str],
|
||
|
|
discovery: Mapping[str, Any],
|
||
|
|
captures: Sequence[Mapping[str, Any]],
|
||
|
|
models: Mapping[str, Mapping[str, Any]],
|
||
|
|
) -> dict[str, Any]:
|
||
|
|
"""Build and validate a common envelope from catalog/evidence records."""
|
||
|
|
|
||
|
|
records = {str(key): dict(value) for key, value in sorted(models.items())}
|
||
|
|
snapshot: dict[str, Any] = {
|
||
|
|
"schema_version": MMO_SCHEMA_VERSION,
|
||
|
|
"inventory": inventory,
|
||
|
|
"adapter": adapter,
|
||
|
|
"fingerprint_fields": list(fingerprint_fields),
|
||
|
|
"as_of": as_of,
|
||
|
|
"dynamic": dynamic,
|
||
|
|
"sources": dict(sorted(sources.items())),
|
||
|
|
"discovery": dict(discovery),
|
||
|
|
"captures": [dict(value) for value in captures],
|
||
|
|
"models": records,
|
||
|
|
"models_sha256": stable_hash(records),
|
||
|
|
}
|
||
|
|
return validate_inventory_snapshot(snapshot)
|
||
|
|
|
||
|
|
|
||
|
|
def rehash_inventory_snapshot(value: Any) -> dict[str, Any]:
|
||
|
|
"""Recalculate the common model-record digest after an intentional review edit."""
|
||
|
|
|
||
|
|
snapshot = _object(value, "inventory snapshot")
|
||
|
|
records = _object(snapshot.get("models"), "inventory snapshot.models")
|
||
|
|
updated = dict(snapshot)
|
||
|
|
updated["models_sha256"] = stable_hash(records)
|
||
|
|
return validate_inventory_snapshot(updated)
|
||
|
|
|
||
|
|
|
||
|
|
def model_record_fingerprint(value: Mapping[str, Any]) -> dict[str, Any]:
|
||
|
|
"""Return the complete, route-aware fingerprint represented by one record.
|
||
|
|
|
||
|
|
Fingerprints intentionally retain upstream evidence that cannot be flattened
|
||
|
|
into the executable catalog (tiered prices, exact endpoint metadata, and
|
||
|
|
gateway-supported parameters). An ID match therefore cannot masquerade as
|
||
|
|
a capability match.
|
||
|
|
"""
|
||
|
|
|
||
|
|
record = _object(value, "inventory model record")
|
||
|
|
catalog = _object(record.get("catalog"), "inventory model record.catalog")
|
||
|
|
evidence = _object(record.get("evidence", {}), "inventory model record.evidence")
|
||
|
|
models_dev = _object(evidence.get("models_dev", {}), "models-dev evidence")
|
||
|
|
docs = _object(evidence.get("docs", {}), "provider docs evidence")
|
||
|
|
codex_runtime = _object(evidence.get("codex_runtime", {}), "Codex runtime evidence")
|
||
|
|
selected_endpoint = _object(evidence.get("selected_endpoint", {}), "selected endpoint evidence")
|
||
|
|
|
||
|
|
raw_pricing = evidence.get("pricing")
|
||
|
|
if raw_pricing is None:
|
||
|
|
raw_pricing = docs.get("pricing")
|
||
|
|
if raw_pricing is None:
|
||
|
|
raw_pricing = models_dev.get("cost")
|
||
|
|
if raw_pricing is None:
|
||
|
|
raw_pricing = {
|
||
|
|
key: catalog[key]
|
||
|
|
for key in (
|
||
|
|
"input_cost_per_million",
|
||
|
|
"output_cost_per_million",
|
||
|
|
"cached_input_cost_per_million",
|
||
|
|
"cache_write_input_cost_per_million",
|
||
|
|
)
|
||
|
|
if key in catalog
|
||
|
|
}
|
||
|
|
if codex_runtime:
|
||
|
|
raw_pricing = {"billing_mode": "chatgpt_subscription"}
|
||
|
|
if selected_endpoint:
|
||
|
|
raw_pricing = {"model": raw_pricing, "selected_endpoint": selected_endpoint["pricing"]}
|
||
|
|
|
||
|
|
limits_evidence: Any = None
|
||
|
|
if "context_length" in evidence or "top_provider" in evidence:
|
||
|
|
limits_evidence = {
|
||
|
|
"context_length": evidence.get("context_length"),
|
||
|
|
"top_provider": evidence.get("top_provider"),
|
||
|
|
}
|
||
|
|
elif models_dev.get("limit") is not None:
|
||
|
|
limits_evidence = models_dev.get("limit")
|
||
|
|
if limits_evidence is None:
|
||
|
|
limits_evidence = {
|
||
|
|
"context_window": catalog.get("context_window"),
|
||
|
|
"max_output_tokens": catalog.get("max_output_tokens"),
|
||
|
|
}
|
||
|
|
if codex_runtime:
|
||
|
|
limits_evidence = {"context_window": codex_runtime.get("context_window")}
|
||
|
|
if selected_endpoint:
|
||
|
|
limits_evidence = {
|
||
|
|
"model": limits_evidence,
|
||
|
|
"selected_endpoint": {
|
||
|
|
"context_length": selected_endpoint["context_length"],
|
||
|
|
"max_completion_tokens": selected_endpoint["max_completion_tokens"],
|
||
|
|
},
|
||
|
|
}
|
||
|
|
|
||
|
|
modalities_evidence: Any = evidence.get("architecture")
|
||
|
|
if modalities_evidence is None:
|
||
|
|
modalities_evidence = models_dev.get("modalities")
|
||
|
|
if modalities_evidence is None:
|
||
|
|
modalities_evidence = {
|
||
|
|
"input": catalog.get("modalities", []),
|
||
|
|
"output": catalog.get("output_modalities", []),
|
||
|
|
}
|
||
|
|
if codex_runtime:
|
||
|
|
modalities_evidence = {
|
||
|
|
"input": codex_runtime.get("input_modalities", []),
|
||
|
|
"output": ["text"],
|
||
|
|
}
|
||
|
|
|
||
|
|
supported_parameters = evidence.get("supported_parameters")
|
||
|
|
if supported_parameters is None:
|
||
|
|
supported_parameters = models_dev.get("supported_parameters")
|
||
|
|
if supported_parameters is None:
|
||
|
|
supported_parameters = sorted(
|
||
|
|
key
|
||
|
|
for key, enabled in (
|
||
|
|
("tools", catalog.get("tool_calling")),
|
||
|
|
("parallel_tool_calls", catalog.get("parallel_tool_calls")),
|
||
|
|
("structured_outputs", catalog.get("structured_output")),
|
||
|
|
("reasoning", catalog.get("reasoning_levels") not in (None, ["none"])),
|
||
|
|
)
|
||
|
|
if enabled
|
||
|
|
)
|
||
|
|
if selected_endpoint:
|
||
|
|
supported_parameters = selected_endpoint["supported_parameters"]
|
||
|
|
if codex_runtime:
|
||
|
|
codex_tools = _object(codex_runtime.get("tools", {}), "Codex runtime tools")
|
||
|
|
supported_parameters = sorted(
|
||
|
|
key for key, value in codex_tools.items() if value not in (None, False, "none")
|
||
|
|
)
|
||
|
|
|
||
|
|
raw_reasoning: Any = evidence.get("reasoning")
|
||
|
|
if raw_reasoning is None and "reasoning" in models_dev:
|
||
|
|
raw_reasoning = {
|
||
|
|
"enabled": models_dev.get("reasoning"),
|
||
|
|
"options": models_dev.get("reasoning_options", []),
|
||
|
|
}
|
||
|
|
if raw_reasoning is None:
|
||
|
|
raw_reasoning = {
|
||
|
|
"levels": catalog.get("reasoning_levels", []),
|
||
|
|
"default": catalog.get("default_reasoning"),
|
||
|
|
"summaries": catalog.get("supports_reasoning_summaries", False),
|
||
|
|
}
|
||
|
|
if codex_runtime:
|
||
|
|
raw_reasoning = codex_runtime.get("reasoning")
|
||
|
|
|
||
|
|
endpoint_metadata: Any = evidence.get("endpoint_metadata")
|
||
|
|
protocol_resolution = evidence.get("protocol_resolution")
|
||
|
|
if endpoint_metadata is None and docs.get("endpoint") is not None:
|
||
|
|
endpoint_metadata = docs.get("endpoint")
|
||
|
|
if protocol_resolution is not None:
|
||
|
|
endpoint_metadata = {
|
||
|
|
"documented_endpoint": endpoint_metadata,
|
||
|
|
"protocol_resolution": _object(
|
||
|
|
protocol_resolution, "provider protocol-resolution evidence"
|
||
|
|
),
|
||
|
|
}
|
||
|
|
if endpoint_metadata is None and models_dev:
|
||
|
|
endpoint_metadata = {
|
||
|
|
"provider": models_dev.get("provider"),
|
||
|
|
"interleaved": models_dev.get("interleaved"),
|
||
|
|
}
|
||
|
|
if selected_endpoint:
|
||
|
|
endpoint_metadata = {
|
||
|
|
"model": endpoint_metadata,
|
||
|
|
"selected_endpoint": {
|
||
|
|
key: selected_endpoint[key]
|
||
|
|
for key in (
|
||
|
|
"model_id",
|
||
|
|
"provider_name",
|
||
|
|
"tag",
|
||
|
|
"quantization",
|
||
|
|
"status",
|
||
|
|
"zdr",
|
||
|
|
)
|
||
|
|
},
|
||
|
|
}
|
||
|
|
if endpoint_metadata is None:
|
||
|
|
endpoint_metadata = {
|
||
|
|
"route": catalog.get("route"),
|
||
|
|
"route_policy": catalog.get("route_policy"),
|
||
|
|
}
|
||
|
|
if codex_runtime:
|
||
|
|
endpoint_metadata = {
|
||
|
|
key: codex_runtime.get(key)
|
||
|
|
for key in (
|
||
|
|
"comp_hash",
|
||
|
|
"multi_agent_version",
|
||
|
|
"service_tiers",
|
||
|
|
"supported_in_api",
|
||
|
|
)
|
||
|
|
}
|
||
|
|
|
||
|
|
deprecation: Any = {
|
||
|
|
"availability": catalog.get("availability"),
|
||
|
|
"expiration_date": evidence.get("expiration_date"),
|
||
|
|
"deprecation_date": docs.get("deprecation_date"),
|
||
|
|
}
|
||
|
|
canonical_slug = (
|
||
|
|
evidence.get("canonical_slug") or models_dev.get("id") or catalog.get("upstream_id")
|
||
|
|
)
|
||
|
|
return {
|
||
|
|
"upstream_id": catalog.get("upstream_id"),
|
||
|
|
"canonical_slug": canonical_slug,
|
||
|
|
"pricing": raw_pricing,
|
||
|
|
"limits": limits_evidence,
|
||
|
|
"modalities": modalities_evidence,
|
||
|
|
"tools": (
|
||
|
|
codex_runtime.get("tools")
|
||
|
|
if codex_runtime
|
||
|
|
else {
|
||
|
|
"tool_calling": models_dev.get("tool_call", catalog.get("tool_calling")),
|
||
|
|
"parallel_tool_calls": catalog.get("parallel_tool_calls"),
|
||
|
|
}
|
||
|
|
),
|
||
|
|
"reasoning": raw_reasoning,
|
||
|
|
"structured_output": (
|
||
|
|
codex_runtime.get("structured_output")
|
||
|
|
if codex_runtime
|
||
|
|
else models_dev.get("structured_output", catalog.get("structured_output"))
|
||
|
|
),
|
||
|
|
"supported_parameters": supported_parameters,
|
||
|
|
"deprecation": deprecation,
|
||
|
|
"endpoint_metadata": endpoint_metadata,
|
||
|
|
}
|
||
|
|
|
||
|
|
|
||
|
|
def model_record_evidence_coverage(value: Mapping[str, Any]) -> dict[str, bool]:
|
||
|
|
"""Report which fingerprint fields were established by observed evidence."""
|
||
|
|
|
||
|
|
record = _object(value, "inventory model record")
|
||
|
|
evidence = _object(record.get("evidence", {}), "inventory model record.evidence")
|
||
|
|
if evidence.get("unverified_live_only") is True:
|
||
|
|
return {field: field == "upstream_id" for field in FULL_FINGERPRINT_FIELDS}
|
||
|
|
if not evidence:
|
||
|
|
return {field: False for field in FULL_FINGERPRINT_FIELDS}
|
||
|
|
models_dev = evidence.get("models_dev")
|
||
|
|
docs = evidence.get("docs")
|
||
|
|
is_openrouter = "supported_parameters" in evidence and "top_provider" in evidence
|
||
|
|
if is_openrouter:
|
||
|
|
return {field: True for field in FULL_FINGERPRINT_FIELDS}
|
||
|
|
if isinstance(models_dev, Mapping) and isinstance(evidence.get("live"), Mapping):
|
||
|
|
# A Models.dev model without an override inherits the access product's
|
||
|
|
# validated default OpenAI-compatible package, so its endpoint family
|
||
|
|
# is still known.
|
||
|
|
endpoint_known = bool(isinstance(docs, Mapping) and docs.get("endpoint")) or bool(
|
||
|
|
models_dev
|
||
|
|
)
|
||
|
|
return {
|
||
|
|
"upstream_id": True,
|
||
|
|
"canonical_slug": True,
|
||
|
|
"pricing": isinstance(models_dev.get("cost"), Mapping)
|
||
|
|
or bool(isinstance(docs, Mapping) and docs.get("pricing")),
|
||
|
|
"limits": isinstance(models_dev.get("limit"), Mapping),
|
||
|
|
"modalities": isinstance(models_dev.get("modalities"), Mapping),
|
||
|
|
"tools": isinstance(models_dev.get("tool_call"), bool),
|
||
|
|
"reasoning": isinstance(models_dev.get("reasoning"), bool),
|
||
|
|
"structured_output": isinstance(models_dev.get("structured_output", False), bool),
|
||
|
|
"supported_parameters": isinstance(models_dev.get("tool_call"), bool),
|
||
|
|
"deprecation": isinstance(docs, Mapping),
|
||
|
|
"endpoint_metadata": endpoint_known,
|
||
|
|
}
|
||
|
|
# Other adapters must opt in with an explicit list. This is what prevents
|
||
|
|
# an authenticated /models response containing only IDs from being called
|
||
|
|
# a complete verification.
|
||
|
|
verified = evidence.get("verified_fingerprint_fields", [])
|
||
|
|
return {field: field in verified for field in FULL_FINGERPRINT_FIELDS}
|
||
|
|
|
||
|
|
|
||
|
|
def compare_inventory_fingerprints(
|
||
|
|
expected: Mapping[str, Any], observed: Mapping[str, Any]
|
||
|
|
) -> dict[str, Any]:
|
||
|
|
"""Compare two normalized snapshots, requiring full observed evidence."""
|
||
|
|
|
||
|
|
expected_snapshot = validate_inventory_snapshot(expected)
|
||
|
|
observed_snapshot = validate_inventory_snapshot(observed)
|
||
|
|
if expected_snapshot["inventory"] != observed_snapshot["inventory"]:
|
||
|
|
raise ValueError("cannot compare snapshots from different inventories")
|
||
|
|
|
||
|
|
def bindings(snapshot: Mapping[str, Any]) -> dict[tuple[str, str], Mapping[str, Any]]:
|
||
|
|
return {
|
||
|
|
(str(record["catalog"]["route"]), str(record["catalog"]["upstream_id"])): record
|
||
|
|
for record in snapshot["models"].values()
|
||
|
|
}
|
||
|
|
|
||
|
|
expected_records = bindings(expected_snapshot)
|
||
|
|
observed_records = bindings(observed_snapshot)
|
||
|
|
missing = sorted(expected_records.keys() - observed_records.keys())
|
||
|
|
unknown = sorted(observed_records.keys() - expected_records.keys())
|
||
|
|
mismatches: list[dict[str, Any]] = []
|
||
|
|
incomplete: list[dict[str, Any]] = []
|
||
|
|
for binding in sorted(expected_records.keys() & observed_records.keys()):
|
||
|
|
expected_fingerprint = model_record_fingerprint(expected_records[binding])
|
||
|
|
observed_fingerprint = model_record_fingerprint(observed_records[binding])
|
||
|
|
changed_fields = [
|
||
|
|
field
|
||
|
|
for field in FULL_FINGERPRINT_FIELDS
|
||
|
|
if expected_fingerprint[field] != observed_fingerprint[field]
|
||
|
|
]
|
||
|
|
if changed_fields:
|
||
|
|
mismatches.append(
|
||
|
|
{
|
||
|
|
"route": binding[0],
|
||
|
|
"upstream_id": binding[1],
|
||
|
|
"fields": changed_fields,
|
||
|
|
"expected_sha256": stable_hash(expected_fingerprint),
|
||
|
|
"observed_sha256": stable_hash(observed_fingerprint),
|
||
|
|
}
|
||
|
|
)
|
||
|
|
coverage = model_record_evidence_coverage(observed_records[binding])
|
||
|
|
missing_evidence = [field for field in FULL_FINGERPRINT_FIELDS if not coverage[field]]
|
||
|
|
if missing_evidence:
|
||
|
|
incomplete.append(
|
||
|
|
{
|
||
|
|
"route": binding[0],
|
||
|
|
"upstream_id": binding[1],
|
||
|
|
"fields": missing_evidence,
|
||
|
|
}
|
||
|
|
)
|
||
|
|
exact = not missing and not unknown and not mismatches and not incomplete
|
||
|
|
return {
|
||
|
|
"expected_count": len(expected_records),
|
||
|
|
"observed_count": len(observed_records),
|
||
|
|
"missing_from_observed": [list(item) for item in missing],
|
||
|
|
"unknown_to_catalog": [list(item) for item in unknown],
|
||
|
|
"fingerprint_mismatches": mismatches,
|
||
|
|
"incomplete_evidence": incomplete,
|
||
|
|
"verified_fields": list(FULL_FINGERPRINT_FIELDS) if exact else [],
|
||
|
|
"exact": exact,
|
||
|
|
}
|
||
|
|
|
||
|
|
|
||
|
|
def route_catalog_key(route_key: str, upstream_id: str, used: set[str]) -> str:
|
||
|
|
"""Derive a route-qualified, bounded key with a deterministic collision suffix."""
|
||
|
|
|
||
|
|
validate_id(route_key, "route id")
|
||
|
|
slug = _nonempty_string(upstream_id, "upstream model id")
|
||
|
|
normalized = (
|
||
|
|
"".join(
|
||
|
|
char if char.isascii() and char.isalnum() else "_" for char in slug.casefold()
|
||
|
|
).strip("_")
|
||
|
|
or "model"
|
||
|
|
)
|
||
|
|
prefix = route_key + "__"
|
||
|
|
if len(prefix) >= 64:
|
||
|
|
raise ValueError(f"route id {route_key!r} is too long for a qualified model key")
|
||
|
|
candidate = prefix + normalized
|
||
|
|
if len(candidate) <= 64 and candidate not in used:
|
||
|
|
return candidate
|
||
|
|
digest = stable_hash(slug)[:10]
|
||
|
|
suffix = "_" + digest
|
||
|
|
candidate = prefix + normalized[: 64 - len(prefix) - len(suffix)] + suffix
|
||
|
|
counter = 1
|
||
|
|
while candidate in used:
|
||
|
|
counter_suffix = f"_{counter}"
|
||
|
|
candidate = candidate[: 64 - len(counter_suffix)] + counter_suffix
|
||
|
|
counter += 1
|
||
|
|
return candidate
|
||
|
|
|
||
|
|
|
||
|
|
def _openrouter_model_rows(document: Any) -> list[dict[str, Any]]:
|
||
|
|
root = _object(document, "OpenRouter models response")
|
||
|
|
raw_rows = root.get("data")
|
||
|
|
if not isinstance(raw_rows, list):
|
||
|
|
raise ValueError("OpenRouter models response.data must be a list")
|
||
|
|
rows = [
|
||
|
|
_object(row, f"OpenRouter models response.data[{index}]")
|
||
|
|
for index, row in enumerate(raw_rows)
|
||
|
|
]
|
||
|
|
if not rows:
|
||
|
|
raise ValueError("OpenRouter models response.data must not be empty")
|
||
|
|
if "total_count" not in root or "links" not in root:
|
||
|
|
raise ValueError("OpenRouter models response lacks required pagination metadata")
|
||
|
|
total_count = root["total_count"]
|
||
|
|
_positive_int(total_count, "OpenRouter models response.total_count", allow_zero=True)
|
||
|
|
if total_count != len(rows):
|
||
|
|
raise ValueError(
|
||
|
|
"OpenRouter models response is paginated or incomplete: "
|
||
|
|
f"total_count={total_count}, rows={len(rows)}"
|
||
|
|
)
|
||
|
|
links_object = _object(root["links"], "OpenRouter models response.links")
|
||
|
|
if "next" not in links_object:
|
||
|
|
raise ValueError("OpenRouter models response.links lacks required next field")
|
||
|
|
if links_object["next"] not in (None, ""):
|
||
|
|
raise ValueError("OpenRouter models response has an unconsumed next page")
|
||
|
|
return rows
|
||
|
|
|
||
|
|
|
||
|
|
def _openrouter_evidence(raw: Mapping[str, Any]) -> dict[str, Any]:
|
||
|
|
upstream_id = _nonempty_string(raw.get("id"), "OpenRouter model id")
|
||
|
|
name = _nonempty_string(raw.get("name"), f"OpenRouter model {upstream_id}.name")
|
||
|
|
context = _positive_int(
|
||
|
|
raw.get("context_length"), f"OpenRouter model {upstream_id}.context_length"
|
||
|
|
)
|
||
|
|
architecture = _object(raw.get("architecture"), f"OpenRouter model {upstream_id}.architecture")
|
||
|
|
input_modalities = _string_list(
|
||
|
|
architecture.get("input_modalities"),
|
||
|
|
f"OpenRouter model {upstream_id}.architecture.input_modalities",
|
||
|
|
allow_empty=False,
|
||
|
|
)
|
||
|
|
output_modalities = _string_list(
|
||
|
|
architecture.get("output_modalities"),
|
||
|
|
f"OpenRouter model {upstream_id}.architecture.output_modalities",
|
||
|
|
allow_empty=False,
|
||
|
|
)
|
||
|
|
parameters = sorted(
|
||
|
|
_string_list(
|
||
|
|
raw.get("supported_parameters"),
|
||
|
|
f"OpenRouter model {upstream_id}.supported_parameters",
|
||
|
|
)
|
||
|
|
)
|
||
|
|
pricing = _object(raw.get("pricing"), f"OpenRouter model {upstream_id}.pricing")
|
||
|
|
top_provider = _object(raw.get("top_provider"), f"OpenRouter model {upstream_id}.top_provider")
|
||
|
|
top_context = top_provider.get("context_length")
|
||
|
|
if top_context is not None:
|
||
|
|
_positive_int(top_context, f"OpenRouter model {upstream_id}.top_provider.context_length")
|
||
|
|
max_output = top_provider.get("max_completion_tokens")
|
||
|
|
if max_output is not None:
|
||
|
|
_positive_int(
|
||
|
|
max_output,
|
||
|
|
f"OpenRouter model {upstream_id}.top_provider.max_completion_tokens",
|
||
|
|
)
|
||
|
|
expiration = raw.get("expiration_date")
|
||
|
|
if expiration is not None:
|
||
|
|
_iso_date(expiration, f"OpenRouter model {upstream_id}.expiration_date")
|
||
|
|
|
||
|
|
evidence: dict[str, Any] = {
|
||
|
|
"id": upstream_id,
|
||
|
|
"canonical_slug": _nonempty_string(
|
||
|
|
raw.get("canonical_slug"), f"OpenRouter model {upstream_id}.canonical_slug"
|
||
|
|
),
|
||
|
|
"name": name,
|
||
|
|
"context_length": context,
|
||
|
|
"architecture": {
|
||
|
|
"input_modalities": input_modalities,
|
||
|
|
"output_modalities": output_modalities,
|
||
|
|
},
|
||
|
|
"supported_parameters": parameters,
|
||
|
|
"pricing": dict(sorted(pricing.items())),
|
||
|
|
"top_provider": {
|
||
|
|
"context_length": top_context,
|
||
|
|
"max_completion_tokens": max_output,
|
||
|
|
"is_moderated": top_provider.get("is_moderated"),
|
||
|
|
},
|
||
|
|
"expiration_date": expiration,
|
||
|
|
"endpoint_metadata": {
|
||
|
|
"links": _object(raw.get("links", {}), f"OpenRouter model {upstream_id}.links"),
|
||
|
|
"per_request_limits": raw.get("per_request_limits"),
|
||
|
|
},
|
||
|
|
}
|
||
|
|
if raw.get("reasoning") is not None:
|
||
|
|
evidence["reasoning"] = _object(
|
||
|
|
raw["reasoning"], f"OpenRouter model {upstream_id}.reasoning"
|
||
|
|
)
|
||
|
|
if raw.get("alias_target") is not None:
|
||
|
|
evidence["alias_target"] = _object(
|
||
|
|
raw["alias_target"], f"OpenRouter model {upstream_id}.alias_target"
|
||
|
|
)
|
||
|
|
return evidence
|
||
|
|
|
||
|
|
|
||
|
|
def _openrouter_zdr_endpoint(document: Any, model_id: str, endpoint_tag: str) -> dict[str, Any]:
|
||
|
|
root = _object(document, "OpenRouter ZDR endpoints response")
|
||
|
|
rows = root.get("data")
|
||
|
|
if not isinstance(rows, list) or not rows:
|
||
|
|
raise ValueError("OpenRouter ZDR endpoints response.data must be non-empty")
|
||
|
|
matches = [
|
||
|
|
_object(row, "OpenRouter ZDR endpoint")
|
||
|
|
for row in rows
|
||
|
|
if isinstance(row, Mapping)
|
||
|
|
and row.get("model_id") == model_id
|
||
|
|
and row.get("tag") == endpoint_tag
|
||
|
|
]
|
||
|
|
if len(matches) != 1:
|
||
|
|
raise ValueError(
|
||
|
|
f"OpenRouter model {model_id} must have exactly one ZDR endpoint {endpoint_tag!r}"
|
||
|
|
)
|
||
|
|
row = matches[0]
|
||
|
|
parameters = sorted(
|
||
|
|
_string_list(
|
||
|
|
row.get("supported_parameters"),
|
||
|
|
f"OpenRouter endpoint {model_id}/{endpoint_tag}.supported_parameters",
|
||
|
|
allow_empty=False,
|
||
|
|
)
|
||
|
|
)
|
||
|
|
required = {"tools", "tool_choice", "reasoning_effort"}
|
||
|
|
if not required.issubset(parameters):
|
||
|
|
raise ValueError(
|
||
|
|
f"OpenRouter ZDR endpoint {endpoint_tag!r} for {model_id} lacks {sorted(required)}"
|
||
|
|
)
|
||
|
|
status = row.get("status")
|
||
|
|
if not isinstance(status, int) or isinstance(status, bool) or status != 0:
|
||
|
|
raise ValueError(f"OpenRouter ZDR endpoint {endpoint_tag!r} for {model_id} is not healthy")
|
||
|
|
context = _positive_int(
|
||
|
|
row.get("context_length"),
|
||
|
|
f"OpenRouter endpoint {model_id}/{endpoint_tag}.context_length",
|
||
|
|
)
|
||
|
|
max_completion = row.get("max_completion_tokens")
|
||
|
|
if max_completion is not None:
|
||
|
|
max_completion = _positive_int(
|
||
|
|
max_completion,
|
||
|
|
f"OpenRouter endpoint {model_id}/{endpoint_tag}.max_completion_tokens",
|
||
|
|
)
|
||
|
|
return {
|
||
|
|
"model_id": model_id,
|
||
|
|
"provider_name": _nonempty_string(
|
||
|
|
row.get("provider_name"),
|
||
|
|
f"OpenRouter endpoint {model_id}/{endpoint_tag}.provider_name",
|
||
|
|
),
|
||
|
|
"tag": endpoint_tag,
|
||
|
|
"quantization": _nonempty_string(
|
||
|
|
row.get("quantization"),
|
||
|
|
f"OpenRouter endpoint {model_id}/{endpoint_tag}.quantization",
|
||
|
|
),
|
||
|
|
"context_length": context,
|
||
|
|
"max_completion_tokens": max_completion,
|
||
|
|
"pricing": _object(
|
||
|
|
row.get("pricing"), f"OpenRouter endpoint {model_id}/{endpoint_tag}.pricing"
|
||
|
|
),
|
||
|
|
"supported_parameters": parameters,
|
||
|
|
"zdr": True,
|
||
|
|
"status": status,
|
||
|
|
}
|
||
|
|
|
||
|
|
|
||
|
|
def _decimal_price(pricing: Mapping[str, Any], field: str, model_id: str) -> Decimal | None:
|
||
|
|
raw = pricing.get(field)
|
||
|
|
if raw is None:
|
||
|
|
return None
|
||
|
|
if isinstance(raw, bool) or not isinstance(raw, (str, int, float)):
|
||
|
|
raise ValueError(f"OpenRouter model {model_id}.pricing.{field} must be numeric")
|
||
|
|
try:
|
||
|
|
value = Decimal(str(raw))
|
||
|
|
except InvalidOperation as exc:
|
||
|
|
raise ValueError(f"OpenRouter model {model_id}.pricing.{field} must be numeric") from exc
|
||
|
|
if not value.is_finite():
|
||
|
|
raise ValueError(f"OpenRouter model {model_id}.pricing.{field} must be finite")
|
||
|
|
return value
|
||
|
|
|
||
|
|
|
||
|
|
def _per_million(value: Decimal, model_id: str, field: str) -> float:
|
||
|
|
converted = float(value * Decimal(1_000_000))
|
||
|
|
if not math.isfinite(converted) or converted < 0:
|
||
|
|
raise ValueError(f"OpenRouter model {model_id}.pricing.{field} is out of range")
|
||
|
|
return converted
|
||
|
|
|
||
|
|
|
||
|
|
def openrouter_catalog_pricing(evidence: Mapping[str, Any]) -> dict[str, float]:
|
||
|
|
"""Translate only scalar text-token prices that the MMO schema can represent."""
|
||
|
|
|
||
|
|
model_id = _nonempty_string(evidence.get("id"), "OpenRouter evidence model id")
|
||
|
|
pricing = _object(evidence.get("pricing"), f"OpenRouter model {model_id}.pricing")
|
||
|
|
prompt = _decimal_price(pricing, "prompt", model_id)
|
||
|
|
completion = _decimal_price(pricing, "completion", model_id)
|
||
|
|
if prompt is None or completion is None:
|
||
|
|
raise ValueError(f"OpenRouter model {model_id} lacks prompt/completion pricing")
|
||
|
|
request = _decimal_price(pricing, "request", model_id) or Decimal(0)
|
||
|
|
reasoning = _decimal_price(pricing, "internal_reasoning", model_id)
|
||
|
|
cache_read = _decimal_price(pricing, "input_cache_read", model_id)
|
||
|
|
cache_write = _decimal_price(pricing, "input_cache_write", model_id)
|
||
|
|
for field, value in (
|
||
|
|
("request", request),
|
||
|
|
("internal_reasoning", reasoning),
|
||
|
|
("input_cache_read", cache_read),
|
||
|
|
("input_cache_write", cache_write),
|
||
|
|
):
|
||
|
|
if value is not None and value < 0:
|
||
|
|
raise ValueError(f"OpenRouter model {model_id}.pricing.{field} must not be negative")
|
||
|
|
negative_tokens = prompt < 0 or completion < 0
|
||
|
|
if negative_tokens:
|
||
|
|
if prompt != Decimal(-1) or completion != Decimal(-1):
|
||
|
|
raise ValueError(
|
||
|
|
f"OpenRouter model {model_id} has inconsistent dynamic token-price sentinels"
|
||
|
|
)
|
||
|
|
return {}
|
||
|
|
overrides = pricing.get("overrides", [])
|
||
|
|
if not isinstance(overrides, list):
|
||
|
|
raise ValueError(f"OpenRouter model {model_id}.pricing.overrides must be a list")
|
||
|
|
if overrides or request != 0 or (reasoning is not None and reasoning != completion):
|
||
|
|
return {}
|
||
|
|
result = {
|
||
|
|
"input_cost_per_million": _per_million(prompt, model_id, "prompt"),
|
||
|
|
"output_cost_per_million": _per_million(completion, model_id, "completion"),
|
||
|
|
}
|
||
|
|
if cache_read is not None:
|
||
|
|
result["cached_input_cost_per_million"] = _per_million(
|
||
|
|
cache_read, model_id, "input_cache_read"
|
||
|
|
)
|
||
|
|
if cache_write is not None:
|
||
|
|
result["cache_write_input_cost_per_million"] = _per_million(
|
||
|
|
cache_write, model_id, "input_cache_write"
|
||
|
|
)
|
||
|
|
return result
|
||
|
|
|
||
|
|
|
||
|
|
def openrouter_reasoning(evidence: Mapping[str, Any]) -> tuple[list[str], str]:
|
||
|
|
"""Return only reasoning efforts explicitly advertised by OpenRouter."""
|
||
|
|
|
||
|
|
model_id = _nonempty_string(evidence.get("id"), "OpenRouter evidence model id")
|
||
|
|
raw = evidence.get("reasoning")
|
||
|
|
if raw is None:
|
||
|
|
return ["none"], "none"
|
||
|
|
reasoning = _object(raw, f"OpenRouter model {model_id}.reasoning")
|
||
|
|
mandatory = reasoning.get("mandatory", False)
|
||
|
|
default_enabled = reasoning.get("default_enabled")
|
||
|
|
if not isinstance(mandatory, bool):
|
||
|
|
raise ValueError(f"OpenRouter model {model_id}.reasoning.mandatory must be boolean")
|
||
|
|
if default_enabled is not None and not isinstance(default_enabled, bool):
|
||
|
|
raise ValueError(
|
||
|
|
f"OpenRouter model {model_id}.reasoning.default_enabled must be boolean or null"
|
||
|
|
)
|
||
|
|
if "supported_efforts" not in reasoning:
|
||
|
|
return ["none"], "none"
|
||
|
|
supported_raw = reasoning["supported_efforts"]
|
||
|
|
all_gateway_efforts = supported_raw is None
|
||
|
|
supported = (
|
||
|
|
list(OPENROUTER_REASONING_ORDER)
|
||
|
|
if all_gateway_efforts
|
||
|
|
else _string_list(
|
||
|
|
supported_raw,
|
||
|
|
f"OpenRouter model {model_id}.reasoning.supported_efforts",
|
||
|
|
allow_empty=False,
|
||
|
|
)
|
||
|
|
)
|
||
|
|
unknown = sorted(set(supported) - set(OPENROUTER_REASONING_ORDER))
|
||
|
|
if unknown:
|
||
|
|
raise ValueError(f"OpenRouter model {model_id} has unknown reasoning efforts: {unknown}")
|
||
|
|
if mandatory and "none" in supported:
|
||
|
|
if all_gateway_efforts:
|
||
|
|
supported.remove("none")
|
||
|
|
else:
|
||
|
|
raise ValueError(
|
||
|
|
f"OpenRouter model {model_id} is mandatory-reasoning but advertises none"
|
||
|
|
)
|
||
|
|
levels = [item for item in OPENROUTER_REASONING_ORDER if item in supported]
|
||
|
|
if not mandatory and "none" not in levels:
|
||
|
|
levels.insert(0, "none")
|
||
|
|
default_effort = reasoning.get("default_effort")
|
||
|
|
if default_effort is not None and default_effort not in supported:
|
||
|
|
raise ValueError(f"OpenRouter model {model_id}.reasoning.default_effort is not supported")
|
||
|
|
if mandatory or default_enabled is True:
|
||
|
|
default = default_effort if isinstance(default_effort, str) else "medium"
|
||
|
|
else:
|
||
|
|
default = "none"
|
||
|
|
if default not in levels:
|
||
|
|
raise ValueError(f"OpenRouter model {model_id} has an unusable default reasoning effort")
|
||
|
|
return levels, default
|
||
|
|
|
||
|
|
|
||
|
|
def openrouter_catalog_record(evidence: Mapping[str, Any], *, as_of: str) -> dict[str, Any]:
|
||
|
|
"""Derive one conservative text-only executable catalog record."""
|
||
|
|
|
||
|
|
audit_date = date.fromisoformat(_iso_date(as_of, "OpenRouter snapshot as_of"))
|
||
|
|
model_id = _nonempty_string(evidence.get("id"), "OpenRouter evidence model id")
|
||
|
|
name = _nonempty_string(evidence.get("name"), f"OpenRouter model {model_id}.name")
|
||
|
|
context = _positive_int(
|
||
|
|
evidence.get("context_length"), f"OpenRouter model {model_id}.context_length"
|
||
|
|
)
|
||
|
|
architecture = _object(
|
||
|
|
evidence.get("architecture"), f"OpenRouter model {model_id}.architecture"
|
||
|
|
)
|
||
|
|
inputs = _string_list(
|
||
|
|
architecture.get("input_modalities"),
|
||
|
|
f"OpenRouter model {model_id}.architecture.input_modalities",
|
||
|
|
allow_empty=False,
|
||
|
|
)
|
||
|
|
outputs = _string_list(
|
||
|
|
architecture.get("output_modalities"),
|
||
|
|
f"OpenRouter model {model_id}.architecture.output_modalities",
|
||
|
|
allow_empty=False,
|
||
|
|
)
|
||
|
|
if "text" not in inputs or "text" not in outputs:
|
||
|
|
raise ValueError(f"OpenRouter model {model_id} is not a text-input/text-output model")
|
||
|
|
parameters = set(
|
||
|
|
_string_list(
|
||
|
|
evidence.get("supported_parameters"),
|
||
|
|
f"OpenRouter model {model_id}.supported_parameters",
|
||
|
|
)
|
||
|
|
)
|
||
|
|
reasoning_levels, default_reasoning = openrouter_reasoning(evidence)
|
||
|
|
expiration = evidence.get("expiration_date")
|
||
|
|
expired = False
|
||
|
|
availability = "current"
|
||
|
|
if expiration is not None:
|
||
|
|
expiration_date = date.fromisoformat(
|
||
|
|
_iso_date(expiration, f"OpenRouter model {model_id}.expiration_date")
|
||
|
|
)
|
||
|
|
expired = expiration_date < audit_date
|
||
|
|
availability = f"{'expired' if expired else 'expires'}-{expiration}"
|
||
|
|
tool_calling = "tools" in parameters
|
||
|
|
catalog: dict[str, Any] = {
|
||
|
|
"maker": infer_model_maker(model_id),
|
||
|
|
"route": "openrouter_openai_chat",
|
||
|
|
"upstream_id": model_id,
|
||
|
|
"display_name": f"{name} via OpenRouter",
|
||
|
|
"description": f"{name} served through OpenRouter's reviewed text-only Switchyard route",
|
||
|
|
"kind": "chat",
|
||
|
|
"agent_compatible": tool_calling and not expired,
|
||
|
|
"context_window": context,
|
||
|
|
"reasoning_levels": reasoning_levels,
|
||
|
|
"default_reasoning": default_reasoning,
|
||
|
|
"modalities": ["text"],
|
||
|
|
"output_modalities": ["text"],
|
||
|
|
"tool_calling": tool_calling,
|
||
|
|
"parallel_tool_calls": tool_calling and "parallel_tool_calls" in parameters,
|
||
|
|
"supports_reasoning_summaries": False,
|
||
|
|
"structured_output": "structured_outputs" in parameters,
|
||
|
|
"availability": availability,
|
||
|
|
"capability_confidence": "openrouter-models-api-snapshot",
|
||
|
|
"source": "openrouter-models-api",
|
||
|
|
"availability_source": "openrouter-models-api",
|
||
|
|
"capability_source": "openrouter-models-api",
|
||
|
|
"pricing_source": "openrouter-models-api",
|
||
|
|
"inventory": "openrouter",
|
||
|
|
"resource_group": "openrouter",
|
||
|
|
}
|
||
|
|
top_provider = _object(
|
||
|
|
evidence.get("top_provider"), f"OpenRouter model {model_id}.top_provider"
|
||
|
|
)
|
||
|
|
max_output = top_provider.get("max_completion_tokens")
|
||
|
|
if max_output is not None:
|
||
|
|
catalog["max_output_tokens"] = _positive_int(
|
||
|
|
max_output, f"OpenRouter model {model_id}.top_provider.max_completion_tokens"
|
||
|
|
)
|
||
|
|
catalog.update(openrouter_catalog_pricing(evidence))
|
||
|
|
selected_endpoint = evidence.get("selected_endpoint")
|
||
|
|
if selected_endpoint is not None:
|
||
|
|
endpoint = _object(selected_endpoint, f"OpenRouter model {model_id}.selected_endpoint")
|
||
|
|
endpoint_tag = _nonempty_string(
|
||
|
|
endpoint.get("tag"), f"OpenRouter model {model_id}.selected_endpoint.tag"
|
||
|
|
)
|
||
|
|
catalog["route_policy"] = {
|
||
|
|
"only": [endpoint_tag],
|
||
|
|
"allow_fallbacks": False,
|
||
|
|
"require_parameters": True,
|
||
|
|
"data_collection": "deny",
|
||
|
|
"zdr": True,
|
||
|
|
}
|
||
|
|
quantization = endpoint.get("quantization")
|
||
|
|
if quantization not in (None, "unknown"):
|
||
|
|
catalog["route_policy"]["quantizations"] = [quantization]
|
||
|
|
return catalog
|
||
|
|
|
||
|
|
|
||
|
|
def build_openrouter_snapshot(
|
||
|
|
document: Any,
|
||
|
|
zdr_document: Any,
|
||
|
|
*,
|
||
|
|
as_of: str,
|
||
|
|
retrieved_at: str,
|
||
|
|
response_sha256: str,
|
||
|
|
zdr_response_sha256: str,
|
||
|
|
source_url: str,
|
||
|
|
zdr_source_url: str,
|
||
|
|
endpoint_selections: Mapping[str, str],
|
||
|
|
) -> dict[str, Any]:
|
||
|
|
"""Normalize an official OpenRouter models response into the common envelope."""
|
||
|
|
|
||
|
|
_sha256(response_sha256, "OpenRouter source response SHA-256")
|
||
|
|
_sha256(zdr_response_sha256, "OpenRouter ZDR source response SHA-256")
|
||
|
|
_utc_timestamp(retrieved_at, "OpenRouter retrieval timestamp")
|
||
|
|
source_url = _https_url(source_url, "OpenRouter models source URL")
|
||
|
|
zdr_source_url = _https_url(zdr_source_url, "OpenRouter ZDR endpoints source URL")
|
||
|
|
sources = dict(OPENROUTER_SOURCE_IDS)
|
||
|
|
sources["openrouter-models-api"] = source_url
|
||
|
|
sources["openrouter-zdr-endpoints"] = zdr_source_url
|
||
|
|
selections = {
|
||
|
|
_nonempty_string(model_id, "OpenRouter endpoint-selection model"): _nonempty_string(
|
||
|
|
tag, "OpenRouter endpoint-selection tag"
|
||
|
|
)
|
||
|
|
for model_id, tag in endpoint_selections.items()
|
||
|
|
}
|
||
|
|
if not selections:
|
||
|
|
raise ValueError("OpenRouter endpoint_selections must not be empty")
|
||
|
|
used: set[str] = set()
|
||
|
|
records: dict[str, dict[str, Any]] = {}
|
||
|
|
seen_ids: set[str] = set()
|
||
|
|
for raw in sorted(_openrouter_model_rows(document), key=lambda row: str(row.get("id", ""))):
|
||
|
|
evidence = _openrouter_evidence(raw)
|
||
|
|
model_id = str(evidence["id"])
|
||
|
|
endpoint_tag = selections.get(model_id)
|
||
|
|
if endpoint_tag is not None:
|
||
|
|
evidence["selected_endpoint"] = _openrouter_zdr_endpoint(
|
||
|
|
zdr_document, model_id, endpoint_tag
|
||
|
|
)
|
||
|
|
if model_id in seen_ids:
|
||
|
|
raise ValueError(f"duplicate OpenRouter model id: {model_id}")
|
||
|
|
seen_ids.add(model_id)
|
||
|
|
architecture = _object(
|
||
|
|
evidence["architecture"], f"OpenRouter model {model_id}.architecture"
|
||
|
|
)
|
||
|
|
if (
|
||
|
|
"text" not in architecture["input_modalities"]
|
||
|
|
or "text" not in architecture["output_modalities"]
|
||
|
|
):
|
||
|
|
continue
|
||
|
|
key = route_catalog_key("openrouter_openai_chat", model_id, used)
|
||
|
|
used.add(key)
|
||
|
|
records[key] = {
|
||
|
|
"catalog": openrouter_catalog_record(evidence, as_of=as_of),
|
||
|
|
"evidence": evidence,
|
||
|
|
}
|
||
|
|
if not records:
|
||
|
|
raise ValueError("OpenRouter response did not contain any text-input/text-output models")
|
||
|
|
missing_selections = sorted(set(selections) - seen_ids)
|
||
|
|
if missing_selections:
|
||
|
|
raise ValueError(
|
||
|
|
"OpenRouter endpoint selections reference unlisted models: "
|
||
|
|
+ ", ".join(missing_selections)
|
||
|
|
)
|
||
|
|
return build_inventory_snapshot(
|
||
|
|
inventory="openrouter",
|
||
|
|
adapter="openrouter_models_api",
|
||
|
|
fingerprint_fields=FULL_FINGERPRINT_FIELDS,
|
||
|
|
as_of=as_of,
|
||
|
|
dynamic=True,
|
||
|
|
sources=sources,
|
||
|
|
discovery={
|
||
|
|
"endpoint": source_url,
|
||
|
|
"endpoint_selections": dict(sorted(selections.items())),
|
||
|
|
},
|
||
|
|
captures=[
|
||
|
|
{
|
||
|
|
"source": "openrouter-models-api",
|
||
|
|
"retrieved_at": retrieved_at,
|
||
|
|
"response_sha256": response_sha256,
|
||
|
|
},
|
||
|
|
{
|
||
|
|
"source": "openrouter-zdr-endpoints",
|
||
|
|
"retrieved_at": retrieved_at,
|
||
|
|
"response_sha256": zdr_response_sha256,
|
||
|
|
},
|
||
|
|
],
|
||
|
|
models=records,
|
||
|
|
)
|
||
|
|
|
||
|
|
|
||
|
|
def opencode_zen_reasoning(metadata: Mapping[str, Any]) -> tuple[list[str], str]:
|
||
|
|
"""Translate only Models.dev effort controls representable by Codex."""
|
||
|
|
|
||
|
|
model_id = _nonempty_string(metadata.get("id"), "OpenCode Zen metadata model id")
|
||
|
|
reasoning = metadata.get("reasoning")
|
||
|
|
if not isinstance(reasoning, bool):
|
||
|
|
raise ValueError(f"OpenCode Zen model {model_id}.reasoning must be boolean")
|
||
|
|
if not reasoning:
|
||
|
|
return ["none"], "none"
|
||
|
|
raw_options = metadata.get("reasoning_options", [])
|
||
|
|
if not isinstance(raw_options, list):
|
||
|
|
raise ValueError(f"OpenCode Zen model {model_id}.reasoning_options must be a list")
|
||
|
|
effort_values: list[str] | None = None
|
||
|
|
has_toggle = False
|
||
|
|
seen_types: set[str] = set()
|
||
|
|
for index, raw_option in enumerate(raw_options):
|
||
|
|
option = _object(raw_option, f"OpenCode Zen model {model_id}.reasoning_options[{index}]")
|
||
|
|
option_type = _nonempty_string(
|
||
|
|
option.get("type"), f"OpenCode Zen model {model_id}.reasoning_options[{index}].type"
|
||
|
|
)
|
||
|
|
if option_type in seen_types:
|
||
|
|
raise ValueError(
|
||
|
|
f"OpenCode Zen model {model_id} has duplicate reasoning option {option_type!r}"
|
||
|
|
)
|
||
|
|
seen_types.add(option_type)
|
||
|
|
if option_type == "effort":
|
||
|
|
effort_values = _string_list(
|
||
|
|
option.get("values"),
|
||
|
|
f"OpenCode Zen model {model_id}.reasoning_options[{index}].values",
|
||
|
|
allow_empty=False,
|
||
|
|
)
|
||
|
|
unknown = sorted(set(effort_values) - set(CODEX_REASONING_ORDER))
|
||
|
|
if unknown:
|
||
|
|
raise ValueError(
|
||
|
|
f"OpenCode Zen model {model_id} has unknown reasoning efforts: {unknown}"
|
||
|
|
)
|
||
|
|
elif option_type == "toggle":
|
||
|
|
has_toggle = True
|
||
|
|
elif option_type != "budget_tokens":
|
||
|
|
raise ValueError(
|
||
|
|
f"OpenCode Zen model {model_id} has unknown reasoning option {option_type!r}"
|
||
|
|
)
|
||
|
|
if effort_values is None:
|
||
|
|
# Toggle and token-budget controls have no faithful Codex effort value.
|
||
|
|
return ["none"], "none"
|
||
|
|
levels = [item for item in CODEX_REASONING_ORDER if item in effort_values]
|
||
|
|
if has_toggle and "none" not in levels:
|
||
|
|
levels.insert(0, "none")
|
||
|
|
if "none" in levels:
|
||
|
|
default = "none"
|
||
|
|
elif "medium" in levels:
|
||
|
|
default = "medium"
|
||
|
|
elif "high" in levels:
|
||
|
|
default = "high"
|
||
|
|
else:
|
||
|
|
default = levels[0]
|
||
|
|
return levels, default
|
||
|
|
|
||
|
|
|
||
|
|
def opencode_zen_catalog_pricing(docs_evidence: Mapping[str, Any]) -> dict[str, float]:
|
||
|
|
"""Retain one non-tiered rate row from an OpenCode billing table."""
|
||
|
|
|
||
|
|
model_id = _nonempty_string(docs_evidence.get("id"), "OpenCode Zen docs model id")
|
||
|
|
raw_rows = docs_evidence.get("pricing", [])
|
||
|
|
if not isinstance(raw_rows, list):
|
||
|
|
raise ValueError(f"OpenCode Zen docs model {model_id}.pricing must be a list")
|
||
|
|
rows = [
|
||
|
|
_object(row, f"OpenCode Zen docs model {model_id}.pricing[{index}]")
|
||
|
|
for index, row in enumerate(raw_rows)
|
||
|
|
]
|
||
|
|
# Multiple distinct rows mean the rate depends on request context, which
|
||
|
|
# the catalog's flat per-million fields cannot represent.
|
||
|
|
if len(rows) != 1:
|
||
|
|
return {}
|
||
|
|
row = rows[0]
|
||
|
|
input_rate = row.get("input")
|
||
|
|
output_rate = row.get("output")
|
||
|
|
if input_rate is None and output_rate is None:
|
||
|
|
# A complete dash-valued provider row means the operator publishes no
|
||
|
|
# scalar token rate. Preserve the raw row as evidence without guessing
|
||
|
|
# that an undocumented price is zero.
|
||
|
|
return {}
|
||
|
|
if (input_rate is None) != (output_rate is None):
|
||
|
|
missing = "input" if input_rate is None else "output"
|
||
|
|
raise ValueError(f"OpenCode Zen docs model {model_id}.{missing} is required")
|
||
|
|
result: dict[str, float] = {}
|
||
|
|
for source, target in (
|
||
|
|
("input", "input_cost_per_million"),
|
||
|
|
("output", "output_cost_per_million"),
|
||
|
|
("cache_read", "cached_input_cost_per_million"),
|
||
|
|
("cache_write", "cache_write_input_cost_per_million"),
|
||
|
|
):
|
||
|
|
value = row.get(source)
|
||
|
|
if value is None:
|
||
|
|
if source in {"input", "output"}:
|
||
|
|
raise ValueError(f"OpenCode Zen docs model {model_id}.{source} is required")
|
||
|
|
continue
|
||
|
|
if (
|
||
|
|
isinstance(value, bool)
|
||
|
|
or not isinstance(value, (int, float))
|
||
|
|
or not math.isfinite(float(value))
|
||
|
|
or value < 0
|
||
|
|
):
|
||
|
|
raise ValueError(
|
||
|
|
f"OpenCode Zen docs model {model_id}.{source} must be non-negative and finite"
|
||
|
|
)
|
||
|
|
result[target] = float(value)
|
||
|
|
return result
|
||
|
|
|
||
|
|
|
||
|
|
def _markdown_table(
|
||
|
|
document: str, heading: str, expected_headers: Sequence[str]
|
||
|
|
) -> list[list[str]]:
|
||
|
|
lines = document.splitlines()
|
||
|
|
try:
|
||
|
|
heading_index = next(index for index, line in enumerate(lines) if line.strip() == heading)
|
||
|
|
except StopIteration as exc:
|
||
|
|
raise ValueError(f"OpenCode Zen docs lack {heading!r}") from exc
|
||
|
|
|
||
|
|
def cells(line: str) -> list[str]:
|
||
|
|
return [item.strip().strip("`") for item in line.strip("|").split("|")]
|
||
|
|
|
||
|
|
tables: list[list[str]] = []
|
||
|
|
current: list[str] = []
|
||
|
|
for line in lines[heading_index + 1 :]:
|
||
|
|
stripped = line.strip()
|
||
|
|
if stripped.startswith("## "):
|
||
|
|
break
|
||
|
|
if stripped.startswith("|") and stripped.endswith("|"):
|
||
|
|
current.append(stripped)
|
||
|
|
elif current:
|
||
|
|
tables.append(current)
|
||
|
|
current = []
|
||
|
|
if current:
|
||
|
|
tables.append(current)
|
||
|
|
table_lines = next(
|
||
|
|
(
|
||
|
|
table
|
||
|
|
for table in tables
|
||
|
|
if len(table) >= 3 and cells(table[0]) == list(expected_headers)
|
||
|
|
),
|
||
|
|
None,
|
||
|
|
)
|
||
|
|
if table_lines is None:
|
||
|
|
observed = [cells(table[0]) for table in tables if table]
|
||
|
|
raise ValueError(
|
||
|
|
f"provider docs {heading!r} lack the required table: "
|
||
|
|
f"expected={list(expected_headers)}, observed={observed}"
|
||
|
|
)
|
||
|
|
headers = cells(table_lines[0])
|
||
|
|
separator = cells(table_lines[1])
|
||
|
|
if len(separator) != len(headers) or any(
|
||
|
|
not re.fullmatch(r":?-{3,}:?", item.replace(" ", "")) for item in separator
|
||
|
|
):
|
||
|
|
raise ValueError(f"OpenCode Zen docs {heading!r} separator is invalid")
|
||
|
|
rows = [cells(line) for line in table_lines[2:]]
|
||
|
|
if any(len(row) != len(headers) for row in rows):
|
||
|
|
raise ValueError(f"OpenCode Zen docs {heading!r} contains a malformed row")
|
||
|
|
return rows
|
||
|
|
|
||
|
|
|
||
|
|
def _opencode_zen_name_key(value: str) -> str:
|
||
|
|
return "".join(character.casefold() for character in value if character.isalnum())
|
||
|
|
|
||
|
|
|
||
|
|
_OPENCODE_PRICING_TIER_SUFFIX = re.compile(
|
||
|
|
r"\s+\((?:off-peak|peak|[<>≤≥]=?\s*[0-9]+(?:\.[0-9]+)?\s*[KMGT]?\s+tokens)\)$",
|
||
|
|
re.IGNORECASE,
|
||
|
|
)
|
||
|
|
|
||
|
|
|
||
|
|
def _opencode_pricing_base_name(value: str) -> str:
|
||
|
|
"""Remove provider-defined context/time tier labels from a model name."""
|
||
|
|
|
||
|
|
base_name = value.strip()
|
||
|
|
while True:
|
||
|
|
stripped = _OPENCODE_PRICING_TIER_SUFFIX.sub("", base_name).strip()
|
||
|
|
if stripped == base_name:
|
||
|
|
return base_name
|
||
|
|
base_name = stripped
|
||
|
|
|
||
|
|
|
||
|
|
def _opencode_zen_price(value: str, label: str) -> float | None:
|
||
|
|
if value == "-":
|
||
|
|
return None
|
||
|
|
if value == "Free":
|
||
|
|
return 0.0
|
||
|
|
match = re.fullmatch(r"\$([0-9]+(?:\.[0-9]+)?)", value)
|
||
|
|
if not match:
|
||
|
|
raise ValueError(f"{label} has an unsupported price {value!r}")
|
||
|
|
return float(match.group(1))
|
||
|
|
|
||
|
|
|
||
|
|
def _opencode_zen_docs(
|
||
|
|
document: Any, metadata_by_id: Mapping[str, Mapping[str, Any]]
|
||
|
|
) -> dict[str, dict[str, Any]]:
|
||
|
|
if not isinstance(document, str) or not document.strip():
|
||
|
|
raise ValueError("OpenCode Zen docs source must be non-empty UTF-8 text")
|
||
|
|
endpoint_rows = _markdown_table(
|
||
|
|
document,
|
||
|
|
"## Endpoints",
|
||
|
|
("Model", "Model ID", "Endpoint", "AI SDK Package"),
|
||
|
|
)
|
||
|
|
pricing_rows = _markdown_table(
|
||
|
|
document,
|
||
|
|
"## Pricing",
|
||
|
|
("Model", "Input", "Output", "Cached Read", "Cached Write"),
|
||
|
|
)
|
||
|
|
deprecation_rows = _markdown_table(
|
||
|
|
document,
|
||
|
|
"### Deprecated models",
|
||
|
|
("Model", "Deprecation date"),
|
||
|
|
)
|
||
|
|
|
||
|
|
docs_by_id: dict[str, dict[str, Any]] = {}
|
||
|
|
name_to_id: dict[str, str] = {}
|
||
|
|
|
||
|
|
def register_name(name: str, model_id: str) -> None:
|
||
|
|
key = _opencode_zen_name_key(name)
|
||
|
|
if not key:
|
||
|
|
raise ValueError(f"OpenCode Zen docs contain an empty model name for {model_id}")
|
||
|
|
previous = name_to_id.get(key)
|
||
|
|
if previous is not None and previous != model_id:
|
||
|
|
raise ValueError(
|
||
|
|
f"OpenCode Zen docs model name {name!r} is ambiguous: {previous}, {model_id}"
|
||
|
|
)
|
||
|
|
name_to_id[key] = model_id
|
||
|
|
|
||
|
|
for model_id, metadata in metadata_by_id.items():
|
||
|
|
register_name(
|
||
|
|
_nonempty_string(metadata.get("name"), f"OpenCode Zen model {model_id}.name"),
|
||
|
|
model_id,
|
||
|
|
)
|
||
|
|
for name, model_id, endpoint, npm in endpoint_rows:
|
||
|
|
model_id = _nonempty_string(model_id, "OpenCode Zen docs endpoint model id")
|
||
|
|
endpoint = _https_url(endpoint, f"OpenCode Zen docs model {model_id}.endpoint")
|
||
|
|
npm = _nonempty_string(npm, f"OpenCode Zen docs model {model_id}.npm")
|
||
|
|
if npm not in OPENCODE_ZEN_ROUTE_BY_NPM:
|
||
|
|
raise ValueError(f"OpenCode Zen docs model {model_id} has unsupported package {npm!r}")
|
||
|
|
expected_suffix = {
|
||
|
|
"@ai-sdk/openai": "/responses",
|
||
|
|
"@ai-sdk/openai-compatible": "/chat/completions",
|
||
|
|
"@ai-sdk/anthropic": "/messages",
|
||
|
|
"@ai-sdk/google": f"/models/{model_id}",
|
||
|
|
}[npm]
|
||
|
|
if endpoint != f"https://opencode.ai/zen/v1{expected_suffix}":
|
||
|
|
raise ValueError(
|
||
|
|
f"OpenCode Zen docs model {model_id} has unexpected endpoint {endpoint}"
|
||
|
|
)
|
||
|
|
evidence = docs_by_id.setdefault(model_id, {"id": model_id})
|
||
|
|
if "endpoint" in evidence:
|
||
|
|
raise ValueError(f"duplicate OpenCode Zen docs endpoint row: {model_id}")
|
||
|
|
evidence["endpoint"] = {"name": name, "url": endpoint, "npm": npm}
|
||
|
|
register_name(name, model_id)
|
||
|
|
|
||
|
|
for label, input_price, output_price, cache_read, cache_write in pricing_rows:
|
||
|
|
base_name = _opencode_pricing_base_name(label)
|
||
|
|
pricing_model_id = name_to_id.get(_opencode_zen_name_key(base_name))
|
||
|
|
if pricing_model_id is None:
|
||
|
|
raise ValueError(f"OpenCode Zen docs pricing model is unresolved: {label!r}")
|
||
|
|
row: dict[str, Any] = {
|
||
|
|
"label": label,
|
||
|
|
"input": _opencode_zen_price(input_price, f"OpenCode Zen docs {label}.input"),
|
||
|
|
"output": _opencode_zen_price(output_price, f"OpenCode Zen docs {label}.output"),
|
||
|
|
}
|
||
|
|
read_rate = _opencode_zen_price(cache_read, f"OpenCode Zen docs {label}.cache_read")
|
||
|
|
write_rate = _opencode_zen_price(cache_write, f"OpenCode Zen docs {label}.cache_write")
|
||
|
|
if read_rate is not None:
|
||
|
|
row["cache_read"] = read_rate
|
||
|
|
if write_rate is not None:
|
||
|
|
row["cache_write"] = write_rate
|
||
|
|
pricing = docs_by_id.setdefault(pricing_model_id, {"id": pricing_model_id}).setdefault(
|
||
|
|
"pricing", []
|
||
|
|
)
|
||
|
|
if not isinstance(pricing, list):
|
||
|
|
raise ValueError(
|
||
|
|
f"OpenCode Zen docs pricing evidence is malformed for {pricing_model_id}"
|
||
|
|
)
|
||
|
|
if row not in pricing:
|
||
|
|
pricing.append(row)
|
||
|
|
|
||
|
|
for name, raw_date in deprecation_rows:
|
||
|
|
deprecated_model_id = name_to_id.get(_opencode_zen_name_key(name))
|
||
|
|
if deprecated_model_id is None:
|
||
|
|
# A deprecated, no-longer-inventoried row is irrelevant to the live
|
||
|
|
# join but remains present in the exact source capture.
|
||
|
|
continue
|
||
|
|
try:
|
||
|
|
deprecation_date = datetime.strptime(raw_date, "%B %d, %Y").date().isoformat()
|
||
|
|
except ValueError as exc:
|
||
|
|
raise ValueError(
|
||
|
|
"OpenCode Zen docs model "
|
||
|
|
f"{deprecated_model_id} has invalid deprecation date {raw_date!r}"
|
||
|
|
) from exc
|
||
|
|
evidence = docs_by_id.setdefault(deprecated_model_id, {"id": deprecated_model_id})
|
||
|
|
if "deprecation_date" in evidence:
|
||
|
|
raise ValueError(f"duplicate OpenCode Zen docs deprecation row: {deprecated_model_id}")
|
||
|
|
evidence["deprecation_date"] = deprecation_date
|
||
|
|
return docs_by_id
|
||
|
|
|
||
|
|
|
||
|
|
def _opencode_zen_listing_rows(document: Any) -> list[dict[str, Any]]:
|
||
|
|
root = _object(document, "OpenCode Zen models response")
|
||
|
|
if root.get("object") != "list":
|
||
|
|
raise ValueError("OpenCode Zen models response.object must be 'list'")
|
||
|
|
raw_rows = root.get("data")
|
||
|
|
if not isinstance(raw_rows, list) or not raw_rows:
|
||
|
|
raise ValueError("OpenCode Zen models response.data must be a non-empty list")
|
||
|
|
return [
|
||
|
|
_object(row, f"OpenCode Zen models response.data[{index}]")
|
||
|
|
for index, row in enumerate(raw_rows)
|
||
|
|
]
|
||
|
|
|
||
|
|
|
||
|
|
def _opencode_zen_live_evidence(row: Mapping[str, Any]) -> dict[str, Any]:
|
||
|
|
"""Validate a live row while excluding its request-time ``created`` value.
|
||
|
|
|
||
|
|
Zen currently assigns the same current Unix timestamp to every model on
|
||
|
|
each listing request. The raw response digest retains that exact capture;
|
||
|
|
catalog-record evidence keeps only stable availability fields so a refresh
|
||
|
|
does not manufacture 62 model changes when the ID set is unchanged.
|
||
|
|
"""
|
||
|
|
|
||
|
|
model_id = _nonempty_string(row.get("id"), "OpenCode Zen live model id")
|
||
|
|
if row.get("object") != "model":
|
||
|
|
raise ValueError(f"OpenCode Zen live model {model_id}.object must be 'model'")
|
||
|
|
if row.get("owned_by") != "opencode":
|
||
|
|
raise ValueError(f"OpenCode Zen live model {model_id}.owned_by must be 'opencode'")
|
||
|
|
_positive_int(row.get("created"), f"OpenCode Zen live model {model_id}.created")
|
||
|
|
return {"id": model_id, "object": "model", "owned_by": "opencode"}
|
||
|
|
|
||
|
|
|
||
|
|
def _opencode_zen_metadata(document: Any) -> dict[str, dict[str, Any]]:
|
||
|
|
root = _object(document, "Models.dev response")
|
||
|
|
provider = _object(root.get("opencode"), "Models.dev opencode provider")
|
||
|
|
if provider.get("id") != "opencode":
|
||
|
|
raise ValueError("Models.dev opencode provider has an unexpected id")
|
||
|
|
if provider.get("api") != "https://opencode.ai/zen/v1":
|
||
|
|
raise ValueError("Models.dev opencode provider has an unexpected API base")
|
||
|
|
if provider.get("env") != ["OPENCODE_API_KEY"]:
|
||
|
|
raise ValueError("Models.dev opencode provider has an unexpected credential contract")
|
||
|
|
models = _object(provider.get("models"), "Models.dev opencode models")
|
||
|
|
if not models:
|
||
|
|
raise ValueError("Models.dev opencode models must not be empty")
|
||
|
|
return {
|
||
|
|
_nonempty_string(key, "Models.dev OpenCode Zen model key"): _object(
|
||
|
|
value, f"Models.dev OpenCode Zen model {key}"
|
||
|
|
)
|
||
|
|
for key, value in models.items()
|
||
|
|
}
|
||
|
|
|
||
|
|
|
||
|
|
def _opencode_zen_catalog_record(
|
||
|
|
metadata: Mapping[str, Any], docs_evidence: Mapping[str, Any], *, as_of: str
|
||
|
|
) -> dict[str, Any]:
|
||
|
|
model_id = _nonempty_string(metadata.get("id"), "OpenCode Zen metadata model id")
|
||
|
|
if docs_evidence.get("id") != model_id:
|
||
|
|
raise ValueError(f"OpenCode Zen docs evidence id mismatch for {model_id}")
|
||
|
|
name = _nonempty_string(metadata.get("name"), f"OpenCode Zen model {model_id}.name")
|
||
|
|
description = _nonempty_string(
|
||
|
|
metadata.get("description"), f"OpenCode Zen model {model_id}.description"
|
||
|
|
)
|
||
|
|
limit = _object(metadata.get("limit"), f"OpenCode Zen model {model_id}.limit")
|
||
|
|
context = _positive_int(limit.get("context"), f"OpenCode Zen model {model_id}.limit.context")
|
||
|
|
output = _positive_int(limit.get("output"), f"OpenCode Zen model {model_id}.limit.output")
|
||
|
|
modalities = _object(metadata.get("modalities"), f"OpenCode Zen model {model_id}.modalities")
|
||
|
|
inputs = _string_list(
|
||
|
|
modalities.get("input"),
|
||
|
|
f"OpenCode Zen model {model_id}.modalities.input",
|
||
|
|
allow_empty=False,
|
||
|
|
)
|
||
|
|
outputs = _string_list(
|
||
|
|
modalities.get("output"),
|
||
|
|
f"OpenCode Zen model {model_id}.modalities.output",
|
||
|
|
allow_empty=False,
|
||
|
|
)
|
||
|
|
if "text" not in inputs or outputs != ["text"]:
|
||
|
|
raise ValueError(f"OpenCode Zen model {model_id} lacks the required text contract")
|
||
|
|
tool_call = metadata.get("tool_call")
|
||
|
|
if not isinstance(tool_call, bool):
|
||
|
|
raise ValueError(f"OpenCode Zen model {model_id}.tool_call must be boolean")
|
||
|
|
structured = metadata.get("structured_output", False)
|
||
|
|
if structured is not None and not isinstance(structured, bool):
|
||
|
|
raise ValueError(f"OpenCode Zen model {model_id}.structured_output must be boolean or null")
|
||
|
|
provider_metadata = metadata.get("provider")
|
||
|
|
if provider_metadata is None:
|
||
|
|
npm = "@ai-sdk/openai-compatible"
|
||
|
|
else:
|
||
|
|
npm = _nonempty_string(
|
||
|
|
_object(provider_metadata, f"OpenCode Zen model {model_id}.provider").get("npm"),
|
||
|
|
f"OpenCode Zen model {model_id}.provider.npm",
|
||
|
|
)
|
||
|
|
route_key = OPENCODE_ZEN_ROUTE_BY_NPM.get(npm)
|
||
|
|
if route_key is None:
|
||
|
|
raise ValueError(f"OpenCode Zen model {model_id} has unsupported provider package {npm!r}")
|
||
|
|
endpoint = docs_evidence.get("endpoint")
|
||
|
|
deprecation_date = docs_evidence.get("deprecation_date")
|
||
|
|
if endpoint is not None:
|
||
|
|
documented_npm = _nonempty_string(
|
||
|
|
_object(endpoint, f"OpenCode Zen docs model {model_id}.endpoint").get("npm"),
|
||
|
|
f"OpenCode Zen docs model {model_id}.endpoint.npm",
|
||
|
|
)
|
||
|
|
if documented_npm != npm:
|
||
|
|
raise ValueError(
|
||
|
|
f"OpenCode Zen model {model_id} protocol mismatch: "
|
||
|
|
f"Models.dev={npm}, docs={documented_npm}"
|
||
|
|
)
|
||
|
|
availability = (
|
||
|
|
"live-undocumented" if endpoint is None and deprecation_date is None else "current"
|
||
|
|
)
|
||
|
|
if deprecation_date is not None:
|
||
|
|
deprecation_date = _iso_date(
|
||
|
|
deprecation_date, f"OpenCode Zen docs model {model_id}.deprecation_date"
|
||
|
|
)
|
||
|
|
state = "deprecated" if deprecation_date <= as_of else "deprecation-scheduled"
|
||
|
|
availability = f"{state}-{deprecation_date}-live-listed"
|
||
|
|
executable = route_key != "opencode_zen_google_catalog"
|
||
|
|
upstream_modalities = ["file" if item == "pdf" else item for item in inputs]
|
||
|
|
upstream_modalities = list(dict.fromkeys(upstream_modalities))
|
||
|
|
reasoning_levels, default_reasoning = opencode_zen_reasoning(metadata)
|
||
|
|
catalog: dict[str, Any] = {
|
||
|
|
"maker": infer_model_maker(model_id),
|
||
|
|
"route": route_key,
|
||
|
|
"upstream_id": model_id,
|
||
|
|
"display_name": f"{name} via OpenCode Zen",
|
||
|
|
"description": description,
|
||
|
|
"kind": "chat" if executable or "image" not in upstream_modalities else "vision_chat",
|
||
|
|
"agent_compatible": executable and tool_call,
|
||
|
|
"context_window": context,
|
||
|
|
"max_output_tokens": output,
|
||
|
|
"reasoning_levels": reasoning_levels,
|
||
|
|
"default_reasoning": default_reasoning,
|
||
|
|
"modalities": ["text"] if executable else upstream_modalities,
|
||
|
|
"output_modalities": ["text"],
|
||
|
|
"tool_calling": tool_call,
|
||
|
|
"parallel_tool_calls": False,
|
||
|
|
"supports_reasoning_summaries": False,
|
||
|
|
"structured_output": structured is True,
|
||
|
|
"availability": availability,
|
||
|
|
"capability_confidence": "opencode-maintained-catalog",
|
||
|
|
"source": "models-dev-opencode-zen",
|
||
|
|
"availability_source": "opencode-zen-models",
|
||
|
|
"capability_source": "models-dev-opencode-zen",
|
||
|
|
"pricing_source": "opencode-zen-docs-source",
|
||
|
|
"inventory": "opencode-zen",
|
||
|
|
"resource_group": "opencode_zen",
|
||
|
|
}
|
||
|
|
catalog.update(opencode_zen_catalog_pricing(docs_evidence))
|
||
|
|
return catalog
|
||
|
|
|
||
|
|
|
||
|
|
def build_opencode_zen_snapshot(
|
||
|
|
listing_document: Any,
|
||
|
|
models_dev_document: Any,
|
||
|
|
docs_document: Any,
|
||
|
|
*,
|
||
|
|
as_of: str,
|
||
|
|
retrieved_at: str,
|
||
|
|
listing_sha256: str,
|
||
|
|
models_dev_sha256: str,
|
||
|
|
docs_sha256: str,
|
||
|
|
listing_url: str,
|
||
|
|
models_dev_url: str,
|
||
|
|
docs_url: str,
|
||
|
|
) -> dict[str, Any]:
|
||
|
|
"""Join live Zen availability with capability and provider-doc evidence."""
|
||
|
|
|
||
|
|
as_of = _iso_date(as_of, "OpenCode Zen snapshot as_of")
|
||
|
|
_sha256(listing_sha256, "OpenCode Zen listing SHA-256")
|
||
|
|
_sha256(models_dev_sha256, "Models.dev response SHA-256")
|
||
|
|
_sha256(docs_sha256, "OpenCode Zen docs source SHA-256")
|
||
|
|
_utc_timestamp(retrieved_at, "OpenCode Zen retrieval timestamp")
|
||
|
|
listing_url = _https_url(listing_url, "OpenCode Zen models source URL")
|
||
|
|
models_dev_url = _https_url(models_dev_url, "Models.dev source URL")
|
||
|
|
docs_url = _https_url(docs_url, "OpenCode Zen docs source URL")
|
||
|
|
metadata_by_id = _opencode_zen_metadata(models_dev_document)
|
||
|
|
docs_by_id = _opencode_zen_docs(docs_document, metadata_by_id)
|
||
|
|
rows = _opencode_zen_listing_rows(listing_document)
|
||
|
|
sources = dict(OPENCODE_ZEN_SOURCE_IDS)
|
||
|
|
sources["opencode-zen-models"] = listing_url
|
||
|
|
sources["models-dev-opencode-zen"] = models_dev_url
|
||
|
|
sources["opencode-zen-docs-source"] = docs_url
|
||
|
|
seen_ids: set[str] = set()
|
||
|
|
used: set[str] = set()
|
||
|
|
records: dict[str, dict[str, Any]] = {}
|
||
|
|
for row in sorted(rows, key=lambda value: str(value.get("id", ""))):
|
||
|
|
live_evidence = _opencode_zen_live_evidence(row)
|
||
|
|
model_id = str(live_evidence["id"])
|
||
|
|
if model_id in seen_ids:
|
||
|
|
raise ValueError(f"duplicate OpenCode Zen live model id: {model_id}")
|
||
|
|
seen_ids.add(model_id)
|
||
|
|
metadata = metadata_by_id.get(model_id)
|
||
|
|
if metadata is None:
|
||
|
|
raise ValueError(f"OpenCode Zen live model lacks Models.dev metadata: {model_id}")
|
||
|
|
if metadata.get("id") != model_id:
|
||
|
|
raise ValueError(f"OpenCode Zen Models.dev id mismatch for {model_id}")
|
||
|
|
docs_evidence = docs_by_id.get(
|
||
|
|
model_id,
|
||
|
|
{"id": model_id, "documented_endpoint": False},
|
||
|
|
)
|
||
|
|
catalog_record = _opencode_zen_catalog_record(metadata, docs_evidence, as_of=as_of)
|
||
|
|
key = route_catalog_key(str(catalog_record["route"]), model_id, used)
|
||
|
|
used.add(key)
|
||
|
|
records[key] = {
|
||
|
|
"catalog": catalog_record,
|
||
|
|
"evidence": {
|
||
|
|
"docs": docs_evidence,
|
||
|
|
"live": live_evidence,
|
||
|
|
"models_dev": dict(sorted(metadata.items())),
|
||
|
|
},
|
||
|
|
}
|
||
|
|
return build_inventory_snapshot(
|
||
|
|
inventory="opencode-zen",
|
||
|
|
adapter="opencode_zen_join",
|
||
|
|
fingerprint_fields=FULL_FINGERPRINT_FIELDS,
|
||
|
|
as_of=as_of,
|
||
|
|
dynamic=True,
|
||
|
|
sources=sources,
|
||
|
|
discovery={"endpoint": listing_url},
|
||
|
|
captures=[
|
||
|
|
{
|
||
|
|
"source": "opencode-zen-models",
|
||
|
|
"retrieved_at": retrieved_at,
|
||
|
|
"response_sha256": listing_sha256,
|
||
|
|
},
|
||
|
|
{
|
||
|
|
"source": "models-dev-opencode-zen",
|
||
|
|
"retrieved_at": retrieved_at,
|
||
|
|
"response_sha256": models_dev_sha256,
|
||
|
|
},
|
||
|
|
{
|
||
|
|
"source": "opencode-zen-docs-source",
|
||
|
|
"retrieved_at": retrieved_at,
|
||
|
|
"response_sha256": docs_sha256,
|
||
|
|
},
|
||
|
|
],
|
||
|
|
models=records,
|
||
|
|
)
|
||
|
|
|
||
|
|
|
||
|
|
def _opencode_go_metadata(document: Any) -> dict[str, dict[str, Any]]:
|
||
|
|
root = _object(document, "Models.dev response")
|
||
|
|
access = _object(root.get("opencode-go"), "Models.dev opencode-go access product")
|
||
|
|
if access.get("id") != "opencode-go":
|
||
|
|
raise ValueError("Models.dev opencode-go access product has an unexpected id")
|
||
|
|
if access.get("api") != "https://opencode.ai/zen/go/v1":
|
||
|
|
raise ValueError("Models.dev opencode-go access product has an unexpected API base")
|
||
|
|
if access.get("env") != ["OPENCODE_API_KEY"]:
|
||
|
|
raise ValueError("Models.dev opencode-go has an unexpected credential contract")
|
||
|
|
models = _object(access.get("models"), "Models.dev opencode-go models")
|
||
|
|
if not models:
|
||
|
|
raise ValueError("Models.dev opencode-go models must not be empty")
|
||
|
|
return {
|
||
|
|
_nonempty_string(key, "Models.dev OpenCode Go model key"): _object(
|
||
|
|
value, f"Models.dev OpenCode Go model {key}"
|
||
|
|
)
|
||
|
|
for key, value in models.items()
|
||
|
|
}
|
||
|
|
|
||
|
|
|
||
|
|
def _opencode_go_docs(document: Any) -> dict[str, dict[str, Any]]:
|
||
|
|
if not isinstance(document, str) or not document.strip():
|
||
|
|
raise ValueError("OpenCode Go docs source must be non-empty UTF-8 text")
|
||
|
|
endpoint_rows = _markdown_table(
|
||
|
|
document,
|
||
|
|
"## Endpoints",
|
||
|
|
("Model", "Model ID", "Endpoint", "AI SDK Package"),
|
||
|
|
)
|
||
|
|
pricing_rows = _markdown_table(
|
||
|
|
document,
|
||
|
|
"## Usage limits",
|
||
|
|
("Model", "Input", "Output", "Cached Read", "Cached Write", "Usage"),
|
||
|
|
)
|
||
|
|
docs_by_id: dict[str, dict[str, Any]] = {}
|
||
|
|
names: dict[str, str] = {}
|
||
|
|
for name, model_id, endpoint, npm in endpoint_rows:
|
||
|
|
model_id = _nonempty_string(model_id, "OpenCode Go docs endpoint model id")
|
||
|
|
route = OPENCODE_GO_ROUTE_BY_NPM.get(npm)
|
||
|
|
if route is None:
|
||
|
|
raise ValueError(f"OpenCode Go docs model {model_id} has unsupported package {npm!r}")
|
||
|
|
expected_suffix = {
|
||
|
|
"@ai-sdk/openai": "/responses",
|
||
|
|
"@ai-sdk/openai-compatible": "/chat/completions",
|
||
|
|
"@ai-sdk/anthropic": "/messages",
|
||
|
|
}[npm]
|
||
|
|
endpoint = _https_url(endpoint, f"OpenCode Go docs model {model_id}.endpoint")
|
||
|
|
if endpoint != f"https://opencode.ai/zen/go/v1{expected_suffix}":
|
||
|
|
raise ValueError(
|
||
|
|
f"OpenCode Go docs model {model_id} has unexpected endpoint {endpoint}"
|
||
|
|
)
|
||
|
|
if model_id in docs_by_id:
|
||
|
|
raise ValueError(f"duplicate OpenCode Go docs endpoint row: {model_id}")
|
||
|
|
docs_by_id[model_id] = {
|
||
|
|
"id": model_id,
|
||
|
|
"endpoint": {"name": name, "url": endpoint, "npm": npm},
|
||
|
|
}
|
||
|
|
key = _opencode_zen_name_key(name)
|
||
|
|
previous = names.get(key)
|
||
|
|
if previous is not None and previous != model_id:
|
||
|
|
raise ValueError(f"OpenCode Go docs model name {name!r} is ambiguous")
|
||
|
|
names[key] = model_id
|
||
|
|
|
||
|
|
for label, input_price, output_price, cache_read, cache_write, usage in pricing_rows:
|
||
|
|
base_name = _opencode_pricing_base_name(label)
|
||
|
|
pricing_model_id = names.get(_opencode_zen_name_key(base_name))
|
||
|
|
if pricing_model_id is None:
|
||
|
|
# A price row may use a typography variant; preserve it only when
|
||
|
|
# its endpoint identity is unambiguous.
|
||
|
|
candidates = [
|
||
|
|
candidate
|
||
|
|
for key, candidate in names.items()
|
||
|
|
if key in _opencode_zen_name_key(base_name)
|
||
|
|
or _opencode_zen_name_key(base_name) in key
|
||
|
|
]
|
||
|
|
if len(set(candidates)) != 1:
|
||
|
|
raise ValueError(f"OpenCode Go docs pricing model is unresolved: {label!r}")
|
||
|
|
pricing_model_id = candidates[0]
|
||
|
|
row: dict[str, Any] = {
|
||
|
|
"label": label,
|
||
|
|
"input": _opencode_zen_price(input_price, f"OpenCode Go docs {label}.input"),
|
||
|
|
"output": _opencode_zen_price(output_price, f"OpenCode Go docs {label}.output"),
|
||
|
|
"usage": _opencode_zen_price(usage, f"OpenCode Go docs {label}.usage"),
|
||
|
|
}
|
||
|
|
read_rate = _opencode_zen_price(cache_read, f"OpenCode Go docs {label}.cache_read")
|
||
|
|
write_rate = _opencode_zen_price(cache_write, f"OpenCode Go docs {label}.cache_write")
|
||
|
|
if read_rate is not None:
|
||
|
|
row["cache_read"] = read_rate
|
||
|
|
if write_rate is not None:
|
||
|
|
row["cache_write"] = write_rate
|
||
|
|
docs_by_id[pricing_model_id].setdefault("pricing", []).append(row)
|
||
|
|
return docs_by_id
|
||
|
|
|
||
|
|
|
||
|
|
def _opencode_go_catalog_record(
|
||
|
|
metadata: Mapping[str, Any], docs_evidence: Mapping[str, Any] | None
|
||
|
|
) -> dict[str, Any]:
|
||
|
|
model_id = _nonempty_string(metadata.get("id"), "OpenCode Go metadata model id")
|
||
|
|
name = _nonempty_string(metadata.get("name"), f"OpenCode Go model {model_id}.name")
|
||
|
|
description = _nonempty_string(
|
||
|
|
metadata.get("description"), f"OpenCode Go model {model_id}.description"
|
||
|
|
)
|
||
|
|
limit = _object(metadata.get("limit"), f"OpenCode Go model {model_id}.limit")
|
||
|
|
context = _positive_int(limit.get("context"), f"OpenCode Go model {model_id}.limit.context")
|
||
|
|
output = _positive_int(limit.get("output"), f"OpenCode Go model {model_id}.limit.output")
|
||
|
|
modalities = _object(metadata.get("modalities"), f"OpenCode Go model {model_id}.modalities")
|
||
|
|
inputs = _string_list(
|
||
|
|
modalities.get("input"), f"OpenCode Go model {model_id}.modalities.input", allow_empty=False
|
||
|
|
)
|
||
|
|
outputs = _string_list(
|
||
|
|
modalities.get("output"),
|
||
|
|
f"OpenCode Go model {model_id}.modalities.output",
|
||
|
|
allow_empty=False,
|
||
|
|
)
|
||
|
|
if "text" not in inputs or outputs != ["text"]:
|
||
|
|
raise ValueError(f"OpenCode Go model {model_id} lacks the required text contract")
|
||
|
|
tool_call = metadata.get("tool_call")
|
||
|
|
if not isinstance(tool_call, bool):
|
||
|
|
raise ValueError(f"OpenCode Go model {model_id}.tool_call must be boolean")
|
||
|
|
structured = metadata.get("structured_output", False)
|
||
|
|
if structured is not None and not isinstance(structured, bool):
|
||
|
|
raise ValueError(f"OpenCode Go model {model_id}.structured_output is invalid")
|
||
|
|
npm, _protocol_resolution = _opencode_go_protocol(metadata, docs_evidence)
|
||
|
|
route_key = OPENCODE_GO_ROUTE_BY_NPM.get(npm)
|
||
|
|
if route_key is None:
|
||
|
|
raise ValueError(f"OpenCode Go model {model_id} has unsupported package {npm!r}")
|
||
|
|
reasoning_levels, default_reasoning = opencode_zen_reasoning(metadata)
|
||
|
|
record: dict[str, Any] = {
|
||
|
|
"maker": infer_model_maker(model_id),
|
||
|
|
"route": route_key,
|
||
|
|
"upstream_id": model_id,
|
||
|
|
"display_name": f"{name} via OpenCode Go",
|
||
|
|
"description": description,
|
||
|
|
"kind": "chat",
|
||
|
|
"agent_compatible": tool_call,
|
||
|
|
"context_window": context,
|
||
|
|
"max_output_tokens": output,
|
||
|
|
"reasoning_levels": reasoning_levels,
|
||
|
|
"default_reasoning": default_reasoning,
|
||
|
|
# The current Switchyard Go transports are text-only even when the
|
||
|
|
# underlying model record advertises richer modalities. The complete
|
||
|
|
# upstream modality set remains in fingerprint evidence.
|
||
|
|
"modalities": ["text"],
|
||
|
|
"output_modalities": ["text"],
|
||
|
|
"tool_calling": tool_call,
|
||
|
|
"parallel_tool_calls": False,
|
||
|
|
"supports_reasoning_summaries": False,
|
||
|
|
"structured_output": structured is True,
|
||
|
|
"availability": "current",
|
||
|
|
"capability_confidence": "opencode-maintained-catalog",
|
||
|
|
"source": "models-dev-opencode-go",
|
||
|
|
"availability_source": "opencode-go-models",
|
||
|
|
"capability_source": "models-dev-opencode-go",
|
||
|
|
"pricing_source": "opencode-go-docs-source",
|
||
|
|
"inventory": "opencode-go",
|
||
|
|
"resource_group": "opencode_go",
|
||
|
|
}
|
||
|
|
# Provider documentation is authoritative for the Go access product.
|
||
|
|
# Time- or context-tiered tables remain exact evidence but cannot be
|
||
|
|
# flattened into the catalog's scalar per-million estimate fields.
|
||
|
|
if docs_evidence is not None:
|
||
|
|
record.update(opencode_zen_catalog_pricing(docs_evidence))
|
||
|
|
return record
|
||
|
|
|
||
|
|
|
||
|
|
def _opencode_go_protocol(
|
||
|
|
metadata: Mapping[str, Any], docs_evidence: Mapping[str, Any] | None
|
||
|
|
) -> tuple[str, dict[str, Any]]:
|
||
|
|
"""Resolve Go's wire protocol and retain disagreements as evidence.
|
||
|
|
|
||
|
|
OpenCode's Go endpoint table is the operator-owned transport contract. The
|
||
|
|
Models.dev package identifies the client adapter OpenCode currently uses,
|
||
|
|
but it can temporarily lag or lead the endpoint table. A disagreement must
|
||
|
|
remain visible in the fingerprint; it must not make an explicitly
|
||
|
|
documented endpoint impossible to represent.
|
||
|
|
"""
|
||
|
|
|
||
|
|
model_id = _nonempty_string(metadata.get("id"), "OpenCode Go metadata model id")
|
||
|
|
provider_metadata = metadata.get("provider")
|
||
|
|
models_dev_npm = (
|
||
|
|
"@ai-sdk/openai-compatible"
|
||
|
|
if provider_metadata is None
|
||
|
|
else _nonempty_string(
|
||
|
|
_object(provider_metadata, f"OpenCode Go model {model_id}.provider").get("npm"),
|
||
|
|
f"OpenCode Go model {model_id}.provider.npm",
|
||
|
|
)
|
||
|
|
)
|
||
|
|
if models_dev_npm not in OPENCODE_GO_ROUTE_BY_NPM:
|
||
|
|
raise ValueError(
|
||
|
|
f"OpenCode Go model {model_id} has unsupported Models.dev package {models_dev_npm!r}"
|
||
|
|
)
|
||
|
|
|
||
|
|
documented_npm: str | None = None
|
||
|
|
if docs_evidence is not None:
|
||
|
|
endpoint = _object(
|
||
|
|
docs_evidence.get("endpoint"), f"OpenCode Go docs model {model_id}.endpoint"
|
||
|
|
)
|
||
|
|
documented_npm = _nonempty_string(
|
||
|
|
endpoint.get("npm"), f"OpenCode Go docs model {model_id}.endpoint.npm"
|
||
|
|
)
|
||
|
|
selected_npm = documented_npm or models_dev_npm
|
||
|
|
if selected_npm not in OPENCODE_GO_ROUTE_BY_NPM:
|
||
|
|
raise ValueError(
|
||
|
|
f"OpenCode Go model {model_id} has unsupported documented package {selected_npm!r}"
|
||
|
|
)
|
||
|
|
return selected_npm, {
|
||
|
|
"authority": "opencode-go-docs-source" if documented_npm else "models-dev-opencode-go",
|
||
|
|
"models_dev_npm": models_dev_npm,
|
||
|
|
"documented_npm": documented_npm,
|
||
|
|
"selected_npm": selected_npm,
|
||
|
|
"disagreement": documented_npm is not None and documented_npm != models_dev_npm,
|
||
|
|
}
|
||
|
|
|
||
|
|
|
||
|
|
def build_opencode_go_snapshot(
|
||
|
|
listing_document: Any,
|
||
|
|
models_dev_document: Any,
|
||
|
|
docs_document: Any,
|
||
|
|
*,
|
||
|
|
as_of: str,
|
||
|
|
retrieved_at: str,
|
||
|
|
listing_sha256: str,
|
||
|
|
models_dev_sha256: str,
|
||
|
|
docs_sha256: str,
|
||
|
|
listing_url: str,
|
||
|
|
models_dev_url: str,
|
||
|
|
docs_url: str,
|
||
|
|
) -> dict[str, Any]:
|
||
|
|
"""Join Go availability, Models.dev capabilities, and operator docs."""
|
||
|
|
|
||
|
|
as_of = _iso_date(as_of, "OpenCode Go snapshot as_of")
|
||
|
|
_sha256(listing_sha256, "OpenCode Go listing SHA-256")
|
||
|
|
_sha256(models_dev_sha256, "Models.dev response SHA-256")
|
||
|
|
_sha256(docs_sha256, "OpenCode Go docs source SHA-256")
|
||
|
|
_utc_timestamp(retrieved_at, "OpenCode Go retrieval timestamp")
|
||
|
|
listing_url = _https_url(listing_url, "OpenCode Go models source URL")
|
||
|
|
models_dev_url = _https_url(models_dev_url, "Models.dev source URL")
|
||
|
|
docs_url = _https_url(docs_url, "OpenCode Go docs source URL")
|
||
|
|
metadata_by_id = _opencode_go_metadata(models_dev_document)
|
||
|
|
docs_by_id = _opencode_go_docs(docs_document)
|
||
|
|
rows = _opencode_zen_listing_rows(listing_document)
|
||
|
|
sources = dict(OPENCODE_GO_SOURCE_IDS)
|
||
|
|
sources["opencode-go-models"] = listing_url
|
||
|
|
sources["models-dev-opencode-go"] = models_dev_url
|
||
|
|
sources["opencode-go-docs-source"] = docs_url
|
||
|
|
seen_ids: set[str] = set()
|
||
|
|
used: set[str] = set()
|
||
|
|
records: dict[str, dict[str, Any]] = {}
|
||
|
|
for row in sorted(rows, key=lambda value: str(value.get("id", ""))):
|
||
|
|
live_evidence = _opencode_zen_live_evidence(row)
|
||
|
|
model_id = str(live_evidence["id"])
|
||
|
|
if model_id in seen_ids:
|
||
|
|
raise ValueError(f"duplicate OpenCode Go live model id: {model_id}")
|
||
|
|
seen_ids.add(model_id)
|
||
|
|
metadata = metadata_by_id.get(model_id)
|
||
|
|
if metadata is None:
|
||
|
|
# Keep the live ID visible in the comparison, but do not infer its
|
||
|
|
# capabilities. The coverage marker makes the verification fail
|
||
|
|
# with an actionable per-field report.
|
||
|
|
route_key = "opencode_go_openai_chat"
|
||
|
|
key = route_catalog_key(route_key, model_id, used)
|
||
|
|
used.add(key)
|
||
|
|
records[key] = {
|
||
|
|
"catalog": {
|
||
|
|
"maker": infer_model_maker(model_id),
|
||
|
|
"route": route_key,
|
||
|
|
"upstream_id": model_id,
|
||
|
|
"display_name": f"Unverified live Go model {model_id}",
|
||
|
|
"description": "Live ID without capability fingerprint evidence",
|
||
|
|
"kind": "chat",
|
||
|
|
"agent_compatible": False,
|
||
|
|
"context_window": 0,
|
||
|
|
"reasoning_levels": ["none"],
|
||
|
|
"default_reasoning": "none",
|
||
|
|
"modalities": ["text"],
|
||
|
|
"output_modalities": ["text"],
|
||
|
|
"tool_calling": False,
|
||
|
|
"parallel_tool_calls": False,
|
||
|
|
"supports_reasoning_summaries": False,
|
||
|
|
"structured_output": False,
|
||
|
|
"availability": "live-unverified",
|
||
|
|
"capability_confidence": "id-only",
|
||
|
|
"source": "opencode-go-models",
|
||
|
|
"availability_source": "opencode-go-models",
|
||
|
|
"capability_source": "opencode-go-models",
|
||
|
|
"pricing_source": "opencode-go-models",
|
||
|
|
"inventory": "opencode-go",
|
||
|
|
"resource_group": "opencode_go",
|
||
|
|
},
|
||
|
|
"evidence": {
|
||
|
|
"live": live_evidence,
|
||
|
|
"unverified_live_only": True,
|
||
|
|
},
|
||
|
|
}
|
||
|
|
continue
|
||
|
|
if metadata.get("id") != model_id:
|
||
|
|
raise ValueError(f"OpenCode Go Models.dev id mismatch for {model_id}")
|
||
|
|
docs_evidence = docs_by_id.get(model_id)
|
||
|
|
catalog_record = _opencode_go_catalog_record(metadata, docs_evidence)
|
||
|
|
_selected_npm, protocol_resolution = _opencode_go_protocol(metadata, docs_evidence)
|
||
|
|
key = route_catalog_key(str(catalog_record["route"]), model_id, used)
|
||
|
|
used.add(key)
|
||
|
|
records[key] = {
|
||
|
|
"catalog": catalog_record,
|
||
|
|
"evidence": {
|
||
|
|
"docs": docs_evidence or {"id": model_id, "documented_endpoint": False},
|
||
|
|
"live": live_evidence,
|
||
|
|
"models_dev": dict(sorted(metadata.items())),
|
||
|
|
"protocol_resolution": protocol_resolution,
|
||
|
|
},
|
||
|
|
}
|
||
|
|
return build_inventory_snapshot(
|
||
|
|
inventory="opencode-go",
|
||
|
|
adapter="opencode_go_join",
|
||
|
|
fingerprint_fields=FULL_FINGERPRINT_FIELDS,
|
||
|
|
as_of=as_of,
|
||
|
|
dynamic=True,
|
||
|
|
sources=sources,
|
||
|
|
discovery={"endpoint": listing_url},
|
||
|
|
captures=[
|
||
|
|
{
|
||
|
|
"source": "opencode-go-models",
|
||
|
|
"retrieved_at": retrieved_at,
|
||
|
|
"response_sha256": listing_sha256,
|
||
|
|
},
|
||
|
|
{
|
||
|
|
"source": "models-dev-opencode-go",
|
||
|
|
"retrieved_at": retrieved_at,
|
||
|
|
"response_sha256": models_dev_sha256,
|
||
|
|
},
|
||
|
|
{
|
||
|
|
"source": "opencode-go-docs-source",
|
||
|
|
"retrieved_at": retrieved_at,
|
||
|
|
"response_sha256": docs_sha256,
|
||
|
|
},
|
||
|
|
],
|
||
|
|
models=records,
|
||
|
|
)
|
||
|
|
|
||
|
|
|
||
|
|
def codex_runtime_evidence(value: Mapping[str, Any]) -> dict[str, Any]:
|
||
|
|
"""Normalize only installed-Codex metadata relevant to route execution."""
|
||
|
|
|
||
|
|
row = _object(value, "Codex model metadata")
|
||
|
|
slug = _nonempty_string(row.get("slug"), "Codex model slug")
|
||
|
|
levels_raw = row.get("supported_reasoning_levels")
|
||
|
|
if not isinstance(levels_raw, list) or not levels_raw:
|
||
|
|
raise ValueError(f"Codex model {slug} lacks supported reasoning levels")
|
||
|
|
levels: list[str] = []
|
||
|
|
for index, item in enumerate(levels_raw):
|
||
|
|
effort = item.get("effort") if isinstance(item, Mapping) else item
|
||
|
|
effort = _nonempty_string(effort, f"Codex model {slug} reasoning level {index}")
|
||
|
|
if effort not in CODEX_REASONING_ORDER or effort in levels:
|
||
|
|
raise ValueError(f"Codex model {slug} has invalid reasoning effort {effort!r}")
|
||
|
|
levels.append(effort)
|
||
|
|
default = _nonempty_string(
|
||
|
|
row.get("default_reasoning_level"), f"Codex model {slug} default reasoning level"
|
||
|
|
)
|
||
|
|
if default not in levels:
|
||
|
|
raise ValueError(f"Codex model {slug} default reasoning level is unsupported")
|
||
|
|
context = _positive_int(row.get("context_window"), f"Codex model {slug}.context_window")
|
||
|
|
modalities = _string_list(
|
||
|
|
row.get("input_modalities"), f"Codex model {slug}.input_modalities", allow_empty=False
|
||
|
|
)
|
||
|
|
if set(modalities) - {"text", "image", "audio", "video", "file"}:
|
||
|
|
raise ValueError(f"Codex model {slug} has unsupported input modalities")
|
||
|
|
parallel = row.get("supports_parallel_tool_calls")
|
||
|
|
if parallel is None:
|
||
|
|
# The bundled source omits this derived installed-client field. Codex's
|
||
|
|
# shell/apply-patch harness is the authoritative positive capability.
|
||
|
|
parallel = bool(
|
||
|
|
row.get("tool_mode") == "code_mode_only"
|
||
|
|
or row.get("shell_type")
|
||
|
|
or row.get("apply_patch_tool_type")
|
||
|
|
)
|
||
|
|
if not isinstance(parallel, bool):
|
||
|
|
raise ValueError(f"Codex model {slug}.supports_parallel_tool_calls is invalid")
|
||
|
|
summaries = row.get("supports_reasoning_summaries")
|
||
|
|
if summaries is None:
|
||
|
|
summaries = "default_reasoning_summary" in row or row.get(
|
||
|
|
"reasoning_summary_format"
|
||
|
|
) not in (None, "none")
|
||
|
|
if not isinstance(summaries, bool):
|
||
|
|
raise ValueError(f"Codex model {slug}.supports_reasoning_summaries is invalid")
|
||
|
|
return {
|
||
|
|
"slug": slug,
|
||
|
|
"comp_hash": row.get("comp_hash"),
|
||
|
|
"context_window": context,
|
||
|
|
"input_modalities": modalities,
|
||
|
|
"reasoning": {"levels": levels, "default": default, "summaries": summaries},
|
||
|
|
"tools": {
|
||
|
|
"shell_type": row.get("shell_type"),
|
||
|
|
"tool_mode": row.get("tool_mode"),
|
||
|
|
"apply_patch_tool_type": row.get("apply_patch_tool_type"),
|
||
|
|
"parallel_tool_calls": parallel,
|
||
|
|
"search": row.get("supports_search_tool"),
|
||
|
|
},
|
||
|
|
"structured_output": False,
|
||
|
|
"visibility": row.get("visibility"),
|
||
|
|
"supported_in_api": row.get("supported_in_api"),
|
||
|
|
"multi_agent_version": row.get("multi_agent_version"),
|
||
|
|
"service_tiers": row.get("service_tiers", []),
|
||
|
|
}
|
||
|
|
|
||
|
|
|
||
|
|
def _codex_model_rows(document: Any) -> list[dict[str, Any]]:
|
||
|
|
root = _object(document, "Codex models document")
|
||
|
|
rows = root.get("models")
|
||
|
|
if not isinstance(rows, list) or not rows:
|
||
|
|
raise ValueError("Codex models document.models must be a non-empty list")
|
||
|
|
return [
|
||
|
|
_object(row, f"Codex models document.models[{index}]") for index, row in enumerate(rows)
|
||
|
|
]
|
||
|
|
|
||
|
|
|
||
|
|
def build_codex_installed_snapshot(
|
||
|
|
document: Any,
|
||
|
|
reviewed_snapshot: Any,
|
||
|
|
*,
|
||
|
|
as_of: str,
|
||
|
|
retrieved_at: str,
|
||
|
|
response_sha256: str,
|
||
|
|
source_url: str,
|
||
|
|
) -> dict[str, Any]:
|
||
|
|
"""Join a reviewed Codex catalog with the exact bundled client metadata."""
|
||
|
|
|
||
|
|
reviewed = validate_inventory_snapshot(reviewed_snapshot, expected_inventory="openai-codex")
|
||
|
|
_sha256(response_sha256, "Codex models source SHA-256")
|
||
|
|
_utc_timestamp(retrieved_at, "Codex models retrieval timestamp")
|
||
|
|
source_url = _https_url(source_url, "Codex models source URL")
|
||
|
|
rows = {
|
||
|
|
str(evidence["slug"]): evidence
|
||
|
|
for evidence in (codex_runtime_evidence(row) for row in _codex_model_rows(document))
|
||
|
|
}
|
||
|
|
records: dict[str, dict[str, Any]] = {}
|
||
|
|
for key, old_record in reviewed["models"].items():
|
||
|
|
catalog = dict(old_record["catalog"])
|
||
|
|
slug = str(catalog["upstream_id"])
|
||
|
|
evidence = rows.get(slug)
|
||
|
|
if evidence is None:
|
||
|
|
raise ValueError(f"reviewed Codex model is absent from bundled source: {slug}")
|
||
|
|
catalog.update(
|
||
|
|
{
|
||
|
|
"capability_confidence": (f"codex-{APP_SERVER_PROTOCOL_CODEX_VERSION}-baseline"),
|
||
|
|
"context_window": evidence["context_window"],
|
||
|
|
"reasoning_levels": evidence["reasoning"]["levels"],
|
||
|
|
"default_reasoning": evidence["reasoning"]["default"],
|
||
|
|
"modalities": evidence["input_modalities"],
|
||
|
|
"parallel_tool_calls": evidence["tools"]["parallel_tool_calls"],
|
||
|
|
"supports_reasoning_summaries": evidence["reasoning"]["summaries"],
|
||
|
|
"structured_output": evidence["structured_output"],
|
||
|
|
}
|
||
|
|
)
|
||
|
|
# The installed catalog does not expose a separate maximum output
|
||
|
|
# limit; carrying an API-product limit here would be false precision.
|
||
|
|
catalog.pop("max_output_tokens", None)
|
||
|
|
records[key] = {
|
||
|
|
"catalog": catalog,
|
||
|
|
"evidence": {
|
||
|
|
"codex_runtime": evidence,
|
||
|
|
"verified_fingerprint_fields": list(FULL_FINGERPRINT_FIELDS),
|
||
|
|
},
|
||
|
|
}
|
||
|
|
sources = dict(reviewed["sources"])
|
||
|
|
sources["openai-codex-client-models"] = source_url
|
||
|
|
return build_inventory_snapshot(
|
||
|
|
inventory="openai-codex",
|
||
|
|
adapter="codex_installed_models_join",
|
||
|
|
fingerprint_fields=FULL_FINGERPRINT_FIELDS,
|
||
|
|
as_of=as_of,
|
||
|
|
dynamic=False,
|
||
|
|
sources=sources,
|
||
|
|
discovery=reviewed["discovery"],
|
||
|
|
captures=[
|
||
|
|
{
|
||
|
|
"source": "openai-codex-client-models",
|
||
|
|
"retrieved_at": retrieved_at,
|
||
|
|
"response_sha256": response_sha256,
|
||
|
|
}
|
||
|
|
],
|
||
|
|
models=records,
|
||
|
|
)
|