Files
2026-08-24 08:11:59 -07:00

448 lines
18 KiB
Python
Executable File

#!/usr/bin/env python3
"""Small JSON-schema subset used for worker output contracts.
The project intentionally avoids a runtime dependency on jsonschema. Profiles
may use the supported, auditable subset documented in PROFILE_SCHEMA.md.
"""
from __future__ import annotations
import json
import math
import re
from collections.abc import Mapping
from typing import Any
from mmo_util import (
strict_json_decoder,
strict_json_loads,
valid_absolute_uri,
validate_json_unicode,
)
SUPPORTED_TYPES = {"object", "array", "string", "integer", "number", "boolean", "null"}
SUPPORTED_FORMATS = {"uri", "date", "date-time"}
_RFC3339_DATE = re.compile(r"([0-9]{4})-([0-9]{2})-([0-9]{2})", re.ASCII)
_RFC3339_DATE_TIME = re.compile(
r"([0-9]{4})-([0-9]{2})-([0-9]{2})"
r"[Tt]([0-9]{2}):([0-9]{2}):([0-9]{2})"
r"(?:\.([0-9]+))?"
r"(?:([Zz])|([+-])([0-9]{2}):([0-9]{2}))",
re.ASCII,
)
def _is_finite_number(value: Any) -> bool:
if isinstance(value, bool) or not isinstance(value, (int, float)):
return False
# Converting an arbitrarily large JSON integer to float can overflow even
# though the integer itself is finite.
return isinstance(value, int) or math.isfinite(value)
def _is_integer_number(value: Any) -> bool:
"""Return whether a JSON number has a zero fractional part."""
return _is_finite_number(value) and (
isinstance(value, int) or (isinstance(value, float) and value.is_integer())
)
def _json_value_key(value: Any) -> tuple[Any, ...] | None:
"""Return a strict JSON value key using JSON Schema equality rules."""
if value is None:
return ("null",)
if isinstance(value, bool):
return ("boolean", value)
if isinstance(value, int):
return ("number", value)
if isinstance(value, float):
return ("number", value) if math.isfinite(value) else None
if isinstance(value, str):
return ("string", value)
if isinstance(value, list):
items = [_json_value_key(item) for item in value]
if any(item is None for item in items):
return None
return ("array", tuple(items))
if isinstance(value, Mapping):
if not all(isinstance(key, str) for key in value):
return None
members = [(key, _json_value_key(value[key])) for key in sorted(value)]
if any(item is None for _key, item in members):
return None
return ("object", tuple(members))
return None
def validate_schema_definition(schema: Any, path: str = "$") -> list[str]:
errors: list[str] = []
if isinstance(schema, bool):
return errors
if not isinstance(schema, Mapping):
return [f"{path}: schema must be an object or boolean"]
if not all(isinstance(key, str) for key in schema):
errors.append(f"{path}: schema keyword names must be strings")
schema_type = schema.get("type")
if schema_type is not None:
if isinstance(schema_type, str):
if schema_type not in SUPPORTED_TYPES:
errors.append(f"{path}.type: unsupported type {schema_type!r}")
elif isinstance(schema_type, list):
if not schema_type:
errors.append(f"{path}.type: array cannot be empty")
invalid = [
item
for item in schema_type
if not isinstance(item, str) or item not in SUPPORTED_TYPES
]
if invalid:
errors.append(f"{path}.type: unsupported types {invalid!r}")
elif len(schema_type) != len(set(schema_type)):
errors.append(f"{path}.type: array contains duplicate types")
else:
errors.append(f"{path}.type: must be a string or array")
for keyword in ("oneOf", "anyOf", "allOf"):
if keyword in schema:
value = schema[keyword]
if not isinstance(value, list) or not value:
errors.append(f"{path}.{keyword}: must be a non-empty array")
else:
for index, item in enumerate(value):
errors.extend(validate_schema_definition(item, f"{path}.{keyword}[{index}]"))
for keyword in ("not", "if", "then", "else"):
if keyword in schema:
errors.extend(validate_schema_definition(schema[keyword], f"{path}.{keyword}"))
if ("then" in schema or "else" in schema) and "if" not in schema:
errors.append(f"{path}: then/else requires if")
if "properties" in schema:
properties = schema["properties"]
if not isinstance(properties, Mapping):
errors.append(f"{path}.properties: must be an object")
else:
for key, value in properties.items():
if not isinstance(key, str):
errors.append(f"{path}.properties: property names must be strings")
continue
errors.extend(validate_schema_definition(value, f"{path}.properties.{key}"))
if "items" in schema:
errors.extend(validate_schema_definition(schema["items"], f"{path}.items"))
if "required" in schema:
required = schema["required"]
if not (isinstance(required, list) and all(isinstance(item, str) for item in required)):
errors.append(f"{path}.required: must be an array of strings")
elif len(required) != len(set(required)):
errors.append(f"{path}.required: contains duplicate property names")
if "enum" in schema:
enum = schema["enum"]
if not isinstance(enum, list) or not enum:
errors.append(f"{path}.enum: must be a non-empty array")
else:
keys = [_json_value_key(item) for item in enum]
if any(key is None for key in keys):
errors.append(f"{path}.enum: values must be valid finite JSON values")
elif len(keys) != len(set(keys)):
errors.append(f"{path}.enum: values must be unique")
if "const" in schema and _json_value_key(schema["const"]) is None:
errors.append(f"{path}.const: must be a valid finite JSON value")
for keyword in ("$schema", "$id", "title", "description"):
if keyword in schema and not isinstance(schema[keyword], str):
errors.append(f"{path}.{keyword}: must be a string")
for keyword in ("minLength", "maxLength", "minItems", "maxItems"):
if keyword in schema:
value = schema[keyword]
if not _is_integer_number(value) or value < 0:
errors.append(f"{path}.{keyword}: must be a non-negative integer")
for minimum, maximum in (("minLength", "maxLength"), ("minItems", "maxItems")):
if (
_is_integer_number(schema.get(minimum))
and _is_integer_number(schema.get(maximum))
and schema[minimum] > schema[maximum]
):
errors.append(f"{path}: {minimum} exceeds {maximum}")
for keyword in ("minimum", "maximum"):
if keyword in schema:
value = schema[keyword]
if not _is_finite_number(value):
errors.append(f"{path}.{keyword}: must be a finite number")
if (
isinstance(schema.get("minimum"), (int, float))
and not isinstance(schema.get("minimum"), bool)
and isinstance(schema.get("maximum"), (int, float))
and not isinstance(schema.get("maximum"), bool)
and schema["minimum"] > schema["maximum"]
):
errors.append(f"{path}: minimum exceeds maximum")
if "uniqueItems" in schema and not isinstance(schema["uniqueItems"], bool):
errors.append(f"{path}.uniqueItems: must be boolean")
if "additionalProperties" in schema:
additional = schema["additionalProperties"]
if isinstance(additional, Mapping):
errors.extend(validate_schema_definition(additional, f"{path}.additionalProperties"))
elif not isinstance(additional, bool):
errors.append(f"{path}.additionalProperties: must be boolean or a schema")
if "pattern" in schema:
pattern = schema["pattern"]
if not isinstance(pattern, str):
errors.append(f"{path}.pattern: must be a string")
else:
try:
re.compile(pattern)
except re.error as exc:
errors.append(f"{path}.pattern: invalid regular expression: {exc}")
if "format" in schema:
schema_format = schema["format"]
if not isinstance(schema_format, str) or schema_format not in SUPPORTED_FORMATS:
errors.append(f"{path}.format: must be one of {sorted(SUPPORTED_FORMATS)}")
known = {
"$schema",
"$id",
"title",
"description",
"type",
"properties",
"required",
"additionalProperties",
"items",
"enum",
"const",
"minLength",
"maxLength",
"minimum",
"maximum",
"minItems",
"maxItems",
"uniqueItems",
"pattern",
"oneOf",
"anyOf",
"allOf",
"not",
"if",
"then",
"else",
"format",
}
unknown = sorted(key for key in schema if isinstance(key, str) and key not in known)
if unknown:
errors.append(f"{path}: unsupported schema keywords: {', '.join(unknown)}")
return errors
def _types(schema: Mapping[str, Any]) -> set[str] | None:
value = schema.get("type")
if value is None:
return None
return {value} if isinstance(value, str) else set(value)
def _matches_type(value: Any, expected: str) -> bool:
if expected == "null":
return value is None
if expected == "boolean":
return isinstance(value, bool)
if expected == "integer":
return _is_integer_number(value)
if expected == "number":
return _is_finite_number(value)
if expected == "string":
return isinstance(value, str)
if expected == "array":
return isinstance(value, list)
if expected == "object":
return isinstance(value, Mapping)
return False
def _rfc3339_month_days(year: int, month: int) -> int:
if not 1 <= month <= 12:
return 0
leap = year % 4 == 0 and (year % 100 != 0 or year % 400 == 0)
month_days = (31, 29 if leap else 28, 31, 30, 31, 30, 31, 31, 30, 31, 30, 31)
return month_days[month - 1]
def _valid_rfc3339_date(year: int, month: int, day: int) -> bool:
return 1 <= day <= _rfc3339_month_days(year, month)
def _shift_rfc3339_date(year: int, month: int, day: int, day_delta: int) -> tuple[int, int, int]:
"""Shift a valid RFC 3339 date by the at-most-one-day offset boundary."""
if day_delta == -1:
if day > 1:
return year, month, day - 1
if month > 1:
month -= 1
else:
year -= 1
month = 12
return year, month, _rfc3339_month_days(year, month)
if day_delta == 1:
if day < _rfc3339_month_days(year, month):
return year, month, day + 1
if month < 12:
return year, month + 1, 1
return year + 1, 1, 1
return year, month, day
def _matches_rfc3339_date(value: str) -> bool:
matched = _RFC3339_DATE.fullmatch(value)
return matched is not None and _valid_rfc3339_date(
int(matched.group(1)), int(matched.group(2)), int(matched.group(3))
)
def _matches_rfc3339_date_time(value: str) -> bool:
matched = _RFC3339_DATE_TIME.fullmatch(value)
if matched is None:
return False
year, month, day, hour, minute, second = map(int, matched.groups()[:6])
if not _valid_rfc3339_date(year, month, day) or hour > 23 or minute > 59 or second > 60:
return False
if matched.group(8) is not None:
offset_minutes = 0
else:
offset_hour = int(matched.group(10))
offset_minute = int(matched.group(11))
if offset_hour > 23 or offset_minute > 59:
return False
offset_minutes = offset_hour * 60 + offset_minute
if matched.group(9) == "-":
offset_minutes = -offset_minutes
if second == 60:
day_delta, utc_minute = divmod(hour * 60 + minute - offset_minutes, 24 * 60)
utc_year, utc_month, utc_day = _shift_rfc3339_date(year, month, day, day_delta)
# RFC 3339 section 5.7 allows :60 only at the end of a month in UTC.
# The local spelling may fall on the adjacent date after applying its
# numeric offset, so validate the shifted UTC calendar date.
return utc_minute == 23 * 60 + 59 and utc_day == _rfc3339_month_days(utc_year, utc_month)
return True
def _matches_format(value: str, schema_format: str) -> bool:
if schema_format == "uri":
return valid_absolute_uri(value)
if schema_format == "date":
return _matches_rfc3339_date(value)
if schema_format == "date-time":
return _matches_rfc3339_date_time(value)
return False
def validate_instance(
value: Any,
schema: Mapping[str, Any] | bool,
path: str = "$",
) -> list[str]:
if schema is True:
return []
if schema is False:
return [f"{path}: value is rejected by false schema"]
errors: list[str] = []
allowed_types = _types(schema)
if allowed_types and not any(_matches_type(value, item) for item in allowed_types):
return [
f"{path}: expected {' or '.join(sorted(allowed_types))}, got {type(value).__name__}"
]
if "const" in schema:
value_key = _json_value_key(value)
const_key = _json_value_key(schema["const"])
if value_key is None or const_key is None or value_key != const_key:
errors.append(f"{path}: expected constant {schema['const']!r}")
if "enum" in schema:
value_key = _json_value_key(value)
enum_keys = {_json_value_key(item) for item in schema["enum"]}
if value_key is None or value_key not in enum_keys:
errors.append(f"{path}: value {value!r} is not in the allowed enum")
if "oneOf" in schema:
matches = [not validate_instance(value, item, path) for item in schema["oneOf"]]
if sum(matches) != 1:
errors.append(f"{path}: value must match exactly one oneOf schema")
if "anyOf" in schema:
if not any(not validate_instance(value, item, path) for item in schema["anyOf"]):
errors.append(f"{path}: value does not match any anyOf schema")
if "allOf" in schema:
for item in schema["allOf"]:
errors.extend(validate_instance(value, item, path))
if "not" in schema and not validate_instance(value, schema["not"], path):
errors.append(f"{path}: value matches prohibited not schema")
if "if" in schema:
branch = "then" if not validate_instance(value, schema["if"], path) else "else"
if branch in schema:
errors.extend(validate_instance(value, schema[branch], path))
if isinstance(value, str):
if len(value) < int(schema.get("minLength", 0)):
errors.append(f"{path}: string is shorter than minLength")
if "maxLength" in schema and len(value) > int(schema["maxLength"]):
errors.append(f"{path}: string is longer than maxLength")
if "pattern" in schema and not re.search(str(schema["pattern"]), value):
errors.append(f"{path}: string does not match required pattern")
if "format" in schema and not _matches_format(value, str(schema["format"])):
errors.append(f"{path}: string does not match {schema['format']} format")
if isinstance(value, (int, float)) and not isinstance(value, bool):
if "minimum" in schema and value < schema["minimum"]:
errors.append(f"{path}: value is below minimum")
if "maximum" in schema and value > schema["maximum"]:
errors.append(f"{path}: value is above maximum")
if isinstance(value, list):
if len(value) < int(schema.get("minItems", 0)):
errors.append(f"{path}: array has fewer than minItems")
if "maxItems" in schema and len(value) > int(schema["maxItems"]):
errors.append(f"{path}: array has more than maxItems")
if schema.get("uniqueItems"):
serialized = [_json_value_key(item) for item in value]
if any(item is None for item in serialized):
errors.append(f"{path}: array items must be valid finite JSON values")
elif len(serialized) != len(set(serialized)):
errors.append(f"{path}: array items must be unique")
item_schema = schema.get("items")
if isinstance(item_schema, (Mapping, bool)):
for index, item in enumerate(value):
errors.extend(validate_instance(item, item_schema, f"{path}[{index}]"))
if isinstance(value, Mapping):
required = schema.get("required", [])
for key in required:
if key not in value:
errors.append(f"{path}: required property {key!r} is missing")
properties = schema.get("properties", {})
additional = schema.get("additionalProperties", True)
for key, item in value.items():
if key in properties:
errors.extend(validate_instance(item, properties[key], f"{path}.{key}"))
elif additional is False:
errors.append(f"{path}: additional property {key!r} is not allowed")
elif isinstance(additional, Mapping):
errors.extend(validate_instance(item, additional, f"{path}.{key}"))
return errors
def extract_json_document(text: str) -> tuple[Any | None, str | None]:
stripped = text.strip()
candidates = [stripped]
fenced = re.findall(r"```(?:json)?\s*(.*?)```", stripped, flags=re.IGNORECASE | re.DOTALL)
candidates.extend(item.strip() for item in fenced)
for candidate in candidates:
if not candidate:
continue
try:
return strict_json_loads(candidate), None
except (json.JSONDecodeError, ValueError):
pass
decoder = strict_json_decoder()
for match in re.finditer(r"[\[{]", stripped):
try:
value, end = decoder.raw_decode(stripped[match.start() :])
except (json.JSONDecodeError, ValueError):
continue
try:
validate_json_unicode(value)
except ValueError:
continue
remaining = stripped[match.start() + end :].strip()
if not remaining or remaining.startswith("```"):
return value, None
return None, "no valid JSON document was found in the final worker response"