448 lines
18 KiB
Python
Executable File
448 lines
18 KiB
Python
Executable File
#!/usr/bin/env python3
|
|
"""Small JSON-schema subset used for worker output contracts.
|
|
|
|
The project intentionally avoids a runtime dependency on jsonschema. Profiles
|
|
may use the supported, auditable subset documented in PROFILE_SCHEMA.md.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import math
|
|
import re
|
|
from collections.abc import Mapping
|
|
from typing import Any
|
|
|
|
from mmo_util import (
|
|
strict_json_decoder,
|
|
strict_json_loads,
|
|
valid_absolute_uri,
|
|
validate_json_unicode,
|
|
)
|
|
|
|
SUPPORTED_TYPES = {"object", "array", "string", "integer", "number", "boolean", "null"}
|
|
SUPPORTED_FORMATS = {"uri", "date", "date-time"}
|
|
_RFC3339_DATE = re.compile(r"([0-9]{4})-([0-9]{2})-([0-9]{2})", re.ASCII)
|
|
_RFC3339_DATE_TIME = re.compile(
|
|
r"([0-9]{4})-([0-9]{2})-([0-9]{2})"
|
|
r"[Tt]([0-9]{2}):([0-9]{2}):([0-9]{2})"
|
|
r"(?:\.([0-9]+))?"
|
|
r"(?:([Zz])|([+-])([0-9]{2}):([0-9]{2}))",
|
|
re.ASCII,
|
|
)
|
|
|
|
|
|
def _is_finite_number(value: Any) -> bool:
|
|
if isinstance(value, bool) or not isinstance(value, (int, float)):
|
|
return False
|
|
# Converting an arbitrarily large JSON integer to float can overflow even
|
|
# though the integer itself is finite.
|
|
return isinstance(value, int) or math.isfinite(value)
|
|
|
|
|
|
def _is_integer_number(value: Any) -> bool:
|
|
"""Return whether a JSON number has a zero fractional part."""
|
|
|
|
return _is_finite_number(value) and (
|
|
isinstance(value, int) or (isinstance(value, float) and value.is_integer())
|
|
)
|
|
|
|
|
|
def _json_value_key(value: Any) -> tuple[Any, ...] | None:
|
|
"""Return a strict JSON value key using JSON Schema equality rules."""
|
|
|
|
if value is None:
|
|
return ("null",)
|
|
if isinstance(value, bool):
|
|
return ("boolean", value)
|
|
if isinstance(value, int):
|
|
return ("number", value)
|
|
if isinstance(value, float):
|
|
return ("number", value) if math.isfinite(value) else None
|
|
if isinstance(value, str):
|
|
return ("string", value)
|
|
if isinstance(value, list):
|
|
items = [_json_value_key(item) for item in value]
|
|
if any(item is None for item in items):
|
|
return None
|
|
return ("array", tuple(items))
|
|
if isinstance(value, Mapping):
|
|
if not all(isinstance(key, str) for key in value):
|
|
return None
|
|
members = [(key, _json_value_key(value[key])) for key in sorted(value)]
|
|
if any(item is None for _key, item in members):
|
|
return None
|
|
return ("object", tuple(members))
|
|
return None
|
|
|
|
|
|
def validate_schema_definition(schema: Any, path: str = "$") -> list[str]:
|
|
errors: list[str] = []
|
|
if isinstance(schema, bool):
|
|
return errors
|
|
if not isinstance(schema, Mapping):
|
|
return [f"{path}: schema must be an object or boolean"]
|
|
if not all(isinstance(key, str) for key in schema):
|
|
errors.append(f"{path}: schema keyword names must be strings")
|
|
schema_type = schema.get("type")
|
|
if schema_type is not None:
|
|
if isinstance(schema_type, str):
|
|
if schema_type not in SUPPORTED_TYPES:
|
|
errors.append(f"{path}.type: unsupported type {schema_type!r}")
|
|
elif isinstance(schema_type, list):
|
|
if not schema_type:
|
|
errors.append(f"{path}.type: array cannot be empty")
|
|
invalid = [
|
|
item
|
|
for item in schema_type
|
|
if not isinstance(item, str) or item not in SUPPORTED_TYPES
|
|
]
|
|
if invalid:
|
|
errors.append(f"{path}.type: unsupported types {invalid!r}")
|
|
elif len(schema_type) != len(set(schema_type)):
|
|
errors.append(f"{path}.type: array contains duplicate types")
|
|
else:
|
|
errors.append(f"{path}.type: must be a string or array")
|
|
for keyword in ("oneOf", "anyOf", "allOf"):
|
|
if keyword in schema:
|
|
value = schema[keyword]
|
|
if not isinstance(value, list) or not value:
|
|
errors.append(f"{path}.{keyword}: must be a non-empty array")
|
|
else:
|
|
for index, item in enumerate(value):
|
|
errors.extend(validate_schema_definition(item, f"{path}.{keyword}[{index}]"))
|
|
for keyword in ("not", "if", "then", "else"):
|
|
if keyword in schema:
|
|
errors.extend(validate_schema_definition(schema[keyword], f"{path}.{keyword}"))
|
|
if ("then" in schema or "else" in schema) and "if" not in schema:
|
|
errors.append(f"{path}: then/else requires if")
|
|
if "properties" in schema:
|
|
properties = schema["properties"]
|
|
if not isinstance(properties, Mapping):
|
|
errors.append(f"{path}.properties: must be an object")
|
|
else:
|
|
for key, value in properties.items():
|
|
if not isinstance(key, str):
|
|
errors.append(f"{path}.properties: property names must be strings")
|
|
continue
|
|
errors.extend(validate_schema_definition(value, f"{path}.properties.{key}"))
|
|
if "items" in schema:
|
|
errors.extend(validate_schema_definition(schema["items"], f"{path}.items"))
|
|
if "required" in schema:
|
|
required = schema["required"]
|
|
if not (isinstance(required, list) and all(isinstance(item, str) for item in required)):
|
|
errors.append(f"{path}.required: must be an array of strings")
|
|
elif len(required) != len(set(required)):
|
|
errors.append(f"{path}.required: contains duplicate property names")
|
|
if "enum" in schema:
|
|
enum = schema["enum"]
|
|
if not isinstance(enum, list) or not enum:
|
|
errors.append(f"{path}.enum: must be a non-empty array")
|
|
else:
|
|
keys = [_json_value_key(item) for item in enum]
|
|
if any(key is None for key in keys):
|
|
errors.append(f"{path}.enum: values must be valid finite JSON values")
|
|
elif len(keys) != len(set(keys)):
|
|
errors.append(f"{path}.enum: values must be unique")
|
|
if "const" in schema and _json_value_key(schema["const"]) is None:
|
|
errors.append(f"{path}.const: must be a valid finite JSON value")
|
|
for keyword in ("$schema", "$id", "title", "description"):
|
|
if keyword in schema and not isinstance(schema[keyword], str):
|
|
errors.append(f"{path}.{keyword}: must be a string")
|
|
for keyword in ("minLength", "maxLength", "minItems", "maxItems"):
|
|
if keyword in schema:
|
|
value = schema[keyword]
|
|
if not _is_integer_number(value) or value < 0:
|
|
errors.append(f"{path}.{keyword}: must be a non-negative integer")
|
|
for minimum, maximum in (("minLength", "maxLength"), ("minItems", "maxItems")):
|
|
if (
|
|
_is_integer_number(schema.get(minimum))
|
|
and _is_integer_number(schema.get(maximum))
|
|
and schema[minimum] > schema[maximum]
|
|
):
|
|
errors.append(f"{path}: {minimum} exceeds {maximum}")
|
|
for keyword in ("minimum", "maximum"):
|
|
if keyword in schema:
|
|
value = schema[keyword]
|
|
if not _is_finite_number(value):
|
|
errors.append(f"{path}.{keyword}: must be a finite number")
|
|
if (
|
|
isinstance(schema.get("minimum"), (int, float))
|
|
and not isinstance(schema.get("minimum"), bool)
|
|
and isinstance(schema.get("maximum"), (int, float))
|
|
and not isinstance(schema.get("maximum"), bool)
|
|
and schema["minimum"] > schema["maximum"]
|
|
):
|
|
errors.append(f"{path}: minimum exceeds maximum")
|
|
if "uniqueItems" in schema and not isinstance(schema["uniqueItems"], bool):
|
|
errors.append(f"{path}.uniqueItems: must be boolean")
|
|
if "additionalProperties" in schema:
|
|
additional = schema["additionalProperties"]
|
|
if isinstance(additional, Mapping):
|
|
errors.extend(validate_schema_definition(additional, f"{path}.additionalProperties"))
|
|
elif not isinstance(additional, bool):
|
|
errors.append(f"{path}.additionalProperties: must be boolean or a schema")
|
|
if "pattern" in schema:
|
|
pattern = schema["pattern"]
|
|
if not isinstance(pattern, str):
|
|
errors.append(f"{path}.pattern: must be a string")
|
|
else:
|
|
try:
|
|
re.compile(pattern)
|
|
except re.error as exc:
|
|
errors.append(f"{path}.pattern: invalid regular expression: {exc}")
|
|
if "format" in schema:
|
|
schema_format = schema["format"]
|
|
if not isinstance(schema_format, str) or schema_format not in SUPPORTED_FORMATS:
|
|
errors.append(f"{path}.format: must be one of {sorted(SUPPORTED_FORMATS)}")
|
|
known = {
|
|
"$schema",
|
|
"$id",
|
|
"title",
|
|
"description",
|
|
"type",
|
|
"properties",
|
|
"required",
|
|
"additionalProperties",
|
|
"items",
|
|
"enum",
|
|
"const",
|
|
"minLength",
|
|
"maxLength",
|
|
"minimum",
|
|
"maximum",
|
|
"minItems",
|
|
"maxItems",
|
|
"uniqueItems",
|
|
"pattern",
|
|
"oneOf",
|
|
"anyOf",
|
|
"allOf",
|
|
"not",
|
|
"if",
|
|
"then",
|
|
"else",
|
|
"format",
|
|
}
|
|
unknown = sorted(key for key in schema if isinstance(key, str) and key not in known)
|
|
if unknown:
|
|
errors.append(f"{path}: unsupported schema keywords: {', '.join(unknown)}")
|
|
return errors
|
|
|
|
|
|
def _types(schema: Mapping[str, Any]) -> set[str] | None:
|
|
value = schema.get("type")
|
|
if value is None:
|
|
return None
|
|
return {value} if isinstance(value, str) else set(value)
|
|
|
|
|
|
def _matches_type(value: Any, expected: str) -> bool:
|
|
if expected == "null":
|
|
return value is None
|
|
if expected == "boolean":
|
|
return isinstance(value, bool)
|
|
if expected == "integer":
|
|
return _is_integer_number(value)
|
|
if expected == "number":
|
|
return _is_finite_number(value)
|
|
if expected == "string":
|
|
return isinstance(value, str)
|
|
if expected == "array":
|
|
return isinstance(value, list)
|
|
if expected == "object":
|
|
return isinstance(value, Mapping)
|
|
return False
|
|
|
|
|
|
def _rfc3339_month_days(year: int, month: int) -> int:
|
|
if not 1 <= month <= 12:
|
|
return 0
|
|
leap = year % 4 == 0 and (year % 100 != 0 or year % 400 == 0)
|
|
month_days = (31, 29 if leap else 28, 31, 30, 31, 30, 31, 31, 30, 31, 30, 31)
|
|
return month_days[month - 1]
|
|
|
|
|
|
def _valid_rfc3339_date(year: int, month: int, day: int) -> bool:
|
|
return 1 <= day <= _rfc3339_month_days(year, month)
|
|
|
|
|
|
def _shift_rfc3339_date(year: int, month: int, day: int, day_delta: int) -> tuple[int, int, int]:
|
|
"""Shift a valid RFC 3339 date by the at-most-one-day offset boundary."""
|
|
|
|
if day_delta == -1:
|
|
if day > 1:
|
|
return year, month, day - 1
|
|
if month > 1:
|
|
month -= 1
|
|
else:
|
|
year -= 1
|
|
month = 12
|
|
return year, month, _rfc3339_month_days(year, month)
|
|
if day_delta == 1:
|
|
if day < _rfc3339_month_days(year, month):
|
|
return year, month, day + 1
|
|
if month < 12:
|
|
return year, month + 1, 1
|
|
return year + 1, 1, 1
|
|
return year, month, day
|
|
|
|
|
|
def _matches_rfc3339_date(value: str) -> bool:
|
|
matched = _RFC3339_DATE.fullmatch(value)
|
|
return matched is not None and _valid_rfc3339_date(
|
|
int(matched.group(1)), int(matched.group(2)), int(matched.group(3))
|
|
)
|
|
|
|
|
|
def _matches_rfc3339_date_time(value: str) -> bool:
|
|
matched = _RFC3339_DATE_TIME.fullmatch(value)
|
|
if matched is None:
|
|
return False
|
|
year, month, day, hour, minute, second = map(int, matched.groups()[:6])
|
|
if not _valid_rfc3339_date(year, month, day) or hour > 23 or minute > 59 or second > 60:
|
|
return False
|
|
if matched.group(8) is not None:
|
|
offset_minutes = 0
|
|
else:
|
|
offset_hour = int(matched.group(10))
|
|
offset_minute = int(matched.group(11))
|
|
if offset_hour > 23 or offset_minute > 59:
|
|
return False
|
|
offset_minutes = offset_hour * 60 + offset_minute
|
|
if matched.group(9) == "-":
|
|
offset_minutes = -offset_minutes
|
|
if second == 60:
|
|
day_delta, utc_minute = divmod(hour * 60 + minute - offset_minutes, 24 * 60)
|
|
utc_year, utc_month, utc_day = _shift_rfc3339_date(year, month, day, day_delta)
|
|
# RFC 3339 section 5.7 allows :60 only at the end of a month in UTC.
|
|
# The local spelling may fall on the adjacent date after applying its
|
|
# numeric offset, so validate the shifted UTC calendar date.
|
|
return utc_minute == 23 * 60 + 59 and utc_day == _rfc3339_month_days(utc_year, utc_month)
|
|
return True
|
|
|
|
|
|
def _matches_format(value: str, schema_format: str) -> bool:
|
|
if schema_format == "uri":
|
|
return valid_absolute_uri(value)
|
|
if schema_format == "date":
|
|
return _matches_rfc3339_date(value)
|
|
if schema_format == "date-time":
|
|
return _matches_rfc3339_date_time(value)
|
|
return False
|
|
|
|
|
|
def validate_instance(
|
|
value: Any,
|
|
schema: Mapping[str, Any] | bool,
|
|
path: str = "$",
|
|
) -> list[str]:
|
|
if schema is True:
|
|
return []
|
|
if schema is False:
|
|
return [f"{path}: value is rejected by false schema"]
|
|
errors: list[str] = []
|
|
allowed_types = _types(schema)
|
|
if allowed_types and not any(_matches_type(value, item) for item in allowed_types):
|
|
return [
|
|
f"{path}: expected {' or '.join(sorted(allowed_types))}, got {type(value).__name__}"
|
|
]
|
|
if "const" in schema:
|
|
value_key = _json_value_key(value)
|
|
const_key = _json_value_key(schema["const"])
|
|
if value_key is None or const_key is None or value_key != const_key:
|
|
errors.append(f"{path}: expected constant {schema['const']!r}")
|
|
if "enum" in schema:
|
|
value_key = _json_value_key(value)
|
|
enum_keys = {_json_value_key(item) for item in schema["enum"]}
|
|
if value_key is None or value_key not in enum_keys:
|
|
errors.append(f"{path}: value {value!r} is not in the allowed enum")
|
|
if "oneOf" in schema:
|
|
matches = [not validate_instance(value, item, path) for item in schema["oneOf"]]
|
|
if sum(matches) != 1:
|
|
errors.append(f"{path}: value must match exactly one oneOf schema")
|
|
if "anyOf" in schema:
|
|
if not any(not validate_instance(value, item, path) for item in schema["anyOf"]):
|
|
errors.append(f"{path}: value does not match any anyOf schema")
|
|
if "allOf" in schema:
|
|
for item in schema["allOf"]:
|
|
errors.extend(validate_instance(value, item, path))
|
|
if "not" in schema and not validate_instance(value, schema["not"], path):
|
|
errors.append(f"{path}: value matches prohibited not schema")
|
|
if "if" in schema:
|
|
branch = "then" if not validate_instance(value, schema["if"], path) else "else"
|
|
if branch in schema:
|
|
errors.extend(validate_instance(value, schema[branch], path))
|
|
if isinstance(value, str):
|
|
if len(value) < int(schema.get("minLength", 0)):
|
|
errors.append(f"{path}: string is shorter than minLength")
|
|
if "maxLength" in schema and len(value) > int(schema["maxLength"]):
|
|
errors.append(f"{path}: string is longer than maxLength")
|
|
if "pattern" in schema and not re.search(str(schema["pattern"]), value):
|
|
errors.append(f"{path}: string does not match required pattern")
|
|
if "format" in schema and not _matches_format(value, str(schema["format"])):
|
|
errors.append(f"{path}: string does not match {schema['format']} format")
|
|
if isinstance(value, (int, float)) and not isinstance(value, bool):
|
|
if "minimum" in schema and value < schema["minimum"]:
|
|
errors.append(f"{path}: value is below minimum")
|
|
if "maximum" in schema and value > schema["maximum"]:
|
|
errors.append(f"{path}: value is above maximum")
|
|
if isinstance(value, list):
|
|
if len(value) < int(schema.get("minItems", 0)):
|
|
errors.append(f"{path}: array has fewer than minItems")
|
|
if "maxItems" in schema and len(value) > int(schema["maxItems"]):
|
|
errors.append(f"{path}: array has more than maxItems")
|
|
if schema.get("uniqueItems"):
|
|
serialized = [_json_value_key(item) for item in value]
|
|
if any(item is None for item in serialized):
|
|
errors.append(f"{path}: array items must be valid finite JSON values")
|
|
elif len(serialized) != len(set(serialized)):
|
|
errors.append(f"{path}: array items must be unique")
|
|
item_schema = schema.get("items")
|
|
if isinstance(item_schema, (Mapping, bool)):
|
|
for index, item in enumerate(value):
|
|
errors.extend(validate_instance(item, item_schema, f"{path}[{index}]"))
|
|
if isinstance(value, Mapping):
|
|
required = schema.get("required", [])
|
|
for key in required:
|
|
if key not in value:
|
|
errors.append(f"{path}: required property {key!r} is missing")
|
|
properties = schema.get("properties", {})
|
|
additional = schema.get("additionalProperties", True)
|
|
for key, item in value.items():
|
|
if key in properties:
|
|
errors.extend(validate_instance(item, properties[key], f"{path}.{key}"))
|
|
elif additional is False:
|
|
errors.append(f"{path}: additional property {key!r} is not allowed")
|
|
elif isinstance(additional, Mapping):
|
|
errors.extend(validate_instance(item, additional, f"{path}.{key}"))
|
|
return errors
|
|
|
|
|
|
def extract_json_document(text: str) -> tuple[Any | None, str | None]:
|
|
stripped = text.strip()
|
|
candidates = [stripped]
|
|
fenced = re.findall(r"```(?:json)?\s*(.*?)```", stripped, flags=re.IGNORECASE | re.DOTALL)
|
|
candidates.extend(item.strip() for item in fenced)
|
|
for candidate in candidates:
|
|
if not candidate:
|
|
continue
|
|
try:
|
|
return strict_json_loads(candidate), None
|
|
except (json.JSONDecodeError, ValueError):
|
|
pass
|
|
decoder = strict_json_decoder()
|
|
for match in re.finditer(r"[\[{]", stripped):
|
|
try:
|
|
value, end = decoder.raw_decode(stripped[match.start() :])
|
|
except (json.JSONDecodeError, ValueError):
|
|
continue
|
|
try:
|
|
validate_json_unicode(value)
|
|
except ValueError:
|
|
continue
|
|
remaining = stripped[match.start() + end :].strip()
|
|
if not remaining or remaining.startswith("```"):
|
|
return value, None
|
|
return None, "no valid JSON document was found in the final worker response"
|