1180 lines
43 KiB
Python
1180 lines
43 KiB
Python
"""Reproducible horizon/max-step matrix runs for one authoritative System XML.
|
|
|
|
The module deliberately delegates every simulation to
|
|
``execute_regression_case``. Consequently each cell inherits the existing
|
|
soft-cancel/hard-termination boundary and derives its XML in child-process
|
|
memory; the source XML is never rewritten.
|
|
|
|
The two comparison axes answer different questions:
|
|
|
|
* same horizon, different maximum steps: ordinary step-size sensitivity;
|
|
* same maximum step, different horizons: prefix invariance with respect to
|
|
``tStop`` (a longer request should not change already-reached checkpoints).
|
|
|
|
This is a diagnostic matrix, not a golden generator. Numerical state values
|
|
are retained in its JSON evidence but are never written back to a manifest.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
from collections.abc import Callable, Mapping, Sequence
|
|
from dataclasses import dataclass
|
|
from datetime import UTC, datetime
|
|
from itertools import combinations
|
|
import json
|
|
import math
|
|
from pathlib import Path
|
|
|
|
from app.simulation.benchmark_regression import (
|
|
DEFAULT_MANIFEST_PATH,
|
|
RegressionCaseRequest,
|
|
RegressionManifestError,
|
|
execute_regression_case,
|
|
load_regression_manifest,
|
|
source_simulation_config,
|
|
)
|
|
|
|
|
|
MATRIX_REPORT_SCHEMA_VERSION = 1
|
|
DEFAULT_HORIZON_CASE_IDS = ("1s", "5s", "10s")
|
|
DEFAULT_COMMON_CHECKPOINT_TIMES = (0.0, 0.04, 0.8, 1.0)
|
|
DEFAULT_EXPECTED_PROJECTION_COUNT = 134
|
|
|
|
MatrixCaseExecutor = Callable[[RegressionCaseRequest], dict[str, object]]
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class MatrixTolerance:
|
|
state_relative: float
|
|
state_absolute: float
|
|
checkpoint_time_absolute_seconds: float
|
|
event_time_absolute_seconds: float
|
|
signal_event_time_absolute_seconds: float
|
|
maximum_scaled_residual: float | None
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class MatrixHorizon:
|
|
case_id: str
|
|
source_variant_id: str | None
|
|
stop_time: float
|
|
checkpoint_times: tuple[float, ...]
|
|
expected_signal_event_times: tuple[float, ...]
|
|
soft_timeout_seconds: float
|
|
hard_timeout_seconds: float
|
|
|
|
|
|
def _finite_positive(value: object, *, field: str) -> float:
|
|
try:
|
|
numeric = float(value)
|
|
except (TypeError, ValueError) as exc:
|
|
raise RegressionManifestError(f"{field} must be numeric.") from exc
|
|
if not math.isfinite(numeric) or numeric <= 0.0:
|
|
raise RegressionManifestError(f"{field} must be finite and positive.")
|
|
return numeric
|
|
|
|
|
|
def _finite_nonnegative(value: object, *, field: str) -> float:
|
|
try:
|
|
numeric = float(value)
|
|
except (TypeError, ValueError) as exc:
|
|
raise RegressionManifestError(f"{field} must be numeric.") from exc
|
|
if not math.isfinite(numeric) or numeric < 0.0:
|
|
raise RegressionManifestError(
|
|
f"{field} must be finite and non-negative."
|
|
)
|
|
return numeric
|
|
|
|
|
|
def _matrix_tolerance(manifest: Mapping[str, object]) -> MatrixTolerance:
|
|
correctness = manifest.get("correctness")
|
|
if not isinstance(correctness, Mapping):
|
|
raise RegressionManifestError("Manifest correctness must be an object.")
|
|
maximum_residual = correctness.get("maximumScaledResidual")
|
|
return MatrixTolerance(
|
|
state_relative=_finite_nonnegative(
|
|
correctness.get("stateRelativeTolerance", 0.0),
|
|
field="correctness.stateRelativeTolerance",
|
|
),
|
|
state_absolute=_finite_nonnegative(
|
|
correctness.get("stateAbsoluteTolerance", 0.0),
|
|
field="correctness.stateAbsoluteTolerance",
|
|
),
|
|
checkpoint_time_absolute_seconds=_finite_nonnegative(
|
|
correctness.get("checkpointTimeAbsoluteToleranceSeconds", 1.0e-12),
|
|
field="correctness.checkpointTimeAbsoluteToleranceSeconds",
|
|
),
|
|
event_time_absolute_seconds=_finite_nonnegative(
|
|
correctness.get("eventTimeAbsoluteToleranceSeconds", 0.0),
|
|
field="correctness.eventTimeAbsoluteToleranceSeconds",
|
|
),
|
|
signal_event_time_absolute_seconds=_finite_nonnegative(
|
|
correctness.get("signalEventTimeAbsoluteToleranceSeconds", 0.0),
|
|
field="correctness.signalEventTimeAbsoluteToleranceSeconds",
|
|
),
|
|
maximum_scaled_residual=(
|
|
_finite_nonnegative(
|
|
maximum_residual,
|
|
field="correctness.maximumScaledResidual",
|
|
)
|
|
if maximum_residual is not None
|
|
else None
|
|
),
|
|
)
|
|
|
|
|
|
def _resolve_horizons(
|
|
manifest: Mapping[str, object],
|
|
requested: Sequence[str | float],
|
|
*,
|
|
additional_checkpoint_times: Sequence[float],
|
|
) -> tuple[MatrixHorizon, ...]:
|
|
variants = manifest.get("variants")
|
|
sequence = manifest.get("sequence")
|
|
if not isinstance(variants, Mapping) or not isinstance(sequence, list):
|
|
raise RegressionManifestError("Manifest variants/sequence are incomplete.")
|
|
extra_checkpoints = tuple(
|
|
_finite_nonnegative(value, field=f"additionalCheckpointTimes[{index}]")
|
|
for index, value in enumerate(additional_checkpoint_times)
|
|
)
|
|
ordered_variants: list[tuple[str, Mapping[str, object]]] = []
|
|
for raw_case_id in sequence:
|
|
variant = variants.get(str(raw_case_id))
|
|
if isinstance(variant, Mapping):
|
|
ordered_variants.append((str(raw_case_id), variant))
|
|
selected: list[MatrixHorizon] = []
|
|
for raw_request in requested:
|
|
request = str(raw_request).strip()
|
|
if request in variants:
|
|
source_variant_id: str | None = request
|
|
variant = variants[request]
|
|
assert isinstance(variant, Mapping)
|
|
stop_time = float(variant["stopTime"])
|
|
else:
|
|
try:
|
|
stop_time = float(request.removesuffix("s"))
|
|
except ValueError as exc:
|
|
raise RegressionManifestError(
|
|
f"Unknown matrix horizon {raw_request!r}."
|
|
) from exc
|
|
stop_time = _finite_positive(stop_time, field=f"horizon[{request}]")
|
|
matches = [
|
|
(str(candidate), variant)
|
|
for candidate, variant in variants.items()
|
|
if isinstance(variant, Mapping)
|
|
and isinstance(variant.get("stopTime"), (int, float))
|
|
and math.isclose(
|
|
float(variant["stopTime"]),
|
|
stop_time,
|
|
rel_tol=0.0,
|
|
abs_tol=max(1.0e-12, 8.0 * math.ulp(max(1.0, abs(stop_time)))),
|
|
)
|
|
]
|
|
if len(matches) > 1:
|
|
raise RegressionManifestError(
|
|
f"Matrix horizon {raw_request!r} identifies multiple variants."
|
|
)
|
|
if matches:
|
|
source_variant_id, variant = matches[0]
|
|
else:
|
|
source_variant_id = None
|
|
variant = None
|
|
case_id = source_variant_id or f"{format(stop_time, '.12g')}s-custom"
|
|
if any(item.case_id == case_id for item in selected):
|
|
raise RegressionManifestError(
|
|
f"Matrix horizon {case_id!r} was selected more than once."
|
|
)
|
|
|
|
timeout_variant = variant
|
|
if timeout_variant is None:
|
|
enclosing = [
|
|
candidate
|
|
for _, candidate in ordered_variants
|
|
if float(candidate["stopTime"]) >= stop_time
|
|
]
|
|
timeout_variant = enclosing[0] if enclosing else ordered_variants[-1][1]
|
|
raw_checkpoints = (
|
|
variant.get("checkpointTimes", [])
|
|
if isinstance(variant, Mapping)
|
|
else []
|
|
)
|
|
checkpoint_candidates = [
|
|
*DEFAULT_COMMON_CHECKPOINT_TIMES,
|
|
*(
|
|
float(value)
|
|
for value in raw_checkpoints
|
|
if isinstance(value, (int, float))
|
|
),
|
|
*extra_checkpoints,
|
|
stop_time,
|
|
]
|
|
checkpoint_times = tuple(
|
|
sorted(
|
|
{
|
|
float(value)
|
|
for value in checkpoint_candidates
|
|
if 0.0 <= float(value) <= stop_time
|
|
}
|
|
)
|
|
)
|
|
if isinstance(variant, Mapping) and isinstance(
|
|
variant.get("expectedSignalEventTimes"), list
|
|
):
|
|
expected_signal_times = tuple(
|
|
float(value) for value in variant["expectedSignalEventTimes"]
|
|
)
|
|
else:
|
|
expected_signal_times = tuple(
|
|
sorted(
|
|
{
|
|
float(value)
|
|
for _, candidate in ordered_variants
|
|
for value in candidate.get("expectedSignalEventTimes", [])
|
|
if isinstance(value, (int, float))
|
|
and float(value) <= stop_time
|
|
}
|
|
)
|
|
)
|
|
selected.append(
|
|
MatrixHorizon(
|
|
case_id=case_id,
|
|
source_variant_id=source_variant_id,
|
|
stop_time=stop_time,
|
|
checkpoint_times=checkpoint_times,
|
|
expected_signal_event_times=expected_signal_times,
|
|
soft_timeout_seconds=_finite_positive(
|
|
timeout_variant.get("softTimeoutSeconds"),
|
|
field=f"horizon[{request}].softTimeoutSeconds",
|
|
),
|
|
hard_timeout_seconds=_finite_positive(
|
|
timeout_variant.get("hardTimeoutSeconds"),
|
|
field=f"horizon[{request}].hardTimeoutSeconds",
|
|
),
|
|
)
|
|
)
|
|
if not selected:
|
|
raise RegressionManifestError("At least one matrix horizon is required.")
|
|
stop_times = [item.stop_time for item in selected]
|
|
if any(right <= left for left, right in zip(stop_times, stop_times[1:])):
|
|
raise RegressionManifestError(
|
|
"Matrix horizons must have strictly increasing stop times."
|
|
)
|
|
return tuple(selected)
|
|
|
|
|
|
def _normalise_max_steps(max_steps: Sequence[float]) -> tuple[float, ...]:
|
|
values = tuple(
|
|
_finite_positive(value, field=f"maxSteps[{index}]")
|
|
for index, value in enumerate(max_steps)
|
|
)
|
|
if not values:
|
|
raise RegressionManifestError("At least one maximum step is required.")
|
|
if len(set(values)) != len(values):
|
|
raise RegressionManifestError("Maximum steps must be unique.")
|
|
return values
|
|
|
|
|
|
def _summary(case: Mapping[str, object]) -> Mapping[str, object] | None:
|
|
worker = case.get("worker")
|
|
if not isinstance(worker, Mapping):
|
|
return None
|
|
summary = worker.get("summary")
|
|
return summary if isinstance(summary, Mapping) else None
|
|
|
|
|
|
def _completed(case: Mapping[str, object], stop_time: float) -> bool:
|
|
summary = _summary(case)
|
|
simulated_until = summary.get("simulatedUntil") if summary is not None else None
|
|
return (
|
|
case.get("outcome") == "completed"
|
|
and summary is not None
|
|
and bool(summary.get("success"))
|
|
and summary.get("status") == "completed"
|
|
and isinstance(simulated_until, (int, float))
|
|
and math.isclose(
|
|
float(simulated_until),
|
|
stop_time,
|
|
rel_tol=0.0,
|
|
abs_tol=max(1.0e-12, 8.0 * math.ulp(max(1.0, abs(stop_time)))),
|
|
)
|
|
)
|
|
|
|
|
|
def _checkpoints(summary: Mapping[str, object] | None) -> list[Mapping[str, object]]:
|
|
if summary is None:
|
|
return []
|
|
contract = summary.get("physicalContract")
|
|
raw = (
|
|
contract.get("checkpoints")
|
|
if isinstance(contract, Mapping)
|
|
else summary.get("checkpoints")
|
|
)
|
|
if not isinstance(raw, list):
|
|
return []
|
|
return [checkpoint for checkpoint in raw if isinstance(checkpoint, Mapping)]
|
|
|
|
|
|
def _event_trace(summary: Mapping[str, object] | None) -> Mapping[str, object]:
|
|
if summary is None:
|
|
return {}
|
|
contract = summary.get("physicalContract")
|
|
raw = (
|
|
contract.get("eventTrace")
|
|
if isinstance(contract, Mapping)
|
|
else summary.get("eventTrace")
|
|
)
|
|
return raw if isinstance(raw, Mapping) else {}
|
|
|
|
|
|
def _diagnostics(summary: Mapping[str, object] | None) -> Mapping[str, object]:
|
|
if summary is None:
|
|
return {}
|
|
raw = summary.get("diagnostics")
|
|
return raw if isinstance(raw, Mapping) else {}
|
|
|
|
|
|
def _scaled_residual(summary: Mapping[str, object] | None) -> float | None:
|
|
pressure_flow = _diagnostics(summary).get("pressureFlow")
|
|
value = (
|
|
pressure_flow.get("maxScaledResidual")
|
|
if isinstance(pressure_flow, Mapping)
|
|
else None
|
|
)
|
|
if isinstance(value, (int, float)) and math.isfinite(float(value)):
|
|
return float(value)
|
|
return None
|
|
|
|
|
|
def _integration_totals(summary: Mapping[str, object] | None) -> dict[str, float]:
|
|
integration = _diagnostics(summary).get("integration")
|
|
totals = integration.get("totals") if isinstance(integration, Mapping) else None
|
|
if not isinstance(totals, Mapping):
|
|
return {}
|
|
return {
|
|
str(key): float(value)
|
|
for key, value in totals.items()
|
|
if isinstance(value, (int, float)) and math.isfinite(float(value))
|
|
}
|
|
|
|
|
|
def _finite_event_times(value: object) -> list[float] | None:
|
|
if not isinstance(value, list):
|
|
return None
|
|
if any(
|
|
not isinstance(item, (int, float)) or not math.isfinite(float(item))
|
|
for item in value
|
|
):
|
|
return None
|
|
return [float(item) for item in value]
|
|
|
|
|
|
def _case_observation(
|
|
case: Mapping[str, object],
|
|
*,
|
|
stop_time: float,
|
|
expected_projection_count: int | None,
|
|
tolerance: MatrixTolerance,
|
|
expected_signal_event_times: Sequence[float] | None,
|
|
) -> tuple[dict[str, object], tuple[str, ...]]:
|
|
summary = _summary(case)
|
|
checkpoints = _checkpoints(summary)
|
|
checkpoint_details: list[dict[str, object]] = []
|
|
issues: list[str] = []
|
|
for checkpoint in checkpoints:
|
|
values = checkpoint.get("stateValues")
|
|
state_values = values if isinstance(values, Mapping) else {}
|
|
nonfinite = [
|
|
str(key)
|
|
for key, value in state_values.items()
|
|
if not isinstance(value, (int, float)) or not math.isfinite(float(value))
|
|
]
|
|
detail = {
|
|
"requestedTime": checkpoint.get("requestedTime"),
|
|
"actualTime": checkpoint.get("actualTime"),
|
|
"available": bool(checkpoint.get("available")),
|
|
"projectionKeyCount": len(state_values),
|
|
"nonfiniteKeyCount": len(nonfinite),
|
|
}
|
|
checkpoint_details.append(detail)
|
|
if not detail["available"]:
|
|
issues.append("checkpointUnavailable")
|
|
if expected_projection_count is not None and len(state_values) != int(
|
|
expected_projection_count
|
|
):
|
|
issues.append("projectionCountMismatch")
|
|
if nonfinite:
|
|
issues.append("nonfiniteProjection")
|
|
if not checkpoints:
|
|
issues.append("checkpointsUnavailable")
|
|
|
|
residual = _scaled_residual(summary)
|
|
if tolerance.maximum_scaled_residual is not None and (
|
|
residual is None or residual > tolerance.maximum_scaled_residual
|
|
):
|
|
issues.append("scaledResidualExceeded")
|
|
|
|
events = _event_trace(summary)
|
|
signal_times = _finite_event_times(events.get("signalEventTimes"))
|
|
if expected_signal_event_times is not None:
|
|
expected = [float(value) for value in expected_signal_event_times]
|
|
if signal_times is None or len(signal_times) != len(expected) or any(
|
|
not math.isclose(
|
|
actual,
|
|
wanted,
|
|
rel_tol=0.0,
|
|
abs_tol=tolerance.signal_event_time_absolute_seconds,
|
|
)
|
|
for actual, wanted in zip(signal_times or (), expected)
|
|
):
|
|
issues.append("signalEventTraceMismatch")
|
|
if events.get("mechanicalTransitionTimesAvailable") is False:
|
|
issues.append("mechanicalTransitionTimesUnavailable")
|
|
|
|
if not _completed(case, stop_time):
|
|
issues.insert(0, "simulationDidNotComplete")
|
|
unique_issues = tuple(dict.fromkeys(issues))
|
|
return (
|
|
{
|
|
"completed": _completed(case, stop_time),
|
|
"checkpointCount": len(checkpoints),
|
|
"expectedProjectionCount": expected_projection_count,
|
|
"checkpoints": checkpoint_details,
|
|
"events": dict(events),
|
|
"maximumScaledResidual": residual,
|
|
"integrationTotals": _integration_totals(summary),
|
|
"passed": not unique_issues,
|
|
"issues": list(unique_issues),
|
|
},
|
|
unique_issues,
|
|
)
|
|
|
|
|
|
def _pair_checkpoints(
|
|
left: Mapping[str, object],
|
|
right: Mapping[str, object],
|
|
*,
|
|
tolerance: MatrixTolerance,
|
|
) -> dict[str, object]:
|
|
left_checkpoints = _checkpoints(_summary(left))
|
|
right_checkpoints = _checkpoints(_summary(right))
|
|
paired: list[tuple[Mapping[str, object], Mapping[str, object]]] = []
|
|
for left_checkpoint in left_checkpoints:
|
|
left_time = left_checkpoint.get("requestedTime")
|
|
if not isinstance(left_time, (int, float)):
|
|
continue
|
|
match = next(
|
|
(
|
|
right_checkpoint
|
|
for right_checkpoint in right_checkpoints
|
|
if isinstance(right_checkpoint.get("requestedTime"), (int, float))
|
|
and math.isclose(
|
|
float(right_checkpoint["requestedTime"]),
|
|
float(left_time),
|
|
rel_tol=0.0,
|
|
abs_tol=tolerance.checkpoint_time_absolute_seconds,
|
|
)
|
|
),
|
|
None,
|
|
)
|
|
if match is not None:
|
|
paired.append((left_checkpoint, match))
|
|
|
|
comparisons: list[dict[str, object]] = []
|
|
mismatch_count = 0
|
|
key_set_mismatch_count = 0
|
|
nonnumeric_count = 0
|
|
maximum_absolute_difference = 0.0
|
|
maximum_relative_difference = 0.0
|
|
for left_checkpoint, right_checkpoint in paired:
|
|
left_values = left_checkpoint.get("stateValues")
|
|
right_values = right_checkpoint.get("stateValues")
|
|
left_mapping = left_values if isinstance(left_values, Mapping) else {}
|
|
right_mapping = right_values if isinstance(right_values, Mapping) else {}
|
|
left_keys = set(map(str, left_mapping))
|
|
right_keys = set(map(str, right_mapping))
|
|
missing_from_left = sorted(right_keys - left_keys)
|
|
missing_from_right = sorted(left_keys - right_keys)
|
|
key_set_mismatch_count += len(missing_from_left) + len(missing_from_right)
|
|
mismatches: list[dict[str, object]] = []
|
|
for key in sorted(left_keys & right_keys):
|
|
left_value = left_mapping.get(key)
|
|
right_value = right_mapping.get(key)
|
|
if not isinstance(left_value, (int, float)) or not isinstance(
|
|
right_value, (int, float)
|
|
) or not math.isfinite(float(left_value)) or not math.isfinite(
|
|
float(right_value)
|
|
):
|
|
nonnumeric_count += 1
|
|
mismatches.append(
|
|
{"key": key, "left": left_value, "right": right_value}
|
|
)
|
|
continue
|
|
left_numeric = float(left_value)
|
|
right_numeric = float(right_value)
|
|
absolute = abs(left_numeric - right_numeric)
|
|
denominator = max(abs(left_numeric), abs(right_numeric))
|
|
relative = absolute / denominator if denominator else 0.0
|
|
maximum_absolute_difference = max(maximum_absolute_difference, absolute)
|
|
maximum_relative_difference = max(maximum_relative_difference, relative)
|
|
if not math.isclose(
|
|
left_numeric,
|
|
right_numeric,
|
|
rel_tol=tolerance.state_relative,
|
|
abs_tol=tolerance.state_absolute,
|
|
):
|
|
mismatches.append(
|
|
{
|
|
"key": key,
|
|
"left": left_numeric,
|
|
"right": right_numeric,
|
|
"absoluteDifference": absolute,
|
|
"relativeDifference": relative,
|
|
}
|
|
)
|
|
mismatch_count += len(mismatches)
|
|
comparisons.append(
|
|
{
|
|
"requestedTime": left_checkpoint.get("requestedTime"),
|
|
"leftProjectionKeyCount": len(left_keys),
|
|
"rightProjectionKeyCount": len(right_keys),
|
|
"missingFromLeft": missing_from_left,
|
|
"missingFromRight": missing_from_right,
|
|
"valueMismatchCount": len(mismatches),
|
|
"mismatches": mismatches,
|
|
}
|
|
)
|
|
passed = bool(paired) and not (
|
|
mismatch_count or key_set_mismatch_count or nonnumeric_count
|
|
)
|
|
return {
|
|
"evaluated": bool(paired),
|
|
"passed": passed,
|
|
"commonCheckpointCount": len(paired),
|
|
"commonCheckpointTimes": [
|
|
pair[0].get("requestedTime") for pair in paired
|
|
],
|
|
"valueMismatchCount": mismatch_count,
|
|
"keySetMismatchCount": key_set_mismatch_count,
|
|
"nonnumericValueCount": nonnumeric_count,
|
|
"maximumAbsoluteDifference": maximum_absolute_difference,
|
|
"maximumRelativeDifference": maximum_relative_difference,
|
|
"checkpoints": comparisons,
|
|
}
|
|
|
|
|
|
def _compare_time_sequences(
|
|
left: object,
|
|
right: object,
|
|
*,
|
|
prefix_stop: float,
|
|
absolute_tolerance: float,
|
|
) -> dict[str, object]:
|
|
left_times = _finite_event_times(left)
|
|
right_times = _finite_event_times(right)
|
|
if left_times is None or right_times is None:
|
|
return {
|
|
"available": False,
|
|
"passed": False,
|
|
"left": left,
|
|
"right": right,
|
|
}
|
|
left_prefix = [
|
|
value for value in left_times if value <= prefix_stop + absolute_tolerance
|
|
]
|
|
right_prefix = [
|
|
value for value in right_times if value <= prefix_stop + absolute_tolerance
|
|
]
|
|
passed = len(left_prefix) == len(right_prefix) and all(
|
|
math.isclose(
|
|
left_value,
|
|
right_value,
|
|
rel_tol=0.0,
|
|
abs_tol=absolute_tolerance,
|
|
)
|
|
for left_value, right_value in zip(left_prefix, right_prefix)
|
|
)
|
|
return {
|
|
"available": True,
|
|
"passed": passed,
|
|
"left": left_prefix,
|
|
"right": right_prefix,
|
|
}
|
|
|
|
|
|
def _pair_events(
|
|
left: Mapping[str, object],
|
|
right: Mapping[str, object],
|
|
*,
|
|
prefix_stop: float,
|
|
tolerance: MatrixTolerance,
|
|
) -> dict[str, object]:
|
|
left_trace = _event_trace(_summary(left))
|
|
right_trace = _event_trace(_summary(right))
|
|
signal = _compare_time_sequences(
|
|
left_trace.get("signalEventTimes"),
|
|
right_trace.get("signalEventTimes"),
|
|
prefix_stop=prefix_stop,
|
|
absolute_tolerance=tolerance.signal_event_time_absolute_seconds,
|
|
)
|
|
mechanical = _compare_time_sequences(
|
|
left_trace.get("mechanicalTransitionTimes"),
|
|
right_trace.get("mechanicalTransitionTimes"),
|
|
prefix_stop=prefix_stop,
|
|
absolute_tolerance=tolerance.event_time_absolute_seconds,
|
|
)
|
|
availability = (
|
|
left_trace.get("mechanicalTransitionTimesAvailable") is not False
|
|
and right_trace.get("mechanicalTransitionTimesAvailable") is not False
|
|
)
|
|
return {
|
|
"evaluated": bool(left_trace) and bool(right_trace),
|
|
"passed": bool(signal["passed"] and mechanical["passed"] and availability),
|
|
"prefixStopTime": prefix_stop,
|
|
"signalEventTimes": signal,
|
|
"mechanicalTransitionTimes": mechanical,
|
|
"mechanicalTransitionTimesAvailable": availability,
|
|
"leftStateTransitionCount": left_trace.get("stateTransitionCount"),
|
|
"rightStateTransitionCount": right_trace.get("stateTransitionCount"),
|
|
}
|
|
|
|
|
|
def _ratio(right: float, left: float) -> float | None:
|
|
if left == 0.0:
|
|
return 1.0 if right == 0.0 else None
|
|
return right / left
|
|
|
|
|
|
def _pair_diagnostics(
|
|
left: Mapping[str, object],
|
|
right: Mapping[str, object],
|
|
*,
|
|
maximum_scaled_residual: float | None,
|
|
) -> dict[str, object]:
|
|
left_residual = _scaled_residual(_summary(left))
|
|
right_residual = _scaled_residual(_summary(right))
|
|
residuals_pass = maximum_scaled_residual is None or (
|
|
left_residual is not None
|
|
and right_residual is not None
|
|
and left_residual <= maximum_scaled_residual
|
|
and right_residual <= maximum_scaled_residual
|
|
)
|
|
left_totals = _integration_totals(_summary(left))
|
|
right_totals = _integration_totals(_summary(right))
|
|
integration: dict[str, dict[str, float | None]] = {}
|
|
for key in sorted(set(left_totals) | set(right_totals)):
|
|
left_value = left_totals.get(key)
|
|
right_value = right_totals.get(key)
|
|
integration[key] = {
|
|
"left": left_value,
|
|
"right": right_value,
|
|
"delta": (
|
|
right_value - left_value
|
|
if left_value is not None and right_value is not None
|
|
else None
|
|
),
|
|
"rightOverLeft": (
|
|
_ratio(right_value, left_value)
|
|
if left_value is not None and right_value is not None
|
|
else None
|
|
),
|
|
}
|
|
return {
|
|
"scaledResidual": {
|
|
"left": left_residual,
|
|
"right": right_residual,
|
|
"limit": maximum_scaled_residual,
|
|
"passed": residuals_pass,
|
|
},
|
|
"integrationTotals": integration,
|
|
}
|
|
|
|
|
|
def compare_matrix_cases(
|
|
left: Mapping[str, object],
|
|
right: Mapping[str, object],
|
|
*,
|
|
tolerance: MatrixTolerance,
|
|
) -> dict[str, object]:
|
|
"""Compare two completed matrix cells on their common time prefix."""
|
|
|
|
left_stop = float(left["stopTime"])
|
|
right_stop = float(right["stopTime"])
|
|
prefix_stop = min(left_stop, right_stop)
|
|
if not _completed(left, left_stop) or not _completed(right, right_stop):
|
|
return {
|
|
"evaluated": False,
|
|
"passed": False,
|
|
"reason": "oneOrBothCasesDidNotComplete",
|
|
}
|
|
state = _pair_checkpoints(left, right, tolerance=tolerance)
|
|
events = _pair_events(
|
|
left,
|
|
right,
|
|
prefix_stop=prefix_stop,
|
|
tolerance=tolerance,
|
|
)
|
|
diagnostics = _pair_diagnostics(
|
|
left,
|
|
right,
|
|
maximum_scaled_residual=tolerance.maximum_scaled_residual,
|
|
)
|
|
passed = bool(
|
|
state["passed"]
|
|
and events["passed"]
|
|
and diagnostics["scaledResidual"]["passed"]
|
|
)
|
|
return {
|
|
"evaluated": True,
|
|
"passed": passed,
|
|
"commonPrefixStopTime": prefix_stop,
|
|
"stateProjection": state,
|
|
"events": events,
|
|
"diagnostics": diagnostics,
|
|
}
|
|
|
|
|
|
def _cell_identity(cell: Mapping[str, object]) -> dict[str, object]:
|
|
return {
|
|
"matrixCaseId": cell.get("matrixCaseId"),
|
|
"horizonCaseId": cell.get("horizonCaseId"),
|
|
"stopTime": cell.get("stopTime"),
|
|
"maxStep": cell.get("maxStep"),
|
|
}
|
|
|
|
|
|
def _comparison_record(
|
|
kind: str,
|
|
left: Mapping[str, object],
|
|
right: Mapping[str, object],
|
|
tolerance: MatrixTolerance,
|
|
) -> dict[str, object]:
|
|
return {
|
|
"kind": kind,
|
|
"left": _cell_identity(left),
|
|
"right": _cell_identity(right),
|
|
**compare_matrix_cases(left, right, tolerance=tolerance),
|
|
}
|
|
|
|
|
|
def _build_comparisons(
|
|
cases: Sequence[Mapping[str, object]],
|
|
*,
|
|
horizon_case_ids: Sequence[str],
|
|
max_steps: Sequence[float],
|
|
tolerance: MatrixTolerance,
|
|
) -> dict[str, object]:
|
|
by_identity = {
|
|
(str(case.get("horizonCaseId")), float(case.get("maxStep"))): case
|
|
for case in cases
|
|
if case.get("outcome") != "deferred"
|
|
and isinstance(case.get("maxStep"), (int, float))
|
|
}
|
|
same_horizon: list[dict[str, object]] = []
|
|
for horizon_case_id in horizon_case_ids:
|
|
for left_step, right_step in combinations(max_steps, 2):
|
|
left = by_identity.get((horizon_case_id, float(left_step)))
|
|
right = by_identity.get((horizon_case_id, float(right_step)))
|
|
if left is not None and right is not None:
|
|
same_horizon.append(
|
|
_comparison_record(
|
|
"sameHorizonAcrossMaxSteps", left, right, tolerance
|
|
)
|
|
)
|
|
same_max_step: list[dict[str, object]] = []
|
|
for max_step in max_steps:
|
|
for left_horizon, right_horizon in combinations(horizon_case_ids, 2):
|
|
left = by_identity.get((left_horizon, float(max_step)))
|
|
right = by_identity.get((right_horizon, float(max_step)))
|
|
if left is not None and right is not None:
|
|
same_max_step.append(
|
|
_comparison_record(
|
|
"sameMaxStepAcrossHorizons", left, right, tolerance
|
|
)
|
|
)
|
|
evaluated = [*same_horizon, *same_max_step]
|
|
comparisons_passed = (
|
|
len(by_identity) == 1 and not evaluated
|
|
) or (
|
|
bool(evaluated)
|
|
and all(
|
|
bool(item.get("evaluated")) and bool(item.get("passed"))
|
|
for item in evaluated
|
|
)
|
|
)
|
|
return {
|
|
"sameHorizonAcrossMaxSteps": same_horizon,
|
|
"sameMaxStepAcrossHorizons": same_max_step,
|
|
"evaluatedCount": sum(bool(item.get("evaluated")) for item in evaluated),
|
|
"failedCount": sum(
|
|
bool(item.get("evaluated")) and not bool(item.get("passed"))
|
|
for item in evaluated
|
|
),
|
|
"passed": comparisons_passed,
|
|
}
|
|
|
|
|
|
def _max_step_slug(value: float) -> str:
|
|
return format(value, ".12g").replace("-", "m").replace(".", "p")
|
|
|
|
|
|
def run_max_step_matrix(
|
|
manifest_path: Path | str = DEFAULT_MANIFEST_PATH,
|
|
*,
|
|
lane: str = "production",
|
|
horizon_case_ids: Sequence[str | float] = DEFAULT_HORIZON_CASE_IDS,
|
|
max_steps: Sequence[float],
|
|
additional_checkpoint_times: Sequence[float] = (),
|
|
soft_timeout_seconds: float | None = None,
|
|
hard_timeout_seconds: float | None = None,
|
|
expected_projection_count: int | None = DEFAULT_EXPECTED_PROJECTION_COUNT,
|
|
stop_after_failed_tier: bool = True,
|
|
case_executor: MatrixCaseExecutor = execute_regression_case,
|
|
) -> dict[str, object]:
|
|
"""Run a staged, sequential max-step matrix without mutating its source XML."""
|
|
|
|
manifest = load_regression_manifest(manifest_path)
|
|
lanes = manifest.get("lanes")
|
|
if not isinstance(lanes, Mapping) or lane not in lanes:
|
|
raise RegressionManifestError(f"Unknown regression lane {lane!r}.")
|
|
lane_config = lanes[lane]
|
|
if not isinstance(lane_config, Mapping):
|
|
raise RegressionManifestError(f"Lane {lane!r} must be an object.")
|
|
selected_horizons = _resolve_horizons(
|
|
manifest,
|
|
horizon_case_ids,
|
|
additional_checkpoint_times=additional_checkpoint_times,
|
|
)
|
|
selected_max_steps = _normalise_max_steps(max_steps)
|
|
if expected_projection_count is not None and expected_projection_count <= 0:
|
|
raise RegressionManifestError("expectedProjectionCount must be positive.")
|
|
soft_override = (
|
|
_finite_positive(soft_timeout_seconds, field="softTimeoutSeconds")
|
|
if soft_timeout_seconds is not None
|
|
else None
|
|
)
|
|
hard_override = (
|
|
_finite_positive(hard_timeout_seconds, field="hardTimeoutSeconds")
|
|
if hard_timeout_seconds is not None
|
|
else None
|
|
)
|
|
if soft_override is not None and hard_override is not None and (
|
|
hard_override <= soft_override
|
|
):
|
|
raise RegressionManifestError(
|
|
"hardTimeoutSeconds must exceed softTimeoutSeconds."
|
|
)
|
|
|
|
source_path = Path(str(manifest["_sourcePath"]))
|
|
source_payload = source_path.read_bytes()
|
|
source_config = source_simulation_config(source_payload)
|
|
sampling_mode = lane_config.get("samplingMode", "source")
|
|
sample_step = (
|
|
float(source_config["sampleStep"])
|
|
if sampling_mode == "source"
|
|
else _finite_positive(lane_config.get("sampleStep"), field="lane.sampleStep")
|
|
)
|
|
instrumentation_mode = str(
|
|
lane_config.get("instrumentationMode", "standard")
|
|
)
|
|
execution = manifest.get("execution")
|
|
if not isinstance(execution, Mapping):
|
|
raise RegressionManifestError("Manifest execution must be an object.")
|
|
raw_environment = execution.get("environment", {})
|
|
if not isinstance(raw_environment, Mapping) or not all(
|
|
isinstance(key, str) and isinstance(value, str)
|
|
for key, value in raw_environment.items()
|
|
):
|
|
raise RegressionManifestError("execution.environment must map strings to strings.")
|
|
environment_overrides = tuple(sorted(raw_environment.items()))
|
|
termination_grace = _finite_positive(
|
|
execution.get("terminationGraceSeconds", 5.0),
|
|
field="execution.terminationGraceSeconds",
|
|
)
|
|
expected_sha256 = str(manifest["source"]["sha256"]) # type: ignore[index]
|
|
tolerance = _matrix_tolerance(manifest)
|
|
|
|
cells: list[dict[str, object]] = []
|
|
tier_decisions: list[dict[str, object]] = []
|
|
later_tiers_enabled = True
|
|
for horizon in selected_horizons:
|
|
horizon_case_id = horizon.case_id
|
|
stop_time = horizon.stop_time
|
|
soft_timeout = soft_override or horizon.soft_timeout_seconds
|
|
hard_timeout = hard_override or horizon.hard_timeout_seconds
|
|
if hard_timeout <= soft_timeout:
|
|
raise RegressionManifestError(
|
|
f"Hard timeout must exceed soft timeout for {horizon_case_id!r}."
|
|
)
|
|
tier_cells: list[dict[str, object]] = []
|
|
if not later_tiers_enabled:
|
|
for max_step in selected_max_steps:
|
|
cell = {
|
|
"matrixCaseId": (
|
|
f"{horizon_case_id}__max_step_{_max_step_slug(max_step)}"
|
|
),
|
|
"horizonCaseId": horizon_case_id,
|
|
"sourceVariantId": horizon.source_variant_id,
|
|
"stopTime": stop_time,
|
|
"sampleStep": sample_step,
|
|
"maxStep": max_step,
|
|
"lane": lane,
|
|
"softTimeoutSeconds": soft_timeout,
|
|
"hardTimeoutSeconds": hard_timeout,
|
|
"outcome": "deferred",
|
|
"reason": "previousTierDidNotPass",
|
|
"matrixAcceptance": {
|
|
"evaluated": False,
|
|
"passed": False,
|
|
"issues": ["previousTierDidNotPass"],
|
|
},
|
|
}
|
|
cells.append(cell)
|
|
tier_cells.append(cell)
|
|
tier_decisions.append(
|
|
{
|
|
"horizonCaseId": horizon_case_id,
|
|
"sourceVariantId": horizon.source_variant_id,
|
|
"stopTime": stop_time,
|
|
"executed": False,
|
|
"passed": False,
|
|
"reason": "previousTierDidNotPass",
|
|
}
|
|
)
|
|
continue
|
|
|
|
for max_step in selected_max_steps:
|
|
matrix_case_id = (
|
|
f"{horizon_case_id}__max_step_{_max_step_slug(max_step)}"
|
|
)
|
|
request = RegressionCaseRequest(
|
|
case_id=matrix_case_id,
|
|
source_path=source_path,
|
|
expected_sha256=expected_sha256,
|
|
lane=lane,
|
|
stop_time=stop_time,
|
|
sample_step=sample_step,
|
|
max_step=max_step,
|
|
checkpoint_times=tuple(
|
|
horizon.checkpoint_times
|
|
),
|
|
soft_timeout_seconds=soft_timeout,
|
|
hard_timeout_seconds=hard_timeout,
|
|
termination_grace_seconds=termination_grace,
|
|
instrumentation_mode=instrumentation_mode,
|
|
environment_overrides=environment_overrides,
|
|
)
|
|
result = case_executor(request)
|
|
base_cell = {
|
|
"matrixCaseId": matrix_case_id,
|
|
"horizonCaseId": horizon_case_id,
|
|
"sourceVariantId": horizon.source_variant_id,
|
|
"stopTime": stop_time,
|
|
"sampleStep": sample_step,
|
|
"maxStep": max_step,
|
|
"lane": lane,
|
|
"softTimeoutSeconds": soft_timeout,
|
|
"hardTimeoutSeconds": hard_timeout,
|
|
**result,
|
|
}
|
|
observation, issues = _case_observation(
|
|
base_cell,
|
|
stop_time=stop_time,
|
|
expected_projection_count=expected_projection_count,
|
|
tolerance=tolerance,
|
|
expected_signal_event_times=horizon.expected_signal_event_times,
|
|
)
|
|
cell = {
|
|
**base_cell,
|
|
"matrixObservation": observation,
|
|
"matrixAcceptance": {
|
|
"evaluated": True,
|
|
"passed": not issues,
|
|
"issues": list(issues),
|
|
},
|
|
}
|
|
cells.append(cell)
|
|
tier_cells.append(cell)
|
|
tier_passed = all(
|
|
bool(cell.get("matrixAcceptance", {}).get("passed"))
|
|
for cell in tier_cells
|
|
if isinstance(cell.get("matrixAcceptance"), Mapping)
|
|
)
|
|
tier_decisions.append(
|
|
{
|
|
"horizonCaseId": horizon_case_id,
|
|
"stopTime": stop_time,
|
|
"executed": True,
|
|
"passed": tier_passed,
|
|
"cellCount": len(tier_cells),
|
|
}
|
|
)
|
|
if stop_after_failed_tier and not tier_passed:
|
|
later_tiers_enabled = False
|
|
|
|
comparisons = _build_comparisons(
|
|
cells,
|
|
horizon_case_ids=[horizon.case_id for horizon in selected_horizons],
|
|
max_steps=selected_max_steps,
|
|
tolerance=tolerance,
|
|
)
|
|
executed_acceptance = [
|
|
cell.get("matrixAcceptance")
|
|
for cell in cells
|
|
if cell.get("outcome") != "deferred"
|
|
and isinstance(cell.get("matrixAcceptance"), Mapping)
|
|
]
|
|
deferred_count = sum(cell.get("outcome") == "deferred" for cell in cells)
|
|
public_manifest = {
|
|
key: value for key, value in manifest.items() if not key.startswith("_")
|
|
}
|
|
overall_passed = (
|
|
bool(executed_acceptance)
|
|
and not deferred_count
|
|
and all(bool(item.get("passed")) for item in executed_acceptance)
|
|
and bool(comparisons["passed"])
|
|
)
|
|
return {
|
|
"schemaVersion": MATRIX_REPORT_SCHEMA_VERSION,
|
|
"reportKind": "maxStepHorizonMatrix",
|
|
"generatedAt": datetime.now(UTC).isoformat(),
|
|
"manifestId": manifest.get("id"),
|
|
"manifestPath": str(manifest["_manifestPath"]),
|
|
"lane": lane,
|
|
"source": {
|
|
**dict(manifest["source"]), # type: ignore[arg-type]
|
|
"resolvedPath": str(source_path),
|
|
"bytes": len(source_payload),
|
|
"simulation": source_config,
|
|
"mutationPolicy": "readOnly; per-cell overrides are child-memory only",
|
|
},
|
|
"configuration": {
|
|
"horizons": [
|
|
{
|
|
"horizonCaseId": horizon.case_id,
|
|
"sourceVariantId": horizon.source_variant_id,
|
|
"stopTime": horizon.stop_time,
|
|
"checkpointTimes": list(horizon.checkpoint_times),
|
|
"defaultSoftTimeoutSeconds": horizon.soft_timeout_seconds,
|
|
"defaultHardTimeoutSeconds": horizon.hard_timeout_seconds,
|
|
}
|
|
for horizon in selected_horizons
|
|
],
|
|
"additionalCheckpointTimes": [
|
|
float(value) for value in additional_checkpoint_times
|
|
],
|
|
"maxSteps": list(selected_max_steps),
|
|
"sampleStep": sample_step,
|
|
"samplingMode": sampling_mode,
|
|
"softTimeoutOverrideSeconds": soft_override,
|
|
"hardTimeoutOverrideSeconds": hard_override,
|
|
"expectedProjectionCount": expected_projection_count,
|
|
"stopAfterFailedTier": stop_after_failed_tier,
|
|
"executionOrder": "horizon-major, max-step serial",
|
|
"tolerance": {
|
|
"stateRelative": tolerance.state_relative,
|
|
"stateAbsolute": tolerance.state_absolute,
|
|
"checkpointTimeAbsoluteSeconds": (
|
|
tolerance.checkpoint_time_absolute_seconds
|
|
),
|
|
"eventTimeAbsoluteSeconds": tolerance.event_time_absolute_seconds,
|
|
"signalEventTimeAbsoluteSeconds": (
|
|
tolerance.signal_event_time_absolute_seconds
|
|
),
|
|
"maximumScaledResidual": tolerance.maximum_scaled_residual,
|
|
},
|
|
},
|
|
"manifest": public_manifest,
|
|
"tierDecisions": tier_decisions,
|
|
"cases": cells,
|
|
"comparisons": comparisons,
|
|
"acceptance": {
|
|
"passed": overall_passed,
|
|
"executedCellCount": len(executed_acceptance),
|
|
"deferredCellCount": deferred_count,
|
|
"caseFailureCount": sum(
|
|
not bool(item.get("passed")) for item in executed_acceptance
|
|
),
|
|
"comparisonFailureCount": comparisons["failedCount"],
|
|
},
|
|
}
|
|
|
|
|
|
def _parse_arguments(argv: Sequence[str] | None = None) -> argparse.Namespace:
|
|
parser = argparse.ArgumentParser(
|
|
description=(
|
|
"Run a bounded 1s/5s/10s matrix over multiple maximum integration steps."
|
|
)
|
|
)
|
|
parser.add_argument("--manifest", type=Path, default=DEFAULT_MANIFEST_PATH)
|
|
parser.add_argument("--lane", default="production")
|
|
parser.add_argument(
|
|
"--horizon",
|
|
action="append",
|
|
default=[],
|
|
help="Manifest case id or stop time; repeat in sequence order (default: 1s,5s,10s).",
|
|
)
|
|
parser.add_argument(
|
|
"--checkpoint",
|
|
type=float,
|
|
action="append",
|
|
default=[],
|
|
help=(
|
|
"Additional checkpoint for every horizon that reaches it; repeat for event-neighbour probes."
|
|
),
|
|
)
|
|
parser.add_argument(
|
|
"--max-step",
|
|
type=float,
|
|
action="append",
|
|
required=True,
|
|
help="Maximum integration step; repeat to form matrix columns.",
|
|
)
|
|
parser.add_argument("--soft-timeout", type=float)
|
|
parser.add_argument("--hard-timeout", type=float)
|
|
parser.add_argument(
|
|
"--expected-projection-count",
|
|
type=int,
|
|
default=DEFAULT_EXPECTED_PROJECTION_COUNT,
|
|
)
|
|
parser.add_argument(
|
|
"--continue-after-failure",
|
|
action="store_true",
|
|
help="Run later horizons even when a preceding horizon tier fails.",
|
|
)
|
|
parser.add_argument("--output", type=Path)
|
|
return parser.parse_args(argv)
|
|
|
|
|
|
def main(argv: Sequence[str] | None = None) -> int:
|
|
arguments = _parse_arguments(argv)
|
|
report = run_max_step_matrix(
|
|
arguments.manifest,
|
|
lane=arguments.lane,
|
|
horizon_case_ids=arguments.horizon or DEFAULT_HORIZON_CASE_IDS,
|
|
max_steps=arguments.max_step,
|
|
additional_checkpoint_times=arguments.checkpoint,
|
|
soft_timeout_seconds=arguments.soft_timeout,
|
|
hard_timeout_seconds=arguments.hard_timeout,
|
|
expected_projection_count=arguments.expected_projection_count,
|
|
stop_after_failed_tier=not arguments.continue_after_failure,
|
|
)
|
|
serialized = json.dumps(report, ensure_ascii=False, indent=2, default=str) + "\n"
|
|
if arguments.output is None:
|
|
print(serialized, end="")
|
|
else:
|
|
arguments.output.parent.mkdir(parents=True, exist_ok=True)
|
|
arguments.output.write_text(serialized, encoding="utf-8")
|
|
print(f"Max-step matrix report written to {arguments.output.resolve()}")
|
|
if report["acceptance"]["passed"]:
|
|
return 0
|
|
if report["acceptance"]["deferredCellCount"] and not report["acceptance"][
|
|
"caseFailureCount"
|
|
]:
|
|
return 2
|
|
return 1
|
|
|
|
|
|
if __name__ == "__main__":
|
|
raise SystemExit(main())
|