"""Reproducible horizon/max-step matrix runs for one authoritative System XML. The module deliberately delegates every simulation to ``execute_regression_case``. Consequently each cell inherits the existing soft-cancel/hard-termination boundary and derives its XML in child-process memory; the source XML is never rewritten. The two comparison axes answer different questions: * same horizon, different maximum steps: ordinary step-size sensitivity; * same maximum step, different horizons: prefix invariance with respect to ``tStop`` (a longer request should not change already-reached checkpoints). This is a diagnostic matrix, not a golden generator. Numerical state values are retained in its JSON evidence but are never written back to a manifest. """ from __future__ import annotations import argparse from collections.abc import Callable, Mapping, Sequence from dataclasses import dataclass from datetime import UTC, datetime from itertools import combinations import json import math from pathlib import Path from app.simulation.benchmark_regression import ( DEFAULT_MANIFEST_PATH, RegressionCaseRequest, RegressionManifestError, execute_regression_case, load_regression_manifest, source_simulation_config, ) MATRIX_REPORT_SCHEMA_VERSION = 1 DEFAULT_HORIZON_CASE_IDS = ("1s", "5s", "10s") DEFAULT_COMMON_CHECKPOINT_TIMES = (0.0, 0.04, 0.8, 1.0) DEFAULT_EXPECTED_PROJECTION_COUNT = 134 MatrixCaseExecutor = Callable[[RegressionCaseRequest], dict[str, object]] @dataclass(frozen=True) class MatrixTolerance: state_relative: float state_absolute: float checkpoint_time_absolute_seconds: float event_time_absolute_seconds: float signal_event_time_absolute_seconds: float maximum_scaled_residual: float | None @dataclass(frozen=True) class MatrixHorizon: case_id: str source_variant_id: str | None stop_time: float checkpoint_times: tuple[float, ...] expected_signal_event_times: tuple[float, ...] soft_timeout_seconds: float hard_timeout_seconds: float def _finite_positive(value: object, *, field: str) -> float: try: numeric = float(value) except (TypeError, ValueError) as exc: raise RegressionManifestError(f"{field} must be numeric.") from exc if not math.isfinite(numeric) or numeric <= 0.0: raise RegressionManifestError(f"{field} must be finite and positive.") return numeric def _finite_nonnegative(value: object, *, field: str) -> float: try: numeric = float(value) except (TypeError, ValueError) as exc: raise RegressionManifestError(f"{field} must be numeric.") from exc if not math.isfinite(numeric) or numeric < 0.0: raise RegressionManifestError( f"{field} must be finite and non-negative." ) return numeric def _matrix_tolerance(manifest: Mapping[str, object]) -> MatrixTolerance: correctness = manifest.get("correctness") if not isinstance(correctness, Mapping): raise RegressionManifestError("Manifest correctness must be an object.") maximum_residual = correctness.get("maximumScaledResidual") return MatrixTolerance( state_relative=_finite_nonnegative( correctness.get("stateRelativeTolerance", 0.0), field="correctness.stateRelativeTolerance", ), state_absolute=_finite_nonnegative( correctness.get("stateAbsoluteTolerance", 0.0), field="correctness.stateAbsoluteTolerance", ), checkpoint_time_absolute_seconds=_finite_nonnegative( correctness.get("checkpointTimeAbsoluteToleranceSeconds", 1.0e-12), field="correctness.checkpointTimeAbsoluteToleranceSeconds", ), event_time_absolute_seconds=_finite_nonnegative( correctness.get("eventTimeAbsoluteToleranceSeconds", 0.0), field="correctness.eventTimeAbsoluteToleranceSeconds", ), signal_event_time_absolute_seconds=_finite_nonnegative( correctness.get("signalEventTimeAbsoluteToleranceSeconds", 0.0), field="correctness.signalEventTimeAbsoluteToleranceSeconds", ), maximum_scaled_residual=( _finite_nonnegative( maximum_residual, field="correctness.maximumScaledResidual", ) if maximum_residual is not None else None ), ) def _resolve_horizons( manifest: Mapping[str, object], requested: Sequence[str | float], *, additional_checkpoint_times: Sequence[float], ) -> tuple[MatrixHorizon, ...]: variants = manifest.get("variants") sequence = manifest.get("sequence") if not isinstance(variants, Mapping) or not isinstance(sequence, list): raise RegressionManifestError("Manifest variants/sequence are incomplete.") extra_checkpoints = tuple( _finite_nonnegative(value, field=f"additionalCheckpointTimes[{index}]") for index, value in enumerate(additional_checkpoint_times) ) ordered_variants: list[tuple[str, Mapping[str, object]]] = [] for raw_case_id in sequence: variant = variants.get(str(raw_case_id)) if isinstance(variant, Mapping): ordered_variants.append((str(raw_case_id), variant)) selected: list[MatrixHorizon] = [] for raw_request in requested: request = str(raw_request).strip() if request in variants: source_variant_id: str | None = request variant = variants[request] assert isinstance(variant, Mapping) stop_time = float(variant["stopTime"]) else: try: stop_time = float(request.removesuffix("s")) except ValueError as exc: raise RegressionManifestError( f"Unknown matrix horizon {raw_request!r}." ) from exc stop_time = _finite_positive(stop_time, field=f"horizon[{request}]") matches = [ (str(candidate), variant) for candidate, variant in variants.items() if isinstance(variant, Mapping) and isinstance(variant.get("stopTime"), (int, float)) and math.isclose( float(variant["stopTime"]), stop_time, rel_tol=0.0, abs_tol=max(1.0e-12, 8.0 * math.ulp(max(1.0, abs(stop_time)))), ) ] if len(matches) > 1: raise RegressionManifestError( f"Matrix horizon {raw_request!r} identifies multiple variants." ) if matches: source_variant_id, variant = matches[0] else: source_variant_id = None variant = None case_id = source_variant_id or f"{format(stop_time, '.12g')}s-custom" if any(item.case_id == case_id for item in selected): raise RegressionManifestError( f"Matrix horizon {case_id!r} was selected more than once." ) timeout_variant = variant if timeout_variant is None: enclosing = [ candidate for _, candidate in ordered_variants if float(candidate["stopTime"]) >= stop_time ] timeout_variant = enclosing[0] if enclosing else ordered_variants[-1][1] raw_checkpoints = ( variant.get("checkpointTimes", []) if isinstance(variant, Mapping) else [] ) checkpoint_candidates = [ *DEFAULT_COMMON_CHECKPOINT_TIMES, *( float(value) for value in raw_checkpoints if isinstance(value, (int, float)) ), *extra_checkpoints, stop_time, ] checkpoint_times = tuple( sorted( { float(value) for value in checkpoint_candidates if 0.0 <= float(value) <= stop_time } ) ) if isinstance(variant, Mapping) and isinstance( variant.get("expectedSignalEventTimes"), list ): expected_signal_times = tuple( float(value) for value in variant["expectedSignalEventTimes"] ) else: expected_signal_times = tuple( sorted( { float(value) for _, candidate in ordered_variants for value in candidate.get("expectedSignalEventTimes", []) if isinstance(value, (int, float)) and float(value) <= stop_time } ) ) selected.append( MatrixHorizon( case_id=case_id, source_variant_id=source_variant_id, stop_time=stop_time, checkpoint_times=checkpoint_times, expected_signal_event_times=expected_signal_times, soft_timeout_seconds=_finite_positive( timeout_variant.get("softTimeoutSeconds"), field=f"horizon[{request}].softTimeoutSeconds", ), hard_timeout_seconds=_finite_positive( timeout_variant.get("hardTimeoutSeconds"), field=f"horizon[{request}].hardTimeoutSeconds", ), ) ) if not selected: raise RegressionManifestError("At least one matrix horizon is required.") stop_times = [item.stop_time for item in selected] if any(right <= left for left, right in zip(stop_times, stop_times[1:])): raise RegressionManifestError( "Matrix horizons must have strictly increasing stop times." ) return tuple(selected) def _normalise_max_steps(max_steps: Sequence[float]) -> tuple[float, ...]: values = tuple( _finite_positive(value, field=f"maxSteps[{index}]") for index, value in enumerate(max_steps) ) if not values: raise RegressionManifestError("At least one maximum step is required.") if len(set(values)) != len(values): raise RegressionManifestError("Maximum steps must be unique.") return values def _summary(case: Mapping[str, object]) -> Mapping[str, object] | None: worker = case.get("worker") if not isinstance(worker, Mapping): return None summary = worker.get("summary") return summary if isinstance(summary, Mapping) else None def _completed(case: Mapping[str, object], stop_time: float) -> bool: summary = _summary(case) simulated_until = summary.get("simulatedUntil") if summary is not None else None return ( case.get("outcome") == "completed" and summary is not None and bool(summary.get("success")) and summary.get("status") == "completed" and isinstance(simulated_until, (int, float)) and math.isclose( float(simulated_until), stop_time, rel_tol=0.0, abs_tol=max(1.0e-12, 8.0 * math.ulp(max(1.0, abs(stop_time)))), ) ) def _checkpoints(summary: Mapping[str, object] | None) -> list[Mapping[str, object]]: if summary is None: return [] contract = summary.get("physicalContract") raw = ( contract.get("checkpoints") if isinstance(contract, Mapping) else summary.get("checkpoints") ) if not isinstance(raw, list): return [] return [checkpoint for checkpoint in raw if isinstance(checkpoint, Mapping)] def _event_trace(summary: Mapping[str, object] | None) -> Mapping[str, object]: if summary is None: return {} contract = summary.get("physicalContract") raw = ( contract.get("eventTrace") if isinstance(contract, Mapping) else summary.get("eventTrace") ) return raw if isinstance(raw, Mapping) else {} def _diagnostics(summary: Mapping[str, object] | None) -> Mapping[str, object]: if summary is None: return {} raw = summary.get("diagnostics") return raw if isinstance(raw, Mapping) else {} def _scaled_residual(summary: Mapping[str, object] | None) -> float | None: pressure_flow = _diagnostics(summary).get("pressureFlow") value = ( pressure_flow.get("maxScaledResidual") if isinstance(pressure_flow, Mapping) else None ) if isinstance(value, (int, float)) and math.isfinite(float(value)): return float(value) return None def _integration_totals(summary: Mapping[str, object] | None) -> dict[str, float]: integration = _diagnostics(summary).get("integration") totals = integration.get("totals") if isinstance(integration, Mapping) else None if not isinstance(totals, Mapping): return {} return { str(key): float(value) for key, value in totals.items() if isinstance(value, (int, float)) and math.isfinite(float(value)) } def _finite_event_times(value: object) -> list[float] | None: if not isinstance(value, list): return None if any( not isinstance(item, (int, float)) or not math.isfinite(float(item)) for item in value ): return None return [float(item) for item in value] def _case_observation( case: Mapping[str, object], *, stop_time: float, expected_projection_count: int | None, tolerance: MatrixTolerance, expected_signal_event_times: Sequence[float] | None, ) -> tuple[dict[str, object], tuple[str, ...]]: summary = _summary(case) checkpoints = _checkpoints(summary) checkpoint_details: list[dict[str, object]] = [] issues: list[str] = [] for checkpoint in checkpoints: values = checkpoint.get("stateValues") state_values = values if isinstance(values, Mapping) else {} nonfinite = [ str(key) for key, value in state_values.items() if not isinstance(value, (int, float)) or not math.isfinite(float(value)) ] detail = { "requestedTime": checkpoint.get("requestedTime"), "actualTime": checkpoint.get("actualTime"), "available": bool(checkpoint.get("available")), "projectionKeyCount": len(state_values), "nonfiniteKeyCount": len(nonfinite), } checkpoint_details.append(detail) if not detail["available"]: issues.append("checkpointUnavailable") if expected_projection_count is not None and len(state_values) != int( expected_projection_count ): issues.append("projectionCountMismatch") if nonfinite: issues.append("nonfiniteProjection") if not checkpoints: issues.append("checkpointsUnavailable") residual = _scaled_residual(summary) if tolerance.maximum_scaled_residual is not None and ( residual is None or residual > tolerance.maximum_scaled_residual ): issues.append("scaledResidualExceeded") events = _event_trace(summary) signal_times = _finite_event_times(events.get("signalEventTimes")) if expected_signal_event_times is not None: expected = [float(value) for value in expected_signal_event_times] if signal_times is None or len(signal_times) != len(expected) or any( not math.isclose( actual, wanted, rel_tol=0.0, abs_tol=tolerance.signal_event_time_absolute_seconds, ) for actual, wanted in zip(signal_times or (), expected) ): issues.append("signalEventTraceMismatch") if events.get("mechanicalTransitionTimesAvailable") is False: issues.append("mechanicalTransitionTimesUnavailable") if not _completed(case, stop_time): issues.insert(0, "simulationDidNotComplete") unique_issues = tuple(dict.fromkeys(issues)) return ( { "completed": _completed(case, stop_time), "checkpointCount": len(checkpoints), "expectedProjectionCount": expected_projection_count, "checkpoints": checkpoint_details, "events": dict(events), "maximumScaledResidual": residual, "integrationTotals": _integration_totals(summary), "passed": not unique_issues, "issues": list(unique_issues), }, unique_issues, ) def _pair_checkpoints( left: Mapping[str, object], right: Mapping[str, object], *, tolerance: MatrixTolerance, ) -> dict[str, object]: left_checkpoints = _checkpoints(_summary(left)) right_checkpoints = _checkpoints(_summary(right)) paired: list[tuple[Mapping[str, object], Mapping[str, object]]] = [] for left_checkpoint in left_checkpoints: left_time = left_checkpoint.get("requestedTime") if not isinstance(left_time, (int, float)): continue match = next( ( right_checkpoint for right_checkpoint in right_checkpoints if isinstance(right_checkpoint.get("requestedTime"), (int, float)) and math.isclose( float(right_checkpoint["requestedTime"]), float(left_time), rel_tol=0.0, abs_tol=tolerance.checkpoint_time_absolute_seconds, ) ), None, ) if match is not None: paired.append((left_checkpoint, match)) comparisons: list[dict[str, object]] = [] mismatch_count = 0 key_set_mismatch_count = 0 nonnumeric_count = 0 maximum_absolute_difference = 0.0 maximum_relative_difference = 0.0 for left_checkpoint, right_checkpoint in paired: left_values = left_checkpoint.get("stateValues") right_values = right_checkpoint.get("stateValues") left_mapping = left_values if isinstance(left_values, Mapping) else {} right_mapping = right_values if isinstance(right_values, Mapping) else {} left_keys = set(map(str, left_mapping)) right_keys = set(map(str, right_mapping)) missing_from_left = sorted(right_keys - left_keys) missing_from_right = sorted(left_keys - right_keys) key_set_mismatch_count += len(missing_from_left) + len(missing_from_right) mismatches: list[dict[str, object]] = [] for key in sorted(left_keys & right_keys): left_value = left_mapping.get(key) right_value = right_mapping.get(key) if not isinstance(left_value, (int, float)) or not isinstance( right_value, (int, float) ) or not math.isfinite(float(left_value)) or not math.isfinite( float(right_value) ): nonnumeric_count += 1 mismatches.append( {"key": key, "left": left_value, "right": right_value} ) continue left_numeric = float(left_value) right_numeric = float(right_value) absolute = abs(left_numeric - right_numeric) denominator = max(abs(left_numeric), abs(right_numeric)) relative = absolute / denominator if denominator else 0.0 maximum_absolute_difference = max(maximum_absolute_difference, absolute) maximum_relative_difference = max(maximum_relative_difference, relative) if not math.isclose( left_numeric, right_numeric, rel_tol=tolerance.state_relative, abs_tol=tolerance.state_absolute, ): mismatches.append( { "key": key, "left": left_numeric, "right": right_numeric, "absoluteDifference": absolute, "relativeDifference": relative, } ) mismatch_count += len(mismatches) comparisons.append( { "requestedTime": left_checkpoint.get("requestedTime"), "leftProjectionKeyCount": len(left_keys), "rightProjectionKeyCount": len(right_keys), "missingFromLeft": missing_from_left, "missingFromRight": missing_from_right, "valueMismatchCount": len(mismatches), "mismatches": mismatches, } ) passed = bool(paired) and not ( mismatch_count or key_set_mismatch_count or nonnumeric_count ) return { "evaluated": bool(paired), "passed": passed, "commonCheckpointCount": len(paired), "commonCheckpointTimes": [ pair[0].get("requestedTime") for pair in paired ], "valueMismatchCount": mismatch_count, "keySetMismatchCount": key_set_mismatch_count, "nonnumericValueCount": nonnumeric_count, "maximumAbsoluteDifference": maximum_absolute_difference, "maximumRelativeDifference": maximum_relative_difference, "checkpoints": comparisons, } def _compare_time_sequences( left: object, right: object, *, prefix_stop: float, absolute_tolerance: float, ) -> dict[str, object]: left_times = _finite_event_times(left) right_times = _finite_event_times(right) if left_times is None or right_times is None: return { "available": False, "passed": False, "left": left, "right": right, } left_prefix = [ value for value in left_times if value <= prefix_stop + absolute_tolerance ] right_prefix = [ value for value in right_times if value <= prefix_stop + absolute_tolerance ] passed = len(left_prefix) == len(right_prefix) and all( math.isclose( left_value, right_value, rel_tol=0.0, abs_tol=absolute_tolerance, ) for left_value, right_value in zip(left_prefix, right_prefix) ) return { "available": True, "passed": passed, "left": left_prefix, "right": right_prefix, } def _pair_events( left: Mapping[str, object], right: Mapping[str, object], *, prefix_stop: float, tolerance: MatrixTolerance, ) -> dict[str, object]: left_trace = _event_trace(_summary(left)) right_trace = _event_trace(_summary(right)) signal = _compare_time_sequences( left_trace.get("signalEventTimes"), right_trace.get("signalEventTimes"), prefix_stop=prefix_stop, absolute_tolerance=tolerance.signal_event_time_absolute_seconds, ) mechanical = _compare_time_sequences( left_trace.get("mechanicalTransitionTimes"), right_trace.get("mechanicalTransitionTimes"), prefix_stop=prefix_stop, absolute_tolerance=tolerance.event_time_absolute_seconds, ) availability = ( left_trace.get("mechanicalTransitionTimesAvailable") is not False and right_trace.get("mechanicalTransitionTimesAvailable") is not False ) return { "evaluated": bool(left_trace) and bool(right_trace), "passed": bool(signal["passed"] and mechanical["passed"] and availability), "prefixStopTime": prefix_stop, "signalEventTimes": signal, "mechanicalTransitionTimes": mechanical, "mechanicalTransitionTimesAvailable": availability, "leftStateTransitionCount": left_trace.get("stateTransitionCount"), "rightStateTransitionCount": right_trace.get("stateTransitionCount"), } def _ratio(right: float, left: float) -> float | None: if left == 0.0: return 1.0 if right == 0.0 else None return right / left def _pair_diagnostics( left: Mapping[str, object], right: Mapping[str, object], *, maximum_scaled_residual: float | None, ) -> dict[str, object]: left_residual = _scaled_residual(_summary(left)) right_residual = _scaled_residual(_summary(right)) residuals_pass = maximum_scaled_residual is None or ( left_residual is not None and right_residual is not None and left_residual <= maximum_scaled_residual and right_residual <= maximum_scaled_residual ) left_totals = _integration_totals(_summary(left)) right_totals = _integration_totals(_summary(right)) integration: dict[str, dict[str, float | None]] = {} for key in sorted(set(left_totals) | set(right_totals)): left_value = left_totals.get(key) right_value = right_totals.get(key) integration[key] = { "left": left_value, "right": right_value, "delta": ( right_value - left_value if left_value is not None and right_value is not None else None ), "rightOverLeft": ( _ratio(right_value, left_value) if left_value is not None and right_value is not None else None ), } return { "scaledResidual": { "left": left_residual, "right": right_residual, "limit": maximum_scaled_residual, "passed": residuals_pass, }, "integrationTotals": integration, } def compare_matrix_cases( left: Mapping[str, object], right: Mapping[str, object], *, tolerance: MatrixTolerance, ) -> dict[str, object]: """Compare two completed matrix cells on their common time prefix.""" left_stop = float(left["stopTime"]) right_stop = float(right["stopTime"]) prefix_stop = min(left_stop, right_stop) if not _completed(left, left_stop) or not _completed(right, right_stop): return { "evaluated": False, "passed": False, "reason": "oneOrBothCasesDidNotComplete", } state = _pair_checkpoints(left, right, tolerance=tolerance) events = _pair_events( left, right, prefix_stop=prefix_stop, tolerance=tolerance, ) diagnostics = _pair_diagnostics( left, right, maximum_scaled_residual=tolerance.maximum_scaled_residual, ) passed = bool( state["passed"] and events["passed"] and diagnostics["scaledResidual"]["passed"] ) return { "evaluated": True, "passed": passed, "commonPrefixStopTime": prefix_stop, "stateProjection": state, "events": events, "diagnostics": diagnostics, } def _cell_identity(cell: Mapping[str, object]) -> dict[str, object]: return { "matrixCaseId": cell.get("matrixCaseId"), "horizonCaseId": cell.get("horizonCaseId"), "stopTime": cell.get("stopTime"), "maxStep": cell.get("maxStep"), } def _comparison_record( kind: str, left: Mapping[str, object], right: Mapping[str, object], tolerance: MatrixTolerance, ) -> dict[str, object]: return { "kind": kind, "left": _cell_identity(left), "right": _cell_identity(right), **compare_matrix_cases(left, right, tolerance=tolerance), } def _build_comparisons( cases: Sequence[Mapping[str, object]], *, horizon_case_ids: Sequence[str], max_steps: Sequence[float], tolerance: MatrixTolerance, ) -> dict[str, object]: by_identity = { (str(case.get("horizonCaseId")), float(case.get("maxStep"))): case for case in cases if case.get("outcome") != "deferred" and isinstance(case.get("maxStep"), (int, float)) } same_horizon: list[dict[str, object]] = [] for horizon_case_id in horizon_case_ids: for left_step, right_step in combinations(max_steps, 2): left = by_identity.get((horizon_case_id, float(left_step))) right = by_identity.get((horizon_case_id, float(right_step))) if left is not None and right is not None: same_horizon.append( _comparison_record( "sameHorizonAcrossMaxSteps", left, right, tolerance ) ) same_max_step: list[dict[str, object]] = [] for max_step in max_steps: for left_horizon, right_horizon in combinations(horizon_case_ids, 2): left = by_identity.get((left_horizon, float(max_step))) right = by_identity.get((right_horizon, float(max_step))) if left is not None and right is not None: same_max_step.append( _comparison_record( "sameMaxStepAcrossHorizons", left, right, tolerance ) ) evaluated = [*same_horizon, *same_max_step] comparisons_passed = ( len(by_identity) == 1 and not evaluated ) or ( bool(evaluated) and all( bool(item.get("evaluated")) and bool(item.get("passed")) for item in evaluated ) ) return { "sameHorizonAcrossMaxSteps": same_horizon, "sameMaxStepAcrossHorizons": same_max_step, "evaluatedCount": sum(bool(item.get("evaluated")) for item in evaluated), "failedCount": sum( bool(item.get("evaluated")) and not bool(item.get("passed")) for item in evaluated ), "passed": comparisons_passed, } def _max_step_slug(value: float) -> str: return format(value, ".12g").replace("-", "m").replace(".", "p") def run_max_step_matrix( manifest_path: Path | str = DEFAULT_MANIFEST_PATH, *, lane: str = "production", horizon_case_ids: Sequence[str | float] = DEFAULT_HORIZON_CASE_IDS, max_steps: Sequence[float], additional_checkpoint_times: Sequence[float] = (), soft_timeout_seconds: float | None = None, hard_timeout_seconds: float | None = None, expected_projection_count: int | None = DEFAULT_EXPECTED_PROJECTION_COUNT, stop_after_failed_tier: bool = True, case_executor: MatrixCaseExecutor = execute_regression_case, ) -> dict[str, object]: """Run a staged, sequential max-step matrix without mutating its source XML.""" manifest = load_regression_manifest(manifest_path) lanes = manifest.get("lanes") if not isinstance(lanes, Mapping) or lane not in lanes: raise RegressionManifestError(f"Unknown regression lane {lane!r}.") lane_config = lanes[lane] if not isinstance(lane_config, Mapping): raise RegressionManifestError(f"Lane {lane!r} must be an object.") selected_horizons = _resolve_horizons( manifest, horizon_case_ids, additional_checkpoint_times=additional_checkpoint_times, ) selected_max_steps = _normalise_max_steps(max_steps) if expected_projection_count is not None and expected_projection_count <= 0: raise RegressionManifestError("expectedProjectionCount must be positive.") soft_override = ( _finite_positive(soft_timeout_seconds, field="softTimeoutSeconds") if soft_timeout_seconds is not None else None ) hard_override = ( _finite_positive(hard_timeout_seconds, field="hardTimeoutSeconds") if hard_timeout_seconds is not None else None ) if soft_override is not None and hard_override is not None and ( hard_override <= soft_override ): raise RegressionManifestError( "hardTimeoutSeconds must exceed softTimeoutSeconds." ) source_path = Path(str(manifest["_sourcePath"])) source_payload = source_path.read_bytes() source_config = source_simulation_config(source_payload) sampling_mode = lane_config.get("samplingMode", "source") sample_step = ( float(source_config["sampleStep"]) if sampling_mode == "source" else _finite_positive(lane_config.get("sampleStep"), field="lane.sampleStep") ) instrumentation_mode = str( lane_config.get("instrumentationMode", "standard") ) execution = manifest.get("execution") if not isinstance(execution, Mapping): raise RegressionManifestError("Manifest execution must be an object.") raw_environment = execution.get("environment", {}) if not isinstance(raw_environment, Mapping) or not all( isinstance(key, str) and isinstance(value, str) for key, value in raw_environment.items() ): raise RegressionManifestError("execution.environment must map strings to strings.") environment_overrides = tuple(sorted(raw_environment.items())) termination_grace = _finite_positive( execution.get("terminationGraceSeconds", 5.0), field="execution.terminationGraceSeconds", ) expected_sha256 = str(manifest["source"]["sha256"]) # type: ignore[index] tolerance = _matrix_tolerance(manifest) cells: list[dict[str, object]] = [] tier_decisions: list[dict[str, object]] = [] later_tiers_enabled = True for horizon in selected_horizons: horizon_case_id = horizon.case_id stop_time = horizon.stop_time soft_timeout = soft_override or horizon.soft_timeout_seconds hard_timeout = hard_override or horizon.hard_timeout_seconds if hard_timeout <= soft_timeout: raise RegressionManifestError( f"Hard timeout must exceed soft timeout for {horizon_case_id!r}." ) tier_cells: list[dict[str, object]] = [] if not later_tiers_enabled: for max_step in selected_max_steps: cell = { "matrixCaseId": ( f"{horizon_case_id}__max_step_{_max_step_slug(max_step)}" ), "horizonCaseId": horizon_case_id, "sourceVariantId": horizon.source_variant_id, "stopTime": stop_time, "sampleStep": sample_step, "maxStep": max_step, "lane": lane, "softTimeoutSeconds": soft_timeout, "hardTimeoutSeconds": hard_timeout, "outcome": "deferred", "reason": "previousTierDidNotPass", "matrixAcceptance": { "evaluated": False, "passed": False, "issues": ["previousTierDidNotPass"], }, } cells.append(cell) tier_cells.append(cell) tier_decisions.append( { "horizonCaseId": horizon_case_id, "sourceVariantId": horizon.source_variant_id, "stopTime": stop_time, "executed": False, "passed": False, "reason": "previousTierDidNotPass", } ) continue for max_step in selected_max_steps: matrix_case_id = ( f"{horizon_case_id}__max_step_{_max_step_slug(max_step)}" ) request = RegressionCaseRequest( case_id=matrix_case_id, source_path=source_path, expected_sha256=expected_sha256, lane=lane, stop_time=stop_time, sample_step=sample_step, max_step=max_step, checkpoint_times=tuple( horizon.checkpoint_times ), soft_timeout_seconds=soft_timeout, hard_timeout_seconds=hard_timeout, termination_grace_seconds=termination_grace, instrumentation_mode=instrumentation_mode, environment_overrides=environment_overrides, ) result = case_executor(request) base_cell = { "matrixCaseId": matrix_case_id, "horizonCaseId": horizon_case_id, "sourceVariantId": horizon.source_variant_id, "stopTime": stop_time, "sampleStep": sample_step, "maxStep": max_step, "lane": lane, "softTimeoutSeconds": soft_timeout, "hardTimeoutSeconds": hard_timeout, **result, } observation, issues = _case_observation( base_cell, stop_time=stop_time, expected_projection_count=expected_projection_count, tolerance=tolerance, expected_signal_event_times=horizon.expected_signal_event_times, ) cell = { **base_cell, "matrixObservation": observation, "matrixAcceptance": { "evaluated": True, "passed": not issues, "issues": list(issues), }, } cells.append(cell) tier_cells.append(cell) tier_passed = all( bool(cell.get("matrixAcceptance", {}).get("passed")) for cell in tier_cells if isinstance(cell.get("matrixAcceptance"), Mapping) ) tier_decisions.append( { "horizonCaseId": horizon_case_id, "stopTime": stop_time, "executed": True, "passed": tier_passed, "cellCount": len(tier_cells), } ) if stop_after_failed_tier and not tier_passed: later_tiers_enabled = False comparisons = _build_comparisons( cells, horizon_case_ids=[horizon.case_id for horizon in selected_horizons], max_steps=selected_max_steps, tolerance=tolerance, ) executed_acceptance = [ cell.get("matrixAcceptance") for cell in cells if cell.get("outcome") != "deferred" and isinstance(cell.get("matrixAcceptance"), Mapping) ] deferred_count = sum(cell.get("outcome") == "deferred" for cell in cells) public_manifest = { key: value for key, value in manifest.items() if not key.startswith("_") } overall_passed = ( bool(executed_acceptance) and not deferred_count and all(bool(item.get("passed")) for item in executed_acceptance) and bool(comparisons["passed"]) ) return { "schemaVersion": MATRIX_REPORT_SCHEMA_VERSION, "reportKind": "maxStepHorizonMatrix", "generatedAt": datetime.now(UTC).isoformat(), "manifestId": manifest.get("id"), "manifestPath": str(manifest["_manifestPath"]), "lane": lane, "source": { **dict(manifest["source"]), # type: ignore[arg-type] "resolvedPath": str(source_path), "bytes": len(source_payload), "simulation": source_config, "mutationPolicy": "readOnly; per-cell overrides are child-memory only", }, "configuration": { "horizons": [ { "horizonCaseId": horizon.case_id, "sourceVariantId": horizon.source_variant_id, "stopTime": horizon.stop_time, "checkpointTimes": list(horizon.checkpoint_times), "defaultSoftTimeoutSeconds": horizon.soft_timeout_seconds, "defaultHardTimeoutSeconds": horizon.hard_timeout_seconds, } for horizon in selected_horizons ], "additionalCheckpointTimes": [ float(value) for value in additional_checkpoint_times ], "maxSteps": list(selected_max_steps), "sampleStep": sample_step, "samplingMode": sampling_mode, "softTimeoutOverrideSeconds": soft_override, "hardTimeoutOverrideSeconds": hard_override, "expectedProjectionCount": expected_projection_count, "stopAfterFailedTier": stop_after_failed_tier, "executionOrder": "horizon-major, max-step serial", "tolerance": { "stateRelative": tolerance.state_relative, "stateAbsolute": tolerance.state_absolute, "checkpointTimeAbsoluteSeconds": ( tolerance.checkpoint_time_absolute_seconds ), "eventTimeAbsoluteSeconds": tolerance.event_time_absolute_seconds, "signalEventTimeAbsoluteSeconds": ( tolerance.signal_event_time_absolute_seconds ), "maximumScaledResidual": tolerance.maximum_scaled_residual, }, }, "manifest": public_manifest, "tierDecisions": tier_decisions, "cases": cells, "comparisons": comparisons, "acceptance": { "passed": overall_passed, "executedCellCount": len(executed_acceptance), "deferredCellCount": deferred_count, "caseFailureCount": sum( not bool(item.get("passed")) for item in executed_acceptance ), "comparisonFailureCount": comparisons["failedCount"], }, } def _parse_arguments(argv: Sequence[str] | None = None) -> argparse.Namespace: parser = argparse.ArgumentParser( description=( "Run a bounded 1s/5s/10s matrix over multiple maximum integration steps." ) ) parser.add_argument("--manifest", type=Path, default=DEFAULT_MANIFEST_PATH) parser.add_argument("--lane", default="production") parser.add_argument( "--horizon", action="append", default=[], help="Manifest case id or stop time; repeat in sequence order (default: 1s,5s,10s).", ) parser.add_argument( "--checkpoint", type=float, action="append", default=[], help=( "Additional checkpoint for every horizon that reaches it; repeat for event-neighbour probes." ), ) parser.add_argument( "--max-step", type=float, action="append", required=True, help="Maximum integration step; repeat to form matrix columns.", ) parser.add_argument("--soft-timeout", type=float) parser.add_argument("--hard-timeout", type=float) parser.add_argument( "--expected-projection-count", type=int, default=DEFAULT_EXPECTED_PROJECTION_COUNT, ) parser.add_argument( "--continue-after-failure", action="store_true", help="Run later horizons even when a preceding horizon tier fails.", ) parser.add_argument("--output", type=Path) return parser.parse_args(argv) def main(argv: Sequence[str] | None = None) -> int: arguments = _parse_arguments(argv) report = run_max_step_matrix( arguments.manifest, lane=arguments.lane, horizon_case_ids=arguments.horizon or DEFAULT_HORIZON_CASE_IDS, max_steps=arguments.max_step, additional_checkpoint_times=arguments.checkpoint, soft_timeout_seconds=arguments.soft_timeout, hard_timeout_seconds=arguments.hard_timeout, expected_projection_count=arguments.expected_projection_count, stop_after_failed_tier=not arguments.continue_after_failure, ) serialized = json.dumps(report, ensure_ascii=False, indent=2, default=str) + "\n" if arguments.output is None: print(serialized, end="") else: arguments.output.parent.mkdir(parents=True, exist_ok=True) arguments.output.write_text(serialized, encoding="utf-8") print(f"Max-step matrix report written to {arguments.output.resolve()}") if report["acceptance"]["passed"]: return 0 if report["acceptance"]["deferredCellCount"] and not report["acceptance"][ "caseFailureCount" ]: return 2 return 1 if __name__ == "__main__": raise SystemExit(main())