Files
lujingze e18399c022 整合求解器活动监控与步长回归证据
同步远端 PNL0003 诊断和大采样网格能力,语义合并活动感知的 60 秒真停滞判定与旧后端 15 分钟兼容兜底。

纳管热路径优化、15 单元运行证据、浏览器与 API 报告,并补充北京时间更新日志和遗留问题。
2026-08-19 16:24:31 +00:00

1248 lines
46 KiB
Python

"""Reproducible horizon/max-step matrix runs for one authoritative System XML.
The module deliberately delegates every simulation to
``execute_regression_case``. Consequently each cell inherits the existing
soft-cancel/hard-termination boundary and derives its XML in child-process
memory; the source XML is never rewritten.
The two comparison axes answer different questions:
* same horizon, different maximum steps: ordinary step-size sensitivity;
* same maximum step, different horizons: prefix invariance with respect to
``tStop`` (a longer request should not change already-reached checkpoints).
This is a diagnostic matrix, not a golden generator. Numerical state values
are retained in its JSON evidence but are never written back to a manifest.
"""
from __future__ import annotations
import argparse
from collections.abc import Callable, Mapping, Sequence
from dataclasses import dataclass
from datetime import UTC, datetime
from itertools import combinations
import json
import math
from pathlib import Path
from app.simulation.benchmark_regression import (
DEFAULT_MANIFEST_PATH,
RegressionCaseRequest,
RegressionManifestError,
execute_regression_case,
load_regression_manifest,
source_simulation_config,
)
MATRIX_REPORT_SCHEMA_VERSION = 1
DEFAULT_HORIZON_CASE_IDS = ("1s", "5s", "10s")
DEFAULT_COMMON_CHECKPOINT_TIMES = (0.0, 0.04, 0.8, 1.0)
DEFAULT_EXPECTED_PROJECTION_COUNT = 134
MatrixCaseExecutor = Callable[[RegressionCaseRequest], dict[str, object]]
@dataclass(frozen=True)
class MatrixTolerance:
state_relative: float
state_absolute: float
checkpoint_time_absolute_seconds: float
event_time_absolute_seconds: float
signal_event_time_absolute_seconds: float
maximum_scaled_residual: float | None
@dataclass(frozen=True)
class MatrixHorizon:
case_id: str
source_variant_id: str | None
stop_time: float
checkpoint_times: tuple[float, ...]
expected_signal_event_times: tuple[float, ...]
soft_timeout_seconds: float
hard_timeout_seconds: float
def _finite_positive(value: object, *, field: str) -> float:
try:
numeric = float(value)
except (TypeError, ValueError) as exc:
raise RegressionManifestError(f"{field} must be numeric.") from exc
if not math.isfinite(numeric) or numeric <= 0.0:
raise RegressionManifestError(f"{field} must be finite and positive.")
return numeric
def _finite_nonnegative(value: object, *, field: str) -> float:
try:
numeric = float(value)
except (TypeError, ValueError) as exc:
raise RegressionManifestError(f"{field} must be numeric.") from exc
if not math.isfinite(numeric) or numeric < 0.0:
raise RegressionManifestError(
f"{field} must be finite and non-negative."
)
return numeric
def _matrix_tolerance(manifest: Mapping[str, object]) -> MatrixTolerance:
correctness = manifest.get("correctness")
if not isinstance(correctness, Mapping):
raise RegressionManifestError("Manifest correctness must be an object.")
maximum_residual = correctness.get("maximumScaledResidual")
return MatrixTolerance(
state_relative=_finite_nonnegative(
correctness.get("stateRelativeTolerance", 0.0),
field="correctness.stateRelativeTolerance",
),
state_absolute=_finite_nonnegative(
correctness.get("stateAbsoluteTolerance", 0.0),
field="correctness.stateAbsoluteTolerance",
),
checkpoint_time_absolute_seconds=_finite_nonnegative(
correctness.get("checkpointTimeAbsoluteToleranceSeconds", 1.0e-12),
field="correctness.checkpointTimeAbsoluteToleranceSeconds",
),
event_time_absolute_seconds=_finite_nonnegative(
correctness.get("eventTimeAbsoluteToleranceSeconds", 0.0),
field="correctness.eventTimeAbsoluteToleranceSeconds",
),
signal_event_time_absolute_seconds=_finite_nonnegative(
correctness.get("signalEventTimeAbsoluteToleranceSeconds", 0.0),
field="correctness.signalEventTimeAbsoluteToleranceSeconds",
),
maximum_scaled_residual=(
_finite_nonnegative(
maximum_residual,
field="correctness.maximumScaledResidual",
)
if maximum_residual is not None
else None
),
)
def _resolve_horizons(
manifest: Mapping[str, object],
requested: Sequence[str | float],
*,
additional_checkpoint_times: Sequence[float],
) -> tuple[MatrixHorizon, ...]:
variants = manifest.get("variants")
sequence = manifest.get("sequence")
if not isinstance(variants, Mapping) or not isinstance(sequence, list):
raise RegressionManifestError("Manifest variants/sequence are incomplete.")
extra_checkpoints = tuple(
_finite_nonnegative(value, field=f"additionalCheckpointTimes[{index}]")
for index, value in enumerate(additional_checkpoint_times)
)
ordered_variants: list[tuple[str, Mapping[str, object]]] = []
for raw_case_id in sequence:
variant = variants.get(str(raw_case_id))
if isinstance(variant, Mapping):
ordered_variants.append((str(raw_case_id), variant))
selected: list[MatrixHorizon] = []
for raw_request in requested:
request = str(raw_request).strip()
if request in variants:
source_variant_id: str | None = request
variant = variants[request]
assert isinstance(variant, Mapping)
stop_time = float(variant["stopTime"])
else:
try:
stop_time = float(request.removesuffix("s"))
except ValueError as exc:
raise RegressionManifestError(
f"Unknown matrix horizon {raw_request!r}."
) from exc
stop_time = _finite_positive(stop_time, field=f"horizon[{request}]")
matches = [
(str(candidate), variant)
for candidate, variant in variants.items()
if isinstance(variant, Mapping)
and isinstance(variant.get("stopTime"), (int, float))
and math.isclose(
float(variant["stopTime"]),
stop_time,
rel_tol=0.0,
abs_tol=max(1.0e-12, 8.0 * math.ulp(max(1.0, abs(stop_time)))),
)
]
if len(matches) > 1:
raise RegressionManifestError(
f"Matrix horizon {raw_request!r} identifies multiple variants."
)
if matches:
source_variant_id, variant = matches[0]
else:
source_variant_id = None
variant = None
case_id = source_variant_id or f"{format(stop_time, '.12g')}s-custom"
if any(item.case_id == case_id for item in selected):
raise RegressionManifestError(
f"Matrix horizon {case_id!r} was selected more than once."
)
timeout_variant = variant
if timeout_variant is None:
enclosing = [
candidate
for _, candidate in ordered_variants
if float(candidate["stopTime"]) >= stop_time
]
timeout_variant = enclosing[0] if enclosing else ordered_variants[-1][1]
raw_checkpoints = (
variant.get("checkpointTimes", [])
if isinstance(variant, Mapping)
else []
)
checkpoint_candidates = [
*DEFAULT_COMMON_CHECKPOINT_TIMES,
*(
float(value)
for value in raw_checkpoints
if isinstance(value, (int, float))
),
*extra_checkpoints,
stop_time,
]
checkpoint_times = tuple(
sorted(
{
float(value)
for value in checkpoint_candidates
if 0.0 <= float(value) <= stop_time
}
)
)
if isinstance(variant, Mapping) and isinstance(
variant.get("expectedSignalEventTimes"), list
):
expected_signal_times = tuple(
float(value) for value in variant["expectedSignalEventTimes"]
)
else:
expected_signal_times = tuple(
sorted(
{
float(value)
for _, candidate in ordered_variants
for value in candidate.get("expectedSignalEventTimes", [])
if isinstance(value, (int, float))
and float(value) <= stop_time
}
)
)
selected.append(
MatrixHorizon(
case_id=case_id,
source_variant_id=source_variant_id,
stop_time=stop_time,
checkpoint_times=checkpoint_times,
expected_signal_event_times=expected_signal_times,
soft_timeout_seconds=_finite_positive(
timeout_variant.get("softTimeoutSeconds"),
field=f"horizon[{request}].softTimeoutSeconds",
),
hard_timeout_seconds=_finite_positive(
timeout_variant.get("hardTimeoutSeconds"),
field=f"horizon[{request}].hardTimeoutSeconds",
),
)
)
if not selected:
raise RegressionManifestError("At least one matrix horizon is required.")
stop_times = [item.stop_time for item in selected]
if any(right <= left for left, right in zip(stop_times, stop_times[1:])):
raise RegressionManifestError(
"Matrix horizons must have strictly increasing stop times."
)
return tuple(selected)
def _normalise_max_steps(max_steps: Sequence[float]) -> tuple[float, ...]:
values = tuple(
_finite_positive(value, field=f"maxSteps[{index}]")
for index, value in enumerate(max_steps)
)
if not values:
raise RegressionManifestError("At least one maximum step is required.")
if len(set(values)) != len(values):
raise RegressionManifestError("Maximum steps must be unique.")
return values
def _summary(case: Mapping[str, object]) -> Mapping[str, object] | None:
worker = case.get("worker")
if not isinstance(worker, Mapping):
return None
summary = worker.get("summary")
return summary if isinstance(summary, Mapping) else None
def _completed(case: Mapping[str, object], stop_time: float) -> bool:
summary = _summary(case)
simulated_until = summary.get("simulatedUntil") if summary is not None else None
return (
case.get("outcome") == "completed"
and summary is not None
and bool(summary.get("success"))
and summary.get("status") == "completed"
and isinstance(simulated_until, (int, float))
and math.isclose(
float(simulated_until),
stop_time,
rel_tol=0.0,
abs_tol=max(1.0e-12, 8.0 * math.ulp(max(1.0, abs(stop_time)))),
)
)
def _checkpoints(summary: Mapping[str, object] | None) -> list[Mapping[str, object]]:
if summary is None:
return []
contract = summary.get("physicalContract")
raw = (
contract.get("checkpoints")
if isinstance(contract, Mapping)
else summary.get("checkpoints")
)
if not isinstance(raw, list):
return []
return [checkpoint for checkpoint in raw if isinstance(checkpoint, Mapping)]
def _event_trace(summary: Mapping[str, object] | None) -> Mapping[str, object]:
if summary is None:
return {}
contract = summary.get("physicalContract")
raw = (
contract.get("eventTrace")
if isinstance(contract, Mapping)
else summary.get("eventTrace")
)
return raw if isinstance(raw, Mapping) else {}
def _diagnostics(summary: Mapping[str, object] | None) -> Mapping[str, object]:
if summary is None:
return {}
raw = summary.get("diagnostics")
return raw if isinstance(raw, Mapping) else {}
def _scaled_residual(summary: Mapping[str, object] | None) -> float | None:
pressure_flow = _diagnostics(summary).get("pressureFlow")
value = (
pressure_flow.get("maxScaledResidual")
if isinstance(pressure_flow, Mapping)
else None
)
if isinstance(value, (int, float)) and math.isfinite(float(value)):
return float(value)
return None
def _integration_totals(summary: Mapping[str, object] | None) -> dict[str, float]:
integration = _diagnostics(summary).get("integration")
totals = integration.get("totals") if isinstance(integration, Mapping) else None
if not isinstance(totals, Mapping):
return {}
return {
str(key): float(value)
for key, value in totals.items()
if isinstance(value, (int, float)) and math.isfinite(float(value))
}
def _finite_event_times(value: object) -> list[float] | None:
if not isinstance(value, list):
return None
if any(
not isinstance(item, (int, float)) or not math.isfinite(float(item))
for item in value
):
return None
return [float(item) for item in value]
def _case_observation(
case: Mapping[str, object],
*,
stop_time: float,
expected_projection_count: int | None,
tolerance: MatrixTolerance,
expected_signal_event_times: Sequence[float] | None,
) -> tuple[dict[str, object], tuple[str, ...]]:
summary = _summary(case)
checkpoints = _checkpoints(summary)
checkpoint_details: list[dict[str, object]] = []
issues: list[str] = []
for checkpoint in checkpoints:
values = checkpoint.get("stateValues")
state_values = values if isinstance(values, Mapping) else {}
nonfinite = [
str(key)
for key, value in state_values.items()
if not isinstance(value, (int, float)) or not math.isfinite(float(value))
]
detail = {
"requestedTime": checkpoint.get("requestedTime"),
"actualTime": checkpoint.get("actualTime"),
"available": bool(checkpoint.get("available")),
"projectionKeyCount": len(state_values),
"nonfiniteKeyCount": len(nonfinite),
}
checkpoint_details.append(detail)
if not detail["available"]:
issues.append("checkpointUnavailable")
requested_time = checkpoint.get("requestedTime")
actual_time = checkpoint.get("actualTime")
if (
not isinstance(requested_time, (int, float))
or not isinstance(actual_time, (int, float))
or not math.isclose(
float(requested_time),
float(actual_time),
rel_tol=0.0,
abs_tol=tolerance.checkpoint_time_absolute_seconds,
)
):
issues.append("checkpointTimeMismatch")
if expected_projection_count is not None and len(state_values) != int(
expected_projection_count
):
issues.append("projectionCountMismatch")
if nonfinite:
issues.append("nonfiniteProjection")
if not checkpoints:
issues.append("checkpointsUnavailable")
residual = _scaled_residual(summary)
if tolerance.maximum_scaled_residual is not None and (
residual is None or residual > tolerance.maximum_scaled_residual
):
issues.append("scaledResidualExceeded")
events = _event_trace(summary)
signal_times = _finite_event_times(events.get("signalEventTimes"))
if expected_signal_event_times is not None:
expected = [float(value) for value in expected_signal_event_times]
if signal_times is None or len(signal_times) != len(expected) or any(
not math.isclose(
actual,
wanted,
rel_tol=0.0,
abs_tol=tolerance.signal_event_time_absolute_seconds,
)
for actual, wanted in zip(signal_times or (), expected)
):
issues.append("signalEventTraceMismatch")
if events.get("mechanicalTransitionTimesAvailable") is False:
issues.append("mechanicalTransitionTimesUnavailable")
if not _completed(case, stop_time):
issues.insert(0, "simulationDidNotComplete")
unique_issues = tuple(dict.fromkeys(issues))
return (
{
"completed": _completed(case, stop_time),
"checkpointCount": len(checkpoints),
"expectedProjectionCount": expected_projection_count,
"checkpoints": checkpoint_details,
"events": dict(events),
"maximumScaledResidual": residual,
"integrationTotals": _integration_totals(summary),
"passed": not unique_issues,
"issues": list(unique_issues),
},
unique_issues,
)
def _pair_checkpoints(
left: Mapping[str, object],
right: Mapping[str, object],
*,
tolerance: MatrixTolerance,
strict_prefix_stop: float | None = None,
) -> dict[str, object]:
left_checkpoints = _checkpoints(_summary(left))
right_checkpoints = _checkpoints(_summary(right))
paired: list[tuple[Mapping[str, object], Mapping[str, object]]] = []
for left_checkpoint in left_checkpoints:
left_time = left_checkpoint.get("requestedTime")
if not isinstance(left_time, (int, float)):
continue
if strict_prefix_stop is not None and not (
float(left_time)
< strict_prefix_stop - tolerance.checkpoint_time_absolute_seconds
):
continue
match = next(
(
right_checkpoint
for right_checkpoint in right_checkpoints
if isinstance(right_checkpoint.get("requestedTime"), (int, float))
and math.isclose(
float(right_checkpoint["requestedTime"]),
float(left_time),
rel_tol=0.0,
abs_tol=tolerance.checkpoint_time_absolute_seconds,
)
),
None,
)
if match is not None:
paired.append((left_checkpoint, match))
comparisons: list[dict[str, object]] = []
mismatch_count = 0
key_set_mismatch_count = 0
nonnumeric_count = 0
maximum_absolute_difference = 0.0
maximum_relative_difference = 0.0
for left_checkpoint, right_checkpoint in paired:
left_values = left_checkpoint.get("stateValues")
right_values = right_checkpoint.get("stateValues")
left_mapping = left_values if isinstance(left_values, Mapping) else {}
right_mapping = right_values if isinstance(right_values, Mapping) else {}
left_keys = set(map(str, left_mapping))
right_keys = set(map(str, right_mapping))
missing_from_left = sorted(right_keys - left_keys)
missing_from_right = sorted(left_keys - right_keys)
key_set_mismatch_count += len(missing_from_left) + len(missing_from_right)
mismatches: list[dict[str, object]] = []
for key in sorted(left_keys & right_keys):
left_value = left_mapping.get(key)
right_value = right_mapping.get(key)
if not isinstance(left_value, (int, float)) or not isinstance(
right_value, (int, float)
) or not math.isfinite(float(left_value)) or not math.isfinite(
float(right_value)
):
nonnumeric_count += 1
mismatches.append(
{"key": key, "left": left_value, "right": right_value}
)
continue
left_numeric = float(left_value)
right_numeric = float(right_value)
absolute = abs(left_numeric - right_numeric)
denominator = max(abs(left_numeric), abs(right_numeric))
relative = absolute / denominator if denominator else 0.0
maximum_absolute_difference = max(maximum_absolute_difference, absolute)
maximum_relative_difference = max(maximum_relative_difference, relative)
if not math.isclose(
left_numeric,
right_numeric,
rel_tol=tolerance.state_relative,
abs_tol=tolerance.state_absolute,
):
mismatches.append(
{
"key": key,
"left": left_numeric,
"right": right_numeric,
"absoluteDifference": absolute,
"relativeDifference": relative,
}
)
mismatch_count += len(mismatches)
comparisons.append(
{
"requestedTime": left_checkpoint.get("requestedTime"),
"leftProjectionKeyCount": len(left_keys),
"rightProjectionKeyCount": len(right_keys),
"missingFromLeft": missing_from_left,
"missingFromRight": missing_from_right,
"valueMismatchCount": len(mismatches),
"mismatches": mismatches,
}
)
passed = bool(paired) and not (
mismatch_count or key_set_mismatch_count or nonnumeric_count
)
return {
"evaluated": bool(paired),
"passed": passed,
"commonCheckpointCount": len(paired),
"commonCheckpointTimes": [
pair[0].get("requestedTime") for pair in paired
],
"strictPrefixStopTime": strict_prefix_stop,
"valueMismatchCount": mismatch_count,
"keySetMismatchCount": key_set_mismatch_count,
"nonnumericValueCount": nonnumeric_count,
"maximumAbsoluteDifference": maximum_absolute_difference,
"maximumRelativeDifference": maximum_relative_difference,
"checkpoints": comparisons,
}
def _compare_time_sequences(
left: object,
right: object,
*,
prefix_stop: float,
absolute_tolerance: float,
strict_prefix: bool,
) -> dict[str, object]:
left_times = _finite_event_times(left)
right_times = _finite_event_times(right)
if left_times is None or right_times is None:
return {
"available": False,
"passed": False,
"left": left,
"right": right,
}
if strict_prefix:
include = lambda value: value < prefix_stop - absolute_tolerance
else:
include = lambda value: value <= prefix_stop + absolute_tolerance
left_prefix = [value for value in left_times if include(value)]
right_prefix = [value for value in right_times if include(value)]
passed = len(left_prefix) == len(right_prefix) and all(
math.isclose(
left_value,
right_value,
rel_tol=0.0,
abs_tol=absolute_tolerance,
)
for left_value, right_value in zip(left_prefix, right_prefix)
)
return {
"available": True,
"passed": passed,
"left": left_prefix,
"right": right_prefix,
}
def _pair_events(
left: Mapping[str, object],
right: Mapping[str, object],
*,
prefix_stop: float,
strict_prefix: bool,
tolerance: MatrixTolerance,
) -> dict[str, object]:
left_trace = _event_trace(_summary(left))
right_trace = _event_trace(_summary(right))
signal = _compare_time_sequences(
left_trace.get("signalEventTimes"),
right_trace.get("signalEventTimes"),
prefix_stop=prefix_stop,
absolute_tolerance=tolerance.signal_event_time_absolute_seconds,
strict_prefix=strict_prefix,
)
mechanical = _compare_time_sequences(
left_trace.get("mechanicalTransitionTimes"),
right_trace.get("mechanicalTransitionTimes"),
prefix_stop=prefix_stop,
absolute_tolerance=tolerance.event_time_absolute_seconds,
strict_prefix=strict_prefix,
)
availability = (
left_trace.get("mechanicalTransitionTimesAvailable") is not False
and right_trace.get("mechanicalTransitionTimesAvailable") is not False
)
return {
"evaluated": bool(left_trace) and bool(right_trace),
"passed": bool(signal["passed"] and mechanical["passed"] and availability),
"prefixStopTime": prefix_stop,
"strictPrefix": strict_prefix,
"signalEventTimes": signal,
"mechanicalTransitionTimes": mechanical,
"mechanicalTransitionTimesAvailable": availability,
"leftStateTransitionCount": left_trace.get("stateTransitionCount"),
"rightStateTransitionCount": right_trace.get("stateTransitionCount"),
}
def _ratio(right: float, left: float) -> float | None:
if left == 0.0:
return 1.0 if right == 0.0 else None
return right / left
def _pair_diagnostics(
left: Mapping[str, object],
right: Mapping[str, object],
*,
maximum_scaled_residual: float | None,
) -> dict[str, object]:
left_residual = _scaled_residual(_summary(left))
right_residual = _scaled_residual(_summary(right))
residuals_pass = maximum_scaled_residual is None or (
left_residual is not None
and right_residual is not None
and left_residual <= maximum_scaled_residual
and right_residual <= maximum_scaled_residual
)
left_totals = _integration_totals(_summary(left))
right_totals = _integration_totals(_summary(right))
integration: dict[str, dict[str, float | None]] = {}
for key in sorted(set(left_totals) | set(right_totals)):
left_value = left_totals.get(key)
right_value = right_totals.get(key)
integration[key] = {
"left": left_value,
"right": right_value,
"delta": (
right_value - left_value
if left_value is not None and right_value is not None
else None
),
"rightOverLeft": (
_ratio(right_value, left_value)
if left_value is not None and right_value is not None
else None
),
}
return {
"scaledResidual": {
"left": left_residual,
"right": right_residual,
"limit": maximum_scaled_residual,
"passed": residuals_pass,
},
"integrationTotals": integration,
}
def compare_matrix_cases(
left: Mapping[str, object],
right: Mapping[str, object],
*,
tolerance: MatrixTolerance,
) -> dict[str, object]:
"""Compare two completed matrix cells on their common time prefix."""
left_stop = float(left["stopTime"])
right_stop = float(right["stopTime"])
prefix_stop = min(left_stop, right_stop)
strict_prefix = not math.isclose(
left_stop,
right_stop,
rel_tol=0.0,
abs_tol=tolerance.checkpoint_time_absolute_seconds,
)
if not _completed(left, left_stop) or not _completed(right, right_stop):
return {
"evaluated": False,
"passed": False,
"reason": "oneOrBothCasesDidNotComplete",
}
state = _pair_checkpoints(
left,
right,
tolerance=tolerance,
strict_prefix_stop=prefix_stop if strict_prefix else None,
)
events = _pair_events(
left,
right,
prefix_stop=prefix_stop,
strict_prefix=strict_prefix,
tolerance=tolerance,
)
diagnostics = _pair_diagnostics(
left,
right,
maximum_scaled_residual=tolerance.maximum_scaled_residual,
)
passed = bool(
state["passed"]
and events["passed"]
and diagnostics["scaledResidual"]["passed"]
)
return {
"evaluated": True,
"passed": passed,
"commonPrefixStopTime": prefix_stop,
"strictPrefix": strict_prefix,
"stateProjection": state,
"events": events,
"diagnostics": diagnostics,
}
def _cell_identity(cell: Mapping[str, object]) -> dict[str, object]:
return {
"matrixCaseId": cell.get("matrixCaseId"),
"horizonCaseId": cell.get("horizonCaseId"),
"stopTime": cell.get("stopTime"),
"maxStep": cell.get("maxStep"),
}
def _comparison_record(
kind: str,
left: Mapping[str, object],
right: Mapping[str, object],
tolerance: MatrixTolerance,
) -> dict[str, object]:
return {
"kind": kind,
"left": _cell_identity(left),
"right": _cell_identity(right),
**compare_matrix_cases(left, right, tolerance=tolerance),
}
def _build_comparisons(
cases: Sequence[Mapping[str, object]],
*,
horizon_case_ids: Sequence[str],
max_steps: Sequence[float],
tolerance: MatrixTolerance,
) -> dict[str, object]:
by_identity = {
(str(case.get("horizonCaseId")), float(case.get("maxStep"))): case
for case in cases
if case.get("outcome") != "deferred"
and isinstance(case.get("maxStep"), (int, float))
}
same_horizon: list[dict[str, object]] = []
for horizon_case_id in horizon_case_ids:
for left_step, right_step in combinations(max_steps, 2):
left = by_identity.get((horizon_case_id, float(left_step)))
right = by_identity.get((horizon_case_id, float(right_step)))
if left is not None and right is not None:
same_horizon.append(
_comparison_record(
"sameHorizonAcrossMaxSteps", left, right, tolerance
)
)
same_max_step: list[dict[str, object]] = []
for max_step in max_steps:
for left_horizon, right_horizon in combinations(horizon_case_ids, 2):
left = by_identity.get((left_horizon, float(max_step)))
right = by_identity.get((right_horizon, float(max_step)))
if left is not None and right is not None:
same_max_step.append(
_comparison_record(
"sameMaxStepAcrossHorizons", left, right, tolerance
)
)
evaluated = [*same_horizon, *same_max_step]
comparisons_passed = (
len(by_identity) == 1 and not evaluated
) or (
bool(evaluated)
and all(
bool(item.get("evaluated")) and bool(item.get("passed"))
for item in evaluated
)
)
return {
"sameHorizonAcrossMaxSteps": same_horizon,
"sameMaxStepAcrossHorizons": same_max_step,
"evaluatedCount": sum(bool(item.get("evaluated")) for item in evaluated),
"failedCount": sum(
bool(item.get("evaluated")) and not bool(item.get("passed"))
for item in evaluated
),
"passed": comparisons_passed,
}
def _max_step_slug(value: float) -> str:
return format(value, ".12g").replace("-", "m").replace(".", "p")
def run_max_step_matrix(
manifest_path: Path | str = DEFAULT_MANIFEST_PATH,
*,
lane: str = "production",
horizon_case_ids: Sequence[str | float] = DEFAULT_HORIZON_CASE_IDS,
max_steps: Sequence[float],
additional_checkpoint_times: Sequence[float] = (),
soft_timeout_seconds: float | None = None,
hard_timeout_seconds: float | None = None,
expected_projection_count: int | None = DEFAULT_EXPECTED_PROJECTION_COUNT,
stop_after_failed_tier: bool = True,
case_executor: MatrixCaseExecutor = execute_regression_case,
) -> dict[str, object]:
"""Run a staged, sequential max-step matrix without mutating its source XML."""
manifest = load_regression_manifest(manifest_path)
lanes = manifest.get("lanes")
if not isinstance(lanes, Mapping) or lane not in lanes:
raise RegressionManifestError(f"Unknown regression lane {lane!r}.")
lane_config = lanes[lane]
if not isinstance(lane_config, Mapping):
raise RegressionManifestError(f"Lane {lane!r} must be an object.")
selected_horizons = _resolve_horizons(
manifest,
horizon_case_ids,
additional_checkpoint_times=additional_checkpoint_times,
)
selected_max_steps = _normalise_max_steps(max_steps)
if expected_projection_count is not None and expected_projection_count <= 0:
raise RegressionManifestError("expectedProjectionCount must be positive.")
soft_override = (
_finite_positive(soft_timeout_seconds, field="softTimeoutSeconds")
if soft_timeout_seconds is not None
else None
)
hard_override = (
_finite_positive(hard_timeout_seconds, field="hardTimeoutSeconds")
if hard_timeout_seconds is not None
else None
)
if soft_override is not None and hard_override is not None and (
hard_override <= soft_override
):
raise RegressionManifestError(
"hardTimeoutSeconds must exceed softTimeoutSeconds."
)
source_path = Path(str(manifest["_sourcePath"]))
source_payload = source_path.read_bytes()
source_config = source_simulation_config(source_payload)
sampling_mode = lane_config.get("samplingMode", "source")
sample_step = (
float(source_config["sampleStep"])
if sampling_mode == "source"
else _finite_positive(lane_config.get("sampleStep"), field="lane.sampleStep")
)
source_start = float(source_config["tStart"])
checkpoint_grid_tolerance = max(
1.0e-12,
8.0 * math.ulp(max(1.0, abs(source_start))),
8.0 * math.ulp(max(1.0, abs(sample_step))),
)
for horizon in selected_horizons:
for checkpoint_time in horizon.checkpoint_times:
if math.isclose(
checkpoint_time,
horizon.stop_time,
rel_tol=0.0,
abs_tol=checkpoint_grid_tolerance,
):
continue
grid_index = round((checkpoint_time - source_start) / sample_step)
grid_time = source_start + grid_index * sample_step
if not math.isclose(
checkpoint_time,
grid_time,
rel_tol=0.0,
abs_tol=checkpoint_grid_tolerance,
):
raise RegressionManifestError(
"Matrix checkpoint "
f"{checkpoint_time:.17g} is not represented by the "
f"{sample_step:.17g} s output grid for "
f"{horizon.case_id!r}; use the event trace for off-grid "
"transition times."
)
instrumentation_mode = str(
lane_config.get("instrumentationMode", "standard")
)
execution = manifest.get("execution")
if not isinstance(execution, Mapping):
raise RegressionManifestError("Manifest execution must be an object.")
raw_environment = execution.get("environment", {})
if not isinstance(raw_environment, Mapping) or not all(
isinstance(key, str) and isinstance(value, str)
for key, value in raw_environment.items()
):
raise RegressionManifestError("execution.environment must map strings to strings.")
environment_overrides = tuple(sorted(raw_environment.items()))
termination_grace = _finite_positive(
execution.get("terminationGraceSeconds", 5.0),
field="execution.terminationGraceSeconds",
)
expected_sha256 = str(manifest["source"]["sha256"]) # type: ignore[index]
tolerance = _matrix_tolerance(manifest)
cells: list[dict[str, object]] = []
tier_decisions: list[dict[str, object]] = []
later_tiers_enabled = True
for horizon in selected_horizons:
horizon_case_id = horizon.case_id
stop_time = horizon.stop_time
soft_timeout = soft_override or horizon.soft_timeout_seconds
hard_timeout = hard_override or horizon.hard_timeout_seconds
if hard_timeout <= soft_timeout:
raise RegressionManifestError(
f"Hard timeout must exceed soft timeout for {horizon_case_id!r}."
)
tier_cells: list[dict[str, object]] = []
if not later_tiers_enabled:
for max_step in selected_max_steps:
cell = {
"matrixCaseId": (
f"{horizon_case_id}__max_step_{_max_step_slug(max_step)}"
),
"horizonCaseId": horizon_case_id,
"sourceVariantId": horizon.source_variant_id,
"stopTime": stop_time,
"sampleStep": sample_step,
"maxStep": max_step,
"lane": lane,
"softTimeoutSeconds": soft_timeout,
"hardTimeoutSeconds": hard_timeout,
"outcome": "deferred",
"reason": "previousTierDidNotPass",
"matrixAcceptance": {
"evaluated": False,
"passed": False,
"issues": ["previousTierDidNotPass"],
},
}
cells.append(cell)
tier_cells.append(cell)
tier_decisions.append(
{
"horizonCaseId": horizon_case_id,
"sourceVariantId": horizon.source_variant_id,
"stopTime": stop_time,
"executed": False,
"passed": False,
"reason": "previousTierDidNotPass",
}
)
continue
for max_step in selected_max_steps:
matrix_case_id = (
f"{horizon_case_id}__max_step_{_max_step_slug(max_step)}"
)
request = RegressionCaseRequest(
case_id=matrix_case_id,
source_path=source_path,
expected_sha256=expected_sha256,
lane=lane,
stop_time=stop_time,
sample_step=sample_step,
max_step=max_step,
checkpoint_times=tuple(
horizon.checkpoint_times
),
soft_timeout_seconds=soft_timeout,
hard_timeout_seconds=hard_timeout,
termination_grace_seconds=termination_grace,
instrumentation_mode=instrumentation_mode,
environment_overrides=environment_overrides,
)
result = case_executor(request)
base_cell = {
"matrixCaseId": matrix_case_id,
"horizonCaseId": horizon_case_id,
"sourceVariantId": horizon.source_variant_id,
"stopTime": stop_time,
"sampleStep": sample_step,
"maxStep": max_step,
"lane": lane,
"softTimeoutSeconds": soft_timeout,
"hardTimeoutSeconds": hard_timeout,
**result,
}
observation, issues = _case_observation(
base_cell,
stop_time=stop_time,
expected_projection_count=expected_projection_count,
tolerance=tolerance,
expected_signal_event_times=horizon.expected_signal_event_times,
)
cell = {
**base_cell,
"matrixObservation": observation,
"matrixAcceptance": {
"evaluated": True,
"passed": not issues,
"issues": list(issues),
},
}
cells.append(cell)
tier_cells.append(cell)
tier_passed = all(
bool(cell.get("matrixAcceptance", {}).get("passed"))
for cell in tier_cells
if isinstance(cell.get("matrixAcceptance"), Mapping)
)
tier_decisions.append(
{
"horizonCaseId": horizon_case_id,
"stopTime": stop_time,
"executed": True,
"passed": tier_passed,
"cellCount": len(tier_cells),
}
)
if stop_after_failed_tier and not tier_passed:
later_tiers_enabled = False
comparisons = _build_comparisons(
cells,
horizon_case_ids=[horizon.case_id for horizon in selected_horizons],
max_steps=selected_max_steps,
tolerance=tolerance,
)
executed_acceptance = [
cell.get("matrixAcceptance")
for cell in cells
if cell.get("outcome") != "deferred"
and isinstance(cell.get("matrixAcceptance"), Mapping)
]
deferred_count = sum(cell.get("outcome") == "deferred" for cell in cells)
public_manifest = {
key: value for key, value in manifest.items() if not key.startswith("_")
}
overall_passed = (
bool(executed_acceptance)
and not deferred_count
and all(bool(item.get("passed")) for item in executed_acceptance)
and bool(comparisons["passed"])
)
return {
"schemaVersion": MATRIX_REPORT_SCHEMA_VERSION,
"reportKind": "maxStepHorizonMatrix",
"generatedAt": datetime.now(UTC).isoformat(),
"manifestId": manifest.get("id"),
"manifestPath": str(manifest["_manifestPath"]),
"lane": lane,
"source": {
**dict(manifest["source"]), # type: ignore[arg-type]
"resolvedPath": str(source_path),
"bytes": len(source_payload),
"simulation": source_config,
"mutationPolicy": "readOnly; per-cell overrides are child-memory only",
},
"configuration": {
"horizons": [
{
"horizonCaseId": horizon.case_id,
"sourceVariantId": horizon.source_variant_id,
"stopTime": horizon.stop_time,
"checkpointTimes": list(horizon.checkpoint_times),
"defaultSoftTimeoutSeconds": horizon.soft_timeout_seconds,
"defaultHardTimeoutSeconds": horizon.hard_timeout_seconds,
}
for horizon in selected_horizons
],
"additionalCheckpointTimes": [
float(value) for value in additional_checkpoint_times
],
"maxSteps": list(selected_max_steps),
"sampleStep": sample_step,
"samplingMode": sampling_mode,
"softTimeoutOverrideSeconds": soft_override,
"hardTimeoutOverrideSeconds": hard_override,
"expectedProjectionCount": expected_projection_count,
"stopAfterFailedTier": stop_after_failed_tier,
"executionOrder": "horizon-major, max-step serial",
"tolerance": {
"stateRelative": tolerance.state_relative,
"stateAbsolute": tolerance.state_absolute,
"checkpointTimeAbsoluteSeconds": (
tolerance.checkpoint_time_absolute_seconds
),
"eventTimeAbsoluteSeconds": tolerance.event_time_absolute_seconds,
"signalEventTimeAbsoluteSeconds": (
tolerance.signal_event_time_absolute_seconds
),
"maximumScaledResidual": tolerance.maximum_scaled_residual,
},
},
"manifest": public_manifest,
"tierDecisions": tier_decisions,
"cases": cells,
"comparisons": comparisons,
"acceptance": {
"passed": overall_passed,
"executedCellCount": len(executed_acceptance),
"deferredCellCount": deferred_count,
"caseFailureCount": sum(
not bool(item.get("passed")) for item in executed_acceptance
),
"comparisonFailureCount": comparisons["failedCount"],
},
}
def _parse_arguments(argv: Sequence[str] | None = None) -> argparse.Namespace:
parser = argparse.ArgumentParser(
description=(
"Run a bounded 1s/5s/10s matrix over multiple maximum integration steps."
)
)
parser.add_argument("--manifest", type=Path, default=DEFAULT_MANIFEST_PATH)
parser.add_argument("--lane", default="production")
parser.add_argument(
"--horizon",
action="append",
default=[],
help="Manifest case id or stop time; repeat in sequence order (default: 1s,5s,10s).",
)
parser.add_argument(
"--checkpoint",
type=float,
action="append",
default=[],
help=(
"Additional checkpoint for every horizon that reaches it; repeat for event-neighbour probes."
),
)
parser.add_argument(
"--max-step",
type=float,
action="append",
required=True,
help="Maximum integration step; repeat to form matrix columns.",
)
parser.add_argument("--soft-timeout", type=float)
parser.add_argument("--hard-timeout", type=float)
parser.add_argument(
"--expected-projection-count",
type=int,
default=DEFAULT_EXPECTED_PROJECTION_COUNT,
)
parser.add_argument(
"--continue-after-failure",
action="store_true",
help="Run later horizons even when a preceding horizon tier fails.",
)
parser.add_argument("--output", type=Path)
return parser.parse_args(argv)
def main(argv: Sequence[str] | None = None) -> int:
arguments = _parse_arguments(argv)
report = run_max_step_matrix(
arguments.manifest,
lane=arguments.lane,
horizon_case_ids=arguments.horizon or DEFAULT_HORIZON_CASE_IDS,
max_steps=arguments.max_step,
additional_checkpoint_times=arguments.checkpoint,
soft_timeout_seconds=arguments.soft_timeout,
hard_timeout_seconds=arguments.hard_timeout,
expected_projection_count=arguments.expected_projection_count,
stop_after_failed_tier=not arguments.continue_after_failure,
)
serialized = json.dumps(report, ensure_ascii=False, indent=2, default=str) + "\n"
if arguments.output is None:
print(serialized, end="")
else:
arguments.output.parent.mkdir(parents=True, exist_ok=True)
arguments.output.write_text(serialized, encoding="utf-8")
print(f"Max-step matrix report written to {arguments.output.resolve()}")
if report["acceptance"]["passed"]:
return 0
if report["acceptance"]["deferredCellCount"] and not report["acceptance"][
"caseFailureCount"
]:
return 2
return 1
if __name__ == "__main__":
raise SystemExit(main())