"""Manifest-driven, progressively bounded System XML regression runner. The parent process launches one child process per requested simulation horizon. It first asks the child to cancel cooperatively at the soft deadline and then terminates it at the hard deadline. This keeps an unexpectedly expensive long-run probe from blocking later optimization work indefinitely. The source XML is immutable and authoritative. Stop time, sampling, and an explicitly configured lane max step are changed only in the child process's in-memory XML copy. """ from __future__ import annotations import argparse from collections.abc import Callable, Mapping, Sequence from dataclasses import dataclass from datetime import UTC, datetime import hashlib import importlib.metadata import json import math import os from pathlib import Path import platform import queue import subprocess import sys import threading from time import monotonic, perf_counter, process_time from typing import Any import xml.etree.ElementTree as ET try: # resource is unavailable on native Windows Python. import resource except ImportError: # pragma: no cover - Windows regression job resource = None # type: ignore[assignment] REPORT_SCHEMA_VERSION = 2 MANIFEST_SCHEMA_VERSION = 1 REGRESSION_GOLDEN_SCHEMA_VERSION = 1 DEFAULT_MANIFEST_PATH = ( Path(__file__).resolve().parents[2] / "tests" / "baselines" / "simulation" / "test_mql_8" / "manifest.json" ) class RegressionManifestError(ValueError): """Raised when a regression manifest is incomplete or inconsistent.""" @dataclass(frozen=True) class RegressionCaseRequest: case_id: str source_path: Path expected_sha256: str lane: str stop_time: float sample_step: float max_step: float checkpoint_times: tuple[float, ...] soft_timeout_seconds: float hard_timeout_seconds: float termination_grace_seconds: float instrumentation_mode: str = "standard" environment_overrides: tuple[tuple[str, str], ...] = () CaseExecutor = Callable[[RegressionCaseRequest], dict[str, object]] def _sha256(payload: bytes) -> str: return hashlib.sha256(payload).hexdigest() def _canonical_json_sha256(value: object) -> str: payload = json.dumps( value, default=str, ensure_ascii=False, sort_keys=True, separators=(",", ":"), ).encode("utf-8") return _sha256(payload) def _valid_sha256(value: object) -> bool: return ( isinstance(value, str) and len(value) == 64 and all(character in "0123456789abcdef" for character in value) ) def _repository_root(manifest_path: Path) -> Path: for candidate in (manifest_path.parent, *manifest_path.parents): if (candidate / "app").is_dir() and (candidate / "tests").is_dir(): return candidate raise RegressionManifestError( f"Could not locate the repository root above {manifest_path}." ) def _finite_positive(value: object, *, field: str) -> float: try: numeric = float(value) except (TypeError, ValueError) as exc: raise RegressionManifestError(f"{field} must be numeric.") from exc if not math.isfinite(numeric) or numeric <= 0.0: raise RegressionManifestError(f"{field} must be finite and positive.") return numeric def _finite_nonnegative(value: object, *, field: str) -> float: try: numeric = float(value) except (TypeError, ValueError) as exc: raise RegressionManifestError(f"{field} must be numeric.") from exc if not math.isfinite(numeric) or numeric < 0.0: raise RegressionManifestError( f"{field} must be finite and non-negative." ) return numeric def load_regression_golden( path: Path | str, *, expected_sha256: str | None = None, repository_root: Path | None = None, ) -> dict[str, object]: """Load one reviewed physical/output contract without trusting its values.""" golden_path = Path(path).resolve() try: payload = golden_path.read_bytes() golden = json.loads(payload) except (OSError, json.JSONDecodeError) as exc: raise RegressionManifestError( f"Could not read regression golden {golden_path}: {exc}" ) from exc actual_sha256 = _sha256(payload) if expected_sha256 is not None and actual_sha256 != expected_sha256: raise RegressionManifestError( "Regression golden hash mismatch: " f"expected {expected_sha256}, received {actual_sha256}." ) if not isinstance(golden, dict): raise RegressionManifestError("Regression golden must be a JSON object.") if golden.get("schemaVersion") != REGRESSION_GOLDEN_SCHEMA_VERSION: raise RegressionManifestError( "Unsupported regression golden schemaVersion: " f"{golden.get('schemaVersion')!r}." ) for field in ("id", "caseId", "lane", "sourceXmlSha256"): if not isinstance(golden.get(field), str) or not golden[field]: raise RegressionManifestError( f"Regression golden {field} must be a non-empty string." ) if not _valid_sha256(golden["sourceXmlSha256"]): raise RegressionManifestError( "Regression golden sourceXmlSha256 must be a lowercase SHA-256 digest." ) approval = golden.get("approval") if not isinstance(approval, Mapping) or approval.get("status") != "approved": raise RegressionManifestError( "Regression golden must have approval.status='approved'." ) provenance = golden.get("provenance") source_report = ( provenance.get("sourceReport") if isinstance(provenance, Mapping) else None ) if not isinstance(source_report, Mapping): raise RegressionManifestError( "Regression golden provenance.sourceReport must be an object." ) report_path_value = source_report.get("path") report_sha256 = source_report.get("sha256") report_generated_at = source_report.get("generatedAt") if not isinstance(report_path_value, str) or not report_path_value: raise RegressionManifestError( "Regression golden source report path must be non-empty." ) if not _valid_sha256(report_sha256): raise RegressionManifestError( "Regression golden source report sha256 must be a lowercase digest." ) if not isinstance(report_generated_at, str) or not report_generated_at: raise RegressionManifestError( "Regression golden source report generatedAt must be non-empty." ) compatibility = source_report.get("metadataCompatibility") if not isinstance(compatibility, Mapping) or compatibility.get("status") not in { "current", "acceptedHistorical", }: raise RegressionManifestError( "Regression golden source report must declare metadata compatibility." ) differences = compatibility.get("differences") if not isinstance(differences, list) or not all( isinstance(value, str) and value for value in differences ): raise RegressionManifestError( "Regression golden metadata compatibility differences must be strings." ) if compatibility.get("status") == "current" and differences: raise RegressionManifestError( "A current source report must not declare metadata differences." ) layout = golden.get("physicalLayout") if not isinstance(layout, Mapping): raise RegressionManifestError( "Regression golden physicalLayout must be an object." ) categories = layout.get("projectionCategories") state_keys = layout.get("stateKeys") if not isinstance(categories, list) or "state" not in categories or not all( isinstance(value, str) and value for value in categories ): raise RegressionManifestError( "Regression golden projectionCategories must contain 'state'." ) if not isinstance(state_keys, list) or not state_keys or not all( isinstance(value, str) and value for value in state_keys ): raise RegressionManifestError( "Regression golden stateKeys must be a non-empty string list." ) if len(set(state_keys)) != len(state_keys): raise RegressionManifestError("Regression golden stateKeys must be unique.") if layout.get("stateKeyLayoutSha256") != _canonical_json_sha256(state_keys): raise RegressionManifestError( "Regression golden stateKeyLayoutSha256 does not match stateKeys." ) tolerance = golden.get("tolerance") if not isinstance(tolerance, Mapping): raise RegressionManifestError( "Regression golden tolerance must be an object." ) _finite_nonnegative( tolerance.get("relative"), field="golden.tolerance.relative" ) _finite_nonnegative( tolerance.get("absolute"), field="golden.tolerance.absolute" ) _finite_nonnegative( tolerance.get("checkpointTimeAbsoluteSeconds"), field="golden.tolerance.checkpointTimeAbsoluteSeconds", ) checkpoints = golden.get("physicalCheckpoints") if not isinstance(checkpoints, list) or not checkpoints: raise RegressionManifestError( "Regression golden physicalCheckpoints must be a non-empty list." ) for index, checkpoint in enumerate(checkpoints): if not isinstance(checkpoint, Mapping): raise RegressionManifestError( f"Regression golden checkpoint {index} must be an object." ) _finite_nonnegative( checkpoint.get("requestedTime"), field=f"golden.physicalCheckpoints[{index}].requestedTime", ) values = checkpoint.get("values") if not isinstance(values, list) or len(values) != len(state_keys): raise RegressionManifestError( f"Regression golden checkpoint {index} has the wrong value layout." ) if any( not isinstance(value, (int, float)) or not math.isfinite(float(value)) for value in values ): raise RegressionManifestError( f"Regression golden checkpoint {index} values must be finite numbers." ) output_contract = golden.get("outputContract") if output_contract is not None: if not isinstance(output_contract, Mapping) or not _valid_sha256( output_contract.get("sha256") ): raise RegressionManifestError( "Regression golden outputContract.sha256 must be a lowercase digest." ) if repository_root is not None: root = repository_root.resolve() report_path = (root / report_path_value).resolve() if not report_path.is_relative_to(root): raise RegressionManifestError( "Regression golden source report must remain inside the repository." ) try: report_payload = report_path.read_bytes() except OSError as exc: raise RegressionManifestError( f"Could not read golden source report {report_path}: {exc}" ) from exc if _sha256(report_payload) != report_sha256: raise RegressionManifestError( "Regression golden source report hash no longer matches provenance." ) try: source_report_document = json.loads(report_payload) except json.JSONDecodeError as exc: raise RegressionManifestError( "Regression golden source report is not valid JSON." ) from exc report_source = ( source_report_document.get("source") if isinstance(source_report_document, Mapping) else None ) if ( not isinstance(report_source, Mapping) or report_source.get("sha256") != golden["sourceXmlSha256"] or source_report_document.get("generatedAt") != report_generated_at ): raise RegressionManifestError( "Regression golden source report identity does not match the golden." ) report_cases = source_report_document.get("cases") matching_cases = ( [ case for case in report_cases if isinstance(case, Mapping) and case.get("caseId") == golden["caseId"] and case.get("lane") == golden["lane"] ] if isinstance(report_cases, list) else [] ) if len(matching_cases) != 1: raise RegressionManifestError( "Regression golden source report does not contain exactly one matching case." ) worker = matching_cases[0].get("worker") summary = worker.get("summary") if isinstance(worker, Mapping) else None report_checkpoints = ( summary.get("physicalContract", {}).get("checkpoints") if isinstance(summary, Mapping) and isinstance(summary.get("physicalContract"), Mapping) else summary.get("checkpoints") if isinstance(summary, Mapping) else None ) if not isinstance(report_checkpoints, list) or len(report_checkpoints) != len( checkpoints ): raise RegressionManifestError( "Regression golden checkpoints do not match the source report." ) for golden_checkpoint, report_checkpoint in zip( checkpoints, report_checkpoints ): report_values = ( report_checkpoint.get("stateValues") if isinstance(report_checkpoint, Mapping) else None ) extracted_values = ( [report_values.get(key) for key in state_keys] if isinstance(report_values, Mapping) and set(report_values) == set(state_keys) else None ) if ( not isinstance(report_checkpoint, Mapping) or report_checkpoint.get("requestedTime") != golden_checkpoint.get("requestedTime") or extracted_values != golden_checkpoint.get("values") ): raise RegressionManifestError( "Regression golden values are not an exact extraction of its source report." ) if output_contract is not None: report_output_contract = ( summary.get("outputContract") if isinstance(summary, Mapping) else None ) if ( not isinstance(report_output_contract, Mapping) or report_output_contract.get("sha256") != output_contract.get("sha256") ): raise RegressionManifestError( "Regression golden output contract does not match its source report." ) golden["_sourceReportPath"] = str(report_path) golden["_path"] = str(golden_path) golden["_sha256"] = actual_sha256 return golden def load_regression_manifest(path: Path | str) -> dict[str, object]: """Load and validate the stable, machine-independent suite definition.""" manifest_path = Path(path).resolve() try: manifest = json.loads(manifest_path.read_text(encoding="utf-8")) except (OSError, json.JSONDecodeError) as exc: raise RegressionManifestError( f"Could not read regression manifest {manifest_path}: {exc}" ) from exc if not isinstance(manifest, dict): raise RegressionManifestError("Regression manifest must be a JSON object.") if manifest.get("schemaVersion") != MANIFEST_SCHEMA_VERSION: raise RegressionManifestError( f"Unsupported regression manifest schemaVersion: " f"{manifest.get('schemaVersion')!r}." ) source = manifest.get("source") if not isinstance(source, dict): raise RegressionManifestError("Manifest source must be an object.") raw_source_path = source.get("path") expected_sha256 = source.get("sha256") if not isinstance(raw_source_path, str) or not raw_source_path: raise RegressionManifestError("Manifest source.path must be non-empty.") if not _valid_sha256(expected_sha256): raise RegressionManifestError( "Manifest source.sha256 must be a lowercase SHA-256 digest." ) repository_root = _repository_root(manifest_path) source_path = (repository_root / raw_source_path).resolve() try: source_payload = source_path.read_bytes() except OSError as exc: raise RegressionManifestError( f"Could not read authoritative XML {source_path}: {exc}" ) from exc actual_sha256 = _sha256(source_payload) if actual_sha256 != expected_sha256: raise RegressionManifestError( "Authoritative XML hash mismatch: " f"expected {expected_sha256}, received {actual_sha256}." ) expected_source_bytes = source.get("bytes") if not isinstance(expected_source_bytes, int) or expected_source_bytes != len( source_payload ): raise RegressionManifestError( "Authoritative XML byte count does not match manifest source.bytes." ) companion = source.get("companionProject") companion_path: Path | None = None if companion is not None: if not isinstance(companion, dict): raise RegressionManifestError( "Manifest source.companionProject must be an object when declared." ) companion_path_value = companion.get("path") companion_sha256 = companion.get("sha256") companion_bytes = companion.get("bytes") if not isinstance(companion_path_value, str) or not companion_path_value: raise RegressionManifestError( "Manifest companion project path must be non-empty." ) if not _valid_sha256(companion_sha256): raise RegressionManifestError( "Manifest companion project sha256 must be a lowercase digest." ) companion_path = (repository_root / companion_path_value).resolve() try: companion_payload = companion_path.read_bytes() except OSError as exc: raise RegressionManifestError( f"Could not read companion project {companion_path}: {exc}" ) from exc if _sha256(companion_payload) != companion_sha256: raise RegressionManifestError("Companion project hash mismatch.") if not isinstance(companion_bytes, int) or companion_bytes != len( companion_payload ): raise RegressionManifestError("Companion project byte count mismatch.") if companion.get("executionInput") is not False: raise RegressionManifestError( "Companion project must be explicitly marked as non-execution input." ) historical_reports = manifest.get("historicalReports", []) if not isinstance(historical_reports, list): raise RegressionManifestError("historicalReports must be a list.") for index, historical in enumerate(historical_reports): if not isinstance(historical, Mapping): raise RegressionManifestError( f"historicalReports[{index}] must be an object." ) historical_path_value = historical.get("path") historical_sha256 = historical.get("sha256") historical_source_sha256 = historical.get("sourceXmlSha256") if ( historical.get("status") != "historicalOnly" or not isinstance(historical_path_value, str) or not historical_path_value or not _valid_sha256(historical_sha256) or not _valid_sha256(historical_source_sha256) or historical_source_sha256 == expected_sha256 or historical.get("compatibleWithCurrentSource") is not False or not isinstance(historical.get("reason"), str) or not historical.get("reason") ): raise RegressionManifestError( f"historicalReports[{index}] is incomplete." ) historical_path = (repository_root / historical_path_value).resolve() try: historical_payload = historical_path.read_bytes() except OSError as exc: raise RegressionManifestError( f"Could not read historical report {historical_path}: {exc}" ) from exc if _sha256(historical_payload) != historical_sha256: raise RegressionManifestError( f"historicalReports[{index}] hash mismatch." ) sequence = manifest.get("sequence") variants = manifest.get("variants") lanes = manifest.get("lanes") execution = manifest.get("execution") if not isinstance(sequence, list) or not sequence: raise RegressionManifestError("Manifest sequence must be a non-empty list.") if not isinstance(variants, dict): raise RegressionManifestError("Manifest variants must be an object.") if not isinstance(lanes, dict) or not lanes: raise RegressionManifestError("Manifest lanes must be a non-empty object.") if not isinstance(execution, dict): raise RegressionManifestError("Manifest execution must be an object.") previous_stop = -math.inf for raw_case_id in sequence: if not isinstance(raw_case_id, str) or raw_case_id not in variants: raise RegressionManifestError( f"Sequence entry {raw_case_id!r} has no matching variant." ) variant = variants[raw_case_id] if not isinstance(variant, dict): raise RegressionManifestError( f"Variant {raw_case_id!r} must be an object." ) stop_time = _finite_positive( variant.get("stopTime"), field=f"variants.{raw_case_id}.stopTime" ) if stop_time <= previous_stop: raise RegressionManifestError( "Variant stop times must be strictly increasing in sequence order." ) previous_stop = stop_time soft_timeout = _finite_positive( variant.get("softTimeoutSeconds"), field=f"variants.{raw_case_id}.softTimeoutSeconds", ) hard_timeout = _finite_positive( variant.get("hardTimeoutSeconds"), field=f"variants.{raw_case_id}.hardTimeoutSeconds", ) if hard_timeout <= soft_timeout: raise RegressionManifestError( f"Variant {raw_case_id!r} hard timeout must exceed its soft timeout." ) prediction_eligible = variant.get("useForRuntimePrediction", True) if not isinstance(prediction_eligible, bool): raise RegressionManifestError( f"Variant {raw_case_id!r} useForRuntimePrediction must be boolean." ) checkpoint_times = variant.get("checkpointTimes", []) if not isinstance(checkpoint_times, list): raise RegressionManifestError( f"Variant {raw_case_id!r} checkpointTimes must be a list." ) for index, checkpoint in enumerate(checkpoint_times): checkpoint_value = _finite_nonnegative( checkpoint, field=( f"variants.{raw_case_id}.checkpointTimes[{index}]" ), ) if checkpoint_value > stop_time: raise RegressionManifestError( f"Variant {raw_case_id!r} checkpoint exceeds stopTime." ) for lane_name, raw_lane in lanes.items(): if not isinstance(lane_name, str) or not isinstance(raw_lane, dict): raise RegressionManifestError("Every lane must be a named object.") sampling_mode = raw_lane.get("samplingMode") if sampling_mode not in {"source", "fixed"}: raise RegressionManifestError( f"Lane {lane_name!r} samplingMode must be source or fixed." ) if sampling_mode == "fixed": _finite_positive( raw_lane.get("sampleStep"), field=f"lanes.{lane_name}.sampleStep", ) max_step_mode = raw_lane.get("maxStepMode", "source") if max_step_mode not in {"source", "fixed"}: raise RegressionManifestError( f"Lane {lane_name!r} maxStepMode must be source or fixed." ) if max_step_mode == "fixed": _finite_positive( raw_lane.get("maxStep"), field=f"lanes.{lane_name}.maxStep", ) instrumentation = raw_lane.get("instrumentationMode", "standard") if instrumentation not in {"off", "standard", "audit"}: raise RegressionManifestError( f"Lane {lane_name!r} has an unsupported instrumentationMode." ) correctness = manifest.get("correctness") if not isinstance(correctness, dict): raise RegressionManifestError("Manifest correctness must be an object.") state_relative_tolerance = _finite_nonnegative( correctness.get("stateRelativeTolerance"), field="correctness.stateRelativeTolerance", ) state_absolute_tolerance = _finite_nonnegative( correctness.get("stateAbsoluteTolerance", 0.0), field="correctness.stateAbsoluteTolerance", ) checkpoint_time_tolerance = _finite_nonnegative( correctness.get("checkpointTimeAbsoluteToleranceSeconds", 1.0e-12), field="correctness.checkpointTimeAbsoluteToleranceSeconds", ) loaded_goldens: dict[str, dict[str, dict[str, object]]] = {} for case_id in sequence: variant = variants[case_id] assert isinstance(variant, dict) raw_goldens = variant.get("goldens", {}) if not isinstance(raw_goldens, dict): raise RegressionManifestError( f"Variant {case_id!r} goldens must be an object keyed by lane." ) for lane_name, reference in raw_goldens.items(): if lane_name not in lanes or not isinstance(reference, Mapping): raise RegressionManifestError( f"Variant {case_id!r} has an invalid golden lane reference." ) golden_path_value = reference.get("path") golden_sha256 = reference.get("sha256") if ( not isinstance(golden_path_value, str) or not golden_path_value or not _valid_sha256(golden_sha256) ): raise RegressionManifestError( f"Variant {case_id!r} golden reference is incomplete." ) golden_path = (repository_root / golden_path_value).resolve() if not golden_path.is_relative_to(repository_root): raise RegressionManifestError( "Regression golden must remain inside the repository." ) golden = load_regression_golden( golden_path, expected_sha256=str(golden_sha256), repository_root=repository_root, ) if ( golden["caseId"] != case_id or golden["lane"] != lane_name or golden["sourceXmlSha256"] != expected_sha256 ): raise RegressionManifestError( f"Variant {case_id!r} golden identity does not match the manifest." ) tolerance = golden["tolerance"] assert isinstance(tolerance, Mapping) if ( float(tolerance["relative"]) != state_relative_tolerance or float(tolerance["absolute"]) != state_absolute_tolerance or float(tolerance["checkpointTimeAbsoluteSeconds"]) != checkpoint_time_tolerance ): raise RegressionManifestError( f"Variant {case_id!r} golden tolerance differs from the manifest." ) expected_times = [float(value) for value in variant["checkpointTimes"]] golden_times = [ float(checkpoint["requestedTime"]) for checkpoint in golden["physicalCheckpoints"] ] if expected_times != golden_times: raise RegressionManifestError( f"Variant {case_id!r} golden checkpoint times differ from manifest." ) loaded_goldens.setdefault(case_id, {})[lane_name] = golden _finite_positive( execution.get("predictionSafetyFactor", 1.0), field="execution.predictionSafetyFactor", ) _finite_positive( execution.get("terminationGraceSeconds", 5.0), field="execution.terminationGraceSeconds", ) if source.get("simulation") != source_simulation_config(source_payload): raise RegressionManifestError( "Manifest source.simulation does not match the authoritative XML." ) manifest["_manifestPath"] = str(manifest_path) manifest["_repositoryRoot"] = str(repository_root) manifest["_sourcePath"] = str(source_path) manifest["_companionPath"] = ( str(companion_path) if companion_path is not None else None ) manifest["_goldens"] = loaded_goldens return manifest def _simulation_element(root: ET.Element) -> ET.Element: simulations = root.findall("./Simulation") if len(simulations) != 1: raise RegressionManifestError( f"Expected exactly one System/Simulation element, received {len(simulations)}." ) return simulations[0] def source_simulation_config(payload: bytes) -> dict[str, object]: """Read the source settings without importing or invoking the simulator.""" try: root = ET.fromstring(payload) except ET.ParseError as exc: raise RegressionManifestError(f"Authoritative XML is not well formed: {exc}") from exc simulation = _simulation_element(root) try: return { "tStart": float(simulation.attrib["tStart"]), "tStop": float(simulation.attrib["tStop"]), "sampleStep": float(simulation.attrib["sampleStep"]), "maxStep": float(simulation.attrib["maxStep"]), "method": simulation.attrib["method"], } except (KeyError, ValueError) as exc: raise RegressionManifestError( "Authoritative XML has an incomplete Simulation configuration." ) from exc def derive_simulation_xml( source_payload: bytes, *, stop_time: float, sample_step: float, max_step: float | None = None, ) -> bytes: """Return an in-memory derivative with explicit, audited setting overrides.""" stop_value = _finite_positive(stop_time, field="stop_time") sample_value = _finite_positive(sample_step, field="sample_step") max_value = ( _finite_positive(max_step, field="max_step") if max_step is not None else None ) try: root = ET.fromstring(source_payload) except ET.ParseError as exc: raise RegressionManifestError(f"Authoritative XML is not well formed: {exc}") from exc simulation = _simulation_element(root) original_attributes = dict(simulation.attrib) simulation.set("tStop", format(stop_value, ".17g")) simulation.set("sampleStep", format(sample_value, ".17g")) if max_value is not None: simulation.set("maxStep", format(max_value, ".17g")) changed_attributes = { key for key in set(original_attributes) | set(simulation.attrib) if original_attributes.get(key) != simulation.attrib.get(key) } if not changed_attributes.issubset({"tStop", "sampleStep", "maxStep"}): raise AssertionError( "In-memory regression derivative changed unexpected Simulation attributes." ) return ET.tostring(root, encoding="utf-8", xml_declaration=True) def _package_version(distribution: str) -> str | None: try: return importlib.metadata.version(distribution) except importlib.metadata.PackageNotFoundError: return None def _repository_snapshot() -> dict[str, object]: repository_root = Path(__file__).resolve().parents[2] def git_output(*arguments: str) -> str | None: try: completed = subprocess.run( ("git", "-C", str(repository_root), *arguments), check=False, capture_output=True, text=True, timeout=5.0, ) except (OSError, subprocess.SubprocessError): return None if completed.returncode != 0: return None return completed.stdout.strip() status = git_output("status", "--short", "--untracked-files=all") return { "root": str(repository_root), "head": git_output("rev-parse", "HEAD"), "branch": git_output("branch", "--show-current"), "dirty": bool(status) if status is not None else None, "status": status.splitlines() if status else [], } def runtime_snapshot() -> dict[str, object]: return { "python": sys.version, "pythonImplementation": platform.python_implementation(), "executable": sys.executable, "platform": platform.platform(), "machine": platform.machine(), "processor": platform.processor(), "cpuCount": os.cpu_count(), "repository": _repository_snapshot(), "packages": { "numpy": _package_version("numpy"), "scipy": _package_version("scipy"), "lxml": _package_version("lxml"), }, "environment": { "SIMULATIONAPP_PROFILE": os.getenv("SIMULATIONAPP_PROFILE"), "SIMULATION_ODE_JACOBIAN_MODE": os.getenv( "SIMULATION_ODE_JACOBIAN_MODE", "scipy" ), "SIMULATIONAPP_PROPERTY_CACHE": os.getenv( "SIMULATIONAPP_PROPERTY_CACHE", "on" ), "SIMULATION_CAUSAL_EXECUTOR_V2": os.getenv( "SIMULATION_CAUSAL_EXECUTOR_V2", "1" ), "SIMULATION_CAUSAL_COORDINATE_KERNEL": os.getenv( "SIMULATION_CAUSAL_COORDINATE_KERNEL", "1" ), "SIMULATION_CAUSAL_FAST_PATH": os.getenv( "SIMULATION_CAUSAL_FAST_PATH", "1" ), }, } def _peak_rss_bytes() -> tuple[int | None, int | None, str]: if resource is None: return None, None, "unavailable" raw_value = int(resource.getrusage(resource.RUSAGE_SELF).ru_maxrss) if sys.platform == "darwin": return raw_value, raw_value, "bytes" return raw_value * 1024, raw_value, "KiB" def _emit_worker_event(payload: Mapping[str, object]) -> None: print( json.dumps(payload, ensure_ascii=False, separators=(",", ":"), default=str), flush=True, ) def _event_trace(diagnostics: Mapping[str, object]) -> dict[str, object]: signal = diagnostics.get("signal") integration = diagnostics.get("integration") signal_times: list[object] = [] segment_trace: list[dict[str, object]] = [] mechanical_transition_times: list[float] = [] mechanical_transition_times_available = True totals: dict[str, object] = {} if isinstance(signal, Mapping): raw_signal_times = signal.get("eventTimes") if isinstance(raw_signal_times, list): signal_times = list(raw_signal_times) if isinstance(integration, Mapping): raw_totals = integration.get("totals") if isinstance(raw_totals, Mapping): totals = dict(raw_totals) raw_segments = integration.get("segments") if isinstance(raw_segments, list): for segment in raw_segments: if not isinstance(segment, Mapping): continue segment_trace.append( { key: segment.get(key) for key in ( "startTime", "requestedStopTime", "simulatedUntil", "solverStartCount", "stateTransitionCount", "stateTransitionTimes", "recoverableRetryCount", ) } ) raw_transition_times = segment.get("stateTransitionTimes") raw_transition_count = segment.get("stateTransitionCount", 0) transition_count = ( int(raw_transition_count) if isinstance(raw_transition_count, (int, float)) else 0 ) if isinstance(raw_transition_times, list): finite_transition_times = [ float(value) for value in raw_transition_times if isinstance(value, (int, float)) and math.isfinite(float(value)) ] mechanical_transition_times.extend(finite_transition_times) if len(finite_transition_times) != transition_count: mechanical_transition_times_available = False elif transition_count != 0: mechanical_transition_times_available = False total_transition_count = totals.get("stateTransitionCount", 0) if isinstance(total_transition_count, (int, float)) and int( total_transition_count ) != len(mechanical_transition_times): mechanical_transition_times_available = False return { "signalEventTimes": signal_times, "stateTransitionCount": totals.get("stateTransitionCount", 0), "mechanicalTransitionTimes": mechanical_transition_times, "solverStartCount": totals.get("solverStartCount", 0), "segments": segment_trace, "mechanicalTransitionTimesAvailable": ( mechanical_transition_times_available ), } def _series_health(series: object) -> dict[str, object]: if not isinstance(series, Mapping): return { "seriesCount": 0, "scalarCount": 0, "nonfiniteCount": 0, "timeStrictlyIncreasing": False, } scalar_count = 0 nonfinite_count = 0 for values in series.values(): if not isinstance(values, list): continue scalar_count += len(values) for value in values: if not isinstance(value, (int, float)) or not math.isfinite(float(value)): nonfinite_count += 1 times = series.get("time") time_values = times if isinstance(times, list) else [] return { "seriesCount": len(series), "scalarCount": scalar_count, "nonfiniteCount": nonfinite_count, "timeStrictlyIncreasing": all( float(first) < float(second) for first, second in zip(time_values, time_values[1:]) ), "timeStart": time_values[0] if time_values else None, "timeEnd": time_values[-1] if time_values else None, } def _state_checkpoints( result: Mapping[str, object], requested_times: Sequence[float], sample_step: float, ) -> list[dict[str, object]]: series = result.get("series") variables = result.get("variables") if not isinstance(series, Mapping) or not isinstance(variables, list): return [] raw_times = series.get("time") if not isinstance(raw_times, list) or not raw_times: return [] state_keys = [ str(variable["key"]) for variable in variables if isinstance(variable, Mapping) and variable.get("category") == "state" and isinstance(variable.get("key"), str) ] tolerance = max(1.0e-12, 0.51 * float(sample_step)) checkpoints: list[dict[str, object]] = [] for requested in requested_times: index = min( range(len(raw_times)), key=lambda candidate: abs(float(raw_times[candidate]) - requested), ) actual_time = float(raw_times[index]) if abs(actual_time - requested) > tolerance: checkpoints.append( { "requestedTime": requested, "available": False, "nearestTime": actual_time, } ) continue values: dict[str, object] = {} for key in state_keys: variable_series = series.get(key) if isinstance(variable_series, list) and index < len(variable_series): values[key] = variable_series[index] checkpoints.append( { "requestedTime": requested, "actualTime": actual_time, "available": True, "stateValues": values, } ) return checkpoints def _output_contract(result: Mapping[str, object]) -> dict[str, object]: """Hash output metadata and shape without including any physical values.""" raw_variables = result.get("variables") variables = ( [dict(variable) for variable in raw_variables if isinstance(variable, Mapping)] if isinstance(raw_variables, list) else [] ) raw_series = result.get("series") series_shape = ( [ { "key": str(key), "length": len(values) if isinstance(values, list) else None, } for key, values in raw_series.items() ] if isinstance(raw_series, Mapping) else [] ) lengths = [entry["length"] for entry in series_shape] numeric_lengths = [value for value in lengths if isinstance(value, int)] contract_payload = { "variables": variables, "seriesShape": series_shape, } time_entry = next( (entry for entry in series_shape if entry["key"] == "time"), None, ) return { "schemaVersion": 1, "sha256": _canonical_json_sha256(contract_payload), "variableMetadataSha256": _canonical_json_sha256(variables), "seriesShapeSha256": _canonical_json_sha256(series_shape), "variableCount": len(variables), "seriesKeyCount": len(series_shape), "sampleCount": time_entry["length"] if time_entry is not None else None, "seriesLengthsConsistent": ( bool(numeric_lengths) and len(numeric_lengths) == len(lengths) and len(set(numeric_lengths)) == 1 ), "containsPhysicalValues": False, "hashPayload": "variables metadata + ordered series keys/lengths", } def summarize_simulation_result( result: Mapping[str, object], *, checkpoint_times: Sequence[float], sample_step: float, ) -> dict[str, object]: diagnostics = result.get("diagnostics") diagnostic_mapping = diagnostics if isinstance(diagnostics, Mapping) else {} final = result.get("final") checkpoints = _state_checkpoints(result, checkpoint_times, sample_step) event_trace = _event_trace(diagnostic_mapping) physical_contract = { "schemaVersion": 1, "projectionCategories": ["state"], "checkpoints": checkpoints, "eventTrace": event_trace, "comparisonMode": "numericTolerance", } return { "success": bool(result.get("success")), "status": result.get("status"), "partial": bool(result.get("partial")), "message": result.get("message"), "simulatedUntil": result.get("simulatedUntil"), "requestedStopTime": result.get("requestedStopTime"), "variableCount": ( len(result["variables"]) if isinstance(result.get("variables"), list) else 0 ), "seriesHealth": _series_health(result.get("series")), "final": dict(final) if isinstance(final, Mapping) else {}, "physicalContract": physical_contract, "outputContract": _output_contract(result), # Compatibility aliases for schema-v1 report consumers. "checkpoints": checkpoints, "diagnostics": dict(diagnostic_mapping), "eventTrace": event_trace, } def _worker_control_listener(cancel_event: threading.Event) -> None: try: for line in sys.stdin: if line.strip().lower() == "cancel": cancel_event.set() return except (OSError, ValueError): return def run_worker(arguments: argparse.Namespace) -> int: """Execute one real simulation inside the bounded child process.""" os.environ["SIMULATIONAPP_PROFILE"] = arguments.instrumentation_mode source_path = Path(arguments.xml).resolve() source_payload = source_path.read_bytes() actual_sha256 = _sha256(source_payload) if actual_sha256 != arguments.expected_sha256: raise RegressionManifestError( "Worker source hash mismatch: " f"expected {arguments.expected_sha256}, received {actual_sha256}." ) derived_xml = derive_simulation_xml( source_payload, stop_time=arguments.stop_time, sample_step=arguments.sample_step, max_step=arguments.max_step, ) cancel_event = threading.Event() threading.Thread( target=_worker_control_listener, args=(cancel_event,), name="regression-soft-cancel-listener", daemon=True, ).start() latest_simulated_time: float | None = None def progress_callback( progress: int, phase: str, message: str, simulated_time: float | None = None, total_time: float | None = None, ) -> None: nonlocal latest_simulated_time if simulated_time is not None and math.isfinite(float(simulated_time)): latest_simulated_time = max( float(simulated_time), latest_simulated_time if latest_simulated_time is not None else -math.inf, ) _emit_worker_event( { "event": "progress", "progress": progress, "phase": phase, "message": message, "simulatedTime": simulated_time, "totalTime": total_time, "wallSeconds": perf_counter() - wall_started, } ) wall_started = perf_counter() cpu_started = process_time() try: from app.main import run_system_xml_simulation result = run_system_xml_simulation( derived_xml, progress_callback=progress_callback, cancel_check=cancel_event.is_set, ) wall_seconds = perf_counter() - wall_started cpu_seconds = process_time() - cpu_started simulated_until = result.get("simulatedUntil") if isinstance(simulated_until, (int, float)) and math.isfinite( float(simulated_until) ): latest_simulated_time = max( float(simulated_until), latest_simulated_time if latest_simulated_time is not None else -math.inf, ) peak_rss_bytes, peak_rss_raw, peak_rss_raw_unit = _peak_rss_bytes() summary = summarize_simulation_result( result, checkpoint_times=arguments.checkpoint_time, sample_step=arguments.sample_step, ) _emit_worker_event( { "event": "result", "outcome": ( "completed" if bool(result.get("success")) and result.get("status") == "completed" else str(result.get("status", "failed")) ), "sourceSha256": actual_sha256, "sourcePath": str(source_path), "derivedConfiguration": { "stopTime": arguments.stop_time, "sampleStep": arguments.sample_step, "maxStep": arguments.max_step, "lane": arguments.lane, "sourceXmlUnmodified": True, }, "wallSeconds": wall_seconds, "cpuSeconds": cpu_seconds, "peakRssBytes": peak_rss_bytes, "peakRssRaw": peak_rss_raw, "peakRssRawUnit": peak_rss_raw_unit, "lastSimulatedTime": latest_simulated_time, "runtime": runtime_snapshot(), "summary": summary, } ) return 0 except BaseException as exc: peak_rss_bytes, peak_rss_raw, peak_rss_raw_unit = _peak_rss_bytes() _emit_worker_event( { "event": "result", "outcome": "error", "errorType": type(exc).__name__, "message": str(exc), "wallSeconds": perf_counter() - wall_started, "cpuSeconds": process_time() - cpu_started, "peakRssBytes": peak_rss_bytes, "peakRssRaw": peak_rss_raw, "peakRssRawUnit": peak_rss_raw_unit, "lastSimulatedTime": latest_simulated_time, "runtime": runtime_snapshot(), } ) return 1 def run_bounded_child_process( command: Sequence[str], *, soft_timeout_seconds: float, hard_timeout_seconds: float, termination_grace_seconds: float, environment: Mapping[str, str] | None = None, ) -> dict[str, object]: """Run a JSON-lines worker with cooperative and forced timeout layers.""" soft_timeout = _finite_positive( soft_timeout_seconds, field="soft_timeout_seconds" ) hard_timeout = _finite_positive( hard_timeout_seconds, field="hard_timeout_seconds" ) grace = _finite_positive( termination_grace_seconds, field="termination_grace_seconds" ) if hard_timeout <= soft_timeout: raise ValueError("hard_timeout_seconds must exceed soft_timeout_seconds.") process = subprocess.Popen( list(command), stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, bufsize=1, env=dict(environment) if environment is not None else None, ) output_queue: queue.Queue[tuple[str, str | None]] = queue.Queue() def read_stream(name: str, stream: Any) -> None: try: for line in stream: output_queue.put((name, line.rstrip("\n"))) finally: output_queue.put((name, None)) stdout_thread = threading.Thread( target=read_stream, args=("stdout", process.stdout), daemon=True, ) stderr_thread = threading.Thread( target=read_stream, args=("stderr", process.stderr), daemon=True, ) stdout_thread.start() stderr_thread.start() started_at = monotonic() soft_cancel_sent = False hard_timeout_reached = False worker_result: dict[str, object] | None = None progress_events: list[dict[str, object]] = [] stderr_lines: list[str] = [] def consume_output() -> None: nonlocal worker_result while True: try: stream_name, line = output_queue.get_nowait() except queue.Empty: return if line is None: continue if stream_name == "stderr": stderr_lines.append(line) continue try: event = json.loads(line) except json.JSONDecodeError: stderr_lines.append(f"[non-json stdout] {line}") continue if not isinstance(event, dict): continue if event.get("event") == "progress": progress_events.append(event) elif event.get("event") == "result": worker_result = event while process.poll() is None: consume_output() elapsed = monotonic() - started_at if not soft_cancel_sent and elapsed >= soft_timeout: soft_cancel_sent = True try: assert process.stdin is not None process.stdin.write("cancel\n") process.stdin.flush() except (BrokenPipeError, OSError, ValueError): pass if elapsed >= hard_timeout: hard_timeout_reached = True process.terminate() try: process.wait(timeout=grace) except subprocess.TimeoutExpired: process.kill() process.wait(timeout=grace) break threading.Event().wait(0.02) try: process.wait(timeout=grace) except subprocess.TimeoutExpired: hard_timeout_reached = True process.kill() process.wait(timeout=grace) consume_output() if process.stdin is not None: try: process.stdin.close() except (BrokenPipeError, OSError, ValueError): pass stdout_thread.join(timeout=grace) stderr_thread.join(timeout=grace) consume_output() for stream in (process.stdout, process.stderr): if stream is not None: stream.close() last_simulated_time: float | None = None for progress in progress_events: value = progress.get("simulatedTime") if isinstance(value, (int, float)) and math.isfinite(float(value)): last_simulated_time = max( float(value), last_simulated_time if last_simulated_time is not None else -math.inf, ) if worker_result is not None: value = worker_result.get("lastSimulatedTime") if isinstance(value, (int, float)) and math.isfinite(float(value)): last_simulated_time = max( float(value), last_simulated_time if last_simulated_time is not None else -math.inf, ) if hard_timeout_reached: outcome = "hard_timeout" elif soft_cancel_sent: outcome = "soft_timeout" elif worker_result is None: outcome = "worker_error" else: outcome = str(worker_result.get("outcome", "worker_error")) return { "outcome": outcome, "softCancelSent": soft_cancel_sent, "hardTimeoutReached": hard_timeout_reached, "orchestrationWallSeconds": monotonic() - started_at, "returnCode": process.returncode, "lastSimulatedTime": last_simulated_time, "progressEventCount": len(progress_events), "lastProgressEvent": progress_events[-1] if progress_events else None, "stderrTail": stderr_lines[-40:], "worker": worker_result, } def execute_regression_case(request: RegressionCaseRequest) -> dict[str, object]: command = [ sys.executable, "-u", "-m", "app.simulation.benchmark_regression", "--worker", "--xml", str(request.source_path), "--expected-sha256", request.expected_sha256, "--lane", request.lane, "--stop-time", format(request.stop_time, ".17g"), "--sample-step", format(request.sample_step, ".17g"), "--max-step", format(request.max_step, ".17g"), "--instrumentation-mode", request.instrumentation_mode, ] for checkpoint in request.checkpoint_times: command.extend(("--checkpoint-time", format(checkpoint, ".17g"))) environment = os.environ.copy() environment["PYTHONDONTWRITEBYTECODE"] = "1" environment.update(dict(request.environment_overrides)) return run_bounded_child_process( command, soft_timeout_seconds=request.soft_timeout_seconds, hard_timeout_seconds=request.hard_timeout_seconds, termination_grace_seconds=request.termination_grace_seconds, environment=environment, ) def _completed_case(result: Mapping[str, object], stop_time: float) -> bool: if result.get("outcome") != "completed": return False worker = result.get("worker") if not isinstance(worker, Mapping): return False summary = worker.get("summary") if not isinstance(summary, Mapping): return False simulated_until = summary.get("simulatedUntil") return ( bool(summary.get("success")) and summary.get("status") == "completed" and isinstance(simulated_until, (int, float)) and math.isclose( float(simulated_until), float(stop_time), rel_tol=0.0, abs_tol=max(1.0e-12, 8.0 * math.ulp(max(abs(stop_time), 1.0))), ) ) def evaluate_regression_golden( summary: Mapping[str, object] | None, golden: Mapping[str, object] | None, ) -> dict[str, object]: """Compare physical values numerically and output shape by contract hash.""" if golden is None: return { "configured": False, "evaluated": False, "passed": None, "issues": [], } provenance = golden["provenance"] assert isinstance(provenance, Mapping) source_report = provenance["sourceReport"] assert isinstance(source_report, Mapping) tolerance = golden["tolerance"] layout = golden["physicalLayout"] assert isinstance(tolerance, Mapping) assert isinstance(layout, Mapping) audit: dict[str, object] = { "configured": True, "evaluated": summary is not None, "passed": False, "goldenId": golden["id"], "goldenSha256": golden.get("_sha256"), "goldenPath": golden.get("_path"), "sourceReport": { "path": source_report.get("path"), "sha256": source_report.get("sha256"), "generatedAt": source_report.get("generatedAt"), "metadataCompatibility": source_report.get("metadataCompatibility"), }, "relativeTolerance": float(tolerance["relative"]), "absoluteTolerance": float(tolerance["absolute"]), "checkpointTimeAbsoluteToleranceSeconds": float( tolerance["checkpointTimeAbsoluteSeconds"] ), "stateKeyLayoutSha256": layout["stateKeyLayoutSha256"], "expectedCheckpointCount": len(golden["physicalCheckpoints"]), "expectedStateValueCountPerCheckpoint": len(layout["stateKeys"]), "comparedValueCount": 0, "maxAbsoluteError": None, "maxToleranceRatio": None, "worstValue": None, "issues": [], } if summary is None: audit["issues"] = ["missingSummaryForGolden"] return audit physical_contract = summary.get("physicalContract") actual_checkpoints = ( physical_contract.get("checkpoints") if isinstance(physical_contract, Mapping) else summary.get("checkpoints") ) expected_checkpoints = golden["physicalCheckpoints"] state_keys = layout["stateKeys"] assert isinstance(expected_checkpoints, list) assert isinstance(state_keys, list) issues: list[str] = [] compared_count = 0 max_absolute_error = 0.0 max_tolerance_ratio = 0.0 worst_value: dict[str, object] | None = None if not isinstance(actual_checkpoints, list) or len(actual_checkpoints) != len( expected_checkpoints ): issues.append("stateCheckpointGoldenCountMismatch") else: relative_tolerance = float(tolerance["relative"]) absolute_tolerance = float(tolerance["absolute"]) time_tolerance = float(tolerance["checkpointTimeAbsoluteSeconds"]) for expected_checkpoint, actual_checkpoint in zip( expected_checkpoints, actual_checkpoints ): assert isinstance(expected_checkpoint, Mapping) if not isinstance(actual_checkpoint, Mapping): issues.append("stateCheckpointGoldenLayoutMismatch") continue requested_time = float(expected_checkpoint["requestedTime"]) actual_requested = actual_checkpoint.get("requestedTime") actual_time = actual_checkpoint.get("actualTime") if ( not isinstance(actual_requested, (int, float)) or not isinstance(actual_time, (int, float)) or not math.isclose( float(actual_requested), requested_time, rel_tol=0.0, abs_tol=time_tolerance, ) or not math.isclose( float(actual_time), requested_time, rel_tol=0.0, abs_tol=time_tolerance, ) ): issues.append("stateCheckpointGoldenTimeMismatch") continue actual_values = actual_checkpoint.get("stateValues") expected_values = expected_checkpoint["values"] assert isinstance(expected_values, list) if not isinstance(actual_values, Mapping) or set(actual_values) != set( state_keys ): issues.append("stateCheckpointGoldenLayoutMismatch") continue for key, expected_value in zip(state_keys, expected_values): actual_value = actual_values[key] if not isinstance(actual_value, (int, float)) or not math.isfinite( float(actual_value) ): issues.append("stateCheckpointGoldenNonfiniteValue") continue expected_numeric = float(expected_value) actual_numeric = float(actual_value) absolute_error = abs(actual_numeric - expected_numeric) allowed_error = absolute_tolerance + relative_tolerance * abs( expected_numeric ) tolerance_ratio = ( absolute_error / allowed_error if allowed_error > 0.0 else 0.0 if absolute_error == 0.0 else math.inf ) compared_count += 1 if absolute_error > max_absolute_error: max_absolute_error = absolute_error if tolerance_ratio > max_tolerance_ratio: max_tolerance_ratio = tolerance_ratio worst_value = { "requestedTime": requested_time, "key": key, "expected": expected_numeric, "actual": actual_numeric, "absoluteError": absolute_error, "allowedError": allowed_error, } if tolerance_ratio > 1.0: issues.append("stateCheckpointGoldenValueMismatch") expected_output_contract = golden.get("outputContract") actual_output_contract = summary.get("outputContract") if expected_output_contract is not None: if not isinstance(actual_output_contract, Mapping): issues.append("missingOutputContract") elif actual_output_contract.get("sha256") != expected_output_contract.get( "sha256" ): issues.append("outputContractMismatch") deduplicated_issues = list(dict.fromkeys(issues)) audit.update( { "passed": not deduplicated_issues, "comparedValueCount": compared_count, "maxAbsoluteError": max_absolute_error if compared_count else None, "maxToleranceRatio": max_tolerance_ratio if compared_count else None, "worstValue": worst_value, "issues": deduplicated_issues, "outputContractExpected": ( dict(expected_output_contract) if isinstance(expected_output_contract, Mapping) else None ), "outputContractActual": ( dict(actual_output_contract) if isinstance(actual_output_contract, Mapping) else None ), } ) return audit def _case_correctness_issues( result: Mapping[str, object], *, variant: Mapping[str, object], correctness: Mapping[str, object], golden_evaluation: Mapping[str, object] | None = None, ) -> tuple[str, ...]: """Evaluate structural/event checks and an optional reviewed golden.""" worker = result.get("worker") summary = worker.get("summary") if isinstance(worker, Mapping) else None if not isinstance(summary, Mapping): return ("missingWorkerSummary",) issues: list[str] = [] health = summary.get("seriesHealth") if bool(correctness.get("requireFiniteSeries", False)): if not isinstance(health, Mapping): issues.append("missingSeriesHealth") elif int(health.get("nonfiniteCount", -1)) != 0: issues.append("nonfiniteSeries") if bool(correctness.get("requireStrictlyIncreasingTimes", False)): if not isinstance(health, Mapping) or not bool( health.get("timeStrictlyIncreasing", False) ): issues.append("sampleTimesNotStrictlyIncreasing") stop_time = float(variant["stopTime"]) if not isinstance(health, Mapping) or not isinstance( health.get("seriesCount"), int ) or int(health["seriesCount"]) <= 0: issues.append("emptySeries") elif not isinstance(health.get("timeEnd"), (int, float)) or not math.isclose( float(health["timeEnd"]), stop_time, rel_tol=0.0, abs_tol=max(1.0e-12, 8.0 * math.ulp(max(abs(stop_time), 1.0))), ): issues.append("seriesDoesNotReachStopTime") checkpoints = summary.get("checkpoints") expected_checkpoint_times = variant.get("checkpointTimes", []) if not isinstance(checkpoints, list) or not isinstance( expected_checkpoint_times, list ): issues.append("missingStateCheckpoints") elif len(checkpoints) != len(expected_checkpoint_times) or any( not isinstance(checkpoint, Mapping) or not bool(checkpoint.get("available", False)) for checkpoint in checkpoints ): issues.append("stateCheckpointUnavailable") elif any( not isinstance(checkpoint.get("stateValues"), Mapping) or not checkpoint["stateValues"] for checkpoint in checkpoints if isinstance(checkpoint, Mapping) ): issues.append("stateCheckpointValuesMissing") maximum_residual = correctness.get("maximumScaledResidual") diagnostics = summary.get("diagnostics") pressure_flow = ( diagnostics.get("pressureFlow") if isinstance(diagnostics, Mapping) else None ) observed_residual = ( pressure_flow.get("maxScaledResidual") if isinstance(pressure_flow, Mapping) else None ) if isinstance(maximum_residual, (int, float)): if not isinstance(observed_residual, (int, float)) or not math.isfinite( float(observed_residual) ): issues.append("missingOrNonfiniteScaledResidual") elif float(observed_residual) > float(maximum_residual): issues.append("scaledResidualExceedsLimit") expected_signal_times = variant.get("expectedSignalEventTimes", []) event_trace = summary.get("eventTrace") actual_signal_times = ( event_trace.get("signalEventTimes") if isinstance(event_trace, Mapping) else None ) signal_tolerance = float( correctness.get("signalEventTimeAbsoluteToleranceSeconds", 1.0e-12) ) if not isinstance(expected_signal_times, list) or not isinstance( actual_signal_times, list ): issues.append("missingSignalEventTrace") elif len(expected_signal_times) != len(actual_signal_times) or any( not isinstance(actual, (int, float)) or not math.isclose( float(actual), float(expected), rel_tol=0.0, abs_tol=signal_tolerance, ) for expected, actual in zip(expected_signal_times, actual_signal_times) ): issues.append("signalEventTraceMismatch") segments = event_trace.get("segments") if isinstance(event_trace, Mapping) else None segment_start_times = ( [ segment.get("startTime") for segment in segments if isinstance(segment, Mapping) ] if isinstance(segments, list) else [] ) if isinstance(expected_signal_times, list) and any( not any( isinstance(actual, (int, float)) and math.isclose( float(actual), float(expected), rel_tol=0.0, abs_tol=signal_tolerance, ) for actual in segment_start_times ) for expected in expected_signal_times ): issues.append("signalEventSegmentMissing") if bool(correctness.get("mechanicalTransitionTimesAvailable", False)): if not isinstance(event_trace, Mapping) or not bool( event_trace.get("mechanicalTransitionTimesAvailable", False) ): issues.append("mechanicalTransitionTimesUnavailable") expected_mechanical_times = variant.get("expectedMechanicalTransitionTimes") if isinstance(expected_mechanical_times, list): actual_mechanical_times = ( event_trace.get("mechanicalTransitionTimes") if isinstance(event_trace, Mapping) else None ) event_tolerance = float( correctness.get("eventTimeAbsoluteToleranceSeconds", 2.0e-5) ) if not isinstance(actual_mechanical_times, list) or len( actual_mechanical_times ) != len(expected_mechanical_times) or any( not isinstance(actual, (int, float)) or not math.isclose( float(actual), float(expected), rel_tol=0.0, abs_tol=event_tolerance, ) for expected, actual in zip( expected_mechanical_times, actual_mechanical_times ) ): issues.append("mechanicalTransitionTraceMismatch") if isinstance(golden_evaluation, Mapping): raw_golden_issues = golden_evaluation.get("issues") if isinstance(raw_golden_issues, list): issues.extend( str(issue) for issue in raw_golden_issues if isinstance(issue, str) ) return tuple(dict.fromkeys(issues)) def _case_wall_seconds(result: Mapping[str, object]) -> float | None: worker = result.get("worker") if isinstance(worker, Mapping): value = worker.get("wallSeconds") if isinstance(value, (int, float)) and float(value) > 0.0: return float(value) value = result.get("orchestrationWallSeconds") if isinstance(value, (int, float)) and float(value) > 0.0: return float(value) return None def run_regression_suite( manifest_path: Path | str = DEFAULT_MANIFEST_PATH, *, lane: str = "production", case_ids: Sequence[str] | None = None, case_executor: CaseExecutor = execute_regression_case, ) -> dict[str, object]: """Run requested horizons in order, deferring unsafe downstream work.""" manifest = load_regression_manifest(manifest_path) lanes = manifest["lanes"] assert isinstance(lanes, dict) if lane not in lanes: raise RegressionManifestError(f"Unknown regression lane {lane!r}.") lane_config = lanes[lane] assert isinstance(lane_config, dict) variants = manifest["variants"] assert isinstance(variants, dict) sequence = manifest["sequence"] assert isinstance(sequence, list) if case_ids is not None: requested = set(case_ids) if not requested: raise RegressionManifestError("No regression variants were selected.") unknown = requested - set(sequence) if unknown: raise RegressionManifestError( f"Unknown requested variants: {', '.join(sorted(unknown))}." ) last_requested_index = max(sequence.index(case_id) for case_id in requested) selected_sequence = list(sequence[: last_requested_index + 1]) else: selected_sequence = list(sequence) source_path = Path(str(manifest["_sourcePath"])) source_payload = source_path.read_bytes() source_config = source_simulation_config(source_payload) sampling_mode = lane_config["samplingMode"] sample_step = ( float(source_config["sampleStep"]) if sampling_mode == "source" else float(lane_config["sampleStep"]) ) max_step_mode = lane_config.get("maxStepMode", "source") max_step = ( float(source_config["maxStep"]) if max_step_mode == "source" else float(lane_config["maxStep"]) ) execution = manifest["execution"] assert isinstance(execution, dict) safety_factor = float(execution.get("predictionSafetyFactor", 1.5)) termination_grace = float(execution.get("terminationGraceSeconds", 5.0)) instrumentation_mode = str( lane_config.get("instrumentationMode", "standard") ) expected_sha256 = str(manifest["source"]["sha256"]) # type: ignore[index] raw_environment = execution.get("environment", {}) if not isinstance(raw_environment, dict) or not all( isinstance(key, str) and isinstance(value, str) for key, value in raw_environment.items() ): raise RegressionManifestError("execution.environment must map strings to strings.") environment_overrides = tuple(sorted(raw_environment.items())) correctness = manifest["correctness"] assert isinstance(correctness, dict) loaded_goldens = manifest.get("_goldens", {}) assert isinstance(loaded_goldens, dict) case_reports: list[dict[str, object]] = [] predecessor_completed = True previous_stop: float | None = None previous_wall: float | None = None deferral_reason: str | None = None for case_id in selected_sequence: variant = variants[case_id] assert isinstance(variant, dict) stop_time = float(variant["stopTime"]) soft_timeout = float(variant["softTimeoutSeconds"]) hard_timeout = float(variant["hardTimeoutSeconds"]) runtime_prediction_eligible = bool( variant.get("useForRuntimePrediction", True) ) predicted_wall: float | None = None if previous_stop is not None and previous_wall is not None: predicted_wall = ( previous_wall * stop_time / previous_stop * safety_factor ) if not predecessor_completed: deferral_reason = deferral_reason or "predecessorDidNotComplete" elif predicted_wall is not None and predicted_wall > soft_timeout: predecessor_completed = False deferral_reason = "predictedWallExceedsSoftBudget" if not predecessor_completed: case_reports.append( { "caseId": case_id, "stopTime": stop_time, "sampleStep": sample_step, "maxStep": max_step, "lane": lane, "outcome": "deferred", "reason": deferral_reason, "predictedWallSeconds": predicted_wall, "softTimeoutSeconds": soft_timeout, "hardTimeoutSeconds": hard_timeout, "runtimePredictionEligible": runtime_prediction_eligible, } ) continue request = RegressionCaseRequest( case_id=case_id, source_path=source_path, expected_sha256=expected_sha256, lane=lane, stop_time=stop_time, sample_step=sample_step, max_step=max_step, checkpoint_times=tuple( float(value) for value in variant.get("checkpointTimes", []) ), soft_timeout_seconds=soft_timeout, hard_timeout_seconds=hard_timeout, termination_grace_seconds=termination_grace, instrumentation_mode=instrumentation_mode, environment_overrides=environment_overrides, ) result = case_executor(request) solver_completed = _completed_case(result, stop_time) worker = result.get("worker") summary = worker.get("summary") if isinstance(worker, Mapping) else None case_goldens = loaded_goldens.get(case_id, {}) golden = ( case_goldens.get(lane) if isinstance(case_goldens, Mapping) else None ) golden_evaluation = evaluate_regression_golden( summary if isinstance(summary, Mapping) else None, golden if isinstance(golden, Mapping) else None, ) correctness_issues = ( _case_correctness_issues( result, variant=variant, correctness=correctness, golden_evaluation=golden_evaluation, ) if solver_completed else () ) report = { "caseId": case_id, "stopTime": stop_time, "sampleStep": sample_step, "maxStep": max_step, "lane": lane, "samplingMode": sampling_mode, "maxStepMode": max_step_mode, "runtimePredictionEligible": runtime_prediction_eligible, "predictedWallSeconds": predicted_wall, "softTimeoutSeconds": soft_timeout, "hardTimeoutSeconds": hard_timeout, **result, "acceptance": { "evaluated": solver_completed, "passed": solver_completed and not correctness_issues, "issues": list(correctness_issues), "regressionGolden": golden_evaluation, }, } if solver_completed and correctness_issues: report["outcome"] = "correctness_failed" case_reports.append(report) predecessor_completed = solver_completed and not correctness_issues if predecessor_completed: if runtime_prediction_eligible: previous_stop = stop_time previous_wall = _case_wall_seconds(result) else: deferral_reason = "predecessorDidNotComplete" public_manifest = { key: value for key, value in manifest.items() if not key.startswith("_") } return { "schemaVersion": REPORT_SCHEMA_VERSION, "generatedAt": datetime.now(UTC).isoformat(), "manifestId": manifest.get("id"), "manifestPath": str(manifest["_manifestPath"]), "lane": lane, "laneDescription": lane_config.get("description"), "source": { **dict(manifest["source"]), # type: ignore[arg-type] "resolvedPath": str(source_path), "companionResolvedPath": manifest.get("_companionPath"), "bytes": len(source_payload), "simulation": source_config, }, "manifest": public_manifest, "cases": case_reports, } def _parse_arguments(argv: Sequence[str] | None = None) -> argparse.Namespace: parser = argparse.ArgumentParser( description="Run progressively bounded System XML regression horizons." ) parser.add_argument("--manifest", type=Path, default=DEFAULT_MANIFEST_PATH) parser.add_argument( "--lane", default="production", help="Manifest lane: production runs approved correctness contracts; solver-only is an explicit coarse-output iteration lane.", ) parser.add_argument("--case", action="append", default=[]) parser.add_argument("--output", type=Path) parser.add_argument("--worker", action="store_true", help=argparse.SUPPRESS) parser.add_argument("--xml", help=argparse.SUPPRESS) parser.add_argument("--expected-sha256", help=argparse.SUPPRESS) parser.add_argument("--stop-time", type=float, help=argparse.SUPPRESS) parser.add_argument("--sample-step", type=float, help=argparse.SUPPRESS) parser.add_argument("--max-step", type=float, help=argparse.SUPPRESS) parser.add_argument( "--checkpoint-time", type=float, action="append", default=[], help=argparse.SUPPRESS, ) parser.add_argument( "--instrumentation-mode", choices=("off", "standard", "audit"), default="standard", help=argparse.SUPPRESS, ) arguments = parser.parse_args(argv) if arguments.worker: missing = [ name for name in ( "xml", "expected_sha256", "stop_time", "sample_step", "max_step", ) if getattr(arguments, name) is None ] if missing: parser.error("Worker arguments missing: " + ", ".join(missing)) return arguments def main(argv: Sequence[str] | None = None) -> int: arguments = _parse_arguments(argv) if arguments.worker: return run_worker(arguments) report = run_regression_suite( arguments.manifest, lane=arguments.lane, case_ids=arguments.case or None, ) serialized = json.dumps(report, ensure_ascii=False, indent=2, default=str) + "\n" if arguments.output is None: print(serialized, end="") else: arguments.output.parent.mkdir(parents=True, exist_ok=True) arguments.output.write_text(serialized, encoding="utf-8") print(f"Regression report written to {arguments.output.resolve()}") outcomes = [case.get("outcome") for case in report["cases"]] if outcomes and all(outcome == "completed" for outcome in outcomes): return 0 if outcomes and all( outcome in {"completed", "deferred"} for outcome in outcomes ): return 2 return 1 if __name__ == "__main__": raise SystemExit(main())