优化雅可比矩阵计算;端口转发情况下仿真结果传输方式优化

This commit is contained in:
lujingze committed 2026-09-12 03:57:31 +00:00
1 parent 3bc4be3c06
commit aa4951b14e
28 files changed
+8038 -23

No files matched your search

+522
View File
@@ -0,0 +1,522 @@
r"""Prepare, then optionally benchmark the production automatic Jacobian.
.venv/bin/python tests/manual/benchmark_native_jacobian.py \
--input tests/data/test-mql-8-corrected.json \
--output-dir test/jacobian-production/benchmark --warmups 1 --repeats 3 \
--verify-jacobian
Add --run to build once and execute. Preparation generates C without compiling
or solving. --verify-jacobian (--verify alias) adds one separate, untimed-for-
statistics full-matrix diagnostic run. Ordinary runs use the production default.
An optional --baseline-executable must point to a frozen historical executable
whose DEFAULT algorithm is the old dense Jacobian. No strategy selector is sent
to either executable. The external binary hash and provenance are recorded; if
it reports a Jacobian mode, it must report dense-difference. With no external
baseline only current-production repeatability is compared. Historical summary
keys dense/auto are retained; dense statistics are empty and speedComparison is
null when no external baseline was supplied.
Both executables receive the same time, sampling and tolerance arguments. The
caller must select a frozen baseline for the same model and embedded absolute
tolerances; recorded provenance and structural checks alone do not prove this.
Parsing and comparisons are outside process timing. Exact payload equality and
signed-zero bit equality are reported separately, without resampling or tolerance
relaxation. --record-differences permits failed external comparisons to be saved
for separate trajectory review; repeatability and verification still must pass.
"""
from __future__ import annotations
import argparse
from array import array
from hashlib import sha256
import json
import math
import os
from pathlib import Path
import platform
import statistics
import struct
import subprocess
import sys
import time
ROOT = Path(__file__).resolve().parents[2]
PAYLOAD = ("series", "final", "finalState")
COUNTERS = ("nfev", "acceptedSteps", "rejectedSteps", "njev", "nlu",
"stateTransitions", "solverStarts", "jacobianRhsCalls",
"jacobianColoredEvals", "jacobianFallbacks", "jacobianChecks",
"jacobianMismatches", "cvodeRhsCalls", "cvodeLinearRhsCalls")
TIMINGS = ("solveSeconds", "solveCpuSeconds", "processWallSeconds")
INVARIANT_METADATA = ("success", "status", "backend", "method", "solver", "sundialsVersion", "simulatedUntil")
MAX_DETAILS = 20
def digest(path: Path) -> str:
with path.open("rb") as stream:
value = sha256()
for block in iter(lambda: stream.read(1024 * 1024), b""):
value.update(block)
return value.hexdigest()
def write_json(path: Path, value: object) -> None:
path.write_text(json.dumps(value, ensure_ascii=False, indent=2, allow_nan=False) + "\n", encoding="utf-8")
def pointer(*parts: object) -> str:
return "/" + "/".join(str(part).replace("~", "~0").replace("/", "~1") for part in parts)
def read_result(path: Path) -> tuple[dict, str]:
def unique(items):
result = {}
for key, value in items:
if key in result:
raise ValueError(f"Duplicate JSON object key: {key!r}")
result[key] = value
return result
def reject(token):
raise ValueError(f"Nonfinite JSON token: {token}")
def finite_float(token):
value = float(token)
if not math.isfinite(value):
raise ValueError(f"Nonfinite JSON number: {token}")
return value
raw = path.read_bytes()
data = json.loads(raw, parse_int=lambda s: -0.0 if s == "-0" else int(s),
parse_float=finite_float, parse_constant=reject, object_pairs_hook=unique)
if not isinstance(data, dict):
raise ValueError("Expected a native result object")
return data, sha256(raw).hexdigest()
def number(value, path: str) -> float:
if isinstance(value, bool) or not isinstance(value, (int, float)):
raise ValueError(f"Expected numeric payload at {path}")
try:
result = float(value)
except (OverflowError, ValueError) as exc:
raise ValueError(f"Cannot represent binary64 at {path}") from exc
if not math.isfinite(result) or (isinstance(value, int) and int(result) != value):
raise ValueError(f"Nonfinite or inexact binary64 at {path}")
return result
def blocks(data: dict):
for key, values in data["series"].items():
yield pointer("series", key), values
for key, value in data["final"].items():
yield pointer("final", key), [value]
yield "/finalState", data["finalState"]
def validate_result(data: dict, prepared: dict, *, external_baseline: bool = False) -> dict:
required_counters = COUNTERS[:7] if external_baseline else COUNTERS
mode_fields = () if external_baseline else ("jacobianMode",)
for key in (*PAYLOAD, *required_counters, *INVARIANT_METADATA, *mode_fields, "maxAcceptedStep", "solveSeconds", "solveCpuSeconds"):
if key not in data:
raise ValueError(f"Missing result field: {key}")
if not isinstance(data["series"], dict) or not isinstance(data["final"], dict) or not isinstance(data["finalState"], list):
raise ValueError("Invalid series/final/finalState structure")
expected = set(prepared["outputKeys"])
if set(data["series"]) != expected | {"time"} or set(data["final"]) != expected:
raise ValueError("Result output key sets do not match the generated manifest")
if len(data["finalState"]) != prepared["stateCount"]:
raise ValueError("finalState length does not match the generated manifest")
times = data["series"]["time"]
if not isinstance(times, list) or not times:
raise ValueError("Full sampled output is required")
for key, values in data["series"].items():
if not isinstance(values, list) or len(values) != len(times):
raise ValueError(f"Series column length differs from time: {key}")
for key in COUNTERS:
if external_baseline and key not in data:
continue
if type(data[key]) is not int or data[key] < 0:
raise ValueError(f"Invalid nonnegative counter: {key}")
if data["success"] is not True or data["status"] != "completed":
raise ValueError(f"Native solve did not complete: {data.get('message')}")
if (data["backend"], data["method"], data["solver"]) != ("native-c", "BDF", "CVODE"):
raise ValueError("Unexpected backend/integrator")
for key in ("simulatedUntil", "maxAcceptedStep", "solveSeconds", "solveCpuSeconds"):
number(data[key], pointer(key))
if data["solveSeconds"] <= 0 or data["solveCpuSeconds"] < 0:
raise ValueError("Invalid reported solve duration")
cells = negative_zero = positive_zero = 0
for path, values in blocks(data):
for index, value in enumerate(values):
value = number(value, path + "/" + str(index))
cells += 1
if value == 0:
if math.copysign(1.0, value) < 0:
negative_zero += 1
else:
positive_zero += 1
cfg = prepared["settings"]
if times[0] != cfg["t_start"] or times[-1] != cfg["t_stop"] or data["simulatedUntil"] != cfg["t_stop"]:
raise ValueError("Result does not cover the entire requested interval")
if any(a >= b for a, b in zip(times, times[1:])):
raise ValueError("Sample times must be strictly increasing")
if data["maxAcceptedStep"] > cfg["max_step"] * (1 + 1e-14):
raise ValueError("Reported accepted step exceeds configured maximum")
# Match the runtime's start + index * sample_step arithmetic exactly.
regular = []
index = 0
while (value := cfg["t_start"] + index * prepared["sampleStep"]) <= cfg["t_stop"]:
regular.append(value)
index += 1
actual_times = set(times)
missing_regular = [value for value in regular if value not in actual_times]
regular_set = set(regular)
extra = [value for value in times if value not in regular_set]
if missing_regular:
raise ValueError(f"Missing regular samples: {missing_regular[:MAX_DETAILS]}")
return {"sampleCount": len(times), "seriesColumns": len(data["series"]),
"finalScalars": len(data["final"]), "finalStateValues": len(data["finalState"]),
"payloadValues": cells, "negativeZeroValues": negative_zero, "positiveZeroValues": positive_zero,
"regularSampleCount": len(regular), "extraSampleTimes": extra,
"extraSampleInterpretation": "Off-grid saved points, usually events; a non-grid final endpoint may also appear.",
"counterAccountingMatches": (data["nfev"] == data["cvodeRhsCalls"] + data["cvodeLinearRhsCalls"] + data["jacobianRhsCalls"]
if all(k in data for k in ("cvodeRhsCalls", "cvodeLinearRhsCalls", "jacobianRhsCalls")) else None),
"missingHistoricalCounters": [k for k in COUNTERS if k not in data]}
def compare_payload(baseline: dict, candidate: dict) -> dict:
"""Exact values and separately exact bits; never compare unaligned series."""
structure = []
for section in ("series", "final"):
left, right = baseline[section], candidate[section]
if left.keys() != right.keys():
structure.append({"path": pointer(section), "missing": sorted(left.keys() - right.keys()), "extra": sorted(right.keys() - left.keys())})
for key in baseline["series"].keys() & candidate["series"].keys():
a, b = len(baseline["series"][key]), len(candidate["series"][key])
if a != b:
structure.append({"path": pointer("series", key), "baselineLength": a, "candidateLength": b})
if len(baseline["finalState"]) != len(candidate["finalState"]):
structure.append({"path": "/finalState", "baselineLength": len(baseline["finalState"]), "candidateLength": len(candidate["finalState"])})
left_times, right_times = baseline["series"]["time"], candidate["series"]["time"]
same_times = left_times == right_times
time_differences = []
for i, (a, b) in enumerate(zip(left_times, right_times)):
if a != b:
time_differences.append({"index": i, "baseline": a, "candidate": b})
if len(time_differences) == MAX_DETAILS:
break
metadata = [{"key": key, "baseline": baseline.get(key), "candidate": candidate.get(key)}
for key in INVARIANT_METADATA if baseline.get(key) != candidate.get(key)]
numerical = bit_count = zero_signs = compared = 0
details = []
compared_blocks = 0
pairs = []
if same_times:
for key in sorted(baseline["series"].keys() & candidate["series"].keys()):
pairs.append((pointer("series", key), baseline["series"][key], candidate["series"][key]))
for key in sorted(baseline["final"].keys() & candidate["final"].keys()):
pairs.append((pointer("final", key), [baseline["final"][key]], [candidate["final"][key]]))
pairs.append(("/finalState", baseline["finalState"], candidate["finalState"]))
for path, left, right in pairs:
if len(left) != len(right):
continue
compared_blocks += 1
compared += len(left)
# Fast whole-block bit check; inspect individual cells only on differences.
if array("d", left).tobytes() == array("d", right).tobytes():
continue
for index, (a, b) in enumerate(zip(left, right, strict=True)):
packed_a, packed_b = struct.pack("<d", a), struct.pack("<d", b)
if packed_a == packed_b:
continue
bit_count += 1
numerical += a != b
zero_signs += a == 0 and b == 0
if len(details) < MAX_DETAILS:
details.append({"path": path if path.startswith("/final/") else path + "/" + str(index), "baseline": a, "candidate": b,
"numericallyEqual": a == b, "baselineBitsLE": packed_a.hex(), "candidateBitsLE": packed_b.hex()})
complete = not structure and same_times
return {"passed": complete and not metadata and numerical == 0,
"comparisonContract": "Exact payload numeric equality; signed-zero differences do not fail numeric equality and are reported separately. Solver work counters are observations, not invariants.",
"structureEqual": not structure, "structureDifferences": structure[:MAX_DETAILS],
"structureDifferenceCount": len(structure), "metadataDifferences": metadata,
"timeAxis": {"numericallyEqual": same_times, "baselineLength": len(left_times), "candidateLength": len(right_times),
"firstIndexDifferences": time_differences, "seriesCompared": same_times,
"interpretation": "Time differences are index diagnostics only; unequal time axes disable series-value comparison. No interpolation or event removal."},
"allPayloadCompared": complete, "comparedBlocks": compared_blocks, "comparedValues": compared,
"numericallyUnequalValues": numerical, "differentBits": bit_count, "signedZeroDifferences": zero_signs,
"allPayloadBitsEqual": complete and bit_count == 0, "firstDifferences": details,
"counterDifferences": {key: {"baseline": baseline.get(key), "candidate": candidate.get(key)}
for key in COUNTERS if baseline.get(key) != candidate.get(key)},
"maxAcceptedStep": {"baseline": baseline.get("maxAcceptedStep"), "candidate": candidate.get("maxAcceptedStep")}}
def stats(values) -> dict:
values = list(values)
return {"n": len(values), "median": statistics.median(values) if values else None,
"min": min(values) if values else None, "max": max(values) if values else None, "values": values}
def prepare(args) -> tuple[dict, object]:
sys.path.insert(0, str(ROOT))
from app.main import compile_system_xml_network
from app.simulation.backends import simulation_config
from app.simulation.native_codegen.compiler import compile_native_program
from app.simulation.native_codegen.input import load_input
from app.simulation.native_codegen.tolerances import state_absolute_tolerance
output = args.output_dir.resolve()
if output == ROOT / "test" or not output.is_relative_to(ROOT / "test"):
raise ValueError("Choose an output subdirectory beneath the repository's ignored test/ directory")
if any((output / name).exists() for name in ("summary.json", "dense", "auto", "verify")):
raise ValueError("Run artifacts already exist; choose a fresh output directory")
external = None
if args.baseline_executable is not None:
executable = args.baseline_executable.resolve()
if not executable.is_file() or not os.access(executable, os.X_OK):
raise ValueError("--baseline-executable must be an existing executable from a frozen historical version")
manifest = executable.parent / "manifest.json"
external = {"executable": str(executable), "sha256": digest(executable),
"manifestPath": str(manifest) if manifest.is_file() else None,
"manifestSha256": digest(manifest) if manifest.is_file() else None,
"strategy": "historical executable default; dense-difference required if reported",
"modelAndEmbeddedToleranceProvenance": "Caller-supplied frozen model; current manifest checks payload dimensions, not full physical equivalence"}
started = time.perf_counter()
xml, document = load_input(args.input)
cfg = simulation_config(document.simulation)
if cfg.method != "BDF" or cfg.rtol != 1e-8 or cfg.atol != 1e-8 or cfg.first_step is not None:
raise ValueError("Input must use BDF, production rtol=1e-8, generated atol and automatic first step")
step = document.simulation.sample_step
if not all(math.isfinite(v) for v in (cfg.t_start, cfg.t_stop, cfg.max_step, step)) or not (cfg.t_stop > cfg.t_start and cfg.max_step > 0 and step > 0):
raise ValueError("Invalid finite simulation interval/step settings")
if (cfg.t_stop - cfg.t_start) / step > 1000000 or cfg.t_start + step == cfg.t_start:
raise ValueError("Invalid or excessive sampling grid")
program = compile_native_program(compile_system_xml_network(document))
if external and external["manifestPath"]:
frozen_manifest = json.loads(Path(external["manifestPath"]).read_text())
if frozen_manifest.get("stateKeys") != list(program.state_keys):
raise ValueError("Frozen baseline manifest state keys/order differ from the current model")
frozen_outputs = {v["key"] for v in frozen_manifest.get("variables", [])}
if frozen_outputs != {v.key for v in program.variables}:
raise ValueError("Frozen baseline manifest output keys differ from the current model")
external["manifestStateAndOutputContractMatched"] = True
output.mkdir(parents=True, exist_ok=True)
(output / "input.xml").write_bytes(xml)
if args.input.suffix.lower() == ".json":
(output / "input.json").write_bytes(args.input.read_bytes())
(output / "model.c").write_text(program.source, encoding="utf-8")
(output / "model.h").write_text(program.header, encoding="utf-8")
write_json(output / "model-contract.json", program.manifest())
source_paths = sorted((ROOT / "native").rglob("*.c")) + sorted((ROOT / "native").rglob("*.h")) + sorted((ROOT / "app/simulation/native_codegen").glob("*.py"))
prepared = {"schemaVersion": 1, "preparedOnly": not args.run, "input": str(args.input.resolve()),
"inputSha256": digest(args.input), "xmlSha256": sha256(xml).hexdigest(),
"scriptSha256": digest(Path(__file__)), "modelSourceSha256": digest(output / "model.c"),
"modelHeaderSha256": digest(output / "model.h"), "settings": vars(cfg), "sampleStep": step,
"stateCount": len(program.state_keys), "stateKeys": list(program.state_keys),
"stateAbsoluteTolerances": [float(state_absolute_tolerance(k)) for k in program.state_keys],
"outputKeys": [v.key for v in program.variables], "modelContract": program.manifest(),
"sourceHashes": {str(p.relative_to(ROOT)): digest(p) for p in source_paths},
"environment": {"platform": platform.platform(), "python": sys.version, "machine": platform.machine()},
"warmupsPerMode": args.warmups, "repeatsPerMode": args.repeats,
"algorithm": "production-automatic", "externalBaseline": external,
"activeModes": ["dense", "auto"] if external else ["auto"],
"verifyRequested": args.verify, "recordDifferences": args.record_differences, "nativeTimeoutSeconds": args.timeout,
"processTimeoutSeconds": args.timeout + 10, "preparationSeconds": time.perf_counter() - started,
"timingContract": "Production automatic executable plus an optional caller-supplied frozen external baseline; complete sampled output. C-reported solve wall/CPU and subprocess creation-through-reap wall only. Build, parsing, validation and comparisons excluded. No independently measured write/projection stage; process-minus-solve is not called write time. Ordinary file writes, no fsync.",
"comparisonContract": "Exact structure, numeric values and time axis for series/final/finalState. Bits, including signed zero, are separately reported. No loosened tolerance or interpolation. Counters may differ.",
"pairOrder": [{"pair": i + 1, "warmup": i < args.warmups,
"modes": (["dense", "auto"] if i % 2 == 0 else ["auto", "dense"]) if external else ["auto"]}
for i in range(args.warmups + args.repeats)]}
write_json(output / "prepared.json", prepared)
return prepared, program
def execute(args, prepared: dict, program) -> dict:
from app.simulation.native_codegen.build import build_native
output = args.output_dir.resolve()
summary = {"schemaVersion": 1, "complete": False, "passed": False, "prepared": prepared,
"errors": [], "rows": [], "pairs": [], "verify": None, "statistics": None,
"externalBaseline": prepared["externalBaseline"],
"speedComparison": None, "strictFailureMeans": "Exact equivalence was not established; retain outputs for independent convergence diagnostics. No automatic acceptance-tolerance change."}
rows, pairs = summary["rows"], summary["pairs"]
def save():
write_json(output / "summary.json", summary)
def run(mode: str, label: str, *, pair=None, warmup=False, diagnostic=False):
directory = output / mode / label
directory.mkdir(parents=True, exist_ok=False)
cfg = prepared["settings"]
executable = Path(prepared["externalBaseline"]["executable"]) if mode == "dense" else build.executable
command = [str(executable), "--method", "BDF",
"--start", str(cfg["t_start"]), "--stop", str(cfg["t_stop"]),
"--sample-step", str(prepared["sampleStep"]), "--max-step", str(cfg["max_step"]),
"--rtol", "1e-8", "--timeout", str(args.timeout),
"--output", str(directory / "result.json"), "--result-index", str(directory / "result-index.json"),
"--cancel-file", str(directory / "cancel.request")]
if diagnostic:
command.append("--verify-jacobian")
row = {"mode": mode, "label": label, "pair": pair, "warmup": warmup, "diagnostic": diagnostic,
"includedInStatistics": not warmup and not diagnostic, "directory": str(directory), "command": command,
"completed": False, "resultValidation": None, "externalBaseline": mode == "dense"}
rows.append(row)
write_json(directory / "command.json", command)
environment = dict(os.environ)
for key in ("NATIVE_COMPUTE_PROFILE", "NATIVE_STAGE_PROFILE"):
environment.pop(key, None)
try:
with (directory / "stdout.log").open("wb") as stdout, (directory / "stderr.log").open("wb") as stderr:
started = time.perf_counter()
try:
process = subprocess.run(command, cwd=executable.parent, env=environment, stdin=subprocess.DEVNULL,
stdout=stdout, stderr=stderr, timeout=args.timeout + 10)
row["exitCode"] = process.returncode
finally:
row["processWallSeconds"] = time.perf_counter() - started
result_path = directory / "result.json"
if not result_path.is_file():
raise RuntimeError(f"No native result: {directory}")
data, result_hash = read_result(result_path)
row.update({key: value for key, value in data.items() if key not in PAYLOAD})
row["resultSha256"] = result_hash
row["resultBytes"] = result_path.stat().st_size
row["resultValidation"] = validate_result(data, prepared, external_baseline=mode == "dense")
if row["exitCode"] != 0:
raise RuntimeError(f"Native exit code {row['exitCode']}: {directory}")
if row["resultValidation"]["counterAccountingMatches"] is False:
raise RuntimeError(f"RHS counter accounting failed: {directory}")
if mode == "dense" and data.get("jacobianMode") not in (None, "dense-difference"):
raise RuntimeError("External baseline is not a frozen default-dense executable; current production cannot emulate the old strategy")
if mode == "dense":
row["historicalStrategyEvidence"] = "reported-dense" if data.get("jacobianMode") else "unreported: caller-supplied frozen provenance"
if diagnostic and (data["jacobianChecks"] <= 0 or data["jacobianMismatches"] != 0 or
data["jacobianChecks"] != data["jacobianColoredEvals"]):
raise RuntimeError("Verify mode did not validate every computed colored Jacobian without mismatches")
row["completed"] = True
print(f"{mode}/{label}: solve={row['solveSeconds']:.6f}s process={row['processWallSeconds']:.6f}s nfev={row['nfev']}", flush=True)
return data
except Exception as exc:
row["error"] = f"{type(exc).__name__}: {exc}"
raise
finally:
write_json(directory / "run.json", row)
save()
try:
# Build current production once. An optional baseline is never built or modified.
build = build_native(program, cache_dir=output / "cache")
summary["build"] = {"executable": str(build.executable), "buildKey": build.manifest["buildKey"],
"cacheHit": build.cache_hit, "seconds": build.seconds, "manifest": build.manifest}
if prepared["externalBaseline"]:
if digest(Path(prepared["externalBaseline"]["executable"])) != prepared["externalBaseline"]["sha256"]:
raise RuntimeError("Frozen external baseline changed after preparation")
if digest(build.executable) == prepared["externalBaseline"]["sha256"]:
raise RuntimeError("External baseline equals the current production executable")
save()
verify_data = run("verify", "validation", diagnostic=True) if args.verify else None
first_auto = first_dense = None
warmup_count = measured_count = 0
for planned in prepared["pairOrder"]:
is_warmup = planned["warmup"]
if is_warmup:
warmup_count += 1
label = f"warmup-{warmup_count}"
else:
measured_count += 1
label = f"run-{measured_count}"
results = {mode: run(mode, label, pair=planned["pair"], warmup=is_warmup) for mode in planned["modes"]}
comparison = compare_payload(results["dense"], results["auto"]) if "dense" in results else None
auto_repeat = compare_payload(first_auto, results["auto"]) if first_auto is not None else None
dense_repeat = compare_payload(first_dense, results["dense"]) if first_dense is not None else None
if first_auto is None:
first_auto = results["auto"]
first_dense = results.get("dense")
if verify_data is not None:
summary["verify"] = compare_payload(results["auto"], verify_data)
summary["verify"]["modes"] = "production default versus --verify-jacobian: diagnostic must preserve trajectory"
verify_data = None
record = {"pair": planned["pair"], "label": label, "warmup": is_warmup, "order": planned["modes"],
"externalBaseline": prepared["externalBaseline"] is not None,
"denseVsAuto": comparison, "denseRepeatVsFirstDense": dense_repeat,
"autoRepeatVsFirstAuto": auto_repeat}
pairs.append(record)
write_json(output / f"comparison-{label}.json", record)
save()
if any(check is not None and not check["passed"] for check in (auto_repeat, dense_repeat, summary["verify"])):
raise RuntimeError(f"Within-algorithm or default/verify reproducibility failed at {label}")
if comparison is not None and not comparison["passed"] and not args.record_differences:
raise RuntimeError(f"Strict external-baseline comparison failed at {label}; raw artifacts retained for convergence diagnostics")
summary["statistics"] = {mode: {key: stats(row[key] for row in rows if row["mode"] == mode and row["includedInStatistics"] and key in row)
for key in (*TIMINGS, *COUNTERS, "resultBytes")} for mode in ("dense", "auto")}
if prepared["externalBaseline"]:
ratios = {}
for metric in TIMINGS:
matched = []
for pair in pairs:
if pair["warmup"]:
continue
selected = {row["mode"]: row for row in rows if row["pair"] == pair["pair"]}
old, new = selected["dense"][metric], selected["auto"][metric]
if old <= 0 or new <= 0:
raise RuntimeError(f"Cannot form a positive-duration comparison for {metric}")
matched.append({"pair": pair["pair"], "dense": old, "auto": new,
"reductionPercent": 100 * (old - new) / old, "speedup": old / new})
old_median = summary["statistics"]["dense"][metric]["median"]
new_median = summary["statistics"]["auto"][metric]["median"]
ratios[metric] = {"pairs": matched,
"pairedReductionPercent": stats(p["reductionPercent"] for p in matched),
"pairedSpeedup": stats(p["speedup"] for p in matched),
"ratioOfGroupMedians": {"denseMedian": old_median, "autoMedian": new_median,
"reductionPercent": 100 * (old_median - new_median) / old_median,
"speedup": old_median / new_median}}
summary["speedComparison"] = {"method": "Alternating serial frozen-external-baseline/production pairs, distinct executables; paired ratios and ratio of group medians are distinct estimates. Warmups and verification excluded.", "externalBaseline": True, "metrics": ratios}
checks = [check for pair in pairs for check in (pair["denseVsAuto"], pair["denseRepeatVsFirstDense"], pair["autoRepeatVsFirstAuto"]) if check is not None]
if summary["verify"] is not None:
checks.append(summary["verify"])
summary["complete"] = True
summary["comparisonCount"] = len(checks)
summary["passed"] = all(check["passed"] for check in checks)
summary["numericalAcceptance"] = ("Exact external-baseline equivalence and repeatability" if summary["passed"] else "Requires separate trajectory/convergence review; no tolerance gate applied") if prepared["externalBaseline"] else "Production repeatability/verification only; no external accuracy comparison"
summary["allPayloadBitsEqual"] = all(check["allPayloadBitsEqual"] for check in checks) if checks else None
save()
except Exception as exc:
summary["errors"].append(f"{type(exc).__name__}: {exc}")
save()
print(summary["errors"][-1], file=sys.stderr, flush=True)
return summary
def main() -> int:
parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
parser.add_argument("--input", type=Path, default=ROOT / "tests/data/test-mql-8-corrected.json")
parser.add_argument("--output-dir", required=True, type=Path)
parser.add_argument("--run", action="store_true", help="Build once and execute; omitted means preparation only")
parser.add_argument("--warmups", type=int, default=1)
parser.add_argument("--repeats", type=int, default=3)
parser.add_argument("--verify-jacobian", "--verify", dest="verify", action="store_true", help="Also execute one full-matrix diagnostic run, excluded from timing statistics")
parser.add_argument("--baseline-executable", type=Path, help="Optional frozen historical executable with default dense strategy; never built or modified by this tool")
parser.add_argument("--record-differences", action="store_true", help="Complete timings while retaining failed strict external-baseline comparisons; does not accept numerical differences")
parser.add_argument("--timeout", type=float, default=120, help="Per-process native timeout; Python allows 10 s exit grace")
if any(arg == "--jacobian" or arg.startswith("--jacobian=") for arg in sys.argv[1:]):
parser.error("--jacobian strategy selection was removed; use the production default or --verify-jacobian. Dense requests require a frozen historical version (benchmark only: --baseline-executable).")
args = parser.parse_args()
if args.warmups < 0 or args.repeats < 1 or not math.isfinite(args.timeout) or args.timeout <= 0:
parser.error("Require warmups >= 0, repeats >= 1, finite timeout > 0")
try:
prepared, program = prepare(args)
except Exception as exc:
print(f"Preparation failed: {type(exc).__name__}: {exc}", file=sys.stderr)
return 2
print(f"Prepared: {args.output_dir.resolve() / 'prepared.json'} (no compilation or solve during preparation)", flush=True)
if not args.run:
return 0
summary = execute(args, prepared, program)
if summary["complete"]:
print(json.dumps(summary["speedComparison"] if summary["speedComparison"] is not None else summary["statistics"]["auto"], ensure_ascii=False, indent=2), flush=True)
return 0 if summary["passed"] or (args.record_differences and summary["complete"]) else 2
if __name__ == "__main__":
raise SystemExit(main())