优化雅可比矩阵计算;端口转发情况下仿真结果传输方式优化
This commit is contained in:
1 parent
3bc4be3c06
commit
aa4951b14e
28 files changed
+8038
-23
No files matched your search
@@ -0,0 +1,522 @@
|
||||
r"""Prepare, then optionally benchmark the production automatic Jacobian.
|
||||
|
||||
.venv/bin/python tests/manual/benchmark_native_jacobian.py \
|
||||
--input tests/data/test-mql-8-corrected.json \
|
||||
--output-dir test/jacobian-production/benchmark --warmups 1 --repeats 3 \
|
||||
--verify-jacobian
|
||||
|
||||
Add --run to build once and execute. Preparation generates C without compiling
|
||||
or solving. --verify-jacobian (--verify alias) adds one separate, untimed-for-
|
||||
statistics full-matrix diagnostic run. Ordinary runs use the production default.
|
||||
|
||||
An optional --baseline-executable must point to a frozen historical executable
|
||||
whose DEFAULT algorithm is the old dense Jacobian. No strategy selector is sent
|
||||
to either executable. The external binary hash and provenance are recorded; if
|
||||
it reports a Jacobian mode, it must report dense-difference. With no external
|
||||
baseline only current-production repeatability is compared. Historical summary
|
||||
keys dense/auto are retained; dense statistics are empty and speedComparison is
|
||||
null when no external baseline was supplied.
|
||||
|
||||
Both executables receive the same time, sampling and tolerance arguments. The
|
||||
caller must select a frozen baseline for the same model and embedded absolute
|
||||
tolerances; recorded provenance and structural checks alone do not prove this.
|
||||
Parsing and comparisons are outside process timing. Exact payload equality and
|
||||
signed-zero bit equality are reported separately, without resampling or tolerance
|
||||
relaxation. --record-differences permits failed external comparisons to be saved
|
||||
for separate trajectory review; repeatability and verification still must pass.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
from array import array
|
||||
from hashlib import sha256
|
||||
import json
|
||||
import math
|
||||
import os
|
||||
from pathlib import Path
|
||||
import platform
|
||||
import statistics
|
||||
import struct
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[2]
|
||||
PAYLOAD = ("series", "final", "finalState")
|
||||
COUNTERS = ("nfev", "acceptedSteps", "rejectedSteps", "njev", "nlu",
|
||||
"stateTransitions", "solverStarts", "jacobianRhsCalls",
|
||||
"jacobianColoredEvals", "jacobianFallbacks", "jacobianChecks",
|
||||
"jacobianMismatches", "cvodeRhsCalls", "cvodeLinearRhsCalls")
|
||||
TIMINGS = ("solveSeconds", "solveCpuSeconds", "processWallSeconds")
|
||||
INVARIANT_METADATA = ("success", "status", "backend", "method", "solver", "sundialsVersion", "simulatedUntil")
|
||||
MAX_DETAILS = 20
|
||||
|
||||
|
||||
def digest(path: Path) -> str:
|
||||
with path.open("rb") as stream:
|
||||
value = sha256()
|
||||
for block in iter(lambda: stream.read(1024 * 1024), b""):
|
||||
value.update(block)
|
||||
return value.hexdigest()
|
||||
|
||||
|
||||
def write_json(path: Path, value: object) -> None:
|
||||
path.write_text(json.dumps(value, ensure_ascii=False, indent=2, allow_nan=False) + "\n", encoding="utf-8")
|
||||
|
||||
|
||||
def pointer(*parts: object) -> str:
|
||||
return "/" + "/".join(str(part).replace("~", "~0").replace("/", "~1") for part in parts)
|
||||
|
||||
|
||||
def read_result(path: Path) -> tuple[dict, str]:
|
||||
def unique(items):
|
||||
result = {}
|
||||
for key, value in items:
|
||||
if key in result:
|
||||
raise ValueError(f"Duplicate JSON object key: {key!r}")
|
||||
result[key] = value
|
||||
return result
|
||||
|
||||
def reject(token):
|
||||
raise ValueError(f"Nonfinite JSON token: {token}")
|
||||
|
||||
def finite_float(token):
|
||||
value = float(token)
|
||||
if not math.isfinite(value):
|
||||
raise ValueError(f"Nonfinite JSON number: {token}")
|
||||
return value
|
||||
|
||||
raw = path.read_bytes()
|
||||
data = json.loads(raw, parse_int=lambda s: -0.0 if s == "-0" else int(s),
|
||||
parse_float=finite_float, parse_constant=reject, object_pairs_hook=unique)
|
||||
if not isinstance(data, dict):
|
||||
raise ValueError("Expected a native result object")
|
||||
return data, sha256(raw).hexdigest()
|
||||
|
||||
|
||||
def number(value, path: str) -> float:
|
||||
if isinstance(value, bool) or not isinstance(value, (int, float)):
|
||||
raise ValueError(f"Expected numeric payload at {path}")
|
||||
try:
|
||||
result = float(value)
|
||||
except (OverflowError, ValueError) as exc:
|
||||
raise ValueError(f"Cannot represent binary64 at {path}") from exc
|
||||
if not math.isfinite(result) or (isinstance(value, int) and int(result) != value):
|
||||
raise ValueError(f"Nonfinite or inexact binary64 at {path}")
|
||||
return result
|
||||
|
||||
|
||||
def blocks(data: dict):
|
||||
for key, values in data["series"].items():
|
||||
yield pointer("series", key), values
|
||||
for key, value in data["final"].items():
|
||||
yield pointer("final", key), [value]
|
||||
yield "/finalState", data["finalState"]
|
||||
|
||||
|
||||
def validate_result(data: dict, prepared: dict, *, external_baseline: bool = False) -> dict:
|
||||
required_counters = COUNTERS[:7] if external_baseline else COUNTERS
|
||||
mode_fields = () if external_baseline else ("jacobianMode",)
|
||||
for key in (*PAYLOAD, *required_counters, *INVARIANT_METADATA, *mode_fields, "maxAcceptedStep", "solveSeconds", "solveCpuSeconds"):
|
||||
if key not in data:
|
||||
raise ValueError(f"Missing result field: {key}")
|
||||
if not isinstance(data["series"], dict) or not isinstance(data["final"], dict) or not isinstance(data["finalState"], list):
|
||||
raise ValueError("Invalid series/final/finalState structure")
|
||||
expected = set(prepared["outputKeys"])
|
||||
if set(data["series"]) != expected | {"time"} or set(data["final"]) != expected:
|
||||
raise ValueError("Result output key sets do not match the generated manifest")
|
||||
if len(data["finalState"]) != prepared["stateCount"]:
|
||||
raise ValueError("finalState length does not match the generated manifest")
|
||||
times = data["series"]["time"]
|
||||
if not isinstance(times, list) or not times:
|
||||
raise ValueError("Full sampled output is required")
|
||||
for key, values in data["series"].items():
|
||||
if not isinstance(values, list) or len(values) != len(times):
|
||||
raise ValueError(f"Series column length differs from time: {key}")
|
||||
for key in COUNTERS:
|
||||
if external_baseline and key not in data:
|
||||
continue
|
||||
if type(data[key]) is not int or data[key] < 0:
|
||||
raise ValueError(f"Invalid nonnegative counter: {key}")
|
||||
if data["success"] is not True or data["status"] != "completed":
|
||||
raise ValueError(f"Native solve did not complete: {data.get('message')}")
|
||||
if (data["backend"], data["method"], data["solver"]) != ("native-c", "BDF", "CVODE"):
|
||||
raise ValueError("Unexpected backend/integrator")
|
||||
for key in ("simulatedUntil", "maxAcceptedStep", "solveSeconds", "solveCpuSeconds"):
|
||||
number(data[key], pointer(key))
|
||||
if data["solveSeconds"] <= 0 or data["solveCpuSeconds"] < 0:
|
||||
raise ValueError("Invalid reported solve duration")
|
||||
cells = negative_zero = positive_zero = 0
|
||||
for path, values in blocks(data):
|
||||
for index, value in enumerate(values):
|
||||
value = number(value, path + "/" + str(index))
|
||||
cells += 1
|
||||
if value == 0:
|
||||
if math.copysign(1.0, value) < 0:
|
||||
negative_zero += 1
|
||||
else:
|
||||
positive_zero += 1
|
||||
cfg = prepared["settings"]
|
||||
if times[0] != cfg["t_start"] or times[-1] != cfg["t_stop"] or data["simulatedUntil"] != cfg["t_stop"]:
|
||||
raise ValueError("Result does not cover the entire requested interval")
|
||||
if any(a >= b for a, b in zip(times, times[1:])):
|
||||
raise ValueError("Sample times must be strictly increasing")
|
||||
if data["maxAcceptedStep"] > cfg["max_step"] * (1 + 1e-14):
|
||||
raise ValueError("Reported accepted step exceeds configured maximum")
|
||||
# Match the runtime's start + index * sample_step arithmetic exactly.
|
||||
regular = []
|
||||
index = 0
|
||||
while (value := cfg["t_start"] + index * prepared["sampleStep"]) <= cfg["t_stop"]:
|
||||
regular.append(value)
|
||||
index += 1
|
||||
actual_times = set(times)
|
||||
missing_regular = [value for value in regular if value not in actual_times]
|
||||
regular_set = set(regular)
|
||||
extra = [value for value in times if value not in regular_set]
|
||||
if missing_regular:
|
||||
raise ValueError(f"Missing regular samples: {missing_regular[:MAX_DETAILS]}")
|
||||
return {"sampleCount": len(times), "seriesColumns": len(data["series"]),
|
||||
"finalScalars": len(data["final"]), "finalStateValues": len(data["finalState"]),
|
||||
"payloadValues": cells, "negativeZeroValues": negative_zero, "positiveZeroValues": positive_zero,
|
||||
"regularSampleCount": len(regular), "extraSampleTimes": extra,
|
||||
"extraSampleInterpretation": "Off-grid saved points, usually events; a non-grid final endpoint may also appear.",
|
||||
"counterAccountingMatches": (data["nfev"] == data["cvodeRhsCalls"] + data["cvodeLinearRhsCalls"] + data["jacobianRhsCalls"]
|
||||
if all(k in data for k in ("cvodeRhsCalls", "cvodeLinearRhsCalls", "jacobianRhsCalls")) else None),
|
||||
"missingHistoricalCounters": [k for k in COUNTERS if k not in data]}
|
||||
|
||||
|
||||
def compare_payload(baseline: dict, candidate: dict) -> dict:
|
||||
"""Exact values and separately exact bits; never compare unaligned series."""
|
||||
structure = []
|
||||
for section in ("series", "final"):
|
||||
left, right = baseline[section], candidate[section]
|
||||
if left.keys() != right.keys():
|
||||
structure.append({"path": pointer(section), "missing": sorted(left.keys() - right.keys()), "extra": sorted(right.keys() - left.keys())})
|
||||
for key in baseline["series"].keys() & candidate["series"].keys():
|
||||
a, b = len(baseline["series"][key]), len(candidate["series"][key])
|
||||
if a != b:
|
||||
structure.append({"path": pointer("series", key), "baselineLength": a, "candidateLength": b})
|
||||
if len(baseline["finalState"]) != len(candidate["finalState"]):
|
||||
structure.append({"path": "/finalState", "baselineLength": len(baseline["finalState"]), "candidateLength": len(candidate["finalState"])})
|
||||
left_times, right_times = baseline["series"]["time"], candidate["series"]["time"]
|
||||
same_times = left_times == right_times
|
||||
time_differences = []
|
||||
for i, (a, b) in enumerate(zip(left_times, right_times)):
|
||||
if a != b:
|
||||
time_differences.append({"index": i, "baseline": a, "candidate": b})
|
||||
if len(time_differences) == MAX_DETAILS:
|
||||
break
|
||||
metadata = [{"key": key, "baseline": baseline.get(key), "candidate": candidate.get(key)}
|
||||
for key in INVARIANT_METADATA if baseline.get(key) != candidate.get(key)]
|
||||
numerical = bit_count = zero_signs = compared = 0
|
||||
details = []
|
||||
compared_blocks = 0
|
||||
pairs = []
|
||||
if same_times:
|
||||
for key in sorted(baseline["series"].keys() & candidate["series"].keys()):
|
||||
pairs.append((pointer("series", key), baseline["series"][key], candidate["series"][key]))
|
||||
for key in sorted(baseline["final"].keys() & candidate["final"].keys()):
|
||||
pairs.append((pointer("final", key), [baseline["final"][key]], [candidate["final"][key]]))
|
||||
pairs.append(("/finalState", baseline["finalState"], candidate["finalState"]))
|
||||
for path, left, right in pairs:
|
||||
if len(left) != len(right):
|
||||
continue
|
||||
compared_blocks += 1
|
||||
compared += len(left)
|
||||
# Fast whole-block bit check; inspect individual cells only on differences.
|
||||
if array("d", left).tobytes() == array("d", right).tobytes():
|
||||
continue
|
||||
for index, (a, b) in enumerate(zip(left, right, strict=True)):
|
||||
packed_a, packed_b = struct.pack("<d", a), struct.pack("<d", b)
|
||||
if packed_a == packed_b:
|
||||
continue
|
||||
bit_count += 1
|
||||
numerical += a != b
|
||||
zero_signs += a == 0 and b == 0
|
||||
if len(details) < MAX_DETAILS:
|
||||
details.append({"path": path if path.startswith("/final/") else path + "/" + str(index), "baseline": a, "candidate": b,
|
||||
"numericallyEqual": a == b, "baselineBitsLE": packed_a.hex(), "candidateBitsLE": packed_b.hex()})
|
||||
complete = not structure and same_times
|
||||
return {"passed": complete and not metadata and numerical == 0,
|
||||
"comparisonContract": "Exact payload numeric equality; signed-zero differences do not fail numeric equality and are reported separately. Solver work counters are observations, not invariants.",
|
||||
"structureEqual": not structure, "structureDifferences": structure[:MAX_DETAILS],
|
||||
"structureDifferenceCount": len(structure), "metadataDifferences": metadata,
|
||||
"timeAxis": {"numericallyEqual": same_times, "baselineLength": len(left_times), "candidateLength": len(right_times),
|
||||
"firstIndexDifferences": time_differences, "seriesCompared": same_times,
|
||||
"interpretation": "Time differences are index diagnostics only; unequal time axes disable series-value comparison. No interpolation or event removal."},
|
||||
"allPayloadCompared": complete, "comparedBlocks": compared_blocks, "comparedValues": compared,
|
||||
"numericallyUnequalValues": numerical, "differentBits": bit_count, "signedZeroDifferences": zero_signs,
|
||||
"allPayloadBitsEqual": complete and bit_count == 0, "firstDifferences": details,
|
||||
"counterDifferences": {key: {"baseline": baseline.get(key), "candidate": candidate.get(key)}
|
||||
for key in COUNTERS if baseline.get(key) != candidate.get(key)},
|
||||
"maxAcceptedStep": {"baseline": baseline.get("maxAcceptedStep"), "candidate": candidate.get("maxAcceptedStep")}}
|
||||
|
||||
|
||||
def stats(values) -> dict:
|
||||
values = list(values)
|
||||
return {"n": len(values), "median": statistics.median(values) if values else None,
|
||||
"min": min(values) if values else None, "max": max(values) if values else None, "values": values}
|
||||
|
||||
|
||||
def prepare(args) -> tuple[dict, object]:
|
||||
sys.path.insert(0, str(ROOT))
|
||||
from app.main import compile_system_xml_network
|
||||
from app.simulation.backends import simulation_config
|
||||
from app.simulation.native_codegen.compiler import compile_native_program
|
||||
from app.simulation.native_codegen.input import load_input
|
||||
from app.simulation.native_codegen.tolerances import state_absolute_tolerance
|
||||
|
||||
output = args.output_dir.resolve()
|
||||
if output == ROOT / "test" or not output.is_relative_to(ROOT / "test"):
|
||||
raise ValueError("Choose an output subdirectory beneath the repository's ignored test/ directory")
|
||||
if any((output / name).exists() for name in ("summary.json", "dense", "auto", "verify")):
|
||||
raise ValueError("Run artifacts already exist; choose a fresh output directory")
|
||||
external = None
|
||||
if args.baseline_executable is not None:
|
||||
executable = args.baseline_executable.resolve()
|
||||
if not executable.is_file() or not os.access(executable, os.X_OK):
|
||||
raise ValueError("--baseline-executable must be an existing executable from a frozen historical version")
|
||||
manifest = executable.parent / "manifest.json"
|
||||
external = {"executable": str(executable), "sha256": digest(executable),
|
||||
"manifestPath": str(manifest) if manifest.is_file() else None,
|
||||
"manifestSha256": digest(manifest) if manifest.is_file() else None,
|
||||
"strategy": "historical executable default; dense-difference required if reported",
|
||||
"modelAndEmbeddedToleranceProvenance": "Caller-supplied frozen model; current manifest checks payload dimensions, not full physical equivalence"}
|
||||
started = time.perf_counter()
|
||||
xml, document = load_input(args.input)
|
||||
cfg = simulation_config(document.simulation)
|
||||
if cfg.method != "BDF" or cfg.rtol != 1e-8 or cfg.atol != 1e-8 or cfg.first_step is not None:
|
||||
raise ValueError("Input must use BDF, production rtol=1e-8, generated atol and automatic first step")
|
||||
step = document.simulation.sample_step
|
||||
if not all(math.isfinite(v) for v in (cfg.t_start, cfg.t_stop, cfg.max_step, step)) or not (cfg.t_stop > cfg.t_start and cfg.max_step > 0 and step > 0):
|
||||
raise ValueError("Invalid finite simulation interval/step settings")
|
||||
if (cfg.t_stop - cfg.t_start) / step > 1000000 or cfg.t_start + step == cfg.t_start:
|
||||
raise ValueError("Invalid or excessive sampling grid")
|
||||
program = compile_native_program(compile_system_xml_network(document))
|
||||
if external and external["manifestPath"]:
|
||||
frozen_manifest = json.loads(Path(external["manifestPath"]).read_text())
|
||||
if frozen_manifest.get("stateKeys") != list(program.state_keys):
|
||||
raise ValueError("Frozen baseline manifest state keys/order differ from the current model")
|
||||
frozen_outputs = {v["key"] for v in frozen_manifest.get("variables", [])}
|
||||
if frozen_outputs != {v.key for v in program.variables}:
|
||||
raise ValueError("Frozen baseline manifest output keys differ from the current model")
|
||||
external["manifestStateAndOutputContractMatched"] = True
|
||||
output.mkdir(parents=True, exist_ok=True)
|
||||
(output / "input.xml").write_bytes(xml)
|
||||
if args.input.suffix.lower() == ".json":
|
||||
(output / "input.json").write_bytes(args.input.read_bytes())
|
||||
(output / "model.c").write_text(program.source, encoding="utf-8")
|
||||
(output / "model.h").write_text(program.header, encoding="utf-8")
|
||||
write_json(output / "model-contract.json", program.manifest())
|
||||
source_paths = sorted((ROOT / "native").rglob("*.c")) + sorted((ROOT / "native").rglob("*.h")) + sorted((ROOT / "app/simulation/native_codegen").glob("*.py"))
|
||||
prepared = {"schemaVersion": 1, "preparedOnly": not args.run, "input": str(args.input.resolve()),
|
||||
"inputSha256": digest(args.input), "xmlSha256": sha256(xml).hexdigest(),
|
||||
"scriptSha256": digest(Path(__file__)), "modelSourceSha256": digest(output / "model.c"),
|
||||
"modelHeaderSha256": digest(output / "model.h"), "settings": vars(cfg), "sampleStep": step,
|
||||
"stateCount": len(program.state_keys), "stateKeys": list(program.state_keys),
|
||||
"stateAbsoluteTolerances": [float(state_absolute_tolerance(k)) for k in program.state_keys],
|
||||
"outputKeys": [v.key for v in program.variables], "modelContract": program.manifest(),
|
||||
"sourceHashes": {str(p.relative_to(ROOT)): digest(p) for p in source_paths},
|
||||
"environment": {"platform": platform.platform(), "python": sys.version, "machine": platform.machine()},
|
||||
"warmupsPerMode": args.warmups, "repeatsPerMode": args.repeats,
|
||||
"algorithm": "production-automatic", "externalBaseline": external,
|
||||
"activeModes": ["dense", "auto"] if external else ["auto"],
|
||||
"verifyRequested": args.verify, "recordDifferences": args.record_differences, "nativeTimeoutSeconds": args.timeout,
|
||||
"processTimeoutSeconds": args.timeout + 10, "preparationSeconds": time.perf_counter() - started,
|
||||
"timingContract": "Production automatic executable plus an optional caller-supplied frozen external baseline; complete sampled output. C-reported solve wall/CPU and subprocess creation-through-reap wall only. Build, parsing, validation and comparisons excluded. No independently measured write/projection stage; process-minus-solve is not called write time. Ordinary file writes, no fsync.",
|
||||
"comparisonContract": "Exact structure, numeric values and time axis for series/final/finalState. Bits, including signed zero, are separately reported. No loosened tolerance or interpolation. Counters may differ.",
|
||||
"pairOrder": [{"pair": i + 1, "warmup": i < args.warmups,
|
||||
"modes": (["dense", "auto"] if i % 2 == 0 else ["auto", "dense"]) if external else ["auto"]}
|
||||
for i in range(args.warmups + args.repeats)]}
|
||||
write_json(output / "prepared.json", prepared)
|
||||
return prepared, program
|
||||
|
||||
|
||||
def execute(args, prepared: dict, program) -> dict:
|
||||
from app.simulation.native_codegen.build import build_native
|
||||
|
||||
output = args.output_dir.resolve()
|
||||
summary = {"schemaVersion": 1, "complete": False, "passed": False, "prepared": prepared,
|
||||
"errors": [], "rows": [], "pairs": [], "verify": None, "statistics": None,
|
||||
"externalBaseline": prepared["externalBaseline"],
|
||||
"speedComparison": None, "strictFailureMeans": "Exact equivalence was not established; retain outputs for independent convergence diagnostics. No automatic acceptance-tolerance change."}
|
||||
rows, pairs = summary["rows"], summary["pairs"]
|
||||
|
||||
def save():
|
||||
write_json(output / "summary.json", summary)
|
||||
|
||||
def run(mode: str, label: str, *, pair=None, warmup=False, diagnostic=False):
|
||||
directory = output / mode / label
|
||||
directory.mkdir(parents=True, exist_ok=False)
|
||||
cfg = prepared["settings"]
|
||||
executable = Path(prepared["externalBaseline"]["executable"]) if mode == "dense" else build.executable
|
||||
command = [str(executable), "--method", "BDF",
|
||||
"--start", str(cfg["t_start"]), "--stop", str(cfg["t_stop"]),
|
||||
"--sample-step", str(prepared["sampleStep"]), "--max-step", str(cfg["max_step"]),
|
||||
"--rtol", "1e-8", "--timeout", str(args.timeout),
|
||||
"--output", str(directory / "result.json"), "--result-index", str(directory / "result-index.json"),
|
||||
"--cancel-file", str(directory / "cancel.request")]
|
||||
if diagnostic:
|
||||
command.append("--verify-jacobian")
|
||||
row = {"mode": mode, "label": label, "pair": pair, "warmup": warmup, "diagnostic": diagnostic,
|
||||
"includedInStatistics": not warmup and not diagnostic, "directory": str(directory), "command": command,
|
||||
"completed": False, "resultValidation": None, "externalBaseline": mode == "dense"}
|
||||
rows.append(row)
|
||||
write_json(directory / "command.json", command)
|
||||
environment = dict(os.environ)
|
||||
for key in ("NATIVE_COMPUTE_PROFILE", "NATIVE_STAGE_PROFILE"):
|
||||
environment.pop(key, None)
|
||||
try:
|
||||
with (directory / "stdout.log").open("wb") as stdout, (directory / "stderr.log").open("wb") as stderr:
|
||||
started = time.perf_counter()
|
||||
try:
|
||||
process = subprocess.run(command, cwd=executable.parent, env=environment, stdin=subprocess.DEVNULL,
|
||||
stdout=stdout, stderr=stderr, timeout=args.timeout + 10)
|
||||
row["exitCode"] = process.returncode
|
||||
finally:
|
||||
row["processWallSeconds"] = time.perf_counter() - started
|
||||
result_path = directory / "result.json"
|
||||
if not result_path.is_file():
|
||||
raise RuntimeError(f"No native result: {directory}")
|
||||
data, result_hash = read_result(result_path)
|
||||
row.update({key: value for key, value in data.items() if key not in PAYLOAD})
|
||||
row["resultSha256"] = result_hash
|
||||
row["resultBytes"] = result_path.stat().st_size
|
||||
row["resultValidation"] = validate_result(data, prepared, external_baseline=mode == "dense")
|
||||
if row["exitCode"] != 0:
|
||||
raise RuntimeError(f"Native exit code {row['exitCode']}: {directory}")
|
||||
if row["resultValidation"]["counterAccountingMatches"] is False:
|
||||
raise RuntimeError(f"RHS counter accounting failed: {directory}")
|
||||
if mode == "dense" and data.get("jacobianMode") not in (None, "dense-difference"):
|
||||
raise RuntimeError("External baseline is not a frozen default-dense executable; current production cannot emulate the old strategy")
|
||||
if mode == "dense":
|
||||
row["historicalStrategyEvidence"] = "reported-dense" if data.get("jacobianMode") else "unreported: caller-supplied frozen provenance"
|
||||
if diagnostic and (data["jacobianChecks"] <= 0 or data["jacobianMismatches"] != 0 or
|
||||
data["jacobianChecks"] != data["jacobianColoredEvals"]):
|
||||
raise RuntimeError("Verify mode did not validate every computed colored Jacobian without mismatches")
|
||||
row["completed"] = True
|
||||
print(f"{mode}/{label}: solve={row['solveSeconds']:.6f}s process={row['processWallSeconds']:.6f}s nfev={row['nfev']}", flush=True)
|
||||
return data
|
||||
except Exception as exc:
|
||||
row["error"] = f"{type(exc).__name__}: {exc}"
|
||||
raise
|
||||
finally:
|
||||
write_json(directory / "run.json", row)
|
||||
save()
|
||||
|
||||
try:
|
||||
# Build current production once. An optional baseline is never built or modified.
|
||||
build = build_native(program, cache_dir=output / "cache")
|
||||
summary["build"] = {"executable": str(build.executable), "buildKey": build.manifest["buildKey"],
|
||||
"cacheHit": build.cache_hit, "seconds": build.seconds, "manifest": build.manifest}
|
||||
if prepared["externalBaseline"]:
|
||||
if digest(Path(prepared["externalBaseline"]["executable"])) != prepared["externalBaseline"]["sha256"]:
|
||||
raise RuntimeError("Frozen external baseline changed after preparation")
|
||||
if digest(build.executable) == prepared["externalBaseline"]["sha256"]:
|
||||
raise RuntimeError("External baseline equals the current production executable")
|
||||
save()
|
||||
verify_data = run("verify", "validation", diagnostic=True) if args.verify else None
|
||||
first_auto = first_dense = None
|
||||
warmup_count = measured_count = 0
|
||||
for planned in prepared["pairOrder"]:
|
||||
is_warmup = planned["warmup"]
|
||||
if is_warmup:
|
||||
warmup_count += 1
|
||||
label = f"warmup-{warmup_count}"
|
||||
else:
|
||||
measured_count += 1
|
||||
label = f"run-{measured_count}"
|
||||
results = {mode: run(mode, label, pair=planned["pair"], warmup=is_warmup) for mode in planned["modes"]}
|
||||
comparison = compare_payload(results["dense"], results["auto"]) if "dense" in results else None
|
||||
auto_repeat = compare_payload(first_auto, results["auto"]) if first_auto is not None else None
|
||||
dense_repeat = compare_payload(first_dense, results["dense"]) if first_dense is not None else None
|
||||
if first_auto is None:
|
||||
first_auto = results["auto"]
|
||||
first_dense = results.get("dense")
|
||||
if verify_data is not None:
|
||||
summary["verify"] = compare_payload(results["auto"], verify_data)
|
||||
summary["verify"]["modes"] = "production default versus --verify-jacobian: diagnostic must preserve trajectory"
|
||||
verify_data = None
|
||||
record = {"pair": planned["pair"], "label": label, "warmup": is_warmup, "order": planned["modes"],
|
||||
"externalBaseline": prepared["externalBaseline"] is not None,
|
||||
"denseVsAuto": comparison, "denseRepeatVsFirstDense": dense_repeat,
|
||||
"autoRepeatVsFirstAuto": auto_repeat}
|
||||
pairs.append(record)
|
||||
write_json(output / f"comparison-{label}.json", record)
|
||||
save()
|
||||
if any(check is not None and not check["passed"] for check in (auto_repeat, dense_repeat, summary["verify"])):
|
||||
raise RuntimeError(f"Within-algorithm or default/verify reproducibility failed at {label}")
|
||||
if comparison is not None and not comparison["passed"] and not args.record_differences:
|
||||
raise RuntimeError(f"Strict external-baseline comparison failed at {label}; raw artifacts retained for convergence diagnostics")
|
||||
summary["statistics"] = {mode: {key: stats(row[key] for row in rows if row["mode"] == mode and row["includedInStatistics"] and key in row)
|
||||
for key in (*TIMINGS, *COUNTERS, "resultBytes")} for mode in ("dense", "auto")}
|
||||
if prepared["externalBaseline"]:
|
||||
ratios = {}
|
||||
for metric in TIMINGS:
|
||||
matched = []
|
||||
for pair in pairs:
|
||||
if pair["warmup"]:
|
||||
continue
|
||||
selected = {row["mode"]: row for row in rows if row["pair"] == pair["pair"]}
|
||||
old, new = selected["dense"][metric], selected["auto"][metric]
|
||||
if old <= 0 or new <= 0:
|
||||
raise RuntimeError(f"Cannot form a positive-duration comparison for {metric}")
|
||||
matched.append({"pair": pair["pair"], "dense": old, "auto": new,
|
||||
"reductionPercent": 100 * (old - new) / old, "speedup": old / new})
|
||||
old_median = summary["statistics"]["dense"][metric]["median"]
|
||||
new_median = summary["statistics"]["auto"][metric]["median"]
|
||||
ratios[metric] = {"pairs": matched,
|
||||
"pairedReductionPercent": stats(p["reductionPercent"] for p in matched),
|
||||
"pairedSpeedup": stats(p["speedup"] for p in matched),
|
||||
"ratioOfGroupMedians": {"denseMedian": old_median, "autoMedian": new_median,
|
||||
"reductionPercent": 100 * (old_median - new_median) / old_median,
|
||||
"speedup": old_median / new_median}}
|
||||
summary["speedComparison"] = {"method": "Alternating serial frozen-external-baseline/production pairs, distinct executables; paired ratios and ratio of group medians are distinct estimates. Warmups and verification excluded.", "externalBaseline": True, "metrics": ratios}
|
||||
checks = [check for pair in pairs for check in (pair["denseVsAuto"], pair["denseRepeatVsFirstDense"], pair["autoRepeatVsFirstAuto"]) if check is not None]
|
||||
if summary["verify"] is not None:
|
||||
checks.append(summary["verify"])
|
||||
summary["complete"] = True
|
||||
summary["comparisonCount"] = len(checks)
|
||||
summary["passed"] = all(check["passed"] for check in checks)
|
||||
summary["numericalAcceptance"] = ("Exact external-baseline equivalence and repeatability" if summary["passed"] else "Requires separate trajectory/convergence review; no tolerance gate applied") if prepared["externalBaseline"] else "Production repeatability/verification only; no external accuracy comparison"
|
||||
summary["allPayloadBitsEqual"] = all(check["allPayloadBitsEqual"] for check in checks) if checks else None
|
||||
save()
|
||||
except Exception as exc:
|
||||
summary["errors"].append(f"{type(exc).__name__}: {exc}")
|
||||
save()
|
||||
print(summary["errors"][-1], file=sys.stderr, flush=True)
|
||||
return summary
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
|
||||
parser.add_argument("--input", type=Path, default=ROOT / "tests/data/test-mql-8-corrected.json")
|
||||
parser.add_argument("--output-dir", required=True, type=Path)
|
||||
parser.add_argument("--run", action="store_true", help="Build once and execute; omitted means preparation only")
|
||||
parser.add_argument("--warmups", type=int, default=1)
|
||||
parser.add_argument("--repeats", type=int, default=3)
|
||||
parser.add_argument("--verify-jacobian", "--verify", dest="verify", action="store_true", help="Also execute one full-matrix diagnostic run, excluded from timing statistics")
|
||||
parser.add_argument("--baseline-executable", type=Path, help="Optional frozen historical executable with default dense strategy; never built or modified by this tool")
|
||||
parser.add_argument("--record-differences", action="store_true", help="Complete timings while retaining failed strict external-baseline comparisons; does not accept numerical differences")
|
||||
parser.add_argument("--timeout", type=float, default=120, help="Per-process native timeout; Python allows 10 s exit grace")
|
||||
if any(arg == "--jacobian" or arg.startswith("--jacobian=") for arg in sys.argv[1:]):
|
||||
parser.error("--jacobian strategy selection was removed; use the production default or --verify-jacobian. Dense requests require a frozen historical version (benchmark only: --baseline-executable).")
|
||||
args = parser.parse_args()
|
||||
if args.warmups < 0 or args.repeats < 1 or not math.isfinite(args.timeout) or args.timeout <= 0:
|
||||
parser.error("Require warmups >= 0, repeats >= 1, finite timeout > 0")
|
||||
try:
|
||||
prepared, program = prepare(args)
|
||||
except Exception as exc:
|
||||
print(f"Preparation failed: {type(exc).__name__}: {exc}", file=sys.stderr)
|
||||
return 2
|
||||
print(f"Prepared: {args.output_dir.resolve() / 'prepared.json'} (no compilation or solve during preparation)", flush=True)
|
||||
if not args.run:
|
||||
return 0
|
||||
summary = execute(args, prepared, program)
|
||||
if summary["complete"]:
|
||||
print(json.dumps(summary["speedComparison"] if summary["speedComparison"] is not None else summary["statistics"]["auto"], ensure_ascii=False, indent=2), flush=True)
|
||||
return 0 if summary["passed"] or (args.record_differences and summary["complete"]) else 2
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
Reference in new issue
Block a user