优化雅可比矩阵计算;端口转发情况下仿真结果传输方式优化
This commit is contained in:
1 parent
3bc4be3c06
commit
aa4951b14e
28 files changed
+8038
-23
No files matched your search
@@ -0,0 +1,522 @@
|
||||
r"""Prepare, then optionally benchmark the production automatic Jacobian.
|
||||
|
||||
.venv/bin/python tests/manual/benchmark_native_jacobian.py \
|
||||
--input tests/data/test-mql-8-corrected.json \
|
||||
--output-dir test/jacobian-production/benchmark --warmups 1 --repeats 3 \
|
||||
--verify-jacobian
|
||||
|
||||
Add --run to build once and execute. Preparation generates C without compiling
|
||||
or solving. --verify-jacobian (--verify alias) adds one separate, untimed-for-
|
||||
statistics full-matrix diagnostic run. Ordinary runs use the production default.
|
||||
|
||||
An optional --baseline-executable must point to a frozen historical executable
|
||||
whose DEFAULT algorithm is the old dense Jacobian. No strategy selector is sent
|
||||
to either executable. The external binary hash and provenance are recorded; if
|
||||
it reports a Jacobian mode, it must report dense-difference. With no external
|
||||
baseline only current-production repeatability is compared. Historical summary
|
||||
keys dense/auto are retained; dense statistics are empty and speedComparison is
|
||||
null when no external baseline was supplied.
|
||||
|
||||
Both executables receive the same time, sampling and tolerance arguments. The
|
||||
caller must select a frozen baseline for the same model and embedded absolute
|
||||
tolerances; recorded provenance and structural checks alone do not prove this.
|
||||
Parsing and comparisons are outside process timing. Exact payload equality and
|
||||
signed-zero bit equality are reported separately, without resampling or tolerance
|
||||
relaxation. --record-differences permits failed external comparisons to be saved
|
||||
for separate trajectory review; repeatability and verification still must pass.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
from array import array
|
||||
from hashlib import sha256
|
||||
import json
|
||||
import math
|
||||
import os
|
||||
from pathlib import Path
|
||||
import platform
|
||||
import statistics
|
||||
import struct
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[2]
|
||||
PAYLOAD = ("series", "final", "finalState")
|
||||
COUNTERS = ("nfev", "acceptedSteps", "rejectedSteps", "njev", "nlu",
|
||||
"stateTransitions", "solverStarts", "jacobianRhsCalls",
|
||||
"jacobianColoredEvals", "jacobianFallbacks", "jacobianChecks",
|
||||
"jacobianMismatches", "cvodeRhsCalls", "cvodeLinearRhsCalls")
|
||||
TIMINGS = ("solveSeconds", "solveCpuSeconds", "processWallSeconds")
|
||||
INVARIANT_METADATA = ("success", "status", "backend", "method", "solver", "sundialsVersion", "simulatedUntil")
|
||||
MAX_DETAILS = 20
|
||||
|
||||
|
||||
def digest(path: Path) -> str:
|
||||
with path.open("rb") as stream:
|
||||
value = sha256()
|
||||
for block in iter(lambda: stream.read(1024 * 1024), b""):
|
||||
value.update(block)
|
||||
return value.hexdigest()
|
||||
|
||||
|
||||
def write_json(path: Path, value: object) -> None:
|
||||
path.write_text(json.dumps(value, ensure_ascii=False, indent=2, allow_nan=False) + "\n", encoding="utf-8")
|
||||
|
||||
|
||||
def pointer(*parts: object) -> str:
|
||||
return "/" + "/".join(str(part).replace("~", "~0").replace("/", "~1") for part in parts)
|
||||
|
||||
|
||||
def read_result(path: Path) -> tuple[dict, str]:
|
||||
def unique(items):
|
||||
result = {}
|
||||
for key, value in items:
|
||||
if key in result:
|
||||
raise ValueError(f"Duplicate JSON object key: {key!r}")
|
||||
result[key] = value
|
||||
return result
|
||||
|
||||
def reject(token):
|
||||
raise ValueError(f"Nonfinite JSON token: {token}")
|
||||
|
||||
def finite_float(token):
|
||||
value = float(token)
|
||||
if not math.isfinite(value):
|
||||
raise ValueError(f"Nonfinite JSON number: {token}")
|
||||
return value
|
||||
|
||||
raw = path.read_bytes()
|
||||
data = json.loads(raw, parse_int=lambda s: -0.0 if s == "-0" else int(s),
|
||||
parse_float=finite_float, parse_constant=reject, object_pairs_hook=unique)
|
||||
if not isinstance(data, dict):
|
||||
raise ValueError("Expected a native result object")
|
||||
return data, sha256(raw).hexdigest()
|
||||
|
||||
|
||||
def number(value, path: str) -> float:
|
||||
if isinstance(value, bool) or not isinstance(value, (int, float)):
|
||||
raise ValueError(f"Expected numeric payload at {path}")
|
||||
try:
|
||||
result = float(value)
|
||||
except (OverflowError, ValueError) as exc:
|
||||
raise ValueError(f"Cannot represent binary64 at {path}") from exc
|
||||
if not math.isfinite(result) or (isinstance(value, int) and int(result) != value):
|
||||
raise ValueError(f"Nonfinite or inexact binary64 at {path}")
|
||||
return result
|
||||
|
||||
|
||||
def blocks(data: dict):
|
||||
for key, values in data["series"].items():
|
||||
yield pointer("series", key), values
|
||||
for key, value in data["final"].items():
|
||||
yield pointer("final", key), [value]
|
||||
yield "/finalState", data["finalState"]
|
||||
|
||||
|
||||
def validate_result(data: dict, prepared: dict, *, external_baseline: bool = False) -> dict:
|
||||
required_counters = COUNTERS[:7] if external_baseline else COUNTERS
|
||||
mode_fields = () if external_baseline else ("jacobianMode",)
|
||||
for key in (*PAYLOAD, *required_counters, *INVARIANT_METADATA, *mode_fields, "maxAcceptedStep", "solveSeconds", "solveCpuSeconds"):
|
||||
if key not in data:
|
||||
raise ValueError(f"Missing result field: {key}")
|
||||
if not isinstance(data["series"], dict) or not isinstance(data["final"], dict) or not isinstance(data["finalState"], list):
|
||||
raise ValueError("Invalid series/final/finalState structure")
|
||||
expected = set(prepared["outputKeys"])
|
||||
if set(data["series"]) != expected | {"time"} or set(data["final"]) != expected:
|
||||
raise ValueError("Result output key sets do not match the generated manifest")
|
||||
if len(data["finalState"]) != prepared["stateCount"]:
|
||||
raise ValueError("finalState length does not match the generated manifest")
|
||||
times = data["series"]["time"]
|
||||
if not isinstance(times, list) or not times:
|
||||
raise ValueError("Full sampled output is required")
|
||||
for key, values in data["series"].items():
|
||||
if not isinstance(values, list) or len(values) != len(times):
|
||||
raise ValueError(f"Series column length differs from time: {key}")
|
||||
for key in COUNTERS:
|
||||
if external_baseline and key not in data:
|
||||
continue
|
||||
if type(data[key]) is not int or data[key] < 0:
|
||||
raise ValueError(f"Invalid nonnegative counter: {key}")
|
||||
if data["success"] is not True or data["status"] != "completed":
|
||||
raise ValueError(f"Native solve did not complete: {data.get('message')}")
|
||||
if (data["backend"], data["method"], data["solver"]) != ("native-c", "BDF", "CVODE"):
|
||||
raise ValueError("Unexpected backend/integrator")
|
||||
for key in ("simulatedUntil", "maxAcceptedStep", "solveSeconds", "solveCpuSeconds"):
|
||||
number(data[key], pointer(key))
|
||||
if data["solveSeconds"] <= 0 or data["solveCpuSeconds"] < 0:
|
||||
raise ValueError("Invalid reported solve duration")
|
||||
cells = negative_zero = positive_zero = 0
|
||||
for path, values in blocks(data):
|
||||
for index, value in enumerate(values):
|
||||
value = number(value, path + "/" + str(index))
|
||||
cells += 1
|
||||
if value == 0:
|
||||
if math.copysign(1.0, value) < 0:
|
||||
negative_zero += 1
|
||||
else:
|
||||
positive_zero += 1
|
||||
cfg = prepared["settings"]
|
||||
if times[0] != cfg["t_start"] or times[-1] != cfg["t_stop"] or data["simulatedUntil"] != cfg["t_stop"]:
|
||||
raise ValueError("Result does not cover the entire requested interval")
|
||||
if any(a >= b for a, b in zip(times, times[1:])):
|
||||
raise ValueError("Sample times must be strictly increasing")
|
||||
if data["maxAcceptedStep"] > cfg["max_step"] * (1 + 1e-14):
|
||||
raise ValueError("Reported accepted step exceeds configured maximum")
|
||||
# Match the runtime's start + index * sample_step arithmetic exactly.
|
||||
regular = []
|
||||
index = 0
|
||||
while (value := cfg["t_start"] + index * prepared["sampleStep"]) <= cfg["t_stop"]:
|
||||
regular.append(value)
|
||||
index += 1
|
||||
actual_times = set(times)
|
||||
missing_regular = [value for value in regular if value not in actual_times]
|
||||
regular_set = set(regular)
|
||||
extra = [value for value in times if value not in regular_set]
|
||||
if missing_regular:
|
||||
raise ValueError(f"Missing regular samples: {missing_regular[:MAX_DETAILS]}")
|
||||
return {"sampleCount": len(times), "seriesColumns": len(data["series"]),
|
||||
"finalScalars": len(data["final"]), "finalStateValues": len(data["finalState"]),
|
||||
"payloadValues": cells, "negativeZeroValues": negative_zero, "positiveZeroValues": positive_zero,
|
||||
"regularSampleCount": len(regular), "extraSampleTimes": extra,
|
||||
"extraSampleInterpretation": "Off-grid saved points, usually events; a non-grid final endpoint may also appear.",
|
||||
"counterAccountingMatches": (data["nfev"] == data["cvodeRhsCalls"] + data["cvodeLinearRhsCalls"] + data["jacobianRhsCalls"]
|
||||
if all(k in data for k in ("cvodeRhsCalls", "cvodeLinearRhsCalls", "jacobianRhsCalls")) else None),
|
||||
"missingHistoricalCounters": [k for k in COUNTERS if k not in data]}
|
||||
|
||||
|
||||
def compare_payload(baseline: dict, candidate: dict) -> dict:
|
||||
"""Exact values and separately exact bits; never compare unaligned series."""
|
||||
structure = []
|
||||
for section in ("series", "final"):
|
||||
left, right = baseline[section], candidate[section]
|
||||
if left.keys() != right.keys():
|
||||
structure.append({"path": pointer(section), "missing": sorted(left.keys() - right.keys()), "extra": sorted(right.keys() - left.keys())})
|
||||
for key in baseline["series"].keys() & candidate["series"].keys():
|
||||
a, b = len(baseline["series"][key]), len(candidate["series"][key])
|
||||
if a != b:
|
||||
structure.append({"path": pointer("series", key), "baselineLength": a, "candidateLength": b})
|
||||
if len(baseline["finalState"]) != len(candidate["finalState"]):
|
||||
structure.append({"path": "/finalState", "baselineLength": len(baseline["finalState"]), "candidateLength": len(candidate["finalState"])})
|
||||
left_times, right_times = baseline["series"]["time"], candidate["series"]["time"]
|
||||
same_times = left_times == right_times
|
||||
time_differences = []
|
||||
for i, (a, b) in enumerate(zip(left_times, right_times)):
|
||||
if a != b:
|
||||
time_differences.append({"index": i, "baseline": a, "candidate": b})
|
||||
if len(time_differences) == MAX_DETAILS:
|
||||
break
|
||||
metadata = [{"key": key, "baseline": baseline.get(key), "candidate": candidate.get(key)}
|
||||
for key in INVARIANT_METADATA if baseline.get(key) != candidate.get(key)]
|
||||
numerical = bit_count = zero_signs = compared = 0
|
||||
details = []
|
||||
compared_blocks = 0
|
||||
pairs = []
|
||||
if same_times:
|
||||
for key in sorted(baseline["series"].keys() & candidate["series"].keys()):
|
||||
pairs.append((pointer("series", key), baseline["series"][key], candidate["series"][key]))
|
||||
for key in sorted(baseline["final"].keys() & candidate["final"].keys()):
|
||||
pairs.append((pointer("final", key), [baseline["final"][key]], [candidate["final"][key]]))
|
||||
pairs.append(("/finalState", baseline["finalState"], candidate["finalState"]))
|
||||
for path, left, right in pairs:
|
||||
if len(left) != len(right):
|
||||
continue
|
||||
compared_blocks += 1
|
||||
compared += len(left)
|
||||
# Fast whole-block bit check; inspect individual cells only on differences.
|
||||
if array("d", left).tobytes() == array("d", right).tobytes():
|
||||
continue
|
||||
for index, (a, b) in enumerate(zip(left, right, strict=True)):
|
||||
packed_a, packed_b = struct.pack("<d", a), struct.pack("<d", b)
|
||||
if packed_a == packed_b:
|
||||
continue
|
||||
bit_count += 1
|
||||
numerical += a != b
|
||||
zero_signs += a == 0 and b == 0
|
||||
if len(details) < MAX_DETAILS:
|
||||
details.append({"path": path if path.startswith("/final/") else path + "/" + str(index), "baseline": a, "candidate": b,
|
||||
"numericallyEqual": a == b, "baselineBitsLE": packed_a.hex(), "candidateBitsLE": packed_b.hex()})
|
||||
complete = not structure and same_times
|
||||
return {"passed": complete and not metadata and numerical == 0,
|
||||
"comparisonContract": "Exact payload numeric equality; signed-zero differences do not fail numeric equality and are reported separately. Solver work counters are observations, not invariants.",
|
||||
"structureEqual": not structure, "structureDifferences": structure[:MAX_DETAILS],
|
||||
"structureDifferenceCount": len(structure), "metadataDifferences": metadata,
|
||||
"timeAxis": {"numericallyEqual": same_times, "baselineLength": len(left_times), "candidateLength": len(right_times),
|
||||
"firstIndexDifferences": time_differences, "seriesCompared": same_times,
|
||||
"interpretation": "Time differences are index diagnostics only; unequal time axes disable series-value comparison. No interpolation or event removal."},
|
||||
"allPayloadCompared": complete, "comparedBlocks": compared_blocks, "comparedValues": compared,
|
||||
"numericallyUnequalValues": numerical, "differentBits": bit_count, "signedZeroDifferences": zero_signs,
|
||||
"allPayloadBitsEqual": complete and bit_count == 0, "firstDifferences": details,
|
||||
"counterDifferences": {key: {"baseline": baseline.get(key), "candidate": candidate.get(key)}
|
||||
for key in COUNTERS if baseline.get(key) != candidate.get(key)},
|
||||
"maxAcceptedStep": {"baseline": baseline.get("maxAcceptedStep"), "candidate": candidate.get("maxAcceptedStep")}}
|
||||
|
||||
|
||||
def stats(values) -> dict:
|
||||
values = list(values)
|
||||
return {"n": len(values), "median": statistics.median(values) if values else None,
|
||||
"min": min(values) if values else None, "max": max(values) if values else None, "values": values}
|
||||
|
||||
|
||||
def prepare(args) -> tuple[dict, object]:
|
||||
sys.path.insert(0, str(ROOT))
|
||||
from app.main import compile_system_xml_network
|
||||
from app.simulation.backends import simulation_config
|
||||
from app.simulation.native_codegen.compiler import compile_native_program
|
||||
from app.simulation.native_codegen.input import load_input
|
||||
from app.simulation.native_codegen.tolerances import state_absolute_tolerance
|
||||
|
||||
output = args.output_dir.resolve()
|
||||
if output == ROOT / "test" or not output.is_relative_to(ROOT / "test"):
|
||||
raise ValueError("Choose an output subdirectory beneath the repository's ignored test/ directory")
|
||||
if any((output / name).exists() for name in ("summary.json", "dense", "auto", "verify")):
|
||||
raise ValueError("Run artifacts already exist; choose a fresh output directory")
|
||||
external = None
|
||||
if args.baseline_executable is not None:
|
||||
executable = args.baseline_executable.resolve()
|
||||
if not executable.is_file() or not os.access(executable, os.X_OK):
|
||||
raise ValueError("--baseline-executable must be an existing executable from a frozen historical version")
|
||||
manifest = executable.parent / "manifest.json"
|
||||
external = {"executable": str(executable), "sha256": digest(executable),
|
||||
"manifestPath": str(manifest) if manifest.is_file() else None,
|
||||
"manifestSha256": digest(manifest) if manifest.is_file() else None,
|
||||
"strategy": "historical executable default; dense-difference required if reported",
|
||||
"modelAndEmbeddedToleranceProvenance": "Caller-supplied frozen model; current manifest checks payload dimensions, not full physical equivalence"}
|
||||
started = time.perf_counter()
|
||||
xml, document = load_input(args.input)
|
||||
cfg = simulation_config(document.simulation)
|
||||
if cfg.method != "BDF" or cfg.rtol != 1e-8 or cfg.atol != 1e-8 or cfg.first_step is not None:
|
||||
raise ValueError("Input must use BDF, production rtol=1e-8, generated atol and automatic first step")
|
||||
step = document.simulation.sample_step
|
||||
if not all(math.isfinite(v) for v in (cfg.t_start, cfg.t_stop, cfg.max_step, step)) or not (cfg.t_stop > cfg.t_start and cfg.max_step > 0 and step > 0):
|
||||
raise ValueError("Invalid finite simulation interval/step settings")
|
||||
if (cfg.t_stop - cfg.t_start) / step > 1000000 or cfg.t_start + step == cfg.t_start:
|
||||
raise ValueError("Invalid or excessive sampling grid")
|
||||
program = compile_native_program(compile_system_xml_network(document))
|
||||
if external and external["manifestPath"]:
|
||||
frozen_manifest = json.loads(Path(external["manifestPath"]).read_text())
|
||||
if frozen_manifest.get("stateKeys") != list(program.state_keys):
|
||||
raise ValueError("Frozen baseline manifest state keys/order differ from the current model")
|
||||
frozen_outputs = {v["key"] for v in frozen_manifest.get("variables", [])}
|
||||
if frozen_outputs != {v.key for v in program.variables}:
|
||||
raise ValueError("Frozen baseline manifest output keys differ from the current model")
|
||||
external["manifestStateAndOutputContractMatched"] = True
|
||||
output.mkdir(parents=True, exist_ok=True)
|
||||
(output / "input.xml").write_bytes(xml)
|
||||
if args.input.suffix.lower() == ".json":
|
||||
(output / "input.json").write_bytes(args.input.read_bytes())
|
||||
(output / "model.c").write_text(program.source, encoding="utf-8")
|
||||
(output / "model.h").write_text(program.header, encoding="utf-8")
|
||||
write_json(output / "model-contract.json", program.manifest())
|
||||
source_paths = sorted((ROOT / "native").rglob("*.c")) + sorted((ROOT / "native").rglob("*.h")) + sorted((ROOT / "app/simulation/native_codegen").glob("*.py"))
|
||||
prepared = {"schemaVersion": 1, "preparedOnly": not args.run, "input": str(args.input.resolve()),
|
||||
"inputSha256": digest(args.input), "xmlSha256": sha256(xml).hexdigest(),
|
||||
"scriptSha256": digest(Path(__file__)), "modelSourceSha256": digest(output / "model.c"),
|
||||
"modelHeaderSha256": digest(output / "model.h"), "settings": vars(cfg), "sampleStep": step,
|
||||
"stateCount": len(program.state_keys), "stateKeys": list(program.state_keys),
|
||||
"stateAbsoluteTolerances": [float(state_absolute_tolerance(k)) for k in program.state_keys],
|
||||
"outputKeys": [v.key for v in program.variables], "modelContract": program.manifest(),
|
||||
"sourceHashes": {str(p.relative_to(ROOT)): digest(p) for p in source_paths},
|
||||
"environment": {"platform": platform.platform(), "python": sys.version, "machine": platform.machine()},
|
||||
"warmupsPerMode": args.warmups, "repeatsPerMode": args.repeats,
|
||||
"algorithm": "production-automatic", "externalBaseline": external,
|
||||
"activeModes": ["dense", "auto"] if external else ["auto"],
|
||||
"verifyRequested": args.verify, "recordDifferences": args.record_differences, "nativeTimeoutSeconds": args.timeout,
|
||||
"processTimeoutSeconds": args.timeout + 10, "preparationSeconds": time.perf_counter() - started,
|
||||
"timingContract": "Production automatic executable plus an optional caller-supplied frozen external baseline; complete sampled output. C-reported solve wall/CPU and subprocess creation-through-reap wall only. Build, parsing, validation and comparisons excluded. No independently measured write/projection stage; process-minus-solve is not called write time. Ordinary file writes, no fsync.",
|
||||
"comparisonContract": "Exact structure, numeric values and time axis for series/final/finalState. Bits, including signed zero, are separately reported. No loosened tolerance or interpolation. Counters may differ.",
|
||||
"pairOrder": [{"pair": i + 1, "warmup": i < args.warmups,
|
||||
"modes": (["dense", "auto"] if i % 2 == 0 else ["auto", "dense"]) if external else ["auto"]}
|
||||
for i in range(args.warmups + args.repeats)]}
|
||||
write_json(output / "prepared.json", prepared)
|
||||
return prepared, program
|
||||
|
||||
|
||||
def execute(args, prepared: dict, program) -> dict:
|
||||
from app.simulation.native_codegen.build import build_native
|
||||
|
||||
output = args.output_dir.resolve()
|
||||
summary = {"schemaVersion": 1, "complete": False, "passed": False, "prepared": prepared,
|
||||
"errors": [], "rows": [], "pairs": [], "verify": None, "statistics": None,
|
||||
"externalBaseline": prepared["externalBaseline"],
|
||||
"speedComparison": None, "strictFailureMeans": "Exact equivalence was not established; retain outputs for independent convergence diagnostics. No automatic acceptance-tolerance change."}
|
||||
rows, pairs = summary["rows"], summary["pairs"]
|
||||
|
||||
def save():
|
||||
write_json(output / "summary.json", summary)
|
||||
|
||||
def run(mode: str, label: str, *, pair=None, warmup=False, diagnostic=False):
|
||||
directory = output / mode / label
|
||||
directory.mkdir(parents=True, exist_ok=False)
|
||||
cfg = prepared["settings"]
|
||||
executable = Path(prepared["externalBaseline"]["executable"]) if mode == "dense" else build.executable
|
||||
command = [str(executable), "--method", "BDF",
|
||||
"--start", str(cfg["t_start"]), "--stop", str(cfg["t_stop"]),
|
||||
"--sample-step", str(prepared["sampleStep"]), "--max-step", str(cfg["max_step"]),
|
||||
"--rtol", "1e-8", "--timeout", str(args.timeout),
|
||||
"--output", str(directory / "result.json"), "--result-index", str(directory / "result-index.json"),
|
||||
"--cancel-file", str(directory / "cancel.request")]
|
||||
if diagnostic:
|
||||
command.append("--verify-jacobian")
|
||||
row = {"mode": mode, "label": label, "pair": pair, "warmup": warmup, "diagnostic": diagnostic,
|
||||
"includedInStatistics": not warmup and not diagnostic, "directory": str(directory), "command": command,
|
||||
"completed": False, "resultValidation": None, "externalBaseline": mode == "dense"}
|
||||
rows.append(row)
|
||||
write_json(directory / "command.json", command)
|
||||
environment = dict(os.environ)
|
||||
for key in ("NATIVE_COMPUTE_PROFILE", "NATIVE_STAGE_PROFILE"):
|
||||
environment.pop(key, None)
|
||||
try:
|
||||
with (directory / "stdout.log").open("wb") as stdout, (directory / "stderr.log").open("wb") as stderr:
|
||||
started = time.perf_counter()
|
||||
try:
|
||||
process = subprocess.run(command, cwd=executable.parent, env=environment, stdin=subprocess.DEVNULL,
|
||||
stdout=stdout, stderr=stderr, timeout=args.timeout + 10)
|
||||
row["exitCode"] = process.returncode
|
||||
finally:
|
||||
row["processWallSeconds"] = time.perf_counter() - started
|
||||
result_path = directory / "result.json"
|
||||
if not result_path.is_file():
|
||||
raise RuntimeError(f"No native result: {directory}")
|
||||
data, result_hash = read_result(result_path)
|
||||
row.update({key: value for key, value in data.items() if key not in PAYLOAD})
|
||||
row["resultSha256"] = result_hash
|
||||
row["resultBytes"] = result_path.stat().st_size
|
||||
row["resultValidation"] = validate_result(data, prepared, external_baseline=mode == "dense")
|
||||
if row["exitCode"] != 0:
|
||||
raise RuntimeError(f"Native exit code {row['exitCode']}: {directory}")
|
||||
if row["resultValidation"]["counterAccountingMatches"] is False:
|
||||
raise RuntimeError(f"RHS counter accounting failed: {directory}")
|
||||
if mode == "dense" and data.get("jacobianMode") not in (None, "dense-difference"):
|
||||
raise RuntimeError("External baseline is not a frozen default-dense executable; current production cannot emulate the old strategy")
|
||||
if mode == "dense":
|
||||
row["historicalStrategyEvidence"] = "reported-dense" if data.get("jacobianMode") else "unreported: caller-supplied frozen provenance"
|
||||
if diagnostic and (data["jacobianChecks"] <= 0 or data["jacobianMismatches"] != 0 or
|
||||
data["jacobianChecks"] != data["jacobianColoredEvals"]):
|
||||
raise RuntimeError("Verify mode did not validate every computed colored Jacobian without mismatches")
|
||||
row["completed"] = True
|
||||
print(f"{mode}/{label}: solve={row['solveSeconds']:.6f}s process={row['processWallSeconds']:.6f}s nfev={row['nfev']}", flush=True)
|
||||
return data
|
||||
except Exception as exc:
|
||||
row["error"] = f"{type(exc).__name__}: {exc}"
|
||||
raise
|
||||
finally:
|
||||
write_json(directory / "run.json", row)
|
||||
save()
|
||||
|
||||
try:
|
||||
# Build current production once. An optional baseline is never built or modified.
|
||||
build = build_native(program, cache_dir=output / "cache")
|
||||
summary["build"] = {"executable": str(build.executable), "buildKey": build.manifest["buildKey"],
|
||||
"cacheHit": build.cache_hit, "seconds": build.seconds, "manifest": build.manifest}
|
||||
if prepared["externalBaseline"]:
|
||||
if digest(Path(prepared["externalBaseline"]["executable"])) != prepared["externalBaseline"]["sha256"]:
|
||||
raise RuntimeError("Frozen external baseline changed after preparation")
|
||||
if digest(build.executable) == prepared["externalBaseline"]["sha256"]:
|
||||
raise RuntimeError("External baseline equals the current production executable")
|
||||
save()
|
||||
verify_data = run("verify", "validation", diagnostic=True) if args.verify else None
|
||||
first_auto = first_dense = None
|
||||
warmup_count = measured_count = 0
|
||||
for planned in prepared["pairOrder"]:
|
||||
is_warmup = planned["warmup"]
|
||||
if is_warmup:
|
||||
warmup_count += 1
|
||||
label = f"warmup-{warmup_count}"
|
||||
else:
|
||||
measured_count += 1
|
||||
label = f"run-{measured_count}"
|
||||
results = {mode: run(mode, label, pair=planned["pair"], warmup=is_warmup) for mode in planned["modes"]}
|
||||
comparison = compare_payload(results["dense"], results["auto"]) if "dense" in results else None
|
||||
auto_repeat = compare_payload(first_auto, results["auto"]) if first_auto is not None else None
|
||||
dense_repeat = compare_payload(first_dense, results["dense"]) if first_dense is not None else None
|
||||
if first_auto is None:
|
||||
first_auto = results["auto"]
|
||||
first_dense = results.get("dense")
|
||||
if verify_data is not None:
|
||||
summary["verify"] = compare_payload(results["auto"], verify_data)
|
||||
summary["verify"]["modes"] = "production default versus --verify-jacobian: diagnostic must preserve trajectory"
|
||||
verify_data = None
|
||||
record = {"pair": planned["pair"], "label": label, "warmup": is_warmup, "order": planned["modes"],
|
||||
"externalBaseline": prepared["externalBaseline"] is not None,
|
||||
"denseVsAuto": comparison, "denseRepeatVsFirstDense": dense_repeat,
|
||||
"autoRepeatVsFirstAuto": auto_repeat}
|
||||
pairs.append(record)
|
||||
write_json(output / f"comparison-{label}.json", record)
|
||||
save()
|
||||
if any(check is not None and not check["passed"] for check in (auto_repeat, dense_repeat, summary["verify"])):
|
||||
raise RuntimeError(f"Within-algorithm or default/verify reproducibility failed at {label}")
|
||||
if comparison is not None and not comparison["passed"] and not args.record_differences:
|
||||
raise RuntimeError(f"Strict external-baseline comparison failed at {label}; raw artifacts retained for convergence diagnostics")
|
||||
summary["statistics"] = {mode: {key: stats(row[key] for row in rows if row["mode"] == mode and row["includedInStatistics"] and key in row)
|
||||
for key in (*TIMINGS, *COUNTERS, "resultBytes")} for mode in ("dense", "auto")}
|
||||
if prepared["externalBaseline"]:
|
||||
ratios = {}
|
||||
for metric in TIMINGS:
|
||||
matched = []
|
||||
for pair in pairs:
|
||||
if pair["warmup"]:
|
||||
continue
|
||||
selected = {row["mode"]: row for row in rows if row["pair"] == pair["pair"]}
|
||||
old, new = selected["dense"][metric], selected["auto"][metric]
|
||||
if old <= 0 or new <= 0:
|
||||
raise RuntimeError(f"Cannot form a positive-duration comparison for {metric}")
|
||||
matched.append({"pair": pair["pair"], "dense": old, "auto": new,
|
||||
"reductionPercent": 100 * (old - new) / old, "speedup": old / new})
|
||||
old_median = summary["statistics"]["dense"][metric]["median"]
|
||||
new_median = summary["statistics"]["auto"][metric]["median"]
|
||||
ratios[metric] = {"pairs": matched,
|
||||
"pairedReductionPercent": stats(p["reductionPercent"] for p in matched),
|
||||
"pairedSpeedup": stats(p["speedup"] for p in matched),
|
||||
"ratioOfGroupMedians": {"denseMedian": old_median, "autoMedian": new_median,
|
||||
"reductionPercent": 100 * (old_median - new_median) / old_median,
|
||||
"speedup": old_median / new_median}}
|
||||
summary["speedComparison"] = {"method": "Alternating serial frozen-external-baseline/production pairs, distinct executables; paired ratios and ratio of group medians are distinct estimates. Warmups and verification excluded.", "externalBaseline": True, "metrics": ratios}
|
||||
checks = [check for pair in pairs for check in (pair["denseVsAuto"], pair["denseRepeatVsFirstDense"], pair["autoRepeatVsFirstAuto"]) if check is not None]
|
||||
if summary["verify"] is not None:
|
||||
checks.append(summary["verify"])
|
||||
summary["complete"] = True
|
||||
summary["comparisonCount"] = len(checks)
|
||||
summary["passed"] = all(check["passed"] for check in checks)
|
||||
summary["numericalAcceptance"] = ("Exact external-baseline equivalence and repeatability" if summary["passed"] else "Requires separate trajectory/convergence review; no tolerance gate applied") if prepared["externalBaseline"] else "Production repeatability/verification only; no external accuracy comparison"
|
||||
summary["allPayloadBitsEqual"] = all(check["allPayloadBitsEqual"] for check in checks) if checks else None
|
||||
save()
|
||||
except Exception as exc:
|
||||
summary["errors"].append(f"{type(exc).__name__}: {exc}")
|
||||
save()
|
||||
print(summary["errors"][-1], file=sys.stderr, flush=True)
|
||||
return summary
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
|
||||
parser.add_argument("--input", type=Path, default=ROOT / "tests/data/test-mql-8-corrected.json")
|
||||
parser.add_argument("--output-dir", required=True, type=Path)
|
||||
parser.add_argument("--run", action="store_true", help="Build once and execute; omitted means preparation only")
|
||||
parser.add_argument("--warmups", type=int, default=1)
|
||||
parser.add_argument("--repeats", type=int, default=3)
|
||||
parser.add_argument("--verify-jacobian", "--verify", dest="verify", action="store_true", help="Also execute one full-matrix diagnostic run, excluded from timing statistics")
|
||||
parser.add_argument("--baseline-executable", type=Path, help="Optional frozen historical executable with default dense strategy; never built or modified by this tool")
|
||||
parser.add_argument("--record-differences", action="store_true", help="Complete timings while retaining failed strict external-baseline comparisons; does not accept numerical differences")
|
||||
parser.add_argument("--timeout", type=float, default=120, help="Per-process native timeout; Python allows 10 s exit grace")
|
||||
if any(arg == "--jacobian" or arg.startswith("--jacobian=") for arg in sys.argv[1:]):
|
||||
parser.error("--jacobian strategy selection was removed; use the production default or --verify-jacobian. Dense requests require a frozen historical version (benchmark only: --baseline-executable).")
|
||||
args = parser.parse_args()
|
||||
if args.warmups < 0 or args.repeats < 1 or not math.isfinite(args.timeout) or args.timeout <= 0:
|
||||
parser.error("Require warmups >= 0, repeats >= 1, finite timeout > 0")
|
||||
try:
|
||||
prepared, program = prepare(args)
|
||||
except Exception as exc:
|
||||
print(f"Preparation failed: {type(exc).__name__}: {exc}", file=sys.stderr)
|
||||
return 2
|
||||
print(f"Prepared: {args.output_dir.resolve() / 'prepared.json'} (no compilation or solve during preparation)", flush=True)
|
||||
if not args.run:
|
||||
return 0
|
||||
summary = execute(args, prepared, program)
|
||||
if summary["complete"]:
|
||||
print(json.dumps(summary["speedComparison"] if summary["speedComparison"] is not None else summary["statistics"]["auto"], ensure_ascii=False, indent=2), flush=True)
|
||||
return 0 if summary["passed"] or (args.record_differences and summary["complete"]) else 2
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -0,0 +1,301 @@
|
||||
r"""Diagnose complete native trajectories without changing acceptance tolerances.
|
||||
|
||||
.venv/bin/python tests/manual/compare_jacobian_trajectories.py \
|
||||
--baseline old/result.json --candidate new/result.json \
|
||||
--manifest new/cache/KEY/manifest.json --output test/jacobian/comparison.json
|
||||
|
||||
Optional --reference tight/result.json compares each trajectory to that supplied
|
||||
reference. Its precision/convergence must be established separately. There is no
|
||||
numerical pass threshold here: successful analysis is not accuracy acceptance.
|
||||
The ordinary grid defaults to 0..10 s at .01 s. Only exactly shared grid times
|
||||
are compared; all off-grid saved points are listed and examined separately.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
from hashlib import sha256
|
||||
import json
|
||||
import math
|
||||
from pathlib import Path
|
||||
import sys
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[2]
|
||||
sys.path.insert(0, str(ROOT))
|
||||
from app.simulation.native_codegen.tolerances import state_absolute_tolerance
|
||||
|
||||
RTOL = 1e-8
|
||||
PAYLOAD = {"series", "final", "finalState"}
|
||||
|
||||
|
||||
def read_json(path: Path):
|
||||
def unique(items):
|
||||
value = {}
|
||||
for key, item in items:
|
||||
if key in value:
|
||||
raise ValueError(f"Duplicate JSON key: {key}")
|
||||
value[key] = item
|
||||
return value
|
||||
|
||||
def reject(token):
|
||||
raise ValueError(f"Nonfinite JSON token: {token}")
|
||||
|
||||
def parsed_float(token):
|
||||
value = float(token)
|
||||
if not math.isfinite(value):
|
||||
raise ValueError(f"Nonfinite JSON number: {token}")
|
||||
return value
|
||||
|
||||
raw = path.read_bytes()
|
||||
value = json.loads(raw, parse_float=parsed_float, parse_int=lambda token: -0.0 if token == "-0" else int(token),
|
||||
parse_constant=reject, object_pairs_hook=unique)
|
||||
return value, {"path": str(path.resolve()), "sha256": sha256(raw).hexdigest(), "bytes": len(raw)}
|
||||
|
||||
|
||||
def finite(value):
|
||||
if isinstance(value, bool) or not isinstance(value, (int, float)):
|
||||
raise ValueError("Expected a finite numeric payload")
|
||||
number = float(value)
|
||||
if not math.isfinite(number) or (isinstance(value, int) and int(number) != value):
|
||||
raise ValueError("Payload is nonfinite or not exactly representable as binary64")
|
||||
return number
|
||||
|
||||
|
||||
def rms(values):
|
||||
return math.hypot(*values) / math.sqrt(len(values)) if values else None
|
||||
|
||||
|
||||
def peak(values, times):
|
||||
i = max(range(len(values)), key=lambda i: abs(values[i]))
|
||||
return {"absolute": abs(values[i]), "value": values[i], "time": times[i]}
|
||||
|
||||
|
||||
def validate(data, metadata, states, start, stop):
|
||||
if not isinstance(data, dict) or data.get("success") is not True or data.get("status") != "completed":
|
||||
raise ValueError("A completed native result is required")
|
||||
if not isinstance(data.get("series"), dict) or not isinstance(data.get("final"), dict) or not isinstance(data.get("finalState"), list):
|
||||
raise ValueError("Missing native series/final/finalState structure")
|
||||
if set(data["series"]) != set(metadata) | {"time"} or set(data["final"]) != set(metadata):
|
||||
raise ValueError("Series/final key sets do not match the supplied manifest")
|
||||
if len(data["finalState"]) != len(states) or not set(states) <= set(metadata):
|
||||
raise ValueError("State keys/order cannot be mapped from the supplied manifest")
|
||||
times = data["series"]["time"]
|
||||
if not isinstance(times, list) or not times:
|
||||
raise ValueError("No saved samples")
|
||||
for key, values in data["series"].items():
|
||||
if not isinstance(values, list) or len(values) != len(times):
|
||||
raise ValueError(f"Invalid sample length: {key}")
|
||||
for value in values:
|
||||
finite(value)
|
||||
for value in (*data["final"].values(), *data["finalState"]):
|
||||
finite(value)
|
||||
if any(a >= b for a, b in zip(times, times[1:])):
|
||||
raise ValueError("Sample times are not strictly increasing")
|
||||
if times[0] != start or times[-1] != stop or data.get("simulatedUntil") != stop:
|
||||
raise ValueError("Result does not span the requested complete interval")
|
||||
|
||||
|
||||
def input_observations(data, metadata, states, grid):
|
||||
series, times = data["series"], data["series"]["time"]
|
||||
fixed = set(grid)
|
||||
mapping = {t: i for i, t in enumerate(times)}
|
||||
extras = [i for i, t in enumerate(times) if t not in fixed]
|
||||
mechanical = [key for key, item in metadata.items() if item.get("scope") == "component"
|
||||
and item.get("quantity") in {"force", "length", "velocity", "acceleration"}]
|
||||
events = []
|
||||
for i in extras:
|
||||
indices = list(range(max(0, i - 1), min(len(times), i + 2)))
|
||||
neighborhood = {}
|
||||
for key, item in metadata.items():
|
||||
group = (item["quantity"], item["unit"])
|
||||
best = max(indices, key=lambda j: abs(series[key][j]))
|
||||
current = neighborhood.get(group)
|
||||
if current is None or abs(series[key][best]) > current["absolute"]:
|
||||
neighborhood[group] = {"quantity": group[0], "unit": group[1], "key": key,
|
||||
"absolute": abs(series[key][best]), "value": series[key][best], "time": times[best]}
|
||||
events.append({"sampleIndex": i, "time": times[i], "neighborSampleTimes": [times[j] for j in indices],
|
||||
"stateSamples": [{"time": times[j], "values": {key: series[key][j] for key in states}} for j in indices],
|
||||
"mechanicalSamples": [{"key": key, "unit": metadata[key]["unit"],
|
||||
"values": [series[key][j] for j in indices],
|
||||
"neighborhoodPeak": peak([series[key][j] for j in indices], [times[j] for j in indices])}
|
||||
for key in mechanical],
|
||||
"neighborhoodQuantityPeaks": list(neighborhood.values())})
|
||||
mass_keys = [key for key in states if key.rsplit(".", 1)[-1] in {"m", "m1", "m2"}
|
||||
and metadata[key]["quantity"] == "mass" and metadata[key]["unit"] == "kg"]
|
||||
conservation = None
|
||||
if mass_keys:
|
||||
totals = [math.fsum(series[key][i] for key in mass_keys) for i in range(len(times))]
|
||||
drift = [abs(value - totals[0]) for value in totals]
|
||||
worst = max(range(len(times)), key=drift.__getitem__)
|
||||
conservation = {"stateKeys": mass_keys, "uniqueMassStateCount": len(mass_keys), "initialKg": totals[0],
|
||||
"finalKg": totals[-1], "maxAbsoluteDriftKg": drift[worst], "worstTime": times[worst],
|
||||
"sampleCount": len(times), "includesOffGridSamples": True,
|
||||
"method": "math.fsum of unique manifest gas mass states; output aliases are not accumulated"}
|
||||
return {"metadata": {k: v for k, v in data.items() if k not in PAYLOAD},
|
||||
"sampleCount": len(times), "seriesVariableCount": len(metadata), "stateCount": len(states),
|
||||
"missingFixedGridTimes": [t for t in grid if t not in mapping],
|
||||
"excludedFromFixedGrid": [{"index": i, "time": times[i]} for i in extras],
|
||||
"extraSampleCount": len(extras), "reportedStateTransitions": data.get("stateTransitions"),
|
||||
"eventInterpretation": "Off-grid saved points are event candidates, not guaranteed event identities. An event on the fixed grid is not distinguishable from ordinary samples. Neighbors are nearest saved samples, not the true pre/post impact limits; a non-grid final endpoint can also be extra.",
|
||||
"events": events, "massConservation": conservation,
|
||||
"final": data["final"], "finalStateByKey": dict(zip(states, data["finalState"], strict=True)),
|
||||
"finalVsLastSeriesUnequalKeys": [key for key in metadata if data["final"][key] != series[key][-1]],
|
||||
"finalStateVsLastSeriesUnequalKeys": [key for key, value in zip(states, data["finalState"], strict=True) if value != series[key][-1]]}
|
||||
|
||||
|
||||
def aggregate(rows):
|
||||
groups = {}
|
||||
for row in rows:
|
||||
group = (row["quantity"], row["unit"])
|
||||
if group not in groups:
|
||||
groups[group] = {"quantity": group[0], "unit": group[1], "variableCount": 0,
|
||||
"sampleValues": 0, "maxAbsoluteError": -1., "norm": 0.}
|
||||
total = groups[group]
|
||||
total["variableCount"] += 1
|
||||
total["sampleValues"] += row["sampleCount"]
|
||||
total["norm"] = math.hypot(total["norm"], row["rmsError"] * math.sqrt(row["sampleCount"]))
|
||||
if row["maxAbsoluteError"] > total["maxAbsoluteError"]:
|
||||
total.update(maxAbsoluteError=row["maxAbsoluteError"], worstKey=row["key"], worstTime=row["worstTime"])
|
||||
for total in groups.values():
|
||||
total["rmsError"] = total.pop("norm") / math.sqrt(total["sampleValues"])
|
||||
return list(groups.values())
|
||||
|
||||
|
||||
def trajectory_comparison(left, right, metadata, states, grid, label):
|
||||
a_times, b_times = left["series"]["time"], right["series"]["time"]
|
||||
a_index, b_index = {t: i for i, t in enumerate(a_times)}, {t: i for i, t in enumerate(b_times)}
|
||||
common = [t for t in grid if t in a_index and t in b_index]
|
||||
if not common:
|
||||
raise ValueError(f"No exactly shared fixed-grid times: {label}")
|
||||
ai, bi = [a_index[t] for t in common], [b_index[t] for t in common]
|
||||
curves = []
|
||||
state_rows = []
|
||||
state_norms = [0.] * len(common)
|
||||
state_set = set(states)
|
||||
for key, item in metadata.items():
|
||||
av = [left["series"][key][i] for i in ai]
|
||||
bv = [right["series"][key][i] for i in bi]
|
||||
errors = [b - a for a, b in zip(av, bv, strict=True)]
|
||||
if not all(math.isfinite(e) for e in errors):
|
||||
raise ValueError(f"Difference exceeds binary64 range: {key}")
|
||||
worst = max(range(len(common)), key=lambda i: abs(errors[i]))
|
||||
row = {"key": key, "quantity": item["quantity"], "unit": item["unit"], "sampleCount": len(common),
|
||||
"maxAbsoluteError": abs(errors[worst]), "rmsError": rms(errors), "worstTime": common[worst],
|
||||
"leftValueAtWorst": av[worst], "rightValueAtWorst": bv[worst],
|
||||
"leftAllSavedPeak": peak(left["series"][key], a_times),
|
||||
"rightAllSavedPeak": peak(right["series"][key], b_times)}
|
||||
curves.append(row)
|
||||
if key in state_set:
|
||||
atol = float(state_absolute_tolerance(key))
|
||||
z = [abs(e) / (atol + RTOL * max(abs(a), abs(b))) for e, a, b in zip(errors, av, bv, strict=True)]
|
||||
at = max(range(len(z)), key=z.__getitem__)
|
||||
state_rows.append({"key": key, "unit": item["unit"], "atol": atol,
|
||||
"maxWeightedError": z[at], "rmsWeightedError": rms(z), "worstTime": common[at],
|
||||
"maxAbsoluteError": row["maxAbsoluteError"], "rmsError": row["rmsError"]})
|
||||
for i, value in enumerate(z):
|
||||
state_norms[i] = math.hypot(state_norms[i], value)
|
||||
wrms = [value / math.sqrt(len(states)) for value in state_norms]
|
||||
weighted_worst = max(state_rows, key=lambda row: row["maxWeightedError"])
|
||||
wrms_at = max(range(len(wrms)), key=wrms.__getitem__)
|
||||
final_rows = []
|
||||
for key, item in metadata.items():
|
||||
a, b = left["final"][key], right["final"][key]
|
||||
error = abs(b - a)
|
||||
final_rows.append({"key": key, "quantity": item["quantity"], "unit": item["unit"],
|
||||
"sampleCount": 1, "maxAbsoluteError": error, "rmsError": error,
|
||||
"worstTime": right["simulatedUntil"], "left": a, "right": b})
|
||||
terminal = []
|
||||
for key, a, b in zip(states, left["finalState"], right["finalState"], strict=True):
|
||||
atol = float(state_absolute_tolerance(key))
|
||||
terminal.append({"key": key, "unit": metadata[key]["unit"], "left": a, "right": b,
|
||||
"absoluteError": abs(b - a), "weightedError": abs(b - a) / (atol + RTOL * max(abs(a), abs(b)))})
|
||||
fixed = set(grid)
|
||||
a_extra, b_extra = [t for t in a_times if t not in fixed], [t for t in b_times if t not in fixed]
|
||||
return {"label": label, "comparedFixedGridTimes": common, "comparedSampleCount": len(common),
|
||||
"expectedFixedGridCount": len(grid), "allFixedGridTimesCompared": len(common) == len(grid),
|
||||
"missingFromLeft": [t for t in grid if t not in a_index], "missingFromRight": [t for t in grid if t not in b_index],
|
||||
"curves": curves, "quantityGroups": aggregate(curves),
|
||||
"stateErrors": {"formula": "abs(right-left)/(state_atol + 1e-8*max(abs(left),abs(right))), independently at each state/time",
|
||||
"interpretation": "Diagnostic normalization only. Local integration tolerances are not global trajectory acceptance thresholds.",
|
||||
"rows": state_rows, "maxWeightedError": weighted_worst["maxWeightedError"],
|
||||
"worstKey": weighted_worst["key"], "worstTime": weighted_worst["worstTime"],
|
||||
"wrmsAtEachComparedTime": wrms, "maxWrms": wrms[wrms_at], "maxWrmsTime": common[wrms_at]},
|
||||
"final": {"rows": final_rows, "quantityGroups": aggregate(final_rows)},
|
||||
"finalState": {"rows": terminal, "maxWeightedError": max(row["weightedError"] for row in terminal),
|
||||
"wrms": rms([row["weightedError"] for row in terminal])},
|
||||
"eventTimeDiagnostics": {"leftExtraTimes": a_extra, "rightExtraTimes": b_extra,
|
||||
"leftCount": len(a_extra), "rightCount": len(b_extra),
|
||||
"reportedTransitions": [left.get("stateTransitions"), right.get("stateTransitions")],
|
||||
"ordinalTimeDifferences": [{"ordinal": i + 1, "leftTime": a, "rightTime": b, "rightMinusLeftSeconds": b - a}
|
||||
for i, (a, b) in enumerate(zip(a_extra, b_extra, strict=True))] if len(a_extra) == len(b_extra) else None,
|
||||
"interpretation": "Equal-count ordinal differences are observations only, not verified physical event matching. Different counts are not paired. Event and neighbor values/peaks are in input observations; no time shifting or interpolation."}}
|
||||
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
|
||||
parser.add_argument("--baseline", required=True, type=Path)
|
||||
parser.add_argument("--candidate", required=True, type=Path)
|
||||
parser.add_argument("--manifest", required=True, type=Path)
|
||||
parser.add_argument("--reference", type=Path)
|
||||
parser.add_argument("--output", required=True, type=Path)
|
||||
parser.add_argument("--start", type=float, default=0.)
|
||||
parser.add_argument("--stop", type=float, default=10.)
|
||||
parser.add_argument("--sample-step", type=float, default=.01)
|
||||
args = parser.parse_args()
|
||||
if not all(math.isfinite(v) for v in (args.start, args.stop, args.sample_step)) or not (args.stop > args.start and args.sample_step > 0):
|
||||
parser.error("Require finite increasing interval and positive sample step")
|
||||
if (args.stop - args.start) / args.sample_step > 1000000 or args.start + args.sample_step == args.start:
|
||||
parser.error("Invalid or excessive sampling grid")
|
||||
output = args.output.resolve()
|
||||
sources = [args.baseline, args.candidate, args.manifest] + ([args.reference] if args.reference else [])
|
||||
if output in {path.resolve() for path in sources}:
|
||||
parser.error("Output must not overwrite an input")
|
||||
report = {"schemaVersion": 1, "complete": False, "errors": [], "inputs": {}, "comparisons": {},
|
||||
"numericalAcceptance": "Not assessed: diagnostic errors only; no tolerance relaxation or automatic pass threshold.",
|
||||
"definitions": {"fixedGrid": "start + integer index * sampleStep; exact floating-point time membership, no interpolation",
|
||||
"quantityRms": "Pooled RMS of every compared value in that quantity/unit group; output aliases are included. Per-curve RMS is also reported.",
|
||||
"peaks": "All saved points including off-grid events. A sampled peak need not be the continuous-time peak.",
|
||||
"reference": "User-supplied reference; native JSON alone does not establish tighter effective tolerances or convergence. Both comparisons retain normalization rtol=1e-8.",
|
||||
"sharedManifest": "Caller must establish identical state order and physical output mapping for every input; one shared manifest is checked against all key sets and lengths."},
|
||||
"scriptSha256": sha256(Path(__file__).read_bytes()).hexdigest(),
|
||||
"stateToleranceSourceSha256": sha256((ROOT / 'app/simulation/native_codegen/tolerances.py').read_bytes()).hexdigest(),
|
||||
"normalizationRtol": RTOL, "grid": {"start": args.start, "stop": args.stop, "sampleStep": args.sample_step}}
|
||||
output.parent.mkdir(parents=True, exist_ok=True)
|
||||
try:
|
||||
manifest, identity = read_json(args.manifest)
|
||||
report["manifest"] = identity
|
||||
states = manifest["stateKeys"]
|
||||
variables = manifest["variables"]
|
||||
metadata = {row["key"]: row for row in variables}
|
||||
if not states or len(states) != len(set(states)) or len(metadata) != len(variables):
|
||||
raise ValueError("Manifest has empty/duplicate state keys or duplicate output keys")
|
||||
if not all(isinstance(row.get("quantity"), str) and isinstance(row.get("unit"), str) for row in variables):
|
||||
raise ValueError("Every manifest output requires quantity and unit metadata")
|
||||
report["stateKeys"] = states
|
||||
report["stateAbsoluteTolerances"] = {key: float(state_absolute_tolerance(key)) for key in states}
|
||||
grid = []
|
||||
i = 0
|
||||
while (t := args.start + i * args.sample_step) <= args.stop:
|
||||
grid.append(t)
|
||||
i += 1
|
||||
data = {}
|
||||
for label, path in [("baseline", args.baseline), ("candidate", args.candidate)] + ([("reference", args.reference)] if args.reference else []):
|
||||
data[label], identity = read_json(path)
|
||||
report["inputs"][label] = identity
|
||||
validate(data[label], metadata, states, args.start, args.stop)
|
||||
report["inputs"][label]["observations"] = input_observations(data[label], metadata, states, grid)
|
||||
report["comparisons"]["candidateVsBaseline"] = trajectory_comparison(data["baseline"], data["candidate"], metadata, states, grid, "candidate minus baseline")
|
||||
if args.reference:
|
||||
for label in ("baseline", "candidate"):
|
||||
report["comparisons"][label + "VsReference"] = trajectory_comparison(data["reference"], data[label], metadata, states, grid, label + " minus supplied reference")
|
||||
report["complete"] = True
|
||||
except (OSError, ValueError, KeyError, TypeError, OverflowError) as exc:
|
||||
report["errors"].append(f"{type(exc).__name__}: {exc}")
|
||||
output.write_text(json.dumps(report, ensure_ascii=False, indent=2, allow_nan=False) + "\n", encoding="utf-8")
|
||||
print(json.dumps({"complete": report["complete"], "errors": report["errors"], "output": str(output),
|
||||
"comparedSamples": {key: value["comparedSampleCount"] for key, value in report["comparisons"].items()},
|
||||
"numericalAcceptance": report["numericalAcceptance"]}, ensure_ascii=False), flush=True)
|
||||
return 0 if report["complete"] else 2
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -13,6 +13,9 @@ unchanged. Full output/state/counter equality is checked outside run timing.
|
||||
Inclusive durations are nested: only exclusiveSeconds may be added. Clock and
|
||||
bookkeeping overhead remain in measured totals; compare against the control.
|
||||
No property/pipe/libc allocation is inferred from this outer-only diagnostic.
|
||||
The production automatic Jacobian is used by default. --verify-jacobian enables
|
||||
full-matrix checking; those diagnostic timings are not ordinary production cost.
|
||||
Recorded --jacobian auto/verify options are translated; dense replay is rejected.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
@@ -32,7 +35,7 @@ ROOT = Path(__file__).resolve().parents[2]
|
||||
sys.path.insert(0, str(ROOT))
|
||||
from app.simulation.native_codegen.build import LIBRARIES, toolchain
|
||||
|
||||
CATEGORIES = ["integration", "cvode_step", "rhs", "poll", "accept", "append", "dense_output", "dense_setup", "dense_solve"]
|
||||
CATEGORIES = ["integration", "cvode_step", "rhs", "poll", "accept", "append", "dense_output", "dense_setup", "dense_solve", "jacobian"]
|
||||
COUNTERS = ["rhs", "linear_rhs", "nonlinear_iterations", "nonlinear_failures"]
|
||||
PROFILE_HEADER = r'''
|
||||
#ifndef NATIVE_COMPUTE_PROFILE_H
|
||||
@@ -145,8 +148,8 @@ def instrument(native: Path) -> None:
|
||||
(native / "include/compute_profile.h").write_text(PROFILE_HEADER.replace("@CATEGORIES@", ", ".join("PROFILE_" + c.upper() for c in CATEGORIES)))
|
||||
(native / "runtime/compute_profile.c").write_text(PROFILE_SOURCE.replace("@NAMES@", ",".join(json.dumps(c) for c in CATEGORIES)))
|
||||
for filename, functions in {
|
||||
"common.c": {"native_rhs": "rhs", "native_poll": "poll", "native_accept": "accept", "native_append": "append"},
|
||||
"cvode_solver.c": {"native_bdf": "integration", "cv_dense": "dense_output"},
|
||||
"common.c": {"native_rhs": "rhs", "native_jacobian_rhs": "rhs", "native_poll": "poll", "native_accept": "accept", "native_append": "append"},
|
||||
"cvode_solver.c": {"native_bdf": "integration", "cv_dense": "dense_output", "cv_jacobian": "jacobian"},
|
||||
"rk45.c": {"native_rk45": "integration"},
|
||||
}.items():
|
||||
path = native / "runtime" / filename
|
||||
@@ -168,7 +171,13 @@ def instrument(native: Path) -> None:
|
||||
path.write_text(text)
|
||||
|
||||
|
||||
def runtime_arguments(stages: Path | None) -> list[str]:
|
||||
def runtime_arguments(stages: Path | None, verify_jacobian: bool = False) -> list[str]:
|
||||
"""Replay numerical options using the production Jacobian only.
|
||||
|
||||
Historical auto becomes the default and verify becomes --verify-jacobian.
|
||||
A historical dense request must run with its frozen historical tool/runtime;
|
||||
silently replaying it with today's automatic algorithm would fake a baseline.
|
||||
"""
|
||||
if stages:
|
||||
original = json.loads(stages.read_text())["process"]["command"]
|
||||
args = original[1:]
|
||||
@@ -178,18 +187,30 @@ def runtime_arguments(stages: Path | None) -> list[str]:
|
||||
value_options = {"--method", "--start", "--stop", "--sample-step", "--max-step", "--rtol", "--timeout"}
|
||||
while index < len(args):
|
||||
key = args[index]
|
||||
if key == "--verify-jacobian":
|
||||
verify_jacobian = True; index += 1; continue
|
||||
if key == "--solve-only":
|
||||
safe.append(key); index += 1; continue
|
||||
if key not in value_options | {"--output", "--result-index", "--cancel-file"} or index + 1 >= len(args):
|
||||
if key not in value_options | {"--jacobian", "--output", "--result-index", "--cancel-file"} or index + 1 >= len(args):
|
||||
raise RuntimeError(f"Unsupported replay argument: {key}")
|
||||
if key in value_options:
|
||||
if key == "--jacobian":
|
||||
recorded = args[index + 1]
|
||||
if recorded == "dense":
|
||||
raise RuntimeError("Historical --jacobian dense requires the frozen historical executable and tool; current production has no legacy strategy selector")
|
||||
if recorded not in {"auto", "verify"}:
|
||||
raise RuntimeError(f"Unsupported historical Jacobian mode: {recorded}")
|
||||
verify_jacobian = verify_jacobian or recorded == "verify"
|
||||
elif key in value_options:
|
||||
safe.extend(args[index:index + 2])
|
||||
index += 2
|
||||
if verify_jacobian:
|
||||
safe.append("--verify-jacobian")
|
||||
return safe
|
||||
|
||||
|
||||
def prepare(args: argparse.Namespace) -> dict:
|
||||
cache, output = args.cache_dir.resolve(), args.output_dir.resolve()
|
||||
numerical_arguments = runtime_arguments(args.request_stages, args.verify_jacobian)
|
||||
if not output.is_relative_to(ROOT / "test"):
|
||||
raise RuntimeError("Diagnostic output must be in the repository's ignored test/ directory")
|
||||
manifest = json.loads((cache / "manifest.json").read_text())
|
||||
@@ -223,7 +244,7 @@ def prepare(args: argparse.Namespace) -> dict:
|
||||
shutil.copy2(cache / name, output / name)
|
||||
instrument(native)
|
||||
command = [compiler, *manifest["compilerFlags"], "-I", str(output), "-I", str(native / "include"), "-I", str(sundials / "include"), str(output / "model.c"), *map(str, sorted(native.rglob("*.c"))), "-Wl,--start-group", *map(str, libraries), "-Wl,--end-group", "-lm", "-o", str(output / "profiled-model")]
|
||||
prepared = {"controlExecutable": str(cache / "model"), "profiledExecutable": str(output / "profiled-model"), "buildCommand": command, "runtimeArguments": runtime_arguments(args.request_stages), "cacheDir": str(cache), "cachedMainDifference": differences, "stateCount": len(manifest["stateKeys"]), "modelSha256": digest(output / "model.c"), "compiler": compiler_version, "warmups": args.warmups, "repeats": args.repeats, "scope": "Inclusive times overlap. Exclusive scope times partition integration including instrumentation overhead. No component or libc sub-cost inference.", "sourceHashes": {str(p.relative_to(output)): digest(p) for p in sorted(native.rglob("*")) if p.is_file()}}
|
||||
prepared = {"controlExecutable": str(cache / "model"), "profiledExecutable": str(output / "profiled-model"), "buildCommand": command, "runtimeArguments": numerical_arguments, "verifyJacobian": "--verify-jacobian" in numerical_arguments, "algorithm": "production-automatic", "cacheDir": str(cache), "cachedMainDifference": differences, "stateCount": len(manifest["stateKeys"]), "modelSha256": digest(output / "model.c"), "compiler": compiler_version, "warmups": args.warmups, "repeats": args.repeats, "scope": "Inclusive times overlap. Exclusive scope times partition integration including instrumentation overhead. No component or libc sub-cost inference.", "sourceHashes": {str(p.relative_to(output)): digest(p) for p in sorted(native.rglob("*")) if p.is_file()}}
|
||||
write_json(output / "prepared.json", prepared)
|
||||
return prepared
|
||||
|
||||
@@ -274,10 +295,11 @@ def execute(args: argparse.Namespace, prepared: dict) -> None:
|
||||
write_json(run / "parity-failure.json", mismatches)
|
||||
raise RuntimeError(f"Numerical/counter parity failed: {mismatches}")
|
||||
row = {"variant": variant, "run": label, "warmup": index < 0, "processWallSeconds": wall, "solveSeconds": result["solveSeconds"], "solveCpuSeconds": result["solveCpuSeconds"], "fullParity": True, "resultBytes": result_path.stat().st_size, "nfev": result["nfev"], "njev": result["njev"], "nlu": result["nlu"], "acceptedSteps": result["acceptedSteps"], "solverStarts": result["solverStarts"]}
|
||||
row.update({key: result[key] for key in ("jacobianMode", "jacobianRhsCalls", "jacobianColoredEvals", "jacobianFallbacks", "jacobianChecks", "jacobianMismatches", "cvodeRhsCalls", "cvodeLinearRhsCalls") if key in result})
|
||||
if variant == "profiled":
|
||||
profile = json.loads(profile_path.read_text())
|
||||
counters = profile["cvodeCounters"]
|
||||
checks = {"counterReadsSucceeded": profile["counterErrors"] == 0, "rhsCountMatches": counters["rhs"] + counters["linear_rhs"] == result["nfev"], "linearRhsEqualsJacobianCountTimesStates": counters["linear_rhs"] == result["njev"] * prepared["stateCount"], "rhsClockCountMatches": profile["scopes"]["integration"]["rhs"]["calls"] == result["nfev"]}
|
||||
checks = {"counterReadsSucceeded": profile["counterErrors"] == 0, "rhsCountMatches": counters["rhs"] + counters["linear_rhs"] + result.get("jacobianRhsCalls", 0) == result["nfev"], "defaultLinearRhsEqualsJacobianCountTimesStates": (counters["linear_rhs"] == result["njev"] * prepared["stateCount"] if result.get("jacobianMode") == "dense-difference" and not result.get("jacobianRhsCalls", 0) else None), "rhsClockCountMatches": profile["scopes"]["integration"]["rhs"]["calls"] == result["nfev"]}
|
||||
row["profile"] = profile
|
||||
row["counterChecks"] = checks
|
||||
if not checks["counterReadsSucceeded"] or not checks["rhsClockCountMatches"] or (result["method"] == "BDF" and not checks["rhsCountMatches"]):
|
||||
@@ -288,7 +310,7 @@ def execute(args: argparse.Namespace, prepared: dict) -> None:
|
||||
print(f"{variant}/{label}: solve={row['solveSeconds']:.6f}s wall={wall:.6f}s parity=true", flush=True)
|
||||
medians = {variant: {key: statistics.median(row[key] for row in rows if row["variant"] == variant and not row["warmup"]) for key in ("processWallSeconds", "solveSeconds", "solveCpuSeconds")} for variant in ("control", "profiled")}
|
||||
overhead = {key: medians["profiled"][key] / medians["control"][key] - 1 for key in medians["control"]}
|
||||
write_json(output / "summary.json", {"prepared": prepared, "runs": rows, "medians": medians, "instrumentationOverheadFraction": overhead, "allFullParity": True, "interpretation": "CVODE linear_rhs counts finite-difference RHS calls independently of model nfev. Its multiplication by stateCount is checked, not assumed. RHS time includes all model work; no Jacobian-specific RHS time is inferred. cvode_step.exclusiveSeconds excludes nested RHS, polling and Dense ops; integration.exclusiveSeconds is outer setup/loop/cleanup work. Dense setup covers the original factorization operation; dense_solve covers the original triangular solve operation. Clock/bookkeeping costs remain in all measured totals. Production control may retain its pre-existing sparse main clocks; cachedMainDifference records this."})
|
||||
write_json(output / "summary.json", {"prepared": prepared, "runs": rows, "medians": medians, "instrumentationOverheadFraction": overhead, "allFullParity": True, "interpretation": "CVODE linear_rhs counts only its built-in finite-difference calls. Custom jacobianRhsCalls are counted separately and included in nfev reconciliation. State-count multiplication applies only to the default callback. The custom jacobian scope includes its canonical base/probe RHS work; these inclusive durations overlap RHS totals. cvode_step.exclusiveSeconds excludes nested RHS, polling and Dense ops; integration.exclusiveSeconds is outer setup/loop/cleanup work. Dense setup covers the original factorization operation; dense_solve covers the original triangular solve operation. Clock/bookkeeping costs remain in all measured totals. Production control may retain its pre-existing sparse main clocks; cachedMainDifference records this."})
|
||||
print(f"Summary: {output / 'summary.json'}", flush=True)
|
||||
|
||||
|
||||
@@ -297,10 +319,13 @@ def main() -> None:
|
||||
parser.add_argument("--cache-dir", required=True, type=Path)
|
||||
parser.add_argument("--request-stages", type=Path)
|
||||
parser.add_argument("--output-dir", required=True, type=Path)
|
||||
parser.add_argument("--verify-jacobian", action="store_true", help="Enable full-matrix diagnostic checks in both control/profiled runs; default uses production automatic Jacobian")
|
||||
parser.add_argument("--run", action="store_true", help="Build and run serial warmups/repeats; default only prepares")
|
||||
parser.add_argument("--warmups", type=int, default=1)
|
||||
parser.add_argument("--repeats", type=int, default=3)
|
||||
parser.add_argument("--process-timeout", type=float, default=360)
|
||||
if any(arg == "--jacobian" or arg.startswith("--jacobian=") for arg in sys.argv[1:]):
|
||||
parser.error("--jacobian strategy selection was removed; use the production default or --verify-jacobian. Dense requests require a frozen historical version (benchmark only: --baseline-executable).")
|
||||
args = parser.parse_args()
|
||||
if args.warmups < 0 or args.repeats < 1:
|
||||
parser.error("warmups must be nonnegative and repeats positive")
|
||||
|
||||
@@ -0,0 +1,407 @@
|
||||
"""Summarize Jacobian timing metadata without opening results or CSV files.
|
||||
|
||||
python3 tests/manual/summarize_jacobian_cost.py --root test/jacobian-20260911
|
||||
|
||||
Requires four complete browser groups (one warmup and three formal runs each)
|
||||
plus the standalone native benchmark. Writes ROOT/cost-summary.json by default.
|
||||
Incomplete evidence overwrites the destination with complete:false and exits 2.
|
||||
Numerical differences are recorded, never interpreted as numerical acceptance.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import hashlib
|
||||
import json
|
||||
import math
|
||||
from pathlib import Path
|
||||
from statistics import median
|
||||
from typing import Any
|
||||
|
||||
REPO = Path(__file__).resolve().parents[2]
|
||||
GROUPS = {
|
||||
"baseline": ("control", None),
|
||||
"optimized": ("control", None),
|
||||
"baseline-profiled": ("profiled", "baseline-source/profiled-backend/requests"),
|
||||
"optimized-profiled": ("profiled", "backend-optimized-profiled/requests"),
|
||||
}
|
||||
GOALS = {"ready": "clickToReadyDomMs", "saved_observed": "clickToIndexedDbObservedMs",
|
||||
"csv_download_saved": "csvClickToDownloadSavedMs"}
|
||||
COUNTERS = ("nfev", "acceptedSteps", "rejectedSteps", "stateTransitions", "solverStarts", "njev", "nlu")
|
||||
JAC_COUNTERS = ("jacobianRhsCalls", "jacobianColoredEvals", "jacobianFallbacks", "jacobianChecks",
|
||||
"jacobianMismatches", "cvodeRhsCalls", "cvodeLinearRhsCalls")
|
||||
IDENTITY = ("backend", "method", "solver", "sundialsVersion", "simulatedUntil", "maxAcceptedStep")
|
||||
NATIVE_TIMES = ("solveSeconds", "solveCpuSeconds", "processWallSeconds", "buildSeconds")
|
||||
C_WALL = ("argumentPreparationSeconds", "initializationSeconds", "integrationSeconds",
|
||||
"finalSampleAndStatusSeconds", "projectionSeconds", "jsonWriteSeconds", "mainTotalSeconds")
|
||||
DEFINITIONS = {
|
||||
"scope": "Jacobian experiment only; metadata summaries, no result arrays or CSV contents are opened. complete means timing evidence is complete, not numerical acceptance.",
|
||||
"statistics": "Formal runs and warmups are separate. Each metric reports n/min/median/max and missing count. A missing baseline Jacobian field means unavailable, not zero.",
|
||||
"browserComparisons": "The separately collected control groups provide end-to-end changes: reduction = 100*(baseline median-optimized median)/baseline median. Run ordinals are not paired trials. Warmups are excluded.",
|
||||
"nativeComparisons": "The native benchmark alternated serial dense/auto pairs with one executable. Its paired statistics and ratio of group medians are kept separately; neither is a browser estimate.",
|
||||
"instrumentation": "Profiled/control group-median differences are observational diagnostics, not isolated instrumentation overhead. Separate collection, scheduling and thermal variation can produce negative increments.",
|
||||
"ready": "Click to DOM-observed completion, not GPU completion. Paint opportunity is separately recorded.",
|
||||
"saved": "Control comparisons use IndexedDB pointer observation, including polling/scheduling latency. Exact instrumented commit timing is a separate metric.",
|
||||
"csv": "CSV click to Playwright download save completion includes automation and filesystem work. It is a separate action after solve, not another segment of click-to-ready.",
|
||||
"backend": "Spans are inclusive wall intervals on the HTTP request axis. Parent/child intervals are retained; same-name spans are summed within a run. Stage percentages use that run's own HTTP duration before aggregation.",
|
||||
"c": "C integration includes solver setup/work, events and sampling. Projection and write are inside C main. CPU and wall are distinct; RHS/Jacobian counts are not CPU shares. No process-minus-solve estimate is named output-write time.",
|
||||
"process": "Standalone process wall is subprocess creation through reap. Browser native.processWallSeconds includes Python result reading after exit; backend observed process lifetime is a separate span including spawn/exit-observation latency.",
|
||||
"overlap": "Browser receive overlaps backend execution/send; parse/decode are within reception. Render, persistence and other tasks can overlap. C main is inside process, which is inside orchestration/worker/HTTP. Never add overlapping stages or stage medians.",
|
||||
"network": "ASGI send-await time and browser outstanding-read time are not pure network measurements.",
|
||||
"cache": "Cache-hit false identifies a cold build; a warmup label alone does not. Formal browser rows must hit cache. Warmup/cold build costs are retained separately. File writes do not imply fsync.",
|
||||
"numerics": "Same input/settings and payload dimensions do not prove curve parity. Different trajectories, event times and work counts are expected between dense and experimental auto; numerical acceptance requires the separate trajectory/convergence review.",
|
||||
"portability": "Artifact paths are relative to this experiment root; repo input paths use repo-relative notation. Only recorded metadata hashes are propagated, not independently rehashed result payloads.",
|
||||
}
|
||||
|
||||
|
||||
def require(condition: bool, message: str) -> None:
|
||||
if not condition:
|
||||
raise ValueError(message)
|
||||
|
||||
|
||||
def number(value: Any) -> bool:
|
||||
return type(value) in (int, float) and math.isfinite(value)
|
||||
|
||||
|
||||
def statistics(values: list[Any]) -> dict:
|
||||
present = [v for v in values if number(v)]
|
||||
require(all(v is None or number(v) for v in values), "Invalid metric value")
|
||||
return {"n": len(present), "missing": len(values) - len(present),
|
||||
"min": min(present) if present else None, "median": median(present) if present else None,
|
||||
"max": max(present) if present else None}
|
||||
|
||||
|
||||
def field(data: dict, name: str, context: str, positive: bool = False) -> float:
|
||||
value = data.get(name)
|
||||
require(number(value) and (value > 0 if positive else value >= 0), f"{context}: invalid {name}")
|
||||
return value
|
||||
|
||||
|
||||
def select(data: dict, keys: tuple | list) -> dict:
|
||||
return {key: data.get(key) for key in keys}
|
||||
|
||||
|
||||
def metrics_summary(rows: list[dict], key: str = "metrics") -> dict:
|
||||
names = sorted({name for row in rows for name in row[key]})
|
||||
return {name: statistics([row[key].get(name) for row in rows]) for name in names}
|
||||
|
||||
|
||||
def ratio(before: dict, after: dict) -> dict:
|
||||
old, new = before["median"], after["median"]
|
||||
require(number(old) and number(new) and old > 0 and new > 0, "Missing comparison medians")
|
||||
return {"statistic": "ratio_of_group_medians", "baseline": before, "optimized": after,
|
||||
"saved": old - new, "durationReductionPercent": (old - new) / old * 100,
|
||||
"speedupRatio": old / new}
|
||||
|
||||
|
||||
class Summary:
|
||||
def __init__(self, root: Path):
|
||||
self.root = root.resolve()
|
||||
self.sources: dict[str, dict] = {}
|
||||
self.ids: set[str] = set()
|
||||
self.metric_definitions: dict[str, dict] = {}
|
||||
|
||||
def read(self, relative: str) -> dict:
|
||||
path = (self.root / relative).resolve()
|
||||
require(path.is_relative_to(self.root), f"Metadata path escapes root: {relative}")
|
||||
require(path.is_file(), f"Incomplete experiment: missing {relative}")
|
||||
require(path.stat().st_size <= 4 * 1024 * 1024, f"Refusing large metadata input: {relative}")
|
||||
raw = path.read_bytes()
|
||||
data = json.loads(raw, parse_constant=lambda token: (_ for _ in ()).throw(ValueError(token)))
|
||||
require(isinstance(data, dict), f"Expected metadata object: {relative}")
|
||||
self.sources[relative] = {"bytes": len(raw), "sha256": hashlib.sha256(raw).hexdigest()}
|
||||
return data
|
||||
|
||||
def metric(self, row: dict, domain: str, name: str, value: Any, unit: str, parent: str | None = None) -> None:
|
||||
require(value is None or number(value), f"Invalid {domain}.{name}")
|
||||
key = f"{domain}.{name}"
|
||||
row["metrics"][key] = value
|
||||
self.metric_definitions[key] = {"unit": unit, "parent": parent,
|
||||
"additive": False, "inclusiveOrOverlapping": True}
|
||||
|
||||
def native_fields(self, row: dict, data: dict, context: str, *, browser: bool) -> dict:
|
||||
require(data.get("success") is True and data.get("status") == "completed", f"{context}: simulation failed")
|
||||
for key in IDENTITY:
|
||||
require(data.get(key) is not None, f"{context}: missing {key}")
|
||||
require(data.get("simulatedUntil") == 10, f"{context}: incomplete simulation endpoint")
|
||||
for key in COUNTERS:
|
||||
require(type(data.get(key)) is int and data[key] >= 0, f"{context}: invalid count {key}")
|
||||
for key in (*COUNTERS, *JAC_COUNTERS):
|
||||
value = data.get(key)
|
||||
require(value is None or type(value) is int and value >= 0, f"{context}: invalid counter {key}")
|
||||
self.metric(row, "native", key, value, "count")
|
||||
for key in NATIVE_TIMES if browser else NATIVE_TIMES[:-1]:
|
||||
self.metric(row, "native", key, field(data, key, context), "CPU_s" if "Cpu" in key else "s")
|
||||
self.metric(row, "native", "maxAcceptedStep", field(data, "maxAcceptedStep", context), "simulation_s")
|
||||
if all(data.get(k) is not None for k in ("nfev", "cvodeRhsCalls", "cvodeLinearRhsCalls", "jacobianRhsCalls")):
|
||||
require(data["nfev"] == sum(data[k] for k in ("cvodeRhsCalls", "cvodeLinearRhsCalls", "jacobianRhsCalls")),
|
||||
f"{context}: RHS accounting mismatch")
|
||||
return select(data, [*IDENTITY, *COUNTERS, *JAC_COUNTERS, *NATIVE_TIMES, "jacobianMode", "buildKey", "cacheHit"])
|
||||
|
||||
def backend(self, row: dict, relative: str) -> dict:
|
||||
data = self.read(relative)
|
||||
require(data.get("id") == row["simulationId"] and data.get("httpStatus") == 200, f"{relative}: request ID/status mismatch")
|
||||
for key, value in row["native"].items():
|
||||
require(data.get("native", {}).get(key) == value, f"{relative}: browser/backend native {key} mismatch")
|
||||
require(data.get("sampleCount") == row["sampleCount"], f"{relative}: sample count mismatch")
|
||||
http = field(data, "httpTotalSeconds", relative, True) * 1000
|
||||
self.metric(row, "backend", "httpTotalMs", http, "ms")
|
||||
totals: dict[str, float] = {}
|
||||
spans = data.get("spans", [])
|
||||
require(bool(spans), f"{relative}: missing spans")
|
||||
annotated = []
|
||||
for index, span in enumerate(spans):
|
||||
start, end = field(span, "startMs", relative), field(span, "endMs", relative)
|
||||
require(start <= end <= http + 1e-5, f"{relative}: span outside HTTP interval")
|
||||
name = span["name"]
|
||||
totals[name] = totals.get(name, 0) + end - start
|
||||
parents = [(other["endMs"] - other["startMs"], j, other["name"]) for j, other in enumerate(spans)
|
||||
if j != index and other["startMs"] <= start and end <= other["endMs"]
|
||||
and (other["startMs"] < start or end < other["endMs"])]
|
||||
parent = min(parents)[2] if parents else "httpTotalMs"
|
||||
annotated.append({"name": name, "startMs": start, "endMs": end, "parent": parent})
|
||||
for name, value in totals.items():
|
||||
self.metric(row, "backendSpan", name, value, "ms", "backend.httpTotalMs (percentage denominator)")
|
||||
self.metric(row, "percentOfHttp", name, value / http * 100, "%", "backend.httpTotalMs")
|
||||
for name in ("requestBodyCompleteMs", "responseHeadersMs", "largeResultBodySendStartMs", "responseBodyCompleteMs"):
|
||||
self.metric(row, "backendPosition", name, data.get(name), "ms_from_request_start")
|
||||
self.metric(row, "backend", "responseSendAwaitSeconds", field(data, "responseSendAwaitSeconds", relative), "s", "backend.httpTotalMs")
|
||||
for name in ("responseBodyBytes", "rawSeriesBytes", "xmlBytes"):
|
||||
self.metric(row, "backend", name, field(data, name, relative, True), "bytes")
|
||||
stages = data.get("nativeStages", {})
|
||||
main = field(stages, "mainTotalSeconds", relative, True)
|
||||
for name in C_WALL:
|
||||
value = field(stages, name, relative)
|
||||
self.metric(row, "cWall", name, value, "s", None if name == "mainTotalSeconds" else "cWall.mainTotalSeconds")
|
||||
if name != "mainTotalSeconds":
|
||||
self.metric(row, "percentOfCMain", name, value / main * 100, "%", "cWall.mainTotalSeconds")
|
||||
for name in ("projectionCpuSeconds", "jsonWriteCpuSeconds"):
|
||||
self.metric(row, "cCpu", name, field(stages, name, relative), "CPU_s")
|
||||
process = data.get("process", {})
|
||||
require(process.get("exitCode") == 0, f"{relative}: child failed")
|
||||
for name in ("childrenUserCpuSeconds", "childrenSystemCpuSeconds"):
|
||||
self.metric(row, "process", name, field(process, name, relative), "CPU_s")
|
||||
for name, phase in data.get("existingPerformance", {}).get("phases", {}).items():
|
||||
self.metric(row, "backendExisting", name, field(phase, "inclusiveNs", relative) / 1e6, "ms", "backend.httpTotalMs")
|
||||
# Store only portable command flags; outputs and temporary filesystem paths are not needed.
|
||||
command = process.get("command", [])
|
||||
flags = {name: command[command.index(name) + 1] for name in ("--method", "--jacobian", "--start", "--stop", "--sample-step", "--max-step", "--rtol", "--timeout") if name in command}
|
||||
return {"source": relative, "simulationId": data["id"], "xmlSha256": data.get("xmlSha256"),
|
||||
"httpStatus": data["httpStatus"], "spans": annotated, "commandFlags": flags,
|
||||
"build": select(data.get("build", {}), ["cacheHit", "buildKey", "reportedSeconds"])}
|
||||
|
||||
def browser_group(self, name: str, mode: str, backend_root: str | None) -> dict:
|
||||
relative = f"browser-{name}/summary.json"
|
||||
data = self.read(relative)
|
||||
require(data.get("errors") == [], f"{name}: browser errors or missing errors field")
|
||||
rows = data.get("rows", [])
|
||||
require(len(rows) == 4 and sorted(r.get("run", -1) for r in rows) == [0, 1, 2, 3], f"Incomplete {name}: expected warmup 1 + formal 3")
|
||||
runs = []
|
||||
for source in sorted(rows, key=lambda r: r["run"]):
|
||||
context = f"{name}/{source['run']}"
|
||||
require(source.get("mode") == mode and source.get("deep") is False, f"{context}: instrumentation mode mismatch")
|
||||
require(source.get("warmup") is (source["run"] == 0), f"{context}: warmup mismatch")
|
||||
sid = source.get("simulationId")
|
||||
require(isinstance(sid, str) and sid and sid not in self.ids and Path(sid).name == sid, f"{context}: missing/duplicate/invalid ID")
|
||||
self.ids.add(sid)
|
||||
row = {"run": source["run"], "warmup": source["warmup"], "simulationId": sid, "metrics": {}}
|
||||
row["native"] = self.native_fields(row, source.get("native", {}), context, browser=True)
|
||||
require(type(row["native"]["cacheHit"]) is bool, f"{context}: missing cache hit evidence")
|
||||
require(source["warmup"] or row["native"]["cacheHit"], f"{context}: formal run contains cold build")
|
||||
require(source.get("restoredIdentical") is True, f"{context}: restore mismatch")
|
||||
for goal in GOALS.values():
|
||||
field(source, goal, context, True)
|
||||
for key, value in source.items():
|
||||
if key.endswith(("Ms", "Bytes")) or key == "streamReadCount":
|
||||
self.metric(row, "frontend", key, value, "bytes" if key.endswith("Bytes") else "count" if key == "streamReadCount" else "ms")
|
||||
for key in ("sampleCount", "variableCount"):
|
||||
row[key] = field(source, key, context, True)
|
||||
row.update(select(source, ["resultSha256", "numericalResultSha256", "csvSha256", "restoredIdentical"]))
|
||||
row["integration"] = select(source.get("integration", {}), ["method", "rtol"])
|
||||
require(row["integration"] == {"method": "BDF", "rtol": 1e-8}, f"{context}: method/rtol changed")
|
||||
row["backend"] = self.backend(row, f"{backend_root}/{sid}/stages.json") if backend_root else None
|
||||
runs.append(row)
|
||||
formal = [r for r in runs if not r["warmup"]]
|
||||
warmup = [r for r in runs if r["warmup"]]
|
||||
return {"source": relative, "mode": mode, **select(data, ["inputSha256", "buildAssetSetSha256", "browser", "node", "scriptSha256"]),
|
||||
"formal": metrics_summary(formal), "warmup": metrics_summary(warmup), "runs": runs,
|
||||
"cache": {"formalHits": sum(r["native"]["cacheHit"] for r in formal), "formalCount": len(formal),
|
||||
"coldRuns": [{"run": r["run"], "warmup": r["warmup"], "buildSeconds": r["native"]["buildSeconds"]}
|
||||
for r in runs if not r["native"]["cacheHit"]]}}
|
||||
|
||||
def native_benchmark(self) -> dict:
|
||||
data = self.read("benchmark/summary.json")
|
||||
require(data.get("complete") is True and data.get("errors") == [], "Native benchmark incomplete or execution errors")
|
||||
prepared = data["prepared"]
|
||||
settings = prepared.get("settings", {})
|
||||
require(all(settings.get(k) == v for k, v in {"method": "BDF", "rtol": 1e-8, "t_start": 0, "t_stop": 10}.items()),
|
||||
"Native method/rtol/time settings changed")
|
||||
require(prepared.get("sampleStep") == 0.01, "Native fixed sample interval changed")
|
||||
require(prepared.get("warmupsPerMode") == 1 and prepared.get("repeatsPerMode") == 3, "Native run count configuration changed")
|
||||
runs = []
|
||||
for source in data.get("rows", []):
|
||||
require(source.get("completed") is True and source.get("exitCode") == 0, "Native run failed")
|
||||
row = {**select(source, ["mode", "label", "pair", "warmup", "diagnostic", "includedInStatistics", "resultBytes", "resultSha256"]), "metrics": {}}
|
||||
row["native"] = self.native_fields(row, source, f"native/{source['mode']}/{source['label']}", browser=False)
|
||||
parts = Path(source["directory"]).parts
|
||||
require("benchmark" in parts, "Native artifact directory lacks benchmark prefix")
|
||||
row["artifactDirectory"] = Path(*parts[parts.index("benchmark"):]).as_posix()
|
||||
row["payloadMetadata"] = source.get("resultValidation")
|
||||
runs.append(row)
|
||||
groups = {}
|
||||
for mode in ("dense", "auto"):
|
||||
formal = [r for r in runs if r["mode"] == mode and r["includedInStatistics"]]
|
||||
warmup = [r for r in runs if r["mode"] == mode and r["warmup"]]
|
||||
require(len(formal) == 3 and len(warmup) == 1, f"Incomplete native {mode} repetitions")
|
||||
groups[mode] = {"formal": metrics_summary(formal), "warmup": metrics_summary(warmup)}
|
||||
verify = [r for r in runs if r["mode"] == "verify"]
|
||||
if prepared.get("verifyRequested"):
|
||||
require(len(verify) == 1 and verify[0]["diagnostic"] and not verify[0]["includedInStatistics"], "Missing separate verify run")
|
||||
return {"source": "benchmark/summary.json", "inputSha256": prepared.get("inputSha256"),
|
||||
"xmlSha256": prepared.get("xmlSha256"), "settings": prepared.get("settings"),
|
||||
"sampleStep": prepared.get("sampleStep"), "stateCount": prepared.get("stateCount"),
|
||||
"timingContract": prepared.get("timingContract"), "environment": prepared.get("environment"),
|
||||
"preparationSeconds": prepared.get("preparationSeconds"),
|
||||
"build": select(data.get("build", {}), ["buildKey", "cacheHit", "seconds"]),
|
||||
"groups": groups, "runs": runs, "speedComparison": data.get("speedComparison"),
|
||||
"strictComparisonPassed": data.get("passed"), "allPayloadBitsEqual": data.get("allPayloadBitsEqual"),
|
||||
"numericalAcceptance": data.get("numericalAcceptance")}
|
||||
|
||||
def compute_profile(self, native: dict) -> dict:
|
||||
relative = "native-compute-profile/summary.json"
|
||||
if not (self.root / relative).exists():
|
||||
return {"available": False, "source": relative,
|
||||
"reason": "Optional compute profile metadata has not been generated."}
|
||||
data = self.read(relative)
|
||||
require(data.get("allFullParity") is True, "Compute profile payload/counter parity not confirmed")
|
||||
prepared = data.get("prepared", {})
|
||||
require(prepared.get("warmups") == 1 and prepared.get("repeats") == 3, "Compute profile run count changed")
|
||||
require(Path(prepared.get("controlExecutable", "")).parent.name == native["build"]["buildKey"],
|
||||
"Compute profile uses a different native control build")
|
||||
runtime = prepared.get("runtimeArguments", [])
|
||||
require("--verify-jacobian" not in runtime and not prepared.get("verifyJacobian"),
|
||||
"Verification compute profile is diagnostic-only, not ordinary production cost")
|
||||
if "--jacobian" in runtime:
|
||||
require(runtime[runtime.index("--jacobian") + 1] == "auto", "Historical compute profile is not auto")
|
||||
else:
|
||||
require(prepared.get("algorithm") == "production-automatic" or
|
||||
(bool(data.get("runs")) and all(r.get("jacobianMode") == "colored-difference" for r in data["runs"])),
|
||||
"Compute profile lacks evidence of the production automatic algorithm")
|
||||
require("--rtol" in runtime and float(runtime[runtime.index("--rtol") + 1]) == 1e-8, "Compute profile rtol changed")
|
||||
rows = []
|
||||
for source in data.get("runs", []):
|
||||
require(source.get("fullParity") is True, "Compute profile run parity failed")
|
||||
row = {**select(source, ["variant", "run", "warmup", "fullParity"]), "metrics": {}}
|
||||
for key in ("processWallSeconds", "solveSeconds", "solveCpuSeconds"):
|
||||
self.metric(row, "computeProfile", key, field(source, key, relative, True),
|
||||
"CPU_s" if "Cpu" in key else "s")
|
||||
for key in ("nfev", "njev", "nlu", "acceptedSteps", "solverStarts"):
|
||||
self.metric(row, "computeCounter", key, field(source, key, relative), "count")
|
||||
if source.get("variant") == "profiled":
|
||||
profile = source.get("profile", {})
|
||||
require(profile.get("counterErrors") == 0, "Compute profile counter read failed")
|
||||
require(all(v is not False for v in source.get("counterChecks", {}).values()), "Compute profile counter check failed")
|
||||
for key, value in profile.get("cvodeCounters", {}).items():
|
||||
self.metric(row, "cvodeCounter", key, value, "count")
|
||||
scopes = profile.get("scopes", {})
|
||||
integration = scopes.get("integration", {})
|
||||
total = field(integration.get("integration", {}), "inclusiveSeconds", relative, True)
|
||||
for region, region_scopes in scopes.items():
|
||||
for scope, values in region_scopes.items():
|
||||
for kind in ("calls", "inclusiveSeconds", "exclusiveSeconds"):
|
||||
self.metric(row, f"scope.{region}.{scope}", kind, field(values, kind, relative),
|
||||
"count" if kind == "calls" else "s")
|
||||
if region == "integration":
|
||||
for kind in ("inclusiveSeconds", "exclusiveSeconds"):
|
||||
self.metric(row, f"scopePercentOfIntegration.{scope}", kind,
|
||||
values[kind] / total * 100, "%", "scope.integration.integration.inclusiveSeconds")
|
||||
exclusive_sum = math.fsum(s["exclusiveSeconds"] for s in integration.values())
|
||||
residual = total - exclusive_sum
|
||||
require(abs(residual) <= max(1e-9, total * 1e-9), "Compute profile scopes do not partition integration")
|
||||
row["exclusivePartition"] = {"integrationSeconds": total, "exclusiveSumSeconds": exclusive_sum,
|
||||
"residualSeconds": residual, "percentSum": exclusive_sum / total * 100}
|
||||
row["counterChecks"] = source.get("counterChecks")
|
||||
rows.append(row)
|
||||
groups = {}
|
||||
for variant in ("control", "profiled"):
|
||||
selected = [r for r in rows if r["variant"] == variant]
|
||||
require(len(selected) == 4 and {r["run"] for r in selected} == {"warmup-1", "run-1", "run-2", "run-3"},
|
||||
f"Incomplete compute profile {variant}")
|
||||
require(all(r["warmup"] is (r["run"] == "warmup-1") for r in selected), "Compute profile warmup labels changed")
|
||||
groups[variant] = {"formal": metrics_summary([r for r in selected if not r["warmup"]]),
|
||||
"warmup": metrics_summary([r for r in selected if r["warmup"]])}
|
||||
paired = []
|
||||
for label in ("run-1", "run-2", "run-3"):
|
||||
pair = {r["variant"]: r for r in rows if r["run"] == label}
|
||||
before = pair["control"]["metrics"]["computeProfile.solveSeconds"]
|
||||
after = pair["profiled"]["metrics"]["computeProfile.solveSeconds"]
|
||||
paired.append({"run": label, "controlSeconds": before, "profiledSeconds": after,
|
||||
"incrementPercent": (after / before - 1) * 100})
|
||||
return {"available": True, "source": relative, "groups": groups, "runs": rows,
|
||||
"runtimeArguments": runtime, "allFullParityRecorded": True,
|
||||
"pairedSolveIncrements": paired,
|
||||
"pairedSolveIncrementPercent": statistics([r["incrementPercent"] for r in paired]),
|
||||
"groupMedianRatioOverheadFraction": data.get("instrumentationOverheadFraction"),
|
||||
"interpretation": data.get("interpretation"),
|
||||
"scopeStatistics": "Exclusive scopes partition EACH run's integration wall time. Percentages are computed within each run before n/min/median/max; summed medians are not an exact total. Inclusive Jacobian contains its nested canonical base/probe RHS and overlaps total RHS.",
|
||||
"nonlinearFailures": "The current auto CVODE nonlinear-convergence-failure counters cover all restart segments. The previous report's 414 described dense CVODE failures in a different experiment. Neither counts pipe-local Newton exhaustion, rejected steps, or completed-run failures; do not use them as a timing share.",
|
||||
"historicalCounterSource": "repo:docs/other/八路网页求解全流程成本评估-2026-09-11.md:141"}
|
||||
|
||||
def summarize(self) -> dict:
|
||||
groups = {name: self.browser_group(name, *config) for name, config in GROUPS.items()}
|
||||
native = self.native_benchmark()
|
||||
compute = self.compute_profile(native)
|
||||
for key in ("inputSha256", "buildAssetSetSha256", "scriptSha256"):
|
||||
values = {g[key] for g in groups.values()}
|
||||
require(len(values) == 1 and isinstance(next(iter(values)), str) and len(next(iter(values))) == 64,
|
||||
f"Browser group identity mismatch/missing: {key}")
|
||||
require(native["inputSha256"] == groups["baseline"]["inputSha256"], "Native/browser input hash mismatch")
|
||||
dims = {(r["sampleCount"], r["variableCount"]) for g in groups.values() for r in g["runs"]}
|
||||
require(len(dims) == 1, "Browser sample/variable dimensions changed")
|
||||
xmls = {r["backend"]["xmlSha256"] for g in groups.values() for r in g["runs"] if r["backend"]}
|
||||
require(len(xmls) == 1 and None not in xmls, "Profiled browser XML input changed")
|
||||
controls = {goal: {"metric": f"frontend.{metric}", "unit": "ms", **ratio(groups["baseline"]["formal"][f"frontend.{metric}"], groups["optimized"]["formal"][f"frontend.{metric}"])}
|
||||
for goal, metric in GOALS.items()}
|
||||
diagnostics = {}
|
||||
for name in ("baseline", "optimized"):
|
||||
diagnostics[name] = {}
|
||||
for goal, metric in GOALS.items():
|
||||
control = groups[name]["formal"][f"frontend.{metric}"]
|
||||
profiled = groups[name + "-profiled"]["formal"][f"frontend.{metric}"]
|
||||
diagnostics[name][goal] = {"control": control, "profiled": profiled,
|
||||
"observedIncrementPercent": (profiled["median"] / control["median"] - 1) * 100,
|
||||
"statistic": "ratio_of_separately_collected_group_medians", "causalOverheadEstimate": False}
|
||||
stage_comparisons = {}
|
||||
for metric in ("native.solveSeconds", "cWall.projectionSeconds", "cWall.jsonWriteSeconds", "backendSpan.native_indexed_result_read", "backend.httpTotalMs"):
|
||||
stage_comparisons[metric] = {"diagnosticOnly": True, **ratio(groups["baseline-profiled"]["formal"][metric], groups["optimized-profiled"]["formal"][metric])}
|
||||
return {"schemaVersion": 1, "complete": True, "errors": [], "definitions": DEFINITIONS,
|
||||
"validation": {"browserInputAndAssetsAndHarnessHashesEqual": True, "nativeBrowserInputHashEqual": True,
|
||||
"profiledBrowserXmlHashesEqual": True, "nativeXmlHashEqualToBrowserXml": native["xmlSha256"] in xmls,
|
||||
"xmlIdentityNote": "Browser and standalone native XML serialization hashes are recorded separately; equality of the imported JSON is verified, semantic equivalence is not established by an XML hash mismatch alone.",
|
||||
"browserDimensionsEqual": True, "sampleCount": next(iter(dims))[0], "variableCount": next(iter(dims))[1],
|
||||
"rtol": 1e-8, "largePayloadParityCheckedHere": False, "crossModeCountersRequiredEqual": False},
|
||||
"browserGroups": groups, "nativeBenchmark": native, "nativeComputeProfile": compute, "controlComparisons": controls,
|
||||
"profiledStageComparisons": stage_comparisons, "instrumentationDiagnostics": diagnostics,
|
||||
"metricDefinitions": self.metric_definitions, "sources": self.sources}
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("--root", type=Path, default=REPO / "test/jacobian-20260911")
|
||||
parser.add_argument("--output", type=Path, help="Default: ROOT/cost-summary.json")
|
||||
args = parser.parse_args()
|
||||
summarizer = Summary(args.root)
|
||||
output = args.output or args.root / "cost-summary.json"
|
||||
try:
|
||||
result = summarizer.summarize()
|
||||
except (OSError, ValueError, KeyError, TypeError, IndexError) as exc:
|
||||
result = {"schemaVersion": 1, "complete": False, "errors": [str(exc)],
|
||||
"definitions": DEFINITIONS, "sources": summarizer.sources}
|
||||
result["scriptSha256"] = hashlib.sha256(Path(__file__).read_bytes()).hexdigest()
|
||||
output.parent.mkdir(parents=True, exist_ok=True)
|
||||
output.write_text(json.dumps(result, ensure_ascii=False, indent=2, allow_nan=False) + "\n", encoding="utf-8")
|
||||
print(json.dumps({"complete": result["complete"], "output": str(output), "errors": result["errors"]}, ensure_ascii=False))
|
||||
return 0 if result["complete"] else 2
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
Reference in new issue
Block a user