Files
SystemSimulationApp/tests/manual/summarize_jacobian_cost.py

408 lines
30 KiB
Python

"""Summarize Jacobian timing metadata without opening results or CSV files.
python3 tests/manual/summarize_jacobian_cost.py --root test/jacobian-20260911
Requires four complete browser groups (one warmup and three formal runs each)
plus the standalone native benchmark. Writes ROOT/cost-summary.json by default.
Incomplete evidence overwrites the destination with complete:false and exits 2.
Numerical differences are recorded, never interpreted as numerical acceptance.
"""
from __future__ import annotations
import argparse
import hashlib
import json
import math
from pathlib import Path
from statistics import median
from typing import Any
REPO = Path(__file__).resolve().parents[2]
GROUPS = {
"baseline": ("control", None),
"optimized": ("control", None),
"baseline-profiled": ("profiled", "baseline-source/profiled-backend/requests"),
"optimized-profiled": ("profiled", "backend-optimized-profiled/requests"),
}
GOALS = {"ready": "clickToReadyDomMs", "saved_observed": "clickToIndexedDbObservedMs",
"csv_download_saved": "csvClickToDownloadSavedMs"}
COUNTERS = ("nfev", "acceptedSteps", "rejectedSteps", "stateTransitions", "solverStarts", "njev", "nlu")
JAC_COUNTERS = ("jacobianRhsCalls", "jacobianColoredEvals", "jacobianFallbacks", "jacobianChecks",
"jacobianMismatches", "cvodeRhsCalls", "cvodeLinearRhsCalls")
IDENTITY = ("backend", "method", "solver", "sundialsVersion", "simulatedUntil", "maxAcceptedStep")
NATIVE_TIMES = ("solveSeconds", "solveCpuSeconds", "processWallSeconds", "buildSeconds")
C_WALL = ("argumentPreparationSeconds", "initializationSeconds", "integrationSeconds",
"finalSampleAndStatusSeconds", "projectionSeconds", "jsonWriteSeconds", "mainTotalSeconds")
DEFINITIONS = {
"scope": "Jacobian experiment only; metadata summaries, no result arrays or CSV contents are opened. complete means timing evidence is complete, not numerical acceptance.",
"statistics": "Formal runs and warmups are separate. Each metric reports n/min/median/max and missing count. A missing baseline Jacobian field means unavailable, not zero.",
"browserComparisons": "The separately collected control groups provide end-to-end changes: reduction = 100*(baseline median-optimized median)/baseline median. Run ordinals are not paired trials. Warmups are excluded.",
"nativeComparisons": "The native benchmark alternated serial dense/auto pairs with one executable. Its paired statistics and ratio of group medians are kept separately; neither is a browser estimate.",
"instrumentation": "Profiled/control group-median differences are observational diagnostics, not isolated instrumentation overhead. Separate collection, scheduling and thermal variation can produce negative increments.",
"ready": "Click to DOM-observed completion, not GPU completion. Paint opportunity is separately recorded.",
"saved": "Control comparisons use IndexedDB pointer observation, including polling/scheduling latency. Exact instrumented commit timing is a separate metric.",
"csv": "CSV click to Playwright download save completion includes automation and filesystem work. It is a separate action after solve, not another segment of click-to-ready.",
"backend": "Spans are inclusive wall intervals on the HTTP request axis. Parent/child intervals are retained; same-name spans are summed within a run. Stage percentages use that run's own HTTP duration before aggregation.",
"c": "C integration includes solver setup/work, events and sampling. Projection and write are inside C main. CPU and wall are distinct; RHS/Jacobian counts are not CPU shares. No process-minus-solve estimate is named output-write time.",
"process": "Standalone process wall is subprocess creation through reap. Browser native.processWallSeconds includes Python result reading after exit; backend observed process lifetime is a separate span including spawn/exit-observation latency.",
"overlap": "Browser receive overlaps backend execution/send; parse/decode are within reception. Render, persistence and other tasks can overlap. C main is inside process, which is inside orchestration/worker/HTTP. Never add overlapping stages or stage medians.",
"network": "ASGI send-await time and browser outstanding-read time are not pure network measurements.",
"cache": "Cache-hit false identifies a cold build; a warmup label alone does not. Formal browser rows must hit cache. Warmup/cold build costs are retained separately. File writes do not imply fsync.",
"numerics": "Same input/settings and payload dimensions do not prove curve parity. Different trajectories, event times and work counts are expected between dense and experimental auto; numerical acceptance requires the separate trajectory/convergence review.",
"portability": "Artifact paths are relative to this experiment root; repo input paths use repo-relative notation. Only recorded metadata hashes are propagated, not independently rehashed result payloads.",
}
def require(condition: bool, message: str) -> None:
if not condition:
raise ValueError(message)
def number(value: Any) -> bool:
return type(value) in (int, float) and math.isfinite(value)
def statistics(values: list[Any]) -> dict:
present = [v for v in values if number(v)]
require(all(v is None or number(v) for v in values), "Invalid metric value")
return {"n": len(present), "missing": len(values) - len(present),
"min": min(present) if present else None, "median": median(present) if present else None,
"max": max(present) if present else None}
def field(data: dict, name: str, context: str, positive: bool = False) -> float:
value = data.get(name)
require(number(value) and (value > 0 if positive else value >= 0), f"{context}: invalid {name}")
return value
def select(data: dict, keys: tuple | list) -> dict:
return {key: data.get(key) for key in keys}
def metrics_summary(rows: list[dict], key: str = "metrics") -> dict:
names = sorted({name for row in rows for name in row[key]})
return {name: statistics([row[key].get(name) for row in rows]) for name in names}
def ratio(before: dict, after: dict) -> dict:
old, new = before["median"], after["median"]
require(number(old) and number(new) and old > 0 and new > 0, "Missing comparison medians")
return {"statistic": "ratio_of_group_medians", "baseline": before, "optimized": after,
"saved": old - new, "durationReductionPercent": (old - new) / old * 100,
"speedupRatio": old / new}
class Summary:
def __init__(self, root: Path):
self.root = root.resolve()
self.sources: dict[str, dict] = {}
self.ids: set[str] = set()
self.metric_definitions: dict[str, dict] = {}
def read(self, relative: str) -> dict:
path = (self.root / relative).resolve()
require(path.is_relative_to(self.root), f"Metadata path escapes root: {relative}")
require(path.is_file(), f"Incomplete experiment: missing {relative}")
require(path.stat().st_size <= 4 * 1024 * 1024, f"Refusing large metadata input: {relative}")
raw = path.read_bytes()
data = json.loads(raw, parse_constant=lambda token: (_ for _ in ()).throw(ValueError(token)))
require(isinstance(data, dict), f"Expected metadata object: {relative}")
self.sources[relative] = {"bytes": len(raw), "sha256": hashlib.sha256(raw).hexdigest()}
return data
def metric(self, row: dict, domain: str, name: str, value: Any, unit: str, parent: str | None = None) -> None:
require(value is None or number(value), f"Invalid {domain}.{name}")
key = f"{domain}.{name}"
row["metrics"][key] = value
self.metric_definitions[key] = {"unit": unit, "parent": parent,
"additive": False, "inclusiveOrOverlapping": True}
def native_fields(self, row: dict, data: dict, context: str, *, browser: bool) -> dict:
require(data.get("success") is True and data.get("status") == "completed", f"{context}: simulation failed")
for key in IDENTITY:
require(data.get(key) is not None, f"{context}: missing {key}")
require(data.get("simulatedUntil") == 10, f"{context}: incomplete simulation endpoint")
for key in COUNTERS:
require(type(data.get(key)) is int and data[key] >= 0, f"{context}: invalid count {key}")
for key in (*COUNTERS, *JAC_COUNTERS):
value = data.get(key)
require(value is None or type(value) is int and value >= 0, f"{context}: invalid counter {key}")
self.metric(row, "native", key, value, "count")
for key in NATIVE_TIMES if browser else NATIVE_TIMES[:-1]:
self.metric(row, "native", key, field(data, key, context), "CPU_s" if "Cpu" in key else "s")
self.metric(row, "native", "maxAcceptedStep", field(data, "maxAcceptedStep", context), "simulation_s")
if all(data.get(k) is not None for k in ("nfev", "cvodeRhsCalls", "cvodeLinearRhsCalls", "jacobianRhsCalls")):
require(data["nfev"] == sum(data[k] for k in ("cvodeRhsCalls", "cvodeLinearRhsCalls", "jacobianRhsCalls")),
f"{context}: RHS accounting mismatch")
return select(data, [*IDENTITY, *COUNTERS, *JAC_COUNTERS, *NATIVE_TIMES, "jacobianMode", "buildKey", "cacheHit"])
def backend(self, row: dict, relative: str) -> dict:
data = self.read(relative)
require(data.get("id") == row["simulationId"] and data.get("httpStatus") == 200, f"{relative}: request ID/status mismatch")
for key, value in row["native"].items():
require(data.get("native", {}).get(key) == value, f"{relative}: browser/backend native {key} mismatch")
require(data.get("sampleCount") == row["sampleCount"], f"{relative}: sample count mismatch")
http = field(data, "httpTotalSeconds", relative, True) * 1000
self.metric(row, "backend", "httpTotalMs", http, "ms")
totals: dict[str, float] = {}
spans = data.get("spans", [])
require(bool(spans), f"{relative}: missing spans")
annotated = []
for index, span in enumerate(spans):
start, end = field(span, "startMs", relative), field(span, "endMs", relative)
require(start <= end <= http + 1e-5, f"{relative}: span outside HTTP interval")
name = span["name"]
totals[name] = totals.get(name, 0) + end - start
parents = [(other["endMs"] - other["startMs"], j, other["name"]) for j, other in enumerate(spans)
if j != index and other["startMs"] <= start and end <= other["endMs"]
and (other["startMs"] < start or end < other["endMs"])]
parent = min(parents)[2] if parents else "httpTotalMs"
annotated.append({"name": name, "startMs": start, "endMs": end, "parent": parent})
for name, value in totals.items():
self.metric(row, "backendSpan", name, value, "ms", "backend.httpTotalMs (percentage denominator)")
self.metric(row, "percentOfHttp", name, value / http * 100, "%", "backend.httpTotalMs")
for name in ("requestBodyCompleteMs", "responseHeadersMs", "largeResultBodySendStartMs", "responseBodyCompleteMs"):
self.metric(row, "backendPosition", name, data.get(name), "ms_from_request_start")
self.metric(row, "backend", "responseSendAwaitSeconds", field(data, "responseSendAwaitSeconds", relative), "s", "backend.httpTotalMs")
for name in ("responseBodyBytes", "rawSeriesBytes", "xmlBytes"):
self.metric(row, "backend", name, field(data, name, relative, True), "bytes")
stages = data.get("nativeStages", {})
main = field(stages, "mainTotalSeconds", relative, True)
for name in C_WALL:
value = field(stages, name, relative)
self.metric(row, "cWall", name, value, "s", None if name == "mainTotalSeconds" else "cWall.mainTotalSeconds")
if name != "mainTotalSeconds":
self.metric(row, "percentOfCMain", name, value / main * 100, "%", "cWall.mainTotalSeconds")
for name in ("projectionCpuSeconds", "jsonWriteCpuSeconds"):
self.metric(row, "cCpu", name, field(stages, name, relative), "CPU_s")
process = data.get("process", {})
require(process.get("exitCode") == 0, f"{relative}: child failed")
for name in ("childrenUserCpuSeconds", "childrenSystemCpuSeconds"):
self.metric(row, "process", name, field(process, name, relative), "CPU_s")
for name, phase in data.get("existingPerformance", {}).get("phases", {}).items():
self.metric(row, "backendExisting", name, field(phase, "inclusiveNs", relative) / 1e6, "ms", "backend.httpTotalMs")
# Store only portable command flags; outputs and temporary filesystem paths are not needed.
command = process.get("command", [])
flags = {name: command[command.index(name) + 1] for name in ("--method", "--jacobian", "--start", "--stop", "--sample-step", "--max-step", "--rtol", "--timeout") if name in command}
return {"source": relative, "simulationId": data["id"], "xmlSha256": data.get("xmlSha256"),
"httpStatus": data["httpStatus"], "spans": annotated, "commandFlags": flags,
"build": select(data.get("build", {}), ["cacheHit", "buildKey", "reportedSeconds"])}
def browser_group(self, name: str, mode: str, backend_root: str | None) -> dict:
relative = f"browser-{name}/summary.json"
data = self.read(relative)
require(data.get("errors") == [], f"{name}: browser errors or missing errors field")
rows = data.get("rows", [])
require(len(rows) == 4 and sorted(r.get("run", -1) for r in rows) == [0, 1, 2, 3], f"Incomplete {name}: expected warmup 1 + formal 3")
runs = []
for source in sorted(rows, key=lambda r: r["run"]):
context = f"{name}/{source['run']}"
require(source.get("mode") == mode and source.get("deep") is False, f"{context}: instrumentation mode mismatch")
require(source.get("warmup") is (source["run"] == 0), f"{context}: warmup mismatch")
sid = source.get("simulationId")
require(isinstance(sid, str) and sid and sid not in self.ids and Path(sid).name == sid, f"{context}: missing/duplicate/invalid ID")
self.ids.add(sid)
row = {"run": source["run"], "warmup": source["warmup"], "simulationId": sid, "metrics": {}}
row["native"] = self.native_fields(row, source.get("native", {}), context, browser=True)
require(type(row["native"]["cacheHit"]) is bool, f"{context}: missing cache hit evidence")
require(source["warmup"] or row["native"]["cacheHit"], f"{context}: formal run contains cold build")
require(source.get("restoredIdentical") is True, f"{context}: restore mismatch")
for goal in GOALS.values():
field(source, goal, context, True)
for key, value in source.items():
if key.endswith(("Ms", "Bytes")) or key == "streamReadCount":
self.metric(row, "frontend", key, value, "bytes" if key.endswith("Bytes") else "count" if key == "streamReadCount" else "ms")
for key in ("sampleCount", "variableCount"):
row[key] = field(source, key, context, True)
row.update(select(source, ["resultSha256", "numericalResultSha256", "csvSha256", "restoredIdentical"]))
row["integration"] = select(source.get("integration", {}), ["method", "rtol"])
require(row["integration"] == {"method": "BDF", "rtol": 1e-8}, f"{context}: method/rtol changed")
row["backend"] = self.backend(row, f"{backend_root}/{sid}/stages.json") if backend_root else None
runs.append(row)
formal = [r for r in runs if not r["warmup"]]
warmup = [r for r in runs if r["warmup"]]
return {"source": relative, "mode": mode, **select(data, ["inputSha256", "buildAssetSetSha256", "browser", "node", "scriptSha256"]),
"formal": metrics_summary(formal), "warmup": metrics_summary(warmup), "runs": runs,
"cache": {"formalHits": sum(r["native"]["cacheHit"] for r in formal), "formalCount": len(formal),
"coldRuns": [{"run": r["run"], "warmup": r["warmup"], "buildSeconds": r["native"]["buildSeconds"]}
for r in runs if not r["native"]["cacheHit"]]}}
def native_benchmark(self) -> dict:
data = self.read("benchmark/summary.json")
require(data.get("complete") is True and data.get("errors") == [], "Native benchmark incomplete or execution errors")
prepared = data["prepared"]
settings = prepared.get("settings", {})
require(all(settings.get(k) == v for k, v in {"method": "BDF", "rtol": 1e-8, "t_start": 0, "t_stop": 10}.items()),
"Native method/rtol/time settings changed")
require(prepared.get("sampleStep") == 0.01, "Native fixed sample interval changed")
require(prepared.get("warmupsPerMode") == 1 and prepared.get("repeatsPerMode") == 3, "Native run count configuration changed")
runs = []
for source in data.get("rows", []):
require(source.get("completed") is True and source.get("exitCode") == 0, "Native run failed")
row = {**select(source, ["mode", "label", "pair", "warmup", "diagnostic", "includedInStatistics", "resultBytes", "resultSha256"]), "metrics": {}}
row["native"] = self.native_fields(row, source, f"native/{source['mode']}/{source['label']}", browser=False)
parts = Path(source["directory"]).parts
require("benchmark" in parts, "Native artifact directory lacks benchmark prefix")
row["artifactDirectory"] = Path(*parts[parts.index("benchmark"):]).as_posix()
row["payloadMetadata"] = source.get("resultValidation")
runs.append(row)
groups = {}
for mode in ("dense", "auto"):
formal = [r for r in runs if r["mode"] == mode and r["includedInStatistics"]]
warmup = [r for r in runs if r["mode"] == mode and r["warmup"]]
require(len(formal) == 3 and len(warmup) == 1, f"Incomplete native {mode} repetitions")
groups[mode] = {"formal": metrics_summary(formal), "warmup": metrics_summary(warmup)}
verify = [r for r in runs if r["mode"] == "verify"]
if prepared.get("verifyRequested"):
require(len(verify) == 1 and verify[0]["diagnostic"] and not verify[0]["includedInStatistics"], "Missing separate verify run")
return {"source": "benchmark/summary.json", "inputSha256": prepared.get("inputSha256"),
"xmlSha256": prepared.get("xmlSha256"), "settings": prepared.get("settings"),
"sampleStep": prepared.get("sampleStep"), "stateCount": prepared.get("stateCount"),
"timingContract": prepared.get("timingContract"), "environment": prepared.get("environment"),
"preparationSeconds": prepared.get("preparationSeconds"),
"build": select(data.get("build", {}), ["buildKey", "cacheHit", "seconds"]),
"groups": groups, "runs": runs, "speedComparison": data.get("speedComparison"),
"strictComparisonPassed": data.get("passed"), "allPayloadBitsEqual": data.get("allPayloadBitsEqual"),
"numericalAcceptance": data.get("numericalAcceptance")}
def compute_profile(self, native: dict) -> dict:
relative = "native-compute-profile/summary.json"
if not (self.root / relative).exists():
return {"available": False, "source": relative,
"reason": "Optional compute profile metadata has not been generated."}
data = self.read(relative)
require(data.get("allFullParity") is True, "Compute profile payload/counter parity not confirmed")
prepared = data.get("prepared", {})
require(prepared.get("warmups") == 1 and prepared.get("repeats") == 3, "Compute profile run count changed")
require(Path(prepared.get("controlExecutable", "")).parent.name == native["build"]["buildKey"],
"Compute profile uses a different native control build")
runtime = prepared.get("runtimeArguments", [])
require("--verify-jacobian" not in runtime and not prepared.get("verifyJacobian"),
"Verification compute profile is diagnostic-only, not ordinary production cost")
if "--jacobian" in runtime:
require(runtime[runtime.index("--jacobian") + 1] == "auto", "Historical compute profile is not auto")
else:
require(prepared.get("algorithm") == "production-automatic" or
(bool(data.get("runs")) and all(r.get("jacobianMode") == "colored-difference" for r in data["runs"])),
"Compute profile lacks evidence of the production automatic algorithm")
require("--rtol" in runtime and float(runtime[runtime.index("--rtol") + 1]) == 1e-8, "Compute profile rtol changed")
rows = []
for source in data.get("runs", []):
require(source.get("fullParity") is True, "Compute profile run parity failed")
row = {**select(source, ["variant", "run", "warmup", "fullParity"]), "metrics": {}}
for key in ("processWallSeconds", "solveSeconds", "solveCpuSeconds"):
self.metric(row, "computeProfile", key, field(source, key, relative, True),
"CPU_s" if "Cpu" in key else "s")
for key in ("nfev", "njev", "nlu", "acceptedSteps", "solverStarts"):
self.metric(row, "computeCounter", key, field(source, key, relative), "count")
if source.get("variant") == "profiled":
profile = source.get("profile", {})
require(profile.get("counterErrors") == 0, "Compute profile counter read failed")
require(all(v is not False for v in source.get("counterChecks", {}).values()), "Compute profile counter check failed")
for key, value in profile.get("cvodeCounters", {}).items():
self.metric(row, "cvodeCounter", key, value, "count")
scopes = profile.get("scopes", {})
integration = scopes.get("integration", {})
total = field(integration.get("integration", {}), "inclusiveSeconds", relative, True)
for region, region_scopes in scopes.items():
for scope, values in region_scopes.items():
for kind in ("calls", "inclusiveSeconds", "exclusiveSeconds"):
self.metric(row, f"scope.{region}.{scope}", kind, field(values, kind, relative),
"count" if kind == "calls" else "s")
if region == "integration":
for kind in ("inclusiveSeconds", "exclusiveSeconds"):
self.metric(row, f"scopePercentOfIntegration.{scope}", kind,
values[kind] / total * 100, "%", "scope.integration.integration.inclusiveSeconds")
exclusive_sum = math.fsum(s["exclusiveSeconds"] for s in integration.values())
residual = total - exclusive_sum
require(abs(residual) <= max(1e-9, total * 1e-9), "Compute profile scopes do not partition integration")
row["exclusivePartition"] = {"integrationSeconds": total, "exclusiveSumSeconds": exclusive_sum,
"residualSeconds": residual, "percentSum": exclusive_sum / total * 100}
row["counterChecks"] = source.get("counterChecks")
rows.append(row)
groups = {}
for variant in ("control", "profiled"):
selected = [r for r in rows if r["variant"] == variant]
require(len(selected) == 4 and {r["run"] for r in selected} == {"warmup-1", "run-1", "run-2", "run-3"},
f"Incomplete compute profile {variant}")
require(all(r["warmup"] is (r["run"] == "warmup-1") for r in selected), "Compute profile warmup labels changed")
groups[variant] = {"formal": metrics_summary([r for r in selected if not r["warmup"]]),
"warmup": metrics_summary([r for r in selected if r["warmup"]])}
paired = []
for label in ("run-1", "run-2", "run-3"):
pair = {r["variant"]: r for r in rows if r["run"] == label}
before = pair["control"]["metrics"]["computeProfile.solveSeconds"]
after = pair["profiled"]["metrics"]["computeProfile.solveSeconds"]
paired.append({"run": label, "controlSeconds": before, "profiledSeconds": after,
"incrementPercent": (after / before - 1) * 100})
return {"available": True, "source": relative, "groups": groups, "runs": rows,
"runtimeArguments": runtime, "allFullParityRecorded": True,
"pairedSolveIncrements": paired,
"pairedSolveIncrementPercent": statistics([r["incrementPercent"] for r in paired]),
"groupMedianRatioOverheadFraction": data.get("instrumentationOverheadFraction"),
"interpretation": data.get("interpretation"),
"scopeStatistics": "Exclusive scopes partition EACH run's integration wall time. Percentages are computed within each run before n/min/median/max; summed medians are not an exact total. Inclusive Jacobian contains its nested canonical base/probe RHS and overlaps total RHS.",
"nonlinearFailures": "The current auto CVODE nonlinear-convergence-failure counters cover all restart segments. The previous report's 414 described dense CVODE failures in a different experiment. Neither counts pipe-local Newton exhaustion, rejected steps, or completed-run failures; do not use them as a timing share.",
"historicalCounterSource": "repo:docs/other/八路网页求解全流程成本评估-2026-09-11.md:141"}
def summarize(self) -> dict:
groups = {name: self.browser_group(name, *config) for name, config in GROUPS.items()}
native = self.native_benchmark()
compute = self.compute_profile(native)
for key in ("inputSha256", "buildAssetSetSha256", "scriptSha256"):
values = {g[key] for g in groups.values()}
require(len(values) == 1 and isinstance(next(iter(values)), str) and len(next(iter(values))) == 64,
f"Browser group identity mismatch/missing: {key}")
require(native["inputSha256"] == groups["baseline"]["inputSha256"], "Native/browser input hash mismatch")
dims = {(r["sampleCount"], r["variableCount"]) for g in groups.values() for r in g["runs"]}
require(len(dims) == 1, "Browser sample/variable dimensions changed")
xmls = {r["backend"]["xmlSha256"] for g in groups.values() for r in g["runs"] if r["backend"]}
require(len(xmls) == 1 and None not in xmls, "Profiled browser XML input changed")
controls = {goal: {"metric": f"frontend.{metric}", "unit": "ms", **ratio(groups["baseline"]["formal"][f"frontend.{metric}"], groups["optimized"]["formal"][f"frontend.{metric}"])}
for goal, metric in GOALS.items()}
diagnostics = {}
for name in ("baseline", "optimized"):
diagnostics[name] = {}
for goal, metric in GOALS.items():
control = groups[name]["formal"][f"frontend.{metric}"]
profiled = groups[name + "-profiled"]["formal"][f"frontend.{metric}"]
diagnostics[name][goal] = {"control": control, "profiled": profiled,
"observedIncrementPercent": (profiled["median"] / control["median"] - 1) * 100,
"statistic": "ratio_of_separately_collected_group_medians", "causalOverheadEstimate": False}
stage_comparisons = {}
for metric in ("native.solveSeconds", "cWall.projectionSeconds", "cWall.jsonWriteSeconds", "backendSpan.native_indexed_result_read", "backend.httpTotalMs"):
stage_comparisons[metric] = {"diagnosticOnly": True, **ratio(groups["baseline-profiled"]["formal"][metric], groups["optimized-profiled"]["formal"][metric])}
return {"schemaVersion": 1, "complete": True, "errors": [], "definitions": DEFINITIONS,
"validation": {"browserInputAndAssetsAndHarnessHashesEqual": True, "nativeBrowserInputHashEqual": True,
"profiledBrowserXmlHashesEqual": True, "nativeXmlHashEqualToBrowserXml": native["xmlSha256"] in xmls,
"xmlIdentityNote": "Browser and standalone native XML serialization hashes are recorded separately; equality of the imported JSON is verified, semantic equivalence is not established by an XML hash mismatch alone.",
"browserDimensionsEqual": True, "sampleCount": next(iter(dims))[0], "variableCount": next(iter(dims))[1],
"rtol": 1e-8, "largePayloadParityCheckedHere": False, "crossModeCountersRequiredEqual": False},
"browserGroups": groups, "nativeBenchmark": native, "nativeComputeProfile": compute, "controlComparisons": controls,
"profiledStageComparisons": stage_comparisons, "instrumentationDiagnostics": diagnostics,
"metricDefinitions": self.metric_definitions, "sources": self.sources}
def main() -> int:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--root", type=Path, default=REPO / "test/jacobian-20260911")
parser.add_argument("--output", type=Path, help="Default: ROOT/cost-summary.json")
args = parser.parse_args()
summarizer = Summary(args.root)
output = args.output or args.root / "cost-summary.json"
try:
result = summarizer.summarize()
except (OSError, ValueError, KeyError, TypeError, IndexError) as exc:
result = {"schemaVersion": 1, "complete": False, "errors": [str(exc)],
"definitions": DEFINITIONS, "sources": summarizer.sources}
result["scriptSha256"] = hashlib.sha256(Path(__file__).read_bytes()).hexdigest()
output.parent.mkdir(parents=True, exist_ok=True)
output.write_text(json.dumps(result, ensure_ascii=False, indent=2, allow_nan=False) + "\n", encoding="utf-8")
print(json.dumps({"complete": result["complete"], "output": str(output), "errors": result["errors"]}, ensure_ascii=False))
return 0 if result["complete"] else 2
if __name__ == "__main__":
raise SystemExit(main())