"""Summarize the four complete C-result-encoding browser groups. .venv/bin/python tests/manual/summarize_native_encoding.py \ --root test/c-result-encoding-20260911 Reads small timing metadata only, never result arrays or CSV contents. Every browser group must contain one warmup and three measured successful runs. Missing/inconsistent evidence exits 2 and writes complete:false plus an empty CSV, so an earlier successful summary cannot masquerade as current evidence. """ from __future__ import annotations import argparse import csv import hashlib import json import math from pathlib import Path from statistics import median import sys from typing import Any REPO = Path(__file__).resolve().parents[2] GROUPS = { "baseline": ("control", None), "baseline-profiled": ("profiled", "baseline-source/profiled-backend/requests"), "optimized": ("control", None), "optimized-profiled": ("profiled", "backend-optimized-profiled/requests"), } GOALS = { "ready": "clickToReadyDomMs", "saved_observed": "clickToIndexedDbObservedMs", "csv_download_saved": "csvClickToDownloadSavedMs", } COUNTERS = ("nfev", "acceptedSteps", "rejectedSteps", "stateTransitions", "solverStarts", "njev", "nlu") NATIVE_IDENTITY = ("backend", "method", "solver", "sundialsVersion", "simulatedUntil", "maxAcceptedStep", *COUNTERS) C_WALL = ("argumentPreparationSeconds", "initializationSeconds", "integrationSeconds", "finalSampleAndStatusSeconds", "projectionSeconds", "jsonWriteSeconds", "mainTotalSeconds") CSV_FIELDS = ("group", "phase", "run", "simulationId", "domain", "metric", "statistic", "value", "unit", "parent", "denominatorValue", "percentOfParent", "baselineValue", "optimizedValue", "count", "missingCount", "inclusion", "source") DEFINITIONS = { "scope": "One C-result-encoding experiment. Baseline/optimized control groups alone provide end-to-end comparisons; profiled C-write comparisons are separate diagnostics.", "statistics": "Run values precede median/min/max. Stage/parent percentages use each run's own denominator before aggregation. Before/after changes use the ratio of independently collected group medians; medians and overlapping stages must not be added.", "comparison": "Groups were collected separately, not as alternating paired trials. Duration reduction = (baseline median - optimized median) / baseline median; speedup = baseline median / optimized median. Run ordinals are not matched pairs. Ordering, scheduling and thermal variability remain possible.", "ready": "Click to DOM-observed successful completion and an enabled Run button, not GPU completion.", "saved": "End-to-end comparisons use pointer polling observation in BOTH control groups; includes polling and scheduling latency. Exact instrumented pointer publication remains a separate profiled metric.", "csv": "Click through Playwright download notification and saveAs completion; includes automation and filesystem work.", "backend": "Backend spans are inclusive wall intervals; children are included in their parents. ASGI send awaits are not pure network time. Response serialization includes metadata encoding and raw numeric-fragment joining.", "cWrite": "C output write includes numeric encoding, stdio writes, close and index writing. CPU and wall are distinct observations; their difference is not an isolated disk-I/O measurement.", "cSolve": "Integration includes CVODE setup, RHS/Jacobian/linear work, events and sampling. Counts are not CPU-time shares. Projection is outside integration and inside C main.", "process": "Native reported processWallSeconds includes Python result reading after child exit; observed process lifetime spans include spawn and exit-observation latency.", "overlap": "Browser reads overlap backend work. Parse/decode lie inside reception; persistence and rendering overlap. Per-stage percentages are inclusive and must not be added.", "warmup": "Warmup rows are retained separately. Formal browser runs require build-cache hits. CacheHit, not the warmup label, identifies cold compilation.", "replay": "Microbenchmark replays preloaded contiguous binary64 values, excluding model projection and production strided access. Its medians are separate and never substituted for end-to-end results.", "validation": "Timing metadata validates group completeness, input/assets, counters, sample/variable counts and recorded success/restore flags. It does not independently prove numerical bitwise parity; use the separate full-result comparison artifact.", } def numeric(value: Any) -> bool: return type(value) in (int, float) and math.isfinite(value) def require(condition: bool, message: str) -> None: if not condition: raise ValueError(message) def stats(values: list[Any]) -> dict: present = [v for v in values if numeric(v)] return {"count": len(present), "missingCount": len(values) - len(present), "values": values, "median": median(present) if present else None, "min": min(present) if present else None, "max": max(present) if present else None} def percent(value: Any, denominator: Any) -> float | None: return value / denominator * 100 if numeric(value) and numeric(denominator) and denominator > 0 else None def finite_field(data: dict, name: str, context: str, *, positive: bool = False) -> float: value = data.get(name) require(numeric(value) and (value > 0 if positive else value >= 0), f"{context}: missing/invalid {name}") return value class Summarizer: def __init__(self, root: Path): self.root = root.resolve() self.sources: dict[str, dict] = {} self.observations: list[dict] = [] self.simulation_ids: set[str] = set() def read(self, relative: str) -> dict: path = (self.root / relative).resolve() require(path.is_relative_to(self.root), f"Metadata path leaves experiment root: {relative}") require(path.is_file(), f"Incomplete experiment: missing {relative}") require(path.stat().st_size <= 4 * 1024 * 1024, f"Refusing large non-metadata input: {relative}") raw = path.read_bytes() value = json.loads(raw, parse_constant=lambda token: (_ for _ in ()).throw(ValueError(f"Invalid JSON number: {token}"))) require(isinstance(value, dict), f"Expected metadata object: {relative}") self.sources[relative] = {"bytes": len(raw), "sha256": hashlib.sha256(raw).hexdigest()} return value def observe(self, run: dict, domain: str, metric: str, value: Any, unit: str = "ms", *, parent: str = "", denominator: Any = None, inclusion: str = "inclusive/overlapping; not additive", source: str = "", baseline: Any = None, optimized: Any = None) -> None: require(value is None or numeric(value), f"Invalid observation {domain}.{metric}: {value!r}") self.observations.append({k: run[k] for k in ("group", "phase", "run", "simulationId")} | { "domain": domain, "metric": metric, "value": value, "unit": unit, "parent": parent, "denominatorValue": denominator, "percentOfParent": percent(value, denominator), "baselineValue": baseline, "optimizedValue": optimized, "inclusion": inclusion, "source": source}) def backend(self, run: dict, relative: str) -> dict: data = self.read(relative) context = f"{run['group']}/{run['run']} backend" require(data.get("id") == run["simulationId"], f"{context}: simulationId mismatch") require(data.get("httpStatus") == 200, f"{context}: HTTP did not succeed") native = run["native"] for key in (*NATIVE_IDENTITY, "solveSeconds", "solveCpuSeconds", "buildKey", "cacheHit"): require(data.get("native", {}).get(key) == native.get(key), f"{context}: browser/backend mismatch for {key}") require(data.get("sampleCount") == run["sampleCount"], f"{context}: backend sampleCount mismatch") http = finite_field(data, "httpTotalSeconds", context, positive=True) * 1000 self.observe(run, "backend", "httpTotalMs", http, source=relative, inclusion=DEFINITIONS["backend"]) spans = data.get("spans", []) require(bool(spans), f"{context}: missing backend spans") totals: dict[str, float] = {} for span in spans: start = finite_field(span, "startMs", context) end = finite_field(span, "endMs", context) require(start <= end <= http + 1e-5, f"{context}: span outside HTTP interval: {span['name']}") totals[span["name"]] = totals.get(span["name"], 0) + end - start for name in ("native_indexed_result_read", "native_process_lifetime_observed", "response_result_json_serialization"): require(name in totals, f"{context}: missing {name}") for name, value in totals.items(): self.observe(run, "backend_span", name, value, parent="httpTotalMs", denominator=http, inclusion=DEFINITIONS["backend"], source=relative) c = data.get("nativeStages", {}) main_ms = finite_field(c, "mainTotalSeconds", context, positive=True) * 1000 for name in C_WALL: self.observe(run, "c_wall", name.removesuffix("Seconds") + "Ms", finite_field(c, name, context) * 1000, parent="cMainMs" if name != "mainTotalSeconds" else "", denominator=main_ms if name != "mainTotalSeconds" else None, inclusion=DEFINITIONS["cWrite"] if name == "jsonWriteSeconds" else DEFINITIONS["cSolve"], source=relative) for name in ("projectionCpuSeconds", "jsonWriteCpuSeconds"): self.observe(run, "c_cpu", name.removesuffix("Seconds") + "Ms", finite_field(c, name, context) * 1000, inclusion="CPU duration; separate from wall intervals", source=relative) self.observe(run, "backend", "responseSendAwaitMs", finite_field(data, "responseSendAwaitSeconds", context) * 1000, parent="httpTotalMs", denominator=http, inclusion=DEFINITIONS["backend"], source=relative) for name in ("rawSeriesBytes", "responseBodyBytes"): self.observe(run, "size", name, finite_field(data, name, context, positive=True), "bytes", source=relative) phases = data.get("existingPerformance", {}).get("phases", {}) for name, phase in phases.items(): self.observe(run, "backend_existing", name, finite_field(phase, "inclusiveNs", context) / 1e6, parent="httpTotalMs", denominator=http, inclusion="Inclusive duration without aligned start/end; not an exclusive extra cost", source=relative) return {"source": relative, "xmlSha256": data.get("xmlSha256"), "httpTotalMs": http, "spans": spans, "nativeStages": c, "process": data.get("process"), "build": data.get("build")} def group(self, name: str, mode: str, backend_root: str | None) -> dict: relative = f"browser-{name}/summary.json" data = self.read(relative) require(data.get("errors") == [], f"{name}: missing errors list or reported browser errors") rows = data.get("rows", []) require(len(rows) == 4, f"Incomplete {name}: expected 1 warmup + 3 measured rows, got {len(rows)}") require(sorted(r.get("run", -1) for r in rows) == [0, 1, 2, 3], f"{name}: unexpected/duplicate run numbers") runs = [] for row in sorted(rows, key=lambda r: r["run"]): context = f"{name}/{row['run']}" require(row.get("mode") == mode and row.get("deep") is False, f"{context}: wrong instrumentation mode") require(row.get("warmup") is (row["run"] == 0), f"{context}: warmup label mismatch") sid = row.get("simulationId") require(isinstance(sid, str) and bool(sid) and sid not in self.simulation_ids, f"{context}: missing/duplicate simulationId") self.simulation_ids.add(sid) native = row.get("native", {}) require(native.get("success") is True and native.get("status") == "completed", f"{context}: native simulation failed") require(row.get("restoredIdentical") is True, f"{context}: restore parity was not confirmed") require(type(native.get("cacheHit")) is bool, f"{context}: missing cacheHit") if row["run"]: require(native["cacheHit"], f"{context}: measured run includes a cold build") for key in NATIVE_IDENTITY: require(key in native and native[key] is not None, f"{context}: missing native {key}") for key in COUNTERS: value = finite_field(native, key, context) require(int(value) == value, f"{context}: noninteger counter {key}") run = {"group": name, "phase": "warmup" if row["warmup"] else "measured", "run": row["run"], "simulationId": sid, "native": native, "original": row, "sampleCount": finite_field(row, "sampleCount", context, positive=True), "variableCount": finite_field(row, "variableCount", context, positive=True)} for key in GOALS.values(): finite_field(row, key, context, positive=True) if mode == "profiled": for key in ("resultParseMs", "synchronousStreamDecodeMs", "clickToIndexedDbCommitMs", "streamBytes"): finite_field(row, key, context, positive=True) for key, value in row.items(): if key.endswith(("Ms", "Bytes")) and (value is None or numeric(value)): self.observe(run, "frontend", key, value, "bytes" if key.endswith("Bytes") else "ms", source=relative) for key in ("solveSeconds", "solveCpuSeconds", "processWallSeconds", "buildSeconds"): self.observe(run, "native", key.removesuffix("Seconds") + "Ms", finite_field(native, key, context) * 1000, source=relative, inclusion=DEFINITIONS["process"] if key == "processWallSeconds" else "Native-reported timing; CPU and wall are separate") run["backend"] = self.backend(run, f"{backend_root}/{sid}/stages.json") if backend_root else None runs.append(run) for key in ("inputSha256", "buildAssetSetSha256"): value = data.get(key) require(isinstance(value, str) and len(value) == 64 and all(c in "0123456789abcdef" for c in value), f"{name}: missing/invalid {key}") require(bool(data.get("servedAssets")), f"{name}: missing served frontend assets") return {"source": relative, "mode": mode, "inputSha256": data["inputSha256"], "buildAssetSetSha256": data["buildAssetSetSha256"], "servedAssets": data["servedAssets"], "browser": data.get("browser"), "node": data.get("node"), "scriptSha256": data.get("scriptSha256"), "sourceDefinitions": data.get("definitions"), "measuredRunCount": 3, "warmupRunCount": 1, "runs": runs} def comparisons(self, groups: dict, before: str, after: str, metrics: dict, *, diagnostic: bool) -> list[dict]: comparison = [] for target, (source_domain, metric) in metrics.items(): summaries = [] for name in (before, after): rows = [o for o in self.observations if o["group"] == name and o["phase"] == "measured" and o["domain"] == source_domain and o["metric"] == metric] require(len(rows) == 3 and all(numeric(r["value"]) and r["value"] > 0 for r in rows), f"Missing comparison metric {name}/{target}") summaries.append(stats([r["value"] for r in sorted(rows, key=lambda r: r["run"])])) baseline, optimized = summaries old, new = baseline["median"], optimized["median"] comparison.append({"target": target, "metric": f"{source_domain}.{metric}", "before": before, "after": after, "diagnosticOnly": diagnostic, "statistic": "ratio_of_group_medians", "definition": DEFINITIONS["comparison"], "source": f"{groups[before]['source']} | {groups[after]['source']}", "baselineMs": baseline, "optimizedMs": optimized, "savedMs": old - new, "durationReductionPercent": (old - new) / old * 100, "speedupRatio": old / new}) return comparison def replay(self) -> dict: relative = "replay/summary.json" data = self.read(relative) require(data.get("allRealFileBinary64Parity") is True, "Replay real-file binary64 verification not complete") supplied = data.get("medians", {}) require(bool(supplied.get("file")), "Replay real-file medians missing") runs = data.get("runs", []) require(bool(runs), "Replay run metadata missing") for row in runs: require(row.get("success") is True, "Replay includes a failed run") run = {"group": f"replay:{row['sink']}:{row['variant']}", "phase": "warmup" if row["warmup"] else "measured", "run": row["run"], "simulationId": ""} for metric, unit in (("wallSeconds", "ms"), ("cpuSeconds", "ms"), ("encodedBytes", "bytes")): value = finite_field(row, metric, run["group"], positive=True) self.observe(run, "replay", metric.removesuffix("Seconds") + "Ms" if unit == "ms" else metric, value * 1000 if unit == "ms" else value, unit, source=relative, inclusion=DEFINITIONS["replay"]) medians = {sink: {variant: {k: value[k] for k in ("wallSeconds", "cpuSeconds", "encodedBytes")} for variant, value in variants.items()} for sink, variants in supplied.items()} return {"source": relative, "suppliedMedians": medians, "runs": runs, "timingContract": data.get("prepared", {}).get("timingContract"), "limitation": data.get("limitation"), "usedForEndToEndComparison": False} def build(self) -> dict: groups = {name: self.group(name, *config) for name, config in GROUPS.items()} reference = groups["baseline"] first = reference["runs"][0] for name, group in groups.items(): for key in ("inputSha256", "buildAssetSetSha256", "browser", "node"): require(group[key] is not None and group[key] == reference[key], f"Cross-group {key} mismatch: {name}") for run in group["runs"]: for key in ("sampleCount", "variableCount"): require(run[key] == first[key], f"Cross-run {key} mismatch: {name}/{run['run']}") for key in NATIVE_IDENTITY: require(run["native"][key] == first["native"][key], f"Cross-run native {key} mismatch: {name}/{run['run']}") xml_hashes = [run["backend"]["xmlSha256"] for group in groups.values() for run in group["runs"] if run["backend"]] require(all(isinstance(value, str) and len(value) == 64 for value in xml_hashes) and len(set(xml_hashes)) == 1, "Profiled input XML SHA missing or mismatched") goals = self.comparisons(groups, "baseline", "optimized", {key: ("frontend", metric) for key, metric in GOALS.items()}, diagnostic=False) stages = self.comparisons(groups, "baseline-profiled", "optimized-profiled", {"write_wall": ("c_wall", "jsonWriteMs"), "write_cpu": ("c_cpu", "jsonWriteCpuMs")}, diagnostic=True) replay = self.replay() buckets: dict[tuple, list[dict]] = {} keys = ("group", "phase", "domain", "metric", "unit", "parent") for observation in self.observations: buckets.setdefault(tuple(observation[key] for key in keys), []).append(observation) aggregates = [] for key, rows in sorted(buckets.items()): require(len({r["run"] for r in rows}) == len(rows), f"Duplicate per-run observation: {key}") aggregates.append(dict(zip(keys, key)) | stats([r["value"] for r in rows]) | { "percentOfParent": stats([r["percentOfParent"] for r in rows]), "runs": [{k: row[k] for k in ("run", "simulationId")} for row in rows]}) return {"schemaVersion": 1, "complete": True, "errors": [], "experimentRoot": str(self.root), "scriptSha256": hashlib.sha256(Path(__file__).read_bytes()).hexdigest(), "definitions": DEFINITIONS, "sourceFiles": self.sources, "groups": groups, "changes": {"endToEnd": goals, "cWriteDiagnostics": stages}, "replay": replay, "observations": self.observations, "aggregates": aggregates, "validation": {"groupCount": 4, "browserRunCount": 16, "warmupCount": 4, "measuredCount": 12, "inputSha256": reference["inputSha256"], "frontendAssetSetSha256": reference["buildAssetSetSha256"], "profiledXmlSha256": xml_hashes[0], "sampleCount": first["sampleCount"], "variableCount": first["variableCount"], "nativeIdentity": {key: first["native"][key] for key in NATIVE_IDENTITY}, "fullNumericalParityIndependentlyChecked": False}} def write_outputs(root: Path, summary: dict) -> None: root.mkdir(parents=True, exist_ok=True) with (root / "timings.csv").open("w", encoding="utf-8", newline="") as stream: writer = csv.DictWriter(stream, fieldnames=CSV_FIELDS) writer.writeheader() for row in summary.get("observations", []): writer.writerow(row | {"statistic": "run", "count": int(numeric(row["value"])), "missingCount": int(row["value"] is None)}) for entry in summary.get("aggregates", []): for statistic in ("median", "min", "max"): writer.writerow({k: entry[k] for k in ("group", "phase", "domain", "metric", "unit", "parent", "count", "missingCount")} | { "statistic": statistic, "value": entry[statistic], "percentOfParent": entry["percentOfParent"][statistic], "inclusion": "Per-run values and percentages aggregated separately; medians are not additive"}) for category, changes in summary.get("changes", {}).items(): for change in changes: for key, unit in (("durationReductionPercent", "percent"), ("speedupRatio", "ratio"), ("savedMs", "ms")): writer.writerow({"group": f"{change['after']}-vs-{change['before']}", "phase": "measured", "domain": f"comparison:{category}", "metric": f"{change['target']}.{key}", "statistic": "ratio_of_group_medians" if key != "savedMs" else "difference_of_group_medians", "value": change[key], "unit": unit, "baselineValue": change["baselineMs"]["median"], "optimizedValue": change["optimizedMs"]["median"], "count": 3, "missingCount": 0, "inclusion": ("Profiled C-write diagnostic only; " if change["diagnosticOnly"] else "") + change["definition"], "source": change["source"]}) (root / "summary.json").write_text(json.dumps(summary, ensure_ascii=False, indent=2, allow_nan=False) + "\n", encoding="utf-8") def main() -> int: parser = argparse.ArgumentParser(description=__doc__) parser.add_argument("--root", type=Path, default=REPO / "test/c-result-encoding-20260911") args = parser.parse_args() root = args.root.resolve() if not root.is_relative_to(REPO / "test"): parser.error("--root must be beneath the repository's ignored test/ directory") summarizer = Summarizer(root) try: summary = summarizer.build() except (OSError, ValueError, KeyError, TypeError) as error: summary = {"schemaVersion": 1, "complete": False, "errors": [str(error)], "experimentRoot": str(root), "sourceFiles": summarizer.sources, "note": "No partial timing statistics are published. Complete/fix all groups and rerun."} write_outputs(root, summary) print(json.dumps(summary, ensure_ascii=False, indent=2), file=sys.stderr) return 2 write_outputs(root, summary) print(json.dumps({"complete": True, "validation": summary["validation"], "observations": len(summary["observations"]), "outputs": [str(root / name) for name in ("summary.json", "timings.csv")]}, ensure_ascii=False, indent=2)) return 0 if __name__ == "__main__": raise SystemExit(main())