优化原生结果编码传输与浏览器缓存,记录八路性能基线
原生结果series通过字节索引直传,C端使用Ryu精确回读编码和64 KiB批量写出;网页采用Float64缓存和CSV工作线程,减少结果处理与保存等待。 补充八路AME曲线核查、全流程分阶段计时、独立编码基准和复现工具,固定后续优化采用修正八路及rtol=1e-8。C写出1.1808→0.1638 s,点击到可查看8.0100→6.9756 s。 验证:最终10项编码专项、29项相关后端回归通过;8份原生结果逐位一致,16次网页结果/CSV/刷新恢复通过。前端构建及缓存/CSV专项在本轮结果处理工作中通过。环境、原始大结果与临时构建不纳入Git。
No files matched your search
@@ -23,6 +23,7 @@ from fastapi.responses import FileResponse, HTMLResponse, StreamingResponse
|
|||||||
from pydantic import BaseModel, ConfigDict, Field, ValidationError
|
from pydantic import BaseModel, ConfigDict, Field, ValidationError
|
||||||
|
|
||||||
from app.simulation.performance import performance_span, profile_phase, profile_run
|
from app.simulation.performance import performance_span, profile_phase, profile_run
|
||||||
|
from app.simulation.native_codegen.transport import NativeSeriesJson, serialize_result_parts
|
||||||
from app.simulation.config import SolverActivityTracker
|
from app.simulation.config import SolverActivityTracker
|
||||||
from app.system_xml import (
|
from app.system_xml import (
|
||||||
SystemXmlDocument,
|
SystemXmlDocument,
|
||||||
@@ -652,7 +653,7 @@ async def simulate_system_xml_stream(request: Request) -> StreamingResponse:
|
|||||||
simulation_id = request.headers.get("x-simulation-id") or uuid4().hex
|
simulation_id = request.headers.get("x-simulation-id") or uuid4().hex
|
||||||
task = _register_simulation_task(simulation_id)
|
task = _register_simulation_task(simulation_id)
|
||||||
return StreamingResponse(
|
return StreamingResponse(
|
||||||
simulation_event_stream(await request.body(), task=task),
|
simulation_event_stream(await request.body(), task=task, raw_series=True),
|
||||||
media_type="application/x-ndjson",
|
media_type="application/x-ndjson",
|
||||||
headers={
|
headers={
|
||||||
"Cache-Control": "no-cache, no-transform",
|
"Cache-Control": "no-cache, no-transform",
|
||||||
@@ -679,13 +680,17 @@ def cancel_system_xml_simulation(
|
|||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
@app.get("/api/system-xml/simulations/{simulation_id}")
|
@app.get("/api/system-xml/simulations/{simulation_id}", response_model=None)
|
||||||
def get_system_xml_simulation(simulation_id: str) -> dict[str, object]:
|
def get_system_xml_simulation(simulation_id: str) -> dict[str, object] | Response:
|
||||||
with SIMULATION_TASKS_LOCK:
|
with SIMULATION_TASKS_LOCK:
|
||||||
task = SIMULATION_TASKS.get(simulation_id)
|
task = SIMULATION_TASKS.get(simulation_id)
|
||||||
if task is None:
|
if task is None:
|
||||||
raise HTTPException(status_code=404, detail="Simulation task was not found.")
|
raise HTTPException(status_code=404, detail="Simulation task was not found.")
|
||||||
return _simulation_task_snapshot(task)
|
snapshot = _simulation_task_snapshot(task)
|
||||||
|
result = snapshot.get("result")
|
||||||
|
if isinstance(result, dict) and isinstance(result.get("series"), NativeSeriesJson):
|
||||||
|
return StreamingResponse(iter(serialize_result_parts(snapshot)), media_type="application/json")
|
||||||
|
return snapshot
|
||||||
|
|
||||||
|
|
||||||
def run_system_xml_simulation(
|
def run_system_xml_simulation(
|
||||||
@@ -693,6 +698,7 @@ def run_system_xml_simulation(
|
|||||||
progress_callback: SimulationProgressEmitter | None = None,
|
progress_callback: SimulationProgressEmitter | None = None,
|
||||||
cancel_check: Callable[[], bool] | None = None,
|
cancel_check: Callable[[], bool] | None = None,
|
||||||
activity_tracker: SolverActivityTracker | None = None,
|
activity_tracker: SolverActivityTracker | None = None,
|
||||||
|
*, raw_series: bool = False,
|
||||||
) -> dict[str, object]:
|
) -> dict[str, object]:
|
||||||
with profile_run() as trace:
|
with profile_run() as trace:
|
||||||
result = _run_system_xml_simulation_profiled(
|
result = _run_system_xml_simulation_profiled(
|
||||||
@@ -700,6 +706,7 @@ def run_system_xml_simulation(
|
|||||||
progress_callback,
|
progress_callback,
|
||||||
cancel_check,
|
cancel_check,
|
||||||
activity_tracker,
|
activity_tracker,
|
||||||
|
raw_series=raw_series,
|
||||||
)
|
)
|
||||||
|
|
||||||
performance = trace.snapshot()
|
performance = trace.snapshot()
|
||||||
@@ -715,6 +722,7 @@ def _run_system_xml_simulation_profiled(
|
|||||||
progress_callback: SimulationProgressEmitter | None = None,
|
progress_callback: SimulationProgressEmitter | None = None,
|
||||||
cancel_check: Callable[[], bool] | None = None,
|
cancel_check: Callable[[], bool] | None = None,
|
||||||
activity_tracker: SolverActivityTracker | None = None,
|
activity_tracker: SolverActivityTracker | None = None,
|
||||||
|
*, raw_series: bool = False,
|
||||||
) -> dict[str, object]:
|
) -> dict[str, object]:
|
||||||
from app.simulation.backends import simulate_network
|
from app.simulation.backends import simulate_network
|
||||||
from app.simulation.results import SimulationPreparationError
|
from app.simulation.results import SimulationPreparationError
|
||||||
@@ -762,6 +770,7 @@ def _run_system_xml_simulation_profiled(
|
|||||||
progress_callback=report_system_progress,
|
progress_callback=report_system_progress,
|
||||||
cancel_check=cancel_check,
|
cancel_check=cancel_check,
|
||||||
activity_tracker=activity_tracker,
|
activity_tracker=activity_tracker,
|
||||||
|
raw_series=raw_series,
|
||||||
)
|
)
|
||||||
except SimulationPreparationError as exc:
|
except SimulationPreparationError as exc:
|
||||||
raise HTTPException(
|
raise HTTPException(
|
||||||
@@ -799,7 +808,7 @@ def _run_system_xml_simulation_profiled(
|
|||||||
"validation": report.as_dict(),
|
"validation": report.as_dict(),
|
||||||
"simulation": document.as_model_data()["simulation"],
|
"simulation": document.as_model_data()["simulation"],
|
||||||
"model": network.as_interface_dict(),
|
"model": network.as_interface_dict(),
|
||||||
**result.as_dict(),
|
**result.as_dict(raw_series=raw_series),
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
@@ -807,7 +816,8 @@ def simulation_event_stream(
|
|||||||
xml_bytes: bytes,
|
xml_bytes: bytes,
|
||||||
*,
|
*,
|
||||||
task: SimulationTaskRecord | None = None,
|
task: SimulationTaskRecord | None = None,
|
||||||
) -> Iterator[str]:
|
raw_series: bool = False,
|
||||||
|
) -> Iterator[str | bytes]:
|
||||||
events: queue.Queue[dict[str, object] | object] = queue.Queue()
|
events: queue.Queue[dict[str, object] | object] = queue.Queue()
|
||||||
finished = object()
|
finished = object()
|
||||||
latest_progress = 0
|
latest_progress = 0
|
||||||
@@ -857,6 +867,7 @@ def simulation_event_stream(
|
|||||||
emit_progress,
|
emit_progress,
|
||||||
task.cancel_event.is_set if task is not None else None,
|
task.cancel_event.is_set if task is not None else None,
|
||||||
activity_tracker,
|
activity_tracker,
|
||||||
|
**({"raw_series": True} if raw_series else {}),
|
||||||
)
|
)
|
||||||
if task is not None:
|
if task is not None:
|
||||||
result = _mark_simulation_task_result(task, result)
|
result = _mark_simulation_task_result(task, result)
|
||||||
@@ -958,6 +969,10 @@ def simulation_event_stream(
|
|||||||
continue
|
continue
|
||||||
if event is finished:
|
if event is finished:
|
||||||
break
|
break
|
||||||
|
if raw_series and isinstance(event, dict) and event.get("event") == "result":
|
||||||
|
yield from serialize_result_parts(event)
|
||||||
|
yield b"\n"
|
||||||
|
else:
|
||||||
yield json.dumps(event, ensure_ascii=False, separators=(",", ":")) + "\n"
|
yield json.dumps(event, ensure_ascii=False, separators=(",", ":")) + "\n"
|
||||||
finally:
|
finally:
|
||||||
if task is not None:
|
if task is not None:
|
||||||
|
|||||||
@@ -30,9 +30,9 @@ def simulation_config(simulation) -> SolveIVPConfig:
|
|||||||
|
|
||||||
|
|
||||||
def simulate_network(network, simulation, *, progress_callback=None,
|
def simulate_network(network, simulation, *, progress_callback=None,
|
||||||
cancel_check=None, activity_tracker=None, backend=None):
|
cancel_check=None, activity_tracker=None, backend=None, raw_series=False):
|
||||||
numeric_engine_name(backend)
|
numeric_engine_name(backend)
|
||||||
config = simulation_config(simulation)
|
config = simulation_config(simulation)
|
||||||
from app.simulation.native_codegen.runner import simulate_native
|
from app.simulation.native_codegen.runner import simulate_native
|
||||||
return simulate_native(network, config, sample_step=simulation.sample_step, progress_callback=progress_callback,
|
return simulate_native(network, config, sample_step=simulation.sample_step, progress_callback=progress_callback,
|
||||||
cancel_check=cancel_check, activity_tracker=activity_tracker)
|
cancel_check=cancel_check, activity_tracker=activity_tracker, raw_series=raw_series)
|
||||||
@@ -52,7 +52,7 @@ def toolchain() -> tuple[str, Path, str]:
|
|||||||
def build_native(program: NativeProgram, *, cache_dir: Path | None = None) -> NativeBuild:
|
def build_native(program: NativeProgram, *, cache_dir: Path | None = None) -> NativeBuild:
|
||||||
start = time.perf_counter()
|
start = time.perf_counter()
|
||||||
compiler, sundials, compiler_version = toolchain()
|
compiler, sundials, compiler_version = toolchain()
|
||||||
runtime = sorted(NATIVE.rglob("*.c")) + sorted((NATIVE / "include").glob("*.h"))
|
runtime = sorted(NATIVE.rglob("*.c")) + sorted((NATIVE / "include").rglob("*.h"))
|
||||||
flags = ["-std=c11", "-O3", "-Wall", "-Wextra", "-Werror", "-ffp-contract=off", "-fno-fast-math"]
|
flags = ["-std=c11", "-O3", "-Wall", "-Wextra", "-Werror", "-ffp-contract=off", "-fno-fast-math"]
|
||||||
executable_name = "model.exe" if os.name == "nt" else "model"
|
executable_name = "model.exe" if os.name == "nt" else "model"
|
||||||
if os.name == "nt":
|
if os.name == "nt":
|
||||||
|
|||||||
@@ -12,13 +12,14 @@ import time
|
|||||||
|
|
||||||
from app.simulation.config import SolveIVPConfig
|
from app.simulation.config import SolveIVPConfig
|
||||||
from app.simulation.results import GenericSimulationResult
|
from app.simulation.results import GenericSimulationResult
|
||||||
|
from .transport import NativeSeriesJson, read_indexed_result
|
||||||
from .build import NativeBuild, build_native
|
from .build import NativeBuild, build_native
|
||||||
from .compiler import NativeCapabilityError, compile_native_program
|
from .compiler import NativeCapabilityError, compile_native_program
|
||||||
|
|
||||||
|
|
||||||
def execute_native(build: NativeBuild, config: SolveIVPConfig, sample_step: float, *,
|
def execute_native(build: NativeBuild, config: SolveIVPConfig, sample_step: float, *,
|
||||||
run_dir: Path, record_samples=True, cancel_check=None,
|
run_dir: Path, record_samples=True, cancel_check=None,
|
||||||
progress_callback=None, activity_tracker=None, timeout=300.0) -> dict:
|
progress_callback=None, activity_tracker=None, timeout=300.0, raw_series=False) -> dict:
|
||||||
if config.method not in ("RK45", "BDF"):
|
if config.method not in ("RK45", "BDF"):
|
||||||
raise NativeCapabilityError(f"Native v1 does not support method {config.method}.")
|
raise NativeCapabilityError(f"Native v1 does not support method {config.method}.")
|
||||||
if not isinstance(config.atol, (int, float)) or config.atol != 1e-8 or config.first_step is not None:
|
if not isinstance(config.atol, (int, float)) or config.atol != 1e-8 or config.first_step is not None:
|
||||||
@@ -26,13 +27,16 @@ def execute_native(build: NativeBuild, config: SolveIVPConfig, sample_step: floa
|
|||||||
run_dir.mkdir(parents=True, exist_ok=True)
|
run_dir.mkdir(parents=True, exist_ok=True)
|
||||||
output = run_dir / "result.json"
|
output = run_dir / "result.json"
|
||||||
cancel_path = run_dir / "cancel.request"
|
cancel_path = run_dir / "cancel.request"
|
||||||
if output.exists() or cancel_path.exists():
|
index_path = run_dir / "result-index.json"
|
||||||
|
if output.exists() or cancel_path.exists() or index_path.exists():
|
||||||
raise ValueError("Native execution requires a fresh run directory.")
|
raise ValueError("Native execution requires a fresh run directory.")
|
||||||
command = [str(build.executable), "--method", config.method,
|
command = [str(build.executable), "--method", config.method,
|
||||||
"--start", str(config.t_start), "--stop", str(config.t_stop),
|
"--start", str(config.t_start), "--stop", str(config.t_stop),
|
||||||
"--sample-step", str(sample_step), "--max-step", str(config.max_step),
|
"--sample-step", str(sample_step), "--max-step", str(config.max_step),
|
||||||
"--rtol", str(config.rtol), "--timeout", str(timeout),
|
"--rtol", str(config.rtol), "--timeout", str(timeout),
|
||||||
"--cancel-file", str(cancel_path.resolve()), "--output", str(output.resolve())]
|
"--cancel-file", str(cancel_path.resolve()), "--output", str(output.resolve())]
|
||||||
|
if raw_series:
|
||||||
|
command.extend(["--result-index", str(index_path.resolve())])
|
||||||
if not record_samples:
|
if not record_samples:
|
||||||
command.append("--solve-only")
|
command.append("--solve-only")
|
||||||
creationflags = subprocess.CREATE_NO_WINDOW if os.name == "nt" else 0
|
creationflags = subprocess.CREATE_NO_WINDOW if os.name == "nt" else 0
|
||||||
@@ -86,9 +90,13 @@ def execute_native(build: NativeBuild, config: SolveIVPConfig, sample_step: floa
|
|||||||
process.stderr.close()
|
process.stderr.close()
|
||||||
if not output.is_file():
|
if not output.is_file():
|
||||||
raise RuntimeError(f"Native worker exited with code {process.returncode} without results; see {run_dir / 'worker.log'}.")
|
raise RuntimeError(f"Native worker exited with code {process.returncode} without results; see {run_dir / 'worker.log'}.")
|
||||||
payload = json.loads(output.read_text(encoding="utf-8"))
|
|
||||||
if process.returncode not in (0, 2):
|
if process.returncode not in (0, 2):
|
||||||
raise RuntimeError(f"Native worker failed with exit code {process.returncode}.")
|
raise RuntimeError(f"Native worker failed with exit code {process.returncode}.")
|
||||||
|
try:
|
||||||
|
payload = (read_indexed_result(output, index_path) if raw_series
|
||||||
|
else json.loads(output.read_text(encoding="utf-8")))
|
||||||
|
except OSError as exc:
|
||||||
|
raise RuntimeError(f"Cannot read native worker result artifacts: {exc}") from exc
|
||||||
payload["processWallSeconds"] = time.perf_counter()-started
|
payload["processWallSeconds"] = time.perf_counter()-started
|
||||||
payload["buildKey"] = build.manifest["buildKey"]
|
payload["buildKey"] = build.manifest["buildKey"]
|
||||||
payload["cacheHit"] = build.cache_hit
|
payload["cacheHit"] = build.cache_hit
|
||||||
@@ -99,7 +107,7 @@ def execute_native(build: NativeBuild, config: SolveIVPConfig, sample_step: floa
|
|||||||
|
|
||||||
|
|
||||||
def simulate_native(network, config, *, sample_step, progress_callback=None,
|
def simulate_native(network, config, *, sample_step, progress_callback=None,
|
||||||
cancel_check=None, activity_tracker=None):
|
cancel_check=None, activity_tracker=None, raw_series=False):
|
||||||
if config.method not in ("RK45", "BDF"):
|
if config.method not in ("RK45", "BDF"):
|
||||||
raise NativeCapabilityError(f"Native v1 does not support method {config.method}.")
|
raise NativeCapabilityError(f"Native v1 does not support method {config.method}.")
|
||||||
if progress_callback:
|
if progress_callback:
|
||||||
@@ -109,7 +117,7 @@ def simulate_native(network, config, *, sample_step, progress_callback=None,
|
|||||||
with tempfile.TemporaryDirectory(prefix="native-simulation-") as directory:
|
with tempfile.TemporaryDirectory(prefix="native-simulation-") as directory:
|
||||||
data = execute_native(build, config, sample_step, run_dir=Path(directory),
|
data = execute_native(build, config, sample_step, run_dir=Path(directory),
|
||||||
cancel_check=cancel_check, progress_callback=progress_callback,
|
cancel_check=cancel_check, progress_callback=progress_callback,
|
||||||
activity_tracker=activity_tracker)
|
activity_tracker=activity_tracker, raw_series=raw_series)
|
||||||
totals = {
|
totals = {
|
||||||
"nfev": data["nfev"], "njev": data["njev"], "nlu": data["nlu"],
|
"nfev": data["nfev"], "njev": data["njev"], "nlu": data["nlu"],
|
||||||
"acceptedStepCount": data["acceptedSteps"], "rejectedStepCount": data["rejectedSteps"],
|
"acceptedStepCount": data["acceptedSteps"], "rejectedStepCount": data["rejectedSteps"],
|
||||||
@@ -124,5 +132,6 @@ def simulate_native(network, config, *, sample_step, progress_callback=None,
|
|||||||
variables=program.variables, series=data["series"], final=data["final"],
|
variables=program.variables, series=data["series"], final=data["final"],
|
||||||
diagnostics={"backend": "native-c", "native": {k: v for k, v in data.items()
|
diagnostics={"backend": "native-c", "native": {k: v for k, v in data.items()
|
||||||
if k not in ("series", "final", "finalState")}, "integration": {"method": config.method, "rtol": config.rtol, "totals": totals},
|
if k not in ("series", "final", "finalState")}, "integration": {"method": config.method, "rtol": config.rtol, "totals": totals},
|
||||||
"stateCount": len(program.state_keys), "sampleCount": len(data["series"]["time"])},
|
"stateCount": len(program.state_keys), "sampleCount": (data["series"].sample_count if isinstance(data["series"], NativeSeriesJson)
|
||||||
|
else len(data["series"].get("time", [])))},
|
||||||
)
|
)
|
||||||
@@ -0,0 +1,63 @@
|
|||||||
|
"""Carry trusted C-generated series JSON without a Python float-array round trip.
|
||||||
|
|
||||||
|
Only native output plus its byte index may construct this transport object. Public
|
||||||
|
JSON still has the ordinary object/array schema; synchronous callers materialize
|
||||||
|
it explicitly. Bytes own their lifetime independently of the worker directory.
|
||||||
|
"""
|
||||||
|
from dataclasses import dataclass
|
||||||
|
import json
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class NativeSeriesJson:
|
||||||
|
data: bytes
|
||||||
|
sample_count: int
|
||||||
|
|
||||||
|
def materialize(self) -> dict[str, list[float]]:
|
||||||
|
return json.loads(self.data)
|
||||||
|
|
||||||
|
|
||||||
|
def read_indexed_result(output: Path, index_path: Path) -> dict:
|
||||||
|
index = json.loads(index_path.read_bytes())
|
||||||
|
if not isinstance(index, dict) or type(index.get('version')) is not int or index['version'] != 1:
|
||||||
|
raise ValueError('Unsupported native result index.')
|
||||||
|
names = ('seriesStart', 'seriesEnd', 'resultBytes', 'sampleCount')
|
||||||
|
if any(type(index.get(key)) is not int for key in names):
|
||||||
|
raise ValueError('Invalid native result index integers.')
|
||||||
|
start, end, length, count = (index[key] for key in names)
|
||||||
|
if not (0 < start < end < length and count >= 0):
|
||||||
|
raise ValueError('Invalid native result index bounds.')
|
||||||
|
if output.stat().st_size != length:
|
||||||
|
raise ValueError('Incomplete native result file.')
|
||||||
|
raw = output.read_bytes()
|
||||||
|
if (len(raw) != length or not raw[:start].endswith(b'"series":')
|
||||||
|
or raw[start:start+1] != b'{' or raw[end-1:end] != b'}'
|
||||||
|
or not raw[end:].startswith(b',"final":')):
|
||||||
|
raise ValueError('Native result index does not match the output layout.')
|
||||||
|
# The C writer supplies exact boundaries. No search through strings or numeric
|
||||||
|
# arrays, and no large json.loads call: only diagnostics/final values are read.
|
||||||
|
metadata = json.loads(raw[:start] + b'{}' + raw[end:])
|
||||||
|
if not isinstance(metadata, dict) or metadata.get('series') != {}:
|
||||||
|
raise ValueError('Invalid native result metadata.')
|
||||||
|
metadata['series'] = NativeSeriesJson(raw[start:end], count)
|
||||||
|
return metadata
|
||||||
|
|
||||||
|
|
||||||
|
def serialize_result_parts(payload: dict) -> tuple[bytes, ...]:
|
||||||
|
"""Return one JSON object in three byte segments, without copying its series.
|
||||||
|
|
||||||
|
The only raw position is payload.result.series, produced by our C writer;
|
||||||
|
all model-provided labels, messages and keys use the standard JSON encoder.
|
||||||
|
"""
|
||||||
|
result = payload.get('result')
|
||||||
|
series = result.get('series') if isinstance(result, dict) else None
|
||||||
|
if not isinstance(series, NativeSeriesJson):
|
||||||
|
return (json.dumps(payload, ensure_ascii=False, separators=(',', ':')).encode('utf-8'),)
|
||||||
|
metadata = {key: value for key, value in result.items() if key != 'series'}
|
||||||
|
outer = {key: value for key, value in payload.items() if key != 'result'}
|
||||||
|
encoded = json.dumps(metadata, ensure_ascii=False, separators=(',', ':')).encode('utf-8')
|
||||||
|
remainder = json.dumps(outer, ensure_ascii=False, separators=(',', ':')).encode('utf-8')
|
||||||
|
prefix = b'{"result":' + encoded[:-1] + (b',' if metadata else b'') + b'"series":'
|
||||||
|
suffix = b'}' + (b',' + remainder[1:] if outer else b'}')
|
||||||
|
return prefix, series.data, suffix
|
||||||
@@ -2,7 +2,10 @@
|
|||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
from dataclasses import dataclass
|
from dataclasses import dataclass
|
||||||
from typing import Literal
|
from typing import TYPE_CHECKING, Literal
|
||||||
|
|
||||||
|
if TYPE_CHECKING:
|
||||||
|
from app.simulation.native_codegen.transport import NativeSeriesJson
|
||||||
from app.simulation.core.metadata import ResultVariableMetadata
|
from app.simulation.core.metadata import ResultVariableMetadata
|
||||||
|
|
||||||
SimulationRunStatus = Literal["completed", "cancelled", "failed"]
|
SimulationRunStatus = Literal["completed", "cancelled", "failed"]
|
||||||
@@ -30,11 +33,15 @@ class GenericSimulationResult:
|
|||||||
simulated_until: float
|
simulated_until: float
|
||||||
requested_stop_time: float
|
requested_stop_time: float
|
||||||
variables: tuple[ResultVariableMetadata, ...]
|
variables: tuple[ResultVariableMetadata, ...]
|
||||||
series: dict[str, list[float]]
|
series: dict[str, list[float]] | NativeSeriesJson
|
||||||
final: dict[str, float]
|
final: dict[str, float]
|
||||||
diagnostics: dict[str, object]
|
diagnostics: dict[str, object]
|
||||||
|
|
||||||
def as_dict(self) -> dict[str, object]:
|
def as_dict(self, *, raw_series: bool = False) -> dict[str, object]:
|
||||||
|
from app.simulation.native_codegen.transport import NativeSeriesJson
|
||||||
|
series = self.series
|
||||||
|
if isinstance(series, NativeSeriesJson) and not raw_series:
|
||||||
|
series = series.materialize()
|
||||||
return {
|
return {
|
||||||
"success": self.success,
|
"success": self.success,
|
||||||
"status": self.status,
|
"status": self.status,
|
||||||
@@ -43,7 +50,7 @@ class GenericSimulationResult:
|
|||||||
"simulatedUntil": self.simulated_until,
|
"simulatedUntil": self.simulated_until,
|
||||||
"requestedStopTime": self.requested_stop_time,
|
"requestedStopTime": self.requested_stop_time,
|
||||||
"variables": [variable.as_dict() for variable in self.variables],
|
"variables": [variable.as_dict() for variable in self.variables],
|
||||||
"series": self.series,
|
"series": series,
|
||||||
"final": self.final,
|
"final": self.final,
|
||||||
"diagnostics": self.diagnostics,
|
"diagnostics": self.diagnostics,
|
||||||
}
|
}
|
||||||
@@ -0,0 +1,156 @@
|
|||||||
|
# C 端结果编码与写出优化(2026-09-11)
|
||||||
|
|
||||||
|
针对上一轮 [八路全流程成本评估](八路网页求解全流程成本评估-2026-09-11.md) 中约1.18 s的C结果写出阶段,先比较编码和缓冲方案,再实现并验证真实网页路径。案例固定为 [test-mql-8-corrected.json](../../tests/data/test-mql-8-corrected.json),未修改方程、局部管流求根、积分器、误差限、采样或前端生产代码。
|
||||||
|
|
||||||
|
已采用 **Ryu精确回读编码 + 64 KiB批量写出**。真实八路模型中,C编码写出墙钟中位数 **1.1808 → 0.1638 s,减少86.13%**;网页点击到结果可查看 **8.0100 → 6.9756 s,减少12.91%**,点击到浏览器缓存保存完成 **8.1280 → 7.0841 s,减少12.84%**。CSV下载保存没有观察到改善(1.1127 → 1.1506 s)。数值精度和求解路径保留,完整原生结果逐位一致,16次网页结果/CSV/刷新恢复核验通过。
|
||||||
|
|
||||||
|
上述网页收益是构建缓存命中的三次正式运行组中位数比较。首次编译另列:新依赖会增加冷编译成本,不能把缓存命中收益直接套用到首次运行。
|
||||||
|
|
||||||
|
## 调研与方案选择
|
||||||
|
|
||||||
|
原路径对约179万个结果数字逐值调用 `fprintf("%s%.17g", …)`。CPU时间与墙钟接近,只说明这一段主要在执行代码,仍需实验区分转换和写入成本。保留JSON合同可以继续使用已有Python原始片段传输、浏览器解析、缓存、CSV和结果文件流程。
|
||||||
|
|
||||||
|
候选比较基于作者源码与官方文档,未采用第三方性能宣传作为本项目的加速证据:
|
||||||
|
|
||||||
|
| 候选 | 本轮判断 | 一手依据 |
|
||||||
|
|---|---|---|
|
||||||
|
| 加大stdio缓存,保留 `fprintf %.17g` | 改动小;必须实测是否能减少主要成本 | 当前 `native/runtime/main.c` 与下方重放实验 |
|
||||||
|
| `snprintf %.17g` 到固定块,再 `fwrite` | 仍使用相同浮点转换;可隔离stdio调用方式的收益 | 下方重放实验 |
|
||||||
|
| Ryu binary64 shortest | 采用;C接口、小型固定依赖、无分配转换,能精确回读原浮点值 | [固定版本源码](https://github.com/ulfjack/ryu/blob/4c0618b0e44f7ef027ebae05d2cc7812048f7c8f/ryu/d2s.c)、[作者说明](https://github.com/ulfjack/ryu/tree/4c0618b0e44f7ef027ebae05d2cc7812048f7c8f)、[边界测试](https://github.com/ulfjack/ryu/blob/4c0618b0e44f7ef027ebae05d2cc7812048f7c8f/ryu/tests/d2s_test.cc) |
|
||||||
|
| yyjson 的Schubfach路径 | 可用作后续对照,但完整 `.c/.h` 约756 kB,本轮不引入完整JSON库 | [0.13.0接口](https://github.com/ibireme/yyjson/blob/6447536015f3d600f3d65323b10976103b337ca7/src/yyjson.h#L1715-L1733) |
|
||||||
|
| 独立yy_double / Dragonbox | 前者属于作者基准仓库抽出版;后者官方实现要求C++11,本轮保留C11构建链 | [yy_double](https://github.com/ibireme/c_numconv_benchmark/tree/bdacf3330e202d7ec3ae552419ea5772bae95dac/vendor/yy_double)、[Dragonbox](https://github.com/jk-jeon/dragonbox/blob/beeeef91cf6fef89a4d4ba5e95d47ca64ccb3a44/README.md) |
|
||||||
|
|
||||||
|
Ryu 的最短转换指足以恢复原始binary64的有效数字,不代表完整JSON字符数一定最少。其裸接口也会生成 `NaN`/`Infinity`;这些不是标准JSON数字,因此包装层必须明确拒绝,而非直接输出。相关规则见 [Ryu源码](https://github.com/ulfjack/ryu/blob/4c0618b0e44f7ef027ebae05d2cc7812048f7c8f/ryu/d2s.c)、[RFC 8259 §6](https://www.rfc-editor.org/rfc/rfc8259.html#section-6)。
|
||||||
|
|
||||||
|
本轮固定Ryu提交 `4c0618b0e44f7ef027ebae05d2cc7812048f7c8f`,原样保留相关C/头文件,选择Boost-1.0许可。源码及来源清单位于 [native/encoding/ryu](../../native/encoding/ryu/),头文件位于 [native/include/ryu](../../native/include/ryu/),许可同时随原生构建的 `THIRD_PARTY_NOTICES.txt` 分发。它是项目源代码依赖,不是本地运行环境;未安装新环境或新增Python/前端依赖。
|
||||||
|
|
||||||
|
## 独立编码与写入实验
|
||||||
|
|
||||||
|
将真实八路原生结果的全部series、final和finalState,共 **1,790,486个binary64**,在计时外转为预加载的连续double数组。每个候选预热一次、正式三次,顺序交替,先写真实文件并逐值按64位模式核对,再单独做 `/dev/null` 输出对照。
|
||||||
|
|
||||||
|
计时从 `fopen` 前到 `fclose` 后,包含编码、缓冲设置和写出;不含输入加载、输出投影、JSON静态键名准备或数值复核,也未调用 `fsync`。连续数据重放不包含生产代码的跨行矩阵读取,不能直接将微基准加速当作网页提速。
|
||||||
|
|
||||||
|
| 编码/写入候选 | 真实文件墙钟 / s | 真实文件CPU / s | `/dev/null`墙钟 / s | 输出字节 |
|
||||||
|
|---|---:|---:|---:|---:|
|
||||||
|
| 原 `fprintf %.17g` | 1.121005 | 1.120823 | 1.108461 | 32,725,281 |
|
||||||
|
| `fprintf` + 1 MiB stdio缓存 | 1.121927 | 1.121777 | 1.107086 | 32,725,281 |
|
||||||
|
| `snprintf %.17g` + 64 KiB批量 | 1.148625 | 1.148510 | 1.087661 | 32,725,281 |
|
||||||
|
| 裸Ryu + 64 KiB批量 | 0.100643 | 0.100637 | 0.071845 | 34,387,794 |
|
||||||
|
|
||||||
|
加大stdio缓存没有改善;保留相同浮点转换的 `snprintf` 批量方案反而稍慢。裸Ryu重放墙钟减少91.02%,且 `/dev/null` 中同样大幅加速,证据支持主要成本在浮点转换,而非仅文件写入等待。CPU/墙钟之差不是独立磁盘耗时。
|
||||||
|
|
||||||
|
裸Ryu总是采用科学计数法,重放文件反而增大约5.08%。因此生产包装层进一步比较普通与科学表示长度;下方生产C写出包含该重排和真实stride读取,不能与裸Ryu的0.1006 s混作同一测量。
|
||||||
|
|
||||||
|
原始依据:[replay/summary.json](../../test/c-result-encoding-20260911/replay/summary.json),每次真实文件与计时记录均保留在 `replay/file/`。工具:[benchmark_native_result_encoding.py](../../tests/manual/benchmark_native_result_encoding.py)。
|
||||||
|
|
||||||
|
## 实现
|
||||||
|
|
||||||
|
- [json_numbers.c](../../native/runtime/json_numbers.c) 调用 `d2s_buffered_n`,用固定64 KiB栈缓冲批量写出数组,支持原输出矩阵的stride;每次调用返回前将自身缓冲交给FILE,保留调用者已有的stdio顺序和 `ftell` 边界。
|
||||||
|
- Ryu输出后只进行十进制token重排:普通表示更短时采用普通表示,否则保留科学计数法。例如 `1.2E1 → 12`、`1E-1 → 0.1`;等长时不改。重排没有浮点运算或再次舍入,也不会展开巨大指数。负零固定输出 `-0.0`,普通Python JSON读取也能保留符号。
|
||||||
|
- [main.c](../../native/runtime/main.c) 中 `series/final/finalState` 接入新编码。状态/整数计数/索引元数据、字符串转义、`--probe/--init` 的既有stdio路径保持原方式;不是所有C数字出口都改成Ryu。
|
||||||
|
- 非有限结果数字、短写、`ferror` 或 `fclose` 失败都会阻止新结果索引发布并返回失败。数值数组未写完整时,不能把已写出的文件前缀视作成功结果。活动runner仍要求每次使用新的输出目录。
|
||||||
|
- [build.py](../../app/simulation/native_codegen/build.py) 递归纳入嵌套头文件哈希,Ryu源码、查找表和许可记录都进入构建身份或随构建分发;不会误用优化前缓存。
|
||||||
|
|
||||||
|
JSON的字段、变量键名、列序、采样数和数值精度保留。数字拼写和文件SHA允许变化;不要求与旧 `%.17g` 的文本逐字相同。没有采用降低精度、减少采样、删列或有损压缩。
|
||||||
|
|
||||||
|
## 真实八路网页验证
|
||||||
|
|
||||||
|
输入SHA256为 `670977bef67e62d9c66e8af497bada208bd72a7301be45128d185d47282cf288`;157元件、178条连接、132状态、1784变量及时间列、1002个采样。0~10 s、0.01 s输出、CVODE BDF、`rtol=1e-8`、`max_step=1e30`及各状态atol下限保持原值。
|
||||||
|
|
||||||
|
旧版是本轮修改前从工作区冻结的代码(含此前结果传输/IDB/CSV优化),不是退回Git的旧后处理。旧版与新版各有无插桩组和阶段诊断组;每组预热一次、正式三次,串行运行。没有在正式计时期间安排其他大型构建、仿真或全量数值比对。前端全部使用相同生产资产,不改源码或构建。
|
||||||
|
|
||||||
|
用户端到端加速取同机无插桩两组;C写出细分取阶段诊断两组。页面“结果可查看”为成功完成DOM且按钮恢复可用,“保存完成”为IndexedDB事务提交后恢复指针发布的观察点。下载计时包括Playwright通知和 `saveAs`;本机回环HTTP不能代表远程网络。
|
||||||
|
|
||||||
|
三次正式运行的**中位数(最小~最大)**,单位s。端到端来自无插桩组,C阶段来自独立诊断组;各组依次采集,未声称同序号是交替配对试验。降幅统一为两组中位数之比,小样本、系统调度与温度波动仍影响结果。
|
||||||
|
|
||||||
|
| 指标 | 优化前 | 优化后 | 用时变化 |
|
||||||
|
|---|---:|---:|---:|
|
||||||
|
| C编码写出(墙钟,诊断组) | 1.1808(1.1529~1.2007) | 0.1638(0.1625~0.1671) | -86.13% |
|
||||||
|
| C编码写出(CPU,诊断组) | 1.1804(1.1528~1.1991) | 0.1638(0.1625~0.1671) | -86.12% |
|
||||||
|
| 点击运行 → 结果可查看 | 8.0100(7.9657~8.0463) | 6.9756(6.9228~6.9863) | -12.91% |
|
||||||
|
| 点击运行 → 缓存保存完成(观察点) | 8.1280(8.0499~8.1684) | 7.0841(7.0053~7.0849) | -12.84% |
|
||||||
|
| CSV点击 → 下载保存 | 1.1127(0.9959~1.3711) | 1.1506(0.8617~1.3136) | +3.41% |
|
||||||
|
| 结果文件点击 → 下载保存 | 1.2254(1.1259~1.3546) | 1.2662(0.9707~1.3405) | +3.33% |
|
||||||
|
| 积分求解(无插桩组) | 6.0926(6.0571~6.0992) | 6.0617(6.0598~6.0752) | -0.51% |
|
||||||
|
|
||||||
|
CSV和结果文件导出依然由已有浏览器路径生成,本轮未改。CSV用时波动范围重叠,不能判定加速;文件大小及SHA在全部16次运行间完全一致(31,820,845字节)。积分耗时的少量变化也不归因于编码:求解代码、计数和输出数值保持一致。
|
||||||
|
|
||||||
|

|
||||||
|
|
||||||
|
生产 `series` 字节数 **32,640,796 → 31,609,208(减少3.16%)**,诊断组HTTP响应体中位数 **34,475,145 → 33,442,734字节(减少2.99%)**。HTTP体另含元数据与进度,随时间文本略有波动;完整浏览器导出结果约33.85 MB,因浏览器重新编码而基本不变。
|
||||||
|
|
||||||
|
阶段诊断组的其他主要区间如下(单位ms,中位数):
|
||||||
|
|
||||||
|
| 阶段 | 优化前 | 优化后 | 边界说明 |
|
||||||
|
|---|---:|---:|---|
|
||||||
|
| 前端提交前预处理 | 18.800 | 22.600 | 模型检查、快照/XML生成及提交准备 |
|
||||||
|
| 后端XML校验 | 23.855 | 22.529 | 请求输入验证 |
|
||||||
|
| 网络编译 | 45.417 | 41.611 | 连接/方程编译 |
|
||||||
|
| 生成C | 59.194 | 60.283 | 生成模型代码 |
|
||||||
|
| 构建缓存校验 | 45.505 | 34.094 | 正式运行全部命中 |
|
||||||
|
| 积分求解 | 6113.293 | 5957.849 | 含积分器、RHS/Jacobian、事件和采样 |
|
||||||
|
| 输出投影 | 85.804 | 79.819 | 重算采样点输出,位于编码前 |
|
||||||
|
| C结果编码写出 | 1180.792 | 163.829 | 含fopen、编码、stdio、fclose及索引 |
|
||||||
|
| Python索引结果读取 | 46.351 | 40.434 | 父区间,含字节读取/小元数据解析 |
|
||||||
|
| HTTP结果事件组装编码 | 32.844 | 33.361 | 元数据JSON与原始series片段拼接 |
|
||||||
|
| 后端HTTP全程 | 7720.603 | 6553.295 | 包含上述后端子区间及ASGI发送等待 |
|
||||||
|
| 浏览器流文本解码 | 27.700 | 27.000 | 同步TextDecoder调用累计 |
|
||||||
|
| 浏览器结果JSON解析 | 99.700 | 104.300 | 单次原始JSON.parse |
|
||||||
|
|
||||||
|
C写出占原生 `main` 时间的比例从 **16.00%降至2.64%**;新版积分占 **96.05%**,输出投影约 **1.29%**。这里先计算每次运行的阶段/父区间比例,再取中位数。浏览器解析未见改善;下一步更大的速度空间仍在求解计算,结果侧剩余CPU时间已明显缩小。
|
||||||
|
|
||||||
|
首次运行单列(无插桩组,各一次,构建缓存未命中):
|
||||||
|
|
||||||
|
| 观察值 / s | 优化前 | 优化后 |
|
||||||
|
|---|---:|---:|
|
||||||
|
| 原生构建 | 3.6580 | 4.3720 |
|
||||||
|
| 点击到结果可查看 | 11.6815 | 11.3810 |
|
||||||
|
| 点击到缓存保存完成 | 11.8062 | 11.5025 |
|
||||||
|
|
||||||
|
本次新构建增加约0.714 s,抵消了大部分编码收益;首次页面运行仅缩短约0.30 s。这是单次冷构建观察,未做重复冷编译统计,不能推广为稳定冷启动提速。旧/新无插桩与诊断程序构建身份不同,均单独预热,不混入正式三次。
|
||||||
|
|
||||||
|
## 正确性、边界与限制
|
||||||
|
|
||||||
|
- 数字编码最终 **10项专项测试通过**。覆盖34,254个有限binary64位模式、正负零、极大极小/次正规数、十进制边界与随机值;默认及 `RYU_ONLY_64_BIT_OPS` 两种路径均逐位回读一致。覆盖64 KiB边界、stride、非有限值拒绝、短写/零写、`/dev/full` 与关闭失败。见 [最终测试日志](../../test/c-result-encoding-20260911/writer-final-tests.log)。
|
||||||
|
- 相关后端传输/取消、代码生成、HTTP、CSV与纯C后端 **29项回归通过**。该次同时运行当时9项数字编码测试,共38项;之后补充普通/科学token选择测试并重新运行最终10项编码测试。见 [回归日志](../../test/c-result-encoding-20260911/backend-tests.log)。
|
||||||
|
- 8次诊断原生结果以一份旧版为基准,其余7份全部 **1,790,486个数值逐位一致**,每份含1,044个负零;series、final和finalState均覆盖,不使用容差或抽样。状态和求解计数也相同,仅排除求解墙钟/CPU元数据。见 [native-bit-parity.json](../../test/c-result-encoding-20260911/native-bit-parity.json)。
|
||||||
|
- 16次真实网页完整series/final、CSV全部单元格、下载结果文件和刷新恢复通过;CSV比较 **28,617,120个单元格**,全部CSV SHA一致。网页比较是解析后的数值严格相等,原生64位检查另行补足负零验证。见 [equality.json](../../test/c-result-encoding-20260911/equality.json)。
|
||||||
|
- 原生CLI额外向 `/dev/full` 写出,返回退出码3且未发布结果索引。见 [写失败验证](../../test/c-result-encoding-20260911/write-failure-check/summary.json)。
|
||||||
|
- 所有网页运行均为1002采样、1784变量,nfev=74,265、接受步6,974、拒绝步454、njev=475、nlu=1,656、事件1、启动4。输入、导入导出参数/连接与前端资产哈希一致。
|
||||||
|
- 与本轮冻结源码比对,已有文件只改变C输出main、原生构建头文件扫描、README和许可;物理内核、积分器、代码生成方程、runner及前端生产资产未变。Ryu引入的C/头文件与Boost许可逐文件哈希等于固定上游快照。见 [source-manifest.json](../../test/c-result-encoding-20260911/source-manifest.json)。
|
||||||
|
|
||||||
|
本轮未改变数值求解算法,因此此前Amesim曲线差异结论保持不变。没有新的Amesim同工况CPU/墙钟数据,不作Amesim速度比较。源码兼容性考虑了GCC/MinGW,已在Linux GCC13.3实测默认及纯64位Ryu路径;本轮没有Windows运行实测。
|
||||||
|
|
||||||
|
后端输出期间仍包含输出投影;Python片段整理、传输、浏览器解析和保存也各有成本。父子区间及并行区间不能相加,独立阶段中位数不保证相加等于总耗时中位数。CSV生产实现本轮未修改,其下载用时差异仅记录为观察,不归因于C编码优化。
|
||||||
|
|
||||||
|
## 文件与复现
|
||||||
|
|
||||||
|
- 本报告:[C端结果编码与写出优化-2026-09-11.md](C端结果编码与写出优化-2026-09-11.md)。
|
||||||
|
- 所有原始产物:[test/c-result-encoding-20260911/](../../test/c-result-encoding-20260911/),Git忽略。
|
||||||
|
- 调研上游快照与SHA:[ryu-upstream/manifest.json](../../test/c-result-encoding-20260911/ryu-upstream/manifest.json);优化前冻结源码:[baseline-source/manifest.json](../../test/c-result-encoding-20260911/baseline-source/manifest.json)。
|
||||||
|
- 网页与后端汇总:[summary.json](../../test/c-result-encoding-20260911/summary.json)、[timings.csv](../../test/c-result-encoding-20260911/timings.csv)。
|
||||||
|
- 数字编码测试:[test_native_json_writer.py](../../tests/test_native_json_writer.py);传输与取消:[test_native_result_transport.py](../../tests/test_native_result_transport.py)。
|
||||||
|
|
||||||
|
当前优化版无插桩网页保留在 **http://127.0.0.1:8027/**,可导入同一八路JSON复查。计时数据、下载、截图、临时构建和上游调研快照都在被Git忽略的 `test/` 下;未安装新环境、提交或推送Git。源码Ryu依赖及许可应作为项目实现保留,不属于应忽略的本地运行环境。
|
||||||
|
|
||||||
|
仓库根目录运行。服务与浏览器应分两个终端启动,输出目录选新路径;下方只是新版复测例子,勿与其他仿真/编译并行。旧版重放使用 `baseline-source/tests/manual/backend_stage_profile.py`,显式 `--frontend-dist frontend/dist`;旧版诊断输出必须置于 `baseline-source` 内,以满足构建器的源码相对路径要求。
|
||||||
|
|
||||||
|
```bash
|
||||||
|
.venv/bin/python -m unittest tests.test_native_json_writer tests.test_native_result_transport tests.test_native_codegen tests.test_generic_system_xml_simulation tests.test_result_csv_export tests.test_native_only_backend -v
|
||||||
|
|
||||||
|
# 终端1:无插桩新版,新的端口和产物目录
|
||||||
|
.venv/bin/python tests/manual/backend_stage_profile.py --plain --port 8029 --output-dir test/c-encoding-recheck/backend
|
||||||
|
|
||||||
|
# 终端2:浏览器沿用已存在的本地运行条件
|
||||||
|
LD_LIBRARY_PATH="$PWD/.venv/native/browser-libs/usr/lib/x86_64-linux-gnu${LD_LIBRARY_PATH:+:$LD_LIBRARY_PATH}" .tools/node-v24.18.0-linux-x64/bin/node tests/manual/browser_stage_profile.mjs --url http://127.0.0.1:8029 --input tests/data/test-mql-8-corrected.json --mode control --runs 3 --output test/c-encoding-recheck/browser
|
||||||
|
```
|
||||||
|
|
||||||
|
诊断组去掉服务的 `--plain`,浏览器使用 `--mode profiled`;请另选新产物目录、端口并串行测试。独立编码实验、计时聚合和完整核对使用:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
.venv/bin/python tests/manual/benchmark_native_result_encoding.py --result-json test/web-cost-20260911/native-compute-profile/control/run-1/result.json --output-dir test/c-encoding-recheck/replay --ryu-root test/c-result-encoding-20260911/ryu-upstream --run --warmups 1 --repeats 3 --dev-null
|
||||||
|
.venv/bin/python tests/manual/summarize_native_encoding.py --root test/c-result-encoding-20260911
|
||||||
|
.venv/bin/python tests/manual/compare_browser_stage_outputs.py test/c-result-encoding-20260911 --native test/web-cost-20260911/native-compute-profile/control/run-1/result.json --group baseline=test/c-result-encoding-20260911/browser-baseline --group baseline-profiled=test/c-result-encoding-20260911/browser-baseline-profiled --group optimized=test/c-result-encoding-20260911/browser-optimized --group optimized-profiled=test/c-result-encoding-20260911/browser-optimized-profiled --output test/c-result-encoding-20260911/equality.json
|
||||||
|
```
|
||||||
|
|
||||||
|
原生逐位工具 [compare_native_result_bits.py](../../tests/manual/compare_native_result_bits.py) 使用 `--baseline 旧结果.json --candidate 新结果.json --output 报告.json`,`--candidate` 可重复。准确输入路径记录于现有 `native-bit-parity.json`。图表由 [plot_encoding.py](../../test/c-result-encoding-20260911/plot_encoding.py) 用系统Python/matplotlib生成,可缩放图为 [overview.svg](assets/2026-09-11/c-result-encoding-20260911-overview.svg)。大结果数值核对安排在全部性能计时结束之后。
|
||||||
|
After Width: | Height: | Size: 184 KiB |
|
After Width: | Height: | Size: 88 KiB |
|
After Width: | Height: | Size: 170 KiB |
|
After Width: | Height: | Size: 107 KiB |
|
After Width: | Height: | Size: 88 KiB |
|
After Width: | Height: | Size: 63 KiB |
|
After Width: | Height: | Size: 159 KiB |
|
After Width: | Height: | Size: 111 KiB |
@@ -0,0 +1,78 @@
|
|||||||
|
# test-mql-8 当前 AME 归档与完整曲线核查
|
||||||
|
|
||||||
|
2026-09-11。只读解析当前输入、AME 归档和已完成的原生运行,不重新求解、编译或修改模型。
|
||||||
|
|
||||||
|
**当前八路归档的数据身份、公开参数和初始状态相互对应,可以作为已保存曲线的观察参考。完整曲线仍有差异,不能判为一致;Amesim 没有可信 CPU/墙钟耗时,速度比较明确 `skip`。** 归档名义容差为 1e-7,本轮原生为 1e-8,亦不构成同精度的性能比较。
|
||||||
|
|
||||||
|
## 本轮来源
|
||||||
|
|
||||||
|
| 文件 | SHA-256 |
|
||||||
|
|---|---|
|
||||||
|
| `tests/data/test_mql.ame` | `1ff0ea4284b9248260eeceb8b27cd0bc14dbccb43554ab8c0b49900d0b944c3d` |
|
||||||
|
| `tests/data/test-mql-8-corrected.json` | `670977bef67e62d9c66e8af497bada208bd72a7301be45128d185d47282cf288` |
|
||||||
|
| `test/mql8-efficiency-20260911/native-production/run-1/result.json` | `12a2018119c8f231056dfd834dfc9ac096945173f3982c98ee2b385140b2c7dc` |
|
||||||
|
|
||||||
|
本轮正式原生 run-1 已完整算到 10 s。其 `series`、`final`、`finalState` 与先前 `test/solver-newton-20260911/mql8/native-run-rtol1e8/execution/result.json` 逐值完全相同,比较产物已经绑定本轮 run-1。见 [逐值相等记录](../../test/mql8-current-archive-audit-20260911/native-result-equivalence.json)。
|
||||||
|
|
||||||
|
历史 `tests/baselines/simulation/test_mql_8/manifest.json` 绑定的是另一个 AME SHA:`cbc3aadd4569a49b3a63e5d66d4143ec16126c0f950df73fb637e07673c20fbb`。该旧基准不用于本轮。另一个现有 `test/mql8-efficiency-20260911/amesim-archive-comparison/comparison.json` 确实绑定当前 1ff0… 归档,但仅覆盖 140 条状态/阀流量曲线;本次补入碰撞力、限位力、间隙及原生额外事件采样。
|
||||||
|
|
||||||
|
## 归档是否自洽
|
||||||
|
|
||||||
|
归档成员均为 `test_mql_` 前缀,子文件 SHA 和字节数完整保存在 [comparison.json](../../test/mql8-current-archive-audit-20260911/comparison.json)。
|
||||||
|
|
||||||
|
| 成员 | 本次核查 |
|
||||||
|
|---|---|
|
||||||
|
| `.cir` | 117 个 COMP、40 个有子模型 LINE,157 个公开元件;以真实端口/接触/管线方向核对 corrected JSON 的全部 178 条连接、1092 个公开参数,均对应 |
|
||||||
|
| `.param` / `.data` | 各 1170 行;重新求值图纸与数据表中的表达式,1015 个直接绑定的公开参数均一致,2 个 LMECHN1 结构字段只保存在图纸 |
|
||||||
|
| 编译 `.c` | 声明 157 个子模型引用、132 个连续状态、24 个离散状态、1278 个变量;GParamInfo 对应全部 1170 行,1154 个具名参数所属模型可核对;1100 个具名 GVarInfo 的模型实例和变量名逐项匹配 `.var` |
|
||||||
|
| `.modelinfo` | 132 个连续状态、24 个离散状态,与编译 C 相同 |
|
||||||
|
| `.state` | 全部 132 个状态 Data_Path 与编译 C 的 GcontStateVarNum 顺序一致;其文件时间较旧本身没有造成状态错配 |
|
||||||
|
| `.var` / `.results` | 1278 个变量行,1116 个保存变量,1100 条具名曲线;1002 个严格递增时间点,覆盖 0–10 s,具名保存值均有限;132 个状态首点逐项对应当前 `.data` 初值 |
|
||||||
|
| `.sim` | 首行 `0 10 0.01 1e+30 1e-07 0.001 4 0.1`;不把未解码的求解器枚举位直接当作 CVODE/BDF 设置 |
|
||||||
|
| `.ameperf` | 仅包含 SC/DISC 的仿真时刻及事件计数,没有 CPU、墙钟或完整运行耗时 |
|
||||||
|
|
||||||
|
`.param` 的第 766、768、770、776 行将介质显示为 `PNGD_HELIUM instance 1`,编译参数表对应 `PNGD00 instance 1`;四行 Data_Path 均属同一 `pn_gas_data`,在审计中明确列作显示名别名。16 个 HIDDEN 参数行、82 个 HIDDEN 编译变量项没有伪造可读名称来宣称逐名核对。
|
||||||
|
|
||||||
|
这些检查未发现当前归档发生四路图纸/八路缓存错配,提供了使用其保存曲线的依据。它仍是归档中的历史运行,未在本轮观察 Amesim 重新运行,不能提供新的控制精度或耗时验收。
|
||||||
|
|
||||||
|
## 变量与误差口径
|
||||||
|
|
||||||
|
覆盖全部 132 个连续状态对应的物理量:56 条压力、56 条温度、10 条位移、10 条速度;另含 8 条阀流量、8 条接触力、20 条质量元件限位力和 8 条接触间隙,共 176 条曲线。
|
||||||
|
|
||||||
|
变量身份通过归档编译 C 的 GVarInfo/GcontStateVarNum、`.var` Data_Path 和图纸拓扑映射确定。单位来自对应 `.var` 标签与仓库保存的元件 C:PNVO001 的 `dm2` 保存单位为 g/s,乘 `-0.001` 转成本系统 port_2 的流入 kg/s;压力为表压 Pa,加 101300 Pa 转为绝对压力;温度 K、位移 m、速度 m/s、力 N 直接比较,间隙 mm 乘 0.001。见 [PNVO001.c](../../tests/data/AmesimModels/help/source/PNVO001.c) 中端口定义及末尾转换、[PNL0001.c](../../tests/data/AmesimModels/help/source/PNL0001.c) 中 `pa = p + PATM`。这些元件源文件用于解释单位,不作为另一个 AME 归档的数据参考。
|
||||||
|
|
||||||
|
LSTP00A 的归档 f2 明确为 f1 的重复量,8 组保存数组逐值相同;本系统两个接口力逐值反号。因此本文以 port_1.f 对 f1;port_2.f 对负 f2,得到相同误差。质量元件的 20 条 Fmin/Fmax 在双方保存点均为零,此项不能代替接触力检查。
|
||||||
|
|
||||||
|
对每一方的原始保存网格,分别将另一方原始曲线线性插值到该网格;不平移时刻、不平滑、不删除事件点。每条曲线输出最大绝对差、采样均方根差、最大差/参考峰值、RMS/参考 RMS,以及参考非零时的逐点相对差;参考为零的点另外保留绝对差。RMS 是所述采样网格上的等权 RMS。原生网格和归档网格分别报告,避免将没有保存的 Amesim 事件状态冒称为实测值。
|
||||||
|
|
||||||
|
## 保存点上的实际差异
|
||||||
|
|
||||||
|
下表是归档 1002 个保存点上的结果,每类列出最大绝对差所在曲线及该曲线 RMS;“峰值归一化”是该差除以该曲线参考峰值,不是逐点相对误差。
|
||||||
|
|
||||||
|
| 量 | 最差曲线 | 最大绝对差 | RMS 差 | 峰值归一化 | 最大差时刻 |
|
||||||
|
|---|---|---:|---:|---:|---:|
|
||||||
|
| 压力 | `amesim_pnl0002_6.p` | 163395.189 Pa | 7170.441 Pa | 0.792631% | 0.01 s |
|
||||||
|
| 温度 | `amesim_pnl0001_13.T` | 105.215279 K | 5.829065 K | 2.033444% | 0.01 s |
|
||||||
|
| 阀流量 | `amesim_pnvo001_7.port_2.m_flow` | 0.002990901 kg/s | 0.000114014 kg/s | 0.766320% | 0.04 s |
|
||||||
|
| 接触力 | `amesim_lstp00a_7.port_1.f` | 5131.970918 N | 225.217793 N | 0.796525% | 0.01 s |
|
||||||
|
| 接触间隙 | `amesim_lstp00a_2.gap` | 5.215260e-8 m | 2.265633e-8 m | 1.112982% | 1.27 s |
|
||||||
|
|
||||||
|
位移的最大绝对差为 `amesim_mecmas21_9.x` 的 53,857,561 m,RMS 为 52,430,984.93 m;该曲线参考终值本身为 7.68e15 m,所以不能脱离物理量级解读。该元件在双方均出现约 8e14 m/s 速度;当前 AME 的 UD00 输入含 1e17 的初始信号,不能把这种极端输入改小后再当成同模型验收。
|
||||||
|
|
||||||
|
各曲线按自身峰值归一化后,位移最差为 `amesim_mecmas21_5.x`:最大差 2.193606e-5 m、RMS 1.406520e-6 m、峰值归一化 0.00592859%;速度最差为 `amesim_mecmas21_10.v`:最大差 2.437279e-4 m/s、RMS 1.964328e-5 m/s、峰值归一化 0.00615392%。两者最大差在 0.98 s。全部逐曲线、逐网格统计见 [curve-differences.csv](../../test/mql8-current-archive-audit-20260911/curve-differences.csv)。
|
||||||
|
|
||||||
|
## 碰撞事件使完整曲线不一致
|
||||||
|
|
||||||
|
原生额外保存 `t = 0.9833956321732664 s`,170000 kg 的 `amesim_mecmas21_10` 此时到达 0.37 m 限位并将速度置零。8 个 LSTP00A 的原始接口力同时约为 **4.034997e11 N**。不能丢弃这一点后宣称完整曲线吻合。
|
||||||
|
|
||||||
|
归档在相邻 0.98 s、0.99 s 保存数据,没有这一事件时刻;将这两个点线性插值得到的接触力约 467936 N。以全部原生保存点为网格,最差接触力最大差为 **4.034993e11 N**,采样 RMS 为 **1.274703e10 N**。同一点,负载速度与归档插值差 2.615678 m/s、位移差 0.008950590 m。这些是保存曲线之间的差异,插值跨越限位事件,不能当作 Amesim 真实事件状态的误差。
|
||||||
|
|
||||||
|
双方保存的负载数据都显示限位发生在 0.98–0.99 s 内。归档 `.ameperf` 没有提供该负载在此区间的精确限位时刻;只有采样数据不足以确定两者精确事件偏移,也无法判断 Amesim 内部是否出现同类极窄接触峰。此前 0.01 s 的压力、温度、接触力误差发生得更早,不能全部归因于 0.9834 s 这一未对齐的保存事件。
|
||||||
|
|
||||||
|
因此本轮结论是“已完成全点差异观察,尚未通过八路曲线一致性验收”。未预设八路接受阈值,也未取得足够密的 Amesim 事件输出,不将这些差异自动判成精度达标。
|
||||||
|
|
||||||
|
## 耗时与复现
|
||||||
|
|
||||||
|
本轮原生 run-1 的纯求解墙钟为 5.797191 s、CPU 为 5.796605 s;这里只标识已有运行,完整性能分析由本轮优化总报告给出。归档中没有可用 Amesim CPU/墙钟记录。当前环境的 PATH、相关环境变量及 Python 模块检查亦未发现可调用 Amesim 安装;归档生成 C 标记为 Simcenter Amesim 2404,不等于当前可调用软件版本。不得用成员修改时间差或 `.ameperf` 的 10 s 仿真终点推算实际耗时,**Amesim 速度比较 `skip`**。
|
||||||
|
|
||||||
|
复现本次只读数值后处理:`.venv/bin/python test/mql8-current-archive-audit-20260911/audit_and_compare.py`。脚本输出来源/成员哈希、参数及状态身份、两种原始网格的误差、额外事件与未验收边界,不启动任何求解器。长期模型选择、先正确性后计时、固定精度及分阶段记录遵循 [优化验证约定](../standard/optimization-benchmark-model.md)。
|
||||||
@@ -0,0 +1,210 @@
|
|||||||
|
# 八路模型计算效率与网页阶段计时(2026-09-11)
|
||||||
|
|
||||||
|
本轮以 `808c484` 的正式代码和 [test-mql-8-corrected.json](../../tests/data/test-mql-8-corrected.json) 为主案例。八路在当前默认精度下完整运行,未退回四路。本轮增加独立诊断脚本和运行记录,没有修改生产算法、工程参数或输出采样。后续遵循 [八路优先的优化验证约定](../standard/optimization-benchmark-model.md)。
|
||||||
|
|
||||||
|
## 运行范围与可复现条件
|
||||||
|
|
||||||
|
- 工程 SHA256:`670977bef67e62d9c66e8af497bada208bd72a7301be45128d185d47282cf288`;CLI XML SHA256:`2804b33f04cabb3c10fcc3cc0843c35de17efbe1b6ee5e5e821edd9a5515a23d`。对应 AME 为 `tests/data/test_mql.ame`,SHA256 `1ff0ea4284b9248260eeceb8b27cd0bc14dbccb43554ab8c0b49900d0b944c3d`。
|
||||||
|
- 157 元件、178 条连接、132 连续状态;返回 1784 个变量、1002 个原始采样点(含事件点),仿真 0~10 s,输出间隔 0.01 s。
|
||||||
|
- CVODE BDF / SUNDIALS 7.4.0,`rtol=1e-8`,`max_step=1e30`;自动首步。实际绝对误差限按状态生成:质量 `1e-14 kg`、内能 `1e-8 J`、速度 `1e-12 m/s`、位移 `1e-12 m`,并非对所有状态统一使用配置对象中的 `atol=1e-8`。数值雅可比、稠密线性求解和事件处理策略保持生产设置。
|
||||||
|
- Linux x86_64 虚拟机,Intel Xeon Silver 4210R @ 2.40 GHz,16 逻辑 CPU、约 30 GiB 内存;GCC 13.3.0、Python 3.12、Node 24.18.0、Chromium 151.0.7922.34。原生进程主要为单核计算,环境与构建哈希保存在原始记录。
|
||||||
|
- 各组预热一次、正式运行三次;中位数与范围单列。三算法组交错执行;网页对照组与分阶段插桩组串行执行,不同时运行诊断求解、绘图或大编译。样本数较少,组间百分之几的差异不作为显著优化结论。
|
||||||
|
|
||||||
|
## 管流算法效率与触顶情况
|
||||||
|
|
||||||
|
三算法使用同一 corrected 八路、物性复用实现和全系统积分精度。固定点版本仅在明确的 `5d5a2e1` 基线上恢复原半步松弛管流循环,其余生产数值逻辑保持一致;不是运行旧版整个应用。前次牛顿取自该提交,当前保护牛顿取自本轮正式内核。此组没有逐次调用计数插桩。
|
||||||
|
|
||||||
|
| 算法 | 纯求解中位数(范围)/ s | 进程全程中位数 / s | RHS 次数 | 接受 / 拒绝步 |
|
||||||
|
|---|---:|---:|---:|---:|
|
||||||
|
| 旧固定点,16 / 64 轮上限 | 12.0976(11.9926~12.1511) | 14.2514 | 119058 | 9647 / 679 |
|
||||||
|
| 前次牛顿 | 5.8183(5.7829~5.8944) | 7.9144 | 74265 | 6974 / 454 |
|
||||||
|
| 当前保护牛顿+二分回退 | 5.9138(5.9115~5.9323) | 7.9698 | 74265 | 6974 / 454 |
|
||||||
|
|
||||||
|
相对旧固定点,当前纯求解时间减少 **51.12%**、速度约 **2.05 倍**,进程全程减少 **44.08%**,RHS 次数减少 **37.62%**。相对前次牛顿,本组耗时增加约 1.64%,不能宣称保护机制带来额外提速;两版牛顿的完整采样、最终输出和状态逐值相同。
|
||||||
|
|
||||||
|
另用未经本轮插桩的正式构建独立运行一次预热、三次正式检查:纯求解中位数 **5.7972 s**(5.7897~5.8214),进程全程 **7.8504 s**(7.8437~7.8685),均完成 10 s;每次 RHS 74265、接受 6974、拒绝 454、雅可比 475、线性分解 1656、状态转换 1、求解器启动 4。该组构建缓存命中,输入准备 0.1493 s、构建/缓存核验 0.03294 s;这些准备时间在三个进程运行前发生一次,不包含在纯求解时间内。单独组的 5.80 s 不与上一组旧算法拼接计算加速比。
|
||||||
|
|
||||||
|
为区分算法本身和积分轨迹变化,另对三版内核进行独立计数诊断:先各自跑完整八路,再将当前轨迹的 **3,081,320 组实际管阻方程输入全部原样重放**给三个版本,不抽样、不去重。诊断耗时不用于上述性能比较。
|
||||||
|
|
||||||
|
| 同一组管阻输入的计数 | 旧固定点 | 前次牛顿 | 当前保护牛顿 |
|
||||||
|
|---|---:|---:|---:|
|
||||||
|
| 解析返回 / 迭代求解 | 0 / 3081320 | 2370685 / 710635 | 相同 |
|
||||||
|
| 累计循环轮数 | 49431317 | 3081555 | 相同 |
|
||||||
|
| 单次最多轮数 | 64 | 6 | 6 |
|
||||||
|
| 进入最后允许轮次 | 3037792(98.587%) | 0 | 0 |
|
||||||
|
| 耗尽额度且未满足各自停止条件 | 3010570(97.704%) | 0 | 0 |
|
||||||
|
| 实际返回值的统一相对残差 > 1e-9 | 3072455 | 0 | 0 |
|
||||||
|
| 二分回退 / 非有限返回 | 0 / 0 | 0 / 0 | 0 / 0 |
|
||||||
|
|
||||||
|
牛顿累计轮数减少 **93.766%**。710635 次迭代调用分别用 3 轮 74290 次、4 轮 361601 次、5 轮 236183 次、6 轮 38561 次。当前案例没有触发二分或失败路径,所以它验证了正常输入的效率和一致性,不能代替极端状态下的保护测试,也不能证明任意系统必然收敛。
|
||||||
|
|
||||||
|
“进入最后一轮”与“耗尽仍未满足条件”分开统计:固定点 PNL00R 的 64 轮分支有 1638 次进入最后一轮、但没有耗尽失败;真正耗尽的都来自其他管路的 16 轮分支。旧循环满足自身停止条件的 70750 次中,61885 次仍不满足当前统一方程残差门槛;相对残差超标不直接代表同等数量的大绝对流量误差。
|
||||||
|
|
||||||
|
各自完整积分轨迹上,旧固定点实际管阻调用 5041645 次、耗尽 4937544 次;两版牛顿均为 3081320 次、耗尽 0 次。零压差 247634 次和 PNL00R 直接层流 235766 次另计,未混入混合摩擦方程残差。当前插桩结果与独立正式构建的 `series/final/finalState` 逐值相同。
|
||||||
|
|
||||||
|
该八路在积分 RHS 中请求管流缓存 2970600 次,命中 **0**;另有 594120 次直接调用。物性复用与这项管流缓存不同;此结果只说明本例可进一步评估缓存查找的成本,不能直接推广到其他拓扑或据此删除缓存。
|
||||||
|
|
||||||
|
## 真实网页主流程与额外操作成本
|
||||||
|
|
||||||
|
使用正式 Vite 静态构建,经 Playwright 驱动 Chromium;无模拟 API、预制结果注入或重复读取响应体。独立对照服务 `127.0.0.1:8013` 不加载诊断插桩;分阶段服务 `127.0.0.1:8012` 仅在隔离 C 主程序中增加稀疏时钟,并包装实际后台函数。两个服务的求解器和元件文件逐字节相同,前端构建集合 SHA256 同为 `2faa6ce32ee9997187870d64af2bc235a830eb7b8def7749ddcb0eaeeed4831e`;计时脚本 SHA 也相同。
|
||||||
|
|
||||||
|
每组预热一次、正式三次。每轮从公开 JSON 导入入口加载同一 corrected 文件,公开导出工程后逐项验证全部参数、端点和仿真设置,然后点击运行;导入与核查不包含在点击运行计时中。每轮继续打开结果页、选择指定管路温度、下载 CSV 与结果文件,再刷新并核对恢复结果。结果页和温度曲线立即打开,允许浏览器 IndexedDB 异步保存与查看过程自然重叠。
|
||||||
|
|
||||||
|
下表单位为秒,格式为 **中位数(最小~最大)**。对照组是常规网页成本的主记录;“首次绘制机会”指可见 DOM 后经过两次 requestAnimationFrame,未测量 GPU 完成时间。
|
||||||
|
|
||||||
|
| 阶段 | 对照组 / s | 分阶段组 / s |
|
||||||
|
|---|---:|---:|
|
||||||
|
| 工程导入 change → 画布首次绘制机会 | 0.2527(0.2273~0.2862) | 0.2312(0.2251~0.2411) |
|
||||||
|
| 点击运行 → 完成状态 DOM | 9.8716(9.7697~9.8826) | 9.7777(9.7358~9.8408) |
|
||||||
|
| 点击运行 → 完成状态首次绘制机会 | 9.8989(9.7840~9.9104) | 9.8005(9.7596~9.8660) |
|
||||||
|
| 点击运行 → 结果页首次绘制机会(自动切页) | 10.1516(10.0600~10.1834) | 10.0589(10.0305~10.1056) |
|
||||||
|
| 点击运行 → 温度曲线首次绘制机会(自动选曲线) | 10.6899(10.5762~10.7431) | 10.6160(10.5916~10.6433) |
|
||||||
|
| 点击运行 → 观察到 IndexedDB 保存完成指针 | 12.2922(11.8303~12.4076) | 12.4344(11.5322~12.4967) |
|
||||||
|
| 进入结果页 → 首次绘制机会 | 0.1910(0.1893~0.1980) | 0.1927(0.1832~0.1946) |
|
||||||
|
| 选择温度 → 曲线首次绘制机会 | 0.0269(0.0262~0.0297) | 0.0276(0.0233~0.0293) |
|
||||||
|
| CSV 点击 → 下载并保存 | 5.6198(5.5717~5.6381) | 5.5732(5.4253~5.6625) |
|
||||||
|
| 结果文件点击 → 下载并保存 | 1.2308(1.2265~1.3213) | 1.2921(1.1552~1.5145) |
|
||||||
|
| 刷新导航 → 已恢复结果首次绘制机会 | 0.5885(0.5406~0.6592) | 0.5487(0.5074~0.5630) |
|
||||||
|
|
||||||
|
无插桩对照组的 C 纯求解中位数为 **6.0653 s**、进程全程 **7.9499 s**;独立 CLI 的 5.7972 s 属另一运行环境下的分组结果,不把差值直接归因于网页某项 CPU 工作。分阶段组点击到完成绘制为 9.8005 s,对照为 9.8989 s,相差约 -0.99%;本组三次样本未见明显端到端插桩增量,也不能把该负差当作加速。持久化和导出波动更大,逐次原始值均保留。
|
||||||
|
|
||||||
|
初次导航只记录每组一次:HTML `loadEventEnd` 对照 0.0971 s、分阶段 0.1086 s;Playwright 观察到工程导入控件时分别为 0.5628 / 0.6064 s,包含自动化观察与调度延迟,不是严格的首次可交互时间。首次工程导入到两帧观察点,对照预热为 0.3716 s;三次重复导入中位为 0.2527 s。页面加载、导入、仿真、查看与导出分别记录,未用自动化操作间隙拼成一个“纯网页耗时”。
|
||||||
|
|
||||||
|
“DOM 稳定”另取 120 ms 无变动再等两帧:结果页中位 0.3710 s、温度曲线 0.1736 s、刷新恢复 0.7575 s(对照)。这包含人为安静窗口,不能把它全称为渲染工作。下载完成包含 Playwright 通知、`saveAs` 与文件系统开销,不等于浏览器内序列化 CPU 时间。
|
||||||
|
|
||||||
|
## 后台内部各阶段
|
||||||
|
|
||||||
|
用每个请求的 `X-Simulation-Id` 将网页与后台记录一一关联。下表为分阶段组正式三次的秒数;所列主要阶段在单个请求内依次发生。各阶段分别取中位数,故其相加不保证等于总耗时中位数。
|
||||||
|
|
||||||
|
| 阶段 | 中位数(最小~最大)/ s |
|
||||||
|
|---|---:|
|
||||||
|
| XML 校验 | 0.0231(0.0222~0.0246) |
|
||||||
|
| 网络编译 | 0.0364(0.0361~0.0465) |
|
||||||
|
| 系统 C 代码生成 | 0.0542(0.0528~0.0548) |
|
||||||
|
| 构建缓存核验(3 次均命中) | 0.0321(0.0308~0.0331) |
|
||||||
|
| C 参数准备 | 0.0000503 |
|
||||||
|
| C 初始化与初始采样 | 0.0001(0.0001~0.0001) |
|
||||||
|
| C 纯积分 | 6.0855(6.0414~6.0994) |
|
||||||
|
| C 最终采样/状态处理 | 0.0000029 |
|
||||||
|
| C 最终及全采样输出投影 | 0.0855(0.0837~0.0861) |
|
||||||
|
| C 结果 JSON 格式化并写文件 | 1.1707(1.1656~1.1945) |
|
||||||
|
| Python 读文件并 UTF-8 解码 | 0.0360(0.0321~0.0429) |
|
||||||
|
| Python 解析原生 JSON | 0.5513(0.5429~0.5588) |
|
||||||
|
| 进程启动/等待/退出等未细分余量 | 0.0029(0.0025~0.0029) |
|
||||||
|
| 结果接口元数据组装 | 0.0330(0.0245~0.1175) |
|
||||||
|
| 最终 NDJSON 结果事件序列化 | 1.1928(1.1862~1.2106) |
|
||||||
|
| 序列化结束 → ASGI 最后响应体发送完成 | 0.1715(0.1489~0.1808) |
|
||||||
|
| HTTP 其余编排/调度余量 | 0.0124(0.0114~0.0155) |
|
||||||
|
|
||||||
|
后台 HTTP 全程中位 **9.5123 s**(9.4458~9.5845)。内部 C main 全程 7.3380 s,Python 原生执行与结果读取包装 7.9194 s,仿真 worker 全程 8.1380 s;这些是包含上表子阶段的父区间,**不能再次相加**。ASGI `send` 的累计 await 为 0.0701 s,可与生成结果的后台任务重叠,亦不代表客户端网络总时间。诊断产物保留另耗约 0.00066 s,已包含在原生执行包装中。
|
||||||
|
|
||||||
|
约 **2.95 s** 用于 C 的 JSON 写出、Python 读取解析及 HTTP 结果 JSON 序列化,约为后台总程的 31%;相比之下输出物理量投影仅约 0.086 s。这里先定位到文本结果处理成本,并未将整个差额笼统归为物性或求根。
|
||||||
|
|
||||||
|
正式网页八次运行均复用已构建的程序。唯一冷构建来自功能预检:编译/核验 **3.4154 s**,构建键 `bcb85b5dce729fe9b7a503919b720be03950ff31839877edd344377d4ee9cfc0` 与后续完整八路分阶段组相同;预检将运行终点设为 0.02 s,只改变运行时选项,未改变编译出的系统程序。这个单次冷编译时间可供首次运行成本参考,但本轮没有测得完整 0~10 s 的冷启动网页总时间,不把预检总时间冒用为正式结果。
|
||||||
|
|
||||||
|
## 浏览器内部、持久化和导出细分
|
||||||
|
|
||||||
|
分阶段组保留了原始 `fetch`、流读取、解码、JSON.parse、IndexedDB 事务和下载锚点的时间戳,没有克隆响应或重复解析。
|
||||||
|
|
||||||
|
| 观察区间 | 中位数 / s | 含义与边界 |
|
||||||
|
|---|---:|---|
|
||||||
|
| 点击运行 → 调用 fetch | 0.0234 | 含前端模型检查、快照/XML 和提交准备,未单独计 XML CPU |
|
||||||
|
| fetch → 收到响应头 | 0.0298 | 收到流式响应头,不表示仿真已经算完 |
|
||||||
|
| 响应头 → 读到流 EOF | 9.6765 | 包含后台计算、传输、消费者处理与调度 |
|
||||||
|
| 累计未完成的流 read 等待 | 9.4636 | 含后台结果尚未生成的等待,不能称为纯网络时间 |
|
||||||
|
| 最终结果的原始 JSON.parse | 0.1125 | 同步解析仅执行一次 |
|
||||||
|
| 全部流消息 JSON.parse | 0.1139 | 已含上一行 |
|
||||||
|
| UTF-8 decode 累计 | 0.0280 | 原始解码调用,不包含分片扫描/拼接 |
|
||||||
|
| 最终 JSON.parse 结束 → 完成 DOM | 0.0430 | 含应用处理和 React 调度,未独立测 React CPU |
|
||||||
|
| 结果解析结束 → IndexedDB 完成指针发布 | 2.6279 | 含批处理、事务等待与调度,可与看图重叠 |
|
||||||
|
| 首次保存事务 → 完成指针发布 | 2.5952 | 仍是异步区间;不是磁盘或主线程独占时间 |
|
||||||
|
| CSV 点击 → fetch | 0.2641 | 包含将整份结果再次 JSON.stringify 的提交准备 |
|
||||||
|
| CSV fetch → 响应头 | 4.6968 | 含浏览器请求准备、后台处理及网络/调度 |
|
||||||
|
| CSV 点击 → Blob 下载锚点 | 5.0143 | 浏览器已经取得并准备好下载数据 |
|
||||||
|
| 结果文件点击 → Blob 下载锚点 | 0.6113 | 包含结果文件序列化及 Blob 准备 |
|
||||||
|
|
||||||
|
最终流式响应约 **33.73 MB**;原生结果 JSON 为 32,725,293 字节,CSV 为 32,346,795 字节,结果文件约 33.85 MB(十进制 MB)。本次是本机回环 HTTP,未模拟远端带宽、TLS 或网络延迟;不能把这些网络相关耗时直接推广到远端部署。
|
||||||
|
|
||||||
|
CSV 后台另有独立请求,按请求顺序和时间戳关联,正式三次分段如下:
|
||||||
|
|
||||||
|
| CSV 后台阶段 | 中位数(最小~最大)/ s |
|
||||||
|
|---|---:|
|
||||||
|
| ASGI 开始 → 收完请求体 | 0.0262(0.0247~0.0285) |
|
||||||
|
| 收完请求体 → CSV 函数开始 | 0.7072(0.7010~0.7138) |
|
||||||
|
| CSV 校验、逐单元格式化与文本组装 | 2.3684(2.3616~2.3885) |
|
||||||
|
| CSV 函数结束 → HTTP 完成 | 0.1252(0.1221~0.1264) |
|
||||||
|
| CSV HTTP 全程(父区间) | 3.2264(3.2177~3.2494) |
|
||||||
|
|
||||||
|
“收完请求体→CSV 函数”约 0.707 s,涵盖路由层 JSON 解码、模型校验及调度,未再将每项 CPU 时间拆开。CSV 浏览器 fetch 到响应头比 ASGI 全程更长;跨进程请求准备、发送及调度的余量未进一步归因。CSV 的完整结果往返及文本组装成本已经实测,后续可据此单独优化。
|
||||||
|
|
||||||
|
## 本轮验证与下一步优先级
|
||||||
|
|
||||||
|
两组共 **8 次**网页运行均完成 10 s,且无 `pageerror`。每轮公开导出的工程参数与连接全部匹配固定输入;结果页、精确单位 K 的温度曲线、CSV、`.simresult` 下载和刷新恢复均通过。8 份 CSV 均为 1785 列 × 1002 数据行,合计 **14,308,560 个数值单元**与独立原生结果精确相等;全部 `series/final` 也精确相等,刷新前后完整结果一致。4 份后台稀疏插桩原生文件的 `series/final/finalState` 均与生产文件逐值相同。56 个唯一气体质量状态用 `math.fsum` 汇总,初始总质量 5.566893015 kg,最大漂移 **1.15463e-14 kg**;所有保存数值有限。
|
||||||
|
|
||||||
|
初次自动化预检曾误用旧控制台 CSS,另一次重复组在刷新后假定建模工程自动恢复而未提交新请求。这两类脚本失败保存在 `browser-stage-smoke/` 与 `browser-control-failed/`,明确排除正式统计。最终脚本使用当前 DockedSimulationConsole、每轮重新导入并核对工程,两个完整正式组均通过。0.02 s 的功能预检只证明操作流程,不混入八路完整计算效率。
|
||||||
|
|
||||||
|
依据本轮数据,后续优先考虑:
|
||||||
|
|
||||||
|
1. **先追踪八路早期压力/温度差和碰撞事件输出。** 参数对齐与局部求根成功不能代替整条曲线正确性;保留八路主案例,继续固定当前输入与精度。
|
||||||
|
2. **减少结果反复文本化。** 当前 C 写 JSON、Python 再解析、HTTP 再序列化合计约 2.95 s;优化这条路径时仍需保留全部变量、采样、浮点数值与输出合同。
|
||||||
|
3. **单独优化 CSV 与浏览器持久化。** CSV 全量回传后台并组装文本明显影响使用时间;持久化约 2 s 以上、且与查看曲线重叠,适合分别评估批大小、复制与调度,不能把保存等待计为绘图耗时。
|
||||||
|
4. **继续剖析 RHS/雅可比等求解成本。** 管流触顶已经为零,当前再削减迭代上限没有实测依据。可先评估本例零命中的管流缓存和 74265 次 RHS 的组成;更改雅可比或缓存策略后仍按同一八路、同精度和完整输出复核。
|
||||||
|
|
||||||
|

|
||||||
|
|
||||||
|
图中三算法与浏览器操作取三次中位数;后台堆叠图选 HTTP 总耗时居中的单个请求,所有片段可相加。浏览器各操作区间不能相加;完整可缩放图为 [SVG](assets/2026-09-11/mql8-efficiency-20260911-stage-costs.svg)。
|
||||||
|
|
||||||
|
|
||||||
|
## 与当前 Amesim 归档的对照
|
||||||
|
|
||||||
|
本次找到可用的八路保存曲线。独立核对当前 AME 内的图纸、参数数据、编译 C、状态与变量索引及初值,未发现此前四路归档中那种图纸/缓存错配;完整说明见 [当前八路归档与曲线核查](test-mql-8当前AME归档与完整曲线核查-2026-09-11.md)。使用的是当前 AME SHA,未套用绑定另一归档的历史八路冻结基准。
|
||||||
|
|
||||||
|
选取全部 132 个连续状态对应的压力、温度、位移与速度,另加阀流量、接触力、限位力和间隙,共 **176 条曲线**;明确映射表压/绝压、SI 单位与接口方向,分别在双方原始保存网格上线性插值对照,不删去原生事件点。下表为 AME 保存网格上各类最差曲线的最大差及该曲线 RMS,不是把全类所有点混算的 RMS。
|
||||||
|
|
||||||
|
| 量 | 最大绝对差 | 对应曲线 RMS |
|
||||||
|
|---|---:|---:|
|
||||||
|
| 压力 | 163395.189 Pa | 7170.441 Pa |
|
||||||
|
| 温度 | 105.215279 K | 5.829065 K |
|
||||||
|
| 阀质量流量 | 0.002990901 kg/s | 0.000114014 kg/s |
|
||||||
|
| 接触力(AME 保存网格) | 5131.971 N | 225.218 N |
|
||||||
|
|
||||||
|
最大压力、温度和此表接触力差在 0.01 s,不能全归因于后续碰撞采样时刻不一致。原生另有 `t=0.9833956321732664 s` 的原始事件点,8 个接触力约 **4.035×10¹¹ N**;AME 只保存相邻 0.98 / 0.99 s 点,无法据其线性插值判断 Amesim 精确事件时刻或是否也有极窄峰。当前 AME 的 UD00 输入含 `1e17` 初始信号,双方公共质量元件都出现极大位移/速度,也不能把“数值有限”当成物理合理性证明。本轮未修改这些 AME 参数。
|
||||||
|
|
||||||
|
因此,**八路运行与计时记录完成,不等于八路曲线一致性验收通过**。目前没有八路专用接受阈值,也缺少足够密的 AME 事件输出;保留差异供后续核查,不为通过而放宽门槛或换四路。
|
||||||
|
|
||||||
|
Amesim **耗时对比跳过**:当前环境没有可调用的 Amesim,归档也没有可信 CPU/墙钟记录;`.ameperf` 中的时间是仿真事件时刻。归档名义容差为 `1e-7`,本轮为 `1e-8`,也不能冒称同精度速度比较。没有重跑 Amesim,没有从 10 s 仿真终点或文件时间戳推算耗时。
|
||||||
|
|
||||||
|
## 记录、脚本与复现
|
||||||
|
|
||||||
|
所有输入快照、编译程序、完整曲线、下载文件和浏览器截图都放在 Git 忽略目录 `test/`,环境仍在 `.venv/native/`,未纳入 Git。报告引用的相对路径在仓库本地可打开;这些大型运行产物不随 Git 克隆传递。
|
||||||
|
|
||||||
|
- [三算法原始基准](../../test/mql8-efficiency-20260911/native-benchmark/summary.json)、[三算法汇总](../../test/mql8-efficiency-20260911/native-summary.json)、[正式构建独立基准](../../test/mql8-efficiency-20260911/native-production/summary.json)。
|
||||||
|
- [完整管阻计数与同输入重放](../../test/solver-newton-20260911/pipe-iteration-profile-8/summary.json)、[插桩与生产逐值核对](../../test/solver-newton-20260911/pipe-iteration-profile-8/production-parity.json)。
|
||||||
|
- [阶段汇总 JSON](../../test/mql8-efficiency-20260911/stage-summary.json)、[阶段汇总 CSV](../../test/mql8-efficiency-20260911/stage-summary.csv)、[网页对照组](../../test/mql8-efficiency-20260911/browser-control/summary.json)、[网页分阶段组](../../test/mql8-efficiency-20260911/browser-profiled/summary.json)。
|
||||||
|
- [网页与 CSV 全量相等核查](../../test/mql8-efficiency-20260911/equality.json)、[后台插桩逐值核查](../../test/mql8-efficiency-20260911/backend-profile-production-parity.json)、[质量守恒](../../test/mql8-efficiency-20260911/mass-conservation.json);每个后台请求在 `test/mql8-efficiency-20260911/backend-profile/requests/<simulationId>/` 保存执行 XML、原生 JSON、日志和阶段时间。
|
||||||
|
- [后台分阶段诊断脚本](../../tests/manual/backend_stage_profile.py)、[真实浏览器计时脚本](../../tests/manual/browser_stage_profile.mjs)、[管阻计数脚本](../../tests/manual/profile_pipe_iterations.py)。[网页结果复核脚本](../../tests/manual/compare_browser_stage_outputs.py)。这些都是手动诊断入口,生产服务无需加载。
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# 八路三算法基准;目录必须为新目录
|
||||||
|
.venv/bin/python tests/manual/benchmark_native_pipe_solver.py \
|
||||||
|
tests/data/test-mql-8-corrected.json --output-dir test/new-mql8-variants --runs 3
|
||||||
|
|
||||||
|
# 正式构建独立基准
|
||||||
|
.venv/bin/python -m app.simulation.native_codegen tests/data/test-mql-8-corrected.json \
|
||||||
|
--output-dir test/new-mql8-production --runs 3 --timeout 120
|
||||||
|
|
||||||
|
# 后台阶段记录服务;需要已有 frontend/dist,选空闲端口
|
||||||
|
.venv/bin/python tests/manual/backend_stage_profile.py \
|
||||||
|
--output-dir test/new-mql8-backend --port 8012
|
||||||
|
# 另一个终端:--plain 为无诊断插桩的对照服务
|
||||||
|
.venv/bin/python tests/manual/backend_stage_profile.py --plain \
|
||||||
|
--output-dir test/new-mql8-control --port 8013 \
|
||||||
|
--frontend-dist test/new-mql8-backend/frontend
|
||||||
|
|
||||||
|
# Chromium 所需依赖仅使用仓库现有忽略环境
|
||||||
|
export LD_LIBRARY_PATH="$PWD/.venv/native/browser-libs/usr/lib/x86_64-linux-gnu${LD_LIBRARY_PATH:+:$LD_LIBRARY_PATH}"
|
||||||
|
.tools/node-v24.18.0-linux-x64/bin/node tests/manual/browser_stage_profile.mjs \
|
||||||
|
--output test/new-mql8-browser-control --url http://127.0.0.1:8013 --mode control --runs 3
|
||||||
|
.tools/node-v24.18.0-linux-x64/bin/node tests/manual/browser_stage_profile.mjs \
|
||||||
|
--output test/new-mql8-browser-profiled --url http://127.0.0.1:8012 --mode profiled --runs 3
|
||||||
|
```
|
||||||
@@ -0,0 +1,158 @@
|
|||||||
|
# 八路结果处理、浏览器保存与 CSV 优化(2026-09-11)
|
||||||
|
|
||||||
|
本轮针对“点击运行→结果可查看”“点击运行→浏览器保存完成”“CSV 下载保存”三个用户关注的时间点实施优化。基线为 `808c484`,主案例仍是 [修正后的八路工程](../../tests/data/test-mql-8-corrected.json),未退回四路。此前各阶段成本见 [八路评估报告](八路模型计算效率与网页阶段计时-2026-09-11.md)。
|
||||||
|
|
||||||
|
## 实现
|
||||||
|
|
||||||
|
### C 数值 JSON 直接进入 HTTP 结果流
|
||||||
|
|
||||||
|
C 继续生成原有 `result.json`,数值仍使用 `%.17g`,新增可选 `--result-index` 小索引文件。索引通过写文件时的实际字节位置标出完整 `series` 对象,包含版本、边界、文件长度和采样数;主结果成功关闭后才发布索引,定位或文件写入失败返回错误。
|
||||||
|
|
||||||
|
网页流式请求启用原生片段传输:Python 校验索引版本、整数边界、文件长度与布局,只解析较小的诊断和最终值;采样数组保留为 C 已生成的 JSON 字节。返回时标准 JSON 编码器处理模型元数据、名称、单位和诊断,原始数值片段直接放入 `result.series`,不再次转换为 Python 浮点数组或重新编码这些数组。浏览器仍收到原有 JSON 对象和普通数字数组,没有改成二进制接口。
|
||||||
|
|
||||||
|
字节内容在临时工作目录删除前已经独立持有,任务保留及重复 GET 查询也可直接返回。默认 Python 调用、同步 HTTP 仿真与独立 CLI 仍提供普通结果对象;取消、失败和部分结果的状态规则保持原合同。原始片段仅来自受控 C 输出,用户模型文字继续经过标准 JSON 转义,不按字符串搜索猜测边界。
|
||||||
|
|
||||||
|
C 的 JSON 数字格式化成本仍存在;本轮着重省去大数组在 Python 中的解析和再次编码。方程、牛顿/二分算法、积分器、误差限、事件点与采样没有修改。
|
||||||
|
|
||||||
|
### 浏览器打包保存与旧缓存兼容
|
||||||
|
|
||||||
|
原保存方式按每个输出列写独立块,八路一次结果有 1785 个数据写入、14 个顺序事务。新方式把短列连成固定容量的 Float64 数据块:完整八路约 7 块,每块最多 2 MiB,加 1 条头记录,在单次原子事务内提交;只有成功提交且仍是最新保存请求时,才更新小型 sessionStorage 指针。
|
||||||
|
|
||||||
|
复制过程分批让出主线程;换结果会停止旧保存,失败不发布新指针,只清理本页拥有的缓存。新记录带 `packed-f64-v1` 标记,读取通过少量 `getAll` 结果恢复;旧 IndexedDB 按列格式及旧 sessionStorage 格式继续可读。记录缺块、重叠、错误长度或未知布局会明确失败,不返回残缺结果。
|
||||||
|
|
||||||
|
### CSV 改在工作线程中生成
|
||||||
|
|
||||||
|
网页不再把全部曲线 JSON 回传给后台 CSV 接口。主线程按最多 1 MiB 的 Float64 数据批次转移给专用 Worker,Worker 按列序生成 CSV Blob;主线程保持可响应,重复点击、切换结果和组件卸载均有任务清理。保留原下载按钮和文件名规则,旧 HTTP CSV 接口仍可单独使用。
|
||||||
|
|
||||||
|
保留 time 首列、变量元数据顺序、原始单位、全部原始采样、UTF-8 BOM、CRLF 与 CSV 转义。数字使用可精确回读的最短十进制并保留负零;例如整数可能省掉 `.0`,所以 CSV 文本及文件 SHA 会变化,数值必须逐单元精确一致。未降低精度、删列或抽样来缩短用时。
|
||||||
|
|
||||||
|
## 验证条件
|
||||||
|
|
||||||
|
- 输入 SHA256:`670977bef67e62d9c66e8af497bada208bd72a7301be45128d185d47282cf288`;157 元件、178 条连接、132 状态、1784 输出变量。
|
||||||
|
- 八路 0~10 s、输出间隔 0.01 s,1002 个原始采样(含事件点);CVODE BDF / SUNDIALS 7.4.0,`rtol=1e-8`、`max_step=1e30`,状态绝对误差限保持质量1e-14、内能1e-8、速度/位移1e-12(各自SI单位)。
|
||||||
|
- 同一 Linux Xeon Silver 4210R 虚拟机、GCC 13.3.0、现有 Python .venv,正式网页使用 Node 24.18.0 构建和 Chromium 151.0.7922.34;未安装新环境或新增生产依赖。
|
||||||
|
- 旧版后台从 Git `808c484` 导出 app/native/schemas 到隔离目录,前端使用上一轮保存的同提交正式资产;优化版使用本轮源码。两者都是真实服务,不注入结果或模拟网络。
|
||||||
|
- 依次运行旧版无插桩、优化版无插桩、优化版分阶段诊断三组;每组预热一次、正式三次。正式计时期间没有安排并行求解、大编译或大文件比对;输出比较留到计时结束。
|
||||||
|
- 每轮先公开导入、导出并核对相同工程,再计点击运行时间;接着进入结果页、选择温度、CSV 和结果文件下载、刷新恢复。浏览器保存完成采用同一个 session 指针提交观察点,允许保存与查看自然重叠。
|
||||||
|
|
||||||
|
## 同机重新测得的前后用时
|
||||||
|
|
||||||
|
下表均取**无插桩组**预热后三次正式运行的中位数,括号为最小~最大值,单位秒。结果可查看以完成状态 DOM 和运行按钮恢复可用为准,另列两帧后的首次绘制机会以便对应此前报告。保存完成由相同 sessionStorage 指针观察,轮询间隔16 ms并含调度延迟;不是提前显示的保存提示。
|
||||||
|
|
||||||
|
| 指标 | 旧版 / s | 优化后 / s | 中位耗时降低 |
|
||||||
|
|---|---:|---:|---:|
|
||||||
|
| 点击运行 → 结果可查看(完成 DOM) | 9.8720(9.8700~9.9832) | 8.0341(7.9848~8.0377) | 18.62% |
|
||||||
|
| 点击运行 → 完成状态首次绘制机会 | 9.8931(9.8904~10.0018) | 8.0547(8.0017~8.0595) | 18.58% |
|
||||||
|
| 点击运行 → 浏览器保存完成观察 | 11.7740(11.6157~12.6169) | 8.1320(8.0843~8.1468) | 30.93% |
|
||||||
|
| CSV 点击 → 下载保存 | 5.6103(5.4994~5.7601) | 1.0051(0.9394~1.0077) | 82.08% |
|
||||||
|
| 刷新 → 结果首次绘制机会 | 0.6095(0.5541~0.6375) | 0.3768(0.3525~0.3980) | 38.18% |
|
||||||
|
|
||||||
|
三个目标均有改善:结果可查看节省约 **1.84 s**,结果保存完成节省约 **3.64 s**,CSV 下载保存节省约 **4.61 s**。这些区间重叠,不能相加为一次仿真的节省量。与上轮历史数据略有差异时,以本轮重新运行的旧版对照为计算比例的依据。
|
||||||
|
|
||||||
|
| 其他观察 | 旧版中位 / s | 优化后中位 / s |
|
||||||
|
|---|---:|---:|
|
||||||
|
| 工程导入到首次绘制机会 | 0.2325 | 0.2445 |
|
||||||
|
| 进入结果页到首次绘制机会 | 0.1825 | 0.1874 |
|
||||||
|
| 选择温度到曲线首次绘制机会 | 0.0286 | 0.0288 |
|
||||||
|
| 结果文件下载保存 | 1.3575 | 1.0118 |
|
||||||
|
| 刷新到结果DOM | 0.4669 | 0.2496 |
|
||||||
|
| 结果就绪DOM到保存指针观察 | 1.9040 | 0.0995 |
|
||||||
|
|
||||||
|
两组 C 纯求解中位数分别为 **6.1083 / 6.1013 s**,进程全程分别为 **8.0204 / 7.4582 s**。每次仍是 RHS 74265、接受6974、拒绝454;本轮收益来自结果处理与保存,而不是改变求解精度或少算输出。结果文件下载不是本轮主要改动,有限样本的时间变化不单独宣称为该功能优化收益。
|
||||||
|
|
||||||
|
下载完成计时包含 Playwright 通知和 saveAs 的文件系统成本;首个 DOM 后的两帧只是绘制机会,未直接测 GPU。正式页面均经本机回环 HTTP 访问,没有模拟远端网络。分阶段组相比优化版无插桩组有几%波动,其纯求解也从约6.10 s变到约5.94 s;CSV下载完成还有自动化/磁盘调度波动,故加速比例严格来自上面的两组无插桩对照。
|
||||||
|
|
||||||
|

|
||||||
|
|
||||||
|
可缩放版本:[SVG](assets/2026-09-11/postprocess-20260911-before-after.svg)。图的“result visible”为完成状态DOM,与主表第一行相同。
|
||||||
|
|
||||||
|
## 优化后的后台阶段
|
||||||
|
|
||||||
|
分阶段服务使用隔离的 C 主程序时钟和真实函数包装;数值内核逐字节相同。按每个 X-Simulation-Id 关联网页与后台,以下为正式三次的秒数(中位、范围)。
|
||||||
|
|
||||||
|
| 阶段 | 中位数(最小~最大)/ s |
|
||||||
|
|---|---:|
|
||||||
|
| XML 校验 | 0.0227(0.0219~0.0232) |
|
||||||
|
| 网络编译 | 0.0393(0.0372~0.0425) |
|
||||||
|
| C 代码生成 | 0.0557(0.0539~0.0934) |
|
||||||
|
| 构建缓存核验(均命中) | 0.0339(0.0327~0.0340) |
|
||||||
|
| C 纯积分 | 5.9407(5.9370~6.0256) |
|
||||||
|
| C 输出投影 | 0.0799(0.0774~0.0813) |
|
||||||
|
| C 原始 JSON 与索引写出 | 1.1643(1.1578~1.1762) |
|
||||||
|
| Python 原始结果文件读字节 | 0.0250(0.0107~0.0269) |
|
||||||
|
| Python 解析小索引与元数据 | 0.0012(0.0011~0.0015) |
|
||||||
|
| 索引读取与片段整理全程(含上两项) | 0.0472(0.0350~0.0486) |
|
||||||
|
| 响应接口元数据组装 | 0.0172(0.0168~0.0176) |
|
||||||
|
| 响应小元数据 JSON 序列化 | 0.0329(0.0328~0.0329) |
|
||||||
|
| 序列化后至 ASGI 最后响应完成 | 0.0747(0.0528~0.1222) |
|
||||||
|
| 后台 HTTP 全程(父区间) | 7.5315(7.5174~7.6975) |
|
||||||
|
|
||||||
|
旧版阶段诊断中,Python 完整结果 JSON 解析约0.551 s、HTTP结果序列化约1.193 s;新路径只解析小元数据约0.0012 s,序列化约0.0329 s。旧阶段分项来自上一轮诊断报告,不用于替代本轮端到端的同机重新比较。C格式化/写出仍约1.16 s,后续还有优化空间。
|
||||||
|
|
||||||
|
原生片段整理全程约0.0472 s,包含读文件、小JSON解析、校验和series字节持有;不能与其子项重复相加。C main全程约7.1957 s,Python原生执行包装约7.2336 s,worker全程约7.4409 s,也都是包含子阶段的父区间。诊断产物保留另约0.00055 s;服务器send、后台生产和浏览器等待可能重叠。
|
||||||
|
|
||||||
|
## 浏览器保存与 CSV 的实际细分
|
||||||
|
|
||||||
|
以下来自优化后的分阶段组,所有时间为观察区间,含相应异步等待与调度。
|
||||||
|
|
||||||
|
| 观察区间 | 中位数 / s |
|
||||||
|
|---|---:|
|
||||||
|
| 点击运行至调用fetch | 0.0187 |
|
||||||
|
| 最终结果 JSON.parse | 0.0982 |
|
||||||
|
| UTF-8 解码总计 | 0.0297 |
|
||||||
|
| 结果解析结束至保存提交指针发布 | 0.1186 |
|
||||||
|
| IndexedDB 单次事务窗口 | 0.0549 |
|
||||||
|
| 完成状态DOM至保存完成观察 | 0.0829 |
|
||||||
|
| CSV工作线程start发送至finish发送 | 0.0681 |
|
||||||
|
| CSV finish发送至完成消息接收 | 0.3889 |
|
||||||
|
| CSV点击至Blob下载锚点 | 0.4893 |
|
||||||
|
| 结果文件点击至Blob下载锚点 | 0.5905 |
|
||||||
|
|
||||||
|
实际八路三次保存均为 **1 个事务、约55 ms**;事务前还需要打包和调度,所以不能把55 ms当成从计算结束到保存完成的全部时间。浏览器仍在同一提交完成后发布恢复指针,没有通过放宽持久化完成标准来提速。
|
||||||
|
|
||||||
|
CSV三次均未发HTTP请求。按批传输约68 ms;finish发出到完成消息收到约389 ms,含剩余工作线程计算、启动/排队和消息传递,不冒称纯Worker CPU时间。点击至Blob准备好约489 ms,实际下载保存的主结论仍使用无插桩组约1.005 s。
|
||||||
|
|
||||||
|
新CSV为31,820,845字节,旧版32,346,795字节,减少约1.63%;主要收益是省去全量JSON往返和后台逐单元格式化,并将生成放在工作线程,不能仅归因于文件体积变小。CSV编码采用Float64精确回读方式,数值一致性单独验证。
|
||||||
|
|
||||||
|
|
||||||
|
## 正确性与回归
|
||||||
|
|
||||||
|
29 项后端相关回归通过,覆盖原生执行、默认同步接口、真实 ASGI 流式响应与重复任务 GET、用户/异常断线取消后的部分结果、心跳、索引损坏与截断文件、JSON 边界转义、独立 C 程序和旧 CSV HTTP 合同。后续补全严格索引版本及 I/O 异常处理后,6 项传输专项再次通过。
|
||||||
|
|
||||||
|
6 项真实 IndexedDB 专项通过,覆盖完整八路形状的值、负零、空列、旧缓存、缺块/重叠、头记录写入失败、保存竞态和跨页面隔离。6 项 CSV 专项通过,包括 1052929 个值的精确回读、最小子正规数、极大数、负零、列顺序和转义、无 CSV HTTP 请求、重复点击与错误重试。专项合成数据耗时只用于功能诊断,不充当正式八路测量。正式 TypeScript/Vite 构建通过。
|
||||||
|
|
||||||
|
本轮发现已有 `tests/test_native_codegen.py` 引用的 `tests/fixtures/native-skill-test.xml` 在当前提交缺失,从 `5d5a2e1:tests/data/native-skill-test.xml` 原样恢复到其现行测试路径,保证相关回归可运行。没有放回浏览器输入目录或改变这份测试资料的物理内容。
|
||||||
|
|
||||||
|
三组共 **12 次**真实八路运行全部完成10 s,无页面错误。每次公开导出工程都与固定输入的全部参数、连接与设置一致;全部 `series`、`final` 与优化前原生基准精确相等。12份CSV合计 **21,462,840 个数值单元**与基准精确相等,各组内部CSV字节稳定;旧新文本差异符合前述编码规则。12次刷新前后的完整结果一致。
|
||||||
|
|
||||||
|
另外,4份分阶段组C原始结果文件的 `series/final/finalState` 均与旧原生基准逐值相同,证明后处理改动没有改变积分轨迹或事件点。这里只比较数值,不要求运行耗时等诊断字段在不同运行间相同。证据:[全量网页/CSV核对](../../test/postprocess-20260911/equality.json)、[C原始状态核对](../../test/postprocess-20260911/native-parity.json)。
|
||||||
|
|
||||||
|
## Amesim 范围
|
||||||
|
|
||||||
|
本轮处理优化不改变计算结果,沿用 [当前八路 AME 归档核查](test-mql-8当前AME归档与完整曲线核查-2026-09-11.md) 的差异记录。完整数值逐值不变后,原有早期压力/温度差及约0.9834 s碰撞力尖峰也会保留;不能把后处理提速解释为八路物理曲线验收通过。
|
||||||
|
|
||||||
|
当前没有可信 Amesim CPU/墙钟记录,也没有可调用的 Amesim 安装,速度对比仍跳过。后续继续使用修正八路,在相同输入、精度与完整采样下核查正确性与用时。
|
||||||
|
|
||||||
|
## 文件与复现
|
||||||
|
|
||||||
|
代码入口为 [原生片段传输](../../app/simulation/native_codegen/transport.py)、[C 结果索引](../../native/runtime/main.c)、[浏览器保存](../../frontend/src/resultPersistence.ts)、[CSV 导出任务](../../frontend/src/resultCsvExport.ts) 和 [CSV Worker](../../frontend/src/resultCsv.worker.ts)。
|
||||||
|
|
||||||
|
输入快照、隔离旧版、构建、运行结果和截图位于 `test/postprocess-20260911/`;环境仍在 `.venv/native/`。这些大型运行产物和环境均在 Git 忽略范围,本轮未提交或推送 Git。
|
||||||
|
|
||||||
|
- [三组逐次与汇总JSON](../../test/postprocess-20260911/summary.json)、[各阶段CSV](../../test/postprocess-20260911/timings.csv)
|
||||||
|
- [旧版网页记录](../../test/postprocess-20260911/browser-baseline/summary.json)、[优化版网页记录](../../test/postprocess-20260911/browser-optimized/summary.json)、[优化版分阶段记录](../../test/postprocess-20260911/browser-profiled/summary.json);每次截图、输入、结果与刷新文件保存在对应组子目录,后台逐请求阶段位于 `backend-profiled/requests/<simulationId>/stages.json`。
|
||||||
|
- [源码与输入清单](../../test/postprocess-20260911/source-manifest.json)
|
||||||
|
- [后端回归](../../test/postprocess-20260911/backend-tests.log)、[传输专项复测](../../test/postprocess-20260911/backend-transport-final.log)、[正式前端构建](../../test/postprocess-20260911/frontend-build.log)
|
||||||
|
- 计时脚本 [backend_stage_profile.py](../../tests/manual/backend_stage_profile.py)、[browser_stage_profile.mjs](../../tests/manual/browser_stage_profile.mjs);数值复核 [compare_browser_stage_outputs.py](../../tests/manual/compare_browser_stage_outputs.py)。
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# 后端相关回归
|
||||||
|
.venv/bin/python -m unittest tests.test_native_result_transport tests.test_native_codegen \
|
||||||
|
tests.test_generic_system_xml_simulation tests.test_result_csv_export tests.test_native_only_backend -v
|
||||||
|
# 当前正式网页的独立服务与计时:两个终端分别运行,输出目录必须为新目录
|
||||||
|
.venv/bin/python tests/manual/backend_stage_profile.py --plain \
|
||||||
|
--output-dir test/new-results-service --port 8021
|
||||||
|
LD_LIBRARY_PATH="$PWD/.venv/native/browser-libs/usr/lib/x86_64-linux-gnu${LD_LIBRARY_PATH:+:$LD_LIBRARY_PATH}" \
|
||||||
|
.tools/node-v24.18.0-linux-x64/bin/node tests/manual/browser_stage_profile.mjs \
|
||||||
|
--url http://127.0.0.1:8021 --mode control --runs 3 --output test/new-results-browser
|
||||||
|
```
|
||||||
@@ -0,0 +1,210 @@
|
|||||||
|
# 八路网页求解全流程成本评估(2026-09-11)
|
||||||
|
|
||||||
|
本轮评估上一轮结果处理优化后的当前版本,目标是分清前端预处理、后端准备与计算、结果输出、浏览器后处理的成本。使用 [test-mql-8-corrected.json](../../tests/data/test-mql-8-corrected.json),没有修改生产算法、物理输入、积分精度或输出点数。此前实现与前后加速见 [结果处理优化报告](八路结果处理与网页保存优化-2026-09-11.md)。
|
||||||
|
|
||||||
|
**当前主要成本集中在后端数值积分和 C 结果写出。** 无插桩正式三次中位数:点击运行到结果可查看 **7.99 s**,到浏览器保存完成 **8.11 s**。分阶段组内逐次计算,积分占点击到可查看时间约 **76.4%**,C结果编码与写出占约 **14.8%**。前端点击后的预处理仅约 **19.6 ms**,不属于当前优先优化的大项。
|
||||||
|
|
||||||
|
## 测量条件与可复现证据
|
||||||
|
|
||||||
|
- 模型 SHA256 为 `670977bef67e62d9c66e8af497bada208bd72a7301be45128d185d47282cf288`;157 元件、178 条连接、132 状态,1784 输出变量和时间列,1002 个原始采样。
|
||||||
|
- 0~10 s,输出间隔0.01 s,CVODE BDF,`rtol=1e-8`、`max_step=1e30`,状态绝对误差下限保持原值;事件和采样策略均不变。
|
||||||
|
- Git 基线 `808c484` 加上一轮尚未提交的结果处理优化。所有上一轮生产源码清单中的文件哈希一致;本轮只增加/完善手动测量工具与报告。环境继续使用已有 `.venv`、`.venv/native`、Node24.18.0、Chromium151.0.7922.34;没有安装新依赖。
|
||||||
|
- 使用真实生产页面,经本机回环 HTTP 提交。每轮公开导入、导出并核对八路工程参数、连接和仿真配置,再点击运行,进入结果页显示温度曲线,保存 CSV 和结果文件,刷新恢复。
|
||||||
|
- 无插桩组、分阶段组各预热一次、正式三次;串行运行,计时期间不安排大型构建、其他仿真或全量文件比对。端到端体验以无插桩组为准。分阶段组用于定位成本,不把其耗时套入无插桩组拆账。
|
||||||
|
- 分阶段后台只在隔离的 C `main.c` 副本增加少量墙钟/CPU时钟,数值实现未改;Python 包装原函数,观察子进程创建、退出及实际读取/响应边界,不额外轮询。C写出 CPU 时钟包括用户态和内核态,不能单独区分格式化、内存访问与系统调用。
|
||||||
|
- 分阶段网页包装原有 fetch、reader、decoder、JSON.parse、IndexedDB 和 Worker 调用,保持每份响应仅读取、解码和解析一次。另有独立 CDP CPU 采样诊断,不混入正式端到端结果。
|
||||||
|
|
||||||
|
“结果可查看”指页面观察到成功完成且运行按钮恢复可用;“绘制机会”指再经过两次 requestAnimationFrame,不等于直接测得 GPU 绘制时间。“浏览器保存完成”指实际 IndexedDB 事务提交后发布恢复指针;无插桩组通过16 ms轮询观察,含调度延迟。“下载保存完成”包括 Playwright 通知及 `saveAs` 的文件系统成本。
|
||||||
|
|
||||||
|
## 网页端到端体验
|
||||||
|
|
||||||
|
下表为无插桩组预热后三次中位数及范围,单位秒。导入工程发生在点击运行之前,不计入仿真总耗时。
|
||||||
|
|
||||||
|
| 用户过程 | 中位数(最小~最大)/ s |
|
||||||
|
|---|---:|
|
||||||
|
| 点击运行 → 结果可查看 | 7.992(7.927~8.104) |
|
||||||
|
| 点击运行 → 完成状态绘制机会 | 8.015(7.944~8.122) |
|
||||||
|
| 点击运行 → 浏览器保存完成观察 | 8.108(8.043~8.218) |
|
||||||
|
| CSV点击 → 下载保存完成 | 0.912(0.884~0.984) |
|
||||||
|
| 结果页签点击 → 绘制机会 | 0.193(0.181~0.240) |
|
||||||
|
| 温度变量点击 → 曲线绘制机会 | 0.029(0.028~0.029) |
|
||||||
|
| 刷新 → 结果绘制机会 | 0.437(0.427~0.452) |
|
||||||
|
| 结果文件点击 → 下载保存完成 | 1.154(1.033~1.223) |
|
||||||
|
|
||||||
|
工程导入到绘制机会为 0.238(0.231~0.247) s。三次完整运行的 C 纯求解中位数为 6.143 s;每次均完成10 s物理区间。
|
||||||
|
|
||||||
|
首次打开无插桩页面至自动化确认导入控件已附加约0.530 s,属于一次独立导航观察,含自动化确认延迟;不等同首帧或完整应用就绪时间,未计入上面的点击运行用时。
|
||||||
|
|
||||||
|
## 前端预处理、接收与后处理
|
||||||
|
|
||||||
|
分阶段组三次正式运行;所有值为毫秒。各行可能重叠。
|
||||||
|
|
||||||
|
| 过程 | 中位数(最小~最大)/ ms | 口径 |
|
||||||
|
|---|---:|---|
|
||||||
|
| 点击→提交fetch:校验、快照、XML和请求准备 | 19.6(18.8~31.9) | 毫秒;同步预处理与提交设置的整体窗口 |
|
||||||
|
| fetch→响应头可供读取 | 29.9(28.7~30.0) | 毫秒;包括请求处理、传输和浏览器调度 |
|
||||||
|
| 流UTF-8解码同步调用之和 | 32.2(30.8~33.0) | 毫秒;与接收过程重叠 |
|
||||||
|
| 最终结果JSON.parse | 107.0(105.5~119.6) | 毫秒;原始单次同步解析 |
|
||||||
|
| 解析结束→完成DOM | 48.2(42.4~58.2) | 毫秒;发布状态、渲染准备与调度 |
|
||||||
|
| 进入结果页→绘制机会 | 182.0(181.1~196.7) | 毫秒;包含系统图准备 |
|
||||||
|
| 温度曲线选择→绘制机会 | 35.7(34.4~36.3) | 毫秒;单条温度曲线 |
|
||||||
|
| 最后分块交付→开始解析 | 36.6(33.3~45.3) | 含最后一次解码、扫描、拼接、trim及调度,不能全归于join |
|
||||||
|
| 解析结束→缓存指针发布 | 135.5(127.0~176.5) | 包含数据打包、让出主线程、数据库打开/提交、页面工作 |
|
||||||
|
| 保存事务窗口(每次1次) | 63.1(50.0~84.5) | 包含同步提交及异步等待,是上一行子区间 |
|
||||||
|
|
||||||
|
从最后一次分块交付到完成DOM为 197.0(191.8~207.3) ms;这比整段 headers→EOF 更接近最终结果到达后的页面处理窗口。后者约 7.866 s,绝大部分与后台执行同时发生,不能称为前端解析用时。
|
||||||
|
|
||||||
|
CSV 点击到 Blob 下载触发为 494.9(487.4~495.9) ms;Worker start→finish投递为 77.3(69.1~86.2) ms,finish投递→完成消息到达为 382.9(377.8~386.7) ms。前一段主线程分批准备与Worker执行重叠,后一段包含Worker编码/Blob生成/消息投递和调度。没有单独测量Worker线程CPU。CSV完整下载体验仍以无插桩组为准。
|
||||||
|
|
||||||
|
### 主线程 CPU 采样补充
|
||||||
|
|
||||||
|
另对真实页面做1 ms间隔的CDP主线程采样,预热一次、正式三次。使用临时生成的hidden source map映射回源码;临时构建的JS与实际服务的生产JS逐字节一致,未替换页面资产。采样组点击→可查看中位8.152 s,独立于上面的无插桩/分阶段组。下表为采样归属估计,精度受约1 ms采样与时钟校准误差限制,不能当作逐函数秒表。
|
||||||
|
|
||||||
|
| 观察窗口 | 可识别的主要工作,采样归属中位 / ms |
|
||||||
|
|---|---|
|
||||||
|
| 点击→fetch(该诊断组27.1 ms) | 模型校验15.0;XML构造7.8;XML原生序列化2.3;项目快照1.2;端点/合同处理1.1 |
|
||||||
|
| 响应头→大结果开始解析 | NDJSON扫描/拼接/分发40.2;UTF-8解码29.2;React更新69.3;其余含进度显示和测量工作 |
|
||||||
|
| 大结果解析开始→保存指针(255.3 ms) | JSON解析归属103.5;IndexedDB请求提交62.2;打包/保存逻辑11.3;React7.9;GC12.7 |
|
||||||
|
| 进入结果页→绘制机会(207.6 ms) | React运行时77.7;ReactFlow系统图52.6;结果视图数据准备9.3;系统图几何处理5.9 |
|
||||||
|
| 选择温度→曲线绘制机会(28.4 ms) | 结果视图8.0;React7.1;曲线数据准备2.6 |
|
||||||
|
| CSV点击→文件保存 | 主线程分批准备和传输36.6;其余可识别主线程工作包括视图与React;不含Worker线程CPU |
|
||||||
|
|
||||||
|
IndexedDB“请求提交”类别包含相应JS调用下的原生同步处理,不代表后台磁盘线程耗时。CPU样本中的JSON解析归属与同步wrapper实测是两种观察,不能相加;其余分类和不同窗口也不能直接相加成总耗时。
|
||||||
|
|
||||||
|
响应头到最终结果解析前约7.947 s的窗口,采样明确空闲约5.592 s,V8 `(program)` 未归属约2.086 s,可识别JS及其下原生调用约0.265 s。其中仍有约0.127 s未归到具体业务函数,含DOM观测/自动化;**不能把2.086 s未知时间或整段7.947 s计作前端业务CPU**。GC、idle、program、unknown-runtime单列,函数active统计排除这些样本。
|
||||||
|
|
||||||
|
结果页准备比单条温度曲线的数据准备更显著;采样主要落在React/ReactFlow与DOM操作,不足以将其直接称为GPU绘制耗时。刷新恢复另有独立profile,使用刷新后的document时间轴。
|
||||||
|
|
||||||
|
采用离线修订的v2分类,不使用会混淆系统样本的初始汇总字段。证据:[cpu-resummary-v2.json](../../test/web-cost-20260911/browser-deep/cpu-resummary-v2.json)、[frontend-cpu-statistics.json](../../test/web-cost-20260911/frontend-cpu-statistics.json)、[sourcemap-verification.json](../../test/web-cost-20260911/sourcemap-verification.json)。原始 `.cpuprofile`、`trace.json` 与网页计时结果均保持原样。
|
||||||
|
|
||||||
|
## 后端准备、求解、输出与响应
|
||||||
|
|
||||||
|
分阶段组三次正式运行;单位秒。包含“其中”的行是父区间子项,不应重复累计。
|
||||||
|
|
||||||
|
| 阶段 | 中位数(最小~最大)/ s |
|
||||||
|
|---|---:|
|
||||||
|
| XML校验 | 0.0225(0.0220~0.0236) |
|
||||||
|
| 网络编译:组件和连接 | 0.0372(0.0365~0.0454) |
|
||||||
|
| C模型代码生成 | 0.0601(0.0540~0.0863) |
|
||||||
|
| 构建缓存校验(命中) | 0.0349(0.0341~0.0417) |
|
||||||
|
| 子进程创建Popen调用 | 0.0006(0.0006~0.0007) |
|
||||||
|
| C初始状态和首样本准备 | 0.0001(0.0001~0.0001) |
|
||||||
|
| C数值积分,含求解器内部工作 | 6.0645(6.0614~6.1547) |
|
||||||
|
| 输出投影:由状态计算全部输出 | 0.0852(0.0852~0.0861) |
|
||||||
|
| C结果JSON及索引格式化/写出 | 1.1752(1.1620~1.1862) |
|
||||||
|
| Python结果索引读取、校验和片段整理 | 0.0440(0.0334~0.0466) |
|
||||||
|
| 其中:原始结果文件读字节 | 0.0230(0.0112~0.0254) |
|
||||||
|
| 其中:小索引和元数据JSON解析 | 0.0012(0.0011~0.0015) |
|
||||||
|
| 响应元数据组装 | 0.0164(0.0164~0.0175) |
|
||||||
|
| 响应包装:元数据编码与字节片段组织 | 0.0354(0.0340~0.0357) |
|
||||||
|
| ASGI send累计await | 0.0794(0.0396~0.0845) |
|
||||||
|
| 后台HTTP总窗口(父区间) | 7.6916(7.6678~7.7199) |
|
||||||
|
|
||||||
|
XML校验、网络编译、C代码生成、缓存检查的同次合计为 0.1630(0.1565~0.1789) s。C初始状态准备只包含 `model_init` 和首样本,CVODE对象创建、初始化和重启仍属于积分区间。
|
||||||
|
|
||||||
|
C输出投影CPU为 0.0852(0.0851~0.0860) s;JSON写出CPU为 **1.1749(1.1619~1.1859) s**,与墙钟 1.1752 s 几乎一致。这支持优先调查数值格式化、逐值stdio调用和内存访问;不能将整段1.18 s称为磁盘等待,也不能把CPU/墙钟差直接当成测得的I/O耗时。
|
||||||
|
|
||||||
|
响应数值原始片段约32.64 MB,整条HTTP响应约34.48 MB,CSV约31.82 MB(十进制MB)。当前后端已避免Python大数组解析/重编码,但C逐值 `fprintf("%s%.17g", …)` 的成本仍保留。
|
||||||
|
|
||||||
|
`processWallSeconds` 包括退出后的Python结果读取;本轮另记录子进程从创建到已有poll/wait首次观察到退出的寿命,不能混称C求解时间。ASGI send窗口也不是纯网络耗时;本机回环测试不代表远端网络。
|
||||||
|
|
||||||
|
**首次编译单独记录:** 分阶段组首轮缓存未命中,构建耗时 **3.446 s**,点击到可查看 **11.447 s**。这是一次冷构建观察,不纳入三次缓存命中的正式中位数,也不当成稳定冷启动统计。修改会影响生成代码的模型内容后,可能重新发生该成本。
|
||||||
|
|
||||||
|
### 积分内部的计算分布
|
||||||
|
|
||||||
|
另做独立原生诊断:复用真正生产缓存可执行文件作control,隔离副本只在调用边界增加嵌套时钟并读取CVODE计数。每个变体预热一次、正式三次,交替串行运行;不是在网页中再开第二个求解任务。
|
||||||
|
|
||||||
|
表中为**排他墙钟时间**,同一次运行中已扣除子调用;每次所有排他区间之和精确等于积分父区间。中位数列来自三次运行,不再相加假装某次总耗时。
|
||||||
|
|
||||||
|
| 积分内部工作 | 排他时间中位 / s | 同次积分占比的中位数 |
|
||||||
|
|---|---:|---:|
|
||||||
|
| 模型RHS:物性、管流局部求根及组件方程等全部模型求值 | 5.559250 | 93.7191% |
|
||||||
|
| Dense矩阵分解 | 0.180638 | 3.0452% |
|
||||||
|
| Dense线性方程回代求解 | 0.110131 | 1.8620% |
|
||||||
|
| CVODE剩余内部工作(已扣RHS、poll、Dense) | 0.069325 | 1.1660% |
|
||||||
|
| 超时/取消/进度轮询 | 0.007586 | 0.1279% |
|
||||||
|
| 积分外层准备、循环和清理余量 | 0.001608 | 0.0272% |
|
||||||
|
| 接受步与事件处理自身(已扣采样和插值) | 0.001412 | 0.0237% |
|
||||||
|
| 保存采样状态 | 0.000806 | 0.0136% |
|
||||||
|
| 稠密输出插值 | 0.000293 | 0.0049% |
|
||||||
|
|
||||||
|
诊断积分中位 **5.931924 s**,配对生产control为 **5.874992 s**;逐对计算的插桩增幅中位约 **0.91%**(0.68%~1.71%)。原始summary的 `instrumentationOverheadFraction` 采用两组中位数之比,估计为0.97%;两者计算口径不同。此处为独立进程诊断,其绝对耗时不直接替代网页组的6.06 s;用作积分内部占比定位。时钟/统计维护成本仍在诊断结果中。
|
||||||
|
|
||||||
|
**关键发现是雅可比所需的方程求值次数。** 每次实际读取CVODE统计均为:
|
||||||
|
|
||||||
|
| 计数 | 实测值 |
|
||||||
|
|---|---:|
|
||||||
|
| 常规RHS求值 | 11,565 |
|
||||||
|
| 线性求解器有限差分RHS求值 | 62,700 |
|
||||||
|
| 合计模型RHS | 74,265 |
|
||||||
|
| 雅可比计算次数 × 状态数 | 475 × 132 = 62,700 |
|
||||||
|
| 非线性迭代 / 非线性收敛失败 | 11,557 / 414 |
|
||||||
|
| Dense分解 / 线性回代调用 | 1,656 / 11,557 |
|
||||||
|
| 分段计数 / 求解器启动 | 4 / 4 |
|
||||||
|
|
||||||
|
有限差分占全部RHS**调用次数的84.43%**。结合全部RHS耗时占积分的93.72%,应优先研究如何减少重复模型求值。没有单独计时“差分RHS子集”,不能据此声称差分恰好占84.43%的积分时间,更不能承诺减少同等比例的总耗时。Dense代数自身合计约4.91%,单独更换矩阵分解实现的潜在收益较受限;雅可比结构改变带来的RHS次数减少属于另一项收益。
|
||||||
|
|
||||||
|
414是CVODE本级非线性收敛失败计数,后续通过重试完成仿真,不是管路局部Newton达到上限的次数。本轮未继续拆分RHS中的物性、管路求根和组件方程,不能将5.56 s全部归给管路求根。
|
||||||
|
|
||||||
|
所有8次原生运行的完整series、final、finalState和求解计数(只排除两个solve计时字段)与生产control精确一致。工具和原始证据:[native_compute_profile.py](../../tests/manual/native_compute_profile.py)、[native-compute-profile/summary.json](../../test/web-cost-20260911/native-compute-profile/summary.json)。
|
||||||
|
|
||||||
|
## 下一步优化顺序
|
||||||
|
|
||||||
|
1. **计算核心:验证减少雅可比差分所需RHS求值。** 研究生成模型的状态依赖结构、可复用的导数和稀疏差分/着色方案;保持当前精度、事件与物理参数,单独验证全曲线和守恒。这是本轮看到的最大计算成本来源,但尚未实现或证明某种替代方案的收益。
|
||||||
|
2. **结果处理:优先优化C数值编码与写出。** 当前仍约1.18 s,明显大于Python读回和网页JSON解析。可比较更高效且精确回读的数值编码、批量输出,或直接数值缓冲传输;需保留原始采样和数值精度。本轮没有改变传输合同。
|
||||||
|
3. **前端后处理:针对结果解析和缓存提交做小幅改进。** JSON.parse约0.11 s,数据打包/IDB提交约0.1 s量级;若优先改善主线程响应,可研究转移解析、减少数组复制。预期端到端空间小于上面两项,收益需实测。
|
||||||
|
4. **模型编辑后的首次运行:单独调查构建缓存。** 本轮冷构建一次约3.45 s;若用户经常改参数/拓扑导致缓存失效,应单独分析编译单元复用和参数是否必须进入生成代码,不能只看缓存命中数据。
|
||||||
|
5. **前端预处理和小项暂缓。** 点击后的整体准备约20 ms;Python元数据解析约1 ms、进程启动不足1 ms、事件采样/轮询均很小,当前不适合优先投入。
|
||||||
|
|
||||||
|

|
||||||
|
|
||||||
|
可缩放图:[cost-breakdown.svg](assets/2026-09-11/web-cost-20260911-cost-breakdown.svg)。三个面板分别使用后端网页组、独立原生诊断组和前端网页组;前端各行存在重叠,不能相加。
|
||||||
|
|
||||||
|
## 数据完整性与边界
|
||||||
|
|
||||||
|
三组共 **12 次网页运行**全部成功,无页面错误;完整曲线、最终输出及 **21,462,840 个CSV数值单元**均与既有C原生结果逐值精确相等,刷新恢复一致。每次 `nfev=74265`、接受步6974、错误测试失败454、Jacobian计数475、线性求解准备1656、状态跳变1、求解器启动4,均与原模型一致。数值比较不含本来就变化的计时诊断。证据见 [equality.json](../../test/web-cost-20260911/equality.json)。CSV文本SHA本轮各组也相同。
|
||||||
|
|
||||||
|
本轮没有可调用的 Amesim 运行环境及同工况可靠墙钟/CPU耗时,因此不作 Amesim 速度比较;此前已记录的 Amesim 曲线差异结论保持原状。本报告是当前网页路径的性能评估,不新增曲线一致性验收结论。
|
||||||
|
|
||||||
|
前端等待与后端执行同时发生,C积分内又包含RHS、雅可比和线性求解;后台响应发送与前端接收/解码也可重叠。表中父子区间及不同线程区间不能直接相加。独立阶段中位数的和也未必等于总耗时中位数。占比需在同一次运行、同一父区间内先计算,再汇总。
|
||||||
|
|
||||||
|
## 文件位置
|
||||||
|
|
||||||
|
- 本报告:[八路网页求解全流程成本评估-2026-09-11.md](八路网页求解全流程成本评估-2026-09-11.md)。
|
||||||
|
- 全部原始产物:[test/web-cost-20260911/](../../test/web-cost-20260911/),此目录被 Git 忽略。
|
||||||
|
- 无插桩网页记录:[browser-control/summary.json](../../test/web-cost-20260911/browser-control/summary.json)。
|
||||||
|
- 分阶段网页记录:[browser-profiled/summary.json](../../test/web-cost-20260911/browser-profiled/summary.json);各轮目录有 `trace.json`、`restore-trace.json`、结果、CSV与截图。
|
||||||
|
- 后端记录:[backend-profiled/requests/](../../test/web-cost-20260911/backend-profiled/requests/),以网页记录的 `simulationId` 关联;每项有真实输入XML、C原始结果、worker日志和 `stages.json`。
|
||||||
|
- 汇总:[summary.json](../../test/web-cost-20260911/summary.json)、[timings.csv](../../test/web-cost-20260911/timings.csv);源码与环境:[source-manifest.json](../../test/web-cost-20260911/source-manifest.json)。
|
||||||
|
- 手动工具:[backend_stage_profile.py](../../tests/manual/backend_stage_profile.py)、[browser_stage_profile.mjs](../../tests/manual/browser_stage_profile.mjs)、[summarize_web_cost.py](../../tests/manual/summarize_web_cost.py)。
|
||||||
|
|
||||||
|
以下命令使用新的输出目录复现;已有输出目录保留原始记录,不覆盖。后台服务需各自保持运行,网页各组与原生诊断串行执行。
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# 无插桩服务 / 后台分阶段服务(分别运行)
|
||||||
|
.venv/bin/python tests/manual/backend_stage_profile.py --plain --port 8024 --output-dir test/web-cost-repeat/backend-control
|
||||||
|
.venv/bin/python tests/manual/backend_stage_profile.py --port 8023 --output-dir test/web-cost-repeat/backend-profiled
|
||||||
|
|
||||||
|
# 网页组:先control,然后profiled;各预热1次、正式3次
|
||||||
|
export LD_LIBRARY_PATH="$PWD/.venv/native/browser-libs/usr/lib/x86_64-linux-gnu${LD_LIBRARY_PATH:+:$LD_LIBRARY_PATH}"
|
||||||
|
.tools/node-v24.18.0-linux-x64/bin/node tests/manual/browser_stage_profile.mjs --url http://127.0.0.1:8024 --output test/web-cost-repeat/browser-control --mode control --runs 3
|
||||||
|
.tools/node-v24.18.0-linux-x64/bin/node tests/manual/browser_stage_profile.mjs --url http://127.0.0.1:8023 --output test/web-cost-repeat/browser-profiled --mode profiled --runs 3
|
||||||
|
.venv/bin/python tests/manual/summarize_web_cost.py --root test/web-cost-repeat
|
||||||
|
```
|
||||||
|
|
||||||
|
深度采样增加 `--deep --cpu-interval-us 1000 --source-map-dir PATH`,使用独立输出目录;source map的生成JS须与生产资产哈希相同。只重新分类现有数据可执行:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
.tools/node-v24.18.0-linux-x64/bin/node tests/manual/browser_stage_profile.mjs --summarize-cpu-only test/web-cost-20260911/browser-deep --source-map-dir test/web-cost-20260911/frontend-sourcemaps/assets
|
||||||
|
```
|
||||||
|
|
||||||
|
原生内部诊断使用生产模型缓存和已记录的请求参数;本轮命令:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
.venv/bin/python tests/manual/native_compute_profile.py \
|
||||||
|
--cache-dir app/data/native-builds/00a1cc841fb4496044c99e13066bbd18784aef134893c3c30b5b984c7aef8eec \
|
||||||
|
--request-stages test/web-cost-20260911/backend-profiled/requests/e18c2e0e-b2e7-4219-9f5a-c09a26cde4fa/stages.json \
|
||||||
|
--output-dir test/native-cost-repeat --run --warmups 1 --repeats 3
|
||||||
|
```
|
||||||
|
|
||||||
|
原生工具目前针对Linux/GCC;所有新产物均位于已忽略的 `test/` 中。手动工具Python语法检查、Node语法检查、离线分类自检和汇总一致性检查通过;没有为本轮只读评估重新改动生产组件或降低精度。
|
||||||
@@ -0,0 +1,31 @@
|
|||||||
|
# 优化验证的模型、数据和计时口径
|
||||||
|
|
||||||
|
本约定记录用户于 2026-09-11 明确的长期偏好,适用于后续仿真正确性检查与性能优化。
|
||||||
|
|
||||||
|
## 默认模型与退路
|
||||||
|
|
||||||
|
优先使用 `tests/data/test-mql-8-corrected.json`,Amesim 来源严格对应 `tests/data/test_mql.ame`。先核实当次文件与物理输入,不按显示名、目录名或历史报告标题认定为同一模型。原始旧工程仅保留审计用途,不重新作为活动输入。
|
||||||
|
|
||||||
|
只有八路模型确实跑不通,而且短期内没有可行修复时,才退回 `tests/data/test-mql-4-corrected.json` 与 `tests/data/test_mql_4.ame`。报告必须记录八路失败的实际终点、诊断、已排查内容及退回原因;四路结果不能代替八路验收。
|
||||||
|
|
||||||
|
## 先核对实际数据,再计时
|
||||||
|
|
||||||
|
逐元件、逐端口、逐参数检查拓扑和物理量,尤其是参考口、表压与绝对压力、SI 换算、力和流量方向、初始化及信号时刻。完整区间实际结果、守恒、事件和所需曲线按明确口径核查。存在数值差异时仍可按用户要求测量并记录运行成本,但必须将其标为性能观察,不能据此宣布正确性验收通过。完成运行和数值有限只证明可运行,不能直接判定曲线一致。
|
||||||
|
|
||||||
|
AME 文件是归档。必须记录外层 SHA-256,并核实 `.cir`、编译 `.c`、`.param`/`.data`、`.modelinfo`、`.var`/`.results`、`.sim` 的子模型、参数布局、变量索引、初值和时间范围相互对应。图纸与缓存不同就不能使用缓存作为当前参考。冻结基准绑定的另一个 AME SHA 不能因为文件同名就移作当前数据。
|
||||||
|
|
||||||
|
曲线对照保留原始采样点和事件点,明确变量映射、单位、方向、插值方法与范围。按量纲报告全时点最大绝对误差、均方根误差及定义清楚的相对误差,并单独说明事件时刻差异。峰值归一化误差不冒称为逐点相对误差;不能删去异常点、平滑尖峰或只选常规采样点来宣称整条曲线一致。缺少同工况数据或验收阈值时,明确标记该项 `skip` 或“仅观察,未验收”。
|
||||||
|
|
||||||
|
Amesim 速度比较需要同模型、同设置、完整区间的真实 CPU/墙钟记录。`.ameperf` 内的仿真事件时间不是运行耗时,归档成员修改时间也不能推导运行耗时。缺少可信耗时则明确 `skip` 速度比较;仍可独立报告满足来源条件的已保存曲线观察。
|
||||||
|
|
||||||
|
## 固定精度与运行条件
|
||||||
|
|
||||||
|
当前用户已批准网页/API 默认 `rtol = 1e-8`。同一性能比较中的精度必须固定,记录实际生效的 `rtol`、`atol`/状态误差下限、求解器、最大步长、输出间隔、起止时间、事件策略及雅可比策略。不能通过放宽精度、缩短区间、减少输出或改变物理参数制造加速结论。精度或模型变化后另建比较组,不能直接用旧组耗时计算优化比例。
|
||||||
|
|
||||||
|
正式计时记录源码版本、输入和原始结果 SHA、构建标识、编译器及积分库版本、硬件与运行环境。固定预热规则、缓存状态和重复次数,同机比较时避免并行求解、大编译等 CPU 干扰;报告逐次耗时及汇总口径。单次完整运行用于功能验证,不据此认定性能提升。
|
||||||
|
|
||||||
|
## 分阶段记录
|
||||||
|
|
||||||
|
分别记录输入加载/校验、代码生成、构建和缓存检查、进程启动、C 纯求解 CPU/墙钟、结果整理/序列化/传输,以及网页接收、持久化、曲线展示和导出。只测到总耗时就称为总耗时,不能把差值未经测量地归于某个阶段。
|
||||||
|
|
||||||
|
每次运行保存实际设置、成功或失败状态、实际终点、采样数量、求值次数、接受/拒绝步数、雅可比/线性分解次数、事件诊断和守恒结果。报告链接到原始产物,明确哪些阶段未测量、哪些比较因数据不足而跳过。后续优化以通过正确性检查后的同口径数据为依据。
|
||||||
@@ -1,5 +1,15 @@
|
|||||||
# 2026-09-11:C 管路验证版合入网页后端
|
# 2026-09-11:C 管路验证版合入网页后端
|
||||||
|
|
||||||
|
同步基线记录:将已验证的原生结果直传、浏览器Float64缓存与CSV工作线程、C端Ryu编码写出,以及八路性能调研/复现工具统一保存到 `system-optimization`。本次同步包含源代码、测试和报告;本地环境、仿真大结果和临时构建继续保留在Git忽略目录。雅可比矩阵优化从这份基线之后单独开展,沿用修正八路与 `rtol=1e-8`。
|
||||||
|
|
||||||
|
C结果编码写出优化完成:采用精确回读的Ryu数字编码与64 KiB批量写出,保留JSON结构、全量采样及binary64数值。相同修正八路,正式运行各三次,C写出1.1808→0.1638 s(-86.13%),点击到可查看8.0100→6.9756 s(-12.91%),点击到缓存保存8.1280→7.0841 s(-12.84%);CSV1.1127→1.1506 s,未观察到改善。冷编译单次3.658→4.372 s,首次网页11.682→11.381 s,已单独披露。10项编码专项、29项相关回归、8份原生结果逐位与16次网页/CSV/恢复核验通过。研究、限制、分阶段数据和复现见 [C端编码与写出报告](../other/C端结果编码与写出优化-2026-09-11.md)。未安装新环境或提交Git。
|
||||||
|
|
||||||
|
补充当前优化版八路全流程成本评估:网页预处理约19.6 ms,后台积分约6.064 s、C结果写出约1.175 s;独立原生诊断中RHS占积分93.72%,实测差分Jacobian需62,700次RHS,占全部调用84.43%。12次网页与8次原生诊断完整数值一致,本轮只增加计时工具和报告;阶段边界、CPU/墙钟、首次编译和前端采样见 [全流程成本评估](../other/八路网页求解全流程成本评估-2026-09-11.md)。
|
||||||
|
|
||||||
|
本轮结果处理优化完成:C的数值series直接进入HTTP结果流,Python不再解析并重新编码大数组;网页保存改为打包Float64块的一次原子事务,CSV改由工作线程本地生成。相同修正八路、相同精度和采样,同机旧版/优化版各预热加三次,点击到结果可查看9.872→8.034 s(-18.62%)、点击到保存完成11.774→8.132 s(-30.93%)、CSV下载保存5.610→1.005 s(-82.08%)。12次完整网页/CSV数值及刷新恢复一致;后端29项、传输6项复测、持久化6项与CSV6项专项通过。记录与实现见 [结果处理优化报告](../other/八路结果处理与网页保存优化-2026-09-11.md)。
|
||||||
|
|
||||||
|
本轮八路性能评估:后续优化优先 `test-mql-8-corrected.json`,只有八路无法运行且短期无解时才退四路。八路同精度三算法基准,当前方案纯求解 5.9138 s,相对旧固定点 12.0976 s 减少 51.12%;相同 3081320 组管阻输入重放,耗尽迭代从 3010570 次降至 0,最多 6 轮。真实网页两组各预热加三次完成;常规组点击到完成绘制中位 9.8989 s,CSV 下载 5.6198 s。已拆分后台与浏览器各阶段,并验证全部网页/CSV 数值与原生结果精确相等。当前 AME 可作保存曲线观察,但仍有早期温压和碰撞峰差异,未判八路曲线验收通过;无可信 Amesim 耗时,速度对比跳过。详见 [八路完整计时报告](../other/八路模型计算效率与网页阶段计时-2026-09-11.md) 与 [当前 AME 核查](../other/test-mql-8当前AME归档与完整曲线核查-2026-09-11.md)。本轮仅新增手动诊断和文档,未改生产算法或环境。
|
||||||
|
|
||||||
数据目录整理:按用户要求,`tests/data` 的 JSON/XML 仅保留可在浏览器导入、且已与对应 AME 参数和端口对齐的 `test-mql-4-corrected.json` 与 `test-mql-8-corrected.json`。其余8份资料逐字节迁到 `tests/fixtures` / `tests/baselines`;活动测试、审计脚本、CI、基准清单路径和当前报告已同步。两个工程真实浏览器导入及XML下载通过,导出的物理参数、连接和仿真设置与工程输入一致;8项路径/清单相关回归通过。具体去向见 [目录说明](../../tests/data/README.md)。
|
数据目录整理:按用户要求,`tests/data` 的 JSON/XML 仅保留可在浏览器导入、且已与对应 AME 参数和端口对齐的 `test-mql-4-corrected.json` 与 `test-mql-8-corrected.json`。其余8份资料逐字节迁到 `tests/fixtures` / `tests/baselines`;活动测试、审计脚本、CI、基准清单路径和当前报告已同步。两个工程真实浏览器导入及XML下载通过,导出的物理参数、连接和仿真设置与工程输入一致;8项路径/清单相关回归通过。具体去向见 [目录说明](../../tests/data/README.md)。
|
||||||
|
|
||||||
最新补充:本轮完成管流牛顿迭代的有限夹根、残差进展检查与二分回退;按用户确认将网页/API 默认 `rtol` 改为 `1e-8`。重新核对四路/八路 AME 图纸并交付 `tests/data/test-mql-4-corrected.json` 与 `tests/data/test-mql-8-corrected.json`;用户复核发现的 P4NODE2 显示方向和 3/4 号端口锚点错误一并修正。相同新精度下,相比局部半步松弛迭代,四路纯求解中位耗时 1.3309→0.9097 s(减少31.6%);72条已有Amesim参考曲线均通过原门槛,最大温差0.00882331 K。38项后端测试与2项P4真实端口几何测试通过;四路最终网页从点击到结果可查看中位3.55 s。八路修正版在新精度下单次完整完成10 s,纯求解5.91 s。最新验收范围、网页证据和八路运行限制以 [本轮完整报告](../other/牛顿管流求根与四路网页验证-2026-09-11.md) 为准;下文旧默认精度和旧输入路径为此前历史记录。
|
最新补充:本轮完成管流牛顿迭代的有限夹根、残差进展检查与二分回退;按用户确认将网页/API 默认 `rtol` 改为 `1e-8`。重新核对四路/八路 AME 图纸并交付 `tests/data/test-mql-4-corrected.json` 与 `tests/data/test-mql-8-corrected.json`;用户复核发现的 P4NODE2 显示方向和 3/4 号端口锚点错误一并修正。相同新精度下,相比局部半步松弛迭代,四路纯求解中位耗时 1.3309→0.9097 s(减少31.6%);72条已有Amesim参考曲线均通过原门槛,最大温差0.00882331 K。38项后端测试与2项P4真实端口几何测试通过;四路最终网页从点击到结果可查看中位3.55 s。八路修正版在新精度下单次完整完成10 s,纯求解5.91 s。最新验收范围、网页证据和八路运行限制以 [本轮完整报告](../other/牛顿管流求根与四路网页验证-2026-09-11.md) 为准;下文旧默认精度和旧输入路径为此前历史记录。
|
||||||
|
|||||||
@@ -1,4 +1,5 @@
|
|||||||
import { useEffect, useId, useMemo, useRef, useState } from "react";
|
import { useEffect, useId, useMemo, useRef, useState } from "react";
|
||||||
|
import { exportResultCsv, resultCsvFilename } from "./resultCsvExport";
|
||||||
import type {
|
import type {
|
||||||
ChangeEvent as ReactChangeEvent,
|
ChangeEvent as ReactChangeEvent,
|
||||||
DragEvent,
|
DragEvent,
|
||||||
@@ -442,6 +443,15 @@ export function SimulationResultsView({
|
|||||||
const [activePaneResize, setActivePaneResize] =
|
const [activePaneResize, setActivePaneResize] =
|
||||||
useState<PaneResizeKind | null>(null);
|
useState<PaneResizeKind | null>(null);
|
||||||
const [csvDownloadPending, setCsvDownloadPending] = useState(false);
|
const [csvDownloadPending, setCsvDownloadPending] = useState(false);
|
||||||
|
const csvExportRef = useRef<AbortController | null>(null);
|
||||||
|
useEffect(() => {
|
||||||
|
setCsvDownloadPending(false);
|
||||||
|
return () => {
|
||||||
|
const pending = csvExportRef.current;
|
||||||
|
csvExportRef.current = null;
|
||||||
|
pending?.abort();
|
||||||
|
};
|
||||||
|
}, [snapshot]);
|
||||||
const [activeMultiPickerWindowId, setActiveMultiPickerWindowId] = useState<
|
const [activeMultiPickerWindowId, setActiveMultiPickerWindowId] = useState<
|
||||||
string | null
|
string | null
|
||||||
>(null);
|
>(null);
|
||||||
@@ -1256,33 +1266,26 @@ export function SimulationResultsView({
|
|||||||
};
|
};
|
||||||
|
|
||||||
const downloadResultsCsv = async () => {
|
const downloadResultsCsv = async () => {
|
||||||
if (csvDownloadPending) {
|
// Ref closes the gap before React renders the disabled button.
|
||||||
return;
|
if (csvExportRef.current) return;
|
||||||
}
|
const controller = new AbortController();
|
||||||
|
csvExportRef.current = controller;
|
||||||
setCsvDownloadPending(true);
|
setCsvDownloadPending(true);
|
||||||
try {
|
try {
|
||||||
const response = await fetch("/api/simulation-results/csv", {
|
const blob = await exportResultCsv(snapshot.result, controller.signal);
|
||||||
method: "POST",
|
if (csvExportRef.current === controller) {
|
||||||
headers: { "Content-Type": "application/json" },
|
downloadBlob(blob, resultCsvFilename(snapshot.project.name));
|
||||||
body: JSON.stringify({
|
|
||||||
projectName: snapshot.project.name,
|
|
||||||
variables: snapshot.result.variables,
|
|
||||||
series: snapshot.result.series,
|
|
||||||
}),
|
|
||||||
});
|
|
||||||
if (!response.ok) {
|
|
||||||
throw new Error(await responseErrorMessage(response));
|
|
||||||
}
|
}
|
||||||
const filename = responseDownloadFilename(
|
|
||||||
response,
|
|
||||||
`${safeFileStem(snapshot.project.name)}-results.csv`,
|
|
||||||
);
|
|
||||||
downloadBlob(await response.blob(), filename);
|
|
||||||
} catch (error) {
|
} catch (error) {
|
||||||
|
if (!controller.signal.aborted && csvExportRef.current === controller) {
|
||||||
window.alert(`CSV 下载失败:${errorMessage(error)}`);
|
window.alert(`CSV 下载失败:${errorMessage(error)}`);
|
||||||
|
}
|
||||||
} finally {
|
} finally {
|
||||||
|
if (csvExportRef.current === controller) {
|
||||||
|
csvExportRef.current = null;
|
||||||
setCsvDownloadPending(false);
|
setCsvDownloadPending(false);
|
||||||
}
|
}
|
||||||
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
return (
|
return (
|
||||||
@@ -5539,31 +5542,6 @@ function compactFileTimestamp(value: string) {
|
|||||||
)}`;
|
)}`;
|
||||||
}
|
}
|
||||||
|
|
||||||
function responseDownloadFilename(response: Response, fallback: string) {
|
|
||||||
const disposition = response.headers.get("Content-Disposition") ?? "";
|
|
||||||
const encoded = /filename\*=UTF-8''([^;]+)/i.exec(disposition)?.[1];
|
|
||||||
if (encoded) {
|
|
||||||
try {
|
|
||||||
return decodeURIComponent(encoded);
|
|
||||||
} catch {
|
|
||||||
return fallback;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return /filename="([^"]+)"/i.exec(disposition)?.[1] ?? fallback;
|
|
||||||
}
|
|
||||||
|
|
||||||
async function responseErrorMessage(response: Response) {
|
|
||||||
try {
|
|
||||||
const payload = (await response.json()) as { detail?: unknown };
|
|
||||||
if (typeof payload.detail === "string") {
|
|
||||||
return payload.detail;
|
|
||||||
}
|
|
||||||
} catch {
|
|
||||||
// Use the HTTP status when the server did not return a JSON error body.
|
|
||||||
}
|
|
||||||
return `${response.status} ${response.statusText}`.trim();
|
|
||||||
}
|
|
||||||
|
|
||||||
function errorMessage(error: unknown) {
|
function errorMessage(error: unknown) {
|
||||||
return error instanceof Error ? error.message : String(error);
|
return error instanceof Error ? error.message : String(error);
|
||||||
}
|
}
|
||||||
@@ -5582,11 +5560,14 @@ function downloadBlob(blob: Blob, filename: string) {
|
|||||||
anchor.href = url;
|
anchor.href = url;
|
||||||
anchor.download = filename;
|
anchor.download = filename;
|
||||||
anchor.style.display = "none";
|
anchor.style.display = "none";
|
||||||
|
try {
|
||||||
document.body.append(anchor);
|
document.body.append(anchor);
|
||||||
anchor.click();
|
anchor.click();
|
||||||
|
} finally {
|
||||||
anchor.remove();
|
anchor.remove();
|
||||||
window.setTimeout(() => URL.revokeObjectURL(url), 1000);
|
window.setTimeout(() => URL.revokeObjectURL(url), 1000);
|
||||||
}
|
}
|
||||||
|
}
|
||||||
|
|
||||||
function colorForVariable(key: string) {
|
function colorForVariable(key: string) {
|
||||||
let hash = 0;
|
let hash = 0;
|
||||||
|
|||||||
@@ -0,0 +1,38 @@
|
|||||||
|
import { buildResultCsvBlob } from "./resultCsvEncoding";
|
||||||
|
import type { ResultCsvWorkerMessage, ResultCsvWorkerResponse } from "./resultCsvExport";
|
||||||
|
|
||||||
|
// Keep DOM and worker entry points on the same app tsconfig without mixing libs.
|
||||||
|
const worker = self as unknown as {
|
||||||
|
onmessage: ((event: MessageEvent<ResultCsvWorkerMessage>) => void) | null;
|
||||||
|
postMessage: (message: ResultCsvWorkerResponse) => void;
|
||||||
|
};
|
||||||
|
let keys: string[] | null = null;
|
||||||
|
let rowCount = 0;
|
||||||
|
let received = 0;
|
||||||
|
let values: Float64Array | null = null;
|
||||||
|
|
||||||
|
worker.onmessage = ({ data }) => {
|
||||||
|
try {
|
||||||
|
if (data.type === "start") {
|
||||||
|
keys = data.keys;
|
||||||
|
rowCount = data.rowCount;
|
||||||
|
values = new Float64Array(keys.length * rowCount);
|
||||||
|
received = 0;
|
||||||
|
} else if (data.type === "chunk") {
|
||||||
|
if (!values || data.offset !== received || received + data.values.length > values.length) {
|
||||||
|
throw new Error("Simulation result CSV data is incomplete.");
|
||||||
|
}
|
||||||
|
values.set(data.values, received);
|
||||||
|
received += data.values.length;
|
||||||
|
} else {
|
||||||
|
if (!keys || !values || received !== values.length) {
|
||||||
|
throw new Error("Simulation result CSV data is incomplete.");
|
||||||
|
}
|
||||||
|
worker.postMessage({ type: "complete", blob: buildResultCsvBlob(keys, rowCount, values) });
|
||||||
|
keys = values = null;
|
||||||
|
}
|
||||||
|
} catch (error) {
|
||||||
|
keys = values = null;
|
||||||
|
worker.postMessage({ type: "error", message: error instanceof Error ? error.message : String(error) });
|
||||||
|
}
|
||||||
|
};
|
||||||
@@ -0,0 +1,37 @@
|
|||||||
|
/** CSV uses raw result units and metadata order, independently of chart units. */
|
||||||
|
export function buildResultCsvBlob(
|
||||||
|
keys: readonly string[],
|
||||||
|
rowCount: number,
|
||||||
|
values: Float64Array,
|
||||||
|
): Blob {
|
||||||
|
if (values.length !== keys.length * rowCount) {
|
||||||
|
throw new Error("Simulation result CSV data is incomplete.");
|
||||||
|
}
|
||||||
|
const escape = (value: string) =>
|
||||||
|
/[",\r\n]/.test(value) ? `"${value.replace(/"/g, '""')}"` : value;
|
||||||
|
const parts: BlobPart[] = ["\ufeff", keys.map(escape).join(","), "\r\n"];
|
||||||
|
const lines: string[] = [];
|
||||||
|
const row = new Array<string>(keys.length);
|
||||||
|
let chunkLength = 0;
|
||||||
|
for (let index = 0; index < rowCount; index++) {
|
||||||
|
for (let column = 0; column < keys.length; column++) {
|
||||||
|
const value = values[column * rowCount + index];
|
||||||
|
if (!Number.isFinite(value)) {
|
||||||
|
throw new Error(`Simulation result column '${keys[column]}' contains non-finite values.`);
|
||||||
|
}
|
||||||
|
// Number.toString() round-trips binary64. Preserve negative zero too;
|
||||||
|
// unlike Python float repr, integral values need no redundant ".0".
|
||||||
|
row[column] = Object.is(value, -0) ? "-0" : String(value);
|
||||||
|
}
|
||||||
|
const line = row.join(",") + "\r\n";
|
||||||
|
lines.push(line);
|
||||||
|
chunkLength += line.length;
|
||||||
|
if (chunkLength >= 1024 * 1024) {
|
||||||
|
parts.push(lines.join(""));
|
||||||
|
lines.length = 0;
|
||||||
|
chunkLength = 0;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (lines.length) parts.push(lines.join(""));
|
||||||
|
return new Blob(parts, { type: "text/csv;charset=utf-8" });
|
||||||
|
}
|
||||||
@@ -0,0 +1,128 @@
|
|||||||
|
export type ResultCsvInput = {
|
||||||
|
variables: readonly { key: string }[];
|
||||||
|
series: Readonly<Record<string, readonly number[]>>;
|
||||||
|
};
|
||||||
|
|
||||||
|
export type ResultCsvWorkerMessage =
|
||||||
|
| { type: "start"; keys: string[]; rowCount: number }
|
||||||
|
| { type: "chunk"; offset: number; values: Float64Array }
|
||||||
|
| { type: "finish" };
|
||||||
|
export type ResultCsvWorkerResponse =
|
||||||
|
| { type: "complete"; blob: Blob }
|
||||||
|
| { type: "error"; message: string };
|
||||||
|
|
||||||
|
function csvColumns(input: ResultCsvInput) {
|
||||||
|
const times = input.series.time;
|
||||||
|
if (!Array.isArray(times) || !times.length) {
|
||||||
|
throw new Error("Simulation results must contain a non-empty time series.");
|
||||||
|
}
|
||||||
|
const variableKeys = input.variables.map((variable) => variable.key);
|
||||||
|
if (!variableKeys.length) {
|
||||||
|
throw new Error("Simulation results do not contain exportable variables.");
|
||||||
|
}
|
||||||
|
if (new Set(variableKeys).size !== variableKeys.length) {
|
||||||
|
throw new Error("Simulation result metadata contains duplicate variable keys.");
|
||||||
|
}
|
||||||
|
const keys = ["time", ...variableKeys];
|
||||||
|
const expectedKeys = new Set(keys);
|
||||||
|
const missing = [...expectedKeys].filter((key) => !Object.hasOwn(input.series, key)).sort();
|
||||||
|
const unknown = Object.keys(input.series).filter((key) => !expectedKeys.has(key)).sort();
|
||||||
|
if (missing.length || unknown.length) {
|
||||||
|
const details = [missing.length ? `missing ${missing.join(", ")}` : "", unknown.length ? `unmapped ${unknown.join(", ")}` : ""].filter(Boolean);
|
||||||
|
throw new Error(`Simulation result columns do not match metadata: ${details.join("; ")}.`);
|
||||||
|
}
|
||||||
|
for (const key of keys) {
|
||||||
|
if (!Array.isArray(input.series[key]) || input.series[key].length !== times.length) {
|
||||||
|
throw new Error(`Simulation result column '${key}' has an inconsistent length.`);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return { keys, rowCount: times.length };
|
||||||
|
}
|
||||||
|
|
||||||
|
export function resultCsvFilename(projectName: string): string {
|
||||||
|
const stem = projectName.replace(/[<>:"/\\|?*\u0000-\u001f]/g, "_").replace(/^[ .]+|[ .]+$/g, "");
|
||||||
|
return `${Array.from(stem).slice(0, 80).join("") || "simulation"}-results.csv`;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Transfer at most 1 MiB per task; never stringify or clone the full result. */
|
||||||
|
export function exportResultCsv(input: ResultCsvInput, signal?: AbortSignal): Promise<Blob> {
|
||||||
|
return new Promise((resolve, reject) => {
|
||||||
|
if (signal?.aborted) {
|
||||||
|
reject(new DOMException("CSV export was cancelled.", "AbortError"));
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
let worker: Worker | undefined;
|
||||||
|
let settled = false;
|
||||||
|
let yieldTimer: ReturnType<typeof setTimeout> | undefined;
|
||||||
|
let resumeYield: (() => void) | undefined;
|
||||||
|
const finish = (error?: unknown, blob?: Blob) => {
|
||||||
|
if (settled) return;
|
||||||
|
settled = true;
|
||||||
|
signal?.removeEventListener("abort", abort);
|
||||||
|
if (yieldTimer !== undefined) clearTimeout(yieldTimer);
|
||||||
|
resumeYield?.();
|
||||||
|
worker?.terminate();
|
||||||
|
if (error !== undefined) reject(error);
|
||||||
|
else resolve(blob!);
|
||||||
|
};
|
||||||
|
const abort = () => finish(new DOMException("CSV export was cancelled.", "AbortError"));
|
||||||
|
signal?.addEventListener("abort", abort, { once: true });
|
||||||
|
const yieldMainThread = () => new Promise<void>((resume) => {
|
||||||
|
resumeYield = resume;
|
||||||
|
yieldTimer = setTimeout(() => {
|
||||||
|
yieldTimer = undefined;
|
||||||
|
resumeYield = undefined;
|
||||||
|
resume();
|
||||||
|
}, 0);
|
||||||
|
});
|
||||||
|
try {
|
||||||
|
const { keys, rowCount } = csvColumns(input);
|
||||||
|
worker = new Worker(new URL("./resultCsv.worker.ts", import.meta.url), { type: "module" });
|
||||||
|
worker.onmessage = ({ data }: MessageEvent<ResultCsvWorkerResponse>) => {
|
||||||
|
if (data.type === "complete" && data.blob instanceof Blob) finish(undefined, data.blob);
|
||||||
|
else finish(new Error(data.type === "error" ? data.message : "CSV worker returned an invalid response."));
|
||||||
|
};
|
||||||
|
worker.onerror = (event) => {
|
||||||
|
event.preventDefault();
|
||||||
|
finish(new Error(event.message || "CSV worker failed."));
|
||||||
|
};
|
||||||
|
worker.onmessageerror = () => finish(new Error("CSV worker response could not be read."));
|
||||||
|
const transfer = async () => {
|
||||||
|
// Let the pending indicator render before copying any numeric payload.
|
||||||
|
await yieldMainThread();
|
||||||
|
if (settled) return;
|
||||||
|
worker!.postMessage({ type: "start", keys, rowCount } satisfies ResultCsvWorkerMessage);
|
||||||
|
const total = keys.length * rowCount;
|
||||||
|
let column = 0;
|
||||||
|
let row = 0;
|
||||||
|
for (let offset = 0; offset < total;) {
|
||||||
|
if (settled) return;
|
||||||
|
const batch = new Float64Array(Math.min(128 * 1024, total - offset));
|
||||||
|
let filled = 0;
|
||||||
|
while (filled < batch.length) {
|
||||||
|
const source = input.series[keys[column]];
|
||||||
|
const count = Math.min(rowCount - row, batch.length - filled);
|
||||||
|
for (let index = 0; index < count; index++) {
|
||||||
|
const value = source[row + index];
|
||||||
|
if (typeof value !== "number") {
|
||||||
|
throw new Error(`Simulation result column '${keys[column]}' contains non-numeric values.`);
|
||||||
|
}
|
||||||
|
batch[filled + index] = value;
|
||||||
|
}
|
||||||
|
row += count;
|
||||||
|
filled += count;
|
||||||
|
if (row === rowCount) { column++; row = 0; }
|
||||||
|
}
|
||||||
|
const transferredCount = batch.length;
|
||||||
|
worker!.postMessage({ type: "chunk", offset, values: batch } satisfies ResultCsvWorkerMessage, [batch.buffer]);
|
||||||
|
offset += transferredCount;
|
||||||
|
if (offset < total) await yieldMainThread();
|
||||||
|
}
|
||||||
|
if (!settled) worker!.postMessage({ type: "finish" } satisfies ResultCsvWorkerMessage);
|
||||||
|
};
|
||||||
|
void transfer().catch(finish);
|
||||||
|
} catch (error) {
|
||||||
|
finish(error);
|
||||||
|
}
|
||||||
|
});
|
||||||
|
}
|
||||||
@@ -2,13 +2,15 @@ import type { SimulationResultsSnapshot } from "./SimulationResultsView";
|
|||||||
|
|
||||||
export const RESULT_SNAPSHOT_KEY = "system-simulation-flow:latest-result";
|
export const RESULT_SNAPSHOT_KEY = "system-simulation-flow:latest-result";
|
||||||
const DATABASE = "system-simulation-results";
|
const DATABASE = "system-simulation-results";
|
||||||
const CHUNK_SIZE = 32768;
|
// Pack short output columns together: an eight-branch result needs seven writes,
|
||||||
const BATCH_VALUES = 131072;
|
// rather than one write per output column and fourteen serial transactions.
|
||||||
|
const PACKED_VALUES = 262144;
|
||||||
type Header = {
|
type Header = {
|
||||||
snapshot: SimulationResultsSnapshot;
|
snapshot: SimulationResultsSnapshot;
|
||||||
lengths: Record<string, number>;
|
lengths: Record<string, number>;
|
||||||
|
layout?: "packed-f64-v1";
|
||||||
};
|
};
|
||||||
type Chunk = { key: IDBValidKey; values: Float64Array };
|
type PackedChunk = { offset: number; values: Float64Array };
|
||||||
let database: Promise<IDBDatabase> | undefined;
|
let database: Promise<IDBDatabase> | undefined;
|
||||||
let saveSequence = 0;
|
let saveSequence = 0;
|
||||||
let pendingSaves = 0;
|
let pendingSaves = 0;
|
||||||
@@ -39,15 +41,17 @@ function openDatabase() {
|
|||||||
return database;
|
return database;
|
||||||
}
|
}
|
||||||
|
|
||||||
function writeBatch(db: IDBDatabase, chunks: Chunk[], header?: [string, Header]) {
|
function writeSnapshot(db: IDBDatabase, cacheId: string, chunks: PackedChunk[], header: Header) {
|
||||||
return new Promise<void>((resolve, reject) => {
|
return new Promise<void>((resolve, reject) => {
|
||||||
const transaction = db.transaction(["headers", "chunks"], "readwrite");
|
const transaction = db.transaction(["headers", "chunks"], "readwrite");
|
||||||
transaction.oncomplete = () => resolve();
|
transaction.oncomplete = () => resolve();
|
||||||
transaction.onabort = () => reject(transaction.error ?? new Error("结果保存事务已中止"));
|
transaction.onabort = () => reject(transaction.error ?? new Error("结果保存事务已中止"));
|
||||||
transaction.onerror = () => {}; // onabort reports the transaction failure once.
|
transaction.onerror = () => {}; // onabort reports the transaction failure once.
|
||||||
try {
|
try {
|
||||||
for (const chunk of chunks) transaction.objectStore("chunks").put(chunk.values, chunk.key);
|
const store = transaction.objectStore("chunks");
|
||||||
if (header) transaction.objectStore("headers").put(header[1], header[0]);
|
for (const chunk of chunks) store.put(chunk, [cacheId, chunk.offset]);
|
||||||
|
transaction.objectStore("headers").put(header, cacheId);
|
||||||
|
transaction.commit?.();
|
||||||
} catch (error) {
|
} catch (error) {
|
||||||
transaction.abort();
|
transaction.abort();
|
||||||
reject(error);
|
reject(error);
|
||||||
@@ -65,7 +69,7 @@ function deleteCache(db: IDBDatabase, cacheId: string) {
|
|||||||
});
|
});
|
||||||
}
|
}
|
||||||
|
|
||||||
/** Publish the small session pointer only after every data chunk is durably committed. */
|
/** Publish the small session pointer only after the atomic snapshot write commits. */
|
||||||
export async function storeResultSnapshot(snapshot: SimulationResultsSnapshot) {
|
export async function storeResultSnapshot(snapshot: SimulationResultsSnapshot) {
|
||||||
const sequence = ++saveSequence;
|
const sequence = ++saveSequence;
|
||||||
const cacheId = typeof crypto.randomUUID === "function"
|
const cacheId = typeof crypto.randomUUID === "function"
|
||||||
@@ -74,49 +78,86 @@ export async function storeResultSnapshot(snapshot: SimulationResultsSnapshot) {
|
|||||||
pendingSaves++;
|
pendingSaves++;
|
||||||
let db: IDBDatabase | undefined;
|
let db: IDBDatabase | undefined;
|
||||||
let committed = false;
|
let committed = false;
|
||||||
|
let writeStarted = false;
|
||||||
try {
|
try {
|
||||||
// Yield before serialization so the ready result and controls can paint first.
|
// Yield before copying so the ready result and controls can paint first.
|
||||||
await new Promise<void>((resolve) => setTimeout(resolve, 0));
|
await new Promise<void>((resolve) => setTimeout(resolve, 0));
|
||||||
if (sequence !== saveSequence) return false;
|
if (sequence !== saveSequence) return false;
|
||||||
|
const columns = Object.entries(snapshot.result.series);
|
||||||
|
const lengths: Record<string, number> = Object.create(null);
|
||||||
|
let total = 0;
|
||||||
|
for (const [name, values] of columns) { lengths[name] = values.length; total += values.length; }
|
||||||
|
const chunks: PackedChunk[] = [];
|
||||||
|
let columnIndex = 0;
|
||||||
|
let columnOffset = 0;
|
||||||
|
for (let offset = 0; offset < total; offset += PACKED_VALUES) {
|
||||||
|
if (sequence !== saveSequence) return false;
|
||||||
|
const values = new Float64Array(Math.min(PACKED_VALUES, total - offset));
|
||||||
|
let written = 0;
|
||||||
|
while (written < values.length) {
|
||||||
|
const source = columns[columnIndex][1];
|
||||||
|
const count = Math.min(source.length - columnOffset, values.length - written);
|
||||||
|
if (columnOffset === 0 && count === source.length) values.set(source, written);
|
||||||
|
else for (let i = 0; i < count; i++) values[written + i] = source[columnOffset + i];
|
||||||
|
written += count;
|
||||||
|
columnOffset += count;
|
||||||
|
if (columnOffset === source.length) { columnIndex++; columnOffset = 0; }
|
||||||
|
}
|
||||||
|
chunks.push({ offset, values });
|
||||||
|
// Keep cancellation responsive without holding an idle IndexedDB transaction open.
|
||||||
|
if (offset + values.length < total) await new Promise<void>((resolve) => setTimeout(resolve, 0));
|
||||||
|
}
|
||||||
|
if (sequence !== saveSequence) return false;
|
||||||
db = await openDatabase();
|
db = await openDatabase();
|
||||||
const lengths: Record<string, number> = Object.create(null);
|
|
||||||
let batch: Chunk[] = [];
|
|
||||||
let batchSize = 0;
|
|
||||||
for (const [name, values] of Object.entries(snapshot.result.series)) {
|
|
||||||
lengths[name] = values.length;
|
|
||||||
for (let offset = 0; offset < values.length; offset += CHUNK_SIZE) {
|
|
||||||
if (sequence !== saveSequence) return false;
|
if (sequence !== saveSequence) return false;
|
||||||
const length = Math.min(CHUNK_SIZE, values.length - offset);
|
writeStarted = true;
|
||||||
const block = new Float64Array(length);
|
await writeSnapshot(db, cacheId, chunks, {
|
||||||
for (let i = 0; i < length; i++) block[i] = values[offset + i];
|
snapshot: { ...snapshot, result: { ...snapshot.result, series: Object.create(null) } },
|
||||||
batch.push({ key: [cacheId, name, offset], values: block });
|
lengths, layout: "packed-f64-v1",
|
||||||
batchSize += length;
|
});
|
||||||
if (batchSize >= BATCH_VALUES) {
|
|
||||||
await writeBatch(db, batch);
|
|
||||||
batch = [];
|
|
||||||
batchSize = 0;
|
|
||||||
await new Promise<void>((resolve) => setTimeout(resolve, 0));
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (sequence !== saveSequence) return false;
|
|
||||||
await writeBatch(db, batch, [cacheId, {
|
|
||||||
snapshot: { ...snapshot, result: { ...snapshot.result, series: Object.create(null) } }, lengths,
|
|
||||||
}]);
|
|
||||||
if (sequence !== saveSequence) return false;
|
if (sequence !== saveSequence) return false;
|
||||||
sessionStorage.setItem(RESULT_SNAPSHOT_KEY, JSON.stringify({ storage: "indexeddb", version: 1, cacheId }));
|
sessionStorage.setItem(RESULT_SNAPSHOT_KEY, JSON.stringify({ storage: "indexeddb", version: 1, cacheId }));
|
||||||
committed = true;
|
committed = true;
|
||||||
// Only retire this page's own previous save, never another tab's loaded snapshot.
|
// Only retire this page's own previous save, never a snapshot loaded from another tab.
|
||||||
const previous = ownedCacheId;
|
const previous = ownedCacheId;
|
||||||
ownedCacheId = cacheId;
|
ownedCacheId = cacheId;
|
||||||
if (previous) void deleteCache(db, previous).catch(() => {});
|
if (previous) void deleteCache(db, previous).catch(() => {});
|
||||||
return true;
|
return true;
|
||||||
} finally {
|
} finally {
|
||||||
pendingSaves--;
|
pendingSaves--;
|
||||||
if (db && !committed) await deleteCache(db, cacheId).catch(() => {});
|
if (db && writeStarted && !committed) await deleteCache(db, cacheId).catch(() => {});
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
function cacheError() { return new Error("结果缓存不完整或分块无效,请重新载入结果文件"); }
|
||||||
|
|
||||||
|
function restorePacked(header: Header, chunks: PackedChunk[]) {
|
||||||
|
const columns = Object.entries(header.lengths);
|
||||||
|
const total = columns.reduce((sum, [, length]) => sum + length, 0);
|
||||||
|
let loaded = 0;
|
||||||
|
let columnIndex = 0;
|
||||||
|
let columnOffset = 0;
|
||||||
|
for (const chunk of chunks) {
|
||||||
|
if (!chunk || chunk.offset !== loaded || !(chunk.values instanceof Float64Array) ||
|
||||||
|
chunk.values.length === 0 || loaded + chunk.values.length > total) throw cacheError();
|
||||||
|
let read = 0;
|
||||||
|
while (read < chunk.values.length) {
|
||||||
|
while (columnIndex < columns.length && columnOffset === columns[columnIndex][1]) {
|
||||||
|
columnIndex++; columnOffset = 0;
|
||||||
|
}
|
||||||
|
if (columnIndex >= columns.length) throw cacheError();
|
||||||
|
const [name, length] = columns[columnIndex];
|
||||||
|
const count = Math.min(length - columnOffset, chunk.values.length - read);
|
||||||
|
const target = header.snapshot.result.series[name];
|
||||||
|
for (let i = 0; i < count; i++) target[columnOffset + i] = chunk.values[read + i];
|
||||||
|
read += count;
|
||||||
|
columnOffset += count;
|
||||||
|
}
|
||||||
|
loaded += chunk.values.length;
|
||||||
|
}
|
||||||
|
if (loaded !== total) throw cacheError();
|
||||||
|
}
|
||||||
|
|
||||||
export async function loadStoredResultSnapshot(): Promise<unknown | null> {
|
export async function loadStoredResultSnapshot(): Promise<unknown | null> {
|
||||||
const raw = sessionStorage.getItem(RESULT_SNAPSHOT_KEY);
|
const raw = sessionStorage.getItem(RESULT_SNAPSHOT_KEY);
|
||||||
if (!raw) return null;
|
if (!raw) return null;
|
||||||
@@ -128,38 +169,54 @@ export async function loadStoredResultSnapshot(): Promise<unknown | null> {
|
|||||||
return new Promise<SimulationResultsSnapshot | null>((resolve, reject) => {
|
return new Promise<SimulationResultsSnapshot | null>((resolve, reject) => {
|
||||||
const transaction = db.transaction(["headers", "chunks"], "readonly");
|
const transaction = db.transaction(["headers", "chunks"], "readonly");
|
||||||
let result: SimulationResultsSnapshot | null = null;
|
let result: SimulationResultsSnapshot | null = null;
|
||||||
let failure: Error | null = null;
|
let failure: unknown = null;
|
||||||
transaction.oncomplete = () => failure ? reject(failure) : resolve(result);
|
transaction.oncomplete = () => failure ? reject(failure) : resolve(result);
|
||||||
transaction.onabort = () => reject(transaction.error);
|
transaction.onabort = () => reject(transaction.error);
|
||||||
const request = transaction.objectStore("headers").get(value.cacheId);
|
const request = transaction.objectStore("headers").get(value.cacheId);
|
||||||
request.onsuccess = () => {
|
request.onsuccess = () => {
|
||||||
const header = request.result as Header | undefined;
|
const header = request.result as Header | undefined;
|
||||||
if (!header) { failure = new Error("结果缓存已丢失,请重新载入结果文件"); return; }
|
if (!header) { failure = new Error("结果缓存已丢失,请重新载入结果文件"); return; }
|
||||||
|
try {
|
||||||
result = header.snapshot;
|
result = header.snapshot;
|
||||||
for (const [name, length] of Object.entries(header.lengths)) {
|
for (const [name, length] of Object.entries(header.lengths)) {
|
||||||
|
if (!Number.isSafeInteger(length) || length < 0) throw cacheError();
|
||||||
result.result.series[name] = new Array<number>(length);
|
result.result.series[name] = new Array<number>(length);
|
||||||
}
|
}
|
||||||
|
const range = IDBKeyRange.bound([value.cacheId], [value.cacheId, []]);
|
||||||
|
if (header.layout === "packed-f64-v1") {
|
||||||
|
// A handful of packed records can be restored with one IndexedDB delivery.
|
||||||
|
const blocks = transaction.objectStore("chunks").getAll(range);
|
||||||
|
blocks.onsuccess = () => {
|
||||||
|
try { restorePacked(header, blocks.result as PackedChunk[]); }
|
||||||
|
catch (error) { failure = error; }
|
||||||
|
};
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
if (header.layout !== undefined) throw cacheError();
|
||||||
|
// Compatibility with the original [cacheId, columnName, offset] records.
|
||||||
const loaded: Record<string, number> = Object.create(null);
|
const loaded: Record<string, number> = Object.create(null);
|
||||||
const cursor = transaction.objectStore("chunks").openCursor(IDBKeyRange.bound([value.cacheId], [value.cacheId, []]));
|
const cursor = transaction.objectStore("chunks").openCursor(range);
|
||||||
cursor.onsuccess = () => {
|
cursor.onsuccess = () => {
|
||||||
const entry = cursor.result;
|
const entry = cursor.result;
|
||||||
if (!entry) {
|
if (!entry) {
|
||||||
for (const [name, length] of Object.entries(header.lengths)) {
|
for (const [name, length] of Object.entries(header.lengths)) {
|
||||||
if ((loaded[name] ?? 0) !== length) failure = new Error("结果缓存不完整,请重新载入结果文件");
|
if ((loaded[name] ?? 0) !== length) failure = cacheError();
|
||||||
}
|
}
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
const [, name, offset] = entry.key as [string, string, number];
|
const [, name, offset] = entry.key as [string, string, number];
|
||||||
const block = entry.value as Float64Array;
|
const block = entry.value as Float64Array;
|
||||||
const values = result!.result.series[name];
|
const values = result!.result.series[name];
|
||||||
if (!values || offset < 0 || offset + block.length > values.length) {
|
if (!values || !(block instanceof Float64Array) || offset !== (loaded[name] ?? 0) ||
|
||||||
failure = new Error("结果缓存分块无效");
|
offset + block.length > values.length) {
|
||||||
|
failure = cacheError();
|
||||||
} else {
|
} else {
|
||||||
for (let i = 0; i < block.length; i++) values[offset + i] = block[i];
|
for (let i = 0; i < block.length; i++) values[offset + i] = block[i];
|
||||||
loaded[name] = (loaded[name] ?? 0) + block.length;
|
loaded[name] = offset + block.length;
|
||||||
}
|
}
|
||||||
entry.continue();
|
entry.continue();
|
||||||
};
|
};
|
||||||
|
} catch (error) { failure = error; }
|
||||||
};
|
};
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
@@ -0,0 +1,208 @@
|
|||||||
|
import { expect, test, type Page } from "@playwright/test";
|
||||||
|
import { buildResultCsvBlob } from "../../src/resultCsvEncoding";
|
||||||
|
import { prepareApp, resultSnapshot } from "./fixtures";
|
||||||
|
|
||||||
|
test("CSV preserves BOM, CRLF, escaped raw keys and every binary64 value", async () => {
|
||||||
|
const numbers = [-0, 0, Number.MIN_VALUE, -Number.MIN_VALUE, Number.MAX_VALUE, -Number.MAX_VALUE,
|
||||||
|
1 + Number.EPSILON, 1e21, 1e-7, Math.PI, 403499745569.73364];
|
||||||
|
const keys = ["time", '力,"端口"\r\nN'];
|
||||||
|
const blob = buildResultCsvBlob(keys, numbers.length, new Float64Array([...numbers, ...numbers]));
|
||||||
|
const bytes = new Uint8Array(await blob.arrayBuffer());
|
||||||
|
expect([...bytes.slice(0, 3)]).toEqual([0xef, 0xbb, 0xbf]);
|
||||||
|
const text = new TextDecoder().decode(bytes);
|
||||||
|
const header = 'time,"力,""端口""\r\nN"\r\n';
|
||||||
|
expect(text.startsWith(header)).toBe(true);
|
||||||
|
const rows = text.slice(header.length).split("\r\n");
|
||||||
|
expect(rows.pop()).toBe("");
|
||||||
|
expect(rows).toHaveLength(numbers.length);
|
||||||
|
rows.forEach((row, i) => row.split(",").forEach((value) => expect(Object.is(Number(value), numbers[i])).toBe(true)));
|
||||||
|
expect(blob.type).toBe("text/csv;charset=utf-8");
|
||||||
|
expect(() => buildResultCsvBlob(["time"], 1, new Float64Array([Infinity]))).toThrow("non-finite");
|
||||||
|
expect(() => buildResultCsvBlob(["time"], 2, new Float64Array([0]))).toThrow("incomplete");
|
||||||
|
});
|
||||||
|
|
||||||
|
async function probePage(page: Page) {
|
||||||
|
await page.route("**/csv-worker-probe", (route) => route.fulfill({ contentType: "text/html", body: "<title>CSV worker probe</title>" }));
|
||||||
|
await page.goto("/csv-worker-probe");
|
||||||
|
}
|
||||||
|
|
||||||
|
test("real Worker exports over one million raw cells exactly in bounded transfers", async ({ page }) => {
|
||||||
|
test.setTimeout(60_000);
|
||||||
|
await probePage(page);
|
||||||
|
let csvRequests = 0;
|
||||||
|
page.on("request", (request) => { if (request.url().includes("/api/simulation-results/csv")) csvRequests++; });
|
||||||
|
const result = await page.evaluate(async () => {
|
||||||
|
// @ts-expect-error Vite serves this browser module.
|
||||||
|
const { exportResultCsv } = await import("/src/resultCsvExport.ts");
|
||||||
|
const OriginalWorker = Worker;
|
||||||
|
let workerCount = 0, terminated = 0, maxTransferBytes = 0, transferCount = 0;
|
||||||
|
window.Worker = class extends OriginalWorker {
|
||||||
|
constructor(url: string | URL, options?: WorkerOptions) { super(url, options); workerCount++; }
|
||||||
|
postMessage(message: any, transfer?: any) {
|
||||||
|
if (message.type === "chunk") { maxTransferBytes = Math.max(maxTransferBytes, message.values.byteLength); transferCount++; }
|
||||||
|
super.postMessage(message, transfer);
|
||||||
|
}
|
||||||
|
terminate() { terminated++; super.terminate(); }
|
||||||
|
};
|
||||||
|
const count = 4097;
|
||||||
|
const variables = Array.from({ length: 256 }, (_, i) => ({ key: `component_${255-i}.pressure`, unit: "Pa", order: i }));
|
||||||
|
const series: Record<string, number[]> = { time: Array.from({ length: count }, (_, i) => i / 1000) };
|
||||||
|
series.time[1] = 0; // Duplicate and event samples are data, not rows to remove.
|
||||||
|
series.time[2100] = 2.099000001;
|
||||||
|
for (let column = variables.length - 1; column >= 0; column--) {
|
||||||
|
series[variables[column].key] = Array.from({ length: count }, (_, i) => Math.sin(i / 17) * (column + 1));
|
||||||
|
}
|
||||||
|
const extremes = [-0, Number.MIN_VALUE, Number.MAX_VALUE, 1 + Number.EPSILON, 403499745569.73364];
|
||||||
|
extremes.forEach((value, i) => { series[variables[0].key][i] = value; });
|
||||||
|
let ticks = 0;
|
||||||
|
const timer = setInterval(() => ticks++, 0);
|
||||||
|
const started = performance.now();
|
||||||
|
const blob = await exportResultCsv({ variables, series });
|
||||||
|
const elapsedMs = performance.now() - started;
|
||||||
|
clearInterval(timer);
|
||||||
|
window.Worker = OriginalWorker;
|
||||||
|
const bytes = new Uint8Array(await blob.arrayBuffer());
|
||||||
|
const text = new TextDecoder().decode(bytes);
|
||||||
|
const lines = text.split("\r\n");
|
||||||
|
const expectedKeys = ["time", ...variables.map((v) => v.key)];
|
||||||
|
const ordered = lines.shift() === expectedKeys.join(",");
|
||||||
|
const trailingNewline = lines.pop() === "";
|
||||||
|
let checked = 0, exact = true;
|
||||||
|
lines.forEach((line, row) => line.split(",").forEach((value, column) => {
|
||||||
|
exact &&= Object.is(Number(value), series[expectedKeys[column]][row]); checked++;
|
||||||
|
}));
|
||||||
|
return { ordered, trailingNewline, exact, checked, rowCount: lines.length,
|
||||||
|
bom: [...bytes.slice(0,3)], workerCount, terminated, maxTransferBytes, transferCount, ticks, elapsedMs };
|
||||||
|
});
|
||||||
|
expect(result).toMatchObject({ ordered: true, trailingNewline: true, exact: true, checked: 257 * 4097,
|
||||||
|
rowCount: 4097, bom: [239,187,191], workerCount: 1, terminated: 1 });
|
||||||
|
expect(result.maxTransferBytes).toBeLessThanOrEqual(1024 * 1024);
|
||||||
|
expect(result.transferCount).toBeGreaterThan(1);
|
||||||
|
expect(result.ticks).toBeGreaterThan(1);
|
||||||
|
expect(csvRequests).toBe(0);
|
||||||
|
console.log(JSON.stringify({ probe: "csv-worker-roundtrip", ...result }));
|
||||||
|
});
|
||||||
|
|
||||||
|
test("CSV rejects bad metadata/data and releases Workers after failure or cancellation", async ({ page }) => {
|
||||||
|
await probePage(page);
|
||||||
|
const result = await page.evaluate(async () => {
|
||||||
|
// @ts-expect-error Vite serves this browser module.
|
||||||
|
const { exportResultCsv, resultCsvFilename } = await import("/src/resultCsvExport.ts");
|
||||||
|
const OriginalWorker = Worker;
|
||||||
|
let created = 0, terminated = 0;
|
||||||
|
window.Worker = class extends OriginalWorker {
|
||||||
|
constructor(url: string | URL, options?: WorkerOptions) { super(url, options); created++; }
|
||||||
|
terminate() { terminated++; super.terminate(); }
|
||||||
|
};
|
||||||
|
const good = () => ({ variables: [{ key: "p" }], series: { time: [0,1], p: [1,2] } });
|
||||||
|
const errors: string[] = [];
|
||||||
|
for (const input of [
|
||||||
|
{ variables: [{ key: "p" }], series: { p: [1] } },
|
||||||
|
{ variables: [], series: { time: [0] } },
|
||||||
|
{ variables: [{ key: "p" }, { key: "p" }], series: { time: [0], p: [1] } },
|
||||||
|
{ variables: [{ key: "p" }], series: { time: [0], extra: [1] } },
|
||||||
|
{ variables: [{ key: "p" }], series: { time: [0], p: [1,2] } },
|
||||||
|
{ variables: [{ key: "p" }], series: { time: [0], p: [Infinity] } },
|
||||||
|
{ variables: [{ key: "p" }], series: { time: [0], p: [null] } },
|
||||||
|
]) {
|
||||||
|
try { await exportResultCsv(input); errors.push("unexpected success"); }
|
||||||
|
catch (error) { errors.push((error as Error).message); }
|
||||||
|
}
|
||||||
|
const aborted = new AbortController(); aborted.abort();
|
||||||
|
const names: string[] = [];
|
||||||
|
try { await exportResultCsv(good(), aborted.signal); } catch (error) { names.push((error as Error).name); }
|
||||||
|
const running = new AbortController();
|
||||||
|
const task = exportResultCsv(good(), running.signal);
|
||||||
|
running.abort();
|
||||||
|
try { await task; } catch (error) { names.push((error as Error).name); }
|
||||||
|
await exportResultCsv(good()); // Failure and abort must not poison a retry.
|
||||||
|
window.Worker = OriginalWorker;
|
||||||
|
return { errors, names, created, terminated,
|
||||||
|
filename: resultCsvFilename(' .储气:/系统. '), unicodeLength: resultCsvFilename("😀".repeat(90)).split("-results")[0].length };
|
||||||
|
});
|
||||||
|
expect(result.errors[0]).toContain("non-empty time");
|
||||||
|
expect(result.errors[1]).toContain("exportable variables");
|
||||||
|
expect(result.errors[2]).toContain("duplicate");
|
||||||
|
expect(result.errors[3]).toContain("missing p; unmapped extra");
|
||||||
|
expect(result.errors[4]).toContain("inconsistent length");
|
||||||
|
expect(result.errors[5]).toContain("non-finite");
|
||||||
|
expect(result.errors[6]).toContain("non-numeric");
|
||||||
|
expect(result.names).toEqual(["AbortError", "AbortError"]);
|
||||||
|
expect(result.created).toBe(4);
|
||||||
|
expect(result.terminated).toBe(result.created);
|
||||||
|
expect(result.filename).toBe("储气__系统-results.csv");
|
||||||
|
expect(result.unicodeLength).toBe(160);
|
||||||
|
});
|
||||||
|
|
||||||
|
async function openResults(page: Page) {
|
||||||
|
await prepareApp(page);
|
||||||
|
await page.addInitScript((snapshot) => {
|
||||||
|
sessionStorage.setItem("system-simulation-flow:latest-result", JSON.stringify(snapshot));
|
||||||
|
}, resultSnapshot);
|
||||||
|
await page.goto("/");
|
||||||
|
await page.getByRole("tab", { name: "结果", exact: true }).click();
|
||||||
|
await expect(page.getByRole("button", { name: "下载结果 CSV", exact: true })).toBeEnabled();
|
||||||
|
}
|
||||||
|
|
||||||
|
test("CSV toolbar downloads current raw data locally without HTTP", async ({ page }) => {
|
||||||
|
await openResults(page);
|
||||||
|
let requests = 0;
|
||||||
|
page.on("request", (request) => { if (request.url().includes("/api/simulation-results/csv")) requests++; });
|
||||||
|
const downloading = page.waitForEvent("download");
|
||||||
|
await page.getByRole("button", { name: "下载结果 CSV", exact: true }).click();
|
||||||
|
const download = await downloading;
|
||||||
|
const stream = await download.createReadStream();
|
||||||
|
const chunks: Buffer[] = [];
|
||||||
|
for await (const chunk of stream!) chunks.push(Buffer.from(chunk));
|
||||||
|
expect(Buffer.concat(chunks).toString("utf8")).toBe("\ufefftime,generic_sensor_1.value\r\n0,0\r\n5,0.5\r\n10,1\r\n");
|
||||||
|
await expect(page.getByRole("button", { name: "下载结果 CSV", exact: true })).toBeEnabled();
|
||||||
|
expect(requests).toBe(0);
|
||||||
|
});
|
||||||
|
|
||||||
|
async function installHeldWorker(page: Page) {
|
||||||
|
await page.evaluate(() => {
|
||||||
|
const original = Worker;
|
||||||
|
(window as any).__csvWorkers = [];
|
||||||
|
window.Worker = function(url: string | URL, options?: WorkerOptions) {
|
||||||
|
if (!String(url).includes("resultCsv.worker")) return new original(url, options);
|
||||||
|
const worker = { onmessage: null, onerror: null, onmessageerror: null, terminated: 0,
|
||||||
|
postMessage() {}, terminate() { this.terminated++; } };
|
||||||
|
(window as any).__csvWorkers.push(worker);
|
||||||
|
return worker;
|
||||||
|
} as any;
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
test("CSV toolbar blocks duplicate clicks and cancels a download when leaving results", async ({ page }) => {
|
||||||
|
await openResults(page);
|
||||||
|
await installHeldWorker(page);
|
||||||
|
const button = page.getByRole("button", { name: "下载结果 CSV", exact: true });
|
||||||
|
let downloads = 0;
|
||||||
|
page.on("download", () => downloads++);
|
||||||
|
await button.evaluate((element: HTMLButtonElement) => { element.click(); element.click(); });
|
||||||
|
await expect(button).toBeDisabled();
|
||||||
|
expect(await page.evaluate(() => (window as any).__csvWorkers.length)).toBe(1);
|
||||||
|
await page.getByRole("tab", { name: "建模", exact: true }).click();
|
||||||
|
expect(await page.evaluate(() => (window as any).__csvWorkers[0].terminated)).toBe(1);
|
||||||
|
await page.evaluate(() => (window as any).__csvWorkers[0].onmessage({ data: { type: "complete", blob: new Blob(["stale"]) } }));
|
||||||
|
expect(downloads).toBe(0);
|
||||||
|
await page.getByRole("tab", { name: "结果", exact: true }).click();
|
||||||
|
await expect(button).toBeEnabled();
|
||||||
|
});
|
||||||
|
|
||||||
|
test("CSV Worker errors release resources and allow the toolbar to retry", async ({ page }) => {
|
||||||
|
await openResults(page);
|
||||||
|
await installHeldWorker(page);
|
||||||
|
const button = page.getByRole("button", { name: "下载结果 CSV", exact: true });
|
||||||
|
await button.click();
|
||||||
|
const dialog = page.waitForEvent("dialog");
|
||||||
|
const fail = page.evaluate(() => (window as any).__csvWorkers[0].onerror({ message: "injected worker error", preventDefault() {} }));
|
||||||
|
const alert = await dialog;
|
||||||
|
expect(alert.message()).toBe("CSV 下载失败:injected worker error");
|
||||||
|
await alert.accept();
|
||||||
|
await fail;
|
||||||
|
await expect(button).toBeEnabled();
|
||||||
|
expect(await page.evaluate(() => (window as any).__csvWorkers[0].terminated)).toBe(1);
|
||||||
|
await button.click();
|
||||||
|
expect(await page.evaluate(() => (window as any).__csvWorkers.length)).toBe(2);
|
||||||
|
});
|
||||||
@@ -0,0 +1,179 @@
|
|||||||
|
import { expect, test, type Page } from "@playwright/test";
|
||||||
|
import { resultSnapshot, prepareApp } from "./fixtures";
|
||||||
|
|
||||||
|
async function probePage(page: Page) {
|
||||||
|
await prepareApp(page);
|
||||||
|
await page.route("**/result-persistence-probe", route => route.fulfill({ contentType: "text/html", body: "<title>Persistence probe</title>" }));
|
||||||
|
await page.goto("/result-persistence-probe");
|
||||||
|
}
|
||||||
|
|
||||||
|
test("eight-branch shaped results save atomically with few writes and restore every number", async ({ page }) => {
|
||||||
|
await probePage(page);
|
||||||
|
const saved = await page.evaluate(async template => {
|
||||||
|
// @ts-expect-error browser-side Vite module
|
||||||
|
const api = await import("/src/resultPersistence.ts");
|
||||||
|
const series: Record<string, number[]> = { emptyFirst: [] };
|
||||||
|
for (let column = 0; column < 1785; column++) {
|
||||||
|
series[column === 0 ? "time" : `output.${column}`] = Array.from({ length: 1002 }, (_, row) => column * 100000 + row + 0.125);
|
||||||
|
}
|
||||||
|
series.emptyLast = [];
|
||||||
|
series["output.17"][31] = -0;
|
||||||
|
const snapshot = { ...template, result: { ...template.result, series } };
|
||||||
|
const originalTransaction = IDBDatabase.prototype.transaction;
|
||||||
|
const originalPut = IDBObjectStore.prototype.put;
|
||||||
|
let writes = 0;
|
||||||
|
let transactions = 0;
|
||||||
|
let pointerBeforeCommit: string | null | undefined;
|
||||||
|
let pointerAtCommit: string | null | undefined;
|
||||||
|
IDBDatabase.prototype.transaction = function (...args: Parameters<typeof originalTransaction>) {
|
||||||
|
const transaction = originalTransaction.apply(this, args);
|
||||||
|
if (this.name === "system-simulation-results" && transaction.mode === "readwrite") {
|
||||||
|
transactions++;
|
||||||
|
pointerBeforeCommit = sessionStorage.getItem(api.RESULT_SNAPSHOT_KEY);
|
||||||
|
transaction.addEventListener("complete", () => { pointerAtCommit = sessionStorage.getItem(api.RESULT_SNAPSHOT_KEY); });
|
||||||
|
}
|
||||||
|
return transaction;
|
||||||
|
};
|
||||||
|
IDBObjectStore.prototype.put = function (...args: Parameters<typeof originalPut>) {
|
||||||
|
writes++;
|
||||||
|
return originalPut.apply(this, args);
|
||||||
|
};
|
||||||
|
const start = performance.now();
|
||||||
|
await api.storeResultSnapshot(snapshot);
|
||||||
|
const saveMs = performance.now() - start;
|
||||||
|
IDBDatabase.prototype.transaction = originalTransaction;
|
||||||
|
IDBObjectStore.prototype.put = originalPut;
|
||||||
|
const readStart = performance.now();
|
||||||
|
const restored = await api.loadStoredResultSnapshot();
|
||||||
|
return { writes, transactions, saveMs, readMs: performance.now() - readStart,
|
||||||
|
pointerBeforeCommit, pointerAtCommit, hasPointer: Boolean(sessionStorage.getItem(api.RESULT_SNAPSHOT_KEY)),
|
||||||
|
pending: api.resultSavePending(),
|
||||||
|
exact: Object.entries(series).every(([key, values]) => values.length === restored.result.series[key].length && values.every((value, i) => Object.is(value, restored.result.series[key][i]))) };
|
||||||
|
}, resultSnapshot);
|
||||||
|
expect(saved).toMatchObject({ transactions: 1, pointerBeforeCommit: null, pointerAtCommit: null, hasPointer: true, pending: false, exact: true });
|
||||||
|
// Thousands of short output columns must not cause thousands of writes or
|
||||||
|
// one transaction per small batch; the workload bound is independent of packing size.
|
||||||
|
expect(saved.writes).toBeLessThan(16);
|
||||||
|
console.log(JSON.stringify({ probe: "eight-branch-packed-persistence", ...saved }));
|
||||||
|
await page.reload();
|
||||||
|
const restored = await page.evaluate(async () => {
|
||||||
|
// @ts-expect-error browser-side Vite module
|
||||||
|
const { loadStoredResultSnapshot } = await import("/src/resultPersistence.ts");
|
||||||
|
const snapshot = await loadStoredResultSnapshot();
|
||||||
|
return { columns: Object.keys(snapshot.result.series).length, last: snapshot.result.series["output.1784"][1001], empty: snapshot.result.series.emptyFirst };
|
||||||
|
});
|
||||||
|
expect(restored).toEqual({ columns: 1787, last: 178401001.125, empty: [] });
|
||||||
|
});
|
||||||
|
|
||||||
|
test("legacy session snapshots and old column chunks remain readable; gaps and overlaps fail", async ({ page }) => {
|
||||||
|
await probePage(page);
|
||||||
|
const result = await page.evaluate(async template => {
|
||||||
|
// @ts-expect-error browser-side Vite module
|
||||||
|
const api = await import("/src/resultPersistence.ts");
|
||||||
|
sessionStorage.setItem(api.RESULT_SNAPSHOT_KEY, JSON.stringify(template));
|
||||||
|
const sessionLegacy = (await api.loadStoredResultSnapshot()).id === template.id;
|
||||||
|
// Create legacy records independently of the new writer and its format.
|
||||||
|
const db = await new Promise<IDBDatabase>((resolve, reject) => {
|
||||||
|
const request = indexedDB.open("system-simulation-results", 1);
|
||||||
|
request.onupgradeneeded = () => { request.result.createObjectStore("headers"); request.result.createObjectStore("chunks"); };
|
||||||
|
request.onsuccess = () => resolve(request.result); request.onerror = () => reject(request.error);
|
||||||
|
});
|
||||||
|
async function legacy(chunks: [number, number[]][]) {
|
||||||
|
await new Promise<void>((resolve, reject) => {
|
||||||
|
const tx = db.transaction(["headers", "chunks"], "readwrite");
|
||||||
|
tx.oncomplete = () => resolve(); tx.onabort = () => reject(tx.error);
|
||||||
|
tx.objectStore("headers").put({ snapshot: { ...template, result: { ...template.result, series: {} } }, lengths: { time: 6, empty: 0 } }, "legacy");
|
||||||
|
tx.objectStore("chunks").clear();
|
||||||
|
for (const [offset, values] of chunks) tx.objectStore("chunks").put(new Float64Array(values), ["legacy", "time", offset]);
|
||||||
|
});
|
||||||
|
sessionStorage.setItem(api.RESULT_SNAPSHOT_KEY, JSON.stringify({ storage: "indexeddb", version: 1, cacheId: "legacy" }));
|
||||||
|
try { return { values: (await api.loadStoredResultSnapshot()).result.series.time, rejected: false }; }
|
||||||
|
catch { return { rejected: true }; }
|
||||||
|
}
|
||||||
|
const complete = await legacy([[0, [1, 2, 3]], [3, [4, 5, 6]]]);
|
||||||
|
const gap = await legacy([[0, [1, 2]], [3, [4, 5, 6]]]);
|
||||||
|
const overlap = await legacy([[0, [1, 2, 3, 4]], [2, [3, 4]]]);
|
||||||
|
db.close();
|
||||||
|
return { sessionLegacy, complete, gap, overlap };
|
||||||
|
}, resultSnapshot);
|
||||||
|
expect(result).toEqual({ sessionLegacy: true, complete: { values: [1, 2, 3, 4, 5, 6], rejected: false }, gap: { rejected: true }, overlap: { rejected: true } });
|
||||||
|
});
|
||||||
|
|
||||||
|
test("in-flight superseded commits and failed writes cannot replace the latest complete pointer", async ({ page }) => {
|
||||||
|
await probePage(page);
|
||||||
|
const result = await page.evaluate(async template => {
|
||||||
|
// @ts-expect-error browser-side Vite module
|
||||||
|
const api = await import("/src/resultPersistence.ts");
|
||||||
|
await api.storeResultSnapshot(template);
|
||||||
|
const stablePointer = sessionStorage.getItem(api.RESULT_SNAPSHOT_KEY);
|
||||||
|
const originalPut = IDBObjectStore.prototype.put;
|
||||||
|
let puts = 0;
|
||||||
|
IDBObjectStore.prototype.put = function (...args: Parameters<typeof originalPut>) {
|
||||||
|
if (++puts === 2) throw new DOMException("Injected header failure after data write", "QuotaExceededError");
|
||||||
|
return originalPut.apply(this, args);
|
||||||
|
};
|
||||||
|
let failed = false;
|
||||||
|
try { await api.storeResultSnapshot({ ...template, id: "failed" }); } catch { failed = true; }
|
||||||
|
IDBObjectStore.prototype.put = originalPut;
|
||||||
|
const preserved = sessionStorage.getItem(api.RESULT_SNAPSHOT_KEY) === stablePointer;
|
||||||
|
const originalTransaction = IDBDatabase.prototype.transaction;
|
||||||
|
let latest: Promise<boolean> | undefined;
|
||||||
|
let armed = true;
|
||||||
|
IDBDatabase.prototype.transaction = function (...args: Parameters<typeof originalTransaction>) {
|
||||||
|
const transaction = originalTransaction.apply(this, args);
|
||||||
|
if (armed && this.name === "system-simulation-results" && transaction.mode === "readwrite") {
|
||||||
|
armed = false;
|
||||||
|
// Replace an older save after its transaction starts but before commit.
|
||||||
|
queueMicrotask(() => { latest = api.storeResultSnapshot({ ...template, id: "newest" }); });
|
||||||
|
}
|
||||||
|
return transaction;
|
||||||
|
};
|
||||||
|
const old = await api.storeResultSnapshot({ ...template, id: "older" });
|
||||||
|
const newer = await latest;
|
||||||
|
IDBDatabase.prototype.transaction = originalTransaction;
|
||||||
|
const restored = await api.loadStoredResultSnapshot();
|
||||||
|
return { failed, preserved, old, newer, id: restored.id, pending: api.resultSavePending() };
|
||||||
|
}, resultSnapshot);
|
||||||
|
expect(result).toEqual({ failed: true, preserved: true, old: false, newer: true, id: "newest", pending: false });
|
||||||
|
});
|
||||||
|
|
||||||
|
test("saving in another page preserves its borrowed cache and corrupt packed records are rejected", async ({ page, context }) => {
|
||||||
|
await probePage(page);
|
||||||
|
const pointer = await page.evaluate(async template => {
|
||||||
|
// @ts-expect-error browser-side Vite module
|
||||||
|
const api = await import("/src/resultPersistence.ts");
|
||||||
|
await api.storeResultSnapshot(template);
|
||||||
|
return sessionStorage.getItem(api.RESULT_SNAPSHOT_KEY)!;
|
||||||
|
}, resultSnapshot);
|
||||||
|
const other = await context.newPage();
|
||||||
|
await probePage(other);
|
||||||
|
await other.evaluate(async ({ template, pointer }) => {
|
||||||
|
// @ts-expect-error browser-side Vite module
|
||||||
|
const api = await import("/src/resultPersistence.ts");
|
||||||
|
sessionStorage.setItem(api.RESULT_SNAPSHOT_KEY, pointer);
|
||||||
|
await api.loadStoredResultSnapshot();
|
||||||
|
await api.storeResultSnapshot({ ...template, id: "other-page" });
|
||||||
|
}, { template: resultSnapshot, pointer });
|
||||||
|
const outcome = await page.evaluate(async () => {
|
||||||
|
// @ts-expect-error browser-side Vite module
|
||||||
|
const api = await import("/src/resultPersistence.ts");
|
||||||
|
const before = await api.loadStoredResultSnapshot();
|
||||||
|
const marker = JSON.parse(sessionStorage.getItem(api.RESULT_SNAPSHOT_KEY)!);
|
||||||
|
const db = await new Promise<IDBDatabase>((resolve, reject) => {
|
||||||
|
const request = indexedDB.open("system-simulation-results", 1);
|
||||||
|
request.onsuccess = () => resolve(request.result); request.onerror = () => reject(request.error);
|
||||||
|
});
|
||||||
|
await new Promise<void>((resolve, reject) => {
|
||||||
|
const tx = db.transaction("chunks", "readwrite");
|
||||||
|
tx.oncomplete = () => resolve(); tx.onabort = () => reject(tx.error);
|
||||||
|
const cursor = tx.objectStore("chunks").openCursor(IDBKeyRange.bound([marker.cacheId], [marker.cacheId, []]));
|
||||||
|
cursor.onsuccess = () => { const entry = cursor.result; if (entry) entry.update({ ...entry.value, offset: 1 }); };
|
||||||
|
});
|
||||||
|
db.close();
|
||||||
|
let rejected = false;
|
||||||
|
try { await api.loadStoredResultSnapshot(); } catch { rejected = true; }
|
||||||
|
return { originalId: before.id, rejected };
|
||||||
|
});
|
||||||
|
expect(outcome).toEqual({ originalId: resultSnapshot.id, rejected: true });
|
||||||
|
await other.close();
|
||||||
|
});
|
||||||
@@ -87,3 +87,11 @@ EXE 不需要 Python、SciPy、XML 或原工程文件。DLL 需要与 EXE 一同
|
|||||||
## AME 对齐后的工程输入
|
## AME 对齐后的工程输入
|
||||||
|
|
||||||
当前用户运行请使用 `tests/data/test-mql-4-corrected.json` 或 `tests/data/test-mql-8-corrected.json`。四路执行 XML 位于 `tests/fixtures/amesim/`,与其 JSON 同步;原八路 JSON 已移入 `tests/fixtures/legacy/` 供差异审计。`test-mql-4-amesim-reference.json` 是曲线基准,不是工程输入。模型复核、P4 端口图形修正、默认 `rtol=1e-8` 的验收和速度记录见 [本轮验证报告](../docs/other/牛顿管流求根与四路网页验证-2026-09-11.md)。
|
当前用户运行请使用 `tests/data/test-mql-4-corrected.json` 或 `tests/data/test-mql-8-corrected.json`。四路执行 XML 位于 `tests/fixtures/amesim/`,与其 JSON 同步;原八路 JSON 已移入 `tests/fixtures/legacy/` 供差异审计。`test-mql-4-amesim-reference.json` 是曲线基准,不是工程输入。模型复核、P4 端口图形修正、默认 `rtol=1e-8` 的验收和速度记录见 [本轮验证报告](../docs/other/牛顿管流求根与四路网页验证-2026-09-11.md)。
|
||||||
|
|
||||||
|
后续优化和网页计时默认使用修正后的八路工程;仅当八路无法运行且短期不能解决时退回四路。选择规则见 [优化验证约定](../docs/standard/optimization-benchmark-model.md),当前效率、迭代触顶与网页分阶段记录见 [八路评估报告](../docs/other/八路模型计算效率与网页阶段计时-2026-09-11.md)。
|
||||||
|
|
||||||
|
## 结果传输与网页后处理
|
||||||
|
|
||||||
|
网页流式请求使用 C 的可选 `--result-index` 字节索引,Python 只解析小元数据,将原有 JSON 的 `series` 原样放入 NDJSON 结果事件;任务结果查询同样支持原样返回。独立 CLI 与默认同步调用仍返回普通 JSON/结果对象,JSON结构、积分和采样保持一致。前端结果保存改为打包 Float64 块的原子事务,CSV 由工作线程本地生成,兼容旧缓存及旧 HTTP CSV 接口。完整结果、取消与刷新回归及八路前后计时见 [结果处理优化报告](../docs/other/八路结果处理与网页保存优化-2026-09-11.md)。
|
||||||
|
|
||||||
|
C结果的 `series/final/finalState` 现使用固定版本Ryu binary64编码及64 KiB批量写出;在精确回读前提下选择更短的普通/科学token,负零保留为 `-0.0`,非有限值或写出失败阻止索引发布。数字文本允许变化,索引元数据保持整数。嵌套Ryu头文件参与缓存哈希;源码和Boost许可随项目保留。本轮研究、逐位验证和八路网页前后计时见 [C结果编码优化报告](../docs/other/C端结果编码与写出优化-2026-09-11.md)。
|
||||||
@@ -87,3 +87,33 @@ The views and opinions of authors expressed herein do not necessarily
|
|||||||
state or reflect those of the United States Government or Lawrence
|
state or reflect those of the United States Government or Lawrence
|
||||||
Livermore National Security, LLC, and shall not be used for advertising
|
Livermore National Security, LLC, and shall not be used for advertising
|
||||||
or product endorsement purposes.
|
or product endorsement purposes.
|
||||||
|
|
||||||
|
Ryu binary64 shortest conversion
|
||||||
|
Copyright 2018 Ulf Adams
|
||||||
|
Upstream: https://github.com/ulfjack/ryu
|
||||||
|
Commit: 4c0618b0e44f7ef027ebae05d2cc7812048f7c8f
|
||||||
|
The vendored ryu files are used under the following license option:
|
||||||
|
|
||||||
|
Boost Software License - Version 1.0 - August 17th, 2003
|
||||||
|
|
||||||
|
Permission is hereby granted, free of charge, to any person or organization
|
||||||
|
obtaining a copy of the software and accompanying documentation covered by
|
||||||
|
this license (the "Software") to use, reproduce, display, distribute,
|
||||||
|
execute, and transmit the Software, and to prepare derivative works of the
|
||||||
|
Software, and to permit third-parties to whom the Software is furnished to
|
||||||
|
do so, all subject to the following:
|
||||||
|
|
||||||
|
The copyright notices in the Software and this entire statement, including
|
||||||
|
the above license grant, this restriction and the following disclaimer,
|
||||||
|
must be included in all copies of the Software, in whole or in part, and
|
||||||
|
all derivative works of the Software, unless such copies or derivative
|
||||||
|
works are solely in the form of machine-executable object code generated by
|
||||||
|
a source language processor.
|
||||||
|
|
||||||
|
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||||
|
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||||
|
FITNESS FOR A PARTICULAR PURPOSE, TITLE AND NON-INFRINGEMENT. IN NO EVENT
|
||||||
|
SHALL THE COPYRIGHT HOLDERS OR ANYONE DISTRIBUTING THE SOFTWARE BE LIABLE
|
||||||
|
FOR ANY DAMAGES OR OTHER LIABILITY, WHETHER IN CONTRACT, TORT OR OTHERWISE,
|
||||||
|
ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
|
||||||
|
DEALINGS IN THE SOFTWARE.
|
||||||
@@ -0,0 +1,23 @@
|
|||||||
|
Boost Software License - Version 1.0 - August 17th, 2003
|
||||||
|
|
||||||
|
Permission is hereby granted, free of charge, to any person or organization
|
||||||
|
obtaining a copy of the software and accompanying documentation covered by
|
||||||
|
this license (the "Software") to use, reproduce, display, distribute,
|
||||||
|
execute, and transmit the Software, and to prepare derivative works of the
|
||||||
|
Software, and to permit third-parties to whom the Software is furnished to
|
||||||
|
do so, all subject to the following:
|
||||||
|
|
||||||
|
The copyright notices in the Software and this entire statement, including
|
||||||
|
the above license grant, this restriction and the following disclaimer,
|
||||||
|
must be included in all copies of the Software, in whole or in part, and
|
||||||
|
all derivative works of the Software, unless such copies or derivative
|
||||||
|
works are solely in the form of machine-executable object code generated by
|
||||||
|
a source language processor.
|
||||||
|
|
||||||
|
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||||
|
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||||
|
FITNESS FOR A PARTICULAR PURPOSE, TITLE AND NON-INFRINGEMENT. IN NO EVENT
|
||||||
|
SHALL THE COPYRIGHT HOLDERS OR ANYONE DISTRIBUTING THE SOFTWARE BE LIABLE
|
||||||
|
FOR ANY DAMAGES OR OTHER LIABILITY, WHETHER IN CONTRACT, TORT OR OTHERWISE,
|
||||||
|
ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
|
||||||
|
DEALINGS IN THE SOFTWARE.
|
||||||
@@ -0,0 +1,58 @@
|
|||||||
|
{
|
||||||
|
"repository": "https://github.com/ulfjack/ryu",
|
||||||
|
"commit": "4c0618b0e44f7ef027ebae05d2cc7812048f7c8f",
|
||||||
|
"files": {
|
||||||
|
"ryu/d2s.c": {
|
||||||
|
"sha256": "d24323c7eb77d63f1e52c415212b50060b776f0883525d907d04954fcd48cf64",
|
||||||
|
"bytes": 17705,
|
||||||
|
"url": "https://raw.githubusercontent.com/ulfjack/ryu/4c0618b0e44f7ef027ebae05d2cc7812048f7c8f/ryu/d2s.c"
|
||||||
|
},
|
||||||
|
"ryu/ryu.h": {
|
||||||
|
"sha256": "b7feab0ba1df5e9ef3d602f386592ad152491405e49b408fb21c0e0e9e6bfb16",
|
||||||
|
"bytes": 1360,
|
||||||
|
"url": "https://raw.githubusercontent.com/ulfjack/ryu/4c0618b0e44f7ef027ebae05d2cc7812048f7c8f/ryu/ryu.h"
|
||||||
|
},
|
||||||
|
"ryu/common.h": {
|
||||||
|
"sha256": "0bbd71d26da6193e678d0776cf418f43f287c73d6fd6725353df0aadf70f2a19",
|
||||||
|
"bytes": 3648,
|
||||||
|
"url": "https://raw.githubusercontent.com/ulfjack/ryu/4c0618b0e44f7ef027ebae05d2cc7812048f7c8f/ryu/common.h"
|
||||||
|
},
|
||||||
|
"ryu/digit_table.h": {
|
||||||
|
"sha256": "8b782573abc0b8554d74163ae6c02f0beb5c30d1e63eaebdb0ddf2c98d817e01",
|
||||||
|
"bytes": 1745,
|
||||||
|
"url": "https://raw.githubusercontent.com/ulfjack/ryu/4c0618b0e44f7ef027ebae05d2cc7812048f7c8f/ryu/digit_table.h"
|
||||||
|
},
|
||||||
|
"ryu/d2s_intrinsics.h": {
|
||||||
|
"sha256": "1d05702f2edacce428223d4b43dd3095c1bd1f3ad30128ce1761d84356dddc7d",
|
||||||
|
"bytes": 13015,
|
||||||
|
"url": "https://raw.githubusercontent.com/ulfjack/ryu/4c0618b0e44f7ef027ebae05d2cc7812048f7c8f/ryu/d2s_intrinsics.h"
|
||||||
|
},
|
||||||
|
"ryu/d2s_full_table.h": {
|
||||||
|
"sha256": "2618f6e5fae6c4443899b184efe3d08295dd267dc9f1a994c983c7caca59ebe6",
|
||||||
|
"bytes": 34501,
|
||||||
|
"url": "https://raw.githubusercontent.com/ulfjack/ryu/4c0618b0e44f7ef027ebae05d2cc7812048f7c8f/ryu/d2s_full_table.h"
|
||||||
|
},
|
||||||
|
"ryu/d2s_small_table.h": {
|
||||||
|
"sha256": "54fec51f1c5eed786a8ce6de881b81582dd71f0812dcd4324b37396617bb0fa9",
|
||||||
|
"bytes": 7641,
|
||||||
|
"url": "https://raw.githubusercontent.com/ulfjack/ryu/4c0618b0e44f7ef027ebae05d2cc7812048f7c8f/ryu/d2s_small_table.h"
|
||||||
|
},
|
||||||
|
"LICENSE-Boost": {
|
||||||
|
"sha256": "c9bff75738922193e67fa726fa225535870d2aa1059f91452c411736284ad566",
|
||||||
|
"bytes": 1338,
|
||||||
|
"url": "https://raw.githubusercontent.com/ulfjack/ryu/4c0618b0e44f7ef027ebae05d2cc7812048f7c8f/LICENSE-Boost"
|
||||||
|
},
|
||||||
|
"LICENSE-Apache2": {
|
||||||
|
"sha256": "c71d239df91726fc519c6eb72d318ec65820627232b2f796219e87dcf35d0ab4",
|
||||||
|
"bytes": 11357,
|
||||||
|
"url": "https://raw.githubusercontent.com/ulfjack/ryu/4c0618b0e44f7ef027ebae05d2cc7812048f7c8f/LICENSE-Apache2"
|
||||||
|
},
|
||||||
|
"README.md": {
|
||||||
|
"sha256": "20572708ea26d19fd596f1c4a44097dbcb491cdf0517024d68bbb5d622d0ac93",
|
||||||
|
"bytes": 14930,
|
||||||
|
"url": "https://raw.githubusercontent.com/ulfjack/ryu/4c0618b0e44f7ef027ebae05d2cc7812048f7c8f/README.md"
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"licenseSelected": "Boost Software License 1.0",
|
||||||
|
"layout": "Upstream ryu/*.h in native/include/ryu; unchanged ryu/d2s.c in native/encoding/ryu; no source edits."
|
||||||
|
}
|
||||||
@@ -0,0 +1,509 @@
|
|||||||
|
// Copyright 2018 Ulf Adams
|
||||||
|
//
|
||||||
|
// The contents of this file may be used under the terms of the Apache License,
|
||||||
|
// Version 2.0.
|
||||||
|
//
|
||||||
|
// (See accompanying file LICENSE-Apache or copy at
|
||||||
|
// http://www.apache.org/licenses/LICENSE-2.0)
|
||||||
|
//
|
||||||
|
// Alternatively, the contents of this file may be used under the terms of
|
||||||
|
// the Boost Software License, Version 1.0.
|
||||||
|
// (See accompanying file LICENSE-Boost or copy at
|
||||||
|
// https://www.boost.org/LICENSE_1_0.txt)
|
||||||
|
//
|
||||||
|
// Unless required by applicable law or agreed to in writing, this software
|
||||||
|
// is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
|
||||||
|
// KIND, either express or implied.
|
||||||
|
|
||||||
|
// Runtime compiler options:
|
||||||
|
// -DRYU_DEBUG Generate verbose debugging output to stdout.
|
||||||
|
//
|
||||||
|
// -DRYU_ONLY_64_BIT_OPS Avoid using uint128_t or 64-bit intrinsics. Slower,
|
||||||
|
// depending on your compiler.
|
||||||
|
//
|
||||||
|
// -DRYU_OPTIMIZE_SIZE Use smaller lookup tables. Instead of storing every
|
||||||
|
// required power of 5, only store every 26th entry, and compute
|
||||||
|
// intermediate values with a multiplication. This reduces the lookup table
|
||||||
|
// size by about 10x (only one case, and only double) at the cost of some
|
||||||
|
// performance. Currently requires MSVC intrinsics.
|
||||||
|
|
||||||
|
#include "ryu/ryu.h"
|
||||||
|
|
||||||
|
#include <assert.h>
|
||||||
|
#include <stdbool.h>
|
||||||
|
#include <stdint.h>
|
||||||
|
#include <stdlib.h>
|
||||||
|
#include <string.h>
|
||||||
|
|
||||||
|
#ifdef RYU_DEBUG
|
||||||
|
#include <inttypes.h>
|
||||||
|
#include <stdio.h>
|
||||||
|
#endif
|
||||||
|
|
||||||
|
#include "ryu/common.h"
|
||||||
|
#include "ryu/digit_table.h"
|
||||||
|
#include "ryu/d2s_intrinsics.h"
|
||||||
|
|
||||||
|
// Include either the small or the full lookup tables depending on the mode.
|
||||||
|
#if defined(RYU_OPTIMIZE_SIZE)
|
||||||
|
#include "ryu/d2s_small_table.h"
|
||||||
|
#else
|
||||||
|
#include "ryu/d2s_full_table.h"
|
||||||
|
#endif
|
||||||
|
|
||||||
|
#define DOUBLE_MANTISSA_BITS 52
|
||||||
|
#define DOUBLE_EXPONENT_BITS 11
|
||||||
|
#define DOUBLE_BIAS 1023
|
||||||
|
|
||||||
|
static inline uint32_t decimalLength17(const uint64_t v) {
|
||||||
|
// This is slightly faster than a loop.
|
||||||
|
// The average output length is 16.38 digits, so we check high-to-low.
|
||||||
|
// Function precondition: v is not an 18, 19, or 20-digit number.
|
||||||
|
// (17 digits are sufficient for round-tripping.)
|
||||||
|
assert(v < 100000000000000000L);
|
||||||
|
if (v >= 10000000000000000L) { return 17; }
|
||||||
|
if (v >= 1000000000000000L) { return 16; }
|
||||||
|
if (v >= 100000000000000L) { return 15; }
|
||||||
|
if (v >= 10000000000000L) { return 14; }
|
||||||
|
if (v >= 1000000000000L) { return 13; }
|
||||||
|
if (v >= 100000000000L) { return 12; }
|
||||||
|
if (v >= 10000000000L) { return 11; }
|
||||||
|
if (v >= 1000000000L) { return 10; }
|
||||||
|
if (v >= 100000000L) { return 9; }
|
||||||
|
if (v >= 10000000L) { return 8; }
|
||||||
|
if (v >= 1000000L) { return 7; }
|
||||||
|
if (v >= 100000L) { return 6; }
|
||||||
|
if (v >= 10000L) { return 5; }
|
||||||
|
if (v >= 1000L) { return 4; }
|
||||||
|
if (v >= 100L) { return 3; }
|
||||||
|
if (v >= 10L) { return 2; }
|
||||||
|
return 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
// A floating decimal representing m * 10^e.
|
||||||
|
typedef struct floating_decimal_64 {
|
||||||
|
uint64_t mantissa;
|
||||||
|
// Decimal exponent's range is -324 to 308
|
||||||
|
// inclusive, and can fit in a short if needed.
|
||||||
|
int32_t exponent;
|
||||||
|
} floating_decimal_64;
|
||||||
|
|
||||||
|
static inline floating_decimal_64 d2d(const uint64_t ieeeMantissa, const uint32_t ieeeExponent) {
|
||||||
|
int32_t e2;
|
||||||
|
uint64_t m2;
|
||||||
|
if (ieeeExponent == 0) {
|
||||||
|
// We subtract 2 so that the bounds computation has 2 additional bits.
|
||||||
|
e2 = 1 - DOUBLE_BIAS - DOUBLE_MANTISSA_BITS - 2;
|
||||||
|
m2 = ieeeMantissa;
|
||||||
|
} else {
|
||||||
|
e2 = (int32_t) ieeeExponent - DOUBLE_BIAS - DOUBLE_MANTISSA_BITS - 2;
|
||||||
|
m2 = (1ull << DOUBLE_MANTISSA_BITS) | ieeeMantissa;
|
||||||
|
}
|
||||||
|
const bool even = (m2 & 1) == 0;
|
||||||
|
const bool acceptBounds = even;
|
||||||
|
|
||||||
|
#ifdef RYU_DEBUG
|
||||||
|
printf("-> %" PRIu64 " * 2^%d\n", m2, e2 + 2);
|
||||||
|
#endif
|
||||||
|
|
||||||
|
// Step 2: Determine the interval of valid decimal representations.
|
||||||
|
const uint64_t mv = 4 * m2;
|
||||||
|
// Implicit bool -> int conversion. True is 1, false is 0.
|
||||||
|
const uint32_t mmShift = ieeeMantissa != 0 || ieeeExponent <= 1;
|
||||||
|
// We would compute mp and mm like this:
|
||||||
|
// uint64_t mp = 4 * m2 + 2;
|
||||||
|
// uint64_t mm = mv - 1 - mmShift;
|
||||||
|
|
||||||
|
// Step 3: Convert to a decimal power base using 128-bit arithmetic.
|
||||||
|
uint64_t vr, vp, vm;
|
||||||
|
int32_t e10;
|
||||||
|
bool vmIsTrailingZeros = false;
|
||||||
|
bool vrIsTrailingZeros = false;
|
||||||
|
if (e2 >= 0) {
|
||||||
|
// I tried special-casing q == 0, but there was no effect on performance.
|
||||||
|
// This expression is slightly faster than max(0, log10Pow2(e2) - 1).
|
||||||
|
const uint32_t q = log10Pow2(e2) - (e2 > 3);
|
||||||
|
e10 = (int32_t) q;
|
||||||
|
const int32_t k = DOUBLE_POW5_INV_BITCOUNT + pow5bits((int32_t) q) - 1;
|
||||||
|
const int32_t i = -e2 + (int32_t) q + k;
|
||||||
|
#if defined(RYU_OPTIMIZE_SIZE)
|
||||||
|
uint64_t pow5[2];
|
||||||
|
double_computeInvPow5(q, pow5);
|
||||||
|
vr = mulShiftAll64(m2, pow5, i, &vp, &vm, mmShift);
|
||||||
|
#else
|
||||||
|
vr = mulShiftAll64(m2, DOUBLE_POW5_INV_SPLIT[q], i, &vp, &vm, mmShift);
|
||||||
|
#endif
|
||||||
|
#ifdef RYU_DEBUG
|
||||||
|
printf("%" PRIu64 " * 2^%d / 10^%u\n", mv, e2, q);
|
||||||
|
printf("V+=%" PRIu64 "\nV =%" PRIu64 "\nV-=%" PRIu64 "\n", vp, vr, vm);
|
||||||
|
#endif
|
||||||
|
if (q <= 21) {
|
||||||
|
// This should use q <= 22, but I think 21 is also safe. Smaller values
|
||||||
|
// may still be safe, but it's more difficult to reason about them.
|
||||||
|
// Only one of mp, mv, and mm can be a multiple of 5, if any.
|
||||||
|
const uint32_t mvMod5 = ((uint32_t) mv) - 5 * ((uint32_t) div5(mv));
|
||||||
|
if (mvMod5 == 0) {
|
||||||
|
vrIsTrailingZeros = multipleOfPowerOf5(mv, q);
|
||||||
|
} else if (acceptBounds) {
|
||||||
|
// Same as min(e2 + (~mm & 1), pow5Factor(mm)) >= q
|
||||||
|
// <=> e2 + (~mm & 1) >= q && pow5Factor(mm) >= q
|
||||||
|
// <=> true && pow5Factor(mm) >= q, since e2 >= q.
|
||||||
|
vmIsTrailingZeros = multipleOfPowerOf5(mv - 1 - mmShift, q);
|
||||||
|
} else {
|
||||||
|
// Same as min(e2 + 1, pow5Factor(mp)) >= q.
|
||||||
|
vp -= multipleOfPowerOf5(mv + 2, q);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
// This expression is slightly faster than max(0, log10Pow5(-e2) - 1).
|
||||||
|
const uint32_t q = log10Pow5(-e2) - (-e2 > 1);
|
||||||
|
e10 = (int32_t) q + e2;
|
||||||
|
const int32_t i = -e2 - (int32_t) q;
|
||||||
|
const int32_t k = pow5bits(i) - DOUBLE_POW5_BITCOUNT;
|
||||||
|
const int32_t j = (int32_t) q - k;
|
||||||
|
#if defined(RYU_OPTIMIZE_SIZE)
|
||||||
|
uint64_t pow5[2];
|
||||||
|
double_computePow5(i, pow5);
|
||||||
|
vr = mulShiftAll64(m2, pow5, j, &vp, &vm, mmShift);
|
||||||
|
#else
|
||||||
|
vr = mulShiftAll64(m2, DOUBLE_POW5_SPLIT[i], j, &vp, &vm, mmShift);
|
||||||
|
#endif
|
||||||
|
#ifdef RYU_DEBUG
|
||||||
|
printf("%" PRIu64 " * 5^%d / 10^%u\n", mv, -e2, q);
|
||||||
|
printf("%u %d %d %d\n", q, i, k, j);
|
||||||
|
printf("V+=%" PRIu64 "\nV =%" PRIu64 "\nV-=%" PRIu64 "\n", vp, vr, vm);
|
||||||
|
#endif
|
||||||
|
if (q <= 1) {
|
||||||
|
// {vr,vp,vm} is trailing zeros if {mv,mp,mm} has at least q trailing 0 bits.
|
||||||
|
// mv = 4 * m2, so it always has at least two trailing 0 bits.
|
||||||
|
vrIsTrailingZeros = true;
|
||||||
|
if (acceptBounds) {
|
||||||
|
// mm = mv - 1 - mmShift, so it has 1 trailing 0 bit iff mmShift == 1.
|
||||||
|
vmIsTrailingZeros = mmShift == 1;
|
||||||
|
} else {
|
||||||
|
// mp = mv + 2, so it always has at least one trailing 0 bit.
|
||||||
|
--vp;
|
||||||
|
}
|
||||||
|
} else if (q < 63) { // TODO(ulfjack): Use a tighter bound here.
|
||||||
|
// We want to know if the full product has at least q trailing zeros.
|
||||||
|
// We need to compute min(p2(mv), p5(mv) - e2) >= q
|
||||||
|
// <=> p2(mv) >= q && p5(mv) - e2 >= q
|
||||||
|
// <=> p2(mv) >= q (because -e2 >= q)
|
||||||
|
vrIsTrailingZeros = multipleOfPowerOf2(mv, q);
|
||||||
|
#ifdef RYU_DEBUG
|
||||||
|
printf("vr is trailing zeros=%s\n", vrIsTrailingZeros ? "true" : "false");
|
||||||
|
#endif
|
||||||
|
}
|
||||||
|
}
|
||||||
|
#ifdef RYU_DEBUG
|
||||||
|
printf("e10=%d\n", e10);
|
||||||
|
printf("V+=%" PRIu64 "\nV =%" PRIu64 "\nV-=%" PRIu64 "\n", vp, vr, vm);
|
||||||
|
printf("vm is trailing zeros=%s\n", vmIsTrailingZeros ? "true" : "false");
|
||||||
|
printf("vr is trailing zeros=%s\n", vrIsTrailingZeros ? "true" : "false");
|
||||||
|
#endif
|
||||||
|
|
||||||
|
// Step 4: Find the shortest decimal representation in the interval of valid representations.
|
||||||
|
int32_t removed = 0;
|
||||||
|
uint8_t lastRemovedDigit = 0;
|
||||||
|
uint64_t output;
|
||||||
|
// On average, we remove ~2 digits.
|
||||||
|
if (vmIsTrailingZeros || vrIsTrailingZeros) {
|
||||||
|
// General case, which happens rarely (~0.7%).
|
||||||
|
for (;;) {
|
||||||
|
const uint64_t vpDiv10 = div10(vp);
|
||||||
|
const uint64_t vmDiv10 = div10(vm);
|
||||||
|
if (vpDiv10 <= vmDiv10) {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
const uint32_t vmMod10 = ((uint32_t) vm) - 10 * ((uint32_t) vmDiv10);
|
||||||
|
const uint64_t vrDiv10 = div10(vr);
|
||||||
|
const uint32_t vrMod10 = ((uint32_t) vr) - 10 * ((uint32_t) vrDiv10);
|
||||||
|
vmIsTrailingZeros &= vmMod10 == 0;
|
||||||
|
vrIsTrailingZeros &= lastRemovedDigit == 0;
|
||||||
|
lastRemovedDigit = (uint8_t) vrMod10;
|
||||||
|
vr = vrDiv10;
|
||||||
|
vp = vpDiv10;
|
||||||
|
vm = vmDiv10;
|
||||||
|
++removed;
|
||||||
|
}
|
||||||
|
#ifdef RYU_DEBUG
|
||||||
|
printf("V+=%" PRIu64 "\nV =%" PRIu64 "\nV-=%" PRIu64 "\n", vp, vr, vm);
|
||||||
|
printf("d-10=%s\n", vmIsTrailingZeros ? "true" : "false");
|
||||||
|
#endif
|
||||||
|
if (vmIsTrailingZeros) {
|
||||||
|
for (;;) {
|
||||||
|
const uint64_t vmDiv10 = div10(vm);
|
||||||
|
const uint32_t vmMod10 = ((uint32_t) vm) - 10 * ((uint32_t) vmDiv10);
|
||||||
|
if (vmMod10 != 0) {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
const uint64_t vpDiv10 = div10(vp);
|
||||||
|
const uint64_t vrDiv10 = div10(vr);
|
||||||
|
const uint32_t vrMod10 = ((uint32_t) vr) - 10 * ((uint32_t) vrDiv10);
|
||||||
|
vrIsTrailingZeros &= lastRemovedDigit == 0;
|
||||||
|
lastRemovedDigit = (uint8_t) vrMod10;
|
||||||
|
vr = vrDiv10;
|
||||||
|
vp = vpDiv10;
|
||||||
|
vm = vmDiv10;
|
||||||
|
++removed;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
#ifdef RYU_DEBUG
|
||||||
|
printf("%" PRIu64 " %d\n", vr, lastRemovedDigit);
|
||||||
|
printf("vr is trailing zeros=%s\n", vrIsTrailingZeros ? "true" : "false");
|
||||||
|
#endif
|
||||||
|
if (vrIsTrailingZeros && lastRemovedDigit == 5 && vr % 2 == 0) {
|
||||||
|
// Round even if the exact number is .....50..0.
|
||||||
|
lastRemovedDigit = 4;
|
||||||
|
}
|
||||||
|
// We need to take vr + 1 if vr is outside bounds or we need to round up.
|
||||||
|
output = vr + ((vr == vm && (!acceptBounds || !vmIsTrailingZeros)) || lastRemovedDigit >= 5);
|
||||||
|
} else {
|
||||||
|
// Specialized for the common case (~99.3%). Percentages below are relative to this.
|
||||||
|
bool roundUp = false;
|
||||||
|
const uint64_t vpDiv100 = div100(vp);
|
||||||
|
const uint64_t vmDiv100 = div100(vm);
|
||||||
|
if (vpDiv100 > vmDiv100) { // Optimization: remove two digits at a time (~86.2%).
|
||||||
|
const uint64_t vrDiv100 = div100(vr);
|
||||||
|
const uint32_t vrMod100 = ((uint32_t) vr) - 100 * ((uint32_t) vrDiv100);
|
||||||
|
roundUp = vrMod100 >= 50;
|
||||||
|
vr = vrDiv100;
|
||||||
|
vp = vpDiv100;
|
||||||
|
vm = vmDiv100;
|
||||||
|
removed += 2;
|
||||||
|
}
|
||||||
|
// Loop iterations below (approximately), without optimization above:
|
||||||
|
// 0: 0.03%, 1: 13.8%, 2: 70.6%, 3: 14.0%, 4: 1.40%, 5: 0.14%, 6+: 0.02%
|
||||||
|
// Loop iterations below (approximately), with optimization above:
|
||||||
|
// 0: 70.6%, 1: 27.8%, 2: 1.40%, 3: 0.14%, 4+: 0.02%
|
||||||
|
for (;;) {
|
||||||
|
const uint64_t vpDiv10 = div10(vp);
|
||||||
|
const uint64_t vmDiv10 = div10(vm);
|
||||||
|
if (vpDiv10 <= vmDiv10) {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
const uint64_t vrDiv10 = div10(vr);
|
||||||
|
const uint32_t vrMod10 = ((uint32_t) vr) - 10 * ((uint32_t) vrDiv10);
|
||||||
|
roundUp = vrMod10 >= 5;
|
||||||
|
vr = vrDiv10;
|
||||||
|
vp = vpDiv10;
|
||||||
|
vm = vmDiv10;
|
||||||
|
++removed;
|
||||||
|
}
|
||||||
|
#ifdef RYU_DEBUG
|
||||||
|
printf("%" PRIu64 " roundUp=%s\n", vr, roundUp ? "true" : "false");
|
||||||
|
printf("vr is trailing zeros=%s\n", vrIsTrailingZeros ? "true" : "false");
|
||||||
|
#endif
|
||||||
|
// We need to take vr + 1 if vr is outside bounds or we need to round up.
|
||||||
|
output = vr + (vr == vm || roundUp);
|
||||||
|
}
|
||||||
|
const int32_t exp = e10 + removed;
|
||||||
|
|
||||||
|
#ifdef RYU_DEBUG
|
||||||
|
printf("V+=%" PRIu64 "\nV =%" PRIu64 "\nV-=%" PRIu64 "\n", vp, vr, vm);
|
||||||
|
printf("O=%" PRIu64 "\n", output);
|
||||||
|
printf("EXP=%d\n", exp);
|
||||||
|
#endif
|
||||||
|
|
||||||
|
floating_decimal_64 fd;
|
||||||
|
fd.exponent = exp;
|
||||||
|
fd.mantissa = output;
|
||||||
|
return fd;
|
||||||
|
}
|
||||||
|
|
||||||
|
static inline int to_chars(const floating_decimal_64 v, const bool sign, char* const result) {
|
||||||
|
// Step 5: Print the decimal representation.
|
||||||
|
int index = 0;
|
||||||
|
if (sign) {
|
||||||
|
result[index++] = '-';
|
||||||
|
}
|
||||||
|
|
||||||
|
uint64_t output = v.mantissa;
|
||||||
|
const uint32_t olength = decimalLength17(output);
|
||||||
|
|
||||||
|
#ifdef RYU_DEBUG
|
||||||
|
printf("DIGITS=%" PRIu64 "\n", v.mantissa);
|
||||||
|
printf("OLEN=%u\n", olength);
|
||||||
|
printf("EXP=%u\n", v.exponent + olength);
|
||||||
|
#endif
|
||||||
|
|
||||||
|
// Print the decimal digits.
|
||||||
|
// The following code is equivalent to:
|
||||||
|
// for (uint32_t i = 0; i < olength - 1; ++i) {
|
||||||
|
// const uint32_t c = output % 10; output /= 10;
|
||||||
|
// result[index + olength - i] = (char) ('0' + c);
|
||||||
|
// }
|
||||||
|
// result[index] = '0' + output % 10;
|
||||||
|
|
||||||
|
uint32_t i = 0;
|
||||||
|
// We prefer 32-bit operations, even on 64-bit platforms.
|
||||||
|
// We have at most 17 digits, and uint32_t can store 9 digits.
|
||||||
|
// If output doesn't fit into uint32_t, we cut off 8 digits,
|
||||||
|
// so the rest will fit into uint32_t.
|
||||||
|
if ((output >> 32) != 0) {
|
||||||
|
// Expensive 64-bit division.
|
||||||
|
const uint64_t q = div1e8(output);
|
||||||
|
uint32_t output2 = ((uint32_t) output) - 100000000 * ((uint32_t) q);
|
||||||
|
output = q;
|
||||||
|
|
||||||
|
const uint32_t c = output2 % 10000;
|
||||||
|
output2 /= 10000;
|
||||||
|
const uint32_t d = output2 % 10000;
|
||||||
|
const uint32_t c0 = (c % 100) << 1;
|
||||||
|
const uint32_t c1 = (c / 100) << 1;
|
||||||
|
const uint32_t d0 = (d % 100) << 1;
|
||||||
|
const uint32_t d1 = (d / 100) << 1;
|
||||||
|
memcpy(result + index + olength - 1, DIGIT_TABLE + c0, 2);
|
||||||
|
memcpy(result + index + olength - 3, DIGIT_TABLE + c1, 2);
|
||||||
|
memcpy(result + index + olength - 5, DIGIT_TABLE + d0, 2);
|
||||||
|
memcpy(result + index + olength - 7, DIGIT_TABLE + d1, 2);
|
||||||
|
i += 8;
|
||||||
|
}
|
||||||
|
uint32_t output2 = (uint32_t) output;
|
||||||
|
while (output2 >= 10000) {
|
||||||
|
#ifdef __clang__ // https://bugs.llvm.org/show_bug.cgi?id=38217
|
||||||
|
const uint32_t c = output2 - 10000 * (output2 / 10000);
|
||||||
|
#else
|
||||||
|
const uint32_t c = output2 % 10000;
|
||||||
|
#endif
|
||||||
|
output2 /= 10000;
|
||||||
|
const uint32_t c0 = (c % 100) << 1;
|
||||||
|
const uint32_t c1 = (c / 100) << 1;
|
||||||
|
memcpy(result + index + olength - i - 1, DIGIT_TABLE + c0, 2);
|
||||||
|
memcpy(result + index + olength - i - 3, DIGIT_TABLE + c1, 2);
|
||||||
|
i += 4;
|
||||||
|
}
|
||||||
|
if (output2 >= 100) {
|
||||||
|
const uint32_t c = (output2 % 100) << 1;
|
||||||
|
output2 /= 100;
|
||||||
|
memcpy(result + index + olength - i - 1, DIGIT_TABLE + c, 2);
|
||||||
|
i += 2;
|
||||||
|
}
|
||||||
|
if (output2 >= 10) {
|
||||||
|
const uint32_t c = output2 << 1;
|
||||||
|
// We can't use memcpy here: the decimal dot goes between these two digits.
|
||||||
|
result[index + olength - i] = DIGIT_TABLE[c + 1];
|
||||||
|
result[index] = DIGIT_TABLE[c];
|
||||||
|
} else {
|
||||||
|
result[index] = (char) ('0' + output2);
|
||||||
|
}
|
||||||
|
|
||||||
|
// Print decimal point if needed.
|
||||||
|
if (olength > 1) {
|
||||||
|
result[index + 1] = '.';
|
||||||
|
index += olength + 1;
|
||||||
|
} else {
|
||||||
|
++index;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Print the exponent.
|
||||||
|
result[index++] = 'E';
|
||||||
|
int32_t exp = v.exponent + (int32_t) olength - 1;
|
||||||
|
if (exp < 0) {
|
||||||
|
result[index++] = '-';
|
||||||
|
exp = -exp;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (exp >= 100) {
|
||||||
|
const int32_t c = exp % 10;
|
||||||
|
memcpy(result + index, DIGIT_TABLE + 2 * (exp / 10), 2);
|
||||||
|
result[index + 2] = (char) ('0' + c);
|
||||||
|
index += 3;
|
||||||
|
} else if (exp >= 10) {
|
||||||
|
memcpy(result + index, DIGIT_TABLE + 2 * exp, 2);
|
||||||
|
index += 2;
|
||||||
|
} else {
|
||||||
|
result[index++] = (char) ('0' + exp);
|
||||||
|
}
|
||||||
|
|
||||||
|
return index;
|
||||||
|
}
|
||||||
|
|
||||||
|
static inline bool d2d_small_int(const uint64_t ieeeMantissa, const uint32_t ieeeExponent,
|
||||||
|
floating_decimal_64* const v) {
|
||||||
|
const uint64_t m2 = (1ull << DOUBLE_MANTISSA_BITS) | ieeeMantissa;
|
||||||
|
const int32_t e2 = (int32_t) ieeeExponent - DOUBLE_BIAS - DOUBLE_MANTISSA_BITS;
|
||||||
|
|
||||||
|
if (e2 > 0) {
|
||||||
|
// f = m2 * 2^e2 >= 2^53 is an integer.
|
||||||
|
// Ignore this case for now.
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (e2 < -52) {
|
||||||
|
// f < 1.
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Since 2^52 <= m2 < 2^53 and 0 <= -e2 <= 52: 1 <= f = m2 / 2^-e2 < 2^53.
|
||||||
|
// Test if the lower -e2 bits of the significand are 0, i.e. whether the fraction is 0.
|
||||||
|
const uint64_t mask = (1ull << -e2) - 1;
|
||||||
|
const uint64_t fraction = m2 & mask;
|
||||||
|
if (fraction != 0) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
// f is an integer in the range [1, 2^53).
|
||||||
|
// Note: mantissa might contain trailing (decimal) 0's.
|
||||||
|
// Note: since 2^53 < 10^16, there is no need to adjust decimalLength17().
|
||||||
|
v->mantissa = m2 >> -e2;
|
||||||
|
v->exponent = 0;
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
int d2s_buffered_n(double f, char* result) {
|
||||||
|
// Step 1: Decode the floating-point number, and unify normalized and subnormal cases.
|
||||||
|
const uint64_t bits = double_to_bits(f);
|
||||||
|
|
||||||
|
#ifdef RYU_DEBUG
|
||||||
|
printf("IN=");
|
||||||
|
for (int32_t bit = 63; bit >= 0; --bit) {
|
||||||
|
printf("%d", (int) ((bits >> bit) & 1));
|
||||||
|
}
|
||||||
|
printf("\n");
|
||||||
|
#endif
|
||||||
|
|
||||||
|
// Decode bits into sign, mantissa, and exponent.
|
||||||
|
const bool ieeeSign = ((bits >> (DOUBLE_MANTISSA_BITS + DOUBLE_EXPONENT_BITS)) & 1) != 0;
|
||||||
|
const uint64_t ieeeMantissa = bits & ((1ull << DOUBLE_MANTISSA_BITS) - 1);
|
||||||
|
const uint32_t ieeeExponent = (uint32_t) ((bits >> DOUBLE_MANTISSA_BITS) & ((1u << DOUBLE_EXPONENT_BITS) - 1));
|
||||||
|
// Case distinction; exit early for the easy cases.
|
||||||
|
if (ieeeExponent == ((1u << DOUBLE_EXPONENT_BITS) - 1u) || (ieeeExponent == 0 && ieeeMantissa == 0)) {
|
||||||
|
return copy_special_str(result, ieeeSign, ieeeExponent, ieeeMantissa);
|
||||||
|
}
|
||||||
|
|
||||||
|
floating_decimal_64 v;
|
||||||
|
const bool isSmallInt = d2d_small_int(ieeeMantissa, ieeeExponent, &v);
|
||||||
|
if (isSmallInt) {
|
||||||
|
// For small integers in the range [1, 2^53), v.mantissa might contain trailing (decimal) zeros.
|
||||||
|
// For scientific notation we need to move these zeros into the exponent.
|
||||||
|
// (This is not needed for fixed-point notation, so it might be beneficial to trim
|
||||||
|
// trailing zeros in to_chars only if needed - once fixed-point notation output is implemented.)
|
||||||
|
for (;;) {
|
||||||
|
const uint64_t q = div10(v.mantissa);
|
||||||
|
const uint32_t r = ((uint32_t) v.mantissa) - 10 * ((uint32_t) q);
|
||||||
|
if (r != 0) {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
v.mantissa = q;
|
||||||
|
++v.exponent;
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
v = d2d(ieeeMantissa, ieeeExponent);
|
||||||
|
}
|
||||||
|
|
||||||
|
return to_chars(v, ieeeSign, result);
|
||||||
|
}
|
||||||
|
|
||||||
|
void d2s_buffered(double f, char* result) {
|
||||||
|
const int index = d2s_buffered_n(f, result);
|
||||||
|
|
||||||
|
// Terminate the string.
|
||||||
|
result[index] = '\0';
|
||||||
|
}
|
||||||
|
|
||||||
|
char* d2s(double f) {
|
||||||
|
char* const result = (char*) malloc(25);
|
||||||
|
d2s_buffered(f, result);
|
||||||
|
return result;
|
||||||
|
}
|
||||||
@@ -0,0 +1,20 @@
|
|||||||
|
#ifndef NATIVE_JSON_NUMBERS_H
|
||||||
|
#define NATIVE_JSON_NUMBERS_H
|
||||||
|
#include <stddef.h>
|
||||||
|
#include <stdio.h>
|
||||||
|
|
||||||
|
#define NATIVE_JSON_DOUBLE_CAPACITY 32
|
||||||
|
|
||||||
|
/* Locale-independent finite binary64 token; no terminating NUL is promised.
|
||||||
|
* Returns its length, or zero for NaN/Infinity. Negative zero is -0.0 so both
|
||||||
|
* Python and JavaScript JSON readers preserve its sign. */
|
||||||
|
int native_json_format_double(char output[NATIVE_JSON_DOUBLE_CAPACITY], double value);
|
||||||
|
|
||||||
|
/* Write a number, or a complete array from values[i*stride]. Each call flushes
|
||||||
|
* its own fixed-size buffer into FILE, but does not fflush/fclose the FILE.
|
||||||
|
* Return zero on invalid input/nonfinite values/write failure. A failed array
|
||||||
|
* may have written a prefix: callers must not publish that document as valid.
|
||||||
|
* The caller owns FILE and must check its final flush/close before publishing. */
|
||||||
|
int native_json_write_number(FILE *file, double value);
|
||||||
|
int native_json_write_array(FILE *file, const double *values, size_t count, size_t stride);
|
||||||
|
#endif
|
||||||
@@ -0,0 +1,114 @@
|
|||||||
|
// Copyright 2018 Ulf Adams
|
||||||
|
//
|
||||||
|
// The contents of this file may be used under the terms of the Apache License,
|
||||||
|
// Version 2.0.
|
||||||
|
//
|
||||||
|
// (See accompanying file LICENSE-Apache or copy at
|
||||||
|
// http://www.apache.org/licenses/LICENSE-2.0)
|
||||||
|
//
|
||||||
|
// Alternatively, the contents of this file may be used under the terms of
|
||||||
|
// the Boost Software License, Version 1.0.
|
||||||
|
// (See accompanying file LICENSE-Boost or copy at
|
||||||
|
// https://www.boost.org/LICENSE_1_0.txt)
|
||||||
|
//
|
||||||
|
// Unless required by applicable law or agreed to in writing, this software
|
||||||
|
// is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
|
||||||
|
// KIND, either express or implied.
|
||||||
|
#ifndef RYU_COMMON_H
|
||||||
|
#define RYU_COMMON_H
|
||||||
|
|
||||||
|
#include <assert.h>
|
||||||
|
#include <stdint.h>
|
||||||
|
#include <string.h>
|
||||||
|
|
||||||
|
#if defined(_M_IX86) || defined(_M_ARM)
|
||||||
|
#define RYU_32_BIT_PLATFORM
|
||||||
|
#endif
|
||||||
|
|
||||||
|
// Returns the number of decimal digits in v, which must not contain more than 9 digits.
|
||||||
|
static inline uint32_t decimalLength9(const uint32_t v) {
|
||||||
|
// Function precondition: v is not a 10-digit number.
|
||||||
|
// (f2s: 9 digits are sufficient for round-tripping.)
|
||||||
|
// (d2fixed: We print 9-digit blocks.)
|
||||||
|
assert(v < 1000000000);
|
||||||
|
if (v >= 100000000) { return 9; }
|
||||||
|
if (v >= 10000000) { return 8; }
|
||||||
|
if (v >= 1000000) { return 7; }
|
||||||
|
if (v >= 100000) { return 6; }
|
||||||
|
if (v >= 10000) { return 5; }
|
||||||
|
if (v >= 1000) { return 4; }
|
||||||
|
if (v >= 100) { return 3; }
|
||||||
|
if (v >= 10) { return 2; }
|
||||||
|
return 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Returns e == 0 ? 1 : [log_2(5^e)]; requires 0 <= e <= 3528.
|
||||||
|
static inline int32_t log2pow5(const int32_t e) {
|
||||||
|
// This approximation works up to the point that the multiplication overflows at e = 3529.
|
||||||
|
// If the multiplication were done in 64 bits, it would fail at 5^4004 which is just greater
|
||||||
|
// than 2^9297.
|
||||||
|
assert(e >= 0);
|
||||||
|
assert(e <= 3528);
|
||||||
|
return (int32_t) ((((uint32_t) e) * 1217359) >> 19);
|
||||||
|
}
|
||||||
|
|
||||||
|
// Returns e == 0 ? 1 : ceil(log_2(5^e)); requires 0 <= e <= 3528.
|
||||||
|
static inline int32_t pow5bits(const int32_t e) {
|
||||||
|
// This approximation works up to the point that the multiplication overflows at e = 3529.
|
||||||
|
// If the multiplication were done in 64 bits, it would fail at 5^4004 which is just greater
|
||||||
|
// than 2^9297.
|
||||||
|
assert(e >= 0);
|
||||||
|
assert(e <= 3528);
|
||||||
|
return (int32_t) (((((uint32_t) e) * 1217359) >> 19) + 1);
|
||||||
|
}
|
||||||
|
|
||||||
|
// Returns e == 0 ? 1 : ceil(log_2(5^e)); requires 0 <= e <= 3528.
|
||||||
|
static inline int32_t ceil_log2pow5(const int32_t e) {
|
||||||
|
return log2pow5(e) + 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Returns floor(log_10(2^e)); requires 0 <= e <= 1650.
|
||||||
|
static inline uint32_t log10Pow2(const int32_t e) {
|
||||||
|
// The first value this approximation fails for is 2^1651 which is just greater than 10^297.
|
||||||
|
assert(e >= 0);
|
||||||
|
assert(e <= 1650);
|
||||||
|
return (((uint32_t) e) * 78913) >> 18;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Returns floor(log_10(5^e)); requires 0 <= e <= 2620.
|
||||||
|
static inline uint32_t log10Pow5(const int32_t e) {
|
||||||
|
// The first value this approximation fails for is 5^2621 which is just greater than 10^1832.
|
||||||
|
assert(e >= 0);
|
||||||
|
assert(e <= 2620);
|
||||||
|
return (((uint32_t) e) * 732923) >> 20;
|
||||||
|
}
|
||||||
|
|
||||||
|
static inline int copy_special_str(char * const result, const bool sign, const bool exponent, const bool mantissa) {
|
||||||
|
if (mantissa) {
|
||||||
|
memcpy(result, "NaN", 3);
|
||||||
|
return 3;
|
||||||
|
}
|
||||||
|
if (sign) {
|
||||||
|
result[0] = '-';
|
||||||
|
}
|
||||||
|
if (exponent) {
|
||||||
|
memcpy(result + sign, "Infinity", 8);
|
||||||
|
return sign + 8;
|
||||||
|
}
|
||||||
|
memcpy(result + sign, "0E0", 3);
|
||||||
|
return sign + 3;
|
||||||
|
}
|
||||||
|
|
||||||
|
static inline uint32_t float_to_bits(const float f) {
|
||||||
|
uint32_t bits = 0;
|
||||||
|
memcpy(&bits, &f, sizeof(float));
|
||||||
|
return bits;
|
||||||
|
}
|
||||||
|
|
||||||
|
static inline uint64_t double_to_bits(const double d) {
|
||||||
|
uint64_t bits = 0;
|
||||||
|
memcpy(&bits, &d, sizeof(double));
|
||||||
|
return bits;
|
||||||
|
}
|
||||||
|
|
||||||
|
#endif // RYU_COMMON_H
|
||||||
@@ -0,0 +1,367 @@
|
|||||||
|
// Copyright 2018 Ulf Adams
|
||||||
|
//
|
||||||
|
// The contents of this file may be used under the terms of the Apache License,
|
||||||
|
// Version 2.0.
|
||||||
|
//
|
||||||
|
// (See accompanying file LICENSE-Apache or copy at
|
||||||
|
// http://www.apache.org/licenses/LICENSE-2.0)
|
||||||
|
//
|
||||||
|
// Alternatively, the contents of this file may be used under the terms of
|
||||||
|
// the Boost Software License, Version 1.0.
|
||||||
|
// (See accompanying file LICENSE-Boost or copy at
|
||||||
|
// https://www.boost.org/LICENSE_1_0.txt)
|
||||||
|
//
|
||||||
|
// Unless required by applicable law or agreed to in writing, this software
|
||||||
|
// is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
|
||||||
|
// KIND, either express or implied.
|
||||||
|
#ifndef RYU_D2S_FULL_TABLE_H
|
||||||
|
#define RYU_D2S_FULL_TABLE_H
|
||||||
|
|
||||||
|
// These tables are generated by PrintDoubleLookupTable.
|
||||||
|
#define DOUBLE_POW5_INV_BITCOUNT 125
|
||||||
|
#define DOUBLE_POW5_BITCOUNT 125
|
||||||
|
|
||||||
|
#define DOUBLE_POW5_INV_TABLE_SIZE 342
|
||||||
|
#define DOUBLE_POW5_TABLE_SIZE 326
|
||||||
|
|
||||||
|
static const uint64_t DOUBLE_POW5_INV_SPLIT[DOUBLE_POW5_INV_TABLE_SIZE][2] = {
|
||||||
|
{ 1u, 2305843009213693952u }, { 11068046444225730970u, 1844674407370955161u },
|
||||||
|
{ 5165088340638674453u, 1475739525896764129u }, { 7821419487252849886u, 1180591620717411303u },
|
||||||
|
{ 8824922364862649494u, 1888946593147858085u }, { 7059937891890119595u, 1511157274518286468u },
|
||||||
|
{ 13026647942995916322u, 1208925819614629174u }, { 9774590264567735146u, 1934281311383406679u },
|
||||||
|
{ 11509021026396098440u, 1547425049106725343u }, { 16585914450600699399u, 1237940039285380274u },
|
||||||
|
{ 15469416676735388068u, 1980704062856608439u }, { 16064882156130220778u, 1584563250285286751u },
|
||||||
|
{ 9162556910162266299u, 1267650600228229401u }, { 7281393426775805432u, 2028240960365167042u },
|
||||||
|
{ 16893161185646375315u, 1622592768292133633u }, { 2446482504291369283u, 1298074214633706907u },
|
||||||
|
{ 7603720821608101175u, 2076918743413931051u }, { 2393627842544570617u, 1661534994731144841u },
|
||||||
|
{ 16672297533003297786u, 1329227995784915872u }, { 11918280793837635165u, 2126764793255865396u },
|
||||||
|
{ 5845275820328197809u, 1701411834604692317u }, { 15744267100488289217u, 1361129467683753853u },
|
||||||
|
{ 3054734472329800808u, 2177807148294006166u }, { 17201182836831481939u, 1742245718635204932u },
|
||||||
|
{ 6382248639981364905u, 1393796574908163946u }, { 2832900194486363201u, 2230074519853062314u },
|
||||||
|
{ 5955668970331000884u, 1784059615882449851u }, { 1075186361522890384u, 1427247692705959881u },
|
||||||
|
{ 12788344622662355584u, 2283596308329535809u }, { 13920024512871794791u, 1826877046663628647u },
|
||||||
|
{ 3757321980813615186u, 1461501637330902918u }, { 10384555214134712795u, 1169201309864722334u },
|
||||||
|
{ 5547241898389809503u, 1870722095783555735u }, { 4437793518711847602u, 1496577676626844588u },
|
||||||
|
{ 10928932444453298728u, 1197262141301475670u }, { 17486291911125277965u, 1915619426082361072u },
|
||||||
|
{ 6610335899416401726u, 1532495540865888858u }, { 12666966349016942027u, 1225996432692711086u },
|
||||||
|
{ 12888448528943286597u, 1961594292308337738u }, { 17689456452638449924u, 1569275433846670190u },
|
||||||
|
{ 14151565162110759939u, 1255420347077336152u }, { 7885109000409574610u, 2008672555323737844u },
|
||||||
|
{ 9997436015069570011u, 1606938044258990275u }, { 7997948812055656009u, 1285550435407192220u },
|
||||||
|
{ 12796718099289049614u, 2056880696651507552u }, { 2858676849947419045u, 1645504557321206042u },
|
||||||
|
{ 13354987924183666206u, 1316403645856964833u }, { 17678631863951955605u, 2106245833371143733u },
|
||||||
|
{ 3074859046935833515u, 1684996666696914987u }, { 13527933681774397782u, 1347997333357531989u },
|
||||||
|
{ 10576647446613305481u, 2156795733372051183u }, { 15840015586774465031u, 1725436586697640946u },
|
||||||
|
{ 8982663654677661702u, 1380349269358112757u }, { 18061610662226169046u, 2208558830972980411u },
|
||||||
|
{ 10759939715039024913u, 1766847064778384329u }, { 12297300586773130254u, 1413477651822707463u },
|
||||||
|
{ 15986332124095098083u, 2261564242916331941u }, { 9099716884534168143u, 1809251394333065553u },
|
||||||
|
{ 14658471137111155161u, 1447401115466452442u }, { 4348079280205103483u, 1157920892373161954u },
|
||||||
|
{ 14335624477811986218u, 1852673427797059126u }, { 7779150767507678651u, 1482138742237647301u },
|
||||||
|
{ 2533971799264232598u, 1185710993790117841u }, { 15122401323048503126u, 1897137590064188545u },
|
||||||
|
{ 12097921058438802501u, 1517710072051350836u }, { 5988988032009131678u, 1214168057641080669u },
|
||||||
|
{ 16961078480698431330u, 1942668892225729070u }, { 13568862784558745064u, 1554135113780583256u },
|
||||||
|
{ 7165741412905085728u, 1243308091024466605u }, { 11465186260648137165u, 1989292945639146568u },
|
||||||
|
{ 16550846638002330379u, 1591434356511317254u }, { 16930026125143774626u, 1273147485209053803u },
|
||||||
|
{ 4951948911778577463u, 2037035976334486086u }, { 272210314680951647u, 1629628781067588869u },
|
||||||
|
{ 3907117066486671641u, 1303703024854071095u }, { 6251387306378674625u, 2085924839766513752u },
|
||||||
|
{ 16069156289328670670u, 1668739871813211001u }, { 9165976216721026213u, 1334991897450568801u },
|
||||||
|
{ 7286864317269821294u, 2135987035920910082u }, { 16897537898041588005u, 1708789628736728065u },
|
||||||
|
{ 13518030318433270404u, 1367031702989382452u }, { 6871453250525591353u, 2187250724783011924u },
|
||||||
|
{ 9186511415162383406u, 1749800579826409539u }, { 11038557946871817048u, 1399840463861127631u },
|
||||||
|
{ 10282995085511086630u, 2239744742177804210u }, { 8226396068408869304u, 1791795793742243368u },
|
||||||
|
{ 13959814484210916090u, 1433436634993794694u }, { 11267656730511734774u, 2293498615990071511u },
|
||||||
|
{ 5324776569667477496u, 1834798892792057209u }, { 7949170070475892320u, 1467839114233645767u },
|
||||||
|
{ 17427382500606444826u, 1174271291386916613u }, { 5747719112518849781u, 1878834066219066582u },
|
||||||
|
{ 15666221734240810795u, 1503067252975253265u }, { 12532977387392648636u, 1202453802380202612u },
|
||||||
|
{ 5295368560860596524u, 1923926083808324180u }, { 4236294848688477220u, 1539140867046659344u },
|
||||||
|
{ 7078384693692692099u, 1231312693637327475u }, { 11325415509908307358u, 1970100309819723960u },
|
||||||
|
{ 9060332407926645887u, 1576080247855779168u }, { 14626963555825137356u, 1260864198284623334u },
|
||||||
|
{ 12335095245094488799u, 2017382717255397335u }, { 9868076196075591040u, 1613906173804317868u },
|
||||||
|
{ 15273158586344293478u, 1291124939043454294u }, { 13369007293925138595u, 2065799902469526871u },
|
||||||
|
{ 7005857020398200553u, 1652639921975621497u }, { 16672732060544291412u, 1322111937580497197u },
|
||||||
|
{ 11918976037903224966u, 2115379100128795516u }, { 5845832015580669650u, 1692303280103036413u },
|
||||||
|
{ 12055363241948356366u, 1353842624082429130u }, { 841837113407818570u, 2166148198531886609u },
|
||||||
|
{ 4362818505468165179u, 1732918558825509287u }, { 14558301248600263113u, 1386334847060407429u },
|
||||||
|
{ 12225235553534690011u, 2218135755296651887u }, { 2401490813343931363u, 1774508604237321510u },
|
||||||
|
{ 1921192650675145090u, 1419606883389857208u }, { 17831303500047873437u, 2271371013423771532u },
|
||||||
|
{ 6886345170554478103u, 1817096810739017226u }, { 1819727321701672159u, 1453677448591213781u },
|
||||||
|
{ 16213177116328979020u, 1162941958872971024u }, { 14873036941900635463u, 1860707134196753639u },
|
||||||
|
{ 15587778368262418694u, 1488565707357402911u }, { 8780873879868024632u, 1190852565885922329u },
|
||||||
|
{ 2981351763563108441u, 1905364105417475727u }, { 13453127855076217722u, 1524291284333980581u },
|
||||||
|
{ 7073153469319063855u, 1219433027467184465u }, { 11317045550910502167u, 1951092843947495144u },
|
||||||
|
{ 12742985255470312057u, 1560874275157996115u }, { 10194388204376249646u, 1248699420126396892u },
|
||||||
|
{ 1553625868034358140u, 1997919072202235028u }, { 8621598323911307159u, 1598335257761788022u },
|
||||||
|
{ 17965325103354776697u, 1278668206209430417u }, { 13987124906400001422u, 2045869129935088668u },
|
||||||
|
{ 121653480894270168u, 1636695303948070935u }, { 97322784715416134u, 1309356243158456748u },
|
||||||
|
{ 14913111714512307107u, 2094969989053530796u }, { 8241140556867935363u, 1675975991242824637u },
|
||||||
|
{ 17660958889720079260u, 1340780792994259709u }, { 17189487779326395846u, 2145249268790815535u },
|
||||||
|
{ 13751590223461116677u, 1716199415032652428u }, { 18379969808252713988u, 1372959532026121942u },
|
||||||
|
{ 14650556434236701088u, 2196735251241795108u }, { 652398703163629901u, 1757388200993436087u },
|
||||||
|
{ 11589965406756634890u, 1405910560794748869u }, { 7475898206584884855u, 2249456897271598191u },
|
||||||
|
{ 2291369750525997561u, 1799565517817278553u }, { 9211793429904618695u, 1439652414253822842u },
|
||||||
|
{ 18428218302589300235u, 2303443862806116547u }, { 7363877012587619542u, 1842755090244893238u },
|
||||||
|
{ 13269799239553916280u, 1474204072195914590u }, { 10615839391643133024u, 1179363257756731672u },
|
||||||
|
{ 2227947767661371545u, 1886981212410770676u }, { 16539753473096738529u, 1509584969928616540u },
|
||||||
|
{ 13231802778477390823u, 1207667975942893232u }, { 6413489186596184024u, 1932268761508629172u },
|
||||||
|
{ 16198837793502678189u, 1545815009206903337u }, { 5580372605318321905u, 1236652007365522670u },
|
||||||
|
{ 8928596168509315048u, 1978643211784836272u }, { 18210923379033183008u, 1582914569427869017u },
|
||||||
|
{ 7190041073742725760u, 1266331655542295214u }, { 436019273762630246u, 2026130648867672343u },
|
||||||
|
{ 7727513048493924843u, 1620904519094137874u }, { 9871359253537050198u, 1296723615275310299u },
|
||||||
|
{ 4726128361433549347u, 2074757784440496479u }, { 7470251503888749801u, 1659806227552397183u },
|
||||||
|
{ 13354898832594820487u, 1327844982041917746u }, { 13989140502667892133u, 2124551971267068394u },
|
||||||
|
{ 14880661216876224029u, 1699641577013654715u }, { 11904528973500979224u, 1359713261610923772u },
|
||||||
|
{ 4289851098633925465u, 2175541218577478036u }, { 18189276137874781665u, 1740432974861982428u },
|
||||||
|
{ 3483374466074094362u, 1392346379889585943u }, { 1884050330976640656u, 2227754207823337509u },
|
||||||
|
{ 5196589079523222848u, 1782203366258670007u }, { 15225317707844309248u, 1425762693006936005u },
|
||||||
|
{ 5913764258841343181u, 2281220308811097609u }, { 8420360221814984868u, 1824976247048878087u },
|
||||||
|
{ 17804334621677718864u, 1459980997639102469u }, { 17932816512084085415u, 1167984798111281975u },
|
||||||
|
{ 10245762345624985047u, 1868775676978051161u }, { 4507261061758077715u, 1495020541582440929u },
|
||||||
|
{ 7295157664148372495u, 1196016433265952743u }, { 7982903447895485668u, 1913626293225524389u },
|
||||||
|
{ 10075671573058298858u, 1530901034580419511u }, { 4371188443704728763u, 1224720827664335609u },
|
||||||
|
{ 14372599139411386667u, 1959553324262936974u }, { 15187428126271019657u, 1567642659410349579u },
|
||||||
|
{ 15839291315758726049u, 1254114127528279663u }, { 3206773216762499739u, 2006582604045247462u },
|
||||||
|
{ 13633465017635730761u, 1605266083236197969u }, { 14596120828850494932u, 1284212866588958375u },
|
||||||
|
{ 4907049252451240275u, 2054740586542333401u }, { 236290587219081897u, 1643792469233866721u },
|
||||||
|
{ 14946427728742906810u, 1315033975387093376u }, { 16535586736504830250u, 2104054360619349402u },
|
||||||
|
{ 5849771759720043554u, 1683243488495479522u }, { 15747863852001765813u, 1346594790796383617u },
|
||||||
|
{ 10439186904235184007u, 2154551665274213788u }, { 15730047152871967852u, 1723641332219371030u },
|
||||||
|
{ 12584037722297574282u, 1378913065775496824u }, { 9066413911450387881u, 2206260905240794919u },
|
||||||
|
{ 10942479943902220628u, 1765008724192635935u }, { 8753983955121776503u, 1412006979354108748u },
|
||||||
|
{ 10317025513452932081u, 2259211166966573997u }, { 874922781278525018u, 1807368933573259198u },
|
||||||
|
{ 8078635854506640661u, 1445895146858607358u }, { 13841606313089133175u, 1156716117486885886u },
|
||||||
|
{ 14767872471458792434u, 1850745787979017418u }, { 746251532941302978u, 1480596630383213935u },
|
||||||
|
{ 597001226353042382u, 1184477304306571148u }, { 15712597221132509104u, 1895163686890513836u },
|
||||||
|
{ 8880728962164096960u, 1516130949512411069u }, { 10793931984473187891u, 1212904759609928855u },
|
||||||
|
{ 17270291175157100626u, 1940647615375886168u }, { 2748186495899949531u, 1552518092300708935u },
|
||||||
|
{ 2198549196719959625u, 1242014473840567148u }, { 18275073973719576693u, 1987223158144907436u },
|
||||||
|
{ 10930710364233751031u, 1589778526515925949u }, { 12433917106128911148u, 1271822821212740759u },
|
||||||
|
{ 8826220925580526867u, 2034916513940385215u }, { 7060976740464421494u, 1627933211152308172u },
|
||||||
|
{ 16716827836597268165u, 1302346568921846537u }, { 11989529279587987770u, 2083754510274954460u },
|
||||||
|
{ 9591623423670390216u, 1667003608219963568u }, { 15051996368420132820u, 1333602886575970854u },
|
||||||
|
{ 13015147745246481542u, 2133764618521553367u }, { 3033420566713364587u, 1707011694817242694u },
|
||||||
|
{ 6116085268112601993u, 1365609355853794155u }, { 9785736428980163188u, 2184974969366070648u },
|
||||||
|
{ 15207286772667951197u, 1747979975492856518u }, { 1097782973908629988u, 1398383980394285215u },
|
||||||
|
{ 1756452758253807981u, 2237414368630856344u }, { 5094511021344956708u, 1789931494904685075u },
|
||||||
|
{ 4075608817075965366u, 1431945195923748060u }, { 6520974107321544586u, 2291112313477996896u },
|
||||||
|
{ 1527430471115325346u, 1832889850782397517u }, { 12289990821117991246u, 1466311880625918013u },
|
||||||
|
{ 17210690286378213644u, 1173049504500734410u }, { 9090360384495590213u, 1876879207201175057u },
|
||||||
|
{ 18340334751822203140u, 1501503365760940045u }, { 14672267801457762512u, 1201202692608752036u },
|
||||||
|
{ 16096930852848599373u, 1921924308174003258u }, { 1809498238053148529u, 1537539446539202607u },
|
||||||
|
{ 12515645034668249793u, 1230031557231362085u }, { 1578287981759648052u, 1968050491570179337u },
|
||||||
|
{ 12330676829633449412u, 1574440393256143469u }, { 13553890278448669853u, 1259552314604914775u },
|
||||||
|
{ 3239480371808320148u, 2015283703367863641u }, { 17348979556414297411u, 1612226962694290912u },
|
||||||
|
{ 6500486015647617283u, 1289781570155432730u }, { 10400777625036187652u, 2063650512248692368u },
|
||||||
|
{ 15699319729512770768u, 1650920409798953894u }, { 16248804598352126938u, 1320736327839163115u },
|
||||||
|
{ 7551343283653851484u, 2113178124542660985u }, { 6041074626923081187u, 1690542499634128788u },
|
||||||
|
{ 12211557331022285596u, 1352433999707303030u }, { 1091747655926105338u, 2163894399531684849u },
|
||||||
|
{ 4562746939482794594u, 1731115519625347879u }, { 7339546366328145998u, 1384892415700278303u },
|
||||||
|
{ 8053925371383123274u, 2215827865120445285u }, { 6443140297106498619u, 1772662292096356228u },
|
||||||
|
{ 12533209867169019542u, 1418129833677084982u }, { 5295740528502789974u, 2269007733883335972u },
|
||||||
|
{ 15304638867027962949u, 1815206187106668777u }, { 4865013464138549713u, 1452164949685335022u },
|
||||||
|
{ 14960057215536570740u, 1161731959748268017u }, { 9178696285890871890u, 1858771135597228828u },
|
||||||
|
{ 14721654658196518159u, 1487016908477783062u }, { 4398626097073393881u, 1189613526782226450u },
|
||||||
|
{ 7037801755317430209u, 1903381642851562320u }, { 5630241404253944167u, 1522705314281249856u },
|
||||||
|
{ 814844308661245011u, 1218164251424999885u }, { 1303750893857992017u, 1949062802279999816u },
|
||||||
|
{ 15800395974054034906u, 1559250241823999852u }, { 5261619149759407279u, 1247400193459199882u },
|
||||||
|
{ 12107939454356961969u, 1995840309534719811u }, { 5997002748743659252u, 1596672247627775849u },
|
||||||
|
{ 8486951013736837725u, 1277337798102220679u }, { 2511075177753209390u, 2043740476963553087u },
|
||||||
|
{ 13076906586428298482u, 1634992381570842469u }, { 14150874083884549109u, 1307993905256673975u },
|
||||||
|
{ 4194654460505726958u, 2092790248410678361u }, { 18113118827372222859u, 1674232198728542688u },
|
||||||
|
{ 3422448617672047318u, 1339385758982834151u }, { 16543964232501006678u, 2143017214372534641u },
|
||||||
|
{ 9545822571258895019u, 1714413771498027713u }, { 15015355686490936662u, 1371531017198422170u },
|
||||||
|
{ 5577825024675947042u, 2194449627517475473u }, { 11840957649224578280u, 1755559702013980378u },
|
||||||
|
{ 16851463748863483271u, 1404447761611184302u }, { 12204946739213931940u, 2247116418577894884u },
|
||||||
|
{ 13453306206113055875u, 1797693134862315907u }, { 3383947335406624054u, 1438154507889852726u },
|
||||||
|
{ 16482362180876329456u, 2301047212623764361u }, { 9496540929959153242u, 1840837770099011489u },
|
||||||
|
{ 11286581558709232917u, 1472670216079209191u }, { 5339916432225476010u, 1178136172863367353u },
|
||||||
|
{ 4854517476818851293u, 1885017876581387765u }, { 3883613981455081034u, 1508014301265110212u },
|
||||||
|
{ 14174937629389795797u, 1206411441012088169u }, { 11611853762797942306u, 1930258305619341071u },
|
||||||
|
{ 5600134195496443521u, 1544206644495472857u }, { 15548153800622885787u, 1235365315596378285u },
|
||||||
|
{ 6430302007287065643u, 1976584504954205257u }, { 16212288050055383484u, 1581267603963364205u },
|
||||||
|
{ 12969830440044306787u, 1265014083170691364u }, { 9683682259845159889u, 2024022533073106183u },
|
||||||
|
{ 15125643437359948558u, 1619218026458484946u }, { 8411165935146048523u, 1295374421166787957u },
|
||||||
|
{ 17147214310975587960u, 2072599073866860731u }, { 10028422634038560045u, 1658079259093488585u },
|
||||||
|
{ 8022738107230848036u, 1326463407274790868u }, { 9147032156827446534u, 2122341451639665389u },
|
||||||
|
{ 11006974540203867551u, 1697873161311732311u }, { 5116230817421183718u, 1358298529049385849u },
|
||||||
|
{ 15564666937357714594u, 2173277646479017358u }, { 1383687105660440706u, 1738622117183213887u },
|
||||||
|
{ 12174996128754083534u, 1390897693746571109u }, { 8411947361780802685u, 2225436309994513775u },
|
||||||
|
{ 6729557889424642148u, 1780349047995611020u }, { 5383646311539713719u, 1424279238396488816u },
|
||||||
|
{ 1235136468979721303u, 2278846781434382106u }, { 15745504434151418335u, 1823077425147505684u },
|
||||||
|
{ 16285752362063044992u, 1458461940118004547u }, { 5649904260166615347u, 1166769552094403638u },
|
||||||
|
{ 5350498001524674232u, 1866831283351045821u }, { 591049586477829062u, 1493465026680836657u },
|
||||||
|
{ 11540886113407994219u, 1194772021344669325u }, { 18673707743239135u, 1911635234151470921u },
|
||||||
|
{ 14772334225162232601u, 1529308187321176736u }, { 8128518565387875758u, 1223446549856941389u },
|
||||||
|
{ 1937583260394870242u, 1957514479771106223u }, { 8928764237799716840u, 1566011583816884978u },
|
||||||
|
{ 14521709019723594119u, 1252809267053507982u }, { 8477339172590109297u, 2004494827285612772u },
|
||||||
|
{ 17849917782297818407u, 1603595861828490217u }, { 6901236596354434079u, 1282876689462792174u },
|
||||||
|
{ 18420676183650915173u, 2052602703140467478u }, { 3668494502695001169u, 1642082162512373983u },
|
||||||
|
{ 10313493231639821582u, 1313665730009899186u }, { 9122891541139893884u, 2101865168015838698u },
|
||||||
|
{ 14677010862395735754u, 1681492134412670958u }, { 673562245690857633u, 1345193707530136767u }
|
||||||
|
};
|
||||||
|
|
||||||
|
static const uint64_t DOUBLE_POW5_SPLIT[DOUBLE_POW5_TABLE_SIZE][2] = {
|
||||||
|
{ 0u, 1152921504606846976u }, { 0u, 1441151880758558720u },
|
||||||
|
{ 0u, 1801439850948198400u }, { 0u, 2251799813685248000u },
|
||||||
|
{ 0u, 1407374883553280000u }, { 0u, 1759218604441600000u },
|
||||||
|
{ 0u, 2199023255552000000u }, { 0u, 1374389534720000000u },
|
||||||
|
{ 0u, 1717986918400000000u }, { 0u, 2147483648000000000u },
|
||||||
|
{ 0u, 1342177280000000000u }, { 0u, 1677721600000000000u },
|
||||||
|
{ 0u, 2097152000000000000u }, { 0u, 1310720000000000000u },
|
||||||
|
{ 0u, 1638400000000000000u }, { 0u, 2048000000000000000u },
|
||||||
|
{ 0u, 1280000000000000000u }, { 0u, 1600000000000000000u },
|
||||||
|
{ 0u, 2000000000000000000u }, { 0u, 1250000000000000000u },
|
||||||
|
{ 0u, 1562500000000000000u }, { 0u, 1953125000000000000u },
|
||||||
|
{ 0u, 1220703125000000000u }, { 0u, 1525878906250000000u },
|
||||||
|
{ 0u, 1907348632812500000u }, { 0u, 1192092895507812500u },
|
||||||
|
{ 0u, 1490116119384765625u }, { 4611686018427387904u, 1862645149230957031u },
|
||||||
|
{ 9799832789158199296u, 1164153218269348144u }, { 12249790986447749120u, 1455191522836685180u },
|
||||||
|
{ 15312238733059686400u, 1818989403545856475u }, { 14528612397897220096u, 2273736754432320594u },
|
||||||
|
{ 13692068767113150464u, 1421085471520200371u }, { 12503399940464050176u, 1776356839400250464u },
|
||||||
|
{ 15629249925580062720u, 2220446049250313080u }, { 9768281203487539200u, 1387778780781445675u },
|
||||||
|
{ 7598665485932036096u, 1734723475976807094u }, { 274959820560269312u, 2168404344971008868u },
|
||||||
|
{ 9395221924704944128u, 1355252715606880542u }, { 2520655369026404352u, 1694065894508600678u },
|
||||||
|
{ 12374191248137781248u, 2117582368135750847u }, { 14651398557727195136u, 1323488980084844279u },
|
||||||
|
{ 13702562178731606016u, 1654361225106055349u }, { 3293144668132343808u, 2067951531382569187u },
|
||||||
|
{ 18199116482078572544u, 1292469707114105741u }, { 8913837547316051968u, 1615587133892632177u },
|
||||||
|
{ 15753982952572452864u, 2019483917365790221u }, { 12152082354571476992u, 1262177448353618888u },
|
||||||
|
{ 15190102943214346240u, 1577721810442023610u }, { 9764256642163156992u, 1972152263052529513u },
|
||||||
|
{ 17631875447420442880u, 1232595164407830945u }, { 8204786253993389888u, 1540743955509788682u },
|
||||||
|
{ 1032610780636961552u, 1925929944387235853u }, { 2951224747111794922u, 1203706215242022408u },
|
||||||
|
{ 3689030933889743652u, 1504632769052528010u }, { 13834660704216955373u, 1880790961315660012u },
|
||||||
|
{ 17870034976990372916u, 1175494350822287507u }, { 17725857702810578241u, 1469367938527859384u },
|
||||||
|
{ 3710578054803671186u, 1836709923159824231u }, { 26536550077201078u, 2295887403949780289u },
|
||||||
|
{ 11545800389866720434u, 1434929627468612680u }, { 14432250487333400542u, 1793662034335765850u },
|
||||||
|
{ 8816941072311974870u, 2242077542919707313u }, { 17039803216263454053u, 1401298464324817070u },
|
||||||
|
{ 12076381983474541759u, 1751623080406021338u }, { 5872105442488401391u, 2189528850507526673u },
|
||||||
|
{ 15199280947623720629u, 1368455531567204170u }, { 9775729147674874978u, 1710569414459005213u },
|
||||||
|
{ 16831347453020981627u, 2138211768073756516u }, { 1296220121283337709u, 1336382355046097823u },
|
||||||
|
{ 15455333206886335848u, 1670477943807622278u }, { 10095794471753144002u, 2088097429759527848u },
|
||||||
|
{ 6309871544845715001u, 1305060893599704905u }, { 12499025449484531656u, 1631326116999631131u },
|
||||||
|
{ 11012095793428276666u, 2039157646249538914u }, { 11494245889320060820u, 1274473528905961821u },
|
||||||
|
{ 532749306367912313u, 1593091911132452277u }, { 5277622651387278295u, 1991364888915565346u },
|
||||||
|
{ 7910200175544436838u, 1244603055572228341u }, { 14499436237857933952u, 1555753819465285426u },
|
||||||
|
{ 8900923260467641632u, 1944692274331606783u }, { 12480606065433357876u, 1215432671457254239u },
|
||||||
|
{ 10989071563364309441u, 1519290839321567799u }, { 9124653435777998898u, 1899113549151959749u },
|
||||||
|
{ 8008751406574943263u, 1186945968219974843u }, { 5399253239791291175u, 1483682460274968554u },
|
||||||
|
{ 15972438586593889776u, 1854603075343710692u }, { 759402079766405302u, 1159126922089819183u },
|
||||||
|
{ 14784310654990170340u, 1448908652612273978u }, { 9257016281882937117u, 1811135815765342473u },
|
||||||
|
{ 16182956370781059300u, 2263919769706678091u }, { 7808504722524468110u, 1414949856066673807u },
|
||||||
|
{ 5148944884728197234u, 1768687320083342259u }, { 1824495087482858639u, 2210859150104177824u },
|
||||||
|
{ 1140309429676786649u, 1381786968815111140u }, { 1425386787095983311u, 1727233711018888925u },
|
||||||
|
{ 6393419502297367043u, 2159042138773611156u }, { 13219259225790630210u, 1349401336733506972u },
|
||||||
|
{ 16524074032238287762u, 1686751670916883715u }, { 16043406521870471799u, 2108439588646104644u },
|
||||||
|
{ 803757039314269066u, 1317774742903815403u }, { 14839754354425000045u, 1647218428629769253u },
|
||||||
|
{ 4714634887749086344u, 2059023035787211567u }, { 9864175832484260821u, 1286889397367007229u },
|
||||||
|
{ 16941905809032713930u, 1608611746708759036u }, { 2730638187581340797u, 2010764683385948796u },
|
||||||
|
{ 10930020904093113806u, 1256727927116217997u }, { 18274212148543780162u, 1570909908895272496u },
|
||||||
|
{ 4396021111970173586u, 1963637386119090621u }, { 5053356204195052443u, 1227273366324431638u },
|
||||||
|
{ 15540067292098591362u, 1534091707905539547u }, { 14813398096695851299u, 1917614634881924434u },
|
||||||
|
{ 13870059828862294966u, 1198509146801202771u }, { 12725888767650480803u, 1498136433501503464u },
|
||||||
|
{ 15907360959563101004u, 1872670541876879330u }, { 14553786618154326031u, 1170419088673049581u },
|
||||||
|
{ 4357175217410743827u, 1463023860841311977u }, { 10058155040190817688u, 1828779826051639971u },
|
||||||
|
{ 7961007781811134206u, 2285974782564549964u }, { 14199001900486734687u, 1428734239102843727u },
|
||||||
|
{ 13137066357181030455u, 1785917798878554659u }, { 11809646928048900164u, 2232397248598193324u },
|
||||||
|
{ 16604401366885338411u, 1395248280373870827u }, { 16143815690179285109u, 1744060350467338534u },
|
||||||
|
{ 10956397575869330579u, 2180075438084173168u }, { 6847748484918331612u, 1362547148802608230u },
|
||||||
|
{ 17783057643002690323u, 1703183936003260287u }, { 17617136035325974999u, 2128979920004075359u },
|
||||||
|
{ 17928239049719816230u, 1330612450002547099u }, { 17798612793722382384u, 1663265562503183874u },
|
||||||
|
{ 13024893955298202172u, 2079081953128979843u }, { 5834715712847682405u, 1299426220705612402u },
|
||||||
|
{ 16516766677914378815u, 1624282775882015502u }, { 11422586310538197711u, 2030353469852519378u },
|
||||||
|
{ 11750802462513761473u, 1268970918657824611u }, { 10076817059714813937u, 1586213648322280764u },
|
||||||
|
{ 12596021324643517422u, 1982767060402850955u }, { 5566670318688504437u, 1239229412751781847u },
|
||||||
|
{ 2346651879933242642u, 1549036765939727309u }, { 7545000868343941206u, 1936295957424659136u },
|
||||||
|
{ 4715625542714963254u, 1210184973390411960u }, { 5894531928393704067u, 1512731216738014950u },
|
||||||
|
{ 16591536947346905892u, 1890914020922518687u }, { 17287239619732898039u, 1181821263076574179u },
|
||||||
|
{ 16997363506238734644u, 1477276578845717724u }, { 2799960309088866689u, 1846595723557147156u },
|
||||||
|
{ 10973347230035317489u, 1154122327223216972u }, { 13716684037544146861u, 1442652909029021215u },
|
||||||
|
{ 12534169028502795672u, 1803316136286276519u }, { 11056025267201106687u, 2254145170357845649u },
|
||||||
|
{ 18439230838069161439u, 1408840731473653530u }, { 13825666510731675991u, 1761050914342066913u },
|
||||||
|
{ 3447025083132431277u, 2201313642927583642u }, { 6766076695385157452u, 1375821026829739776u },
|
||||||
|
{ 8457595869231446815u, 1719776283537174720u }, { 10571994836539308519u, 2149720354421468400u },
|
||||||
|
{ 6607496772837067824u, 1343575221513417750u }, { 17482743002901110588u, 1679469026891772187u },
|
||||||
|
{ 17241742735199000331u, 2099336283614715234u }, { 15387775227926763111u, 1312085177259197021u },
|
||||||
|
{ 5399660979626290177u, 1640106471573996277u }, { 11361262242960250625u, 2050133089467495346u },
|
||||||
|
{ 11712474920277544544u, 1281333180917184591u }, { 10028907631919542777u, 1601666476146480739u },
|
||||||
|
{ 7924448521472040567u, 2002083095183100924u }, { 14176152362774801162u, 1251301934489438077u },
|
||||||
|
{ 3885132398186337741u, 1564127418111797597u }, { 9468101516160310080u, 1955159272639746996u },
|
||||||
|
{ 15140935484454969608u, 1221974545399841872u }, { 479425281859160394u, 1527468181749802341u },
|
||||||
|
{ 5210967620751338397u, 1909335227187252926u }, { 17091912818251750210u, 1193334516992033078u },
|
||||||
|
{ 12141518985959911954u, 1491668146240041348u }, { 15176898732449889943u, 1864585182800051685u },
|
||||||
|
{ 11791404716994875166u, 1165365739250032303u }, { 10127569877816206054u, 1456707174062540379u },
|
||||||
|
{ 8047776328842869663u, 1820883967578175474u }, { 836348374198811271u, 2276104959472719343u },
|
||||||
|
{ 7440246761515338900u, 1422565599670449589u }, { 13911994470321561530u, 1778206999588061986u },
|
||||||
|
{ 8166621051047176104u, 2222758749485077483u }, { 2798295147690791113u, 1389224218428173427u },
|
||||||
|
{ 17332926989895652603u, 1736530273035216783u }, { 17054472718942177850u, 2170662841294020979u },
|
||||||
|
{ 8353202440125167204u, 1356664275808763112u }, { 10441503050156459005u, 1695830344760953890u },
|
||||||
|
{ 3828506775840797949u, 2119787930951192363u }, { 86973725686804766u, 1324867456844495227u },
|
||||||
|
{ 13943775212390669669u, 1656084321055619033u }, { 3594660960206173375u, 2070105401319523792u },
|
||||||
|
{ 2246663100128858359u, 1293815875824702370u }, { 12031700912015848757u, 1617269844780877962u },
|
||||||
|
{ 5816254103165035138u, 2021587305976097453u }, { 5941001823691840913u, 1263492066235060908u },
|
||||||
|
{ 7426252279614801142u, 1579365082793826135u }, { 4671129331091113523u, 1974206353492282669u },
|
||||||
|
{ 5225298841145639904u, 1233878970932676668u }, { 6531623551432049880u, 1542348713665845835u },
|
||||||
|
{ 3552843420862674446u, 1927935892082307294u }, { 16055585193321335241u, 1204959932551442058u },
|
||||||
|
{ 10846109454796893243u, 1506199915689302573u }, { 18169322836923504458u, 1882749894611628216u },
|
||||||
|
{ 11355826773077190286u, 1176718684132267635u }, { 9583097447919099954u, 1470898355165334544u },
|
||||||
|
{ 11978871809898874942u, 1838622943956668180u }, { 14973589762373593678u, 2298278679945835225u },
|
||||||
|
{ 2440964573842414192u, 1436424174966147016u }, { 3051205717303017741u, 1795530218707683770u },
|
||||||
|
{ 13037379183483547984u, 2244412773384604712u }, { 8148361989677217490u, 1402757983365377945u },
|
||||||
|
{ 14797138505523909766u, 1753447479206722431u }, { 13884737113477499304u, 2191809349008403039u },
|
||||||
|
{ 15595489723564518921u, 1369880843130251899u }, { 14882676136028260747u, 1712351053912814874u },
|
||||||
|
{ 9379973133180550126u, 2140438817391018593u }, { 17391698254306313589u, 1337774260869386620u },
|
||||||
|
{ 3292878744173340370u, 1672217826086733276u }, { 4116098430216675462u, 2090272282608416595u },
|
||||||
|
{ 266718509671728212u, 1306420176630260372u }, { 333398137089660265u, 1633025220787825465u },
|
||||||
|
{ 5028433689789463235u, 2041281525984781831u }, { 10060300083759496378u, 1275800953740488644u },
|
||||||
|
{ 12575375104699370472u, 1594751192175610805u }, { 1884160825592049379u, 1993438990219513507u },
|
||||||
|
{ 17318501580490888525u, 1245899368887195941u }, { 7813068920331446945u, 1557374211108994927u },
|
||||||
|
{ 5154650131986920777u, 1946717763886243659u }, { 915813323278131534u, 1216698602428902287u },
|
||||||
|
{ 14979824709379828129u, 1520873253036127858u }, { 9501408849870009354u, 1901091566295159823u },
|
||||||
|
{ 12855909558809837702u, 1188182228934474889u }, { 2234828893230133415u, 1485227786168093612u },
|
||||||
|
{ 2793536116537666769u, 1856534732710117015u }, { 8663489100477123587u, 1160334207943823134u },
|
||||||
|
{ 1605989338741628675u, 1450417759929778918u }, { 11230858710281811652u, 1813022199912223647u },
|
||||||
|
{ 9426887369424876662u, 2266277749890279559u }, { 12809333633531629769u, 1416423593681424724u },
|
||||||
|
{ 16011667041914537212u, 1770529492101780905u }, { 6179525747111007803u, 2213161865127226132u },
|
||||||
|
{ 13085575628799155685u, 1383226165704516332u }, { 16356969535998944606u, 1729032707130645415u },
|
||||||
|
{ 15834525901571292854u, 2161290883913306769u }, { 2979049660840976177u, 1350806802445816731u },
|
||||||
|
{ 17558870131333383934u, 1688508503057270913u }, { 8113529608884566205u, 2110635628821588642u },
|
||||||
|
{ 9682642023980241782u, 1319147268013492901u }, { 16714988548402690132u, 1648934085016866126u },
|
||||||
|
{ 11670363648648586857u, 2061167606271082658u }, { 11905663298832754689u, 1288229753919426661u },
|
||||||
|
{ 1047021068258779650u, 1610287192399283327u }, { 15143834390605638274u, 2012858990499104158u },
|
||||||
|
{ 4853210475701136017u, 1258036869061940099u }, { 1454827076199032118u, 1572546086327425124u },
|
||||||
|
{ 1818533845248790147u, 1965682607909281405u }, { 3442426662494187794u, 1228551629943300878u },
|
||||||
|
{ 13526405364972510550u, 1535689537429126097u }, { 3072948650933474476u, 1919611921786407622u },
|
||||||
|
{ 15755650962115585259u, 1199757451116504763u }, { 15082877684217093670u, 1499696813895630954u },
|
||||||
|
{ 9630225068416591280u, 1874621017369538693u }, { 8324733676974063502u, 1171638135855961683u },
|
||||||
|
{ 5794231077790191473u, 1464547669819952104u }, { 7242788847237739342u, 1830684587274940130u },
|
||||||
|
{ 18276858095901949986u, 2288355734093675162u }, { 16034722328366106645u, 1430222333808546976u },
|
||||||
|
{ 1596658836748081690u, 1787777917260683721u }, { 6607509564362490017u, 2234722396575854651u },
|
||||||
|
{ 1823850468512862308u, 1396701497859909157u }, { 6891499104068465790u, 1745876872324886446u },
|
||||||
|
{ 17837745916940358045u, 2182346090406108057u }, { 4231062170446641922u, 1363966306503817536u },
|
||||||
|
{ 5288827713058302403u, 1704957883129771920u }, { 6611034641322878003u, 2131197353912214900u },
|
||||||
|
{ 13355268687681574560u, 1331998346195134312u }, { 16694085859601968200u, 1664997932743917890u },
|
||||||
|
{ 11644235287647684442u, 2081247415929897363u }, { 4971804045566108824u, 1300779634956185852u },
|
||||||
|
{ 6214755056957636030u, 1625974543695232315u }, { 3156757802769657134u, 2032468179619040394u },
|
||||||
|
{ 6584659645158423613u, 1270292612261900246u }, { 17454196593302805324u, 1587865765327375307u },
|
||||||
|
{ 17206059723201118751u, 1984832206659219134u }, { 6142101308573311315u, 1240520129162011959u },
|
||||||
|
{ 3065940617289251240u, 1550650161452514949u }, { 8444111790038951954u, 1938312701815643686u },
|
||||||
|
{ 665883850346957067u, 1211445438634777304u }, { 832354812933696334u, 1514306798293471630u },
|
||||||
|
{ 10263815553021896226u, 1892883497866839537u }, { 17944099766707154901u, 1183052186166774710u },
|
||||||
|
{ 13206752671529167818u, 1478815232708468388u }, { 16508440839411459773u, 1848519040885585485u },
|
||||||
|
{ 12623618533845856310u, 1155324400553490928u }, { 15779523167307320387u, 1444155500691863660u },
|
||||||
|
{ 1277659885424598868u, 1805194375864829576u }, { 1597074856780748586u, 2256492969831036970u },
|
||||||
|
{ 5609857803915355770u, 1410308106144398106u }, { 16235694291748970521u, 1762885132680497632u },
|
||||||
|
{ 1847873790976661535u, 2203606415850622041u }, { 12684136165428883219u, 1377254009906638775u },
|
||||||
|
{ 11243484188358716120u, 1721567512383298469u }, { 219297180166231438u, 2151959390479123087u },
|
||||||
|
{ 7054589765244976505u, 1344974619049451929u }, { 13429923224983608535u, 1681218273811814911u },
|
||||||
|
{ 12175718012802122765u, 2101522842264768639u }, { 14527352785642408584u, 1313451776415480399u },
|
||||||
|
{ 13547504963625622826u, 1641814720519350499u }, { 12322695186104640628u, 2052268400649188124u },
|
||||||
|
{ 16925056528170176201u, 1282667750405742577u }, { 7321262604930556539u, 1603334688007178222u },
|
||||||
|
{ 18374950293017971482u, 2004168360008972777u }, { 4566814905495150320u, 1252605225005607986u },
|
||||||
|
{ 14931890668723713708u, 1565756531257009982u }, { 9441491299049866327u, 1957195664071262478u },
|
||||||
|
{ 1289246043478778550u, 1223247290044539049u }, { 6223243572775861092u, 1529059112555673811u },
|
||||||
|
{ 3167368447542438461u, 1911323890694592264u }, { 1979605279714024038u, 1194577431684120165u },
|
||||||
|
{ 7086192618069917952u, 1493221789605150206u }, { 18081112809442173248u, 1866527237006437757u },
|
||||||
|
{ 13606538515115052232u, 1166579523129023598u }, { 7784801107039039482u, 1458224403911279498u },
|
||||||
|
{ 507629346944023544u, 1822780504889099373u }, { 5246222702107417334u, 2278475631111374216u },
|
||||||
|
{ 3278889188817135834u, 1424047269444608885u }, { 8710297504448807696u, 1780059086805761106u }
|
||||||
|
};
|
||||||
|
|
||||||
|
#endif // RYU_D2S_FULL_TABLE_H
|
||||||
@@ -0,0 +1,357 @@
|
|||||||
|
// Copyright 2018 Ulf Adams
|
||||||
|
//
|
||||||
|
// The contents of this file may be used under the terms of the Apache License,
|
||||||
|
// Version 2.0.
|
||||||
|
//
|
||||||
|
// (See accompanying file LICENSE-Apache or copy at
|
||||||
|
// http://www.apache.org/licenses/LICENSE-2.0)
|
||||||
|
//
|
||||||
|
// Alternatively, the contents of this file may be used under the terms of
|
||||||
|
// the Boost Software License, Version 1.0.
|
||||||
|
// (See accompanying file LICENSE-Boost or copy at
|
||||||
|
// https://www.boost.org/LICENSE_1_0.txt)
|
||||||
|
//
|
||||||
|
// Unless required by applicable law or agreed to in writing, this software
|
||||||
|
// is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
|
||||||
|
// KIND, either express or implied.
|
||||||
|
#ifndef RYU_D2S_INTRINSICS_H
|
||||||
|
#define RYU_D2S_INTRINSICS_H
|
||||||
|
|
||||||
|
#include <assert.h>
|
||||||
|
#include <stdint.h>
|
||||||
|
|
||||||
|
// Defines RYU_32_BIT_PLATFORM if applicable.
|
||||||
|
#include "ryu/common.h"
|
||||||
|
|
||||||
|
// ABSL avoids uint128_t on Win32 even if __SIZEOF_INT128__ is defined.
|
||||||
|
// Let's do the same for now.
|
||||||
|
#if defined(__SIZEOF_INT128__) && !defined(_MSC_VER) && !defined(RYU_ONLY_64_BIT_OPS)
|
||||||
|
#define HAS_UINT128
|
||||||
|
#elif defined(_MSC_VER) && !defined(RYU_ONLY_64_BIT_OPS) && defined(_M_X64)
|
||||||
|
#define HAS_64_BIT_INTRINSICS
|
||||||
|
#endif
|
||||||
|
|
||||||
|
#if defined(HAS_UINT128)
|
||||||
|
typedef __uint128_t uint128_t;
|
||||||
|
#endif
|
||||||
|
|
||||||
|
#if defined(HAS_64_BIT_INTRINSICS)
|
||||||
|
|
||||||
|
#include <intrin.h>
|
||||||
|
|
||||||
|
static inline uint64_t umul128(const uint64_t a, const uint64_t b, uint64_t* const productHi) {
|
||||||
|
return _umul128(a, b, productHi);
|
||||||
|
}
|
||||||
|
|
||||||
|
// Returns the lower 64 bits of (hi*2^64 + lo) >> dist, with 0 < dist < 64.
|
||||||
|
static inline uint64_t shiftright128(const uint64_t lo, const uint64_t hi, const uint32_t dist) {
|
||||||
|
// For the __shiftright128 intrinsic, the shift value is always
|
||||||
|
// modulo 64.
|
||||||
|
// In the current implementation of the double-precision version
|
||||||
|
// of Ryu, the shift value is always < 64. (In the case
|
||||||
|
// RYU_OPTIMIZE_SIZE == 0, the shift value is in the range [49, 58].
|
||||||
|
// Otherwise in the range [2, 59].)
|
||||||
|
// However, this function is now also called by s2d, which requires supporting
|
||||||
|
// the larger shift range (TODO: what is the actual range?).
|
||||||
|
// Check this here in case a future change requires larger shift
|
||||||
|
// values. In this case this function needs to be adjusted.
|
||||||
|
assert(dist < 64);
|
||||||
|
return __shiftright128(lo, hi, (unsigned char) dist);
|
||||||
|
}
|
||||||
|
|
||||||
|
#else // defined(HAS_64_BIT_INTRINSICS)
|
||||||
|
|
||||||
|
static inline uint64_t umul128(const uint64_t a, const uint64_t b, uint64_t* const productHi) {
|
||||||
|
// The casts here help MSVC to avoid calls to the __allmul library function.
|
||||||
|
const uint32_t aLo = (uint32_t)a;
|
||||||
|
const uint32_t aHi = (uint32_t)(a >> 32);
|
||||||
|
const uint32_t bLo = (uint32_t)b;
|
||||||
|
const uint32_t bHi = (uint32_t)(b >> 32);
|
||||||
|
|
||||||
|
const uint64_t b00 = (uint64_t)aLo * bLo;
|
||||||
|
const uint64_t b01 = (uint64_t)aLo * bHi;
|
||||||
|
const uint64_t b10 = (uint64_t)aHi * bLo;
|
||||||
|
const uint64_t b11 = (uint64_t)aHi * bHi;
|
||||||
|
|
||||||
|
const uint32_t b00Lo = (uint32_t)b00;
|
||||||
|
const uint32_t b00Hi = (uint32_t)(b00 >> 32);
|
||||||
|
|
||||||
|
const uint64_t mid1 = b10 + b00Hi;
|
||||||
|
const uint32_t mid1Lo = (uint32_t)(mid1);
|
||||||
|
const uint32_t mid1Hi = (uint32_t)(mid1 >> 32);
|
||||||
|
|
||||||
|
const uint64_t mid2 = b01 + mid1Lo;
|
||||||
|
const uint32_t mid2Lo = (uint32_t)(mid2);
|
||||||
|
const uint32_t mid2Hi = (uint32_t)(mid2 >> 32);
|
||||||
|
|
||||||
|
const uint64_t pHi = b11 + mid1Hi + mid2Hi;
|
||||||
|
const uint64_t pLo = ((uint64_t)mid2Lo << 32) | b00Lo;
|
||||||
|
|
||||||
|
*productHi = pHi;
|
||||||
|
return pLo;
|
||||||
|
}
|
||||||
|
|
||||||
|
static inline uint64_t shiftright128(const uint64_t lo, const uint64_t hi, const uint32_t dist) {
|
||||||
|
// We don't need to handle the case dist >= 64 here (see above).
|
||||||
|
assert(dist < 64);
|
||||||
|
assert(dist > 0);
|
||||||
|
return (hi << (64 - dist)) | (lo >> dist);
|
||||||
|
}
|
||||||
|
|
||||||
|
#endif // defined(HAS_64_BIT_INTRINSICS)
|
||||||
|
|
||||||
|
#if defined(RYU_32_BIT_PLATFORM)
|
||||||
|
|
||||||
|
// Returns the high 64 bits of the 128-bit product of a and b.
|
||||||
|
static inline uint64_t umulh(const uint64_t a, const uint64_t b) {
|
||||||
|
// Reuse the umul128 implementation.
|
||||||
|
// Optimizers will likely eliminate the instructions used to compute the
|
||||||
|
// low part of the product.
|
||||||
|
uint64_t hi;
|
||||||
|
umul128(a, b, &hi);
|
||||||
|
return hi;
|
||||||
|
}
|
||||||
|
|
||||||
|
// On 32-bit platforms, compilers typically generate calls to library
|
||||||
|
// functions for 64-bit divisions, even if the divisor is a constant.
|
||||||
|
//
|
||||||
|
// E.g.:
|
||||||
|
// https://bugs.llvm.org/show_bug.cgi?id=37932
|
||||||
|
// https://gcc.gnu.org/bugzilla/show_bug.cgi?id=17958
|
||||||
|
// https://gcc.gnu.org/bugzilla/show_bug.cgi?id=37443
|
||||||
|
//
|
||||||
|
// The functions here perform division-by-constant using multiplications
|
||||||
|
// in the same way as 64-bit compilers would do.
|
||||||
|
//
|
||||||
|
// NB:
|
||||||
|
// The multipliers and shift values are the ones generated by clang x64
|
||||||
|
// for expressions like x/5, x/10, etc.
|
||||||
|
|
||||||
|
static inline uint64_t div5(const uint64_t x) {
|
||||||
|
return umulh(x, 0xCCCCCCCCCCCCCCCDu) >> 2;
|
||||||
|
}
|
||||||
|
|
||||||
|
static inline uint64_t div10(const uint64_t x) {
|
||||||
|
return umulh(x, 0xCCCCCCCCCCCCCCCDu) >> 3;
|
||||||
|
}
|
||||||
|
|
||||||
|
static inline uint64_t div100(const uint64_t x) {
|
||||||
|
return umulh(x >> 2, 0x28F5C28F5C28F5C3u) >> 2;
|
||||||
|
}
|
||||||
|
|
||||||
|
static inline uint64_t div1e8(const uint64_t x) {
|
||||||
|
return umulh(x, 0xABCC77118461CEFDu) >> 26;
|
||||||
|
}
|
||||||
|
|
||||||
|
static inline uint64_t div1e9(const uint64_t x) {
|
||||||
|
return umulh(x >> 9, 0x44B82FA09B5A53u) >> 11;
|
||||||
|
}
|
||||||
|
|
||||||
|
static inline uint32_t mod1e9(const uint64_t x) {
|
||||||
|
// Avoid 64-bit math as much as possible.
|
||||||
|
// Returning (uint32_t) (x - 1000000000 * div1e9(x)) would
|
||||||
|
// perform 32x64-bit multiplication and 64-bit subtraction.
|
||||||
|
// x and 1000000000 * div1e9(x) are guaranteed to differ by
|
||||||
|
// less than 10^9, so their highest 32 bits must be identical,
|
||||||
|
// so we can truncate both sides to uint32_t before subtracting.
|
||||||
|
// We can also simplify (uint32_t) (1000000000 * div1e9(x)).
|
||||||
|
// We can truncate before multiplying instead of after, as multiplying
|
||||||
|
// the highest 32 bits of div1e9(x) can't affect the lowest 32 bits.
|
||||||
|
return ((uint32_t) x) - 1000000000 * ((uint32_t) div1e9(x));
|
||||||
|
}
|
||||||
|
|
||||||
|
#else // defined(RYU_32_BIT_PLATFORM)
|
||||||
|
|
||||||
|
static inline uint64_t div5(const uint64_t x) {
|
||||||
|
return x / 5;
|
||||||
|
}
|
||||||
|
|
||||||
|
static inline uint64_t div10(const uint64_t x) {
|
||||||
|
return x / 10;
|
||||||
|
}
|
||||||
|
|
||||||
|
static inline uint64_t div100(const uint64_t x) {
|
||||||
|
return x / 100;
|
||||||
|
}
|
||||||
|
|
||||||
|
static inline uint64_t div1e8(const uint64_t x) {
|
||||||
|
return x / 100000000;
|
||||||
|
}
|
||||||
|
|
||||||
|
static inline uint64_t div1e9(const uint64_t x) {
|
||||||
|
return x / 1000000000;
|
||||||
|
}
|
||||||
|
|
||||||
|
static inline uint32_t mod1e9(const uint64_t x) {
|
||||||
|
return (uint32_t) (x - 1000000000 * div1e9(x));
|
||||||
|
}
|
||||||
|
|
||||||
|
#endif // defined(RYU_32_BIT_PLATFORM)
|
||||||
|
|
||||||
|
static inline uint32_t pow5Factor(uint64_t value) {
|
||||||
|
const uint64_t m_inv_5 = 14757395258967641293u; // 5 * m_inv_5 = 1 (mod 2^64)
|
||||||
|
const uint64_t n_div_5 = 3689348814741910323u; // #{ n | n = 0 (mod 2^64) } = 2^64 / 5
|
||||||
|
uint32_t count = 0;
|
||||||
|
for (;;) {
|
||||||
|
assert(value != 0);
|
||||||
|
value *= m_inv_5;
|
||||||
|
if (value > n_div_5)
|
||||||
|
break;
|
||||||
|
++count;
|
||||||
|
}
|
||||||
|
return count;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Returns true if value is divisible by 5^p.
|
||||||
|
static inline bool multipleOfPowerOf5(const uint64_t value, const uint32_t p) {
|
||||||
|
// I tried a case distinction on p, but there was no performance difference.
|
||||||
|
return pow5Factor(value) >= p;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Returns true if value is divisible by 2^p.
|
||||||
|
static inline bool multipleOfPowerOf2(const uint64_t value, const uint32_t p) {
|
||||||
|
assert(value != 0);
|
||||||
|
assert(p < 64);
|
||||||
|
// __builtin_ctzll doesn't appear to be faster here.
|
||||||
|
return (value & ((1ull << p) - 1)) == 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
// We need a 64x128-bit multiplication and a subsequent 128-bit shift.
|
||||||
|
// Multiplication:
|
||||||
|
// The 64-bit factor is variable and passed in, the 128-bit factor comes
|
||||||
|
// from a lookup table. We know that the 64-bit factor only has 55
|
||||||
|
// significant bits (i.e., the 9 topmost bits are zeros). The 128-bit
|
||||||
|
// factor only has 124 significant bits (i.e., the 4 topmost bits are
|
||||||
|
// zeros).
|
||||||
|
// Shift:
|
||||||
|
// In principle, the multiplication result requires 55 + 124 = 179 bits to
|
||||||
|
// represent. However, we then shift this value to the right by j, which is
|
||||||
|
// at least j >= 115, so the result is guaranteed to fit into 179 - 115 = 64
|
||||||
|
// bits. This means that we only need the topmost 64 significant bits of
|
||||||
|
// the 64x128-bit multiplication.
|
||||||
|
//
|
||||||
|
// There are several ways to do this:
|
||||||
|
// 1. Best case: the compiler exposes a 128-bit type.
|
||||||
|
// We perform two 64x64-bit multiplications, add the higher 64 bits of the
|
||||||
|
// lower result to the higher result, and shift by j - 64 bits.
|
||||||
|
//
|
||||||
|
// We explicitly cast from 64-bit to 128-bit, so the compiler can tell
|
||||||
|
// that these are only 64-bit inputs, and can map these to the best
|
||||||
|
// possible sequence of assembly instructions.
|
||||||
|
// x64 machines happen to have matching assembly instructions for
|
||||||
|
// 64x64-bit multiplications and 128-bit shifts.
|
||||||
|
//
|
||||||
|
// 2. Second best case: the compiler exposes intrinsics for the x64 assembly
|
||||||
|
// instructions mentioned in 1.
|
||||||
|
//
|
||||||
|
// 3. We only have 64x64 bit instructions that return the lower 64 bits of
|
||||||
|
// the result, i.e., we have to use plain C.
|
||||||
|
// Our inputs are less than the full width, so we have three options:
|
||||||
|
// a. Ignore this fact and just implement the intrinsics manually.
|
||||||
|
// b. Split both into 31-bit pieces, which guarantees no internal overflow,
|
||||||
|
// but requires extra work upfront (unless we change the lookup table).
|
||||||
|
// c. Split only the first factor into 31-bit pieces, which also guarantees
|
||||||
|
// no internal overflow, but requires extra work since the intermediate
|
||||||
|
// results are not perfectly aligned.
|
||||||
|
#if defined(HAS_UINT128)
|
||||||
|
|
||||||
|
// Best case: use 128-bit type.
|
||||||
|
static inline uint64_t mulShift64(const uint64_t m, const uint64_t* const mul, const int32_t j) {
|
||||||
|
const uint128_t b0 = ((uint128_t) m) * mul[0];
|
||||||
|
const uint128_t b2 = ((uint128_t) m) * mul[1];
|
||||||
|
return (uint64_t) (((b0 >> 64) + b2) >> (j - 64));
|
||||||
|
}
|
||||||
|
|
||||||
|
static inline uint64_t mulShiftAll64(const uint64_t m, const uint64_t* const mul, const int32_t j,
|
||||||
|
uint64_t* const vp, uint64_t* const vm, const uint32_t mmShift) {
|
||||||
|
// m <<= 2;
|
||||||
|
// uint128_t b0 = ((uint128_t) m) * mul[0]; // 0
|
||||||
|
// uint128_t b2 = ((uint128_t) m) * mul[1]; // 64
|
||||||
|
//
|
||||||
|
// uint128_t hi = (b0 >> 64) + b2;
|
||||||
|
// uint128_t lo = b0 & 0xffffffffffffffffull;
|
||||||
|
// uint128_t factor = (((uint128_t) mul[1]) << 64) + mul[0];
|
||||||
|
// uint128_t vpLo = lo + (factor << 1);
|
||||||
|
// *vp = (uint64_t) ((hi + (vpLo >> 64)) >> (j - 64));
|
||||||
|
// uint128_t vmLo = lo - (factor << mmShift);
|
||||||
|
// *vm = (uint64_t) ((hi + (vmLo >> 64) - (((uint128_t) 1ull) << 64)) >> (j - 64));
|
||||||
|
// return (uint64_t) (hi >> (j - 64));
|
||||||
|
*vp = mulShift64(4 * m + 2, mul, j);
|
||||||
|
*vm = mulShift64(4 * m - 1 - mmShift, mul, j);
|
||||||
|
return mulShift64(4 * m, mul, j);
|
||||||
|
}
|
||||||
|
|
||||||
|
#elif defined(HAS_64_BIT_INTRINSICS)
|
||||||
|
|
||||||
|
static inline uint64_t mulShift64(const uint64_t m, const uint64_t* const mul, const int32_t j) {
|
||||||
|
// m is maximum 55 bits
|
||||||
|
uint64_t high1; // 128
|
||||||
|
const uint64_t low1 = umul128(m, mul[1], &high1); // 64
|
||||||
|
uint64_t high0; // 64
|
||||||
|
umul128(m, mul[0], &high0); // 0
|
||||||
|
const uint64_t sum = high0 + low1;
|
||||||
|
if (sum < high0) {
|
||||||
|
++high1; // overflow into high1
|
||||||
|
}
|
||||||
|
return shiftright128(sum, high1, j - 64);
|
||||||
|
}
|
||||||
|
|
||||||
|
static inline uint64_t mulShiftAll64(const uint64_t m, const uint64_t* const mul, const int32_t j,
|
||||||
|
uint64_t* const vp, uint64_t* const vm, const uint32_t mmShift) {
|
||||||
|
*vp = mulShift64(4 * m + 2, mul, j);
|
||||||
|
*vm = mulShift64(4 * m - 1 - mmShift, mul, j);
|
||||||
|
return mulShift64(4 * m, mul, j);
|
||||||
|
}
|
||||||
|
|
||||||
|
#else // !defined(HAS_UINT128) && !defined(HAS_64_BIT_INTRINSICS)
|
||||||
|
|
||||||
|
static inline uint64_t mulShift64(const uint64_t m, const uint64_t* const mul, const int32_t j) {
|
||||||
|
// m is maximum 55 bits
|
||||||
|
uint64_t high1; // 128
|
||||||
|
const uint64_t low1 = umul128(m, mul[1], &high1); // 64
|
||||||
|
uint64_t high0; // 64
|
||||||
|
umul128(m, mul[0], &high0); // 0
|
||||||
|
const uint64_t sum = high0 + low1;
|
||||||
|
if (sum < high0) {
|
||||||
|
++high1; // overflow into high1
|
||||||
|
}
|
||||||
|
return shiftright128(sum, high1, j - 64);
|
||||||
|
}
|
||||||
|
|
||||||
|
// This is faster if we don't have a 64x64->128-bit multiplication.
|
||||||
|
static inline uint64_t mulShiftAll64(uint64_t m, const uint64_t* const mul, const int32_t j,
|
||||||
|
uint64_t* const vp, uint64_t* const vm, const uint32_t mmShift) {
|
||||||
|
m <<= 1;
|
||||||
|
// m is maximum 55 bits
|
||||||
|
uint64_t tmp;
|
||||||
|
const uint64_t lo = umul128(m, mul[0], &tmp);
|
||||||
|
uint64_t hi;
|
||||||
|
const uint64_t mid = tmp + umul128(m, mul[1], &hi);
|
||||||
|
hi += mid < tmp; // overflow into hi
|
||||||
|
|
||||||
|
const uint64_t lo2 = lo + mul[0];
|
||||||
|
const uint64_t mid2 = mid + mul[1] + (lo2 < lo);
|
||||||
|
const uint64_t hi2 = hi + (mid2 < mid);
|
||||||
|
*vp = shiftright128(mid2, hi2, (uint32_t) (j - 64 - 1));
|
||||||
|
|
||||||
|
if (mmShift == 1) {
|
||||||
|
const uint64_t lo3 = lo - mul[0];
|
||||||
|
const uint64_t mid3 = mid - mul[1] - (lo3 > lo);
|
||||||
|
const uint64_t hi3 = hi - (mid3 > mid);
|
||||||
|
*vm = shiftright128(mid3, hi3, (uint32_t) (j - 64 - 1));
|
||||||
|
} else {
|
||||||
|
const uint64_t lo3 = lo + lo;
|
||||||
|
const uint64_t mid3 = mid + mid + (lo3 < lo);
|
||||||
|
const uint64_t hi3 = hi + hi + (mid3 < mid);
|
||||||
|
const uint64_t lo4 = lo3 - mul[0];
|
||||||
|
const uint64_t mid4 = mid3 - mul[1] - (lo4 > lo3);
|
||||||
|
const uint64_t hi4 = hi3 - (mid4 > mid3);
|
||||||
|
*vm = shiftright128(mid4, hi4, (uint32_t) (j - 64));
|
||||||
|
}
|
||||||
|
|
||||||
|
return shiftright128(mid, hi, (uint32_t) (j - 64 - 1));
|
||||||
|
}
|
||||||
|
|
||||||
|
#endif // HAS_64_BIT_INTRINSICS
|
||||||
|
|
||||||
|
#endif // RYU_D2S_INTRINSICS_H
|
||||||
@@ -0,0 +1,186 @@
|
|||||||
|
// Copyright 2018 Ulf Adams
|
||||||
|
//
|
||||||
|
// The contents of this file may be used under the terms of the Apache License,
|
||||||
|
// Version 2.0.
|
||||||
|
//
|
||||||
|
// (See accompanying file LICENSE-Apache or copy at
|
||||||
|
// http://www.apache.org/licenses/LICENSE-2.0)
|
||||||
|
//
|
||||||
|
// Alternatively, the contents of this file may be used under the terms of
|
||||||
|
// the Boost Software License, Version 1.0.
|
||||||
|
// (See accompanying file LICENSE-Boost or copy at
|
||||||
|
// https://www.boost.org/LICENSE_1_0.txt)
|
||||||
|
//
|
||||||
|
// Unless required by applicable law or agreed to in writing, this software
|
||||||
|
// is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
|
||||||
|
// KIND, either express or implied.
|
||||||
|
#ifndef RYU_D2S_SMALL_TABLE_H
|
||||||
|
#define RYU_D2S_SMALL_TABLE_H
|
||||||
|
|
||||||
|
#include <assert.h>
|
||||||
|
|
||||||
|
// Defines HAS_UINT128 and uint128_t if applicable.
|
||||||
|
#include "ryu/d2s_intrinsics.h"
|
||||||
|
|
||||||
|
// These tables are generated by PrintDoubleLookupTable.
|
||||||
|
#define DOUBLE_POW5_INV_BITCOUNT 125
|
||||||
|
#define DOUBLE_POW5_BITCOUNT 125
|
||||||
|
|
||||||
|
static const uint64_t DOUBLE_POW5_INV_SPLIT2[15][2] = {
|
||||||
|
{ 1u, 2305843009213693952u },
|
||||||
|
{ 5955668970331000884u, 1784059615882449851u },
|
||||||
|
{ 8982663654677661702u, 1380349269358112757u },
|
||||||
|
{ 7286864317269821294u, 2135987035920910082u },
|
||||||
|
{ 7005857020398200553u, 1652639921975621497u },
|
||||||
|
{ 17965325103354776697u, 1278668206209430417u },
|
||||||
|
{ 8928596168509315048u, 1978643211784836272u },
|
||||||
|
{ 10075671573058298858u, 1530901034580419511u },
|
||||||
|
{ 597001226353042382u, 1184477304306571148u },
|
||||||
|
{ 1527430471115325346u, 1832889850782397517u },
|
||||||
|
{ 12533209867169019542u, 1418129833677084982u },
|
||||||
|
{ 5577825024675947042u, 2194449627517475473u },
|
||||||
|
{ 11006974540203867551u, 1697873161311732311u },
|
||||||
|
{ 10313493231639821582u, 1313665730009899186u },
|
||||||
|
{ 12701016819766672773u, 2032799256770390445u }
|
||||||
|
};
|
||||||
|
static const uint32_t POW5_INV_OFFSETS[22] = {
|
||||||
|
0x54544554, 0x04055545, 0x10041000, 0x00400414, 0x40010000, 0x41155555,
|
||||||
|
0x00000454, 0x00010044, 0x40000000, 0x44000041, 0x50454450, 0x55550054,
|
||||||
|
0x51655554, 0x40004000, 0x01000001, 0x00010500, 0x51515411, 0x05555554,
|
||||||
|
0x50411500, 0x40040000, 0x05040110, 0x00000000
|
||||||
|
};
|
||||||
|
|
||||||
|
static const uint64_t DOUBLE_POW5_SPLIT2[13][2] = {
|
||||||
|
{ 0u, 1152921504606846976u },
|
||||||
|
{ 0u, 1490116119384765625u },
|
||||||
|
{ 1032610780636961552u, 1925929944387235853u },
|
||||||
|
{ 7910200175544436838u, 1244603055572228341u },
|
||||||
|
{ 16941905809032713930u, 1608611746708759036u },
|
||||||
|
{ 13024893955298202172u, 2079081953128979843u },
|
||||||
|
{ 6607496772837067824u, 1343575221513417750u },
|
||||||
|
{ 17332926989895652603u, 1736530273035216783u },
|
||||||
|
{ 13037379183483547984u, 2244412773384604712u },
|
||||||
|
{ 1605989338741628675u, 1450417759929778918u },
|
||||||
|
{ 9630225068416591280u, 1874621017369538693u },
|
||||||
|
{ 665883850346957067u, 1211445438634777304u },
|
||||||
|
{ 14931890668723713708u, 1565756531257009982u }
|
||||||
|
};
|
||||||
|
static const uint32_t POW5_OFFSETS[21] = {
|
||||||
|
0x00000000, 0x00000000, 0x00000000, 0x00000000, 0x40000000, 0x59695995,
|
||||||
|
0x55545555, 0x56555515, 0x41150504, 0x40555410, 0x44555145, 0x44504540,
|
||||||
|
0x45555550, 0x40004000, 0x96440440, 0x55565565, 0x54454045, 0x40154151,
|
||||||
|
0x55559155, 0x51405555, 0x00000105
|
||||||
|
};
|
||||||
|
|
||||||
|
#define POW5_TABLE_SIZE 26
|
||||||
|
static const uint64_t DOUBLE_POW5_TABLE[POW5_TABLE_SIZE] = {
|
||||||
|
1ull, 5ull, 25ull, 125ull, 625ull, 3125ull, 15625ull, 78125ull, 390625ull,
|
||||||
|
1953125ull, 9765625ull, 48828125ull, 244140625ull, 1220703125ull, 6103515625ull,
|
||||||
|
30517578125ull, 152587890625ull, 762939453125ull, 3814697265625ull,
|
||||||
|
19073486328125ull, 95367431640625ull, 476837158203125ull,
|
||||||
|
2384185791015625ull, 11920928955078125ull, 59604644775390625ull,
|
||||||
|
298023223876953125ull //, 1490116119384765625ull
|
||||||
|
};
|
||||||
|
|
||||||
|
#if defined(HAS_UINT128)
|
||||||
|
|
||||||
|
// Computes 5^i in the form required by Ryu, and stores it in the given pointer.
|
||||||
|
static inline void double_computePow5(const uint32_t i, uint64_t* const result) {
|
||||||
|
const uint32_t base = i / POW5_TABLE_SIZE;
|
||||||
|
const uint32_t base2 = base * POW5_TABLE_SIZE;
|
||||||
|
const uint32_t offset = i - base2;
|
||||||
|
const uint64_t* const mul = DOUBLE_POW5_SPLIT2[base];
|
||||||
|
if (offset == 0) {
|
||||||
|
result[0] = mul[0];
|
||||||
|
result[1] = mul[1];
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
const uint64_t m = DOUBLE_POW5_TABLE[offset];
|
||||||
|
const uint128_t b0 = ((uint128_t) m) * mul[0];
|
||||||
|
const uint128_t b2 = ((uint128_t) m) * mul[1];
|
||||||
|
const uint32_t delta = pow5bits(i) - pow5bits(base2);
|
||||||
|
const uint128_t shiftedSum = (b0 >> delta) + (b2 << (64 - delta)) + ((POW5_OFFSETS[i / 16] >> ((i % 16) << 1)) & 3);
|
||||||
|
result[0] = (uint64_t) shiftedSum;
|
||||||
|
result[1] = (uint64_t) (shiftedSum >> 64);
|
||||||
|
}
|
||||||
|
|
||||||
|
// Computes 5^-i in the form required by Ryu, and stores it in the given pointer.
|
||||||
|
static inline void double_computeInvPow5(const uint32_t i, uint64_t* const result) {
|
||||||
|
const uint32_t base = (i + POW5_TABLE_SIZE - 1) / POW5_TABLE_SIZE;
|
||||||
|
const uint32_t base2 = base * POW5_TABLE_SIZE;
|
||||||
|
const uint32_t offset = base2 - i;
|
||||||
|
const uint64_t* const mul = DOUBLE_POW5_INV_SPLIT2[base]; // 1/5^base2
|
||||||
|
if (offset == 0) {
|
||||||
|
result[0] = mul[0];
|
||||||
|
result[1] = mul[1];
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
const uint64_t m = DOUBLE_POW5_TABLE[offset]; // 5^offset
|
||||||
|
const uint128_t b0 = ((uint128_t) m) * (mul[0] - 1);
|
||||||
|
const uint128_t b2 = ((uint128_t) m) * mul[1]; // 1/5^base2 * 5^offset = 1/5^(base2-offset) = 1/5^i
|
||||||
|
const uint32_t delta = pow5bits(base2) - pow5bits(i);
|
||||||
|
assert(i / 16 < sizeof(POW5_INV_OFFSETS) / sizeof(POW5_INV_OFFSETS[0]));
|
||||||
|
const uint128_t shiftedSum =
|
||||||
|
((b0 >> delta) + (b2 << (64 - delta))) + 1 + ((POW5_INV_OFFSETS[i / 16] >> ((i % 16) << 1)) & 3);
|
||||||
|
result[0] = (uint64_t) shiftedSum;
|
||||||
|
result[1] = (uint64_t) (shiftedSum >> 64);
|
||||||
|
}
|
||||||
|
|
||||||
|
#else // defined(HAS_UINT128)
|
||||||
|
|
||||||
|
// Computes 5^i in the form required by Ryu, and stores it in the given pointer.
|
||||||
|
static inline void double_computePow5(const uint32_t i, uint64_t* const result) {
|
||||||
|
const uint32_t base = i / POW5_TABLE_SIZE;
|
||||||
|
const uint32_t base2 = base * POW5_TABLE_SIZE;
|
||||||
|
const uint32_t offset = i - base2;
|
||||||
|
const uint64_t* const mul = DOUBLE_POW5_SPLIT2[base];
|
||||||
|
if (offset == 0) {
|
||||||
|
result[0] = mul[0];
|
||||||
|
result[1] = mul[1];
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
const uint64_t m = DOUBLE_POW5_TABLE[offset];
|
||||||
|
uint64_t high1;
|
||||||
|
const uint64_t low1 = umul128(m, mul[1], &high1);
|
||||||
|
uint64_t high0;
|
||||||
|
const uint64_t low0 = umul128(m, mul[0], &high0);
|
||||||
|
const uint64_t sum = high0 + low1;
|
||||||
|
if (sum < high0) {
|
||||||
|
++high1; // overflow into high1
|
||||||
|
}
|
||||||
|
// high1 | sum | low0
|
||||||
|
const uint32_t delta = pow5bits(i) - pow5bits(base2);
|
||||||
|
result[0] = shiftright128(low0, sum, delta) + ((POW5_OFFSETS[i / 16] >> ((i % 16) << 1)) & 3);
|
||||||
|
result[1] = shiftright128(sum, high1, delta);
|
||||||
|
}
|
||||||
|
|
||||||
|
// Computes 5^-i in the form required by Ryu, and stores it in the given pointer.
|
||||||
|
static inline void double_computeInvPow5(const uint32_t i, uint64_t* const result) {
|
||||||
|
const uint32_t base = (i + POW5_TABLE_SIZE - 1) / POW5_TABLE_SIZE;
|
||||||
|
const uint32_t base2 = base * POW5_TABLE_SIZE;
|
||||||
|
const uint32_t offset = base2 - i;
|
||||||
|
const uint64_t* const mul = DOUBLE_POW5_INV_SPLIT2[base]; // 1/5^base2
|
||||||
|
if (offset == 0) {
|
||||||
|
result[0] = mul[0];
|
||||||
|
result[1] = mul[1];
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
const uint64_t m = DOUBLE_POW5_TABLE[offset];
|
||||||
|
uint64_t high1;
|
||||||
|
const uint64_t low1 = umul128(m, mul[1], &high1);
|
||||||
|
uint64_t high0;
|
||||||
|
const uint64_t low0 = umul128(m, mul[0] - 1, &high0);
|
||||||
|
const uint64_t sum = high0 + low1;
|
||||||
|
if (sum < high0) {
|
||||||
|
++high1; // overflow into high1
|
||||||
|
}
|
||||||
|
// high1 | sum | low0
|
||||||
|
const uint32_t delta = pow5bits(base2) - pow5bits(i);
|
||||||
|
assert(i / 16 < sizeof(POW5_INV_OFFSETS) / sizeof(POW5_INV_OFFSETS[0]));
|
||||||
|
result[0] = shiftright128(low0, sum, delta) + 1 + ((POW5_INV_OFFSETS[i / 16] >> ((i % 16) << 1)) & 3);
|
||||||
|
result[1] = shiftright128(sum, high1, delta);
|
||||||
|
}
|
||||||
|
|
||||||
|
#endif // defined(HAS_UINT128)
|
||||||
|
|
||||||
|
#endif // RYU_D2S_SMALL_TABLE_H
|
||||||
@@ -0,0 +1,35 @@
|
|||||||
|
// Copyright 2018 Ulf Adams
|
||||||
|
//
|
||||||
|
// The contents of this file may be used under the terms of the Apache License,
|
||||||
|
// Version 2.0.
|
||||||
|
//
|
||||||
|
// (See accompanying file LICENSE-Apache or copy at
|
||||||
|
// http://www.apache.org/licenses/LICENSE-2.0)
|
||||||
|
//
|
||||||
|
// Alternatively, the contents of this file may be used under the terms of
|
||||||
|
// the Boost Software License, Version 1.0.
|
||||||
|
// (See accompanying file LICENSE-Boost or copy at
|
||||||
|
// https://www.boost.org/LICENSE_1_0.txt)
|
||||||
|
//
|
||||||
|
// Unless required by applicable law or agreed to in writing, this software
|
||||||
|
// is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
|
||||||
|
// KIND, either express or implied.
|
||||||
|
#ifndef RYU_DIGIT_TABLE_H
|
||||||
|
#define RYU_DIGIT_TABLE_H
|
||||||
|
|
||||||
|
// A table of all two-digit numbers. This is used to speed up decimal digit
|
||||||
|
// generation by copying pairs of digits into the final output.
|
||||||
|
static const char DIGIT_TABLE[200] = {
|
||||||
|
'0','0','0','1','0','2','0','3','0','4','0','5','0','6','0','7','0','8','0','9',
|
||||||
|
'1','0','1','1','1','2','1','3','1','4','1','5','1','6','1','7','1','8','1','9',
|
||||||
|
'2','0','2','1','2','2','2','3','2','4','2','5','2','6','2','7','2','8','2','9',
|
||||||
|
'3','0','3','1','3','2','3','3','3','4','3','5','3','6','3','7','3','8','3','9',
|
||||||
|
'4','0','4','1','4','2','4','3','4','4','4','5','4','6','4','7','4','8','4','9',
|
||||||
|
'5','0','5','1','5','2','5','3','5','4','5','5','5','6','5','7','5','8','5','9',
|
||||||
|
'6','0','6','1','6','2','6','3','6','4','6','5','6','6','6','7','6','8','6','9',
|
||||||
|
'7','0','7','1','7','2','7','3','7','4','7','5','7','6','7','7','7','8','7','9',
|
||||||
|
'8','0','8','1','8','2','8','3','8','4','8','5','8','6','8','7','8','8','8','9',
|
||||||
|
'9','0','9','1','9','2','9','3','9','4','9','5','9','6','9','7','9','8','9','9'
|
||||||
|
};
|
||||||
|
|
||||||
|
#endif // RYU_DIGIT_TABLE_H
|
||||||
@@ -0,0 +1,46 @@
|
|||||||
|
// Copyright 2018 Ulf Adams
|
||||||
|
//
|
||||||
|
// The contents of this file may be used under the terms of the Apache License,
|
||||||
|
// Version 2.0.
|
||||||
|
//
|
||||||
|
// (See accompanying file LICENSE-Apache or copy at
|
||||||
|
// http://www.apache.org/licenses/LICENSE-2.0)
|
||||||
|
//
|
||||||
|
// Alternatively, the contents of this file may be used under the terms of
|
||||||
|
// the Boost Software License, Version 1.0.
|
||||||
|
// (See accompanying file LICENSE-Boost or copy at
|
||||||
|
// https://www.boost.org/LICENSE_1_0.txt)
|
||||||
|
//
|
||||||
|
// Unless required by applicable law or agreed to in writing, this software
|
||||||
|
// is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
|
||||||
|
// KIND, either express or implied.
|
||||||
|
#ifndef RYU_H
|
||||||
|
#define RYU_H
|
||||||
|
|
||||||
|
#ifdef __cplusplus
|
||||||
|
extern "C" {
|
||||||
|
#endif
|
||||||
|
|
||||||
|
#include <inttypes.h>
|
||||||
|
|
||||||
|
int d2s_buffered_n(double f, char* result);
|
||||||
|
void d2s_buffered(double f, char* result);
|
||||||
|
char* d2s(double f);
|
||||||
|
|
||||||
|
int f2s_buffered_n(float f, char* result);
|
||||||
|
void f2s_buffered(float f, char* result);
|
||||||
|
char* f2s(float f);
|
||||||
|
|
||||||
|
int d2fixed_buffered_n(double d, uint32_t precision, char* result);
|
||||||
|
void d2fixed_buffered(double d, uint32_t precision, char* result);
|
||||||
|
char* d2fixed(double d, uint32_t precision);
|
||||||
|
|
||||||
|
int d2exp_buffered_n(double d, uint32_t precision, char* result);
|
||||||
|
void d2exp_buffered(double d, uint32_t precision, char* result);
|
||||||
|
char* d2exp(double d, uint32_t precision);
|
||||||
|
|
||||||
|
#ifdef __cplusplus
|
||||||
|
}
|
||||||
|
#endif
|
||||||
|
|
||||||
|
#endif // RYU_H
|
||||||
@@ -0,0 +1,83 @@
|
|||||||
|
#include "json_numbers.h"
|
||||||
|
#include "ryu/ryu.h"
|
||||||
|
#include <float.h>
|
||||||
|
#include <stdint.h>
|
||||||
|
#include <string.h>
|
||||||
|
|
||||||
|
_Static_assert(sizeof(double)==8 && DBL_MANT_DIG==53 && DBL_MAX_EXP==1024,
|
||||||
|
"Native JSON numbers require IEEE-754 binary64");
|
||||||
|
|
||||||
|
int native_json_format_double(char output[NATIVE_JSON_DOUBLE_CAPACITY], double value) {
|
||||||
|
uint64_t bits;
|
||||||
|
memcpy(&bits,&value,sizeof(bits));
|
||||||
|
uint64_t magnitude=bits & UINT64_C(0x7fffffffffffffff);
|
||||||
|
if (magnitude>=UINT64_C(0x7ff0000000000000)) return 0;
|
||||||
|
if (magnitude==0) {
|
||||||
|
if (bits>>63) { memcpy(output,"-0.0",4); return 4; }
|
||||||
|
output[0]='0'; return 1;
|
||||||
|
}
|
||||||
|
int length=d2s_buffered_n(value,output);
|
||||||
|
/* Ryu returns shortest significant digits in scientific notation. Use
|
||||||
|
ordinary notation only when its JSON token is shorter. This moves an
|
||||||
|
exact decimal point; it never recomputes or rounds the floating value. */
|
||||||
|
char digits[17];
|
||||||
|
int negative=output[0]=='-', count=0, pos=negative;
|
||||||
|
while (output[pos]!='E') {
|
||||||
|
if (output[pos]!='.') digits[count++]=output[pos];
|
||||||
|
pos++;
|
||||||
|
}
|
||||||
|
pos++;
|
||||||
|
int exponent_negative=output[pos]=='-';
|
||||||
|
if (exponent_negative) pos++;
|
||||||
|
int exponent=0;
|
||||||
|
while (pos<length) exponent=exponent*10+(output[pos++]-'0');
|
||||||
|
if (exponent_negative) exponent=-exponent;
|
||||||
|
int point=exponent+1;
|
||||||
|
int fixed_length=negative+(point<=0 ? 2-point+count : point>=count ? point : count+1);
|
||||||
|
if (fixed_length>=length) return length;
|
||||||
|
pos=0;
|
||||||
|
if (negative) output[pos++]='-';
|
||||||
|
if (point<=0) {
|
||||||
|
output[pos++]='0'; output[pos++]='.';
|
||||||
|
for (int i=0;i<-point;i++) output[pos++]='0';
|
||||||
|
memcpy(output+pos,digits,(size_t)count); pos+=count;
|
||||||
|
} else {
|
||||||
|
for (int i=0;i<count;i++) {
|
||||||
|
if (i==point) output[pos++]='.';
|
||||||
|
output[pos++]=digits[i];
|
||||||
|
}
|
||||||
|
for (int i=count;i<point;i++) output[pos++]='0';
|
||||||
|
}
|
||||||
|
return pos;
|
||||||
|
}
|
||||||
|
|
||||||
|
static int write_bytes(FILE *file, const char *data, size_t length) {
|
||||||
|
return fwrite(data,1,length,file)==length && !ferror(file);
|
||||||
|
}
|
||||||
|
|
||||||
|
int native_json_write_number(FILE *file, double value) {
|
||||||
|
if (!file || ferror(file)) return 0;
|
||||||
|
char text[NATIVE_JSON_DOUBLE_CAPACITY];
|
||||||
|
int length=native_json_format_double(text,value);
|
||||||
|
return length>0 && write_bytes(file,text,(size_t)length);
|
||||||
|
}
|
||||||
|
|
||||||
|
int native_json_write_array(FILE *file, const double *values, size_t count, size_t stride) {
|
||||||
|
if (!file || ferror(file) || (count && !values)) return 0;
|
||||||
|
if (count>1 && (!stride || stride>(size_t)PTRDIFF_MAX/sizeof(double)/(count-1))) return 0;
|
||||||
|
char buffer[65536];
|
||||||
|
size_t used=1;
|
||||||
|
buffer[0]='[';
|
||||||
|
for (size_t i=0;i<count;i++) {
|
||||||
|
if (used>sizeof(buffer)-NATIVE_JSON_DOUBLE_CAPACITY-2) {
|
||||||
|
if (!write_bytes(file,buffer,used)) return 0;
|
||||||
|
used=0;
|
||||||
|
}
|
||||||
|
if (i) buffer[used++]=',';
|
||||||
|
int length=native_json_format_double(buffer+used,values[i*stride]);
|
||||||
|
if (!length) return 0;
|
||||||
|
used+=(size_t)length;
|
||||||
|
}
|
||||||
|
buffer[used++]=']';
|
||||||
|
return write_bytes(file,buffer,used);
|
||||||
|
}
|
||||||
@@ -1,4 +1,5 @@
|
|||||||
#include "runtime.h"
|
#include "runtime.h"
|
||||||
|
#include "json_numbers.h"
|
||||||
#include <sundials/sundials_config.h>
|
#include <sundials/sundials_config.h>
|
||||||
#include <stdio.h>
|
#include <stdio.h>
|
||||||
#include <stdlib.h>
|
#include <stdlib.h>
|
||||||
@@ -30,7 +31,19 @@ static int probe(void) {
|
|||||||
}
|
}
|
||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
static int write_result(NativeRun *r, const char *path) {
|
static int write_result_index(const char *path, long series_start, long series_end,
|
||||||
|
long result_bytes, size_t sample_count) {
|
||||||
|
if(series_start<0 || series_end<series_start || result_bytes<series_end) return 0;
|
||||||
|
FILE *index=fopen(path,"wb"); if(!index) return 0;
|
||||||
|
int ok=fprintf(index,"{\"version\":1,\"seriesStart\":%ld,\"seriesEnd\":%ld,"
|
||||||
|
"\"resultBytes\":%ld,\"sampleCount\":%zu}\n",
|
||||||
|
series_start,series_end,result_bytes,sample_count)>=0;
|
||||||
|
if(ferror(index)) ok=0;
|
||||||
|
if(fclose(index)) ok=0;
|
||||||
|
return ok;
|
||||||
|
}
|
||||||
|
static int write_result(NativeRun *r, const char *path, const char *index_path) {
|
||||||
|
if(index_path && !strcmp(path,index_path)) return 0;
|
||||||
double dy[NSTATES], final[NOUTPUTS];
|
double dy[NSTATES], final[NOUTPUTS];
|
||||||
int final_ok=model_eval(r->final_time,r->final_state,dy,final);
|
int final_ok=model_eval(r->final_time,r->final_state,dy,final);
|
||||||
size_t length=r->count*NOUTPUTS;
|
size_t length=r->count*NOUTPUTS;
|
||||||
@@ -53,30 +66,48 @@ static int write_result(NativeRun *r, const char *path) {
|
|||||||
r->options.bdf?"BDF":"RK45",r->options.bdf?"CVODE":"Dormand-Prince 5(4)",SUNDIALS_VERSION,
|
r->options.bdf?"BDF":"RK45",r->options.bdf?"CVODE":"Dormand-Prince 5(4)",SUNDIALS_VERSION,
|
||||||
r->final_time,r->solve_seconds,r->solve_cpu_seconds,r->nfev,r->accepted,r->rejected,
|
r->final_time,r->solve_seconds,r->solve_cpu_seconds,r->nfev,r->accepted,r->rejected,
|
||||||
r->events,r->starts,r->njev,r->nlu,r->max_accepted_step);
|
r->events,r->starts,r->njev,r->nlu,r->max_accepted_step);
|
||||||
|
/* Binary-mode positions delimit the complete series object, including
|
||||||
|
both braces. The optional index avoids scanning or parsing its values. */
|
||||||
|
long series_start=-1,series_end=-1,result_bytes=-1;
|
||||||
|
if(index_path) {
|
||||||
|
series_start=ftell(f);
|
||||||
|
if(series_start>0) series_start--; else series_start=-1;
|
||||||
|
}
|
||||||
|
int output_ok=1;
|
||||||
if (r->count) {
|
if (r->count) {
|
||||||
fprintf(f,"\"time\":[");
|
fprintf(f,"\"time\":");
|
||||||
for (size_t i=0;i<r->count;i++) fprintf(f,"%s%.17g",i?",":"",r->times[i]);
|
output_ok=native_json_write_array(f,r->times,r->count,1);
|
||||||
fputc(']',f);
|
for (int j=0;j<NOUTPUTS && output_ok;j++) {
|
||||||
for (int j=0;j<NOUTPUTS;j++) {
|
fputc(',',f); json_string(f,model_output_keys[j]); fputc(':',f);
|
||||||
fputc(',',f); json_string(f,model_output_keys[j]); fprintf(f,":[");
|
output_ok=native_json_write_array(f,values+j,r->count,NOUTPUTS);
|
||||||
for (size_t i=0;i<r->count;i++) fprintf(f,"%s%.17g",i?",":"",values[i*NOUTPUTS+j]);
|
|
||||||
fputc(']',f);
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
fprintf(f,"},\"final\":{");
|
fputc('}',f);
|
||||||
if (final_ok) for (int j=0;j<NOUTPUTS;j++) {
|
if(index_path) series_end=ftell(f);
|
||||||
|
fprintf(f,",\"final\":{");
|
||||||
|
if (final_ok) for (int j=0;j<NOUTPUTS && output_ok;j++) {
|
||||||
if (j) fputc(',',f);
|
if (j) fputc(',',f);
|
||||||
json_string(f,model_output_keys[j]); fprintf(f,":%.17g",final[j]);
|
json_string(f,model_output_keys[j]); fputc(':',f);
|
||||||
|
output_ok=native_json_write_number(f,final[j]);
|
||||||
}
|
}
|
||||||
fprintf(f,"},\"finalState\":"); vector(f,r->final_state,NSTATES);
|
fprintf(f,"},\"finalState\":");
|
||||||
fprintf(f,"}\n"); int ok=!ferror(f); if (fclose(f)) ok=0;
|
if (output_ok) output_ok=native_json_write_array(f,r->final_state,NSTATES,1);
|
||||||
|
fprintf(f,"}\n"); int ok=output_ok && !ferror(f);
|
||||||
|
if(index_path) {
|
||||||
|
result_bytes=ftell(f);
|
||||||
|
if(series_start<0 || series_end<0 || result_bytes<0) ok=0;
|
||||||
|
}
|
||||||
|
if (fclose(f)) ok=0;
|
||||||
|
/* Publish the index only after the complete result was successfully
|
||||||
|
flushed and closed. An index I/O failure is a failed result write. */
|
||||||
|
if(ok && index_path) ok=write_result_index(index_path,series_start,series_end,result_bytes,r->count);
|
||||||
free(values); return ok;
|
free(values); return ok;
|
||||||
}
|
}
|
||||||
|
|
||||||
int main(int argc, char **argv) {
|
int main(int argc, char **argv) {
|
||||||
NativeRun r={0};
|
NativeRun r={0};
|
||||||
r.options=(NativeOptions){0,10,.02,.001,1e-6,300,0,1,NULL};
|
r.options=(NativeOptions){0,10,.02,.001,1e-6,300,0,1,NULL};
|
||||||
const char *output="result.json";
|
const char *output="result.json", *index_path=NULL;
|
||||||
for (int i=1;i<argc;i++) {
|
for (int i=1;i<argc;i++) {
|
||||||
const char *arg=argv[i];
|
const char *arg=argv[i];
|
||||||
if (!strcmp(arg,"--probe")) return probe();
|
if (!strcmp(arg,"--probe")) return probe();
|
||||||
@@ -88,6 +119,7 @@ int main(int argc, char **argv) {
|
|||||||
if (i+1==argc) return 64;
|
if (i+1==argc) return 64;
|
||||||
const char *value=argv[++i];
|
const char *value=argv[++i];
|
||||||
if (!strcmp(arg,"--output")) output=value;
|
if (!strcmp(arg,"--output")) output=value;
|
||||||
|
else if (!strcmp(arg,"--result-index")) index_path=value;
|
||||||
else if (!strcmp(arg,"--cancel-file")) r.options.cancel_path=value;
|
else if (!strcmp(arg,"--cancel-file")) r.options.cancel_path=value;
|
||||||
else if (!strcmp(arg,"--method")) {
|
else if (!strcmp(arg,"--method")) {
|
||||||
if (strcmp(value,"RK45") && strcmp(value,"BDF")) return 64;
|
if (strcmp(value,"RK45") && strcmp(value,"BDF")) return 64;
|
||||||
@@ -113,7 +145,7 @@ int main(int argc, char **argv) {
|
|||||||
((r.options.stop-r.options.start)/r.options.sample_step>1000000 ||
|
((r.options.stop-r.options.start)/r.options.sample_step>1000000 ||
|
||||||
((r.options.stop-r.options.start)/r.options.sample_step+1024)*(NSTATES+NOUTPUTS)*sizeof(double)>268435456)) return 64;
|
((r.options.stop-r.options.start)/r.options.sample_step+1024)*(NSTATES+NOUTPUTS)*sizeof(double)>268435456)) return 64;
|
||||||
native_solve(&r);
|
native_solve(&r);
|
||||||
int saved=write_result(&r,output);
|
int saved=write_result(&r,output,index_path);
|
||||||
int code=saved?(r.status==2?2:0):3;
|
int code=saved?(r.status==2?2:0):3;
|
||||||
native_run_free(&r); return code;
|
native_run_free(&r); return code;
|
||||||
}
|
}
|
||||||
@@ -9,6 +9,8 @@
|
|||||||
|
|
||||||
网页当前只支持导入工程 JSON;XML 可从网页下载,用作后端输入。网页使用原生 BDF 和当前默认 `rtol=1e-8`,与 AME 内部积分器及历史数值设置不能视为相同。这里的参数对齐指模型物理参数与初值,不声称不同仿真器所有数值输出完全一致。
|
网页当前只支持导入工程 JSON;XML 可从网页下载,用作后端输入。网页使用原生 BDF 和当前默认 `rtol=1e-8`,与 AME 内部积分器及历史数值设置不能视为相同。这里的参数对齐指模型物理参数与初值,不声称不同仿真器所有数值输出完全一致。
|
||||||
|
|
||||||
|
后续优化测试优先使用八路 corrected 工程;仅当八路跑不通且短期无法解决时,再使用四路。具体记录要求见 [优化验证约定](../../docs/standard/optimization-benchmark-model.md)。
|
||||||
|
|
||||||
2026-09-11 整理时,两份工程内容保持不变。其他 JSON/XML 已按用途移出本目录,移动前后字节数及 SHA256 相同:
|
2026-09-11 整理时,两份工程内容保持不变。其他 JSON/XML 已按用途移出本目录,移动前后字节数及 SHA256 相同:
|
||||||
|
|
||||||
| 原文件 | 当前路径 | 用途 |
|
| 原文件 | 当前路径 | 用途 |
|
||||||
|
|||||||
@@ -0,0 +1,325 @@
|
|||||||
|
<?xml version="1.0" encoding="UTF-8"?>
|
||||||
|
<System name="skill-test" schemaVersion="3" unitSystem="SI">
|
||||||
|
<Simulation tStart="0" tStop="10" sampleStep="0.02" maxStep="0.001" method="RK45"/>
|
||||||
|
<Components>
|
||||||
|
<Component id="amesim_pnch023_1" type="amesim_pnch023" modelVersion="0.1.0">
|
||||||
|
<Parameter name="gi" value="1"/>
|
||||||
|
<Parameter name="cvol" value="0.057"/>
|
||||||
|
<Parameter name="kth" value="0"/>
|
||||||
|
<Parameter name="sth" value="0.1"/>
|
||||||
|
<Parameter name="extemp" value="293.15"/>
|
||||||
|
<Parameter name="p0" value="15300000"/>
|
||||||
|
<Parameter name="T0" value="293.15"/>
|
||||||
|
</Component>
|
||||||
|
<Component id="amesim_pnvo001_1" type="amesim_pnvo001" modelVersion="0.2.0">
|
||||||
|
<Parameter name="gi" value="1"/>
|
||||||
|
<Parameter name="cq" value="0.45"/>
|
||||||
|
<Parameter name="area0" value="0.0000785"/>
|
||||||
|
<Parameter name="Cv" value="0.5"/>
|
||||||
|
<Parameter name="Kv" value="0.4"/>
|
||||||
|
<Parameter name="flowset" value="1"/>
|
||||||
|
<Parameter name="opening0" value="0"/>
|
||||||
|
</Component>
|
||||||
|
<Component id="amesim_step0_1" type="amesim_step0" modelVersion="0.1.0">
|
||||||
|
<Parameter name="initial" value="0"/>
|
||||||
|
<Parameter name="final" value="1"/>
|
||||||
|
<Parameter name="time" value="0.04"/>
|
||||||
|
</Component>
|
||||||
|
<Component id="amesim_pnch012_1" type="amesim_pnch012" modelVersion="0.1.0">
|
||||||
|
<Parameter name="gi" value="1"/>
|
||||||
|
<Parameter name="cvol0" value="0.015"/>
|
||||||
|
<Parameter name="kth" value="1500"/>
|
||||||
|
<Parameter name="sth" value="0.7"/>
|
||||||
|
<Parameter name="extemp" value="293.15"/>
|
||||||
|
<Parameter name="p0" value="100000"/>
|
||||||
|
<Parameter name="T0" value="293.15"/>
|
||||||
|
<Parameter name="vol1" value="0"/>
|
||||||
|
<Parameter name="vol2" value="0"/>
|
||||||
|
<Parameter name="vol3" value="0"/>
|
||||||
|
<Parameter name="vol4" value="0"/>
|
||||||
|
<Parameter name="dvol1" value="0"/>
|
||||||
|
<Parameter name="dvol2" value="0"/>
|
||||||
|
<Parameter name="dvol3" value="0"/>
|
||||||
|
<Parameter name="dvol4" value="0"/>
|
||||||
|
</Component>
|
||||||
|
<Component id="amesim_pnpl01_2" type="amesim_pnpl01" modelVersion="0.1.0"/>
|
||||||
|
<Component id="amesim_pnpl01_3" type="amesim_pnpl01" modelVersion="0.1.0"/>
|
||||||
|
<Component id="amesim_pnpl01_4" type="amesim_pnpl01" modelVersion="0.1.0"/>
|
||||||
|
<Component id="amesim_ud00_1" type="amesim_ud00" modelVersion="0.2.0">
|
||||||
|
<Parameter name="tstart" value="0"/>
|
||||||
|
<Parameter name="start1" value="100000000000000000"/>
|
||||||
|
<Parameter name="end1" value="100000000000000000"/>
|
||||||
|
<Parameter name="t1" value="0.8"/>
|
||||||
|
<Parameter name="start2" value="49000"/>
|
||||||
|
<Parameter name="end2" value="49000"/>
|
||||||
|
<Parameter name="t2" value="10"/>
|
||||||
|
<Parameter name="start3" value="1"/>
|
||||||
|
<Parameter name="end3" value="1"/>
|
||||||
|
<Parameter name="t3" value="0"/>
|
||||||
|
<Parameter name="start4" value="1"/>
|
||||||
|
<Parameter name="end4" value="1"/>
|
||||||
|
<Parameter name="t4" value="0"/>
|
||||||
|
<Parameter name="start5" value="1"/>
|
||||||
|
<Parameter name="end5" value="1"/>
|
||||||
|
<Parameter name="t5" value="0"/>
|
||||||
|
<Parameter name="start6" value="1"/>
|
||||||
|
<Parameter name="end6" value="1"/>
|
||||||
|
<Parameter name="t6" value="0"/>
|
||||||
|
<Parameter name="start7" value="1"/>
|
||||||
|
<Parameter name="end7" value="1"/>
|
||||||
|
<Parameter name="t7" value="0"/>
|
||||||
|
<Parameter name="start8" value="1"/>
|
||||||
|
<Parameter name="end8" value="1"/>
|
||||||
|
<Parameter name="t8" value="0"/>
|
||||||
|
<Parameter name="nstages" value="2"/>
|
||||||
|
<Parameter name="iscyclic" value="0"/>
|
||||||
|
</Component>
|
||||||
|
<Component id="amesim_forc_1" type="amesim_forc" modelVersion="0.2.0">
|
||||||
|
<Parameter name="direction" value="1"/>
|
||||||
|
</Component>
|
||||||
|
<Component id="amesim_pnrp17_1" type="amesim_pnrp17" modelVersion="0.1.0">
|
||||||
|
<Parameter name="gi" value="1"/>
|
||||||
|
<Parameter name="dp" value="0.2"/>
|
||||||
|
<Parameter name="dr" value="0.001"/>
|
||||||
|
<Parameter name="x0" value="0"/>
|
||||||
|
</Component>
|
||||||
|
<Component id="amesim_helium_medium_1" type="amesim_helium_medium" modelVersion="0.1.0">
|
||||||
|
<Parameter name="gi" value="1"/>
|
||||||
|
<Parameter name="property_model" value="0"/>
|
||||||
|
</Component>
|
||||||
|
<Component id="amesim_pnch012_2" type="amesim_pnch012" modelVersion="0.1.0">
|
||||||
|
<Parameter name="gi" value="1"/>
|
||||||
|
<Parameter name="cvol0" value="0.015"/>
|
||||||
|
<Parameter name="kth" value="1500"/>
|
||||||
|
<Parameter name="sth" value="0.7"/>
|
||||||
|
<Parameter name="extemp" value="293.15"/>
|
||||||
|
<Parameter name="p0" value="100000"/>
|
||||||
|
<Parameter name="T0" value="293.15"/>
|
||||||
|
<Parameter name="vol1" value="0"/>
|
||||||
|
<Parameter name="vol2" value="0"/>
|
||||||
|
<Parameter name="vol3" value="0"/>
|
||||||
|
<Parameter name="vol4" value="0"/>
|
||||||
|
<Parameter name="dvol1" value="0"/>
|
||||||
|
<Parameter name="dvol2" value="0"/>
|
||||||
|
<Parameter name="dvol3" value="0"/>
|
||||||
|
<Parameter name="dvol4" value="0"/>
|
||||||
|
</Component>
|
||||||
|
<Component id="amesim_pnpl01_6" type="amesim_pnpl01" modelVersion="0.1.0"/>
|
||||||
|
<Component id="amesim_pnpl01_7" type="amesim_pnpl01" modelVersion="0.1.0"/>
|
||||||
|
<Component id="amesim_pnpl01_8" type="amesim_pnpl01" modelVersion="0.1.0"/>
|
||||||
|
<Component id="amesim_pnpl01_9" type="amesim_pnpl01" modelVersion="0.1.0"/>
|
||||||
|
<Component id="amesim_mecmas21_2" type="amesim_mecmas21" modelVersion="0.2.0">
|
||||||
|
<Parameter name="mass" value="50"/>
|
||||||
|
<Parameter name="fstick" value="0"/>
|
||||||
|
<Parameter name="fcoul" value="0"/>
|
||||||
|
<Parameter name="rvisc" value="0"/>
|
||||||
|
<Parameter name="wind" value="0"/>
|
||||||
|
<Parameter name="dvel" value="0.000001"/>
|
||||||
|
<Parameter name="restdvel" value="0.000001"/>
|
||||||
|
<Parameter name="restcoeff" value="0.65"/>
|
||||||
|
<Parameter name="astrib" value="0.001"/>
|
||||||
|
<Parameter name="xmin" value="-1"/>
|
||||||
|
<Parameter name="Kbmin" value="1000000000"/>
|
||||||
|
<Parameter name="Dbmin" value="10000"/>
|
||||||
|
<Parameter name="Pdmin" value="0.0001"/>
|
||||||
|
<Parameter name="xmax" value="0.8"/>
|
||||||
|
<Parameter name="Kbmax" value="1000000000"/>
|
||||||
|
<Parameter name="Dbmax" value="10000"/>
|
||||||
|
<Parameter name="Pdmax" value="0.0001"/>
|
||||||
|
<Parameter name="theta" value="0"/>
|
||||||
|
<Parameter name="useFriction" value="1"/>
|
||||||
|
<Parameter name="stoptype" value="4"/>
|
||||||
|
<Parameter name="discContactOption" value="1"/>
|
||||||
|
<Parameter name="strib" value="1"/>
|
||||||
|
<Parameter name="frictionType" value="1"/>
|
||||||
|
<Parameter name="v0" value="0"/>
|
||||||
|
<Parameter name="x0" value="0"/>
|
||||||
|
</Component>
|
||||||
|
<Component id="amesim_f000_1" type="amesim_f000" modelVersion="0.1.0"/>
|
||||||
|
<Component id="amesim_f000_2" type="amesim_f000" modelVersion="0.1.0"/>
|
||||||
|
<Component id="amesim_lstp00a_1" type="amesim_lstp00a" modelVersion="0.2.0">
|
||||||
|
<Parameter name="na" value="10"/>
|
||||||
|
<Parameter name="gap0" value="0"/>
|
||||||
|
<Parameter name="kcont" value="100000000000"/>
|
||||||
|
<Parameter name="G" value="85700000000"/>
|
||||||
|
<Parameter name="sdiam" value="0.02"/>
|
||||||
|
<Parameter name="wdiam" value="0.002"/>
|
||||||
|
<Parameter name="rcont" value="100000000000"/>
|
||||||
|
<Parameter name="Pdis" value="1e-7"/>
|
||||||
|
<Parameter name="stiffmode" value="1"/>
|
||||||
|
<Parameter name="discContactOption" value="1"/>
|
||||||
|
</Component>
|
||||||
|
<Component id="amesim_forc_2" type="amesim_forc" modelVersion="0.2.0">
|
||||||
|
<Parameter name="direction" value="1"/>
|
||||||
|
</Component>
|
||||||
|
<Component id="amesim_ud00_2" type="amesim_ud00" modelVersion="0.2.0">
|
||||||
|
<Parameter name="tstart" value="0"/>
|
||||||
|
<Parameter name="start1" value="1000000000000"/>
|
||||||
|
<Parameter name="end1" value="1000000000000"/>
|
||||||
|
<Parameter name="t1" value="0.8"/>
|
||||||
|
<Parameter name="start2" value="0"/>
|
||||||
|
<Parameter name="end2" value="0"/>
|
||||||
|
<Parameter name="t2" value="10"/>
|
||||||
|
<Parameter name="start3" value="1"/>
|
||||||
|
<Parameter name="end3" value="1"/>
|
||||||
|
<Parameter name="t3" value="0"/>
|
||||||
|
<Parameter name="start4" value="1"/>
|
||||||
|
<Parameter name="end4" value="1"/>
|
||||||
|
<Parameter name="t4" value="0"/>
|
||||||
|
<Parameter name="start5" value="1"/>
|
||||||
|
<Parameter name="end5" value="1"/>
|
||||||
|
<Parameter name="t5" value="0"/>
|
||||||
|
<Parameter name="start6" value="1"/>
|
||||||
|
<Parameter name="end6" value="1"/>
|
||||||
|
<Parameter name="t6" value="0"/>
|
||||||
|
<Parameter name="start7" value="1"/>
|
||||||
|
<Parameter name="end7" value="1"/>
|
||||||
|
<Parameter name="t7" value="0"/>
|
||||||
|
<Parameter name="start8" value="1"/>
|
||||||
|
<Parameter name="end8" value="1"/>
|
||||||
|
<Parameter name="t8" value="0"/>
|
||||||
|
<Parameter name="nstages" value="2"/>
|
||||||
|
<Parameter name="iscyclic" value="0"/>
|
||||||
|
</Component>
|
||||||
|
<Component id="amesim_mecmas21_5" type="amesim_mecmas21" modelVersion="0.2.0">
|
||||||
|
<Parameter name="mass" value="170000"/>
|
||||||
|
<Parameter name="fstick" value="0"/>
|
||||||
|
<Parameter name="fcoul" value="0"/>
|
||||||
|
<Parameter name="rvisc" value="0"/>
|
||||||
|
<Parameter name="wind" value="0"/>
|
||||||
|
<Parameter name="dvel" value="0.000001"/>
|
||||||
|
<Parameter name="restdvel" value="0.000001"/>
|
||||||
|
<Parameter name="restcoeff" value="0.65"/>
|
||||||
|
<Parameter name="astrib" value="0.001"/>
|
||||||
|
<Parameter name="xmin" value="0"/>
|
||||||
|
<Parameter name="Kbmin" value="1000000000"/>
|
||||||
|
<Parameter name="Dbmin" value="10000"/>
|
||||||
|
<Parameter name="Pdmin" value="0.0001"/>
|
||||||
|
<Parameter name="xmax" value="0.37"/>
|
||||||
|
<Parameter name="Kbmax" value="1000000000"/>
|
||||||
|
<Parameter name="Dbmax" value="10000"/>
|
||||||
|
<Parameter name="Pdmax" value="0.0001"/>
|
||||||
|
<Parameter name="theta" value="0"/>
|
||||||
|
<Parameter name="useFriction" value="1"/>
|
||||||
|
<Parameter name="stoptype" value="1"/>
|
||||||
|
<Parameter name="discContactOption" value="1"/>
|
||||||
|
<Parameter name="strib" value="1"/>
|
||||||
|
<Parameter name="frictionType" value="1"/>
|
||||||
|
<Parameter name="v0" value="0"/>
|
||||||
|
<Parameter name="x0" value="0"/>
|
||||||
|
</Component>
|
||||||
|
<Component id="amesim_mecmas21_7" type="amesim_mecmas21" modelVersion="0.2.0">
|
||||||
|
<Parameter name="mass" value="90000"/>
|
||||||
|
<Parameter name="fstick" value="0"/>
|
||||||
|
<Parameter name="fcoul" value="0"/>
|
||||||
|
<Parameter name="rvisc" value="0"/>
|
||||||
|
<Parameter name="wind" value="0"/>
|
||||||
|
<Parameter name="dvel" value="0.000001"/>
|
||||||
|
<Parameter name="restdvel" value="0.000001"/>
|
||||||
|
<Parameter name="restcoeff" value="0.65"/>
|
||||||
|
<Parameter name="astrib" value="0.001"/>
|
||||||
|
<Parameter name="xmin" value="-0.72"/>
|
||||||
|
<Parameter name="Kbmin" value="1000000000"/>
|
||||||
|
<Parameter name="Dbmin" value="10000"/>
|
||||||
|
<Parameter name="Pdmin" value="0.0001"/>
|
||||||
|
<Parameter name="xmax" value="0"/>
|
||||||
|
<Parameter name="Kbmax" value="1000000000"/>
|
||||||
|
<Parameter name="Dbmax" value="10000"/>
|
||||||
|
<Parameter name="Pdmax" value="0.0001"/>
|
||||||
|
<Parameter name="theta" value="0"/>
|
||||||
|
<Parameter name="useFriction" value="1"/>
|
||||||
|
<Parameter name="stoptype" value="1"/>
|
||||||
|
<Parameter name="discContactOption" value="1"/>
|
||||||
|
<Parameter name="strib" value="1"/>
|
||||||
|
<Parameter name="frictionType" value="1"/>
|
||||||
|
<Parameter name="v0" value="0"/>
|
||||||
|
<Parameter name="x0" value="0"/>
|
||||||
|
</Component>
|
||||||
|
</Components>
|
||||||
|
<Connections>
|
||||||
|
<Connection id="edge-amesim_pnch023_1-port_2-amesim_pnvo001_1-port_2-1786524999267">
|
||||||
|
<Endpoint component="amesim_pnch023_1" port="port_2"/>
|
||||||
|
<Endpoint component="amesim_pnvo001_1" port="port_2"/>
|
||||||
|
</Connection>
|
||||||
|
<Connection id="edge-amesim_pnvo001_1-port_3-amesim_pnch012_1-port_1-1786525009973">
|
||||||
|
<Endpoint component="amesim_pnvo001_1" port="port_3"/>
|
||||||
|
<Endpoint component="amesim_pnch012_1" port="port_1"/>
|
||||||
|
</Connection>
|
||||||
|
<Connection id="edge-contact-amesim_pnpl01_2-port_1-amesim_pnch012_1-port_2-1786525035706-0">
|
||||||
|
<Endpoint component="amesim_pnpl01_2" port="port_1"/>
|
||||||
|
<Endpoint component="amesim_pnch012_1" port="port_2"/>
|
||||||
|
</Connection>
|
||||||
|
<Connection id="edge-contact-amesim_pnpl01_3-port_1-amesim_pnch012_1-port_3-1786525054054-0">
|
||||||
|
<Endpoint component="amesim_pnpl01_3" port="port_1"/>
|
||||||
|
<Endpoint component="amesim_pnch012_1" port="port_3"/>
|
||||||
|
</Connection>
|
||||||
|
<Connection id="edge-contact-amesim_pnpl01_4-port_1-amesim_pnch012_1-port_4-1786525061349-0">
|
||||||
|
<Endpoint component="amesim_pnpl01_4" port="port_1"/>
|
||||||
|
<Endpoint component="amesim_pnch012_1" port="port_4"/>
|
||||||
|
</Connection>
|
||||||
|
<Connection id="edge-amesim_step0_1-out-amesim_pnvo001_1-res-1786525074494">
|
||||||
|
<Endpoint component="amesim_step0_1" port="out"/>
|
||||||
|
<Endpoint component="amesim_pnvo001_1" port="res"/>
|
||||||
|
</Connection>
|
||||||
|
<Connection id="edge-amesim_ud00_1-out-amesim_forc_1-res-1786525088412">
|
||||||
|
<Endpoint component="amesim_ud00_1" port="out"/>
|
||||||
|
<Endpoint component="amesim_forc_1" port="res"/>
|
||||||
|
</Connection>
|
||||||
|
<Connection id="edge-contact-amesim_pnpl01_6-port_1-amesim_pnch023_1-port_1-1786525164428-0">
|
||||||
|
<Endpoint component="amesim_pnpl01_6" port="port_1"/>
|
||||||
|
<Endpoint component="amesim_pnch023_1" port="port_1"/>
|
||||||
|
</Connection>
|
||||||
|
<Connection id="edge-contact-amesim_pnpl01_7-port_1-amesim_pnch012_2-port_1-1786525181969-0">
|
||||||
|
<Endpoint component="amesim_pnpl01_7" port="port_1"/>
|
||||||
|
<Endpoint component="amesim_pnch012_2" port="port_1"/>
|
||||||
|
</Connection>
|
||||||
|
<Connection id="edge-contact-amesim_pnpl01_8-port_1-amesim_pnch012_2-port_4-1786525185493-0">
|
||||||
|
<Endpoint component="amesim_pnpl01_8" port="port_1"/>
|
||||||
|
<Endpoint component="amesim_pnch012_2" port="port_4"/>
|
||||||
|
</Connection>
|
||||||
|
<Connection id="edge-contact-amesim_pnpl01_9-port_1-amesim_pnch012_2-port_2-1786525188482-0">
|
||||||
|
<Endpoint component="amesim_pnpl01_9" port="port_1"/>
|
||||||
|
<Endpoint component="amesim_pnch012_2" port="port_2"/>
|
||||||
|
</Connection>
|
||||||
|
<Connection id="edge-amesim_mecmas21_2-port_1-amesim_pnrp17_1-port_2-1786525203778">
|
||||||
|
<Endpoint component="amesim_mecmas21_2" port="port_1"/>
|
||||||
|
<Endpoint component="amesim_pnrp17_1" port="port_2"/>
|
||||||
|
</Connection>
|
||||||
|
<Connection id="edge-amesim_pnrp17_1-port_5-amesim_lstp00a_1-port_1-1786525213548">
|
||||||
|
<Endpoint component="amesim_pnrp17_1" port="port_5"/>
|
||||||
|
<Endpoint component="amesim_lstp00a_1" port="port_1"/>
|
||||||
|
</Connection>
|
||||||
|
<Connection id="edge-amesim_ud00_2-out-amesim_forc_2-res-1786525226761">
|
||||||
|
<Endpoint component="amesim_ud00_2" port="out"/>
|
||||||
|
<Endpoint component="amesim_forc_2" port="res"/>
|
||||||
|
</Connection>
|
||||||
|
<Connection id="edge-contact-amesim_pnrp17_1-port_1-amesim_pnch012_2-port_3-1786525729832-1">
|
||||||
|
<Endpoint component="amesim_pnrp17_1" port="port_1"/>
|
||||||
|
<Endpoint component="amesim_pnch012_2" port="port_3"/>
|
||||||
|
</Connection>
|
||||||
|
<Connection id="edge-amesim_forc_1-port_2-amesim_mecmas21_7-port_2-1786525915163">
|
||||||
|
<Endpoint component="amesim_forc_1" port="port_2"/>
|
||||||
|
<Endpoint component="amesim_mecmas21_7" port="port_2"/>
|
||||||
|
</Connection>
|
||||||
|
<Connection id="edge-amesim_mecmas21_7-port_1-amesim_pnrp17_1-port_3-1786525916652">
|
||||||
|
<Endpoint component="amesim_mecmas21_7" port="port_1"/>
|
||||||
|
<Endpoint component="amesim_pnrp17_1" port="port_3"/>
|
||||||
|
</Connection>
|
||||||
|
<Connection id="edge-amesim_mecmas21_5-port_1-amesim_forc_2-port_2-1786525922010">
|
||||||
|
<Endpoint component="amesim_mecmas21_5" port="port_1"/>
|
||||||
|
<Endpoint component="amesim_forc_2" port="port_2"/>
|
||||||
|
</Connection>
|
||||||
|
<Connection id="edge-amesim_lstp00a_1-port_2-amesim_mecmas21_5-port_2-1786525924224">
|
||||||
|
<Endpoint component="amesim_lstp00a_1" port="port_2"/>
|
||||||
|
<Endpoint component="amesim_mecmas21_5" port="port_2"/>
|
||||||
|
</Connection>
|
||||||
|
<Connection id="edge-contact-amesim_f000_2-port_1-amesim_pnrp17_1-port_4-1788429018340-0">
|
||||||
|
<Endpoint component="amesim_f000_2" port="port_1"/>
|
||||||
|
<Endpoint component="amesim_pnrp17_1" port="port_4"/>
|
||||||
|
</Connection>
|
||||||
|
<Connection id="edge-contact-amesim_f000_1-port_1-amesim_mecmas21_2-port_2-1788429022851-0">
|
||||||
|
<Endpoint component="amesim_f000_1" port="port_1"/>
|
||||||
|
<Endpoint component="amesim_mecmas21_2" port="port_2"/>
|
||||||
|
</Connection>
|
||||||
|
</Connections>
|
||||||
|
</System>
|
||||||
@@ -0,0 +1,300 @@
|
|||||||
|
"""Serve the real app with opt-in, request-scoped stage measurements.
|
||||||
|
|
||||||
|
All C changes are sparse clocks in an isolated copy of runtime/main.c. Component
|
||||||
|
kernels and the solver are copied unchanged. No production modules are edited.
|
||||||
|
Use --plain for the uninstrumented HTTP/browser control. Output must be fresh.
|
||||||
|
"""
|
||||||
|
from __future__ import annotations
|
||||||
|
import argparse
|
||||||
|
from contextvars import ContextVar
|
||||||
|
from functools import wraps
|
||||||
|
from hashlib import sha256
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import resource
|
||||||
|
from pathlib import Path
|
||||||
|
import shutil
|
||||||
|
import subprocess
|
||||||
|
import sys
|
||||||
|
import threading
|
||||||
|
import time
|
||||||
|
from uuid import uuid4
|
||||||
|
|
||||||
|
ROOT = Path(__file__).resolve().parents[2]
|
||||||
|
sys.path.insert(0, str(ROOT))
|
||||||
|
LOCAL = threading.local()
|
||||||
|
REQUEST = ContextVar('benchmark_request', default=None)
|
||||||
|
|
||||||
|
|
||||||
|
def isolate_runtime(output):
|
||||||
|
target = output/'native'
|
||||||
|
shutil.copytree(ROOT/'native', target)
|
||||||
|
p = target/'runtime/main.c'; source = p.read_text()
|
||||||
|
source = '#include <time.h>\n' + source
|
||||||
|
signature = next(line for line in source.splitlines() if line.startswith('static int write_result(NativeRun *r,'))
|
||||||
|
source = source.replace(signature,
|
||||||
|
'static double profile_projection_seconds, profile_json_write_seconds;\n'
|
||||||
|
'static double profile_projection_cpu_seconds, profile_json_write_cpu_seconds;\n'
|
||||||
|
+ signature + '\n'
|
||||||
|
' double profile_output_start=native_wall_time();\n'
|
||||||
|
' double profile_output_cpu_start=(double)clock()/CLOCKS_PER_SEC;')
|
||||||
|
source = source.replace(' FILE *f=fopen(path,"wb");',
|
||||||
|
' profile_projection_seconds=native_wall_time()-profile_output_start;\n'
|
||||||
|
' profile_projection_cpu_seconds=(double)clock()/CLOCKS_PER_SEC-profile_output_cpu_start;\n'
|
||||||
|
' double profile_write_start=native_wall_time();\n'
|
||||||
|
' double profile_write_cpu_start=(double)clock()/CLOCKS_PER_SEC;\n FILE *f=fopen(path,"wb");')
|
||||||
|
source = source.replace(' free(values); return ok;',
|
||||||
|
' profile_json_write_seconds=native_wall_time()-profile_write_start;\n'
|
||||||
|
' profile_json_write_cpu_seconds=(double)clock()/CLOCKS_PER_SEC-profile_write_cpu_start;\n'
|
||||||
|
' free(values); return ok;')
|
||||||
|
source = source.replace('int main(int argc, char **argv) {',
|
||||||
|
'int main(int argc, char **argv) {\n double profile_main_start=native_wall_time();')
|
||||||
|
source = source.replace(' native_solve(&r);',
|
||||||
|
' double profile_solve_call_start=native_wall_time();\n native_solve(&r);\n'
|
||||||
|
' double profile_solve_call_end=native_wall_time();')
|
||||||
|
source = source.replace(' native_run_free(&r); return code;', r''' native_run_free(&r);
|
||||||
|
fprintf(stderr,"{\"event\":\"native-stage-profile\",\"mainStartMonotonic\":%.17g,"
|
||||||
|
"\"solveCallStartMonotonic\":%.17g,\"integrationStartMonotonic\":%.17g,"
|
||||||
|
"\"integrationSeconds\":%.17g,\"solveCallEndMonotonic\":%.17g,"
|
||||||
|
"\"argumentPreparationSeconds\":%.17g,\"initializationSeconds\":%.17g,"
|
||||||
|
"\"finalSampleAndStatusSeconds\":%.17g,\"projectionSeconds\":%.17g,"
|
||||||
|
"\"jsonWriteSeconds\":%.17g,\"projectionCpuSeconds\":%.17g,\"jsonWriteCpuSeconds\":%.17g,\"mainTotalSeconds\":%.17g}\n",
|
||||||
|
profile_main_start,profile_solve_call_start,r.wall_start,r.solve_seconds,profile_solve_call_end,
|
||||||
|
profile_solve_call_start-profile_main_start,r.wall_start-profile_solve_call_start,
|
||||||
|
profile_solve_call_end-r.wall_start-r.solve_seconds,profile_projection_seconds,
|
||||||
|
profile_json_write_seconds,profile_projection_cpu_seconds,profile_json_write_cpu_seconds,
|
||||||
|
native_wall_time()-profile_main_start);
|
||||||
|
return code;''')
|
||||||
|
if source.count('native-stage-profile') != 1:
|
||||||
|
raise ValueError('Unrecognized runtime layout')
|
||||||
|
p.write_text(source)
|
||||||
|
# The measurement copy must preserve all numerical files verbatim.
|
||||||
|
for original in (ROOT/'native').rglob('*'):
|
||||||
|
if original.is_file() and original.relative_to(ROOT/'native').as_posix() != 'runtime/main.c':
|
||||||
|
assert original.read_bytes() == (target/original.relative_to(ROOT/'native')).read_bytes()
|
||||||
|
return target
|
||||||
|
|
||||||
|
|
||||||
|
class Profile:
|
||||||
|
def __init__(self, output):
|
||||||
|
self.output=output; self.records={}; self.trackers={}; self.results={}
|
||||||
|
|
||||||
|
def current(self):
|
||||||
|
return getattr(LOCAL,'record',None) or REQUEST.get()
|
||||||
|
|
||||||
|
def record(self, record, name, start, end):
|
||||||
|
record['spans'].append({'name':name,'startMs':(start-record['startNs'])/1e6,
|
||||||
|
'endMs':(end-record['startNs'])/1e6,'seconds':(end-start)/1e9})
|
||||||
|
|
||||||
|
def wrap(self, function, name):
|
||||||
|
@wraps(function)
|
||||||
|
def measured(*args, **kwargs):
|
||||||
|
record=self.current()
|
||||||
|
if record is None: return function(*args, **kwargs)
|
||||||
|
start=time.perf_counter_ns()
|
||||||
|
try:return function(*args, **kwargs)
|
||||||
|
finally:self.record(record,name,start,time.perf_counter_ns())
|
||||||
|
return measured
|
||||||
|
|
||||||
|
def install(self, api, builder, runner):
|
||||||
|
profile=self
|
||||||
|
api.validate_system_xml_document=self.wrap(api.validate_system_xml_document,'xml_validation')
|
||||||
|
api.compile_system_xml_network=self.wrap(api.compile_system_xml_network,'network_compilation')
|
||||||
|
runner.compile_native_program=self.wrap(runner.compile_native_program,'c_generation')
|
||||||
|
if hasattr(runner,'read_indexed_result'):
|
||||||
|
runner.read_indexed_result=self.wrap(runner.read_indexed_result,'native_indexed_result_read')
|
||||||
|
from app.simulation.native_codegen import transport
|
||||||
|
class TransportJson:
|
||||||
|
def __getattr__(self,key):return getattr(json,key)
|
||||||
|
def loads(self,value,*args,**kwargs):
|
||||||
|
return profile.wrap(json.loads,'native_result_metadata_json_parse')(value,*args,**kwargs)
|
||||||
|
transport.json=TransportJson()
|
||||||
|
api.serialize_result_parts=self.wrap(api.serialize_result_parts,'response_result_json_serialization')
|
||||||
|
original_read_bytes=Path.read_bytes
|
||||||
|
def read_bytes(path,*args,**kwargs):
|
||||||
|
name='native_result_read_bytes' if path.name=='result.json' else 'native_result_index_read_bytes'
|
||||||
|
if path.name in ('result.json','result-index.json') and self.current() is not None:
|
||||||
|
return self.wrap(original_read_bytes,name)(path,*args,**kwargs)
|
||||||
|
return original_read_bytes(path,*args,**kwargs)
|
||||||
|
Path.read_bytes=read_bytes
|
||||||
|
original_build=runner.build_native
|
||||||
|
def isolated_build(program, **kwargs):
|
||||||
|
kwargs['cache_dir']=self.output/'cache'
|
||||||
|
result=original_build(program,**kwargs)
|
||||||
|
rec=self.current()
|
||||||
|
if rec is not None:
|
||||||
|
rec['build']={'cacheHit':result.cache_hit,'buildKey':result.manifest['buildKey'],
|
||||||
|
'reportedSeconds':result.seconds,'executable':str(result.executable)}
|
||||||
|
return result
|
||||||
|
runner.build_native=self.wrap(isolated_build,'native_build_or_cache_validation')
|
||||||
|
builder.build_native=runner.build_native
|
||||||
|
# Observe process creation and reaping without polling more often or
|
||||||
|
# changing the production stderr-reader/cancellation loop.
|
||||||
|
class RunnerSubprocess:
|
||||||
|
def __getattr__(self, key): return getattr(subprocess, key)
|
||||||
|
def Popen(self, *args, **kwargs):
|
||||||
|
rec=profile.current()
|
||||||
|
if rec is None: return subprocess.Popen(*args, **kwargs)
|
||||||
|
start=time.perf_counter_ns()
|
||||||
|
usage=resource.getrusage(resource.RUSAGE_CHILDREN)
|
||||||
|
process=subprocess.Popen(*args, **kwargs)
|
||||||
|
profile.record(rec,'native_process_spawn',start,time.perf_counter_ns())
|
||||||
|
rec['process']={'pid':process.pid,'command':list(args[0]),
|
||||||
|
'startMs':(start-rec['startNs'])/1e6}
|
||||||
|
original_poll,original_wait=process.poll,process.wait
|
||||||
|
def observe(code):
|
||||||
|
if code is not None and 'exitObservedMs' not in rec['process']:
|
||||||
|
now=time.perf_counter_ns()
|
||||||
|
after=resource.getrusage(resource.RUSAGE_CHILDREN)
|
||||||
|
rec['process'].update(exitCode=code,exitObservedMs=(now-rec['startNs'])/1e6,
|
||||||
|
childrenUserCpuSeconds=after.ru_utime-usage.ru_utime,
|
||||||
|
childrenSystemCpuSeconds=after.ru_stime-usage.ru_stime)
|
||||||
|
profile.record(rec,'native_process_lifetime_observed',start,now)
|
||||||
|
return code
|
||||||
|
def poll(*a,**k):return observe(original_poll(*a,**k))
|
||||||
|
def wait(*a,**k):return observe(original_wait(*a,**k))
|
||||||
|
process.poll,process.wait=poll,wait
|
||||||
|
return process
|
||||||
|
runner.subprocess=RunnerSubprocess()
|
||||||
|
original_execute=runner.execute_native
|
||||||
|
def execute(*args, **kwargs):
|
||||||
|
data=original_execute(*args,**kwargs)
|
||||||
|
rec=self.current()
|
||||||
|
if rec is not None:
|
||||||
|
rec['native']={k:v for k,v in data.items() if k not in ('series','final','finalState')}
|
||||||
|
series=data['series']
|
||||||
|
rec['sampleCount']=series.sample_count if hasattr(series,'sample_count') else len(series.get('time',[]))
|
||||||
|
if hasattr(series,'data'):rec['rawSeriesBytes']=len(series.data)
|
||||||
|
started=time.perf_counter_ns()
|
||||||
|
directory=self.output/'requests'/rec['id'];directory.mkdir(parents=True,exist_ok=True)
|
||||||
|
source=kwargs['run_dir']/'result.json';target=directory/'native-result.json'
|
||||||
|
try:os.link(source,target)
|
||||||
|
except OSError:shutil.copyfile(source,target)
|
||||||
|
shutil.copyfile(kwargs['run_dir']/'worker.log',directory/'worker.log')
|
||||||
|
index=kwargs['run_dir']/'result-index.json'
|
||||||
|
if index.exists():shutil.copyfile(index,directory/'result-index.json')
|
||||||
|
self.record(rec,'profile_artifact_preservation',started,time.perf_counter_ns())
|
||||||
|
return data
|
||||||
|
runner.execute_native=self.wrap(execute,'native_execution_with_result_read')
|
||||||
|
runner.simulate_native=self.wrap(runner.simulate_native,'native_orchestration_total')
|
||||||
|
original_read=Path.read_text
|
||||||
|
@wraps(original_read)
|
||||||
|
def read_text(path,*args,**kwargs):
|
||||||
|
if path.name=='result.json' and self.current() is not None:
|
||||||
|
return self.wrap(original_read,'native_result_read_utf8')(path,*args,**kwargs)
|
||||||
|
return original_read(path,*args,**kwargs)
|
||||||
|
Path.read_text=read_text
|
||||||
|
class RunnerJson:
|
||||||
|
def __getattr__(self,key):return getattr(json,key)
|
||||||
|
def loads(self,text,*args,**kwargs):
|
||||||
|
rec=profile.current()
|
||||||
|
if rec is not None and len(text)>100000:
|
||||||
|
return profile.wrap(json.loads,'native_result_json_parse')(text,*args,**kwargs)
|
||||||
|
value=json.loads(text,*args,**kwargs)
|
||||||
|
if rec is not None and isinstance(value,dict) and value.get('event')=='native-stage-profile':
|
||||||
|
rec['nativeStages']=value
|
||||||
|
return value
|
||||||
|
runner.json=RunnerJson()
|
||||||
|
original_run=api.run_system_xml_simulation
|
||||||
|
def run(xml,*args,**kwargs):
|
||||||
|
tracker=kwargs.get('activity_tracker') or (args[2] if len(args)>2 else None)
|
||||||
|
rec=self.trackers.get(id(tracker)) or REQUEST.get()
|
||||||
|
if rec is None:return original_run(xml,*args,**kwargs)
|
||||||
|
LOCAL.record=rec
|
||||||
|
started=time.perf_counter_ns()
|
||||||
|
try:
|
||||||
|
result=original_run(xml,*args,**kwargs)
|
||||||
|
self.results[id(result)]=rec
|
||||||
|
rec['existingPerformance']=result['diagnostics'].get('performance')
|
||||||
|
return result
|
||||||
|
finally:
|
||||||
|
self.record(rec,'simulation_worker_total',started,time.perf_counter_ns())
|
||||||
|
LOCAL.record=None
|
||||||
|
api.run_system_xml_simulation=run
|
||||||
|
original_stream=api.simulation_event_stream
|
||||||
|
def stream(xml,*,task=None,**kwargs):
|
||||||
|
rec=self.records.get(task.simulation_id) if task else None
|
||||||
|
if rec is not None:
|
||||||
|
self.trackers[id(task.activity_tracker)]=rec
|
||||||
|
rec['xmlSha256']=sha256(xml).hexdigest();rec['xmlBytes']=len(xml)
|
||||||
|
directory=self.output/'requests'/rec['id'];directory.mkdir(parents=True,exist_ok=True)
|
||||||
|
(directory/'input.xml').write_bytes(xml)
|
||||||
|
yield from original_stream(xml,task=task,**kwargs)
|
||||||
|
api.simulation_event_stream=stream
|
||||||
|
class ApiJson:
|
||||||
|
def __getattr__(self,key):return getattr(json,key)
|
||||||
|
def dumps(self,value,*args,**kwargs):
|
||||||
|
rec=profile.results.get(id(value.get('result'))) if isinstance(value,dict) and value.get('event')=='result' else None
|
||||||
|
if rec is None:return json.dumps(value,*args,**kwargs)
|
||||||
|
start=time.perf_counter_ns();result=json.dumps(value,*args,**kwargs)
|
||||||
|
profile.record(rec,'response_result_json_serialization',start,time.perf_counter_ns())
|
||||||
|
rec['resultJsonCharacters']=len(result)
|
||||||
|
return result
|
||||||
|
api.json=ApiJson()
|
||||||
|
api.build_simulation_results_csv=self.wrap(api.build_simulation_results_csv,'csv_assembly')
|
||||||
|
|
||||||
|
def app(self, underlying):
|
||||||
|
async def wrapped(scope,receive,send):
|
||||||
|
if scope['type']!='http' or scope.get('path') not in ('/api/system-xml/simulate-stream','/api/simulation-results/csv'):
|
||||||
|
return await underlying(scope,receive,send)
|
||||||
|
headers=dict(scope.get('headers',[]));ident=headers.get(b'x-simulation-id',uuid4().hex.encode()).decode()
|
||||||
|
if not all(c.isalnum() or c in '-_' for c in ident): raise ValueError('Invalid profiling request ID')
|
||||||
|
rec={'id':ident,'path':scope['path'],'startNs':time.perf_counter_ns(),'spans':[],
|
||||||
|
'responseBodyBytes':0,'responseSendAwaitSeconds':0.0}
|
||||||
|
self.records[ident]=rec;token=REQUEST.set(rec)
|
||||||
|
async def observed_receive():
|
||||||
|
message=await receive()
|
||||||
|
if message['type']=='http.request' and not message.get('more_body',False):
|
||||||
|
rec['requestBodyCompleteMs']=(time.perf_counter_ns()-rec['startNs'])/1e6
|
||||||
|
return message
|
||||||
|
async def observed_send(message):
|
||||||
|
start=time.perf_counter_ns()
|
||||||
|
if message['type']=='http.response.start':
|
||||||
|
rec['responseHeadersMs']=(start-rec['startNs'])/1e6;rec['httpStatus']=message['status']
|
||||||
|
if message['type']=='http.response.body':
|
||||||
|
rec['responseBodyBytes']+=len(message.get('body',b''))
|
||||||
|
if len(message.get('body',b''))>100000:
|
||||||
|
rec['largeResultBodySendStartMs']=(start-rec['startNs'])/1e6
|
||||||
|
await send(message)
|
||||||
|
rec['responseSendAwaitSeconds']+=(time.perf_counter_ns()-start)/1e9
|
||||||
|
if message['type']=='http.response.body' and not message.get('more_body',False):
|
||||||
|
rec['responseBodyCompleteMs']=(time.perf_counter_ns()-rec['startNs'])/1e6
|
||||||
|
try:await underlying(scope,observed_receive,observed_send)
|
||||||
|
finally:
|
||||||
|
rec['httpTotalSeconds']=(time.perf_counter_ns()-rec['startNs'])/1e9
|
||||||
|
REQUEST.reset(token)
|
||||||
|
target=self.output/'requests'/ident;target.mkdir(parents=True,exist_ok=True)
|
||||||
|
(target/'stages.json').write_text(json.dumps(rec,ensure_ascii=False,indent=2)+'\n')
|
||||||
|
return wrapped
|
||||||
|
|
||||||
|
|
||||||
|
def main():
|
||||||
|
parser=argparse.ArgumentParser(description=__doc__)
|
||||||
|
parser.add_argument('--output-dir',type=Path,required=True)
|
||||||
|
parser.add_argument('--port',type=int,default=8012)
|
||||||
|
parser.add_argument('--plain',action='store_true')
|
||||||
|
parser.add_argument('--frontend-dist',type=Path,default=ROOT/'frontend/dist')
|
||||||
|
args=parser.parse_args();out=args.output_dir.resolve()
|
||||||
|
if out.exists():parser.error('Choose a fresh output directory')
|
||||||
|
out.mkdir(parents=True)
|
||||||
|
# Snapshot static assets so another workspace build cannot change a run.
|
||||||
|
shutil.copytree(args.frontend_dist,out/'frontend')
|
||||||
|
if not args.plain:os.environ['SIMULATIONAPP_PROFILE']='standard'
|
||||||
|
import app.main as api
|
||||||
|
from app.simulation.native_codegen import build as builder,runner
|
||||||
|
api.FRONTEND_DIST_DIR=out/'frontend'
|
||||||
|
profile=None
|
||||||
|
if not args.plain:
|
||||||
|
builder.NATIVE=isolate_runtime(out)
|
||||||
|
profile=Profile(out);profile.install(api,builder,runner)
|
||||||
|
metadata={'gitHead':subprocess.check_output(['git','rev-parse','HEAD'],cwd=ROOT,text=True).strip(),
|
||||||
|
'mode':'plain' if args.plain else 'profile','port':args.port,
|
||||||
|
'python':sys.version,'platform':sys.platform,
|
||||||
|
'frontendFiles':{str(p.relative_to(out/'frontend')):sha256(p.read_bytes()).hexdigest() for p in (out/'frontend').rglob('*') if p.is_file()},
|
||||||
|
'productionKernelSha256':sha256((ROOT/'native/components/kernels.c').read_bytes()).hexdigest()}
|
||||||
|
(out/'environment.json').write_text(json.dumps(metadata,indent=2)+'\n')
|
||||||
|
import uvicorn
|
||||||
|
uvicorn.run(profile.app(api.app) if profile else api.app,host='127.0.0.1',port=args.port)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__=='__main__':main()
|
||||||
@@ -0,0 +1,318 @@
|
|||||||
|
"""Replay native result encoding in isolated C programs, without model evaluation.
|
||||||
|
|
||||||
|
Prepare only by default. Explicit --run builds a small C replay and serially
|
||||||
|
runs one warmup and three measured real-file outputs per variant. --dev-null
|
||||||
|
adds sink-only runs after real-file verification; these never stand in for I/O.
|
||||||
|
|
||||||
|
.venv/bin/python tests/manual/benchmark_native_result_encoding.py \
|
||||||
|
--result-json test/web-cost-20260911/native-compute-profile/control/run-1/result.json \
|
||||||
|
--output-dir test/c-result-encoding-20260911 --ryu-root /path/to/ryu --run
|
||||||
|
|
||||||
|
ryu-root must contain ryu/d2s.c and ryu/ryu.h (the ryu/ subdirectory itself is
|
||||||
|
also accepted). No dependency is downloaded and no production source is edited.
|
||||||
|
Every series cell, final scalar and finalState cell is encoded in C. Static
|
||||||
|
metadata, JSON structure and escaped keys are prepared outside timing. Values
|
||||||
|
are loaded contiguously before timing; this isolates decimal encoding/write
|
||||||
|
cost, excluding projection and the production writer's strided matrix reads.
|
||||||
|
Timers include fopen, buffer setup, formatting, write, flush and fclose, but
|
||||||
|
not fsync durability. All real-file outputs are parsed and compared as complete
|
||||||
|
binary64 values, including signed zero, outside the measured interval.
|
||||||
|
"""
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
from array import array
|
||||||
|
from hashlib import sha256
|
||||||
|
import json
|
||||||
|
import math
|
||||||
|
import os
|
||||||
|
from pathlib import Path
|
||||||
|
import shutil
|
||||||
|
import statistics
|
||||||
|
import struct
|
||||||
|
import subprocess
|
||||||
|
import sys
|
||||||
|
import time
|
||||||
|
|
||||||
|
ROOT = Path(__file__).resolve().parents[2]
|
||||||
|
VARIANTS = ["fprintf-default", "fprintf-1m", "snprintf-64k", "ryu-64k"]
|
||||||
|
C_SOURCE = r'''
|
||||||
|
#include <stdio.h>
|
||||||
|
#include <stdlib.h>
|
||||||
|
#include <stdint.h>
|
||||||
|
#include <string.h>
|
||||||
|
#include <math.h>
|
||||||
|
#include <time.h>
|
||||||
|
#if HAVE_RYU
|
||||||
|
#include "ryu/ryu.h"
|
||||||
|
#endif
|
||||||
|
#include "replay-layout.h"
|
||||||
|
#define CHUNK (64u*1024u)
|
||||||
|
typedef struct { FILE *f; char block[CHUNK]; size_t used; unsigned long long bytes; int failed; } Writer;
|
||||||
|
static double wall_now(void) { struct timespec t; if(clock_gettime(CLOCK_MONOTONIC,&t))exit(72); return t.tv_sec+t.tv_nsec*1e-9; }
|
||||||
|
static double cpu_now(void) { struct timespec t; if(clock_gettime(CLOCK_PROCESS_CPUTIME_ID,&t))exit(72); return t.tv_sec+t.tv_nsec*1e-9; }
|
||||||
|
static void flush_block(Writer *w) {
|
||||||
|
if(w->used && fwrite(w->block,1,w->used,w->f)!=w->used)w->failed=1;
|
||||||
|
w->used=0;
|
||||||
|
}
|
||||||
|
static void block_bytes(Writer *w,const char *text,size_t n) {
|
||||||
|
w->bytes+=n;
|
||||||
|
while(n) {
|
||||||
|
size_t left=CHUNK-w->used, part=n<left?n:left;
|
||||||
|
memcpy(w->block+w->used,text,part); w->used+=part; text+=part; n-=part;
|
||||||
|
if(w->used==CHUNK)flush_block(w);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
static void direct_bytes(Writer *w,const char *text,size_t n) {
|
||||||
|
w->bytes+=n;
|
||||||
|
if(fwrite(text,1,n,w->f)!=n)w->failed=1;
|
||||||
|
}
|
||||||
|
static void encode(Writer *w,const double *values,int mode) {
|
||||||
|
for(size_t g=0;g<GROUP_COUNT;g++) {
|
||||||
|
const Group *group=&groups[g];
|
||||||
|
if(mode<2)direct_bytes(w,group->prefix,group->prefix_length);
|
||||||
|
else block_bytes(w,group->prefix,group->prefix_length);
|
||||||
|
for(size_t i=0;i<group->count;i++) {
|
||||||
|
double value=values[group->offset+i];
|
||||||
|
if(mode<2) {
|
||||||
|
/* Same number/separator formatting call as native main.c. */
|
||||||
|
int n=fprintf(w->f,"%s%.17g",i?",":"",value);
|
||||||
|
if(n<0)w->failed=1; else w->bytes+=(unsigned)n;
|
||||||
|
} else if(mode==2) {
|
||||||
|
/* Format directly into the remaining batch buffer. */
|
||||||
|
if(CHUNK-w->used<64)flush_block(w);
|
||||||
|
int n=snprintf(w->block+w->used,CHUNK-w->used,"%s%.17g",i?",":"",value);
|
||||||
|
if(n<0 || (size_t)n>=CHUNK-w->used){w->failed=1;return;}
|
||||||
|
w->used+=(unsigned)n;w->bytes+=(unsigned)n;
|
||||||
|
} else {
|
||||||
|
#if HAVE_RYU
|
||||||
|
if(CHUNK-w->used<64)flush_block(w);
|
||||||
|
if(i){w->block[w->used++]=',';w->bytes++;}
|
||||||
|
int n=d2s_buffered_n(value,w->block+w->used);
|
||||||
|
if(n<1 || n>32){w->failed=1;return;}
|
||||||
|
w->used+=(unsigned)n;w->bytes+=(unsigned)n;
|
||||||
|
#else
|
||||||
|
w->failed=1;return;
|
||||||
|
#endif
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if(mode<2)direct_bytes(w,tail,TAIL_LENGTH);
|
||||||
|
else {block_bytes(w,tail,TAIL_LENGTH);flush_block(w);}
|
||||||
|
}
|
||||||
|
int main(int argc,char **argv) {
|
||||||
|
if(argc!=4)return 64;
|
||||||
|
int mode=-1;
|
||||||
|
const char *names[]={"fprintf-default","fprintf-1m","snprintf-64k","ryu-64k"};
|
||||||
|
for(int i=0;i<4;i++)if(!strcmp(argv[1],names[i]))mode=i;
|
||||||
|
if(mode<0 || (mode==3 && !HAVE_RYU))return 64;
|
||||||
|
if(sizeof(double)!=8 || sizeof(uint64_t)!=8)return 65;
|
||||||
|
FILE *input=fopen(argv[2],"rb"); if(!input)return 66;
|
||||||
|
double *values=malloc(VALUE_COUNT*sizeof(double));
|
||||||
|
if(!values){fclose(input);return 67;}
|
||||||
|
int loaded=fread(values,sizeof(double),VALUE_COUNT,input)==VALUE_COUNT && fgetc(input)==EOF && !ferror(input);
|
||||||
|
if(fclose(input))loaded=0;
|
||||||
|
if(!loaded){free(values);return 68;}
|
||||||
|
for(size_t i=0;i<VALUE_COUNT;i++)if(!isfinite(values[i])){free(values);return 69;}
|
||||||
|
Writer *writer=calloc(1,sizeof(Writer)); char *stdio_buffer=malloc(1024u*1024u);
|
||||||
|
if(!writer || !stdio_buffer){free(values);free(writer);free(stdio_buffer);return 67;}
|
||||||
|
/* Loading, allocation and input validation are intentionally outside timing. */
|
||||||
|
double wall_start=wall_now(), cpu_start=cpu_now();
|
||||||
|
writer->f=fopen(argv[3],"wb");
|
||||||
|
if(!writer->f){free(values);free(writer);free(stdio_buffer);return 70;}
|
||||||
|
if(mode==1 && setvbuf(writer->f,stdio_buffer,_IOFBF,1024u*1024u))writer->failed=1;
|
||||||
|
/* Manual batch variants use identical unbuffered FILE sinks. */
|
||||||
|
if(mode>=2 && setvbuf(writer->f,NULL,_IONBF,0))writer->failed=1;
|
||||||
|
if(!writer->failed)encode(writer,values,mode);
|
||||||
|
if(ferror(writer->f))writer->failed=1;
|
||||||
|
if(fclose(writer->f))writer->failed=1;
|
||||||
|
double cpu_seconds=cpu_now()-cpu_start, wall_seconds=wall_now()-wall_start;
|
||||||
|
printf("{\"variant\":\"%s\",\"wallSeconds\":%.17g,\"cpuSeconds\":%.17g,\"encodedBytes\":%llu,\"valueCount\":%zu,\"success\":%s}\n",
|
||||||
|
names[mode],wall_seconds,cpu_seconds,writer->bytes,(size_t)VALUE_COUNT,writer->failed?"false":"true");
|
||||||
|
int code=writer->failed?71:0;
|
||||||
|
free(values);free(stdio_buffer);free(writer);return code;
|
||||||
|
}
|
||||||
|
'''
|
||||||
|
|
||||||
|
|
||||||
|
def digest(path: Path) -> str:
|
||||||
|
return sha256(path.read_bytes()).hexdigest()
|
||||||
|
|
||||||
|
|
||||||
|
def write_json(path: Path, value: object) -> None:
|
||||||
|
path.write_text(json.dumps(value, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
|
||||||
|
|
||||||
|
|
||||||
|
def read_json_numbers(path: Path) -> dict:
|
||||||
|
# JSON's integer spelling -0 must retain the sign before conversion to double.
|
||||||
|
return json.loads(path.read_bytes(), parse_int=lambda token: -0.0 if token == "-0" else int(token))
|
||||||
|
|
||||||
|
|
||||||
|
def c_string(value: bytes) -> str:
|
||||||
|
return '"' + ''.join(f"\\x{byte:02x}" for byte in value) + '"'
|
||||||
|
|
||||||
|
|
||||||
|
def numeric_blocks(result: dict) -> list[tuple[str, list]]:
|
||||||
|
blocks = [(f"series/{key}", value) for key, value in result["series"].items()]
|
||||||
|
blocks.extend((f"final/{key}", [value]) for key, value in result["final"].items())
|
||||||
|
blocks.append(("finalState", result["finalState"]))
|
||||||
|
return blocks
|
||||||
|
|
||||||
|
|
||||||
|
def prepare(args: argparse.Namespace) -> dict:
|
||||||
|
output = args.output_dir.resolve()
|
||||||
|
if not output.is_relative_to(ROOT / "test"):
|
||||||
|
raise RuntimeError("Output must be beneath the repository's ignored test/ directory")
|
||||||
|
if sys.byteorder != "little" or array('d').itemsize != 8 or not sys.platform.startswith("linux"):
|
||||||
|
raise RuntimeError("This isolated replay currently requires Linux and little-endian binary64")
|
||||||
|
result = read_json_numbers(args.result_json)
|
||||||
|
if not isinstance(result.get("series"), dict) or not isinstance(result.get("final"), dict) or not isinstance(result.get("finalState"), list):
|
||||||
|
raise RuntimeError("Input must be a complete native result.json")
|
||||||
|
blocks = numeric_blocks(result)
|
||||||
|
all_values = array('d')
|
||||||
|
descriptors = []
|
||||||
|
for name, values in blocks:
|
||||||
|
if not isinstance(values, list):
|
||||||
|
raise RuntimeError(f"Expected numeric array: {name}")
|
||||||
|
for value in values:
|
||||||
|
if isinstance(value, bool) or not isinstance(value, (int, float)) or not math.isfinite(value):
|
||||||
|
raise RuntimeError(f"Nonfinite or nonnumeric input: {name}")
|
||||||
|
if isinstance(value, int) and int(float(value)) != value:
|
||||||
|
raise RuntimeError(f"Integer does not fit exactly in binary64: {name}")
|
||||||
|
descriptors.append({"name": name, "offset": len(all_values), "count": len(values)})
|
||||||
|
all_values.extend(values)
|
||||||
|
if not all_values:
|
||||||
|
raise RuntimeError("No numeric output values")
|
||||||
|
metadata = {key: value for key, value in result.items() if key not in {"series", "final", "finalState"}}
|
||||||
|
pending = json.dumps(metadata, ensure_ascii=True, separators=(",", ":"), allow_nan=False)[:-1]
|
||||||
|
pending += ("," if metadata else "") + '"series":{'
|
||||||
|
prefixes = []
|
||||||
|
first = True
|
||||||
|
for key in result["series"]:
|
||||||
|
pending += ("" if first else ",") + json.dumps(key, ensure_ascii=True) + ":["
|
||||||
|
prefixes.append(pending.encode()); pending = "]"; first = False
|
||||||
|
pending += '},"final":{'
|
||||||
|
first = True
|
||||||
|
for key in result["final"]:
|
||||||
|
pending += ("" if first else ",") + json.dumps(key, ensure_ascii=True) + ":"
|
||||||
|
prefixes.append(pending.encode()); pending = ""; first = False
|
||||||
|
pending += '},"finalState":['
|
||||||
|
prefixes.append(pending.encode())
|
||||||
|
tail = b"]}\n"
|
||||||
|
output.mkdir(parents=True, exist_ok=True)
|
||||||
|
raw = output / "values.f64le"
|
||||||
|
raw.write_bytes(all_values.tobytes())
|
||||||
|
header = ["/* Generated test data: all numeric payload cells, no projection. */", "typedef struct { const char *prefix; size_t prefix_length, offset, count; } Group;", f"#define GROUP_COUNT {len(descriptors)}u", f"#define VALUE_COUNT {len(all_values)}u", "static const Group groups[]={"]
|
||||||
|
for desc, prefix in zip(descriptors, prefixes, strict=True):
|
||||||
|
header.append(f" {{{c_string(prefix)},{len(prefix)}u,{desc['offset']}u,{desc['count']}u}},")
|
||||||
|
header += ["};", f"static const char tail[]={c_string(tail)};", f"#define TAIL_LENGTH {len(tail)}u"]
|
||||||
|
(output / "replay-layout.h").write_text("\n".join(header) + "\n")
|
||||||
|
(output / "replay.c").write_text(C_SOURCE)
|
||||||
|
compiler = os.environ.get("SIMULATION_NATIVE_CC") or shutil.which("gcc")
|
||||||
|
if not compiler:
|
||||||
|
raise RuntimeError("GCC is required")
|
||||||
|
ryu = args.ryu_root.resolve() if args.ryu_root else None
|
||||||
|
if ryu and not (ryu / "ryu/d2s.c").is_file() and (ryu / "d2s.c").is_file():
|
||||||
|
ryu = ryu.parent
|
||||||
|
if ryu and not all((ryu / name).is_file() for name in ("ryu/d2s.c", "ryu/ryu.h")):
|
||||||
|
raise RuntimeError("--ryu-root must contain ryu/d2s.c and ryu/ryu.h")
|
||||||
|
command = [compiler, "-std=c11", "-O3", "-Wall", "-Wextra", "-Werror", "-ffp-contract=off", "-fno-fast-math", "-D_POSIX_C_SOURCE=200809L", f"-DHAVE_RYU={int(ryu is not None)}", "-I", str(output), str(output / "replay.c")]
|
||||||
|
ryu_hashes = {}
|
||||||
|
if ryu:
|
||||||
|
command += ["-I", str(ryu), str(ryu / "ryu/d2s.c")]
|
||||||
|
ryu_hashes = {str(p.relative_to(ryu)): digest(p) for p in sorted((ryu / "ryu").glob("*")) if p.is_file() and p.suffix in {".c", ".h"}}
|
||||||
|
command += ["-lm", "-o", str(output / "replay")]
|
||||||
|
prepared = {"sourceResult": str(args.result_json.resolve()), "sourceSha256": digest(args.result_json), "sourceBytes": args.result_json.stat().st_size, "rawValuesSha256": digest(raw), "rawValueBytes": raw.stat().st_size, "valueCount": len(all_values), "seriesColumns": len(result["series"]), "seriesValues": sum(len(v) for v in result["series"].values()), "finalValues": len(result["final"]), "finalStateValues": len(result["finalState"]), "blocks": descriptors, "variants": VARIANTS if ryu else VARIANTS[:3], "ryuRoot": str(ryu) if ryu else None, "ryuSourceHashes": ryu_hashes, "buildCommand": command, "compiler": subprocess.run([compiler, "--version"], capture_output=True, text=True, check=True).stdout.splitlines()[0], "warmups": args.warmups, "repeats": args.repeats, "devNullRequested": args.dev_null, "precisionContract": "All finite payload values must decode to identical little-endian binary64 bytes, including signed zero. Shortest output may have different length/exponent spelling.", "timingContract": "C wall/process-CPU from before fopen through fclose, including buffer setup, all numeric payload formatting and writing. Excludes extraction, preload, allocation, static JSON framing preparation and verification. Ordinary files/page cache; no fsync durability. Contiguous replay does not reproduce production matrix strides or output projection. Block variants both use a 64KiB application buffer with unbuffered FILE sink."}
|
||||||
|
write_json(output / "prepared.json", prepared)
|
||||||
|
return prepared
|
||||||
|
|
||||||
|
|
||||||
|
def verify_file(path: Path, expected: dict, raw: bytes, prepared: dict) -> dict:
|
||||||
|
actual = read_json_numbers(path)
|
||||||
|
# This catches missing columns, changed metadata, duplicates in array values,
|
||||||
|
# order differences, and scalar value drift before the exact signed-zero pass.
|
||||||
|
if actual != expected:
|
||||||
|
raise RuntimeError(f"Full result value/structure parity failed: {path}")
|
||||||
|
actual_blocks = numeric_blocks(actual)
|
||||||
|
if [name for name, _ in actual_blocks] != [d["name"] for d in prepared["blocks"]]:
|
||||||
|
raise RuntimeError(f"Numeric block order differs: {path}")
|
||||||
|
negative_zeroes = 0
|
||||||
|
for (_, values), desc in zip(actual_blocks, prepared["blocks"], strict=True):
|
||||||
|
binary = array('d', values).tobytes()
|
||||||
|
start, end = desc["offset"] * 8, (desc["offset"] + desc["count"]) * 8
|
||||||
|
if binary != raw[start:end]:
|
||||||
|
raise RuntimeError(f"Binary64 parity failed at {desc['name']}: {path}")
|
||||||
|
negative_zeroes += sum(value == 0 and math.copysign(1.0, value) < 0 for value in values)
|
||||||
|
return {"fullResultParity": True, "allPayloadBinary64Parity": True, "checkedValues": prepared["valueCount"], "negativeZeroCount": negative_zeroes, "sha256": digest(path)}
|
||||||
|
|
||||||
|
|
||||||
|
def execute(args: argparse.Namespace, prepared: dict) -> None:
|
||||||
|
output = args.output_dir.resolve()
|
||||||
|
built = subprocess.run(prepared["buildCommand"], capture_output=True, text=True, timeout=120)
|
||||||
|
(output / "build.log").write_text(built.stdout + built.stderr)
|
||||||
|
if built.returncode:
|
||||||
|
raise RuntimeError(f"Compilation failed: {output / 'build.log'}")
|
||||||
|
expected = read_json_numbers(args.result_json)
|
||||||
|
raw = (output / "values.f64le").read_bytes()
|
||||||
|
rows = []
|
||||||
|
for sink in (["file", "dev-null"] if args.dev_null else ["file"]):
|
||||||
|
for index in range(-args.warmups, args.repeats):
|
||||||
|
label = f"warmup-{index + args.warmups + 1}" if index < 0 else f"run-{index + 1}"
|
||||||
|
for variant in prepared["variants"]:
|
||||||
|
run = output / sink / variant / label
|
||||||
|
run.mkdir(parents=True, exist_ok=True)
|
||||||
|
target = run / "result.json" if sink == "file" else Path("/dev/null")
|
||||||
|
started = time.perf_counter()
|
||||||
|
process = subprocess.run([str(output / "replay"), variant, str(output / "values.f64le"), str(target)], capture_output=True, text=True, timeout=120)
|
||||||
|
process_wall = time.perf_counter() - started
|
||||||
|
(run / "stdout.log").write_text(process.stdout)
|
||||||
|
(run / "stderr.log").write_text(process.stderr)
|
||||||
|
if process.returncode:
|
||||||
|
raise RuntimeError(f"Replay failed ({process.returncode}): {run}")
|
||||||
|
record = json.loads(process.stdout)
|
||||||
|
if record["success"] is not True:
|
||||||
|
raise RuntimeError(f"Encoding reported failure: {run}")
|
||||||
|
record.update(sink=sink, run=label, warmup=index < 0, processWallSeconds=process_wall)
|
||||||
|
if sink == "file":
|
||||||
|
if target.stat().st_size != record["encodedBytes"]:
|
||||||
|
raise RuntimeError(f"Written byte count differs: {target}")
|
||||||
|
record["verification"] = verify_file(target, expected, raw, prepared)
|
||||||
|
else:
|
||||||
|
record["verification"] = {"actualSinkFileReadback": False, "sameEncoderPassedRealFileReadback": True}
|
||||||
|
write_json(run / "run.json", record)
|
||||||
|
rows.append(record)
|
||||||
|
print(f"{sink}/{variant}/{label}: wall={record['wallSeconds']:.6f}s cpu={record['cpuSeconds']:.6f}s bytes={record['encodedBytes']}", flush=True)
|
||||||
|
medians = {}
|
||||||
|
for sink in {row["sink"] for row in rows}:
|
||||||
|
medians[sink] = {}
|
||||||
|
for variant in prepared["variants"]:
|
||||||
|
selected = [row for row in rows if row["sink"] == sink and row["variant"] == variant and not row["warmup"]]
|
||||||
|
medians[sink][variant] = {key: statistics.median(row[key] for row in selected) for key in ("wallSeconds", "cpuSeconds", "encodedBytes")}
|
||||||
|
baseline = medians[sink]["fprintf-default"]
|
||||||
|
for data in medians[sink].values():
|
||||||
|
data["wallReductionFractionVsDefault"] = 1 - data["wallSeconds"] / baseline["wallSeconds"]
|
||||||
|
data["byteReductionFractionVsDefault"] = 1 - data["encodedBytes"] / baseline["encodedBytes"]
|
||||||
|
write_json(output / "summary.json", {"prepared": prepared, "runs": rows, "medians": medians, "allRealFileBinary64Parity": True, "limitation": "This replay isolates formatting and ordinary file writes on preloaded contiguous doubles. It is not an end-to-end native/application speedup and excludes projection, strided output reads and durable storage flush. /dev/null metrics, if present, are separate sink-only observations."})
|
||||||
|
print(f"Summary: {output / 'summary.json'}", flush=True)
|
||||||
|
|
||||||
|
|
||||||
|
def main() -> None:
|
||||||
|
parser = argparse.ArgumentParser(description=__doc__)
|
||||||
|
parser.add_argument("--result-json", required=True, type=Path)
|
||||||
|
parser.add_argument("--output-dir", type=Path, default=ROOT / "test/c-result-encoding-20260911")
|
||||||
|
parser.add_argument("--ryu-root", type=Path)
|
||||||
|
parser.add_argument("--run", action="store_true")
|
||||||
|
parser.add_argument("--dev-null", action="store_true")
|
||||||
|
parser.add_argument("--warmups", type=int, default=1)
|
||||||
|
parser.add_argument("--repeats", type=int, default=3)
|
||||||
|
args = parser.parse_args()
|
||||||
|
if args.warmups < 0 or args.repeats < 1:
|
||||||
|
parser.error("warmups must be nonnegative and repeats positive")
|
||||||
|
prepared = prepare(args)
|
||||||
|
print(f"Prepared {prepared['valueCount']} binary64 values and {len(prepared['variants'])} variants: {args.output_dir.resolve()}", flush=True)
|
||||||
|
if args.run:
|
||||||
|
execute(args, prepared)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
@@ -0,0 +1,967 @@
|
|||||||
|
// Real production-page profiling. No route mocks, response cloning, or duplicate body parsing.
|
||||||
|
// All stage timestamps use the active document's performance.now(). A reload starts a new axis.
|
||||||
|
import { chromium } from '../../frontend/node_modules/playwright/index.mjs';
|
||||||
|
import fs from 'node:fs/promises';
|
||||||
|
import path from 'node:path';
|
||||||
|
import assert from 'node:assert/strict';
|
||||||
|
import { createHash } from 'node:crypto';
|
||||||
|
|
||||||
|
const usage = `node tests/manual/browser_stage_profile.mjs --output DIR [--input tests/data/test-mql-8-corrected.json] [--url http://127.0.0.1:8011] [--runs 3 (0 for one smoke run)] [--mode both|profiled|control] [--deep] [--cpu-interval-us 1000] [--source-map-dir DIR] [--check]
|
||||||
|
Offline only: node tests/manual/browser_stage_profile.mjs --summarize-cpu-only EXISTING_DIRECTORY [--source-map-dir DIR] [--check]
|
||||||
|
Each mode runs one warmup followed by RUNS measured runs, sequentially. --check validates inputs without launching a browser.
|
||||||
|
Optional --deep (alias --cpu-profile) records renderer-main-thread .cpuprofile diagnostics separately from ordinary endpoint timing.
|
||||||
|
--source-map-dir accepts an offline hidden-source-map build only when its generated JS bytes exactly match served assets.
|
||||||
|
Run with the repository Node 24 and Chromium runtime library environment. No application code is modified.`;
|
||||||
|
const args = process.argv.slice(2);
|
||||||
|
if (args.includes('--help')) { console.log(usage); process.exit(0); }
|
||||||
|
const options = { input: 'tests/data/test-mql-8-corrected.json', url: 'http://127.0.0.1:8011', runs: '3', mode: 'both', cpuIntervalUs: '1000' };
|
||||||
|
for (let i = 0; i < args.length; i++) {
|
||||||
|
if (['--deep', '--cpu-profile'].includes(args[i])) { options.deep = true; continue; }
|
||||||
|
if (args[i] === '--check') { options.check = true; continue; }
|
||||||
|
const cliKey = args[i].replace(/^--/, '');
|
||||||
|
const key = ({ 'cpu-interval-us': 'cpuIntervalUs', 'source-map-dir': 'sourceMapDir', 'summarize-cpu-only': 'summarizeCpuOnly' })[cliKey] ?? cliKey;
|
||||||
|
if (!['input', 'output', 'url', 'runs', 'mode', 'cpuIntervalUs', 'sourceMapDir', 'summarizeCpuOnly'].includes(key) || !args[i + 1]) throw new Error(usage);
|
||||||
|
options[key] = args[++i];
|
||||||
|
}
|
||||||
|
if (!['both', 'profiled', 'control'].includes(options.mode) || !/^(0|[1-9]\d*)$/.test(options.runs)) throw new Error(usage);
|
||||||
|
if (!/^\d+$/.test(options.cpuIntervalUs) || Number(options.cpuIntervalUs) < 100 || Number(options.cpuIntervalUs) > 100000) throw new Error('--cpu-interval-us must be between 100 and 100000.');
|
||||||
|
if (options.sourceMapDir && !options.deep && !options.summarizeCpuOnly) throw new Error('--source-map-dir requires --deep or --summarize-cpu-only.');
|
||||||
|
const sha = data => createHash('sha256').update(data).digest('hex');
|
||||||
|
let inputText; let project; let curveNodeId;
|
||||||
|
if (!options.summarizeCpuOnly) {
|
||||||
|
inputText = await fs.readFile(options.input, 'utf8');
|
||||||
|
project = JSON.parse(inputText);
|
||||||
|
assert.equal(project.projectSchemaVersion, 1);
|
||||||
|
assert.ok(project.nodes.length && project.edges.length);
|
||||||
|
curveNodeId = project.nodes.find(n => n.id === 'amesim_pnl0002_10')?.id
|
||||||
|
?? project.nodes.find(n => n.data.modelType === 'amesim_pnl0002')?.id;
|
||||||
|
assert.ok(curveNodeId, 'A PNL0002 temperature component is required.');
|
||||||
|
if (!options.check) {
|
||||||
|
if (!options.output) throw new Error(usage);
|
||||||
|
await fs.mkdir(options.output, { recursive: true });
|
||||||
|
await fs.writeFile(path.join(options.output, 'input.json'), inputText);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Serialized by Playwright. Only the profiled mode wraps production APIs; control observes
|
||||||
|
// clicks and DOM readiness, with storage-pointer polling at 16 ms resolution.
|
||||||
|
function installStageObserver({ profiled }) {
|
||||||
|
const storageKey = 'system-simulation-flow:latest-result';
|
||||||
|
const nativeParse = JSON.parse;
|
||||||
|
const data = { profiled, timeOrigin: performance.timeOrigin, marks: {}, requests: [], reads: [], parses: [],
|
||||||
|
decodes: [], transactions: [], workers: [], downloads: [], longTasks: [], streamActive: false, activeRun: false };
|
||||||
|
window.__stageProfile = data;
|
||||||
|
const mark = (name, value = performance.now()) => {
|
||||||
|
if (data.marks[name] === undefined) {
|
||||||
|
data.marks[name] = value;
|
||||||
|
performance.mark(`stage:${name}`, { startTime: value });
|
||||||
|
}
|
||||||
|
return value;
|
||||||
|
};
|
||||||
|
data.mark = mark;
|
||||||
|
data.reset = () => {
|
||||||
|
Object.assign(data, { marks: {}, requests: [], reads: [], parses: [], decodes: [], transactions: [], workers: [], runFailure: undefined,
|
||||||
|
downloads: [], longTasks: [], streamActive: false, activeRun: true });
|
||||||
|
performance.clearMarks();
|
||||||
|
};
|
||||||
|
data.armClick = (name, selector) => {
|
||||||
|
const listener = event => {
|
||||||
|
if (event.target instanceof Element && event.target.closest(selector)) {
|
||||||
|
mark(name);
|
||||||
|
document.removeEventListener('click', listener, true);
|
||||||
|
}
|
||||||
|
};
|
||||||
|
document.addEventListener('click', listener, true);
|
||||||
|
};
|
||||||
|
data.watchRunReady = () => {
|
||||||
|
const previous = sessionStorage.getItem(storageKey);
|
||||||
|
let observedBusy = false;
|
||||||
|
const observer = new MutationObserver(check);
|
||||||
|
observer.observe(document.documentElement, { subtree: true, childList: true, attributes: true, characterData: true });
|
||||||
|
function check() {
|
||||||
|
if (data.marks.runClick === undefined) return;
|
||||||
|
const failure = document.querySelector('.simulation-console-dock-progress.error, .simulation-console-progress.error');
|
||||||
|
if (failure) { data.runFailure = failure.textContent; mark('runFailure'); observer.disconnect(); return; }
|
||||||
|
const button = document.querySelector('button[aria-label="运行仿真"]');
|
||||||
|
observedBusy ||= Boolean(button?.disabled || document.querySelector('.simulation-console-dock-progress.running, .simulation-console-progress.running'));
|
||||||
|
const success = document.querySelector('.simulation-console-dock-progress.success, .simulation-console-progress.success');
|
||||||
|
if (observedBusy && button && !button.disabled && success?.textContent.includes('仿真完成')) {
|
||||||
|
mark('resultReadyDom');
|
||||||
|
data.streamActive = false;
|
||||||
|
requestAnimationFrame(() => requestAnimationFrame(() => mark('resultReadyPaintOpportunity')));
|
||||||
|
observer.disconnect();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
const poll = () => {
|
||||||
|
const raw = sessionStorage.getItem(storageKey);
|
||||||
|
if (raw && raw !== previous && nativeParse(raw).storage === 'indexeddb') {
|
||||||
|
mark('indexedDbPointerObserved');
|
||||||
|
} else setTimeout(poll, 16);
|
||||||
|
};
|
||||||
|
poll();
|
||||||
|
};
|
||||||
|
// DOM quiet is an operational threshold, not measured GPU work. No SVG serialization.
|
||||||
|
data.watchDom = (name, selector, quietMs = 120) => {
|
||||||
|
let target;
|
||||||
|
let quietTimer;
|
||||||
|
let quietObserver;
|
||||||
|
const foundObserver = new MutationObserver(find);
|
||||||
|
function find() {
|
||||||
|
const next = document.querySelector(selector);
|
||||||
|
if (!next || !next.getClientRects().length) return;
|
||||||
|
target = next;
|
||||||
|
mark(`${name}Dom`);
|
||||||
|
foundObserver.disconnect();
|
||||||
|
requestAnimationFrame(() => requestAnimationFrame(() => mark(`${name}PaintOpportunity`)));
|
||||||
|
const settle = () => {
|
||||||
|
clearTimeout(quietTimer);
|
||||||
|
data.marks[`${name}LastMutation`] = performance.now();
|
||||||
|
quietTimer = setTimeout(() => requestAnimationFrame(() => requestAnimationFrame(() => {
|
||||||
|
// A later mutation cancels the pending quiet interval as well as its timer.
|
||||||
|
if (performance.now() - data.marks[`${name}LastMutation`] < quietMs) return;
|
||||||
|
mark(`${name}Stable`);
|
||||||
|
quietObserver.disconnect();
|
||||||
|
})), quietMs);
|
||||||
|
};
|
||||||
|
quietObserver = new MutationObserver(settle);
|
||||||
|
quietObserver.observe(target, { subtree: true, childList: true, attributes: true, characterData: true });
|
||||||
|
settle();
|
||||||
|
}
|
||||||
|
foundObserver.observe(document, { subtree: true, childList: true, attributes: true });
|
||||||
|
find();
|
||||||
|
};
|
||||||
|
data.watchImport = expectedName => {
|
||||||
|
const previous = [...document.querySelectorAll('[data-entry-id]')].at(-1)?.getAttribute('data-entry-id');
|
||||||
|
const fileInput = document.querySelector('input[type="file"][accept*=".json"]');
|
||||||
|
fileInput.addEventListener('change', () => mark('importChange'), { once: true, capture: true });
|
||||||
|
const observer = new MutationObserver(() => {
|
||||||
|
if (data.marks.importChange === undefined) return;
|
||||||
|
const entry = [...document.querySelectorAll('[data-entry-id]')].at(-1);
|
||||||
|
if (entry?.getAttribute('data-entry-id') !== previous && entry?.textContent.includes('已导入工程') &&
|
||||||
|
[...document.querySelectorAll('input')].some(input => input.value === expectedName)) {
|
||||||
|
mark('importReadyDom');
|
||||||
|
requestAnimationFrame(() => requestAnimationFrame(() => mark('importReadyPaintOpportunity')));
|
||||||
|
observer.disconnect();
|
||||||
|
}
|
||||||
|
});
|
||||||
|
observer.observe(document, { subtree: true, childList: true, characterData: true });
|
||||||
|
};
|
||||||
|
data.snapshot = () => ({ ...data, mark: undefined, reset: undefined, armClick: undefined, watchRunReady: undefined,
|
||||||
|
watchDom: undefined, watchImport: undefined, snapshot: undefined, resources: performance.getEntriesByType('resource')
|
||||||
|
.filter(e => /simulate-stream|simulation-results\/csv/.test(e.name))
|
||||||
|
.map(e => ({ name: e.name, startTime: e.startTime, requestStart: e.requestStart, responseStart: e.responseStart,
|
||||||
|
responseEnd: e.responseEnd, duration: e.duration, transferSize: e.transferSize,
|
||||||
|
encodedBodySize: e.encodedBodySize, decodedBodySize: e.decodedBodySize })) });
|
||||||
|
if (location.hash === '#/results' && sessionStorage.getItem(storageKey)) {
|
||||||
|
data.watchDom('restoredResults', '.results-shell .results-system-panel');
|
||||||
|
}
|
||||||
|
if (!profiled) return;
|
||||||
|
|
||||||
|
const bodies = new WeakMap();
|
||||||
|
const originalFetch = window.fetch;
|
||||||
|
window.fetch = function (...args) {
|
||||||
|
const url = typeof args[0] === 'string' ? args[0] : args[0] instanceof Request ? args[0].url : String(args[0]);
|
||||||
|
const kind = url.includes('/api/system-xml/simulate-stream') ? 'simulation'
|
||||||
|
: url.includes('/api/simulation-results/csv') ? 'csv' : null;
|
||||||
|
if (!kind) return Reflect.apply(originalFetch, this, args);
|
||||||
|
const row = { kind, fetchStart: performance.now(), simulationId: new Headers(args[1]?.headers ?? (args[0] instanceof Request ? args[0].headers : undefined)).get('X-Simulation-Id'), requestStringLength: typeof args[1]?.body === 'string' ? args[1].body.length : null };
|
||||||
|
data.requests.push(row);
|
||||||
|
if (kind === 'simulation') { mark('fetchStart', row.fetchStart); data.streamActive = true; }
|
||||||
|
return Reflect.apply(originalFetch, this, args).then(response => {
|
||||||
|
row.headers = performance.now(); row.status = response.status;
|
||||||
|
if (kind === 'simulation') mark('headers', row.headers);
|
||||||
|
if (response.body) bodies.set(response.body, row);
|
||||||
|
return response;
|
||||||
|
});
|
||||||
|
};
|
||||||
|
const originalGetReader = ReadableStream.prototype.getReader;
|
||||||
|
ReadableStream.prototype.getReader = function (...args) {
|
||||||
|
const reader = Reflect.apply(originalGetReader, this, args);
|
||||||
|
const request = bodies.get(this);
|
||||||
|
if (!request) return reader;
|
||||||
|
const originalRead = reader.read;
|
||||||
|
reader.read = function (...readArgs) {
|
||||||
|
const row = { kind: request.kind, start: performance.now() };
|
||||||
|
return Reflect.apply(originalRead, this, readArgs).then(value => {
|
||||||
|
row.end = performance.now(); row.bytes = value.value?.byteLength ?? 0; row.done = value.done;
|
||||||
|
data.reads.push(row);
|
||||||
|
if (request.kind === 'simulation') {
|
||||||
|
if (row.bytes) { mark('firstChunk', row.end); data.marks.lastChunk = row.end; }
|
||||||
|
if (row.done) mark('streamEof', row.end);
|
||||||
|
}
|
||||||
|
return value;
|
||||||
|
});
|
||||||
|
};
|
||||||
|
return reader;
|
||||||
|
};
|
||||||
|
JSON.parse = function (...args) {
|
||||||
|
if (!data.streamActive) return Reflect.apply(nativeParse, this, args);
|
||||||
|
const start = performance.now();
|
||||||
|
const parsed = Reflect.apply(nativeParse, this, args);
|
||||||
|
const end = performance.now();
|
||||||
|
if (parsed && ['progress', 'result', 'error'].includes(parsed.event)) {
|
||||||
|
data.parses.push({ event: parsed.event, phase: parsed.phase, start, end, characters: typeof args[0] === 'string' ? args[0].length : null });
|
||||||
|
if (parsed.event === 'result') {
|
||||||
|
mark('resultParseStart', start); mark('resultParseEnd', end);
|
||||||
|
// This microtask is only a checkpoint after the current consumer continuation,
|
||||||
|
// not a claim that all EOF/finally/publish/React work has completed.
|
||||||
|
queueMicrotask(() => mark('resultConsumerMicrotaskCheckpoint'));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return parsed;
|
||||||
|
};
|
||||||
|
const originalDecode = TextDecoder.prototype.decode;
|
||||||
|
TextDecoder.prototype.decode = function (...args) {
|
||||||
|
if (!data.streamActive) return Reflect.apply(originalDecode, this, args);
|
||||||
|
const start = performance.now();
|
||||||
|
const value = Reflect.apply(originalDecode, this, args);
|
||||||
|
data.decodes.push({ start, end: performance.now(), bytes: args[0]?.byteLength ?? 0 });
|
||||||
|
return value;
|
||||||
|
};
|
||||||
|
const originalTransaction = IDBDatabase.prototype.transaction;
|
||||||
|
IDBDatabase.prototype.transaction = function (...args) {
|
||||||
|
const transaction = Reflect.apply(originalTransaction, this, args);
|
||||||
|
if (this.name === 'system-simulation-results') {
|
||||||
|
const row = { start: performance.now(), mode: transaction.mode, stores: Array.from(transaction.objectStoreNames) };
|
||||||
|
data.transactions.push(row);
|
||||||
|
transaction.addEventListener('complete', () => { row.end = performance.now(); row.outcome = 'complete'; });
|
||||||
|
transaction.addEventListener('abort', () => { row.end = performance.now(); row.outcome = 'abort'; });
|
||||||
|
}
|
||||||
|
return transaction;
|
||||||
|
};
|
||||||
|
const originalSetItem = Storage.prototype.setItem;
|
||||||
|
Storage.prototype.setItem = function (...args) {
|
||||||
|
const value = Reflect.apply(originalSetItem, this, args);
|
||||||
|
if (this === sessionStorage && args[0] === storageKey && data.activeRun) mark('indexedDbCommittedPointer');
|
||||||
|
return value;
|
||||||
|
};
|
||||||
|
const NativeWorker = window.Worker;
|
||||||
|
window.Worker = new Proxy(NativeWorker, {
|
||||||
|
construct(target, args, newTarget) {
|
||||||
|
const url = String(args[0]);
|
||||||
|
const isCsv = /resultCsv/i.test(url);
|
||||||
|
const start = performance.now();
|
||||||
|
const worker = Reflect.construct(target, args, newTarget);
|
||||||
|
if (!isCsv) return worker;
|
||||||
|
const row = { url, constructStart: start, constructEnd: performance.now(), posts: [] };
|
||||||
|
data.workers.push(row);
|
||||||
|
worker.addEventListener('message', event => {
|
||||||
|
const message = event.data;
|
||||||
|
if (message?.type === 'complete') { row.completeReceived = performance.now(); row.blobBytes = message.blob?.size ?? null; }
|
||||||
|
if (message?.type === 'error') { row.errorReceived = performance.now(); row.error = message.message; }
|
||||||
|
});
|
||||||
|
const originalPost = worker.postMessage;
|
||||||
|
worker.postMessage = function (...postArgs) {
|
||||||
|
const message = postArgs[0];
|
||||||
|
// Read scalar metadata before transfer detaches the original buffer. Never
|
||||||
|
// inspect/copy values, Blob contents, or inject code into the worker.
|
||||||
|
const post = { type: message?.type, start: performance.now(), offset: message?.offset,
|
||||||
|
bytes: message?.values?.byteLength ?? 0 };
|
||||||
|
const result = Reflect.apply(originalPost, this, postArgs);
|
||||||
|
post.end = performance.now();
|
||||||
|
row.posts.push(post);
|
||||||
|
return result;
|
||||||
|
};
|
||||||
|
return worker;
|
||||||
|
},
|
||||||
|
});
|
||||||
|
const originalAnchorClick = HTMLAnchorElement.prototype.click;
|
||||||
|
HTMLAnchorElement.prototype.click = function (...args) {
|
||||||
|
if (this.download) data.downloads.push({ name: this.download, anchorClick: performance.now(), blobUrl: this.href.startsWith('blob:') });
|
||||||
|
return Reflect.apply(originalAnchorClick, this, args);
|
||||||
|
};
|
||||||
|
if (PerformanceObserver.supportedEntryTypes.includes('longtask')) {
|
||||||
|
new PerformanceObserver(list => {
|
||||||
|
for (const e of list.getEntries()) if (data.activeRun) data.longTasks.push({ start: e.startTime, duration: e.duration, name: e.name });
|
||||||
|
}).observe({ type: 'longtask', buffered: false });
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Optional sampling diagnostics. These deliberately remain independent of production
|
||||||
|
// code and existing stage marks. Analysis/source-map loading runs after all timings.
|
||||||
|
async function calibrateCpuClock(session) {
|
||||||
|
const samples = [];
|
||||||
|
for (let i = 0; i < 3; i++) {
|
||||||
|
const read = async () => (await session.send('Runtime.evaluate', {
|
||||||
|
expression: '({now: performance.now(), timeOrigin: performance.timeOrigin})', returnByValue: true,
|
||||||
|
})).result.value;
|
||||||
|
const before = await read();
|
||||||
|
const metrics = await session.send('Performance.getMetrics');
|
||||||
|
const after = await read();
|
||||||
|
const timestamp = metrics.metrics.find(metric => metric.name === 'Timestamp')?.value;
|
||||||
|
if (timestamp === undefined || before.timeOrigin !== after.timeOrigin) continue;
|
||||||
|
samples.push({ timeOrigin: before.timeOrigin, pageBeforeMs: before.now, pageAfterMs: after.now,
|
||||||
|
cdpTimestampMs: timestamp * 1000, offsetMs: timestamp * 1000 - (before.now + after.now) / 2,
|
||||||
|
uncertaintyMs: (after.now - before.now) / 2 });
|
||||||
|
}
|
||||||
|
if (!samples.length) throw new Error('Unable to calibrate CDP sampling against the page clock.');
|
||||||
|
samples.sort((a, b) => a.uncertaintyMs - b.uncertaintyMs);
|
||||||
|
return { chosen: samples[0], probes: samples };
|
||||||
|
}
|
||||||
|
|
||||||
|
async function createCpuRecorder(context, page, intervalUs) {
|
||||||
|
const session = await context.newCDPSession(page);
|
||||||
|
await session.send('Performance.enable', { timeDomain: 'timeTicks' });
|
||||||
|
await session.send('Profiler.enable');
|
||||||
|
await session.send('Profiler.setSamplingInterval', { interval: intervalUs });
|
||||||
|
let startCalibration;
|
||||||
|
let running = false;
|
||||||
|
return {
|
||||||
|
async start() {
|
||||||
|
if (running) throw new Error('CPU profiler already running.');
|
||||||
|
startCalibration = await calibrateCpuClock(session);
|
||||||
|
await session.send('Profiler.start');
|
||||||
|
running = true;
|
||||||
|
},
|
||||||
|
async stop() {
|
||||||
|
const { profile } = await session.send('Profiler.stop');
|
||||||
|
running = false;
|
||||||
|
const endCalibration = await calibrateCpuClock(session);
|
||||||
|
return { profile, calibration: { start: startCalibration, end: endCalibration }, intervalUs };
|
||||||
|
},
|
||||||
|
async close() {
|
||||||
|
if (running) await session.send('Profiler.stop').catch(() => {});
|
||||||
|
await session.detach().catch(() => {});
|
||||||
|
},
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
const cpuFunctionCategories = {
|
||||||
|
buildSystemXml: 'xml_generation', validateModel: 'model_validation', checkModel: 'model_validation',
|
||||||
|
componentParameterValidationMessage: 'model_validation', projectExecutionContractIssues: 'model_validation',
|
||||||
|
resolveSimulationConfig: 'model_validation', modelValidationSignature: 'model_validation',
|
||||||
|
projectConnectionMetadata: 'model_contract_and_endpoints', buildCurrentProject: 'project_snapshot_copy',
|
||||||
|
buildProjectPayload: 'project_snapshot_copy', cloneValue: 'project_snapshot_copy',
|
||||||
|
publishSimulationResult: 'result_publication', normalizeSimulationProgressEvent: 'progress_normalization',
|
||||||
|
storeResultSnapshot: 'persistence_pack_and_save', writeSnapshot: 'indexeddb_request_submission',
|
||||||
|
restorePacked: 'persistence_unpack', loadStoredResultSnapshot: 'persistence_restore',
|
||||||
|
openDatabase: 'indexeddb_open', deleteCache: 'persistence_cleanup',
|
||||||
|
};
|
||||||
|
|
||||||
|
const cpuSourceRanges = new Map();
|
||||||
|
function originalFunctionAt(content, line) {
|
||||||
|
if (!content) return null;
|
||||||
|
let ranges = cpuSourceRanges.get(content);
|
||||||
|
if (!ranges) {
|
||||||
|
const lines = content.split('\n');
|
||||||
|
const declarations = [];
|
||||||
|
for (let i = 0; i < lines.length; i++) {
|
||||||
|
const match = lines[i].match(/^(\s*)(?:(?:export\s+)?(?:async\s+)?function\s+|const\s+)([A-Za-z_$][\w$]*)/);
|
||||||
|
if (match) declarations.push({ name: match[2], indent: match[1].length, line: i + 1 });
|
||||||
|
}
|
||||||
|
ranges = [];
|
||||||
|
for (const [index, declaration] of declarations.entries()) {
|
||||||
|
if (!(declaration.name in cpuFunctionCategories)) continue;
|
||||||
|
const next = declarations.slice(index + 1).find(peer => peer.indent <= declaration.indent);
|
||||||
|
ranges.push({ name: declaration.name, startLine: declaration.line, endLineExclusive: next?.line ?? lines.length + 1 });
|
||||||
|
}
|
||||||
|
cpuSourceRanges.set(content, ranges);
|
||||||
|
}
|
||||||
|
// sourcesContent comes from the byte-verified bundle. Declaration ranges are
|
||||||
|
// cached once per source, rather than rescanning App.tsx for every sampled frame.
|
||||||
|
return ranges.findLast(range => line >= range.startLine && line < range.endLineExclusive) ?? null;
|
||||||
|
}
|
||||||
|
|
||||||
|
async function loadVerifiedCpuMaps(sourceMapDirectory, servedAssets) {
|
||||||
|
const consumers = new Map();
|
||||||
|
const evidence = [];
|
||||||
|
if (!sourceMapDirectory) return { consumers, evidence };
|
||||||
|
const module = await import('../../frontend/node_modules/source-map-js/source-map.js');
|
||||||
|
const { SourceMapConsumer } = module.default ?? module;
|
||||||
|
const root = path.resolve(sourceMapDirectory);
|
||||||
|
for (const asset of servedAssets.filter(asset => new URL(asset.url).pathname.endsWith('.js'))) {
|
||||||
|
const pathname = decodeURIComponent(new URL(asset.url).pathname);
|
||||||
|
const candidates = [path.resolve(root, `.${pathname}`), path.join(root, path.basename(pathname))];
|
||||||
|
let generated = candidates[0];
|
||||||
|
for (const candidate of candidates) {
|
||||||
|
if (!candidate.startsWith(`${root}${path.sep}`)) continue;
|
||||||
|
if (await fs.stat(candidate).then(stat => stat.isFile()).catch(() => false)) { generated = candidate; break; }
|
||||||
|
}
|
||||||
|
if (!generated.startsWith(`${root}${path.sep}`)) continue;
|
||||||
|
const mapPath = `${generated}.map`;
|
||||||
|
try {
|
||||||
|
const generatedBytes = await fs.readFile(generated);
|
||||||
|
const generatedSha256 = sha(generatedBytes);
|
||||||
|
if (generatedSha256 !== asset.sha256) {
|
||||||
|
evidence.push({ url: asset.url, generated, status: 'rejected-generated-bytes-differ', generatedSha256, servedSha256: asset.sha256 });
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
const mapBytes = await fs.readFile(mapPath);
|
||||||
|
const consumer = new SourceMapConsumer(JSON.parse(mapBytes));
|
||||||
|
consumers.set(asset.url, consumer);
|
||||||
|
evidence.push({ url: asset.url, generated, mapPath, status: 'verified', generatedSha256, mapSha256: sha(mapBytes) });
|
||||||
|
} catch (error) {
|
||||||
|
evidence.push({ url: asset.url, generated, mapPath, status: 'unavailable', reason: String(error) });
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return { consumers, evidence };
|
||||||
|
}
|
||||||
|
|
||||||
|
function cpuFrameInfo(frame, maps, cache) {
|
||||||
|
const key = `${frame.scriptId}:${frame.url}:${frame.lineNumber}:${frame.columnNumber}:${frame.functionName}`;
|
||||||
|
if (cache.has(key)) return cache.get(key);
|
||||||
|
let original = null;
|
||||||
|
let sourceFunction = null;
|
||||||
|
const consumer = maps.get(frame.url);
|
||||||
|
if (consumer && frame.lineNumber >= 0 && frame.columnNumber >= 0) {
|
||||||
|
const position = consumer.originalPositionFor({ line: frame.lineNumber + 1, column: frame.columnNumber });
|
||||||
|
if (position.source && position.line) {
|
||||||
|
original = position;
|
||||||
|
sourceFunction = originalFunctionAt(consumer.sourceContentFor(position.source, true), position.line);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
const name = sourceFunction?.name ?? original?.name ?? frame.functionName;
|
||||||
|
const source = original?.source ?? frame.url ?? '';
|
||||||
|
let category = 'unclassified';
|
||||||
|
let basis = 'unclassified';
|
||||||
|
if (frame.functionName === '(idle)') { category = 'idle'; basis = 'v8-system-frame'; }
|
||||||
|
else if (frame.functionName === '(garbage collector)') { category = 'garbage_collection'; basis = 'v8-system-frame'; }
|
||||||
|
else if (['(program)', '(root)'].includes(frame.functionName)) { category = 'unclassified_program'; basis = 'v8-system-frame'; }
|
||||||
|
else if (name in cpuFunctionCategories) { category = cpuFunctionCategories[name]; basis = sourceFunction ? 'verified-map-declaration-range' : original ? 'verified-map-name' : 'runtime-function-name'; }
|
||||||
|
else if (/JSON\.parse/.test(name)) { category = 'json_parse'; basis = 'runtime-function-name'; }
|
||||||
|
else if (/TextDecoder.*decode/.test(name)) { category = 'utf8_decode'; basis = 'runtime-function-name'; }
|
||||||
|
else if (/XMLSerializer|serializeToString/.test(name)) { category = 'xml_native_serialization'; basis = 'runtime-function-name'; }
|
||||||
|
else if (/structuredClone/.test(name)) { category = 'structured_clone'; basis = 'runtime-function-name'; }
|
||||||
|
else if (/react-dom|react\/cjs|scheduler\/cjs/.test(source)) { category = 'react_runtime'; basis = original ? 'verified-map-source' : 'runtime-source-url'; }
|
||||||
|
else if (/@xyflow/.test(source)) { category = 'diagram_reactflow'; basis = original ? 'verified-map-source' : 'runtime-source-url'; }
|
||||||
|
else if (/ndjsonStream\./.test(source)) { category = 'ndjson_scan_join_dispatch'; basis = 'verified-map-source'; }
|
||||||
|
else if (/resultPersistence\./.test(source)) { category = 'persistence_other'; basis = 'verified-map-source'; }
|
||||||
|
else if (/resultCsvExport\./.test(source)) { category = 'csv_prepare_and_transfer'; basis = 'verified-map-source'; }
|
||||||
|
else if (/chartData\.|resultEventSeries\./.test(source)) { category = 'chart_data_preparation'; basis = 'verified-map-source'; }
|
||||||
|
else if (/SimulationResultsView\./.test(source)) { category = 'results_view_preparation'; basis = 'verified-map-source'; }
|
||||||
|
else if (/componentSymbols|edgeRouting|ContactAwareEdge/.test(source)) { category = 'diagram_geometry'; basis = 'verified-map-source'; }
|
||||||
|
else if (/App\.tsx/.test(source)) { category = 'application_other'; basis = 'verified-map-source'; }
|
||||||
|
else if (/__playwright|utilityScript|evaluate@/.test(source + name)) { category = 'measurement_or_automation'; basis = 'runtime-frame'; }
|
||||||
|
const result = { key, functionName: frame.functionName, hasScript: Boolean(frame.scriptId && frame.scriptId !== '0'), generated: { url: frame.url, line: frame.lineNumber + 1, column: frame.columnNumber },
|
||||||
|
original, sourceFunction, category, classificationBasis: basis };
|
||||||
|
cache.set(key, result);
|
||||||
|
return result;
|
||||||
|
}
|
||||||
|
|
||||||
|
function cpuWindows(trace, kind) {
|
||||||
|
const m = trace.marks;
|
||||||
|
const endCommit = m.indexedDbCommittedPointer ?? m.indexedDbPointerObserved;
|
||||||
|
const windows = kind === 'restore' ? [
|
||||||
|
['restore_navigation_to_dom', 0, m.restoredResultsDom],
|
||||||
|
['restore_navigation_to_paint_opportunity', 0, m.restoredResultsPaintOpportunity],
|
||||||
|
] : [
|
||||||
|
['import', m.importChange, m.importReadyPaintOpportunity],
|
||||||
|
['pre_submit', m.runClick, m.fetchStart],
|
||||||
|
['response_before_result_parse', m.headers, m.resultParseStart],
|
||||||
|
['result_json_parse', m.resultParseStart, m.resultParseEnd],
|
||||||
|
['result_publish_to_dom', m.resultParseEnd, m.resultReadyDom],
|
||||||
|
['result_ready_to_saved_pointer', m.resultReadyDom, endCommit],
|
||||||
|
['result_parse_through_saved_pointer', m.resultParseStart, endCommit],
|
||||||
|
['result_parsed_to_saved_pointer', m.resultParseEnd, endCommit],
|
||||||
|
['result_tab_preparation', m.resultTabClick, m.resultsPaintOpportunity],
|
||||||
|
['temperature_chart_preparation', m.temperatureClick, m.curvePaintOpportunity],
|
||||||
|
['csv_export_through_download_saved', m.csvClick, m.csvDownloadSaved],
|
||||||
|
['result_export_through_download_saved', m.resultFileClick, m.resultFileDownloadSaved],
|
||||||
|
['whole_run_click_through_download_saved', m.runClick, m.resultFileDownloadSaved],
|
||||||
|
];
|
||||||
|
return windows.filter(([, start, end]) => Number.isFinite(start) && Number.isFinite(end) && end >= start)
|
||||||
|
.map(([name, startMs, endMs]) => ({ name, startMs, endMs, wallMs: endMs - startMs }));
|
||||||
|
}
|
||||||
|
|
||||||
|
function cpuExecutionState(stack) {
|
||||||
|
const leaf = stack[0];
|
||||||
|
if (!leaf) return 'unknown_runtime';
|
||||||
|
if (['idle', 'garbage_collection', 'unclassified_program'].includes(leaf.category)) return leaf.category;
|
||||||
|
if (leaf.hasScript) return 'active_js';
|
||||||
|
// Native call frames (for example IDBObjectStore.put or Blob) may be sampled
|
||||||
|
// under a real JS caller. A bare runtime/native frame has no such evidence.
|
||||||
|
return stack.some(frame => frame.hasScript) ? 'active_native_call' : 'unknown_runtime';
|
||||||
|
}
|
||||||
|
|
||||||
|
function summarizeCpu(record, maps) {
|
||||||
|
const { profile, trace, intervalUs, calibration, kind } = record;
|
||||||
|
const anchors = [calibration.start.chosen, calibration.end.chosen].filter(anchor => anchor.timeOrigin === trace.timeOrigin);
|
||||||
|
if (!anchors.length) throw new Error('CPU profile calibration does not match its document timeOrigin.');
|
||||||
|
const anchor = anchors.reduce((best, candidate) => candidate.uncertaintyMs < best.uncertaintyMs ? candidate : best);
|
||||||
|
const offsetSpreadMs = Math.max(...anchors.map(a => a.offsetMs)) - Math.min(...anchors.map(a => a.offsetMs));
|
||||||
|
const frameCache = new Map();
|
||||||
|
const nodes = new Map(profile.nodes.map(node => [node.id, { ...node, info: cpuFrameInfo(node.callFrame, maps, frameCache) }]));
|
||||||
|
const parent = new Map();
|
||||||
|
for (const node of profile.nodes) for (const child of node.children ?? []) parent.set(child, node.id);
|
||||||
|
const stacks = new Map();
|
||||||
|
for (const node of profile.nodes) {
|
||||||
|
const stack = []; let id = node.id;
|
||||||
|
while (nodes.has(id) && stack.length < 256) { stack.push(nodes.get(id).info); id = parent.get(id); }
|
||||||
|
stacks.set(node.id, stack);
|
||||||
|
}
|
||||||
|
let clockMs = profile.startTime / 1000 - anchor.offsetMs;
|
||||||
|
const samples = (profile.samples ?? []).map((id, index) => {
|
||||||
|
const deltaMs = (profile.timeDeltas?.[index] ?? intervalUs) / 1000;
|
||||||
|
const startMs = clockMs; clockMs += deltaMs;
|
||||||
|
// Sampling gaps can include OS descheduling; never credit a long unsampled
|
||||||
|
// gap wholesale to the currently sampled JavaScript function.
|
||||||
|
return { id, startMs, endMs: clockMs, representedMs: Math.min(deltaMs, 2 * intervalUs / 1000) };
|
||||||
|
});
|
||||||
|
const windows = cpuWindows(trace, kind).map(window => {
|
||||||
|
const categories = new Map(); const functions = new Map();
|
||||||
|
const executionSampleMs = { active_js: 0, active_native_call: 0, idle: 0,
|
||||||
|
garbage_collection: 0, unclassified_program: 0, unknown_runtime: 0 };
|
||||||
|
let sampleCount = 0; let representedMs = 0; let samplingGapMs = 0;
|
||||||
|
for (const sample of samples) {
|
||||||
|
const overlap = Math.max(0, Math.min(window.endMs, sample.endMs) - Math.max(window.startMs, sample.startMs));
|
||||||
|
if (!overlap) continue;
|
||||||
|
const interval = sample.endMs - sample.startMs;
|
||||||
|
const weight = interval > 0 ? overlap * sample.representedMs / interval : 0;
|
||||||
|
sampleCount++; representedMs += weight; samplingGapMs += Math.max(0, overlap - weight);
|
||||||
|
const stack = stacks.get(sample.id) ?? [];
|
||||||
|
const state = cpuExecutionState(stack);
|
||||||
|
executionSampleMs[state] += weight;
|
||||||
|
const active = state === 'active_js' || state === 'active_native_call';
|
||||||
|
// Never walk up to V8 (root)/(program) to classify unknown JavaScript.
|
||||||
|
// GC/idle/program samples also never accrue to an application caller.
|
||||||
|
const activeStack = stack.filter(frame => frame.classificationBasis !== 'v8-system-frame');
|
||||||
|
const owner = activeStack.find(frame => !['unclassified', 'application_other'].includes(frame.category)) ?? activeStack[0];
|
||||||
|
const category = active ? owner?.category ?? 'unclassified' : state;
|
||||||
|
categories.set(category, (categories.get(category) ?? 0) + weight);
|
||||||
|
if (!active) continue;
|
||||||
|
const seen = new Set();
|
||||||
|
for (const [index, frame] of activeStack.entries()) {
|
||||||
|
if (seen.has(frame.key)) continue;
|
||||||
|
seen.add(frame.key);
|
||||||
|
const entry = functions.get(frame.key) ?? { ...frame, selfActiveEstimatedMs: 0, inclusiveActiveEstimatedMs: 0 };
|
||||||
|
if (index === 0) entry.selfActiveEstimatedMs += weight;
|
||||||
|
entry.inclusiveActiveEstimatedMs += weight;
|
||||||
|
functions.set(frame.key, entry);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return { ...window, sampleCount, representedSampleMs: representedMs, executionSampleMs,
|
||||||
|
activeJsAndNativeCallSampleMs: executionSampleMs.active_js + executionSampleMs.active_native_call,
|
||||||
|
samplingGapMs, unattributedWallMs: Math.max(0, window.wallMs - representedMs - samplingGapMs),
|
||||||
|
exclusiveOwnerCategories: Object.fromEntries([...categories].sort((a, b) => b[1] - a[1])),
|
||||||
|
functions: [...functions.values()].sort((a, b) => b.inclusiveActiveEstimatedMs - a.inclusiveActiveEstimatedMs) };
|
||||||
|
});
|
||||||
|
return { schemaVersion: 2, kind, intervalUs, timeOrigin: trace.timeOrigin, profileStartUs: profile.startTime, profileEndUs: profile.endTime,
|
||||||
|
profilePageStartMs: profile.startTime / 1000 - anchor.offsetMs, profilePageEndMs: profile.endTime / 1000 - anchor.offsetMs,
|
||||||
|
clockCalibration: { anchor, offsetSpreadMs, probes: calibration }, windows,
|
||||||
|
limitations: [
|
||||||
|
'Sampling estimates renderer-main-thread execution attribution; it is not a function stopwatch or an OS thread CPU clock.',
|
||||||
|
'representedSampleMs includes idle, GC and unknown runtime samples. It is NOT an active CPU total.',
|
||||||
|
'activeJsAndNativeCallSampleMs includes sampled JavaScript and native calls with a JavaScript ancestor; bare native/runtime frames remain unknown.',
|
||||||
|
'V8 (program)/(root) samples are unclassified_program, not evidence of JavaScript or native CPU activity. Do not attribute this interval to receiving, parsing, painting or another residual category.',
|
||||||
|
'Each sample represents at most two configured sampling intervals; larger gaps are unassigned scheduling/sampling gaps.',
|
||||||
|
'Function selfActiveEstimatedMs/inclusiveActiveEstimatedMs exclude idle, GC, program and unknown-runtime samples even if they have application ancestors.',
|
||||||
|
'Active inclusive estimates include active children and cannot be added together; stage windows also overlap and cannot be added.',
|
||||||
|
'Synchronous JSON.parse/decoder durations in trace.json remain direct wrapper measurements; do not add sampling estimates to them.',
|
||||||
|
'Renderer main-thread profiling excludes CSV Worker CPU, IndexedDB background/disk work, compositor/GPU work and backend CPU.',
|
||||||
|
'React/source categories describe JavaScript preparation, not actual paint completion.',
|
||||||
|
'Without byte-verified source maps, minified function ownership stays unknown; original declaration ranges are a source attribution aid, not exact instruction boundaries.',
|
||||||
|
'Clock alignment uses CDP timeTicks and bracketing page-clock reads; uncertainty and before/after offset spread are reported.',
|
||||||
|
'Deep profiling perturbs execution; compare endpoint timings using separate runs without --deep.',
|
||||||
|
] };
|
||||||
|
}
|
||||||
|
|
||||||
|
function cpuSummaryMarkdown(summary, filename) {
|
||||||
|
const lines = [`CPU sampling diagnostic: ${filename}`, '', `Schema ${summary.schemaVersion}; interval ${summary.intervalUs} us. Main-thread samples only.`,
|
||||||
|
`Clock uncertainty: ${summary.clockCalibration.anchor.uncertaintyMs.toFixed(3)} ms; offset spread: ${summary.clockCalibration.offsetSpreadMs.toFixed(3)} ms.`, '',
|
||||||
|
'Active = JavaScript plus native calls sampled under JavaScript. Program/unknown and idle are not counted as active CPU.', '',
|
||||||
|
'| Window (overlaps allowed) | Wall ms | Active sample ms | Idle ms | GC ms | Program/unknown ms | Sampling gap ms | Leading active owner categories |',
|
||||||
|
'|---|---:|---:|---:|---:|---:|---:|---|'];
|
||||||
|
for (const window of summary.windows) {
|
||||||
|
const states = window.executionSampleMs;
|
||||||
|
const top = Object.entries(window.exclusiveOwnerCategories).filter(([name]) => !['idle', 'garbage_collection', 'unclassified_program', 'unknown_runtime'].includes(name))
|
||||||
|
.slice(0, 5).map(([name, ms]) => `${name}: ${ms.toFixed(2)}`).join('; ');
|
||||||
|
lines.push(`| ${window.name} | ${window.wallMs.toFixed(2)} | ${window.activeJsAndNativeCallSampleMs.toFixed(2)} | ${states.idle.toFixed(2)} | ${states.garbage_collection.toFixed(2)} | ${(states.unclassified_program + states.unknown_runtime).toFixed(2)} | ${window.samplingGapMs.toFixed(2)} | ${top} |`);
|
||||||
|
}
|
||||||
|
for (const window of summary.windows) {
|
||||||
|
lines.push('', `Top active sampled functions: ${window.name}`, '');
|
||||||
|
lines.push(...window.functions.filter(frame => frame.selfActiveEstimatedMs > 0).sort((a, b) => b.selfActiveEstimatedMs - a.selfActiveEstimatedMs).slice(0, 8).map(frame => {
|
||||||
|
const location = frame.original ? `${frame.original.source}:${frame.original.line}:${frame.original.column}`
|
||||||
|
: `${frame.generated.url || '(native/injected)'}:${frame.generated.line}:${frame.generated.column}`;
|
||||||
|
return `- ${frame.sourceFunction?.name ?? frame.original?.name ?? frame.functionName ?? '(anonymous)'} — active self ${frame.selfActiveEstimatedMs.toFixed(2)} ms, active inclusive ${frame.inclusiveActiveEstimatedMs.toFixed(2)} ms; ${location}; ${frame.classificationBasis}`;
|
||||||
|
}));
|
||||||
|
}
|
||||||
|
lines.push('', 'Limitations:', '', ...summary.limitations.map(text => `- ${text}`), '');
|
||||||
|
return lines.join('\n');
|
||||||
|
}
|
||||||
|
|
||||||
|
// Small synthetic sampling contract: nested GC is separate, a bare program frame
|
||||||
|
// remains unknown, minified JS cannot inherit (root), and long gaps stay unassigned.
|
||||||
|
function checkCpuSummaryContract() {
|
||||||
|
const frame = (functionName, scriptId = '0') => ({ functionName, scriptId, url: scriptId === '0' ? '' : 'fixture.js', lineNumber: scriptId === '0' ? -1 : 0, columnNumber: 0 });
|
||||||
|
const profile = { startTime: 0, endTime: 14000, nodes: [
|
||||||
|
{ id: 1, callFrame: frame('(root)'), children: [2, 3, 4, 8] },
|
||||||
|
{ id: 2, callFrame: frame('(idle)') }, { id: 3, callFrame: frame('(program)') },
|
||||||
|
{ id: 4, callFrame: frame('buildSystemXml', '1'), children: [5, 6] },
|
||||||
|
{ id: 5, callFrame: frame('(garbage collector)') },
|
||||||
|
{ id: 6, callFrame: frame('anonymousMinified', '1'), children: [7] },
|
||||||
|
{ id: 7, callFrame: frame('put') }, { id: 8, callFrame: frame('unresolvedNative') },
|
||||||
|
], samples: [2, 3, 4, 5, 6, 7, 8], timeDeltas: [1000, 1000, 1000, 1000, 1000, 1000, 8000] };
|
||||||
|
const chosen = { timeOrigin: 0, offsetMs: 0, uncertaintyMs: 0 };
|
||||||
|
const summary = summarizeCpu({ profile, kind: 'interaction', trace: { timeOrigin: 0, marks: { runClick: 0, fetchStart: 14 } },
|
||||||
|
calibration: { start: { chosen }, end: { chosen } }, intervalUs: 1000 }, new Map());
|
||||||
|
const window = summary.windows[0];
|
||||||
|
assert.deepEqual(window.executionSampleMs, { active_js: 2, active_native_call: 1, idle: 1, garbage_collection: 1, unclassified_program: 1, unknown_runtime: 2 });
|
||||||
|
assert.equal(window.activeJsAndNativeCallSampleMs, 3);
|
||||||
|
assert.equal(window.samplingGapMs, 6);
|
||||||
|
assert.equal(window.exclusiveOwnerCategories.xml_generation, 3);
|
||||||
|
assert.equal(window.functions.find(item => item.functionName === 'buildSystemXml').inclusiveActiveEstimatedMs, 3);
|
||||||
|
assert.equal(window.functions.some(item => item.classificationBasis === 'v8-system-frame'), false);
|
||||||
|
// A JS leaf directly under root must be active with unknown ownership.
|
||||||
|
const isolated = structuredClone(profile);
|
||||||
|
isolated.nodes[0].children.push(9);
|
||||||
|
isolated.nodes.push({ id: 9, callFrame: frame('minifiedUnknown', '1') });
|
||||||
|
isolated.samples = [9]; isolated.timeDeltas = [1000]; isolated.endTime = 1000;
|
||||||
|
const unknown = summarizeCpu({ profile: isolated, kind: 'interaction', trace: { timeOrigin: 0, marks: { runClick: 0, fetchStart: 1 } },
|
||||||
|
calibration: { start: { chosen }, end: { chosen } }, intervalUs: 1000 }, new Map()).windows[0];
|
||||||
|
assert.equal(unknown.exclusiveOwnerCategories.unclassified, 1);
|
||||||
|
assert.equal(unknown.executionSampleMs.unclassified_program, 0);
|
||||||
|
}
|
||||||
|
|
||||||
|
function cpuMeasuredMedians(records) {
|
||||||
|
const median = values => { const sorted = values.sort((a, b) => a - b); const mid = Math.floor(sorted.length / 2);
|
||||||
|
return sorted.length % 2 ? sorted[mid] : (sorted[mid - 1] + sorted[mid]) / 2; };
|
||||||
|
const groups = new Map();
|
||||||
|
for (const record of records.filter(record => /(?:^|[/\\])(?:control|profiled)-run-\d+(?:[/\\]|$)/.test(record.profile))) {
|
||||||
|
const mode = record.profile.match(/(?:control|profiled)-run-\d+/)[0].split('-')[0];
|
||||||
|
for (const window of record.summary.windows) {
|
||||||
|
const key = `${mode}:${record.kind}:${window.name}`;
|
||||||
|
if (!groups.has(key)) groups.set(key, []);
|
||||||
|
groups.get(key).push(window);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return Object.fromEntries([...groups].map(([key, windows]) => {
|
||||||
|
const categories = new Set(windows.flatMap(window => Object.keys(window.exclusiveOwnerCategories)));
|
||||||
|
return [key, { samples: windows.length,
|
||||||
|
...Object.fromEntries(['wallMs', 'activeJsAndNativeCallSampleMs', 'representedSampleMs', 'samplingGapMs', 'unattributedWallMs']
|
||||||
|
.map(field => [field, median(windows.map(window => window[field]))])),
|
||||||
|
executionSampleMs: Object.fromEntries(Object.keys(windows[0].executionSampleMs).map(state => [state, median(windows.map(window => window.executionSampleMs[state]))])),
|
||||||
|
exclusiveOwnerCategories: Object.fromEntries([...categories].map(category => [category, median(windows.map(window => window.exclusiveOwnerCategories[category] ?? 0))])
|
||||||
|
.sort((a, b) => b[1] - a[1])),
|
||||||
|
}];
|
||||||
|
}));
|
||||||
|
}
|
||||||
|
|
||||||
|
async function summarizeExistingCpu(directory, sourceMapDirectory) {
|
||||||
|
const summaryFile = path.join(directory, 'summary.json');
|
||||||
|
const timingBytes = await fs.readFile(summaryFile);
|
||||||
|
const timing = JSON.parse(timingBytes);
|
||||||
|
assert.ok(timing.cpuDiagnostics?.profiles?.length, 'No captured CPU profiles in summary.json.');
|
||||||
|
const sourceDirectory = sourceMapDirectory ?? timing.cpuDiagnostics.sourceMaps.find(item => item.status === 'verified')?.generated;
|
||||||
|
const { consumers, evidence: sourceMaps } = await loadVerifiedCpuMaps(sourceMapDirectory ?? (sourceDirectory && path.dirname(sourceDirectory)), timing.servedAssets);
|
||||||
|
const records = [];
|
||||||
|
const index = { schemaVersion: 2, browserLaunched: false, sourceTimingSummary: summaryFile, sourceTimingSha256: sha(timingBytes),
|
||||||
|
scriptSha256: sha(await fs.readFile(new URL(import.meta.url))), sourceMaps, profiles: [] };
|
||||||
|
for (const entry of timing.cpuDiagnostics.profiles) {
|
||||||
|
// Prefer paths relative to the supplied directory so captured runs can be moved.
|
||||||
|
const profileFile = path.join(directory, path.basename(path.dirname(entry.profile)), path.basename(entry.profile));
|
||||||
|
const oldSummaryFile = path.join(path.dirname(profileFile), path.basename(entry.summary));
|
||||||
|
const traceFile = path.join(path.dirname(profileFile), entry.kind === 'restore' ? 'restore-trace.json' : 'trace.json');
|
||||||
|
const [profileBytes, previousBytes, traceBytes] = await Promise.all([fs.readFile(profileFile), fs.readFile(oldSummaryFile), fs.readFile(traceFile)]);
|
||||||
|
const previous = JSON.parse(previousBytes);
|
||||||
|
const record = { profile: JSON.parse(profileBytes), trace: JSON.parse(traceBytes), kind: entry.kind,
|
||||||
|
intervalUs: previous.intervalUs, calibration: previous.clockCalibration.probes };
|
||||||
|
const summary = summarizeCpu(record, consumers);
|
||||||
|
const output = profileFile.replace(/\.cpuprofile$/, '-cpu-summary-v2.json');
|
||||||
|
await fs.writeFile(output, JSON.stringify(summary, null, 2));
|
||||||
|
await fs.writeFile(output.replace(/\.json$/, '.md'), cpuSummaryMarkdown(summary, profileFile));
|
||||||
|
index.profiles.push({ profile: profileFile, profileSha256: sha(profileBytes), trace: traceFile, traceSha256: sha(traceBytes),
|
||||||
|
sourceSummary: oldSummaryFile, sourceSummarySha256: sha(previousBytes), summary: output, kind: entry.kind });
|
||||||
|
records.push({ profile: profileFile, kind: entry.kind, summary });
|
||||||
|
// Offline analysis must never change recorded timing, traces or raw profiles.
|
||||||
|
assert.equal(sha(await fs.readFile(profileFile)), sha(profileBytes));
|
||||||
|
assert.equal(sha(await fs.readFile(traceFile)), sha(traceBytes));
|
||||||
|
}
|
||||||
|
index.measuredMedians = cpuMeasuredMedians(records);
|
||||||
|
index.medianDefinition = 'Per-field medians over measured runs only; warmups excluded. Median columns and overlapping windows are not additive. Active includes JS plus native calls under JS, excluding idle/GC/program/unknown runtime.';
|
||||||
|
assert.equal(sha(await fs.readFile(summaryFile)), sha(timingBytes));
|
||||||
|
const indexFile = path.join(directory, 'cpu-resummary-v2.json');
|
||||||
|
await fs.writeFile(indexFile, JSON.stringify(index, null, 2));
|
||||||
|
console.log(JSON.stringify({ indexFile, profiles: records.length, originalTimingAndProfilesUnchanged: true, browserLaunched: false }));
|
||||||
|
}
|
||||||
|
|
||||||
|
if (options.check) {
|
||||||
|
checkCpuSummaryContract();
|
||||||
|
if (options.summarizeCpuOnly) {
|
||||||
|
const manifest = JSON.parse(await fs.readFile(path.join(options.summarizeCpuOnly, 'summary.json')));
|
||||||
|
assert.ok(manifest.cpuDiagnostics?.profiles?.length);
|
||||||
|
console.log(JSON.stringify({ directory: options.summarizeCpuOnly, profiles: manifest.cpuDiagnostics.profiles.length,
|
||||||
|
cpuClassificationContractPassed: true, browserLaunched: false }));
|
||||||
|
} else {
|
||||||
|
console.log(JSON.stringify({ input: options.input, inputSha256: sha(inputText), nodes: project.nodes.length,
|
||||||
|
edges: project.edges.length, curveNodeId, mode: options.mode, deep: Boolean(options.deep), cpuIntervalUs: Number(options.cpuIntervalUs),
|
||||||
|
sourceMapDir: options.sourceMapDir ?? null, warmupsPerMode: 1, measuredRunsPerMode: Number(options.runs),
|
||||||
|
cpuClassificationContractPassed: true, browserLaunched: false }));
|
||||||
|
}
|
||||||
|
process.exit(0);
|
||||||
|
}
|
||||||
|
if (options.summarizeCpuOnly) {
|
||||||
|
checkCpuSummaryContract();
|
||||||
|
await summarizeExistingCpu(options.summarizeCpuOnly, options.sourceMapDir);
|
||||||
|
process.exit(0);
|
||||||
|
}
|
||||||
|
|
||||||
|
const definitions = {
|
||||||
|
clock: 'All timestamps are performance.now() in the stated document timeOrigin; reload uses a separate time axis.',
|
||||||
|
clickToFetch: 'Includes model checking, snapshot construction, XML generation and submission setup; not isolated XML CPU time.',
|
||||||
|
headersAndReads: 'fetch resolution and consumer read delivery. Outstanding-read intervals include backend production, transport and browser scheduling; not pure network time.',
|
||||||
|
unobservedCpu: 'NDJSON fragment scanning/join/trim and React handler CPU are not isolated. Their residual intervals can also include scheduling and cannot be attributed wholesale to parsing, transport or drawing.',
|
||||||
|
jsonParse: 'Only the original synchronous JSON.parse call, called once per application parse. No duplicate body read, decode, scan or parse.',
|
||||||
|
resultReady: 'First DOM observation of successful completion plus an enabled run button after busy state. React state/handler boundaries are not directly instrumented.',
|
||||||
|
indexedDb: 'Profiled: session pointer publication immediately after all save transactions commit. Control: pointer polling, up to 16 ms plus scheduling delay. Transaction windows also include asynchronous waiting and may include old-cache cleanup.',
|
||||||
|
render: 'First visible DOM, then two requestAnimationFrame callbacks (paint opportunity, not GPU completion); stable means scoped DOM quiet for 120 ms followed by two frames.',
|
||||||
|
export: 'User click to blob anchor invocation (profiled) and Playwright download completion observed back on the page clock. Completion includes automation notification and filesystem saveAs overhead.',
|
||||||
|
worker: 'Profiled only: native Worker construction, original postMessage calls with unchanged transfer lists, and complete-message receipt on the document clock. No worker injection, payload copy or Blob read. Finish-post to receipt includes worker scheduling/encoding/Blob creation/message delivery, not isolated worker CPU. Main-thread preparation between posts overlaps worker activity and includes deliberate yields.',
|
||||||
|
control: 'Same user workflow and minimal click/DOM observations, without fetch/reader/JSON/decoder/IDB/Storage/Worker/anchor wrappers or long-task observer.',
|
||||||
|
overlap: 'Intervals overlap. Do not add read waits, parse, main-thread tasks, rendering, persistence, backend native time or residual differences as exclusive costs.',
|
||||||
|
import: 'Every run reimports the fixed input before runClick. Timing is native file-input change through import-success DOM and two frames. Public project export then verifies node/edge counts, all parameters and endpoints outside simulation timing.',
|
||||||
|
warmup: 'One complete warmup per mode excluded from measured summaries. Modes run sequentially; ordering/cache/thermal effects remain possible.',
|
||||||
|
};
|
||||||
|
const cpuRecords = [];
|
||||||
|
const rows = [];
|
||||||
|
const errors = [];
|
||||||
|
const browser = await chromium.launch({ headless: true });
|
||||||
|
const evidence = { input: path.resolve(options.input), inputSha256: sha(inputText), baseURL: options.url,
|
||||||
|
browser: browser.version(), node: process.version, deep: Boolean(options.deep), cpuIntervalUs: options.deep ? Number(options.cpuIntervalUs) : null, initialNavigation: {}, scriptSha256: sha(await fs.readFile(new URL(import.meta.url))), definitions, rows, errors, servedAssets: [] };
|
||||||
|
const writeSummary = () => fs.writeFile(path.join(options.output, 'summary.json'), JSON.stringify(evidence, null, 2));
|
||||||
|
const waitMark = async (page, name) => {
|
||||||
|
await page.waitForFunction(key => window.__stageProfile?.marks[key] !== undefined || window.__stageProfile?.marks.runFailure !== undefined, name, { timeout: 180000 });
|
||||||
|
const failure = await page.evaluate(() => window.__stageProfile.runFailure);
|
||||||
|
if (failure) throw new Error(`Page simulation failed: ${failure}`);
|
||||||
|
};
|
||||||
|
const pageMark = (page, name) => page.evaluate(key => window.__stageProfile.mark(key), name);
|
||||||
|
const arm = (page, name, selector) => page.evaluate(({ name, selector }) => window.__stageProfile.armClick(name, selector), { name, selector });
|
||||||
|
const watch = (page, name, selector) => page.evaluate(({ name, selector }) => window.__stageProfile.watchDom(name, selector), { name, selector });
|
||||||
|
const clickTab = async (page, name) => { await page.getByRole('tab', { name }).click(); };
|
||||||
|
const resultDigest = result => sha(JSON.stringify(result));
|
||||||
|
const delta = (m, a, b) => m[a] === undefined || m[b] === undefined ? null : m[b] - m[a];
|
||||||
|
function metrics(trace) {
|
||||||
|
const m = trace.marks;
|
||||||
|
const csvAnchor = trace.downloads.find(d => d.name.endsWith('.csv'))?.anchorClick;
|
||||||
|
const resultAnchor = trace.downloads.find(d => d.name.endsWith('.simresult'))?.anchorClick;
|
||||||
|
const csvRequest = trace.requests.find(r => r.kind === 'csv');
|
||||||
|
const worker = trace.workers?.[0];
|
||||||
|
const workerStart = worker?.posts.find(p => p.type === 'start');
|
||||||
|
const workerFinish = worker?.posts.find(p => p.type === 'finish');
|
||||||
|
return {
|
||||||
|
importToDomMs: delta(m, 'importChange', 'importReadyDom'),
|
||||||
|
importToPaintOpportunityMs: delta(m, 'importChange', 'importReadyPaintOpportunity'),
|
||||||
|
clickToFetchMs: delta(m, 'runClick', 'fetchStart'), fetchToHeadersMs: delta(m, 'fetchStart', 'headers'),
|
||||||
|
headersToEofMs: delta(m, 'headers', 'streamEof'), clickToReadyDomMs: delta(m, 'runClick', 'resultReadyDom'),
|
||||||
|
clickToReadyPaintOpportunityMs: delta(m, 'runClick', 'resultReadyPaintOpportunity'),
|
||||||
|
resultParseMs: delta(m, 'resultParseStart', 'resultParseEnd'),
|
||||||
|
resultParseEndToReadyDomMs: delta(m, 'resultParseEnd', 'resultReadyDom'),
|
||||||
|
clickToIndexedDbCommitMs: delta(m, 'runClick', 'indexedDbCommittedPointer'),
|
||||||
|
clickToIndexedDbObservedMs: delta(m, 'runClick', 'indexedDbPointerObserved'),
|
||||||
|
resultTabToDomMs: delta(m, 'resultTabClick', 'resultsDom'),
|
||||||
|
resultTabToPaintOpportunityMs: delta(m, 'resultTabClick', 'resultsPaintOpportunity'),
|
||||||
|
resultTabToDomStableMs: delta(m, 'resultTabClick', 'resultsStable'),
|
||||||
|
curveSelectToDomMs: delta(m, 'temperatureClick', 'curveDom'),
|
||||||
|
curveSelectToPaintOpportunityMs: delta(m, 'temperatureClick', 'curvePaintOpportunity'),
|
||||||
|
curveSelectToDomStableMs: delta(m, 'temperatureClick', 'curveStable'),
|
||||||
|
csvClickToWorkerConstructMs: worker && m.csvClick !== undefined ? worker.constructStart - m.csvClick : null,
|
||||||
|
csvWorkerConstructMs: worker ? worker.constructEnd - worker.constructStart : null,
|
||||||
|
csvWorkerStartToFinishPostMs: workerStart && workerFinish ? workerFinish.end - workerStart.start : null,
|
||||||
|
csvWorkerFinishPostToCompleteReceivedMs: workerFinish && worker.completeReceived !== undefined ? worker.completeReceived - workerFinish.end : null,
|
||||||
|
csvWorkerStartToCompleteReceivedMs: workerStart && worker.completeReceived !== undefined ? worker.completeReceived - workerStart.start : null,
|
||||||
|
csvWorkerPostSyncTotalMs: worker ? worker.posts.reduce((total, post) => total + post.end - post.start, 0) : null,
|
||||||
|
csvWorkerTransferredBytes: worker ? worker.posts.reduce((total, post) => total + post.bytes, 0) : null,
|
||||||
|
csvClickToFetchMs: csvRequest && m.csvClick !== undefined ? csvRequest.fetchStart - m.csvClick : null,
|
||||||
|
csvFetchToHeadersMs: csvRequest?.headers === undefined ? null : csvRequest.headers - csvRequest.fetchStart,
|
||||||
|
csvClickToBlobAnchorMs: csvAnchor === undefined ? null : csvAnchor - m.csvClick,
|
||||||
|
resultClickToBlobAnchorMs: resultAnchor === undefined ? null : resultAnchor - m.resultFileClick,
|
||||||
|
csvClickToDownloadSavedMs: delta(m, 'csvClick', 'csvDownloadSaved'),
|
||||||
|
resultClickToDownloadSavedMs: delta(m, 'resultFileClick', 'resultFileDownloadSaved'),
|
||||||
|
streamReadCount: trace.profiled ? trace.reads.filter(r => r.kind === 'simulation').length : null,
|
||||||
|
streamBytes: trace.profiled ? trace.reads.reduce((n, r) => n + (r.kind === 'simulation' ? r.bytes : 0), 0) : null,
|
||||||
|
streamOutstandingReadMs: trace.profiled ? trace.reads.reduce((n, r) => n + (r.kind === 'simulation' ? r.end - r.start : 0), 0) : null,
|
||||||
|
synchronousStreamJsonParseMs: trace.profiled ? trace.parses.reduce((n, r) => n + r.end - r.start, 0) : null,
|
||||||
|
synchronousStreamDecodeMs: trace.profiled ? trace.decodes.reduce((n, r) => n + r.end - r.start, 0) : null,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
try {
|
||||||
|
for (const mode of options.mode === 'both' ? ['control', 'profiled'] : [options.mode]) {
|
||||||
|
const context = await browser.newContext({ viewport: { width: 1600, height: 1000 }, acceptDownloads: true });
|
||||||
|
await context.addInitScript(installStageObserver, { profiled: mode === 'profiled' });
|
||||||
|
const page = await context.newPage();
|
||||||
|
const cpuRecorder = options.deep ? await createCpuRecorder(context, page, Number(options.cpuIntervalUs)) : null;
|
||||||
|
page.setDefaultTimeout(30000);
|
||||||
|
const simulationRequests = [];
|
||||||
|
const workerUrls = new Set();
|
||||||
|
page.on('worker', worker => workerUrls.add(worker.url()));
|
||||||
|
page.on('request', request => {
|
||||||
|
if (request.url().includes('/api/system-xml/simulate-stream')) {
|
||||||
|
simulationRequests.push({ url: request.url(), simulationId: request.headers()['x-simulation-id'] ?? null });
|
||||||
|
}
|
||||||
|
});
|
||||||
|
page.on('pageerror', error => errors.push({ mode, error: String(error) }));
|
||||||
|
page.on('dialog', dialog => { errors.push({ mode, dialog: dialog.message() }); void dialog.dismiss(); });
|
||||||
|
try {
|
||||||
|
await page.goto(options.url);
|
||||||
|
await page.locator('input[type="file"][accept*=".json"]').waitFor({ state: 'attached' });
|
||||||
|
evidence.initialNavigation[mode] = await page.evaluate(() => ({ timeOrigin: performance.timeOrigin,
|
||||||
|
appControlObservedAt: performance.now(), navigation: performance.getEntriesByType('navigation').map(e => e.toJSON()) }));
|
||||||
|
const assetUrls = await page.evaluate(() => [...document.querySelectorAll('script[src],link[rel="stylesheet"][href]')]
|
||||||
|
.map(e => e.src || e.href));
|
||||||
|
if (assetUrls.some(url => /@vite\/client|\/src\//.test(url))) throw new Error('Expected a production build, found Vite development assets.');
|
||||||
|
for (const url of assetUrls) {
|
||||||
|
const response = await context.request.get(url);
|
||||||
|
assert.ok(response.ok(), `Asset HTTP ${response.status()}: ${url}`);
|
||||||
|
const bytes = await response.body();
|
||||||
|
const previous = evidence.servedAssets.find(asset => asset.url === url);
|
||||||
|
if (previous) assert.equal(previous.sha256, sha(bytes), 'Production asset changed between modes.');
|
||||||
|
else evidence.servedAssets.push({ url, sha256: sha(bytes), bytes: bytes.length });
|
||||||
|
}
|
||||||
|
assert.ok(evidence.servedAssets.length, 'No production assets found.');
|
||||||
|
evidence.buildAssetSetSha256 = sha(JSON.stringify(evidence.servedAssets
|
||||||
|
.map(({ url, ...asset }) => ({ path: new URL(url).pathname, ...asset }))
|
||||||
|
.sort((a, b) => a.path.localeCompare(b.path))));
|
||||||
|
for (let run = 0; run <= Number(options.runs); run++) {
|
||||||
|
const prefix = `${mode}-${run === 0 ? 'warmup' : `run-${run}`}`;
|
||||||
|
const runDir = path.join(options.output, prefix);
|
||||||
|
await fs.mkdir(runDir, { recursive: true });
|
||||||
|
await clickTab(page, '建模');
|
||||||
|
const expand = page.getByRole('button', { name: '展开仿真控制台', exact: true });
|
||||||
|
if (await expand.isVisible()) await expand.click();
|
||||||
|
const requestOffset = simulationRequests.length;
|
||||||
|
if (cpuRecorder) await cpuRecorder.start();
|
||||||
|
await page.evaluate(name => { window.__stageProfile.reset(); window.__stageProfile.watchImport(name); },
|
||||||
|
path.basename(options.input, path.extname(options.input)));
|
||||||
|
await page.locator('input[type="file"][accept*=".json"]').setInputFiles(path.resolve(options.input));
|
||||||
|
await waitMark(page, 'importReadyPaintOpportunity');
|
||||||
|
// Verify the public project export outside simulation timing; importing after each
|
||||||
|
// reload avoids assuming that result recovery also restores the modeling workspace.
|
||||||
|
const projectDownload = page.waitForEvent('download');
|
||||||
|
await page.getByRole('button', { name: '导出工程 JSON', exact: true }).click();
|
||||||
|
await (await projectDownload).saveAs(path.join(runDir, 'imported-project.json'));
|
||||||
|
const imported = JSON.parse(await fs.readFile(path.join(runDir, 'imported-project.json'), 'utf8'));
|
||||||
|
assert.equal(imported.nodes.length, project.nodes.length);
|
||||||
|
assert.equal(imported.edges.length, project.edges.length);
|
||||||
|
assert.equal(imported.name, path.basename(options.input, path.extname(options.input)));
|
||||||
|
assert.deepEqual(imported.simulation, project.simulation);
|
||||||
|
for (const node of project.nodes) assert.deepEqual(imported.nodes.find(n => n.id === node.id)?.data.parameters, node.data.parameters);
|
||||||
|
for (const edge of project.edges) {
|
||||||
|
const actual = imported.edges.find(e => e.id === edge.id);
|
||||||
|
for (const key of ['source', 'target', 'sourceHandle', 'targetHandle']) assert.equal(actual?.[key], edge[key]);
|
||||||
|
}
|
||||||
|
await page.evaluate(() => window.__stageProfile.watchRunReady());
|
||||||
|
await arm(page, 'runClick', 'button[aria-label="运行仿真"]');
|
||||||
|
await page.getByRole('button', { name: '运行仿真', exact: true }).click();
|
||||||
|
await waitMark(page, 'resultReadyPaintOpportunity');
|
||||||
|
// Open results immediately; persistence is allowed to overlap exactly as in real use.
|
||||||
|
await arm(page, 'resultTabClick', '[role="tab"]');
|
||||||
|
await watch(page, 'results', '.results-shell .results-system-panel');
|
||||||
|
await clickTab(page, /^结果/);
|
||||||
|
await waitMark(page, 'resultsStable');
|
||||||
|
await page.getByRole('button', { name: '适应系统图窗口', exact: true }).click();
|
||||||
|
await page.locator(`.results-system-panel .react-flow__node[data-id=${JSON.stringify(curveNodeId)}]`).click();
|
||||||
|
// Remove persisted chart windows outside the curve-selection timing interval.
|
||||||
|
const close = page.locator('.result-chart-window .result-chart-window-actions button.close');
|
||||||
|
while (await close.count()) await close.first().click();
|
||||||
|
assert.equal(await page.locator('.results-chart-panel svg[data-result-chart="true"]').count(), 0,
|
||||||
|
'Curve timing requires no existing chart; update close-window selector if the UI changed.');
|
||||||
|
await arm(page, 'temperatureClick', '.results-variable-list button');
|
||||||
|
await watch(page, 'curve', '.results-chart-panel svg[data-result-chart="true"]');
|
||||||
|
await page.locator('.results-variable-list button').filter({ has: page.locator('small', { hasText: /^K$/ }) }).first().click();
|
||||||
|
await waitMark(page, 'curveStable');
|
||||||
|
await waitMark(page, 'indexedDbPointerObserved');
|
||||||
|
const saveDownload = async (buttonName, clickName, completedName, filename) => {
|
||||||
|
await arm(page, clickName, 'button');
|
||||||
|
const pending = page.waitForEvent('download', { timeout: 180000 });
|
||||||
|
await page.getByRole('button', { name: buttonName, exact: true }).click();
|
||||||
|
const download = await pending;
|
||||||
|
await download.saveAs(path.join(runDir, filename));
|
||||||
|
await pageMark(page, completedName);
|
||||||
|
assert.equal(await download.failure(), null);
|
||||||
|
};
|
||||||
|
await saveDownload('下载结果 CSV', 'csvClick', 'csvDownloadSaved', 'result.csv');
|
||||||
|
await saveDownload('下载结果文件', 'resultFileClick', 'resultFileDownloadSaved', 'result.simresult');
|
||||||
|
const trace = await page.evaluate(() => window.__stageProfile.snapshot());
|
||||||
|
if (cpuRecorder) {
|
||||||
|
const capture = await cpuRecorder.stop();
|
||||||
|
const file = path.join(runDir, 'interaction.cpuprofile');
|
||||||
|
await fs.writeFile(file, JSON.stringify(capture.profile));
|
||||||
|
cpuRecords.push({ ...capture, trace, kind: 'interaction', file });
|
||||||
|
}
|
||||||
|
const exportedBytes = await fs.readFile(path.join(runDir, 'result.simresult'));
|
||||||
|
const exported = JSON.parse(exportedBytes);
|
||||||
|
const result = exported.snapshot.result;
|
||||||
|
assert.equal(result.success, true, result.message);
|
||||||
|
assert.equal(result.simulatedUntil, Number(project.simulation.t_stop));
|
||||||
|
const expectedDigest = resultDigest(result);
|
||||||
|
const csv = await fs.readFile(path.join(runDir, 'result.csv'));
|
||||||
|
await fs.writeFile(path.join(runDir, 'trace.json'), JSON.stringify(trace, null, 2));
|
||||||
|
await page.screenshot({ path: path.join(runDir, 'result.png'), fullPage: true });
|
||||||
|
// Reload the real persisted snapshot without changing its session pointer or result.
|
||||||
|
if (cpuRecorder) await cpuRecorder.start();
|
||||||
|
await page.reload();
|
||||||
|
await page.evaluate(() => { window.__stageProfile.activeRun = true; });
|
||||||
|
await waitMark(page, 'restoredResultsStable');
|
||||||
|
await page.getByRole('button', { name: '下载结果文件', exact: true }).waitFor();
|
||||||
|
if (cpuRecorder) {
|
||||||
|
const restoreCpuTrace = await page.evaluate(() => window.__stageProfile.snapshot());
|
||||||
|
const capture = await cpuRecorder.stop();
|
||||||
|
const file = path.join(runDir, 'restore.cpuprofile');
|
||||||
|
await fs.writeFile(file, JSON.stringify(capture.profile));
|
||||||
|
cpuRecords.push({ ...capture, trace: restoreCpuTrace, kind: 'restore', file });
|
||||||
|
}
|
||||||
|
const restoredDownload = page.waitForEvent('download');
|
||||||
|
await page.getByRole('button', { name: '下载结果文件', exact: true }).click();
|
||||||
|
await (await restoredDownload).saveAs(path.join(runDir, 'restored.simresult'));
|
||||||
|
const restoredBytes = await fs.readFile(path.join(runDir, 'restored.simresult'));
|
||||||
|
const restored = JSON.parse(restoredBytes);
|
||||||
|
assert.equal(resultDigest(restored.snapshot.result), expectedDigest, 'Restored result changed.');
|
||||||
|
const restoreTrace = await page.evaluate(() => ({ ...window.__stageProfile.snapshot(),
|
||||||
|
navigation: performance.getEntriesByType('navigation').map(e => e.toJSON()) }));
|
||||||
|
await fs.writeFile(path.join(runDir, 'restore-trace.json'), JSON.stringify(restoreTrace, null, 2));
|
||||||
|
const requests = simulationRequests.slice(requestOffset);
|
||||||
|
assert.equal(requests.length, 1, 'Expected exactly one real simulation request.');
|
||||||
|
assert.ok(requests[0].simulationId, 'Missing X-Simulation-Id correlation key.');
|
||||||
|
const row = { mode, deep: Boolean(options.deep), run, warmup: run === 0, simulationId: requests[0].simulationId, timeOrigin: trace.timeOrigin, ...metrics(trace),
|
||||||
|
restoreNavigationToDomMs: restoreTrace.marks.restoredResultsDom,
|
||||||
|
restoreNavigationToPaintOpportunityMs: restoreTrace.marks.restoredResultsPaintOpportunity,
|
||||||
|
restoreNavigationToDomStableMs: restoreTrace.marks.restoredResultsStable,
|
||||||
|
sampleCount: result.series.time.length, variableCount: result.variables.length,
|
||||||
|
native: result.diagnostics.native, integration: result.diagnostics.integration,
|
||||||
|
resultSha256: sha(exportedBytes), numericalResultSha256: expectedDigest, resultBytes: exportedBytes.length,
|
||||||
|
csvSha256: sha(csv), csvBytes: csv.length, restoredIdentical: true, artifacts: prefix };
|
||||||
|
rows.push(row);
|
||||||
|
await writeSummary();
|
||||||
|
console.log(JSON.stringify(row));
|
||||||
|
if (errors.length) throw new Error(`Browser errors: ${JSON.stringify(errors)}`);
|
||||||
|
}
|
||||||
|
// Collect actual worker build evidence after all timing intervals. Worker code
|
||||||
|
// is loaded dynamically and does not appear among the initial document tags.
|
||||||
|
for (const url of workerUrls) {
|
||||||
|
if (!/^https?:/.test(url)) continue;
|
||||||
|
const response = await context.request.get(url);
|
||||||
|
assert.ok(response.ok(), `Worker asset HTTP ${response.status()}: ${url}`);
|
||||||
|
const bytes = await response.body();
|
||||||
|
const previous = evidence.servedAssets.find(asset => asset.url === url);
|
||||||
|
if (previous) assert.equal(previous.sha256, sha(bytes), 'Worker build changed between modes.');
|
||||||
|
else evidence.servedAssets.push({ url, sha256: sha(bytes), bytes: bytes.length });
|
||||||
|
}
|
||||||
|
evidence.buildAssetSetSha256 = sha(JSON.stringify(evidence.servedAssets
|
||||||
|
.map(({ url, ...asset }) => ({ path: new URL(url).pathname, ...asset }))
|
||||||
|
.sort((a, b) => a.path.localeCompare(b.path))));
|
||||||
|
await writeSummary();
|
||||||
|
} catch (error) {
|
||||||
|
await page.screenshot({ path: path.join(options.output, `${mode}-failure.png`), fullPage: true }).catch(() => {});
|
||||||
|
const trace = await page.evaluate(() => window.__stageProfile?.snapshot()).catch(() => null);
|
||||||
|
const body = await page.locator('body').innerText().catch(() => null);
|
||||||
|
await fs.writeFile(path.join(options.output, `${mode}-failure.json`), JSON.stringify({ error: error.stack, trace, errors, body }, null, 2));
|
||||||
|
throw error;
|
||||||
|
} finally { if (cpuRecorder) await cpuRecorder.close(); await context.close(); }
|
||||||
|
}
|
||||||
|
if (cpuRecords.length) {
|
||||||
|
const { consumers, evidence: sourceMaps } = await loadVerifiedCpuMaps(options.sourceMapDir, evidence.servedAssets);
|
||||||
|
evidence.cpuDiagnostics = { sourceMaps, profiles: [] };
|
||||||
|
for (const record of cpuRecords) {
|
||||||
|
const summary = summarizeCpu(record, consumers);
|
||||||
|
const output = record.file.replace(/\.cpuprofile$/, '-cpu-summary.json');
|
||||||
|
await fs.writeFile(output, JSON.stringify(summary, null, 2));
|
||||||
|
await fs.writeFile(output.replace(/\.json$/, '.md'), cpuSummaryMarkdown(summary, record.file));
|
||||||
|
evidence.cpuDiagnostics.profiles.push({ profile: record.file, summary: output, kind: record.kind });
|
||||||
|
}
|
||||||
|
}
|
||||||
|
const median = values => { const v = values.filter(n => typeof n === 'number').sort((a, b) => a - b);
|
||||||
|
return v.length ? v.length % 2 ? v[Math.floor(v.length / 2)] : (v[v.length / 2 - 1] + v[v.length / 2]) / 2 : null; };
|
||||||
|
evidence.measuredMedians = Object.fromEntries(['control', 'profiled'].map(mode => [mode,
|
||||||
|
Object.fromEntries(Object.keys(rows.find(r => r.mode === mode) ?? {}).filter(k => k.endsWith('Ms'))
|
||||||
|
.map(k => [k, median(rows.filter(r => r.mode === mode && !r.warmup).map(r => r[k]))]))]));
|
||||||
|
await writeSummary();
|
||||||
|
} finally { await writeSummary(); await browser.close(); }
|
||||||
@@ -0,0 +1,135 @@
|
|||||||
|
"""Compare all real browser CSV/result/restore values with a native execution.
|
||||||
|
|
||||||
|
Run only after performance measurements finish; JSON/CSV verification is CPU intensive.
|
||||||
|
No tolerance or resampling is used. Diagnostic timing fields intentionally differ.
|
||||||
|
"""
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import csv
|
||||||
|
import hashlib
|
||||||
|
import json
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
|
||||||
|
def sha256(path: Path) -> str:
|
||||||
|
digest = hashlib.sha256()
|
||||||
|
with path.open('rb') as source:
|
||||||
|
for chunk in iter(lambda: source.read(1024 * 1024), b''):
|
||||||
|
digest.update(chunk)
|
||||||
|
return digest.hexdigest()
|
||||||
|
|
||||||
|
|
||||||
|
def main() -> None:
|
||||||
|
parser = argparse.ArgumentParser(description=__doc__)
|
||||||
|
parser.add_argument('directory', type=Path)
|
||||||
|
parser.add_argument('--native', type=Path, help='Full native result baseline; defaults to DIRECTORY/native-production/run-1/result.json')
|
||||||
|
parser.add_argument('--group', action='append', default=[], metavar='LABEL=PATH', help='Explicit browser output group; repeat for old/new/profiled groups')
|
||||||
|
parser.add_argument('--output', type=Path, help='Defaults to DIRECTORY/equality.json')
|
||||||
|
args = parser.parse_args()
|
||||||
|
base = args.directory
|
||||||
|
native_path = args.native or base / 'native-production/run-1/result.json'
|
||||||
|
groups = []
|
||||||
|
for group in args.group:
|
||||||
|
label, separator, directory = group.partition('=')
|
||||||
|
if not separator or not label or not directory:
|
||||||
|
parser.error('--group requires LABEL=PATH')
|
||||||
|
groups.append((label, Path(directory)))
|
||||||
|
if not groups:
|
||||||
|
groups = [(mode, base / f'browser-{mode}') for mode in ('control', 'profiled')]
|
||||||
|
assert len({label for label, _ in groups}) == len(groups), 'Group labels must be unique'
|
||||||
|
native = json.loads(native_path.read_text())
|
||||||
|
assert native['success'] and native['simulatedUntil'] == 10
|
||||||
|
expected_series, expected_final = native['series'], native['final']
|
||||||
|
point_count = len(expected_series['time'])
|
||||||
|
assert len(expected_series) == 1785 and point_count == 1002
|
||||||
|
report = {
|
||||||
|
'native': str(native_path), 'nativeSha256': sha256(native_path),
|
||||||
|
'comparison': 'Numeric equality using == on parsed numbers, with identical keys and lengths; no tolerance, conversion, interpolation or rounding. Diagnostic timings are excluded.',
|
||||||
|
'sampleCount': point_count, 'seriesCountIncludingTime': len(expected_series),
|
||||||
|
'finalValueCount': len(expected_final), 'runs': [], 'groups': [],
|
||||||
|
}
|
||||||
|
input_hashes, build_hashes, csv_hashes = set(), set(), set()
|
||||||
|
for label, group in groups:
|
||||||
|
group_csv_hashes = set()
|
||||||
|
summary = json.loads((group / 'summary.json').read_text())
|
||||||
|
assert not summary['errors'], summary['errors']
|
||||||
|
assert len(summary['rows']) == 4
|
||||||
|
assert sum(bool(row['warmup']) for row in summary['rows']) == 1
|
||||||
|
assert len({row['run'] for row in summary['rows']}) == 4
|
||||||
|
input_hashes.add(summary['inputSha256'])
|
||||||
|
build_hashes.add(summary['buildAssetSetSha256'])
|
||||||
|
for row in summary['rows']:
|
||||||
|
directory = group / row['artifacts']
|
||||||
|
result_path = directory / 'result.simresult'
|
||||||
|
result = json.loads(result_path.read_text())['snapshot']['result']
|
||||||
|
assert result['success'] and result['simulatedUntil'] == 10
|
||||||
|
assert set(result['series']) == set(expected_series), directory
|
||||||
|
assert result['series'] == expected_series, f'{directory}: series differs from native'
|
||||||
|
assert result['final'] == expected_final, f'{directory}: final differs from native'
|
||||||
|
variables = [variable['key'] for variable in result['variables']]
|
||||||
|
assert len(variables) == 1784 and len(set(variables)) == 1784
|
||||||
|
restored_path = directory / 'restored.simresult'
|
||||||
|
restored = json.loads(restored_path.read_text())['snapshot']['result']
|
||||||
|
assert restored == result, f'{directory}: restored result changed'
|
||||||
|
del restored
|
||||||
|
csv_path = directory / 'result.csv'
|
||||||
|
checked = 0
|
||||||
|
with csv_path.open(newline='', encoding='utf-8-sig') as source:
|
||||||
|
reader = csv.reader(source)
|
||||||
|
headers = next(reader)
|
||||||
|
assert headers == ['time', *variables], f'{directory}: CSV variable order differs'
|
||||||
|
assert len(headers) == 1785 and set(headers) == set(expected_series)
|
||||||
|
columns = [expected_series[key] for key in headers]
|
||||||
|
row_count = 0
|
||||||
|
for index, cells in enumerate(reader):
|
||||||
|
assert index < point_count and len(cells) == len(headers), (directory, index)
|
||||||
|
for column_index, (cell, column) in enumerate(zip(cells, columns, strict=True)):
|
||||||
|
assert float(cell) == column[index], (directory, index, headers[column_index], cell, column[index])
|
||||||
|
checked += 1
|
||||||
|
row_count += 1
|
||||||
|
assert row_count == point_count
|
||||||
|
csv_hash = sha256(csv_path)
|
||||||
|
csv_hashes.add(csv_hash)
|
||||||
|
group_csv_hashes.add(csv_hash)
|
||||||
|
item = {
|
||||||
|
'group': label, 'mode': row['mode'], 'run': row['run'], 'warmup': row['warmup'],
|
||||||
|
'simulationId': row['simulationId'], 'directory': str(directory),
|
||||||
|
'seriesComparedValues': sum(map(len, expected_series.values())),
|
||||||
|
'finalComparedValues': len(expected_final), 'csvComparedCells': checked,
|
||||||
|
'csvColumns': len(headers), 'csvRows': row_count,
|
||||||
|
'seriesExactlyEqualToNative': True, 'finalExactlyEqualToNative': True,
|
||||||
|
'csvAllCellsExactlyEqualToNative': True, 'restoredResultExactlyEqual': True,
|
||||||
|
'resultSha256': sha256(result_path), 'restoredSha256': sha256(restored_path),
|
||||||
|
'csvSha256': csv_hash,
|
||||||
|
}
|
||||||
|
report['runs'].append(item)
|
||||||
|
print(json.dumps({'group': label, 'mode': row['mode'], 'run': row['run'], 'checkedCsvCells': checked, 'exact': True}), flush=True)
|
||||||
|
assert len(group_csv_hashes) == 1, f'{label}: CSV bytes differ between repeated runs'
|
||||||
|
report['groups'].append({
|
||||||
|
'label': label, 'directory': str(group), 'runCount': len(summary['rows']),
|
||||||
|
'buildAssetSetSha256': summary['buildAssetSetSha256'],
|
||||||
|
'inputSha256': summary['inputSha256'], 'allCsvFilesByteIdentical': True,
|
||||||
|
'csvSha256': next(iter(group_csv_hashes)),
|
||||||
|
})
|
||||||
|
assert len(input_hashes) == 1
|
||||||
|
# Python and JavaScript emit different round-tripping spellings (e.g. 0.0/0).
|
||||||
|
# Across versions CSV numeric equality is required; text hashes are per group.
|
||||||
|
report.update({
|
||||||
|
'allPassed': True, 'inputSha256': next(iter(input_hashes)),
|
||||||
|
'buildAssetSetSha256': next(iter(build_hashes)) if len(build_hashes) == 1 else None,
|
||||||
|
'buildAssetSetSha256Values': sorted(build_hashes),
|
||||||
|
'allGroupsCsvFilesByteIdentical': True,
|
||||||
|
'allCsvFilesByteIdentical': len(csv_hashes) == 1,
|
||||||
|
'csvSha256': next(iter(csv_hashes)) if len(csv_hashes) == 1 else None,
|
||||||
|
'totalCsvCellsCompared': sum(row['csvComparedCells'] for row in report['runs']),
|
||||||
|
'totalSeriesValuesCompared': sum(row['seriesComparedValues'] for row in report['runs']),
|
||||||
|
'totalFinalValuesCompared': sum(row['finalComparedValues'] for row in report['runs']),
|
||||||
|
})
|
||||||
|
output = args.output or base / 'equality.json'
|
||||||
|
output.write_text(json.dumps(report, indent=2) + '\n')
|
||||||
|
print(json.dumps({'output': str(output), 'allPassed': True, 'totalCsvCellsCompared': report['totalCsvCellsCompared']}))
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == '__main__':
|
||||||
|
main()
|
||||||
@@ -0,0 +1,236 @@
|
|||||||
|
"""Compare complete native numerical results as binary64, outside benchmark timing.
|
||||||
|
|
||||||
|
.venv/bin/python tests/manual/compare_native_result_bits.py \
|
||||||
|
--baseline old/result.json --candidate new/run-1/result.json \
|
||||||
|
--candidate new/run-2/result.json --output test/result-bit-parity.json
|
||||||
|
|
||||||
|
JSON number spelling may change. In particular, the integer token -0 must be
|
||||||
|
parsed as negative floating zero before packing. All series columns (including
|
||||||
|
time), final scalars and finalState entries are compared without sampling.
|
||||||
|
Solver status/configuration/counters must also agree. Only solve wall/CPU timing
|
||||||
|
metadata is intentionally ignored. Nonfinite payload or metadata numbers fail.
|
||||||
|
"""
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
from hashlib import sha256
|
||||||
|
import json
|
||||||
|
import math
|
||||||
|
from pathlib import Path
|
||||||
|
import struct
|
||||||
|
import sys
|
||||||
|
|
||||||
|
PAYLOAD_KEYS = ("series", "final", "finalState")
|
||||||
|
FLOAT_METADATA = ("simulatedUntil", "maxAcceptedStep")
|
||||||
|
COUNT_METADATA = ("nfev", "acceptedSteps", "rejectedSteps", "njev", "nlu", "stateTransitions", "solverStarts")
|
||||||
|
VALUE_METADATA = ("success", "status", "message", "method", "backend", "solver", "sundialsVersion")
|
||||||
|
TIMING_METADATA = ("solveSeconds", "solveCpuSeconds")
|
||||||
|
NEGATIVE_ZERO = struct.pack("<Q", 1 << 63)
|
||||||
|
POSITIVE_ZERO = b"\0" * 8
|
||||||
|
|
||||||
|
|
||||||
|
class ComparisonError(ValueError):
|
||||||
|
def __init__(self, path: str, reason: str, **details: object) -> None:
|
||||||
|
super().__init__(f"{path}: {reason}")
|
||||||
|
self.detail = {"path": path, "reason": reason, **details}
|
||||||
|
|
||||||
|
|
||||||
|
def pointer(*parts: object) -> str:
|
||||||
|
return "/" + "/".join(str(part).replace("~", "~0").replace("/", "~1") for part in parts)
|
||||||
|
|
||||||
|
|
||||||
|
def read_result(path: Path) -> tuple[dict, str]:
|
||||||
|
raw = path.read_bytes()
|
||||||
|
|
||||||
|
def unique_object(items: list[tuple[str, object]]) -> dict:
|
||||||
|
result = {}
|
||||||
|
for key, value in items:
|
||||||
|
if key in result:
|
||||||
|
raise ComparisonError("/", "Duplicate JSON object key", key=key)
|
||||||
|
result[key] = value
|
||||||
|
return result
|
||||||
|
|
||||||
|
def reject_constant(token: str) -> object:
|
||||||
|
raise ComparisonError("/", "Nonfinite JSON token", token=token)
|
||||||
|
|
||||||
|
value = json.loads(raw, parse_int=lambda token: -0.0 if token == "-0" else int(token),
|
||||||
|
parse_constant=reject_constant, object_pairs_hook=unique_object)
|
||||||
|
if not isinstance(value, dict):
|
||||||
|
raise ComparisonError("/", "Expected a native result object")
|
||||||
|
return value, sha256(raw).hexdigest()
|
||||||
|
|
||||||
|
|
||||||
|
def bits(value: object, path: str) -> bytes:
|
||||||
|
if isinstance(value, bool) or not isinstance(value, (int, float)):
|
||||||
|
raise ComparisonError(path, "Expected a finite numeric value", actualType=type(value).__name__)
|
||||||
|
try:
|
||||||
|
number = float(value)
|
||||||
|
except (OverflowError, ValueError):
|
||||||
|
raise ComparisonError(path, "Number cannot be represented as finite binary64") from None
|
||||||
|
if not math.isfinite(number):
|
||||||
|
raise ComparisonError(path, "Nonfinite binary64 value")
|
||||||
|
return struct.pack("<d", number)
|
||||||
|
|
||||||
|
|
||||||
|
def ensure_finite_tree(value: object, path: str = "") -> None:
|
||||||
|
"""Reject overflow-to-infinity tokens even in metadata excluded from parity."""
|
||||||
|
if isinstance(value, dict):
|
||||||
|
for key, child in value.items():
|
||||||
|
ensure_finite_tree(child, path + pointer(key))
|
||||||
|
elif isinstance(value, list):
|
||||||
|
for index, child in enumerate(value):
|
||||||
|
ensure_finite_tree(child, path + pointer(index))
|
||||||
|
elif isinstance(value, (int, float)) and not isinstance(value, bool):
|
||||||
|
bits(value, path or "/")
|
||||||
|
|
||||||
|
|
||||||
|
def validate_result(result: dict) -> dict:
|
||||||
|
required = set(PAYLOAD_KEYS + FLOAT_METADATA + COUNT_METADATA + VALUE_METADATA)
|
||||||
|
if missing := required - result.keys():
|
||||||
|
raise ComparisonError("/", "Missing native result fields", missing=sorted(missing))
|
||||||
|
if not isinstance(result["series"], dict) or not isinstance(result["final"], dict):
|
||||||
|
raise ComparisonError("/", "series and final must be objects")
|
||||||
|
if not isinstance(result["finalState"], list):
|
||||||
|
raise ComparisonError("/finalState", "Expected an array")
|
||||||
|
for key, values in result["series"].items():
|
||||||
|
if not isinstance(values, list):
|
||||||
|
raise ComparisonError(pointer("series", key), "Expected a numeric array")
|
||||||
|
if result["series"] and "time" not in result["series"]:
|
||||||
|
raise ComparisonError("/series", "Nonempty series has no time column")
|
||||||
|
if result["series"]:
|
||||||
|
samples = len(result["series"]["time"])
|
||||||
|
for key, values in result["series"].items():
|
||||||
|
if len(values) != samples:
|
||||||
|
raise ComparisonError(pointer("series", key), "Column length differs from time", expectedLength=samples, actualLength=len(values))
|
||||||
|
for key in COUNT_METADATA:
|
||||||
|
value = result[key]
|
||||||
|
if isinstance(value, bool) or not isinstance(value, int) or value < 0:
|
||||||
|
raise ComparisonError(pointer(key), "Expected a nonnegative integer solver counter")
|
||||||
|
if not isinstance(result["success"], bool):
|
||||||
|
raise ComparisonError("/success", "Expected a boolean")
|
||||||
|
for key in VALUE_METADATA[1:]:
|
||||||
|
if not isinstance(result[key], str):
|
||||||
|
raise ComparisonError(pointer(key), "Expected string metadata")
|
||||||
|
ensure_finite_tree(result)
|
||||||
|
for key in FLOAT_METADATA:
|
||||||
|
bits(result[key], pointer(key))
|
||||||
|
# Scalars in final and all payload cells must be numeric, never bool/null.
|
||||||
|
count = negative_zeroes = positive_zeroes = 0
|
||||||
|
for path, value in payload_values(result):
|
||||||
|
packed = bits(value, path)
|
||||||
|
count += 1
|
||||||
|
negative_zeroes += packed == NEGATIVE_ZERO
|
||||||
|
positive_zeroes += packed == POSITIVE_ZERO
|
||||||
|
return {"payloadValues": count, "seriesColumns": len(result["series"]),
|
||||||
|
"samples": len(result["series"].get("time", [])), "finalScalars": len(result["final"]),
|
||||||
|
"finalStateValues": len(result["finalState"]), "negativeZeroValues": negative_zeroes,
|
||||||
|
"positiveZeroValues": positive_zeroes}
|
||||||
|
|
||||||
|
|
||||||
|
def payload_values(result: dict):
|
||||||
|
for key, values in result["series"].items():
|
||||||
|
for index, value in enumerate(values):
|
||||||
|
yield pointer("series", key, index), value
|
||||||
|
for key, value in result["final"].items():
|
||||||
|
yield pointer("final", key), value
|
||||||
|
for index, value in enumerate(result["finalState"]):
|
||||||
|
yield pointer("finalState", index), value
|
||||||
|
|
||||||
|
|
||||||
|
def match_keys(baseline: dict, candidate: dict, path: str) -> None:
|
||||||
|
if baseline.keys() != candidate.keys():
|
||||||
|
raise ComparisonError(path, "Object key sets differ", missing=sorted(baseline.keys() - candidate.keys()),
|
||||||
|
extra=sorted(candidate.keys() - baseline.keys()))
|
||||||
|
|
||||||
|
|
||||||
|
def compare(baseline: dict, candidate: dict) -> dict:
|
||||||
|
# Validate the complete structure before comparing any payload bit patterns.
|
||||||
|
match_keys(baseline, candidate, "/")
|
||||||
|
for key in ("series", "final"):
|
||||||
|
match_keys(baseline[key], candidate[key], pointer(key))
|
||||||
|
for key, values in baseline["series"].items():
|
||||||
|
if len(values) != len(candidate["series"][key]):
|
||||||
|
raise ComparisonError(pointer("series", key), "Array lengths differ", baselineLength=len(values), candidateLength=len(candidate["series"][key]))
|
||||||
|
if len(baseline["finalState"]) != len(candidate["finalState"]):
|
||||||
|
raise ComparisonError("/finalState", "Array lengths differ", baselineLength=len(baseline["finalState"]), candidateLength=len(candidate["finalState"]))
|
||||||
|
metadata_comparisons = 0
|
||||||
|
for key in VALUE_METADATA + COUNT_METADATA:
|
||||||
|
if baseline[key] != candidate[key]:
|
||||||
|
raise ComparisonError(pointer(key), "Solver metadata or counter differs", baseline=baseline[key], candidate=candidate[key])
|
||||||
|
metadata_comparisons += 1
|
||||||
|
for key in FLOAT_METADATA:
|
||||||
|
left, right = bits(baseline[key], pointer(key)), bits(candidate[key], pointer(key))
|
||||||
|
if left != right:
|
||||||
|
raise ComparisonError(pointer(key), "Numeric metadata binary64 bits differ", baselineBitsLE=left.hex(), candidateBitsLE=right.hex())
|
||||||
|
metadata_comparisons += 1
|
||||||
|
# Any future top-level metadata field must also agree unless explicitly timed.
|
||||||
|
known = set(PAYLOAD_KEYS + FLOAT_METADATA + COUNT_METADATA + VALUE_METADATA + TIMING_METADATA)
|
||||||
|
for key in baseline.keys() - known:
|
||||||
|
if baseline[key] != candidate[key]:
|
||||||
|
raise ComparisonError(pointer(key), "Additional metadata differs")
|
||||||
|
metadata_comparisons += 1
|
||||||
|
comparisons = negative_zeroes = positive_zeroes = 0
|
||||||
|
|
||||||
|
def compare_number(left_value: object, right_value: object, path: str) -> None:
|
||||||
|
nonlocal comparisons, negative_zeroes, positive_zeroes
|
||||||
|
left, right = bits(left_value, path), bits(right_value, path)
|
||||||
|
if left != right:
|
||||||
|
raise ComparisonError(path, "Payload binary64 bits differ", baselineBitsLE=left.hex(), candidateBitsLE=right.hex(),
|
||||||
|
baselineValue=repr(left_value), candidateValue=repr(right_value),
|
||||||
|
signedZeroMismatch=left in (POSITIVE_ZERO, NEGATIVE_ZERO) and right in (POSITIVE_ZERO, NEGATIVE_ZERO))
|
||||||
|
comparisons += 1
|
||||||
|
negative_zeroes += left == NEGATIVE_ZERO
|
||||||
|
positive_zeroes += left == POSITIVE_ZERO
|
||||||
|
|
||||||
|
# Look up columns by their validated key; JSON object order is immaterial.
|
||||||
|
for key, values in baseline["series"].items():
|
||||||
|
for index, (left, right) in enumerate(zip(values, candidate["series"][key], strict=True)):
|
||||||
|
compare_number(left, right, pointer("series", key, index))
|
||||||
|
for key, value in baseline["final"].items():
|
||||||
|
compare_number(value, candidate["final"][key], pointer("final", key))
|
||||||
|
for index, (left, right) in enumerate(zip(baseline["finalState"], candidate["finalState"], strict=True)):
|
||||||
|
compare_number(left, right, pointer("finalState", index))
|
||||||
|
return {"passed": True, "comparisons": comparisons, "metadataComparisons": metadata_comparisons,
|
||||||
|
"negativeZeroComparisons": negative_zeroes, "positiveZeroComparisons": positive_zeroes,
|
||||||
|
"allPayloadBinary64BitsEqual": True}
|
||||||
|
|
||||||
|
|
||||||
|
def main() -> int:
|
||||||
|
parser = argparse.ArgumentParser(description=__doc__)
|
||||||
|
parser.add_argument("--baseline", required=True, type=Path)
|
||||||
|
parser.add_argument("--candidate", required=True, action="append", type=Path)
|
||||||
|
parser.add_argument("--output", required=True, type=Path)
|
||||||
|
args = parser.parse_args()
|
||||||
|
if args.output.resolve() in {args.baseline.resolve(), *(path.resolve() for path in args.candidate)}:
|
||||||
|
parser.error("--output must differ from every input file")
|
||||||
|
report = {"version": 1, "baseline": str(args.baseline.resolve()), "candidates": [], "allPassed": False,
|
||||||
|
"comparisonContract": "Exact finite binary64 payload bits, including signed zero; complete structure plus solver metadata/counters. JSON object ordering is ignored. solveSeconds and solveCpuSeconds are excluded. Parsing and comparison are diagnostic work outside benchmark timing."}
|
||||||
|
try:
|
||||||
|
baseline, baseline_hash = read_result(args.baseline)
|
||||||
|
report["baselineSha256"] = baseline_hash
|
||||||
|
report["baselineStatistics"] = validate_result(baseline)
|
||||||
|
for path in args.candidate:
|
||||||
|
item = {"path": str(path.resolve()), "passed": False}
|
||||||
|
try:
|
||||||
|
candidate, candidate_hash = read_result(path)
|
||||||
|
item["sha256"] = candidate_hash
|
||||||
|
item["statistics"] = validate_result(candidate)
|
||||||
|
item.update(compare(baseline, candidate))
|
||||||
|
except ComparisonError as error:
|
||||||
|
item["error"] = error.detail
|
||||||
|
except (OSError, ValueError, TypeError) as error:
|
||||||
|
item["error"] = {"reason": str(error), "type": type(error).__name__}
|
||||||
|
report["candidates"].append(item)
|
||||||
|
report["allPassed"] = all(item["passed"] for item in report["candidates"])
|
||||||
|
except ComparisonError as error:
|
||||||
|
report["baselineError"] = error.detail
|
||||||
|
except (OSError, ValueError, TypeError) as error:
|
||||||
|
report["baselineError"] = {"reason": str(error), "type": type(error).__name__}
|
||||||
|
args.output.parent.mkdir(parents=True, exist_ok=True)
|
||||||
|
args.output.write_text(json.dumps(report, ensure_ascii=False, indent=2, allow_nan=False) + "\n", encoding="utf-8")
|
||||||
|
print(json.dumps({"allPassed": report["allPassed"], "candidates": len(report["candidates"]), "output": str(args.output.resolve())}))
|
||||||
|
return 0 if report["allPassed"] else 1
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
sys.exit(main())
|
||||||
@@ -0,0 +1,314 @@
|
|||||||
|
"""Isolated, Linux/GCC-only CVODE cost diagnostic; never a production benchmark.
|
||||||
|
|
||||||
|
Example (prepare only by default; --run builds and executes serially)::
|
||||||
|
|
||||||
|
.venv/bin/python tests/manual/native_compute_profile.py \
|
||||||
|
--cache-dir test/.../cache/BUILD_KEY --request-stages test/.../stages.json \
|
||||||
|
--output-dir test/native-compute-profile --run --warmups 1 --repeats 3
|
||||||
|
|
||||||
|
The cached executable is the unmodified control. Only a private native source
|
||||||
|
copy receives wall-clock scopes and sparse CVODE counter reads. The generated
|
||||||
|
model and numerical expressions, compiler FP flags, solver and libraries stay
|
||||||
|
unchanged. Full output/state/counter equality is checked outside run timing.
|
||||||
|
Inclusive durations are nested: only exclusiveSeconds may be added. Clock and
|
||||||
|
bookkeeping overhead remain in measured totals; compare against the control.
|
||||||
|
No property/pipe/libc allocation is inferred from this outer-only diagnostic.
|
||||||
|
"""
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
from hashlib import sha256
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
from pathlib import Path
|
||||||
|
import re
|
||||||
|
import shutil
|
||||||
|
import statistics
|
||||||
|
import subprocess
|
||||||
|
import sys
|
||||||
|
import time
|
||||||
|
|
||||||
|
ROOT = Path(__file__).resolve().parents[2]
|
||||||
|
sys.path.insert(0, str(ROOT))
|
||||||
|
from app.simulation.native_codegen.build import LIBRARIES, toolchain
|
||||||
|
|
||||||
|
CATEGORIES = ["integration", "cvode_step", "rhs", "poll", "accept", "append", "dense_output", "dense_setup", "dense_solve"]
|
||||||
|
COUNTERS = ["rhs", "linear_rhs", "nonlinear_iterations", "nonlinear_failures"]
|
||||||
|
PROFILE_HEADER = r'''
|
||||||
|
#ifndef NATIVE_COMPUTE_PROFILE_H
|
||||||
|
#define NATIVE_COMPUTE_PROFILE_H
|
||||||
|
#include <stddef.h>
|
||||||
|
enum { @CATEGORIES@, PROFILE_CATEGORY_COUNT };
|
||||||
|
typedef struct ProfileScope { double start, children; int id, domain, active; struct ProfileScope *parent; } ProfileScope;
|
||||||
|
ProfileScope profile_begin(int id);
|
||||||
|
void profile_link(ProfileScope *scope);
|
||||||
|
void profile_end(ProfileScope *scope);
|
||||||
|
void profile_counter(int slot, int status, long int value);
|
||||||
|
void profile_counter_segment(void);
|
||||||
|
void profile_dump(void);
|
||||||
|
#define PROFILE_SCOPE(id) ProfileScope profile_scope __attribute__((cleanup(profile_end)))=profile_begin(id); profile_link(&profile_scope)
|
||||||
|
#endif
|
||||||
|
'''
|
||||||
|
PROFILE_SOURCE = r'''
|
||||||
|
#include "compute_profile.h"
|
||||||
|
#include <stdio.h>
|
||||||
|
#include <stdlib.h>
|
||||||
|
#include <time.h>
|
||||||
|
#include <stdint.h>
|
||||||
|
typedef struct { unsigned long long count; double inclusive, exclusive; } ProfileTotal;
|
||||||
|
static ProfileTotal totals[2][PROFILE_CATEGORY_COUNT];
|
||||||
|
static ProfileScope *parent;
|
||||||
|
static unsigned long long counters[4], segments, counter_errors;
|
||||||
|
static double now(void) { struct timespec t; clock_gettime(CLOCK_MONOTONIC,&t); return t.tv_sec+t.tv_nsec*1e-9; }
|
||||||
|
ProfileScope profile_begin(int id) {
|
||||||
|
ProfileScope s={0}; s.id=id; s.domain=(id==PROFILE_INTEGRATION || (parent && parent->domain));
|
||||||
|
s.parent=parent; s.active=1; s.start=now(); return s;
|
||||||
|
}
|
||||||
|
void profile_link(ProfileScope *s) { parent=s; }
|
||||||
|
void profile_end(ProfileScope *s) {
|
||||||
|
if(!s->active) return;
|
||||||
|
double elapsed=now()-s->start;
|
||||||
|
if(parent!=s) { fputs("Invalid profile scope nesting\n",stderr); exit(74); }
|
||||||
|
ProfileTotal *t=&totals[s->domain][s->id];
|
||||||
|
t->count++; t->inclusive+=elapsed; t->exclusive+=elapsed-s->children;
|
||||||
|
parent=s->parent;
|
||||||
|
if(parent) parent->children+=elapsed;
|
||||||
|
s->active=0;
|
||||||
|
}
|
||||||
|
void profile_counter(int slot,int status,long int value) {
|
||||||
|
if(status || value<0) counter_errors++; else counters[slot]+=(unsigned long long)value;
|
||||||
|
}
|
||||||
|
void profile_counter_segment(void) { segments++; }
|
||||||
|
void profile_dump(void) {
|
||||||
|
const char *path=getenv("NATIVE_COMPUTE_PROFILE"); if(!path)return;
|
||||||
|
FILE *f=fopen(path,"wb"); if(!f){perror(path);exit(73);}
|
||||||
|
const char *names[]={@NAMES@};
|
||||||
|
fprintf(f,"{\"version\":1,\"counterSegments\":%llu,\"counterErrors\":%llu,\"cvodeCounters\":{",segments,counter_errors);
|
||||||
|
const char *counter_names[]={"rhs","linear_rhs","nonlinear_iterations","nonlinear_failures"};
|
||||||
|
for(int i=0;i<4;i++)fprintf(f,"%s\"%s\":%llu",i?",":"",counter_names[i],counters[i]);
|
||||||
|
fprintf(f,"},\"scopes\":{");
|
||||||
|
for(int d=0;d<2;d++) {
|
||||||
|
fprintf(f,"%s\"%s\":{",d?",":"",d?"integration":"outsideIntegration");
|
||||||
|
for(int i=0;i<PROFILE_CATEGORY_COUNT;i++) {
|
||||||
|
ProfileTotal *t=&totals[d][i];
|
||||||
|
fprintf(f,"%s\"%s\":{\"calls\":%llu,\"inclusiveSeconds\":%.17g,\"exclusiveSeconds\":%.17g}",
|
||||||
|
i?",":"",names[i],t->count,t->inclusive,t->exclusive);
|
||||||
|
}
|
||||||
|
fputc('}',f);
|
||||||
|
}
|
||||||
|
fprintf(f,"}}\n"); int ok=!ferror(f); if(fclose(f))ok=0; if(!ok)exit(73);
|
||||||
|
}
|
||||||
|
'''
|
||||||
|
LINEAR_WRAPPERS = r'''
|
||||||
|
/* Preserve the exact original Dense ops; only the call boundary is timed. */
|
||||||
|
static int (*profile_original_setup)(SUNLinearSolver,SUNMatrix);
|
||||||
|
static int (*profile_original_solve)(SUNLinearSolver,SUNMatrix,N_Vector,N_Vector,sunrealtype);
|
||||||
|
static int profile_dense_setup(SUNLinearSolver linear,SUNMatrix matrix) {
|
||||||
|
PROFILE_SCOPE(PROFILE_DENSE_SETUP);
|
||||||
|
return profile_original_setup(linear,matrix);
|
||||||
|
}
|
||||||
|
static int profile_dense_solve(SUNLinearSolver linear,SUNMatrix matrix,N_Vector x,N_Vector b,sunrealtype tolerance) {
|
||||||
|
PROFILE_SCOPE(PROFILE_DENSE_SOLVE);
|
||||||
|
return profile_original_solve(linear,matrix,x,b,tolerance);
|
||||||
|
}
|
||||||
|
static int profile_cvode(void *solver,sunrealtype end,N_Vector y,sunrealtype *next,int task) {
|
||||||
|
PROFILE_SCOPE(PROFILE_CVODE_STEP);
|
||||||
|
return CVode(solver,end,y,next,task);
|
||||||
|
}
|
||||||
|
'''
|
||||||
|
|
||||||
|
|
||||||
|
def digest(path: Path) -> str:
|
||||||
|
return sha256(path.read_bytes()).hexdigest()
|
||||||
|
|
||||||
|
|
||||||
|
def write_json(path: Path, value: object) -> None:
|
||||||
|
path.write_text(json.dumps(value, indent=2, ensure_ascii=False) + "\n", encoding="utf-8")
|
||||||
|
|
||||||
|
|
||||||
|
def replace_once(text: str, old: str, new: str) -> str:
|
||||||
|
if text.count(old) != 1:
|
||||||
|
raise RuntimeError(f"Source anchor count changed: {old!r}")
|
||||||
|
return text.replace(old, new)
|
||||||
|
|
||||||
|
|
||||||
|
def scope_function(text: str, function: str, category: str) -> str:
|
||||||
|
pattern = rf"(?m)^[A-Za-z_][A-Za-z0-9_ \t*]*\b{re.escape(function)}\s*\([^;{{}}]*\)\s*\{{"
|
||||||
|
matches = list(re.finditer(pattern, text))
|
||||||
|
if len(matches) != 1:
|
||||||
|
raise RuntimeError(f"Cannot identify unique function {function}")
|
||||||
|
pos = matches[0].end()
|
||||||
|
return text[:pos] + f"\n PROFILE_SCOPE(PROFILE_{category.upper()});" + text[pos:]
|
||||||
|
|
||||||
|
|
||||||
|
def instrument(native: Path) -> None:
|
||||||
|
(native / "include/compute_profile.h").write_text(PROFILE_HEADER.replace("@CATEGORIES@", ", ".join("PROFILE_" + c.upper() for c in CATEGORIES)))
|
||||||
|
(native / "runtime/compute_profile.c").write_text(PROFILE_SOURCE.replace("@NAMES@", ",".join(json.dumps(c) for c in CATEGORIES)))
|
||||||
|
for filename, functions in {
|
||||||
|
"common.c": {"native_rhs": "rhs", "native_poll": "poll", "native_accept": "accept", "native_append": "append"},
|
||||||
|
"cvode_solver.c": {"native_bdf": "integration", "cv_dense": "dense_output"},
|
||||||
|
"rk45.c": {"native_rk45": "integration"},
|
||||||
|
}.items():
|
||||||
|
path = native / "runtime" / filename
|
||||||
|
text = '#include "compute_profile.h"\n' + path.read_text()
|
||||||
|
for function, category in functions.items():
|
||||||
|
text = scope_function(text, function, category)
|
||||||
|
if filename == "cvode_solver.c":
|
||||||
|
text = replace_once(text, "typedef struct { void *solver;", LINEAR_WRAPPERS + "\ntypedef struct { void *solver;")
|
||||||
|
text = replace_once(text, " if (!linear) goto cleanup;", " if (!linear) goto cleanup;\n profile_original_setup=linear->ops->setup; profile_original_solve=linear->ops->solve;\n linear->ops->setup=profile_dense_setup; linear->ops->solve=profile_dense_solve;")
|
||||||
|
text = replace_once(text, "int flag=CVode(solver,end,y,&next,CV_ONE_STEP);", "int flag=profile_cvode(solver,end,y,&next,CV_ONE_STEP);")
|
||||||
|
extra = "\n profile_counter_segment();\n"
|
||||||
|
for slot, api in enumerate(("CVodeGetNumRhsEvals", "CVodeGetNumLinRhsEvals", "CVodeGetNumNonlinSolvIters", "CVodeGetNumNonlinSolvConvFails")):
|
||||||
|
extra += f" value=0; int profile_status_{slot}={api}(solver,&value); profile_counter({slot},profile_status_{slot},value);\n"
|
||||||
|
text = replace_once(text, "CVodeGetNumLinSolvSetups(solver,&value); r->nlu+=(unsigned long)value;", "CVodeGetNumLinSolvSetups(solver,&value); r->nlu+=(unsigned long)value;" + extra)
|
||||||
|
path.write_text(text)
|
||||||
|
path = native / "runtime/main.c"
|
||||||
|
text = '#include "compute_profile.h"\n' + path.read_text()
|
||||||
|
text = replace_once(text, " native_run_free(&r); return code;", " native_run_free(&r); profile_dump(); return code;")
|
||||||
|
path.write_text(text)
|
||||||
|
|
||||||
|
|
||||||
|
def runtime_arguments(stages: Path | None) -> list[str]:
|
||||||
|
if stages:
|
||||||
|
original = json.loads(stages.read_text())["process"]["command"]
|
||||||
|
args = original[1:]
|
||||||
|
else:
|
||||||
|
args = ["--method", "BDF", "--start", "0", "--stop", "10", "--sample-step", ".01", "--max-step", "1e30", "--rtol", "1e-8", "--timeout", "300"]
|
||||||
|
safe, index = [], 0
|
||||||
|
value_options = {"--method", "--start", "--stop", "--sample-step", "--max-step", "--rtol", "--timeout"}
|
||||||
|
while index < len(args):
|
||||||
|
key = args[index]
|
||||||
|
if key == "--solve-only":
|
||||||
|
safe.append(key); index += 1; continue
|
||||||
|
if key not in value_options | {"--output", "--result-index", "--cancel-file"} or index + 1 >= len(args):
|
||||||
|
raise RuntimeError(f"Unsupported replay argument: {key}")
|
||||||
|
if key in value_options:
|
||||||
|
safe.extend(args[index:index + 2])
|
||||||
|
index += 2
|
||||||
|
return safe
|
||||||
|
|
||||||
|
|
||||||
|
def prepare(args: argparse.Namespace) -> dict:
|
||||||
|
cache, output = args.cache_dir.resolve(), args.output_dir.resolve()
|
||||||
|
if not output.is_relative_to(ROOT / "test"):
|
||||||
|
raise RuntimeError("Diagnostic output must be in the repository's ignored test/ directory")
|
||||||
|
manifest = json.loads((cache / "manifest.json").read_text())
|
||||||
|
for name in ("model", "model.c", "model.h"):
|
||||||
|
if digest(cache / name) != manifest["artifacts"][name]:
|
||||||
|
raise RuntimeError(f"Cache artifact integrity failure: {name}")
|
||||||
|
# Reject numerical/runtime drift; a sparse timing-only cached main is allowed
|
||||||
|
# because control and our current writer share the numeric model contract.
|
||||||
|
differences = []
|
||||||
|
for name, expected in manifest["sourceHashes"].items():
|
||||||
|
relative = name.split("native/", 1)[-1]
|
||||||
|
current = ROOT / "native" / relative
|
||||||
|
if digest(current) != expected:
|
||||||
|
differences.append(relative)
|
||||||
|
if any(name != "runtime/main.c" for name in differences):
|
||||||
|
raise RuntimeError(f"Cached numerical sources differ from current sources: {differences}")
|
||||||
|
compiler, sundials, compiler_version = toolchain()
|
||||||
|
if not sys.platform.startswith("linux"):
|
||||||
|
raise RuntimeError("This test-only cleanup-scope profiler requires Linux/GCC")
|
||||||
|
libraries = [sundials / "lib" / f"libsundials_{name}.a" for name in LIBRARIES]
|
||||||
|
for name, expected in manifest["dependencyHashes"].items():
|
||||||
|
path = sundials / ("include" if "/" in name else "lib") / name
|
||||||
|
if digest(path) != expected:
|
||||||
|
raise RuntimeError(f"SUNDIALS dependency changed: {name}")
|
||||||
|
if compiler_version != manifest["compiler"]:
|
||||||
|
raise RuntimeError("Use the cached model's compiler version for this comparison")
|
||||||
|
output.mkdir(parents=True, exist_ok=True)
|
||||||
|
native = output / "native"
|
||||||
|
shutil.copytree(ROOT / "native", native, dirs_exist_ok=True)
|
||||||
|
for name in ("model.c", "model.h", "manifest.json"):
|
||||||
|
shutil.copy2(cache / name, output / name)
|
||||||
|
instrument(native)
|
||||||
|
command = [compiler, *manifest["compilerFlags"], "-I", str(output), "-I", str(native / "include"), "-I", str(sundials / "include"), str(output / "model.c"), *map(str, sorted(native.rglob("*.c"))), "-Wl,--start-group", *map(str, libraries), "-Wl,--end-group", "-lm", "-o", str(output / "profiled-model")]
|
||||||
|
prepared = {"controlExecutable": str(cache / "model"), "profiledExecutable": str(output / "profiled-model"), "buildCommand": command, "runtimeArguments": runtime_arguments(args.request_stages), "cacheDir": str(cache), "cachedMainDifference": differences, "stateCount": len(manifest["stateKeys"]), "modelSha256": digest(output / "model.c"), "compiler": compiler_version, "warmups": args.warmups, "repeats": args.repeats, "scope": "Inclusive times overlap. Exclusive scope times partition integration including instrumentation overhead. No component or libc sub-cost inference.", "sourceHashes": {str(p.relative_to(output)): digest(p) for p in sorted(native.rglob("*")) if p.is_file()}}
|
||||||
|
write_json(output / "prepared.json", prepared)
|
||||||
|
return prepared
|
||||||
|
|
||||||
|
|
||||||
|
def parity_payload(result: dict) -> dict:
|
||||||
|
return {key: value for key, value in result.items() if key not in {"solveSeconds", "solveCpuSeconds"}}
|
||||||
|
|
||||||
|
|
||||||
|
def execute(args: argparse.Namespace, prepared: dict) -> None:
|
||||||
|
output = args.output_dir.resolve()
|
||||||
|
build = subprocess.run(prepared["buildCommand"], capture_output=True, text=True, timeout=180)
|
||||||
|
(output / "build.log").write_text(build.stdout + build.stderr)
|
||||||
|
if build.returncode:
|
||||||
|
raise RuntimeError(f"Compilation failed: {output / 'build.log'}")
|
||||||
|
baseline = None
|
||||||
|
rows = []
|
||||||
|
# Serial paired control/profile runs; warmups excluded from overhead figures.
|
||||||
|
for index in range(-args.warmups, args.repeats):
|
||||||
|
label = f"warmup-{index + args.warmups + 1}" if index < 0 else f"run-{index + 1}"
|
||||||
|
for variant in ("control", "profiled"):
|
||||||
|
run = output / variant / label
|
||||||
|
run.mkdir(parents=True, exist_ok=True)
|
||||||
|
result_path, profile_path = run / "result.json", run / "profile.json"
|
||||||
|
for stale in (result_path, profile_path, run / "cancel.request"):
|
||||||
|
stale.unlink(missing_ok=True)
|
||||||
|
command = [prepared[f"{variant}Executable"], *prepared["runtimeArguments"], "--output", str(result_path), "--result-index", str(run / "result-index.json"), "--cancel-file", str(run / "cancel.request")]
|
||||||
|
environment = dict(os.environ)
|
||||||
|
environment.pop("NATIVE_COMPUTE_PROFILE", None)
|
||||||
|
# Disable independent sparse-stage profilers in a cached control.
|
||||||
|
environment.pop("NATIVE_STAGE_PROFILE", None)
|
||||||
|
if variant == "profiled":
|
||||||
|
environment["NATIVE_COMPUTE_PROFILE"] = str(profile_path)
|
||||||
|
started = time.perf_counter()
|
||||||
|
process = subprocess.run(command, env=environment, capture_output=True, timeout=args.process_timeout)
|
||||||
|
wall = time.perf_counter() - started
|
||||||
|
(run / "stdout.log").write_bytes(process.stdout)
|
||||||
|
(run / "stderr.log").write_bytes(process.stderr)
|
||||||
|
if process.returncode:
|
||||||
|
raise RuntimeError(f"{variant}/{label} exit {process.returncode}; see stderr.log")
|
||||||
|
result = json.loads(result_path.read_bytes())
|
||||||
|
if result.get("success") is not True:
|
||||||
|
raise RuntimeError(f"{variant}/{label} did not complete")
|
||||||
|
comparable = parity_payload(result)
|
||||||
|
if baseline is None:
|
||||||
|
baseline = comparable
|
||||||
|
if comparable != baseline:
|
||||||
|
mismatches = [k for k in baseline.keys() | comparable.keys() if baseline.get(k) != comparable.get(k)]
|
||||||
|
write_json(run / "parity-failure.json", mismatches)
|
||||||
|
raise RuntimeError(f"Numerical/counter parity failed: {mismatches}")
|
||||||
|
row = {"variant": variant, "run": label, "warmup": index < 0, "processWallSeconds": wall, "solveSeconds": result["solveSeconds"], "solveCpuSeconds": result["solveCpuSeconds"], "fullParity": True, "resultBytes": result_path.stat().st_size, "nfev": result["nfev"], "njev": result["njev"], "nlu": result["nlu"], "acceptedSteps": result["acceptedSteps"], "solverStarts": result["solverStarts"]}
|
||||||
|
if variant == "profiled":
|
||||||
|
profile = json.loads(profile_path.read_text())
|
||||||
|
counters = profile["cvodeCounters"]
|
||||||
|
checks = {"counterReadsSucceeded": profile["counterErrors"] == 0, "rhsCountMatches": counters["rhs"] + counters["linear_rhs"] == result["nfev"], "linearRhsEqualsJacobianCountTimesStates": counters["linear_rhs"] == result["njev"] * prepared["stateCount"], "rhsClockCountMatches": profile["scopes"]["integration"]["rhs"]["calls"] == result["nfev"]}
|
||||||
|
row["profile"] = profile
|
||||||
|
row["counterChecks"] = checks
|
||||||
|
if not checks["counterReadsSucceeded"] or not checks["rhsClockCountMatches"] or (result["method"] == "BDF" and not checks["rhsCountMatches"]):
|
||||||
|
write_json(run / "counter-failure.json", row)
|
||||||
|
raise RuntimeError(f"Unexpected profiling counters: {checks}")
|
||||||
|
rows.append(row)
|
||||||
|
write_json(run / "run.json", row)
|
||||||
|
print(f"{variant}/{label}: solve={row['solveSeconds']:.6f}s wall={wall:.6f}s parity=true", flush=True)
|
||||||
|
medians = {variant: {key: statistics.median(row[key] for row in rows if row["variant"] == variant and not row["warmup"]) for key in ("processWallSeconds", "solveSeconds", "solveCpuSeconds")} for variant in ("control", "profiled")}
|
||||||
|
overhead = {key: medians["profiled"][key] / medians["control"][key] - 1 for key in medians["control"]}
|
||||||
|
write_json(output / "summary.json", {"prepared": prepared, "runs": rows, "medians": medians, "instrumentationOverheadFraction": overhead, "allFullParity": True, "interpretation": "CVODE linear_rhs counts finite-difference RHS calls independently of model nfev. Its multiplication by stateCount is checked, not assumed. RHS time includes all model work; no Jacobian-specific RHS time is inferred. cvode_step.exclusiveSeconds excludes nested RHS, polling and Dense ops; integration.exclusiveSeconds is outer setup/loop/cleanup work. Dense setup covers the original factorization operation; dense_solve covers the original triangular solve operation. Clock/bookkeeping costs remain in all measured totals. Production control may retain its pre-existing sparse main clocks; cachedMainDifference records this."})
|
||||||
|
print(f"Summary: {output / 'summary.json'}", flush=True)
|
||||||
|
|
||||||
|
|
||||||
|
def main() -> None:
|
||||||
|
parser = argparse.ArgumentParser(description=__doc__)
|
||||||
|
parser.add_argument("--cache-dir", required=True, type=Path)
|
||||||
|
parser.add_argument("--request-stages", type=Path)
|
||||||
|
parser.add_argument("--output-dir", required=True, type=Path)
|
||||||
|
parser.add_argument("--run", action="store_true", help="Build and run serial warmups/repeats; default only prepares")
|
||||||
|
parser.add_argument("--warmups", type=int, default=1)
|
||||||
|
parser.add_argument("--repeats", type=int, default=3)
|
||||||
|
parser.add_argument("--process-timeout", type=float, default=360)
|
||||||
|
args = parser.parse_args()
|
||||||
|
if args.warmups < 0 or args.repeats < 1:
|
||||||
|
parser.error("warmups must be nonnegative and repeats positive")
|
||||||
|
prepared = prepare(args)
|
||||||
|
print(f"Prepared: {args.output_dir.resolve() / 'prepared.json'}", flush=True)
|
||||||
|
if args.run:
|
||||||
|
execute(args, prepared)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
@@ -0,0 +1,440 @@
|
|||||||
|
"""Instrument isolated pipe solvers; these runs are diagnostics, never benchmarks.
|
||||||
|
|
||||||
|
Prepare: python tests/manual/profile_pipe_iterations.py --output-dir test/pipe-profile --prepare-only
|
||||||
|
Run the prepared diagnostic programs: use the same command without --prepare-only.
|
||||||
|
All mutations except this test helper stay below the ignored output directory.
|
||||||
|
The replay contains EVERY resistance-law input from the guarded solver's RHS
|
||||||
|
trajectory, including its scalar low-Re analytic branch. Cache hits, zero dp,
|
||||||
|
and the separate PNL00R analytic law are counted but do not enter this replay.
|
||||||
|
"""
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
from hashlib import sha256
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
from pathlib import Path
|
||||||
|
import subprocess
|
||||||
|
import sys
|
||||||
|
|
||||||
|
ROOT = Path(__file__).resolve().parents[2]
|
||||||
|
sys.path.insert(0, str(ROOT))
|
||||||
|
from app.main import compile_system_xml_network
|
||||||
|
from app.simulation.backends import simulation_config
|
||||||
|
from app.simulation.native_codegen import build as builder
|
||||||
|
from app.simulation.native_codegen.compiler import compile_native_program
|
||||||
|
from app.simulation.native_codegen.input import load_input
|
||||||
|
|
||||||
|
VARIANTS = ('guarded-newton', 'previous-newton', 'fixed-point')
|
||||||
|
FIELDS = '''cache_requests cache_hits cache_misses flow_calls zero_pressure_calls
|
||||||
|
pnl00r_analytic_calls resistance_calls scalar_analytic_calls iterative_calls
|
||||||
|
iterations_total iterations_max reached_last_iteration exhausted_limit
|
||||||
|
algorithm_converged failed_before_iteration finite_returns nonfinite_returns
|
||||||
|
bisections residual_pass residual_fail residual_nonfinite
|
||||||
|
residual_fail_after_algorithm_converged residual_pass_after_limit
|
||||||
|
invalid_inputs upper_bracket_evaluations upper_bracket_exhausted
|
||||||
|
bisection_after_poor_progress bisection_invalid_or_outside float_stagnation
|
||||||
|
'''.split()
|
||||||
|
|
||||||
|
PROFILE_PREFIX = r'''
|
||||||
|
/* Test-only instrumentation, injected in an isolated source tree. */
|
||||||
|
#include <stdio.h>
|
||||||
|
#include <stdlib.h>
|
||||||
|
#include <string.h>
|
||||||
|
#include <stdint.h>
|
||||||
|
#define PROFILE_VARIANT "@VARIANT@"
|
||||||
|
#define PROFILE_FIXED @FIXED@
|
||||||
|
#define PROFILE_LIMIT(kind) @LIMIT@
|
||||||
|
#define PROFILE_FIELDS(X) @FIELDS@
|
||||||
|
typedef struct {
|
||||||
|
#define PROFILE_DECLARE(name) unsigned long long name;
|
||||||
|
PROFILE_FIELDS(PROFILE_DECLARE)
|
||||||
|
#undef PROFILE_DECLARE
|
||||||
|
unsigned long long histogram[129];
|
||||||
|
double maximum_relative_residual;
|
||||||
|
} PipeProfile;
|
||||||
|
static PipeProfile profile_stats[2][4];
|
||||||
|
static unsigned long long profile_rhs_calls,profile_capture_records;
|
||||||
|
static int profile_in_rhs,profile_kind,profile_last_exhausted;
|
||||||
|
static FILE *profile_capture;
|
||||||
|
static PipeProfile *profile_bucket(int kind) {
|
||||||
|
if(kind<0 || kind>3){fprintf(stderr,"Unexpected pipe kind %d\n",kind);exit(71);}
|
||||||
|
return &profile_stats[profile_in_rhs?0:1][kind];
|
||||||
|
}
|
||||||
|
void pipe_profile_rhs_enter(void){profile_in_rhs=1;profile_rhs_calls++;}
|
||||||
|
void pipe_profile_rhs_leave(void){profile_in_rhs=0;}
|
||||||
|
static void profile_add(PipeProfile *total,const PipeProfile *value) {
|
||||||
|
#define PROFILE_ADD(name) total->name+=value->name;
|
||||||
|
PROFILE_FIELDS(PROFILE_ADD)
|
||||||
|
#undef PROFILE_ADD
|
||||||
|
if(value->iterations_max>total->iterations_max)total->iterations_max=value->iterations_max;
|
||||||
|
for(int i=0;i<129;i++)total->histogram[i]+=value->histogram[i];
|
||||||
|
if(value->maximum_relative_residual>total->maximum_relative_residual)
|
||||||
|
total->maximum_relative_residual=value->maximum_relative_residual;
|
||||||
|
}
|
||||||
|
static void profile_write_bucket(FILE *f,const PipeProfile *value) {
|
||||||
|
fprintf(f,"{");
|
||||||
|
#define PROFILE_WRITE(name) fprintf(f,"\"" #name "\":%llu,",value->name);
|
||||||
|
PROFILE_FIELDS(PROFILE_WRITE)
|
||||||
|
#undef PROFILE_WRITE
|
||||||
|
fprintf(f,"\"maximum_relative_residual\":%.17g,\"iteration_histogram\":{",value->maximum_relative_residual);
|
||||||
|
int comma=0;
|
||||||
|
for(int i=0;i<129;i++)if(value->histogram[i]) {
|
||||||
|
fprintf(f,"%s\"%d\":%llu",comma?",":"",i,value->histogram[i]);comma=1;
|
||||||
|
}
|
||||||
|
fprintf(f,"}}");
|
||||||
|
}
|
||||||
|
static void profile_dump(void) {
|
||||||
|
const char *path=getenv("PIPE_PROFILE_JSON");
|
||||||
|
if(profile_capture){if(fclose(profile_capture))exit(73);profile_capture=NULL;}
|
||||||
|
if(!path)return;
|
||||||
|
FILE *f=fopen(path,"wb");if(!f){perror(path);exit(73);}
|
||||||
|
fprintf(f,"{\"variant\":\"%s\",\"rhs_calls\":%llu,\"capture_records\":%llu,",PROFILE_VARIANT,profile_rhs_calls,profile_capture_records);
|
||||||
|
for(int scope=0;scope<2;scope++) {
|
||||||
|
PipeProfile total={0};
|
||||||
|
for(int kind=0;kind<4;kind++)profile_add(&total,&profile_stats[scope][kind]);
|
||||||
|
/* iterations_max is a maximum, unlike the additive counters. */
|
||||||
|
total.iterations_max=0;
|
||||||
|
for(int kind=0;kind<4;kind++)if(profile_stats[scope][kind].iterations_max>total.iterations_max)
|
||||||
|
total.iterations_max=profile_stats[scope][kind].iterations_max;
|
||||||
|
fprintf(f,"%s\"%s\":{\"total\":",scope?",":"",scope?"non_rhs":"rhs");
|
||||||
|
profile_write_bucket(f,&total);fprintf(f,",\"by_kind\":{");
|
||||||
|
for(int kind=0;kind<4;kind++) {
|
||||||
|
fprintf(f,"%s\"%d\":",kind?",":"",kind);profile_write_bucket(f,&profile_stats[scope][kind]);
|
||||||
|
}
|
||||||
|
fprintf(f,"}}");
|
||||||
|
}
|
||||||
|
fprintf(f,"}\n");if(fclose(f))exit(73);
|
||||||
|
}
|
||||||
|
void pipe_profile_install(void) {
|
||||||
|
const char *path=getenv("PIPE_PROFILE_CAPTURE");
|
||||||
|
if(path){profile_capture=fopen(path,"wb");if(!profile_capture){perror(path);exit(73);}}
|
||||||
|
if(atexit(profile_dump)){fprintf(stderr,"Cannot register profile writer\n");exit(73);}
|
||||||
|
}
|
||||||
|
static void profile_save_input(double base,double d,double length,double rr,double den,int kind) {
|
||||||
|
if(profile_in_rhs && profile_capture) {
|
||||||
|
/* Six IEEE doubles, native endian; no sampling or deduplication. */
|
||||||
|
double input[]={base,d,length,rr,den,(double)kind};
|
||||||
|
if(fwrite(input,sizeof(input),1,profile_capture)!=1){perror("capture");exit(73);}
|
||||||
|
profile_capture_records++;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
'''
|
||||||
|
|
||||||
|
PROFILE_SOLVE = r'''
|
||||||
|
/* Independently factored Darcy law: no production slope or solver status is
|
||||||
|
consulted. Long double reduces rounding noise in the returned-q residual. */
|
||||||
|
static long double profile_reference_friction(long double re,long double rr) {
|
||||||
|
if(!(re>0))return NAN;
|
||||||
|
long double laminar=64/re;
|
||||||
|
if(re<=89.96829989L)return laminar;
|
||||||
|
long double smooth=powl(-1.8L*log10l(6.9L/re),-2),turbulent=smooth;
|
||||||
|
if(rr>0) {
|
||||||
|
long double fully_rough=powl(-2*log10l(rr/3.7L),-2);
|
||||||
|
long double weight=1/(1+powl(180/(re*rr),2));
|
||||||
|
turbulent=(1-weight)*smooth+weight*fully_rough;
|
||||||
|
}
|
||||||
|
long double blend=powl((re-89.96829989L)/2741.96700831L,8.37293695L);
|
||||||
|
return (laminar+blend*turbulent)/(1+blend);
|
||||||
|
}
|
||||||
|
static void profile_returned_residual(PipeProfile *s,double q,double den,double K,double rr,
|
||||||
|
int converged,int exhausted) {
|
||||||
|
if(!isfinite(q)){s->nonfinite_returns++;s->residual_nonfinite++;return;}
|
||||||
|
s->finite_returns++;
|
||||||
|
long double re=fabsl((long double)q/(den/4)),residual;
|
||||||
|
if(K==0 && q==0)residual=0;
|
||||||
|
else residual=fabsl(re*re*profile_reference_friction(re,rr)/K-1);
|
||||||
|
if(!isfinite(residual)){s->residual_nonfinite++;return;}
|
||||||
|
if(residual>s->maximum_relative_residual)s->maximum_relative_residual=(double)residual;
|
||||||
|
if(residual<=1e-9L){s->residual_pass++;if(exhausted)s->residual_pass_after_limit++;}
|
||||||
|
else {s->residual_fail++;if(converged)s->residual_fail_after_algorithm_converged++;}
|
||||||
|
}
|
||||||
|
static double profile_solve(double base,double d,double length,double rr,double den,int kind) {
|
||||||
|
PipeProfile *s=profile_bucket(kind);
|
||||||
|
double K=pow(4*base/den,2)*d/length;
|
||||||
|
profile_save_input(base,d,length,rr,den,kind);
|
||||||
|
s->resistance_calls++;
|
||||||
|
if(!(K>=0 && rr>=0 && den/4>0) || !isfinite(K) || !isfinite(rr) || !isfinite(den/4))s->invalid_inputs++;
|
||||||
|
int iterations=0,converged=0,bisections=0,exhausted=0;
|
||||||
|
double q;
|
||||||
|
profile_kind=kind;profile_last_exhausted=0;
|
||||||
|
#if PROFILE_FIXED
|
||||||
|
double rough_limit=pipe_rough_limit(rr);
|
||||||
|
q=sqrt(d/(length*.02))*base;
|
||||||
|
for(int i=0;i<PROFILE_LIMIT(kind);i++) {
|
||||||
|
iterations=i+1;
|
||||||
|
double next=sqrt(d/(length*pipe_friction_prepared(4*fabs(q)/den,rr,rough_limit)))*base;
|
||||||
|
if(fabs(next-q)<=fmax(1e-12,fabs(q)*1e-9)){q=next;converged=1;break;}
|
||||||
|
q=.5*(q+next);
|
||||||
|
}
|
||||||
|
exhausted=!converged;
|
||||||
|
#else
|
||||||
|
NativePipeSolve status;
|
||||||
|
q=native_pipe_resistance(K,rr,den/4,&status)*den/4;
|
||||||
|
iterations=status.iterations;converged=status.converged;bisections=status.bisections;
|
||||||
|
exhausted=profile_last_exhausted;
|
||||||
|
#endif
|
||||||
|
if(iterations<0 || iterations>128){fprintf(stderr,"Unexpected iteration count\n");exit(71);}
|
||||||
|
s->histogram[iterations]++;s->iterations_total+=(unsigned)iterations;s->bisections+=(unsigned)bisections;
|
||||||
|
if((unsigned)iterations>s->iterations_max)s->iterations_max=(unsigned)iterations;
|
||||||
|
if(iterations)s->iterative_calls++;
|
||||||
|
else if(converged)s->scalar_analytic_calls++;
|
||||||
|
else s->failed_before_iteration++;
|
||||||
|
if(iterations==PROFILE_LIMIT(kind))s->reached_last_iteration++;
|
||||||
|
if(exhausted)s->exhausted_limit++;
|
||||||
|
if(converged)s->algorithm_converged++;
|
||||||
|
profile_returned_residual(s,q,den,K,rr,converged,exhausted);
|
||||||
|
return q;
|
||||||
|
}
|
||||||
|
'''
|
||||||
|
|
||||||
|
REPLAY_MAIN = r'''
|
||||||
|
#include "native/components/kernels.c"
|
||||||
|
int main(int argc,char **argv) {
|
||||||
|
if(argc!=2)return 64;
|
||||||
|
pipe_profile_install();profile_in_rhs=1;
|
||||||
|
FILE *f=fopen(argv[1],"rb");if(!f){perror(argv[1]);return 73;}
|
||||||
|
double input[6];size_t count;
|
||||||
|
while((count=fread(input,1,sizeof(input),f))==sizeof(input)) {
|
||||||
|
profile_solve(input[0],input[1],input[2],input[3],input[4],(int)input[5]);
|
||||||
|
}
|
||||||
|
int failed=count || ferror(f);fclose(f);return failed?74:0;
|
||||||
|
}
|
||||||
|
'''
|
||||||
|
|
||||||
|
|
||||||
|
def replace_once(source: str, old: str, new: str) -> str:
|
||||||
|
if source.count(old) != 1:
|
||||||
|
raise ValueError(f'Expected exactly one audited source fragment: {old[:100]!r}')
|
||||||
|
return source.replace(old, new, 1)
|
||||||
|
|
||||||
|
|
||||||
|
def instrument(source: str, variant: str) -> str:
|
||||||
|
fixed = variant == 'fixed-point'
|
||||||
|
limit = '(kind==0?64:16)' if fixed else '80' if variant == 'previous-newton' else '128'
|
||||||
|
prefix = PROFILE_PREFIX.replace('@VARIANT@', variant).replace('@FIXED@', str(int(fixed)))
|
||||||
|
prefix = prefix.replace('@LIMIT@', limit).replace('@FIELDS@', ' '.join(f'X({name})' for name in FIELDS))
|
||||||
|
source = replace_once(source, '#include <stddef.h>', '#include <stddef.h>\n' + prefix)
|
||||||
|
# Exhaustion is marked at the actual loop fall-through, not inferred from
|
||||||
|
# visiting the last allowed iteration (which can still converge).
|
||||||
|
start = source.index('double native_pipe_resistance(')
|
||||||
|
end = source.index('double native_pipe_flow(', start)
|
||||||
|
resistance = source[start:end]
|
||||||
|
ending = ' return NAN;\n}\n'
|
||||||
|
if not resistance.endswith(ending):
|
||||||
|
raise ValueError('Unexpected resistance function ending')
|
||||||
|
resistance = resistance[:-len(ending)] + ' profile_last_exhausted=1;return NAN;\n}\n'
|
||||||
|
if variant == 'guarded-newton':
|
||||||
|
resistance = replace_once(resistance, ' double value=hi*hi*pipe_friction_prepared(hi,rr,rough);',
|
||||||
|
' profile_bucket(profile_kind)->upper_bracket_evaluations++;\n double value=hi*hi*pipe_friction_prepared(hi,rr,rough);')
|
||||||
|
resistance = replace_once(resistance, ' if(!bracketed)return NAN;',
|
||||||
|
' if(!bracketed){profile_bucket(profile_kind)->upper_bracket_exhausted++;return NAN;}')
|
||||||
|
resistance = replace_once(resistance, ' if(bisect) {',
|
||||||
|
''' if(bisect) {
|
||||||
|
if(previous_newton && fabs(F)>.5*previous_residual)profile_bucket(profile_kind)->bisection_after_poor_progress++;
|
||||||
|
if(!(slope>0) || !isfinite(slope) || !isfinite(next) || next<=lo || next>=hi)
|
||||||
|
profile_bucket(profile_kind)->bisection_invalid_or_outside++;''')
|
||||||
|
resistance = replace_once(resistance, ' if(next<=lo || next>=hi) {',
|
||||||
|
' if(next<=lo || next>=hi) {\n profile_bucket(profile_kind)->float_stagnation++;')
|
||||||
|
elif variant == 'previous-newton':
|
||||||
|
# Count each evaluation of the original loop condition without changing
|
||||||
|
# its short-circuit behaviour or the original upper endpoint arithmetic.
|
||||||
|
resistance = replace_once(resistance, 'i<128 && hi*hi*pipe_friction_prepared(hi,rr,rough)<K',
|
||||||
|
'i<128 && (profile_bucket(profile_kind)->upper_bracket_evaluations++,hi*hi*pipe_friction_prepared(hi,rr,rough)<K)')
|
||||||
|
resistance = replace_once(resistance, 'status->bisections++;',
|
||||||
|
'status->bisections++;profile_bucket(profile_kind)->bisection_invalid_or_outside++;')
|
||||||
|
source = source[:start] + resistance + PROFILE_SOLVE + source[end:]
|
||||||
|
source = replace_once(source, ' if(fabs(p1-p2)<=1e-8) return 0;',
|
||||||
|
''' PipeProfile *profile=profile_bucket(kind);profile->flow_calls++;
|
||||||
|
if(fabs(p1-p2)<=1e-8){profile->zero_pressure_calls++;return 0;}''')
|
||||||
|
source = replace_once(source, ' if(4*lam/den<=1000) return sign*lam;',
|
||||||
|
' if(4*lam/den<=1000){profile->pnl00r_analytic_calls++;return sign*lam;}')
|
||||||
|
source = replace_once(source,
|
||||||
|
' double base=area*p*cm/sqrt(T),K=pow(4*base/den,2)*d/length;\n return sign*native_pipe_resistance(K,rr,den/4,NULL)*den/4;',
|
||||||
|
' double base=area*p*cm/sqrt(T);\n return sign*profile_solve(base,d,length,rr,den,kind);')
|
||||||
|
source = replace_once(source, ' if(cache->valid && cache->p1==p1 && cache->p2==p2 && cache->T==T &&',
|
||||||
|
' PipeProfile *profile=profile_bucket(kind);profile->cache_requests++;\n if(cache->valid && cache->p1==p1 && cache->p2==p2 && cache->T==T &&')
|
||||||
|
source = replace_once(source, ' return cache->flow;\n double result=properties?',
|
||||||
|
' {profile->cache_hits++;return cache->flow;}\n profile->cache_misses++;\n double result=properties?')
|
||||||
|
return source
|
||||||
|
|
||||||
|
|
||||||
|
def git(*arguments: str) -> str:
|
||||||
|
return subprocess.check_output(['git', *arguments], cwd=ROOT, text=True).strip()
|
||||||
|
|
||||||
|
|
||||||
|
def prepare(args, out: Path) -> dict:
|
||||||
|
if (out / 'prepared.json').exists():
|
||||||
|
metadata = json.loads((out / 'prepared.json').read_text())
|
||||||
|
if metadata['input_sha256'] != sha256(args.input.read_bytes()).hexdigest():
|
||||||
|
raise ValueError('Prepared model no longer matches the input file')
|
||||||
|
if metadata['script_sha256'] != sha256(Path(__file__).read_bytes()).hexdigest():
|
||||||
|
raise ValueError('Diagnostic helper changed; choose a fresh output directory')
|
||||||
|
return metadata
|
||||||
|
previous = git('rev-parse', args.previous_ref)
|
||||||
|
current = git('rev-parse', args.current_ref)
|
||||||
|
xml, doc = load_input(args.input)
|
||||||
|
config = simulation_config(doc.simulation)
|
||||||
|
if config.rtol != 1e-8 or config.t_start != 0 or config.t_stop != 10:
|
||||||
|
raise ValueError('This experiment requires API default rtol=1e-8 and 0–10 s model settings')
|
||||||
|
program = compile_native_program(compile_system_xml_network(doc))
|
||||||
|
out.mkdir(parents=True, exist_ok=True)
|
||||||
|
(out / 'input.xml').write_bytes(xml)
|
||||||
|
(out / 'input.json').write_bytes(args.input.read_bytes())
|
||||||
|
paths = git('ls-tree', '-r', '--name-only', current, 'native').splitlines()
|
||||||
|
metadata = dict(input=str(args.input.resolve()), input_sha256=sha256(args.input.read_bytes()).hexdigest(),
|
||||||
|
xml_sha256=sha256(xml).hexdigest(), script_sha256=sha256(Path(__file__).read_bytes()).hexdigest(),
|
||||||
|
previous_revision=previous, current_revision=current, settings=vars(config),
|
||||||
|
sample_step=doc.simulation.sample_step, variants={},
|
||||||
|
measurement_note='Instrumented times are diagnostic overhead and MUST NOT be used as production benchmark results.',
|
||||||
|
counter_scope='rhs counts native_rhs/model_eval calls including rejected trials and Jacobian differences; non_rhs includes initialization/output/probe evaluations.',
|
||||||
|
replay_scope='All guarded RHS resistance calls, including scalar analytic low-Re cases; no sampling or deduplication. Cache hits, zero pressure difference and direct PNL00R analytic calls are excluded and counted separately.',
|
||||||
|
capture_format='Native-endian IEEE-754 binary64 records: base, diameter, length, relative roughness, den=pi*d*mu, kind (six doubles, 48 bytes). Replay on the same host.',
|
||||||
|
residual_test='At actual returned q, independently factored long-double f(Re) evaluates abs(Re^2*f(Re)/K-1)<=1e-9. Separate from each algorithm stopping rule.')
|
||||||
|
original_native = builder.NATIVE
|
||||||
|
try:
|
||||||
|
for variant in VARIANTS:
|
||||||
|
directory = out / variant
|
||||||
|
revision = current if variant == 'guarded-newton' else previous
|
||||||
|
for name in paths:
|
||||||
|
target = directory / name
|
||||||
|
target.parent.mkdir(parents=True, exist_ok=True)
|
||||||
|
target.write_bytes(subprocess.check_output(['git', 'show', f'{revision}:{name}'], cwd=ROOT))
|
||||||
|
kernel = directory / 'native/components/kernels.c'
|
||||||
|
original_hash = sha256(kernel.read_bytes()).hexdigest()
|
||||||
|
kernel.write_text(instrument(kernel.read_text(), variant))
|
||||||
|
common = directory / 'native/runtime/common.c'
|
||||||
|
common.write_text(replace_once(common.read_text(), ' return model_eval(t,y,dy,w);',
|
||||||
|
''' extern void pipe_profile_rhs_enter(void),pipe_profile_rhs_leave(void);
|
||||||
|
pipe_profile_rhs_enter();int ok=model_eval(t,y,dy,w);pipe_profile_rhs_leave();return ok;'''))
|
||||||
|
main = directory / 'native/runtime/main.c'
|
||||||
|
main.write_text(replace_once(main.read_text(), 'int main(int argc, char **argv) {',
|
||||||
|
'int main(int argc, char **argv) {\n extern void pipe_profile_install(void);pipe_profile_install();'))
|
||||||
|
builder.NATIVE = directory / 'native'
|
||||||
|
build = builder.build_native(program, cache_dir=out / 'cache')
|
||||||
|
replay_source = directory / 'replay.c'
|
||||||
|
replay_source.write_text(REPLAY_MAIN)
|
||||||
|
replay = directory / ('replay.exe' if os.name == 'nt' else 'replay')
|
||||||
|
compiler, _, _ = builder.toolchain()
|
||||||
|
command = [compiler, '-std=c11', '-O3', '-Wall', '-Wextra', '-Werror', '-ffp-contract=off',
|
||||||
|
'-fno-fast-math', '-I', str(directory / 'native/include'), str(replay_source), '-lm', '-o', str(replay)]
|
||||||
|
compiled = subprocess.run(command, capture_output=True, text=True, timeout=60)
|
||||||
|
(directory / 'replay-build.log').write_text(compiled.stdout + compiled.stderr)
|
||||||
|
if compiled.returncode:
|
||||||
|
raise RuntimeError(f'Replay compilation failed: {compiled.stderr}')
|
||||||
|
metadata['variants'][variant] = dict(executable=str(build.executable), replay=str(replay),
|
||||||
|
build_key=build.manifest['buildKey'], original_kernel_sha256=original_hash,
|
||||||
|
instrumented_kernel_sha256=sha256(kernel.read_bytes()).hexdigest())
|
||||||
|
print(f'Prepared {variant}', flush=True)
|
||||||
|
finally:
|
||||||
|
builder.NATIVE = original_native
|
||||||
|
(out / 'prepared.json').write_text(json.dumps(metadata, ensure_ascii=False, indent=2) + '\n')
|
||||||
|
return metadata
|
||||||
|
|
||||||
|
|
||||||
|
def load_stats(path: Path) -> dict:
|
||||||
|
stats = json.loads(path.read_text())
|
||||||
|
for scope in ('rhs', 'non_rhs'):
|
||||||
|
for row in [stats[scope]['total'], *stats[scope]['by_kind'].values()]:
|
||||||
|
assert row['cache_requests'] == row['cache_hits'] + row['cache_misses']
|
||||||
|
assert row['flow_calls'] == row['zero_pressure_calls'] + row['pnl00r_analytic_calls'] + row['resistance_calls'] or not row['flow_calls']
|
||||||
|
assert row['resistance_calls'] == row['iterative_calls'] + row['scalar_analytic_calls'] + row['failed_before_iteration']
|
||||||
|
assert row['resistance_calls'] == sum(row['iteration_histogram'].values())
|
||||||
|
assert row['iterations_total'] == sum(int(k) * v for k, v in row['iteration_histogram'].items())
|
||||||
|
assert row['resistance_calls'] == row['finite_returns'] + row['nonfinite_returns']
|
||||||
|
assert row['resistance_calls'] == row['residual_pass'] + row['residual_fail'] + row['residual_nonfinite']
|
||||||
|
assert row['exhausted_limit'] <= row['reached_last_iteration']
|
||||||
|
return stats
|
||||||
|
|
||||||
|
|
||||||
|
def run(metadata: dict, out: Path, timeout: float):
|
||||||
|
summary_path = out / 'summary.json'
|
||||||
|
if summary_path.exists():
|
||||||
|
raise ValueError('Diagnostic results already exist; choose a fresh output directory')
|
||||||
|
settings = metadata['settings']
|
||||||
|
capture = out / 'guarded-rhs-resistance-inputs.bin'
|
||||||
|
rows = {}
|
||||||
|
for variant in VARIANTS:
|
||||||
|
directory = out / variant / 'trajectory'
|
||||||
|
directory.mkdir(parents=True, exist_ok=False)
|
||||||
|
result_path = directory / 'result.json'
|
||||||
|
stats_path = directory / 'profile.json'
|
||||||
|
env = os.environ.copy()
|
||||||
|
env.pop('PIPE_PROFILE_CAPTURE', None)
|
||||||
|
env['PIPE_PROFILE_JSON'] = str(stats_path)
|
||||||
|
if variant == 'guarded-newton':
|
||||||
|
env['PIPE_PROFILE_CAPTURE'] = str(capture)
|
||||||
|
executable = metadata['variants'][variant]['executable']
|
||||||
|
command = [executable, '--method', settings['method'], '--start', str(settings['t_start']),
|
||||||
|
'--stop', str(settings['t_stop']), '--sample-step', str(metadata['sample_step']),
|
||||||
|
'--max-step', str(settings['max_step']), '--rtol', str(settings['rtol']),
|
||||||
|
'--timeout', str(timeout), '--output', str(result_path)]
|
||||||
|
print(f'Starting diagnostic trajectory: {variant}', flush=True)
|
||||||
|
with (directory / 'worker.log').open('w') as log:
|
||||||
|
try:
|
||||||
|
process = subprocess.run(command, cwd=Path(executable).parent, env=env,
|
||||||
|
stdout=subprocess.DEVNULL, stderr=log, timeout=timeout+15)
|
||||||
|
exit_code = process.returncode
|
||||||
|
except subprocess.TimeoutExpired:
|
||||||
|
exit_code = 'external-timeout'
|
||||||
|
data = json.loads(result_path.read_text()) if result_path.exists() else {}
|
||||||
|
# Deliberately omit measured times from the cross-variant summary.
|
||||||
|
trajectory = {key:data.get(key) for key in ('success', 'status', 'message', 'simulatedUntil',
|
||||||
|
'nfev', 'acceptedSteps', 'rejectedSteps', 'njev', 'nlu', 'stateTransitions')}
|
||||||
|
trajectory['exit_code'] = exit_code
|
||||||
|
rows[variant] = dict(trajectory=trajectory, profile=load_stats(stats_path) if stats_path.exists() else None)
|
||||||
|
print(json.dumps(dict(variant=variant, **trajectory), ensure_ascii=False), flush=True)
|
||||||
|
if capture.stat().st_size % 48:
|
||||||
|
raise ValueError('Truncated replay capture')
|
||||||
|
count = capture.stat().st_size // 48
|
||||||
|
guarded = rows['guarded-newton']['profile']
|
||||||
|
if guarded is None or count != guarded['rhs']['total']['resistance_calls'] or count != guarded['capture_records']:
|
||||||
|
raise ValueError('Capture does not contain every guarded RHS resistance call')
|
||||||
|
for variant in VARIANTS:
|
||||||
|
directory = out / variant / 'replay-results'
|
||||||
|
directory.mkdir(exist_ok=False)
|
||||||
|
stats_path = directory / 'profile.json'
|
||||||
|
env = os.environ.copy()
|
||||||
|
env.pop('PIPE_PROFILE_CAPTURE', None)
|
||||||
|
env['PIPE_PROFILE_JSON'] = str(stats_path)
|
||||||
|
print(f'Replaying all {count} common inputs: {variant}', flush=True)
|
||||||
|
with (directory / 'worker.log').open('w') as log:
|
||||||
|
subprocess.run([metadata['variants'][variant]['replay'], str(capture)], env=env,
|
||||||
|
stdout=subprocess.DEVNULL, stderr=log, timeout=300, check=True)
|
||||||
|
replay = load_stats(stats_path)
|
||||||
|
if replay['rhs']['total']['resistance_calls'] != count:
|
||||||
|
raise ValueError('Replay count mismatch')
|
||||||
|
rows[variant]['replay'] = replay
|
||||||
|
# Identical input replay must reproduce every resistance-solve counter for
|
||||||
|
# the guarded solver, excluding cache/flow bookkeeping performed upstream.
|
||||||
|
exempt = {'cache_requests', 'cache_hits', 'cache_misses', 'flow_calls', 'zero_pressure_calls', 'pnl00r_analytic_calls'}
|
||||||
|
for kind in ('total', '0', '1', '2', '3'):
|
||||||
|
actual = guarded['rhs']['total'] if kind == 'total' else guarded['rhs']['by_kind'][kind]
|
||||||
|
replay = rows['guarded-newton']['replay']['rhs']['total'] if kind == 'total' else rows['guarded-newton']['replay']['rhs']['by_kind'][kind]
|
||||||
|
assert {k:v for k,v in actual.items() if k not in exempt} == {k:v for k,v in replay.items() if k not in exempt}
|
||||||
|
summary = dict(metadata=metadata, capture=dict(path=str(capture), records=count, bytes=capture.stat().st_size,
|
||||||
|
sha256=sha256(capture.read_bytes()).hexdigest(), sampling='none'), variants=rows)
|
||||||
|
summary_path.write_text(json.dumps(summary, ensure_ascii=False, indent=2) + '\n')
|
||||||
|
print(f'Diagnostic profile complete: {summary_path}', flush=True)
|
||||||
|
|
||||||
|
|
||||||
|
def main():
|
||||||
|
parser = argparse.ArgumentParser(description=__doc__)
|
||||||
|
parser.add_argument('--input', type=Path, default=ROOT / 'tests/data/test-mql-8-corrected.json')
|
||||||
|
parser.add_argument('--output-dir', type=Path, required=True)
|
||||||
|
parser.add_argument('--previous-ref', default='5d5a2e1')
|
||||||
|
parser.add_argument('--current-ref', default='808c484')
|
||||||
|
parser.add_argument('--timeout', type=float, default=120)
|
||||||
|
parser.add_argument('--prepare-only', action='store_true')
|
||||||
|
args = parser.parse_args()
|
||||||
|
out = args.output_dir.resolve()
|
||||||
|
# Keep diagnostic native copies out of production sources and tracked data.
|
||||||
|
if not out.is_relative_to(ROOT / 'test'):
|
||||||
|
parser.error('--output-dir must be below the ignored repository test/ directory')
|
||||||
|
metadata = prepare(args, out)
|
||||||
|
if not args.prepare_only:
|
||||||
|
run(metadata, out, args.timeout)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == '__main__':
|
||||||
|
main()
|
||||||
@@ -0,0 +1,342 @@
|
|||||||
|
"""Summarize the four complete C-result-encoding browser groups.
|
||||||
|
|
||||||
|
.venv/bin/python tests/manual/summarize_native_encoding.py \
|
||||||
|
--root test/c-result-encoding-20260911
|
||||||
|
|
||||||
|
Reads small timing metadata only, never result arrays or CSV contents. Every
|
||||||
|
browser group must contain one warmup and three measured successful runs.
|
||||||
|
Missing/inconsistent evidence exits 2 and writes complete:false plus an empty
|
||||||
|
CSV, so an earlier successful summary cannot masquerade as current evidence.
|
||||||
|
"""
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import csv
|
||||||
|
import hashlib
|
||||||
|
import json
|
||||||
|
import math
|
||||||
|
from pathlib import Path
|
||||||
|
from statistics import median
|
||||||
|
import sys
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
REPO = Path(__file__).resolve().parents[2]
|
||||||
|
GROUPS = {
|
||||||
|
"baseline": ("control", None),
|
||||||
|
"baseline-profiled": ("profiled", "baseline-source/profiled-backend/requests"),
|
||||||
|
"optimized": ("control", None),
|
||||||
|
"optimized-profiled": ("profiled", "backend-optimized-profiled/requests"),
|
||||||
|
}
|
||||||
|
GOALS = {
|
||||||
|
"ready": "clickToReadyDomMs",
|
||||||
|
"saved_observed": "clickToIndexedDbObservedMs",
|
||||||
|
"csv_download_saved": "csvClickToDownloadSavedMs",
|
||||||
|
}
|
||||||
|
COUNTERS = ("nfev", "acceptedSteps", "rejectedSteps", "stateTransitions", "solverStarts", "njev", "nlu")
|
||||||
|
NATIVE_IDENTITY = ("backend", "method", "solver", "sundialsVersion", "simulatedUntil", "maxAcceptedStep", *COUNTERS)
|
||||||
|
C_WALL = ("argumentPreparationSeconds", "initializationSeconds", "integrationSeconds",
|
||||||
|
"finalSampleAndStatusSeconds", "projectionSeconds", "jsonWriteSeconds", "mainTotalSeconds")
|
||||||
|
CSV_FIELDS = ("group", "phase", "run", "simulationId", "domain", "metric", "statistic", "value", "unit",
|
||||||
|
"parent", "denominatorValue", "percentOfParent", "baselineValue", "optimizedValue",
|
||||||
|
"count", "missingCount", "inclusion", "source")
|
||||||
|
DEFINITIONS = {
|
||||||
|
"scope": "One C-result-encoding experiment. Baseline/optimized control groups alone provide end-to-end comparisons; profiled C-write comparisons are separate diagnostics.",
|
||||||
|
"statistics": "Run values precede median/min/max. Stage/parent percentages use each run's own denominator before aggregation. Before/after changes use the ratio of independently collected group medians; medians and overlapping stages must not be added.",
|
||||||
|
"comparison": "Groups were collected separately, not as alternating paired trials. Duration reduction = (baseline median - optimized median) / baseline median; speedup = baseline median / optimized median. Run ordinals are not matched pairs. Ordering, scheduling and thermal variability remain possible.",
|
||||||
|
"ready": "Click to DOM-observed successful completion and an enabled Run button, not GPU completion.",
|
||||||
|
"saved": "End-to-end comparisons use pointer polling observation in BOTH control groups; includes polling and scheduling latency. Exact instrumented pointer publication remains a separate profiled metric.",
|
||||||
|
"csv": "Click through Playwright download notification and saveAs completion; includes automation and filesystem work.",
|
||||||
|
"backend": "Backend spans are inclusive wall intervals; children are included in their parents. ASGI send awaits are not pure network time. Response serialization includes metadata encoding and raw numeric-fragment joining.",
|
||||||
|
"cWrite": "C output write includes numeric encoding, stdio writes, close and index writing. CPU and wall are distinct observations; their difference is not an isolated disk-I/O measurement.",
|
||||||
|
"cSolve": "Integration includes CVODE setup, RHS/Jacobian/linear work, events and sampling. Counts are not CPU-time shares. Projection is outside integration and inside C main.",
|
||||||
|
"process": "Native reported processWallSeconds includes Python result reading after child exit; observed process lifetime spans include spawn and exit-observation latency.",
|
||||||
|
"overlap": "Browser reads overlap backend work. Parse/decode lie inside reception; persistence and rendering overlap. Per-stage percentages are inclusive and must not be added.",
|
||||||
|
"warmup": "Warmup rows are retained separately. Formal browser runs require build-cache hits. CacheHit, not the warmup label, identifies cold compilation.",
|
||||||
|
"replay": "Microbenchmark replays preloaded contiguous binary64 values, excluding model projection and production strided access. Its medians are separate and never substituted for end-to-end results.",
|
||||||
|
"validation": "Timing metadata validates group completeness, input/assets, counters, sample/variable counts and recorded success/restore flags. It does not independently prove numerical bitwise parity; use the separate full-result comparison artifact.",
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def numeric(value: Any) -> bool:
|
||||||
|
return type(value) in (int, float) and math.isfinite(value)
|
||||||
|
|
||||||
|
|
||||||
|
def require(condition: bool, message: str) -> None:
|
||||||
|
if not condition:
|
||||||
|
raise ValueError(message)
|
||||||
|
|
||||||
|
|
||||||
|
def stats(values: list[Any]) -> dict:
|
||||||
|
present = [v for v in values if numeric(v)]
|
||||||
|
return {"count": len(present), "missingCount": len(values) - len(present), "values": values,
|
||||||
|
"median": median(present) if present else None,
|
||||||
|
"min": min(present) if present else None, "max": max(present) if present else None}
|
||||||
|
|
||||||
|
|
||||||
|
def percent(value: Any, denominator: Any) -> float | None:
|
||||||
|
return value / denominator * 100 if numeric(value) and numeric(denominator) and denominator > 0 else None
|
||||||
|
|
||||||
|
|
||||||
|
def finite_field(data: dict, name: str, context: str, *, positive: bool = False) -> float:
|
||||||
|
value = data.get(name)
|
||||||
|
require(numeric(value) and (value > 0 if positive else value >= 0), f"{context}: missing/invalid {name}")
|
||||||
|
return value
|
||||||
|
|
||||||
|
|
||||||
|
class Summarizer:
|
||||||
|
def __init__(self, root: Path):
|
||||||
|
self.root = root.resolve()
|
||||||
|
self.sources: dict[str, dict] = {}
|
||||||
|
self.observations: list[dict] = []
|
||||||
|
self.simulation_ids: set[str] = set()
|
||||||
|
|
||||||
|
def read(self, relative: str) -> dict:
|
||||||
|
path = (self.root / relative).resolve()
|
||||||
|
require(path.is_relative_to(self.root), f"Metadata path leaves experiment root: {relative}")
|
||||||
|
require(path.is_file(), f"Incomplete experiment: missing {relative}")
|
||||||
|
require(path.stat().st_size <= 4 * 1024 * 1024, f"Refusing large non-metadata input: {relative}")
|
||||||
|
raw = path.read_bytes()
|
||||||
|
value = json.loads(raw, parse_constant=lambda token: (_ for _ in ()).throw(ValueError(f"Invalid JSON number: {token}")))
|
||||||
|
require(isinstance(value, dict), f"Expected metadata object: {relative}")
|
||||||
|
self.sources[relative] = {"bytes": len(raw), "sha256": hashlib.sha256(raw).hexdigest()}
|
||||||
|
return value
|
||||||
|
|
||||||
|
def observe(self, run: dict, domain: str, metric: str, value: Any, unit: str = "ms", *,
|
||||||
|
parent: str = "", denominator: Any = None, inclusion: str = "inclusive/overlapping; not additive",
|
||||||
|
source: str = "", baseline: Any = None, optimized: Any = None) -> None:
|
||||||
|
require(value is None or numeric(value), f"Invalid observation {domain}.{metric}: {value!r}")
|
||||||
|
self.observations.append({k: run[k] for k in ("group", "phase", "run", "simulationId")} | {
|
||||||
|
"domain": domain, "metric": metric, "value": value, "unit": unit, "parent": parent,
|
||||||
|
"denominatorValue": denominator, "percentOfParent": percent(value, denominator),
|
||||||
|
"baselineValue": baseline, "optimizedValue": optimized, "inclusion": inclusion, "source": source})
|
||||||
|
|
||||||
|
def backend(self, run: dict, relative: str) -> dict:
|
||||||
|
data = self.read(relative)
|
||||||
|
context = f"{run['group']}/{run['run']} backend"
|
||||||
|
require(data.get("id") == run["simulationId"], f"{context}: simulationId mismatch")
|
||||||
|
require(data.get("httpStatus") == 200, f"{context}: HTTP did not succeed")
|
||||||
|
native = run["native"]
|
||||||
|
for key in (*NATIVE_IDENTITY, "solveSeconds", "solveCpuSeconds", "buildKey", "cacheHit"):
|
||||||
|
require(data.get("native", {}).get(key) == native.get(key), f"{context}: browser/backend mismatch for {key}")
|
||||||
|
require(data.get("sampleCount") == run["sampleCount"], f"{context}: backend sampleCount mismatch")
|
||||||
|
http = finite_field(data, "httpTotalSeconds", context, positive=True) * 1000
|
||||||
|
self.observe(run, "backend", "httpTotalMs", http, source=relative, inclusion=DEFINITIONS["backend"])
|
||||||
|
spans = data.get("spans", [])
|
||||||
|
require(bool(spans), f"{context}: missing backend spans")
|
||||||
|
totals: dict[str, float] = {}
|
||||||
|
for span in spans:
|
||||||
|
start = finite_field(span, "startMs", context)
|
||||||
|
end = finite_field(span, "endMs", context)
|
||||||
|
require(start <= end <= http + 1e-5, f"{context}: span outside HTTP interval: {span['name']}")
|
||||||
|
totals[span["name"]] = totals.get(span["name"], 0) + end - start
|
||||||
|
for name in ("native_indexed_result_read", "native_process_lifetime_observed", "response_result_json_serialization"):
|
||||||
|
require(name in totals, f"{context}: missing {name}")
|
||||||
|
for name, value in totals.items():
|
||||||
|
self.observe(run, "backend_span", name, value, parent="httpTotalMs", denominator=http,
|
||||||
|
inclusion=DEFINITIONS["backend"], source=relative)
|
||||||
|
c = data.get("nativeStages", {})
|
||||||
|
main_ms = finite_field(c, "mainTotalSeconds", context, positive=True) * 1000
|
||||||
|
for name in C_WALL:
|
||||||
|
self.observe(run, "c_wall", name.removesuffix("Seconds") + "Ms", finite_field(c, name, context) * 1000,
|
||||||
|
parent="cMainMs" if name != "mainTotalSeconds" else "",
|
||||||
|
denominator=main_ms if name != "mainTotalSeconds" else None,
|
||||||
|
inclusion=DEFINITIONS["cWrite"] if name == "jsonWriteSeconds" else DEFINITIONS["cSolve"], source=relative)
|
||||||
|
for name in ("projectionCpuSeconds", "jsonWriteCpuSeconds"):
|
||||||
|
self.observe(run, "c_cpu", name.removesuffix("Seconds") + "Ms", finite_field(c, name, context) * 1000,
|
||||||
|
inclusion="CPU duration; separate from wall intervals", source=relative)
|
||||||
|
self.observe(run, "backend", "responseSendAwaitMs", finite_field(data, "responseSendAwaitSeconds", context) * 1000,
|
||||||
|
parent="httpTotalMs", denominator=http, inclusion=DEFINITIONS["backend"], source=relative)
|
||||||
|
for name in ("rawSeriesBytes", "responseBodyBytes"):
|
||||||
|
self.observe(run, "size", name, finite_field(data, name, context, positive=True), "bytes", source=relative)
|
||||||
|
phases = data.get("existingPerformance", {}).get("phases", {})
|
||||||
|
for name, phase in phases.items():
|
||||||
|
self.observe(run, "backend_existing", name, finite_field(phase, "inclusiveNs", context) / 1e6,
|
||||||
|
parent="httpTotalMs", denominator=http,
|
||||||
|
inclusion="Inclusive duration without aligned start/end; not an exclusive extra cost", source=relative)
|
||||||
|
return {"source": relative, "xmlSha256": data.get("xmlSha256"), "httpTotalMs": http,
|
||||||
|
"spans": spans, "nativeStages": c, "process": data.get("process"), "build": data.get("build")}
|
||||||
|
|
||||||
|
def group(self, name: str, mode: str, backend_root: str | None) -> dict:
|
||||||
|
relative = f"browser-{name}/summary.json"
|
||||||
|
data = self.read(relative)
|
||||||
|
require(data.get("errors") == [], f"{name}: missing errors list or reported browser errors")
|
||||||
|
rows = data.get("rows", [])
|
||||||
|
require(len(rows) == 4, f"Incomplete {name}: expected 1 warmup + 3 measured rows, got {len(rows)}")
|
||||||
|
require(sorted(r.get("run", -1) for r in rows) == [0, 1, 2, 3], f"{name}: unexpected/duplicate run numbers")
|
||||||
|
runs = []
|
||||||
|
for row in sorted(rows, key=lambda r: r["run"]):
|
||||||
|
context = f"{name}/{row['run']}"
|
||||||
|
require(row.get("mode") == mode and row.get("deep") is False, f"{context}: wrong instrumentation mode")
|
||||||
|
require(row.get("warmup") is (row["run"] == 0), f"{context}: warmup label mismatch")
|
||||||
|
sid = row.get("simulationId")
|
||||||
|
require(isinstance(sid, str) and bool(sid) and sid not in self.simulation_ids, f"{context}: missing/duplicate simulationId")
|
||||||
|
self.simulation_ids.add(sid)
|
||||||
|
native = row.get("native", {})
|
||||||
|
require(native.get("success") is True and native.get("status") == "completed", f"{context}: native simulation failed")
|
||||||
|
require(row.get("restoredIdentical") is True, f"{context}: restore parity was not confirmed")
|
||||||
|
require(type(native.get("cacheHit")) is bool, f"{context}: missing cacheHit")
|
||||||
|
if row["run"]:
|
||||||
|
require(native["cacheHit"], f"{context}: measured run includes a cold build")
|
||||||
|
for key in NATIVE_IDENTITY:
|
||||||
|
require(key in native and native[key] is not None, f"{context}: missing native {key}")
|
||||||
|
for key in COUNTERS:
|
||||||
|
value = finite_field(native, key, context)
|
||||||
|
require(int(value) == value, f"{context}: noninteger counter {key}")
|
||||||
|
run = {"group": name, "phase": "warmup" if row["warmup"] else "measured", "run": row["run"],
|
||||||
|
"simulationId": sid, "native": native, "original": row,
|
||||||
|
"sampleCount": finite_field(row, "sampleCount", context, positive=True),
|
||||||
|
"variableCount": finite_field(row, "variableCount", context, positive=True)}
|
||||||
|
for key in GOALS.values():
|
||||||
|
finite_field(row, key, context, positive=True)
|
||||||
|
if mode == "profiled":
|
||||||
|
for key in ("resultParseMs", "synchronousStreamDecodeMs", "clickToIndexedDbCommitMs", "streamBytes"):
|
||||||
|
finite_field(row, key, context, positive=True)
|
||||||
|
for key, value in row.items():
|
||||||
|
if key.endswith(("Ms", "Bytes")) and (value is None or numeric(value)):
|
||||||
|
self.observe(run, "frontend", key, value, "bytes" if key.endswith("Bytes") else "ms", source=relative)
|
||||||
|
for key in ("solveSeconds", "solveCpuSeconds", "processWallSeconds", "buildSeconds"):
|
||||||
|
self.observe(run, "native", key.removesuffix("Seconds") + "Ms", finite_field(native, key, context) * 1000,
|
||||||
|
source=relative, inclusion=DEFINITIONS["process"] if key == "processWallSeconds" else "Native-reported timing; CPU and wall are separate")
|
||||||
|
run["backend"] = self.backend(run, f"{backend_root}/{sid}/stages.json") if backend_root else None
|
||||||
|
runs.append(run)
|
||||||
|
for key in ("inputSha256", "buildAssetSetSha256"):
|
||||||
|
value = data.get(key)
|
||||||
|
require(isinstance(value, str) and len(value) == 64 and all(c in "0123456789abcdef" for c in value), f"{name}: missing/invalid {key}")
|
||||||
|
require(bool(data.get("servedAssets")), f"{name}: missing served frontend assets")
|
||||||
|
return {"source": relative, "mode": mode, "inputSha256": data["inputSha256"],
|
||||||
|
"buildAssetSetSha256": data["buildAssetSetSha256"], "servedAssets": data["servedAssets"],
|
||||||
|
"browser": data.get("browser"), "node": data.get("node"), "scriptSha256": data.get("scriptSha256"),
|
||||||
|
"sourceDefinitions": data.get("definitions"), "measuredRunCount": 3, "warmupRunCount": 1, "runs": runs}
|
||||||
|
|
||||||
|
def comparisons(self, groups: dict, before: str, after: str, metrics: dict, *, diagnostic: bool) -> list[dict]:
|
||||||
|
comparison = []
|
||||||
|
for target, (source_domain, metric) in metrics.items():
|
||||||
|
summaries = []
|
||||||
|
for name in (before, after):
|
||||||
|
rows = [o for o in self.observations if o["group"] == name and o["phase"] == "measured"
|
||||||
|
and o["domain"] == source_domain and o["metric"] == metric]
|
||||||
|
require(len(rows) == 3 and all(numeric(r["value"]) and r["value"] > 0 for r in rows),
|
||||||
|
f"Missing comparison metric {name}/{target}")
|
||||||
|
summaries.append(stats([r["value"] for r in sorted(rows, key=lambda r: r["run"])]))
|
||||||
|
baseline, optimized = summaries
|
||||||
|
old, new = baseline["median"], optimized["median"]
|
||||||
|
comparison.append({"target": target, "metric": f"{source_domain}.{metric}", "before": before, "after": after,
|
||||||
|
"diagnosticOnly": diagnostic, "statistic": "ratio_of_group_medians", "definition": DEFINITIONS["comparison"],
|
||||||
|
"source": f"{groups[before]['source']} | {groups[after]['source']}",
|
||||||
|
"baselineMs": baseline, "optimizedMs": optimized, "savedMs": old - new,
|
||||||
|
"durationReductionPercent": (old - new) / old * 100, "speedupRatio": old / new})
|
||||||
|
return comparison
|
||||||
|
|
||||||
|
def replay(self) -> dict:
|
||||||
|
relative = "replay/summary.json"
|
||||||
|
data = self.read(relative)
|
||||||
|
require(data.get("allRealFileBinary64Parity") is True, "Replay real-file binary64 verification not complete")
|
||||||
|
supplied = data.get("medians", {})
|
||||||
|
require(bool(supplied.get("file")), "Replay real-file medians missing")
|
||||||
|
runs = data.get("runs", [])
|
||||||
|
require(bool(runs), "Replay run metadata missing")
|
||||||
|
for row in runs:
|
||||||
|
require(row.get("success") is True, "Replay includes a failed run")
|
||||||
|
run = {"group": f"replay:{row['sink']}:{row['variant']}", "phase": "warmup" if row["warmup"] else "measured",
|
||||||
|
"run": row["run"], "simulationId": ""}
|
||||||
|
for metric, unit in (("wallSeconds", "ms"), ("cpuSeconds", "ms"), ("encodedBytes", "bytes")):
|
||||||
|
value = finite_field(row, metric, run["group"], positive=True)
|
||||||
|
self.observe(run, "replay", metric.removesuffix("Seconds") + "Ms" if unit == "ms" else metric,
|
||||||
|
value * 1000 if unit == "ms" else value, unit, source=relative, inclusion=DEFINITIONS["replay"])
|
||||||
|
medians = {sink: {variant: {k: value[k] for k in ("wallSeconds", "cpuSeconds", "encodedBytes")}
|
||||||
|
for variant, value in variants.items()} for sink, variants in supplied.items()}
|
||||||
|
return {"source": relative, "suppliedMedians": medians, "runs": runs,
|
||||||
|
"timingContract": data.get("prepared", {}).get("timingContract"), "limitation": data.get("limitation"),
|
||||||
|
"usedForEndToEndComparison": False}
|
||||||
|
|
||||||
|
def build(self) -> dict:
|
||||||
|
groups = {name: self.group(name, *config) for name, config in GROUPS.items()}
|
||||||
|
reference = groups["baseline"]
|
||||||
|
first = reference["runs"][0]
|
||||||
|
for name, group in groups.items():
|
||||||
|
for key in ("inputSha256", "buildAssetSetSha256", "browser", "node"):
|
||||||
|
require(group[key] is not None and group[key] == reference[key], f"Cross-group {key} mismatch: {name}")
|
||||||
|
for run in group["runs"]:
|
||||||
|
for key in ("sampleCount", "variableCount"):
|
||||||
|
require(run[key] == first[key], f"Cross-run {key} mismatch: {name}/{run['run']}")
|
||||||
|
for key in NATIVE_IDENTITY:
|
||||||
|
require(run["native"][key] == first["native"][key], f"Cross-run native {key} mismatch: {name}/{run['run']}")
|
||||||
|
xml_hashes = [run["backend"]["xmlSha256"] for group in groups.values() for run in group["runs"] if run["backend"]]
|
||||||
|
require(all(isinstance(value, str) and len(value) == 64 for value in xml_hashes) and len(set(xml_hashes)) == 1,
|
||||||
|
"Profiled input XML SHA missing or mismatched")
|
||||||
|
goals = self.comparisons(groups, "baseline", "optimized",
|
||||||
|
{key: ("frontend", metric) for key, metric in GOALS.items()}, diagnostic=False)
|
||||||
|
stages = self.comparisons(groups, "baseline-profiled", "optimized-profiled",
|
||||||
|
{"write_wall": ("c_wall", "jsonWriteMs"), "write_cpu": ("c_cpu", "jsonWriteCpuMs")}, diagnostic=True)
|
||||||
|
replay = self.replay()
|
||||||
|
buckets: dict[tuple, list[dict]] = {}
|
||||||
|
keys = ("group", "phase", "domain", "metric", "unit", "parent")
|
||||||
|
for observation in self.observations:
|
||||||
|
buckets.setdefault(tuple(observation[key] for key in keys), []).append(observation)
|
||||||
|
aggregates = []
|
||||||
|
for key, rows in sorted(buckets.items()):
|
||||||
|
require(len({r["run"] for r in rows}) == len(rows), f"Duplicate per-run observation: {key}")
|
||||||
|
aggregates.append(dict(zip(keys, key)) | stats([r["value"] for r in rows]) | {
|
||||||
|
"percentOfParent": stats([r["percentOfParent"] for r in rows]),
|
||||||
|
"runs": [{k: row[k] for k in ("run", "simulationId")} for row in rows]})
|
||||||
|
return {"schemaVersion": 1, "complete": True, "errors": [], "experimentRoot": str(self.root),
|
||||||
|
"scriptSha256": hashlib.sha256(Path(__file__).read_bytes()).hexdigest(), "definitions": DEFINITIONS,
|
||||||
|
"sourceFiles": self.sources, "groups": groups, "changes": {"endToEnd": goals, "cWriteDiagnostics": stages},
|
||||||
|
"replay": replay, "observations": self.observations, "aggregates": aggregates,
|
||||||
|
"validation": {"groupCount": 4, "browserRunCount": 16, "warmupCount": 4, "measuredCount": 12,
|
||||||
|
"inputSha256": reference["inputSha256"], "frontendAssetSetSha256": reference["buildAssetSetSha256"],
|
||||||
|
"profiledXmlSha256": xml_hashes[0], "sampleCount": first["sampleCount"], "variableCount": first["variableCount"],
|
||||||
|
"nativeIdentity": {key: first["native"][key] for key in NATIVE_IDENTITY},
|
||||||
|
"fullNumericalParityIndependentlyChecked": False}}
|
||||||
|
|
||||||
|
|
||||||
|
def write_outputs(root: Path, summary: dict) -> None:
|
||||||
|
root.mkdir(parents=True, exist_ok=True)
|
||||||
|
with (root / "timings.csv").open("w", encoding="utf-8", newline="") as stream:
|
||||||
|
writer = csv.DictWriter(stream, fieldnames=CSV_FIELDS)
|
||||||
|
writer.writeheader()
|
||||||
|
for row in summary.get("observations", []):
|
||||||
|
writer.writerow(row | {"statistic": "run", "count": int(numeric(row["value"])), "missingCount": int(row["value"] is None)})
|
||||||
|
for entry in summary.get("aggregates", []):
|
||||||
|
for statistic in ("median", "min", "max"):
|
||||||
|
writer.writerow({k: entry[k] for k in ("group", "phase", "domain", "metric", "unit", "parent", "count", "missingCount")} | {
|
||||||
|
"statistic": statistic, "value": entry[statistic], "percentOfParent": entry["percentOfParent"][statistic],
|
||||||
|
"inclusion": "Per-run values and percentages aggregated separately; medians are not additive"})
|
||||||
|
for category, changes in summary.get("changes", {}).items():
|
||||||
|
for change in changes:
|
||||||
|
for key, unit in (("durationReductionPercent", "percent"), ("speedupRatio", "ratio"), ("savedMs", "ms")):
|
||||||
|
writer.writerow({"group": f"{change['after']}-vs-{change['before']}", "phase": "measured",
|
||||||
|
"domain": f"comparison:{category}", "metric": f"{change['target']}.{key}",
|
||||||
|
"statistic": "ratio_of_group_medians" if key != "savedMs" else "difference_of_group_medians",
|
||||||
|
"value": change[key], "unit": unit, "baselineValue": change["baselineMs"]["median"],
|
||||||
|
"optimizedValue": change["optimizedMs"]["median"], "count": 3, "missingCount": 0,
|
||||||
|
"inclusion": ("Profiled C-write diagnostic only; " if change["diagnosticOnly"] else "") + change["definition"],
|
||||||
|
"source": change["source"]})
|
||||||
|
(root / "summary.json").write_text(json.dumps(summary, ensure_ascii=False, indent=2, allow_nan=False) + "\n", encoding="utf-8")
|
||||||
|
|
||||||
|
|
||||||
|
def main() -> int:
|
||||||
|
parser = argparse.ArgumentParser(description=__doc__)
|
||||||
|
parser.add_argument("--root", type=Path, default=REPO / "test/c-result-encoding-20260911")
|
||||||
|
args = parser.parse_args()
|
||||||
|
root = args.root.resolve()
|
||||||
|
if not root.is_relative_to(REPO / "test"):
|
||||||
|
parser.error("--root must be beneath the repository's ignored test/ directory")
|
||||||
|
summarizer = Summarizer(root)
|
||||||
|
try:
|
||||||
|
summary = summarizer.build()
|
||||||
|
except (OSError, ValueError, KeyError, TypeError) as error:
|
||||||
|
summary = {"schemaVersion": 1, "complete": False, "errors": [str(error)],
|
||||||
|
"experimentRoot": str(root), "sourceFiles": summarizer.sources,
|
||||||
|
"note": "No partial timing statistics are published. Complete/fix all groups and rerun."}
|
||||||
|
write_outputs(root, summary)
|
||||||
|
print(json.dumps(summary, ensure_ascii=False, indent=2), file=sys.stderr)
|
||||||
|
return 2
|
||||||
|
write_outputs(root, summary)
|
||||||
|
print(json.dumps({"complete": True, "validation": summary["validation"],
|
||||||
|
"observations": len(summary["observations"]), "outputs": [str(root / name) for name in ("summary.json", "timings.csv")]}, ensure_ascii=False, indent=2))
|
||||||
|
return 0
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
raise SystemExit(main())
|
||||||
@@ -0,0 +1,385 @@
|
|||||||
|
"""Summarize one web-cost experiment without reading numerical result files.
|
||||||
|
|
||||||
|
Usage: .venv/bin/python tests/manual/summarize_web_cost.py --root test/web-cost-20260911
|
||||||
|
Only small summary/trace/stages/environment JSON files are read. Warmups and
|
||||||
|
control/profiled groups stay separate. This script does not compute speedups.
|
||||||
|
"""
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import csv
|
||||||
|
import hashlib
|
||||||
|
import json
|
||||||
|
import math
|
||||||
|
from pathlib import Path
|
||||||
|
from statistics import median
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
REPO = Path(__file__).resolve().parents[2]
|
||||||
|
AXIS = (
|
||||||
|
("runClick", "click"), ("fetchStart", "fetch"), ("headers", "headers"),
|
||||||
|
("lastChunk", "last_chunk"), ("resultParseStart", "parse_start"),
|
||||||
|
("resultParseEnd", "parse_end"), ("streamEof", "eof"),
|
||||||
|
("resultReadyDom", "ready"), ("indexedDbCommittedPointer", "commit"),
|
||||||
|
)
|
||||||
|
# Prune the reporting tree at these declared boundaries. The indexed-read and
|
||||||
|
# process-lifetime spans are atomic ONLY in this coarse HTTP partition; their
|
||||||
|
# measured children remain visible in the separate inclusive span hierarchy.
|
||||||
|
HTTP_BOUNDARIES = {
|
||||||
|
"xml_validation", "network_compilation", "c_generation",
|
||||||
|
"native_build_or_cache_validation", "native_process_lifetime_observed",
|
||||||
|
"native_indexed_result_read", "profile_artifact_preservation",
|
||||||
|
"response_result_json_serialization",
|
||||||
|
}
|
||||||
|
C_WALL_FIELDS = (
|
||||||
|
"argumentPreparationSeconds", "initializationSeconds", "integrationSeconds",
|
||||||
|
"finalSampleAndStatusSeconds", "projectionSeconds", "jsonWriteSeconds",
|
||||||
|
)
|
||||||
|
DEFINITIONS = {
|
||||||
|
"scope": "One current experiment; control/profiled are observation modes, not before/after implementations. No speedup is computed.",
|
||||||
|
"statistics": "Median/min/max are calculated from individual runs. Percentages divide each run by its own stated denominator before aggregation. Medians need not add to a median total.",
|
||||||
|
"warmup": "All warmup rows are retained separately and excluded from measured statistics. Cold build is identified by cacheHit=false, not by the warmup label.",
|
||||||
|
"frontendClock": "Browser performance.now() within one trace timeOrigin. Reload/restore has another time axis; no timestamps are subtracted across documents or across browser/backend clocks.",
|
||||||
|
"waterfall": "Adjacent requested marks are subtracted without clipping negatives. Missing marks remain null; commit is the instrumented pointer publication, never replaced by the control polling mark.",
|
||||||
|
"waterfallPercent": "Each adjacent interval / same-run click-to-commit. Negative intervals remain negative and reveal overlapping completion order; they are not exclusive CPU costs.",
|
||||||
|
"backendClock": "Backend spans use request-relative perf_counter_ns. Parent links are inferred by interval containment and indicate inclusive wall intervals, not a traced call stack or exclusive CPU work.",
|
||||||
|
"httpPartition": "Pruned reporting leaves use HTTP_BOUNDARIES, checked for overlap and containment before partitioning. Their union is subtracted from HTTP total to produce unnamed other time. Spawn/read/parse children must not be added again. Response assembly has a duration but no aligned timestamp and remains in other.",
|
||||||
|
"backendOther": "Unclassified HTTP wall intervals include uninstrumented work, scheduling, response assembly, inter-stage gaps and sending. They are not all transport time or CPU work.",
|
||||||
|
"process": "Observed lifetime runs from Popen entry to the first existing poll/wait reporting exit; includes spawn and exit-observation delay. The result processWallSeconds also includes Python result reading. Child CPU is a RUSAGE_CHILDREN delta and assumes no unrelated child is reaped in the same parent interval.",
|
||||||
|
"cInitialization": "The separately measured initialization covers model_init and first sample. CVODE allocation/init/reinit and cleanup are inside integration, along with RHS, Jacobian/linear solve, events and sampling.",
|
||||||
|
"cOutput": "Projection includes final/sample model_eval and allocation. JSON write includes float formatting, stdio, file close and index writing. CPU and wall are separately reported; their difference is not a measured disk-I/O stage.",
|
||||||
|
"cParent": "C phase wall percentages use same-run C main wall; CPU percentages use the same-run observed child user+system CPU. Initial/final/remaining CPU is not individually measured.",
|
||||||
|
"stream": "Read-wait wall overlaps backend production/transport/browser scheduling. Decode and JSON.parse are within headers-to-EOF. NDJSON scanning/join/trim, callbacks, GC and scheduler time are not independently timed.",
|
||||||
|
"persistence": "Save preparation/IndexedDB run in the background with readiness and rendering. Transaction windows include asynchronous waiting; pointer commit is publication after successful writes, control observation uses polling.",
|
||||||
|
"renderExport": "Two requestAnimationFrame callbacks give a paint opportunity, not GPU completion. Download completion includes automation delivery and saveAs. CSV Worker transfer/preparation overlaps Worker activity; finish-post-to-receipt is not isolated Worker CPU.",
|
||||||
|
"unmeasured": "No exclusive breakdown of click preprocessing, NDJSON join/trim, React/GC, CVODE internals, C formatting versus file writes, browser network stack or GPU work is invented. Deeper/native-only experiments are intentionally not read here.",
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def number(value: Any) -> bool:
|
||||||
|
return type(value) in (int, float) and math.isfinite(value)
|
||||||
|
|
||||||
|
|
||||||
|
def ratio(value: Any, denominator: Any) -> float | None:
|
||||||
|
return 100.0 * value / denominator if number(value) and number(denominator) and denominator > 0 else None
|
||||||
|
|
||||||
|
|
||||||
|
def difference(marks: dict, left: str, right: str) -> float | None:
|
||||||
|
a, b = marks.get(left), marks.get(right)
|
||||||
|
return b - a if number(a) and number(b) else None
|
||||||
|
|
||||||
|
|
||||||
|
def statistics(values: list[Any]) -> dict:
|
||||||
|
valid = [float(value) for value in values if number(value)]
|
||||||
|
return {"count": len(valid), "missingCount": len(values) - len(valid),
|
||||||
|
"median": median(valid) if valid else None,
|
||||||
|
"min": min(valid) if valid else None, "max": max(valid) if valid else None,
|
||||||
|
"values": values}
|
||||||
|
|
||||||
|
|
||||||
|
def union_length(intervals: list[tuple[float, float]]) -> float:
|
||||||
|
total, end = 0.0, -math.inf
|
||||||
|
for start, stop in sorted(intervals):
|
||||||
|
total += max(0.0, stop - max(start, end))
|
||||||
|
end = max(end, stop)
|
||||||
|
return total
|
||||||
|
|
||||||
|
|
||||||
|
def span_hierarchy(raw_spans: list[dict], http_ms: float) -> list[dict]:
|
||||||
|
spans = [{"id": "http", "name": "http_total", "startMs": 0.0,
|
||||||
|
"endMs": http_ms, "durationMs": http_ms, "source": "httpTotalSeconds"}]
|
||||||
|
for index, span in enumerate(raw_spans):
|
||||||
|
start, end = span.get("startMs"), span.get("endMs")
|
||||||
|
if not number(start) or not number(end) or end < start:
|
||||||
|
raise ValueError(f"Invalid backend span: {span}")
|
||||||
|
spans.append({"id": f"span-{index}", "name": span["name"], "startMs": start,
|
||||||
|
"endMs": end, "durationMs": end - start,
|
||||||
|
"recordedSeconds": span.get("seconds"), "source": "spans"})
|
||||||
|
for current in spans:
|
||||||
|
candidates = [other for other in spans if other["id"] != current["id"]
|
||||||
|
and other["startMs"] <= current["startMs"]
|
||||||
|
and other["endMs"] >= current["endMs"]
|
||||||
|
and (other["durationMs"] > current["durationMs"]
|
||||||
|
or other["id"] == "http")]
|
||||||
|
parent = min(candidates, key=lambda s: s["durationMs"]) if current["id"] != "http" and candidates else None
|
||||||
|
current["parentId"] = parent["id"] if parent else None
|
||||||
|
current["parentName"] = parent["name"] if parent else None
|
||||||
|
current["percentOfParent"] = ratio(current["durationMs"], parent["durationMs"]) if parent else None
|
||||||
|
current["percentOfHttp"] = ratio(current["durationMs"], http_ms)
|
||||||
|
current["inclusion"] = "inclusive interval; do not add its children"
|
||||||
|
for current in spans:
|
||||||
|
children = [s for s in spans if s["parentId"] == current["id"]]
|
||||||
|
current["children"] = [s["id"] for s in children]
|
||||||
|
current["uncoveredByDirectChildrenMs"] = current["durationMs"] - union_length(
|
||||||
|
[(s["startMs"], s["endMs"]) for s in children])
|
||||||
|
current["partiallyOverlaps"] = [s["id"] for s in spans if s["id"] != current["id"]
|
||||||
|
and max(s["startMs"], current["startMs"]) < min(s["endMs"], current["endMs"])
|
||||||
|
and not (s["startMs"] <= current["startMs"] and s["endMs"] >= current["endMs"])
|
||||||
|
and not (current["startMs"] <= s["startMs"] and current["endMs"] >= s["endMs"])]
|
||||||
|
return spans
|
||||||
|
|
||||||
|
|
||||||
|
class Summary:
|
||||||
|
def __init__(self, root: Path):
|
||||||
|
self.root = root.resolve()
|
||||||
|
self.sources: dict[str, dict] = {}
|
||||||
|
self.observations: list[dict] = []
|
||||||
|
self.warnings: list[str] = []
|
||||||
|
|
||||||
|
def read(self, path: Path) -> dict:
|
||||||
|
path = path.resolve()
|
||||||
|
path.relative_to(self.root)
|
||||||
|
size = path.stat().st_size
|
||||||
|
if size > 4 * 1024 * 1024:
|
||||||
|
raise ValueError(f"Expected small metadata JSON, refusing {path} ({size} bytes)")
|
||||||
|
raw = path.read_bytes()
|
||||||
|
self.sources[str(path.relative_to(self.root))] = {
|
||||||
|
"sha256": hashlib.sha256(raw).hexdigest(), "bytes": len(raw)}
|
||||||
|
return json.loads(raw)
|
||||||
|
|
||||||
|
def observe(self, run: dict, domain: str, metric: str, value: Any,
|
||||||
|
unit: str = "ms", *, parent: str = "", denominator: Any = None,
|
||||||
|
inclusion: str = "inclusive or overlapping; not additive", source: str = "") -> None:
|
||||||
|
if value is not None and not number(value):
|
||||||
|
raise ValueError(f"Non-numeric metric {domain}.{metric}: {value!r}")
|
||||||
|
self.observations.append({"group": run["group"], "phase": run["phase"],
|
||||||
|
"run": run["run"], "simulationId": run["simulationId"], "domain": domain,
|
||||||
|
"metric": metric, "value": value, "unit": unit, "parent": parent,
|
||||||
|
"denominatorValue": denominator if number(denominator) else None,
|
||||||
|
"percentOfParent": ratio(value, denominator), "inclusion": inclusion, "source": source})
|
||||||
|
|
||||||
|
def backend(self, run: dict, data: dict, source: str) -> dict:
|
||||||
|
if data.get("id") != run["simulationId"]:
|
||||||
|
raise ValueError(f"Simulation ID mismatch in {source}")
|
||||||
|
http_ms = data["httpTotalSeconds"] * 1000
|
||||||
|
hierarchy = span_hierarchy(data.get("spans", []), http_ms)
|
||||||
|
by_name: dict[str, list[dict]] = {}
|
||||||
|
for span in hierarchy:
|
||||||
|
by_name.setdefault(span["name"], []).append(span)
|
||||||
|
for name, occurrences in by_name.items():
|
||||||
|
parent_names = sorted({s["parentName"] or "" for s in occurrences})
|
||||||
|
self.observe(run, "backend_spans", name, sum(s["durationMs"] for s in occurrences),
|
||||||
|
parent="http_total", denominator=http_ms, source=source,
|
||||||
|
inclusion=f"inclusive sum of {len(occurrences)} call(s); interval parents: {', '.join(parent_names)}")
|
||||||
|
chosen = [s for s in hierarchy if s["name"] in HTTP_BOUNDARIES]
|
||||||
|
intervals = [(s["startMs"], s["endMs"]) for s in chosen]
|
||||||
|
covered = union_length(intervals)
|
||||||
|
overlaps = sum(stop - start for start, stop in intervals) - covered
|
||||||
|
outside = [s["id"] for s in chosen if s["startMs"] < 0 or s["endMs"] > http_ms]
|
||||||
|
partition_ok = overlaps <= 1e-6 and not outside
|
||||||
|
partition = {"valid": partition_ok, "scope": "http_total", "totalMs": http_ms,
|
||||||
|
"selectedSpanIds": [s["id"] for s in chosen], "measuredUnionMs": covered,
|
||||||
|
"overlapMs": overlaps, "outsideHttpSpanIds": outside,
|
||||||
|
"otherMs": http_ms - covered if not outside else None,
|
||||||
|
"note": DEFINITIONS["httpPartition"], "segments": []}
|
||||||
|
if partition_ok:
|
||||||
|
partition_durations: dict[str, float] = {}
|
||||||
|
for span in sorted(chosen, key=lambda s: s["startMs"]):
|
||||||
|
partition["segments"].append({"metric": span["name"], "startMs": span["startMs"],
|
||||||
|
"endMs": span["endMs"], "durationMs": span["durationMs"],
|
||||||
|
"percentOfHttp": ratio(span["durationMs"], http_ms)})
|
||||||
|
partition_durations[span["name"]] = partition_durations.get(span["name"], 0) + span["durationMs"]
|
||||||
|
for name, duration in partition_durations.items():
|
||||||
|
self.observe(run, "http_partition", name, duration,
|
||||||
|
parent="http_total", denominator=http_ms, inclusion="non-overlapping at declared reporting depth", source=source)
|
||||||
|
self.observe(run, "http_partition", "other_unclassified", partition["otherMs"],
|
||||||
|
parent="http_total", denominator=http_ms, inclusion=DEFINITIONS["backendOther"], source=source)
|
||||||
|
else:
|
||||||
|
self.warnings.append(f"{run['simulationId']}: HTTP partition disabled; selected spans overlap or leave request bounds")
|
||||||
|
for key, value in data.items():
|
||||||
|
if key.endswith("Ms") or key in ("responseBodyBytes", "rawSeriesBytes", "xmlBytes", "sampleCount"):
|
||||||
|
self.observe(run, "backend_metrics", key, value,
|
||||||
|
"ms" if key.endswith("Ms") else "bytes" if key.endswith("Bytes") else "count", source=source)
|
||||||
|
elif key == "responseSendAwaitSeconds":
|
||||||
|
self.observe(run, "backend_metrics", key, value * 1000,
|
||||||
|
inclusion="ASGI send waits overlap HTTP and are not pure network time", source=source)
|
||||||
|
phases = data.get("existingPerformance", {}).get("phases", {})
|
||||||
|
for key, phase in phases.items():
|
||||||
|
self.observe(run, "backend_existing_performance", key, phase["inclusiveNs"] / 1e6,
|
||||||
|
parent="http_total", denominator=http_ms, source=source,
|
||||||
|
inclusion="inclusive duration without aligned start/end; excluded from HTTP partition")
|
||||||
|
c = dict(data.get("nativeStages", {}))
|
||||||
|
native = data.get("native", {})
|
||||||
|
c_main = c.get("mainTotalSeconds")
|
||||||
|
for key in C_WALL_FIELDS:
|
||||||
|
value = c.get(key)
|
||||||
|
self.observe(run, "c_wall", key, value * 1000 if number(value) else None,
|
||||||
|
parent="c_main_wall", denominator=c_main * 1000 if number(c_main) else None,
|
||||||
|
inclusion="sequential C phase wall duration; C main lies within observed process lifetime", source=source)
|
||||||
|
self.observe(run, "c_wall", "mainTotalSeconds", c_main * 1000 if number(c_main) else None,
|
||||||
|
inclusion="inclusive C main; does not include loader/exit observation", source=source)
|
||||||
|
c_other = c_main - sum(c[key] for key in C_WALL_FIELDS) if number(c_main) and all(number(c.get(k)) for k in C_WALL_FIELDS) else None
|
||||||
|
self.observe(run, "c_wall", "other_unclassified", c_other * 1000 if number(c_other) else None,
|
||||||
|
parent="c_main_wall", denominator=c_main * 1000 if number(c_main) else None,
|
||||||
|
inclusion="C main minus sequential measured wall phases, without clipping negative differences", source=source)
|
||||||
|
process = dict(data.get("process", {}))
|
||||||
|
child_cpu = (process["childrenUserCpuSeconds"] + process["childrenSystemCpuSeconds"]
|
||||||
|
if all(number(process.get(k)) for k in ("childrenUserCpuSeconds", "childrenSystemCpuSeconds")) else None)
|
||||||
|
for key, value in {"integrationCpuSeconds": native.get("solveCpuSeconds"),
|
||||||
|
"projectionCpuSeconds": c.get("projectionCpuSeconds"),
|
||||||
|
"jsonWriteCpuSeconds": c.get("jsonWriteCpuSeconds")}.items():
|
||||||
|
self.observe(run, "c_cpu", key, value * 1000 if number(value) else None,
|
||||||
|
parent="observed_child_cpu", denominator=child_cpu * 1000 if number(child_cpu) else None,
|
||||||
|
inclusion="CPU duration, separate from wall partition", source=source)
|
||||||
|
for key in ("startMs", "exitObservedMs", "childrenUserCpuSeconds", "childrenSystemCpuSeconds"):
|
||||||
|
value = process.get(key)
|
||||||
|
self.observe(run, "process", key, value * 1000 if number(value) and key.endswith("Seconds") else value,
|
||||||
|
inclusion=DEFINITIONS["process"], source=source)
|
||||||
|
self.observe(run, "process", "observed_lifetime", difference(process, "startMs", "exitObservedMs"),
|
||||||
|
inclusion="includes spawn; excludes subsequent Python result read", source=source)
|
||||||
|
self.observe(run, "process", "observed_child_cpu", child_cpu * 1000 if number(child_cpu) else None,
|
||||||
|
inclusion=DEFINITIONS["process"], source=source)
|
||||||
|
return {"source": source, "httpTotalMs": http_ms, "spanHierarchy": hierarchy,
|
||||||
|
"httpPartition": partition, "cStages": c, "process": process,
|
||||||
|
"build": data.get("build"), "native": native,
|
||||||
|
"existingPerformance": data.get("existingPerformance")}
|
||||||
|
|
||||||
|
def group(self, mode: str, expected_runs: int) -> dict:
|
||||||
|
folder = self.root / f"browser-{mode}"
|
||||||
|
source = folder / "summary.json"
|
||||||
|
summary = self.read(source)
|
||||||
|
rows = summary.get("rows", [])
|
||||||
|
seen: set[str] = set()
|
||||||
|
runs = []
|
||||||
|
for original in rows:
|
||||||
|
if original.get("mode") != mode:
|
||||||
|
raise ValueError(f"Unexpected mode {original.get('mode')!r} in {source}")
|
||||||
|
sid = original["simulationId"]
|
||||||
|
if sid in seen:
|
||||||
|
raise ValueError(f"Duplicate simulationId {sid} in {source}")
|
||||||
|
seen.add(sid)
|
||||||
|
run = {"group": mode, "phase": "warmup" if original.get("warmup") else "measured",
|
||||||
|
"run": original["run"], "simulationId": sid, "original": original}
|
||||||
|
trace_path = folder / original["artifacts"] / "trace.json"
|
||||||
|
trace = self.read(trace_path)
|
||||||
|
if trace.get("timeOrigin") != original.get("timeOrigin"):
|
||||||
|
raise ValueError(f"Browser timeOrigin mismatch: {trace_path}")
|
||||||
|
marks = trace.get("marks", {})
|
||||||
|
if mode == "profiled":
|
||||||
|
request_ids = {r.get("simulationId") for r in trace.get("requests", []) if r.get("kind") == "simulation"}
|
||||||
|
if request_ids != {sid}:
|
||||||
|
raise ValueError(f"Browser trace request ID mismatch: {trace_path}")
|
||||||
|
total = difference(marks, "runClick", "indexedDbCommittedPointer")
|
||||||
|
waterfall = []
|
||||||
|
for (left, left_name), (right, right_name) in zip(AXIS, AXIS[1:]):
|
||||||
|
value = difference(marks, left, right)
|
||||||
|
metric = f"{left_name}_to_{right_name}"
|
||||||
|
entry = {"metric": metric, "fromMark": left, "toMark": right,
|
||||||
|
"startMs": marks.get(left), "endMs": marks.get(right), "durationMs": value,
|
||||||
|
"percentOfClickToCommit": ratio(value, total), "negative": value is not None and value < 0}
|
||||||
|
waterfall.append(entry)
|
||||||
|
self.observe(run, "frontend_waterfall", metric, value, parent="click_to_commit",
|
||||||
|
denominator=total, source=str(trace_path.relative_to(self.root)), inclusion=DEFINITIONS["waterfall"])
|
||||||
|
complete = all(number(part["durationMs"]) for part in waterfall)
|
||||||
|
if complete and not math.isclose(sum(p["durationMs"] for p in waterfall), total, abs_tol=1e-6):
|
||||||
|
raise ValueError(f"Waterfall does not telescope for {sid}")
|
||||||
|
run["frontend"] = {"timeOrigin": trace["timeOrigin"], "marks": marks,
|
||||||
|
"waterfall": waterfall, "clickToCommitMs": total, "waterfallComplete": complete,
|
||||||
|
"negativeIntervals": [part["metric"] for part in waterfall if part["negative"]]}
|
||||||
|
for key, value in original.items():
|
||||||
|
if key in ("run", "timeOrigin") or isinstance(value, bool):
|
||||||
|
continue
|
||||||
|
if number(value) or value is None:
|
||||||
|
unit = "ms" if key.endswith("Ms") else "bytes" if key.endswith("Bytes") else "count" if key.endswith("Count") else "number"
|
||||||
|
self.observe(run, "frontend_metrics", key, value, unit, source=str(source.relative_to(self.root)))
|
||||||
|
native = original.get("native", {})
|
||||||
|
for key, value in native.items():
|
||||||
|
if number(value):
|
||||||
|
unit = "ms" if key.endswith("Seconds") else "simulated_s" if key in ("simulatedUntil", "maxAcceptedStep") else "count"
|
||||||
|
self.observe(run, "native_reported", key, value * 1000 if key.endswith("Seconds") else value, unit,
|
||||||
|
source=str(source.relative_to(self.root)), inclusion="original browser native diagnostic; processWallSeconds includes result read")
|
||||||
|
run["buildCost"] = {"cacheHit": native.get("cacheHit"), "buildKey": native.get("buildKey"),
|
||||||
|
"reportedSeconds": native.get("buildSeconds"),
|
||||||
|
"classification": "cache_hit" if native.get("cacheHit") is True else "cold_build" if native.get("cacheHit") is False else "unknown"}
|
||||||
|
self.observe(run, "build", run["buildCost"]["classification"],
|
||||||
|
native["buildSeconds"] * 1000 if number(native.get("buildSeconds")) else None,
|
||||||
|
source=str(source.relative_to(self.root)), inclusion="warmups separate; cacheHit determines cold/cache classification")
|
||||||
|
backend_path = self.root / "backend-profiled/requests" / sid / "stages.json"
|
||||||
|
if mode == "profiled" and backend_path.exists():
|
||||||
|
backend_data = self.read(backend_path)
|
||||||
|
run["backend"] = self.backend(run, backend_data, str(backend_path.relative_to(self.root)))
|
||||||
|
for key in ("buildKey", "nfev", "acceptedSteps", "solveSeconds"):
|
||||||
|
if backend_data.get("native", {}).get(key) != native.get(key):
|
||||||
|
raise ValueError(f"Browser/backend native diagnostic mismatch for {sid}: {key}")
|
||||||
|
else:
|
||||||
|
run["backend"] = None
|
||||||
|
if mode == "profiled": self.warnings.append(f"{sid}: completed browser row has no backend stages.json yet")
|
||||||
|
runs.append(run)
|
||||||
|
measured = sum(r["phase"] == "measured" for r in runs)
|
||||||
|
if measured != expected_runs:
|
||||||
|
self.warnings.append(f"{mode}: {measured} measured run(s), expected {expected_runs}")
|
||||||
|
if summary.get("errors"):
|
||||||
|
self.warnings.append(f"{mode}: browser summary contains errors; inspect source before interpreting results")
|
||||||
|
environment = self.root / f"backend-{mode}" / "environment.json"
|
||||||
|
return {"source": str(source.relative_to(self.root)), "browser": summary.get("browser"),
|
||||||
|
"node": summary.get("node"), "input": summary.get("input"), "inputSha256": summary.get("inputSha256"),
|
||||||
|
"buildAssetSetSha256": summary.get("buildAssetSetSha256"), "servedAssets": summary.get("servedAssets"),
|
||||||
|
"sourceDefinitions": summary.get("definitions"), "initialNavigation": summary.get("initialNavigation"),
|
||||||
|
"backendEnvironment": self.read(environment) if environment.exists() else None,
|
||||||
|
"measuredRunCount": measured, "warmupRunCount": len(runs) - measured, "runs": runs}
|
||||||
|
|
||||||
|
def aggregate(self) -> list[dict]:
|
||||||
|
buckets: dict[tuple, list[dict]] = {}
|
||||||
|
for observation in self.observations:
|
||||||
|
key = tuple(observation[k] for k in ("group", "phase", "domain", "metric", "unit", "parent"))
|
||||||
|
buckets.setdefault(key, []).append(observation)
|
||||||
|
result = []
|
||||||
|
for key, rows in sorted(buckets.items()):
|
||||||
|
entry = dict(zip(("group", "phase", "domain", "metric", "unit", "parent"), key))
|
||||||
|
entry.update(statistics([r["value"] for r in rows]))
|
||||||
|
entry["percentOfParent"] = statistics([r["percentOfParent"] for r in rows])
|
||||||
|
entry["runs"] = [{"run": r["run"], "simulationId": r["simulationId"]} for r in rows]
|
||||||
|
result.append(entry)
|
||||||
|
return result
|
||||||
|
|
||||||
|
def write(self, expected_runs: int) -> dict:
|
||||||
|
groups = {mode: self.group(mode, expected_runs) for mode in ("control", "profiled")}
|
||||||
|
if groups["control"]["inputSha256"] != groups["profiled"]["inputSha256"]:
|
||||||
|
raise ValueError("Control/profiled input SHA mismatch")
|
||||||
|
if groups["control"]["buildAssetSetSha256"] != groups["profiled"]["buildAssetSetSha256"]:
|
||||||
|
self.warnings.append("Control/profiled frontend asset sets differ")
|
||||||
|
aggregates = self.aggregate()
|
||||||
|
build_costs = [{"group": mode, "run": r["run"], "phase": r["phase"],
|
||||||
|
"simulationId": r["simulationId"], **r["buildCost"]}
|
||||||
|
for mode, group in groups.items() for r in group["runs"]]
|
||||||
|
output = {"schemaVersion": 1, "experimentRoot": str(self.root),
|
||||||
|
"script": str(Path(__file__).resolve().relative_to(REPO)),
|
||||||
|
"scriptSha256": hashlib.sha256(Path(__file__).read_bytes()).hexdigest(),
|
||||||
|
"definitions": DEFINITIONS, "warnings": self.warnings, "sourceFiles": self.sources,
|
||||||
|
"groups": groups, "observations": self.observations, "aggregates": aggregates,
|
||||||
|
"complete": not self.warnings, "buildCosts": build_costs,
|
||||||
|
"warmupBuildCosts": [cost for cost in build_costs if cost["phase"] == "warmup"],
|
||||||
|
"coldBuildCosts": [cost for cost in build_costs if cost["classification"] == "cold_build"]}
|
||||||
|
(self.root / "summary.json").write_text(json.dumps(output, ensure_ascii=False, indent=2, allow_nan=False) + "\n")
|
||||||
|
fields = ["group", "phase", "run", "simulationId", "domain", "metric", "statistic", "value", "unit",
|
||||||
|
"parent", "denominatorValue", "percentOfParent", "count", "missingCount", "inclusion", "source"]
|
||||||
|
with (self.root / "timings.csv").open("w", encoding="utf-8", newline="") as stream:
|
||||||
|
writer = csv.DictWriter(stream, fieldnames=fields)
|
||||||
|
writer.writeheader()
|
||||||
|
for row in self.observations:
|
||||||
|
writer.writerow({**row, "statistic": "run", "count": int(number(row["value"])), "missingCount": int(row["value"] is None)})
|
||||||
|
for entry in aggregates:
|
||||||
|
for stat in ("median", "min", "max"):
|
||||||
|
writer.writerow({**{key: entry[key] for key in ("group", "phase", "domain", "metric", "unit", "parent")},
|
||||||
|
"statistic": stat, "value": entry[stat], "percentOfParent": entry["percentOfParent"][stat],
|
||||||
|
"count": entry["count"], "missingCount": entry["missingCount"],
|
||||||
|
"inclusion": "values and same-run percentages aggregated separately; do not add medians"})
|
||||||
|
return output
|
||||||
|
|
||||||
|
|
||||||
|
def main() -> None:
|
||||||
|
parser = argparse.ArgumentParser(description=__doc__)
|
||||||
|
parser.add_argument("--root", type=Path, default=REPO / "test/web-cost-20260911")
|
||||||
|
parser.add_argument("--expected-runs", type=int, default=3)
|
||||||
|
args = parser.parse_args()
|
||||||
|
if args.expected_runs < 1:
|
||||||
|
parser.error("--expected-runs must be positive")
|
||||||
|
summary = Summary(args.root).write(args.expected_runs)
|
||||||
|
print(json.dumps({"root": summary["experimentRoot"], "complete": summary["complete"],
|
||||||
|
"runs": {mode: {"measured": value["measuredRunCount"], "warmup": value["warmupRunCount"]}
|
||||||
|
for mode, value in summary["groups"].items()},
|
||||||
|
"observations": len(summary["observations"]), "warnings": summary["warnings"]}, ensure_ascii=False, indent=2))
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
@@ -0,0 +1,367 @@
|
|||||||
|
"""Standalone JSON-number contracts; no model, solver or SUNDIALS is required.
|
||||||
|
|
||||||
|
The decimal text may change between correct shortest encoders. These tests use
|
||||||
|
Python's independent JSON decoder and binary64 bits, including the sign of zero,
|
||||||
|
instead of comparing against the old %.17g spelling.
|
||||||
|
"""
|
||||||
|
import ctypes
|
||||||
|
import json
|
||||||
|
import math
|
||||||
|
import os
|
||||||
|
from pathlib import Path
|
||||||
|
import random
|
||||||
|
import re
|
||||||
|
import shlex
|
||||||
|
import shutil
|
||||||
|
import struct
|
||||||
|
import subprocess
|
||||||
|
import tempfile
|
||||||
|
import unittest
|
||||||
|
|
||||||
|
|
||||||
|
ROOT = Path(__file__).resolve().parents[1]
|
||||||
|
JSON_NUMBER = re.compile(rb'-?(?:0|[1-9][0-9]*)(?:\.[0-9]+)?(?:[eE][+-]?[0-9]+)?\Z')
|
||||||
|
BUFFER_BYTES = 64 * 1024
|
||||||
|
SIGN = 1 << 63
|
||||||
|
FRACTION_MASK = (1 << 52) - 1
|
||||||
|
|
||||||
|
|
||||||
|
def double_from_bits(bits):
|
||||||
|
return struct.unpack('=d', struct.pack('=Q', bits))[0]
|
||||||
|
|
||||||
|
|
||||||
|
def double_bits(value):
|
||||||
|
return struct.unpack('=Q', struct.pack('=d', value))[0]
|
||||||
|
|
||||||
|
|
||||||
|
def finite_patterns(random_count=10000):
|
||||||
|
# Every finite exponent binade, with exact powers and significand edges.
|
||||||
|
values = {0, SIGN}
|
||||||
|
for exponent in range(0x7ff):
|
||||||
|
for fraction in (0, 1, (1 << 51) - 1, 1 << 51, FRACTION_MASK):
|
||||||
|
bits = (exponent << 52) | fraction
|
||||||
|
values.update((bits, bits | SIGN))
|
||||||
|
# Decimal carry/notation boundaries and adjacent representable values.
|
||||||
|
for exponent in range(-323, 309):
|
||||||
|
center = float(f'1e{exponent}')
|
||||||
|
for value in (math.nextafter(center, 0), center, math.nextafter(center, math.inf)):
|
||||||
|
if math.isfinite(value):
|
||||||
|
values.update((double_bits(value), double_bits(-value)))
|
||||||
|
random_source = random.Random(0x5259555F4A534F4E)
|
||||||
|
added = 0
|
||||||
|
while added < random_count:
|
||||||
|
bits = random_source.getrandbits(64)
|
||||||
|
if (bits >> 52) & 0x7ff != 0x7ff:
|
||||||
|
values.add(bits)
|
||||||
|
added += 1
|
||||||
|
return sorted(values)
|
||||||
|
|
||||||
|
|
||||||
|
class WriteStatus(ctypes.Structure):
|
||||||
|
_fields_ = [
|
||||||
|
('opened', ctypes.c_int), ('written', ctypes.c_int), ('closed', ctypes.c_int),
|
||||||
|
('start', ctypes.c_longlong), ('end', ctypes.c_longlong), ('final_position', ctypes.c_longlong),
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
|
HARNESS = r'''
|
||||||
|
#include <stdio.h>
|
||||||
|
#include <stdint.h>
|
||||||
|
#include <stddef.h>
|
||||||
|
#include <string.h>
|
||||||
|
#include "json_numbers.h"
|
||||||
|
|
||||||
|
/* Faults intercept only the production writer's fwrite calls in this temporary
|
||||||
|
translation unit. Real stdio still writes the accepted prefix. */
|
||||||
|
int test_write_fault_mode = 0;
|
||||||
|
int test_write_call_count = 0;
|
||||||
|
#ifdef TEST_FWRITE_FAULTS
|
||||||
|
static size_t test_fwrite(const void *data, size_t size, size_t count, FILE *stream) {
|
||||||
|
test_write_call_count++;
|
||||||
|
if (test_write_fault_mode == 1 ||
|
||||||
|
(test_write_fault_mode == 3 && test_write_call_count >= 2)) return 0;
|
||||||
|
if (test_write_fault_mode == 2)
|
||||||
|
return count ? fwrite(data, size, count - 1, stream) : 0;
|
||||||
|
return fwrite(data, size, count, stream);
|
||||||
|
}
|
||||||
|
#define fwrite test_fwrite
|
||||||
|
#include "json_numbers.c"
|
||||||
|
#undef fwrite
|
||||||
|
#endif
|
||||||
|
|
||||||
|
typedef struct {
|
||||||
|
int opened, written, closed;
|
||||||
|
long long start, end, final_position;
|
||||||
|
} TestWriteStatus;
|
||||||
|
|
||||||
|
void test_array_file(const char *path, const double *values, size_t count,
|
||||||
|
size_t stride, size_t prefix_bytes, TestWriteStatus *status) {
|
||||||
|
memset(status, 0, sizeof(*status));
|
||||||
|
FILE *stream = fopen(path, "wb");
|
||||||
|
if (!stream) return;
|
||||||
|
status->opened = 1;
|
||||||
|
for (size_t i = 0; i < prefix_bytes; i++) fputc('p', stream);
|
||||||
|
status->start = (long long)ftell(stream);
|
||||||
|
status->written = native_json_write_array(stream, values, count, stride);
|
||||||
|
status->end = (long long)ftell(stream);
|
||||||
|
if (status->written) fputs("TAIL", stream);
|
||||||
|
status->final_position = (long long)ftell(stream);
|
||||||
|
status->closed = fclose(stream) == 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
void test_number_file(const char *path, double value, TestWriteStatus *status) {
|
||||||
|
memset(status, 0, sizeof(*status));
|
||||||
|
FILE *stream = fopen(path, "wb");
|
||||||
|
if (!stream) return;
|
||||||
|
status->opened = 1;
|
||||||
|
status->written = native_json_write_number(stream, value);
|
||||||
|
status->end = status->final_position = (long long)ftell(stream);
|
||||||
|
status->closed = fclose(stream) == 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/* The number writer flushes its own block to FILE, not FILE's stdio buffer.
|
||||||
|
The caller must still propagate a delayed fclose failure. */
|
||||||
|
int test_delayed_close_failure(const char *path, int *written, int *closed) {
|
||||||
|
FILE *stream = fopen(path, "wb");
|
||||||
|
if (!stream) return 0;
|
||||||
|
char buffer[4096];
|
||||||
|
if (setvbuf(stream, buffer, _IOFBF, sizeof(buffer))) { fclose(stream); return 0; }
|
||||||
|
*written = native_json_write_number(stream, 0.1);
|
||||||
|
*closed = fclose(stream) == 0;
|
||||||
|
return 1;
|
||||||
|
}
|
||||||
|
'''
|
||||||
|
|
||||||
|
|
||||||
|
class NativeJsonWriterTests(unittest.TestCase):
|
||||||
|
@classmethod
|
||||||
|
def setUpClass(cls):
|
||||||
|
command = shlex.split(os.environ.get('CC', ''))
|
||||||
|
if not command:
|
||||||
|
compiler = shutil.which('gcc') or shutil.which('clang')
|
||||||
|
if not compiler:
|
||||||
|
raise unittest.SkipTest('A native C compiler is required')
|
||||||
|
command = [compiler]
|
||||||
|
cls.compiler = command
|
||||||
|
cls.directory = tempfile.TemporaryDirectory(prefix='native-json-writer-')
|
||||||
|
cls.addClassCleanup(cls.directory.cleanup)
|
||||||
|
cls.root = Path(cls.directory.name)
|
||||||
|
cls.library = cls.build_library('ordinary', faults=False)
|
||||||
|
cls.fault_library = cls.build_library('faults', faults=True)
|
||||||
|
cls.portable_library = cls.build_library('portable-64-bit', faults=False, only_64_bit=True)
|
||||||
|
cls.fault_mode = ctypes.c_int.in_dll(cls.fault_library, 'test_write_fault_mode')
|
||||||
|
cls.fault_calls = ctypes.c_int.in_dll(cls.fault_library, 'test_write_call_count')
|
||||||
|
|
||||||
|
@classmethod
|
||||||
|
def build_library(cls, name, *, faults, only_64_bit=False):
|
||||||
|
source = cls.root / f'{name}.c'
|
||||||
|
source.write_text(HARNESS)
|
||||||
|
library_path = cls.root / (name + ('.dll' if os.name == 'nt' else '.so'))
|
||||||
|
command = cls.compiler + ['-std=c11', '-O2', '-Wall', '-Wextra', '-Werror',
|
||||||
|
'-ffp-contract=off', '-fno-fast-math', '-shared']
|
||||||
|
if os.name != 'nt':
|
||||||
|
command.append('-fPIC')
|
||||||
|
if faults:
|
||||||
|
command.append('-DTEST_FWRITE_FAULTS')
|
||||||
|
if only_64_bit:
|
||||||
|
command.append('-DRYU_ONLY_64_BIT_OPS')
|
||||||
|
command += ['-I', str(ROOT / 'native/include'), '-I', str(ROOT / 'native/runtime'),
|
||||||
|
str(source)]
|
||||||
|
if not faults:
|
||||||
|
command.append(str(ROOT / 'native/runtime/json_numbers.c'))
|
||||||
|
command += [str(ROOT / 'native/encoding/ryu/d2s.c'), '-lm', '-o', str(library_path)]
|
||||||
|
compiled = subprocess.run(command, capture_output=True, text=True, timeout=60)
|
||||||
|
if compiled.returncode:
|
||||||
|
raise AssertionError(compiled.stderr)
|
||||||
|
library = ctypes.CDLL(str(library_path))
|
||||||
|
if os.name == 'nt':
|
||||||
|
import _ctypes
|
||||||
|
cls.addClassCleanup(_ctypes.FreeLibrary, library._handle)
|
||||||
|
library.native_json_format_double.argtypes = [ctypes.c_void_p, ctypes.c_double]
|
||||||
|
library.native_json_format_double.restype = ctypes.c_int
|
||||||
|
library.native_json_write_number.argtypes = [ctypes.c_void_p, ctypes.c_double]
|
||||||
|
library.native_json_write_number.restype = ctypes.c_int
|
||||||
|
library.native_json_write_array.argtypes = [ctypes.c_void_p, ctypes.POINTER(ctypes.c_double), ctypes.c_size_t, ctypes.c_size_t]
|
||||||
|
library.native_json_write_array.restype = ctypes.c_int
|
||||||
|
library.test_array_file.argtypes = [ctypes.c_char_p, ctypes.POINTER(ctypes.c_double),
|
||||||
|
ctypes.c_size_t, ctypes.c_size_t, ctypes.c_size_t,
|
||||||
|
ctypes.POINTER(WriteStatus)]
|
||||||
|
library.test_array_file.restype = None
|
||||||
|
library.test_number_file.argtypes = [ctypes.c_char_p, ctypes.c_double, ctypes.POINTER(WriteStatus)]
|
||||||
|
library.test_number_file.restype = None
|
||||||
|
library.test_delayed_close_failure.argtypes = [ctypes.c_char_p, ctypes.POINTER(ctypes.c_int), ctypes.POINTER(ctypes.c_int)]
|
||||||
|
library.test_delayed_close_failure.restype = ctypes.c_int
|
||||||
|
return library
|
||||||
|
|
||||||
|
def encode(self, value, library=None):
|
||||||
|
# Sentinels bracket the promised 32-byte output, with no assumption that
|
||||||
|
# the returned token is NUL-terminated.
|
||||||
|
storage = (ctypes.c_ubyte * 34)(*([0xA5] * 34))
|
||||||
|
length = (library or self.library).native_json_format_double(ctypes.byref(storage, 1), value)
|
||||||
|
self.assertEqual((storage[0], storage[33]), (0xA5, 0xA5))
|
||||||
|
self.assertGreater(length, 0)
|
||||||
|
self.assertLessEqual(length, 32)
|
||||||
|
return bytes(storage[1:1 + length])
|
||||||
|
|
||||||
|
def assert_roundtrip(self, text, expected_bits):
|
||||||
|
self.assertRegex(text, JSON_NUMBER, f'Invalid JSON token for {expected_bits:016x}')
|
||||||
|
decoded = json.loads(text)
|
||||||
|
self.assertEqual(double_bits(float(decoded)), expected_bits,
|
||||||
|
f'{expected_bits:016x} became {text!r} then {decoded!r}')
|
||||||
|
|
||||||
|
def write_array(self, values, *, count=None, stride=1, prefix=0, library=None):
|
||||||
|
array = (ctypes.c_double * len(values))(*values) if values else None
|
||||||
|
status = WriteStatus()
|
||||||
|
output = self.root / 'array.json'
|
||||||
|
(library or self.library).test_array_file(os.fsencode(output), array,
|
||||||
|
len(values) if count is None else count, stride, prefix, ctypes.byref(status))
|
||||||
|
self.assertTrue(status.opened)
|
||||||
|
return output.read_bytes(), status
|
||||||
|
|
||||||
|
def test_binary64_boundaries_and_seeded_random_values_roundtrip(self):
|
||||||
|
for bits in finite_patterns():
|
||||||
|
self.assert_roundtrip(self.encode(double_from_bits(bits)), bits)
|
||||||
|
|
||||||
|
def test_64_bit_fallback_roundtrips_the_same_binary64_corpus(self):
|
||||||
|
patterns = finite_patterns()
|
||||||
|
for bits in patterns:
|
||||||
|
self.assert_roundtrip(self.encode(double_from_bits(bits), self.portable_library), bits)
|
||||||
|
values = [double_from_bits(bits) for bits in patterns]
|
||||||
|
ordinary, ordinary_status = self.write_array(values)
|
||||||
|
portable, portable_status = self.write_array(values, library=self.portable_library)
|
||||||
|
self.assertTrue(ordinary_status.written and ordinary_status.closed)
|
||||||
|
self.assertTrue(portable_status.written and portable_status.closed)
|
||||||
|
self.assertEqual(portable, ordinary)
|
||||||
|
|
||||||
|
def test_plain_decimal_is_used_only_when_it_shortens_the_token(self):
|
||||||
|
# Tie cases deliberately keep scientific notation; far exponents must
|
||||||
|
# never be expanded to hundreds of zeroes in the 32-byte destination.
|
||||||
|
cases = ((10.0, b'10'), (12.0, b'12'), (-12.0, b'-12'),
|
||||||
|
(.1, b'0.1'), (-.1, b'-0.1'), (123.45, b'123.45'),
|
||||||
|
(100.0, b'1E2'), (.01, b'1E-2'), (1e100, b'1E100'),
|
||||||
|
(double_from_bits(1), b'5E-324'))
|
||||||
|
for library in (self.library, self.portable_library):
|
||||||
|
for value, expected in cases:
|
||||||
|
with self.subTest(value=value, expected=expected):
|
||||||
|
token = self.encode(value, library)
|
||||||
|
self.assertEqual(token, expected)
|
||||||
|
self.assert_roundtrip(token, double_bits(value))
|
||||||
|
|
||||||
|
def test_zero_sign_and_single_number_file(self):
|
||||||
|
self.assertEqual(self.encode(0.0), b'0')
|
||||||
|
self.assertEqual(self.encode(-0.0), b'-0.0')
|
||||||
|
output = self.root / 'number.json'
|
||||||
|
for bits in (0, SIGN, 1, SIGN | 1, 0x0010000000000000, 0x7fefffffffffffff):
|
||||||
|
status = WriteStatus()
|
||||||
|
self.library.test_number_file(os.fsencode(output), double_from_bits(bits), ctypes.byref(status))
|
||||||
|
self.assertTrue(status.opened and status.written and status.closed)
|
||||||
|
data = output.read_bytes()
|
||||||
|
self.assertEqual(status.end, len(data))
|
||||||
|
self.assert_roundtrip(data, bits)
|
||||||
|
|
||||||
|
def test_array_crosses_block_boundaries_without_changing_offsets(self):
|
||||||
|
cases = [([], 2), ([0.0] * 32767, BUFFER_BYTES - 1),
|
||||||
|
([-0.0] + [0.0] * 32765, BUFFER_BYTES),
|
||||||
|
([0.0] * 32768, BUFFER_BYTES + 1),
|
||||||
|
([double_from_bits(bits) for bits in finite_patterns(0)[::3]], None)]
|
||||||
|
for values, expected_length in cases:
|
||||||
|
for prefix in (0, 37, BUFFER_BYTES - 1):
|
||||||
|
with self.subTest(values=len(values), expected_length=expected_length, prefix=prefix):
|
||||||
|
data, status = self.write_array(values, prefix=prefix)
|
||||||
|
self.assertTrue(status.written and status.closed)
|
||||||
|
self.assertEqual(status.start, prefix)
|
||||||
|
self.assertEqual(data[:prefix], b'p' * prefix)
|
||||||
|
self.assertEqual(data[status.end:], b'TAIL')
|
||||||
|
self.assertEqual(status.final_position, len(data))
|
||||||
|
token = data[status.start:status.end]
|
||||||
|
if expected_length is not None:
|
||||||
|
self.assertEqual(len(token), expected_length)
|
||||||
|
decoded = json.loads(token)
|
||||||
|
self.assertEqual([double_bits(float(v)) for v in decoded], [double_bits(v) for v in values])
|
||||||
|
|
||||||
|
def test_strided_array_keeps_selected_column_order(self):
|
||||||
|
selected = [double_from_bits(bits) for bits in finite_patterns(0)[::13]]
|
||||||
|
for stride in (1, 3, 17):
|
||||||
|
with self.subTest(stride=stride):
|
||||||
|
values = [math.nan] * (len(selected) * stride)
|
||||||
|
values[::stride] = selected
|
||||||
|
data, status = self.write_array(values, count=len(selected), stride=stride)
|
||||||
|
self.assertTrue(status.written and status.closed)
|
||||||
|
decoded = json.loads(data[:status.end])
|
||||||
|
self.assertEqual([double_bits(float(v)) for v in decoded], [double_bits(v) for v in selected])
|
||||||
|
|
||||||
|
def test_invalid_file_pointer_and_array_bounds_fail_before_access(self):
|
||||||
|
value = (ctypes.c_double * 1)(.1)
|
||||||
|
self.assertEqual(self.library.native_json_write_number(None, .1), 0)
|
||||||
|
self.assertEqual(self.library.native_json_write_array(None, value, 1, 1), 0)
|
||||||
|
for values, count, stride in (([], 1, 1), ([.1], 2, 0),
|
||||||
|
([.1], 2, ctypes.c_size_t(-1).value)):
|
||||||
|
with self.subTest(count=count, stride=stride):
|
||||||
|
data, status = self.write_array(values, count=count, stride=stride)
|
||||||
|
self.assertTrue(status.opened and status.closed)
|
||||||
|
self.assertFalse(status.written)
|
||||||
|
self.assertEqual(data, b'')
|
||||||
|
# A single sample never advances its pointer; an enormous stride is safe.
|
||||||
|
data, status = self.write_array([.1], count=1, stride=ctypes.c_size_t(-1).value)
|
||||||
|
self.assertTrue(status.written and status.closed)
|
||||||
|
self.assertEqual(json.loads(data[:status.end]), [.1])
|
||||||
|
|
||||||
|
def test_nonfinite_numbers_fail_instead_of_emitting_invalid_json(self):
|
||||||
|
output = self.root / 'nonfinite.json'
|
||||||
|
for bits in (0x7ff0000000000000, 0xfff0000000000000,
|
||||||
|
0x7ff8000000000000, 0xfff8000000000001, 0x7ff0000000000001):
|
||||||
|
value = double_from_bits(bits)
|
||||||
|
with self.subTest(bits=f'{bits:016x}'):
|
||||||
|
token = ctypes.create_string_buffer(32)
|
||||||
|
self.assertEqual(self.library.native_json_format_double(token, value), 0)
|
||||||
|
status = WriteStatus()
|
||||||
|
self.library.test_number_file(os.fsencode(output), value, ctypes.byref(status))
|
||||||
|
self.assertTrue(status.opened and status.closed)
|
||||||
|
self.assertFalse(status.written)
|
||||||
|
self.assertEqual(output.read_bytes(), b'')
|
||||||
|
for finite_prefix in ([], [0.0] * (BUFFER_BYTES + 1)):
|
||||||
|
data, status = self.write_array(finite_prefix + [value])
|
||||||
|
self.assertFalse(status.written)
|
||||||
|
self.assertTrue(status.closed)
|
||||||
|
self.assertNotIn(b'NaN', data)
|
||||||
|
self.assertNotIn(b'Infinity', data)
|
||||||
|
|
||||||
|
def test_zero_and_short_writes_are_reported_without_stdio_error_flag(self):
|
||||||
|
# The injected fwrite can return short without setting FILE's error bit;
|
||||||
|
# relying only on ferror/fclose would incorrectly report success.
|
||||||
|
output = self.root / 'fault-number.json'
|
||||||
|
try:
|
||||||
|
for mode in (1, 2):
|
||||||
|
with self.subTest(mode=mode):
|
||||||
|
self.fault_mode.value = mode
|
||||||
|
self.fault_calls.value = 0
|
||||||
|
status = WriteStatus()
|
||||||
|
self.fault_library.test_number_file(os.fsencode(output), .1, ctypes.byref(status))
|
||||||
|
self.assertTrue(status.opened and status.closed)
|
||||||
|
self.assertFalse(status.written)
|
||||||
|
self.assertGreater(self.fault_calls.value, 0)
|
||||||
|
self.fault_calls.value = 0
|
||||||
|
_, status = self.write_array([1.0, 2.0], library=self.fault_library)
|
||||||
|
self.assertFalse(status.written)
|
||||||
|
self.assertTrue(status.closed)
|
||||||
|
self.fault_mode.value = 3
|
||||||
|
self.fault_calls.value = 0
|
||||||
|
data, status = self.write_array([0.0] * (BUFFER_BYTES * 2), library=self.fault_library)
|
||||||
|
self.assertFalse(status.written)
|
||||||
|
self.assertTrue(status.closed)
|
||||||
|
self.assertGreaterEqual(self.fault_calls.value, 2)
|
||||||
|
self.assertTrue(data, 'The initial successful block must survive a later write failure')
|
||||||
|
finally:
|
||||||
|
self.fault_mode.value = 0
|
||||||
|
self.fault_calls.value = 0
|
||||||
|
|
||||||
|
@unittest.skipUnless(os.name == 'posix' and Path('/dev/full').exists(), '/dev/full is needed for a real delayed I/O failure')
|
||||||
|
def test_caller_must_check_delayed_fclose_failure(self):
|
||||||
|
written, closed = ctypes.c_int(), ctypes.c_int()
|
||||||
|
self.assertEqual(self.library.test_delayed_close_failure(b'/dev/full', ctypes.byref(written), ctypes.byref(closed)), 1)
|
||||||
|
self.assertEqual(written.value, 1, 'A small buffered token should not force FILE fflush')
|
||||||
|
self.assertEqual(closed.value, 0, 'FILE close must expose the delayed device error')
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == '__main__':
|
||||||
|
unittest.main()
|
||||||
@@ -0,0 +1,143 @@
|
|||||||
|
"""The HTTP fast path must preserve real native values and task semantics."""
|
||||||
|
from dataclasses import replace
|
||||||
|
import json
|
||||||
|
from pathlib import Path
|
||||||
|
import tempfile
|
||||||
|
from types import SimpleNamespace
|
||||||
|
import unittest
|
||||||
|
from unittest.mock import patch
|
||||||
|
from uuid import uuid4
|
||||||
|
|
||||||
|
import asyncio
|
||||||
|
from app.main import app, _register_simulation_task, _request_simulation_task_cancel, simulation_event_stream
|
||||||
|
from app.simulation.backends import simulation_config
|
||||||
|
from app.simulation.native_codegen.build import build_native
|
||||||
|
from app.simulation.native_codegen.compiler import compile_native_program
|
||||||
|
from app.simulation.native_codegen.input import load_input
|
||||||
|
from app.simulation.native_codegen.runner import execute_native
|
||||||
|
from app.simulation.native_codegen.transport import NativeSeriesJson, read_indexed_result, serialize_result_parts
|
||||||
|
from app.main import compile_system_xml_network
|
||||||
|
|
||||||
|
|
||||||
|
class AsgiClient:
|
||||||
|
"""Exercise real routing/response bodies without an optional HTTP client dependency."""
|
||||||
|
def __init__(self, application): self.application = application
|
||||||
|
def post(self, path, *, content=b'', headers=None): return self.request('POST', path, content, headers)
|
||||||
|
def get(self, path): return self.request('GET', path, b'', None)
|
||||||
|
def request(self, method, path, content, headers):
|
||||||
|
async def run():
|
||||||
|
messages = []
|
||||||
|
scope = {'type':'http', 'asgi':{'version':'3.0','spec_version':'2.4'},
|
||||||
|
'http_version':'1.1','method':method,'scheme':'http','path':path,
|
||||||
|
'raw_path':path.encode(),'query_string':b'', 'root_path':'',
|
||||||
|
'headers':[(k.lower().encode(),v.encode()) for k,v in (headers or {}).items()],
|
||||||
|
'server':('testserver',80),'client':('127.0.0.1',1234)}
|
||||||
|
async def receive(): return {'type':'http.request','body':content,'more_body':False}
|
||||||
|
async def send(message): messages.append(message)
|
||||||
|
await self.application(scope,receive,send)
|
||||||
|
status = next(m['status'] for m in messages if m['type']=='http.response.start')
|
||||||
|
body = b''.join(m.get('body',b'') for m in messages if m['type']=='http.response.body')
|
||||||
|
return SimpleNamespace(status_code=status,content=body,json=lambda:json.loads(body))
|
||||||
|
return asyncio.run(run())
|
||||||
|
|
||||||
|
|
||||||
|
class NativeResultTransportTests(unittest.TestCase):
|
||||||
|
@classmethod
|
||||||
|
def setUpClass(cls):
|
||||||
|
cls.temp = tempfile.TemporaryDirectory(prefix='test-native-transport-')
|
||||||
|
cls.root = Path(cls.temp.name)
|
||||||
|
cls.xml = Path('tests/fixtures/native-skill-test.xml').read_bytes().replace(b'tStop="10"', b'tStop="0.1"')
|
||||||
|
source = cls.root/'input.xml'; source.write_bytes(cls.xml)
|
||||||
|
_, document = load_input(source)
|
||||||
|
cls.config = simulation_config(document.simulation)
|
||||||
|
cls.build = build_native(compile_native_program(compile_system_xml_network(document)))
|
||||||
|
cls.normal = execute_native(cls.build, cls.config, .001, run_dir=cls.root/'ordinary')
|
||||||
|
cls.indexed = execute_native(cls.build, cls.config, .001, run_dir=cls.root/'indexed', raw_series=True)
|
||||||
|
|
||||||
|
@classmethod
|
||||||
|
def tearDownClass(cls):
|
||||||
|
cls.temp.cleanup()
|
||||||
|
|
||||||
|
def test_real_native_series_are_equal_without_large_python_parse(self):
|
||||||
|
self.assertIsInstance(self.indexed['series'], NativeSeriesJson)
|
||||||
|
self.assertEqual(self.indexed['series'].materialize(), self.normal['series'])
|
||||||
|
self.assertEqual(self.indexed['series'].sample_count, len(self.normal['series']['time']))
|
||||||
|
for key in ('final', 'finalState', 'nfev', 'acceptedSteps', 'rejectedSteps'):
|
||||||
|
self.assertEqual(self.indexed[key], self.normal[key])
|
||||||
|
self.assertGreater(len(self.indexed['series'].data), 100000)
|
||||||
|
original = json.loads
|
||||||
|
sizes = []
|
||||||
|
def small_only(value):
|
||||||
|
sizes.append(len(value))
|
||||||
|
self.assertLess(len(value), 50000, 'Raw series was decoded through Python')
|
||||||
|
return original(value)
|
||||||
|
with patch('app.simulation.native_codegen.transport.json', SimpleNamespace(loads=small_only)):
|
||||||
|
payload = read_indexed_result(self.root/'indexed/result.json', self.root/'indexed/result-index.json')
|
||||||
|
self.assertEqual(len(sizes), 2) # index and small metadata only
|
||||||
|
self.assertEqual(payload['series'].data, self.indexed['series'].data)
|
||||||
|
|
||||||
|
def test_real_http_stream_and_retained_task_get_keep_same_schema(self):
|
||||||
|
client = AsgiClient(app)
|
||||||
|
ident = 'transport-'+uuid4().hex
|
||||||
|
response = client.post('/api/system-xml/simulate-stream', content=self.xml, headers={'X-Simulation-Id':ident})
|
||||||
|
self.assertEqual(response.status_code, 200)
|
||||||
|
events = [json.loads(line) for line in response.content.splitlines()]
|
||||||
|
result = next(event['result'] for event in events if event['event'] == 'result')
|
||||||
|
self.assertTrue(result['success'])
|
||||||
|
self.assertEqual(result['simulatedUntil'], .1)
|
||||||
|
self.assertEqual(result['diagnostics']['sampleCount'], len(result['series']['time']))
|
||||||
|
# The bytes survive deletion of the worker directory and repeated task reads.
|
||||||
|
for _ in range(2):
|
||||||
|
retained = client.get('/api/system-xml/simulations/'+ident)
|
||||||
|
self.assertEqual(retained.status_code, 200)
|
||||||
|
self.assertEqual(retained.json()['result'], result)
|
||||||
|
synchronous = client.post('/api/system-xml/simulate', content=self.xml).json()
|
||||||
|
for key in ('series', 'final', 'variables', 'model', 'simulation'):
|
||||||
|
self.assertEqual(synchronous[key], result[key])
|
||||||
|
|
||||||
|
def test_cancelled_raw_stream_keeps_partial_result_and_public_status(self):
|
||||||
|
for reason, status in [('user', 'stopped'), ('stalled', 'stalled')]:
|
||||||
|
task = _register_simulation_task('transport-'+uuid4().hex)
|
||||||
|
_request_simulation_task_cancel(task, reason)
|
||||||
|
body = b''.join(part.encode() if isinstance(part, str) else part
|
||||||
|
for part in simulation_event_stream(self.xml, task=task, raw_series=True))
|
||||||
|
result = next(event['result'] for event in map(json.loads, body.splitlines()) if event['event']=='result')
|
||||||
|
self.assertEqual(result['status'], status)
|
||||||
|
self.assertTrue(result['partial'])
|
||||||
|
self.assertLess(result['simulatedUntil'], .1)
|
||||||
|
self.assertEqual(result['series']['time'][-1], result['simulatedUntil'])
|
||||||
|
self.assertEqual(AsgiClient(app).get('/api/system-xml/simulations/'+task.simulation_id).json()['result'], result)
|
||||||
|
|
||||||
|
def test_index_corruption_or_truncated_output_is_rejected(self):
|
||||||
|
directory = self.root/'corrupt'; directory.mkdir(exist_ok=True)
|
||||||
|
data = (self.root/'indexed/result.json').read_bytes()
|
||||||
|
original = json.loads((self.root/'indexed/result-index.json').read_bytes())
|
||||||
|
output = directory/'result.json'; index = directory/'index.json'
|
||||||
|
output.write_bytes(data)
|
||||||
|
for change in ({'version':2}, {'version':True}, {'version':1.0}, {'seriesStart':True}, {'seriesStart':-1}, {'seriesEnd':len(data)+1},
|
||||||
|
{'resultBytes':len(data)-1}, {'sampleCount':-1}, {'seriesStart':original['seriesStart']+1}):
|
||||||
|
with self.subTest(change=change):
|
||||||
|
index.write_text(json.dumps(original | change))
|
||||||
|
with self.assertRaises(ValueError): read_indexed_result(output,index)
|
||||||
|
index.write_text(json.dumps(original)); output.write_bytes(data[:-2])
|
||||||
|
with self.assertRaises(ValueError): read_indexed_result(output,index)
|
||||||
|
|
||||||
|
def test_json_framing_and_escaping_cannot_confuse_raw_series(self):
|
||||||
|
values = {'time':[0,1], 'odd\\"},"final":{\n温度':[1e-300,-0.0]}
|
||||||
|
raw = NativeSeriesJson(json.dumps(values,ensure_ascii=False).encode(),2)
|
||||||
|
for payload in ({'result':{'series':raw}},
|
||||||
|
{'event':'result','message':'\n"series":{},"result":null',
|
||||||
|
'result':{'series':raw,'label':'\\"雪\n','success':True}}):
|
||||||
|
encoded = b''.join(serialize_result_parts(payload))
|
||||||
|
expected = dict(payload, result=dict(payload['result'], series=values))
|
||||||
|
self.assertEqual(json.loads(encoded), expected)
|
||||||
|
self.assertEqual(json.loads(b''.join(serialize_result_parts({'event':'progress'}))), {'event':'progress'})
|
||||||
|
|
||||||
|
def test_solve_only_raw_series_is_empty_object(self):
|
||||||
|
result = execute_native(self.build, replace(self.config,t_stop=.001),.001,
|
||||||
|
run_dir=self.root/'solve-only', raw_series=True,record_samples=False)
|
||||||
|
self.assertEqual(result['series'].data,b'{}')
|
||||||
|
self.assertEqual(result['series'].sample_count,0)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == '__main__': unittest.main()
|
||||||