优化原生结果编码传输与浏览器缓存,记录八路性能基线
原生结果series通过字节索引直传,C端使用Ryu精确回读编码和64 KiB批量写出;网页采用Float64缓存和CSV工作线程,减少结果处理与保存等待。 补充八路AME曲线核查、全流程分阶段计时、独立编码基准和复现工具,固定后续优化采用修正八路及rtol=1e-8。C写出1.1808→0.1638 s,点击到可查看8.0100→6.9756 s。 验证:最终10项编码专项、29项相关后端回归通过;8份原生结果逐位一致,16次网页结果/CSV/刷新恢复通过。前端构建及缓存/CSV专项在本轮结果处理工作中通过。环境、原始大结果与临时构建不纳入Git。
This commit is contained in:
1 parent
808c484f5b
commit
3bc4be3c06
55 files changed
+20522
-141
No files matched your search
@@ -0,0 +1,314 @@
|
||||
"""Isolated, Linux/GCC-only CVODE cost diagnostic; never a production benchmark.
|
||||
|
||||
Example (prepare only by default; --run builds and executes serially)::
|
||||
|
||||
.venv/bin/python tests/manual/native_compute_profile.py \
|
||||
--cache-dir test/.../cache/BUILD_KEY --request-stages test/.../stages.json \
|
||||
--output-dir test/native-compute-profile --run --warmups 1 --repeats 3
|
||||
|
||||
The cached executable is the unmodified control. Only a private native source
|
||||
copy receives wall-clock scopes and sparse CVODE counter reads. The generated
|
||||
model and numerical expressions, compiler FP flags, solver and libraries stay
|
||||
unchanged. Full output/state/counter equality is checked outside run timing.
|
||||
Inclusive durations are nested: only exclusiveSeconds may be added. Clock and
|
||||
bookkeeping overhead remain in measured totals; compare against the control.
|
||||
No property/pipe/libc allocation is inferred from this outer-only diagnostic.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
from hashlib import sha256
|
||||
import json
|
||||
import os
|
||||
from pathlib import Path
|
||||
import re
|
||||
import shutil
|
||||
import statistics
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[2]
|
||||
sys.path.insert(0, str(ROOT))
|
||||
from app.simulation.native_codegen.build import LIBRARIES, toolchain
|
||||
|
||||
CATEGORIES = ["integration", "cvode_step", "rhs", "poll", "accept", "append", "dense_output", "dense_setup", "dense_solve"]
|
||||
COUNTERS = ["rhs", "linear_rhs", "nonlinear_iterations", "nonlinear_failures"]
|
||||
PROFILE_HEADER = r'''
|
||||
#ifndef NATIVE_COMPUTE_PROFILE_H
|
||||
#define NATIVE_COMPUTE_PROFILE_H
|
||||
#include <stddef.h>
|
||||
enum { @CATEGORIES@, PROFILE_CATEGORY_COUNT };
|
||||
typedef struct ProfileScope { double start, children; int id, domain, active; struct ProfileScope *parent; } ProfileScope;
|
||||
ProfileScope profile_begin(int id);
|
||||
void profile_link(ProfileScope *scope);
|
||||
void profile_end(ProfileScope *scope);
|
||||
void profile_counter(int slot, int status, long int value);
|
||||
void profile_counter_segment(void);
|
||||
void profile_dump(void);
|
||||
#define PROFILE_SCOPE(id) ProfileScope profile_scope __attribute__((cleanup(profile_end)))=profile_begin(id); profile_link(&profile_scope)
|
||||
#endif
|
||||
'''
|
||||
PROFILE_SOURCE = r'''
|
||||
#include "compute_profile.h"
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <time.h>
|
||||
#include <stdint.h>
|
||||
typedef struct { unsigned long long count; double inclusive, exclusive; } ProfileTotal;
|
||||
static ProfileTotal totals[2][PROFILE_CATEGORY_COUNT];
|
||||
static ProfileScope *parent;
|
||||
static unsigned long long counters[4], segments, counter_errors;
|
||||
static double now(void) { struct timespec t; clock_gettime(CLOCK_MONOTONIC,&t); return t.tv_sec+t.tv_nsec*1e-9; }
|
||||
ProfileScope profile_begin(int id) {
|
||||
ProfileScope s={0}; s.id=id; s.domain=(id==PROFILE_INTEGRATION || (parent && parent->domain));
|
||||
s.parent=parent; s.active=1; s.start=now(); return s;
|
||||
}
|
||||
void profile_link(ProfileScope *s) { parent=s; }
|
||||
void profile_end(ProfileScope *s) {
|
||||
if(!s->active) return;
|
||||
double elapsed=now()-s->start;
|
||||
if(parent!=s) { fputs("Invalid profile scope nesting\n",stderr); exit(74); }
|
||||
ProfileTotal *t=&totals[s->domain][s->id];
|
||||
t->count++; t->inclusive+=elapsed; t->exclusive+=elapsed-s->children;
|
||||
parent=s->parent;
|
||||
if(parent) parent->children+=elapsed;
|
||||
s->active=0;
|
||||
}
|
||||
void profile_counter(int slot,int status,long int value) {
|
||||
if(status || value<0) counter_errors++; else counters[slot]+=(unsigned long long)value;
|
||||
}
|
||||
void profile_counter_segment(void) { segments++; }
|
||||
void profile_dump(void) {
|
||||
const char *path=getenv("NATIVE_COMPUTE_PROFILE"); if(!path)return;
|
||||
FILE *f=fopen(path,"wb"); if(!f){perror(path);exit(73);}
|
||||
const char *names[]={@NAMES@};
|
||||
fprintf(f,"{\"version\":1,\"counterSegments\":%llu,\"counterErrors\":%llu,\"cvodeCounters\":{",segments,counter_errors);
|
||||
const char *counter_names[]={"rhs","linear_rhs","nonlinear_iterations","nonlinear_failures"};
|
||||
for(int i=0;i<4;i++)fprintf(f,"%s\"%s\":%llu",i?",":"",counter_names[i],counters[i]);
|
||||
fprintf(f,"},\"scopes\":{");
|
||||
for(int d=0;d<2;d++) {
|
||||
fprintf(f,"%s\"%s\":{",d?",":"",d?"integration":"outsideIntegration");
|
||||
for(int i=0;i<PROFILE_CATEGORY_COUNT;i++) {
|
||||
ProfileTotal *t=&totals[d][i];
|
||||
fprintf(f,"%s\"%s\":{\"calls\":%llu,\"inclusiveSeconds\":%.17g,\"exclusiveSeconds\":%.17g}",
|
||||
i?",":"",names[i],t->count,t->inclusive,t->exclusive);
|
||||
}
|
||||
fputc('}',f);
|
||||
}
|
||||
fprintf(f,"}}\n"); int ok=!ferror(f); if(fclose(f))ok=0; if(!ok)exit(73);
|
||||
}
|
||||
'''
|
||||
LINEAR_WRAPPERS = r'''
|
||||
/* Preserve the exact original Dense ops; only the call boundary is timed. */
|
||||
static int (*profile_original_setup)(SUNLinearSolver,SUNMatrix);
|
||||
static int (*profile_original_solve)(SUNLinearSolver,SUNMatrix,N_Vector,N_Vector,sunrealtype);
|
||||
static int profile_dense_setup(SUNLinearSolver linear,SUNMatrix matrix) {
|
||||
PROFILE_SCOPE(PROFILE_DENSE_SETUP);
|
||||
return profile_original_setup(linear,matrix);
|
||||
}
|
||||
static int profile_dense_solve(SUNLinearSolver linear,SUNMatrix matrix,N_Vector x,N_Vector b,sunrealtype tolerance) {
|
||||
PROFILE_SCOPE(PROFILE_DENSE_SOLVE);
|
||||
return profile_original_solve(linear,matrix,x,b,tolerance);
|
||||
}
|
||||
static int profile_cvode(void *solver,sunrealtype end,N_Vector y,sunrealtype *next,int task) {
|
||||
PROFILE_SCOPE(PROFILE_CVODE_STEP);
|
||||
return CVode(solver,end,y,next,task);
|
||||
}
|
||||
'''
|
||||
|
||||
|
||||
def digest(path: Path) -> str:
|
||||
return sha256(path.read_bytes()).hexdigest()
|
||||
|
||||
|
||||
def write_json(path: Path, value: object) -> None:
|
||||
path.write_text(json.dumps(value, indent=2, ensure_ascii=False) + "\n", encoding="utf-8")
|
||||
|
||||
|
||||
def replace_once(text: str, old: str, new: str) -> str:
|
||||
if text.count(old) != 1:
|
||||
raise RuntimeError(f"Source anchor count changed: {old!r}")
|
||||
return text.replace(old, new)
|
||||
|
||||
|
||||
def scope_function(text: str, function: str, category: str) -> str:
|
||||
pattern = rf"(?m)^[A-Za-z_][A-Za-z0-9_ \t*]*\b{re.escape(function)}\s*\([^;{{}}]*\)\s*\{{"
|
||||
matches = list(re.finditer(pattern, text))
|
||||
if len(matches) != 1:
|
||||
raise RuntimeError(f"Cannot identify unique function {function}")
|
||||
pos = matches[0].end()
|
||||
return text[:pos] + f"\n PROFILE_SCOPE(PROFILE_{category.upper()});" + text[pos:]
|
||||
|
||||
|
||||
def instrument(native: Path) -> None:
|
||||
(native / "include/compute_profile.h").write_text(PROFILE_HEADER.replace("@CATEGORIES@", ", ".join("PROFILE_" + c.upper() for c in CATEGORIES)))
|
||||
(native / "runtime/compute_profile.c").write_text(PROFILE_SOURCE.replace("@NAMES@", ",".join(json.dumps(c) for c in CATEGORIES)))
|
||||
for filename, functions in {
|
||||
"common.c": {"native_rhs": "rhs", "native_poll": "poll", "native_accept": "accept", "native_append": "append"},
|
||||
"cvode_solver.c": {"native_bdf": "integration", "cv_dense": "dense_output"},
|
||||
"rk45.c": {"native_rk45": "integration"},
|
||||
}.items():
|
||||
path = native / "runtime" / filename
|
||||
text = '#include "compute_profile.h"\n' + path.read_text()
|
||||
for function, category in functions.items():
|
||||
text = scope_function(text, function, category)
|
||||
if filename == "cvode_solver.c":
|
||||
text = replace_once(text, "typedef struct { void *solver;", LINEAR_WRAPPERS + "\ntypedef struct { void *solver;")
|
||||
text = replace_once(text, " if (!linear) goto cleanup;", " if (!linear) goto cleanup;\n profile_original_setup=linear->ops->setup; profile_original_solve=linear->ops->solve;\n linear->ops->setup=profile_dense_setup; linear->ops->solve=profile_dense_solve;")
|
||||
text = replace_once(text, "int flag=CVode(solver,end,y,&next,CV_ONE_STEP);", "int flag=profile_cvode(solver,end,y,&next,CV_ONE_STEP);")
|
||||
extra = "\n profile_counter_segment();\n"
|
||||
for slot, api in enumerate(("CVodeGetNumRhsEvals", "CVodeGetNumLinRhsEvals", "CVodeGetNumNonlinSolvIters", "CVodeGetNumNonlinSolvConvFails")):
|
||||
extra += f" value=0; int profile_status_{slot}={api}(solver,&value); profile_counter({slot},profile_status_{slot},value);\n"
|
||||
text = replace_once(text, "CVodeGetNumLinSolvSetups(solver,&value); r->nlu+=(unsigned long)value;", "CVodeGetNumLinSolvSetups(solver,&value); r->nlu+=(unsigned long)value;" + extra)
|
||||
path.write_text(text)
|
||||
path = native / "runtime/main.c"
|
||||
text = '#include "compute_profile.h"\n' + path.read_text()
|
||||
text = replace_once(text, " native_run_free(&r); return code;", " native_run_free(&r); profile_dump(); return code;")
|
||||
path.write_text(text)
|
||||
|
||||
|
||||
def runtime_arguments(stages: Path | None) -> list[str]:
|
||||
if stages:
|
||||
original = json.loads(stages.read_text())["process"]["command"]
|
||||
args = original[1:]
|
||||
else:
|
||||
args = ["--method", "BDF", "--start", "0", "--stop", "10", "--sample-step", ".01", "--max-step", "1e30", "--rtol", "1e-8", "--timeout", "300"]
|
||||
safe, index = [], 0
|
||||
value_options = {"--method", "--start", "--stop", "--sample-step", "--max-step", "--rtol", "--timeout"}
|
||||
while index < len(args):
|
||||
key = args[index]
|
||||
if key == "--solve-only":
|
||||
safe.append(key); index += 1; continue
|
||||
if key not in value_options | {"--output", "--result-index", "--cancel-file"} or index + 1 >= len(args):
|
||||
raise RuntimeError(f"Unsupported replay argument: {key}")
|
||||
if key in value_options:
|
||||
safe.extend(args[index:index + 2])
|
||||
index += 2
|
||||
return safe
|
||||
|
||||
|
||||
def prepare(args: argparse.Namespace) -> dict:
|
||||
cache, output = args.cache_dir.resolve(), args.output_dir.resolve()
|
||||
if not output.is_relative_to(ROOT / "test"):
|
||||
raise RuntimeError("Diagnostic output must be in the repository's ignored test/ directory")
|
||||
manifest = json.loads((cache / "manifest.json").read_text())
|
||||
for name in ("model", "model.c", "model.h"):
|
||||
if digest(cache / name) != manifest["artifacts"][name]:
|
||||
raise RuntimeError(f"Cache artifact integrity failure: {name}")
|
||||
# Reject numerical/runtime drift; a sparse timing-only cached main is allowed
|
||||
# because control and our current writer share the numeric model contract.
|
||||
differences = []
|
||||
for name, expected in manifest["sourceHashes"].items():
|
||||
relative = name.split("native/", 1)[-1]
|
||||
current = ROOT / "native" / relative
|
||||
if digest(current) != expected:
|
||||
differences.append(relative)
|
||||
if any(name != "runtime/main.c" for name in differences):
|
||||
raise RuntimeError(f"Cached numerical sources differ from current sources: {differences}")
|
||||
compiler, sundials, compiler_version = toolchain()
|
||||
if not sys.platform.startswith("linux"):
|
||||
raise RuntimeError("This test-only cleanup-scope profiler requires Linux/GCC")
|
||||
libraries = [sundials / "lib" / f"libsundials_{name}.a" for name in LIBRARIES]
|
||||
for name, expected in manifest["dependencyHashes"].items():
|
||||
path = sundials / ("include" if "/" in name else "lib") / name
|
||||
if digest(path) != expected:
|
||||
raise RuntimeError(f"SUNDIALS dependency changed: {name}")
|
||||
if compiler_version != manifest["compiler"]:
|
||||
raise RuntimeError("Use the cached model's compiler version for this comparison")
|
||||
output.mkdir(parents=True, exist_ok=True)
|
||||
native = output / "native"
|
||||
shutil.copytree(ROOT / "native", native, dirs_exist_ok=True)
|
||||
for name in ("model.c", "model.h", "manifest.json"):
|
||||
shutil.copy2(cache / name, output / name)
|
||||
instrument(native)
|
||||
command = [compiler, *manifest["compilerFlags"], "-I", str(output), "-I", str(native / "include"), "-I", str(sundials / "include"), str(output / "model.c"), *map(str, sorted(native.rglob("*.c"))), "-Wl,--start-group", *map(str, libraries), "-Wl,--end-group", "-lm", "-o", str(output / "profiled-model")]
|
||||
prepared = {"controlExecutable": str(cache / "model"), "profiledExecutable": str(output / "profiled-model"), "buildCommand": command, "runtimeArguments": runtime_arguments(args.request_stages), "cacheDir": str(cache), "cachedMainDifference": differences, "stateCount": len(manifest["stateKeys"]), "modelSha256": digest(output / "model.c"), "compiler": compiler_version, "warmups": args.warmups, "repeats": args.repeats, "scope": "Inclusive times overlap. Exclusive scope times partition integration including instrumentation overhead. No component or libc sub-cost inference.", "sourceHashes": {str(p.relative_to(output)): digest(p) for p in sorted(native.rglob("*")) if p.is_file()}}
|
||||
write_json(output / "prepared.json", prepared)
|
||||
return prepared
|
||||
|
||||
|
||||
def parity_payload(result: dict) -> dict:
|
||||
return {key: value for key, value in result.items() if key not in {"solveSeconds", "solveCpuSeconds"}}
|
||||
|
||||
|
||||
def execute(args: argparse.Namespace, prepared: dict) -> None:
|
||||
output = args.output_dir.resolve()
|
||||
build = subprocess.run(prepared["buildCommand"], capture_output=True, text=True, timeout=180)
|
||||
(output / "build.log").write_text(build.stdout + build.stderr)
|
||||
if build.returncode:
|
||||
raise RuntimeError(f"Compilation failed: {output / 'build.log'}")
|
||||
baseline = None
|
||||
rows = []
|
||||
# Serial paired control/profile runs; warmups excluded from overhead figures.
|
||||
for index in range(-args.warmups, args.repeats):
|
||||
label = f"warmup-{index + args.warmups + 1}" if index < 0 else f"run-{index + 1}"
|
||||
for variant in ("control", "profiled"):
|
||||
run = output / variant / label
|
||||
run.mkdir(parents=True, exist_ok=True)
|
||||
result_path, profile_path = run / "result.json", run / "profile.json"
|
||||
for stale in (result_path, profile_path, run / "cancel.request"):
|
||||
stale.unlink(missing_ok=True)
|
||||
command = [prepared[f"{variant}Executable"], *prepared["runtimeArguments"], "--output", str(result_path), "--result-index", str(run / "result-index.json"), "--cancel-file", str(run / "cancel.request")]
|
||||
environment = dict(os.environ)
|
||||
environment.pop("NATIVE_COMPUTE_PROFILE", None)
|
||||
# Disable independent sparse-stage profilers in a cached control.
|
||||
environment.pop("NATIVE_STAGE_PROFILE", None)
|
||||
if variant == "profiled":
|
||||
environment["NATIVE_COMPUTE_PROFILE"] = str(profile_path)
|
||||
started = time.perf_counter()
|
||||
process = subprocess.run(command, env=environment, capture_output=True, timeout=args.process_timeout)
|
||||
wall = time.perf_counter() - started
|
||||
(run / "stdout.log").write_bytes(process.stdout)
|
||||
(run / "stderr.log").write_bytes(process.stderr)
|
||||
if process.returncode:
|
||||
raise RuntimeError(f"{variant}/{label} exit {process.returncode}; see stderr.log")
|
||||
result = json.loads(result_path.read_bytes())
|
||||
if result.get("success") is not True:
|
||||
raise RuntimeError(f"{variant}/{label} did not complete")
|
||||
comparable = parity_payload(result)
|
||||
if baseline is None:
|
||||
baseline = comparable
|
||||
if comparable != baseline:
|
||||
mismatches = [k for k in baseline.keys() | comparable.keys() if baseline.get(k) != comparable.get(k)]
|
||||
write_json(run / "parity-failure.json", mismatches)
|
||||
raise RuntimeError(f"Numerical/counter parity failed: {mismatches}")
|
||||
row = {"variant": variant, "run": label, "warmup": index < 0, "processWallSeconds": wall, "solveSeconds": result["solveSeconds"], "solveCpuSeconds": result["solveCpuSeconds"], "fullParity": True, "resultBytes": result_path.stat().st_size, "nfev": result["nfev"], "njev": result["njev"], "nlu": result["nlu"], "acceptedSteps": result["acceptedSteps"], "solverStarts": result["solverStarts"]}
|
||||
if variant == "profiled":
|
||||
profile = json.loads(profile_path.read_text())
|
||||
counters = profile["cvodeCounters"]
|
||||
checks = {"counterReadsSucceeded": profile["counterErrors"] == 0, "rhsCountMatches": counters["rhs"] + counters["linear_rhs"] == result["nfev"], "linearRhsEqualsJacobianCountTimesStates": counters["linear_rhs"] == result["njev"] * prepared["stateCount"], "rhsClockCountMatches": profile["scopes"]["integration"]["rhs"]["calls"] == result["nfev"]}
|
||||
row["profile"] = profile
|
||||
row["counterChecks"] = checks
|
||||
if not checks["counterReadsSucceeded"] or not checks["rhsClockCountMatches"] or (result["method"] == "BDF" and not checks["rhsCountMatches"]):
|
||||
write_json(run / "counter-failure.json", row)
|
||||
raise RuntimeError(f"Unexpected profiling counters: {checks}")
|
||||
rows.append(row)
|
||||
write_json(run / "run.json", row)
|
||||
print(f"{variant}/{label}: solve={row['solveSeconds']:.6f}s wall={wall:.6f}s parity=true", flush=True)
|
||||
medians = {variant: {key: statistics.median(row[key] for row in rows if row["variant"] == variant and not row["warmup"]) for key in ("processWallSeconds", "solveSeconds", "solveCpuSeconds")} for variant in ("control", "profiled")}
|
||||
overhead = {key: medians["profiled"][key] / medians["control"][key] - 1 for key in medians["control"]}
|
||||
write_json(output / "summary.json", {"prepared": prepared, "runs": rows, "medians": medians, "instrumentationOverheadFraction": overhead, "allFullParity": True, "interpretation": "CVODE linear_rhs counts finite-difference RHS calls independently of model nfev. Its multiplication by stateCount is checked, not assumed. RHS time includes all model work; no Jacobian-specific RHS time is inferred. cvode_step.exclusiveSeconds excludes nested RHS, polling and Dense ops; integration.exclusiveSeconds is outer setup/loop/cleanup work. Dense setup covers the original factorization operation; dense_solve covers the original triangular solve operation. Clock/bookkeeping costs remain in all measured totals. Production control may retain its pre-existing sparse main clocks; cachedMainDifference records this."})
|
||||
print(f"Summary: {output / 'summary.json'}", flush=True)
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("--cache-dir", required=True, type=Path)
|
||||
parser.add_argument("--request-stages", type=Path)
|
||||
parser.add_argument("--output-dir", required=True, type=Path)
|
||||
parser.add_argument("--run", action="store_true", help="Build and run serial warmups/repeats; default only prepares")
|
||||
parser.add_argument("--warmups", type=int, default=1)
|
||||
parser.add_argument("--repeats", type=int, default=3)
|
||||
parser.add_argument("--process-timeout", type=float, default=360)
|
||||
args = parser.parse_args()
|
||||
if args.warmups < 0 or args.repeats < 1:
|
||||
parser.error("warmups must be nonnegative and repeats positive")
|
||||
prepared = prepare(args)
|
||||
print(f"Prepared: {args.output_dir.resolve() / 'prepared.json'}", flush=True)
|
||||
if args.run:
|
||||
execute(args, prepared)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in new issue
Block a user