原生结果series通过字节索引直传,C端使用Ryu精确回读编码和64 KiB批量写出;网页采用Float64缓存和CSV工作线程,减少结果处理与保存等待。 补充八路AME曲线核查、全流程分阶段计时、独立编码基准和复现工具,固定后续优化采用修正八路及rtol=1e-8。C写出1.1808→0.1638 s,点击到可查看8.0100→6.9756 s。 验证:最终10项编码专项、29项相关后端回归通过;8份原生结果逐位一致,16次网页结果/CSV/刷新恢复通过。前端构建及缓存/CSV专项在本轮结果处理工作中通过。环境、原始大结果与临时构建不纳入Git。
319 lines
18 KiB
Python
319 lines
18 KiB
Python
"""Replay native result encoding in isolated C programs, without model evaluation.
|
|
|
|
Prepare only by default. Explicit --run builds a small C replay and serially
|
|
runs one warmup and three measured real-file outputs per variant. --dev-null
|
|
adds sink-only runs after real-file verification; these never stand in for I/O.
|
|
|
|
.venv/bin/python tests/manual/benchmark_native_result_encoding.py \
|
|
--result-json test/web-cost-20260911/native-compute-profile/control/run-1/result.json \
|
|
--output-dir test/c-result-encoding-20260911 --ryu-root /path/to/ryu --run
|
|
|
|
ryu-root must contain ryu/d2s.c and ryu/ryu.h (the ryu/ subdirectory itself is
|
|
also accepted). No dependency is downloaded and no production source is edited.
|
|
Every series cell, final scalar and finalState cell is encoded in C. Static
|
|
metadata, JSON structure and escaped keys are prepared outside timing. Values
|
|
are loaded contiguously before timing; this isolates decimal encoding/write
|
|
cost, excluding projection and the production writer's strided matrix reads.
|
|
Timers include fopen, buffer setup, formatting, write, flush and fclose, but
|
|
not fsync durability. All real-file outputs are parsed and compared as complete
|
|
binary64 values, including signed zero, outside the measured interval.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
from array import array
|
|
from hashlib import sha256
|
|
import json
|
|
import math
|
|
import os
|
|
from pathlib import Path
|
|
import shutil
|
|
import statistics
|
|
import struct
|
|
import subprocess
|
|
import sys
|
|
import time
|
|
|
|
ROOT = Path(__file__).resolve().parents[2]
|
|
VARIANTS = ["fprintf-default", "fprintf-1m", "snprintf-64k", "ryu-64k"]
|
|
C_SOURCE = r'''
|
|
#include <stdio.h>
|
|
#include <stdlib.h>
|
|
#include <stdint.h>
|
|
#include <string.h>
|
|
#include <math.h>
|
|
#include <time.h>
|
|
#if HAVE_RYU
|
|
#include "ryu/ryu.h"
|
|
#endif
|
|
#include "replay-layout.h"
|
|
#define CHUNK (64u*1024u)
|
|
typedef struct { FILE *f; char block[CHUNK]; size_t used; unsigned long long bytes; int failed; } Writer;
|
|
static double wall_now(void) { struct timespec t; if(clock_gettime(CLOCK_MONOTONIC,&t))exit(72); return t.tv_sec+t.tv_nsec*1e-9; }
|
|
static double cpu_now(void) { struct timespec t; if(clock_gettime(CLOCK_PROCESS_CPUTIME_ID,&t))exit(72); return t.tv_sec+t.tv_nsec*1e-9; }
|
|
static void flush_block(Writer *w) {
|
|
if(w->used && fwrite(w->block,1,w->used,w->f)!=w->used)w->failed=1;
|
|
w->used=0;
|
|
}
|
|
static void block_bytes(Writer *w,const char *text,size_t n) {
|
|
w->bytes+=n;
|
|
while(n) {
|
|
size_t left=CHUNK-w->used, part=n<left?n:left;
|
|
memcpy(w->block+w->used,text,part); w->used+=part; text+=part; n-=part;
|
|
if(w->used==CHUNK)flush_block(w);
|
|
}
|
|
}
|
|
static void direct_bytes(Writer *w,const char *text,size_t n) {
|
|
w->bytes+=n;
|
|
if(fwrite(text,1,n,w->f)!=n)w->failed=1;
|
|
}
|
|
static void encode(Writer *w,const double *values,int mode) {
|
|
for(size_t g=0;g<GROUP_COUNT;g++) {
|
|
const Group *group=&groups[g];
|
|
if(mode<2)direct_bytes(w,group->prefix,group->prefix_length);
|
|
else block_bytes(w,group->prefix,group->prefix_length);
|
|
for(size_t i=0;i<group->count;i++) {
|
|
double value=values[group->offset+i];
|
|
if(mode<2) {
|
|
/* Same number/separator formatting call as native main.c. */
|
|
int n=fprintf(w->f,"%s%.17g",i?",":"",value);
|
|
if(n<0)w->failed=1; else w->bytes+=(unsigned)n;
|
|
} else if(mode==2) {
|
|
/* Format directly into the remaining batch buffer. */
|
|
if(CHUNK-w->used<64)flush_block(w);
|
|
int n=snprintf(w->block+w->used,CHUNK-w->used,"%s%.17g",i?",":"",value);
|
|
if(n<0 || (size_t)n>=CHUNK-w->used){w->failed=1;return;}
|
|
w->used+=(unsigned)n;w->bytes+=(unsigned)n;
|
|
} else {
|
|
#if HAVE_RYU
|
|
if(CHUNK-w->used<64)flush_block(w);
|
|
if(i){w->block[w->used++]=',';w->bytes++;}
|
|
int n=d2s_buffered_n(value,w->block+w->used);
|
|
if(n<1 || n>32){w->failed=1;return;}
|
|
w->used+=(unsigned)n;w->bytes+=(unsigned)n;
|
|
#else
|
|
w->failed=1;return;
|
|
#endif
|
|
}
|
|
}
|
|
}
|
|
if(mode<2)direct_bytes(w,tail,TAIL_LENGTH);
|
|
else {block_bytes(w,tail,TAIL_LENGTH);flush_block(w);}
|
|
}
|
|
int main(int argc,char **argv) {
|
|
if(argc!=4)return 64;
|
|
int mode=-1;
|
|
const char *names[]={"fprintf-default","fprintf-1m","snprintf-64k","ryu-64k"};
|
|
for(int i=0;i<4;i++)if(!strcmp(argv[1],names[i]))mode=i;
|
|
if(mode<0 || (mode==3 && !HAVE_RYU))return 64;
|
|
if(sizeof(double)!=8 || sizeof(uint64_t)!=8)return 65;
|
|
FILE *input=fopen(argv[2],"rb"); if(!input)return 66;
|
|
double *values=malloc(VALUE_COUNT*sizeof(double));
|
|
if(!values){fclose(input);return 67;}
|
|
int loaded=fread(values,sizeof(double),VALUE_COUNT,input)==VALUE_COUNT && fgetc(input)==EOF && !ferror(input);
|
|
if(fclose(input))loaded=0;
|
|
if(!loaded){free(values);return 68;}
|
|
for(size_t i=0;i<VALUE_COUNT;i++)if(!isfinite(values[i])){free(values);return 69;}
|
|
Writer *writer=calloc(1,sizeof(Writer)); char *stdio_buffer=malloc(1024u*1024u);
|
|
if(!writer || !stdio_buffer){free(values);free(writer);free(stdio_buffer);return 67;}
|
|
/* Loading, allocation and input validation are intentionally outside timing. */
|
|
double wall_start=wall_now(), cpu_start=cpu_now();
|
|
writer->f=fopen(argv[3],"wb");
|
|
if(!writer->f){free(values);free(writer);free(stdio_buffer);return 70;}
|
|
if(mode==1 && setvbuf(writer->f,stdio_buffer,_IOFBF,1024u*1024u))writer->failed=1;
|
|
/* Manual batch variants use identical unbuffered FILE sinks. */
|
|
if(mode>=2 && setvbuf(writer->f,NULL,_IONBF,0))writer->failed=1;
|
|
if(!writer->failed)encode(writer,values,mode);
|
|
if(ferror(writer->f))writer->failed=1;
|
|
if(fclose(writer->f))writer->failed=1;
|
|
double cpu_seconds=cpu_now()-cpu_start, wall_seconds=wall_now()-wall_start;
|
|
printf("{\"variant\":\"%s\",\"wallSeconds\":%.17g,\"cpuSeconds\":%.17g,\"encodedBytes\":%llu,\"valueCount\":%zu,\"success\":%s}\n",
|
|
names[mode],wall_seconds,cpu_seconds,writer->bytes,(size_t)VALUE_COUNT,writer->failed?"false":"true");
|
|
int code=writer->failed?71:0;
|
|
free(values);free(stdio_buffer);free(writer);return code;
|
|
}
|
|
'''
|
|
|
|
|
|
def digest(path: Path) -> str:
|
|
return sha256(path.read_bytes()).hexdigest()
|
|
|
|
|
|
def write_json(path: Path, value: object) -> None:
|
|
path.write_text(json.dumps(value, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
|
|
|
|
|
|
def read_json_numbers(path: Path) -> dict:
|
|
# JSON's integer spelling -0 must retain the sign before conversion to double.
|
|
return json.loads(path.read_bytes(), parse_int=lambda token: -0.0 if token == "-0" else int(token))
|
|
|
|
|
|
def c_string(value: bytes) -> str:
|
|
return '"' + ''.join(f"\\x{byte:02x}" for byte in value) + '"'
|
|
|
|
|
|
def numeric_blocks(result: dict) -> list[tuple[str, list]]:
|
|
blocks = [(f"series/{key}", value) for key, value in result["series"].items()]
|
|
blocks.extend((f"final/{key}", [value]) for key, value in result["final"].items())
|
|
blocks.append(("finalState", result["finalState"]))
|
|
return blocks
|
|
|
|
|
|
def prepare(args: argparse.Namespace) -> dict:
|
|
output = args.output_dir.resolve()
|
|
if not output.is_relative_to(ROOT / "test"):
|
|
raise RuntimeError("Output must be beneath the repository's ignored test/ directory")
|
|
if sys.byteorder != "little" or array('d').itemsize != 8 or not sys.platform.startswith("linux"):
|
|
raise RuntimeError("This isolated replay currently requires Linux and little-endian binary64")
|
|
result = read_json_numbers(args.result_json)
|
|
if not isinstance(result.get("series"), dict) or not isinstance(result.get("final"), dict) or not isinstance(result.get("finalState"), list):
|
|
raise RuntimeError("Input must be a complete native result.json")
|
|
blocks = numeric_blocks(result)
|
|
all_values = array('d')
|
|
descriptors = []
|
|
for name, values in blocks:
|
|
if not isinstance(values, list):
|
|
raise RuntimeError(f"Expected numeric array: {name}")
|
|
for value in values:
|
|
if isinstance(value, bool) or not isinstance(value, (int, float)) or not math.isfinite(value):
|
|
raise RuntimeError(f"Nonfinite or nonnumeric input: {name}")
|
|
if isinstance(value, int) and int(float(value)) != value:
|
|
raise RuntimeError(f"Integer does not fit exactly in binary64: {name}")
|
|
descriptors.append({"name": name, "offset": len(all_values), "count": len(values)})
|
|
all_values.extend(values)
|
|
if not all_values:
|
|
raise RuntimeError("No numeric output values")
|
|
metadata = {key: value for key, value in result.items() if key not in {"series", "final", "finalState"}}
|
|
pending = json.dumps(metadata, ensure_ascii=True, separators=(",", ":"), allow_nan=False)[:-1]
|
|
pending += ("," if metadata else "") + '"series":{'
|
|
prefixes = []
|
|
first = True
|
|
for key in result["series"]:
|
|
pending += ("" if first else ",") + json.dumps(key, ensure_ascii=True) + ":["
|
|
prefixes.append(pending.encode()); pending = "]"; first = False
|
|
pending += '},"final":{'
|
|
first = True
|
|
for key in result["final"]:
|
|
pending += ("" if first else ",") + json.dumps(key, ensure_ascii=True) + ":"
|
|
prefixes.append(pending.encode()); pending = ""; first = False
|
|
pending += '},"finalState":['
|
|
prefixes.append(pending.encode())
|
|
tail = b"]}\n"
|
|
output.mkdir(parents=True, exist_ok=True)
|
|
raw = output / "values.f64le"
|
|
raw.write_bytes(all_values.tobytes())
|
|
header = ["/* Generated test data: all numeric payload cells, no projection. */", "typedef struct { const char *prefix; size_t prefix_length, offset, count; } Group;", f"#define GROUP_COUNT {len(descriptors)}u", f"#define VALUE_COUNT {len(all_values)}u", "static const Group groups[]={"]
|
|
for desc, prefix in zip(descriptors, prefixes, strict=True):
|
|
header.append(f" {{{c_string(prefix)},{len(prefix)}u,{desc['offset']}u,{desc['count']}u}},")
|
|
header += ["};", f"static const char tail[]={c_string(tail)};", f"#define TAIL_LENGTH {len(tail)}u"]
|
|
(output / "replay-layout.h").write_text("\n".join(header) + "\n")
|
|
(output / "replay.c").write_text(C_SOURCE)
|
|
compiler = os.environ.get("SIMULATION_NATIVE_CC") or shutil.which("gcc")
|
|
if not compiler:
|
|
raise RuntimeError("GCC is required")
|
|
ryu = args.ryu_root.resolve() if args.ryu_root else None
|
|
if ryu and not (ryu / "ryu/d2s.c").is_file() and (ryu / "d2s.c").is_file():
|
|
ryu = ryu.parent
|
|
if ryu and not all((ryu / name).is_file() for name in ("ryu/d2s.c", "ryu/ryu.h")):
|
|
raise RuntimeError("--ryu-root must contain ryu/d2s.c and ryu/ryu.h")
|
|
command = [compiler, "-std=c11", "-O3", "-Wall", "-Wextra", "-Werror", "-ffp-contract=off", "-fno-fast-math", "-D_POSIX_C_SOURCE=200809L", f"-DHAVE_RYU={int(ryu is not None)}", "-I", str(output), str(output / "replay.c")]
|
|
ryu_hashes = {}
|
|
if ryu:
|
|
command += ["-I", str(ryu), str(ryu / "ryu/d2s.c")]
|
|
ryu_hashes = {str(p.relative_to(ryu)): digest(p) for p in sorted((ryu / "ryu").glob("*")) if p.is_file() and p.suffix in {".c", ".h"}}
|
|
command += ["-lm", "-o", str(output / "replay")]
|
|
prepared = {"sourceResult": str(args.result_json.resolve()), "sourceSha256": digest(args.result_json), "sourceBytes": args.result_json.stat().st_size, "rawValuesSha256": digest(raw), "rawValueBytes": raw.stat().st_size, "valueCount": len(all_values), "seriesColumns": len(result["series"]), "seriesValues": sum(len(v) for v in result["series"].values()), "finalValues": len(result["final"]), "finalStateValues": len(result["finalState"]), "blocks": descriptors, "variants": VARIANTS if ryu else VARIANTS[:3], "ryuRoot": str(ryu) if ryu else None, "ryuSourceHashes": ryu_hashes, "buildCommand": command, "compiler": subprocess.run([compiler, "--version"], capture_output=True, text=True, check=True).stdout.splitlines()[0], "warmups": args.warmups, "repeats": args.repeats, "devNullRequested": args.dev_null, "precisionContract": "All finite payload values must decode to identical little-endian binary64 bytes, including signed zero. Shortest output may have different length/exponent spelling.", "timingContract": "C wall/process-CPU from before fopen through fclose, including buffer setup, all numeric payload formatting and writing. Excludes extraction, preload, allocation, static JSON framing preparation and verification. Ordinary files/page cache; no fsync durability. Contiguous replay does not reproduce production matrix strides or output projection. Block variants both use a 64KiB application buffer with unbuffered FILE sink."}
|
|
write_json(output / "prepared.json", prepared)
|
|
return prepared
|
|
|
|
|
|
def verify_file(path: Path, expected: dict, raw: bytes, prepared: dict) -> dict:
|
|
actual = read_json_numbers(path)
|
|
# This catches missing columns, changed metadata, duplicates in array values,
|
|
# order differences, and scalar value drift before the exact signed-zero pass.
|
|
if actual != expected:
|
|
raise RuntimeError(f"Full result value/structure parity failed: {path}")
|
|
actual_blocks = numeric_blocks(actual)
|
|
if [name for name, _ in actual_blocks] != [d["name"] for d in prepared["blocks"]]:
|
|
raise RuntimeError(f"Numeric block order differs: {path}")
|
|
negative_zeroes = 0
|
|
for (_, values), desc in zip(actual_blocks, prepared["blocks"], strict=True):
|
|
binary = array('d', values).tobytes()
|
|
start, end = desc["offset"] * 8, (desc["offset"] + desc["count"]) * 8
|
|
if binary != raw[start:end]:
|
|
raise RuntimeError(f"Binary64 parity failed at {desc['name']}: {path}")
|
|
negative_zeroes += sum(value == 0 and math.copysign(1.0, value) < 0 for value in values)
|
|
return {"fullResultParity": True, "allPayloadBinary64Parity": True, "checkedValues": prepared["valueCount"], "negativeZeroCount": negative_zeroes, "sha256": digest(path)}
|
|
|
|
|
|
def execute(args: argparse.Namespace, prepared: dict) -> None:
|
|
output = args.output_dir.resolve()
|
|
built = subprocess.run(prepared["buildCommand"], capture_output=True, text=True, timeout=120)
|
|
(output / "build.log").write_text(built.stdout + built.stderr)
|
|
if built.returncode:
|
|
raise RuntimeError(f"Compilation failed: {output / 'build.log'}")
|
|
expected = read_json_numbers(args.result_json)
|
|
raw = (output / "values.f64le").read_bytes()
|
|
rows = []
|
|
for sink in (["file", "dev-null"] if args.dev_null else ["file"]):
|
|
for index in range(-args.warmups, args.repeats):
|
|
label = f"warmup-{index + args.warmups + 1}" if index < 0 else f"run-{index + 1}"
|
|
for variant in prepared["variants"]:
|
|
run = output / sink / variant / label
|
|
run.mkdir(parents=True, exist_ok=True)
|
|
target = run / "result.json" if sink == "file" else Path("/dev/null")
|
|
started = time.perf_counter()
|
|
process = subprocess.run([str(output / "replay"), variant, str(output / "values.f64le"), str(target)], capture_output=True, text=True, timeout=120)
|
|
process_wall = time.perf_counter() - started
|
|
(run / "stdout.log").write_text(process.stdout)
|
|
(run / "stderr.log").write_text(process.stderr)
|
|
if process.returncode:
|
|
raise RuntimeError(f"Replay failed ({process.returncode}): {run}")
|
|
record = json.loads(process.stdout)
|
|
if record["success"] is not True:
|
|
raise RuntimeError(f"Encoding reported failure: {run}")
|
|
record.update(sink=sink, run=label, warmup=index < 0, processWallSeconds=process_wall)
|
|
if sink == "file":
|
|
if target.stat().st_size != record["encodedBytes"]:
|
|
raise RuntimeError(f"Written byte count differs: {target}")
|
|
record["verification"] = verify_file(target, expected, raw, prepared)
|
|
else:
|
|
record["verification"] = {"actualSinkFileReadback": False, "sameEncoderPassedRealFileReadback": True}
|
|
write_json(run / "run.json", record)
|
|
rows.append(record)
|
|
print(f"{sink}/{variant}/{label}: wall={record['wallSeconds']:.6f}s cpu={record['cpuSeconds']:.6f}s bytes={record['encodedBytes']}", flush=True)
|
|
medians = {}
|
|
for sink in {row["sink"] for row in rows}:
|
|
medians[sink] = {}
|
|
for variant in prepared["variants"]:
|
|
selected = [row for row in rows if row["sink"] == sink and row["variant"] == variant and not row["warmup"]]
|
|
medians[sink][variant] = {key: statistics.median(row[key] for row in selected) for key in ("wallSeconds", "cpuSeconds", "encodedBytes")}
|
|
baseline = medians[sink]["fprintf-default"]
|
|
for data in medians[sink].values():
|
|
data["wallReductionFractionVsDefault"] = 1 - data["wallSeconds"] / baseline["wallSeconds"]
|
|
data["byteReductionFractionVsDefault"] = 1 - data["encodedBytes"] / baseline["encodedBytes"]
|
|
write_json(output / "summary.json", {"prepared": prepared, "runs": rows, "medians": medians, "allRealFileBinary64Parity": True, "limitation": "This replay isolates formatting and ordinary file writes on preloaded contiguous doubles. It is not an end-to-end native/application speedup and excludes projection, strided output reads and durable storage flush. /dev/null metrics, if present, are separate sink-only observations."})
|
|
print(f"Summary: {output / 'summary.json'}", flush=True)
|
|
|
|
|
|
def main() -> None:
|
|
parser = argparse.ArgumentParser(description=__doc__)
|
|
parser.add_argument("--result-json", required=True, type=Path)
|
|
parser.add_argument("--output-dir", type=Path, default=ROOT / "test/c-result-encoding-20260911")
|
|
parser.add_argument("--ryu-root", type=Path)
|
|
parser.add_argument("--run", action="store_true")
|
|
parser.add_argument("--dev-null", action="store_true")
|
|
parser.add_argument("--warmups", type=int, default=1)
|
|
parser.add_argument("--repeats", type=int, default=3)
|
|
args = parser.parse_args()
|
|
if args.warmups < 0 or args.repeats < 1:
|
|
parser.error("warmups must be nonnegative and repeats positive")
|
|
prepared = prepare(args)
|
|
print(f"Prepared {prepared['valueCount']} binary64 values and {len(prepared['variants'])} variants: {args.output_dir.resolve()}", flush=True)
|
|
if args.run:
|
|
execute(args, prepared)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|