相较上一版 Jacobian 确定性复用更新,本次补齐事件边界一致性、结果两侧采样及接触事件定位;保留已有物性复用和组件力学公式。 - 统一 UD00 信号求值与下一事件查询的绝对时间边界,修复循环边界浮点舍入导致的阶段错位、重复或漏报,并覆盖零时长、多阶段及长周期场景。 - 引入原生输出语义 v2:保留规则网格真实时间,补充内部时间事件和状态事件的左邻及事件后采样,按保存时间、状态和离散模式重放结果。 - 两条代码生成路径均发出 LSTP 接触描述,默认定位间隙过零及非负力模式的力截断;仅在接受事件时更新防重复记录,增加 contactEvents 诊断计数。 - 补充 MASS/LSTP 独立事件实验、八路全曲线与驱动阶段配对评估,以及 Amesim 不连续点输出对照和力差定位报告;MASS 新增释放机制仍保留为独立实验。 - 保存局部 probe、context 访问与回退、shadow replay、R288 real skip/typed replay 及阀门数值尾部诊断工具和报告;未证明净收益的实验不启用为生产默认优化。 - 更新原生运行说明和元件建模规范,补充信号边界、输出语义、接触事件和实验依赖回归测试。 验证:五组专项回归共 34 项全部通过;37 个待提交 Python 文件语法检查通过;git diff --cached --check 通过。
245 lines
18 KiB
Python
245 lines
18 KiB
Python
"""Read archived profiling only; write a reproducible fallback cost inventory.
|
||
|
||
No worker build, simulation, native code edit, or extrapolated speedup claim.
|
||
Run from any directory with the repository's Python interpreter.
|
||
"""
|
||
from collections import defaultdict
|
||
import hashlib
|
||
import json
|
||
from pathlib import Path
|
||
import re
|
||
import statistics
|
||
|
||
ROOT = Path(__file__).resolve().parents[2]
|
||
SOURCE = ROOT / "test/context-fallback-20260917"
|
||
OUT = ROOT / "test/fallback-profitability-20260917"
|
||
DOCS = ROOT / "tests/manual"
|
||
J = 896
|
||
MANIFEST = {}
|
||
|
||
|
||
def read(path):
|
||
raw = path.read_bytes()
|
||
MANIFEST[path.relative_to(ROOT).as_posix()] = hashlib.sha256(raw).hexdigest()
|
||
return json.loads(raw)
|
||
|
||
|
||
def close(a, b):
|
||
assert abs(a - b) <= 1e-12 * max(1, abs(a), abs(b)), (a, b)
|
||
|
||
|
||
def table(headers, rows):
|
||
def cell(x):
|
||
return str(x).replace("|", "\\|").replace("\n", " ")
|
||
return "\n| " + " | ".join(headers) + " |\n| " + " | ".join(["---"] * len(headers)) + " |\n" + "\n".join("| " + " | ".join(map(cell, r)) + " |" for r in rows) + "\n"
|
||
|
||
|
||
def labels(xs, prefix=""):
|
||
return ",".join(prefix + str(x) for x in xs)
|
||
|
||
|
||
def span_stats(xs):
|
||
return dict(rounds=xs, mean=statistics.mean(xs), median=statistics.median(xs), min=min(xs), max=max(xs))
|
||
|
||
|
||
def main():
|
||
a = read(SOURCE / "analysis.json")
|
||
plan = read(SOURCE / "plan.json")
|
||
typed = read(ROOT / "test/r288-typed-replay-20260917/performance-summary.json")
|
||
layered = read(ROOT / "test/local-probe-profile-direct-20260917/comparison.json")
|
||
P = typed["metrics"]["C"]["stages"]["median"]
|
||
H = typed["paired"]["C"]["metadataIncrementUs"]["median"]
|
||
|
||
def budgets(c, m):
|
||
# Strict inequality is required for a positive saving. Negative budgets
|
||
# are retained: they mean no nonnegative implementation cost can fit.
|
||
return dict(originalUs=c, structuralReusePerJacobian=m,
|
||
captureMaxAtZeroProbeUs=m*c,
|
||
captureMaxAtProbeUs={str(p): m*(c-p) for p in (0.1, 0.25, 0.5, 2, 5, 10, P)},
|
||
probeMaxAtTypedCaptureUs=c-H/m,
|
||
typedZeroCapturePossible=c>P,
|
||
typedWithSharedCapturePossible=c>P+H/m,
|
||
typedScenarioNetMs=J*(m*(c-P)-H)/1000)
|
||
|
||
native = {}
|
||
for pos, lines in enumerate(plan["code"]):
|
||
native[pos] = sorted(set(re.findall(r"\b(native_\w+)\s*\(", "\n".join(lines))))
|
||
for op in a["operations"]:
|
||
pos = op["position"]
|
||
assert bool(native[pos]) == (plan["versions"][pos] != plan["versions"][pos+1])
|
||
|
||
pairs = {(x["group"], x["region"]): x for x in a["groupRegions"] if x["failures"]}
|
||
gos = {(x["group"], x["position"]): x for x in a["groupOperations"]}
|
||
assert len(pairs) == 156 and len(gos) == len(a["groupOperations"]) == 5157
|
||
assert sum(x["failures"] for x in pairs.values()) == 139776
|
||
assert sum(x["count"] for x in gos.values()) == 4620672
|
||
assert all(x["failures"] == J for x in pairs.values())
|
||
assert all(x["inputDiffOps"] == x["outputDiffOps"] == x["exitOutputDiff"] == 0 and x["exitContextDiff"] == J for x in pairs.values())
|
||
region_runs, op_runs = [], []
|
||
for i in range(3):
|
||
r = read(SOURCE / f"regions-{i}/context.json")
|
||
o = read(SOURCE / f"ops-{i}/context.json")
|
||
assert r["sampled"] == o["sampled"] == 112
|
||
region_runs.append({(v["group"], v["region"]): v["regionTicks"] * J/r["sampled"]/r["frequency"] for v in r["regions"]})
|
||
op_runs.append({(g, pos): ticks * J/o["sampled"]/o["frequency"] for g, pos, n, ticks in o["operations"]})
|
||
for key, x in pairs.items():
|
||
close(x["seconds"], statistics.mean(r[key] for r in region_runs))
|
||
for key, x in gos.items():
|
||
close(x["seconds"], statistics.mean(o[key] for o in op_runs))
|
||
|
||
def composition(ops):
|
||
ns = sum(x["seconds"] for x in ops if native[x["position"]])
|
||
alg = sum(x["seconds"] for x in ops if not native[x["position"]])
|
||
return dict(nativeContainingOperationSeconds=ns, pureAlgebraAliasOperationSeconds=alg)
|
||
|
||
operations = []
|
||
for x in sorted(a["operations"], key=lambda x: -x["seconds"]):
|
||
pos = x["position"]
|
||
members = [o for (g, p), o in gos.items() if p == pos]
|
||
close(x["seconds"], sum(o["seconds"] for o in members))
|
||
assert x["count"] == len(x["groups"])*J
|
||
calls = native[pos]
|
||
kernel_family = any(c in calls for c in ("native_pipe_flow_cached_context", "native_medium_orifice_context"))
|
||
operation = dict(**x, nativeCalls=calls, code=plan["code"][pos],
|
||
nativeOperationCount=int(bool(calls)), operationCount=1,
|
||
contextFreeAlgebraAlias=not calls,
|
||
decision="A;停止R288后续优化" if pos == 16 else "A;E只测纯数值尾部" if kernel_family else "A" if calls else "A;C合并代数段",
|
||
baselineSharing="显式输入/输出同值;跨组记录共享仅为结构上限,消费字段/分支未普遍验证",
|
||
budget=budgets(x["seconds"]*1e6/x["count"], len(x["groups"])),
|
||
roundSeconds=span_stats([sum(run[g, pos] for g in x["groups"]) for run in op_runs]),
|
||
groupMeanUsRange=[min(o["seconds"]*1e6/o["count"] for o in members), max(o["seconds"]*1e6/o["count"] for o in members)],
|
||
functionTimingScope="仅有 regionKernels 的区间函数归因,无本 position 专属函数计时",
|
||
**composition(members))
|
||
operations.append(operation)
|
||
regions, group_regions = [], []
|
||
for r in sorted((r for r in a["regions"] if r["failures"]), key=lambda r: -r["seconds"]):
|
||
rid = r["id"]
|
||
members = [x for x in gos.values() if x["region"] == rid]
|
||
comp = composition(members)
|
||
close(r["operationSeconds"], sum(comp.values()))
|
||
native_positions = [p for p in range(r["start"], r["end"]) if native[p]]
|
||
assert native_positions == r["nativeOperationPositions"]
|
||
count = r["end"]-r["start"]
|
||
kernels = sorted((dict(name=plan["kernels"][k["kernel"]][1], **k) for k in a["regionKernels"] if k["region"] == rid), key=lambda k: -k["exclusiveSeconds"])
|
||
m = len(r["failedGroups"])
|
||
op_us = r["operationSeconds"]*1e6/r["failures"]
|
||
decision = "A" if count == 1 else "C/E预算筛选;native默认A"
|
||
if rid in (3, 480, 485, 490):
|
||
decision += ";D仅融合区间的待证假设"
|
||
region = dict(**r, operationCount=count, nativeOperationCount=len(native_positions),
|
||
nativeCalls=sorted(set(c for p in native_positions for c in native[p])),
|
||
kernels=kernels, decision=decision, **comp,
|
||
baselineSharing="同一 baseline 区间结构可供多个组引用;语义兼容未证" if m>1 else "仅一个失败组,无跨失败组摊销",
|
||
budget=budgets(r["meanUs"], m), operationModeBudget=budgets(op_us, m),
|
||
bothModesTypedWithSharedCapturePossible=min(r["meanUs"], op_us)>P+H/m,
|
||
roundMeanUs=span_stats([sum(run[g, rid] for g in r["failedGroups"])*1e6/r["failures"] for run in region_runs]))
|
||
regions.append(region)
|
||
for g in r["failedGroups"]:
|
||
x = pairs[g, rid]
|
||
group_regions.append(dict(**x, operationCount=count, nativeOperationCount=len(native_positions),
|
||
nativeCalls=region["nativeCalls"],
|
||
functionTimingScope=f"R{rid}汇总,无group级函数拆分", decision=decision,
|
||
**composition([v for v in members if v["group"] == g]),
|
||
budget=budgets(x["seconds"]*1e6/J, m),
|
||
# The net over J*m applies to a homogeneous-cost
|
||
# scenario, not the measured total for this group.
|
||
sharedBudgetNote="m为整个interval共享上限;本group成本的预算是假设同成本的情景,不是全interval实测净收益",
|
||
roundMeanUs=span_stats([run[g, rid]*1e6/J for run in region_runs])))
|
||
assert len(regions) == 115 and len(operations) == 343
|
||
close(sum(r["seconds"] for r in regions), a["totals"]["regionSeconds"])
|
||
close(sum(o["seconds"] for o in operations), a["totals"]["operationSeconds"])
|
||
close(sum(r["pureAlgebraAliasOperationSeconds"] for r in regions), sum(o["pureAlgebraAliasOperationSeconds"] for o in operations))
|
||
|
||
cohorts = []
|
||
for p in (2, 5, P, 10):
|
||
for capture in (0, H):
|
||
eligible = [r for r in regions if r["meanUs"] > p+capture/r["budget"]["structuralReusePerJacobian"]]
|
||
cohorts.append(dict(probeCostUs=p, captureUs=capture, regionIds=[r["id"] for r in eligible],
|
||
count=len(eligible), originalSeconds=sum(r["seconds"] for r in eligible),
|
||
originalShare=sum(r["seconds"] for r in eligible)/a["totals"]["regionSeconds"],
|
||
hypotheticalNetSeconds=sum(r["seconds"]-r["failures"]*p/1e6-J*capture/1e6 for r in eligible),
|
||
assumption="每个完整interval只付一次固定P,baseline捕获H跨所有失败组共享,100%成功;不等于逐operation replay,也不是预测"))
|
||
total_comp = composition(list(gos.values()))
|
||
pure_spans = []
|
||
for r in regions:
|
||
pos = r["start"]
|
||
while pos < r["end"]:
|
||
if native[pos]:
|
||
pos += 1
|
||
continue
|
||
start = pos
|
||
while pos < r["end"] and not native[pos]:
|
||
pos += 1
|
||
members = [gos[g, p] for g in r["failedGroups"] for p in range(start, pos)]
|
||
seconds = sum(x["seconds"] for x in members)
|
||
pure_spans.append(dict(region=r["id"], groups=r["failedGroups"], start=start, end=pos,
|
||
operationCount=pos-start, count=r["failures"], seconds=seconds,
|
||
meanUs=seconds*1e6/r["failures"]))
|
||
close(sum(s["seconds"] for s in pure_spans), total_comp["pureAlgebraAliasOperationSeconds"])
|
||
pure_spans.sort(key=lambda s: -s["seconds"])
|
||
kernel_cohort = [o for o in operations if any(c in o["nativeCalls"] for c in ("native_pipe_flow_cached_context", "native_medium_orifice_context"))]
|
||
summary = dict(**a["totals"], **total_comp, intervals=len(regions), groupIntervals=len(pairs),
|
||
operations=len(operations), groupOperations=len(gos),
|
||
maxOperationMeanUs=max(o["budget"]["originalUs"] for o in operations),
|
||
maxGroupOperationMeanUs=max(o["seconds"]*1e6/o["count"] for o in gos.values()),
|
||
pureSpanCount=len(pure_spans),
|
||
pureSpansAbove2UsSeconds=sum(s["seconds"] for s in pure_spans if s["meanUs"]>2),
|
||
kernelContainingPositions=[o["position"] for o in kernel_cohort],
|
||
kernelContainingOriginalSeconds=sum(o["seconds"] for o in kernel_cohort),
|
||
layeredFallbackSeconds=layered["main"]["totals"]["correctedSeconds"]["schedule_context_fallback"]*J/layered["main"]["samples"],
|
||
wholeContextSuccesses=a["totals"]["compares"]-a["totals"]["failures"])
|
||
group_ops = []
|
||
for x in a["groupOperations"]:
|
||
op = next(o for o in operations if o["position"] == x["position"])
|
||
group_ops.append(dict(**x, nativeCalls=op["nativeCalls"], decision=op["decision"],
|
||
meanUs=x["seconds"]*1e6/x["count"],
|
||
structuralReusePerJacobian=op["budget"]["structuralReusePerJacobian"]))
|
||
out = dict(scope="archived 0–10 s; diagnostics only", summary=summary,
|
||
referenceCosts=dict(typedProbeUs=P, typedCaptureIncrementUs=H, metrics=typed["metrics"], paired=typed["paired"]),
|
||
assumptions=dict(sharing="structural upper bound, not established semantic reuse", successRate=1,
|
||
originalCost="sampled instrumented mean; not an uninstrumented lower bound",
|
||
budgets="strictly less for profit; no incremental failed-guard/fallback-dispatch cost assumed"),
|
||
regions=regions, groupRegions=group_regions, operations=operations, groupOperations=group_ops,
|
||
pureSpans=pure_spans, groups=a["groups"], kernels=a["kernels"], scenarios=cohorts,
|
||
successfulContextRegions=[r for r in a["regions"] if r["contextual"] and not r["failures"]],
|
||
inputSha256=MANIFEST,
|
||
checks=["139776 interval failures", "4620672 fallback operation executions", "no duplicate (group,position)",
|
||
"156 failing group/interval pairs; 115 intervals; 343 positions; 5157 group/position pairs",
|
||
"raw three-run timing reconstruction matches archived analysis", "native-call and context-version classification agree",
|
||
"region, group and operation partitions conserve totals", "all pure spans conserve algebra time"])
|
||
OUT.mkdir(parents=True, exist_ok=True)
|
||
(OUT / "analysis.json").write_text(json.dumps(out, ensure_ascii=False, indent=2)+"\n", encoding="utf-8")
|
||
|
||
preamble = "# 全部 fallback 区间及预算\n\n由 `analyze_fallback_profitability.py` 从历史数据生成。ID、group、position 为0基,范围为[start,end)。ms为完整896个Jacobian折算累计;µs为每次fallback。按累计原计算时间排序。\n\nP=probe总开销,H=每个Jacobian新增baseline捕获,m=结构上最多共享的失败组数。盈利要求 P+H/m<C;Hmax=m(C-P)。所有预算是严格上限,不代表实际可达到;负数表示不可能。m不是已验证成功次数,若只能组内独立捕获则m=1。\n\n原区间时间、op时间和function时间来自不同抽样运行,不相加;纯代数是完全不含native调用的operation计时,native内部代数耗时未知。完整JSON另含全部5157条group/position映射、三轮原始折算值及函数表。\n"
|
||
doc = [preamble, "## 115个失败区间:成本和组成\n", table(
|
||
["R / 范围", "group", "次数", "累计ms", "均值µs / 三轮min–max", "ops/native", "op累计ms / 纯代数ms", "主要native", "主要function(exclusive排序)", "m / 共享", "选择"],
|
||
[(f"R{r['id']} [{r['start']},{r['end']})", labels(r['failedGroups']), r['failures'], f"{r['seconds']*1e3:.4f}",
|
||
f"{r['meanUs']:.3f} / {r['roundMeanUs']['min']:.3f}–{r['roundMeanUs']['max']:.3f}", f"{r['operationCount']}/{r['nativeOperationCount']}",
|
||
f"{r['operationSeconds']*1e3:.4f} / {r['pureAlgebraAliasOperationSeconds']*1e3:.4f}", labels(r['nativeCalls']),
|
||
labels([k['name'] for k in r['kernels'] if k['count']][:3]), f"{len(r['failedGroups'])} / 条件式", r['decision']) for r in regions]),
|
||
"\n## 每个区间的break-even边界\n\nPtyped、Htyped取R288最新批次中位数,仅作成本量级情景。区间预算假设一套融合机制处理整个区间,不能按每个native重复付P后仍使用本预算。\n", table(
|
||
["R", "Cµs", "m", "Hmax(P=0)µs", "Hmax(P=2/5/10)µs", "Pmax(Htyped)µs", "Ptyped,H=0可行", "Ptyped,Htyped可行", "op模式也支持该情景"],
|
||
[(r['id'], f"{r['meanUs']:.3f}", len(r['failedGroups']), f"{r['budget']['captureMaxAtZeroProbeUs']:.3f}",
|
||
"/".join(f"{r['budget']['captureMaxAtProbeUs'][str(p)]:.3f}" for p in (2,5,10)), f"{r['budget']['probeMaxAtTypedCaptureUs']:.3f}",
|
||
r['budget']['typedZeroCapturePossible'], r['budget']['typedWithSharedCapturePossible'], r['bothModesTypedWithSharedCapturePossible']) for r in regions]),
|
||
"\n## 全部156条group/interval成本\n", table(
|
||
["group", "R", "次数", "原计算ms", "µs/次", "ops/native", "纯代数ms", "m上限", "Pmax(Htyped/m)µs"],
|
||
[(r['group'], r['region'], r['failures'], f"{r['seconds']*1e3:.4f}", f"{r['budget']['originalUs']:.3f}", f"{r['operationCount']}/{r['nativeOperationCount']}",
|
||
f"{r['pureAlgebraAliasOperationSeconds']*1e3:.4f}", r['budget']['structuralReusePerJacobian'], f"{r['budget']['probeMaxAtTypedCaptureUs']:.3f}") for r in sorted(group_regions,key=lambda x:-x['seconds'])]),
|
||
"\n## 全部纯代数连续段(区间内切分候选)\n", table(
|
||
["R", "group", "范围", "ops", "执行次数", "累计ms", "每段µs / 允许的最大新增成本"],
|
||
[(s['region'], labels(s['groups']), f"[{s['start']},{s['end']})", s['operationCount'], s['count'], f"{s['seconds']*1e3:.4f}", f"{s['meanUs']:.3f}") for s in pure_spans])]
|
||
(DOCS / "fallback_profitability_intervals.md").write_text("\n".join(doc), encoding="utf-8")
|
||
doc = ["# 全部343个 fallback operation\n\n按跨group累计原计算时间排序。一个position计作一个operation,native=1表示原语句含native调用(不等于动态native调用总次数)。纯代数operation的纯代数耗时等于其累计时间;native语句内部的代数/函数耗时没有独立position级测量。具体(group,position)到R的关联见JSON groupOperations,不将各列group和R做笛卡尔积。\n\n所有m都是每Jacobian结构可共享上限,非语义证明。Hmax/Pmax使用与区间表相同的严格盈亏公式。逐operation的7.109µs typed量级,即使H=0也全部不盈利。A=原计算;C=批量代数段切分;E=只评估纯kernel memo,绝非跳过native的context副作用。\n", table(
|
||
["position / 原ID", "operation", "R", "group", "次数", "累计ms", "每次µs", "native", "主要native", "m", "Hmax(P=0)µs", "Hmax(P=2/5/10)µs", "Pmax(Htyped)µs", "选择"],
|
||
[(f"{o['position']}/{o['operationId']}", o['name'], labels(o['regions'],'R'), labels(o['groups']), o['count'], f"{o['seconds']*1e3:.4f}",
|
||
f"{o['budget']['originalUs']:.3f}", o['nativeOperationCount'], labels(o['nativeCalls']) or '纯代数/alias', o['budget']['structuralReusePerJacobian'],
|
||
f"{o['budget']['captureMaxAtZeroProbeUs']:.3f}", "/".join(f"{o['budget']['captureMaxAtProbeUs'][str(p)]:.3f}" for p in (2,5,10)),
|
||
f"{o['budget']['probeMaxAtTypedCaptureUs']:.3f}", o['decision']) for o in operations])]
|
||
(DOCS / "fallback_profitability_operations.md").write_text("\n".join(doc), encoding="utf-8")
|
||
print(json.dumps(dict(summary=summary, scenarios=cohorts, topPureSpans=pure_spans[:8], checks=out['checks']), ensure_ascii=False, indent=2))
|
||
|
||
|
||
if __name__ == "__main__":
|
||
main()
|