feat: extend stall timeout and remove sample cap

This commit is contained in:
huojiarong committed 2026-08-19 11:34:31 +00:00
1 parent 27f9f4add8
commit eb6ea70e19
9 files changed
+105 -45

No files matched your search

+17 -25
View File
@@ -4,6 +4,7 @@ from collections.abc import Callable
from dataclasses import dataclass, replace from dataclasses import dataclass, replace
from math import floor, isfinite from math import floor, isfinite
import os import os
from sys import maxsize
from typing import Literal from typing import Literal
from app.simulation.core.base import Component, DynamicComponent from app.simulation.core.base import Component, DynamicComponent
@@ -312,14 +313,7 @@ def simulation_preparation_issues(
def simulation_sample_times( def simulation_sample_times(
config: SolveIVPConfig, config: SolveIVPConfig,
step: float, step: float,
*,
max_points: int = 10001,
) -> list[float]: ) -> list[float]:
if max_points < 2:
raise SimulationSampleTimeError(
"SIMULATION_SAMPLE_LIMIT_INVALID",
"Simulation sample limit must allow at least two points.",
)
t_start = float(config.t_start) t_start = float(config.t_start)
t_stop = float(config.t_stop) t_stop = float(config.t_stop)
if not isfinite(t_start) or not isfinite(t_stop): if not isfinite(t_start) or not isfinite(t_stop):
@@ -344,33 +338,31 @@ def simulation_sample_times(
"Simulation stop time must be greater than start time.", "Simulation stop time must be greater than start time.",
) )
# Bound the grid before dividing by a potentially tiny step or allocating
# the result list. This avoids both float-to-int overflow and an OOM-sized
# ``range``/list when input comes from an external System XML document.
maximum_interval_count = max_points - 1
if step < duration / maximum_interval_count:
raise SimulationSampleTimeError(
"SIMULATION_SAMPLE_COUNT_EXCEEDED",
f"Simulation sample count exceeds the limit of {max_points}; "
"increase sampleStep.",
)
ratio = duration / step ratio = duration / step
if not isfinite(ratio): if not isfinite(ratio):
raise SimulationSampleTimeError( raise SimulationSampleTimeError(
"SIMULATION_SAMPLE_COUNT_EXCEEDED", "SIMULATION_SAMPLE_COUNT_UNREPRESENTABLE",
f"Simulation sample count exceeds the limit of {max_points}; " "Simulation sample count cannot be represented by this runtime; "
"increase sampleStep.",
)
interval_count = floor(ratio)
# There is no product-level point cap. Still reject a collection that the
# Python runtime cannot index before multiplying by the potentially huge
# interval count or allocating the output grid.
if interval_count > maxsize - 2:
raise SimulationSampleTimeError(
"SIMULATION_SAMPLE_COUNT_UNREPRESENTABLE",
"Simulation sample count cannot be represented by this runtime; "
"increase sampleStep.", "increase sampleStep.",
) )
interval_count = int(floor(ratio))
last_regular_time = t_start + interval_count * step last_regular_time = t_start + interval_count * step
append_stop = last_regular_time < t_stop append_stop = last_regular_time < t_stop
requested_point_count = interval_count + 1 + int(append_stop) requested_point_count = interval_count + 1 + int(append_stop)
if requested_point_count > max_points: if requested_point_count > maxsize:
raise SimulationSampleTimeError( raise SimulationSampleTimeError(
"SIMULATION_SAMPLE_COUNT_EXCEEDED", "SIMULATION_SAMPLE_COUNT_UNREPRESENTABLE",
f"Simulation requests {requested_point_count} samples; " "Simulation sample count cannot be represented by this runtime; "
f"the limit is {max_points}.", "increase sampleStep.",
) )
times = [t_start] times = [t_start]
@@ -263,11 +263,11 @@ signal、stream 和外部 volume resolver 已把静态组件列表、端口引
| 代数残差容差 | `1e-7` | 压力流量快速路径/接受标准 | | 代数残差容差 | `1e-7` | 压力流量快速路径/接受标准 |
| 代数最大评估 | `500` | 单次 `least_squares` 上限 | | 代数最大评估 | `500` | 单次 `least_squares` 上限 |
| stream 容差/迭代 | `1e-9 / 100` | 焓传播固定点 | | stream 容差/迭代 | `1e-9 / 100` | 焓传播固定点 |
| 采样数上限 | `10001` | 限制输出样本,不限制 RHS 次数或事件数 | | 采样数上限 | 无固定业务上限 | 输出规模受运行时可表示范围和可用资源约束,不限制 RHS 次数或事件数 |
前端/Pydantic 默认值见 `frontend/src/App.tsx` 的仿真默认配置和 `app/main.py:131-137`;通用路径构造 `SolveIVPConfig` 见 `app/main.py:684-701`;采样网格见 `app/simulation/systems/generic.py:204-225`。 前端/Pydantic 默认值见 `frontend/src/App.tsx` 的仿真默认配置和 `app/main.py:131-137`;通用路径构造 `SolveIVPConfig` 见 `app/main.py:684-701`;采样网格见 `app/simulation/systems/generic.py:204-225`。
**[已实现]** `sampleStep` 生成的采样网格会确保包含 `t_stop`,并在超过 10001 点时拒绝;它不会把 BDF 变成固定步算法。实际 RHS 次数由自适应误差控制、Jacobian 估计、拒绝步、事件重启和 `max_step` 共同决定。 **[已实现]** `sampleStep` 生成的采样网格会确保包含 `t_stop`,不再设置固定点数上限;仅在数值非有限、当前运行时无法表示点数或时间无法严格递增时预先拒绝。它不会把 BDF 变成固定步算法。实际 RHS 次数由自适应误差控制、Jacobian 估计、拒绝步、事件重启和 `max_step` 共同决定。
### 6.3 逐步推进分派 ### 6.3 逐步推进分派
@@ -384,10 +384,10 @@ STEP0、UD00 等信号源提供离散事件时刻。积分器先推进到事件
前端规则: 前端规则:
- 30 秒没有收到任何字节:连接超时; - 30 秒没有收到任何字节:连接超时;
- 60 秒只收到心跳而没有真实积分进度:判定 stalled 并请求取消; - 15 分钟只收到心跳而没有真实积分进度:判定 stalled 并请求取消;
- 正常运行不是轮询,轮询仅用于异常恢复。 - 正常运行不是轮询,轮询仅用于异常恢复。
**[发现]** 合法但单个已接受步/闭合超过 60 秒时,前端可能误判停滞。后端结果事件的 `phase` 使用 `completed/stopped/stalled/failed`,前端事件类型却声明 `"complete"`;运行时当前没有按该字段做严格校验,所以契约漂移尚未直接报错(`app/main.py:825-840`、`frontend/src/App.tsx` 的流式事件类型)。 **[已缓解]** 合法但单个已接受步/闭合超过原 60 秒阈值时,前端会误判停滞;浏览器警钟现延长为 15 分钟,30 秒断流检测保持不变。该警钟仍以“最后一次非心跳积分进度”为依据,尚未细分 RHS、Jacobian 和闭合活动。后端结果事件的 `phase` 使用 `completed/stopped/stalled/failed`,前端事件类型却声明 `"complete"`;运行时当前没有按该字段做严格校验,所以契约漂移尚未直接报错(`app/main.py:825-840`、`frontend/src/App.tsx` 的流式事件类型)。
### 9.2 开发和部署连接数 ### 9.2 开发和部署连接数
@@ -424,7 +424,7 @@ O(组件 + 连接 + 代数结构)
+ O(采样数 × 公开结果变量数) + O(采样数 × 公开结果变量数)
``` ```
当前采样上限是 10001。放大因素包括: 当前不设置固定采样点数上限,调用方必须根据模型输出变量数和可用内存选择 `sampleStep`。放大因素包括:
- 积分状态矩阵与后处理 `series` 在后处理阶段同时存在; - 积分状态矩阵与后处理 `series` 在后处理阶段同时存在;
- 最终完整结果保存在全局任务记录中,又被编码为一个大型 NDJSON 行; - 最终完整结果保存在全局任务记录中,又被编码为一个大型 NDJSON 行;
@@ -642,7 +642,7 @@ PYTHONPATH=. .venv/bin/python -m app.simulation.benchmark_regression \
- [ ] 正常慢步不会被误判为死锁,真实无活动能在约定时间内终止并给出诊断。 - [ ] 正常慢步不会被误判为死锁,真实无活动能在约定时间内终止并给出诊断。
- [ ] 取消请求在每个主要阶段都能在有界时间内生效。 - [ ] 取消请求在每个主要阶段都能在有界时间内生效。
- [ ] 并发压力下服务仍能响应健康检查和新请求拒绝/排队逻辑。 - [ ] 并发压力下服务仍能响应健康检查和新请求拒绝/排队逻辑。
- [ ] 权威 `0.2 s / 0.001 s` 浏览器路径完成且不发生假超时;只要内部活动持续,60 s 无接受步不得触发 `SOLVER_STALLED`,真正无活动仍能按约定上限停止并返回最后阶段与计数。 - [ ] 权威 `0.2 s / 0.001 s` 浏览器路径完成且不发生假超时;只要内部活动持续,15 分钟无接受步不得触发 `SOLVER_STALLED`,真正无活动仍能按约定上限停止并返回最后阶段与计数。
### OPT-09 建立 10 s 长时验证与模式覆盖 ### OPT-09 建立 10 s 长时验证与模式覆盖
+5 -4
View File
@@ -101,7 +101,7 @@ System
| --- | --- | --- | | --- | --- | --- |
| `tStart` | 仿真开始时刻 | 必须是有限数值 | | `tStart` | 仿真开始时刻 | 必须是有限数值 |
| `tStop` | 仿真结束时刻 | 必须有限且大于 `tStart` | | `tStop` | 仿真结束时刻 | 必须有限且大于 `tStart` |
| `sampleStep` | 结果相邻采样点的时间间隔 | 必须大于 0,且整个区间最多生成 10001 个采样点 | | `sampleStep` | 结果相邻采样点的时间间隔 | 必须大于 0;不设置固定的采样点数上限 |
| `maxStep` | 自适应积分器单个内部步的上限 | 必须大于 0 | | `maxStep` | 自适应积分器单个内部步的上限 | 必须大于 0 |
| `method` | 积分方法 | `RK45/RK23/DOP853/Radau/BDF/LSODA` | | `method` | 积分方法 | `RK45/RK23/DOP853/Radau/BDF/LSODA` |
@@ -111,9 +111,10 @@ System
- `maxStep` 限制求解器内部一次最多前进多久; - `maxStep` 限制求解器内部一次最多前进多久;
- 自适应求解器可以因为误差、事件或试探状态失败而走得比 `maxStep` 更短。 - 自适应求解器可以因为误差、事件或试探状态失败而走得比 `maxStep` 更短。
采样点数量会在创建时间数组前计算。若区间长度不可表示为有限数、请求超过 10001 采样点数量会在创建时间数组前计算。若区间长度不可表示为有限数、点数超过当前
点,或在当前浮点精度下无法得到包含 `tStart/tStop` 的严格递增时间序列,输入会在 运行时可表示的集合大小,或在当前浮点精度下无法得到包含 `tStart/tStop` 的严格
仿真前被拒绝,不会把超大或重复的 `t_eval` 交给积分器。 递增时间序列,输入会在仿真前被拒绝。采样点不再受固定业务上限约束,但结果内存、
序列化体积和浏览器负载仍会随“采样点数 × 输出变量数”线性增长。
工程 JSON 为兼容现有前端仍把采样字段命名为 `simulation.step`;导出 v3 时必须映射为 `Simulation/@sampleStep`。 工程 JSON 为兼容现有前端仍把采样字段命名为 `simulation.step`;导出 v3 时必须映射为 `Simulation/@sampleStep`。
+7 -4
View File
@@ -108,6 +108,10 @@ import {
LMECHN1_MAX_RIGHT_PORT_COUNT, LMECHN1_MAX_RIGHT_PORT_COUNT,
lmechn1RightPortCount, lmechn1RightPortCount,
} from "./componentSymbols/mechanical"; } from "./componentSymbols/mechanical";
import {
solverStallTimeoutMessage,
solverStallTimeoutReached,
} from "./simulationTimeout";
import { import {
CONTACT_AWARE_EDGE_TYPE, CONTACT_AWARE_EDGE_TYPE,
contactAwareEdgeTypes, contactAwareEdgeTypes,
@@ -559,7 +563,6 @@ const PALETTE_COLLAPSED_SECTIONS_KEY =
const PALETTE_ICON_PREVIEW_DELAY_MS = 500; const PALETTE_ICON_PREVIEW_DELAY_MS = 500;
const CONSOLE_VIEWPORT_MARGIN = 8; const CONSOLE_VIEWPORT_MARGIN = 8;
const SIMULATION_STREAM_IDLE_TIMEOUT_MS = 30_000; const SIMULATION_STREAM_IDLE_TIMEOUT_MS = 30_000;
const SIMULATION_SOLVER_STALL_TIMEOUT_MS = 60_000;
const GRID_SIZE = CANVAS_GRID_SIZE; const GRID_SIZE = CANVAS_GRID_SIZE;
const MODELING_DEFAULT_EDGE_OPTIONS = { const MODELING_DEFAULT_EDGE_OPTIONS = {
interactionWidth: 28, interactionWidth: 28,
@@ -7401,7 +7404,7 @@ function FlowWorkbench() {
appendConsoleEntry( appendConsoleEntry(
"error", "error",
error.code === "SOLVER_STALLED" error.code === "SOLVER_STALLED"
? "求解器连续 60 秒没有接受新积分步,正在终止任务并恢复部分结果" ? solverStallTimeoutMessage("recovering")
: "超过 30 秒未收到后端数据,正在终止任务并恢复部分结果", : "超过 30 秒未收到后端数据,正在终止任务并恢复部分结果",
); );
try { try {
@@ -10936,10 +10939,10 @@ async function streamSystemSimulation(
if (event.event === "progress") { if (event.event === "progress") {
if ( if (
event.heartbeat === true && event.heartbeat === true &&
Date.now() - lastSolverProgressAt >= SIMULATION_SOLVER_STALL_TIMEOUT_MS solverStallTimeoutReached(lastSolverProgressAt)
) { ) {
throw new SimulationStreamError( throw new SimulationStreamError(
"求解器连续 60 秒没有接受新的积分步,任务可能已经卡死", solverStallTimeoutMessage("detected"),
[], [],
undefined, undefined,
"SOLVER_STALLED", "SOLVER_STALLED",
+19
View File
@@ -0,0 +1,19 @@
export const SIMULATION_SOLVER_STALL_TIMEOUT_MINUTES = 15;
export const SIMULATION_SOLVER_STALL_TIMEOUT_MS =
SIMULATION_SOLVER_STALL_TIMEOUT_MINUTES * 60_000;
export function solverStallTimeoutReached(
lastSolverProgressAt: number,
now: number = Date.now(),
) {
return now - lastSolverProgressAt >= SIMULATION_SOLVER_STALL_TIMEOUT_MS;
}
export function solverStallTimeoutMessage(
context: "detected" | "recovering",
) {
const prefix = `求解器连续 ${SIMULATION_SOLVER_STALL_TIMEOUT_MINUTES} 分钟没有接受新的积分步`;
return context === "recovering"
? `${prefix},正在终止任务并恢复部分结果`
: `${prefix},任务可能已经卡死`;
}
@@ -0,0 +1,21 @@
import { expect, test } from "@playwright/test";
import {
SIMULATION_SOLVER_STALL_TIMEOUT_MINUTES,
SIMULATION_SOLVER_STALL_TIMEOUT_MS,
solverStallTimeoutMessage,
solverStallTimeoutReached,
} from "../../src/simulationTimeout";
test("solver stall watchdog allows the validated long-running window", () => {
expect(SIMULATION_SOLVER_STALL_TIMEOUT_MINUTES).toBe(15);
expect(SIMULATION_SOLVER_STALL_TIMEOUT_MS).toBe(900_000);
expect(solverStallTimeoutReached(1_000, 900_999)).toBe(false);
expect(solverStallTimeoutReached(1_000, 901_000)).toBe(true);
});
test("solver stall messages use the configured duration", () => {
expect(solverStallTimeoutMessage("detected")).toContain("连续 15 分钟");
expect(solverStallTimeoutMessage("recovering")).toContain("连续 15 分钟");
expect(solverStallTimeoutMessage("recovering")).toContain("恢复部分结果");
});
@@ -475,6 +475,20 @@ class GenericSystemXmlSimulationTests(unittest.TestCase):
"port_a", "port_a",
) )
def test_raw_system_xml_serializes_more_than_legacy_sample_limit(self) -> None:
project = chain_project()
project.simulation.step = 0.0000005
xml = build_reactflow_system_xml(project)
response = asyncio.run(simulate_system_xml(xml_request(xml)))
round_tripped = json.loads(json.dumps(response))
self.assertTrue(round_tripped["success"])
self.assertEqual(round_tripped["diagnostics"]["sampleCount"], 20001)
self.assertEqual(len(round_tripped["series"]["time"]), 20001)
self.assertEqual(round_tripped["series"]["time"][0], 0.0)
self.assertEqual(round_tripped["series"]["time"][-1], 0.01)
def test_streaming_endpoint_events_have_monotonic_progress_and_result(self) -> None: def test_streaming_endpoint_events_have_monotonic_progress_and_result(self) -> None:
project = chain_project() project = chain_project()
xml = build_reactflow_system_xml(project) xml = build_reactflow_system_xml(project)
+16 -6
View File
@@ -107,15 +107,22 @@ class SimulationSampleTimeSafetyTests(unittest.TestCase):
[1.0, 1.25], [1.0, 1.25],
) )
def test_exact_point_limit_is_allowed(self) -> None: def test_grid_can_exceed_the_legacy_point_limit(self) -> None:
times = simulation_sample_times( times = simulation_sample_times(
SolveIVPConfig(t_start=0.0, t_stop=1.0), SolveIVPConfig(t_start=0.0, t_stop=5.0),
0.0001, 0.0001,
) )
self.assertEqual(len(times), 10001) self.assertEqual(len(times), 50001)
self.assertEqual(times[0], 0.0) self.assertEqual(times[0], 0.0)
self.assertEqual(times[-1], 1.0) self.assertEqual(times[-1], 5.0)
def test_system_xml_accepts_more_than_the_legacy_point_limit(self) -> None:
report = validate_system_xml_document(
_system_xml(t_start="0", t_stop="5", sample_step="0.0001")
)
self.assertTrue(report.valid, report.issues)
def test_tiny_step_is_rejected_before_an_oversized_grid_is_allocated(self) -> None: def test_tiny_step_is_rejected_before_an_oversized_grid_is_allocated(self) -> None:
with self.assertRaises(SimulationSampleTimeError) as caught: with self.assertRaises(SimulationSampleTimeError) as caught:
@@ -124,7 +131,10 @@ class SimulationSampleTimeSafetyTests(unittest.TestCase):
1.0e-300, 1.0e-300,
) )
self.assertEqual(caught.exception.code, "SIMULATION_SAMPLE_COUNT_EXCEEDED") self.assertEqual(
caught.exception.code,
"SIMULATION_SAMPLE_COUNT_UNREPRESENTABLE",
)
def test_non_finite_derived_duration_has_a_stable_error(self) -> None: def test_non_finite_derived_duration_has_a_stable_error(self) -> None:
with self.assertRaises(SimulationSampleTimeError) as caught: with self.assertRaises(SimulationSampleTimeError) as caught:
@@ -153,7 +163,7 @@ class SimulationSampleTimeSafetyTests(unittest.TestCase):
cases = ( cases = (
( (
_system_xml(t_start="0", t_stop="1", sample_step="1e-300"), _system_xml(t_start="0", t_stop="1", sample_step="1e-300"),
"SIMULATION_SAMPLE_COUNT_EXCEEDED", "SIMULATION_SAMPLE_COUNT_UNREPRESENTABLE",
), ),
( (
_system_xml( _system_xml(