From eb6ea70e198e2b0f6dc8cc112d6fdfb914aa84c7 Mon Sep 17 00:00:00 2001 From: huojiarong Date: Wed, 19 Aug 2026 11:34:31 +0000 Subject: [PATCH] feat: extend stall timeout and remove sample cap --- app/simulation/systems/generic.py | 42 ++++++++----------- docs/other/后端求解逻辑与效率优化调研.md | 10 ++--- docs/other/求解器性能优化任务清单.md | 2 +- docs/standard/system-xml-v3.md | 9 ++-- frontend/src/App.tsx | 11 +++-- frontend/src/simulationTimeout.ts | 19 +++++++++ frontend/tests/e2e/simulation-timeout.spec.ts | 21 ++++++++++ tests/test_generic_system_xml_simulation.py | 14 +++++++ tests/test_simulation_safety.py | 22 +++++++--- 9 files changed, 105 insertions(+), 45 deletions(-) create mode 100644 frontend/src/simulationTimeout.ts create mode 100644 frontend/tests/e2e/simulation-timeout.spec.ts diff --git a/app/simulation/systems/generic.py b/app/simulation/systems/generic.py index eebee7a..985c7ea 100644 --- a/app/simulation/systems/generic.py +++ b/app/simulation/systems/generic.py @@ -4,6 +4,7 @@ from collections.abc import Callable from dataclasses import dataclass, replace from math import floor, isfinite import os +from sys import maxsize from typing import Literal from app.simulation.core.base import Component, DynamicComponent @@ -312,14 +313,7 @@ def simulation_preparation_issues( def simulation_sample_times( config: SolveIVPConfig, step: float, - *, - max_points: int = 10001, ) -> list[float]: - if max_points < 2: - raise SimulationSampleTimeError( - "SIMULATION_SAMPLE_LIMIT_INVALID", - "Simulation sample limit must allow at least two points.", - ) t_start = float(config.t_start) t_stop = float(config.t_stop) if not isfinite(t_start) or not isfinite(t_stop): @@ -344,33 +338,31 @@ def simulation_sample_times( "Simulation stop time must be greater than start time.", ) - # Bound the grid before dividing by a potentially tiny step or allocating - # the result list. This avoids both float-to-int overflow and an OOM-sized - # ``range``/list when input comes from an external System XML document. - maximum_interval_count = max_points - 1 - if step < duration / maximum_interval_count: - raise SimulationSampleTimeError( - "SIMULATION_SAMPLE_COUNT_EXCEEDED", - f"Simulation sample count exceeds the limit of {max_points}; " - "increase sampleStep.", - ) - ratio = duration / step if not isfinite(ratio): raise SimulationSampleTimeError( - "SIMULATION_SAMPLE_COUNT_EXCEEDED", - f"Simulation sample count exceeds the limit of {max_points}; " + "SIMULATION_SAMPLE_COUNT_UNREPRESENTABLE", + "Simulation sample count cannot be represented by this runtime; " + "increase sampleStep.", + ) + interval_count = floor(ratio) + # There is no product-level point cap. Still reject a collection that the + # Python runtime cannot index before multiplying by the potentially huge + # interval count or allocating the output grid. + if interval_count > maxsize - 2: + raise SimulationSampleTimeError( + "SIMULATION_SAMPLE_COUNT_UNREPRESENTABLE", + "Simulation sample count cannot be represented by this runtime; " "increase sampleStep.", ) - interval_count = int(floor(ratio)) last_regular_time = t_start + interval_count * step append_stop = last_regular_time < t_stop requested_point_count = interval_count + 1 + int(append_stop) - if requested_point_count > max_points: + if requested_point_count > maxsize: raise SimulationSampleTimeError( - "SIMULATION_SAMPLE_COUNT_EXCEEDED", - f"Simulation requests {requested_point_count} samples; " - f"the limit is {max_points}.", + "SIMULATION_SAMPLE_COUNT_UNREPRESENTABLE", + "Simulation sample count cannot be represented by this runtime; " + "increase sampleStep.", ) times = [t_start] diff --git a/docs/other/后端求解逻辑与效率优化调研.md b/docs/other/后端求解逻辑与效率优化调研.md index d1999dc..e14d70d 100644 --- a/docs/other/后端求解逻辑与效率优化调研.md +++ b/docs/other/后端求解逻辑与效率优化调研.md @@ -263,11 +263,11 @@ signal、stream 和外部 volume resolver 已把静态组件列表、端口引 | 代数残差容差 | `1e-7` | 压力流量快速路径/接受标准 | | 代数最大评估 | `500` | 单次 `least_squares` 上限 | | stream 容差/迭代 | `1e-9 / 100` | 焓传播固定点 | -| 采样数上限 | `10001` | 限制输出样本,不限制 RHS 次数或事件数 | +| 采样数上限 | 无固定业务上限 | 输出规模受运行时可表示范围和可用资源约束,不限制 RHS 次数或事件数 | 前端/Pydantic 默认值见 `frontend/src/App.tsx` 的仿真默认配置和 `app/main.py:131-137`;通用路径构造 `SolveIVPConfig` 见 `app/main.py:684-701`;采样网格见 `app/simulation/systems/generic.py:204-225`。 -**[已实现]** `sampleStep` 生成的采样网格会确保包含 `t_stop`,并在超过 10001 点时拒绝;它不会把 BDF 变成固定步算法。实际 RHS 次数由自适应误差控制、Jacobian 估计、拒绝步、事件重启和 `max_step` 共同决定。 +**[已实现]** `sampleStep` 生成的采样网格会确保包含 `t_stop`,不再设置固定点数上限;仅在数值非有限、当前运行时无法表示点数或时间无法严格递增时预先拒绝。它不会把 BDF 变成固定步算法。实际 RHS 次数由自适应误差控制、Jacobian 估计、拒绝步、事件重启和 `max_step` 共同决定。 ### 6.3 逐步推进分派 @@ -384,10 +384,10 @@ STEP0、UD00 等信号源提供离散事件时刻。积分器先推进到事件 前端规则: - 30 秒没有收到任何字节:连接超时; -- 60 秒只收到心跳而没有真实积分进度:判定 stalled 并请求取消; +- 15 分钟只收到心跳而没有真实积分进度:判定 stalled 并请求取消; - 正常运行不是轮询,轮询仅用于异常恢复。 -**[发现]** 合法但单个已接受步/闭合超过 60 秒时,前端可能误判停滞。后端结果事件的 `phase` 使用 `completed/stopped/stalled/failed`,前端事件类型却声明 `"complete"`;运行时当前没有按该字段做严格校验,所以契约漂移尚未直接报错(`app/main.py:825-840`、`frontend/src/App.tsx` 的流式事件类型)。 +**[已缓解]** 合法但单个已接受步/闭合超过原 60 秒阈值时,前端会误判停滞;浏览器警钟现延长为 15 分钟,30 秒断流检测保持不变。该警钟仍以“最后一次非心跳积分进度”为依据,尚未细分 RHS、Jacobian 和闭合活动。后端结果事件的 `phase` 使用 `completed/stopped/stalled/failed`,前端事件类型却声明 `"complete"`;运行时当前没有按该字段做严格校验,所以契约漂移尚未直接报错(`app/main.py:825-840`、`frontend/src/App.tsx` 的流式事件类型)。 ### 9.2 开发和部署连接数 @@ -424,7 +424,7 @@ O(组件 + 连接 + 代数结构) + O(采样数 × 公开结果变量数) ``` -当前采样上限是 10001。放大因素包括: +当前不设置固定采样点数上限,调用方必须根据模型输出变量数和可用内存选择 `sampleStep`。放大因素包括: - 积分状态矩阵与后处理 `series` 在后处理阶段同时存在; - 最终完整结果保存在全局任务记录中,又被编码为一个大型 NDJSON 行; diff --git a/docs/other/求解器性能优化任务清单.md b/docs/other/求解器性能优化任务清单.md index e9e892e..da3aec5 100644 --- a/docs/other/求解器性能优化任务清单.md +++ b/docs/other/求解器性能优化任务清单.md @@ -642,7 +642,7 @@ PYTHONPATH=. .venv/bin/python -m app.simulation.benchmark_regression \ - [ ] 正常慢步不会被误判为死锁,真实无活动能在约定时间内终止并给出诊断。 - [ ] 取消请求在每个主要阶段都能在有界时间内生效。 - [ ] 并发压力下服务仍能响应健康检查和新请求拒绝/排队逻辑。 -- [ ] 权威 `0.2 s / 0.001 s` 浏览器路径完成且不发生假超时;只要内部活动持续,60 s 无接受步不得触发 `SOLVER_STALLED`,真正无活动仍能按约定上限停止并返回最后阶段与计数。 +- [ ] 权威 `0.2 s / 0.001 s` 浏览器路径完成且不发生假超时;只要内部活动持续,15 分钟无接受步不得触发 `SOLVER_STALLED`,真正无活动仍能按约定上限停止并返回最后阶段与计数。 ### OPT-09 建立 10 s 长时验证与模式覆盖 diff --git a/docs/standard/system-xml-v3.md b/docs/standard/system-xml-v3.md index 0dff855..325f26c 100644 --- a/docs/standard/system-xml-v3.md +++ b/docs/standard/system-xml-v3.md @@ -101,7 +101,7 @@ System | --- | --- | --- | | `tStart` | 仿真开始时刻 | 必须是有限数值 | | `tStop` | 仿真结束时刻 | 必须有限且大于 `tStart` | -| `sampleStep` | 结果相邻采样点的时间间隔 | 必须大于 0,且整个区间最多生成 10001 个采样点 | +| `sampleStep` | 结果相邻采样点的时间间隔 | 必须大于 0;不设置固定的采样点数上限 | | `maxStep` | 自适应积分器单个内部步的上限 | 必须大于 0 | | `method` | 积分方法 | `RK45/RK23/DOP853/Radau/BDF/LSODA` | @@ -111,9 +111,10 @@ System - `maxStep` 限制求解器内部一次最多前进多久; - 自适应求解器可以因为误差、事件或试探状态失败而走得比 `maxStep` 更短。 -采样点数量会在创建时间数组前计算。若区间长度不可表示为有限数、请求超过 10001 -点,或在当前浮点精度下无法得到包含 `tStart/tStop` 的严格递增时间序列,输入会在 -仿真前被拒绝,不会把超大或重复的 `t_eval` 交给积分器。 +采样点数量会在创建时间数组前计算。若区间长度不可表示为有限数、点数超过当前 +运行时可表示的集合大小,或在当前浮点精度下无法得到包含 `tStart/tStop` 的严格 +递增时间序列,输入会在仿真前被拒绝。采样点不再受固定业务上限约束,但结果内存、 +序列化体积和浏览器负载仍会随“采样点数 × 输出变量数”线性增长。 工程 JSON 为兼容现有前端仍把采样字段命名为 `simulation.step`;导出 v3 时必须映射为 `Simulation/@sampleStep`。 diff --git a/frontend/src/App.tsx b/frontend/src/App.tsx index 3e5aca0..9cc828f 100644 --- a/frontend/src/App.tsx +++ b/frontend/src/App.tsx @@ -108,6 +108,10 @@ import { LMECHN1_MAX_RIGHT_PORT_COUNT, lmechn1RightPortCount, } from "./componentSymbols/mechanical"; +import { + solverStallTimeoutMessage, + solverStallTimeoutReached, +} from "./simulationTimeout"; import { CONTACT_AWARE_EDGE_TYPE, contactAwareEdgeTypes, @@ -559,7 +563,6 @@ const PALETTE_COLLAPSED_SECTIONS_KEY = const PALETTE_ICON_PREVIEW_DELAY_MS = 500; const CONSOLE_VIEWPORT_MARGIN = 8; const SIMULATION_STREAM_IDLE_TIMEOUT_MS = 30_000; -const SIMULATION_SOLVER_STALL_TIMEOUT_MS = 60_000; const GRID_SIZE = CANVAS_GRID_SIZE; const MODELING_DEFAULT_EDGE_OPTIONS = { interactionWidth: 28, @@ -7401,7 +7404,7 @@ function FlowWorkbench() { appendConsoleEntry( "error", error.code === "SOLVER_STALLED" - ? "求解器连续 60 秒没有接受新积分步,正在终止任务并恢复部分结果" + ? solverStallTimeoutMessage("recovering") : "超过 30 秒未收到后端数据,正在终止任务并恢复部分结果", ); try { @@ -10936,10 +10939,10 @@ async function streamSystemSimulation( if (event.event === "progress") { if ( event.heartbeat === true && - Date.now() - lastSolverProgressAt >= SIMULATION_SOLVER_STALL_TIMEOUT_MS + solverStallTimeoutReached(lastSolverProgressAt) ) { throw new SimulationStreamError( - "求解器连续 60 秒没有接受新的积分步,任务可能已经卡死", + solverStallTimeoutMessage("detected"), [], undefined, "SOLVER_STALLED", diff --git a/frontend/src/simulationTimeout.ts b/frontend/src/simulationTimeout.ts new file mode 100644 index 0000000..0422bec --- /dev/null +++ b/frontend/src/simulationTimeout.ts @@ -0,0 +1,19 @@ +export const SIMULATION_SOLVER_STALL_TIMEOUT_MINUTES = 15; +export const SIMULATION_SOLVER_STALL_TIMEOUT_MS = + SIMULATION_SOLVER_STALL_TIMEOUT_MINUTES * 60_000; + +export function solverStallTimeoutReached( + lastSolverProgressAt: number, + now: number = Date.now(), +) { + return now - lastSolverProgressAt >= SIMULATION_SOLVER_STALL_TIMEOUT_MS; +} + +export function solverStallTimeoutMessage( + context: "detected" | "recovering", +) { + const prefix = `求解器连续 ${SIMULATION_SOLVER_STALL_TIMEOUT_MINUTES} 分钟没有接受新的积分步`; + return context === "recovering" + ? `${prefix},正在终止任务并恢复部分结果` + : `${prefix},任务可能已经卡死`; +} diff --git a/frontend/tests/e2e/simulation-timeout.spec.ts b/frontend/tests/e2e/simulation-timeout.spec.ts new file mode 100644 index 0000000..c6137bc --- /dev/null +++ b/frontend/tests/e2e/simulation-timeout.spec.ts @@ -0,0 +1,21 @@ +import { expect, test } from "@playwright/test"; + +import { + SIMULATION_SOLVER_STALL_TIMEOUT_MINUTES, + SIMULATION_SOLVER_STALL_TIMEOUT_MS, + solverStallTimeoutMessage, + solverStallTimeoutReached, +} from "../../src/simulationTimeout"; + +test("solver stall watchdog allows the validated long-running window", () => { + expect(SIMULATION_SOLVER_STALL_TIMEOUT_MINUTES).toBe(15); + expect(SIMULATION_SOLVER_STALL_TIMEOUT_MS).toBe(900_000); + expect(solverStallTimeoutReached(1_000, 900_999)).toBe(false); + expect(solverStallTimeoutReached(1_000, 901_000)).toBe(true); +}); + +test("solver stall messages use the configured duration", () => { + expect(solverStallTimeoutMessage("detected")).toContain("连续 15 分钟"); + expect(solverStallTimeoutMessage("recovering")).toContain("连续 15 分钟"); + expect(solverStallTimeoutMessage("recovering")).toContain("恢复部分结果"); +}); diff --git a/tests/test_generic_system_xml_simulation.py b/tests/test_generic_system_xml_simulation.py index 0412a45..38b24a5 100644 --- a/tests/test_generic_system_xml_simulation.py +++ b/tests/test_generic_system_xml_simulation.py @@ -475,6 +475,20 @@ class GenericSystemXmlSimulationTests(unittest.TestCase): "port_a", ) + def test_raw_system_xml_serializes_more_than_legacy_sample_limit(self) -> None: + project = chain_project() + project.simulation.step = 0.0000005 + xml = build_reactflow_system_xml(project) + + response = asyncio.run(simulate_system_xml(xml_request(xml))) + round_tripped = json.loads(json.dumps(response)) + + self.assertTrue(round_tripped["success"]) + self.assertEqual(round_tripped["diagnostics"]["sampleCount"], 20001) + self.assertEqual(len(round_tripped["series"]["time"]), 20001) + self.assertEqual(round_tripped["series"]["time"][0], 0.0) + self.assertEqual(round_tripped["series"]["time"][-1], 0.01) + def test_streaming_endpoint_events_have_monotonic_progress_and_result(self) -> None: project = chain_project() xml = build_reactflow_system_xml(project) diff --git a/tests/test_simulation_safety.py b/tests/test_simulation_safety.py index f288fe8..1af6cab 100644 --- a/tests/test_simulation_safety.py +++ b/tests/test_simulation_safety.py @@ -107,15 +107,22 @@ class SimulationSampleTimeSafetyTests(unittest.TestCase): [1.0, 1.25], ) - def test_exact_point_limit_is_allowed(self) -> None: + def test_grid_can_exceed_the_legacy_point_limit(self) -> None: times = simulation_sample_times( - SolveIVPConfig(t_start=0.0, t_stop=1.0), + SolveIVPConfig(t_start=0.0, t_stop=5.0), 0.0001, ) - self.assertEqual(len(times), 10001) + self.assertEqual(len(times), 50001) self.assertEqual(times[0], 0.0) - self.assertEqual(times[-1], 1.0) + self.assertEqual(times[-1], 5.0) + + def test_system_xml_accepts_more_than_the_legacy_point_limit(self) -> None: + report = validate_system_xml_document( + _system_xml(t_start="0", t_stop="5", sample_step="0.0001") + ) + + self.assertTrue(report.valid, report.issues) def test_tiny_step_is_rejected_before_an_oversized_grid_is_allocated(self) -> None: with self.assertRaises(SimulationSampleTimeError) as caught: @@ -124,7 +131,10 @@ class SimulationSampleTimeSafetyTests(unittest.TestCase): 1.0e-300, ) - self.assertEqual(caught.exception.code, "SIMULATION_SAMPLE_COUNT_EXCEEDED") + self.assertEqual( + caught.exception.code, + "SIMULATION_SAMPLE_COUNT_UNREPRESENTABLE", + ) def test_non_finite_derived_duration_has_a_stable_error(self) -> None: with self.assertRaises(SimulationSampleTimeError) as caught: @@ -153,7 +163,7 @@ class SimulationSampleTimeSafetyTests(unittest.TestCase): cases = ( ( _system_xml(t_start="0", t_stop="1", sample_step="1e-300"), - "SIMULATION_SAMPLE_COUNT_EXCEEDED", + "SIMULATION_SAMPLE_COUNT_UNREPRESENTABLE", ), ( _system_xml(