Drive small until big

This commit is contained in:
whoisdomi
2026-09-14 10:46:04 -05:00
parent 64f8b75551
commit e369bdcc6a
11 changed files with 500 additions and 72 deletions
+1
View File
@@ -132,6 +132,7 @@ struct OnroadEvent @0xc4fa6047f024e718 {
audioFeedback @97;
bigModelLoading @100;
bigModelFailed @102;
bigModelPending @103;
soundsUnavailableDEPRECATED @47;
stockLkasDEPRECATED @98;
+1
View File
@@ -177,6 +177,7 @@ inline static std::unordered_map<std::string, ParamKeyAttributes> keys = {
{"UsbGpuActive", {CLEAR_ON_MANAGER_START | CLEAR_ON_OFFROAD_TRANSITION, BOOL}},
{"UsbGpuCompiled", {CLEAR_ON_MANAGER_START | CLEAR_ON_OFFROAD_TRANSITION, BOOL}},
{"UsbGpuLoading", {CLEAR_ON_MANAGER_START | CLEAR_ON_OFFROAD_TRANSITION, BOOL}},
{"UsbGpuPending", {CLEAR_ON_MANAGER_START | CLEAR_ON_OFFROAD_TRANSITION, BOOL}},
{"UsbGpuPresent", {CLEAR_ON_MANAGER_START | CLEAR_ON_OFFROAD_TRANSITION, BOOL}},
{"Version", {PERSISTENT, STRING}},
+185 -43
View File
@@ -5,6 +5,7 @@ from functools import cached_property
import json
import os
import struct
import threading
import usb1
from openpilot.system.hardware import HARDWARE, TICI
os.environ['GMMU'] = '0'
@@ -25,7 +26,7 @@ from openpilot.common.swaglog import cloudlog
from openpilot.common.params import Params
from openpilot.common.filter_simple import FirstOrderFilter
from openpilot.common.file_chunker import file_chunked_exists, open_file_chunked
from openpilot.common.realtime import config_realtime_process, DT_MDL
from openpilot.common.realtime import config_realtime_process, set_core_affinity, DT_MDL
from openpilot.common.transformations.camera import DEVICE_CAMERAS
from openpilot.common.transformations.model import get_warp_matrix
from openpilot.system.camerad.cameras.nv12_info import get_nv12_info
@@ -105,6 +106,15 @@ EXTERNAL_GPU_POWER_WAIT_TIMEOUT_SECONDS = 60.0
EXTERNAL_GPU_POWER_LOG_INTERVAL_SECONDS = 10.0
LAT_SMOOTH_BP = [2.0, 8.0]
# The background big-model loader must leave modeld's SCHED_FIFO core. Thread affinity is
# inherited from config_realtime_process(7, 54), so without this the loader would sit behind
# the 20 Hz publish loop on core 7 (shared with dmonitoringmodeld) and barely run at all.
BIG_MODEL_LOADER_CORES = {6}
class BigModelLoadCancelled(Exception):
"""Raised inside the background loader when modeld no longer wants the big model."""
def _set_hcq_wait_timeout(timeout_ms: int) -> None:
"""Update tinygrad's cached HCQ timeout for the external-GPU load/run phase."""
@@ -155,8 +165,10 @@ def _egmp_vehicle_ready(can_messages, bus: int) -> bool:
)
def wait_for_external_gpu_power_ready(CP=None) -> None:
def wait_for_external_gpu_power_ready(CP=None, cancel=None) -> None:
"""Wait out vehicle startup power transitions before initializing Chestnut."""
if cancel is not None and cancel.is_set():
raise BigModelLoadCancelled("cancelled before waiting for external GPU power")
device_type = HARDWARE.get_device_type()
egmp_bus = _egmp_ready_bus(CP)
services = ["pandaStates", "peripheralState"] + (["can"] if egmp_bus is not None else [])
@@ -167,6 +179,8 @@ def wait_for_external_gpu_power_ready(CP=None) -> None:
wait_started = time.monotonic()
while True:
if cancel is not None and cancel.is_set():
raise BigModelLoadCancelled("cancelled while waiting for external GPU power")
sm.update(1000)
now = time.monotonic()
if egmp_bus is not None and sm.updated["can"] and _egmp_vehicle_ready(sm["can"], egmp_bus):
@@ -854,33 +868,111 @@ def _load_model_state(cam_w: int, cam_h: int, selected_model: str, external_gpu_
)
def _load_external_gpu_model(cam_w: int, cam_h: int, selected_model: str, model_version: str = "",
CP=None, demo: bool = False) -> ModelState | None:
"""Load and warm the USB-GPU model without running another tinygrad model concurrently."""
candidate = None
try:
if not demo:
wait_for_external_gpu_power_ready(CP)
_set_hcq_wait_timeout(BIG_MODEL_LOAD_WAIT_TIMEOUT_MS)
wait_usbgpu_link()
candidate = ModelState(
cam_w,
cam_h,
True,
model_id_override=selected_model,
write_model_version=False,
model_version_override=model_version,
)
if not candidate.uses_external_gpu:
raise RuntimeError("external GPU model resolved to the builtin model")
candidate.warmup()
return candidate
except Exception:
cloudlog.exception("external GPU model load or warmup failed")
return None
finally:
_close_tinygrad_disk_cache_connection()
_set_hcq_wait_timeout(BIG_MODEL_RUN_WAIT_TIMEOUT_MS)
def _big_model_swap_allowed(engaged: bool, carcontrol_alive: bool, carstate_alive: bool,
vipc_dropped_frames: int, live_calib_seen: bool) -> bool:
"""Swapping models rebuilds the temporal input queues, so it must never happen under actuation.
Fails closed: without fresh carControl/carState we cannot trust that we are disengaged.
"""
return (
not engaged
and carcontrol_alive
and carstate_alive
and vipc_dropped_frames == 0
and live_calib_seen
)
class BigModelLoader:
"""Load and warm the Chestnut model on a background thread while the small model drives.
Everything this touches must be safe against a concurrently running small model, so unlike
the synchronous path it never calls _set_hcq_wait_timeout (hoisted to modeld startup) or
_isolate_next_model_artifact_load (which evicts buffer UOps from a tinygrad global cache).
"""
def __init__(self, cam_w: int, cam_h: int, CP=None, demo: bool = False):
self.cam_w = cam_w
self.cam_h = cam_h
self.CP = CP
self.demo = demo
self._thread: threading.Thread | None = None
self._cancel = threading.Event()
self._lock = threading.Lock()
self._result: ModelState | None = None
self._error = ""
self._model_id = ""
self._model_version = ""
self.started_t = 0.0
@property
def in_progress(self) -> bool:
return self._thread is not None and self._thread.is_alive()
def start(self, model_id: str, model_version: str = "") -> bool:
if self.in_progress:
return False
with self._lock:
self._result = None
self._error = ""
self._model_id = model_id
self._model_version = model_version
self._cancel.clear()
self.started_t = time.monotonic()
self._thread = threading.Thread(target=self._run, name="big_model_loader", daemon=True)
self._thread.start()
return True
def cancel(self) -> None:
self._cancel.set()
def take(self) -> tuple[ModelState | None, str]:
"""Hand over the finished load exactly once: (model, error). Both empty while in flight."""
if self._thread is None or self._thread.is_alive():
return None, ""
self._thread = None
with self._lock:
result, error = self._result, self._error
self._result = None
self._error = ""
return result, error
def _run(self) -> None:
try:
# Leave modeld's realtime core before doing any work; see BIG_MODEL_LOADER_CORES.
set_core_affinity(sorted(BIG_MODEL_LOADER_CORES))
if not self.demo:
wait_for_external_gpu_power_ready(self.CP, cancel=self._cancel)
if self._cancel.is_set():
raise BigModelLoadCancelled("cancelled before the external GPU link check")
wait_usbgpu_link()
candidate = ModelState(
self.cam_w,
self.cam_h,
True,
model_id_override=self._model_id,
write_model_version=False,
model_version_override=self._model_version,
)
if not candidate.uses_external_gpu:
raise RuntimeError("external GPU model resolved to the builtin model")
candidate.warmup()
if self._cancel.is_set():
raise BigModelLoadCancelled("cancelled after warmup")
with self._lock:
self._result = candidate
cloudlog.warning(f"background big model load finished in {time.monotonic() - self.started_t:.1f}s")
except BigModelLoadCancelled as exc:
cloudlog.warning(f"background big model load cancelled: {exc}")
with self._lock:
self._error = str(exc)
except Exception as exc:
cloudlog.exception("background big model load failed")
with self._lock:
self._error = str(exc) or exc.__class__.__name__
finally:
# tinygrad's disk cache handle is thread-local, so this closes only the loader's.
_close_tinygrad_disk_cache_connection()
def _model_versions() -> dict[str, str]:
@@ -1058,9 +1150,16 @@ def main(demo=False):
external_artifact = MODELS_PATH / f"{big_model_id}_driving_tinygrad.pkl"
external_artifact_ready = external_model_selected and file_chunked_exists(external_artifact)
external_gpu_requested = usbgpu_present_now and (bool(big_model_id) or model_lab_ready)
# The big model is loaded in the background while the small model drives, so the HCQ
# watchdog is set once here and never mutated again: it is process-global and shared by
# the QCOM and AMD devices, so changing it mid-drive would also move the small model's.
if external_gpu_requested:
_set_hcq_wait_timeout(BIG_MODEL_LOAD_WAIT_TIMEOUT_MS)
params.put_bool("UsbGpuPresent", usbgpu_present_now)
params.put_bool("UsbGpuCompiled", external_artifact_ready or model_lab_ready)
params.put_bool("UsbGpuActive", False)
params.put_bool("UsbGpuPending", False)
params.put_bool("UsbGpuLoading", external_gpu_requested)
_set_model_lab_runtime(
params,
@@ -1147,15 +1246,8 @@ def main(demo=False):
else:
CP = messaging.log_from_bytes(params.get("CarParams", block=True), car.CarParams)
big_model = _load_external_gpu_model(
vipc_client_main.width,
vipc_client_main.height,
selected_model,
selected_model_version,
CP,
demo,
)
# Start on the small model so openpilot is drivable immediately; the big model is loaded
# in the background below and promoted once the driver disengages.
small_model = _load_model_state(
vipc_client_main.width,
vipc_client_main.height,
@@ -1165,10 +1257,7 @@ def main(demo=False):
small_model_version,
False,
)
model = big_model if big_model is not None else small_model
if big_model is not None:
params.put("ModelVersion", model.policy_generation)
params.put("DrivingModelVersion", model.policy_generation)
model = small_model
else:
model = _load_model_state(
vipc_client_main.width,
@@ -1183,9 +1272,15 @@ def main(demo=False):
set_runtime_model_params(params, model.model_id, model.policy_generation)
external_gpu_active = model_lab_active or model.uses_external_gpu
# Load the big model in the background so the small model can drive in the meantime.
big_loader = None
if external_gpu_requested and not model_lab_active and big_model_id:
big_loader = BigModelLoader(vipc_client_main.width, vipc_client_main.height, CP, demo)
big_loader.start(big_model_id, big_model_version)
params.put_bool("UsbGpuCompiled", external_artifact_ready or model_lab_ready)
params.put_bool("UsbGpuActive", external_gpu_active)
params.put_bool("UsbGpuLoading", False)
params.put_bool("UsbGpuPending", False)
params.put_bool("UsbGpuLoading", big_loader is not None)
_set_model_lab_runtime(
params,
requested=model_lab_requested,
@@ -1340,6 +1435,48 @@ def main(demo=False):
chestnut_state.big = False
cloudlog.error(f"Model Laboratory stopped: {model_lab_error}")
# Collect a finished background big-model load, then promote it once the driver is
# disengaged. Swapping rebuilds the temporal input queues, so it must not happen
# under actuation.
if big_loader is not None and not big_loader.in_progress:
loaded_big_model, big_load_error = big_loader.take()
big_loader = None
params.put_bool("UsbGpuLoading", False)
if loaded_big_model is not None:
big_model = loaded_big_model
params.put_bool("UsbGpuPending", True)
cloudlog.warning("big model ready; waiting for the driver to disengage before using it")
else:
params.put_bool("UsbGpuPending", False)
params.put_bool("UsbGpuActive", False)
cloudlog.error(f"big model unavailable, staying on the small model: {big_load_error}")
if big_model is not None and model is not big_model and _big_model_swap_allowed(
sm["carControl"].enabled,
sm.alive["carControl"],
sm.alive["carState"],
vipc_dropped_frames,
live_calib_seen,
):
big_model._reset_state()
model = big_model
external_gpu_active = True
# Nothing from the small model carries over: re-arm the frame-drop warmup and drop
# the rolling probability buffers and previous action, which are model specific.
run_count = 0
frame_dropped_filter.x = 0.
publish_state = PublishState()
prev_action = log.ModelDataV2.Action()
params.put("ModelVersion", model.policy_generation)
params.put("DrivingModelVersion", model.policy_generation)
set_runtime_model_params(params, model.model_id, model.policy_generation)
params.put_bool("UsbGpuActive", True)
params.put_bool("UsbGpuPending", False)
params.put_bool("UsbGpuLoading", False)
if chestnut_state is not None:
chestnut_state.big = True
cloudlog.warning(f"now driving on the big model {model.model_id}")
frame_drop_ratio = frames_dropped / (1 + frames_dropped)
dropped_frame = vipc_dropped_frames > 0
if dropped_frame and (model.can_prepare_only or (model_lab_longitudinal is not None and model_lab_longitudinal.can_prepare_only)):
@@ -1433,11 +1570,16 @@ def main(demo=False):
cloudlog.exception("external GPU model failed, falling back to active small model")
model = small_model
big_model = None
# A failed big model is not retried in this drive, so stop any load still in flight.
if big_loader is not None:
big_loader.cancel()
big_loader = None
params.put_bool("UsbGpuActive", False)
external_gpu_active = False
params.put("ModelVersion", model.policy_generation)
params.put("DrivingModelVersion", model.policy_generation)
set_runtime_model_params(params, model.model_id, model.policy_generation)
params.put_bool("UsbGpuPending", False)
params.put_bool("UsbGpuLoading", False)
if chestnut_state is not None:
chestnut_state.big = False
+197 -16
View File
@@ -1,5 +1,6 @@
import io
import struct
import threading
from types import MethodType
from types import SimpleNamespace
@@ -269,45 +270,225 @@ def test_tinygrad_empty_thread_local_cache_holder_is_safe(monkeypatch):
assert tinygrad_helpers._db_connection is holder
def test_external_gpu_load_finishes_before_native_model_can_start(monkeypatch):
calls = []
def _stub_big_model_loader(monkeypatch, calls, *, uses_external_gpu=True, model_error=None):
class FakeModelState:
uses_external_gpu = True
def __init__(self, cam_w, cam_h, external_gpu_active, model_id_override, write_model_version,
model_version_override):
calls.append(("model", cam_w, cam_h, external_gpu_active, model_id_override,
write_model_version, model_version_override))
if model_error is not None:
raise model_error
self.uses_external_gpu = uses_external_gpu
def warmup(self):
calls.append("warmup")
monkeypatch.setattr(modeld, "set_core_affinity", lambda cores: calls.append(("affinity", tuple(cores))))
monkeypatch.setattr(modeld, "wait_usbgpu_link", lambda: calls.append("link"))
monkeypatch.setattr(modeld, "wait_for_external_gpu_power_ready", lambda CP: calls.append(("power", CP)))
monkeypatch.setattr(modeld, "_set_hcq_wait_timeout", lambda timeout: calls.append(("timeout", timeout)))
monkeypatch.setattr(modeld, "wait_for_external_gpu_power_ready",
lambda CP, cancel=None: calls.append(("power", CP)))
monkeypatch.setattr(modeld, "_close_tinygrad_disk_cache_connection", lambda: calls.append("close_cache"))
monkeypatch.setattr(modeld, "ModelState", FakeModelState)
monkeypatch.setattr(
modeld,
"tinygrad_dev_config",
lambda *_args: (_ for _ in ()).throw(AssertionError("runtime must not change tinygrad's process-global DEV")),
)
return FakeModelState
loaded = modeld._load_external_gpu_model(1928, 1208, "big-model", "v15", "car-params")
assert isinstance(loaded, FakeModelState)
def test_background_big_model_load_leaves_the_running_model_untouched(monkeypatch):
calls = []
fake_model_state = _stub_big_model_loader(monkeypatch, calls)
# The small model is already driving on these process-global settings, so the background
# load must not touch tinygrad's DEV, its HCQ watchdog, or its shared buffer UOp cache.
for name, detail in (
("tinygrad_dev_config", "runtime must not change tinygrad's process-global DEV"),
("_set_hcq_wait_timeout", "the background load must not move the running model's HCQ watchdog"),
("_isolate_next_model_artifact_load", "the background load must not evict the running model's buffers"),
):
monkeypatch.setattr(modeld, name, lambda *_args, _d=detail: (_ for _ in ()).throw(AssertionError(_d)))
loader = modeld.BigModelLoader(1928, 1208, "car-params")
assert loader.start("big-model", "v15")
loader._thread.join(timeout=10)
loaded, error = loader.take()
assert isinstance(loaded, fake_model_state)
assert error == ""
assert calls == [
("affinity", tuple(sorted(modeld.BIG_MODEL_LOADER_CORES))),
("power", "car-params"),
("timeout", modeld.BIG_MODEL_LOAD_WAIT_TIMEOUT_MS),
"link",
("model", 1928, 1208, True, "big-model", False, "v15"),
"warmup",
"close_cache",
("timeout", modeld.BIG_MODEL_RUN_WAIT_TIMEOUT_MS),
]
def test_background_big_model_loader_runs_off_modelds_realtime_core():
# modeld is SCHED_FIFO on core 7 and threads inherit its affinity, so a loader left there
# would be starved behind the 20 Hz publish loop.
assert 7 not in modeld.BIG_MODEL_LOADER_CORES
assert modeld.BIG_MODEL_LOADER_CORES
@pytest.mark.parametrize("failure", [
dict(model_error=RuntimeError("artifact is corrupt")),
dict(uses_external_gpu=False),
])
def test_background_big_model_load_reports_failure_without_a_model(monkeypatch, failure):
calls = []
_stub_big_model_loader(monkeypatch, calls, **failure)
loader = modeld.BigModelLoader(1928, 1208, "car-params")
loader.start("big-model", "v15")
loader._thread.join(timeout=10)
loaded, error = loader.take()
assert loaded is None
assert error
assert not loader.in_progress
# Even a failed load must release the loader thread's own tinygrad cache handle.
assert "close_cache" in calls
def test_background_big_model_load_is_handed_over_exactly_once(monkeypatch):
calls = []
fake_model_state = _stub_big_model_loader(monkeypatch, calls)
loader = modeld.BigModelLoader(1928, 1208, "car-params")
loader.start("big-model", "v15")
loader._thread.join(timeout=10)
assert isinstance(loader.take()[0], fake_model_state)
# A second take must not promote the same model again.
assert loader.take() == (None, "")
def test_cancelled_big_model_load_never_builds_a_model(monkeypatch):
calls = []
def cancelling_power_wait(CP, cancel=None):
calls.append(("power", CP))
cancel.set()
raise modeld.BigModelLoadCancelled("cancelled while waiting for external GPU power")
_stub_big_model_loader(monkeypatch, calls)
monkeypatch.setattr(modeld, "wait_for_external_gpu_power_ready", cancelling_power_wait)
loader = modeld.BigModelLoader(1928, 1208, "car-params")
loader.start("big-model", "v15")
loader._thread.join(timeout=10)
loaded, error = loader.take()
assert loaded is None
assert error
assert not any(call[0] == "model" for call in calls if isinstance(call, tuple))
def test_external_gpu_power_wait_aborts_when_the_loader_is_cancelled(monkeypatch):
cancel = threading.Event()
cancel.set()
monkeypatch.setattr(modeld, "SubMaster",
lambda *_a, **_k: pytest.fail("a cancelled load must not start waiting on power"))
with pytest.raises(modeld.BigModelLoadCancelled):
modeld.wait_for_external_gpu_power_ready("car-params", cancel=cancel)
def test_big_model_promotion_waits_for_the_driver_to_disengage(monkeypatch):
"""Drive the real collect-and-promote blocks lifted out of main()'s loop."""
import ast
from pathlib import Path
source = (Path(modeld.__file__).with_name("modeld.py")).read_text(encoding="utf-8")
main_fn = next(n for n in ast.parse(source).body if isinstance(n, ast.FunctionDef) and n.name == "main")
def names(node):
return {n.id for n in ast.walk(node) if isinstance(n, ast.Name)}
collect_block = next(n for n in ast.walk(main_fn)
if isinstance(n, ast.If) and "big_loader" in names(n.test)
and "in_progress" in ast.dump(n.test))
promote_block = next(n for n in ast.walk(main_fn)
if isinstance(n, ast.If) and "_big_model_swap_allowed" in names(n.test))
promote = compile(ast.Module(body=[collect_block, promote_block], type_ignores=[]),
"<promotion>", "exec")
class FakeModel:
model_id = "big-model"
policy_generation = "v15"
def __init__(self):
self.reset = 0
def _reset_state(self):
self.reset += 1
class FakeLoader:
def __init__(self, model):
self.in_progress = True
self._model = model
def take(self):
return self._model, ""
written = {}
big = FakeModel()
loader = FakeLoader(big)
small = FakeModel()
small.model_id = "small-model"
scope = {
"big_loader": loader, "big_model": None, "model": small, "small_model": small,
"params": SimpleNamespace(put_bool=lambda k, v: written.__setitem__(k, v),
put=lambda k, v: written.__setitem__(k, v)),
"cloudlog": SimpleNamespace(warning=lambda *_a: None, error=lambda *_a: None),
"vipc_dropped_frames": 0, "live_calib_seen": True, "external_gpu_active": False,
"run_count": 99, "frame_dropped_filter": SimpleNamespace(x=5.0),
"PublishState": lambda: "fresh", "publish_state": "stale",
"prev_action": "stale", "log": modeld.log, "chestnut_state": SimpleNamespace(big=False),
"set_runtime_model_params": lambda *_a: None,
"_big_model_swap_allowed": modeld._big_model_swap_allowed,
}
scope["sm"] = type("SM", (), {
"__getitem__": lambda _s, k: SimpleNamespace(enabled=scope["_engaged"]),
"alive": {"carControl": True, "carState": True},
})()
# Still loading: nothing is collected and the small model keeps driving.
scope["_engaged"] = True
exec(promote, scope) # noqa: S102
assert scope["model"] is small and scope["big_model"] is None
# Load finishes while engaged: the model is collected but must NOT be promoted.
loader.in_progress = False
exec(promote, scope) # noqa: S102
assert scope["big_model"] is big
assert scope["model"] is small, "must not swap models while the driver is engaged"
assert written["UsbGpuPending"] is True
assert written["UsbGpuLoading"] is False
assert big.reset == 0
# The driver disengages: now the big model takes over.
scope["_engaged"] = False
exec(promote, scope) # noqa: S102
assert scope["model"] is big
assert big.reset == 1, "temporal queues must be reset before the big model drives"
assert scope["run_count"] == 0 and scope["frame_dropped_filter"].x == 0.
assert scope["publish_state"] == "fresh" and scope["prev_action"] != "stale"
assert written["UsbGpuActive"] is True and written["UsbGpuPending"] is False
assert scope["chestnut_state"].big is True
@pytest.mark.parametrize("kwargs,expected", [
(dict(), True),
(dict(engaged=True), False),
(dict(carcontrol_alive=False), False),
(dict(carstate_alive=False), False),
(dict(vipc_dropped_frames=1), False),
(dict(live_calib_seen=False), False),
])
def test_big_model_swap_only_while_disengaged_with_fresh_state(kwargs, expected):
defaults = dict(engaged=False, carcontrol_alive=True, carstate_alive=True,
vipc_dropped_frames=0, live_calib_seen=True)
assert modeld._big_model_swap_allowed(**(defaults | kwargs)) is expected
def test_external_gpu_nonfinite_outputs_trigger_fallback(monkeypatch):
class FakeTensor:
@staticmethod
+8 -1
View File
@@ -533,7 +533,14 @@ EVENTS: dict[int, dict[str, Alert | AlertCallbackType]] = {
},
EventName.bigModelLoading: {
ET.NO_ENTRY: NoEntryAlert("Big Model Loading"),
ET.PERMANENT: NormalPermanentAlert("Big Model Loading",
"Driving on the small model"),
},
EventName.bigModelPending: {
ET.PERMANENT: NormalPermanentAlert("Big Model Ready",
"Disengage and re-engage to use it",
duration=10.),
},
EventName.bigModelFailed: {
+16 -5
View File
@@ -255,7 +255,7 @@ class SelfdriveD:
self.big_model_attempted = False
self.big_model_active = False
self.big_model_failed = False
self.big_model_ready_t = 0.
self.big_model_swap_t = 0.
self.experimental_mode = False
self.ecu_disable_failed = False
self.ecu_disable_failed_checked = not (
@@ -394,15 +394,23 @@ class SelfdriveD:
loading = self.params.get_bool("UsbGpuLoading")
if loading:
self.big_model_attempted = True
if self.big_model_loading and not loading:
self.big_model_ready_t = time.monotonic()
self.big_model_loading = loading
if loading:
self.events.add(EventName.bigModelLoading)
# The big model loads in the background while the small model drives, so it sits
# loaded-but-unused until the driver disengages. That is a success, not a failure.
pending = self.params.get_bool("UsbGpuPending")
if pending:
self.big_model_attempted = True
self.events.add(EventName.bigModelPending)
big_active = self.params.get("UsbGpuActive")
if self.big_model_active != (big_active is True):
self.big_model_swap_t = time.monotonic()
model_unavailable = self.big_model_active and self.sm.seen['modelV2'] and not self.sm.alive['modelV2']
big_failed = self.big_model_attempted and not loading and (big_active is False or model_unavailable)
big_failed = self.big_model_attempted and not loading and not pending and \
(big_active is False or model_unavailable)
if big_failed and not self.big_model_failed:
self.events.add(EventName.bigModelFailed)
self.big_model_failed = big_failed
@@ -697,7 +705,10 @@ class SelfdriveD:
(contains_event_type(self.events, self.starpilot_events, ET.SOFT_DISABLE) or
contains_event_type(self.events, self.starpilot_events, ET.IMMEDIATE_DISABLE))
no_system_errors = (not has_disable_events) or (len(self.events) == num_events)
big_model_settling = self.big_model_loading or time.monotonic() < self.big_model_ready_t + 5.
# modeld publishes on the small model throughout the background load, so only the model
# swap itself can briefly disturb the stream. Suppressing for the whole load would hide
# a genuinely dead modeld for as long as the load takes.
big_model_settling = time.monotonic() < self.big_model_swap_t + 2.
all_checks = self.sm.all_checks()
all_alive = self.sm.all_alive() if not all_checks else True
all_freq_ok = self.sm.all_freq_ok() if not all_checks else True
@@ -0,0 +1,60 @@
import ast
from pathlib import Path
import pytest
from cereal import log
from openpilot.selfdrive.selfdrived.events import EVENTS, ET
EventName = log.OnroadEvent.EventName
def test_big_model_loading_does_not_block_engagement():
# The big model loads in the background while the small model drives, so waiting for it
# must never keep the driver from engaging.
assert ET.NO_ENTRY not in EVENTS[EventName.bigModelLoading]
assert ET.PERMANENT in EVENTS[EventName.bigModelLoading]
def test_big_model_pending_is_advisory_only():
pending = EVENTS[EventName.bigModelPending]
assert set(pending) == {ET.PERMANENT}
def test_big_model_failure_still_disengages():
assert ET.SOFT_DISABLE in EVENTS[EventName.bigModelFailed]
def _big_failed(*, attempted, loading, pending, big_active, model_unavailable=False):
"""Mirror of selfdrived's big_failed expression, extracted from the source."""
source = (Path(__file__).parents[1] / "selfdrived.py").read_text(encoding="utf-8")
tree = ast.parse(source)
assign = next(
node for node in ast.walk(tree)
if isinstance(node, ast.Assign)
and any(isinstance(t, ast.Name) and t.id == "big_failed" for t in node.targets)
)
scope = {
"self": type("S", (), {"big_model_attempted": attempted})(),
"loading": loading,
"pending": pending,
"big_active": big_active,
"model_unavailable": model_unavailable,
}
return eval(compile(ast.Expression(assign.value), "<big_failed>", "eval"), {}, scope) # noqa: S307
@pytest.mark.parametrize("state,expected", [
# A loaded model waiting for the driver to disengage is a success, not a failure.
(dict(attempted=True, loading=False, pending=True, big_active=False), False),
(dict(attempted=True, loading=True, pending=False, big_active=False), False),
(dict(attempted=True, loading=False, pending=False, big_active=True), False),
# Genuine failures must still be reported.
(dict(attempted=True, loading=False, pending=False, big_active=False), True),
(dict(attempted=True, loading=False, pending=False, big_active=True,
model_unavailable=True), True),
# Never report a failure for a big model that was never attempted.
(dict(attempted=False, loading=False, pending=False, big_active=False), False),
])
def test_pending_big_model_is_not_reported_as_failed(state, expected):
assert _big_failed(**state) is expected
@@ -62,12 +62,17 @@ class ModelSourceWidget(LayoutWidget):
return self.SIZE
@staticmethod
def _big_model_failed(active: bool | None, usbgpu: bool, model_seen: bool, model_alive: bool) -> bool:
def _big_model_failed(active: bool | None, usbgpu: bool, model_seen: bool, model_alive: bool,
pending: bool = False) -> bool:
# A model that finished loading but is waiting for the driver to disengage is not a failure.
if pending and usbgpu:
return False
return active is False or not usbgpu or (active is True and model_seen and not model_alive)
@staticmethod
def _status_for(loading: bool, small_model_engaged: bool, big_failed: bool) -> ModelSourceStatus:
if loading:
def _status_for(loading: bool, small_model_engaged: bool, big_failed: bool,
pending: bool = False) -> ModelSourceStatus:
if loading or pending:
return ModelSourceStatus.LOADING
if small_model_engaged:
return ModelSourceStatus.FALLBACK_ENGAGED
@@ -84,16 +89,18 @@ class ModelSourceWidget(LayoutWidget):
model_seen = sm.recv_frame["modelV2"] > ui_state.started_frame
model_alive = sm.alive["modelV2"] if model_seen else True
loading = ui_state.usbgpu_loading
big_failed = self._big_model_failed(ui_state.usbgpu_active, ui_state.usbgpu, model_seen, model_alive)
pending = ui_state.usbgpu_pending
big_failed = self._big_model_failed(ui_state.usbgpu_active, ui_state.usbgpu, model_seen, model_alive, pending)
engaged = sm["selfdriveState"].enabled
if engaged and not self._engaged and not loading and ui_state.usbgpu_active is not True and model_seen:
if engaged and not self._engaged and not loading and not pending \
and ui_state.usbgpu_active is not True and model_seen:
self._small_model_engaged = True
if engaged != self._engaged:
self._fade_time = rl.get_time() if engaged else 0.0
self._engaged = engaged
self._small_model_engaged &= big_failed
self._status = self._status_for(loading, self._small_model_engaged, big_failed)
self._status = self._status_for(loading, self._small_model_engaged, big_failed, pending)
def _render(self, rect: rl.Rectangle) -> None:
if self._status is None:
+4 -1
View File
@@ -344,8 +344,11 @@ class Soundd:
if now - self.model_ready_last_check >= 0.25:
self.model_ready_last_check = now
try:
# The big model now loads in the background, so it is announced when it becomes
# usable (pending the driver's next disengage) rather than only once it is running.
self.model_ready_pending = self.model_ready_chime.update(
active=self.model_ready_params.get_bool("UsbGpuActive"),
active=(self.model_ready_params.get_bool("UsbGpuPending") or
self.model_ready_params.get_bool("UsbGpuActive")),
loading=self.model_ready_params.get_bool("UsbGpuLoading"),
onroad=self.model_ready_params.get_bool("IsOnroad"),
enabled=self.model_ready_params.get_bool("GpuModelReadySound"), now=now)
@@ -32,6 +32,18 @@ def test_model_source_failure_detection_matches_the_backend_state_contract():
assert not failed(True, True, False, True)
def test_model_source_shows_a_pending_big_model_as_still_loading():
# The big model loads in the background, so "loaded, waiting for a disengage" must read
# as in-progress rather than as a failure.
status = model_source.ModelSourceWidget._status_for
failed = model_source.ModelSourceWidget._big_model_failed
assert status(False, False, False, True) is model_source.ModelSourceStatus.LOADING
assert not failed(False, True, False, True, True)
# Chestnut going away while pending is still a genuine failure.
assert failed(False, False, False, True, True)
def test_model_source_latches_small_model_engagement_until_the_big_model_recovers(monkeypatch):
widget = object.__new__(model_source.ModelSourceWidget)
widget._small_model_engaged = False
@@ -50,6 +62,7 @@ def test_model_source_latches_small_model_engagement_until_the_big_model_recover
usbgpu_compiled=True,
usbgpu_active=False,
usbgpu_loading=False,
usbgpu_pending=False,
),
)
monkeypatch.setattr(model_source.rl, "get_time", lambda: 42.0)
+2
View File
@@ -92,6 +92,7 @@ class UIState:
self.usbgpu_compiled: bool = self.params.get_bool("UsbGpuCompiled")
self.usbgpu_active: bool = self.params.get_bool("UsbGpuActive")
self.usbgpu_loading: bool = self.params.get_bool("UsbGpuLoading")
self.usbgpu_pending: bool = self.params.get_bool("UsbGpuPending")
self.started: bool = False
self.ignition: bool = False
self.recording_audio: bool = False
@@ -215,6 +216,7 @@ class UIState:
self.usbgpu_compiled = params.get_bool("UsbGpuCompiled")
self.usbgpu_active = params.get_bool("UsbGpuActive")
self.usbgpu_loading = params.get_bool("UsbGpuLoading")
self.usbgpu_pending = params.get_bool("UsbGpuPending")
self.switchback_mode_enabled = self.params_memory.get_bool("SwitchbackModeEnabled") if self.started else False
self.conditional_status = self.params_memory.get_int("CEStatus", default=0) if self.started else 0
mark_progress("ui.update.after_state_params")