From e369bdcc6a572dda0e36b4349885dd63c15d131a Mon Sep 17 00:00:00 2001 From: whoisdomi Date: Mon, 14 Sep 2026 10:46:04 -0500 Subject: [PATCH] Drive small until big --- cereal/log.capnp | 1 + common/params_keys.h | 1 + selfdrive/modeld/modeld.py | 228 ++++++++++++++---- selfdrive/modeld/tests/test_usbgpu_helpers.py | 213 ++++++++++++++-- selfdrive/selfdrived/events.py | 9 +- selfdrive/selfdrived/selfdrived.py | 21 +- .../tests/test_big_model_engagement.py | 60 +++++ .../onroad/starpilot/widgets/model_source.py | 19 +- selfdrive/ui/soundd.py | 5 +- .../ui/tests/test_model_source_widget.py | 13 + selfdrive/ui/ui_state.py | 2 + 11 files changed, 500 insertions(+), 72 deletions(-) create mode 100644 selfdrive/selfdrived/tests/test_big_model_engagement.py diff --git a/cereal/log.capnp b/cereal/log.capnp index 8179fa5275..0a6210bf75 100644 --- a/cereal/log.capnp +++ b/cereal/log.capnp @@ -132,6 +132,7 @@ struct OnroadEvent @0xc4fa6047f024e718 { audioFeedback @97; bigModelLoading @100; bigModelFailed @102; + bigModelPending @103; soundsUnavailableDEPRECATED @47; stockLkasDEPRECATED @98; diff --git a/common/params_keys.h b/common/params_keys.h index fc68da5409..ac43026c3f 100644 --- a/common/params_keys.h +++ b/common/params_keys.h @@ -177,6 +177,7 @@ inline static std::unordered_map keys = { {"UsbGpuActive", {CLEAR_ON_MANAGER_START | CLEAR_ON_OFFROAD_TRANSITION, BOOL}}, {"UsbGpuCompiled", {CLEAR_ON_MANAGER_START | CLEAR_ON_OFFROAD_TRANSITION, BOOL}}, {"UsbGpuLoading", {CLEAR_ON_MANAGER_START | CLEAR_ON_OFFROAD_TRANSITION, BOOL}}, + {"UsbGpuPending", {CLEAR_ON_MANAGER_START | CLEAR_ON_OFFROAD_TRANSITION, BOOL}}, {"UsbGpuPresent", {CLEAR_ON_MANAGER_START | CLEAR_ON_OFFROAD_TRANSITION, BOOL}}, {"Version", {PERSISTENT, STRING}}, diff --git a/selfdrive/modeld/modeld.py b/selfdrive/modeld/modeld.py index 797b29ea25..3c660457d7 100755 --- a/selfdrive/modeld/modeld.py +++ b/selfdrive/modeld/modeld.py @@ -5,6 +5,7 @@ from functools import cached_property import json import os import struct +import threading import usb1 from openpilot.system.hardware import HARDWARE, TICI os.environ['GMMU'] = '0' @@ -25,7 +26,7 @@ from openpilot.common.swaglog import cloudlog from openpilot.common.params import Params from openpilot.common.filter_simple import FirstOrderFilter from openpilot.common.file_chunker import file_chunked_exists, open_file_chunked -from openpilot.common.realtime import config_realtime_process, DT_MDL +from openpilot.common.realtime import config_realtime_process, set_core_affinity, DT_MDL from openpilot.common.transformations.camera import DEVICE_CAMERAS from openpilot.common.transformations.model import get_warp_matrix from openpilot.system.camerad.cameras.nv12_info import get_nv12_info @@ -105,6 +106,15 @@ EXTERNAL_GPU_POWER_WAIT_TIMEOUT_SECONDS = 60.0 EXTERNAL_GPU_POWER_LOG_INTERVAL_SECONDS = 10.0 LAT_SMOOTH_BP = [2.0, 8.0] +# The background big-model loader must leave modeld's SCHED_FIFO core. Thread affinity is +# inherited from config_realtime_process(7, 54), so without this the loader would sit behind +# the 20 Hz publish loop on core 7 (shared with dmonitoringmodeld) and barely run at all. +BIG_MODEL_LOADER_CORES = {6} + + +class BigModelLoadCancelled(Exception): + """Raised inside the background loader when modeld no longer wants the big model.""" + def _set_hcq_wait_timeout(timeout_ms: int) -> None: """Update tinygrad's cached HCQ timeout for the external-GPU load/run phase.""" @@ -155,8 +165,10 @@ def _egmp_vehicle_ready(can_messages, bus: int) -> bool: ) -def wait_for_external_gpu_power_ready(CP=None) -> None: +def wait_for_external_gpu_power_ready(CP=None, cancel=None) -> None: """Wait out vehicle startup power transitions before initializing Chestnut.""" + if cancel is not None and cancel.is_set(): + raise BigModelLoadCancelled("cancelled before waiting for external GPU power") device_type = HARDWARE.get_device_type() egmp_bus = _egmp_ready_bus(CP) services = ["pandaStates", "peripheralState"] + (["can"] if egmp_bus is not None else []) @@ -167,6 +179,8 @@ def wait_for_external_gpu_power_ready(CP=None) -> None: wait_started = time.monotonic() while True: + if cancel is not None and cancel.is_set(): + raise BigModelLoadCancelled("cancelled while waiting for external GPU power") sm.update(1000) now = time.monotonic() if egmp_bus is not None and sm.updated["can"] and _egmp_vehicle_ready(sm["can"], egmp_bus): @@ -854,33 +868,111 @@ def _load_model_state(cam_w: int, cam_h: int, selected_model: str, external_gpu_ ) -def _load_external_gpu_model(cam_w: int, cam_h: int, selected_model: str, model_version: str = "", - CP=None, demo: bool = False) -> ModelState | None: - """Load and warm the USB-GPU model without running another tinygrad model concurrently.""" - candidate = None - try: - if not demo: - wait_for_external_gpu_power_ready(CP) - _set_hcq_wait_timeout(BIG_MODEL_LOAD_WAIT_TIMEOUT_MS) - wait_usbgpu_link() - candidate = ModelState( - cam_w, - cam_h, - True, - model_id_override=selected_model, - write_model_version=False, - model_version_override=model_version, - ) - if not candidate.uses_external_gpu: - raise RuntimeError("external GPU model resolved to the builtin model") - candidate.warmup() - return candidate - except Exception: - cloudlog.exception("external GPU model load or warmup failed") - return None - finally: - _close_tinygrad_disk_cache_connection() - _set_hcq_wait_timeout(BIG_MODEL_RUN_WAIT_TIMEOUT_MS) +def _big_model_swap_allowed(engaged: bool, carcontrol_alive: bool, carstate_alive: bool, + vipc_dropped_frames: int, live_calib_seen: bool) -> bool: + """Swapping models rebuilds the temporal input queues, so it must never happen under actuation. + + Fails closed: without fresh carControl/carState we cannot trust that we are disengaged. + """ + return ( + not engaged + and carcontrol_alive + and carstate_alive + and vipc_dropped_frames == 0 + and live_calib_seen + ) + + +class BigModelLoader: + """Load and warm the Chestnut model on a background thread while the small model drives. + + Everything this touches must be safe against a concurrently running small model, so unlike + the synchronous path it never calls _set_hcq_wait_timeout (hoisted to modeld startup) or + _isolate_next_model_artifact_load (which evicts buffer UOps from a tinygrad global cache). + """ + + def __init__(self, cam_w: int, cam_h: int, CP=None, demo: bool = False): + self.cam_w = cam_w + self.cam_h = cam_h + self.CP = CP + self.demo = demo + self._thread: threading.Thread | None = None + self._cancel = threading.Event() + self._lock = threading.Lock() + self._result: ModelState | None = None + self._error = "" + self._model_id = "" + self._model_version = "" + self.started_t = 0.0 + + @property + def in_progress(self) -> bool: + return self._thread is not None and self._thread.is_alive() + + def start(self, model_id: str, model_version: str = "") -> bool: + if self.in_progress: + return False + with self._lock: + self._result = None + self._error = "" + self._model_id = model_id + self._model_version = model_version + self._cancel.clear() + self.started_t = time.monotonic() + self._thread = threading.Thread(target=self._run, name="big_model_loader", daemon=True) + self._thread.start() + return True + + def cancel(self) -> None: + self._cancel.set() + + def take(self) -> tuple[ModelState | None, str]: + """Hand over the finished load exactly once: (model, error). Both empty while in flight.""" + if self._thread is None or self._thread.is_alive(): + return None, "" + self._thread = None + with self._lock: + result, error = self._result, self._error + self._result = None + self._error = "" + return result, error + + def _run(self) -> None: + try: + # Leave modeld's realtime core before doing any work; see BIG_MODEL_LOADER_CORES. + set_core_affinity(sorted(BIG_MODEL_LOADER_CORES)) + if not self.demo: + wait_for_external_gpu_power_ready(self.CP, cancel=self._cancel) + if self._cancel.is_set(): + raise BigModelLoadCancelled("cancelled before the external GPU link check") + wait_usbgpu_link() + candidate = ModelState( + self.cam_w, + self.cam_h, + True, + model_id_override=self._model_id, + write_model_version=False, + model_version_override=self._model_version, + ) + if not candidate.uses_external_gpu: + raise RuntimeError("external GPU model resolved to the builtin model") + candidate.warmup() + if self._cancel.is_set(): + raise BigModelLoadCancelled("cancelled after warmup") + with self._lock: + self._result = candidate + cloudlog.warning(f"background big model load finished in {time.monotonic() - self.started_t:.1f}s") + except BigModelLoadCancelled as exc: + cloudlog.warning(f"background big model load cancelled: {exc}") + with self._lock: + self._error = str(exc) + except Exception as exc: + cloudlog.exception("background big model load failed") + with self._lock: + self._error = str(exc) or exc.__class__.__name__ + finally: + # tinygrad's disk cache handle is thread-local, so this closes only the loader's. + _close_tinygrad_disk_cache_connection() def _model_versions() -> dict[str, str]: @@ -1058,9 +1150,16 @@ def main(demo=False): external_artifact = MODELS_PATH / f"{big_model_id}_driving_tinygrad.pkl" external_artifact_ready = external_model_selected and file_chunked_exists(external_artifact) external_gpu_requested = usbgpu_present_now and (bool(big_model_id) or model_lab_ready) + # The big model is loaded in the background while the small model drives, so the HCQ + # watchdog is set once here and never mutated again: it is process-global and shared by + # the QCOM and AMD devices, so changing it mid-drive would also move the small model's. + if external_gpu_requested: + _set_hcq_wait_timeout(BIG_MODEL_LOAD_WAIT_TIMEOUT_MS) + params.put_bool("UsbGpuPresent", usbgpu_present_now) params.put_bool("UsbGpuCompiled", external_artifact_ready or model_lab_ready) params.put_bool("UsbGpuActive", False) + params.put_bool("UsbGpuPending", False) params.put_bool("UsbGpuLoading", external_gpu_requested) _set_model_lab_runtime( params, @@ -1147,15 +1246,8 @@ def main(demo=False): else: CP = messaging.log_from_bytes(params.get("CarParams", block=True), car.CarParams) - big_model = _load_external_gpu_model( - vipc_client_main.width, - vipc_client_main.height, - selected_model, - selected_model_version, - CP, - demo, - ) - + # Start on the small model so openpilot is drivable immediately; the big model is loaded + # in the background below and promoted once the driver disengages. small_model = _load_model_state( vipc_client_main.width, vipc_client_main.height, @@ -1165,10 +1257,7 @@ def main(demo=False): small_model_version, False, ) - model = big_model if big_model is not None else small_model - if big_model is not None: - params.put("ModelVersion", model.policy_generation) - params.put("DrivingModelVersion", model.policy_generation) + model = small_model else: model = _load_model_state( vipc_client_main.width, @@ -1183,9 +1272,15 @@ def main(demo=False): set_runtime_model_params(params, model.model_id, model.policy_generation) external_gpu_active = model_lab_active or model.uses_external_gpu + # Load the big model in the background so the small model can drive in the meantime. + big_loader = None + if external_gpu_requested and not model_lab_active and big_model_id: + big_loader = BigModelLoader(vipc_client_main.width, vipc_client_main.height, CP, demo) + big_loader.start(big_model_id, big_model_version) params.put_bool("UsbGpuCompiled", external_artifact_ready or model_lab_ready) params.put_bool("UsbGpuActive", external_gpu_active) - params.put_bool("UsbGpuLoading", False) + params.put_bool("UsbGpuPending", False) + params.put_bool("UsbGpuLoading", big_loader is not None) _set_model_lab_runtime( params, requested=model_lab_requested, @@ -1340,6 +1435,48 @@ def main(demo=False): chestnut_state.big = False cloudlog.error(f"Model Laboratory stopped: {model_lab_error}") + # Collect a finished background big-model load, then promote it once the driver is + # disengaged. Swapping rebuilds the temporal input queues, so it must not happen + # under actuation. + if big_loader is not None and not big_loader.in_progress: + loaded_big_model, big_load_error = big_loader.take() + big_loader = None + params.put_bool("UsbGpuLoading", False) + if loaded_big_model is not None: + big_model = loaded_big_model + params.put_bool("UsbGpuPending", True) + cloudlog.warning("big model ready; waiting for the driver to disengage before using it") + else: + params.put_bool("UsbGpuPending", False) + params.put_bool("UsbGpuActive", False) + cloudlog.error(f"big model unavailable, staying on the small model: {big_load_error}") + + if big_model is not None and model is not big_model and _big_model_swap_allowed( + sm["carControl"].enabled, + sm.alive["carControl"], + sm.alive["carState"], + vipc_dropped_frames, + live_calib_seen, + ): + big_model._reset_state() + model = big_model + external_gpu_active = True + # Nothing from the small model carries over: re-arm the frame-drop warmup and drop + # the rolling probability buffers and previous action, which are model specific. + run_count = 0 + frame_dropped_filter.x = 0. + publish_state = PublishState() + prev_action = log.ModelDataV2.Action() + params.put("ModelVersion", model.policy_generation) + params.put("DrivingModelVersion", model.policy_generation) + set_runtime_model_params(params, model.model_id, model.policy_generation) + params.put_bool("UsbGpuActive", True) + params.put_bool("UsbGpuPending", False) + params.put_bool("UsbGpuLoading", False) + if chestnut_state is not None: + chestnut_state.big = True + cloudlog.warning(f"now driving on the big model {model.model_id}") + frame_drop_ratio = frames_dropped / (1 + frames_dropped) dropped_frame = vipc_dropped_frames > 0 if dropped_frame and (model.can_prepare_only or (model_lab_longitudinal is not None and model_lab_longitudinal.can_prepare_only)): @@ -1433,11 +1570,16 @@ def main(demo=False): cloudlog.exception("external GPU model failed, falling back to active small model") model = small_model big_model = None + # A failed big model is not retried in this drive, so stop any load still in flight. + if big_loader is not None: + big_loader.cancel() + big_loader = None params.put_bool("UsbGpuActive", False) external_gpu_active = False params.put("ModelVersion", model.policy_generation) params.put("DrivingModelVersion", model.policy_generation) set_runtime_model_params(params, model.model_id, model.policy_generation) + params.put_bool("UsbGpuPending", False) params.put_bool("UsbGpuLoading", False) if chestnut_state is not None: chestnut_state.big = False diff --git a/selfdrive/modeld/tests/test_usbgpu_helpers.py b/selfdrive/modeld/tests/test_usbgpu_helpers.py index 3a68f2a9f1..8d6e272ebf 100644 --- a/selfdrive/modeld/tests/test_usbgpu_helpers.py +++ b/selfdrive/modeld/tests/test_usbgpu_helpers.py @@ -1,5 +1,6 @@ import io import struct +import threading from types import MethodType from types import SimpleNamespace @@ -269,45 +270,225 @@ def test_tinygrad_empty_thread_local_cache_holder_is_safe(monkeypatch): assert tinygrad_helpers._db_connection is holder -def test_external_gpu_load_finishes_before_native_model_can_start(monkeypatch): - calls = [] - +def _stub_big_model_loader(monkeypatch, calls, *, uses_external_gpu=True, model_error=None): class FakeModelState: - uses_external_gpu = True - def __init__(self, cam_w, cam_h, external_gpu_active, model_id_override, write_model_version, model_version_override): calls.append(("model", cam_w, cam_h, external_gpu_active, model_id_override, write_model_version, model_version_override)) + if model_error is not None: + raise model_error + self.uses_external_gpu = uses_external_gpu def warmup(self): calls.append("warmup") + monkeypatch.setattr(modeld, "set_core_affinity", lambda cores: calls.append(("affinity", tuple(cores)))) monkeypatch.setattr(modeld, "wait_usbgpu_link", lambda: calls.append("link")) - monkeypatch.setattr(modeld, "wait_for_external_gpu_power_ready", lambda CP: calls.append(("power", CP))) - monkeypatch.setattr(modeld, "_set_hcq_wait_timeout", lambda timeout: calls.append(("timeout", timeout))) + monkeypatch.setattr(modeld, "wait_for_external_gpu_power_ready", + lambda CP, cancel=None: calls.append(("power", CP))) monkeypatch.setattr(modeld, "_close_tinygrad_disk_cache_connection", lambda: calls.append("close_cache")) monkeypatch.setattr(modeld, "ModelState", FakeModelState) - monkeypatch.setattr( - modeld, - "tinygrad_dev_config", - lambda *_args: (_ for _ in ()).throw(AssertionError("runtime must not change tinygrad's process-global DEV")), - ) + return FakeModelState - loaded = modeld._load_external_gpu_model(1928, 1208, "big-model", "v15", "car-params") - assert isinstance(loaded, FakeModelState) +def test_background_big_model_load_leaves_the_running_model_untouched(monkeypatch): + calls = [] + fake_model_state = _stub_big_model_loader(monkeypatch, calls) + # The small model is already driving on these process-global settings, so the background + # load must not touch tinygrad's DEV, its HCQ watchdog, or its shared buffer UOp cache. + for name, detail in ( + ("tinygrad_dev_config", "runtime must not change tinygrad's process-global DEV"), + ("_set_hcq_wait_timeout", "the background load must not move the running model's HCQ watchdog"), + ("_isolate_next_model_artifact_load", "the background load must not evict the running model's buffers"), + ): + monkeypatch.setattr(modeld, name, lambda *_args, _d=detail: (_ for _ in ()).throw(AssertionError(_d))) + + loader = modeld.BigModelLoader(1928, 1208, "car-params") + assert loader.start("big-model", "v15") + loader._thread.join(timeout=10) + + loaded, error = loader.take() + assert isinstance(loaded, fake_model_state) + assert error == "" assert calls == [ + ("affinity", tuple(sorted(modeld.BIG_MODEL_LOADER_CORES))), ("power", "car-params"), - ("timeout", modeld.BIG_MODEL_LOAD_WAIT_TIMEOUT_MS), "link", ("model", 1928, 1208, True, "big-model", False, "v15"), "warmup", "close_cache", - ("timeout", modeld.BIG_MODEL_RUN_WAIT_TIMEOUT_MS), ] +def test_background_big_model_loader_runs_off_modelds_realtime_core(): + # modeld is SCHED_FIFO on core 7 and threads inherit its affinity, so a loader left there + # would be starved behind the 20 Hz publish loop. + assert 7 not in modeld.BIG_MODEL_LOADER_CORES + assert modeld.BIG_MODEL_LOADER_CORES + + +@pytest.mark.parametrize("failure", [ + dict(model_error=RuntimeError("artifact is corrupt")), + dict(uses_external_gpu=False), +]) +def test_background_big_model_load_reports_failure_without_a_model(monkeypatch, failure): + calls = [] + _stub_big_model_loader(monkeypatch, calls, **failure) + + loader = modeld.BigModelLoader(1928, 1208, "car-params") + loader.start("big-model", "v15") + loader._thread.join(timeout=10) + + loaded, error = loader.take() + assert loaded is None + assert error + assert not loader.in_progress + # Even a failed load must release the loader thread's own tinygrad cache handle. + assert "close_cache" in calls + + +def test_background_big_model_load_is_handed_over_exactly_once(monkeypatch): + calls = [] + fake_model_state = _stub_big_model_loader(monkeypatch, calls) + + loader = modeld.BigModelLoader(1928, 1208, "car-params") + loader.start("big-model", "v15") + loader._thread.join(timeout=10) + + assert isinstance(loader.take()[0], fake_model_state) + # A second take must not promote the same model again. + assert loader.take() == (None, "") + + +def test_cancelled_big_model_load_never_builds_a_model(monkeypatch): + calls = [] + + def cancelling_power_wait(CP, cancel=None): + calls.append(("power", CP)) + cancel.set() + raise modeld.BigModelLoadCancelled("cancelled while waiting for external GPU power") + + _stub_big_model_loader(monkeypatch, calls) + monkeypatch.setattr(modeld, "wait_for_external_gpu_power_ready", cancelling_power_wait) + + loader = modeld.BigModelLoader(1928, 1208, "car-params") + loader.start("big-model", "v15") + loader._thread.join(timeout=10) + + loaded, error = loader.take() + assert loaded is None + assert error + assert not any(call[0] == "model" for call in calls if isinstance(call, tuple)) + + +def test_external_gpu_power_wait_aborts_when_the_loader_is_cancelled(monkeypatch): + cancel = threading.Event() + cancel.set() + monkeypatch.setattr(modeld, "SubMaster", + lambda *_a, **_k: pytest.fail("a cancelled load must not start waiting on power")) + + with pytest.raises(modeld.BigModelLoadCancelled): + modeld.wait_for_external_gpu_power_ready("car-params", cancel=cancel) + + +def test_big_model_promotion_waits_for_the_driver_to_disengage(monkeypatch): + """Drive the real collect-and-promote blocks lifted out of main()'s loop.""" + import ast + from pathlib import Path + + source = (Path(modeld.__file__).with_name("modeld.py")).read_text(encoding="utf-8") + main_fn = next(n for n in ast.parse(source).body if isinstance(n, ast.FunctionDef) and n.name == "main") + def names(node): + return {n.id for n in ast.walk(node) if isinstance(n, ast.Name)} + + collect_block = next(n for n in ast.walk(main_fn) + if isinstance(n, ast.If) and "big_loader" in names(n.test) + and "in_progress" in ast.dump(n.test)) + promote_block = next(n for n in ast.walk(main_fn) + if isinstance(n, ast.If) and "_big_model_swap_allowed" in names(n.test)) + promote = compile(ast.Module(body=[collect_block, promote_block], type_ignores=[]), + "", "exec") + + class FakeModel: + model_id = "big-model" + policy_generation = "v15" + + def __init__(self): + self.reset = 0 + + def _reset_state(self): + self.reset += 1 + + class FakeLoader: + def __init__(self, model): + self.in_progress = True + self._model = model + + def take(self): + return self._model, "" + + written = {} + big = FakeModel() + loader = FakeLoader(big) + small = FakeModel() + small.model_id = "small-model" + scope = { + "big_loader": loader, "big_model": None, "model": small, "small_model": small, + "params": SimpleNamespace(put_bool=lambda k, v: written.__setitem__(k, v), + put=lambda k, v: written.__setitem__(k, v)), + "cloudlog": SimpleNamespace(warning=lambda *_a: None, error=lambda *_a: None), + "vipc_dropped_frames": 0, "live_calib_seen": True, "external_gpu_active": False, + "run_count": 99, "frame_dropped_filter": SimpleNamespace(x=5.0), + "PublishState": lambda: "fresh", "publish_state": "stale", + "prev_action": "stale", "log": modeld.log, "chestnut_state": SimpleNamespace(big=False), + "set_runtime_model_params": lambda *_a: None, + "_big_model_swap_allowed": modeld._big_model_swap_allowed, + } + scope["sm"] = type("SM", (), { + "__getitem__": lambda _s, k: SimpleNamespace(enabled=scope["_engaged"]), + "alive": {"carControl": True, "carState": True}, + })() + + # Still loading: nothing is collected and the small model keeps driving. + scope["_engaged"] = True + exec(promote, scope) # noqa: S102 + assert scope["model"] is small and scope["big_model"] is None + + # Load finishes while engaged: the model is collected but must NOT be promoted. + loader.in_progress = False + exec(promote, scope) # noqa: S102 + assert scope["big_model"] is big + assert scope["model"] is small, "must not swap models while the driver is engaged" + assert written["UsbGpuPending"] is True + assert written["UsbGpuLoading"] is False + assert big.reset == 0 + + # The driver disengages: now the big model takes over. + scope["_engaged"] = False + exec(promote, scope) # noqa: S102 + assert scope["model"] is big + assert big.reset == 1, "temporal queues must be reset before the big model drives" + assert scope["run_count"] == 0 and scope["frame_dropped_filter"].x == 0. + assert scope["publish_state"] == "fresh" and scope["prev_action"] != "stale" + assert written["UsbGpuActive"] is True and written["UsbGpuPending"] is False + assert scope["chestnut_state"].big is True + + +@pytest.mark.parametrize("kwargs,expected", [ + (dict(), True), + (dict(engaged=True), False), + (dict(carcontrol_alive=False), False), + (dict(carstate_alive=False), False), + (dict(vipc_dropped_frames=1), False), + (dict(live_calib_seen=False), False), +]) +def test_big_model_swap_only_while_disengaged_with_fresh_state(kwargs, expected): + defaults = dict(engaged=False, carcontrol_alive=True, carstate_alive=True, + vipc_dropped_frames=0, live_calib_seen=True) + assert modeld._big_model_swap_allowed(**(defaults | kwargs)) is expected + + def test_external_gpu_nonfinite_outputs_trigger_fallback(monkeypatch): class FakeTensor: @staticmethod diff --git a/selfdrive/selfdrived/events.py b/selfdrive/selfdrived/events.py index cb5f4c64b1..ecac0c5e91 100644 --- a/selfdrive/selfdrived/events.py +++ b/selfdrive/selfdrived/events.py @@ -533,7 +533,14 @@ EVENTS: dict[int, dict[str, Alert | AlertCallbackType]] = { }, EventName.bigModelLoading: { - ET.NO_ENTRY: NoEntryAlert("Big Model Loading"), + ET.PERMANENT: NormalPermanentAlert("Big Model Loading", + "Driving on the small model"), + }, + + EventName.bigModelPending: { + ET.PERMANENT: NormalPermanentAlert("Big Model Ready", + "Disengage and re-engage to use it", + duration=10.), }, EventName.bigModelFailed: { diff --git a/selfdrive/selfdrived/selfdrived.py b/selfdrive/selfdrived/selfdrived.py index aea14c1a93..bcec82270d 100644 --- a/selfdrive/selfdrived/selfdrived.py +++ b/selfdrive/selfdrived/selfdrived.py @@ -255,7 +255,7 @@ class SelfdriveD: self.big_model_attempted = False self.big_model_active = False self.big_model_failed = False - self.big_model_ready_t = 0. + self.big_model_swap_t = 0. self.experimental_mode = False self.ecu_disable_failed = False self.ecu_disable_failed_checked = not ( @@ -394,15 +394,23 @@ class SelfdriveD: loading = self.params.get_bool("UsbGpuLoading") if loading: self.big_model_attempted = True - if self.big_model_loading and not loading: - self.big_model_ready_t = time.monotonic() self.big_model_loading = loading if loading: self.events.add(EventName.bigModelLoading) + # The big model loads in the background while the small model drives, so it sits + # loaded-but-unused until the driver disengages. That is a success, not a failure. + pending = self.params.get_bool("UsbGpuPending") + if pending: + self.big_model_attempted = True + self.events.add(EventName.bigModelPending) + big_active = self.params.get("UsbGpuActive") + if self.big_model_active != (big_active is True): + self.big_model_swap_t = time.monotonic() model_unavailable = self.big_model_active and self.sm.seen['modelV2'] and not self.sm.alive['modelV2'] - big_failed = self.big_model_attempted and not loading and (big_active is False or model_unavailable) + big_failed = self.big_model_attempted and not loading and not pending and \ + (big_active is False or model_unavailable) if big_failed and not self.big_model_failed: self.events.add(EventName.bigModelFailed) self.big_model_failed = big_failed @@ -697,7 +705,10 @@ class SelfdriveD: (contains_event_type(self.events, self.starpilot_events, ET.SOFT_DISABLE) or contains_event_type(self.events, self.starpilot_events, ET.IMMEDIATE_DISABLE)) no_system_errors = (not has_disable_events) or (len(self.events) == num_events) - big_model_settling = self.big_model_loading or time.monotonic() < self.big_model_ready_t + 5. + # modeld publishes on the small model throughout the background load, so only the model + # swap itself can briefly disturb the stream. Suppressing for the whole load would hide + # a genuinely dead modeld for as long as the load takes. + big_model_settling = time.monotonic() < self.big_model_swap_t + 2. all_checks = self.sm.all_checks() all_alive = self.sm.all_alive() if not all_checks else True all_freq_ok = self.sm.all_freq_ok() if not all_checks else True diff --git a/selfdrive/selfdrived/tests/test_big_model_engagement.py b/selfdrive/selfdrived/tests/test_big_model_engagement.py new file mode 100644 index 0000000000..d22039f1b2 --- /dev/null +++ b/selfdrive/selfdrived/tests/test_big_model_engagement.py @@ -0,0 +1,60 @@ +import ast +from pathlib import Path + +import pytest + +from cereal import log +from openpilot.selfdrive.selfdrived.events import EVENTS, ET + +EventName = log.OnroadEvent.EventName + + +def test_big_model_loading_does_not_block_engagement(): + # The big model loads in the background while the small model drives, so waiting for it + # must never keep the driver from engaging. + assert ET.NO_ENTRY not in EVENTS[EventName.bigModelLoading] + assert ET.PERMANENT in EVENTS[EventName.bigModelLoading] + + +def test_big_model_pending_is_advisory_only(): + pending = EVENTS[EventName.bigModelPending] + assert set(pending) == {ET.PERMANENT} + + +def test_big_model_failure_still_disengages(): + assert ET.SOFT_DISABLE in EVENTS[EventName.bigModelFailed] + + +def _big_failed(*, attempted, loading, pending, big_active, model_unavailable=False): + """Mirror of selfdrived's big_failed expression, extracted from the source.""" + source = (Path(__file__).parents[1] / "selfdrived.py").read_text(encoding="utf-8") + tree = ast.parse(source) + assign = next( + node for node in ast.walk(tree) + if isinstance(node, ast.Assign) + and any(isinstance(t, ast.Name) and t.id == "big_failed" for t in node.targets) + ) + scope = { + "self": type("S", (), {"big_model_attempted": attempted})(), + "loading": loading, + "pending": pending, + "big_active": big_active, + "model_unavailable": model_unavailable, + } + return eval(compile(ast.Expression(assign.value), "", "eval"), {}, scope) # noqa: S307 + + +@pytest.mark.parametrize("state,expected", [ + # A loaded model waiting for the driver to disengage is a success, not a failure. + (dict(attempted=True, loading=False, pending=True, big_active=False), False), + (dict(attempted=True, loading=True, pending=False, big_active=False), False), + (dict(attempted=True, loading=False, pending=False, big_active=True), False), + # Genuine failures must still be reported. + (dict(attempted=True, loading=False, pending=False, big_active=False), True), + (dict(attempted=True, loading=False, pending=False, big_active=True, + model_unavailable=True), True), + # Never report a failure for a big model that was never attempted. + (dict(attempted=False, loading=False, pending=False, big_active=False), False), +]) +def test_pending_big_model_is_not_reported_as_failed(state, expected): + assert _big_failed(**state) is expected diff --git a/selfdrive/ui/onroad/starpilot/widgets/model_source.py b/selfdrive/ui/onroad/starpilot/widgets/model_source.py index b610715e81..7b25208e41 100644 --- a/selfdrive/ui/onroad/starpilot/widgets/model_source.py +++ b/selfdrive/ui/onroad/starpilot/widgets/model_source.py @@ -62,12 +62,17 @@ class ModelSourceWidget(LayoutWidget): return self.SIZE @staticmethod - def _big_model_failed(active: bool | None, usbgpu: bool, model_seen: bool, model_alive: bool) -> bool: + def _big_model_failed(active: bool | None, usbgpu: bool, model_seen: bool, model_alive: bool, + pending: bool = False) -> bool: + # A model that finished loading but is waiting for the driver to disengage is not a failure. + if pending and usbgpu: + return False return active is False or not usbgpu or (active is True and model_seen and not model_alive) @staticmethod - def _status_for(loading: bool, small_model_engaged: bool, big_failed: bool) -> ModelSourceStatus: - if loading: + def _status_for(loading: bool, small_model_engaged: bool, big_failed: bool, + pending: bool = False) -> ModelSourceStatus: + if loading or pending: return ModelSourceStatus.LOADING if small_model_engaged: return ModelSourceStatus.FALLBACK_ENGAGED @@ -84,16 +89,18 @@ class ModelSourceWidget(LayoutWidget): model_seen = sm.recv_frame["modelV2"] > ui_state.started_frame model_alive = sm.alive["modelV2"] if model_seen else True loading = ui_state.usbgpu_loading - big_failed = self._big_model_failed(ui_state.usbgpu_active, ui_state.usbgpu, model_seen, model_alive) + pending = ui_state.usbgpu_pending + big_failed = self._big_model_failed(ui_state.usbgpu_active, ui_state.usbgpu, model_seen, model_alive, pending) engaged = sm["selfdriveState"].enabled - if engaged and not self._engaged and not loading and ui_state.usbgpu_active is not True and model_seen: + if engaged and not self._engaged and not loading and not pending \ + and ui_state.usbgpu_active is not True and model_seen: self._small_model_engaged = True if engaged != self._engaged: self._fade_time = rl.get_time() if engaged else 0.0 self._engaged = engaged self._small_model_engaged &= big_failed - self._status = self._status_for(loading, self._small_model_engaged, big_failed) + self._status = self._status_for(loading, self._small_model_engaged, big_failed, pending) def _render(self, rect: rl.Rectangle) -> None: if self._status is None: diff --git a/selfdrive/ui/soundd.py b/selfdrive/ui/soundd.py index e2ede4fdf4..7bc5c6a82d 100644 --- a/selfdrive/ui/soundd.py +++ b/selfdrive/ui/soundd.py @@ -344,8 +344,11 @@ class Soundd: if now - self.model_ready_last_check >= 0.25: self.model_ready_last_check = now try: + # The big model now loads in the background, so it is announced when it becomes + # usable (pending the driver's next disengage) rather than only once it is running. self.model_ready_pending = self.model_ready_chime.update( - active=self.model_ready_params.get_bool("UsbGpuActive"), + active=(self.model_ready_params.get_bool("UsbGpuPending") or + self.model_ready_params.get_bool("UsbGpuActive")), loading=self.model_ready_params.get_bool("UsbGpuLoading"), onroad=self.model_ready_params.get_bool("IsOnroad"), enabled=self.model_ready_params.get_bool("GpuModelReadySound"), now=now) diff --git a/selfdrive/ui/tests/test_model_source_widget.py b/selfdrive/ui/tests/test_model_source_widget.py index f3d1848f5e..9aa83c0941 100644 --- a/selfdrive/ui/tests/test_model_source_widget.py +++ b/selfdrive/ui/tests/test_model_source_widget.py @@ -32,6 +32,18 @@ def test_model_source_failure_detection_matches_the_backend_state_contract(): assert not failed(True, True, False, True) +def test_model_source_shows_a_pending_big_model_as_still_loading(): + # The big model loads in the background, so "loaded, waiting for a disengage" must read + # as in-progress rather than as a failure. + status = model_source.ModelSourceWidget._status_for + failed = model_source.ModelSourceWidget._big_model_failed + + assert status(False, False, False, True) is model_source.ModelSourceStatus.LOADING + assert not failed(False, True, False, True, True) + # Chestnut going away while pending is still a genuine failure. + assert failed(False, False, False, True, True) + + def test_model_source_latches_small_model_engagement_until_the_big_model_recovers(monkeypatch): widget = object.__new__(model_source.ModelSourceWidget) widget._small_model_engaged = False @@ -50,6 +62,7 @@ def test_model_source_latches_small_model_engagement_until_the_big_model_recover usbgpu_compiled=True, usbgpu_active=False, usbgpu_loading=False, + usbgpu_pending=False, ), ) monkeypatch.setattr(model_source.rl, "get_time", lambda: 42.0) diff --git a/selfdrive/ui/ui_state.py b/selfdrive/ui/ui_state.py index 6fc17569d5..75b5f8459e 100644 --- a/selfdrive/ui/ui_state.py +++ b/selfdrive/ui/ui_state.py @@ -92,6 +92,7 @@ class UIState: self.usbgpu_compiled: bool = self.params.get_bool("UsbGpuCompiled") self.usbgpu_active: bool = self.params.get_bool("UsbGpuActive") self.usbgpu_loading: bool = self.params.get_bool("UsbGpuLoading") + self.usbgpu_pending: bool = self.params.get_bool("UsbGpuPending") self.started: bool = False self.ignition: bool = False self.recording_audio: bool = False @@ -215,6 +216,7 @@ class UIState: self.usbgpu_compiled = params.get_bool("UsbGpuCompiled") self.usbgpu_active = params.get_bool("UsbGpuActive") self.usbgpu_loading = params.get_bool("UsbGpuLoading") + self.usbgpu_pending = params.get_bool("UsbGpuPending") self.switchback_mode_enabled = self.params_memory.get_bool("SwitchbackModeEnabled") if self.started else False self.conditional_status = self.params_memory.get_int("CEStatus", default=0) if self.started else 0 mark_progress("ui.update.after_state_params")