diff --git a/cereal/log.capnp b/cereal/log.capnp index 8179fa5275..0a6210bf75 100644 --- a/cereal/log.capnp +++ b/cereal/log.capnp @@ -132,6 +132,7 @@ struct OnroadEvent @0xc4fa6047f024e718 { audioFeedback @97; bigModelLoading @100; bigModelFailed @102; + bigModelPending @103; soundsUnavailableDEPRECATED @47; stockLkasDEPRECATED @98; diff --git a/common/params_keys.h b/common/params_keys.h index d94d7f22d8..957058f5dc 100644 --- a/common/params_keys.h +++ b/common/params_keys.h @@ -177,6 +177,7 @@ inline static std::unordered_map keys = { {"UsbGpuActive", {CLEAR_ON_MANAGER_START | CLEAR_ON_OFFROAD_TRANSITION, BOOL}}, {"UsbGpuCompiled", {CLEAR_ON_MANAGER_START | CLEAR_ON_OFFROAD_TRANSITION, BOOL}}, {"UsbGpuLoading", {CLEAR_ON_MANAGER_START | CLEAR_ON_OFFROAD_TRANSITION, BOOL}}, + {"UsbGpuPending", {CLEAR_ON_MANAGER_START | CLEAR_ON_OFFROAD_TRANSITION, BOOL}}, {"UsbGpuPresent", {CLEAR_ON_MANAGER_START | CLEAR_ON_OFFROAD_TRANSITION, BOOL}}, {"Version", {PERSISTENT, STRING}}, diff --git a/selfdrive/modeld/modeld.py b/selfdrive/modeld/modeld.py index 797b29ea25..1be68efb10 100755 --- a/selfdrive/modeld/modeld.py +++ b/selfdrive/modeld/modeld.py @@ -5,6 +5,7 @@ from functools import cached_property import json import os import struct +import threading import usb1 from openpilot.system.hardware import HARDWARE, TICI os.environ['GMMU'] = '0' @@ -25,7 +26,7 @@ from openpilot.common.swaglog import cloudlog from openpilot.common.params import Params from openpilot.common.filter_simple import FirstOrderFilter from openpilot.common.file_chunker import file_chunked_exists, open_file_chunked -from openpilot.common.realtime import config_realtime_process, DT_MDL +from openpilot.common.realtime import config_realtime_process, set_core_affinity, DT_MDL from openpilot.common.transformations.camera import DEVICE_CAMERAS from openpilot.common.transformations.model import get_warp_matrix from openpilot.system.camerad.cameras.nv12_info import get_nv12_info @@ -105,6 +106,23 @@ EXTERNAL_GPU_POWER_WAIT_TIMEOUT_SECONDS = 60.0 EXTERNAL_GPU_POWER_LOG_INTERVAL_SECONDS = 10.0 LAT_SMOOTH_BP = [2.0, 8.0] +# The background big-model loader must leave modeld's SCHED_FIFO core, because thread +# affinity is inherited from config_realtime_process(7, 54) and the loader would otherwise +# sit behind the 20 Hz publish loop. Every other core is spoken for as well -- 4 is +# card/controlsd, 5 is plannerd/radard/selfdrived, 6 is camerad (system/camerad/main.cc), +# 7 is modeld and dmonitoringmodeld -- so the loader floats across the little cores, which +# run the non-realtime locationd/UI work, instead of pinning on top of one critical process. +BIG_MODEL_LOADER_CORES = {0, 1, 2, 3} + +# A healthy load takes ~30 s once vehicle power is stable. If the AMD device never comes up +# the tinygrad calls can block indefinitely, so give up rather than reporting "loading" +# forever while the small model quietly keeps driving. +BIG_MODEL_LOAD_TIMEOUT_SECONDS = 150.0 + + +class BigModelLoadCancelled(Exception): + """Raised inside the background loader when modeld no longer wants the big model.""" + def _set_hcq_wait_timeout(timeout_ms: int) -> None: """Update tinygrad's cached HCQ timeout for the external-GPU load/run phase.""" @@ -155,8 +173,10 @@ def _egmp_vehicle_ready(can_messages, bus: int) -> bool: ) -def wait_for_external_gpu_power_ready(CP=None) -> None: +def wait_for_external_gpu_power_ready(CP=None, cancel=None) -> None: """Wait out vehicle startup power transitions before initializing Chestnut.""" + if cancel is not None and cancel.is_set(): + raise BigModelLoadCancelled("cancelled before waiting for external GPU power") device_type = HARDWARE.get_device_type() egmp_bus = _egmp_ready_bus(CP) services = ["pandaStates", "peripheralState"] + (["can"] if egmp_bus is not None else []) @@ -167,6 +187,8 @@ def wait_for_external_gpu_power_ready(CP=None) -> None: wait_started = time.monotonic() while True: + if cancel is not None and cancel.is_set(): + raise BigModelLoadCancelled("cancelled while waiting for external GPU power") sm.update(1000) now = time.monotonic() if egmp_bus is not None and sm.updated["can"] and _egmp_vehicle_ready(sm["can"], egmp_bus): @@ -854,33 +876,123 @@ def _load_model_state(cam_w: int, cam_h: int, selected_model: str, external_gpu_ ) -def _load_external_gpu_model(cam_w: int, cam_h: int, selected_model: str, model_version: str = "", - CP=None, demo: bool = False) -> ModelState | None: - """Load and warm the USB-GPU model without running another tinygrad model concurrently.""" - candidate = None - try: - if not demo: - wait_for_external_gpu_power_ready(CP) - _set_hcq_wait_timeout(BIG_MODEL_LOAD_WAIT_TIMEOUT_MS) - wait_usbgpu_link() - candidate = ModelState( - cam_w, - cam_h, - True, - model_id_override=selected_model, - write_model_version=False, - model_version_override=model_version, - ) - if not candidate.uses_external_gpu: - raise RuntimeError("external GPU model resolved to the builtin model") - candidate.warmup() - return candidate - except Exception: - cloudlog.exception("external GPU model load or warmup failed") - return None - finally: - _close_tinygrad_disk_cache_connection() - _set_hcq_wait_timeout(BIG_MODEL_RUN_WAIT_TIMEOUT_MS) +def _big_model_swap_allowed(engaged: bool, carcontrol_alive: bool, carstate_alive: bool, + vipc_dropped_frames: int, live_calib_seen: bool) -> bool: + """Swapping models rebuilds the temporal input queues, so it must never happen under actuation. + + Fails closed: without fresh carControl/carState we cannot trust that we are disengaged. + """ + return ( + not engaged + and carcontrol_alive + and carstate_alive + and vipc_dropped_frames == 0 + and live_calib_seen + ) + + +class BigModelLoader: + """Load and warm the Chestnut model on a background thread while the small model drives. + + Everything this touches must be safe against a concurrently running small model, so unlike + the synchronous path it never calls _set_hcq_wait_timeout (hoisted to modeld startup) or + _isolate_next_model_artifact_load (which evicts buffer UOps from a tinygrad global cache). + """ + + def __init__(self, cam_w: int, cam_h: int, CP=None, demo: bool = False): + self.cam_w = cam_w + self.cam_h = cam_h + self.CP = CP + self.demo = demo + self._thread: threading.Thread | None = None + self._cancel = threading.Event() + self._lock = threading.Lock() + self._result: ModelState | None = None + self._error = "" + self._model_id = "" + self._model_version = "" + self.started_t = 0.0 + + @property + def in_progress(self) -> bool: + return self._thread is not None and self._thread.is_alive() + + @property + def timed_out(self) -> bool: + """A load stuck inside tinygrad cannot be interrupted, so the caller gives up on it. + + The thread is a daemon and keeps running, but nothing will consume its result. + """ + return self.in_progress and time.monotonic() - self.started_t > BIG_MODEL_LOAD_TIMEOUT_SECONDS + + def start(self, model_id: str, model_version: str = "") -> bool: + if self.in_progress: + return False + with self._lock: + self._result = None + self._error = "" + self._model_id = model_id + self._model_version = model_version + self._cancel.clear() + self.started_t = time.monotonic() + self._thread = threading.Thread(target=self._run, name="big_model_loader", daemon=True) + self._thread.start() + return True + + def cancel(self) -> None: + self._cancel.set() + + def take(self) -> tuple[ModelState | None, str]: + """Hand over the finished load exactly once: (model, error). Both empty while in flight.""" + if self._thread is None or self._thread.is_alive(): + return None, "" + self._thread = None + with self._lock: + result, error = self._result, self._error + self._result = None + self._error = "" + return result, error + + def _run(self) -> None: + try: + # Leave modeld's realtime core before doing any work; see BIG_MODEL_LOADER_CORES. + set_core_affinity(sorted(BIG_MODEL_LOADER_CORES)) + if not self.demo: + wait_for_external_gpu_power_ready(self.CP, cancel=self._cancel) + if self._cancel.is_set(): + raise BigModelLoadCancelled("cancelled before the external GPU link check") + wait_usbgpu_link() + candidate = ModelState( + self.cam_w, + self.cam_h, + True, + model_id_override=self._model_id, + write_model_version=False, + model_version_override=self._model_version, + ) + if not candidate.uses_external_gpu: + raise RuntimeError("external GPU model resolved to the builtin model") + # warmup() runs the camera warp on QCOM, shared with the driving small model, so it + # costs some jitter here. It still belongs on this thread: doing it at promotion + # instead blocks the publish loop for ~13 s in one stretch, which trips commIssue and + # blocks the driver from engaging exactly when the model becomes available. + candidate.warmup() + if self._cancel.is_set(): + raise BigModelLoadCancelled("cancelled after warmup") + with self._lock: + self._result = candidate + cloudlog.warning(f"background big model load finished in {time.monotonic() - self.started_t:.1f}s") + except BigModelLoadCancelled as exc: + cloudlog.warning(f"background big model load cancelled: {exc}") + with self._lock: + self._error = str(exc) + except Exception as exc: + cloudlog.exception("background big model load failed") + with self._lock: + self._error = str(exc) or exc.__class__.__name__ + finally: + # tinygrad's disk cache handle is thread-local, so this closes only the loader's. + _close_tinygrad_disk_cache_connection() def _model_versions() -> dict[str, str]: @@ -1058,9 +1170,16 @@ def main(demo=False): external_artifact = MODELS_PATH / f"{big_model_id}_driving_tinygrad.pkl" external_artifact_ready = external_model_selected and file_chunked_exists(external_artifact) external_gpu_requested = usbgpu_present_now and (bool(big_model_id) or model_lab_ready) + # The big model is loaded in the background while the small model drives, so the HCQ + # watchdog is set once here and never mutated again: it is process-global and shared by + # the QCOM and AMD devices, so changing it mid-drive would also move the small model's. + if external_gpu_requested: + _set_hcq_wait_timeout(BIG_MODEL_LOAD_WAIT_TIMEOUT_MS) + params.put_bool("UsbGpuPresent", usbgpu_present_now) params.put_bool("UsbGpuCompiled", external_artifact_ready or model_lab_ready) params.put_bool("UsbGpuActive", False) + params.put_bool("UsbGpuPending", False) params.put_bool("UsbGpuLoading", external_gpu_requested) _set_model_lab_runtime( params, @@ -1147,15 +1266,8 @@ def main(demo=False): else: CP = messaging.log_from_bytes(params.get("CarParams", block=True), car.CarParams) - big_model = _load_external_gpu_model( - vipc_client_main.width, - vipc_client_main.height, - selected_model, - selected_model_version, - CP, - demo, - ) - + # Start on the small model so openpilot is drivable immediately; the big model is loaded + # in the background below and promoted once the driver disengages. small_model = _load_model_state( vipc_client_main.width, vipc_client_main.height, @@ -1165,10 +1277,7 @@ def main(demo=False): small_model_version, False, ) - model = big_model if big_model is not None else small_model - if big_model is not None: - params.put("ModelVersion", model.policy_generation) - params.put("DrivingModelVersion", model.policy_generation) + model = small_model else: model = _load_model_state( vipc_client_main.width, @@ -1183,9 +1292,15 @@ def main(demo=False): set_runtime_model_params(params, model.model_id, model.policy_generation) external_gpu_active = model_lab_active or model.uses_external_gpu + # Load the big model in the background so the small model can drive in the meantime. + big_loader = None + if external_gpu_requested and not model_lab_active and big_model_id: + big_loader = BigModelLoader(vipc_client_main.width, vipc_client_main.height, CP, demo) + big_loader.start(big_model_id, big_model_version) params.put_bool("UsbGpuCompiled", external_artifact_ready or model_lab_ready) params.put_bool("UsbGpuActive", external_gpu_active) - params.put_bool("UsbGpuLoading", False) + params.put_bool("UsbGpuPending", False) + params.put_bool("UsbGpuLoading", big_loader is not None) _set_model_lab_runtime( params, requested=model_lab_requested, @@ -1340,6 +1455,61 @@ def main(demo=False): chestnut_state.big = False cloudlog.error(f"Model Laboratory stopped: {model_lab_error}") + # Collect a finished background big-model load, then promote it once the driver is + # disengaged. Swapping rebuilds the temporal input queues, so it must not happen + # under actuation. + if big_loader is not None and big_loader.timed_out: + big_loader.cancel() + big_loader = None + params.put_bool("UsbGpuLoading", False) + params.put_bool("UsbGpuPending", False) + params.put_bool("UsbGpuActive", False) + cloudlog.error(f"big model load exceeded {BIG_MODEL_LOAD_TIMEOUT_SECONDS:.0f}s, " + "staying on the small model") + elif big_loader is not None and not big_loader.in_progress: + loaded_big_model, big_load_error = big_loader.take() + big_loader = None + if loaded_big_model is not None: + big_model = loaded_big_model + # Raise pending before clearing loading: selfdrived polls these separately at 100 Hz + # and reads "neither loading nor pending" as a failed load. + params.put_bool("UsbGpuPending", True) + params.put_bool("UsbGpuLoading", False) + cloudlog.warning("big model ready; waiting for the driver to disengage before using it") + else: + params.put_bool("UsbGpuPending", False) + params.put_bool("UsbGpuActive", False) + params.put_bool("UsbGpuLoading", False) + cloudlog.error(f"big model unavailable, staying on the small model: {big_load_error}") + + if big_model is not None and model is not big_model and _big_model_swap_allowed( + sm["carControl"].enabled, + sm.alive["carControl"], + sm.alive["carState"], + vipc_dropped_frames, + live_calib_seen, + ): + # The loader already warmed this model, so the swap itself is just a pointer change + # plus a queue reset; it must stay cheap because it runs inside the publish loop. + big_model._reset_state() + model = big_model + external_gpu_active = True + # Nothing from the small model carries over: re-arm the frame-drop warmup and drop + # the rolling probability buffers and previous action, which are model specific. + run_count = 0 + frame_dropped_filter.x = 0. + publish_state = PublishState() + prev_action = log.ModelDataV2.Action() + params.put("ModelVersion", model.policy_generation) + params.put("DrivingModelVersion", model.policy_generation) + set_runtime_model_params(params, model.model_id, model.policy_generation) + params.put_bool("UsbGpuActive", True) + params.put_bool("UsbGpuPending", False) + params.put_bool("UsbGpuLoading", False) + if chestnut_state is not None: + chestnut_state.big = True + cloudlog.warning(f"now driving on the big model {model.model_id}") + frame_drop_ratio = frames_dropped / (1 + frames_dropped) dropped_frame = vipc_dropped_frames > 0 if dropped_frame and (model.can_prepare_only or (model_lab_longitudinal is not None and model_lab_longitudinal.can_prepare_only)): @@ -1357,6 +1527,9 @@ def main(demo=False): try: send_chestnut = ( chestnut_state is not None and + # Telemetry shares the USB device with the model transfer, so polling it while a + # background load is in flight stalls that transfer until it times out. + (big_loader is None or not big_loader.in_progress) and run_count % round(ModelConstants.MODEL_FREQ / SERVICE_LIST["chestnutState"].frequency) == 0 ) if model_lab_longitudinal is not None: @@ -1433,11 +1606,16 @@ def main(demo=False): cloudlog.exception("external GPU model failed, falling back to active small model") model = small_model big_model = None + # A failed big model is not retried in this drive, so stop any load still in flight. + if big_loader is not None: + big_loader.cancel() + big_loader = None params.put_bool("UsbGpuActive", False) external_gpu_active = False params.put("ModelVersion", model.policy_generation) params.put("DrivingModelVersion", model.policy_generation) set_runtime_model_params(params, model.model_id, model.policy_generation) + params.put_bool("UsbGpuPending", False) params.put_bool("UsbGpuLoading", False) if chestnut_state is not None: chestnut_state.big = False diff --git a/selfdrive/modeld/tests/test_usbgpu_helpers.py b/selfdrive/modeld/tests/test_usbgpu_helpers.py index 3a68f2a9f1..86046ca2c4 100644 --- a/selfdrive/modeld/tests/test_usbgpu_helpers.py +++ b/selfdrive/modeld/tests/test_usbgpu_helpers.py @@ -1,5 +1,6 @@ import io import struct +import threading from types import MethodType from types import SimpleNamespace @@ -269,45 +270,349 @@ def test_tinygrad_empty_thread_local_cache_holder_is_safe(monkeypatch): assert tinygrad_helpers._db_connection is holder -def test_external_gpu_load_finishes_before_native_model_can_start(monkeypatch): - calls = [] - +def _stub_big_model_loader(monkeypatch, calls, *, uses_external_gpu=True, model_error=None): class FakeModelState: - uses_external_gpu = True - def __init__(self, cam_w, cam_h, external_gpu_active, model_id_override, write_model_version, model_version_override): calls.append(("model", cam_w, cam_h, external_gpu_active, model_id_override, write_model_version, model_version_override)) + if model_error is not None: + raise model_error + self.uses_external_gpu = uses_external_gpu def warmup(self): calls.append("warmup") + monkeypatch.setattr(modeld, "set_core_affinity", lambda cores: calls.append(("affinity", tuple(cores)))) monkeypatch.setattr(modeld, "wait_usbgpu_link", lambda: calls.append("link")) - monkeypatch.setattr(modeld, "wait_for_external_gpu_power_ready", lambda CP: calls.append(("power", CP))) - monkeypatch.setattr(modeld, "_set_hcq_wait_timeout", lambda timeout: calls.append(("timeout", timeout))) + monkeypatch.setattr(modeld, "wait_for_external_gpu_power_ready", + lambda CP, cancel=None: calls.append(("power", CP))) monkeypatch.setattr(modeld, "_close_tinygrad_disk_cache_connection", lambda: calls.append("close_cache")) monkeypatch.setattr(modeld, "ModelState", FakeModelState) - monkeypatch.setattr( - modeld, - "tinygrad_dev_config", - lambda *_args: (_ for _ in ()).throw(AssertionError("runtime must not change tinygrad's process-global DEV")), - ) + return FakeModelState - loaded = modeld._load_external_gpu_model(1928, 1208, "big-model", "v15", "car-params") - assert isinstance(loaded, FakeModelState) +def test_background_big_model_load_leaves_the_running_model_untouched(monkeypatch): + calls = [] + fake_model_state = _stub_big_model_loader(monkeypatch, calls) + # The small model is already driving on these process-global settings, so the background + # load must not touch tinygrad's DEV, its HCQ watchdog, or its shared buffer UOp cache. + for name, detail in ( + ("tinygrad_dev_config", "runtime must not change tinygrad's process-global DEV"), + ("_set_hcq_wait_timeout", "the background load must not move the running model's HCQ watchdog"), + ("_isolate_next_model_artifact_load", "the background load must not evict the running model's buffers"), + ): + monkeypatch.setattr(modeld, name, lambda *_args, _d=detail: (_ for _ in ()).throw(AssertionError(_d))) + + loader = modeld.BigModelLoader(1928, 1208, "car-params") + assert loader.start("big-model", "v15") + loader._thread.join(timeout=10) + + loaded, error = loader.take() + assert isinstance(loaded, fake_model_state) + assert error == "" + # warmup() belongs here rather than at promotion: see + # test_big_model_warmup_stays_on_the_background_loader. assert calls == [ + ("affinity", tuple(sorted(modeld.BIG_MODEL_LOADER_CORES))), ("power", "car-params"), - ("timeout", modeld.BIG_MODEL_LOAD_WAIT_TIMEOUT_MS), "link", ("model", 1928, 1208, True, "big-model", False, "v15"), "warmup", "close_cache", - ("timeout", modeld.BIG_MODEL_RUN_WAIT_TIMEOUT_MS), ] +def test_big_model_load_gives_up_instead_of_loading_forever(monkeypatch): + # A load stuck inside tinygrad cannot be interrupted, so the timeout is what stops modeld + # reporting "loading" for the rest of the drive while the small model quietly drives. + release = threading.Event() + calls = [] + + class HangingModelState: + uses_external_gpu = True + + def __init__(self, *_a, **_k): + release.wait(timeout=10) + + def warmup(self): + pass + + monkeypatch.setattr(modeld, "set_core_affinity", lambda cores: None) + monkeypatch.setattr(modeld, "wait_usbgpu_link", lambda: None) + monkeypatch.setattr(modeld, "wait_for_external_gpu_power_ready", lambda CP, cancel=None: None) + monkeypatch.setattr(modeld, "_close_tinygrad_disk_cache_connection", lambda: calls.append("close")) + monkeypatch.setattr(modeld, "ModelState", HangingModelState) + + loader = modeld.BigModelLoader(1928, 1208, "car-params") + loader.start("big-model", "v15") + try: + assert loader.in_progress + assert not loader.timed_out + + # Pretend the load has been running well past its budget. + loader.started_t -= modeld.BIG_MODEL_LOAD_TIMEOUT_SECONDS + 1 + assert loader.timed_out + # take() must not hand over a model from a load we already gave up on. + assert loader.take() == (None, "") + finally: + release.set() + loader._thread.join(timeout=10) + + +def test_big_model_load_timeout_leaves_room_for_a_normal_load(): + # A healthy load measured ~26 s on device; the budget must not cut those off. + assert modeld.BIG_MODEL_LOAD_TIMEOUT_SECONDS > 60 + + +def test_big_model_warmup_stays_on_the_background_loader(): + """Warming at promotion blocks the publish loop in one long stretch; warming on the + loader spreads the same work out while the small model is still driving. + + Measured with warmup at promotion: modelV2 went silent for 13.0 s and 12.8 s on two + drives, starting the instant the model was collected, which tripped commIssue and blocked + the driver from engaging. Measured with warmup on the loader: worst in-load gap was + 1.3-2.9 s, spread across the load. Neither is free, but only the second one leaves + modeld publishing when the driver wants to engage. + """ + import ast + from pathlib import Path + + source = (Path(modeld.__file__).with_name("modeld.py")).read_text(encoding="utf-8") + tree = ast.parse(source) + + loader = next(n for n in ast.walk(tree) + if isinstance(n, ast.ClassDef) and n.name == "BigModelLoader") + assert "warmup" in ast.dump(loader), "the loader thread must warm the big model" + + main_fn = next(n for n in tree.body if isinstance(n, ast.FunctionDef) and n.name == "main") + promote = next(n for n in ast.walk(main_fn) + if isinstance(n, ast.If) and "_big_model_swap_allowed" in ast.dump(n.test)) + assert "warmup" not in ast.dump(promote), \ + "promotion must not warm the big model; it runs inside the 20 Hz publish loop" + + +def test_chestnut_telemetry_is_suppressed_while_a_background_load_runs(): + """Telemetry shares Chestnut's USB device with the model weight transfer. + + ChestnutState._read_ina() issues USB control reads on the same device tinygrad streams + weights over. The old code loaded before the publish loop existed so the two never + overlapped; loading in the background makes them concurrent, which stalls the transfer + until it times out. Captured on four drives: the load never completed while telemetry + was polling at 10 Hz, and chestnutState first appeared only after the load on the one + drive that succeeded. + """ + import ast + from pathlib import Path + + source = (Path(modeld.__file__).with_name("modeld.py")).read_text(encoding="utf-8") + main_fn = next(n for n in ast.parse(source).body + if isinstance(n, ast.FunctionDef) and n.name == "main") + assign = next( + node for node in ast.walk(main_fn) + if isinstance(node, ast.Assign) + and any(isinstance(t, ast.Name) and t.id == "send_chestnut" for t in node.targets) + ) + guard = ast.dump(assign.value) + assert "big_loader" in guard and "in_progress" in guard, \ + "chestnutState polling must be gated on the background loader being idle" + + +def test_background_big_model_loader_avoids_the_realtime_cores(): + """The loader must not share a core with modeld or the camera pipeline. + + modeld is SCHED_FIFO on core 7 and threads inherit its affinity, so a loader left there is + starved behind the 20 Hz publish loop. Putting it on camerad's core instead stalls frame + delivery: measured p95 on roadCameraState went 50.8 ms -> 108.9 ms for the whole load. + """ + assert modeld.BIG_MODEL_LOADER_CORES + + reserved = { + 7: "modeld (config_realtime_process(7, 54)) and dmonitoringmodeld", + 6: "camerad (system/camerad/main.cc set_core_affinity({6}))", + 5: "plannerd, radard, selfdrived, starpilot_process", + 4: "card and controlsd", + } + for core, owner in reserved.items(): + assert core not in modeld.BIG_MODEL_LOADER_CORES, f"core {core} belongs to {owner}" + + # camerad pins itself in C++, which is easy to miss when auditing Python callers. + import re + from pathlib import Path + + camerad_main = Path(modeld.__file__).parents[2] / "system" / "camerad" / "main.cc" + pinned = re.search(r"set_core_affinity\(\{([0-9,\s]+)\}\)", camerad_main.read_text(encoding="utf-8")) + assert pinned, "could not find camerad's core affinity" + camerad_cores = {int(c) for c in pinned.group(1).split(",") if c.strip()} + assert not (camerad_cores & modeld.BIG_MODEL_LOADER_CORES), \ + f"the loader must not share camerad's cores {sorted(camerad_cores)}" + + +@pytest.mark.parametrize("failure", [ + dict(model_error=RuntimeError("artifact is corrupt")), + dict(uses_external_gpu=False), +]) +def test_background_big_model_load_reports_failure_without_a_model(monkeypatch, failure): + calls = [] + _stub_big_model_loader(monkeypatch, calls, **failure) + + loader = modeld.BigModelLoader(1928, 1208, "car-params") + loader.start("big-model", "v15") + loader._thread.join(timeout=10) + + loaded, error = loader.take() + assert loaded is None + assert error + assert not loader.in_progress + # Even a failed load must release the loader thread's own tinygrad cache handle. + assert "close_cache" in calls + + +def test_background_big_model_load_is_handed_over_exactly_once(monkeypatch): + calls = [] + fake_model_state = _stub_big_model_loader(monkeypatch, calls) + + loader = modeld.BigModelLoader(1928, 1208, "car-params") + loader.start("big-model", "v15") + loader._thread.join(timeout=10) + + assert isinstance(loader.take()[0], fake_model_state) + # A second take must not promote the same model again. + assert loader.take() == (None, "") + + +def test_cancelled_big_model_load_never_builds_a_model(monkeypatch): + calls = [] + + def cancelling_power_wait(CP, cancel=None): + calls.append(("power", CP)) + cancel.set() + raise modeld.BigModelLoadCancelled("cancelled while waiting for external GPU power") + + _stub_big_model_loader(monkeypatch, calls) + monkeypatch.setattr(modeld, "wait_for_external_gpu_power_ready", cancelling_power_wait) + + loader = modeld.BigModelLoader(1928, 1208, "car-params") + loader.start("big-model", "v15") + loader._thread.join(timeout=10) + + loaded, error = loader.take() + assert loaded is None + assert error + assert not any(call[0] == "model" for call in calls if isinstance(call, tuple)) + + +def test_external_gpu_power_wait_aborts_when_the_loader_is_cancelled(monkeypatch): + cancel = threading.Event() + cancel.set() + monkeypatch.setattr(modeld, "SubMaster", + lambda *_a, **_k: pytest.fail("a cancelled load must not start waiting on power")) + + with pytest.raises(modeld.BigModelLoadCancelled): + modeld.wait_for_external_gpu_power_ready("car-params", cancel=cancel) + + +def test_big_model_promotion_waits_for_the_driver_to_disengage(monkeypatch): + """Drive the real collect-and-promote blocks lifted out of main()'s loop.""" + import ast + from pathlib import Path + + source = (Path(modeld.__file__).with_name("modeld.py")).read_text(encoding="utf-8") + main_fn = next(n for n in ast.parse(source).body if isinstance(n, ast.FunctionDef) and n.name == "main") + def names(node): + return {n.id for n in ast.walk(node) if isinstance(n, ast.Name)} + + collect_block = next(n for n in ast.walk(main_fn) + if isinstance(n, ast.If) and "big_loader" in names(n.test) + and "in_progress" in ast.dump(n.test)) + promote_block = next(n for n in ast.walk(main_fn) + if isinstance(n, ast.If) and "_big_model_swap_allowed" in names(n.test)) + promote = compile(ast.Module(body=[collect_block, promote_block], type_ignores=[]), + "", "exec") + + class FakeModel: + model_id = "big-model" + policy_generation = "v15" + + def __init__(self): + self.reset = 0 + + def warmup(self): + # The real warmup() ends with _reset_state(); it runs at promotion, not on the loader. + self.reset += 1 + + def _reset_state(self): + self.reset += 1 + + class FakeLoader: + def __init__(self, model): + self.in_progress = True + self._model = model + + def take(self): + return self._model, "" + + written = {} + big = FakeModel() + loader = FakeLoader(big) + small = FakeModel() + small.model_id = "small-model" + scope = { + "big_loader": loader, "big_model": None, "model": small, "small_model": small, + "params": SimpleNamespace(put_bool=lambda k, v: written.__setitem__(k, v), + put=lambda k, v: written.__setitem__(k, v)), + "cloudlog": SimpleNamespace(warning=lambda *_a: None, error=lambda *_a: None), + "vipc_dropped_frames": 0, "live_calib_seen": True, "external_gpu_active": False, + "run_count": 99, "frame_dropped_filter": SimpleNamespace(x=5.0), + "PublishState": lambda: "fresh", "publish_state": "stale", + "prev_action": "stale", "log": modeld.log, "chestnut_state": SimpleNamespace(big=False), + "set_runtime_model_params": lambda *_a: None, + "_big_model_swap_allowed": modeld._big_model_swap_allowed, + } + scope["sm"] = type("SM", (), { + "__getitem__": lambda _s, k: SimpleNamespace(enabled=scope["_engaged"]), + "alive": {"carControl": True, "carState": True}, + })() + + # Still loading: nothing is collected and the small model keeps driving. + scope["_engaged"] = True + exec(promote, scope) # noqa: S102 + assert scope["model"] is small and scope["big_model"] is None + + # Load finishes while engaged: the model is collected but must NOT be promoted. + loader.in_progress = False + exec(promote, scope) # noqa: S102 + assert scope["big_model"] is big + assert scope["model"] is small, "must not swap models while the driver is engaged" + assert written["UsbGpuPending"] is True + assert written["UsbGpuLoading"] is False + assert big.reset == 0 + + # The driver disengages: now the big model takes over. + scope["_engaged"] = False + exec(promote, scope) # noqa: S102 + assert scope["model"] is big + assert big.reset == 1, "the big model must be warmed and reset before it drives" + assert scope["run_count"] == 0 and scope["frame_dropped_filter"].x == 0. + assert scope["publish_state"] == "fresh" and scope["prev_action"] != "stale" + assert written["UsbGpuActive"] is True and written["UsbGpuPending"] is False + assert scope["chestnut_state"].big is True + + +@pytest.mark.parametrize("kwargs,expected", [ + (dict(), True), + (dict(engaged=True), False), + (dict(carcontrol_alive=False), False), + (dict(carstate_alive=False), False), + (dict(vipc_dropped_frames=1), False), + (dict(live_calib_seen=False), False), +]) +def test_big_model_swap_only_while_disengaged_with_fresh_state(kwargs, expected): + defaults = dict(engaged=False, carcontrol_alive=True, carstate_alive=True, + vipc_dropped_frames=0, live_calib_seen=True) + assert modeld._big_model_swap_allowed(**(defaults | kwargs)) is expected + + def test_external_gpu_nonfinite_outputs_trigger_fallback(monkeypatch): class FakeTensor: @staticmethod diff --git a/selfdrive/selfdrived/events.py b/selfdrive/selfdrived/events.py index cb5f4c64b1..09c5351d23 100644 --- a/selfdrive/selfdrived/events.py +++ b/selfdrive/selfdrived/events.py @@ -532,9 +532,14 @@ EVENTS: dict[int, dict[str, Alert | AlertCallbackType]] = { "Ensure road ahead is clear"), }, - EventName.bigModelLoading: { - ET.NO_ENTRY: NoEntryAlert("Big Model Loading"), - }, + # bigModelLoading and bigModelPending carry no alerts on purpose: the blinking eGPU icon + # already says the big model is loading, and its absence says it is ready. + EventName.bigModelLoading: {}, + + # bigModelPending carries no alert on purpose: the eGPU icon clearing, and then turning + # green on the next engage, is the driver's cue. A banner here announced the model before + # it was usable. + EventName.bigModelPending: {}, EventName.bigModelFailed: { ET.SOFT_DISABLE: soft_disable_alert("Big Model Failed"), diff --git a/selfdrive/selfdrived/selfdrived.py b/selfdrive/selfdrived/selfdrived.py index aea14c1a93..b33cd627ae 100644 --- a/selfdrive/selfdrived/selfdrived.py +++ b/selfdrive/selfdrived/selfdrived.py @@ -255,7 +255,7 @@ class SelfdriveD: self.big_model_attempted = False self.big_model_active = False self.big_model_failed = False - self.big_model_ready_t = 0. + self.big_model_swap_t = 0. self.experimental_mode = False self.ecu_disable_failed = False self.ecu_disable_failed_checked = not ( @@ -394,15 +394,25 @@ class SelfdriveD: loading = self.params.get_bool("UsbGpuLoading") if loading: self.big_model_attempted = True - if self.big_model_loading and not loading: - self.big_model_ready_t = time.monotonic() self.big_model_loading = loading + # No alert while loading: the blinking eGPU icon already carries that state, and the + # small model is driving normally. big_model_loading still gates the frame-drop warning. if loading: self.events.add(EventName.bigModelLoading) + # The big model loads in the background while the small model drives, so it sits + # loaded-but-unused until the driver disengages. That is a success, not a failure, and + # it raises no alert: the driver sees it through the eGPU icon, not a banner. + pending = self.params.get_bool("UsbGpuPending") + if pending: + self.big_model_attempted = True + big_active = self.params.get("UsbGpuActive") + if self.big_model_active != (big_active is True): + self.big_model_swap_t = time.monotonic() model_unavailable = self.big_model_active and self.sm.seen['modelV2'] and not self.sm.alive['modelV2'] - big_failed = self.big_model_attempted and not loading and (big_active is False or model_unavailable) + big_failed = self.big_model_attempted and not loading and not pending and \ + (big_active is False or model_unavailable) if big_failed and not self.big_model_failed: self.events.add(EventName.bigModelFailed) self.big_model_failed = big_failed @@ -697,10 +707,21 @@ class SelfdriveD: (contains_event_type(self.events, self.starpilot_events, ET.SOFT_DISABLE) or contains_event_type(self.events, self.starpilot_events, ET.IMMEDIATE_DISABLE)) no_system_errors = (not has_disable_events) or (len(self.events) == num_events) - big_model_settling = self.big_model_loading or time.monotonic() < self.big_model_ready_t + 5. + # The model swap briefly disturbs the stream, so forgive everything around it. + big_model_settling = time.monotonic() < self.big_model_swap_t + 2. all_checks = self.sm.all_checks() all_alive = self.sm.all_alive() if not all_checks else True all_freq_ok = self.sm.all_freq_ok() if not all_checks else True + # Loading the big model realizes its graph on the same QCOM GPU the small model drives + # on, which costs modeld frames for part of the load and drags modelV2 below its target + # rate. commIssueAvgFreq is NO_ENTRY, so that would block engaging on a small model that + # is working fine. Forgive only a slow modelV2, and only while a load is in flight -- + # anything dying, and any other service, still reports normally. + if self.big_model_loading and not all_checks and all_alive and not all_freq_ok: + slow = {s for s, freq_ok in self.sm.freq_ok.items() if not freq_ok} + if slow and slow <= {'modelV2', 'drivingModelData', 'cameraOdometry'} and self.sm.all_valid(): + all_freq_ok = True + all_checks = True report_comm_issue, self.valid_only_comm_issue_frames = evaluate_comm_issue( all_checks, all_alive, all_freq_ok, self.valid_only_comm_issue_frames, ) @@ -798,7 +819,10 @@ class SelfdriveD: # TODO: fix simulator if not SIMULATION or REPLAY: - if self.sm['modelV2'].frameDropPerc > 20: + # Loading the big model realizes its graph on the same QCOM GPU the small model is + # driving on, which costs real frames for part of the load. The small model keeps + # publishing throughout, so warn about the drops only once the load is out of the way. + if self.sm['modelV2'].frameDropPerc > 20 and not self.big_model_loading: self.events.add(EventName.modeldLagging) # Decrement personality on configured steering-wheel button presses diff --git a/selfdrive/ui/onroad/starpilot/widgets/model_source.py b/selfdrive/ui/onroad/starpilot/widgets/model_source.py index b610715e81..eb78312eca 100644 --- a/selfdrive/ui/onroad/starpilot/widgets/model_source.py +++ b/selfdrive/ui/onroad/starpilot/widgets/model_source.py @@ -62,12 +62,19 @@ class ModelSourceWidget(LayoutWidget): return self.SIZE @staticmethod - def _big_model_failed(active: bool | None, usbgpu: bool, model_seen: bool, model_alive: bool) -> bool: + def _big_model_failed(active: bool | None, usbgpu: bool, model_seen: bool, model_alive: bool, + pending: bool = False) -> bool: + # A model that finished loading but is waiting for the driver to disengage is not a failure. + if pending and usbgpu: + return False return active is False or not usbgpu or (active is True and model_seen and not model_alive) @staticmethod - def _status_for(loading: bool, small_model_engaged: bool, big_failed: bool) -> ModelSourceStatus: - if loading: + def _status_for(loading: bool, small_model_engaged: bool, big_failed: bool, + pending: bool = False) -> ModelSourceStatus: + # Pending means the big model finished loading and is waiting for the next disengage, + # so the blinking loading icon stops: its absence is what tells the driver it is ready. + if loading and not pending: return ModelSourceStatus.LOADING if small_model_engaged: return ModelSourceStatus.FALLBACK_ENGAGED @@ -84,16 +91,18 @@ class ModelSourceWidget(LayoutWidget): model_seen = sm.recv_frame["modelV2"] > ui_state.started_frame model_alive = sm.alive["modelV2"] if model_seen else True loading = ui_state.usbgpu_loading - big_failed = self._big_model_failed(ui_state.usbgpu_active, ui_state.usbgpu, model_seen, model_alive) + pending = ui_state.usbgpu_pending + big_failed = self._big_model_failed(ui_state.usbgpu_active, ui_state.usbgpu, model_seen, model_alive, pending) engaged = sm["selfdriveState"].enabled - if engaged and not self._engaged and not loading and ui_state.usbgpu_active is not True and model_seen: + if engaged and not self._engaged and not loading and not pending \ + and ui_state.usbgpu_active is not True and model_seen: self._small_model_engaged = True if engaged != self._engaged: self._fade_time = rl.get_time() if engaged else 0.0 self._engaged = engaged self._small_model_engaged &= big_failed - self._status = self._status_for(loading, self._small_model_engaged, big_failed) + self._status = self._status_for(loading, self._small_model_engaged, big_failed, pending) def _render(self, rect: rl.Rectangle) -> None: if self._status is None: diff --git a/selfdrive/ui/soundd.py b/selfdrive/ui/soundd.py index e2ede4fdf4..7bc5c6a82d 100644 --- a/selfdrive/ui/soundd.py +++ b/selfdrive/ui/soundd.py @@ -344,8 +344,11 @@ class Soundd: if now - self.model_ready_last_check >= 0.25: self.model_ready_last_check = now try: + # The big model now loads in the background, so it is announced when it becomes + # usable (pending the driver's next disengage) rather than only once it is running. self.model_ready_pending = self.model_ready_chime.update( - active=self.model_ready_params.get_bool("UsbGpuActive"), + active=(self.model_ready_params.get_bool("UsbGpuPending") or + self.model_ready_params.get_bool("UsbGpuActive")), loading=self.model_ready_params.get_bool("UsbGpuLoading"), onroad=self.model_ready_params.get_bool("IsOnroad"), enabled=self.model_ready_params.get_bool("GpuModelReadySound"), now=now) diff --git a/selfdrive/ui/tests/test_model_source_widget.py b/selfdrive/ui/tests/test_model_source_widget.py index f3d1848f5e..f0733dcad5 100644 --- a/selfdrive/ui/tests/test_model_source_widget.py +++ b/selfdrive/ui/tests/test_model_source_widget.py @@ -32,6 +32,19 @@ def test_model_source_failure_detection_matches_the_backend_state_contract(): assert not failed(True, True, False, True) +def test_model_source_blinks_while_loading_then_clears_once_the_big_model_is_ready(): + # The blinking icon is the driver's "still loading" cue, and its disappearance is how they + # know the next disengage/engage will pick up the big model. + status = model_source.ModelSourceWidget._status_for + failed = model_source.ModelSourceWidget._big_model_failed + + assert status(True, False, False, False) is model_source.ModelSourceStatus.LOADING + assert status(True, False, False, True) is not model_source.ModelSourceStatus.LOADING + assert not failed(False, True, False, True, True) + # Chestnut going away while pending is still a genuine failure. + assert failed(False, False, False, True, True) + + def test_model_source_latches_small_model_engagement_until_the_big_model_recovers(monkeypatch): widget = object.__new__(model_source.ModelSourceWidget) widget._small_model_engaged = False @@ -50,6 +63,7 @@ def test_model_source_latches_small_model_engagement_until_the_big_model_recover usbgpu_compiled=True, usbgpu_active=False, usbgpu_loading=False, + usbgpu_pending=False, ), ) monkeypatch.setattr(model_source.rl, "get_time", lambda: 42.0) diff --git a/selfdrive/ui/ui_state.py b/selfdrive/ui/ui_state.py index 10c22d6384..2905a87ac9 100644 --- a/selfdrive/ui/ui_state.py +++ b/selfdrive/ui/ui_state.py @@ -97,6 +97,7 @@ class UIState: self.usbgpu_compiled: bool = self.params.get_bool("UsbGpuCompiled") self.usbgpu_active: bool = self.params.get_bool("UsbGpuActive") self.usbgpu_loading: bool = self.params.get_bool("UsbGpuLoading") + self.usbgpu_pending: bool = self.params.get_bool("UsbGpuPending") self.started: bool = False self.ignition: bool = False self.recording_audio: bool = False @@ -220,6 +221,7 @@ class UIState: self.usbgpu_compiled = params.get_bool("UsbGpuCompiled") self.usbgpu_active = params.get_bool("UsbGpuActive") self.usbgpu_loading = params.get_bool("UsbGpuLoading") + self.usbgpu_pending = params.get_bool("UsbGpuPending") self.switchback_mode_enabled = self.params_memory.get_bool("SwitchbackModeEnabled") if self.started else False self.conditional_status = self.params_memory.get_int("CEStatus", default=0) if self.started else 0 mark_progress("ui.update.after_state_params")