mirror of
https://github.com/firestar5683/StarPilot.git
synced 2026-09-28 10:23:49 +08:00
Drive small until big
This commit is contained in:
@@ -132,6 +132,7 @@ struct OnroadEvent @0xc4fa6047f024e718 {
|
||||
audioFeedback @97;
|
||||
bigModelLoading @100;
|
||||
bigModelFailed @102;
|
||||
bigModelPending @103;
|
||||
|
||||
soundsUnavailableDEPRECATED @47;
|
||||
stockLkasDEPRECATED @98;
|
||||
|
||||
@@ -177,6 +177,7 @@ inline static std::unordered_map<std::string, ParamKeyAttributes> keys = {
|
||||
{"UsbGpuActive", {CLEAR_ON_MANAGER_START | CLEAR_ON_OFFROAD_TRANSITION, BOOL}},
|
||||
{"UsbGpuCompiled", {CLEAR_ON_MANAGER_START | CLEAR_ON_OFFROAD_TRANSITION, BOOL}},
|
||||
{"UsbGpuLoading", {CLEAR_ON_MANAGER_START | CLEAR_ON_OFFROAD_TRANSITION, BOOL}},
|
||||
{"UsbGpuPending", {CLEAR_ON_MANAGER_START | CLEAR_ON_OFFROAD_TRANSITION, BOOL}},
|
||||
{"UsbGpuPresent", {CLEAR_ON_MANAGER_START | CLEAR_ON_OFFROAD_TRANSITION, BOOL}},
|
||||
{"Version", {PERSISTENT, STRING}},
|
||||
|
||||
|
||||
+185
-43
@@ -5,6 +5,7 @@ from functools import cached_property
|
||||
import json
|
||||
import os
|
||||
import struct
|
||||
import threading
|
||||
import usb1
|
||||
from openpilot.system.hardware import HARDWARE, TICI
|
||||
os.environ['GMMU'] = '0'
|
||||
@@ -25,7 +26,7 @@ from openpilot.common.swaglog import cloudlog
|
||||
from openpilot.common.params import Params
|
||||
from openpilot.common.filter_simple import FirstOrderFilter
|
||||
from openpilot.common.file_chunker import file_chunked_exists, open_file_chunked
|
||||
from openpilot.common.realtime import config_realtime_process, DT_MDL
|
||||
from openpilot.common.realtime import config_realtime_process, set_core_affinity, DT_MDL
|
||||
from openpilot.common.transformations.camera import DEVICE_CAMERAS
|
||||
from openpilot.common.transformations.model import get_warp_matrix
|
||||
from openpilot.system.camerad.cameras.nv12_info import get_nv12_info
|
||||
@@ -105,6 +106,15 @@ EXTERNAL_GPU_POWER_WAIT_TIMEOUT_SECONDS = 60.0
|
||||
EXTERNAL_GPU_POWER_LOG_INTERVAL_SECONDS = 10.0
|
||||
LAT_SMOOTH_BP = [2.0, 8.0]
|
||||
|
||||
# The background big-model loader must leave modeld's SCHED_FIFO core. Thread affinity is
|
||||
# inherited from config_realtime_process(7, 54), so without this the loader would sit behind
|
||||
# the 20 Hz publish loop on core 7 (shared with dmonitoringmodeld) and barely run at all.
|
||||
BIG_MODEL_LOADER_CORES = {6}
|
||||
|
||||
|
||||
class BigModelLoadCancelled(Exception):
|
||||
"""Raised inside the background loader when modeld no longer wants the big model."""
|
||||
|
||||
|
||||
def _set_hcq_wait_timeout(timeout_ms: int) -> None:
|
||||
"""Update tinygrad's cached HCQ timeout for the external-GPU load/run phase."""
|
||||
@@ -155,8 +165,10 @@ def _egmp_vehicle_ready(can_messages, bus: int) -> bool:
|
||||
)
|
||||
|
||||
|
||||
def wait_for_external_gpu_power_ready(CP=None) -> None:
|
||||
def wait_for_external_gpu_power_ready(CP=None, cancel=None) -> None:
|
||||
"""Wait out vehicle startup power transitions before initializing Chestnut."""
|
||||
if cancel is not None and cancel.is_set():
|
||||
raise BigModelLoadCancelled("cancelled before waiting for external GPU power")
|
||||
device_type = HARDWARE.get_device_type()
|
||||
egmp_bus = _egmp_ready_bus(CP)
|
||||
services = ["pandaStates", "peripheralState"] + (["can"] if egmp_bus is not None else [])
|
||||
@@ -167,6 +179,8 @@ def wait_for_external_gpu_power_ready(CP=None) -> None:
|
||||
wait_started = time.monotonic()
|
||||
|
||||
while True:
|
||||
if cancel is not None and cancel.is_set():
|
||||
raise BigModelLoadCancelled("cancelled while waiting for external GPU power")
|
||||
sm.update(1000)
|
||||
now = time.monotonic()
|
||||
if egmp_bus is not None and sm.updated["can"] and _egmp_vehicle_ready(sm["can"], egmp_bus):
|
||||
@@ -854,33 +868,111 @@ def _load_model_state(cam_w: int, cam_h: int, selected_model: str, external_gpu_
|
||||
)
|
||||
|
||||
|
||||
def _load_external_gpu_model(cam_w: int, cam_h: int, selected_model: str, model_version: str = "",
|
||||
CP=None, demo: bool = False) -> ModelState | None:
|
||||
"""Load and warm the USB-GPU model without running another tinygrad model concurrently."""
|
||||
candidate = None
|
||||
try:
|
||||
if not demo:
|
||||
wait_for_external_gpu_power_ready(CP)
|
||||
_set_hcq_wait_timeout(BIG_MODEL_LOAD_WAIT_TIMEOUT_MS)
|
||||
wait_usbgpu_link()
|
||||
candidate = ModelState(
|
||||
cam_w,
|
||||
cam_h,
|
||||
True,
|
||||
model_id_override=selected_model,
|
||||
write_model_version=False,
|
||||
model_version_override=model_version,
|
||||
)
|
||||
if not candidate.uses_external_gpu:
|
||||
raise RuntimeError("external GPU model resolved to the builtin model")
|
||||
candidate.warmup()
|
||||
return candidate
|
||||
except Exception:
|
||||
cloudlog.exception("external GPU model load or warmup failed")
|
||||
return None
|
||||
finally:
|
||||
_close_tinygrad_disk_cache_connection()
|
||||
_set_hcq_wait_timeout(BIG_MODEL_RUN_WAIT_TIMEOUT_MS)
|
||||
def _big_model_swap_allowed(engaged: bool, carcontrol_alive: bool, carstate_alive: bool,
|
||||
vipc_dropped_frames: int, live_calib_seen: bool) -> bool:
|
||||
"""Swapping models rebuilds the temporal input queues, so it must never happen under actuation.
|
||||
|
||||
Fails closed: without fresh carControl/carState we cannot trust that we are disengaged.
|
||||
"""
|
||||
return (
|
||||
not engaged
|
||||
and carcontrol_alive
|
||||
and carstate_alive
|
||||
and vipc_dropped_frames == 0
|
||||
and live_calib_seen
|
||||
)
|
||||
|
||||
|
||||
class BigModelLoader:
|
||||
"""Load and warm the Chestnut model on a background thread while the small model drives.
|
||||
|
||||
Everything this touches must be safe against a concurrently running small model, so unlike
|
||||
the synchronous path it never calls _set_hcq_wait_timeout (hoisted to modeld startup) or
|
||||
_isolate_next_model_artifact_load (which evicts buffer UOps from a tinygrad global cache).
|
||||
"""
|
||||
|
||||
def __init__(self, cam_w: int, cam_h: int, CP=None, demo: bool = False):
|
||||
self.cam_w = cam_w
|
||||
self.cam_h = cam_h
|
||||
self.CP = CP
|
||||
self.demo = demo
|
||||
self._thread: threading.Thread | None = None
|
||||
self._cancel = threading.Event()
|
||||
self._lock = threading.Lock()
|
||||
self._result: ModelState | None = None
|
||||
self._error = ""
|
||||
self._model_id = ""
|
||||
self._model_version = ""
|
||||
self.started_t = 0.0
|
||||
|
||||
@property
|
||||
def in_progress(self) -> bool:
|
||||
return self._thread is not None and self._thread.is_alive()
|
||||
|
||||
def start(self, model_id: str, model_version: str = "") -> bool:
|
||||
if self.in_progress:
|
||||
return False
|
||||
with self._lock:
|
||||
self._result = None
|
||||
self._error = ""
|
||||
self._model_id = model_id
|
||||
self._model_version = model_version
|
||||
self._cancel.clear()
|
||||
self.started_t = time.monotonic()
|
||||
self._thread = threading.Thread(target=self._run, name="big_model_loader", daemon=True)
|
||||
self._thread.start()
|
||||
return True
|
||||
|
||||
def cancel(self) -> None:
|
||||
self._cancel.set()
|
||||
|
||||
def take(self) -> tuple[ModelState | None, str]:
|
||||
"""Hand over the finished load exactly once: (model, error). Both empty while in flight."""
|
||||
if self._thread is None or self._thread.is_alive():
|
||||
return None, ""
|
||||
self._thread = None
|
||||
with self._lock:
|
||||
result, error = self._result, self._error
|
||||
self._result = None
|
||||
self._error = ""
|
||||
return result, error
|
||||
|
||||
def _run(self) -> None:
|
||||
try:
|
||||
# Leave modeld's realtime core before doing any work; see BIG_MODEL_LOADER_CORES.
|
||||
set_core_affinity(sorted(BIG_MODEL_LOADER_CORES))
|
||||
if not self.demo:
|
||||
wait_for_external_gpu_power_ready(self.CP, cancel=self._cancel)
|
||||
if self._cancel.is_set():
|
||||
raise BigModelLoadCancelled("cancelled before the external GPU link check")
|
||||
wait_usbgpu_link()
|
||||
candidate = ModelState(
|
||||
self.cam_w,
|
||||
self.cam_h,
|
||||
True,
|
||||
model_id_override=self._model_id,
|
||||
write_model_version=False,
|
||||
model_version_override=self._model_version,
|
||||
)
|
||||
if not candidate.uses_external_gpu:
|
||||
raise RuntimeError("external GPU model resolved to the builtin model")
|
||||
candidate.warmup()
|
||||
if self._cancel.is_set():
|
||||
raise BigModelLoadCancelled("cancelled after warmup")
|
||||
with self._lock:
|
||||
self._result = candidate
|
||||
cloudlog.warning(f"background big model load finished in {time.monotonic() - self.started_t:.1f}s")
|
||||
except BigModelLoadCancelled as exc:
|
||||
cloudlog.warning(f"background big model load cancelled: {exc}")
|
||||
with self._lock:
|
||||
self._error = str(exc)
|
||||
except Exception as exc:
|
||||
cloudlog.exception("background big model load failed")
|
||||
with self._lock:
|
||||
self._error = str(exc) or exc.__class__.__name__
|
||||
finally:
|
||||
# tinygrad's disk cache handle is thread-local, so this closes only the loader's.
|
||||
_close_tinygrad_disk_cache_connection()
|
||||
|
||||
|
||||
def _model_versions() -> dict[str, str]:
|
||||
@@ -1058,9 +1150,16 @@ def main(demo=False):
|
||||
external_artifact = MODELS_PATH / f"{big_model_id}_driving_tinygrad.pkl"
|
||||
external_artifact_ready = external_model_selected and file_chunked_exists(external_artifact)
|
||||
external_gpu_requested = usbgpu_present_now and (bool(big_model_id) or model_lab_ready)
|
||||
# The big model is loaded in the background while the small model drives, so the HCQ
|
||||
# watchdog is set once here and never mutated again: it is process-global and shared by
|
||||
# the QCOM and AMD devices, so changing it mid-drive would also move the small model's.
|
||||
if external_gpu_requested:
|
||||
_set_hcq_wait_timeout(BIG_MODEL_LOAD_WAIT_TIMEOUT_MS)
|
||||
|
||||
params.put_bool("UsbGpuPresent", usbgpu_present_now)
|
||||
params.put_bool("UsbGpuCompiled", external_artifact_ready or model_lab_ready)
|
||||
params.put_bool("UsbGpuActive", False)
|
||||
params.put_bool("UsbGpuPending", False)
|
||||
params.put_bool("UsbGpuLoading", external_gpu_requested)
|
||||
_set_model_lab_runtime(
|
||||
params,
|
||||
@@ -1147,15 +1246,8 @@ def main(demo=False):
|
||||
else:
|
||||
CP = messaging.log_from_bytes(params.get("CarParams", block=True), car.CarParams)
|
||||
|
||||
big_model = _load_external_gpu_model(
|
||||
vipc_client_main.width,
|
||||
vipc_client_main.height,
|
||||
selected_model,
|
||||
selected_model_version,
|
||||
CP,
|
||||
demo,
|
||||
)
|
||||
|
||||
# Start on the small model so openpilot is drivable immediately; the big model is loaded
|
||||
# in the background below and promoted once the driver disengages.
|
||||
small_model = _load_model_state(
|
||||
vipc_client_main.width,
|
||||
vipc_client_main.height,
|
||||
@@ -1165,10 +1257,7 @@ def main(demo=False):
|
||||
small_model_version,
|
||||
False,
|
||||
)
|
||||
model = big_model if big_model is not None else small_model
|
||||
if big_model is not None:
|
||||
params.put("ModelVersion", model.policy_generation)
|
||||
params.put("DrivingModelVersion", model.policy_generation)
|
||||
model = small_model
|
||||
else:
|
||||
model = _load_model_state(
|
||||
vipc_client_main.width,
|
||||
@@ -1183,9 +1272,15 @@ def main(demo=False):
|
||||
set_runtime_model_params(params, model.model_id, model.policy_generation)
|
||||
|
||||
external_gpu_active = model_lab_active or model.uses_external_gpu
|
||||
# Load the big model in the background so the small model can drive in the meantime.
|
||||
big_loader = None
|
||||
if external_gpu_requested and not model_lab_active and big_model_id:
|
||||
big_loader = BigModelLoader(vipc_client_main.width, vipc_client_main.height, CP, demo)
|
||||
big_loader.start(big_model_id, big_model_version)
|
||||
params.put_bool("UsbGpuCompiled", external_artifact_ready or model_lab_ready)
|
||||
params.put_bool("UsbGpuActive", external_gpu_active)
|
||||
params.put_bool("UsbGpuLoading", False)
|
||||
params.put_bool("UsbGpuPending", False)
|
||||
params.put_bool("UsbGpuLoading", big_loader is not None)
|
||||
_set_model_lab_runtime(
|
||||
params,
|
||||
requested=model_lab_requested,
|
||||
@@ -1340,6 +1435,48 @@ def main(demo=False):
|
||||
chestnut_state.big = False
|
||||
cloudlog.error(f"Model Laboratory stopped: {model_lab_error}")
|
||||
|
||||
# Collect a finished background big-model load, then promote it once the driver is
|
||||
# disengaged. Swapping rebuilds the temporal input queues, so it must not happen
|
||||
# under actuation.
|
||||
if big_loader is not None and not big_loader.in_progress:
|
||||
loaded_big_model, big_load_error = big_loader.take()
|
||||
big_loader = None
|
||||
params.put_bool("UsbGpuLoading", False)
|
||||
if loaded_big_model is not None:
|
||||
big_model = loaded_big_model
|
||||
params.put_bool("UsbGpuPending", True)
|
||||
cloudlog.warning("big model ready; waiting for the driver to disengage before using it")
|
||||
else:
|
||||
params.put_bool("UsbGpuPending", False)
|
||||
params.put_bool("UsbGpuActive", False)
|
||||
cloudlog.error(f"big model unavailable, staying on the small model: {big_load_error}")
|
||||
|
||||
if big_model is not None and model is not big_model and _big_model_swap_allowed(
|
||||
sm["carControl"].enabled,
|
||||
sm.alive["carControl"],
|
||||
sm.alive["carState"],
|
||||
vipc_dropped_frames,
|
||||
live_calib_seen,
|
||||
):
|
||||
big_model._reset_state()
|
||||
model = big_model
|
||||
external_gpu_active = True
|
||||
# Nothing from the small model carries over: re-arm the frame-drop warmup and drop
|
||||
# the rolling probability buffers and previous action, which are model specific.
|
||||
run_count = 0
|
||||
frame_dropped_filter.x = 0.
|
||||
publish_state = PublishState()
|
||||
prev_action = log.ModelDataV2.Action()
|
||||
params.put("ModelVersion", model.policy_generation)
|
||||
params.put("DrivingModelVersion", model.policy_generation)
|
||||
set_runtime_model_params(params, model.model_id, model.policy_generation)
|
||||
params.put_bool("UsbGpuActive", True)
|
||||
params.put_bool("UsbGpuPending", False)
|
||||
params.put_bool("UsbGpuLoading", False)
|
||||
if chestnut_state is not None:
|
||||
chestnut_state.big = True
|
||||
cloudlog.warning(f"now driving on the big model {model.model_id}")
|
||||
|
||||
frame_drop_ratio = frames_dropped / (1 + frames_dropped)
|
||||
dropped_frame = vipc_dropped_frames > 0
|
||||
if dropped_frame and (model.can_prepare_only or (model_lab_longitudinal is not None and model_lab_longitudinal.can_prepare_only)):
|
||||
@@ -1433,11 +1570,16 @@ def main(demo=False):
|
||||
cloudlog.exception("external GPU model failed, falling back to active small model")
|
||||
model = small_model
|
||||
big_model = None
|
||||
# A failed big model is not retried in this drive, so stop any load still in flight.
|
||||
if big_loader is not None:
|
||||
big_loader.cancel()
|
||||
big_loader = None
|
||||
params.put_bool("UsbGpuActive", False)
|
||||
external_gpu_active = False
|
||||
params.put("ModelVersion", model.policy_generation)
|
||||
params.put("DrivingModelVersion", model.policy_generation)
|
||||
set_runtime_model_params(params, model.model_id, model.policy_generation)
|
||||
params.put_bool("UsbGpuPending", False)
|
||||
params.put_bool("UsbGpuLoading", False)
|
||||
if chestnut_state is not None:
|
||||
chestnut_state.big = False
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
import io
|
||||
import struct
|
||||
import threading
|
||||
from types import MethodType
|
||||
from types import SimpleNamespace
|
||||
|
||||
@@ -269,45 +270,225 @@ def test_tinygrad_empty_thread_local_cache_holder_is_safe(monkeypatch):
|
||||
assert tinygrad_helpers._db_connection is holder
|
||||
|
||||
|
||||
def test_external_gpu_load_finishes_before_native_model_can_start(monkeypatch):
|
||||
calls = []
|
||||
|
||||
def _stub_big_model_loader(monkeypatch, calls, *, uses_external_gpu=True, model_error=None):
|
||||
class FakeModelState:
|
||||
uses_external_gpu = True
|
||||
|
||||
def __init__(self, cam_w, cam_h, external_gpu_active, model_id_override, write_model_version,
|
||||
model_version_override):
|
||||
calls.append(("model", cam_w, cam_h, external_gpu_active, model_id_override,
|
||||
write_model_version, model_version_override))
|
||||
if model_error is not None:
|
||||
raise model_error
|
||||
self.uses_external_gpu = uses_external_gpu
|
||||
|
||||
def warmup(self):
|
||||
calls.append("warmup")
|
||||
|
||||
monkeypatch.setattr(modeld, "set_core_affinity", lambda cores: calls.append(("affinity", tuple(cores))))
|
||||
monkeypatch.setattr(modeld, "wait_usbgpu_link", lambda: calls.append("link"))
|
||||
monkeypatch.setattr(modeld, "wait_for_external_gpu_power_ready", lambda CP: calls.append(("power", CP)))
|
||||
monkeypatch.setattr(modeld, "_set_hcq_wait_timeout", lambda timeout: calls.append(("timeout", timeout)))
|
||||
monkeypatch.setattr(modeld, "wait_for_external_gpu_power_ready",
|
||||
lambda CP, cancel=None: calls.append(("power", CP)))
|
||||
monkeypatch.setattr(modeld, "_close_tinygrad_disk_cache_connection", lambda: calls.append("close_cache"))
|
||||
monkeypatch.setattr(modeld, "ModelState", FakeModelState)
|
||||
monkeypatch.setattr(
|
||||
modeld,
|
||||
"tinygrad_dev_config",
|
||||
lambda *_args: (_ for _ in ()).throw(AssertionError("runtime must not change tinygrad's process-global DEV")),
|
||||
)
|
||||
return FakeModelState
|
||||
|
||||
loaded = modeld._load_external_gpu_model(1928, 1208, "big-model", "v15", "car-params")
|
||||
|
||||
assert isinstance(loaded, FakeModelState)
|
||||
def test_background_big_model_load_leaves_the_running_model_untouched(monkeypatch):
|
||||
calls = []
|
||||
fake_model_state = _stub_big_model_loader(monkeypatch, calls)
|
||||
# The small model is already driving on these process-global settings, so the background
|
||||
# load must not touch tinygrad's DEV, its HCQ watchdog, or its shared buffer UOp cache.
|
||||
for name, detail in (
|
||||
("tinygrad_dev_config", "runtime must not change tinygrad's process-global DEV"),
|
||||
("_set_hcq_wait_timeout", "the background load must not move the running model's HCQ watchdog"),
|
||||
("_isolate_next_model_artifact_load", "the background load must not evict the running model's buffers"),
|
||||
):
|
||||
monkeypatch.setattr(modeld, name, lambda *_args, _d=detail: (_ for _ in ()).throw(AssertionError(_d)))
|
||||
|
||||
loader = modeld.BigModelLoader(1928, 1208, "car-params")
|
||||
assert loader.start("big-model", "v15")
|
||||
loader._thread.join(timeout=10)
|
||||
|
||||
loaded, error = loader.take()
|
||||
assert isinstance(loaded, fake_model_state)
|
||||
assert error == ""
|
||||
assert calls == [
|
||||
("affinity", tuple(sorted(modeld.BIG_MODEL_LOADER_CORES))),
|
||||
("power", "car-params"),
|
||||
("timeout", modeld.BIG_MODEL_LOAD_WAIT_TIMEOUT_MS),
|
||||
"link",
|
||||
("model", 1928, 1208, True, "big-model", False, "v15"),
|
||||
"warmup",
|
||||
"close_cache",
|
||||
("timeout", modeld.BIG_MODEL_RUN_WAIT_TIMEOUT_MS),
|
||||
]
|
||||
|
||||
|
||||
def test_background_big_model_loader_runs_off_modelds_realtime_core():
|
||||
# modeld is SCHED_FIFO on core 7 and threads inherit its affinity, so a loader left there
|
||||
# would be starved behind the 20 Hz publish loop.
|
||||
assert 7 not in modeld.BIG_MODEL_LOADER_CORES
|
||||
assert modeld.BIG_MODEL_LOADER_CORES
|
||||
|
||||
|
||||
@pytest.mark.parametrize("failure", [
|
||||
dict(model_error=RuntimeError("artifact is corrupt")),
|
||||
dict(uses_external_gpu=False),
|
||||
])
|
||||
def test_background_big_model_load_reports_failure_without_a_model(monkeypatch, failure):
|
||||
calls = []
|
||||
_stub_big_model_loader(monkeypatch, calls, **failure)
|
||||
|
||||
loader = modeld.BigModelLoader(1928, 1208, "car-params")
|
||||
loader.start("big-model", "v15")
|
||||
loader._thread.join(timeout=10)
|
||||
|
||||
loaded, error = loader.take()
|
||||
assert loaded is None
|
||||
assert error
|
||||
assert not loader.in_progress
|
||||
# Even a failed load must release the loader thread's own tinygrad cache handle.
|
||||
assert "close_cache" in calls
|
||||
|
||||
|
||||
def test_background_big_model_load_is_handed_over_exactly_once(monkeypatch):
|
||||
calls = []
|
||||
fake_model_state = _stub_big_model_loader(monkeypatch, calls)
|
||||
|
||||
loader = modeld.BigModelLoader(1928, 1208, "car-params")
|
||||
loader.start("big-model", "v15")
|
||||
loader._thread.join(timeout=10)
|
||||
|
||||
assert isinstance(loader.take()[0], fake_model_state)
|
||||
# A second take must not promote the same model again.
|
||||
assert loader.take() == (None, "")
|
||||
|
||||
|
||||
def test_cancelled_big_model_load_never_builds_a_model(monkeypatch):
|
||||
calls = []
|
||||
|
||||
def cancelling_power_wait(CP, cancel=None):
|
||||
calls.append(("power", CP))
|
||||
cancel.set()
|
||||
raise modeld.BigModelLoadCancelled("cancelled while waiting for external GPU power")
|
||||
|
||||
_stub_big_model_loader(monkeypatch, calls)
|
||||
monkeypatch.setattr(modeld, "wait_for_external_gpu_power_ready", cancelling_power_wait)
|
||||
|
||||
loader = modeld.BigModelLoader(1928, 1208, "car-params")
|
||||
loader.start("big-model", "v15")
|
||||
loader._thread.join(timeout=10)
|
||||
|
||||
loaded, error = loader.take()
|
||||
assert loaded is None
|
||||
assert error
|
||||
assert not any(call[0] == "model" for call in calls if isinstance(call, tuple))
|
||||
|
||||
|
||||
def test_external_gpu_power_wait_aborts_when_the_loader_is_cancelled(monkeypatch):
|
||||
cancel = threading.Event()
|
||||
cancel.set()
|
||||
monkeypatch.setattr(modeld, "SubMaster",
|
||||
lambda *_a, **_k: pytest.fail("a cancelled load must not start waiting on power"))
|
||||
|
||||
with pytest.raises(modeld.BigModelLoadCancelled):
|
||||
modeld.wait_for_external_gpu_power_ready("car-params", cancel=cancel)
|
||||
|
||||
|
||||
def test_big_model_promotion_waits_for_the_driver_to_disengage(monkeypatch):
|
||||
"""Drive the real collect-and-promote blocks lifted out of main()'s loop."""
|
||||
import ast
|
||||
from pathlib import Path
|
||||
|
||||
source = (Path(modeld.__file__).with_name("modeld.py")).read_text(encoding="utf-8")
|
||||
main_fn = next(n for n in ast.parse(source).body if isinstance(n, ast.FunctionDef) and n.name == "main")
|
||||
def names(node):
|
||||
return {n.id for n in ast.walk(node) if isinstance(n, ast.Name)}
|
||||
|
||||
collect_block = next(n for n in ast.walk(main_fn)
|
||||
if isinstance(n, ast.If) and "big_loader" in names(n.test)
|
||||
and "in_progress" in ast.dump(n.test))
|
||||
promote_block = next(n for n in ast.walk(main_fn)
|
||||
if isinstance(n, ast.If) and "_big_model_swap_allowed" in names(n.test))
|
||||
promote = compile(ast.Module(body=[collect_block, promote_block], type_ignores=[]),
|
||||
"<promotion>", "exec")
|
||||
|
||||
class FakeModel:
|
||||
model_id = "big-model"
|
||||
policy_generation = "v15"
|
||||
|
||||
def __init__(self):
|
||||
self.reset = 0
|
||||
|
||||
def _reset_state(self):
|
||||
self.reset += 1
|
||||
|
||||
class FakeLoader:
|
||||
def __init__(self, model):
|
||||
self.in_progress = True
|
||||
self._model = model
|
||||
|
||||
def take(self):
|
||||
return self._model, ""
|
||||
|
||||
written = {}
|
||||
big = FakeModel()
|
||||
loader = FakeLoader(big)
|
||||
small = FakeModel()
|
||||
small.model_id = "small-model"
|
||||
scope = {
|
||||
"big_loader": loader, "big_model": None, "model": small, "small_model": small,
|
||||
"params": SimpleNamespace(put_bool=lambda k, v: written.__setitem__(k, v),
|
||||
put=lambda k, v: written.__setitem__(k, v)),
|
||||
"cloudlog": SimpleNamespace(warning=lambda *_a: None, error=lambda *_a: None),
|
||||
"vipc_dropped_frames": 0, "live_calib_seen": True, "external_gpu_active": False,
|
||||
"run_count": 99, "frame_dropped_filter": SimpleNamespace(x=5.0),
|
||||
"PublishState": lambda: "fresh", "publish_state": "stale",
|
||||
"prev_action": "stale", "log": modeld.log, "chestnut_state": SimpleNamespace(big=False),
|
||||
"set_runtime_model_params": lambda *_a: None,
|
||||
"_big_model_swap_allowed": modeld._big_model_swap_allowed,
|
||||
}
|
||||
scope["sm"] = type("SM", (), {
|
||||
"__getitem__": lambda _s, k: SimpleNamespace(enabled=scope["_engaged"]),
|
||||
"alive": {"carControl": True, "carState": True},
|
||||
})()
|
||||
|
||||
# Still loading: nothing is collected and the small model keeps driving.
|
||||
scope["_engaged"] = True
|
||||
exec(promote, scope) # noqa: S102
|
||||
assert scope["model"] is small and scope["big_model"] is None
|
||||
|
||||
# Load finishes while engaged: the model is collected but must NOT be promoted.
|
||||
loader.in_progress = False
|
||||
exec(promote, scope) # noqa: S102
|
||||
assert scope["big_model"] is big
|
||||
assert scope["model"] is small, "must not swap models while the driver is engaged"
|
||||
assert written["UsbGpuPending"] is True
|
||||
assert written["UsbGpuLoading"] is False
|
||||
assert big.reset == 0
|
||||
|
||||
# The driver disengages: now the big model takes over.
|
||||
scope["_engaged"] = False
|
||||
exec(promote, scope) # noqa: S102
|
||||
assert scope["model"] is big
|
||||
assert big.reset == 1, "temporal queues must be reset before the big model drives"
|
||||
assert scope["run_count"] == 0 and scope["frame_dropped_filter"].x == 0.
|
||||
assert scope["publish_state"] == "fresh" and scope["prev_action"] != "stale"
|
||||
assert written["UsbGpuActive"] is True and written["UsbGpuPending"] is False
|
||||
assert scope["chestnut_state"].big is True
|
||||
|
||||
|
||||
@pytest.mark.parametrize("kwargs,expected", [
|
||||
(dict(), True),
|
||||
(dict(engaged=True), False),
|
||||
(dict(carcontrol_alive=False), False),
|
||||
(dict(carstate_alive=False), False),
|
||||
(dict(vipc_dropped_frames=1), False),
|
||||
(dict(live_calib_seen=False), False),
|
||||
])
|
||||
def test_big_model_swap_only_while_disengaged_with_fresh_state(kwargs, expected):
|
||||
defaults = dict(engaged=False, carcontrol_alive=True, carstate_alive=True,
|
||||
vipc_dropped_frames=0, live_calib_seen=True)
|
||||
assert modeld._big_model_swap_allowed(**(defaults | kwargs)) is expected
|
||||
|
||||
|
||||
def test_external_gpu_nonfinite_outputs_trigger_fallback(monkeypatch):
|
||||
class FakeTensor:
|
||||
@staticmethod
|
||||
|
||||
@@ -533,7 +533,14 @@ EVENTS: dict[int, dict[str, Alert | AlertCallbackType]] = {
|
||||
},
|
||||
|
||||
EventName.bigModelLoading: {
|
||||
ET.NO_ENTRY: NoEntryAlert("Big Model Loading"),
|
||||
ET.PERMANENT: NormalPermanentAlert("Big Model Loading",
|
||||
"Driving on the small model"),
|
||||
},
|
||||
|
||||
EventName.bigModelPending: {
|
||||
ET.PERMANENT: NormalPermanentAlert("Big Model Ready",
|
||||
"Disengage and re-engage to use it",
|
||||
duration=10.),
|
||||
},
|
||||
|
||||
EventName.bigModelFailed: {
|
||||
|
||||
@@ -255,7 +255,7 @@ class SelfdriveD:
|
||||
self.big_model_attempted = False
|
||||
self.big_model_active = False
|
||||
self.big_model_failed = False
|
||||
self.big_model_ready_t = 0.
|
||||
self.big_model_swap_t = 0.
|
||||
self.experimental_mode = False
|
||||
self.ecu_disable_failed = False
|
||||
self.ecu_disable_failed_checked = not (
|
||||
@@ -394,15 +394,23 @@ class SelfdriveD:
|
||||
loading = self.params.get_bool("UsbGpuLoading")
|
||||
if loading:
|
||||
self.big_model_attempted = True
|
||||
if self.big_model_loading and not loading:
|
||||
self.big_model_ready_t = time.monotonic()
|
||||
self.big_model_loading = loading
|
||||
if loading:
|
||||
self.events.add(EventName.bigModelLoading)
|
||||
|
||||
# The big model loads in the background while the small model drives, so it sits
|
||||
# loaded-but-unused until the driver disengages. That is a success, not a failure.
|
||||
pending = self.params.get_bool("UsbGpuPending")
|
||||
if pending:
|
||||
self.big_model_attempted = True
|
||||
self.events.add(EventName.bigModelPending)
|
||||
|
||||
big_active = self.params.get("UsbGpuActive")
|
||||
if self.big_model_active != (big_active is True):
|
||||
self.big_model_swap_t = time.monotonic()
|
||||
model_unavailable = self.big_model_active and self.sm.seen['modelV2'] and not self.sm.alive['modelV2']
|
||||
big_failed = self.big_model_attempted and not loading and (big_active is False or model_unavailable)
|
||||
big_failed = self.big_model_attempted and not loading and not pending and \
|
||||
(big_active is False or model_unavailable)
|
||||
if big_failed and not self.big_model_failed:
|
||||
self.events.add(EventName.bigModelFailed)
|
||||
self.big_model_failed = big_failed
|
||||
@@ -697,7 +705,10 @@ class SelfdriveD:
|
||||
(contains_event_type(self.events, self.starpilot_events, ET.SOFT_DISABLE) or
|
||||
contains_event_type(self.events, self.starpilot_events, ET.IMMEDIATE_DISABLE))
|
||||
no_system_errors = (not has_disable_events) or (len(self.events) == num_events)
|
||||
big_model_settling = self.big_model_loading or time.monotonic() < self.big_model_ready_t + 5.
|
||||
# modeld publishes on the small model throughout the background load, so only the model
|
||||
# swap itself can briefly disturb the stream. Suppressing for the whole load would hide
|
||||
# a genuinely dead modeld for as long as the load takes.
|
||||
big_model_settling = time.monotonic() < self.big_model_swap_t + 2.
|
||||
all_checks = self.sm.all_checks()
|
||||
all_alive = self.sm.all_alive() if not all_checks else True
|
||||
all_freq_ok = self.sm.all_freq_ok() if not all_checks else True
|
||||
|
||||
@@ -0,0 +1,60 @@
|
||||
import ast
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
from cereal import log
|
||||
from openpilot.selfdrive.selfdrived.events import EVENTS, ET
|
||||
|
||||
EventName = log.OnroadEvent.EventName
|
||||
|
||||
|
||||
def test_big_model_loading_does_not_block_engagement():
|
||||
# The big model loads in the background while the small model drives, so waiting for it
|
||||
# must never keep the driver from engaging.
|
||||
assert ET.NO_ENTRY not in EVENTS[EventName.bigModelLoading]
|
||||
assert ET.PERMANENT in EVENTS[EventName.bigModelLoading]
|
||||
|
||||
|
||||
def test_big_model_pending_is_advisory_only():
|
||||
pending = EVENTS[EventName.bigModelPending]
|
||||
assert set(pending) == {ET.PERMANENT}
|
||||
|
||||
|
||||
def test_big_model_failure_still_disengages():
|
||||
assert ET.SOFT_DISABLE in EVENTS[EventName.bigModelFailed]
|
||||
|
||||
|
||||
def _big_failed(*, attempted, loading, pending, big_active, model_unavailable=False):
|
||||
"""Mirror of selfdrived's big_failed expression, extracted from the source."""
|
||||
source = (Path(__file__).parents[1] / "selfdrived.py").read_text(encoding="utf-8")
|
||||
tree = ast.parse(source)
|
||||
assign = next(
|
||||
node for node in ast.walk(tree)
|
||||
if isinstance(node, ast.Assign)
|
||||
and any(isinstance(t, ast.Name) and t.id == "big_failed" for t in node.targets)
|
||||
)
|
||||
scope = {
|
||||
"self": type("S", (), {"big_model_attempted": attempted})(),
|
||||
"loading": loading,
|
||||
"pending": pending,
|
||||
"big_active": big_active,
|
||||
"model_unavailable": model_unavailable,
|
||||
}
|
||||
return eval(compile(ast.Expression(assign.value), "<big_failed>", "eval"), {}, scope) # noqa: S307
|
||||
|
||||
|
||||
@pytest.mark.parametrize("state,expected", [
|
||||
# A loaded model waiting for the driver to disengage is a success, not a failure.
|
||||
(dict(attempted=True, loading=False, pending=True, big_active=False), False),
|
||||
(dict(attempted=True, loading=True, pending=False, big_active=False), False),
|
||||
(dict(attempted=True, loading=False, pending=False, big_active=True), False),
|
||||
# Genuine failures must still be reported.
|
||||
(dict(attempted=True, loading=False, pending=False, big_active=False), True),
|
||||
(dict(attempted=True, loading=False, pending=False, big_active=True,
|
||||
model_unavailable=True), True),
|
||||
# Never report a failure for a big model that was never attempted.
|
||||
(dict(attempted=False, loading=False, pending=False, big_active=False), False),
|
||||
])
|
||||
def test_pending_big_model_is_not_reported_as_failed(state, expected):
|
||||
assert _big_failed(**state) is expected
|
||||
@@ -62,12 +62,17 @@ class ModelSourceWidget(LayoutWidget):
|
||||
return self.SIZE
|
||||
|
||||
@staticmethod
|
||||
def _big_model_failed(active: bool | None, usbgpu: bool, model_seen: bool, model_alive: bool) -> bool:
|
||||
def _big_model_failed(active: bool | None, usbgpu: bool, model_seen: bool, model_alive: bool,
|
||||
pending: bool = False) -> bool:
|
||||
# A model that finished loading but is waiting for the driver to disengage is not a failure.
|
||||
if pending and usbgpu:
|
||||
return False
|
||||
return active is False or not usbgpu or (active is True and model_seen and not model_alive)
|
||||
|
||||
@staticmethod
|
||||
def _status_for(loading: bool, small_model_engaged: bool, big_failed: bool) -> ModelSourceStatus:
|
||||
if loading:
|
||||
def _status_for(loading: bool, small_model_engaged: bool, big_failed: bool,
|
||||
pending: bool = False) -> ModelSourceStatus:
|
||||
if loading or pending:
|
||||
return ModelSourceStatus.LOADING
|
||||
if small_model_engaged:
|
||||
return ModelSourceStatus.FALLBACK_ENGAGED
|
||||
@@ -84,16 +89,18 @@ class ModelSourceWidget(LayoutWidget):
|
||||
model_seen = sm.recv_frame["modelV2"] > ui_state.started_frame
|
||||
model_alive = sm.alive["modelV2"] if model_seen else True
|
||||
loading = ui_state.usbgpu_loading
|
||||
big_failed = self._big_model_failed(ui_state.usbgpu_active, ui_state.usbgpu, model_seen, model_alive)
|
||||
pending = ui_state.usbgpu_pending
|
||||
big_failed = self._big_model_failed(ui_state.usbgpu_active, ui_state.usbgpu, model_seen, model_alive, pending)
|
||||
engaged = sm["selfdriveState"].enabled
|
||||
|
||||
if engaged and not self._engaged and not loading and ui_state.usbgpu_active is not True and model_seen:
|
||||
if engaged and not self._engaged and not loading and not pending \
|
||||
and ui_state.usbgpu_active is not True and model_seen:
|
||||
self._small_model_engaged = True
|
||||
if engaged != self._engaged:
|
||||
self._fade_time = rl.get_time() if engaged else 0.0
|
||||
self._engaged = engaged
|
||||
self._small_model_engaged &= big_failed
|
||||
self._status = self._status_for(loading, self._small_model_engaged, big_failed)
|
||||
self._status = self._status_for(loading, self._small_model_engaged, big_failed, pending)
|
||||
|
||||
def _render(self, rect: rl.Rectangle) -> None:
|
||||
if self._status is None:
|
||||
|
||||
@@ -344,8 +344,11 @@ class Soundd:
|
||||
if now - self.model_ready_last_check >= 0.25:
|
||||
self.model_ready_last_check = now
|
||||
try:
|
||||
# The big model now loads in the background, so it is announced when it becomes
|
||||
# usable (pending the driver's next disengage) rather than only once it is running.
|
||||
self.model_ready_pending = self.model_ready_chime.update(
|
||||
active=self.model_ready_params.get_bool("UsbGpuActive"),
|
||||
active=(self.model_ready_params.get_bool("UsbGpuPending") or
|
||||
self.model_ready_params.get_bool("UsbGpuActive")),
|
||||
loading=self.model_ready_params.get_bool("UsbGpuLoading"),
|
||||
onroad=self.model_ready_params.get_bool("IsOnroad"),
|
||||
enabled=self.model_ready_params.get_bool("GpuModelReadySound"), now=now)
|
||||
|
||||
@@ -32,6 +32,18 @@ def test_model_source_failure_detection_matches_the_backend_state_contract():
|
||||
assert not failed(True, True, False, True)
|
||||
|
||||
|
||||
def test_model_source_shows_a_pending_big_model_as_still_loading():
|
||||
# The big model loads in the background, so "loaded, waiting for a disengage" must read
|
||||
# as in-progress rather than as a failure.
|
||||
status = model_source.ModelSourceWidget._status_for
|
||||
failed = model_source.ModelSourceWidget._big_model_failed
|
||||
|
||||
assert status(False, False, False, True) is model_source.ModelSourceStatus.LOADING
|
||||
assert not failed(False, True, False, True, True)
|
||||
# Chestnut going away while pending is still a genuine failure.
|
||||
assert failed(False, False, False, True, True)
|
||||
|
||||
|
||||
def test_model_source_latches_small_model_engagement_until_the_big_model_recovers(monkeypatch):
|
||||
widget = object.__new__(model_source.ModelSourceWidget)
|
||||
widget._small_model_engaged = False
|
||||
@@ -50,6 +62,7 @@ def test_model_source_latches_small_model_engagement_until_the_big_model_recover
|
||||
usbgpu_compiled=True,
|
||||
usbgpu_active=False,
|
||||
usbgpu_loading=False,
|
||||
usbgpu_pending=False,
|
||||
),
|
||||
)
|
||||
monkeypatch.setattr(model_source.rl, "get_time", lambda: 42.0)
|
||||
|
||||
@@ -92,6 +92,7 @@ class UIState:
|
||||
self.usbgpu_compiled: bool = self.params.get_bool("UsbGpuCompiled")
|
||||
self.usbgpu_active: bool = self.params.get_bool("UsbGpuActive")
|
||||
self.usbgpu_loading: bool = self.params.get_bool("UsbGpuLoading")
|
||||
self.usbgpu_pending: bool = self.params.get_bool("UsbGpuPending")
|
||||
self.started: bool = False
|
||||
self.ignition: bool = False
|
||||
self.recording_audio: bool = False
|
||||
@@ -215,6 +216,7 @@ class UIState:
|
||||
self.usbgpu_compiled = params.get_bool("UsbGpuCompiled")
|
||||
self.usbgpu_active = params.get_bool("UsbGpuActive")
|
||||
self.usbgpu_loading = params.get_bool("UsbGpuLoading")
|
||||
self.usbgpu_pending = params.get_bool("UsbGpuPending")
|
||||
self.switchback_mode_enabled = self.params_memory.get_bool("SwitchbackModeEnabled") if self.started else False
|
||||
self.conditional_status = self.params_memory.get_int("CEStatus", default=0) if self.started else 0
|
||||
mark_progress("ui.update.after_state_params")
|
||||
|
||||
Reference in New Issue
Block a user