small before big 2

This commit is contained in:
whoisdomi
2026-09-17 11:34:33 -05:00
parent b06d8baa56
commit e84eafd9d1
10 changed files with 617 additions and 75 deletions
+1
View File
@@ -132,6 +132,7 @@ struct OnroadEvent @0xc4fa6047f024e718 {
audioFeedback @97;
bigModelLoading @100;
bigModelFailed @102;
bigModelPending @103;
soundsUnavailableDEPRECATED @47;
stockLkasDEPRECATED @98;
+1
View File
@@ -177,6 +177,7 @@ inline static std::unordered_map<std::string, ParamKeyAttributes> keys = {
{"UsbGpuActive", {CLEAR_ON_MANAGER_START | CLEAR_ON_OFFROAD_TRANSITION, BOOL}},
{"UsbGpuCompiled", {CLEAR_ON_MANAGER_START | CLEAR_ON_OFFROAD_TRANSITION, BOOL}},
{"UsbGpuLoading", {CLEAR_ON_MANAGER_START | CLEAR_ON_OFFROAD_TRANSITION, BOOL}},
{"UsbGpuPending", {CLEAR_ON_MANAGER_START | CLEAR_ON_OFFROAD_TRANSITION, BOOL}},
{"UsbGpuPresent", {CLEAR_ON_MANAGER_START | CLEAR_ON_OFFROAD_TRANSITION, BOOL}},
{"Version", {PERSISTENT, STRING}},
+221 -43
View File
@@ -5,6 +5,7 @@ from functools import cached_property
import json
import os
import struct
import threading
import usb1
from openpilot.system.hardware import HARDWARE, TICI
os.environ['GMMU'] = '0'
@@ -25,7 +26,7 @@ from openpilot.common.swaglog import cloudlog
from openpilot.common.params import Params
from openpilot.common.filter_simple import FirstOrderFilter
from openpilot.common.file_chunker import file_chunked_exists, open_file_chunked
from openpilot.common.realtime import config_realtime_process, DT_MDL
from openpilot.common.realtime import config_realtime_process, set_core_affinity, DT_MDL
from openpilot.common.transformations.camera import DEVICE_CAMERAS
from openpilot.common.transformations.model import get_warp_matrix
from openpilot.system.camerad.cameras.nv12_info import get_nv12_info
@@ -105,6 +106,23 @@ EXTERNAL_GPU_POWER_WAIT_TIMEOUT_SECONDS = 60.0
EXTERNAL_GPU_POWER_LOG_INTERVAL_SECONDS = 10.0
LAT_SMOOTH_BP = [2.0, 8.0]
# The background big-model loader must leave modeld's SCHED_FIFO core, because thread
# affinity is inherited from config_realtime_process(7, 54) and the loader would otherwise
# sit behind the 20 Hz publish loop. Every other core is spoken for as well -- 4 is
# card/controlsd, 5 is plannerd/radard/selfdrived, 6 is camerad (system/camerad/main.cc),
# 7 is modeld and dmonitoringmodeld -- so the loader floats across the little cores, which
# run the non-realtime locationd/UI work, instead of pinning on top of one critical process.
BIG_MODEL_LOADER_CORES = {0, 1, 2, 3}
# A healthy load takes ~30 s once vehicle power is stable. If the AMD device never comes up
# the tinygrad calls can block indefinitely, so give up rather than reporting "loading"
# forever while the small model quietly keeps driving.
BIG_MODEL_LOAD_TIMEOUT_SECONDS = 150.0
class BigModelLoadCancelled(Exception):
"""Raised inside the background loader when modeld no longer wants the big model."""
def _set_hcq_wait_timeout(timeout_ms: int) -> None:
"""Update tinygrad's cached HCQ timeout for the external-GPU load/run phase."""
@@ -155,8 +173,10 @@ def _egmp_vehicle_ready(can_messages, bus: int) -> bool:
)
def wait_for_external_gpu_power_ready(CP=None) -> None:
def wait_for_external_gpu_power_ready(CP=None, cancel=None) -> None:
"""Wait out vehicle startup power transitions before initializing Chestnut."""
if cancel is not None and cancel.is_set():
raise BigModelLoadCancelled("cancelled before waiting for external GPU power")
device_type = HARDWARE.get_device_type()
egmp_bus = _egmp_ready_bus(CP)
services = ["pandaStates", "peripheralState"] + (["can"] if egmp_bus is not None else [])
@@ -167,6 +187,8 @@ def wait_for_external_gpu_power_ready(CP=None) -> None:
wait_started = time.monotonic()
while True:
if cancel is not None and cancel.is_set():
raise BigModelLoadCancelled("cancelled while waiting for external GPU power")
sm.update(1000)
now = time.monotonic()
if egmp_bus is not None and sm.updated["can"] and _egmp_vehicle_ready(sm["can"], egmp_bus):
@@ -854,33 +876,123 @@ def _load_model_state(cam_w: int, cam_h: int, selected_model: str, external_gpu_
)
def _load_external_gpu_model(cam_w: int, cam_h: int, selected_model: str, model_version: str = "",
CP=None, demo: bool = False) -> ModelState | None:
"""Load and warm the USB-GPU model without running another tinygrad model concurrently."""
candidate = None
try:
if not demo:
wait_for_external_gpu_power_ready(CP)
_set_hcq_wait_timeout(BIG_MODEL_LOAD_WAIT_TIMEOUT_MS)
wait_usbgpu_link()
candidate = ModelState(
cam_w,
cam_h,
True,
model_id_override=selected_model,
write_model_version=False,
model_version_override=model_version,
)
if not candidate.uses_external_gpu:
raise RuntimeError("external GPU model resolved to the builtin model")
candidate.warmup()
return candidate
except Exception:
cloudlog.exception("external GPU model load or warmup failed")
return None
finally:
_close_tinygrad_disk_cache_connection()
_set_hcq_wait_timeout(BIG_MODEL_RUN_WAIT_TIMEOUT_MS)
def _big_model_swap_allowed(engaged: bool, carcontrol_alive: bool, carstate_alive: bool,
vipc_dropped_frames: int, live_calib_seen: bool) -> bool:
"""Swapping models rebuilds the temporal input queues, so it must never happen under actuation.
Fails closed: without fresh carControl/carState we cannot trust that we are disengaged.
"""
return (
not engaged
and carcontrol_alive
and carstate_alive
and vipc_dropped_frames == 0
and live_calib_seen
)
class BigModelLoader:
"""Load and warm the Chestnut model on a background thread while the small model drives.
Everything this touches must be safe against a concurrently running small model, so unlike
the synchronous path it never calls _set_hcq_wait_timeout (hoisted to modeld startup) or
_isolate_next_model_artifact_load (which evicts buffer UOps from a tinygrad global cache).
"""
def __init__(self, cam_w: int, cam_h: int, CP=None, demo: bool = False):
self.cam_w = cam_w
self.cam_h = cam_h
self.CP = CP
self.demo = demo
self._thread: threading.Thread | None = None
self._cancel = threading.Event()
self._lock = threading.Lock()
self._result: ModelState | None = None
self._error = ""
self._model_id = ""
self._model_version = ""
self.started_t = 0.0
@property
def in_progress(self) -> bool:
return self._thread is not None and self._thread.is_alive()
@property
def timed_out(self) -> bool:
"""A load stuck inside tinygrad cannot be interrupted, so the caller gives up on it.
The thread is a daemon and keeps running, but nothing will consume its result.
"""
return self.in_progress and time.monotonic() - self.started_t > BIG_MODEL_LOAD_TIMEOUT_SECONDS
def start(self, model_id: str, model_version: str = "") -> bool:
if self.in_progress:
return False
with self._lock:
self._result = None
self._error = ""
self._model_id = model_id
self._model_version = model_version
self._cancel.clear()
self.started_t = time.monotonic()
self._thread = threading.Thread(target=self._run, name="big_model_loader", daemon=True)
self._thread.start()
return True
def cancel(self) -> None:
self._cancel.set()
def take(self) -> tuple[ModelState | None, str]:
"""Hand over the finished load exactly once: (model, error). Both empty while in flight."""
if self._thread is None or self._thread.is_alive():
return None, ""
self._thread = None
with self._lock:
result, error = self._result, self._error
self._result = None
self._error = ""
return result, error
def _run(self) -> None:
try:
# Leave modeld's realtime core before doing any work; see BIG_MODEL_LOADER_CORES.
set_core_affinity(sorted(BIG_MODEL_LOADER_CORES))
if not self.demo:
wait_for_external_gpu_power_ready(self.CP, cancel=self._cancel)
if self._cancel.is_set():
raise BigModelLoadCancelled("cancelled before the external GPU link check")
wait_usbgpu_link()
candidate = ModelState(
self.cam_w,
self.cam_h,
True,
model_id_override=self._model_id,
write_model_version=False,
model_version_override=self._model_version,
)
if not candidate.uses_external_gpu:
raise RuntimeError("external GPU model resolved to the builtin model")
# warmup() runs the camera warp on QCOM, shared with the driving small model, so it
# costs some jitter here. It still belongs on this thread: doing it at promotion
# instead blocks the publish loop for ~13 s in one stretch, which trips commIssue and
# blocks the driver from engaging exactly when the model becomes available.
candidate.warmup()
if self._cancel.is_set():
raise BigModelLoadCancelled("cancelled after warmup")
with self._lock:
self._result = candidate
cloudlog.warning(f"background big model load finished in {time.monotonic() - self.started_t:.1f}s")
except BigModelLoadCancelled as exc:
cloudlog.warning(f"background big model load cancelled: {exc}")
with self._lock:
self._error = str(exc)
except Exception as exc:
cloudlog.exception("background big model load failed")
with self._lock:
self._error = str(exc) or exc.__class__.__name__
finally:
# tinygrad's disk cache handle is thread-local, so this closes only the loader's.
_close_tinygrad_disk_cache_connection()
def _model_versions() -> dict[str, str]:
@@ -1058,9 +1170,16 @@ def main(demo=False):
external_artifact = MODELS_PATH / f"{big_model_id}_driving_tinygrad.pkl"
external_artifact_ready = external_model_selected and file_chunked_exists(external_artifact)
external_gpu_requested = usbgpu_present_now and (bool(big_model_id) or model_lab_ready)
# The big model is loaded in the background while the small model drives, so the HCQ
# watchdog is set once here and never mutated again: it is process-global and shared by
# the QCOM and AMD devices, so changing it mid-drive would also move the small model's.
if external_gpu_requested:
_set_hcq_wait_timeout(BIG_MODEL_LOAD_WAIT_TIMEOUT_MS)
params.put_bool("UsbGpuPresent", usbgpu_present_now)
params.put_bool("UsbGpuCompiled", external_artifact_ready or model_lab_ready)
params.put_bool("UsbGpuActive", False)
params.put_bool("UsbGpuPending", False)
params.put_bool("UsbGpuLoading", external_gpu_requested)
_set_model_lab_runtime(
params,
@@ -1147,15 +1266,8 @@ def main(demo=False):
else:
CP = messaging.log_from_bytes(params.get("CarParams", block=True), car.CarParams)
big_model = _load_external_gpu_model(
vipc_client_main.width,
vipc_client_main.height,
selected_model,
selected_model_version,
CP,
demo,
)
# Start on the small model so openpilot is drivable immediately; the big model is loaded
# in the background below and promoted once the driver disengages.
small_model = _load_model_state(
vipc_client_main.width,
vipc_client_main.height,
@@ -1165,10 +1277,7 @@ def main(demo=False):
small_model_version,
False,
)
model = big_model if big_model is not None else small_model
if big_model is not None:
params.put("ModelVersion", model.policy_generation)
params.put("DrivingModelVersion", model.policy_generation)
model = small_model
else:
model = _load_model_state(
vipc_client_main.width,
@@ -1183,9 +1292,15 @@ def main(demo=False):
set_runtime_model_params(params, model.model_id, model.policy_generation)
external_gpu_active = model_lab_active or model.uses_external_gpu
# Load the big model in the background so the small model can drive in the meantime.
big_loader = None
if external_gpu_requested and not model_lab_active and big_model_id:
big_loader = BigModelLoader(vipc_client_main.width, vipc_client_main.height, CP, demo)
big_loader.start(big_model_id, big_model_version)
params.put_bool("UsbGpuCompiled", external_artifact_ready or model_lab_ready)
params.put_bool("UsbGpuActive", external_gpu_active)
params.put_bool("UsbGpuLoading", False)
params.put_bool("UsbGpuPending", False)
params.put_bool("UsbGpuLoading", big_loader is not None)
_set_model_lab_runtime(
params,
requested=model_lab_requested,
@@ -1340,6 +1455,61 @@ def main(demo=False):
chestnut_state.big = False
cloudlog.error(f"Model Laboratory stopped: {model_lab_error}")
# Collect a finished background big-model load, then promote it once the driver is
# disengaged. Swapping rebuilds the temporal input queues, so it must not happen
# under actuation.
if big_loader is not None and big_loader.timed_out:
big_loader.cancel()
big_loader = None
params.put_bool("UsbGpuLoading", False)
params.put_bool("UsbGpuPending", False)
params.put_bool("UsbGpuActive", False)
cloudlog.error(f"big model load exceeded {BIG_MODEL_LOAD_TIMEOUT_SECONDS:.0f}s, "
"staying on the small model")
elif big_loader is not None and not big_loader.in_progress:
loaded_big_model, big_load_error = big_loader.take()
big_loader = None
if loaded_big_model is not None:
big_model = loaded_big_model
# Raise pending before clearing loading: selfdrived polls these separately at 100 Hz
# and reads "neither loading nor pending" as a failed load.
params.put_bool("UsbGpuPending", True)
params.put_bool("UsbGpuLoading", False)
cloudlog.warning("big model ready; waiting for the driver to disengage before using it")
else:
params.put_bool("UsbGpuPending", False)
params.put_bool("UsbGpuActive", False)
params.put_bool("UsbGpuLoading", False)
cloudlog.error(f"big model unavailable, staying on the small model: {big_load_error}")
if big_model is not None and model is not big_model and _big_model_swap_allowed(
sm["carControl"].enabled,
sm.alive["carControl"],
sm.alive["carState"],
vipc_dropped_frames,
live_calib_seen,
):
# The loader already warmed this model, so the swap itself is just a pointer change
# plus a queue reset; it must stay cheap because it runs inside the publish loop.
big_model._reset_state()
model = big_model
external_gpu_active = True
# Nothing from the small model carries over: re-arm the frame-drop warmup and drop
# the rolling probability buffers and previous action, which are model specific.
run_count = 0
frame_dropped_filter.x = 0.
publish_state = PublishState()
prev_action = log.ModelDataV2.Action()
params.put("ModelVersion", model.policy_generation)
params.put("DrivingModelVersion", model.policy_generation)
set_runtime_model_params(params, model.model_id, model.policy_generation)
params.put_bool("UsbGpuActive", True)
params.put_bool("UsbGpuPending", False)
params.put_bool("UsbGpuLoading", False)
if chestnut_state is not None:
chestnut_state.big = True
cloudlog.warning(f"now driving on the big model {model.model_id}")
frame_drop_ratio = frames_dropped / (1 + frames_dropped)
dropped_frame = vipc_dropped_frames > 0
if dropped_frame and (model.can_prepare_only or (model_lab_longitudinal is not None and model_lab_longitudinal.can_prepare_only)):
@@ -1357,6 +1527,9 @@ def main(demo=False):
try:
send_chestnut = (
chestnut_state is not None and
# Telemetry shares the USB device with the model transfer, so polling it while a
# background load is in flight stalls that transfer until it times out.
(big_loader is None or not big_loader.in_progress) and
run_count % round(ModelConstants.MODEL_FREQ / SERVICE_LIST["chestnutState"].frequency) == 0
)
if model_lab_longitudinal is not None:
@@ -1433,11 +1606,16 @@ def main(demo=False):
cloudlog.exception("external GPU model failed, falling back to active small model")
model = small_model
big_model = None
# A failed big model is not retried in this drive, so stop any load still in flight.
if big_loader is not None:
big_loader.cancel()
big_loader = None
params.put_bool("UsbGpuActive", False)
external_gpu_active = False
params.put("ModelVersion", model.policy_generation)
params.put("DrivingModelVersion", model.policy_generation)
set_runtime_model_params(params, model.model_id, model.policy_generation)
params.put_bool("UsbGpuPending", False)
params.put_bool("UsbGpuLoading", False)
if chestnut_state is not None:
chestnut_state.big = False
+321 -16
View File
@@ -1,5 +1,6 @@
import io
import struct
import threading
from types import MethodType
from types import SimpleNamespace
@@ -269,45 +270,349 @@ def test_tinygrad_empty_thread_local_cache_holder_is_safe(monkeypatch):
assert tinygrad_helpers._db_connection is holder
def test_external_gpu_load_finishes_before_native_model_can_start(monkeypatch):
calls = []
def _stub_big_model_loader(monkeypatch, calls, *, uses_external_gpu=True, model_error=None):
class FakeModelState:
uses_external_gpu = True
def __init__(self, cam_w, cam_h, external_gpu_active, model_id_override, write_model_version,
model_version_override):
calls.append(("model", cam_w, cam_h, external_gpu_active, model_id_override,
write_model_version, model_version_override))
if model_error is not None:
raise model_error
self.uses_external_gpu = uses_external_gpu
def warmup(self):
calls.append("warmup")
monkeypatch.setattr(modeld, "set_core_affinity", lambda cores: calls.append(("affinity", tuple(cores))))
monkeypatch.setattr(modeld, "wait_usbgpu_link", lambda: calls.append("link"))
monkeypatch.setattr(modeld, "wait_for_external_gpu_power_ready", lambda CP: calls.append(("power", CP)))
monkeypatch.setattr(modeld, "_set_hcq_wait_timeout", lambda timeout: calls.append(("timeout", timeout)))
monkeypatch.setattr(modeld, "wait_for_external_gpu_power_ready",
lambda CP, cancel=None: calls.append(("power", CP)))
monkeypatch.setattr(modeld, "_close_tinygrad_disk_cache_connection", lambda: calls.append("close_cache"))
monkeypatch.setattr(modeld, "ModelState", FakeModelState)
monkeypatch.setattr(
modeld,
"tinygrad_dev_config",
lambda *_args: (_ for _ in ()).throw(AssertionError("runtime must not change tinygrad's process-global DEV")),
)
return FakeModelState
loaded = modeld._load_external_gpu_model(1928, 1208, "big-model", "v15", "car-params")
assert isinstance(loaded, FakeModelState)
def test_background_big_model_load_leaves_the_running_model_untouched(monkeypatch):
calls = []
fake_model_state = _stub_big_model_loader(monkeypatch, calls)
# The small model is already driving on these process-global settings, so the background
# load must not touch tinygrad's DEV, its HCQ watchdog, or its shared buffer UOp cache.
for name, detail in (
("tinygrad_dev_config", "runtime must not change tinygrad's process-global DEV"),
("_set_hcq_wait_timeout", "the background load must not move the running model's HCQ watchdog"),
("_isolate_next_model_artifact_load", "the background load must not evict the running model's buffers"),
):
monkeypatch.setattr(modeld, name, lambda *_args, _d=detail: (_ for _ in ()).throw(AssertionError(_d)))
loader = modeld.BigModelLoader(1928, 1208, "car-params")
assert loader.start("big-model", "v15")
loader._thread.join(timeout=10)
loaded, error = loader.take()
assert isinstance(loaded, fake_model_state)
assert error == ""
# warmup() belongs here rather than at promotion: see
# test_big_model_warmup_stays_on_the_background_loader.
assert calls == [
("affinity", tuple(sorted(modeld.BIG_MODEL_LOADER_CORES))),
("power", "car-params"),
("timeout", modeld.BIG_MODEL_LOAD_WAIT_TIMEOUT_MS),
"link",
("model", 1928, 1208, True, "big-model", False, "v15"),
"warmup",
"close_cache",
("timeout", modeld.BIG_MODEL_RUN_WAIT_TIMEOUT_MS),
]
def test_big_model_load_gives_up_instead_of_loading_forever(monkeypatch):
# A load stuck inside tinygrad cannot be interrupted, so the timeout is what stops modeld
# reporting "loading" for the rest of the drive while the small model quietly drives.
release = threading.Event()
calls = []
class HangingModelState:
uses_external_gpu = True
def __init__(self, *_a, **_k):
release.wait(timeout=10)
def warmup(self):
pass
monkeypatch.setattr(modeld, "set_core_affinity", lambda cores: None)
monkeypatch.setattr(modeld, "wait_usbgpu_link", lambda: None)
monkeypatch.setattr(modeld, "wait_for_external_gpu_power_ready", lambda CP, cancel=None: None)
monkeypatch.setattr(modeld, "_close_tinygrad_disk_cache_connection", lambda: calls.append("close"))
monkeypatch.setattr(modeld, "ModelState", HangingModelState)
loader = modeld.BigModelLoader(1928, 1208, "car-params")
loader.start("big-model", "v15")
try:
assert loader.in_progress
assert not loader.timed_out
# Pretend the load has been running well past its budget.
loader.started_t -= modeld.BIG_MODEL_LOAD_TIMEOUT_SECONDS + 1
assert loader.timed_out
# take() must not hand over a model from a load we already gave up on.
assert loader.take() == (None, "")
finally:
release.set()
loader._thread.join(timeout=10)
def test_big_model_load_timeout_leaves_room_for_a_normal_load():
# A healthy load measured ~26 s on device; the budget must not cut those off.
assert modeld.BIG_MODEL_LOAD_TIMEOUT_SECONDS > 60
def test_big_model_warmup_stays_on_the_background_loader():
"""Warming at promotion blocks the publish loop in one long stretch; warming on the
loader spreads the same work out while the small model is still driving.
Measured with warmup at promotion: modelV2 went silent for 13.0 s and 12.8 s on two
drives, starting the instant the model was collected, which tripped commIssue and blocked
the driver from engaging. Measured with warmup on the loader: worst in-load gap was
1.3-2.9 s, spread across the load. Neither is free, but only the second one leaves
modeld publishing when the driver wants to engage.
"""
import ast
from pathlib import Path
source = (Path(modeld.__file__).with_name("modeld.py")).read_text(encoding="utf-8")
tree = ast.parse(source)
loader = next(n for n in ast.walk(tree)
if isinstance(n, ast.ClassDef) and n.name == "BigModelLoader")
assert "warmup" in ast.dump(loader), "the loader thread must warm the big model"
main_fn = next(n for n in tree.body if isinstance(n, ast.FunctionDef) and n.name == "main")
promote = next(n for n in ast.walk(main_fn)
if isinstance(n, ast.If) and "_big_model_swap_allowed" in ast.dump(n.test))
assert "warmup" not in ast.dump(promote), \
"promotion must not warm the big model; it runs inside the 20 Hz publish loop"
def test_chestnut_telemetry_is_suppressed_while_a_background_load_runs():
"""Telemetry shares Chestnut's USB device with the model weight transfer.
ChestnutState._read_ina() issues USB control reads on the same device tinygrad streams
weights over. The old code loaded before the publish loop existed so the two never
overlapped; loading in the background makes them concurrent, which stalls the transfer
until it times out. Captured on four drives: the load never completed while telemetry
was polling at 10 Hz, and chestnutState first appeared only after the load on the one
drive that succeeded.
"""
import ast
from pathlib import Path
source = (Path(modeld.__file__).with_name("modeld.py")).read_text(encoding="utf-8")
main_fn = next(n for n in ast.parse(source).body
if isinstance(n, ast.FunctionDef) and n.name == "main")
assign = next(
node for node in ast.walk(main_fn)
if isinstance(node, ast.Assign)
and any(isinstance(t, ast.Name) and t.id == "send_chestnut" for t in node.targets)
)
guard = ast.dump(assign.value)
assert "big_loader" in guard and "in_progress" in guard, \
"chestnutState polling must be gated on the background loader being idle"
def test_background_big_model_loader_avoids_the_realtime_cores():
"""The loader must not share a core with modeld or the camera pipeline.
modeld is SCHED_FIFO on core 7 and threads inherit its affinity, so a loader left there is
starved behind the 20 Hz publish loop. Putting it on camerad's core instead stalls frame
delivery: measured p95 on roadCameraState went 50.8 ms -> 108.9 ms for the whole load.
"""
assert modeld.BIG_MODEL_LOADER_CORES
reserved = {
7: "modeld (config_realtime_process(7, 54)) and dmonitoringmodeld",
6: "camerad (system/camerad/main.cc set_core_affinity({6}))",
5: "plannerd, radard, selfdrived, starpilot_process",
4: "card and controlsd",
}
for core, owner in reserved.items():
assert core not in modeld.BIG_MODEL_LOADER_CORES, f"core {core} belongs to {owner}"
# camerad pins itself in C++, which is easy to miss when auditing Python callers.
import re
from pathlib import Path
camerad_main = Path(modeld.__file__).parents[2] / "system" / "camerad" / "main.cc"
pinned = re.search(r"set_core_affinity\(\{([0-9,\s]+)\}\)", camerad_main.read_text(encoding="utf-8"))
assert pinned, "could not find camerad's core affinity"
camerad_cores = {int(c) for c in pinned.group(1).split(",") if c.strip()}
assert not (camerad_cores & modeld.BIG_MODEL_LOADER_CORES), \
f"the loader must not share camerad's cores {sorted(camerad_cores)}"
@pytest.mark.parametrize("failure", [
dict(model_error=RuntimeError("artifact is corrupt")),
dict(uses_external_gpu=False),
])
def test_background_big_model_load_reports_failure_without_a_model(monkeypatch, failure):
calls = []
_stub_big_model_loader(monkeypatch, calls, **failure)
loader = modeld.BigModelLoader(1928, 1208, "car-params")
loader.start("big-model", "v15")
loader._thread.join(timeout=10)
loaded, error = loader.take()
assert loaded is None
assert error
assert not loader.in_progress
# Even a failed load must release the loader thread's own tinygrad cache handle.
assert "close_cache" in calls
def test_background_big_model_load_is_handed_over_exactly_once(monkeypatch):
calls = []
fake_model_state = _stub_big_model_loader(monkeypatch, calls)
loader = modeld.BigModelLoader(1928, 1208, "car-params")
loader.start("big-model", "v15")
loader._thread.join(timeout=10)
assert isinstance(loader.take()[0], fake_model_state)
# A second take must not promote the same model again.
assert loader.take() == (None, "")
def test_cancelled_big_model_load_never_builds_a_model(monkeypatch):
calls = []
def cancelling_power_wait(CP, cancel=None):
calls.append(("power", CP))
cancel.set()
raise modeld.BigModelLoadCancelled("cancelled while waiting for external GPU power")
_stub_big_model_loader(monkeypatch, calls)
monkeypatch.setattr(modeld, "wait_for_external_gpu_power_ready", cancelling_power_wait)
loader = modeld.BigModelLoader(1928, 1208, "car-params")
loader.start("big-model", "v15")
loader._thread.join(timeout=10)
loaded, error = loader.take()
assert loaded is None
assert error
assert not any(call[0] == "model" for call in calls if isinstance(call, tuple))
def test_external_gpu_power_wait_aborts_when_the_loader_is_cancelled(monkeypatch):
cancel = threading.Event()
cancel.set()
monkeypatch.setattr(modeld, "SubMaster",
lambda *_a, **_k: pytest.fail("a cancelled load must not start waiting on power"))
with pytest.raises(modeld.BigModelLoadCancelled):
modeld.wait_for_external_gpu_power_ready("car-params", cancel=cancel)
def test_big_model_promotion_waits_for_the_driver_to_disengage(monkeypatch):
"""Drive the real collect-and-promote blocks lifted out of main()'s loop."""
import ast
from pathlib import Path
source = (Path(modeld.__file__).with_name("modeld.py")).read_text(encoding="utf-8")
main_fn = next(n for n in ast.parse(source).body if isinstance(n, ast.FunctionDef) and n.name == "main")
def names(node):
return {n.id for n in ast.walk(node) if isinstance(n, ast.Name)}
collect_block = next(n for n in ast.walk(main_fn)
if isinstance(n, ast.If) and "big_loader" in names(n.test)
and "in_progress" in ast.dump(n.test))
promote_block = next(n for n in ast.walk(main_fn)
if isinstance(n, ast.If) and "_big_model_swap_allowed" in names(n.test))
promote = compile(ast.Module(body=[collect_block, promote_block], type_ignores=[]),
"<promotion>", "exec")
class FakeModel:
model_id = "big-model"
policy_generation = "v15"
def __init__(self):
self.reset = 0
def warmup(self):
# The real warmup() ends with _reset_state(); it runs at promotion, not on the loader.
self.reset += 1
def _reset_state(self):
self.reset += 1
class FakeLoader:
def __init__(self, model):
self.in_progress = True
self._model = model
def take(self):
return self._model, ""
written = {}
big = FakeModel()
loader = FakeLoader(big)
small = FakeModel()
small.model_id = "small-model"
scope = {
"big_loader": loader, "big_model": None, "model": small, "small_model": small,
"params": SimpleNamespace(put_bool=lambda k, v: written.__setitem__(k, v),
put=lambda k, v: written.__setitem__(k, v)),
"cloudlog": SimpleNamespace(warning=lambda *_a: None, error=lambda *_a: None),
"vipc_dropped_frames": 0, "live_calib_seen": True, "external_gpu_active": False,
"run_count": 99, "frame_dropped_filter": SimpleNamespace(x=5.0),
"PublishState": lambda: "fresh", "publish_state": "stale",
"prev_action": "stale", "log": modeld.log, "chestnut_state": SimpleNamespace(big=False),
"set_runtime_model_params": lambda *_a: None,
"_big_model_swap_allowed": modeld._big_model_swap_allowed,
}
scope["sm"] = type("SM", (), {
"__getitem__": lambda _s, k: SimpleNamespace(enabled=scope["_engaged"]),
"alive": {"carControl": True, "carState": True},
})()
# Still loading: nothing is collected and the small model keeps driving.
scope["_engaged"] = True
exec(promote, scope) # noqa: S102
assert scope["model"] is small and scope["big_model"] is None
# Load finishes while engaged: the model is collected but must NOT be promoted.
loader.in_progress = False
exec(promote, scope) # noqa: S102
assert scope["big_model"] is big
assert scope["model"] is small, "must not swap models while the driver is engaged"
assert written["UsbGpuPending"] is True
assert written["UsbGpuLoading"] is False
assert big.reset == 0
# The driver disengages: now the big model takes over.
scope["_engaged"] = False
exec(promote, scope) # noqa: S102
assert scope["model"] is big
assert big.reset == 1, "the big model must be warmed and reset before it drives"
assert scope["run_count"] == 0 and scope["frame_dropped_filter"].x == 0.
assert scope["publish_state"] == "fresh" and scope["prev_action"] != "stale"
assert written["UsbGpuActive"] is True and written["UsbGpuPending"] is False
assert scope["chestnut_state"].big is True
@pytest.mark.parametrize("kwargs,expected", [
(dict(), True),
(dict(engaged=True), False),
(dict(carcontrol_alive=False), False),
(dict(carstate_alive=False), False),
(dict(vipc_dropped_frames=1), False),
(dict(live_calib_seen=False), False),
])
def test_big_model_swap_only_while_disengaged_with_fresh_state(kwargs, expected):
defaults = dict(engaged=False, carcontrol_alive=True, carstate_alive=True,
vipc_dropped_frames=0, live_calib_seen=True)
assert modeld._big_model_swap_allowed(**(defaults | kwargs)) is expected
def test_external_gpu_nonfinite_outputs_trigger_fallback(monkeypatch):
class FakeTensor:
@staticmethod
+8 -3
View File
@@ -532,9 +532,14 @@ EVENTS: dict[int, dict[str, Alert | AlertCallbackType]] = {
"Ensure road ahead is clear"),
},
EventName.bigModelLoading: {
ET.NO_ENTRY: NoEntryAlert("Big Model Loading"),
},
# bigModelLoading and bigModelPending carry no alerts on purpose: the blinking eGPU icon
# already says the big model is loading, and its absence says it is ready.
EventName.bigModelLoading: {},
# bigModelPending carries no alert on purpose: the eGPU icon clearing, and then turning
# green on the next engage, is the driver's cue. A banner here announced the model before
# it was usable.
EventName.bigModelPending: {},
EventName.bigModelFailed: {
ET.SOFT_DISABLE: soft_disable_alert("Big Model Failed"),
+30 -6
View File
@@ -255,7 +255,7 @@ class SelfdriveD:
self.big_model_attempted = False
self.big_model_active = False
self.big_model_failed = False
self.big_model_ready_t = 0.
self.big_model_swap_t = 0.
self.experimental_mode = False
self.ecu_disable_failed = False
self.ecu_disable_failed_checked = not (
@@ -394,15 +394,25 @@ class SelfdriveD:
loading = self.params.get_bool("UsbGpuLoading")
if loading:
self.big_model_attempted = True
if self.big_model_loading and not loading:
self.big_model_ready_t = time.monotonic()
self.big_model_loading = loading
# No alert while loading: the blinking eGPU icon already carries that state, and the
# small model is driving normally. big_model_loading still gates the frame-drop warning.
if loading:
self.events.add(EventName.bigModelLoading)
# The big model loads in the background while the small model drives, so it sits
# loaded-but-unused until the driver disengages. That is a success, not a failure, and
# it raises no alert: the driver sees it through the eGPU icon, not a banner.
pending = self.params.get_bool("UsbGpuPending")
if pending:
self.big_model_attempted = True
big_active = self.params.get("UsbGpuActive")
if self.big_model_active != (big_active is True):
self.big_model_swap_t = time.monotonic()
model_unavailable = self.big_model_active and self.sm.seen['modelV2'] and not self.sm.alive['modelV2']
big_failed = self.big_model_attempted and not loading and (big_active is False or model_unavailable)
big_failed = self.big_model_attempted and not loading and not pending and \
(big_active is False or model_unavailable)
if big_failed and not self.big_model_failed:
self.events.add(EventName.bigModelFailed)
self.big_model_failed = big_failed
@@ -697,10 +707,21 @@ class SelfdriveD:
(contains_event_type(self.events, self.starpilot_events, ET.SOFT_DISABLE) or
contains_event_type(self.events, self.starpilot_events, ET.IMMEDIATE_DISABLE))
no_system_errors = (not has_disable_events) or (len(self.events) == num_events)
big_model_settling = self.big_model_loading or time.monotonic() < self.big_model_ready_t + 5.
# The model swap briefly disturbs the stream, so forgive everything around it.
big_model_settling = time.monotonic() < self.big_model_swap_t + 2.
all_checks = self.sm.all_checks()
all_alive = self.sm.all_alive() if not all_checks else True
all_freq_ok = self.sm.all_freq_ok() if not all_checks else True
# Loading the big model realizes its graph on the same QCOM GPU the small model drives
# on, which costs modeld frames for part of the load and drags modelV2 below its target
# rate. commIssueAvgFreq is NO_ENTRY, so that would block engaging on a small model that
# is working fine. Forgive only a slow modelV2, and only while a load is in flight --
# anything dying, and any other service, still reports normally.
if self.big_model_loading and not all_checks and all_alive and not all_freq_ok:
slow = {s for s, freq_ok in self.sm.freq_ok.items() if not freq_ok}
if slow and slow <= {'modelV2', 'drivingModelData', 'cameraOdometry'} and self.sm.all_valid():
all_freq_ok = True
all_checks = True
report_comm_issue, self.valid_only_comm_issue_frames = evaluate_comm_issue(
all_checks, all_alive, all_freq_ok, self.valid_only_comm_issue_frames,
)
@@ -798,7 +819,10 @@ class SelfdriveD:
# TODO: fix simulator
if not SIMULATION or REPLAY:
if self.sm['modelV2'].frameDropPerc > 20:
# Loading the big model realizes its graph on the same QCOM GPU the small model is
# driving on, which costs real frames for part of the load. The small model keeps
# publishing throughout, so warn about the drops only once the load is out of the way.
if self.sm['modelV2'].frameDropPerc > 20 and not self.big_model_loading:
self.events.add(EventName.modeldLagging)
# Decrement personality on configured steering-wheel button presses
@@ -62,12 +62,19 @@ class ModelSourceWidget(LayoutWidget):
return self.SIZE
@staticmethod
def _big_model_failed(active: bool | None, usbgpu: bool, model_seen: bool, model_alive: bool) -> bool:
def _big_model_failed(active: bool | None, usbgpu: bool, model_seen: bool, model_alive: bool,
pending: bool = False) -> bool:
# A model that finished loading but is waiting for the driver to disengage is not a failure.
if pending and usbgpu:
return False
return active is False or not usbgpu or (active is True and model_seen and not model_alive)
@staticmethod
def _status_for(loading: bool, small_model_engaged: bool, big_failed: bool) -> ModelSourceStatus:
if loading:
def _status_for(loading: bool, small_model_engaged: bool, big_failed: bool,
pending: bool = False) -> ModelSourceStatus:
# Pending means the big model finished loading and is waiting for the next disengage,
# so the blinking loading icon stops: its absence is what tells the driver it is ready.
if loading and not pending:
return ModelSourceStatus.LOADING
if small_model_engaged:
return ModelSourceStatus.FALLBACK_ENGAGED
@@ -84,16 +91,18 @@ class ModelSourceWidget(LayoutWidget):
model_seen = sm.recv_frame["modelV2"] > ui_state.started_frame
model_alive = sm.alive["modelV2"] if model_seen else True
loading = ui_state.usbgpu_loading
big_failed = self._big_model_failed(ui_state.usbgpu_active, ui_state.usbgpu, model_seen, model_alive)
pending = ui_state.usbgpu_pending
big_failed = self._big_model_failed(ui_state.usbgpu_active, ui_state.usbgpu, model_seen, model_alive, pending)
engaged = sm["selfdriveState"].enabled
if engaged and not self._engaged and not loading and ui_state.usbgpu_active is not True and model_seen:
if engaged and not self._engaged and not loading and not pending \
and ui_state.usbgpu_active is not True and model_seen:
self._small_model_engaged = True
if engaged != self._engaged:
self._fade_time = rl.get_time() if engaged else 0.0
self._engaged = engaged
self._small_model_engaged &= big_failed
self._status = self._status_for(loading, self._small_model_engaged, big_failed)
self._status = self._status_for(loading, self._small_model_engaged, big_failed, pending)
def _render(self, rect: rl.Rectangle) -> None:
if self._status is None:
+4 -1
View File
@@ -344,8 +344,11 @@ class Soundd:
if now - self.model_ready_last_check >= 0.25:
self.model_ready_last_check = now
try:
# The big model now loads in the background, so it is announced when it becomes
# usable (pending the driver's next disengage) rather than only once it is running.
self.model_ready_pending = self.model_ready_chime.update(
active=self.model_ready_params.get_bool("UsbGpuActive"),
active=(self.model_ready_params.get_bool("UsbGpuPending") or
self.model_ready_params.get_bool("UsbGpuActive")),
loading=self.model_ready_params.get_bool("UsbGpuLoading"),
onroad=self.model_ready_params.get_bool("IsOnroad"),
enabled=self.model_ready_params.get_bool("GpuModelReadySound"), now=now)
@@ -32,6 +32,19 @@ def test_model_source_failure_detection_matches_the_backend_state_contract():
assert not failed(True, True, False, True)
def test_model_source_blinks_while_loading_then_clears_once_the_big_model_is_ready():
# The blinking icon is the driver's "still loading" cue, and its disappearance is how they
# know the next disengage/engage will pick up the big model.
status = model_source.ModelSourceWidget._status_for
failed = model_source.ModelSourceWidget._big_model_failed
assert status(True, False, False, False) is model_source.ModelSourceStatus.LOADING
assert status(True, False, False, True) is not model_source.ModelSourceStatus.LOADING
assert not failed(False, True, False, True, True)
# Chestnut going away while pending is still a genuine failure.
assert failed(False, False, False, True, True)
def test_model_source_latches_small_model_engagement_until_the_big_model_recovers(monkeypatch):
widget = object.__new__(model_source.ModelSourceWidget)
widget._small_model_engaged = False
@@ -50,6 +63,7 @@ def test_model_source_latches_small_model_engagement_until_the_big_model_recover
usbgpu_compiled=True,
usbgpu_active=False,
usbgpu_loading=False,
usbgpu_pending=False,
),
)
monkeypatch.setattr(model_source.rl, "get_time", lambda: 42.0)
+2
View File
@@ -97,6 +97,7 @@ class UIState:
self.usbgpu_compiled: bool = self.params.get_bool("UsbGpuCompiled")
self.usbgpu_active: bool = self.params.get_bool("UsbGpuActive")
self.usbgpu_loading: bool = self.params.get_bool("UsbGpuLoading")
self.usbgpu_pending: bool = self.params.get_bool("UsbGpuPending")
self.started: bool = False
self.ignition: bool = False
self.recording_audio: bool = False
@@ -220,6 +221,7 @@ class UIState:
self.usbgpu_compiled = params.get_bool("UsbGpuCompiled")
self.usbgpu_active = params.get_bool("UsbGpuActive")
self.usbgpu_loading = params.get_bool("UsbGpuLoading")
self.usbgpu_pending = params.get_bool("UsbGpuPending")
self.switchback_mode_enabled = self.params_memory.get_bool("SwitchbackModeEnabled") if self.started else False
self.conditional_status = self.params_memory.get_int("CEStatus", default=0) if self.started else 0
mark_progress("ui.update.after_state_params")