Use a precompiled eGPU driving model (#38930)

* Ship precompiled eGPU model and camera warps

Compile f78ed37d-afad-4dbc-8050-40ea885eedde/12864 through xx/ml_tools/openpilot_compile using the pinned tinygrad version.

* Precompile the existing master driving model

Use the unchanged master ONNX (SHA-256 6fee5937923c74848df4a63f6239eb6331c6274dd4bdb7a5d6ec0388a8b543d5) instead of updating the trained model.

* Compile camera warps on device

* Remove obsolete ONNX chunking and big model build check

* Chunk model artifacts only during release packaging

* Require model and camera warps for Chestnut readiness

* Recompile precompiled CPU helpers for the runtime host

* Ship the eGPU model with an ARM submission helper

* Exempt model pickles from the build product size limit
This commit is contained in:
Harald Schäfer
2026-09-16 08:13:45 -07:00
committed by GitHub
parent 81ae1a2e2d
commit 6080cc6168
17 changed files with 39 additions and 72 deletions
+13 -42
View File
@@ -1,18 +1,14 @@
import glob
import os
import shutil
import tempfile
import time
from SCons.Script import Action, Value
from openpilot.common.file_chunker import chunk_file, get_chunk_targets, get_existing_chunks, open_file_chunked
from openpilot.common.transformations.camera import _ar_ox_fisheye, _os_fisheye
from openpilot.common.transformations.model import MEDMODEL_INPUT_SIZE, DM_INPUT_SIZE
from openpilot.selfdrive.modeld.helpers import chestnut_present, modeld_pkl_path
from openpilot.selfdrive.modeld.helpers import chestnut_present
from openpilot.system.camerad.cameras.nv12_info import get_nv12_info
Import('env', 'arch')
chunker_file = File("#openpilot/common/file_chunker.py")
lenv = env.Clone()
lenv.PrependENVPath('PYTHONPATH', Dir('#tinygrad_repo').abspath)
@@ -20,11 +16,6 @@ tinygrad_root = env.Dir("#").abspath
tinygrad_files = ["#"+x for x in glob.glob(env.Dir("#tinygrad_repo").relpath + "/**", recursive=True, root_dir=tinygrad_root)
if 'pycache' not in x and os.path.isfile(os.path.join(tinygrad_root, x))]
def estimate_pickle_max_size(onnx_size):
# QCOM programs for models with spatial recurrent features can approach 2x
# the ONNX size. Overestimating only adds an empty trailing chunk.
return 2.0 * onnx_size + 10 * 1024 * 1024
camera_configs = [(c.width, c.height) for c in (_ar_ox_fisheye, _os_fisheye)]
if arch == 'comma_arm64':
@@ -47,7 +38,7 @@ compiler = Dir('#tinygrad_repo/examples/openpilot').abspath
# CPU 7 is isolated with isolcpus on AGNOS, so explicitly pin the compiler to it.
taskset = 'taskset -c 7 ' if arch == 'comma_arm64' else ''
def chestnut_action(command, pkl=None, chunks=()):
def chestnut_action(command):
def do_compile(target, source, env):
from openpilot.system.hardware.chestnut.flash import link_up
# chestnut can enumerate before its PCIe link is up due to varying 12V power behavior across cars
@@ -56,47 +47,27 @@ def chestnut_action(command, pkl=None, chunks=()):
break
time.sleep(1)
else:
print("Chestnut not ready, skipping big model build")
print("Chestnut not ready, skipping warp build")
return
if ret := env.Execute(command):
return ret
if chunks:
chunk_file(pkl, chunks)
return env.Execute(command)
return Action(do_compile, " [CHESTNUT] $TARGET")
def compile_model(onnx_path, pkl_path, flags, chestnut=False):
def compile_model(onnx_path, pkl_path):
onnx_path, target_pkl_path = File(onnx_path).abspath, File(pkl_path).abspath
onnx_deps = get_existing_chunks(onnx_path)
cmd = (f'{flags} {mac_brew_string} {taskset}python3 "{compiler}/compile_onnx.py" '
f'"{{onnx}}" "{target_pkl_path}" --device-input "*" --out-of-band --benchmark-runs 1')
def do_compile(target, source, env):
if os.path.isfile(onnx_path):
return env.Execute(cmd.format(onnx=onnx_path))
# TODO: Remove ONNX chunk reassembly once models are precompiled.
with tempfile.NamedTemporaryFile(dir=os.path.dirname(onnx_path), suffix='.onnx') as tmp, open_file_chunked(onnx_path) as src:
shutil.copyfileobj(src, tmp)
tmp.flush()
return env.Execute(cmd.format(onnx=tmp.name))
compile_action = Action(do_compile, " [ONNX] $TARGET")
onnx_sizes_sum = sum(os.path.getsize(f) for f in onnx_deps)
chunk_targets = get_chunk_targets(target_pkl_path, estimate_pickle_max_size(onnx_sizes_sum))
def do_chunk(target, source, env, pkl=target_pkl_path, chunks=chunk_targets):
chunk_file(pkl, chunks)
actions = chestnut_action(compile_action, target_pkl_path, chunk_targets) if chestnut else [compile_action, Action(do_chunk, " [CHUNK] $TARGET")]
node = lenv.Command(
chunk_targets,
tinygrad_files + onnx_deps + [Value(cmd), Value(chunk_targets), chunker_file],
actions,
cmd = (f'{tg_flags} {mac_brew_string} {taskset}python3 "{compiler}/compile_onnx.py" '
f'"{onnx_path}" "{target_pkl_path}" --device-input "*" --out-of-band --benchmark-runs 1')
lenv.Command(
target_pkl_path,
tinygrad_files + [onnx_path, Value(cmd)],
Action(cmd, " [ONNX] $TARGET"),
)
if chestnut:
lenv.SideEffect(chestnut_lock, node)
compile_model('models/dmonitoring_model.onnx', 'models/dmonitoring_model_tinygrad.pkl', tg_flags)
compile_model('models/dmonitoring_model.onnx', 'models/dmonitoring_model_tinygrad.pkl')
compile_model('models/driving_supercombo.onnx', 'models/driving_tinygrad.pkl')
model_w, model_h = MEDMODEL_INPUT_SIZE
for chestnut in [False, True] if CHESTNUT else [False]:
file_prefix, cmd_flags = ('big_', chestnut_tg_flags) if chestnut else ('', tg_flags)
compile_model(f'models/{file_prefix}driving_supercombo.onnx', modeld_pkl_path(chestnut), cmd_flags, chestnut)
for cam_w, cam_h in camera_configs:
warp_pkl_path = File(f"models/{file_prefix}driving_warp_{cam_w}x{cam_h}_tinygrad.pkl").abspath
stride, y_height, uv_height, _ = get_nv12_info(cam_w, cam_h)