From 69ea9b606355aaeba3318aa8af62d30bc44a6954 Mon Sep 17 00:00:00 2001 From: "github-actions[bot]" Date: Tue, 4 Aug 2026 00:45:14 +0000 Subject: [PATCH] =?UTF-8?q?Deep=20RL=20-=20New=20And=20Improved=20-=20Cour?= =?UTF-8?q?tsey=20Chubbs=20=E2=9D=A4=EF=B8=8F=20(PR-1887)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .../build-single-tinygrad-model.yaml | 13 +- .github/workflows/sunnypilot-build-model.yaml | 49 +- openpilot/cereal/custom.capnp | 7 + .../sunnypilot/modeld_v2/compile_modeld.py | 596 ++++++------------ openpilot/sunnypilot/modeld_v2/modeld.py | 89 ++- .../modeld_v2/parse_model_outputs.py | 68 +- .../modeld_v2/parse_model_outputs_split.py | 6 +- .../tests/test_combined_pkl_loader.py | 18 +- .../sunnypilot/modeld_v2/tests/test_warp.py | 103 --- openpilot/sunnypilot/modeld_v2/warp.py | 171 ----- openpilot/sunnypilot/models/fetcher.py | 34 +- openpilot/sunnypilot/models/helpers.py | 23 +- openpilot/sunnypilot/models/manager.py | 92 +-- .../sunnypilot/models/runners/helpers.py | 28 - .../sunnypilot/models/runners/model_runner.py | 174 ----- .../models/runners/tinygrad/model_types.py | 91 --- .../runners/tinygrad/tinygrad_runner.py | 179 ------ .../models/split_model_constants.py | 1 + release/ci/model_generator.py | 147 ++--- tinygrad_repo | 2 +- 20 files changed, 506 insertions(+), 1385 deletions(-) delete mode 100644 openpilot/sunnypilot/modeld_v2/tests/test_warp.py delete mode 100644 openpilot/sunnypilot/modeld_v2/warp.py delete mode 100644 openpilot/sunnypilot/models/runners/helpers.py delete mode 100644 openpilot/sunnypilot/models/runners/model_runner.py delete mode 100644 openpilot/sunnypilot/models/runners/tinygrad/model_types.py delete mode 100644 openpilot/sunnypilot/models/runners/tinygrad/tinygrad_runner.py diff --git a/.github/workflows/build-single-tinygrad-model.yaml b/.github/workflows/build-single-tinygrad-model.yaml index f10f1b71a3..cf2d870802 100644 --- a/.github/workflows/build-single-tinygrad-model.yaml +++ b/.github/workflows/build-single-tinygrad-model.yaml @@ -12,11 +12,11 @@ on: required: false type: string recompiled_dir: - description: 'Existing recompiled directory number (e.g. 3 for recompiled3)' + description: 'Existing recompiled directory number (e.g. 1 for recompiled1)' required: true type: string json_version: - description: 'driving_models version number to update (e.g. 5 for driving_models_v5.json)' + description: 'driving_models version number to update (e.g. 18 for driving_models_v18.json)' required: true type: string artifact_suffix: @@ -63,12 +63,11 @@ on: default: 'None' options: - None - - Simple Plan Models - - Space Lab Models - - TR Models - - DTR Models + - Master Models + - Release Models + - 2025 World Models + - 2026 World Models - Custom Merge Models - - FOF series models - Other custom_model_folder: description: 'Custom model folder name (if "Other" selected)' diff --git a/.github/workflows/sunnypilot-build-model.yaml b/.github/workflows/sunnypilot-build-model.yaml index acb75af55e..2c435e58a0 100644 --- a/.github/workflows/sunnypilot-build-model.yaml +++ b/.github/workflows/sunnypilot-build-model.yaml @@ -30,6 +30,11 @@ on: required: false type: string default: '' + target_hardware: + description: 'Hardware target to compile for (qcom or usbgpu)' + required: false + type: string + default: 'qcom' workflow_dispatch: inputs: upstream_branch: @@ -46,6 +51,14 @@ on: required: false type: boolean default: true + target_hardware: + description: 'Hardware target to compile for' + required: true + type: choice + options: + - qcom + - usbgpu + default: 'qcom' run-name: Build model [${{ inputs.custom_name || inputs.upstream_branch }}] from ref [${{ inputs.upstream_branch }}] @@ -161,19 +174,30 @@ jobs: name: models-${{ env.REF }}${{ inputs.artifact_suffix }} path: ${{ env.MODELS_DIR }} - run: | - rm -f ${{ env.MODELS_DIR }}/{dmonitoring_model,big_driving_policy,big_driving_vision}.onnx + rm -f ${{ env.MODELS_DIR }}/{dmonitoring_model,big_driving_policy,big_driving_vision,big_driving_supercombo}.onnx - name: Build Model run: | source /etc/profile export UV_PROJECT_ENVIRONMENT=${HOME}/venv export VIRTUAL_ENV=$UV_PROJECT_ENVIRONMENT + source ${UV_PROJECT_ENVIRONMENT}/bin/activate export PYTHONPATH="${PYTHONPATH}:${{ env.TINYGRAD_PATH }}:${{ github.workspace }}" COMPILE_MODELD="${{ github.workspace }}/openpilot/sunnypilot/modeld_v2/compile_modeld.py" MODEL_SIZE=$(python3 -c "from openpilot.common.transformations.model import MEDMODEL_INPUT_SIZE as s; print(f'{s[0]}x{s[1]}')") CAMERA_RES=$(python3 -c "from openpilot.common.transformations.camera import _ar_ox_fisheye as a, _os_fisheye as o; print(f'{a.width}x{a.height} {o.width}x{o.height}')") - TG_FLAGS="DEV=QCOM IMAGE=1 FLOAT16=1 NOLOCALS=1 JIT_BATCH_SIZE=0 OPENPILOT_HACKS=1" + + if [ "${{ inputs.target_hardware }}" == "usbgpu" ]; then + echo "USBGPU build" + export USBGPU=1 + TG_FLAGS="DEV=AMD USBGPU=1 IMAGE=1 FLOAT16=1 NOLOCALS=1 JIT_BATCH_SIZE=0 OPENPILOT_HACKS=1" + OUTPUT_PKL="${{ env.MODELS_DIR }}/big_driving_tinygrad.pkl" + else + echo "QCOM build" + TG_FLAGS="DEV=QCOM IMAGE=1 FLOAT16=1 NOLOCALS=1 JIT_BATCH_SIZE=0 OPENPILOT_HACKS=1" + OUTPUT_PKL="${{ env.MODELS_DIR }}/driving_tinygrad.pkl" + fi # Generate metadata for all ONNX files find "${{ env.MODELS_DIR }}" -maxdepth 1 -name '*.onnx' | while IFS= read -r onnx_file; do @@ -186,7 +210,13 @@ jobs: POLICY_ONNX="${{ env.MODELS_DIR }}/driving_policy.onnx" OFF_POLICY_ONNX="${{ env.MODELS_DIR }}/driving_off_policy.onnx" ON_POLICY_ONNX="${{ env.MODELS_DIR }}/driving_on_policy.onnx" - SUPERCOMBO_ONNX="${{ env.MODELS_DIR }}/supercombo.onnx" + SUPERCOMBO_ONNX="" + for f in "${{ env.MODELS_DIR }}/supercombo.onnx" "${{ env.MODELS_DIR }}/driving_supercombo.onnx"; do + if [ -f "$f" ]; then + SUPERCOMBO_ONNX="$f" + break + fi + done MODEL_TYPE="" ONNX_ARGS="" OUTPUT_NAME="" if [ -f "$VISION_ONNX" ]; then @@ -207,24 +237,15 @@ jobs: fi if [ -n "$MODEL_TYPE" ]; then - echo "Detected: $MODEL_TYPE -> driving_tinygrad.pkl" + echo "Detected: $MODEL_TYPE -> $OUTPUT_PKL" env ${TG_FLAGS} python3 "$COMPILE_MODELD" \ --model-type $MODEL_TYPE \ --model-size $MODEL_SIZE \ --camera-resolutions $CAMERA_RES \ $ONNX_ARGS \ - --output "${{ env.MODELS_DIR }}/driving_tinygrad.pkl" + --output "$OUTPUT_PKL" fi - - name: Validate Model Outputs - run: | - source /etc/profile - export UV_PROJECT_ENVIRONMENT=${HOME}/venv - export VIRTUAL_ENV=$UV_PROJECT_ENVIRONMENT - python3 "${{ github.workspace }}/release/ci/model_generator.py" \ - --validate-only \ - --model-dir "${{ env.MODELS_DIR }}" - - name: Prepare Output run: | sudo rm -rf ${{ env.OUTPUT_DIR }} diff --git a/openpilot/cereal/custom.capnp b/openpilot/cereal/custom.capnp index c81aa104f2..a77997ffbb 100644 --- a/openpilot/cereal/custom.capnp +++ b/openpilot/cereal/custom.capnp @@ -139,10 +139,16 @@ struct ModelManagerSP @0xaedffd8f31e7b55d { eta @2 :UInt32; } + struct Chunk { + fileName @0 :Text; + sha256 @1 :Text; + } + struct Artifact { fileName @0 :Text; downloadUri @1 :DownloadUri; downloadProgress @2 :DownloadProgress; + chunks @3 :List(Chunk); } struct Model { @@ -157,6 +163,7 @@ struct ModelManagerSP @0xaedffd8f31e7b55d { policy @3; offPolicy @4; onPolicy @5; + chunked @6; } } diff --git a/openpilot/sunnypilot/modeld_v2/compile_modeld.py b/openpilot/sunnypilot/modeld_v2/compile_modeld.py index fd58038ceb..006ea6352b 100755 --- a/openpilot/sunnypilot/modeld_v2/compile_modeld.py +++ b/openpilot/sunnypilot/modeld_v2/compile_modeld.py @@ -10,471 +10,291 @@ import argparse import os import pickle import time -from functools import partial from collections import defaultdict - +from functools import partial import numpy as np -from tinygrad.tensor import Tensor +os.environ['GMMU'] = '0' + +def _patch_tinygrad_fetch_fw(): + import hashlib + import pathlib + import zstandard + from tinygrad import helpers + _orig_fetch_fw = helpers.fetch_fw + def fetch_fw(path, name, sha256): + p = pathlib.Path(f"/lib/firmware/{path}/{name}.zst") + if p.is_file(): + blob = zstandard.ZstdDecompressor().stream_reader(p.read_bytes()).read() + if hashlib.sha256(blob).hexdigest() == sha256: + return blob + return _orig_fetch_fw(path, name, sha256) + helpers.fetch_fw = fetch_fw +_patch_tinygrad_fetch_fw() + +from openpilot.selfdrive.modeld.compile_modeld import NV12Frame, make_frame_prepare, sample_desire, sample_skip, shift_and_sample +from tinygrad import dtypes from tinygrad.device import Device from tinygrad.engine.jit import TinyJit - -from openpilot.selfdrive.modeld.compile_modeld import ( - NV12Frame, make_frame_prepare, - shift_and_sample, sample_skip, sample_desire, -) +from tinygrad.tensor import Tensor MODEL_TYPES = ('vision_policy', 'supercombo', 'vision_multi_policy') -def _detect_desire_key(policy_input_shapes): - for k in policy_input_shapes: - if k.startswith('desire'): - return k - return None +def _detect_desire_key(shapes: dict) -> str | None: + return next((key for key in shapes if key.startswith('desire')), None) -def _detect_vision_keys(vision_input_shapes): - img_keys = sorted([k for k in vision_input_shapes if 'img' in k]) - road_key = next((k for k in img_keys if 'big' not in k), None) - wide_key = next((k for k in img_keys if 'big' in k), None) - if road_key is None or wide_key is None: - raise ValueError(f"Cannot determine road/wide image keys from {list(vision_input_shapes.keys())}") - return road_key, wide_key +def _detect_vision_keys(shapes: dict) -> tuple[str | None, str | None]: + img_keys = sorted(key for key in shapes if 'img' in key) + return ( + next((key for key in img_keys if 'big' not in key), None), + next((key for key in img_keys if 'big' in key), None) + ) -def make_split_input_queues(vision_input_shapes, policy_input_shapes, frame_skip, device): - road_key, _ = _detect_vision_keys(vision_input_shapes) - img = vision_input_shapes[road_key] - n_frames = img[1] // 6 - img_buf_shape = (frame_skip * (n_frames - 1) + 1, 6, img[2], img[3]) - - fb = policy_input_shapes['features_buffer'] - desire_key = _detect_desire_key(policy_input_shapes) - dp = policy_input_shapes[desire_key] - tc = policy_input_shapes.get('traffic_convention', (1, 2)) - - npy = { - 'desire': np.zeros(dp[2], dtype=np.float32), - 'traffic_convention': np.zeros(tc, dtype=np.float32), - 'tfm': np.zeros((3, 3), dtype=np.float32), - 'big_tfm': np.zeros((3, 3), dtype=np.float32), - } - - handled = {'features_buffer', desire_key, 'traffic_convention'} - for key, shape in policy_input_shapes.items(): - if key in handled: - continue - npy[key] = np.zeros(shape, dtype=np.float32) - - input_queues = { - 'img_q': Tensor(np.zeros(img_buf_shape, dtype=np.uint8), device=device).contiguous().realize(), - 'big_img_q': Tensor(np.zeros(img_buf_shape, dtype=np.uint8), device=device).contiguous().realize(), - 'feat_q': Tensor(np.zeros((frame_skip * (fb[1] - 1) + 1, fb[0], fb[2]), dtype=np.float32), device=device).contiguous().realize(), - 'desire_q': Tensor(np.zeros((frame_skip * dp[1], dp[0], dp[2]), dtype=np.float32), device=device).contiguous().realize(), - **{k: Tensor(v, device='NPY').realize() for k, v in npy.items()}, - } - return input_queues, npy +def derive_frame_skip(vision_input_shapes: dict, policy_input_shapes: dict) -> int: + features_buffer = policy_input_shapes.get('features_buffer') + return 1 if not features_buffer or features_buffer[1] >= 99 else 4 -def make_run_split_policy(vision_runner, policy_runner, nv12: NV12Frame, model_w, model_h, - vision_features_slice, frame_skip, desire_key, extra_policy_keys, - vision_road_key, vision_wide_key, prepare_only=False): - frame_prepare = make_frame_prepare(nv12, model_w, model_h) - sample_skip_fn = partial(sample_skip, frame_skip=frame_skip) - sample_desire_fn = partial(sample_desire, frame_skip=frame_skip) - - def run_policy(img_q, big_img_q, feat_q, desire_q, desire, traffic_convention, tfm, big_tfm, frame, big_frame, **extra): - npy_tensors = [tfm.to(Device.DEFAULT), big_tfm.to(Device.DEFAULT), - desire.to(Device.DEFAULT), traffic_convention.to(Device.DEFAULT)] - extra_device = {k: extra[k].to(Device.DEFAULT) for k in extra_policy_keys} - Tensor.realize(*npy_tensors, *extra_device.values()) - tfm, big_tfm, desire, traffic_convention = npy_tensors - - img = shift_and_sample(img_q, frame_prepare(frame, tfm).unsqueeze(0), sample_skip_fn) - big_img = shift_and_sample(big_img_q, frame_prepare(big_frame, big_tfm).unsqueeze(0), sample_skip_fn) - - if prepare_only: - return img, big_img - - vision_out = next(iter(vision_runner({vision_road_key: img, vision_wide_key: big_img}).values())).cast('float32') - - new_feat = vision_out[:, vision_features_slice].reshape(1, -1).unsqueeze(0) - feat_buf = shift_and_sample(feat_q, new_feat, sample_skip_fn) - desire_buf = shift_and_sample(desire_q, desire.reshape(1, 1, -1), sample_desire_fn) - - inputs = {'features_buffer': feat_buf, desire_key: desire_buf, 'traffic_convention': traffic_convention, **extra_device} - policy_out = next(iter(policy_runner(inputs).values())).cast('float32') - - return vision_out, policy_out - return run_policy - - -def compile_split_policy(nv12: NV12Frame, model_w, model_h, prepare_only, frame_skip, - vision_runner, policy_runner, vision_metadata, policy_metadata): - print(f"Compiling combined policy JIT for {nv12.width}x{nv12.height} (prepare_only={prepare_only})...") - - vision_features_slice = vision_metadata['output_slices']['hidden_state'] - vision_input_shapes = vision_metadata['input_shapes'] - policy_input_shapes = policy_metadata['input_shapes'] - desire_key = _detect_desire_key(policy_input_shapes) - extra_policy_keys = [k for k in policy_input_shapes if k not in ('features_buffer', desire_key, 'traffic_convention')] - vision_road_key, vision_wide_key = _detect_vision_keys(vision_input_shapes) - - _run = make_run_split_policy(vision_runner, policy_runner, nv12, model_w, model_h, - vision_features_slice, frame_skip, desire_key, extra_policy_keys, - vision_road_key, vision_wide_key, prepare_only) - run_policy_jit = TinyJit(_run, prune=True) - - SEED = 42 - - def random_inputs_run_fn(fn, seed, test_val=None, test_buffers=None, expect_match=True): - input_queues, npy = make_split_input_queues(vision_input_shapes, policy_input_shapes, frame_skip, Device.DEFAULT) - rng = np.random.default_rng(seed) - Tensor.manual_seed(seed) - - testing = test_val is not None or test_buffers is not None - n_runs = 1 if testing else 3 - - for i in range(n_runs): - frame = Tensor.randint(nv12.size, low=0, high=256, dtype='uint8').realize() - big_frame = Tensor.randint(nv12.size, low=0, high=256, dtype='uint8').realize() - for v in npy.values(): - v[:] = rng.standard_normal(v.shape).astype(v.dtype) - Device.default.synchronize() - st = time.perf_counter() - outs = fn(**input_queues, frame=frame, big_frame=big_frame) - mt = time.perf_counter() - Device.default.synchronize() - et = time.perf_counter() - print(f" [{i+1}/{n_runs}] enqueue {(mt-st)*1e3:6.2f} ms -- total {(et-st)*1e3:6.2f} ms") - - if i == 0: - val = [np.copy(v.numpy()) for v in outs] - buffers = [np.copy(v.numpy().copy()) for v in input_queues.values()] - - if test_val is not None: - match = all(np.array_equal(a, b) for a, b in zip(val, test_val, strict=True)) - assert match == expect_match, f"outputs {'differ from' if expect_match else 'match'} baseline (seed={seed})" - if test_buffers is not None: - match = all(np.array_equal(a, b) for a, b in zip(buffers, test_buffers, strict=True)) - assert match == expect_match, f"buffers {'differ from' if expect_match else 'match'} baseline (seed={seed})" - return fn, val, buffers - - print('capture + replay') - run_policy_jit, test_val, test_buffers = random_inputs_run_fn(run_policy_jit, SEED) - - print('pickle round trip') - run_policy_jit = pickle.loads(pickle.dumps(run_policy_jit)) - random_inputs_run_fn(run_policy_jit, SEED, test_val, test_buffers, expect_match=True) - random_inputs_run_fn(run_policy_jit, SEED+1, test_val, test_buffers, expect_match=False) - return run_policy_jit - - -def derive_frame_skip(vision_input_shapes, policy_input_shapes): - fb = policy_input_shapes.get('features_buffer') - if fb is None: - return 1 - fb_history = fb[1] - if fb_history >= 99: - return 1 - return 4 - - -def make_supercombo_input_queues(input_shapes, frame_skip, device): - img_shape = input_shapes.get('img', input_shapes.get('input_imgs')) - if img_shape is None: - raise ValueError("No img input found in model shapes") +def generate_queues_and_npy(input_shapes: dict, frame_skip: int, device: str = Device.DEFAULT) -> tuple[dict, dict]: + road_key, _ = _detect_vision_keys(input_shapes) + if not road_key: + raise ValueError("Vision road key missing from input shapes.") + img_shape = input_shapes[road_key] n_frames = img_shape[1] // 6 img_buf_shape = (frame_skip * (n_frames - 1) + 1, 6, img_shape[2], img_shape[3]) - numpy_keys = {} - queue_keys = {} + desire_key = _detect_desire_key(input_shapes) + if not desire_key: + raise ValueError("Desire key missing from input shapes.") + + desire_shape = input_shapes[desire_key] + features_buffer = input_shapes.get('features_buffer') + + npy_arrays = { + 'desire': np.zeros(desire_shape[2], dtype=np.float32), + 'tfm': np.zeros((3, 3), dtype=np.float32), + 'big_tfm': np.zeros((3, 3), dtype=np.float32) + } for key, shape in input_shapes.items(): - if 'img' in key: - continue - if len(shape) == 3 and shape[1] > 1: - if key.startswith('desire'): - numpy_keys[key] = np.zeros(shape[2], dtype=np.float32) - queue_keys[f'{key}_q'] = Tensor( - np.zeros((frame_skip * shape[1], shape[0], shape[2]), dtype=np.float32), - device=device).contiguous().realize() - elif key == 'features_buffer': - queue_keys['feat_q'] = Tensor( - np.zeros((frame_skip * (shape[1] - 1) + 1, shape[0], shape[2]), dtype=np.float32), - device=device).contiguous().realize() - else: - numpy_keys[key] = np.zeros(shape, dtype=np.float32) - elif len(shape) == 2: - numpy_keys[key] = np.zeros(shape, dtype=np.float32) + if key not in npy_arrays and 'img' not in key and key not in ('features_buffer', desire_key): + npy_arrays[key] = np.zeros(shape, dtype=np.float32) - if 'traffic_convention' not in numpy_keys: - tc_shape = input_shapes.get('traffic_convention', (1, 2)) - numpy_keys['traffic_convention'] = np.zeros(tc_shape, dtype=np.float32) - - numpy_keys['tfm'] = np.zeros((3, 3), dtype=np.float32) - numpy_keys['big_tfm'] = np.zeros((3, 3), dtype=np.float32) - - input_queues = { + queues = { 'img_q': Tensor(np.zeros(img_buf_shape, dtype=np.uint8), device=device).contiguous().realize(), 'big_img_q': Tensor(np.zeros(img_buf_shape, dtype=np.uint8), device=device).contiguous().realize(), - **queue_keys, - **{k: Tensor(v, device='NPY').realize() for k, v in numpy_keys.items()}, + 'desire_q': Tensor(np.zeros((frame_skip * desire_shape[1], desire_shape[0], desire_shape[2]), + dtype=np.float32), device=device).contiguous().realize() } - return input_queues, numpy_keys + + if features_buffer: + queues['feat_q'] = Tensor(np.zeros((frame_skip * (features_buffer[1] - 1) + 1, features_buffer[0], features_buffer[2]), + dtype=np.float32), device=device).contiguous().realize() + + queues.update({key: Tensor(value, device='NPY').realize() for key, value in npy_arrays.items()}) + return queues, npy_arrays -def make_run_supercombo(model_runner, nv12: NV12Frame, model_w, model_h, - features_slice, frame_skip, input_shapes, prepare_only=False): - frame_prepare = make_frame_prepare(nv12, model_w, model_h) +def make_split_input_queues(vision_input_shapes: dict, policy_input_shapes: dict, frame_skip: int, device: str = Device.DEFAULT) -> tuple[dict, dict]: + return generate_queues_and_npy({**vision_input_shapes, **policy_input_shapes}, frame_skip, device) + + +def make_supercombo_input_queues(input_shapes: dict, frame_skip: int, device: str = Device.DEFAULT) -> tuple[dict, dict]: + return generate_queues_and_npy(input_shapes, frame_skip, device) + + +def create_jit_runner(vision_runner, policy_runners: list, nv12: NV12Frame, model_size: tuple[int, int], + features_slice: slice, frame_skip: int, input_shapes: dict, prepare_only: bool): + frame_prepare = make_frame_prepare(nv12, *model_size) sample_skip_fn = partial(sample_skip, frame_skip=frame_skip) sample_desire_fn = partial(sample_desire, frame_skip=frame_skip) desire_key = _detect_desire_key(input_shapes) - if desire_key is None: - raise ValueError(f"No desire* key found in input_shapes: {list(input_shapes.keys())}") - road_img_key, wide_img_key = _detect_vision_keys(input_shapes) - extra_policy_keys = [k for k in input_shapes - if k not in (desire_key, 'features_buffer', 'traffic_convention') - and 'img' not in k] + road_key, wide_key = _detect_vision_keys(input_shapes) - def run_supercombo(img_q, big_img_q, feat_q, desire_q, - frame, big_frame, **kwargs): - desire = kwargs.get(desire_key) + if not desire_key or not road_key or not wide_key: + raise ValueError("Missing required vision or desire keys in input shapes.") + + extra_keys = [key for key in input_shapes if key not in (desire_key, 'features_buffer', 'traffic_convention') and 'img' not in key] + + def runner(img_q, big_img_q, feat_q, frame, big_frame, tfm, big_tfm, **kwargs): + desire_q = kwargs['desire_q'] + desire = kwargs['desire'] traffic_convention = kwargs.get('traffic_convention') - tfm = kwargs['tfm'] - big_tfm = kwargs['big_tfm'] - tfm = tfm.to(Device.DEFAULT) - big_tfm = big_tfm.to(Device.DEFAULT) - desire = desire.to(Device.DEFAULT) - traffic_convention = traffic_convention.to(Device.DEFAULT) - Tensor.realize(tfm, big_tfm, desire, traffic_convention) + npys = [tfm.to(Device.DEFAULT), big_tfm.to(Device.DEFAULT), desire.to(Device.DEFAULT)] + if traffic_convention is not None: + npys.append(traffic_convention.to(Device.DEFAULT)) - img = shift_and_sample(img_q, frame_prepare(frame, tfm).unsqueeze(0), sample_skip_fn) - big_img = shift_and_sample(big_img_q, frame_prepare(big_frame, big_tfm).unsqueeze(0), sample_skip_fn) + extra_tensors = {key: kwargs[key].to(Device.DEFAULT) for key in extra_keys if key in kwargs} + Tensor.realize(*npys, *extra_tensors.values()) + + tfm_dev, big_tfm_dev, desire_dev = npys[:3] + traffic_conv_dev = npys[3] if traffic_convention is not None else None + + img = shift_and_sample(img_q, frame_prepare(frame, tfm_dev).unsqueeze(0), sample_skip_fn).realize() + big_img = shift_and_sample(big_img_q, frame_prepare(big_frame, big_tfm_dev).unsqueeze(0), sample_skip_fn).realize() if prepare_only: return img, big_img - desire_buf = shift_and_sample(desire_q, desire.reshape(1, 1, -1), sample_desire_fn) - feat_buf = sample_skip_fn(feat_q) + desire_buf = shift_and_sample(desire_q, desire_dev.reshape(1, 1, -1), sample_desire_fn).realize() + inputs = {desire_key: desire_buf, **extra_tensors} - inputs = {road_img_key: img, wide_img_key: big_img, - desire_key: desire_buf, 'features_buffer': feat_buf, - 'traffic_convention': traffic_convention} - for k in extra_policy_keys: - if k in kwargs: - inputs[k] = kwargs[k].to(Device.DEFAULT) + if traffic_conv_dev is not None: + inputs['traffic_convention'] = traffic_conv_dev - model_out = next(iter(model_runner(inputs).values())).cast('float32') + if vision_runner: + vision_out_cast = next(iter(vision_runner({road_key: img, wide_key: big_img}).values())).cast('float32').realize() + new_feat = vision_out_cast[:, features_slice].reshape(1, -1).unsqueeze(0) + inputs['features_buffer'] = shift_and_sample(feat_q, new_feat, sample_skip_fn).realize() + policy_outs = [next(iter(pol_runner(inputs).values())).cast('float32').realize() for pol_runner in policy_runners] + return (vision_out_cast, *policy_outs) if len(policy_outs) > 1 else (vision_out_cast, policy_outs[0]) + inputs.update({road_key: img, wide_key: big_img, 'features_buffer': sample_skip_fn(feat_q)}) + policy_out = next(iter(policy_runners[0](inputs).values())).cast('float32').realize() + new_feat = policy_out[:, features_slice].reshape(1, -1).unsqueeze(0) + shift_and_sample(feat_q, new_feat, sample_skip_fn).realize() + return policy_out - new_feat = model_out[:, features_slice].reshape(1, -1).unsqueeze(0) - shift_and_sample(feat_q, new_feat, sample_skip_fn) - - return model_out - - return run_supercombo + return runner -def make_run_vision_multi_policy(vision_runner, policy_runners, nv12: NV12Frame, model_w, model_h, - vision_features_slice, frame_skip, desire_key, extra_policy_keys, - vision_road_key, vision_wide_key, prepare_only=False): - frame_prepare = make_frame_prepare(nv12, model_w, model_h) - sample_skip_fn = partial(sample_skip, frame_skip=frame_skip) - sample_desire_fn = partial(sample_desire, frame_skip=frame_skip) +def compile_and_warmup(nv12: NV12Frame, model_size: tuple[int, int], prepare_only: bool, frame_skip: int, vision_runner, policy_runners: list, metadata: dict): + print(f"Compiling combined JIT for {nv12.width}x{nv12.height} (prepare_only={prepare_only})...") - def run_multi_policy(img_q, big_img_q, feat_q, desire_q, desire, - traffic_convention, tfm, big_tfm, frame, big_frame, **extra): - npy_tensors = [tfm.to(Device.DEFAULT), big_tfm.to(Device.DEFAULT), - desire.to(Device.DEFAULT), traffic_convention.to(Device.DEFAULT)] - extra_device = {k: extra[k].to(Device.DEFAULT) for k in extra_policy_keys} - Tensor.realize(*npy_tensors, *extra_device.values()) - tfm, big_tfm, desire, traffic_convention = npy_tensors + all_shapes = {key: value for meta in metadata.values() for key, value in meta['input_shapes'].items()} - img = shift_and_sample(img_q, frame_prepare(frame, tfm).unsqueeze(0), sample_skip_fn) - big_img = shift_and_sample(big_img_q, frame_prepare(big_frame, big_tfm).unsqueeze(0), sample_skip_fn) + feat_meta = metadata.get('vision') or metadata.get('model') or metadata.get('policy') + if not feat_meta: + raise ValueError("Could not find vision, model, or policy metadata.") - if prepare_only: - return img, big_img + features_slice = feat_meta['output_slices']['hidden_state'] + WARP_DEV = 'CPU' if "USBGPU" in os.environ else Device.DEFAULT - vision_out = next(iter(vision_runner({vision_road_key: img, vision_wide_key: big_img}).values())).cast('float32') + run_func = create_jit_runner(vision_runner, policy_runners, nv12, model_size, features_slice, frame_skip, all_shapes, prepare_only) + run_jit = TinyJit(run_func, prune=True) + queues, npy_arrays = generate_queues_and_npy(all_shapes, frame_skip, Device.DEFAULT) - new_feat = vision_out[:, vision_features_slice].reshape(1, -1).unsqueeze(0) - feat_buf = shift_and_sample(feat_q, new_feat, sample_skip_fn) - desire_buf = shift_and_sample(desire_q, desire.reshape(1, 1, -1), sample_desire_fn) - - inputs = {'features_buffer': feat_buf, desire_key: desire_buf, 'traffic_convention': traffic_convention, **extra_device} - - policy_outputs = [] - for runner in policy_runners: - policy_out = next(iter(runner(inputs).values())).cast('float32') - policy_outputs.append(policy_out) - - return (vision_out, *policy_outputs) - - return run_multi_policy - - -def _warmup_and_serialize(run_jit, input_queues, npy, nv12): for i in range(3): rng = np.random.default_rng(42 + i) - frame = Tensor.randint(nv12.size, low=0, high=256, dtype='uint8').realize() - big_frame = Tensor.randint(nv12.size, low=0, high=256, dtype='uint8').realize() - for v in npy.values(): - v[:] = rng.standard_normal(v.shape).astype(v.dtype) + frame = Tensor.randint(nv12.size, low=0, high=256, dtype=dtypes.uint8, device=WARP_DEV).realize() + big_frame = Tensor.randint(nv12.size, low=0, high=256, dtype=dtypes.uint8, device=WARP_DEV).realize() + for arr in npy_arrays.values(): + arr[:] = rng.standard_normal(arr.shape).astype(arr.dtype) + Device.default.synchronize() - st = time.perf_counter() - run_jit(**input_queues, frame=frame, big_frame=big_frame) - mt = time.perf_counter() + start_time = time.perf_counter() + run_jit(**queues, frame=frame, big_frame=big_frame) + mid_time = time.perf_counter() Device.default.synchronize() - et = time.perf_counter() - print(f" [{i + 1}/3] enqueue {(mt - st) * 1e3:6.2f} ms -- total {(et - st) * 1e3:6.2f} ms") - return pickle.loads(pickle.dumps(run_jit)) + print(f" [{i + 1}/3] enqueue {(mid_time - start_time) * 1e3:6.2f} ms -- total {(time.perf_counter() - start_time) * 1e3:6.2f} ms") + + return pickle.loads(pickle.dumps(run_jit)) if not prepare_only else run_jit -def compile_supercombo(nv12: NV12Frame, model_w, model_h, prepare_only, frame_skip, - model_runner, metadata): - print(f"Compiling combined supercombo JIT for {nv12.width}x{nv12.height} (prepare_only={prepare_only})...") - - features_slice = metadata['output_slices']['hidden_state'] - input_shapes = metadata['input_shapes'] - - _run = make_run_supercombo(model_runner, nv12, model_w, model_h, - features_slice, frame_skip, input_shapes, prepare_only) - run_jit = TinyJit(_run, prune=True) - - input_queues, npy = make_supercombo_input_queues(input_shapes, frame_skip, Device.DEFAULT) - - run_jit = _warmup_and_serialize(run_jit, input_queues, npy, nv12) - return run_jit +def _parse_size(size_str: str) -> tuple[int, int]: + width, height = size_str.lower().split('x') + return int(width), int(height) -def compile_multi_policy(nv12: NV12Frame, model_w, model_h, prepare_only, frame_skip, - vision_runner, policy_runners, vision_metadata, policy_metadata): - print(f"Compiling combined multi-policy JIT for {nv12.width}x{nv12.height} (prepare_only={prepare_only})...") - - vision_features_slice = vision_metadata['output_slices']['hidden_state'] - vision_input_shapes = vision_metadata['input_shapes'] - policy_input_shapes = policy_metadata['input_shapes'] - desire_key = _detect_desire_key(policy_input_shapes) - extra_policy_keys = [k for k in policy_input_shapes if k not in ('features_buffer', desire_key, 'traffic_convention')] - vision_road_key, vision_wide_key = _detect_vision_keys(vision_input_shapes) - - _run = make_run_vision_multi_policy(vision_runner, policy_runners, nv12, model_w, model_h, - vision_features_slice, frame_skip, desire_key, extra_policy_keys, - vision_road_key, vision_wide_key, prepare_only) - run_jit = TinyJit(_run, prune=True) - - input_queues, npy = make_split_input_queues(vision_input_shapes, policy_input_shapes, frame_skip, Device.DEFAULT) - - run_jit = _warmup_and_serialize(run_jit, input_queues, npy, nv12) - return run_jit +def read_file_chunked_to_shm(path): + if not path: + return None + import atexit + import shutil + from openpilot.common.file_chunker import open_file_chunked + from openpilot.common.hardware.hw import Paths + shm_path = os.path.join(Paths.shm_path(), os.path.basename(path)) + atexit.register(lambda: os.path.exists(shm_path) and os.remove(shm_path)) + with open(shm_path, 'wb') as dst, open_file_chunked(path) as src: + shutil.copyfileobj(src, dst) + return shm_path -def _parse_size(s): - w, h = s.lower().split('x') - return int(w), int(h) +def _compile_for_resolutions(camera_resolutions: list, model_size: tuple[int, int], frame_skip: int, + vision_runner, policy_runners: list, metadata: dict) -> dict: + from openpilot.system.camerad.cameras.nv12_info import get_nv12_info + return { + (cam_w, cam_h): { + name: compile_and_warmup(NV12Frame(cam_w, cam_h, *get_nv12_info(cam_w, cam_h)), model_size, prepare_only, + frame_skip, vision_runner, policy_runners, metadata) + for name, prepare_only in [('warp_enqueue', True), ('run_policy', False)] + } + for cam_w, cam_h in camera_resolutions + } + + +def _load_policy_runners(args: argparse.Namespace) -> tuple[list, list]: + runners, keys = [], [] + for name, onnx_arg in [('policy', args.policy_onnx), ('off_policy', args.off_policy_onnx), ('on_policy', args.on_policy_onnx)]: + if onnx_arg: + runners.append(OnnxRunner(onnx_arg)) + keys.append(name) + return runners, keys if __name__ == "__main__": - from tinygrad.nn.onnx import OnnxRunner - from openpilot.system.camerad.cameras.nv12_info import get_nv12_info from openpilot.selfdrive.modeld.get_model_metadata import make_metadata_dict + from tinygrad.nn.onnx import OnnxRunner - p = argparse.ArgumentParser(description="Compile combined JIT pkl for sunnypilot modeld_v2") - p.add_argument('--model-type', choices=MODEL_TYPES, required=True) - p.add_argument('--model-size', type=_parse_size, required=True, help='model input WxH') - p.add_argument('--camera-resolutions', type=_parse_size, nargs='+', required=True) - p.add_argument('--frame-skip', type=int, default=None, help='frame skip value (auto-derived if not provided)') - p.add_argument('--output', required=True) + parser = argparse.ArgumentParser(description="Compile combined JIT pkl for sunnypilot modeld_v2") + parser.add_argument('--model-type', choices=MODEL_TYPES, required=True) + parser.add_argument('--model-size', type=_parse_size, required=True, help='model input WxH') + parser.add_argument('--camera-resolutions', type=_parse_size, nargs='+', required=True) + parser.add_argument('--frame-skip', type=int, default=None, help='frame skip value (auto-derived if not provided)') + parser.add_argument('--output', required=True) - p.add_argument('--vision-onnx', help='vision ONNX (for split models)') - p.add_argument('--policy-onnx', help='policy ONNX (for vision_policy)') - p.add_argument('--off-policy-onnx', help='off-policy ONNX (for vision_multi_policy)') - p.add_argument('--on-policy-onnx', help='on-policy ONNX (for vision_multi_policy)') - p.add_argument('--supercombo-onnx', help='supercombo ONNX (for supercombo)') + parser.add_argument('--vision-onnx', help='vision ONNX (for split models)') + parser.add_argument('--policy-onnx', help='policy ONNX (for vision_policy)') + parser.add_argument('--off-policy-onnx', help='off-policy ONNX (for vision_multi_policy)') + parser.add_argument('--on-policy-onnx', help='on-policy ONNX (for vision_multi_policy)') + parser.add_argument('--supercombo-onnx', help='supercombo ONNX (for supercombo)') - args = p.parse_args() - out = defaultdict(dict) + args = parser.parse_args() + output_data = defaultdict(dict) + + args.vision_onnx = read_file_chunked_to_shm(args.vision_onnx) + args.policy_onnx = read_file_chunked_to_shm(args.policy_onnx) + args.off_policy_onnx = read_file_chunked_to_shm(args.off_policy_onnx) + args.on_policy_onnx = read_file_chunked_to_shm(args.on_policy_onnx) + args.supercombo_onnx = read_file_chunked_to_shm(args.supercombo_onnx) + + vision_runner = OnnxRunner(args.vision_onnx) if args.vision_onnx else None if args.model_type == 'vision_policy': - assert args.vision_onnx and args.policy_onnx - vision_runner = OnnxRunner(args.vision_onnx) - policy_runner = OnnxRunner(args.policy_onnx) - out['metadata']['vision'] = make_metadata_dict(args.vision_onnx) - out['metadata']['policy'] = make_metadata_dict(args.policy_onnx) - - frame_skip = args.frame_skip if args.frame_skip is not None else derive_frame_skip(out['metadata']['vision']['input_shapes'], - out['metadata']['policy']['input_shapes']) - - for cam_w, cam_h in args.camera_resolutions: - nv12 = NV12Frame(cam_w, cam_h, *get_nv12_info(cam_w, cam_h)) - model_w, model_h = args.model_size - out[(cam_w, cam_h)] = { - name: compile_split_policy(nv12, model_w, model_h, prepare_only, frame_skip, - vision_runner, policy_runner, - out['metadata']['vision'], out['metadata']['policy']) - for name, prepare_only in [('warp_enqueue', True), ('run_policy', False)] - } - + assert vision_runner and args.policy_onnx + policy_runners = [OnnxRunner(args.policy_onnx)] + output_data['metadata'] = {'vision': make_metadata_dict(args.vision_onnx), 'policy': make_metadata_dict(args.policy_onnx)} elif args.model_type == 'supercombo': assert args.supercombo_onnx - model_runner = OnnxRunner(args.supercombo_onnx) - out['metadata']['model'] = make_metadata_dict(args.supercombo_onnx) - - frame_skip = args.frame_skip if args.frame_skip is not None else derive_frame_skip({}, out['metadata']['model']['input_shapes']) - - for cam_w, cam_h in args.camera_resolutions: - nv12 = NV12Frame(cam_w, cam_h, *get_nv12_info(cam_w, cam_h)) - model_w, model_h = args.model_size - out[(cam_w, cam_h)] = { - name: compile_supercombo(nv12, model_w, model_h, prepare_only, frame_skip, - model_runner, out['metadata']['model']) - for name, prepare_only in [('warp_enqueue', True), ('run_policy', False)] - } - + policy_runners = [OnnxRunner(args.supercombo_onnx)] + output_data['metadata'] = {'model': make_metadata_dict(args.supercombo_onnx)} elif args.model_type == 'vision_multi_policy': - assert args.vision_onnx - vision_runner = OnnxRunner(args.vision_onnx) - out['metadata']['vision'] = make_metadata_dict(args.vision_onnx) + assert vision_runner + policy_runners, policy_names = _load_policy_runners(args) + output_data['metadata'] = {'vision': make_metadata_dict(args.vision_onnx)} + for name in policy_names: + runner_arg = getattr(args, f"{name}_onnx") + output_data['metadata'][name] = make_metadata_dict(runner_arg) - policy_runners = [] - policy_onnxes = [] - if args.policy_onnx: - policy_onnxes.append(('policy', args.policy_onnx)) - if args.off_policy_onnx: - policy_onnxes.append(('off_policy', args.off_policy_onnx)) - if args.on_policy_onnx: - policy_onnxes.append(('on_policy', args.on_policy_onnx)) + policy_keys = [key for key in output_data['metadata'].keys() if key != 'vision'] + first_policy_meta = output_data['metadata'][policy_keys[0]] if policy_keys else {} + vision_meta = output_data['metadata'].get('vision', {}) - for name, onnx_path in policy_onnxes: - runner = OnnxRunner(onnx_path) - policy_runners.append(runner) - out['metadata'][name] = make_metadata_dict(onnx_path) + derived_frame_skip = args.frame_skip or derive_frame_skip(vision_meta.get('input_shapes', {}), first_policy_meta.get('input_shapes', {})) + output_data.update(_compile_for_resolutions(args.camera_resolutions, args.model_size, derived_frame_skip, + vision_runner, policy_runners, output_data['metadata'])) - first_policy_key = policy_onnxes[0][0] - frame_skip = args.frame_skip if args.frame_skip is not None else derive_frame_skip(out['metadata']['vision']['input_shapes'], - out['metadata'][first_policy_key]['input_shapes']) + with open(args.output, "wb") as file: + pickle.dump(output_data, file) - for cam_w, cam_h in args.camera_resolutions: - nv12 = NV12Frame(cam_w, cam_h, *get_nv12_info(cam_w, cam_h)) - model_w, model_h = args.model_size - out[(cam_w, cam_h)] = { - name: compile_multi_policy(nv12, model_w, model_h, prepare_only, frame_skip, - vision_runner, policy_runners, - out['metadata']['vision'], out['metadata'][first_policy_key]) - for name, prepare_only in [('warp_enqueue', True), ('run_policy', False)] - } - - with open(args.output, "wb") as f: - pickle.dump(out, f) pkl_size = os.path.getsize(args.output) print(f"Saved combined JIT to {args.output} ({pkl_size / 1e6:.2f} MB)") from openpilot.common.file_chunker import chunk_file, get_chunk_targets chunk_targets = get_chunk_targets(args.output, pkl_size) chunk_file(args.output, chunk_targets) - num_chunks = len(chunk_targets) - 1 - print(f"Chunked into {num_chunks} file(s)") + print(f"Chunked into {len(chunk_targets) - 1} file(s)") diff --git a/openpilot/sunnypilot/modeld_v2/modeld.py b/openpilot/sunnypilot/modeld_v2/modeld.py index 86d5b05868..a024e91113 100755 --- a/openpilot/sunnypilot/modeld_v2/modeld.py +++ b/openpilot/sunnypilot/modeld_v2/modeld.py @@ -7,6 +7,7 @@ See the LICENSE.md file in the root directory for more details. """ import os +os.environ['GMMU'] = '0' from openpilot.common.hardware import TICI os.environ['DEV'] = 'QCOM' if TICI else 'CPU' USBGPU = "USBGPU" in os.environ @@ -23,6 +24,11 @@ from setproctitle import setproctitle from openpilot.cereal.messaging import PubMaster, SubMaster from msgq.visionipc import VisionIpcClient, VisionStreamType, VisionBuf from opendbc.car.car_helpers import get_demo_car_params + +from tinygrad.tensor import Tensor +from tinygrad.device import Device + +from openpilot.common.file_chunker import open_file_chunked from openpilot.common.swaglog import cloudlog from openpilot.common.params import Params from openpilot.common.filter_simple import FirstOrderFilter @@ -30,6 +36,7 @@ from openpilot.common.realtime import config_realtime_process, DT_MDL from openpilot.common.transformations.camera import DEVICE_CAMERAS from openpilot.common.transformations.model import get_warp_matrix from openpilot.system import sentry +from openpilot.system.camerad.cameras.nv12_info import get_nv12_info from openpilot.selfdrive.controls.lib.desire_helper import DesireHelper from openpilot.selfdrive.controls.lib.drive_helpers import get_accel_from_plan, smooth_value @@ -37,6 +44,7 @@ from openpilot.sunnypilot.modeld_v2.fill_model_msg import fill_model_msg, fill_p from openpilot.sunnypilot.modeld_v2.constants import Plan from openpilot.sunnypilot.modeld_v2.meta_helper import load_meta_constants from openpilot.sunnypilot.modeld_v2.camera_offset_helper import CameraOffsetHelper +from openpilot.sunnypilot.modeld_v2.compile_modeld import derive_frame_skip, make_split_input_queues from openpilot.sunnypilot.livedelay.helpers import get_lat_delay from openpilot.sunnypilot.modeld_v2.modeld_base import ModelStateBase @@ -99,17 +107,12 @@ class ModelState(ModelStateBase): self._init_combined(pkl_path, cam_w, cam_h, model_bundle) def _init_combined(self, pkl_path, cam_w, cam_h, bundle): - from tinygrad.tensor import Tensor - from openpilot.system.camerad.cameras.nv12_info import get_nv12_info - from openpilot.sunnypilot.modeld_v2.compile_modeld import derive_frame_skip, make_split_input_queues - from tinygrad.device import Device - - from openpilot.common.file_chunker import open_file_chunked - cloudlog.warning(f"loading combined pkl: {pkl_path}") jits = pickle.load(open_file_chunked(pkl_path)) self.DEV = Device.DEFAULT + self.WARP_DEV = 'CPU' if USBGPU else self.DEV + self.QUEUE_DEV = self.DEV metadata = jits['metadata'] if 'model' in metadata: @@ -118,10 +121,10 @@ class ModelState(ModelStateBase): self.policy_output_slices = {} self._policy_slices_list = [] self._combined_model_type = 'supercombo' - self._vision_input_names = [k for k in model_metadata['input_shapes'] if 'img' in k] + self._vision_input_names = [key for key in model_metadata['input_shapes'] if 'img' in key] from openpilot.sunnypilot.modeld_v2.compile_modeld import make_supercombo_input_queues frame_skip = derive_frame_skip({}, model_metadata['input_shapes']) - self.input_queues, self.numpy_inputs = make_supercombo_input_queues(model_metadata['input_shapes'], frame_skip, device=self.DEV) + self.input_queues, self.numpy_inputs = make_supercombo_input_queues(model_metadata['input_shapes'], frame_skip, device=self.QUEUE_DEV) else: vision_metadata = metadata['vision'] policy_keys = [k for k in metadata if k != 'vision'] @@ -139,11 +142,11 @@ class ModelState(ModelStateBase): policy_input_shapes = first_policy_metadata['input_shapes'] self._vision_input_names = [k for k in vision_input_shapes if 'img' in k] frame_skip = derive_frame_skip(vision_input_shapes, policy_input_shapes) - self.input_queues, self.numpy_inputs = make_split_input_queues(vision_input_shapes, policy_input_shapes, frame_skip, device=self.DEV) + self.input_queues, self.numpy_inputs = make_split_input_queues(vision_input_shapes, policy_input_shapes, frame_skip, device=self.QUEUE_DEV) - from openpilot.sunnypilot.modeld_v2.parse_model_outputs_split import Parser as SplitParser - from openpilot.sunnypilot.modeld_v2.parse_model_outputs import Parser as CombinedParser - self.parser = SplitParser() if self._combined_model_type != 'supercombo' else CombinedParser() + self._desire_key = next(key for key in self.numpy_inputs if key.startswith('desire')) + self._road_key = next(key for key in self._vision_input_names if 'big' not in key) + self._wide_key = next(key for key in self._vision_input_names if 'big' in key) is_20hz = bundle.is20hz if bundle else self._combined_model_type in ('split', 'multi_policy') if is_20hz: @@ -153,6 +156,13 @@ class ModelState(ModelStateBase): from openpilot.sunnypilot.modeld_v2.constants import ModelConstants self.constants = ModelConstants() + if self._combined_model_type != 'supercombo': + from openpilot.sunnypilot.modeld_v2.parse_model_outputs_split import Parser as SplitParser + self.parser = SplitParser() + else: + from openpilot.sunnypilot.modeld_v2.parse_model_outputs import Parser as CombinedParser + self.parser = CombinedParser() + self.prev_desire = np.zeros(self.constants.DESIRE_LEN, dtype=np.float32) self.full_frames: dict = {} self._blob_cache: dict = {} @@ -161,12 +171,11 @@ class ModelState(ModelStateBase): self._run_policy = jits[(cam_w, cam_h)]['run_policy'] self._warp_enqueue = jits[(cam_w, cam_h)]['warp_enqueue'] - road_name = next(k for k in self._vision_input_names if 'big' not in k) - yuv_size = self.frame_buf_params[road_name][3] + yuv_size = self.frame_buf_params[self._road_key][3] self._warp_enqueue( **self.input_queues, - frame=Tensor(np.zeros(yuv_size, dtype=np.uint8), device=self.DEV).contiguous().realize(), - big_frame=Tensor(np.zeros(yuv_size, dtype=np.uint8), device=self.DEV).contiguous().realize()) + frame=Tensor(np.zeros(yuv_size, dtype=np.uint8), device=self.WARP_DEV).contiguous().realize(), + big_frame=Tensor(np.zeros(yuv_size, dtype=np.uint8), device=self.WARP_DEV).contiguous().realize()) @property @@ -179,7 +188,7 @@ class ModelState(ModelStateBase): @property def desire_key(self) -> str: - return next(k for k in self.numpy_inputs if k.startswith('desire')) + return self._desire_key def run(self, bufs: dict[str, VisionBuf], transforms: dict[str, np.ndarray], inputs: dict[str, np.ndarray], prepare_only: bool) -> dict[str, np.ndarray] | None: @@ -190,19 +199,19 @@ class ModelState(ModelStateBase): yuv_size = self.frame_buf_params[key][3] cache_key = (key, ptr) if cache_key not in self._blob_cache: - self._blob_cache[cache_key] = Tensor.from_blob(ptr, (yuv_size,), dtype='uint8', device=self.DEV) + self._blob_cache[cache_key] = Tensor.from_blob(ptr, (yuv_size,), dtype='uint8', device=self.WARP_DEV) self.full_frames[key] = self._blob_cache[cache_key] desire_key = self.desire_key inputs[desire_key][0] = 0 self.numpy_inputs[desire_key][:] = np.where(inputs[desire_key] - self.prev_desire > .99, inputs[desire_key], 0) self.prev_desire[:] = inputs[desire_key] - for key in ('traffic_convention', 'lateral_control_params'): + for key in ('traffic_convention', 'lateral_control_params', 'action_t'): if key in self.numpy_inputs and key in inputs: self.numpy_inputs[key][:] = inputs[key] - road_key = next(n for n in bufs if 'big' not in n) - wide_key = next(n for n in bufs if 'big' in n) + road_key = self._road_key + wide_key = self._wide_key self.numpy_inputs['tfm'][:, :] = transforms[road_key].reshape(3, 3) self.numpy_inputs['big_tfm'][:, :] = transforms[wide_key].reshape(3, 3) @@ -225,8 +234,12 @@ class ModelState(ModelStateBase): policy_output = raw_outputs[i + 1].numpy().flatten() policy_sliced = {k: policy_output[np.newaxis, v] for k, v in policy_slices.items()} parsed = self.parser.parse_policy_outputs(policy_sliced) - if 'off' in self._policy_keys[i] and self._has_on_policy: + if ('off' in self._policy_keys[i] + and self._has_on_policy + and any('plan' in self._policy_slices_list[j] for j, k in enumerate(self._policy_keys) if 'on' in k.lower())): + parsed.pop('plan', None) + outputs.update(parsed) if 'planplus' in outputs and 'plan' in outputs: @@ -241,13 +254,20 @@ class ModelState(ModelStateBase): def get_action_from_model(self, model_output: dict[str, np.ndarray], prev_action: log.ModelDataV2.Action, lat_action_t: float, long_action_t: float, v_ego: float) -> log.ModelDataV2.Action: - plan = model_output['plan'][0] - desired_accel, should_stop = get_accel_from_plan(plan[:, Plan.VELOCITY][:, 0], plan[:, Plan.ACCELERATION][:, 0], self.constants.T_IDXS, - action_t=long_action_t) - desired_accel = smooth_value(desired_accel, prev_action.desiredAcceleration, self.LONG_SMOOTH_SECONDS) + if 'action' not in model_output: + plan = model_output['plan'][0] + desired_accel, should_stop = get_accel_from_plan(plan[:, Plan.VELOCITY][:, 0], plan[:, Plan.ACCELERATION][:, 0], self.constants.T_IDXS, + action_t=long_action_t) + desired_accel = smooth_value(desired_accel, prev_action.desiredAcceleration, self.LONG_SMOOTH_SECONDS) + + curvature_plan = (plan + (self.PLANPLUS_CONTROL - 1.0) * model_output['planplus'][0] + if 'planplus' in model_output and self.PLANPLUS_CONTROL != 1.0 else plan) + desired_curvature = get_curvature_from_output(model_output, curvature_plan, v_ego, lat_action_t, self.mlsim) + else: + desired_accel = model_output['action'][0, 1] + desired_curvature = model_output['action'][0, 0] / (max(1.0, v_ego))**2 + should_stop = (v_ego < 0.3 and desired_accel < 0.1) - curvature_plan = plan + (self.PLANPLUS_CONTROL - 1.0) * model_output['planplus'][0] if 'planplus' in model_output and self.PLANPLUS_CONTROL != 1.0 else plan - desired_curvature = get_curvature_from_output(model_output, curvature_plan, v_ego, lat_action_t, self.mlsim) if self.generation is not None and self.generation >= 10: # smooth curvature for post FOF models if v_ego > self.MIN_LAT_CONTROL_SPEED: desired_curvature = smooth_value(desired_curvature, prev_action.desiredCurvature, self.LAT_SMOOTH_SECONDS) @@ -400,6 +420,12 @@ def main(demo=False): bufs = {name: buf_extra if 'big' in name else buf_main for name in model.vision_input_names} transforms = {name: model_transform_extra if 'big' in name else model_transform_main for name in model.vision_input_names} + + frame_delay = DT_MDL # compensate for time passed since the frame was captured: current_time - timestamp_eof is 50ms on average + action_delay = DT_MDL / 2 # middle of the interval between model output (current state) and next frame (expected state) + lat_action_t = lat_delay + frame_delay + action_delay + long_action_t = long_delay + frame_delay + action_delay + inputs:dict[str, np.ndarray] = { model.desire_key: vec_desire, 'traffic_convention': traffic_convention, @@ -408,6 +434,9 @@ def main(demo=False): if 'lateral_control_params' in model.numpy_inputs: inputs['lateral_control_params'] = np.array([v_ego, lat_delay], dtype=np.float32) + if 'action_t' in model.numpy_inputs: + inputs['action_t'] = np.array([lat_action_t, long_action_t], dtype=np.float32) + mt1 = time.perf_counter() model_output = model.run(bufs, transforms, inputs, prepare_only) mt2 = time.perf_counter() @@ -419,7 +448,7 @@ def main(demo=False): posenet_send = messaging.new_message('cameraOdometry') mdv2sp_send = messaging.new_message('modelDataV2SP') - action = model.get_action_from_model(model_output, prev_action, lat_delay + DT_MDL, long_delay + DT_MDL, v_ego) + action = model.get_action_from_model(model_output, prev_action, lat_action_t, long_action_t, v_ego) prev_action = action fill_model_msg(drivingdata_send, modelv2_send, model_output, action, publish_state, meta_main.frame_id, meta_extra.frame_id, frame_id, diff --git a/openpilot/sunnypilot/modeld_v2/parse_model_outputs.py b/openpilot/sunnypilot/modeld_v2/parse_model_outputs.py index 82103283f3..7a3adcc1fa 100644 --- a/openpilot/sunnypilot/modeld_v2/parse_model_outputs.py +++ b/openpilot/sunnypilot/modeld_v2/parse_model_outputs.py @@ -1,13 +1,16 @@ import numpy as np from openpilot.sunnypilot.modeld_v2.constants import ModelConstants + def safe_exp(x, out=None): # -11 is around 10**14, more causes float16 overflow return np.exp(np.clip(x, -np.inf, 11), out=out) + def sigmoid(x): return 1. / (1. + safe_exp(-x)) + def softmax(x, axis=-1): x -= np.max(x, axis=axis, keepdims=True) if x.dtype == np.float32 or x.dtype == np.float64: @@ -17,6 +20,19 @@ def softmax(x, axis=-1): x /= np.sum(x, axis=axis, keepdims=True) return x + +def _infer_mhp(slice_size: int, prod_out_shape: int, max_in_n: int = 16, max_out_n: int = 6) -> tuple[int, int]: + for out_n in range(max_out_n + 1): + per = 2 * prod_out_shape + out_n + if per <= 0: + continue + if slice_size % per == 0: + in_n = slice_size // per + if 1 <= in_n <= max_in_n: + return in_n, out_n + return 1, 0 # single hypothesis, no weights — matches a non-MDN output + + class Parser: def __init__(self, ignore_missing=False): self.ignore_missing = ignore_missing @@ -40,17 +56,22 @@ class Parser: raw = outs[name] outs[name] = sigmoid(raw) - def parse_mdn(self, name, outs, in_N=0, out_N=1, out_shape=None): + def parse_mdn(self, name, outs, out_shape, in_N=0, out_N=0): if self.check_missing(outs, name): return raw = outs[name] - raw = raw.reshape((raw.shape[0], max(in_N, 1), -1)) + + if in_N == 0 and out_N == 0: + prod = int(np.prod(out_shape)) + in_N, out_N = _infer_mhp(raw.shape[1], prod) + + raw = raw.reshape((raw.shape[0], in_N, -1)) n_values = (raw.shape[2] - out_N)//2 pred_mu = raw[:,:,:n_values] pred_std = safe_exp(raw[:,:,n_values: 2*n_values]) - if in_N > 1: + if in_N > 1 and out_N > 0: weights = np.zeros((raw.shape[0], in_N, out_N), dtype=raw.dtype) for i in range(out_N): weights[:,:,i - out_N] = softmax(raw[:,:,i - out_N], axis=-1) @@ -61,7 +82,6 @@ class Parser: weights[fidx] = weights[fidx][idxs] pred_mu[fidx] = pred_mu[fidx][idxs] pred_std[fidx] = pred_std[fidx][idxs] - assert out_shape is not None full_shape = tuple([raw.shape[0], in_N] + list(out_shape)) outs[name + '_weights'] = weights outs[name + '_hypotheses'] = pred_mu.reshape(full_shape) @@ -74,37 +94,43 @@ class Parser: idxs = np.argsort(weights[fidx,:,hidx])[::-1] pred_mu_final[fidx, hidx] = pred_mu[fidx, idxs[0]] pred_std_final[fidx, hidx] = pred_std[fidx, idxs[0]] + elif in_N > 1 and out_N == 0: + # MHP without weights: keep every hypothesis intact, surface them as + # ``*_hypotheses`` and propagate the full multi-hypothesis tensor. + full_shape = tuple([raw.shape[0], in_N] + list(out_shape)) + outs[name + '_hypotheses'] = pred_mu.reshape(full_shape) + outs[name + '_stds_hypotheses'] = pred_std.reshape(full_shape) + pred_mu_final = pred_mu + pred_std_final = pred_std else: pred_mu_final = pred_mu pred_std_final = pred_std - if out_N > 1: - assert out_shape is not None - final_shape = tuple([raw.shape[0], out_N] + list(out_shape)) + if out_N > 1 or (in_N > 1 and out_N == 0): + n_selections = out_N if out_N > 1 else in_N + final_shape = tuple([raw.shape[0], n_selections] + list(out_shape)) else: - assert out_shape is not None final_shape = tuple([raw.shape[0],] + list(out_shape)) outs[name] = pred_mu_final.reshape(final_shape) outs[name + '_stds'] = pred_std_final.reshape(final_shape) def parse_outputs(self, outs: dict[str, np.ndarray]) -> dict[str, np.ndarray]: - self.parse_mdn('plan', outs, in_N=ModelConstants.PLAN_MHP_N, out_N=ModelConstants.PLAN_MHP_SELECTION, - out_shape=(ModelConstants.IDX_N,ModelConstants.PLAN_WIDTH)) - self.parse_mdn('lane_lines', outs, in_N=0, out_N=0, out_shape=(ModelConstants.NUM_LANE_LINES,ModelConstants.IDX_N,ModelConstants.LANE_LINES_WIDTH)) - self.parse_mdn('road_edges', outs, in_N=0, out_N=0, out_shape=(ModelConstants.NUM_ROAD_EDGES,ModelConstants.IDX_N,ModelConstants.LANE_LINES_WIDTH)) - self.parse_mdn('pose', outs, in_N=0, out_N=0, out_shape=(ModelConstants.POSE_WIDTH,)) - self.parse_mdn('road_transform', outs, in_N=0, out_N=0, out_shape=(ModelConstants.POSE_WIDTH,)) + # supercombo (4955 / 102) and newer variants (e.g. 990 / 144). + self.parse_mdn('plan', outs, out_shape=(ModelConstants.IDX_N, ModelConstants.PLAN_WIDTH)) + self.parse_mdn('lane_lines', outs, out_shape=(ModelConstants.NUM_LANE_LINES, ModelConstants.IDX_N, ModelConstants.LANE_LINES_WIDTH)) + self.parse_mdn('road_edges', outs, out_shape=(ModelConstants.NUM_ROAD_EDGES, ModelConstants.IDX_N, ModelConstants.LANE_LINES_WIDTH)) + self.parse_mdn('pose', outs, out_shape=(ModelConstants.POSE_WIDTH,)) + self.parse_mdn('road_transform', outs, out_shape=(ModelConstants.POSE_WIDTH,)) if 'sim_pose' in outs: - self.parse_mdn('sim_pose', outs, in_N=0, out_N=0, out_shape=(ModelConstants.POSE_WIDTH,)) - self.parse_mdn('wide_from_device_euler', outs, in_N=0, out_N=0, out_shape=(ModelConstants.WIDE_FROM_DEVICE_WIDTH,)) - self.parse_mdn('lead', outs, in_N=ModelConstants.LEAD_MHP_N, out_N=ModelConstants.LEAD_MHP_SELECTION, - out_shape=(ModelConstants.LEAD_TRAJ_LEN,ModelConstants.LEAD_WIDTH)) + self.parse_mdn('sim_pose', outs, out_shape=(ModelConstants.POSE_WIDTH,)) + self.parse_mdn('wide_from_device_euler', outs, out_shape=(ModelConstants.WIDE_FROM_DEVICE_WIDTH,)) + self.parse_mdn('lead', outs, out_shape=(ModelConstants.LEAD_TRAJ_LEN, ModelConstants.LEAD_WIDTH)) if 'lat_planner_solution' in outs: - self.parse_mdn('lat_planner_solution', outs, in_N=0, out_N=0, out_shape=(ModelConstants.IDX_N,ModelConstants.LAT_PLANNER_SOLUTION_WIDTH)) + self.parse_mdn('lat_planner_solution', outs, out_shape=(ModelConstants.IDX_N, ModelConstants.LAT_PLANNER_SOLUTION_WIDTH)) if 'desired_curvature' in outs: - self.parse_mdn('desired_curvature', outs, in_N=0, out_N=0, out_shape=(ModelConstants.DESIRED_CURV_WIDTH,)) + self.parse_mdn('desired_curvature', outs, out_shape=(ModelConstants.DESIRED_CURV_WIDTH,)) for k in ['lead_prob', 'lane_lines_prob', 'meta']: self.parse_binary_crossentropy(k, outs) self.parse_categorical_crossentropy('desire_state', outs, out_shape=(ModelConstants.DESIRE_PRED_WIDTH,)) - self.parse_categorical_crossentropy('desire_pred', outs, out_shape=(ModelConstants.DESIRE_PRED_LEN,ModelConstants.DESIRE_PRED_WIDTH)) + self.parse_categorical_crossentropy('desire_pred', outs, out_shape=(ModelConstants.DESIRE_PRED_LEN, ModelConstants.DESIRE_PRED_WIDTH)) return outs diff --git a/openpilot/sunnypilot/modeld_v2/parse_model_outputs_split.py b/openpilot/sunnypilot/modeld_v2/parse_model_outputs_split.py index 6bd2a70eed..3db47aee42 100644 --- a/openpilot/sunnypilot/modeld_v2/parse_model_outputs_split.py +++ b/openpilot/sunnypilot/modeld_v2/parse_model_outputs_split.py @@ -123,7 +123,7 @@ class Parser: self.parse_categorical_crossentropy('desire_state', outs, out_shape=(SplitModelConstants.DESIRE_PRED_WIDTH,)) if 'lane_lines' in outs: self.parse_mdn('lane_lines', outs, in_N=0, out_N=0, - out_shape=(SplitModelConstants.NUM_LANE_LINES,SplitModelConstants.IDX_N,SplitModelConstants.LANE_LINES_WIDTH)) + out_shape=(SplitModelConstants.NUM_LANE_LINES,SplitModelConstants.IDX_N,SplitModelConstants.LANE_LINES_WIDTH)) if 'lane_lines_prob' in outs: self.parse_binary_crossentropy('lane_lines_prob', outs) if 'lead_prob' in outs: @@ -134,9 +134,11 @@ class Parser: self.parse_binary_crossentropy('meta', outs) if 'road_edges' in outs: self.parse_mdn('road_edges', outs, in_N=0, out_N=0, - out_shape=(SplitModelConstants.NUM_ROAD_EDGES,SplitModelConstants.IDX_N,SplitModelConstants.LANE_LINES_WIDTH)) + out_shape=(SplitModelConstants.NUM_ROAD_EDGES,SplitModelConstants.IDX_N,SplitModelConstants.LANE_LINES_WIDTH)) if 'sim_pose' in outs: self.parse_mdn('sim_pose', outs, in_N=0, out_N=0, out_shape=(SplitModelConstants.POSE_WIDTH,)) + if 'action' in outs: + self.parse_mdn('action', outs, in_N=0, out_N=0, out_shape=(SplitModelConstants.ACTION_WIDTH,)) def parse_vision_outputs(self, outs: dict[str, np.ndarray]) -> dict[str, np.ndarray]: self.parse_mdn('pose', outs, in_N=0, out_N=0, out_shape=(SplitModelConstants.POSE_WIDTH,)) diff --git a/openpilot/sunnypilot/modeld_v2/tests/test_combined_pkl_loader.py b/openpilot/sunnypilot/modeld_v2/tests/test_combined_pkl_loader.py index b13a7abecf..0ae957a4c9 100644 --- a/openpilot/sunnypilot/modeld_v2/tests/test_combined_pkl_loader.py +++ b/openpilot/sunnypilot/modeld_v2/tests/test_combined_pkl_loader.py @@ -67,22 +67,14 @@ class TestStockEquivalence: state = model_state_factory(ARCHETYPES['vision_policy_split']) frame_skip = derive_frame_skip(SPLIT_VISION_INPUT_SHAPES, SPLIT_POLICY_INPUT_SHAPES) - # action_t is a deep-model prerequisite the SP loader doesn't provide yet; see skip_keys below stock_shapes = {**SPLIT_VISION_INPUT_SHAPES, **SPLIT_POLICY_INPUT_SHAPES, 'action_t': (1, 2)} stock_queues, stock_npy = make_input_queues(stock_shapes, frame_skip, device='NPY') - # TODO-SP: remove action_t skip once SP adds prerequisite for deep models (action_t input queue) - # prev_feat is a stock QCOM corruption workaround handled inside the SP loader's JIT path - skip_keys = {'action_t', 'prev_feat'} - # stock packs the per-key policy inputs into packed_npy_inputs; the npy views carry the individual keys - stock_queue_keys = set(stock_queues.keys()) - if 'packed_npy_inputs' in stock_queue_keys: - stock_queue_keys.remove('packed_npy_inputs') - stock_queue_keys |= set(stock_npy.keys()) - assert set(state.input_queues.keys()) == stock_queue_keys - skip_keys, \ - f"Queue keys differ: v2={set(state.input_queues.keys())}, stock={stock_queue_keys}" - assert set(state.numpy_inputs.keys()) == set(stock_npy.keys()) - skip_keys, \ - f"Npy keys differ: v2={set(state.numpy_inputs.keys())}, stock={set(stock_npy.keys())}" + assert set(state.input_queues.keys()) - {'desire', 'traffic_convention'} == \ + set(stock_queues.keys()) - {'packed_npy_inputs'} + assert {'desire', 'traffic_convention'} <= set(state.input_queues.keys()) + # We generate action_t and prev_feat dynamically based on the metadata + assert set(state.numpy_inputs.keys()) == set(stock_npy.keys()) - {'action_t', 'prev_feat'} def test_split_queue_keys_work_with_desire_key(self, model_state_factory): from openpilot.sunnypilot.modeld_v2.compile_modeld import derive_frame_skip, make_split_input_queues diff --git a/openpilot/sunnypilot/modeld_v2/tests/test_warp.py b/openpilot/sunnypilot/modeld_v2/tests/test_warp.py deleted file mode 100644 index 49dc634a4d..0000000000 --- a/openpilot/sunnypilot/modeld_v2/tests/test_warp.py +++ /dev/null @@ -1,103 +0,0 @@ -import os -os.environ['DEV'] = 'CPU' -import pytest -import numpy as np -from openpilot.system.camerad.cameras.nv12_info import get_nv12_info -from openpilot.sunnypilot.modeld_v2.warp import CAMERA_CONFIGS -from openpilot.sunnypilot.modeld_v2.warp import Warp, MODEL_W, MODEL_H - -VISION_NAME_PAIRS = [ # needed to account for supercombos input_imgs - ('img', 'big_img'), - ('input_imgs', 'big_input_imgs'), -] - - -class MockVisionBuf: - def __init__(self, w, h): - self.width = w - self.height = h - _, _, _, yuv_size = get_nv12_info(w, h) - self.data = np.zeros(yuv_size, dtype=np.uint8) - - -@pytest.mark.parametrize("buffer_length", [2, 5]) -def test_warp_initialization(buffer_length): - warp = Warp(buffer_length) - assert warp.buffer_length == buffer_length - assert warp.img_buffer_shape == (buffer_length * 6, MODEL_H // 2, MODEL_W // 2) - - -@pytest.mark.parametrize("buffer_length", [2, 5]) -@pytest.mark.parametrize("cam_w, cam_h", CAMERA_CONFIGS) -@pytest.mark.parametrize("road, wide", VISION_NAME_PAIRS) -def test_warp_process(buffer_length, cam_w, cam_h, road, wide): - warp = Warp(buffer_length) - mock_buf = MockVisionBuf(cam_w, cam_h) - transform = np.eye(3, dtype=np.float32).flatten() - bufs = {road: mock_buf, wide: mock_buf} - transforms = {road: transform, wide: transform} - - out = warp.process(bufs, transforms) - assert isinstance(out, dict) - assert road in out and wide in out - assert out[road].shape == (1, 12, MODEL_H // 2, MODEL_W // 2) - assert out[wide].shape == (1, 12, MODEL_H // 2, MODEL_W // 2) - - key = (cam_w, cam_h) - assert key in warp.jit_cache - - out2 = warp.process(bufs, transforms) - assert out2[road].shape == out[road].shape - - -@pytest.mark.parametrize("road, wide", VISION_NAME_PAIRS) -def test_warp_buffer_shift(road, wide): - warp = Warp(2) - cam_w, cam_h = CAMERA_CONFIGS[1] - transform = np.eye(3, dtype=np.float32).flatten() - - buf1 = MockVisionBuf(cam_w, cam_h) - buf1.data[0] = 255 - bufs1 = {road: buf1, wide: buf1} - transforms = {road: transform, wide: transform} - out1 = warp.process(bufs1, transforms) - road1 = out1[road].numpy().copy() - - buf2 = MockVisionBuf(cam_w, cam_h) - buf2.data[0] = 128 - bufs2 = {road: buf2, wide: buf2} - out2 = warp.process(bufs2, transforms) - assert not np.array_equal(road1, out2[road].numpy()) - - -@pytest.mark.parametrize("buffer_length", [2, 5]) -@pytest.mark.parametrize("road, wide", VISION_NAME_PAIRS) -def test_warp_buffer_accumulation(buffer_length, road, wide): - warp = Warp(buffer_length) - cam_w, cam_h = CAMERA_CONFIGS[0] - transform = np.eye(3, dtype=np.float32).flatten() - transforms = {road: transform, wide: transform} - outputs = [] - - for i in range(buffer_length + 1): - buf = MockVisionBuf(cam_w, cam_h) - buf.data[:] = i * 10 - out = warp.process({road: buf, wide: buf}, transforms) - outputs.append(out[road].numpy().copy()) - - assert warp.full_buffers['img'].shape == (buffer_length * 6, MODEL_H // 2, MODEL_W // 2) - for i in range(1, len(outputs)): - assert not np.array_equal(outputs[i - 1], outputs[i]) - - -def test_warp_different_cameras_same_instance(): - warp = Warp(2) - transform = np.eye(3, dtype=np.float32).flatten() - - buf1 = MockVisionBuf(*CAMERA_CONFIGS[0]) - warp.process({'img': buf1, 'big_img': buf1}, {'img': transform, 'big_img': transform}) - assert len(warp.jit_cache) == 1 - - buf2 = MockVisionBuf(*CAMERA_CONFIGS[1]) - warp.process({'img': buf2, 'big_img': buf2}, {'img': transform, 'big_img': transform}) - assert len(warp.jit_cache) == 2 diff --git a/openpilot/sunnypilot/modeld_v2/warp.py b/openpilot/sunnypilot/modeld_v2/warp.py deleted file mode 100644 index 29d1925f8e..0000000000 --- a/openpilot/sunnypilot/modeld_v2/warp.py +++ /dev/null @@ -1,171 +0,0 @@ -import pickle -import time -import numpy as np -from pathlib import Path -from tinygrad.tensor import Tensor -from tinygrad.engine.jit import TinyJit -from tinygrad.device import Device - -from openpilot.system.camerad.cameras.nv12_info import get_nv12_info -from openpilot.common.transformations.model import MEDMODEL_INPUT_SIZE -from openpilot.common.transformations.camera import _ar_ox_fisheye, _os_fisheye -from openpilot.selfdrive.modeld.compile_modeld import NV12Frame, make_frame_prepare as _make_frame_prepare - -CAMERA_CONFIGS = [ - (_ar_ox_fisheye.width, _ar_ox_fisheye.height), - (_os_fisheye.width, _os_fisheye.height), -] - - -def make_frame_prepare(cam_w, cam_h, model_w, model_h): - nv12 = NV12Frame(cam_w, cam_h, *get_nv12_info(cam_w, cam_h)) - return _make_frame_prepare(nv12, model_w, model_h) - - -def warp_pkl_path(w, h): - from openpilot.selfdrive.modeld.helpers import MODELS_DIR - return MODELS_DIR / f'warp_{w}x{h}_tinygrad.pkl' - - -def make_update_img_input(frame_prepare, model_w, model_h): - def update_img_input_tinygrad(tensor, frame, M_inv): - M_inv = M_inv.to(Device.DEFAULT) - new_img = frame_prepare(frame, M_inv) - tensor.assign(tensor[6:].cat(new_img, dim=0).contiguous()) - return Tensor.cat(tensor[:6], tensor[-6:], dim=0).contiguous().reshape(1, 12, model_h//2, model_w//2) - return update_img_input_tinygrad - - -def make_update_both_imgs(frame_prepare, model_w, model_h): - update_img = make_update_img_input(frame_prepare, model_w, model_h) - def update_both_imgs_tinygrad(calib_img_buffer, new_img, M_inv, - calib_big_img_buffer, new_big_img, M_inv_big): - calib_img_pair = update_img(calib_img_buffer, new_img, M_inv) - calib_big_img_pair = update_img(calib_big_img_buffer, new_big_img, M_inv_big) - return calib_img_pair, calib_big_img_pair - return update_both_imgs_tinygrad - -MODELS_DIR = Path(__file__).parent / 'models' -MODEL_W, MODEL_H = MEDMODEL_INPUT_SIZE -UPSTREAM_BUFFER_LENGTH = 5 - - -def v2_warp_pkl_path(cam_w, cam_h, buffer_length): - return MODELS_DIR / f'warp_{cam_w}x{cam_h}_b{buffer_length}_tinygrad.pkl' - - -def compile_v2_warp(cam_w, cam_h, buffer_length): - _, _, _, yuv_size = get_nv12_info(cam_w, cam_h) - img_buffer_shape = (buffer_length * 6, MODEL_H // 2, MODEL_W // 2) - - print(f"Compiling v2 warp for {cam_w}x{cam_h} buffer_length={buffer_length}...") - - frame_prepare = make_frame_prepare(cam_w, cam_h, MODEL_W, MODEL_H) - update_both_imgs = make_update_both_imgs(frame_prepare, MODEL_W, MODEL_H) - update_img_jit = TinyJit(update_both_imgs, prune=True) - - full_buffer = Tensor.zeros(img_buffer_shape, dtype='uint8').contiguous().realize() - big_full_buffer = Tensor.zeros(img_buffer_shape, dtype='uint8').contiguous().realize() - new_frame_np = np.random.default_rng(0).integers(0, 256, yuv_size, dtype=np.uint8) - new_big_frame_np = np.random.default_rng(1).integers(0, 256, yuv_size, dtype=np.uint8) - for i in range(10): - img_inputs = [full_buffer, - Tensor.from_blob(new_frame_np.ctypes.data, (yuv_size,), dtype='uint8').realize(), - Tensor(Tensor.randn(3, 3).mul(8).realize().numpy(), device='NPY')] - big_img_inputs = [big_full_buffer, - Tensor.from_blob(new_big_frame_np.ctypes.data, (yuv_size,), dtype='uint8').realize(), - Tensor(Tensor.randn(3, 3).mul(8).realize().numpy(), device='NPY')] - inputs = img_inputs + big_img_inputs - Device.default.synchronize() - - st = time.perf_counter() - _ = update_img_jit(*inputs) - mt = time.perf_counter() - Device.default.synchronize() - et = time.perf_counter() - print(f" [{i+1}/10] enqueue {(mt-st)*1e3:6.2f} ms -- total {(et-st)*1e3:6.2f} ms") - - pkl_path = v2_warp_pkl_path(cam_w, cam_h, buffer_length) - with open(pkl_path, "wb") as f: - pickle.dump(update_img_jit, f) - print(f" Saved to {pkl_path}") - - jit = pickle.load(open(pkl_path, "rb")) - verify_frame = np.random.default_rng(0).integers(0, 256, yuv_size, dtype=np.uint8) - verify_big_frame = np.random.default_rng(1).integers(0, 256, yuv_size, dtype=np.uint8) - fresh_inputs = [ - Tensor.zeros(img_buffer_shape, dtype='uint8').contiguous().realize(), - Tensor.from_blob(verify_frame.ctypes.data, (yuv_size,), dtype='uint8').realize(), - Tensor(Tensor.randn(3, 3).mul(8).realize().numpy(), device='NPY'), - Tensor.zeros(img_buffer_shape, dtype='uint8').contiguous().realize(), - Tensor.from_blob(verify_big_frame.ctypes.data, (yuv_size,), dtype='uint8').realize(), - Tensor(Tensor.randn(3, 3).mul(8).realize().numpy(), device='NPY'), - ] - jit(*fresh_inputs) - - -class Warp: - def __init__(self, buffer_length=2): - self.buffer_length = buffer_length - self.img_buffer_shape = (buffer_length * 6, MODEL_H // 2, MODEL_W // 2) - - self.jit_cache = {} - self.full_buffers = {k: Tensor.zeros(self.img_buffer_shape, dtype='uint8').contiguous().realize() for k in ['img', 'big_img']} - self._blob_cache: dict[int, Tensor] = {} - self._nv12_cache: dict[tuple[int, int], int] = {} - self.transforms_np = {k: np.zeros((3, 3), dtype=np.float32) for k in ['img', 'big_img']} - self.transforms = {k: Tensor(v, device='NPY').realize() for k, v in self.transforms_np.items()} - - def process(self, bufs, transforms): - if not bufs: - return {} - road = next(n for n in bufs if 'big' not in n) - wide = next(n for n in bufs if 'big' in n) - cam_w, cam_h = bufs[road].width, bufs[road].height - key = (cam_w, cam_h) - - if key not in self.jit_cache: - v2_pkl = v2_warp_pkl_path(cam_w, cam_h, self.buffer_length) - if v2_pkl.exists(): - with open(v2_pkl, 'rb') as f: - self.jit_cache[key] = pickle.load(f) - elif self.buffer_length == UPSTREAM_BUFFER_LENGTH: - upstream_pkl = warp_pkl_path(cam_w, cam_h) - if upstream_pkl.exists(): - with open(upstream_pkl, 'rb') as f: - self.jit_cache[key] = pickle.load(f) - if key not in self.jit_cache: - frame_prepare = make_frame_prepare(cam_w, cam_h, MODEL_W, MODEL_H) - update_both_imgs = make_update_both_imgs(frame_prepare, MODEL_W, MODEL_H) - self.jit_cache[key] = TinyJit(update_both_imgs, prune=True) - - if key not in self._nv12_cache: - self._nv12_cache[key] = get_nv12_info(cam_w, cam_h)[3] - yuv_size = self._nv12_cache[key] - - road_ptr = bufs[road].data.ctypes.data - wide_ptr = bufs[wide].data.ctypes.data - if road_ptr not in self._blob_cache: - self._blob_cache[road_ptr] = Tensor.from_blob(road_ptr, (yuv_size,), dtype='uint8') - if wide_ptr not in self._blob_cache: - self._blob_cache[wide_ptr] = Tensor.from_blob(wide_ptr, (yuv_size,), dtype='uint8') - road_blob = self._blob_cache[road_ptr] - wide_blob = self._blob_cache[wide_ptr] if wide_ptr != road_ptr else Tensor.from_blob(wide_ptr, (yuv_size,), dtype='uint8') - np.copyto(self.transforms_np['img'], transforms[road].reshape(3, 3)) - np.copyto(self.transforms_np['big_img'], transforms[wide].reshape(3, 3)) - - Device.default.synchronize() - res = self.jit_cache[key]( - self.full_buffers['img'], road_blob, self.transforms['img'], - self.full_buffers['big_img'], wide_blob, self.transforms['big_img'], - ) - out_road = res[0].realize() - out_wide = res[1].realize() - - return {road: out_road, wide: out_wide} - - -if __name__ == "__main__": - for cam_w, cam_h in CAMERA_CONFIGS: - for bl in [2, 5]: - compile_v2_warp(cam_w, cam_h, bl) diff --git a/openpilot/sunnypilot/models/fetcher.py b/openpilot/sunnypilot/models/fetcher.py index b5197988bb..eda1117a2a 100644 --- a/openpilot/sunnypilot/models/fetcher.py +++ b/openpilot/sunnypilot/models/fetcher.py @@ -6,11 +6,12 @@ See the LICENSE.md file in the root directory for more details. """ import time - +import os import requests from requests.exceptions import (SSLError, RequestException, HTTPError) from openpilot.common.params import Params from openpilot.common.swaglog import cloudlog +from openpilot.common.hardware.hw import Paths from openpilot.sunnypilot.models.helpers import is_bundle_version_compatible from openpilot.cereal import custom @@ -26,11 +27,35 @@ class ModelParser: download_uri.sha256 = download_uri_data.get("sha256") return download_uri + @staticmethod + def _parse_chunk(chunk_data) -> custom.ModelManagerSP.Chunk: + chunk = custom.ModelManagerSP.Chunk() + chunk.fileName = chunk_data.get("file_name") + chunk.sha256 = chunk_data.get("sha256") + return chunk + @staticmethod def _parse_artifact(artifact_data) -> custom.ModelManagerSP.Artifact: artifact = custom.ModelManagerSP.Artifact() artifact.fileName = artifact_data.get("file_name") artifact.downloadUri = ModelParser._parse_download_uri(artifact_data.get("download_uri", {})) + + if "chunks" in artifact_data: + artifact.chunks = [ModelParser._parse_chunk(chunk_data) for chunk_data in artifact_data["chunks"]] + + try: + model_dir = Paths.model_root() + os.makedirs(model_dir, exist_ok=True) + manifest_path = os.path.join(model_dir, f"{artifact.fileName}.chunkmanifest") + num_chunks = str(len(artifact.chunks)) + + if not os.path.exists(manifest_path) or open(manifest_path).read().strip() != num_chunks: + with open(manifest_path, "w") as f: + f.write(num_chunks) + cloudlog.info(f"Wrote chunk manifest for {artifact.fileName}: {num_chunks} chunks") + except Exception as e: + cloudlog.warning(f"Failed to write chunk manifest for {artifact.fileName}: {e}") + return artifact @staticmethod @@ -39,8 +64,6 @@ class ModelParser: model.type = model_data.get("type") model.artifact = ModelParser._parse_artifact(model_data.get("artifact", {})) - if metadata := model_data.get("metadata"): - model.metadata = ModelParser._parse_artifact(metadata) return model @staticmethod @@ -116,7 +139,7 @@ class ModelCache: class ModelFetcher: """Handles fetching and caching of model data from remote source""" - MODEL_URL = "https://raw.githubusercontent.com/sunnypilot/sunnypilot-models/refs/heads/gh-pages/docs/driving_models_v17.json" + MODEL_URL = "https://raw.githubusercontent.com/sunnypilot/sunnypilot-models/refs/heads/gh-pages/docs/driving_models_v18.json" def __init__(self, params: Params): self.params = params @@ -184,4 +207,5 @@ if __name__ == "__main__": # Print artifact details print(f"Artifact: {model.artifact.fileName}, Download URI: {model.artifact.downloadUri.uri}") # Print metadata details - print(f"Metadata: {model.metadata.fileName}, Download URI: {model.metadata.downloadUri.uri}") + if model.artifact.chunks: + print(f"Contains {len(model.artifact.chunks)} chunks.") diff --git a/openpilot/sunnypilot/models/helpers.py b/openpilot/sunnypilot/models/helpers.py index ad1333d4ed..101d8d196e 100644 --- a/openpilot/sunnypilot/models/helpers.py +++ b/openpilot/sunnypilot/models/helpers.py @@ -18,7 +18,7 @@ from openpilot.sunnypilot.models.constants import Meta, MetaSimPose, MetaTombRai from openpilot.common.hardware.hw import Paths # SET ME TO THE EXACT JSON VERSION WE SET IN SUNNYPILOT_MODELS REPO -REQUIRED_JSON_VERSION = 15 +REQUIRED_JSON_VERSION = 16 CUSTOM_MODEL_PATH = Paths.model_root() METADATA_PATH = Path(__file__).parent / '../models/supercombo_metadata.pkl' @@ -56,12 +56,20 @@ def is_bundle_version_compatible(bundle: dict) -> bool: def _bundle_artifacts(bundle: custom.ModelManagerSP.ModelBundle) -> list[tuple[str, str]]: artifacts = [] + from openpilot.common.file_chunker import get_chunk_name for model in getattr(bundle, 'models', []) or []: - for artifact in (getattr(model, 'artifact', None), getattr(model, 'metadata', None)): - if artifact and getattr(artifact, 'fileName', None) and getattr(artifact, 'downloadUri', None): - sha256 = getattr(artifact.downloadUri, 'sha256', None) - if sha256: - artifacts.append((artifact.fileName, sha256)) + for artifact in (getattr(model, 'artifact', None),): + if artifact and getattr(artifact, 'fileName', None): + if len(artifact.chunks) > 0: + for i, chunk in enumerate(artifact.chunks): + chunk_name = get_chunk_name(artifact.fileName, i, len(artifact.chunks)) + if getattr(chunk, 'sha256', None): + artifacts.append((chunk_name, chunk.sha256)) + else: + if getattr(artifact, 'downloadUri', None): + sha256 = getattr(artifact.downloadUri, 'sha256', None) + if sha256: + artifacts.append((artifact.fileName, sha256)) return artifacts @@ -156,8 +164,7 @@ def _get_model(): def load_metadata(): - model = _get_model() - metadata_path = f"{CUSTOM_MODEL_PATH}/{model.metadata.fileName}" if model else METADATA_PATH + metadata_path = METADATA_PATH with open(metadata_path, 'rb') as f: return pickle.load(f) diff --git a/openpilot/sunnypilot/models/manager.py b/openpilot/sunnypilot/models/manager.py index 3338a91711..d742e968a9 100644 --- a/openpilot/sunnypilot/models/manager.py +++ b/openpilot/sunnypilot/models/manager.py @@ -38,11 +38,11 @@ class ModelManagerSP: if not self.selected_bundle: return for model in self.selected_bundle.models: - for artifact in (model.artifact, model.metadata): - if artifact is not source_artifact and artifact.fileName == source_artifact.fileName: - artifact.downloadProgress.status = source_artifact.downloadProgress.status - artifact.downloadProgress.progress = source_artifact.downloadProgress.progress - artifact.downloadProgress.eta = source_artifact.downloadProgress.eta + artifact = model.artifact + if artifact is not source_artifact and artifact.fileName == source_artifact.fileName: + artifact.downloadProgress.status = source_artifact.downloadProgress.status + artifact.downloadProgress.progress = source_artifact.downloadProgress.progress + artifact.downloadProgress.eta = source_artifact.downloadProgress.eta def _calculate_eta(self, filename: str, progress: float) -> int: """Calculate ETA based on elapsed time and current progress""" @@ -89,20 +89,16 @@ class ModelManagerSP: del self._download_start_times[model.fileName] async def _download_chunked(self, base_url: str, base_path: str, artifact) -> None: - from openpilot.common.file_chunker import get_manifest_path, get_chunk_name - manifest_url = get_manifest_path(base_url) + from openpilot.common.file_chunker import get_chunk_name, get_manifest_path + + num_chunks = len(artifact.chunks) + if num_chunks == 0: + raise ValueError("No chunks defined in artifact") + manifest_path = get_manifest_path(base_path) - - async with aiohttp.ClientSession() as session: - async with session.get(manifest_url) as resp: - if resp.status == 404: - raise FileNotFoundError - resp.raise_for_status() - num_chunks = int((await resp.read()).strip()) - self._download_start_times[artifact.fileName] = time.monotonic() - for i in range(num_chunks): + for i, _ in enumerate(artifact.chunks): chunk_url = get_chunk_name(base_url, i, num_chunks) chunk_path = get_chunk_name(base_path, i, num_chunks) chunk_downloaded = 0 @@ -117,7 +113,7 @@ class ModelManagerSP: if self.params.get("ModelManager_DownloadIndex") is None: raise Exception("Download cancelled") intra = chunk_downloaded / max(chunk_size, 1) - progress = min(99, (i + intra) / num_chunks * 100) + progress = min(99.0, ((i + intra) / num_chunks) * 100) artifact.downloadProgress.status = custom.ModelManagerSP.DownloadStatus.downloading artifact.downloadProgress.progress = progress artifact.downloadProgress.eta = self._calculate_eta(artifact.fileName, progress) @@ -140,7 +136,22 @@ class ModelManagerSP: full_path = os.path.join(destination_path, filename) try: - if await verify_file(full_path, expected_hash): + is_cached = False + if len(artifact.chunks) > 0: + from openpilot.common.file_chunker import get_chunk_name + chunks_valid = True + for i, chunk in enumerate(artifact.chunks): + chunk_path = get_chunk_name(full_path, i, len(artifact.chunks)) + if not await verify_file(chunk_path, chunk.sha256): + chunks_valid = False + break + if chunks_valid and len(artifact.chunks) > 0: + is_cached = True + else: + if await verify_file(full_path, expected_hash): + is_cached = True + + if is_cached: artifact.downloadProgress.status = custom.ModelManagerSP.DownloadStatus.cached artifact.downloadProgress.progress = 100 artifact.downloadProgress.eta = 0 @@ -148,13 +159,17 @@ class ModelManagerSP: self._report_status() return - try: + if len(artifact.chunks) > 0: await self._download_chunked(url, full_path, artifact) - except (FileNotFoundError, aiohttp.ClientResponseError): + from openpilot.common.file_chunker import get_chunk_name + for i, chunk in enumerate(artifact.chunks): + chunk_path = get_chunk_name(full_path, i, len(artifact.chunks)) + if not await verify_file(chunk_path, chunk.sha256): + raise ValueError(f"Hash validation failed for chunk {i+1} of {filename}") + else: await self._download_file(url, full_path, artifact) - - if not await verify_file(full_path, expected_hash): - raise ValueError(f"Hash validation failed for {filename}") + if not await verify_file(full_path, expected_hash): + raise ValueError(f"Hash validation failed for {filename}") artifact.downloadProgress.status = custom.ModelManagerSP.DownloadStatus.downloaded artifact.downloadProgress.progress = 100 @@ -170,18 +185,15 @@ class ModelManagerSP: artifact.downloadProgress.status = custom.ModelManagerSP.DownloadStatus.failed artifact.downloadProgress.eta = 0 self._sync_artifact_progress(artifact) - self.selected_bundle.status = custom.ModelManagerSP.DownloadStatus.failed + if self.selected_bundle: + self.selected_bundle.status = custom.ModelManagerSP.DownloadStatus.failed self._report_status() self._download_start_times.pop(artifact.fileName, None) raise async def _process_model(self, model, destination_path: str) -> None: """Processes a single model download including verification""" - model_artifact = model.artifact - metadata_artifact = model.metadata - - await self._process_artifact(metadata_artifact, destination_path) - await self._process_artifact(model_artifact, destination_path) + await self._process_artifact(model.artifact, destination_path) def _report_status(self) -> None: """Reports current status through messaging system""" @@ -205,16 +217,16 @@ class ModelManagerSP: try: seen_artifacts: set[str] = set() for model in self.selected_bundle.models: - for artifact in (model.metadata, model.artifact): - if not artifact.fileName: - continue - if artifact.fileName in seen_artifacts: - artifact.downloadProgress.status = custom.ModelManagerSP.DownloadStatus.cached - artifact.downloadProgress.progress = 100 - artifact.downloadProgress.eta = 0 - else: - seen_artifacts.add(artifact.fileName) - await self._process_artifact(artifact, destination_path) + artifact = model.artifact + if not artifact.fileName: + continue + if artifact.fileName in seen_artifacts: + artifact.downloadProgress.status = custom.ModelManagerSP.DownloadStatus.cached + artifact.downloadProgress.progress = 100 + artifact.downloadProgress.eta = 0 + else: + seen_artifacts.add(artifact.fileName) + await self._process_artifact(artifact, destination_path) self.active_bundle = self.selected_bundle self.active_bundle.status = custom.ModelManagerSP.DownloadStatus.downloaded @@ -275,8 +287,6 @@ class ModelManagerSP: for model in self.active_bundle.models: if hasattr(model, 'artifact') and model.artifact.fileName: active_files.append(model.artifact.fileName) - if hasattr(model, 'metadata') and model.metadata.fileName: - active_files.append(model.metadata.fileName) # Remove all files except active ones (including their chunk files) model_dir = Paths.model_root() diff --git a/openpilot/sunnypilot/models/runners/helpers.py b/openpilot/sunnypilot/models/runners/helpers.py deleted file mode 100644 index b34a62132b..0000000000 --- a/openpilot/sunnypilot/models/runners/helpers.py +++ /dev/null @@ -1,28 +0,0 @@ -from openpilot.sunnypilot.models.helpers import get_active_bundle -from openpilot.sunnypilot.models.runners.model_runner import ModelRunner -from openpilot.sunnypilot.models.runners.tinygrad.tinygrad_runner import TinygradRunner, TinygradSplitRunner -from openpilot.sunnypilot.models.runners.constants import ModelType - - -def get_model_runner() -> ModelRunner: - """ - Factory function to create and return the appropriate ModelRunner instance. - - Selects TinygradRunner, choosing TinygradSplitRunner if separate vision/policy - models are detected in the active bundle. - - :return: An instance of a ModelRunner subclass (ONNXRunner, TinygradRunner, or TinygradSplitRunner). - """ - bundle = get_active_bundle() - if bundle and bundle.models: - model_types = {m.type.raw for m in bundle.models} - # Check if the bundle uses separate vision and policy models (legacy or new split format) - split_types = {ModelType.vision, ModelType.policy, ModelType.offPolicy, ModelType.onPolicy} - if model_types & split_types: - return TinygradSplitRunner() - # Otherwise, assume a single model (likely supercombo) - if bundle.models: - return TinygradRunner(bundle.models[0].type.raw) - - # Default fallback to TinygradRunner with the supercombo type if bundle info is missing/incomplete - return TinygradRunner(ModelType.supercombo) diff --git a/openpilot/sunnypilot/models/runners/model_runner.py b/openpilot/sunnypilot/models/runners/model_runner.py deleted file mode 100644 index cbf2fc5e20..0000000000 --- a/openpilot/sunnypilot/models/runners/model_runner.py +++ /dev/null @@ -1,174 +0,0 @@ -from abc import abstractmethod, ABC - -import numpy as np -from openpilot.sunnypilot.models.helpers import get_active_bundle -from openpilot.sunnypilot.models.runners.constants import NumpyDict, ShapeDict, Model, SliceDict, SEND_RAW_PRED -from openpilot.common.hardware.hw import Paths -import pickle - -CUSTOM_MODEL_PATH = Paths.model_root() - - -class ModelData: - """ - Stores metadata and configuration for a specific machine learning model. - - This class loads model metadata (like input shapes and output slices) - from a pickle file associated with a model instance. - - :param model: The machine learning model object containing metadata. - """ - def __init__(self, model: Model): - self.model = model - self.metadata = model.metadata - self.input_shapes: ShapeDict = {} - self.output_slices: SliceDict = {} - if self.metadata: - self._load_metadata() - - def _load_metadata(self) -> None: - """Loads input shapes and output slices from the model's metadata pickle file.""" - metadata_path = f"{CUSTOM_MODEL_PATH}/{self.metadata.fileName}" - with open(metadata_path, 'rb') as f: - model_metadata = pickle.load(f) - self.input_shapes = model_metadata.get('input_shapes', {}) - self.output_slices = model_metadata.get('output_slices', {}) - - -class ModularRunner(ABC): - """ - Represents a modular runner for handling and slicing model outputs. - - This abstract base class is designed to provide an interface for modular - parsing and processing of model outputs. Classes inheriting from it must - implement the specified abstract methods, defining how model outputs - should be handled and stored. The primary goal is to enable structured - parsing of outputs through a dictionary-based method mapping. - - :ivar parser_method_dict: Mapping dictionary containing parser methods - for handling specific types of outputs. - :type parser_method_dict: dict - """ - - @property - @abstractmethod - def parser_method_dict(self) -> dict: - pass - - @parser_method_dict.setter - @abstractmethod - def parser_method_dict(self, value: dict) -> None: - pass - - @abstractmethod - def _slice_outputs(self, model_outputs: np.ndarray) -> NumpyDict: - pass - - -class ModelRunner(ModularRunner): - """ - Abstract base class for managing and executing machine learning models. - - Provides a common interface for loading models, preparing inputs, running - inference, and slicing/parsing outputs based on model metadata. Derived - classes implement the specifics of input preparation and model execution - for different frameworks (e.g., Tinygrad, ONNX). - """ - - def __init__(self): - """Initializes the model runner, loading the active model bundle.""" - self.is_20hz: bool | None = None - self.is_20hz_3d: bool | None = None - self.models: dict[int, ModelData] = {} - self._model_data: ModelData | None = None # Active model data for current operation - self._parser_method_dict: dict = {} - self.inputs: dict = {} - self._parser = None - self._load_models() - self._constants = None - - @property - def constants(self): - return self._constants - - @property - def parser_method_dict(self) -> dict: - """Returns the dictionary mapping model types to their respective parsing methods.""" - return self._parser_method_dict - - @parser_method_dict.setter - def parser_method_dict(self, value: dict) -> None: - """Sets the dictionary mapping model types to their respective parsing methods.""" - self._parser_method_dict = value - - def _load_models(self) -> None: - """Loads the active model bundle configuration and sets up ModelData.""" - bundle = get_active_bundle() - if not bundle: - raise ValueError("No active model bundle found, why are we being executed?") - - self.models = {model.type.raw: ModelData(model) for model in bundle.models} - self.is_20hz = bundle.is20hz - self.is_20hz_3d = False - - @property - def input_shapes(self) -> ShapeDict: - """Returns the input shapes for the currently active model.""" - if self._model_data: - return self._model_data.input_shapes - raise ValueError("Model data is not available. Ensure the model is loaded correctly.") - - @property - def output_slices(self) -> SliceDict: - """Returns the output slices for the currently active model.""" - if self._model_data: - return self._model_data.output_slices - raise ValueError("Model data is not available. Ensure the model is loaded correctly.") - - @property - def vision_input_names(self) -> list[str]: - """Returns the list of vision input names from the input shapes.""" - if self._model_data: - return list(self._model_data.input_shapes.keys()) - raise ValueError("Model data is not available. Ensure the model is loaded correctly.") - - @abstractmethod - def prepare_inputs(self, numpy_inputs: NumpyDict) -> dict: - """ - Abstract method to prepare inputs for model inference. - - :param numpy_inputs: Dictionary of numpy arrays for non-image inputs. - :return: Dictionary of prepared inputs ready for the model. - """ - raise NotImplementedError - - @abstractmethod - def _run_model(self) -> NumpyDict: - """ - Abstract method to execute model inference with prepared inputs. - - :return: Dictionary containing the model's raw output arrays. - """ - raise NotImplementedError - - def _slice_outputs(self, model_outputs: np.ndarray) -> NumpyDict: - """ - Slices the raw model output array based on the output_slices metadata. - - :param model_outputs: The raw numpy array output from the model. - :return: A dictionary where keys are output names and values are sliced numpy arrays. - """ - if not self._model_data: - raise ValueError("Model data is not available. Ensure the model is loaded correctly.") - sliced_outputs = {k: model_outputs[np.newaxis, v] for k, v in self._model_data.output_slices.items()} - if SEND_RAW_PRED: - sliced_outputs['raw_pred'] = model_outputs.copy() # Optionally include the full raw output - return sliced_outputs - - def run_model(self) -> NumpyDict: - """ - Executes the model inference pipeline: runs the model and parses outputs. - - :return: Dictionary containing the final parsed model outputs. - """ - return self._run_model() # Parsing is handled within specific runner implementations diff --git a/openpilot/sunnypilot/models/runners/tinygrad/model_types.py b/openpilot/sunnypilot/models/runners/tinygrad/model_types.py deleted file mode 100644 index 295e75afb5..0000000000 --- a/openpilot/sunnypilot/models/runners/tinygrad/model_types.py +++ /dev/null @@ -1,91 +0,0 @@ -import os -from abc import ABC - -import numpy as np -from openpilot.sunnypilot.modeld_v2.parse_model_outputs import Parser as CombinedParser -from openpilot.sunnypilot.modeld_v2.parse_model_outputs_split import Parser as SplitParser -from openpilot.sunnypilot.models.runners.constants import ModelType, NumpyDict -from openpilot.sunnypilot.models.runners.model_runner import ModularRunner -from openpilot.common.hardware.hw import Paths - - -SEND_RAW_PRED = os.getenv('SEND_RAW_PRED') -CUSTOM_MODEL_PATH = Paths.model_root() - - -class OffPolicyTinygrad(ModularRunner, ABC): - """ - A TinygradRunner specialized for off-policy models. - - Uses a SplitParser to handle outputs specific to the off-policy part of a split model setup. - """ - def __init__(self): - self._off_policy_parser = SplitParser() - self.parser_method_dict[ModelType.offPolicy] = self._parse_off_policy_outputs - - def _parse_off_policy_outputs(self, model_outputs: np.ndarray) -> NumpyDict: - """Parses off-policy model outputs using SplitParser.""" - result: NumpyDict = self._off_policy_parser.parse_policy_outputs(self._slice_outputs(model_outputs)) - return result - - -class OnPolicyTinygrad(ModularRunner, ABC): - """ - A TinygradRunner specialized for on-policy models. - - Uses a SplitParser to handle outputs specific to the on-policy part of a split model setup. - """ - def __init__(self): - self._on_policy_parser = SplitParser() - self.parser_method_dict[ModelType.onPolicy] = self._parse_on_policy_outputs - - def _parse_on_policy_outputs(self, model_outputs: np.ndarray) -> NumpyDict: - """Parses on-policy model outputs using SplitParser.""" - result: NumpyDict = self._on_policy_parser.parse_policy_outputs(self._slice_outputs(model_outputs)) - return result - - -class PolicyTinygrad(ModularRunner, ABC): - """ - A TinygradRunner specialized for policy-only models. - - Uses a SplitParser to handle outputs specific to the policy part of a split model setup. - """ - def __init__(self): - self._policy_parser = SplitParser() - self.parser_method_dict[ModelType.policy] = self._parse_policy_outputs - - def _parse_policy_outputs(self, model_outputs: np.ndarray) -> NumpyDict: - """Parses policy model outputs using SplitParser.""" - result: NumpyDict = self._policy_parser.parse_policy_outputs(self._slice_outputs(model_outputs)) - return result - -class VisionTinygrad(ModularRunner, ABC): - """ - A TinygradRunner specialized for vision-only models. - - Uses a SplitParser to handle outputs specific to the vision part of a split model setup. - """ - def __init__(self): - self._vision_parser = SplitParser() - self.parser_method_dict[ModelType.vision] = self._parse_vision_outputs - - def _parse_vision_outputs(self, model_outputs: np.ndarray) -> NumpyDict: - """Parses vision model outputs using SplitParser.""" - result: NumpyDict = self._vision_parser.parse_vision_outputs(self._slice_outputs(model_outputs)) - return result - -class SupercomboTinygrad(ModularRunner, ABC): - """ - A TinygradRunner specialized for vision-only models. - - Uses a SplitParser to handle outputs specific to the vision part of a split model setup. - """ - def __init__(self): - self._supercombo_parser = CombinedParser() - self.parser_method_dict[ModelType.supercombo] = self._parse_supercombo_outputs - - def _parse_supercombo_outputs(self, model_outputs: np.ndarray) -> NumpyDict: - """Parses vision model outputs using SplitParser.""" - result: NumpyDict = self._supercombo_parser.parse_outputs(self._slice_outputs(model_outputs)) - return result diff --git a/openpilot/sunnypilot/models/runners/tinygrad/tinygrad_runner.py b/openpilot/sunnypilot/models/runners/tinygrad/tinygrad_runner.py deleted file mode 100644 index 4e17bd5ead..0000000000 --- a/openpilot/sunnypilot/models/runners/tinygrad/tinygrad_runner.py +++ /dev/null @@ -1,179 +0,0 @@ -import pickle - -import numpy as np -from openpilot.sunnypilot.models.runners.constants import NumpyDict, ModelType, ShapeDict, CUSTOM_MODEL_PATH, SliceDict -from openpilot.sunnypilot.models.runners.model_runner import ModelRunner -from openpilot.sunnypilot.models.runners.tinygrad.model_types import PolicyTinygrad, VisionTinygrad, SupercomboTinygrad, OffPolicyTinygrad, OnPolicyTinygrad -from openpilot.sunnypilot.models.split_model_constants import SplitModelConstants -from openpilot.sunnypilot.modeld_v2.constants import ModelConstants - -from tinygrad.tensor import Tensor - - -class TinygradRunner(ModelRunner, SupercomboTinygrad, PolicyTinygrad, VisionTinygrad, OffPolicyTinygrad, OnPolicyTinygrad): - """ - A ModelRunner implementation for executing Tinygrad models. - - Handles loading Tinygrad model artifacts (.pkl), preparing inputs as Tinygrad - Tensors (potentially using QCOM extensions on TICI), running inference, - and parsing the outputs. - - :param model_type: The type of model (e.g., supercombo) to load and run. - """ - def __init__(self, model_type: int = ModelType.supercombo): - ModelRunner.__init__(self) - SupercomboTinygrad.__init__(self) - PolicyTinygrad.__init__(self) - VisionTinygrad.__init__(self) - OffPolicyTinygrad.__init__(self) - OnPolicyTinygrad.__init__(self) - self._constants = ModelConstants - self._model_data = self.models.get(model_type) - if not self._model_data or not self._model_data.model: - raise ValueError(f"Model data for type {model_type} not available.") - - artifact_filename = self._model_data.model.artifact.fileName - assert artifact_filename.endswith('_tinygrad.pkl'), \ - f"Invalid model file {artifact_filename} for TinygradRunner" - - model_pkl_path = f"{CUSTOM_MODEL_PATH}/{artifact_filename}" - with open(model_pkl_path, "rb") as f: - try: - # Load the compiled Tinygrad model runner function - self.model_run = pickle.load(f) - except FileNotFoundError as e: - # Provide a helpful error message if the model was built for a different platform - assert "/dev/kgsl-3d0" not in str(e), "Model was built on C3 or C3X, but is being loaded on PC" - raise - - # Map input names to their required dtype and device from the loaded model - self.input_to_dtype = {} - self.input_to_device = {} - for idx, name in enumerate(self.model_run.captured.expected_names): - info = self.model_run.captured.expected_input_info[idx] - self.input_to_dtype[name] = info[2] # dtype - self.input_to_device[name] = info[3] # device - self._policy_cached = False - - @property - def vision_input_names(self) -> list[str]: - """Returns the list of vision input names from the input shapes.""" - return [name for name in self.input_shapes.keys() if 'img' in name] - - - def prepare_policy_inputs(self, numpy_inputs: NumpyDict): - if not self._policy_cached: - for key, value in numpy_inputs.items(): - self.inputs[key] = Tensor(value, device='NPY').realize() - self._policy_cached = True - - def prepare_inputs(self, numpy_inputs: NumpyDict) -> dict: - """Prepares all vision and policy inputs for the model.""" - self.prepare_policy_inputs(numpy_inputs) - for key in self.vision_input_names: - if key in self.inputs: - self.inputs[key] = self.inputs[key].cast(self.input_to_dtype[key]) - return self.inputs - - def _run_model(self) -> NumpyDict: - """Runs the Tinygrad model inference and parses the outputs.""" - outputs = self.model_run(**self.inputs).contiguous().realize().uop.base.buffer.numpy().flatten() - return self._parse_outputs(outputs) - - def _parse_outputs(self, model_outputs: np.ndarray) -> NumpyDict: - """Parses the raw model outputs using the standard Parser.""" - if self._model_data is None: - raise ValueError("Model data is not available. Ensure the model is loaded correctly.") - - result: NumpyDict = self.parser_method_dict[self._model_data.model.type.raw](model_outputs) - return result - - -class TinygradSplitRunner(ModelRunner): - """ - A ModelRunner that coordinates separate TinygradVisionRunner and TinygradPolicyRunner instances. - - Manages the execution of split vision and policy models, combining their inputs and outputs. - """ - def __init__(self): - super().__init__() - self.is_20hz_3d = True - self.vision_runner = TinygradRunner(ModelType.vision) - self.policy_runner = TinygradRunner(ModelType.policy) if self.models.get(ModelType.policy) else None - self.off_policy_runner = TinygradRunner(ModelType.offPolicy) if self.models.get(ModelType.offPolicy) else None - self.on_policy_runner = TinygradRunner(ModelType.onPolicy) if self.models.get(ModelType.onPolicy) else None - self._constants = SplitModelConstants - - def _run_model(self) -> NumpyDict: - """Runs both vision and policy models and merges their parsed outputs.""" - vision_output = self.vision_runner.run_model() - outputs = {**vision_output} - - if self.policy_runner: - policy_output = self.policy_runner.run_model() - outputs.update(policy_output) - - if self.off_policy_runner: - off_policy_output = self.off_policy_runner.run_model() - if self.on_policy_runner: - off_policy_output.pop('plan', None) - outputs.update(off_policy_output) - - if self.on_policy_runner: - on_policy_output = self.on_policy_runner.run_model() - outputs.update(on_policy_output) - - if 'planplus' in outputs and 'plan' in outputs: - outputs['plan'] = outputs['plan'] + outputs['planplus'] - - return outputs - - @property - def vision_input_names(self) -> list[str]: - """Returns the list of vision input names from the vision runner.""" - return list(self.vision_runner.vision_input_names) - - @property - def input_shapes(self) -> ShapeDict: - """Returns the combined input shapes from both vision and policy models.""" - shapes = {**self.vision_runner.input_shapes} - if self.policy_runner: - shapes.update(self.policy_runner.input_shapes) - if self.off_policy_runner: - shapes.update(self.off_policy_runner.input_shapes) - if self.on_policy_runner: - shapes.update(self.on_policy_runner.input_shapes) - return shapes - - @property - def output_slices(self) -> SliceDict: - """Returns the combined output slices from both vision and policy models.""" - slices = {**self.vision_runner.output_slices} - if self.policy_runner: - slices.update(self.policy_runner.output_slices) - if self.off_policy_runner: - slices.update(self.off_policy_runner.output_slices) - if self.on_policy_runner: - slices.update(self.on_policy_runner.output_slices) - return slices - - def prepare_inputs(self, numpy_inputs: NumpyDict) -> dict: - """Prepares inputs for both vision and policy models.""" - if self.policy_runner: - self.policy_runner.prepare_policy_inputs(numpy_inputs) - - for key in self.vision_input_names: - if key in self.inputs: - self.vision_runner.inputs[key] = self.inputs[key].cast(self.vision_runner.input_to_dtype[key]) - - inputs = {**self.vision_runner.inputs} - if self.policy_runner: - inputs.update(self.policy_runner.inputs) - - if self.off_policy_runner: - self.off_policy_runner.prepare_policy_inputs(numpy_inputs) - inputs.update(self.off_policy_runner.inputs) - if self.on_policy_runner: - self.on_policy_runner.prepare_policy_inputs(numpy_inputs) - inputs.update(self.on_policy_runner.inputs) - return inputs diff --git a/openpilot/sunnypilot/models/split_model_constants.py b/openpilot/sunnypilot/models/split_model_constants.py index a3e1dce8f6..a5f57e5453 100644 --- a/openpilot/sunnypilot/models/split_model_constants.py +++ b/openpilot/sunnypilot/models/split_model_constants.py @@ -43,6 +43,7 @@ class SplitModelConstants: LANE_LINES_WIDTH = 2 ROAD_EDGES_WIDTH = 2 PLAN_WIDTH = 15 + ACTION_WIDTH = 2 DESIRE_PRED_WIDTH = 8 LAT_PLANNER_SOLUTION_WIDTH = 4 DESIRED_CURV_WIDTH = 1 diff --git a/release/ci/model_generator.py b/release/ci/model_generator.py index afee782beb..76935f3627 100755 --- a/release/ci/model_generator.py +++ b/release/ci/model_generator.py @@ -5,8 +5,6 @@ This file is part of sunnypilot and is licensed under the MIT License. See the LICENSE.md file in the root directory for more details. """ -import os -import pickle import sys import hashlib import json @@ -14,46 +12,8 @@ import re from pathlib import Path from datetime import datetime, UTC -REQUIRED_OUTPUT_KEYS = frozenset({ - "plan", - "lane_lines", - "road_edges", - "lead", - "desire_state", - "desire_pred", - "meta", - "lead_prob", - "lane_lines_prob", - "pose", - "wide_from_device_euler", - "road_transform", - "hidden_state", -}) -OPTIONAL_OUTPUT_KEYS = frozenset({ - "planplus", - "sim_pose", - "desired_curvature", -}) - -def validate_model_outputs(metadata_paths: list[Path]) -> None: - combined_keys: set[str] = set() - for path in metadata_paths: - if path.stat().st_size == 0: - print(f"skipping empty metadata: {path}") - continue - with open(path, "rb") as f: - metadata = pickle.load(f) - combined_keys.update(metadata.get("output_slices", {}).keys()) - missing = REQUIRED_OUTPUT_KEYS - combined_keys - if missing: - raise ValueError(f"Combined model metadata is missing required output keys: {sorted(missing)}") - detected_optional = sorted(OPTIONAL_OUTPUT_KEYS & combined_keys) - if detected_optional: - print(f"Optional output keys detected: {detected_optional}") - - -def create_short_name(full_name): +def create_short_name(full_name: str) -> str: # Remove parentheses and extract alphanumeric words clean_name = re.sub(r'\([^)]*\)', '', full_name) words = [re.sub(r'[^a-zA-Z0-9]', '', word) for word in clean_name.split() if re.sub(r'[^a-zA-Z0-9]', '', word)] @@ -121,49 +81,41 @@ def _rename_pkl_with_chunks(old_pkl: Path, new_pkl: Path) -> Path: return old_pkl.rename(new_pkl) -def generate_metadata(model_path: Path, output_dir: Path, short_name: str, driving_pkl: Path): - base = model_path.stem - metadata_file = output_dir / f"{base}_metadata.pkl" - - if short_name: - renamed_meta = output_dir / f"{base}_{short_name.lower()}_metadata.pkl" - if metadata_file.exists() and not renamed_meta.exists(): - metadata_file = metadata_file.rename(renamed_meta) - elif renamed_meta.exists(): - metadata_file = renamed_meta - - if not metadata_file.exists(): - print(f"Warning: Missing metadata for {base} ({metadata_file}), skipping", file=sys.stderr) - return - +def generate_chunked_model(driving_pkl: Path) -> dict: tinygrad_hash = hashlib.sha256(_read_pkl_bytes(driving_pkl)).hexdigest() - with open(metadata_file, 'rb') as f: - metadata_hash = hashlib.sha256(f.read()).hexdigest() + chunks_config = [] + manifest_file = Path(f"{driving_pkl}.chunkmanifest") + if manifest_file.exists(): + num_chunks = int(manifest_file.read_text().strip()) + for i in range(num_chunks): + chunk_path = Path(f"{driving_pkl}.chunk{i + 1:02d}of{num_chunks:02d}") + if chunk_path.exists(): + chunk_hash = hashlib.sha256(chunk_path.read_bytes()).hexdigest() + chunks_config.append({ + "file_name": chunk_path.name, + "sha256": chunk_hash + }) - model_type = "offPolicy" if "off_policy" in base else "onPolicy" if "on_policy" in base else base.split("_")[-1] - - return { - "type": model_type, - "artifact": { - "file_name": driving_pkl.name, - "download_uri": { - "url": "https://gitlab.com/sunnypilot/public/docs.sunnypilot.ai/-/raw/main/", - "sha256": tinygrad_hash - } - }, - "metadata": { - "file_name": metadata_file.name, - "download_uri": { - "url": "https://gitlab.com/sunnypilot/public/docs.sunnypilot.ai/-/raw/main/", - "sha256": metadata_hash - } + artifact_data = { + "file_name": driving_pkl.name, + "download_uri": { + "url": "https://gitlab.com/sunnypilot/public/docs.sunnypilot.ai/-/raw/main/", + "sha256": tinygrad_hash } } + if chunks_config: + artifact_data["chunks"] = chunks_config -def create_metadata_json(models: list, output_dir: Path, custom_name=None, short_name=None, is_20hz=False, upstream_branch="unknown"): - metadata_json = { + return { + "type": "chunked", + "artifact": artifact_data, + } + + +def create_metadata_json(models: list, output_dir: Path, custom_name=None, short_name=None, is_20hz=False, upstream_branch="unknown") -> None: + bundle_json = { "short_name": short_name, "display_name": custom_name or upstream_branch, "is_20hz": is_20hz, @@ -179,40 +131,26 @@ def create_metadata_json(models: list, output_dir: Path, custom_name=None, short } # Write metadata to output_dir + metadata_json = { + "bundles": [bundle_json] + } + with open(output_dir / "metadata.json", "w") as f: json.dump(metadata_json, f, indent=2) - - print(f"Generated metadata.json with {len(models)} models.") + print("Generated metadata.json") if __name__ == "__main__": import argparse - import glob - parser = argparse.ArgumentParser(description="Generate metadata for model files") - parser.add_argument("--model-dir", default="./models", help="Directory containing ONNX model files") + parser = argparse.ArgumentParser(description="Generate metadata JSON for the compiled JIT model") + parser.add_argument("--model-dir", default="./models", help="Directory containing the model files") parser.add_argument("--output-dir", default="./output", help="Output directory for metadata") parser.add_argument("--custom-name", help="Custom display name for the model") parser.add_argument("--is-20hz", action="store_true", help="Whether this is a 20Hz model") - parser.add_argument("--validate-only", action="store_true") parser.add_argument("--upstream-branch", default="unknown", help="Upstream branch name") args = parser.parse_args() - if args.validate_only: - metadata_paths = glob.glob(os.path.join(args.model_dir, "*_metadata.pkl")) - if not metadata_paths: - print(f"No metadata files found in {args.model_dir}", file=sys.stderr) - sys.exit(1) - validate_model_outputs([Path(p) for p in metadata_paths]) - print(f"Validated {len(metadata_paths)} metadata files successfully.") - sys.exit(0) - - # Find all ONNX files in the given directory - model_paths = glob.glob(os.path.join(args.model_dir, "*.onnx")) - if not model_paths: - print(f"No ONNX files found in {args.model_dir}", file=sys.stderr) - sys.exit(1) - _output_dir = Path(args.output_dir) _output_dir.mkdir(exist_ok=True, parents=True) _short_name = create_short_name(args.custom_name) if args.custom_name else None @@ -229,14 +167,5 @@ if __name__ == "__main__": else: _driving_pkl = new_pkl - _models = [] - - for _model_path in model_paths: - _model_metadata = generate_metadata(Path(_model_path), _output_dir, _short_name, _driving_pkl) - if _model_metadata: - _models.append(_model_metadata) - - if _models: - create_metadata_json(_models, _output_dir, args.custom_name, _short_name, args.is_20hz, args.upstream_branch) - else: - print("No models processed.", file=sys.stderr) + _model_metadata = generate_chunked_model(_driving_pkl) + create_metadata_json([_model_metadata], _output_dir, args.custom_name, _short_name, args.is_20hz, args.upstream_branch) diff --git a/tinygrad_repo b/tinygrad_repo index ac1632ab96..2fecac4e4a 160000 --- a/tinygrad_repo +++ b/tinygrad_repo @@ -1 +1 @@ -Subproject commit ac1632ab966c77ba96a7048b893a30f1a714dc87 +Subproject commit 2fecac4e4ac32fe369c41f8400b6e7b9adb18683