diff --git a/.github/workflows/fleet_safety.yaml b/.github/workflows/fleet_safety.yaml new file mode 100644 index 0000000000..0580a71515 --- /dev/null +++ b/.github/workflows/fleet_safety.yaml @@ -0,0 +1,56 @@ +name: Fleet controller safety + +on: + workflow_dispatch: + pull_request: + paths: + - 'opendbc_repo/**' + - 'selfdrive/car/**' + - 'starpilot/car/**' + - 'starpilot/controls/**' + - 'cereal/**' + - '.github/workflows/fleet_safety.yaml' + +permissions: + contents: read + +jobs: + harness: + runs-on: ubuntu-24.04 + steps: + - uses: actions/checkout@v4 + - uses: actions/setup-python@v5 + with: + python-version: '3.12' + - run: python -m pip install -r selfdrive/car/tests/fleet_requirements.txt + - name: Audit accounting and runner failures + run: >- + python -m pytest --noconftest -o addopts='' -q + selfdrive/car/tests/test_fleet_safety_core.py + selfdrive/car/tests/test_fleet_safety_runner.py + opendbc_repo/opendbc/safety/tests/safety_replay/test_replay_drive.py + recorded_fleet: + runs-on: ubuntu-24.04 + timeout-minutes: 90 + strategy: + fail-fast: false + matrix: + shard: [0, 1, 2, 3, 4, 5, 6, 7] + steps: + - uses: actions/checkout@v4 + - uses: actions/setup-python@v5 + with: + python-version: '3.12' + - run: python -m pip install -r selfdrive/car/tests/fleet_requirements.txt + - name: Current controllers against freshly compiled release safety + run: >- + python -m selfdrive.car.tests.fleet_safety --all --release + --shard-count 8 --shard-index ${{ matrix.shard }} + --out selfdrive/car/tests/fleet_results/ci + - uses: actions/upload-artifact@v4 + if: always() + with: + name: fleet-safety-${{ matrix.shard }} + path: | + selfdrive/car/tests/fleet_results/ci/**/*.json + selfdrive/car/tests/fleet_results/ci/**/*.log diff --git a/docs/FLEET_SAFETY_RESULTS_20260921.md b/docs/FLEET_SAFETY_RESULTS_20260921.md new file mode 100644 index 0000000000..dae82087b9 --- /dev/null +++ b/docs/FLEET_SAFETY_RESULTS_20260921.md @@ -0,0 +1,86 @@ +# Fleet offline audit — September 21, 2026 + +This is a first-pass fault and coverage inventory, not fleet driving clearance. +No hardware was accessed and no controller or safety policy was changed by this +audit. Other work was concurrently modifying the checkout; per-case source and +library hashes identify the tested implementations. + +The full debug-library run visited all 345 platforms. It evaluated both recorded +and AOL-main scenarios for 287 registered routes, plus 108 missing-route entries: + +| Result | Cases | +|---|---:| +| Pass within the stated scope | 103 | +| Failed checks, requiring triage | 123 | +| Missing coverage | 288 | +| Could not evaluate | 168 | +| Total | 682 | + +These are case counts, not numbers of unsafe cars. Missing coverage includes all +108 platforms without routes and segments without active transitions. Evaluation +errors include unavailable recordings and identities that do not match the +registered platform after the existing fingerprint migration. Old logs must be +normalized explicitly, not quietly substituted for another car. + +Full local evidence is under +`selfdrive/car/tests/fleet_results/full_fleet/results.json`, with per-shard build, +case and worker logs. Generated evidence is ignored by Git. + +The follow-up full release-library run also completed 682 cases: **102 pass, +153 failed, 289 uncovered, 138 evaluation errors**. Evidence is under +`selfdrive/car/tests/fleet_results/release_fleet_checked/results.json`. +The two batches had different download availability and ran against a changing +working tree; their count difference is not an isolated debug-versus-release +experiment. Both correctly exit nonzero. The release batch is not fleet clearance. + +## Confirmed distinctions from failure triage + +**Hyundai Custin — controller/safety capability mismatch.** The registered +segment `0bbe367c98fa1538/2023-09-16--00-16-49/2` contains 600 camera-bus +messages at 0x53e, all eight bytes, none six. `CarInterfaceBase.get_starpilot_params` +in `opendbc_repo/opendbc/car/interfaces.py` enables HAS_LKAS12 by address alone. +Hyundai `CarState.update` and `CarController.update` then produce six-byte LKAS12 +replacements. In `opendbc_repo/opendbc/safety/modes/hyundai.h`, +`hyundai_rx_all_hook` only enables replacement after receiving a six-byte camera +message, and `hyundai_tx_hook` correctly rejects the unsolicited replacement. +Both scenarios reject 5,799 such packets. A fresh-library probe also reproduced +the six-byte/eight-byte distinction. Repair requires a controller capability and +parser regression test; do not broaden the safety allowlist to hide the mismatch. + +**Honda Civic Bosch — incompatible historical control requests.** All 5,997 +recorded requests in the inspected 2020 fixture have enabled/latActive/longActive +false but resume true; all 6,000 cruise-state CAN messages are disabled. Current +controller output is RES_ACCEL at 0x296, which current safety correctly blocks. +The AOL probe suppresses resume and passes. Investigate historical command/schema +semantics before calling this a current steering defect. + +**Ford Escape — mid-segment initialization artifact.** The first cruise-enabled +0x165 enables controls, but the immediately following 0x202 reaches +`speed_mismatch_check` before safety has nonzero speed history. Controls are +revoked; cruise stays enabled for the entire segment so `pcm_cruise_check` sees +no new rising edge. Relay health remains good. This reproduces with both recorded +and default alternative experience. A two-second scoring warmup does not repair +the latch. This needs recorded preroll/initialization coverage, not force-setting +`controls_allowed` or changing vehicle safety. + +## Tesla and AOL scope + +In the full debug-library run, the Model 3 route and the second Model Y route +passed both scenarios. The first Model Y route lacked a lateral transition; +its AOL case also lacked requested steering under AOL-only safety permission. +Model X was uncovered as a current dashcam-only configuration. The Model S HW1 +and Pre-AP entries have no registered routes. None of these findings reproduces +or disproves the exact hackathon oscillation without its trace. + +The separate actual StarPilotCard synthetic-input suite passed 964 checks with +72 explicit active-sequence gaps. All 345 disabled configurations were checked; +309 platforms completed both active modes. These tests check state-machine gates +and stable sequences, not the entire selfdrived-to-Panda feedback loop. + +The test-harness regressions pass 34 tests. They cover pre-hook AOL authorization, +strict configuration/bus routing, rejected active packets, expected negative +checks, empty activity, worker crashes, stale reports and build failures. + +See `FLEET_SAFETY_TESTING.md` for commands, CI scope and limitations. The workflow +has been added locally but not published or run on GitHub, and branch protection +has not been changed. The full fleet is not green. diff --git a/docs/FLEET_SAFETY_TESTING.md b/docs/FLEET_SAFETY_TESTING.md new file mode 100644 index 0000000000..b7a85b7b29 --- /dev/null +++ b/docs/FLEET_SAFETY_TESTING.md @@ -0,0 +1,112 @@ +# Offline fleet controller and safety checks + +The runner in `selfdrive/car/tests/fleet_safety.py` enumerates every platform in +the current checkout and uses `opendbc.car.tests.routes`. It runs current +CarInterface/CarState/CarController code against recorded CAN and actuator +requests, then checks each newly generated CAN packet with freshly compiled +current safety hooks. It never connects to a Panda or starts vehicle processes. + +Run from the repository root with the repository Python environment: + +```sh +python -m selfdrive.car.tests.fleet_safety --inventory +python -m selfdrive.car.tests.fleet_safety --platform TESLA_MODEL_Y --release +python -m selfdrive.car.tests.fleet_safety --all --release +``` + +An isolated dependency set is in `selfdrive/car/tests/fleet_requirements.txt`. +The runner needs a C compiler. It does not use a previously staged libsafety. +Without `--release`, the safety library enables ALLOW_DEBUG, like the existing +safety unit tests. Release checks are needed as well: a debug-only hook must not +be mistaken for an available production configuration. Release host builds retain +unused-variable warnings without treating that specific diagnostic as an error. + +For parallel runs, use separate output directories: + +```sh +python -m selfdrive.car.tests.fleet_safety --all --release \ + --shard-count 8 --shard-index 0 --out selfdrive/car/tests/fleet_results/shard_0 +``` + +Run indices 0 through 7. Each has its own library copies, worker processes, +parameter namespace, logs and results. `--local-log` allows a local rlog for one +explicitly selected platform. Route IDs and old fingerprint aliases must match; +the harness does not silently treat another vehicle's log as that platform. + +## What is checked + +- Recorded commands under the current default feature configuration. +- A separate AOL MAIN-availability controller/safety probe, using recorded + actuator values with longitudinal requests and cruise button requests off. +- Actual per-Panda safety model, CP/FPCP safety-param OR, alternative-experience + OR, and strict four-bus routing, matching production configuration assembly. +- Incoming CAN, current CarState validity and safety receive health. +- Every emitted TX, including inactive-state packets; unexpected rejection fails. +- Pre-hook normal/AOL/longitudinal permissions, so a rejection that revokes + authorization cannot disappear from the failure accounting. +- Active requests, accepted active TX, engagement transitions, and AOL-only + safety authorization coverage. Sparse commands and absent transitions cannot + qualify as complete coverage. + +There is a two-second unscored fixture startup interval. Controller and safety +history still receive messages during it. The harness does not force safety +authorization or clear a relay fault to manufacture a passing result. + +## Results are deliberately strict + +`pass` means the case satisfied these specific checks and coverage requirements. +`failed` means a hook/health check failed and needs investigation. `uncovered` +means the scenario was not demonstrated, including missing routes, dashcam-only +interfaces and segments without transitions. `error` means the case could not be +evaluated, such as download failure or mismatched fixture identity. Anything +other than pass makes the command exit nonzero. Existing `non_tested_cars` +exemptions remain visible coverage gaps. + +Reports include frame counters, bounded rejected packet evidence, source hashes, +effective safety configurations and build provenance. Worker results carry a +unique execution ID; a crash, stale result or inconsistent exit code cannot be +reused as a pass. JSON and logs live under the ignored `fleet_results` directory. + +The AOL probe is **not** a complete simulation of StarPilotCard, selfdrived, +controls mismatch handling or a vehicle ECU. Recorded commands may also reflect +historical settings different from current defaults. A blocked historical resume +request is not automatically a steering bug. Investigate each failure before +changing code. Do not widen safety permissions to make tests green. + +## Continuous integration and remaining coverage + +`starpilot/controls/tests/test_fleet_aol.py` separately exercises the actual +StarPilotCard state machine with isolated synthetic inputs for every platform. +It tests feature-off behavior, steady engagement, AOL-only operation, brake +pause, native/StarPilot immediate-disable alerts, calibration and gear gates. +It uses empty firmware/fingerprint fixtures, so optional vehicle configurations +are not covered by these sequences. Run it with a compatible built host runtime: + +```sh +python -m pytest --noconftest -o addopts='' -q starpilot/controls/tests/test_fleet_aol.py +``` + +The first run passed 964 checks and explicitly skipped 72 active sequences: +309 platforms exercised both active modes; 36 platforms had two gaps each +(30 dashcam-only, one notCar, four Volvo policy exclusions, one Pre-AP external +authorization dependency). All 345 feature-off checks passed. A separate +Pre-AP authorization input boundary test is synthetic, not proof of actual +Panda authorization. These tests were run against the current working tree, +including concurrent Pre-AP changes; they do not certify an earlier commit. +The lightweight CI workflow below does not build the native runtime required +by this separate state-machine suite. + +`.github/workflows/fleet_safety.yaml` adds harness tests and eight release-mode +recorded-route shards on relevant pull requests and manual runs. Missing coverage +is not converted to a skip or allowed failure. The current fleet is not green; +this workflow will expose that fact. It has not been executed on GitHub from this +local task. Requiring it for merge also needs repository branch protection; a +workflow file alone does not change repository settings. + +As of the first September 21 inventory there are 345 platforms, 287 registered +routes across 237 platforms, and 108 platforms with no registered route. A route +entry does not guarantee valid, downloadable logs or all necessary maneuvers. +Every optional harness, longitudinal mode, safety parameter, firmware generation +and AOL configuration still needs explicit coverage. Offline checks reduce +blind spots; they do not certify every physical vehicle or reproduce an incident +whose CAN trace was not retained. diff --git a/selfdrive/car/tests/fleet_requirements.txt b/selfdrive/car/tests/fleet_requirements.txt new file mode 100644 index 0000000000..ac161f6c4c --- /dev/null +++ b/selfdrive/car/tests/fleet_requirements.txt @@ -0,0 +1,25 @@ +atomicwrites==1.4.1 +certifi==2026.7.22 +cffi==2.1.1 +charset-normalizer==3.5.1 +crcmod==1.7 +idna==3.20 +iniconfig==2.3.0 +numpy==2.5.3 +packaging==26.3 +pluggy==1.6.0 +psutil==7.2.2 +pycapnp==2.1.0 +pycparser==3.0 +pycryptodome==3.23.0 +pygments==2.21.0 +pyjwt==2.14.0 +pyserial==3.5 +pytest==9.1.1 +pyzmq==27.2.0 +requests==2.34.2 +smbus2==0.6.1 +tenacity==9.1.4 +tqdm==4.70.1 +urllib3==2.8.0 +zstandard==0.25.0 diff --git a/selfdrive/car/tests/fleet_results/.gitignore b/selfdrive/car/tests/fleet_results/.gitignore new file mode 100644 index 0000000000..d6b7ef32c8 --- /dev/null +++ b/selfdrive/car/tests/fleet_results/.gitignore @@ -0,0 +1,2 @@ +* +!.gitignore diff --git a/selfdrive/car/tests/fleet_safety.py b/selfdrive/car/tests/fleet_safety.py new file mode 100644 index 0000000000..e1bc21a180 --- /dev/null +++ b/selfdrive/car/tests/fleet_safety.py @@ -0,0 +1,348 @@ +"""Offline current-controller -> current-safety checks using the registered fleet routes. + +No Panda connection, vehicle processes, or actuation. Missing evidence is not a pass. +Run with: python -m selfdrive.car.tests.fleet_safety --help +""" +import argparse +from collections import Counter, defaultdict +from dataclasses import asdict +import hashlib +import importlib.util +import json +import os +from pathlib import Path +import shutil +import subprocess +import sys +import tempfile +from types import SimpleNamespace +import uuid + +ROOT = Path(__file__).resolve().parents[3] +DEFAULT_OUT = Path(__file__).parent / 'fleet_results' + + +def digest(path): + return hashlib.sha256(Path(path).read_bytes()).hexdigest() + + +def inventory(): + from opendbc.car.values import PLATFORMS + from opendbc.car.tests.routes import routes, non_tested_cars + registered = defaultdict(list) + for route in routes: + registered[str(route.car_model)].append(dict(route=route.route, segment=route.segment)) + exempt = set(map(str, non_tested_cars)) + return [dict(platform=str(platform), routes=registered[str(platform)], + declared_without_route=str(platform) in exempt, + status='pending' if registered[str(platform)] else 'uncovered') + for platform in sorted(PLATFORMS)] + + +def build_safety(out, release=False): + source = ROOT / 'opendbc_repo/opendbc/safety/tests/libsafety/safety.c' + target = out / ('safety_release' if release else 'safety_debug') + target.mkdir(parents=True, exist_ok=True) + command = [os.environ.get('CC', 'cc'), '-shared', '-fPIC', '-std=gnu11', '-Wall', '-Werror', + '-I', str(ROOT/'opendbc_repo'), str(source), '-o', str(target/'libsafety.so')] + if not release: + command.insert(1, '-DALLOW_DEBUG') + else: + # Some hooks declare constants only consumed by ALLOW_DEBUG branches. + # Keep the compiler diagnostic, without making a host-only warning prevent + # checking which configurations the release registry actually supports. + command.insert(1, '-Wno-error=unused-variable') + result = subprocess.run(command, capture_output=True, text=True, timeout=120) + (target/'build.log').write_text(result.stdout + result.stderr) + result.check_returncode() + shutil.copy2(source.with_name('libsafety_py.py'), target/'libsafety_py.py') + headers = sorted((ROOT/'opendbc_repo/opendbc/safety').rglob('*.h')) + provenance = dict(command=command, release=release, library_sha256=digest(target/'libsafety.so'), + source_sha256=digest(source), headers={str(p.relative_to(ROOT)): digest(p) for p in headers}) + (target/'build.json').write_text(json.dumps(provenance, indent=2)) + return target + + +def load_safety(directory, panda_index): + # Each Panda gets independent C globals, never multiple handles to one library. + target = directory.parent / f'{directory.name}_panda_{panda_index}' + target.mkdir(exist_ok=True) + for name in ('libsafety.so', 'libsafety_py.py'): + shutil.copy2(directory/name, target/name) + spec = importlib.util.spec_from_file_location(f'fleet_safety_panda_{panda_index}', target/'libsafety_py.py') + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + +def test_toggles(scenario): + # Explicit fixture; do not read the operator's installed settings. + return SimpleNamespace(always_on_lateral=scenario != 'recorded', always_on_lateral_main=scenario == 'aol-main', + always_on_lateral_lkas=False, car_model='', cluster_offset=1.0, + disable_openpilot_long=False, force_fingerprint=False, lock_doors=False, + sng_hack=False, subaru_sng=False, subaru_sng_manual_parking_brake=False, + tesla_cooperative_steering=False, unlock_doors=False, vEgoStopping=0.5, volt_sng=False) + + +def read_route(route, segment, local_log=None): + from openpilot.tools.lib.logreader import LogReader, openpilotci_source + errors = [] + for seg in ([segment] if segment is not None else [2, 1, 0]): + identifier = local_log or f'{route}/{seg}' + try: + selected = [m for m in LogReader(identifier, sources=[openpilotci_source], sort_by_time=True) + if m.which() in ('can', 'carControl', 'carParams')] + if not any(m.which() == 'carParams' for m in selected): + raise ValueError('No recorded CarParams') + if not any(m.which() == 'carControl' for m in selected): + raise ValueError('No recorded control requests; active coverage unavailable') + return selected, identifier + except Exception as error: + errors.append(f'{identifier}: {type(error).__name__}: {error}') + if local_log: + break + raise RuntimeError('; '.join(errors)) + + +def run_case(platform, route, segment, scenario, safety_dir, local_log=None): + from opendbc.car import gen_empty_fingerprint + from opendbc.car.can_definitions import CanData + from opendbc.car.car_helpers import interfaces + from opendbc.car.fingerprints import MIGRATION + from opendbc.safety import ALTERNATIVE_EXPERIENCE + from openpilot.selfdrive.car.tests.fleet_safety_core import effective_safety_configs, TxAudit + + messages, identifier = read_route(route, segment, local_log) + recorded = next(m.carParams for m in messages if m.which() == 'carParams') + detected = MIGRATION.get(recorded.carFingerprint, recorded.carFingerprint) + if detected != platform: + raise ValueError(f'Fixture identity mismatch: {detected} != {platform}') + fingerprint = gen_empty_fingerprint() + for msg in messages: + if msg.which() == 'can': + for frame in msg.can: + if frame.src < 64: + fingerprint.setdefault(frame.src, {})[frame.address] = len(frame.dat) + toggles = test_toggles(scenario) + interface = interfaces[platform] + cp = interface.get_params(platform, fingerprint, list(recorded.carFw), + recorded.openpilotLongitudinalControl, False, docs=False, starpilot_toggles=toggles) + fp = interface.get_starpilot_params(platform, fingerprint, list(recorded.carFw), cp, toggles) + if cp.dashcamOnly or cp.notCar: + return dict(status='uncovered', reason='Current interface is dashcamOnly/notCar', platform=platform, scenario=scenario) + if scenario == 'aol-main': + cp.alternativeExperience |= ALTERNATIVE_EXPERIENCE.ALWAYS_ON_LATERAL + fp.alternativeExperience |= ALTERNATIVE_EXPERIENCE.ALWAYS_ON_LATERAL + configs = effective_safety_configs(cp, fp) + audits = {} + modules = {} + for cfg in configs: + if cfg.safety_model in (0, 19): # SILENT/noOutput; any attempted TX below is a failure. + continue + module = load_safety(safety_dir, cfg.panda_index) + safety = module.libsafety + if safety.set_safety_hooks(cfg.safety_model, cfg.safety_param) != 0: + raise ValueError(f'Safety configuration unavailable in built library: {cfg}') + safety.init_tests() + safety.set_alternative_experience(cfg.alternative_experience) + modules[cfg.panda_index] = module + audits[cfg.panda_index] = TxAudit(safety, module.make_CANPacket, config=cfg) + ci = interface(cp, fp) + from inspect import getfile + imported = {getfile(interface), getfile(type(ci.CC)), getfile(type(ci.CS))} + for path in imported: + if not Path(path).resolve().is_relative_to(ROOT/'opendbc_repo/opendbc/car'): + raise RuntimeError('Controller imports outside the tested checkout: ' + path) + source_hashes = {str(Path(p).resolve().relative_to(ROOT)): digest(p) for p in imported} + first = messages[0].logMonoTime + pending = [] + coverage = Counter() + errors = [] + previous = None + previous_command = None + warmed = False + for msg in messages: + elapsed = (msg.logMonoTime-first)/1e9 + if elapsed > 2 and not warmed: + # Keep safety/controller history established during startup, but report the + # scored interval separately from the explicit two-second fixture warmup. + audits = {cfg.panda_index: TxAudit(modules[cfg.panda_index].libsafety, + modules[cfg.panda_index].make_CANPacket, config=cfg) + for cfg in configs if cfg.panda_index in modules} + warmed = True + for module in modules.values(): + module.libsafety.set_timer((msg.logMonoTime//1000) & 0xffffffff) + if msg.which() == 'can': + rx = [CanData(f.address, bytes(f.dat), f.src) for f in msg.can if f.src < 64] + pending.append((msg.logMonoTime, rx)) + for frame in rx: + index = frame.src//4 + if index not in modules: + continue + module = modules[index] + accepted = module.libsafety.safety_rx_hook(module.make_CANPacket(frame.address, frame.src-index*4, frame.dat)) + if not accepted and elapsed > 2: + coverage['rx_rejections'] += 1 + continue + if msg.which() != 'carControl': + continue + cs, _ = ci.update(pending, toggles) + pending.clear() + cc = msg.carControl.as_builder() + if previous_command is not None and msg.logMonoTime-previous_command > 100_000_000: + coverage['command_gaps_over_100ms'] += 1 + previous_command = msg.logMonoTime + if scenario == 'aol-main': + # This is a controller/safety MAIN-availability probe, not a substitute for + # testing StarPilotCard's separate LKAS-button latch or full selfdrived loop. + cc.enabled = False + cc.longActive = False + cc.cruiseControl.cancel = False + cc.cruiseControl.resume = False + cc.latActive = bool(cs.canValid and cs.cruiseState.available and + str(cs.gearShifter) in ('drive', 'sport', 'low', 'eco', 'manumatic') and + not (cs.steerFaultTemporary or cs.steerFaultPermanent or cs.steeringDisengage or cs.brakePressed) and + (cp.steerAtStandstill or not cs.standstill)) + state = (bool(cc.latActive), bool(cc.longActive), bool(cs.brakePressed), bool(cs.gasPressed), bool(cs.standstill)) + if elapsed > 2: + coverage['control_frames'] += 1 + coverage['lat_active_frames'] += bool(cc.latActive) + coverage['long_active_frames'] += bool(cc.longActive) + coverage['lat_only_frames'] += bool(cc.latActive and not cc.longActive) + coverage['brake_frames'] += bool(cs.brakePressed) + coverage['gas_frames'] += bool(cs.gasPressed) + coverage['standstill_frames'] += bool(cs.standstill) + coverage['invalid_carstate_frames'] += not cs.canValid + if previous is not None and previous[0] != state[0]: + coverage['lateral_transitions'] += 1 + for module in modules.values(): + module.libsafety.safety_tick_current_safety_config() + coverage['invalid_safety_rx_frames'] += not module.libsafety.safety_config_valid() + previous = state + _, frames = ci.apply(cc.as_reader(), msg.logMonoTime, toggles) + for addr, data, bus in frames: + if bus//4 not in audits: + errors.append(f'Unconfigured outgoing bus {bus} at {msg.logMonoTime}') + continue + audit = audits[bus//4] + audit.check(addr, bus, bytes(data), msg.logMonoTime, scenario, + requested_lat_active=bool(cc.latActive), requested_long_active=bool(cc.longActive)) + summaries = {str(index): audit.summary() for index, audit in audits.items()} + missing = [name for name in ('lat_active_frames', 'lateral_transitions') if coverage[name] == 0] + if coverage['lat_active_frames'] < 100: + missing.append('at_least_100_active_control_frames') + if coverage['command_gaps_over_100ms']: + missing.append('continuous_recorded_control_timing') + if not summaries or any(s['status'] == 'uncovered' for s in summaries.values()): + missing.append('active_tx_authorization_and_request') + if scenario == 'aol-main' and coverage['lat_only_frames'] == 0: + missing.append('lat_only_frames') + if scenario == 'aol-main' and not any(s['requested_aol_only_tx'] for s in summaries.values()): + missing.append('requested_steering_with_aol_only_safety_permission') + failed = bool(errors or any(s['status'] == 'failed' for s in summaries.values()) or + any(coverage[k] for k in ('rx_rejections', 'invalid_carstate_frames', 'invalid_safety_rx_frames'))) + status = 'failed' if failed else 'uncovered' if missing else 'pass' + return dict(platform=platform, route=identifier, scenario=scenario, status=status, + missing_coverage=missing, coverage=dict(coverage), errors=errors[:30], + safety_configs=[asdict(c) for c in configs], panda_results=summaries, source_hashes=source_hashes, + warmup_seconds=2, + configuration_policy='Recorded actuator requests under current default settings, or explicit AOL MAIN probe', + recorded_alternative_experience=int(recorded.alternativeExperience), + limitations=['Controller/safety only: no modeld, selfdrived/AOL latch, UI or physical ECU simulation', + 'Uses recorded actuator requests; does not test new model outputs or every optional feature']) + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument('--inventory', action='store_true') + parser.add_argument('--platform', action='append', default=[]) + parser.add_argument('--all', action='store_true') + parser.add_argument('--shard-count', type=int, default=1) + parser.add_argument('--shard-index', type=int, default=0) + parser.add_argument('--scenario', action='append', choices=['recorded', 'aol-main']) + parser.add_argument('--out', type=Path, default=DEFAULT_OUT) + parser.add_argument('--release', action='store_true', help='Build without ALLOW_DEBUG') + parser.add_argument('--local-log', help='Explicit local rlog, requires one selected platform') + parser.add_argument('--case-timeout', type=float, default=180) + parser.add_argument('--worker', type=Path, help=argparse.SUPPRESS) + args = parser.parse_args() + if args.shard_count < 1 or not 0 <= args.shard_index < args.shard_count: + parser.error('Require 0 <= shard-index < shard-count') + args.out = args.out.resolve() + args.out.mkdir(parents=True, exist_ok=True) + if args.worker: + case = json.loads(args.worker.read_text()) + # Isolate any Params access by interfaces from real operator settings. + with tempfile.TemporaryDirectory(prefix='fleet-params-') as params: + os.environ.update(PARAMS_ROOT=params, OPENPILOT_PREFIX='fleet-offline-'+str(os.getpid())) + try: + result = run_case(case['platform'], case['route'], case['segment'], case['scenario'], + Path(case['safety_dir']), case.get('local_log')) + except Exception as error: + import traceback + result = dict(status='error', error=repr(error), traceback=traceback.format_exc()) + result['execution_id'] = case['execution_id'] + args.worker.with_suffix('.result.json').write_text(json.dumps(result, indent=2)) + return 0 if result['status'] == 'pass' else 1 + entries = inventory() + report = dict(platforms=entries, total_platforms=len(entries), + with_routes=sum(bool(e['routes']) for e in entries), + registered_routes=sum(len(e['routes']) for e in entries), + scope='offline controller/safety compatibility; not full vehicle validation') + (args.out/'inventory.json').write_text(json.dumps(report, indent=2)) + if args.inventory: + print(json.dumps({k:v for k,v in report.items() if k != 'platforms'}, indent=2)) + return 0 + if not args.all and not args.platform: + parser.error('Choose --all or --platform; --inventory only lists coverage') + unknown = set(args.platform)-{e['platform'] for e in entries} + if unknown: + parser.error('Unknown platforms: '+', '.join(sorted(unknown))) + if args.local_log and (len(args.platform) != 1 or args.all): + parser.error('--local-log requires exactly one --platform') + (args.out/'results.json').write_text(json.dumps(dict(cases=[], counts={}, state='running'))) + try: + safety_dir = build_safety(args.out, args.release) + except (subprocess.SubprocessError, OSError) as error: + failure = dict(status='error', reason='Safety build failed', error=repr(error)) + (args.out/'results.json').write_text(json.dumps(dict(cases=[failure], counts={'error': 1}), indent=2)) + print(failure['reason'] + ': ' + failure['error'], flush=True) + return 1 + selected = [e for e in entries if args.all or e['platform'] in args.platform] + selected = selected[args.shard_index::args.shard_count] + results = [] + for entry in selected: + registered = entry['routes'] + if args.local_log: + registered = [dict(route='local', segment=0)] + if not registered: + results.append(dict(platform=entry['platform'], status='uncovered', reason='No registered route')) + continue + for route in registered: + for scenario in args.scenario or ['recorded', 'aol-main']: + case = dict(platform=entry['platform'], **route, scenario=scenario, safety_dir=str(safety_dir), + local_log=args.local_log, execution_id=uuid.uuid4().hex) + path = args.out / f'case_{len(results):04d}.json' + path.write_text(json.dumps(case)) + path.with_suffix('.result.json').unlink(missing_ok=True) + print(f"Checking {entry['platform']} {scenario} {route['route']}", flush=True) + with path.with_suffix('.log').open('w') as log: + try: + completed = subprocess.run([sys.executable, '-m', 'selfdrive.car.tests.fleet_safety', '--worker', str(path)], + cwd=ROOT, stdout=log, stderr=subprocess.STDOUT, timeout=args.case_timeout, check=False) + result = json.loads(path.with_suffix('.result.json').read_text()) + if result.get('execution_id') != case['execution_id'] or completed.returncode != (0 if result.get('status') == 'pass' else 1): + raise ValueError('Worker identity/exit status does not match its report') + except (subprocess.TimeoutExpired, OSError, ValueError) as error: + result = dict(status='error', error=repr(error)) + results.append({**case, **result}) + print(' '+result['status'], flush=True) + (args.out/'results.json').write_text(json.dumps(dict(cases=results, counts=dict(Counter(r['status'] for r in results))), indent=2)) + counts = Counter(r['status'] for r in results) + (args.out/'results.json').write_text(json.dumps(dict(cases=results, counts=dict(counts)), indent=2)) + print(json.dumps(dict(counts), indent=2)) + return 0 if results and all(r['status'] == 'pass' for r in results) else 1 + + +if __name__ == '__main__': + raise SystemExit(main()) diff --git a/selfdrive/car/tests/fleet_safety_core.py b/selfdrive/car/tests/fleet_safety_core.py index 8f36e1c23b..a0c85a032e 100644 --- a/selfdrive/car/tests/fleet_safety_core.py +++ b/selfdrive/car/tests/fleet_safety_core.py @@ -167,7 +167,8 @@ class TxAudit: for i, r in enumerate(records): if r.failure and len(failures) < failure_limit: failures.append({"index": i, "addr": r.addr, "bus": r.bus, "data_hex": r.data.hex(), "timestamp_ns": r.timestamp_ns, - "scenario": r.scenario, "reason": r.failure}) + "scenario": r.scenario, "reason": r.failure, + "permissions": r.permissions.label, "requested_mode": r.requested_mode}) permissions = [r.permissions.label for r in records] requested = [r.requested_mode for r in records] diff --git a/selfdrive/car/tests/test_fleet_safety_core.py b/selfdrive/car/tests/test_fleet_safety_core.py index bd14f042f9..3117ae11ce 100644 --- a/selfdrive/car/tests/test_fleet_safety_core.py +++ b/selfdrive/car/tests/test_fleet_safety_core.py @@ -80,7 +80,7 @@ class TestEffectiveSafetyConfigs(unittest.TestCase): (params(), params([config(1, -1)]), {}), (params(alternative=65535), params(), {}), (params(), params(), {"panda_count": -1})): - with self.subTest(cp=cp, fp=fp, kwargs=kwargs), self.assertRaises(ValueError): + with self.subTest(case=repr((cp, fp, kwargs))), self.assertRaises(ValueError): effective_safety_configs(cp, fp, **kwargs) diff --git a/selfdrive/car/tests/test_fleet_safety_runner.py b/selfdrive/car/tests/test_fleet_safety_runner.py new file mode 100644 index 0000000000..c0310a51f6 --- /dev/null +++ b/selfdrive/car/tests/test_fleet_safety_runner.py @@ -0,0 +1,112 @@ +"""Runner orchestration tests: no native libraries, route downloads, or vehicle IO.""" +import importlib.util +import json +from pathlib import Path +import sys +from types import SimpleNamespace + +import pytest + + +@pytest.fixture +def runner(): + spec = importlib.util.spec_from_file_location("fleet_safety_runner_test", Path(__file__).with_name("fleet_safety.py")) + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + +def test_inventory_missing_route_remains_uncovered(runner, monkeypatch): + monkeypatch.setitem(sys.modules, "opendbc.car.values", SimpleNamespace(PLATFORMS={"covered": None, "exempt": None, "missing": None})) + monkeypatch.setitem(sys.modules, "opendbc.car.tests.routes", SimpleNamespace( + routes=[SimpleNamespace(car_model="covered", route="route-a", segment=2)], non_tested_cars=["exempt"])) + entries = {entry["platform"]: entry for entry in runner.inventory()} + assert entries["covered"]["status"] == "pending" + assert entries["covered"]["routes"] == [{"route": "route-a", "segment": 2}] + assert entries["exempt"]["declared_without_route"] + assert entries["exempt"]["status"] == "uncovered" + assert entries["missing"]["status"] == "uncovered" + + +def configure_main(runner, monkeypatch, tmp_path, routes): + monkeypatch.setattr(sys, "argv", ["fleet_safety", "--all", "--scenario", "recorded", "--out", str(tmp_path)]) + monkeypatch.setattr(runner, "inventory", lambda: [dict(platform="test-platform", routes=routes, + declared_without_route=False, status="pending" if routes else "uncovered")]) + monkeypatch.setattr(runner, "build_safety", lambda *args: tmp_path / "mock-safety") + + +def test_missing_route_fails_run(runner, monkeypatch, tmp_path): + configure_main(runner, monkeypatch, tmp_path, []) + (tmp_path / "results.json").write_text(json.dumps(dict(cases=[], counts={"pass": 1}))) + monkeypatch.setattr(runner.subprocess, "run", lambda *args, **kwargs: pytest.fail("No route may launch a worker")) + assert runner.main() == 1 + report = json.loads((tmp_path / "results.json").read_text()) + assert report["counts"] == {"uncovered": 1} + + +def test_failed_build_cannot_leave_stale_pass(runner, monkeypatch, tmp_path): + configure_main(runner, monkeypatch, tmp_path, []) + (tmp_path / "results.json").write_text(json.dumps(dict(cases=[], counts={"pass": 1}))) + + def failed_build(*args): + raise OSError("compiler unavailable") + + monkeypatch.setattr(runner, "build_safety", failed_build) + assert runner.main() == 1 + report = json.loads((tmp_path / "results.json").read_text()) + assert report["counts"] == {"error": 1} + assert report["cases"][0]["reason"] == "Safety build failed" + + +@pytest.mark.parametrize("status,returncode,expected", [ + ("pass", 0, 0), ("failed", 1, 1), ("uncovered", 1, 1), ("error", 1, 1), +]) +def test_worker_status_controls_overall_result(runner, monkeypatch, tmp_path, status, returncode, expected): + configure_main(runner, monkeypatch, tmp_path, [dict(route="registered-route", segment=2)]) + + def worker(command, **kwargs): + path = Path(command[command.index("--worker") + 1]) + case = json.loads(path.read_text()) + path.with_suffix(".result.json").write_text(json.dumps(dict(status=status, execution_id=case["execution_id"]))) + return SimpleNamespace(returncode=returncode) + + monkeypatch.setattr(runner.subprocess, "run", worker) + assert runner.main() == expected + report = json.loads((tmp_path / "results.json").read_text()) + assert report["counts"] == {status: 1} + + +def test_crashed_worker_cannot_reuse_previous_pass(runner, monkeypatch, tmp_path): + configure_main(runner, monkeypatch, tmp_path, [dict(route="new-route", segment=2)]) + (tmp_path / "case_0000.result.json").write_text(json.dumps(dict(status="pass", route="old-route"))) + monkeypatch.setattr(runner.subprocess, "run", lambda *args, **kwargs: SimpleNamespace(returncode=1)) + assert runner.main() == 1 + report = json.loads((tmp_path / "results.json").read_text()) + assert report["counts"] == {"error": 1} + + +def test_worker_nonzero_exit_cannot_report_pass(runner, monkeypatch, tmp_path): + configure_main(runner, monkeypatch, tmp_path, [dict(route="registered-route", segment=2)]) + + def worker(command, **kwargs): + path = Path(command[command.index("--worker") + 1]) + case = json.loads(path.read_text()) + path.with_suffix(".result.json").write_text(json.dumps(dict(status="pass", execution_id=case["execution_id"]))) + return SimpleNamespace(returncode=1) + + monkeypatch.setattr(runner.subprocess, "run", worker) + assert runner.main() == 1 + + +def test_wrong_execution_identity_cannot_report_pass(runner, monkeypatch, tmp_path): + configure_main(runner, monkeypatch, tmp_path, [dict(route="registered-route", segment=2)]) + + def worker(command, **kwargs): + path = Path(command[command.index("--worker") + 1]) + path.with_suffix(".result.json").write_text(json.dumps(dict(status="pass", execution_id="other-run"))) + return SimpleNamespace(returncode=0) + + monkeypatch.setattr(runner.subprocess, "run", worker) + assert runner.main() == 1 + report = json.loads((tmp_path / "results.json").read_text()) + assert report["counts"] == {"error": 1}