From 6bfd34eab1bd40e6260233daf990736c90c584d4 Mon Sep 17 00:00:00 2001 From: firestar5683 <168790843+firestar5683@users.noreply.github.com> Date: Fri, 18 Sep 2026 22:52:31 -0700 Subject: [PATCH] Prepare private community route normalization, caching and blind judging --- roadscore/judging/README.md | 162 +++++++++++++ roadscore/judging/analyze.py | 119 ++++++++++ roadscore/judging/index.html | 36 +++ roadscore/judging/prepare.py | 364 ++++++++++++++++++++++++++++++ roadscore/judging/test_prepare.py | 122 ++++++++++ 5 files changed, 803 insertions(+) create mode 100644 roadscore/judging/README.md create mode 100644 roadscore/judging/analyze.py create mode 100644 roadscore/judging/index.html create mode 100644 roadscore/judging/prepare.py create mode 100644 roadscore/judging/test_prepare.py diff --git a/roadscore/judging/README.md b/roadscore/judging/README.md new file mode 100644 index 0000000000..1b1fd00eea --- /dev/null +++ b/roadscore/judging/README.md @@ -0,0 +1,162 @@ +# Community judging preparation + +Agent 3 scope: Mac-only route preparation. This package has no SSH, device, replay, +composer, worker or audio-output control. Final RoadScore generation is deliberately +absent. Do not start it as part of preparation. + +## Current handoff + +The repository guidance, STATUS.md, README.md and EVENT.md were reviewed. No applicable +AGENTS.md was found in the root or RoadScore tree. The only repository AGENTS.md, +under tinygrad_repo/, was also read: it specifies tinygrad test/typecheck/lint commands +and prohibits commit amendments. No tinygrad changes were made. EVENT.md records the native +prerequisite pass, but subjective listening and community judging remain pending. +The user explicitly authorized community preparation while reserving hardware for +another agent. + +The user subsequently identified `ROADSCORE_COMMA_HACK_7_CONTEXT.md` in the integration +checkout. It was read in full; its private appendix supplies the community entries. +The eleven context candidates plus the explicitly reported duplicate reconstruct +twelve submission instances, normalized to eleven distinct candidates, preserving the duplicate's +provenance and the integration owner's existing labels and deterministic seeds. +All live data, mappings, metadata and decisions are in ignored local results/. +Final generation is not authorized and has not been run. + +Connect semantics were checked against upstream commit +[`5c7f276`](https://github.com/commaai/connect/blob/5c7f27617371a2503c360ff1d08bf19966d2527c/src/url.js#L27): +URL suffixes are seconds, not segment indices. Its current +[route reducer](https://github.com/commaai/connect/blob/5c7f27617371a2503c360ff1d08bf19966d2527c/src/reducers/globalState.js#L432) +uses the full route when only a start suffix is supplied. The user explicitly chose +that full-route behavior for the ambiguous single-timestamp submission. The explicit +two-timestamp submission retains its exact end-exclusive interval. Private handoff +files record the concrete ranges without exposing them in tracked documentation. + +## Input and normalization + +Store inputs in an ignored `roadscore/results/community_INPUT/submissions.private.json`. +Use a JSON array of route strings or objects with `route`, optional `submitter`, `note`, +and selection fields. For example (synthetic identifiers only): + +```json +[ + {"route":"0000000000000000/2026-01-01--00-00-00", "segments":[2,3]}, + {"route":"0000000000000001|2026-01-01--00-00-00", "start_s":60, "end_s":120} +] +``` + +Supported inputs: `DONGLE/ROUTE`, `DONGLE|ROUTE`, HTTPS comma/Konik Connect route links. In **bare native identifiers**, +`--N` or `/N` selects one segment and `/START:STOP` an end-exclusive segment slice. +In **Connect URLs**, `/START/END` denotes seconds, and a lone positive `/START` +remains pending intent confirmation because the current web page displays the full route. +Explicit `segments` are sorted and deduplicated. No selection means the full route. +Time bounds are route-relative seconds, with exclusive end. URL queries/fragments, +open native slices, dash ranges, conflicting selections and mixed time/segment selections +block preparation rather than guessing. Convert those using the submitter's intent +and retain the original text/decision in `note`. Never infer a time range from a URL +without checking its semantics. + +Exact normalized route+selection duplicates collapse, retaining submission references. +Different selections of the same route remain distinct and require an explicit +`--acknowledge-same-route-selections` decision; the tool does not silently merge them. +Eleven unique selections are required for A–K. A failed attempt writes only its +normalization report; correct the source and use a fresh output directory. + +```sh +python3 roadscore/judging/prepare.py prepare \ + --submissions roadscore/results/community_INPUT/submissions.private.json \ + --output roadscore/results/community_BATCH +``` + +The private manifest preserves source checksum, exact selection and provenance. +Use `--existing-manifest PATH` to preserve a previously issued mapping and seeds. +Otherwise a random permutation is drawn once with SystemRandom and frozen on disk. +Seeds always derive from SHA256 of `roadscore-judging-v1:` plus canonical route, +first four bytes big endian; labels and submission order do not affect seeds. The tool +refuses an existing output directory. The blind batch ID hashes the private mapping; +no identity or submitter is included in the blind folder. Commit only source; all +batch outputs belong under ignored results/. Do not publish recordings or mappings. + +## Cache and characterize on the Mac + +Local cache sources are read-only. Copies are written under the chosen batch only; +no existing score archives, driver camera, weights or generation assets are copied. + +```sh +python3 roadscore/judging/prepare.py cache \ + --manifest roadscore/results/community_BATCH/manifest.private.json \ + --source /Users/dominickthompson/Desktop/RoadScore/routes +``` + +Missing rlog/front camera and segment holes remain explicit failures. qlog/qcamera +are retained as diagnostic alternatives but do not satisfy the full-data gate. +Each copied file has byte count and SHA-256; corrupt cached copies are repaired from +the source. Local presence cannot prove remote EOF, so extent stays unverified. + +For uncached routes, run the following using an existing **Mac** Python environment +with openpilot imports, native API authentication and requests available: + +```sh +python3 roadscore/judging/prepare.py metadata \ + --manifest roadscore/results/community_BATCH/manifest.private.json +python3 roadscore/judging/prepare.py fetch --logs-only \ + --manifest roadscore/results/community_BATCH/manifest.private.json +python3 roadscore/judging/analyze.py \ + --manifest roadscore/results/community_BATCH/manifest.private.json +``` + +Fetch uses the repository's authenticated route API host fallback and HTTPS assets. +It downloads selected full logs and front camera, reuses verified files, +and saves no tokens or signed URLs. Authentication, access and download failures are +reported by exception type without leaking signed URL text. A fetched listing proves +only the available remote extent, not that the owner uploaded every recorded segment. +The default free-space reserve is 8 GiB, checked before and during each download. +Use `--logs-only` for characterization first, `--preview-video` for smaller qcamera +previews, or omit both for full front video when space allows. Missing rlogs fall back +to qlogs for partial characterization; they remain failures of the full-log gate. Metadata/analysis do not declare missing video ready. No hardware access or +score generation follows fetch. + +Time-to-segment conversion is nominal 60 seconds. Remote fetch also obtains segment +zero as a log-only clock reference for time selections. Analysis uses original log +timestamps and an end-exclusive cut; the reference segment never contributes to a +later excerpt's statistics. Local-only caches lacking segment zero remain blocked. +Camera decode and timestamp alignment still require later replay validation. + +Analysis checks original log hashes, inventories valid message types/rates/gaps, +records timestamp reversals, speed/steering distributions and navigation/blinker +sample coverage. The private reports remain outside the blind folder. These are +technical proxies, not route quality rankings. They must never be imported into the +causal music runtime or used to select a favorable seed. Missing data is not a pass. + +## Blind judging and later handoff + +Open `blind/index.html` locally and load its adjacent `batch.json`. The page uses no +network requests and supports A–K navigation, explicit local media attachment, +1–5 / not-observable ratings, notes, browser-local save and JSON export tied to the +batch ID and anonymous judging session. Playback never autostarts. Browser storage +may be unavailable on file URLs; export remains available. Reopened recordings must +be reattached and acknowledged as reviewed. Blank scores are not zeros. Only give +judges the blind folder and deliberately anonymized approved recordings; never serve +the parent batch directory. Road footage itself can reveal recognizable locations. + +Before the hardware owner later runs final generation, review each private selection, +cache completeness/extent, message coverage and clock ambiguity. Use the same Prism configuration, quality/gesture policy and deterministic route-derived +seed policy consistently across candidates. One official run per candidate is allowed +unless technically invalid; automatic normal quality rerolls remain allowed. Preserve +failed runs; do not choose a winning seed from offline route characterization. Use the +existing event replay audit for timing/causality evidence and collect human ratings +separately. Keep identity reveal and any winner decision deferred until judging ends. + +## Validation + +```sh +python3 -m unittest discover -s roadscore/judging -p 'test_*.py' +``` + +Tests cover normalization/deduplication, explicit range semantics, frozen blind mapping, +privacy boundary, missing/corrupt cache data, and technical analysis validity/time cuts. +No real routes, network, device access or music generation are needed for tests. + +Nine synthetic preparation/cache/analysis tests pass, including resumable authenticated +fetch mocks and Connect/native range distinctions. Python compilation and JavaScript +syntax checks pass. Interactive browser verification was not completed: the browser +security policy rejected the local file URL. No workaround or audio playback was used. diff --git a/roadscore/judging/analyze.py b/roadscore/judging/analyze.py new file mode 100644 index 0000000000..17af20ef84 --- /dev/null +++ b/roadscore/judging/analyze.py @@ -0,0 +1,119 @@ +"""Offline technical characterization only. Never imported by the composer.""" +import argparse +from collections import Counter +import json +import math +from pathlib import Path + +from prepare import digest, write_json + +REQUIRED = ('carState', 'modelV2', 'roadCameraState') + + +def summarize(events, start_s=0, end_s=None): + counts, valid_counts, first, last, gaps = Counter(), Counter(), {}, {}, {} + speeds, steering, signals, navigation = [], [], 0, 0 + origin = None + turns, stops, prior_signal, prior_stopped = 0, 0, False, False + maneuvers = Counter() + previous = None + reversals = 0 + for event in events: + ns, kind, valid, data = event + if origin is None: + origin = ns + t = (ns - origin) / 1e9 + if previous is not None and ns < previous: + reversals += 1 + previous = ns + if t < start_s or (end_s is not None and t >= end_s): + continue + counts[kind] += 1 + valid_counts[kind] += bool(valid) + if kind in last: + gaps[kind] = max(gaps.get(kind, 0), t - last[kind]) + first.setdefault(kind, t) + last[kind] = t + if not valid: + continue + if kind == 'carState': + for field, values in [('vEgo', speeds), ('steeringAngleDeg', steering)]: + value = data.get(field) + if isinstance(value, (float, int)) and math.isfinite(value): + values.append(float(value)) + signal = bool(data.get('leftBlinker') or data.get('rightBlinker')) + stopped = bool(data.get('standstill')) + turns += signal and not prior_signal + stops += stopped and not prior_stopped + prior_signal, prior_stopped = signal, stopped + signals += signal + if kind == 'navInstruction': + navigation += 1 + maneuvers[str(data.get('maneuverType', 'unspecified'))] += 1 + def stats(values): + if not values: + return None + values = sorted(values) + return {'min': values[0], 'median': values[len(values) // 2], 'max': values[-1], 'samples': len(values)} + coverage = {k: {'count': counts[k], 'valid_count': valid_counts[k], 'first_s': first[k], 'last_s': last[k], + 'max_gap_s': gaps.get(k), 'mean_hz': (counts[k] - 1) / (last[k] - first[k]) if last[k] > first[k] else None} + for k in sorted(counts)} + return {'message_coverage': coverage, 'required_valid_messages_present': all(valid_counts[k] > 0 for k in REQUIRED), + 'timestamp_reversals': reversals, 'speed_m_s': stats(speeds), 'steering_degrees': stats(steering), + 'blinker_active_samples': signals, 'blinker_activation_edges': turns, + 'standstill_entry_edges': stops, 'nav_maneuver_sample_counts': dict(maneuvers), 'valid_navigation_messages': navigation, + 'observed_span_s': max(last.values()) - min(first.values()) if last else 0, + 'limitations': ['Offline characterization only; never feed future annotations into runtime.', + 'Presence is not camera decode, navigation accuracy, musical acceptance, or replay timing validation.', + 'Blinker counts are samples, not distinct turns; steering is a proxy, not ground truth.', + 'Origin is first decoded log message; segment numbering is only nominal 60s.']} + + +def read_events(paths): + from openpilot.tools.lib.logreader import _LogFileReader + for path in paths: + for event in _LogFileReader(str(path), sort_by_time=False): + kind = event.which() + data = getattr(event, kind).to_dict() if kind in ('carState', 'navInstruction') else {} + yield event.logMonoTime, kind, event.valid, data + + +def analyze(manifest): + batch = json.loads(manifest.read_text()) + for row in batch['routes']: + cache = row.get('cache', {}) + paths = {} + try: + for asset in cache.get('files', []): + path = manifest.parent / asset['path'] + if path.name.startswith(('rlog', 'qlog')): + if not path.is_file() or digest(path) != asset['sha256']: + raise ValueError('Original log cache checksum mismatch') + paths.setdefault(asset['segment'], path) + if not paths: + raise ValueError('No full logs available') + selection = row['selection'] + first_segment = min(paths) + # Time selections need segment zero to establish actual route-relative origin. + if selection['segments'] is None and first_segment != 0: + raise ValueError('Time/full-route selection needs segment zero to establish its clock origin') + report = summarize(read_events([paths[s] for s in sorted(paths)]), selection['start_s'], selection['end_s']) + report['segments'] = sorted(paths) + report['selection'] = selection + report['complete_selected_logs'] = all(s in paths and paths[s].name.startswith('rlog') for s in cache.get('selected_segments', [])) + report['qlog_fallback_segments'] = [s for s in sorted(paths) if paths[s].name.startswith('qlog')] + report['clock_reference_is_full_log'] = paths[first_segment].name.startswith('rlog') + report['clock_scope'] = 'selected first log' if selection['segments'] is not None else 'route first log' + row['analysis_status'] = 'characterized' if report['required_valid_messages_present'] and report['complete_selected_logs'] else 'incomplete' + except Exception as e: + row['analysis_status'] = 'blocked' + report = {'status': 'blocked', 'reason_type': type(e).__name__, 'reason': str(e), 'note': 'Check cache coverage and local logreader dependencies. No generation attempted.'} + write_json(manifest.parent / 'analysis' / (row['label'] + '.private.json'), report) + print(row['label'], row['analysis_status'], flush=True) + write_json(manifest, batch) + + +if __name__ == '__main__': + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument('--manifest', type=Path, required=True) + analyze(parser.parse_args().manifest) diff --git a/roadscore/judging/index.html b/roadscore/judging/index.html new file mode 100644 index 0000000000..725f2712ff --- /dev/null +++ b/roadscore/judging/index.html @@ -0,0 +1,36 @@ + +
+Eleven anonymous routes. Judge the musical experience and how it responds to the drive.
+ + + +Choose the batch.json supplied with this page. This ties exported ratings to the frozen anonymous mapping without revealing identities.
+ + + + diff --git a/roadscore/judging/prepare.py b/roadscore/judging/prepare.py new file mode 100644 index 0000000000..94658b1dc8 --- /dev/null +++ b/roadscore/judging/prepare.py @@ -0,0 +1,364 @@ +"""Mac-only judging preparation. No replay, generation, device or audio control.""" +import argparse +import hashlib +import json +import math +import os +from pathlib import Path +import random +import re +import shutil +import string +import sys +from urllib.parse import unquote, urlparse + +ROOT = Path(__file__).resolve().parents[1] +sys.path.insert(0, str(ROOT.parent)) +LETTERS = list(string.ascii_uppercase[:11]) +ROUTE = re.compile(r'([0-9a-fA-F]{16})[|/]([a-zA-Z0-9_-]{20})(.*)') +ASSETS = ('rlog.zst', 'rlog.bz2', 'rlog', 'qlog.zst', 'qlog.bz2', 'qlog', 'fcamera.hevc', 'qcamera.ts') + + +def digest(path): + h = hashlib.sha256() + with Path(path).open('rb') as f: + for block in iter(lambda: f.read(1024 * 1024), b''): + h.update(block) + return h.hexdigest() + + +def write_json(path, data): + path = Path(path) + path.parent.mkdir(parents=True, exist_ok=True) + tmp = path.with_suffix(path.suffix + '.tmp') + tmp.write_text(json.dumps(data, indent=2, sort_keys=True) + '\n') + os.chmod(tmp, 0o600) + tmp.replace(path) + + +def normalize(row): + if isinstance(row, str): + row = {'route': row} + if not isinstance(row, dict) or set(row) - {'route', 'segments', 'start_s', 'end_s', 'submitter', 'note'}: + raise ValueError('Unknown submission fields; use the documented schema') + if not isinstance(row.get('route'), str): + raise ValueError('route must be a string') + raw = row['route'].strip() + value = unquote(raw) + is_connect = '://' in value + if is_connect: + url = urlparse(value) + if url.scheme != 'https' or url.netloc not in ('connect.comma.ai', 'connect.konik.ai'): + raise ValueError('Only HTTPS Connect route links are supported') + if url.query or url.fragment: + raise ValueError('URL query/fragment needs an explicit range decision; remove it and provide start_s/end_s or segments') + value = url.path.strip('/') + match = ROUTE.fullmatch(value) + if not match: + raise ValueError('Expected DONGLE/ROUTE, DONGLE|ROUTE or a Connect route link') + dongle, route, suffix = match.groups() + segments = None + resolution = 'resolved syntax; metadata pending' + if is_connect and suffix: + times = re.fullmatch(r'/(\d+)(?:/(\d+))?', suffix) + if not times: + raise ValueError('Connect suffixes are seconds: /START or /START/END') + row = dict(row) + if 'segments' in row or 'start_s' in row or 'end_s' in row: + raise ValueError('Conflicting URL and explicit selections') + row['start_s'] = int(times[1]) + row['end_s'] = int(times[2]) if times[2] is not None else None + if times[2] is None and int(times[1]) > 0: + resolution = 'pending single-timestamp intent: current Connect displays full route without end' + suffix = '' + if suffix: + single = re.fullmatch(r'(?:--|/)(\d+)', suffix) + span = re.fullmatch(r'/(\d+):(\d+)', suffix) + if single: + segments = [int(single[1])] + elif span and int(span[2]) > int(span[1]): + segments = list(range(int(span[1]), int(span[2]))) + else: + raise ValueError('Ambiguous segment suffix; use /N, --N, /START:STOP (exclusive) or explicit segments') + if 'segments' in row: + explicit = row['segments'] + if not isinstance(explicit, list) or not explicit or any(type(x) is not int or x < 0 for x in explicit): + raise ValueError('segments must be a nonempty list of nonnegative integers') + explicit = sorted(set(explicit)) + if segments is not None and explicit != segments: + raise ValueError('Conflicting segment selections') + segments = explicit + start, end = row.get('start_s', 0), row.get('end_s') + for v in (start, end): + if v is not None and (isinstance(v, bool) or not isinstance(v, (int, float)) or not math.isfinite(v) or v < 0): + raise ValueError('Time ranges must contain finite nonnegative seconds') + if start is None or (end is not None and end <= start): + raise ValueError('end_s must be greater than start_s') + if segments is not None and ('start_s' in row or 'end_s' in row): + raise ValueError('Combining segment and time ranges requires a single explicit selection decision') + selection = {'segments': segments, 'start_s': float(start), 'end_s': None if end is None else float(end), + 'time_basis': 'route_start', 'end_exclusive': True} + return {'route': dongle.lower() + '/' + route, 'selection': selection, 'range_resolution': resolution, + 'submission': {'raw': raw, 'submitter': row.get('submitter'), 'note': row.get('note')}} + + +def normalize_all(rows): + unique, duplicates, errors = [], [], [] + keys = {} + for n, row in enumerate(rows, 1): + try: + item = normalize(row) + key = json.dumps([item['route'], item['selection']], sort_keys=True) + if key in keys: + duplicates.append({'submission': n, 'duplicate_of': keys[key], 'provenance': item['submission']}) + else: + item['submission_number'] = n + unique.append(item) + keys[key] = n + except (ValueError, KeyError, TypeError) as e: + errors.append({'submission': n, 'reason': str(e)}) + overlaps = [] + for i, left in enumerate(unique): + for right in unique[i + 1:]: + if left['route'] == right['route']: + overlaps.append([left['submission_number'], right['submission_number']]) + return {'routes': unique, 'duplicates': duplicates, 'errors': errors, 'same_route_different_selection': overlaps} + + +def prepare(source, output, acknowledge=False, existing=None): + rows = json.loads(source.read_text()) + if not isinstance(rows, list): + raise ValueError('Input must be a JSON array') + if output.exists(): + raise ValueError('Output already exists; mapping is immutable. Use a new batch directory') + report = normalize_all(rows) + output.mkdir(parents=True, mode=0o700) + write_json(output / 'normalization.private.json', report) + if report['errors'] or len(report['routes']) != 11 or (report['same_route_different_selection'] and not acknowledge): + raise ValueError('Batch blocked: inspect normalization.private.json; require 11 unique selections, no errors, and acknowledge same-route selections') + shuffled = report['routes'][:] + random.SystemRandom().shuffle(shuffled) + if existing: + inherited = json.loads(existing.read_text()) + prior = inherited.get('submissions', inherited.get('routes', [])) + if {r['route'] for r in prior} != {r['route'] for r in shuffled} or sorted(r['label'] for r in prior) != LETTERS: + raise ValueError('Existing mapping does not match this candidate set') + lookup = {r['route']: r for r in prior} + shuffled.sort(key=lambda r: lookup[r['route']]['label']) + for label, row in zip(LETTERS, shuffled): + row.update({'seed': int.from_bytes(hashlib.sha256(('roadscore-judging-v1:' + row['route']).encode()).digest()[:4], 'big'), 'label': label, 'cache_status': 'not_checked', 'analysis_status': 'not_run', 'score_status': 'not_generated'}) + if existing: + for row in shuffled: + if row['seed'] != lookup[row['route']]['seed']: + raise ValueError('Existing seed violates the deterministic route-derived policy') + batch = {'schema_version': 1, 'input_sha256': digest(source), 'generation_authorized': False, + 'mapping_policy': 'preserved existing mapping' if existing else 'SystemRandom shuffle, frozen at creation; no seed/music optimization', 'routes': shuffled} + if existing: + batch['inherited_manifest_sha256'] = digest(existing) + batch['configuration'] = inherited.get('configuration') + write_json(output / 'manifest.private.json', batch) + write_json(output / 'blind_mapping.private.json', {r['label']: {'route': r['route'], 'selection': r['selection'], 'submission': r['submission']} for r in shuffled}) + blind = output / 'blind' + blind.mkdir(mode=0o700) + shutil.copyfile(Path(__file__).with_name('index.html'), blind / 'index.html') + write_json(blind / 'batch.json', {'batch_id': digest(output / 'blind_mapping.private.json'), 'labels': LETTERS}) + return batch + + +def selected_segments(row, available): + selection = row['selection'] + if selection['segments'] is not None: + wanted = selection['segments'] + elif selection['end_s'] is not None: + wanted = list(range(int(selection['start_s'] // 60), math.ceil(selection['end_s'] / 60))) + else: + # Full/open-ended routes must begin at the selected nominal minute with no holes. + first = int(selection['start_s'] // 60) + wanted = list(range(first, max(available, default=first - 1) + 1)) + return wanted + + +def cache_local(manifest, roots): + """Copy only selected original logs/front video, never prior generated scores.""" + batch = json.loads(manifest.read_text()) + for row in batch['routes']: + dongle, name = row['route'].split('/') + sources = {} + for root in roots: + for parent in (root / dongle / name, root / dongle, root): + if (parent / '.acquiring').exists(): + continue + for folder in sorted(parent.glob(name + '--*')): + suffix = folder.name[len(name) + 2:] + if folder.is_dir() and suffix.isdigit(): + sources.setdefault(int(suffix), []).append(folder) + wanted = selected_segments(row, sources) + files, missing = [], [] + for segment in wanted: + segment_assets = [] + for asset in ASSETS: + src = next((p / asset for p in sources.get(segment, []) if (p / asset).is_file() and (p / asset).stat().st_size), None) + if src is None: + continue + dest = manifest.parent / 'cache' / row['label'] / (name + '--' + str(segment)) / asset + dest.parent.mkdir(parents=True, exist_ok=True) + checksum = digest(src) + if not dest.exists() or digest(dest) != checksum: + tmp = dest.with_suffix(dest.suffix + '.partial') + shutil.copyfile(src, tmp) + if digest(tmp) != checksum: + raise ValueError('Cache copy checksum mismatch') + tmp.replace(dest) + files.append({'segment': segment, 'path': str(dest.relative_to(manifest.parent)), 'bytes': dest.stat().st_size, 'sha256': checksum}) + segment_assets.append(asset) + if not any(a.startswith('rlog') for a in segment_assets): + missing.append({'segment': segment, 'asset': 'rlog'}) + if 'fcamera.hevc' not in segment_assets: + missing.append({'segment': segment, 'asset': 'fcamera.hevc'}) + row['cache'] = {'files': files, 'selected_segments': wanted, 'missing': missing, + 'extent_verified_against_remote_listing': False, + 'note': 'Nominal 60s segment selection; actual timestamps must be checked in analysis. Local presence does not prove remote EOF.'} + row['cache_status'] = 'local_assets_present_extent_unverified' if wanted and not missing else 'incomplete' + write_json(manifest, batch) + return batch + + +def fetch_remote(manifest, logs_only=False, reserve_gib=8, preview_video=False): + """Explicit Mac HTTPS download. Native auth/host fallback, no device transport.""" + from openpilot.tools.lib.api import APIError, CommaApi, route_api_hosts + from openpilot.tools.lib.auth_config import get_token + import requests + batch = json.loads(manifest.read_text()) + for row in batch['routes']: + row['fetch_status'] = 'failed' + try: + listing = None + for host in route_api_hosts(): + try: + listing = CommaApi(get_token(host), host=host).get('v1/route/' + row['route'].replace('/', '|') + '/files', timeout=30) + break + except APIError as e: + if e.status_code != 404: + raise + if listing is None: + raise ValueError('No remote listing') + assets = {} + for urls in listing.values(): + if not isinstance(urls, list): + continue + for url in urls: + parsed = urlparse(url) + parts = unquote(parsed.path).split('/') + if parsed.scheme != 'https' or len(parts) < 2 or not parts[-2].isdigit() or parts[-1] not in ASSETS: + continue + assets[(int(parts[-2]), parts[-1])] = url + wanted = selected_segments(row, [s for s, _ in assets]) + download_segments = sorted(set(wanted + ([0] if row['selection']['segments'] is None else []))) + name = row['route'].split('/')[1] + files, missing = [], [] + previous_files = {f['path']: f for f in row.get('cache', {}).get('files', [])} + row['cache'] = {'files': files, 'selected_segments': wanted, 'missing': missing, + 'extent_verified_against_remote_listing': True, 'remote_segments': sorted({s for s, _ in assets})} + row['cache_status'] = 'acquiring' + for segment in download_segments: + # Prefer full logs and road camera; qlog/qcamera are diagnostic fallbacks. + chosen = [] + groups = (ASSETS[:6],) if logs_only or segment not in wanted else (ASSETS[:6], ASSETS[7:] if preview_video else ASSETS[6:7]) + for alternatives in groups: + chosen += [next((a for a in alternatives if (segment, a) in assets), None)] + for asset in filter(None, chosen): + dest = manifest.parent / 'cache' / row['label'] / (name + '--' + str(segment)) / asset + dest.parent.mkdir(parents=True, exist_ok=True) + previous = previous_files.get(str(dest.relative_to(manifest.parent))) + if not previous or not dest.is_file() or digest(dest) != previous['sha256']: + if shutil.disk_usage(manifest.parent).free < reserve_gib * 1024**3: + raise OSError('Disk reserve reached') + tmp = dest.with_suffix(dest.suffix + '.partial') + with requests.get(assets[(segment, asset)], stream=True, timeout=(20, 90)) as response: + response.raise_for_status() + with tmp.open('wb') as out: + for chunk in response.iter_content(1024 * 1024): + if shutil.disk_usage(manifest.parent).free - len(chunk) < reserve_gib * 1024**3: + raise OSError('Disk reserve reached') + out.write(chunk) + if not tmp.stat().st_size: + raise ValueError('Empty remote asset') + tmp.replace(dest) + files.append({'segment': segment, 'path': str(dest.relative_to(manifest.parent)), 'bytes': dest.stat().st_size, 'sha256': digest(dest), 'clock_reference_only': segment not in wanted}) + write_json(manifest, batch) + if not any(a in chosen for a in ASSETS[:3]): + missing.append({'segment': segment, 'asset': 'rlog'}) + if segment in wanted and 'fcamera.hevc' not in chosen: + missing.append({'segment': segment, 'asset': 'fcamera.hevc'}) + row['cache'] = {'files': files, 'selected_segments': wanted, 'missing': missing, + 'extent_verified_against_remote_listing': True, 'remote_segments': sorted({s for s, _ in assets})} + row['cache_status'] = 'available_listing_cached' if wanted and not missing else 'incomplete' + row['fetch_status'] = 'completed' + print(row['label'], 'cached', len(files), 'files', flush=True) + except Exception as e: + print(row['label'], 'fetch failed', type(e).__name__, flush=True) + # Exception text can contain signed URLs or authentication material. + row['fetch_status'] = 'failed:' + type(e).__name__ + row['cache_status'] = 'fetch_failed_needs_review' + write_json(manifest, batch) + return batch + + +def metadata(manifest): + """Read route extent metadata without coordinates, tokens or signed URL persistence.""" + from openpilot.tools.lib.api import APIError, CommaApi, route_api_hosts + from openpilot.tools.lib.auth_config import get_token + batch = json.loads(manifest.read_text()) + out = {} + for row in batch['routes']: + for host in route_api_hosts(): + try: + meta = CommaApi(get_token(host), host=host).get('v1/route/' + row['route'].replace('/', '|'), timeout=30) + fields = ('duration', 'start_time', 'end_time', 'maxqlog', 'maxlog', 'maxcamera', 'maxqcamera', + 'start_time_utc_millis', 'end_time_utc_millis', 'size') + out[row['label']] = {'status': 'available', 'metadata': {k: meta[k] for k in fields if k in meta}} + break + except APIError as e: + out[row['label']] = {'status': 'failed', 'status_code': e.status_code} + if e.status_code != 404: + break + except Exception as e: + out[row['label']] = {'status': 'failed', 'type': type(e).__name__} + break + write_json(manifest.parent / 'route_metadata.private.json', out) + return out + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + sub = parser.add_subparsers(dest='action', required=True) + p = sub.add_parser('prepare') + p.add_argument('--submissions', required=True, type=Path) + p.add_argument('--output', required=True, type=Path) + p.add_argument('--acknowledge-same-route-selections', action='store_true') + p.add_argument('--existing-manifest', type=Path) + p = sub.add_parser('cache') + p.add_argument('--manifest', required=True, type=Path) + p.add_argument('--source', action='append', default=[], type=Path) + p = sub.add_parser('fetch') + p.add_argument('--manifest', required=True, type=Path) + modes = p.add_mutually_exclusive_group() + modes.add_argument('--logs-only', action='store_true') + modes.add_argument('--preview-video', action='store_true', help='Cache smaller qcamera previews; full-camera readiness remains false') + p.add_argument('--reserve-gib', type=float, default=8) + p = sub.add_parser('metadata') + p.add_argument('--manifest', required=True, type=Path) + args = parser.parse_args() + if args.action == 'prepare': + prepare(args.submissions, args.output, args.acknowledge_same_route_selections, args.existing_manifest) + elif args.action == 'cache': + cache_local(args.manifest, args.source) + elif args.action == 'fetch': + fetch_remote(args.manifest, args.logs_only, args.reserve_gib, args.preview_video) + else: + metadata(args.manifest) + print('Preparation operation complete. No RoadScore generation was run.') + + +if __name__ == '__main__': + main() diff --git a/roadscore/judging/test_prepare.py b/roadscore/judging/test_prepare.py new file mode 100644 index 0000000000..8fd8a92b61 --- /dev/null +++ b/roadscore/judging/test_prepare.py @@ -0,0 +1,122 @@ +import json +from pathlib import Path +import tempfile +import unittest +from unittest.mock import patch +from types import SimpleNamespace + +from prepare import cache_local, normalize, normalize_all, prepare, selected_segments, fetch_remote +from analyze import summarize + +ROUTE = '0000000000000000/2026-01-01--00-00-00' + + +class PreparationTests(unittest.TestCase): + def test_normalization_and_duplicate_provenance(self): + result = normalize_all([ROUTE, ROUTE.replace('/', '|'), 'https://connect.comma.ai/' + ROUTE + '/']) + self.assertEqual(len(result['routes']), 1) + self.assertEqual(len(result['duplicates']), 2) + self.assertEqual(result['errors'], []) + + def test_segment_semantics(self): + self.assertEqual(normalize(ROUTE + '--3')['selection']['segments'], [3]) + self.assertEqual(normalize(ROUTE + '/3:5')['selection']['segments'], [3, 4]) + for suffix in ('/3-5', '/3:', '/5:3', '?t=3', '/-1'): + with self.assertRaises(ValueError): + normalize('https://connect.comma.ai/' + ROUTE + suffix) + with self.assertRaises(ValueError): + normalize({'route': ROUTE + '--3', 'segments': [4]}) + with self.assertRaises(ValueError): + normalize({'route': ROUTE, 'start_s': float('nan')}) + + def test_connect_seconds_are_not_segment_numbers(self): + item = normalize('https://connect.comma.ai/' + ROUTE + '/60') + self.assertEqual(item['selection']['start_s'], 60) + self.assertIsNone(item['selection']['segments']) + self.assertIn('pending', item['range_resolution']) + item = normalize('https://connect.comma.ai/' + ROUTE + '/1572/2029') + self.assertEqual(item['selection']['end_s'] - item['selection']['start_s'], 457) + self.assertEqual(selected_segments(item, range(41)), list(range(26, 34))) + + def test_ranges_are_distinct_not_silently_merged(self): + report = normalize_all([ROUTE, ROUTE + '/2', ROUTE + '/3']) + self.assertEqual(len(report['routes']), 3) + self.assertEqual(len(report['same_route_different_selection']), 3) + row = normalize({'route': ROUTE, 'start_s': 60, 'end_s': 120}) + self.assertEqual(selected_segments(row, [0, 1, 2]), [1]) + + def test_frozen_mapping_and_blind_boundary(self): + with tempfile.TemporaryDirectory() as tmp: + root = Path(tmp) + source = root / 'input.json' + source.write_text(json.dumps([f'{i:016x}/2026-01-01--00-00-00' for i in range(11)])) + output = root / 'batch' + result = prepare(source, output) + self.assertEqual([x['label'] for x in result['routes']], list('ABCDEFGHIJK')) + self.assertFalse(result['generation_authorized']) + with self.assertRaises(ValueError): + prepare(source, output) + for f in (output / 'blind').iterdir(): + self.assertNotIn('2026-01-01', f.read_text()) + self.assertNotIn('submitter', f.read_text()) + + def test_wrong_count_and_invalid_input_block_mapping(self): + with tempfile.TemporaryDirectory() as tmp: + root = Path(tmp); source = root / 'input.json' + source.write_text(json.dumps([ROUTE, 'invalid'])) + with self.assertRaises(ValueError): + prepare(source, root / 'batch') + self.assertFalse((root / 'batch/blind_mapping.private.json').exists()) + self.assertTrue((root / 'batch/normalization.private.json').exists()) + + def test_cache_gap_and_corruption(self): + with tempfile.TemporaryDirectory() as tmp: + root = Path(tmp); sources = root / 'sources'; manifest = root / 'manifest.private.json' + for segment in [0, 2]: + folder = sources / (ROUTE.split('/')[1] + '--' + str(segment)); folder.mkdir(parents=True) + (folder / 'rlog').write_bytes(b'original log') + (folder / 'fcamera.hevc').write_bytes(b'original video') + row = normalize(ROUTE); row['label'] = 'A' + manifest.write_text(json.dumps({'routes': [row]})) + cached = cache_local(manifest, [sources])['routes'][0] + self.assertEqual(cached['cache_status'], 'incomplete') + self.assertEqual(cached['cache']['selected_segments'], [0, 1, 2]) + target = root / cached['cache']['files'][0]['path']; target.write_bytes(b'corrupt') + cache_local(manifest, [sources]) + self.assertEqual(target.read_bytes(), b'original log') + + def test_fetch_preserves_clock_reference_and_resumes_without_download(self): + class Response: + def __enter__(self): return self + def __exit__(self, *args): pass + def raise_for_status(self): pass + def iter_content(self, size): return iter([b'original route log']) + listing = {'rlogs': [f'https://example.com/route/{n}/rlog.zst' for n in range(3)]} + class API: + def __init__(self, *args, **kwargs): pass + def get(self, *args, **kwargs): return listing + with tempfile.TemporaryDirectory() as tmp: + manifest = Path(tmp) / 'manifest.private.json' + row = normalize({'route': ROUTE, 'start_s': 120, 'end_s': 180}); row['label'] = 'A' + manifest.write_text(json.dumps({'routes': [row]})) + modules = {'openpilot.tools.lib.api': SimpleNamespace(APIError=RuntimeError, CommaApi=API, route_api_hosts=lambda: ['https://api.example.com']), + 'openpilot.tools.lib.auth_config': SimpleNamespace(get_token=lambda host: None)} + with patch.dict('sys.modules', modules), patch('requests.get', return_value=Response()) as get: + result = fetch_remote(manifest, logs_only=True, reserve_gib=0)['routes'][0] + self.assertEqual(get.call_count, 2) + self.assertEqual(result['cache']['selected_segments'], [2]) + self.assertTrue(result['cache']['files'][0]['clock_reference_only']) + self.assertEqual(result['cache_status'], 'incomplete') # no video, never a replay pass + fetch_remote(manifest, logs_only=True, reserve_gib=0) + self.assertEqual(get.call_count, 2) + + def test_technical_analysis_does_not_count_invalid_or_future_data(self): + events = [(0, 'carState', True, {'vEgo': 1}), (1_000_000_000, 'modelV2', False, {}), + (2_000_000_000, 'carState', True, {'vEgo': 5}), (3_000_000_000, 'carState', True, {'vEgo': 100})] + result = summarize(events, end_s=3) + self.assertEqual(result['speed_m_s']['max'], 5) + self.assertFalse(result['required_valid_messages_present']) + + +if __name__ == '__main__': + unittest.main()