diff --git a/roadscore/experiments/ace_chestnut_20260916/ace_worker.py b/roadscore/experiments/ace_chestnut_20260916/ace_worker.py index 537a83b118..4cf12844f1 100644 --- a/roadscore/experiments/ace_chestnut_20260916/ace_worker.py +++ b/roadscore/experiments/ace_chestnut_20260916/ace_worker.py @@ -15,7 +15,7 @@ P=Path(__file__).resolve().parent;R=P.parents[1];G=R/'generated' sys.path.insert(0,str(R/'prototype')) from ace_profiles import selected from generation_seed import configured_seed,sample_seed -base_seed=configured_seed() +base_seed=configured_seed(required=True) from quality_gate import QualifiedGenerator,POLICY from link_health import LinkProbe from tinygrad import Device @@ -52,7 +52,7 @@ try: initial=None;last=None;preparation=[];slot=0 while initial is None or len(initial)/48000` chooses one fresh unsigned 32-bit session seed +before launching preparation or replay. It prints the seed and records it in +`session_seed.json`, settings and the completed launch record. Native and remote +workers receive the same seed. Ambient legacy seed variables do not silently +pin ordinary launches to the reference seed. + +`./onroad --roadscore --roadscore-seed 73921` explicitly repeats the +session's sampling stream. Preparation and continuation seeds retain the existing +`roadscore-sample-v1` derivation. The worker rejects missing seeds rather than +falling back to 33602. A running worker with another seed remains protected from +accidental reuse; this change does not kill another owner's service or implement +resident reconditioning. + +Official judging passes its existing route-derived seed explicitly and tags its +origin. The launcher rejects a mismatch. Route-to-seed derivation, quality gates, +retry policy and historical ledgers are unchanged; a changed implementation still +requires a new verified judging freeze before new official runs. + +Normal replay now runs to route EOF unless `--duration` is supplied. Showcase +selection does not introduce route allowlists or special event timestamps. + +CPU tests establish seed selection, deterministic derivation, judging isolation +and propagation; they do not establish musical quality or hardware acceptance. +Repeatability also requires the same conditioning/model/code and causal input +sequence. Archives retain the exact heard output and accepted generation decisions +when exact historical playback is required. + +Pending gated acceptance: ordinary-command launches of the same route with fresh +seeds must produce different compositions under one unchanged strategy; explicit +seed repeats must reproduce musical generation under matching inputs/configuration. +Prompt/tensor adaptation and presentation/UI integration are separate candidates. +No claim of end-to-end acceptance is made before the reviewed hardware run. diff --git a/roadscore/prototype/app.py b/roadscore/prototype/app.py index 91752e355b..f2a81c48d1 100644 --- a/roadscore/prototype/app.py +++ b/roadscore/prototype/app.py @@ -37,7 +37,7 @@ from render_policy import selected as render_mode_selected, validate as validate render_mode=render_mode_selected() from generation_budget import GenerationBudget from generation_seed import configured_seed,sample_seed -base_seed=configured_seed();generation_index=0 +base_seed=configured_seed(required=choice()=="ace");generation_index=0 if base_seed is not None: if choice()!="ace":raise ValueError("Deterministic session seeds currently require ACE") prepared=json.loads((root/"generated/ace_initial.json").read_text()) diff --git a/roadscore/prototype/generation_seed.py b/roadscore/prototype/generation_seed.py index 04f5ba8508..2c65ef0806 100644 --- a/roadscore/prototype/generation_seed.py +++ b/roadscore/prototype/generation_seed.py @@ -3,9 +3,11 @@ import hashlib import os -def configured_seed(): +def configured_seed(required=False): value = os.environ.get('ROADSCORE_GENERATION_SEED') if value is None: + if required: + raise ValueError("ACE requires a session seed; launch through ./onroad --roadscore or provide an explicit seed") return None seed = int(value) if not 0 <= seed < 2**32: diff --git a/roadscore/prototype/normal_onroad.py b/roadscore/prototype/normal_onroad.py index 605de82ee8..8143b3104b 100644 --- a/roadscore/prototype/normal_onroad.py +++ b/roadscore/prototype/normal_onroad.py @@ -6,13 +6,19 @@ No RoadScore event annotations, route allowlist, custom camera drawing, or model import argparse,os,subprocess,time,signal,shlex,json from pathlib import Path from clock_sync import measure +from session_seed import select_session, seed_argument, seed_environment, remote_assignments R=Path(__file__).resolve().parents[1] native=Path('/TICI').exists() def interrupt(*_):raise KeyboardInterrupt signal.signal(signal.SIGTERM,interrupt) -p=argparse.ArgumentParser();p.add_argument('--render-mode',choices=['current','gold-core'],default='current',help='Explicit ACE output-only gold core bypass');p.add_argument('route',nargs='?');p.add_argument('--routeid');p.add_argument('--roadscore',action='store_true',required=True);p.add_argument('--replay',action='store_true',help='Play recorded final score without Chestnut');p.add_argument('--start',type=int,default=0);p.add_argument('--duration',type=float,default=180);p.add_argument('--audible',action='store_true',help='Compatibility flag; output is audible by default outside automated sessions');p.add_argument('--muted',action='store_true');p.add_argument('--no-overlay',action='store_true');p.add_argument('--capture-ui',action='store_true',help='Record the normal UI internally without speaker output');p.add_argument('--audio-device',default=None,help='Development host output device; default is the system output');p.add_argument('--transport-only',action='store_true');p.add_argument('--headless',action='store_true');p.add_argument('--runtime',type=Path,default=Path('/data/openpilot') if native else Path(os.environ.get('ROADSCORE_RUNTIME','/Users/dominickthompson/starpilot/.host_runtime/darwin/worktree')));p.add_argument('--bench',default=device_target());p.add_argument('--composer',choices=['sa3','ace'],default=choice(),help='ACE Prism is the event default; SA3 is an explicit fallback');p.add_argument('--profile',choices=['prism','aurora'],default='prism');a=p.parse_args() +p=argparse.ArgumentParser();p.add_argument('--roadscore-seed',type=seed_argument,help='Reproduce an ACE session; normal launches choose a fresh seed');p.add_argument('--render-mode',choices=['current','gold-core'],default='current',help='Explicit ACE output-only gold core bypass');p.add_argument('route',nargs='?');p.add_argument('--routeid');p.add_argument('--roadscore',action='store_true',required=True);p.add_argument('--replay',action='store_true',help='Play recorded final score without Chestnut');p.add_argument('--start',type=int,default=0);p.add_argument('--duration',type=float,default=float('inf'),help='Optional duration limit; normally replay to route EOF');p.add_argument('--audible',action='store_true',help='Compatibility flag; output is audible by default outside automated sessions');p.add_argument('--muted',action='store_true');p.add_argument('--no-overlay',action='store_true');p.add_argument('--capture-ui',action='store_true',help='Record the normal UI internally without speaker output');p.add_argument('--audio-device',default=None,help='Development host output device; default is the system output');p.add_argument('--transport-only',action='store_true');p.add_argument('--headless',action='store_true');p.add_argument('--runtime',type=Path,default=Path('/data/openpilot') if native else Path(os.environ.get('ROADSCORE_RUNTIME','/Users/dominickthompson/starpilot/.host_runtime/darwin/worktree')));p.add_argument('--bench',default=device_target());p.add_argument('--composer',choices=['sa3','ace'],default=choice(),help='ACE Prism is the event default; SA3 is an explicit fallback');p.add_argument('--profile',choices=['prism','aurora'],default='prism');a=p.parse_args() if a.replay and a.render_mode!='current':raise SystemExit('Stored scores retain their recorded rendering; do not apply gold-core to a finished mix') if a.render_mode=='gold-core' and a.composer!='ace':raise SystemExit('Gold core requires ACE') +if a.roadscore_seed is not None and (a.replay or a.composer!='ace'):raise SystemExit('--roadscore-seed applies only to fresh ACE generation') +session=None +if not a.replay and a.composer=='ace': + judging_seed=os.environ.get('ROADSCORE_GENERATION_SEED') if os.environ.get('ROADSCORE_SEED_ORIGIN')=='judging-route' and a.roadscore_seed is not None else None + session=select_session(a.roadscore_seed,judging_seed=judging_seed) from settings import Settings,resolve_route a.routeid=resolve_route(p,a.route,a.routeid) settings=Settings(mode='stored' if a.replay else 'generate',muted=a.muted,overlay=not a.no_overlay,output_device=a.audio_device) @@ -24,7 +30,11 @@ if native: from native_ownership import verify_offroad verify_offroad() out=R/'results'/('normal_'+str(int(time.time())));out.mkdir();env=os.environ.copy();env.update(PYTHONDONTWRITEBYTECODE='1',ZMQ='1',OPENPILOT_ZMQ_NAMESPACE='roadscore-native-'+str(os.getpid()),PARAMS_ROOT=str(out/'params'),BASEDIR=str(rt),NOBOARD='1',SIMULATION='1',SKIP_FW_QUERY='1',BIG='0',SP_ALLOW_DESKTOP_FAKE_WIFI='0',SP_ALLOW_DESKTOP_FAKE_BLUETOOTH='0',SP_ONROAD_NAV_DEMO='0',SP_ONROAD_CEM_DEMO='0') -(out/'settings.json').write_text(json.dumps({**settings.snapshot(a.headless),'composer':a.composer,'profile':a.profile,'render_mode':a.render_mode},indent=2)) +if session: + env.update(seed_environment(session));(out/'session_seed.json').write_text(json.dumps(session,indent=2));print('RoadScore session seed:',session['generation_seed'],'('+session['seed_origin']+')',flush=True) +else: + env.pop('ROADSCORE_GENERATION_SEED',None);env.pop('ROADSCORE_SEED_ORIGIN',None) +(out/'settings.json').write_text(json.dumps({**settings.snapshot(a.headless),'composer':a.composer,'profile':a.profile,'render_mode':a.render_mode,**(session or {})},indent=2)) env['ROADSCORE_RENDER_MODE']=a.render_mode env['ROADSCORE_COMPOSER']=a.composer env['ROADSCORE_ACE_PROFILE']=a.profile @@ -78,7 +88,7 @@ try: if sender.poll() is not None or time.monotonic()>deadline:raise RuntimeError('Stored audio failed; see stored_audio.log') time.sleep(.1) else: - receiver_command=(['env','-u','ZMQ','ROADSCORE_AUDIBLE='+('1' if a.audible else '0'),'bash',str(R/'prototype/native_receiver.sh'),a.routeid] if native else ['ssh',a.bench,'ROADSCORE_RENDER_MODE='+a.render_mode+' ROADSCORE_COMPOSER='+a.composer+' ROADSCORE_ACE_PROFILE='+a.profile+' ROADSCORE_PCM_RETURN=1 '+('ROADSCORE_NO_GENERATION=1 ' if a.transport_only else '')+'bash /data/roadscore/prototype/native_receiver.sh '+shlex.quote(a.routeid)]) + receiver_command=(['env','-u','ZMQ','ROADSCORE_AUDIBLE='+('1' if a.audible else '0'),'bash',str(R/'prototype/native_receiver.sh'),a.routeid] if native else ['ssh',a.bench,(remote_assignments(session) if session else '')+'ROADSCORE_RENDER_MODE='+a.render_mode+' ROADSCORE_COMPOSER='+a.composer+' ROADSCORE_ACE_PROFILE='+a.profile+' ROADSCORE_PCM_RETURN=1 '+('ROADSCORE_NO_GENERATION=1 ' if a.transport_only else '')+'bash /data/roadscore/prototype/native_receiver.sh '+shlex.quote(a.routeid)]) receiver=launch(receiver_command,'receiver',stdin=subprocess.PIPE) deadline=time.monotonic()+(1560 if a.composer=='ace' else 420) while b'BRIDGE_READY' not in (out/'receiver.log').read_bytes(): @@ -121,7 +131,7 @@ try: time.sleep(.5) if sender.poll() not in [None,0]:raise RuntimeError('Replay sender failed') if sender.poll() is None:os.killpg(sender.pid,signal.SIGTERM);sender.wait(timeout=5) - (out/'launch.json').write_text(json.dumps({'render_mode':a.render_mode,'route':a.routeid,'mode':'stored-score' if a.replay else 'fresh-generation','host':'comma' if native else 'development host','compute':'none' if a.replay else ('local Chestnut' if native else 'remote Chestnut'),'local_cache':str(local) if local else None,'native_replay_args':args,'end_reason':end_reason,'duration_wall':time.monotonic()-started,'startup_seconds':started-launch_started,'headless':a.headless,'ui':'existing selfdrive/ui/ui.py','muted':not a.audible},indent=2)) + (out/'launch.json').write_text(json.dumps({**(session or {}),'render_mode':a.render_mode,'route':a.routeid,'mode':'stored-score' if a.replay else 'fresh-generation','host':'comma' if native else 'development host','compute':'none' if a.replay else ('local Chestnut' if native else 'remote Chestnut'),'local_cache':str(local) if local else None,'native_replay_args':args,'end_reason':end_reason,'duration_wall':time.monotonic()-started,'startup_seconds':started-launch_started,'headless':a.headless,'ui':'existing selfdrive/ui/ui.py','muted':not a.audible},indent=2)) if not a.replay: receiver.wait(timeout=35) if not (out/'replay_origin.json').exists():raise RuntimeError('Replay delivered no model clock; refusing empty score') @@ -135,7 +145,7 @@ try: subprocess.run([str(py),str(R/'prototype/archive_native.py'),str(out),a.routeid,str(a.start)],env=env,cwd=rt,check=True) if not native: # Preserve generated decisions with the exact host presentation audio. - for filename in ['summary.json','jobs.jsonl','boundaries.jsonl','ending.json','bridge.json','trace.jsonl','runtime_manifest.json','song_form.json','gesture_grid.json','gestures.json','composition.json','quality_events.jsonl']: + for filename in ['summary.json','jobs.jsonl','boundaries.jsonl','ending.json','bridge.json','trace.jsonl','runtime_manifest.json','song_form.json','gesture_grid.json','gestures.json','composition.json','quality_events.jsonl','shaker_grid.json','shaker_events.json','core_apex_events.json']: subprocess.run(['scp',a.bench+':/data/roadscore/results/current/'+filename,str(out/filename)],stdout=subprocess.DEVNULL,stderr=subprocess.DEVNULL) subprocess.run(['scp','-r',a.bench+':/data/roadscore/results/current/quality',str(out/'quality')],stdout=subprocess.DEVNULL,stderr=subprocess.DEVNULL) if a.composer=='ace':subprocess.run(['scp',a.bench+':/data/roadscore/generated/ace_link.jsonl',str(out/'ace_link.jsonl')],stdout=subprocess.DEVNULL,stderr=subprocess.DEVNULL) diff --git a/roadscore/prototype/session_seed.py b/roadscore/prototype/session_seed.py new file mode 100644 index 0000000000..85504ec7e1 --- /dev/null +++ b/roadscore/prototype/session_seed.py @@ -0,0 +1,40 @@ +"""One persisted seed decision per normal launch; official judging stays explicit.""" +import secrets + +MAX_SEED = 2**32 + + +def seed_argument(value): + try: + seed = int(value) + except (TypeError, ValueError) as error: + raise ValueError('RoadScore seed must be an unsigned 32-bit integer') from error + if isinstance(value, bool) or str(seed) != str(value) or not 0 <= seed < MAX_SEED: + raise ValueError('RoadScore seed must be an unsigned 32-bit integer') + return seed + + +def select_session(explicit=None, *, judging_seed=None, random_bits=None): + if judging_seed is not None: + expected = seed_argument(judging_seed) + if explicit is None or seed_argument(explicit) != expected: + raise ValueError('Official judging requires its unchanged route-derived CLI seed') + seed, origin = expected, 'judging-route' + elif explicit is not None: + seed, origin = seed_argument(explicit), 'explicit' + else: + seed, origin = (random_bits or secrets.randbits)(32), 'fresh-session' + seed = seed_argument(seed) + return {'generation_seed': seed, 'seed_origin': origin, + 'sample_seed_policy': 'roadscore-sample-v1', + 'reproduce_cli': ['--roadscore-seed', str(seed)]} + + +def seed_environment(session): + return {'ROADSCORE_GENERATION_SEED': str(session['generation_seed']), + 'ROADSCORE_SEED_ORIGIN': session['seed_origin']} + + +def remote_assignments(session): + # Values are constrained to uint32 and the closed origin set above. + return ' '.join(key + '=' + value for key, value in seed_environment(session).items()) + ' ' diff --git a/roadscore/prototype/test_session_seed.py b/roadscore/prototype/test_session_seed.py new file mode 100644 index 0000000000..fc67788f41 --- /dev/null +++ b/roadscore/prototype/test_session_seed.py @@ -0,0 +1,59 @@ +import os +import shlex +import unittest +from unittest.mock import Mock, patch +from generation_seed import configured_seed, sample_seed +from session_seed import seed_argument, select_session, seed_environment, remote_assignments + + +class SessionSeedTests(unittest.TestCase): + def test_each_normal_launch_draws_once_and_does_not_inherit_gold(self): + draw = Mock(side_effect=[18007211, 28899402, 79121330]) + with patch.dict(os.environ, {'ROADSCORE_GENERATION_SEED': '33602'}): + sessions = [select_session(random_bits=draw) for _ in range(3)] + self.assertEqual(draw.call_count, 3) + self.assertEqual(len({s['generation_seed'] for s in sessions}), 3) + self.assertTrue(all(s['seed_origin'] == 'fresh-session' for s in sessions)) + streams = [tuple(sample_seed(s['generation_seed'], 'prepare', i) for i in range(4)) for s in sessions] + self.assertEqual(len(set(streams)), 3) + + def test_explicit_session_repeats_initial_and_continuation_streams(self): + forbidden = Mock(side_effect=AssertionError('Explicit seed must not draw randomness')) + a = select_session(73921, random_bits=forbidden) + b = select_session('73921', random_bits=forbidden) + self.assertEqual(a, b) + for phase in ('prepare', 'continuation'): + self.assertEqual([sample_seed(a['generation_seed'], phase, i) for i in range(8)], + [sample_seed(b['generation_seed'], phase, i) for i in range(8)]) + self.assertEqual(a['reproduce_cli'], ['--roadscore-seed', '73921']) + + def test_official_seed_is_preserved_and_conflict_refused(self): + s = select_session(114992, judging_seed='114992') + self.assertEqual(s['seed_origin'], 'judging-route') + self.assertEqual(s['generation_seed'], 114992) + for explicit in (None, 114993): + with self.assertRaises(ValueError): + select_session(explicit, judging_seed=114992) + + def test_native_and_remote_receive_identical_seed_environment(self): + for source in (select_session(0), select_session(2**32-1, judging_seed=2**32-1)): + native = seed_environment(source) + remote = dict(part.split('=', 1) for part in shlex.split(remote_assignments(source))) + self.assertEqual(remote, native) + with patch.dict(os.environ, remote, clear=True): + self.assertEqual(configured_seed(required=True), source['generation_seed']) + + def test_worker_cannot_fall_back_to_diagnostic_seed(self): + with patch.dict(os.environ, {}, clear=True): + self.assertIsNone(configured_seed()) # read-only legacy callers remain compatible + with self.assertRaisesRegex(ValueError, 'session seed'): + configured_seed(required=True) + + def test_reject_invalid_seed_without_random_or_shell_input(self): + for bad in (-1, 2**32, '1.5', 'nan', '$(anything)', True): + with self.assertRaises(ValueError): + seed_argument(bad) + + +if __name__ == '__main__': + unittest.main() diff --git a/roadscore/tools/judging_run.py b/roadscore/tools/judging_run.py index fa91eb6cab..27a7bb9db1 100644 --- a/roadscore/tools/judging_run.py +++ b/roadscore/tools/judging_run.py @@ -105,6 +105,7 @@ def main(): json.dump(record, stream, indent=2) os.chmod(ledger, 0o600) env['ROADSCORE_GENERATION_SEED'] = str(row['seed']) + env['ROADSCORE_SEED_ORIGIN'] = 'judging-route' service = ROOT / 'prototype/worker_service.py' try: if not args.resume_preparation: @@ -136,6 +137,7 @@ def main(): save(ledger, record) command = [str(ROOT.parent / 'onroad'), '--routeid', row['route'], '--roadscore', '--composer', 'ace', '--profile', 'prism', '--muted', + '--roadscore-seed', str(row['seed']), '--start', str(row.get('start_seconds', 0)), '--duration', str(row.get('duration_seconds', 86400))] with (out / f'console_{args.label}.log').open('wb') as log: result = subprocess.run(command, env=env, cwd=ROOT.parent, stdout=log, stderr=subprocess.STDOUT)