Compare commits

..

5 Commits

Author SHA1 Message Date
firestarsdog 089fd38fc2 Add minimal ELM327 RFCOMM support 2026-09-20 19:55:18 -04:00
whoisdomi f15a1974d5 Ioniq 6 turn blips 2026-09-19 21:01:03 -05:00
whoisdomi 08139a021a Car Date/Time fallback when gps/wifi not available 2026-09-19 21:01:02 -05:00
whoisdomi fbe982f47b Ioniq 6 Date/Time DBC Signal
Added date/time signal from Ioniq 6 can
2026-09-19 21:01:01 -05:00
firestarsdog 5e6e978438 Gen2 Bolt HSA Fix? Maybe?
0x315 : Mode 1 when not engaged, not Mode 9
2026-09-19 18:25:59 -04:00
609 changed files with 540 additions and 32164 deletions
-3
View File
@@ -134,6 +134,3 @@ Pipfile
!rednose_repo/rednose/helpers/ekf_sym_pyx.so
!panda/board/obj/
!panda/board/obj/**
# Private hackathon context and community route submissions
/ROADSCORE_COMMA_HACK_7_CONTEXT.md
-6
View File
@@ -3,10 +3,4 @@
set -euo pipefail
ROOT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
# RoadScore owns its replay/audio lifecycle only when explicitly requested.
for argument in "$@"; do
if [[ "$argument" == "--roadscore" ]]; then
exec "${ROOT_DIR}/roadscore/onroad" "$@"
fi
done
exec "${ROOT_DIR}/scripts/host_tool_runner.sh" onroad "$@"
+1 -1
View File
@@ -1197,7 +1197,7 @@ class CarController(CarControllerBase):
# cannot linger after a disengage or main-off event.
if should_send_bolt_acc_pedal_friction:
can_sends.append(gmcan.create_friction_brake_command(
self.packer_ch, friction_brake_bus, experiment_brake, idx, bolt_acc_pedal_friction_main_on,
self.packer_ch, friction_brake_bus, experiment_brake, idx, bolt_acc_pedal_friction_main_on and CC.longActive,
near_stop, at_full_stop, self.CP))
if self.CP.carFingerprint not in CC_ONLY_CAR:
friction_brake_bus = get_friction_brake_bus(self.CP)
@@ -441,12 +441,6 @@ def suppress_redundant_gv70_brake_cancel(CP, brake_pressed: bool, lat_active: bo
)
def clear_ioniq_6_torque_when_request_inactive(CP, apply_torque: int, apply_steer_req: bool) -> int:
if CP.carFingerprint == CAR.HYUNDAI_IONIQ_6 and not apply_steer_req:
return 0
return apply_torque
class CarController(CarControllerBase):
def __init__(self, dbc_names, CP):
super().__init__(dbc_names, CP)
@@ -642,8 +636,6 @@ class CarController(CarControllerBase):
if not CC.latActive:
apply_torque = 0
apply_torque = clear_ioniq_6_torque_when_request_inactive(self.CP, apply_torque, apply_steer_req)
# Hold torque with induced temporary fault when cutting the actuation bit
# FIXME: we don't use this with CAN FD?
torque_fault = CC.latActive and not apply_steer_req
@@ -20,8 +20,7 @@ from opendbc.car.hyundai.carcontroller import CarController, CANCEL_BUTTON_DELAY
should_track_stop_accel_directly_for_car, \
preserve_stock_canfd_lfa_status, \
preserve_stock_canfd_lkas_status, \
suppress_redundant_gv70_brake_cancel, \
clear_ioniq_6_torque_when_request_inactive
suppress_redundant_gv70_brake_cancel
from opendbc.car.hyundai.carstate import CarState, decode_canfd_camera_lead, decode_ioniq_6_blindspot_radar_state, \
get_canfd_cruise_available
from opendbc.car.hyundai.interface import CarInterface, KIA_EV9_ACCEL_MAX, get_communication_control_request
@@ -680,14 +679,7 @@ class TestHyundaiFingerprint:
assert not (CP.flags & HyundaiFlags.CANFD_LKA_STEERING)
assert bool(CP.flags & HyundaiFlags.CANFD_CAMERA_SCC)
def test_ioniq_6_clears_torque_with_inactive_safety_request(self):
ioniq_6_cp = SimpleNamespace(carFingerprint=CAR.HYUNDAI_IONIQ_6)
other_cp = SimpleNamespace(carFingerprint=CAR.KIA_EV6)
assert clear_ioniq_6_torque_when_request_inactive(ioniq_6_cp, -409, False) == 0
assert clear_ioniq_6_torque_when_request_inactive(ioniq_6_cp, -409, True) == -409
assert clear_ioniq_6_torque_when_request_inactive(other_cp, -409, False) == -409
def test_palisade_2023_uses_can_canfd_blended_layout(self):
palisade_2023 = CarInterface.get_params(CAR.HYUNDAI_PALISADE_2023, gen_empty_fingerprint(), [], True, False, False, None)
assert palisade_2023.flags & HyundaiFlags.CAN_CANFD_BLENDED
assert DBC[palisade_2023.carFingerprint][Bus.pt] == "hyundai_palisade_2023_generated"
@@ -727,10 +727,14 @@ BO_ 1259 LOCAL_TIME2: 8 XXX
SG_ NEW_SIGNAL_3 : 39|1@0+ (1,0) [0|1] "" XXX
BO_ 1264 LOCAL_TIME: 8 XXX
SG_ HOURS : 12|5@0+ (1,0) [0|31] "" XXX
SG_ MINUTES : 21|6@0+ (1,0) [0|63] "" XXX
SG_ SECONDS : 31|8@0+ (1,0) [0|59] "" XXX
SG_ HOURS : 8|8@1+ (1,0) [0|23] "" XXX
SG_ MINUTES : 16|8@1+ (1,0) [0|59] "" XXX
SG_ SECONDS : 24|8@1+ (1,0) [0|59] "" XXX
SG_ MONTH : 34|4@1+ (1,0) [1|12] "" XXX
SG_ YEAR : 40|8@1+ (1,2000) [2000|2255] "" XXX
SG_ DAY : 48|8@1+ (1,0) [1|31] "" XXX
CM_ BO_ 1264 "Cluster wall clock, 1Hz. Local time, not UTC. All 0xFF until the cluster initializes.";
CM_ SG_ 96 BRAKE_PRESSURE "User applied brake pedal pressure. Ramps from computer applied pressure on falling edge of cruise. Cruise cancels if !=0";
CM_ SG_ 101 BRAKE_POSITION "User applied brake pedal position, max is ~700. Signed on some vehicles";
CM_ SG_ 203 ADAS_ActvACISta "ADAS Active AngleControlInterface State";
@@ -964,10 +964,14 @@ BO_ 1259 LOCAL_TIME2: 8 XXX
SG_ NEW_SIGNAL_3 : 39|1@0+ (1,0) [0|1] "" XXX
BO_ 1264 LOCAL_TIME: 8 XXX
SG_ HOURS : 12|5@0+ (1,0) [0|31] "" XXX
SG_ MINUTES : 21|6@0+ (1,0) [0|63] "" XXX
SG_ SECONDS : 31|8@0+ (1,0) [0|59] "" XXX
SG_ HOURS : 8|8@1+ (1,0) [0|23] "" XXX
SG_ MINUTES : 16|8@1+ (1,0) [0|59] "" XXX
SG_ SECONDS : 24|8@1+ (1,0) [0|59] "" XXX
SG_ MONTH : 34|4@1+ (1,0) [1|12] "" XXX
SG_ YEAR : 40|8@1+ (1,2000) [2000|2255] "" XXX
SG_ DAY : 48|8@1+ (1,0) [1|31] "" XXX
CM_ BO_ 1264 "Cluster wall clock, 1Hz. Local time, not UTC. All 0xFF until the cluster initializes.";
CM_ SG_ 96 BRAKE_PRESSURE "User applied brake pedal pressure. Ramps from computer applied pressure on falling edge of cruise. Cruise cancels if !=0";
CM_ SG_ 101 BRAKE_POSITION "User applied brake pedal position, max is ~700. Signed on some vehicles";
CM_ SG_ 203 ADAS_ActvACISta "ADAS Active AngleControlInterface State";
+2
View File
@@ -179,6 +179,8 @@ testpaths = [
"system/tests",
"system/ubloxd",
"system/webrtc",
"starpilot/system/bluetooth/tests",
"starpilot/system/obd/tests",
"tools/lib/tests",
"tools/replay",
"tools/cabana",
-45
View File
@@ -1,45 +0,0 @@
routes/
results/
*.wav
*.npy
*.npz
*.hevc
*.private.json
__pycache__/
.DS_Store
assets/
*.mp4
*.png
.analysis-venv/
.session-muted
experiments/**/vendor/
experiments/**/venv/
experiments/**/models/
experiments/**/cache/
.cache/
# Event runtime data remain local
generated/
native_build/
external/
*.safetensors
*.pt
*.pth
*.onnx
*.flac
*.log
*.jsonl
**/weights*/
**/vae_weights/
**/.venv/
worker_power*.json
runtime.json
# Event scratch and private fixture configuration
tmp/
*.private.env
-147
View File
@@ -1,147 +0,0 @@
> **September16 stability update from b719139:** See [ACE_STABILITY.md](ACE_STABILITY.md) for the current acceptance result and [the port review](results/ace_stability_20260916/index.html). The engineering history below predates the stronger paired tests. Warm fixed-window generation and multiple replay/soak runs pass, but official-reference interior silence and three intermittent GPU/link failures keep overall acceptance open. No general alignment or mixed-precision optimization was promoted.
# ACE-Step Chestnut engineering — from1c00780
Human accepted ACE's musical superiority and seamless verse→chorus example. ACE is now primary composition candidate; SA3 stays intact as fallback. YuE's unwanted vocals, silence and execution burden remove it from this pass's priorities. All physical output stays muted; work remains private/offroad.
## First measured boundary
Exact safetensor element census: main decoder/DiT1,575,458,880 parameters (3,150,917,760 FP16 bytes); condition encoder608,367,616; detokenizer105,011,776; audio tokenizer105,032,198; null embedding2048. This supports testing staged residency instead of assuming the entire package must fit.
The first trained sliding-attention block executes on Chestnut. After54.25s initial kernel compilation and1.45s capture, repeated execution took15.8–16.0ms for160 tokens and48 conditioning tokens. Peak process RSS145.35MiB; tracked GPU139.83MB; loaded parameter bytes134.26MB. Relative output RMSE0.000890 and cosine0.9999983 against Torch FP16. The initial fixed absolute-error limit0.05 failed (max0.25), because these synthetic trained-weight activations reach170.5; peak-normalized error is0.001466. Preserve this failed assertion and use scale-aware metrics plus full real-input validation next. This is a component result, not generated-audio success.
Operator path so far: dense matmul, FP32 RMS reduction with FP16 output, split-half RoPE, grouped-query attention, bidirectional sliding mask, cross-attention, gated SiLU feedforward and AdaLN. All execute through native tinygrad; no PyTorch compatibility layer. Full/sliding attention at long duration is performance-sensitive. Input/output convolution uses non-overlapping two-frame patches and can be represented exactly by reshape/linear transforms.
```mermaid
flowchart LR
A[Text + section tags] --> B[Text embedding + lyric/timbre encoder]
R[Reference audio] --> V[VAE encode]
V --> B
A --> P[Optional 5Hz planner]
P --> D[Detokenizer to25Hz hints]
B --> C[Compact conditioning tensor]
D --> L[Context latents + masks]
C --> M[Chestnut:24-layer DiT,8 flow steps]
L --> M
M --> O[25Hz audio latents]
O --> Q[VAE decode]
Q --> PCM[PCM → gestures → playback/archive]
```
Initial experiment uses the Mac to prepare/cache actual conditioning and reference data, keeping planner/encoders off Chestnut while validating the heavy DiT. This is host-assisted research, not a permanent Mac requirement or proof that comma CPU can perform all preparation within budget. Main-model residency, activations and decode will be measured before choosing quantization or staging.
## Full trained stack milestone
The actual 15-second conditioning path has 375 latent frames and 243 conditioning tokens. The FP16 native 24-layer DiT matches an independent official FP16 MPS forward at relative RMSE 0.005884 and peak-normalized error 0.009549. Both satisfy the preselected 1% criteria. Main weights: 3,150,917,760 bytes; tracked end allocation: 3,160,789,808 bytes; process peak host RSS: 179.14 MiB. These are tracked allocation/end values, not proof of modeld coexistence.
Initial default tinygrad kernels: 32.41 s weight load; 43.88 s first forward; 22.47 s capture; 4.97 s warm forward. Actual eight-step generation: 40.40 s for 15 s audio (latent RTF 2.69). Official host VAE has decoded these latents to a WAV; this first audio path is explicitly host-assisted and is not a local standalone deployment. Initial PCM export clipped peaks above one; the listening copy uses documented uniform peak normalization.
A concrete optimization hypothesis is under test: tinygrad defaults TC_OPT=0 and rejects matrix-tile padding unless TC_OPT>=2. The initial synthetic block had tile-aligned lengths160/48, whereas real inputs have lengths188/243. Enable the existing safe padding optimizer, repeat numerical validation, and measure before changing the model architecture or reducing precision. Preserve both original and optimized evidence.
## Matrix padding result
`TC_OPT=2` enables existing tensor-core tile padding without changing the model. It reduced warm DiT forward from4.97s to0.685s and actual15s eight-step generation to5.872s (latent RTF0.3915). The first optimized attempt failed the FP16 peak-normalized1% gate at1.0896%, despite improved RMSE. An independent comparison against the already-captured official FP32 output passes the unchanged1% limits: RMSE0.4284%, peak-relative0.6511%, cosine0.9999908. Both attempts and the precision audit are preserved. This is numerical evidence, not subjective listening approval.
The sampler presently uses eight-step Euler with DCW disabled; the initial official reference audio used default DCW. Numerical DiT comparisons are unaffected by that sampler distinction. Listening comparisons must identify it. Native VAE also passes: synthetic reference relative RMSE0.2214%, warm0.481s for1.28s output,216MB tracked GPU and284MiB host peak. Actual full-length decode and duration scaling are under measurement.
## Reference, continuation, and the host boundary
The initial human-approved transition is reference-conditioned generation, not proof of exact prefix continuation. This pass captures actual official repaint requests:12s of already-generated musical context plus28s to generate; explicit mask has300 preserved and700 generated latent frames. The native conditioning prefix exactly equals the clean source prefix. Repaint injects the noised clean source for the first4 of8 Euler steps, then performs the official12-frame boundary blend. DCW is explicitly disabled in these matched repaint requests. This is a small sampler extension, not another model/runtime.
Potential self-contained runtime for a prepared identity: persist text/reference encoder output and the silence/context template, put the previous generated latent tail into the new prefix, and generate/decode on Chestnut. Preparing a new identity/arbitrary text still needs the unported planner/conditioning stack; this pass currently prepares it on the Mac. Shipping prepared embeddings does not demonstrate on-comma arbitrary-prompt preparation. No permanent Mac transport is inherently needed by the DiT/VAE loop, but that loop still needs validation and replay integration.
## Quantization scope
FP16 fits. A bounded per-row INT8 storage experiment is prepared (271 matrices, payload1,578,352,768bytes versus3,150,917,760bytes). It transfers INT8 and dequantizes once on GPU into resident FP16. Thus it may improve startup/transport and trades weight fidelity; it does **not** claim lower resident FP16 memory or native INT8 matrix multiplication. Only retain it if benchmark and listening evidence justify the trade. No lower-bit or broad compatibility work is planned merely to force a pass.
## First complete decoded benchmarks and stability limit
15s:5.913s generation +1.508s native decode =7.420s, RTF0.495.30s:10.717s +1.828s =12.546s, RTF0.418. Actual full native VAE output matches official FP32 decoding of the same generated latents at relative RMSE0.000890. Listening approval remains pending. Payload normalization is documented and preserves dynamics.
Cold work remains substantial: main load~32s, VAE~6s, initial forward~72s plus26s capture; first15s VAE call205s plus8s capture. These are not included in warm RTF. A readiness/prebuffer stage is mandatory.
The combined15→30→45 run **hung during45s sampling after a valid1.781s warm forward**: USB copyin drain exceeded10s, then GPU timeline failed. No45s generated/decode result is claimed. Owned process was terminated, power snapshot restored, and a new isolated retry was launched with explicit synchronization before timestep USB transfer and a bounded decoder. The root cause is not yet established; do not call this a stable continuous or production pass based only on warm throughput.
At30s the end tracked allocation was5.007GB and host peak520MiB. Retaining full-length decoder graphs at several lengths is unnecessary. The bounded decoder uses375latent-frame windows,250-frame cores, and at least62frames of interior context, discarding overlap rather than audio crossfading. It will be compared with a full decode; exact chunk equivalence is not assumed.
For arbitrary identities, native Qwen preparation remains unported. Both actual configs have28layers,8KVheads,128head dimension. An FP16 autoregressive KV cache costs114,688bytes per token (~235MB at2048tokens); text encoder can be run without persistent decode cache. Planner, text encoder, conditioning encoder, DiT, and VAE can have separate lifetimes. Present Mac preparation for60/90s measured28.74/27.75s, excluding service initialization. Do not treat those as comma CPU measurements.
## Minimum runtime/buffer design
- Prepare identity and role embeddings before playback; retain only compact tensors. This is currently a Mac development step and should be clearly surfaced as preparation, not hidden inside a claimed comma benchmark.
- Keep the FP16 DiT resident, plus a bounded-window VAE. Heavy waveform generation stays on Chestnut. CPU owns seeds, small sampler bookkeeping, request scheduling, and final arrangement/capture.
- Generate a first30–60s section before starting playback. A cold device is not ready after simply loading weights: graph construction/capture must also complete.
- For later sections, preserve the actual previous12s latent tail and repaint the next28s. Keep at least generation-time plus jitter in the playback buffer. Measure RTF against **new28s**, not the40s output containing its old prefix.
- Current nav horizon may request an outro while the destination is still tens of seconds away; model trajectory gestures still own curve timing. Never inspect future recorded route points to schedule music.
- Keep final PCM/clock/archive independent of the composer. A generic result needs new-audio boundaries, context duration, wall time, identity/preparation provenance, and capability metadata; SA3-specific latent-frame constants must not be reused for ACE.
- SA3 remains an intact alternate backend. An ACE→SA3 continuous handoff would require a validated codec/conditioning bridge and musical review; it is not an automatic fallback merely because both models make WAVs.
For RTF `r`, steady buffer slope is `1/r - 1` seconds per second. `r<1` can build margin; `r>1` drains any finite prebuffer eventually (lifetime `B0/(1-1/r)`). A large initial buffer cannot make a persistently slower composer continuously viable. The current15/30s warm device results have margin, but the USB hang prevents claiming sustained reliability yet. Additional Mac preparation costs must be included if generating new conditions per request; prepared-identity reuse is a different, explicit capability.
## Bounded-decoder duration results
Synchronized45s:14.578s generation +7.941s decode =22.518s (RTF0.500), tracked high-water4,196,265,570bytes; host405MiB. Repeated30s:10.801+4.591=15.392s (RTF0.513). Bounded30s PCM is bit-identical to the prior full native30s decode. Bounded45s versus official FP32 VAE: relative RMSE0.000856 and peak-relative0.007521.
60s exhibits a shape-dependent slowdown: warm forward8.781s, generation70.711s +decode9.357s =80.068s (RTF1.334), tracked peak4.219GB, host471MiB. A single explicit masked token-alignment experiment is prepared to test whether this is an avoidable kernel-shape issue.
90s produced finite(1,2250,64) latents, then the reused bounded decoder hit a30s GPU timeline timeout during output readback. Exact90s sampling time was not persisted before that failure, so do not infer an exact RTF from the warm-forward projection. Owned process was terminated and power restored. This second failure means explicit step synchronization is **not** established as a complete fix. Do not promote the new model into normal replay based on one short throughput pass. The next bounded check is a fixed-size sequential continuation loop, not a broad USB driver rewrite.
## Alignment gate and continued scope
The PCIe link remained down after the90s hang. A bounded offroad recovery using the installed controller's `set_pcie_power` protocol restored LTSSM0x78. No firmware was flashed, and the bench CPU power snapshot was restored. First alignment launch failed on link-down; the recovered launch is separate evidence.
Explicit32-token masked alignment reduced60s warm forward from8.78s to5.80s. Relative RMSE against official FP32 is0.003766 and cosine0.999992, but peak-relative error0.012502 failed the preselected1% maximum-error gate. The largest discrepancy is at frame1300 (not the padded tail), but no exact cause is established. **No aligned60s generation pass is claimed; it was not promoted into the prepared composer.** The shorter demonstrated configurations are sufficient to continue the bounded fixed-duration loop test without a major optimization project.
## Gesture listening derivatives
V2 introduces unpitched electronic click/tick/body/brush/riser/sweep voices and distinct call/answer timbres. There is no validated current-chord estimator, so the added voices do not synthesize guessed pitches. Turn signal has an activation phrase, lighter once-per-bar sustained pattern, and a finite release. The unchanged original bank is preserved.
Controlled60s A/B: zero late events; signal phrases26→7. Full archived571.6s dry-road remix: zero late events; signal phrases222→59, all started gesture events282→130; same54,873,600 channel samples with5 limiter-level samples in both versions. This uses each archived semantic state only at its original elapsed time, and is explicitly an offline listening derivative, not a new native/Mac replay or fresh ACE-generation acceptance run. Human salience/annoyance review is still required.
## INT8 storage decision
The native per-row INT8 round trip matches the equivalent quantized official forward (RMSE0.004861, peak-relative0.007073). It generated15s latents in5.991s, compared with5.872–5.913s for FP16. Load36.93s versus~32s; resident weights remain3.151GB by design. Original-weight drift is~4.14% relative RMSE against FP32. Thus the smaller stored payload provides neither a measured runtime nor resident-memory benefit here. **Keep FP16; do not promote INT8 or pursue lower bits just to claim quantization.** Its audio/metadata are preserved as an exploratory comparison.
## Sequential prepared backend passed
Condition-padding validation against official FP32: RMSE0.004092, peak-relative0.006095. A fresh native verse30s followed by four actual12s-prefix/28s-new repaint sections completed on one resident DiT/VAE instance. Every preserved prefix is bit-exact outside the12-frame blend. New sections were prechorus, chorus, bridge, and outro; concatenated listening duration142s. Warm complete call walls: chorus20.970s, bridge20.661s, outro20.364s per28s new audio (RTF0.749/0.738/0.727, including call overhead). Peak tracked allocation4,214,994,090bytes; host471MiB. First30s shape preparation was285s inside generate, and first40s shape124s; weight loading another38.74s. These cold costs must happen before replay playback.
The same resident decoder also recovered the prior90s latents to PCM in14.344s, confirming those latents were usable. It does not restore the missing original90s generation timing. The stable fixed-shape run supports a prepared/prebuffered experimental architecture; earlier link failures still preclude production/coexistence claims.
## Opt-in replay integration under test
`--composer ace` selects the prepared ACE worker; no flag preserves SA3. The renderer accepts explicit PCM append boundaries instead of reusing SA3 latent-rate constants. ACE initializes a58s generated buffer, then requests28s-new continuations when the existing buffer reaches30s. Current navigation chooses existing section intents; immediate events use V2 gestures. A weak tonal estimate uses source-tail release instead of guessing a cadence chord. Archive provenance identifies the actual backend. Existing route resolver, cereal bridge, PCM transport, clock guards, and stored-score mechanics remain in place.
68 unit tests pass including legacy/default selection, PCM boundaries, invalid-rate rejection, and weak-tonality behavior. Real Mac/native replay tests are next; no live acceptance result is inferred from unit tests or the synthetic section-flow run.
## Replay integration evidence (current pass)
The first Mac ACE replay used the existing normal camera/UI and cached community route for180s. It completed five fresh continuations in21.29–23.29s each (RTF0.760–0.832 per28s new audio), with minimum composer buffer6.6s, zero fallback loops, zero renderer underflows, and no request timestamp beyond its causal cutoff. Camera/path/lane drawing, navigation and seven curve activations were observed. However host delivery had46 starved callbacks and21,257 discarded late frames. **This is not a clean Mac audio acceptance pass.** Delays also occurred before any generation request; observed transport latency reached344ms against the unchanged250ms presentation buffer. No output timing tolerance was relaxed. Original capture and failed audit remain in`results/normal_1789585496`.
Mac stored-score regression independently passed120s with exact contiguous samples, no output flags, and0.604ms maximum clock-alignment error. This confirms the stored path; it does not repair the failed fresh transport capture.
Resident ACE preparation took447.8s from supervisor start to ready, including generating the initial58s and warming both generation shapes. This is a real pre-playback cost. During native replay the composer used~491MiB RSS (plus two~88MiB compiler helpers); the audio app~288MiB, replay~217MiB and UI~144MiB. System available RAM at that sample was815MiB without swap. These are a snapshot and process peaks where recorded, not a guarantee of memory safety with modeld.
The pre-existing resident supervisor initially reported a start timeout because its process detector only recognized SA3, while the ACE worker continued preparing successfully. The detector now selects the requested backend and checks both known worker owners. Existing readiness/GPU ownership guards prevent starting another model over an occupied accelerator. The first timeout is preserved rather than counted as a clean cold-launch result.
Native comma ACE regression completed the entire arrival route:254.3s captured, seven accepted fresh sections at21.57–21.81s per28s new (RTF0.770–0.779), minimum buffer8.1s, zero fallback loops/underflows/output flags, no generation errors, and no input timestamp beyond request cutoff. One already-generated closing request was correctly discarded after navigation revision. Normal camera received5,098 frames; paths/lanes rendered; ten curve activations and navigation were present. The normal default backend was not changed.
The causal arrival run played41.2s of generated outro before the final cadence trigger at239.15s. Weak tonal confidence0.0155 selected the source-tail release instead of a guessed chord. Final cadence ended at246.31s. This establishes scheduling and source provenance; whether the ending is convincing remains a human listening decision. Full final PCM and arrival excerpt are stored under the owning route's archive. No route assets were redownloaded or converted into future-aware runtime inputs.
## What this establishes, and what it does not
| Question | Evidence / remaining limit |
|---|---|
| Meaningful ACE execution and music on Chestnut? | Yes: trained FP16 DiT and native VAE, numerical reference checks, actual WAVs and a full native replay. |
| Practical precision/residency? | FP16 fits with bounded decoding. Keep DiT/VAE resident; prepared embeddings replace live planner/encoder residency. INT8 storage was not beneficial. |
| Continuous throughput? |28s new in~21.7s during native replay gives~6.3s per-cycle margin. Classify as demonstrated Level3 prebuffer viability, with Level4 throughput evidence on this bounded run; prior hangs prevent a sustained-reliability or Level5 claim. |
| Permanent Mac dependence? | The prepared-identity composer loop runs entirely on comma/Chestnut. New arbitrary-prompt preparation remains unported; do not call the whole product standalone yet. |
| Seamless transitions and convincing outro? | Actual latent-prefix continuation and sequential native flow are preserved, and early outro playback is verified. Musical acceptance still needs a human. |
| V2 signal/curve/nav gestures? | New timbres and finite signal activation/sustain/release patterns are integrated. Source-derived/unpitched voices avoid guessed harmony. Salience and annoyance remain listening questions. |
| In-rhythm / in-key guarantee? | No validated current-chord tracker. Gesture grid is estimated at initial preparation; tempo/phase tracking across evolving ACE sections is not yet proven. Strong harmonic claims are withheld. |
| Curve structural preparation + local payoff? | Nav chooses longer-horizon section intents; causal model forecasts drive local fill/apex gestures. There is no validated guarantee that a chorus lands exactly on every curve. A~6s curve forecast alone is too late for a~22s composer plus queued audio. |
| Lane-change gesture? | A listening example exists; automatic lane-change event wiring is not implemented. |
| SA3 hybrid/fallback? | SA3 remains a selectable independent backend. Seamless ACE→SA3 emergency handoff would require a validated codec/conditioning bridge and is not implemented or needed by the successful native buffer run. |
| modeld coexistence? | Untested. Peak tracked ACE allocation~4.215GB leaves nominal~4.325GB of8.540GB accelerator capacity, not guaranteed usable headroom. No GPU scheduling/preemption or modeld deadlines were evaluated. |
Current Mac preparation timings (separate from Chestnut):30s target22.20s total including4.38s planner;45s21.64s/6.19s planner;60s28.74s/7.58s planner;90s27.75s/11.21s planner. Initialization~34s is separate. Remainder includes encoding/conditioning and preparation overhead, not a clean isolated encoder benchmark. These are not comma CPU measurements.
Runtime division: Mac performs development-time planner/text/reference preparation; comma CPU handles request policy, seeds, small sampler bookkeeping, latent-prefix selection, gestures and PCM/archive; Chestnut executes all heavy DiT denoising and VAE waveform decoding. No remote/cloud generation is used.
## Final preservation checks
Native stored ACE replay passed120s: contiguous sample digest verified, zero output flags,8.104ms maximum clock error, no generation invoked by stored playback. A separately owned SA3 worker was preparing in parallel; the stored path itself did not request it.
Fresh default SA3 on the Mac passed the same community route for120s (114.5s captured): three accepted generations at20.325–20.392s per26.006s new, RTF0.782–0.784; no fallback/underflow/starvation/late frames/output flags; maximum host alignment error0.524ms; six curve activations and camera/path/lane drawing verified. This demonstrates the fallback still works after the adapter changes. It does **not** establish a root cause or a fix for the earlier ACE Mac transport failure. That requires a separate reproducible timing investigation before promoting Mac ACE replay as dependable.
All route launches used their normal IDs/resolver and causal cereal path. Existing arbitrary-route and archival tests remain intact; no new route-specific music rules, timestamps, or allowlist entries were introduced. Native full-route ACE final PCM is owned by its route library directory; listening derivatives are also route-owned. The prior unseen-route first-attempt evidence was not overwritten.
Cleanup verified: both owned resident services stopped; no composer/audio worker left running; normal comma UI restored; real`IsOnroad=False`; CPU settings exactly match the saved pre-worker snapshot. Both session mute locks remain. Mac output volume0/muted; its default output is built-in speakers, not Bluetooth. No speaker output or Bluetooth audio was intentionally enabled. All model artifacts and route scores stay private/local. Public StarPilot has no changes from this work.
Validation:68 integration/pure unit tests,5 V2 gesture tests, targeted worker-selection tests and the real-route future-mutation causality check passed. Listening-page local links resolve and no audio element autoplays. Subjective physical listening remains explicitly deferred.
-77
View File
@@ -1,77 +0,0 @@
# Prism demo hardening
Prism is the primary ACE identity; Aurora is the selectable backup; Circuit is archived and excluded from active selection. The native comma path is the recommended demo path. The quality gate, bounded retries and accepted-music holds have passed their focused hardware tests. **The intermittent GPU/link fault is not fixed, and this is not a production-reliability certification.** Prism/Aurora route checks, SA3, Mac/native stored playback and final shutdown checks all passed.
All work is private under Desktop/RoadScore and `/data/roadscore`. Baseline `08ba22e` is preserved as `ace-demo-baseline-08ba22e`. Public StarPilot and MusicGen are untouched.
## Measured results
| Check | Result |
|---|---|
| Native Prism soak | 30 accepted continuations; 840 seconds of new music in 694.97 seconds, **0.827 RTF**; no retries |
| Soak continuity | All 30 latent prefixes exact outside the 12-frame blend; longest quiet span 0.6 seconds in the 868-second joined render |
| Final-policy full curve replay | **0.888 RTF** including qualification/persistence; 363 seconds of new music in 322.19 seconds; 13 jobs |
| Final-policy memory | 5,789,487,104 tracked Chestnut bytes including allocator cache; worker host high-water mark 761 MiB |
| Cold Prism preparation | 511.9 seconds; 112 seconds of accepted starting music; four jobs, no retries |
| Arrival route | 254.1 seconds captured; seven jobs; one successful reroll; minimum buffer 42.9 seconds |
| Community route | 570.3 seconds; 19 jobs; one successful reroll; minimum buffer 41.3 seconds |
| Corrected curve route | 402.9 seconds; 13 jobs; minimum buffer 64.8 seconds; **no two-second quiet spans** |
| Normal native replay | All three Prism routes reached EOF with camera/path/lane/navigation data and causal delivery verified; zero underflows, emergency fallbacks or output flags |
| Focused section flow | 204 seconds, six fresh transitions, no retries; longest quiet span 0.3 seconds |
| Composer-stop containment | 176.7 seconds; four accepted-music holds; zero underflows/output flags; no two-second quiet spans; DEGRADED reported |
| Mac fresh retest | 176.4 seconds; zero missed frames/starvation/output flags; maximum alignment error 0.923 ms |
| Mac stored playback | 120.01 seconds; exact contiguous samples; no generation; zero flags; maximum alignment error 1.125 ms |
| Aurora full route | 254.2 seconds; seven jobs, no retries/holds/output errors; minimum buffer 64.6 seconds; profile archive verified |
| SA3 regression | 116.9 seconds, four fresh jobs at 0.758–0.777 RTF; no underflows/fallbacks/output flags; CPU restoration verified |
| Native stored regression | 120.02 seconds of the Aurora archive; exact contiguous samples, no generation, zero flags; maximum alignment error 8.708 ms |
The 30-section soak used the first gate revision. Every saved section was rechecked with the final playback-scale policy and remained accepted. The final policy was then measured in real replay, including the full curve run. The soak's conservative 90-second buffer audit is **post-run modeling**, not live playback evidence. Its standalone 1,059 MiB host peak includes accumulating the whole 14-minute WAV; this is distinct from the resident-worker measurement above.
The final curve run spent 192.80 seconds in sampling and 104.17 seconds in decode. DiT and VAE execute on Chestnut; quality inspection, arrangement, gestures, playback and file handling execute on CPU. Prepared profile embeddings are supplied from the existing Mac preparation workflow. No Mac inference or cloud generation participates in native replay.
## Pre-commit quality and continuity
Only newly committed material is assessed; retained context and discarded lookahead cannot trigger rejection. Version `prism-precommit-v2` previews the existing neutral output DSP at the fixed 0.65 output gain, without route/event inputs. It measures DC-removed 100 ms RMS windows, using conservative relative thresholds and absolute bounds. Non-outro material is rejected for a two-second near-silent interval, with one window of alignment allowance, or a four-second severe collapse. Sparse and uniformly soft material supplies its own reference scale.
Intentional outro fades use a separate policy. Terminal fade and sparse final phrases are allowed; non-finite/broken output, entirely silent output and absurdly long leading silence are rejected. The arrival test rejected 7.8 seconds of leading outro silence and accepted its reroll before playback.
The original attempt plus at most two deterministic rerolls preserves the same source and role. Each attempt requires the measured estimate, at least 30 seconds, plus 10 seconds of reserve before its playback deadline. Rejected WAVs, latents, seeds, spans, reasons, wall times and remaining buffer are retained in the route-owned quality archive.
Known-case results:
- Community chorus: all three attempts failed, with 5.8/5.2/7.1-second raw-policy gaps, taking 70.26 seconds. No successful reroll is claimed.
- Curve verse: original 2.7-second gap rejected; first reroll accepted with a 0.1-second gap; 46.37 seconds total.
- New playback-scale counterexample: original rejected; reroll accepted in 48.04 seconds total with 42.4 seconds of the test reserve remaining. Official Mac reference reproduces both decisions and the measured 1.9/1.6-second preview spans.
The 11 fixtures include recorded good/bad outputs and explicitly labeled soft/rest transforms. The 20 focused tests pass. Existing regression coverage also passed, including the causal future-mutation test in the proper openpilot host runtime.
The section grammar remains verse → prechorus → chorus → verse → bridge → chorus, with current delivered navigation controlling the outro. Prism's bridge keeps the same tonal center and contrasts arrangement/timbre. Three additional reference bridge variants are available for listening; the previously selected conservative conditioning remains active. No new harmony experiment was promoted without human review.
Curve Apex V4 adds an unpitched crash, low transient and stereo flutter. V3 turn waveforms remain unchanged. Controlled V4 rendering had no late events, peak 0.718 and essentially unchanged overall RMS relative to V3. **Musical salience and subjective continuity still require listening.**
## Failures preserved, fixes and limits
The first curve archive passed runtime counters but contained a two-second final-output quiet passage. Raw decoder thresholds missed it. The final gate measures through neutral playback processing and accounts for window alignment. The original archive remains labeled `curve_v1`; the corrected full curve archive contains no two-second quiet spans. No route-specific timestamp or musical parameter was added.
The first injected worker stop held music but caused four output underflows: each immediately followed construction of a hold while holding the callback lock. Buffer construction now occurs outside that lock, as for normal continuations. The repeated test completed four holds with no output errors after CPU restoration. Holds are labeled reused accepted music, not fresh generation. No seamless ACE→SA3 handoff or in-playback reset is implemented. This controlled stop test does not prove every possible USB failure will be harmless to the host.
The first Mac fresh run had a 1.138-second packet-delay burst exceeding its 250 ms reserve, producing 151 starved callbacks and 71,088 skipped samples despite healthy composition. A quiet-host retest passed with no transport-code or latency change. The initiating network/host-scheduling cause is unresolved; this is why native playback is preferred for the demo.
The smaller GPU reproducer repeats one fixed 375-frame VAE graph with resident input, no DiT, replay or full-file assembly. Back-to-back readback failed after 22 successes; compute-only failed after two successes, before PCM readback. A one-second-idle comparison passed 40 iterations with identical PCM hashes. Both genuine failures showed a healthy link after submission, then an inaccessible GPU and link register `0x0` **before the driver timeout handler**, while the USB bridge remained readable. Corrupted timeline reads are not counted as progress. This narrows the interval but does not identify root cause, and diagnostic sampling can alter timing.
Two separate cold benchmark attempts omitted `TC_OPT=2`; their upload-drain failures are preserved as invalid harness configurations, not production ACE regressions. Correctly configured real generation and replay passed as reported above. The private worker retains fixed initial-30/continuation-45 shapes, pre-job link checks, a resident model and a timeout snapshot that avoids the driver's interrupt/reset storm. There is no silent power cycling. See [GPU findings](results/ace_demo_20260916/GPU_FINDINGS.md).
Aurora cold preparation exposed another operational issue: normal screen-off power saving offlined the large CPU cores after the supervisor configured them. This first backup preparation took 846.5s and required an explicit CPU-only correction. Repository code ties that action to ignition-off/screen-off power saving, separately from thermal state. The private supervisor now maintains only its CPU-online settings while the real device remains offroad, without calling amplifier/power-save APIs or reasserting clock limits over later thermal decisions. It restores its saved settings on exit. The helper was exercised against the actual offlined cores; three guard tests cover no writes onroad and stopping writes on an onroad transition. The SA3 run exercised a real screen-off transition: the supervisor automatically restored all four CPU-online bits, completed replay without output errors and restored its saved settings on exit. The correction did not call amplifier APIs or reassert clock limits.
No live driving or simultaneous modeld coexistence was tested. Arbitrary new identity preparation on the comma is not demonstrated. Repeated bad samples can still force explicitly reported musical holds; the measurements do not establish indefinite fresh composition under arbitrary failures.
## Review and operation
- [Focused nine-section listening page](results/ace_demo_20260916/index.html)
- [Launch and recovery checklist](results/ace_demo_20260916/DEMO_CHECKLIST.md)
- [Route evidence](results/ace_demo_20260916/replay_results.json)
- [Final-policy performance](results/ace_demo_20260916/v2_performance.json)
- [Exact source/seed reference comparison](results/ace_demo_20260916/gain_reference_audit.json)
- [Corrected containment audit](results/ace_demo_20260916/containment_audit.json)
[Final validation](results/ace_demo_20260916/validation.json) and [cleanup evidence](results/ace_demo_20260916/cleanup.json) are saved. All owned workers/replays are stopped, GPU/session/display/service locks are free, the normal offroad UI and manager are restored, and saved CPU settings match their restoration snapshot. Both session mute locks remain active and Mac system mute is on. No speaker or Bluetooth output was intentionally enabled; the bench remained muted. Physical speaker confirmation is deferred to the user.
-85
View File
@@ -1,85 +0,0 @@
# ACE port stability — September 16 investigation
**The fixed-window native path is faster than playback and has passed substantial repeatability/replay tests. Overall acceptance remains blocked by model-generated interior silence and an unresolved GPU/link failure.** Do not call this production-stable or claim every transition is seamless.
[Listening review](results/ace_stability_20260916/index.html) · [Detailed evidence](results/ace_stability_20260916/INVESTIGATION.md)
All work is private under Desktop/RoadScore or the offroad bench. Baseline tag: `ace-stability-baseline-b719139`. No new models, training, public StarPilot changes, or physical audio output.
## What matches the reference
The deterministic harness exports the actual conditioning, explicit noise, source prefix/mask, eight diffusion steps, final latents and decoded PCM. Official Mac FP32 and native Chestnut FP16 consume the same rounded boundary values. Three new identities—Prism, Aurora and Circuit—have paired 30/45/60s outputs. Final latent relative RMS differences range approximately 0.45–2.75%; the page preserves localized differences for human listening rather than labeling them inaudible.
Five identical native Prism30 runs have identical latent and PCM hashes, with RTF 0.5295–0.5320. The first thirteen jobs of separate continuation processes are also identical until adaptive endpoint choices make their subsequent inputs diverge.
Same-latent VAE comparisons at 30/45/60s and the matched transition have 0.070–0.088% waveform RMS error, envelope correlation above 0.9999999 and zero measured sample lag in every one-second window. The tested decoder does not introduce the suspected rhythm displacement.
Teacher-forced Circuit30 forwards isolate native velocity error at 0.877% initially, falling to 0.231–0.269% in later steps. Capturing all 24 layers at steps 0 and 7 found gradual small FP16 differences, without an abrupt order-of-magnitude break at sliding/full-attention boundaries. End-to-end diffusion amplifies those differences. Mixed FP32 residual accumulation improves two examples but worsens a third; it was not promoted. Artificial conditioning padding is masked; actual prepared tokens are all valid.
Repaint semantics match official behavior: 25 latent frames/s, 1920 PCM samples/frame, mask true means generate, a 12-frame boundary blend, and exact source preservation outside the blend. Source prefixes are taken from the actual committed previous latent tail.
The human-approved old Mac transition cannot be reproduced exactly: original random reference crops/VAE sampling state were not saved, and the single-item planner path did not obey the supplied seed. Original planner codes were recovered. The review supplies both a closer recovered-code reconstruction and a new original-settings pair with fully captured identical inputs. Neither is falsely labeled the identical old performance.
## Silence: two mechanisms, one still unresolved
The old long terminal silences appear in the official reference before decoding or concatenation. The planner predicted endings, and copying those silent tails carried them into subsequent prefixes. No-fade text and longer declared duration alone did not fix this.
The private ACE candidate now generates a 45s lookahead window, preserves 8s, and normally commits through 36s: 28s of new material. A generated-energy endpoint check trims a premature terminal fade before selecting the next source tail. Intentional outros may fade. This repaired the tested initial section flow and completed a 20-job adaptive soak.
**This is not a general no-silence guarantee.** Full-route audio later exposed a 5.1s community chorus gap and a 2.7s curve-run gap. The exact community input reproduces the gap on official Mac, with committed latent RMS difference 1.070%. Its preceding source tail remains active. The curve verse gap is also reproduced on official Mac (committed latent RMS difference0.770%); this is not confined to chorus prompts. Four alternate seeds still produce 2.2–7.1s quiet; changing the role to verse also fails. Stronger/weaker repaint injection does not remove it. A 16s prefix fixes that one example but leaves too little new music to sustain real-time output. A 4s prefix fails with a nine-second gap on another seed. Refreshing the timbre reference from the actual preceding audio changed exactly one conditioning token but also failed across three seeds. No such workaround was promoted.
The remaining demonstrated blocker is model/conditioning-dependent leading/interior silence, not transport starvation, final concatenation, an inherited silent prefix, or a native-only VAE error. The endpoint guard cannot detect it. Full musical acceptance remains failed; rerolls and crossfades are not presented as a repair.
## Speed and bounded runtime
| Generated duration | Warm sampling + decode | RTF |
|---|---:|---:|
| 15s | 9.16s | 0.611 |
| 20s | 10.53s | 0.527 |
| 24s | 32.01s | 1.334 |
| 28s | 16.71s | 0.597 |
| 30s | 15.96s | 0.532 |
| 32s | 39.15s | 1.223 |
| 36s | 49.26s | 1.368 |
| 40s | 20.78s | 0.520 |
| 45s | 22.68s | 0.504 |
| 48s | 35.98s | 0.750 |
| 52s | 57.41s | 1.104 |
| 56s | decode failed | unavailable |
| 60s | 79.93s | 1.332 |
These are full-duration RTFs; continuation must divide by retained **new** music. The adaptive 20-job chain retained 555s in 454.17s sampling/decode (RTF 0.818). Worst warm full-job RTF was 0.964. Two endpoints trimmed to 32/35s. Host high-water was 478.6MiB; sampled physical GPU allocation including allocator cache reached 5.791GB. First fixed-window soak also completed 20 continuations without a hang.
Cold resident startup took roughly 758s to prepare a 56s buffer. First model job wall time was 538.37s, second 177.22s, plus loading/startup. Warm RTF does not imply fast launch: prepare the resident worker in advance.
Duration performance is non-monotonic and depends on sequence/tile shapes. A padded 60s experiment improved throughput, but failed the earlier single-forward 1% peak-error criterion. This pass's accumulated final-latent metrics are different measurements, not a retroactive pass. Alignment changes were not promoted.
## GPU failure remains a blocker
Three preserved 56s decode failures required explicit offroad link recovery. The first followed a duration sweep. An isolated VAE-only repeat then reproduced failure on its third decode, after two bit-identical successful decodes, using only 2.470GB physical allocation and explicit transfer/compute fences. No DiT or shape changes were needed. The sixth fixed decoder chunk finished upload, then waited unsuccessfully for compute timeline 383 (observed zero).
After terminating that failed process, USB bridge config remained readable, link register B450 was 0x58 rather than healthy 0x78, and GPU config reads returned Unsupported Request. Existing power-cycle recovery restored 0x78. This establishes link failure after the incident; it does not prove which component initiated it. The final VAE-only probe completed four identical decodes (cold177.63s, warm9.21–9.52s), then failed on repeat5/chunk4 after upload: timeline473, observed0. The diagnostic read B450=0x0 after the driver hang handler but before decoder/process teardown; after terminating the owned process it was0x58, with Unsupported Request from GPU config. Recovery restored0x78. This failure occurred at a different repeat/chunk than the earlier probe, confirming intermittency rather than a deterministic third-repeat boundary. It used2.470GB allocation and385.4MiB host high-water before failure. Fences alone are not a fix. No firmware work or silent auto-recovery was added.
## Replay evidence
All three complete native runs reached EOF with normal camera/path/lane UI, navigation, causal cereal and fresh Chestnut generation. None had an underflow, fallback, output flag, future-input violation or GPU hang.
| Route fixture | Captured | Accepted jobs | Minimum buffer |
|---|---:|---:|---:|
| Arrival | 253.9s | 7 | 5.4s |
| Community | 572.4s | 19 | 5.9s |
| Curve | 403.2s | 13 | 3.9s |
The first curve attempt was partial and remains separately preserved. Full-route transport passes do not override the musical silence failures above. Archived route-owned scores remain private.
Native stored playback of the new arrival archive passed 120s with no generation, zero output flags, contiguous samples and maximum measured clock alignment 10.67ms. Mac stored playback passed 90s with zero flags and contiguous samples: 0.438ms maximum alignment on the earlier normal-UI run, 2.5ms on the new archive's explicit headless run. A new normal-UI Mac launch failed because its display was asleep; that failure is retained, not disguised as a UI pass.
Mac fresh ACE transport passed a180s headless launch (176.5s source packets, six fresh jobs): zero late frames/starved callbacks/output flags,0.34ms maximum presentation error. All1,765packets were logged before drops; max delivery154ms, p99≈103ms. During-generation delivery p99≈103.3ms versus103.5ms outside generation gives no evidence that this run’s ACE calls caused the earlier transport failure. The earlier344ms delivery spike remains preserved; normal Mac UI transport under load is not cleared by a headless pass. The250ms buffer and acceptance thresholds are unchanged. Fresh native SA3 fallback normal_1789600873 also passed a120s launch:116.6s captured, three fresh jobs atRTF0.776–0.780, minimum buffer9.326s, zero underflow/fallback/output flag/future-input violation, normal camera/path/lane UI verified. Cold readiness180.09s.
## Profiles, gestures and limits
Three prepared profiles have initial/verse/prechorus/chorus/bridge/outro tensors, 6.1–6.7MB each plus shared trained weights. Identity preparation remains Mac-assisted; runtime DiT, Euler sampling and VAE execute on Chestnut, with CPU noise generation, endpoint checks and arrangement. Aurora and Circuit each completed all six native initial/section jobs using actual prior outputs, with exact prefix preservation. Maximum new-region quiet was0.1s/0.2s respectively in these bounded chains; retained-new continuation RTF was0.799–0.889. Combined host high-water478.8MiB and sampled physical allocator occupancy5.792GB. Their official reference chains also pass the energy check. These are prepared demo candidates, not human-approved profiles or full-route-tested substitutes for Prism.
Bridges request contrast through texture/instrumentation while preserving tonal center, without forced modulation. V3 adds a brighter rim/shaker turn motif and layered unpitched curve impact. Twelve gesture tests passed; V2 remains selectable. Human salience and musical fidelity require listening. The private unit suite passed 69 checks during this pass.
SA3 remains the default fallback path; MusicGen is untouched. Set `ROADSCORE_ACE_WINDOWED=0` for the prior ACE wrapper. Session mute locks remain active. No speakers or Bluetooth output have been enabled. Final checks confirm no owned workers/replays remain; GPU/session/display locks are free, the normal UI and manager are running, and the bench is offroad. The supervisor’s before/after CPU snapshots match. Normal offroad power management subsequently offlined the big cores; no further override was applied. Mac system mute is true, default output remains built-in (muted), no Bluetooth audio output is listed, and both session mute locks remain. No speaker or Bluetooth output was intentionally enabled; bench tests remained muted.
-75
View File
@@ -1,75 +0,0 @@
# Composition strategy pass — from a7dc34f / overnight-review
Human rejected the prior macro-form, splices and arrival preparation. Preserve engineering foundation; do not protect SA3 out of sunk cost. All physical audio stays muted; session locks retained. Private Desktop/RoadScore work only. No fine-tuning, invalid graph snapshots, public edits, or speculative Chestnut ports.
Priorities: one aggressive instrumental K-pop/game-score SA3 control; screen 3–5 alternatives cheaply; run ACE-Step1.5 locally if feasible and at most one justified additional candidate; generated transitions and actual outro; deterministic tempo-aware gesture family and causal turn-signal/curve/nav integration; regress Mac/native/stored replay; one listening page and precise evidence.
Development hardware: Mac M1 Max32GB,32GPU cores;~43GiB disk free at start. Alternative downloads/installations stay isolated and bounded. A musical winner requires human listening; claims in model documentation are not our measured results.
## Checkpoint during execution
- Fresh SA3 control: nine trained-weight outputs; five deliberately extreme roles share one fresh identity. Three transitions are generated from existing context with no end anchor. All remain human-listening candidates.
- ACE-Step1.5 turbo + MLX planner/DiT/VAE successfully generated a 90s structured piece, five reference-conditioned roles and a 40s transition. First90s call108.03s (RTF1.20); not a Chestnut measurement. Initialization248.42s includes model acquisition. Peak process RSS8.10GiB understates total unified-memory pressure; system swap reached17.38GiB. No musical winner declared.
- YuE2 is the single additional executed candidate, using official MPS support with native symbolic planning. Bounded to90s semantic material and20min runtime. No Chestnut port attempted.
- Native arrival run `normal_1789575069`:253.9s, eight fresh jobs, zero fallbacks/underflows. First outro-conditioned material203.964s; cadence trigger239.2s. Navigation clearing reset the initial outro counter, despite35.236s of archived outro-conditioned playback. Original evidence retained; accounting fix now separates delivered musical history from navigation intent. Concluding musical behavior is still unverified.
- New source-derived percussion layer: finite turn-signal/curve/nav/arrival phrases, estimated pulse phase, sample-addressed scheduling, cancellation recorded for stored replay, source-dependent level/headroom. Synthetic turn-signal example115ms onset, zero scheduling lateness. Estimated beat/bar/key values are not ground truth.
-61 tests pass including real-log future-mutation causality regression. Initial test invocations had environment import failures; corrected test environment uses native runtime plus analysis libraries.
Physical output remains muted throughout. Alternative models are isolated under experiments; resident Chestnut worker stays exclusive and unchanged. No MusicGen edits.
## Model screen and porting decision
The musical winner is **not selected**. Listen to the new [single review page](results/composition_20260916/index.html). Runtime/role labels cannot establish that a chorus is convincing or that transitions sound natural. No candidate has earned Tier3 on human musical evidence; no speculative Chestnut port was started.
| Candidate | Interface and evidence | Execution / decision |
|---|---|---|
| SA3 Small-Music control | Native tinygrad DiT + decoder, cached CPU text conditions, audio-prefix inpainting. One fresh identity, five extreme roles and three explicitly generated transitions. | Demonstrated Chestnut runtime retained. Control roles:28.05s unique audio in about19.3s. Current live continuations:26.006s unique in about19.9–20.3s,RTF0.77–0.78. Clear macro-form remains a listening question. |
| ACE-Step1.5 turbo | Official hybrid planner + flow/diffusion model; instrumental tags, tempo/key metadata, reference audio; upstream repaint/cover interfaces. We tested full-form generation and reference-conditioned roles, not live repaint. | Seven actual local Mac outputs.90s/108.03s,RTF1.20;30s roles take63.15–91.40s,RTF2.11–3.05, including reference encoding;40s transition73.33s,RTF1.83. Preserve for human evaluation as a possible longer-horizon composer. |
| YuE2-3B | Native symbolic ABC melody/chord planning, autoregressive semantic audio tokens, non-autoregressive acoustic flow synthesis, VAE. Instrumental request through style/structure text, not a verified dedicated no-vocal mode. | Actual MPS end-to-end output81.279s/374.55s,RTF4.61. Neither score nor semantic stream hit our token cap. However37.7s of100ms windows are below−50dBFS, much of the latter half; the symbolic plan choseF minor despiteD minor conditioning. Keep original evidence; no further investment this pass. |
| MAGNeT | Documented masked-token generation,300M/1.5B family,10/30s clips, EnCodec decode. No documented full-song structural interface in the inspected guide. | Cheap screen only. Does not present an obvious solution to this pass's missing macro-form; no model downloaded/executed. This is an interface-based screen, not a listening rejection. |
| LeVo2 / SongGeneration | Full-song model claims surfaced in search, but the official repository/raw source returned unavailable during this run. | Cheap screen blocked on source availability; exact failure saved in`levo_access.txt`. Do not infer model quality or claim an executed benchmark. |
Primary sources: [ACE-Step1.5 official implementation](https://github.com/ACE-Step/ACE-Step-1.5), [YuE official implementation](https://github.com/multimodal-art-projection/YuE), [MAGNeT official guide](https://github.com/facebookresearch/audiocraft/blob/main/docs/MAGNET.md), [SongGeneration official repository checked](https://github.com/tencent-ailab/SongGeneration). Local model revisions, sizes and YuE integrity hashes are retained in the experiment/results. No vendor source was patched; only isolated research runners were added.
### Why not port the alternatives yet?
ACE's inspected turbo configuration has24 decoder layers,2048 hidden width,16 query/8 KV heads, alternating sliding/full attention, rotary embeddings, RMS normalization, gated feedforwards, and extra lyric/timbre conditioning. Its local packages occupy4.463GiB for the core,3.498GiB for the1.7B planner,1.125GiB for the text embedding model and0.314GiB for the VAE. These are on-disk footprints, not peak allocations. The narrow matrix/attention/convolution primitives are familiar, but this is a much larger conditioning/decoder stack than the SA3 port. Complete simultaneous residency is not established on the8GB Chestnut; host3.5GB makes staging/offload difficult. Quantization, serial model lifetimes or a separate local planner would need engineering and measurement. A2B core does not automatically imply an impossible port, nor does familiar attention imply real-time performance.
YuE's inspected backbone has28 layers,2048 hidden width,16 query/8 KV heads,6144 feedforward width and184704 vocabulary. Audio uses64-channel latents at25Hz and a convolutional VAE. Its official MPS path actually ran; a general PyTorch compatibility layer was unnecessary. Nevertheless it combines sequential planning/semantic generation with32 acoustic flow steps. The unquantized backbone file is7,261,441,640 bytes plus530,512,720-byte VAE, before activations/KV cache. Mac MPS driver allocation at the end was18,031,886,336 bytes (~16.79GiB), **not a measured peak**. The configured16GiB budget does not enforce a hard MPS cap in this implementation. It is not a demonstrated viable8GB live backend.
ACE peak process RSS was8,291.9MiB, while unified allocations plus other system use caused substantial swap (17.38GiB observed). YuE RSS was1,275.6MiB despite the large MPS driver allocation: RSS alone is particularly misleading here. Alternative runs were serialized, never run concurrently with each other. Their model operations executed on Mac GPU via MLX/MPS; Python/tokenization, file handling and some conditioning/offload work execute on CPU. ACE uses its upstream mixed MLX/PyTorch path. YuE's operator fallback environment was enabled, so this is not proof every operator ran on GPU. Neither is a Chestnut benchmark.
SA3 worker process peak RSS ~175–178MiB and tracked post-decode allocation1,111.85MiB are not whole-bench RAM or verified peak VRAM. Its DiT and decoder execute on Chestnut; CPU handles already-cached text conditioning, noise, replay, resampling, mixing and gestures. Text encoding was performed separately on CPU before the runs. Existing process-cold127–185s and resident15–24s launcher evidence remains historical; this pass's native curve resident launch14.62s is recorded. Alternative initialization timings are not apples-to-apples: ACE248.42s includes downloads; YuE7.42s resolves/checks local files, with lazy model loading included in the generation wall timer. No warm YuE repeat was run merely to improve a number.
## Minimum viable architecture recommendation, conditional on listening
Keep the model-independent route/replay → causal input → final PCM → host/archive pipeline. Split musical control by time horizon:
- **Immediate / next beat:** source-derived deterministic percussion gestures. Turn signal uses an eighth-note motif whose first entrance is quantized to the next sixteenth. Curve preparation uses a one-bar fill and a predicted-peak accent. Navigation uses a small phrase, with a cooldown. No new guessed chord/key is introduced.
- **Tens of seconds ahead:** cached text conditions and prefix-conditioned generated transitions. Current navigation can select a chorus/bridge transition or concluding material far enough ahead to survive generation and existing buffered audio. We do not pretend the neural model can create a new chorus six seconds before a curve.
- **Longer-form composition:** if ACE's actual examples win human listening, investigate its longer-form material as a local prebuffered composer or arrangement guide. SA3/Chestnut would still generate meaningful continuing material, and deterministic gestures would own precise timing. This hybrid is a candidate architecture, not a demonstrated coherent multi-model composition system.
The live prototype now uses continuous generated windows, not the previously rejected independent section bank. It still overlaps retained context at decoder boundaries; this cannot guarantee identical tempo, harmony or groove across windows. Three context-included transition outputs expose the underlying transition for listening rather than hiding it with seam tuning. The alternative runners own model-specific preparation and decoding;`backends.py` records their actual capabilities/results. They are not selectable live workers. The existing transport/score format remains final PCM, avoiding alternative-model imports in the replay process.
The earlier rock example's human impact cannot be re-proved while muted. The earlier dense-source report does document stable107–109BPM pulse, denser transient vocabulary and no premature silence in its continuous run. A plausible lesson is stronger attack contrast and headroom for event punctuation, not simply more distortion. The present source-derived hats/fills/crashes test that hypothesis; this is explicitly an inference awaiting listening.
### What the new controls do and do not establish
- A persistent identity is a source prefix and repeated musical target, not a verified recurring motif.
- A source-derived drum bank is stylistically related by material and approximate tempo. Filtered source attacks can still contain pitched leakage. We deliberately avoid claiming a new in-key bass/synth part from unreliable key estimates.
- Gesture start frames are exact in the captured sample stream. The source's beat/downbeat inference is uncertain and can drift across subsequent generated windows. Input polling and host output buffering add latency (Mac target buffer250ms); synthetic115ms scheduling is not an acoustic115ms measurement.
- Curve payoff now follows revisions to the currently delivered model forecast until it is within one beat. Already-started audio is never moved. Curvature forecast and physical steering onset are different quantities; no future recorded steering drives runtime.
- Outro conditioning begins well before the final cadence in the demonstrated arrival run. This is necessary evidence, not proof of musically conclusive phrasing. Original counter failure and the corrected history/intent separation are documented rather than rewriting old records.
- Physical speakers, subjective form, instrumental adherence, groove/key continuity and a convincing ending all remain human verification items. No fine-tuning, model port, or subjective acceptance was claimed.
## Replay validation and known capture failures
Native arrival253.9s:8 fresh jobs,0 fallbacks/renderer underflows/output flags;81 repeated turn-signal phrase starts,7 curve preparation/payoff pairs,3 navigation cues and1 arrival preparation. Native curve176.3s:6 fresh jobs,0 fallbacks/underflows/output flags;27 repeated signal phrase starts and4 curve pairs. These counts are phrases, **not distinct signal activations**. Both had zero sample scheduling lateness. Repeated signal phrases are intentionally queued up to one beat ahead; their queue wait is not the first-signal response latency.
A native video-recording attempt failed the existing1x replay-clock guard after20.8s audio. Retrying without recording passed. A Mac live recording attempt had a persistent~94ms reported DAC offset beginning at42.5s, despite zero PortAudio flags/starvation/recovery. It was stopped at159.4s rather than presented as a synchronized pass. Recording load is a plausible contributor, not a proven root cause. Both failures and original logs remain in[failures.json](results/composition_20260916/failures.json); no timing tolerance was weakened. A full no-recording Mac rerun is being validated, followed by separate stored-score capture.
Stored overlay now derives road phase/lead only from decisions whose archived sample time has been reached; cancelled future gestures stay cancelled. The normal camera/path/lane UI remains the existing replay UI. Capture screenshots now wait for a ready onroad display instead of preserving the preparation screen. No host PCM scheduling or jitter logic changed; its friendly style label was the only host-audio edit.
Final regression: full community route`normal_1789576104` completed571.6s,21 accepted fresh jobs, median unique-audioRTF0.7795, zero host starvation/late samples/output flags and max host alignment0.324ms. Live outro counter correctly recorded21.653s before cadence. Normal camera/path/lane drawing remained active. Mac stored capture`normal_1789576767` verified contiguous samples with zero output flags and0.583ms maximum clock error; VFR review uses10,978 recorded UI timestamps over186.46s including startup, with149ms worst frame gap. Native stored arrival`normal_1789576781` verified contiguous samples with zero flags and9.90ms maximum clock error. Neither stored run invoked generation. The review video is the archived fresh score, not a new generation presented as live.
Shutdown: owned resident worker stopped; power snapshots match exactly. Normal bench UI running, realIsOnroad=false. Mac system mute and both session locks remain; every test stream was muted. No Bluetooth output or speaker output was enabled. ALSA mixer queries were unavailable even through the permitted read attempt, so no hardware-mixer-register verification is claimed; native muted PCM and session-policy evidence are retained. Public StarPilot tree was not edited; MusicGen remains untouched.
-15
View File
@@ -1,15 +0,0 @@
# Private RoadScore control boundary
`prototype/settings.py` holds host session settings: enabled, style, generate/stored mode, mute, overlay, output device and profile. `settings.json` records the effective session configuration. The current musical worker still reads `runtime.json`; a future Galaxy adapter should validate and atomically update that same musical configuration rather than add a second set of musical defaults. Style switching must also stage matching conditioning/source material and readiness before committing it.
Product output is audible by default. Explicit `--muted`, headless operation, `ROADSCORE_FORCE_MUTE=1`, or the private `.session-muted` marker suppresses output. The session marker overrides direct calls to all three audio endpoints. **This overnight session leaves the marker in place on both hosts.** Remove it only when the human resumes audible testing. Bluetooth is not selected, connected, or configured by RoadScore. A future output adapter may own a user-selected device; it must preserve the host's sample clock, mute policy and output ownership.
The optional overlay runs inside the existing normal UI through a private wrapper. Readiness uses the existing native GPU loading/green assets and existing font; the small RoadScore card is custom. It does not change the driving model's GPU state or impersonate driving-model readiness. The status schema has no dependency on historical routes, so the same consumer can display future live readiness.
Generated sources and decisions are private. Demo score archives remain under the route parent, so deleting that parent removes its scores. Future real driving should rotate lossless audio chunks and decision metadata under each native realdata segment, with shared session identity and original monotonic timestamps. Segment pruning should remove that segment's audio naturally. Do not maintain a second undeletable global audio library; do not feed recorded future events into the music engine.
Experimental form playback is controlled by `song_form_experimental` in runtime.json. It is not an assertion that human musical acceptance has passed. Generated candidate banks provide low-latency section choices; fresh local GPU continuations replenish material at phrase boundaries. The predictor cannot wait for a ~19-second GPU job when an event is only ~5 seconds away. Metadata distinguishes requested section, scheduled landing, actual sample landing and which fresh generation job was heard. Beat and bar detection remain estimated, not musically verified.
For an explicitly prepared bench demo, `./roadscore-worker start` owns a resident normal worker, `./roadscore-worker status` reports readiness, and `./roadscore-worker stop` restores its CPU state. Normal `./onroad ROUTE --roadscore` reuses that worker. This is opt-in: ordinary launches still clean up workers they start. The service refuses to take over an externally owned worker and stops on real onroad state. It is not a solution for concurrent live driving inference. Overnight cleanup stops all workers; no resident service is intentionally left running.
Archive selection preserves the most complete demo score: a short regression no longer replaces a full-route recording as the default stored replay. `last_generated.json` identifies the newest experiment; `latest.json` identifies the preferred coverage. Every session remains available under the route-owned archive. Reaching EOF after starting mid-route is not marked as a complete route recording.
-43
View File
@@ -1,43 +0,0 @@
# Horizon continuity experiment
Human review rejected the previous Nocturne preference and earlier-tail trimming. That rejection is the baseline for this pass; the old zero-underrun/near-silence measurements did not establish a good song.
## What changed
The fixed 324-frame, eight-step SA3 worker now optionally inpaints **between two active pieces of musical context**. Four seconds of the preceding passage remain at the front. A four-second identity anchor, selected from the already-generated opening's active middle, remains at the back. The same anchor returns about every 26 seconds. It is explicitly reused music, and its ranges are shaded purple in the listening plots.
The model generates the roughly 21.92-second interior between those anchors. The runtime appends about 26.01 seconds per job, including the reused 4.09-second anchor. The retained prefix is removed except for a two-second common-context crossfade. No silence removal, time stretching, gain repair, stem separation, or hidden offline arrangement is applied. Even latent-frame boundaries preserve the decoder's two-frame grouping. MusicGen, native SA3 model weights and the public repository remain untouched.
The initial Horizon opening stops at 21.92 seconds, before its rejected terminal fade. This trimming alone is **not** the solution: a matched-seed prefix-only control still fades near the end, whereas adding the trailing inpaint anchor keeps that ending active. The probe holds prompt, duration, noise seed, input prefix and sampler constant for that comparison.
Ongoing conceptual duration remains 120 seconds. The runtime uses a small phrase-level prompt-embedding blend between the existing base and development prompts: establish, explore, develop, build, release. These are generation intentions, not verified musical states or guaranteed linear intensity control. Navigation can choose a related closing prompt. There is no new style catalog. Horizon is the current bench default.
## Evidence and performance
`results/continuity/listen.html` contains the new review and energy plots, with the rejected raw Horizon/chamber regression cases inside the diagnostics section. The controlled anchored journey is 125.945 seconds. True new-material boundaries are 21.920, 47.926, 73.932 and 99.939 seconds; the end anchors are separately shaded. Human judgment is still required at every join.
Four warm probe jobs took 19.268–19.428 seconds. Playback RTF is 0.741–0.747; RTF against newly generated interior alone is 0.879–0.886. Each job adds 6.58–6.74 seconds of buffer headroom when the reused anchor is included. About 84% of each appended passage is newly denoised material. Startup buffering does not count as sustainable throughput.
The old ~0.97 useful RTF came from discarding the last ~2 seconds while retaining ~8 seconds of context: only 19.97 usable seconds remained per job. The older untrimmed ~0.878 figure used ~22 seconds. The new strategy improves useful musical yield, not GPU speed. Its novel-material RTF remains similar to the old untrimmed case.
The first cold job of the final worker took 96.750 seconds including graph capture; the second warm-up took 22.505 seconds. These job timers exclude initial model/weight loading. Sixteen accepted live jobs took 19.231–19.441 seconds (median 19.341), with decoder execution around 0.7 seconds. The 210.1-second long capture accepted seven live continuations, held at least 10.24 seconds buffered, and had zero fallback loops/output flags and no -50 dBFS/100 ms gaps in dry or wet music. All four runs together captured 517.6 seconds; this is finite validation, not an indefinite soak test. CPU handles cached conditioning, noise, replay, resampling, arranging and the Conductor. Chestnut runs DiT and decoder. Resource measurements are worker peak RSS and tracked GPU allocation sampled after decode, **not total system RAM or a verified peak-VRAM trace**. Exact live metrics are in `results/continuity/live_metrics.json`.
Measured positive headroom and the live captures support sustainable average operation at this bench load; they do not guarantee indefinite operation, all seeds, all prompts, thermal conditions or coexistence with modeld. No simultaneous driving inference was attempted.
## Curve and arrival
The Conductor and curve detector are unchanged. The 100.1-second Horizon curve capture retains 6.208 seconds of callback-measured anticipation. Reported DAC buffering adds ~0.100 seconds, yielding ~6.108 seconds by that estimate; the speaker was muted, so neither is acoustic evidence. Video packaging accounts for this delay using actual callback timestamps.
Actual route203 cereal shows reverse near 232.2 seconds, navigation invalidation near 241 seconds, braking reverse near 243.9 seconds and park near 255.1 seconds. The runtime now requires recent arrive context within 40 m, reverse below 2 m/s with braking sustained 0.8 seconds, and explicit navigation invalidation; park/standstill or a sustained stop remain fallbacks. It does not treat reverse alone, lost navigation alone, or EOF as arrival. A new valid route clears the context.
The verified trigger is 244.654812 seconds. Rendering starts around 244.935743; the reported DAC estimate is 245.035743. The difference includes local harmony analysis and callback scheduling. Two related closing jobs were accepted before resolution, and the source contains no -50 dBFS/100 ms gaps before the ending. A five-second source-informed sonority then decays. It may still be stylistically imperfect; listen to the generated closing and exact landing together.
## Limits and regression scope
There is no claim yet that the boundaries are hard to hear, the anchor sounds like a natural refrain, the arc develops convincingly, or the synthesized final chord is the right cadence. No amount of numerical evidence substitutes for that listening decision.
Musical logic contains no route IDs, excerpt timestamps, future route reads or precomputed event locations. The same generated-source policy was run on a second cached route. Route IDs and excerpt times occur only in test/publisher selection and offline evidence. The privacy guard remains active. Thirteen detector/musical unit tests pass, including reverse-without-destination and route-reset cases. The public tree and custom display were not edited.
A validation-shell edit while that shell was running caused an extra trailing command error after all three initial captures were saved; the captures and audits completed. The final script passes shell syntax validation, and the longer run was launched separately. This was an orchestration error, not an audio/runtime failure.
Normal existing-onroad replay integration and safe live modeld coexistence remain future work. No new custom visualization was built.
-51
View File
@@ -1,51 +0,0 @@
# RoadScore demo
RoadScore makes an adaptive soundtrack from road context. The native demo shows a recorded drive in the normal comma UI while ACE-Step generates new musical passages on Chestnut. Prism is the prepared musical profile. RoadScore supplies the native port, musical continuation, buffering, quality checks, immediate gestures and synchronization; ACE-Step is a pretrained model, not a model trained by this project.
The conductor receives only road messages already delivered by replay. It requests future passages and uses immediate, source-derived gestures for short road events. Generation takes time: the buffer lets the current score continue while the next passage is being made. A curve gesture does not imply a complete new neural composition in that instant.
## Reading the overlay
RoadScore uses a native-style icon and label group in the free area below driver monitoring and current speed, left of the speed-limit sign. A 50-pixel musical mark matches the native steering wheel and accompanies a fixed 20-pixel RoadScore title and 14-pixel supporting identity. Local text shadows preserve contrast without a panel covering the camera. Normal buffer/backend detail is omitted; degraded state remains explicit. The group yields to actual native navigation cards and alerts.
| Display | Meaning |
| --- | --- |
| Prism / Aurora | Reported prepared musical profile. Prism is the primary candidate. |
| INTENT: VERSE > PRECHORUS | Current section conditioning and the next scheduled section, when provided by the runtime. These are musical intentions, not verified structural labels. |
| ACE · CHESTNUT | Reported composer and compute backend. This is not an independent GPU health probe. |
| PREPARING | The status does not yet report playback readiness. Cold preparation is separate from warm generation time. |
| READY | The runtime reports ready and no degraded condition is present. This does not certify listening quality or live-driving readiness. |
| GENERATING | A fresh passage is in flight. The elapsed time is job time, not an estimated countdown. Buffered music may continue playing. |
| DEGRADED | The runtime reports a failure, quality issue or accepted-music hold. This label takes priority even if a generation job is still in flight. |
| Holding accepted music | Previously accepted music is being reused. It is not fresh generation. |
| BUFFER 64s | Reported queued playback time. It can include accepted-music holds; it is not a fresh-music counter. `--` means unavailable. |
| STORED SCORE · NO COMPUTE | Archived score playback. It must not be presented as fresh Chestnut generation. |
Native text uses ASCII labels and measured truncation to avoid unsupported glyphs or text spilling outside the panel. The middle dot in the backend label is drawn geometrically. Long section labels end with `...`; full source values remain in the existing status/audit JSON.
## Presenter explanation
“Prism is the musical identity. The recorded road messages guide the score as they arrive. ACE generates future passages locally on Chestnut, while the buffer keeps playback moving. Short gestures can react sooner than a new generated passage. READY means the runtime is ready; GENERATING means a new passage is being made. If generation cannot keep up or a quality check fails, DEGRADED makes that visible, and the system can reuse accepted music.”
Use [EVENT.md](EVENT.md) for measured event claims. The recorded 45 W native pass was muted and does not establish physical audio, Bluetooth, live driving or modeld coexistence. Historical performance reports describe a different bench. Operational commands belong in [README.md](README.md#run-on-the-event-comma) and are for the agent/operator who owns the hardware.
## Offline UI review
From the repository root, with a Python environment containing Pillow:
```sh
python3 -m unittest discover -s roadscore/tests -v
python3 roadscore/tools/preview_overlay.py --output roadscore/results/ui-polish/overlay-states.png
```
The preview renders six full 536 × 240 synthetic canvases with the repository's Inter bitmap font atlas: preparation, ready, generating, accepted-music hold, failed composer and stored score. It uses CPU image drawing and does not open a window, contact a device, start a worker, run replay or open an audio stream. It is a layout review, not a native-rendering or hardware-validation claim. The output stays in ignored `results/`.
Road gestures take priority over job timing in the detail row; degraded explanations take priority over both. The startup label uses the selected Prism/Aurora profile; stored playback is labeled as such from preparation onward.
The active native overlay is `prototype/overlay.py`; the pure presentation helper is `prototype/overlay_view.py`. `prototype/index.html` and `prototype/native_display.py` are older excerpt-specific bench interfaces, not the normal onroad demo. The historical style selector is not an ACE profile selector. Keep those older entrypoints distinct when presenting the current demo.
The contextual line names reported musical cues: Turn signal / percussion, Curve ahead / build, Curve apex / impact, Navigation turn / accent, and arrival/stop/resume cues. Active cues take priority over queued cues; queued cues are explicitly prefixed Next. A brief curve apex takes priority over ongoing signal percussion. Unknown kinds are labeled Music cue without guessing a road cause. The archived player supplies these fields from the original scheduler timing.
The score ribbon yields completely to native selfdrive/StarPilot alerts and their fade-out. Optional `ROADSCORE_CAPTURE_EVENTS=1` captures the first displayed active cue of each kind into the ignored replay output for UI review; it does not alter status or music.
The supporting action line stays at a fixed anchor. Brief completed cues may remain for up to 2.5 seconds labeled Recent; queued cues must remain present for 0.4 seconds before display. New active cues and degraded state update immediately. This changes presentation only; raw scheduler status and capture timing remain unchanged.
-47
View File
@@ -1,47 +0,0 @@
# Horizon Drive — denser continuity and musical finalization
## Human baseline and preservation
The review of e3477e2 provisionally accepts sparse Horizon's active-context/anchor continuity and harmonic development. It likes the final cadence but rejects its abrupt entry. This pass preserves that architecture and cadence, and tests a deliberately more transient-rich source. `dense-before-e3477e2` preserves the prior implementation; its recordings remain under results/continuity.
## Focused conditioning experiment
One new identity, Horizon Drive, requests a 108 BPM instrumental electronic/post-rock band with drum kit, bass, rhythm guitar, lead motif and synth counterlines. Eight cached text conditions cover base, explore, develop, build, peak, release, closing and approach. Encoding is local CPU T5Gemma from existing weights; no network or model changes. These words are intentions, not proof that each instrument was produced.
Two fresh seeds were tested. Seed 7301 was selected before the live test for stronger high-frequency transient content and a pulse estimate close to the requested groove. Against sparse Horizon's first 21.92 seconds, its transient detector reports 3.19 vs 1.92 peaks/second, high-band energy share 0.141 vs 0.024, normalized spectral flux 0.109 vs 0.063, and estimated pulse ~107.4 BPM. Seed 7302 also estimates ~107.4 BPM with 3.24 peaks/second. These metrics are diagnostics; they do not identify drums/bass or establish taste. Both seeds are in the collapsed listening diagnostics.
The selected opening is a route-independent generated identity seed. It is not a pre-generated route soundtrack. All subsequent accepted continuations were generated during the live captures; startup uses one matching prewarm as before. This is still the private prototype workflow, not a finished cold arbitrary-route launcher.
## Dense continuity stress test
First, the existing architecture and old prompt-blend/Conductor ran unchanged with the dense source for 210.2 seconds. Seven live jobs were accepted; no fallback loops or PortAudio flags; minimum buffer 10.24 seconds. Playback RTF median ~0.746. No -50 dBFS/100 ms windows appeared before EOF. Pulse estimates across 20-second sections stay near 107–109 BPM, with transient rates roughly 2.2–3.2/second.
Every transition into new model material and every audible suffix-anchor entry is checked in `boundary_metrics.json`: 21.920, 47.926, 73.932, 99.939, 125.945, 151.951, 177.958 and 203.964 seconds. Both the model-created transition into the suffix anchor and the next continuation boundary matter; listening plots shade all anchor intervals. Estimated pulse displacement across the initial stress-test joins ranges from ~0.001 to ~0.150 beats. That is an approximate analysis window result, not proof of sample/beat alignment or inaudibility. Suffix-anchor entries in the initial stress run show estimates up to ~0.257 beats. The later arrangement run shows ~0.35–0.36 beats near 43.84, 95.85 and 199.88 seconds. An interior control window also reaches ~0.311 beats, so these estimates are noisy and confounded by changing instrumentation. Those entries are explicit listening flags, not declared inaudible joins. No silence removal, stretch, gain repair or offline seam editing was applied. This evidence did not justify redesigning the architecture before human listening.
The unchanged per-job budget is 324 latent frames, 44 retained prefix frames, 44 reused suffix-anchor frames, and 236 newly denoised interior frames. Each job appends 26.006 seconds: 21.920 new plus 4.087 reused anchor. The common-context splice remains two seconds. The anchor still repeats; dense human listening must judge whether that repetition becomes conspicuous.
## Arrangement and immediate road response
After the stress test, Horizon Drive's live sequence uses explicit explore/develop/build/peak/release text conditions instead of interpolating base/development embeddings. Closing overrides this cycle using already-available navigation. The prompts request changes in bass movement, percussion density, guitar/synth layers and harmonic tension; they cannot guarantee those outcomes. The initial source and prewarm establish the groove. Accepted-job count advances the cycle, so time/route IDs are not baked into musical logic.
The curve detector is untouched. `DrivingDSP` is enabled only for Horizon Drive: it extracts transient-rich high and low bands from the generated mix, stores short non-feedback delays at estimated eighth/sixteenth intervals, and brings in modest reprises as anticipation develops. A short attack reinforcement marks the event, and release reduces the additions. Existing phrase echoes remain. Severity comes from the detector's predicted lateral-acceleration strength, bounded to modest intensity. This does not provide stems, new drum composition or a guaranteed beat grid. It is a deterministic source-derived rhythmic texture; listening must determine whether it feels like a build or merely extra echoes. No stock drum samples or universal drop.
## Cadence runway
The causal vehicle trigger is unchanged from e3477e2. At that trigger, no more generation is requested, road-event embellishments subside, and the current musical thought is allowed a bounded continuation. A 10 ms spectral-flux pulse estimator examines already-generated recent music. When periodicity confidence is at least .25, it evaluates beat-grid opportunities 0.8–3.5 seconds ahead; otherwise it evaluates local energy-release opportunities. It favors a decayed passage before the landing and avoids an impending attack. It reads only audio already in the buffer, never future route events.
This is **not bar or phrase recognition**. Sparse old guitar also yields apparently high autocorrelation confidence, so confidence alone is insufficient to establish a real beat. A synthetic 120 BPM transient test verifies timing mechanics; the dense source's estimate near requested 108 BPM provides supporting but incomplete empirical evidence. The live delay and alternatives are recorded in ending.json.
The final five-second sonority synthesis is unchanged. Its harmony is estimated from the generated audio leading up to the selected landing, and its start is sample-aligned inside the audio callback. Unlike a naive delayed-start change, samples before that future entry remain unchanged and do not play the cadence early. The existing 300 ms overlap into the cadence is preserved. Tests cover early-playback prevention, bounded delay, weak-pulse release selection, and lack of queued audio. No fixed arbitrary arrival delay or later vehicle trigger was substituted.
## Review and limitations
`results/dense/listen.html` presents the three primary audio tests directly: long dense live continuity, live curve, and phrase-aware live arrival. Timing markers are optional; video links are explicitly labeled optional. Diagnostics and seed candidates are collapsed. All artifacts remain local/private. Human musical success is not claimed.
The implementation still cannot isolate true drum/bass stems, prove exact bar boundaries, guarantee instrumentation or demonstrate modeld coexistence. No new onroad visualization, launcher architecture, training, model shopping or public StarPilot edits occurred.
## Capture/verification limitations
The long final capture and curve dry PCM diagnostics each contain one full-scale channel sample after resampling. Their actual conducted listening captures contain no full-scale samples. Files were not repaired or gain-adjusted. The same-source curve diagnostic differs by about 5.1% RMS from the previous effect over its 27-second window; this establishes that the added processing is present, not that it is compelling.
The local browser rejected file-URL access during layout verification. No workaround was used. All three primary players, WAV headers/durations, local audio/video links, and absence of autoplay were checked statically; browser rendering was not verified. Initial harness setup needed the existing soundfile import path, and video packaging needed the installed ffmpeg absolute path because the shell PATH omitted it. Both were corrected without new dependencies.
-115
View File
@@ -1,115 +0,0 @@
# COMMA_HACK 7 event integration
Pre-event private baseline: `8a957d1`, preserved as tag `pre-event-8a957d1` in Desktop/RoadScore. The original source, routes, models and evidence remain there unchanged.
Event source lives under `roadscore/` in StarPilot on `RoadScore`. Existing module layout is retained to reduce migration risk. Generated artifacts, routes and model weights are excluded from Git. Historical reports describe the pre-event bench, not event validation.
Target: `comma@192.168.63.143`. Automated runs remain physically muted. Physical listening and live driving require attended validation; no live safety claims.
Status: source migrated on `RoadScore`; official example passed; bounded GPU recovery investigation and full Prism validation in progress. See current checkpoint below.
## Night-one checkpoint: hardware connection interrupted
- Source migration committed; event comma switched from `Dom` at `b990a776b2` to `RoadScore` at `e0aa4543aa` (verify full revision on reconnect). Existing active-theme changes on the comma were preserved.
- Event inventory: comma four (`mici`), SDM845, AGNOS `19.6.20`, kernel `4.9.103`, 3606 MiB RAM, no swap, roughly 82 GiB free on `/data`; CPU0–3 online at inventory. Real `IsOnroad` was false.
- USB showed only root hubs and Quectel modem. **No Chestnut detected**. GPU capacity/link/firmware cannot be reported yet.
- `bluetoothctl show` timed out without controller details; `pactl` was unavailable. ALSA listed the onboard sdm845 card. No Bluetooth sink was selected and no audio stream was opened.
- Official repository cloned; its setup first failed because `/home` is a 100 MB overlay. Restarted with cache and temporary files under `/data`. The first cache was preserved under `.cache/uv-first-attempt`. Installation completion is not verified.
- Replay build started; compiler warnings were recorded, but executable completion is not verified.
- ACE profiles transferred (approximately 20 MB). Weight transfer reached approximately 585–601 MB before SSH reset; it is incomplete. VAE and fixed reproducer fixture still need verification/staging.
- At the network interruption, no RoadScore GPU worker or replay had been started. No model timing or event hardware success is claimed.
- Mac migration tests: 20 hardening tests plus 61 core/settings/clock/archive tests passed. Logs are private under `roadscore/results/event_night_one/`.
- The first inventory script encountered an offline CPU-policy sysfs read error. The committed tool now records unavailable reads as null instead of aborting.
## Resume sequence
1. Confirm device IP and Chestnut power/USB connection. Set `ROADSCORE_DEVICE` if changed.
2. Bring the event branch up to the latest local commit, preserving the device's theme changes. Do not push private data. `/data/roadscore` is a compatibility symlink to the repo subdirectory.
3. On comma: `bash /data/roadscore/tools/bootstrap_event.sh`. This verifies real offroad state, keeps automated audio muted, finishes dependencies and builds replay. It does not start GPU work.
4. On Mac: `python3 roadscore/tools/stage_event_assets.py`. This resumes ACE weights/profiles and the exact saved decoder fixture from Desktop. `--known-route DONGLE/ROUTE` optionally stages an already-used regression route without score archives. No judging submission acquisition before the Phase 1 gate.
5. Finish the official repo's `uv sync --locked --python 3.12` with `UV_CACHE_DIR=/data/roadscore/.cache/uv TMPDIR=/data/roadscore/tmp`, then `tools/setup.py`. Stage `models/yolo26n.onnx` from the already-exported official Mac example.
6. Run `bash /data/roadscore/tools/official_chestnut.sh`; it refuses a busy GPU, verifies offroad, saves logs, and runs official USB validation plus YOLO. Stop and preserve any failure.
7. Start `prototype/worker_service.py start --composer ace --profile prism`; capture cold preparation and warm continuation metrics, then stop the owned worker before the decoder reproducer.
8. Run `tools/decoder_repro.py`; use its fixed input and unique output folder. If the link fails, ask an engineer before further hardware debugging. Then native replay, attended audio checks, and the remaining ordered brief.
SA3 source and local converted weights exist, but its event assets/environment have not been fully staged or validated. Do not call it an available event fallback yet. Aurora's prepared profile transferred but likewise has no event-generation result.
Phase 2 has **not started**: no official judging runs, no identity reveal, no winner, and no submitted-route characterization or seed selection. Its required event hardware/native replay gate is unmet. Live adapter work, modeld coexistence and physical listening remain pending; no live drive is authorized by a test result.
### Latest connection update
The device briefly returned, and the event branch was fast-forwarded successfully to `cfdd608238`, including the repo-level `./onroad --roadscore` dispatch. Setup and model transfer were resumed. Connectivity failed again; the resumed transfer exited on SSH timeout. The last observed USB inventory still contained no Chestnut. This is an intermittent connection/hardware blocker, not a completed event bringup. The bootstrap/replay and official setup resume logs are on the comma; their final status must be checked after reconnection.
The root launcher dispatch was tested in isolation: explicit `--roadscore` routes to the integrated launcher with all arguments preserved; ordinary replay routes to the original host runner. Both cases passed. Total completed local checks: 81 regression tests and 2 launcher cases. No judging runs were attempted.
### Hardware interruption finding
A subsequent read confirmed the same hostname `comma-14765f`, the `/data/roadscore` symlink, deployed `cfdd608`, and approximately 80 GiB free. Its uptime had reset to about one minute, so at least one actual reboot occurred; the cause is unknown. Fresh SSH then timed out again. No automatic power cycle or GPU reset was performed by this task. Further heavy setup is paused pending stable power/connectivity and Chestnut enumeration. Stale local SSH clients were closed; remote final process state cannot be certified while disconnected. No RoadScore worker, replay, audio stream, or judging run was started by this pass.
## Current event checkpoint
Source and dependencies are deployed on the event comma. Both working trees use the `RoadScore` branch. Changes remain local or are copied directly to the comma; no further publishing is authorized.
- Official Chestnut USB and YOLO examples passed on gfx1200, firmware ed4e39b7-CLEAN, with 8,539,602,944 bytes reported VRAM.
- All 744 staged ACE weight/profile files passed SHA256 verification. Native replay built successfully after repairing interrupted zero-length objects.
- First Prism attempt failed during the second VAE invocation. The first decode/readback succeeded, but no completed music asset was produced. GPU config became inaccessible, while the bridge remained readable. Timeline data repeated the bridge identifier. No driver reset was attempted by the timeout handler.
- The user reported handling the hardware and confirmed no engineer is available. An explicit offroad controller power recovery restored link 0x78. A controlled decoder-only repeat test is underway; no event Prism or native replay pass is claimed yet.
- All tests remain physically muted. No Bluetooth output was enabled.
## Route privacy
A read-only check found the RoadScore branch already present on the public repository at 51ea668d73. No RoadScore route recordings, generated scores, or result directories are tracked, but historical reports and helpers exposed real route identifiers. The user was informed. Identifiers have been removed from the local working files; historical fixture tools now require private environment settings. This does not remove identifiers from existing public Git history or change comma Connect permissions. No remote history rewrite or push was performed. The exact local audit is in ignored `results/event_night_one/route_privacy_audit.json`.
The controlled default decoder test reproduced the fault: cold call 163.49s, capture call 8.10s, both identical finite PCM; repeat index 2 lost GPU access. Peak tracked allocation was 2,470,055,936 bytes, far below reported capacity. A no-graph diagnostic (`JIT=2`) is in progress. This is a repeated event-unit failure, not a memory-capacity pass/fail inference or an identified root cause.
### Controlled decoder result: 30 W cap
Normal graph execution at 30 W completed 40/40 identical finite PCM outputs. Cold call 164.416s, capture 9.363s, warm median 2.760s (max 2.791s), host peak 321.10 MiB, tracked GPU peak 2,470,055,936 bytes. All link samples remained healthy; CPU settings restored. The no-graph attempt instead stalled on call index 1 with GPU configuration still readable and timeline 238 versus target 288.
This is evidence of load-sensitive behavior, not proof of a faulty supply. Full Prism preparation now runs under the same 30 W cap, recorded in the service state and generation provenance. Default model settings remain unchanged. No official-runtime comparison is needed unless the full workload fails.
The UI working change exposes current job elapsed time through native/host status transport. It still requires native replay validation. Privacy cleanup, diagnostics and UI changes remain uncommitted in both RoadScore working trees, with no push.
User-reported power source: approximately 120 W, 12 V brick; rating and delivered voltage/current have not been independently measured. The cap result does not establish that the supply is undersized.
### Full Prism preparation and native integration
At 30 W, Prism prepared 112s of accepted audio in 591.215s from worker start, with 38.228s model load and 783,356 KiB peak host RSS. All four initial sections passed without rerolls. Warm continuation required approximately 44s per 28s of new material (RTF 1.57–1.58); this cap is stable so far but is not faster than playback.
The first native replay rendered camera/path/lanes and the overlay, then its audio process failed after 22.1s when an ACE request accessed an absent legacy SA3 anchor. No underflow occurred before the exception. Evidence remains in `results/normal_1789793308/failed_score`. The general fix excludes SA3 anchors/cache checks from ACE requests without changing its boundary DSP policy. Native receiver/launcher supervision now propagates score-process failures. A same-route muted retry is running under `results/normal_1789793554`. Its first fresh music request succeeded in reaching the worker. No full replay pass is claimed until EOF and audit.
### Replay retry outcomes and next bounded power test
- `normal_1789793554`: full 254.1s EOF, 5,091 camera frames, navigation, 10 curves, zero future-input violations, zero underflow, all blocks muted. Failed fresh-generation acceptance: one real quality rejection followed by a zero-attempt deadline deferral was incorrectly counted as a second bad generation. The healthy worker was marked failed and accepted music was held.
- General correction: `GenerationBudget` distinguishes deadline deferrals from quality failures and uses measured warm generation cost to reserve time with accepted-music holds. Three targeted regression tests pass.
- `normal_1789793863`: accepted three fresh continuations (~46s each), no GPU fault. Later holds could not preserve the exact latent context after an intentionally faded closing section; generation was deliberately suspended. At route time ~227.5s, the separate causal replay bridge detected 1.35s timing drift and stopped. This is not a clean native pass. Original current-run files are retained in its `failed_score` folder.
- The replay timing guard has not been relaxed. Phase 2 remains unstarted. No modeld coexistence or live driving claim.
- Resident Prism was stopped cleanly before a bounded 45 W fixed-decoder test (40 calls). The default ACE worker cap stays 30 W unless explicitly overridden. Further musical logic changes wait on this performance/stability comparison.
ACE is now the event default; explicit SA3 selection is preserved. Composer tests pass with the existing analysis environment (the system Python lacked SciPy). Native launcher supervision and UI state changes remain local/unpublished.
### Coexistence investigation (read-only, not executed)
Actual native Params report `Model=DrivingModel=rdf43`, version v15, real `IsOnroad=false`. The built-in model path uses the QCOM backend and does not select an external-GPU artifact merely because Chestnut is connected. Model Lab configuration must still be checked before any test.
The existing `selfdrive/test/process_replay/process_replay.py` supports isolated `modeld` execution in an `OpenpilotPrefix`, feeding road/wide camera frames and device/calibration/car state; it publishes modelV2, drivingModelData and cameraOdometry. The cached known route contains `fcamera.hevc`, `ecamera.hevc` and rlog. A bounded test should call that local harness directly and record execution times first alone, then during ACE generation. Do not use the CI model-replay report/upload entrypoint for private routes. No coexistence or live-driving test has been run.
### 45 W full preparation
The explicitly capped 45 W resident worker completed 112 seconds of accepted Prism audio in 530.441 seconds, including 38.190 seconds model load. Peak host RSS was 758,308 KiB; tracked allocation reached 5,789,487,104 bytes. Initial 28-second music required 13.651 seconds generation plus 6.206 seconds decode (0.709 RTF excluding cold compile). Warm continuation required 19.51–19.59 seconds generation plus 10.30–10.32 seconds decode per 28 new seconds (1.065–1.068 compute RTF; 30.12–30.42 seconds wall). This does not establish sustained faster-than-playback continuation.
Native muted replay `normal_1789795416` is running without continuous UI video encoding. Camera/path auditing and the overlay image remain enabled; the causal timing guard is unchanged. No event native pass is claimed before its final audit. Exact preparation metadata is preserved privately in `results/event_night_one/prism_power45/`.
### First clean event native replay — 45 W
`normal_1789795416` passed the instrumented native gate through final-segment EOF: 254.1 seconds captured, six accepted fresh generation jobs, zero accepted-music holds, zero underflows, zero emergency fallbacks, no worker failure, zero output flags and all blocks muted. Camera accepted 5,088 frames; path/lane drawing, navigation and ten curve activations were observed. All request timestamps were causal. Maximum source-clock drift was 24.671 ms. Continuous UI video encoding was disabled; the overlay snapshot and UI audit remain available. This single pass does not prove that encoding caused the earlier timing failure.
The local private evidence is `results/event_night_one/prism_power45/`: preparation metadata, native audit, overlay and captured score. The reusable post-run audit correctly rejects the earlier full-route failure and has three focused negative-evidence tests. It does not claim human musical approval, physical speaker/Bluetooth validation or modeld coexistence. Phase 2 has not yet begun; the native prerequisite is now met. No upload or push was performed.
### Private judging batch setup
After local checkpoint `f75a98181f`, all eleven unique submissions returned native route-file listings. Private complete-route caching has begun. The private manifest fixes anonymous labels, route-derived seeds, common Prism/45 W/muted settings and objective excerpt rules before any judging playback. Exact identities and acquisition logs remain ignored/private.
Checkpoint `963d069778` adds optional deterministic sampling: separate preparation/continuation seed streams, request sequence independent of wall-clock job filenames, and rejection of a mismatched resident preparation. Legacy sampling remains unchanged unless explicitly configured. Three seed tests and six composer tests pass. Submission A's first attempt is preparing; the single-attempt runner waits for its matching preparation and complete ordinary route cache, then performs muted native replay. It refuses automatic reruns. No listening scores or winner exist. Range-disambiguation entries remain blocked from launch until their metadata decision is recorded.
-84
View File
@@ -1,84 +0,0 @@
# Local event integration candidate
This candidate combines the compact UI, judging preparation/export fixes, and
fail-closed judging preflight. No hardware validation follows from local tests.
## Hardware handoff and first run
Only the explicitly authorized hardware owner may execute this sequence.
1. Inventory the current device revision, working changes, real offroad state,
owned processes, GPU/link health, mute safeguards, caches and original attempt
ledgers. Preserve the device's existing theme changes. Do not stop another
owner's processes. Reuse verified device caches before transferring assets.
2. Stage only the reviewed source diff from `90bd992ced` to this candidate using
a local patch/direct transfer. Check applicability first; no remote Git push,
no replacement of generated files, routes, private ledgers or frozen policies.
3. Verify actual firmware power state, not just the environment. Historical caps
may persist: an unset `AM_POWER_LIMIT` requests driver default but does not
prove the physical limit reverted. Establish the supported stock/default
setting and read it back before calling a run full speed. No speculative
resets or wattage sweeps. Preserve any reproduced link failure before using
an explicitly versioned practical mitigation.
4. Prepare Prism on the device with `AM_POWER_LIMIT` removed from the environment
(the ACE worker no longer supplies an implicit 30 W cap). Use the established
regression route for the first full-speed native replay, muted, with the
integrated overlay. Keep UI capture/audit and the unchanged causal guard.
5. Validate visible READY/GENERATING/DEGRADED, buffer and gesture behavior. Then
perform Bluetooth listening only with the user attending and explicitly
verified sink/volume. Do not globally remove mute safeguards.
Native preparation command, only after the checks above:
```sh
env -u AM_POWER_LIMIT /usr/local/venv/bin/python /data/roadscore/prototype/worker_service.py start --composer ace --profile prism
```
Native regression invocation uses the already verified private route variable:
```sh
env -u AM_POWER_LIMIT /data/openpilot/onroad --routeid "$REGRESSION_ROUTE" --roadscore --composer ace --profile prism --muted
```
For every significant run, save requested and verified actual power mode/cap,
preparation timing, warm generation RTF, decoder time, GPU/link samples,
underflows, accepted-music holds and minimum playback buffer. Retain native
summary, trace, audio callback log, UI audit and preparation/generation metadata.
Missing telemetry is unknown, never a passing measurement. Investigate callback
misses if reproduced and materially blocking; the old A underflow did not show a
GPU fault or depleted buffer.
## Versioned judging continuation
Do not invoke the legacy batch scheduler: it uses the historical unversioned
ledger directory. Invoke the guarded runner one ready label at a time under the
single hardware owner's supervision.
The new handoff schema requires an explicit `configuration.policy_version`,
`hardware_mode: full-speed`, `power_limit_watts: null`, generation authorization,
and per-label readiness with no blockers. Capped mitigation requires a distinct
policy version and explicit numeric limit. The runner strips inherited power
caps for full-speed mode and refuses legacy manifests for new attempts.
After reconciling device cache contents, create a new private runtime handoff
from the prepared export. Retain mapping/seeds and the explicit full-route and
457-second range decisions. Do not relabel, tune per route, or substitute a
partial log silently. Freeze the reviewed common runtime, quality/gesture and
profile file hashes in `configuration.freeze.private.json` beside that handoff;
it must be nonempty and every hash must verify before any attempt starts.
New ledgers and preparations live under
`results/community_judging/<policy_version>/`. Historical A's technically invalid
result and B's `paused_by_user` preparation remain unchanged. B may begin its
first replay in the new common version; A's technically invalid repeat must
retain a provenance reference to the original. All comparable official results
must use this same new policy; never mix the old 45 W run into a full-speed cohort.
```sh
/usr/local/venv/bin/python /data/roadscore/tools/judging_run.py "$PRIVATE_MANIFEST" --label "$LABEL"
```
A finished process is not necessarily a technical pass: require the native audit
before listing a valid demo. Preserve invalid audio as diagnostic evidence,
separate from the valid listening shortlist. Human ranking remains the user's.
Live input/coexistence changes are separate and must not be staged prematurely.
-75
View File
@@ -1,75 +0,0 @@
# Private integration pass
Music remains frozen at f3216fa: ignition_arc, existing continuation, 0.8 curve threshold. Human listening review remains pending.
## First unseen attempt — immutable evidence
`results/integration/first_attempt/` preserves command, f3216fa launcher, console/exit status, native host logs and bench capture. The supplied route `<private-route-id>` was launched by route ID only, without prior route inspection, caching, manual excerpt or source changes. Exit 0 without intervention: 3,631 accepted camera frames, 9,860 nonempty path/lane draws, navigation, eight curve activations, six new Chestnut jobs, 174.6 audio seconds, zero underruns/repeats. This first attempt predates host audio return, so capture was on the muted bench. UI rendering is instrumented, not visually reviewed.
Only after preserving this result was the complete native file listing cached locally: 60 files, 1,214,246,322 bytes, ten segments, road/wide/driver/low-resolution cameras plus rlog/qlog. Private signed URLs are not retained in the library manifest.
## Ownership
Mac: existing native replay and existing UI own route resolution, video and original cereal time. Ordered SSH forwards only already-published music services to the bench. The shared app and Chestnut generate/render final PCM. A Unix socket returns bounded final stereo float32 blocks through SSH. The Mac owns its audio device and FLAC capture. `--audible` means Mac speakers; the remote comma is always muted.
PCM packets carry the existing callback/source clock. The host measures remote monotonic offset, schedules output at callback time +250ms, caps the receive queue at two seconds, and records target/actual DAC estimates and late/drop counters. There is no unbounded streaming buffer. Default automation runs mute the output after capturing the identical rendered samples. A 66.2-second test had zero late frames, zero starved callbacks, zero output flags, and a maximum queue of three 100ms blocks. Physical speaker audibility and acoustic AV timing have not been listened to.
Comma: same launcher detects `/TICI`, selects the existing source tree and a privately built native replay executable. Same cereal bridge, app, musical core and generation scheduling; local output. Actual existing UI temporarily owns display; original manager resumes on cleanup. Native hardware acceptance subsequently passed; see the completed results below.
## Library and archive lifecycle
`./routes-tool inventory` reports the private library. `./routes-tool fetch DONGLE/ROUTE` uses the same authenticated native route API endpoints and fallback order, downloading all compatible listed log/camera files without printing signed URLs. Native replay reads ordinary `<route>--<segment>` directories under `routes/<dongle>/<route>/`. Resolution is library, configured `ROADSCORE_ROUTE_PATHS`/native realdata, then native remote resolution. No allowlist and no route-specific runtime features.
Scores live inside the same route parent at `roadscore/<session>/score.flac`, with timing, model/job boundaries, phase/curve/ending decisions and metadata. Actual rendered output is saved; replay does not regenerate it. A route-owned latest pointer selects the last completed score. `./routes-tool delete DONGLE/ROUTE --confirm-route DONGLE/ROUTE` explicitly removes the private route parent and its owned scores. It is implemented but no real routes have been deleted.
The repository's `system/loggerd/deleter.py` recursively deletes individual realdata segment directories, not a route parent. Therefore future live recording must rotate audio and metadata into corresponding segment-owned subdirectories. The demo library is not falsely claimed to be integrated with live realdata pruning. No live recording integration is implemented in this pass.
`./onroad --routeid DONGLE/ROUTE --roadscore --replay` selects a stored final score and bypasses receiver, clock SSH and worker startup. Playback anchors to the original first model logMonoTime and recorded host DAC/file origin. The first 30-second Mac stored-score test passes with no output flags. Pause/seek/rate discontinuities fail rather than silently desynchronizing.
## Native build notes
The device lacked a replay executable. `prototype/build_native_replay.py` compiles existing openpilot replay/msgq/VisionIPC sources and generated cereal headers into `/data/roadscore/native_build`, linking managed AGNOS dependencies. It does not modify `/data/openpilot`. The first device adapter run exposed an absent private msgq directory; the general launcher now creates its namespace before subscribing. The subsequent acceptance results and software-decoder correction are recorded below.
## Human next step
Do the listening review at results/unattended/listen.html before the next musical prompt.
## Completed native and stored-score acceptance
The actual comma UI passed first on the existing fixture (1,301 camera frames, 2,351 nonempty path draws, 3,206 lane draws; 66.3 audio seconds, two accepted jobs, no underruns/repeats). Normal UI ownership was restored afterward.
The community fixture then exposed `VIDIOC_STREAMON CAPTURE failed` in Qualcomm's hardware decoder. This is preserved in `results/integration/native_decoder_failure`. Native mode now consistently selects the existing software decoder. The retry (`results/integration/native_community`) passed: 84.7 audio seconds, two jobs, no underruns/repeats, 1,732 accepted camera frames and 3,007 nonempty path/lane draws. Its final score archived automatically. Empty-model/empty-audio sessions now fail before archival. None of these later integration fixes changes the preserved first unseen Mac result.
The same final contiguous samples were verified during stored-score playback on both hosts. Mac: 0.19ms maximum alignment error, invalid/unreachable compute-host argument, no generation. Comma: 14.3ms maximum alignment error with the generation worker stopped. Both had zero output flags. `stored_score.py` advances a contiguous sample cursor and verifies its PCM digest against the original FLAC sample range. These are instrumented output-path checks; acoustic output was not heard. A virtual loopback input stalled in CoreAudio setup and was terminated without collecting audio.
An intentional SIGTERM stopped native replay and restored the ordinary UI (PID134958 observed afterward). The external integration worker was stopped and `POWER_RESTORED` verified before the no-worker native stored-score test. The subsequent full-route Mac test failed closed on excessive transport delay; its worker was stopped. The final native full-route test completed; details follow.
## Final host scheduling correction
Callback-timestamp jitter initially accumulated up to27ms of inserted spacing in a174.4-second returned stream. The corrected host scheduler anchors the first block to the measured clock, then presents the contiguous PCM sample sequence. It checks sequence continuity and records DAC timing error, queue occupancy, output flags and late/starvation counters. A subsequent84.5-second run had exactly4,800 frames between block starts, zero inserted spacing, and0.36ms maximum measured scheduling error. The fixed added delay is250ms; LAN clock-offset uncertainty adds a few milliseconds. Saved-score playback uses the same contiguous-sample principle.
## Natural EOF
Repository inspection found that native headless `--no-loop` waits after the final segment instead of terminating. `replay_end.py` supervises the native player's own final-segment exhaustion/status and closes the cereal pipe after exhaustion. Segment bounds are never sent to the musical runtime. No ending is armed by EOF; an already-triggered cadence may finish. Intermediate segment waits and paused replay do not count as EOF. Unit coverage includes those distinctions. Full-route empirical validation subsequently passed natural EOF handling; see below.
## Ownership and privacy details
RoadScore made no source edits in the public Mac StarPilot checkout. Final inspection found unrelated concurrent changes there; they were left untouched. The bench had pre-existing July28 active-theme symlink differences; they were left untouched. All new compilation outputs and Python integration code are in private RoadScore directories. No upload, push or PR occurred. The private route library contains170 verified compatible files, totaling3,099,393,010 bytes. Original first-attempt and historical development evidence remains separate from deletable demo-library entries.
The complete musical core, worker, prompts/styles, continuation and detector remain byte-for-byte unchanged from f3216fa. Only rendering/presentation, transport, supervision, cache and archive plumbing changed.
## Longer Mac run — explicit failed acceptance
`results/normal_1789519804` rendered309.6 seconds and completed11 jobs without renderer underruns/repeats, then stopped. The last model packet took approximately767ms from host send timestamp to bench receipt, above the750ms source-clock tolerance. The audio return recorded97 starved10ms callbacks and969ms maximum timing error. This identifies an end-to-end transport/OS scheduling delay; radio alone is not proven responsible. The wrapper failed closed and did not replace the last good archive. `results/integration/long_mac_failure.json` preserves the outcome. Do not call this a full-route Mac success. Longer network-jitter recovery remains outside the completed work. The final native test removes the network from the critical replay/render/audio path.
## Final full native route and cleanup
The whole cached community route completed through the actual native UI and local Chestnut (`results/integration/native_full`, session `normal_1789520509`). Cold startup173.898s; replay575.288s; final rendered audio571.9s. There were21 accepted jobs at19.685–20.067s each (median19.884s), zero emergency repeats and one active output underrun at90.2s. The observed callback gap was about305ms; it did not overlap an app GC event. No claim is made that the hardware dropout is repaired or reproduced by the saved PCM. The archive explicitly marks its timing quality as non-clean and retains per-block DAC timing.
The existing arrival detector triggered at563.672s from recent destination context and parked standstill. It selected a3.126s runway and finished the cadence at audio571.726s. No route-specific timestamp/configuration was added. Native final-segment exhaustion ended replay naturally; EOF itself did not arm an ending.
The actual UI accepted11,270 camera frames, made21,795 nonempty path and22,014 lane draws, and transitioned offroad at the recorded ending. This remains instrumented rather than visually observed. Worker processes were absent after cleanup, `POWER_RESTORED` was recorded, saved/before and restored/after CPU JSON matched exactly, and the ordinary UI was running again (PID139744).
The full score is copied to `routes/<private-route-id>/roadscore/normal_1789520509/` on Mac and comma. This contains actual lossless rendered audio, timing,21 generation records, navigation/curve/arrival decisions, implementation provenance and explicit output-quality flags. It is the latest-score selection. Human listening and physical screen/audio confirmation remain pending.
A final cross-host test replayed the full native-produced score on Mac for10seconds with an invalid compute-host address, zero output flags, and contiguous source samples verified=True. It correctly disclosed the source recording's output interruption.
-57
View File
@@ -1,57 +0,0 @@
From the Mac, start the saved showcase on the comma and in a native Mac onroad window:
```sh
./onroad --roadscore route1 --demo
```
Add `--fullscreen` to fill the Mac display while preserving the native layout:
```sh
./onroad --roadscore route1 --demo --fullscreen
```
Press **F** or **F11** to toggle fullscreen. **Escape** returns to the window without stopping playback. The native aspect ratio is preserved with letterboxing; this does not select a different driving UI.
The command prepares both local native replay engines, then releases their start barriers together. The comma primes its local route cache and first camera frame before publishing road state. The Mac decodes its own cached video and follows the comma’s route playhead. Audio plays from the comma only, through its selected output. No ACE generation or model warm-up occurs. The previous single-device saved-replay setup took 17.13 seconds for loading, replay parameters and UI startup; measure the paired path separately.
Use Galaxy for engagement and turn-signal controls. The Mac follows the same applied presentation selections. No browser control page opens and no video is streamed from the comma. Small Galaxy status updates keep the local playheads approximately aligned; this is independent playback, not frame-exact mirroring. The buttons affect replay appearance and musical presentation only, never vehicle control. Ctrl+C stops both owned demo sessions and leaves the resident ACE worker intact.
Left, Right and Off replace the recorded turn signals until Recorded is selected again. This applies to the music, native arrows and ordinary lane-change prompts on both displays. Off releases the shaker smoothly without starting more pulses. Important native alerts retain priority. Engagement and signal selections are independent.
Each registered alias uses its own matching route and preserved ACE/Prism performance. Replace `route1` in the command with another ready alias; no manual audio staging is needed. Readiness is recorded in the local catalog, and unprepared entries are rejected before playback. Never substitute another route's music archive to bypass this check.
The separately registered `route5` option uses route1's same footage with an explicitly staged gold-seed soundtrack. Its opening comes from the preserved original seed33602, followed by that recording's existing continuations. This is saved standalone music with real replay-driven presentation, not a claim that the music was freshly composed for route1. Its manifest pins the original route recording's clock, ending and hashes, and requires exactly the same sample count. Route1 remains the unchanged default.
```sh
./onroad --roadscore route5 --demo --fullscreen
```
Route5 retains engagement and motion effects throughout. Signal motifs are most reliable after25 seconds; uncertain musical grids suppress rhythmic additions. The curve near2:01.5 retains its bass build/return but omits the extra quantized cut/impact. Route1 retains its original staging.
The event recordings are approximately 5:02 for route1, 5:06 for route2, 7:33 for route3 and 1:45 for route4. Route1 remains the primary showcase. Route3 includes an accepted-music extension at the ending: a two-second blend followed by 6.3 seconds of repeated accepted material. Route4's original recording logged one output-scheduling underflow; its saved PCM is continuous, and the subsequent 20-second saved-playback check had no output flags.
Route2 completed its full saved recording on the Mac, including the recorded tail after the last road frame. Route3 and route4 use the same saved path and have separate archive checks and short playback checks; this is not a claim of full-route physical listening acceptance. Galaxy's Off, opposite-direction override and Recorded reset were checked against real route2 events on the comma, including the native arrow and prompt.
The local configuration is ignored by Git: `roadscore/assets/demo_catalog.json` stores the explicit comma address, local controls port and registered routes. The comma has its own catalog pointing to persistent local assets. Each entry needs the route identity, a compatible completed archive and an optional matching curve plan. No discovery or automatic target switching occurs.
The Mac launcher checks the complete runtime message registry before choosing its session namespace. This prevents the UI's own publishers, such as `uiDebug`, from hashing to the same TCP port as replay messages. The camera port is also reserved. The selected namespace is recorded in the Mac session's `launch.json`.
The dual-native mode completed a full301.7-second FiiO audible run with Galaxy controls reaching both screens and successful cleanup. Tracking skew was85ms median and171ms at the95th percentile. Network delays sometimes paused corrections while replay continued. There were no callback or clock exceptions, but28 audio status flags and21 forward clock corrections were recorded, so completion does not establish glitch-free output. Fullscreen also passed an8-second native Mac replay check with no output flags or clock exceptions. The public JLab speaker still needs its own physical output/timing check.
For a Mac-only interactive fallback, use:
```sh
./onroad --roadscore route1 --prepared-showcase --unpaired
```
This opens native onroad replay on the Mac and processes the same preserved core with the current presentation layer. It needs the route cached on the Mac and ignored `roadscore/assets/prepared_showcase.json`, containing `route`, `archive` and optional `curve_plan`. The Mac uses its selected system audio output. A comma Bluetooth correction does not automatically apply to a Mac output.
Compatible archives contain `dry.wav`, `launch.json`, `audio_blocks.jsonl`, `replay_origin.json` and `rhythm_timeline.json`. Original dry PCM is used at unity gain, with the recorded model/DAC clock. The final mix is not processed twice. `--score-archive PATH` explicitly selects another compatible archive for the independent mode.
Both saved launchers support `--muted`, `--duration SECONDS` and `--no-browser`. Paired `--check` contacts Galaxy to verify the target is offroad; independent `--check` validates local prerequisites only. The controls server binds to loopback and refuses an occupied port. The paired target must be an explicit private IPv4 address on Galaxy port 8082.
The independent mode can follow Galaxy controls with `--paired-comma URL`, but controls alone do not synchronize playback. The paired `--demo` launcher additionally holds both starts and enables playhead following. The Mac follows the comma; it never adjusts the comma's audio clock to match its screen. If the Mac replay fails after playback begins, the launcher records a degraded display and lets healthy comma audio finish. Temporary Galaxy status failures during playback retry without restarting either replay. After four consecutive failures, the Mac closes and leaves comma playback running to its local end; the terminal reports that its status is unknown. Reconnect to Galaxy to check or stop it. Retry and recovery evidence is saved in `peer_monitor.json`. Startup failure, verified native failure, ownership changes and Ctrl+C still attempt to stop the owned pair. `--demo --screen-mirror` explicitly enables JPEG capture for the earlier browser mirror. Normal paired playback does not capture or encode a video feed.
Prepared native Bluetooth playback requests250ms of output buffering while retaining20ms callbacks. The backend may round this request; both requested and actual latency are recorded in `prepared_summary.json`, along with callback gaps, processing times and underflow flags. DAC timestamps continue to drive presentation timing, so no second manual delay is added for this queue. Mac, system-speaker and calibration defaults are unchanged.
Ordinary `./onroad --roadscore route1` remains the fresh-generation path. Gold music and historical recordings are preserved. The deliverable is interactive replay; an MP4 is not a substitute.
-61
View File
@@ -1,61 +0,0 @@
# Overnight continuation from e8f3069
Session safety: `.session-muted` on Mac and bench overrides every audio endpoint. Mac system output muted. No Bluetooth operations. All launches explicitly muted; speaker verification deferred to human.
Human confirms native normal comma UI works. Preserve this result and all integration fixtures. New music research prioritizes breakbeat explicit song sections; prior musical freeze superseded.
Work queue (unchecked means not demonstrated):
- [ ] P0–P8: engineering experiments and captures complete; identifiable roles/coherence require human listening; late prediction cases documented
- [x] P9: full Mac fresh runs; measured stability, bounded recovery unit-tested (not arbitrary outage guarantee)
- [x] P10/P13/P14: compact intent/readiness overlay, native assets reused
- [x] P11/P12: profiles and residency measurements complete;30–45s target NOT met
- [x] P15/P16: shared settings, positional CLI, explicit automated mute; Bluetooth extension documented
- [x] P17/P18: form archive, stored no-GPU regression, live storage design
- [x] P19–P21: three routes and native replay regression; host ownership
- [ ] P22–P25: long-form artifacts, transition diagnostics and listening gate complete; audible coherence/downbeats pending human; no fine-tuning
- [x] Final cleanup, worker-exit CPU restoration, ordinary UI restoration, mute audit, accurate STATUS
Experiments: `results/overnight/`. Existing evidence remains unchanged. No acoustic musical judgments can be verified unattended; generated captures and measured properties are evidence for subsequent human review.
First checks: 32 tests passed. Positional muted stored replay passed on Mac for 15 seconds with an invalid bench host: no GPU invoked, contiguous samples verified, zero output flags, 0.146 ms maximum alignment error. Overlay wrapper retained the real normal UI. Musical role labels remain unverified pending listening; runtime planner is experimental and not enabled by default.
Checkpoint a001dcb adds an opt-in generated-section bank. Cached candidates provide immediate road-event response; fresh GPU continuations replace role material only at phrase boundaries. These are generated sources, not prerecorded substitutes. Roles, inferred bars and harmonic coherence remain listening hypotheses. Current full Mac run uses this experimental mode; it must not be called a musical pass merely because the scheduler works.
First controlled results: strong anchors retained highly similar waveforms (verse/chorus correlation 0.823, verse/bridge 0.896); 22-frame anchors increased contrast (0.267 and 0.552). The common pulse estimator reports 92.3 BPM despite the requested 138 BPM, demonstrating tempo/meter ambiguity or prompt noncompliance. No true downbeat labels are available. A second pass will avoid imposing a potentially incompatible key on the reference and use chained generated context.
Measured worker startup after removing full second generation: 127.017 s (imports/input 3.795, DiT load 17.224, decoder load 6.448, first generation 95.367, decoder capture 3.918). This is a process-cold restart with existing compiler cache, not a post-boot measurement. Opt-in trusted local TinyJit snapshot experiment prepared; no default snapshot loading enabled.
39 unit tests pass, including sample-exact section landing, fresh-material replacement only on a phrase boundary, causal curve cooldown, arrival supersession and bounded late-packet skipping without timeline drift. These establish mechanics, not musical coherence. Mac system output reports muted with volume zero; selected device is built-in MacBook speakers. Bench ALSA exposes onboard MultiMedia1; RoadScore has not enabled or selected Bluetooth output.
Full Mac community result `normal_1789525740`: natural route exhaustion, 571.8 audio seconds, 21 fresh jobs, eleven fresh job IDs observed at played phrase boundaries, zero renderer underruns/emergency repeats, zero host starvation/late frames/output flags/sequence gaps. Minimum source buffer 10.05 s; max host alignment error 0.606 ms; max measured PCM transport delay 140.9 ms. All jobs satisfy logged causal input cutoffs. Score archived with decisions and song-form events. This clean run does not guarantee arbitrary network outages are solved; bounded recovery is unit-tested and would mark any concealed loss explicitly.
Stored replay of that score remains no-GPU and exact-sample verified. The normal UI overlay now uses the GUI render hook, native GPU textures and smaller ASCII-safe labels; a silent VFR capture uses observed UI-frame and audio-sample timestamps (`normal_1789526728/synchronized.mp4`). No acoustic validation claimed.
Snapshot restoration experiment: 62.532 s ready (22.292 s restore, 36.410 s initial generation, 0.759 s decoder preparation), peak host 1301 MiB vs ~175 MiB uncached worker. First-run comparison differed. Two identical steady-state restored requests match each other bit-for-bit; uncached steady-state cross-check is still required before enabling restore by default. Snapshot stays opt-in.
First full-run steering-proxy audit exposes late structural requests on some curves; this is not a complete before-steering pass. The next generic scheduler regression uses integrated *current model-predicted* turning angle to recognize substantial turns earlier, requires at least three seconds of predicted lead, and begins the preparatory fill on the next inferred beat while reserving the major landing for a bar. No route timestamps/labels/configuration are introduced. Offline proxy definitions and all unmatched/late cases are preserved.
**Snapshot validation failure and quarantine:** uncached steady-state requests match each other, but differ from restored steady-state requests. Therefore the 62.5 s restore is NOT an accepted startup path. Follow-up/style/text-only outputs made by that worker were moved under `results/overnight/rejected_graph_restore` on both hosts and removed from the primary review. They must not be used to judge musical capability. The normal-runtime eighteen-candidate screen and full Mac route are unaffected. Source inspection found this tinygrad revision does not advance `UOp.unique_num` during unpickling, and the worker allocated input buffers before restoring saved buffers; a narrowly scoped collision-prevention retry is prepared but not yet validated. Normal-worker regeneration is required.
Native curve regression `normal_1789527539`: warm launcher 15.016 s, 176.3 rendered seconds, zero output underruns/emergency repeats. Existing normal UI retained. On the unambiguous matched steering-proxy turn, request at 129.9 s, PRECHORUS waveform at 130.65 s, steering onset at 135.8 s, CHORUS landing at 137.8 s: preparatory material starts 5.15 s before the proxy. A second requested bridge falls beyond the captured window and is not counted as demonstrated. Native archival initially omitted the new form file; the general copy list is fixed and this completed run's exact preserved form metadata was recovered into its score archive.
Native arrival regression `normal_1789527744`: natural final-segment exhaustion, 254.3 rendered seconds, warm launch 16.502 s, nine fresh jobs, no output underruns or emergency repeats. All five active roles appeared (VERSE/PRECHORUS/CHORUS/BRIDGE/OUTRO), arrival triggered at replay t=239.146 s, and cadence/silence were captured. Minimum generation buffer 9.526 s; all job input cutoffs causal. Both newly generated native scores are now mirrored into the private Mac route-owned library; ordinary route data was not redownloaded.
Final normal-runtime followups regenerated successfully: seven chained candidates, six synthwave/funk candidates and three text-only controls. Their provenance ties each result to the normal worker log. Snapshot collision-prevention retry still failed: base/chorus/bridge became identical. The optimization is rejected, snapshot quarantined, normal worker restored. No musical evaluation should use the rejected outputs.
Second full Mac run `normal_1789528719`:571.8s,21jobs,seven fresh job IDs played,zero host starvation/late frames/flags/sequence gaps,zero renderer underflows/repeats. Minimum buffer10.283s,max alignment0.709ms,max transport123.3ms. Native stored replay `normal_1789529531` with no worker:contiguous samples verified,zero flags,no generation,3.375ms max alignment.
Audit exposed an exact-block-boundary preparation handoff defect: control could clear a due prechorus before the callback consumed it. Complete preparation+landing plans are now queued atomically once, with explicit arrival supersession. Two targeted callback tests added;48tests pass. Further full-route regression running before final acceptance.
Resident lifecycle first pass: explicit service reused by native replay,56.3s captured with zero underruns/repeats/output flags; launcher preserved service ownership. Stop succeeded and restored saved CPU settings exactly. Short capture remained `last_generated`; preferred176.3s demo was not replaced. Preparation was197.036s, not a warm-launch measurement: the replay joined while service was still preparing. The service's one-second safety loop spawned a1.3-second native Python import each time; changed to a cached native Params reader with fresh IsOnroad reads.49unit tests pass, including reader freshness. New preparation/profile and full queued-plan regression follow.
Resident preparation after cached Params reader:185.432s (imports3.690,DiT17.807,decoder8.755,first-generation148.449,decoder-capture6.256). This is slower than earlier127s process-cold measurements; no stable lower bound or post-boot claim. The change removes measured monitor overhead but does not establish the30–45s target. Third full Mac regression started only after READY, so its launcher time is a true resident-worker measurement.
**Final queued-plan regression `normal_1789530053`:**571.8audio seconds,naturalEOF,21fresh jobs,eightfresh job IDs heard,all11scheduledprechoruses reachedactualwaveforms. Zero host starvation/lateframes/outputflags/sequencegaps;zero renderer underflows/emergencyrepeats. Maxalignment0.525ms,maxtransport108.8ms,minbuffer8.926s;warmresidentlauncher23.568s. Median generation20.199s/21.920unique seconds(RTF0.921),hostpeak178.16MiB,trackedGPU1111.85MiB. Seven unambiguous community steering matches:three useful3.20/5.75/3.30s leads,two near-zero0.20/0.05s,two late1.50/1.15s. Mechanics fixed;prediction of steering entry and human visual timing are not universal passes.
Resident service stopped,ready markerremoved,CPU before/aftermatched. Final score mirrored privately tobench. Final Mac stored replay `normal_1789530685` used an unavailable compute address:40seconds,exactcontiguoussamples,zeroflags,no generation,maxalignment0.229ms. Silent VFR UI/audio capture41.022s saved;video and all listening media have no autoplay. Final native stored regression follows below.
Final native stored replay `normal_1789530751`:latest community score copied privately from Mac,15s,exact contiguous samples,zero flags,no generator,maxalignment2.250ms. Native ordinary UI restored(PID174039,no replay prefix). No owned worker/service/app/replay processes;no ready marker;realIsOnroad false. Worker exit before/after CPU snapshots matched. A subsequent live snapshot had big cores offline, matching native hardwared power-save behavior; we did not undo manager power saving. Initial cleanup assertion expected live equality too strictly; corrected evidence explicitly distinguishes restoration at exit from subsequent manager state (`bench_cleanup_final.json`).
Final Mac system output volume0,mutedtrue;both.session-mutedlocks retained. No speaker output intentionally enabled,no Bluetooth output enabled,bench remained muted. Public tree status identical to mid-run snapshot.49tests pass;review links/poster/media validated,all local,no autoplay. STATUS.md is authoritative for current results and specific remaining human/model/runtime limitations. No push/upload/PR,MusicGen untouched.
-44
View File
@@ -1,44 +0,0 @@
# Verified Future-Horizon Findings
Baseline: architecture investigation in the preceding conversation, StarPilot 64f8b75551196af9149ff92c47331ef3c3ff31a5. Model trajectories: 33 quadratic offsets 0–10 s; lane/edge positions 0–192 m without times. Longitudinal ego plan: 17 samples through 2.5 s; MPC lead trajectories: 13 through 10 s. navInstruction includes current guidance and up to three maneuvers; navRoute is event-driven full geometry without timed trajectory. Full rlogs required; modelV2/navRoute absent from qlogs. Mapd producer source is not available here; determine actual usefulness from logs.
Sources: selfdrive/modeld/constants.py and fill_model_msg.py; selfdrive/controls/lib/longitudinal_planner.py; cereal/services.py, log.capnp, custom.capnp; starpilot/navigation/navigationd.py and route_engine.py; tools/replay/replay.cc; system/loggerd/loggerd.cc.
# Implementation baseline
Private external workspace only. No uploads/push/PRs. Use cached original rlogs and existing LogReader/messaging. Replay historical model outputs; never run driving inference concurrently with SA3. Runtime detector/music process receives published events only and never reads route files. Historical logMonoTime controls availability; timestampEof + trajectory offsets establishes observation-relative prediction time. Never use end-of-file as arrival. Offline event selection may identify excerpts but not provide runtime triggers.
One event first: sustained model-predicted curve, then slowdown if needed. Predicted lateral acceleration = orientationRate.z * velocity.x. Require persistence ~0.3 s and avoid low-speed artifacts; initial 0.8 m/s² curve threshold, 0.5 release. Validate actual anticipation using later vehicle motion in offline evaluation only.
One exclusive Chestnut worker runs existing native SA3, fixed 30 s. Measured warm generation+decode 18.53 s; CPU prompt encode 3.76 s; continuation 19.34 s per ~22 new seconds; cold startup ~177 s. Cache conditioning, reuse ~8 s tail, prewarm, buffer ~52 s, request at 30 s remaining. Deterministic generated-tail loop if late. No finite buffer guarantees survival of a GPU hang. No model shopping or optimization absent integration blocker.
CPU Conductor: anticipation → event → release, brightness/filter/gain/rhythmic envelopes applied to locally generated music. No steering-to-pan or speed-to-tempo. Six-second-ahead curves cannot depend on new SA3 output. Navigation can influence later composition only when already available with sufficient lead (roughly 30–52 s under buffer policy). No precise neural cadence/tempo/stem promises.
# Order and acceptance
Inventory all three routes lightly, choose defensible curve, get sequential on-device replay and generated audio audible early. Validate >=2 s anticipatory response, no future leakage, repeatability, timestamps. Local git checkpoint. Then continuous generation, late-job fallback, synchronized local UI/video, navigation/reroute, arrival, slowdown, polish. Prefer existing replay executable; if unavailable use narrow paced LogReader publisher. Isolated messaging prefix. No production changes.
Test future-mutation invariance, missing/invalid/stale messages, persistence, low speed, source ages, audio continuity/clipping/deadlines, reroute invalidation, arrival independent of file ending. Ten-minute rehearsal target. Standalone local debug UI acceptable. Record actual results and failures in STATUS.md. Keep best known runnable checkpoint.
# Empirical implementation notes
- The bench has no compiled tools/replay/replay executable. A narrow original-message publisher uses existing LogReader/cereal instead. One-segment prefetch avoids a measured ~1.2 s segment-load delay; audio result resampling occurs in a background thread.
- Recording commit c3e4ec630f41c4baa43254a90f718abd1bf764a1 parses with the checkout schema. carState.yawRate is zero in these recordings; offline physical verification uses steering and video, never that zero field.
- Route 202 is the primary bend demonstration. Route 201 has four valid route-geometry updates; route 203 supplies arrival validation.
- Camera frame zero is ~1.451 s after route-202 log origin. The selected video export accounts for this offset; frame-count agreement alone did not establish temporal alignment.
- First curve anticipation occurs at 135.275 s; sustained steering onset is ~141.53–141.58 s depending on sampling, ~6.26 s later. Runtime has no copy of the steering-evaluation timestamp.
- Locally encoded approach/closing prompts are cached so repeated CPU encoding does not consume the 2.7 s continuation throughput margin.
- Bench output is now muted by user instruction; rendered audio is recorded before zeroing speaker samples. The capture is not acoustic evidence.
- Arrival prototype uses the navigation producer's creeping-speed scale (<2 m/s), not a strict parked-state threshold. A strict <1 m/s rule missed route 203 before nav invalidated at arrival. Valid arrival-zone context held for 3 s triggers a 6 s fade. Harmonic cadence remains unproven.
- A long run sustained 575.8 s, 23 accepted continuations and no output flags. Rejecting stale navigation-conditioned work can intentionally require a generated-source loop even while the worker is faster than playback.
- Bound a continuing curve peak to its initial estimate ±2 s and matching direction, avoiding indefinite drift into a later bend. Exact physical-apex alignment remains a listening/video tuning task.
- Audio sample counts alone have a callback-boundary ambiguity relative to replay. Final evidence records callback monotonic wall time against replay origin; it does not claim acoustic output timing.
# Current continuity pass (supersedes prior quality preference)
Human review rejected Nocturne and the earlier endpoint trimming/level-match approach. Horizon is the primary source. The private experiment now uses the existing inpaint mask to preserve active four-second context at both ends of a 324-frame window. The trailing anchor is a disclosed recurring fragment selected from already-generated opening audio. Chestnut generates the ~21.92-second interior; ~4.09 seconds are reused identity material. This is a hybrid composition strategy, not 26 seconds of wholly novel output. No silence repair, stretching or route-specific soundtrack selection.
The four-job raw Horizon probe produces 125.945 seconds including its initial source. Warm jobs take 19.27–19.43 seconds: useful playback RTF .741–.747 and unique-material RTF .879–.886. This supports live validation, not subjective acceptance or an indefinite performance guarantee. Compare actual envelope plots and listen across disclosed boundaries in results/continuity/listen.html. Original failed artifacts and tag continuity-before-865e43d are preserved.
After the raw probe, validate the identical runtime policy muted on the existing curve, arrival and a second route. Keep the detector unchanged. A small prompt-embedding blend supplies establish/explore/develop/build/release intent; it cannot guarantee a perceptual trajectory. Arrival investigation found reverse at ~232.2 s and park at ~255.1 s. Do not equate the first reverse with the exact ending: require recent destination context plus sustained low-speed braking reverse and subsequent navigation invalidation, or a verified stop/park fallback. Continue the generated closing while that maneuver begins.
Normal onroad replay integration and live modeld coexistence remain later milestones. Do not extend native_display. This pass runs without display takeover and packages the existing timestamp-aligned road video for private listening.
# Dense musical pass
Human review provisionally accepts the e3477e2 sparse Horizon continuity and likes its cadence. Preserve the active-context/anchor method. Stress-test one energetic Horizon Drive identity with percussion/bass/layered-arrangement prompts; no alternate models, training, route-specific tuning, onroad UI or production work. First unchanged dense stress run completed 210.2 seconds with seven jobs and zero flags/loops. Then use explicit arrangement-stage prompts, a modest generated-transient reprise vocabulary, and a 0.8–3.5-second pulse/release-aware runway into the same cadence. The causal vehicle trigger and road detector remain unchanged. See DENSE.md and results/dense for the new gate and disclosed anchor-entry timing flags.
-53
View File
@@ -1,53 +0,0 @@
# Latest human baseline and dense test
Human review provisionally passes e3477e2's sparse Horizon continuity and harmonic development, and likes the final cadence. It requests a denser rhythmic stress test and a less abrupt cadence entry. This supersedes the older rejection notes below where they refer to the improved anchor architecture. Current pass: [DENSE.md](DENSE.md), [STATUS.md](STATUS.md), `results/dense/listen.html`. No new subjective acceptance is claimed.
# Current listening pass supersedes the numerical preference below
Human review rejected Nocturne, earlier-endpoint trimming and level matching as a continuity solution. **Horizon is now primary.** See [CONTINUITY.md](CONTINUITY.md), [STATUS.md](STATUS.md) and `results/continuity/listen.html` for the new two-sided inpainting experiment, disclosed recurring anchors, live captures and performance. No subjective musical success is claimed. The remainder of this document is preserved historical evidence from the rejected pass.
# Musical-quality pass
Preserved baseline: local tag `quality-baseline-149c3cb`, original `results/final_timed`, curve/arrival videos and sustained run. Repeated baseline replay before integration changes: 100.1 s, three accepted continuations, zero fallbacks/output flags, 6.218 s rendered lead. Evidence: `results/quality/baseline_replay`.
Human feedback: causal curve felt anticipated; musical quality not accepted. Generic identity, apparent cutouts/disjoint continuations, excessive volume swell, and premature fade ending require improvement. Numeric continuity is not subjective acceptance.
## Findings and bounded changes
The original final_timed dry capture contains 22.8 s below -50 dBFS (100 ms RMS windows), versus 0.2 s in the new 100 s style-switch run. Original generated WAVs themselves contain long near-silent passages; the Conductor and PortAudio were not their primary cause. Several passages fade before the retained tail ends, so continuation inherits an ending rather than ongoing musical context. A matched-seed fresh test reduced near silence from 2.6 s to 0.0 s by conditioning for a longer piece while keeping the 30 s compute shape. Continuing from the old faded tail still produced 6.8 s. These results establish an integration problem; they do not explain every model-internal pause.
The narrow chamber experiment retains ~8 s of context but ends the usable window at latent frame 301 (~27.96 s), before the troublesome final tail. The next continuation uses exactly that same endpoint. Crossfade duration remains two seconds. Three new passages contained 0.2 / 0.0 / 0.0 s below -50 dBFS in their usable new regions. Context RMS matching corrects independent normalization; silence is never amplified as a reference. Shared-prefix correlations in the exploratory sequences were positive, so broad phase cancellation was not the dominant gap mechanism.
Tradeoff: each trimmed window contributes 19.969 s instead of 22.012 s. Generation is roughly 19.4–20.3 s, depending on simultaneous display/I/O load; useful RTF straddles 1.0. Do not reuse the old 0.878 RTF claim for this quality mode. A matching prewarm restores ~48 s startup buffer. Fallback repeats the current generated tail to preserve context rather than substituting the original seed. It waits until two seconds remain; the first combined run's five-second threshold inserted an unnecessary loop shortly before completion.
Four identities were each tested with initial/base/development/closing material: Nocturne (chamber), Horizon (post-rock), Orbit (analog), Canopy (organic ambient). These are four candidates, not four accepted strong styles. The chamber direction was strongest numerically. Untrimmed 96 s sequences contained 5.6 / 11.7 / 25.7 / 14.0 s below -50 dBFS respectively, including intended closing sections. Orbit and Horizon remain poor continuity candidates in these tests. Nocturne is the default; user listening must determine actual distinctiveness and musical merit.
The causal detector is unchanged. A small alternative Conductor keeps more of the original full-band material and adds delayed phrase echoes, with much less gain/brightness modulation. Long-range navigation still selects development/closing conditioning for later material. This does not prove arrangement, harmony or exact rhythmic control. No new model, training, weight changes or model optimization were performed.
Arrival now separates preparation from ending. Recent valid arrive guidance within 40 m arms the system; speed below 0.15 m/s for two seconds triggers a final gesture. Context can survive navigation invalidation for 30 s, but a new valid route clears it. Sequential evaluation triggers at route 255.257926352 s, versus the old fade at 234.505508006 s. A five-second source-informed final sonority estimates pitch classes from music already played. It is a candidate resolution, not a guaranteed tonic cadence. EOF cannot trigger it; an already-started gesture may finish after replay ends.
The runtime style selector affects future requests. In the verified 100 s run, a manual test request at 146.026 s changed the playing identity at 175.927 s. Zero underruns/fallbacks. Audio source cutoffs and muted-output assertions passed. These manual test times are not runtime road-event triggers.
## Listening
Open `results/quality/listen.html`. It contains matched-seed duration examples, four style candidates, same-source Conductor A/B, continuation-level A/B, and a short arrival A/B. All files remain local/private and require a click to play. Final integrated review files will be listed in STATUS.md.
## Display investigation
The repository's `./onroad` is a host-worktree launcher. The bench has no built replay executable. Full normal-UI replay additionally needs camera VisionIPC and appropriate UI services; that was not made a dependency. The private fallback uses the already installed pyray and OpenCV to show the timestamp-aligned road video and actual RoadScore/model decision state. It does not draw invented lane/path geometry or run the driving model. The original UI manager is paused temporarily while its display process is replaced, then resumed; no production files or onroad parameters are changed. Initial tests exposed a pyray pointer conversion and an incorrect large-screen assumption; both are recorded failures, with corrected final testing required.
## Live integration checks
The first combined display/music run had one early fallback and no underruns. A matching-style prewarm plus corrected buffer policy produced a second 100.1 s run with zero fallbacks/underruns, three accepted continuations, 6.209 s rendered anticipation lead, and a verified future style change. Its dry near-silence total was 0.2 s, wet 1.0 s, compared with original dry 22.8 s / wet 24.9 s. Different prompts/seeds mean this integrated comparison is not a controlled isolated musical preference test; matched-seed and same-source A/B files are provided separately.
The full live arrival run triggered at 255.258 s, held the final gesture beyond log EOF, and captured 90.6 s. It had zero underruns and one background source-loop extension after the ending had begun; that unnecessary source work was then removed. The ring-only phase now advances the audio clock without fetching more source or requesting new generation. A final short arrival regression and a final cold native-display run verify this change and the corrected small-screen layout.
Musical limitations remain: no isolated stems, guaranteed beat grid, verified tonal center, precise cadence, or accepted four-style lineup. The deterministic final sonority can be stylistically unlike its source. Echoes may feel like an effect rather than genuine arrangement development. Long-term quality and source activity remain model-dependent. None of these are hidden by the zero-underrun metric.
## Final acceptance
The final cold native-display run passed: 100.1 s, four live continuations, zero output flags/fallbacks, zero near-silent 100 ms windows in both dry/wet captures, and 6.2108 s anticipation lead. Pinning presentation to cores 0–3 restored median 19.284 s generation / 19.969 s usable material (0.966 RTF). This is a narrow throughput margin, not an indefinite-playback guarantee. The original UI and CPU settings were restored; RoadScore processes were stopped. Final detector/musical tests passed.
The corrected 536×240 native layout is visible in `results/quality/native_display_final.png`. Active software frame age was median 73 ms / p95 115 ms / max 379 ms; no future frames. Post-EOF last-frame hold is recorded separately, not counted as live synchronization error. These are software timestamps, not photon/acoustic measurements.
Best integrated reviews: `results/quality/curve_review.html` and `results/quality/arrival_review.html`. The final arrival regression captured 35.8 s, zero output flags/loops and the complete final gesture. It used prewarmed source; the longer 90.6 s arrival run separately verifies live closing generation. Subjective musical acceptance is still yours.
-144
View File
@@ -1,144 +0,0 @@
# RoadScore at COMMA_HACK 7
RoadScore turns causally delivered road context into a locally generated adaptive score. ACE-Step 1.5 turbo is the pretrained base model; **Prism** is the primary prepared musical profile and Aurora is an explicit alternative.
Our work includes the native tinygrad Chestnut port, continuation and buffering, pre-playback quality checks and bounded rerolls, immediate deterministic gestures, replay integration, UI, and private score archives. We did not train ACE-Step.
The pipeline is:
recorded cereal → semantic road state → section requests → ACE on Chestnut → quality checks → PCM arrangement and gestures → output → route-owned archive
Only already-delivered messages enter musical decisions. The ordinary replay engine owns route resolution, camera video, path/lane display, and playback timing. Existing generated music may be held when fresh material is unavailable; holds are recorded and are not counted as new generation.
## Demo and source guide
- [Interactive saved showcase](MAC_SHOWCASE.md): `./onroad --roadscore route1 --demo --fullscreen` on the Mac starts synchronized native replays with Galaxy controls and comma audio. Saved demos need no model warm-up.
- [Demo explanation and overlay legend](DEMO.md): what the audience sees and what each state means.
- [Event evidence](EVENT.md): measurements and preserved failures on the event hardware; read the latest checkpoint.
- [Status history](STATUS.md): event summary followed by explicitly historical pre-event results.
- `prototype/overlay.py` and `prototype/overlay_view.py`: native overlay integration and read-only presentation.
- `prototype/`: score runtime, composition, gestures, transport and archival code.
- `tools/`: development and event utilities; [offline UI preview](DEMO.md#offline-ui-review) needs no device.
- `tests/`: isolated UI tests. Existing runtime tests remain beside their modules and in `experiments/`.
- `experiments/`, `chestnut_music/`, `chestnut_stable_audio/`: research and historical backend work.
- `routes/`, `results/`, model assets: private local data, excluded from Git.
## Event status
Source lives in this repository on the `RoadScore` branch and on the event comma at `/data/openpilot/roadscore`. `/data/roadscore` is a compatibility symlink. See [EVENT.md](EVENT.md) for current measurements and preserved failures.
The latest recorded checkpoint in [EVENT.md](EVENT.md#first-clean-event-native-replay--45-w) reports a complete 254.1-second muted native Prism replay at an explicitly selected 45 W: six accepted fresh jobs, no holds, no underflows and no worker failure. This is one instrumented pass, not a production-reliability claim. Live/modeld coexistence and physical speaker/Bluetooth validation remain pending. Earlier 30 W results and failed runs are preserved in that report.
## Run on the event comma
```sh
cd /data/openpilot
python3 roadscore/prototype/worker_service.py start --composer ace --profile prism
python3 roadscore/prototype/worker_service.py status
./onroad --routeid '<dongle>/<route>' --roadscore --muted --duration 3600
```
Wait for the resident worker to be ready before replay. Initial preparation can take about ten minutes; warm throughput is a separate measurement. The duration is a watchdog, and recorded route EOF ends normal playback. ACE is now the default composer. SA3 requires explicit selection and separately staged assets; it is not currently a verified event fallback.
The ACE worker defaults to the tested 30 W cap. An explicit `AM_POWER_LIMIT` override is experimental and must be recorded with performance results. Set `ROADSCORE_DEVICE` or `--bench` for host-side device selection. Stop only the owned worker with `worker_service.py stop`.
Automated sessions must use `--muted` and keep `.session-muted`. This restriction does not change the product's eventual normal audible behavior. Attended listening needs a deliberately selected output; do not infer Bluetooth or speaker validation from captured audio.
## Private data
Routes, model weights, generated audio, result logs, and local fixture settings are ignored by Git. Do not publish them. Historical source on the public branch exposed real route identifiers; those identifiers have been removed from the current working files, but existing public history is unchanged. Local redaction does not change comma Connect sharing permissions.
Historical fixture helpers require private environment settings instead of real IDs in source. These helpers are not the arbitrary-route runtime. Keep any `fixtures.private.env` local.
<details>
<summary>Historical pre-event notes (superseded, not event validation)</summary>
## Historical prototype notes
The older instructions and measurements below are preserved for history. Their Horizon identity, special excerpt launchers and early replay limitations are superseded by the current `./onroad` path above.
# Latest listening gate: Horizon Drive
Run `./review_roadscore.sh` for the three primary audio tests: dense live continuity, energetic live curve, and phrase-aware live arrival. Timing/anchor markers are optional, and synchronized video links are labeled separately. See STATUS.md and DENSE.md. The bench remains muted by default; all work is private under this Desktop folder.
Earlier sections below are historical. The active-context/anchor system is preserved; the new identity and cadence runway are the current listening candidate.
# Current review: Horizon continuity
Run `./review_roadscore.sh` to open the new private listening lab. Generation boundaries and reused four-second identity anchors are marked. See STATUS.md and CONTINUITY.md for current measurements and limitations. The bench default is Horizon with rolling inpainting; the speaker remains muted. Musical acceptance is pending human listening.
The sections below describe the prior prototype and remain for command/reference history; older Nocturne preference and trimming claims are superseded by the current status.
# RoadScore — private bench prototype
Everything here is local/private. No remotes or uploads are configured. The public StarPilot checkout is unchanged.
## Run the current musical-quality demo
From this directory on the Mac:
```sh
./run_roadscore_onroad.sh
```
Requires the already-staged bench at `comma@192.168.3.111`, its existing Chestnut and local SA3 assets. No route downloads or dependency installs occur. Default: route 202, seconds 110–210. The recorded drive appears on the comma display with a small RoadScore overlay. Open `http://192.168.3.111:8088` during playback to select a future musical style. `./run_roadscore_demo.sh` runs without taking over the display. Cold preparation takes roughly 2–3 minutes; an existing healthy worker is reused.
**The bench speaker stays silent by default.** Actual rendered music is saved to `results/latest/heard.wav` after the run. Open that file locally when ready to listen. `results/roadscore_causal_demo.mp4` is a preserved synchronized video/audio demonstration of the successful 100-second run.
Other excerpts (numeric start/duration are replay selections, never runtime event triggers):
```sh
./run_roadscore_demo.sh <private-route-name> 170 87
./run_roadscore_demo.sh <private-route-name> 0 600
```
The native display and live video panel currently support the default route-202 excerpt only; it hides video for other routes rather than showing unrelated footage. Audio/event handling works independently of video.
## What is live and what is recorded
The dedicated replay process alone reads rlogs and republishes original cereal messages at original timing. The runtime consumes those messages. It does not read offline event selections or future route outputs. A Python audit guard additionally denies runtime route-file opens. Future-mutation tests cover the actual selected route's event detector.
Chestnut generates SA3 continuations locally in a single worker. The unchanged curve detector controls a lighter brightness/level envelope plus phrase echoes. Navigation can bias future generated development/closing passages. It does not claim that SA3 newly composes in response to a six-second-ahead curve. Navigation can choose cached broad approach/closing conditioning for later blocks; exact keys, instruments, tempo and cadences are not controlled.
In the current musical mode, valid `arrive` guidance within 40 m arms an ending; speed below 0.15 m/s for two seconds triggers it. Context expires after 30 s and resets on a new valid route. The final sonority uses estimated harmony from music already played; it is not a guaranteed tonic cadence. An already-triggered gesture can finish after log EOF. EOF itself never triggers arrival.
## Files and evidence
- `PLAN.md`: architecture baseline and implementation deviations.
- `STATUS.md`: actual test results, limitations and checkpoints.
- `prototype/`: narrow publisher, runtime, model worker, tests and local server.
- `routes/`: immutable cached original route artifacts and preflight validation.
- `results/`: saved renders, traces, model timing, comparisons and demo video.
- `chestnut_stable_audio/`, `chestnut_music/`: prior isolated feasibility work; MusicGen unchanged.
On the bench everything is under `/data/roadscore`, with pretrained SA3 files reused from `/data/sa3-feasibility`. Existing openpilot modules are imported read-only. The worker temporarily enables capped big CPU cores while offroad and restores the previous settings when stopped by its wrapper.
## Restart/failure behavior
Each run archives the previous `/data/roadscore/results/current` directory. The supervisor refuses duplicate invocations and reuses only the specific existing SA3 worker. Failed/late generation extends generated audio rather than blocking the output callback. Errors and fallback counts remain visible. A new run ignores older job IDs. Seek/pause/speed changes and live onroad integration are outside this prototype.
The output WAV records the rendered callback buffers before muting; it is not a microphone recording of the speaker. Physical acoustic output was not subjectively evaluated during unattended testing.
## Review without the bench
Open `results/curve_review.html` or `results/arrival_review.html` locally. Press play to hear the saved render; neither page auto-plays sound or makes network requests. The adjacent MP4 files also play directly. Keep these private because they contain your road footage.
Tests use the existing bench Python environment (the Mac system Python does not include scipy):
```sh
ssh comma@192.168.3.111 'PYTHONPATH=/data/roadscore/prototype:/data/roadscore-feasibility/venv/lib/python3.12/site-packages /usr/local/venv/bin/python -m unittest discover -s /data/roadscore/prototype -p test_core.py'
```
## Musical-quality listening pass
Open `results/quality/curve_review.html` for the best combined demo, `results/quality/arrival_review.html` for the later ending, and `results/quality/listen.html` for short A/B files and four distinct style candidates. Nocturne is the default because it was strongest numerically; the others still need listening and have known continuity problems. See `QUALITY.md` for controlled experiments and `STATUS.md` for the latest integrated evidence.
The native presentation uses the bench’s existing pyray/OpenCV and temporarily hands display ownership back and forth with the original UI manager. It shows actual road video and RoadScore/model decisions; normal onroad lane/path rendering is not integrated. The original UI resumes on exit.
Style selection affects future generation, so expect roughly 30–50 seconds before a request reaches playback. The current piece continues in the meantime. All four seeds and prompt embeddings are already staged locally.
Quality mode keeps the 30 s model shape, conditions ongoing passages for a longer piece and uses a ~27.96 s active window with ~8 s retained context. That produces ~19.97 s of usable new audio per ~19.4–20.3 s job in the initial combined tests. The final display-pinned run measured 19.28 s (0.966 useful RTF). This remains near real time with a narrow margin; generated-tail loops remain necessary protection for longer or slower runs. No guaranteed indefinitely loop-free claim.
Baseline rollback: tag `quality-baseline-149c3cb` and the original `results/final_timed` artifacts remain untouched. Only local checkpoints exist.
</details>
-31
View File
@@ -1,31 +0,0 @@
# Replay showcase operator flow
Scope as of September 20: one showcase route, current ACE/Prism, replay only. Freeze live-car work and unrelated feature development. Preserve the gold excerpt and accepted soundtrack archives. The operator is back at the event. Hardware inspection is allowed for the designated owner; current access must be verified before claiming readiness. No device behavior should be inferred from the local rehearsal recording.
On the event comma, while offroad:
```sh
./onroad --roadscore route1
```
The private local favorites file resolves route1 to showcase E and route2 to backup A. No private route identities belong in this document. Normal launches use a fresh logged session seed. For this staged showcase, the user permits choosing and fixing a seed with `--roadscore-seed <seed>`; the same override supports reproduction and diagnostics. Current native operation uses prepared generic role conditioning and fresh ACE sampling, not a fresh semantic-planner pass per launch.
Mac seed auditions use `roadscore/experiments/ace_chestnut_20260916/audition_cached_initial_mps.py` with the matching native conditioning bank, exported weights and noise policy. They shortlist initial passages only. The verified baseline closely matches native PCM but is not bit-identical across backends. The selected seed still needs a complete native listening rehearsal; an accepted opening does not establish continuation or full-song quality. Audition audio and provenance stay in the Desktop rehearsal package.
In Galaxy's RoadScore section, **Simulate disengage** gives the replay UI its native blue AOL appearance and contains the music; **Simulate engage** restores enabled UI/music. **Recorded engagement** resets that selection. **Left signal**, **Right signal**, **Signals off**, and **Use recorded signals** separately drive native arrows and the existing rhythm-aware shaker. The controls change only the isolated replay UI view and music: no recorded messages, vehicle controls or real Params are rewritten. They follow acknowledged audible-time state. Restore both selections independently. Device code checks passed; an audible/visual operator rehearsal remains required before judging. Proposed engagement sequence: contain at0:10, open at0:15, recorded at0:19.
The next presentation revision uses native **Steer Left/Right** prompts labeled **Replay simulation**, with existing alerts taking priority. Conservative-v4 strengthens the engagement contrast and mixes the shaker after containment so its high frequencies remain audible. Manual direction changes can restart the motif after a one-bar cooldown. Stale inputs and uncertain beat timing still suppress percussion; a button press is not proof that a shaker pulse rendered.
For this explicitly staged showcase, an optional private `assets/showcase_curve_plan.json` identifies measured route events and the exact replay start. Only a matching native fresh v4 replay may apply it; live input and official judging ignore it. It shapes the existing song's bass during the approach and restores full bandwidth at the next suitable eighth-note boundary around the measured apex. The original composer state is unchanged. The plan and its hash remain in the captured archive. Ordinary routes without a matching plan retain their causal conductor timing.
Use the protected recorded fallback if preparation or hardware is unreliable. The local rehearsal package is `Desktop/RoadScore/demos/showcase-rehearsal/`; `SHOW DEMO.command` opens the route E presentation cut over matching camera footage: 5:00.35, unchanged soundtrack until a two-second closing fade. The unedited exact soundtrack remains protected. The full judge briefing, evidence inventory, timestamp script, and cheat sheet are in that package. This is recorded playback, not fresh generation. The original 315.4-second recording includes an approximately 15-second historical EOF tail; the source app ended with an input-stall error despite uninterrupted audio. It is not the latest device run and has neither native UI nor the newest presentation refinements.
Do not infer audible accents from source alerts alone. In the locked soundtrack, the first strong signal motif begins about 0:24.6, curve payoffs occur around 0:52.7 and 2:01.3, and recorded engagement switches at 2:10.8 and 2:15.9. Fresh generations may have different beat phase. Let the music establish and leave silence around those demonstrations.
If anything fails, switch to the labeled recording or the approved 30-second stopped-motion comparison; do not change vehicle controls, model settings or power limits during judging. Current live-car readiness is not established. Physical Bluetooth synchronization and new replay-button operation still need an operator check. The inherited-session-lock fix is installed; the recovered worker must cold-start once, and subsequent canceled sessions should release their receiver lease without losing the resident worker.
## Presentation
Lead with: “RoadScore makes the drive part of the music.” Explain the model/runtime split before playback, then leave silence for the melody, signal motif and curve payoff. The core line is: “The model composes ahead; the presentation layer reacts now.” Use the local SPEAKING_CARD.md and TODAY_RUN_SHEET.md for the timestamped presenter flow.
Advance knowledge is allowed for this staged showcase. Keep the recorded source events real and disclose manual music simulation. Do not change ordinary fresh-seed behavior or official judging policy to make this presentation repeatable. A route-specific demonstration does not prove arbitrary-route or live-car readiness.
-140
View File
@@ -1,140 +0,0 @@
User acceptance update, 2026-09-19: the user reports the converted-video standalone demo worked and requests it be protected as the demo baseline. Preserve that run before any musical or presentation changes. The comma became unreachable before its exact run ID, completed archive and metrics could be retrieved; archival verification remains pending. Next requested work is Bluetooth timing and the ending. The installed Galaxy tool measures tap offset but currently applies only a delay to displayed music cues, not global replay/audio synchronization. Route E has all eight original and converted videos locally; route A has original HD video; other community routes are not provisioned on the comma.
2026-09-19 current standalone status: the integrated HD demo is not yet passing. Native ACE resident reuse passed same-seed PCM/latent equivalence and fresh-seed variation without model reload or graph compilation. First accepted audio took 16.8–17.3 seconds; 110 seconds of accepted buffer took 91.3–92.1 seconds at 100 W. The normal native command now selects resident preparation automatically.
JLab Bluetooth output started during user-operated replay. The latest run produced 36.4 seconds of audio with no underflows, but HD replay stalled at the first segment boundary and the synchronization guard stopped it. Initial cache/frame priming fixed the earlier startup stall. A separate native decode fixture reproduces only 17.2–17.4 fps against the required 20 fps; sustained HD decoding is the current blocker. Decoder changes require measured frame equivalence and throughput before another user test. The warm worker, generated music, original gold and fallback assets remain preserved. User owns the next replay launch.
Galaxy RoadScore settings and tap calibration exist only in an isolated local branch and are not installed on the device. They require adaptation from PulseAudio to the device's BlueALSA output. They are deferred behind the replay fix.
Mac archived replay passed actual UI capture and exact final-sample playback against a complete historical archive. Presentation passed 33 offline tests; an actual 27-second opening gives a coherent estimated beat after removing the former 30-second minimum. Native alert accents are sparse, beat-aligned and suppressed during competing cues. Subjective cue salience still requires listening.
# Event checkpoint — clean 45 W native replay
Prism completed a full 254.1-second muted route on the event comma + Chestnut with six accepted fresh music jobs, no holds, no underflows, no worker failure, and clean causal timing. Camera, path, lanes, navigation and ten curve activations were recorded. See [event evidence](EVENT.md). This is an instrumented replay pass; physical listening, Bluetooth and modeld coexistence remain unverified. Community judging has not begun.
# Event checkpoint — RoadScore branch
Current event work and exact failures are recorded in [EVENT.md](EVENT.md). Source is migrated. Official Chestnut validation and 30 W Prism preparation pass; native replay still has unresolved timing/ending-context failures. A 45 W fixed-decoder comparison passed 40/40 and full-model testing is starting. No event-baseline or judging-batch pass is claimed. All automated audio remains muted; no files have been pushed by this work.
Everything below describes preserved **pre-event** evidence, not the current event unit.
---
# Current result — Prism demo hardening from 08ba22e
**Prism is the primary native demo candidate; Aurora is the verified selectable backup. Circuit is archived.** The requested instrumented regressions pass. The intermittent GPU/link fault remains unresolved, so this is not a production-reliability claim.
- The final pre-commit gate catches committed quiet gaps before playback, preserves soft passages/intentional outro fades, and permits at most two same-role deterministic rerolls within the buffer deadline. Known bad native/reference cases and real route rerolls are preserved.
- Thirty native continuations produced 840s of new music in 694.97s (0.827 RTF); all remain accepted by the final policy. Final-policy full curve replay measured 0.888 RTF including qualification/persistence, with 64.8s minimum buffer and no two-second quiet span.
- Full native Prism arrival, curve and community routes passed normal UI/camera/path/lane/causal checks without underflows, emergency fallbacks or output flags. Aurora completed a full 254.2s route and archived correctly. Its startup exposed screen-off CPU interference; the private CPU-only supervisor fix subsequently corrected that transition automatically during SA3 preparation.
- The first composer-stop test exposed callback blocking. After the fix, four accepted-music holds completed with zero output errors and truthful DEGRADED status. No reset or model handoff occurs during playback.
- The smaller fixed-chunk decoder reproducer localizes genuine link loss to after submit and before the driver timeout handler. Compute-only also fails; a 40-repeat idle comparison passes. Root cause is still unknown.
- Mac fresh retest, SA3 and Mac/native stored replay pass. The first Mac fresh test missed deadlines during a 1.138s transport burst; retain that operational risk and prefer native playback for the demo.
- Curve Apex V4 and the conservative bridge are available in the focused review. Human musical/salience judgment and live modeld coexistence remain unverified. Twenty focused tests and the existing regression coverage pass.
[Evidence report](ACE_DEMO_HARDENING.md) · [Focused listening review](results/ace_demo_20260916/index.html) · [Launch/recovery checklist](results/ace_demo_20260916/DEMO_CHECKLIST.md) · [Validation](results/ace_demo_20260916/validation.json).
**Clean shutdown verified:** owned workers/replays stopped; locks free; normal offroad UI/manager and saved CPU settings restored. Mac and bench remain muted, both session mute locks are present, and no speaker or Bluetooth output was intentionally enabled. Public StarPilot is clean; MusicGen is untouched; route/audio artifacts remain private/local.
---
The records below are historical and are superseded by the current evidence above.
# Historical result — ACE stability pass from b719139
**ACE remains the strongest composition candidate, but the full stability/continuity gate has not passed.** Read the [current report](ACE_STABILITY.md) and [Mac/Chestnut listening review](results/ace_stability_20260916/index.html). Earlier checkpoint claims below are superseded by this pass’s evidence.
- Deterministic Mac/native harness, three new identities at30/45/60s, five identical native repeats, same-latent VAE tests, layer/step diagnostics and the15–60s shape sweep are complete.
- Fixed45s continuations retained555s of new music in454.17s compute (RTF0.818). Two20-job soaks passed. Three complete native routes reached EOF with39accepted fresh jobs total and zero underflow/fallback/hang. Native stored playback and a headless Mac fresh ACE transport retest passed.
- Aurora and Circuit each completed six linked native sections, with exact prefixes and at most0.1/0.2s measured quiet. Their prepared bundles are available; human approval and full-route tests for these two profiles remain pending.
- The terminal-fade guard repairs the earlier flow, but full Prism routes exposed5.1s and2.7s interior gaps. Both reproduce with identical inputs in the official reference. Seed/role/prefix/preservation/refreshed-timbre counterfactuals did not provide a reliable real-time fix. **Musical continuity remains failed.**
- Three56s decoder/runtime failures are preserved, including two VAE-only probes at2.470GB allocation. The latest captured an unhealthy link read before process teardown. Recovery restored the bench; initiating cause remains unresolved. **Do not claim production reliability.**
- V3 signal/curve gestures are rendered and tested; perceived salience and subjective port fidelity require human listening. Exact reconstruction of the old approved performance is blocked by missing original random preparation state; honest recovered-code/new-boundary A/Bs are supplied.
- Cold ACE buffer preparation took about12.6minutes. Warm throughput does not imply instant readiness. No live driving/modeld coexistence claim follows from offroad replay.
Fresh native SA3 fallback passed:116.6s captured, three jobs atRTF0.776–0.780, zero underflows/fallbacks/output flags. All owned workers/replays have stopped; the normal offroad bench UI is restored and CPU restoration is verified. Mac and bench remained muted; no speaker or Bluetooth output was intentionally enabled. Public StarPilot is clean; route/audio files remain private/local. MusicGen is untouched.
---
## Previous checkpoint record (b719139; historical)
# RoadScore — ACE Chestnut result
**ACE now generates and decodes music locally on Chestnut faster than playback at useful short horizons.** Keep it as the primary experimental composer; SA3 remains the working default/fallback. The prepared-identity continuation loop needs no Mac during generation. Arbitrary new identity preparation is still a Mac development step, not a demonstrated comma CPU capability.
[Start with the listening review](results/ace_chestnut_20260916/index.html). It includes native music, a142s verse→prechorus→chorus→bridge→outro, arrival, distinct V2 gestures, and the actual native road score. Nothing autoplays. Human approval applies to the earlier Mac ACE reference; **native continuity, endings, and gesture salience remain unapproved until listening**.
| Measurement | Result |
|---|---|
| FP16 generation + native decode |15s in7.42s;30s in12.55s full decode or15.39s bounded decode;45s in22.52s |
| Warm continuation |28s genuinely new material in20.36–20.97s isolated;21.57–21.81s during native replay |
| Longer horizons |60s takes80.07s;90s latents decoded only after recovery, so no complete90s RTF claim |
| Memory |~4.215GB tracked GPU peak; composer~471–493MiB host peak/RSS, plus compiler helpers and replay/audio/UI |
| Cold readiness |447.8s including initial58s score and both graph shapes; preparation must precede playback |
| Native ACE replay |Full254.3s arrival route,7 accepted fresh sections, minimum buffer8.1s; zero fallbacks, underflows, or output flags; camera/path/lanes/nav/10 curve activations verified |
| Arrival |41.2s of generated outro heard before cadence; weak harmonic confidence used source-tail release, not an invented chord |
| Mac fresh ACE |Composer kept up, but PCM transport failed clean timing:46 starved callbacks /21,257 late samples. Preserved as a failure, not a pass |
| Stored-score regression |Mac120s: exact samples, zero flags,0.604ms alignment. Native ACE120s: exact samples, zero flags,8.104ms |
| Fresh SA3 fallback |Mac120s,3 new sections atRTF0.782–0.784, zero starvation/late frames/output flags;0.524ms alignment |
| Precision |Native numerical checks pass against official references. INT8 storage offered no speed/resident-memory benefit; rejected. Explicit32-token alignment failed the unchanged maximum-error gate; rejected |
Use the existing command with `--composer ace` to select the experimental prepared backend. Omitting it keeps SA3. Runtime consumes causal cereal messages, not future route files. Route-owned archives remain private and compatible with stored-score replay. No public StarPilot or MusicGen edits.
**Readiness classification: Level3, demonstrated prebuffer viability; Level4 throughput evidence on a full offroad route, not a reliability or production claim.** Two earlier stress-test GPU hangs required recovery. modeld coexistence and real driving are untested. ACE→SA3 seamless handoff is not implemented. A Mac transport-timing issue remains explicit. Final regression and cleanup details follow in the engineering report.
[Architecture, experiments and limitations](ACE_CHESTNUT.md) · [Native replay evidence](results/ace_chestnut_20260916/native_replay_audit.json) · [Mac failed capture evidence](results/ace_chestnut_20260916/mac_replay_audit.json).
Final checks:68 unit tests and5 V2 gesture tests pass; native ACE and default SA3 fresh regressions pass as described above; stored playback passes on both hosts. Owned workers are stopped, normal bench UI is restored, real offroad state verified, and CPU settings restored to their saved snapshot. **Mac and bench remained muted; no speaker or Bluetooth audio was enabled.**
## Previous pass (historical)
# RoadScore — composition strategy review
Continue from private checkpoint`a7dc34f` /`overnight-review`. The human rejected the previous section-bank music. That verdict stands; the old engineering successes do not establish musical acceptance.
**Start with [the new listening review](results/composition_20260916/index.html).** One page contains SA3/ACE comparisons, YuE's limited result, generated transitions, source-derived gestures, actual arrival excerpts, replay evidence and timings. Nothing autoplays. All musical judgments remain pending human listening.
## Recommendation
Keep SA3 as the demonstrated fast Chestnut backend, but do **not** assume it has solved composition. Listen to ACE-Step's90-second structured piece and outro next to the single aggressive SA3 control. ACE is a plausible longer-horizon composer candidate; it has not earned a musical win or a Chestnut port. YuE ran but was slow and much of its latter half was near-silent. No fine-tuning, compatibility layer, or speculative accelerator port was undertaken.
The minimum prototype separates longer-horizon generated material from immediate source-derived musical gestures. It now uses continuous generated transitions rather than the rejected independent section bank. This is still an experiment, not a claim of convincing verse/chorus hierarchy or seamless multi-model composition.
## What changed and what was measured
| Area | Evidence |
|---|---|
| SA3 control | One fresh K-pop/game-score identity, five extreme roles, three context-conditioned transitions.28.05s role outputs take about19.3s. Live26.006s continuations run at medianRTF0.77–0.78 on Chestnut. |
| ACE-Step1.5 | Official MLX stack on local M1 Max.90s in108.03s (RTF1.20); five30s reference-conditioned roles take63–91s;40s transition73.33s. Actual decoded WAVs retained. |
| YuE2 | Official MPS implementation generated81.279s in374.55s (RTF4.61), without hitting token caps.37.7s near silence; symbolic keyF minor despiteD-minor prompt. No further porting investment. |
| Gesture bank | Source-derived hats, fills, crashes and finite nav/arrival phrases. Tempo/phase estimated; no new guessed chord pitches. Synthetic first-signal onset115ms, zero scheduling lateness. Physical response also includes polling/output buffering. |
| Causal road controls | Native signal, curve and nav inputs produced gestures. Upcoming curve payoff follows current forecast revisions until one beat away; no future route data. Stored overlay exposes only decisions already reached on its sample timeline. |
| Arrival | Native arrival:35.236s of outro-conditioned material before cadence. Full community run:21.653s, with corrected counter. Whether either sounds like an intentional outro is unverified. |
| Native fresh replay | Arrival253.9s/8 jobs and curve176.3s/6 jobs; zero fallback loops, renderer underflows or output flags. Camera/path/lane evidence preserved. |
| Mac fresh replay | Full community route571.6s/21 jobs, normal UI, zero starvation/late samples/output flags, maximum alignment0.324ms. Route-owned archive complete. |
| Capture limitation | Fresh video-recording attempts failed timing checks on both hosts; originals retained. No tolerances weakened. Separate stored-score video validation follows below. |
| Tests |63 passed, including real-log future-mutation causality, gesture scheduling/cancellation, outro accounting and archived decision visibility. |
[Full architecture, model screen and limitations](COMPOSITION.md) · [Measured replay evidence](results/composition_20260916/regressions.json) · [Preserved failures](results/composition_20260916/failures.json) · [Local route inventory](results/composition_20260916/route_library.json).
## Musical acceptance questions
- Obvious SA3 section hierarchy? **Pending listening.**
- Better generated transitions and fewer audible splices? **Pending listening.** Context-generated comparisons exist; no seam-quality claim.
- Alternative structurally better than SA3? **Pending listening.** ACE is the next comparison, not a declared winner.
- Alternative ready for Chestnut? **No demonstrated port.** Memory/runtime estimates do not establish impossibility or a pass.
- Useful hybrid? **A plausible direction, partly prototyped.** Immediate gestures and long-horizon SA3 conditioning run; a coherent ACE→SA3 composition system has not been demonstrated.
- Immediate signal / curve punctuation? **Sample-timed prototype demonstrated.** Beat/key accuracy and musical effectiveness remain unverified.
- Actual outro before cadence? **Generated outro-conditioned material demonstrably played first.** Its musical concluding behavior still requires a human.
- Existing replay intact? **Fresh Mac/native tests passed without video recording.** Recording failures are explicit; stored tests and final cleanup are appended below.
All work remains private in Desktop/RoadScore and the offroad bench. Cached routes were not redownloaded or published. MusicGen and the public StarPilot source were untouched. Product audible defaults were preserved; this development session remains muted.
## Stored replay and shutdown — complete
- Mac stored replay with normal-UI video:180.03s, exact contiguous archived samples, no generation or output flags, maximum clock alignment0.583ms. [Synchronized video](results/composition_20260916/road_synchronized.mp4). Captured UI has occasional frame gaps (maximum149ms); timestamps are preserved.
- Native stored arrival:253.84s of callbacks, contiguous archived samples verified, no generation or output flags, maximum clock alignment9.90ms. Normal bench UI restored afterward.
- Resident Chestnut worker stopped; saved CPU settings restored exactly. Real bench state remains offroad.
- Mac output muted, both session mute locks retained, all tested audio streams muted. **No speaker output or Bluetooth audio output was intentionally enabled; bench test output remained muted.** Physical listening is deferred.
Source/strategy checkpoints this pass:`08246ed`,`0b72f6e`,`9ee1dbb`,`dc27742`; final private checkpoint recorded in Git. Earlier rejected music and both failed capture attempts remain available as evidence.
-20
View File
@@ -1,20 +0,0 @@
# Experiment ledger
Baseline: `b132721`, preserved tag `unattended-before-b132721`.
1. **Genre screen:** six distinct prompt families, same seed8101,30.093 seconds each, approximately19.3 seconds generation. Electronic, synthwave and funk selected for continuation comparisons on objective rhythmic/spectral behavior; no subjective winner asserted.
2. **Development controls:** fixed versus refreshed anchor, same seeds/section requests. Refreshed anchor did not clearly increase measured development. More contrasting prompts alone also had modest effects.
3. **Anchor removal:** harmonic/timbral distances increase substantially, but18.3 seconds fall below−50dB and repeated joins approach silence. Rejected. No silence repair or time stretching was applied.
4. **Short anchor:**22 latent frames /2.043 seconds instead of44 /4.087; four continuations total125.945 seconds. Zero100ms near-silence blocks, modestly greater final-section harmonic change. Preserved for listening; live default unchanged.
5. **Curve audit:** all three full cached routes scanned offline. Fixed steering definitions, raw prediction candidates, persistence, activation, commands and audio residuals are recorded. Future labels are evaluation-only and never runtime inputs. The0.6 trial created an extra event and delayed a subsequent sharp turn; it was not adopted and the bench default is restored to0.8. Low-speed sharp turns remain failures; do not infer success from convenient gentle curves alone.
6. **Style responses:** same source/captured event stream A/B for electronic, synthwave and funk. Deterministic processing changes existing generated material; it does not supply prerecorded songs or percussion assets.
7. **Native replay:** first transport test found a native all-service socket collision; an explicit UI/music service set avoids it. The initial set omitted encode-index services, preventing camera delivery. Adding roadEncodeIdx fixes this. This debugging history matters: early native recordings are not proof that camera reception worked.
8. **Native evidence:** third-route run confirms camera reception. Final run further confirms actual nonempty path/lane and camera-texture draw calls. No custom renderer or modeld rerun.
9. **Clock correction:** first-message wall anchoring showed up to0.17 seconds of misleading source/output alignment. Per-message delivery plus an SSH round-trip clock measurement replaces it in final diagnostics. A later offset discontinuity after the completed capture was excluded, not hidden.
10. **Deadline regression:** native_cold and native_gc are rejected as clean playback demonstrations because of output flags. GC instrumentation links every miss to85–96ms full collections. Direct result lookup plus pre-stream GC freezing yields a clean180.2-second final capture. Dynamic GC remains enabled; finalization unfreezes the heap.
11. **Arrival:** final different-route capture is clean, same causal trigger,1.691-second source-dependent runway, five-second resolving gesture. Original liked Horizon cadence is retained.
12. **Shutdown:** manually owned worker stopped, launcher-owned cold cleanup independently exercised, saved CPU states equal restored states. All captured audio blocks are muted.
Primary evidence lives in `results/unattended/listen.html`, `native_final`, `native_final_arrival`, `genre_screen.json`, `development_metrics.json`, `development_boundaries.json`, `curve_timing.json`, `threshold_sensitivity.json` and the synchronized timing JSON files. Raw diagnostics and rejected experiments remain secondary.
Current recommendation: keep native SA3 with active context and fixed anchor for the demo; offer several generated musical identities and early style-specific event realizations. Treat richer composition and universal turn anticipation as unresolved product work. Run an unseen compatible community route as the next acceptance test.
-122
View File
@@ -1,122 +0,0 @@
# Chestnut music-generation feasibility gate
## Verdict and current status
**Local neural music generation is demonstrated. Conditional GO for an asynchronous, replay-based hackathon demo.** This is an isolated experiment, not RoadScore production code.
On 2026-09-13 America/Chicago (2026-09-14 UTC), comma `192.168.3.111` generated a 2-second clip and two different 8-second clips from instrumental text conditioning. The entire MusicGen music-token transformer executed on Chestnut; text encoding, sampling, and waveform decoding executed on the comma CPU. No remote inference or prerecorded audio input was used. Network access downloaded dependencies and pretrained weights only.
The generated WAVs are the deliverables, not merely transformer outputs:
- [8 seconds, seed 123](results/clip_8s_seed123.wav)
- [8 seconds, seed 124](results/clip_8s_seed124.wav)
- [Initial 2-second output](results/first_2s.wav)
These files are valid, finite, non-silent audio with no clipped samples. **Perceptual musical quality has not been independently verified by the agent, which cannot listen to audio in this session.** A listening check was requested from the user; no judgment was received before this report. The demonstrated technical gate must not be confused with proof of polished musical quality, coherent stems, or long-form composition.
The practical decision:
1. **Can Chestnut generate music locally?** Yes: text-conditioned MusicGen inference and decoded WAV output completed on this device.
2. **Best demonstrated model/runtime?** MusicGen Small FP16, a narrow native tinygrad decoder with fixed KV storage and TinyJit, and CPU PyTorch for T5/EnCodec. The generic tinygrad PyTorch bridge was abandoned after bounded attempts.
3. **How fast?** Warm 8-second generation: 50.74 seconds on the GPU path plus 18.11 seconds for CPU decode, 68.85 seconds total. Real-time factor 8.61: a value above one means slower than playback.
4. **Coexist with driving inference?** Independent processes concurrently owning Chestnut did not work. A second owner failed with `libusb_set_configuration: Resource busy`. No driving-model deadline, preemption, thermal-soak, or combined-VRAM test was performed. Coexistence is not established.
5. **Another local architecture if coexistence fails?** Use recorded model messages during the replay demo and let Chestnut generate musical material asynchronously. Keep the model loaded; arrange, loop, and schedule its locally generated output while it generates further material. For live use, QCOM driving inference with Chestnut dedicated to music is an untested candidate; a shared GPU owner/scheduler is substantial additional work. Neither is a proven driving-safe fallback.
**Do not pivot solely because local music generation was thought impossible. Do pivot or change scope if simultaneous live driving inference on Chestnut, continuous newly generated audio at playback speed, or high-quality isolated synchronized stems are mandatory this weekend. Those gates have not passed.**
## Measurements
Hardware: ARM64 AGNOS, Linux 4.9.103, roughly 3.5 GiB host RAM, no swap; Chestnut `gfx1200`, `USBIface`, reported VRAM capacity 8,539,602,944 bytes, USB link 5000 Mbit/s. Repository revision: `6b8bb279d4a6bf5ca4c5e92a8f45edd37bba1bf7` on host and device. The vendored tinygrad belongs to that tree.
For the main benchmark, CPUs 4–7 were temporarily enabled, set to the normal openpilot 1,689,600 kHz cap, and used with four PyTorch threads. They are normally offline while offroad. No openpilot processes were stopped or driving modes activated. Original CPU settings were restored; [before](results/power_before.json) and [after](results/power_after.json) agree.
| Measurement | First 8-second run | Second 8-second run, same loaded process |
|---|---:|---:|
| Audio duration / format | 8 s, 32 kHz mono PCM16 | 8 s, 32 kHz mono PCM16 |
| Music-token generation | 100.60 s, includes JIT warmup | 50.74 s |
| CPU waveform decode | 18.73 s | 18.11 s |
| Generation + decode | 119.33 s | 68.85 s |
| Real-time factor | 14.92 | 8.61 |
| Steady token-step wall time | 0.1261 s | 0.1259 s |
| Output peak amplitude | 0.888 | 0.732 |
| Clipped-sample fraction | 0 | 0 |
Full cold process launch to the first 8-second WAV: **167.92 seconds**, including imports, checkpoint loading, text encoding, a test-only CPU reference forward, GPU weight placement, compilation, generation, and decoding. CPU text conditioning separately measured **0.684 seconds**. The warm run reused the same conditioning with a different random seed; warm prompt changes were not benchmarked.
Sampled process RSS high-water mark: **2,116,182,016 bytes (2.12 GB)**, including the test-only reference computation. Runtime snapshots were about 1.15 GB after CPU transformer release and 1.68 GB after the second decode. Sampling was every 50 ms, so this is not an exact kernel-accounted RSS maximum.
Peak tinygrad-tracked active buffer allocation: **928,465,340 bytes (0.93 GB)**. This is an allocation-counter estimate, principally Chestnut tensors, **not full physical VRAM utilization**; allocator caches, runtime reservations, and other allocations are not fully represented. Full peak physical VRAM was not measured.
The initial 2-second attempt used the offroad little cores, two PyTorch threads, and an earlier JIT setup. It took 133.30 seconds for generation, 14.40 seconds for decoding, and 234.48 seconds from process start. Do not extrapolate that cold result to steady performance.
Raw evidence: [main measurements](results/bench8_result.json), [run log](results/bench8.log), [initial measurements](results/first_2s.json), [environment versions](results/requirements.freeze.txt).
## Actual execution boundary
```text
Text prompt
-> comma CPU: tokenizer + pretrained T5 text encoder + projection
-> Chestnut: embeddings, positions, all 24 MusicGen transformer layers,
self/cross-attention, KV cache, feed-forward layers, output heads
-> comma CPU: classifier-free guidance, top-k sampling, codebook delay pattern
-> Chestnut: next autoregressive step, repeating until all codes are generated
-> comma CPU: pretrained EnCodec waveform decoder
-> 32 kHz mono WAV on /data/roadscore-feasibility
```
The CPU reference logits are used only for verification, never for selecting generated tokens. The native generation implementation constructs its numerical tensors on `AMD`; the measured device interface is `USBIface`. TinyJit captures the GPU work for subsequent steps. Kernel counter values include graph dispatches and are not a count of every internal GPU shader during graph replay.
The model is [Meta MusicGen Small](https://huggingface.co/facebook/musicgen-small), pinned to revision `4c8334b02c6ec4e8664a91979669a501ec497792`. Architecture and delay-pattern handling follow [Transformers v4.46.3 MusicGen](https://github.com/huggingface/transformers/blob/v4.46.3/src/transformers/models/musicgen/modeling_musicgen.py). Model weights retain their original license (CC-BY-NC-4.0); upstream implementation attribution and licensing apply to reused architecture/code.
## What was tried and what was rejected
- **Normal ACE-Step / ROCm:** not a demonstrated path. The device does not expose `/dev/kfd`; Chestnut uses tinygrad's custom USB transport. ACE-Step's documented PyTorch/ROCm installation is not a bridge to that transport. ACE-Step was not installed or benchmarked.
- **tinygrad PyTorch bridge + MusicGen:** installed PyTorch 2.10 CPU ARM64 and compiled the existing extension successfully. A bridge GPU smoke test passed. The complete transformer produced excessive graph/scheduling overhead before the first token. An interrupted attempt reported an AMD synchronization failure during finalization. A layer-boundary materialization attempt also failed to produce a first token within the allotted several minutes and was stopped. No general compatibility layer was developed.
- **Checkpoint loading:** the original 2.36 GB safetensors mapping failed with `Cannot allocate memory`. A streaming conversion to nineteen <=64 MiB target FP16 shards (1.18 GB total) solved loading without changing kernel memory policy or adding swap.
- **Native tinygrad subset:** a small explicit implementation of the existing MusicGen decoder, using existing tensor operations, fixed KV caches, and TinyJit, completed end-to-end generation.
- **Other candidates:** Stable Audio Open Small and TinyMusician were inspected as alternatives. They were not benchmarked once the native MusicGen path succeeded. No speed or compatibility claims about them on Chestnut are justified by this experiment.
## Correctness and remaining limits
The first four input/output steps of **both** runs were checked against CPU Transformers using the same FP16 weights, text conditioning, and token inputs, including JIT execution and cache reuse. All eight checks passed cosine similarity >0.999; observed minimum was **0.999966**. Mean absolute logit errors ranged from 0.0018 to 0.0256. This is approximate FP16 agreement, not bitwise equivalence or an exhaustive model-port validation. See [validation](results/validation.json) and saved step arrays.
Both full outputs decoded successfully and had finite, nonzero samples without clipping. These checks do not establish listening quality, absence of incidental vocals, genre adherence, or harmonic compatibility between independently generated clips. The WAV files, rather than claims about their aesthetic quality, are provided for review.
The USB contention result is in [coexistence probe](results/coexist_lock.log). Separate ownership is blocked; a process priority or VRAM budget alone does not solve it. A single-owner scheduler would need measured deadlines and safe bounded GPU work. Music token-step wall time exceeds the 50 ms model frame interval; GPU-only execution time and available interleaving headroom have not been isolated.
No real-time audio-output jitter, route alignment, camera replay, modeld coexistence, or continuous thermal-soak test was attempted. Those are outside this generation gate. The device's onroad state remained false. No RoadScore daemon, UI, cereal schema, controls integration, or production source change was introduced.
## Reproduce on this device
Everything installed on-device is under `/data/roadscore-feasibility`, including its own venv, package/model caches, scripts, weights, logs, and WAVs. openpilot's Python environment was not changed. Benchmarks are offroad only.
The already-provisioned device can repeat the main benchmark:
```bash
ssh comma@192.168.3.111 \
'/usr/local/venv/bin/python /data/roadscore-feasibility/power_bench.py --seconds 8 --repeats 2 --tag bench8'
```
This temporarily enables the big CPU cores, pins the experimental process to them, and restores prior CPU settings on exit while still offroad. Reusing the same tag overwrites that tag's previous outputs on-device; choose another tag to preserve them. The repository's copied evidence remains unchanged. Do not start this while another GPU owner is active. Repeats reuse the prompt and change the seed.
To provision a new isolated experiment directory, copy this directory's Python/shell scripts and `results/requirements.freeze.txt` to `/data/roadscore-feasibility`, then on the comma:
```bash
mkdir -p /data/roadscore-feasibility/cache /data/roadscore-feasibility/tmp
export UV_CACHE_DIR=/data/roadscore-feasibility/cache/uv
export TMPDIR=/data/roadscore-feasibility/tmp
uv venv --python /usr/bin/python3.12 /data/roadscore-feasibility/venv
uv pip install --python /data/roadscore-feasibility/venv/bin/python \
-r /data/roadscore-feasibility/requirements.freeze.txt
/data/roadscore-feasibility/venv/bin/python /data/roadscore-feasibility/download.py
/data/roadscore-feasibility/venv/bin/python /data/roadscore-feasibility/shard.py
/usr/local/venv/bin/python /data/roadscore-feasibility/power_bench.py --seconds 8 --repeats 2 --tag bench8
```
`run_native.sh` sets the tinygrad path to the installed repository and selects `DEV=USB+AMD:LLVM`. `verify_native.py` checks `bench8` artifacts against the CPU reference; run it in the experimental venv after generation. The existing generic bridge is not needed by the demonstrated native path.
## Hackathon implication
The viable scope is a persistent local generator producing short material ahead of a deterministic arranger. With the measured configuration, plan approximately **one new 8-second asset per 69 seconds** at steady state, after a roughly three-minute cold startup. Keep locally generated material playing through repeats/transitions while further material is generated. Pitch/key/BPM matching, seamless looping, and stem controllability remain musical engineering tasks; MusicGen does not provide proven synchronized isolated stems here.
A replay demo can dedicate Chestnut to music and consume recorded openpilot predictions, satisfying genuine local generation without competing GPU owners. A live system requiring Chestnut driving inference and simultaneous music generation has **not** passed feasibility. Do not begin the full RoadScore build until the user accepts the audible material and this asynchronous/replay scope.
@@ -1,125 +0,0 @@
# Chestnut musical controllability experiment
This is an isolated feasibility experiment, not Orchestra/RoadScore implementation. No replay, Galaxy, driving integration, or playback service is added.
## Recommendation
Use **one Chestnut-generated musical identity, a persistent phase-aligned loop derived from it, and deterministic intensity/brightness/swell controls** as the minimum viable Orchestra. Preserve generated EnCodec tokens and request audio-prefix-conditioned continuations asynchronously as optional future material. Audition/select new passages before replacing the bed. Do not make event timing or persistence depend on a new model response.
MusicGen Small supports a real continuation path with a small extension to the existing native tinygrad harness. This is stronger than independent text prompts: every branch has the same exact eight-second audio-token prefix. However, neither the model API nor this small experiment establishes guaranteed motif/key/tempo retention in newly generated material, a reliable intensity knob, independently editable instruments, or a cadence on demand. Prefix identity is not evidence that the generated suffix is musically coherent. Human listening is decisive.
The hybrid approach preserves source identity by construction; it controls energy and texture without asking the model to rewrite the song at each event. This is a narrower musical claim than generating orchestrated stems or harmonic tension/resolution. If those are mandatory, this gate is still unproven.
## Result and validation
**Conditional recommendation: hybrid Orchestra.** Real continuation and changed prompt conditioning work on Chestnut. Dependable model-only intensity, instrumentation, key/BPM, and harmonic resolution are not established. The continuous source-preserving DSP demo is the strongest demonstrated route to persistence and controllable energy. Human listening acceptance remains open; no audio-quality judgment is claimed by the coding agent.
The experiment began at 02:53 UTC on 14 September 2026 and finished within the 35-minute bound. No second model or general runtime extension was attempted. CPU settings were restored exactly after both generation and verification; before/after JSON snapshots match.
All four branches retained their 400-frame (eight-second) source-token prefix exactly. Their full delay masks match the installed official Transformers implementation. A separate short-prefix cached-execution test compared the last four positions of a 16-position sequence against CPU Transformers under low and high conditioning, including JIT reuse after replacing the prompt. Cosine similarity was 0.99999356 / 0.99999386, with mean absolute logit error about 0.0106. This is an approximate FP16 correctness check, not exhaustive validation of every generated token. All prompt lengths were 49–59 tokens, within the 64-token padded input.
The three newly generated branch suffixes differ on 94–96% of their codec-token positions. They are actual new generations, not copies of the prefix or each other. Token difference does not measure musical difference. WAV headers, finite/non-silent output, clipping, and hashes were independently checked; see `results/wav_manifest.json`.
## What was tested
Hardware/runtime are unchanged from the [first gate](../README.md): comma four CPU plus Chestnut through tinygrad USBIface, MusicGen Small FP16 weights. All 24 autoregressive music transformer layers execute on Chestnut. CPU handles T5 text conditioning, sampling/masking, and EnCodec waveform decoding. The offline DSP experiment also runs on the comma CPU. No external audio or remote inference is used.
1. Generate a new 12-second base passage on Chestnut.
2. Retain its final eight seconds of EnCodec tokens.
3. Teacher-force those exact tokens through the decoder, then generate eight new seconds under neutral, low-intensity, or high-intensity text conditioning.
4. Take the high branch's eight newly generated seconds as another prefix and generate an eight-second release.
5. Derive periodic low/neutral/high DSP variants from one passage of the generated base, with identical timing and notes. Render repeated loops and a continuous rise/release listening example.
Neutral/low/high share the same prefix and sampling seed (321). The common prompt requests a repeating synth motif, soft strings, A minor, 100 BPM, and instrumental cinematic downtempo. Only the arrangement/mood suffix changes. This is one paired branching experiment, not a statistically meaningful prompt-adherence study.
The extension adds no model layers and leaves `native_decoder.py` unchanged. It supplies forced prefix tokens using the official delayed-codebook convention and updates preallocated cross-attention/mask buffer contents between prompts. Keeping the same buffers matters because TinyJit captures those buffers. The self-attention cache is overwritten from position zero for each branch. Prefix processing is sequential and costs roughly as much as generating the same duration; batched prefill or prefix-cache snapshots were deliberately not implemented.
## Measured performance and prompt response
| Run | Retained + new audio | GPU loop including prefix/sampling | CPU decode | Entire run including conditioning-buffer update |
|---|---:|---:|---:|---:|
| base | 0 + 12 s | 124.32 s | 27.59 s | 157.31 s |
| neutral | 8 + 8 s | 100.32 s | 35.07 s | 138.43 s |
| low | 8 + 8 s | 100.36 s | 34.77 s | 138.16 s |
| high | 8 + 8 s | 100.39 s | 34.55 s | 137.98 s |
| release | 8 + 8 s | 100.24 s | 34.50 s | 137.76 s |
The base includes cold JIT execution. Later runs reuse the JIT and change cross-attention conditioning buffers. CPU text encodings were computed beforehand in 1.12–1.17 seconds per prompt; add that for an unseen prompt. The entire-run column includes roughly 3 seconds to update GPU conditioning buffers but excludes model loading and those precomputed text encodings. GPU-loop time includes CPU sampling and USB transfers; it is not pure shader time.
Warm continuations therefore add eight seconds of music in approximately **139 seconds including a fresh prompt** (~17.4 times the new audio duration). About 49.7 seconds are spent ingesting the retained prefix through the first new-token logits. Steady steps remain 0.125 seconds, close to the first gate: the added latency comes from processing context and decoding twice as much audio, not a large steady-step slowdown.
Peak sampled process RSS was **2,125,008,896 bytes (2.13 GB)**, including CPU reference checks/loading; peak tracked tinygrad buffers were **1,014,611,536 bytes (1.01 GB)**, not full physical VRAM. The five generation runs completed in about 13 minutes after imports, including model preparation and CPU reference checks. No output contained nonfinite values or clipped samples.
On newly generated suffixes, RMS was neutral **0.1886**, low **0.2032**, high **0.1701**, release **0.1714**. Spectral centroid was neutral **1365 Hz**, low **1193 Hz**, high **1316 Hz**, release **1164 Hz**. Thus the low prompt produced a darker spectrum, but the high prompt did not exceed neutral on either of these coarse intensity proxies. They do not measure musical quality or perceived intensity; this is evidence against treating prompt labels as calibrated controls.
New-suffix pitch-class profile cosine similarity to the base was **0.991 neutral / 0.997 low / 0.967 high**; release-to-high was **0.995**. These descriptive numbers support tonal relatedness, but cannot establish a shared melody, exact key, functional harmony, or perceptually seamless switching. The first 7.8 decoded seconds of neutral/low/high were sample-identical. Differences near the prefix's decoded end illustrate why retaining exact tokens does not make every codec boundary sample identical.
## Listening guide
Files are in [results](results/). Names `*_full.wav` include the retained prefix; names `*_new.wav` contain only newly generated material. No prefix duration is counted as new generation.
- [Base identity, 12 seconds](results/base_full.wav).
- [Neutral continuation](results/neutral_full.wav), [lower-intensity prompt](results/low_full.wav), [higher-intensity prompt](results/high_full.wav): each is 16 seconds, with identical token conditioning for the first eight seconds and new music beginning at 8.0 seconds. Judge the transition and the new suffix, not only the shared opening.
- [Release continuation](results/release_full.wav): high branch material for eight seconds, then the release attempt.
- [Model continuation chain](results/model_tension_release_chain.wav): base → high → release, 28 seconds. New high material begins at 12 seconds; release begins at 20 seconds. A 120 ms blend in the shared decoded context handles codec boundary differences without shifting the musical timeline.
- [Hybrid persistent score](results/hybrid_tension_release.wav): one generated source loop, continuous low → high → low DSP adaptation. This directly tests the recommended minimum architecture.
- [Low loop](results/loop_low.wav), [neutral loop](results/loop_neutral.wav), [high loop](results/loop_high.wav): four cycles each, all using the same source phase. Compare repeat boundaries and energy.
- [Matched model comparison](results/model_variants_comparison.wav): neutral, low, then high new material, eight seconds each with half-second gaps.
- `neutral_matched.wav`, `low_matched.wav`, `high_matched.wav`, and `release_matched.wav` normalize the new suffixes to the same RMS target, subject to recorded peak limiting. RMS matching is not perceptual LUFS normalization.
Listening questions: Does the model continuation retain the same song? Do low/high differ in the intended direction beyond loudness? Does the release feel like a resolution or just another passage? Does the hybrid loop repeat unobtrusively? Is its energy change expressive enough for the project? The automated measurements cannot answer these questions.
## DSP loop experiment
The primary loop uses a **4.864-second** source period, starting at 1.568 seconds in the generated base, with a 120 ms seam blend. This corresponds to eight hypothesized beats at 98.68 BPM. Autocorrelation's strongest candidate was 66.96 BPM; the near-100 candidate was also strong and was selected using the requested-tempo prior. The [alternative loop](results/loop_alternative_tempo.wav) retains the strongest autocorrelation interpretation (7.168-second period). Neither choice establishes the downbeat or harmonic phrase boundary.
The main [hybrid demo](results/hybrid_tension_release.wav) lasts **38.912 seconds**: low for 0–9.728 s, rising energy to 19.456 s, high through 24.320 s, release through 34.048 s, then low. A short reverse swell made from the generated source precedes the peak. No new instrument or outside audio is added. Level and spectral changes act on the same notes and phase; they are not separately generated stems.
The low/neutral/high loops have RMS **0.0993 / 0.1497 / 0.1824** with a common headroom gain. The periodic seam's sample jump is 0.0124, below the largest interior sample step (0.2508); this rules out a large simple splice jump, not an audible rhythmic/harmonic mismatch. The entire offline analysis-and-WAV rendering script took 5.56 seconds on the restored comma CPU configuration, excluding imports. That is not a real-time playback/deadline benchmark.
## Capability boundaries
| Requirement | Model-only evidence | Minimum hybrid capability |
|---|---|---|
| Persistent identity | Exact source-token prefix; new suffix identity requires listening | Repeat and transform the same generated phrase; notes/timing retained |
| Low/high intensity | Paired text-conditioned branches; intended direction is not guaranteed | Continuous gain and spectral balance, with common phase |
| Looping/crossfades | Generation is not constrained to close a musical loop | Select a candidate periodic passage, soften seam, repeat; phrase/downbeat quality still needs listening |
| Tension/release | High → release continuation attempt | Timed rise/fall in energy and source-derived reverse swell; not a guaranteed harmonic cadence |
| Switching without changing songs | Possible through continuation, unproven across arbitrary assets | Crossfade aligned DSP variants of the same loop; avoid arbitrary branch layering |
| Tempo/key | Text requests only; no hard BPM/key input in this Small checkpoint | Retain actual source timing/pitches; validate/annotate them before adding harmony or bar-level events |
| Instrumentation | Text can request arrangement; mixed mono output, no instrument isolation | Filter/mix texture controls, not independent string/drum stems |
The loop extractor's onset autocorrelation and chroma statistics are rough short-clip descriptors. They are not validated beat/downbeat/key estimates. A seam blend prevents a simple splice discontinuity but cannot fix a broken chord progression. No time stretching or pitch shifting is used; there is no claim the source obeys the requested 100 BPM or A minor.
## Smallest Orchestra design implied by the evidence
- A session stores one chosen generated phrase, its original audio codes, loop boundaries, and a common sample/phrase clock. Add human-verified tempo/key only if needed; otherwise avoid inventing harmony over the mix.
- The timing-critical side reads the existing source and smoothly automates intensity, spectral balance, and source-derived swells. Keep all variants phase-aligned. Hold the current bed when generation is late or fails.
- A separate non-critical generation task requests continuations from recent source tokens. Preserve the initial identity anchor too, so long continuation chains do not become the only available context. Long-term drift prevention using that anchor is a proposal, not a tested conditioning feature.
- Treat each generated continuation as a candidate sequential passage, not a synchronized stem. Listen/select before adding it to the usable pool for the demo. Replace a bed only at an acceptable phrase boundary; a generic crossfade does not establish harmonic compatibility.
- Chestnut continues to contribute actual musical composition. DSP makes that locally generated composition persistent and responsive. No prerecorded library is substituted.
Do not add a large melody model, source-separation model, symbolic composer, key-control training, or generalized PyTorch compatibility work to the MVP. The earlier exclusive Chestnut ownership limitation remains: this experiment does not prove coexistence with driving inference.
## Sources checked before implementation
[Transformers MusicGen documentation, v4.46.3](https://huggingface.co/docs/transformers/v4.46.3/en/model_doc/musicgen) documents audio-prompted continuation and the shared prompt/output duration limit. [The matching source](https://github.com/huggingface/transformers/blob/v4.46.3/src/transformers/models/musicgen/modeling_musicgen.py) supplies the exact BOS and delayed-codebook semantics used here.
[AudioCraft's model catalog](https://github.com/facebookresearch/audiocraft/blob/main/docs/MUSICGEN.md) distinguishes the 300M Small checkpoint from the 1.5B Melody checkpoint with chroma conditioning. Melody guidance is not a switch we can enable in these Small weights. A larger-model migration was outside this bounded experiment.
## Reproduction on the provisioned device
The scripts assume the first gate's isolated environment and weights under `/data/roadscore-feasibility`. Copy `probe.py` there as `controllability_probe.py`, alongside `native_decoder.py`, `run_probe.sh`, and `power_probe.py`. Run offroad with Chestnut available:
```sh
ssh comma@192.168.3.111 '/usr/local/venv/bin/python /data/roadscore-feasibility/power_probe.py'
```
The wrapper temporarily enables the big CPU cores at the normal openpilot cap, then restores the prior settings. Its generation timeout is 1400 seconds. Output names are fixed; another run overwrites the device's controllability outputs. `arrange.py` derives only offline listening artifacts from those outputs. `verify.py` checks prefix/mask construction and cached GPU logits against the CPU implementation; it is a separate exclusive GPU run. Copy those files plus `power_verify.py` and `run_verify.sh` into the same device directory, then run:
```sh
ssh comma@192.168.3.111 '/usr/local/venv/bin/python /data/roadscore-feasibility/power_verify.py'
ssh comma@192.168.3.111 '/data/roadscore-feasibility/venv/bin/python /data/roadscore-feasibility/arrange.py'
```
`result.json`, `arrangement.json`, `validation.json`, logs, and CPU setting snapshots accompany the WAVs. Full physical VRAM utilization is not available; tracked tinygrad buffers must not be described as complete device memory consumption. This experiment does not include real-time playback scheduling, long-duration thermal tests, or human-rated musical acceptance.
@@ -1,135 +0,0 @@
"""Offline listening experiment: generated-source loops and continuous DSP controls.
No external audio, source separation, event/replay/UI integration, or production service.
"""
import json, time
from pathlib import Path
import numpy as np
import soundfile as sf
from scipy import signal
ROOT=Path('/data/roadscore-feasibility/controllability')
SR=32000
start=time.monotonic()
def load(name):
x,sr=sf.read(ROOT/f'{name}.wav'); assert sr==SR and x.ndim==1
return x
def analysis(x):
f,t,z=signal.stft(x,SR,nperseg=4096,noverlap=3584)
mag=np.abs(z)
# Descriptive chroma, not a reliable key detector or a musical-quality metric.
keep=(f>=65)&(f<2100)
midi=np.rint(69+12*np.log2(f[keep]/440)).astype(int)
chroma=np.bincount(midi%12,weights=(mag[keep]**2).mean(axis=1),minlength=12)
chroma=chroma/(np.linalg.norm(chroma)+1e-12)
flux=np.maximum(np.diff(np.log1p(mag*100),axis=1),0).mean(axis=0)
flux=np.maximum(flux-np.median(flux),0)
rate=SR/512
corr=signal.correlate(flux,flux,mode='full')[len(flux)-1:]
candidates=[]
for k in signal.find_peaks(corr)[0]:
bpm=60*rate/k
if 60<=bpm<=160: candidates.append((float(corr[k]/max(corr[0],1e-12)),float(bpm)))
candidates.sort(reverse=True)
return {'rms':float(np.sqrt(np.mean(x*x))), 'peak':float(np.max(np.abs(x))),
'centroid_hz':float((f[:,None]*mag).sum()/max(mag.sum(),1e-12)),
'high_band_power_fraction':float((mag[f>2000]**2).sum()/max((mag**2).sum(),1e-12)),
'chroma_unit':chroma.tolist(), 'tempo_candidates':candidates[:5]},flux
def save(name,x):
assert np.isfinite(x).all()
# Fixed common gain is used for aligned DSP variants. Any exceptional peak limit is recorded.
gain=min(1.,.95/max(np.max(np.abs(x)),1e-12))
sf.write(ROOT/f'{name}.wav',x*gain,SR,subtype='PCM_16')
return {'seconds':len(x)/SR,'peak_before_limit':float(np.max(np.abs(x))),'file_gain':gain}
metrics={}
for name in ['base','neutral','low','high','release']:
x=load(name+'_new')
metrics[name],_=analysis(x)
target=.10
matched=x*(target/max(metrics[name]['rms'],1e-12))
save(name+'_matched',matched)
save('model_variants_comparison',np.concatenate([
load('neutral_matched'),np.zeros(SR//2),load('low_matched'),np.zeros(SR//2),load('high_matched')]))
base=load('base_full')
base_stats,flux=analysis(base)
# Autocorrelation is ambiguous on short music. Choose strongest candidate only as a loop hypothesis.
tempo_candidates=base_stats['tempo_candidates']
strongest_bpm=tempo_candidates[0][1] if tempo_candidates else 100.
# Use the requested tempo only to choose among strong measured candidates; not as ground truth.
near_requested=[(strength,bpm) for strength,bpm in tempo_candidates
if abs(bpm-100)<5 and strength>=.8*tempo_candidates[0][0]]
bpm=near_requested[0][1] if near_requested else strongest_bpm
fade=round(.12*SR)
peaks=signal.find_peaks(flux,distance=round(.25*SR/512))[0]*512
def make_loop(tempo):
length=round(8*60/tempo*SR)
candidates=[int(p) for p in peaks if p>SR and p+length+fade<len(base)]
if not candidates: candidates=[SR]
def seam_cost(p):
a=base[p:p+fade];b=base[p+length:p+length+fade]
return float(np.mean((a-b)**2)/(np.mean(a*a+b*b)+1e-12))
p=min(candidates,key=seam_cost)
x=base[p:p+length+fade]
w=.5-.5*np.cos(np.linspace(0,np.pi,fade))
loop=np.concatenate([x[:fade]*w+x[length:length+fade]*(1-w),x[fade:length]])
return loop,p,length,seam_cost(p)
loop,p,length,seam_score=make_loop(bpm)
alternative,alternative_start,alternative_length,alternative_cost=make_loop(strongest_bpm)
save('loop_alternative_tempo',np.tile(.78*alternative,4))
# Circular filtering preserves the same source phase and periodic boundary.
def filt(x,cutoff,kind):
sos=signal.butter(2,cutoff,btype=kind,fs=SR,output='sos')
return signal.sosfiltfilt(sos,np.tile(x,3))[len(x):2*len(x)]
soft=filt(loop,1500,'lowpass')
bright=filt(loop,2200,'highpass')
low=.52*soft
neutral=.78*loop
high=.95*loop+.22*bright
common_gain=min(1.,.85/max(np.max(np.abs(y)) for y in [low,neutral,high]))
low*=common_gain;neutral*=common_gain;high*=common_gain
files={}
for name,y in [('loop_low',low),('loop_neutral',neutral),('loop_high',high)]:
files[name]=save(name,np.tile(y,4))
metrics[name],_=analysis(y)
# One shared phase across 8 repeated phrases; intensity rises and falls continuously.
N=len(loop)*8
t=np.arange(N)/SR
phrase=len(loop)/SR
control=np.interp(t,np.array([0,2,4,5,7,8])*phrase,[0,0,1,1,0,0])
control=control*control*(3-2*control)
l=np.tile(low,8);h=np.tile(high,8)
adaptive=l*(1-control)+h*control
# Add a pitched-source-derived reverse swell before peak; no synth, sample library, or new notes.
swell_len=min(round(1.2*SR),len(loop))
swell=loop[-swell_len:][::-1]*np.linspace(0,1,swell_len)**2*.14*common_gain
peak_sample=4*len(loop)
adaptive[peak_sample-swell_len:peak_sample]+=swell
# Presentation fade at file boundaries only; loop seams remain exposed internally.
adaptive[:1600]*=np.linspace(0,1,1600);adaptive[-1600:]*=np.linspace(1,0,1600)
files['hybrid_tension_release']=save('hybrid_tension_release',adaptive)
# Same passage A/Bs, with common generated prefix retained. Model branches are not crossfade-aligned stems.
for name in ['neutral','low','high']:
full=load(name+'_full')
files[name+'_context']=save(name+'_context',full)
# Continuation chain: base (12s), high new (8s), release new (8s).
# Use 120ms aligned overlap in already-shared prefix, preserve timeline duration.
def append_continuation(previous,full,prefix_seconds=8):
ov=round(.12*SR); boundary=round(prefix_seconds*SR)
w=.5-.5*np.cos(np.linspace(0,np.pi,ov))
result=previous.copy()
result[-ov:]=result[-ov:]*(1-w)+full[boundary-ov:boundary]*w
return np.concatenate([result,full[boundary:]])
chain=append_continuation(base,load('high_full'))
chain=append_continuation(chain,load('release_full'))
files['model_tension_release_chain']=save('model_tension_release_chain',chain)
for name in ['neutral','low','high']:
metrics[name]['chroma_cosine_to_base']=float(np.dot(metrics[name]['chroma_unit'],metrics['base']['chroma_unit']))
metrics['release']['chroma_cosine_to_high']=float(np.dot(metrics['release']['chroma_unit'],metrics['high']['chroma_unit']))
result={'analysis_warning':'Short-clip tempo/chroma are descriptive hypotheses, not verified BPM/key or listening quality.',
'metrics':metrics,'loop':{'bpm_hypothesis':bpm,'beats_assumed':8,'requested_bpm_prior':100,'strongest_autocorrelation_bpm':strongest_bpm,'alternative_period_seconds':alternative_length/SR,'start_seconds':p/SR,'period_seconds':length/SR,
'overlap_seconds':fade/SR,'seam_cost':seam_score,'common_gain':common_gain,
'seam_sample_jump':float(abs(loop[-1]-loop[0])),'max_interior_sample_jump':float(np.max(np.abs(np.diff(loop))))},
'files':files,'elapsed_seconds':time.monotonic()-start}
(ROOT/'arrangement.json').write_text(json.dumps(result,indent=2))
print(json.dumps(result,indent=2))
@@ -1,61 +0,0 @@
"""Temporary offroad benchmark CPU settings; restore on exit. No manager changes."""
import json
import os
import signal
import subprocess
import sys
from pathlib import Path
root = Path('/data/roadscore-feasibility')
def is_offroad():
check = subprocess.run(
['/usr/local/venv/bin/python', '-c',
'from openpilot.common.params import Params; raise SystemExit(int(Params().get_bool("IsOnroad")))'],
cwd='/data/openpilot', check=False,
)
return check.returncode == 0
if not is_offroad():
raise SystemExit('This isolated benchmark requires a verified offroad device.')
paths = [Path(f'/sys/devices/system/cpu/cpu{i}/online') for i in range(4, 8)]
paths += [Path('/sys/devices/system/cpu/cpufreq/policy4') / n for n in ['scaling_max_freq', 'scaling_governor']]
saved = {str(p): p.read_text().strip() for p in paths}
(root / 'controllability_power_before.json').write_text(json.dumps(saved, indent=2))
def write(path, value):
subprocess.run(['sudo', '-n', 'tee', str(path)], input=value + '\n', text=True, stdout=subprocess.DEVNULL, check=True)
def stop(sig, frame):
raise KeyboardInterrupt
signal.signal(signal.SIGTERM, stop)
child = None
rc = 1
try:
for path in paths[:4]:
write(path, '1')
write(paths[4], '1689600')
write(paths[5], 'performance')
print('POWER', json.dumps({str(p): p.read_text().strip() for p in paths}), flush=True)
child = subprocess.Popen(['taskset', '-c', '4-7', 'bash', str(root / 'run_probe.sh'), *sys.argv[1:]], start_new_session=True)
rc = child.wait()
finally:
try:
if child is not None and child.poll() is None:
os.killpg(child.pid, signal.SIGTERM)
try:
child.wait(timeout=15)
except subprocess.TimeoutExpired:
os.killpg(child.pid, signal.SIGKILL)
child.wait()
finally:
if is_offroad():
for path in reversed(paths):
write(path, saved[str(path)])
(root / 'controllability_power_after.json').write_text(json.dumps({str(p): p.read_text().strip() for p in paths}, indent=2))
print('POWER_RESTORED', flush=True)
else:
print('Device no longer verified offroad; leaving active CPU settings to hardwared.', flush=True)
sys.exit(rc)
@@ -1,61 +0,0 @@
"""Temporary offroad benchmark CPU settings; restore on exit. No manager changes."""
import json
import os
import signal
import subprocess
import sys
from pathlib import Path
root = Path('/data/roadscore-feasibility')
def is_offroad():
check = subprocess.run(
['/usr/local/venv/bin/python', '-c',
'from openpilot.common.params import Params; raise SystemExit(int(Params().get_bool("IsOnroad")))'],
cwd='/data/openpilot', check=False,
)
return check.returncode == 0
if not is_offroad():
raise SystemExit('This isolated benchmark requires a verified offroad device.')
paths = [Path(f'/sys/devices/system/cpu/cpu{i}/online') for i in range(4, 8)]
paths += [Path('/sys/devices/system/cpu/cpufreq/policy4') / n for n in ['scaling_max_freq', 'scaling_governor']]
saved = {str(p): p.read_text().strip() for p in paths}
(root / 'controllability_verify_power_before.json').write_text(json.dumps(saved, indent=2))
def write(path, value):
subprocess.run(['sudo', '-n', 'tee', str(path)], input=value + '\n', text=True, stdout=subprocess.DEVNULL, check=True)
def stop(sig, frame):
raise KeyboardInterrupt
signal.signal(signal.SIGTERM, stop)
child = None
rc = 1
try:
for path in paths[:4]:
write(path, '1')
write(paths[4], '1689600')
write(paths[5], 'performance')
print('POWER', json.dumps({str(p): p.read_text().strip() for p in paths}), flush=True)
child = subprocess.Popen(['taskset', '-c', '4-7', 'bash', str(root / 'run_verify.sh'), *sys.argv[1:]], start_new_session=True)
rc = child.wait()
finally:
try:
if child is not None and child.poll() is None:
os.killpg(child.pid, signal.SIGTERM)
try:
child.wait(timeout=15)
except subprocess.TimeoutExpired:
os.killpg(child.pid, signal.SIGKILL)
child.wait()
finally:
if is_offroad():
for path in reversed(paths):
write(path, saved[str(path)])
(root / 'controllability_verify_power_after.json').write_text(json.dumps({str(p): p.read_text().strip() for p in paths}, indent=2))
print('POWER_RESTORED', flush=True)
else:
print('Device no longer verified offroad; leaving active CPU settings to hardwared.', flush=True)
sys.exit(rc)
@@ -1,141 +0,0 @@
"""Bounded continuation probe. Experimental only; CPU text/codec, Chestnut decoder."""
import gc, json, os, threading, time
from pathlib import Path
import psutil
import numpy as np
import torch
import soundfile as sf
from transformers import AutoProcessor, MusicgenForConditionalGeneration
from transformers.utils import logging
from tinygrad import Tensor, Device
from tinygrad.helpers import GlobalCounters
from native_decoder import NativeDecoder
ROOT = Path('/data/roadscore-feasibility')
OUT = ROOT / 'controllability'
OUT.mkdir(exist_ok=True)
start = time.monotonic()
peak_rss = peak_buffers = 0
proc = psutil.Process()
def monitor():
global peak_rss, peak_buffers
while True:
peak_rss = max(peak_rss, proc.memory_info().rss)
peak_buffers = max(peak_buffers, GlobalCounters.mem_used)
time.sleep(.05)
threading.Thread(target=monitor, daemon=True).start()
def log(*args): print(round(time.monotonic()-start, 3), *args, flush=True)
logging.set_verbosity_error()
torch.set_num_threads(4)
common = 'Instrumental cinematic downtempo electronic score, repeating warm synthesizer motif, soft strings, A minor, 100 BPM, no vocals. '
prompts = {
'base': common + 'Steady restrained drums, calm flowing mood, medium intensity.',
'neutral': common + 'Steady restrained drums, calm flowing mood, medium intensity.',
'low': common + 'Very gentle sparse arrangement, soft sustained pads, minimal percussion, quiet relaxed mood.',
'high': common + 'Intense driving drums, urgent pulsing synthesizers, rising strings, dramatic tension and energy.',
'release': common + 'Drums recede, gentle sustained strings and warm pads, calm peaceful resolution, tension dissolves.',
}
path = str(ROOT/'musicgen-small-fp16')
processor = AutoProcessor.from_pretrained(path, local_files_only=True)
model = MusicgenForConditionalGeneration.from_pretrained(path, local_files_only=True, torch_dtype=torch.float16,
low_cpu_mem_usage=True, attn_implementation='eager').eval()
model.text_encoder.float(); model.enc_to_dec_proj.float(); model.audio_encoder.float()
log('LOADED')
conditions = {}; references = {}; conditioning_seconds = {}
with torch.no_grad():
for name, prompt in prompts.items():
t0 = time.monotonic()
inputs = processor(text=[prompt], padding='max_length', max_length=64, truncation=True, return_tensors='pt')
enc = model.text_encoder(**inputs).last_hidden_state
enc = model.enc_to_dec_proj(torch.cat([enc, torch.zeros_like(enc)], 0))
mask = torch.cat([inputs['attention_mask'], torch.zeros_like(inputs['attention_mask'])], 0)
enc = (enc * mask[..., None]).half()
conditions[name] = (enc, mask)
conditioning_seconds[name] = time.monotonic()-t0
references[name] = model.decoder(input_ids=torch.full((8,1),2048,dtype=torch.long),
encoder_hidden_states=enc, encoder_attention_mask=mask, use_cache=True).logits[:,-1].float().numpy()
log('CONDITION', name, conditioning_seconds[name])
# Retain official delay-mask builder without retaining the full CPU transformer.
build_mask = model.decoder.build_delay_pattern_mask
patterns = {}
# Empty base and dummy prefix establish mask layout before deleting CPU weights.
for n, prefix_n in [(600, 0), (800, 400)]:
inp = torch.cat([torch.full((4,1),2048,dtype=torch.long), torch.zeros((4,prefix_n),dtype=torch.long)], 1)
_, patterns[(n,prefix_n)] = build_mask(inp,2048,n+4)
native = NativeDecoder(model.decoder, *conditions['base'], 804)
del build_mask, model.decoder, model.text_encoder, model.enc_to_dec_proj
gc.collect()
log('GPU_LOADED', type(Device['AMD'].iface).__name__)
results = {'prompts':prompts, 'conditioning_seconds':conditioning_seconds, 'runs':[],
'hardware':{'backend':type(Device['AMD'].iface).__name__, 'arch':Device['AMD'].arch},
'cpu_affinity':list(os.sched_getaffinity(0)), 'torch_threads':torch.get_num_threads()}
def set_condition(name):
enc, mask = conditions[name]
native.enc.assign(Tensor(enc.numpy(),device='AMD')).realize()
native.mask.assign(Tensor(((1-mask.numpy())*-65504.).astype(np.float16),device='AMD').reshape(2,1,1,-1)).realize()
# Assign existing buffers: TinyJit must read changed content, not stale captured objects.
for i, (dstk,dstv) in enumerate(native.cross):
p = f'model.decoder.layers.{i}.encoder_attn.'
k = native.linear(native.enc,p+'k_proj').reshape(2,-1,16,64).transpose(1,2)
v = native.linear(native.enc,p+'v_proj').reshape(2,-1,16,64).transpose(1,2)
dstk.assign(k).realize(); dstv.assign(v).realize()
all_codes = {}
for name in ['base','neutral','low','high','release']:
run_start = time.monotonic()
set_condition(name)
condition_update_seconds = time.monotonic()-run_start
prefix = None if name=='base' else all_codes['high' if name=='release' else 'base'][...,-400:]
prefix_n = 0 if prefix is None else prefix.shape[-1]
n = 600 if name=='base' else 800
delay = patterns[(n,prefix_n)].clone()
if prefix is not None:
for c in range(4): delay[c,c+1:c+1+prefix_n] = prefix[0,0,c]
ids = torch.full((4,1),2048,dtype=torch.long)
# Same seed for neutral/low/high: paired branch experiment, not a diversity trick.
torch.manual_seed(321)
steps=[]; k0=GlobalCounters.kernel_count; g0=time.monotonic(); first_new=None
for t in range(1,n+4):
t0=time.monotonic()
raw=native(ids[:,-1:].repeat(2,1),t-1)
if t==1:
ref=references[name]
cosine=float(np.dot(raw.ravel(),ref.ravel())/(np.linalg.norm(raw)*np.linalg.norm(ref)))
assert np.isfinite(raw).all() and cosine>.999, (name,cosine)
log('REFERENCE',name,cosine)
if (delay[:,t]==-1).any():
if first_new is None: first_new=time.monotonic()-g0
logits=torch.from_numpy(raw); logits=logits[4:]+3*(logits[:4]-logits[4:])
vals,idx=logits.topk(250,dim=-1)
nxt=idx.gather(-1,torch.multinomial(vals.softmax(-1),1)).squeeze(-1)
nxt=torch.where(delay[:,t]!=-1,delay[:,t],nxt)
else: nxt=delay[:,t]
ids=torch.cat([ids,nxt[:,None]],1)
steps.append(time.monotonic()-t0)
if t%100==0: log('STEP',name,t,'elapsed',time.monotonic()-g0)
generation_seconds=time.monotonic()-g0
codes=ids[ids!=2048].reshape(1,1,4,-1)
assert codes.shape[-1]==n
if prefix is not None: assert torch.equal(codes[...,:prefix_n],prefix)
all_codes[name]=codes
torch.save(codes,OUT/f'{name}_codes.pt')
t0=time.monotonic()
audio=model.audio_encoder.decode(codes,audio_scales=[None]).audio_values[0,0].float().numpy()
decode_seconds=time.monotonic()-t0
assert np.isfinite(audio).all()
sf.write(OUT/f'{name}_full.wav',audio,32000,subtype='PCM_16')
sf.write(OUT/f'{name}_new.wav',audio[prefix_n*640:],32000,subtype='PCM_16')
run={'name':name,'prefix_source':None if prefix is None else ('high' if name=='release' else 'base'),
'prefix_seconds':prefix_n/50,'new_seconds':(n-prefix_n)/50,'total_audio_seconds':len(audio)/32000,
'condition_update_seconds':condition_update_seconds,'generation_seconds':generation_seconds,
'decode_seconds':decode_seconds,'total_run_seconds':time.monotonic()-run_start,
'prefix_plus_first_new_logits_seconds':first_new,'steady_step_seconds':float(np.mean(steps[4:])),
'rtf_per_new_audio':(generation_seconds+decode_seconds)/((n-prefix_n)/50),
'reference_cosine':cosine,'prefix_tokens_exact':prefix is not None,
'peak_host_rss_bytes':peak_rss,'peak_tracked_buffer_bytes':peak_buffers,
'gpu_dispatch_counter_delta':GlobalCounters.kernel_count-k0,
'rms':float(np.sqrt(np.mean(audio**2))),'peak':float(np.max(np.abs(audio))),
'clip_fraction':float(np.mean(np.abs(audio)>=1))}
results['runs'].append(run);results['elapsed_seconds']=time.monotonic()-start
(OUT/'result.json').write_text(json.dumps(results,indent=2))
log('RESULT',json.dumps(run))
log('COMPLETE')
@@ -1,13 +0,0 @@
#!/usr/bin/env bash
set -euo pipefail
cd /data/openpilot/tinygrad_repo
export PATH=/data/roadscore-feasibility/venv/bin:/usr/bin:/bin
export PYTHONPATH=/data/openpilot/tinygrad_repo
export DEV=USB+AMD:LLVM
export MAX_JOBS=1 CXX=clang++
export TORCH_EXTENSIONS_DIR=/data/roadscore-feasibility/cache/torch_extensions
export XDG_CACHE_HOME=/data/roadscore-feasibility/cache
export HF_HOME=/data/roadscore-feasibility/cache/huggingface
export TMPDIR=/data/roadscore-feasibility/tmp
export OMP_NUM_THREADS=2 OPENBLAS_NUM_THREADS=2
exec timeout 1400 python -u /data/roadscore-feasibility/controllability_probe.py "$@"
@@ -1,13 +0,0 @@
#!/usr/bin/env bash
set -euo pipefail
cd /data/openpilot/tinygrad_repo
export PATH=/data/roadscore-feasibility/venv/bin:/usr/bin:/bin
export PYTHONPATH=/data/openpilot/tinygrad_repo
export DEV=USB+AMD:LLVM
export MAX_JOBS=1 CXX=clang++
export TORCH_EXTENSIONS_DIR=/data/roadscore-feasibility/cache/torch_extensions
export XDG_CACHE_HOME=/data/roadscore-feasibility/cache
export HF_HOME=/data/roadscore-feasibility/cache/huggingface
export TMPDIR=/data/roadscore-feasibility/tmp
export OMP_NUM_THREADS=2 OPENBLAS_NUM_THREADS=2
exec timeout 300 python -u /data/roadscore-feasibility/verify.py "$@"
@@ -1,73 +0,0 @@
"""Check official prefix-delay mapping and cached continuation logits after prompt replacement."""
import gc,json,time
from pathlib import Path
import numpy as np
import torch
from transformers import AutoProcessor,MusicgenForConditionalGeneration
from tinygrad import Tensor
from native_decoder import NativeDecoder
ROOT=Path('/data/roadscore-feasibility');OUT=ROOT/'controllability'
result=json.loads((OUT/'result.json').read_text())
torch.set_num_threads(4)
path=str(ROOT/'musicgen-small-fp16')
model=MusicgenForConditionalGeneration.from_pretrained(path,local_files_only=True,torch_dtype=torch.float16,low_cpu_mem_usage=True,attn_implementation='eager').eval()
processor=AutoProcessor.from_pretrained(path,local_files_only=True)
model.text_encoder.float();model.enc_to_dec_proj.float()
checks={};inputs={};conditions={};refs={}
prompt_lengths={name:len(processor.tokenizer(prompt)['input_ids']) for name,prompt in result['prompts'].items()}
assert max(prompt_lengths.values())<=64, prompt_lengths
(OUT/'prompt_token_lengths.json').write_text(json.dumps(prompt_lengths,indent=2))
with torch.no_grad():
for name in ['neutral','low','high','release']:
parent='high' if name=='release' else 'base'
prefix=torch.load(OUT/f'{parent}_codes.pt',weights_only=True)[...,-400:]
child=torch.load(OUT/f'{name}_codes.pt',weights_only=True)
assert torch.equal(child[...,:400],prefix)
initial,mask=model.decoder.build_delay_pattern_mask(torch.cat([torch.full((4,1),2048,dtype=torch.long),prefix.reshape(4,400)],1),2048,804)
manual=torch.full((4,804),-1,dtype=torch.long)
for c in range(4):
manual[c,:c+1]=2048;manual[c,c+1:c+401]=prefix[0,0,c];manual[c,801+c:]=2048
assert torch.equal(manual,mask)
checks[name]={'exact_prefix_codes':True,'official_delay_mask_equal':True}
if name not in ['low','high']:continue
# Shorten the test prefix to 8 frames, then check four sampled continuation inputs.
# Retain actual generated token inputs but rebuild official BOS/delay positions.
short=prefix[...,:8].reshape(4,8)
_,shortmask=model.decoder.build_delay_pattern_mask(torch.cat([torch.full((4,1),2048,dtype=torch.long),short],1),2048,32)
seq=shortmask[:,:16].clone()
replacement=child.reshape(4,-1)[:,400:416]
seq=torch.where(seq==-1,replacement,seq)
inputs[name]=seq.repeat(2,1)
txt=processor(text=[result['prompts'][name]],padding='max_length',max_length=64,truncation=True,return_tensors='pt')
enc=model.text_encoder(**txt).last_hidden_state
enc=model.enc_to_dec_proj(torch.cat([enc,torch.zeros_like(enc)],0))
mask=torch.cat([txt['attention_mask'],torch.zeros_like(txt['attention_mask'])],0)
enc=(enc*mask[...,None]).half();conditions[name]=(enc,mask)
refs[name]=model.decoder(input_ids=inputs[name],encoder_hidden_states=enc,encoder_attention_mask=mask,use_cache=True).logits[:,-4:].float().numpy()
native=NativeDecoder(model.decoder,*conditions['low'],32)
del model;gc.collect()
for name in ['low','high']:
enc,mask=conditions[name]
native.enc.assign(Tensor(enc.numpy(),device='AMD')).realize()
native.mask.assign(Tensor(((1-mask.numpy())*-65504.).astype(np.float16),device='AMD').reshape(2,1,1,-1)).realize()
for i,(k,v) in enumerate(native.cross):
p=f'model.decoder.layers.{i}.encoder_attn.'
k.assign(native.linear(native.enc,p+'k_proj').reshape(2,-1,16,64).transpose(1,2)).realize()
v.assign(native.linear(native.enc,p+'v_proj').reshape(2,-1,16,64).transpose(1,2)).realize()
raws=[]
for pos in range(16):
raw=native(inputs[name][:,pos:pos+1],pos)
if pos>=12: raws.append(raw)
raw=np.stack(raws,axis=1);ref=refs[name]
checks[name]['cached_continuation_cosine']=float(np.dot(raw.ravel(),ref.ravel())/(np.linalg.norm(raw)*np.linalg.norm(ref)))
checks[name]['cached_continuation_mae']=float(np.abs(raw-ref).mean())
assert checks[name]['cached_continuation_cosine']>.999,checks[name]
print(name,checks[name],flush=True)
(OUT/'validation.json').write_text(json.dumps(checks,indent=2))
divergence={}
for a,b in [('neutral','low'),('neutral','high'),('low','high')]:
ca=torch.load(OUT/f'{a}_codes.pt',weights_only=True)[...,400:]
cb=torch.load(OUT/f'{b}_codes.pt',weights_only=True)[...,400:]
divergence[f'{a}_vs_{b}']=float((ca!=cb).float().mean())
(OUT/'token_divergence.json').write_text(json.dumps(divergence,indent=2))
print('CONTINUATION_CHECKS_PASSED',flush=True)
-12
View File
@@ -1,12 +0,0 @@
"""Download model weights only; all inference runs locally. Run in experiment venv."""
import os
os.environ.setdefault('HF_HOME', '/data/roadscore-feasibility/cache/huggingface')
os.environ.setdefault('HF_HUB_DISABLE_XET', '1')
from huggingface_hub import snapshot_download
print(snapshot_download(
'facebook/musicgen-small',
revision='4c8334b02c6ec4e8664a91979669a501ec497792',
local_dir='/data/roadscore-feasibility/musicgen-small',
allow_patterns=['*.json', '*.model', 'model.safetensors'],
max_workers=2,
))
@@ -1,67 +0,0 @@
"""Isolated hardware gate; CPU text/codec, actual Chestnut music transformer."""
import time,json,threading,gc,argparse,os
from pathlib import Path
import psutil
p=argparse.ArgumentParser()
p.add_argument('--seconds',type=float,default=8);p.add_argument('--repeats',type=int,default=2)
p.add_argument('--prompt',default='Instrumental cinematic electronic music, warm synthesizer chords, melodic strings, steady drums, no vocals')
p.add_argument('--tag',default='bench8');a=p.parse_args()
root=Path('/data/roadscore-feasibility'); start=time.monotonic(); stages={};peak_rss=0;peak_buffers=0
proc=psutil.Process()
def monitor():
global peak_rss,peak_buffers
while True:
peak_rss=max(peak_rss,proc.memory_info().rss)
if 'GlobalCounters' in globals(): peak_buffers=max(peak_buffers,GlobalCounters.mem_used)
time.sleep(.05)
threading.Thread(target=monitor,daemon=True).start()
def mark(s):
stages[s]=time.monotonic()-start;print(s,round(stages[s],3),'RSS_MB',round(proc.memory_info().rss/1e6),flush=True)
import torch,numpy as np,soundfile as sf
from transformers import AutoProcessor,MusicgenForConditionalGeneration
from transformers.utils import logging
from tinygrad import Device
from tinygrad.helpers import GlobalCounters
from native_decoder import NativeDecoder
logging.set_verbosity_error();mark('imports');torch.set_num_threads(4)
path=str(root/'musicgen-small-fp16')
processor=AutoProcessor.from_pretrained(path,local_files_only=True)
model=MusicgenForConditionalGeneration.from_pretrained(path,local_files_only=True,torch_dtype=torch.float16,low_cpu_mem_usage=True,attn_implementation='eager').eval();mark('loaded')
model.text_encoder.float();model.enc_to_dec_proj.float();model.audio_encoder.float();gc.collect()
with torch.no_grad():
t0=time.monotonic()
inputs=processor(text=[a.prompt],padding=True,return_tensors='pt')
enc=model.text_encoder(**inputs).last_hidden_state
enc=torch.cat([enc,torch.zeros_like(enc)],0);enc=model.enc_to_dec_proj(enc)
mask=torch.cat([inputs['attention_mask'],torch.zeros_like(inputs['attention_mask'])],0)
enc=(enc*mask[...,None]).half();conditioning_seconds=time.monotonic()-t0;mark('conditioned')
n=round(a.seconds*50);length=n+4
initial,delay=model.decoder.build_delay_pattern_mask(torch.full((4,1),2048,dtype=torch.long),2048,length)
reference=model.decoder(input_ids=initial.repeat(2,1),encoder_hidden_states=enc,encoder_attention_mask=mask,use_cache=True,return_dict=True).logits[:,-1].float().numpy();mark('reference')
native=NativeDecoder(model.decoder,enc,mask,length)
del model.decoder;gc.collect();mark('gpu_loaded')
dev=Device['AMD'];hardware=dict(backend=type(dev.iface).__name__,arch=dev.arch,vram_bytes=dev.iface.dev_impl.vram_size)
runs=[]
for repeat in range(a.repeats):
seed=123+repeat;torch.manual_seed(seed);ids=initial.clone();step_seconds=[];k0=GlobalCounters.kernel_count;g0=time.monotonic()
for t in range(1,length):
t0=time.monotonic();x=ids[:,-1:].repeat(2,1);raw=native(x,t-1)
if t<=4: np.savez(root/f'{a.tag}_r{repeat}_step{t}.npz',inputs=x.numpy(),logits=raw)
if t==1:
compare=dict(mae=float(np.abs(raw-reference).mean()),max_error=float(np.abs(raw-reference).max()),cosine=float(np.dot(raw.ravel(),reference.ravel())/(np.linalg.norm(raw)*np.linalg.norm(reference))))
print('REFERENCE_COMPARE',json.dumps(compare),flush=True)
assert np.isfinite(raw).all() and compare['cosine']>.999,compare
logits=torch.from_numpy(raw);logits=logits[4:]+3.0*(logits[:4]-logits[4:])
vals,idx=logits.topk(250,dim=-1);nxt=idx.gather(-1,torch.multinomial(vals.softmax(-1),1)).squeeze(-1)
nxt=torch.where(delay[:,t]!=-1,delay[:,t],nxt);ids=torch.cat([ids,nxt[:,None]],dim=1)
step_seconds.append(time.monotonic()-t0)
if t<=4 or t%50==0: print('RUN',repeat,'STEP',t,'elapsed',time.monotonic()-g0,'kernels',GlobalCounters.kernel_count-k0,flush=True)
generation_seconds=time.monotonic()-g0;mark(f'generated_{repeat}')
codes=ids[ids!=2048].reshape(1,1,4,-1);torch.save(codes,root/f'{a.tag}_r{repeat}_codes.pt')
t0=time.monotonic();audio=model.audio_encoder.decode(codes,audio_scales=[None]).audio_values[0,0].float().numpy();decode_seconds=time.monotonic()-t0
assert np.isfinite(audio).all()
out=root/f'{a.tag}_r{repeat}.wav';sf.write(out,audio,32000,subtype='PCM_16');mark(f'decoded_{repeat}')
run=dict(seed=seed,generation_seconds=generation_seconds,decode_seconds=decode_seconds,audio_seconds=len(audio)/32000,rtf=(generation_seconds+decode_seconds)/(len(audio)/32000),step_seconds=step_seconds,steady_seconds_per_step=float(np.mean(step_seconds[4:])),peak_host_rss_bytes=peak_rss,peak_tinygrad_buffer_bytes=peak_buffers,active_tinygrad_buffer_bytes=GlobalCounters.mem_used,gpu_kernel_counter_delta=GlobalCounters.kernel_count-k0,reference_compare=compare,audio_rms=float(np.sqrt(np.mean(audio**2))),audio_peak=float(np.max(np.abs(audio))),clip_fraction=float(np.mean(np.abs(audio)>=1)),output=str(out))
runs.append(run)
result=dict(model='facebook/musicgen-small',revision='4c8334b02c6ec4e8664a91979669a501ec497792',prompt=a.prompt,hardware=hardware,cpu_affinity=list(os.sched_getaffinity(0)),torch_threads=torch.get_num_threads(),conditioning_seconds=conditioning_seconds,stages=stages,runs=runs)
(root/f'{a.tag}_result.json').write_text(json.dumps(result,indent=2));print('RESULT',json.dumps({k:v for k,v in run.items() if k!='step_seconds'}),flush=True)
@@ -1,56 +0,0 @@
"""Narrow experimental MusicGen Small decoder, weights from Transformers. No training."""
import numpy as np
from tinygrad import Tensor, TinyJit, Variable, dtypes
class NativeDecoder:
def __init__(self, decoder, conditioning, mask, max_length):
self.weights={}
for name,p in decoder.named_parameters():
self.weights[name]=Tensor(p.detach().numpy().copy(),device='AMD').realize()
self.max_length=max_length
self.enc=Tensor(conditioning.numpy().astype(np.float16),device='AMD').realize()
self.mask=Tensor(((1-mask.numpy())*-65504.).astype(np.float16),device='AMD').reshape(2,1,1,-1).realize()
self.caches=[]; self.cross=[]
for i in range(24):
prefix=f'model.decoder.layers.{i}.'
self.caches.append(Tensor.zeros(2,2,16,max_length,64,device='AMD',dtype=dtypes.float16).contiguous().realize())
k=self.linear(self.enc,prefix+'encoder_attn.k_proj').reshape(2,-1,16,64).transpose(1,2)
v=self.linear(self.enc,prefix+'encoder_attn.v_proj').reshape(2,-1,16,64).transpose(1,2)
k.realize(v); self.cross.append((k,v))
self.run_jit=TinyJit(self.forward)
def linear(self,x,prefix):
out=x.linear(self.weights[prefix+'.weight'].T)
if prefix+'.bias' in self.weights: out=out+self.weights[prefix+'.bias']
return out
def norm(self,x,prefix):
return x.float().layernorm(eps=1e-5).cast(x.dtype)*self.weights[prefix+'.weight']+self.weights[prefix+'.bias']
def attention(self,q,k,v):
scores=(q*0.125)@k.transpose(-1,-2)
return scores.float().softmax(-1).cast(q.dtype)@v
def forward(self,ids,pos):
x=sum(self.weights[f'model.decoder.embed_tokens.{c}.weight'][ids[:,c]] for c in range(4)).reshape(2,1,1024)
x=x+self.weights['model.decoder.embed_positions.weights'][pos:pos+1]
for i in range(24):
p=f'model.decoder.layers.{i}.'
z=self.norm(x,p+'self_attn_layer_norm')
q=self.linear(z,p+'self_attn.q_proj').reshape(2,1,16,64).transpose(1,2)
k=self.linear(z,p+'self_attn.k_proj').reshape(2,1,16,64).transpose(1,2)
v=self.linear(z,p+'self_attn.v_proj').reshape(2,1,16,64).transpose(1,2)
cache=self.caches[i]
cache[:,:,:,pos:pos+1,:].assign(Tensor.stack(k,v)).realize()
a=self.attention(q,cache[0,:,:,:pos+1,:],cache[1,:,:,:pos+1,:]).transpose(1,2).reshape(2,1,1024)
x=x+self.linear(a,p+'self_attn.out_proj')
z=self.norm(x,p+'encoder_attn_layer_norm')
q=self.linear(z,p+'encoder_attn.q_proj').reshape(2,1,16,64).transpose(1,2)
k,v=self.cross[i]
scores=(q*0.125)@k.transpose(-1,-2)+self.mask
a=(scores.float().softmax(-1).cast(q.dtype)@v).transpose(1,2).reshape(2,1,1024)
x=x+self.linear(a,p+'encoder_attn.out_proj')
z=self.norm(x,p+'final_layer_norm')
x=(x+self.linear(self.linear(z,p+'fc1').gelu(approximate='none'),p+'fc2')).realize()
x=self.norm(x,'model.decoder.layer_norm')
return Tensor.stack(*[self.linear(x,f'lm_heads.{c}') for c in range(4)],dim=1).reshape(8,2048).realize()
def __call__(self,ids,pos):
t=Tensor(ids.numpy().astype(np.int32).reshape(2,4),device='AMD').realize()
result=self.run_jit(t,Variable('pos',0,self.max_length-1).bind(pos))
return result.float().numpy()
-61
View File
@@ -1,61 +0,0 @@
"""Temporary offroad benchmark CPU settings; restore on exit. No manager changes."""
import json
import os
import signal
import subprocess
import sys
from pathlib import Path
root = Path('/data/roadscore-feasibility')
def is_offroad():
check = subprocess.run(
['/usr/local/venv/bin/python', '-c',
'from openpilot.common.params import Params; raise SystemExit(int(Params().get_bool("IsOnroad")))'],
cwd='/data/openpilot', check=False,
)
return check.returncode == 0
if not is_offroad():
raise SystemExit('This isolated benchmark requires a verified offroad device.')
paths = [Path(f'/sys/devices/system/cpu/cpu{i}/online') for i in range(4, 8)]
paths += [Path('/sys/devices/system/cpu/cpufreq/policy4') / n for n in ['scaling_max_freq', 'scaling_governor']]
saved = {str(p): p.read_text().strip() for p in paths}
(root / 'power_before.json').write_text(json.dumps(saved, indent=2))
def write(path, value):
subprocess.run(['sudo', '-n', 'tee', str(path)], input=value + '\n', text=True, stdout=subprocess.DEVNULL, check=True)
def stop(sig, frame):
raise KeyboardInterrupt
signal.signal(signal.SIGTERM, stop)
child = None
rc = 1
try:
for path in paths[:4]:
write(path, '1')
write(paths[4], '1689600')
write(paths[5], 'performance')
print('POWER', json.dumps({str(p): p.read_text().strip() for p in paths}), flush=True)
child = subprocess.Popen(['taskset', '-c', '4-7', 'bash', str(root / 'run_native.sh'), *sys.argv[1:]], start_new_session=True)
rc = child.wait()
finally:
try:
if child is not None and child.poll() is None:
os.killpg(child.pid, signal.SIGTERM)
try:
child.wait(timeout=15)
except subprocess.TimeoutExpired:
os.killpg(child.pid, signal.SIGKILL)
child.wait()
finally:
if is_offroad():
for path in reversed(paths):
write(path, saved[str(path)])
(root / 'power_after.json').write_text(json.dumps({str(p): p.read_text().strip() for p in paths}, indent=2))
print('POWER_RESTORED', flush=True)
else:
print('Device no longer verified offroad; leaving active CPU settings to hardwared.', flush=True)
sys.exit(rc)
-13
View File
@@ -1,13 +0,0 @@
#!/usr/bin/env bash
set -euo pipefail
cd /data/openpilot/tinygrad_repo
export PATH=/data/roadscore-feasibility/venv/bin:/usr/bin:/bin
export PYTHONPATH=/data/openpilot/tinygrad_repo
export DEV=USB+AMD:LLVM
export MAX_JOBS=1 CXX=clang++
export TORCH_EXTENSIONS_DIR=/data/roadscore-feasibility/cache/torch_extensions
export XDG_CACHE_HOME=/data/roadscore-feasibility/cache
export HF_HOME=/data/roadscore-feasibility/cache/huggingface
export TMPDIR=/data/roadscore-feasibility/tmp
export OMP_NUM_THREADS=2 OPENBLAS_NUM_THREADS=2
exec timeout 900 python -u /data/roadscore-feasibility/musicgen_native.py "$@"
-26
View File
@@ -1,26 +0,0 @@
import json,struct,shutil
from pathlib import Path
import numpy as np
from safetensors.numpy import save_file
src=Path('/data/roadscore-feasibility/musicgen-small'); dst=Path('/data/roadscore-feasibility/musicgen-small-fp16'); dst.mkdir(exist_ok=True)
for p in list(src.glob('*.json'))+list(src.glob('*.model')): shutil.copy2(p,dst/p.name)
weight_map={}; shard={}; size=0; total=0; idx=0
def flush():
global shard,size,idx
if not shard: return
name=f'model-{idx:03}.safetensors'; save_file(shard,str(dst/name),metadata={'format':'pt'})
weight_map.update({k:name for k in shard}); print(name,size,flush=True);idx+=1;shard={};size=0
with open(src/'model.safetensors','rb') as f:
n=struct.unpack('<Q',f.read(8))[0]; h=json.loads(f.read(n)); start=8+n
for key,info in h.items():
if key=='__metadata__': continue
lo,hi=info['data_offsets'];f.seek(start+lo)
dt={'F32':np.float32,'F16':np.float16,'I64':np.int64,'I32':np.int32}[info['dtype']]
arr=np.frombuffer(f.read(hi-lo),dtype=dt).reshape(info['shape'])
if dt==np.float32: arr=arr.astype(np.float16)
if size+arr.nbytes>64*1024**2: flush()
shard[key]=arr;size+=arr.nbytes;total+=arr.nbytes
flush()
(dst/'model.safetensors.index.json').write_text(json.dumps({'metadata':{'total_size':total},'weight_map':weight_map}))
print('DONE',total)
-29
View File
@@ -1,29 +0,0 @@
import time,json
from pathlib import Path
import numpy as np,torch
from transformers import AutoProcessor,MusicgenForConditionalGeneration
from transformers.utils import logging
logging.set_verbosity_error();torch.set_num_threads(2)
root=Path('/data/roadscore-feasibility');path=str(root/'musicgen-small-fp16')
report=json.loads((root/'bench8_result.json').read_text())
processor=AutoProcessor.from_pretrained(path,local_files_only=True)
model=MusicgenForConditionalGeneration.from_pretrained(path,local_files_only=True,torch_dtype=torch.float16,low_cpu_mem_usage=True,attn_implementation='eager').eval()
model.text_encoder.float();model.enc_to_dec_proj.float()
checks=[]
with torch.no_grad():
inputs=processor(text=[report['prompt']],padding=True,return_tensors='pt')
enc=model.text_encoder(**inputs).last_hidden_state
enc=torch.cat([enc,torch.zeros_like(enc)],0);enc=model.enc_to_dec_proj(enc)
mask=torch.cat([inputs['attention_mask'],torch.zeros_like(inputs['attention_mask'])],0)
enc=(enc*mask[...,None]).half()
for r in range(2):
cache=None
for step in range(1,5):
d=np.load(root/f'bench8_r{r}_step{step}.npz')
out=model.decoder(input_ids=torch.from_numpy(d['inputs']),encoder_hidden_states=enc,encoder_attention_mask=mask,past_key_values=cache,use_cache=True,return_dict=True)
cache=out.past_key_values;ref=out.logits[:,-1].float().numpy();actual=d['logits']
c=dict(run=r,step=step,mae=float(np.abs(ref-actual).mean()),max_error=float(np.abs(ref-actual).max()),cosine=float(np.dot(ref.ravel(),actual.ravel())/(np.linalg.norm(ref)*np.linalg.norm(actual))))
checks.append(c);print(c,flush=True)
assert np.isfinite(actual).all() and c['cosine']>.999,c
(root/'validation.json').write_text(json.dumps({'passed':True,'checks':checks},indent=2))
print('ALL_CACHED_STEPS_MATCH',flush=True)
@@ -1,113 +0,0 @@
# Stable Audio on Chestnut: bounded feasibility investigation
## Decision
**The faster-than-playback gate is unresolved because model-weight access is blocked, not because an architectural incompatibility was found.** No Stable Audio WAV was generated, and no end-to-end model timing is claimed.
Stable Audio 3 Small-Music is a credible candidate for a narrow native tinygrad implementation. Representative diffusion and audio-decoder blocks ran on the actual Chestnut USB AMD backend with encouraging warm latency. This supports pursuing one full generation after weight access is available; it does not justify replacing the working MusicGen implementation yet. Keep the MusicGen hybrid as the demonstrated fallback. Do not begin RoadScore production implementation from this result.
Only the two requested model families were investigated. MusicGen was not optimized or modified. All additions are in this experimental directory; the device scratch directory is `/data/sa3-feasibility`. CPU settings were restored after the hardware test.
## Blocking evidence
Authenticated requests using the already-configured Hugging Face credential received **HTTP 403, account not in the authorized list**, for both:
- [Stable Audio 3 Small-Music](https://huggingface.co/stabilityai/stable-audio-3-small-music)
- [Stable Audio Open Small](https://huggingface.co/stabilityai/stable-audio-open-small)
The credential itself was not printed, copied to the device, or saved in results. See [access.json](results/access.json). The access requests were raised during the investigation and the checks were repeated before reporting. The owner needs to accept the repository access conditions with the account used by that credential; fine-grained token permissions may also need adjustment. No alternate restricted download route was attempted.
SA3 redistributes its T5Gemma assets in the same repository, so access to that repository is also needed for its released text-conditioning path. The public repository file listing was available; the actual checkpoint/configuration files were not.
## Actual architecture inspected
Primary implementation: [Stability-AI/stable-audio-3](https://github.com/Stability-AI/stable-audio-3), pinned source commit `779434a908193105335fd8d833418603625b2859`.
The most useful reference for a narrow port is the project's own self-contained [MLX DiT implementation](https://github.com/Stability-AI/stable-audio-3/blob/779434a908193105335fd8d833418603625b2859/optimized/mlx/models/defs/dit_mlx.py), plus its SAME-S and sampling implementations. MLX itself will not run on Chestnut; the reusable part is the explicitly expressed model math and weight mapping.
| Component | Released implementation | Narrow tinygrad route |
|---|---|---|
| Diffusion model | 20 layers, width 1024, 16 heads, 64-dimensional heads, standard self/cross attention, 4096-wide SwiGLU inner dimension | Native matmuls, RMS normalization, RoPE, FP32 softmax, SiLU, residuals and conditional scale/shift/gates |
| Conditioning | 768-dimensional T5Gemma embeddings, learned prompt padding, duration and timestep Fourier embeddings, 64 memory tokens | CPU tokenizer/text encoder initially; port encoder only if CPU latency prevents the gate |
| Sampling | Eight ping-pong rectified-flow steps; normal inference example uses guidance 1 | A small explicit update loop; keep latents on Chestnut between steps |
| Inpainting | Per-position mask plus 256-channel source latents, projected into each DiT layer | Existing linear and masking operations; no new attention primitive |
| SAME-S decoder | Six 768-wide transformer blocks, differential attention, DyT normalization, SwiGLU, shifted local chunks | Two attention products and subtraction; tanh affine normalization; reshape/slice/cat; pre-fused weight-normalized convolution |
| Waveform reconstruction | 256-dimensional latents → 16 stereo audio patches per latent → 4096 waveform samples per latent | Native decoder, then reshape and CPU WAV write; no separate autoregressive vocoder |
At 44.1 kHz, one latent represents 4096 samples (~10.77 latent frames/second). Eight denoising passes process the whole latent sequence, rather than hundreds or thousands of sequential token predictions. This is the architectural reason to expect a substantially different speed regime from MusicGen. It is not proof of a particular wall time.
References: [SAME-S decoder](https://github.com/Stability-AI/stable-audio-3/blob/779434a908193105335fd8d833418603625b2859/optimized/mlx/models/defs/same_s_decoder.py), [sampling and waveform reconstruction](https://github.com/Stability-AI/stable-audio-3/blob/779434a908193105335fd8d833418603625b2859/optimized/mlx/models/defs/sa3_pipeline.py), [T5Gemma encoder](https://github.com/Stability-AI/stable-audio-3/blob/779434a908193105335fd8d833418603625b2859/optimized/mlx/models/defs/t5gemma_mlx.py).
## tinygrad / USB compatibility assessment
No obvious fundamental operator blocker was found for Small-Music. The repository already provides matmul, softmax, sin/cos, tanh, SiLU, reductions, slicing/reshaping/concatenation, convolutions and TinyJit. RMSNorm and DyT can be expressed directly. Differential attention is two ordinary attention results subtracted; it does not require a custom CUDA kernel. Weight normalization can be folded into convolution weights during conversion.
The synthetic test actually exercised the attention, normalization, RoPE, feed-forward, conditional gating and local-conditioning projections on `USBIface`, `gfx1200`. It did **not** exercise the entire decoder's chunk-shift/reconstruction pipeline, the final convolution, real checkpoint loading, the sampler, or text encoding. There is no numerical parity claim against the actual SA3 checkpoint.
The existing transport is tinygrad's custom USB AMD route. CUDA/TensorRT, Apple's CoreML/MLX, and the CPU LiteRT/XNNPACK release are not drop-in Chestnut backends. Small-Music uses standard attention; the Medium model's different backend requirements are not a reason to reject Small. A general PyTorch backend compatibility layer is unnecessary for the proposed route.
## Measured synthetic hardware test — not music generation
[Probe](layer_probe.py), [raw timings](results/layer_probe.json), [log](results/layer_probe.log).
| Representative block | Shape | First call, compilation included | Capture call | Warm median, three replays |
|---|---|---:|---:|---:|
| SA3 DiT | `[1, 384, 1024]` (320 latents + 64 memory positions) | 22.558 s | 1.231 s | **10.774 ms** |
| SAME-S decoder | `[160, 34, 768]` (first-half chunk layout for 320 latents) | 21.699 s | 0.940 s | **23.331 ms** |
The represented latent length would correspond to 29.72 seconds of audio **if it were generated and decoded**. No audio was generated by this probe. The weights and inputs are deterministic random arrays, not downloaded trained parameters. Device synchronization is included in timings. Synthetic block initialization took 3.52 and 1.75 seconds respectively.
Simple arithmetic, `20 × 8 × 10.774 ms + 6 × 23.331 ms`, gives about **1.86 seconds for those repeated block workloads**. This is only a screening extrapolation. It omits text encoding, input/output projections, sampler work, codec mapping and layout changes, loading, full-model memory/cache effects, and distinct learned weights. It is neither a measured generation time nor a reliable lower/upper bound. It is encouraging enough that rejecting SA3 as inherently too slow would be unjustified.
Peak sampled host RSS for the probe was **185,114,624 bytes**; peak tracked tinygrad buffers were **178,425,616 bytes**. Those are **single-block test allocations, not model memory requirements**. They must not be compared with the full MusicGen model's RAM/VRAM as though they represented equivalent workloads. Complete physical VRAM usage was not measured.
## Requested success-gate record
| Metric | SA3 Small-Music | Stable Audio Open Small |
|---|---|---|
| Actual generated duration | None | None |
| End-to-end generation time | Not measured: weights inaccessible | Not measured: weights inaccessible |
| Waveform decode time | Not measured | Not measured |
| Cold/warm full-model performance | Not measured | Not measured |
| Full-model host RAM | Not measured | Not measured |
| Full-model tracked Chestnut allocation | Not measured | Not measured |
| Faster-than-playback success | **Not demonstrated** | **Not demonstrated** |
| CPU versus Chestnut | Only synthetic block math ran on Chestnut; host prepared inputs and dispatched | Source inspection only |
For an actual SA3 attempt, the preferred initial split is CPU text/tokenization plus Chestnut DiT **and SAME-S decode**, followed by CPU WAV writing. Leaving a slow general PyTorch audio decoder on the comma CPU could erase the diffusion speed advantage. Text encoding must be included in a fresh-prompt benchmark; cached-conditioning results should be reported separately.
## Continuation and audio conditioning
SA3 explicitly supports both. Its [SAME-S encoder](https://github.com/Stability-AI/stable-audio-3/blob/779434a908193105335fd8d833418603625b2859/optimized/mlx/models/defs/same_s_encoder.py) reuses the decoder's six-block transformer machinery, with the direction and projections changed. Audio-to-audio initializes from encoded audio with chosen noise. Inpainting/continuation supplies masked source latents to the DiT; the published sampler can restore retained latents after sampling.
That is a small extension **after** a correct text-to-audio/decoder port, rather than a new generative architecture. Reusing previously generated SA3 latents may avoid waveform re-encoding. MusicGen codec tokens cannot be used as SA3 latents; a MusicGen WAV would need SAME-S encoding. Musical continuity, boundary quality and conditioning speed remain untested here.
## Secondary candidate assessment
[Stable Audio Open Small](https://huggingface.co/stabilityai/stable-audio-open-small) is also a credible few-step design, but a less compelling first choice for this music project. Its model card explicitly rates its sound-effect/field-recording performance above music. It supports short stereo generation and eight-step ping-pong sampling.
Its [paper](https://arxiv.org/html/2505.08175v1) describes a 16-layer, 1024-wide diffusion transformer, a 64-channel 21.5 Hz latent space, a separate audio autoencoder and T5 text encoder. The paper also demonstrates audio-to-audio initialization. It does not establish SA3-style trained inpainting as an interchangeable feature of this checkpoint.
The CPU headline is not a comma benchmark: the paper's edge result uses a Vivo X200 Pro with Cortex-X925/X4/A720 cores and 12 GB RAM, plus selective dynamic INT8. Reported runtime falls from 15.3 to 6.6 seconds and peak RAM from 6.5 to 3.6 GB. That newer phone and memory budget cannot be equated with this comma's ~3.5 GiB RAM and older CPU cores. A Chestnut native port could still be fast; the CPU result alone does not establish that.
The [official Arm example](https://github.com/ARM-software/ML-examples/tree/main/kleidiai-examples/audiogen) exports conditioning, DiT and autoencoder separately and runs them through LiteRT/XNNPACK. Source commit inspected: `0e453847ea28043194c32791a591d0f5a8a5ae2e`. Stable Audio Tools source commit inspected: `3241adba4fc2a85cf5b29d9eb68d42f40a28e820`. The export requires the gated model configuration and weights; no ready-to-run local path was available without those assets. No secondary hardware benchmark was attempted.
## Port size and decisive next test
**Complexity estimate: moderate, model-specific work; not a general runtime project.** SA3 needs a weight loader/converter, full DiT, SAME-S decoder, exact sampling/conditioning glue, and stage-level reference comparisons. The reference code makes a few hundred lines of core tensor math plausible, but validation and dependency integration are meaningful additional work. A one-session port is plausible, not guaranteed. The synthetic block probe is not that port.
Once access is enabled, the narrow experiment should be limited to one fixed ~30-second shape, batch one, eight steps, guidance one, and two seeds. Stream/shard weights to avoid loading a multi-gigabyte FP32 checkpoint and a full duplicate in host RAM. Keep the working MusicGen Python environment intact; T5Gemma requires a different/newer text runtime or a narrow native encoder.
Validate one DiT forward and a decoder slice against the released reference before judging generated audio. Measure cold startup to WAV, fresh-prompt warm generation, cached-prompt generation, GPU decode, host RSS and tracked allocations separately. Require text → trained Chestnut inference → listenable WAV, with warm end-to-end time below audio duration. If weights or correctness cannot be established in the bounded attempt, retain MusicGen rather than inferring success from layer arithmetic.
Full-model GPU sharing with openpilot is still unproven; nothing here changes the earlier exclusive-device-ownership finding.
## Reproduce the completed block probe
The probe reuses only the first gate's isolated Python environment and tinygrad source. Copy this directory's `layer_probe.py`, `run_probe.sh`, and `power_probe.py` into `/data/sa3-feasibility`, then, while offroad and with Chestnut available:
```sh
ssh comma@192.168.3.111 '/usr/local/venv/bin/python /data/sa3-feasibility/power_probe.py'
```
The wrapper temporarily enables the normal capped big-core configuration, enforces a six-minute subprocess timeout, and restores the prior CPU settings. Before/after snapshots match. No driving parameters, manager processes, production code, or MusicGen implementation were changed.
-89
View File
@@ -1,89 +0,0 @@
# SA3 Small-Music trained-weight Chestnut experiment
## Result
**Warm 30-second generation passes the numerical speed gate: 18.53 seconds to a playable WAV, RTF 0.618.** The actual trained diffusion model and SAME-S waveform decoder ran on Chestnut through native tinygrad USB AMD (`USBIface`, `gfx1200`). Human listening acceptance is still required; waveform validity and reference agreement do not establish musical quality by themselves.
CPU text encoding took 3.761 seconds. Adding this measured stage gives 22.29 seconds / 30 seconds = **0.743 RTF**. This is a sum of separately measured stages, not a fresh prompt re-encoded during each warm run. The encoder was unloaded before GPU model loading to limit host memory. A production process that reloads the encoder for each changed prompt will incur additional startup cost.
No RoadScore production code or MusicGen code/environment was modified. The secondary model was not pursued because SA3 now demonstrates the requested speed. This is a bounded feasibility prototype, not a production integration.
## Listening files
- [30 seconds, seed 991](trained_results/trained_30s_r0.wav)
- [30 seconds, seed 992](trained_results/trained_30s_r1.wav)
- [30 seconds, seed 993, fully warm benchmark](trained_results/trained_30s_r2.wav)
- [12 seconds, cold shape benchmark](trained_results/trained_12s_r0.wav)
Prompt: “Instrumental cinematic electronic music, warm synthesizer chords, melodic strings, steady drums, no vocals”. Stereo 44.1 kHz PCM16. All outputs were checked for finite, non-silent samples. Peak normalization before PCM export is recorded as `output_gain`; raw peaks sometimes exceed 1.0.
## Measured performance
| Test | Audio | Diffusion | GPU decode | Total to WAV | RTF |
|---|---:|---:|---:|---:|---:|
| First baseline output, compilation included | 30 s | 100.277 s | 23.235 s | 124.064 s | 4.135 |
| Second baseline output, decoder JIT capture | 30 s | 17.678 s | 3.966 s | 21.952 s | 0.732 |
| Third baseline output, fully warm | 30 s | 17.528 s | 0.695 s | 18.529 s | **0.618** |
| First 12-second shape, compilation included | 12 s | 63.814 s | 23.280 s | 87.427 s | 7.286 |
| Independent fixed-30 retry, fully warm | 30 s | 18.084 s | 0.706 s | 19.092 s | **0.636** |
Cold process start to first WAV was **177.157 seconds**, including imports, text model loading/encoding, GPU weight upload and compilation. That is not faster than playback. Diagnostic reference dumps add overhead to first runs. Warm totals include GPU synchronization, waveform transfer, WAV writing and latent saving. Download/conversion are excluded. The independent retry reuses cached text conditioning and adds persistent input buffers; this change did not improve speed materially. All three retry WAVs are byte-identical to their corresponding baseline WAVs (see the WAV manifest).
**12 seconds did not pass.** The next 12-second denoising run took roughly 15 seconds before decoding, and its decoder hung/timed out during JIT capture after changing shapes in the same process. No complete warm 12-second timing is claimed. The experimental process was terminated and Chestnut reset for the fixed-30 retry; three fixed-30 outputs then completed. Cause is not established. Prefer one fixed duration per worker for a prototype, while treating this as an unresolved runtime reliability problem.
## Memory and execution split
- Peak sampled **main-process** host RSS: **2,042,085,376 bytes (2.04 GB)**. Compiler subprocesses are excluded, so this is not total system peak RAM. The cached-text retry used about 196 MB main-process RSS; this excludes the earlier CPU text stage.
- Peak tracked tinygrad buffers across baseline shapes: **1,218,829,230 bytes (1.22 GB)**. This is allocator accounting, not complete physical VRAM measurement. Physical VRAM was not independently measured.
- CPU: tokenizer and FP32 T5Gemma encoder, duration/timestep Fourier features, NumPy random noise, scheduling/compilation, transfers, normalization and WAV writing.
- Chestnut: FP16 trained 20-layer DiT with selected FP32 normalization/softmax/math, all eight denoising passes and sampler tensor updates, trained SAME-S decoder and waveform reconstruction.
- Host big cores 4–7 were temporarily enabled at the normal 1,689,600 kHz cap/performance governor, offroad. Wrappers restore the prior configuration; no manager processes were stopped.
Warm dispatch instrumentation returned from the DiT call in about 0.084 seconds, while synchronized whole steps took about 2.2–2.4 seconds. The remaining time includes asynchronous GPU execution and eager sampler/runtime overhead; it cannot all be attributed to CPU or USB. Fusing sampler arithmetic into the captured graph is a plausible small follow-up, but is unmeasured and unnecessary for the demonstrated 30-second threshold. No major optimization effort was undertaken.
## Correctness and port scope
The [native port](native_sa3.py) implements model-specific tensor math and TinyJit; no general PyTorch compatibility layer or tinygrad backend changes were needed. [Conversion](convert_weights.py) streams tensors to FP16 NPY files and folds decoder weight normalization. The pretrained audio encoder is omitted: basic text generation does not require it.
[Reference validation](trained_results/validation.json) compares first DiT forwards and decoder prefix slices for both durations against the released CPU PyTorch implementations using converted FP16 weights promoted to FP32:
| Comparison | Cosine similarity | Relative RMSE |
|---|---:|---:|
| DiT 30 s | 0.999768 | 0.02154 |
| DiT 12 s | 0.999702 | 0.02444 |
| Decoder 30 s | 0.999725 | 0.02364 |
| Decoder 12 s | 0.999853 | 0.01756 |
These checks support the narrow port's numerical correctness; they are not exhaustive per-step/full-waveform parity or a listening evaluation. Decoder checks use eight input latent frames and the first four decoded frames to avoid truncated-context boundary effects.
Pinned trained model: `stabilityai/stable-audio-3-small-music`, revision `0fef1392cd842149a2b6d445e181c97608faac06`. Released source: `Stability-AI/stable-audio-3`, commit `779434a908193105335fd8d833418603625b2859`. Model assets stay in scratch storage, not this repository. Existing authenticated access was used without logging credentials.
See [the prior architecture/access investigation](ARCHITECTURE_GATE.md) for source links and historical synthetic results. Its weight-access blocker is now resolved; its synthetic arithmetic is superseded by these trained-weight measurements.
## Continuation result
[Warm continuation WAV](trained_results/continuation_30s_r2.wav), [alternate continuation](trained_results/continuation_30s_r1.wav), [source WAV](trained_results/trained_30s_r0.wav), [raw results](trained_results/continuation_result.json).
The model received the first 86 source latent frames (7.9877 seconds) with the trained inpainting mask/local conditioning. It generated the remainder of a 30-second output, then retained latents were restored before decoding. All three saved continuations retain those 86 latent frames exactly. Full raw waveform equality at the boundary is not claimed: decoder context and independently reported peak normalization can affect PCM output.
The fully warm run generated **22.0123 seconds of new material** in **19.3357 seconds total**, including 18.3408 seconds of diffusion and 0.6893 seconds of GPU decoding. **RTF per new audio = 0.8784**; RTF against the entire 30-second file = 0.6445. Cached text was used. The decoder-capture run took 22.3218 seconds (1.0141 RTF per new audio), so prewarming matters. This establishes execution and speed, not perceptual continuity. Listen around the eight-second transition.
No waveform encoder was ported: this inexpensive continuation path reuses SA3-generated source latents. Arbitrary external audio conditioning still requires the SAME-S encoder and is not demonstrated. The continuation worker completed three outputs and restored CPU settings.
## Recommendation
Use **SA3 at a fixed approximately 30-second duration as the leading experimental generator**, subject to human listening acceptance and runtime reliability work. Preserve MusicGen as the known-working fallback. Prewarm the model and generate buffered music; do not promise immediate cold-start or subsecond reaction.
This does **not** establish coexistence with openpilot inference. Exclusive GPU ownership, total host memory including compiler workers, concurrent workload scheduling and driving-time behavior remain untested. Do not interpret an offroad speed pass as authorization or evidence for production deployment.
## Reproduction
Device scratch root: `/data/sa3-feasibility`. [Benchmark](benchmark_sa3.py), [offroad/CPU wrapper](power_trained.py), [runtime environment](run_trained.sh), [CPU reference check](verify_trained.py). The isolated SA3 venv has its own Transformers dependencies; it reads existing Torch dependencies without modifying MusicGen's venv.
With model conversion and environment already staged, run:
```sh
ssh comma@192.168.3.111 '/usr/local/venv/bin/python /data/sa3-feasibility/power_trained.py --durations 30 --repeats 3 --tag reproduced'
```
For cached conditioning add `--conditioning /data/sa3-feasibility/trained_results/conditioning.npy`. Cached runs do not measure fresh prompt encoding. Keep one shape per process given the observed mixed-shape failure. Raw evidence: [baseline JSON](trained_results/trained_result.json), [fixed-30 retry JSON](trained_results/stable30_result.json), and logs in `trained_results/`.
@@ -1,97 +0,0 @@
"""Actual trained SA3 on Chestnut. CPU text, GPU DiT + SAME-S, CPU file output."""
import argparse,gc,json,os,threading,time
from pathlib import Path
import numpy as np,psutil
start=time.monotonic();peak_rss=peak_buffers=0;proc=psutil.Process()
def monitor():
global peak_rss,peak_buffers
while True:
peak_rss=max(peak_rss,proc.memory_info().rss)
if 'GlobalCounters' in globals():peak_buffers=max(peak_buffers,GlobalCounters.mem_used)
time.sleep(.02)
threading.Thread(target=monitor,daemon=True).start()
def log(*args):print(round(time.monotonic()-start,3),*args,flush=True)
p=argparse.ArgumentParser();p.add_argument('--durations',default='30,12');p.add_argument('--repeats',type=int,default=3)
p.add_argument('--prompt',default='Instrumental cinematic electronic music, warm synthesizer chords, melodic strings, steady drums, no vocals')
p.add_argument('--stable-inputs',action='store_true');p.add_argument('--prefix-latents',default='');p.add_argument('--keep-seconds',type=float,default=8);p.add_argument('--tag',default='trained');p.add_argument('--conditioning',default='');a=p.parse_args()
ROOT=Path('/data/sa3-feasibility');OUT=ROOT/'trained_results';OUT.mkdir(exist_ok=True)
import soundfile as sf
from tinygrad import Device,Tensor,dtypes
from tinygrad.helpers import GlobalCounters
from native_sa3 import DiT,Decoder,tensor,fourier
text_time=0
if a.conditioning:
enc=np.load(a.conditioning)
else:
import torch
from transformers import AutoTokenizer,AutoConfig,T5GemmaEncoderModel
torch.set_num_threads(4)
path=str(ROOT/'t5gemma-b-b-ul2');t0=time.monotonic()
tok=AutoTokenizer.from_pretrained(path,local_files_only=True,use_fast=False)
cfg=AutoConfig.from_pretrained(path,local_files_only=True);cfg.is_encoder_decoder=False
text_model=T5GemmaEncoderModel.from_pretrained(path,config=cfg,local_files_only=True,torch_dtype=torch.float32,low_cpu_mem_usage=True).eval()
log('TEXT_LOADED',time.monotonic()-t0)
with torch.no_grad():
t0=time.monotonic();inputs=tok([a.prompt],padding='max_length',max_length=256,truncation=True,return_tensors='pt')
enc=text_model(**inputs).last_hidden_state.numpy()
mask=inputs['attention_mask'].numpy()[...,None]
padding=np.load(ROOT/'native/conditioner.conditioners.prompt.padding_embedding.npy').astype(np.float32)
enc=enc*mask+padding*(1-mask);text_time=time.monotonic()-t0
np.save(OUT/'conditioning.npy',enc.astype(np.float16));log('TEXT_ENCODED',text_time)
del text_model,tok,cfg,inputs;gc.collect()
log('BEFORE_GPU_LOAD')
t0=time.monotonic();dit=DiT(ROOT/'native',persistent=a.stable_inputs);decoder=Decoder(ROOT/'native');load_gpu=time.monotonic()-t0
log('GPU_LOADED',load_gpu,type(Device['AMD'].iface).__name__)
results={'model':'stabilityai/stable-audio-3-small-music','revision':'0fef1392cd842149a2b6d445e181c97608faac06','prompt':a.prompt,
'hardware':{'backend':type(Device['AMD'].iface).__name__,'arch':Device['AMD'].arch},'text_encoding_seconds':text_time,
'cached_text_input':bool(a.conditioning),'stable_input_buffers':a.stable_inputs,'gpu_load_seconds':load_gpu,'cpu_affinity':list(os.sched_getaffinity(0)),'runs':[]}
for duration in map(float,a.durations.split(',')):
n=int(np.ceil(duration*44100/4096/2)*2)
number=fourier([duration/384]).astype(np.float32)
w=np.load(ROOT/'native/conditioner.conditioners.seconds_total.embedder.embedding.1.weight.npy').astype(np.float32)
b=np.load(ROOT/'native/conditioner.conditioners.seconds_total.embedder.embedding.1.bias.npy').astype(np.float32)
g=(number@w.T+b).astype(np.float16)
prefix=None;keep_frames=0
if a.prefix_latents:
source=np.load(a.prefix_latents);keep_frames=min(round(a.keep_seconds*44100/4096),source.shape[1],n-2)
prefix=np.zeros((1,n,256),dtype=np.float16);prefix[:,:keep_frames]=source[:,:keep_frames]
cond=tensor(np.concatenate([enc.astype(np.float16),g[:,None,:]],axis=1));glob=tensor(g);local=tensor(np.zeros((1,n,257),dtype=np.float16))
if prefix is not None:
mask_np=np.zeros((1,n,1),dtype=np.float16);mask_np[:,:keep_frames]=1
local=tensor(np.concatenate([mask_np,prefix],axis=-1));known=tensor(prefix);keep=tensor(mask_np)
sigmas=np.linspace(1,0,9,dtype=np.float32);sigmas=1/(1+np.exp(2-sigmas*8.2));sigmas[0]=1;sigmas[-1]=0
# CPU Fourier features computed with FP32 before transferring the small timestep inputs.
tfs=[tensor(fourier([t])) for t in sigmas[:-1]]
for repeat in range(a.repeats):
seed=991+repeat;rng=np.random.default_rng(seed);runstart=time.monotonic()
x=tensor(rng.standard_normal((1,n,256)).astype(np.float16));steps=[];dispatch=[]
for i in range(8):
t0=time.monotonic();v=dit(x,tfs[i],cond,glob,local);dispatch.append(time.monotonic()-t0)
if repeat==0 and i==0:
np.savez(OUT/f'first_step_{int(duration)}.npz',x=x.numpy(),tf=fourier([sigmas[i]]),cond=cond.numpy(),g=g,v=v.numpy())
clean=x.float()-float(sigmas[i])*v.float()
if i<7:
noise=tensor(rng.standard_normal((1,n,256)).astype(np.float16))
x=((1-float(sigmas[i+1]))*clean+float(sigmas[i+1])*noise.float()).cast(dtypes.float16).realize()
else:x=clean.cast(dtypes.float16).realize()
Device['AMD'].synchronize();steps.append(time.monotonic()-t0);log('STEP',duration,repeat,i,steps[-1])
if prefix is not None:x=(x*(1-keep)+known*keep).realize()
gen=time.monotonic()-runstart
t0=time.monotonic();audio_tensor=decoder(x);Device['AMD'].synchronize();decode_gpu=time.monotonic()-t0
t0=time.monotonic();audio=audio_tensor.float().numpy()[0,:,:round(duration*44100)].T;transfer=time.monotonic()-t0
assert np.isfinite(audio).all()
if repeat==0:
np.savez(OUT/f'decoder_probe_{int(duration)}.npz',latents=x.numpy()[:,:8],audio=audio[:4*4096])
peak=float(np.abs(audio).max());rms=float(np.sqrt(np.mean(audio**2)))
assert rms>1e-5,(peak,rms)
# Preserve raw model scale unless clipping would occur; report any output gain.
gain=min(1.,.98/max(peak,1e-9));name=f'{a.tag}_{int(duration)}s_r{repeat}'
sf.write(OUT/(name+'.wav'),audio*gain,44100,subtype='PCM_16')
np.save(OUT/(name+'_latents.npy'),x.numpy())
total=time.monotonic()-runstart
row={'duration_seconds':len(audio)/44100,'retained_seconds':keep_frames*4096/44100,'new_seconds':len(audio)/44100-keep_frames*4096/44100,'latent_frames':n,'seed':seed,'repeat':repeat,'generation_seconds':gen,'step_seconds':steps,'dit_call_seconds':dispatch,
'decode_gpu_seconds':decode_gpu,'waveform_transfer_seconds':transfer,'total_to_wav_seconds':total,
'rtf':total/(len(audio)/44100),'rtf_per_new_audio':total/(len(audio)/44100-keep_frames*4096/44100),'fresh_prompt_rtf':(total+text_time)/(len(audio)/44100),'process_elapsed_seconds':time.monotonic()-start,
'raw_audio_rms':rms,'raw_audio_peak':peak,'output_gain':gain,'peak_host_rss_bytes':peak_rss,'peak_tracked_buffer_bytes':peak_buffers,'output':name+'.wav'}
results['runs'].append(row);(OUT/(a.tag+'_result.json')).write_text(json.dumps(results,indent=2));log('RESULT',json.dumps(row))
log('COMPLETE')
@@ -1,22 +0,0 @@
"""Bounded-memory safetensors -> FP16 NPY native weights; no pickle deserialization."""
import json,struct
from pathlib import Path
import numpy as np
ROOT=Path('/tmp/chestnut-sa3-weights');OUT=ROOT/'native';OUT.mkdir(exist_ok=True)
with (ROOT/'model.safetensors').open('rb') as f:
n=struct.unpack('<Q',f.read(8))[0];h=json.loads(f.read(n));base=8+n
def get(k):
v=h[k];f.seek(base+v['data_offsets'][0]);a=np.frombuffer(f.read(v['data_offsets'][1]-v['data_offsets'][0]),dtype={'F32':'<f4','F16':'<f2'}[v['dtype']]).reshape(v['shape'])
return a
count=0
for k,v in h.items():
if k=='__metadata__':continue
if not k.startswith(('model.model.','pretransform.model.decoder.','pretransform.model.bottleneck.running_std','conditioner.')):continue
np.save(OUT/(k+'.npy'),get(k).astype(np.float16));count+=1
prefix='pretransform.model.decoder.layers.3.mapping'
w=get(prefix+'.weight_v');g=get(prefix+'.weight_g')
np.save(OUT/(prefix+'.weight.npy'),(w*g/np.sqrt(np.sum(w*w,axis=(1,2),keepdims=True))).astype(np.float16))
print('NATIVE_TENSORS',count,'BYTES',sum(p.stat().st_size for p in OUT.glob('*.npy')))
with (ROOT/'t5gemma-b-b-ul2/model.safetensors').open('rb') as f:
n=struct.unpack('<Q',f.read(8))[0];h=json.loads(f.read(n))
print('TEXT_KEYS',list(h)[:12]);print('TEXT_DTYPES',set(v.get('dtype') for v in h.values()))
@@ -1,16 +0,0 @@
"""Download authorized, pinned weights; auth header is not forwarded on redirects."""
import json,urllib.request
from pathlib import Path
ROOT=Path('/tmp/chestnut-sa3-weights');ROOT.mkdir(exist_ok=True)
REPO='stabilityai/stable-audio-3-small-music';REV='0fef1392cd842149a2b6d445e181c97608faac06'
token=(Path.home()/'.cache/huggingface/token').read_text().strip()
files=['model_config.json','model.safetensors','t5gemma-b-b-ul2/config.json','t5gemma-b-b-ul2/model.safetensors','t5gemma-b-b-ul2/tokenizer.model','t5gemma-b-b-ul2/tokenizer_config.json','t5gemma-b-b-ul2/special_tokens_map.json']
for filename in files:
out=ROOT/filename;out.parent.mkdir(exist_ok=True)
if out.exists():print('EXISTS',filename,flush=True);continue
req=urllib.request.Request(f'https://huggingface.co/{REPO}/resolve/{REV}/{filename}')
req.add_unredirected_header('Authorization','Bearer '+token)
with urllib.request.urlopen(req,timeout=60) as r,out.with_suffix(out.suffix+'.tmp').open('wb') as f:
while buf:=r.read(8*1024*1024):f.write(buf)
out.with_suffix(out.suffix+'.tmp').rename(out)
print('DOWNLOADED',filename,out.stat().st_size,flush=True)
@@ -1,76 +0,0 @@
"""Synthetic SA3 layer-shape probe. NOT model inference or a music benchmark.
Shapes from Stability-AI/stable-audio-3 optimized/mlx models, see report.
"""
import time,json,threading,os
from pathlib import Path
import numpy as np,psutil
from tinygrad import Tensor,TinyJit,Device,dtypes
from tinygrad.helpers import GlobalCounters
ROOT=Path('/data/sa3-feasibility');ROOT.mkdir(exist_ok=True)
rng=np.random.default_rng(1709)
peak_rss=peak_buffers=0
proc=psutil.Process()
def monitor():
global peak_rss,peak_buffers
while True:
peak_rss=max(peak_rss,proc.memory_info().rss);peak_buffers=max(peak_buffers,GlobalCounters.mem_used);time.sleep(.02)
threading.Thread(target=monitor,daemon=True).start()
def random(shape,scale=.02): return Tensor((rng.standard_normal(shape)*scale).astype(np.float16),device='AMD').realize()
def norm(x,eps): return (x.float()*(x.float().square().mean(-1,keepdim=True)+eps).rsqrt()).cast(x.dtype)
def rope(x,c,s):
a,b=x[...,:16],x[...,16:32]
return (a*c-b*s).cat(a*s+b*c,x[...,32:],dim=-1)
class Block:
def __init__(self,kind,seq):
f=np.arange(seq)[:,None]/10000**(np.arange(0,32,2)/32)
self.cos=Tensor(np.cos(f).astype(np.float16),device='AMD').realize();self.sin=Tensor(np.sin(f).astype(np.float16),device='AMD').realize()
self.kind=kind;self.dim=1024 if kind=='dit' else 768;d=self.dim
self.heads=d//64;self.inner=4096 if kind=='dit' else 2304
self.wqkv=random((d,(3 if kind=='dit' else 5)*d));self.wo=random((d,d))
self.wff=random((d,2*self.inner));self.bff=random((2*self.inner,));self.wout=random((self.inner,d));self.bout=random((d,))
if kind=='dit':
self.wq=random((d,d));self.wkv=random((d,2*d));self.wcross=random((d,d))
self.wlocal=random((257,d));self.blocal=random((d,));self.wlocal2=random((d,d));self.blocal2=random((d,))
self.jit=TinyJit(self.forward)
def attention(self,q,k,v): return (((q*.125)@k.transpose(-1,-2)).float().softmax(-1).cast(q.dtype)@v)
def forward(self,x,context,local,g):
B,T,D=x.shape
def heads(y):return y.reshape(B,T,self.heads,64).transpose(1,2)
if self.kind=='dit':
ss=g.reshape(B,1,6,D);a,b,c,d,e,f=[ss[:,:,i] for i in range(6)]
z=norm(x,1e-5)*(1+a)+b
q,k,v=[heads(t) for t in (z@self.wqkv).chunk(3,dim=-1)]
q=rope(norm(q,1e-6),self.cos,self.sin);k=rope(norm(k,1e-6),self.cos,self.sin)
x=x+(self.attention(q,k,v).transpose(1,2).reshape(B,T,D)@self.wo)*(1-c).sigmoid()
q=norm(heads(norm(x,1e-5)@self.wq),1e-6)
k,v=[t.reshape(B,-1,self.heads,64).transpose(1,2) for t in (context@self.wkv).chunk(2,dim=-1)]
k=norm(k,1e-6)
x=x+self.attention(q,k,v).transpose(1,2).reshape(B,T,D)@self.wcross
x=x+((local@self.wlocal+self.blocal).silu()@self.wlocal2+self.blocal2)
z=norm(x,1e-5)*(1+d)+e
else:
q,k,v,qd,kd=[heads(t) for t in (x.tanh()@self.wqkv).chunk(5,dim=-1)]
q,k,qd,kd=[rope(t.tanh(),self.cos,self.sin) for t in [q,k,qd,kd]]
y=self.attention(q,k,v)-self.attention(qd,kd,v)
x=x+y.transpose(1,2).reshape(B,T,D)@self.wo
z=x.tanh()
val,gate=(z@self.wff+self.bff).chunk(2,dim=-1)
y=(val*gate.silu())@self.wout+self.bout
if self.kind=='dit':y=y*(1-f).sigmoid()
return (x+y).realize()
results={'warning':'Random-weight representative blocks only. No text conditioning, complete model, decoded audio, or gate success. Repeated-layer extrapolation is not measured end-to-end latency.','runs':[]}
for kind,shape in [('dit',(1,384,1024)),('same_s_decoder',(160,34,768))]:
t0=time.monotonic();block=Block(kind,shape[1]);x=random(shape,.2)
context=random((shape[0],257 if kind=='dit' else 1,shape[-1]));local=random((shape[0],shape[1],257))
g=random((shape[0],6*shape[-1]));init=time.monotonic()-t0
times=[]
for i in range(6):
t0=time.monotonic();out=block.jit(x,context,local,g);Device['AMD'].synchronize();dt=time.monotonic()-t0
times.append(dt);print(kind,i,dt,flush=True)
assert np.isfinite(out.numpy()).all()
run={'kind':kind,'shape':shape,'initialization_seconds':init,'calls_seconds':times,'warm_median_seconds':float(np.median(times[3:])),
'peak_host_rss_bytes':peak_rss,'peak_tracked_buffer_bytes':peak_buffers}
results['runs'].append(run)
results['hardware']={'backend':type(Device['AMD'].iface).__name__,'arch':Device['AMD'].arch}
results['cpu_affinity']=list(os.sched_getaffinity(0))
(ROOT/'layer_probe.json').write_text(json.dumps(results,indent=2));print('RESULT',json.dumps(run),flush=True)
@@ -1,118 +0,0 @@
"""Narrow trained-weight SA3 Small-Music port. No PyTorch backend bridge.
Reference: Stability-AI/stable-audio-3 @ 779434a908193105335fd8d833418603625b2859.
"""
from pathlib import Path
import numpy as np
from tinygrad import Tensor,TinyJit,dtypes
def tensor(a):return Tensor(np.asarray(a),device='AMD').realize()
def fourier(values):
f=np.exp(np.linspace(np.log(.5),np.log(10000.),128,dtype=np.float32))*np.float32(2*np.pi)
args=np.asarray(values,dtype=np.float32).reshape(-1,1)*f
return np.concatenate([np.cos(args),np.sin(args)],axis=-1).astype(np.float16)
class Weights:
def __init__(self,path,prefix):
self.w={}
for f in sorted(Path(path).glob(prefix+'*.npy')):
self.w[f.stem[len(prefix):]]=tensor(np.load(f))
def linear(self,x,p):
w=self.w[p+'.weight'];y=x.cast(w.dtype)@w.T
if p+'.bias' in self.w:y=y+self.w[p+'.bias'].cast(y.dtype)
return y
def mlp(self,x,p):return self.linear(self.linear(x,p+'.0').silu(),p+'.2')
def rms(self,x,p,eps=1e-5):
z=x.float();z=z*(z.square().mean(-1,keepdim=True)+eps).rsqrt()
return (z*self.w[p+'.gamma'].float()).cast(x.dtype)
def dyt(self,x,p):
return (self.w[p+'.gamma'].float()*(self.w[p+'.alpha'].float()*x.float()).tanh()+self.w[p+'.beta'].float()).cast(x.dtype)
def ff(self,x,p):
a,b=self.linear(x,p+'.ff.0.proj').chunk(2,dim=-1)
return self.linear(a*b.silu(),p+'.ff.2')
@staticmethod
def attention(q,k,v):
return (((q*.125)@k.transpose(-1,-2)).float().softmax(-1).cast(v.dtype)@v)
@staticmethod
def rope_tables(n):
inv=1./10000**(np.arange(0,32,2,dtype=np.float32)/32)
f=np.arange(n,dtype=np.float32)[:,None]*inv[None,:]
return tensor(np.cos(f)),tensor(np.sin(f))
@staticmethod
def rope(x,cs):
c,s=cs;a,b=x[...,:16].float(),x[...,16:32].float()
return (a*c-b*s).cat(a*s+b*c,dim=-1).cast(x.dtype).cat(x[...,32:],dim=-1)
class DiT(Weights):
def __init__(self,path,persistent=False):
super().__init__(path,'model.model.')
self.jits={};self.ropes={};self.inputs={};self.persistent=persistent
def forward(self,x,tf,cond,global_cond,local):
B,T,D=x.shape
context=self.mlp(cond,'to_cond_embed')
g=self.mlp(global_cond,'to_global_embed')+self.mlp(tf,'to_timestep_embed')
g=self.mlp(g,'transformer.global_cond_embedder')
x=x+x@self.w['preprocess_conv.weight'][:,:,0].T
x=self.linear(x,'transformer.project_in')
x=self.w['transformer.memory_tokens'].unsqueeze(0).expand(B,64,1024).cat(x,dim=1)
N=T+64;cs=self.ropes[T]
def heads(a):return a.reshape(B,-1,16,64).transpose(1,2)
for i in range(20):
p=f'transformer.layers.{i}.'
a,b,c,d,e,f=(g+self.w[p+'to_scale_shift_gate']).unsqueeze(1).chunk(6,dim=-1)
z=self.rms(x,p+'pre_norm')*(1+a)+b
q,k,v=[heads(t) for t in self.linear(z,p+'self_attn.to_qkv').chunk(3,dim=-1)]
q=self.rope(self.rms(q,p+'self_attn.q_norm',1e-6),cs)
k=self.rope(self.rms(k,p+'self_attn.k_norm',1e-6),cs)
att=self.attention(q,k,v).transpose(1,2).reshape(B,N,1024)
x=x+self.linear(att,p+'self_attn.to_out')*(1-c).sigmoid()
q=self.rms(heads(self.linear(self.rms(x,p+'cross_attend_norm'),p+'cross_attn.to_q')),p+'cross_attn.q_norm',1e-6)
k,v=[heads(t) for t in self.linear(context,p+'cross_attn.to_kv').chunk(2,dim=-1)]
k=self.rms(k,p+'cross_attn.k_norm',1e-6)
att=self.attention(q,k,v).transpose(1,2).reshape(B,N,1024)
x=x+self.linear(att,p+'cross_attn.to_out')
loc=self.mlp(local,p+'to_local_embed').pad(((0,0),(64,0),(0,0)))
x=x+loc
x=(x+self.ff(self.rms(x,p+'ff_norm')*(1+d)+e,p+'ff')*(1-f).sigmoid()).realize()
x=self.linear(x[:,64:],'transformer.project_out')
return (x+x@self.w['postprocess_conv.weight'][:,:,0].T).realize()
def __call__(self,x,tf,cond,g,local):
n=x.shape[1]
if n not in self.jits:
self.ropes[n]=self.rope_tables(n+64);self.jits[n]=TinyJit(self.forward)
if self.persistent:
if n not in self.inputs:self.inputs[n]=(x.clone().realize(),tf.clone().realize())
xi,ti=self.inputs[n];xi.assign(x).realize();ti.assign(tf).realize()
return self.jits[n](xi,ti,cond,g,local)
return self.jits[n](x,tf,cond,g,local)
class Decoder(Weights):
def __init__(self,path):
super().__init__(path,'pretransform.model.decoder.')
self.std=tensor(np.load(Path(path)/'pretransform.model.bottleneck.running_std.npy'))
self.ropes=self.rope_tables(34);self.jits={}
# Converter folds weight normalization before casting.
def block(self,x,i):
B,T,D=x.shape;p=f'layers.3.transformers.{i}.'
def heads(a):return a.reshape(B,T,12,64).transpose(1,2)
q,k,v,qd,kd=[heads(t) for t in self.linear(self.dyt(x,p+'pre_norm'),p+'self_attn.to_qkv').chunk(5,dim=-1)]
q,qd=[self.rope(self.dyt(a,p+'self_attn.q_norm'),self.ropes) for a in [q,qd]]
k,kd=[self.rope(self.dyt(a,p+'self_attn.k_norm'),self.ropes) for a in [k,kd]]
att=(self.attention(q,k,v)-self.attention(qd,kd,v)).transpose(1,2).reshape(B,T,D)
x=x+self.linear(att,p+'self_attn.to_out')
return (x+self.ff(self.dyt(x,p+'ff_norm'),p+'ff')).realize()
def forward(self,latents):
B,T,C=latents.shape
x=self.linear((latents.float()*self.std.float()).cast(latents.dtype),'layers.1')
nt=self.w['layers.3.new_tokens'].reshape(1,1,1,768).expand(B,T,16,768)
x=x.unsqueeze(2).cat(nt,dim=2).reshape(B*T//2,34,768)
for i in range(3):x=self.block(x,i)
x=x.reshape(B,T*17,768)
x=x[:,:17].cat(x,x[:,-17:],dim=1).reshape(B*(T//2+1),34,768)
for i in range(3,6):x=self.block(x,i)
x=x.reshape(B,(T+2)*17,768)[:,17:-17].reshape(B,T,17,768)[:,:,1:].reshape(B,T*16,768)
padded=x.pad(((0,0),(1,1),(0,0)));w=self.w['layers.3.mapping.weight']
out=sum(padded[:,i:i+T*16]@w[:,:,i].T for i in range(3))+self.w['layers.3.mapping.bias']
return out.reshape(B,T*16,2,256).permute(0,2,1,3).reshape(B,2,T*4096).realize()
def __call__(self,x):
n=x.shape[1]
if n not in self.jits:self.jits[n]=TinyJit(self.forward)
return self.jits[n](x)
@@ -1,61 +0,0 @@
"""Temporary offroad benchmark CPU settings; restore on exit. No manager changes."""
import json
import os
import signal
import subprocess
import sys
from pathlib import Path
root = Path('/data/sa3-feasibility')
def is_offroad():
check = subprocess.run(
['/usr/local/venv/bin/python', '-c',
'from openpilot.common.params import Params; raise SystemExit(int(Params().get_bool("IsOnroad")))'],
cwd='/data/openpilot', check=False,
)
return check.returncode == 0
if not is_offroad():
raise SystemExit('This isolated benchmark requires a verified offroad device.')
paths = [Path(f'/sys/devices/system/cpu/cpu{i}/online') for i in range(4, 8)]
paths += [Path('/sys/devices/system/cpu/cpufreq/policy4') / n for n in ['scaling_max_freq', 'scaling_governor']]
saved = {str(p): p.read_text().strip() for p in paths}
(root / 'power_before.json').write_text(json.dumps(saved, indent=2))
def write(path, value):
subprocess.run(['sudo', '-n', 'tee', str(path)], input=value + '\n', text=True, stdout=subprocess.DEVNULL, check=True)
def stop(sig, frame):
raise KeyboardInterrupt
signal.signal(signal.SIGTERM, stop)
child = None
rc = 1
try:
for path in paths[:4]:
write(path, '1')
write(paths[4], '1689600')
write(paths[5], 'performance')
print('POWER', json.dumps({str(p): p.read_text().strip() for p in paths}), flush=True)
child = subprocess.Popen(['taskset', '-c', '4-7', 'bash', str(root / 'run_probe.sh'), *sys.argv[1:]], start_new_session=True)
rc = child.wait()
finally:
try:
if child is not None and child.poll() is None:
os.killpg(child.pid, signal.SIGTERM)
try:
child.wait(timeout=15)
except subprocess.TimeoutExpired:
os.killpg(child.pid, signal.SIGKILL)
child.wait()
finally:
if is_offroad():
for path in reversed(paths):
write(path, saved[str(path)])
(root / 'power_after.json').write_text(json.dumps({str(p): p.read_text().strip() for p in paths}, indent=2))
print('POWER_RESTORED', flush=True)
else:
print('Device no longer verified offroad; leaving active CPU settings to hardwared.', flush=True)
sys.exit(rc)
@@ -1,61 +0,0 @@
"""Temporary offroad benchmark CPU settings; restore on exit. No manager changes."""
import json
import os
import signal
import subprocess
import sys
from pathlib import Path
root = Path('/data/sa3-feasibility')
def is_offroad():
check = subprocess.run(
['/usr/local/venv/bin/python', '-c',
'from openpilot.common.params import Params; raise SystemExit(int(Params().get_bool("IsOnroad")))'],
cwd='/data/openpilot', check=False,
)
return check.returncode == 0
if not is_offroad():
raise SystemExit('This isolated benchmark requires a verified offroad device.')
paths = [Path(f'/sys/devices/system/cpu/cpu{i}/online') for i in range(4, 8)]
paths += [Path('/sys/devices/system/cpu/cpufreq/policy4') / n for n in ['scaling_max_freq', 'scaling_governor']]
saved = {str(p): p.read_text().strip() for p in paths}
(root / 'trained_power_before.json').write_text(json.dumps(saved, indent=2))
def write(path, value):
subprocess.run(['sudo', '-n', 'tee', str(path)], input=value + '\n', text=True, stdout=subprocess.DEVNULL, check=True)
def stop(sig, frame):
raise KeyboardInterrupt
signal.signal(signal.SIGTERM, stop)
child = None
rc = 1
try:
for path in paths[:4]:
write(path, '1')
write(paths[4], '1689600')
write(paths[5], 'performance')
print('POWER', json.dumps({str(p): p.read_text().strip() for p in paths}), flush=True)
child = subprocess.Popen(['taskset', '-c', '4-7', 'bash', str(root / 'run_trained.sh'), *sys.argv[1:]], start_new_session=True)
rc = child.wait()
finally:
try:
if child is not None and child.poll() is None:
os.killpg(child.pid, signal.SIGTERM)
try:
child.wait(timeout=15)
except subprocess.TimeoutExpired:
os.killpg(child.pid, signal.SIGKILL)
child.wait()
finally:
if is_offroad():
for path in reversed(paths):
write(path, saved[str(path)])
(root / 'trained_power_after.json').write_text(json.dumps({str(p): p.read_text().strip() for p in paths}, indent=2))
print('POWER_RESTORED', flush=True)
else:
print('Device no longer verified offroad; leaving active CPU settings to hardwared.', flush=True)
sys.exit(rc)
@@ -1,61 +0,0 @@
"""Temporary offroad benchmark CPU settings; restore on exit. No manager changes."""
import json
import os
import signal
import subprocess
import sys
from pathlib import Path
root = Path('/data/sa3-feasibility')
def is_offroad():
check = subprocess.run(
['/usr/local/venv/bin/python', '-c',
'from openpilot.common.params import Params; raise SystemExit(int(Params().get_bool("IsOnroad")))'],
cwd='/data/openpilot', check=False,
)
return check.returncode == 0
if not is_offroad():
raise SystemExit('This isolated benchmark requires a verified offroad device.')
paths = [Path(f'/sys/devices/system/cpu/cpu{i}/online') for i in range(4, 8)]
paths += [Path('/sys/devices/system/cpu/cpufreq/policy4') / n for n in ['scaling_max_freq', 'scaling_governor']]
saved = {str(p): p.read_text().strip() for p in paths}
(root / 'verify_trained_power_before.json').write_text(json.dumps(saved, indent=2))
def write(path, value):
subprocess.run(['sudo', '-n', 'tee', str(path)], input=value + '\n', text=True, stdout=subprocess.DEVNULL, check=True)
def stop(sig, frame):
raise KeyboardInterrupt
signal.signal(signal.SIGTERM, stop)
child = None
rc = 1
try:
for path in paths[:4]:
write(path, '1')
write(paths[4], '1689600')
write(paths[5], 'performance')
print('POWER', json.dumps({str(p): p.read_text().strip() for p in paths}), flush=True)
child = subprocess.Popen(['taskset', '-c', '4-7', 'bash', str(root / 'run_verify_trained.sh'), *sys.argv[1:]], start_new_session=True)
rc = child.wait()
finally:
try:
if child is not None and child.poll() is None:
os.killpg(child.pid, signal.SIGTERM)
try:
child.wait(timeout=15)
except subprocess.TimeoutExpired:
os.killpg(child.pid, signal.SIGKILL)
child.wait()
finally:
if is_offroad():
for path in reversed(paths):
write(path, saved[str(path)])
(root / 'verify_trained_power_after.json').write_text(json.dumps({str(p): p.read_text().strip() for p in paths}, indent=2))
print('POWER_RESTORED', flush=True)
else:
print('Device no longer verified offroad; leaving active CPU settings to hardwared.', flush=True)
sys.exit(rc)
@@ -1,13 +0,0 @@
#!/usr/bin/env bash
set -euo pipefail
cd /data/openpilot/tinygrad_repo
export PATH=/data/roadscore-feasibility/venv/bin:/usr/bin:/bin
export PYTHONPATH=/data/openpilot/tinygrad_repo
export DEV=USB+AMD:LLVM
export MAX_JOBS=1 CXX=clang++
export TORCH_EXTENSIONS_DIR=/data/roadscore-feasibility/cache/torch_extensions
export XDG_CACHE_HOME=/data/roadscore-feasibility/cache
export HF_HOME=/data/roadscore-feasibility/cache/huggingface
export TMPDIR=/data/roadscore-feasibility/tmp
export OMP_NUM_THREADS=2 OPENBLAS_NUM_THREADS=2
exec timeout 360 python -u /data/sa3-feasibility/layer_probe.py "$@"
@@ -1,13 +0,0 @@
#!/usr/bin/env bash
set -euo pipefail
cd /data/openpilot/tinygrad_repo
export PATH=/data/sa3-feasibility/venv/bin:/usr/bin:/bin
export PYTHONPATH=/data/openpilot/tinygrad_repo
export DEV=USB+AMD:LLVM
export MAX_JOBS=1 CXX=clang++
export TORCH_EXTENSIONS_DIR=/data/roadscore-feasibility/cache/torch_extensions
export XDG_CACHE_HOME=/data/roadscore-feasibility/cache
export HF_HOME=/data/roadscore-feasibility/cache/huggingface
export TMPDIR=/data/roadscore-feasibility/tmp
export OMP_NUM_THREADS=2 OPENBLAS_NUM_THREADS=2
exec timeout 1500 python -u /data/sa3-feasibility/benchmark_sa3.py "$@"
@@ -1,13 +0,0 @@
#!/usr/bin/env bash
set -euo pipefail
cd /data/openpilot/tinygrad_repo
export PATH=/data/sa3-feasibility/venv/bin:/usr/bin:/bin
export PYTHONPATH=/data/openpilot/tinygrad_repo
export DEV=USB+AMD:LLVM
export MAX_JOBS=1 CXX=clang++
export TORCH_EXTENSIONS_DIR=/data/roadscore-feasibility/cache/torch_extensions
export XDG_CACHE_HOME=/data/roadscore-feasibility/cache
export HF_HOME=/data/roadscore-feasibility/cache/huggingface
export TMPDIR=/data/roadscore-feasibility/tmp
export OMP_NUM_THREADS=2 OPENBLAS_NUM_THREADS=2
exec timeout 600 python -u /data/sa3-feasibility/verify_trained.py "$@"
@@ -1,61 +0,0 @@
"""Independent released PyTorch model references versus saved native GPU outputs."""
import gc,json,sys,types,time
from pathlib import Path
import numpy as np,torch
ROOT=Path('/data/sa3-feasibility');OUT=ROOT/'trained_results';W=ROOT/'native'
torch.set_num_threads(4)
# Import the released model code without importing its unrelated application/deployment dependencies.
for name,rel in [('stable_audio_3','stable_audio_3'),('stable_audio_3.models','stable_audio_3/models')]:
mod=types.ModuleType(name);mod.__path__=[str(ROOT/'reference_source'/rel)];sys.modules[name]=mod
from stable_audio_3.models.dit import DiffusionTransformer
from same_s_decoder_torch import SAMESDecoder
def load(k):return torch.from_numpy(np.load(W/(k+'.npy'))).float()
def compare(a,b):
a=a.astype(np.float64).ravel();b=b.astype(np.float64).ravel()
return {'cosine':float(np.dot(a,b)/(np.linalg.norm(a)*np.linalg.norm(b))), 'mae':float(np.mean(np.abs(a-b))),
'relative_rmse':float(np.linalg.norm(a-b)/np.linalg.norm(b)),'max_error':float(np.max(np.abs(a-b)))}
results={};t0=time.monotonic()
cfg=json.loads((ROOT/'model_config.json').read_text())['model']['diffusion']['config']
with torch.device('meta'):model=DiffusionTransformer(**cfg,diffusion_objective='rf_denoiser')
# Materialize one tensor at a time, avoiding an extra full state-dict copy.
for name,param in list(model.named_parameters())+list(model.named_buffers()):
path=W/('model.model.'+name+'.npy')
if not path.exists():raise RuntimeError('Missing reference tensor '+name)
parts=name.split('.');parent=model
for part in parts[:-1]:parent=getattr(parent,part)
value=load('model.model.'+name)
if isinstance(param,torch.nn.Parameter):value=torch.nn.Parameter(value,requires_grad=False)
setattr(parent,parts[-1],value)
model.eval();print('REFERENCE_DIT_LOADED',time.monotonic()-t0,flush=True)
with torch.no_grad():
for seconds in [30,12]:
file=OUT/f'first_step_{seconds}.npz'
if not file.exists():continue
d=np.load(file);x=torch.from_numpy(d['x']).float().transpose(1,2)
ref=model(x,torch.ones(1),cross_attn_cond=torch.from_numpy(d['cond']).float(),global_embed=torch.from_numpy(d['g']).float(),local_add_cond=torch.zeros(1,257,x.shape[-1])).transpose(1,2).numpy()
c=compare(d['v'],ref);results[f'dit_{seconds}']=c;print('DIT_COMPARE',seconds,c,flush=True)
assert c['cosine']>.995 and c['relative_rmse']<.1,c
del model;gc.collect()
model=SAMESDecoder(output_audio=True).eval()
sd={}
sd['project_in.weight']=load('pretransform.model.decoder.layers.1.weight');sd['project_in.bias']=load('pretransform.model.decoder.layers.1.bias')
sd['new_tokens']=load('pretransform.model.decoder.layers.3.new_tokens')
sd['mapping.weight']=load('pretransform.model.decoder.layers.3.mapping.weight');sd['mapping.bias']=load('pretransform.model.decoder.layers.3.mapping.bias')
sd['running_std']=load('pretransform.model.bottleneck.running_std')
for name in model.state_dict():
if not name.startswith('blocks.'):continue
key=name.replace('blocks.','pretransform.model.decoder.layers.3.transformers.').replace('.attn.','.self_attn.').replace('.ff.glu_proj.','.ff.ff.0.proj.').replace('.ff.proj_out.','.ff.ff.2.')
sd[name]=load(key)
missing,unexpected=model.load_state_dict(sd,strict=False)
assert set(missing)<= {'rope_cos','rope_sin'} and not unexpected,(missing,unexpected)
del sd;gc.collect()
for seconds in [30,12]:
file=OUT/f'decoder_probe_{seconds}.npz'
if not file.exists():continue
d=np.load(file);ref=model(torch.from_numpy(d['latents']).float().transpose(1,2))[0,:,:4*4096].transpose(0,1).numpy()
c=compare(d['audio'],ref);results[f'decoder_{seconds}']=c;print('DECODER_COMPARE',seconds,c,flush=True)
assert c['cosine']>.99 and c['relative_rmse']<.15,c
results['elapsed_seconds']=time.monotonic()-t0
(OUT/'validation.json').write_text(json.dumps(results,indent=2))
print('TRAINED_REFERENCE_CHECKS_PASSED',flush=True)
@@ -1,69 +0,0 @@
# Normal-session hook planning
Candidate source only: no v2 plans/tensors, audio, or native integration yet. The
normal command remains `./onroad --roadscore <route>`. GOLD33602 and the approved
planned60 are references, never default inputs to a new composition.
**Integration API**
1. The normal caller owns session seed selection: fresh random seed per launch,
logged for reproduction; explicit override repeats it; official judging keeps
its existing deterministic route seed policy.
2. `HostHookAdapter(existing_assets_root).fingerprints()` identifies existing
model files, official ACE source and adapter. It hashes severalGB once, without
model generation/downloads. The adapter lazily loads one serialized Mac model.
3. `request_plan(session_seed, plan_index, profile, section, window_seconds,
model_fingerprint, preparation_fingerprint, hook_reference_sha256,
committed_prefix_sha256, previous_plan_sha256)` creates the complete request.
These are keyword arguments. Initial index0 has no context hashes; subsequent
requests require all three. Supported bounded windows are30/45/60seconds.
4. `PlanCache(root).resolve(request, adapter, sources=...)` prepares automatically
on a miss and returns `(directory, cache_hit)`. For continuation, sources maps
`hook_reference` to this run's accepted hook WAV and `committed_prefix` to a
NumPy1×200×64 file containing its latest committed8s latent. Initial sources
are empty. Native worker binding to the returned directory remains unimplemented;
do not overwrite existing profile directories as a substitute.
Semantic seed is separately derived with
`roadscore-semantic-plan-v1:<session_seed>:<plan_index>:<profile>`. Native sampling
keeps the normal runtime's generation seed policy. Cache identity includes the
session, role, duration, prompts, versions and musical context hashes. Fresh seeds
cannot reuse another run's plan; identical reproduction requests can. Cached
assets contain semantic codes and verified tensors, never old PCM. Invalid/stale
entries fail explicitly. `next_section()` offers verse/build/chorus/verse/bridge/
chorus progression; current arrival intent selects outro. Advance after acceptance.
**Musical behavior and preparation**
Each composition establishes its own compact4–6note hook and rhythmic answer.
Verses fragment it, builds intensify, choruses return the complete motif, bridges
transform rhythm/register, and outros resolve it. Profile identity persists in
all prompts: Prism128BPM/Dminor, or Aurora116BPM/Aminor. Actual hook audio and the
committed prefix anchor subsequent plans; no future route events or fixed route
schedule enters the request.
The host adapter runs semantic planning, captures tensors before audio diffusion,
then combines planned future hints with the committed8s prefix and existing
12frame/.5 repaint recipe. This new combination requires native validation.
Changing prompt strings alone does not update running conditioning. No v2
preparation or generation has run, and existing worker/CLI files are untouched.
Planning requires the previous accepted prefix, so it adds a causal dependency.
Historical60s LM work alone took≈13s. Start preparation as soon as that prefix
exists while buffered music plays; include planning in buffer budgets and measure
sustained throughput before claiming uninterrupted arbitrary-duration operation.
Use the best validated normal composer until this candidate qualifies.
**Predeclared validation after combined-plan approval**
- CPU: fresh-seed separation, exact cache reuse, context/version invalidation,
missing-code/PCM rejection, exact prefix and unchanged future planned hints.
- Fixed test session seeds:11701223,280714055,3910408210, same route/policy/gates.
Each gets an initial and at least two chained windows. Keep every result,
failure and automatic reroll; no seed substitutions or best-seed selection.
- Check actual planner codes/seeds, reproducibility, input chain and hook identity;
measure preparation/generation/decode/RTF, buffer minimum, holds/underflows and
GPU/link health through the normal command.
- User assesses all three: memorable melody, recognizable return, meaningful
development, no unrelated lead or loop collapse. Include only validated pieces
in the combined demo; presentation remains subordinate.
@@ -1,88 +0,0 @@
# Prism hook candidate — proposal, not deployed
Base: `06c8e7ffe8` (isolated GOLD60 diagnostic), itself based on `a12d1b7854`.
Protected GOLD and user-approved planned60 assets remain unchanged.
## Musical intent
`prism_hook_spec.py` defines one shared full Prism identity and hook contract for
initial, verse, build/prechorus, chorus, bridge and outro. Every role retains
128 BPM, D minor, crystal pluck/glass lead timbre, syncopated bass and tight drums.
The hook has a recognizable four-note rising contour, signature syncopated rhythm
and short answering phrase. Verses expose fragments; builds intensify them;
choruses state the complete original; the bridge transforms rhythm/register while
keeping identity; the return restores the original phrase and outro resolves it.
This replaces the old generic restrained-verse emphasis and keeps specific hook
and instrumental instructions present across all role prompts. It cannot guarantee
memorability or uniqueness without listening. Fixed seeds intentionally reproduce
the same composition; route-derived seeds distinguish comparable route runs.
## Proposed first demo composition
The first candidate is one coherent LM-planned60-second composition, not repeated
8-second-context verses. At the requested128BPM,32bars occupy60seconds:
| Target time | Role | Musical development |
|---|---|---|
|0–3.75s|Introduction,2bars|State recognizable hook|
|3.75–18.75s|Verse,8bars|Lighter fragments, breathing space and answer|
|18.75–26.25s|Build,4bars|Rising register/subdivisions of same motif|
|26.25–41.25s|Chorus,8bars|Complete melody, fuller bass and drums|
|41.25–48.75s|Bridge,4bars|Spacious contrasting transformation|
|48.75–56.25s|Chorus reprise,4bars|Return the unmistakable original|
|56.25–60s|Outro,2bars|Answer and resolution|
These are prompt targets, **not verified generated timestamps**. Do not force
audio cuts or treat these times as detected beats/section boundaries. An actual
generated performance may not follow them. The fixed narrative is independent
of recorded future route events; operator excerpt selection does not grant
runtime access to future telemetry. Current delivered conditions may later inform
causal adaptation, but this candidate does not implement it.
Prefer a60-second route excerpt for this first musical test. A90–120second route
requires a newly prepared longer semantic plan, larger native shape validation,
or a tested continuation strategy. Do not loop this60s piece, stretch its timing,
or restart its initial plan and call that coherent longer-form composition.
Presentation is subordinate: proposed cue/engagement lanes must not overwrite
the hook or rhythmic phase. If a cue conflicts, omit/simplify it. This composition
lane does not add signal, curve, engagement or other sonification. The master
will combine the route, composition, one signal motif, one curve treatment and
one engagement transition in the operator proposal before new generation.
## Actual preparation status and path
**Prompt specification only. No new conditioning tensors, semantic codes, native
audio, or deployed behavior yet.** Three CPU specification tests pass. Local Mac
ACE packages and existing9.4GB model assets are available; no model downloads are
needed. New preparation is paused for the combined demo-plan review.
`prepare_prism_hook.py` is an opt-in Mac preparation tool. Default mode saves the
specification only. `--prepare` initializes the existing official ACE/1.7BLM,
uses `thinking=True` and fixed seed33602 (or deterministic route-derived seed),
records actual returned semantic codes/seed/LM costs, then captures full60s
encoder/context tensors immediately before DiT diffusion. It refuses repaint,
missing semantic planning, nonfinite/wrong-duration tensors and an unavailable
MLX interception path. It never executes native hardware or intentionally
generates audio. Candidate output uses a new private directory; protected files
and active profiles are never replaced. Exact prompt and tensor hashes are saved.
The full composition caption/section text are what this first preparation feeds
into the model. Per-role continuation contracts are saved for future use but
**are not prepared role tensors** and are not wired into current worker startup.
Editing these strings cannot alter deployed prepared embeddings. All role-based
continuation restoration remains future work after the full-plan musical test.
Historical preparation recorded12.4s for the old60s LM plan, plus text/conditioning
and cold model load; budget minutes rather than promise that runtime. Longer
hook text may change encoder length/cost. Tensor output should be a few MiB
(1500×128 float32 context≈0.73MiB plus variable-length encoder≈1–severalMiB), not
new multi-GB weights. Model loading can consume substantial RAM; current observed
diskfree was≈7GB. The approved past run is the playback fallback while preparing.
After review: run one captured plan, check its actual codes/tensors and provenance;
have hardware owner review a candidate-capable isolated native harness (the GOLD
probe intentionally pins original asset hashes and must reject this new case);
then generate one fixed-seed sample with rawgain processing, inspect/listen,
and only afterward integrate subordinate presentation. No seed auditions or
per-route cherry-picking. Master/user musical acceptance remains the gate.
@@ -1,59 +0,0 @@
"""Prepared-condition ACE backend. Owns no audio device, route resolver, or input clock.
All heavy generation/decode runs on Chestnut. New identity preparation is not ported.
Caller must hold the shared GPU lock and establish offroad power supervision.
"""
from pathlib import Path
import json,time
import numpy as np
from tinygrad import TinyJit,Device
from native_ace import DiT,tensor,time_features
from native_vae import VAE
from chunk_decode import ChunkDecoder
class Composer:
capabilities={'supports_full_song':True,'supports_reference':True,'supports_continuation':'repaint latent prefix','supports_transition':True,'supports_outro':True,'supports_streaming':False,'preparation':'precomputed identity/role embeddings; arbitrary prompt preparation remains host-assisted','precision':'FP16, TC_OPT=2','live_rtf':None,'coexistence_validated':False}
def __init__(self,root):
self.root=Path(root);self.dit=DiT(self.root/'weights');self.vae=VAE(self.root/'vae_weights');self.graphs={};self.decoder=ChunkDecoder(self.vae);self.decoder_ready=False
def prepare(self,case,previous=None):
p=self.root/case;cond=np.load(p/'encoder_hidden_states.npy').astype(np.float16);context=np.load(p/'context_latents.npy').astype(np.float16);n=context.shape[1]
valid=np.load(p/'encoder_attention_mask.npy').astype(bool)
width=max(256,((cond.shape[1]+31)//32)*32)
mask=np.full((1,1,1,width),-np.inf,np.float16);mask[0,0,0,:cond.shape[1]]=np.where(valid[0],0,-np.inf)
cond=np.pad(cond,((0,0),(0,width-cond.shape[1]),(0,0)))
repaint=(p/'sampler_repaint_mask.npy').exists();source=keep=blend=None;settings={}
if repaint:
settings=json.loads((p/'sampler.json').read_text());source=np.load(p/'sampler_clean_src_latents.npy').astype(np.float16);keep=np.load(p/'sampler_repaint_mask.npy').astype(bool)
if previous is not None:
preserved=int(np.flatnonzero(keep[0])[0]);assert previous.shape[1]>=preserved
source[:,:preserved]=previous[:,-preserved:];context[:,:preserved,:64]=source[:,:preserved]
blend=keep.astype(np.float16)
for row in blend:
ids=np.flatnonzero(row);left,right=int(ids[0]),int(ids[-1])+1;cf=int(settings.get('repaint_crossfade_frames',12));lo=max(0,left-cf);hi=min(n,right+cf)
if left>lo:row[lo:left]=np.linspace(0,1,left-lo+2)[1:-1]
if hi>right:row[right:hi]=np.linspace(1,0,hi-right+2)[1:-1]
return cond,context,mask,source,keep,blend,settings
def generate(self,case,seed,previous=None):
started=time.monotonic();cond,context,mask,source,keep,blend,settings=self.prepare(case,previous);n=context.shape[1]
self.dit.prepare_shape(n);key=(n,cond.shape[1]);cold=key not in self.graphs
fn=self.graphs.setdefault(key,TinyJit(self.dit.forward))
noise=np.random.default_rng(seed).standard_normal((1,n,64)).astype(np.float16)
x,c,ctx,cm,tf,rf=[tensor(v) for v in [noise,cond,context,mask,time_features([1]),time_features([0])]]
if source is not None:src,km,bm,nt=[tensor(v) for v in [source,keep[...,None],blend[...,None],noise]]
# Warmup is explicit and included in cold wall time, excluded from warm compute.
graph_warmup_started=time.monotonic()
if cold:
for _ in range(3):fn(x,c,ctx,tf,rf,cm);Device['AMD'].synchronize()
graph_warmup_seconds=time.monotonic()-graph_warmup_started if cold else 0.
t=time.monotonic();schedule=np.linspace(1,0,9).tolist()
for i in range(8):
Device['AMD'].synchronize();tf.assign(tensor(time_features([schedule[i]]))).realize();v=fn(x,c,ctx,tf,rf,cm);x.assign(x-v*(schedule[i]-schedule[i+1])).realize()
if source is not None and i<7 and i<round(float(settings.get('repaint_injection_ratio',.5))*8):x.assign(km.where(x,nt*schedule[i+1]+src*(1-schedule[i+1]))).realize()
if source is not None:x.assign(bm*x+(1-bm)*src).realize()
Device['AMD'].synchronize();generation=time.monotonic()-t;latent=x.numpy();decode_cold=not self.decoder_ready
decode_warmup_started=time.monotonic()
if decode_cold:
for _ in range(2):self.decoder.decode(latent[:,:min(375,n)])
self.decoder_ready=True
decode_warmup_seconds=time.monotonic()-decode_warmup_started if decode_cold else 0.
wave,decode_seconds=self.decoder.decode(latent)
prefix=int(np.flatnonzero(keep[0])[0]) if keep is not None else 0
return wave,latent,{'case':case,'seed':seed,'graph_warmup_seconds':graph_warmup_seconds,'decode_warmup_seconds':decode_warmup_seconds,'cold':cold or decode_cold,'wall_seconds':time.monotonic()-started,'generation_seconds':generation,'decode_seconds':decode_seconds,'duration':n/25,'new_seconds':(n-prefix)/25,'warm_rtf_new_audio':(generation+decode_seconds)/((n-prefix)/25),'prefix_seconds':prefix/25,'preparation_host':'prepared embeddings; no Mac required during this generation','sampler':'Euler8/DCW off','subjective_acceptance':None}
@@ -1,157 +0,0 @@
"""Opt-in, offroad resident ACE composer. No route reads or physical audio output."""
import os
os.environ.setdefault('TC_OPT','2')
# Full-speed is the event default. A cap must be an explicit failure mitigation.
import time,json,fcntl,signal,traceback,sys,resource
BOOT=time.monotonic()
from pathlib import Path
import numpy as np,soundfile as sf
windowed=os.environ.get('ROADSCORE_ACE_WINDOWED','1')=='1'
if windowed:
from window_runtime import Composer
else:
from ace_runtime import Composer
P=Path(__file__).resolve().parent;R=P.parents[1];G=R/'generated'
sys.path.insert(0,str(R/'prototype'))
from ace_profiles import selected
from generation_seed import configured_seed,sample_seed
base_seed=configured_seed(required=True)
from quality_gate import QualifiedGenerator,POLICY,HOOK_POLICY
from link_health import LinkProbe
from hook_service import Client
from planned_composition import PlannedComposition
composition_policy=os.environ.get('ROADSCORE_COMPOSITION_POLICY','prepared-v1')
if composition_policy not in ('prepared-v1','hook-v2','hook-cache-v1'):raise ValueError('Unknown composition policy')
if composition_policy in ('hook-v2','hook-cache-v1') and not windowed:raise ValueError('Hook planning requires windowed native sampler')
if composition_policy=='hook-cache-v1':
from cached_composition import CachedComposition,validate_bank
validate_bank(Path(os.environ['ROADSCORE_PLAN_BANK']),profile=selected())
resident=os.environ.get('ROADSCORE_RESIDENT')=='1'
if resident and composition_policy!='hook-cache-v1':raise ValueError('Resident mode requires local current conditioning')
if resident and (G/'session_request.json').exists():raise RuntimeError('Unacknowledged resident request requires owner inspection before restart')
from contextlib import nullcontext
from resident_session import session_lease,validate_session
from startup_buffer import initial_target,session_target,SESSION_PROTOCOL
boot_initial_buffer_target=initial_buffer_target=initial_target(os.environ,composition_policy,POLICY.initial_buffer_seconds)
from tinygrad import Device
profile=selected();preparation_id=f'{profile}_{time.time_ns()}'
lock=open(G/'gpu.lock','w');fcntl.flock(lock,fcntl.LOCK_EX|fcntl.LOCK_NB)
ready=G/'worker_ready';ready.unlink(missing_ok=True)
def stop(*_):raise KeyboardInterrupt
signal.signal(signal.SIGTERM,stop)
def write_json(path,data):
tmp=path.with_suffix('.tmp');tmp.write_text(json.dumps(data,indent=2));tmp.replace(path)
def save_wave(path,wave):sf.write(path,wave*POLICY.output_gain,48000,subtype='FLOAT')
try:
write_json(G/'ace_worker_state.json',{'pid':os.getpid(),'generation_seed':base_seed,'composition_policy':composition_policy,'profile':profile,'phase':'preparing'})
load_started=time.monotonic()
c=Composer(P,profile=profile) if windowed else Composer(P)
model_load_seconds=time.monotonic()-load_started
probe=LinkProbe(Device['AMD'],G/'ace_link.jsonl');probe.install_failure_hook(contain=True)
if windowed:c.decoder.trace=probe.trace
planned=None
if composition_policy in ('hook-v2','hook-cache-v1'):
from ace_runtime import Composer as NativeComposer
from window_policy import retained_end
if composition_policy=='hook-cache-v1':
planned=CachedComposition(Path(os.environ['ROADSCORE_PLAN_BANK']),G/'hook_sessions'/preparation_id,base_seed,profile)
else:
planned=PlannedComposition(Client(os.environ['ROADSCORE_PLANNER_URL'],os.environ['ROADSCORE_PLANNER_TOKEN']),G/'hook_sessions'/preparation_id,base_seed,profile)
def planned_generate(role,seed,previous):return planned.generate(lambda case,seed,previous:NativeComposer.generate(c,case,seed,previous),seed,previous,retained_end)
sample=planned_generate if planned else c.generate
def generate(role,seed,previous):
probe.preflight(role=role,seed=seed)
try:
wave,latent,stats=sample(role,seed,previous);stats['power_limit_watts']=float(os.environ['AM_POWER_LIMIT']) if os.environ.get('AM_POWER_LIMIT') else None;stats['link_session']=probe.session;stats['host_peak_rss_kib']=resource.getrusage(resource.RUSAGE_SELF).ru_maxrss;stats['tracked_allocation_bytes']=probe.last.get('allocator_bytes');return wave,latent,stats
except Exception as e:
probe.sample('generation_exception',error=str(e),role=role,seed=seed);raise
qualified=QualifiedGenerator(generate,policy=HOOK_POLICY if planned else POLICY)
def record_for(job):
folder=G/'quality'/str(job);folder.mkdir(parents=True,exist_ok=True)
def record(attempt,wave,latent,row):
stem=folder/f'attempt_{attempt}'
save_wave(stem.with_suffix('.wav'),wave);np.save(stem.with_suffix('.npy'),latent);write_json(stem.with_suffix('.json'),row)
return {'directory':str(folder),'stem':stem.name,'output_gain':POLICY.output_gain}
return record
def prepare_session(selection=None):
global profile,base_seed,preparation_id,planned,qualified,initial_buffer_target
if selection is not None:
new_profile,new_seed,new_policy,bank_hash=validate_session(selection)
next_buffer_target=session_target(selection,new_policy,boot_initial_buffer_target)
if new_policy!=composition_policy or bank_hash!=planned.bank_hash:raise ValueError('Resident conditioning identity changed')
if (G/'request.json').exists() or (G/'busy').exists():raise RuntimeError('Pending continuation prevents session reset')
next_id=f'{new_profile}_{time.time_ns()}'
next_plan=CachedComposition(Path(os.environ['ROADSCORE_PLAN_BANK']),G/'hook_sessions'/next_id,new_seed,new_profile)
if next_plan.bank_hash!=bank_hash:raise ValueError('Conditioning bank changed on disk')
ready.unlink(missing_ok=True)
profile,base_seed,preparation_id,planned=new_profile,new_seed,next_id,next_plan
initial_buffer_target=next_buffer_target
qualified=QualifiedGenerator(generate,policy=HOOK_POLICY)
session_started=time.monotonic() if selection else BOOT
write_json(G/'ace_worker_state.json',{'pid':os.getpid(),'generation_seed':base_seed,'composition_policy':composition_policy,'profile':profile,'phase':'preparing','accepted_chunks':0,'accepted_buffer_seconds':0.,'elapsed_seconds':0.,'resident_reused':bool(selection)})
print('ACE_PREPARING',profile,flush=True)
initial=None;last=None;preparation=[];slot=0;first_accepted_audio_seconds=None
while initial is None or len(initial)/48000<initial_buffer_target:
role=('initial' if windowed else 'verse') if initial is None else ('verse' if windowed else 'repaint_verse')
if planned:role=planned.begin(last)
wave,last_new,stats=qualified.run(role,sample_seed(base_seed,"prepare",slot),last,record=record_for('prepare_'+preparation_id+'_'+str(slot)))
preparation.append(stats)
if wave is None:raise RuntimeError('Preparation rejected after bounded quality retries; inspect generated/quality')
if planned:planned.accept(wave,last_new)
if initial is None:
first_accepted_audio_seconds=time.monotonic()-session_started;initial=wave.copy()
else:
overlap=2*48000;prefix=round(stats['prefix_seconds']*48000);alpha=np.linspace(0,1,overlap)[:,None]
initial[-overlap:]=initial[-overlap:]*(1-alpha)+wave[prefix-overlap:prefix]*alpha
initial=np.concatenate([initial,wave[prefix:]])
last=last_new;slot+=1
write_json(G/'ace_worker_state.json',{'pid':os.getpid(),'generation_seed':base_seed,'composition_policy':composition_policy,'profile':profile,'phase':'preparing','accepted_chunks':slot,'accepted_buffer_seconds':len(initial)/48000,'first_accepted_audio_seconds':first_accepted_audio_seconds,'elapsed_seconds':time.monotonic()-session_started})
if slot>8:raise RuntimeError('Initial buffer did not fill within bounded preparation')
save_wave(G/'ace_initial.wav',initial);np.save(G/'ace_initial.npy',last)
write_json(G/'ace_initial.json',{'generation_seed':base_seed,'composition_policy':composition_policy,'composer':'ace','startup_seconds':time.monotonic()-session_started,'first_accepted_audio_seconds':first_accepted_audio_seconds,'model_load_seconds':0. if selection else model_load_seconds,'resident_model_load_seconds':model_load_seconds,'worker_uptime_seconds':time.monotonic()-BOOT,'resident_reused':bool(selection),'resident_capable':resident,'resident_session_protocol':SESSION_PROTOCOL if resident else None,'resident_model_identity':str(id(c)),'conditioning_bank_sha256':planned.bank_hash if resident else None,'host_peak_rss_kib':resource.getrusage(resource.RUSAGE_SELF).ru_maxrss,'duration':len(initial)/48000,'initial_buffer_target_seconds':initial_buffer_target,'source_identity':'kpop_control','prepared_profile':profile if windowed else 'legacy','preparation_id':preparation_id,'continuation_policy':'quality-gated fixed lookahead','generation':preparation,'prepared_identity':True,'output_gain':POLICY.output_gain,'created_wall':time.time()})
write_json(G/'ace_worker_state.json',{'pid':os.getpid(),'generation_seed':base_seed,'composition_policy':composition_policy,'profile':profile,'phase':'READY','initial_buffer_seconds':len(initial)/48000})
(G/'request.json').unlink(missing_ok=True);ready.write_text('ace');print('ACE_READY',flush=True)
with session_lease(G) if resident else nullcontext():prepare_session()
while True:
session_request=G/'session_request.json'
if resident and session_request.exists():
selection=json.loads(session_request.read_text());session_request.unlink()
try:
with session_lease(G):prepare_session(selection)
write_json(G/'session_result.json',{**selection,'phase':'READY','preparation_id':preparation_id,'applied_initial_buffer_target_seconds':initial_buffer_target})
except BlockingIOError:
write_json(G/'session_result.json',{'id':selection.get('id'),'error':'Playback owns the resident session'})
except Exception as error:
write_json(G/'session_result.json',{'id':selection.get('id'),'error':str(error)})
ready.unlink(missing_ok=True)
raise
continue
request=G/'request.json'
if not request.exists():time.sleep(.1);continue
req=json.loads(request.read_text());request.unlink()
if req.get('composer')!='ace':raise ValueError('ACE worker received a request for a different backend')
if req.get('profile',profile)!=profile:raise ValueError('Resident ACE profile mismatch; stop and prepare requested profile before replay')
if req.get('identity')!='kpop_control':raise ValueError('ACE experimental worker only has the prepared kpop_control identity')
if req.get('generation_seed')!=base_seed:raise ValueError('Generation request does not match prepared session seed')
job=int(req['id']);busy=G/'busy';busy.write_text(str(job));start=time.monotonic()
try:
case={'base':'repaint_verse','approach':'repaint_chorus','bridge_transition':'repaint_bridge','closing':'repaint_outro'}.get(req['conditioning'])
if case is None and req['conditioning'] in ('verse','prechorus','chorus','bridge','outro'):case=req['conditioning'] if windowed else 'repaint_'+req['conditioning']
if case is None:raise ValueError('Unsupported ACE section intent')
if windowed:case=case.removeprefix('repaint_')
previous=np.load(req['latents'])
if planned:case=planned.begin(previous,arrival=case=='outro')
deadline=req.get('playback_deadline_monotonic')
wave,latent,stats=qualified.run(case,int(req['seed']),previous,deadline=deadline,record=record_for(job))
if wave is None:
write_json(G/f'result_{job}.json',{**req,**stats,'id':job,'composer':'ace','seconds':time.monotonic()-start});continue
if planned:planned.accept(wave,latent)
wav=G/f'ace_job_{job}.wav';lat=G/f'ace_job_{job}.npy';save_wave(wav,wave);np.save(lat,latent)
result={**req,**stats,'id':job,'composer':'ace','backend':'ACE-Step1.5 turbo native Chestnut, prepared identity','wav':str(wav),'latents':str(lat),'seconds':time.monotonic()-start,'retained_seconds':stats['prefix_seconds'],'new_audio_start_frame':round(stats['prefix_seconds']*48000),'overlap_frames':96000,'sample_rate':48000,'output_gain':POLICY.output_gain,'conditioning':req['conditioning']}
write_json(G/f'result_{job}.json',result);print('ACE_RESULT',job,result['seconds'],flush=True)
except Exception as e:
write_json(G/f'result_{job}.json',{**req,'error':str(e),'composer':'ace'});traceback.print_exc();raise
finally:busy.unlink(missing_ok=True)
finally:
ready.unlink(missing_ok=True)
write_json(G/'ace_worker_state.json',{'pid':os.getpid(),'generation_seed':base_seed,'composition_policy':composition_policy,'profile':profile,'phase':'Stopped'})
@@ -1,20 +0,0 @@
import os,time,json,fcntl,resource
from pathlib import Path
import numpy as np
from native_ace import DiT,tensor,time_features
from tinygrad import TinyJit,Device
from tinygrad.helpers import GlobalCounters
P=Path(__file__).resolve().parent;O=P/'reference60';prefix=os.environ.get('ACE_RESULT_PREFIX','');lock=open('/data/roadscore/generated/gpu.lock','w');fcntl.flock(lock,fcntl.LOCK_EX|fcntl.LOCK_NB)
t=time.monotonic();m=DiT(P/'weights');loaded=time.monotonic()-t;print('WEIGHTS_READY',loaded,GlobalCounters.mem_used,flush=True)
x,cond,ctx=[tensor(np.load(O/(k+'.npy')).astype(np.float16)) for k in ['hidden_states','encoder_hidden_states','context_latents']];m.prepare_shape(x.shape[1],align_tokens=32);tf=tensor(time_features([1]));rf=tensor(time_features([0]));fn=TinyJit(m.forward);times=[]
for i in range(4):
t=time.monotonic();out=fn(x,cond,ctx,tf,rf);Device['AMD'].synchronize();times.append(time.monotonic()-t);print('FULL',i,times[-1],GlobalCounters.mem_used,flush=True)
y=out.numpy().astype(np.float32);np.save(P/(prefix+'full_actual.npy'),y);reference_file=os.environ.get('ACE_REFERENCE','velocity_fp16.npy');ref=np.load(O/reference_file);error=y-ref
report={'reference_file':reference_file,'tc_opt':os.environ.get('TC_OPT','0'),'alignment_tokens':32,'sampler':'8-step Euler; DCW disabled','load_seconds':loaded,'full_forward_seconds':times,'relative_rmse':float(np.linalg.norm(error)/np.linalg.norm(ref)),'peak_relative_error':float(abs(error).max()/abs(ref).max()),'max_abs_error':float(abs(error).max()),'finite':bool(np.isfinite(y).all()),'peak_host_mib':resource.getrusage(resource.RUSAGE_SELF).ru_maxrss/1024,'tracked_gpu_bytes':GlobalCounters.mem_used,'weights_bytes':sum(w.numel()*w.dtype.itemsize for w in m.w.values()),'audio_seconds':x.shape[1]/25,'projection_8steps_rtf':8*times[-1]/(x.shape[1]/25)}
(P/(prefix+'full_result.json')).write_text(json.dumps(report,indent=2));print(report,flush=True)
assert report['finite'] and report['relative_rmse']<.01 and report['peak_relative_error']<.01
# Eight trained-weight flow steps, identical initial noise and cached conditioning.
schedule=np.linspace(1,0,9,dtype=np.float32).tolist();start=time.monotonic()
for i in range(8):
Device['AMD'].synchronize();tf.assign(tensor(time_features([schedule[i]]))).realize();v=fn(x,cond,ctx,tf,rf);x.assign(x-v*(schedule[i]-schedule[i+1])).realize()
Device['AMD'].synchronize();elapsed=time.monotonic()-start;np.save(P/(prefix+'generated_latents.npy'),x.numpy());report.update(generation_seconds=elapsed,generation_rtf=elapsed/(x.shape[1]/25));(P/(prefix+'full_result.json')).write_text(json.dumps(report,indent=2));print('GENERATION',elapsed,flush=True)
@@ -1,262 +0,0 @@
"""Bounded offline Mac initials from the exact current native conditioning bank.
Uses the official Torch DiT layers with native exported FP16 weights and explicit
native masks/time/rotary inputs. MPS arithmetic and FP32 VAE decoding differ from
Chestnut: this is a listening shortlist, never an exact cross-backend replay.
No planner, download, route reader, audio device, or hardware connection is used.
"""
import argparse
import hashlib
import json
import os
from pathlib import Path
import sys
import time
os.environ.update(HF_HUB_OFFLINE='1', TRANSFORMERS_OFFLINE='1', HF_HUB_DISABLE_TELEMETRY='1',
TOKENIZERS_PARALLELISM='false', PYTORCH_ENABLE_MPS_FALLBACK='1')
import numpy as np
HERE = Path(__file__).resolve().parent
sys.path.insert(0, str(HERE.parents[1] / 'prototype'))
from cached_composition import validate_bank
from generation_seed import sample_seed
from host_hook_adapter import HostHookAdapter
from quality_gate import HOOK_POLICY, inspect
from window_policy import retained_end
BANK_SHA256 = 'd8570881b873a093ea509cafd8592e6b78ced880e9cc00d8f5fbfc8e67e39daf'
BASELINE = 1496885951
def digest(path):
with Path(path).open('rb') as stream:
return hashlib.file_digest(stream, 'sha256').hexdigest()
def array_digest(value):
return hashlib.sha256(value.tobytes()).hexdigest()
def time_features(value):
# Same NumPy FP16 feature construction as native_ace.time_features.
f = np.exp(-np.log(10000) * np.arange(128, dtype=np.float32) / 128)
v = np.asarray(value, dtype=np.float32).reshape(-1, 1) * 1000 * f[None, :]
return np.concatenate((np.cos(v), np.sin(v)), -1).astype(np.float16)
class Reference:
def __init__(self, assets, cond, context, valid):
import torch
from diffusers import AutoencoderOobleck
from safetensors import safe_open
model_root = assets / 'experiments/composition_20260916/models/ace/checkpoints'
exports = assets / 'experiments/ace_chestnut_20260916'
sys.path.insert(0, str(model_root / 'acestep-v15-turbo'))
from configuration_acestep_v15 import AceStepConfig
from modeling_acestep_v15_turbo import AceStepDiTModel
self.torch = torch
config = AceStepConfig.from_pretrained(model_root / 'acestep-v15-turbo', local_files_only=True)
config._attn_implementation = 'sdpa'
with torch.device('meta'):
self.model = AceStepDiTModel(config)
weights = {p.stem: torch.from_numpy(np.load(p, allow_pickle=False))
for p in sorted((exports / 'weights').glob('*.npy'))}
# Prove that the exported native DiT tensors are this bank's checkpoint.
with safe_open(str(model_root / 'acestep-v15-turbo/model.safetensors'), framework='pt', device='cpu') as source:
expected = {key.removeprefix('decoder.') for key in source.keys() if key.startswith('decoder.')}
if set(weights) != expected:
raise ValueError('Incomplete or unexpected native DiT export')
for name, value in weights.items():
if value.dtype != torch.float16 or not torch.equal(value, source.get_tensor('decoder.' + name).half()):
raise ValueError('Native DiT export differs from matching checkpoint: ' + name)
self.model.load_state_dict(weights, assign=True)
# Rotary values are supplied explicitly below, so discard the meta buffer.
self.model.rotary_emb = torch.nn.Identity()
self.model = self.model.to('mps').eval()
del weights
vae = AutoencoderOobleck.from_pretrained(model_root / 'vae', local_files_only=True).eval()
for module in vae.decoder.modules():
if hasattr(module, 'weight_g'):
torch.nn.utils.remove_weight_norm(module)
decoder_weights = {p.stem: torch.from_numpy(np.load(p, allow_pickle=False))
for p in sorted((exports / 'vae_weights').glob('*.npy'))}
reference_weights = vae.decoder.state_dict()
if set(decoder_weights) != set(reference_weights):
raise ValueError('Incomplete or unexpected native VAE export')
for name, value in decoder_weights.items():
if value.dtype != torch.float16 or not torch.equal(value, reference_weights[name].half()):
raise ValueError('Native VAE export differs from matching checkpoint: ' + name)
vae.decoder.load_state_dict(decoder_weights, assign=True)
self.decoder = vae.decoder.float().to('mps').eval()
del vae, decoder_weights, reference_weights
self.cond = self.tensor(cond)
self.context = self.tensor(context)
self.cross_mask = self.tensor(np.where(valid[:, None, None, :], 0, -np.inf).astype(np.float16))
n = context.shape[1] // 2
positions = np.arange(n, dtype=np.float32)[:, None]
freq = positions / (1000000 ** (np.arange(0, 128, 2, dtype=np.float32) / 128))[None, :]
freq = np.concatenate((freq, freq), -1)[None]
self.rotary = (self.tensor(np.cos(freq).astype(np.float16)), self.tensor(np.sin(freq).astype(np.float16)))
index = np.arange(n)
self.local_mask = self.tensor(np.where(abs(index[:, None] - index[None, :]) <= 128, 0, -np.inf).astype(np.float16)[None, None])
def tensor(self, value):
return self.torch.from_numpy(value).to('mps')
def velocity(self, latent, timestep):
"""Official layers with the exact masks/features used by native_ace.DiT.
The vendor model.forward resets caller masks, so use its layers directly
to retain the native padded-conditioning mask without changing vendor code.
"""
m = self.model
torch = self.torch
def embedding(block, value):
z = block.linear_2(block.act1(block.linear_1(self.tensor(time_features([value])))))
return z, block.time_proj(block.act2(z)).reshape(1, 6, 2048)
t, projection = embedding(m.time_embed, timestep)
r, projection_r = embedding(m.time_embed_r, 0)
hidden = m.proj_in(torch.cat((self.context, latent), dim=-1))
condition = m.condition_embedder(self.cond)
for index, layer in enumerate(m.layers):
hidden = layer(hidden, self.rotary, projection + projection_r,
attention_mask=self.local_mask if index % 2 == 0 else None,
encoder_hidden_states=condition, encoder_attention_mask=self.cross_mask,
use_cache=False)[0]
shift, scale = (m.scale_shift_table + (t + r).unsqueeze(1)).chunk(2, dim=1)
return m.proj_out((m.norm_out(hidden) * (1 + scale) + shift).type_as(hidden))
def generate(self, noise, progress):
torch = self.torch
started = time.monotonic()
with torch.inference_mode():
x = self.tensor(noise)
for i, timestep in enumerate(np.linspace(1, 0, 9)[:-1]):
step = time.monotonic()
x = x - self.velocity(x, float(timestep)) * .125
torch.mps.synchronize()
if not torch.isfinite(x).all().item():
raise ValueError('Nonfinite diffusion latent')
progress(dict(step=i, seconds=time.monotonic() - step))
latent = x.cpu().numpy()
diffusion = time.monotonic() - started
started = time.monotonic()
# Native ChunkDecoder layout: 375-frame windows, 250-frame core.
n, window, core = latent.shape[1], 375, 250
halo = (window - core) // 2
parts = []
for start in range(0, n, core):
left = max(0, min(start - halo, n - window))
end = min(start + core, n)
chunk = self.tensor(latent[:, left:left + window].transpose(0, 2, 1).copy()).float()
decoded = self.decoder(chunk).cpu().numpy()[0].T
parts.append(decoded[(start - left) * 1920:(end - left) * 1920])
wave = np.concatenate(parts)
decode = time.monotonic() - started
return wave, latent, diffusion, decode
def main():
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument('--assets-root', type=Path, required=True)
parser.add_argument('--bank', type=Path, required=True)
parser.add_argument('--output', type=Path, required=True)
parser.add_argument('--seeds', type=int, nargs='+', default=[BASELINE, 3277374468])
parser.add_argument('--check-only', action='store_true')
args = parser.parse_args()
if not 1 <= len(args.seeds) <= 8 or len(set(args.seeds)) != len(args.seeds) or any(not 0 <= n < 2**32 for n in args.seeds):
parser.error('One to eight distinct uint32 seeds are required')
args.output.mkdir(parents=True, exist_ok=False)
report = {'status': 'verifying_inputs', 'backend': 'Mac PyTorch MPS; exported FP16 DiT; exported-weight FP32 VAE',
'cross_backend_guarantee': False, 'musical_acceptance': 'pending user listening; no best seed selected',
'composition_policy': 'hook-cache-v1', 'profile': 'prism', 'presentation_policy': 'conservative-v3',
'presentation_scope': 'isolated core audio, gain0.65; route DSP/UI/synchronization require native verification',
'sampler': {'steps': 8, 'method': 'Euler', 'schedule': np.linspace(1, 0, 9).tolist(), 'shift': 1.,
'dcw_enabled': False, 'time_embed_r_value': 0., 'noise': 'NumPy default_rng PCG64 standard_normal cast FP16',
'model_seconds': 30, 'target_commit_seconds': 28, 'prefix_seconds': 0, 'rerolls_in_this_audition': 0},
'runs': [], 'source_sha256': {p.name: digest(p) for p in [Path(__file__), HERE / 'native_ace.py', HERE / 'ace_runtime.py',
HERE / 'chunk_decode.py', HERE / 'quality_gate.py', HERE / 'window_policy.py', HERE.parents[1] / 'prototype/generation_seed.py']}}
def save():
(args.output / 'audition.json').write_text(json.dumps(report, indent=2))
save()
try:
if digest(args.bank / 'bank.json') != BANK_SHA256:
raise ValueError('Exact current-native bank missing: expected ' + BANK_SHA256)
manifest = validate_bank(args.bank, 'prism')
identities = HostHookAdapter(args.assets_root, preparation_only=True).fingerprints()
if any(identities[key] != manifest['roles']['initial']['request'][key] for key in identities):
raise ValueError('Local checkpoint/preparation fingerprints differ from exact bank')
initial = args.bank / 'initial'
cond = np.load(initial / 'encoder_hidden_states.npy', allow_pickle=False).astype(np.float16)
context = np.load(initial / 'context_latents.npy', allow_pickle=False).astype(np.float16)
valid = np.load(initial / 'encoder_attention_mask.npy', allow_pickle=False).astype(bool)
if context.shape != (1, 750, 128) or (initial / 'sampler_repaint_mask.npy').exists():
raise ValueError('Expected unpainted30s native initial')
width = max(256, ((cond.shape[1] + 31) // 32) * 32)
pad = width - cond.shape[1]
cond = np.pad(cond, ((0, 0), (0, pad), (0, 0)))
valid = np.pad(valid, ((0, 0), (0, pad)))
import torch
import mlx.core as mx
import soundfile as sf
import acestep
if not torch.backends.mps.is_available():
raise RuntimeError('MPS is unavailable')
expected_package = args.assets_root / 'experiments/composition_20260916/vendor/ACE-Step-1.5/acestep'
if Path(acestep.__file__).resolve().parent != expected_package.resolve():
raise ValueError('ACE import differs from fingerprinted package')
report.update(status='inputs_verified', conditioning_bank=str(args.bank), conditioning_bank_sha256=BANK_SHA256,
model_fingerprints=identities, conditioning_tensor_sha256=manifest['roles']['initial']['sha256'],
versions={'numpy': np.__version__, 'torch': torch.__version__, 'mlx': str(mx.__file__), 'soundfile': sf.__version__},
prepared_inputs={name: {'shape': list(a.shape), 'dtype': str(a.dtype), 'sha256': array_digest(a)}
for name, a in [('encoder_hidden_states', cond), ('context_latents', context), ('encoder_attention_mask', valid)]})
save()
if args.check_only:
return
started = time.monotonic()
reference = Reference(args.assets_root, cond, context, valid)
report.update(status='generating', exported_weights_match_checkpoint=True, model_load_seconds=time.monotonic() - started)
save()
for session_seed in args.seeds:
directory = args.output / str(session_seed)
directory.mkdir()
noise_seed = sample_seed(session_seed, 'prepare', 0)
noise = np.random.default_rng(noise_seed).standard_normal((1, 750, 64)).astype(np.float16)
np.save(directory / 'noise.npy', noise)
row = {'session_seed': session_seed, 'sample_seed': noise_seed, 'phase': 'prepare', 'index': 0,
'noise_sha256': array_digest(noise), 'attempt': 0, 'steps': [], 'status': 'generating'}
report['runs'].append(row)
save()
def progress(value):
row['steps'].append(value)
save()
print(json.dumps({'session_seed': session_seed, **value}), flush=True)
wave, latent, generation, decode = reference.generate(noise, progress)
endpoint_error = None
try:
frames, endpoint = retained_end(wave, 48000, 0, 28)
except ValueError as error:
frames, endpoint_error = 700, str(error)
endpoint = {'rejected': endpoint_error}
committed = wave[:frames * 1920]
quality = inspect(committed, 48000, 0, role='initial', policy=HOOK_POLICY, endpoint_error=endpoint_error)
sf.write(directory / 'initial_raw.wav', committed, 48000, subtype='FLOAT')
sf.write(directory / 'initial_listen.wav', committed * np.float32(.65), 48000, subtype='FLOAT')
np.save(directory / 'committed_latents.npy', latent[:, :frames])
row.update(status='technical_pass_listening_pending' if quality['accepted'] else 'quality_rejected_no_reroll',
generation_seconds=generation, decode_seconds=decode, endpoint=endpoint, quality=quality,
committed_seconds=len(committed) / 48000, raw_pcm_sha256=array_digest(committed),
raw_peak=float(np.abs(committed).max()), listen_file=str(directory / 'initial_listen.wav'))
save()
print(json.dumps({'session_seed': session_seed, 'status': row['status'], 'seconds': generation + decode}), flush=True)
report['status'] = 'complete_listening_pending'
save()
except Exception as error:
report.update(status='stopped', error=str(error))
save()
raise
if __name__ == '__main__':
main()
@@ -1,200 +0,0 @@
"""Three bounded Mac song auditions with fresh semantic plans, preserving every attempt.
Prepare and render are separate processes so the musical planner is unloaded before
the verified native-weight MPS decoder loads. No route input, audio output, network,
hardware access, automatic rerolls, or changes to the runtime conditioning bank.
"""
import argparse
import hashlib
import json
import os
from pathlib import Path
import resource
import secrets
import sys
import time
os.environ.update(HF_HUB_OFFLINE='1', TRANSFORMERS_OFFLINE='1',
HF_HUB_DISABLE_TELEMETRY='1', TOKENIZERS_PARALLELISM='false',
PYTORCH_ENABLE_MPS_FALLBACK='1')
import numpy as np
HERE = Path(__file__).resolve().parent
sys.path.insert(0, str(HERE.parents[1] / 'prototype'))
from hook_planning import PlanCache, PlanRequest, digest, request_plan, validate_prepared
from host_hook_adapter import HostHookAdapter
from generation_seed import sample_seed
from quality_gate import HOOK_POLICY, inspect
from window_policy import retained_end
def save(root, report):
report['peak_process_rss_bytes'] = max(report.get('peak_process_rss_bytes', 0),
resource.getrusage(resource.RUSAGE_SELF).ru_maxrss)
(root / 'audition.json').write_text(json.dumps(report, indent=2))
def prepare(args):
args.output.mkdir(parents=True, exist_ok=False)
seeds = [secrets.randbits(32) for _ in range(3)]
if len(set(seeds)) != 3:
raise RuntimeError('Fresh-seed collision; do not silently substitute candidates')
report = dict(status='preparing', profile='prism', attempts_per_song=1,
preparation='Fresh actual semantic plan for each song; same current v4 groove prompt',
backend='Mac preparation-only MLX planner, then official Torch MPS DiT with native exported weights',
listening_scope='Opening excerpts only; no route, continuations, gestures or presentation DSP',
gain=.65, musical_acceptance='User listening required; no winner selected',
source_sha256={p.name: digest(p) for p in (Path(__file__), HERE/'host_hook_adapter.py',
HERE/'audition_cached_initial_mps.py', HERE.parents[1]/'prototype/hook_planning.py')},
songs=[dict(label=f'Song {i+1}', session_seed=seed, attempt=0, status='registered')
for i, seed in enumerate(seeds)])
save(args.output, report)
started = time.monotonic()
try:
adapter = HostHookAdapter(args.assets_root, preparation_only=True)
identities = adapter.fingerprints()
report['model_fingerprints'] = identities
cache = PlanCache(args.output/'plans')
for row in report['songs']:
request = request_plan(session_seed=row['session_seed'], plan_index=0,
profile='prism', section='initial', window_seconds=30, **identities)
row.update(status='preparing', semantic_seed=request.semantic_seed, request=request.identity())
save(args.output, report)
tick = time.monotonic()
directory, hit = cache.resolve(request, adapter)
plan = json.loads((directory/'semantic_plan.json').read_text())
row.update(status='prepared', plan_key=request.cache_key, plan_seconds=time.monotonic()-tick,
plan_cache_hit=hit, plan_sha256=digest(directory/'semantic_plan.json'),
audio_codes_sha256=hashlib.sha256(plan['audio_codes'].encode()).hexdigest())
save(args.output, report)
print(json.dumps({k: row[k] for k in ('label','session_seed','semantic_seed','status','plan_seconds')}), flush=True)
report.update(status='prepared', preparation_seconds=time.monotonic()-started,
preparation_peak_rss_bytes=resource.getrusage(resource.RUSAGE_SELF).ru_maxrss)
if len({row['audio_codes_sha256'] for row in report['songs']}) != 3:
raise ValueError('Planner produced duplicate semantic code sequences; do not claim distinct plans')
save(args.output, report)
except BaseException as error:
report.update(status='preparation_failed', error=repr(error))
save(args.output, report)
raise
def render(args):
from audition_cached_initial_mps import Reference, array_digest
import soundfile as sf
report = json.loads((args.output/'audition.json').read_text())
if report['status'] != 'prepared':
raise ValueError('Render requires an untouched complete preparation batch')
started = time.monotonic()
reference = None
try:
report['status'] = 'rendering'
save(args.output, report)
for i, row in enumerate(report['songs']):
directory = args.output / f'song_{i+1}'
directory.mkdir(exist_ok=False)
plan = args.output / 'plans' / row['plan_key']
row['conditioning_sha256'] = validate_prepared(PlanRequest(**row['request']), plan)
cond = np.load(plan/'encoder_hidden_states.npy', allow_pickle=False).astype(np.float16)
context = np.load(plan/'context_latents.npy', allow_pickle=False).astype(np.float16)
valid = np.load(plan/'encoder_attention_mask.npy', allow_pickle=False).astype(bool)
width = max(256, ((cond.shape[1]+31)//32)*32)
cond = np.pad(cond, ((0,0),(0,width-cond.shape[1]),(0,0)))
valid = np.pad(valid, ((0,0),(0,width-valid.shape[1])))
if context.shape != (1,750,128):
raise ValueError('Unexpected initial shape')
if reference is None:
tick = time.monotonic()
reference = Reference(args.assets_root, cond, context, valid)
report['render_model_load_seconds'] = time.monotonic()-tick
else:
reference.cond = reference.tensor(cond)
reference.context = reference.tensor(context)
reference.cross_mask = reference.tensor(np.where(valid[:,None,None,:],0,-np.inf).astype(np.float16))
noise_seed = sample_seed(row['session_seed'], 'prepare', 0)
noise = np.random.default_rng(noise_seed).standard_normal((1,750,64)).astype(np.float16)
np.save(directory/'noise.npy', noise)
row.update(status='generating', sample_seed=noise_seed, noise_sha256=array_digest(noise), steps=[])
save(args.output, report)
def progress(value):
row['steps'].append(value)
save(args.output, report)
print(json.dumps(dict(label=row['label'], **value)), flush=True)
wave, latent, generation, decode = reference.generate(noise, progress)
endpoint_error = None
try:
frames, endpoint = retained_end(wave, 48000, 0, 28)
except ValueError as error:
frames, endpoint_error = 700, str(error)
endpoint = {'rejected': endpoint_error}
committed = wave[:frames*1920]
quality = inspect(committed, 48000, 0, role='initial', policy=HOOK_POLICY, endpoint_error=endpoint_error)
sf.write(directory/'raw.wav', committed, 48000, subtype='FLOAT')
sf.write(directory/'listen.wav', committed*np.float32(.65), 48000, subtype='FLOAT')
np.save(directory/'committed_latents.npy', latent[:,:frames])
row.update(status='technical_pass_listening_pending' if quality['accepted'] else 'technical_reject_no_reroll',
generation_seconds=generation, decode_seconds=decode, quality=quality,
endpoint=endpoint, seconds=len(committed)/48000, raw_pcm_sha256=array_digest(committed),
listen_file=str(directory/'listen.wav'), raw_peak=float(np.abs(committed).max()))
save(args.output, report)
print(json.dumps({k:row[k] for k in ('label','status','seconds','generation_seconds','decode_seconds')}), flush=True)
report.update(status='complete_listening_pending', render_seconds=time.monotonic()-started,
render_peak_rss_bytes=resource.getrusage(resource.RUSAGE_SELF).ru_maxrss)
save(args.output, report)
finalize_auditions(args.output, report)
write_page(args.output, report)
except BaseException as error:
report.update(status='render_failed', error=repr(error))
save(args.output, report)
raise
def finalize_auditions(root, report):
"""One common gain for every candidate and the reference, with raw PCM retained."""
import soundfile as sf
for row in report['songs']:
path = Path(row['listen_file'])
original = path.with_name('listen_original_gain_065.wav')
if original.exists():
raise FileExistsError('Audition leveling already finalized; preserve it')
path.rename(original)
wave, rate = sf.read(path.with_name('raw.wav'), dtype='float32', always_2d=True)
listen = wave*np.float32(.60)
if not np.isfinite(listen).all() or np.abs(listen).max() >= 1:
raise ValueError('Unsafe listening peak; no automatic per-song normalization')
sf.write(path, listen, rate, subtype='FLOAT')
row.update(listening_peak=float(np.abs(listen).max()), original_gain_065_file=str(original))
reference = root.parent/'mac-seeds/mps-initials/1496885951/initial_listen.wav'
wave, rate = sf.read(reference, dtype='float32', always_2d=True)
sf.write(root/'reference_A.wav', wave*np.float32(.60/.65), rate, subtype='FLOAT')
report.update(gain=.60, gain_note='Common0.60 audition gain for all songs and A; original0.65 files and raw generation retained',
reference=dict(source=str(reference), sha256=digest(reference), source_gain=.65),
finalization_source_sha256=digest(Path(__file__)))
save(root, report)
def write_page(root, report):
cards = ''.join(f'''<article><span class="tag">New semantic plan · {r['seconds']:.1f}s</span>
<h2>{r['label']}</h2><audio controls preload="metadata" src="song_{i+1}/listen.wav"></audio>
<p>Seed {r['session_seed']} · {'Technical checks passed' if r['quality']['accepted'] else 'Technical check failed; preserved for review'}</p></article>'''
for i,r in enumerate(report['songs']))
page = '''<!doctype html><html lang="en"><meta charset="utf-8"><meta name="viewport" content="width=device-width, initial-scale=1">
<title>RoadScore · new songs</title><style>body{margin:0;background:#101317;color:#edf2f7;font:18px system-ui,sans-serif}main{max-width:900px;margin:60px auto;padding:0 24px}h1{font-size:44px;letter-spacing:-1.5px;margin-bottom:12px}p{color:#afbbc8;line-height:1.6}article{border:1px solid #37414b;border-radius:18px;padding:24px;margin:22px 0;background:#191f27}audio{width:100%}.tag{font-size:13px;color:#75e2dc}h2{margin:10px 0 20px}a{color:#75e2dc}</style>
<main><span class="tag">ROADSCORE · PRISM</span><h1>Three new song ideas</h1><p>Each opening has a new musical plan and seed, using the same groove-focused Prism strategy. These are core music only, with the same gain and no added cues. Pick the hook and groove you like; none is selected automatically.</p>''' + cards + '''
<article><span class="tag">Current reference</span><h2>A · current showcase</h2><audio controls preload="metadata" src="reference_A.wav"></audio><p>The earlier A–H set varied the backing under one fixed musical plan. The three above generate new plans.</p></article>
<p>Opening auditions, not complete route scores. A selected new song needs its own matching continuation bank and full-route verification before replacing the preserved demo. Mac arithmetic is not bit-identical to Chestnut.</p></main>
<script>const players=[...document.querySelectorAll('audio')];players.forEach(p=>p.addEventListener('play',()=>players.forEach(q=>{if(q!==p)q.pause()})));</script></html>'''
(root/'index.html').write_text(page)
def main():
p = argparse.ArgumentParser(description=__doc__)
p.add_argument('phase', choices=('prepare','render'))
p.add_argument('--assets-root', type=Path, required=True)
p.add_argument('--output', type=Path, required=True)
args = p.parse_args()
{'prepare':prepare,'render':render}[args.phase](args)
if __name__ == '__main__':
main()
@@ -1,33 +0,0 @@
"""Actual prepared conditions, native ACE DiT; no route or physical audio access."""
import os,time,json,fcntl,resource,gc
from pathlib import Path
import numpy as np
from native_ace import DiT,tensor,time_features
from native_vae import VAE
from chunk_decode import ChunkDecoder
import track_memory
from tinygrad import TinyJit,Device
from tinygrad.helpers import GlobalCounters
P=Path(__file__).resolve().parent
lock=open('/data/roadscore/generated/gpu.lock','w');fcntl.flock(lock,fcntl.LOCK_EX|fcntl.LOCK_NB)
t=time.monotonic();m=DiT(P/'weights');load=time.monotonic()-t;t=time.monotonic();vae=VAE(P/'vae_weights');vae_load=time.monotonic()-t;decoder=ChunkDecoder(vae)
for duration in [int(v) for v in os.environ.get('ACE_DURATIONS','45,60,90').split(',')]:
O=P/f'reference{duration}';x,cond,ctx=[tensor(np.load(O/(k+'.npy')).astype(np.float16)) for k in ['hidden_states','encoder_hidden_states','context_latents']]
m.prepare_shape(x.shape[1]);tf=tensor(time_features([1]));rf=tensor(time_features([0]));fn=TinyJit(m.forward);times=[]
for i in range(3):
t=time.monotonic();y=fn(x,cond,ctx,tf,rf);Device['AMD'].synchronize();times.append(time.monotonic()-t);print('WARMUP',duration,i,times[-1],flush=True)
schedule=np.linspace(1,0,9).tolist();t=time.monotonic()
for i in range(8):
Device['AMD'].synchronize();tf.assign(tensor(time_features([schedule[i]]))).realize();v=fn(x,cond,ctx,tf,rf);x.assign(x-v*(schedule[i]-schedule[i+1])).realize()
Device['AMD'].synchronize();elapsed=time.monotonic()-t;a=x.numpy();np.save(O/'chestnut_latents.npy',a)
r={'duration':x.shape[1]/25,'generation_seconds':elapsed,'latent_rtf':elapsed/(x.shape[1]/25),'load_seconds':load,'forward_seconds':times,'tc_opt':os.environ.get('TC_OPT'),'tracked_gpu_bytes':GlobalCounters.mem_used,'host_peak_mib':resource.getrusage(resource.RUSAGE_SELF).ru_maxrss/1024,'finite':bool(np.isfinite(a).all())}
# Fixed375-frame decode windows retain ample context and cap resident workspace.
if not decoder.graphs:
for _ in range(2):decoder.decode(a[:,:min(375,a.shape[1])])
pcm,decode_time=decoder.decode(a);decode_times=[decode_time];np.save(O/'bounded_pcm.npy',pcm)
r.update(vae_load_seconds=vae_load,decode_seconds=decode_times,total_warm_seconds=elapsed+decode_time,total_warm_rtf=(elapsed+decode_time)/r['duration'],tracked_gpu_bytes=GlobalCounters.mem_used,tracked_gpu_peak_bytes=track_memory.peak,host_peak_mib=resource.getrusage(resource.RUSAGE_SELF).ru_maxrss/1024,pcm_finite=bool(np.isfinite(pcm).all()),sampler='8-step Euler, DCW disabled; synchronize before timestep USB copy',conditioning='actual Mac-prepared inputs; excluded from warm device timing',decoder='375latent window/250core, discard context')
if (O/'chestnut_pcm.npy').exists():
ref=np.load(O/'chestnut_pcm.npy');r['chunk_relative_rmse']=float(np.linalg.norm(pcm-ref)/np.linalg.norm(ref));assert r['chunk_relative_rmse']<.01
(O/'bounded_result.json').write_text(json.dumps(r,indent=2));print('RESULT',r,flush=True)
assert r['finite']
del fn,x,cond,ctx,tf,rf,y,v;gc.collect()
@@ -1,29 +0,0 @@
"""Actual prepared conditions, native ACE DiT; no route or physical audio access."""
import os,time,json,fcntl,resource,gc
from pathlib import Path
import numpy as np
from native_ace import DiT,tensor,time_features
from native_vae import VAE
from tinygrad import TinyJit,Device
from tinygrad.helpers import GlobalCounters
P=Path(__file__).resolve().parent
lock=open('/data/roadscore/generated/gpu.lock','w');fcntl.flock(lock,fcntl.LOCK_EX|fcntl.LOCK_NB)
t=time.monotonic();m=DiT(P/'weights');load=time.monotonic()-t;t=time.monotonic();vae=VAE(P/'vae_weights');vae_load=time.monotonic()-t
for duration in [int(v) for v in os.environ.get('ACE_DURATIONS','30,45,60,90').split(',')]:
O=P/f'reference{duration}';x,cond,ctx=[tensor(np.load(O/(k+'.npy')).astype(np.float16)) for k in ['hidden_states','encoder_hidden_states','context_latents']]
m.prepare_shape(x.shape[1]);tf=tensor(time_features([1]));rf=tensor(time_features([0]));fn=TinyJit(m.forward);times=[]
for i in range(3):
t=time.monotonic();y=fn(x,cond,ctx,tf,rf);Device['AMD'].synchronize();times.append(time.monotonic()-t);print('WARMUP',duration,i,times[-1],flush=True)
schedule=np.linspace(1,0,9).tolist();t=time.monotonic()
for i in range(8):
tf.assign(tensor(time_features([schedule[i]]))).realize();v=fn(x,cond,ctx,tf,rf);x.assign(x-v*(schedule[i]-schedule[i+1])).realize()
Device['AMD'].synchronize();elapsed=time.monotonic()-t;a=x.numpy();np.save(O/'chestnut_latents.npy',a)
r={'duration':x.shape[1]/25,'generation_seconds':elapsed,'latent_rtf':elapsed/(x.shape[1]/25),'load_seconds':load,'forward_seconds':times,'tc_opt':os.environ.get('TC_OPT'),'tracked_gpu_bytes':GlobalCounters.mem_used,'host_peak_mib':resource.getrusage(resource.RUSAGE_SELF).ru_maxrss/1024,'finite':bool(np.isfinite(a).all())}
decode=TinyJit(vae.forward);z=tensor(a.transpose(0,2,1));decode_times=[]
for j in range(3):
t=time.monotonic();wave=decode(z);Device['AMD'].synchronize();decode_times.append(time.monotonic()-t);print('DECODE',duration,j,decode_times[-1],flush=True)
pcm=wave.numpy()[0].T.astype(np.float32);np.save(O/'chestnut_pcm.npy',pcm)
r.update(vae_load_seconds=vae_load,decode_seconds=decode_times,total_warm_seconds=elapsed+decode_times[-1],total_warm_rtf=(elapsed+decode_times[-1])/r['duration'],tracked_gpu_bytes=GlobalCounters.mem_used,host_peak_mib=resource.getrusage(resource.RUSAGE_SELF).ru_maxrss/1024,pcm_finite=bool(np.isfinite(pcm).all()),sampler='8-step Euler, DCW disabled',conditioning='actual Mac-prepared inputs; excluded from warm device timing')
(O/'chestnut_result.json').write_text(json.dumps(r,indent=2));print('RESULT',r,flush=True)
assert r['finite']
del fn,x,cond,ctx,tf,rf,y,v,decode,z,wave;gc.collect()
@@ -1,42 +0,0 @@
"""Actual prepared conditions, native ACE DiT; no route or physical audio access."""
import os,time,json,fcntl,resource,gc
from pathlib import Path
import numpy as np
from native_ace import DiT,tensor,time_features
from native_vae import VAE
from tinygrad import TinyJit,Device
from tinygrad.helpers import GlobalCounters
import track_memory
P=Path(__file__).resolve().parent
lock=open('/data/roadscore/generated/gpu.lock','w');fcntl.flock(lock,fcntl.LOCK_EX|fcntl.LOCK_NB)
t=time.monotonic();m=DiT(P/'weights');load=time.monotonic()-t;t=time.monotonic();vae=VAE(P/'vae_weights');vae_load=time.monotonic()-t
for case in os.environ.get('ACE_CASES','verse,prechorus,chorus,bridge,outro,repaint_transition,repaint_outro').split(','):
O=P/case;duration=np.load(O/'hidden_states.npy').shape[1]/25;
assert np.all(np.load(O/'encoder_attention_mask.npy')), 'Masked conditioning needs explicit mask port'
x,cond,ctx=[tensor(np.load(O/(k+'.npy')).astype(np.float16)) for k in ['hidden_states','encoder_hidden_states','context_latents']]
m.prepare_shape(x.shape[1]);sampler=json.loads((O/'sampler.json').read_text()) if (O/'sampler.json').exists() else {};repaint=(O/'sampler_repaint_mask.npy').exists();noise=x.clone().realize()
if repaint:
source=tensor(np.load(O/'sampler_clean_src_latents.npy').astype(np.float16));mask_np=np.load(O/'sampler_repaint_mask.npy').astype(bool);mask=tensor(mask_np[...,None]);soft=mask_np.astype(np.float32)
for row in soft:
ids=np.flatnonzero(row);left,right=int(ids[0]),int(ids[-1])+1;cf=int(sampler.get('repaint_crossfade_frames',10));start=max(0,left-cf);end=min(len(row),right+cf)
if left>start:row[start:left]=np.linspace(0,1,left-start+2)[1:-1]
if end>right:row[right:end]=np.linspace(1,0,end-right+2)[1:-1]
blend=tensor(soft[...,None].astype(np.float16))
tf=tensor(time_features([1]));rf=tensor(time_features([0]));fn=TinyJit(m.forward);times=[]
for i in range(3):
t=time.monotonic();y=fn(x,cond,ctx,tf,rf);Device['AMD'].synchronize();times.append(time.monotonic()-t);print('WARMUP',duration,i,times[-1],flush=True)
schedule=np.linspace(1,0,9).tolist();t=time.monotonic()
for i in range(8):
Device['AMD'].synchronize();tf.assign(tensor(time_features([schedule[i]]))).realize();v=fn(x,cond,ctx,tf,rf);x.assign(x-v*(schedule[i]-schedule[i+1])).realize()
if repaint and i<7 and i<round(float(sampler.get('repaint_injection_ratio',.5))*8):x.assign(mask.where(x,noise*schedule[i+1]+source*(1-schedule[i+1]))).realize()
if repaint:x.assign(blend*x+(1-blend)*source).realize()
Device['AMD'].synchronize();elapsed=time.monotonic()-t;a=x.numpy();np.save(O/'chestnut_latents.npy',a)
r={'case':case,'repaint':repaint,'duration':x.shape[1]/25,'generation_seconds':elapsed,'latent_rtf':elapsed/(x.shape[1]/25),'load_seconds':load,'forward_seconds':times,'tc_opt':os.environ.get('TC_OPT'),'tracked_gpu_bytes':GlobalCounters.mem_used,'host_peak_mib':resource.getrusage(resource.RUSAGE_SELF).ru_maxrss/1024,'finite':bool(np.isfinite(a).all())}
decode=TinyJit(vae.forward);z=tensor(a.transpose(0,2,1));decode_times=[]
for j in range(3):
t=time.monotonic();wave=decode(z);Device['AMD'].synchronize();decode_times.append(time.monotonic()-t);print('DECODE',duration,j,decode_times[-1],flush=True)
pcm=wave.numpy()[0].T.astype(np.float32);np.save(O/'chestnut_pcm.npy',pcm)
r.update(tracked_gpu_peak_bytes=track_memory.peak,vae_load_seconds=vae_load,decode_seconds=decode_times,total_warm_seconds=elapsed+decode_times[-1],total_warm_rtf=(elapsed+decode_times[-1])/r['duration'],tracked_gpu_bytes=GlobalCounters.mem_used,host_peak_mib=resource.getrusage(resource.RUSAGE_SELF).ru_maxrss/1024,pcm_finite=bool(np.isfinite(pcm).all()),sampler='8-step Euler, DCW disabled',conditioning='actual Mac-prepared inputs; excluded from warm device timing')
(O/'chestnut_result.json').write_text(json.dumps(r,indent=2));print('RESULT',r,flush=True)
assert r['finite']
del fn,x,cond,ctx,tf,rf,y,v,decode,z,wave;gc.collect()
@@ -1,42 +0,0 @@
from pathlib import Path
import json,html
P=Path(__file__).resolve().parent;R=P.parents[1];O=R/'results/ace_chestnut_20260916'
def audio(path,label):return f'<figure><figcaption>{html.escape(label)}</figcaption><audio controls preload="none" src="{html.escape(path)}"></audio></figure>'
p=['''<!doctype html><meta charset="utf-8"><meta name="viewport" content="width=device-width"><title>RoadScore · ACE Chestnut</title><style>body{font:17px/1.5 system-ui;background:#10141d;color:#e7edf5;max-width:1050px;margin:auto;padding:36px}h1,h2{line-height:1.2}h2{margin-top:48px;color:#95dbca}a{color:#95c9ff}audio{width:100%}figure{margin:20px 0;padding:18px;background:#1c2533;border-radius:12px}table{border-collapse:collapse;width:100%;font-size:15px}td,th{padding:10px;text-align:left;border-bottom:1px solid #354052}.note{padding:18px;background:#243249;border-radius:12px}nav{display:flex;gap:18px;flex-wrap:wrap}</style><h1>ACE-Step on Chestnut</h1><p>Private engineering and listening review. All files were generated/rendered with physical audio muted. Nothing autoplays.</p><nav>''']
p+=['<a href="#'+s+'">'+s+'</a>' for s in 'ABCDEFG'];p+=['</nav><p class="note">Human preference is established for the original Mac ACE music. The new native port, transitions, outro, and gestures still require listening approval. Prepared conditioning is currently made on the Mac; heavy generation and decoded audio below run on Chestnut. No modeld coexistence approval. The first45s run hit a USB timeout; an explicitly synchronized45s retry completed, but a90s decode later hung. Sustained stability remains unresolved.</p>']
p+=['<h2 id="A">A · Accepted ACE reference</h2>',audio('../composition_20260916/ace/structured90.wav','Known-good 90-second Mac reference'),audio('../composition_20260916/ace/verse_to_chorus.wav','Human-approved verse → chorus')]
p+=['<h2 id="B">B · Native Chestnut music</h2><p>FP16, existing tinygrad tensor-core padding, eight-step Euler; DCW disabled. Times below exclude conditioning preparation and startup. Cold compilation is several minutes.</p><table><tr><th>Audio</th><th>Generation</th><th>Decode</th><th>Total</th><th>RTF</th><th>Tracked GPU</th></tr>']
for d in [15,30,45,60,90]:
q=O/f'reference{d}'/'bounded_result.json'
if not q.exists():q=O/f'reference{d}'/'chestnut_result.json'
if q.exists():
r=json.loads(q.read_text());p.append(f'<tr><td>{d}s</td><td>{r["generation_seconds"]:.2f}s</td><td>{r["decode_seconds"][-1]:.2f}s</td><td>{r["total_warm_seconds"]:.2f}s</td><td>{r["total_warm_rtf"]:.3f}</td><td>{r["tracked_gpu_bytes"]/1e9:.2f} GB</td></tr>')
p.append('</table>')
for d in [15,30,45,60,90]:
if (O/f'reference{d}/chestnut.wav').exists():p.append(audio(f'reference{d}/chestnut.wav',f'{d}s · native Chestnut generation + decode'+(' · decode recovered in a later process; exact RTF unavailable' if d==90 else '')))
p+=['<h2 id="C">C · Precision and decoder comparison</h2><p>Full DiT: relative RMSE0.428% versus official FP32. Native full-length VAE: relative RMSE0.089%. The first optimized FP16-reference maximum-error check failed narrowly; the preserved FP32 audit is linked below. Quantized weights are an unapproved experiment, not the default.</p><p><a href="tc2_precision_audit.json">Precision audit</a> · <a href="reference15/decode_comparison.json">Decoder audit</a></p>',audio('reference15/official_decode.wav','Same Chestnut latents · official Mac decoder')]
p.append('<p>INT8 storage was rejected: slower startup, unchanged resident memory, extra weight error. Evidence is retained without promoting it as a listening candidate.</p>')
p+=['<h2 id="D">D · Section continuity</h2><p>Reference-conditioned independent roles are not proof of continuation. New sequential flow uses the previous section’s actual latent tail as the preserved prefix.</p>']
if (O/'flow/section_flow.wav').exists():p.append(audio('flow/section_flow.wav','Native verse → prechorus → chorus → bridge → outro'))
else:p.append('<p>Sequential native flow is still under test.</p>')
p+=['<h2 id="E">E · Arrival</h2><p>Existing musical context → generated outro → source-derived final cadence. The explanatory render is separate from a road acceptance run.</p>']
if (O/'flow/arrival.wav').exists():p.append(audio('flow/arrival.wav','Native outro and deterministic final cadence'))
else:p.append('<p>Native repaint outro is still under test.</p>')
p+=['<h2 id="F">F · Gestures V2</h2><p>Controlled synthetic timeline: signal2–14s; curve preparation20s / predicted peak26s; navigation32s; lane-change example36s; stop40s; resume44s; outro53s. V2 changes timbre and phrase structure, not merely gain. No guessed notes are introduced.</p>',audio('gestures/dry.wav','Dry ACE source'),audio('gestures/v1.wav','Previous gesture bank'),audio('gestures/v2.wav','V2 · same source and events')]
for kind in ['turn_signal','turn_signal_sustain','turn_signal_off','curve_prepare','curve_apex','navigation_turn','lane_change','stop','resume']:p.append(audio('gestures/v2_'+kind+'.wav',kind.replace('_',' ').title()))
p+=[audio('road_gestures/v1.flac','Full archived road capture · previous gestures'),audio('road_gestures/v2.flac','Same archived dry music and causal states · gestures V2')]
p+=['<h2 id="G">G · Normal replay and fallback</h2><p>ACE is opt-in with <code>--composer ace</code>; the normal default remains SA3. These are fresh runtime tests, separate from the archived gesture remix. All physical outputs were muted. No modeld/real-driving coexistence test was performed.</p>']
for name,label in [('mac_replay_audit.json','Mac fresh ACE'),('native_replay_audit.json','Native comma fresh ACE'),('sa3_replay_audit.json','Mac fresh fallback SA3')]:
q=O/name
if not q.exists():continue
r=json.loads(q.read_text());h=r['host_audio'];rtf=r.get('rtf_per_new_audio',[])
summary=f"{r['audio_seconds']:.1f}s captured; {r['completed_jobs']} new sections; RTF {min(rtf):.3f}–{max(rtf):.3f}" if rtf else 'No fresh generation measured'
p.append(f'<p><strong>{label}: {"instrumented pass" if r["instrumented_pass"] else "not a clean pass"}</strong>. {summary}. Composer fallbacks {r["fallbacks"]}; renderer underflows {r["underflows"]}; host late samples {h.get("late_frames",0)}; host starvation callbacks {h.get("starved_callbacks",0)}; output flags {h.get("portaudio_flags",0)}. <a href="{name}">Full evidence</a>.</p>')
for name,label in [('mac_stored_summary.json','Mac stored fallback score'),('native_stored_summary.json','Native stored ACE score')]:
q=O/name
if q.exists():
r=json.loads(q.read_text());p.append(f'<p>{label}: contiguous samples verified={r["contiguous_samples_verified"]}; output flags={r["portaudio_flags"]}; clock alignment error={r["max_clock_alignment_error_seconds"]*1000:.2f}ms; generation invoked={r["generation_invoked"]}. <a href="{name}">Evidence</a>.</p>')
if (O/'road_demo.json').exists():
r=json.loads((O/'road_demo.json').read_text());p.append(audio(r['audio'],'Actual native ACE route score · gestures and arrival'));p.append(audio(r['audio'].replace('score.flac','arrival_excerpt.wav'),'Actual road arrival excerpt · 178–252s'))
p.append('<p>The Mac fresh ACE run preserved a transport-timing failure. It is not hidden by the successful model throughput or contiguous stored playback. Subjective continuity, ending quality and gesture salience remain for human review.</p><p><a href="../../ACE_CHESTNUT.md">Engineering report and limitations</a> · <a href="../composition_20260916/index.html">Previous road evidence</a> · <a href="regression_route.log">Actual route causality check</a></p>')
(O/'index.html').write_text('\n'.join(p));print(O/'index.html')
@@ -1,28 +0,0 @@
"""Run official local ACE and capture the exact boundary around its heavy DiT."""
import os,sys,time,json,resource
from pathlib import Path
import numpy as np,torch
P=Path(__file__).resolve().parent;R=P.parents[1];BASE=P.parent/'composition_20260916';O=R/'results/ace_chestnut_20260916/reference15';O.mkdir(parents=True,exist_ok=True)
os.environ.update(HF_HOME=str(BASE/'cache/hf'),HF_HUB_DISABLE_TELEMETRY='1',PYTORCH_ENABLE_MPS_FALLBACK='1',TOKENIZERS_PARALLELISM='false')
from acestep.handler import AceStepHandler
from acestep.llm_inference import LLMHandler
from acestep.inference import GenerationParams,GenerationConfig,generate_music
sys.path.insert(0,str(R/'prototype'));from composition_sa3 import BASE as TARGET,ROLES
h=AceStepHandler();lm=LLMHandler();start=time.monotonic()
status,ok=h.initialize_service(str(BASE/'models/ace'),config_path='acestep-v15-turbo',device='mps',use_mlx_dit=False,offload_to_cpu=True);assert ok,status
status,ok=lm.initialize(str(BASE/'models/ace/checkpoints'),'acestep-5Hz-lm-1.7B',backend='mlx',device='mps');assert ok,status
calls=[]
def pre(module,args,kwargs):
i=len(calls);t=time.monotonic();calls.append({'start':t,'timestep':kwargs['timestep'].float().cpu().tolist()})
if i==0:
for key,value in kwargs.items():
if isinstance(value,torch.Tensor):np.save(O/(key+'.npy'),value.detach().float().cpu().numpy())
(O/'inputs.json').write_text(json.dumps({k:{'shape':list(v.shape),'dtype':str(v.dtype)} for k,v in kwargs.items() if isinstance(v,torch.Tensor)},indent=2))
def post(module,args,kwargs,output):
calls[-1]['seconds']=time.monotonic()-calls[-1]['start']
np.save(O/('velocity_'+str(len(calls)-1)+'.npy'),output[0].detach().float().cpu().numpy())
h.model.decoder.register_forward_pre_hook(pre,with_kwargs=True);h.model.decoder.register_forward_hook(post,with_kwargs=True)
params=GenerationParams(caption=TARGET+ROLES['verse_to_chorus'],lyrics='[Instrumental]\n[Verse]\n[Pre-Chorus]\n[Chorus]',instrumental=True,bpm=128,keyscale='D minor',timesignature='4',duration=15,inference_steps=8,seed=12601,thinking=True,use_cot_caption=False,use_cot_metas=False,use_cot_language=False,reference_audio=str(R/'results/composition_20260916/ace/structured90.wav'))
t=time.monotonic();result=generate_music(h,lm,params,GenerationConfig(batch_size=1,allow_lm_batch=False,use_random_seed=False,seeds=[12601],audio_format='wav'),save_dir=str(O))
report={'success':result.success,'error':result.error,'wall_seconds':time.monotonic()-t,'prepare_seconds':t-start,'dit_calls':calls,'audios':result.audios,'extra':result.extra_outputs,'peak_host_mib':resource.getrusage(resource.RUSAGE_SELF).ru_maxrss/2**20}
(O/'result.json').write_text(json.dumps(report,indent=2,default=str));print('RESULT',report['success'],report['error'],flush=True)
@@ -1,38 +0,0 @@
"""Bound decoder workspace with overlapping latent windows; discard context, no crossfade."""
import time
import numpy as np
from tinygrad import TinyJit,Device
from native_ace import tensor
class ChunkDecoder:
def __init__(self,vae,window=375,core=250):
self.vae=vae;self.window=window;self.core=core;self.graphs={}
def decode(self,latent):
n=latent.shape[1];w=min(n,self.window);fn=self.graphs.setdefault(w,TinyJit(self.vae.forward));halo=(w-min(self.core,w))//2;parts=[];start_time=time.monotonic()
for start in range(0,n,min(self.core,w)):
left=max(0,min(start-halo,n-w));end=min(start+self.core,n);z=tensor(latent[:,left:left+w].transpose(0,2,1).astype(np.float16));a=fn(z).numpy()[0].T.astype(np.float32);parts.append(a[(start-left)*1920:(end-left)*1920])
Device['AMD'].synchronize();return np.concatenate(parts),time.monotonic()-start_time
class FencedChunkDecoder(ChunkDecoder):
"""Bounded decoder with explicit completion at each USB/compute boundary.
Optional trace is diagnostic only; it does not change the arithmetic or windows.
This is not a claim that the earlier USB/GPU failure's root cause is repaired.
"""
def __init__(self,vae,window=375,core=250,trace=None):
super().__init__(vae,window,core);self.trace=trace
def decode(self,latent):
n=latent.shape[1];w=min(n,self.window);fn=self.graphs.setdefault(w,TinyJit(self.vae.forward));halo=(w-min(self.core,w))//2;parts=[];started=time.monotonic();dev=Device['AMD']
def mark(stage,start):
if self.trace:self.trace(stage,start,n,getattr(dev,'timeline_value',None))
for start in range(0,n,min(self.core,w)):
left=max(0,min(start-halo,n-w));end=min(start+self.core,n);stage='upload';mark('upload_begin',start)
try:
z=tensor(latent[:,left:left+w].transpose(0,2,1).astype(np.float16));dev.synchronize();mark('upload_done',start);stage='execute'
result=fn(z);mark('compute_submitted',start);dev.synchronize()
mark('execute_done',start);stage='readback'
mark('readback_begin',start);a=result.numpy()[0].T.astype(np.float32);mark('readback_done',start);parts.append(a[(start-left)*1920:(end-left)*1920])
except Exception:
# Read through the existing owner before teardown; never recover silently.
try:print('VAE_FAILURE_LINK',hex(dev.iface.pci_dev.usb.read(0xB450,1)[0]),'stage',stage,'latent_start',start,flush=True)
except Exception as diagnostic:print('VAE_FAILURE_LINK_UNREADABLE',str(diagnostic),flush=True)
raise
dev.synchronize();return np.concatenate(parts),time.monotonic()-started
@@ -1,16 +0,0 @@
import os,json,time,fcntl,resource
from pathlib import Path
import numpy as np
from native_vae import VAE
from chunk_decode import ChunkDecoder
import track_memory
P=Path(__file__).resolve().parent;lock=open('/data/roadscore/generated/gpu.lock','w');fcntl.flock(lock,fcntl.LOCK_EX|fcntl.LOCK_NB);vae=VAE(P/'vae_weights');decoder=ChunkDecoder(vae)
for case in os.environ.get('ACE_CASES','reference30,reference90').split(','):
O=P/case;latent=np.load(O/'chestnut_latents.npy');times=[]
for i in range(2):
pcm,t=decoder.decode(latent);times.append(t);print('CHUNK',case,i,t,flush=True)
np.save(O/'chunk_pcm.npy',pcm);r={'seconds':times,'duration':len(pcm)/48000,'rtf':times[-1]/(len(pcm)/48000),'tracked_gpu_peak_bytes':track_memory.peak,'host_peak_mib':resource.getrusage(resource.RUSAGE_SELF).ru_maxrss/1024,'window_latent_frames':375,'core_latent_frames':250,'finite':bool(np.isfinite(pcm).all())}
if (O/'chestnut_pcm.npy').exists():
ref=np.load(O/'chestnut_pcm.npy');r.update(relative_rmse=float(np.linalg.norm(pcm-ref)/np.linalg.norm(ref)),max_abs_error=float(abs(pcm-ref).max()))
(O/'chunk_result.json').write_text(json.dumps(r,indent=2));print(r,flush=True)
if 'relative_rmse' in r:assert r['relative_rmse']<.01
@@ -1,11 +0,0 @@
from pathlib import Path
import time,json
import torch,numpy as np,soundfile as sf
from diffusers import AutoencoderOobleck
P=Path(__file__).resolve().parent;O=P.parents[1]/'results/ace_chestnut_20260916/reference45'
m=AutoencoderOobleck.from_pretrained(P.parent/'composition_20260916/models/ace/checkpoints/vae').eval().to('mps');x=torch.from_numpy(np.load(O/'chestnut_latents.npy').astype(np.float32)).transpose(1,2).to('mps');t=time.monotonic()
with torch.no_grad():y=m.decode(x).sample
torch.mps.synchronize();elapsed=time.monotonic()-t
a=y[0].cpu().numpy().T;b=np.load(O/"bounded_pcm.npy");e=b-a
r={"official_decode_seconds":elapsed,"relative_rmse":float(np.linalg.norm(e)/np.linalg.norm(a)),"peak_relative_error":float(abs(e).max()/abs(a).max()),"cosine":float(np.vdot(a,b)/np.linalg.norm(a)/np.linalg.norm(b))}
(O/"bounded_decode_comparison.json").write_text(json.dumps(r,indent=2));sf.write(O/"official_decode.wav",a*min(1,.98/abs(a).max()),48000,subtype="PCM_24");print(r)
@@ -1,11 +0,0 @@
from pathlib import Path
import time,json
import torch,numpy as np,soundfile as sf
from diffusers import AutoencoderOobleck
P=Path(__file__).resolve().parent;O=P.parents[1]/'results/ace_chestnut_20260916/reference15'
m=AutoencoderOobleck.from_pretrained(P.parent/'composition_20260916/models/ace/checkpoints/vae').eval().to('mps');x=torch.from_numpy(np.load(O/'chestnut_latents.npy').astype(np.float32)).transpose(1,2).to('mps');t=time.monotonic()
with torch.no_grad():y=m.decode(x).sample
torch.mps.synchronize();elapsed=time.monotonic()-t
a=y[0].cpu().numpy().T;b=np.load(O/"chestnut_pcm.npy");e=b-a
r={"official_decode_seconds":elapsed,"relative_rmse":float(np.linalg.norm(e)/np.linalg.norm(a)),"peak_relative_error":float(abs(e).max()/abs(a).max()),"cosine":float(np.vdot(a,b)/np.linalg.norm(a)/np.linalg.norm(b))}
(O/"decode_comparison.json").write_text(json.dumps(r,indent=2));sf.write(O/"official_decode.wav",a*min(1,.98/abs(a).max()),48000,subtype="PCM_24");print(r)
@@ -1,12 +0,0 @@
from pathlib import Path
import time,json,resource
import numpy as np,torch,soundfile as sf
from diffusers import AutoencoderOobleck
P=Path(__file__).resolve().parent;R=P.parents[1];O=R/'results/ace_chestnut_20260916'
t=time.monotonic();m=AutoencoderOobleck.from_pretrained(P.parent/'composition_20260916/models/ace/checkpoints/vae').eval().to('mps');loaded=time.monotonic()-t
x=torch.from_numpy(np.load(O/'generated_latents.npy').astype(np.float32)).transpose(1,2).to('mps');t=time.monotonic()
with torch.no_grad():y=m.decode(x).sample
torch.mps.synchronize();elapsed=time.monotonic()-t
a=y[0].cpu().numpy().T;gain=min(1.0,.98/float(abs(a).max()));sf.write(O/"chestnut15_host_decode_normalized.wav",a*gain,48000,subtype="PCM_24")
r={"decoder":"official FP32 MPS; main generation Chestnut", "listening_gain":gain,"load_seconds":loaded,"decode_seconds":elapsed,"audio_seconds":len(a)/48000,"peak":float(abs(a).max()),"rms":float(np.sqrt(np.mean(a*a))),"finite":bool(np.isfinite(a).all()),"host_peak_mib":resource.getrusage(resource.RUSAGE_SELF).ru_maxrss/2**20}
(O/"host_decode.json").write_text(json.dumps(r,indent=2));print(r,flush=True)
@@ -1,18 +0,0 @@
"""File-only Chestnut ACE VAE decoding. No audio output API."""
import os,time,json,resource,fcntl,gc
from pathlib import Path
import numpy as np
from tinygrad import TinyJit,Device
from tinygrad.helpers import GlobalCounters
from native_ace import tensor
from native_vae import VAE
P=Path(__file__).resolve().parent;lock=open('/data/roadscore/generated/gpu.lock','w');fcntl.flock(lock,fcntl.LOCK_EX|fcntl.LOCK_NB)
t=time.monotonic();m=VAE(P/'vae_weights');load=time.monotonic()-t
for relative in os.environ.get('ACE_LATENTS','tc2_generated_latents.npy').split(','):
path=P/relative;raw=np.load(path).astype(np.float16).transpose(0,2,1);x=tensor(raw);fn=TinyJit(m.forward);times=[]
for i in range(3):
t=time.monotonic();y=fn(x);Device['AMD'].synchronize();times.append(time.monotonic()-t);print('DECODE',relative,i,times[-1],flush=True)
a=y.numpy()[0].T.astype(np.float32);np.save(path.with_name(path.stem+'_pcm.npy'),a)
r={'input':relative,'load_seconds':load,'decode_seconds':times,'duration':len(a)/48000,'warm_decode_rtf':times[-1]/(len(a)/48000),'tracked_gpu_bytes':GlobalCounters.mem_used,'host_peak_mib':resource.getrusage(resource.RUSAGE_SELF).ru_maxrss/1024,'peak':float(abs(a).max()),'rms':float(np.sqrt(np.mean(a*a))),'finite':bool(np.isfinite(a).all())};path.with_name(path.stem+'_decode.json').write_text(json.dumps(r,indent=2));print(r,flush=True)
assert r['finite']
del fn,x,y;gc.collect()
@@ -1,24 +0,0 @@
"""Export only the DiT and validate a trained block without loading the whole stack."""
from pathlib import Path
import sys,json,time
import numpy as np,torch
from safetensors import safe_open
P=Path(__file__).resolve().parent;M=P.parent/'composition_20260916/models/ace/checkpoints/acestep-v15-turbo';O=P/'weights';O.mkdir(exist_ok=True)
sys.path.insert(0,str(M))
from configuration_acestep_v15 import AceStepConfig
from modeling_acestep_v15_turbo import AceStepDiTLayer
config=AceStepConfig.from_pretrained(M);config._attn_implementation='sdpa'
counts={};block={}
with safe_open(str(M/'model.safetensors'),framework='pt',device='cpu') as f:
for k in f.keys():
shape=f.get_slice(k).get_shape();num=int(np.prod(shape));prefix=k.split('.')[0];counts[prefix]=counts.get(prefix,0)+num
if not k.startswith('decoder.'):continue
w=f.get_tensor(k).to(torch.float16);name=k[len('decoder.'):];np.save(O/(name+'.npy'),w.numpy())
if k.startswith('decoder.layers.0.'):block[k[len('decoder.layers.0.'):]]=w
(P/'tensor_census.json').write_text(json.dumps({'parameters_by_component':counts,'fp16_bytes_by_component':{k:v*2 for k,v in counts.items()}},indent=2));print(counts,flush=True)
torch.manual_seed(421);layer=AceStepDiTLayer(config,0).half().eval();layer.load_state_dict(block)
N=160;E=48;x=torch.randn(1,N,2048).half()*.3;cond=torch.randn(1,E,2048).half()*.2;temb=torch.randn(1,6,2048).half()*.1
freq=torch.arange(N)[:,None].float()*(1/(1000000**(torch.arange(0,128,2).float()/128)))[None,:];freq=torch.cat((freq,freq),-1)[None];cs=(freq.cos().half(),freq.sin().half());idx=torch.arange(N);mask=torch.where((idx[:,None]-idx[None,:]).abs()<=128,0.,-float('inf')).half()[None,None]
with torch.no_grad():y=layer(x,cs,temb,attention_mask=mask,encoder_hidden_states=cond)[0]
for k,v in dict(x=x,cond=cond,temb=temb,cos=cs[0],sin=cs[1],mask=mask,expected=y).items():np.save(P/(k+'.npy'),v.float().numpy() if k=='expected' else v.numpy())
print('REFERENCE_READY',y.shape,flush=True)
@@ -1,13 +0,0 @@
"""Per-output-row symmetric int8 storage; dequantize once into FP16 GPU weights.
This tests transport/startup and quality, not reduced resident matmul precision.
"""
from pathlib import Path
import numpy as np,json
P=Path(__file__).resolve().parent;O=P/'weights_int8';O.mkdir(exist_ok=True);before=after=0;count=0
for path in (P/'weights').glob('*.npy'):
a=np.load(path);before+=a.nbytes
if a.ndim==2 and path.stem.endswith('.weight'):
scale=np.maximum(np.max(abs(a.astype(np.float32)),axis=1,keepdims=True)/127,1e-8).astype(np.float16)
q=np.clip(np.round(a/scale),-127,127).astype(np.int8);np.savez(O/(path.stem+'.npz'),q=q,scale=scale);after+=q.nbytes+scale.nbytes;count+=1
else:np.save(O/path.name,a);after+=a.nbytes
(P/'int8_storage.json').write_text(json.dumps({'scheme':'row symmetric int8, FP16 scale, GPU dequantize once at load; FP16 resident computation','baseline_payload_bytes':before,'quantized_payload_bytes':after,'quantized_matrices':count},indent=2));print(before,after,count,flush=True)
@@ -1,12 +0,0 @@
from pathlib import Path
import json
import numpy as np,torch
from diffusers import AutoencoderOobleck
P=Path(__file__).resolve().parent;M=P.parent/'composition_20260916/models/ace/checkpoints/vae';O=P/'vae_weights';O.mkdir(exist_ok=True)
m=AutoencoderOobleck.from_pretrained(M).eval()
for module in m.decoder.modules():
if hasattr(module,'weight_g'):torch.nn.utils.remove_weight_norm(module)
for k,v in m.decoder.state_dict().items():np.save(O/(k+'.npy'),v.detach().half().numpy())
torch.manual_seed(20);x=torch.randn(1,64,32)*.2
with torch.no_grad():y=m.decoder(x)
np.save(P/'vae_input.npy',x.numpy());np.save(P/'vae_expected.npy',y.numpy());print({'weights':sum(v.numel()*2 for v in m.decoder.parameters()),'input':list(x.shape),'output':list(y.shape)},flush=True)
@@ -1,29 +0,0 @@
"""Sequential fresh native composition; each next section sees only prior generated latents."""
import os,fcntl,json,time,resource
from pathlib import Path
import numpy as np
from ace_runtime import Composer
from native_ace import tensor,time_features
from tinygrad import Device
import track_memory
P=Path(__file__).resolve().parent;O=P/'flow';O.mkdir(exist_ok=True);lock=open('/data/roadscore/generated/gpu.lock','w');fcntl.flock(lock,fcntl.LOCK_EX|fcntl.LOCK_NB)
t=time.monotonic();c=Composer(P);load=time.monotonic()-t;previous=None;reports=[]
# Validate newly-added condition padding/masking against the saved official FP32 boundary.
cond,context,mask,*_=c.prepare('reference15');c.dit.prepare_shape(context.shape[1]);x=tensor(np.load(P/'reference15/hidden_states.npy').astype(np.float16));y=c.dit.forward(x,tensor(cond),tensor(context),tensor(time_features([1])),tensor(time_features([0])),tensor(mask)).numpy().astype(np.float32);ref=np.load(P/'reference15/velocity_0.npy');e=y-ref
check={'relative_rmse':float(np.linalg.norm(e)/np.linalg.norm(ref)),'peak_relative_error':float(abs(e).max()/abs(ref).max())};(O/'padding_validation.json').write_text(json.dumps(check,indent=2));print('PADDED_REFERENCE',check,flush=True);assert check['relative_rmse']<.01 and check['peak_relative_error']<.01
for i,case in enumerate(os.environ.get('ACE_CASES','verse,repaint_prechorus,repaint_chorus,repaint_bridge,repaint_outro').split(',')):
wave,latent,r=c.generate(case,22601+i,previous);np.save(O/(case+'_pcm.npy'),wave);np.save(O/(case+'_latents.npy'),latent);r.update(load_seconds=load,tracked_gpu_peak_bytes=track_memory.peak,peak_host_mib=resource.getrusage(resource.RUSAGE_SELF).ru_maxrss/1024)
if previous is not None:
# Before boundary blend starts, preserved latent prefix must remain exactly equal.
count=round(r['prefix_seconds']*25);r['preserved_prefix_max_error']=float(abs(latent[:,:count-12]-previous[:,-count:-12]).max());assert r['preserved_prefix_max_error']==0
assert np.isfinite(wave).all();reports.append(r);(O/'results.json').write_text(json.dumps(reports,indent=2));print('SECTION',r,flush=True);previous=latent
quant=P/'int8_generated_latents.npy'
if quant.exists():
pcm,elapsed=c.decoder.decode(np.load(quant));np.save(O/'int8_pcm.npy',pcm);(O/'int8_decode.json').write_text(json.dumps({'decode_seconds':elapsed,'duration':len(pcm)/48000,'decoder':'native bounded Chestnut; shared warm decoder'},indent=2))
for label in ['aligned60','reference90']:
path=P/'aligned60_generated_latents.npy' if label=='aligned60' else P/'reference90/chestnut_latents.npy'
if path.exists():
pcm,elapsed=c.decoder.decode(np.load(path));np.save(O/(label+'_pcm.npy'),pcm);(O/(label+'_decode.json')).write_text(json.dumps({'decode_seconds':elapsed,'duration':len(pcm)/48000,'decoder':'native bounded Chestnut; shared warm decoder'},indent=2));print('ADDITIONAL_DECODE',label,elapsed,flush=True)
@@ -1,20 +0,0 @@
import os,time,json,fcntl,resource
from pathlib import Path
import numpy as np
from native_ace import DiT,tensor,time_features
from tinygrad import TinyJit,Device
from tinygrad.helpers import GlobalCounters
P=Path(__file__).resolve().parent;O=P/'reference15';prefix=os.environ.get('ACE_RESULT_PREFIX','');lock=open('/data/roadscore/generated/gpu.lock','w');fcntl.flock(lock,fcntl.LOCK_EX|fcntl.LOCK_NB)
t=time.monotonic();m=DiT(P/'weights');loaded=time.monotonic()-t;print('WEIGHTS_READY',loaded,GlobalCounters.mem_used,flush=True)
x,cond,ctx=[tensor(np.load(O/(k+'.npy')).astype(np.float16)) for k in ['hidden_states','encoder_hidden_states','context_latents']];m.prepare_shape(x.shape[1]);tf=tensor(time_features([1]));rf=tensor(time_features([0]));fn=TinyJit(m.forward);times=[]
for i in range(4):
t=time.monotonic();out=fn(x,cond,ctx,tf,rf);Device['AMD'].synchronize();times.append(time.monotonic()-t);print('FULL',i,times[-1],GlobalCounters.mem_used,flush=True)
y=out.numpy().astype(np.float32);np.save(P/(prefix+'full_actual.npy'),y);reference_file=os.environ.get('ACE_REFERENCE','velocity_fp16.npy');ref=np.load(O/reference_file);error=y-ref
report={'reference_file':reference_file,'tc_opt':os.environ.get('TC_OPT','0'),'sampler':'8-step Euler; DCW disabled','load_seconds':loaded,'full_forward_seconds':times,'relative_rmse':float(np.linalg.norm(error)/np.linalg.norm(ref)),'peak_relative_error':float(abs(error).max()/abs(ref).max()),'max_abs_error':float(abs(error).max()),'finite':bool(np.isfinite(y).all()),'peak_host_mib':resource.getrusage(resource.RUSAGE_SELF).ru_maxrss/1024,'tracked_gpu_bytes':GlobalCounters.mem_used,'weights_bytes':sum(w.numel()*w.dtype.itemsize for w in m.w.values()),'audio_seconds':x.shape[1]/25,'projection_8steps_rtf':8*times[-1]/(x.shape[1]/25)}
(P/(prefix+'full_result.json')).write_text(json.dumps(report,indent=2));print(report,flush=True)
assert report['finite'] and report['relative_rmse']<.01 and report['peak_relative_error']<.01
# Eight trained-weight flow steps, identical initial noise and cached conditioning.
schedule=np.linspace(1,0,9,dtype=np.float32).tolist();start=time.monotonic()
for i in range(8):
tf.assign(tensor(time_features([schedule[i]]))).realize();v=fn(x,cond,ctx,tf,rf);x.assign(x-v*(schedule[i]-schedule[i+1])).realize()
Device['AMD'].synchronize();elapsed=time.monotonic()-start;np.save(P/(prefix+'generated_latents.npy'),x.numpy());report.update(generation_seconds=elapsed,generation_rtf=elapsed/(x.shape[1]/25));(P/(prefix+'full_result.json')).write_text(json.dumps(report,indent=2));print('GENERATION',elapsed,flush=True)
@@ -1,15 +0,0 @@
from pathlib import Path
import sys,time,json
import numpy as np,torch
P=Path(__file__).resolve().parent;R=P.parents[1];O=R/'results/ace_chestnut_20260916/reference15';M=P.parent/'composition_20260916/models/ace/checkpoints/acestep-v15-turbo';sys.path.insert(0,str(M))
from configuration_acestep_v15 import AceStepConfig
from modeling_acestep_v15_turbo import AceStepDiTModel
c=AceStepConfig.from_pretrained(M);c._attn_implementation='sdpa'
with torch.device('meta'):m=AceStepDiTModel(c)
w={p.stem:torch.from_numpy(np.load(p)) for p in (P/'weights').glob('*.npy')};m.load_state_dict(w,assign=True);del w
from transformers.models.qwen3.modeling_qwen3 import Qwen3RotaryEmbedding
m.rotary_emb=Qwen3RotaryEmbedding(c);m=m.to('mps').eval()
a={k:torch.tensor(np.load(O/(k+'.npy')),device='mps',dtype=torch.float16) for k in ['hidden_states','timestep','timestep_r','attention_mask','encoder_hidden_states','encoder_attention_mask','context_latents']}
with torch.no_grad():
t=time.monotonic();y=m(**a,use_cache=False)[0];torch.mps.synchronize();elapsed=time.monotonic()-t
np.save(O/'velocity_fp16.npy',y.float().cpu().numpy());print({'seconds':elapsed,'shape':list(y.shape),'peak':y.abs().max().item()},flush=True)
@@ -1,214 +0,0 @@
"""Lazy official Mac semantic planner adapter for hook_planning.PlanCache.
No audio diffusion. Not deployed/validated on native until a combined plan is
approved. Continuation hint+prefix composition is an explicit candidate algorithm.
"""
import json
import hashlib
import os
from pathlib import Path
import random
import sys
import threading
import numpy as np
class Captured(BaseException):
pass
def finish_capture(completed, result, request, sources, output):
from hook_planning import validate_prepared
if not completed:
raise RuntimeError('Semantic preparation did not reach the completed capture boundary')
if result is not None and (result.success or str(result.error) != "'outputs'"):
raise RuntimeError(f'Unexpected result after capture: {result.success}: {result.error}')
validate_prepared(request, output, sources)
def continuation_inputs(context, prefix):
"""Keep planned future hints; overwrite only exact already-committed8s prefix."""
if context.ndim != 3 or context.shape[0] != 1 or context.shape[2] != 128:
raise ValueError('Expected native context shape')
if prefix.shape != (1, 200, 64) or not np.isfinite(prefix).all() or context.shape[1] <= 200:
raise ValueError('Expected exactly8s committed latent prefix, never whole previous song')
result = context.copy()
mask = np.arange(context.shape[1])[None, :] >= 200
source = np.zeros((1, context.shape[1], 64), dtype=context.dtype)
source[:, :200] = prefix
result[:, :200, :64] = prefix
result[:, :, 64:] = mask[:, :, None]
return result, source, mask
class PreparationOnlyDecoder:
"""Dispatch marker, never a model: accidental diffusion must fail closed."""
def __call__(self, *args, **kwargs):
raise RuntimeError('Host diffusion disabled in preparation-only service')
def release_preparation_decoder(handler):
# Official code selects the capture boundary only when mlx_decoder is not None.
# Require real initialization first; do not fake a successful model conversion.
if not handler.use_mlx_dit or handler.mlx_decoder is None or handler.model.decoder is not None:
raise RuntimeError('Release requires verified MLX conversion and disabled Torch decoder')
handler.mlx_decoder = PreparationOnlyDecoder()
class HostHookAdapter:
"""One serialized host model instance; initialize only on actual cache miss."""
def __init__(self, assets_root, *, preparation_only=False):
self.preparation_only = preparation_only
self.base = Path(assets_root).resolve() / 'experiments/composition_20260916'
self.handler = self.lm = None
self.lock = threading.Lock()
self._fingerprints = None
def fingerprints(self):
"""Call once before constructing requests; does not load models or generate."""
if self._fingerprints is None:
def tree_hash(root):
paths = sorted(p for p in root.rglob('*') if p.is_file() and '__pycache__' not in p.parts)
if not paths:
raise FileNotFoundError(f'No identity files at {root}')
state = hashlib.sha256()
for path in paths:
with path.open('rb') as stream:
value = hashlib.file_digest(stream, 'sha256').hexdigest()
state.update(str(path.relative_to(root)).encode() + b'\0' + value.encode() + b'\n')
return state.hexdigest()
model = tree_hash(self.base / 'models/ace/checkpoints')
official = tree_hash(self.base / 'vendor/ACE-Step-1.5/acestep')
local = hashlib.sha256(Path(__file__).read_bytes()).hexdigest()
preparation = hashlib.sha256((official + local + ('preparation-only-v1' if self.preparation_only else 'full-host-v1')).encode()).hexdigest()
self._fingerprints = {'model_fingerprint': model, 'preparation_fingerprint': preparation}
return dict(self._fingerprints)
def _load(self):
if self.handler is not None:
return
if sys.platform != 'darwin':
raise RuntimeError('Official host semantic preparation currently requires Mac')
if not (self.base / 'models/ace/checkpoints').is_dir():
raise FileNotFoundError('Existing local model assets required; no downloads')
os.environ.update(HF_HOME=str(self.base / 'cache/hf'), HF_HUB_OFFLINE='1',
HF_HUB_DISABLE_TELEMETRY='1', TOKENIZERS_PARALLELISM='false')
import acestep
if Path(acestep.__file__).resolve().parent != (self.base / 'vendor/ACE-Step-1.5/acestep').resolve():
raise RuntimeError('Imported ACE package differs from fingerprinted preparation source')
from acestep.handler import AceStepHandler
from acestep.llm_inference import LLMHandler
handler, lm = AceStepHandler(), LLMHandler()
status, ok = lm.initialize(str(self.base / 'models/ace/checkpoints'), 'acestep-5Hz-lm-1.7B', backend='mlx', device='mps')
if not ok:
raise RuntimeError(status)
original_init = handler._init_mlx_dit
def initialize_without_duplicate(*args, **kwargs):
ok = original_init(*args, **kwargs)
if not ok or handler.mlx_decoder is None:
raise RuntimeError('MLX conversion failed; refusing Torch diffusion fallback')
import gc
import torch
handler.model.decoder = None
def no_torch_diffusion(*args, **kwargs):
raise RuntimeError('Torch diffusion disabled after verified MLX conversion')
handler.model.generate_audio = no_torch_diffusion
gc.collect()
torch.mps.empty_cache()
return ok
handler._init_mlx_dit = initialize_without_duplicate
status, ok = handler.initialize_service(str(self.base / 'models/ace'), config_path='acestep-v15-turbo',
device='mps', use_mlx_dit=True, offload_to_cpu=True, offload_dit_to_cpu=True)
if not ok or not handler.use_mlx_dit or handler.mlx_decoder is None:
raise RuntimeError(f'No safe MLX preparation boundary: {status}')
if self.preparation_only:
release_preparation_decoder(handler)
import gc
import mlx.core as mx
import torch
gc.collect()
mx.clear_cache()
torch.mps.empty_cache()
self.handler, self.lm = handler, lm
def __call__(self, request, sources, output):
with self.lock:
identities = self.fingerprints()
if (request.model_fingerprint != identities['model_fingerprint'] or
request.preparation_fingerprint != identities['preparation_fingerprint']):
raise ValueError('Request fingerprint does not match this host preparation implementation/models')
self._load()
import torch
import mlx.core as mx
from acestep.inference import GenerationParams, GenerationConfig, generate_music
random.seed(request.semantic_seed)
np.random.seed(request.semantic_seed)
torch.manual_seed(request.semantic_seed)
mx.random.seed(request.semantic_seed)
output = Path(output)
original_diffusion = self.handler._mlx_run_diffusion
original_plan = self.lm.generate_with_stop_condition
captured_plan = {}
completed = threading.Event()
def plan(*args, **kwargs):
result = original_plan(*args, **kwargs)
if not result.get('success') or not result.get('audio_codes'):
raise RuntimeError('Planner failed; no generic/static conditioning fallback')
captured_plan.update(semantic_seed=request.semantic_seed, audio_codes=result['audio_codes'],
actual_seeds=kwargs.get('seeds'), caption=kwargs.get('caption'), lyrics=kwargs.get('lyrics'),
time_costs=result.get('extra_outputs', {}).get('time_costs'))
if captured_plan['actual_seeds'] != [request.semantic_seed]:
raise ValueError('Host planner ignored requested reproducibility seed')
return result
def capture(*unused, **kwargs):
if not captured_plan:
raise ValueError('No actual semantic plan was generated')
if kwargs.get('repaint_mask') is not None:
raise ValueError('Unexpected upstream repaint; refuse ambiguous semantic prefix')
arrays = {name: kwargs[name].detach().float().cpu().numpy()
for name in ('encoder_hidden_states', 'encoder_attention_mask', 'context_latents')}
if arrays['context_latents'].shape != (1, request.window_seconds * 25, 128):
raise ValueError('Unexpected planned duration')
if request.prefix_seconds:
prefix = np.load(sources['committed_prefix'], allow_pickle=False)
context, source, mask = continuation_inputs(arrays['context_latents'], prefix)
arrays.update(context_latents=context, sampler_clean_src_latents=source, sampler_repaint_mask=mask)
(output / 'sampler.json').write_text(json.dumps({
'repaint_crossfade_frames': 12, 'repaint_injection_ratio': .5,
'semantic_prefix_policy': 'generated plan hints with exact committed8s prefix; candidate unvalidated native'}))
for name, array in arrays.items():
np.save(output / (name + '.npy'), array)
(output / 'semantic_plan.json').write_text(json.dumps(captured_plan, indent=2))
(output / 'prepared.json').write_text(json.dumps({
'request_key': request.cache_key, 'semantic_seed': request.semantic_seed,
'semantic_plan_present': True, 'audio_diffusion_called': False,
'host_preparation_only': self.preparation_only,
'conditioning_strategy': 'full semantic planning per bounded window; actual hook audio reference; committed latent prefix',
'native_validated': False, 'request': request.identity()}, indent=2))
from hook_planning import validate_prepared
validate_prepared(request, output, sources)
completed.set()
raise Captured()
self.handler._mlx_run_diffusion = capture
self.lm.generate_with_stop_condition = plan
try:
params = GenerationParams(caption=request.caption, lyrics=request.lyrics,
instrumental=True, bpm=request.bpm, keyscale=request.keyscale, timesignature='4',
duration=request.window_seconds, inference_steps=8, seed=request.semantic_seed,
thinking=True, dcw_enabled=False, use_cot_caption=False, use_cot_metas=False,
use_cot_language=False, reference_audio=str(sources['hook_reference']) if sources else None)
try:
result = generate_music(self.handler, self.lm, params,
GenerationConfig(batch_size=1, allow_lm_batch=False, use_random_seed=False,
seeds=[request.semantic_seed], audio_format='wav'), save_dir=None)
except Captured:
finish_capture(completed.is_set(), None, request, sources, output)
return
finish_capture(completed.is_set(), result, request, sources, output)
finally:
self.handler._mlx_run_diffusion = original_diffusion
self.lm.generate_with_stop_condition = original_plan
@@ -1,22 +0,0 @@
import os,time,json,fcntl,resource
from pathlib import Path
import numpy as np
from native_ace import DiT,tensor,time_features
from tinygrad import TinyJit,Device
from tinygrad.helpers import GlobalCounters
P=Path(__file__).resolve().parent;O=P/'reference15';prefix=os.environ.get('ACE_RESULT_PREFIX','');lock=open('/data/roadscore/generated/gpu.lock','w');fcntl.flock(lock,fcntl.LOCK_EX|fcntl.LOCK_NB)
t=time.monotonic();m=DiT(P/'weights_int8');loaded=time.monotonic()-t;print('WEIGHTS_READY',loaded,GlobalCounters.mem_used,flush=True)
x,cond,ctx=[tensor(np.load(O/(k+'.npy')).astype(np.float16)) for k in ['hidden_states','encoder_hidden_states','context_latents']];m.prepare_shape(x.shape[1]);tf=tensor(time_features([1]));rf=tensor(time_features([0]));fn=TinyJit(m.forward);times=[]
for i in range(4):
t=time.monotonic();out=fn(x,cond,ctx,tf,rf);Device['AMD'].synchronize();times.append(time.monotonic()-t);print('FULL',i,times[-1],GlobalCounters.mem_used,flush=True)
y=out.numpy().astype(np.float32);np.save(P/(prefix+'full_actual.npy'),y);reference_file=os.environ.get('ACE_REFERENCE','velocity_fp16.npy');ref=np.load(O/reference_file);error=y-ref
report={'reference_file':reference_file,'tc_opt':os.environ.get('TC_OPT','0'),'sampler':'8-step Euler; DCW disabled','load_seconds':loaded,'full_forward_seconds':times,'relative_rmse':float(np.linalg.norm(error)/np.linalg.norm(ref)),'peak_relative_error':float(abs(error).max()/abs(ref).max()),'max_abs_error':float(abs(error).max()),'finite':bool(np.isfinite(y).all()),'peak_host_mib':resource.getrusage(resource.RUSAGE_SELF).ru_maxrss/1024,'tracked_gpu_bytes':GlobalCounters.mem_used,'weights_bytes':sum(w.numel()*w.dtype.itemsize for w in m.w.values()),'audio_seconds':x.shape[1]/25,'projection_8steps_rtf':8*times[-1]/(x.shape[1]/25)}
(P/(prefix+'full_result.json')).write_text(json.dumps(report,indent=2));print(report,flush=True)
# Exploratory storage quantization; loosened quality rejection guard is NOT equivalence.
original=np.load(O/'velocity_0.npy');report['original_fp32_weight_drift_rmse']=float(np.linalg.norm(y-original)/np.linalg.norm(original));report['quality_status']='unapproved quantization candidate; implementation comparison uses quantized official reference';report['equivalence_1pct']=report['relative_rmse']<.01 and report['peak_relative_error']<.01
assert report['finite'] and report['relative_rmse']<.01 and report['peak_relative_error']<.015
# Eight trained-weight flow steps, identical initial noise and cached conditioning.
schedule=np.linspace(1,0,9,dtype=np.float32).tolist();start=time.monotonic()
for i in range(8):
Device['AMD'].synchronize();tf.assign(tensor(time_features([schedule[i]]))).realize();v=fn(x,cond,ctx,tf,rf);x.assign(x-v*(schedule[i]-schedule[i+1])).realize()
Device['AMD'].synchronize();elapsed=time.monotonic()-start;np.save(P/(prefix+'generated_latents.npy'),x.numpy());report.update(generation_seconds=elapsed,generation_rtf=elapsed/(x.shape[1]/25));(P/(prefix+'full_result.json')).write_text(json.dumps(report,indent=2));print('GENERATION',elapsed,flush=True)
@@ -1,22 +0,0 @@
from pathlib import Path
import sys,time,json
import numpy as np,torch
P=Path(__file__).resolve().parent;R=P.parents[1];O=R/'results/ace_chestnut_20260916/reference15';M=P.parent/'composition_20260916/models/ace/checkpoints/acestep-v15-turbo';sys.path.insert(0,str(M))
from configuration_acestep_v15 import AceStepConfig
from modeling_acestep_v15_turbo import AceStepDiTModel
c=AceStepConfig.from_pretrained(M);c._attn_implementation='sdpa'
with torch.device('meta'):m=AceStepDiTModel(c)
w={}
for p in (P/'weights_int8').iterdir():
if p.suffix=='.npy':a=np.load(p)
elif p.suffix=='.npz':
with np.load(p) as packed:a=(packed['q'].astype(np.float16)*packed['scale']).astype(np.float16)
else:continue
w[p.stem]=torch.from_numpy(a)
m.load_state_dict(w,assign=True);del w
from transformers.models.qwen3.modeling_qwen3 import Qwen3RotaryEmbedding
m.rotary_emb=Qwen3RotaryEmbedding(c);m=m.to('mps').eval()
a={k:torch.tensor(np.load(O/(k+'.npy')),device='mps',dtype=torch.float16) for k in ['hidden_states','timestep','timestep_r','attention_mask','encoder_hidden_states','encoder_attention_mask','context_latents']}
with torch.no_grad():
t=time.monotonic();y=m(**a,use_cache=False)[0];torch.mps.synchronize();elapsed=time.monotonic()-t
np.save(O/'velocity_int8_reference.npy',y.float().cpu().numpy());print({'seconds':elapsed,'shape':list(y.shape),'peak':y.abs().max().item()},flush=True)
@@ -1,6 +0,0 @@
{
"scheme": "row symmetric int8, FP16 scale, GPU dequantize once at load; FP16 resident computation",
"baseline_payload_bytes": 3150917760,
"quantized_payload_bytes": 1578352768,
"quantized_matrices": 271
}
@@ -1,20 +0,0 @@
"""Post-run evidence only; never an input to the composer or replay."""
import json,sys,collections
from pathlib import Path
def read(p,default=None):return json.loads(p.read_text()) if p.exists() else default
def lines(p):return [json.loads(x) for x in p.read_text().splitlines()] if p.exists() else []
def audit(p):
s=read(p/'summary.json',{});h=read(p/'host_audio_summary.json',{});launch=read(p/'launch.json',{});jobs=s.get('generation',[]);trace=lines(p/'trace.jsonl');requests=lines(p/'jobs.jsonl');ui=lines(p/'ui_audit.jsonl');blocks=[x for x in lines(p/'host_audio.jsonl') if 'audio_s' in x];g=read(p/'gestures.json',{})
result={'run':str(p),'launch':launch,'audio_seconds':s.get('audio_seconds'),'completed_jobs':len(jobs),'generation_seconds':[j.get('seconds') for j in jobs],'rtf_per_new_audio':[j['seconds']/j['new_seconds'] for j in jobs if 'seconds' in j and j.get('new_seconds')],'generation_errors':[j for j in jobs if j.get('error')],'minimum_buffer_seconds':min([x['buffered'] for x in trace] or [0]),'fallbacks':s.get('fallbacks'),'underflows':s.get('underflows'),'worker_failed':any(x.get('worker_failed') for x in trace),'host_audio':h,'all_blocks_muted':bool(blocks) and all(x.get('muted') and x.get('presentation_muted',True) for x in blocks),'input_time_violations':sum(any(v>r['cutoff_ns'] for v in r['input_times'].values()) for r in requests),'request_roles':dict(collections.Counter(r['conditioning'] for r in requests)),'navigation_present':any(x.get('nav',{}).get('valid') for x in trace),'curve_activations':len({x['activation'] for x in trace if x.get('kind')=='curve' and x.get('activation') is not None}),'ui_last':ui[-1] if ui else None,'ending':read(p/'ending.json'),'composition':read(p/'composition.json'),'gestures':g,'runtime_manifest':read(p/'runtime_manifest.json')}
result['quality_rejected_jobs']=sum(bool(j.get('quality_rejected')) for j in jobs)
result['quality_rejections']=sum(j.get('quality_rejections',0) for j in jobs)
result['rerolls']=sum(j.get('rerolls',0) for j in jobs)
result['accepted_jobs']=sum(j.get('quality_accepted',not j.get('error')) for j in jobs)
result['safe_extensions']=s.get('safe_extensions',[])
result['instrumented_pass']=bool(jobs) and not (result['generation_errors'] or result['fallbacks'] or result['underflows'] or result['worker_failed'] or result['input_time_violations'] or any(h.get(k,0) for k in ['late_frames','starved_callbacks','portaudio_flags'])) and result['all_blocks_muted']
return result
if __name__=='__main__':
p=Path(sys.argv[1]);r=audit(p);out=Path(sys.argv[2]);out.write_text(json.dumps(r,indent=2));print(json.dumps({k:v for k,v in r.items() if k not in ['composition','gestures','runtime_manifest']},indent=2))
@@ -1,31 +0,0 @@
"""Read-only link diagnostics through the current GPU owner. Never resets hardware."""
import json,time,os
from pathlib import Path
class LinkUnhealthy(RuntimeError):pass
class LinkProbe:
def __init__(self,device,path):
self.session=f'{os.getpid()}_{time.time_ns()}';self.dev=device;self.path=Path(path);self.path.parent.mkdir(parents=True,exist_ok=True);self.last=None
def sample(self,stage,**extra):
dev=self.dev;row={'session':self.session,'owner_pid':os.getpid(),'stage':stage,'monotonic':time.monotonic(),'wall':time.time(),'timeline_target':getattr(dev,'timeline_value',0)-1,'architecture':getattr(dev,'arch',None),'vram_capacity_bytes':getattr(dev.iface.dev_impl,'vram_size',None),**extra}
controller=dev.iface.pci_dev.usb
for name,read in [('link_register',lambda:hex(controller.read(0xB450,1)[0])),('gpu_config',lambda:hex(controller.pcie_cfg_req(0,bus=4))),('bridge_config',lambda:hex(controller.pcie_cfg_req(0,bus=1))),('timeline_observed',lambda:int(dev.timeline_signal.value))]:
try:row[name]=read()
except Exception as e:row[name+'_error']=str(e)
try:row['allocator_bytes']=sum(size for size,_,_,free in dev.iface.dev_impl.mm.pa_allocator.blocks.values() if not free)
except Exception as e:row['allocator_error']=str(e)
row['healthy']=row.get('link_register')=='0x78' and int(row.get('gpu_config','0'),16)&0xffff==0x1002
with self.path.open('a') as f:f.write(json.dumps(row)+'\n')
self.last=row;return row
def preflight(self,**extra):
row=self.sample('before_generation_job',**extra)
if not row['healthy']:raise LinkUnhealthy('Chestnut link preflight failed; no generation submitted; explicit offroad recovery required')
return row
def install_failure_hook(self,contain=False):
original=self.dev.on_device_hang
def before_driver_handler():
self.sample('before_driver_timeout_handler')
if contain:raise LinkUnhealthy('GPU timeout captured; driver interrupt-reset handler skipped; explicit offroad recovery required')
try:return original()
finally:self.sample('after_driver_timeout_handler')
self.dev.on_device_hang=before_driver_handler
def trace(self,stage,start,n,timeline):self.sample(stage,latent_start=start,latent_count=n,submission_timeline=timeline)
@@ -1,65 +0,0 @@
"""Narrow ACE DiT port. Equations follow official ACE-Step Apache-2.0 implementation."""
from pathlib import Path
import numpy as np
from tinygrad import Tensor,TinyJit,dtypes
def tensor(x):return Tensor(np.asarray(x),device='AMD').realize()
class DiT:
def __init__(self,path,layers=24):
self.w={};self.layers=layers
for p in sorted(list(Path(path).glob('*.npy'))+list(Path(path).glob('*.npz'))):
if p.stem.startswith('layers.') and int(p.stem.split('.')[1])>=layers:continue
if p.suffix=='.npz':
with np.load(p) as packed:self.w[p.stem]=(tensor(packed['q']).cast(dtypes.float16)*tensor(packed['scale'])).realize()
else:self.w[p.stem]=tensor(np.load(p))
def linear(self,x,p):
y=x@self.w[p+'.weight'].T
return y+self.w[p+'.bias'] if p+'.bias' in self.w else y
def rms(self,x,p):
z=x.float();return (z*(z.square().mean(-1,keepdim=True)+1e-6).rsqrt()).cast(x.dtype)*self.w[p+'.weight']
def attention(self,x,p,cond=None,cs=None,mask=None):
b,n,_=x.shape;source=x if cond is None else cond
def heads(z,h):return z.reshape(b,-1,h,128).transpose(1,2)
q=self.rms(heads(self.linear(x,p+'.q_proj'),16),p+'.q_norm');k=self.rms(heads(self.linear(source,p+'.k_proj'),8),p+'.k_norm');v=heads(self.linear(source,p+'.v_proj'),8)
if cs is not None:
c,s=cs
def rope(z):return z*c+(-z[...,64:]).cat(z[...,:64],dim=-1)*s
q,k=rope(q),rope(k)
k=k.unsqueeze(2).expand(b,8,2,k.shape[2],128).reshape(b,16,-1,128);v=v.unsqueeze(2).expand(b,8,2,v.shape[2],128).reshape(b,16,-1,128)
scores=(q*(128**-.5))@k.transpose(-1,-2)
if mask is not None:scores=scores+mask
z=(scores.float().softmax(-1).cast(v.dtype)@v).transpose(1,2).reshape(b,n,2048)
return self.linear(z,p+'.o_proj')
def block(self,x,cond,temb,cs,mask,i=0,cond_mask=None):
p=f'layers.{i}';shift,scale,gate,shift_ff,scale_ff,gate_ff=(self.w[p+'.scale_shift_table']+temb).chunk(6,dim=1)
z=(self.rms(x,p+'.self_attn_norm')*(1+scale)+shift).cast(x.dtype)
x=(x+self.attention(z,p+'.self_attn',cs=cs,mask=mask)*gate).cast(x.dtype)
x=x+self.attention(self.rms(x,p+'.cross_attn_norm'),p+'.cross_attn',cond=cond,mask=cond_mask)
z=(self.rms(x,p+'.mlp_norm')*(1+scale_ff)+shift_ff).cast(x.dtype)
ff=self.linear(self.linear(z,p+'.mlp.gate_proj').silu()*self.linear(z,p+'.mlp.up_proj'),p+'.mlp.down_proj')
return (x+ff*gate_ff).cast(x.dtype).realize()
def time_embedding(self,tf,p):
t=self.linear(self.linear(tf,p+'.linear_1').silu(),p+'.linear_2')
return t,self.linear(t.silu(),p+'.time_proj').reshape(1,6,2048)
def prepare_shape(self,n,align_tokens=0):
self.original=n;valid=(n+1)//2;N=((valid+align_tokens-1)//align_tokens)*align_tokens if align_tokens else valid;self.padded_latents=2*N
f=np.arange(N,dtype=np.float32)[:,None]/(1000000**(np.arange(0,128,2,dtype=np.float32)/128))[None,:];f=np.concatenate((f,f),-1)[None,None]
self.cs=(tensor(np.cos(f).astype(np.float16)),tensor(np.sin(f).astype(np.float16)))
i=np.arange(N);self.mask=tensor(np.where((abs(i[:,None]-i[None,:])<=128)&(i[None,:]<valid),0,-np.inf).astype(np.float16)[None,None]);self.fullmask=tensor(np.where(i<valid,0,-np.inf).astype(np.float16)[None,None,None,:]) if N!=valid else None
def forward(self,x,cond,context,tf,rf,cond_mask=None):
b,n,_=x.shape
t,p=self.time_embedding(tf,'time_embed');tr,pr=self.time_embedding(rf,'time_embed_r');t=t+tr;p=p+pr
x=context.cat(x,dim=-1)
if self.padded_latents>n:x=x.pad(((0,0),(0,self.padded_latents-n),(0,0)))
N=x.shape[1]//2
x=x.reshape(b,N,2,192).permute(0,1,3,2).reshape(b,N,384)@self.w['proj_in.1.weight'].reshape(2048,384).T+self.w['proj_in.1.bias']
cond=self.linear(cond,'condition_embedder')
for i in range(self.layers):x=self.block(x,cond,p,self.cs,self.mask if i%2==0 else self.fullmask,i,cond_mask)
shift,scale=(self.w['scale_shift_table']+t.unsqueeze(1)).chunk(2,dim=1)
x=(self.rms(x,'norm_out')*(1+scale)+shift).cast(x.dtype)
x=x@self.w['proj_out.1.weight'].reshape(2048,128)
return (x.reshape(b,N,64,2).permute(0,1,3,2).reshape(b,N*2,64)+self.w['proj_out.1.bias'])[:,:n].realize()
def time_features(t):
f=np.exp(-np.log(10000)*np.arange(128,dtype=np.float32)/128);v=np.asarray(t,dtype=np.float32).reshape(-1,1)*1000*f[None,:]
return np.concatenate((np.cos(v),np.sin(v)),-1).astype(np.float16)
@@ -1,20 +0,0 @@
"""ACE Oobleck decoder, folded weight normalization and explicit precision boundary."""
from pathlib import Path
import numpy as np
from tinygrad import TinyJit
from native_ace import tensor
class VAE:
def __init__(self,path):self.w={p.stem:tensor(np.load(p)) for p in sorted(Path(path).glob('*.npy'))}
def snake(self,x,p):
z=x.float();a=self.w[p+'.alpha'].float().exp();b=self.w[p+'.beta'].float().exp()
return (z+(a*z).sin().square()/(b+1e-9)).cast(x.dtype)
def conv(self,x,p,padding=0,dilation=1):return x.conv2d(self.w[p+'.weight'],self.w.get(p+'.bias'),padding=padding,dilation=dilation).realize()
def residual(self,x,p,d):
y=self.conv(self.snake(x,p+'.snake1'),p+'.conv1',3*d,d)
return (x+self.conv(self.snake(y,p+'.snake2'),p+'.conv2')).realize()
def forward(self,x):
x=self.conv(x,'conv1',3)
for i,stride in enumerate([10,6,4,4,2]):
p=f'block.{i}';x=self.snake(x,p+'.snake1').conv_transpose2d(self.w[p+'.conv_t1.weight'],self.w[p+'.conv_t1.bias'],stride=stride,padding=(stride+1)//2).realize()
for j,d in enumerate([1,3,9]):x=self.residual(x,p+f'.res_unit{j+1}',d)
return self.conv(self.snake(x,'snake1'),'conv2',3)
@@ -1,51 +0,0 @@
"""Run official local ACE and capture the exact boundary around its heavy DiT."""
import os,sys,time,json,resource
from pathlib import Path
import numpy as np,torch
P=Path(__file__).resolve().parent;R=P.parents[1];BASE=P.parent/'composition_20260916';O=R/'results/ace_chestnut_20260916/reference15';O.mkdir(parents=True,exist_ok=True)
os.environ.update(HF_HOME=str(BASE/'cache/hf'),HF_HUB_DISABLE_TELEMETRY='1',PYTORCH_ENABLE_MPS_FALLBACK='1',TOKENIZERS_PARALLELISM='false')
from acestep.handler import AceStepHandler
from acestep.llm_inference import LLMHandler
from acestep.inference import GenerationParams,GenerationConfig,generate_music
sys.path.insert(0,str(R/'prototype'));from composition_sa3 import BASE as TARGET,ROLES
h=AceStepHandler();lm=LLMHandler();start=time.monotonic()
status,ok=h.initialize_service(str(BASE/'models/ace'),config_path='acestep-v15-turbo',device='mps',use_mlx_dit=False,offload_to_cpu=True);assert ok,status
status,ok=lm.initialize(str(BASE/'models/ace/checkpoints'),'acestep-5Hz-lm-1.7B',backend='mlx',device='mps');assert ok,status
calls=[]
class BoundaryCaptured(Exception): pass
def pre(module,args,kwargs):
i=len(calls);t=time.monotonic();calls.append({'start':t,'timestep':kwargs['timestep'].float().cpu().tolist()})
if i==0:
for key,value in kwargs.items():
if isinstance(value,torch.Tensor):np.save(O/(key+'.npy'),value.detach().float().cpu().numpy())
(O/'inputs.json').write_text(json.dumps({k:{'shape':list(v.shape),'dtype':str(v.dtype)} for k,v in kwargs.items() if isinstance(v,torch.Tensor)},indent=2))
raise BoundaryCaptured('Prepared actual conditioning saved; skip host generation')
def post(module,args,kwargs,output):
calls[-1]['seconds']=time.monotonic()-calls[-1]['start']
np.save(O/('velocity_'+str(len(calls)-1)+'.npy'),output[0].detach().float().cpu().numpy())
h.model.decoder.register_forward_pre_hook(pre,with_kwargs=True);h.model.decoder.register_forward_hook(post,with_kwargs=True)
import soundfile as sf,functools
original_generate=h.model.generate_audio
@functools.wraps(original_generate)
def capture_sampler(*args,**kwargs):
for key in ['clean_src_latents','repaint_mask']:
if isinstance(kwargs.get(key),torch.Tensor):np.save(O/('sampler_'+key+'.npy'),kwargs[key].detach().float().cpu().numpy())
(O/'sampler.json').write_text(json.dumps({k:v for k,v in kwargs.items() if isinstance(v,(str,int,float,bool)) or v is None},indent=2))
return original_generate(*args,**kwargs)
h.model.generate_audio=capture_sampler
source,sr=sf.read(R/'results/composition_20260916/ace/verse_to_chorus.wav',always_2d=True,dtype='float32')
# Only already-generated context; no route or future musical continuation is read.
padded=np.zeros((40*sr,2),np.float32);padded[:12*sr]=source[-12*sr:]
source_path=R/'results/ace_chestnut_20260916/repaint_source.wav';sf.write(source_path,padded,sr,subtype='FLOAT')
for role in ['repaint_prechorus','repaint_chorus','repaint_bridge']:
repaint=role.startswith('repaint');duration=40 if repaint else 30
O=R/f'results/ace_chestnut_20260916/{role}';O.mkdir(parents=True,exist_ok=True);calls.clear()
intent=role.removeprefix('repaint_')
structure='[Instrumental]\n['+intent.title()+']'
extra={'task_type':'repaint','src_audio':str(source_path),'repainting_start':12,'repainting_end':40,'chunk_mask_mode':'explicit','thinking':False} if repaint else {'thinking':True}
params=GenerationParams(caption=TARGET+ROLES[intent],lyrics=structure,instrumental=True,bpm=128,keyscale='D minor',timesignature='4',duration=duration,inference_steps=8,seed=12607,dcw_enabled=False,use_cot_caption=False,use_cot_metas=False,use_cot_language=False,reference_audio=str(R/'results/composition_20260916/ace/structured90.wav'),**extra)
t=time.monotonic()
try: result=generate_music(h,lm,params,GenerationConfig(batch_size=1,allow_lm_batch=False,use_random_seed=False,seeds=[12607],audio_format='wav'),save_dir=str(O))
except BoundaryCaptured: pass
(O/'preparation.json').write_text(json.dumps({'seconds':time.monotonic()-t,'duration':duration,'role':role,'captured':(O/'context_latents.npy').exists(),'host':'Mac MPS + MLX planner','sampler_dcw':False},indent=2));print('PREPARED',role,time.monotonic()-t,flush=True)
@@ -1,34 +0,0 @@
"""Run official local ACE and capture the exact boundary around its heavy DiT."""
import os,sys,time,json,resource
from pathlib import Path
import numpy as np,torch
P=Path(__file__).resolve().parent;R=P.parents[1];BASE=P.parent/'composition_20260916';O=R/'results/ace_chestnut_20260916/reference15';O.mkdir(parents=True,exist_ok=True)
os.environ.update(HF_HOME=str(BASE/'cache/hf'),HF_HUB_DISABLE_TELEMETRY='1',PYTORCH_ENABLE_MPS_FALLBACK='1',TOKENIZERS_PARALLELISM='false')
from acestep.handler import AceStepHandler
from acestep.llm_inference import LLMHandler
from acestep.inference import GenerationParams,GenerationConfig,generate_music
sys.path.insert(0,str(R/'prototype'));from composition_sa3 import BASE as TARGET,ROLES
h=AceStepHandler();lm=LLMHandler();start=time.monotonic()
status,ok=h.initialize_service(str(BASE/'models/ace'),config_path='acestep-v15-turbo',device='mps',use_mlx_dit=False,offload_to_cpu=True);assert ok,status
status,ok=lm.initialize(str(BASE/'models/ace/checkpoints'),'acestep-5Hz-lm-1.7B',backend='mlx',device='mps');assert ok,status
calls=[]
class BoundaryCaptured(Exception): pass
def pre(module,args,kwargs):
i=len(calls);t=time.monotonic();calls.append({'start':t,'timestep':kwargs['timestep'].float().cpu().tolist()})
if i==0:
for key,value in kwargs.items():
if isinstance(value,torch.Tensor):np.save(O/(key+'.npy'),value.detach().float().cpu().numpy())
(O/'inputs.json').write_text(json.dumps({k:{'shape':list(v.shape),'dtype':str(v.dtype)} for k,v in kwargs.items() if isinstance(v,torch.Tensor)},indent=2))
raise BoundaryCaptured('Prepared actual conditioning saved; skip host generation')
def post(module,args,kwargs,output):
calls[-1]['seconds']=time.monotonic()-calls[-1]['start']
np.save(O/('velocity_'+str(len(calls)-1)+'.npy'),output[0].detach().float().cpu().numpy())
h.model.decoder.register_forward_pre_hook(pre,with_kwargs=True);h.model.decoder.register_forward_hook(post,with_kwargs=True)
for duration in [30,45,60,90]:
O=R/f'results/ace_chestnut_20260916/reference{duration}';O.mkdir(parents=True,exist_ok=True);calls.clear()
params=GenerationParams(caption=TARGET+ROLES['verse_to_chorus'],lyrics='[Instrumental]\n[Verse]\n[Pre-Chorus]\n[Chorus]',instrumental=True,bpm=128,keyscale='D minor',timesignature='4',duration=duration,inference_steps=8,seed=12601,thinking=True,use_cot_caption=False,use_cot_metas=False,use_cot_language=False,reference_audio=str(R/'results/composition_20260916/ace/structured90.wav'))
t=time.monotonic()
try: result=generate_music(h,lm,params,GenerationConfig(batch_size=1,allow_lm_batch=False,use_random_seed=False,seeds=[12601],audio_format='wav'),save_dir=str(O))
except BoundaryCaptured: pass
(O/'preparation.json').write_text(json.dumps({'seconds':time.monotonic()-t,'duration':duration,'captured':(O/'context_latents.npy').exists(),'host':'Mac MPS + MLX planner','initialization_seconds':t-start if duration==30 else None},indent=2))
print('PREPARED',duration,time.monotonic()-t,flush=True)
@@ -1,65 +0,0 @@
"""Prepare real generic current plans once; never generate or cache playback PCM."""
import argparse
import json
from pathlib import Path
import shutil
import sys
import tempfile
import time
import numpy as np
sys.path.insert(0,str(Path(__file__).resolve().parents[2]/'prototype'))
from hook_planning import PlanCache, VERSION, digest, request_plan, validate_prepared
from cached_composition import SCHEMA, ROLES, validate_bank
from host_hook_adapter import HostHookAdapter
def export_bank(output,plans,prefix,reference_hash,seed):
output=Path(output);output.mkdir(exist_ok=False)
manifest={'schema':SCHEMA,'profile':'prism','plan_version':VERSION,'preparation_seed':seed,
'reference_audio_sha256':reference_hash,'reference_mode':'fixed-preparation-audio','contains_pcm':False,
'runtime_seed_policy':'fresh logged native diffusion seed per normal launch; semantic plans reused',
'runtime_context_policy':'overwrite retained8s with actual committed fresh latent tail',
'scope':'generic demo optimization; same bundle for route1 and route2; no route/event inputs',
'roles':{}}
for role,(request,directory) in plans.items():
dest=output/role;dest.mkdir()
sources=None if role=='initial' else {'committed_prefix':prefix}
hashes=validate_prepared(request,directory,sources)
for name in hashes:shutil.copyfile(directory/name,dest/name)
if role!='initial':
shutil.copyfile(prefix,dest/'preparation_prefix.npy');hashes['preparation_prefix.npy']=digest(prefix)
manifest['roles'][role]={'request':request.identity(),'plan_key':request.cache_key,'sha256':hashes}
(output/'bank.json').write_text(json.dumps(manifest,indent=2))
validate_bank(output)
return manifest
def main():
parser=argparse.ArgumentParser();parser.add_argument('--assets-root',type=Path,required=True);parser.add_argument('--reference-batch',type=Path,required=True);parser.add_argument('--output',type=Path,required=True);args=parser.parse_args()
if args.output.exists():raise FileExistsError('Do not overwrite prepared conditioning')
report=json.loads((args.reference_batch/'validation.json').read_text())
seed=report['sessions'][0]['generation_seed'];first=report['runs'][0]
if first['generation_seed']!=seed or not first['windows'][0]['quality']['accepted']:raise ValueError('First registered session has no accepted initial reference')
reference=args.reference_batch/'session_1/hook_reference.wav';latent=args.reference_batch/'session_1/quality/00_committed.npy'
initial_key=first['windows'][0]['plan_key']
original=json.loads((args.reference_batch/'plans'/initial_key/'prepared.json').read_text())['request']
if original['version']!=VERSION or original['section']!='initial':raise ValueError('Reference is not current initial strategy')
adapter=HostHookAdapter(args.assets_root,preparation_only=True);identities=adapter.fingerprints()
args.output.parent.mkdir(parents=True,exist_ok=True)
cache=PlanCache(args.output.parent/'current-planner-cache')
with tempfile.TemporaryDirectory(prefix='bank-prefix-',dir=args.output.parent) as scratch:
prefix=Path(scratch)/'prefix.npy';np.save(prefix,np.load(latent,allow_pickle=False)[:,-200:].copy())
sources={'hook_reference':reference,'committed_prefix':prefix}
plans={};started=time.monotonic()
indices={'initial':0,'verse':1,'prechorus':2,'chorus':3,'bridge':5,'outro':7}
for role in ROLES:
if shutil.disk_usage(args.output.parent).free<1024**3:raise RuntimeError('Less than1GiB free; refusing preparation')
context={} if role=='initial' else dict(hook_reference_sha256=digest(reference),committed_prefix_sha256=digest(prefix),previous_plan_sha256=plans['initial'][0].cache_key)
request=request_plan(session_seed=seed,plan_index=indices[role],profile='prism',section=role,window_seconds=30 if role=='initial' else 45,**identities,**context)
tick=time.monotonic();directory,hit=cache.resolve(request,adapter,sources={} if role=='initial' else sources)
plans[role]=(request,directory)
print(json.dumps({'role':role,'plan_key':request.cache_key,'cache_hit':hit,'seconds':time.monotonic()-tick}),flush=True)
export_bank(args.output,plans,prefix,digest(reference),seed)
print(json.dumps({'complete':True,'output':str(args.output),'seconds':time.monotonic()-started,'contains_pcm':False}),flush=True)
if __name__=='__main__':main()
@@ -1,145 +0,0 @@
"""Mac-only semantic planning/tensor capture. Stops BEFORE audio diffusion or decoding."""
import argparse
import hashlib
import json
import os
from pathlib import Path
import random
import sys
import time
import traceback
from prism_hook_spec import spec
class BoundaryCaptured(BaseException):
pass
def main():
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument('--assets-root', type=Path, required=True, help='Existing RoadScore assets; read only')
parser.add_argument('--output-root', type=Path, required=True, help='New private demo artifacts only')
parser.add_argument('--route-seed', type=int)
parser.add_argument('--prepare', action='store_true', help='Run one host semantic plan; no audio diffusion')
args = parser.parse_args()
recipe = spec(args.route_seed)
out = args.output_root.resolve() / f'prism_hook_v1_{time.time_ns()}'
out.mkdir(parents=True, exist_ok=False)
(out / 'hook_spec.json').write_text(json.dumps(recipe, indent=2))
state = {'phase': 'spec_only', 'output': str(out), 'spec_sha256': recipe['spec_sha256'],
'audio_generated': False, 'native_deployed': False, 'continuation_tensors_prepared': False}
def record():
(out / 'preparation.json').write_text(json.dumps(state, indent=2))
record()
if not args.prepare:
print(json.dumps(state))
return
if sys.platform != 'darwin':
raise RuntimeError('This host-preparation entry point is Mac-only; no device execution')
base = args.assets_root.resolve() / 'experiments/composition_20260916'
if not (base / 'models/ace/checkpoints').is_dir():
raise FileNotFoundError('Existing local ACE checkpoints missing; no download attempted')
os.environ.update(HF_HOME=str(base / 'cache/hf'), HF_HUB_OFFLINE='1', HF_HUB_DISABLE_TELEMETRY='1',
TOKENIZERS_PARALLELISM='false', PYTORCH_ENABLE_MPS_FALLBACK='1')
started = time.monotonic()
state.update(phase='loading', started_wall=time.time())
record()
try:
import numpy as np
import torch
import mlx.core as mx
from acestep.handler import AceStepHandler
from acestep.llm_inference import LLMHandler
from acestep.inference import GenerationParams, GenerationConfig, generate_music
seed = recipe['seed']
random.seed(seed)
np.random.seed(seed)
torch.manual_seed(seed)
mx.random.seed(seed)
handler, lm = AceStepHandler(), LLMHandler()
status, ok = handler.initialize_service(str(base / 'models/ace'), config_path='acestep-v15-turbo',
device='mps', use_mlx_dit=True, offload_to_cpu=True, offload_dit_to_cpu=False)
if not ok:
raise RuntimeError(status)
if not handler.use_mlx_dit or handler.mlx_decoder is None:
raise RuntimeError('MLX boundary unavailable; refuse fallback audio generation')
status, ok = lm.initialize(str(base / 'models/ace/checkpoints'), 'acestep-5Hz-lm-1.7B', backend='mlx', device='mps')
if not ok:
raise RuntimeError(status)
original_plan = lm.generate_with_stop_condition
def record_plan(*values, **kwargs):
result = original_plan(*values, **kwargs)
codes = result.get('audio_codes', '')
if not result.get('success') or not codes:
raise RuntimeError('Semantic planner did not return audio codes')
(out / 'semantic_plan.json').write_text(json.dumps({
'caption': kwargs.get('caption'), 'lyrics': kwargs.get('lyrics'),
'seeds': kwargs.get('seeds'), 'infer_type': kwargs.get('infer_type'),
'temperature': kwargs.get('temperature'), 'cfg_scale': kwargs.get('cfg_scale'),
'audio_codes': codes, 'metadata': result.get('metadata'),
'time_costs': result.get('extra_outputs', {}).get('time_costs'),
}, indent=2))
state['semantic_plan_present'] = True
return result
lm.generate_with_stop_condition = record_plan
state.update(phase='planning', load_seconds=time.monotonic() - started)
record()
def capture(*unused, **kwargs):
required = ('encoder_hidden_states', 'encoder_attention_mask', 'context_latents')
arrays = {}
for key in required:
tensor = kwargs.get(key)
if not isinstance(tensor, torch.Tensor):
raise ValueError(f'Missing prepared boundary tensor: {key}')
array = tensor.detach().float().cpu().numpy()
if not np.isfinite(array).all():
raise ValueError(f'Non-finite {key}')
arrays[key] = array
if arrays['context_latents'].shape != (1, 1500, 128):
raise ValueError('Preparation did not produce a full60s context')
if kwargs.get('repaint_mask') is not None:
raise ValueError('Full planned composition must not silently become repaint')
if not state.get('semantic_plan_present'):
raise ValueError('Refuse unplanned conditioning')
case = dict(recipe, name='prism_hook_v1_60', reference_audio=None,
sampler={k: v for k, v in kwargs.items() if isinstance(v, (str, int, float, bool)) or v is None},
tensor_shapes={k: list(v.shape) for k, v in arrays.items()})
for key, array in arrays.items():
np.save(out / (key + '.npy'), array)
(out / 'case.json').write_text(json.dumps(case, indent=2))
state.update(phase='prepared_boundary', audio_diffusion_called=False,
boundary_sha256={p.name: hashlib.sha256(p.read_bytes()).hexdigest()
for p in out.glob('*.npy')},
tensor_shapes=case['tensor_shapes'], elapsed_seconds=time.monotonic() - started)
record()
raise BoundaryCaptured()
handler._mlx_run_diffusion = capture
params = GenerationParams(caption=recipe['caption'], lyrics=recipe['lyrics'], instrumental=True,
bpm=128, keyscale='D minor', timesignature='4', duration=60, inference_steps=8, seed=seed,
thinking=True, dcw_enabled=False, use_cot_caption=False, use_cot_metas=False, use_cot_language=False)
try:
result = generate_music(handler, lm, params,
GenerationConfig(batch_size=1, allow_lm_batch=False, use_random_seed=False, seeds=[seed], audio_format='wav'),
save_dir=str(out / 'unused_audio'))
except BoundaryCaptured:
pass
else:
raise RuntimeError(f'Expected boundary capture, got success={result.success}: {result.error}')
if state['phase'] != 'prepared_boundary':
raise RuntimeError('No native-compatible tensor capture')
print(json.dumps(state))
except BaseException as error:
state.update(phase='failed', error=repr(error), traceback=traceback.format_exc(), elapsed_seconds=time.monotonic() - started)
record()
raise
if __name__ == '__main__':
main()
@@ -1,51 +0,0 @@
"""Run official local ACE and capture the exact boundary around its heavy DiT."""
import os,sys,time,json,resource
from pathlib import Path
import numpy as np,torch
P=Path(__file__).resolve().parent;R=P.parents[1];BASE=P.parent/'composition_20260916';O=R/'results/ace_chestnut_20260916/reference15';O.mkdir(parents=True,exist_ok=True)
os.environ.update(HF_HOME=str(BASE/'cache/hf'),HF_HUB_DISABLE_TELEMETRY='1',PYTORCH_ENABLE_MPS_FALLBACK='1',TOKENIZERS_PARALLELISM='false')
from acestep.handler import AceStepHandler
from acestep.llm_inference import LLMHandler
from acestep.inference import GenerationParams,GenerationConfig,generate_music
sys.path.insert(0,str(R/'prototype'));from composition_sa3 import BASE as TARGET,ROLES
h=AceStepHandler();lm=LLMHandler();start=time.monotonic()
status,ok=h.initialize_service(str(BASE/'models/ace'),config_path='acestep-v15-turbo',device='mps',use_mlx_dit=False,offload_to_cpu=True);assert ok,status
status,ok=lm.initialize(str(BASE/'models/ace/checkpoints'),'acestep-5Hz-lm-1.7B',backend='mlx',device='mps');assert ok,status
calls=[]
class BoundaryCaptured(Exception): pass
def pre(module,args,kwargs):
i=len(calls);t=time.monotonic();calls.append({'start':t,'timestep':kwargs['timestep'].float().cpu().tolist()})
if i==0:
for key,value in kwargs.items():
if isinstance(value,torch.Tensor):np.save(O/(key+'.npy'),value.detach().float().cpu().numpy())
(O/'inputs.json').write_text(json.dumps({k:{'shape':list(v.shape),'dtype':str(v.dtype)} for k,v in kwargs.items() if isinstance(v,torch.Tensor)},indent=2))
raise BoundaryCaptured('Prepared actual conditioning saved; skip host generation')
def post(module,args,kwargs,output):
calls[-1]['seconds']=time.monotonic()-calls[-1]['start']
np.save(O/('velocity_'+str(len(calls)-1)+'.npy'),output[0].detach().float().cpu().numpy())
h.model.decoder.register_forward_pre_hook(pre,with_kwargs=True);h.model.decoder.register_forward_hook(post,with_kwargs=True)
import soundfile as sf,functools
original_generate=h.model.generate_audio
@functools.wraps(original_generate)
def capture_sampler(*args,**kwargs):
for key in ['clean_src_latents','repaint_mask']:
if isinstance(kwargs.get(key),torch.Tensor):np.save(O/('sampler_'+key+'.npy'),kwargs[key].detach().float().cpu().numpy())
(O/'sampler.json').write_text(json.dumps({k:v for k,v in kwargs.items() if isinstance(v,(str,int,float,bool)) or v is None},indent=2))
return original_generate(*args,**kwargs)
h.model.generate_audio=capture_sampler
source,sr=sf.read(R/'results/composition_20260916/ace/verse_to_chorus.wav',always_2d=True,dtype='float32')
# Only already-generated context; no route or future musical continuation is read.
padded=np.zeros((40*sr,2),np.float32);padded[:12*sr]=source[-12*sr:]
source_path=R/'results/ace_chestnut_20260916/repaint_source.wav';sf.write(source_path,padded,sr,subtype='FLOAT')
for role in ['verse','prechorus','chorus','bridge','outro','repaint_transition','repaint_outro']:
repaint=role.startswith('repaint');duration=40 if repaint else 30
O=R/f'results/ace_chestnut_20260916/{role}';O.mkdir(parents=True,exist_ok=True);calls.clear()
intent='song_to_outro' if role=='repaint_outro' else 'chorus_to_bridge' if role=='repaint_transition' else role
structure='[Instrumental]\n[Chorus]\n[Outro]' if intent=='song_to_outro' else '[Instrumental]\n[Chorus]\n[Bridge]' if repaint else '[Instrumental]\n['+role.title()+']'
extra={'task_type':'repaint','src_audio':str(source_path),'repainting_start':12,'repainting_end':40,'chunk_mask_mode':'explicit','thinking':False} if repaint else {'thinking':True}
params=GenerationParams(caption=TARGET+ROLES[intent],lyrics=structure,instrumental=True,bpm=128,keyscale='D minor',timesignature='4',duration=duration,inference_steps=8,seed=12607,dcw_enabled=False,use_cot_caption=False,use_cot_metas=False,use_cot_language=False,reference_audio=str(R/'results/composition_20260916/ace/structured90.wav'),**extra)
t=time.monotonic()
try: result=generate_music(h,lm,params,GenerationConfig(batch_size=1,allow_lm_batch=False,use_random_seed=False,seeds=[12607],audio_format='wav'),save_dir=str(O))
except BoundaryCaptured: pass
(O/'preparation.json').write_text(json.dumps({'seconds':time.monotonic()-t,'duration':duration,'role':role,'captured':(O/'context_latents.npy').exists(),'host':'Mac MPS + MLX planner','sampler_dcw':False},indent=2));print('PREPARED',role,time.monotonic()-t,flush=True)
@@ -1,50 +0,0 @@
"""Run official local ACE and capture the exact boundary around its heavy DiT."""
import os,sys,time,json,resource
from pathlib import Path
import numpy as np,torch
P=Path(__file__).resolve().parent;R=P.parents[1];BASE=P.parent/'composition_20260916';O=R/'results/ace_chestnut_20260916/reference15';O.mkdir(parents=True,exist_ok=True)
os.environ.update(HF_HOME=str(BASE/'cache/hf'),HF_HUB_DISABLE_TELEMETRY='1',PYTORCH_ENABLE_MPS_FALLBACK='1',TOKENIZERS_PARALLELISM='false')
from acestep.handler import AceStepHandler
from acestep.llm_inference import LLMHandler
from acestep.inference import GenerationParams,GenerationConfig,generate_music
sys.path.insert(0,str(R/'prototype'));from composition_sa3 import BASE as TARGET,ROLES
h=AceStepHandler();lm=LLMHandler();start=time.monotonic()
status,ok=h.initialize_service(str(BASE/'models/ace'),config_path='acestep-v15-turbo',device='mps',use_mlx_dit=False,offload_to_cpu=True);assert ok,status
calls=[]
class BoundaryCaptured(Exception): pass
def pre(module,args,kwargs):
i=len(calls);t=time.monotonic();calls.append({'start':t,'timestep':kwargs['timestep'].float().cpu().tolist()})
if i==0:
for key,value in kwargs.items():
if isinstance(value,torch.Tensor):np.save(O/(key+'.npy'),value.detach().float().cpu().numpy())
(O/'inputs.json').write_text(json.dumps({k:{'shape':list(v.shape),'dtype':str(v.dtype)} for k,v in kwargs.items() if isinstance(v,torch.Tensor)},indent=2))
raise BoundaryCaptured('Prepared actual conditioning saved; skip host generation')
def post(module,args,kwargs,output):
calls[-1]['seconds']=time.monotonic()-calls[-1]['start']
np.save(O/('velocity_'+str(len(calls)-1)+'.npy'),output[0].detach().float().cpu().numpy())
h.model.decoder.register_forward_pre_hook(pre,with_kwargs=True);h.model.decoder.register_forward_hook(post,with_kwargs=True)
import soundfile as sf,functools
original_generate=h.model.generate_audio
@functools.wraps(original_generate)
def capture_sampler(*args,**kwargs):
for key in ['clean_src_latents','repaint_mask']:
if isinstance(kwargs.get(key),torch.Tensor):np.save(O/('sampler_'+key+'.npy'),kwargs[key].detach().float().cpu().numpy())
(O/'sampler.json').write_text(json.dumps({k:v for k,v in kwargs.items() if isinstance(v,(str,int,float,bool)) or v is None},indent=2))
return original_generate(*args,**kwargs)
h.model.generate_audio=capture_sampler
source,sr=sf.read(R/'results/composition_20260916/ace/verse_to_chorus.wav',always_2d=True,dtype='float32')
# Only already-generated context; no route or future musical continuation is read.
padded=np.zeros((40*sr,2),np.float32);padded[:12*sr]=source[-12*sr:]
source_path=R/'results/ace_chestnut_20260916/repaint_source.wav';sf.write(source_path,padded,sr,subtype='FLOAT')
for role in ['repaint_verse']:
repaint=role.startswith('repaint');duration=40 if repaint else 30
O=R/f'results/ace_chestnut_20260916/{role}';O.mkdir(parents=True,exist_ok=True);calls.clear()
intent=role.removeprefix('repaint_')
structure='[Instrumental]\n['+intent.title()+']'
extra={'task_type':'repaint','src_audio':str(source_path),'repainting_start':12,'repainting_end':40,'chunk_mask_mode':'explicit','thinking':False} if repaint else {'thinking':True}
params=GenerationParams(caption=TARGET+ROLES[intent],lyrics=structure,instrumental=True,bpm=128,keyscale='D minor',timesignature='4',duration=duration,inference_steps=8,seed=12607,dcw_enabled=False,use_cot_caption=False,use_cot_metas=False,use_cot_language=False,reference_audio=str(R/'results/composition_20260916/ace/structured90.wav'),**extra)
t=time.monotonic()
try: result=generate_music(h,lm,params,GenerationConfig(batch_size=1,allow_lm_batch=False,use_random_seed=False,seeds=[12607],audio_format='wav'),save_dir=str(O))
except BoundaryCaptured: pass
(O/'preparation.json').write_text(json.dumps({'seconds':time.monotonic()-t,'duration':duration,'role':role,'captured':(O/'context_latents.npy').exists(),'host':'Mac MPS + MLX planner','sampler_dcw':False},indent=2));print('PREPARED',role,time.monotonic()-t,flush=True)
@@ -1,75 +0,0 @@
"""Shared musical intent for a candidate; this alone does not change prepared tensors."""
import hashlib
import json
VERSION = 'prism-hook-v1'
IDENTITY = (
'Instrumental polished K-pop and modern electronic game score. No vocals, singing or speech. '
'128 BPM, D minor, 4/4. Tight electronic drums, punchy rubbery syncopated bass, '
'crystal pluck arpeggios and bright glass synth leads. '
)
HOOK = (
'Give this composition one distinctive, immediately memorable four-note rising synth hook '
'with a catchy syncopated rhythm and a short answering phrase. Establish its recognizable '
'melodic contour and rhythm early. Keep that SAME hook identity throughout this song: '
'develop its rhythm, register, accompaniment and dynamics, then bring back the clear original '
'phrase at each chorus. Make the motif specific to this composition, not a generic running '
'arpeggio. Leave breathing space between hook statements. Do not replace it with unrelated '
'lead melodies or mechanically repeat an unchanged loop. '
)
ROLES = {
'initial': 'Introduce the hook clearly, develop a lighter verse, rise into a build and deliver a confident first chorus. ',
'verse': 'Develop a lighter verse using recognizable fragments and a quieter call-and-response of the established hook over an active bass groove. Vary the accompaniment and leave space for the hook return. ',
'prechorus': 'Build tension with shorter recognizable hook fragments, rising register and denser drum subdivisions. Aim the phrase toward the chorus; do not introduce a new main melody. ',
'chorus': 'State the complete established hook prominently with its original contour and signature rhythm; answer it with the same phrase in a wider register and fuller bass/drums. Make this a clear melodic payoff, not merely louder texture. ',
'bridge': 'Create contrast by reducing the arrangement and transforming the same hook into a spacious half-time rhythmic answer in compatible glass/pluck timbre. Retain its melodic identity and D minor harmony, then rebuild toward the original chorus hook. ',
'outro': 'Return to the recognizable complete hook, answer and resolve its last phrase, then thin the arrangement into a gentle intentional conclusion. ',
}
TIMELINE = [
('initial', 2, 'Clear hook introduction'),
('verse', 8, 'Hook fragments and lighter call-and-response'),
('prechorus', 4, 'Rising fragment development'),
('chorus', 8, 'Complete hook and melodic payoff'),
('bridge', 4, 'Contrasting transformation of same motif'),
('chorus', 4, 'Recognizable original hook reprise'),
('outro', 2, 'Motif answer and resolution'),
]
def spec(route_seed=None):
if route_seed is not None and (type(route_seed) is not int or not 0 <= route_seed < 2**32):
raise ValueError('route_seed must be an unsigned 32-bit integer')
seed = 33602 if route_seed is None else int.from_bytes(
hashlib.sha256(f'roadscore-sample-v1:{route_seed}:prepare:0'.encode()).digest()[:4], 'big')
interior = 'This is a continuing interior section: preserve the pulse and hand off naturally without a terminal fade or silence. '
prompts = {role: IDENTITY + HOOK + text + (interior if role not in ('initial', 'outro') else '')
for role, text in ROLES.items()}
cursor = 0
timeline = []
for role, bars, intent in TIMELINE:
duration = bars * 4 * 60 / 128
timeline.append({'role': role, 'bars': bars, 'start_seconds': cursor,
'end_seconds': cursor + duration, 'intent': intent})
cursor += duration
result = {
'version': VERSION, 'profile': 'prism', 'bpm': 128, 'keyscale': 'D minor', 'timesignature': '4',
'duration': 60, 'seed': seed, 'seed_mode': 'fixed-reference' if route_seed is None else 'route-derived',
'route_seed': route_seed, 'thinking': True, 'role_captions': prompts,
'role_lyrics': {r: '[Instrumental]\n[' + ('Pre-Chorus' if r == 'prechorus' else r.title()) + ']'
for r in ROLES if r != 'initial'},
'caption': IDENTITY + HOOK + (
'Compose a coherent one-minute arc: establish the hook, develop a lighter verse, '
'build anticipation, reveal a full chorus, give a brief contrasting bridge variation, '
'then reprise the original hook and resolve naturally. Each section should change '
'the musical arrangement while retaining the song\'s identity. '),
'lyrics': '[Instrumental]\n[Intro]\n[Verse]\n[Pre-Chorus]\n[Chorus]\n[Bridge]\n[Chorus]\n[Outro]',
'desired_timeline': timeline,
'timeline_status': '32-bar intent at 128 BPM, not verified generated section boundaries; never force cuts to these timestamps',
'continuation_policy': 'role contracts only; no prepared continuation tensors or runtime deployment claimed',
'uniqueness_limit': 'Distinct hook is musical intent, not a metric guarantee; fixed seeds intentionally reproduce a composition.',
}
result['caption'] += ('Target a 32-bar arc: 2-bar introduction, 8-bar verse, 4-bar build, '
'8-bar chorus, 4-bar bridge variation, 4-bar chorus reprise and 2-bar resolution. ')
result['role_lyrics']['initial'] = result['lyrics']
result['spec_sha256'] = hashlib.sha256(json.dumps(result, sort_keys=True).encode()).hexdigest()
return result
@@ -1,15 +0,0 @@
import os,time,json,fcntl,resource
from pathlib import Path
import numpy as np
from native_ace import DiT,tensor
from tinygrad import TinyJit,Device
from tinygrad.helpers import GlobalCounters
P=Path(__file__).resolve().parent;lock=open('/data/roadscore/generated/gpu.lock','w');fcntl.flock(lock,fcntl.LOCK_EX|fcntl.LOCK_NB)
t=time.monotonic();m=DiT(P/'weights',layers=1);load=time.monotonic()-t
x,c,temb,co,si,mask=[tensor(np.load(P/(n+'.npy'))) for n in ['x','cond','temb','cos','sin','mask']];fn=TinyJit(lambda x,c,t,co,si,ma:m.block(x,c,t,(co.unsqueeze(1),si.unsqueeze(1)),ma))
times=[]
for i in range(4):
t=time.monotonic();out=fn(x,c,temb,co,si,mask);Device['AMD'].synchronize();times.append(time.monotonic()-t);print('BLOCK',i,times[-1],flush=True)
y=out.numpy().astype(np.float32);ref=np.load(P/'expected.npy');np.save(P/'block_actual.npy',y)
d={'load_seconds':load,'block_times':times,'relative_rmse':float(np.sqrt(np.mean((y-ref)**2))/np.sqrt(np.mean(ref**2))),'max_abs_error':float(abs(y-ref).max()),'finite':bool(np.isfinite(y).all()),'tracked_gpu_bytes':GlobalCounters.mem_used,'peak_host_mib':resource.getrusage(resource.RUSAGE_SELF).ru_maxrss/1024,'weights_bytes':sum(w.numel()*w.dtype.itemsize for w in m.w.values())};(P/'block_result.json').write_text(json.dumps(d,indent=2));print(d,flush=True)
assert d['finite'] and d['relative_rmse']<.01 and d['max_abs_error']<.05
@@ -1,35 +0,0 @@
{
"name": "aurora45_bridge",
"caption": "Instrumental polished K-pop and modern electronic game score. No vocals, no singing, no speech. Continuous tight drum groove, punchy bass, memorable recurring hook, polished dynamic arrangement. 116 BPM, A minor. Warm analog synths, octave bass ostinato, shimmering bell answers and a lyrical three-note descending lead motif; syncopated dance-pop drums. Contrasting bridge in the same tonal center, preserve groove identity, different texture and instrumentation, new lead timbre, then rebuild. No modulation, silence or fade.",
"duration": 45,
"declared_condition_duration": 120,
"seed": 77203,
"reference_audio": "/Users/dominickthompson/Desktop/RoadScore/results/ace_stability_20260916/cases/aurora_60/official/audio.wav",
"thinking": false,
"sampler": {
"infer_method": "ode",
"shift": 1.0,
"timesteps": null,
"infer_steps": 8,
"guidance_scale": 1.0,
"cfg_interval_start": 0.0,
"cfg_interval_end": 1.0,
"audio_cover_strength": 1.0,
"encoder_hidden_states_non_cover": null,
"encoder_attention_mask_non_cover": null,
"context_latents_non_cover": null,
"sampler_mode": "euler",
"velocity_norm_threshold": 0.0,
"velocity_ema_factor": 0.0,
"dcw_enabled": false,
"dcw_mode": "double",
"dcw_scaler": 0.05,
"dcw_high_scaler": 0.02,
"dcw_wavelet": "haar",
"retake_seed": null,
"retake_variance": 0.0,
"repaint_crossfade_frames": 12,
"repaint_injection_ratio": 0.5
},
"preparation": "official120s source/task conditioning; first45s bounded window;8s prefix"
}
@@ -1,25 +0,0 @@
{
"infer_method": "ode",
"shift": 1.0,
"timesteps": null,
"infer_steps": 8,
"guidance_scale": 1.0,
"cfg_interval_start": 0.0,
"cfg_interval_end": 1.0,
"audio_cover_strength": 1.0,
"encoder_hidden_states_non_cover": null,
"encoder_attention_mask_non_cover": null,
"context_latents_non_cover": null,
"sampler_mode": "euler",
"velocity_norm_threshold": 0.0,
"velocity_ema_factor": 0.0,
"dcw_enabled": false,
"dcw_mode": "double",
"dcw_scaler": 0.05,
"dcw_high_scaler": 0.02,
"dcw_wavelet": "haar",
"retake_seed": null,
"retake_variance": 0.0,
"repaint_crossfade_frames": 12,
"repaint_injection_ratio": 0.5
}

Some files were not shown because too many files have changed in this diff Show More