mirror of
https://github.com/firestar5683/StarPilot.git
synced 2026-10-06 06:14:02 +08:00
182 lines
12 KiB
Python
182 lines
12 KiB
Python
"""Read-only audio screening and local listening page for every validation session."""
|
|
import argparse
|
|
import hashlib
|
|
import html
|
|
import json
|
|
import os
|
|
from pathlib import Path
|
|
|
|
import numpy as np
|
|
from scipy.io import wavfile
|
|
from scipy.signal import resample_poly, stft, find_peaks
|
|
|
|
|
|
def load(path):
|
|
rate, pcm = wavfile.read(path)
|
|
if np.issubdtype(pcm.dtype, np.integer):
|
|
pcm = pcm.astype(np.float64) / (2 ** (np.iinfo(pcm.dtype).bits - 1))
|
|
return rate, np.asarray(pcm, dtype=np.float64)
|
|
|
|
|
|
def rms(x):
|
|
return float(np.sqrt(np.mean(np.square(x)))) if x.size else 0.
|
|
|
|
|
|
def screen(path):
|
|
rate, pcm = load(path)
|
|
if not np.isfinite(pcm).all():
|
|
return {'file': path.name, 'error': 'Non-finite samples'}
|
|
mono = pcm.mean(axis=1) if pcm.ndim > 1 else pcm
|
|
hop = rate // 10
|
|
levels = np.array([rms(pcm[i:i+hop]) for i in range(0, len(pcm), hop)])
|
|
quiet = levels < 1e-4
|
|
longest = count = 0
|
|
for value in quiet:
|
|
count = count + 1 if value else 0
|
|
longest = max(longest, count)
|
|
fingerprints = {}
|
|
duplicates = []
|
|
for i in range(0, len(pcm) - rate + 1, rate):
|
|
block = pcm[i:i+rate]
|
|
if rms(block) < 1e-4:
|
|
continue
|
|
digest = hashlib.sha256(block.tobytes()).hexdigest()
|
|
if digest in fingerprints:
|
|
duplicates.append([fingerprints[digest], i/rate])
|
|
else:
|
|
fingerprints[digest] = i/rate
|
|
down = resample_poly(mono, 1, 4) if rate == 48000 else mono
|
|
sr = rate // 4 if rate == 48000 else rate
|
|
length = 4 * sr
|
|
phrases = [down[i:i+length] for i in range(0, len(down)-length+1, length)]
|
|
best = None
|
|
for i, left in enumerate(phrases):
|
|
for j in range(i+2, len(phrases)):
|
|
right = phrases[j]
|
|
norm = np.linalg.norm(left) * np.linalg.norm(right)
|
|
score = float(np.dot(left, right) / norm) if norm > 1e-9 else 0.
|
|
if best is None or abs(score) > abs(best['correlation']):
|
|
best = dict(start_seconds=[i*4, j*4], correlation=score)
|
|
frequencies, _, spectrum = stft(down, fs=sr, nperseg=2048, noverlap=1536)
|
|
power = np.abs(spectrum) ** 2
|
|
mean_power = power.mean(axis=1)
|
|
centroid = float(np.dot(frequencies, mean_power) / max(mean_power.sum(), 1e-15))
|
|
return dict(file=path.name, sha256=hashlib.sha256(path.read_bytes()).hexdigest(),
|
|
seconds=len(pcm)/rate, rate=rate, frames=len(pcm), peak=float(np.max(abs(pcm))),
|
|
rms=rms(pcm), samples_at_or_above_full_scale=int(np.count_nonzero(abs(pcm) >= 1)),
|
|
samples_above_0999=int(np.count_nonzero(abs(pcm) >= .999)),
|
|
silence_fraction_100ms=float(quiet.mean()), longest_quiet_seconds=longest*.1,
|
|
exact_repeated_nonsilent_1s_blocks=duplicates,
|
|
strongest_aligned_4s_waveform_match=best, spectral_centroid_hz=centroid)
|
|
|
|
|
|
def assembly_check(pcm, parts, rate):
|
|
expected = np.concatenate(parts)
|
|
if pcm.shape != expected.shape:
|
|
return {'status': 'pending or inconsistent length', 'core_frames': len(pcm), 'window_frames': len(expected)}
|
|
seams = np.cumsum([len(wave) for wave in parts])[:-1]
|
|
include = np.ones(len(pcm), dtype=bool)
|
|
for seam in seams:
|
|
include[max(0, seam-2*rate):seam] = False
|
|
residual = float(np.max(abs(pcm[include]-expected[include]))) if include.any() else None
|
|
return {'status': 'matches outside crossfades' if residual is not None and residual <= 5e-7 else 'unexpected difference outside crossfades',
|
|
'excluded_regions_seconds': [[float(max(0, i/rate-2)), float(i/rate)] for i in seams],
|
|
'max_residual_outside_crossfades': residual,
|
|
'limit': 'Saved windows omit regenerated prefixes. The expected 2s crossfade regions cannot be reconstructed from these files; float32 resume roundoff allowed up to 5e-7.'}
|
|
|
|
|
|
def contrast(pcm, rate):
|
|
hop = max(1, rate // 100)
|
|
envelope = np.array([rms(pcm[i:i+hop]) for i in range(0, len(pcm)-hop+1, hop)])
|
|
onset = np.maximum(0, np.diff(envelope))
|
|
threshold = np.median(onset) + 3*np.median(abs(onset-np.median(onset)))
|
|
peaks, _ = find_peaks(onset, height=max(float(threshold), .001), distance=12)
|
|
return {'rms': rms(pcm), 'peak': float(np.max(abs(pcm))),
|
|
'envelope_variation': float(envelope.std()/max(envelope.mean(), 1e-12)),
|
|
'detected_onsets_per_second': float(len(peaks)/(len(pcm)/rate))}
|
|
|
|
|
|
def compare_initial(current, other):
|
|
rate, pcm = load(current)
|
|
other_rate, original = load(other)
|
|
if rate != other_rate:
|
|
return {'error': 'Sample rates differ; comparison skipped'}
|
|
n = min(len(pcm), len(original))
|
|
a, b = contrast(pcm[:n], rate), contrast(original[:n], rate)
|
|
return {'matched_start_seconds': n/rate, 'current': a, 'reference': b,
|
|
'rms_change_db': float(20*np.log10(max(a['rms'], 1e-12)/max(b['rms'], 1e-12))),
|
|
'reference_sha256': hashlib.sha256(other.read_bytes()).hexdigest(),
|
|
'limits': 'Fixed gain, same-length opening comparison. Onsets use 10ms RMS changes with a 120ms minimum separation; density and envelope variation do not establish groove or musical quality.'}
|
|
|
|
|
|
def main():
|
|
parser = argparse.ArgumentParser(description=__doc__)
|
|
parser.add_argument('directory', type=Path)
|
|
parser.add_argument('--baseline', type=Path)
|
|
parser.add_argument('--reference', type=Path)
|
|
args = parser.parse_args()
|
|
root = args.directory
|
|
validation = json.loads((root / 'validation.json').read_text())
|
|
report = {'scope': 'All sessions retained, original samples/gain unchanged; no listening judgment.',
|
|
'limits': 'Silence is channel-combined RMS below -80 dBFS in 100ms blocks. Repetition compares exact non-silent 1s blocks and zero-lag 4s waveform blocks; it does not measure memorable melody, shifted phrases, or musical quality. RMS/centroid section contrast can reflect level/timbre alone.',
|
|
'sessions': []}
|
|
baseline = json.loads((args.baseline / 'validation.json').read_text()) if args.baseline else None
|
|
cards = []
|
|
for number, seed in enumerate(validation['sessions'], 1):
|
|
folder = root / f'session_{number}'
|
|
core = folder / 'core.wav'
|
|
windows = sorted(folder.glob('[0-9][0-9]_*.wav'))
|
|
run = validation.get('runs', [])[number-1] if len(validation.get('runs', [])) >= number else {}
|
|
status = run.get('status', 'not started')
|
|
finished = status in ('technical_pass_listening_pending', 'quality_failed')
|
|
entry = {'generation_status': status, 'session_finished': finished, 'session': number, 'seed': seed['generation_seed'], 'core_available': core.exists(),
|
|
'windows': [screen(p) for p in windows]}
|
|
entry['section_contrast'] = [dict(from_file=a['file'], to_file=b['file'],
|
|
rms_change_db=float(20*np.log10(max(b['rms'], 1e-12)/max(a['rms'], 1e-12))),
|
|
centroid_change_hz=b['spectral_centroid_hz']-a['spectral_centroid_hz'])
|
|
for a, b in zip(entry['windows'], entry['windows'][1:]) if 'error' not in a and 'error' not in b]
|
|
if core.exists():
|
|
entry['core'] = screen(core)
|
|
rate, pcm = load(core)
|
|
parts = [load(p) for p in windows]
|
|
entry['assembly'] = assembly_check(pcm, [wave for _, wave in parts], rate) if parts and all(r == rate for r, _ in parts) else {'status': 'sample rate mismatch or missing windows'}
|
|
seams = np.cumsum([len(wave) for _, wave in parts])[:-1]
|
|
entry['seams'] = [dict(seconds=float(i/rate), sample_step=float(np.max(abs(pcm[i]-pcm[i-1])))) for i in seams]
|
|
initial = folder / '00_initial.wav'
|
|
original = args.baseline / folder.name / '00_initial.wav' if args.baseline else None
|
|
if initial.exists():
|
|
if original and original.exists() and baseline['sessions'][number-1]['generation_seed'] == seed['generation_seed']:
|
|
entry['same_seed_initial_comparison'] = compare_initial(initial, original)
|
|
if args.reference:
|
|
entry['gold_initial_comparison'] = compare_initial(initial, args.reference)
|
|
report['sessions'].append(entry)
|
|
title = f'Session {number}' + (' · Complete' if status == 'technical_pass_listening_pending' else ' · Stopped early' if status == 'quality_failed' else ' · Partial' if core.exists() else ' · Pending')
|
|
core_label = 'Full session' if status == 'technical_pass_listening_pending' else 'Available audio (partial)'
|
|
choices = ([core] if core.exists() else []) + windows
|
|
options = ''.join(f'<option value="session_{number}/{html.escape(p.name)}">{html.escape(core_label if p == core else p.stem.replace("_", " "))}</option>' for p in choices)
|
|
if original and original.exists() and 'same_seed_initial_comparison' in entry:
|
|
options += f'<option value="{html.escape(os.path.relpath(original, root))}">Earlier version · same seed · opening</option>'
|
|
player = (f'<select aria-label="Session {number} section">{options}</select><audio controls preload="metadata" src="session_{number}/{html.escape(choices[0].name)}"></audio>'
|
|
if choices else '<p class="pending">Session audio is not available yet.</p>')
|
|
metrics = entry.get('core', {})
|
|
summary = (f'{metrics["seconds"]:.1f}s · peak {metrics["peak"]:.3f} · {metrics["samples_at_or_above_full_scale"]} full-scale samples'
|
|
if metrics and 'error' not in metrics else 'Screening pending')
|
|
cards.append(f'<section><h2>{title}</h2><p class="seed">Seed {seed["generation_seed"]}</p>{player}<p>{summary}</p></section>')
|
|
if args.reference:
|
|
cards.append(f'<section><h2>Protected GOLD reference</h2><p>Original 28-second reference, unchanged gain.</p><audio controls preload="metadata" src="{html.escape(os.path.relpath(args.reference, root))}"></audio></section>')
|
|
(root / 'audio_screening.json').write_text(json.dumps(report, indent=2))
|
|
page = '''<!doctype html><html lang="en"><meta charset="utf-8"><meta name="viewport" content="width=device-width,initial-scale=1"><title>RoadScore · Fresh hook review</title><style>
|
|
:root{color-scheme:dark;font:16px system-ui;background:#101417;color:#eef2f4}body{max-width:980px;margin:40px auto;padding:0 24px}h1{font-size:32px;margin-bottom:12px}p{line-height:1.55;color:#c1cbd1}section{padding:24px;border:1px solid #344149;border-radius:16px;margin:20px 0}h2{margin:0}audio{width:100%;margin:12px 0}.seed{font:13px ui-monospace;color:#9cabb5}a{color:#91d7c5}select{padding:8px;margin-top:16px;background:#26383e;color:white;border:1px solid #61767f;border-radius:8px}li{padding:6px}button{background:#26383e;color:white;border:1px solid #61767f;border-radius:8px;padding:10px 16px;cursor:pointer}
|
|
</style><h1>Fresh hook review</h1><p>Three fresh sessions, one policy. Listen for a recognizable hook, meaningful development, and a convincing return. All outputs stay here; no seed was selected for sounding best.</p><p>Mac MLX audition — native runtime equivalence is unverified. Audio is unchanged and never starts automatically. One player runs at a time.</p><button id="stop">Pause all</button>'''+''.join(cards)+'''<p><a href="audio_screening.json">Objective screening</a> · <a href="validation.json">Generation record</a></p><p>Clipping, silence, repeated-block and section-contrast checks flag technical properties. They do not establish musical quality or a memorable melody.</p><script>
|
|
const players=[...document.querySelectorAll('audio')];
|
|
for(const player of players)player.addEventListener('play',()=>{for(const other of players)if(other!==player)other.pause()});
|
|
for(const select of document.querySelectorAll('select'))select.addEventListener('change',()=>{const player=select.parentElement.querySelector('audio');player.pause();player.src=select.value;player.load()});
|
|
document.querySelector('#stop').addEventListener('click',()=>players.forEach(player=>player.pause()));
|
|
</script></html>'''
|
|
(root / 'listen.html').write_text(page)
|
|
print(json.dumps([{'session': e['session'], 'windows': len(e['windows']), 'core': e['core_available']} for e in report['sessions']]))
|
|
|
|
|
|
if __name__ == '__main__':
|
|
main()
|