{ "method": "frame-level cross-entropy distillation (scripts/train-ctc-aligner/distill_ctc.py)", "teacher": { "model": "facebook/wav2vec2-lv-60-espeak-cv-ft", "license": "Apache-2.0", "ships": false }, "corpus": { "synthetic": "corpus_out_big/{supertonic,kokoro} (~1,960 utts; corpus LOST post-training — see docs/rca/2026-07-22-ctc-aligner-training-corpus-loss.md in the source repo)", "natural": "LibriSpeech train-clean-100 (CC-BY 4.0)", "hours_approx": 3 }, "feature_extractor": "Rust-parity MFCC-39, 16 kHz, 13+delta+deltadelta (M0 gate)", "architecture": "ConvCTC ch=256 depthwise-separable dilated (dilations 1,1,2,2,4,4,8,8); 39 ARPABET + blank; ONNX opset 17, external data", "eval": { "median_onset_error_ms": 7, "pct_within_20ms": 94, "pct_within_50ms": 99, "pct_within_100ms": 99, "degenerate_span_rate": 0.0004, "degenerate_span_rate_raw": 0.018, "degenerate_span_note": "raw 177/9650; 173 coincide with teacher pauses (artifact), 4 real student errors (0.04%)", "eval_set": "REGENERATED 2026-07-22 — original held-out set lost (RCA in source repo). 150 committed sentences (eval_texts_v2.txt) x Supertonic+Kokoro = 300 utts / 9,650 onsets, fully unseen by the student. Not comparable to development-time figures.", "eval_gate": "scripts/train-ctc-aligner/eval_rigorous.py (OMOTE_HOLDOUT_FROM=0)" }, "license": "Apache-2.0", "trained_at": "2026-06-30" }