phonon-2-coreml / config.json
alexwengg's picture
Phonon-2 Core ML: 5 exact five-value encoders (sparse-g8 default), decoder, joint, preprocessor, vocab, attribution
ab5f1e8 verified
Raw History Blame Contribute Delete
1.63 kB
{
"base_model": "FermionResearch/Phonon-2",
"original_base_model": "nvidia/parakeet-tdt-0.6b-v3",
"contract": "parakeet-tdt-0.6b-v3-coreml",
"language": ["en"],
"minimum_os": "iOS18/macOS15",
"encoders": {
"Encoder.mlmodelc": {
"weight_format": "five_value_sparse_mask_lut6_8rows_fp16",
"size_mb": 321,
"recommended_compute_units": "cpuAndNeuralEngine",
"note": "default; exact; ANE 18.6 ms per 15 s window (159x RTFx); first ANE load compiles ~1 min; GPU load materializes the sparse weights (~150 s every launch) - use Encoder_lut3 for GPU"
},
"Encoder_sparse-g4.mlmodelc": {
"weight_format": "five_value_sparse_mask_lut4_4rows_fp16",
"size_mb": 246,
"recommended_compute_units": "cpuAndNeuralEngine",
"note": "exact; ANE 24.3 ms (140x RTFx, v3 speed); same GPU load caveat"
},
"Encoder_sparse-g1.mlmodelc": {
"weight_format": "five_value_sparse_mask_lut2_per_row_fp16",
"size_mb": 176,
"recommended_compute_units": "cpuAndNeuralEngine",
"note": "exact; smallest; ANE 70 ms (~70x RTFx); same GPU load caveat"
},
"Encoder_lut6.mlmodelc": {
"weight_format": "five_value_lut6_8rows_fp16",
"size_mb": 470,
"recommended_compute_units": "any",
"note": "exact dense palette; ANE 18.6 ms, GPU 16 ms, loads in under a second on either"
},
"Encoder_lut3.mlmodelc": {
"weight_format": "five_value_lut3_per_row_fp16",
"size_mb": 253,
"recommended_compute_units": "cpuAndGPU",
"note": "exact dense palette; GPU 16 ms with sub-second load; 72 ms on the ANE"
}
}
}