{ "model": "Nemotron-3-Diarization (streaming Sortformer, 8 speakers)", "license": "openmdw-1.1", "source": { "repo": "nvidia/Nemotron-3-Diarization", "revision": "f667ed73aee57d40cc39428eb768b4fd87a0a29e", "file": "model.safetensors", "sha256": "c074d86335b3b794f8fa5edc25594558f128bdb3914d27806a3a5a2e44963cb6", "bytes": 396954592, "reference": "transformers Nemotron3DiarizationForAudioFrameClassification (fp32)" }, "toolchain": { "coreai-torch": "0.4.1", "coreai-core": "1.0.0b2", "torch": "2.9.0", "macos": "27.0 (26A428)" }, "graph": { "function": "main", "inputs": { "packed": "[1, T, 512] float32: [speaker cache | FIFO | chunk rows (+ look-ahead)] left-packed at rows [0, L); rows [L, T) are ignored (zero them)", "valid": "[1, T] float32: 1.0 for rows < L, 0.0 after" }, "outputs": { "logits": "[1, T*8, 8] float32 speaker logits at 10 ms; rows [0, L*8) are real" }, "T": { "streaming": 541, "offline": 684 }, "host_side": "sigmoid, the 8x average pool to encoder rate, the speaker cache (AOSC + FIFO), segmentation" }, "host": { "frame_s": 0.01, "encoder_frame_s": 0.08, "subsampling": 8, "num_speakers": 8, "hidden": 512, "profiles": { "streaming": { "T": 541, "fifo_length": 264, "update_period": 222, "modes": { "low_latency": { "chunk": 9, "lookahead": 4 }, "very_low_latency": { "chunk": 6, "lookahead": 2 }, "ultra_low_latency": { "chunk": 3, "lookahead": 1 } }, "chunking": "per chunk mel from its own audio slice: chunk 0 = audio[:((c+la)*8-1)*160+200] center=True; chunk k>=1 starts at 160*k*c*8-256 and holds (c+la)*8*160+400 samples, center=False; the first chunk that would run past the audio takes the rest, is the last (no look-ahead) and emits its true mel frame count" }, "offline": { "T": 684, "fifo_length": 40, "update_period": 300, "chunk": 340, "lookahead": 40, "chunking": "whole-recording mel (center=True) -> embeds -> chunks of 340 rows + up to 40 look-ahead rows; output cut to the mel frame count" } }, "step": "rows = [cache | FIFO | chunk]; logits = graph(rows)[:L*8]; probs = avg_pool8(sigmoid(logits)); emit logits[(n_cache+n_fifo)*8 : +min(n_chunk*8, n_emit)]; then the cache update", "speaker_cache": { "length": 264, "silence_frames_per_speaker": 1, "prediction_score_threshold": 0.25, "latest_frames_score_boost": 0.05, "min_positive_scores": 16, "strong_boost": { "frames": 24, "add": 1.3862943611198906 }, "weak_boost": { "frames": 48, "add": 0.6931471805599453 }, "update": "transformers Nemotron3DiarizationSpeakerCache.update/_compress @ 4b28d51, line by line (host_loop.py)", "score_dtype": "float64 from the float32 probabilities (transformers: float32; see host_loop.py)", "topk_ties": "lower index first", "sentinel": "the silence row (row N: silence_embeds, zero probs)" }, "segments": "transformers extract_speaker_dict: per speaker, runs of sigmoid(logits) > 0.5; overlaps kept" }, "mel": { "sample_rate": 16000, "preemphasis": 0.9700000286102295, "preemphasis_first_sample": "kept", "n_fft": 512, "win_length": 400, "hop": 160, "window": "hann_window_400.f32le, centered in 512 (zeros at [0,56) and [456,512))", "power": "|rFFT|^2 (as sqrt(re^2+im^2)^2 in float32)", "n_mels": 128, "fmin": 0.0, "fmax": 8000.0, "norm": "slaney", "log": "log(mel + 2^-24)", "log_guard": 5.960464477539063e-08, "normalize": null, "center_pad": "256 zeros each side when center=True (first chunk / offline), none otherwise", "valid_frames": "center: n // 160; not center: (n - 512) // 160 + 1", "stacking": "8 frames -> 1024 (zero-pad the frame count to a multiple of 8 in the log-mel domain)" }, "assets": { "embedder_projection.f32le": { "shape": [ 512, 1024 ], "what": "model.audio_tower.embedder.projection.weight (Linear 1024->512, no bias); embeds = stacked @ W.T", "sha256": "add02ccfa230d319e82b55a49e6e2b0ae3d01fb326621b10e04e5aa5dd03d651", "dtype": "float32 little-endian, C order" }, "silence_embeds.f32le": { "shape": [ 512 ], "what": "silence_embeds: the speaker cache's silence row (1 per speaker)", "sha256": "d4417b3c0eabdf7c47032fac2b5b5a7ee83d819a6ddda8fd8eaf74e2b5cc4ac7", "dtype": "float32 little-endian, C order" }, "mel_filters_128x257.f32le": { "shape": [ 128, 257 ], "what": "librosa.filters.mel(sr=16000, n_fft=512, n_mels=128, fmin=0, fmax=8000, norm='slaney'); bit-identical to the transformers feature extractor's", "sha256": "bce5ec5f194a5913f6508cee5a85512e7bad2352db8fc28f5c6ff75af8b09137", "dtype": "float32 little-endian, C order" }, "hann_window_400.f32le": { "shape": [ 400 ], "what": "torch.hann_window(400, periodic=False); bit-identical to the feature extractor's", "sha256": "c427e2029118cf789649e5a4d439b6115d0dd0cbf95dcd22f65e3c848add8c5b", "dtype": "float32 little-endian, C order" } }, "bundles": { "n3d_streaming_float16.aimodel": { "profile": "streaming", "T": 541, "dtype": "float16", "target": "macOS 27, Apple silicon GPU (compiled at load)", "compute_preference": "gpu", "bytes": 197982173, "mb": 198.0, "tree_sha256": "52563c95a40394c0b587e1550495b12bbb424dfcbfb5983a857f38db237d38f0", "contract": { "function": "main", "inputs": { "packed": { "shape": [ 1, 541, 512 ], "dtype": "float32" }, "valid": { "shape": [ 1, 541 ], "dtype": "float32" } }, "outputs": { "logits": { "shape": [ 1, 4328, 8 ], "dtype": "float32" } }, "states": [] } }, "n3d_streaming_float16.h18p.aimodelc": { "profile": "streaming", "T": 541, "dtype": "float16", "target": "iOS 27, iPhone 17 Pro GPU (compiled ahead of time, h18p)", "compute_preference": "gpu", "bytes": 198238645, "mb": 198.2, "tree_sha256": "1e2840464bc246ab7508e1d5e4bf1b6a29e2f55159ff8fabf65c609303ccdb2d", "compiled_from": "n3d_streaming_float16.aimodel", "contract": "that of n3d_streaming_float16.aimodel", "aot": { "command": "xcrun coreai-build compile n3d_streaming_float16.aimodel --output