mlboydaisuke's picture
Add files using upload-large-folder tool
e919ff6 verified
Raw History Blame Contribute Delete
8.29 kB
{
"model": "Nemotron-3-Diarization (streaming Sortformer, 8 speakers)",
"license": "openmdw-1.1",
"source": {
"repo": "nvidia/Nemotron-3-Diarization",
"revision": "f667ed73aee57d40cc39428eb768b4fd87a0a29e",
"file": "model.safetensors",
"sha256": "c074d86335b3b794f8fa5edc25594558f128bdb3914d27806a3a5a2e44963cb6",
"bytes": 396954592,
"reference": "transformers Nemotron3DiarizationForAudioFrameClassification (fp32)"
},
"toolchain": {
"coreai-torch": "0.4.1",
"coreai-core": "1.0.0b2",
"torch": "2.9.0",
"macos": "27.0 (26A428)"
},
"graph": {
"function": "main",
"inputs": {
"packed": "[1, T, 512] float32: [speaker cache | FIFO | chunk rows (+ look-ahead)] left-packed at rows [0, L); rows [L, T) are ignored (zero them)",
"valid": "[1, T] float32: 1.0 for rows < L, 0.0 after"
},
"outputs": {
"logits": "[1, T*8, 8] float32 speaker logits at 10 ms; rows [0, L*8) are real"
},
"T": {
"streaming": 541,
"offline": 684
},
"host_side": "sigmoid, the 8x average pool to encoder rate, the speaker cache (AOSC + FIFO), segmentation"
},
"host": {
"frame_s": 0.01,
"encoder_frame_s": 0.08,
"subsampling": 8,
"num_speakers": 8,
"hidden": 512,
"profiles": {
"streaming": {
"T": 541,
"fifo_length": 264,
"update_period": 222,
"modes": {
"low_latency": {
"chunk": 9,
"lookahead": 4
},
"very_low_latency": {
"chunk": 6,
"lookahead": 2
},
"ultra_low_latency": {
"chunk": 3,
"lookahead": 1
}
},
"chunking": "per chunk mel from its own audio slice: chunk 0 = audio[:((c+la)*8-1)*160+200] center=True; chunk k>=1 starts at 160*k*c*8-256 and holds (c+la)*8*160+400 samples, center=False; the first chunk that would run past the audio takes the rest, is the last (no look-ahead) and emits its true mel frame count"
},
"offline": {
"T": 684,
"fifo_length": 40,
"update_period": 300,
"chunk": 340,
"lookahead": 40,
"chunking": "whole-recording mel (center=True) -> embeds -> chunks of 340 rows + up to 40 look-ahead rows; output cut to the mel frame count"
}
},
"step": "rows = [cache | FIFO | chunk]; logits = graph(rows)[:L*8]; probs = avg_pool8(sigmoid(logits)); emit logits[(n_cache+n_fifo)*8 : +min(n_chunk*8, n_emit)]; then the cache update",
"speaker_cache": {
"length": 264,
"silence_frames_per_speaker": 1,
"prediction_score_threshold": 0.25,
"latest_frames_score_boost": 0.05,
"min_positive_scores": 16,
"strong_boost": {
"frames": 24,
"add": 1.3862943611198906
},
"weak_boost": {
"frames": 48,
"add": 0.6931471805599453
},
"update": "transformers Nemotron3DiarizationSpeakerCache.update/_compress @ 4b28d51, line by line (host_loop.py)",
"score_dtype": "float64 from the float32 probabilities (transformers: float32; see host_loop.py)",
"topk_ties": "lower index first",
"sentinel": "the silence row (row N: silence_embeds, zero probs)"
},
"segments": "transformers extract_speaker_dict: per speaker, runs of sigmoid(logits) > 0.5; overlaps kept"
},
"mel": {
"sample_rate": 16000,
"preemphasis": 0.9700000286102295,
"preemphasis_first_sample": "kept",
"n_fft": 512,
"win_length": 400,
"hop": 160,
"window": "hann_window_400.f32le, centered in 512 (zeros at [0,56) and [456,512))",
"power": "|rFFT|^2 (as sqrt(re^2+im^2)^2 in float32)",
"n_mels": 128,
"fmin": 0.0,
"fmax": 8000.0,
"norm": "slaney",
"log": "log(mel + 2^-24)",
"log_guard": 5.960464477539063e-08,
"normalize": null,
"center_pad": "256 zeros each side when center=True (first chunk / offline), none otherwise",
"valid_frames": "center: n // 160; not center: (n - 512) // 160 + 1",
"stacking": "8 frames -> 1024 (zero-pad the frame count to a multiple of 8 in the log-mel domain)"
},
"assets": {
"embedder_projection.f32le": {
"shape": [
512,
1024
],
"what": "model.audio_tower.embedder.projection.weight (Linear 1024->512, no bias); embeds = stacked @ W.T",
"sha256": "add02ccfa230d319e82b55a49e6e2b0ae3d01fb326621b10e04e5aa5dd03d651",
"dtype": "float32 little-endian, C order"
},
"silence_embeds.f32le": {
"shape": [
512
],
"what": "silence_embeds: the speaker cache's silence row (1 per speaker)",
"sha256": "d4417b3c0eabdf7c47032fac2b5b5a7ee83d819a6ddda8fd8eaf74e2b5cc4ac7",
"dtype": "float32 little-endian, C order"
},
"mel_filters_128x257.f32le": {
"shape": [
128,
257
],
"what": "librosa.filters.mel(sr=16000, n_fft=512, n_mels=128, fmin=0, fmax=8000, norm='slaney'); bit-identical to the transformers feature extractor's",
"sha256": "bce5ec5f194a5913f6508cee5a85512e7bad2352db8fc28f5c6ff75af8b09137",
"dtype": "float32 little-endian, C order"
},
"hann_window_400.f32le": {
"shape": [
400
],
"what": "torch.hann_window(400, periodic=False); bit-identical to the feature extractor's",
"sha256": "c427e2029118cf789649e5a4d439b6115d0dd0cbf95dcd22f65e3c848add8c5b",
"dtype": "float32 little-endian, C order"
}
},
"bundles": {
"n3d_streaming_float16.aimodel": {
"profile": "streaming",
"T": 541,
"dtype": "float16",
"target": "macOS 27, Apple silicon GPU (compiled at load)",
"compute_preference": "gpu",
"bytes": 197982173,
"mb": 198.0,
"tree_sha256": "52563c95a40394c0b587e1550495b12bbb424dfcbfb5983a857f38db237d38f0",
"contract": {
"function": "main",
"inputs": {
"packed": {
"shape": [
1,
541,
512
],
"dtype": "float32"
},
"valid": {
"shape": [
1,
541
],
"dtype": "float32"
}
},
"outputs": {
"logits": {
"shape": [
1,
4328,
8
],
"dtype": "float32"
}
},
"states": []
}
},
"n3d_streaming_float16.h18p.aimodelc": {
"profile": "streaming",
"T": 541,
"dtype": "float16",
"target": "iOS 27, iPhone 17 Pro GPU (compiled ahead of time, h18p)",
"compute_preference": "gpu",
"bytes": 198238645,
"mb": 198.2,
"tree_sha256": "1e2840464bc246ab7508e1d5e4bf1b6a29e2f55159ff8fabf65c609303ccdb2d",
"compiled_from": "n3d_streaming_float16.aimodel",
"contract": "that of n3d_streaming_float16.aimodel",
"aot": {
"command": "xcrun coreai-build compile n3d_streaming_float16.aimodel --output <dir> --platform iOS --architecture h18p --preferred-compute gpu --min-deployment-version 27.0",
"coreai_build": "3600.83.1",
"ane_regions": 0
}
},
"n3d_offline_float16.aimodel": {
"profile": "offline",
"T": 684,
"dtype": "float16",
"target": "macOS 27, Apple silicon GPU (compiled at load)",
"compute_preference": "gpu",
"bytes": 198018769,
"mb": 198.0,
"tree_sha256": "93c99942a3b311824c90c6ccbdc0042eb259331a12a60ae414040963d001210d",
"contract": {
"function": "main",
"inputs": {
"packed": {
"shape": [
1,
684,
512
],
"dtype": "float32"
},
"valid": {
"shape": [
1,
684
],
"dtype": "float32"
}
},
"outputs": {
"logits": {
"shape": [
1,
5472,
8
],
"dtype": "float32"
}
},
"states": []
}
},
"n3d_offline_float16.h18p.aimodelc": {
"profile": "offline",
"T": 684,
"dtype": "float16",
"target": "iOS 27, iPhone 17 Pro GPU (compiled ahead of time, h18p)",
"compute_preference": "gpu",
"bytes": 198275246,
"mb": 198.3,
"tree_sha256": "8494215fba0bddc0d3a6b99673664ce6826d6e47882bdfde744abd7506be10ff",
"compiled_from": "n3d_offline_float16.aimodel",
"contract": "that of n3d_offline_float16.aimodel",
"aot": {
"command": "xcrun coreai-build compile n3d_offline_float16.aimodel --output <dir> --platform iOS --architecture h18p --preferred-compute gpu --min-deployment-version 27.0",
"coreai_build": "3600.83.1",
"ane_regions": 0
}
}
},
"notes": [
"compute_preference: load with SpecializationOptions(preferredComputeUnitKind: .gpu); the .h18p.aimodelc bundles were also compiled with --preferred-compute gpu.",
"The float32 bundles are for parity only (CPU) and are not shipped: export_n3d.py --dtype float32 [--profile offline] rebuilds them."
]
}