Download metadata.json from mlboydaisuke/Nemotron-3-Diarization-CoreAI: direct link, hf CLI and curl.
- Browser
- Download file 8.29 kB
-
https://huggingface.co/mlboydaisuke/Nemotron-3-Diarization-CoreAI/resolve/main/metadata.json
- Command line
-
hf download hf://mlboydaisuke/Nemotron-3-Diarization-CoreAI/metadata.json
-
curl -L -o metadata.json https://huggingface.co/mlboydaisuke/Nemotron-3-Diarization-CoreAI/resolve/main/metadata.json
8.29 kB
| { | |
| "model": "Nemotron-3-Diarization (streaming Sortformer, 8 speakers)", | |
| "license": "openmdw-1.1", | |
| "source": { | |
| "repo": "nvidia/Nemotron-3-Diarization", | |
| "revision": "f667ed73aee57d40cc39428eb768b4fd87a0a29e", | |
| "file": "model.safetensors", | |
| "sha256": "c074d86335b3b794f8fa5edc25594558f128bdb3914d27806a3a5a2e44963cb6", | |
| "bytes": 396954592, | |
| "reference": "transformers Nemotron3DiarizationForAudioFrameClassification (fp32)" | |
| }, | |
| "toolchain": { | |
| "coreai-torch": "0.4.1", | |
| "coreai-core": "1.0.0b2", | |
| "torch": "2.9.0", | |
| "macos": "27.0 (26A428)" | |
| }, | |
| "graph": { | |
| "function": "main", | |
| "inputs": { | |
| "packed": "[1, T, 512] float32: [speaker cache | FIFO | chunk rows (+ look-ahead)] left-packed at rows [0, L); rows [L, T) are ignored (zero them)", | |
| "valid": "[1, T] float32: 1.0 for rows < L, 0.0 after" | |
| }, | |
| "outputs": { | |
| "logits": "[1, T*8, 8] float32 speaker logits at 10 ms; rows [0, L*8) are real" | |
| }, | |
| "T": { | |
| "streaming": 541, | |
| "offline": 684 | |
| }, | |
| "host_side": "sigmoid, the 8x average pool to encoder rate, the speaker cache (AOSC + FIFO), segmentation" | |
| }, | |
| "host": { | |
| "frame_s": 0.01, | |
| "encoder_frame_s": 0.08, | |
| "subsampling": 8, | |
| "num_speakers": 8, | |
| "hidden": 512, | |
| "profiles": { | |
| "streaming": { | |
| "T": 541, | |
| "fifo_length": 264, | |
| "update_period": 222, | |
| "modes": { | |
| "low_latency": { | |
| "chunk": 9, | |
| "lookahead": 4 | |
| }, | |
| "very_low_latency": { | |
| "chunk": 6, | |
| "lookahead": 2 | |
| }, | |
| "ultra_low_latency": { | |
| "chunk": 3, | |
| "lookahead": 1 | |
| } | |
| }, | |
| "chunking": "per chunk mel from its own audio slice: chunk 0 = audio[:((c+la)*8-1)*160+200] center=True; chunk k>=1 starts at 160*k*c*8-256 and holds (c+la)*8*160+400 samples, center=False; the first chunk that would run past the audio takes the rest, is the last (no look-ahead) and emits its true mel frame count" | |
| }, | |
| "offline": { | |
| "T": 684, | |
| "fifo_length": 40, | |
| "update_period": 300, | |
| "chunk": 340, | |
| "lookahead": 40, | |
| "chunking": "whole-recording mel (center=True) -> embeds -> chunks of 340 rows + up to 40 look-ahead rows; output cut to the mel frame count" | |
| } | |
| }, | |
| "step": "rows = [cache | FIFO | chunk]; logits = graph(rows)[:L*8]; probs = avg_pool8(sigmoid(logits)); emit logits[(n_cache+n_fifo)*8 : +min(n_chunk*8, n_emit)]; then the cache update", | |
| "speaker_cache": { | |
| "length": 264, | |
| "silence_frames_per_speaker": 1, | |
| "prediction_score_threshold": 0.25, | |
| "latest_frames_score_boost": 0.05, | |
| "min_positive_scores": 16, | |
| "strong_boost": { | |
| "frames": 24, | |
| "add": 1.3862943611198906 | |
| }, | |
| "weak_boost": { | |
| "frames": 48, | |
| "add": 0.6931471805599453 | |
| }, | |
| "update": "transformers Nemotron3DiarizationSpeakerCache.update/_compress @ 4b28d51, line by line (host_loop.py)", | |
| "score_dtype": "float64 from the float32 probabilities (transformers: float32; see host_loop.py)", | |
| "topk_ties": "lower index first", | |
| "sentinel": "the silence row (row N: silence_embeds, zero probs)" | |
| }, | |
| "segments": "transformers extract_speaker_dict: per speaker, runs of sigmoid(logits) > 0.5; overlaps kept" | |
| }, | |
| "mel": { | |
| "sample_rate": 16000, | |
| "preemphasis": 0.9700000286102295, | |
| "preemphasis_first_sample": "kept", | |
| "n_fft": 512, | |
| "win_length": 400, | |
| "hop": 160, | |
| "window": "hann_window_400.f32le, centered in 512 (zeros at [0,56) and [456,512))", | |
| "power": "|rFFT|^2 (as sqrt(re^2+im^2)^2 in float32)", | |
| "n_mels": 128, | |
| "fmin": 0.0, | |
| "fmax": 8000.0, | |
| "norm": "slaney", | |
| "log": "log(mel + 2^-24)", | |
| "log_guard": 5.960464477539063e-08, | |
| "normalize": null, | |
| "center_pad": "256 zeros each side when center=True (first chunk / offline), none otherwise", | |
| "valid_frames": "center: n // 160; not center: (n - 512) // 160 + 1", | |
| "stacking": "8 frames -> 1024 (zero-pad the frame count to a multiple of 8 in the log-mel domain)" | |
| }, | |
| "assets": { | |
| "embedder_projection.f32le": { | |
| "shape": [ | |
| 512, | |
| 1024 | |
| ], | |
| "what": "model.audio_tower.embedder.projection.weight (Linear 1024->512, no bias); embeds = stacked @ W.T", | |
| "sha256": "add02ccfa230d319e82b55a49e6e2b0ae3d01fb326621b10e04e5aa5dd03d651", | |
| "dtype": "float32 little-endian, C order" | |
| }, | |
| "silence_embeds.f32le": { | |
| "shape": [ | |
| 512 | |
| ], | |
| "what": "silence_embeds: the speaker cache's silence row (1 per speaker)", | |
| "sha256": "d4417b3c0eabdf7c47032fac2b5b5a7ee83d819a6ddda8fd8eaf74e2b5cc4ac7", | |
| "dtype": "float32 little-endian, C order" | |
| }, | |
| "mel_filters_128x257.f32le": { | |
| "shape": [ | |
| 128, | |
| 257 | |
| ], | |
| "what": "librosa.filters.mel(sr=16000, n_fft=512, n_mels=128, fmin=0, fmax=8000, norm='slaney'); bit-identical to the transformers feature extractor's", | |
| "sha256": "bce5ec5f194a5913f6508cee5a85512e7bad2352db8fc28f5c6ff75af8b09137", | |
| "dtype": "float32 little-endian, C order" | |
| }, | |
| "hann_window_400.f32le": { | |
| "shape": [ | |
| 400 | |
| ], | |
| "what": "torch.hann_window(400, periodic=False); bit-identical to the feature extractor's", | |
| "sha256": "c427e2029118cf789649e5a4d439b6115d0dd0cbf95dcd22f65e3c848add8c5b", | |
| "dtype": "float32 little-endian, C order" | |
| } | |
| }, | |
| "bundles": { | |
| "n3d_streaming_float16.aimodel": { | |
| "profile": "streaming", | |
| "T": 541, | |
| "dtype": "float16", | |
| "target": "macOS 27, Apple silicon GPU (compiled at load)", | |
| "compute_preference": "gpu", | |
| "bytes": 197982173, | |
| "mb": 198.0, | |
| "tree_sha256": "52563c95a40394c0b587e1550495b12bbb424dfcbfb5983a857f38db237d38f0", | |
| "contract": { | |
| "function": "main", | |
| "inputs": { | |
| "packed": { | |
| "shape": [ | |
| 1, | |
| 541, | |
| 512 | |
| ], | |
| "dtype": "float32" | |
| }, | |
| "valid": { | |
| "shape": [ | |
| 1, | |
| 541 | |
| ], | |
| "dtype": "float32" | |
| } | |
| }, | |
| "outputs": { | |
| "logits": { | |
| "shape": [ | |
| 1, | |
| 4328, | |
| 8 | |
| ], | |
| "dtype": "float32" | |
| } | |
| }, | |
| "states": [] | |
| } | |
| }, | |
| "n3d_streaming_float16.h18p.aimodelc": { | |
| "profile": "streaming", | |
| "T": 541, | |
| "dtype": "float16", | |
| "target": "iOS 27, iPhone 17 Pro GPU (compiled ahead of time, h18p)", | |
| "compute_preference": "gpu", | |
| "bytes": 198238645, | |
| "mb": 198.2, | |
| "tree_sha256": "1e2840464bc246ab7508e1d5e4bf1b6a29e2f55159ff8fabf65c609303ccdb2d", | |
| "compiled_from": "n3d_streaming_float16.aimodel", | |
| "contract": "that of n3d_streaming_float16.aimodel", | |
| "aot": { | |
| "command": "xcrun coreai-build compile n3d_streaming_float16.aimodel --output <dir> --platform iOS --architecture h18p --preferred-compute gpu --min-deployment-version 27.0", | |
| "coreai_build": "3600.83.1", | |
| "ane_regions": 0 | |
| } | |
| }, | |
| "n3d_offline_float16.aimodel": { | |
| "profile": "offline", | |
| "T": 684, | |
| "dtype": "float16", | |
| "target": "macOS 27, Apple silicon GPU (compiled at load)", | |
| "compute_preference": "gpu", | |
| "bytes": 198018769, | |
| "mb": 198.0, | |
| "tree_sha256": "93c99942a3b311824c90c6ccbdc0042eb259331a12a60ae414040963d001210d", | |
| "contract": { | |
| "function": "main", | |
| "inputs": { | |
| "packed": { | |
| "shape": [ | |
| 1, | |
| 684, | |
| 512 | |
| ], | |
| "dtype": "float32" | |
| }, | |
| "valid": { | |
| "shape": [ | |
| 1, | |
| 684 | |
| ], | |
| "dtype": "float32" | |
| } | |
| }, | |
| "outputs": { | |
| "logits": { | |
| "shape": [ | |
| 1, | |
| 5472, | |
| 8 | |
| ], | |
| "dtype": "float32" | |
| } | |
| }, | |
| "states": [] | |
| } | |
| }, | |
| "n3d_offline_float16.h18p.aimodelc": { | |
| "profile": "offline", | |
| "T": 684, | |
| "dtype": "float16", | |
| "target": "iOS 27, iPhone 17 Pro GPU (compiled ahead of time, h18p)", | |
| "compute_preference": "gpu", | |
| "bytes": 198275246, | |
| "mb": 198.3, | |
| "tree_sha256": "8494215fba0bddc0d3a6b99673664ce6826d6e47882bdfde744abd7506be10ff", | |
| "compiled_from": "n3d_offline_float16.aimodel", | |
| "contract": "that of n3d_offline_float16.aimodel", | |
| "aot": { | |
| "command": "xcrun coreai-build compile n3d_offline_float16.aimodel --output <dir> --platform iOS --architecture h18p --preferred-compute gpu --min-deployment-version 27.0", | |
| "coreai_build": "3600.83.1", | |
| "ane_regions": 0 | |
| } | |
| } | |
| }, | |
| "notes": [ | |
| "compute_preference: load with SpecializationOptions(preferredComputeUnitKind: .gpu); the .h18p.aimodelc bundles were also compiled with --preferred-compute gpu.", | |
| "The float32 bundles are for parity only (CPU) and are not shipped: export_n3d.py --dtype float32 [--profile offline] rebuilds them." | |
| ] | |
| } | |