Automatic Speech Recognition
WhisperKit
Core ML
speakerkit
pyannote
diarization
speaker-diarization
whisper
asr
quantized
Instructions to use argmaxinc/speakerkit-pro with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- WhisperKit
How to use argmaxinc/speakerkit-pro with WhisperKit:
# Install CLI with Homebrew on macOS device brew install whisperkit-cli # View all available inference options whisperkit-cli transcribe --help # Download and run inference using whisper base model whisperkit-cli transcribe --audio-path /path/to/audio.mp3 # Or use your preferred model variant whisperkit-cli transcribe --model "large-v3" --model-prefix "distil" --audio-path /path/to/audio.mp3 --verbose
- Notebooks
- Google Colab
- Kaggle
Upload nemotron-3-diarization
#6
by arda-argmax - opened
- sortformer/nemotron-3-diarization/684_74MB/AudioConformerPreEncoder.mlmodelc/analytics/coremldata.bin +3 -0
- sortformer/nemotron-3-diarization/684_74MB/AudioConformerPreEncoder.mlmodelc/coremldata.bin +3 -0
- sortformer/nemotron-3-diarization/684_74MB/AudioConformerPreEncoder.mlmodelc/metadata.json +77 -0
- sortformer/nemotron-3-diarization/684_74MB/AudioConformerPreEncoder.mlmodelc/model.mil +35 -0
- sortformer/nemotron-3-diarization/684_74MB/AudioConformerPreEncoder.mlmodelc/weights/weight.bin +3 -0
- sortformer/nemotron-3-diarization/684_74MB/LICENSE-OpenMDW-1.1.txt +19 -0
- sortformer/nemotron-3-diarization/684_74MB/LICENSE_NOTICE.txt +7 -0
- sortformer/nemotron-3-diarization/684_74MB/MelSpectrogram.mlmodelc/analytics/coremldata.bin +3 -0
- sortformer/nemotron-3-diarization/684_74MB/MelSpectrogram.mlmodelc/coremldata.bin +3 -0
- sortformer/nemotron-3-diarization/684_74MB/MelSpectrogram.mlmodelc/metadata.json +76 -0
- sortformer/nemotron-3-diarization/684_74MB/MelSpectrogram.mlmodelc/model.mil +75 -0
- sortformer/nemotron-3-diarization/684_74MB/MelSpectrogram.mlmodelc/weights/weight.bin +3 -0
- sortformer/nemotron-3-diarization/684_74MB/NOTICE.txt +12 -0
- sortformer/nemotron-3-diarization/684_74MB/SortformerFullEncoder.mlmodelc/analytics/coremldata.bin +3 -0
- sortformer/nemotron-3-diarization/684_74MB/SortformerFullEncoder.mlmodelc/coremldata.bin +3 -0
- sortformer/nemotron-3-diarization/684_74MB/SortformerFullEncoder.mlmodelc/metadata.json +138 -0
- sortformer/nemotron-3-diarization/684_74MB/SortformerFullEncoder.mlmodelc/model.mil +0 -0
- sortformer/nemotron-3-diarization/684_74MB/SortformerFullEncoder.mlmodelc/weights/weight.bin +3 -0
sortformer/nemotron-3-diarization/684_74MB/AudioConformerPreEncoder.mlmodelc/analytics/coremldata.bin
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:6b24c23158fd40de42695f7978744f2df0e5a851486c6ebd3d25e2150aba540c
|
| 3 |
+
size 243
|
sortformer/nemotron-3-diarization/684_74MB/AudioConformerPreEncoder.mlmodelc/coremldata.bin
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:c55d8bef97c564412d48e8c4c2358a2ef2dc2504447b8851a814269e6ffecc97
|
| 3 |
+
size 459
|
sortformer/nemotron-3-diarization/684_74MB/AudioConformerPreEncoder.mlmodelc/metadata.json
ADDED
|
@@ -0,0 +1,77 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
[
|
| 2 |
+
{
|
| 3 |
+
"metadataOutputVersion" : "3.0",
|
| 4 |
+
"storagePrecision" : "Float16",
|
| 5 |
+
"outputSchema" : [
|
| 6 |
+
{
|
| 7 |
+
"hasShapeFlexibility" : "0",
|
| 8 |
+
"isOptional" : "0",
|
| 9 |
+
"dataType" : "Float16",
|
| 10 |
+
"formattedType" : "MultiArray (Float16 1 × 512 × 1 × 685)",
|
| 11 |
+
"shortDescription" : "",
|
| 12 |
+
"shape" : "[1, 512, 1, 685]",
|
| 13 |
+
"name" : "downsampled_melspectrogram_features",
|
| 14 |
+
"type" : "MultiArray"
|
| 15 |
+
},
|
| 16 |
+
{
|
| 17 |
+
"hasShapeFlexibility" : "0",
|
| 18 |
+
"isOptional" : "0",
|
| 19 |
+
"dataType" : "Float16",
|
| 20 |
+
"formattedType" : "MultiArray (Float16 1 × 512 × 1 × 1)",
|
| 21 |
+
"shortDescription" : "",
|
| 22 |
+
"shape" : "[1, 512, 1, 1]",
|
| 23 |
+
"name" : "learnable_sil_emb",
|
| 24 |
+
"type" : "MultiArray"
|
| 25 |
+
}
|
| 26 |
+
],
|
| 27 |
+
"modelParameters" : [
|
| 28 |
+
|
| 29 |
+
],
|
| 30 |
+
"specificationVersion" : 9,
|
| 31 |
+
"mlProgramOperationTypeHistogram" : {
|
| 32 |
+
"Ios18.reshape" : 2,
|
| 33 |
+
"Pad" : 1,
|
| 34 |
+
"Ios18.transpose" : 1,
|
| 35 |
+
"Ios18.conv" : 1,
|
| 36 |
+
"Ios18.sliceByIndex" : 2,
|
| 37 |
+
"Ios18.mul" : 1,
|
| 38 |
+
"Ios18.add" : 1
|
| 39 |
+
},
|
| 40 |
+
"computePrecision" : "Mixed (Float16, Int32)",
|
| 41 |
+
"isUpdatable" : "0",
|
| 42 |
+
"stateSchema" : [
|
| 43 |
+
|
| 44 |
+
],
|
| 45 |
+
"availability" : {
|
| 46 |
+
"macOS" : "15.0",
|
| 47 |
+
"tvOS" : "18.0",
|
| 48 |
+
"visionOS" : "2.0",
|
| 49 |
+
"watchOS" : "11.0",
|
| 50 |
+
"iOS" : "18.0",
|
| 51 |
+
"macCatalyst" : "18.0"
|
| 52 |
+
},
|
| 53 |
+
"modelType" : {
|
| 54 |
+
"name" : "MLModelType_mlProgram"
|
| 55 |
+
},
|
| 56 |
+
"userDefinedMetadata" : {
|
| 57 |
+
"com.github.apple.coremltools.conversion_date" : "2026-09-21",
|
| 58 |
+
"com.github.apple.coremltools.source" : "torch==2.8.0",
|
| 59 |
+
"com.github.apple.coremltools.version" : "9.0",
|
| 60 |
+
"com.github.apple.coremltools.source_dialect" : "TorchScript"
|
| 61 |
+
},
|
| 62 |
+
"inputSchema" : [
|
| 63 |
+
{
|
| 64 |
+
"hasShapeFlexibility" : "0",
|
| 65 |
+
"isOptional" : "0",
|
| 66 |
+
"dataType" : "Float16",
|
| 67 |
+
"formattedType" : "MultiArray (Float16 1 × 1 × 5473 × 128)",
|
| 68 |
+
"shortDescription" : "",
|
| 69 |
+
"shape" : "[1, 1, 5473, 128]",
|
| 70 |
+
"name" : "melspectrogram_features",
|
| 71 |
+
"type" : "MultiArray"
|
| 72 |
+
}
|
| 73 |
+
],
|
| 74 |
+
"generatedClassName" : "AudioConformerPreEncoder",
|
| 75 |
+
"method" : "predict"
|
| 76 |
+
}
|
| 77 |
+
]
|
sortformer/nemotron-3-diarization/684_74MB/AudioConformerPreEncoder.mlmodelc/model.mil
ADDED
|
@@ -0,0 +1,35 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
program(1.3)
|
| 2 |
+
[buildInfo = dict<string, string>({{"coremlc-component-MIL", "3600.16.1"}, {"coremlc-version", "3600.25.1"}, {"coremltools-component-torch", "2.8.0"}, {"coremltools-source-dialect", "TorchScript"}, {"coremltools-version", "9.0"}})]
|
| 3 |
+
{
|
| 4 |
+
func main<ios18>(tensor<fp16, [1, 1, 5473, 128]> melspectrogram_features) {
|
| 5 |
+
tensor<int32, [8]> padded_pad_0 = const()[name = string("padded_pad_0"), val = tensor<int32, [8]>([0, 0, 0, 0, 0, 7, 0, 0])];
|
| 6 |
+
string padded_mode_0 = const()[name = string("padded_mode_0"), val = string("constant")];
|
| 7 |
+
fp16 const_0_to_fp16 = const()[name = string("const_0_to_fp16"), val = fp16(0x0p+0)];
|
| 8 |
+
tensor<fp16, [1, 1, 5480, 128]> padded_cast_fp16 = pad(constant_val = const_0_to_fp16, mode = padded_mode_0, pad = padded_pad_0, x = melspectrogram_features)[name = string("padded_cast_fp16")];
|
| 9 |
+
tensor<int32, [4]> var_16 = const()[name = string("op_16"), val = tensor<int32, [4]>([1, 685, 8, 128])];
|
| 10 |
+
tensor<fp16, [1, 685, 8, 128]> grouped_cast_fp16 = reshape(shape = var_16, x = padded_cast_fp16)[name = string("grouped_cast_fp16")];
|
| 11 |
+
tensor<int32, [4]> var_22 = const()[name = string("op_22"), val = tensor<int32, [4]>([0, 2, 3, 1])];
|
| 12 |
+
tensor<int32, [4]> var_28 = const()[name = string("op_28"), val = tensor<int32, [4]>([1, 1024, 1, 685])];
|
| 13 |
+
tensor<fp16, [1, 8, 128, 685]> transposed_cast_fp16 = transpose(perm = var_22, x = grouped_cast_fp16)[name = string("transpose_0")];
|
| 14 |
+
tensor<fp16, [1, 1024, 1, 685]> input_cast_fp16 = reshape(shape = var_28, x = transposed_cast_fp16)[name = string("input_cast_fp16")];
|
| 15 |
+
string embeddings_pad_type_0 = const()[name = string("embeddings_pad_type_0"), val = string("valid")];
|
| 16 |
+
tensor<int32, [2]> embeddings_strides_0 = const()[name = string("embeddings_strides_0"), val = tensor<int32, [2]>([1, 1])];
|
| 17 |
+
tensor<int32, [4]> embeddings_pad_0 = const()[name = string("embeddings_pad_0"), val = tensor<int32, [4]>([0, 0, 0, 0])];
|
| 18 |
+
tensor<int32, [2]> embeddings_dilations_0 = const()[name = string("embeddings_dilations_0"), val = tensor<int32, [2]>([1, 1])];
|
| 19 |
+
int32 embeddings_groups_0 = const()[name = string("embeddings_groups_0"), val = int32(1)];
|
| 20 |
+
tensor<fp16, [512, 1024, 1, 1]> proj_weight_to_fp16 = const()[name = string("proj_weight_to_fp16"), val = tensor<fp16, [512, 1024, 1, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(64)))];
|
| 21 |
+
tensor<fp16, [1, 512, 1, 685]> downsampled_melspectrogram_features = conv(dilations = embeddings_dilations_0, groups = embeddings_groups_0, pad = embeddings_pad_0, pad_type = embeddings_pad_type_0, strides = embeddings_strides_0, weight = proj_weight_to_fp16, x = input_cast_fp16)[name = string("embeddings_cast_fp16")];
|
| 22 |
+
tensor<int32, [4]> var_50_begin_0 = const()[name = string("op_50_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 0])];
|
| 23 |
+
tensor<int32, [4]> var_50_end_0 = const()[name = string("op_50_end_0"), val = tensor<int32, [4]>([1, 1, 1, 685])];
|
| 24 |
+
tensor<bool, [4]> var_50_end_mask_0 = const()[name = string("op_50_end_mask_0"), val = tensor<bool, [4]>([true, false, true, true])];
|
| 25 |
+
tensor<fp16, [1, 1, 1, 685]> var_50_cast_fp16 = slice_by_index(begin = var_50_begin_0, end = var_50_end_0, end_mask = var_50_end_mask_0, x = downsampled_melspectrogram_features)[name = string("op_50_cast_fp16")];
|
| 26 |
+
tensor<int32, [4]> var_60_begin_0 = const()[name = string("op_60_begin_0"), val = tensor<int32, [4]>([0, 0, 0, 0])];
|
| 27 |
+
tensor<int32, [4]> var_60_end_0 = const()[name = string("op_60_end_0"), val = tensor<int32, [4]>([1, 1, 1, 1])];
|
| 28 |
+
tensor<bool, [4]> var_60_end_mask_0 = const()[name = string("op_60_end_mask_0"), val = tensor<bool, [4]>([true, true, true, false])];
|
| 29 |
+
tensor<fp16, [1, 1, 1, 1]> var_60_cast_fp16 = slice_by_index(begin = var_60_begin_0, end = var_60_end_0, end_mask = var_60_end_mask_0, x = var_50_cast_fp16)[name = string("op_60_cast_fp16")];
|
| 30 |
+
fp16 var_61_to_fp16 = const()[name = string("op_61_to_fp16"), val = fp16(0x0p+0)];
|
| 31 |
+
tensor<fp16, [1, 1, 1, 1]> anchor_cast_fp16 = mul(x = var_60_cast_fp16, y = var_61_to_fp16)[name = string("anchor_cast_fp16")];
|
| 32 |
+
tensor<fp16, [1, 512, 1, 1]> silence_embedding_to_fp16 = const()[name = string("silence_embedding_to_fp16"), val = tensor<fp16, [1, 512, 1, 1]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1048704)))];
|
| 33 |
+
tensor<fp16, [1, 512, 1, 1]> learnable_sil_emb = add(x = silence_embedding_to_fp16, y = anchor_cast_fp16)[name = string("op_64_cast_fp16")];
|
| 34 |
+
} -> (downsampled_melspectrogram_features, learnable_sil_emb);
|
| 35 |
+
}
|
sortformer/nemotron-3-diarization/684_74MB/AudioConformerPreEncoder.mlmodelc/weights/weight.bin
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:5e9389bb9150edcaebef5705d30ea3e6f8f86f28dd2ff62165ff54b3c8ad9128
|
| 3 |
+
size 1049792
|
sortformer/nemotron-3-diarization/684_74MB/LICENSE-OpenMDW-1.1.txt
ADDED
|
@@ -0,0 +1,19 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
OpenMDW License Agreement, version 1.1 (OpenMDW-1.1)
|
| 2 |
+
|
| 3 |
+
By exercising rights granted to you under this agreement, you accept and agree to its terms.
|
| 4 |
+
|
| 5 |
+
As used in this agreement, “Model Materials” means the materials provided to you under this agreement, consisting of: (1) one or more machine learning models (including architecture and parameters); and (2) all related artifacts (including associated data, documentation and software) that are provided to you hereunder.
|
| 6 |
+
|
| 7 |
+
Subject to your compliance with this agreement, permission is hereby granted, free of charge, to deal in the Model Materials without restriction, including under all copyright, patent, database, and trade secret rights included or embodied therein.
|
| 8 |
+
|
| 9 |
+
If you distribute any portion of the Model Materials, you shall retain in your distribution (1) a copy of this agreement, and (2) all copyright notices and other notices of origin included in the Model Materials that are applicable to your distribution.
|
| 10 |
+
|
| 11 |
+
If you file, maintain, or voluntarily participate in a lawsuit against any person or entity asserting that the Model Materials directly or indirectly infringe any patent or copyright, then all rights and grants made to you hereunder are terminated, unless that lawsuit was in response to a corresponding lawsuit first brought against you.
|
| 12 |
+
|
| 13 |
+
This agreement does not impose any restrictions or obligations with respect to any use, modification, or sharing of any outputs generated by using the Model Materials.
|
| 14 |
+
|
| 15 |
+
THE MODEL MATERIALS ARE PROVIDED “AS IS”, WITHOUT WARRANTY OF ANY KIND, EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE, TITLE, NONINFRINGEMENT, ACCURACY, OR THE ABSENCE OF LATENT OR OTHER DEFECTS OR ERRORS, WHETHER OR NOT DISCOVERABLE, ALL TO THE GREATEST EXTENT PERMISSIBLE UNDER APPLICABLE LAW.
|
| 16 |
+
|
| 17 |
+
YOU ARE SOLELY RESPONSIBLE FOR (1) CLEARING RIGHTS OF OTHER PERSONS THAT MAY APPLY TO THE MODEL MATERIALS OR ANY USE THEREOF, INCLUDING WITHOUT LIMITATION ANY PERSON’S COPYRIGHTS OR OTHER RIGHTS INCLUDED OR EMBODIED IN THE MODEL MATERIALS; (2) OBTAINING ANY NECESSARY CONSENTS, PERMISSIONS OR OTHER RIGHTS REQUIRED FOR ANY USE OF THE MODEL MATERIALS; OR (3) PERFORMING ANY DUE DILIGENCE OR UNDERTAKING ANY OTHER INVESTIGATIONS INTO THE MODEL MATERIALS OR ANYTHING INCORPORATED OR EMBODIED THEREIN.
|
| 18 |
+
|
| 19 |
+
IN NO EVENT SHALL THE PROVIDERS OF THE MODEL MATERIALS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE MODEL MATERIALS, THE USE THEREOF OR OTHER DEALINGS THEREIN.
|
sortformer/nemotron-3-diarization/684_74MB/LICENSE_NOTICE.txt
ADDED
|
@@ -0,0 +1,7 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
Argmax proprietary and confidential. Under NDA.
|
| 2 |
+
|
| 3 |
+
Copyright 2026 Argmax, Inc. All rights reserved.
|
| 4 |
+
|
| 5 |
+
Unauthorized access, copying, use, distribution, and or commercialization of this file, via any medium or means is strictly prohibited.
|
| 6 |
+
|
| 7 |
+
Please contact Argmax for licensing information at info@argmaxinc.com.
|
sortformer/nemotron-3-diarization/684_74MB/MelSpectrogram.mlmodelc/analytics/coremldata.bin
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:d9a01852976442093092bd56cfebf6f3b95184ead3ee93c64e3550e03c421ea5
|
| 3 |
+
size 243
|
sortformer/nemotron-3-diarization/684_74MB/MelSpectrogram.mlmodelc/coremldata.bin
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:eee895cb3ec071917f4848ee4424e310fcd3844bfb3bceab6b31ddd2807ac93e
|
| 3 |
+
size 390
|
sortformer/nemotron-3-diarization/684_74MB/MelSpectrogram.mlmodelc/metadata.json
ADDED
|
@@ -0,0 +1,76 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
[
|
| 2 |
+
{
|
| 3 |
+
"metadataOutputVersion" : "3.0",
|
| 4 |
+
"storagePrecision" : "Float32",
|
| 5 |
+
"outputSchema" : [
|
| 6 |
+
{
|
| 7 |
+
"hasShapeFlexibility" : "0",
|
| 8 |
+
"isOptional" : "0",
|
| 9 |
+
"dataType" : "Float16",
|
| 10 |
+
"formattedType" : "MultiArray (Float16 1 × 1 × 5473 × 128)",
|
| 11 |
+
"shortDescription" : "",
|
| 12 |
+
"shape" : "[1, 1, 5473, 128]",
|
| 13 |
+
"name" : "melspectrogram_features",
|
| 14 |
+
"type" : "MultiArray"
|
| 15 |
+
}
|
| 16 |
+
],
|
| 17 |
+
"modelParameters" : [
|
| 18 |
+
|
| 19 |
+
],
|
| 20 |
+
"specificationVersion" : 9,
|
| 21 |
+
"mlProgramOperationTypeHistogram" : {
|
| 22 |
+
"Ios18.square" : 2,
|
| 23 |
+
"Ios18.mul" : 1,
|
| 24 |
+
"Ios18.conv" : 2,
|
| 25 |
+
"Ios18.expandDims" : 4,
|
| 26 |
+
"Ios18.sub" : 1,
|
| 27 |
+
"Ios18.matmul" : 1,
|
| 28 |
+
"Ios18.log" : 1,
|
| 29 |
+
"Ios18.concat" : 1,
|
| 30 |
+
"Ios18.add" : 2,
|
| 31 |
+
"Ios18.sliceByIndex" : 3,
|
| 32 |
+
"Ios18.cast" : 2,
|
| 33 |
+
"Ios18.transpose" : 1,
|
| 34 |
+
"Ios18.squeeze" : 2,
|
| 35 |
+
"Ios18.reshape" : 2,
|
| 36 |
+
"Identity" : 1,
|
| 37 |
+
"Pad" : 1
|
| 38 |
+
},
|
| 39 |
+
"computePrecision" : "Mixed (Float16, Float32, Int32)",
|
| 40 |
+
"isUpdatable" : "0",
|
| 41 |
+
"stateSchema" : [
|
| 42 |
+
|
| 43 |
+
],
|
| 44 |
+
"availability" : {
|
| 45 |
+
"macOS" : "15.0",
|
| 46 |
+
"tvOS" : "18.0",
|
| 47 |
+
"visionOS" : "2.0",
|
| 48 |
+
"watchOS" : "11.0",
|
| 49 |
+
"iOS" : "18.0",
|
| 50 |
+
"macCatalyst" : "18.0"
|
| 51 |
+
},
|
| 52 |
+
"modelType" : {
|
| 53 |
+
"name" : "MLModelType_mlProgram"
|
| 54 |
+
},
|
| 55 |
+
"userDefinedMetadata" : {
|
| 56 |
+
"com.github.apple.coremltools.conversion_date" : "2026-08-28",
|
| 57 |
+
"com.github.apple.coremltools.source" : "torch==2.8.0",
|
| 58 |
+
"com.github.apple.coremltools.version" : "9.0",
|
| 59 |
+
"com.github.apple.coremltools.source_dialect" : "TorchScript"
|
| 60 |
+
},
|
| 61 |
+
"inputSchema" : [
|
| 62 |
+
{
|
| 63 |
+
"hasShapeFlexibility" : "0",
|
| 64 |
+
"isOptional" : "0",
|
| 65 |
+
"dataType" : "Float16",
|
| 66 |
+
"formattedType" : "MultiArray (Float16 875520)",
|
| 67 |
+
"shortDescription" : "",
|
| 68 |
+
"shape" : "[875520]",
|
| 69 |
+
"name" : "audio",
|
| 70 |
+
"type" : "MultiArray"
|
| 71 |
+
}
|
| 72 |
+
],
|
| 73 |
+
"generatedClassName" : "MelSpectrogram",
|
| 74 |
+
"method" : "predict"
|
| 75 |
+
}
|
| 76 |
+
]
|
sortformer/nemotron-3-diarization/684_74MB/MelSpectrogram.mlmodelc/model.mil
ADDED
|
@@ -0,0 +1,75 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
program(1.3)
|
| 2 |
+
[buildInfo = dict<string, string>({{"coremlc-component-MIL", "3510.2.1"}, {"coremlc-version", "3500.32.1"}, {"coremltools-component-torch", "2.8.0"}, {"coremltools-source-dialect", "TorchScript"}, {"coremltools-version", "9.0"}})]
|
| 3 |
+
{
|
| 4 |
+
func main<ios18>(tensor<fp16, [875520]> audio) {
|
| 5 |
+
string cast_0_dtype_0 = const()[name = string("cast_0_dtype_0"), val = string("fp32")];
|
| 6 |
+
tensor<fp32, [128, 257]> mel_filters = const()[name = string("mel_filters"), val = tensor<fp32, [128, 257]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(64)))];
|
| 7 |
+
tensor<int32, [1]> var_8_begin_0 = const()[name = string("op_8_begin_0"), val = tensor<int32, [1]>([0])];
|
| 8 |
+
tensor<int32, [1]> var_8_end_0 = const()[name = string("op_8_end_0"), val = tensor<int32, [1]>([1])];
|
| 9 |
+
tensor<bool, [1]> var_8_end_mask_0 = const()[name = string("op_8_end_mask_0"), val = tensor<bool, [1]>([false])];
|
| 10 |
+
tensor<fp32, [875520]> cast_0 = cast(dtype = cast_0_dtype_0, x = audio)[name = string("cast_6")];
|
| 11 |
+
tensor<fp32, [1]> var_8 = slice_by_index(begin = var_8_begin_0, end = var_8_end_0, end_mask = var_8_end_mask_0, x = cast_0)[name = string("op_8")];
|
| 12 |
+
tensor<int32, [1]> var_13_begin_0 = const()[name = string("op_13_begin_0"), val = tensor<int32, [1]>([1])];
|
| 13 |
+
tensor<int32, [1]> var_13_end_0 = const()[name = string("op_13_end_0"), val = tensor<int32, [1]>([875520])];
|
| 14 |
+
tensor<bool, [1]> var_13_end_mask_0 = const()[name = string("op_13_end_mask_0"), val = tensor<bool, [1]>([true])];
|
| 15 |
+
tensor<fp32, [875519]> var_13 = slice_by_index(begin = var_13_begin_0, end = var_13_end_0, end_mask = var_13_end_mask_0, x = cast_0)[name = string("op_13")];
|
| 16 |
+
tensor<int32, [1]> var_18_begin_0 = const()[name = string("op_18_begin_0"), val = tensor<int32, [1]>([0])];
|
| 17 |
+
tensor<int32, [1]> var_18_end_0 = const()[name = string("op_18_end_0"), val = tensor<int32, [1]>([875519])];
|
| 18 |
+
tensor<bool, [1]> var_18_end_mask_0 = const()[name = string("op_18_end_mask_0"), val = tensor<bool, [1]>([false])];
|
| 19 |
+
tensor<fp32, [875519]> var_18 = slice_by_index(begin = var_18_begin_0, end = var_18_end_0, end_mask = var_18_end_mask_0, x = cast_0)[name = string("op_18")];
|
| 20 |
+
fp32 var_19 = const()[name = string("op_19"), val = fp32(0x1.f0a3d8p-1)];
|
| 21 |
+
tensor<fp32, [875519]> var_20 = mul(x = var_18, y = var_19)[name = string("op_20")];
|
| 22 |
+
tensor<fp32, [875519]> var_22 = sub(x = var_13, y = var_20)[name = string("op_22")];
|
| 23 |
+
int32 var_24 = const()[name = string("op_24"), val = int32(0)];
|
| 24 |
+
bool input_1_interleave_0 = const()[name = string("input_1_interleave_0"), val = bool(false)];
|
| 25 |
+
tensor<fp32, [875520]> input_1 = concat(axis = var_24, interleave = input_1_interleave_0, values = (var_8, var_22))[name = string("input_1")];
|
| 26 |
+
tensor<int32, [3]> var_32 = const()[name = string("op_32"), val = tensor<int32, [3]>([1, 1, 875520])];
|
| 27 |
+
tensor<fp32, [1, 1, 875520]> input_3 = reshape(shape = var_32, x = input_1)[name = string("input_3")];
|
| 28 |
+
fp32 const_1 = const()[name = string("const_1"), val = fp32(0x0p+0)];
|
| 29 |
+
tensor<int32, [6]> input_5_pad_0 = const()[name = string("input_5_pad_0"), val = tensor<int32, [6]>([0, 0, 0, 0, 256, 256])];
|
| 30 |
+
string input_5_mode_0 = const()[name = string("input_5_mode_0"), val = string("constant")];
|
| 31 |
+
tensor<fp32, [1, 1, 876032]> input_5 = pad(constant_val = const_1, mode = input_5_mode_0, pad = input_5_pad_0, x = input_3)[name = string("input_5")];
|
| 32 |
+
tensor<int32, [1]> var_44 = const()[name = string("op_44"), val = tensor<int32, [1]>([876032])];
|
| 33 |
+
tensor<fp32, [876032]> input = reshape(shape = var_44, x = input_5)[name = string("input")];
|
| 34 |
+
tensor<int32, [1]> expand_dims_0_axes_0 = const()[name = string("expand_dims_0_axes_0"), val = tensor<int32, [1]>([0])];
|
| 35 |
+
tensor<fp32, [1, 876032]> expand_dims_0 = expand_dims(axes = expand_dims_0_axes_0, x = input)[name = string("expand_dims_0")];
|
| 36 |
+
tensor<fp32, [257, 1, 512]> expand_dims_1 = const()[name = string("expand_dims_1"), val = tensor<fp32, [257, 1, 512]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(131712)))];
|
| 37 |
+
tensor<fp32, [257, 1, 512]> expand_dims_2 = const()[name = string("expand_dims_2"), val = tensor<fp32, [257, 1, 512]>(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(658112)))];
|
| 38 |
+
tensor<int32, [1]> expand_dims_3 = const()[name = string("expand_dims_3"), val = tensor<int32, [1]>([160])];
|
| 39 |
+
tensor<int32, [1]> expand_dims_4_axes_0 = const()[name = string("expand_dims_4_axes_0"), val = tensor<int32, [1]>([1])];
|
| 40 |
+
tensor<fp32, [1, 1, 876032]> expand_dims_4 = expand_dims(axes = expand_dims_4_axes_0, x = expand_dims_0)[name = string("expand_dims_4")];
|
| 41 |
+
string conv_0_pad_type_0 = const()[name = string("conv_0_pad_type_0"), val = string("valid")];
|
| 42 |
+
tensor<int32, [2]> conv_0_pad_0 = const()[name = string("conv_0_pad_0"), val = tensor<int32, [2]>([0, 0])];
|
| 43 |
+
tensor<int32, [1]> conv_0_dilations_0 = const()[name = string("conv_0_dilations_0"), val = tensor<int32, [1]>([1])];
|
| 44 |
+
int32 conv_0_groups_0 = const()[name = string("conv_0_groups_0"), val = int32(1)];
|
| 45 |
+
tensor<fp32, [1, 257, 5473]> conv_0 = conv(dilations = conv_0_dilations_0, groups = conv_0_groups_0, pad = conv_0_pad_0, pad_type = conv_0_pad_type_0, strides = expand_dims_3, weight = expand_dims_1, x = expand_dims_4)[name = string("conv_0")];
|
| 46 |
+
string conv_1_pad_type_0 = const()[name = string("conv_1_pad_type_0"), val = string("valid")];
|
| 47 |
+
tensor<int32, [2]> conv_1_pad_0 = const()[name = string("conv_1_pad_0"), val = tensor<int32, [2]>([0, 0])];
|
| 48 |
+
tensor<int32, [1]> conv_1_dilations_0 = const()[name = string("conv_1_dilations_0"), val = tensor<int32, [1]>([1])];
|
| 49 |
+
int32 conv_1_groups_0 = const()[name = string("conv_1_groups_0"), val = int32(1)];
|
| 50 |
+
tensor<fp32, [1, 257, 5473]> conv_1 = conv(dilations = conv_1_dilations_0, groups = conv_1_groups_0, pad = conv_1_pad_0, pad_type = conv_1_pad_type_0, strides = expand_dims_3, weight = expand_dims_2, x = expand_dims_4)[name = string("conv_1")];
|
| 51 |
+
tensor<int32, [1]> squeeze_0_axes_0 = const()[name = string("squeeze_0_axes_0"), val = tensor<int32, [1]>([0])];
|
| 52 |
+
tensor<fp32, [257, 5473]> squeeze_0 = squeeze(axes = squeeze_0_axes_0, x = conv_0)[name = string("squeeze_0")];
|
| 53 |
+
tensor<int32, [1]> squeeze_1_axes_0 = const()[name = string("squeeze_1_axes_0"), val = tensor<int32, [1]>([0])];
|
| 54 |
+
tensor<fp32, [257, 5473]> squeeze_1 = squeeze(axes = squeeze_1_axes_0, x = conv_1)[name = string("squeeze_1")];
|
| 55 |
+
tensor<fp32, [257, 5473]> square_0 = square(x = squeeze_0)[name = string("square_0")];
|
| 56 |
+
tensor<fp32, [257, 5473]> square_1 = square(x = squeeze_1)[name = string("square_1")];
|
| 57 |
+
tensor<fp32, [257, 5473]> add_1 = add(x = square_0, y = square_1)[name = string("add_1")];
|
| 58 |
+
tensor<fp32, [257, 5473]> magnitudes = identity(x = add_1)[name = string("magnitudes")];
|
| 59 |
+
bool mel_spec_1_transpose_x_0 = const()[name = string("mel_spec_1_transpose_x_0"), val = bool(false)];
|
| 60 |
+
bool mel_spec_1_transpose_y_0 = const()[name = string("mel_spec_1_transpose_y_0"), val = bool(false)];
|
| 61 |
+
tensor<fp32, [128, 5473]> mel_spec_1 = matmul(transpose_x = mel_spec_1_transpose_x_0, transpose_y = mel_spec_1_transpose_y_0, x = mel_filters, y = magnitudes)[name = string("mel_spec_1")];
|
| 62 |
+
fp32 var_59 = const()[name = string("op_59"), val = fp32(0x1p-24)];
|
| 63 |
+
tensor<fp32, [128, 5473]> var_60 = add(x = mel_spec_1, y = var_59)[name = string("op_60")];
|
| 64 |
+
fp32 mel_spec_epsilon_0 = const()[name = string("mel_spec_epsilon_0"), val = fp32(0x1p-149)];
|
| 65 |
+
tensor<fp32, [128, 5473]> mel_spec = log(epsilon = mel_spec_epsilon_0, x = var_60)[name = string("mel_spec")];
|
| 66 |
+
tensor<int32, [2]> var_62_perm_0 = const()[name = string("op_62_perm_0"), val = tensor<int32, [2]>([1, 0])];
|
| 67 |
+
tensor<int32, [1]> var_64_axes_0 = const()[name = string("op_64_axes_0"), val = tensor<int32, [1]>([0])];
|
| 68 |
+
tensor<fp32, [5473, 128]> var_62 = transpose(perm = var_62_perm_0, x = mel_spec)[name = string("transpose_0")];
|
| 69 |
+
tensor<fp32, [1, 5473, 128]> var_64 = expand_dims(axes = var_64_axes_0, x = var_62)[name = string("op_64")];
|
| 70 |
+
tensor<int32, [1]> var_66_axes_0 = const()[name = string("op_66_axes_0"), val = tensor<int32, [1]>([1])];
|
| 71 |
+
tensor<fp32, [1, 1, 5473, 128]> var_66 = expand_dims(axes = var_66_axes_0, x = var_64)[name = string("op_66")];
|
| 72 |
+
string cast_4_dtype_0 = const()[name = string("cast_4_dtype_0"), val = string("fp16")];
|
| 73 |
+
tensor<fp16, [1, 1, 5473, 128]> melspectrogram_features = cast(dtype = cast_4_dtype_0, x = var_66)[name = string("cast_5")];
|
| 74 |
+
} -> (melspectrogram_features);
|
| 75 |
+
}
|
sortformer/nemotron-3-diarization/684_74MB/MelSpectrogram.mlmodelc/weights/weight.bin
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:78e51b9f5e73fb2db50441b974dc20c4a68793440a61ab65166f2a356efcda60
|
| 3 |
+
size 1184512
|
sortformer/nemotron-3-diarization/684_74MB/NOTICE.txt
ADDED
|
@@ -0,0 +1,12 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
NOTICE
|
| 2 |
+
|
| 3 |
+
This folder contains a Core ML conversion by Argmax, Inc. of Nemotron 3
|
| 4 |
+
Diarization, a speaker diarization model by NVIDIA Corporation.
|
| 5 |
+
|
| 6 |
+
Original checkpoint: https://huggingface.co/nvidia/Nemotron-3-Diarization
|
| 7 |
+
|
| 8 |
+
The original model is licensed under the OpenMDW License Agreement, version 1.1
|
| 9 |
+
(https://openmdw.ai/license/1-1/). A copy is included as LICENSE-OpenMDW-1.1.txt.
|
| 10 |
+
|
| 11 |
+
LICENSE_NOTICE.txt covers Argmax's conversion and packaging of this folder.
|
| 12 |
+
The NVIDIA model it contains remains under the OpenMDW License Agreement.
|
sortformer/nemotron-3-diarization/684_74MB/SortformerFullEncoder.mlmodelc/analytics/coremldata.bin
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:efd7ae1770348425117fc7c301b074d6695ba2ff0bfe317ceb701a2739579498
|
| 3 |
+
size 243
|
sortformer/nemotron-3-diarization/684_74MB/SortformerFullEncoder.mlmodelc/coremldata.bin
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:f70c86c377770675fc58b451a651ac73bc51bdff1df2d393cb89a13bd536956e
|
| 3 |
+
size 652
|
sortformer/nemotron-3-diarization/684_74MB/SortformerFullEncoder.mlmodelc/metadata.json
ADDED
|
@@ -0,0 +1,138 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
[
|
| 2 |
+
{
|
| 3 |
+
"metadataOutputVersion" : "3.0",
|
| 4 |
+
"storagePrecision" : "Mixed (Float16, Palettized (6 bits), UInt6)",
|
| 5 |
+
"outputSchema" : [
|
| 6 |
+
{
|
| 7 |
+
"hasShapeFlexibility" : "0",
|
| 8 |
+
"isOptional" : "0",
|
| 9 |
+
"dataType" : "Float16",
|
| 10 |
+
"formattedType" : "MultiArray (Float16 1 × 8 × 1 × 684)",
|
| 11 |
+
"shortDescription" : "",
|
| 12 |
+
"shape" : "[1, 8, 1, 684]",
|
| 13 |
+
"name" : "raw_speaker_preds",
|
| 14 |
+
"type" : "MultiArray"
|
| 15 |
+
},
|
| 16 |
+
{
|
| 17 |
+
"hasShapeFlexibility" : "0",
|
| 18 |
+
"isOptional" : "0",
|
| 19 |
+
"dataType" : "Float16",
|
| 20 |
+
"formattedType" : "MultiArray (Float16 1 × 8 × 1 × 684)",
|
| 21 |
+
"shortDescription" : "",
|
| 22 |
+
"shape" : "[1, 8, 1, 684]",
|
| 23 |
+
"name" : "speaker_sigmoids",
|
| 24 |
+
"type" : "MultiArray"
|
| 25 |
+
},
|
| 26 |
+
{
|
| 27 |
+
"hasShapeFlexibility" : "0",
|
| 28 |
+
"isOptional" : "0",
|
| 29 |
+
"dataType" : "Float16",
|
| 30 |
+
"formattedType" : "MultiArray (Float16 1 × 8 × 1 × 5472)",
|
| 31 |
+
"shortDescription" : "",
|
| 32 |
+
"shape" : "[1, 8, 1, 5472]",
|
| 33 |
+
"name" : "speaker_sigmoids_highres",
|
| 34 |
+
"type" : "MultiArray"
|
| 35 |
+
}
|
| 36 |
+
],
|
| 37 |
+
"modelParameters" : [
|
| 38 |
+
|
| 39 |
+
],
|
| 40 |
+
"specificationVersion" : 9,
|
| 41 |
+
"mlProgramOperationTypeHistogram" : {
|
| 42 |
+
"Ios18.constexprLutToDense" : 191,
|
| 43 |
+
"Ios18.batchNorm" : 64,
|
| 44 |
+
"Ios18.conv" : 190,
|
| 45 |
+
"Ios18.expandDims" : 6,
|
| 46 |
+
"Ios18.sub" : 2,
|
| 47 |
+
"Ios18.matmul" : 62,
|
| 48 |
+
"Ios18.gelu" : 31,
|
| 49 |
+
"Ios18.concat" : 62,
|
| 50 |
+
"Ios18.relu" : 2,
|
| 51 |
+
"Ios18.add" : 186,
|
| 52 |
+
"Ios18.sigmoid" : 1,
|
| 53 |
+
"Ios18.softmax" : 31,
|
| 54 |
+
"Ios18.sliceByIndex" : 124,
|
| 55 |
+
"Ios18.layerNorm" : 64,
|
| 56 |
+
"Ios18.transpose" : 1,
|
| 57 |
+
"Ios18.avgPool" : 2,
|
| 58 |
+
"Ios18.reshape" : 127,
|
| 59 |
+
"Ios18.mul" : 220
|
| 60 |
+
},
|
| 61 |
+
"computePrecision" : "Mixed (Float16, Int32)",
|
| 62 |
+
"isUpdatable" : "0",
|
| 63 |
+
"stateSchema" : [
|
| 64 |
+
|
| 65 |
+
],
|
| 66 |
+
"availability" : {
|
| 67 |
+
"macOS" : "15.0",
|
| 68 |
+
"tvOS" : "18.0",
|
| 69 |
+
"visionOS" : "2.0",
|
| 70 |
+
"watchOS" : "11.0",
|
| 71 |
+
"iOS" : "18.0",
|
| 72 |
+
"macCatalyst" : "18.0"
|
| 73 |
+
},
|
| 74 |
+
"modelType" : {
|
| 75 |
+
"name" : "MLModelType_mlProgram"
|
| 76 |
+
},
|
| 77 |
+
"userDefinedMetadata" : {
|
| 78 |
+
"com.github.apple.coremltools.conversion_date" : "2026-09-21",
|
| 79 |
+
"com.github.apple.coremltools.source" : "torch==2.8.0",
|
| 80 |
+
"com.github.apple.coremltools.version" : "9.0",
|
| 81 |
+
"com.github.apple.coremltools.source_dialect" : "TorchScript"
|
| 82 |
+
},
|
| 83 |
+
"inputSchema" : [
|
| 84 |
+
{
|
| 85 |
+
"hasShapeFlexibility" : "0",
|
| 86 |
+
"isOptional" : "0",
|
| 87 |
+
"dataType" : "Float16",
|
| 88 |
+
"formattedType" : "MultiArray (Float16 1 × 512 × 1 × 684)",
|
| 89 |
+
"shortDescription" : "",
|
| 90 |
+
"shape" : "[1, 512, 1, 684]",
|
| 91 |
+
"name" : "downsampled_melspectrogram_features",
|
| 92 |
+
"type" : "MultiArray"
|
| 93 |
+
},
|
| 94 |
+
{
|
| 95 |
+
"hasShapeFlexibility" : "0",
|
| 96 |
+
"isOptional" : "0",
|
| 97 |
+
"dataType" : "Float16",
|
| 98 |
+
"formattedType" : "MultiArray (Float16 1 × 684)",
|
| 99 |
+
"shortDescription" : "",
|
| 100 |
+
"shape" : "[1, 684]",
|
| 101 |
+
"name" : "conformer_encoder_padding_mask",
|
| 102 |
+
"type" : "MultiArray"
|
| 103 |
+
},
|
| 104 |
+
{
|
| 105 |
+
"hasShapeFlexibility" : "0",
|
| 106 |
+
"isOptional" : "0",
|
| 107 |
+
"dataType" : "Float16",
|
| 108 |
+
"formattedType" : "MultiArray (Float16 1 × 1 × 684 × 684)",
|
| 109 |
+
"shortDescription" : "",
|
| 110 |
+
"shape" : "[1, 1, 684, 684]",
|
| 111 |
+
"name" : "conformer_encoder_qk_mask",
|
| 112 |
+
"type" : "MultiArray"
|
| 113 |
+
},
|
| 114 |
+
{
|
| 115 |
+
"hasShapeFlexibility" : "0",
|
| 116 |
+
"isOptional" : "0",
|
| 117 |
+
"dataType" : "Float16",
|
| 118 |
+
"formattedType" : "MultiArray (Float16 1 × 684)",
|
| 119 |
+
"shortDescription" : "",
|
| 120 |
+
"shape" : "[1, 684]",
|
| 121 |
+
"name" : "transformer_encoder_mask",
|
| 122 |
+
"type" : "MultiArray"
|
| 123 |
+
},
|
| 124 |
+
{
|
| 125 |
+
"hasShapeFlexibility" : "0",
|
| 126 |
+
"isOptional" : "0",
|
| 127 |
+
"dataType" : "Float16",
|
| 128 |
+
"formattedType" : "MultiArray (Float16 1 × 1 × 1 × 1)",
|
| 129 |
+
"shortDescription" : "",
|
| 130 |
+
"shape" : "[1, 1, 1, 1]",
|
| 131 |
+
"name" : "input_1",
|
| 132 |
+
"type" : "MultiArray"
|
| 133 |
+
}
|
| 134 |
+
],
|
| 135 |
+
"generatedClassName" : "SortformerFullEncoder",
|
| 136 |
+
"method" : "predict"
|
| 137 |
+
}
|
| 138 |
+
]
|
sortformer/nemotron-3-diarization/684_74MB/SortformerFullEncoder.mlmodelc/model.mil
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
sortformer/nemotron-3-diarization/684_74MB/SortformerFullEncoder.mlmodelc/weights/weight.bin
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:0e934e5dc095ae12847379ce47f51b36c4db4a5c5a6bafff01bbc46f6e3abfaf
|
| 3 |
+
size 74362752
|