Audio Classification
Transformers
Safetensors
multilingual
wav2vec2-dual-hypersphere
audio-deepfake
deepfake-detection
deepfake
voice-cloning
anti-spoofing
asvspoof
wav2vec2
speech
audio
synthetic-voice
voice-conversion
tts-detection
trust-and-safety
security
SoTA
Modotte
custom_code
Instructions to use Modotte/AIRealNet-Audio with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use Modotte/AIRealNet-Audio with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("audio-classification", model="Modotte/AIRealNet-Audio", trust_remote_code=True)# Load model directly from transformers import AutoModelForAudioClassification model = AutoModelForAudioClassification.from_pretrained("Modotte/AIRealNet-Audio", trust_remote_code=True, device_map="auto") - Notebooks
- Google Colab
- Kaggle
Download modeling.py from Modotte/AIRealNet-Audio: direct link, hf CLI and curl.
- Browser
- Download file 5.07 kB
-
https://huggingface.co/Modotte/AIRealNet-Audio/resolve/main/modeling.py
- Command line
-
hf download hf://Modotte/AIRealNet-Audio/modeling.py
-
curl -L -o modeling.py https://huggingface.co/Modotte/AIRealNet-Audio/resolve/main/modeling.py
5.07 kB
| from typing import Optional, Tuple, Union | |
| import torch | |
| import torch.nn as nn | |
| import torch.nn.functional as F | |
| from transformers import Wav2Vec2Model, Wav2Vec2PreTrainedModel | |
| from transformers.modeling_outputs import SequenceClassifierOutput | |
| from transformers.models.auto import AutoConfig, AutoModelForAudioClassification | |
| from .configuration import Wav2Vec2DualHypersphereConfig | |
| class Wav2Vec2DualHypersphereForAudioClassification(Wav2Vec2PreTrainedModel): | |
| config_class = Wav2Vec2DualHypersphereConfig | |
| def __init__(self, config: Wav2Vec2DualHypersphereConfig): | |
| super().__init__(config) | |
| self.num_labels = config.num_labels | |
| self.wav2vec2 = Wav2Vec2Model(config) | |
| proj_dim = getattr(config, "classifier_proj_size", 256) | |
| self.projector = nn.Linear(config.hidden_size, proj_dim) | |
| self.dropout = nn.Dropout(getattr(config, "final_dropout", 0.1)) | |
| self.classifier = nn.Linear(proj_dim, self.num_labels) | |
| # Initialize weights and apply final processing | |
| self.post_init() | |
| def freeze_feature_extractor(self): | |
| self.wav2vec2.feature_extractor._freeze_parameters() | |
| def forward( | |
| self, | |
| input_values: Optional[torch.Tensor] = None, | |
| attention_mask: Optional[torch.Tensor] = None, | |
| output_attentions: Optional[bool] = None, | |
| output_hidden_states: Optional[bool] = None, | |
| return_dict: Optional[bool] = None, | |
| labels: Optional[torch.Tensor] = None, | |
| ) -> Union[Tuple, SequenceClassifierOutput]: | |
| r""" | |
| labels (`torch.LongTensor` of shape `(batch_size,)`, *optional*): | |
| Labels for computing the sequence classification/regression loss. | |
| """ | |
| return_dict = return_dict if return_dict is not None else self.config.return_dict | |
| outputs = self.wav2vec2( | |
| input_values, | |
| attention_mask=attention_mask, | |
| output_attentions=output_attentions, | |
| output_hidden_states=output_hidden_states, | |
| return_dict=return_dict, | |
| ) | |
| hidden_states = outputs[0] | |
| # 1. Temporal Pooling with Attention Mask support | |
| if attention_mask is not None: | |
| mask = self.wav2vec2._get_feature_vector_attention_mask( | |
| hidden_states.shape[1], attention_mask | |
| ) | |
| mask = mask.unsqueeze(-1).expand_as(hidden_states) | |
| pooled_768 = (hidden_states * mask).sum(dim=1) / mask.sum(dim=1).clamp(min=1e-9) | |
| else: | |
| pooled_768 = hidden_states.mean(dim=1) | |
| # 2. First Hypersphere Projection (L2 Normalize 768-D) | |
| norm_pooled_768 = F.normalize(pooled_768, p=2, dim=-1) | |
| # 3. Intermediate Projection & Dropout | |
| projected_256 = self.dropout(self.projector(norm_pooled_768)) | |
| # 4. Second Hypersphere Projection (L2 Normalize 256-D) | |
| norm_projected_256 = F.normalize(projected_256, p=2, dim=-1) | |
| # 5. Final Classification Logits | |
| logits = self.classifier(norm_projected_256) | |
| loss = None | |
| if labels is not None: | |
| if self.config.problem_type is None: | |
| if self.num_labels == 1: | |
| self.config.problem_type = "regression" | |
| elif self.num_labels > 1 and (labels.dtype == torch.long or labels.dtype == torch.int): | |
| self.config.problem_type = "single_label_classification" | |
| else: | |
| self.config.problem_type = "multi_label_classification" | |
| if self.config.problem_type == "regression": | |
| loss_fct = nn.MSELoss() | |
| if self.num_labels == 1: | |
| loss = loss_fct(logits.squeeze(), labels.squeeze()) | |
| else: | |
| loss = loss_fct(logits, labels) | |
| elif self.config.problem_type == "single_label_classification": | |
| loss_fct = nn.CrossEntropyLoss() | |
| loss = loss_fct(logits.view(-1, self.num_labels), labels.view(-1)) | |
| elif self.config.problem_type == "multi_label_classification": | |
| loss_fct = nn.BCEWithLogitsLoss() | |
| loss = loss_fct(logits, labels) | |
| if not return_dict: | |
| output = (logits,) + outputs[2:] | |
| return ((loss,) + output) if loss is not None else output | |
| return SequenceClassifierOutput( | |
| loss=loss, | |
| logits=logits, | |
| hidden_states=outputs.hidden_states if hasattr(outputs, "hidden_states") else None, | |
| attentions=outputs.attentions if hasattr(outputs, "attentions") else None, | |
| ) | |
| # Register model & config with Auto classes | |
| AutoConfig.register("wav2vec2-dual-hypersphere", Wav2Vec2DualHypersphereConfig) | |
| AutoModelForAudioClassification.register( | |
| Wav2Vec2DualHypersphereConfig, Wav2Vec2DualHypersphereForAudioClassification | |
| ) | |
| Wav2Vec2DualHypersphereClassifier = Wav2Vec2DualHypersphereForAudioClassification | |
| __all__ = [ | |
| "Wav2Vec2DualHypersphereConfig", | |
| "Wav2Vec2DualHypersphereForAudioClassification", | |
| "Wav2Vec2DualHypersphereClassifier", | |
| ] |