Instructions to use ylacombe/accent-classifier with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use ylacombe/accent-classifier with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("audio-classification", model="ylacombe/accent-classifier")# Load model directly from transformers import AutoProcessor, AutoModelForAudioClassification processor = AutoProcessor.from_pretrained("ylacombe/accent-classifier") model = AutoModelForAudioClassification.from_pretrained("ylacombe/accent-classifier", device_map="auto") - Notebooks
- Google Colab
- Kaggle
| { | |
| "_name_or_path": "/fsx/yoach/accent_classification_output_final_v3_smaller_dataset/", | |
| "activation_dropout": 0.0, | |
| "adapter_attn_dim": 16, | |
| "adapter_kernel_size": 3, | |
| "adapter_stride": 2, | |
| "add_adapter": false, | |
| "apply_spec_augment": true, | |
| "architectures": [ | |
| "Wav2Vec2ForSequenceClassification" | |
| ], | |
| "attention_dropout": 0.0, | |
| "bos_token_id": 1, | |
| "classifier_proj_size": 1024, | |
| "codevector_dim": 1024, | |
| "contrastive_logits_temperature": 0.1, | |
| "conv_bias": true, | |
| "conv_dim": [ | |
| 512, | |
| 512, | |
| 512, | |
| 512, | |
| 512, | |
| 512, | |
| 512 | |
| ], | |
| "conv_kernel": [ | |
| 10, | |
| 3, | |
| 3, | |
| 3, | |
| 3, | |
| 2, | |
| 2 | |
| ], | |
| "conv_stride": [ | |
| 5, | |
| 2, | |
| 2, | |
| 2, | |
| 2, | |
| 2, | |
| 2 | |
| ], | |
| "ctc_loss_reduction": "mean", | |
| "ctc_zero_infinity": false, | |
| "diversity_loss_weight": 0.1, | |
| "do_stable_layer_norm": true, | |
| "eos_token_id": 2, | |
| "feat_extract_activation": "gelu", | |
| "feat_extract_dropout": 0.0, | |
| "feat_extract_norm": "layer", | |
| "feat_proj_dropout": 0.0, | |
| "feat_quantizer_dropout": 0.0, | |
| "final_dropout": 0.0, | |
| "finetuning_task": "audio-classification", | |
| "hidden_act": "gelu", | |
| "hidden_dropout": 0.0, | |
| "hidden_size": 1280, | |
| "id2label": { | |
| "0": "American", | |
| "1": "Australian", | |
| "2": "Canadian", | |
| "3": "Chinese", | |
| "4": "Czech", | |
| "5": "Dutch", | |
| "6": "Eastern european", | |
| "7": "English", | |
| "8": "Estonian", | |
| "9": "Finnish", | |
| "10": "French", | |
| "11": "German", | |
| "12": "Hungarian", | |
| "13": "Indian", | |
| "14": "Irish", | |
| "15": "Italian", | |
| "16": "Jamaican", | |
| "17": "Latin american", | |
| "18": "Malaysian", | |
| "19": "New zealand", | |
| "20": "Polish", | |
| "21": "Romanian", | |
| "22": "Scottish", | |
| "23": "Singaporean", | |
| "24": "Slovak", | |
| "25": "South african", | |
| "26": "Spanish", | |
| "27": "Welsh" | |
| }, | |
| "initializer_range": 0.02, | |
| "intermediate_size": 5120, | |
| "label2id": { | |
| "American": "0", | |
| "Australian": "1", | |
| "Canadian": "2", | |
| "Chinese": "3", | |
| "Czech": "4", | |
| "Dutch": "5", | |
| "Eastern european": "6", | |
| "English": "7", | |
| "Estonian": "8", | |
| "Finnish": "9", | |
| "French": "10", | |
| "German": "11", | |
| "Hungarian": "12", | |
| "Indian": "13", | |
| "Irish": "14", | |
| "Italian": "15", | |
| "Jamaican": "16", | |
| "Latin american": "17", | |
| "Malaysian": "18", | |
| "New zealand": "19", | |
| "Polish": "20", | |
| "Romanian": "21", | |
| "Scottish": "22", | |
| "Singaporean": "23", | |
| "Slovak": "24", | |
| "South african": "25", | |
| "Spanish": "26", | |
| "Welsh": "27" | |
| }, | |
| "layer_norm_eps": 1e-05, | |
| "layerdrop": 0.0, | |
| "mask_feature_length": 10, | |
| "mask_feature_min_masks": 0, | |
| "mask_feature_prob": 0.0, | |
| "mask_time_length": 10, | |
| "mask_time_min_masks": 2, | |
| "mask_time_prob": 0.05, | |
| "model_type": "wav2vec2", | |
| "num_adapter_layers": 3, | |
| "num_attention_heads": 16, | |
| "num_codevector_groups": 2, | |
| "num_codevectors_per_group": 320, | |
| "num_conv_pos_embedding_groups": 16, | |
| "num_conv_pos_embeddings": 128, | |
| "num_feat_extract_layers": 7, | |
| "num_hidden_layers": 48, | |
| "num_negatives": 100, | |
| "output_hidden_size": 1280, | |
| "pad_token_id": 0, | |
| "proj_codevector_dim": 1024, | |
| "tdnn_dilation": [ | |
| 1, | |
| 2, | |
| 3, | |
| 1, | |
| 1 | |
| ], | |
| "tdnn_dim": [ | |
| 512, | |
| 512, | |
| 512, | |
| 512, | |
| 1500 | |
| ], | |
| "tdnn_kernel": [ | |
| 5, | |
| 3, | |
| 3, | |
| 1, | |
| 1 | |
| ], | |
| "torch_dtype": "float32", | |
| "transformers_version": "4.40.2", | |
| "use_weighted_layer_sum": false, | |
| "vocab_size": 154, | |
| "xvector_output_dim": 512 | |
| } | |