Export ONNX version of model 'sanchit-gandhi/whisper-base-ft-common-language-id', on 2024-08-30 01:35:14 CST

Browse files

Files changed (4) hide show

README.md +17 -0
config.json +246 -0
model.onnx +3 -0
preprocessor_config.json +14 -0

README.md ADDED Viewed

	@@ -0,0 +1,17 @@

+---
+base_model: sanchit-gandhi/whisper-base-ft-common-language-id
+datasets:
+- common_language
+license: apache-2.0
+metrics:
+- accuracy
+model-index:
+- name: whisper-base-ft-common-language-id
+  results: []
+tags:
+- audio-classification
+- generated_from_trainer
+---
+This is the ONNX exported version of [sanchit-gandhi/whisper-base-ft-common-language-id](https://huggingface.co/sanchit-gandhi/whisper-base-ft-common-language-id).

config.json ADDED Viewed

	@@ -0,0 +1,246 @@

+{
+  "_name_or_path": "sanchit-gandhi/whisper-base-ft-common-language-id",
+  "activation_dropout": 0.0,
+  "activation_function": "gelu",
+  "apply_spec_augment": false,
+  "architectures": [
+    "WhisperForAudioClassification"
+  ],
+  "attention_dropout": 0.0,
+  "begin_suppress_tokens": [
+    220,
+    50257
+  ],
+  "bos_token_id": 50257,
+  "classifier_proj_size": 256,
+  "d_model": 512,
+  "decoder_attention_heads": 8,
+  "decoder_ffn_dim": 2048,
+  "decoder_layerdrop": 0.0,
+  "decoder_layers": 6,
+  "decoder_start_token_id": 50258,
+  "dropout": 0.0,
+  "encoder_attention_heads": 8,
+  "encoder_ffn_dim": 2048,
+  "encoder_layerdrop": 0.0,
+  "encoder_layers": 6,
+  "eos_token_id": 50257,
+  "finetuning_task": "audio-classification",
+  "forced_decoder_ids": [
+    [
+      1,
+      50259
+    ],
+    [
+      2,
+      50359
+    ],
+    [
+      3,
+      50363
+    ]
+  ],
+  "id2label": {
+    "0": "Arabic",
+    "1": "Basque",
+    "2": "Breton",
+    "3": "Catalan",
+    "4": "Chinese_China",
+    "5": "Chinese_Hongkong",
+    "6": "Chinese_Taiwan",
+    "7": "Chuvash",
+    "8": "Czech",
+    "9": "Dhivehi",
+    "10": "Dutch",
+    "11": "English",
+    "12": "Esperanto",
+    "13": "Estonian",
+    "14": "French",
+    "15": "Frisian",
+    "16": "Georgian",
+    "17": "German",
+    "18": "Greek",
+    "19": "Hakha_Chin",
+    "20": "Indonesian",
+    "21": "Interlingua",
+    "22": "Italian",
+    "23": "Japanese",
+    "24": "Kabyle",
+    "25": "Kinyarwanda",
+    "26": "Kyrgyz",
+    "27": "Latvian",
+    "28": "Maltese",
+    "29": "Mangolian",
+    "30": "Persian",
+    "31": "Polish",
+    "32": "Portuguese",
+    "33": "Romanian",
+    "34": "Romansh_Sursilvan",
+    "35": "Russian",
+    "36": "Sakha",
+    "37": "Slovenian",
+    "38": "Spanish",
+    "39": "Swedish",
+    "40": "Tamil",
+    "41": "Tatar",
+    "42": "Turkish",
+    "43": "Ukranian",
+    "44": "Welsh"
+  },
+  "init_std": 0.02,
+  "is_encoder_decoder": true,
+  "label2id": {
+    "Arabic": "0",
+    "Basque": "1",
+    "Breton": "2",
+    "Catalan": "3",
+    "Chinese_China": "4",
+    "Chinese_Hongkong": "5",
+    "Chinese_Taiwan": "6",
+    "Chuvash": "7",
+    "Czech": "8",
+    "Dhivehi": "9",
+    "Dutch": "10",
+    "English": "11",
+    "Esperanto": "12",
+    "Estonian": "13",
+    "French": "14",
+    "Frisian": "15",
+    "Georgian": "16",
+    "German": "17",
+    "Greek": "18",
+    "Hakha_Chin": "19",
+    "Indonesian": "20",
+    "Interlingua": "21",
+    "Italian": "22",
+    "Japanese": "23",
+    "Kabyle": "24",
+    "Kinyarwanda": "25",
+    "Kyrgyz": "26",
+    "Latvian": "27",
+    "Maltese": "28",
+    "Mangolian": "29",
+    "Persian": "30",
+    "Polish": "31",
+    "Portuguese": "32",
+    "Romanian": "33",
+    "Romansh_Sursilvan": "34",
+    "Russian": "35",
+    "Sakha": "36",
+    "Slovenian": "37",
+    "Spanish": "38",
+    "Swedish": "39",
+    "Tamil": "40",
+    "Tatar": "41",
+    "Turkish": "42",
+    "Ukranian": "43",
+    "Welsh": "44"
+  },
+  "mask_feature_length": 10,
+  "mask_feature_min_masks": 0,
+  "mask_feature_prob": 0.0,
+  "mask_time_length": 10,
+  "mask_time_min_masks": 2,
+  "mask_time_prob": 0.05,
+  "max_length": 448,
+  "max_source_positions": 1500,
+  "max_target_positions": 448,
+  "median_filter_width": 7,
+  "model_type": "whisper",
+  "num_hidden_layers": 6,
+  "num_mel_bins": 80,
+  "pad_token_id": 50257,
+  "scale_embedding": false,
+  "suppress_tokens": [
+    1,
+    2,
+    7,
+    8,
+    9,
+    10,
+    14,
+    25,
+    26,
+    27,
+    28,
+    29,
+    31,
+    58,
+    59,
+    60,
+    61,
+    62,
+    63,
+    90,
+    91,
+    92,
+    93,
+    359,
+    503,
+    522,
+    542,
+    873,
+    893,
+    902,
+    918,
+    922,
+    931,
+    1350,
+    1853,
+    1982,
+    2460,
+    2627,
+    3246,
+    3253,
+    3268,
+    3536,
+    3846,
+    3961,
+    4183,
+    4667,
+    6585,
+    6647,
+    7273,
+    9061,
+    9383,
+    10428,
+    10929,
+    11938,
+    12033,
+    12331,
+    12562,
+    13793,
+    14157,
+    14635,
+    15265,
+    15618,
+    16553,
+    16604,
+    18362,
+    18956,
+    20075,
+    21675,
+    22520,
+    26130,
+    26161,
+    26435,
+    28279,
+    29464,
+    31650,
+    32302,
+    32470,
+    36865,
+    42863,
+    47425,
+    49870,
+    50254,
+    50258,
+    50360,
+    50361,
+    50362
+  ],
+  "transformers_version": "4.43.4",
+  "use_cache": true,
+  "use_weighted_layer_sum": false,
+  "vocab_size": 51865
+}

model.onnx ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:7d2203033195dd4033b067b5efecef7d9d5e67321072cabac07a1f9e075e1b94
+size 83029854

preprocessor_config.json ADDED Viewed

	@@ -0,0 +1,14 @@

+{
+  "chunk_length": 30,
+  "feature_extractor_type": "WhisperFeatureExtractor",
+  "feature_size": 80,
+  "hop_length": 160,
+  "n_fft": 400,
+  "n_samples": 480000,
+  "nb_max_frames": 3000,
+  "padding_side": "right",
+  "padding_value": 0.0,
+  "processor_class": "WhisperProcessor",
+  "return_attention_mask": false,
+  "sampling_rate": 16000
+}