first

Browse files

Files changed (14) hide show

.gitignore +1 -0
.run_experiment2.sh.un~ +0 -0
create_student_model.py +226 -0
nb-distil-large-init/added_tokens.json +1611 -0
nb-distil-large-init/config.json +288 -0
nb-distil-large-init/flax_model.msgpack +3 -0
nb-distil-large-init/generation_config.json +270 -0
nb-distil-large-init/merges.txt +0 -0
nb-distil-large-init/preprocessor_config.json +14 -0
nb-distil-large-init/special_tokens_map.json +139 -0
nb-distil-large-init/tokenizer_config.json +0 -0
nb-distil-large-init/vocab.json +0 -0
run_distillation.py +2172 -0
run_experiment2.sh +41 -0

.gitignore ADDED Viewed

	@@ -0,0 +1 @@


1	+ wandb/

.run_experiment2.sh.un~ ADDED Viewed

Binary file (2.95 kB). View file

create_student_model.py ADDED Viewed

	@@ -0,0 +1,226 @@

+#!/usr/bin/env python
+# coding=utf-8
+# Copyright 2023 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+"""
+Initialise a student Whisper model from a pre-trained teacher model for
+teacher-student distillation.
+"""
+import argparse
+import copy
+import logging
+import jax
+import numpy as np
+from flax.core import freeze, unfreeze
+from transformers import GenerationConfig, WhisperFeatureExtractor, WhisperProcessor
+from distil_whisper import FlaxWhisperForConditionalGeneration
+logger = logging.getLogger(__name__)
+def parse_args():
+    parser = argparse.ArgumentParser(
+        description="Initialise a student Whisper model from a teacher model, copying the relevant layer weights and adjusting the processor as necessary."
+    )
+    parser.add_argument(
+        "--teacher_checkpoint",
+        type=str,
+        required=True,
+        help="The HF Hub ID of the teacher checkpoint.",
+    )
+    parser.add_argument(
+        "--subfolder",
+        type=str,
+        default="",
+        help="In case the relevant teacher weights are located inside a subfolder of the model repo on huggingface.co, you "
+        "can specify the folder name here.",
+    )
+    parser.add_argument(
+        "--encoder_layers",
+        type=int,
+        default=None,
+        help="Number of encoder layers to use in the student model. Defaults to all layers from the teacher.",
+    )
+    parser.add_argument(
+        "--decoder_layers",
+        type=int,
+        default=2,
+        help="Number of decoder layers to use in the student model. Defaults to 2 layers.",
+    )
+    parser.add_argument(
+        "--max_source_positions",
+        type=int,
+        default=None,
+        help="The maximum sequence length of log-mel filter-bank features that this model might ever be used with. Can "
+        "be used to create a student model with a shorter context length than the teacher model. Defaults to the number "
+        "of source positions in the teacher model (1500).",
+    )
+    parser.add_argument(
+        "--save_dir",
+        type=str,
+        required=True,
+        help="Where to save the student weights and processor.",
+    )
+    parser.add_argument(
+        "--push_to_hub",
+        type=bool,
+        required=False,
+        default=False,
+        help="Whether to push the student weights and processor to the Hub.",
+    )
+    parser.add_argument(
+        "--cache_dir",
+        type=str,
+        default=None,
+        help="Where to store the pretrained models downloaded from huggingface.co",
+    )
+    args = parser.parse_args()
+    return args
+def init_student_model_from_teacher(
+    teacher_checkpoint,
+    encoder_layers=None,
+    decoder_layers=2,
+    max_source_positions=None,
+    save_dir=None,
+    push_to_hub=None,
+    cache_dir=None,
+    subfolder="",
+):
+    teacher_model, teacher_params = FlaxWhisperForConditionalGeneration.from_pretrained(
+        teacher_checkpoint,
+        _do_init=False,
+        cache_dir=cache_dir,
+        subfolder=subfolder,
+    )
+    processor = WhisperProcessor.from_pretrained(teacher_checkpoint)
+    generation_config = GenerationConfig.from_pretrained(teacher_checkpoint)
+    teacher_config = teacher_model.config
+    teacher_encoder_layers = teacher_config.encoder_layers
+    teacher_decoder_layers = teacher_config.decoder_layers
+    student_config = copy.deepcopy(teacher_config)
+    student_config.update(
+        {
+            "encoder_layers": encoder_layers if encoder_layers is not None else teacher_encoder_layers,
+            "decoder_layers": decoder_layers,
+            "max_source_positions": (
+                max_source_positions if max_source_positions is not None else student_config.max_source_positions
+            ),
+        }
+    )
+    encoder_mapping = np.linspace(0, teacher_encoder_layers - 1, student_config.encoder_layers, dtype=int)
+    encoder_mapping[-1] = teacher_encoder_layers - 1
+    encoder_map = {}
+    for student_layer, teacher_layer in enumerate(encoder_mapping):
+        encoder_map[str(teacher_layer)] = str(student_layer)
+    decoder_mapping = np.linspace(0, teacher_decoder_layers - 1, student_config.decoder_layers, dtype=int)
+    decoder_mapping[-1] = teacher_decoder_layers - 1
+    decoder_map = {}
+    for student_layer, teacher_layer in enumerate(decoder_mapping):
+        decoder_map[str(teacher_layer)] = str(student_layer)
+    # init the student params from the teacher model
+    student_params = unfreeze(teacher_params)
+    student_params["model"]["decoder"]["layers"] = {}
+    for layer in teacher_params["model"]["decoder"]["layers"]:
+        if layer in decoder_map:
+            # re-introduce pre-defined layers from the teacher
+            student_params["model"]["decoder"]["layers"][decoder_map[layer]] = teacher_params["model"]["decoder"][
+                "layers"
+            ][layer]
+    if encoder_layers is not None:
+        student_params["model"]["encoder"]["layers"] = {}
+        for layer in teacher_params["model"]["encoder"]["layers"]:
+            if layer in encoder_map:
+                # re-introduce pre-defined layers from the teacher
+                student_params["model"]["encoder"]["layers"][encoder_map[layer]] = teacher_params["model"]["encoder"][
+                    "layers"
+                ][layer]
+    if max_source_positions is not None:
+        # slice the first MAX_SOURCE_POSITIONS embedding weights
+        student_params["model"]["encoder"]["embed_positions"]["embedding"] = teacher_params["model"]["encoder"][
+            "embed_positions"
+        ]["embedding"][: student_config.max_source_positions, :]
+        # update the feature extractor to handle the new input length
+        chunk_length = int(student_config.max_source_positions * 2 / 100)
+        processor.feature_extractor = WhisperFeatureExtractor(chunk_length=chunk_length)
+    # remove the teacher params and model
+    del teacher_params, teacher_model
+    # save the converted weights and model
+    student_params = freeze(student_params)
+    student_model = FlaxWhisperForConditionalGeneration(student_config, _do_init=False)
+    if save_dir is not None:
+        student_model.save_pretrained(save_dir, params=student_params)
+        # we also need to correctly save the processor and generation config
+        processor.save_pretrained(save_dir)
+        generation_config.save_pretrained(save_dir)
+    # check we can do a forward pass with the saved model - first load the weights and processor
+    logger.info("Checking we can load the saved model...")
+    student_model, student_params = FlaxWhisperForConditionalGeneration.from_pretrained(
+        save_dir,
+        _do_init=False,
+    )
+    processor = WhisperProcessor.from_pretrained(save_dir)
+    # define some random inputs
+    input_features = processor(np.ones(16000), sampling_rate=16000, return_tensors="np").input_features
+    decoder_start_token_id = student_model.config.decoder_start_token_id
+    decoder_input_ids = np.ones((input_features.shape[0], 1)) * decoder_start_token_id
+    # do a forward pass - outputs will be gibberish for the initialised model so we can't check them
+    logger.info("Checking we can run the converted model forward...")
+    _ = student_model(input_features, decoder_input_ids=decoder_input_ids, params=student_params).logits
+    logger.info("Conversion successful!")
+    if push_to_hub:
+        student_model.push_to_hub(save_dir, params=student_params)
+        processor.push_to_hub(save_dir)
+        generation_config.push_to_hub(save_dir)
+if __name__ == "__main__":
+    args = parse_args()
+    # Set the verbosity to info of the logger - we only want one process per machine to log things on the screen
+    logger.setLevel(logging.INFO if jax.process_index() == 0 else logging.ERROR)
+    init_student_model_from_teacher(
+        teacher_checkpoint=args.teacher_checkpoint,
+        encoder_layers=args.encoder_layers,
+        decoder_layers=args.decoder_layers,
+        max_source_positions=args.max_source_positions,
+        save_dir=args.save_dir,
+        push_to_hub=args.push_to_hub,
+        cache_dir=args.cache_dir,
+        subfolder=args.subfolder,
+    )

nb-distil-large-init/added_tokens.json ADDED Viewed

	@@ -0,0 +1,1611 @@

+{
+  "<|0.00|>": 50365,
+  "<|0.02|>": 50366,
+  "<|0.04|>": 50367,
+  "<|0.06|>": 50368,
+  "<|0.08|>": 50369,
+  "<|0.10|>": 50370,
+  "<|0.12|>": 50371,
+  "<|0.14|>": 50372,
+  "<|0.16|>": 50373,
+  "<|0.18|>": 50374,
+  "<|0.20|>": 50375,
+  "<|0.22|>": 50376,
+  "<|0.24|>": 50377,
+  "<|0.26|>": 50378,
+  "<|0.28|>": 50379,
+  "<|0.30|>": 50380,
+  "<|0.32|>": 50381,
+  "<|0.34|>": 50382,
+  "<|0.36|>": 50383,
+  "<|0.38|>": 50384,
+  "<|0.40|>": 50385,
+  "<|0.42|>": 50386,
+  "<|0.44|>": 50387,
+  "<|0.46|>": 50388,
+  "<|0.48|>": 50389,
+  "<|0.50|>": 50390,
+  "<|0.52|>": 50391,
+  "<|0.54|>": 50392,
+  "<|0.56|>": 50393,
+  "<|0.58|>": 50394,
+  "<|0.60|>": 50395,
+  "<|0.62|>": 50396,
+  "<|0.64|>": 50397,
+  "<|0.66|>": 50398,
+  "<|0.68|>": 50399,
+  "<|0.70|>": 50400,
+  "<|0.72|>": 50401,
+  "<|0.74|>": 50402,
+  "<|0.76|>": 50403,
+  "<|0.78|>": 50404,
+  "<|0.80|>": 50405,
+  "<|0.82|>": 50406,
+  "<|0.84|>": 50407,
+  "<|0.86|>": 50408,
+  "<|0.88|>": 50409,
+  "<|0.90|>": 50410,
+  "<|0.92|>": 50411,
+  "<|0.94|>": 50412,
+  "<|0.96|>": 50413,
+  "<|0.98|>": 50414,
+  "<|1.00|>": 50415,
+  "<|1.02|>": 50416,
+  "<|1.04|>": 50417,
+  "<|1.06|>": 50418,
+  "<|1.08|>": 50419,
+  "<|1.10|>": 50420,
+  "<|1.12|>": 50421,
+  "<|1.14|>": 50422,
+  "<|1.16|>": 50423,
+  "<|1.18|>": 50424,
+  "<|1.20|>": 50425,
+  "<|1.22|>": 50426,
+  "<|1.24|>": 50427,
+  "<|1.26|>": 50428,
+  "<|1.28|>": 50429,
+  "<|1.30|>": 50430,
+  "<|1.32|>": 50431,
+  "<|1.34|>": 50432,
+  "<|1.36|>": 50433,
+  "<|1.38|>": 50434,
+  "<|1.40|>": 50435,
+  "<|1.42|>": 50436,
+  "<|1.44|>": 50437,
+  "<|1.46|>": 50438,
+  "<|1.48|>": 50439,
+  "<|1.50|>": 50440,
+  "<|1.52|>": 50441,
+  "<|1.54|>": 50442,
+  "<|1.56|>": 50443,
+  "<|1.58|>": 50444,
+  "<|1.60|>": 50445,
+  "<|1.62|>": 50446,
+  "<|1.64|>": 50447,
+  "<|1.66|>": 50448,
+  "<|1.68|>": 50449,
+  "<|1.70|>": 50450,
+  "<|1.72|>": 50451,
+  "<|1.74|>": 50452,
+  "<|1.76|>": 50453,
+  "<|1.78|>": 50454,
+  "<|1.80|>": 50455,
+  "<|1.82|>": 50456,
+  "<|1.84|>": 50457,
+  "<|1.86|>": 50458,
+  "<|1.88|>": 50459,
+  "<|1.90|>": 50460,
+  "<|1.92|>": 50461,
+  "<|1.94|>": 50462,
+  "<|1.96|>": 50463,
+  "<|1.98|>": 50464,
+  "<|10.00|>": 50865,
+  "<|10.02|>": 50866,
+  "<|10.04|>": 50867,
+  "<|10.06|>": 50868,
+  "<|10.08|>": 50869,
+  "<|10.10|>": 50870,
+  "<|10.12|>": 50871,
+  "<|10.14|>": 50872,
+  "<|10.16|>": 50873,
+  "<|10.18|>": 50874,
+  "<|10.20|>": 50875,
+  "<|10.22|>": 50876,
+  "<|10.24|>": 50877,
+  "<|10.26|>": 50878,
+  "<|10.28|>": 50879,
+  "<|10.30|>": 50880,
+  "<|10.32|>": 50881,
+  "<|10.34|>": 50882,
+  "<|10.36|>": 50883,
+  "<|10.38|>": 50884,
+  "<|10.40|>": 50885,
+  "<|10.42|>": 50886,
+  "<|10.44|>": 50887,
+  "<|10.46|>": 50888,
+  "<|10.48|>": 50889,
+  "<|10.50|>": 50890,
+  "<|10.52|>": 50891,
+  "<|10.54|>": 50892,
+  "<|10.56|>": 50893,
+  "<|10.58|>": 50894,
+  "<|10.60|>": 50895,
+  "<|10.62|>": 50896,
+  "<|10.64|>": 50897,
+  "<|10.66|>": 50898,
+  "<|10.68|>": 50899,
+  "<|10.70|>": 50900,
+  "<|10.72|>": 50901,
+  "<|10.74|>": 50902,
+  "<|10.76|>": 50903,
+  "<|10.78|>": 50904,
+  "<|10.80|>": 50905,
+  "<|10.82|>": 50906,
+  "<|10.84|>": 50907,
+  "<|10.86|>": 50908,
+  "<|10.88|>": 50909,
+  "<|10.90|>": 50910,
+  "<|10.92|>": 50911,
+  "<|10.94|>": 50912,
+  "<|10.96|>": 50913,
+  "<|10.98|>": 50914,
+  "<|11.00|>": 50915,
+  "<|11.02|>": 50916,
+  "<|11.04|>": 50917,
+  "<|11.06|>": 50918,
+  "<|11.08|>": 50919,
+  "<|11.10|>": 50920,
+  "<|11.12|>": 50921,
+  "<|11.14|>": 50922,
+  "<|11.16|>": 50923,
+  "<|11.18|>": 50924,
+  "<|11.20|>": 50925,
+  "<|11.22|>": 50926,
+  "<|11.24|>": 50927,
+  "<|11.26|>": 50928,
+  "<|11.28|>": 50929,
+  "<|11.30|>": 50930,
+  "<|11.32|>": 50931,
+  "<|11.34|>": 50932,
+  "<|11.36|>": 50933,
+  "<|11.38|>": 50934,
+  "<|11.40|>": 50935,
+  "<|11.42|>": 50936,
+  "<|11.44|>": 50937,
+  "<|11.46|>": 50938,
+  "<|11.48|>": 50939,
+  "<|11.50|>": 50940,
+  "<|11.52|>": 50941,
+  "<|11.54|>": 50942,
+  "<|11.56|>": 50943,
+  "<|11.58|>": 50944,
+  "<|11.60|>": 50945,
+  "<|11.62|>": 50946,
+  "<|11.64|>": 50947,
+  "<|11.66|>": 50948,
+  "<|11.68|>": 50949,
+  "<|11.70|>": 50950,
+  "<|11.72|>": 50951,
+  "<|11.74|>": 50952,
+  "<|11.76|>": 50953,
+  "<|11.78|>": 50954,
+  "<|11.80|>": 50955,
+  "<|11.82|>": 50956,
+  "<|11.84|>": 50957,
+  "<|11.86|>": 50958,
+  "<|11.88|>": 50959,
+  "<|11.90|>": 50960,
+  "<|11.92|>": 50961,
+  "<|11.94|>": 50962,
+  "<|11.96|>": 50963,
+  "<|11.98|>": 50964,
+  "<|12.00|>": 50965,
+  "<|12.02|>": 50966,
+  "<|12.04|>": 50967,
+  "<|12.06|>": 50968,
+  "<|12.08|>": 50969,
+  "<|12.10|>": 50970,
+  "<|12.12|>": 50971,
+  "<|12.14|>": 50972,
+  "<|12.16|>": 50973,
+  "<|12.18|>": 50974,
+  "<|12.20|>": 50975,
+  "<|12.22|>": 50976,
+  "<|12.24|>": 50977,
+  "<|12.26|>": 50978,
+  "<|12.28|>": 50979,
+  "<|12.30|>": 50980,
+  "<|12.32|>": 50981,
+  "<|12.34|>": 50982,
+  "<|12.36|>": 50983,
+  "<|12.38|>": 50984,
+  "<|12.40|>": 50985,
+  "<|12.42|>": 50986,
+  "<|12.44|>": 50987,
+  "<|12.46|>": 50988,
+  "<|12.48|>": 50989,
+  "<|12.50|>": 50990,
+  "<|12.52|>": 50991,
+  "<|12.54|>": 50992,
+  "<|12.56|>": 50993,
+  "<|12.58|>": 50994,
+  "<|12.60|>": 50995,
+  "<|12.62|>": 50996,
+  "<|12.64|>": 50997,
+  "<|12.66|>": 50998,
+  "<|12.68|>": 50999,
+  "<|12.70|>": 51000,
+  "<|12.72|>": 51001,
+  "<|12.74|>": 51002,
+  "<|12.76|>": 51003,
+  "<|12.78|>": 51004,
+  "<|12.80|>": 51005,
+  "<|12.82|>": 51006,
+  "<|12.84|>": 51007,
+  "<|12.86|>": 51008,
+  "<|12.88|>": 51009,
+  "<|12.90|>": 51010,
+  "<|12.92|>": 51011,
+  "<|12.94|>": 51012,
+  "<|12.96|>": 51013,
+  "<|12.98|>": 51014,
+  "<|13.00|>": 51015,
+  "<|13.02|>": 51016,
+  "<|13.04|>": 51017,
+  "<|13.06|>": 51018,
+  "<|13.08|>": 51019,
+  "<|13.10|>": 51020,
+  "<|13.12|>": 51021,
+  "<|13.14|>": 51022,
+  "<|13.16|>": 51023,
+  "<|13.18|>": 51024,
+  "<|13.20|>": 51025,
+  "<|13.22|>": 51026,
+  "<|13.24|>": 51027,
+  "<|13.26|>": 51028,
+  "<|13.28|>": 51029,
+  "<|13.30|>": 51030,
+  "<|13.32|>": 51031,
+  "<|13.34|>": 51032,
+  "<|13.36|>": 51033,
+  "<|13.38|>": 51034,
+  "<|13.40|>": 51035,
+  "<|13.42|>": 51036,
+  "<|13.44|>": 51037,
+  "<|13.46|>": 51038,
+  "<|13.48|>": 51039,
+  "<|13.50|>": 51040,
+  "<|13.52|>": 51041,
+  "<|13.54|>": 51042,
+  "<|13.56|>": 51043,
+  "<|13.58|>": 51044,
+  "<|13.60|>": 51045,
+  "<|13.62|>": 51046,
+  "<|13.64|>": 51047,
+  "<|13.66|>": 51048,
+  "<|13.68|>": 51049,
+  "<|13.70|>": 51050,
+  "<|13.72|>": 51051,
+  "<|13.74|>": 51052,
+  "<|13.76|>": 51053,
+  "<|13.78|>": 51054,
+  "<|13.80|>": 51055,
+  "<|13.82|>": 51056,
+  "<|13.84|>": 51057,
+  "<|13.86|>": 51058,
+  "<|13.88|>": 51059,
+  "<|13.90|>": 51060,
+  "<|13.92|>": 51061,
+  "<|13.94|>": 51062,
+  "<|13.96|>": 51063,
+  "<|13.98|>": 51064,
+  "<|14.00|>": 51065,
+  "<|14.02|>": 51066,
+  "<|14.04|>": 51067,
+  "<|14.06|>": 51068,
+  "<|14.08|>": 51069,
+  "<|14.10|>": 51070,
+  "<|14.12|>": 51071,
+  "<|14.14|>": 51072,
+  "<|14.16|>": 51073,
+  "<|14.18|>": 51074,
+  "<|14.20|>": 51075,
+  "<|14.22|>": 51076,
+  "<|14.24|>": 51077,
+  "<|14.26|>": 51078,
+  "<|14.28|>": 51079,
+  "<|14.30|>": 51080,
+  "<|14.32|>": 51081,
+  "<|14.34|>": 51082,
+  "<|14.36|>": 51083,
+  "<|14.38|>": 51084,
+  "<|14.40|>": 51085,
+  "<|14.42|>": 51086,
+  "<|14.44|>": 51087,
+  "<|14.46|>": 51088,
+  "<|14.48|>": 51089,
+  "<|14.50|>": 51090,
+  "<|14.52|>": 51091,
+  "<|14.54|>": 51092,
+  "<|14.56|>": 51093,
+  "<|14.58|>": 51094,
+  "<|14.60|>": 51095,
+  "<|14.62|>": 51096,
+  "<|14.64|>": 51097,
+  "<|14.66|>": 51098,
+  "<|14.68|>": 51099,
+  "<|14.70|>": 51100,
+  "<|14.72|>": 51101,
+  "<|14.74|>": 51102,
+  "<|14.76|>": 51103,
+  "<|14.78|>": 51104,
+  "<|14.80|>": 51105,
+  "<|14.82|>": 51106,
+  "<|14.84|>": 51107,
+  "<|14.86|>": 51108,
+  "<|14.88|>": 51109,
+  "<|14.90|>": 51110,
+  "<|14.92|>": 51111,
+  "<|14.94|>": 51112,
+  "<|14.96|>": 51113,
+  "<|14.98|>": 51114,
+  "<|15.00|>": 51115,
+  "<|15.02|>": 51116,
+  "<|15.04|>": 51117,
+  "<|15.06|>": 51118,
+  "<|15.08|>": 51119,
+  "<|15.10|>": 51120,
+  "<|15.12|>": 51121,
+  "<|15.14|>": 51122,
+  "<|15.16|>": 51123,
+  "<|15.18|>": 51124,
+  "<|15.20|>": 51125,
+  "<|15.22|>": 51126,
+  "<|15.24|>": 51127,
+  "<|15.26|>": 51128,
+  "<|15.28|>": 51129,
+  "<|15.30|>": 51130,
+  "<|15.32|>": 51131,
+  "<|15.34|>": 51132,
+  "<|15.36|>": 51133,
+  "<|15.38|>": 51134,
+  "<|15.40|>": 51135,
+  "<|15.42|>": 51136,
+  "<|15.44|>": 51137,
+  "<|15.46|>": 51138,
+  "<|15.48|>": 51139,
+  "<|15.50|>": 51140,
+  "<|15.52|>": 51141,
+  "<|15.54|>": 51142,
+  "<|15.56|>": 51143,
+  "<|15.58|>": 51144,
+  "<|15.60|>": 51145,
+  "<|15.62|>": 51146,
+  "<|15.64|>": 51147,
+  "<|15.66|>": 51148,
+  "<|15.68|>": 51149,
+  "<|15.70|>": 51150,
+  "<|15.72|>": 51151,
+  "<|15.74|>": 51152,
+  "<|15.76|>": 51153,
+  "<|15.78|>": 51154,
+  "<|15.80|>": 51155,
+  "<|15.82|>": 51156,
+  "<|15.84|>": 51157,
+  "<|15.86|>": 51158,
+  "<|15.88|>": 51159,
+  "<|15.90|>": 51160,
+  "<|15.92|>": 51161,
+  "<|15.94|>": 51162,
+  "<|15.96|>": 51163,
+  "<|15.98|>": 51164,
+  "<|16.00|>": 51165,
+  "<|16.02|>": 51166,
+  "<|16.04|>": 51167,
+  "<|16.06|>": 51168,
+  "<|16.08|>": 51169,
+  "<|16.10|>": 51170,
+  "<|16.12|>": 51171,
+  "<|16.14|>": 51172,
+  "<|16.16|>": 51173,
+  "<|16.18|>": 51174,
+  "<|16.20|>": 51175,
+  "<|16.22|>": 51176,
+  "<|16.24|>": 51177,
+  "<|16.26|>": 51178,
+  "<|16.28|>": 51179,
+  "<|16.30|>": 51180,
+  "<|16.32|>": 51181,
+  "<|16.34|>": 51182,
+  "<|16.36|>": 51183,
+  "<|16.38|>": 51184,
+  "<|16.40|>": 51185,
+  "<|16.42|>": 51186,
+  "<|16.44|>": 51187,
+  "<|16.46|>": 51188,
+  "<|16.48|>": 51189,
+  "<|16.50|>": 51190,
+  "<|16.52|>": 51191,
+  "<|16.54|>": 51192,
+  "<|16.56|>": 51193,
+  "<|16.58|>": 51194,
+  "<|16.60|>": 51195,
+  "<|16.62|>": 51196,
+  "<|16.64|>": 51197,
+  "<|16.66|>": 51198,
+  "<|16.68|>": 51199,
+  "<|16.70|>": 51200,
+  "<|16.72|>": 51201,
+  "<|16.74|>": 51202,
+  "<|16.76|>": 51203,
+  "<|16.78|>": 51204,
+  "<|16.80|>": 51205,
+  "<|16.82|>": 51206,
+  "<|16.84|>": 51207,
+  "<|16.86|>": 51208,
+  "<|16.88|>": 51209,
+  "<|16.90|>": 51210,
+  "<|16.92|>": 51211,
+  "<|16.94|>": 51212,
+  "<|16.96|>": 51213,
+  "<|16.98|>": 51214,
+  "<|17.00|>": 51215,
+  "<|17.02|>": 51216,
+  "<|17.04|>": 51217,
+  "<|17.06|>": 51218,
+  "<|17.08|>": 51219,
+  "<|17.10|>": 51220,
+  "<|17.12|>": 51221,
+  "<|17.14|>": 51222,
+  "<|17.16|>": 51223,
+  "<|17.18|>": 51224,
+  "<|17.20|>": 51225,
+  "<|17.22|>": 51226,
+  "<|17.24|>": 51227,
+  "<|17.26|>": 51228,
+  "<|17.28|>": 51229,
+  "<|17.30|>": 51230,
+  "<|17.32|>": 51231,
+  "<|17.34|>": 51232,
+  "<|17.36|>": 51233,
+  "<|17.38|>": 51234,
+  "<|17.40|>": 51235,
+  "<|17.42|>": 51236,
+  "<|17.44|>": 51237,
+  "<|17.46|>": 51238,
+  "<|17.48|>": 51239,
+  "<|17.50|>": 51240,
+  "<|17.52|>": 51241,
+  "<|17.54|>": 51242,
+  "<|17.56|>": 51243,
+  "<|17.58|>": 51244,
+  "<|17.60|>": 51245,
+  "<|17.62|>": 51246,
+  "<|17.64|>": 51247,
+  "<|17.66|>": 51248,
+  "<|17.68|>": 51249,
+  "<|17.70|>": 51250,
+  "<|17.72|>": 51251,
+  "<|17.74|>": 51252,
+  "<|17.76|>": 51253,
+  "<|17.78|>": 51254,
+  "<|17.80|>": 51255,
+  "<|17.82|>": 51256,
+  "<|17.84|>": 51257,
+  "<|17.86|>": 51258,
+  "<|17.88|>": 51259,
+  "<|17.90|>": 51260,
+  "<|17.92|>": 51261,
+  "<|17.94|>": 51262,
+  "<|17.96|>": 51263,
+  "<|17.98|>": 51264,
+  "<|18.00|>": 51265,
+  "<|18.02|>": 51266,
+  "<|18.04|>": 51267,
+  "<|18.06|>": 51268,
+  "<|18.08|>": 51269,
+  "<|18.10|>": 51270,
+  "<|18.12|>": 51271,
+  "<|18.14|>": 51272,
+  "<|18.16|>": 51273,
+  "<|18.18|>": 51274,
+  "<|18.20|>": 51275,
+  "<|18.22|>": 51276,
+  "<|18.24|>": 51277,
+  "<|18.26|>": 51278,
+  "<|18.28|>": 51279,
+  "<|18.30|>": 51280,
+  "<|18.32|>": 51281,
+  "<|18.34|>": 51282,
+  "<|18.36|>": 51283,
+  "<|18.38|>": 51284,
+  "<|18.40|>": 51285,
+  "<|18.42|>": 51286,
+  "<|18.44|>": 51287,
+  "<|18.46|>": 51288,
+  "<|18.48|>": 51289,
+  "<|18.50|>": 51290,
+  "<|18.52|>": 51291,
+  "<|18.54|>": 51292,
+  "<|18.56|>": 51293,
+  "<|18.58|>": 51294,
+  "<|18.60|>": 51295,
+  "<|18.62|>": 51296,
+  "<|18.64|>": 51297,
+  "<|18.66|>": 51298,
+  "<|18.68|>": 51299,
+  "<|18.70|>": 51300,
+  "<|18.72|>": 51301,
+  "<|18.74|>": 51302,
+  "<|18.76|>": 51303,
+  "<|18.78|>": 51304,
+  "<|18.80|>": 51305,
+  "<|18.82|>": 51306,
+  "<|18.84|>": 51307,
+  "<|18.86|>": 51308,
+  "<|18.88|>": 51309,
+  "<|18.90|>": 51310,
+  "<|18.92|>": 51311,
+  "<|18.94|>": 51312,
+  "<|18.96|>": 51313,
+  "<|18.98|>": 51314,
+  "<|19.00|>": 51315,
+  "<|19.02|>": 51316,
+  "<|19.04|>": 51317,
+  "<|19.06|>": 51318,
+  "<|19.08|>": 51319,
+  "<|19.10|>": 51320,
+  "<|19.12|>": 51321,
+  "<|19.14|>": 51322,
+  "<|19.16|>": 51323,
+  "<|19.18|>": 51324,
+  "<|19.20|>": 51325,
+  "<|19.22|>": 51326,
+  "<|19.24|>": 51327,
+  "<|19.26|>": 51328,
+  "<|19.28|>": 51329,
+  "<|19.30|>": 51330,
+  "<|19.32|>": 51331,
+  "<|19.34|>": 51332,
+  "<|19.36|>": 51333,
+  "<|19.38|>": 51334,
+  "<|19.40|>": 51335,
+  "<|19.42|>": 51336,
+  "<|19.44|>": 51337,
+  "<|19.46|>": 51338,
+  "<|19.48|>": 51339,
+  "<|19.50|>": 51340,
+  "<|19.52|>": 51341,
+  "<|19.54|>": 51342,
+  "<|19.56|>": 51343,
+  "<|19.58|>": 51344,
+  "<|19.60|>": 51345,
+  "<|19.62|>": 51346,
+  "<|19.64|>": 51347,
+  "<|19.66|>": 51348,
+  "<|19.68|>": 51349,
+  "<|19.70|>": 51350,
+  "<|19.72|>": 51351,
+  "<|19.74|>": 51352,
+  "<|19.76|>": 51353,
+  "<|19.78|>": 51354,
+  "<|19.80|>": 51355,
+  "<|19.82|>": 51356,
+  "<|19.84|>": 51357,
+  "<|19.86|>": 51358,
+  "<|19.88|>": 51359,
+  "<|19.90|>": 51360,
+  "<|19.92|>": 51361,
+  "<|19.94|>": 51362,
+  "<|19.96|>": 51363,
+  "<|19.98|>": 51364,
+  "<|2.00|>": 50465,
+  "<|2.02|>": 50466,
+  "<|2.04|>": 50467,
+  "<|2.06|>": 50468,
+  "<|2.08|>": 50469,
+  "<|2.10|>": 50470,
+  "<|2.12|>": 50471,
+  "<|2.14|>": 50472,
+  "<|2.16|>": 50473,
+  "<|2.18|>": 50474,
+  "<|2.20|>": 50475,
+  "<|2.22|>": 50476,
+  "<|2.24|>": 50477,
+  "<|2.26|>": 50478,
+  "<|2.28|>": 50479,
+  "<|2.30|>": 50480,
+  "<|2.32|>": 50481,
+  "<|2.34|>": 50482,
+  "<|2.36|>": 50483,
+  "<|2.38|>": 50484,
+  "<|2.40|>": 50485,
+  "<|2.42|>": 50486,
+  "<|2.44|>": 50487,
+  "<|2.46|>": 50488,
+  "<|2.48|>": 50489,
+  "<|2.50|>": 50490,
+  "<|2.52|>": 50491,
+  "<|2.54|>": 50492,
+  "<|2.56|>": 50493,
+  "<|2.58|>": 50494,
+  "<|2.60|>": 50495,
+  "<|2.62|>": 50496,
+  "<|2.64|>": 50497,
+  "<|2.66|>": 50498,
+  "<|2.68|>": 50499,
+  "<|2.70|>": 50500,
+  "<|2.72|>": 50501,
+  "<|2.74|>": 50502,
+  "<|2.76|>": 50503,
+  "<|2.78|>": 50504,
+  "<|2.80|>": 50505,
+  "<|2.82|>": 50506,
+  "<|2.84|>": 50507,
+  "<|2.86|>": 50508,
+  "<|2.88|>": 50509,
+  "<|2.90|>": 50510,
+  "<|2.92|>": 50511,
+  "<|2.94|>": 50512,
+  "<|2.96|>": 50513,
+  "<|2.98|>": 50514,
+  "<|20.00|>": 51365,
+  "<|20.02|>": 51366,
+  "<|20.04|>": 51367,
+  "<|20.06|>": 51368,
+  "<|20.08|>": 51369,
+  "<|20.10|>": 51370,
+  "<|20.12|>": 51371,
+  "<|20.14|>": 51372,
+  "<|20.16|>": 51373,
+  "<|20.18|>": 51374,
+  "<|20.20|>": 51375,
+  "<|20.22|>": 51376,
+  "<|20.24|>": 51377,
+  "<|20.26|>": 51378,
+  "<|20.28|>": 51379,
+  "<|20.30|>": 51380,
+  "<|20.32|>": 51381,
+  "<|20.34|>": 51382,
+  "<|20.36|>": 51383,
+  "<|20.38|>": 51384,
+  "<|20.40|>": 51385,
+  "<|20.42|>": 51386,
+  "<|20.44|>": 51387,
+  "<|20.46|>": 51388,
+  "<|20.48|>": 51389,
+  "<|20.50|>": 51390,
+  "<|20.52|>": 51391,
+  "<|20.54|>": 51392,
+  "<|20.56|>": 51393,
+  "<|20.58|>": 51394,
+  "<|20.60|>": 51395,
+  "<|20.62|>": 51396,
+  "<|20.64|>": 51397,
+  "<|20.66|>": 51398,
+  "<|20.68|>": 51399,
+  "<|20.70|>": 51400,
+  "<|20.72|>": 51401,
+  "<|20.74|>": 51402,
+  "<|20.76|>": 51403,
+  "<|20.78|>": 51404,
+  "<|20.80|>": 51405,
+  "<|20.82|>": 51406,
+  "<|20.84|>": 51407,
+  "<|20.86|>": 51408,
+  "<|20.88|>": 51409,
+  "<|20.90|>": 51410,
+  "<|20.92|>": 51411,
+  "<|20.94|>": 51412,
+  "<|20.96|>": 51413,
+  "<|20.98|>": 51414,
+  "<|21.00|>": 51415,
+  "<|21.02|>": 51416,
+  "<|21.04|>": 51417,
+  "<|21.06|>": 51418,
+  "<|21.08|>": 51419,
+  "<|21.10|>": 51420,
+  "<|21.12|>": 51421,
+  "<|21.14|>": 51422,
+  "<|21.16|>": 51423,
+  "<|21.18|>": 51424,
+  "<|21.20|>": 51425,
+  "<|21.22|>": 51426,
+  "<|21.24|>": 51427,
+  "<|21.26|>": 51428,
+  "<|21.28|>": 51429,
+  "<|21.30|>": 51430,
+  "<|21.32|>": 51431,
+  "<|21.34|>": 51432,
+  "<|21.36|>": 51433,
+  "<|21.38|>": 51434,
+  "<|21.40|>": 51435,
+  "<|21.42|>": 51436,
+  "<|21.44|>": 51437,
+  "<|21.46|>": 51438,
+  "<|21.48|>": 51439,
+  "<|21.50|>": 51440,
+  "<|21.52|>": 51441,
+  "<|21.54|>": 51442,
+  "<|21.56|>": 51443,
+  "<|21.58|>": 51444,
+  "<|21.60|>": 51445,
+  "<|21.62|>": 51446,
+  "<|21.64|>": 51447,
+  "<|21.66|>": 51448,
+  "<|21.68|>": 51449,
+  "<|21.70|>": 51450,
+  "<|21.72|>": 51451,
+  "<|21.74|>": 51452,
+  "<|21.76|>": 51453,
+  "<|21.78|>": 51454,
+  "<|21.80|>": 51455,
+  "<|21.82|>": 51456,
+  "<|21.84|>": 51457,
+  "<|21.86|>": 51458,
+  "<|21.88|>": 51459,
+  "<|21.90|>": 51460,
+  "<|21.92|>": 51461,
+  "<|21.94|>": 51462,
+  "<|21.96|>": 51463,
+  "<|21.98|>": 51464,
+  "<|22.00|>": 51465,
+  "<|22.02|>": 51466,
+  "<|22.04|>": 51467,
+  "<|22.06|>": 51468,
+  "<|22.08|>": 51469,
+  "<|22.10|>": 51470,
+  "<|22.12|>": 51471,
+  "<|22.14|>": 51472,
+  "<|22.16|>": 51473,
+  "<|22.18|>": 51474,
+  "<|22.20|>": 51475,
+  "<|22.22|>": 51476,
+  "<|22.24|>": 51477,
+  "<|22.26|>": 51478,
+  "<|22.28|>": 51479,
+  "<|22.30|>": 51480,
+  "<|22.32|>": 51481,
+  "<|22.34|>": 51482,
+  "<|22.36|>": 51483,
+  "<|22.38|>": 51484,
+  "<|22.40|>": 51485,
+  "<|22.42|>": 51486,
+  "<|22.44|>": 51487,
+  "<|22.46|>": 51488,
+  "<|22.48|>": 51489,
+  "<|22.50|>": 51490,
+  "<|22.52|>": 51491,
+  "<|22.54|>": 51492,
+  "<|22.56|>": 51493,
+  "<|22.58|>": 51494,
+  "<|22.60|>": 51495,
+  "<|22.62|>": 51496,
+  "<|22.64|>": 51497,
+  "<|22.66|>": 51498,
+  "<|22.68|>": 51499,
+  "<|22.70|>": 51500,
+  "<|22.72|>": 51501,
+  "<|22.74|>": 51502,
+  "<|22.76|>": 51503,
+  "<|22.78|>": 51504,
+  "<|22.80|>": 51505,
+  "<|22.82|>": 51506,
+  "<|22.84|>": 51507,
+  "<|22.86|>": 51508,
+  "<|22.88|>": 51509,
+  "<|22.90|>": 51510,
+  "<|22.92|>": 51511,
+  "<|22.94|>": 51512,
+  "<|22.96|>": 51513,
+  "<|22.98|>": 51514,
+  "<|23.00|>": 51515,
+  "<|23.02|>": 51516,
+  "<|23.04|>": 51517,
+  "<|23.06|>": 51518,
+  "<|23.08|>": 51519,
+  "<|23.10|>": 51520,
+  "<|23.12|>": 51521,
+  "<|23.14|>": 51522,
+  "<|23.16|>": 51523,
+  "<|23.18|>": 51524,
+  "<|23.20|>": 51525,
+  "<|23.22|>": 51526,
+  "<|23.24|>": 51527,
+  "<|23.26|>": 51528,
+  "<|23.28|>": 51529,
+  "<|23.30|>": 51530,
+  "<|23.32|>": 51531,
+  "<|23.34|>": 51532,
+  "<|23.36|>": 51533,
+  "<|23.38|>": 51534,
+  "<|23.40|>": 51535,
+  "<|23.42|>": 51536,
+  "<|23.44|>": 51537,
+  "<|23.46|>": 51538,
+  "<|23.48|>": 51539,
+  "<|23.50|>": 51540,
+  "<|23.52|>": 51541,
+  "<|23.54|>": 51542,
+  "<|23.56|>": 51543,
+  "<|23.58|>": 51544,
+  "<|23.60|>": 51545,
+  "<|23.62|>": 51546,
+  "<|23.64|>": 51547,
+  "<|23.66|>": 51548,
+  "<|23.68|>": 51549,
+  "<|23.70|>": 51550,
+  "<|23.72|>": 51551,
+  "<|23.74|>": 51552,
+  "<|23.76|>": 51553,
+  "<|23.78|>": 51554,
+  "<|23.80|>": 51555,
+  "<|23.82|>": 51556,
+  "<|23.84|>": 51557,
+  "<|23.86|>": 51558,
+  "<|23.88|>": 51559,
+  "<|23.90|>": 51560,
+  "<|23.92|>": 51561,
+  "<|23.94|>": 51562,
+  "<|23.96|>": 51563,
+  "<|23.98|>": 51564,
+  "<|24.00|>": 51565,
+  "<|24.02|>": 51566,
+  "<|24.04|>": 51567,
+  "<|24.06|>": 51568,
+  "<|24.08|>": 51569,
+  "<|24.10|>": 51570,
+  "<|24.12|>": 51571,
+  "<|24.14|>": 51572,
+  "<|24.16|>": 51573,
+  "<|24.18|>": 51574,
+  "<|24.20|>": 51575,
+  "<|24.22|>": 51576,
+  "<|24.24|>": 51577,
+  "<|24.26|>": 51578,
+  "<|24.28|>": 51579,
+  "<|24.30|>": 51580,
+  "<|24.32|>": 51581,
+  "<|24.34|>": 51582,
+  "<|24.36|>": 51583,
+  "<|24.38|>": 51584,
+  "<|24.40|>": 51585,
+  "<|24.42|>": 51586,
+  "<|24.44|>": 51587,
+  "<|24.46|>": 51588,
+  "<|24.48|>": 51589,
+  "<|24.50|>": 51590,
+  "<|24.52|>": 51591,
+  "<|24.54|>": 51592,
+  "<|24.56|>": 51593,
+  "<|24.58|>": 51594,
+  "<|24.60|>": 51595,
+  "<|24.62|>": 51596,
+  "<|24.64|>": 51597,
+  "<|24.66|>": 51598,
+  "<|24.68|>": 51599,
+  "<|24.70|>": 51600,
+  "<|24.72|>": 51601,
+  "<|24.74|>": 51602,
+  "<|24.76|>": 51603,
+  "<|24.78|>": 51604,
+  "<|24.80|>": 51605,
+  "<|24.82|>": 51606,
+  "<|24.84|>": 51607,
+  "<|24.86|>": 51608,
+  "<|24.88|>": 51609,
+  "<|24.90|>": 51610,
+  "<|24.92|>": 51611,
+  "<|24.94|>": 51612,
+  "<|24.96|>": 51613,
+  "<|24.98|>": 51614,
+  "<|25.00|>": 51615,
+  "<|25.02|>": 51616,
+  "<|25.04|>": 51617,
+  "<|25.06|>": 51618,
+  "<|25.08|>": 51619,
+  "<|25.10|>": 51620,
+  "<|25.12|>": 51621,
+  "<|25.14|>": 51622,
+  "<|25.16|>": 51623,
+  "<|25.18|>": 51624,
+  "<|25.20|>": 51625,
+  "<|25.22|>": 51626,
+  "<|25.24|>": 51627,
+  "<|25.26|>": 51628,
+  "<|25.28|>": 51629,
+  "<|25.30|>": 51630,
+  "<|25.32|>": 51631,
+  "<|25.34|>": 51632,
+  "<|25.36|>": 51633,
+  "<|25.38|>": 51634,
+  "<|25.40|>": 51635,
+  "<|25.42|>": 51636,
+  "<|25.44|>": 51637,
+  "<|25.46|>": 51638,
+  "<|25.48|>": 51639,
+  "<|25.50|>": 51640,
+  "<|25.52|>": 51641,
+  "<|25.54|>": 51642,
+  "<|25.56|>": 51643,
+  "<|25.58|>": 51644,
+  "<|25.60|>": 51645,
+  "<|25.62|>": 51646,
+  "<|25.64|>": 51647,
+  "<|25.66|>": 51648,
+  "<|25.68|>": 51649,
+  "<|25.70|>": 51650,
+  "<|25.72|>": 51651,
+  "<|25.74|>": 51652,
+  "<|25.76|>": 51653,
+  "<|25.78|>": 51654,
+  "<|25.80|>": 51655,
+  "<|25.82|>": 51656,
+  "<|25.84|>": 51657,
+  "<|25.86|>": 51658,
+  "<|25.88|>": 51659,
+  "<|25.90|>": 51660,
+  "<|25.92|>": 51661,
+  "<|25.94|>": 51662,
+  "<|25.96|>": 51663,
+  "<|25.98|>": 51664,
+  "<|26.00|>": 51665,
+  "<|26.02|>": 51666,
+  "<|26.04|>": 51667,
+  "<|26.06|>": 51668,
+  "<|26.08|>": 51669,
+  "<|26.10|>": 51670,
+  "<|26.12|>": 51671,
+  "<|26.14|>": 51672,
+  "<|26.16|>": 51673,
+  "<|26.18|>": 51674,
+  "<|26.20|>": 51675,
+  "<|26.22|>": 51676,
+  "<|26.24|>": 51677,
+  "<|26.26|>": 51678,
+  "<|26.28|>": 51679,
+  "<|26.30|>": 51680,
+  "<|26.32|>": 51681,
+  "<|26.34|>": 51682,
+  "<|26.36|>": 51683,
+  "<|26.38|>": 51684,
+  "<|26.40|>": 51685,
+  "<|26.42|>": 51686,
+  "<|26.44|>": 51687,
+  "<|26.46|>": 51688,
+  "<|26.48|>": 51689,
+  "<|26.50|>": 51690,
+  "<|26.52|>": 51691,
+  "<|26.54|>": 51692,
+  "<|26.56|>": 51693,
+  "<|26.58|>": 51694,
+  "<|26.60|>": 51695,
+  "<|26.62|>": 51696,
+  "<|26.64|>": 51697,
+  "<|26.66|>": 51698,
+  "<|26.68|>": 51699,
+  "<|26.70|>": 51700,
+  "<|26.72|>": 51701,
+  "<|26.74|>": 51702,
+  "<|26.76|>": 51703,
+  "<|26.78|>": 51704,
+  "<|26.80|>": 51705,
+  "<|26.82|>": 51706,
+  "<|26.84|>": 51707,
+  "<|26.86|>": 51708,
+  "<|26.88|>": 51709,
+  "<|26.90|>": 51710,
+  "<|26.92|>": 51711,
+  "<|26.94|>": 51712,
+  "<|26.96|>": 51713,
+  "<|26.98|>": 51714,
+  "<|27.00|>": 51715,
+  "<|27.02|>": 51716,
+  "<|27.04|>": 51717,
+  "<|27.06|>": 51718,
+  "<|27.08|>": 51719,
+  "<|27.10|>": 51720,
+  "<|27.12|>": 51721,
+  "<|27.14|>": 51722,
+  "<|27.16|>": 51723,
+  "<|27.18|>": 51724,
+  "<|27.20|>": 51725,
+  "<|27.22|>": 51726,
+  "<|27.24|>": 51727,
+  "<|27.26|>": 51728,
+  "<|27.28|>": 51729,
+  "<|27.30|>": 51730,
+  "<|27.32|>": 51731,
+  "<|27.34|>": 51732,
+  "<|27.36|>": 51733,
+  "<|27.38|>": 51734,
+  "<|27.40|>": 51735,
+  "<|27.42|>": 51736,
+  "<|27.44|>": 51737,
+  "<|27.46|>": 51738,
+  "<|27.48|>": 51739,
+  "<|27.50|>": 51740,
+  "<|27.52|>": 51741,
+  "<|27.54|>": 51742,
+  "<|27.56|>": 51743,
+  "<|27.58|>": 51744,
+  "<|27.60|>": 51745,
+  "<|27.62|>": 51746,
+  "<|27.64|>": 51747,
+  "<|27.66|>": 51748,
+  "<|27.68|>": 51749,
+  "<|27.70|>": 51750,
+  "<|27.72|>": 51751,
+  "<|27.74|>": 51752,
+  "<|27.76|>": 51753,
+  "<|27.78|>": 51754,
+  "<|27.80|>": 51755,
+  "<|27.82|>": 51756,
+  "<|27.84|>": 51757,
+  "<|27.86|>": 51758,
+  "<|27.88|>": 51759,
+  "<|27.90|>": 51760,
+  "<|27.92|>": 51761,
+  "<|27.94|>": 51762,
+  "<|27.96|>": 51763,
+  "<|27.98|>": 51764,
+  "<|28.00|>": 51765,
+  "<|28.02|>": 51766,
+  "<|28.04|>": 51767,
+  "<|28.06|>": 51768,
+  "<|28.08|>": 51769,
+  "<|28.10|>": 51770,
+  "<|28.12|>": 51771,
+  "<|28.14|>": 51772,
+  "<|28.16|>": 51773,
+  "<|28.18|>": 51774,
+  "<|28.20|>": 51775,
+  "<|28.22|>": 51776,
+  "<|28.24|>": 51777,
+  "<|28.26|>": 51778,
+  "<|28.28|>": 51779,
+  "<|28.30|>": 51780,
+  "<|28.32|>": 51781,
+  "<|28.34|>": 51782,
+  "<|28.36|>": 51783,
+  "<|28.38|>": 51784,
+  "<|28.40|>": 51785,
+  "<|28.42|>": 51786,
+  "<|28.44|>": 51787,
+  "<|28.46|>": 51788,
+  "<|28.48|>": 51789,
+  "<|28.50|>": 51790,
+  "<|28.52|>": 51791,
+  "<|28.54|>": 51792,
+  "<|28.56|>": 51793,
+  "<|28.58|>": 51794,
+  "<|28.60|>": 51795,
+  "<|28.62|>": 51796,
+  "<|28.64|>": 51797,
+  "<|28.66|>": 51798,
+  "<|28.68|>": 51799,
+  "<|28.70|>": 51800,
+  "<|28.72|>": 51801,
+  "<|28.74|>": 51802,
+  "<|28.76|>": 51803,
+  "<|28.78|>": 51804,
+  "<|28.80|>": 51805,
+  "<|28.82|>": 51806,
+  "<|28.84|>": 51807,
+  "<|28.86|>": 51808,
+  "<|28.88|>": 51809,
+  "<|28.90|>": 51810,
+  "<|28.92|>": 51811,
+  "<|28.94|>": 51812,
+  "<|28.96|>": 51813,
+  "<|28.98|>": 51814,
+  "<|29.00|>": 51815,
+  "<|29.02|>": 51816,
+  "<|29.04|>": 51817,
+  "<|29.06|>": 51818,
+  "<|29.08|>": 51819,
+  "<|29.10|>": 51820,
+  "<|29.12|>": 51821,
+  "<|29.14|>": 51822,
+  "<|29.16|>": 51823,
+  "<|29.18|>": 51824,
+  "<|29.20|>": 51825,
+  "<|29.22|>": 51826,
+  "<|29.24|>": 51827,
+  "<|29.26|>": 51828,
+  "<|29.28|>": 51829,
+  "<|29.30|>": 51830,
+  "<|29.32|>": 51831,
+  "<|29.34|>": 51832,
+  "<|29.36|>": 51833,
+  "<|29.38|>": 51834,
+  "<|29.40|>": 51835,
+  "<|29.42|>": 51836,
+  "<|29.44|>": 51837,
+  "<|29.46|>": 51838,
+  "<|29.48|>": 51839,
+  "<|29.50|>": 51840,
+  "<|29.52|>": 51841,
+  "<|29.54|>": 51842,
+  "<|29.56|>": 51843,
+  "<|29.58|>": 51844,
+  "<|29.60|>": 51845,
+  "<|29.62|>": 51846,
+  "<|29.64|>": 51847,
+  "<|29.66|>": 51848,
+  "<|29.68|>": 51849,
+  "<|29.70|>": 51850,
+  "<|29.72|>": 51851,
+  "<|29.74|>": 51852,
+  "<|29.76|>": 51853,
+  "<|29.78|>": 51854,
+  "<|29.80|>": 51855,
+  "<|29.82|>": 51856,
+  "<|29.84|>": 51857,
+  "<|29.86|>": 51858,
+  "<|29.88|>": 51859,
+  "<|29.90|>": 51860,
+  "<|29.92|>": 51861,
+  "<|29.94|>": 51862,
+  "<|29.96|>": 51863,
+  "<|29.98|>": 51864,
+  "<|3.00|>": 50515,
+  "<|3.02|>": 50516,
+  "<|3.04|>": 50517,
+  "<|3.06|>": 50518,
+  "<|3.08|>": 50519,
+  "<|3.10|>": 50520,
+  "<|3.12|>": 50521,
+  "<|3.14|>": 50522,
+  "<|3.16|>": 50523,
+  "<|3.18|>": 50524,
+  "<|3.20|>": 50525,
+  "<|3.22|>": 50526,
+  "<|3.24|>": 50527,
+  "<|3.26|>": 50528,
+  "<|3.28|>": 50529,
+  "<|3.30|>": 50530,
+  "<|3.32|>": 50531,
+  "<|3.34|>": 50532,
+  "<|3.36|>": 50533,
+  "<|3.38|>": 50534,
+  "<|3.40|>": 50535,
+  "<|3.42|>": 50536,
+  "<|3.44|>": 50537,
+  "<|3.46|>": 50538,
+  "<|3.48|>": 50539,
+  "<|3.50|>": 50540,
+  "<|3.52|>": 50541,
+  "<|3.54|>": 50542,
+  "<|3.56|>": 50543,
+  "<|3.58|>": 50544,
+  "<|3.60|>": 50545,
+  "<|3.62|>": 50546,
+  "<|3.64|>": 50547,
+  "<|3.66|>": 50548,
+  "<|3.68|>": 50549,
+  "<|3.70|>": 50550,
+  "<|3.72|>": 50551,
+  "<|3.74|>": 50552,
+  "<|3.76|>": 50553,
+  "<|3.78|>": 50554,
+  "<|3.80|>": 50555,
+  "<|3.82|>": 50556,
+  "<|3.84|>": 50557,
+  "<|3.86|>": 50558,
+  "<|3.88|>": 50559,
+  "<|3.90|>": 50560,
+  "<|3.92|>": 50561,
+  "<|3.94|>": 50562,
+  "<|3.96|>": 50563,
+  "<|3.98|>": 50564,
+  "<|30.00|>": 51865,
+  "<|4.00|>": 50565,
+  "<|4.02|>": 50566,
+  "<|4.04|>": 50567,
+  "<|4.06|>": 50568,
+  "<|4.08|>": 50569,
+  "<|4.10|>": 50570,
+  "<|4.12|>": 50571,
+  "<|4.14|>": 50572,
+  "<|4.16|>": 50573,
+  "<|4.18|>": 50574,
+  "<|4.20|>": 50575,
+  "<|4.22|>": 50576,
+  "<|4.24|>": 50577,
+  "<|4.26|>": 50578,
+  "<|4.28|>": 50579,
+  "<|4.30|>": 50580,
+  "<|4.32|>": 50581,
+  "<|4.34|>": 50582,
+  "<|4.36|>": 50583,
+  "<|4.38|>": 50584,
+  "<|4.40|>": 50585,
+  "<|4.42|>": 50586,
+  "<|4.44|>": 50587,
+  "<|4.46|>": 50588,
+  "<|4.48|>": 50589,
+  "<|4.50|>": 50590,
+  "<|4.52|>": 50591,
+  "<|4.54|>": 50592,
+  "<|4.56|>": 50593,
+  "<|4.58|>": 50594,
+  "<|4.60|>": 50595,
+  "<|4.62|>": 50596,
+  "<|4.64|>": 50597,
+  "<|4.66|>": 50598,
+  "<|4.68|>": 50599,
+  "<|4.70|>": 50600,
+  "<|4.72|>": 50601,
+  "<|4.74|>": 50602,
+  "<|4.76|>": 50603,
+  "<|4.78|>": 50604,
+  "<|4.80|>": 50605,
+  "<|4.82|>": 50606,
+  "<|4.84|>": 50607,
+  "<|4.86|>": 50608,
+  "<|4.88|>": 50609,
+  "<|4.90|>": 50610,
+  "<|4.92|>": 50611,
+  "<|4.94|>": 50612,
+  "<|4.96|>": 50613,
+  "<|4.98|>": 50614,
+  "<|5.00|>": 50615,
+  "<|5.02|>": 50616,
+  "<|5.04|>": 50617,
+  "<|5.06|>": 50618,
+  "<|5.08|>": 50619,
+  "<|5.10|>": 50620,
+  "<|5.12|>": 50621,
+  "<|5.14|>": 50622,
+  "<|5.16|>": 50623,
+  "<|5.18|>": 50624,
+  "<|5.20|>": 50625,
+  "<|5.22|>": 50626,
+  "<|5.24|>": 50627,
+  "<|5.26|>": 50628,
+  "<|5.28|>": 50629,
+  "<|5.30|>": 50630,
+  "<|5.32|>": 50631,
+  "<|5.34|>": 50632,
+  "<|5.36|>": 50633,
+  "<|5.38|>": 50634,
+  "<|5.40|>": 50635,
+  "<|5.42|>": 50636,
+  "<|5.44|>": 50637,
+  "<|5.46|>": 50638,
+  "<|5.48|>": 50639,
+  "<|5.50|>": 50640,
+  "<|5.52|>": 50641,
+  "<|5.54|>": 50642,
+  "<|5.56|>": 50643,
+  "<|5.58|>": 50644,
+  "<|5.60|>": 50645,
+  "<|5.62|>": 50646,
+  "<|5.64|>": 50647,
+  "<|5.66|>": 50648,
+  "<|5.68|>": 50649,
+  "<|5.70|>": 50650,
+  "<|5.72|>": 50651,
+  "<|5.74|>": 50652,
+  "<|5.76|>": 50653,
+  "<|5.78|>": 50654,
+  "<|5.80|>": 50655,
+  "<|5.82|>": 50656,
+  "<|5.84|>": 50657,
+  "<|5.86|>": 50658,
+  "<|5.88|>": 50659,
+  "<|5.90|>": 50660,
+  "<|5.92|>": 50661,
+  "<|5.94|>": 50662,
+  "<|5.96|>": 50663,
+  "<|5.98|>": 50664,
+  "<|6.00|>": 50665,
+  "<|6.02|>": 50666,
+  "<|6.04|>": 50667,
+  "<|6.06|>": 50668,
+  "<|6.08|>": 50669,
+  "<|6.10|>": 50670,
+  "<|6.12|>": 50671,
+  "<|6.14|>": 50672,
+  "<|6.16|>": 50673,
+  "<|6.18|>": 50674,
+  "<|6.20|>": 50675,
+  "<|6.22|>": 50676,
+  "<|6.24|>": 50677,
+  "<|6.26|>": 50678,
+  "<|6.28|>": 50679,
+  "<|6.30|>": 50680,
+  "<|6.32|>": 50681,
+  "<|6.34|>": 50682,
+  "<|6.36|>": 50683,
+  "<|6.38|>": 50684,
+  "<|6.40|>": 50685,
+  "<|6.42|>": 50686,
+  "<|6.44|>": 50687,
+  "<|6.46|>": 50688,
+  "<|6.48|>": 50689,
+  "<|6.50|>": 50690,
+  "<|6.52|>": 50691,
+  "<|6.54|>": 50692,
+  "<|6.56|>": 50693,
+  "<|6.58|>": 50694,
+  "<|6.60|>": 50695,
+  "<|6.62|>": 50696,
+  "<|6.64|>": 50697,
+  "<|6.66|>": 50698,
+  "<|6.68|>": 50699,
+  "<|6.70|>": 50700,
+  "<|6.72|>": 50701,
+  "<|6.74|>": 50702,
+  "<|6.76|>": 50703,
+  "<|6.78|>": 50704,
+  "<|6.80|>": 50705,
+  "<|6.82|>": 50706,
+  "<|6.84|>": 50707,
+  "<|6.86|>": 50708,
+  "<|6.88|>": 50709,
+  "<|6.90|>": 50710,
+  "<|6.92|>": 50711,
+  "<|6.94|>": 50712,
+  "<|6.96|>": 50713,
+  "<|6.98|>": 50714,
+  "<|7.00|>": 50715,
+  "<|7.02|>": 50716,
+  "<|7.04|>": 50717,
+  "<|7.06|>": 50718,
+  "<|7.08|>": 50719,
+  "<|7.10|>": 50720,
+  "<|7.12|>": 50721,
+  "<|7.14|>": 50722,
+  "<|7.16|>": 50723,
+  "<|7.18|>": 50724,
+  "<|7.20|>": 50725,
+  "<|7.22|>": 50726,
+  "<|7.24|>": 50727,
+  "<|7.26|>": 50728,
+  "<|7.28|>": 50729,
+  "<|7.30|>": 50730,
+  "<|7.32|>": 50731,
+  "<|7.34|>": 50732,
+  "<|7.36|>": 50733,
+  "<|7.38|>": 50734,
+  "<|7.40|>": 50735,
+  "<|7.42|>": 50736,
+  "<|7.44|>": 50737,
+  "<|7.46|>": 50738,
+  "<|7.48|>": 50739,
+  "<|7.50|>": 50740,
+  "<|7.52|>": 50741,
+  "<|7.54|>": 50742,
+  "<|7.56|>": 50743,
+  "<|7.58|>": 50744,
+  "<|7.60|>": 50745,
+  "<|7.62|>": 50746,
+  "<|7.64|>": 50747,
+  "<|7.66|>": 50748,
+  "<|7.68|>": 50749,
+  "<|7.70|>": 50750,
+  "<|7.72|>": 50751,
+  "<|7.74|>": 50752,
+  "<|7.76|>": 50753,
+  "<|7.78|>": 50754,
+  "<|7.80|>": 50755,
+  "<|7.82|>": 50756,
+  "<|7.84|>": 50757,
+  "<|7.86|>": 50758,
+  "<|7.88|>": 50759,
+  "<|7.90|>": 50760,
+  "<|7.92|>": 50761,
+  "<|7.94|>": 50762,
+  "<|7.96|>": 50763,
+  "<|7.98|>": 50764,
+  "<|8.00|>": 50765,
+  "<|8.02|>": 50766,
+  "<|8.04|>": 50767,
+  "<|8.06|>": 50768,
+  "<|8.08|>": 50769,
+  "<|8.10|>": 50770,
+  "<|8.12|>": 50771,
+  "<|8.14|>": 50772,
+  "<|8.16|>": 50773,
+  "<|8.18|>": 50774,
+  "<|8.20|>": 50775,
+  "<|8.22|>": 50776,
+  "<|8.24|>": 50777,
+  "<|8.26|>": 50778,
+  "<|8.28|>": 50779,
+  "<|8.30|>": 50780,
+  "<|8.32|>": 50781,
+  "<|8.34|>": 50782,
+  "<|8.36|>": 50783,
+  "<|8.38|>": 50784,
+  "<|8.40|>": 50785,
+  "<|8.42|>": 50786,
+  "<|8.44|>": 50787,
+  "<|8.46|>": 50788,
+  "<|8.48|>": 50789,
+  "<|8.50|>": 50790,
+  "<|8.52|>": 50791,
+  "<|8.54|>": 50792,
+  "<|8.56|>": 50793,
+  "<|8.58|>": 50794,
+  "<|8.60|>": 50795,
+  "<|8.62|>": 50796,
+  "<|8.64|>": 50797,
+  "<|8.66|>": 50798,
+  "<|8.68|>": 50799,
+  "<|8.70|>": 50800,
+  "<|8.72|>": 50801,
+  "<|8.74|>": 50802,
+  "<|8.76|>": 50803,
+  "<|8.78|>": 50804,
+  "<|8.80|>": 50805,
+  "<|8.82|>": 50806,
+  "<|8.84|>": 50807,
+  "<|8.86|>": 50808,
+  "<|8.88|>": 50809,
+  "<|8.90|>": 50810,
+  "<|8.92|>": 50811,
+  "<|8.94|>": 50812,
+  "<|8.96|>": 50813,
+  "<|8.98|>": 50814,
+  "<|9.00|>": 50815,
+  "<|9.02|>": 50816,
+  "<|9.04|>": 50817,
+  "<|9.06|>": 50818,
+  "<|9.08|>": 50819,
+  "<|9.10|>": 50820,
+  "<|9.12|>": 50821,
+  "<|9.14|>": 50822,
+  "<|9.16|>": 50823,
+  "<|9.18|>": 50824,
+  "<|9.20|>": 50825,
+  "<|9.22|>": 50826,
+  "<|9.24|>": 50827,
+  "<|9.26|>": 50828,
+  "<|9.28|>": 50829,
+  "<|9.30|>": 50830,
+  "<|9.32|>": 50831,
+  "<|9.34|>": 50832,
+  "<|9.36|>": 50833,
+  "<|9.38|>": 50834,
+  "<|9.40|>": 50835,
+  "<|9.42|>": 50836,
+  "<|9.44|>": 50837,
+  "<|9.46|>": 50838,
+  "<|9.48|>": 50839,
+  "<|9.50|>": 50840,
+  "<|9.52|>": 50841,
+  "<|9.54|>": 50842,
+  "<|9.56|>": 50843,
+  "<|9.58|>": 50844,
+  "<|9.60|>": 50845,
+  "<|9.62|>": 50846,
+  "<|9.64|>": 50847,
+  "<|9.66|>": 50848,
+  "<|9.68|>": 50849,
+  "<|9.70|>": 50850,
+  "<|9.72|>": 50851,
+  "<|9.74|>": 50852,
+  "<|9.76|>": 50853,
+  "<|9.78|>": 50854,
+  "<|9.80|>": 50855,
+  "<|9.82|>": 50856,
+  "<|9.84|>": 50857,
+  "<|9.86|>": 50858,
+  "<|9.88|>": 50859,
+  "<|9.90|>": 50860,
+  "<|9.92|>": 50861,
+  "<|9.94|>": 50862,
+  "<|9.96|>": 50863,
+  "<|9.98|>": 50864,
+  "<|af|>": 50327,
+  "<|am|>": 50334,
+  "<|ar|>": 50272,
+  "<|as|>": 50350,
+  "<|az|>": 50304,
+  "<|ba|>": 50355,
+  "<|be|>": 50330,
+  "<|bg|>": 50292,
+  "<|bn|>": 50302,
+  "<|bo|>": 50347,
+  "<|br|>": 50309,
+  "<|bs|>": 50315,
+  "<|ca|>": 50270,
+  "<|cs|>": 50283,
+  "<|cy|>": 50297,
+  "<|da|>": 50285,
+  "<|de|>": 50261,
+  "<|el|>": 50281,
+  "<|endoftext|>": 50257,
+  "<|en|>": 50259,
+  "<|es|>": 50262,
+  "<|et|>": 50307,
+  "<|eu|>": 50310,
+  "<|fa|>": 50300,
+  "<|fi|>": 50277,
+  "<|fo|>": 50338,
+  "<|fr|>": 50265,
+  "<|gl|>": 50319,
+  "<|gu|>": 50333,
+  "<|haw|>": 50352,
+  "<|ha|>": 50354,
+  "<|he|>": 50279,
+  "<|hi|>": 50276,
+  "<|hr|>": 50291,
+  "<|ht|>": 50339,
+  "<|hu|>": 50286,
+  "<|hy|>": 50312,
+  "<|id|>": 50275,
+  "<|is|>": 50311,
+  "<|it|>": 50274,
+  "<|ja|>": 50266,
+  "<|jw|>": 50356,
+  "<|ka|>": 50329,
+  "<|kk|>": 50316,
+  "<|km|>": 50323,
+  "<|kn|>": 50306,
+  "<|ko|>": 50264,
+  "<|la|>": 50294,
+  "<|lb|>": 50345,
+  "<|ln|>": 50353,
+  "<|lo|>": 50336,
+  "<|lt|>": 50293,
+  "<|lv|>": 50301,
+  "<|mg|>": 50349,
+  "<|mi|>": 50295,
+  "<|mk|>": 50308,
+  "<|ml|>": 50296,
+  "<|mn|>": 50314,
+  "<|mr|>": 50320,
+  "<|ms|>": 50282,
+  "<|mt|>": 50343,
+  "<|my|>": 50346,
+  "<|ne|>": 50313,
+  "<|nl|>": 50271,
+  "<|nn|>": 50342,
+  "<|nospeech|>": 50363,
+  "<|notimestamps|>": 50364,
+  "<|no|>": 50288,
+  "<|oc|>": 50328,
+  "<|pa|>": 50321,
+  "<|pl|>": 50269,
+  "<|ps|>": 50340,
+  "<|pt|>": 50267,
+  "<|ro|>": 50284,
+  "<|ru|>": 50263,
+  "<|sa|>": 50344,
+  "<|sd|>": 50332,
+  "<|si|>": 50322,
+  "<|sk|>": 50298,
+  "<|sl|>": 50305,
+  "<|sn|>": 50324,
+  "<|so|>": 50326,
+  "<|sq|>": 50317,
+  "<|sr|>": 50303,
+  "<|startoflm|>": 50361,
+  "<|startofprev|>": 50362,
+  "<|startoftranscript|>": 50258,
+  "<|su|>": 50357,
+  "<|sv|>": 50273,
+  "<|sw|>": 50318,
+  "<|ta|>": 50287,
+  "<|te|>": 50299,
+  "<|tg|>": 50331,
+  "<|th|>": 50289,
+  "<|tk|>": 50341,
+  "<|tl|>": 50348,
+  "<|transcribe|>": 50360,
+  "<|translate|>": 50359,
+  "<|tr|>": 50268,
+  "<|tt|>": 50351,
+  "<|uk|>": 50280,
+  "<|ur|>": 50290,
+  "<|uz|>": 50337,
+  "<|vi|>": 50278,
+  "<|yi|>": 50335,
+  "<|yo|>": 50325,
+  "<|yue|>": 50358,
+  "<|zh|>": 50260
+}

nb-distil-large-init/config.json ADDED Viewed

	@@ -0,0 +1,288 @@

+{
+  "_name_or_path": "./",
+  "activation_dropout": 0.1,
+  "activation_function": "gelu",
+  "alignment_heads": [
+    [
+      7,
+      0
+    ],
+    [
+      10,
+      17
+    ],
+    [
+      12,
+      18
+    ],
+    [
+      13,
+      12
+    ],
+    [
+      16,
+      1
+    ],
+    [
+      17,
+      14
+    ],
+    [
+      19,
+      11
+    ],
+    [
+      21,
+      4
+    ],
+    [
+      24,
+      1
+    ],
+    [
+      25,
+      6
+    ]
+  ],
+  "apply_spec_augment": false,
+  "architectures": [
+    "WhisperForConditionalGeneration"
+  ],
+  "attention_dropout": 0,
+  "begin_suppress_tokens": [
+    220,
+    50257
+  ],
+  "bos_token_id": 50257,
+  "classifier_proj_size": 256,
+  "d_model": 1280,
+  "decoder_attention_heads": 20,
+  "decoder_ffn_dim": 5120,
+  "decoder_layerdrop": 0,
+  "decoder_layers": 2,
+  "decoder_start_token_id": 50258,
+  "dropout": 0,
+  "encoder_attention_heads": 20,
+  "encoder_ffn_dim": 5120,
+  "encoder_layerdrop": 0,
+  "encoder_layers": 32,
+  "eos_token_id": 50257,
+  "init_std": 0.02,
+  "is_encoder_decoder": true,
+  "lang_ids": [
+    50259,
+    50260,
+    50261,
+    50262,
+    50263,
+    50264,
+    50265,
+    50266,
+    50267,
+    50268,
+    50269,
+    50270,
+    50271,
+    50272,
+    50273,
+    50274,
+    50275,
+    50276,
+    50277,
+    50278,
+    50279,
+    50280,
+    50281,
+    50282,
+    50283,
+    50284,
+    50285,
+    50286,
+    50287,
+    50288,
+    50289,
+    50290,
+    50291,
+    50292,
+    50293,
+    50294,
+    50295,
+    50296,
+    50297,
+    50298,
+    50299,
+    50300,
+    50301,
+    50302,
+    50303,
+    50304,
+    50305,
+    50306,
+    50307,
+    50308,
+    50309,
+    50310,
+    50311,
+    50312,
+    50313,
+    50314,
+    50315,
+    50316,
+    50317,
+    50318,
+    50319,
+    50320,
+    50321,
+    50322,
+    50323,
+    50324,
+    50325,
+    50326,
+    50327,
+    50328,
+    50329,
+    50330,
+    50331,
+    50332,
+    50333,
+    50334,
+    50335,
+    50336,
+    50337,
+    50338,
+    50339,
+    50340,
+    50341,
+    50342,
+    50343,
+    50344,
+    50345,
+    50346,
+    50347,
+    50348,
+    50349,
+    50350,
+    50351,
+    50352,
+    50353,
+    50354,
+    50355,
+    50356,
+    50357,
+    50358
+  ],
+  "mask_feature_length": 10,
+  "mask_feature_min_masks": 0,
+  "mask_feature_prob": 0,
+  "mask_time_length": 10,
+  "mask_time_min_masks": 2,
+  "mask_time_prob": 0.05,
+  "max_length": 448,
+  "max_source_positions": 1500,
+  "max_target_positions": 448,
+  "median_filter_width": 7,
+  "model_type": "whisper",
+  "num_hidden_layers": 32,
+  "num_mel_bins": 128,
+  "pad_token_id": 50256,
+  "scale_embedding": false,
+  "suppress_ids": [
+    1,
+    2,
+    7,
+    8,
+    9,
+    10,
+    14,
+    25,
+    26,
+    27,
+    28,
+    29,
+    31,
+    58,
+    59,
+    60,
+    61,
+    62,
+    63,
+    90,
+    91,
+    92,
+    93,
+    359,
+    503,
+    522,
+    542,
+    873,
+    893,
+    902,
+    918,
+    922,
+    931,
+    1350,
+    1853,
+    1982,
+    2460,
+    2627,
+    3246,
+    3253,
+    3268,
+    3536,
+    3846,
+    3961,
+    4183,
+    4667,
+    6585,
+    6647,
+    7273,
+    9061,
+    9383,
+    10428,
+    10929,
+    11938,
+    12033,
+    12331,
+    12562,
+    13793,
+    14157,
+    14635,
+    15265,
+    15618,
+    16553,
+    16604,
+    18362,
+    18956,
+    20075,
+    21675,
+    22520,
+    26130,
+    26161,
+    26435,
+    28279,
+    29464,
+    31650,
+    32302,
+    32470,
+    36865,
+    42863,
+    47425,
+    49870,
+    50254,
+    50258,
+    50359,
+    50360,
+    50361,
+    50362,
+    50363
+  ],
+  "suppress_ids_begin": [
+    220,
+    50257
+  ],
+  "torch_dtype": "float32",
+  "transformers_version": "4.46.2",
+  "use_cache": true,
+  "use_weighted_layer_sum": false,
+  "vocab_size": 51866
+}

nb-distil-large-init/flax_model.msgpack ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:60f608eb7887b643bfb0d6b11d3ad8564c648c296a90c1e558aa61075b1f2839
+size 1512831199

nb-distil-large-init/generation_config.json ADDED Viewed

	@@ -0,0 +1,270 @@

+{
+  "alignment_heads": [
+    [
+      7,
+      0
+    ],
+    [
+      10,
+      17
+    ],
+    [
+      12,
+      18
+    ],
+    [
+      13,
+      12
+    ],
+    [
+      16,
+      1
+    ],
+    [
+      17,
+      14
+    ],
+    [
+      19,
+      11
+    ],
+    [
+      21,
+      4
+    ],
+    [
+      24,
+      1
+    ],
+    [
+      25,
+      6
+    ]
+  ],
+  "begin_suppress_tokens": [
+    220,
+    50257
+  ],
+  "bos_token_id": 50257,
+  "decoder_start_token_id": 50258,
+  "eos_token_id": 50257,
+  "forced_decoder_ids": [
+    [
+      1,
+      50288
+    ],
+    [
+      2,
+      50360
+    ],
+    [
+      3,
+      50364
+    ]
+  ],
+  "is_multilingual": true,
+  "lang_to_id": {
+    "<|af|>": 50327,
+    "<|am|>": 50334,
+    "<|ar|>": 50272,
+    "<|as|>": 50350,
+    "<|az|>": 50304,
+    "<|ba|>": 50355,
+    "<|be|>": 50330,
+    "<|bg|>": 50292,
+    "<|bn|>": 50302,
+    "<|bo|>": 50347,
+    "<|br|>": 50309,
+    "<|bs|>": 50315,
+    "<|ca|>": 50270,
+    "<|cs|>": 50283,
+    "<|cy|>": 50297,
+    "<|da|>": 50285,
+    "<|de|>": 50261,
+    "<|el|>": 50281,
+    "<|en|>": 50259,
+    "<|es|>": 50262,
+    "<|et|>": 50307,
+    "<|eu|>": 50310,
+    "<|fa|>": 50300,
+    "<|fi|>": 50277,
+    "<|fo|>": 50338,
+    "<|fr|>": 50265,
+    "<|gl|>": 50319,
+    "<|gu|>": 50333,
+    "<|haw|>": 50352,
+    "<|ha|>": 50354,
+    "<|he|>": 50279,
+    "<|hi|>": 50276,
+    "<|hr|>": 50291,
+    "<|ht|>": 50339,
+    "<|hu|>": 50286,
+    "<|hy|>": 50312,
+    "<|id|>": 50275,
+    "<|is|>": 50311,
+    "<|it|>": 50274,
+    "<|ja|>": 50266,
+    "<|jw|>": 50356,
+    "<|ka|>": 50329,
+    "<|kk|>": 50316,
+    "<|km|>": 50323,
+    "<|kn|>": 50306,
+    "<|ko|>": 50264,
+    "<|la|>": 50294,
+    "<|lb|>": 50345,
+    "<|ln|>": 50353,
+    "<|lo|>": 50336,
+    "<|lt|>": 50293,
+    "<|lv|>": 50301,
+    "<|mg|>": 50349,
+    "<|mi|>": 50295,
+    "<|mk|>": 50308,
+    "<|ml|>": 50296,
+    "<|mn|>": 50314,
+    "<|mr|>": 50320,
+    "<|ms|>": 50282,
+    "<|mt|>": 50343,
+    "<|my|>": 50346,
+    "<|ne|>": 50313,
+    "<|nl|>": 50271,
+    "<|nn|>": 50342,
+    "<|no|>": 50288,
+    "<|oc|>": 50328,
+    "<|pa|>": 50321,
+    "<|pl|>": 50269,
+    "<|ps|>": 50340,
+    "<|pt|>": 50267,
+    "<|ro|>": 50284,
+    "<|ru|>": 50263,
+    "<|sa|>": 50344,
+    "<|sd|>": 50332,
+    "<|si|>": 50322,
+    "<|sk|>": 50298,
+    "<|sl|>": 50305,
+    "<|sn|>": 50324,
+    "<|so|>": 50326,
+    "<|sq|>": 50317,
+    "<|sr|>": 50303,
+    "<|su|>": 50357,
+    "<|sv|>": 50273,
+    "<|sw|>": 50318,
+    "<|ta|>": 50287,
+    "<|te|>": 50299,
+    "<|tg|>": 50331,
+    "<|th|>": 50289,
+    "<|tk|>": 50341,
+    "<|tl|>": 50348,
+    "<|tr|>": 50268,
+    "<|tt|>": 50351,
+    "<|uk|>": 50280,
+    "<|ur|>": 50290,
+    "<|uz|>": 50337,
+    "<|vi|>": 50278,
+    "<|yi|>": 50335,
+    "<|yo|>": 50325,
+    "<|yue|>": 50358,
+    "<|zh|>": 50260
+  },
+  "language": "<|no|>",
+  "max_initial_timestamp_index": 1,
+  "max_length": 448,
+  "no_timestamps_token_id": 50364,
+  "pad_token_id": 50257,
+  "return_timestamps": false,
+  "suppress_tokens": [
+    1,
+    2,
+    7,
+    8,
+    9,
+    10,
+    14,
+    25,
+    26,
+    27,
+    28,
+    29,
+    31,
+    58,
+    59,
+    60,
+    61,
+    62,
+    63,
+    90,
+    91,
+    92,
+    93,
+    359,
+    503,
+    522,
+    542,
+    873,
+    893,
+    902,
+    918,
+    922,
+    931,
+    1350,
+    1853,
+    1982,
+    2460,
+    2627,
+    3246,
+    3253,
+    3268,
+    3536,
+    3846,
+    3961,
+    4183,
+    4667,
+    6585,
+    6647,
+    7273,
+    9061,
+    9383,
+    10428,
+    10929,
+    11938,
+    12033,
+    12331,
+    12562,
+    13793,
+    14157,
+    14635,
+    15265,
+    15618,
+    16553,
+    16604,
+    18362,
+    18956,
+    20075,
+    21675,
+    22520,
+    26130,
+    26161,
+    26435,
+    28279,
+    29464,
+    31650,
+    32302,
+    32470,
+    36865,
+    42863,
+    47425,
+    49870,
+    50254,
+    50258,
+    50359,
+    50360,
+    50361,
+    50362,
+    50363
+  ],
+  "task": "transcribe",
+  "task_to_id": {
+    "transcribe": 50360,
+    "translate": 50359
+  },
+  "transformers_version": "4.46.2"
+}

nb-distil-large-init/merges.txt ADDED Viewed

The diff for this file is too large to render. See raw diff

nb-distil-large-init/preprocessor_config.json ADDED Viewed

	@@ -0,0 +1,14 @@

+{
+  "chunk_length": 30,
+  "feature_extractor_type": "WhisperFeatureExtractor",
+  "feature_size": 128,
+  "hop_length": 160,
+  "n_fft": 400,
+  "n_samples": 480000,
+  "nb_max_frames": 3000,
+  "padding_side": "right",
+  "padding_value": 0.0,
+  "processor_class": "WhisperProcessor",
+  "return_attention_mask": false,
+  "sampling_rate": 16000
+}

nb-distil-large-init/special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,139 @@

+{
+  "additional_special_tokens": [
+    "<|startoftranscript|>",
+    "<|en|>",
+    "<|zh|>",
+    "<|de|>",
+    "<|es|>",
+    "<|ru|>",
+    "<|ko|>",
+    "<|fr|>",
+    "<|ja|>",
+    "<|pt|>",
+    "<|tr|>",
+    "<|pl|>",
+    "<|ca|>",
+    "<|nl|>",
+    "<|ar|>",
+    "<|sv|>",
+    "<|it|>",
+    "<|id|>",
+    "<|hi|>",
+    "<|fi|>",
+    "<|vi|>",
+    "<|he|>",
+    "<|uk|>",
+    "<|el|>",
+    "<|ms|>",
+    "<|cs|>",
+    "<|ro|>",
+    "<|da|>",
+    "<|hu|>",
+    "<|ta|>",
+    "<|no|>",
+    "<|th|>",
+    "<|ur|>",
+    "<|hr|>",
+    "<|bg|>",
+    "<|lt|>",
+    "<|la|>",
+    "<|mi|>",
+    "<|ml|>",
+    "<|cy|>",
+    "<|sk|>",
+    "<|te|>",
+    "<|fa|>",
+    "<|lv|>",
+    "<|bn|>",
+    "<|sr|>",
+    "<|az|>",
+    "<|sl|>",
+    "<|kn|>",
+    "<|et|>",
+    "<|mk|>",
+    "<|br|>",
+    "<|eu|>",
+    "<|is|>",
+    "<|hy|>",
+    "<|ne|>",
+    "<|mn|>",
+    "<|bs|>",
+    "<|kk|>",
+    "<|sq|>",
+    "<|sw|>",
+    "<|gl|>",
+    "<|mr|>",
+    "<|pa|>",
+    "<|si|>",
+    "<|km|>",
+    "<|sn|>",
+    "<|yo|>",
+    "<|so|>",
+    "<|af|>",
+    "<|oc|>",
+    "<|ka|>",
+    "<|be|>",
+    "<|tg|>",
+    "<|sd|>",
+    "<|gu|>",
+    "<|am|>",
+    "<|yi|>",
+    "<|lo|>",
+    "<|uz|>",
+    "<|fo|>",
+    "<|ht|>",
+    "<|ps|>",
+    "<|tk|>",
+    "<|nn|>",
+    "<|mt|>",
+    "<|sa|>",
+    "<|lb|>",
+    "<|my|>",
+    "<|bo|>",
+    "<|tl|>",
+    "<|mg|>",
+    "<|as|>",
+    "<|tt|>",
+    "<|haw|>",
+    "<|ln|>",
+    "<|ha|>",
+    "<|ba|>",
+    "<|jw|>",
+    "<|su|>",
+    "<|yue|>",
+    "<|translate|>",
+    "<|transcribe|>",
+    "<|startoflm|>",
+    "<|startofprev|>",
+    "<|nospeech|>",
+    "<|notimestamps|>"
+  ],
+  "bos_token": {
+    "content": "<|endoftext|>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "eos_token": {
+    "content": "<|endoftext|>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "pad_token": {
+    "content": "<|endoftext|>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "unk_token": {
+    "content": "<|endoftext|>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  }
+}

nb-distil-large-init/tokenizer_config.json ADDED Viewed

The diff for this file is too large to render. See raw diff

nb-distil-large-init/vocab.json ADDED Viewed

The diff for this file is too large to render. See raw diff

run_distillation.py ADDED Viewed

	@@ -0,0 +1,2172 @@

+#!/usr/bin/env python
+# coding=utf-8
+# Copyright 2023 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+"""
+ Training the Whisper model for sequence to sequence speech recognition via teacher-student distillation.
+"""
+# You can also adapt this script for your own distillation tasks. Pointers for this are left as comments.
+import logging
+import os
+import re
+import shutil
+import string
+import sys
+import time
+from dataclasses import dataclass, field
+from functools import partial
+from pathlib import Path
+from typing import Any, Callable, Dict, List, Optional, Union
+import datasets
+import evaluate
+import flax
+import jax
+import jax.numpy as jnp
+import numpy as np
+import optax
+import torch
+import transformers
+from datasets import (
+    DatasetDict,
+    IterableDataset,
+    IterableDatasetDict,
+    concatenate_datasets,
+    interleave_datasets,
+    load_dataset,
+)
+from datasets.distributed import split_dataset_by_node
+from flax import jax_utils, traverse_util
+from flax.jax_utils import pad_shard_unpad, unreplicate
+from flax.serialization import from_bytes, to_bytes
+from flax.training import train_state
+from flax.training.common_utils import get_metrics, onehot, shard, shard_prng_key
+from huggingface_hub import Repository, create_repo
+from jax.experimental.compilation_cache import compilation_cache as cc
+from optax._src import linear_algebra
+from torch.utils.data import DataLoader
+from torchdata.datapipes.iter import IterableWrapper
+from tqdm import tqdm
+from transformers import (
+    AddedToken,
+    HfArgumentParser,
+    Seq2SeqTrainingArguments,
+    WhisperConfig,
+    WhisperFeatureExtractor,
+    WhisperProcessor,
+    WhisperTokenizerFast,
+    is_tensorboard_available,
+    is_wandb_available,
+    set_seed,
+)
+from transformers.file_utils import get_full_repo_name
+from transformers.modeling_flax_outputs import FlaxBaseModelOutput
+from transformers.models.whisper.english_normalizer import BasicTextNormalizer,EnglishTextNormalizer
+from transformers.utils import check_min_version, send_example_telemetry
+from transformers.utils.versions import require_version
+from distil_whisper import FlaxWhisperForConditionalGeneration
+# Will error if the minimal version of Transformers is not installed. Remove at your own risks.
+check_min_version("4.27.0.dev0")
+require_version(
+    "datasets>=1.18.0",
+    "To fix: pip install -r examples/flax/speech-recogintion/requirements.txt",
+)
+logger = logging.getLogger(__name__)
+@flax.struct.dataclass
+class ModelArguments:
+    """
+    Arguments pertaining to which model/config/tokenizer we are going to fine-tune from.
+    """
+    model_name_or_path: str = field(
+        metadata={"help": ("Path to pretrained student model or model identifier from huggingface.co/models")}
+    )
+    teacher_model_name_or_path: str = field(
+        metadata={"help": ("Path to pretrained teacher model or model identifier from huggingface.co/models")}
+    )
+    config_name: Optional[str] = field(
+        default=None,
+        metadata={"help": "Pretrained config name or path if not the same as model_name"},
+    )
+    tokenizer_name: Optional[str] = field(
+        default=None,
+        metadata={"help": "Pretrained tokenizer name or path if not the same as model_name"},
+    )
+    feature_extractor_name: Optional[str] = field(
+        default=None,
+        metadata={"help": "feature extractor name or path if not the same as model_name"},
+    )
+    cache_dir: Optional[str] = field(
+        default=None,
+        metadata={"help": ("Where to store the pretrained models downloaded from huggingface.co")},
+    )
+    use_fast_tokenizer: bool = field(
+        default=True,
+        metadata={"help": ("Whether to use one of the fast tokenizer (backed by the tokenizers library) or not.")},
+    )
+    model_revision: str = field(
+        default="main",
+        metadata={"help": ("The specific model version to use (can be a branch name, tag name or commit id).")},
+    )
+    subfolder: str = field(
+        default="",
+        metadata={
+            "help": "In case the relevant files are located inside a subfolder of the model repo on huggingface.co, you can"
+            "specify the folder name here."
+        },
+    )
+    use_auth_token: bool = field(
+        default=False,
+        metadata={
+            "help": (
+                "Will use the token generated when running `transformers-cli login`"
+                " (necessary to use this script with private models)."
+            )
+        },
+    )
+    dtype: Optional[str] = field(
+        default="float32",
+        metadata={
+            "help": (
+                "Floating-point format in which the model weights should be initialized"
+                " and trained. Choose one of `[float32, float16, bfloat16]`."
+            )
+        },
+    )
+    load_with_scan_weights: bool = field(
+        default=False,
+        metadata={
+            "help": "Whether the pre-trained checkpoint has its weights stored in scan format. Set to True for scanned "
+            "weights, defaults to False for non-scan (unrolled) weights."
+        },
+    )
+    activation_dropout: float = field(
+        default=0.0,
+        metadata={"help": "The dropout ratio for activations inside the fully connected layer."},
+    )
+    attention_dropout: float = field(
+        default=0.0,
+        metadata={"help": "The dropout ratio for the attention probabilities."},
+    )
+    dropout: float = field(
+        default=0.0,
+        metadata={
+            "help": "The dropout probability for all fully connected layers in the embeddings, encoder, and pooler."
+        },
+    )
+@flax.struct.dataclass
+class DataTrainingArguments:
+    """
+    Arguments pertaining to what data we are going to input our model for training and eval.
+    """
+    train_dataset_name: str = field(
+        default=None,
+        metadata={
+            "help": "The name of the training dataset to use (via the datasets library). Load and combine "
+            "multiple datasets by separating dataset ids by a '+' symbol. For example, to load and combine "
+            " librispeech and common voice, set `train_dataset_name='librispeech_asr+common_voice'`."
+        },
+    )
+    train_dataset_config_name: Optional[str] = field(
+        default=None,
+        metadata={
+            "help": "The configuration name of the training dataset to use (via the datasets library). Load and combine "
+            "multiple datasets by separating dataset configs by a '+' symbol."
+        },
+    )
+    train_dataset_samples: str = field(
+        default=None,
+        metadata={
+            "help": "Number of samples in the training data. Load and combine "
+            "multiple datasets by separating dataset samples by a '+' symbol."
+        },
+    )
+    eval_dataset_name: str = field(
+        default=None,
+        metadata={
+            "help": "The name of the evaluation dataset to use (via the datasets library). Defaults to the training dataset name if unspecified."
+        },
+    )
+    eval_dataset_config_name: Optional[str] = field(
+        default=None,
+        metadata={
+            "help": "The configuration name of the evaluation dataset to use (via the datasets library). Defaults to the training dataset config name if unspecified"
+        },
+    )
+    dataset_cache_dir: Optional[str] = field(
+        default=None,
+        metadata={"help": "Path to cache directory for saving and loading datasets"},
+    )
+    overwrite_cache: bool = field(
+        default=False,
+        metadata={"help": "Overwrite the cached training and evaluation sets"},
+    )
+    preprocessing_num_workers: Optional[int] = field(
+        default=None,
+        metadata={"help": "The number of processes to use for the preprocessing."},
+    )
+    max_train_samples: Optional[int] = field(
+        default=None,
+        metadata={
+            "help": (
+                "For debugging purposes or quicker training, truncate the number of"
+                " training examples to this value if set."
+            )
+        },
+    )
+    max_eval_samples: Optional[int] = field(
+        default=None,
+        metadata={
+            "help": (
+                "For debugging purposes or quicker training, truncate the number of"
+                " evaluation examples to this value if set."
+            )
+        },
+    )
+    audio_column_name: str = field(
+        default="audio",
+        metadata={"help": ("The name of the dataset column containing the audio data. Defaults to 'audio'")},
+    )
+    train_text_column_name: str = field(
+        default="whisper_transcript",
+        metadata={
+            "help": (
+                "The name of the dataset column containing the text data. Defaults to"
+                " 'whisper_transcript'which is the pseudo-labelled Whisper"
+                " transcription data."
+            )
+        },
+    )
+    eval_text_column_name: str = field(
+        default="text",
+        metadata={
+            "help": (
+                "The name of the dataset column containing the text data. Defaults to"
+                " 'text', which is the original text data"
+            )
+        },
+    )
+    max_duration_in_seconds: float = field(
+        default=30.0,
+        metadata={"help": ("Filter audio files that are longer than `max_duration_in_seconds` seconds")},
+    )
+    min_duration_in_seconds: float = field(
+        default=0.0,
+        metadata={"help": ("Filter audio files that are shorter than `min_duration_in_seconds` seconds")},
+    )
+    max_label_length: int = field(
+        default=128,
+        metadata={"help": "Truncate transcriptions that are longer `max_label_length` tokens."},
+    )
+    pad_target_to_multiple_of: Optional[int] = field(
+        default=None,
+        metadata={
+            "help": (
+                "If set will pad the target sequence to a multiple of the provided"
+                " value. This is important to avoid triggering recompilations on TPU."
+                " If unspecified, will default to padding the targets to max length."
+            )
+        },
+    )
+    preprocessing_only: bool = field(
+        default=False,
+        metadata={
+            "help": (
+                "Whether to only do data preprocessing and skip training. This is"
+                " especially useful when data preprocessing errors out in distributed"
+                " training due to timeout. In this case, one should run the"
+                " preprocessing in a non-distributed setup with"
+                " `preprocessing_only=True` so that the cached datasets can"
+                " consequently be loaded in distributed training"
+            )
+        },
+    )
+    train_split_name: str = field(
+        default="train",
+        metadata={
+            "help": ("The name of the training data set split to use (via the datasets library). Defaults to 'train'")
+        },
+    )
+    eval_split_name: str = field(
+        default="validation",
+        metadata={
+            "help": (
+                "The name of the evaluation data set split to use (via the datasets"
+                " library). Defaults to 'validation'"
+            )
+        },
+    )
+    wandb_project: str = field(
+        default="distil-whisper",
+        metadata={"help": "The name of the wandb project."},
+    )
+    wandb_name: str = field(
+        default=None,
+        metadata={"help": "The name of the wandb run."},
+    )
+    wandb_job_type: str = field(
+        default="distil-whisper",
+        metadata={"help": "The name of the wandb job type."},
+    )
+    wandb_dir: str = field(
+        default=None,
+        metadata={"help": "The absolute path to save the wandb logs."},
+    )
+    save_code_to_wandb: bool = field(
+        default=False,
+        metadata={
+            "help": (
+                "Whether to save main script to wandb. This is valuable for improving"
+                " experiment reproducibility and to diff code across experiments in"
+                " the UI."
+            )
+        },
+    )
+    streaming: bool = field(
+        default=True,
+        metadata={"help": "Whether to use Datasets' streaming mode to load and the data."},
+    )
+    wer_threshold: float = field(
+        default=None,
+        metadata={
+            "help": "Filter training data with Whisper transcriptions that have greater than `wer_threshold` "
+            "WER with the normalised transcriptions."
+        },
+    )
+    prefetch_size: int = field(
+        default=0,
+        metadata={"help": "Number of samples to pre-fetch if using an iterable dataset."},
+    )
+    timestamp_probability: float = field(
+        default=0.5, metadata={"help": "Probability for training on timestamped tokens if the data contains it."}
+    )
+    return_timestamps: bool = field(
+        default=False, metadata={"help": "Whether or not to predict timestamps in the generation step."}
+    )
+    round_timestamps: bool = field(
+        default=False,
+        metadata={
+            "help": "Whether or not to round the timestamp tokens to the nearest tenth of a second."
+            "By default, Whisper predicts timestamps to the nearest hundredth of a second."
+            "Reducing the timestamp precision to one tenth of a second simplifies the timestamp"
+            "prediction task, at the expense of timestamp granularity."
+        },
+    )
+@dataclass
+class FlaxSeq2SeqTrainingArguments(Seq2SeqTrainingArguments):
+    use_scan: Optional[bool] = field(
+        default=True,
+        metadata={
+            "help": (
+                "Whether or not to use `scan_with_axes` over the encoder and decoder blocks. Using scan results "
+                "in faster compile times and more efficient memory use during training, since all of the layers "
+                "in the encoder/decoder are stacked, and we perform a lax.scan over the stacked block to index "
+                "each layer. However, it results in slower inference time due to the overhead of stacking the "
+                "layers this way. Thus, we **always** default to disabling scan for the inference step."
+            )
+        },
+    )
+    freeze_encoder: Optional[bool] = field(
+        default=False,
+        metadata={
+            "help": (
+                "Whether to freeze the entire encoder model. Only recommended when the entire encoder has been "
+                "copied from the teacher model."
+            )
+        },
+    )
+    temperature: Optional[float] = field(
+        default=2.0, metadata={"help": "Temperature to anneal the logits when computing the softmax."}
+    )
+    kl_weight: Optional[float] = field(
+        default=1.0,
+        metadata={
+            "help": (
+                "Weighting assigned to the MSE loss in the KD formulation. MSE loss is "
+                "computed between the teacher-student hidden states and attentions."
+            )
+        },
+    )
+    mse_weight: Optional[float] = field(
+        default=0.0,
+        metadata={
+            "help": (
+                "Weighting assigned to the MSE loss in the KD formulation. MSE loss is "
+                "computed between the teacher-student hidden states and attentions."
+            )
+        },
+    )
+    precision: Optional[str] = field(
+        default="half_mixed",
+        metadata={
+            "help": (
+                "Precision with which run training, Can be one of `full`, `half_mixed` or `full_mixed`, the latter two"
+                "of which enable *mixed-precision* training. **Note that this only specifies the dtype of the computation "
+                "and optimizer state. It does not influence the dtype of model parameters.** An explanation of the three "
+                "settings is provided below:"
+                "   1. Full precision: forward pass, backward pass and optimiser states all in float32."
+                "   2. Half mixed precision: forward pass in bfloat16, backward pass and optimiser states in float32. This "
+                "   corresponds to setting the dtype argument to bfloat16 when instantiating the model."
+                "   3. Full mixed precision: forward pass, backward pass and optimiser states all in bfloat16. The dtype "
+                "   argument is set to bfloat16 for the forward pass, and the gradients computed with respect to the bfloat16 "
+                "   parameters in the backward pass (giving bfloat16 gradients). The new optimiser states and parameter "
+                "   updates are computed in float32 by upcasting the bfloat16 gradients and optimiser states to float32 "
+                "   prior to the optimiser update step. The optimiser states are returned in float32 (but not saved to "
+                "   memory) and then downcasted to bfloat16 (saved to memory) for the subsequent train step."
+                "For further details, refer to https://github.com/deepmind/optax/discussions/336"
+            )
+        },
+    )
+    compilation_cache: Optional[bool] = field(
+        default=False,
+        metadata={
+            "help": (
+                "Whether to enable the JAX (experimental) compilation cache. The compilation step is *cached* the "
+                "first time it is run. Successive compilation steps for the same function utilise the cache to reduce"
+                "the compilation time."
+            )
+        },
+    )
+    save_train_state: Optional[bool] = field(
+        default=False,
+        metadata={
+            "help": "Whether or not to save the Flax Train State on each `save_steps` steps. Required if you intend"
+            "to resume training from partial training runs. If False, only the model weights will be saved."
+            "If True, both the model weights and Flax Train state will be saved."
+        },
+    )
+def shift_tokens_right(label_ids: np.array, decoder_start_token_id: int) -> np.ndarray:
+    """
+    Shift label ids one token to the right.
+    """
+    shifted_label_ids = np.zeros_like(label_ids)
+    shifted_label_ids[:, 1:] = label_ids[:, :-1]
+    shifted_label_ids[:, 0] = decoder_start_token_id
+    return shifted_label_ids
+@flax.struct.dataclass
+class FlaxDataCollatorSpeechSeq2SeqWithPadding:
+    """
+    Data collator that will dynamically pad the inputs received.
+    Args:
+        processor ([`Wav2Vec2Processor`])
+            The processor used for proccessing the data.
+        decoder_start_token_id (:obj: `int`)
+            The start-of-sequence token id of the decoder.
+        decoder_prev_token_id (:obj: `int`)
+            The start-of-prompt token id of the decoder
+        input_padding (:obj:`bool`, :obj:`str` or :class:`~transformers.tokenization_utils_base.PaddingStrategy`, `optional`, defaults to :obj:`True`):
+            Select a strategy to pad the returned input sequences (according to the model's padding side and padding index)
+            among:
+            * :obj:`True` or :obj:`'longest'`: Pad to the longest sequence in the batch (or no padding if only a single
+              sequence if provided).
+            * :obj:`'max_length'`: Pad to a maximum length specified with the argument :obj:`max_length` or to the
+              maximum acceptable input length for the model if that argument is not provided.
+            * :obj:`False` or :obj:`'do_not_pad'` (default): No padding (i.e., can output a batch with sequences of
+              different lengths).
+        target_padding (:obj:`bool`, :obj:`str` or :class:`~transformers.tokenization_utils_base.PaddingStrategy`, `optional`, defaults to :obj:`True`):
+            Select a strategy to pad the returned target sequences (according to the model's padding side and padding index).
+            See above for details.
+        max_target_length (:obj:`int`, `optional`):
+            Maximum length of the ``labels`` of the returned list and optionally padding length (see above).
+    """
+    processor: Any
+    decoder_start_token_id: int
+    decoder_prev_token_id: int
+    input_padding: Union[bool, str] = "max_length"
+    target_padding: Union[bool, str] = "max_length"
+    max_target_length: Optional[int] = None
+    def __call__(self, features: List[Dict[str, Union[List[int], np.ndarray]]]) -> Dict[str, np.ndarray]:
+        # split inputs and labels since they have to be of different lengths and need
+        # different padding methods
+        model_input_name = self.processor.model_input_names[0]
+        # dataloader returns a list of features which we convert to a dict
+        input_features = {model_input_name: [feature[model_input_name] for feature in features]}
+        label_features = {"input_ids": [feature["labels"] for feature in features]}
+        # reformat list to dict and set to pytorch format
+        batch = self.processor.feature_extractor.pad(
+            input_features,
+            padding=self.input_padding,
+            return_tensors="np",
+        )
+        labels_batch = self.processor.tokenizer.pad(
+            label_features,
+            max_length=self.max_target_length,
+            padding=self.target_padding,
+            return_tensors="np",
+        )
+        # if bos token is appended in previous tokenization step,
+        # cut bos token here as it's append later anyways
+        labels = labels_batch["input_ids"]
+        if set(np.unique(labels[:, 0])).issubset({self.decoder_start_token_id, self.decoder_prev_token_id}):
+            decoder_input_ids = labels[:, :-1]
+            labels = labels[:, 1:]
+            labels_batch.attention_mask = labels_batch.attention_mask[:, 1:]
+        else:
+            decoder_input_ids = shift_tokens_right(labels, self.decoder_start_token_id)
+        # replace padding with -100 to ignore correctly when computing the loss
+        labels = np.ma.array(labels, mask=np.not_equal(labels_batch.attention_mask, 1))
+        labels = labels.filled(fill_value=-100)
+        # replace initial prompt tokens with -100 to ignore correctly when computing the loss
+        bos_index = np.argmax(labels == self.decoder_start_token_id, axis=1)
+        prompt_mask = np.arange(labels.shape[1]) < bos_index[:, None]
+        labels = np.where(prompt_mask, -100, labels)
+        batch["labels"] = labels
+        batch["decoder_input_ids"] = decoder_input_ids
+        return batch
+def get_data_loader(
+    seed: int,
+    dataset: IterableDataset,
+    batch_size: int,
+    data_collator: FlaxDataCollatorSpeechSeq2SeqWithPadding,
+    shuffle: bool = False,
+    drop_last: bool = True,
+    dataloader_num_workers: int = 0,
+    skip_batches: int = 0,
+    pin_memory: bool = True,
+    prefetch_size: int = 0,
+) -> DataLoader:
+    """
+    Returns batches of size `batch_size` from `dataset`. If `drop_last` is set to `False`, the final batch may be incomplete,
+    and range in size from 1 to `batch_size`. Shuffle batches if `shuffle` is `True`.
+    Args:
+        seed (int): Numpy seed for generating pseudo random numbers. Used if shuffling the dataset.
+        dataset (IterableDataset): streaming dataset from which to load the data.
+        batch_size (int): how many samples per batch to load.
+        data_collator (FlaxDataCollatorSpeechSeq2SeqWithPadding, optional): merges a list of samples to form a
+            mini-batch of Tensor(s).  Used when using batched loading from a map-style dataset.
+        shuffle (bool, optional): set to `True` to have the batches reshuffled.
+        drop_last (bool, optional): set to ``True`` to drop the last incomplete batch,
+            if the dataset size is not divisible by the batch size. If ``False`` and
+            the size of dataset is not divisible by the batch size, then the last batch
+            will be smaller. (default: ``False``)
+        dataloader_num_workers (int, optional): how many subprocesses to use for data
+            loading. ``0`` means that the data will be loaded in the main process.
+            (default: ``0``)
+        skip_batches (int, optional): Efficiently skip the first `skip_batches`.
+        pin_memory (bool, optional): If ``True``, the data loader will copy Tensors
+            into device/CUDA pinned memory before returning them.  If your data elements
+            are a custom type, or your :attr:`collate_fn` returns a batch that is a custom type,
+            see the example below.
+    """
+    if shuffle:
+        dataset = dataset.shuffle(seed)
+    if skip_batches > 0:
+        dataset = dataset.skip(skip_batches * batch_size)
+    if prefetch_size > 0:
+        dataset = IterableWrapper(dataset)
+        dataset = dataset.prefetch(prefetch_size)
+    num_of_hosts = jax.process_count()
+    dataset = split_dataset_by_node(dataset, rank=jax.process_index(), world_size=num_of_hosts)
+    assert batch_size % num_of_hosts == 0, "Batch size must be divisible by the number of hosts."
+    if dataset.n_shards < dataloader_num_workers:
+        dataloader_num_workers = dataset.n_shards
+    data_loader = DataLoader(
+        dataset,
+        batch_size=batch_size //num_of_hosts,
+        drop_last=drop_last,
+        pin_memory=pin_memory,
+        collate_fn=data_collator,
+        num_workers=dataloader_num_workers,
+    )
+    return data_loader
+def sorted_checkpoints(output_dir=None, checkpoint_prefix="checkpoint", use_mtime=False) -> List[str]:
+    ordering_and_checkpoint_path = []
+    glob_checkpoints = [str(x) for x in Path(output_dir).glob(f"{checkpoint_prefix}-*") if os.path.isdir(x)]
+    for path in glob_checkpoints:
+        if use_mtime:
+            ordering_and_checkpoint_path.append((os.path.getmtime(path), path))
+        else:
+            regex_match = re.match(f".*{checkpoint_prefix}-([0-9]+)", path)
+            if regex_match is not None and regex_match.groups() is not None:
+                ordering_and_checkpoint_path.append((int(regex_match.groups()[0]), path))
+    checkpoints_sorted = sorted(ordering_and_checkpoint_path)
+    checkpoints_sorted = [checkpoint[1] for checkpoint in checkpoints_sorted]
+    return checkpoints_sorted
+def rotate_checkpoints(
+    save_total_limit=None, use_mtime=False, output_dir=None, checkpoint_prefix="checkpoint"
+) -> None:
+    if save_total_limit is None or save_total_limit <= 0:
+        return
+    # Check if we should delete older checkpoint(s)
+    checkpoints_sorted = sorted_checkpoints(
+        use_mtime=use_mtime, output_dir=output_dir, checkpoint_prefix=checkpoint_prefix
+    )
+    if len(checkpoints_sorted) <= save_total_limit:
+        return
+    number_of_checkpoints_to_delete = max(0, len(checkpoints_sorted) - save_total_limit)
+    checkpoints_to_be_deleted = checkpoints_sorted[:number_of_checkpoints_to_delete]
+    for checkpoint in checkpoints_to_be_deleted:
+        logger.info(f"Deleting older checkpoint [{checkpoint}] due to args.save_total_limit")
+        shutil.rmtree(checkpoint, ignore_errors=True)
+def to_fp32(t):
+    return jax.tree_map(lambda x: x.astype(jnp.float32) if x.dtype == jnp.bfloat16 else x, t)
+def to_bf16(t):
+    return jax.tree_map(lambda x: x.astype(jnp.bfloat16) if x.dtype == jnp.float32 else x, t)
+class TrainState(train_state.TrainState):
+    dropout_rng: jnp.ndarray
+    max_grad_norm: float
+    def apply_gradients(self, *, grads, to_dtype: to_fp32, **kwargs):
+        """Updates `step`, `params`, `opt_state` and `**kwargs` in return value, clipping the
+        gradients by the maximum grad norm.
+        Note that internally this function calls `.tx.update()` followed by a call
+        to `optax.apply_updates()` to update `params` and `opt_state`.
+        Args:
+          grads: Gradients that have the same pytree structure as `.params`.
+          **kwargs: Additional dataclass attributes that should be `.replace()`-ed.
+        Returns:
+          An updated instance of `self` with `step` incremented by one, `params`
+          and `opt_state` updated by applying `grads`, and additional attributes
+          replaced as specified by `kwargs`.
+        """
+        # clip gradients by global l2 norm
+        casted_max_grad_norm = to_dtype(self.max_grad_norm)
+        g_norm = linear_algebra.global_norm(grads)
+        g_norm = jnp.maximum(casted_max_grad_norm, g_norm)
+        grads = jax.tree_map(lambda t: (t / g_norm) * casted_max_grad_norm, grads)
+        # perform update step in fp32 and subsequently downcast optimizer states if mixed precision training
+        # grads and opt_state in bf16 (need to upcast), params in fp32 (leave as is)
+        updates, new_opt_state = self.tx.update(to_fp32(grads), to_fp32(self.opt_state), self.params)
+        new_params = optax.apply_updates(self.params, updates)
+        return self.replace(
+            step=self.step + 1,
+            params=new_params,
+            opt_state=to_dtype(new_opt_state),
+            **kwargs,
+        )
+    @classmethod
+    def create(cls, *, apply_fn, params, tx, to_dtype: to_fp32, **kwargs):
+        """Creates a new instance with `step=0` and initialized `opt_state`."""
+        # downcast optimizer state to bf16 if mixed-precision training
+        opt_state = tx.init(to_dtype(params))
+        return cls(
+            step=0,
+            apply_fn=apply_fn,
+            params=params,
+            tx=tx,
+            opt_state=opt_state,
+            **kwargs,
+        )
+    def replicate(self):
+        return jax_utils.replicate(self).replace(dropout_rng=shard_prng_key(self.dropout_rng))
+    def unreplicate(self):
+        return jax_utils.unreplicate(self)
+    def save_state(self, output_dir, save_total_limit=None, checkpoint_prefix="checkpoint"):
+        step = int(jax.device_get(unreplicate(self.step)))
+        serialized_state = to_bytes(self.unreplicate())
+        output_file = Path(os.path.join(output_dir, f"{checkpoint_prefix}-{step}", "train_state.msgpack"))
+        output_file.parent.mkdir(exist_ok=True, parents=True)
+        with output_file.open("wb") as f:
+            f.write(serialized_state)
+        logger.info(f"Flax train state saved in {output_file}")
+        rotate_checkpoints(
+            save_total_limit=save_total_limit, output_dir=output_dir, checkpoint_prefix=checkpoint_prefix
+        )
+def save_hf_weights(
+    student_state: TrainState,
+    student_model: FlaxWhisperForConditionalGeneration,
+    processor: WhisperProcessor,
+    output_dir: str,
+    cur_step: int,
+    total_train_steps: int,
+    use_scan: bool = True,
+    checkpoint_prefix: str = "checkpoint",
+) -> None:
+    # always disable scan in the params / model so that we can load from PyTorch directly - this is a no-op if we're not using scan for training
+    student_state_params = unreplicate(student_state.params)
+    student_state_params = student_model.convert_scan_to_unroll(student_state_params)
+    student_params = jax.device_get(student_state_params)
+    student_model.disable_scan()
+    if cur_step != total_train_steps:
+        output_dir = os.path.join(output_dir, f"{checkpoint_prefix}-{cur_step}")
+        os.makedirs(output_dir, exist_ok=True)
+    student_model.save_pretrained(output_dir, params=student_params)
+    processor.save_pretrained(output_dir)
+    # re-enable scan only if required for training
+    if use_scan:
+        student_model.enable_scan()
+def write_train_metric(summary_writer, train_metrics, train_time, step, logging_steps):
+    summary_writer.scalar("train/time", train_time, step)
+    # Check if train_metrics is empty
+    if not train_metrics:
+        print("DEBUG: train_metrics is empty; This is probably a bug that needs fixing.")
+        return  # Early exit if train_metrics is empty to avoid further processing
+    train_metrics = get_metrics(train_metrics)
+    for key, vals in train_metrics.items():
+        steps_arr = np.arange(0, step, logging_steps)[-len(vals) :]
+        tag = f"train/{key}"
+        for i, val in enumerate(vals):
+            summary_writer.scalar(tag, val, steps_arr[i])
+def write_eval_metric(summary_writer, eval_metrics, step, prefix="eval"):
+    for metric_name, value in eval_metrics.items():
+        summary_writer.scalar(f"{prefix}/{metric_name}", value, step)
+def write_wandb_metric(wandb_logger, metrics, train_time, step, epoch, prefix="train"):
+    log_metrics = {}
+    for k, v in metrics.items():
+        log_metrics[f"{prefix}/{k}"] = v
+    log_metrics[f"{prefix}/time"] = train_time
+    log_metrics[f"{prefix}/epoch"] = epoch
+    wandb_logger.log(log_metrics, step)
+def write_wandb_pred(
+    wandb_logger, pred_str, label_str, norm_pred_str, norm_label_str, cur_step, prefix="eval", num_lines=200000
+):
+    # pretty name for current step: step 50000 -> step 50k
+    cur_step_pretty = f"{int(cur_step // 1000)}k" if cur_step > 1000 else cur_step
+    # convert str data to a wandb compatible format
+    str_data = [[label_str[i], pred_str[i], norm_label_str[i], norm_pred_str[i]] for i in range(len(pred_str))]
+    # log as a table with the appropriate headers
+    wandb_logger.log(
+        {
+            f"predictions/{prefix.replace('/', '-')}-step-{cur_step_pretty}": wandb_logger.Table(
+                columns=["Target", "Pred", "Norm Target", "Norm Pred"], data=str_data[:num_lines]
+            )
+        },
+        cur_step,
+    )
+    # log incorrect normalised predictions
+    str_data = np.asarray(str_data)
+    str_data_incorrect = str_data[str_data[:, -2] != str_data[:, -1]]
+    # log as a table with the appropriate headers
+    wandb_logger.log(
+        {
+            f"incorrect_predictions/{prefix.replace('/', '-')}-step-{cur_step_pretty}": wandb_logger.Table(
+                columns=["Target", "Pred", "Norm Target", "Norm Pred"], data=str_data_incorrect[:num_lines]
+            )
+        },
+        cur_step,
+    )
+def create_learning_rate_fn(
+    num_train_steps: int, lr_scheduler_type: str, num_warmup_steps: int, learning_rate: float
+) -> Callable[[int], jnp.array]:
+    """Returns a linear warmup, linear_decay learning rate function."""
+    lr_scheduler_types = ("linear", "constant_with_warmup")
+    if lr_scheduler_type not in lr_scheduler_types:
+        raise ValueError(
+            f"lr_scheduler_type of type {lr_scheduler_type} not supported, choose from {lr_scheduler_types}."
+        )
+    warmup_fn = optax.linear_schedule(init_value=0.0, end_value=learning_rate, transition_steps=num_warmup_steps)
+    decay_fn = optax.linear_schedule(
+        init_value=learning_rate,
+        end_value=0 if lr_scheduler_type == "linear" else learning_rate,
+        transition_steps=num_train_steps - num_warmup_steps,
+    )
+    schedule_fn = optax.join_schedules(schedules=[warmup_fn, decay_fn], boundaries=[num_warmup_steps])
+    return schedule_fn
+def convert_dataset_str_to_list(
+    dataset_names,
+    dataset_config_names,
+    splits=None,
+    text_column_names=None,
+    dataset_samples=None,
+    default_split="train",
+):
+    if isinstance(dataset_names, str):
+        dataset_names = dataset_names.split("+")
+        # we assume that all the datasets we're using derive from the distil-whisper org on the Hub - prepend the org name if necessary
+        for i in range(len(dataset_names)):
+            ds_name = dataset_names[i]
+            dataset_names[i] = f"distil-whisper/{ds_name}" if "/" not in ds_name else ds_name
+        dataset_config_names = dataset_config_names.split("+")
+        splits = splits.split("+") if splits is not None else None
+        text_column_names = text_column_names.split("+") if text_column_names is not None else None
+        dataset_samples = dataset_samples.split("+") if dataset_samples is not None else None
+    # basic checks to ensure we've got the right number of datasets/configs/splits/columns/probs
+    if len(dataset_names) != len(dataset_config_names):
+        raise ValueError(
+            f"Ensure one config is passed for each dataset, got {len(dataset_names)} datasets and"
+            f" {len(dataset_config_names)} configs."
+        )
+    if splits is not None and len(splits) != len(dataset_names):
+        raise ValueError(
+            f"Ensure one split is passed for each dataset, got {len(dataset_names)} datasets and {len(splits)} splits."
+        )
+    if text_column_names is not None and len(text_column_names) != len(dataset_names):
+        raise ValueError(
+            f"Ensure one text column name is passed for each dataset, got {len(dataset_names)} datasets and"
+            f" {len(text_column_names)} text column names."
+        )
+    if dataset_samples is not None:
+        if len(dataset_samples) != len(dataset_names):
+            raise ValueError(
+                f"Ensure one sample is passed for each dataset, got {len(dataset_names)} datasets and "
+                f"{len(dataset_samples)} samples."
+            )
+        dataset_samples = [float(ds_sample) for ds_sample in dataset_samples]
+    else:
+        dataset_samples = [None] * len(dataset_names)
+    text_column_names = (
+        text_column_names if text_column_names is not None else ["text" for _ in range(len(dataset_names))]
+    )
+    splits = splits if splits is not None else [default_split for _ in range(len(dataset_names))]
+    dataset_names_dict = []
+    for i, ds_name in enumerate(dataset_names):
+        dataset_names_dict.append(
+            {
+                "name": ds_name,
+                "config": dataset_config_names[i],
+                "split": splits[i],
+                "text_column_name": text_column_names[i],
+                "samples": dataset_samples[i],
+            }
+        )
+    return dataset_names_dict
+def load_multiple_datasets(
+    dataset_names: Union[List, str],
+    dataset_config_names: Union[List, str],
+    splits: Optional[Union[List, str]] = None,
+    text_column_names: Optional[List] = None,
+    sampling_rate: Optional[int] = 16000,
+    stopping_strategy: Optional[str] = "first_exhausted",
+    dataset_samples: Optional[Union[List, np.array]] = None,
+    streaming: bool = True,
+    seed: int = None,
+    **kwargs,
+) -> IterableDataset:
+    dataset_names_dict = convert_dataset_str_to_list(
+        dataset_names, dataset_config_names, splits, text_column_names, dataset_samples
+    )
+    if dataset_samples is not None:
+        dataset_samples = [ds_dict["samples"] for ds_dict in dataset_names_dict]
+        probabilities = np.array(dataset_samples) / np.sum(dataset_samples)
+    else:
+        probabilities = None
+    if len(dataset_names_dict) == 1:
+        dataset_dict = dataset_names_dict[0]
+        # we have a single dataset so just return it as is
+        return load_dataset(
+            dataset_dict["name"],
+            dataset_dict["config"],
+            split=dataset_dict["split"],
+            streaming=streaming,
+            **kwargs,
+        )
+    all_datasets = []
+    # iterate over the datasets we want to interleave
+    for dataset_dict in tqdm(dataset_names_dict, desc="Combining datasets..."):
+        dataset = load_dataset(
+            dataset_dict["name"],
+            dataset_dict["config"],
+            split=dataset_dict["split"],
+            streaming=streaming,
+            **kwargs,
+        )
+        # resample to specified sampling rate
+        dataset = dataset.cast_column("audio", datasets.features.Audio(sampling_rate))
+        dataset = dataset.remove_columns(
+            set(dataset.features.keys()) - {"audio", dataset_dict["text_column_name"], "whisper_transcript"}
+        )
+        all_datasets.append(dataset)
+    if streaming:
+        interleaved_dataset = interleave_datasets(
+            all_datasets,
+            stopping_strategy=stopping_strategy,
+            probabilities=probabilities,
+            seed=seed,
+        )
+    else:
+        interleaved_dataset = concatenate_datasets(all_datasets)
+    return interleaved_dataset
+def get_layers_to_supervise(student_layers: int, teacher_layers: int) -> dict:
+    """Helper function to map the student layer i to the teacher layer j whose output we'd like them to emulate. Used
+    for MSE loss terms in distillation (hidden-states and activations). Student layers are paired with teacher layers
+    in equal increments, e.g. for a 12-layer model distilled to a 3-layer model, student layer 0 emulates teacher layer
+    3 (such that it behaves like the first 4 teacher layers), student layer 1 emulates teacher layer 7, and student layer
+    2 emulates teacher layer 11. This mapping is summarised by the dictionary: {0: 3, 1: 7, 2: 11}, which is precisely
+    the output of this function for the arguments (student_layers=3, teacher_layers=12)."""
+    layer_intervals = np.linspace(teacher_layers // student_layers - 1, teacher_layers - 1, student_layers, dtype=int)
+    layer_intervals[-1] = teacher_layers - 1
+    layer_map = {}
+    for student_layer, teacher_layer in enumerate(layer_intervals):
+        layer_map[student_layer] = teacher_layer
+    return layer_map
+class FlaxWhisperFeatureExtractor(WhisperFeatureExtractor):
+    def _np_extract_fbank_features(self, waveform: np.array) -> np.ndarray:
+        """
+        Compute the log-mel spectrogram of the provided audio using torch filters. Using the torch implementation
+        computes stft filter banks approx 5x faster than its numpy counterpart, which is the native implementation
+        in transformers, and matches to within 1e-5 abs tolerance.
+        """
+        waveform = torch.from_numpy(waveform).type(torch.float32)
+        window = torch.hann_window(self.n_fft)
+        stft = torch.stft(waveform, self.n_fft, self.hop_length, window=window, return_complex=True)
+        magnitudes = stft[..., :-1].abs() ** 2
+        mel_filters = torch.from_numpy(self.mel_filters).type(torch.float32)
+        mel_spec = mel_filters.T @ magnitudes
+        log_spec = torch.clamp(mel_spec, min=1e-10).log10()
+        log_spec = torch.maximum(log_spec, log_spec.max() - 8.0)
+        log_spec = (log_spec + 4.0) / 4.0
+        return log_spec.numpy()
+def main():
+    # 1. Parse input arguments
+    # See all possible arguments in src/transformers/training_args.py
+    # or by passing the --help flag to this script.
+    # We now keep distinct sets of args, for a cleaner separation of concerns.
+    parser = HfArgumentParser((ModelArguments, DataTrainingArguments, FlaxSeq2SeqTrainingArguments))
+    if len(sys.argv) == 2 and sys.argv[1].endswith(".json"):
+        # If we pass only one argument to the script and it's the path to a json file,
+        # let's parse it to get our arguments.
+        model_args, data_args, training_args = parser.parse_json_file(json_file=os.path.abspath(sys.argv[1]))
+    else:
+        model_args, data_args, training_args = parser.parse_args_into_dataclasses()
+    # Sending telemetry. Tracking the example usage helps us better allocate resources to maintain them. The
+    # information sent is the one passed as arguments along with your JAX/Flax versions.
+    send_example_telemetry("run_flax_speech_recognition_seq2seq", model_args, data_args, framework="flax")
+    # 2. Define remote logging - do this early so that we get the full traceback on our remote logs
+    # Enable tensorboard only on the master node
+    has_tensorboard = is_tensorboard_available()
+    if has_tensorboard:
+        if jax.process_index() == 0:
+            try:
+                from flax.metrics.tensorboard import SummaryWriter
+                summary_writer = SummaryWriter(log_dir=os.path.join(Path(training_args.output_dir), "runs"))
+            except ImportError as ie:
+                has_tensorboard = False
+                logger.warning(
+                    "Unable to display metrics through TensorBoard because some package" f" are not installed: {ie}"
+                )
+    else:
+        logger.warning(
+            "Unable to display metrics through TensorBoard because the package is not"
+            " installed: Please run `pip install tensorboard` to enable."
+        )
+    # Enable wandb only on the master node
+    has_wandb = is_wandb_available()
+    if has_wandb:
+        import wandb as wandb_logger
+        # Set up wandb run
+        if jax.process_index() == 0:
+            wandb_logger.init(
+                project=data_args.wandb_project,
+                name=data_args.wandb_name,
+                job_type=data_args.wandb_job_type,
+                dir=data_args.wandb_dir,
+                save_code=data_args.save_code_to_wandb,
+            )
+    else:
+        logger.warning("Wandb logging requires wandb to be installed. Run `pip install wandb` to enable.")
+    # 3. Setup local logging
+    # Make one log on every process with the configuration for debugging.
+    logging.basicConfig(
+        format="%(asctime)s - %(levelname)s - %(name)s - %(message)s",
+        datefmt="%m/%d/%Y %H:%M:%S",
+        handlers=[logging.StreamHandler(sys.stdout)],
+    )
+    # Set the verbosity to info of the Transformers logger.
+    # We only want one process per machine to log things on the screen.
+    logger.setLevel(logging.INFO if jax.process_index() == 0 else logging.ERROR)
+    if jax.process_index() == 0:
+        datasets.utils.logging.set_verbosity_warning()
+        transformers.utils.logging.set_verbosity_info()
+    else:
+        datasets.utils.logging.set_verbosity_error()
+        transformers.utils.logging.set_verbosity_error()
+    logger.info("Training/evaluation parameters %s", training_args)
+    # Check the output dir is valid
+    if (
+        os.path.exists(training_args.output_dir)
+        and os.listdir(training_args.output_dir)
+        and training_args.do_train
+        and not training_args.overwrite_output_dir
+    ):
+        raise ValueError(
+            f"Output directory ({training_args.output_dir}) already exists and is not"
+            " empty. Use `--overwrite_output_dir` to overcome."
+        )
+    # 4. Handle the repository creation
+    if training_args.push_to_hub:
+        if training_args.hub_model_id is None:
+            repo_name = get_full_repo_name(
+                Path(training_args.output_dir).absolute().name,
+                token=training_args.hub_token,
+            )
+        else:
+            repo_name = training_args.hub_model_id
+        create_repo(repo_name, exist_ok=True, token=training_args.hub_token)
+        repo = Repository(
+            training_args.output_dir,
+            clone_from=repo_name,
+            token=training_args.hub_token,
+        )
+    if training_args.compilation_cache:
+        cc.initialize_cache(os.path.join(model_args.cache_dir, "jax_cache"))
+    # 5. Load dataset
+    raw_datasets = IterableDatasetDict() if data_args.streaming else DatasetDict()
+    # set seed for determinism
+    set_seed(training_args.seed)
+    if training_args.do_train:
+        raw_datasets["train"] = load_multiple_datasets(
+            data_args.train_dataset_name,
+            data_args.train_dataset_config_name,
+            splits=data_args.train_split_name,
+            streaming=data_args.streaming,
+            dataset_samples=data_args.train_dataset_samples,
+            seed=training_args.seed,
+            trust_remote_code=True,
+            cache_dir=data_args.dataset_cache_dir,
+            token=True if model_args.use_auth_token else None,
+        )
+    if training_args.do_eval:
+        dataset_names_dict = convert_dataset_str_to_list(
+            data_args.eval_dataset_name if data_args.eval_dataset_name else data_args.train_dataset_name,
+            (
+                data_args.eval_dataset_config_name
+                if data_args.eval_dataset_config_name
+                else data_args.train_dataset_config_name
+            ),
+            splits=data_args.eval_split_name,
+            text_column_names=data_args.eval_text_column_name,
+        )
+        all_eval_splits = []
+        if len(dataset_names_dict) == 1:
+            # load a single eval set
+            dataset_dict = dataset_names_dict[0]
+            all_eval_splits.append("eval")
+            raw_datasets["eval"] = load_dataset(
+                dataset_dict["name"],
+                dataset_dict["config"],
+                split=dataset_dict["split"],
+                trust_remote_code=True,
+                cache_dir=data_args.dataset_cache_dir,
+                token=True if model_args.use_auth_token else None,
+                streaming=data_args.streaming,
+            )
+        else:
+            # load multiple eval sets
+            for dataset_dict in dataset_names_dict:
+                if dataset_dict["name"] == "esb/diagnostic-dataset":
+                    # for the ESB diagnostic dataset, the dataset name is effectively the config
+                    pretty_name = f"{dataset_dict['config']}-diagnostic/{dataset_dict['split']}"
+                else:
+                    pretty_name = f"{dataset_dict['name'].split('/')[-1]}/{dataset_dict['split'].replace('.', '-')}"
+                all_eval_splits.append(pretty_name)
+                raw_datasets[pretty_name] = load_dataset(
+                    dataset_dict["name"],
+                    dataset_dict["config"],
+                    split=dataset_dict["split"],
+                    cache_dir=data_args.dataset_cache_dir,
+                    token=True if model_args.use_auth_token else None,
+                    trust_remote_code=True,
+                    streaming=data_args.streaming,
+                )
+                features = raw_datasets[pretty_name].features.keys()
+                if "text" not in features:
+                    raw_datasets[pretty_name] = raw_datasets[pretty_name].rename_column(
+                        dataset_dict["text_column_name"], "text"
+                    )
+                raw_datasets[pretty_name] = raw_datasets[pretty_name].remove_columns(
+                    set(raw_datasets[pretty_name].features.keys()) - {"audio", "text"}
+                )
+    if not training_args.do_train and not training_args.do_eval:
+        raise ValueError(
+            "Cannot not train and not do evaluation. At least one of training or evaluation has to be performed."
+        )
+    raw_datasets_train_features = list(raw_datasets["train"].features.keys())
+    if data_args.audio_column_name not in raw_datasets_train_features:
+        raise ValueError(
+            f"--audio_column_name '{data_args.audio_column_name}' not found in dataset"
+            f" '{data_args.dataset_name}'. Make sure to set `--audio_column_name` to"
+            " the correct audio column - one of"
+            f" {', '.join(raw_datasets_train_features)}."
+        )
+    if data_args.train_text_column_name not in raw_datasets_train_features:
+        raise ValueError(
+            f"--train_text_column_name {data_args.train_text_column_name} not found in dataset"
+            f" '{data_args.dataset_name}'. Make sure to set `--train_text_column_name` to the"
+            " correct text column - one of"
+            f" {', '.join(raw_datasets_train_features)}."
+        )
+    # 6. Load pretrained model, tokenizer, and feature extractor
+    config = WhisperConfig.from_pretrained(
+        (model_args.config_name if model_args.config_name else model_args.model_name_or_path),
+        cache_dir=model_args.cache_dir,
+        revision=model_args.model_revision,
+        token=True if model_args.use_auth_token else None,
+    )
+    feature_extractor = FlaxWhisperFeatureExtractor.from_pretrained(
+        (model_args.feature_extractor_name if model_args.feature_extractor_name else model_args.model_name_or_path),
+        cache_dir=model_args.cache_dir,
+        revision=model_args.model_revision,
+        token=True if model_args.use_auth_token else None,
+    )
+    tokenizer = WhisperTokenizerFast.from_pretrained(
+        (model_args.tokenizer_name if model_args.tokenizer_name else model_args.model_name_or_path),
+        cache_dir=model_args.cache_dir,
+        use_fast=model_args.use_fast_tokenizer,
+        revision=model_args.model_revision,
+        token=True if model_args.use_auth_token else None,
+    )
+    # override timestamp tokens until tokenizer issues are fixed in transformers
+    timestamps = [AddedToken("<|%.2f|>" % (i * 0.02), lstrip=False, rstrip=False) for i in range(1500 + 1)]
+    tokenizer.add_tokens(timestamps)
+    config.update(
+        {
+            "activation_dropout": model_args.activation_dropout,
+            "attention_dropout": model_args.attention_dropout,
+            "dropout": model_args.dropout,
+        }
+    )
+    if training_args.precision == "full_mixed":
+        # forward pass, backward pass and optimiser states in bf16
+        dtype = jnp.bfloat16
+        to_dtype = to_bf16
+    elif training_args.precision == "half_mixed" or model_args.dtype == "bfloat16":
+        # forward pass in bf16, backward pass and optimiser states in fp32
+        dtype = jnp.bfloat16
+        to_dtype = to_fp32
+    else:
+        if training_args.precision != "full":
+            raise ValueError(
+                f"`precision` should be one of: `full`, `half_mixed` or `full_mixed`, got {training_args.precision}"
+            )
+        # forward pass, backward pass and optimiser states in fp32
+        dtype = jnp.float32
+        to_dtype = to_fp32
+    student_model, student_params = FlaxWhisperForConditionalGeneration.from_pretrained(
+        model_args.model_name_or_path,
+        config=config,
+        dtype=dtype,
+        cache_dir=model_args.cache_dir,
+        revision=model_args.model_revision,
+        subfolder=model_args.subfolder,
+        token=True if model_args.use_auth_token else None,
+        _do_init=False,
+        use_scan=model_args.load_with_scan_weights,
+    )
+    teacher_model, teacher_params = FlaxWhisperForConditionalGeneration.from_pretrained(
+        model_args.teacher_model_name_or_path,
+        # config=config,
+        dtype=dtype,
+        cache_dir=model_args.cache_dir,
+        # revision=model_args.model_revision,
+        token=True if model_args.use_auth_token else None,
+        _do_init=False,
+    )
+    if student_model.config.decoder_start_token_id is None or teacher_model.config.decoder_start_token_id is None:
+        raise ValueError(
+            f"Make sure that `config.decoder_start_token_id` is correctly defined for both the "
+            f"student and teacher model. Got {student_model.config.decoder_start_token_id} for the "
+            f"student and {teacher_model.config.decoder_start_token_id} for the teacher."
+        )
+    # enable scan / gradient checkpointing if necessary
+    if training_args.use_scan:
+        student_model.enable_scan()  # to enable scan in the nn.Module
+        student_params = student_model.convert_unroll_to_scan(student_params)  # to convert the unrolled params to scan
+        teacher_model.enable_scan()  # faster compile time (even though we don't train the teacher)
+        teacher_params = teacher_model.convert_unroll_to_scan(teacher_params)
+    if training_args.gradient_checkpointing:
+        student_model.enable_gradient_checkpointing()  # to enable checkpointing in the nn.Module, there is no change to the params structure
+        teacher_model.enable_gradient_checkpointing()
+    if hasattr(teacher_model.generation_config, "is_multilingual") and teacher_model.generation_config.is_multilingual:
+        # We need to set the language and task ids for previously multilingual checkpoints - for now we hardcode this to Norwegian
+        tokenizer.set_prefix_tokens(language="Norwegian", task="transcribe", predict_timestamps=False)
+        student_model.generation_config.update(
+            **{
+                "language": "<|no|>",
+                "task": "transcribe",
+            }
+        )
+    # 7. Resample speech dataset: `datasets` takes care of automatically loading and resampling the audio,
+    # so we just need to set the correct target sampling rate.
+    raw_datasets = raw_datasets.cast_column(
+        data_args.audio_column_name,
+        datasets.features.Audio(sampling_rate=feature_extractor.sampling_rate),
+    )
+    # 8. Preprocessing the datasets.
+    # We need to read the audio files as arrays and tokenize the targets.
+    max_input_length = int(data_args.max_duration_in_seconds * feature_extractor.sampling_rate)
+    min_input_length = int(data_args.min_duration_in_seconds * feature_extractor.sampling_rate)
+    max_label_length = (
+        data_args.max_label_length if data_args.max_label_length is not None else student_model.config.max_length
+    )
+    audio_column_name = data_args.audio_column_name
+    num_workers = data_args.preprocessing_num_workers
+    dataloader_num_workers = training_args.dataloader_num_workers
+    dataloader_prefetch_size = data_args.prefetch_size
+    train_text_column_name = data_args.train_text_column_name
+    eval_text_column_name = "text"
+    model_input_name = feature_extractor.model_input_names[0]
+    #normalizer = BasicTextNormalizer(tokenizer.english_spelling_normalizer)
+    normalizer = BasicTextNormalizer()
+    wer_threshold = data_args.wer_threshold
+    round_timestamps = data_args.round_timestamps
+    if training_args.do_train and data_args.max_train_samples is not None:
+        raw_datasets["train"] = (
+            raw_datasets["train"].take(data_args.max_train_samples)
+            if data_args.streaming
+            else raw_datasets["train"].select(range(data_args.max_train_samples))
+        )
+    if training_args.do_eval and data_args.max_eval_samples is not None:
+        for eval_split in all_eval_splits:
+            raw_datasets[eval_split] = (
+                raw_datasets[eval_split].take(data_args.max_eval_samples)
+                if data_args.streaming
+                else raw_datasets[eval_split].select(range(data_args.max_eval_samples))
+            )
+    # 10.3: filter training data based on WER threshold -> this is KEY to good distillation performance
+    def is_wer_in_range(ground_truth, whisper_transcript):
+        norm_ground_truth = normalizer(ground_truth)
+        if whisper_transcript is not None and whisper_transcript.upper() == whisper_transcript:
+            # filter entirely upper-case transcriptions: these are erroneous generations from large-v3
+            return False
+        elif len(norm_ground_truth) == 0 and len(normalizer(whisper_transcript)) == 0:
+            return True
+        elif len(norm_ground_truth.strip()) > 0 and whisper_transcript is not None and len(normalizer(whisper_transcript).strip()) > 0:
+            norm_whisper_transcript = normalizer(whisper_transcript)
+            wer = 100 * metric.compute(predictions=[norm_whisper_transcript], references=[norm_ground_truth])
+            return wer < wer_threshold
+        else:
+            # filter automatically since we cant know WER
+            return False
+    filter_by_wer_threshold = partial(
+        raw_datasets["train"].filter,
+        function=is_wer_in_range,
+        input_columns=[eval_text_column_name, train_text_column_name],
+    )
+    if wer_threshold is not None:
+        raw_datasets["train"] = (
+            filter_by_wer_threshold(num_proc=num_workers, desc="filtering train dataset by wer")
+            if not data_args.streaming
+            else filter_by_wer_threshold()
+        )
+    def has_timestamp_tokens(input_str):
+        """
+        Identify whether the input string contains timestamp tokens, of the form <|0.00|>, by searching for
+        pairs of left and right-angle brackets.
+        """
+        return bool(re.search("\<[^\>]*\>", input_str))
+    def round_timestamp_tokens(input_str: str, ndigits: int = 1):
+        timestamps = re.findall("\<[^\>]*\>", input_str, re.DOTALL)
+        for token in timestamps:
+            # extract time digits from timestamp token, e.g. <|6.24|> to 6.24
+            time_digit = token[2:-2]
+            # round to specified number of digits, e.g. 6.24 to 6.2
+            time_digit = round(float(time_digit), ndigits=ndigits)
+            # replace in original string with the same precision, e.g. <|6.24|> to <|6.20|>
+            input_str = input_str.replace(token, "<|{:.2f}|>".format(time_digit))
+        return input_str
+    def prepare_train_dataset(batch):
+        # process audio input
+        sample = batch[audio_column_name]
+        inputs = feature_extractor(sample["array"], sampling_rate=sample["sampling_rate"])
+        batch[model_input_name] = inputs.get(model_input_name)[0]
+        batch["input_length"] = len(sample["array"])
+        # process text targets
+        input_str = batch[train_text_column_name]
+        # prompt & timestamp processing: for now, we only do one or the other
+        if input_str.startswith("<|startoftranscript|>") or input_str.startswith("<|startofprev|>"):
+            # prompted target text already has special ids added, so don't add them here
+            batch["labels"] = tokenizer(input_str, add_special_tokens=False).input_ids
+            return batch
+        has_timestamps = has_timestamp_tokens(input_str)
+        if has_timestamps:
+            predict_timestamps = bool(np.random.binomial(1, data_args.timestamp_probability))
+            if not predict_timestamps:
+                # filter timestamp token ids if not part of the prediction task
+                input_str = tokenizer._filter_timestamp_ids(input_str)
+            elif round_timestamps:
+                input_str = round_timestamp_tokens(input_str)
+        else:
+            predict_timestamps = False
+        tokenizer.set_prefix_tokens(language="Norwegian", task="transcribe", predict_timestamps=predict_timestamps)
+        input_ids = tokenizer(input_str).input_ids
+        batch["labels"] = input_ids
+        return batch
+    def prepare_eval_dataset(batch):
+        # process audio
+        sample = batch[audio_column_name]
+        inputs = feature_extractor(sample["array"], sampling_rate=sample["sampling_rate"])
+        # process audio length
+        batch[model_input_name] = inputs.get(model_input_name)[0]
+        batch["input_length"] = len(sample["array"])
+        # process targets
+        input_str = batch[eval_text_column_name]
+        batch["labels"] = tokenizer(input_str).input_ids
+        return batch
+    vectorized_datasets = IterableDatasetDict() if data_args.streaming else DatasetDict()
+    if training_args.do_train:
+        map_fn_train = partial(
+            raw_datasets["train"].map, function=prepare_train_dataset, remove_columns=raw_datasets_train_features
+        )
+        vectorized_datasets["train"] = (
+            map_fn_train(num_proc=num_workers, desc="preprocess train dataset")
+            if not data_args.streaming
+            else map_fn_train()
+        )
+    if training_args.do_eval:
+        for eval_split in all_eval_splits:
+            raw_datasets_eval_features = list(raw_datasets[eval_split].features.keys())
+            map_fn_eval = partial(
+                raw_datasets[eval_split].map, function=prepare_eval_dataset, remove_columns=raw_datasets_eval_features
+            )
+            vectorized_datasets[eval_split] = (
+                map_fn_eval(num_proc=num_workers, desc="preprocess eval dataset")
+                if not data_args.streaming
+                else map_fn_eval()
+            )
+    # filter training data with inputs longer than max_input_length
+    def is_audio_in_length_range(length):
+        return min_input_length < length < max_input_length
+    filter_by_audio_fn = partial(
+        vectorized_datasets.filter, function=is_audio_in_length_range, input_columns=["input_length"]
+    )
+    vectorized_datasets = (
+        filter_by_audio_fn(num_proc=num_workers, desc="filtering train dataset by audio length")
+        if not data_args.streaming
+        else filter_by_audio_fn()
+    )
+    # filter training data with labels longer than max_label_length
+    def is_labels_in_length_range(labels):
+        return 0 < len(labels) < max_label_length
+    filter_by_labels_fn = partial(
+        vectorized_datasets.filter, function=is_labels_in_length_range, input_columns=["labels"]
+    )
+    vectorized_datasets = (
+        filter_by_labels_fn(num_proc=num_workers, desc="filtering train dataset")
+        if not data_args.streaming
+        else filter_by_labels_fn()
+    )
+    # for large datasets it is advised to run the preprocessing on a
+    # single machine first with `args.preprocessing_only` since there will mostly likely
+    # be a timeout when running the script in distributed mode.
+    # In a second step `args.preprocessing_only` can then be set to `False` to load the
+    # cached dataset
+    if data_args.preprocessing_only:
+        cache = {k: v.cache_files for k, v in vectorized_datasets.items()}
+        logger.info(f"Data preprocessing finished. Files cached at {cache}.")
+        return
+    # 8. Load Metric
+    metric = evaluate.load("wer")
+    # convention is that we space all punctuation *except* apostrophes
+    all_punctuation = list(string.punctuation.replace("'", ""))
+    return_timestamps = data_args.return_timestamps if data_args.timestamp_probability > 0 else False
+    def compute_metrics(preds, labels):
+        # replace padded labels by the padding token
+        for idx in range(len(labels)):
+            labels[idx][labels[idx] == -100] = tokenizer.pad_token_id
+        pred_str = tokenizer.batch_decode(preds, skip_special_tokens=True, decode_with_timestamps=return_timestamps)
+        # we do not want to group tokens when computing the metrics
+        label_str = tokenizer.batch_decode(labels, skip_special_tokens=True)
+        # space punctuation for orthographic WER (c.f. ESB paper https://arxiv.org/abs/2210.13352)
+        spaced_pred_str = [
+            pred_str[i].replace(punctuation, f" {punctuation} ")
+            for punctuation in all_punctuation
+            for i in range(len(pred_str))
+        ]
+        spaced_label_str = [
+            label_str[i].replace(punctuation, f" {punctuation} ")
+            for punctuation in all_punctuation
+            for i in range(len(label_str))
+        ]
+        wer_ortho = 100 * metric.compute(predictions=spaced_pred_str, references=spaced_label_str)
+        norm_pred_str, norm_label_str = [], []
+        # Iterate through all predictions and labels
+        for pred, label in zip(pred_str, label_str):
+            # Normalize the prediction and label
+            normalized_pred = normalizer(pred)
+            normalized_label = normalizer(label)
+            # If either normalized string is empty after normalization, replace with "<|nospeech|>"
+            if not normalized_pred.strip():
+                normalized_pred = "<|nospeech|>"
+            if not normalized_label.strip():
+                normalized_label = "<|nospeech|>"
+            norm_pred_str.append(normalized_pred)
+            norm_label_str.append(normalized_label)
+        # Replace original strings with "<|nocaptions|>" where necessary for consistency
+        pred_str = [pred if len(pred.strip()) > 0 else "<|nospeech|>" for pred in pred_str]
+        label_str = [label if len(label.strip()) > 0 else "<|nospeech|>" for label in label_str]
+        # Compute WER using all entries, including those with "<|nocaptions|>"
+        wer = 100 * metric.compute(predictions=norm_pred_str, references=norm_label_str)
+        return {"wer": wer, "wer_ortho": wer_ortho}, pred_str, label_str, norm_pred_str, norm_label_str
+    # 9. Save feature extractor, tokenizer, config and generation config
+    feature_extractor.save_pretrained(training_args.output_dir)
+    tokenizer.save_pretrained(training_args.output_dir)
+    config.save_pretrained(training_args.output_dir)
+    student_model.generation_config.save_pretrained(
+        training_args.output_dir
+    )  # generation config stays bound to model to make it easy to jit
+    processor = WhisperProcessor.from_pretrained(training_args.output_dir)
+    data_collator = FlaxDataCollatorSpeechSeq2SeqWithPadding(
+        processor=processor,
+        decoder_start_token_id=student_model.config.decoder_start_token_id,  # <|startoftranscript|>
+        decoder_prev_token_id=tokenizer.all_special_ids[-3],  # <|startofprev|>
+        input_padding="longest",
+        target_padding="max_length",
+        max_target_length=max_label_length,
+    )
+    # Initialize our training
+    rng = jax.random.PRNGKey(training_args.seed)
+    rng, dropout_rng = jax.random.split(rng)
+    # Store some constants
+    train_batch_size = int(training_args.per_device_train_batch_size) * jax.device_count()
+    gradient_accumulation_steps = int(training_args.gradient_accumulation_steps)
+    per_device_eval_batch_size = int(training_args.per_device_eval_batch_size)
+    eval_batch_size = per_device_eval_batch_size * jax.device_count()
+    if not data_args.streaming and training_args.max_steps < 0:
+        num_epochs = int(training_args.num_train_epochs)
+        steps_per_epoch = len(vectorized_datasets["train"]) // train_batch_size
+        total_train_steps = steps_per_epoch * num_epochs
+    elif training_args.max_steps > 0:
+        logger.info("max_steps is given, it will override any value given in num_train_epochs")
+        total_train_steps = int(training_args.max_steps)
+        # Setting a very large number of epochs so we go as many times as necessary over the iterator.
+        num_epochs = sys.maxsize
+        steps_per_epoch = total_train_steps
+    else:
+        raise ValueError("max_steps must be specified when training with a streaming (iterable) dataset")
+    if training_args.eval_steps is None:
+        logger.info(
+            f"eval_steps is not set, evaluating at the end of {'each epoch' if not data_args.streaming else 'training'}"
+        )
+        eval_steps = steps_per_epoch
+    else:
+        eval_steps = training_args.eval_steps
+    # Create learning rate schedule
+    linear_decay_lr_schedule_fn = create_learning_rate_fn(
+        total_train_steps * gradient_accumulation_steps,
+        training_args.lr_scheduler_type,
+        training_args.warmup_steps * gradient_accumulation_steps,
+        training_args.learning_rate,
+    )
+    # We use Optax's "masking" functionality to not apply weight decay
+    # to bias and LayerNorm scale parameters. decay_mask_fn returns a
+    # mask boolean with the same structure as the parameters.
+    # The mask is True for parameters that should be decayed.
+    def decay_mask_fn(params):
+        flat_params = traverse_util.flatten_dict(params)
+        # find out all LayerNorm parameters
+        layer_norm_candidates = [
+            "layer_norm",
+            "self_attn_layer_norm",
+            "final_layer_norm",
+            "encoder_attn_layer_norm",
+        ]
+        layer_norm_named_params = {
+            layer[-2:]
+            for layer_norm_name in layer_norm_candidates
+            for layer in flat_params.keys()
+            if layer_norm_name in "".join(layer).lower()
+        }
+        flat_mask = {path: path[-1] != "bias" and path[-2:] not in layer_norm_named_params for path in flat_params}
+        return traverse_util.unflatten_dict(flat_mask)
+    # create adam optimizer
+    adamw = optax.adamw(
+        learning_rate=linear_decay_lr_schedule_fn,
+        b1=training_args.adam_beta1,
+        b2=training_args.adam_beta2,
+        eps=training_args.adam_epsilon,
+        weight_decay=training_args.weight_decay,
+        mask=decay_mask_fn,
+    )
+    if gradient_accumulation_steps > 1:
+        # accumulate gradients and apply once every k steps
+        adamw = optax.MultiSteps(adamw, every_k_schedule=gradient_accumulation_steps)
+    share_hidden_states = training_args.freeze_encoder and student_model.config.d_model == teacher_model.config.d_model
+    encoder_layer_mapping = get_layers_to_supervise(
+        student_model.config.encoder_layers, teacher_model.config.encoder_layers
+    )
+    decoder_layer_mapping = get_layers_to_supervise(
+        student_model.config.decoder_layers, teacher_model.config.decoder_layers
+    )
+    # Setup train state
+    student_state = TrainState.create(
+        apply_fn=student_model.decode if share_hidden_states else student_model.__call__,
+        params=student_params,
+        tx=adamw,
+        to_dtype=to_dtype,
+        dropout_rng=dropout_rng,
+        max_grad_norm=training_args.max_grad_norm,
+    )
+    if training_args.resume_from_checkpoint is not None:
+        if os.path.isfile(os.path.join(training_args.resume_from_checkpoint, "train_state.msgpack")):
+            logger.info(
+                f"Checkpoint detected, resuming training at {training_args.resume_from_checkpoint}. To avoid "
+                "this behavior, omit the resume_from_checkpoint argument."
+            )
+            with Path(os.path.join(training_args.resume_from_checkpoint, "train_state.msgpack")).open("rb") as f:
+                student_state = from_bytes(student_state, f.read())
+        else:
+            logger.warning(
+                f"Checkpoint {training_args.resume_from_checkpoint} not detected, training from scratch. Ensure "
+                f"you pass the path to a folder with a valid checkpoint for your model."
+            )
+    def cross_entropy_loss(logits, labels):
+        vocab_size = logits.shape[-1]
+        # optax onehot always returns a float32 device array, need to downcast if performing mixed precision training
+        onehot_targets = to_dtype(onehot(labels, vocab_size))
+        loss = optax.softmax_cross_entropy(logits, onehot_targets)
+        # ignore padded tokens from loss, i.e. where labels are not set to -100
+        padding = labels >= 0
+        loss = loss * padding
+        loss = loss.sum()
+        num_labels = padding.sum()
+        return loss, num_labels
+    # temperature smoothed kl-divergence
+    def kl_divergence(target_distribution, log_predicted_distribution, labels, eps=1e-20):
+        divergence = -target_distribution * (log_predicted_distribution - jnp.log(target_distribution + eps))
+        # ignore padded tokens from divergence, i.e. where labels are not set to -100
+        padding_mask = labels >= 0
+        padding_mask = jnp.expand_dims(padding_mask, axis=-1)
+        divergence = (divergence * padding_mask).sum()
+        return to_dtype(divergence)  # respect the dtype of the backprop
+    def mean_square_error_loss(student_outputs, teacher_outputs):
+        mse = dtype(0.0)
+        # tie encoder embeddings
+        mse += jnp.mean(
+            jnp.square(teacher_outputs.encoder_hidden_states[0] - student_outputs.encoder_hidden_states[0])
+        )
+        for student_layer_id, teacher_layer_id in encoder_layer_mapping.items():
+            # offset the hidden-state layer ids by 1 to account for the extra embedding hidden-state
+            student_hidden_state = student_outputs.encoder_hidden_states[student_layer_id + 1]
+            teacher_hidden_state = teacher_outputs.encoder_hidden_states[teacher_layer_id + 1]
+            mse += jnp.mean(jnp.square(teacher_hidden_state - student_hidden_state))
+            # student_attention = student_outputs.encoder_attentions[student_layer_id]
+            # teacher_attention = teacher_outputs.encoder_attentions[teacher_layer_id]
+            # mse += jnp.mean(jnp.square(student_attention - teacher_attention))
+        # tie decoder embeddings
+        mse += jnp.mean(
+            jnp.square(teacher_outputs.decoder_hidden_states[0] - student_outputs.decoder_hidden_states[0])
+        )
+        for student_layer_id, teacher_layer_id in decoder_layer_mapping.items():
+            # offset the hidden-state layer ids by 1 to account for the extra embedding hidden-state
+            student_hidden_state = student_outputs.decoder_hidden_states[student_layer_id + 1]
+            teacher_hidden_state = teacher_outputs.decoder_hidden_states[teacher_layer_id + 1]
+            mse += jnp.mean(jnp.square(teacher_hidden_state - student_hidden_state))
+            # student_attention = student_outputs.decoder_attentions[student_layer_id]
+            # teacher_attention = teacher_outputs.decoder_attentions[teacher_layer_id]
+            # mse += jnp.mean(jnp.square(student_attention - teacher_attention))
+            # student_cross_attention = student_outputs.cross_attentions[student_layer_id]
+            # teacher_cross_attention = teacher_outputs.cross_attentions[teacher_layer_id]
+            # mse += jnp.mean(jnp.square(student_cross_attention - teacher_cross_attention))
+        return to_dtype(mse)  # respect the dtype of the backprop
+    # Define gradient update step fn
+    def train_step(
+        student_state,
+        teacher_params,
+        batch,
+        freeze_encoder,
+        share_hidden_states,
+        temperature=2.0,
+    ):
+        dropout_rng, new_dropout_rng = jax.random.split(student_state.dropout_rng)
+        def compute_loss(student_params):
+            labels = batch.pop("labels")
+            output_hidden_states = not share_hidden_states and training_args.mse_weight > 0.0
+            teacher_outputs = teacher_model(
+                **batch,
+                params=teacher_params,
+                freeze_encoder=True,
+                output_hidden_states=output_hidden_states,
+                train=False,
+            )
+            if share_hidden_states:
+                # if the student and teacher share the same frozen encoder then we don't have to recompute the
+                # encoder hidden-states for the student model, we can just re-use from the teacher
+                encoder_hidden_states = jax.lax.stop_gradient(teacher_outputs.encoder_last_hidden_state)
+                encoder_outputs = FlaxBaseModelOutput(last_hidden_state=encoder_hidden_states)
+                student_outputs = student_state.apply_fn(
+                    decoder_input_ids=batch["decoder_input_ids"],
+                    encoder_outputs=encoder_outputs,
+                    params=student_params,
+                    dropout_rng=dropout_rng,
+                    train=True,
+                )
+            else:
+                # do the full forward pass for the student model (encoder + decoder)
+                student_outputs = student_state.apply_fn(
+                    **batch,
+                    params=student_params,
+                    dropout_rng=dropout_rng,
+                    freeze_encoder=freeze_encoder,
+                    output_hidden_states=output_hidden_states,
+                    train=True,
+                )
+            # CE (data) loss
+            ce_loss, num_labels = cross_entropy_loss(student_outputs.logits, labels)
+            # rescale by temperature to ensure gradients scale correctly
+            teacher_distribution = jax.nn.softmax(teacher_outputs.logits / temperature, axis=-1)
+            # ensure no information flow backwards through teacher
+            teacher_distribution = jax.lax.stop_gradient(teacher_distribution)
+            # log softmax of student predictions for numerical stability
+            student_distribution = jax.nn.log_softmax(student_outputs.logits / temperature, axis=-1)
+            # KL-divergence loss (scaled by temperature)
+            kl_loss = kl_divergence(teacher_distribution, student_distribution, labels) * temperature**2
+            # MSE loss between enc-dec hidden-states and attentions
+            mse_loss = (
+                mean_square_error_loss(student_outputs, teacher_outputs)
+                if output_hidden_states
+                else jnp.zeros_like(kl_loss)
+            )
+            # use DistilBart formulation - only tune the MSE weight and take remaining HPs from DistilBERT
+            ce_weight = 0.8 if training_args.kl_weight > 0 else 1.0
+            loss = ce_weight * ce_loss + training_args.kl_weight * kl_loss + training_args.mse_weight * mse_loss
+            return loss, (
+                ce_loss,
+                kl_loss,
+                mse_loss,
+                num_labels,
+            )
+        grad_fn = jax.value_and_grad(compute_loss, has_aux=True)
+        (loss, (ce_loss, kl_loss, mse_loss, num_labels)), grad = grad_fn(to_dtype(student_state.params))
+        # true loss = total loss / total samples
+        loss = jax.lax.psum(loss, "batch")
+        num_labels = jax.lax.psum(num_labels, "batch")
+        loss = jax.tree_util.tree_map(lambda x: x / num_labels, loss)
+        # true grad = total grad / total samples
+        grad = jax.lax.psum(grad, "batch")
+        grad = jax.tree_util.tree_map(lambda x: x / num_labels, grad)
+        new_state = student_state.apply_gradients(grads=grad, dropout_rng=new_dropout_rng, to_dtype=to_dtype)
+        # CE/KL/MSE losses for logging
+        ce_loss = jax.lax.psum(ce_loss, "batch")
+        ce_loss = jax.tree_util.tree_map(lambda x: x / num_labels, ce_loss)
+        kl_loss = jax.lax.psum(kl_loss, "batch")
+        kl_loss = jax.tree_util.tree_map(lambda x: x / num_labels, kl_loss)
+        mse_loss = jax.lax.psum(mse_loss, "batch")
+        mse_loss = jax.tree_util.tree_map(lambda x: x / num_labels, mse_loss)
+        metrics = {
+            "loss": loss,
+            "learning_rate": linear_decay_lr_schedule_fn(student_state.step),
+            "ce_loss": ce_loss,
+            "kl_loss": kl_loss,
+            "mse_loss": mse_loss,
+        }
+        return new_state, metrics
+    # Define eval fn
+    def eval_step(student_params, teacher_params, batch):
+        labels = batch.pop("labels")
+        output_hidden_states = not share_hidden_states and training_args.mse_weight > 0
+        student_outputs = student_model(
+            **batch,
+            params=student_params,
+            output_hidden_states=output_hidden_states,
+            train=False,
+        )
+        student_distribution = jax.nn.log_softmax(student_outputs.logits, axis=-1)
+        ce_loss, num_labels = cross_entropy_loss(student_outputs.logits, labels)
+        teacher_outputs = teacher_model(
+            **batch,
+            params=teacher_params,
+            output_hidden_states=output_hidden_states,
+            train=False,
+        )
+        teacher_distribution = jax.nn.softmax(teacher_outputs.logits, axis=-1)
+        # temperature is always 1 for eval
+        kl_loss = kl_divergence(teacher_distribution, student_distribution, labels)
+        mse_loss = (
+            mean_square_error_loss(student_outputs, teacher_outputs)
+            if output_hidden_states
+            else jnp.zeros_like(kl_loss)
+        )
+        ce_weight = 0.8 if training_args.kl_weight > 0 else 1.0
+        loss = ce_weight * ce_loss + training_args.kl_weight * kl_loss + training_args.mse_weight * mse_loss
+        # true loss = total loss / total samples
+        loss = jax.lax.psum(loss, "batch")
+        num_labels = jax.lax.psum(num_labels, "batch")
+        loss = jax.tree_util.tree_map(lambda x: x / num_labels, loss)
+        # CE/KL/MSE losses for logging
+        ce_loss = jax.lax.psum(ce_loss, "batch")
+        ce_loss = jax.tree_util.tree_map(lambda x: x / num_labels, ce_loss)
+        kl_loss = jax.lax.psum(kl_loss, "batch")
+        kl_loss = jax.tree_util.tree_map(lambda x: x / num_labels, kl_loss)
+        mse_loss = jax.lax.psum(mse_loss, "batch")
+        mse_loss = jax.tree_util.tree_map(lambda x: x / num_labels, mse_loss)
+        metrics = {"loss": loss, "ce_loss": ce_loss, "kl_loss": kl_loss, "mse_loss": mse_loss}
+        return metrics
+    # Define generation function
+    num_beams = (
+        training_args.generation_num_beams
+        if training_args.generation_num_beams is not None
+        else student_model.config.num_beams
+    )
+    # forcing the language and task tokens helps the model in its generations
+    gen_kwargs = {
+        "max_length": max_label_length,
+        "num_beams": num_beams,
+        "language": "<|en|>",
+        "task": "transcribe",
+        "return_timestamps": return_timestamps,
+    }
+    def generate_step(student_params, batch):
+        output_ids = student_model.generate(
+            batch[model_input_name],
+            attention_mask=batch.get("attention_mask"),
+            params=student_params,
+            **gen_kwargs,
+        )
+        return output_ids.sequences
+    # Replicate the train state on each device
+    student_state = student_state.replicate()
+    # Replicate the teacher params on each device
+    teacher_params = jax_utils.replicate(teacher_params)
+    # Create parallel version of the train and eval step
+    p_train_step = jax.pmap(
+        train_step,
+        "batch",
+        in_axes=(0, 0, 0, None, None, None),
+        donate_argnums=(0,),
+        static_broadcasted_argnums=(
+            3,
+            4,
+        ),
+    )
+    p_eval_step = jax.pmap(eval_step, "batch")
+    p_generate_step = jax.pmap(generate_step, "batch")
+    logger.info("***** Running training *****")
+    logger.info(f"  Num examples = {total_train_steps * train_batch_size * gradient_accumulation_steps}")
+    logger.info("  Instantaneous batch size per device =" f" {training_args.per_device_train_batch_size}")
+    logger.info("  Gradient accumulation steps =" f" {gradient_accumulation_steps}")
+    logger.info(
+        f"  Total train batch size (w. parallel & distributed) = {train_batch_size * gradient_accumulation_steps}"
+    )
+    logger.info(f"  Total optimization steps = {total_train_steps}")
+    # ======================== Training ================================
+    train_time = 0
+    train_start = time.time()
+    train_metrics = []
+    batches_to_skip = jax.device_get(unreplicate(student_state.step))
+    cur_step = int(batches_to_skip)  # will be zero if starting from scratch
+    epochs_trained = batches_to_skip // steps_per_epoch
+    steps_trained_progress_bar = tqdm(range(total_train_steps), desc="Train steps ... ", position=0)
+    steps_trained_progress_bar.update(batches_to_skip)
+    continue_training = True
+    minibatch_steps = 0
+    if batches_to_skip > 0:
+        logger.info("  Continuing training from checkpoint, will skip to saved global_step")
+        logger.info(f"  Continuing training from epoch {epochs_trained}")
+        logger.info(f"  Continuing training from global step {batches_to_skip}")
+    # Generate a training data loader by shuffling sampling indices from the train dataset
+    train_loader = get_data_loader(
+        training_args.seed,
+        vectorized_datasets["train"],
+        batch_size=train_batch_size,
+        data_collator=data_collator,
+        dataloader_num_workers=dataloader_num_workers,
+        skip_batches=batches_to_skip,
+        prefetch_size=dataloader_prefetch_size,
+    )
+    for epoch in range(epochs_trained, num_epochs):
+        if hasattr(train_loader, "dataset") and isinstance(train_loader.dataset, IterableDataset):
+            train_loader.dataset.set_epoch(epoch)
+        for batch in train_loader:
+            minibatch_steps += 1
+            update_step = minibatch_steps == gradient_accumulation_steps
+            if update_step:
+                steps_trained_progress_bar.update(1)
+                cur_step += 1
+                minibatch_steps = 0
+            batch = shard(batch.data)
+            student_state, train_metric = p_train_step(
+                student_state,
+                teacher_params,
+                batch,
+                training_args.freeze_encoder,
+                share_hidden_states,
+                training_args.temperature,
+            )
+            if cur_step % training_args.logging_steps == 0 and update_step:
+                train_metrics.append(train_metric)
+                train_metric_to_write = unreplicate(train_metric)
+                steps_trained_progress_bar.write(
+                    f"Step... ({cur_step} / {total_train_steps} | Loss:"
+                    f" {train_metric_to_write['loss']}, Learning Rate:"
+                    f" {train_metric_to_write['learning_rate']})"
+                )
+                if has_wandb and jax.process_index() == 0:
+                    write_wandb_metric(
+                        wandb_logger,
+                        train_metric_to_write,
+                        train_time + time.time() - train_start,
+                        cur_step,
+                        epoch,
+                        prefix="train",
+                    )
+            # save checkpoint and weights after each save_steps and at the end of training
+            if (cur_step % training_args.save_steps == 0 and update_step) or cur_step == total_train_steps:
+                if jax.process_index() == 0:
+                    save_hf_weights(
+                        student_state,
+                        student_model,
+                        processor,
+                        training_args.output_dir,
+                        cur_step,
+                        total_train_steps,
+                        use_scan=training_args.use_scan,
+                    )
+                    if training_args.save_train_state:
+                        student_state.save_state(
+                            training_args.output_dir, save_total_limit=training_args.save_total_limit
+                        )
+                    if training_args.push_to_hub:
+                        repo.push_to_hub(
+                            commit_message=f"Saving train state of step {cur_step}",
+                            blocking=False,
+                        )
+            if training_args.do_eval and (
+                (cur_step % eval_steps == 0 and update_step) or cur_step == total_train_steps
+            ):
+                train_time += time.time() - train_start
+                # ======================== Evaluating ==============================
+                for eval_split in all_eval_splits:
+                    eval_metrics = []
+                    eval_preds = []
+                    eval_labels = []
+                    eval_start = time.time()
+                    eval_loader = get_data_loader(
+                        training_args.seed,
+                        vectorized_datasets[eval_split],
+                        batch_size=eval_batch_size,
+                        data_collator=data_collator,
+                        shuffle=False,
+                        drop_last=False,
+                        dataloader_num_workers=dataloader_num_workers,
+                    )
+                    for batch in tqdm(eval_loader, desc=f"Evaluating {eval_split}...", position=2):
+                        # Model forward
+                        labels = batch["labels"]
+                        metrics = pad_shard_unpad(
+                            p_eval_step,
+                            static_argnums=(
+                                0,
+                                1,
+                            ),
+                            static_return=True,
+                        )(
+                            student_state.params,
+                            teacher_params,
+                            batch.data,
+                            min_device_batch=per_device_eval_batch_size,
+                        )
+                        eval_metrics.append(metrics)
+                        # generation
+                        if training_args.predict_with_generate:
+                            generated_ids = pad_shard_unpad(p_generate_step)(
+                                student_state.params, batch.data, min_device_batch=per_device_eval_batch_size
+                            )
+                            eval_preds.extend(jax.device_get(generated_ids.reshape(-1, gen_kwargs["max_length"])))
+                            eval_labels.extend(labels)
+                    eval_time = time.time() - eval_start
+                    # normalize eval metrics
+                    eval_metrics = get_metrics(eval_metrics)
+                    eval_metrics = jax.tree_util.tree_map(jnp.mean, eval_metrics)
+                    # compute WER metric
+                    wer_desc = ""
+                    if training_args.predict_with_generate:
+                        wer_metric, pred_str, label_str, norm_pred_str, norm_label_str = compute_metrics(
+                            eval_preds, eval_labels
+                        )
+                        eval_metrics.update(wer_metric)
+                        wer_desc = " ".join([f"Eval {key}: {value} |" for key, value in wer_metric.items()])
+                    # Print metrics and update progress bar
+                    steps_trained_progress_bar.write(
+                        f"Eval results for step ({cur_step} / {total_train_steps} | Eval Loss: {eval_metrics['loss']} |"
+                        f" {wer_desc})"
+                    )
+                    if has_tensorboard and jax.process_index() == 0:
+                        write_eval_metric(
+                            summary_writer,
+                            eval_metrics,
+                            cur_step,
+                            prefix=eval_split,
+                        )
+                    if has_wandb and jax.process_index() == 0:
+                        write_wandb_metric(wandb_logger, eval_metrics, eval_time, cur_step, epoch, prefix=eval_split)
+                        if training_args.predict_with_generate:
+                            write_wandb_pred(
+                                wandb_logger,
+                                pred_str,
+                                label_str,
+                                norm_pred_str,
+                                norm_label_str,
+                                cur_step,
+                                prefix=eval_split,
+                            )
+                if has_tensorboard and jax.process_index() == 0:
+                    # we'll only log to tensorboard every eval steps
+                    write_train_metric(
+                        summary_writer,
+                        train_metrics,
+                        train_time,
+                        cur_step,
+                        training_args.logging_steps,
+                    )
+                # flush the train metrics
+                train_start = time.time()
+                train_metrics = []
+            # break condition
+            if cur_step == total_train_steps:
+                continue_training = False
+                break
+        if not continue_training:
+            break
+if __name__ == "__main__":
+    main()

run_experiment2.sh ADDED Viewed

	@@ -0,0 +1,41 @@

+#!/usr/bin/env bash
+TOKENIZERS_PARALLELISM=false python3 run_distillation.py \
+  --model_name_or_path "./nb-distil-large-init" \
+  --teacher_model_name_or_path "NbAiLab/nb-whisper-large" \
+  --train_dataset_name "NbAiLab/annotated_ncc_speech_styling_v2_vad3_distil_postLv2" \
+  --train_dataset_config_name "" \
+  --train_split_name "train" \
+  --eval_dataset_name "NbAiLab/ncc_speech_v7" \
+  --eval_dataset_config_name "" \
+  --eval_split_name "validation_norwegian_fleurs" \
+  --eval_steps 500 \
+  --save_steps 5000 \
+  --warmup_steps 1000 \
+  --learning_rate 0.0003 \
+  --lr_scheduler_type "constant_with_warmup" \
+  --logging_steps 500 \
+  --save_total_limit 1 \
+  --max_steps 200000 \
+  --wer_threshold 10 \
+  --per_device_train_batch_size 4\
+  --per_device_eval_batch_size 4 \
+  --dataloader_num_workers 8 \
+  --dtype "bfloat16" \
+  --output_dir "./nb-distil-whisper-larg7-flax6" \
+  --do_train \
+  --do_eval \
+  --use_scan \
+  --gradient_checkpointing \
+  --overwrite_output_dir \
+  --predict_with_generate \
+  --freeze_encoder \
+  --streaming \
+  --use_auth_token \
+  --report_to "wandb" \
+  --wandb_project "nb-distil-whisper-large-fleurseval" \
+  --wandb_name "flax_experiment1_bs4_v5_1e4_wer10" \
+  --save_code_to_wandb \
+  --save_train_state \
+  --hub_model_id "NbAiLab/nb-distil-whisper-large-flax6"7\
+  --push_to_hub