distil-whisper-large-v3-de / run_labelling.sh
sanchit-gandhi's picture
Saving train state of step 5000
f969b47 verified
raw
history blame contribute delete
No virus
849 Bytes
#!/usr/bin/env bash
accelerate launch run_pseudo_labelling.py \
--model_name_or_path "openai/whisper-large-v3" \
--dataset_name "mozilla-foundation/common_voice_16_1" \
--dataset_config_name "de" \
--dataset_split_name "train+validation+test" \
--text_column_name "sentence" \
--id_column_name "path" \
--output_dir "../common_voice_16_1_de_pseudo_labelled" \
--wandb_project "distil-whisper-labelling" \
--per_device_eval_batch_size 64 \
--dtype "bfloat16" \
--attn_implementation "sdpa" \
--logging_steps 500 \
--max_label_length 256 \
--concatenate_audio \
--preprocessing_batch_size 256 \
--preprocessing_num_workers 16 \
--dataloader_num_workers 8 \
--report_to "wandb" \
--language "de" \
--task "transcribe" \
--return_timestamps \
--streaming False \
--generation_num_beams 1 \
--push_to_hub