Add model

Browse files

Files changed (6) hide show

README.md +93 -0
config.json +106 -0
fairseq/model.pt +3 -0
preprocessor_config.json +9 -0
pytorch_model.bin +3 -0
rinna.png +0 -0

README.md ADDED Viewed

	@@ -0,0 +1,93 @@

+---
+thumbnail: https://github.com/rinnakk/japanese-pretrained-models/blob/master/rinna.png
+language: ja
+license: apache-2.0
+datasets: reazon-research/reazonspeech
+pipeline_tag: feature-extraction
+inference: false
+tags:
+  - wav2vec2
+  - speech
+---
+# `rinna/japanese-wav2vec2-base`
+![rinna-icon](./rinna.png)
+# Overview
+This is a Japanese wav2vec 2.0 Base model trained by [rinna Co., Ltd.](https://rinna.co.jp/)
+* **Model summary**
+  The model architecture is the same as the [original wav2vec 2.0 Base model](https://huggingface.co/facebook/wav2vec2-base), which contains 12 transformer layers with 12 attention heads.
+  The model was trained using code from the [official repository](https://github.com/facebookresearch/fairseq/tree/main/examples/wav2vec), and the detailed training configuration can be found in the same repository and the [original paper](https://proceedings.neurips.cc/paper/2020/hash/92d1e1eb1cd6f9fba3227870bb6d7f07-Abstract.html).
+* **Training**
+  The model was trained on approximately 19,000 hours of following Japanese speech corpus ReazonSpeech v1.
+  - [ReazonSpeech](https://huggingface.co/datasets/reazon-research/reazonspeech)
+* **Contributors**
+  - [Yukiya Hono](https://huggingface.co/yky-h)
+  - [Kentaro Mitsui](https://huggingface.co/Kentaro321)
+  - [Kei Sawada](https://huggingface.co/keisawada)
+---
+# How to use the model
+```python
+import soundfile as sf
+from transformers import AutoFeatureExtractor, AutoModel
+model_name = "rinna/japanese-wav2vec2-base"
+feature_extractor = AutoFeatureExtractor.from_pretrained(model_name)
+model = AutoModel.from_pretrained(model_name)
+model.eval()
+raw_speech_16kHz, sr = sf.read(audio_file)
+inputs = feature_extractor(
+    raw_speech_16kHz,
+    return_tensors="pt",
+    sampling_rate=sr,
+)
+outputs = model(**inputs)
+print(f"Input:  {inputs.input_values.size()}")  # [1, #samples]
+print(f"Output: {outputs.last_hidden_state.size()}")  # [1, #frames, 768]
+```
+A fairseq checkpoint file can also be available [here](https://huggingface.co/rinna/japanese-wav2vec2-base/tree/main/fairseq).
+---
+# How to cite
+```bibtex
+@misc{rinna-japanese-wav2vec2-base,
+  title={rinna/japanese-wav2vec2-base},
+  author={Hono, Yukiya and Mitsui, Kentaro and Sawada, Kei},
+  url={https://huggingface.co/rinna/japanese-wav2vec2-base}
+}
+```
+---
+# Citations
+```bibtex
+@inproceedings{baevski2020wav2vec,
+  title={wav2vec 2.0: A Framework for Self-Supervised Learning of Speech Representations},
+  author={Baevski, Alexei and Zhou, Yuhao and Mohamed, Abdelrahman and Auli, Michael},
+  booktitle={Advances in Neural Information Processing Systems},
+  volume={33},
+  pages={12449--12460},
+  year={2020},
+  url={https://proceedings.neurips.cc/paper/2020/hash/92d1e1eb1cd6f9fba3227870bb6d7f07-Abstract.html}
+}
+```
+---
+# License
+[The Apache 2.0 license](https://www.apache.org/licenses/LICENSE-2.0)

config.json ADDED Viewed

	@@ -0,0 +1,106 @@

+{
+  "_name_or_path": "rinna/japanese-wav2vec2-base",
+  "activation_dropout": 0.1,
+  "adapter_kernel_size": 3,
+  "adapter_stride": 2,
+  "add_adapter": false,
+  "apply_spec_augment": true,
+  "architectures": [
+    "Wav2Vec2ForPreTraining"
+  ],
+  "attention_dropout": 0.1,
+  "bos_token_id": 1,
+  "classifier_proj_size": 256,
+  "codevector_dim": 256,
+  "contrastive_logits_temperature": 0.1,
+  "conv_bias": false,
+  "conv_dim": [
+    512,
+    512,
+    512,
+    512,
+    512,
+    512,
+    512
+  ],
+  "conv_kernel": [
+    10,
+    3,
+    3,
+    3,
+    3,
+    2,
+    2
+  ],
+  "conv_stride": [
+    5,
+    2,
+    2,
+    2,
+    2,
+    2,
+    2
+  ],
+  "ctc_loss_reduction": "sum",
+  "ctc_zero_infinity": false,
+  "diversity_loss_weight": 0.1,
+  "do_stable_layer_norm": false,
+  "eos_token_id": 2,
+  "feat_extract_activation": "gelu",
+  "feat_extract_norm": "group",
+  "feat_proj_dropout": 0.0,
+  "feat_quantizer_dropout": 0.0,
+  "final_dropout": 0.1,
+  "hidden_act": "gelu",
+  "hidden_dropout": 0.1,
+  "hidden_size": 768,
+  "initializer_range": 0.02,
+  "intermediate_size": 3072,
+  "layer_norm_eps": 1e-05,
+  "layerdrop": 0.1,
+  "mask_feature_length": 10,
+  "mask_feature_min_masks": 0,
+  "mask_feature_prob": 0.0,
+  "mask_time_length": 10,
+  "mask_time_min_masks": 2,
+  "mask_time_prob": 0.05,
+  "model_type": "wav2vec2",
+  "num_adapter_layers": 3,
+  "num_attention_heads": 12,
+  "num_codevector_groups": 2,
+  "num_codevectors_per_group": 320,
+  "num_conv_pos_embedding_groups": 16,
+  "num_conv_pos_embeddings": 128,
+  "num_feat_extract_layers": 7,
+  "num_hidden_layers": 12,
+  "num_negatives": 100,
+  "output_hidden_size": 768,
+  "pad_token_id": 0,
+  "proj_codevector_dim": 256,
+  "tdnn_dilation": [
+    1,
+    2,
+    3,
+    1,
+    1
+  ],
+  "tdnn_dim": [
+    512,
+    512,
+    512,
+    512,
+    1500
+  ],
+  "tdnn_kernel": [
+    5,
+    3,
+    3,
+    1,
+    1
+  ],
+  "torch_dtype": "float32",
+  "transformers_version": "4.28.1",
+  "use_weighted_layer_sum": false,
+  "vocab_size": 32,
+  "xvector_output_dim": 512
+}

fairseq/model.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:acea47a3380a25d90ead7a4706849b08779937d1113fd517d92c34255d308f54
+size 380266381

preprocessor_config.json ADDED Viewed

	@@ -0,0 +1,9 @@

+{
+  "do_normalize": false,
+  "feature_extractor_type": "Wav2Vec2FeatureExtractor",
+  "feature_size": 1,
+  "padding_side": "right",
+  "padding_value": 0.0,
+  "return_attention_mask": false,
+  "sampling_rate": 16000
+}

pytorch_model.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:5f256c3ad63fafda4ccd06d9ce1830096304547d23fa0f4833592907293018c1
+size 380250485

rinna.png ADDED Viewed