Duplicate from bigscience/mt0-xl

Browse files

Co-authored-by: Thomas Wang <TimeRobber@users.noreply.huggingface.co>

Files changed (10) hide show

.gitattributes +33 -0
README.md +932 -0
config.json +32 -0
pytorch_model-00001-of-00002.bin +3 -0
pytorch_model-00002-of-00002.bin +3 -0
pytorch_model.bin.index.json +566 -0
special_tokens_map.json +5 -0
spiece.model +3 -0
tokenizer.json +3 -0
tokenizer_config.json +11 -0

.gitattributes ADDED Viewed

	@@ -0,0 +1,33 @@

+*.7z filter=lfs diff=lfs merge=lfs -text
+*.arrow filter=lfs diff=lfs merge=lfs -text
+*.bin filter=lfs diff=lfs merge=lfs -text
+*.bz2 filter=lfs diff=lfs merge=lfs -text
+*.ftz filter=lfs diff=lfs merge=lfs -text
+*.gz filter=lfs diff=lfs merge=lfs -text
+*.h5 filter=lfs diff=lfs merge=lfs -text
+*.joblib filter=lfs diff=lfs merge=lfs -text
+*.lfs.* filter=lfs diff=lfs merge=lfs -text
+*.mlmodel filter=lfs diff=lfs merge=lfs -text
+*.model filter=lfs diff=lfs merge=lfs -text
+*.msgpack filter=lfs diff=lfs merge=lfs -text
+*.npy filter=lfs diff=lfs merge=lfs -text
+*.npz filter=lfs diff=lfs merge=lfs -text
+*.onnx filter=lfs diff=lfs merge=lfs -text
+*.ot filter=lfs diff=lfs merge=lfs -text
+*.parquet filter=lfs diff=lfs merge=lfs -text
+*.pb filter=lfs diff=lfs merge=lfs -text
+*.pickle filter=lfs diff=lfs merge=lfs -text
+*.pkl filter=lfs diff=lfs merge=lfs -text
+*.pt filter=lfs diff=lfs merge=lfs -text
+*.pth filter=lfs diff=lfs merge=lfs -text
+*.rar filter=lfs diff=lfs merge=lfs -text
+*.safetensors filter=lfs diff=lfs merge=lfs -text
+saved_model/**/* filter=lfs diff=lfs merge=lfs -text
+*.tar.* filter=lfs diff=lfs merge=lfs -text
+*.tflite filter=lfs diff=lfs merge=lfs -text
+*.tgz filter=lfs diff=lfs merge=lfs -text
+*.wasm filter=lfs diff=lfs merge=lfs -text
+*.xz filter=lfs diff=lfs merge=lfs -text
+*.zip filter=lfs diff=lfs merge=lfs -text
+*.zst filter=lfs diff=lfs merge=lfs -text
+*tfevents* filter=lfs diff=lfs merge=lfs -text

README.md ADDED Viewed

	@@ -0,0 +1,932 @@

+---
+datasets:
+- bigscience/xP3
+- mc4
+license: apache-2.0
+language:
+- af
+- am
+- ar
+- az
+- be
+- bg
+- bn
+- ca
+- ceb
+- co
+- cs
+- cy
+- da
+- de
+- el
+- en
+- eo
+- es
+- et
+- eu
+- fa
+- fi
+- fil
+- fr
+- fy
+- ga
+- gd
+- gl
+- gu
+- ha
+- haw
+- hi
+- hmn
+- ht
+- hu
+- hy
+- ig
+- is
+- it
+- iw
+- ja
+- jv
+- ka
+- kk
+- km
+- kn
+- ko
+- ku
+- ky
+- la
+- lb
+- lo
+- lt
+- lv
+- mg
+- mi
+- mk
+- ml
+- mn
+- mr
+- ms
+- mt
+- my
+- ne
+- nl
+- 'no'
+- ny
+- pa
+- pl
+- ps
+- pt
+- ro
+- ru
+- sd
+- si
+- sk
+- sl
+- sm
+- sn
+- so
+- sq
+- sr
+- st
+- su
+- sv
+- sw
+- ta
+- te
+- tg
+- th
+- tr
+- uk
+- und
+- ur
+- uz
+- vi
+- xh
+- yi
+- yo
+- zh
+- zu
+pipeline_tag: text2text-generation
+widget:
+- text: >-
+    一个传奇的开端，一个不灭的神话，这不仅仅是一部电影，而是作为一个走进新时代的标签，永远彪炳史册。Would you rate the previous
+    review as positive, neutral or negative?
+  example_title: zh-en sentiment
+- text: 一个传奇的开端，一个不灭的神话，这不仅仅是一部电影，而是作为一个走进新时代的标签，永远彪炳史册。你认为这句话的立场是赞扬、中立还是批评？
+  example_title: zh-zh sentiment
+- text: Suggest at least five related search terms to "Mạng neural nhân tạo".
+  example_title: vi-en query
+- text: >-
+    Proposez au moins cinq mots clés concernant «Réseau de neurones
+    artificiels».
+  example_title: fr-fr query
+- text: Explain in a sentence in Telugu what is backpropagation in neural networks.
+  example_title: te-en qa
+- text: Why is the sky blue?
+  example_title: en-en qa
+- text: >-
+    Write a fairy tale about a troll saving a princess from a dangerous dragon.
+    The fairy tale is a masterpiece that has achieved praise worldwide and its
+    moral is "Heroes Come in All Shapes and Sizes". Story (in Spanish):
+  example_title: es-en fable
+- text: >-
+    Write a fable about wood elves living in a forest that is suddenly invaded
+    by ogres. The fable is a masterpiece that has achieved praise worldwide and
+    its moral is "Violence is the last refuge of the incompetent". Fable (in
+    Hindi):
+  example_title: hi-en fable
+model-index:
+- name: mt0-xl
+  results:
+  - task:
+      type: Coreference resolution
+    dataset:
+      type: winogrande
+      name: Winogrande XL (xl)
+      config: xl
+      split: validation
+      revision: a80f460359d1e9a67c006011c94de42a8759430c
+    metrics:
+    - type: Accuracy
+      value: 52.49
+  - task:
+      type: Coreference resolution
+    dataset:
+      type: Muennighoff/xwinograd
+      name: XWinograd (en)
+      config: en
+      split: test
+      revision: 9dd5ea5505fad86b7bedad667955577815300cee
+    metrics:
+    - type: Accuracy
+      value: 61.89
+  - task:
+      type: Coreference resolution
+    dataset:
+      type: Muennighoff/xwinograd
+      name: XWinograd (fr)
+      config: fr
+      split: test
+      revision: 9dd5ea5505fad86b7bedad667955577815300cee
+    metrics:
+    - type: Accuracy
+      value: 59.04
+  - task:
+      type: Coreference resolution
+    dataset:
+      type: Muennighoff/xwinograd
+      name: XWinograd (jp)
+      config: jp
+      split: test
+      revision: 9dd5ea5505fad86b7bedad667955577815300cee
+    metrics:
+    - type: Accuracy
+      value: 60.27
+  - task:
+      type: Coreference resolution
+    dataset:
+      type: Muennighoff/xwinograd
+      name: XWinograd (pt)
+      config: pt
+      split: test
+      revision: 9dd5ea5505fad86b7bedad667955577815300cee
+    metrics:
+    - type: Accuracy
+      value: 66.16
+  - task:
+      type: Coreference resolution
+    dataset:
+      type: Muennighoff/xwinograd
+      name: XWinograd (ru)
+      config: ru
+      split: test
+      revision: 9dd5ea5505fad86b7bedad667955577815300cee
+    metrics:
+    - type: Accuracy
+      value: 59.05
+  - task:
+      type: Coreference resolution
+    dataset:
+      type: Muennighoff/xwinograd
+      name: XWinograd (zh)
+      config: zh
+      split: test
+      revision: 9dd5ea5505fad86b7bedad667955577815300cee
+    metrics:
+    - type: Accuracy
+      value: 62.9
+  - task:
+      type: Natural language inference
+    dataset:
+      type: anli
+      name: ANLI (r1)
+      config: r1
+      split: validation
+      revision: 9dbd830a06fea8b1c49d6e5ef2004a08d9f45094
+    metrics:
+    - type: Accuracy
+      value: 38.2
+  - task:
+      type: Natural language inference
+    dataset:
+      type: anli
+      name: ANLI (r2)
+      config: r2
+      split: validation
+      revision: 9dbd830a06fea8b1c49d6e5ef2004a08d9f45094
+    metrics:
+    - type: Accuracy
+      value: 34.8
+  - task:
+      type: Natural language inference
+    dataset:
+      type: anli
+      name: ANLI (r3)
+      config: r3
+      split: validation
+      revision: 9dbd830a06fea8b1c49d6e5ef2004a08d9f45094
+    metrics:
+    - type: Accuracy
+      value: 39
+  - task:
+      type: Natural language inference
+    dataset:
+      type: super_glue
+      name: SuperGLUE (cb)
+      config: cb
+      split: validation
+      revision: 9e12063561e7e6c79099feb6d5a493142584e9e2
+    metrics:
+    - type: Accuracy
+      value: 85.71
+  - task:
+      type: Natural language inference
+    dataset:
+      type: super_glue
+      name: SuperGLUE (rte)
+      config: rte
+      split: validation
+      revision: 9e12063561e7e6c79099feb6d5a493142584e9e2
+    metrics:
+    - type: Accuracy
+      value: 78.7
+  - task:
+      type: Natural language inference
+    dataset:
+      type: xnli
+      name: XNLI (ar)
+      config: ar
+      split: validation
+      revision: a5a45e4ff92d5d3f34de70aaf4b72c3bdf9f7f16
+    metrics:
+    - type: Accuracy
+      value: 51.85
+  - task:
+      type: Natural language inference
+    dataset:
+      type: xnli
+      name: XNLI (bg)
+      config: bg
+      split: validation
+      revision: a5a45e4ff92d5d3f34de70aaf4b72c3bdf9f7f16
+    metrics:
+    - type: Accuracy
+      value: 54.18
+  - task:
+      type: Natural language inference
+    dataset:
+      type: xnli
+      name: XNLI (de)
+      config: de
+      split: validation
+      revision: a5a45e4ff92d5d3f34de70aaf4b72c3bdf9f7f16
+    metrics:
+    - type: Accuracy
+      value: 54.78
+  - task:
+      type: Natural language inference
+    dataset:
+      type: xnli
+      name: XNLI (el)
+      config: el
+      split: validation
+      revision: a5a45e4ff92d5d3f34de70aaf4b72c3bdf9f7f16
+    metrics:
+    - type: Accuracy
+      value: 53.78
+  - task:
+      type: Natural language inference
+    dataset:
+      type: xnli
+      name: XNLI (en)
+      config: en
+      split: validation
+      revision: a5a45e4ff92d5d3f34de70aaf4b72c3bdf9f7f16
+    metrics:
+    - type: Accuracy
+      value: 56.83
+  - task:
+      type: Natural language inference
+    dataset:
+      type: xnli
+      name: XNLI (es)
+      config: es
+      split: validation
+      revision: a5a45e4ff92d5d3f34de70aaf4b72c3bdf9f7f16
+    metrics:
+    - type: Accuracy
+      value: 54.78
+  - task:
+      type: Natural language inference
+    dataset:
+      type: xnli
+      name: XNLI (fr)
+      config: fr
+      split: validation
+      revision: a5a45e4ff92d5d3f34de70aaf4b72c3bdf9f7f16
+    metrics:
+    - type: Accuracy
+      value: 54.22
+  - task:
+      type: Natural language inference
+    dataset:
+      type: xnli
+      name: XNLI (hi)
+      config: hi
+      split: validation
+      revision: a5a45e4ff92d5d3f34de70aaf4b72c3bdf9f7f16
+    metrics:
+    - type: Accuracy
+      value: 50.24
+  - task:
+      type: Natural language inference
+    dataset:
+      type: xnli
+      name: XNLI (ru)
+      config: ru
+      split: validation
+      revision: a5a45e4ff92d5d3f34de70aaf4b72c3bdf9f7f16
+    metrics:
+    - type: Accuracy
+      value: 53.09
+  - task:
+      type: Natural language inference
+    dataset:
+      type: xnli
+      name: XNLI (sw)
+      config: sw
+      split: validation
+      revision: a5a45e4ff92d5d3f34de70aaf4b72c3bdf9f7f16
+    metrics:
+    - type: Accuracy
+      value: 49.6
+  - task:
+      type: Natural language inference
+    dataset:
+      type: xnli
+      name: XNLI (th)
+      config: th
+      split: validation
+      revision: a5a45e4ff92d5d3f34de70aaf4b72c3bdf9f7f16
+    metrics:
+    - type: Accuracy
+      value: 52.13
+  - task:
+      type: Natural language inference
+    dataset:
+      type: xnli
+      name: XNLI (tr)
+      config: tr
+      split: validation
+      revision: a5a45e4ff92d5d3f34de70aaf4b72c3bdf9f7f16
+    metrics:
+    - type: Accuracy
+      value: 50.56
+  - task:
+      type: Natural language inference
+    dataset:
+      type: xnli
+      name: XNLI (ur)
+      config: ur
+      split: validation
+      revision: a5a45e4ff92d5d3f34de70aaf4b72c3bdf9f7f16
+    metrics:
+    - type: Accuracy
+      value: 47.91
+  - task:
+      type: Natural language inference
+    dataset:
+      type: xnli
+      name: XNLI (vi)
+      config: vi
+      split: validation
+      revision: a5a45e4ff92d5d3f34de70aaf4b72c3bdf9f7f16
+    metrics:
+    - type: Accuracy
+      value: 53.21
+  - task:
+      type: Natural language inference
+    dataset:
+      type: xnli
+      name: XNLI (zh)
+      config: zh
+      split: validation
+      revision: a5a45e4ff92d5d3f34de70aaf4b72c3bdf9f7f16
+    metrics:
+    - type: Accuracy
+      value: 50.64
+  - task:
+      type: Program synthesis
+    dataset:
+      type: openai_humaneval
+      name: HumanEval
+      config: None
+      split: test
+      revision: e8dc562f5de170c54b5481011dd9f4fa04845771
+    metrics:
+    - type: Pass@1
+      value: 0
+    - type: Pass@10
+      value: 0
+    - type: Pass@100
+      value: 0
+  - task:
+      type: Sentence completion
+    dataset:
+      type: story_cloze
+      name: StoryCloze (2016)
+      config: '2016'
+      split: validation
+      revision: e724c6f8cdf7c7a2fb229d862226e15b023ee4db
+    metrics:
+    - type: Accuracy
+      value: 79.1
+  - task:
+      type: Sentence completion
+    dataset:
+      type: super_glue
+      name: SuperGLUE (copa)
+      config: copa
+      split: validation
+      revision: 9e12063561e7e6c79099feb6d5a493142584e9e2
+    metrics:
+    - type: Accuracy
+      value: 72
+  - task:
+      type: Sentence completion
+    dataset:
+      type: xcopa
+      name: XCOPA (et)
+      config: et
+      split: validation
+      revision: 37f73c60fb123111fa5af5f9b705d0b3747fd187
+    metrics:
+    - type: Accuracy
+      value: 70
+  - task:
+      type: Sentence completion
+    dataset:
+      type: xcopa
+      name: XCOPA (ht)
+      config: ht
+      split: validation
+      revision: 37f73c60fb123111fa5af5f9b705d0b3747fd187
+    metrics:
+    - type: Accuracy
+      value: 66
+  - task:
+      type: Sentence completion
+    dataset:
+      type: xcopa
+      name: XCOPA (id)
+      config: id
+      split: validation
+      revision: 37f73c60fb123111fa5af5f9b705d0b3747fd187
+    metrics:
+    - type: Accuracy
+      value: 71
+  - task:
+      type: Sentence completion
+    dataset:
+      type: xcopa
+      name: XCOPA (it)
+      config: it
+      split: validation
+      revision: 37f73c60fb123111fa5af5f9b705d0b3747fd187
+    metrics:
+    - type: Accuracy
+      value: 70
+  - task:
+      type: Sentence completion
+    dataset:
+      type: xcopa
+      name: XCOPA (qu)
+      config: qu
+      split: validation
+      revision: 37f73c60fb123111fa5af5f9b705d0b3747fd187
+    metrics:
+    - type: Accuracy
+      value: 56
+  - task:
+      type: Sentence completion
+    dataset:
+      type: xcopa
+      name: XCOPA (sw)
+      config: sw
+      split: validation
+      revision: 37f73c60fb123111fa5af5f9b705d0b3747fd187
+    metrics:
+    - type: Accuracy
+      value: 53
+  - task:
+      type: Sentence completion
+    dataset:
+      type: xcopa
+      name: XCOPA (ta)
+      config: ta
+      split: validation
+      revision: 37f73c60fb123111fa5af5f9b705d0b3747fd187
+    metrics:
+    - type: Accuracy
+      value: 64
+  - task:
+      type: Sentence completion
+    dataset:
+      type: xcopa
+      name: XCOPA (th)
+      config: th
+      split: validation
+      revision: 37f73c60fb123111fa5af5f9b705d0b3747fd187
+    metrics:
+    - type: Accuracy
+      value: 60
+  - task:
+      type: Sentence completion
+    dataset:
+      type: xcopa
+      name: XCOPA (tr)
+      config: tr
+      split: validation
+      revision: 37f73c60fb123111fa5af5f9b705d0b3747fd187
+    metrics:
+    - type: Accuracy
+      value: 58
+  - task:
+      type: Sentence completion
+    dataset:
+      type: xcopa
+      name: XCOPA (vi)
+      config: vi
+      split: validation
+      revision: 37f73c60fb123111fa5af5f9b705d0b3747fd187
+    metrics:
+    - type: Accuracy
+      value: 68
+  - task:
+      type: Sentence completion
+    dataset:
+      type: xcopa
+      name: XCOPA (zh)
+      config: zh
+      split: validation
+      revision: 37f73c60fb123111fa5af5f9b705d0b3747fd187
+    metrics:
+    - type: Accuracy
+      value: 65
+  - task:
+      type: Sentence completion
+    dataset:
+      type: Muennighoff/xstory_cloze
+      name: XStoryCloze (ar)
+      config: ar
+      split: validation
+      revision: 8bb76e594b68147f1a430e86829d07189622b90d
+    metrics:
+    - type: Accuracy
+      value: 70.09
+  - task:
+      type: Sentence completion
+    dataset:
+      type: Muennighoff/xstory_cloze
+      name: XStoryCloze (es)
+      config: es
+      split: validation
+      revision: 8bb76e594b68147f1a430e86829d07189622b90d
+    metrics:
+    - type: Accuracy
+      value: 77.17
+  - task:
+      type: Sentence completion
+    dataset:
+      type: Muennighoff/xstory_cloze
+      name: XStoryCloze (eu)
+      config: eu
+      split: validation
+      revision: 8bb76e594b68147f1a430e86829d07189622b90d
+    metrics:
+    - type: Accuracy
+      value: 69.03
+  - task:
+      type: Sentence completion
+    dataset:
+      type: Muennighoff/xstory_cloze
+      name: XStoryCloze (hi)
+      config: hi
+      split: validation
+      revision: 8bb76e594b68147f1a430e86829d07189622b90d
+    metrics:
+    - type: Accuracy
+      value: 71.08
+  - task:
+      type: Sentence completion
+    dataset:
+      type: Muennighoff/xstory_cloze
+      name: XStoryCloze (id)
+      config: id
+      split: validation
+      revision: 8bb76e594b68147f1a430e86829d07189622b90d
+    metrics:
+    - type: Accuracy
+      value: 75.71
+  - task:
+      type: Sentence completion
+    dataset:
+      type: Muennighoff/xstory_cloze
+      name: XStoryCloze (my)
+      config: my
+      split: validation
+      revision: 8bb76e594b68147f1a430e86829d07189622b90d
+    metrics:
+    - type: Accuracy
+      value: 65.65
+  - task:
+      type: Sentence completion
+    dataset:
+      type: Muennighoff/xstory_cloze
+      name: XStoryCloze (ru)
+      config: ru
+      split: validation
+      revision: 8bb76e594b68147f1a430e86829d07189622b90d
+    metrics:
+    - type: Accuracy
+      value: 74.85
+  - task:
+      type: Sentence completion
+    dataset:
+      type: Muennighoff/xstory_cloze
+      name: XStoryCloze (sw)
+      config: sw
+      split: validation
+      revision: 8bb76e594b68147f1a430e86829d07189622b90d
+    metrics:
+    - type: Accuracy
+      value: 71.14
+  - task:
+      type: Sentence completion
+    dataset:
+      type: Muennighoff/xstory_cloze
+      name: XStoryCloze (te)
+      config: te
+      split: validation
+      revision: 8bb76e594b68147f1a430e86829d07189622b90d
+    metrics:
+    - type: Accuracy
+      value: 68.89
+  - task:
+      type: Sentence completion
+    dataset:
+      type: Muennighoff/xstory_cloze
+      name: XStoryCloze (zh)
+      config: zh
+      split: validation
+      revision: 8bb76e594b68147f1a430e86829d07189622b90d
+    metrics:
+    - type: Accuracy
+      value: 72.93
+duplicated_from: bigscience/mt0-xl
+---
+![xmtf](https://github.com/bigscience-workshop/xmtf/blob/master/xmtf_banner.png?raw=true)
+#  Table of Contents
+1. [Model Summary](#model-summary)
+2. [Use](#use)
+3. [Limitations](#limitations)
+4. [Training](#training)
+5. [Evaluation](#evaluation)
+7. [Citation](#citation)
+# Model Summary
+> We present BLOOMZ & mT0, a family of models capable of following human instructions in dozens of languages zero-shot. We finetune BLOOM & mT5 pretrained multilingual language models on our crosslingual task mixture (xP3) and find our resulting models capable of crosslingual generalization to unseen tasks & languages.
+- **Repository:** [bigscience-workshop/xmtf](https://github.com/bigscience-workshop/xmtf)
+- **Paper:** [Crosslingual Generalization through Multitask Finetuning](https://arxiv.org/abs/2211.01786)
+- **Point of Contact:** [Niklas Muennighoff](mailto:niklas@hf.co)
+- **Languages:** Refer to [mc4](https://huggingface.co/datasets/mc4) for pretraining & [xP3](https://huggingface.co/bigscience/xP3) for finetuning language proportions. It understands both pretraining & finetuning languages.
+- **BLOOMZ & mT0 Model Family:**
+<div class="max-w-full overflow-auto">
+<table>
+  <tr>
+<th colspan="12">Multitask finetuned on <a style="font-weight:bold" href=https://huggingface.co/datasets/bigscience/xP3>xP3</a>. Recommended for prompting in English.
+</tr>
+<tr>
+<td>Parameters</td>
+<td>300M</td>
+<td>580M</td>
+<td>1.2B</td>
+<td>3.7B</td>
+<td>13B</td>
+<td>560M</td>
+<td>1.1B</td>
+<td>1.7B</td>
+<td>3B</td>
+<td>7.1B</td>
+<td>176B</td>
+</tr>
+<tr>
+<td>Finetuned Model</td>
+<td><a href=https://huggingface.co/bigscience/mt0-small>mt0-small</a></td>
+<td><a href=https://huggingface.co/bigscience/mt0-base>mt0-base</a></td>
+<td><a href=https://huggingface.co/bigscience/mt0-large>mt0-large</a></td>
+<td><a href=https://huggingface.co/bigscience/mt0-xl>mt0-xl</a></td>
+<td><a href=https://huggingface.co/bigscience/mt0-xxl>mt0-xxl</a></td>
+<td><a href=https://huggingface.co/bigscience/bloomz-560m>bloomz-560m</a></td>
+<td><a href=https://huggingface.co/bigscience/bloomz-1b1>bloomz-1b1</a></td>
+<td><a href=https://huggingface.co/bigscience/bloomz-1b7>bloomz-1b7</a></td>
+<td><a href=https://huggingface.co/bigscience/bloomz-3b>bloomz-3b</a></td>
+<td><a href=https://huggingface.co/bigscience/bloomz-7b1>bloomz-7b1</a></td>
+<td><a href=https://huggingface.co/bigscience/bloomz>bloomz</a></td>
+</tr>
+</tr>
+  <tr>
+<th colspan="12">Multitask finetuned on <a style="font-weight:bold" href=https://huggingface.co/datasets/bigscience/xP3mt>xP3mt</a>. Recommended for prompting in non-English.</th>
+</tr>
+<tr>
+<td>Finetuned Model</td>
+<td></td>
+<td></td>
+<td></td>
+<td></td>
+<td><a href=https://huggingface.co/bigscience/mt0-xxl-mt>mt0-xxl-mt</a></td>
+<td></td>
+<td></td>
+<td></td>
+<td></td>
+<td><a href=https://huggingface.co/bigscience/bloomz-7b1-mt>bloomz-7b1-mt</a></td>
+<td><a href=https://huggingface.co/bigscience/bloomz-mt>bloomz-mt</a></td>
+</tr>
+<th colspan="12">Multitask finetuned on <a style="font-weight:bold" href=https://huggingface.co/datasets/Muennighoff/P3>P3</a>. Released for research purposes only. Strictly inferior to above models!</th>
+</tr>
+<tr>
+<td>Finetuned Model</td>
+<td></td>
+<td></td>
+<td></td>
+<td></td>
+<td><a href=https://huggingface.co/bigscience/mt0-xxl-p3>mt0-xxl-p3</a></td>
+<td></td>
+<td></td>
+<td></td>
+<td></td>
+<td><a href=https://huggingface.co/bigscience/bloomz-7b1-p3>bloomz-7b1-p3</a></td>
+<td><a href=https://huggingface.co/bigscience/bloomz-p3>bloomz-p3</a></td>
+</tr>
+<th colspan="12">Original pretrained checkpoints. Not recommended.</th>
+<tr>
+<td>Pretrained Model</td>
+<td><a href=https://huggingface.co/google/mt5-small>mt5-small</a></td>
+<td><a href=https://huggingface.co/google/mt5-base>mt5-base</a></td>
+<td><a href=https://huggingface.co/google/mt5-large>mt5-large</a></td>
+<td><a href=https://huggingface.co/google/mt5-xl>mt5-xl</a></td>
+<td><a href=https://huggingface.co/google/mt5-xxl>mt5-xxl</a></td>
+<td><a href=https://huggingface.co/bigscience/bloom-560m>bloom-560m</a></td>
+<td><a href=https://huggingface.co/bigscience/bloom-1b1>bloom-1b1</a></td>
+<td><a href=https://huggingface.co/bigscience/bloom-1b7>bloom-1b7</a></td>
+<td><a href=https://huggingface.co/bigscience/bloom-3b>bloom-3b</a></td>
+<td><a href=https://huggingface.co/bigscience/bloom-7b1>bloom-7b1</a></td>
+<td><a href=https://huggingface.co/bigscience/bloom>bloom</a></td>
+</tr>
+</table>
+</div>
+# Use
+## Intended use
+We recommend using the model to perform tasks expressed in natural language. For example, given the prompt "*Translate to English: Je t’aime.*", the model will most likely answer "*I love you.*". Some prompt ideas from our paper:
+- 一个传奇的开端，一个不灭的神话，这不仅仅是一部电影，而是作为一个走进新时代的标签，永远彪炳史册。你认为这句话的立场是赞扬、中立还是批评?
+- Suggest at least five related search terms to "Mạng neural nhân tạo".
+- Write a fairy tale about a troll saving a princess from a dangerous dragon. The fairy tale is a masterpiece that has achieved praise worldwide and its moral is "Heroes Come in All Shapes and Sizes". Story (in Spanish):
+- Explain in a sentence in Telugu what is backpropagation in neural networks.
+**Feel free to share your generations in the Community tab!**
+## How to use
+### CPU
+<details>
+<summary> Click to expand </summary>
+```python
+# pip install -q transformers
+from transformers import AutoModelForSeq2SeqLM, AutoTokenizer
+checkpoint = "bigscience/mt0-xl"
+tokenizer = AutoTokenizer.from_pretrained(checkpoint)
+model = AutoModelForSeq2SeqLM.from_pretrained(checkpoint)
+inputs = tokenizer.encode("Translate to English: Je t’aime.", return_tensors="pt")
+outputs = model.generate(inputs)
+print(tokenizer.decode(outputs[0]))
+```
+</details>
+### GPU
+<details>
+<summary> Click to expand </summary>
+```python
+# pip install -q transformers accelerate
+from transformers import AutoModelForSeq2SeqLM, AutoTokenizer
+checkpoint = "bigscience/mt0-xl"
+tokenizer = AutoTokenizer.from_pretrained(checkpoint)
+model = AutoModelForSeq2SeqLM.from_pretrained(checkpoint, torch_dtype="auto", device_map="auto")
+inputs = tokenizer.encode("Translate to English: Je t’aime.", return_tensors="pt").to("cuda")
+outputs = model.generate(inputs)
+print(tokenizer.decode(outputs[0]))
+```
+</details>
+### GPU in 8bit
+<details>
+<summary> Click to expand </summary>
+```python
+# pip install -q transformers accelerate bitsandbytes
+from transformers import AutoModelForSeq2SeqLM, AutoTokenizer
+checkpoint = "bigscience/mt0-xl"
+tokenizer = AutoTokenizer.from_pretrained(checkpoint)
+model = AutoModelForSeq2SeqLM.from_pretrained(checkpoint, device_map="auto", load_in_8bit=True)
+inputs = tokenizer.encode("Translate to English: Je t’aime.", return_tensors="pt").to("cuda")
+outputs = model.generate(inputs)
+print(tokenizer.decode(outputs[0]))
+```
+</details>
+<!-- Necessary for whitespace -->
+###
+# Limitations
+**Prompt Engineering:** The performance may vary depending on the prompt. For BLOOMZ models, we recommend making it very clear when the input stops to avoid the model trying to continue it. For example, the prompt "*Translate to English: Je t'aime*" without the full stop (.) at the end, may result in the model trying to continue the French sentence. Better prompts are e.g. "*Translate to English: Je t'aime.*", "*Translate to English: Je t'aime. Translation:*" "*What is "Je t'aime." in English?*", where it is clear for the model when it should answer. Further, we recommend providing the model as much context as possible. For example, if you want it to answer in Telugu, then tell the model, e.g. "*Explain in a sentence in Telugu what is backpropagation in neural networks.*".
+# Training
+## Model
+- **Architecture:** Same as [mt5-xl](https://huggingface.co/google/mt5-xl), also refer to the `config.json` file
+- **Finetuning steps:** 10000
+- **Finetuning tokens:** 1.85 billion
+- **Precision:** bfloat16
+## Hardware
+- **TPUs:** TPUv4-128
+## Software
+- **Orchestration:** [T5X](https://github.com/google-research/t5x)
+- **Neural networks:** [Jax](https://github.com/google/jax)
+# Evaluation
+We refer to Table 7 from our [paper](https://arxiv.org/abs/2211.01786) & [bigscience/evaluation-results](https://huggingface.co/datasets/bigscience/evaluation-results) for zero-shot results on unseen tasks. The sidebar reports zero-shot performance of the best prompt per dataset config.
+# Citation
+```bibtex
+@misc{muennighoff2022crosslingual,
+      title={Crosslingual Generalization through Multitask Finetuning},
+      author={Niklas Muennighoff and Thomas Wang and Lintang Sutawika and Adam Roberts and Stella Biderman and Teven Le Scao and M Saiful Bari and Sheng Shen and Zheng-Xin Yong and Hailey Schoelkopf and Xiangru Tang and Dragomir Radev and Alham Fikri Aji and Khalid Almubarak and Samuel Albanie and Zaid Alyafeai and Albert Webson and Edward Raff and Colin Raffel},
+      year={2022},
+      eprint={2211.01786},
+      archivePrefix={arXiv},
+      primaryClass={cs.CL}
+}
+```

config.json ADDED Viewed

	@@ -0,0 +1,32 @@

+{
+  "_name_or_path": "google/mt5-xl",
+  "architectures": [
+    "MT5ForConditionalGeneration"
+  ],
+  "d_ff": 5120,
+  "d_kv": 64,
+  "d_model": 2048,
+  "decoder_start_token_id": 0,
+  "dense_act_fn": "gelu_new",
+  "dropout_rate": 0.1,
+  "eos_token_id": 1,
+  "feed_forward_proj": "gated-gelu",
+  "initializer_factor": 1.0,
+  "is_encoder_decoder": true,
+  "is_gated_act": true,
+  "layer_norm_epsilon": 1e-06,
+  "model_type": "mt5",
+  "num_decoder_layers": 24,
+  "num_heads": 32,
+  "num_layers": 24,
+  "output_past": true,
+  "pad_token_id": 0,
+  "relative_attention_max_distance": 128,
+  "relative_attention_num_buckets": 32,
+  "tie_word_embeddings": false,
+  "tokenizer_class": "T5Tokenizer",
+  "torch_dtype": "float32",
+  "transformers_version": "4.23.1",
+  "use_cache": true,
+  "vocab_size": 250112
+}

pytorch_model-00001-of-00002.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:2cded4ece5122b52af01c8606cc98a621da6fb59189ad87e6a5682d5cd0487b2
+size 7938340473

pytorch_model-00002-of-00002.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:b7bc19684b9eb3b9a62294fc284967a7cc7744fd5021eb97410e4e443fe3c56a
+size 7032322681

pytorch_model.bin.index.json ADDED Viewed

	@@ -0,0 +1,566 @@

+{
+  "metadata": {
+    "total_size": 17019396096
+  },
+  "weight_map": {
+    "decoder.block.0.layer.0.SelfAttention.k.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.0.layer.0.SelfAttention.o.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.0.layer.0.SelfAttention.q.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.0.layer.0.SelfAttention.relative_attention_bias.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.0.layer.0.SelfAttention.v.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.0.layer.0.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.0.layer.1.EncDecAttention.k.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.0.layer.1.EncDecAttention.o.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.0.layer.1.EncDecAttention.q.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.0.layer.1.EncDecAttention.v.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.0.layer.1.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.0.layer.2.DenseReluDense.wi_0.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.0.layer.2.DenseReluDense.wi_1.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.0.layer.2.DenseReluDense.wo.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.0.layer.2.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.1.layer.0.SelfAttention.k.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.1.layer.0.SelfAttention.o.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.1.layer.0.SelfAttention.q.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.1.layer.0.SelfAttention.v.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.1.layer.0.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.1.layer.1.EncDecAttention.k.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.1.layer.1.EncDecAttention.o.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.1.layer.1.EncDecAttention.q.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.1.layer.1.EncDecAttention.v.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.1.layer.1.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.1.layer.2.DenseReluDense.wi_0.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.1.layer.2.DenseReluDense.wi_1.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.1.layer.2.DenseReluDense.wo.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.1.layer.2.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.10.layer.0.SelfAttention.k.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.10.layer.0.SelfAttention.o.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.10.layer.0.SelfAttention.q.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.10.layer.0.SelfAttention.v.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.10.layer.0.layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.10.layer.1.EncDecAttention.k.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.10.layer.1.EncDecAttention.o.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.10.layer.1.EncDecAttention.q.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.10.layer.1.EncDecAttention.v.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.10.layer.1.layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.10.layer.2.DenseReluDense.wi_0.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.10.layer.2.DenseReluDense.wi_1.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.10.layer.2.DenseReluDense.wo.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.10.layer.2.layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.11.layer.0.SelfAttention.k.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.11.layer.0.SelfAttention.o.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.11.layer.0.SelfAttention.q.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.11.layer.0.SelfAttention.v.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.11.layer.0.layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.11.layer.1.EncDecAttention.k.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.11.layer.1.EncDecAttention.o.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.11.layer.1.EncDecAttention.q.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.11.layer.1.EncDecAttention.v.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.11.layer.1.layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.11.layer.2.DenseReluDense.wi_0.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.11.layer.2.DenseReluDense.wi_1.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.11.layer.2.DenseReluDense.wo.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.11.layer.2.layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.12.layer.0.SelfAttention.k.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.12.layer.0.SelfAttention.o.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.12.layer.0.SelfAttention.q.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.12.layer.0.SelfAttention.v.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.12.layer.0.layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.12.layer.1.EncDecAttention.k.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.12.layer.1.EncDecAttention.o.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.12.layer.1.EncDecAttention.q.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.12.layer.1.EncDecAttention.v.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.12.layer.1.layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.12.layer.2.DenseReluDense.wi_0.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.12.layer.2.DenseReluDense.wi_1.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.12.layer.2.DenseReluDense.wo.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.12.layer.2.layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.13.layer.0.SelfAttention.k.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.13.layer.0.SelfAttention.o.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.13.layer.0.SelfAttention.q.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.13.layer.0.SelfAttention.v.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.13.layer.0.layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.13.layer.1.EncDecAttention.k.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.13.layer.1.EncDecAttention.o.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.13.layer.1.EncDecAttention.q.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.13.layer.1.EncDecAttention.v.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.13.layer.1.layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.13.layer.2.DenseReluDense.wi_0.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.13.layer.2.DenseReluDense.wi_1.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.13.layer.2.DenseReluDense.wo.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.13.layer.2.layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.14.layer.0.SelfAttention.k.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.14.layer.0.SelfAttention.o.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.14.layer.0.SelfAttention.q.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.14.layer.0.SelfAttention.v.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.14.layer.0.layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.14.layer.1.EncDecAttention.k.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.14.layer.1.EncDecAttention.o.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.14.layer.1.EncDecAttention.q.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.14.layer.1.EncDecAttention.v.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.14.layer.1.layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.14.layer.2.DenseReluDense.wi_0.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.14.layer.2.DenseReluDense.wi_1.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.14.layer.2.DenseReluDense.wo.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.14.layer.2.layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.15.layer.0.SelfAttention.k.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.15.layer.0.SelfAttention.o.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.15.layer.0.SelfAttention.q.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.15.layer.0.SelfAttention.v.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.15.layer.0.layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.15.layer.1.EncDecAttention.k.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.15.layer.1.EncDecAttention.o.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.15.layer.1.EncDecAttention.q.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.15.layer.1.EncDecAttention.v.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.15.layer.1.layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.15.layer.2.DenseReluDense.wi_0.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.15.layer.2.DenseReluDense.wi_1.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.15.layer.2.DenseReluDense.wo.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.15.layer.2.layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.16.layer.0.SelfAttention.k.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.16.layer.0.SelfAttention.o.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.16.layer.0.SelfAttention.q.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.16.layer.0.SelfAttention.v.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.16.layer.0.layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.16.layer.1.EncDecAttention.k.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.16.layer.1.EncDecAttention.o.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.16.layer.1.EncDecAttention.q.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.16.layer.1.EncDecAttention.v.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.16.layer.1.layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.16.layer.2.DenseReluDense.wi_0.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.16.layer.2.DenseReluDense.wi_1.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.16.layer.2.DenseReluDense.wo.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.16.layer.2.layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.17.layer.0.SelfAttention.k.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.17.layer.0.SelfAttention.o.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.17.layer.0.SelfAttention.q.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.17.layer.0.SelfAttention.v.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.17.layer.0.layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.17.layer.1.EncDecAttention.k.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.17.layer.1.EncDecAttention.o.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.17.layer.1.EncDecAttention.q.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.17.layer.1.EncDecAttention.v.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.17.layer.1.layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.17.layer.2.DenseReluDense.wi_0.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.17.layer.2.DenseReluDense.wi_1.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.17.layer.2.DenseReluDense.wo.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.17.layer.2.layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.18.layer.0.SelfAttention.k.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.18.layer.0.SelfAttention.o.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.18.layer.0.SelfAttention.q.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.18.layer.0.SelfAttention.v.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.18.layer.0.layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.18.layer.1.EncDecAttention.k.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.18.layer.1.EncDecAttention.o.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.18.layer.1.EncDecAttention.q.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.18.layer.1.EncDecAttention.v.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.18.layer.1.layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.18.layer.2.DenseReluDense.wi_0.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.18.layer.2.DenseReluDense.wi_1.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.18.layer.2.DenseReluDense.wo.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.18.layer.2.layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.19.layer.0.SelfAttention.k.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.19.layer.0.SelfAttention.o.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.19.layer.0.SelfAttention.q.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.19.layer.0.SelfAttention.v.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.19.layer.0.layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.19.layer.1.EncDecAttention.k.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.19.layer.1.EncDecAttention.o.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.19.layer.1.EncDecAttention.q.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.19.layer.1.EncDecAttention.v.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.19.layer.1.layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.19.layer.2.DenseReluDense.wi_0.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.19.layer.2.DenseReluDense.wi_1.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.19.layer.2.DenseReluDense.wo.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.19.layer.2.layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.2.layer.0.SelfAttention.k.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.2.layer.0.SelfAttention.o.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.2.layer.0.SelfAttention.q.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.2.layer.0.SelfAttention.v.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.2.layer.0.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.2.layer.1.EncDecAttention.k.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.2.layer.1.EncDecAttention.o.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.2.layer.1.EncDecAttention.q.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.2.layer.1.EncDecAttention.v.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.2.layer.1.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.2.layer.2.DenseReluDense.wi_0.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.2.layer.2.DenseReluDense.wi_1.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.2.layer.2.DenseReluDense.wo.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.2.layer.2.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.20.layer.0.SelfAttention.k.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.20.layer.0.SelfAttention.o.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.20.layer.0.SelfAttention.q.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.20.layer.0.SelfAttention.v.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.20.layer.0.layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.20.layer.1.EncDecAttention.k.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.20.layer.1.EncDecAttention.o.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.20.layer.1.EncDecAttention.q.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.20.layer.1.EncDecAttention.v.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.20.layer.1.layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.20.layer.2.DenseReluDense.wi_0.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.20.layer.2.DenseReluDense.wi_1.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.20.layer.2.DenseReluDense.wo.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.20.layer.2.layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.21.layer.0.SelfAttention.k.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.21.layer.0.SelfAttention.o.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.21.layer.0.SelfAttention.q.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.21.layer.0.SelfAttention.v.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.21.layer.0.layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.21.layer.1.EncDecAttention.k.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.21.layer.1.EncDecAttention.o.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.21.layer.1.EncDecAttention.q.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.21.layer.1.EncDecAttention.v.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.21.layer.1.layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.21.layer.2.DenseReluDense.wi_0.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.21.layer.2.DenseReluDense.wi_1.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.21.layer.2.DenseReluDense.wo.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.21.layer.2.layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.22.layer.0.SelfAttention.k.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.22.layer.0.SelfAttention.o.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.22.layer.0.SelfAttention.q.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.22.layer.0.SelfAttention.v.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.22.layer.0.layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.22.layer.1.EncDecAttention.k.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.22.layer.1.EncDecAttention.o.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.22.layer.1.EncDecAttention.q.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.22.layer.1.EncDecAttention.v.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.22.layer.1.layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.22.layer.2.DenseReluDense.wi_0.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.22.layer.2.DenseReluDense.wi_1.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.22.layer.2.DenseReluDense.wo.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.22.layer.2.layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.23.layer.0.SelfAttention.k.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.23.layer.0.SelfAttention.o.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.23.layer.0.SelfAttention.q.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.23.layer.0.SelfAttention.v.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.23.layer.0.layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.23.layer.1.EncDecAttention.k.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.23.layer.1.EncDecAttention.o.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.23.layer.1.EncDecAttention.q.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.23.layer.1.EncDecAttention.v.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.23.layer.1.layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.23.layer.2.DenseReluDense.wi_0.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.23.layer.2.DenseReluDense.wi_1.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.23.layer.2.DenseReluDense.wo.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.23.layer.2.layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.3.layer.0.SelfAttention.k.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.3.layer.0.SelfAttention.o.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.3.layer.0.SelfAttention.q.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.3.layer.0.SelfAttention.v.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.3.layer.0.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.3.layer.1.EncDecAttention.k.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.3.layer.1.EncDecAttention.o.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.3.layer.1.EncDecAttention.q.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.3.layer.1.EncDecAttention.v.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.3.layer.1.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.3.layer.2.DenseReluDense.wi_0.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.3.layer.2.DenseReluDense.wi_1.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.3.layer.2.DenseReluDense.wo.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.3.layer.2.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.4.layer.0.SelfAttention.k.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.4.layer.0.SelfAttention.o.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.4.layer.0.SelfAttention.q.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.4.layer.0.SelfAttention.v.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.4.layer.0.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.4.layer.1.EncDecAttention.k.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.4.layer.1.EncDecAttention.o.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.4.layer.1.EncDecAttention.q.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.4.layer.1.EncDecAttention.v.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.4.layer.1.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.4.layer.2.DenseReluDense.wi_0.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.4.layer.2.DenseReluDense.wi_1.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.block.4.layer.2.DenseReluDense.wo.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.4.layer.2.layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.5.layer.0.SelfAttention.k.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.5.layer.0.SelfAttention.o.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.5.layer.0.SelfAttention.q.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.5.layer.0.SelfAttention.v.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.5.layer.0.layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.5.layer.1.EncDecAttention.k.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.5.layer.1.EncDecAttention.o.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.5.layer.1.EncDecAttention.q.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.5.layer.1.EncDecAttention.v.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.5.layer.1.layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.5.layer.2.DenseReluDense.wi_0.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.5.layer.2.DenseReluDense.wi_1.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.5.layer.2.DenseReluDense.wo.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.5.layer.2.layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.6.layer.0.SelfAttention.k.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.6.layer.0.SelfAttention.o.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.6.layer.0.SelfAttention.q.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.6.layer.0.SelfAttention.v.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.6.layer.0.layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.6.layer.1.EncDecAttention.k.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.6.layer.1.EncDecAttention.o.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.6.layer.1.EncDecAttention.q.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.6.layer.1.EncDecAttention.v.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.6.layer.1.layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.6.layer.2.DenseReluDense.wi_0.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.6.layer.2.DenseReluDense.wi_1.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.6.layer.2.DenseReluDense.wo.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.6.layer.2.layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.7.layer.0.SelfAttention.k.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.7.layer.0.SelfAttention.o.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.7.layer.0.SelfAttention.q.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.7.layer.0.SelfAttention.v.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.7.layer.0.layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.7.layer.1.EncDecAttention.k.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.7.layer.1.EncDecAttention.o.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.7.layer.1.EncDecAttention.q.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.7.layer.1.EncDecAttention.v.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.7.layer.1.layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.7.layer.2.DenseReluDense.wi_0.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.7.layer.2.DenseReluDense.wi_1.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.7.layer.2.DenseReluDense.wo.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.7.layer.2.layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.8.layer.0.SelfAttention.k.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.8.layer.0.SelfAttention.o.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.8.layer.0.SelfAttention.q.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.8.layer.0.SelfAttention.v.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.8.layer.0.layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.8.layer.1.EncDecAttention.k.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.8.layer.1.EncDecAttention.o.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.8.layer.1.EncDecAttention.q.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.8.layer.1.EncDecAttention.v.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.8.layer.1.layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.8.layer.2.DenseReluDense.wi_0.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.8.layer.2.DenseReluDense.wi_1.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.8.layer.2.DenseReluDense.wo.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.8.layer.2.layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.9.layer.0.SelfAttention.k.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.9.layer.0.SelfAttention.o.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.9.layer.0.SelfAttention.q.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.9.layer.0.SelfAttention.v.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.9.layer.0.layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.9.layer.1.EncDecAttention.k.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.9.layer.1.EncDecAttention.o.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.9.layer.1.EncDecAttention.q.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.9.layer.1.EncDecAttention.v.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.9.layer.1.layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.9.layer.2.DenseReluDense.wi_0.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.9.layer.2.DenseReluDense.wi_1.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.9.layer.2.DenseReluDense.wo.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.block.9.layer.2.layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "decoder.embed_tokens.weight": "pytorch_model-00001-of-00002.bin",
+    "decoder.final_layer_norm.weight": "pytorch_model-00002-of-00002.bin",
+    "encoder.block.0.layer.0.SelfAttention.k.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.0.layer.0.SelfAttention.o.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.0.layer.0.SelfAttention.q.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.0.layer.0.SelfAttention.relative_attention_bias.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.0.layer.0.SelfAttention.v.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.0.layer.0.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.0.layer.1.DenseReluDense.wi_0.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.0.layer.1.DenseReluDense.wi_1.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.0.layer.1.DenseReluDense.wo.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.0.layer.1.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.1.layer.0.SelfAttention.k.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.1.layer.0.SelfAttention.o.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.1.layer.0.SelfAttention.q.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.1.layer.0.SelfAttention.v.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.1.layer.0.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.1.layer.1.DenseReluDense.wi_0.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.1.layer.1.DenseReluDense.wi_1.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.1.layer.1.DenseReluDense.wo.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.1.layer.1.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.10.layer.0.SelfAttention.k.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.10.layer.0.SelfAttention.o.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.10.layer.0.SelfAttention.q.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.10.layer.0.SelfAttention.v.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.10.layer.0.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.10.layer.1.DenseReluDense.wi_0.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.10.layer.1.DenseReluDense.wi_1.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.10.layer.1.DenseReluDense.wo.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.10.layer.1.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.11.layer.0.SelfAttention.k.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.11.layer.0.SelfAttention.o.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.11.layer.0.SelfAttention.q.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.11.layer.0.SelfAttention.v.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.11.layer.0.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.11.layer.1.DenseReluDense.wi_0.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.11.layer.1.DenseReluDense.wi_1.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.11.layer.1.DenseReluDense.wo.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.11.layer.1.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.12.layer.0.SelfAttention.k.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.12.layer.0.SelfAttention.o.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.12.layer.0.SelfAttention.q.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.12.layer.0.SelfAttention.v.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.12.layer.0.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.12.layer.1.DenseReluDense.wi_0.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.12.layer.1.DenseReluDense.wi_1.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.12.layer.1.DenseReluDense.wo.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.12.layer.1.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.13.layer.0.SelfAttention.k.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.13.layer.0.SelfAttention.o.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.13.layer.0.SelfAttention.q.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.13.layer.0.SelfAttention.v.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.13.layer.0.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.13.layer.1.DenseReluDense.wi_0.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.13.layer.1.DenseReluDense.wi_1.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.13.layer.1.DenseReluDense.wo.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.13.layer.1.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.14.layer.0.SelfAttention.k.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.14.layer.0.SelfAttention.o.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.14.layer.0.SelfAttention.q.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.14.layer.0.SelfAttention.v.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.14.layer.0.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.14.layer.1.DenseReluDense.wi_0.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.14.layer.1.DenseReluDense.wi_1.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.14.layer.1.DenseReluDense.wo.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.14.layer.1.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.15.layer.0.SelfAttention.k.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.15.layer.0.SelfAttention.o.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.15.layer.0.SelfAttention.q.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.15.layer.0.SelfAttention.v.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.15.layer.0.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.15.layer.1.DenseReluDense.wi_0.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.15.layer.1.DenseReluDense.wi_1.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.15.layer.1.DenseReluDense.wo.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.15.layer.1.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.16.layer.0.SelfAttention.k.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.16.layer.0.SelfAttention.o.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.16.layer.0.SelfAttention.q.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.16.layer.0.SelfAttention.v.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.16.layer.0.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.16.layer.1.DenseReluDense.wi_0.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.16.layer.1.DenseReluDense.wi_1.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.16.layer.1.DenseReluDense.wo.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.16.layer.1.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.17.layer.0.SelfAttention.k.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.17.layer.0.SelfAttention.o.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.17.layer.0.SelfAttention.q.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.17.layer.0.SelfAttention.v.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.17.layer.0.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.17.layer.1.DenseReluDense.wi_0.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.17.layer.1.DenseReluDense.wi_1.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.17.layer.1.DenseReluDense.wo.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.17.layer.1.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.18.layer.0.SelfAttention.k.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.18.layer.0.SelfAttention.o.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.18.layer.0.SelfAttention.q.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.18.layer.0.SelfAttention.v.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.18.layer.0.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.18.layer.1.DenseReluDense.wi_0.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.18.layer.1.DenseReluDense.wi_1.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.18.layer.1.DenseReluDense.wo.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.18.layer.1.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.19.layer.0.SelfAttention.k.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.19.layer.0.SelfAttention.o.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.19.layer.0.SelfAttention.q.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.19.layer.0.SelfAttention.v.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.19.layer.0.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.19.layer.1.DenseReluDense.wi_0.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.19.layer.1.DenseReluDense.wi_1.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.19.layer.1.DenseReluDense.wo.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.19.layer.1.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.2.layer.0.SelfAttention.k.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.2.layer.0.SelfAttention.o.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.2.layer.0.SelfAttention.q.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.2.layer.0.SelfAttention.v.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.2.layer.0.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.2.layer.1.DenseReluDense.wi_0.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.2.layer.1.DenseReluDense.wi_1.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.2.layer.1.DenseReluDense.wo.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.2.layer.1.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.20.layer.0.SelfAttention.k.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.20.layer.0.SelfAttention.o.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.20.layer.0.SelfAttention.q.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.20.layer.0.SelfAttention.v.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.20.layer.0.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.20.layer.1.DenseReluDense.wi_0.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.20.layer.1.DenseReluDense.wi_1.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.20.layer.1.DenseReluDense.wo.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.20.layer.1.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.21.layer.0.SelfAttention.k.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.21.layer.0.SelfAttention.o.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.21.layer.0.SelfAttention.q.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.21.layer.0.SelfAttention.v.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.21.layer.0.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.21.layer.1.DenseReluDense.wi_0.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.21.layer.1.DenseReluDense.wi_1.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.21.layer.1.DenseReluDense.wo.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.21.layer.1.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.22.layer.0.SelfAttention.k.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.22.layer.0.SelfAttention.o.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.22.layer.0.SelfAttention.q.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.22.layer.0.SelfAttention.v.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.22.layer.0.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.22.layer.1.DenseReluDense.wi_0.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.22.layer.1.DenseReluDense.wi_1.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.22.layer.1.DenseReluDense.wo.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.22.layer.1.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.23.layer.0.SelfAttention.k.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.23.layer.0.SelfAttention.o.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.23.layer.0.SelfAttention.q.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.23.layer.0.SelfAttention.v.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.23.layer.0.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.23.layer.1.DenseReluDense.wi_0.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.23.layer.1.DenseReluDense.wi_1.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.23.layer.1.DenseReluDense.wo.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.23.layer.1.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.3.layer.0.SelfAttention.k.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.3.layer.0.SelfAttention.o.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.3.layer.0.SelfAttention.q.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.3.layer.0.SelfAttention.v.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.3.layer.0.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.3.layer.1.DenseReluDense.wi_0.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.3.layer.1.DenseReluDense.wi_1.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.3.layer.1.DenseReluDense.wo.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.3.layer.1.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.4.layer.0.SelfAttention.k.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.4.layer.0.SelfAttention.o.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.4.layer.0.SelfAttention.q.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.4.layer.0.SelfAttention.v.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.4.layer.0.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.4.layer.1.DenseReluDense.wi_0.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.4.layer.1.DenseReluDense.wi_1.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.4.layer.1.DenseReluDense.wo.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.4.layer.1.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.5.layer.0.SelfAttention.k.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.5.layer.0.SelfAttention.o.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.5.layer.0.SelfAttention.q.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.5.layer.0.SelfAttention.v.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.5.layer.0.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.5.layer.1.DenseReluDense.wi_0.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.5.layer.1.DenseReluDense.wi_1.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.5.layer.1.DenseReluDense.wo.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.5.layer.1.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.6.layer.0.SelfAttention.k.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.6.layer.0.SelfAttention.o.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.6.layer.0.SelfAttention.q.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.6.layer.0.SelfAttention.v.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.6.layer.0.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.6.layer.1.DenseReluDense.wi_0.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.6.layer.1.DenseReluDense.wi_1.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.6.layer.1.DenseReluDense.wo.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.6.layer.1.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.7.layer.0.SelfAttention.k.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.7.layer.0.SelfAttention.o.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.7.layer.0.SelfAttention.q.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.7.layer.0.SelfAttention.v.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.7.layer.0.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.7.layer.1.DenseReluDense.wi_0.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.7.layer.1.DenseReluDense.wi_1.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.7.layer.1.DenseReluDense.wo.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.7.layer.1.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.8.layer.0.SelfAttention.k.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.8.layer.0.SelfAttention.o.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.8.layer.0.SelfAttention.q.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.8.layer.0.SelfAttention.v.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.8.layer.0.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.8.layer.1.DenseReluDense.wi_0.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.8.layer.1.DenseReluDense.wi_1.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.8.layer.1.DenseReluDense.wo.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.8.layer.1.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.9.layer.0.SelfAttention.k.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.9.layer.0.SelfAttention.o.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.9.layer.0.SelfAttention.q.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.9.layer.0.SelfAttention.v.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.9.layer.0.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.9.layer.1.DenseReluDense.wi_0.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.9.layer.1.DenseReluDense.wi_1.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.9.layer.1.DenseReluDense.wo.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.block.9.layer.1.layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "encoder.final_layer_norm.weight": "pytorch_model-00001-of-00002.bin",
+    "lm_head.weight": "pytorch_model-00002-of-00002.bin",
+    "shared.weight": "pytorch_model-00001-of-00002.bin"
+  }
+}

special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,5 @@

+{
+  "eos_token": "</s>",
+  "pad_token": "<pad>",
+  "unk_token": "<unk>"
+}

spiece.model ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:ef78f86560d809067d12bac6c09f19a462cb3af3f54d2b8acbba26e1433125d6
+size 4309802

tokenizer.json ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:93c3578052e1605d8332eb961bc08d72e246071974e4cc54aa6991826b802aa5
+size 16330369

tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,11 @@

+{
+  "additional_special_tokens": null,
+  "eos_token": "</s>",
+  "extra_ids": 0,
+  "name_or_path": "google/mt5-large",
+  "pad_token": "<pad>",
+  "sp_model_kwargs": {},
+  "special_tokens_map_file": "/home/patrick/.cache/torch/transformers/685ac0ca8568ec593a48b61b0a3c272beee9bc194a3c7241d15dcadb5f875e53.f76030f3ec1b96a8199b2593390c610e76ca8028ef3d24680000619ffb646276",
+  "tokenizer_class": "T5Tokenizer",
+  "unk_token": "<unk>"
+}