Add new example

#6
by ybelkada HF staff - opened
This view is limited to 50 files because it contains too many changes.  See the raw diff here.
Files changed (50) hide show
  1. README.md +9 -17
  2. config.json +2 -2
  3. evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/ar/Answer_Given_options/results.json +9 -0
  4. evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/ar/Choose_Story_Ending/results.json +9 -0
  5. evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/ar/Generate_Ending/results.json +9 -0
  6. evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/ar/Novel_Correct_Ending/results.json +9 -0
  7. evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/ar/Story_Continuation_and_Options/results.json +9 -0
  8. evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/es/Answer_Given_options/results.json +9 -0
  9. evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/es/Choose_Story_Ending/results.json +9 -0
  10. evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/es/Generate_Ending/results.json +9 -0
  11. evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/es/Novel_Correct_Ending/results.json +9 -0
  12. evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/es/Story_Continuation_and_Options/results.json +9 -0
  13. evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/eu/Answer_Given_options/results.json +9 -0
  14. evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/eu/Choose_Story_Ending/results.json +9 -0
  15. evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/eu/Generate_Ending/results.json +9 -0
  16. evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/eu/Novel_Correct_Ending/results.json +9 -0
  17. evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/eu/Story_Continuation_and_Options/results.json +9 -0
  18. evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/hi/Answer_Given_options/results.json +9 -0
  19. evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/hi/Choose_Story_Ending/results.json +9 -0
  20. evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/hi/Generate_Ending/results.json +9 -0
  21. evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/hi/Novel_Correct_Ending/results.json +9 -0
  22. evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/hi/Story_Continuation_and_Options/results.json +9 -0
  23. evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/id/Answer_Given_options/results.json +9 -0
  24. evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/id/Choose_Story_Ending/results.json +9 -0
  25. evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/id/Generate_Ending/results.json +9 -0
  26. evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/id/Novel_Correct_Ending/results.json +9 -0
  27. evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/id/Story_Continuation_and_Options/results.json +9 -0
  28. evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/zh/Answer_Given_options/results.json +9 -0
  29. evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/zh/Choose_Story_Ending/results.json +9 -0
  30. evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/zh/Generate_Ending/results.json +9 -0
  31. evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/zh/Novel_Correct_Ending/results.json +9 -0
  32. evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/zh/Story_Continuation_and_Options/results.json +9 -0
  33. evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/en/Replace/results.json +9 -0
  34. evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/en/True_or_False/results.json +9 -0
  35. evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/en/does_underscore_refer_to/results.json +9 -0
  36. evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/en/stand_for/results.json +9 -0
  37. evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/en/underscore_refer_to/results.json +9 -0
  38. evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/fr/Replace/results.json +9 -0
  39. evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/fr/True_or_False/results.json +9 -0
  40. evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/fr/does_underscore_refer_to/results.json +9 -0
  41. evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/fr/stand_for/results.json +9 -0
  42. evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/fr/underscore_refer_to/results.json +9 -0
  43. evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/pt/Replace/results.json +9 -0
  44. evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/pt/True_or_False/results.json +9 -0
  45. evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/pt/does_underscore_refer_to/results.json +9 -0
  46. evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/pt/stand_for/results.json +9 -0
  47. evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/pt/underscore_refer_to/results.json +9 -0
  48. evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/zh/Replace/results.json +9 -0
  49. evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/zh/True_or_False/results.json +9 -0
  50. evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/zh/does_underscore_refer_to/results.json +9 -0
README.md CHANGED
@@ -64,7 +64,6 @@ programming_language:
64
  - Scala
65
  - TypeScript
66
  pipeline_tag: text-generation
67
- inference: false
68
  widget:
69
  - text: "一个传奇的开端,一个不灭的神话,这不仅仅是一部电影,而是作为一个走进新时代的标签,永远彪炳史册。Would you rate the previous review as positive, neutral or negative?"
70
  example_title: "zh-en sentiment"
@@ -86,13 +85,6 @@ widget:
86
  example_title: "es-en fable"
87
  - text: "Write a fable about wood elves living in a forest that is suddenly invaded by ogres. The fable is a masterpiece that has achieved praise worldwide and its moral is \"Violence is the last refuge of the incompetent\". Fable (in Hindi):"
88
  example_title: "hi-en fable"
89
- - text: "How many sides does a rectangle and heptagon have, when
90
- combined? Answer this question with some math.
91
- Ein Rechteck hat 4 Seiten. Ein Siebeneck hat 7 Seiten.
92
- In Kombination haben sie 4 + 7 = 11 Seiten.
93
- كم عدد الأضلاع التي يجمعها المربع والمثلث؟
94
- Répondez à cette question en chinois."
95
- example_title: "en-de-ar-fr-zh math"
96
  model-index:
97
  - name: bloomz
98
  results:
@@ -684,7 +676,6 @@ model-index:
684
  - **Languages:** Refer to [bloom](https://huggingface.co/bigscience/bloom) for pretraining & [xP3](https://huggingface.co/datasets/bigscience/xP3) for finetuning language proportions. It understands both pretraining & finetuning languages.
685
  - **BLOOMZ & mT0 Model Family:**
686
 
687
- <div class="max-w-full overflow-auto">
688
  <table>
689
  <tr>
690
  <th colspan="12">Multitask finetuned on <a style="font-weight:bold" href=https://huggingface.co/datasets/bigscience/xP3>xP3</a>. Recommended for prompting in English.
@@ -705,8 +696,8 @@ model-index:
705
  </tr>
706
  <tr>
707
  <td>Finetuned Model</td>
708
- <td><a href=https://huggingface.co/bigscience/mt0-small>mt0-small</a></td>
709
  <td><a href=https://huggingface.co/bigscience/mt0-base>mt0-base</a></td>
 
710
  <td><a href=https://huggingface.co/bigscience/mt0-large>mt0-large</a></td>
711
  <td><a href=https://huggingface.co/bigscience/mt0-xl>mt0-xl</a></td>
712
  <td><a href=https://huggingface.co/bigscience/mt0-xxl>mt0-xxl</a></td>
@@ -754,8 +745,8 @@ model-index:
754
  <th colspan="12">Original pretrained checkpoints. Not recommended.</th>
755
  <tr>
756
  <td>Pretrained Model</td>
757
- <td><a href=https://huggingface.co/google/mt5-small>mt5-small</a></td>
758
  <td><a href=https://huggingface.co/google/mt5-base>mt5-base</a></td>
 
759
  <td><a href=https://huggingface.co/google/mt5-large>mt5-large</a></td>
760
  <td><a href=https://huggingface.co/google/mt5-xl>mt5-xl</a></td>
761
  <td><a href=https://huggingface.co/google/mt5-xxl>mt5-xxl</a></td>
@@ -767,7 +758,6 @@ model-index:
767
  <td><a href=https://huggingface.co/bigscience/bloom>bloom</a></td>
768
  </tr>
769
  </table>
770
- </div>
771
 
772
 
773
  # Use
@@ -883,10 +873,12 @@ We refer to Table 7 from our [paper](https://arxiv.org/abs/2211.01786) & [bigsci
883
 
884
  # Citation
885
  ```bibtex
886
- @article{muennighoff2022crosslingual,
887
- title={Crosslingual generalization through multitask finetuning},
888
- author={Muennighoff, Niklas and Wang, Thomas and Sutawika, Lintang and Roberts, Adam and Biderman, Stella and Scao, Teven Le and Bari, M Saiful and Shen, Sheng and Yong, Zheng-Xin and Schoelkopf, Hailey and others},
889
- journal={arXiv preprint arXiv:2211.01786},
890
- year={2022}
 
 
891
  }
892
  ```
64
  - Scala
65
  - TypeScript
66
  pipeline_tag: text-generation
 
67
  widget:
68
  - text: "一个传奇的开端,一个不灭的神话,这不仅仅是一部电影,而是作为一个走进新时代的标签,永远彪炳史册。Would you rate the previous review as positive, neutral or negative?"
69
  example_title: "zh-en sentiment"
85
  example_title: "es-en fable"
86
  - text: "Write a fable about wood elves living in a forest that is suddenly invaded by ogres. The fable is a masterpiece that has achieved praise worldwide and its moral is \"Violence is the last refuge of the incompetent\". Fable (in Hindi):"
87
  example_title: "hi-en fable"
 
 
 
 
 
 
 
88
  model-index:
89
  - name: bloomz
90
  results:
676
  - **Languages:** Refer to [bloom](https://huggingface.co/bigscience/bloom) for pretraining & [xP3](https://huggingface.co/datasets/bigscience/xP3) for finetuning language proportions. It understands both pretraining & finetuning languages.
677
  - **BLOOMZ & mT0 Model Family:**
678
 
 
679
  <table>
680
  <tr>
681
  <th colspan="12">Multitask finetuned on <a style="font-weight:bold" href=https://huggingface.co/datasets/bigscience/xP3>xP3</a>. Recommended for prompting in English.
696
  </tr>
697
  <tr>
698
  <td>Finetuned Model</td>
 
699
  <td><a href=https://huggingface.co/bigscience/mt0-base>mt0-base</a></td>
700
+ <td><a href=https://huggingface.co/bigscience/mt0-small>mt0-small</a></td>
701
  <td><a href=https://huggingface.co/bigscience/mt0-large>mt0-large</a></td>
702
  <td><a href=https://huggingface.co/bigscience/mt0-xl>mt0-xl</a></td>
703
  <td><a href=https://huggingface.co/bigscience/mt0-xxl>mt0-xxl</a></td>
745
  <th colspan="12">Original pretrained checkpoints. Not recommended.</th>
746
  <tr>
747
  <td>Pretrained Model</td>
 
748
  <td><a href=https://huggingface.co/google/mt5-base>mt5-base</a></td>
749
+ <td><a href=https://huggingface.co/google/mt5-small>mt5-small</a></td>
750
  <td><a href=https://huggingface.co/google/mt5-large>mt5-large</a></td>
751
  <td><a href=https://huggingface.co/google/mt5-xl>mt5-xl</a></td>
752
  <td><a href=https://huggingface.co/google/mt5-xxl>mt5-xxl</a></td>
758
  <td><a href=https://huggingface.co/bigscience/bloom>bloom</a></td>
759
  </tr>
760
  </table>
 
761
 
762
 
763
  # Use
873
 
874
  # Citation
875
  ```bibtex
876
+ @misc{muennighoff2022crosslingual,
877
+ title={Crosslingual Generalization through Multitask Finetuning},
878
+ author={Niklas Muennighoff and Thomas Wang and Lintang Sutawika and Adam Roberts and Stella Biderman and Teven Le Scao and M Saiful Bari and Sheng Shen and Zheng-Xin Yong and Hailey Schoelkopf and Xiangru Tang and Dragomir Radev and Alham Fikri Aji and Khalid Almubarak and Samuel Albanie and Zaid Alyafeai and Albert Webson and Edward Raff and Colin Raffel},
879
+ year={2022},
880
+ eprint={2211.01786},
881
+ archivePrefix={arXiv},
882
+ primaryClass={cs.CL}
883
  }
884
  ```
config.json CHANGED
@@ -2,7 +2,7 @@
2
  "apply_residual_connection_post_layernorm": false,
3
  "attention_dropout": 0.0,
4
  "architectures": [
5
- "BloomForCausalLM"
6
  ],
7
  "attention_softmax_in_fp32": true,
8
  "seq_length": 2048,
@@ -22,4 +22,4 @@
22
  "transformers_version": "4.21.0",
23
  "use_cache": true,
24
  "vocab_size": 250880
25
- }
2
  "apply_residual_connection_post_layernorm": false,
3
  "attention_dropout": 0.0,
4
  "architectures": [
5
+ "BloomModel"
6
  ],
7
  "attention_softmax_in_fp32": true,
8
  "seq_length": 2048,
22
  "transformers_version": "4.21.0",
23
  "use_cache": true,
24
  "vocab_size": 250880
25
+ }
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/ar/Answer_Given_options/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xstory_cloze",
3
+ "dataset_config_name": "ar",
4
+ "template_name": "Answer Given options",
5
+ "evaluation": {
6
+ "accuracy": 0.7968232958305758
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='ar', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Answer Given options', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/ar/Choose_Story_Ending/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xstory_cloze",
3
+ "dataset_config_name": "ar",
4
+ "template_name": "Choose Story Ending",
5
+ "evaluation": {
6
+ "accuracy": 0.9232296492389146
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='ar', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Choose Story Ending', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/ar/Generate_Ending/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xstory_cloze",
3
+ "dataset_config_name": "ar",
4
+ "template_name": "Generate Ending",
5
+ "evaluation": {
6
+ "accuracy": 0.6677696889477167
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='ar', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Generate Ending', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/ar/Novel_Correct_Ending/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xstory_cloze",
3
+ "dataset_config_name": "ar",
4
+ "template_name": "Novel Correct Ending",
5
+ "evaluation": {
6
+ "accuracy": 0.9265387160820648
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='ar', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Novel Correct Ending', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/ar/Story_Continuation_and_Options/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xstory_cloze",
3
+ "dataset_config_name": "ar",
4
+ "template_name": "Story Continuation and Options",
5
+ "evaluation": {
6
+ "accuracy": 0.9126406353408338
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='ar', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Story Continuation and Options', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/es/Answer_Given_options/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xstory_cloze",
3
+ "dataset_config_name": "es",
4
+ "template_name": "Answer Given options",
5
+ "evaluation": {
6
+ "accuracy": 0.8729318332230311
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='es', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Answer Given options', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/es/Choose_Story_Ending/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xstory_cloze",
3
+ "dataset_config_name": "es",
4
+ "template_name": "Choose Story Ending",
5
+ "evaluation": {
6
+ "accuracy": 0.9417604235605559
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='es', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Choose Story Ending', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/es/Generate_Ending/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xstory_cloze",
3
+ "dataset_config_name": "es",
4
+ "template_name": "Generate Ending",
5
+ "evaluation": {
6
+ "accuracy": 0.7359364659166115
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='es', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Generate Ending', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/es/Novel_Correct_Ending/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xstory_cloze",
3
+ "dataset_config_name": "es",
4
+ "template_name": "Novel Correct Ending",
5
+ "evaluation": {
6
+ "accuracy": 0.9430840502978161
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='es', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Novel Correct Ending', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/es/Story_Continuation_and_Options/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xstory_cloze",
3
+ "dataset_config_name": "es",
4
+ "template_name": "Story Continuation and Options",
5
+ "evaluation": {
6
+ "accuracy": 0.9318332230311053
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='es', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Story Continuation and Options', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/eu/Answer_Given_options/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xstory_cloze",
3
+ "dataset_config_name": "eu",
4
+ "template_name": "Answer Given options",
5
+ "evaluation": {
6
+ "accuracy": 0.7054930509596293
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='eu', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Answer Given options', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/eu/Choose_Story_Ending/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xstory_cloze",
3
+ "dataset_config_name": "eu",
4
+ "template_name": "Choose Story Ending",
5
+ "evaluation": {
6
+ "accuracy": 0.8663136995367307
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='eu', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Choose Story Ending', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/eu/Generate_Ending/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xstory_cloze",
3
+ "dataset_config_name": "eu",
4
+ "template_name": "Generate Ending",
5
+ "evaluation": {
6
+ "accuracy": 0.6320317670416943
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='eu', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Generate Ending', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/eu/Novel_Correct_Ending/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xstory_cloze",
3
+ "dataset_config_name": "eu",
4
+ "template_name": "Novel Correct Ending",
5
+ "evaluation": {
6
+ "accuracy": 0.8689609530112509
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='eu', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Novel Correct Ending', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/eu/Story_Continuation_and_Options/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xstory_cloze",
3
+ "dataset_config_name": "eu",
4
+ "template_name": "Story Continuation and Options",
5
+ "evaluation": {
6
+ "accuracy": 0.8524156187954997
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='eu', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Story Continuation and Options', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/hi/Answer_Given_options/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xstory_cloze",
3
+ "dataset_config_name": "hi",
4
+ "template_name": "Answer Given options",
5
+ "evaluation": {
6
+ "accuracy": 0.798808735936466
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='hi', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Answer Given options', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/hi/Choose_Story_Ending/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xstory_cloze",
3
+ "dataset_config_name": "hi",
4
+ "template_name": "Choose Story Ending",
5
+ "evaluation": {
6
+ "accuracy": 0.8702845797485109
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='hi', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Choose Story Ending', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/hi/Generate_Ending/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xstory_cloze",
3
+ "dataset_config_name": "hi",
4
+ "template_name": "Generate Ending",
5
+ "evaluation": {
6
+ "accuracy": 0.6604897418927862
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='hi', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Generate Ending', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/hi/Novel_Correct_Ending/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xstory_cloze",
3
+ "dataset_config_name": "hi",
4
+ "template_name": "Novel Correct Ending",
5
+ "evaluation": {
6
+ "accuracy": 0.8788881535407015
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='hi', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Novel Correct Ending', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/hi/Story_Continuation_and_Options/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xstory_cloze",
3
+ "dataset_config_name": "hi",
4
+ "template_name": "Story Continuation and Options",
5
+ "evaluation": {
6
+ "accuracy": 0.870946393117141
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='hi', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Story Continuation and Options', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/id/Answer_Given_options/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xstory_cloze",
3
+ "dataset_config_name": "id",
4
+ "template_name": "Answer Given options",
5
+ "evaluation": {
6
+ "accuracy": 0.8557246856386499
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='id', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Answer Given options', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/id/Choose_Story_Ending/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xstory_cloze",
3
+ "dataset_config_name": "id",
4
+ "template_name": "Choose Story Ending",
5
+ "evaluation": {
6
+ "accuracy": 0.9212442091330245
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='id', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Choose Story Ending', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/id/Generate_Ending/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xstory_cloze",
3
+ "dataset_config_name": "id",
4
+ "template_name": "Generate Ending",
5
+ "evaluation": {
6
+ "accuracy": 0.7041694242223693
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='id', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Generate Ending', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/id/Novel_Correct_Ending/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xstory_cloze",
3
+ "dataset_config_name": "id",
4
+ "template_name": "Novel Correct Ending",
5
+ "evaluation": {
6
+ "accuracy": 0.9205823957643945
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='id', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Novel Correct Ending', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/id/Story_Continuation_and_Options/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xstory_cloze",
3
+ "dataset_config_name": "id",
4
+ "template_name": "Story Continuation and Options",
5
+ "evaluation": {
6
+ "accuracy": 0.9066843150231635
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='id', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Story Continuation and Options', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/zh/Answer_Given_options/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xstory_cloze",
3
+ "dataset_config_name": "zh",
4
+ "template_name": "Answer Given options",
5
+ "evaluation": {
6
+ "accuracy": 0.900066181336863
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Answer Given options', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/zh/Choose_Story_Ending/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xstory_cloze",
3
+ "dataset_config_name": "zh",
4
+ "template_name": "Choose Story Ending",
5
+ "evaluation": {
6
+ "accuracy": 0.9232296492389146
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Choose Story Ending', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/zh/Generate_Ending/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xstory_cloze",
3
+ "dataset_config_name": "zh",
4
+ "template_name": "Generate Ending",
5
+ "evaluation": {
6
+ "accuracy": 0.684976836532098
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Generate Ending', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/zh/Novel_Correct_Ending/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xstory_cloze",
3
+ "dataset_config_name": "zh",
4
+ "template_name": "Novel Correct Ending",
5
+ "evaluation": {
6
+ "accuracy": 0.9311714096624751
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Novel Correct Ending', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/zh/Story_Continuation_and_Options/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xstory_cloze",
3
+ "dataset_config_name": "zh",
4
+ "template_name": "Story Continuation and Options",
5
+ "evaluation": {
6
+ "accuracy": 0.9199205823957644
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Story Continuation and Options', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/en/Replace/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xwinograd",
3
+ "dataset_config_name": "en",
4
+ "template_name": "Replace",
5
+ "evaluation": {
6
+ "accuracy": 0.6847311827956989
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='en', dataset_name='Muennighoff/xwinograd', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Replace', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/en/True_or_False/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xwinograd",
3
+ "dataset_config_name": "en",
4
+ "template_name": "True or False",
5
+ "evaluation": {
6
+ "accuracy": 0.5135483870967742
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='en', dataset_name='Muennighoff/xwinograd', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='True or False', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/en/does_underscore_refer_to/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xwinograd",
3
+ "dataset_config_name": "en",
4
+ "template_name": "does underscore refer to",
5
+ "evaluation": {
6
+ "accuracy": 0.6787096774193548
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='en', dataset_name='Muennighoff/xwinograd', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='does underscore refer to', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/en/stand_for/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xwinograd",
3
+ "dataset_config_name": "en",
4
+ "template_name": "stand for",
5
+ "evaluation": {
6
+ "accuracy": 0.5053763440860215
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='en', dataset_name='Muennighoff/xwinograd', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='stand for', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/en/underscore_refer_to/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xwinograd",
3
+ "dataset_config_name": "en",
4
+ "template_name": "underscore refer to",
5
+ "evaluation": {
6
+ "accuracy": 0.690752688172043
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='en', dataset_name='Muennighoff/xwinograd', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='underscore refer to', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/fr/Replace/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xwinograd",
3
+ "dataset_config_name": "fr",
4
+ "template_name": "Replace",
5
+ "evaluation": {
6
+ "accuracy": 0.6506024096385542
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='fr', dataset_name='Muennighoff/xwinograd', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Replace', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/fr/True_or_False/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xwinograd",
3
+ "dataset_config_name": "fr",
4
+ "template_name": "True or False",
5
+ "evaluation": {
6
+ "accuracy": 0.4939759036144578
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='fr', dataset_name='Muennighoff/xwinograd', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='True or False', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/fr/does_underscore_refer_to/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xwinograd",
3
+ "dataset_config_name": "fr",
4
+ "template_name": "does underscore refer to",
5
+ "evaluation": {
6
+ "accuracy": 0.6867469879518072
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='fr', dataset_name='Muennighoff/xwinograd', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='does underscore refer to', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/fr/stand_for/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xwinograd",
3
+ "dataset_config_name": "fr",
4
+ "template_name": "stand for",
5
+ "evaluation": {
6
+ "accuracy": 0.46987951807228917
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='fr', dataset_name='Muennighoff/xwinograd', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='stand for', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/fr/underscore_refer_to/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xwinograd",
3
+ "dataset_config_name": "fr",
4
+ "template_name": "underscore refer to",
5
+ "evaluation": {
6
+ "accuracy": 0.6626506024096386
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='fr', dataset_name='Muennighoff/xwinograd', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='underscore refer to', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/pt/Replace/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xwinograd",
3
+ "dataset_config_name": "pt",
4
+ "template_name": "Replace",
5
+ "evaluation": {
6
+ "accuracy": 0.6349809885931559
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='pt', dataset_name='Muennighoff/xwinograd', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='Replace', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/pt/True_or_False/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xwinograd",
3
+ "dataset_config_name": "pt",
4
+ "template_name": "True or False",
5
+ "evaluation": {
6
+ "accuracy": 0.4866920152091255
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='pt', dataset_name='Muennighoff/xwinograd', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='True or False', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/pt/does_underscore_refer_to/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xwinograd",
3
+ "dataset_config_name": "pt",
4
+ "template_name": "does underscore refer to",
5
+ "evaluation": {
6
+ "accuracy": 0.6387832699619772
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='pt', dataset_name='Muennighoff/xwinograd', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='does underscore refer to', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/pt/stand_for/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xwinograd",
3
+ "dataset_config_name": "pt",
4
+ "template_name": "stand for",
5
+ "evaluation": {
6
+ "accuracy": 0.49429657794676807
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='pt', dataset_name='Muennighoff/xwinograd', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='stand for', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/pt/underscore_refer_to/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xwinograd",
3
+ "dataset_config_name": "pt",
4
+ "template_name": "underscore refer to",
5
+ "evaluation": {
6
+ "accuracy": 0.6425855513307985
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='pt', dataset_name='Muennighoff/xwinograd', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='underscore refer to', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/zh/Replace/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xwinograd",
3
+ "dataset_config_name": "zh",
4
+ "template_name": "Replace",
5
+ "evaluation": {
6
+ "accuracy": 0.6865079365079365
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xwinograd', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Replace', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/zh/True_or_False/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xwinograd",
3
+ "dataset_config_name": "zh",
4
+ "template_name": "True or False",
5
+ "evaluation": {
6
+ "accuracy": 0.5277777777777778
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xwinograd', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='True or False', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/zh/does_underscore_refer_to/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xwinograd",
3
+ "dataset_config_name": "zh",
4
+ "template_name": "does underscore refer to",
5
+ "evaluation": {
6
+ "accuracy": 0.6884920634920635
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xwinograd', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='does underscore refer to', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }