Muennighoff commited on
Commit
a202384
1 Parent(s): c900c5a

Remove eval (Moved to bigscience/evaluation-results)

Browse files
This view is limited to 50 files because it contains too many changes.   See raw diff
Files changed (50) hide show
  1. evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/ar/Answer_Given_options/results.json +0 -9
  2. evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/ar/Choose_Story_Ending/results.json +0 -9
  3. evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/ar/Generate_Ending/results.json +0 -9
  4. evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/ar/Novel_Correct_Ending/results.json +0 -9
  5. evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/ar/Story_Continuation_and_Options/results.json +0 -9
  6. evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/es/Answer_Given_options/results.json +0 -9
  7. evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/es/Choose_Story_Ending/results.json +0 -9
  8. evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/es/Generate_Ending/results.json +0 -9
  9. evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/es/Novel_Correct_Ending/results.json +0 -9
  10. evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/es/Story_Continuation_and_Options/results.json +0 -9
  11. evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/eu/Answer_Given_options/results.json +0 -9
  12. evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/eu/Choose_Story_Ending/results.json +0 -9
  13. evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/eu/Generate_Ending/results.json +0 -9
  14. evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/eu/Novel_Correct_Ending/results.json +0 -9
  15. evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/eu/Story_Continuation_and_Options/results.json +0 -9
  16. evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/hi/Answer_Given_options/results.json +0 -9
  17. evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/hi/Choose_Story_Ending/results.json +0 -9
  18. evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/hi/Generate_Ending/results.json +0 -9
  19. evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/hi/Novel_Correct_Ending/results.json +0 -9
  20. evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/hi/Story_Continuation_and_Options/results.json +0 -9
  21. evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/id/Answer_Given_options/results.json +0 -9
  22. evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/id/Choose_Story_Ending/results.json +0 -9
  23. evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/id/Generate_Ending/results.json +0 -9
  24. evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/id/Novel_Correct_Ending/results.json +0 -9
  25. evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/id/Story_Continuation_and_Options/results.json +0 -9
  26. evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/zh/Answer_Given_options/results.json +0 -9
  27. evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/zh/Choose_Story_Ending/results.json +0 -9
  28. evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/zh/Generate_Ending/results.json +0 -9
  29. evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/zh/Novel_Correct_Ending/results.json +0 -9
  30. evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/zh/Story_Continuation_and_Options/results.json +0 -9
  31. evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xwinograd/en/Replace/results.json +0 -9
  32. evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xwinograd/en/True_or_False/results.json +0 -9
  33. evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xwinograd/en/does_underscore_refer_to/results.json +0 -9
  34. evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xwinograd/en/stand_for/results.json +0 -9
  35. evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xwinograd/en/underscore_refer_to/results.json +0 -9
  36. evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xwinograd/fr/Replace/results.json +0 -9
  37. evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xwinograd/fr/True_or_False/results.json +0 -9
  38. evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xwinograd/fr/does_underscore_refer_to/results.json +0 -9
  39. evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xwinograd/fr/stand_for/results.json +0 -9
  40. evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xwinograd/fr/underscore_refer_to/results.json +0 -9
  41. evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xwinograd/pt/Replace/results.json +0 -9
  42. evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xwinograd/pt/True_or_False/results.json +0 -9
  43. evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xwinograd/pt/does_underscore_refer_to/results.json +0 -9
  44. evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xwinograd/pt/stand_for/results.json +0 -9
  45. evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xwinograd/pt/underscore_refer_to/results.json +0 -9
  46. evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xwinograd/zh/Replace/results.json +0 -9
  47. evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xwinograd/zh/True_or_False/results.json +0 -9
  48. evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xwinograd/zh/does_underscore_refer_to/results.json +0 -9
  49. evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xwinograd/zh/stand_for/results.json +0 -9
  50. evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xwinograd/zh/underscore_refer_to/results.json +0 -9
evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/ar/Answer_Given_options/results.json DELETED
@@ -1,9 +0,0 @@
1
- {
2
- "dataset_name": "Muennighoff/xstory_cloze",
3
- "dataset_config_name": "ar",
4
- "template_name": "Answer Given options",
5
- "evaluation": {
6
- "accuracy": 0.6896095301125083
7
- },
8
- "arguments": "Namespace(config_name=None, dataset_config_name='ar', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Answer Given options', tokenizer_name=None, use_slow_tokenizer=False)"
9
- }
 
 
 
 
 
 
 
 
 
 
evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/ar/Choose_Story_Ending/results.json DELETED
@@ -1,9 +0,0 @@
1
- {
2
- "dataset_name": "Muennighoff/xstory_cloze",
3
- "dataset_config_name": "ar",
4
- "template_name": "Choose Story Ending",
5
- "evaluation": {
6
- "accuracy": 0.8378557246856386
7
- },
8
- "arguments": "Namespace(config_name=None, dataset_config_name='ar', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Choose Story Ending', tokenizer_name=None, use_slow_tokenizer=False)"
9
- }
 
 
 
 
 
 
 
 
 
 
evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/ar/Generate_Ending/results.json DELETED
@@ -1,9 +0,0 @@
1
- {
2
- "dataset_name": "Muennighoff/xstory_cloze",
3
- "dataset_config_name": "ar",
4
- "template_name": "Generate Ending",
5
- "evaluation": {
6
- "accuracy": 0.5956320317670417
7
- },
8
- "arguments": "Namespace(config_name=None, dataset_config_name='ar', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Generate Ending', tokenizer_name=None, use_slow_tokenizer=False)"
9
- }
 
 
 
 
 
 
 
 
 
 
evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/ar/Novel_Correct_Ending/results.json DELETED
@@ -1,9 +0,0 @@
1
- {
2
- "dataset_name": "Muennighoff/xstory_cloze",
3
- "dataset_config_name": "ar",
4
- "template_name": "Novel Correct Ending",
5
- "evaluation": {
6
- "accuracy": 0.8213103904698875
7
- },
8
- "arguments": "Namespace(config_name=None, dataset_config_name='ar', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Novel Correct Ending', tokenizer_name=None, use_slow_tokenizer=False)"
9
- }
 
 
 
 
 
 
 
 
 
 
evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/ar/Story_Continuation_and_Options/results.json DELETED
@@ -1,9 +0,0 @@
1
- {
2
- "dataset_name": "Muennighoff/xstory_cloze",
3
- "dataset_config_name": "ar",
4
- "template_name": "Story Continuation and Options",
5
- "evaluation": {
6
- "accuracy": 0.8219722038385175
7
- },
8
- "arguments": "Namespace(config_name=None, dataset_config_name='ar', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Story Continuation and Options', tokenizer_name=None, use_slow_tokenizer=False)"
9
- }
 
 
 
 
 
 
 
 
 
 
evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/es/Answer_Given_options/results.json DELETED
@@ -1,9 +0,0 @@
1
- {
2
- "dataset_name": "Muennighoff/xstory_cloze",
3
- "dataset_config_name": "es",
4
- "template_name": "Answer Given options",
5
- "evaluation": {
6
- "accuracy": 0.7683653209794837
7
- },
8
- "arguments": "Namespace(config_name=None, dataset_config_name='es', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Answer Given options', tokenizer_name=None, use_slow_tokenizer=False)"
9
- }
 
 
 
 
 
 
 
 
 
 
evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/es/Choose_Story_Ending/results.json DELETED
@@ -1,9 +0,0 @@
1
- {
2
- "dataset_name": "Muennighoff/xstory_cloze",
3
- "dataset_config_name": "es",
4
- "template_name": "Choose Story Ending",
5
- "evaluation": {
6
- "accuracy": 0.886168100595632
7
- },
8
- "arguments": "Namespace(config_name=None, dataset_config_name='es', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Choose Story Ending', tokenizer_name=None, use_slow_tokenizer=False)"
9
- }
 
 
 
 
 
 
 
 
 
 
evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/es/Generate_Ending/results.json DELETED
@@ -1,9 +0,0 @@
1
- {
2
- "dataset_name": "Muennighoff/xstory_cloze",
3
- "dataset_config_name": "es",
4
- "template_name": "Generate Ending",
5
- "evaluation": {
6
- "accuracy": 0.6724023825281271
7
- },
8
- "arguments": "Namespace(config_name=None, dataset_config_name='es', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Generate Ending', tokenizer_name=None, use_slow_tokenizer=False)"
9
- }
 
 
 
 
 
 
 
 
 
 
evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/es/Novel_Correct_Ending/results.json DELETED
@@ -1,9 +0,0 @@
1
- {
2
- "dataset_name": "Muennighoff/xstory_cloze",
3
- "dataset_config_name": "es",
4
- "template_name": "Novel Correct Ending",
5
- "evaluation": {
6
- "accuracy": 0.8676373262739907
7
- },
8
- "arguments": "Namespace(config_name=None, dataset_config_name='es', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Novel Correct Ending', tokenizer_name=None, use_slow_tokenizer=False)"
9
- }
 
 
 
 
 
 
 
 
 
 
evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/es/Story_Continuation_and_Options/results.json DELETED
@@ -1,9 +0,0 @@
1
- {
2
- "dataset_name": "Muennighoff/xstory_cloze",
3
- "dataset_config_name": "es",
4
- "template_name": "Story Continuation and Options",
5
- "evaluation": {
6
- "accuracy": 0.8769027134348114
7
- },
8
- "arguments": "Namespace(config_name=None, dataset_config_name='es', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Story Continuation and Options', tokenizer_name=None, use_slow_tokenizer=False)"
9
- }
 
 
 
 
 
 
 
 
 
 
evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/eu/Answer_Given_options/results.json DELETED
@@ -1,9 +0,0 @@
1
- {
2
- "dataset_name": "Muennighoff/xstory_cloze",
3
- "dataset_config_name": "eu",
4
- "template_name": "Answer Given options",
5
- "evaluation": {
6
- "accuracy": 0.6082064857710126
7
- },
8
- "arguments": "Namespace(config_name=None, dataset_config_name='eu', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Answer Given options', tokenizer_name=None, use_slow_tokenizer=False)"
9
- }
 
 
 
 
 
 
 
 
 
 
evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/eu/Choose_Story_Ending/results.json DELETED
@@ -1,9 +0,0 @@
1
- {
2
- "dataset_name": "Muennighoff/xstory_cloze",
3
- "dataset_config_name": "eu",
4
- "template_name": "Choose Story Ending",
5
- "evaluation": {
6
- "accuracy": 0.7266710787557908
7
- },
8
- "arguments": "Namespace(config_name=None, dataset_config_name='eu', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Choose Story Ending', tokenizer_name=None, use_slow_tokenizer=False)"
9
- }
 
 
 
 
 
 
 
 
 
 
evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/eu/Generate_Ending/results.json DELETED
@@ -1,9 +0,0 @@
1
- {
2
- "dataset_name": "Muennighoff/xstory_cloze",
3
- "dataset_config_name": "eu",
4
- "template_name": "Generate Ending",
5
- "evaluation": {
6
- "accuracy": 0.5552614162806089
7
- },
8
- "arguments": "Namespace(config_name=None, dataset_config_name='eu', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Generate Ending', tokenizer_name=None, use_slow_tokenizer=False)"
9
- }
 
 
 
 
 
 
 
 
 
 
evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/eu/Novel_Correct_Ending/results.json DELETED
@@ -1,9 +0,0 @@
1
- {
2
- "dataset_name": "Muennighoff/xstory_cloze",
3
- "dataset_config_name": "eu",
4
- "template_name": "Novel Correct Ending",
5
- "evaluation": {
6
- "accuracy": 0.700198544010589
7
- },
8
- "arguments": "Namespace(config_name=None, dataset_config_name='eu', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Novel Correct Ending', tokenizer_name=None, use_slow_tokenizer=False)"
9
- }
 
 
 
 
 
 
 
 
 
 
evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/eu/Story_Continuation_and_Options/results.json DELETED
@@ -1,9 +0,0 @@
1
- {
2
- "dataset_name": "Muennighoff/xstory_cloze",
3
- "dataset_config_name": "eu",
4
- "template_name": "Story Continuation and Options",
5
- "evaluation": {
6
- "accuracy": 0.7107875579086698
7
- },
8
- "arguments": "Namespace(config_name=None, dataset_config_name='eu', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Story Continuation and Options', tokenizer_name=None, use_slow_tokenizer=False)"
9
- }
 
 
 
 
 
 
 
 
 
 
evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/hi/Answer_Given_options/results.json DELETED
@@ -1,9 +0,0 @@
1
- {
2
- "dataset_name": "Muennighoff/xstory_cloze",
3
- "dataset_config_name": "hi",
4
- "template_name": "Answer Given options",
5
- "evaluation": {
6
- "accuracy": 0.6366644606221046
7
- },
8
- "arguments": "Namespace(config_name=None, dataset_config_name='hi', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Answer Given options', tokenizer_name=None, use_slow_tokenizer=False)"
9
- }
 
 
 
 
 
 
 
 
 
 
evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/hi/Choose_Story_Ending/results.json DELETED
@@ -1,9 +0,0 @@
1
- {
2
- "dataset_name": "Muennighoff/xstory_cloze",
3
- "dataset_config_name": "hi",
4
- "template_name": "Choose Story Ending",
5
- "evaluation": {
6
- "accuracy": 0.7882197220383852
7
- },
8
- "arguments": "Namespace(config_name=None, dataset_config_name='hi', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Choose Story Ending', tokenizer_name=None, use_slow_tokenizer=False)"
9
- }
 
 
 
 
 
 
 
 
 
 
evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/hi/Generate_Ending/results.json DELETED
@@ -1,9 +0,0 @@
1
- {
2
- "dataset_name": "Muennighoff/xstory_cloze",
3
- "dataset_config_name": "hi",
4
- "template_name": "Generate Ending",
5
- "evaluation": {
6
- "accuracy": 0.5982792852415619
7
- },
8
- "arguments": "Namespace(config_name=None, dataset_config_name='hi', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Generate Ending', tokenizer_name=None, use_slow_tokenizer=False)"
9
- }
 
 
 
 
 
 
 
 
 
 
evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/hi/Novel_Correct_Ending/results.json DELETED
@@ -1,9 +0,0 @@
1
- {
2
- "dataset_name": "Muennighoff/xstory_cloze",
3
- "dataset_config_name": "hi",
4
- "template_name": "Novel Correct Ending",
5
- "evaluation": {
6
- "accuracy": 0.7485109199205824
7
- },
8
- "arguments": "Namespace(config_name=None, dataset_config_name='hi', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Novel Correct Ending', tokenizer_name=None, use_slow_tokenizer=False)"
9
- }
 
 
 
 
 
 
 
 
 
 
evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/hi/Story_Continuation_and_Options/results.json DELETED
@@ -1,9 +0,0 @@
1
- {
2
- "dataset_name": "Muennighoff/xstory_cloze",
3
- "dataset_config_name": "hi",
4
- "template_name": "Story Continuation and Options",
5
- "evaluation": {
6
- "accuracy": 0.7683653209794837
7
- },
8
- "arguments": "Namespace(config_name=None, dataset_config_name='hi', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Story Continuation and Options', tokenizer_name=None, use_slow_tokenizer=False)"
9
- }
 
 
 
 
 
 
 
 
 
 
evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/id/Answer_Given_options/results.json DELETED
@@ -1,9 +0,0 @@
1
- {
2
- "dataset_name": "Muennighoff/xstory_cloze",
3
- "dataset_config_name": "id",
4
- "template_name": "Answer Given options",
5
- "evaluation": {
6
- "accuracy": 0.7385837193911317
7
- },
8
- "arguments": "Namespace(config_name=None, dataset_config_name='id', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Answer Given options', tokenizer_name=None, use_slow_tokenizer=False)"
9
- }
 
 
 
 
 
 
 
 
 
 
evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/id/Choose_Story_Ending/results.json DELETED
@@ -1,9 +0,0 @@
1
- {
2
- "dataset_name": "Muennighoff/xstory_cloze",
3
- "dataset_config_name": "id",
4
- "template_name": "Choose Story Ending",
5
- "evaluation": {
6
- "accuracy": 0.8332230311052283
7
- },
8
- "arguments": "Namespace(config_name=None, dataset_config_name='id', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Choose Story Ending', tokenizer_name=None, use_slow_tokenizer=False)"
9
- }
 
 
 
 
 
 
 
 
 
 
evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/id/Generate_Ending/results.json DELETED
@@ -1,9 +0,0 @@
1
- {
2
- "dataset_name": "Muennighoff/xstory_cloze",
3
- "dataset_config_name": "id",
4
- "template_name": "Generate Ending",
5
- "evaluation": {
6
- "accuracy": 0.6293845135671741
7
- },
8
- "arguments": "Namespace(config_name=None, dataset_config_name='id', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Generate Ending', tokenizer_name=None, use_slow_tokenizer=False)"
9
- }
 
 
 
 
 
 
 
 
 
 
evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/id/Novel_Correct_Ending/results.json DELETED
@@ -1,9 +0,0 @@
1
- {
2
- "dataset_name": "Muennighoff/xstory_cloze",
3
- "dataset_config_name": "id",
4
- "template_name": "Novel Correct Ending",
5
- "evaluation": {
6
- "accuracy": 0.7816015883520847
7
- },
8
- "arguments": "Namespace(config_name=None, dataset_config_name='id', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Novel Correct Ending', tokenizer_name=None, use_slow_tokenizer=False)"
9
- }
 
 
 
 
 
 
 
 
 
 
evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/id/Story_Continuation_and_Options/results.json DELETED
@@ -1,9 +0,0 @@
1
- {
2
- "dataset_name": "Muennighoff/xstory_cloze",
3
- "dataset_config_name": "id",
4
- "template_name": "Story Continuation and Options",
5
- "evaluation": {
6
- "accuracy": 0.8226340172071476
7
- },
8
- "arguments": "Namespace(config_name=None, dataset_config_name='id', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Story Continuation and Options', tokenizer_name=None, use_slow_tokenizer=False)"
9
- }
 
 
 
 
 
 
 
 
 
 
evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/zh/Answer_Given_options/results.json DELETED
@@ -1,9 +0,0 @@
1
- {
2
- "dataset_name": "Muennighoff/xstory_cloze",
3
- "dataset_config_name": "zh",
4
- "template_name": "Answer Given options",
5
- "evaluation": {
6
- "accuracy": 0.7498345466578424
7
- },
8
- "arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Answer Given options', tokenizer_name=None, use_slow_tokenizer=False)"
9
- }
 
 
 
 
 
 
 
 
 
 
evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/zh/Choose_Story_Ending/results.json DELETED
@@ -1,9 +0,0 @@
1
- {
2
- "dataset_name": "Muennighoff/xstory_cloze",
3
- "dataset_config_name": "zh",
4
- "template_name": "Choose Story Ending",
5
- "evaluation": {
6
- "accuracy": 0.8583719391131701
7
- },
8
- "arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Choose Story Ending', tokenizer_name=None, use_slow_tokenizer=False)"
9
- }
 
 
 
 
 
 
 
 
 
 
evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/zh/Generate_Ending/results.json DELETED
@@ -1,9 +0,0 @@
1
- {
2
- "dataset_name": "Muennighoff/xstory_cloze",
3
- "dataset_config_name": "zh",
4
- "template_name": "Generate Ending",
5
- "evaluation": {
6
- "accuracy": 0.6227663798808736
7
- },
8
- "arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Generate Ending', tokenizer_name=None, use_slow_tokenizer=False)"
9
- }
 
 
 
 
 
 
 
 
 
 
evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/zh/Novel_Correct_Ending/results.json DELETED
@@ -1,9 +0,0 @@
1
- {
2
- "dataset_name": "Muennighoff/xstory_cloze",
3
- "dataset_config_name": "zh",
4
- "template_name": "Novel Correct Ending",
5
- "evaluation": {
6
- "accuracy": 0.8405029781601588
7
- },
8
- "arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Novel Correct Ending', tokenizer_name=None, use_slow_tokenizer=False)"
9
- }
 
 
 
 
 
 
 
 
 
 
evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xstory_cloze/zh/Story_Continuation_and_Options/results.json DELETED
@@ -1,9 +0,0 @@
1
- {
2
- "dataset_name": "Muennighoff/xstory_cloze",
3
- "dataset_config_name": "zh",
4
- "template_name": "Story Continuation and Options",
5
- "evaluation": {
6
- "accuracy": 0.8385175380542687
7
- },
8
- "arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Story Continuation and Options', tokenizer_name=None, use_slow_tokenizer=False)"
9
- }
 
 
 
 
 
 
 
 
 
 
evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xwinograd/en/Replace/results.json DELETED
@@ -1,9 +0,0 @@
1
- {
2
- "dataset_name": "Muennighoff/xwinograd",
3
- "dataset_config_name": "en",
4
- "template_name": "Replace",
5
- "evaluation": {
6
- "accuracy": 0.6576344086021505
7
- },
8
- "arguments": "Namespace(config_name=None, dataset_config_name='en', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='Replace', tokenizer_name=None, use_slow_tokenizer=False)"
9
- }
 
 
 
 
 
 
 
 
 
 
evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xwinograd/en/True_or_False/results.json DELETED
@@ -1,9 +0,0 @@
1
- {
2
- "dataset_name": "Muennighoff/xwinograd",
3
- "dataset_config_name": "en",
4
- "template_name": "True or False",
5
- "evaluation": {
6
- "accuracy": 0.5187096774193548
7
- },
8
- "arguments": "Namespace(config_name=None, dataset_config_name='en', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='True or False', tokenizer_name=None, use_slow_tokenizer=False)"
9
- }
 
 
 
 
 
 
 
 
 
 
evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xwinograd/en/does_underscore_refer_to/results.json DELETED
@@ -1,9 +0,0 @@
1
- {
2
- "dataset_name": "Muennighoff/xwinograd",
3
- "dataset_config_name": "en",
4
- "template_name": "does underscore refer to",
5
- "evaluation": {
6
- "accuracy": 0.5931182795698925
7
- },
8
- "arguments": "Namespace(config_name=None, dataset_config_name='en', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='does underscore refer to', tokenizer_name=None, use_slow_tokenizer=False)"
9
- }
 
 
 
 
 
 
 
 
 
 
evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xwinograd/en/stand_for/results.json DELETED
@@ -1,9 +0,0 @@
1
- {
2
- "dataset_name": "Muennighoff/xwinograd",
3
- "dataset_config_name": "en",
4
- "template_name": "stand for",
5
- "evaluation": {
6
- "accuracy": 0.5070967741935484
7
- },
8
- "arguments": "Namespace(config_name=None, dataset_config_name='en', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='stand for', tokenizer_name=None, use_slow_tokenizer=False)"
9
- }
 
 
 
 
 
 
 
 
 
 
evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xwinograd/en/underscore_refer_to/results.json DELETED
@@ -1,9 +0,0 @@
1
- {
2
- "dataset_name": "Muennighoff/xwinograd",
3
- "dataset_config_name": "en",
4
- "template_name": "underscore refer to",
5
- "evaluation": {
6
- "accuracy": 0.6210752688172043
7
- },
8
- "arguments": "Namespace(config_name=None, dataset_config_name='en', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='underscore refer to', tokenizer_name=None, use_slow_tokenizer=False)"
9
- }
 
 
 
 
 
 
 
 
 
 
evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xwinograd/fr/Replace/results.json DELETED
@@ -1,9 +0,0 @@
1
- {
2
- "dataset_name": "Muennighoff/xwinograd",
3
- "dataset_config_name": "fr",
4
- "template_name": "Replace",
5
- "evaluation": {
6
- "accuracy": 0.5180722891566265
7
- },
8
- "arguments": "Namespace(config_name=None, dataset_config_name='fr', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='Replace', tokenizer_name=None, use_slow_tokenizer=False)"
9
- }
 
 
 
 
 
 
 
 
 
 
evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xwinograd/fr/True_or_False/results.json DELETED
@@ -1,9 +0,0 @@
1
- {
2
- "dataset_name": "Muennighoff/xwinograd",
3
- "dataset_config_name": "fr",
4
- "template_name": "True or False",
5
- "evaluation": {
6
- "accuracy": 0.5301204819277109
7
- },
8
- "arguments": "Namespace(config_name=None, dataset_config_name='fr', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='True or False', tokenizer_name=None, use_slow_tokenizer=False)"
9
- }
 
 
 
 
 
 
 
 
 
 
evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xwinograd/fr/does_underscore_refer_to/results.json DELETED
@@ -1,9 +0,0 @@
1
- {
2
- "dataset_name": "Muennighoff/xwinograd",
3
- "dataset_config_name": "fr",
4
- "template_name": "does underscore refer to",
5
- "evaluation": {
6
- "accuracy": 0.5542168674698795
7
- },
8
- "arguments": "Namespace(config_name=None, dataset_config_name='fr', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='does underscore refer to', tokenizer_name=None, use_slow_tokenizer=False)"
9
- }
 
 
 
 
 
 
 
 
 
 
evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xwinograd/fr/stand_for/results.json DELETED
@@ -1,9 +0,0 @@
1
- {
2
- "dataset_name": "Muennighoff/xwinograd",
3
- "dataset_config_name": "fr",
4
- "template_name": "stand for",
5
- "evaluation": {
6
- "accuracy": 0.5180722891566265
7
- },
8
- "arguments": "Namespace(config_name=None, dataset_config_name='fr', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='stand for', tokenizer_name=None, use_slow_tokenizer=False)"
9
- }
 
 
 
 
 
 
 
 
 
 
evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xwinograd/fr/underscore_refer_to/results.json DELETED
@@ -1,9 +0,0 @@
1
- {
2
- "dataset_name": "Muennighoff/xwinograd",
3
- "dataset_config_name": "fr",
4
- "template_name": "underscore refer to",
5
- "evaluation": {
6
- "accuracy": 0.5421686746987951
7
- },
8
- "arguments": "Namespace(config_name=None, dataset_config_name='fr', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='underscore refer to', tokenizer_name=None, use_slow_tokenizer=False)"
9
- }
 
 
 
 
 
 
 
 
 
 
evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xwinograd/pt/Replace/results.json DELETED
@@ -1,9 +0,0 @@
1
- {
2
- "dataset_name": "Muennighoff/xwinograd",
3
- "dataset_config_name": "pt",
4
- "template_name": "Replace",
5
- "evaluation": {
6
- "accuracy": 0.5741444866920152
7
- },
8
- "arguments": "Namespace(config_name=None, dataset_config_name='pt', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='Replace', tokenizer_name=None, use_slow_tokenizer=False)"
9
- }
 
 
 
 
 
 
 
 
 
 
evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xwinograd/pt/True_or_False/results.json DELETED
@@ -1,9 +0,0 @@
1
- {
2
- "dataset_name": "Muennighoff/xwinograd",
3
- "dataset_config_name": "pt",
4
- "template_name": "True or False",
5
- "evaluation": {
6
- "accuracy": 0.4790874524714829
7
- },
8
- "arguments": "Namespace(config_name=None, dataset_config_name='pt', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='True or False', tokenizer_name=None, use_slow_tokenizer=False)"
9
- }
 
 
 
 
 
 
 
 
 
 
evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xwinograd/pt/does_underscore_refer_to/results.json DELETED
@@ -1,9 +0,0 @@
1
- {
2
- "dataset_name": "Muennighoff/xwinograd",
3
- "dataset_config_name": "pt",
4
- "template_name": "does underscore refer to",
5
- "evaluation": {
6
- "accuracy": 0.55893536121673
7
- },
8
- "arguments": "Namespace(config_name=None, dataset_config_name='pt', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='does underscore refer to', tokenizer_name=None, use_slow_tokenizer=False)"
9
- }
 
 
 
 
 
 
 
 
 
 
evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xwinograd/pt/stand_for/results.json DELETED
@@ -1,9 +0,0 @@
1
- {
2
- "dataset_name": "Muennighoff/xwinograd",
3
- "dataset_config_name": "pt",
4
- "template_name": "stand for",
5
- "evaluation": {
6
- "accuracy": 0.5209125475285171
7
- },
8
- "arguments": "Namespace(config_name=None, dataset_config_name='pt', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='stand for', tokenizer_name=None, use_slow_tokenizer=False)"
9
- }
 
 
 
 
 
 
 
 
 
 
evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xwinograd/pt/underscore_refer_to/results.json DELETED
@@ -1,9 +0,0 @@
1
- {
2
- "dataset_name": "Muennighoff/xwinograd",
3
- "dataset_config_name": "pt",
4
- "template_name": "underscore refer to",
5
- "evaluation": {
6
- "accuracy": 0.5437262357414449
7
- },
8
- "arguments": "Namespace(config_name=None, dataset_config_name='pt', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='underscore refer to', tokenizer_name=None, use_slow_tokenizer=False)"
9
- }
 
 
 
 
 
 
 
 
 
 
evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xwinograd/zh/Replace/results.json DELETED
@@ -1,9 +0,0 @@
1
- {
2
- "dataset_name": "Muennighoff/xwinograd",
3
- "dataset_config_name": "zh",
4
- "template_name": "Replace",
5
- "evaluation": {
6
- "accuracy": 0.626984126984127
7
- },
8
- "arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='Replace', tokenizer_name=None, use_slow_tokenizer=False)"
9
- }
 
 
 
 
 
 
 
 
 
 
evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xwinograd/zh/True_or_False/results.json DELETED
@@ -1,9 +0,0 @@
1
- {
2
- "dataset_name": "Muennighoff/xwinograd",
3
- "dataset_config_name": "zh",
4
- "template_name": "True or False",
5
- "evaluation": {
6
- "accuracy": 0.503968253968254
7
- },
8
- "arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='True or False', tokenizer_name=None, use_slow_tokenizer=False)"
9
- }
 
 
 
 
 
 
 
 
 
 
evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xwinograd/zh/does_underscore_refer_to/results.json DELETED
@@ -1,9 +0,0 @@
1
- {
2
- "dataset_name": "Muennighoff/xwinograd",
3
- "dataset_config_name": "zh",
4
- "template_name": "does underscore refer to",
5
- "evaluation": {
6
- "accuracy": 0.5436507936507936
7
- },
8
- "arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='does underscore refer to', tokenizer_name=None, use_slow_tokenizer=False)"
9
- }
 
 
 
 
 
 
 
 
 
 
evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xwinograd/zh/stand_for/results.json DELETED
@@ -1,9 +0,0 @@
1
- {
2
- "dataset_name": "Muennighoff/xwinograd",
3
- "dataset_config_name": "zh",
4
- "template_name": "stand for",
5
- "evaluation": {
6
- "accuracy": 0.49007936507936506
7
- },
8
- "arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='stand for', tokenizer_name=None, use_slow_tokenizer=False)"
9
- }
 
 
 
 
 
 
 
 
 
 
evaluation_bloommz-7b1/evaluation_l1/Muennighoff_xwinograd/zh/underscore_refer_to/results.json DELETED
@@ -1,9 +0,0 @@
1
- {
2
- "dataset_name": "Muennighoff/xwinograd",
3
- "dataset_config_name": "zh",
4
- "template_name": "underscore refer to",
5
- "evaluation": {
6
- "accuracy": 0.5535714285714286
7
- },
8
- "arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/6b3t0/tr13f-6b3-ml-t0-lmtoks341b-t0toks4b-xp3mt/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='underscore refer to', tokenizer_name=None, use_slow_tokenizer=False)"
9
- }