Muennighoff commited on
Commit
1e1eb0a
·
1 Parent(s): 08eb31b
This view is limited to 50 files because it contains too many changes.   See raw diff
Files changed (50) hide show
  1. evaluation_l1/Muennighoff_xstory_cloze/ar/Choose_Story_Ending/results.json +9 -0
  2. evaluation_l1/Muennighoff_xstory_cloze/ar/Generate_Ending/results.json +9 -0
  3. evaluation_l1/Muennighoff_xstory_cloze/es/Answer_Given_options/results.json +9 -0
  4. evaluation_l1/Muennighoff_xstory_cloze/es/Choose_Story_Ending/results.json +9 -0
  5. evaluation_l1/Muennighoff_xstory_cloze/es/Generate_Ending/results.json +9 -0
  6. evaluation_l1/Muennighoff_xstory_cloze/es/Novel_Correct_Ending/results.json +9 -0
  7. evaluation_l1/Muennighoff_xstory_cloze/es/Story_Continuation_and_Options/results.json +9 -0
  8. evaluation_l1/Muennighoff_xstory_cloze/eu/Story_Continuation_and_Options/results.json +9 -0
  9. evaluation_l1/Muennighoff_xwinograd/fr/underscore_refer_to/results.json +9 -0
  10. evaluation_l1/Muennighoff_xwinograd/zh/True_or_False/results.json +9 -0
  11. evaluation_l1/Muennighoff_xwinograd/zh/does_underscore_refer_to/results.json +9 -0
  12. evaluation_l1/Muennighoff_xwinograd/zh/stand_for/results.json +9 -0
  13. evaluation_l1/xnli/sw/can_we_infer/results.json +9 -0
  14. evaluation_l1/xnli/ur/GPT-3_style/results.json +9 -0
  15. evaluation_l1/xnli/ur/MNLI_crowdsource/results.json +9 -0
  16. evaluation_l1/xnli/ur/can_we_infer/results.json +9 -0
  17. evaluation_l1/xnli/ur/guaranteed_possible_impossible/results.json +9 -0
  18. evaluation_l1/xnli/ur/justified_in_saying/results.json +9 -0
  19. evaluation_l2/Muennighoff_xwinograd/jp/underscore_refer_to/results.json +9 -0
  20. evaluation_l2/xnli/tr/can_we_infer/results.json +9 -0
  21. evaluation_l2/xnli/tr/justified_in_saying/results.json +9 -0
  22. evaluation_xnlimtht/xnli/es/can_we_infer_esmt/results.json +9 -0
  23. evaluation_xnlimtht/xnli/fr/GPT-3_style_frht/results.json +9 -0
  24. evaluation_xnlimtht/xnli/fr/MNLI_crowdsource_frht/results.json +9 -0
  25. evaluation_xnlimtht/xnli/fr/MNLI_crowdsource_frmt/results.json +9 -0
  26. evaluation_xnlimtht/xnli/fr/can_we_infer_frht/results.json +9 -0
  27. evaluation_xnlimtht/xnli/fr/guaranteed_possible_impossible_frht/results.json +9 -0
  28. evaluation_xnlimtht/xnli/fr/justified_in_saying_frht/results.json +9 -0
  29. evaluation_xnlimtht/xnli/zh/can_we_infer_zhmt/results.json +9 -0
  30. evaluation_xnlimtht/xnli/zh/justified_in_saying_zhmt/results.json +9 -0
  31. {evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/ar/Answer_Given_options_armt/results.json +0 -0
  32. {evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/ar/Choose_Story_Ending_armt/results.json +0 -0
  33. {evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/ar/Generate_Ending_armt/results.json +0 -0
  34. {evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/ar/Novel_Correct_Ending_armt/results.json +0 -0
  35. {evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/ar/Story_Continuation_and_Options_armt/results.json +0 -0
  36. {evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/es/Answer_Given_options_esmt/results.json +0 -0
  37. {evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/es/Choose_Story_Ending_esmt/results.json +0 -0
  38. {evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/es/Generate_Ending_esmt/results.json +0 -0
  39. {evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/es/Novel_Correct_Ending_esmt/results.json +0 -0
  40. {evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/es/Story_Continuation_and_Options_esmt/results.json +0 -0
  41. {evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/eu/Answer_Given_options_eumt/results.json +0 -0
  42. {evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/eu/Choose_Story_Ending_eumt/results.json +0 -0
  43. {evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/eu/Generate_Ending_eumt/results.json +0 -0
  44. {evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/eu/Novel_Correct_Ending_eumt/results.json +0 -0
  45. {evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/eu/Story_Continuation_and_Options_eumt/results.json +0 -0
  46. {evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/hi/Answer_Given_options_himt/results.json +0 -0
  47. {evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/hi/Choose_Story_Ending_himt/results.json +0 -0
  48. {evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/hi/Generate_Ending_himt/results.json +0 -0
  49. {evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/hi/Novel_Correct_Ending_himt/results.json +0 -0
  50. {evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/hi/Story_Continuation_and_Options_himt/results.json +0 -0
evaluation_l1/Muennighoff_xstory_cloze/ar/Choose_Story_Ending/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xstory_cloze",
3
+ "dataset_config_name": "ar",
4
+ "template_name": "Choose Story Ending",
5
+ "evaluation": {
6
+ "accuracy": 0.8060886829913965
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='ar', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Choose Story Ending', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_l1/Muennighoff_xstory_cloze/ar/Generate_Ending/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xstory_cloze",
3
+ "dataset_config_name": "ar",
4
+ "template_name": "Generate Ending",
5
+ "evaluation": {
6
+ "accuracy": 0.5684976836532097
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='ar', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Generate Ending', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_l1/Muennighoff_xstory_cloze/es/Answer_Given_options/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xstory_cloze",
3
+ "dataset_config_name": "es",
4
+ "template_name": "Answer Given options",
5
+ "evaluation": {
6
+ "accuracy": 0.7498345466578424
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='es', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Answer Given options', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_l1/Muennighoff_xstory_cloze/es/Choose_Story_Ending/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xstory_cloze",
3
+ "dataset_config_name": "es",
4
+ "template_name": "Choose Story Ending",
5
+ "evaluation": {
6
+ "accuracy": 0.8590337524818001
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='es', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Choose Story Ending', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_l1/Muennighoff_xstory_cloze/es/Generate_Ending/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xstory_cloze",
3
+ "dataset_config_name": "es",
4
+ "template_name": "Generate Ending",
5
+ "evaluation": {
6
+ "accuracy": 0.6260754467240238
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='es', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Generate Ending', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_l1/Muennighoff_xstory_cloze/es/Novel_Correct_Ending/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xstory_cloze",
3
+ "dataset_config_name": "es",
4
+ "template_name": "Novel Correct Ending",
5
+ "evaluation": {
6
+ "accuracy": 0.8166776968894772
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='es', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Novel Correct Ending', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_l1/Muennighoff_xstory_cloze/es/Story_Continuation_and_Options/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xstory_cloze",
3
+ "dataset_config_name": "es",
4
+ "template_name": "Story Continuation and Options",
5
+ "evaluation": {
6
+ "accuracy": 0.8352084712111185
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='es', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Story Continuation and Options', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_l1/Muennighoff_xstory_cloze/eu/Story_Continuation_and_Options/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xstory_cloze",
3
+ "dataset_config_name": "eu",
4
+ "template_name": "Story Continuation and Options",
5
+ "evaluation": {
6
+ "accuracy": 0.7074784910655195
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='eu', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Story Continuation and Options', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_l1/Muennighoff_xwinograd/fr/underscore_refer_to/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xwinograd",
3
+ "dataset_config_name": "fr",
4
+ "template_name": "underscore refer to",
5
+ "evaluation": {
6
+ "accuracy": 0.5301204819277109
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='fr', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='underscore refer to', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_l1/Muennighoff_xwinograd/zh/True_or_False/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xwinograd",
3
+ "dataset_config_name": "zh",
4
+ "template_name": "True or False",
5
+ "evaluation": {
6
+ "accuracy": 0.5515873015873016
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='True or False', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_l1/Muennighoff_xwinograd/zh/does_underscore_refer_to/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xwinograd",
3
+ "dataset_config_name": "zh",
4
+ "template_name": "does underscore refer to",
5
+ "evaluation": {
6
+ "accuracy": 0.5357142857142857
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='does underscore refer to', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_l1/Muennighoff_xwinograd/zh/stand_for/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xwinograd",
3
+ "dataset_config_name": "zh",
4
+ "template_name": "stand for",
5
+ "evaluation": {
6
+ "accuracy": 0.5238095238095238
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='stand for', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_l1/xnli/sw/can_we_infer/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "xnli",
3
+ "dataset_config_name": "sw",
4
+ "template_name": "can we infer",
5
+ "evaluation": {
6
+ "accuracy": 0.41767068273092367
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='sw', dataset_name='xnli', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='can we infer', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_l1/xnli/ur/GPT-3_style/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "xnli",
3
+ "dataset_config_name": "ur",
4
+ "template_name": "GPT-3 style",
5
+ "evaluation": {
6
+ "accuracy": 0.4903614457831325
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='ur', dataset_name='xnli', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='GPT-3 style', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_l1/xnli/ur/MNLI_crowdsource/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "xnli",
3
+ "dataset_config_name": "ur",
4
+ "template_name": "MNLI crowdsource",
5
+ "evaluation": {
6
+ "accuracy": 0.36666666666666664
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='ur', dataset_name='xnli', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='MNLI crowdsource', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_l1/xnli/ur/can_we_infer/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "xnli",
3
+ "dataset_config_name": "ur",
4
+ "template_name": "can we infer",
5
+ "evaluation": {
6
+ "accuracy": 0.43654618473895584
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='ur', dataset_name='xnli', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='can we infer', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_l1/xnli/ur/guaranteed_possible_impossible/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "xnli",
3
+ "dataset_config_name": "ur",
4
+ "template_name": "guaranteed/possible/impossible",
5
+ "evaluation": {
6
+ "accuracy": 0.3345381526104418
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='ur', dataset_name='xnli', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='guaranteed/possible/impossible', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_l1/xnli/ur/justified_in_saying/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "xnli",
3
+ "dataset_config_name": "ur",
4
+ "template_name": "justified in saying",
5
+ "evaluation": {
6
+ "accuracy": 0.42891566265060244
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='ur', dataset_name='xnli', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='justified in saying', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_l2/Muennighoff_xwinograd/jp/underscore_refer_to/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "Muennighoff/xwinograd",
3
+ "dataset_config_name": "jp",
4
+ "template_name": "underscore refer to",
5
+ "evaluation": {
6
+ "accuracy": 0.4848800834202294
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='jp', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='underscore refer to', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_l2/xnli/tr/can_we_infer/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "xnli",
3
+ "dataset_config_name": "tr",
4
+ "template_name": "can we infer",
5
+ "evaluation": {
6
+ "accuracy": 0.3530120481927711
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='tr', dataset_name='xnli', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='can we infer', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_l2/xnli/tr/justified_in_saying/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "xnli",
3
+ "dataset_config_name": "tr",
4
+ "template_name": "justified in saying",
5
+ "evaluation": {
6
+ "accuracy": 0.36265060240963853
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='tr', dataset_name='xnli', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='justified in saying', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_xnlimtht/xnli/es/can_we_infer_esmt/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "xnli",
3
+ "dataset_config_name": "es",
4
+ "template_name": "can we infer_esmt",
5
+ "evaluation": {
6
+ "accuracy": 0.3333333333333333
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='es', dataset_name='xnli', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='es', template_name='can we infer_esmt', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_xnlimtht/xnli/fr/GPT-3_style_frht/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "xnli",
3
+ "dataset_config_name": "fr",
4
+ "template_name": "GPT-3 style_frht",
5
+ "evaluation": {
6
+ "accuracy": 0.3405622489959839
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='fr', dataset_name='xnli', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='fr', template_name='GPT-3 style_frht', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_xnlimtht/xnli/fr/MNLI_crowdsource_frht/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "xnli",
3
+ "dataset_config_name": "fr",
4
+ "template_name": "MNLI crowdsource_frht",
5
+ "evaluation": {
6
+ "accuracy": 0.344578313253012
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='fr', dataset_name='xnli', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='fr', template_name='MNLI crowdsource_frht', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_xnlimtht/xnli/fr/MNLI_crowdsource_frmt/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "xnli",
3
+ "dataset_config_name": "fr",
4
+ "template_name": "MNLI crowdsource_frmt",
5
+ "evaluation": {
6
+ "accuracy": 0.3333333333333333
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='fr', dataset_name='xnli', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='fr', template_name='MNLI crowdsource_frmt', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_xnlimtht/xnli/fr/can_we_infer_frht/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "xnli",
3
+ "dataset_config_name": "fr",
4
+ "template_name": "can we infer_frht",
5
+ "evaluation": {
6
+ "accuracy": 0.5188755020080321
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='fr', dataset_name='xnli', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='fr', template_name='can we infer_frht', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_xnlimtht/xnli/fr/guaranteed_possible_impossible_frht/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "xnli",
3
+ "dataset_config_name": "fr",
4
+ "template_name": "guaranteed/possible/impossible_frht",
5
+ "evaluation": {
6
+ "accuracy": 0.3457831325301205
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='fr', dataset_name='xnli', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='fr', template_name='guaranteed/possible/impossible_frht', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_xnlimtht/xnli/fr/justified_in_saying_frht/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "xnli",
3
+ "dataset_config_name": "fr",
4
+ "template_name": "justified in saying_frht",
5
+ "evaluation": {
6
+ "accuracy": 0.4843373493975904
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='fr', dataset_name='xnli', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='fr', template_name='justified in saying_frht', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_xnlimtht/xnli/zh/can_we_infer_zhmt/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "xnli",
3
+ "dataset_config_name": "zh",
4
+ "template_name": "can we infer_zhmt",
5
+ "evaluation": {
6
+ "accuracy": 0.3405622489959839
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='xnli', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='zh', template_name='can we infer_zhmt', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
evaluation_xnlimtht/xnli/zh/justified_in_saying_zhmt/results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "xnli",
3
+ "dataset_config_name": "zh",
4
+ "template_name": "justified in saying_zhmt",
5
+ "evaluation": {
6
+ "accuracy": 0.334136546184739
7
+ },
8
+ "arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='xnli', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/2b5t0/bloomz-3b/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='zh', template_name='justified in saying_zhmt', tokenizer_name=None, use_slow_tokenizer=False)"
9
+ }
{evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/ar/Answer_Given_options_armt/results.json RENAMED
File without changes
{evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/ar/Choose_Story_Ending_armt/results.json RENAMED
File without changes
{evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/ar/Generate_Ending_armt/results.json RENAMED
File without changes
{evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/ar/Novel_Correct_Ending_armt/results.json RENAMED
File without changes
{evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/ar/Story_Continuation_and_Options_armt/results.json RENAMED
File without changes
{evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/es/Answer_Given_options_esmt/results.json RENAMED
File without changes
{evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/es/Choose_Story_Ending_esmt/results.json RENAMED
File without changes
{evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/es/Generate_Ending_esmt/results.json RENAMED
File without changes
{evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/es/Novel_Correct_Ending_esmt/results.json RENAMED
File without changes
{evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/es/Story_Continuation_and_Options_esmt/results.json RENAMED
File without changes
{evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/eu/Answer_Given_options_eumt/results.json RENAMED
File without changes
{evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/eu/Choose_Story_Ending_eumt/results.json RENAMED
File without changes
{evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/eu/Generate_Ending_eumt/results.json RENAMED
File without changes
{evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/eu/Novel_Correct_Ending_eumt/results.json RENAMED
File without changes
{evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/eu/Story_Continuation_and_Options_eumt/results.json RENAMED
File without changes
{evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/hi/Answer_Given_options_himt/results.json RENAMED
File without changes
{evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/hi/Choose_Story_Ending_himt/results.json RENAMED
File without changes
{evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/hi/Generate_Ending_himt/results.json RENAMED
File without changes
{evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/hi/Novel_Correct_Ending_himt/results.json RENAMED
File without changes
{evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/hi/Story_Continuation_and_Options_himt/results.json RENAMED
File without changes