Muennighoff
commited on
Commit
•
efac57d
1
Parent(s):
f7d5648
Add eval
Browse filesThis view is limited to 50 files because it contains too many changes.
See raw diff
- evaluation_l1/Muennighoff_xstory_cloze/hi/Choose_Story_Ending/results.json +9 -0
- evaluation_l1/Muennighoff_xstory_cloze/hi/Generate_Ending/results.json +9 -0
- evaluation_l1/Muennighoff_xstory_cloze/hi/Novel_Correct_Ending/results.json +9 -0
- evaluation_l1/Muennighoff_xstory_cloze/zh/Answer_Given_options/results.json +9 -0
- evaluation_l1/Muennighoff_xstory_cloze/zh/Choose_Story_Ending/results.json +9 -0
- evaluation_l1/Muennighoff_xstory_cloze/zh/Generate_Ending/results.json +9 -0
- evaluation_l1/Muennighoff_xstory_cloze/zh/Novel_Correct_Ending/results.json +9 -0
- evaluation_l1/Muennighoff_xstory_cloze/zh/Story_Continuation_and_Options/results.json +9 -0
- evaluation_l1/Muennighoff_xwinograd/en/Replace/results.json +9 -0
- evaluation_l1/Muennighoff_xwinograd/en/underscore_refer_to/results.json +9 -0
- evaluation_l1/Muennighoff_xwinograd/pt/True_or_False/results.json +9 -0
- evaluation_l1/Muennighoff_xwinograd/zh/Replace/results.json +9 -0
- evaluation_l1/Muennighoff_xwinograd/zh/stand_for/results.json +9 -0
- evaluation_l1/Muennighoff_xwinograd/zh/underscore_refer_to/results.json +9 -0
- evaluation_l1/xcopa/id/cause_effect/results.json +9 -0
- evaluation_l1/xcopa/id/plausible_alternatives/results.json +9 -0
- evaluation_l1/xcopa/zh/plausible_alternatives/results.json +9 -0
- evaluation_l1/xnli/en/can_we_infer/results.json +9 -0
- evaluation_l1/xnli/es/GPT-3_style/results.json +9 -0
- evaluation_l1/xnli/es/MNLI_crowdsource/results.json +9 -0
- evaluation_l1/xnli/es/guaranteed_possible_impossible/results.json +9 -0
- evaluation_l1/xnli/es/justified_in_saying/results.json +9 -0
- evaluation_l2/xnli/th/can_we_infer/results.json +9 -0
- evaluation_l2/xnli/tr/GPT-3_style/results.json +9 -0
- evaluation_l2/xnli/tr/MNLI_crowdsource/results.json +9 -0
- evaluation_l2/xnli/tr/can_we_infer/results.json +9 -0
- evaluation_l2/xnli/tr/guaranteed_possible_impossible/results.json +9 -0
- evaluation_l2/xnli/tr/justified_in_saying/results.json +9 -0
- evaluation_xnlimtht/xnli/ar/GPT-3_style_arht/results.json +9 -0
- evaluation_xnlimtht/xnli/ar/justified_in_saying_arht/results.json +9 -0
- evaluation_xnlimtht/xnli/ur/GPT-3_style_urht/results.json +9 -0
- evaluation_xnlimtht/xnli/ur/MNLI_crowdsource_urht/results.json +9 -0
- evaluation_xnlimtht/xnli/ur/can_we_infer_urht/results.json +9 -0
- evaluation_xnlimtht/xnli/ur/guaranteed_possible_impossible_urht/results.json +9 -0
- evaluation_xnlimtht/xnli/ur/justified_in_saying_urht/results.json +9 -0
- evaluation_xnlimtht/xnli/vi/GPT-3_style_viht/results.json +9 -0
- evaluation_xnlimtht/xnli/vi/MNLI_crowdsource_viht/results.json +9 -0
- {evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/ar/Answer_Given_options_armt/results.json +0 -0
- {evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/ar/Choose_Story_Ending_armt/results.json +0 -0
- {evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/ar/Generate_Ending_armt/results.json +0 -0
- {evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/ar/Novel_Correct_Ending_armt/results.json +0 -0
- {evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/ar/Story_Continuation_and_Options_armt/results.json +0 -0
- {evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/es/Answer_Given_options_esmt/results.json +0 -0
- {evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/es/Choose_Story_Ending_esmt/results.json +0 -0
- {evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/es/Generate_Ending_esmt/results.json +0 -0
- {evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/es/Novel_Correct_Ending_esmt/results.json +0 -0
- {evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/es/Story_Continuation_and_Options_esmt/results.json +0 -0
- {evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/eu/Answer_Given_options_eumt/results.json +0 -0
- {evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/eu/Choose_Story_Ending_eumt/results.json +0 -0
- {evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/eu/Generate_Ending_eumt/results.json +0 -0
evaluation_l1/Muennighoff_xstory_cloze/hi/Choose_Story_Ending/results.json
ADDED
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"dataset_name": "Muennighoff/xstory_cloze",
|
3 |
+
"dataset_config_name": "hi",
|
4 |
+
"template_name": "Choose Story Ending",
|
5 |
+
"evaluation": {
|
6 |
+
"accuracy": 0.5274652547981469
|
7 |
+
},
|
8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='hi', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Choose Story Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
+
}
|
evaluation_l1/Muennighoff_xstory_cloze/hi/Generate_Ending/results.json
ADDED
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"dataset_name": "Muennighoff/xstory_cloze",
|
3 |
+
"dataset_config_name": "hi",
|
4 |
+
"template_name": "Generate Ending",
|
5 |
+
"evaluation": {
|
6 |
+
"accuracy": 0.5519523494374586
|
7 |
+
},
|
8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='hi', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Generate Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
+
}
|
evaluation_l1/Muennighoff_xstory_cloze/hi/Novel_Correct_Ending/results.json
ADDED
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"dataset_name": "Muennighoff/xstory_cloze",
|
3 |
+
"dataset_config_name": "hi",
|
4 |
+
"template_name": "Novel Correct Ending",
|
5 |
+
"evaluation": {
|
6 |
+
"accuracy": 0.5016545334215751
|
7 |
+
},
|
8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='hi', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Novel Correct Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
+
}
|
evaluation_l1/Muennighoff_xstory_cloze/zh/Answer_Given_options/results.json
ADDED
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"dataset_name": "Muennighoff/xstory_cloze",
|
3 |
+
"dataset_config_name": "zh",
|
4 |
+
"template_name": "Answer Given options",
|
5 |
+
"evaluation": {
|
6 |
+
"accuracy": 0.5367306419589676
|
7 |
+
},
|
8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Answer Given options', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
+
}
|
evaluation_l1/Muennighoff_xstory_cloze/zh/Choose_Story_Ending/results.json
ADDED
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"dataset_name": "Muennighoff/xstory_cloze",
|
3 |
+
"dataset_config_name": "zh",
|
4 |
+
"template_name": "Choose Story Ending",
|
5 |
+
"evaluation": {
|
6 |
+
"accuracy": 0.5817339510258107
|
7 |
+
},
|
8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Choose Story Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
+
}
|
evaluation_l1/Muennighoff_xstory_cloze/zh/Generate_Ending/results.json
ADDED
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"dataset_name": "Muennighoff/xstory_cloze",
|
3 |
+
"dataset_config_name": "zh",
|
4 |
+
"template_name": "Generate Ending",
|
5 |
+
"evaluation": {
|
6 |
+
"accuracy": 0.5671740569159497
|
7 |
+
},
|
8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Generate Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
+
}
|
evaluation_l1/Muennighoff_xstory_cloze/zh/Novel_Correct_Ending/results.json
ADDED
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"dataset_name": "Muennighoff/xstory_cloze",
|
3 |
+
"dataset_config_name": "zh",
|
4 |
+
"template_name": "Novel Correct Ending",
|
5 |
+
"evaluation": {
|
6 |
+
"accuracy": 0.57180675049636
|
7 |
+
},
|
8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Novel Correct Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
+
}
|
evaluation_l1/Muennighoff_xstory_cloze/zh/Story_Continuation_and_Options/results.json
ADDED
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"dataset_name": "Muennighoff/xstory_cloze",
|
3 |
+
"dataset_config_name": "zh",
|
4 |
+
"template_name": "Story Continuation and Options",
|
5 |
+
"evaluation": {
|
6 |
+
"accuracy": 0.5731303772336201
|
7 |
+
},
|
8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Story Continuation and Options', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
+
}
|
evaluation_l1/Muennighoff_xwinograd/en/Replace/results.json
ADDED
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"dataset_name": "Muennighoff/xwinograd",
|
3 |
+
"dataset_config_name": "en",
|
4 |
+
"template_name": "Replace",
|
5 |
+
"evaluation": {
|
6 |
+
"accuracy": 0.5040860215053763
|
7 |
+
},
|
8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='en', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='Replace', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
+
}
|
evaluation_l1/Muennighoff_xwinograd/en/underscore_refer_to/results.json
ADDED
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"dataset_name": "Muennighoff/xwinograd",
|
3 |
+
"dataset_config_name": "en",
|
4 |
+
"template_name": "underscore refer to",
|
5 |
+
"evaluation": {
|
6 |
+
"accuracy": 0.501505376344086
|
7 |
+
},
|
8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='en', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='underscore refer to', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
+
}
|
evaluation_l1/Muennighoff_xwinograd/pt/True_or_False/results.json
ADDED
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"dataset_name": "Muennighoff/xwinograd",
|
3 |
+
"dataset_config_name": "pt",
|
4 |
+
"template_name": "True or False",
|
5 |
+
"evaluation": {
|
6 |
+
"accuracy": 0.4790874524714829
|
7 |
+
},
|
8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='pt', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='True or False', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
+
}
|
evaluation_l1/Muennighoff_xwinograd/zh/Replace/results.json
ADDED
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"dataset_name": "Muennighoff/xwinograd",
|
3 |
+
"dataset_config_name": "zh",
|
4 |
+
"template_name": "Replace",
|
5 |
+
"evaluation": {
|
6 |
+
"accuracy": 0.5515873015873016
|
7 |
+
},
|
8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='Replace', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
+
}
|
evaluation_l1/Muennighoff_xwinograd/zh/stand_for/results.json
ADDED
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"dataset_name": "Muennighoff/xwinograd",
|
3 |
+
"dataset_config_name": "zh",
|
4 |
+
"template_name": "stand for",
|
5 |
+
"evaluation": {
|
6 |
+
"accuracy": 0.49404761904761907
|
7 |
+
},
|
8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='stand for', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
+
}
|
evaluation_l1/Muennighoff_xwinograd/zh/underscore_refer_to/results.json
ADDED
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"dataset_name": "Muennighoff/xwinograd",
|
3 |
+
"dataset_config_name": "zh",
|
4 |
+
"template_name": "underscore refer to",
|
5 |
+
"evaluation": {
|
6 |
+
"accuracy": 0.5436507936507936
|
7 |
+
},
|
8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='underscore refer to', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
+
}
|
evaluation_l1/xcopa/id/cause_effect/results.json
ADDED
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"dataset_name": "xcopa",
|
3 |
+
"dataset_config_name": "id",
|
4 |
+
"template_name": "cause_effect",
|
5 |
+
"evaluation": {
|
6 |
+
"accuracy": 0.59
|
7 |
+
},
|
8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='id', dataset_name='xcopa', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='cause_effect', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
+
}
|
evaluation_l1/xcopa/id/plausible_alternatives/results.json
ADDED
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"dataset_name": "xcopa",
|
3 |
+
"dataset_config_name": "id",
|
4 |
+
"template_name": "plausible_alternatives",
|
5 |
+
"evaluation": {
|
6 |
+
"accuracy": 0.6
|
7 |
+
},
|
8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='id', dataset_name='xcopa', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='plausible_alternatives', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
+
}
|
evaluation_l1/xcopa/zh/plausible_alternatives/results.json
ADDED
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"dataset_name": "xcopa",
|
3 |
+
"dataset_config_name": "zh",
|
4 |
+
"template_name": "plausible_alternatives",
|
5 |
+
"evaluation": {
|
6 |
+
"accuracy": 0.52
|
7 |
+
},
|
8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='xcopa', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='plausible_alternatives', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
+
}
|
evaluation_l1/xnli/en/can_we_infer/results.json
ADDED
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"dataset_name": "xnli",
|
3 |
+
"dataset_config_name": "en",
|
4 |
+
"template_name": "can we infer",
|
5 |
+
"evaluation": {
|
6 |
+
"accuracy": 0.4710843373493976
|
7 |
+
},
|
8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='en', dataset_name='xnli', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='can we infer', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
+
}
|
evaluation_l1/xnli/es/GPT-3_style/results.json
ADDED
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"dataset_name": "xnli",
|
3 |
+
"dataset_config_name": "es",
|
4 |
+
"template_name": "GPT-3 style",
|
5 |
+
"evaluation": {
|
6 |
+
"accuracy": 0.42771084337349397
|
7 |
+
},
|
8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='es', dataset_name='xnli', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='GPT-3 style', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
+
}
|
evaluation_l1/xnli/es/MNLI_crowdsource/results.json
ADDED
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"dataset_name": "xnli",
|
3 |
+
"dataset_config_name": "es",
|
4 |
+
"template_name": "MNLI crowdsource",
|
5 |
+
"evaluation": {
|
6 |
+
"accuracy": 0.35903614457831323
|
7 |
+
},
|
8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='es', dataset_name='xnli', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='MNLI crowdsource', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
+
}
|
evaluation_l1/xnli/es/guaranteed_possible_impossible/results.json
ADDED
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"dataset_name": "xnli",
|
3 |
+
"dataset_config_name": "es",
|
4 |
+
"template_name": "guaranteed/possible/impossible",
|
5 |
+
"evaluation": {
|
6 |
+
"accuracy": 0.3333333333333333
|
7 |
+
},
|
8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='es', dataset_name='xnli', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='guaranteed/possible/impossible', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
+
}
|
evaluation_l1/xnli/es/justified_in_saying/results.json
ADDED
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"dataset_name": "xnli",
|
3 |
+
"dataset_config_name": "es",
|
4 |
+
"template_name": "justified in saying",
|
5 |
+
"evaluation": {
|
6 |
+
"accuracy": 0.41244979919678715
|
7 |
+
},
|
8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='es', dataset_name='xnli', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='justified in saying', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
+
}
|
evaluation_l2/xnli/th/can_we_infer/results.json
ADDED
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"dataset_name": "xnli",
|
3 |
+
"dataset_config_name": "th",
|
4 |
+
"template_name": "can we infer",
|
5 |
+
"evaluation": {
|
6 |
+
"accuracy": 0.42931726907630524
|
7 |
+
},
|
8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='th', dataset_name='xnli', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='can we infer', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
+
}
|
evaluation_l2/xnli/tr/GPT-3_style/results.json
ADDED
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"dataset_name": "xnli",
|
3 |
+
"dataset_config_name": "tr",
|
4 |
+
"template_name": "GPT-3 style",
|
5 |
+
"evaluation": {
|
6 |
+
"accuracy": 0.3397590361445783
|
7 |
+
},
|
8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='tr', dataset_name='xnli', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='GPT-3 style', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
+
}
|
evaluation_l2/xnli/tr/MNLI_crowdsource/results.json
ADDED
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"dataset_name": "xnli",
|
3 |
+
"dataset_config_name": "tr",
|
4 |
+
"template_name": "MNLI crowdsource",
|
5 |
+
"evaluation": {
|
6 |
+
"accuracy": 0.3389558232931727
|
7 |
+
},
|
8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='tr', dataset_name='xnli', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='MNLI crowdsource', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
+
}
|
evaluation_l2/xnli/tr/can_we_infer/results.json
ADDED
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"dataset_name": "xnli",
|
3 |
+
"dataset_config_name": "tr",
|
4 |
+
"template_name": "can we infer",
|
5 |
+
"evaluation": {
|
6 |
+
"accuracy": 0.3751004016064257
|
7 |
+
},
|
8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='tr', dataset_name='xnli', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='can we infer', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
+
}
|
evaluation_l2/xnli/tr/guaranteed_possible_impossible/results.json
ADDED
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"dataset_name": "xnli",
|
3 |
+
"dataset_config_name": "tr",
|
4 |
+
"template_name": "guaranteed/possible/impossible",
|
5 |
+
"evaluation": {
|
6 |
+
"accuracy": 0.3333333333333333
|
7 |
+
},
|
8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='tr', dataset_name='xnli', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='guaranteed/possible/impossible', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
+
}
|
evaluation_l2/xnli/tr/justified_in_saying/results.json
ADDED
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"dataset_name": "xnli",
|
3 |
+
"dataset_config_name": "tr",
|
4 |
+
"template_name": "justified in saying",
|
5 |
+
"evaluation": {
|
6 |
+
"accuracy": 0.3413654618473896
|
7 |
+
},
|
8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='tr', dataset_name='xnli', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='justified in saying', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
+
}
|
evaluation_xnlimtht/xnli/ar/GPT-3_style_arht/results.json
ADDED
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"dataset_name": "xnli",
|
3 |
+
"dataset_config_name": "ar",
|
4 |
+
"template_name": "GPT-3 style_arht",
|
5 |
+
"evaluation": {
|
6 |
+
"accuracy": 0.3401606425702811
|
7 |
+
},
|
8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='ar', dataset_name='xnli', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='ar', template_name='GPT-3 style_arht', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
+
}
|
evaluation_xnlimtht/xnli/ar/justified_in_saying_arht/results.json
ADDED
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"dataset_name": "xnli",
|
3 |
+
"dataset_config_name": "ar",
|
4 |
+
"template_name": "justified in saying_arht",
|
5 |
+
"evaluation": {
|
6 |
+
"accuracy": 0.3706827309236948
|
7 |
+
},
|
8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='ar', dataset_name='xnli', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='ar', template_name='justified in saying_arht', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
+
}
|
evaluation_xnlimtht/xnli/ur/GPT-3_style_urht/results.json
ADDED
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"dataset_name": "xnli",
|
3 |
+
"dataset_config_name": "ur",
|
4 |
+
"template_name": "GPT-3 style_urht",
|
5 |
+
"evaluation": {
|
6 |
+
"accuracy": 0.3337349397590361
|
7 |
+
},
|
8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='ur', dataset_name='xnli', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='ur', template_name='GPT-3 style_urht', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
+
}
|
evaluation_xnlimtht/xnli/ur/MNLI_crowdsource_urht/results.json
ADDED
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"dataset_name": "xnli",
|
3 |
+
"dataset_config_name": "ur",
|
4 |
+
"template_name": "MNLI crowdsource_urht",
|
5 |
+
"evaluation": {
|
6 |
+
"accuracy": 0.3333333333333333
|
7 |
+
},
|
8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='ur', dataset_name='xnli', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='ur', template_name='MNLI crowdsource_urht', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
+
}
|
evaluation_xnlimtht/xnli/ur/can_we_infer_urht/results.json
ADDED
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"dataset_name": "xnli",
|
3 |
+
"dataset_config_name": "ur",
|
4 |
+
"template_name": "can we infer_urht",
|
5 |
+
"evaluation": {
|
6 |
+
"accuracy": 0.3385542168674699
|
7 |
+
},
|
8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='ur', dataset_name='xnli', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='ur', template_name='can we infer_urht', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
+
}
|
evaluation_xnlimtht/xnli/ur/guaranteed_possible_impossible_urht/results.json
ADDED
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"dataset_name": "xnli",
|
3 |
+
"dataset_config_name": "ur",
|
4 |
+
"template_name": "guaranteed/possible/impossible_urht",
|
5 |
+
"evaluation": {
|
6 |
+
"accuracy": 0.3317269076305221
|
7 |
+
},
|
8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='ur', dataset_name='xnli', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='ur', template_name='guaranteed/possible/impossible_urht', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
+
}
|
evaluation_xnlimtht/xnli/ur/justified_in_saying_urht/results.json
ADDED
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"dataset_name": "xnli",
|
3 |
+
"dataset_config_name": "ur",
|
4 |
+
"template_name": "justified in saying_urht",
|
5 |
+
"evaluation": {
|
6 |
+
"accuracy": 0.336144578313253
|
7 |
+
},
|
8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='ur', dataset_name='xnli', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='ur', template_name='justified in saying_urht', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
+
}
|
evaluation_xnlimtht/xnli/vi/GPT-3_style_viht/results.json
ADDED
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"dataset_name": "xnli",
|
3 |
+
"dataset_config_name": "vi",
|
4 |
+
"template_name": "GPT-3 style_viht",
|
5 |
+
"evaluation": {
|
6 |
+
"accuracy": 0.41686746987951806
|
7 |
+
},
|
8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='vi', dataset_name='xnli', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='vi', template_name='GPT-3 style_viht', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
+
}
|
evaluation_xnlimtht/xnli/vi/MNLI_crowdsource_viht/results.json
ADDED
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"dataset_name": "xnli",
|
3 |
+
"dataset_config_name": "vi",
|
4 |
+
"template_name": "MNLI crowdsource_viht",
|
5 |
+
"evaluation": {
|
6 |
+
"accuracy": 0.3710843373493976
|
7 |
+
},
|
8 |
+
"arguments": "Namespace(config_name=None, dataset_config_name='vi', dataset_name='xnli', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', nospace=False, output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='vi', template_name='MNLI crowdsource_viht', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
+
}
|
{evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/ar/Answer_Given_options_armt/results.json
RENAMED
File without changes
|
{evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/ar/Choose_Story_Ending_armt/results.json
RENAMED
File without changes
|
{evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/ar/Generate_Ending_armt/results.json
RENAMED
File without changes
|
{evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/ar/Novel_Correct_Ending_armt/results.json
RENAMED
File without changes
|
{evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/ar/Story_Continuation_and_Options_armt/results.json
RENAMED
File without changes
|
{evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/es/Answer_Given_options_esmt/results.json
RENAMED
File without changes
|
{evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/es/Choose_Story_Ending_esmt/results.json
RENAMED
File without changes
|
{evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/es/Generate_Ending_esmt/results.json
RENAMED
File without changes
|
{evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/es/Novel_Correct_Ending_esmt/results.json
RENAMED
File without changes
|
{evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/es/Story_Continuation_and_Options_esmt/results.json
RENAMED
File without changes
|
{evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/eu/Answer_Given_options_eumt/results.json
RENAMED
File without changes
|
{evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/eu/Choose_Story_Ending_eumt/results.json
RENAMED
File without changes
|
{evaluation_xcopawinostorymt → evaluation_xwinostorycopamt}/Muennighoff_xstory_cloze/eu/Generate_Ending_eumt/results.json
RENAMED
File without changes
|