Muennighoff commited on Oct 8, 2022

Commit

d733d97

1 Parent(s): 209130f

Add files

Browse files

This view is limited to 50 files because it contains too many changes. See raw diff

Files changed (50) hide show

.gitattributes +1 -0
config.json +30 -0
evaluation_l1/Muennighoff_xstory_cloze/ar/Answer_Given_options/results.json +9 -0
evaluation_l1/Muennighoff_xstory_cloze/ar/Choose_Story_Ending/results.json +9 -0
evaluation_l1/Muennighoff_xstory_cloze/ar/Generate_Ending/results.json +9 -0
evaluation_l1/Muennighoff_xstory_cloze/ar/Novel_Correct_Ending/results.json +9 -0
evaluation_l1/Muennighoff_xstory_cloze/ar/Story_Continuation_and_Options/results.json +9 -0
evaluation_l1/Muennighoff_xstory_cloze/es/Answer_Given_options/results.json +9 -0
evaluation_l1/Muennighoff_xstory_cloze/es/Choose_Story_Ending/results.json +9 -0
evaluation_l1/Muennighoff_xstory_cloze/es/Generate_Ending/results.json +9 -0
evaluation_l1/Muennighoff_xstory_cloze/es/Novel_Correct_Ending/results.json +9 -0
evaluation_l1/Muennighoff_xstory_cloze/es/Story_Continuation_and_Options/results.json +9 -0
evaluation_l1/Muennighoff_xstory_cloze/eu/Answer_Given_options/results.json +9 -0
evaluation_l1/Muennighoff_xstory_cloze/eu/Choose_Story_Ending/results.json +9 -0
evaluation_l1/Muennighoff_xstory_cloze/eu/Generate_Ending/results.json +9 -0
evaluation_l1/Muennighoff_xstory_cloze/eu/Novel_Correct_Ending/results.json +9 -0
evaluation_l1/Muennighoff_xstory_cloze/eu/Story_Continuation_and_Options/results.json +9 -0
evaluation_l1/Muennighoff_xstory_cloze/hi/Answer_Given_options/results.json +9 -0
evaluation_l1/Muennighoff_xstory_cloze/hi/Story_Continuation_and_Options/results.json +9 -0
evaluation_l1/Muennighoff_xstory_cloze/id/Answer_Given_options/results.json +9 -0
evaluation_l1/Muennighoff_xstory_cloze/id/Choose_Story_Ending/results.json +9 -0
evaluation_l1/Muennighoff_xstory_cloze/id/Generate_Ending/results.json +9 -0
evaluation_l1/Muennighoff_xstory_cloze/id/Novel_Correct_Ending/results.json +9 -0
evaluation_l1/Muennighoff_xstory_cloze/id/Story_Continuation_and_Options/results.json +9 -0
evaluation_l1/Muennighoff_xwinograd/en/True_or_False/results.json +9 -0
evaluation_l1/Muennighoff_xwinograd/en/does_underscore_refer_to/results.json +9 -0
evaluation_l1/Muennighoff_xwinograd/en/stand_for/results.json +9 -0
evaluation_l1/Muennighoff_xwinograd/fr/Replace/results.json +9 -0
evaluation_l1/Muennighoff_xwinograd/fr/True_or_False/results.json +9 -0
evaluation_l1/Muennighoff_xwinograd/fr/does_underscore_refer_to/results.json +9 -0
evaluation_l1/Muennighoff_xwinograd/fr/stand_for/results.json +9 -0
evaluation_l1/Muennighoff_xwinograd/fr/underscore_refer_to/results.json +9 -0
evaluation_l1/Muennighoff_xwinograd/pt/Replace/results.json +9 -0
evaluation_l1/Muennighoff_xwinograd/pt/does_underscore_refer_to/results.json +9 -0
evaluation_l1/Muennighoff_xwinograd/pt/stand_for/results.json +9 -0
evaluation_l1/Muennighoff_xwinograd/pt/underscore_refer_to/results.json +9 -0
evaluation_l1/Muennighoff_xwinograd/zh/True_or_False/results.json +9 -0
evaluation_l1/Muennighoff_xwinograd/zh/does_underscore_refer_to/results.json +9 -0
evaluation_l1/anli/dev_r1/GPT-3_style/results.json +9 -0
evaluation_l1/anli/dev_r1/MNLI_crowdsource/results.json +9 -0
evaluation_l1/anli/dev_r1/can_we_infer/results.json +9 -0
evaluation_l1/anli/dev_r1/guaranteed_possible_impossible/results.json +9 -0
evaluation_l1/anli/dev_r1/justified_in_saying/results.json +9 -0
evaluation_l1/anli/dev_r2/GPT-3_style/results.json +9 -0
evaluation_l1/anli/dev_r2/MNLI_crowdsource/results.json +9 -0
evaluation_l1/anli/dev_r2/can_we_infer/results.json +9 -0
evaluation_l1/anli/dev_r2/guaranteed_possible_impossible/results.json +9 -0
evaluation_l1/anli/dev_r2/justified_in_saying/results.json +9 -0
evaluation_l1/anli/dev_r3/GPT-3_style/results.json +9 -0
evaluation_l1/anli/dev_r3/MNLI_crowdsource/results.json +9 -0

.gitattributes CHANGED Viewed

@@ -30,3 +30,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
 *.zip filter=lfs diff=lfs merge=lfs -text
 *.zst filter=lfs diff=lfs merge=lfs -text
 *tfevents* filter=lfs diff=lfs merge=lfs -text

 *.zip filter=lfs diff=lfs merge=lfs -text
 *.zst filter=lfs diff=lfs merge=lfs -text
 *tfevents* filter=lfs diff=lfs merge=lfs -text
+tokenizer.json filter=lfs diff=lfs merge=lfs -text

config.json ADDED Viewed

	@@ -0,0 +1,30 @@

+{
+  "apply_residual_connection_post_layernorm": false,
+  "attention_dropout": 0.0,
+  "attention_softmax_in_fp32": true,
+  "bias_dropout_fusion": true,
+  "bos_token_id": 1,
+  "architectures": [
+    "BloomModel"
+  ],
+  "eos_token_id": 2,
+  "pad_token_id": 3,
+  "unk_token_id": 0,
+  "hidden_dropout": 0.0,
+  "initializer_range": 0.02,
+  "layer_norm_epsilon": 1e-05,
+  "masked_softmax_fusion": true,
+  "model_type": "bloom",
+  "n_embed": 1536,
+  "n_inner": null,
+  "n_layer": 24,
+  "num_attention_heads": 16,
+  "offset_alibi": 100,
+  "pretraining_tp": 1,
+  "seq_length": 2048,
+  "skip_bias_add": true,
+  "skip_bias_add_qkv": false,
+  "transformers_version": "4.20.0",
+  "use_cache": true,
+  "vocab_size": 250880
+}

evaluation_l1/Muennighoff_xstory_cloze/ar/Answer_Given_options/results.json ADDED Viewed

	@@ -0,0 +1,9 @@

+{
+  "dataset_name": "Muennighoff/xstory_cloze",
+  "dataset_config_name": "ar",
+  "template_name": "Answer Given options",
+  "evaluation": {
+    "accuracy": 0.47650562541363334
+  },
+  "arguments": "Namespace(config_name=None, dataset_config_name='ar', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Answer Given options', tokenizer_name=None, use_slow_tokenizer=False)"
+}

evaluation_l1/Muennighoff_xstory_cloze/ar/Choose_Story_Ending/results.json ADDED Viewed

	@@ -0,0 +1,9 @@

+{
+  "dataset_name": "Muennighoff/xstory_cloze",
+  "dataset_config_name": "ar",
+  "template_name": "Choose Story Ending",
+  "evaluation": {
+    "accuracy": 0.5208471211118465
+  },
+  "arguments": "Namespace(config_name=None, dataset_config_name='ar', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Choose Story Ending', tokenizer_name=None, use_slow_tokenizer=False)"
+}

evaluation_l1/Muennighoff_xstory_cloze/ar/Generate_Ending/results.json ADDED Viewed

	@@ -0,0 +1,9 @@

+{
+  "dataset_name": "Muennighoff/xstory_cloze",
+  "dataset_config_name": "ar",
+  "template_name": "Generate Ending",
+  "evaluation": {
+    "accuracy": 0.5354070152217075
+  },
+  "arguments": "Namespace(config_name=None, dataset_config_name='ar', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Generate Ending', tokenizer_name=None, use_slow_tokenizer=False)"
+}

evaluation_l1/Muennighoff_xstory_cloze/ar/Novel_Correct_Ending/results.json ADDED Viewed

	@@ -0,0 +1,9 @@

+{
+  "dataset_name": "Muennighoff/xstory_cloze",
+  "dataset_config_name": "ar",
+  "template_name": "Novel Correct Ending",
+  "evaluation": {
+    "accuracy": 0.5062872270019855
+  },
+  "arguments": "Namespace(config_name=None, dataset_config_name='ar', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Novel Correct Ending', tokenizer_name=None, use_slow_tokenizer=False)"
+}

evaluation_l1/Muennighoff_xstory_cloze/ar/Story_Continuation_and_Options/results.json ADDED Viewed

	@@ -0,0 +1,9 @@

+{
+  "dataset_name": "Muennighoff/xstory_cloze",
+  "dataset_config_name": "ar",
+  "template_name": "Story Continuation and Options",
+  "evaluation": {
+    "accuracy": 0.5228325612177366
+  },
+  "arguments": "Namespace(config_name=None, dataset_config_name='ar', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Story Continuation and Options', tokenizer_name=None, use_slow_tokenizer=False)"
+}

evaluation_l1/Muennighoff_xstory_cloze/es/Answer_Given_options/results.json ADDED Viewed

	@@ -0,0 +1,9 @@

+{
+  "dataset_name": "Muennighoff/xstory_cloze",
+  "dataset_config_name": "es",
+  "template_name": "Answer Given options",
+  "evaluation": {
+    "accuracy": 0.514228987425546
+  },
+  "arguments": "Namespace(config_name=None, dataset_config_name='es', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Answer Given options', tokenizer_name=None, use_slow_tokenizer=False)"
+}

evaluation_l1/Muennighoff_xstory_cloze/es/Choose_Story_Ending/results.json ADDED Viewed

	@@ -0,0 +1,9 @@

+{
+  "dataset_name": "Muennighoff/xstory_cloze",
+  "dataset_config_name": "es",
+  "template_name": "Choose Story Ending",
+  "evaluation": {
+    "accuracy": 0.5724685638649901
+  },
+  "arguments": "Namespace(config_name=None, dataset_config_name='es', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Choose Story Ending', tokenizer_name=None, use_slow_tokenizer=False)"
+}

evaluation_l1/Muennighoff_xstory_cloze/es/Generate_Ending/results.json ADDED Viewed

	@@ -0,0 +1,9 @@

+{
+  "dataset_name": "Muennighoff/xstory_cloze",
+  "dataset_config_name": "es",
+  "template_name": "Generate Ending",
+  "evaluation": {
+    "accuracy": 0.5804103242885507
+  },
+  "arguments": "Namespace(config_name=None, dataset_config_name='es', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Generate Ending', tokenizer_name=None, use_slow_tokenizer=False)"
+}

evaluation_l1/Muennighoff_xstory_cloze/es/Novel_Correct_Ending/results.json ADDED Viewed

	@@ -0,0 +1,9 @@

+{
+  "dataset_name": "Muennighoff/xstory_cloze",
+  "dataset_config_name": "es",
+  "template_name": "Novel Correct Ending",
+  "evaluation": {
+    "accuracy": 0.5539377895433488
+  },
+  "arguments": "Namespace(config_name=None, dataset_config_name='es', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Novel Correct Ending', tokenizer_name=None, use_slow_tokenizer=False)"
+}

evaluation_l1/Muennighoff_xstory_cloze/es/Story_Continuation_and_Options/results.json ADDED Viewed

	@@ -0,0 +1,9 @@

+{
+  "dataset_name": "Muennighoff/xstory_cloze",
+  "dataset_config_name": "es",
+  "template_name": "Story Continuation and Options",
+  "evaluation": {
+    "accuracy": 0.5598941098610192
+  },
+  "arguments": "Namespace(config_name=None, dataset_config_name='es', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Story Continuation and Options', tokenizer_name=None, use_slow_tokenizer=False)"
+}

evaluation_l1/Muennighoff_xstory_cloze/eu/Answer_Given_options/results.json ADDED Viewed

	@@ -0,0 +1,9 @@

+{
+  "dataset_name": "Muennighoff/xstory_cloze",
+  "dataset_config_name": "eu",
+  "template_name": "Answer Given options",
+  "evaluation": {
+    "accuracy": 0.44209133024487096
+  },
+  "arguments": "Namespace(config_name=None, dataset_config_name='eu', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Answer Given options', tokenizer_name=None, use_slow_tokenizer=False)"
+}

evaluation_l1/Muennighoff_xstory_cloze/eu/Choose_Story_Ending/results.json ADDED Viewed

	@@ -0,0 +1,9 @@

+{
+  "dataset_name": "Muennighoff/xstory_cloze",
+  "dataset_config_name": "eu",
+  "template_name": "Choose Story Ending",
+  "evaluation": {
+    "accuracy": 0.47782925215089345
+  },
+  "arguments": "Namespace(config_name=None, dataset_config_name='eu', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Choose Story Ending', tokenizer_name=None, use_slow_tokenizer=False)"
+}

evaluation_l1/Muennighoff_xstory_cloze/eu/Generate_Ending/results.json ADDED Viewed

	@@ -0,0 +1,9 @@

+{
+  "dataset_name": "Muennighoff/xstory_cloze",
+  "dataset_config_name": "eu",
+  "template_name": "Generate Ending",
+  "evaluation": {
+    "accuracy": 0.5221707478491066
+  },
+  "arguments": "Namespace(config_name=None, dataset_config_name='eu', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Generate Ending', tokenizer_name=None, use_slow_tokenizer=False)"
+}

evaluation_l1/Muennighoff_xstory_cloze/eu/Novel_Correct_Ending/results.json ADDED Viewed

	@@ -0,0 +1,9 @@

+{
+  "dataset_name": "Muennighoff/xstory_cloze",
+  "dataset_config_name": "eu",
+  "template_name": "Novel Correct Ending",
+  "evaluation": {
+    "accuracy": 0.4632693580410324
+  },
+  "arguments": "Namespace(config_name=None, dataset_config_name='eu', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Novel Correct Ending', tokenizer_name=None, use_slow_tokenizer=False)"
+}

evaluation_l1/Muennighoff_xstory_cloze/eu/Story_Continuation_and_Options/results.json ADDED Viewed

	@@ -0,0 +1,9 @@

+{
+  "dataset_name": "Muennighoff/xstory_cloze",
+  "dataset_config_name": "eu",
+  "template_name": "Story Continuation and Options",
+  "evaluation": {
+    "accuracy": 0.4672402382528127
+  },
+  "arguments": "Namespace(config_name=None, dataset_config_name='eu', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Story Continuation and Options', tokenizer_name=None, use_slow_tokenizer=False)"
+}

evaluation_l1/Muennighoff_xstory_cloze/hi/Answer_Given_options/results.json ADDED Viewed

	@@ -0,0 +1,9 @@

+{
+  "dataset_name": "Muennighoff/xstory_cloze",
+  "dataset_config_name": "hi",
+  "template_name": "Answer Given options",
+  "evaluation": {
+    "accuracy": 0.49702183984116477
+  },
+  "arguments": "Namespace(config_name=None, dataset_config_name='hi', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Answer Given options', tokenizer_name=None, use_slow_tokenizer=False)"
+}

evaluation_l1/Muennighoff_xstory_cloze/hi/Story_Continuation_and_Options/results.json ADDED Viewed

	@@ -0,0 +1,9 @@

+{
+  "dataset_name": "Muennighoff/xstory_cloze",
+  "dataset_config_name": "hi",
+  "template_name": "Story Continuation and Options",
+  "evaluation": {
+    "accuracy": 0.5181998676373263
+  },
+  "arguments": "Namespace(config_name=None, dataset_config_name='hi', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Story Continuation and Options', tokenizer_name=None, use_slow_tokenizer=False)"
+}

evaluation_l1/Muennighoff_xstory_cloze/id/Answer_Given_options/results.json ADDED Viewed

	@@ -0,0 +1,9 @@

+{
+  "dataset_name": "Muennighoff/xstory_cloze",
+  "dataset_config_name": "id",
+  "template_name": "Answer Given options",
+  "evaluation": {
+    "accuracy": 0.4990072799470549
+  },
+  "arguments": "Namespace(config_name=None, dataset_config_name='id', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Answer Given options', tokenizer_name=None, use_slow_tokenizer=False)"
+}

evaluation_l1/Muennighoff_xstory_cloze/id/Choose_Story_Ending/results.json ADDED Viewed

	@@ -0,0 +1,9 @@

+{
+  "dataset_name": "Muennighoff/xstory_cloze",
+  "dataset_config_name": "id",
+  "template_name": "Choose Story Ending",
+  "evaluation": {
+    "accuracy": 0.5645268034414295
+  },
+  "arguments": "Namespace(config_name=None, dataset_config_name='id', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Choose Story Ending', tokenizer_name=None, use_slow_tokenizer=False)"
+}

evaluation_l1/Muennighoff_xstory_cloze/id/Generate_Ending/results.json ADDED Viewed

	@@ -0,0 +1,9 @@

+{
+  "dataset_name": "Muennighoff/xstory_cloze",
+  "dataset_config_name": "id",
+  "template_name": "Generate Ending",
+  "evaluation": {
+    "accuracy": 0.5797485109199206
+  },
+  "arguments": "Namespace(config_name=None, dataset_config_name='id', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Generate Ending', tokenizer_name=None, use_slow_tokenizer=False)"
+}

evaluation_l1/Muennighoff_xstory_cloze/id/Novel_Correct_Ending/results.json ADDED Viewed

	@@ -0,0 +1,9 @@

+{
+  "dataset_name": "Muennighoff/xstory_cloze",
+  "dataset_config_name": "id",
+  "template_name": "Novel Correct Ending",
+  "evaluation": {
+    "accuracy": 0.513567174056916
+  },
+  "arguments": "Namespace(config_name=None, dataset_config_name='id', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Novel Correct Ending', tokenizer_name=None, use_slow_tokenizer=False)"
+}

evaluation_l1/Muennighoff_xstory_cloze/id/Story_Continuation_and_Options/results.json ADDED Viewed

	@@ -0,0 +1,9 @@

+{
+  "dataset_name": "Muennighoff/xstory_cloze",
+  "dataset_config_name": "id",
+  "template_name": "Story Continuation and Options",
+  "evaluation": {
+    "accuracy": 0.5579086697551291
+  },
+  "arguments": "Namespace(config_name=None, dataset_config_name='id', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Story Continuation and Options', tokenizer_name=None, use_slow_tokenizer=False)"
+}

evaluation_l1/Muennighoff_xwinograd/en/True_or_False/results.json ADDED Viewed

	@@ -0,0 +1,9 @@

+{
+  "dataset_name": "Muennighoff/xwinograd",
+  "dataset_config_name": "en",
+  "template_name": "True or False",
+  "evaluation": {
+    "accuracy": 0.5049462365591398
+  },
+  "arguments": "Namespace(config_name=None, dataset_config_name='en', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='True or False', tokenizer_name=None, use_slow_tokenizer=False)"
+}

evaluation_l1/Muennighoff_xwinograd/en/does_underscore_refer_to/results.json ADDED Viewed

	@@ -0,0 +1,9 @@

+{
+  "dataset_name": "Muennighoff/xwinograd",
+  "dataset_config_name": "en",
+  "template_name": "does underscore refer to",
+  "evaluation": {
+    "accuracy": 0.4997849462365591
+  },
+  "arguments": "Namespace(config_name=None, dataset_config_name='en', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='does underscore refer to', tokenizer_name=None, use_slow_tokenizer=False)"
+}

evaluation_l1/Muennighoff_xwinograd/en/stand_for/results.json ADDED Viewed

	@@ -0,0 +1,9 @@

+{
+  "dataset_name": "Muennighoff/xwinograd",
+  "dataset_config_name": "en",
+  "template_name": "stand for",
+  "evaluation": {
+    "accuracy": 0.5006451612903225
+  },
+  "arguments": "Namespace(config_name=None, dataset_config_name='en', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='stand for', tokenizer_name=None, use_slow_tokenizer=False)"
+}

evaluation_l1/Muennighoff_xwinograd/fr/Replace/results.json ADDED Viewed

	@@ -0,0 +1,9 @@

+{
+  "dataset_name": "Muennighoff/xwinograd",
+  "dataset_config_name": "fr",
+  "template_name": "Replace",
+  "evaluation": {
+    "accuracy": 0.5301204819277109
+  },
+  "arguments": "Namespace(config_name=None, dataset_config_name='fr', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='Replace', tokenizer_name=None, use_slow_tokenizer=False)"
+}

evaluation_l1/Muennighoff_xwinograd/fr/True_or_False/results.json ADDED Viewed

	@@ -0,0 +1,9 @@

+{
+  "dataset_name": "Muennighoff/xwinograd",
+  "dataset_config_name": "fr",
+  "template_name": "True or False",
+  "evaluation": {
+    "accuracy": 0.5662650602409639
+  },
+  "arguments": "Namespace(config_name=None, dataset_config_name='fr', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='True or False', tokenizer_name=None, use_slow_tokenizer=False)"
+}

evaluation_l1/Muennighoff_xwinograd/fr/does_underscore_refer_to/results.json ADDED Viewed

	@@ -0,0 +1,9 @@

+{
+  "dataset_name": "Muennighoff/xwinograd",
+  "dataset_config_name": "fr",
+  "template_name": "does underscore refer to",
+  "evaluation": {
+    "accuracy": 0.5421686746987951
+  },
+  "arguments": "Namespace(config_name=None, dataset_config_name='fr', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='does underscore refer to', tokenizer_name=None, use_slow_tokenizer=False)"
+}

evaluation_l1/Muennighoff_xwinograd/fr/stand_for/results.json ADDED Viewed

	@@ -0,0 +1,9 @@

+{
+  "dataset_name": "Muennighoff/xwinograd",
+  "dataset_config_name": "fr",
+  "template_name": "stand for",
+  "evaluation": {
+    "accuracy": 0.5180722891566265
+  },
+  "arguments": "Namespace(config_name=None, dataset_config_name='fr', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='stand for', tokenizer_name=None, use_slow_tokenizer=False)"
+}

evaluation_l1/Muennighoff_xwinograd/fr/underscore_refer_to/results.json ADDED Viewed

	@@ -0,0 +1,9 @@

+{
+  "dataset_name": "Muennighoff/xwinograd",
+  "dataset_config_name": "fr",
+  "template_name": "underscore refer to",
+  "evaluation": {
+    "accuracy": 0.5301204819277109
+  },
+  "arguments": "Namespace(config_name=None, dataset_config_name='fr', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='underscore refer to', tokenizer_name=None, use_slow_tokenizer=False)"
+}

evaluation_l1/Muennighoff_xwinograd/pt/Replace/results.json ADDED Viewed

	@@ -0,0 +1,9 @@

+{
+  "dataset_name": "Muennighoff/xwinograd",
+  "dataset_config_name": "pt",
+  "template_name": "Replace",
+  "evaluation": {
+    "accuracy": 0.5019011406844106
+  },
+  "arguments": "Namespace(config_name=None, dataset_config_name='pt', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='Replace', tokenizer_name=None, use_slow_tokenizer=False)"
+}

evaluation_l1/Muennighoff_xwinograd/pt/does_underscore_refer_to/results.json ADDED Viewed

	@@ -0,0 +1,9 @@

+{
+  "dataset_name": "Muennighoff/xwinograd",
+  "dataset_config_name": "pt",
+  "template_name": "does underscore refer to",
+  "evaluation": {
+    "accuracy": 0.49049429657794674
+  },
+  "arguments": "Namespace(config_name=None, dataset_config_name='pt', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='does underscore refer to', tokenizer_name=None, use_slow_tokenizer=False)"
+}

evaluation_l1/Muennighoff_xwinograd/pt/stand_for/results.json ADDED Viewed

	@@ -0,0 +1,9 @@

+{
+  "dataset_name": "Muennighoff/xwinograd",
+  "dataset_config_name": "pt",
+  "template_name": "stand for",
+  "evaluation": {
+    "accuracy": 0.5095057034220533
+  },
+  "arguments": "Namespace(config_name=None, dataset_config_name='pt', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='stand for', tokenizer_name=None, use_slow_tokenizer=False)"
+}

evaluation_l1/Muennighoff_xwinograd/pt/underscore_refer_to/results.json ADDED Viewed

	@@ -0,0 +1,9 @@

+{
+  "dataset_name": "Muennighoff/xwinograd",
+  "dataset_config_name": "pt",
+  "template_name": "underscore refer to",
+  "evaluation": {
+    "accuracy": 0.5133079847908745
+  },
+  "arguments": "Namespace(config_name=None, dataset_config_name='pt', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='underscore refer to', tokenizer_name=None, use_slow_tokenizer=False)"
+}

evaluation_l1/Muennighoff_xwinograd/zh/True_or_False/results.json ADDED Viewed

	@@ -0,0 +1,9 @@

+{
+  "dataset_name": "Muennighoff/xwinograd",
+  "dataset_config_name": "zh",
+  "template_name": "True or False",
+  "evaluation": {
+    "accuracy": 0.5138888888888888
+  },
+  "arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='True or False', tokenizer_name=None, use_slow_tokenizer=False)"
+}

evaluation_l1/Muennighoff_xwinograd/zh/does_underscore_refer_to/results.json ADDED Viewed

	@@ -0,0 +1,9 @@

+{
+  "dataset_name": "Muennighoff/xwinograd",
+  "dataset_config_name": "zh",
+  "template_name": "does underscore refer to",
+  "evaluation": {
+    "accuracy": 0.5396825396825397
+  },
+  "arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xwinograd', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='does underscore refer to', tokenizer_name=None, use_slow_tokenizer=False)"
+}

evaluation_l1/anli/dev_r1/GPT-3_style/results.json ADDED Viewed

	@@ -0,0 +1,9 @@

+{
+  "dataset_name": "anli",
+  "dataset_config_name": "dev_r1",
+  "template_name": "GPT-3 style",
+  "evaluation": {
+    "accuracy": 0.291
+  },
+  "arguments": "Namespace(config_name=None, dataset_config_name='dev_r1', dataset_name='anli', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='dev_r1', target_max_length=256, template_config_name=None, template_name='GPT-3 style', tokenizer_name=None, use_slow_tokenizer=False)"
+}

evaluation_l1/anli/dev_r1/MNLI_crowdsource/results.json ADDED Viewed

	@@ -0,0 +1,9 @@

+{
+  "dataset_name": "anli",
+  "dataset_config_name": "dev_r1",
+  "template_name": "MNLI crowdsource",
+  "evaluation": {
+    "accuracy": 0.309
+  },
+  "arguments": "Namespace(config_name=None, dataset_config_name='dev_r1', dataset_name='anli', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='dev_r1', target_max_length=256, template_config_name=None, template_name='MNLI crowdsource', tokenizer_name=None, use_slow_tokenizer=False)"
+}

evaluation_l1/anli/dev_r1/can_we_infer/results.json ADDED Viewed

	@@ -0,0 +1,9 @@

+{
+  "dataset_name": "anli",
+  "dataset_config_name": "dev_r1",
+  "template_name": "can we infer",
+  "evaluation": {
+    "accuracy": 0.281
+  },
+  "arguments": "Namespace(config_name=None, dataset_config_name='dev_r1', dataset_name='anli', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='dev_r1', target_max_length=256, template_config_name=None, template_name='can we infer', tokenizer_name=None, use_slow_tokenizer=False)"
+}

evaluation_l1/anli/dev_r1/guaranteed_possible_impossible/results.json ADDED Viewed

	@@ -0,0 +1,9 @@

+{
+  "dataset_name": "anli",
+  "dataset_config_name": "dev_r1",
+  "template_name": "guaranteed/possible/impossible",
+  "evaluation": {
+    "accuracy": 0.333
+  },
+  "arguments": "Namespace(config_name=None, dataset_config_name='dev_r1', dataset_name='anli', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='dev_r1', target_max_length=256, template_config_name=None, template_name='guaranteed/possible/impossible', tokenizer_name=None, use_slow_tokenizer=False)"
+}

evaluation_l1/anli/dev_r1/justified_in_saying/results.json ADDED Viewed

	@@ -0,0 +1,9 @@

+{
+  "dataset_name": "anli",
+  "dataset_config_name": "dev_r1",
+  "template_name": "justified in saying",
+  "evaluation": {
+    "accuracy": 0.288
+  },
+  "arguments": "Namespace(config_name=None, dataset_config_name='dev_r1', dataset_name='anli', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='dev_r1', target_max_length=256, template_config_name=None, template_name='justified in saying', tokenizer_name=None, use_slow_tokenizer=False)"
+}

evaluation_l1/anli/dev_r2/GPT-3_style/results.json ADDED Viewed

	@@ -0,0 +1,9 @@

+{
+  "dataset_name": "anli",
+  "dataset_config_name": "dev_r2",
+  "template_name": "GPT-3 style",
+  "evaluation": {
+    "accuracy": 0.33
+  },
+  "arguments": "Namespace(config_name=None, dataset_config_name='dev_r2', dataset_name='anli', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='dev_r2', target_max_length=256, template_config_name=None, template_name='GPT-3 style', tokenizer_name=None, use_slow_tokenizer=False)"
+}

evaluation_l1/anli/dev_r2/MNLI_crowdsource/results.json ADDED Viewed

	@@ -0,0 +1,9 @@

+{
+  "dataset_name": "anli",
+  "dataset_config_name": "dev_r2",
+  "template_name": "MNLI crowdsource",
+  "evaluation": {
+    "accuracy": 0.335
+  },
+  "arguments": "Namespace(config_name=None, dataset_config_name='dev_r2', dataset_name='anli', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='dev_r2', target_max_length=256, template_config_name=None, template_name='MNLI crowdsource', tokenizer_name=None, use_slow_tokenizer=False)"
+}

evaluation_l1/anli/dev_r2/can_we_infer/results.json ADDED Viewed

	@@ -0,0 +1,9 @@

+{
+  "dataset_name": "anli",
+  "dataset_config_name": "dev_r2",
+  "template_name": "can we infer",
+  "evaluation": {
+    "accuracy": 0.332
+  },
+  "arguments": "Namespace(config_name=None, dataset_config_name='dev_r2', dataset_name='anli', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='dev_r2', target_max_length=256, template_config_name=None, template_name='can we infer', tokenizer_name=None, use_slow_tokenizer=False)"
+}

evaluation_l1/anli/dev_r2/guaranteed_possible_impossible/results.json ADDED Viewed

	@@ -0,0 +1,9 @@

+{
+  "dataset_name": "anli",
+  "dataset_config_name": "dev_r2",
+  "template_name": "guaranteed/possible/impossible",
+  "evaluation": {
+    "accuracy": 0.333
+  },
+  "arguments": "Namespace(config_name=None, dataset_config_name='dev_r2', dataset_name='anli', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='dev_r2', target_max_length=256, template_config_name=None, template_name='guaranteed/possible/impossible', tokenizer_name=None, use_slow_tokenizer=False)"
+}

evaluation_l1/anli/dev_r2/justified_in_saying/results.json ADDED Viewed

	@@ -0,0 +1,9 @@

+{
+  "dataset_name": "anli",
+  "dataset_config_name": "dev_r2",
+  "template_name": "justified in saying",
+  "evaluation": {
+    "accuracy": 0.331
+  },
+  "arguments": "Namespace(config_name=None, dataset_config_name='dev_r2', dataset_name='anli', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='dev_r2', target_max_length=256, template_config_name=None, template_name='justified in saying', tokenizer_name=None, use_slow_tokenizer=False)"
+}

evaluation_l1/anli/dev_r3/GPT-3_style/results.json ADDED Viewed

	@@ -0,0 +1,9 @@

+{
+  "dataset_name": "anli",
+  "dataset_config_name": "dev_r3",
+  "template_name": "GPT-3 style",
+  "evaluation": {
+    "accuracy": 0.3308333333333333
+  },
+  "arguments": "Namespace(config_name=None, dataset_config_name='dev_r3', dataset_name='anli', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='dev_r3', target_max_length=256, template_config_name=None, template_name='GPT-3 style', tokenizer_name=None, use_slow_tokenizer=False)"
+}

evaluation_l1/anli/dev_r3/MNLI_crowdsource/results.json ADDED Viewed

	@@ -0,0 +1,9 @@

+{
+  "dataset_name": "anli",
+  "dataset_config_name": "dev_r3",
+  "template_name": "MNLI crowdsource",
+  "evaluation": {
+    "accuracy": 0.345
+  },
+  "arguments": "Namespace(config_name=None, dataset_config_name='dev_r3', dataset_name='anli', debug=False, dtype='float16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1', output_dir='/gpfsscratch/rech/six/commun/experiments/muennighoff/bloomckpt/760mt0/bloomz-1b1/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='dev_r3', target_max_length=256, template_config_name=None, template_name='MNLI crowdsource', tokenizer_name=None, use_slow_tokenizer=False)"
+}