|
{ |
|
"results": { |
|
"blimp": { |
|
"acc,none": 0.6829850746268656, |
|
"acc_stderr,none": 0.23934230120935937, |
|
"alias": "blimp" |
|
}, |
|
"blimp_adjunct_island": { |
|
"acc,none": 0.645, |
|
"acc_stderr,none": 0.015139491543780529, |
|
"alias": " - blimp_adjunct_island" |
|
}, |
|
"blimp_anaphor_gender_agreement": { |
|
"acc,none": 0.778, |
|
"acc_stderr,none": 0.013148721948877364, |
|
"alias": " - blimp_anaphor_gender_agreement" |
|
}, |
|
"blimp_anaphor_number_agreement": { |
|
"acc,none": 0.93, |
|
"acc_stderr,none": 0.008072494358323502, |
|
"alias": " - blimp_anaphor_number_agreement" |
|
}, |
|
"blimp_animate_subject_passive": { |
|
"acc,none": 0.75, |
|
"acc_stderr,none": 0.013699915608779773, |
|
"alias": " - blimp_animate_subject_passive" |
|
}, |
|
"blimp_animate_subject_trans": { |
|
"acc,none": 0.823, |
|
"acc_stderr,none": 0.012075463420375061, |
|
"alias": " - blimp_animate_subject_trans" |
|
}, |
|
"blimp_causative": { |
|
"acc,none": 0.67, |
|
"acc_stderr,none": 0.014876872027456732, |
|
"alias": " - blimp_causative" |
|
}, |
|
"blimp_complex_NP_island": { |
|
"acc,none": 0.46, |
|
"acc_stderr,none": 0.01576859691439438, |
|
"alias": " - blimp_complex_NP_island" |
|
}, |
|
"blimp_coordinate_structure_constraint_complex_left_branch": { |
|
"acc,none": 0.34, |
|
"acc_stderr,none": 0.014987482264363935, |
|
"alias": " - blimp_coordinate_structure_constraint_complex_left_branch" |
|
}, |
|
"blimp_coordinate_structure_constraint_object_extraction": { |
|
"acc,none": 0.661, |
|
"acc_stderr,none": 0.01497675877162034, |
|
"alias": " - blimp_coordinate_structure_constraint_object_extraction" |
|
}, |
|
"blimp_determiner_noun_agreement_1": { |
|
"acc,none": 0.961, |
|
"acc_stderr,none": 0.0061250727764261, |
|
"alias": " - blimp_determiner_noun_agreement_1" |
|
}, |
|
"blimp_determiner_noun_agreement_2": { |
|
"acc,none": 0.974, |
|
"acc_stderr,none": 0.005034813735318201, |
|
"alias": " - blimp_determiner_noun_agreement_2" |
|
}, |
|
"blimp_determiner_noun_agreement_irregular_1": { |
|
"acc,none": 0.821, |
|
"acc_stderr,none": 0.0121287306057191, |
|
"alias": " - blimp_determiner_noun_agreement_irregular_1" |
|
}, |
|
"blimp_determiner_noun_agreement_irregular_2": { |
|
"acc,none": 0.886, |
|
"acc_stderr,none": 0.01005510343582333, |
|
"alias": " - blimp_determiner_noun_agreement_irregular_2" |
|
}, |
|
"blimp_determiner_noun_agreement_with_adj_2": { |
|
"acc,none": 0.93, |
|
"acc_stderr,none": 0.008072494358323497, |
|
"alias": " - blimp_determiner_noun_agreement_with_adj_2" |
|
}, |
|
"blimp_determiner_noun_agreement_with_adj_irregular_1": { |
|
"acc,none": 0.781, |
|
"acc_stderr,none": 0.013084731950262014, |
|
"alias": " - blimp_determiner_noun_agreement_with_adj_irregular_1" |
|
}, |
|
"blimp_determiner_noun_agreement_with_adj_irregular_2": { |
|
"acc,none": 0.853, |
|
"acc_stderr,none": 0.011203415395160331, |
|
"alias": " - blimp_determiner_noun_agreement_with_adj_irregular_2" |
|
}, |
|
"blimp_determiner_noun_agreement_with_adjective_1": { |
|
"acc,none": 0.943, |
|
"acc_stderr,none": 0.007335175853706819, |
|
"alias": " - blimp_determiner_noun_agreement_with_adjective_1" |
|
}, |
|
"blimp_distractor_agreement_relational_noun": { |
|
"acc,none": 0.453, |
|
"acc_stderr,none": 0.015749255189977593, |
|
"alias": " - blimp_distractor_agreement_relational_noun" |
|
}, |
|
"blimp_distractor_agreement_relative_clause": { |
|
"acc,none": 0.405, |
|
"acc_stderr,none": 0.015531136990453035, |
|
"alias": " - blimp_distractor_agreement_relative_clause" |
|
}, |
|
"blimp_drop_argument": { |
|
"acc,none": 0.721, |
|
"acc_stderr,none": 0.014190150117612035, |
|
"alias": " - blimp_drop_argument" |
|
}, |
|
"blimp_ellipsis_n_bar_1": { |
|
"acc,none": 0.688, |
|
"acc_stderr,none": 0.014658474370509012, |
|
"alias": " - blimp_ellipsis_n_bar_1" |
|
}, |
|
"blimp_ellipsis_n_bar_2": { |
|
"acc,none": 0.555, |
|
"acc_stderr,none": 0.015723301886760934, |
|
"alias": " - blimp_ellipsis_n_bar_2" |
|
}, |
|
"blimp_existential_there_object_raising": { |
|
"acc,none": 0.807, |
|
"acc_stderr,none": 0.012486268734370145, |
|
"alias": " - blimp_existential_there_object_raising" |
|
}, |
|
"blimp_existential_there_quantifiers_1": { |
|
"acc,none": 0.971, |
|
"acc_stderr,none": 0.0053091606857569775, |
|
"alias": " - blimp_existential_there_quantifiers_1" |
|
}, |
|
"blimp_existential_there_quantifiers_2": { |
|
"acc,none": 0.95, |
|
"acc_stderr,none": 0.006895472974897896, |
|
"alias": " - blimp_existential_there_quantifiers_2" |
|
}, |
|
"blimp_existential_there_subject_raising": { |
|
"acc,none": 0.836, |
|
"acc_stderr,none": 0.011715000693181309, |
|
"alias": " - blimp_existential_there_subject_raising" |
|
}, |
|
"blimp_expletive_it_object_raising": { |
|
"acc,none": 0.734, |
|
"acc_stderr,none": 0.013979965645145162, |
|
"alias": " - blimp_expletive_it_object_raising" |
|
}, |
|
"blimp_inchoative": { |
|
"acc,none": 0.499, |
|
"acc_stderr,none": 0.015819268290576814, |
|
"alias": " - blimp_inchoative" |
|
}, |
|
"blimp_intransitive": { |
|
"acc,none": 0.688, |
|
"acc_stderr,none": 0.014658474370509008, |
|
"alias": " - blimp_intransitive" |
|
}, |
|
"blimp_irregular_past_participle_adjectives": { |
|
"acc,none": 0.882, |
|
"acc_stderr,none": 0.010206869264381798, |
|
"alias": " - blimp_irregular_past_participle_adjectives" |
|
}, |
|
"blimp_irregular_past_participle_verbs": { |
|
"acc,none": 0.884, |
|
"acc_stderr,none": 0.010131468138756986, |
|
"alias": " - blimp_irregular_past_participle_verbs" |
|
}, |
|
"blimp_irregular_plural_subject_verb_agreement_1": { |
|
"acc,none": 0.78, |
|
"acc_stderr,none": 0.013106173040661778, |
|
"alias": " - blimp_irregular_plural_subject_verb_agreement_1" |
|
}, |
|
"blimp_irregular_plural_subject_verb_agreement_2": { |
|
"acc,none": 0.795, |
|
"acc_stderr,none": 0.012772554096113126, |
|
"alias": " - blimp_irregular_plural_subject_verb_agreement_2" |
|
}, |
|
"blimp_left_branch_island_echo_question": { |
|
"acc,none": 0.384, |
|
"acc_stderr,none": 0.015387682761897068, |
|
"alias": " - blimp_left_branch_island_echo_question" |
|
}, |
|
"blimp_left_branch_island_simple_question": { |
|
"acc,none": 0.48, |
|
"acc_stderr,none": 0.015806639423035167, |
|
"alias": " - blimp_left_branch_island_simple_question" |
|
}, |
|
"blimp_matrix_question_npi_licensor_present": { |
|
"acc,none": 0.255, |
|
"acc_stderr,none": 0.013790038620872839, |
|
"alias": " - blimp_matrix_question_npi_licensor_present" |
|
}, |
|
"blimp_npi_present_1": { |
|
"acc,none": 0.399, |
|
"acc_stderr,none": 0.01549319331316291, |
|
"alias": " - blimp_npi_present_1" |
|
}, |
|
"blimp_npi_present_2": { |
|
"acc,none": 0.454, |
|
"acc_stderr,none": 0.01575221038877184, |
|
"alias": " - blimp_npi_present_2" |
|
}, |
|
"blimp_only_npi_licensor_present": { |
|
"acc,none": 0.0, |
|
"acc_stderr,none": 0.0, |
|
"alias": " - blimp_only_npi_licensor_present" |
|
}, |
|
"blimp_only_npi_scope": { |
|
"acc,none": 0.091, |
|
"acc_stderr,none": 0.009099549538400222, |
|
"alias": " - blimp_only_npi_scope" |
|
}, |
|
"blimp_passive_1": { |
|
"acc,none": 0.889, |
|
"acc_stderr,none": 0.009938701010583726, |
|
"alias": " - blimp_passive_1" |
|
}, |
|
"blimp_passive_2": { |
|
"acc,none": 0.883, |
|
"acc_stderr,none": 0.010169287802713329, |
|
"alias": " - blimp_passive_2" |
|
}, |
|
"blimp_principle_A_c_command": { |
|
"acc,none": 0.499, |
|
"acc_stderr,none": 0.015819268290576817, |
|
"alias": " - blimp_principle_A_c_command" |
|
}, |
|
"blimp_principle_A_case_1": { |
|
"acc,none": 1.0, |
|
"acc_stderr,none": 0.0, |
|
"alias": " - blimp_principle_A_case_1" |
|
}, |
|
"blimp_principle_A_case_2": { |
|
"acc,none": 0.914, |
|
"acc_stderr,none": 0.008870325962594766, |
|
"alias": " - blimp_principle_A_case_2" |
|
}, |
|
"blimp_principle_A_domain_1": { |
|
"acc,none": 0.972, |
|
"acc_stderr,none": 0.005219506034410043, |
|
"alias": " - blimp_principle_A_domain_1" |
|
}, |
|
"blimp_principle_A_domain_2": { |
|
"acc,none": 0.666, |
|
"acc_stderr,none": 0.014922019523732965, |
|
"alias": " - blimp_principle_A_domain_2" |
|
}, |
|
"blimp_principle_A_domain_3": { |
|
"acc,none": 0.478, |
|
"acc_stderr,none": 0.015803979428161957, |
|
"alias": " - blimp_principle_A_domain_3" |
|
}, |
|
"blimp_principle_A_reconstruction": { |
|
"acc,none": 0.208, |
|
"acc_stderr,none": 0.012841374572096914, |
|
"alias": " - blimp_principle_A_reconstruction" |
|
}, |
|
"blimp_regular_plural_subject_verb_agreement_1": { |
|
"acc,none": 0.878, |
|
"acc_stderr,none": 0.010354864712936717, |
|
"alias": " - blimp_regular_plural_subject_verb_agreement_1" |
|
}, |
|
"blimp_regular_plural_subject_verb_agreement_2": { |
|
"acc,none": 0.788, |
|
"acc_stderr,none": 0.012931481864938038, |
|
"alias": " - blimp_regular_plural_subject_verb_agreement_2" |
|
}, |
|
"blimp_sentential_negation_npi_licensor_present": { |
|
"acc,none": 0.999, |
|
"acc_stderr,none": 0.0010000000000000115, |
|
"alias": " - blimp_sentential_negation_npi_licensor_present" |
|
}, |
|
"blimp_sentential_negation_npi_scope": { |
|
"acc,none": 0.446, |
|
"acc_stderr,none": 0.015726771166750354, |
|
"alias": " - blimp_sentential_negation_npi_scope" |
|
}, |
|
"blimp_sentential_subject_island": { |
|
"acc,none": 0.368, |
|
"acc_stderr,none": 0.0152580735615218, |
|
"alias": " - blimp_sentential_subject_island" |
|
}, |
|
"blimp_superlative_quantifiers_1": { |
|
"acc,none": 0.656, |
|
"acc_stderr,none": 0.015029633724408947, |
|
"alias": " - blimp_superlative_quantifiers_1" |
|
}, |
|
"blimp_superlative_quantifiers_2": { |
|
"acc,none": 0.859, |
|
"acc_stderr,none": 0.011010914595992445, |
|
"alias": " - blimp_superlative_quantifiers_2" |
|
}, |
|
"blimp_tough_vs_raising_1": { |
|
"acc,none": 0.38, |
|
"acc_stderr,none": 0.015356947477797585, |
|
"alias": " - blimp_tough_vs_raising_1" |
|
}, |
|
"blimp_tough_vs_raising_2": { |
|
"acc,none": 0.839, |
|
"acc_stderr,none": 0.011628164696727204, |
|
"alias": " - blimp_tough_vs_raising_2" |
|
}, |
|
"blimp_transitive": { |
|
"acc,none": 0.787, |
|
"acc_stderr,none": 0.01295371756673722, |
|
"alias": " - blimp_transitive" |
|
}, |
|
"blimp_wh_island": { |
|
"acc,none": 0.617, |
|
"acc_stderr,none": 0.015380102325652702, |
|
"alias": " - blimp_wh_island" |
|
}, |
|
"blimp_wh_questions_object_gap": { |
|
"acc,none": 0.596, |
|
"acc_stderr,none": 0.015524980677122581, |
|
"alias": " - blimp_wh_questions_object_gap" |
|
}, |
|
"blimp_wh_questions_subject_gap": { |
|
"acc,none": 0.895, |
|
"acc_stderr,none": 0.009698921026024977, |
|
"alias": " - blimp_wh_questions_subject_gap" |
|
}, |
|
"blimp_wh_questions_subject_gap_long_distance": { |
|
"acc,none": 0.92, |
|
"acc_stderr,none": 0.008583336977753653, |
|
"alias": " - blimp_wh_questions_subject_gap_long_distance" |
|
}, |
|
"blimp_wh_vs_that_no_gap": { |
|
"acc,none": 0.961, |
|
"acc_stderr,none": 0.006125072776426117, |
|
"alias": " - blimp_wh_vs_that_no_gap" |
|
}, |
|
"blimp_wh_vs_that_no_gap_long_distance": { |
|
"acc,none": 0.974, |
|
"acc_stderr,none": 0.005034813735318215, |
|
"alias": " - blimp_wh_vs_that_no_gap_long_distance" |
|
}, |
|
"blimp_wh_vs_that_with_gap": { |
|
"acc,none": 0.289, |
|
"acc_stderr,none": 0.014341711358296184, |
|
"alias": " - blimp_wh_vs_that_with_gap" |
|
}, |
|
"blimp_wh_vs_that_with_gap_long_distance": { |
|
"acc,none": 0.082, |
|
"acc_stderr,none": 0.00868051561552374, |
|
"alias": " - blimp_wh_vs_that_with_gap_long_distance" |
|
} |
|
}, |
|
"groups": { |
|
"blimp": { |
|
"acc,none": 0.6829850746268656, |
|
"acc_stderr,none": 0.23934230120935937, |
|
"alias": "blimp" |
|
} |
|
}, |
|
"configs": { |
|
"blimp_adjunct_island": { |
|
"task": "blimp_adjunct_island", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "adjunct_island", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_anaphor_gender_agreement": { |
|
"task": "blimp_anaphor_gender_agreement", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "anaphor_gender_agreement", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_anaphor_number_agreement": { |
|
"task": "blimp_anaphor_number_agreement", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "anaphor_number_agreement", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_animate_subject_passive": { |
|
"task": "blimp_animate_subject_passive", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "animate_subject_passive", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_animate_subject_trans": { |
|
"task": "blimp_animate_subject_trans", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "animate_subject_trans", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_causative": { |
|
"task": "blimp_causative", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "causative", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_complex_NP_island": { |
|
"task": "blimp_complex_NP_island", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "complex_NP_island", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_coordinate_structure_constraint_complex_left_branch": { |
|
"task": "blimp_coordinate_structure_constraint_complex_left_branch", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "coordinate_structure_constraint_complex_left_branch", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_coordinate_structure_constraint_object_extraction": { |
|
"task": "blimp_coordinate_structure_constraint_object_extraction", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "coordinate_structure_constraint_object_extraction", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_determiner_noun_agreement_1": { |
|
"task": "blimp_determiner_noun_agreement_1", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "determiner_noun_agreement_1", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_determiner_noun_agreement_2": { |
|
"task": "blimp_determiner_noun_agreement_2", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "determiner_noun_agreement_2", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_determiner_noun_agreement_irregular_1": { |
|
"task": "blimp_determiner_noun_agreement_irregular_1", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "determiner_noun_agreement_irregular_1", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_determiner_noun_agreement_irregular_2": { |
|
"task": "blimp_determiner_noun_agreement_irregular_2", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "determiner_noun_agreement_irregular_2", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_determiner_noun_agreement_with_adj_2": { |
|
"task": "blimp_determiner_noun_agreement_with_adj_2", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "determiner_noun_agreement_with_adj_2", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_determiner_noun_agreement_with_adj_irregular_1": { |
|
"task": "blimp_determiner_noun_agreement_with_adj_irregular_1", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "determiner_noun_agreement_with_adj_irregular_1", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_determiner_noun_agreement_with_adj_irregular_2": { |
|
"task": "blimp_determiner_noun_agreement_with_adj_irregular_2", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "determiner_noun_agreement_with_adj_irregular_2", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_determiner_noun_agreement_with_adjective_1": { |
|
"task": "blimp_determiner_noun_agreement_with_adjective_1", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "determiner_noun_agreement_with_adjective_1", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_distractor_agreement_relational_noun": { |
|
"task": "blimp_distractor_agreement_relational_noun", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "distractor_agreement_relational_noun", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_distractor_agreement_relative_clause": { |
|
"task": "blimp_distractor_agreement_relative_clause", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "distractor_agreement_relative_clause", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_drop_argument": { |
|
"task": "blimp_drop_argument", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "drop_argument", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_ellipsis_n_bar_1": { |
|
"task": "blimp_ellipsis_n_bar_1", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "ellipsis_n_bar_1", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_ellipsis_n_bar_2": { |
|
"task": "blimp_ellipsis_n_bar_2", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "ellipsis_n_bar_2", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_existential_there_object_raising": { |
|
"task": "blimp_existential_there_object_raising", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "existential_there_object_raising", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_existential_there_quantifiers_1": { |
|
"task": "blimp_existential_there_quantifiers_1", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "existential_there_quantifiers_1", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_existential_there_quantifiers_2": { |
|
"task": "blimp_existential_there_quantifiers_2", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "existential_there_quantifiers_2", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_existential_there_subject_raising": { |
|
"task": "blimp_existential_there_subject_raising", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "existential_there_subject_raising", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_expletive_it_object_raising": { |
|
"task": "blimp_expletive_it_object_raising", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "expletive_it_object_raising", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_inchoative": { |
|
"task": "blimp_inchoative", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "inchoative", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_intransitive": { |
|
"task": "blimp_intransitive", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "intransitive", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_irregular_past_participle_adjectives": { |
|
"task": "blimp_irregular_past_participle_adjectives", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "irregular_past_participle_adjectives", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_irregular_past_participle_verbs": { |
|
"task": "blimp_irregular_past_participle_verbs", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "irregular_past_participle_verbs", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_irregular_plural_subject_verb_agreement_1": { |
|
"task": "blimp_irregular_plural_subject_verb_agreement_1", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "irregular_plural_subject_verb_agreement_1", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_irregular_plural_subject_verb_agreement_2": { |
|
"task": "blimp_irregular_plural_subject_verb_agreement_2", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "irregular_plural_subject_verb_agreement_2", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_left_branch_island_echo_question": { |
|
"task": "blimp_left_branch_island_echo_question", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "left_branch_island_echo_question", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_left_branch_island_simple_question": { |
|
"task": "blimp_left_branch_island_simple_question", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "left_branch_island_simple_question", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_matrix_question_npi_licensor_present": { |
|
"task": "blimp_matrix_question_npi_licensor_present", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "matrix_question_npi_licensor_present", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_npi_present_1": { |
|
"task": "blimp_npi_present_1", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "npi_present_1", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_npi_present_2": { |
|
"task": "blimp_npi_present_2", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "npi_present_2", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_only_npi_licensor_present": { |
|
"task": "blimp_only_npi_licensor_present", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "only_npi_licensor_present", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_only_npi_scope": { |
|
"task": "blimp_only_npi_scope", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "only_npi_scope", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_passive_1": { |
|
"task": "blimp_passive_1", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "passive_1", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_passive_2": { |
|
"task": "blimp_passive_2", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "passive_2", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_principle_A_c_command": { |
|
"task": "blimp_principle_A_c_command", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "principle_A_c_command", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_principle_A_case_1": { |
|
"task": "blimp_principle_A_case_1", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "principle_A_case_1", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_principle_A_case_2": { |
|
"task": "blimp_principle_A_case_2", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "principle_A_case_2", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_principle_A_domain_1": { |
|
"task": "blimp_principle_A_domain_1", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "principle_A_domain_1", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_principle_A_domain_2": { |
|
"task": "blimp_principle_A_domain_2", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "principle_A_domain_2", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_principle_A_domain_3": { |
|
"task": "blimp_principle_A_domain_3", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "principle_A_domain_3", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_principle_A_reconstruction": { |
|
"task": "blimp_principle_A_reconstruction", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "principle_A_reconstruction", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_regular_plural_subject_verb_agreement_1": { |
|
"task": "blimp_regular_plural_subject_verb_agreement_1", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "regular_plural_subject_verb_agreement_1", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_regular_plural_subject_verb_agreement_2": { |
|
"task": "blimp_regular_plural_subject_verb_agreement_2", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "regular_plural_subject_verb_agreement_2", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_sentential_negation_npi_licensor_present": { |
|
"task": "blimp_sentential_negation_npi_licensor_present", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "sentential_negation_npi_licensor_present", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_sentential_negation_npi_scope": { |
|
"task": "blimp_sentential_negation_npi_scope", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "sentential_negation_npi_scope", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_sentential_subject_island": { |
|
"task": "blimp_sentential_subject_island", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "sentential_subject_island", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_superlative_quantifiers_1": { |
|
"task": "blimp_superlative_quantifiers_1", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "superlative_quantifiers_1", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_superlative_quantifiers_2": { |
|
"task": "blimp_superlative_quantifiers_2", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "superlative_quantifiers_2", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_tough_vs_raising_1": { |
|
"task": "blimp_tough_vs_raising_1", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "tough_vs_raising_1", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_tough_vs_raising_2": { |
|
"task": "blimp_tough_vs_raising_2", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "tough_vs_raising_2", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_transitive": { |
|
"task": "blimp_transitive", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "transitive", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_wh_island": { |
|
"task": "blimp_wh_island", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "wh_island", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_wh_questions_object_gap": { |
|
"task": "blimp_wh_questions_object_gap", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "wh_questions_object_gap", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_wh_questions_subject_gap": { |
|
"task": "blimp_wh_questions_subject_gap", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "wh_questions_subject_gap", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_wh_questions_subject_gap_long_distance": { |
|
"task": "blimp_wh_questions_subject_gap_long_distance", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "wh_questions_subject_gap_long_distance", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_wh_vs_that_no_gap": { |
|
"task": "blimp_wh_vs_that_no_gap", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "wh_vs_that_no_gap", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_wh_vs_that_no_gap_long_distance": { |
|
"task": "blimp_wh_vs_that_no_gap_long_distance", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "wh_vs_that_no_gap_long_distance", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_wh_vs_that_with_gap": { |
|
"task": "blimp_wh_vs_that_with_gap", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "wh_vs_that_with_gap", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
}, |
|
"blimp_wh_vs_that_with_gap_long_distance": { |
|
"task": "blimp_wh_vs_that_with_gap_long_distance", |
|
"group": "blimp", |
|
"dataset_path": "blimp", |
|
"dataset_name": "wh_vs_that_with_gap_long_distance", |
|
"validation_split": "train", |
|
"doc_to_text": "", |
|
"doc_to_target": 0, |
|
"doc_to_choice": "{{[sentence_good, sentence_bad]}}", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 0, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc" |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": true, |
|
"doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
} |
|
}, |
|
"versions": { |
|
"blimp": "N/A", |
|
"blimp_adjunct_island": 1.0, |
|
"blimp_anaphor_gender_agreement": 1.0, |
|
"blimp_anaphor_number_agreement": 1.0, |
|
"blimp_animate_subject_passive": 1.0, |
|
"blimp_animate_subject_trans": 1.0, |
|
"blimp_causative": 1.0, |
|
"blimp_complex_NP_island": 1.0, |
|
"blimp_coordinate_structure_constraint_complex_left_branch": 1.0, |
|
"blimp_coordinate_structure_constraint_object_extraction": 1.0, |
|
"blimp_determiner_noun_agreement_1": 1.0, |
|
"blimp_determiner_noun_agreement_2": 1.0, |
|
"blimp_determiner_noun_agreement_irregular_1": 1.0, |
|
"blimp_determiner_noun_agreement_irregular_2": 1.0, |
|
"blimp_determiner_noun_agreement_with_adj_2": 1.0, |
|
"blimp_determiner_noun_agreement_with_adj_irregular_1": 1.0, |
|
"blimp_determiner_noun_agreement_with_adj_irregular_2": 1.0, |
|
"blimp_determiner_noun_agreement_with_adjective_1": 1.0, |
|
"blimp_distractor_agreement_relational_noun": 1.0, |
|
"blimp_distractor_agreement_relative_clause": 1.0, |
|
"blimp_drop_argument": 1.0, |
|
"blimp_ellipsis_n_bar_1": 1.0, |
|
"blimp_ellipsis_n_bar_2": 1.0, |
|
"blimp_existential_there_object_raising": 1.0, |
|
"blimp_existential_there_quantifiers_1": 1.0, |
|
"blimp_existential_there_quantifiers_2": 1.0, |
|
"blimp_existential_there_subject_raising": 1.0, |
|
"blimp_expletive_it_object_raising": 1.0, |
|
"blimp_inchoative": 1.0, |
|
"blimp_intransitive": 1.0, |
|
"blimp_irregular_past_participle_adjectives": 1.0, |
|
"blimp_irregular_past_participle_verbs": 1.0, |
|
"blimp_irregular_plural_subject_verb_agreement_1": 1.0, |
|
"blimp_irregular_plural_subject_verb_agreement_2": 1.0, |
|
"blimp_left_branch_island_echo_question": 1.0, |
|
"blimp_left_branch_island_simple_question": 1.0, |
|
"blimp_matrix_question_npi_licensor_present": 1.0, |
|
"blimp_npi_present_1": 1.0, |
|
"blimp_npi_present_2": 1.0, |
|
"blimp_only_npi_licensor_present": 1.0, |
|
"blimp_only_npi_scope": 1.0, |
|
"blimp_passive_1": 1.0, |
|
"blimp_passive_2": 1.0, |
|
"blimp_principle_A_c_command": 1.0, |
|
"blimp_principle_A_case_1": 1.0, |
|
"blimp_principle_A_case_2": 1.0, |
|
"blimp_principle_A_domain_1": 1.0, |
|
"blimp_principle_A_domain_2": 1.0, |
|
"blimp_principle_A_domain_3": 1.0, |
|
"blimp_principle_A_reconstruction": 1.0, |
|
"blimp_regular_plural_subject_verb_agreement_1": 1.0, |
|
"blimp_regular_plural_subject_verb_agreement_2": 1.0, |
|
"blimp_sentential_negation_npi_licensor_present": 1.0, |
|
"blimp_sentential_negation_npi_scope": 1.0, |
|
"blimp_sentential_subject_island": 1.0, |
|
"blimp_superlative_quantifiers_1": 1.0, |
|
"blimp_superlative_quantifiers_2": 1.0, |
|
"blimp_tough_vs_raising_1": 1.0, |
|
"blimp_tough_vs_raising_2": 1.0, |
|
"blimp_transitive": 1.0, |
|
"blimp_wh_island": 1.0, |
|
"blimp_wh_questions_object_gap": 1.0, |
|
"blimp_wh_questions_subject_gap": 1.0, |
|
"blimp_wh_questions_subject_gap_long_distance": 1.0, |
|
"blimp_wh_vs_that_no_gap": 1.0, |
|
"blimp_wh_vs_that_no_gap_long_distance": 1.0, |
|
"blimp_wh_vs_that_with_gap": 1.0, |
|
"blimp_wh_vs_that_with_gap_long_distance": 1.0 |
|
}, |
|
"n-shot": { |
|
"blimp": 0, |
|
"blimp_adjunct_island": 0, |
|
"blimp_anaphor_gender_agreement": 0, |
|
"blimp_anaphor_number_agreement": 0, |
|
"blimp_animate_subject_passive": 0, |
|
"blimp_animate_subject_trans": 0, |
|
"blimp_causative": 0, |
|
"blimp_complex_NP_island": 0, |
|
"blimp_coordinate_structure_constraint_complex_left_branch": 0, |
|
"blimp_coordinate_structure_constraint_object_extraction": 0, |
|
"blimp_determiner_noun_agreement_1": 0, |
|
"blimp_determiner_noun_agreement_2": 0, |
|
"blimp_determiner_noun_agreement_irregular_1": 0, |
|
"blimp_determiner_noun_agreement_irregular_2": 0, |
|
"blimp_determiner_noun_agreement_with_adj_2": 0, |
|
"blimp_determiner_noun_agreement_with_adj_irregular_1": 0, |
|
"blimp_determiner_noun_agreement_with_adj_irregular_2": 0, |
|
"blimp_determiner_noun_agreement_with_adjective_1": 0, |
|
"blimp_distractor_agreement_relational_noun": 0, |
|
"blimp_distractor_agreement_relative_clause": 0, |
|
"blimp_drop_argument": 0, |
|
"blimp_ellipsis_n_bar_1": 0, |
|
"blimp_ellipsis_n_bar_2": 0, |
|
"blimp_existential_there_object_raising": 0, |
|
"blimp_existential_there_quantifiers_1": 0, |
|
"blimp_existential_there_quantifiers_2": 0, |
|
"blimp_existential_there_subject_raising": 0, |
|
"blimp_expletive_it_object_raising": 0, |
|
"blimp_inchoative": 0, |
|
"blimp_intransitive": 0, |
|
"blimp_irregular_past_participle_adjectives": 0, |
|
"blimp_irregular_past_participle_verbs": 0, |
|
"blimp_irregular_plural_subject_verb_agreement_1": 0, |
|
"blimp_irregular_plural_subject_verb_agreement_2": 0, |
|
"blimp_left_branch_island_echo_question": 0, |
|
"blimp_left_branch_island_simple_question": 0, |
|
"blimp_matrix_question_npi_licensor_present": 0, |
|
"blimp_npi_present_1": 0, |
|
"blimp_npi_present_2": 0, |
|
"blimp_only_npi_licensor_present": 0, |
|
"blimp_only_npi_scope": 0, |
|
"blimp_passive_1": 0, |
|
"blimp_passive_2": 0, |
|
"blimp_principle_A_c_command": 0, |
|
"blimp_principle_A_case_1": 0, |
|
"blimp_principle_A_case_2": 0, |
|
"blimp_principle_A_domain_1": 0, |
|
"blimp_principle_A_domain_2": 0, |
|
"blimp_principle_A_domain_3": 0, |
|
"blimp_principle_A_reconstruction": 0, |
|
"blimp_regular_plural_subject_verb_agreement_1": 0, |
|
"blimp_regular_plural_subject_verb_agreement_2": 0, |
|
"blimp_sentential_negation_npi_licensor_present": 0, |
|
"blimp_sentential_negation_npi_scope": 0, |
|
"blimp_sentential_subject_island": 0, |
|
"blimp_superlative_quantifiers_1": 0, |
|
"blimp_superlative_quantifiers_2": 0, |
|
"blimp_tough_vs_raising_1": 0, |
|
"blimp_tough_vs_raising_2": 0, |
|
"blimp_transitive": 0, |
|
"blimp_wh_island": 0, |
|
"blimp_wh_questions_object_gap": 0, |
|
"blimp_wh_questions_subject_gap": 0, |
|
"blimp_wh_questions_subject_gap_long_distance": 0, |
|
"blimp_wh_vs_that_no_gap": 0, |
|
"blimp_wh_vs_that_no_gap_long_distance": 0, |
|
"blimp_wh_vs_that_with_gap": 0, |
|
"blimp_wh_vs_that_with_gap_long_distance": 0 |
|
}, |
|
"config": { |
|
"model": "hf", |
|
"model_args": "pretrained=/home/bastian/Dokumente/teenie_llamas/models/final_20", |
|
"batch_size": 1, |
|
"batch_sizes": [], |
|
"device": "cuda", |
|
"use_cache": null, |
|
"limit": null, |
|
"bootstrap_iters": 100000, |
|
"gen_kwargs": null |
|
}, |
|
"git_hash": null |
|
} |