{ "results": { "blimp": { "acc,none": 0.659910447761194, "acc_stderr,none": 0.16415593005353757, "alias": "blimp" }, "blimp_adjunct_island": { "acc,none": 0.609, "acc_stderr,none": 0.015438826294681782, "alias": " - blimp_adjunct_island" }, "blimp_anaphor_gender_agreement": { "acc,none": 0.68, "acc_stderr,none": 0.01475865230357487, "alias": " - blimp_anaphor_gender_agreement" }, "blimp_anaphor_number_agreement": { "acc,none": 0.854, "acc_stderr,none": 0.011171786285496496, "alias": " - blimp_anaphor_number_agreement" }, "blimp_animate_subject_passive": { "acc,none": 0.67, "acc_stderr,none": 0.014876872027456732, "alias": " - blimp_animate_subject_passive" }, "blimp_animate_subject_trans": { "acc,none": 0.707, "acc_stderr,none": 0.014399942998441275, "alias": " - blimp_animate_subject_trans" }, "blimp_causative": { "acc,none": 0.613, "acc_stderr,none": 0.015410011955493932, "alias": " - blimp_causative" }, "blimp_complex_NP_island": { "acc,none": 0.532, "acc_stderr,none": 0.015786868759359005, "alias": " - blimp_complex_NP_island" }, "blimp_coordinate_structure_constraint_complex_left_branch": { "acc,none": 0.443, "acc_stderr,none": 0.0157161699532041, "alias": " - blimp_coordinate_structure_constraint_complex_left_branch" }, "blimp_coordinate_structure_constraint_object_extraction": { "acc,none": 0.591, "acc_stderr,none": 0.015555094373257942, "alias": " - blimp_coordinate_structure_constraint_object_extraction" }, "blimp_determiner_noun_agreement_1": { "acc,none": 0.917, "acc_stderr,none": 0.00872852720607479, "alias": " - blimp_determiner_noun_agreement_1" }, "blimp_determiner_noun_agreement_2": { "acc,none": 0.929, "acc_stderr,none": 0.008125578442487909, "alias": " - blimp_determiner_noun_agreement_2" }, "blimp_determiner_noun_agreement_irregular_1": { "acc,none": 0.74, "acc_stderr,none": 0.013877773329774166, "alias": " - blimp_determiner_noun_agreement_irregular_1" }, "blimp_determiner_noun_agreement_irregular_2": { "acc,none": 0.838, "acc_stderr,none": 0.011657267771304415, "alias": " - blimp_determiner_noun_agreement_irregular_2" }, "blimp_determiner_noun_agreement_with_adj_2": { "acc,none": 0.855, "acc_stderr,none": 0.011139977517890145, "alias": " - blimp_determiner_noun_agreement_with_adj_2" }, "blimp_determiner_noun_agreement_with_adj_irregular_1": { "acc,none": 0.795, "acc_stderr,none": 0.01277255409611311, "alias": " - blimp_determiner_noun_agreement_with_adj_irregular_1" }, "blimp_determiner_noun_agreement_with_adj_irregular_2": { "acc,none": 0.803, "acc_stderr,none": 0.012583693787968132, "alias": " - blimp_determiner_noun_agreement_with_adj_irregular_2" }, "blimp_determiner_noun_agreement_with_adjective_1": { "acc,none": 0.886, "acc_stderr,none": 0.010055103435823335, "alias": " - blimp_determiner_noun_agreement_with_adjective_1" }, "blimp_distractor_agreement_relational_noun": { "acc,none": 0.506, "acc_stderr,none": 0.015818160898606715, "alias": " - blimp_distractor_agreement_relational_noun" }, "blimp_distractor_agreement_relative_clause": { "acc,none": 0.477, "acc_stderr,none": 0.0158025542467261, "alias": " - blimp_distractor_agreement_relative_clause" }, "blimp_drop_argument": { "acc,none": 0.71, "acc_stderr,none": 0.014356395999905685, "alias": " - blimp_drop_argument" }, "blimp_ellipsis_n_bar_1": { "acc,none": 0.588, "acc_stderr,none": 0.015572363292015095, "alias": " - blimp_ellipsis_n_bar_1" }, "blimp_ellipsis_n_bar_2": { "acc,none": 0.543, "acc_stderr,none": 0.01576069159013638, "alias": " - blimp_ellipsis_n_bar_2" }, "blimp_existential_there_object_raising": { "acc,none": 0.673, "acc_stderr,none": 0.014842213153411244, "alias": " - blimp_existential_there_object_raising" }, "blimp_existential_there_quantifiers_1": { "acc,none": 0.912, "acc_stderr,none": 0.00896305396259209, "alias": " - blimp_existential_there_quantifiers_1" }, "blimp_existential_there_quantifiers_2": { "acc,none": 0.724, "acc_stderr,none": 0.014142984975740663, "alias": " - blimp_existential_there_quantifiers_2" }, "blimp_existential_there_subject_raising": { "acc,none": 0.765, "acc_stderr,none": 0.01341472903024712, "alias": " - blimp_existential_there_subject_raising" }, "blimp_expletive_it_object_raising": { "acc,none": 0.64, "acc_stderr,none": 0.015186527932040122, "alias": " - blimp_expletive_it_object_raising" }, "blimp_inchoative": { "acc,none": 0.551, "acc_stderr,none": 0.01573679276875202, "alias": " - blimp_inchoative" }, "blimp_intransitive": { "acc,none": 0.664, "acc_stderr,none": 0.014944140233795021, "alias": " - blimp_intransitive" }, "blimp_irregular_past_participle_adjectives": { "acc,none": 0.76, "acc_stderr,none": 0.013512312258920826, "alias": " - blimp_irregular_past_participle_adjectives" }, "blimp_irregular_past_participle_verbs": { "acc,none": 0.799, "acc_stderr,none": 0.01267910721461733, "alias": " - blimp_irregular_past_participle_verbs" }, "blimp_irregular_plural_subject_verb_agreement_1": { "acc,none": 0.681, "acc_stderr,none": 0.014746404865473477, "alias": " - blimp_irregular_plural_subject_verb_agreement_1" }, "blimp_irregular_plural_subject_verb_agreement_2": { "acc,none": 0.783, "acc_stderr,none": 0.01304151375727071, "alias": " - blimp_irregular_plural_subject_verb_agreement_2" }, "blimp_left_branch_island_echo_question": { "acc,none": 0.434, "acc_stderr,none": 0.015680876566375054, "alias": " - blimp_left_branch_island_echo_question" }, "blimp_left_branch_island_simple_question": { "acc,none": 0.559, "acc_stderr,none": 0.01570877989424268, "alias": " - blimp_left_branch_island_simple_question" }, "blimp_matrix_question_npi_licensor_present": { "acc,none": 0.271, "acc_stderr,none": 0.014062601350986186, "alias": " - blimp_matrix_question_npi_licensor_present" }, "blimp_npi_present_1": { "acc,none": 0.472, "acc_stderr,none": 0.015794475789511476, "alias": " - blimp_npi_present_1" }, "blimp_npi_present_2": { "acc,none": 0.526, "acc_stderr,none": 0.015797897758042766, "alias": " - blimp_npi_present_2" }, "blimp_only_npi_licensor_present": { "acc,none": 0.576, "acc_stderr,none": 0.01563548747140519, "alias": " - blimp_only_npi_licensor_present" }, "blimp_only_npi_scope": { "acc,none": 0.621, "acc_stderr,none": 0.015349091002225349, "alias": " - blimp_only_npi_scope" }, "blimp_passive_1": { "acc,none": 0.797, "acc_stderr,none": 0.012726073744598288, "alias": " - blimp_passive_1" }, "blimp_passive_2": { "acc,none": 0.756, "acc_stderr,none": 0.01358854843788143, "alias": " - blimp_passive_2" }, "blimp_principle_A_c_command": { "acc,none": 0.561, "acc_stderr,none": 0.01570113134540077, "alias": " - blimp_principle_A_c_command" }, "blimp_principle_A_case_1": { "acc,none": 0.997, "acc_stderr,none": 0.001730316154346938, "alias": " - blimp_principle_A_case_1" }, "blimp_principle_A_case_2": { "acc,none": 0.745, "acc_stderr,none": 0.013790038620872845, "alias": " - blimp_principle_A_case_2" }, "blimp_principle_A_domain_1": { "acc,none": 0.932, "acc_stderr,none": 0.007964887911291603, "alias": " - blimp_principle_A_domain_1" }, "blimp_principle_A_domain_2": { "acc,none": 0.615, "acc_stderr,none": 0.015395194445410805, "alias": " - blimp_principle_A_domain_2" }, "blimp_principle_A_domain_3": { "acc,none": 0.484, "acc_stderr,none": 0.015811198373114878, "alias": " - blimp_principle_A_domain_3" }, "blimp_principle_A_reconstruction": { "acc,none": 0.392, "acc_stderr,none": 0.015445859463771302, "alias": " - blimp_principle_A_reconstruction" }, "blimp_regular_plural_subject_verb_agreement_1": { "acc,none": 0.765, "acc_stderr,none": 0.013414729030247116, "alias": " - blimp_regular_plural_subject_verb_agreement_1" }, "blimp_regular_plural_subject_verb_agreement_2": { "acc,none": 0.729, "acc_stderr,none": 0.014062601350986182, "alias": " - blimp_regular_plural_subject_verb_agreement_2" }, "blimp_sentential_negation_npi_licensor_present": { "acc,none": 0.848, "acc_stderr,none": 0.01135891830347528, "alias": " - blimp_sentential_negation_npi_licensor_present" }, "blimp_sentential_negation_npi_scope": { "acc,none": 0.426, "acc_stderr,none": 0.01564508768811381, "alias": " - blimp_sentential_negation_npi_scope" }, "blimp_sentential_subject_island": { "acc,none": 0.391, "acc_stderr,none": 0.015438826294681787, "alias": " - blimp_sentential_subject_island" }, "blimp_superlative_quantifiers_1": { "acc,none": 0.517, "acc_stderr,none": 0.015810153729833427, "alias": " - blimp_superlative_quantifiers_1" }, "blimp_superlative_quantifiers_2": { "acc,none": 0.58, "acc_stderr,none": 0.015615500115072956, "alias": " - blimp_superlative_quantifiers_2" }, "blimp_tough_vs_raising_1": { "acc,none": 0.336, "acc_stderr,none": 0.014944140233795021, "alias": " - blimp_tough_vs_raising_1" }, "blimp_tough_vs_raising_2": { "acc,none": 0.774, "acc_stderr,none": 0.013232501619085348, "alias": " - blimp_tough_vs_raising_2" }, "blimp_transitive": { "acc,none": 0.691, "acc_stderr,none": 0.0146196009772065, "alias": " - blimp_transitive" }, "blimp_wh_island": { "acc,none": 0.636, "acc_stderr,none": 0.015222868840522022, "alias": " - blimp_wh_island" }, "blimp_wh_questions_object_gap": { "acc,none": 0.612, "acc_stderr,none": 0.015417317979911076, "alias": " - blimp_wh_questions_object_gap" }, "blimp_wh_questions_subject_gap": { "acc,none": 0.784, "acc_stderr,none": 0.013019735539307804, "alias": " - blimp_wh_questions_subject_gap" }, "blimp_wh_questions_subject_gap_long_distance": { "acc,none": 0.806, "acc_stderr,none": 0.012510816141264354, "alias": " - blimp_wh_questions_subject_gap_long_distance" }, "blimp_wh_vs_that_no_gap": { "acc,none": 0.864, "acc_stderr,none": 0.010845350230472988, "alias": " - blimp_wh_vs_that_no_gap" }, "blimp_wh_vs_that_no_gap_long_distance": { "acc,none": 0.902, "acc_stderr,none": 0.009406619184621238, "alias": " - blimp_wh_vs_that_no_gap_long_distance" }, "blimp_wh_vs_that_with_gap": { "acc,none": 0.381, "acc_stderr,none": 0.015364734787007436, "alias": " - blimp_wh_vs_that_with_gap" }, "blimp_wh_vs_that_with_gap_long_distance": { "acc,none": 0.196, "acc_stderr,none": 0.012559527926707387, "alias": " - blimp_wh_vs_that_with_gap_long_distance" } }, "groups": { "blimp": { "acc,none": 0.659910447761194, "acc_stderr,none": 0.16415593005353757, "alias": "blimp" } }, "configs": { "blimp_adjunct_island": { "task": "blimp_adjunct_island", "group": "blimp", "dataset_path": "blimp", "dataset_name": "adjunct_island", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_anaphor_gender_agreement": { "task": "blimp_anaphor_gender_agreement", "group": "blimp", "dataset_path": "blimp", "dataset_name": "anaphor_gender_agreement", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_anaphor_number_agreement": { "task": "blimp_anaphor_number_agreement", "group": "blimp", "dataset_path": "blimp", "dataset_name": "anaphor_number_agreement", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_animate_subject_passive": { "task": "blimp_animate_subject_passive", "group": "blimp", "dataset_path": "blimp", "dataset_name": "animate_subject_passive", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_animate_subject_trans": { "task": "blimp_animate_subject_trans", "group": "blimp", "dataset_path": "blimp", "dataset_name": "animate_subject_trans", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_causative": { "task": "blimp_causative", "group": "blimp", "dataset_path": "blimp", "dataset_name": "causative", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_complex_NP_island": { "task": "blimp_complex_NP_island", "group": "blimp", "dataset_path": "blimp", "dataset_name": "complex_NP_island", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_coordinate_structure_constraint_complex_left_branch": { "task": "blimp_coordinate_structure_constraint_complex_left_branch", "group": "blimp", "dataset_path": "blimp", "dataset_name": "coordinate_structure_constraint_complex_left_branch", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_coordinate_structure_constraint_object_extraction": { "task": "blimp_coordinate_structure_constraint_object_extraction", "group": "blimp", "dataset_path": "blimp", "dataset_name": "coordinate_structure_constraint_object_extraction", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_determiner_noun_agreement_1": { "task": "blimp_determiner_noun_agreement_1", "group": "blimp", "dataset_path": "blimp", "dataset_name": "determiner_noun_agreement_1", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_determiner_noun_agreement_2": { "task": "blimp_determiner_noun_agreement_2", "group": "blimp", "dataset_path": "blimp", "dataset_name": "determiner_noun_agreement_2", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_determiner_noun_agreement_irregular_1": { "task": "blimp_determiner_noun_agreement_irregular_1", "group": "blimp", "dataset_path": "blimp", "dataset_name": "determiner_noun_agreement_irregular_1", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_determiner_noun_agreement_irregular_2": { "task": "blimp_determiner_noun_agreement_irregular_2", "group": "blimp", "dataset_path": "blimp", "dataset_name": "determiner_noun_agreement_irregular_2", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_determiner_noun_agreement_with_adj_2": { "task": "blimp_determiner_noun_agreement_with_adj_2", "group": "blimp", "dataset_path": "blimp", "dataset_name": "determiner_noun_agreement_with_adj_2", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_determiner_noun_agreement_with_adj_irregular_1": { "task": "blimp_determiner_noun_agreement_with_adj_irregular_1", "group": "blimp", "dataset_path": "blimp", "dataset_name": "determiner_noun_agreement_with_adj_irregular_1", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_determiner_noun_agreement_with_adj_irregular_2": { "task": "blimp_determiner_noun_agreement_with_adj_irregular_2", "group": "blimp", "dataset_path": "blimp", "dataset_name": "determiner_noun_agreement_with_adj_irregular_2", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_determiner_noun_agreement_with_adjective_1": { "task": "blimp_determiner_noun_agreement_with_adjective_1", "group": "blimp", "dataset_path": "blimp", "dataset_name": "determiner_noun_agreement_with_adjective_1", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_distractor_agreement_relational_noun": { "task": "blimp_distractor_agreement_relational_noun", "group": "blimp", "dataset_path": "blimp", "dataset_name": "distractor_agreement_relational_noun", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_distractor_agreement_relative_clause": { "task": "blimp_distractor_agreement_relative_clause", "group": "blimp", "dataset_path": "blimp", "dataset_name": "distractor_agreement_relative_clause", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_drop_argument": { "task": "blimp_drop_argument", "group": "blimp", "dataset_path": "blimp", "dataset_name": "drop_argument", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_ellipsis_n_bar_1": { "task": "blimp_ellipsis_n_bar_1", "group": "blimp", "dataset_path": "blimp", "dataset_name": "ellipsis_n_bar_1", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_ellipsis_n_bar_2": { "task": "blimp_ellipsis_n_bar_2", "group": "blimp", "dataset_path": "blimp", "dataset_name": "ellipsis_n_bar_2", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_existential_there_object_raising": { "task": "blimp_existential_there_object_raising", "group": "blimp", "dataset_path": "blimp", "dataset_name": "existential_there_object_raising", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_existential_there_quantifiers_1": { "task": "blimp_existential_there_quantifiers_1", "group": "blimp", "dataset_path": "blimp", "dataset_name": "existential_there_quantifiers_1", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_existential_there_quantifiers_2": { "task": "blimp_existential_there_quantifiers_2", "group": "blimp", "dataset_path": "blimp", "dataset_name": "existential_there_quantifiers_2", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_existential_there_subject_raising": { "task": "blimp_existential_there_subject_raising", "group": "blimp", "dataset_path": "blimp", "dataset_name": "existential_there_subject_raising", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_expletive_it_object_raising": { "task": "blimp_expletive_it_object_raising", "group": "blimp", "dataset_path": "blimp", "dataset_name": "expletive_it_object_raising", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_inchoative": { "task": "blimp_inchoative", "group": "blimp", "dataset_path": "blimp", "dataset_name": "inchoative", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_intransitive": { "task": "blimp_intransitive", "group": "blimp", "dataset_path": "blimp", "dataset_name": "intransitive", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_irregular_past_participle_adjectives": { "task": "blimp_irregular_past_participle_adjectives", "group": "blimp", "dataset_path": "blimp", "dataset_name": "irregular_past_participle_adjectives", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_irregular_past_participle_verbs": { "task": "blimp_irregular_past_participle_verbs", "group": "blimp", "dataset_path": "blimp", "dataset_name": "irregular_past_participle_verbs", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_irregular_plural_subject_verb_agreement_1": { "task": "blimp_irregular_plural_subject_verb_agreement_1", "group": "blimp", "dataset_path": "blimp", "dataset_name": "irregular_plural_subject_verb_agreement_1", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_irregular_plural_subject_verb_agreement_2": { "task": "blimp_irregular_plural_subject_verb_agreement_2", "group": "blimp", "dataset_path": "blimp", "dataset_name": "irregular_plural_subject_verb_agreement_2", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_left_branch_island_echo_question": { "task": "blimp_left_branch_island_echo_question", "group": "blimp", "dataset_path": "blimp", "dataset_name": "left_branch_island_echo_question", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_left_branch_island_simple_question": { "task": "blimp_left_branch_island_simple_question", "group": "blimp", "dataset_path": "blimp", "dataset_name": "left_branch_island_simple_question", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_matrix_question_npi_licensor_present": { "task": "blimp_matrix_question_npi_licensor_present", "group": "blimp", "dataset_path": "blimp", "dataset_name": "matrix_question_npi_licensor_present", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_npi_present_1": { "task": "blimp_npi_present_1", "group": "blimp", "dataset_path": "blimp", "dataset_name": "npi_present_1", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_npi_present_2": { "task": "blimp_npi_present_2", "group": "blimp", "dataset_path": "blimp", "dataset_name": "npi_present_2", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_only_npi_licensor_present": { "task": "blimp_only_npi_licensor_present", "group": "blimp", "dataset_path": "blimp", "dataset_name": "only_npi_licensor_present", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_only_npi_scope": { "task": "blimp_only_npi_scope", "group": "blimp", "dataset_path": "blimp", "dataset_name": "only_npi_scope", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_passive_1": { "task": "blimp_passive_1", "group": "blimp", "dataset_path": "blimp", "dataset_name": "passive_1", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_passive_2": { "task": "blimp_passive_2", "group": "blimp", "dataset_path": "blimp", "dataset_name": "passive_2", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_principle_A_c_command": { "task": "blimp_principle_A_c_command", "group": "blimp", "dataset_path": "blimp", "dataset_name": "principle_A_c_command", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_principle_A_case_1": { "task": "blimp_principle_A_case_1", "group": "blimp", "dataset_path": "blimp", "dataset_name": "principle_A_case_1", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_principle_A_case_2": { "task": "blimp_principle_A_case_2", "group": "blimp", "dataset_path": "blimp", "dataset_name": "principle_A_case_2", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_principle_A_domain_1": { "task": "blimp_principle_A_domain_1", "group": "blimp", "dataset_path": "blimp", "dataset_name": "principle_A_domain_1", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_principle_A_domain_2": { "task": "blimp_principle_A_domain_2", "group": "blimp", "dataset_path": "blimp", "dataset_name": "principle_A_domain_2", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_principle_A_domain_3": { "task": "blimp_principle_A_domain_3", "group": "blimp", "dataset_path": "blimp", "dataset_name": "principle_A_domain_3", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_principle_A_reconstruction": { "task": "blimp_principle_A_reconstruction", "group": "blimp", "dataset_path": "blimp", "dataset_name": "principle_A_reconstruction", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_regular_plural_subject_verb_agreement_1": { "task": "blimp_regular_plural_subject_verb_agreement_1", "group": "blimp", "dataset_path": "blimp", "dataset_name": "regular_plural_subject_verb_agreement_1", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_regular_plural_subject_verb_agreement_2": { "task": "blimp_regular_plural_subject_verb_agreement_2", "group": "blimp", "dataset_path": "blimp", "dataset_name": "regular_plural_subject_verb_agreement_2", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_sentential_negation_npi_licensor_present": { "task": "blimp_sentential_negation_npi_licensor_present", "group": "blimp", "dataset_path": "blimp", "dataset_name": "sentential_negation_npi_licensor_present", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_sentential_negation_npi_scope": { "task": "blimp_sentential_negation_npi_scope", "group": "blimp", "dataset_path": "blimp", "dataset_name": "sentential_negation_npi_scope", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_sentential_subject_island": { "task": "blimp_sentential_subject_island", "group": "blimp", "dataset_path": "blimp", "dataset_name": "sentential_subject_island", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_superlative_quantifiers_1": { "task": "blimp_superlative_quantifiers_1", "group": "blimp", "dataset_path": "blimp", "dataset_name": "superlative_quantifiers_1", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_superlative_quantifiers_2": { "task": "blimp_superlative_quantifiers_2", "group": "blimp", "dataset_path": "blimp", "dataset_name": "superlative_quantifiers_2", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_tough_vs_raising_1": { "task": "blimp_tough_vs_raising_1", "group": "blimp", "dataset_path": "blimp", "dataset_name": "tough_vs_raising_1", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_tough_vs_raising_2": { "task": "blimp_tough_vs_raising_2", "group": "blimp", "dataset_path": "blimp", "dataset_name": "tough_vs_raising_2", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_transitive": { "task": "blimp_transitive", "group": "blimp", "dataset_path": "blimp", "dataset_name": "transitive", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_wh_island": { "task": "blimp_wh_island", "group": "blimp", "dataset_path": "blimp", "dataset_name": "wh_island", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_wh_questions_object_gap": { "task": "blimp_wh_questions_object_gap", "group": "blimp", "dataset_path": "blimp", "dataset_name": "wh_questions_object_gap", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_wh_questions_subject_gap": { "task": "blimp_wh_questions_subject_gap", "group": "blimp", "dataset_path": "blimp", "dataset_name": "wh_questions_subject_gap", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_wh_questions_subject_gap_long_distance": { "task": "blimp_wh_questions_subject_gap_long_distance", "group": "blimp", "dataset_path": "blimp", "dataset_name": "wh_questions_subject_gap_long_distance", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_wh_vs_that_no_gap": { "task": "blimp_wh_vs_that_no_gap", "group": "blimp", "dataset_path": "blimp", "dataset_name": "wh_vs_that_no_gap", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_wh_vs_that_no_gap_long_distance": { "task": "blimp_wh_vs_that_no_gap_long_distance", "group": "blimp", "dataset_path": "blimp", "dataset_name": "wh_vs_that_no_gap_long_distance", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_wh_vs_that_with_gap": { "task": "blimp_wh_vs_that_with_gap", "group": "blimp", "dataset_path": "blimp", "dataset_name": "wh_vs_that_with_gap", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_wh_vs_that_with_gap_long_distance": { "task": "blimp_wh_vs_that_with_gap_long_distance", "group": "blimp", "dataset_path": "blimp", "dataset_name": "wh_vs_that_with_gap_long_distance", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc" } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } } }, "versions": { "blimp": "N/A", "blimp_adjunct_island": 1.0, "blimp_anaphor_gender_agreement": 1.0, "blimp_anaphor_number_agreement": 1.0, "blimp_animate_subject_passive": 1.0, "blimp_animate_subject_trans": 1.0, "blimp_causative": 1.0, "blimp_complex_NP_island": 1.0, "blimp_coordinate_structure_constraint_complex_left_branch": 1.0, "blimp_coordinate_structure_constraint_object_extraction": 1.0, "blimp_determiner_noun_agreement_1": 1.0, "blimp_determiner_noun_agreement_2": 1.0, "blimp_determiner_noun_agreement_irregular_1": 1.0, "blimp_determiner_noun_agreement_irregular_2": 1.0, "blimp_determiner_noun_agreement_with_adj_2": 1.0, "blimp_determiner_noun_agreement_with_adj_irregular_1": 1.0, "blimp_determiner_noun_agreement_with_adj_irregular_2": 1.0, "blimp_determiner_noun_agreement_with_adjective_1": 1.0, "blimp_distractor_agreement_relational_noun": 1.0, "blimp_distractor_agreement_relative_clause": 1.0, "blimp_drop_argument": 1.0, "blimp_ellipsis_n_bar_1": 1.0, "blimp_ellipsis_n_bar_2": 1.0, "blimp_existential_there_object_raising": 1.0, "blimp_existential_there_quantifiers_1": 1.0, "blimp_existential_there_quantifiers_2": 1.0, "blimp_existential_there_subject_raising": 1.0, "blimp_expletive_it_object_raising": 1.0, "blimp_inchoative": 1.0, "blimp_intransitive": 1.0, "blimp_irregular_past_participle_adjectives": 1.0, "blimp_irregular_past_participle_verbs": 1.0, "blimp_irregular_plural_subject_verb_agreement_1": 1.0, "blimp_irregular_plural_subject_verb_agreement_2": 1.0, "blimp_left_branch_island_echo_question": 1.0, "blimp_left_branch_island_simple_question": 1.0, "blimp_matrix_question_npi_licensor_present": 1.0, "blimp_npi_present_1": 1.0, "blimp_npi_present_2": 1.0, "blimp_only_npi_licensor_present": 1.0, "blimp_only_npi_scope": 1.0, "blimp_passive_1": 1.0, "blimp_passive_2": 1.0, "blimp_principle_A_c_command": 1.0, "blimp_principle_A_case_1": 1.0, "blimp_principle_A_case_2": 1.0, "blimp_principle_A_domain_1": 1.0, "blimp_principle_A_domain_2": 1.0, "blimp_principle_A_domain_3": 1.0, "blimp_principle_A_reconstruction": 1.0, "blimp_regular_plural_subject_verb_agreement_1": 1.0, "blimp_regular_plural_subject_verb_agreement_2": 1.0, "blimp_sentential_negation_npi_licensor_present": 1.0, "blimp_sentential_negation_npi_scope": 1.0, "blimp_sentential_subject_island": 1.0, "blimp_superlative_quantifiers_1": 1.0, "blimp_superlative_quantifiers_2": 1.0, "blimp_tough_vs_raising_1": 1.0, "blimp_tough_vs_raising_2": 1.0, "blimp_transitive": 1.0, "blimp_wh_island": 1.0, "blimp_wh_questions_object_gap": 1.0, "blimp_wh_questions_subject_gap": 1.0, "blimp_wh_questions_subject_gap_long_distance": 1.0, "blimp_wh_vs_that_no_gap": 1.0, "blimp_wh_vs_that_no_gap_long_distance": 1.0, "blimp_wh_vs_that_with_gap": 1.0, "blimp_wh_vs_that_with_gap_long_distance": 1.0 }, "n-shot": { "blimp": 0, "blimp_adjunct_island": 0, "blimp_anaphor_gender_agreement": 0, "blimp_anaphor_number_agreement": 0, "blimp_animate_subject_passive": 0, "blimp_animate_subject_trans": 0, "blimp_causative": 0, "blimp_complex_NP_island": 0, "blimp_coordinate_structure_constraint_complex_left_branch": 0, "blimp_coordinate_structure_constraint_object_extraction": 0, "blimp_determiner_noun_agreement_1": 0, "blimp_determiner_noun_agreement_2": 0, "blimp_determiner_noun_agreement_irregular_1": 0, "blimp_determiner_noun_agreement_irregular_2": 0, "blimp_determiner_noun_agreement_with_adj_2": 0, "blimp_determiner_noun_agreement_with_adj_irregular_1": 0, "blimp_determiner_noun_agreement_with_adj_irregular_2": 0, "blimp_determiner_noun_agreement_with_adjective_1": 0, "blimp_distractor_agreement_relational_noun": 0, "blimp_distractor_agreement_relative_clause": 0, "blimp_drop_argument": 0, "blimp_ellipsis_n_bar_1": 0, "blimp_ellipsis_n_bar_2": 0, "blimp_existential_there_object_raising": 0, "blimp_existential_there_quantifiers_1": 0, "blimp_existential_there_quantifiers_2": 0, "blimp_existential_there_subject_raising": 0, "blimp_expletive_it_object_raising": 0, "blimp_inchoative": 0, "blimp_intransitive": 0, "blimp_irregular_past_participle_adjectives": 0, "blimp_irregular_past_participle_verbs": 0, "blimp_irregular_plural_subject_verb_agreement_1": 0, "blimp_irregular_plural_subject_verb_agreement_2": 0, "blimp_left_branch_island_echo_question": 0, "blimp_left_branch_island_simple_question": 0, "blimp_matrix_question_npi_licensor_present": 0, "blimp_npi_present_1": 0, "blimp_npi_present_2": 0, "blimp_only_npi_licensor_present": 0, "blimp_only_npi_scope": 0, "blimp_passive_1": 0, "blimp_passive_2": 0, "blimp_principle_A_c_command": 0, "blimp_principle_A_case_1": 0, "blimp_principle_A_case_2": 0, "blimp_principle_A_domain_1": 0, "blimp_principle_A_domain_2": 0, "blimp_principle_A_domain_3": 0, "blimp_principle_A_reconstruction": 0, "blimp_regular_plural_subject_verb_agreement_1": 0, "blimp_regular_plural_subject_verb_agreement_2": 0, "blimp_sentential_negation_npi_licensor_present": 0, "blimp_sentential_negation_npi_scope": 0, "blimp_sentential_subject_island": 0, "blimp_superlative_quantifiers_1": 0, "blimp_superlative_quantifiers_2": 0, "blimp_tough_vs_raising_1": 0, "blimp_tough_vs_raising_2": 0, "blimp_transitive": 0, "blimp_wh_island": 0, "blimp_wh_questions_object_gap": 0, "blimp_wh_questions_subject_gap": 0, "blimp_wh_questions_subject_gap_long_distance": 0, "blimp_wh_vs_that_no_gap": 0, "blimp_wh_vs_that_no_gap_long_distance": 0, "blimp_wh_vs_that_with_gap": 0, "blimp_wh_vs_that_with_gap_long_distance": 0 }, "config": { "model": "hf", "model_args": "pretrained=/home/bastian/Dokumente/weenie_llamas/models/final_20", "batch_size": 1, "batch_sizes": [], "device": "cuda", "use_cache": null, "limit": null, "bootstrap_iters": 100000, "gen_kwargs": null }, "git_hash": null }