{ "results": { "blimp": { "acc,none": 0.7868059701492538, "acc_stderr,none": 0.0014126246990605982, "alias": "blimp" }, "blimp_adjunct_island": { "alias": " - blimp_adjunct_island", "acc,none": 0.837, "acc_stderr,none": 0.011686212712746828 }, "blimp_anaphor_gender_agreement": { "alias": " - blimp_anaphor_gender_agreement", "acc,none": 0.957, "acc_stderr,none": 0.006418114379799741 }, "blimp_anaphor_number_agreement": { "alias": " - blimp_anaphor_number_agreement", "acc,none": 0.988, "acc_stderr,none": 0.0034449771940998565 }, "blimp_animate_subject_passive": { "alias": " - blimp_animate_subject_passive", "acc,none": 0.746, "acc_stderr,none": 0.01377220656516854 }, "blimp_animate_subject_trans": { "alias": " - blimp_animate_subject_trans", "acc,none": 0.887, "acc_stderr,none": 0.010016552866696846 }, "blimp_causative": { "alias": " - blimp_causative", "acc,none": 0.747, "acc_stderr,none": 0.01375427861358708 }, "blimp_complex_NP_island": { "alias": " - blimp_complex_NP_island", "acc,none": 0.578, "acc_stderr,none": 0.01562562511262068 }, "blimp_coordinate_structure_constraint_complex_left_branch": { "alias": " - blimp_coordinate_structure_constraint_complex_left_branch", "acc,none": 0.624, "acc_stderr,none": 0.015325105508898134 }, "blimp_coordinate_structure_constraint_object_extraction": { "alias": " - blimp_coordinate_structure_constraint_object_extraction", "acc,none": 0.789, "acc_stderr,none": 0.012909130321042094 }, "blimp_determiner_noun_agreement_1": { "alias": " - blimp_determiner_noun_agreement_1", "acc,none": 0.981, "acc_stderr,none": 0.004319451082910638 }, "blimp_determiner_noun_agreement_2": { "alias": " - blimp_determiner_noun_agreement_2", "acc,none": 0.983, "acc_stderr,none": 0.004089954489689096 }, "blimp_determiner_noun_agreement_irregular_1": { "alias": " - blimp_determiner_noun_agreement_irregular_1", "acc,none": 0.935, "acc_stderr,none": 0.007799733061832025 }, "blimp_determiner_noun_agreement_irregular_2": { "alias": " - blimp_determiner_noun_agreement_irregular_2", "acc,none": 0.952, "acc_stderr,none": 0.006763264133666654 }, "blimp_determiner_noun_agreement_with_adj_2": { "alias": " - blimp_determiner_noun_agreement_with_adj_2", "acc,none": 0.943, "acc_stderr,none": 0.0073351758537068225 }, "blimp_determiner_noun_agreement_with_adj_irregular_1": { "alias": " - blimp_determiner_noun_agreement_with_adj_irregular_1", "acc,none": 0.894, "acc_stderr,none": 0.00973955126578513 }, "blimp_determiner_noun_agreement_with_adj_irregular_2": { "alias": " - blimp_determiner_noun_agreement_with_adj_irregular_2", "acc,none": 0.938, "acc_stderr,none": 0.007629823996280309 }, "blimp_determiner_noun_agreement_with_adjective_1": { "alias": " - blimp_determiner_noun_agreement_with_adjective_1", "acc,none": 0.955, "acc_stderr,none": 0.006558812241406112 }, "blimp_distractor_agreement_relational_noun": { "alias": " - blimp_distractor_agreement_relational_noun", "acc,none": 0.91, "acc_stderr,none": 0.00905439020486645 }, "blimp_distractor_agreement_relative_clause": { "alias": " - blimp_distractor_agreement_relative_clause", "acc,none": 0.809, "acc_stderr,none": 0.012436787112179486 }, "blimp_drop_argument": { "alias": " - blimp_drop_argument", "acc,none": 0.748, "acc_stderr,none": 0.013736254390651145 }, "blimp_ellipsis_n_bar_1": { "alias": " - blimp_ellipsis_n_bar_1", "acc,none": 0.855, "acc_stderr,none": 0.011139977517890145 }, "blimp_ellipsis_n_bar_2": { "alias": " - blimp_ellipsis_n_bar_2", "acc,none": 0.86, "acc_stderr,none": 0.010978183844357796 }, "blimp_existential_there_object_raising": { "alias": " - blimp_existential_there_object_raising", "acc,none": 0.844, "acc_stderr,none": 0.01148023500612238 }, "blimp_existential_there_quantifiers_1": { "alias": " - blimp_existential_there_quantifiers_1", "acc,none": 0.977, "acc_stderr,none": 0.004742730594656801 }, "blimp_existential_there_quantifiers_2": { "alias": " - blimp_existential_there_quantifiers_2", "acc,none": 0.219, "acc_stderr,none": 0.013084731950262024 }, "blimp_existential_there_subject_raising": { "alias": " - blimp_existential_there_subject_raising", "acc,none": 0.883, "acc_stderr,none": 0.010169287802713329 }, "blimp_expletive_it_object_raising": { "alias": " - blimp_expletive_it_object_raising", "acc,none": 0.784, "acc_stderr,none": 0.013019735539307804 }, "blimp_inchoative": { "alias": " - blimp_inchoative", "acc,none": 0.684, "acc_stderr,none": 0.014709193056057128 }, "blimp_intransitive": { "alias": " - blimp_intransitive", "acc,none": 0.789, "acc_stderr,none": 0.012909130321042095 }, "blimp_irregular_past_participle_adjectives": { "alias": " - blimp_irregular_past_participle_adjectives", "acc,none": 0.937, "acc_stderr,none": 0.007687007876286409 }, "blimp_irregular_past_participle_verbs": { "alias": " - blimp_irregular_past_participle_verbs", "acc,none": 0.888, "acc_stderr,none": 0.009977753031397247 }, "blimp_irregular_plural_subject_verb_agreement_1": { "alias": " - blimp_irregular_plural_subject_verb_agreement_1", "acc,none": 0.893, "acc_stderr,none": 0.00977991035984717 }, "blimp_irregular_plural_subject_verb_agreement_2": { "alias": " - blimp_irregular_plural_subject_verb_agreement_2", "acc,none": 0.884, "acc_stderr,none": 0.010131468138756998 }, "blimp_left_branch_island_echo_question": { "alias": " - blimp_left_branch_island_echo_question", "acc,none": 0.446, "acc_stderr,none": 0.015726771166750354 }, "blimp_left_branch_island_simple_question": { "alias": " - blimp_left_branch_island_simple_question", "acc,none": 0.592, "acc_stderr,none": 0.015549205052920671 }, "blimp_matrix_question_npi_licensor_present": { "alias": " - blimp_matrix_question_npi_licensor_present", "acc,none": 0.418, "acc_stderr,none": 0.015605111967541949 }, "blimp_npi_present_1": { "alias": " - blimp_npi_present_1", "acc,none": 0.584, "acc_stderr,none": 0.015594460144140601 }, "blimp_npi_present_2": { "alias": " - blimp_npi_present_2", "acc,none": 0.621, "acc_stderr,none": 0.01534909100222535 }, "blimp_only_npi_licensor_present": { "alias": " - blimp_only_npi_licensor_present", "acc,none": 0.917, "acc_stderr,none": 0.008728527206074789 }, "blimp_only_npi_scope": { "alias": " - blimp_only_npi_scope", "acc,none": 0.495, "acc_stderr,none": 0.015818508944436652 }, "blimp_passive_1": { "alias": " - blimp_passive_1", "acc,none": 0.871, "acc_stderr,none": 0.010605256784796577 }, "blimp_passive_2": { "alias": " - blimp_passive_2", "acc,none": 0.873, "acc_stderr,none": 0.010534798620855762 }, "blimp_principle_A_c_command": { "alias": " - blimp_principle_A_c_command", "acc,none": 0.673, "acc_stderr,none": 0.014842213153411245 }, "blimp_principle_A_case_1": { "alias": " - blimp_principle_A_case_1", "acc,none": 1.0, "acc_stderr,none": 0.0 }, "blimp_principle_A_case_2": { "alias": " - blimp_principle_A_case_2", "acc,none": 0.972, "acc_stderr,none": 0.005219506034410044 }, "blimp_principle_A_domain_1": { "alias": " - blimp_principle_A_domain_1", "acc,none": 0.958, "acc_stderr,none": 0.006346359293033839 }, "blimp_principle_A_domain_2": { "alias": " - blimp_principle_A_domain_2", "acc,none": 0.704, "acc_stderr,none": 0.014442734941575018 }, "blimp_principle_A_domain_3": { "alias": " - blimp_principle_A_domain_3", "acc,none": 0.583, "acc_stderr,none": 0.015599819048769616 }, "blimp_principle_A_reconstruction": { "alias": " - blimp_principle_A_reconstruction", "acc,none": 0.348, "acc_stderr,none": 0.01507060460376841 }, "blimp_regular_plural_subject_verb_agreement_1": { "alias": " - blimp_regular_plural_subject_verb_agreement_1", "acc,none": 0.919, "acc_stderr,none": 0.008632121032139974 }, "blimp_regular_plural_subject_verb_agreement_2": { "alias": " - blimp_regular_plural_subject_verb_agreement_2", "acc,none": 0.881, "acc_stderr,none": 0.01024421514533666 }, "blimp_sentential_negation_npi_licensor_present": { "alias": " - blimp_sentential_negation_npi_licensor_present", "acc,none": 0.945, "acc_stderr,none": 0.007212976294639244 }, "blimp_sentential_negation_npi_scope": { "alias": " - blimp_sentential_negation_npi_scope", "acc,none": 0.661, "acc_stderr,none": 0.01497675877162034 }, "blimp_sentential_subject_island": { "alias": " - blimp_sentential_subject_island", "acc,none": 0.494, "acc_stderr,none": 0.015818160898606715 }, "blimp_superlative_quantifiers_1": { "alias": " - blimp_superlative_quantifiers_1", "acc,none": 0.683, "acc_stderr,none": 0.014721675438880213 }, "blimp_superlative_quantifiers_2": { "alias": " - blimp_superlative_quantifiers_2", "acc,none": 0.801, "acc_stderr,none": 0.012631649083099177 }, "blimp_tough_vs_raising_1": { "alias": " - blimp_tough_vs_raising_1", "acc,none": 0.584, "acc_stderr,none": 0.015594460144140598 }, "blimp_tough_vs_raising_2": { "alias": " - blimp_tough_vs_raising_2", "acc,none": 0.886, "acc_stderr,none": 0.010055103435823332 }, "blimp_transitive": { "alias": " - blimp_transitive", "acc,none": 0.852, "acc_stderr,none": 0.011234866364235239 }, "blimp_wh_island": { "alias": " - blimp_wh_island", "acc,none": 0.72, "acc_stderr,none": 0.014205696104091487 }, "blimp_wh_questions_object_gap": { "alias": " - blimp_wh_questions_object_gap", "acc,none": 0.818, "acc_stderr,none": 0.012207580637662164 }, "blimp_wh_questions_subject_gap": { "alias": " - blimp_wh_questions_subject_gap", "acc,none": 0.968, "acc_stderr,none": 0.005568393575081367 }, "blimp_wh_questions_subject_gap_long_distance": { "alias": " - blimp_wh_questions_subject_gap_long_distance", "acc,none": 0.92, "acc_stderr,none": 0.008583336977753651 }, "blimp_wh_vs_that_no_gap": { "alias": " - blimp_wh_vs_that_no_gap", "acc,none": 0.976, "acc_stderr,none": 0.004842256441727098 }, "blimp_wh_vs_that_no_gap_long_distance": { "alias": " - blimp_wh_vs_that_no_gap_long_distance", "acc,none": 0.977, "acc_stderr,none": 0.004742730594656805 }, "blimp_wh_vs_that_with_gap": { "alias": " - blimp_wh_vs_that_with_gap", "acc,none": 0.566, "acc_stderr,none": 0.015680876566375058 }, "blimp_wh_vs_that_with_gap_long_distance": { "alias": " - blimp_wh_vs_that_with_gap_long_distance", "acc,none": 0.312, "acc_stderr,none": 0.014658474370508996 } }, "groups": { "blimp": { "acc,none": 0.7868059701492538, "acc_stderr,none": 0.0014126246990605982, "alias": "blimp" } }, "group_subtasks": { "blimp": [ "blimp_adjunct_island", "blimp_anaphor_gender_agreement", "blimp_anaphor_number_agreement", "blimp_animate_subject_passive", "blimp_animate_subject_trans", "blimp_causative", "blimp_complex_NP_island", "blimp_coordinate_structure_constraint_complex_left_branch", "blimp_coordinate_structure_constraint_object_extraction", "blimp_determiner_noun_agreement_1", "blimp_determiner_noun_agreement_2", "blimp_determiner_noun_agreement_irregular_1", "blimp_determiner_noun_agreement_irregular_2", "blimp_determiner_noun_agreement_with_adj_2", "blimp_determiner_noun_agreement_with_adj_irregular_1", "blimp_determiner_noun_agreement_with_adj_irregular_2", "blimp_determiner_noun_agreement_with_adjective_1", "blimp_distractor_agreement_relational_noun", "blimp_distractor_agreement_relative_clause", "blimp_drop_argument", "blimp_ellipsis_n_bar_1", "blimp_ellipsis_n_bar_2", "blimp_existential_there_object_raising", "blimp_existential_there_quantifiers_1", "blimp_existential_there_quantifiers_2", "blimp_existential_there_subject_raising", "blimp_expletive_it_object_raising", "blimp_inchoative", "blimp_intransitive", "blimp_irregular_past_participle_adjectives", "blimp_irregular_past_participle_verbs", "blimp_irregular_plural_subject_verb_agreement_1", "blimp_irregular_plural_subject_verb_agreement_2", "blimp_left_branch_island_echo_question", "blimp_left_branch_island_simple_question", "blimp_matrix_question_npi_licensor_present", "blimp_npi_present_1", "blimp_npi_present_2", "blimp_only_npi_licensor_present", "blimp_only_npi_scope", "blimp_passive_1", "blimp_passive_2", "blimp_principle_A_c_command", "blimp_principle_A_case_1", "blimp_principle_A_case_2", "blimp_principle_A_domain_1", "blimp_principle_A_domain_2", "blimp_principle_A_domain_3", "blimp_principle_A_reconstruction", "blimp_regular_plural_subject_verb_agreement_1", "blimp_regular_plural_subject_verb_agreement_2", "blimp_sentential_negation_npi_licensor_present", "blimp_sentential_negation_npi_scope", "blimp_sentential_subject_island", "blimp_superlative_quantifiers_1", "blimp_superlative_quantifiers_2", "blimp_tough_vs_raising_1", "blimp_tough_vs_raising_2", "blimp_transitive", "blimp_wh_island", "blimp_wh_questions_object_gap", "blimp_wh_questions_subject_gap", "blimp_wh_questions_subject_gap_long_distance", "blimp_wh_vs_that_no_gap", "blimp_wh_vs_that_no_gap_long_distance", "blimp_wh_vs_that_with_gap", "blimp_wh_vs_that_with_gap_long_distance" ] }, "configs": { "blimp_adjunct_island": { "task": "blimp_adjunct_island", "dataset_path": "blimp", "dataset_name": "adjunct_island", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_anaphor_gender_agreement": { "task": "blimp_anaphor_gender_agreement", "dataset_path": "blimp", "dataset_name": "anaphor_gender_agreement", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_anaphor_number_agreement": { "task": "blimp_anaphor_number_agreement", "dataset_path": "blimp", "dataset_name": "anaphor_number_agreement", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_animate_subject_passive": { "task": "blimp_animate_subject_passive", "dataset_path": "blimp", "dataset_name": "animate_subject_passive", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_animate_subject_trans": { "task": "blimp_animate_subject_trans", "dataset_path": "blimp", "dataset_name": "animate_subject_trans", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_causative": { "task": "blimp_causative", "dataset_path": "blimp", "dataset_name": "causative", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_complex_NP_island": { "task": "blimp_complex_NP_island", "dataset_path": "blimp", "dataset_name": "complex_NP_island", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_coordinate_structure_constraint_complex_left_branch": { "task": "blimp_coordinate_structure_constraint_complex_left_branch", "dataset_path": "blimp", "dataset_name": "coordinate_structure_constraint_complex_left_branch", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_coordinate_structure_constraint_object_extraction": { "task": "blimp_coordinate_structure_constraint_object_extraction", "dataset_path": "blimp", "dataset_name": "coordinate_structure_constraint_object_extraction", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_determiner_noun_agreement_1": { "task": "blimp_determiner_noun_agreement_1", "dataset_path": "blimp", "dataset_name": "determiner_noun_agreement_1", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_determiner_noun_agreement_2": { "task": "blimp_determiner_noun_agreement_2", "dataset_path": "blimp", "dataset_name": "determiner_noun_agreement_2", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_determiner_noun_agreement_irregular_1": { "task": "blimp_determiner_noun_agreement_irregular_1", "dataset_path": "blimp", "dataset_name": "determiner_noun_agreement_irregular_1", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_determiner_noun_agreement_irregular_2": { "task": "blimp_determiner_noun_agreement_irregular_2", "dataset_path": "blimp", "dataset_name": "determiner_noun_agreement_irregular_2", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_determiner_noun_agreement_with_adj_2": { "task": "blimp_determiner_noun_agreement_with_adj_2", "dataset_path": "blimp", "dataset_name": "determiner_noun_agreement_with_adj_2", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_determiner_noun_agreement_with_adj_irregular_1": { "task": "blimp_determiner_noun_agreement_with_adj_irregular_1", "dataset_path": "blimp", "dataset_name": "determiner_noun_agreement_with_adj_irregular_1", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_determiner_noun_agreement_with_adj_irregular_2": { "task": "blimp_determiner_noun_agreement_with_adj_irregular_2", "dataset_path": "blimp", "dataset_name": "determiner_noun_agreement_with_adj_irregular_2", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_determiner_noun_agreement_with_adjective_1": { "task": "blimp_determiner_noun_agreement_with_adjective_1", "dataset_path": "blimp", "dataset_name": "determiner_noun_agreement_with_adjective_1", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_distractor_agreement_relational_noun": { "task": "blimp_distractor_agreement_relational_noun", "dataset_path": "blimp", "dataset_name": "distractor_agreement_relational_noun", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_distractor_agreement_relative_clause": { "task": "blimp_distractor_agreement_relative_clause", "dataset_path": "blimp", "dataset_name": "distractor_agreement_relative_clause", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_drop_argument": { "task": "blimp_drop_argument", "dataset_path": "blimp", "dataset_name": "drop_argument", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_ellipsis_n_bar_1": { "task": "blimp_ellipsis_n_bar_1", "dataset_path": "blimp", "dataset_name": "ellipsis_n_bar_1", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_ellipsis_n_bar_2": { "task": "blimp_ellipsis_n_bar_2", "dataset_path": "blimp", "dataset_name": "ellipsis_n_bar_2", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_existential_there_object_raising": { "task": "blimp_existential_there_object_raising", "dataset_path": "blimp", "dataset_name": "existential_there_object_raising", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_existential_there_quantifiers_1": { "task": "blimp_existential_there_quantifiers_1", "dataset_path": "blimp", "dataset_name": "existential_there_quantifiers_1", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_existential_there_quantifiers_2": { "task": "blimp_existential_there_quantifiers_2", "dataset_path": "blimp", "dataset_name": "existential_there_quantifiers_2", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_existential_there_subject_raising": { "task": "blimp_existential_there_subject_raising", "dataset_path": "blimp", "dataset_name": "existential_there_subject_raising", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_expletive_it_object_raising": { "task": "blimp_expletive_it_object_raising", "dataset_path": "blimp", "dataset_name": "expletive_it_object_raising", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_inchoative": { "task": "blimp_inchoative", "dataset_path": "blimp", "dataset_name": "inchoative", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_intransitive": { "task": "blimp_intransitive", "dataset_path": "blimp", "dataset_name": "intransitive", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_irregular_past_participle_adjectives": { "task": "blimp_irregular_past_participle_adjectives", "dataset_path": "blimp", "dataset_name": "irregular_past_participle_adjectives", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_irregular_past_participle_verbs": { "task": "blimp_irregular_past_participle_verbs", "dataset_path": "blimp", "dataset_name": "irregular_past_participle_verbs", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_irregular_plural_subject_verb_agreement_1": { "task": "blimp_irregular_plural_subject_verb_agreement_1", "dataset_path": "blimp", "dataset_name": "irregular_plural_subject_verb_agreement_1", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_irregular_plural_subject_verb_agreement_2": { "task": "blimp_irregular_plural_subject_verb_agreement_2", "dataset_path": "blimp", "dataset_name": "irregular_plural_subject_verb_agreement_2", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_left_branch_island_echo_question": { "task": "blimp_left_branch_island_echo_question", "dataset_path": "blimp", "dataset_name": "left_branch_island_echo_question", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_left_branch_island_simple_question": { "task": "blimp_left_branch_island_simple_question", "dataset_path": "blimp", "dataset_name": "left_branch_island_simple_question", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_matrix_question_npi_licensor_present": { "task": "blimp_matrix_question_npi_licensor_present", "dataset_path": "blimp", "dataset_name": "matrix_question_npi_licensor_present", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_npi_present_1": { "task": "blimp_npi_present_1", "dataset_path": "blimp", "dataset_name": "npi_present_1", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_npi_present_2": { "task": "blimp_npi_present_2", "dataset_path": "blimp", "dataset_name": "npi_present_2", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_only_npi_licensor_present": { "task": "blimp_only_npi_licensor_present", "dataset_path": "blimp", "dataset_name": "only_npi_licensor_present", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_only_npi_scope": { "task": "blimp_only_npi_scope", "dataset_path": "blimp", "dataset_name": "only_npi_scope", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_passive_1": { "task": "blimp_passive_1", "dataset_path": "blimp", "dataset_name": "passive_1", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_passive_2": { "task": "blimp_passive_2", "dataset_path": "blimp", "dataset_name": "passive_2", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_principle_A_c_command": { "task": "blimp_principle_A_c_command", "dataset_path": "blimp", "dataset_name": "principle_A_c_command", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_principle_A_case_1": { "task": "blimp_principle_A_case_1", "dataset_path": "blimp", "dataset_name": "principle_A_case_1", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_principle_A_case_2": { "task": "blimp_principle_A_case_2", "dataset_path": "blimp", "dataset_name": "principle_A_case_2", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_principle_A_domain_1": { "task": "blimp_principle_A_domain_1", "dataset_path": "blimp", "dataset_name": "principle_A_domain_1", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_principle_A_domain_2": { "task": "blimp_principle_A_domain_2", "dataset_path": "blimp", "dataset_name": "principle_A_domain_2", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_principle_A_domain_3": { "task": "blimp_principle_A_domain_3", "dataset_path": "blimp", "dataset_name": "principle_A_domain_3", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_principle_A_reconstruction": { "task": "blimp_principle_A_reconstruction", "dataset_path": "blimp", "dataset_name": "principle_A_reconstruction", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_regular_plural_subject_verb_agreement_1": { "task": "blimp_regular_plural_subject_verb_agreement_1", "dataset_path": "blimp", "dataset_name": "regular_plural_subject_verb_agreement_1", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_regular_plural_subject_verb_agreement_2": { "task": "blimp_regular_plural_subject_verb_agreement_2", "dataset_path": "blimp", "dataset_name": "regular_plural_subject_verb_agreement_2", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_sentential_negation_npi_licensor_present": { "task": "blimp_sentential_negation_npi_licensor_present", "dataset_path": "blimp", "dataset_name": "sentential_negation_npi_licensor_present", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_sentential_negation_npi_scope": { "task": "blimp_sentential_negation_npi_scope", "dataset_path": "blimp", "dataset_name": "sentential_negation_npi_scope", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_sentential_subject_island": { "task": "blimp_sentential_subject_island", "dataset_path": "blimp", "dataset_name": "sentential_subject_island", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_superlative_quantifiers_1": { "task": "blimp_superlative_quantifiers_1", "dataset_path": "blimp", "dataset_name": "superlative_quantifiers_1", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_superlative_quantifiers_2": { "task": "blimp_superlative_quantifiers_2", "dataset_path": "blimp", "dataset_name": "superlative_quantifiers_2", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_tough_vs_raising_1": { "task": "blimp_tough_vs_raising_1", "dataset_path": "blimp", "dataset_name": "tough_vs_raising_1", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_tough_vs_raising_2": { "task": "blimp_tough_vs_raising_2", "dataset_path": "blimp", "dataset_name": "tough_vs_raising_2", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_transitive": { "task": "blimp_transitive", "dataset_path": "blimp", "dataset_name": "transitive", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_wh_island": { "task": "blimp_wh_island", "dataset_path": "blimp", "dataset_name": "wh_island", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_wh_questions_object_gap": { "task": "blimp_wh_questions_object_gap", "dataset_path": "blimp", "dataset_name": "wh_questions_object_gap", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_wh_questions_subject_gap": { "task": "blimp_wh_questions_subject_gap", "dataset_path": "blimp", "dataset_name": "wh_questions_subject_gap", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_wh_questions_subject_gap_long_distance": { "task": "blimp_wh_questions_subject_gap_long_distance", "dataset_path": "blimp", "dataset_name": "wh_questions_subject_gap_long_distance", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_wh_vs_that_no_gap": { "task": "blimp_wh_vs_that_no_gap", "dataset_path": "blimp", "dataset_name": "wh_vs_that_no_gap", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_wh_vs_that_no_gap_long_distance": { "task": "blimp_wh_vs_that_no_gap_long_distance", "dataset_path": "blimp", "dataset_name": "wh_vs_that_no_gap_long_distance", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_wh_vs_that_with_gap": { "task": "blimp_wh_vs_that_with_gap", "dataset_path": "blimp", "dataset_name": "wh_vs_that_with_gap", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } }, "blimp_wh_vs_that_with_gap_long_distance": { "task": "blimp_wh_vs_that_with_gap_long_distance", "dataset_path": "blimp", "dataset_name": "wh_vs_that_with_gap_long_distance", "validation_split": "train", "doc_to_text": "", "doc_to_target": 0, "unsafe_code": false, "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "description": "", "target_delimiter": " ", "fewshot_delimiter": "\n\n", "num_fewshot": 0, "metric_list": [ { "metric": "acc", "aggregation": "mean", "higher_is_better": true } ], "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "metadata": { "version": 1.0 } } }, "versions": { "blimp": 2.0, "blimp_adjunct_island": 1.0, "blimp_anaphor_gender_agreement": 1.0, "blimp_anaphor_number_agreement": 1.0, "blimp_animate_subject_passive": 1.0, "blimp_animate_subject_trans": 1.0, "blimp_causative": 1.0, "blimp_complex_NP_island": 1.0, "blimp_coordinate_structure_constraint_complex_left_branch": 1.0, "blimp_coordinate_structure_constraint_object_extraction": 1.0, "blimp_determiner_noun_agreement_1": 1.0, "blimp_determiner_noun_agreement_2": 1.0, "blimp_determiner_noun_agreement_irregular_1": 1.0, "blimp_determiner_noun_agreement_irregular_2": 1.0, "blimp_determiner_noun_agreement_with_adj_2": 1.0, "blimp_determiner_noun_agreement_with_adj_irregular_1": 1.0, "blimp_determiner_noun_agreement_with_adj_irregular_2": 1.0, "blimp_determiner_noun_agreement_with_adjective_1": 1.0, "blimp_distractor_agreement_relational_noun": 1.0, "blimp_distractor_agreement_relative_clause": 1.0, "blimp_drop_argument": 1.0, "blimp_ellipsis_n_bar_1": 1.0, "blimp_ellipsis_n_bar_2": 1.0, "blimp_existential_there_object_raising": 1.0, "blimp_existential_there_quantifiers_1": 1.0, "blimp_existential_there_quantifiers_2": 1.0, "blimp_existential_there_subject_raising": 1.0, "blimp_expletive_it_object_raising": 1.0, "blimp_inchoative": 1.0, "blimp_intransitive": 1.0, "blimp_irregular_past_participle_adjectives": 1.0, "blimp_irregular_past_participle_verbs": 1.0, "blimp_irregular_plural_subject_verb_agreement_1": 1.0, "blimp_irregular_plural_subject_verb_agreement_2": 1.0, "blimp_left_branch_island_echo_question": 1.0, "blimp_left_branch_island_simple_question": 1.0, "blimp_matrix_question_npi_licensor_present": 1.0, "blimp_npi_present_1": 1.0, "blimp_npi_present_2": 1.0, "blimp_only_npi_licensor_present": 1.0, "blimp_only_npi_scope": 1.0, "blimp_passive_1": 1.0, "blimp_passive_2": 1.0, "blimp_principle_A_c_command": 1.0, "blimp_principle_A_case_1": 1.0, "blimp_principle_A_case_2": 1.0, "blimp_principle_A_domain_1": 1.0, "blimp_principle_A_domain_2": 1.0, "blimp_principle_A_domain_3": 1.0, "blimp_principle_A_reconstruction": 1.0, "blimp_regular_plural_subject_verb_agreement_1": 1.0, "blimp_regular_plural_subject_verb_agreement_2": 1.0, "blimp_sentential_negation_npi_licensor_present": 1.0, "blimp_sentential_negation_npi_scope": 1.0, "blimp_sentential_subject_island": 1.0, "blimp_superlative_quantifiers_1": 1.0, "blimp_superlative_quantifiers_2": 1.0, "blimp_tough_vs_raising_1": 1.0, "blimp_tough_vs_raising_2": 1.0, "blimp_transitive": 1.0, "blimp_wh_island": 1.0, "blimp_wh_questions_object_gap": 1.0, "blimp_wh_questions_subject_gap": 1.0, "blimp_wh_questions_subject_gap_long_distance": 1.0, "blimp_wh_vs_that_no_gap": 1.0, "blimp_wh_vs_that_no_gap_long_distance": 1.0, "blimp_wh_vs_that_with_gap": 1.0, "blimp_wh_vs_that_with_gap_long_distance": 1.0 }, "n-shot": { "blimp_adjunct_island": 0, "blimp_anaphor_gender_agreement": 0, "blimp_anaphor_number_agreement": 0, "blimp_animate_subject_passive": 0, "blimp_animate_subject_trans": 0, "blimp_causative": 0, "blimp_complex_NP_island": 0, "blimp_coordinate_structure_constraint_complex_left_branch": 0, "blimp_coordinate_structure_constraint_object_extraction": 0, "blimp_determiner_noun_agreement_1": 0, "blimp_determiner_noun_agreement_2": 0, "blimp_determiner_noun_agreement_irregular_1": 0, "blimp_determiner_noun_agreement_irregular_2": 0, "blimp_determiner_noun_agreement_with_adj_2": 0, "blimp_determiner_noun_agreement_with_adj_irregular_1": 0, "blimp_determiner_noun_agreement_with_adj_irregular_2": 0, "blimp_determiner_noun_agreement_with_adjective_1": 0, "blimp_distractor_agreement_relational_noun": 0, "blimp_distractor_agreement_relative_clause": 0, "blimp_drop_argument": 0, "blimp_ellipsis_n_bar_1": 0, "blimp_ellipsis_n_bar_2": 0, "blimp_existential_there_object_raising": 0, "blimp_existential_there_quantifiers_1": 0, "blimp_existential_there_quantifiers_2": 0, "blimp_existential_there_subject_raising": 0, "blimp_expletive_it_object_raising": 0, "blimp_inchoative": 0, "blimp_intransitive": 0, "blimp_irregular_past_participle_adjectives": 0, "blimp_irregular_past_participle_verbs": 0, "blimp_irregular_plural_subject_verb_agreement_1": 0, "blimp_irregular_plural_subject_verb_agreement_2": 0, "blimp_left_branch_island_echo_question": 0, "blimp_left_branch_island_simple_question": 0, "blimp_matrix_question_npi_licensor_present": 0, "blimp_npi_present_1": 0, "blimp_npi_present_2": 0, "blimp_only_npi_licensor_present": 0, "blimp_only_npi_scope": 0, "blimp_passive_1": 0, "blimp_passive_2": 0, "blimp_principle_A_c_command": 0, "blimp_principle_A_case_1": 0, "blimp_principle_A_case_2": 0, "blimp_principle_A_domain_1": 0, "blimp_principle_A_domain_2": 0, "blimp_principle_A_domain_3": 0, "blimp_principle_A_reconstruction": 0, "blimp_regular_plural_subject_verb_agreement_1": 0, "blimp_regular_plural_subject_verb_agreement_2": 0, "blimp_sentential_negation_npi_licensor_present": 0, "blimp_sentential_negation_npi_scope": 0, "blimp_sentential_subject_island": 0, "blimp_superlative_quantifiers_1": 0, "blimp_superlative_quantifiers_2": 0, "blimp_tough_vs_raising_1": 0, "blimp_tough_vs_raising_2": 0, "blimp_transitive": 0, "blimp_wh_island": 0, "blimp_wh_questions_object_gap": 0, "blimp_wh_questions_subject_gap": 0, "blimp_wh_questions_subject_gap_long_distance": 0, "blimp_wh_vs_that_no_gap": 0, "blimp_wh_vs_that_no_gap_long_distance": 0, "blimp_wh_vs_that_with_gap": 0, "blimp_wh_vs_that_with_gap_long_distance": 0 }, "higher_is_better": { "blimp": { "acc": true }, "blimp_adjunct_island": { "acc": true }, "blimp_anaphor_gender_agreement": { "acc": true }, "blimp_anaphor_number_agreement": { "acc": true }, "blimp_animate_subject_passive": { "acc": true }, "blimp_animate_subject_trans": { "acc": true }, "blimp_causative": { "acc": true }, "blimp_complex_NP_island": { "acc": true }, "blimp_coordinate_structure_constraint_complex_left_branch": { "acc": true }, "blimp_coordinate_structure_constraint_object_extraction": { "acc": true }, "blimp_determiner_noun_agreement_1": { "acc": true }, "blimp_determiner_noun_agreement_2": { "acc": true }, "blimp_determiner_noun_agreement_irregular_1": { "acc": true }, "blimp_determiner_noun_agreement_irregular_2": { "acc": true }, "blimp_determiner_noun_agreement_with_adj_2": { "acc": true }, "blimp_determiner_noun_agreement_with_adj_irregular_1": { "acc": true }, "blimp_determiner_noun_agreement_with_adj_irregular_2": { "acc": true }, "blimp_determiner_noun_agreement_with_adjective_1": { "acc": true }, "blimp_distractor_agreement_relational_noun": { "acc": true }, "blimp_distractor_agreement_relative_clause": { "acc": true }, "blimp_drop_argument": { "acc": true }, "blimp_ellipsis_n_bar_1": { "acc": true }, "blimp_ellipsis_n_bar_2": { "acc": true }, "blimp_existential_there_object_raising": { "acc": true }, "blimp_existential_there_quantifiers_1": { "acc": true }, "blimp_existential_there_quantifiers_2": { "acc": true }, "blimp_existential_there_subject_raising": { "acc": true }, "blimp_expletive_it_object_raising": { "acc": true }, "blimp_inchoative": { "acc": true }, "blimp_intransitive": { "acc": true }, "blimp_irregular_past_participle_adjectives": { "acc": true }, "blimp_irregular_past_participle_verbs": { "acc": true }, "blimp_irregular_plural_subject_verb_agreement_1": { "acc": true }, "blimp_irregular_plural_subject_verb_agreement_2": { "acc": true }, "blimp_left_branch_island_echo_question": { "acc": true }, "blimp_left_branch_island_simple_question": { "acc": true }, "blimp_matrix_question_npi_licensor_present": { "acc": true }, "blimp_npi_present_1": { "acc": true }, "blimp_npi_present_2": { "acc": true }, "blimp_only_npi_licensor_present": { "acc": true }, "blimp_only_npi_scope": { "acc": true }, "blimp_passive_1": { "acc": true }, "blimp_passive_2": { "acc": true }, "blimp_principle_A_c_command": { "acc": true }, "blimp_principle_A_case_1": { "acc": true }, "blimp_principle_A_case_2": { "acc": true }, "blimp_principle_A_domain_1": { "acc": true }, "blimp_principle_A_domain_2": { "acc": true }, "blimp_principle_A_domain_3": { "acc": true }, "blimp_principle_A_reconstruction": { "acc": true }, "blimp_regular_plural_subject_verb_agreement_1": { "acc": true }, "blimp_regular_plural_subject_verb_agreement_2": { "acc": true }, "blimp_sentential_negation_npi_licensor_present": { "acc": true }, "blimp_sentential_negation_npi_scope": { "acc": true }, "blimp_sentential_subject_island": { "acc": true }, "blimp_superlative_quantifiers_1": { "acc": true }, "blimp_superlative_quantifiers_2": { "acc": true }, "blimp_tough_vs_raising_1": { "acc": true }, "blimp_tough_vs_raising_2": { "acc": true }, "blimp_transitive": { "acc": true }, "blimp_wh_island": { "acc": true }, "blimp_wh_questions_object_gap": { "acc": true }, "blimp_wh_questions_subject_gap": { "acc": true }, "blimp_wh_questions_subject_gap_long_distance": { "acc": true }, "blimp_wh_vs_that_no_gap": { "acc": true }, "blimp_wh_vs_that_no_gap_long_distance": { "acc": true }, "blimp_wh_vs_that_with_gap": { "acc": true }, "blimp_wh_vs_that_with_gap_long_distance": { "acc": true } }, "n-samples": { "blimp_adjunct_island": { "original": 1000, "effective": 1000 }, "blimp_anaphor_gender_agreement": { "original": 1000, "effective": 1000 }, "blimp_anaphor_number_agreement": { "original": 1000, "effective": 1000 }, "blimp_animate_subject_passive": { "original": 1000, "effective": 1000 }, "blimp_animate_subject_trans": { "original": 1000, "effective": 1000 }, "blimp_causative": { "original": 1000, "effective": 1000 }, "blimp_complex_NP_island": { "original": 1000, "effective": 1000 }, "blimp_coordinate_structure_constraint_complex_left_branch": { "original": 1000, "effective": 1000 }, "blimp_coordinate_structure_constraint_object_extraction": { "original": 1000, "effective": 1000 }, "blimp_determiner_noun_agreement_1": { "original": 1000, "effective": 1000 }, "blimp_determiner_noun_agreement_2": { "original": 1000, "effective": 1000 }, "blimp_determiner_noun_agreement_irregular_1": { "original": 1000, "effective": 1000 }, "blimp_determiner_noun_agreement_irregular_2": { "original": 1000, "effective": 1000 }, "blimp_determiner_noun_agreement_with_adj_2": { "original": 1000, "effective": 1000 }, "blimp_determiner_noun_agreement_with_adj_irregular_1": { "original": 1000, "effective": 1000 }, "blimp_determiner_noun_agreement_with_adj_irregular_2": { "original": 1000, "effective": 1000 }, "blimp_determiner_noun_agreement_with_adjective_1": { "original": 1000, "effective": 1000 }, "blimp_distractor_agreement_relational_noun": { "original": 1000, "effective": 1000 }, "blimp_distractor_agreement_relative_clause": { "original": 1000, "effective": 1000 }, "blimp_drop_argument": { "original": 1000, "effective": 1000 }, "blimp_ellipsis_n_bar_1": { "original": 1000, "effective": 1000 }, "blimp_ellipsis_n_bar_2": { "original": 1000, "effective": 1000 }, "blimp_existential_there_object_raising": { "original": 1000, "effective": 1000 }, "blimp_existential_there_quantifiers_1": { "original": 1000, "effective": 1000 }, "blimp_existential_there_quantifiers_2": { "original": 1000, "effective": 1000 }, "blimp_existential_there_subject_raising": { "original": 1000, "effective": 1000 }, "blimp_expletive_it_object_raising": { "original": 1000, "effective": 1000 }, "blimp_inchoative": { "original": 1000, "effective": 1000 }, "blimp_intransitive": { "original": 1000, "effective": 1000 }, "blimp_irregular_past_participle_adjectives": { "original": 1000, "effective": 1000 }, "blimp_irregular_past_participle_verbs": { "original": 1000, "effective": 1000 }, "blimp_irregular_plural_subject_verb_agreement_1": { "original": 1000, "effective": 1000 }, "blimp_irregular_plural_subject_verb_agreement_2": { "original": 1000, "effective": 1000 }, "blimp_left_branch_island_echo_question": { "original": 1000, "effective": 1000 }, "blimp_left_branch_island_simple_question": { "original": 1000, "effective": 1000 }, "blimp_matrix_question_npi_licensor_present": { "original": 1000, "effective": 1000 }, "blimp_npi_present_1": { "original": 1000, "effective": 1000 }, "blimp_npi_present_2": { "original": 1000, "effective": 1000 }, "blimp_only_npi_licensor_present": { "original": 1000, "effective": 1000 }, "blimp_only_npi_scope": { "original": 1000, "effective": 1000 }, "blimp_passive_1": { "original": 1000, "effective": 1000 }, "blimp_passive_2": { "original": 1000, "effective": 1000 }, "blimp_principle_A_c_command": { "original": 1000, "effective": 1000 }, "blimp_principle_A_case_1": { "original": 1000, "effective": 1000 }, "blimp_principle_A_case_2": { "original": 1000, "effective": 1000 }, "blimp_principle_A_domain_1": { "original": 1000, "effective": 1000 }, "blimp_principle_A_domain_2": { "original": 1000, "effective": 1000 }, "blimp_principle_A_domain_3": { "original": 1000, "effective": 1000 }, "blimp_principle_A_reconstruction": { "original": 1000, "effective": 1000 }, "blimp_regular_plural_subject_verb_agreement_1": { "original": 1000, "effective": 1000 }, "blimp_regular_plural_subject_verb_agreement_2": { "original": 1000, "effective": 1000 }, "blimp_sentential_negation_npi_licensor_present": { "original": 1000, "effective": 1000 }, "blimp_sentential_negation_npi_scope": { "original": 1000, "effective": 1000 }, "blimp_sentential_subject_island": { "original": 1000, "effective": 1000 }, "blimp_superlative_quantifiers_1": { "original": 1000, "effective": 1000 }, "blimp_superlative_quantifiers_2": { "original": 1000, "effective": 1000 }, "blimp_tough_vs_raising_1": { "original": 1000, "effective": 1000 }, "blimp_tough_vs_raising_2": { "original": 1000, "effective": 1000 }, "blimp_transitive": { "original": 1000, "effective": 1000 }, "blimp_wh_island": { "original": 1000, "effective": 1000 }, "blimp_wh_questions_object_gap": { "original": 1000, "effective": 1000 }, "blimp_wh_questions_subject_gap": { "original": 1000, "effective": 1000 }, "blimp_wh_questions_subject_gap_long_distance": { "original": 1000, "effective": 1000 }, "blimp_wh_vs_that_no_gap": { "original": 1000, "effective": 1000 }, "blimp_wh_vs_that_no_gap_long_distance": { "original": 1000, "effective": 1000 }, "blimp_wh_vs_that_with_gap": { "original": 1000, "effective": 1000 }, "blimp_wh_vs_that_with_gap_long_distance": { "original": 1000, "effective": 1000 } }, "config": { "model": "hf", "model_args": "pretrained=outputs/fw57M-tied/42/fw57M_Surprisal_bytespanP0-5T30_64000/.cache/eval_model", "model_num_parameters": 105785088, "model_dtype": "torch.bfloat16", "model_revision": "main", "model_sha": "", "batch_size": 1, "batch_sizes": [], "device": null, "use_cache": null, "limit": null, "bootstrap_iters": 100000, "gen_kwargs": null, "random_seed": 0, "numpy_seed": 1234, "torch_seed": 1234, "fewshot_seed": 1234 }, "git_hash": "778f288", "date": 1749634957.8060253, "pretty_env_info": "'NoneType' object has no attribute 'splitlines'", "transformers_version": "4.52.4", "upper_git_hash": null, "tokenizer_pad_token": [ "<|padding|>", "0" ], "tokenizer_eos_token": [ "<|endoftext|>", "1" ], "tokenizer_bos_token": [ "<|endoftext|>", "1" ], "eot_token_id": 1, "max_length": 2048, "task_hashes": {}, "model_source": "hf", "model_name": "outputs/fw57M-tied/42/fw57M_Surprisal_bytespanP0-5T30_64000/.cache/eval_model", "model_name_sanitized": "outputs__fw57M-tied__42__fw57M_Surprisal_bytespanP0-5T30_64000__.cache__eval_model", "system_instruction": null, "system_instruction_sha": null, "fewshot_as_multiturn": false, "chat_template": null, "chat_template_sha": null, "start_time": 292086.209868551, "end_time": 293447.576254385, "total_evaluation_time_seconds": "1361.3663858340005" }