{ "config_general": { "lighteval_sha": "?", "num_fewshot_seeds": 1, "override_batch_size": 1, "max_samples": null, "job_id": "", "start_time": 749130.091599447, "end_time": 749875.457005619, "total_evaluation_time_secondes": "745.3654061720008", "model_name": "alignment-handbook/zephyr-2b-gemma-dpo-mix6-beta-0.05", "model_sha": "c7eb627c6e4167652aa357d22b4ce2a1b8e2d8b3", "model_dtype": "torch.bfloat16", "model_size": "4.68 GB", "config": null }, "results": { "lighteval|mmlu:abstract_algebra|5": { "acc": 0.27, "acc_stderr": 0.0446196043338474 }, "lighteval|mmlu:anatomy|5": { "acc": 0.4444444444444444, "acc_stderr": 0.04292596718256981 }, "lighteval|mmlu:astronomy|5": { "acc": 0.3815789473684211, "acc_stderr": 0.03953173377749194 }, "lighteval|mmlu:business_ethics|5": { "acc": 0.4, "acc_stderr": 0.04923659639173309 }, "lighteval|mmlu:clinical_knowledge|5": { "acc": 0.4716981132075472, "acc_stderr": 0.030723535249006107 }, "lighteval|mmlu:college_biology|5": { "acc": 0.4375, "acc_stderr": 0.04148415739394154 }, "lighteval|mmlu:college_chemistry|5": { "acc": 0.29, "acc_stderr": 0.04560480215720683 }, "lighteval|mmlu:college_computer_science|5": { "acc": 0.33, "acc_stderr": 0.047258156262526045 }, "lighteval|mmlu:college_mathematics|5": { "acc": 0.36, "acc_stderr": 0.04824181513244218 }, "lighteval|mmlu:college_medicine|5": { "acc": 0.37572254335260113, "acc_stderr": 0.03692820767264867 }, "lighteval|mmlu:college_physics|5": { "acc": 0.19607843137254902, "acc_stderr": 0.03950581861179963 }, "lighteval|mmlu:computer_security|5": { "acc": 0.54, "acc_stderr": 0.05009082659620332 }, "lighteval|mmlu:conceptual_physics|5": { "acc": 0.33617021276595743, "acc_stderr": 0.030881618520676942 }, "lighteval|mmlu:econometrics|5": { "acc": 0.30701754385964913, "acc_stderr": 0.0433913832257986 }, "lighteval|mmlu:electrical_engineering|5": { "acc": 0.46206896551724136, "acc_stderr": 0.041546596717075474 }, "lighteval|mmlu:elementary_mathematics|5": { "acc": 0.30158730158730157, "acc_stderr": 0.023636975996101796 }, "lighteval|mmlu:formal_logic|5": { "acc": 0.38095238095238093, "acc_stderr": 0.043435254289490986 }, "lighteval|mmlu:global_facts|5": { "acc": 0.29, "acc_stderr": 0.045604802157206845 }, "lighteval|mmlu:high_school_biology|5": { "acc": 0.3967741935483871, "acc_stderr": 0.02783123160576795 }, "lighteval|mmlu:high_school_chemistry|5": { "acc": 0.3399014778325123, "acc_stderr": 0.0333276906841079 }, "lighteval|mmlu:high_school_computer_science|5": { "acc": 0.43, "acc_stderr": 0.049756985195624284 }, "lighteval|mmlu:high_school_european_history|5": { "acc": 0.38181818181818183, "acc_stderr": 0.03793713171165633 }, "lighteval|mmlu:high_school_geography|5": { "acc": 0.51010101010101, "acc_stderr": 0.035616254886737454 }, "lighteval|mmlu:high_school_government_and_politics|5": { "acc": 0.5077720207253886, "acc_stderr": 0.03608003225569653 }, "lighteval|mmlu:high_school_macroeconomics|5": { "acc": 0.35128205128205126, "acc_stderr": 0.024203665177902796 }, "lighteval|mmlu:high_school_mathematics|5": { "acc": 0.27037037037037037, "acc_stderr": 0.02708037281514566 }, "lighteval|mmlu:high_school_microeconomics|5": { "acc": 0.37815126050420167, "acc_stderr": 0.031499305777849054 }, "lighteval|mmlu:high_school_physics|5": { "acc": 0.2847682119205298, "acc_stderr": 0.03684881521389023 }, "lighteval|mmlu:high_school_psychology|5": { "acc": 0.5522935779816514, "acc_stderr": 0.021319754962425455 }, "lighteval|mmlu:high_school_statistics|5": { "acc": 0.26851851851851855, "acc_stderr": 0.0302252261600124 }, "lighteval|mmlu:high_school_us_history|5": { "acc": 0.3284313725490196, "acc_stderr": 0.03296245110172229 }, "lighteval|mmlu:high_school_world_history|5": { "acc": 0.38396624472573837, "acc_stderr": 0.031658678064106674 }, "lighteval|mmlu:human_aging|5": { "acc": 0.3991031390134529, "acc_stderr": 0.03286745312567961 }, "lighteval|mmlu:human_sexuality|5": { "acc": 0.5114503816793893, "acc_stderr": 0.043841400240780176 }, "lighteval|mmlu:international_law|5": { "acc": 0.5867768595041323, "acc_stderr": 0.04495087843548408 }, "lighteval|mmlu:jurisprudence|5": { "acc": 0.4537037037037037, "acc_stderr": 0.04812917324536823 }, "lighteval|mmlu:logical_fallacies|5": { "acc": 0.39263803680981596, "acc_stderr": 0.03836740907831029 }, "lighteval|mmlu:machine_learning|5": { "acc": 0.42857142857142855, "acc_stderr": 0.04697113923010212 }, "lighteval|mmlu:management|5": { "acc": 0.5339805825242718, "acc_stderr": 0.04939291447273481 }, "lighteval|mmlu:marketing|5": { "acc": 0.6025641025641025, "acc_stderr": 0.03205953453789293 }, "lighteval|mmlu:medical_genetics|5": { "acc": 0.43, "acc_stderr": 0.049756985195624284 }, "lighteval|mmlu:miscellaneous|5": { "acc": 0.5274584929757343, "acc_stderr": 0.017852981266633948 }, "lighteval|mmlu:moral_disputes|5": { "acc": 0.430635838150289, "acc_stderr": 0.026658800273672373 }, "lighteval|mmlu:moral_scenarios|5": { "acc": 0.23798882681564246, "acc_stderr": 0.014242630070574903 }, "lighteval|mmlu:nutrition|5": { "acc": 0.48366013071895425, "acc_stderr": 0.028614624752805403 }, "lighteval|mmlu:philosophy|5": { "acc": 0.4340836012861736, "acc_stderr": 0.02815023224453559 }, "lighteval|mmlu:prehistory|5": { "acc": 0.43209876543209874, "acc_stderr": 0.02756301097160668 }, "lighteval|mmlu:professional_accounting|5": { "acc": 0.30141843971631205, "acc_stderr": 0.027374128882631157 }, "lighteval|mmlu:professional_law|5": { "acc": 0.31747066492829207, "acc_stderr": 0.01188889206880931 }, "lighteval|mmlu:professional_medicine|5": { "acc": 0.29411764705882354, "acc_stderr": 0.027678468642144707 }, "lighteval|mmlu:professional_psychology|5": { "acc": 0.3660130718954248, "acc_stderr": 0.019488025745529658 }, "lighteval|mmlu:public_relations|5": { "acc": 0.37272727272727274, "acc_stderr": 0.046313813194254635 }, "lighteval|mmlu:security_studies|5": { "acc": 0.5020408163265306, "acc_stderr": 0.0320089533497105 }, "lighteval|mmlu:sociology|5": { "acc": 0.5522388059701493, "acc_stderr": 0.03516184772952167 }, "lighteval|mmlu:us_foreign_policy|5": { "acc": 0.58, "acc_stderr": 0.049604496374885836 }, "lighteval|mmlu:virology|5": { "acc": 0.3855421686746988, "acc_stderr": 0.03789134424611548 }, "lighteval|mmlu:world_religions|5": { "acc": 0.52046783625731, "acc_stderr": 0.038316105328219316 }, "lighteval|mmlu:_average|5": { "acc": 0.4041354033264851, "acc_stderr": 0.03607264368393054 } }, "versions": { "lighteval|mmlu:abstract_algebra|5": 0, "lighteval|mmlu:anatomy|5": 0, "lighteval|mmlu:astronomy|5": 0, "lighteval|mmlu:business_ethics|5": 0, "lighteval|mmlu:clinical_knowledge|5": 0, "lighteval|mmlu:college_biology|5": 0, "lighteval|mmlu:college_chemistry|5": 0, "lighteval|mmlu:college_computer_science|5": 0, "lighteval|mmlu:college_mathematics|5": 0, "lighteval|mmlu:college_medicine|5": 0, "lighteval|mmlu:college_physics|5": 0, "lighteval|mmlu:computer_security|5": 0, "lighteval|mmlu:conceptual_physics|5": 0, "lighteval|mmlu:econometrics|5": 0, "lighteval|mmlu:electrical_engineering|5": 0, "lighteval|mmlu:elementary_mathematics|5": 0, "lighteval|mmlu:formal_logic|5": 0, "lighteval|mmlu:global_facts|5": 0, "lighteval|mmlu:high_school_biology|5": 0, "lighteval|mmlu:high_school_chemistry|5": 0, "lighteval|mmlu:high_school_computer_science|5": 0, "lighteval|mmlu:high_school_european_history|5": 0, "lighteval|mmlu:high_school_geography|5": 0, "lighteval|mmlu:high_school_government_and_politics|5": 0, "lighteval|mmlu:high_school_macroeconomics|5": 0, "lighteval|mmlu:high_school_mathematics|5": 0, "lighteval|mmlu:high_school_microeconomics|5": 0, "lighteval|mmlu:high_school_physics|5": 0, "lighteval|mmlu:high_school_psychology|5": 0, "lighteval|mmlu:high_school_statistics|5": 0, "lighteval|mmlu:high_school_us_history|5": 0, "lighteval|mmlu:high_school_world_history|5": 0, "lighteval|mmlu:human_aging|5": 0, "lighteval|mmlu:human_sexuality|5": 0, "lighteval|mmlu:international_law|5": 0, "lighteval|mmlu:jurisprudence|5": 0, "lighteval|mmlu:logical_fallacies|5": 0, "lighteval|mmlu:machine_learning|5": 0, "lighteval|mmlu:management|5": 0, "lighteval|mmlu:marketing|5": 0, "lighteval|mmlu:medical_genetics|5": 0, "lighteval|mmlu:miscellaneous|5": 0, "lighteval|mmlu:moral_disputes|5": 0, "lighteval|mmlu:moral_scenarios|5": 0, "lighteval|mmlu:nutrition|5": 0, "lighteval|mmlu:philosophy|5": 0, "lighteval|mmlu:prehistory|5": 0, "lighteval|mmlu:professional_accounting|5": 0, "lighteval|mmlu:professional_law|5": 0, "lighteval|mmlu:professional_medicine|5": 0, "lighteval|mmlu:professional_psychology|5": 0, "lighteval|mmlu:public_relations|5": 0, "lighteval|mmlu:security_studies|5": 0, "lighteval|mmlu:sociology|5": 0, "lighteval|mmlu:us_foreign_policy|5": 0, "lighteval|mmlu:virology|5": 0, "lighteval|mmlu:world_religions|5": 0 }, "config_tasks": { "lighteval|mmlu:abstract_algebra": { "name": "mmlu:abstract_algebra", "prompt_function": "mmlu_harness", "hf_repo": "lighteval/mmlu", "hf_subset": "abstract_algebra", "metric": [ "loglikelihood_acc" ], "hf_avail_splits": [ "auxiliary_train", "test", "validation", "dev" ], "evaluation_splits": [ "test" ], "few_shots_split": "dev", "few_shots_select": "sequential", "generation_size": 1, "stop_sequence": [ "\n" ], "output_regex": null, "frozen": false, "suite": [ "lighteval", "mmlu" ], "original_num_docs": 100, "effective_num_docs": 100 }, "lighteval|mmlu:anatomy": { "name": "mmlu:anatomy", "prompt_function": "mmlu_harness", "hf_repo": "lighteval/mmlu", "hf_subset": "anatomy", "metric": [ "loglikelihood_acc" ], "hf_avail_splits": [ "auxiliary_train", "test", "validation", "dev" ], "evaluation_splits": [ "test" ], "few_shots_split": "dev", "few_shots_select": "sequential", "generation_size": 1, "stop_sequence": [ "\n" ], "output_regex": null, "frozen": false, "suite": [ "lighteval", "mmlu" ], "original_num_docs": 135, "effective_num_docs": 135 }, "lighteval|mmlu:astronomy": { "name": "mmlu:astronomy", "prompt_function": "mmlu_harness", "hf_repo": "lighteval/mmlu", "hf_subset": "astronomy", "metric": [ "loglikelihood_acc" ], "hf_avail_splits": [ "auxiliary_train", "test", "validation", "dev" ], "evaluation_splits": [ "test" ], "few_shots_split": "dev", "few_shots_select": "sequential", "generation_size": 1, "stop_sequence": [ "\n" ], "output_regex": null, "frozen": false, "suite": [ "lighteval", "mmlu" ], "original_num_docs": 152, "effective_num_docs": 152 }, "lighteval|mmlu:business_ethics": { "name": "mmlu:business_ethics", "prompt_function": "mmlu_harness", "hf_repo": "lighteval/mmlu", "hf_subset": "business_ethics", "metric": [ "loglikelihood_acc" ], "hf_avail_splits": [ "auxiliary_train", "test", "validation", "dev" ], "evaluation_splits": [ "test" ], "few_shots_split": "dev", "few_shots_select": "sequential", "generation_size": 1, "stop_sequence": [ "\n" ], "output_regex": null, "frozen": false, "suite": [ "lighteval", "mmlu" ], "original_num_docs": 100, "effective_num_docs": 100 }, "lighteval|mmlu:clinical_knowledge": { "name": "mmlu:clinical_knowledge", "prompt_function": "mmlu_harness", "hf_repo": "lighteval/mmlu", "hf_subset": "clinical_knowledge", "metric": [ "loglikelihood_acc" ], "hf_avail_splits": [ "auxiliary_train", "test", "validation", "dev" ], "evaluation_splits": [ "test" ], "few_shots_split": "dev", "few_shots_select": "sequential", "generation_size": 1, "stop_sequence": [ "\n" ], "output_regex": null, "frozen": false, "suite": [ "lighteval", "mmlu" ], "original_num_docs": 265, "effective_num_docs": 265 }, "lighteval|mmlu:college_biology": { "name": "mmlu:college_biology", "prompt_function": "mmlu_harness", "hf_repo": "lighteval/mmlu", "hf_subset": "college_biology", "metric": [ "loglikelihood_acc" ], "hf_avail_splits": [ "auxiliary_train", "test", "validation", "dev" ], "evaluation_splits": [ "test" ], "few_shots_split": "dev", "few_shots_select": "sequential", "generation_size": 1, "stop_sequence": [ "\n" ], "output_regex": null, "frozen": false, "suite": [ "lighteval", "mmlu" ], "original_num_docs": 144, "effective_num_docs": 144 }, "lighteval|mmlu:college_chemistry": { "name": "mmlu:college_chemistry", "prompt_function": "mmlu_harness", "hf_repo": "lighteval/mmlu", "hf_subset": "college_chemistry", "metric": [ "loglikelihood_acc" ], "hf_avail_splits": [ "auxiliary_train", "test", "validation", "dev" ], "evaluation_splits": [ "test" ], "few_shots_split": "dev", "few_shots_select": "sequential", "generation_size": 1, "stop_sequence": [ "\n" ], "output_regex": null, "frozen": false, "suite": [ "lighteval", "mmlu" ], "original_num_docs": 100, "effective_num_docs": 100 }, "lighteval|mmlu:college_computer_science": { "name": "mmlu:college_computer_science", "prompt_function": "mmlu_harness", "hf_repo": "lighteval/mmlu", "hf_subset": "college_computer_science", "metric": [ "loglikelihood_acc" ], "hf_avail_splits": [ "auxiliary_train", "test", "validation", "dev" ], "evaluation_splits": [ "test" ], "few_shots_split": "dev", "few_shots_select": "sequential", "generation_size": 1, "stop_sequence": [ "\n" ], "output_regex": null, "frozen": false, "suite": [ "lighteval", "mmlu" ], "original_num_docs": 100, "effective_num_docs": 100 }, "lighteval|mmlu:college_mathematics": { "name": "mmlu:college_mathematics", "prompt_function": "mmlu_harness", "hf_repo": "lighteval/mmlu", "hf_subset": "college_mathematics", "metric": [ "loglikelihood_acc" ], "hf_avail_splits": [ "auxiliary_train", "test", "validation", "dev" ], "evaluation_splits": [ "test" ], "few_shots_split": "dev", "few_shots_select": "sequential", "generation_size": 1, "stop_sequence": [ "\n" ], "output_regex": null, "frozen": false, "suite": [ "lighteval", "mmlu" ], "original_num_docs": 100, "effective_num_docs": 100 }, "lighteval|mmlu:college_medicine": { "name": "mmlu:college_medicine", "prompt_function": "mmlu_harness", "hf_repo": "lighteval/mmlu", "hf_subset": "college_medicine", "metric": [ "loglikelihood_acc" ], "hf_avail_splits": [ "auxiliary_train", "test", "validation", "dev" ], "evaluation_splits": [ "test" ], "few_shots_split": "dev", "few_shots_select": "sequential", "generation_size": 1, "stop_sequence": [ "\n" ], "output_regex": null, "frozen": false, "suite": [ "lighteval", "mmlu" ], "original_num_docs": 173, "effective_num_docs": 173 }, "lighteval|mmlu:college_physics": { "name": "mmlu:college_physics", "prompt_function": "mmlu_harness", "hf_repo": "lighteval/mmlu", "hf_subset": "college_physics", "metric": [ "loglikelihood_acc" ], "hf_avail_splits": [ "auxiliary_train", "test", "validation", "dev" ], "evaluation_splits": [ "test" ], "few_shots_split": "dev", "few_shots_select": "sequential", "generation_size": 1, "stop_sequence": [ "\n" ], "output_regex": null, "frozen": false, "suite": [ "lighteval", "mmlu" ], "original_num_docs": 102, "effective_num_docs": 102 }, "lighteval|mmlu:computer_security": { "name": "mmlu:computer_security", "prompt_function": "mmlu_harness", "hf_repo": "lighteval/mmlu", "hf_subset": "computer_security", "metric": [ "loglikelihood_acc" ], "hf_avail_splits": [ "auxiliary_train", "test", "validation", "dev" ], "evaluation_splits": [ "test" ], "few_shots_split": "dev", "few_shots_select": "sequential", "generation_size": 1, "stop_sequence": [ "\n" ], "output_regex": null, "frozen": false, "suite": [ "lighteval", "mmlu" ], "original_num_docs": 100, "effective_num_docs": 100 }, "lighteval|mmlu:conceptual_physics": { "name": "mmlu:conceptual_physics", "prompt_function": "mmlu_harness", "hf_repo": "lighteval/mmlu", "hf_subset": "conceptual_physics", "metric": [ "loglikelihood_acc" ], "hf_avail_splits": [ "auxiliary_train", "test", "validation", "dev" ], "evaluation_splits": [ "test" ], "few_shots_split": "dev", "few_shots_select": "sequential", "generation_size": 1, "stop_sequence": [ "\n" ], "output_regex": null, "frozen": false, "suite": [ "lighteval", "mmlu" ], "original_num_docs": 235, "effective_num_docs": 235 }, "lighteval|mmlu:econometrics": { "name": "mmlu:econometrics", "prompt_function": "mmlu_harness", "hf_repo": "lighteval/mmlu", "hf_subset": "econometrics", "metric": [ "loglikelihood_acc" ], "hf_avail_splits": [ "auxiliary_train", "test", "validation", "dev" ], "evaluation_splits": [ "test" ], "few_shots_split": "dev", "few_shots_select": "sequential", "generation_size": 1, "stop_sequence": [ "\n" ], "output_regex": null, "frozen": false, "suite": [ "lighteval", "mmlu" ], "original_num_docs": 114, "effective_num_docs": 114 }, "lighteval|mmlu:electrical_engineering": { "name": "mmlu:electrical_engineering", "prompt_function": "mmlu_harness", "hf_repo": "lighteval/mmlu", "hf_subset": "electrical_engineering", "metric": [ "loglikelihood_acc" ], "hf_avail_splits": [ "auxiliary_train", "test", "validation", "dev" ], "evaluation_splits": [ "test" ], "few_shots_split": "dev", "few_shots_select": "sequential", "generation_size": 1, "stop_sequence": [ "\n" ], "output_regex": null, "frozen": false, "suite": [ "lighteval", "mmlu" ], "original_num_docs": 145, "effective_num_docs": 145 }, "lighteval|mmlu:elementary_mathematics": { "name": "mmlu:elementary_mathematics", "prompt_function": "mmlu_harness", "hf_repo": "lighteval/mmlu", "hf_subset": "elementary_mathematics", "metric": [ "loglikelihood_acc" ], "hf_avail_splits": [ "auxiliary_train", "test", "validation", "dev" ], "evaluation_splits": [ "test" ], "few_shots_split": "dev", "few_shots_select": "sequential", "generation_size": 1, "stop_sequence": [ "\n" ], "output_regex": null, "frozen": false, "suite": [ "lighteval", "mmlu" ], "original_num_docs": 378, "effective_num_docs": 378 }, "lighteval|mmlu:formal_logic": { "name": "mmlu:formal_logic", "prompt_function": "mmlu_harness", "hf_repo": "lighteval/mmlu", "hf_subset": "formal_logic", "metric": [ "loglikelihood_acc" ], "hf_avail_splits": [ "auxiliary_train", "test", "validation", "dev" ], "evaluation_splits": [ "test" ], "few_shots_split": "dev", "few_shots_select": "sequential", "generation_size": 1, "stop_sequence": [ "\n" ], "output_regex": null, "frozen": false, "suite": [ "lighteval", "mmlu" ], "original_num_docs": 126, "effective_num_docs": 126 }, "lighteval|mmlu:global_facts": { "name": "mmlu:global_facts", "prompt_function": "mmlu_harness", "hf_repo": "lighteval/mmlu", "hf_subset": "global_facts", "metric": [ "loglikelihood_acc" ], "hf_avail_splits": [ "auxiliary_train", "test", "validation", "dev" ], "evaluation_splits": [ "test" ], "few_shots_split": "dev", "few_shots_select": "sequential", "generation_size": 1, "stop_sequence": [ "\n" ], "output_regex": null, "frozen": false, "suite": [ "lighteval", "mmlu" ], "original_num_docs": 100, "effective_num_docs": 100 }, "lighteval|mmlu:high_school_biology": { "name": "mmlu:high_school_biology", "prompt_function": "mmlu_harness", "hf_repo": "lighteval/mmlu", "hf_subset": "high_school_biology", "metric": [ "loglikelihood_acc" ], "hf_avail_splits": [ "auxiliary_train", "test", "validation", "dev" ], "evaluation_splits": [ "test" ], "few_shots_split": "dev", "few_shots_select": "sequential", "generation_size": 1, "stop_sequence": [ "\n" ], "output_regex": null, "frozen": false, "suite": [ "lighteval", "mmlu" ], "original_num_docs": 310, "effective_num_docs": 310 }, "lighteval|mmlu:high_school_chemistry": { "name": "mmlu:high_school_chemistry", "prompt_function": "mmlu_harness", "hf_repo": "lighteval/mmlu", "hf_subset": "high_school_chemistry", "metric": [ "loglikelihood_acc" ], "hf_avail_splits": [ "auxiliary_train", "test", "validation", "dev" ], "evaluation_splits": [ "test" ], "few_shots_split": "dev", "few_shots_select": "sequential", "generation_size": 1, "stop_sequence": [ "\n" ], "output_regex": null, "frozen": false, "suite": [ "lighteval", "mmlu" ], "original_num_docs": 203, "effective_num_docs": 203 }, "lighteval|mmlu:high_school_computer_science": { "name": "mmlu:high_school_computer_science", "prompt_function": "mmlu_harness", "hf_repo": "lighteval/mmlu", "hf_subset": "high_school_computer_science", "metric": [ "loglikelihood_acc" ], "hf_avail_splits": [ "auxiliary_train", "test", "validation", "dev" ], "evaluation_splits": [ "test" ], "few_shots_split": "dev", "few_shots_select": "sequential", "generation_size": 1, "stop_sequence": [ "\n" ], "output_regex": null, "frozen": false, "suite": [ "lighteval", "mmlu" ], "original_num_docs": 100, "effective_num_docs": 100 }, "lighteval|mmlu:high_school_european_history": { "name": "mmlu:high_school_european_history", "prompt_function": "mmlu_harness", "hf_repo": "lighteval/mmlu", "hf_subset": "high_school_european_history", "metric": [ "loglikelihood_acc" ], "hf_avail_splits": [ "auxiliary_train", "test", "validation", "dev" ], "evaluation_splits": [ "test" ], "few_shots_split": "dev", "few_shots_select": "sequential", "generation_size": 1, "stop_sequence": [ "\n" ], "output_regex": null, "frozen": false, "suite": [ "lighteval", "mmlu" ], "original_num_docs": 165, "effective_num_docs": 165 }, "lighteval|mmlu:high_school_geography": { "name": "mmlu:high_school_geography", "prompt_function": "mmlu_harness", "hf_repo": "lighteval/mmlu", "hf_subset": "high_school_geography", "metric": [ "loglikelihood_acc" ], "hf_avail_splits": [ "auxiliary_train", "test", "validation", "dev" ], "evaluation_splits": [ "test" ], "few_shots_split": "dev", "few_shots_select": "sequential", "generation_size": 1, "stop_sequence": [ "\n" ], "output_regex": null, "frozen": false, "suite": [ "lighteval", "mmlu" ], "original_num_docs": 198, "effective_num_docs": 198 }, "lighteval|mmlu:high_school_government_and_politics": { "name": "mmlu:high_school_government_and_politics", "prompt_function": "mmlu_harness", "hf_repo": "lighteval/mmlu", "hf_subset": "high_school_government_and_politics", "metric": [ "loglikelihood_acc" ], "hf_avail_splits": [ "auxiliary_train", "test", "validation", "dev" ], "evaluation_splits": [ "test" ], "few_shots_split": "dev", "few_shots_select": "sequential", "generation_size": 1, "stop_sequence": [ "\n" ], "output_regex": null, "frozen": false, "suite": [ "lighteval", "mmlu" ], "original_num_docs": 193, "effective_num_docs": 193 }, "lighteval|mmlu:high_school_macroeconomics": { "name": "mmlu:high_school_macroeconomics", "prompt_function": "mmlu_harness", "hf_repo": "lighteval/mmlu", "hf_subset": "high_school_macroeconomics", "metric": [ "loglikelihood_acc" ], "hf_avail_splits": [ "auxiliary_train", "test", "validation", "dev" ], "evaluation_splits": [ "test" ], "few_shots_split": "dev", "few_shots_select": "sequential", "generation_size": 1, "stop_sequence": [ "\n" ], "output_regex": null, "frozen": false, "suite": [ "lighteval", "mmlu" ], "original_num_docs": 390, "effective_num_docs": 390 }, "lighteval|mmlu:high_school_mathematics": { "name": "mmlu:high_school_mathematics", "prompt_function": "mmlu_harness", "hf_repo": "lighteval/mmlu", "hf_subset": "high_school_mathematics", "metric": [ "loglikelihood_acc" ], "hf_avail_splits": [ "auxiliary_train", "test", "validation", "dev" ], "evaluation_splits": [ "test" ], "few_shots_split": "dev", "few_shots_select": "sequential", "generation_size": 1, "stop_sequence": [ "\n" ], "output_regex": null, "frozen": false, "suite": [ "lighteval", "mmlu" ], "original_num_docs": 270, "effective_num_docs": 270 }, "lighteval|mmlu:high_school_microeconomics": { "name": "mmlu:high_school_microeconomics", "prompt_function": "mmlu_harness", "hf_repo": "lighteval/mmlu", "hf_subset": "high_school_microeconomics", "metric": [ "loglikelihood_acc" ], "hf_avail_splits": [ "auxiliary_train", "test", "validation", "dev" ], "evaluation_splits": [ "test" ], "few_shots_split": "dev", "few_shots_select": "sequential", "generation_size": 1, "stop_sequence": [ "\n" ], "output_regex": null, "frozen": false, "suite": [ "lighteval", "mmlu" ], "original_num_docs": 238, "effective_num_docs": 238 }, "lighteval|mmlu:high_school_physics": { "name": "mmlu:high_school_physics", "prompt_function": "mmlu_harness", "hf_repo": "lighteval/mmlu", "hf_subset": "high_school_physics", "metric": [ "loglikelihood_acc" ], "hf_avail_splits": [ "auxiliary_train", "test", "validation", "dev" ], "evaluation_splits": [ "test" ], "few_shots_split": "dev", "few_shots_select": "sequential", "generation_size": 1, "stop_sequence": [ "\n" ], "output_regex": null, "frozen": false, "suite": [ "lighteval", "mmlu" ], "original_num_docs": 151, "effective_num_docs": 151 }, "lighteval|mmlu:high_school_psychology": { "name": "mmlu:high_school_psychology", "prompt_function": "mmlu_harness", "hf_repo": "lighteval/mmlu", "hf_subset": "high_school_psychology", "metric": [ "loglikelihood_acc" ], "hf_avail_splits": [ "auxiliary_train", "test", "validation", "dev" ], "evaluation_splits": [ "test" ], "few_shots_split": "dev", "few_shots_select": "sequential", "generation_size": 1, "stop_sequence": [ "\n" ], "output_regex": null, "frozen": false, "suite": [ "lighteval", "mmlu" ], "original_num_docs": 545, "effective_num_docs": 545 }, "lighteval|mmlu:high_school_statistics": { "name": "mmlu:high_school_statistics", "prompt_function": "mmlu_harness", "hf_repo": "lighteval/mmlu", "hf_subset": "high_school_statistics", "metric": [ "loglikelihood_acc" ], "hf_avail_splits": [ "auxiliary_train", "test", "validation", "dev" ], "evaluation_splits": [ "test" ], "few_shots_split": "dev", "few_shots_select": "sequential", "generation_size": 1, "stop_sequence": [ "\n" ], "output_regex": null, "frozen": false, "suite": [ "lighteval", "mmlu" ], "original_num_docs": 216, "effective_num_docs": 216 }, "lighteval|mmlu:high_school_us_history": { "name": "mmlu:high_school_us_history", "prompt_function": "mmlu_harness", "hf_repo": "lighteval/mmlu", "hf_subset": "high_school_us_history", "metric": [ "loglikelihood_acc" ], "hf_avail_splits": [ "auxiliary_train", "test", "validation", "dev" ], "evaluation_splits": [ "test" ], "few_shots_split": "dev", "few_shots_select": "sequential", "generation_size": 1, "stop_sequence": [ "\n" ], "output_regex": null, "frozen": false, "suite": [ "lighteval", "mmlu" ], "original_num_docs": 204, "effective_num_docs": 204 }, "lighteval|mmlu:high_school_world_history": { "name": "mmlu:high_school_world_history", "prompt_function": "mmlu_harness", "hf_repo": "lighteval/mmlu", "hf_subset": "high_school_world_history", "metric": [ "loglikelihood_acc" ], "hf_avail_splits": [ "auxiliary_train", "test", "validation", "dev" ], "evaluation_splits": [ "test" ], "few_shots_split": "dev", "few_shots_select": "sequential", "generation_size": 1, "stop_sequence": [ "\n" ], "output_regex": null, "frozen": false, "suite": [ "lighteval", "mmlu" ], "original_num_docs": 237, "effective_num_docs": 237 }, "lighteval|mmlu:human_aging": { "name": "mmlu:human_aging", "prompt_function": "mmlu_harness", "hf_repo": "lighteval/mmlu", "hf_subset": "human_aging", "metric": [ "loglikelihood_acc" ], "hf_avail_splits": [ "auxiliary_train", "test", "validation", "dev" ], "evaluation_splits": [ "test" ], "few_shots_split": "dev", "few_shots_select": "sequential", "generation_size": 1, "stop_sequence": [ "\n" ], "output_regex": null, "frozen": false, "suite": [ "lighteval", "mmlu" ], "original_num_docs": 223, "effective_num_docs": 223 }, "lighteval|mmlu:human_sexuality": { "name": "mmlu:human_sexuality", "prompt_function": "mmlu_harness", "hf_repo": "lighteval/mmlu", "hf_subset": "human_sexuality", "metric": [ "loglikelihood_acc" ], "hf_avail_splits": [ "auxiliary_train", "test", "validation", "dev" ], "evaluation_splits": [ "test" ], "few_shots_split": "dev", "few_shots_select": "sequential", "generation_size": 1, "stop_sequence": [ "\n" ], "output_regex": null, "frozen": false, "suite": [ "lighteval", "mmlu" ], "original_num_docs": 131, "effective_num_docs": 131 }, "lighteval|mmlu:international_law": { "name": "mmlu:international_law", "prompt_function": "mmlu_harness", "hf_repo": "lighteval/mmlu", "hf_subset": "international_law", "metric": [ "loglikelihood_acc" ], "hf_avail_splits": [ "auxiliary_train", "test", "validation", "dev" ], "evaluation_splits": [ "test" ], "few_shots_split": "dev", "few_shots_select": "sequential", "generation_size": 1, "stop_sequence": [ "\n" ], "output_regex": null, "frozen": false, "suite": [ "lighteval", "mmlu" ], "original_num_docs": 121, "effective_num_docs": 121 }, "lighteval|mmlu:jurisprudence": { "name": "mmlu:jurisprudence", "prompt_function": "mmlu_harness", "hf_repo": "lighteval/mmlu", "hf_subset": "jurisprudence", "metric": [ "loglikelihood_acc" ], "hf_avail_splits": [ "auxiliary_train", "test", "validation", "dev" ], "evaluation_splits": [ "test" ], "few_shots_split": "dev", "few_shots_select": "sequential", "generation_size": 1, "stop_sequence": [ "\n" ], "output_regex": null, "frozen": false, "suite": [ "lighteval", "mmlu" ], "original_num_docs": 108, "effective_num_docs": 108 }, "lighteval|mmlu:logical_fallacies": { "name": "mmlu:logical_fallacies", "prompt_function": "mmlu_harness", "hf_repo": "lighteval/mmlu", "hf_subset": "logical_fallacies", "metric": [ "loglikelihood_acc" ], "hf_avail_splits": [ "auxiliary_train", "test", "validation", "dev" ], "evaluation_splits": [ "test" ], "few_shots_split": "dev", "few_shots_select": "sequential", "generation_size": 1, "stop_sequence": [ "\n" ], "output_regex": null, "frozen": false, "suite": [ "lighteval", "mmlu" ], "original_num_docs": 163, "effective_num_docs": 163 }, "lighteval|mmlu:machine_learning": { "name": "mmlu:machine_learning", "prompt_function": "mmlu_harness", "hf_repo": "lighteval/mmlu", "hf_subset": "machine_learning", "metric": [ "loglikelihood_acc" ], "hf_avail_splits": [ "auxiliary_train", "test", "validation", "dev" ], "evaluation_splits": [ "test" ], "few_shots_split": "dev", "few_shots_select": "sequential", "generation_size": 1, "stop_sequence": [ "\n" ], "output_regex": null, "frozen": false, "suite": [ "lighteval", "mmlu" ], "original_num_docs": 112, "effective_num_docs": 112 }, "lighteval|mmlu:management": { "name": "mmlu:management", "prompt_function": "mmlu_harness", "hf_repo": "lighteval/mmlu", "hf_subset": "management", "metric": [ "loglikelihood_acc" ], "hf_avail_splits": [ "auxiliary_train", "test", "validation", "dev" ], "evaluation_splits": [ "test" ], "few_shots_split": "dev", "few_shots_select": "sequential", "generation_size": 1, "stop_sequence": [ "\n" ], "output_regex": null, "frozen": false, "suite": [ "lighteval", "mmlu" ], "original_num_docs": 103, "effective_num_docs": 103 }, "lighteval|mmlu:marketing": { "name": "mmlu:marketing", "prompt_function": "mmlu_harness", "hf_repo": "lighteval/mmlu", "hf_subset": "marketing", "metric": [ "loglikelihood_acc" ], "hf_avail_splits": [ "auxiliary_train", "test", "validation", "dev" ], "evaluation_splits": [ "test" ], "few_shots_split": "dev", "few_shots_select": "sequential", "generation_size": 1, "stop_sequence": [ "\n" ], "output_regex": null, "frozen": false, "suite": [ "lighteval", "mmlu" ], "original_num_docs": 234, "effective_num_docs": 234 }, "lighteval|mmlu:medical_genetics": { "name": "mmlu:medical_genetics", "prompt_function": "mmlu_harness", "hf_repo": "lighteval/mmlu", "hf_subset": "medical_genetics", "metric": [ "loglikelihood_acc" ], "hf_avail_splits": [ "auxiliary_train", "test", "validation", "dev" ], "evaluation_splits": [ "test" ], "few_shots_split": "dev", "few_shots_select": "sequential", "generation_size": 1, "stop_sequence": [ "\n" ], "output_regex": null, "frozen": false, "suite": [ "lighteval", "mmlu" ], "original_num_docs": 100, "effective_num_docs": 100 }, "lighteval|mmlu:miscellaneous": { "name": "mmlu:miscellaneous", "prompt_function": "mmlu_harness", "hf_repo": "lighteval/mmlu", "hf_subset": "miscellaneous", "metric": [ "loglikelihood_acc" ], "hf_avail_splits": [ "auxiliary_train", "test", "validation", "dev" ], "evaluation_splits": [ "test" ], "few_shots_split": "dev", "few_shots_select": "sequential", "generation_size": 1, "stop_sequence": [ "\n" ], "output_regex": null, "frozen": false, "suite": [ "lighteval", "mmlu" ], "original_num_docs": 783, "effective_num_docs": 783 }, "lighteval|mmlu:moral_disputes": { "name": "mmlu:moral_disputes", "prompt_function": "mmlu_harness", "hf_repo": "lighteval/mmlu", "hf_subset": "moral_disputes", "metric": [ "loglikelihood_acc" ], "hf_avail_splits": [ "auxiliary_train", "test", "validation", "dev" ], "evaluation_splits": [ "test" ], "few_shots_split": "dev", "few_shots_select": "sequential", "generation_size": 1, "stop_sequence": [ "\n" ], "output_regex": null, "frozen": false, "suite": [ "lighteval", "mmlu" ], "original_num_docs": 346, "effective_num_docs": 346 }, "lighteval|mmlu:moral_scenarios": { "name": "mmlu:moral_scenarios", "prompt_function": "mmlu_harness", "hf_repo": "lighteval/mmlu", "hf_subset": "moral_scenarios", "metric": [ "loglikelihood_acc" ], "hf_avail_splits": [ "auxiliary_train", "test", "validation", "dev" ], "evaluation_splits": [ "test" ], "few_shots_split": "dev", "few_shots_select": "sequential", "generation_size": 1, "stop_sequence": [ "\n" ], "output_regex": null, "frozen": false, "suite": [ "lighteval", "mmlu" ], "original_num_docs": 895, "effective_num_docs": 895 }, "lighteval|mmlu:nutrition": { "name": "mmlu:nutrition", "prompt_function": "mmlu_harness", "hf_repo": "lighteval/mmlu", "hf_subset": "nutrition", "metric": [ "loglikelihood_acc" ], "hf_avail_splits": [ "auxiliary_train", "test", "validation", "dev" ], "evaluation_splits": [ "test" ], "few_shots_split": "dev", "few_shots_select": "sequential", "generation_size": 1, "stop_sequence": [ "\n" ], "output_regex": null, "frozen": false, "suite": [ "lighteval", "mmlu" ], "original_num_docs": 306, "effective_num_docs": 306 }, "lighteval|mmlu:philosophy": { "name": "mmlu:philosophy", "prompt_function": "mmlu_harness", "hf_repo": "lighteval/mmlu", "hf_subset": "philosophy", "metric": [ "loglikelihood_acc" ], "hf_avail_splits": [ "auxiliary_train", "test", "validation", "dev" ], "evaluation_splits": [ "test" ], "few_shots_split": "dev", "few_shots_select": "sequential", "generation_size": 1, "stop_sequence": [ "\n" ], "output_regex": null, "frozen": false, "suite": [ "lighteval", "mmlu" ], "original_num_docs": 311, "effective_num_docs": 311 }, "lighteval|mmlu:prehistory": { "name": "mmlu:prehistory", "prompt_function": "mmlu_harness", "hf_repo": "lighteval/mmlu", "hf_subset": "prehistory", "metric": [ "loglikelihood_acc" ], "hf_avail_splits": [ "auxiliary_train", "test", "validation", "dev" ], "evaluation_splits": [ "test" ], "few_shots_split": "dev", "few_shots_select": "sequential", "generation_size": 1, "stop_sequence": [ "\n" ], "output_regex": null, "frozen": false, "suite": [ "lighteval", "mmlu" ], "original_num_docs": 324, "effective_num_docs": 324 }, "lighteval|mmlu:professional_accounting": { "name": "mmlu:professional_accounting", "prompt_function": "mmlu_harness", "hf_repo": "lighteval/mmlu", "hf_subset": "professional_accounting", "metric": [ "loglikelihood_acc" ], "hf_avail_splits": [ "auxiliary_train", "test", "validation", "dev" ], "evaluation_splits": [ "test" ], "few_shots_split": "dev", "few_shots_select": "sequential", "generation_size": 1, "stop_sequence": [ "\n" ], "output_regex": null, "frozen": false, "suite": [ "lighteval", "mmlu" ], "original_num_docs": 282, "effective_num_docs": 282 }, "lighteval|mmlu:professional_law": { "name": "mmlu:professional_law", "prompt_function": "mmlu_harness", "hf_repo": "lighteval/mmlu", "hf_subset": "professional_law", "metric": [ "loglikelihood_acc" ], "hf_avail_splits": [ "auxiliary_train", "test", "validation", "dev" ], "evaluation_splits": [ "test" ], "few_shots_split": "dev", "few_shots_select": "sequential", "generation_size": 1, "stop_sequence": [ "\n" ], "output_regex": null, "frozen": false, "suite": [ "lighteval", "mmlu" ], "original_num_docs": 1534, "effective_num_docs": 1534 }, "lighteval|mmlu:professional_medicine": { "name": "mmlu:professional_medicine", "prompt_function": "mmlu_harness", "hf_repo": "lighteval/mmlu", "hf_subset": "professional_medicine", "metric": [ "loglikelihood_acc" ], "hf_avail_splits": [ "auxiliary_train", "test", "validation", "dev" ], "evaluation_splits": [ "test" ], "few_shots_split": "dev", "few_shots_select": "sequential", "generation_size": 1, "stop_sequence": [ "\n" ], "output_regex": null, "frozen": false, "suite": [ "lighteval", "mmlu" ], "original_num_docs": 272, "effective_num_docs": 272 }, "lighteval|mmlu:professional_psychology": { "name": "mmlu:professional_psychology", "prompt_function": "mmlu_harness", "hf_repo": "lighteval/mmlu", "hf_subset": "professional_psychology", "metric": [ "loglikelihood_acc" ], "hf_avail_splits": [ "auxiliary_train", "test", "validation", "dev" ], "evaluation_splits": [ "test" ], "few_shots_split": "dev", "few_shots_select": "sequential", "generation_size": 1, "stop_sequence": [ "\n" ], "output_regex": null, "frozen": false, "suite": [ "lighteval", "mmlu" ], "original_num_docs": 612, "effective_num_docs": 612 }, "lighteval|mmlu:public_relations": { "name": "mmlu:public_relations", "prompt_function": "mmlu_harness", "hf_repo": "lighteval/mmlu", "hf_subset": "public_relations", "metric": [ "loglikelihood_acc" ], "hf_avail_splits": [ "auxiliary_train", "test", "validation", "dev" ], "evaluation_splits": [ "test" ], "few_shots_split": "dev", "few_shots_select": "sequential", "generation_size": 1, "stop_sequence": [ "\n" ], "output_regex": null, "frozen": false, "suite": [ "lighteval", "mmlu" ], "original_num_docs": 110, "effective_num_docs": 110 }, "lighteval|mmlu:security_studies": { "name": "mmlu:security_studies", "prompt_function": "mmlu_harness", "hf_repo": "lighteval/mmlu", "hf_subset": "security_studies", "metric": [ "loglikelihood_acc" ], "hf_avail_splits": [ "auxiliary_train", "test", "validation", "dev" ], "evaluation_splits": [ "test" ], "few_shots_split": "dev", "few_shots_select": "sequential", "generation_size": 1, "stop_sequence": [ "\n" ], "output_regex": null, "frozen": false, "suite": [ "lighteval", "mmlu" ], "original_num_docs": 245, "effective_num_docs": 245 }, "lighteval|mmlu:sociology": { "name": "mmlu:sociology", "prompt_function": "mmlu_harness", "hf_repo": "lighteval/mmlu", "hf_subset": "sociology", "metric": [ "loglikelihood_acc" ], "hf_avail_splits": [ "auxiliary_train", "test", "validation", "dev" ], "evaluation_splits": [ "test" ], "few_shots_split": "dev", "few_shots_select": "sequential", "generation_size": 1, "stop_sequence": [ "\n" ], "output_regex": null, "frozen": false, "suite": [ "lighteval", "mmlu" ], "original_num_docs": 201, "effective_num_docs": 201 }, "lighteval|mmlu:us_foreign_policy": { "name": "mmlu:us_foreign_policy", "prompt_function": "mmlu_harness", "hf_repo": "lighteval/mmlu", "hf_subset": "us_foreign_policy", "metric": [ "loglikelihood_acc" ], "hf_avail_splits": [ "auxiliary_train", "test", "validation", "dev" ], "evaluation_splits": [ "test" ], "few_shots_split": "dev", "few_shots_select": "sequential", "generation_size": 1, "stop_sequence": [ "\n" ], "output_regex": null, "frozen": false, "suite": [ "lighteval", "mmlu" ], "original_num_docs": 100, "effective_num_docs": 100 }, "lighteval|mmlu:virology": { "name": "mmlu:virology", "prompt_function": "mmlu_harness", "hf_repo": "lighteval/mmlu", "hf_subset": "virology", "metric": [ "loglikelihood_acc" ], "hf_avail_splits": [ "auxiliary_train", "test", "validation", "dev" ], "evaluation_splits": [ "test" ], "few_shots_split": "dev", "few_shots_select": "sequential", "generation_size": 1, "stop_sequence": [ "\n" ], "output_regex": null, "frozen": false, "suite": [ "lighteval", "mmlu" ], "original_num_docs": 166, "effective_num_docs": 166 }, "lighteval|mmlu:world_religions": { "name": "mmlu:world_religions", "prompt_function": "mmlu_harness", "hf_repo": "lighteval/mmlu", "hf_subset": "world_religions", "metric": [ "loglikelihood_acc" ], "hf_avail_splits": [ "auxiliary_train", "test", "validation", "dev" ], "evaluation_splits": [ "test" ], "few_shots_split": "dev", "few_shots_select": "sequential", "generation_size": 1, "stop_sequence": [ "\n" ], "output_regex": null, "frozen": false, "suite": [ "lighteval", "mmlu" ], "original_num_docs": 171, "effective_num_docs": 171 } }, "summary_tasks": { "lighteval|mmlu:abstract_algebra|5": { "hashes": { "hash_examples": "4c76229e00c9c0e9", "hash_full_prompts": "a45d01c3409c889c", "hash_input_tokens": "fc11398ca4e995e6", "hash_cont_tokens": "dadea1de19dee95c" }, "truncated": 0, "non_truncated": 100, "padded": 400, "non_padded": 0, "effective_few_shots": 5.0, "num_truncated_few_shots": 0 }, "lighteval|mmlu:anatomy|5": { "hashes": { "hash_examples": "6a1f8104dccbd33b", "hash_full_prompts": "e245c6600e03cc32", "hash_input_tokens": "0e63aad739f5d777", "hash_cont_tokens": "96c2bab19c75f48d" }, "truncated": 0, "non_truncated": 135, "padded": 540, "non_padded": 0, "effective_few_shots": 5.0, "num_truncated_few_shots": 0 }, "lighteval|mmlu:astronomy|5": { "hashes": { "hash_examples": "1302effa3a76ce4c", "hash_full_prompts": "390f9bddf857ad04", "hash_input_tokens": "53afd9483d456920", "hash_cont_tokens": "6cc2d6fb43989c46" }, "truncated": 0, "non_truncated": 152, "padded": 608, "non_padded": 0, "effective_few_shots": 5.0, "num_truncated_few_shots": 0 }, "lighteval|mmlu:business_ethics|5": { "hashes": { "hash_examples": "03cb8bce5336419a", "hash_full_prompts": "5504f893bc4f2fa1", "hash_input_tokens": "1d0d99c2f7f95728", "hash_cont_tokens": "dadea1de19dee95c" }, "truncated": 0, "non_truncated": 100, "padded": 400, "non_padded": 0, "effective_few_shots": 5.0, "num_truncated_few_shots": 0 }, "lighteval|mmlu:clinical_knowledge|5": { "hashes": { "hash_examples": "ffbb9c7b2be257f9", "hash_full_prompts": "106ad0bab4b90b78", "hash_input_tokens": "6abbbf267dbe9940", "hash_cont_tokens": "4566966a1e601b6c" }, "truncated": 0, "non_truncated": 265, "padded": 1060, "non_padded": 0, "effective_few_shots": 5.0, "num_truncated_few_shots": 0 }, "lighteval|mmlu:college_biology|5": { "hashes": { "hash_examples": "3ee77f176f38eb8e", "hash_full_prompts": "59f9bdf2695cb226", "hash_input_tokens": "803196bfad4a393a", "hash_cont_tokens": "4ea00cd7b2f74799" }, "truncated": 0, "non_truncated": 144, "padded": 576, "non_padded": 0, "effective_few_shots": 5.0, "num_truncated_few_shots": 0 }, "lighteval|mmlu:college_chemistry|5": { "hashes": { "hash_examples": "ce61a69c46d47aeb", "hash_full_prompts": "3cac9b759fcff7a0", "hash_input_tokens": "87bd9eea77de9a9a", "hash_cont_tokens": "dadea1de19dee95c" }, "truncated": 0, "non_truncated": 100, "padded": 400, "non_padded": 0, "effective_few_shots": 5.0, "num_truncated_few_shots": 0 }, "lighteval|mmlu:college_computer_science|5": { "hashes": { "hash_examples": "32805b52d7d5daab", "hash_full_prompts": "010b0cca35070130", "hash_input_tokens": "b6775c67bfa0c782", "hash_cont_tokens": "dadea1de19dee95c" }, "truncated": 0, "non_truncated": 100, "padded": 400, "non_padded": 0, "effective_few_shots": 5.0, "num_truncated_few_shots": 0 }, "lighteval|mmlu:college_mathematics|5": { "hashes": { "hash_examples": "55da1a0a0bd33722", "hash_full_prompts": "511422eb9eefc773", "hash_input_tokens": "cbd8a9d6bbda7b3c", "hash_cont_tokens": "dadea1de19dee95c" }, "truncated": 0, "non_truncated": 100, "padded": 400, "non_padded": 0, "effective_few_shots": 5.0, "num_truncated_few_shots": 0 }, "lighteval|mmlu:college_medicine|5": { "hashes": { "hash_examples": "c33e143163049176", "hash_full_prompts": "c8cc1a82a51a046e", "hash_input_tokens": "b3c40eab0fb83731", "hash_cont_tokens": "aed3e7fd8adea27e" }, "truncated": 0, "non_truncated": 173, "padded": 692, "non_padded": 0, "effective_few_shots": 5.0, "num_truncated_few_shots": 0 }, "lighteval|mmlu:college_physics|5": { "hashes": { "hash_examples": "ebdab1cdb7e555df", "hash_full_prompts": "e40721b5059c5818", "hash_input_tokens": "c69c0bfb74e99180", "hash_cont_tokens": "1ca37bb9b8be1c5d" }, "truncated": 0, "non_truncated": 102, "padded": 408, "non_padded": 0, "effective_few_shots": 5.0, "num_truncated_few_shots": 0 }, "lighteval|mmlu:computer_security|5": { "hashes": { "hash_examples": "a24fd7d08a560921", "hash_full_prompts": "946c9be5964ac44a", "hash_input_tokens": "70914e4af05d09b4", "hash_cont_tokens": "dadea1de19dee95c" }, "truncated": 0, "non_truncated": 100, "padded": 400, "non_padded": 0, "effective_few_shots": 5.0, "num_truncated_few_shots": 0 }, "lighteval|mmlu:conceptual_physics|5": { "hashes": { "hash_examples": "8300977a79386993", "hash_full_prompts": "506a4f6094cc40c9", "hash_input_tokens": "dcb90ef41648f505", "hash_cont_tokens": "26db9e6e7dfdac00" }, "truncated": 0, "non_truncated": 235, "padded": 940, "non_padded": 0, "effective_few_shots": 5.0, "num_truncated_few_shots": 0 }, "lighteval|mmlu:econometrics|5": { "hashes": { "hash_examples": "ddde36788a04a46f", "hash_full_prompts": "4ed2703f27f1ed05", "hash_input_tokens": "ef8da4b8e9eb5a76", "hash_cont_tokens": "2ef49b394cfb87e1" }, "truncated": 0, "non_truncated": 114, "padded": 456, "non_padded": 0, "effective_few_shots": 5.0, "num_truncated_few_shots": 0 }, "lighteval|mmlu:electrical_engineering|5": { "hashes": { "hash_examples": "acbc5def98c19b3f", "hash_full_prompts": "d8f4b3e11c23653c", "hash_input_tokens": "1a5e9d41be2d9981", "hash_cont_tokens": "adb5a1c5d57fbb41" }, "truncated": 0, "non_truncated": 145, "padded": 580, "non_padded": 0, "effective_few_shots": 5.0, "num_truncated_few_shots": 0 }, "lighteval|mmlu:elementary_mathematics|5": { "hashes": { "hash_examples": "146e61d07497a9bd", "hash_full_prompts": "256d111bd15647ff", "hash_input_tokens": "e0d51d86d03e1394", "hash_cont_tokens": "d0782f141bcc895b" }, "truncated": 0, "non_truncated": 378, "padded": 1512, "non_padded": 0, "effective_few_shots": 5.0, "num_truncated_few_shots": 0 }, "lighteval|mmlu:formal_logic|5": { "hashes": { "hash_examples": "8635216e1909a03f", "hash_full_prompts": "1171d04f3b1a11f5", "hash_input_tokens": "4c75b7f176e01a01", "hash_cont_tokens": "315a91fa1f805c93" }, "truncated": 0, "non_truncated": 126, "padded": 504, "non_padded": 0, "effective_few_shots": 5.0, "num_truncated_few_shots": 0 }, "lighteval|mmlu:global_facts|5": { "hashes": { "hash_examples": "30b315aa6353ee47", "hash_full_prompts": "a7e56dbc074c7529", "hash_input_tokens": "b83cb180a97c221d", "hash_cont_tokens": "dadea1de19dee95c" }, "truncated": 0, "non_truncated": 100, "padded": 400, "non_padded": 0, "effective_few_shots": 5.0, "num_truncated_few_shots": 0 }, "lighteval|mmlu:high_school_biology|5": { "hashes": { "hash_examples": "c9136373af2180de", "hash_full_prompts": "ad6e859ed978e04a", "hash_input_tokens": "179a2ab8e131445a", "hash_cont_tokens": "715bc46d18155135" }, "truncated": 0, "non_truncated": 310, "padded": 1240, "non_padded": 0, "effective_few_shots": 5.0, "num_truncated_few_shots": 0 }, "lighteval|mmlu:high_school_chemistry|5": { "hashes": { "hash_examples": "b0661bfa1add6404", "hash_full_prompts": "6eb9c04bcc8a8f2a", "hash_input_tokens": "1e6a4441b61eb8f6", "hash_cont_tokens": "3d12f9b93cc609a2" }, "truncated": 0, "non_truncated": 203, "padded": 812, "non_padded": 0, "effective_few_shots": 5.0, "num_truncated_few_shots": 0 }, "lighteval|mmlu:high_school_computer_science|5": { "hashes": { "hash_examples": "80fc1d623a3d665f", "hash_full_prompts": "8e51bc91c81cf8dd", "hash_input_tokens": "4df816916ded3a8c", "hash_cont_tokens": "dadea1de19dee95c" }, "truncated": 0, "non_truncated": 100, "padded": 400, "non_padded": 0, "effective_few_shots": 5.0, "num_truncated_few_shots": 0 }, "lighteval|mmlu:high_school_european_history|5": { "hashes": { "hash_examples": "854da6e5af0fe1a1", "hash_full_prompts": "664a1f16c9f3195c", "hash_input_tokens": "317d565e995cda09", "hash_cont_tokens": "6d9c47e593859ccd" }, "truncated": 0, "non_truncated": 165, "padded": 656, "non_padded": 4, "effective_few_shots": 5.0, "num_truncated_few_shots": 0 }, "lighteval|mmlu:high_school_geography|5": { "hashes": { "hash_examples": "7dc963c7acd19ad8", "hash_full_prompts": "f3acf911f4023c8a", "hash_input_tokens": "0f17bdb1600d33f7", "hash_cont_tokens": "84097c7fa87dfe61" }, "truncated": 0, "non_truncated": 198, "padded": 792, "non_padded": 0, "effective_few_shots": 5.0, "num_truncated_few_shots": 0 }, "lighteval|mmlu:high_school_government_and_politics|5": { "hashes": { "hash_examples": "1f675dcdebc9758f", "hash_full_prompts": "066254feaa3158ae", "hash_input_tokens": "ac3cca039d98e159", "hash_cont_tokens": "86d43dfe026b5e6e" }, "truncated": 0, "non_truncated": 193, "padded": 772, "non_padded": 0, "effective_few_shots": 5.0, "num_truncated_few_shots": 0 }, "lighteval|mmlu:high_school_macroeconomics|5": { "hashes": { "hash_examples": "2fb32cf2d80f0b35", "hash_full_prompts": "19a7fa502aa85c95", "hash_input_tokens": "3e795472fd70b8e9", "hash_cont_tokens": "99f5469b1de9a21b" }, "truncated": 0, "non_truncated": 390, "padded": 1560, "non_padded": 0, "effective_few_shots": 5.0, "num_truncated_few_shots": 0 }, "lighteval|mmlu:high_school_mathematics|5": { "hashes": { "hash_examples": "fd6646fdb5d58a1f", "hash_full_prompts": "4f704e369778b5b0", "hash_input_tokens": "37e154ab071591d5", "hash_cont_tokens": "e215c84aa19ccb33" }, "truncated": 0, "non_truncated": 270, "padded": 1078, "non_padded": 2, "effective_few_shots": 5.0, "num_truncated_few_shots": 0 }, "lighteval|mmlu:high_school_microeconomics|5": { "hashes": { "hash_examples": "2118f21f71d87d84", "hash_full_prompts": "4350f9e2240f8010", "hash_input_tokens": "02d65d5e1ee6dea9", "hash_cont_tokens": "dc8017437d84c710" }, "truncated": 0, "non_truncated": 238, "padded": 952, "non_padded": 0, "effective_few_shots": 5.0, "num_truncated_few_shots": 0 }, "lighteval|mmlu:high_school_physics|5": { "hashes": { "hash_examples": "dc3ce06378548565", "hash_full_prompts": "5dc0d6831b66188f", "hash_input_tokens": "6f0c932d12edce11", "hash_cont_tokens": "b8152fcdcf86c673" }, "truncated": 0, "non_truncated": 151, "padded": 596, "non_padded": 8, "effective_few_shots": 5.0, "num_truncated_few_shots": 0 }, "lighteval|mmlu:high_school_psychology|5": { "hashes": { "hash_examples": "c8d1d98a40e11f2f", "hash_full_prompts": "af2b097da6d50365", "hash_input_tokens": "0e444eb7ba0a1fb0", "hash_cont_tokens": "ac45cbb9009f81d9" }, "truncated": 0, "non_truncated": 545, "padded": 2168, "non_padded": 12, "effective_few_shots": 5.0, "num_truncated_few_shots": 0 }, "lighteval|mmlu:high_school_statistics|5": { "hashes": { "hash_examples": "666c8759b98ee4ff", "hash_full_prompts": "c757694421d6d68d", "hash_input_tokens": "4e1485b614b2dc7f", "hash_cont_tokens": "9c9b68ee68272b16" }, "truncated": 0, "non_truncated": 216, "padded": 864, "non_padded": 0, "effective_few_shots": 5.0, "num_truncated_few_shots": 0 }, "lighteval|mmlu:high_school_us_history|5": { "hashes": { "hash_examples": "95fef1c4b7d3f81e", "hash_full_prompts": "e34a028d0ddeec5e", "hash_input_tokens": "b836c43a53625ee3", "hash_cont_tokens": "cec285b624c15c10" }, "truncated": 0, "non_truncated": 204, "padded": 816, "non_padded": 0, "effective_few_shots": 5.0, "num_truncated_few_shots": 0 }, "lighteval|mmlu:high_school_world_history|5": { "hashes": { "hash_examples": "7e5085b6184b0322", "hash_full_prompts": "1fa3d51392765601", "hash_input_tokens": "bb11d024e2405b72", "hash_cont_tokens": "2c02128f8f2f7539" }, "truncated": 0, "non_truncated": 237, "padded": 948, "non_padded": 0, "effective_few_shots": 5.0, "num_truncated_few_shots": 0 }, "lighteval|mmlu:human_aging|5": { "hashes": { "hash_examples": "c17333e7c7c10797", "hash_full_prompts": "cac900721f9a1a94", "hash_input_tokens": "2a1e5a167a3788c9", "hash_cont_tokens": "faa94c4ec8e7be4e" }, "truncated": 0, "non_truncated": 223, "padded": 892, "non_padded": 0, "effective_few_shots": 5.0, "num_truncated_few_shots": 0 }, "lighteval|mmlu:human_sexuality|5": { "hashes": { "hash_examples": "4edd1e9045df5e3d", "hash_full_prompts": "0d6567bafee0a13c", "hash_input_tokens": "73b98b906cf7ce7f", "hash_cont_tokens": "d642d34719fa5ff6" }, "truncated": 0, "non_truncated": 131, "padded": 524, "non_padded": 0, "effective_few_shots": 5.0, "num_truncated_few_shots": 0 }, "lighteval|mmlu:international_law|5": { "hashes": { "hash_examples": "db2fa00d771a062a", "hash_full_prompts": "d018f9116479795e", "hash_input_tokens": "5f7cf71ef19fdf7d", "hash_cont_tokens": "f0d54717d3cdc783" }, "truncated": 0, "non_truncated": 121, "padded": 484, "non_padded": 0, "effective_few_shots": 5.0, "num_truncated_few_shots": 0 }, "lighteval|mmlu:jurisprudence|5": { "hashes": { "hash_examples": "e956f86b124076fe", "hash_full_prompts": "1487e89a10ec58b7", "hash_input_tokens": "0f30607df3aa1190", "hash_cont_tokens": "d766ae8c3d361559" }, "truncated": 0, "non_truncated": 108, "padded": 432, "non_padded": 0, "effective_few_shots": 5.0, "num_truncated_few_shots": 0 }, "lighteval|mmlu:logical_fallacies|5": { "hashes": { "hash_examples": "956e0e6365ab79f1", "hash_full_prompts": "677785b2181f9243", "hash_input_tokens": "ac2bcfdf302d6dcd", "hash_cont_tokens": "0fcca855210b4243" }, "truncated": 0, "non_truncated": 163, "padded": 652, "non_padded": 0, "effective_few_shots": 5.0, "num_truncated_few_shots": 0 }, "lighteval|mmlu:machine_learning|5": { "hashes": { "hash_examples": "397997cc6f4d581e", "hash_full_prompts": "769ee14a2aea49bb", "hash_input_tokens": "3d634b614f766363", "hash_cont_tokens": "8b369a2ff9235b9d" }, "truncated": 0, "non_truncated": 112, "padded": 448, "non_padded": 0, "effective_few_shots": 5.0, "num_truncated_few_shots": 0 }, "lighteval|mmlu:management|5": { "hashes": { "hash_examples": "2bcbe6f6ca63d740", "hash_full_prompts": "cb1ff9dac9582144", "hash_input_tokens": "d2728b0835c2fa6d", "hash_cont_tokens": "c77ad5f59321afa5" }, "truncated": 0, "non_truncated": 103, "padded": 412, "non_padded": 0, "effective_few_shots": 5.0, "num_truncated_few_shots": 0 }, "lighteval|mmlu:marketing|5": { "hashes": { "hash_examples": "8ddb20d964a1b065", "hash_full_prompts": "9fc2114a187ad9a2", "hash_input_tokens": "9472fa5111070553", "hash_cont_tokens": "c94db408fe712d9b" }, "truncated": 0, "non_truncated": 234, "padded": 936, "non_padded": 0, "effective_few_shots": 5.0, "num_truncated_few_shots": 0 }, "lighteval|mmlu:medical_genetics|5": { "hashes": { "hash_examples": "182a71f4763d2cea", "hash_full_prompts": "46a616fa51878959", "hash_input_tokens": "53f9c4977b0be4e0", "hash_cont_tokens": "dadea1de19dee95c" }, "truncated": 0, "non_truncated": 100, "padded": 400, "non_padded": 0, "effective_few_shots": 5.0, "num_truncated_few_shots": 0 }, "lighteval|mmlu:miscellaneous|5": { "hashes": { "hash_examples": "4c404fdbb4ca57fc", "hash_full_prompts": "0813e1be36dbaae1", "hash_input_tokens": "fca7aac8daf1d0c7", "hash_cont_tokens": "60215a6f77eaf4d9" }, "truncated": 0, "non_truncated": 783, "padded": 3132, "non_padded": 0, "effective_few_shots": 5.0, "num_truncated_few_shots": 0 }, "lighteval|mmlu:moral_disputes|5": { "hashes": { "hash_examples": "60cbd2baa3fea5c9", "hash_full_prompts": "1d14adebb9b62519", "hash_input_tokens": "e06669b20b6dba74", "hash_cont_tokens": "3ca55f92255c9f21" }, "truncated": 0, "non_truncated": 346, "padded": 1384, "non_padded": 0, "effective_few_shots": 5.0, "num_truncated_few_shots": 0 }, "lighteval|mmlu:moral_scenarios|5": { "hashes": { "hash_examples": "fd8b0431fbdd75ef", "hash_full_prompts": "b80d3d236165e3de", "hash_input_tokens": "d22a130cb0ce4eec", "hash_cont_tokens": "a82e76a0738dc6ac" }, "truncated": 0, "non_truncated": 895, "padded": 3551, "non_padded": 29, "effective_few_shots": 5.0, "num_truncated_few_shots": 0 }, "lighteval|mmlu:nutrition|5": { "hashes": { "hash_examples": "71e55e2b829b6528", "hash_full_prompts": "2bfb18e5fab8dea7", "hash_input_tokens": "6213f514742fc41d", "hash_cont_tokens": "b683842a2cf7cdd6" }, "truncated": 0, "non_truncated": 306, "padded": 1224, "non_padded": 0, "effective_few_shots": 5.0, "num_truncated_few_shots": 0 }, "lighteval|mmlu:philosophy|5": { "hashes": { "hash_examples": "a6d489a8d208fa4b", "hash_full_prompts": "e8c0d5b6dae3ccc8", "hash_input_tokens": "99ddb7e2f24852cc", "hash_cont_tokens": "a545f25ae279a135" }, "truncated": 0, "non_truncated": 311, "padded": 1244, "non_padded": 0, "effective_few_shots": 5.0, "num_truncated_few_shots": 0 }, "lighteval|mmlu:prehistory|5": { "hashes": { "hash_examples": "6cc50f032a19acaa", "hash_full_prompts": "4a6a1d3ab1bf28e4", "hash_input_tokens": "246ab4e3ab88967a", "hash_cont_tokens": "5a5ebca069b16663" }, "truncated": 0, "non_truncated": 324, "padded": 1268, "non_padded": 28, "effective_few_shots": 5.0, "num_truncated_few_shots": 0 }, "lighteval|mmlu:professional_accounting|5": { "hashes": { "hash_examples": "50f57ab32f5f6cea", "hash_full_prompts": "e60129bd2d82ffc6", "hash_input_tokens": "aaeb137f42b60e30", "hash_cont_tokens": "e45018e60164d208" }, "truncated": 0, "non_truncated": 282, "padded": 1120, "non_padded": 8, "effective_few_shots": 5.0, "num_truncated_few_shots": 0 }, "lighteval|mmlu:professional_law|5": { "hashes": { "hash_examples": "a8fdc85c64f4b215", "hash_full_prompts": "0dbb1d9b72dcea03", "hash_input_tokens": "a4dd0c29f47b7e84", "hash_cont_tokens": "b11002d08c03f837" }, "truncated": 0, "non_truncated": 1534, "padded": 6136, "non_padded": 0, "effective_few_shots": 5.0, "num_truncated_few_shots": 0 }, "lighteval|mmlu:professional_medicine|5": { "hashes": { "hash_examples": "c373a28a3050a73a", "hash_full_prompts": "5e040f9ca68b089e", "hash_input_tokens": "4e14a4f7fcb794ad", "hash_cont_tokens": "11ce4c2ab1132810" }, "truncated": 0, "non_truncated": 272, "padded": 1088, "non_padded": 0, "effective_few_shots": 5.0, "num_truncated_few_shots": 0 }, "lighteval|mmlu:professional_psychology|5": { "hashes": { "hash_examples": "bf5254fe818356af", "hash_full_prompts": "b386ecda8b87150e", "hash_input_tokens": "d81a045694559382", "hash_cont_tokens": "3835bfc898aacaa0" }, "truncated": 0, "non_truncated": 612, "padded": 2448, "non_padded": 0, "effective_few_shots": 5.0, "num_truncated_few_shots": 0 }, "lighteval|mmlu:public_relations|5": { "hashes": { "hash_examples": "b66d52e28e7d14e0", "hash_full_prompts": "fe43562263e25677", "hash_input_tokens": "1d492df812b3c419", "hash_cont_tokens": "1692112db1aec618" }, "truncated": 0, "non_truncated": 110, "padded": 440, "non_padded": 0, "effective_few_shots": 5.0, "num_truncated_few_shots": 0 }, "lighteval|mmlu:security_studies|5": { "hashes": { "hash_examples": "514c14feaf000ad9", "hash_full_prompts": "27d4a2ac541ef4b9", "hash_input_tokens": "edb25052e8b3c231", "hash_cont_tokens": "9801a1ce7f762a8b" }, "truncated": 0, "non_truncated": 245, "padded": 980, "non_padded": 0, "effective_few_shots": 5.0, "num_truncated_few_shots": 0 }, "lighteval|mmlu:sociology|5": { "hashes": { "hash_examples": "f6c9bc9d18c80870", "hash_full_prompts": "c072ea7d1a1524f2", "hash_input_tokens": "d10e1fc02e9bb000", "hash_cont_tokens": "277e7d5b38c0960d" }, "truncated": 0, "non_truncated": 201, "padded": 804, "non_padded": 0, "effective_few_shots": 5.0, "num_truncated_few_shots": 0 }, "lighteval|mmlu:us_foreign_policy|5": { "hashes": { "hash_examples": "ed7b78629db6678f", "hash_full_prompts": "341a97ca3e4d699d", "hash_input_tokens": "357e68691f7bb5be", "hash_cont_tokens": "dadea1de19dee95c" }, "truncated": 0, "non_truncated": 100, "padded": 397, "non_padded": 3, "effective_few_shots": 5.0, "num_truncated_few_shots": 0 }, "lighteval|mmlu:virology|5": { "hashes": { "hash_examples": "bc52ffdc3f9b994a", "hash_full_prompts": "651d471e2eb8b5e9", "hash_input_tokens": "b38fa14ee2b9cc9d", "hash_cont_tokens": "a4a0852e6fb42244" }, "truncated": 0, "non_truncated": 166, "padded": 664, "non_padded": 0, "effective_few_shots": 5.0, "num_truncated_few_shots": 0 }, "lighteval|mmlu:world_religions|5": { "hashes": { "hash_examples": "ecdb4a4f94f62930", "hash_full_prompts": "3773f03542ce44a3", "hash_input_tokens": "e2e0b330ff7c67d5", "hash_cont_tokens": "c96f2973fdf12010" }, "truncated": 0, "non_truncated": 171, "padded": 684, "non_padded": 0, "effective_few_shots": 5.0, "num_truncated_few_shots": 0 } }, "summary_general": { "hashes": { "hash_examples": "341a076d0beb7048", "hash_full_prompts": "a5c8f2b7ff4f5ae2", "hash_input_tokens": "7d5d2fb20602eddc", "hash_cont_tokens": "28aa09e44eee2d3e" }, "truncated": 0, "non_truncated": 14042, "padded": 56074, "non_padded": 94, "num_truncated_few_shots": 0 } }