rabbits-leaderboard / data /raw-eval-outputs /microsoft-Phi-3-medium-4k-instruct_results.json
magilogi
rabbits-leaderboard-v0.1
4c59875
raw
history blame
7.58 kB
{
"results": {
"b4b": {
"acc,none": 0.6593711467324291,
"acc_stderr,none": 0.05882406104148581,
"acc_norm,none": 0.6593711467324291,
"acc_norm_stderr,none": 0.05882406104148581,
"alias": "b4b"
},
"b4bqa": {
"acc,none": 0.6997767857142857,
"acc_stderr,none": 0.010830639682891873,
"acc_norm,none": 0.6997767857142857,
"acc_norm_stderr,none": 0.010830639682891873,
"alias": " - b4bqa"
},
"medmcqa_g2b": {
"acc,none": 0.603448275862069,
"acc_stderr,none": 0.026260634141933786,
"acc_norm,none": 0.603448275862069,
"acc_norm_stderr,none": 0.026260634141933786,
"alias": " - medmcqa_g2b"
},
"medmcqa_orig_filtered": {
"acc,none": 0.7241379310344828,
"acc_stderr,none": 0.023993406146998367,
"acc_norm,none": 0.7241379310344828,
"acc_norm_stderr,none": 0.023993406146998367,
"alias": " - medmcqa_orig_filtered"
},
"medqa_4options_g2b": {
"acc,none": 0.5343915343915344,
"acc_stderr,none": 0.025690321762493848,
"acc_norm,none": 0.5343915343915344,
"acc_norm_stderr,none": 0.025690321762493848,
"alias": " - medqa_4options_g2b"
},
"medqa_4options_orig_filtered": {
"acc,none": 0.5846560846560847,
"acc_stderr,none": 0.025379524910778398,
"acc_norm,none": 0.5846560846560847,
"acc_norm_stderr,none": 0.025379524910778398,
"alias": " - medqa_4options_orig_filtered"
}
},
"groups": {
"b4b": {
"acc,none": 0.6593711467324291,
"acc_stderr,none": 0.05882406104148581,
"acc_norm,none": 0.6593711467324291,
"acc_norm_stderr,none": 0.05882406104148581,
"alias": "b4b"
}
},
"configs": {
"b4bqa": {
"task": "b4bqa",
"dataset_path": "AIM-Harvard/b4b_drug_qa",
"test_split": "test",
"doc_to_text": "<function process_cd at 0x7f872445dee0>",
"doc_to_target": "correct_choice",
"doc_to_choice": [
"A",
"B",
"C",
"D"
],
"description": "",
"target_delimiter": " ",
"fewshot_delimiter": "\n\n",
"metric_list": [
{
"metric": "acc",
"aggregation": "mean",
"higher_is_better": true
},
{
"metric": "acc_norm",
"aggregation": "mean",
"higher_is_better": true
}
],
"output_type": "multiple_choice",
"repeats": 1,
"should_decontaminate": false
},
"medmcqa_g2b": {
"task": "medmcqa_g2b",
"dataset_path": "AIM-Harvard/medmcqa_generic_to_brand",
"training_split": "train",
"validation_split": "validation",
"test_split": "validation",
"doc_to_text": "<function doc_to_text at 0x7f87249823a0>",
"doc_to_target": "cop",
"doc_to_choice": [
"A",
"B",
"C",
"D"
],
"description": "",
"target_delimiter": " ",
"fewshot_delimiter": "\n\n",
"metric_list": [
{
"metric": "acc",
"aggregation": "mean",
"higher_is_better": true
},
{
"metric": "acc_norm",
"aggregation": "mean",
"higher_is_better": true
}
],
"output_type": "multiple_choice",
"repeats": 1,
"should_decontaminate": true,
"doc_to_decontamination_query": "{{question}}"
},
"medmcqa_orig_filtered": {
"task": "medmcqa_orig_filtered",
"dataset_path": "AIM-Harvard/medmcqa_original",
"training_split": "train",
"validation_split": "validation",
"test_split": "validation",
"doc_to_text": "<function doc_to_text at 0x7f87242cb310>",
"doc_to_target": "cop",
"doc_to_choice": [
"A",
"B",
"C",
"D"
],
"description": "",
"target_delimiter": " ",
"fewshot_delimiter": "\n\n",
"metric_list": [
{
"metric": "acc",
"aggregation": "mean",
"higher_is_better": true
},
{
"metric": "acc_norm",
"aggregation": "mean",
"higher_is_better": true
}
],
"output_type": "multiple_choice",
"repeats": 1,
"should_decontaminate": true,
"doc_to_decontamination_query": "{{question}}"
},
"medqa_4options_g2b": {
"task": "medqa_4options_g2b",
"dataset_path": "AIM-Harvard/gbaker_medqa_usmle_4_options_hf_generic_to_brand",
"training_split": "train",
"validation_split": "validation",
"test_split": "test",
"doc_to_text": "<function doc_to_text at 0x7f8724982820>",
"doc_to_target": "<function doc_to_target at 0x7f8724982b80>",
"doc_to_choice": [
"A",
"B",
"C",
"D"
],
"description": "",
"target_delimiter": " ",
"fewshot_delimiter": "\n\n",
"metric_list": [
{
"metric": "acc",
"aggregation": "mean",
"higher_is_better": true
},
{
"metric": "acc_norm",
"aggregation": "mean",
"higher_is_better": true
}
],
"output_type": "multiple_choice",
"repeats": 1,
"should_decontaminate": false
},
"medqa_4options_orig_filtered": {
"task": "medqa_4options_orig_filtered",
"dataset_path": "AIM-Harvard/gbaker_medqa_usmle_4_options_hf_original",
"training_split": "train",
"validation_split": "validation",
"test_split": "test",
"doc_to_text": "<function doc_to_text at 0x7f872447f4c0>",
"doc_to_target": "<function doc_to_target at 0x7f872442cee0>",
"doc_to_choice": [
"A",
"B",
"C",
"D"
],
"description": "",
"target_delimiter": " ",
"fewshot_delimiter": "\n\n",
"metric_list": [
{
"metric": "acc",
"aggregation": "mean",
"higher_is_better": true
},
{
"metric": "acc_norm",
"aggregation": "mean",
"higher_is_better": true
}
],
"output_type": "multiple_choice",
"repeats": 1,
"should_decontaminate": false
}
},
"versions": {
"b4b": "N/A",
"b4bqa": "Yaml",
"medmcqa_g2b": "Yaml",
"medmcqa_orig_filtered": "Yaml",
"medqa_4options_g2b": "Yaml",
"medqa_4options_orig_filtered": "Yaml"
},
"n-shot": {
"b4b": 0,
"b4bqa": 0,
"medmcqa_g2b": 0,
"medmcqa_orig_filtered": 0,
"medqa_4options_g2b": 0,
"medqa_4options_orig_filtered": 0
},
"config": {
"model": "hf",
"model_args": "pretrained=microsoft/Phi-3-medium-4k-instruct,load_in_4bit=True",
"batch_size": "auto:64",
"batch_sizes": [
8,
16,
32,
32,
32,
32,
32,
32,
32,
32,
32,
32,
64,
64,
64,
64,
64,
64,
64,
64,
64,
64,
64,
64,
64,
64,
64,
64,
64,
64,
64,
64,
64,
64,
64,
64,
64,
64,
64,
64,
64,
64,
64,
64,
64,
64,
64,
64,
64,
64,
64,
64,
64,
64,
64,
64,
64,
64,
64,
64,
64,
64,
64,
64,
64
],
"device": "cuda:0",
"use_cache": null,
"limit": null,
"bootstrap_iters": 100000,
"gen_kwargs": null
},
"git_hash": "928c7657"
}