Spaces:

AIM-Harvard
/

rabbits-leaderboard

Runtime error

rabbits-leaderboard / data /raw-eval-outputs /mistralai-Mistral-7B-v0.3_results.json

magilogi

rabbits-leaderboard-v0.1

4c59875 6 months ago

7.55 kB

	{
	"results": {
	"b4b": {
	"acc,none": 0.6199136868064118,
	"acc_stderr,none": 0.0837373393352743,
	"acc_norm,none": 0.6199136868064118,
	"acc_norm_stderr,none": 0.0837373393352743,
	"alias": "b4b"
	},
	"b4bqa": {
	"acc,none": 0.703125,
	"acc_stderr,none": 0.010795811437682205,
	"acc_norm,none": 0.703125,
	"acc_norm_stderr,none": 0.010795811437682205,
	"alias": " - b4bqa"
	},
	"medmcqa_g2b": {
	"acc,none": 0.4827586206896552,
	"acc_stderr,none": 0.026825443578224806,
	"acc_norm,none": 0.4827586206896552,
	"acc_norm_stderr,none": 0.026825443578224806,
	"alias": " - medmcqa_g2b"
	},
	"medmcqa_orig_filtered": {
	"acc,none": 0.5689655172413793,
	"acc_stderr,none": 0.026584851780353615,
	"acc_norm,none": 0.5689655172413793,
	"acc_norm_stderr,none": 0.026584851780353615,
	"alias": " - medmcqa_orig_filtered"
	},
	"medqa_4options_g2b": {
	"acc,none": 0.48677248677248675,
	"acc_stderr,none": 0.025742297289575142,
	"acc_norm,none": 0.48677248677248675,
	"acc_norm_stderr,none": 0.025742297289575142,
	"alias": " - medqa_4options_g2b"
	},
	"medqa_4options_orig_filtered": {
	"acc,none": 0.5317460317460317,
	"acc_stderr,none": 0.0256993528321318,
	"acc_norm,none": 0.5317460317460317,
	"acc_norm_stderr,none": 0.0256993528321318,
	"alias": " - medqa_4options_orig_filtered"
	}
	},
	"groups": {
	"b4b": {
	"acc,none": 0.6199136868064118,
	"acc_stderr,none": 0.0837373393352743,
	"acc_norm,none": 0.6199136868064118,
	"acc_norm_stderr,none": 0.0837373393352743,
	"alias": "b4b"
	}
	},
	"configs": {
	"b4bqa": {
	"task": "b4bqa",
	"dataset_path": "AIM-Harvard/b4b_drug_qa",
	"test_split": "test",
	"doc_to_text": "<function process_cd at 0x7f6e18c5ff70>",
	"doc_to_target": "correct_choice",
	"doc_to_choice": [
	"A",
	"B",
	"C",
	"D"
	],
	"description": "",
	"target_delimiter": " ",
	"fewshot_delimiter": "\n\n",
	"metric_list": [
	{
	"metric": "acc",
	"aggregation": "mean",
	"higher_is_better": true
	},
	{
	"metric": "acc_norm",
	"aggregation": "mean",
	"higher_is_better": true
	}
	],
	"output_type": "multiple_choice",
	"repeats": 1,
	"should_decontaminate": false
	},
	"medmcqa_g2b": {
	"task": "medmcqa_g2b",
	"dataset_path": "AIM-Harvard/medmcqa_generic_to_brand",
	"training_split": "train",
	"validation_split": "validation",
	"test_split": "validation",
	"doc_to_text": "<function doc_to_text at 0x7f6e1919e430>",
	"doc_to_target": "cop",
	"doc_to_choice": [
	"A",
	"B",
	"C",
	"D"
	],
	"description": "",
	"target_delimiter": " ",
	"fewshot_delimiter": "\n\n",
	"metric_list": [
	{
	"metric": "acc",
	"aggregation": "mean",
	"higher_is_better": true
	},
	{
	"metric": "acc_norm",
	"aggregation": "mean",
	"higher_is_better": true
	}
	],
	"output_type": "multiple_choice",
	"repeats": 1,
	"should_decontaminate": true,
	"doc_to_decontamination_query": "{{question}}"
	},
	"medmcqa_orig_filtered": {
	"task": "medmcqa_orig_filtered",
	"dataset_path": "AIM-Harvard/medmcqa_original",
	"training_split": "train",
	"validation_split": "validation",
	"test_split": "validation",
	"doc_to_text": "<function doc_to_text at 0x7f6e18acb3a0>",
	"doc_to_target": "cop",
	"doc_to_choice": [
	"A",
	"B",
	"C",
	"D"
	],
	"description": "",
	"target_delimiter": " ",
	"fewshot_delimiter": "\n\n",
	"metric_list": [
	{
	"metric": "acc",
	"aggregation": "mean",
	"higher_is_better": true
	},
	{
	"metric": "acc_norm",
	"aggregation": "mean",
	"higher_is_better": true
	}
	],
	"output_type": "multiple_choice",
	"repeats": 1,
	"should_decontaminate": true,
	"doc_to_decontamination_query": "{{question}}"
	},
	"medqa_4options_g2b": {
	"task": "medqa_4options_g2b",
	"dataset_path": "AIM-Harvard/gbaker_medqa_usmle_4_options_hf_generic_to_brand",
	"training_split": "train",
	"validation_split": "validation",
	"test_split": "test",
	"doc_to_text": "<function doc_to_text at 0x7f6e1919e8b0>",
	"doc_to_target": "<function doc_to_target at 0x7f6e1919ec10>",
	"doc_to_choice": [
	"A",
	"B",
	"C",
	"D"
	],
	"description": "",
	"target_delimiter": " ",
	"fewshot_delimiter": "\n\n",
	"metric_list": [
	{
	"metric": "acc",
	"aggregation": "mean",
	"higher_is_better": true
	},
	{
	"metric": "acc_norm",
	"aggregation": "mean",
	"higher_is_better": true
	}
	],
	"output_type": "multiple_choice",
	"repeats": 1,
	"should_decontaminate": false
	},
	"medqa_4options_orig_filtered": {
	"task": "medqa_4options_orig_filtered",
	"dataset_path": "AIM-Harvard/gbaker_medqa_usmle_4_options_hf_original",
	"training_split": "train",
	"validation_split": "validation",
	"test_split": "test",
	"doc_to_text": "<function doc_to_text at 0x7f6e18c80550>",
	"doc_to_target": "<function doc_to_target at 0x7f6e18c2ff70>",
	"doc_to_choice": [
	"A",
	"B",
	"C",
	"D"
	],
	"description": "",
	"target_delimiter": " ",
	"fewshot_delimiter": "\n\n",
	"metric_list": [
	{
	"metric": "acc",
	"aggregation": "mean",
	"higher_is_better": true
	},
	{
	"metric": "acc_norm",
	"aggregation": "mean",
	"higher_is_better": true
	}
	],
	"output_type": "multiple_choice",
	"repeats": 1,
	"should_decontaminate": false
	}
	},
	"versions": {
	"b4b": "N/A",
	"b4bqa": "Yaml",
	"medmcqa_g2b": "Yaml",
	"medmcqa_orig_filtered": "Yaml",
	"medqa_4options_g2b": "Yaml",
	"medqa_4options_orig_filtered": "Yaml"
	},
	"n-shot": {
	"b4b": 0,
	"b4bqa": 0,
	"medmcqa_g2b": 0,
	"medmcqa_orig_filtered": 0,
	"medqa_4options_g2b": 0,
	"medqa_4options_orig_filtered": 0
	},
	"config": {
	"model": "hf",
	"model_args": "pretrained=mistralai/Mistral-7B-v0.3,load_in_4bit=True",
	"batch_size": "auto:64",
	"batch_sizes": [
	32,
	64,
	64,
	64,
	64,
	64,
	64,
	64,
	64,
	64,
	64,
	64,
	64,
	64,
	64,
	64,
	64,
	64,
	64,
	64,
	64,
	64,
	64,
	64,
	64,
	64,
	64,
	64,
	64,
	64,
	64,
	64,
	64,
	64,
	64,
	64,
	64,
	64,
	64,
	64,
	64,
	64,
	64,
	64,
	64,
	64,
	64,
	64,
	64,
	64,
	64,
	64,
	64,
	64,
	64,
	64,
	64,
	64,
	64,
	64,
	64,
	64,
	64,
	64,
	64
	],
	"device": "cuda:0",
	"use_cache": null,
	"limit": null,
	"bootstrap_iters": 100000,
	"gen_kwargs": null
	},
	"git_hash": "928c7657"
	}