|
{ |
|
"results": { |
|
"hellaswag": { |
|
"acc,none": 0.6504680342561243, |
|
"acc_stderr,none": 0.004758476684324035, |
|
"acc_norm,none": 0.8483369846644094, |
|
"acc_norm_stderr,none": 0.0035796087435066605, |
|
"alias": "hellaswag" |
|
} |
|
}, |
|
"configs": { |
|
"hellaswag": { |
|
"task": "hellaswag", |
|
"group": [ |
|
"multiple_choice" |
|
], |
|
"dataset_path": "/lustre07/scratch/gagan30/arocr/meta-llama/self_rewarding_models/eval/hellaswag", |
|
"training_split": "train", |
|
"validation_split": "validation", |
|
"process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n", |
|
"doc_to_text": "{{query}}", |
|
"doc_to_target": "{{label}}", |
|
"doc_to_choice": "choices", |
|
"description": "", |
|
"target_delimiter": " ", |
|
"fewshot_delimiter": "\n\n", |
|
"num_fewshot": 10, |
|
"metric_list": [ |
|
{ |
|
"metric": "acc", |
|
"aggregation": "mean", |
|
"higher_is_better": true |
|
}, |
|
{ |
|
"metric": "acc_norm", |
|
"aggregation": "mean", |
|
"higher_is_better": true |
|
} |
|
], |
|
"output_type": "multiple_choice", |
|
"repeats": 1, |
|
"should_decontaminate": false, |
|
"metadata": { |
|
"version": 1.0 |
|
} |
|
} |
|
}, |
|
"versions": { |
|
"hellaswag": 1.0 |
|
}, |
|
"n-shot": { |
|
"hellaswag": 10 |
|
}, |
|
"config": { |
|
"model": "vllm", |
|
"model_args": "pretrained=/lustre07/scratch/gagan30/arocr/meta-llama/self_rewarding_models/Voyage-dpo-1,tensor_parallel_size=1,dtype=auto,gpu_memory_utilization=0.9,data_parallel_size=1,max_model_len=4096", |
|
"batch_size": "auto:128", |
|
"batch_sizes": [], |
|
"device": "cuda", |
|
"use_cache": "/lustre07/scratch/gagan30/arocr/cache/", |
|
"limit": null, |
|
"bootstrap_iters": 100000, |
|
"gen_kwargs": null |
|
}, |
|
"git_hash": null |
|
} |