Commit 88486e57 authored by lintangsutawika's avatar lintangsutawika
Browse files

Merge branch 'group-agg-rework' of...

Merge branch 'group-agg-rework' of https://github.com/EleutherAI/lm-evaluation-harness into multiprompt
parents 5971f2ca ba73d131
"dataset_name": "philosophy" "dataset_name": "philosophy"
"description": "The following are multiple choice questions (with answers) about philosophy.\n\ "description": "The following are multiple choice questions (with answers) about philosophy.\n\
\n" \n"
"group": "mmlu_humanities_tasks" "tag": "mmlu_humanities_tasks"
"include": "_default_template_yaml" "include": "_default_template_yaml"
"task": "mmlu_philosophy" "task": "mmlu_philosophy"
"task_alias": "philosophy" "task_alias": "philosophy"
"dataset_name": "prehistory" "dataset_name": "prehistory"
"description": "The following are multiple choice questions (with answers) about prehistory.\n\ "description": "The following are multiple choice questions (with answers) about prehistory.\n\
\n" \n"
"group": "mmlu_humanities_tasks" "tag": "mmlu_humanities_tasks"
"include": "_default_template_yaml" "include": "_default_template_yaml"
"task": "mmlu_prehistory" "task": "mmlu_prehistory"
"task_alias": "prehistory" "task_alias": "prehistory"
"dataset_name": "professional_accounting" "dataset_name": "professional_accounting"
"description": "The following are multiple choice questions (with answers) about professional\ "description": "The following are multiple choice questions (with answers) about professional\
\ accounting.\n\n" \ accounting.\n\n"
"group": "mmlu_other_tasks" "tag": "mmlu_other_tasks"
"include": "_default_template_yaml" "include": "_default_template_yaml"
"task": "mmlu_professional_accounting" "task": "mmlu_professional_accounting"
"task_alias": "professional_accounting" "task_alias": "professional_accounting"
"dataset_name": "professional_law" "dataset_name": "professional_law"
"description": "The following are multiple choice questions (with answers) about professional\ "description": "The following are multiple choice questions (with answers) about professional\
\ law.\n\n" \ law.\n\n"
"group": "mmlu_humanities_tasks" "tag": "mmlu_humanities_tasks"
"include": "_default_template_yaml" "include": "_default_template_yaml"
"task": "mmlu_professional_law" "task": "mmlu_professional_law"
"task_alias": "professional_law" "task_alias": "professional_law"
"dataset_name": "professional_medicine" "dataset_name": "professional_medicine"
"description": "The following are multiple choice questions (with answers) about professional\ "description": "The following are multiple choice questions (with answers) about professional\
\ medicine.\n\n" \ medicine.\n\n"
"group": "mmlu_other_tasks" "tag": "mmlu_other_tasks"
"include": "_default_template_yaml" "include": "_default_template_yaml"
"task": "mmlu_professional_medicine" "task": "mmlu_professional_medicine"
"task_alias": "professional_medicine" "task_alias": "professional_medicine"
"dataset_name": "professional_psychology" "dataset_name": "professional_psychology"
"description": "The following are multiple choice questions (with answers) about professional\ "description": "The following are multiple choice questions (with answers) about professional\
\ psychology.\n\n" \ psychology.\n\n"
"group": "mmlu_social_sciences_tasks" "tag": "mmlu_social_sciences_tasks"
"include": "_default_template_yaml" "include": "_default_template_yaml"
"task": "mmlu_professional_psychology" "task": "mmlu_professional_psychology"
"task_alias": "professional_psychology" "task_alias": "professional_psychology"
"dataset_name": "public_relations" "dataset_name": "public_relations"
"description": "The following are multiple choice questions (with answers) about public\ "description": "The following are multiple choice questions (with answers) about public\
\ relations.\n\n" \ relations.\n\n"
"group": "mmlu_social_sciences_tasks" "tag": "mmlu_social_sciences_tasks"
"include": "_default_template_yaml" "include": "_default_template_yaml"
"task": "mmlu_public_relations" "task": "mmlu_public_relations"
"task_alias": "public_relations" "task_alias": "public_relations"
"dataset_name": "security_studies" "dataset_name": "security_studies"
"description": "The following are multiple choice questions (with answers) about security\ "description": "The following are multiple choice questions (with answers) about security\
\ studies.\n\n" \ studies.\n\n"
"group": "mmlu_social_sciences_tasks" "tag": "mmlu_social_sciences_tasks"
"include": "_default_template_yaml" "include": "_default_template_yaml"
"task": "mmlu_security_studies" "task": "mmlu_security_studies"
"task_alias": "security_studies" "task_alias": "security_studies"
"dataset_name": "sociology" "dataset_name": "sociology"
"description": "The following are multiple choice questions (with answers) about sociology.\n\ "description": "The following are multiple choice questions (with answers) about sociology.\n\
\n" \n"
"group": "mmlu_social_sciences_tasks" "tag": "mmlu_social_sciences_tasks"
"include": "_default_template_yaml" "include": "_default_template_yaml"
"task": "mmlu_sociology" "task": "mmlu_sociology"
"task_alias": "sociology" "task_alias": "sociology"
"dataset_name": "us_foreign_policy" "dataset_name": "us_foreign_policy"
"description": "The following are multiple choice questions (with answers) about us\ "description": "The following are multiple choice questions (with answers) about us\
\ foreign policy.\n\n" \ foreign policy.\n\n"
"group": "mmlu_social_sciences_tasks" "tag": "mmlu_social_sciences_tasks"
"include": "_default_template_yaml" "include": "_default_template_yaml"
"task": "mmlu_us_foreign_policy" "task": "mmlu_us_foreign_policy"
"task_alias": "us_foreign_policy" "task_alias": "us_foreign_policy"
"dataset_name": "virology" "dataset_name": "virology"
"description": "The following are multiple choice questions (with answers) about virology.\n\ "description": "The following are multiple choice questions (with answers) about virology.\n\
\n" \n"
"group": "mmlu_other_tasks" "tag": "mmlu_other_tasks"
"include": "_default_template_yaml" "include": "_default_template_yaml"
"task": "mmlu_virology" "task": "mmlu_virology"
"task_alias": "virology" "task_alias": "virology"
"dataset_name": "world_religions" "dataset_name": "world_religions"
"description": "The following are multiple choice questions (with answers) about world\ "description": "The following are multiple choice questions (with answers) about world\
\ religions.\n\n" \ religions.\n\n"
"group": "mmlu_humanities_tasks" "tag": "mmlu_humanities_tasks"
"include": "_default_template_yaml" "include": "_default_template_yaml"
"task": "mmlu_world_religions" "task": "mmlu_world_religions"
"task_alias": "world_religions" "task_alias": "world_religions"
group: mmlu_flan_cot_fewshot group: mmlu_flan_cot_fewshot
group_alias: mmlu (flan style, fewshot cot)
task: task:
- mmlu_flan_cot_fewshot_stem - group: stem
- mmlu_flan_cot_fewshot_other task:
- mmlu_flan_cot_fewshot_social_sciences - mmlu_flan_cot_fewshot_stem
- mmlu_flan_cot_fewshot_humanities aggregate_metric_list:
group_config: - metric: acc
aggregate_metric: True weight_by_size: True
weight_by_size: True - group: other
task:
- mmlu_flan_cot_fewshot_other
aggregate_metric_list:
- metric: acc
weight_by_size: True
- group: social sciences
task:
- mmlu_flan_cot_fewshot_social_sciences
aggregate_metric_list:
- metric: acc
weight_by_size: True
- group: humanities
task:
- mmlu_flan_cot_fewshot_humanities
aggregate_metric_list:
- metric: acc
weight_by_size: True
aggregate_metric_list:
- metric: acc
weight_by_size: True
metadata:
version: 1
...@@ -27,3 +27,5 @@ metric_list: ...@@ -27,3 +27,5 @@ metric_list:
ignore_punctuation: true ignore_punctuation: true
metadata: metadata:
version: 1.0 version: 1.0
dataset_kwargs:
trust_remote_code: true
...@@ -54,6 +54,6 @@ fewshot_config: ...@@ -54,6 +54,6 @@ fewshot_config:
not have any roots. For c = 2 the polynomial x^2 + 2 has two roots at x = 1 not have any roots. For c = 2 the polynomial x^2 + 2 has two roots at x = 1
and x = 2. Hence Z_3[x]/(x^2 + c) is a field if and only if c = 1. The answer and x = 2. Hence Z_3[x]/(x^2 + c) is a field if and only if c = 1. The answer
is (B).' is (B).'
group: mmlu_flan_cot_fewshot_stem tag: mmlu_flan_cot_fewshot_stem
include: _mmlu_flan_cot_fewshot_template_yaml include: _mmlu_flan_cot_fewshot_template_yaml
task: mmlu_flan_cot_fewshot_abstract_algebra task: mmlu_flan_cot_fewshot_abstract_algebra
...@@ -70,6 +70,6 @@ fewshot_config: ...@@ -70,6 +70,6 @@ fewshot_config:
\ origin of the hyoid bone are the second and the third pharyngeal arches\u2014\ \ origin of the hyoid bone are the second and the third pharyngeal arches\u2014\
this information is covered in the last option (D). Therefore, we conclude that\ this information is covered in the last option (D). Therefore, we conclude that\
\ (D) must be the correct answer. The answer is (D).\n\n" \ (D) must be the correct answer. The answer is (D).\n\n"
group: mmlu_flan_cot_fewshot_stem tag: mmlu_flan_cot_fewshot_stem
include: _mmlu_flan_cot_fewshot_template_yaml include: _mmlu_flan_cot_fewshot_template_yaml
task: mmlu_flan_cot_fewshot_anatomy task: mmlu_flan_cot_fewshot_anatomy
...@@ -65,6 +65,6 @@ fewshot_config: ...@@ -65,6 +65,6 @@ fewshot_config:
because it explains that the surface is red due to the rusted materials on the because it explains that the surface is red due to the rusted materials on the
surface and the red color comes from the rust. So the correct option is (A). surface and the red color comes from the rust. So the correct option is (A).
The answer is (A).' The answer is (A).'
group: mmlu_flan_cot_fewshot_stem tag: mmlu_flan_cot_fewshot_stem
include: _mmlu_flan_cot_fewshot_template_yaml include: _mmlu_flan_cot_fewshot_template_yaml
task: mmlu_flan_cot_fewshot_astronomy task: mmlu_flan_cot_fewshot_astronomy
...@@ -70,6 +70,6 @@ fewshot_config: ...@@ -70,6 +70,6 @@ fewshot_config:
\ moral arguments relating to: negative *externalities*, the *power* that corporations\ \ moral arguments relating to: negative *externalities*, the *power* that corporations\
\ possess and the *mutual independence* of business and society. The answer\ \ possess and the *mutual independence* of business and society. The answer\
\ is (D).\n\n" \ is (D).\n\n"
group: mmlu_flan_cot_fewshot_other tag: mmlu_flan_cot_fewshot_other
include: _mmlu_flan_cot_fewshot_template_yaml include: _mmlu_flan_cot_fewshot_template_yaml
task: mmlu_flan_cot_fewshot_business_ethics task: mmlu_flan_cot_fewshot_business_ethics
...@@ -43,6 +43,6 @@ fewshot_config: ...@@ -43,6 +43,6 @@ fewshot_config:
target: 'Let''s think step by step. We refer to Wikipedia articles on clinical target: 'Let''s think step by step. We refer to Wikipedia articles on clinical
knowledge for help. The energy for muscular contraction is provided by ATP (adenosine knowledge for help. The energy for muscular contraction is provided by ATP (adenosine
triphosphate), which is the powerhouse of the cell. The answer is (A).' triphosphate), which is the powerhouse of the cell. The answer is (A).'
group: mmlu_flan_cot_fewshot_other tag: mmlu_flan_cot_fewshot_other
include: _mmlu_flan_cot_fewshot_template_yaml include: _mmlu_flan_cot_fewshot_template_yaml
task: mmlu_flan_cot_fewshot_clinical_knowledge task: mmlu_flan_cot_fewshot_clinical_knowledge
...@@ -70,6 +70,6 @@ fewshot_config: ...@@ -70,6 +70,6 @@ fewshot_config:
that have different origins, which is not the case for the human and bird forearms, that have different origins, which is not the case for the human and bird forearms,
which rules out (D). Humans and birds do belong to the same clade - a group which rules out (D). Humans and birds do belong to the same clade - a group
of organisms composed of a common ancestor. The answer is (C).' of organisms composed of a common ancestor. The answer is (C).'
group: mmlu_flan_cot_fewshot_stem tag: mmlu_flan_cot_fewshot_stem
include: _mmlu_flan_cot_fewshot_template_yaml include: _mmlu_flan_cot_fewshot_template_yaml
task: mmlu_flan_cot_fewshot_college_biology task: mmlu_flan_cot_fewshot_college_biology
Markdown is supported
0% or .
You are about to add 0 people to the discussion. Proceed with caution.
Finish editing this message first!
Please register or to comment