Merge branch 'EleutherAI:main' into main

da211969 · Jess · GitHub · 1b97e487 · 801322e0 · da211969
Unverified Commit da211969 authored Jun 28, 2024 by Jess Committed by GitHub Jun 28, 2024
20 changed files
--- a/lm_eval/tasks/bertaqa/bertaqa_en_mt_latxa-70b-v1.1.yaml
+++ b/lm_eval/tasks/bertaqa/bertaqa_en_mt_latxa-70b-v1.1.yaml
+task: bertaqa_en_mt_latxa-70b-v1.1
+include: _bertaqa_template
+dataset_name: en_mt_latxa-70b-v1.1
+doc_to_text: "Question: {{question}}\nA: {{candidates[0]}}\nB: {{candidates[1]}}\nC: {{candidates[2]}}\nAnswer:"
--- a/lm_eval/tasks/bertaqa/bertaqa_en_mt_latxa-70b-v1.yaml
+++ b/lm_eval/tasks/bertaqa/bertaqa_en_mt_latxa-70b-v1.yaml
+task: bertaqa_en_mt_latxa-70b-v1
+include: _bertaqa_template
+dataset_name: en_mt_latxa-70b-v1
+doc_to_text: "Question: {{question}}\nA: {{candidates[0]}}\nB: {{candidates[1]}}\nC: {{candidates[2]}}\nAnswer:"
--- a/lm_eval/tasks/bertaqa/bertaqa_en_mt_latxa-7b-v1.1.yaml
+++ b/lm_eval/tasks/bertaqa/bertaqa_en_mt_latxa-7b-v1.1.yaml
+task: bertaqa_en_mt_latxa-7b-v1.1
+include: _bertaqa_template
+dataset_name: en_mt_latxa-7b-v1.1
+doc_to_text: "Question: {{question}}\nA: {{candidates[0]}}\nB: {{candidates[1]}}\nC: {{candidates[2]}}\nAnswer:"
--- a/lm_eval/tasks/bertaqa/bertaqa_en_mt_latxa-7b-v1.yaml
+++ b/lm_eval/tasks/bertaqa/bertaqa_en_mt_latxa-7b-v1.yaml
+task: bertaqa_en_mt_latxa-7b-v1
+include: _bertaqa_template
+dataset_name: en_mt_latxa-7b-v1
+doc_to_text: "Question: {{question}}\nA: {{candidates[0]}}\nB: {{candidates[1]}}\nC: {{candidates[2]}}\nAnswer:"
--- a/lm_eval/tasks/bertaqa/bertaqa_en_mt_llama-2-13b.yaml
+++ b/lm_eval/tasks/bertaqa/bertaqa_en_mt_llama-2-13b.yaml
+task: bertaqa_en_mt_llama-2-13b
+include: _bertaqa_template
+dataset_name: en_mt_llama-2-13b
+doc_to_text: "Question: {{question}}\nA: {{candidates[0]}}\nB: {{candidates[1]}}\nC: {{candidates[2]}}\nAnswer:"
--- a/lm_eval/tasks/bertaqa/bertaqa_en_mt_llama-2-70b.yaml
+++ b/lm_eval/tasks/bertaqa/bertaqa_en_mt_llama-2-70b.yaml
+task: bertaqa_en_mt_llama-2-70b
+include: _bertaqa_template
+dataset_name: en_mt_llama-2-70b
+doc_to_text: "Question: {{question}}\nA: {{candidates[0]}}\nB: {{candidates[1]}}\nC: {{candidates[2]}}\nAnswer:"
--- a/lm_eval/tasks/bertaqa/bertaqa_en_mt_llama-2-7b.yaml
+++ b/lm_eval/tasks/bertaqa/bertaqa_en_mt_llama-2-7b.yaml
+task: bertaqa_en_mt_llama-2-7b
+include: _bertaqa_template
+dataset_name: en_mt_llama-2-7b
+doc_to_text: "Question: {{question}}\nA: {{candidates[0]}}\nB: {{candidates[1]}}\nC: {{candidates[2]}}\nAnswer:"
--- a/lm_eval/tasks/bertaqa/bertaqa_en_mt_madlad.yaml
+++ b/lm_eval/tasks/bertaqa/bertaqa_en_mt_madlad.yaml
+task: bertaqa_en_mt_madlad
+include: _bertaqa_template
+dataset_name: en_mt_madlad
+doc_to_text: "Question: {{question}}\nA: {{candidates[0]}}\nB: {{candidates[1]}}\nC: {{candidates[2]}}\nAnswer:"
--- a/lm_eval/tasks/bertaqa/bertaqa_en_mt_nllb.yaml
+++ b/lm_eval/tasks/bertaqa/bertaqa_en_mt_nllb.yaml
+task: bertaqa_en_mt_nllb
+include: _bertaqa_template
+dataset_name: en_mt_nllb
+doc_to_text: "Question: {{question}}\nA: {{candidates[0]}}\nB: {{candidates[1]}}\nC: {{candidates[2]}}\nAnswer:"
--- a/lm_eval/tasks/bertaqa/bertaqa_eu.yaml
+++ b/lm_eval/tasks/bertaqa/bertaqa_eu.yaml
+task: bertaqa_eu
+include: _bertaqa_template
+dataset_name: eu
+doc_to_text: "Galdera: {{question}}\nA: {{candidates[0]}}\nB: {{candidates[1]}}\nC: {{candidates[2]}}\nErantzuna:"
--- a/lm_eval/tasks/bigbench/generate_tasks.py
+++ b/lm_eval/tasks/bigbench/generate_tasks.py
 import os

+import datasets
 import yaml


@@ -173,6 +174,11 @@ all_subtasks = [
    "word_unscrambling",
 ]

+skip_tasks = [
+    "simple_arithmetic_json_multiple_choice",
+    "simple_arithmetic_multiple_targets_json",
+]
+

 def main() -> None:
    for path, task_type in zip(
@@ -183,11 +189,29 @@ def main() -> None:
        for task in all_subtasks:
            file_name = f"{task}.yaml"
            try:
+                template_file = task_type
+                if path == "multiple_choice":
+                    print(f"Checking {task} for multiple choices")
+                    if task in skip_tasks:
+                        continue
+                    data = datasets.load_dataset("hails/bigbench", task + "_zero_shot")
+                    multiple_choice_targets = data["default"][0][
+                        "multiple_choice_targets"
+                    ]
+                    if len(multiple_choice_targets) == 0:
+                        continue
+                    else:
+                        template_file = "multiple_choice_template_b_yaml"
+                        if set(data["default"][0]["targets"]) < set(
+                            multiple_choice_targets
+                        ):
+                            template_file = "multiple_choice_template_a_yaml"
+
                with open(f"{path}/{file_name}", "w", encoding="utf-8") as f:
                    f.write("# Generated by utils.py\n")
                    yaml.dump(
                        {
-                            "include": f"../{task_type}",
+                            "include": f"../{template_file}",
                            "task": "bigbench_"
                            + task
                            + "_{}".format(task_type.split("_template_yaml")[0]),

--- a/lm_eval/tasks/bigbench/multiple_choice/abstract_narrative_understanding.yaml
+++ b/lm_eval/tasks/bigbench/multiple_choice/abstract_narrative_understanding.yaml
 # Generated by utils.py
 dataset_name: abstract_narrative_understanding_zero_shot
-include: ../multiple_choice_template_yaml
+include: ../multiple_choice_template_a_yaml
 task: bigbench_abstract_narrative_understanding_multiple_choice
--- a/lm_eval/tasks/bigbench/multiple_choice/anachronisms.yaml
+++ b/lm_eval/tasks/bigbench/multiple_choice/anachronisms.yaml
 # Generated by utils.py
 dataset_name: anachronisms_zero_shot
-include: ../multiple_choice_template_yaml
+include: ../multiple_choice_template_a_yaml
 task: bigbench_anachronisms_multiple_choice
--- a/lm_eval/tasks/bigbench/multiple_choice/analogical_similarity.yaml
+++ b/lm_eval/tasks/bigbench/multiple_choice/analogical_similarity.yaml
 # Generated by utils.py
 dataset_name: analogical_similarity_zero_shot
-include: ../multiple_choice_template_yaml
+include: ../multiple_choice_template_a_yaml
 task: bigbench_analogical_similarity_multiple_choice
--- a/lm_eval/tasks/bigbench/multiple_choice/analytic_entailment.yaml
+++ b/lm_eval/tasks/bigbench/multiple_choice/analytic_entailment.yaml
 # Generated by utils.py
 dataset_name: analytic_entailment_zero_shot
-include: ../multiple_choice_template_yaml
+include: ../multiple_choice_template_a_yaml
 task: bigbench_analytic_entailment_multiple_choice
--- a/lm_eval/tasks/bigbench/multiple_choice/arithmetic.yaml
+++ b/lm_eval/tasks/bigbench/multiple_choice/arithmetic.yaml
 # Generated by utils.py
 dataset_name: arithmetic_zero_shot
-include: ../multiple_choice_template_yaml
+include: ../multiple_choice_template_a_yaml
 task: bigbench_arithmetic_multiple_choice
--- a/lm_eval/tasks/bigbench/multiple_choice/ascii_word_recognition.yaml
+++ b/lm_eval/tasks/bigbench/multiple_choice/ascii_word_recognition.yaml
-# Generated by utils.py
-dataset_name: ascii_word_recognition_zero_shot
-include: ../multiple_choice_template_yaml
-task: bigbench_ascii_word_recognition_multiple_choice
--- a/lm_eval/tasks/bigbench/multiple_choice/authorship_verification.yaml
+++ b/lm_eval/tasks/bigbench/multiple_choice/authorship_verification.yaml
 # Generated by utils.py
 dataset_name: authorship_verification_zero_shot
-include: ../multiple_choice_template_yaml
+include: ../multiple_choice_template_a_yaml
 task: bigbench_authorship_verification_multiple_choice
--- a/lm_eval/tasks/bigbench/multiple_choice/auto_categorization.yaml
+++ b/lm_eval/tasks/bigbench/multiple_choice/auto_categorization.yaml
-# Generated by utils.py
-dataset_name: auto_categorization_zero_shot
-include: ../multiple_choice_template_yaml
-task: bigbench_auto_categorization_multiple_choice
--- a/lm_eval/tasks/bigbench/multiple_choice/auto_debugging.yaml
+++ b/lm_eval/tasks/bigbench/multiple_choice/auto_debugging.yaml
-# Generated by utils.py
-dataset_name: auto_debugging_zero_shot
-include: ../multiple_choice_template_yaml
-task: bigbench_auto_debugging_multiple_choice