Merge branch 'big-refactor' into process-whitespace

3317c627 · lintangsutawika · 0a694910 · 8eab2a58 · 3317c627 · 3317c627
Commit 3317c627 authored Aug 16, 2023 by lintangsutawika
9 changed files
--- a/lm_eval/tasks/paws-x/paws_fr.yaml
+++ b/lm_eval/tasks/paws-x/paws_fr.yaml
+# Generated by utils.py
+dataset_name: fr
+doc_to_choice: '{{[sentence1+", n''est-ce pas? Oui, "+sentence2, sentence1+", n''est-ce
+  pas? No, "+sentence2]}}'
+doc_to_text: ''
+include: pawsx_template_yaml
+task: paws_fr
--- a/lm_eval/tasks/paws-x/paws_ja.yaml
+++ b/lm_eval/tasks/paws-x/paws_ja.yaml
+# Generated by utils.py
+dataset_name: ja
+doc_to_choice: '{{[sentence1+", ですね? はい, "+sentence2, sentence1+", ですね? いいえ, "+sentence2]}}'
+doc_to_text: ''
+include: pawsx_template_yaml
+task: paws_ja
--- a/lm_eval/tasks/paws-x/paws_ko.yaml
+++ b/lm_eval/tasks/paws-x/paws_ko.yaml
+# Generated by utils.py
+dataset_name: ko
+doc_to_choice: '{{[sentence1+", 맞죠? 예, "+sentence2, sentence1+", 맞죠? 아니요, "+sentence2]}}'
+doc_to_text: ''
+include: pawsx_template_yaml
+task: paws_ko
--- a/lm_eval/tasks/paws-x/paws_zh.yaml
+++ b/lm_eval/tasks/paws-x/paws_zh.yaml
+# Generated by utils.py
+dataset_name: zh
+doc_to_choice: '{{[sentence1+", 对吧? 是, "+sentence2, sentence1+", 对吧? 不是, "+sentence2]}}'
+doc_to_text: ''
+include: pawsx_template_yaml
+task: paws_zh
--- a/lm_eval/tasks/paws-x/pawsx_template_yaml
+++ b/lm_eval/tasks/paws-x/pawsx_template_yaml
+# This file will be included in the generated language-specific task configs.
+# It doesn't have a yaml file extension as it is not meant to be imported directly
+# by the harness.
+group: pawsx
+task: null
+dataset_path: paws-x
+dataset_name: null
+output_type: multiple_choice
+training_split: train
+validation_split: validation
+test_split: test
+doc_to_text: null
+doc_to_target: label
+doc_to_choice: null
+metric_list:
+  - metric: acc
+    aggregation: mean
+    higher_is_better: true
--- a/lm_eval/tasks/paws-x/utils.py
+++ b/lm_eval/tasks/paws-x/utils.py
+import argparse
+from typing import Dict, List
+
+import yaml
+
+
+# Different languages that are part of xnli.
+# These correspond to dataset names (Subsets) on HuggingFace.
+# A yaml file is generated by this script for each language.
+
+LANGUAGES = {
+    "de": {  # German
+        "QUESTION_WORD": "richtig",
+        "YES": "Ja",
+        "NO": "Nein",
+    },
+    "en": {  # English
+        "QUESTION_WORD": "right",
+        "YES": "Yes",
+        "NO": "No",
+    },
+    "es": {  # Spanish
+        "QUESTION_WORD": "verdad",
+        "YES": "Sí",
+        "NO": "No",
+    },
+    "fr": {  # French
+        "QUESTION_WORD": "n'est-ce pas",
+        "YES": "Oui",
+        "NO": "No",
+    },
+    "ja": {  # Japanese
+        "QUESTION_WORD": "ですね",
+        "YES": "はい",
+        "NO": "いいえ",
+    },
+    "ko": {  # Korean
+        "QUESTION_WORD": "맞죠",
+        "YES": "예",
+        "NO": "아니요",
+    },
+    "zh": {  # Chinese
+        "QUESTION_WORD": "对吧",
+        "YES": "是",
+        "NO": "不是",
+    },
+}
+
+
+def gen_lang_yamls(output_dir: str, overwrite: bool) -> None:
+    """
+    Generate a yaml file for each language.
+
+    :param output_dir: The directory to output the files to.
+    :param overwrite: Whether to overwrite files if they already exist.
+    """
+    err = []
+    for lang in LANGUAGES.keys():
+        file_name = f"paws_{lang}.yaml"
+        try:
+            QUESTION_WORD = LANGUAGES[lang]["QUESTION_WORD"]
+            YES = LANGUAGES[lang]["YES"]
+            NO = LANGUAGES[lang]["NO"]
+            with open(
+                f"{output_dir}/{file_name}", "w" if overwrite else "x", encoding="utf8"
+            ) as f:
+                f.write("# Generated by utils.py\n")
+                yaml.dump(
+                    {
+                        "include": "pawsx_template_yaml",
+                        "dataset_name": lang,
+                        "task": f"paws_{lang}",
+                        "doc_to_text": "",
+                        "doc_to_choice": f"{{{{["
+                        f"""sentence1+\", {QUESTION_WORD}? {YES}, \"+sentence2,"""
+                        f""" sentence1+\", {QUESTION_WORD}? {NO}, \"+sentence2"""
+                        f"]}}}}",
+                    },
+                    f,
+                    allow_unicode=True,
+                )
+        except FileExistsError:
+            err.append(file_name)
+
+    if len(err) > 0:
+        raise FileExistsError(
+            "Files were not created because they already exist (use --overwrite flag):"
+            f" {', '.join(err)}"
+        )
+
+
+def main() -> None:
+    """Parse CLI args and generate language-specific yaml files."""
+    parser = argparse.ArgumentParser()
+    parser.add_argument(
+        "--overwrite",
+        default=False,
+        action="store_true",
+        help="Overwrite files if they already exist",
+    )
+    parser.add_argument(
+        "--output-dir", default=".", help="Directory to write yaml files to"
+    )
+    args = parser.parse_args()
+
+    gen_lang_yamls(output_dir=args.output_dir, overwrite=args.overwrite)
+
+
+if __name__ == "__main__":
+    main()
--- a/lm_eval/tasks/super_glue/rte/default.yaml
+++ b/lm_eval/tasks/super_glue/rte/default.yaml
 group:
  - super-glue-lm-eval-v1
-task: rte
+task: sglue_rte
 dataset_path: super_glue
 dataset_name: rte
 output_type: multiple_choice

--- a/lm_eval/tasks/xstorycloze/README.md
+++ b/lm_eval/tasks/xstorycloze/README.md
-# XStoryCloze
+# Trivia QA

 ### Paper

-Title: `Few-shot Learning with Multilingual Language Models`
-Abstract: https://arxiv.org/abs/2112.10668
+Title: `TriviaQA: A Large Scale Distantly Supervised Challenge Dataset for Reading Comprehension`
+Abstract: https://arxiv.org/abs/1705.03551

-XStoryCloze consists of the professionally translated version of the [English StoryCloze dataset](https://cs.rochester.edu/nlp/rocstories/) (Spring 2016 version) to 10 non-English languages. This dataset is released by Meta AI.
+TriviaQA is a reading comprehension dataset containing over 650K question-answer-evidence
+triples. TriviaQA includes 95K question-answer pairs authored by trivia enthusiasts
+and independently gathered evidence documents, six per question on average, that provide
+high quality distant supervision for answering the questions.

-Homepage: https://github.com/facebookresearch/fairseq/pull/4820
+Homepage: https://nlp.cs.washington.edu/triviaqa/


 ### Citation

 ```
-@article{DBLP:journals/corr/abs-2112-10668,
-  author    = {Xi Victoria Lin and
-               Todor Mihaylov and
-               Mikel Artetxe and
-               Tianlu Wang and
-               Shuohui Chen and
-               Daniel Simig and
-               Myle Ott and
-               Naman Goyal and
-               Shruti Bhosale and
-               Jingfei Du and
-               Ramakanth Pasunuru and
-               Sam Shleifer and
-               Punit Singh Koura and
-               Vishrav Chaudhary and
-               Brian O'Horo and
-               Jeff Wang and
-               Luke Zettlemoyer and
-               Zornitsa Kozareva and
-               Mona T. Diab and
-               Veselin Stoyanov and
-               Xian Li},
-  title     = {Few-shot Learning with Multilingual Language Models},
-  journal   = {CoRR},
-  volume    = {abs/2112.10668},
-  year      = {2021},
-  url       = {https://arxiv.org/abs/2112.10668},
-  eprinttype = {arXiv},
-  eprint    = {2112.10668},
-  timestamp = {Tue, 04 Jan 2022 15:59:27 +0100},
-  biburl    = {https://dblp.org/rec/journals/corr/abs-2112-10668.bib},
-  bibsource = {dblp computer science bibliography, https://dblp.org}
+@InProceedings{JoshiTriviaQA2017,
+    author = {Joshi, Mandar and Choi, Eunsol and Weld, Daniel S. and Zettlemoyer, Luke},
+    title = {TriviaQA: A Large Scale Distantly Supervised Challenge Dataset for Reading Comprehension},
+    booktitle = {Proceedings of the 55th Annual Meeting of the Association for Computational Linguistics},
+    month = {July},
+    year = {2017},
+    address = {Vancouver, Canada},
+    publisher = {Association for Computational Linguistics},
 }
 ```

 ### Subtasks

 List or describe tasks defined in this folder, and their names here:
-* `task_name`: `1-sentence description of what this particular task does`
-* `task_name2`: .....
+* `triviaqa`: `Generate and answer based on the question.`

 ### Checklist


--- a/lm_eval/tasks/triviaqa/default.yaml
+++ b/lm_eval/tasks/triviaqa/default.yaml
+task: triviaqa
+dataset_path: trivia_qa
+dataset_name: rc.nocontext
+output_type: greedy_until
+training_split: train
+validation_split: validation
+doc_to_text: "Question: {{question}}?\nAnswer:"
+doc_to_target: "{{answer.aliases}}"
+should_decontaminate: true
+doc_to_decontamination_query: question
+generation_kwargs:
+  until:
+    - "\n"
+    - "."
+    - ","
+  do_sample: false
+  temperature: 0.0
+filter_list:
+  - name: remove_whitespace
+    filter:
+      - function: remove_whitespace
+      - function: take_first
+target_delimiter: " "
+metric_list:
+  - metric: exact_match
+    aggregation: mean
+    higher_is_better: true
+    ignore_case: true
+    ignore_punctuation: true