add flan_cot_zeroshot

c06b0d6e · lintangsutawika · 13940f1e · c06b0d6e · c06b0d6e · c06b0d6e
Commit c06b0d6e authored Sep 04, 2023 by lintangsutawika
20 changed files
--- a/lm_eval/tasks/bbh/flan_cot_zeroshot/_flan_cot_zeroshot_template_yaml
+++ b/lm_eval/tasks/bbh/flan_cot_zeroshot/_flan_cot_zeroshot_template_yaml
+group: bbh_flan_zeroshot
+dataset_path: lukaemon/bbh
+output_type: greedy_until
+test_split: test
+doc_to_target: "{{target}}"
+metric_list:
+  - metric: exact_match
+    aggregation: mean
+    higher_is_better: true
+    ignore_case: true
+    ignore_punctuation: true
+generation_kwargs:
+  until:
+    - "</s>"
+  do_sample: false
+  temperature: 0.0
+filter_list:
+  - name: "get-answer"
+    filter:
+      - function: "regex"
+        regex_pattern: "(?<=The answer is )(.*)(?=.)"
+      - function: "take_first"
\ No newline at end of file
--- a/lm_eval/tasks/bbh/flan_cot_zeroshot/boolean_expressions.yaml
+++ b/lm_eval/tasks/bbh/flan_cot_zeroshot/boolean_expressions.yaml
+"dataset_name": "boolean_expressions"
+"description": "Evaluate the result of a random Boolean expression.\n\n"
+"doc_to_text": "Q: {{input}}\nA: Let's think step by step.\n"
+"include": "_template_yaml"
+"task": "bbh_flan_cot_zeroshot_boolean_expressions"
--- a/lm_eval/tasks/bbh/flan_cot_zeroshot/causal_judgement.yaml
+++ b/lm_eval/tasks/bbh/flan_cot_zeroshot/causal_judgement.yaml
+"dataset_name": "causal_judgement"
+"description": "Answer questions about causal attribution.\n\n"
+"doc_to_text": "Q: {{input}}\nA: Let's think step by step.\n"
+"include": "_template_yaml"
+"task": "bbh_flan_cot_zeroshot_causal_judgement"
--- a/lm_eval/tasks/bbh/flan_cot_zeroshot/date_understanding.yaml
+++ b/lm_eval/tasks/bbh/flan_cot_zeroshot/date_understanding.yaml
+"dataset_name": "date_understanding"
+"description": "Infer the date from context.\n\n"
+"doc_to_text": "Q: {{input}}\nA: Let's think step by step.\n"
+"include": "_template_yaml"
+"task": "bbh_flan_cot_zeroshot_date_understanding"
--- a/lm_eval/tasks/bbh/flan_cot_zeroshot/disambiguation_qa.yaml
+++ b/lm_eval/tasks/bbh/flan_cot_zeroshot/disambiguation_qa.yaml
+"dataset_name": "disambiguation_qa"
+"description": "Clarify the meaning of sentences with ambiguous pronouns.\n\n"
+"doc_to_text": "Q: {{input}}\nA: Let's think step by step.\n"
+"include": "_template_yaml"
+"task": "bbh_flan_cot_zeroshot_disambiguation_qa"
--- a/lm_eval/tasks/bbh/flan_cot_zeroshot/dyck_languages.yaml
+++ b/lm_eval/tasks/bbh/flan_cot_zeroshot/dyck_languages.yaml
+"dataset_name": "dyck_languages"
+"description": "Correctly close a Dyck-n word.\n\n"
+"doc_to_text": "Q: {{input}}\nA: Let's think step by step.\n"
+"include": "_template_yaml"
+"task": "bbh_flan_cot_zeroshot_dyck_languages"
--- a/lm_eval/tasks/bbh/flan_cot_zeroshot/formal_fallacies.yaml
+++ b/lm_eval/tasks/bbh/flan_cot_zeroshot/formal_fallacies.yaml
+"dataset_name": "formal_fallacies"
+"description": "Distinguish deductively valid arguments from formal fallacies.\n\n"
+"doc_to_text": "Q: {{input}}\nA: Let's think step by step.\n"
+"include": "_template_yaml"
+"task": "bbh_flan_cot_zeroshot_formal_fallacies"
--- a/lm_eval/tasks/bbh/flan_cot_zeroshot/geometric_shapes.yaml
+++ b/lm_eval/tasks/bbh/flan_cot_zeroshot/geometric_shapes.yaml
+"dataset_name": "geometric_shapes"
+"description": "Name geometric shapes from their SVG paths.\n\n"
+"doc_to_text": "Q: {{input}}\nA: Let's think step by step.\n"
+"include": "_template_yaml"
+"task": "bbh_flan_cot_zeroshot_geometric_shapes"
--- a/lm_eval/tasks/bbh/flan_cot_zeroshot/hyperbaton.yaml
+++ b/lm_eval/tasks/bbh/flan_cot_zeroshot/hyperbaton.yaml
+"dataset_name": "hyperbaton"
+"description": "Order adjectives correctly in English sentences.\n\n"
+"doc_to_text": "Q: {{input}}\nA: Let's think step by step.\n"
+"include": "_template_yaml"
+"task": "bbh_flan_cot_zeroshot_hyperbaton"
--- a/lm_eval/tasks/bbh/flan_cot_zeroshot/logical_deduction_five_objects.yaml
+++ b/lm_eval/tasks/bbh/flan_cot_zeroshot/logical_deduction_five_objects.yaml
+"dataset_name": "logical_deduction_five_objects"
+"description": "A logical deduction task which requires deducing the order of a sequence of objects.\n\n"
+"doc_to_text": "Q: {{input}}\nA: Let's think step by step.\n"
+"include": "_template_yaml"
+"task": "bbh_flan_cot_zeroshot_logical_deduction_five_objects"
--- a/lm_eval/tasks/bbh/flan_cot_zeroshot/logical_deduction_seven_objects.yaml
+++ b/lm_eval/tasks/bbh/flan_cot_zeroshot/logical_deduction_seven_objects.yaml
+"dataset_name": "logical_deduction_seven_objects"
+"description": "A logical deduction task which requires deducing the order of a sequence of objects.\n\n"
+"doc_to_text": "Q: {{input}}\nA: Let's think step by step.\n"
+"include": "_template_yaml"
+"task": "bbh_flan_cot_zeroshot_logical_deduction_seven_objects"
--- a/lm_eval/tasks/bbh/flan_cot_zeroshot/logical_deduction_three_objects.yaml
+++ b/lm_eval/tasks/bbh/flan_cot_zeroshot/logical_deduction_three_objects.yaml
+"dataset_name": "logical_deduction_three_objects"
+"description": "A logical deduction task which requires deducing the order of a sequence of objects.\n\n"
+"doc_to_text": "Q: {{input}}\nA: Let's think step by step.\n"
+"include": "_template_yaml"
+"task": "bbh_flan_cot_zeroshot_logical_deduction_three_objects"
--- a/lm_eval/tasks/bbh/flan_cot_zeroshot/movie_recommendation.yaml
+++ b/lm_eval/tasks/bbh/flan_cot_zeroshot/movie_recommendation.yaml
+"dataset_name": "movie_recommendation"
+"description": "Recommend movies similar to the given list of movies.\n\n"
+"doc_to_text": "Q: {{input}}\nA: Let's think step by step.\n"
+"include": "_template_yaml"
+"task": "bbh_flan_cot_zeroshot_movie_recommendation"
--- a/lm_eval/tasks/bbh/flan_cot_zeroshot/multistep_arithmetic_two.yaml
+++ b/lm_eval/tasks/bbh/flan_cot_zeroshot/multistep_arithmetic_two.yaml
+"dataset_name": "multistep_arithmetic_two"
+"description": "Solve multi-step arithmetic problems.\n\n"
+"doc_to_text": "Q: {{input}}\nA: Let's think step by step.\n"
+"include": "_template_yaml"
+"task": "bbh_flan_cot_zeroshot_multistep_arithmetic_two"
--- a/lm_eval/tasks/bbh/flan_cot_zeroshot/navigate.yaml
+++ b/lm_eval/tasks/bbh/flan_cot_zeroshot/navigate.yaml
+"dataset_name": "navigate"
+"description": "Given a series of navigation instructions, determine whether one would end up back at the starting point.\n\n"
+"doc_to_text": "Q: {{input}}\nA: Let's think step by step.\n"
+"include": "_template_yaml"
+"task": "bbh_flan_cot_zeroshot_navigate"
--- a/lm_eval/tasks/bbh/flan_cot_zeroshot/object_counting.yaml
+++ b/lm_eval/tasks/bbh/flan_cot_zeroshot/object_counting.yaml
+"dataset_name": "object_counting"
+"description": "Questions that involve enumerating objects and asking the model to count them.\n\n"
+"doc_to_text": "Q: {{input}}\nA: Let's think step by step.\n"
+"include": "_template_yaml"
+"task": "bbh_flan_cot_zeroshot_object_counting"
--- a/lm_eval/tasks/bbh/flan_cot_zeroshot/penguins_in_a_table.yaml
+++ b/lm_eval/tasks/bbh/flan_cot_zeroshot/penguins_in_a_table.yaml
+"dataset_name": "penguins_in_a_table"
+"description": "Answer questions about a table of penguins and their attributes.\n\n"
+"doc_to_text": "Q: {{input}}\nA: Let's think step by step.\n"
+"include": "_template_yaml"
+"task": "bbh_flan_cot_zeroshot_penguins_in_a_table"
--- a/lm_eval/tasks/bbh/flan_cot_zeroshot/reasoning_about_colored_objects.yaml
+++ b/lm_eval/tasks/bbh/flan_cot_zeroshot/reasoning_about_colored_objects.yaml
+"dataset_name": "reasoning_about_colored_objects"
+"description": "Answer extremely simple questions about the colors of objects on a surface.\n\n"
+"doc_to_text": "Q: {{input}}\nA: Let's think step by step.\n"
+"include": "_template_yaml"
+"task": "bbh_flan_cot_zeroshot_reasoning_about_colored_objects"
--- a/lm_eval/tasks/bbh/flan_cot_zeroshot/ruin_names.yaml
+++ b/lm_eval/tasks/bbh/flan_cot_zeroshot/ruin_names.yaml
+"dataset_name": "ruin_names"
+"description": "Select the humorous edit that 'ruins' the input movie or musical artist name.\n\n"
+"doc_to_text": "Q: {{input}}\nA: Let's think step by step.\n"
+"include": "_template_yaml"
+"task": "bbh_flan_cot_zeroshot_ruin_names"
--- a/lm_eval/tasks/bbh/flan_cot_zeroshot/salient_translation_error_detection.yaml
+++ b/lm_eval/tasks/bbh/flan_cot_zeroshot/salient_translation_error_detection.yaml
+"dataset_name": "salient_translation_error_detection"
+"description": "Detect the type of error in an English translation of a German source sentence.\n\n"
+"doc_to_text": "Q: {{input}}\nA: Let's think step by step.\n"
+"include": "_template_yaml"
+"task": "bbh_flan_cot_zeroshot_salient_translation_error_detection"