compass_math.py 1.98 KB
Newer Older
1
2
3
# This summarizer is used for `./datasets/compassbench_v1_math/compassbench_v1_math_gen`

compassbench_v1_math_groups = [
Fengzhe Zhou's avatar
Fengzhe Zhou committed
4
5
6
7
    {'name': 'math_acc_1_and_fill_in_blank', 'subsets': [['compassbench_v1_math-high-single_choice_cn', 'acc_1'], ['compassbench_v1_math-high-single_choice_en', 'acc_1'], ['compassbench_v1_math-middle-single_choice_cn', 'acc_1'], ['compassbench_v1_math-middle-single_choice_en', 'acc_1'], ['compassbench_v1_math-primary-cloze_cn', 'accuracy'], ['compassbench_v1_math-primary-cloze_en', 'accuracy']]},
    {'name': 'math_perf_4_and_fill_in_blank', 'subsets': [['compassbench_v1_math-high-single_choice_cn', 'perf_4'], ['compassbench_v1_math-high-single_choice_en', 'perf_4'], ['compassbench_v1_math-middle-single_choice_cn', 'perf_4'], ['compassbench_v1_math-middle-single_choice_en', 'perf_4'], ['compassbench_v1_math-primary-cloze_cn', 'accuracy'], ['compassbench_v1_math-primary-cloze_en', 'accuracy']]},
    {'name': 'math_perf_4_and_fill_in_blank_cn', 'subsets': [['compassbench_v1_math-high-single_choice_cn', 'perf_4'], ['compassbench_v1_math-middle-single_choice_cn', 'perf_4'], ['compassbench_v1_math-primary-cloze_cn', 'accuracy']]},
    {'name': 'math_perf_4_and_fill_in_blank_en', 'subsets': [['compassbench_v1_math-high-single_choice_en', 'perf_4'], ['compassbench_v1_math-middle-single_choice_en', 'perf_4'], ['compassbench_v1_math-primary-cloze_en', 'accuracy']]},
8
9
10
11
12
13
]


summarizer = dict(
    dataset_abbrs=[
        'math_perf_4_and_fill_in_blank',
Fengzhe Zhou's avatar
Fengzhe Zhou committed
14
15
        'math_perf_4_and_fill_in_blank_cn',
        'math_perf_4_and_fill_in_blank_en',
16
17
18
19
20
21
22
23
24
        ['compassbench_v1_math-high-single_choice_cn', 'perf_4'],
        ['compassbench_v1_math-high-single_choice_en', 'perf_4'],
        ['compassbench_v1_math-middle-single_choice_cn', 'perf_4'],
        ['compassbench_v1_math-middle-single_choice_en', 'perf_4'],
        ['compassbench_v1_math-primary-cloze_cn', 'accuracy'],
        ['compassbench_v1_math-primary-cloze_en', 'accuracy'],
    ],
    summary_groups=compassbench_v1_math_groups,
)