Skip to content

Commit f2af493

Browse files
authored
[Feature] Add ATC Choice Version (#1019)
* Squashed commit of the following: commit c48ad19 Author: DseidLi <[email protected]> Date: Tue Apr 2 16:57:43 2024 +0800 add atc_choice commit 3ac6efe Merge: 2d4e559 8e3a9c3 Author: DseidLi <[email protected]> Date: Tue Apr 2 16:41:38 2024 +0800 Merge branch 'atc_choice' into atc_add_choice commit 8e3a9c3 Merge: 150a036 0a6a03f Author: DseidLi <[email protected]> Date: Tue Mar 26 04:47:07 2024 +0800 Merge branch 'main' into atc_choice Conflicts: configs/summarizers/needlebench.py opencompass/datasets/needlebench/multi.py opencompass/datasets/needlebench/origin.py opencompass/datasets/needlebench/parallel.py commit 150a036 Merge: 8d6ac9a 940dd18 Author: DseidLi <[email protected]> Date: Wed Mar 20 03:49:08 2024 +0800 Merge branch 'needlebench_fix' into atc_choice commit 8d6ac9a Author: DseidLi <[email protected]> Date: Wed Mar 20 03:41:49 2024 +0800 optimize needlebench code commit 940dd18 Author: DseidLi <[email protected]> Date: Wed Mar 20 03:39:46 2024 +0800 fix vllm commit d8be687 Merge: ecad78a 2527fda Author: DseidLi <[email protected]> Date: Tue Mar 19 21:07:08 2024 +0800 Merge remote-tracking branch 'origin/add_1M_dataset' into atc_choice commit 2527fda Author: DseidLi <[email protected]> Date: Tue Mar 19 16:03:40 2024 +0800 add model configs commit 75425ac Author: DseidLi <[email protected]> Date: Tue Mar 19 16:02:15 2024 +0800 add prompt postion args commit 367ba1b Author: DseidLi <[email protected]> Date: Wed Feb 28 21:40:00 2024 +0800 add Needlebench-1000K configs commit ecad78a Author: DseidLi <[email protected]> Date: Thu Mar 14 22:08:32 2024 +0800 fix atc commit 08772c0 Merge: 9f3f8cf caf1cf8 Author: DseidLi <[email protected]> Date: Thu Mar 14 22:07:28 2024 +0800 Merge branch 'main' into atc_choice Conflicts: configs/datasets/needlebench/readme.md configs/datasets/needlebench/readme_zh-CN.md configs/summarizers/needlebench.py opencompass/datasets/needlebench/atc.py opencompass/summarizers/needlebench.py commit 9f3f8cf Author: DseidLi <[email protected]> Date: Thu Mar 14 21:35:53 2024 +0800 add atc-choice test commit 52be7c1 Author: DseidLi <[email protected]> Date: Wed Mar 6 02:54:15 2024 +0800 update needlebench randomseed and add vllm qwen14b commit fc1effc Merge: 4e747ed caf1cf8 Author: DseidLi <[email protected]> Date: Wed Mar 6 02:51:14 2024 +0800 Merge branch 'main' into add_model_configs commit 31834f9 Merge: 7dabc82 120bf8b Author: DseidLi <[email protected]> Date: Sun Mar 3 23:29:42 2024 +0800 Merge branch 'main' of https://github.com/open-compass/opencompass into atc_choice commit 4e747ed Author: DseidLi <[email protected]> Date: Sun Mar 3 22:15:25 2024 +0800 add internlm2-lmdeploy model and gemma configs commit 7dabc82 Author: DseidLi <[email protected]> Date: Sat Mar 2 17:26:15 2024 +0800 add atc choice version -ZH commit 996f8ae Author: DseidLi <[email protected]> Date: Wed Feb 28 16:58:56 2024 +0800 update readme for needlebench commit f7266e8 Author: DseidLi <[email protected]> Date: Wed Feb 28 16:44:53 2024 +0800 move readme.md commit 1c73756 Author: DseidLi <[email protected]> Date: Wed Feb 28 16:38:31 2024 +0800 fix linting error commit b6524f3 Author: DseidLi <[email protected]> Date: Wed Feb 28 16:33:51 2024 +0800 lint summarizer commit c0d1190 Author: DseidLi <[email protected]> Date: Wed Feb 28 16:29:03 2024 +0800 add needlebench intro, fix summarizer commit 0965baf Author: DseidLi <[email protected]> Date: Mon Feb 26 13:31:26 2024 +0800 fix bug in needlebench summarizer commit 5d32b31 Author: DseidLi <[email protected]> Date: Sat Feb 24 03:19:08 2024 +0800 update act prompt commit af82a7f Merge: 32bf9fe 53fe788 Author: DseidLi <[email protected]> Date: Fri Feb 23 17:50:32 2024 +0800 Merge remote-tracking branch 'upstream/main' into needlebench commit 32bf9fe Author: DseidLi <[email protected]> Date: Fri Feb 23 17:31:32 2024 +0800 simplify needlebench 32k, 128k, 200k for eval commit a7cb025 Author: DseidLi <[email protected]> Date: Fri Feb 23 14:48:58 2024 +0800 add needlebench * fix summarizer * remove repeated code * remove chinese comments
1 parent b50d163 commit f2af493

File tree

10 files changed

+720
-9
lines changed

10 files changed

+720
-9
lines changed
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,134 @@
1+
from opencompass.openicl.icl_prompt_template import PromptTemplate
2+
from opencompass.openicl.icl_retriever import ZeroRetriever
3+
from opencompass.openicl.icl_inferencer import GenInferencer
4+
from opencompass.openicl.icl_evaluator import CircularEvaluator
5+
from opencompass.datasets.needlebench.atc_choice import NeedleBenchATCDataset
6+
from opencompass.utils.text_postprocessors import first_option_postprocess
7+
8+
# ----------------------- Prompt Content----------------------- #
9+
10+
few_shot_prompts = {
11+
'single_choice_prompts': {
12+
"single_choice_cn": [
13+
dict(role='HUMAN', prompt='问题: 毕东作为刘红的爸爸,对刘红的成长有重要影响。 \n在上面提供的打乱的家族关系文本中,"刘红"的能够向上追溯到的最年长的亲人是谁?\nA. 毕东\nB. 刘红\nC. 李清亮\nD. 王展\n'),
14+
dict(role='BOT', prompt='回答: A'),
15+
dict(role='HUMAN', prompt='问题: 梅建平作为黄玉兰的姥姥,对黄玉兰的成长有重要影响。梅建平的妈妈是朱丽丽。蓝鑫把黄玉兰视为其母亲。焦慧不仅是朱丽丽的妈妈,还是朱丽丽的榜样。 \n在上面提供的打乱的家族关系文本中,"蓝鑫"的能够向上追溯到的最年长的亲人是谁?\nA. 梅建平\nB. 朱丽丽\nC. 蓝鑫\nD. 焦慧\n'),
16+
dict(role='BOT', prompt='回答: D'),
17+
dict(role='HUMAN', prompt='问题: 毕东把柳金凤视为其姥姥。奉兵作为柳金凤的妈妈,对柳金凤的成长有重要影响。余萍把杨颖视为其爸爸。毕东在郭建华的生命中扮演着父亲的角色。常宁的外公是余萍。刘慧是郭建华所生的孩子。刘慧在杨颖的生命中扮演着外公的角色。 \n在上面提供的打乱的家族关系文本中,"常宁"的能够向上追溯到的最年长的亲人是谁?\nA. 柳金凤\nB. 毕东\nC. 奉兵\nD. 余萍\n'),
18+
dict(role='BOT', prompt='回答: C'),
19+
dict(role='HUMAN', prompt='问题: 魏丽丽在谢平的生命中扮演着奶奶的角色。郭兵是魏阳的姥姥。谢平是郑玉珍的外婆。丁颖把武波视为其外公。丁颖在李建国的生命中扮演着外婆的角色。武波的父亲是刘瑜。许玲把余桂芳视为其父亲。刘瑜把许玲视为其爷爷。李建国对郭兵来说,不只是一个爷爷,还是一个朋友。魏丽丽的外公是魏阳。 \n在上面提供的打乱的家族关系文本中,"郑玉珍"的能够向上追溯到的最年长的亲人是谁?\nA. 魏丽丽\nB. 刘瑜\nC. 李建国\nD. 余桂芳\n'),
20+
dict(role='BOT', prompt='回答: D'),
21+
dict(role='HUMAN', prompt='问题: {question}'),
22+
],
23+
"single_choice_en": [
24+
dict(role='HUMAN', prompt="Question: Jasmine Lane plays the role of James Hill's father in James Hill's life. \nGiven the scrambled family relationships described above, who is the eldest relative that 'James Hill' can trace back to in the context?\nA. Jasmine Lane\nB. James Hill\nC. Christopher Flores\nD. Paula Reeves\n"),
25+
dict(role='BOT', prompt='Answer: A'),
26+
dict(role='HUMAN', prompt="Question: For Michael Goodwin, Annette Valdez is not just a dad, but also a friend.For Annette Valdez, Mary Sanders is not just a father, but also a friend. \nGiven the scrambled family relationships described above, who is the eldest relative that 'Michael Goodwin' can trace back to in the context?\nA. Michael Goodwin\nB. Annette Valdez\nC. Mary Sanders\nD. James Mitchell\n"),
27+
dict(role='BOT', prompt='Answer: C'),
28+
dict(role='HUMAN', prompt="Question: Laura Garcia considers Mr. Stephen Valdez DDS as their mom.Mr. Stephen Valdez DDS's maternal grandfather is Elizabeth Smith.Laura Garcia is Ashley Johnson's mom. \nGiven the scrambled family relationships described above, who is the eldest relative that 'Ashley Johnson' can trace back to in the context?\nA. Ashley Johnson\nB. Mr. Stephen Valdez DDS\nC. Laura Garcia\nD. Elizabeth Smith\n"),
29+
dict(role='BOT', prompt='Answer: D'),
30+
dict(role='HUMAN', prompt="Question: Justin Sanchez's paternal grandfather is Laurie Bean.For Laurie Bean, Christina Kim is not just a mother, but also a friend.Jessica Bass, as Christina Kim's maternal grandfather, has a significant impact on Christina Kim's upbringing.Sarah Hampton considers Justin Sanchez as their father. \nGiven the scrambled family relationships described above, who is the eldest relative that 'Sarah Hampton' can trace back to in the context?\nA. Christina Kim\nB. Jessica Bass\nC. Laurie Bean\nD. Justin Sanchez\n"),
31+
dict(role='BOT', prompt='Answer: B'),
32+
dict(role='HUMAN', prompt='Question: {question}'),
33+
],
34+
"single_choice_cn_reasoning": [
35+
dict(role='HUMAN', prompt='问题: 毕东作为刘红的爸爸,对刘红的成长有重要影响。 \n在上面提供的打乱的家族关系文本中,"刘红"的能够向上追溯到的最年长的亲人是谁?\nA. 毕东\nB. 刘红\nC. 李清亮\nD. 王展\n'),
36+
dict(role='BOT', prompt='回答: 根据提供的信息,我们可以构建以下家族关系链:\n\n- 刘红的爸爸是毕东。\n\n因此,"刘红"能够向上追溯到的最年长的亲人是毕东。所以答案是:A'),
37+
38+
dict(role='HUMAN', prompt='问题: 梅建平作为黄玉兰的姥姥,对黄玉兰的成长有重要影响。梅建平的妈妈是朱丽丽。蓝鑫把黄玉兰视为其母亲。焦慧不仅是朱丽丽的妈妈,还是朱丽丽的榜样。 \n在上面提供的打乱的家族关系文本中,"蓝鑫"的能够向上追溯到的最年长的亲人是谁?\nA. 梅建平\nB. 朱丽丽\nC. 蓝鑫\nD. 焦慧\n'),
39+
dict(role='BOT', prompt='回答: 根据提供的信息,我们可以构建以下家族关系链:\n\n- 黄玉兰的姥姥是梅建平。\n- 梅建平的妈妈是朱丽丽。\n- 朱丽丽的妈妈是焦慧。\n- 蓝鑫把黄玉兰视为其母亲。\n\n因此,"蓝鑫"能够向上追溯到的最年长的亲人是焦慧。所以答案是:D'),
40+
41+
dict(role='HUMAN', prompt='问题: 毕东把柳金凤视为其姥姥。奉兵作为柳金凤的妈妈,对柳金凤的成长有重要影响。余萍把杨颖视为其爸爸。毕东在郭建华的生命中扮演着父亲的角色。常宁的外公是余萍。刘慧是郭建华所生的孩子。刘慧在杨颖的生命中扮演着外公的角色。 \n在上面提供的打乱的家族关系文本中,"常宁"的能够向上追溯到的最年长的亲人是谁?\nA. 柳金凤\nB. 毕东\nC. 奉兵\nD. 余萍\n'),
42+
dict(role='BOT', prompt='回答: 根据提供的信息,我们可以构建以下家族关系链:\n\n- 常宁的外公是余萍。\n- 余萍把杨颖视为其爸爸。\n- 杨颖的外公是刘慧。\n- 刘慧是郭建华所生的孩子。\n- 郭建华的父亲是毕东。\n- 毕东的姥姥是柳金凤。\n- 柳金凤的妈妈是奉兵。\n\n因此,"常宁"能够向上追溯到的最年长的亲人是奉兵。所以答案是:C'),
43+
44+
dict(role='HUMAN', prompt='问题: 魏丽丽在谢平的生命中扮演着奶奶的角色。郭兵是魏阳的姥姥。谢平是郑玉珍的外婆。丁颖把武波视为其外公。丁颖在李建国的生命中扮演着外婆的角色。武波的父亲是刘瑜。许玲把余桂芳视为其父亲。刘瑜把许玲视为其爷爷。李建国对郭兵来说,不只是一个爷爷,还是一个朋友。魏丽丽的外公是魏阳。 \n在上面提供的打乱的家族关系文本中,"郑玉珍"的能够向上追溯到的最年长的亲人是谁?\nA. 魏丽丽\nB. 刘瑜\nC. 李建国\nD. 余桂芳\n'),
45+
dict(role='BOT', prompt='回答: 根据提供的信息,我们可以构建以下家族关系链:\n\n- 郑玉珍的外婆是谢平。\n- 谢平的奶奶是魏丽丽。\n- 魏丽丽的外公是魏阳。 \n- 魏阳的姥姥是郭兵。\n- 郭兵的爷爷是李建国。 \n- 李建国的外婆是丁颖。 \n- 丁颖的外公是武波。 \n- 武波的父亲是刘瑜。\n- 刘瑜的爷爷是许玲。\n- 许玲的父亲是余桂芳。 \n\n因此,"郑玉珍"能够向上追溯到的最年长的亲人是余桂芳。所以答案是:D'),
46+
dict(role='HUMAN', prompt='问题: {question}'),
47+
],
48+
"single_choice_en_reasoning": [
49+
dict(role='HUMAN', prompt="Question: Sharon House, as Jessica Stewart's father, has a significant impact on Jessica Stewart's upbringing. \nGiven the scrambled family relationships described above, who is the eldest relative that 'Jessica Stewart' can trace back to in the context?\nA. Jack Burch\nB. Jessica Stewart\nC. Sharon House\nD. Carolyn Jackson\n"),
50+
dict(role='BOT', prompt="Answer: Based on the provided information, we can construct the following family relationship chain:\n\n- Sharon House, as Jessica Stewart's father, has a significant impact on Jessica Stewart's upbringing.\n\nTherefore, the eldest relative that 'Jessica Stewart' can trace back to in the context is Sharon House. The answer is: C"),
51+
dict(role='HUMAN', prompt="Question: For Robert Hill, Mikayla Scott is not just a paternal grandfather, but also a friend.Jacob Oconnor's paternal grandmother is Robert Hill. \nGiven the scrambled family relationships described above, who is the eldest relative that 'Jacob Oconnor' can trace back to in the context?\nA. Laura Holland\nB. Robert Hill\nC. Jacob Oconnor\nD. Mikayla Scott\n"),
52+
dict(role='BOT', prompt="Answer: Based on the provided information, we can construct the following family relationship chain:\n\n-Jacob Oconnor's paternal grandmother is Robert Hill. \n- For Robert Hill, Mikayla Scott is not just a paternal grandfather, but also a friend.\n\nTherefore, the eldest relative that 'Jacob Oconnor' can trace back to in the context is Mikayla Scott. The answer is: D"),
53+
dict(role='HUMAN', prompt="Question: Misty Moore plays the role of Barbara Fuentes's maternal grandfather in Barbara Fuentes's life.Jennifer Garcia, as Michael Martinez's grandmother, has a significant impact on Michael Martinez's upbringing.Michael Martinez is not only Misty Moore's father but also Misty Moore's role model. \nGiven the scrambled family relationships described above, who is the eldest relative that 'Barbara Fuentes' can trace back to in the context?\nA. Michael Martinez\nB. Jennifer Garcia\nC. Misty Moore\nD. Barbara Fuentes\n"),
54+
dict(role='BOT', prompt="Answer: Based on the provided information, we can construct the following family relationship chain:\n- Misty Moore plays the role of Barbara Fuentes's maternal grandfather in Barbara Fuentes's life. \n- Michael Martinez is not only Misty Moore's father but also Misty Moore's role model.\n- Jennifer Garcia, as Michael Martinez's grandmother, has a significant impact on Michael Martinez's upbringing.\n\nTherefore, the eldest relative that 'Barbara Fuentes' can trace back to in the context is Jennifer Garcia. The answer is: B"),
55+
dict(role='HUMAN', prompt="Question: Carlos Smith, as Mary Gay's grandfather, has a significant impact on Mary Gay's upbringing.Victor Dean considers Mary Gay as their grandfather.Marcus Miller, as Carlos Smith's paternal grandfather, has a significant impact on Carlos Smith's upbringing.Victor Dean is not only Danielle Yates's maternal grandmother but also Danielle Yates's role model.Danielle Yates is not only David Hernandez's paternal grandmother but also David Hernandez's role model.David Hernandez is Jennifer Williams's mom. \nGiven the scrambled family relationships described above, who is the eldest relative that 'Jennifer Williams' can trace back to in the context?\nA. Marcus Miller\nB. Carlos Smith\nC. Mary Gay\nD. Victor Dean\n"),
56+
dict(role='BOT', prompt="Answer: Based on the provided information, we can construct the following family relationship chain:\n\n- David Hernandez is Jennifer Williams's mom.\n- Danielle Yates is not only David Hernandez's paternal grandmother but also David Hernandez's role model.\n- Victor Dean is not only Danielle Yates's maternal grandmother but also Danielle Yates's role model.\n- Victor Dean considers Mary Gay as their grandfather. \n- Carlos Smith, as Mary Gay's grandfather, has a significant impact on Mary Gay's upbringing.\n- Marcus Miller, as Carlos Smith's paternal grandfather, has a significant impact on Carlos Smith's upbringing.\n\nTherefore, the eldest relative that 'Jennifer Williams' can trace back to in the context is Marcus Miller. The answer is: A"),
57+
dict(role='HUMAN', prompt='Question: {question}'),
58+
],
59+
},
60+
}
61+
62+
# ----------------------- Prompt Settings ----------------------- #
63+
needle_num_list = list(range(2, 20, 1))
64+
names_path = './data/needlebench/names.json'
65+
66+
repeats = 10
67+
68+
# Use Zero-Shot or not
69+
with_few_shot = True
70+
71+
# Max for this dataset is 4, should be set with `with_few_shot`
72+
few_shot_samples = 4
73+
74+
# Generate reasoning path or not, only for single choice
75+
with_reasoning = True
76+
77+
# Use circular evaluation or not
78+
with_circular_eval = True
79+
80+
needlebench_prompts = few_shot_prompts
81+
single_choice_prompts = needlebench_prompts['single_choice_prompts']
82+
83+
# Set few shot prompt number
84+
for _name in list(single_choice_prompts.keys()):
85+
if with_few_shot:
86+
assert few_shot_samples > 0 and few_shot_samples <= 4
87+
single_choice_prompts[_name] = \
88+
single_choice_prompts[_name][- few_shot_samples * 2 - 1:]
89+
90+
# ----------------------- Dataset Settings ----------------------- #
91+
92+
needlebench_datasets = []
93+
94+
95+
needlebench_atc_reader_cfg = dict(input_columns=["question"],
96+
output_column="answer")
97+
98+
for _name in list(single_choice_prompts.keys()):
99+
100+
needlebench_atc_infer_cfg = dict(
101+
prompt_template=dict(
102+
type=PromptTemplate,
103+
template=dict(
104+
round=(single_choice_prompts[_name])),
105+
),
106+
retriever=dict(type=ZeroRetriever),
107+
inferencer=dict(type=GenInferencer,),
108+
)
109+
110+
needlebench_atc_eval_cfg = dict(
111+
evaluator=dict(type=CircularEvaluator),
112+
pred_postprocessor=dict(type=first_option_postprocess, options='ABCD'))
113+
114+
current_needle_num_list = needle_num_list
115+
116+
for num_needles in current_needle_num_list:
117+
abbr = (f'NeedleBenchATCDataset-'
118+
f'{num_needles}Needle-{"EN" if "en" in _name else "ZH"}')
119+
language = "English" if "en" in _name else "Chinese"
120+
if 'reasoning' in _name:
121+
abbr += '-Reasoning'
122+
dataset_dict = {
123+
'abbr': abbr,
124+
'type': NeedleBenchATCDataset,
125+
'path': names_path,
126+
'num_needles': num_needles,
127+
'language': language,
128+
'repeats': repeats,
129+
'with_circular': with_circular_eval,
130+
'reader_cfg': needlebench_atc_reader_cfg,
131+
'infer_cfg': needlebench_atc_infer_cfg,
132+
'eval_cfg': needlebench_atc_eval_cfg
133+
}
134+
needlebench_datasets.append(dataset_dict)

0 commit comments

Comments
 (0)