|
| 1 | +from opencompass.openicl.icl_prompt_template import PromptTemplate |
| 2 | +from opencompass.openicl.icl_retriever import ZeroRetriever |
| 3 | +from opencompass.openicl.icl_inferencer import GenInferencer |
| 4 | +from opencompass.openicl.icl_evaluator import CircularEvaluator |
| 5 | +from opencompass.datasets.needlebench.atc_choice import NeedleBenchATCDataset |
| 6 | +from opencompass.utils.text_postprocessors import first_option_postprocess |
| 7 | + |
| 8 | +# ----------------------- Prompt Content----------------------- # |
| 9 | + |
| 10 | +few_shot_prompts = { |
| 11 | + 'single_choice_prompts': { |
| 12 | + "single_choice_cn": [ |
| 13 | + dict(role='HUMAN', prompt='问题: 毕东作为刘红的爸爸,对刘红的成长有重要影响。 \n在上面提供的打乱的家族关系文本中,"刘红"的能够向上追溯到的最年长的亲人是谁?\nA. 毕东\nB. 刘红\nC. 李清亮\nD. 王展\n'), |
| 14 | + dict(role='BOT', prompt='回答: A'), |
| 15 | + dict(role='HUMAN', prompt='问题: 梅建平作为黄玉兰的姥姥,对黄玉兰的成长有重要影响。梅建平的妈妈是朱丽丽。蓝鑫把黄玉兰视为其母亲。焦慧不仅是朱丽丽的妈妈,还是朱丽丽的榜样。 \n在上面提供的打乱的家族关系文本中,"蓝鑫"的能够向上追溯到的最年长的亲人是谁?\nA. 梅建平\nB. 朱丽丽\nC. 蓝鑫\nD. 焦慧\n'), |
| 16 | + dict(role='BOT', prompt='回答: D'), |
| 17 | + dict(role='HUMAN', prompt='问题: 毕东把柳金凤视为其姥姥。奉兵作为柳金凤的妈妈,对柳金凤的成长有重要影响。余萍把杨颖视为其爸爸。毕东在郭建华的生命中扮演着父亲的角色。常宁的外公是余萍。刘慧是郭建华所生的孩子。刘慧在杨颖的生命中扮演着外公的角色。 \n在上面提供的打乱的家族关系文本中,"常宁"的能够向上追溯到的最年长的亲人是谁?\nA. 柳金凤\nB. 毕东\nC. 奉兵\nD. 余萍\n'), |
| 18 | + dict(role='BOT', prompt='回答: C'), |
| 19 | + dict(role='HUMAN', prompt='问题: 魏丽丽在谢平的生命中扮演着奶奶的角色。郭兵是魏阳的姥姥。谢平是郑玉珍的外婆。丁颖把武波视为其外公。丁颖在李建国的生命中扮演着外婆的角色。武波的父亲是刘瑜。许玲把余桂芳视为其父亲。刘瑜把许玲视为其爷爷。李建国对郭兵来说,不只是一个爷爷,还是一个朋友。魏丽丽的外公是魏阳。 \n在上面提供的打乱的家族关系文本中,"郑玉珍"的能够向上追溯到的最年长的亲人是谁?\nA. 魏丽丽\nB. 刘瑜\nC. 李建国\nD. 余桂芳\n'), |
| 20 | + dict(role='BOT', prompt='回答: D'), |
| 21 | + dict(role='HUMAN', prompt='问题: {question}'), |
| 22 | + ], |
| 23 | + "single_choice_en": [ |
| 24 | + dict(role='HUMAN', prompt="Question: Jasmine Lane plays the role of James Hill's father in James Hill's life. \nGiven the scrambled family relationships described above, who is the eldest relative that 'James Hill' can trace back to in the context?\nA. Jasmine Lane\nB. James Hill\nC. Christopher Flores\nD. Paula Reeves\n"), |
| 25 | + dict(role='BOT', prompt='Answer: A'), |
| 26 | + dict(role='HUMAN', prompt="Question: For Michael Goodwin, Annette Valdez is not just a dad, but also a friend.For Annette Valdez, Mary Sanders is not just a father, but also a friend. \nGiven the scrambled family relationships described above, who is the eldest relative that 'Michael Goodwin' can trace back to in the context?\nA. Michael Goodwin\nB. Annette Valdez\nC. Mary Sanders\nD. James Mitchell\n"), |
| 27 | + dict(role='BOT', prompt='Answer: C'), |
| 28 | + dict(role='HUMAN', prompt="Question: Laura Garcia considers Mr. Stephen Valdez DDS as their mom.Mr. Stephen Valdez DDS's maternal grandfather is Elizabeth Smith.Laura Garcia is Ashley Johnson's mom. \nGiven the scrambled family relationships described above, who is the eldest relative that 'Ashley Johnson' can trace back to in the context?\nA. Ashley Johnson\nB. Mr. Stephen Valdez DDS\nC. Laura Garcia\nD. Elizabeth Smith\n"), |
| 29 | + dict(role='BOT', prompt='Answer: D'), |
| 30 | + dict(role='HUMAN', prompt="Question: Justin Sanchez's paternal grandfather is Laurie Bean.For Laurie Bean, Christina Kim is not just a mother, but also a friend.Jessica Bass, as Christina Kim's maternal grandfather, has a significant impact on Christina Kim's upbringing.Sarah Hampton considers Justin Sanchez as their father. \nGiven the scrambled family relationships described above, who is the eldest relative that 'Sarah Hampton' can trace back to in the context?\nA. Christina Kim\nB. Jessica Bass\nC. Laurie Bean\nD. Justin Sanchez\n"), |
| 31 | + dict(role='BOT', prompt='Answer: B'), |
| 32 | + dict(role='HUMAN', prompt='Question: {question}'), |
| 33 | + ], |
| 34 | + "single_choice_cn_reasoning": [ |
| 35 | + dict(role='HUMAN', prompt='问题: 毕东作为刘红的爸爸,对刘红的成长有重要影响。 \n在上面提供的打乱的家族关系文本中,"刘红"的能够向上追溯到的最年长的亲人是谁?\nA. 毕东\nB. 刘红\nC. 李清亮\nD. 王展\n'), |
| 36 | + dict(role='BOT', prompt='回答: 根据提供的信息,我们可以构建以下家族关系链:\n\n- 刘红的爸爸是毕东。\n\n因此,"刘红"能够向上追溯到的最年长的亲人是毕东。所以答案是:A'), |
| 37 | + |
| 38 | + dict(role='HUMAN', prompt='问题: 梅建平作为黄玉兰的姥姥,对黄玉兰的成长有重要影响。梅建平的妈妈是朱丽丽。蓝鑫把黄玉兰视为其母亲。焦慧不仅是朱丽丽的妈妈,还是朱丽丽的榜样。 \n在上面提供的打乱的家族关系文本中,"蓝鑫"的能够向上追溯到的最年长的亲人是谁?\nA. 梅建平\nB. 朱丽丽\nC. 蓝鑫\nD. 焦慧\n'), |
| 39 | + dict(role='BOT', prompt='回答: 根据提供的信息,我们可以构建以下家族关系链:\n\n- 黄玉兰的姥姥是梅建平。\n- 梅建平的妈妈是朱丽丽。\n- 朱丽丽的妈妈是焦慧。\n- 蓝鑫把黄玉兰视为其母亲。\n\n因此,"蓝鑫"能够向上追溯到的最年长的亲人是焦慧。所以答案是:D'), |
| 40 | + |
| 41 | + dict(role='HUMAN', prompt='问题: 毕东把柳金凤视为其姥姥。奉兵作为柳金凤的妈妈,对柳金凤的成长有重要影响。余萍把杨颖视为其爸爸。毕东在郭建华的生命中扮演着父亲的角色。常宁的外公是余萍。刘慧是郭建华所生的孩子。刘慧在杨颖的生命中扮演着外公的角色。 \n在上面提供的打乱的家族关系文本中,"常宁"的能够向上追溯到的最年长的亲人是谁?\nA. 柳金凤\nB. 毕东\nC. 奉兵\nD. 余萍\n'), |
| 42 | + dict(role='BOT', prompt='回答: 根据提供的信息,我们可以构建以下家族关系链:\n\n- 常宁的外公是余萍。\n- 余萍把杨颖视为其爸爸。\n- 杨颖的外公是刘慧。\n- 刘慧是郭建华所生的孩子。\n- 郭建华的父亲是毕东。\n- 毕东的姥姥是柳金凤。\n- 柳金凤的妈妈是奉兵。\n\n因此,"常宁"能够向上追溯到的最年长的亲人是奉兵。所以答案是:C'), |
| 43 | + |
| 44 | + dict(role='HUMAN', prompt='问题: 魏丽丽在谢平的生命中扮演着奶奶的角色。郭兵是魏阳的姥姥。谢平是郑玉珍的外婆。丁颖把武波视为其外公。丁颖在李建国的生命中扮演着外婆的角色。武波的父亲是刘瑜。许玲把余桂芳视为其父亲。刘瑜把许玲视为其爷爷。李建国对郭兵来说,不只是一个爷爷,还是一个朋友。魏丽丽的外公是魏阳。 \n在上面提供的打乱的家族关系文本中,"郑玉珍"的能够向上追溯到的最年长的亲人是谁?\nA. 魏丽丽\nB. 刘瑜\nC. 李建国\nD. 余桂芳\n'), |
| 45 | + dict(role='BOT', prompt='回答: 根据提供的信息,我们可以构建以下家族关系链:\n\n- 郑玉珍的外婆是谢平。\n- 谢平的奶奶是魏丽丽。\n- 魏丽丽的外公是魏阳。 \n- 魏阳的姥姥是郭兵。\n- 郭兵的爷爷是李建国。 \n- 李建国的外婆是丁颖。 \n- 丁颖的外公是武波。 \n- 武波的父亲是刘瑜。\n- 刘瑜的爷爷是许玲。\n- 许玲的父亲是余桂芳。 \n\n因此,"郑玉珍"能够向上追溯到的最年长的亲人是余桂芳。所以答案是:D'), |
| 46 | + dict(role='HUMAN', prompt='问题: {question}'), |
| 47 | + ], |
| 48 | + "single_choice_en_reasoning": [ |
| 49 | + dict(role='HUMAN', prompt="Question: Sharon House, as Jessica Stewart's father, has a significant impact on Jessica Stewart's upbringing. \nGiven the scrambled family relationships described above, who is the eldest relative that 'Jessica Stewart' can trace back to in the context?\nA. Jack Burch\nB. Jessica Stewart\nC. Sharon House\nD. Carolyn Jackson\n"), |
| 50 | + dict(role='BOT', prompt="Answer: Based on the provided information, we can construct the following family relationship chain:\n\n- Sharon House, as Jessica Stewart's father, has a significant impact on Jessica Stewart's upbringing.\n\nTherefore, the eldest relative that 'Jessica Stewart' can trace back to in the context is Sharon House. The answer is: C"), |
| 51 | + dict(role='HUMAN', prompt="Question: For Robert Hill, Mikayla Scott is not just a paternal grandfather, but also a friend.Jacob Oconnor's paternal grandmother is Robert Hill. \nGiven the scrambled family relationships described above, who is the eldest relative that 'Jacob Oconnor' can trace back to in the context?\nA. Laura Holland\nB. Robert Hill\nC. Jacob Oconnor\nD. Mikayla Scott\n"), |
| 52 | + dict(role='BOT', prompt="Answer: Based on the provided information, we can construct the following family relationship chain:\n\n-Jacob Oconnor's paternal grandmother is Robert Hill. \n- For Robert Hill, Mikayla Scott is not just a paternal grandfather, but also a friend.\n\nTherefore, the eldest relative that 'Jacob Oconnor' can trace back to in the context is Mikayla Scott. The answer is: D"), |
| 53 | + dict(role='HUMAN', prompt="Question: Misty Moore plays the role of Barbara Fuentes's maternal grandfather in Barbara Fuentes's life.Jennifer Garcia, as Michael Martinez's grandmother, has a significant impact on Michael Martinez's upbringing.Michael Martinez is not only Misty Moore's father but also Misty Moore's role model. \nGiven the scrambled family relationships described above, who is the eldest relative that 'Barbara Fuentes' can trace back to in the context?\nA. Michael Martinez\nB. Jennifer Garcia\nC. Misty Moore\nD. Barbara Fuentes\n"), |
| 54 | + dict(role='BOT', prompt="Answer: Based on the provided information, we can construct the following family relationship chain:\n- Misty Moore plays the role of Barbara Fuentes's maternal grandfather in Barbara Fuentes's life. \n- Michael Martinez is not only Misty Moore's father but also Misty Moore's role model.\n- Jennifer Garcia, as Michael Martinez's grandmother, has a significant impact on Michael Martinez's upbringing.\n\nTherefore, the eldest relative that 'Barbara Fuentes' can trace back to in the context is Jennifer Garcia. The answer is: B"), |
| 55 | + dict(role='HUMAN', prompt="Question: Carlos Smith, as Mary Gay's grandfather, has a significant impact on Mary Gay's upbringing.Victor Dean considers Mary Gay as their grandfather.Marcus Miller, as Carlos Smith's paternal grandfather, has a significant impact on Carlos Smith's upbringing.Victor Dean is not only Danielle Yates's maternal grandmother but also Danielle Yates's role model.Danielle Yates is not only David Hernandez's paternal grandmother but also David Hernandez's role model.David Hernandez is Jennifer Williams's mom. \nGiven the scrambled family relationships described above, who is the eldest relative that 'Jennifer Williams' can trace back to in the context?\nA. Marcus Miller\nB. Carlos Smith\nC. Mary Gay\nD. Victor Dean\n"), |
| 56 | + dict(role='BOT', prompt="Answer: Based on the provided information, we can construct the following family relationship chain:\n\n- David Hernandez is Jennifer Williams's mom.\n- Danielle Yates is not only David Hernandez's paternal grandmother but also David Hernandez's role model.\n- Victor Dean is not only Danielle Yates's maternal grandmother but also Danielle Yates's role model.\n- Victor Dean considers Mary Gay as their grandfather. \n- Carlos Smith, as Mary Gay's grandfather, has a significant impact on Mary Gay's upbringing.\n- Marcus Miller, as Carlos Smith's paternal grandfather, has a significant impact on Carlos Smith's upbringing.\n\nTherefore, the eldest relative that 'Jennifer Williams' can trace back to in the context is Marcus Miller. The answer is: A"), |
| 57 | + dict(role='HUMAN', prompt='Question: {question}'), |
| 58 | + ], |
| 59 | + }, |
| 60 | +} |
| 61 | + |
| 62 | +# ----------------------- Prompt Settings ----------------------- # |
| 63 | +needle_num_list = list(range(2, 20, 1)) |
| 64 | +names_path = './data/needlebench/names.json' |
| 65 | + |
| 66 | +repeats = 10 |
| 67 | + |
| 68 | +# Use Zero-Shot or not |
| 69 | +with_few_shot = True |
| 70 | + |
| 71 | +# Max for this dataset is 4, should be set with `with_few_shot` |
| 72 | +few_shot_samples = 4 |
| 73 | + |
| 74 | +# Generate reasoning path or not, only for single choice |
| 75 | +with_reasoning = True |
| 76 | + |
| 77 | +# Use circular evaluation or not |
| 78 | +with_circular_eval = True |
| 79 | + |
| 80 | +needlebench_prompts = few_shot_prompts |
| 81 | +single_choice_prompts = needlebench_prompts['single_choice_prompts'] |
| 82 | + |
| 83 | +# Set few shot prompt number |
| 84 | +for _name in list(single_choice_prompts.keys()): |
| 85 | + if with_few_shot: |
| 86 | + assert few_shot_samples > 0 and few_shot_samples <= 4 |
| 87 | + single_choice_prompts[_name] = \ |
| 88 | + single_choice_prompts[_name][- few_shot_samples * 2 - 1:] |
| 89 | + |
| 90 | +# ----------------------- Dataset Settings ----------------------- # |
| 91 | + |
| 92 | +needlebench_datasets = [] |
| 93 | + |
| 94 | + |
| 95 | +needlebench_atc_reader_cfg = dict(input_columns=["question"], |
| 96 | + output_column="answer") |
| 97 | + |
| 98 | +for _name in list(single_choice_prompts.keys()): |
| 99 | + |
| 100 | + needlebench_atc_infer_cfg = dict( |
| 101 | + prompt_template=dict( |
| 102 | + type=PromptTemplate, |
| 103 | + template=dict( |
| 104 | + round=(single_choice_prompts[_name])), |
| 105 | + ), |
| 106 | + retriever=dict(type=ZeroRetriever), |
| 107 | + inferencer=dict(type=GenInferencer,), |
| 108 | + ) |
| 109 | + |
| 110 | + needlebench_atc_eval_cfg = dict( |
| 111 | + evaluator=dict(type=CircularEvaluator), |
| 112 | + pred_postprocessor=dict(type=first_option_postprocess, options='ABCD')) |
| 113 | + |
| 114 | + current_needle_num_list = needle_num_list |
| 115 | + |
| 116 | + for num_needles in current_needle_num_list: |
| 117 | + abbr = (f'NeedleBenchATCDataset-' |
| 118 | + f'{num_needles}Needle-{"EN" if "en" in _name else "ZH"}') |
| 119 | + language = "English" if "en" in _name else "Chinese" |
| 120 | + if 'reasoning' in _name: |
| 121 | + abbr += '-Reasoning' |
| 122 | + dataset_dict = { |
| 123 | + 'abbr': abbr, |
| 124 | + 'type': NeedleBenchATCDataset, |
| 125 | + 'path': names_path, |
| 126 | + 'num_needles': num_needles, |
| 127 | + 'language': language, |
| 128 | + 'repeats': repeats, |
| 129 | + 'with_circular': with_circular_eval, |
| 130 | + 'reader_cfg': needlebench_atc_reader_cfg, |
| 131 | + 'infer_cfg': needlebench_atc_infer_cfg, |
| 132 | + 'eval_cfg': needlebench_atc_eval_cfg |
| 133 | + } |
| 134 | + needlebench_datasets.append(dataset_dict) |
0 commit comments