-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathgenerate_dataset.py
More file actions
109 lines (95 loc) · 4.34 KB
/
Copy pathgenerate_dataset.py
File metadata and controls
109 lines (95 loc) · 4.34 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
from pathlib import Path
import json
from invariance_bench.question_generation import generate_order_question, generate_reshuffle_question, generate_reconstruction_question
output_dir = Path(f"./data/datasets/")
output_dir.mkdir(parents=True, exist_ok=True)
# INIT_NUM_ELEMENTS = 5
# MAX_NUM_ELEMENTS = 50
# STEP_SIZE = 5
BASE_NUM_ELEMENTS = 2
MAX_EXPONENT = 11
INIT_EXPONENT = 2
# generate original questions
original_questions = []
equivalent_questions = []
reshuffle_questions = []
reconstruction_org_questions = []
reconstruction_eqv_questions = []
"""
Data format:
{
"question": ...,
"answer": ...,
"is_fwd": ...,
"num_elements": ...,
}
"""
comb = [(True, True), (True, False), (False, True), (False, False)]
for i in range(4):
print(f"Generating questions for combination {i+1} of {len(comb)}")
print(f"Combination: {comb[i]}")
for exp in range(INIT_EXPONENT, MAX_EXPONENT+1, 1):
n = BASE_NUM_ELEMENTS ** exp
print(f"Generating questions for n = {n}")
for j in range(100):
question = generate_order_question(n, is_fwd=comb[i][0], answer=comb[i][1],
min_distance=min(n, int(0.2*n)))
original_questions.append({
"question": question[0],
"answer": 'yes' if question[3] else 'no',
"is_fwd": question[2],
"num_elements": n
})
equivalent_questions.append({
"question": question[1],
"answer": 'yes' if question[3] else 'no',
"is_fwd": question[2],
"num_elements": n
})
reshuffle = generate_reshuffle_question(n, is_fwd=comb[i][0], answer=comb[i][1])
reshuffle_questions.append({
"question": reshuffle[1],
"answer": 'yes' if reshuffle[3] else 'no',
"is_fwd": reshuffle[2],
"num_elements": n
})
# Reconstruction questions (only need one per ordering, not per is_fwd/answer)
if i == 0: # generate once per n per sample
recon = generate_reconstruction_question(n, min_distance=min(n, int(0.2*n)))
reconstruction_org_questions.append({
"question": recon[0],
"true_order": recon[2],
"num_elements": n
})
reconstruction_eqv_questions.append({
"question": recon[1],
"true_order": recon[2],
"num_elements": n
})
# # save original questions as jsonl
# with open(output_dir / f"original_questions_minElement{INIT_NUM_ELEMENTS}_maxElement{MAX_NUM_ELEMENTS}_stepSize{STEP_SIZE}.jsonl", "w") as f:
# for question in original_questions:
# f.write(json.dumps(question) + "\n")
# # save equivalent questions as jsonl
# with open(output_dir / f"equivalent_questions_minElement{INIT_NUM_ELEMENTS}_maxElement{MAX_NUM_ELEMENTS}_stepSize{STEP_SIZE}.jsonl", "w") as f:
# for question in equivalent_questions:
# f.write(json.dumps(question) + "\n")
# save original questions as jsonl
with open(output_dir / f"original_questions_base{BASE_NUM_ELEMENTS}_maxExp{MAX_EXPONENT}_minExp{INIT_EXPONENT}.jsonl", "w") as f:
for question in original_questions:
f.write(json.dumps(question) + "\n")
# save equivalent questions as jsonl
with open(output_dir / f"equivalent_questions_base{BASE_NUM_ELEMENTS}_maxExp{MAX_EXPONENT}_minExp{INIT_EXPONENT}.jsonl", "w") as f:
for question in equivalent_questions:
f.write(json.dumps(question) + "\n")
# save reshuffle-only questions as jsonl
with open(output_dir / f"reshuffle_questions_base{BASE_NUM_ELEMENTS}_maxExp{MAX_EXPONENT}_minExp{INIT_EXPONENT}.jsonl", "w") as f:
for question in reshuffle_questions:
f.write(json.dumps(question) + "\n")
# save reconstruction questions as jsonl
with open(output_dir / f"reconstruction_original_base{BASE_NUM_ELEMENTS}_maxExp{MAX_EXPONENT}_minExp{INIT_EXPONENT}.jsonl", "w") as f:
for question in reconstruction_org_questions:
f.write(json.dumps(question) + "\n")
with open(output_dir / f"reconstruction_equivalent_base{BASE_NUM_ELEMENTS}_maxExp{MAX_EXPONENT}_minExp{INIT_EXPONENT}.jsonl", "w") as f:
for question in reconstruction_eqv_questions:
f.write(json.dumps(question) + "\n")