Repository navigation
Expand file tree
/
Copy pathtest_binaryClass.py
More file actions
124 lines (97 loc) · 5.48 KB
/
Copy pathtest_binaryClass.py
File metadata and controls
124 lines (97 loc) · 5.48 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
import argparse
from sklearn.model_selection import KFold, train_test_split
from transformers import AutoTokenizer
from transformers import AutoModelForSequenceClassification, logging
from utils import load_binary_training_data, datasetify, ProbTrainer
from utils import find_best_trial_hyperparam
from torch import cuda
from utils import load_binary_training_data, datasetify, ProbTrainer, getPredictionForTrainTest_binary
logging.set_verbosity_warning()
parser = argparse.ArgumentParser()
parser.add_argument('--in_train', dest='in_train', type=str,
help='Add filename for training data')
parser.add_argument('--feature', dest='FEATURE', nargs='?', const="",
help='Column name of feature, default is "" which leads to a creation of "title_abstract-feature" ')
parser.add_argument('--label', dest='LABEL', nargs='?', const="0 relevant-relevant",
help='Column name of label, default is "0 relevant-relevant"')
parser.add_argument('--in_model', dest='in_model', type=str,
help='Add filename for untrained model')
parser.add_argument('--optimal_hyperparameter', dest='optimal_hyperparameter', action='store_true',
help='Flag if hyperparameter should be optimized')
parser.add_argument('--no_optimal_hyperparameter', dest='optimal_hyperparameter', action='store_false',
help='Flag if hyperparameter should not be optimized')
parser.add_argument('--folds', dest='folds', type=int, nargs='?', const=4, default=4,
help='Number of folds for cross-validation. The complete dataset is used.'
)
parser.add_argument('--out_hyperparam', dest='out_hyperparam', type=str,
help='Add filename for storage of test runs during hyperaram search.')
parser.add_argument('--out_train', dest='out_train', type=str,
help='Add filename for predictions on training data')
parser.add_argument('--out_test', dest='out_test', type=str,
help='Add filename for predictions on test dataset')
parser.add_argument('--out_tmp_files', dest='out_tmp_files', type=str,
help='Add directory for storage of large tmp-files used for hyperparam-tuning.')
if __name__ == "__main__":
args = parser.parse_args()
LABEL = args.LABEL
FEATURE = args.FEATURE
FOLDS = args.folds
print('hyperparameter optimization:', args.optimal_hyperparameter)
print("label:", LABEL)
print("feature:", FEATURE)
print('folds:', FOLDS)
all_train_df, LABEL, FEATURE = load_binary_training_data(args.in_train, LABEL=LABEL, FEATURE=FEATURE)
print("label:", LABEL)
print("feature:", FEATURE)
tokenizer = AutoTokenizer.from_pretrained(args.in_model)
model = AutoModelForSequenceClassification.from_pretrained(args.in_model, num_labels=2)
if cuda.is_available():
model.cuda()
all_test_runs = []
all_train_runs = []
kf = KFold(n_splits=FOLDS, shuffle=True, random_state=43581)
count = 0
for train_index, test_index in kf.split(all_train_df):
print(f"Round {count} of {FOLDS}")
train, test = all_train_df.iloc[train_index], all_train_df.iloc[test_index]
if args.optimal_hyperparameter:
print("Hyperparameter search...")
# eval is used to find optimal hyperparameter
train_small, eval = train_test_split(train, test_size=0.1, random_state=43581)
best_trial = find_best_trial_hyperparam(train_df=train_small,
test_df=eval,
model=model,
tokenizer=tokenizer,
balanced=False,
model_tmp_dir=f"{args.out_tmp_files}/tmp_trainer",
ray_log_dir=f"{args.out_tmp_files}/raytune",
FEATURE=FEATURE,
label=LABEL)
best_trial_params = best_trial.hyperparameters
with open(args.out_hyperparam, "a") as f:
f.write(f'run: {count}\n')
f.write(f'{best_trial_params}\n\n')
print("\Training...")
trainer = ProbTrainer(model=model,
train_dataset=datasetify(train[FEATURE],
tokenizer,
train[LABEL].values),
balanced=False,
)
if args.optimal_hyperparameter:
trainer.train(trial=best_trial_params)
else:
trainer.train()
print("\nEvaluating...")
test_one_run = getPredictionForTrainTest_binary(test, trainer, FEATURE, LABEL, count)
train_one_run = getPredictionForTrainTest_binary(train, trainer, FEATURE, LABEL, count)
all_test_runs.append(test_one_run)
all_train_runs.append(train_one_run)
count += 1
# store test result
all_test_runs_df = pd.concat(all_test_runs)
all_train_runs_df = pd.concat(all_train_runs)
all_test_runs_df.to_csv(args.out_test, index=False)
all_train_runs_df.to_csv(args.out_train, index=False)
print("Predictions for test run on test sets saved to", args.out_test)
print("Predictions for test run on train sets saved to", args.out_train)