Repository navigation
Expand file tree
/
Copy pathgood_fit.py
More file actions
106 lines (89 loc) · 5.41 KB
/
Copy pathgood_fit.py
File metadata and controls
106 lines (89 loc) · 5.41 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
import pandas as pd
import numpy as np
from sklearn.metrics \
import mean_squared_error, mean_absolute_error, accuracy_score, precision_score, recall_score, f1_score
from sklearn.model_selection import ParameterGrid, cross_validate
from sklearn.base import clone
# Authors: Mark Wang <markswang@uchicago.edu>,
# Kan Liu <liukan07@berkeley.edu>.
class GoodFit:
def __init__(self, objective):
assert objective in ['reg', 'clf'], "The objective has to be either 'reg' or 'clf'."
# if objective == 'reg':
# assert set(criteria).issubset(['mse', 'mae']), \
# "The criteria for regression models has to be from the list ['mse', 'mae']."
# else:
# assert set(criteria).issubset(['acc', 'recall', 'precision', 'f1']), \
# "The criteria for regression models has to be from the list ['acc', 'recall', 'precision', 'f1']."
self.objective = objective
# self.criteria = criteria
print("GoodFit(objective={})".format(self.objective))
def find_the_spot(self, model, params_grid, X_train, y_train, X_test, y_test, crossv=5):
assert type(params_grid) is dict, "The params_grid needs to a dictionary in which the keys are the possible " \
"parameters of the model, and the values are lists of possible instances of " \
"the parameters."
assert set(params_grid.keys()).issubset(list(model.get_params().keys())), "Parameter in params_grid not " \
"applicable to current model."
for value in params_grid.values():
assert type(value) is list, "Values in params_grid needs to be lists of possible instances of each parameter."
assert (type(X_train) is np.ndarray) | (type(X_train) is pd.DataFrame) | (type(X_train) is pd.Series), \
"Data needs to be either a Pandas DataFrame, Series or a Numpy array."
assert (type(y_train) is np.ndarray) | (type(y_train) is pd.DataFrame)| (type(y_train) is pd.Series), \
"Data needs to be either a Pandas DataFrame, Series or a Numpy array."
assert (type(X_test) is np.ndarray) | (type(X_test) is pd.DataFrame)| (type(X_test) is pd.Series), \
"Data needs to be either a Pandas DataFrame, Series or a Numpy array."
assert (type(y_test) is np.ndarray) | (type(y_test) is pd.DataFrame)| (type(y_test) is pd.Series), \
"Data needs to be either a Pandas DataFrame, Series or a Numpy array."
assert X_test.shape[1] == X_train.shape[1], "The number of columns of X_train and X_test must be equal."
self.base_model_ = model
self.params_grid = params_grid
list_of_param_combinations = ParameterGrid(params_grid)
results = pd.DataFrame(list_of_param_combinations)
if self.objective == 'reg':
train_mse = []
test_mse = []
train_mae = []
test_mae = []
estimators = []
for combo in list_of_param_combinations:
model_new = clone(self.base_model_).set_params(**combo)
model_new.fit(X_train, y_train)
y_pred_train = model_new.predict(X_train)
y_pred_test = model_new.predict(X_test)
mse_train = mean_squared_error(y_train, y_pred_train)
mse_test = mean_squared_error(y_test, y_pred_test)
mae_train = mean_absolute_error(y_train, y_pred_train)
mae_test = mean_absolute_error(y_test, y_pred_test)
train_mse.append(mse_train)
test_mse.append(mse_test)
train_mae.append(mae_train)
test_mae.append(mae_test)
estimators.append(model_new)
MSE_diff = [i - j for i, j in zip(train_mse, test_mse)]
MAE_diff = [i - j for i, j in zip(train_mae, test_mae)]
self.estimators_ = estimators
scores = pd.DataFrame({'MSE_train':train_mse, 'MSE_test':test_mse, 'MSE_diff':MSE_diff,
'MAE_train':train_mae, 'MAE_test':test_mae, 'MAE_diff':MAE_diff})
self.results_ = results.join(scores)
return self.results_
else:
training_acc = []
testing_acc = []
estimators = []
for combo in list_of_param_combinations:
model_new = clone(self.base_model_).set_params(**combo)
model_new.fit(X_train, y_train)
y_pred_train = model_new.predict(X_train)
y_pred_test = model_new.predict(X_test)
scores_train = cross_validate(model_new, X_train, y_train, cv=crossv, scoring=['accuracy'], n_jobs=-1)
acc_train = np.mean(scores_train['test_accuracy'])
scores_test = cross_validate(model_new, X_test, y_test, cv=crossv, scoring=['accuracy'], n_jobs=-1)
acc_test = np.mean(scores_test['test_accuracy'])
training_acc.append(acc_train)
testing_acc.append(acc_test)
estimators.append(model_new)
acc_diff = [i - j for i, j in zip(training_acc, testing_acc)]
self.estimators_ = estimators
scores = pd.DataFrame({'ACC_train': training_acc, 'ACC_test': testing_acc, 'ACC_diff': acc_diff})
self.results_ = results.join(scores)
return self.results_