我想将sklearn.svm.LinearSVC
子类化,并将其用作sklearn.model_selection.GridSearchCV
的估计量。我之前在子类化方面遇到了一些问题,我想我已经根据我以前的post和选择的答案修复了它
但是,现在我的目标是创建一个sklearn.kernel_approximation.RBFSampler
对象作为新类的属性。这是一个例子,我这里有一个更广泛的问题:
GridSearchCV
一起使用,如何基于传递到构造函数中的参数值(或缺少参数值)创建属性?到目前为止,我已经尝试了以下方法:
from sklearn.datasets import make_classification
from sklearn.svm import LinearSVC
from sklearn.model_selection import GridSearchCV
from sklearn.kernel_approximation import RBFSampler
from sklearn.datasets import load_breast_cancer
RANDOM_STATE = 123
class LinearSVCSub(LinearSVC):
def __init__(self, penalty='l2', loss='squared_hinge', sampler_gamma=None, sampler_n=None,
dual=True, tol=0.0001, C=1.0, multi_class='ovr', fit_intercept=True, intercept_scaling=1,
class_weight=None, verbose=0, random_state=None, max_iter=1000):
super(LinearSVCSub, self).__init__(penalty=penalty, loss=loss, dual=dual, tol=tol,
C=C, multi_class=multi_class, fit_intercept=fit_intercept,
intercept_scaling=intercept_scaling, class_weight=class_weight,
verbose=verbose, random_state=random_state, max_iter=max_iter)
self.sampler_gamma = sampler_gamma
self.sampler_n = sampler_n
# I have also tried a conditional statement here instead of
# within a separate function create_sampler()
self.sampler = create_sampler()
def fit(self, X, y, sample_weight=None):
X = self.transform_this(X)
super(LinearSVCSub, self).fit(X, y, sample_weight)
return self
def predict(self, X):
X = self.transform_this(X)
return super(LinearSVCSub, self).predict(X)
def score(self, X, y, sample_weight=None):
X = self.transform_this(X)
return super(LinearSVCSub, self).score(X, y, sample_weight)
def decision_function(self, X):
X = self.transform_this(X)
return super(LinearSVCSub, self).decision_function(X)
def transform_this(self, X):
if self.sampler is not None:
X = sampler.fit_transform(X)
return X
def create_sampler(self):
# If sampler_gamma and sampler_n have been given, create a sampler
if (self.sampler_gamma is not None) and (self.sampler_n is not None):
sampler = RBFSampler(gamma=self.sampler_gamma, n_components=self.sampler_n)
else:
sampler = None
return sampler
if __name__ == '__main__':
data = load_breast_cancer()
X, y = data.data, data.target
# Parameter tuning with custom LinearSVC
param_grid = {'C': [0.00001, 0.0005],
'dual': (True, False), 'random_state': [RANDOM_STATE],
'sampler_gamma': [0.90, 0.60, 0.30],
'sampler_n': [10, 200]}
gs_model = GridSearchCV(estimator=LinearSVCSub(), verbose=1, param_grid=param_grid,
scoring='roc_auc', n_jobs=-1, cv=2)
gs_model.fit(X, y)
gs_model.cv_results_
但是,正如我所了解到的here,GridSearchCV首先使用默认值启动estimator对象,并且具有与^{
另外,我从上述代码中得到的错误是:
---------------------------------------------------------------------------
NameError Traceback (most recent call last)
<ipython-input-6-a11420cc931e> in <module>
66 'sampler_n': [10, 200]}
67
---> 68 gs_model = GridSearchCV(estimator=LinearSVCSub(), verbose=1, param_grid=param_grid,
69 scoring='roc_auc', n_jobs=-1, cv=2)
70 gs_model.fit(X, y)
<ipython-input-6-a11420cc931e> in __init__(self, penalty, loss, sampler_gamma, sampler_n, dual, tol, C, multi_class, fit_intercept, intercept_scaling, class_weight, verbose, random_state, max_iter)
21 self.sampler_n = sampler_n
22
---> 23 self.sampler = create_sampler()
24
25
NameError: name 'create_sampler' is not defined
__init__
构造函数作为容器来存储属性李>相关问题 更多 >
编程相关推荐