- pandas dataframe
- numpy
- scisparseのhstack(連結)したやつ
- 解決法は後述
sparse matrixはkfoldできないので、細工が必要
# from sklearn.model_selection import train_test_split
# from sklearn.metrics import mean_squared_log_error
from sklearn.metrics import mean_squared_error
from math import sqrt
import lightgbm as lgb
from sklearn.cross_validation import KFold
def get_oof(clf, x_train, y, x_test):
NFOLDS=5
SEED=71
kf = KFold(len(x_train), n_folds=NFOLDS, shuffle=True, random_state=SEED)
oof_train = np.zeros((len(x_train),))
oof_test = np.zeros((len(x_test),))
oof_test_skf = np.empty((NFOLDS, len(x_test)))
lgbm_params = {
'task': 'train',
'boosting_type': 'gbdt',
'objective': 'regression',
'metric': 'rmse',
# 'max_depth': 15,
'num_leaves': 20,
'feature_fraction': 0.9,
'bagging_fraction': 0.75,
'bagging_freq': 4,
'learning_rate': 0.016*5,
#'max_bin':1023,
'verbose': 0
}
for i, (train_index, test_index) in enumerate(kf):
print('\nFold {}'.format(i))
x_tr = x_train[train_index]
y_tr = y[train_index]
y_te = y[test_index]
x_te = x_train[test_index]
lgtrain = lgb.Dataset(x_tr, y_tr)
#feature_name=x_train.columns.tolist())
lgvalid = lgb.Dataset(x_te, y_te)
#feature_name=x_train.columns.tolist())
#categorical_feature = categorical)
lgb_clf = lgb.train(
lgbm_params,
lgtrain,
num_boost_round=20000,
valid_sets=[lgtrain, lgvalid],
valid_names=['train','valid'],
early_stopping_rounds=50,
verbose_eval=50
)
oof_train[test_index] = lgb_clf.predict(x_te)
oof_test_skf[i, :] = lgb_clf.predict(x_test)
oof_test[:] = oof_test_skf.mean(axis=0)
return oof_train, oof_test
#Ridge oof method from Faron's kernel
#I was using this to analyze my vectorization, but figured it would be interesting to add the results back into the dataset
#It doesn't really add much to the score, but it does help lightgbm converge faster
oof_train, oof_test = get_oof(None, trainX, np.log1p(trainy), testX)
rms = sqrt(mean_squared_error(trainy, oof_train))
print('LGB OOF RMSE: {}'.format(rms))
print("Modeling Stage")
preds = np.concatenate([oof_test])
sub = pd.read_csv('../input/sample_submission.csv')
sub['target'] = np.around(np.expm1(preds), 0)
sub.to_csv('sub_et2.csv', index=False)scikit-learnのkfoldのindexerでインデックス指定でslicingできないので、銃所を逆にして、このようにする
from sklearn.metrics import mean_squared_error
from math import sqrt
def get_oof(clf, x_trains, y, x_tests):
oof_train = np.zeros((ntrain,))
oof_test = np.zeros((ntest,))
oof_test_skf = np.empty((NFOLDS, ntest))
lgbm_params = {
'task': 'train',
'boosting_type': 'gbdt',
'objective': 'regression',
'metric': 'rmse',
# 'max_depth': 15,
'num_leaves': 270,
'feature_fraction': 0.5,
'bagging_fraction': 0.75,
'bagging_freq': 4,
'learning_rate': 0.016*6,
#'max_bin':1023,
'verbose': 0
}
for i, (train_index, test_index) in enumerate(kf):
print('\nFold {}'.format(i))
x_tr = hstack([x_trains[0][train_index], x_trains[1][train_index]])
y_tr = y[train_index]
y_te = y[test_index]
x_te = hstack([x_trains[0][test_index], x_trains[1][test_index]])
lgtrain = lgb.Dataset(x_tr, y_tr,
feature_name=df.columns.tolist() + vectorizer.get_feature_names(),
categorical_feature = categorical)
lgvalid = lgb.Dataset(x_te, y_te,
feature_name=df.columns.tolist() + vectorizer.get_feature_names(),
categorical_feature = categorical)
lgb_clf = lgb.train(
lgbm_params,
lgtrain,
num_boost_round=20000,
valid_sets=[lgtrain, lgvalid],
valid_names=['train','valid'],
early_stopping_rounds=50,
verbose_eval=50
)
oof_train[test_index] = lgb_clf.predict(x_te)
x_test = hstack([x_tests[0], x_tests[1]])
oof_test_skf[i, :] = lgb_clf.predict(x_test)
oof_test[:] = oof_test_skf.mean(axis=0)
return oof_train.reshape(-1, 1), oof_test.reshape(-1, 1)
#Ridge oof method from Faron's kernel
#I was using this to analyze my vectorization, but figured it would be interesting to add the results back into the dataset
#It doesn't really add much to the score, but it does help lightgbm converge faster
oof_train, oof_test = get_oof(None, [csr_matrix(df.loc[traindex,:].values),ready_df[0:traindex.shape[0]]], y, [csr_matrix(df.loc[testdex,:].values),ready_df[traindex.shape[0]:]])
rms = sqrt(mean_squared_error(y, oof_train))
print('LGB OOF RMSE: {}'.format(rms))
print("Modeling Stage")
preds = np.concatenate([oof_train, oof_test])
df['lgb_preds_next'] = preds