Skip to content

Instantly share code, notes, and snippets.

@GINK03
Last active June 19, 2018 03:52
Show Gist options
  • Select an option

  • Save GINK03/c204d31c9bd246609c43d4949452d52d to your computer and use it in GitHub Desktop.

Select an option

Save GINK03/c204d31c9bd246609c43d4949452d52d to your computer and use it in GitHub Desktop.
lightgbm-kfold

lightgbmでkfoldする

対応するデータタイプ

  • pandas dataframe
  • numpy

対応していないデータタイプ

  • scisparseのhstack(連結)したやつ
  • 解決法は後述

注意

sparse matrixはkfoldできないので、細工が必要

例

# from sklearn.model_selection import train_test_split
# from sklearn.metrics import mean_squared_log_error
from sklearn.metrics import mean_squared_error
from math import sqrt
import lightgbm as lgb
from sklearn.cross_validation import KFold


def get_oof(clf, x_train, y, x_test):
    NFOLDS=5
    SEED=71
    kf = KFold(len(x_train), n_folds=NFOLDS, shuffle=True, random_state=SEED)
    oof_train = np.zeros((len(x_train),))
    oof_test = np.zeros((len(x_test),))
    oof_test_skf = np.empty((NFOLDS, len(x_test)))
    lgbm_params =  {
        'task': 'train',
        'boosting_type': 'gbdt',
        'objective': 'regression',
        'metric': 'rmse',
        # 'max_depth': 15,
        'num_leaves': 20,
        'feature_fraction': 0.9,
        'bagging_fraction': 0.75,
        'bagging_freq': 4,
        'learning_rate': 0.016*5,
        #'max_bin':1023,
        'verbose': 0
    }
    for i, (train_index, test_index) in enumerate(kf):
        print('\nFold {}'.format(i))
        x_tr = x_train[train_index]
        y_tr = y[train_index]
        y_te = y[test_index]
        x_te = x_train[test_index]
        lgtrain = lgb.Dataset(x_tr, y_tr)
                    #feature_name=x_train.columns.tolist())
        lgvalid = lgb.Dataset(x_te, y_te)
                    #feature_name=x_train.columns.tolist())
                    #categorical_feature = categorical)
        lgb_clf = lgb.train(
            lgbm_params,
            lgtrain,
            num_boost_round=20000,
            valid_sets=[lgtrain, lgvalid],
            valid_names=['train','valid'],
            early_stopping_rounds=50,
            verbose_eval=50
        )
        oof_train[test_index] = lgb_clf.predict(x_te)
        oof_test_skf[i, :]    = lgb_clf.predict(x_test)

    oof_test[:] = oof_test_skf.mean(axis=0)
    return oof_train, oof_test
#Ridge oof method from Faron's kernel
#I was using this to analyze my vectorization, but figured it would be interesting to add the results back into the dataset
#It doesn't really add much to the score, but it does help lightgbm converge faster
oof_train, oof_test = get_oof(None, trainX, np.log1p(trainy), testX)
rms = sqrt(mean_squared_error(trainy, oof_train))
print('LGB OOF RMSE: {}'.format(rms))
print("Modeling Stage")
preds = np.concatenate([oof_test])
sub = pd.read_csv('../input/sample_submission.csv')
sub['target'] = np.around(np.expm1(preds), 0)
sub.to_csv('sub_et2.csv', index=False)

hstackのkfold

scikit-learnのkfoldのindexerでインデックス指定でslicingできないので、銃所を逆にして、このようにする

from sklearn.metrics import mean_squared_error
from math import sqrt
def get_oof(clf, x_trains, y, x_tests):
    oof_train = np.zeros((ntrain,))
    oof_test = np.zeros((ntest,))
    oof_test_skf = np.empty((NFOLDS, ntest))
    lgbm_params =  {
        'task': 'train',
        'boosting_type': 'gbdt',
        'objective': 'regression',
        'metric': 'rmse',
        # 'max_depth': 15,
        'num_leaves': 270,
        'feature_fraction': 0.5,
        'bagging_fraction': 0.75,
        'bagging_freq': 4,
        'learning_rate': 0.016*6, 
        #'max_bin':1023,
        'verbose': 0
    }  
    for i, (train_index, test_index) in enumerate(kf):
        print('\nFold {}'.format(i))
        x_tr = hstack([x_trains[0][train_index], x_trains[1][train_index]])
        y_tr = y[train_index]
        y_te = y[test_index]
        x_te = hstack([x_trains[0][test_index], x_trains[1][test_index]])
        lgtrain = lgb.Dataset(x_tr, y_tr,
                  feature_name=df.columns.tolist() + vectorizer.get_feature_names(),
                  categorical_feature = categorical)
        lgvalid = lgb.Dataset(x_te, y_te,
                  feature_name=df.columns.tolist() + vectorizer.get_feature_names(),
                  categorical_feature = categorical)
        lgb_clf = lgb.train(
            lgbm_params,
            lgtrain,
            num_boost_round=20000,
            valid_sets=[lgtrain, lgvalid],
            valid_names=['train','valid'],
            early_stopping_rounds=50,
            verbose_eval=50
        )
        oof_train[test_index] = lgb_clf.predict(x_te)
        x_test = hstack([x_tests[0], x_tests[1]])
        oof_test_skf[i, :]    = lgb_clf.predict(x_test)

    oof_test[:] = oof_test_skf.mean(axis=0)
    return oof_train.reshape(-1, 1), oof_test.reshape(-1, 1)

#Ridge oof method from Faron's kernel
#I was using this to analyze my vectorization, but figured it would be interesting to add the results back into the dataset
#It doesn't really add much to the score, but it does help lightgbm converge faster
oof_train, oof_test = get_oof(None, [csr_matrix(df.loc[traindex,:].values),ready_df[0:traindex.shape[0]]], y, [csr_matrix(df.loc[testdex,:].values),ready_df[traindex.shape[0]:]])
rms = sqrt(mean_squared_error(y, oof_train))
print('LGB OOF RMSE: {}'.format(rms))
print("Modeling Stage")
preds = np.concatenate([oof_train, oof_test])
df['lgb_preds_next'] = preds
Sign up for free to join this conversation on GitHub. Already have an account? Sign in to comment