{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 8.6 안전 운전자 예측 경진대회 성능 개선 III : LightGBM과 XGBoost 앙상블","metadata":{"papermill":{"duration":0.022779,"end_time":"2021-08-09T04:29:36.356459","exception":false,"start_time":"2021-08-09T04:29:36.33368","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"- [안전 운전자 예측 경진대회 링크](https://www.kaggle.com/c/porto-seguro-safe-driver-prediction)\n- [모델링 코드 참고 링크](https://www.kaggle.com/xiaozhouwang/2nd-place-lightgbm-solution)","metadata":{"papermill":{"duration":0.021359,"end_time":"2021-08-09T04:29:36.402047","exception":false,"start_time":"2021-08-09T04:29:36.380688","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import pandas as pd\n\n# 데이터 경로\ndata_path = '/kaggle/input/porto-seguro-safe-driver-prediction/'\n\ntrain = pd.read_csv(data_path + 'train.csv', index_col='id')\ntest = pd.read_csv(data_path + 'test.csv', index_col='id')\nsubmission = pd.read_csv(data_path + 'sample_submission.csv', index_col='id')","metadata":{"papermill":{"duration":10.778988,"end_time":"2021-08-09T04:29:47.253204","exception":false,"start_time":"2021-08-09T04:29:36.474216","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-06T14:28:36.106872Z","iopub.execute_input":"2022-08-06T14:28:36.107797Z","iopub.status.idle":"2022-08-06T14:28:46.356637Z","shell.execute_reply.started":"2022-08-06T14:28:36.107649Z","shell.execute_reply":"2022-08-06T14:28:46.355626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_data = pd.concat([train, test], ignore_index=True)\nall_data = all_data.drop('target', axis=1) # 타깃값 제거\n\nall_features = all_data.columns # 전체 피처","metadata":{"papermill":{"duration":2.503854,"end_time":"2021-08-09T04:29:49.786437","exception":false,"start_time":"2021-08-09T04:29:47.282583","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-06T14:28:46.358543Z","iopub.execute_input":"2022-08-06T14:28:46.358877Z","iopub.status.idle":"2022-08-06T14:28:47.543861Z","shell.execute_reply.started":"2022-08-06T14:28:46.358835Z","shell.execute_reply":"2022-08-06T14:28:47.542819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import OneHotEncoder\n\ncat_features = [feature for feature in all_features if 'cat' in feature] # 명목형 피처\n\n# 원-핫 인코딩 적용\nonehot_encoder = OneHotEncoder()\nencoded_cat_matrix = onehot_encoder.fit_transform(all_data[cat_features]) ","metadata":{"papermill":{"duration":3.244069,"end_time":"2021-08-09T04:29:53.052723","exception":false,"start_time":"2021-08-09T04:29:49.808654","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-06T14:28:47.545103Z","iopub.execute_input":"2022-08-06T14:28:47.545365Z","iopub.status.idle":"2022-08-06T14:28:50.553576Z","shell.execute_reply.started":"2022-08-06T14:28:47.545338Z","shell.execute_reply":"2022-08-06T14:28:50.552319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# '데이터 하나당 결측값 개수'를 파생 피로 추가\nall_data['num_missing'] = (all_data==-1).sum(axis=1)","metadata":{"papermill":{"duration":0.285664,"end_time":"2021-08-09T04:29:53.360128","exception":false,"start_time":"2021-08-09T04:29:53.074464","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-06T14:28:50.555533Z","iopub.execute_input":"2022-08-06T14:28:50.555832Z","iopub.status.idle":"2022-08-06T14:28:50.765144Z","shell.execute_reply.started":"2022-08-06T14:28:50.555799Z","shell.execute_reply":"2022-08-06T14:28:50.764134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 명목형 피처, calc 분류 피처를 제외한 피처\nremaining_features = [feature for feature in all_features \n                      if ('cat' not in feature and 'calc' not in feature)] \n# num_missing을 remaining_features에 추가\nremaining_features.append('num_missing')","metadata":{"papermill":{"duration":0.030541,"end_time":"2021-08-09T04:29:53.412767","exception":false,"start_time":"2021-08-09T04:29:53.382226","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-06T14:28:50.766526Z","iopub.execute_input":"2022-08-06T14:28:50.766785Z","iopub.status.idle":"2022-08-06T14:28:50.771925Z","shell.execute_reply.started":"2022-08-06T14:28:50.766757Z","shell.execute_reply":"2022-08-06T14:28:50.771007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 분류가 ind인 피처\nind_features = [feature for feature in all_features if 'ind' in feature]\n\nis_first_feature = True\nfor ind_feature in ind_features:\n    if is_first_feature:\n        all_data['mix_ind'] = all_data[ind_feature].astype(str) + '_'\n        is_first_feature = False\n    else:\n        all_data['mix_ind'] += all_data[ind_feature].astype(str) + '_'","metadata":{"papermill":{"duration":41.802628,"end_time":"2021-08-09T04:30:35.236825","exception":false,"start_time":"2021-08-09T04:29:53.434197","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-06T14:28:50.773193Z","iopub.execute_input":"2022-08-06T14:28:50.773418Z","iopub.status.idle":"2022-08-06T14:29:16.087652Z","shell.execute_reply.started":"2022-08-06T14:28:50.773392Z","shell.execute_reply":"2022-08-06T14:29:16.086721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_count_features = []\nfor feature in cat_features+['mix_ind']:\n    val_counts_dict = all_data[feature].value_counts().to_dict()\n    all_data[f'{feature}_count'] = all_data[feature].apply(lambda x: \n                                                           val_counts_dict[x])\n    cat_count_features.append(f'{feature}_count')","metadata":{"papermill":{"duration":14.823851,"end_time":"2021-08-09T04:30:50.082511","exception":false,"start_time":"2021-08-09T04:30:35.25866","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-06T14:29:16.089461Z","iopub.execute_input":"2022-08-06T14:29:16.090132Z","iopub.status.idle":"2022-08-06T14:29:26.269272Z","shell.execute_reply.started":"2022-08-06T14:29:16.090081Z","shell.execute_reply":"2022-08-06T14:29:26.268112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_count_features","metadata":{"papermill":{"duration":0.035492,"end_time":"2021-08-09T04:30:50.140514","exception":false,"start_time":"2021-08-09T04:30:50.105022","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-06T14:29:26.270398Z","iopub.execute_input":"2022-08-06T14:29:26.270669Z","iopub.status.idle":"2022-08-06T14:29:26.279010Z","shell.execute_reply.started":"2022-08-06T14:29:26.270638Z","shell.execute_reply":"2022-08-06T14:29:26.278323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy import sparse\n\n# 필요 없는 피처들\ndrop_features = ['ps_ind_14', 'ps_ind_10_bin', 'ps_ind_11_bin', \n                 'ps_ind_12_bin', 'ps_ind_13_bin', 'ps_car_14']\n\n# remaining_features, cat_count_features에서 drop_features를 제거한 데이터\nall_data_remaining = all_data[remaining_features+cat_count_features].drop(drop_features, axis=1)\n\n# 데이터 합치기\nall_data_sprs = sparse.hstack([sparse.csr_matrix(all_data_remaining),\n                               encoded_cat_matrix],\n                              format='csr')","metadata":{"papermill":{"duration":8.173816,"end_time":"2021-08-09T04:30:58.338156","exception":false,"start_time":"2021-08-09T04:30:50.16434","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-06T14:29:26.280173Z","iopub.execute_input":"2022-08-06T14:29:26.280776Z","iopub.status.idle":"2022-08-06T14:29:32.446137Z","shell.execute_reply.started":"2022-08-06T14:29:26.280738Z","shell.execute_reply":"2022-08-06T14:29:32.445036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_train = len(train) # 훈련 데이터 개수\n\n# 훈련 데이터와 테스트 데이터 나누기\nX = all_data_sprs[:num_train]\nX_test = all_data_sprs[num_train:]\n\ny = train['target'].values","metadata":{"papermill":{"duration":1.02836,"end_time":"2021-08-09T04:30:59.391155","exception":false,"start_time":"2021-08-09T04:30:58.362795","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-06T14:29:32.448757Z","iopub.execute_input":"2022-08-06T14:29:32.449224Z","iopub.status.idle":"2022-08-06T14:29:33.663838Z","shell.execute_reply.started":"2022-08-06T14:29:32.449181Z","shell.execute_reply":"2022-08-06T14:29:33.662903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\n\ndef eval_gini(y_true, y_pred):\n    # 실제값과 예측값의 크기가 같은지 확인 (값이 다르면 오류 발생)\n    assert y_true.shape == y_pred.shape\n\n    n_samples = y_true.shape[0]                      # 데이터 개수\n    L_mid = np.linspace(1 / n_samples, 1, n_samples) # 대각선 값\n\n    # 1) 예측값에 대한 지니계수\n    pred_order = y_true[y_pred.argsort()] # y_pred 크기순으로 y_true 값 정렬\n    L_pred = np.cumsum(pred_order) / np.sum(pred_order) # 로렌츠 곡선\n    G_pred = np.sum(L_mid - L_pred)       # 예측 값에 대한 지니계수\n\n    # 2) 예측이 완벽할 때 지니계수\n    true_order = y_true[y_true.argsort()] # y_true 크기순으로 y_true 값 정렬\n    L_true = np.cumsum(true_order) / np.sum(true_order) # 로렌츠 곡선\n    G_true = np.sum(L_mid - L_true)       # 예측이 완벽할 때 지니계수\n\n    # 정규화된 지니계수\n    return G_pred / G_true","metadata":{"papermill":{"duration":0.032496,"end_time":"2021-08-09T04:30:59.445472","exception":false,"start_time":"2021-08-09T04:30:59.412976","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-06T14:29:33.665094Z","iopub.execute_input":"2022-08-06T14:29:33.665368Z","iopub.status.idle":"2022-08-06T14:29:33.673203Z","shell.execute_reply.started":"2022-08-06T14:29:33.665335Z","shell.execute_reply":"2022-08-06T14:29:33.672122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# LightGBM용 gini() 함수\ndef gini_lgb(preds, dtrain):\n    labels = dtrain.get_label()\n    return 'gini', eval_gini(labels, preds), True","metadata":{"papermill":{"duration":0.029034,"end_time":"2021-08-09T04:30:59.496363","exception":false,"start_time":"2021-08-09T04:30:59.467329","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-06T14:29:33.674517Z","iopub.execute_input":"2022-08-06T14:29:33.674791Z","iopub.status.idle":"2022-08-06T14:29:33.689576Z","shell.execute_reply.started":"2022-08-06T14:29:33.674762Z","shell.execute_reply":"2022-08-06T14:29:33.688754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# XGBoost용 gini() 함수\ndef gini_xgb(preds, dtrain):\n    labels = dtrain.get_label()\n    return 'gini', eval_gini(labels, preds)","metadata":{"papermill":{"duration":0.029908,"end_time":"2021-08-09T04:30:59.548052","exception":false,"start_time":"2021-08-09T04:30:59.518144","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-06T14:29:33.690707Z","iopub.execute_input":"2022-08-06T14:29:33.691490Z","iopub.status.idle":"2022-08-06T14:29:33.707812Z","shell.execute_reply.started":"2022-08-06T14:29:33.691450Z","shell.execute_reply":"2022-08-06T14:29:33.706962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import StratifiedKFold\n\n# 층화 K 폴드 교차 검증기 생성\nfolds = StratifiedKFold(n_splits=5, shuffle=True, random_state=1991)","metadata":{"papermill":{"duration":0.083121,"end_time":"2021-08-09T04:30:59.653108","exception":false,"start_time":"2021-08-09T04:30:59.569987","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-06T14:29:33.709144Z","iopub.execute_input":"2022-08-06T14:29:33.709851Z","iopub.status.idle":"2022-08-06T14:29:33.776863Z","shell.execute_reply.started":"2022-08-06T14:29:33.709812Z","shell.execute_reply":"2022-08-06T14:29:33.776184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"max_params_lgb = {\n    'bagging_fraction': 0.6213108174593661,\n    'feature_fraction': 0.608712929970154,\n    'lambda_l1': 0.7040436794880651,\n    'lambda_l2': 0.9832619845547939,\n    'min_child_samples': 9,\n    'min_child_weight': 36.10036444740457,\n    'num_leaves': 40,\n    'objective': 'binary',\n    'learning_rate': 0.005,\n    'bagging_freq': 1,\n    'force_row_wise': True,\n    'random_state': 1991\n}","metadata":{"papermill":{"duration":0.030792,"end_time":"2021-08-09T04:30:59.706286","exception":false,"start_time":"2021-08-09T04:30:59.675494","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-06T14:29:33.778179Z","iopub.execute_input":"2022-08-06T14:29:33.778688Z","iopub.status.idle":"2022-08-06T14:29:33.783882Z","shell.execute_reply.started":"2022-08-06T14:29:33.778650Z","shell.execute_reply":"2022-08-06T14:29:33.783124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import lightgbm as lgb\n\n# OOF 방식으로 훈련된 모델로 검증 데이터 타깃값을 예측한 확률을 담을 1차원 배열\noof_val_preds_lgb = np.zeros(X.shape[0]) \n# OOF 방식으로 훈련된 모델로 테스트 데이터 타깃값을 예측한 확률을 담을 1차원 배열\noof_test_preds_lgb = np.zeros(X_test.shape[0]) \n\n# OOF 방식으로 모델 훈련, 검증, 예측\nfor idx, (train_idx, valid_idx) in enumerate(folds.split(X, y)):\n    # 각 폴드를 구분하는 문구 출력\n    print('#'*40, f'폴드 {idx+1} / 폴드 {folds.n_splits}', '#'*40)\n    \n    # 훈련용 데이터, 검증용 데이터 설정\n    X_train, y_train = X[train_idx], y[train_idx] # 훈련용 데이터\n    X_valid, y_valid = X[valid_idx], y[valid_idx] # 검증용 데이터\n\n    # LightGBM 전용 데이터셋 생성\n    dtrain = lgb.Dataset(X_train, y_train) # LightGBM 전용 훈련 데이터셋\n    dvalid = lgb.Dataset(X_valid, y_valid) # LightGBM 전용 검증 데이터셋\n                          \n    # LightGBM 모델 훈련\n    lgb_model = lgb.train(params=max_params_lgb,     # 최적 하이퍼파라미터\n                          train_set=dtrain,          # 훈련 데이터셋\n                          num_boost_round=2500,      # 부스팅 반복 횟수\n                          valid_sets=dvalid,         # 성능 평가용 검증 데이터셋\n                          feval=gini_lgb,            # 검증용 평가지표\n                          early_stopping_rounds=300, # 조기종료 조건\n                          verbose_eval=100)          # 100번째마다 점수 출력\n    \n    # 테스트 데이터를 활용해 OOF 예측\n    oof_test_preds_lgb += lgb_model.predict(X_test)/folds.n_splits\n    \n    # 모델 성능 평가를 위한 검증 데이터 타깃값 예측 \n    oof_val_preds_lgb[valid_idx] += lgb_model.predict(X_valid)\n    \n    # 검증 데이터 예측확률에 대한 정규화 지니계수\n    gini_score = eval_gini(y_valid, oof_val_preds_lgb[valid_idx])\n    print(f'폴드 {idx+1} 지니계수 : {gini_score}\\n')","metadata":{"papermill":{"duration":1446.543006,"end_time":"2021-08-09T04:55:06.27113","exception":false,"start_time":"2021-08-09T04:30:59.728124","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-06T14:29:33.785170Z","iopub.execute_input":"2022-08-06T14:29:33.785824Z","iopub.status.idle":"2022-08-06T14:53:13.371732Z","shell.execute_reply.started":"2022-08-06T14:29:33.785790Z","shell.execute_reply":"2022-08-06T14:53:13.370654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"max_params_xgb = {\n    'colsample_bytree': 0.8843124587484356,\n    'gamma': 10.452246227672624,\n    'max_depth': 7,\n    'min_child_weight': 6.494091293383359,\n    'reg_alpha': 8.551838810159788,\n    'reg_lambda': 1.3814765995549108,\n    'scale_pos_weight': 1.423280772455086,\n    'subsample': 0.7001630536555632,\n    'objective': 'binary:logistic',\n    'learning_rate': 0.02,\n    'random_state': 1991\n}","metadata":{"papermill":{"duration":0.072942,"end_time":"2021-08-09T04:55:06.406925","exception":false,"start_time":"2021-08-09T04:55:06.333983","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-06T14:53:13.374320Z","iopub.execute_input":"2022-08-06T14:53:13.374790Z","iopub.status.idle":"2022-08-06T14:53:13.381468Z","shell.execute_reply.started":"2022-08-06T14:53:13.374746Z","shell.execute_reply":"2022-08-06T14:53:13.380408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import xgboost as xgb\n\n# OOF 방식으로 훈련된 모델로 검증 데이터 타깃값을 예측한 확률을 담을 1차원 배열\noof_val_preds_xgb = np.zeros(X.shape[0]) \n# OOF 방식으로 훈련된 모델로 테스트 데이터 타깃값을 예측한 확률을 담을 1차원 배열\noof_test_preds_xgb = np.zeros(X_test.shape[0]) \n\n# OOF 방식으로 모델 훈련, 검증, 예측\nfor idx, (train_idx, valid_idx) in enumerate(folds.split(X, y)):\n    # 각 폴드를 구분하는 문구 출력\n    print('#'*40, f'폴드 {idx+1} / 폴드 {folds.n_splits}', '#'*40)\n    \n    # 훈련용 데이터, 검증용 데이터 설정\n    X_train, y_train = X[train_idx], y[train_idx]\n    X_valid, y_valid = X[valid_idx], y[valid_idx]\n\n    # XGBoost 전용 데이터셋 생성 \n    dtrain = xgb.DMatrix(X_train, y_train)\n    dvalid = xgb.DMatrix(X_valid, y_valid)\n    dtest = xgb.DMatrix(X_test)\n\n    # XGBoost 모델 훈련\n    xgb_model = xgb.train(params=max_params_xgb, \n                          dtrain=dtrain,\n                          num_boost_round=2000,\n                          evals=[(dvalid, 'valid')],\n                          maximize=True,\n                          feval=gini_xgb,\n                          early_stopping_rounds=200,\n                          verbose_eval=100)\n\n    # 모델 성능이 가장 좋을 때의 부스팅 반복 횟수 저장\n    best_iter = xgb_model.best_iteration\n\n    # 테스트 데이터를 활용해 OOF 예측\n    oof_test_preds_xgb += xgb_model.predict(dtest,\n                                            iteration_range=(0, best_iter))/folds.n_splits\n    \n    # 모델 성능 평가를 위한 검증 데이터 타깃값 예측 \n    oof_val_preds_xgb[valid_idx] += xgb_model.predict(dvalid, \n                                                      iteration_range=(0, best_iter))\n    \n    # 검증 데이터 예측확률에 대한 정규화 지니계수\n    gini_score = eval_gini(y_valid, oof_val_preds_xgb[valid_idx])\n    print(f'폴드 {idx+1} 지니계수 : {gini_score}\\n')","metadata":{"papermill":{"duration":6003.519634,"end_time":"2021-08-09T06:35:09.988925","exception":false,"start_time":"2021-08-09T04:55:06.469291","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-06T14:53:13.383323Z","iopub.execute_input":"2022-08-06T14:53:13.383934Z","iopub.status.idle":"2022-08-06T16:32:24.870871Z","shell.execute_reply.started":"2022-08-06T14:53:13.383889Z","shell.execute_reply":"2022-08-06T16:32:24.869655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('LightGBM OOF 검증 데이터 지니계수 :', eval_gini(y, oof_val_preds_lgb))","metadata":{"papermill":{"duration":0.229657,"end_time":"2021-08-09T06:35:10.310934","exception":false,"start_time":"2021-08-09T06:35:10.081277","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-06T16:32:24.873657Z","iopub.execute_input":"2022-08-06T16:32:24.875014Z","iopub.status.idle":"2022-08-06T16:32:24.990921Z","shell.execute_reply.started":"2022-08-06T16:32:24.874926Z","shell.execute_reply":"2022-08-06T16:32:24.990040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('XGBoost OOF 검증 데이터 지니계수 :', eval_gini(y, oof_val_preds_xgb))","metadata":{"papermill":{"duration":0.224694,"end_time":"2021-08-09T06:35:10.62757","exception":false,"start_time":"2021-08-09T06:35:10.402876","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-06T16:32:24.992599Z","iopub.execute_input":"2022-08-06T16:32:24.992844Z","iopub.status.idle":"2022-08-06T16:32:25.107819Z","shell.execute_reply.started":"2022-08-06T16:32:24.992816Z","shell.execute_reply":"2022-08-06T16:32:25.106825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 8.6.1 앙상블 수행","metadata":{"papermill":{"duration":0.093207,"end_time":"2021-08-09T06:35:10.814218","exception":false,"start_time":"2021-08-09T06:35:10.721011","status":"completed"},"tags":[]}},{"cell_type":"code","source":"oof_test_preds = oof_test_preds_lgb * 0.5 + oof_test_preds_xgb * 0.5","metadata":{"papermill":{"duration":0.109942,"end_time":"2021-08-09T06:35:11.01693","exception":false,"start_time":"2021-08-09T06:35:10.906988","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-06T16:32:25.109420Z","iopub.execute_input":"2022-08-06T16:32:25.109649Z","iopub.status.idle":"2022-08-06T16:32:25.118132Z","shell.execute_reply.started":"2022-08-06T16:32:25.109623Z","shell.execute_reply":"2022-08-06T16:32:25.117115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 8.6.2 예측 및 결과 제출","metadata":{"papermill":{"duration":0.093316,"end_time":"2021-08-09T06:35:11.202147","exception":false,"start_time":"2021-08-09T06:35:11.108831","status":"completed"},"tags":[]}},{"cell_type":"code","source":"submission['target'] = oof_test_preds\nsubmission.to_csv('submission.csv')","metadata":{"papermill":{"duration":3.838334,"end_time":"2021-08-09T06:35:15.132398","exception":false,"start_time":"2021-08-09T06:35:11.294064","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-06T16:32:25.119814Z","iopub.execute_input":"2022-08-06T16:32:25.120265Z","iopub.status.idle":"2022-08-06T16:32:27.524103Z","shell.execute_reply.started":"2022-08-06T16:32:25.120027Z","shell.execute_reply":"2022-08-06T16:32:27.523020Z"},"trusted":true},"execution_count":null,"outputs":[]}]}