{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-06T08:06:50.674186Z","iopub.execute_input":"2022-08-06T08:06:50.675145Z","iopub.status.idle":"2022-08-06T08:06:50.721553Z","shell.execute_reply.started":"2022-08-06T08:06:50.675027Z","shell.execute_reply":"2022-08-06T08:06:50.720257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\ndata_path = '/kaggle/input/porto-seguro-safe-driver-prediction/'\n\ntrain = pd.read_csv(data_path + 'train.csv', index_col = 'id')\ntest = pd.read_csv(data_path + 'test.csv', index_col = 'id')\nsubmission = pd.read_csv(data_path + 'sample_submission.csv', index_col = 'id')","metadata":{"execution":{"iopub.status.busy":"2022-08-06T08:07:55.355847Z","iopub.execute_input":"2022-08-06T08:07:55.357064Z","iopub.status.idle":"2022-08-06T08:08:05.524420Z","shell.execute_reply.started":"2022-08-06T08:07:55.357008Z","shell.execute_reply":"2022-08-06T08:08:05.523055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 8.3.1 피처 엔지니어링","metadata":{}},{"cell_type":"markdown","source":"데이터 합치기","metadata":{}},{"cell_type":"code","source":"all_data = pd.concat([train, test], ignore_index=True)\nall_data = all_data.drop('target', axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T08:10:01.123402Z","iopub.execute_input":"2022-08-06T08:10:01.123830Z","iopub.status.idle":"2022-08-06T08:10:02.614756Z","shell.execute_reply.started":"2022-08-06T08:10:01.123798Z","shell.execute_reply":"2022-08-06T08:10:02.613248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_data","metadata":{"execution":{"iopub.status.busy":"2022-08-06T08:10:58.372270Z","iopub.execute_input":"2022-08-06T08:10:58.372737Z","iopub.status.idle":"2022-08-06T08:10:58.754628Z","shell.execute_reply.started":"2022-08-06T08:10:58.372700Z","shell.execute_reply":"2022-08-06T08:10:58.753409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_features = all_data.columns\nall_features","metadata":{"execution":{"iopub.status.busy":"2022-08-06T08:11:13.069670Z","iopub.execute_input":"2022-08-06T08:11:13.070104Z","iopub.status.idle":"2022-08-06T08:11:13.078997Z","shell.execute_reply.started":"2022-08-06T08:11:13.070070Z","shell.execute_reply":"2022-08-06T08:11:13.077764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"명목형 피처 원-핫 인코딩","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import OneHotEncoder\n\ncat_features = [feature for feature in all_features if 'cat' in feature]\ncat_features","metadata":{"execution":{"iopub.status.busy":"2022-08-06T08:12:22.043815Z","iopub.execute_input":"2022-08-06T08:12:22.044703Z","iopub.status.idle":"2022-08-06T08:12:22.097314Z","shell.execute_reply.started":"2022-08-06T08:12:22.044652Z","shell.execute_reply":"2022-08-06T08:12:22.096260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"onehot_encoder = OneHotEncoder()\n\nencoded_cat_matrix = onehot_encoder.fit_transform(all_data[cat_features])\n\nencoded_cat_matrix","metadata":{"execution":{"iopub.status.busy":"2022-08-06T08:12:53.765399Z","iopub.execute_input":"2022-08-06T08:12:53.765934Z","iopub.status.idle":"2022-08-06T08:12:56.286944Z","shell.execute_reply.started":"2022-08-06T08:12:53.765884Z","shell.execute_reply":"2022-08-06T08:12:56.285580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"필요 없는 피처 제거","metadata":{}},{"cell_type":"code","source":"drop_features = ['ps_ind_14', 'ps_ind_10_bin', 'ps_ind_11_bin',\n                 'ps_ind_12_bin', 'ps_ind_13_bin', 'ps_car_14']\n\nremaining_features = [feature for feature in all_features if ('cat' not in feature and \n                                                              'calc' not in feature and\n                                                              feature not in drop_features)]","metadata":{"execution":{"iopub.status.busy":"2022-08-06T08:15:45.199027Z","iopub.execute_input":"2022-08-06T08:15:45.199594Z","iopub.status.idle":"2022-08-06T08:15:45.206829Z","shell.execute_reply.started":"2022-08-06T08:15:45.199547Z","shell.execute_reply":"2022-08-06T08:15:45.205461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy import sparse\n\nall_data_sprs = sparse.hstack([sparse.csr_matrix(all_data[remaining_features]), encoded_cat_matrix], format='csr')","metadata":{"execution":{"iopub.status.busy":"2022-08-06T08:16:38.789087Z","iopub.execute_input":"2022-08-06T08:16:38.790774Z","iopub.status.idle":"2022-08-06T08:16:41.543975Z","shell.execute_reply.started":"2022-08-06T08:16:38.790690Z","shell.execute_reply":"2022-08-06T08:16:41.542387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"데이터 나누기","metadata":{}},{"cell_type":"code","source":"num_train = len(train)\n\nX = all_data_sprs[:num_train]\nX_test = all_data_sprs[num_train:]\n\ny = train['target'].values","metadata":{"execution":{"iopub.status.busy":"2022-08-06T08:17:54.187491Z","iopub.execute_input":"2022-08-06T08:17:54.188020Z","iopub.status.idle":"2022-08-06T08:17:55.176797Z","shell.execute_reply.started":"2022-08-06T08:17:54.187948Z","shell.execute_reply":"2022-08-06T08:17:55.175507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 8.3.2 평가지표 계산 함수 작성","metadata":{}},{"cell_type":"code","source":"import numpy as np\n\ndef eval_gini(y_true, y_pred):\n    assert y_true.shape == y_pred.shape\n    \n    n_samples = y_true.shape[0]\n    L_mid = np.linspace(1 / n_samples, 1, n_samples)\n    \n    # 1) 예측값에 대한 지니계수\n    pred_order = y_true[y_pred.argsort()]\n    L_pred = np.cumsum(pred_order) / np.sum(pred_order)\n    G_pred = np.sum(L_mid - L_pred)\n    \n    # 2) 예측이 완벽할 때 지니계수\n    true_order = y_true[y_true.argsort()]\n    L_true = np.cumsum(true_order) / np.sum(true_order)\n    G_true = np.sum(L_mid - L_true)\n    \n    # 정규화된 지니계수\n    return G_pred / G_true","metadata":{"execution":{"iopub.status.busy":"2022-08-06T08:42:13.020182Z","iopub.execute_input":"2022-08-06T08:42:13.020615Z","iopub.status.idle":"2022-08-06T08:42:13.029467Z","shell.execute_reply.started":"2022-08-06T08:42:13.020582Z","shell.execute_reply":"2022-08-06T08:42:13.028035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# LightGBM용 gini() 함수\ndef gini(preds, dtrain):\n    labels = dtrain.get_label()\n    return 'gini', eval_gini(labels, preds), True","metadata":{"execution":{"iopub.status.busy":"2022-08-06T08:42:22.361809Z","iopub.execute_input":"2022-08-06T08:42:22.362283Z","iopub.status.idle":"2022-08-06T08:42:22.368517Z","shell.execute_reply.started":"2022-08-06T08:42:22.362243Z","shell.execute_reply":"2022-08-06T08:42:22.367082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 8.3.3 모델 훈련 및 성능 검증","metadata":{}},{"cell_type":"markdown","source":"OOF 예측 방식\n- Out of Fold prediction 이란 K 폴드 교차 검증을 수행하면서 각 폴드마다 테스트 데이터로 예측하는 방식","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import StratifiedKFold\n\nfolds = StratifiedKFold(n_splits=5, shuffle=True, random_state=1991)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T08:42:22.899725Z","iopub.execute_input":"2022-08-06T08:42:22.901085Z","iopub.status.idle":"2022-08-06T08:42:22.908345Z","shell.execute_reply.started":"2022-08-06T08:42:22.901014Z","shell.execute_reply":"2022-08-06T08:42:22.906250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"params = {'objective' : 'binary',\n          'learning_rate' : 0.01,\n          'force_row_wise' : True,\n          'random_state': 0}","metadata":{"execution":{"iopub.status.busy":"2022-08-06T08:42:23.143904Z","iopub.execute_input":"2022-08-06T08:42:23.144705Z","iopub.status.idle":"2022-08-06T08:42:23.150530Z","shell.execute_reply.started":"2022-08-06T08:42:23.144660Z","shell.execute_reply":"2022-08-06T08:42:23.149205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oof_val_preds = np.zeros(X.shape[0])\n\noof_test_preds = np.zeros(X_test.shape[0])","metadata":{"execution":{"iopub.status.busy":"2022-08-06T08:42:23.336714Z","iopub.execute_input":"2022-08-06T08:42:23.337202Z","iopub.status.idle":"2022-08-06T08:42:23.345786Z","shell.execute_reply.started":"2022-08-06T08:42:23.337155Z","shell.execute_reply":"2022-08-06T08:42:23.344634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import lightgbm as lgb\n\nfor idx, (train_idx, valid_idx) in enumerate(folds.split(X, y)):\n    # 각 포드를 구분하는 문구 출력\n    print('#' * 40, f'폴드 {idx+1} / 폴드 {folds.n_splits}', '#' * 40)\n    \n    # 훈련용 데이터, 검증용 데이터 설정\n    X_train, y_train = X[train_idx], y[train_idx]\n    X_valid, y_valid = X[valid_idx], y[valid_idx]\n    \n    dtrain = lgb.Dataset(X_train, y_train)\n    dvalid = lgb.Dataset(X_valid, y_valid)\n    \n    lgb_model = lgb.train(params = params,\n                          train_set = dtrain,\n                          num_boost_round = 1000,# 부스팅 반복 횟수\n                          valid_sets = dvalid, # 성능 평가용 검증 데이터셋\n                          feval = gini, # 검증용 평가 지표\n                          early_stopping_rounds = 100, # 조기종료 조건\n                          verbose_eval = 100) # 100 번째마다 점수 출력\n    \n    oof_test_preds += lgb_model.predict(X_test)/folds.n_splits\n    oof_val_preds[valid_idx] += lgb_model.predict(X_valid)\n    \n    gini_score = eval_gini(y_valid, oof_val_preds[valid_idx])\n    print(f'폴드 {idx+1} 지니계수: {gini_score}\\n')","metadata":{"execution":{"iopub.status.busy":"2022-08-06T08:42:23.731911Z","iopub.execute_input":"2022-08-06T08:42:23.733147Z","iopub.status.idle":"2022-08-06T08:48:05.793835Z","shell.execute_reply.started":"2022-08-06T08:42:23.733098Z","shell.execute_reply":"2022-08-06T08:48:05.792148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('OOF 검증 데이터 지니계수 :', eval_gini(y, oof_val_preds))","metadata":{"execution":{"iopub.status.busy":"2022-08-06T08:48:36.701610Z","iopub.execute_input":"2022-08-06T08:48:36.702137Z","iopub.status.idle":"2022-08-06T08:48:36.824856Z","shell.execute_reply.started":"2022-08-06T08:48:36.702094Z","shell.execute_reply":"2022-08-06T08:48:36.823209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 8.3.4 예측 및 결과 제출","metadata":{}},{"cell_type":"code","source":"submission['target'] = oof_test_preds\nsubmission.to_csv('submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-06T08:49:01.875020Z","iopub.execute_input":"2022-08-06T08:49:01.875479Z","iopub.status.idle":"2022-08-06T08:49:04.264644Z","shell.execute_reply.started":"2022-08-06T08:49:01.875444Z","shell.execute_reply":"2022-08-06T08:49:04.263264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.read_csv('submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-06T08:49:10.852891Z","iopub.execute_input":"2022-08-06T08:49:10.853346Z","iopub.status.idle":"2022-08-06T08:49:11.081153Z","shell.execute_reply.started":"2022-08-06T08:49:10.853310Z","shell.execute_reply":"2022-08-06T08:49:11.079729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}