{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\n\ndata_path = '/kaggle/input/porto-seguro-safe-driver-prediction/'\ntrain = pd.read_csv(data_path + 'train.csv', index_col = 'id')\ntest = pd.read_csv(data_path + 'test.csv', index_col = 'id')\nsubmission = pd.read_csv(data_path + 'sample_submission.csv', index_col = 'id')\n\nprint(train.shape, test.shape)\ndisplay(train.head(5))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-04T15:30:15.105200Z","iopub.execute_input":"2022-08-04T15:30:15.105760Z","iopub.status.idle":"2022-08-04T15:30:26.932402Z","shell.execute_reply.started":"2022-08-04T15:30:15.105644Z","shell.execute_reply":"2022-08-04T15:30:26.931390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T15:30:26.934433Z","iopub.execute_input":"2022-08-04T15:30:26.935124Z","iopub.status.idle":"2022-08-04T15:30:27.035758Z","shell.execute_reply.started":"2022-08-04T15:30:26.935084Z","shell.execute_reply":"2022-08-04T15:30:27.034373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 결측값 찾아보기 : bar(), missingno","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport missingno as msno\nimport matplotlib.pyplot  as plt\n\ntrain_copy = train.copy().replace(-1, np.NaN)\nmsno.bar(df = train_copy.iloc[:, 1:29], figsize = (13, 6))\nplt.show()\nmsno.bar(df = train_copy.iloc[:, 29:], figsize = (13, 6))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T15:30:27.037719Z","iopub.execute_input":"2022-08-04T15:30:27.038269Z","iopub.status.idle":"2022-08-04T15:30:33.608098Z","shell.execute_reply.started":"2022-08-04T15:30:27.038230Z","shell.execute_reply":"2022-08-04T15:30:33.606880Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"msno.matrix(df = train_copy.iloc[:, 1:29], figsize = (13, 6))\nplt.show()\nmsno.matrix(df = train_copy.iloc[:, 29:], figsize = (13, 6))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T15:30:33.611077Z","iopub.execute_input":"2022-08-04T15:30:33.611499Z","iopub.status.idle":"2022-08-04T15:30:45.805546Z","shell.execute_reply.started":"2022-08-04T15:30:33.611460Z","shell.execute_reply":"2022-08-04T15:30:45.804065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.columns","metadata":{"execution":{"iopub.status.busy":"2022-08-04T15:30:45.807241Z","iopub.execute_input":"2022-08-04T15:30:45.807684Z","iopub.status.idle":"2022-08-04T15:30:45.816385Z","shell.execute_reply.started":"2022-08-04T15:30:45.807642Z","shell.execute_reply":"2022-08-04T15:30:45.815284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def resumetable(df):\n    print(f'데이터 세트 형상: {df.shape}')\n    summary = pd.DataFrame(df.dtypes, columns=['데이터 타입'])\n    summary['결측값 개수'] = (df==1).sum().values\n    summary['고윳값 개수'] = df.nunique().values\n    summary['데이터 종류'] = None\n    for col in df.columns:\n        if 'bin' in col or col =='target':\n            summary.loc[col, '데이터 종류'] = '이진형'\n        elif 'cat' in col or col =='target':\n            summary.loc[col, '데이터 종류'] = '명목형'\n        elif df[col].dtype == float:\n            summary.loc[col, '데이터 종류'] = '연속형'\n        elif df[col].dtype == int:\n            summary.loc[col, '데이터 종류'] = '순서형'\n            \n    \n    return summary\n\nresumetable(train)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T15:30:45.818330Z","iopub.execute_input":"2022-08-04T15:30:45.819215Z","iopub.status.idle":"2022-08-04T15:30:46.197840Z","shell.execute_reply.started":"2022-08-04T15:30:45.819169Z","shell.execute_reply":"2022-08-04T15:30:46.196473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"summary = resumetable(train)\ndisplay(summary[summary[\"데이터 종류\"]=='이진형'].index)\ndisplay(summary[summary[\"데이터 종류\"]=='명목형'].index)\ndisplay(summary[summary[\"데이터 종류\"]=='연속형'].index)\ndisplay(summary[summary[\"데이터 종류\"]=='순서형'].index)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T15:30:46.199998Z","iopub.execute_input":"2022-08-04T15:30:46.200518Z","iopub.status.idle":"2022-08-04T15:30:46.570904Z","shell.execute_reply.started":"2022-08-04T15:30:46.200469Z","shell.execute_reply":"2022-08-04T15:30:46.569791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 데이터 시각화","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib as mpl\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2022-08-04T15:30:46.572730Z","iopub.execute_input":"2022-08-04T15:30:46.573670Z","iopub.status.idle":"2022-08-04T15:30:46.579579Z","shell.execute_reply.started":"2022-08-04T15:30:46.573629Z","shell.execute_reply":"2022-08-04T15:30:46.578288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def write_percent(ax, total_size):\n    '''도형 객체를 순회하며 막대 상단에 타깃값 비율 표시'''\n    for patch in ax.patches:\n        height = patch.get_height()     # 도형 높이(데이터 개수)\n        width = patch.get_width()       # 도형 너비\n        left_coord = patch.get_x()      # 도형 왼쪽 테두리의 x축 위치\n        percent = height/total_size*100 # 타깃값 비율\n        \n        # (x, y) 좌표에 텍스트 입력 \n        ax.text(x=left_coord + width/2.0,    # x축 위치\n                y=height + total_size*0.001, # y축 위치\n                s=f'{percent:1.1f}%',        # 입력 텍스트\n                ha='center')                 # 가운데 정렬\n\nmpl.rc('font',size = 15)\nplt.figure(figsize=(7, 6))\n\nax = sns.countplot(x='target', data=train)\nwrite_percent(ax, len(train)) # 비율 표시\nax.set_title('Target Distribution');\n","metadata":{"execution":{"iopub.status.busy":"2022-08-04T15:30:46.581334Z","iopub.execute_input":"2022-08-04T15:30:46.581826Z","iopub.status.idle":"2022-08-04T15:30:46.883447Z","shell.execute_reply.started":"2022-08-04T15:30:46.581776Z","shell.execute_reply":"2022-08-04T15:30:46.882468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.gridspec as gridspec\n\ndef plot_target_ratio_by_features(df, features, \n                                  num_rows, num_cols, size=(12, 18)):\n    mpl.rc('font', size = 9)\n    grid = gridspec.GridSpec(num_rows, num_cols)\n    plt.subplots_adjust(wspace = 0.3, hspace = 0.3)\n    \n    for idx, feature in enumerate(features):\n        ax = plt.subplot(grid[idx])\n        sns.barplot(x = feature, y='target', data = df, palette = 'Set2', ax = ax)\n        \nbin_features = summary[summary['데이터 종류']=='이진형'].index\nplot_target_ratio_by_features(train, bin_features, 6, 3)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T15:30:46.887971Z","iopub.execute_input":"2022-08-04T15:30:46.888690Z","iopub.status.idle":"2022-08-04T15:35:27.251174Z","shell.execute_reply.started":"2022-08-04T15:30:46.888645Z","shell.execute_reply":"2022-08-04T15:35:27.249740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nom_features = summary[summary['데이터 종류']=='명목형'].index\nplot_target_ratio_by_features(train, nom_features, 7, 2)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T15:35:27.253033Z","iopub.execute_input":"2022-08-04T15:35:27.253473Z","iopub.status.idle":"2022-08-04T15:38:40.677573Z","shell.execute_reply.started":"2022-08-04T15:35:27.253432Z","shell.execute_reply":"2022-08-04T15:38:40.676118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ord_features = summary[summary['데이터 종류']=='순서형'].index\nplot_target_ratio_by_features(train, ord_features, 8, 2, (12, 20))","metadata":{"execution":{"iopub.status.busy":"2022-08-04T15:38:40.679822Z","iopub.execute_input":"2022-08-04T15:38:40.680585Z","iopub.status.idle":"2022-08-04T15:41:17.667308Z","shell.execute_reply.started":"2022-08-04T15:38:40.680534Z","shell.execute_reply":"2022-08-04T15:41:17.665870Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(pd.cut([1.0, 1.5, 2.1, 2.7, 3.5, 4.0], 3))\ncont_features = summary[summary['데이터 종류']=='연속형'].index\nplt.figure(figsize = (12, 16))\ngrid = gridspec.GridSpec(5, 2)\nplt.subplots_adjust(wspace = 0.2, hspace = 0.4)\n\nfor idx, cont_feature in enumerate(cont_features):\n    train[cont_feature] = pd.cut(train[cont_feature], 5)\n    ax = plt.subplot(grid[idx])\n    sns.barplot(x = cont_feature, y = 'target', data = train, palette = 'Set2', ax = ax)\n    ax.tick_params(axis = 'x', labelrotation = 10)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-04T15:41:17.669322Z","iopub.execute_input":"2022-08-04T15:41:17.670819Z","iopub.status.idle":"2022-08-04T15:43:16.557359Z","shell.execute_reply.started":"2022-08-04T15:41:17.670742Z","shell.execute_reply":"2022-08-04T15:43:16.556268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_copy.shape#(595212, 58)\ntrain_copy = train_copy.dropna()\ntrain_copy.shape#(124931, 58)\n\nplt.figure(figsize = (10, 8))\ncont_corr = train_copy[cont_features].corr()\nsns.heatmap(cont_corr , annot=True, cmap = 'OrRd')\n\n#ps_car_14 제거, ","metadata":{"execution":{"iopub.status.busy":"2022-08-04T15:43:16.558682Z","iopub.execute_input":"2022-08-04T15:43:16.559127Z","iopub.status.idle":"2022-08-04T15:43:17.925555Z","shell.execute_reply.started":"2022-08-04T15:43:16.559086Z","shell.execute_reply":"2022-08-04T15:43:17.923962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\n# 데이터 경로\ndata_path = '/kaggle/input/porto-seguro-safe-driver-prediction/'\n\ntrain = pd.read_csv(data_path + 'train.csv', index_col='id')\ntest = pd.read_csv(data_path + 'test.csv', index_col='id')\nsubmission = pd.read_csv(data_path + 'sample_submission.csv', index_col='id')","metadata":{"execution":{"iopub.status.busy":"2022-08-04T16:52:10.778856Z","iopub.execute_input":"2022-08-04T16:52:10.780305Z","iopub.status.idle":"2022-08-04T16:52:20.928790Z","shell.execute_reply.started":"2022-08-04T16:52:10.780174Z","shell.execute_reply":"2022-08-04T16:52:20.927589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_data = pd.concat([train, test], ignore_index=True)\nall_data = all_data.drop('target', axis=1) # 타깃값 제거\nall_features = all_data.columns # 전체 피처\nall_features","metadata":{"execution":{"iopub.status.busy":"2022-08-04T16:52:56.632254Z","iopub.execute_input":"2022-08-04T16:52:56.633070Z","iopub.status.idle":"2022-08-04T16:52:58.703597Z","shell.execute_reply.started":"2022-08-04T16:52:56.633021Z","shell.execute_reply":"2022-08-04T16:52:58.702735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#명목형 피처 원핫인코딩\nfrom sklearn.preprocessing import OneHotEncoder\n\n# 명목형 피처 추출\ncat_features = [feature for feature in all_features if 'cat' in feature] \n\nonehot_encoder = OneHotEncoder() # 원-핫 인코더 객체 생성\n# 인코딩\nencoded_cat_matrix = onehot_encoder.fit_transform(all_data[cat_features]) \n\n# 제거\n# 추가로 제거할 피처\ndrop_features = ['ps_ind_14', 'ps_ind_10_bin', 'ps_ind_11_bin', \n                 'ps_ind_12_bin', 'ps_ind_13_bin', 'ps_car_14']\n\n# '1) 명목형 피처, 2) calc 분류의 피처, 3) 추가 제거할 피처'를 제외한 피처\nremaining_features = [feature for feature in all_features \n                      if ('cat' not in feature and \n                          'calc' not in feature and \n                          feature not in drop_features)]\n\n\nencoded_cat_matrix","metadata":{"execution":{"iopub.status.busy":"2022-08-04T16:53:44.508644Z","iopub.execute_input":"2022-08-04T16:53:44.509859Z","iopub.status.idle":"2022-08-04T16:53:47.321772Z","shell.execute_reply.started":"2022-08-04T16:53:44.509814Z","shell.execute_reply":"2022-08-04T16:53:47.320427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy import sparse\n\nall_data_sprs = sparse.hstack([sparse.csr_matrix(all_data[remaining_features]),\n                               encoded_cat_matrix],\n                              format='csr')\nall_data_sprs","metadata":{"execution":{"iopub.status.busy":"2022-08-04T17:01:37.743954Z","iopub.execute_input":"2022-08-04T17:01:37.744448Z","iopub.status.idle":"2022-08-04T17:01:40.363755Z","shell.execute_reply.started":"2022-08-04T17:01:37.744408Z","shell.execute_reply":"2022-08-04T17:01:40.362447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_train = len(train) # 훈련 데이터 개수\n\n# 훈련 데이터와 테스트 데이터 나누기\nX = all_data_sprs[:num_train]\nX_test = all_data_sprs[num_train:]\n\ny = train['target'].values\ny","metadata":{"execution":{"iopub.status.busy":"2022-08-04T17:01:43.633376Z","iopub.execute_input":"2022-08-04T17:01:43.633801Z","iopub.status.idle":"2022-08-04T17:01:44.406445Z","shell.execute_reply.started":"2022-08-04T17:01:43.633762Z","shell.execute_reply":"2022-08-04T17:01:44.405216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 8.3.2 평가지표 계산 함수 작성\nimport numpy as np\n\ndef eval_gini(y_true, y_pred):\n    # 실제값과 예측값의 크기가 같은지 확인 (값이 다르면 오류 발생)\n    assert y_true.shape == y_pred.shape\n\n    n_samples = y_true.shape[0]                      # 데이터 개수\n    L_mid = np.linspace(1 / n_samples, 1, n_samples) # 대각선 값\n\n    # 1) 예측값에 대한 지니계수\n    pred_order = y_true[y_pred.argsort()] # y_pred 크기순으로 y_true 값 정렬\n    L_pred = np.cumsum(pred_order) / np.sum(pred_order) # 로렌츠 곡선\n    G_pred = np.sum(L_mid - L_pred)       # 예측 값에 대한 지니계수\n\n    # 2) 예측이 완벽할 때 지니계수\n    true_order = y_true[y_true.argsort()] # y_true 크기순으로 y_true 값 정렬\n    L_true = np.cumsum(true_order) / np.sum(true_order) # 로렌츠 곡선\n    G_true = np.sum(L_mid - L_true)       # 예측이 완벽할 때 지니계수\n\n    # 정규화된 지니계수\n    return G_pred / G_true\n\ndef gini(preds, dtrain):\n    labels = dtrain.get_label()\n    return 'gini', eval_gini(labels, preds), True # 반환값","metadata":{"execution":{"iopub.status.busy":"2022-08-04T17:01:47.507308Z","iopub.execute_input":"2022-08-04T17:01:47.507771Z","iopub.status.idle":"2022-08-04T17:01:47.517543Z","shell.execute_reply.started":"2022-08-04T17:01:47.507732Z","shell.execute_reply":"2022-08-04T17:01:47.516330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 8.3.3 모델 훈련 및 성능 검증\nfrom sklearn.model_selection import StratifiedKFold\n\n# 층화 K 폴드 교차 검증기\nfolds = StratifiedKFold(n_splits=5, shuffle=True, random_state=1991)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T17:01:49.135726Z","iopub.execute_input":"2022-08-04T17:01:49.136145Z","iopub.status.idle":"2022-08-04T17:01:49.203821Z","shell.execute_reply.started":"2022-08-04T17:01:49.136110Z","shell.execute_reply":"2022-08-04T17:01:49.202475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"folds","metadata":{"execution":{"iopub.status.busy":"2022-08-04T17:01:57.936869Z","iopub.execute_input":"2022-08-04T17:01:57.937312Z","iopub.status.idle":"2022-08-04T17:01:57.944631Z","shell.execute_reply.started":"2022-08-04T17:01:57.937273Z","shell.execute_reply":"2022-08-04T17:01:57.943338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"params = {'objective': 'binary',\n          'learning_rate': 0.01,\n          'force_row_wise': True,\n          'random_state': 0}","metadata":{"execution":{"iopub.status.busy":"2022-08-04T17:02:01.289789Z","iopub.execute_input":"2022-08-04T17:02:01.290924Z","iopub.status.idle":"2022-08-04T17:02:01.295997Z","shell.execute_reply.started":"2022-08-04T17:02:01.290879Z","shell.execute_reply":"2022-08-04T17:02:01.295029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# OOF 방식으로 훈련된 모델로 검증 데이터 타깃값을 예측한 확률을 담을 1차원 배열\noof_val_preds = np.zeros(X.shape[0]) \n# OOF 방식으로 훈련된 모델로 테스트 데이터 타깃값을 예측한 확률을 담을 1차원 배열\noof_test_preds = np.zeros(X_test.shape[0]) ","metadata":{"execution":{"iopub.status.busy":"2022-08-04T17:02:01.897481Z","iopub.execute_input":"2022-08-04T17:02:01.898689Z","iopub.status.idle":"2022-08-04T17:02:01.905373Z","shell.execute_reply.started":"2022-08-04T17:02:01.898625Z","shell.execute_reply":"2022-08-04T17:02:01.904378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import lightgbm as lgb\n\n# OOF 방식으로 모델 훈련, 검증, 예측\nfor idx, (train_idx, valid_idx) in enumerate(folds.split(X, y)):\n    # 각 폴드를 구분하는 문구 출력\n    print('#'*40, f'폴드 {idx+1} / 폴드 {folds.n_splits}', '#'*40)\n    \n    # 훈련용 데이터, 검증용 데이터 설정 \n    X_train, y_train = X[train_idx], y[train_idx] # 훈련용 데이터\n    X_valid, y_valid = X[valid_idx], y[valid_idx] # 검증용 데이터\n\n    # LightGBM 전용 데이터셋 생성 \n    dtrain = lgb.Dataset(X_train, y_train) # LightGBM 전용 훈련 데이터셋\n    dvalid = lgb.Dataset(X_valid, y_valid) # LightGBM 전용 검증 데이터셋\n\n    # LightGBM 모델 훈련 \n    lgb_model = lgb.train(params=params,        # 훈련용 하이퍼파라미터\n                          train_set=dtrain,     # 훈련 데이터셋\n                          num_boost_round=1000, # 부스팅 반복 횟수\n                          valid_sets=dvalid,    # 성능 평가용 검증 데이터셋\n                          feval=gini,           # 검증용 평가지표\n                          early_stopping_rounds=100, # 조기종료 조건\n                          verbose_eval=100)     # 100번째마다 점수 출력\n    \n    # 테스트 데이터를 활용해 OOF 예측\n    oof_test_preds += lgb_model.predict(X_test)/folds.n_splits\n    \n    # 모델 성능 평가를 위한 검증 데이터 타깃값 예측\n    oof_val_preds[valid_idx] += lgb_model.predict(X_valid)\n    \n    # 검증 데이터 예측 확률에 대한 정규화 지니계수 \n    gini_score = eval_gini(y_valid, oof_val_preds[valid_idx])\n    print(f'폴드 {idx+1} 지니계수 : {gini_score}\\n')","metadata":{"execution":{"iopub.status.busy":"2022-08-04T17:02:05.666147Z","iopub.execute_input":"2022-08-04T17:02:05.666845Z","iopub.status.idle":"2022-08-04T17:07:47.961441Z","shell.execute_reply.started":"2022-08-04T17:02:05.666805Z","shell.execute_reply":"2022-08-04T17:07:47.960264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('OOF 검증 데이터 지니계수:', eval_gini(y, oof_val_preds))\nsubmission['target'] = oof_test_preds\nsubmission.to_csv('submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-04T17:10:50.147331Z","iopub.execute_input":"2022-08-04T17:10:50.147859Z","iopub.status.idle":"2022-08-04T17:10:52.636181Z","shell.execute_reply.started":"2022-08-04T17:10:50.147815Z","shell.execute_reply":"2022-08-04T17:10:52.634930Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"submission","metadata":{"execution":{"iopub.status.busy":"2022-08-04T17:11:52.091504Z","iopub.execute_input":"2022-08-04T17:11:52.092017Z","iopub.status.idle":"2022-08-04T17:11:52.110445Z","shell.execute_reply.started":"2022-08-04T17:11:52.091978Z","shell.execute_reply":"2022-08-04T17:11:52.109321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}