{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n        \n\ndata_path = '/kaggle/input/porto-seguro-safe-driver-prediction/'\n\ntrain = pd.read_csv(data_path + 'train.csv', index_col = 'id')\ntest = pd.read_csv(data_path + 'test.csv', index_col = 'id')\nsubmission = pd.read_csv(data_path + 'sample_submission.csv', index_col = 'id')\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-03T15:30:09.540586Z","iopub.execute_input":"2022-08-03T15:30:09.541067Z","iopub.status.idle":"2022-08-03T15:30:17.730485Z","shell.execute_reply.started":"2022-08-03T15:30:09.541033Z","shell.execute_reply":"2022-08-03T15:30:17.729528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 베이스라인 모델","metadata":{}},{"cell_type":"code","source":"# LightGB 사용\n# 훈련과 예측 동시에 ( OOF 예측 방식 때문)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:30:17.732245Z","iopub.execute_input":"2022-08-03T15:30:17.732794Z","iopub.status.idle":"2022-08-03T15:30:17.736964Z","shell.execute_reply.started":"2022-08-03T15:30:17.732759Z","shell.execute_reply":"2022-08-03T15:30:17.735972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 피처 엔지니어링","metadata":{}},{"cell_type":"code","source":"all_data = pd.concat([train, test], ignore_index = True)\nall_data = all_data.drop('target', axis = 1)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:30:17.738618Z","iopub.execute_input":"2022-08-03T15:30:17.739036Z","iopub.status.idle":"2022-08-03T15:30:19.037807Z","shell.execute_reply.started":"2022-08-03T15:30:17.739001Z","shell.execute_reply":"2022-08-03T15:30:19.036469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 합친 데이터에 있는 모든 변수들을 all_features에 저장\nall_features = all_data.columns\nall_features","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:30:19.040898Z","iopub.execute_input":"2022-08-03T15:30:19.041298Z","iopub.status.idle":"2022-08-03T15:30:19.050230Z","shell.execute_reply.started":"2022-08-03T15:30:19.041264Z","shell.execute_reply":"2022-08-03T15:30:19.048933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"명목형 피처 원핫 인코딩\n\n명목형은 제거할 피처는 없었던 것으로 기억","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import OneHotEncoder\n\ncat_features = [feature for feature in all_features if 'cat' in feature]\n\nonehot_encoder = OneHotEncoder()\n\nencoded_cat_matrix = onehot_encoder.fit_transform(all_data[cat_features])\n\nencoded_cat_matrix","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:30:19.051780Z","iopub.execute_input":"2022-08-03T15:30:19.052658Z","iopub.status.idle":"2022-08-03T15:30:21.374941Z","shell.execute_reply.started":"2022-08-03T15:30:19.052620Z","shell.execute_reply":"2022-08-03T15:30:21.373513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"필요 없는 피처 제거: calc 피처들과 ind 10~14, car 14, cat은 위에서 인코딩했으니까 제외","metadata":{}},{"cell_type":"code","source":"drop_features = ['ps_ind_14', 'ps_ind_10_bin', 'ps_ind_11_bin', 'ps_ind_12_bin', 'ps_ind_13_bin', 'ps_car_14']\n\nremaining_features = [feature for feature in all_features\n                     if ('cat' not in feature and\n                        'calc' not in feature and\n                        feature not in drop_features)]","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:30:21.376457Z","iopub.execute_input":"2022-08-03T15:30:21.376941Z","iopub.status.idle":"2022-08-03T15:30:21.384813Z","shell.execute_reply.started":"2022-08-03T15:30:21.376872Z","shell.execute_reply":"2022-08-03T15:30:21.383213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"매트릭스와 리메이닝 합치기","metadata":{}},{"cell_type":"code","source":"from scipy import sparse\n\nall_data_sprs = sparse.hstack([sparse.csr_matrix(all_data[remaining_features]), encoded_cat_matrix], format = 'csr')\n# csr_matrix()로 리메이닝 피쳐스 csr로 바꾸기, 매트릭스는 원래 csr, 그 두개를 csr 이용하여 수평 방향으로 csr 형식으로 합치기","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:30:21.386485Z","iopub.execute_input":"2022-08-03T15:30:21.387740Z","iopub.status.idle":"2022-08-03T15:30:24.557701Z","shell.execute_reply.started":"2022-08-03T15:30:21.387692Z","shell.execute_reply":"2022-08-03T15:30:24.556327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"데이터 나누기","metadata":{}},{"cell_type":"code","source":"num_train = len(train)\n\nX = all_data_sprs[:num_train]\nX_test = all_data_sprs[num_train:]\n\ny = train['target'].values","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:30:24.559410Z","iopub.execute_input":"2022-08-03T15:30:24.559930Z","iopub.status.idle":"2022-08-03T15:30:25.342807Z","shell.execute_reply.started":"2022-08-03T15:30:24.559858Z","shell.execute_reply":"2022-08-03T15:30:25.341536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 평가지표 계산함수 작성_지니계수\n\n경제학에서 지니계수는 소득 수준의 불평등 정도를 나타내지만 (소득낮은순으로 인구 비율을 가로축으로 두고 소득 누적 점유율을 세로축으로 둔 다음 인구 누적 비율 당 소득 누적 점유율을 선으로 연결한 로렌츠 곡선 아래의 도형 넓이 / 소득 완전 균등 분배시 나타나는 45도 직선 아래의 도형 넓이) 데이터분석에서는 예축값을 크기순으로 정렬한다음 같은 원리로 구한다는데, 이는 (2*ROcAUC -1)과 같다고 함. 결국 ROC로 구하는 상황과 유사하다고 생각하면 됨. ","metadata":{}},{"cell_type":"markdown","source":"(정규화 지니계수) = (예측 값에 대한 지니계수) / (예측이 완벽할 때의 지니계수)  -> 1에 가까울수록 예측이 잘 된 것","metadata":{}},{"cell_type":"code","source":"# 실제 타깃값 y_true와 예측 타깃값 y_pred를 이용하여 정규화 지니계수 반환하도록 하기\n\nimport numpy as np\n\ndef eval_gini(y_true, y_pred):\n    assert y_true.shape == y_pred.shape                  # y_true의 크기와 y_pred의 크기가 서로 같은지 확인 (다르면 오류)\n    \n    n_samples = y_true.shape[0]                          # 데이터 개수\n    L_mid = np.linspace(1/n_samples, 1, n_samples)       # 대각선 값 np.linspce(구간 시작점, 구간 끝점, 시작점과 끝점을 균일한 몇 개의 구간으로 나눌 지)\n    \n    # 예측값에 대한 지니계수\n    pred_order = y_true[y_pred.argsort()]                # y_pred 크기순으로 y_true 정렬\n    L_pred = np.cumsum(pred_order) / np.sum(pred_order)  # 로렌츠 곡선\n    G_pred = np.sum(L_mid - L_pred)                      # 예측값에 대한 지니계수 (예측값에 대한 지니계수 * 완전 균등할 때의 삼각형 넓이? 어차피 정규화 때 나눌거라 생략?) \n    \n    # 예측이 완벽할 때 지니계수\n    true_order = y_true[y_true.argsort()]                # y_true 크기순으로 y_true 정렬\n    L_true = np.cumsum(true_order) / np.sum(true_order)  # 로렌츠 곡선\n    G_true = np.sum(L_mid - L_true)                      # 예측이 완벽할 때의 지니계수 (예측이 완벽할 때의 지니계수 * 완전 균등할 때의 삼각형 넓이? 어차피 정규화 때 나눌거라 생략?)  \n    \n    return G_pred / G_true","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:30:25.344371Z","iopub.execute_input":"2022-08-03T15:30:25.344726Z","iopub.status.idle":"2022-08-03T15:30:25.354562Z","shell.execute_reply.started":"2022-08-03T15:30:25.344695Z","shell.execute_reply":"2022-08-03T15:30:25.352960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 모델 훈련 시 검증 파라미터에 전달하기 위한 함수\n\ndef gini(preds, dtrain):                             # preds = 예측 확률값\n    labels = dtrain.get_label()                      # .get_label() \".\" 앞의 데이터셋의 타깃값 반환 = 실제 타깃값 labels\n    return 'gini', eval_gini(labels, preds), True    # return '평가지표 이름', 평가 점수 계산 함수, 평가 점수가 높을수록 좋은지 여부","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:30:25.358651Z","iopub.execute_input":"2022-08-03T15:30:25.361352Z","iopub.status.idle":"2022-08-03T15:30:25.370465Z","shell.execute_reply.started":"2022-08-03T15:30:25.361305Z","shell.execute_reply":"2022-08-03T15:30:25.369493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 모델 훈련 및 성능 검증","metadata":{}},{"cell_type":"markdown","source":"oof 예측 방식: (과대적합 방지, 서로 다른 모델의 평균을 이용하는 앙상블 효과)\n\n1) K폴드 교차 검증(전체 훈련 데이터를 k개 그룹으로 나누고, 한 개는 검증데이터, 나머지 k-1개는 훈련데이터로 지정함)을 수행하면서 \n\n2) 각 폴드마다(검증데이터가 계속 바뀌며) \n\n3) 훈련데이터로 모델을 훈련하고(즉 모델이 계속 새로 만들어짐), \n\n4) 검증 데이터로 모델 성능을 측정하며, \n\n5) 테스트 데이터로 예측. \n\n6) k개 그룹의 검증 데이터로 예측한 확률을 훈련 데이터 실제 타깃값과 비교해 성능 평가점수 계산하고, \n\n7) 실제 테스트 데이터로 k개 예측확률 평균 구하여 제출","metadata":{}},{"cell_type":"markdown","source":"# OOF 방식으로 LightGBM 훈련","metadata":{}},{"cell_type":"markdown","source":" 층화 kfold 교차검증기 생성","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import StratifiedKFold\n\nfolds = StratifiedKFold(n_splits = 5, shuffle = True, random_state = 1991) # n_splits = k값, shuffle = 데이터 섞기","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:30:25.372455Z","iopub.execute_input":"2022-08-03T15:30:25.372995Z","iopub.status.idle":"2022-08-03T15:30:25.383394Z","shell.execute_reply.started":"2022-08-03T15:30:25.372947Z","shell.execute_reply":"2022-08-03T15:30:25.381781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"LightGBM 하이퍼파라미터 설정","metadata":{}},{"cell_type":"code","source":"params = {'objective': 'binary',    # 보험금을 청구하느냐 마느냐 이진분류\n          'learning_rate' : 0.01,   # 학습률\n          'force_row_wise': True, \n          'random_state': 0}        # 랜덤시드","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:30:25.384977Z","iopub.execute_input":"2022-08-03T15:30:25.385966Z","iopub.status.idle":"2022-08-03T15:30:25.398345Z","shell.execute_reply.started":"2022-08-03T15:30:25.385896Z","shell.execute_reply":"2022-08-03T15:30:25.397283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# \"나중에 oof 방식으로 훈련된 모델로 검증 데이터 타깃값을 예측한 확률\"을 담을 1차원 배열 생성\noof_val_preds = np.zeros(X.shape[0])         # k폴드로 하면 모든 데이터가 한 번 씩 훈련 데이터가 되니 사이즈는 X의 크기로\n\n# \"나중에 oof 방식으로 훈련된 모델로 테스트 데이터 타깃값을 예측한 확률\"을 담을 1차원 배열 생성\noof_test_preds = np.zeros(X_test.shape[0])   # 테스트 데이터를 활용해 예측하여 최종 제출할 것이기 때문에 테스트 개수는 X_test.shape[0]\n\n# np.zeros는 지정 개수만큼 0으로 채운 배열 반환","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:30:25.399966Z","iopub.execute_input":"2022-08-03T15:30:25.400634Z","iopub.status.idle":"2022-08-03T15:30:25.412375Z","shell.execute_reply.started":"2022-08-03T15:30:25.400601Z","shell.execute_reply":"2022-08-03T15:30:25.411103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"LightGBM 모델 훈련","metadata":{}},{"cell_type":"code","source":"import lightgbm as lgb\n\nfor idx, (train_idx, valid_idx) in enumerate(folds.split(X, y)):   # kfold 검증기로 나눈 데이터에 각각 train_idx, valid_idx로 할당\n    print('#'*40, f'폴드{idx+1} / 폴드 {folds.n_splits}', '#'*40)   # 각 폴드를 구분하는 문구 출력\n    \n    X_train, y_train = X[train_idx], y[train_idx] # 훈련용 데이터 설정\n    X_valid, y_valid = X[valid_idx], y[valid_idx] # 검증용 데이터 설정\n    \n    dtrain = lgb.Dataset(X_train, y_train) # LightGBM 전용 훈련 데이터셋 \n    dvalid = lgb.Dataset(X_valid, y_valid) # LightGBM 전용 검증 데이터셋\n    \n    # LightGBM 모델 훈련\n    lgb_model = lgb.train(params = params,                # 훈련용 하이퍼파라미터\n                          train_set = dtrain,             # 훈련 데이터셋\n                          num_boost_round = 1000,         # 부스팅 반복 횟수\n                          valid_sets = dvalid,            # 성능 평가용 검증 데이터셋\n                          feval = gini,                   # 검증 평가지표\n                          early_stopping_rounds = 100,    # 조기종료 조건, 현재까지 나온 최고치의 100번 이내로 최고치 갱신되지 않으면 조기종료\n                          verbose_eval = 100)             # 100번째마다 점수 출력\n    \n    # 테스트 데이터를 활용해 oof 예측\n    oof_test_preds += lgb_model.predict(X_test)/folds.n_splits  # 테스트 데이터를 활용해서 만들어진 모델로 oof 예측 (확률값 예측하고 이 과정 5번 반복시켜서 계속 더하므로 미리 5로 나누기)\n    oof_val_preds[valid_idx] += lgb_model.predict(X_valid)      # 모델 성능 평가를 위한 검증 데이터 타깃값 예측 (oof_val_preds[valid_idx]는 위에서 만든 1차원 배열이므로 +를 쓰지만 각 칸에 하나씩 넣어주는 형태)\n    \n    # 지니계수\n    gini_score = eval_gini(y_valid, oof_val_preds[valid_idx])   # 검증 데이터의 실제 y값과 훈련한 모델로 예측한 검증데이터의 y값을 비교하여 정규화 지니계수 생성\n    print(f'폴드{idx+1} 지니계수: {gini_score}\\n')\n    \n    \n    # 훈련 및 검증용 데이터는 lgb.Dataset으로 변환한 dtrain, dvalid를 사용했지만 마지막에 .predic를 활용해 예측할 때는 원본 데이터(X_test, X_valid) 그대로 사용","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:30:25.414252Z","iopub.execute_input":"2022-08-03T15:30:25.415139Z","iopub.status.idle":"2022-08-03T15:36:15.312141Z","shell.execute_reply.started":"2022-08-03T15:30:25.415089Z","shell.execute_reply":"2022-08-03T15:36:15.311043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"로그손실도 출력된 이유는 lightgbm의 기본 평가지표이기 때문","metadata":{}},{"cell_type":"code","source":"print('OOF 검증 데이터 지니계수: ', eval_gini(y, oof_val_preds))","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:36:15.313948Z","iopub.execute_input":"2022-08-03T15:36:15.315647Z","iopub.status.idle":"2022-08-03T15:36:15.450338Z","shell.execute_reply.started":"2022-08-03T15:36:15.315544Z","shell.execute_reply":"2022-08-03T15:36:15.449046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 예측 및 결과 제출","metadata":{}},{"cell_type":"code","source":"submission['target'] = oof_test_preds\nsubmission.to_csv('submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:37:15.481897Z","iopub.execute_input":"2022-08-03T15:37:15.482350Z","iopub.status.idle":"2022-08-03T15:37:17.818915Z","shell.execute_reply.started":"2022-08-03T15:37:15.482314Z","shell.execute_reply":"2022-08-03T15:37:17.817629Z"},"trusted":true},"execution_count":null,"outputs":[]}]}