{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n        \n\ndata_path = '/kaggle/input/porto-seguro-safe-driver-prediction/'\ntrain = pd.read_csv(data_path + 'train.csv', index_col = 'id')\ntest = pd.read_csv(data_path + 'test.csv', index_col = 'id')\nsubmission = pd.read_csv(data_path + 'sample_submission.csv', index_col = 'id')\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-02T04:40:04.872372Z","iopub.execute_input":"2022-08-02T04:40:04.872845Z","iopub.status.idle":"2022-08-02T04:40:19.380105Z","shell.execute_reply.started":"2022-08-02T04:40:04.872742Z","shell.execute_reply":"2022-08-02T04:40:19.375659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape, test.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-02T04:40:19.385545Z","iopub.execute_input":"2022-08-02T04:40:19.386765Z","iopub.status.idle":"2022-08-02T04:40:19.419372Z","shell.execute_reply.started":"2022-08-02T04:40:19.386609Z","shell.execute_reply":"2022-08-02T04:40:19.415958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T04:40:19.422889Z","iopub.execute_input":"2022-08-02T04:40:19.423883Z","iopub.status.idle":"2022-08-02T04:40:19.468534Z","shell.execute_reply.started":"2022-08-02T04:40:19.423691Z","shell.execute_reply":"2022-08-02T04:40:19.466993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T04:40:19.471861Z","iopub.execute_input":"2022-08-02T04:40:19.472319Z","iopub.status.idle":"2022-08-02T04:40:19.500420Z","shell.execute_reply.started":"2022-08-02T04:40:19.472283Z","shell.execute_reply":"2022-08-02T04:40:19.499224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T04:40:19.502435Z","iopub.execute_input":"2022-08-02T04:40:19.503451Z","iopub.status.idle":"2022-08-02T04:40:19.522539Z","shell.execute_reply.started":"2022-08-02T04:40:19.503398Z","shell.execute_reply":"2022-08-02T04:40:19.521482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"타깃값: 운전자가 보험금을 청구할 확률 -> 이것에 따라 추후 보험회사는 보험금을 다르게 징수하게 됨. ","metadata":{}},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T04:53:52.976603Z","iopub.execute_input":"2022-08-02T04:53:52.980803Z","iopub.status.idle":"2022-08-02T04:53:53.084153Z","shell.execute_reply.started":"2022-08-02T04:53:52.980626Z","shell.execute_reply":"2022-08-02T04:53:53.082476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"피처명에 다양한 정보가 포함되어있음을 알 수 있음. \n\nps: 기본\n\nind, reg, car, calc: 분류\n\n분류별 일련번호\n\nbin, cat, 생략: 이진피처, 명목형 피처, 순서/연속형 피처","metadata":{}},{"cell_type":"markdown","source":"# 결측값 파악\n\n결측값이 있는 자리에 -1이 입력되어 있어서 결측값이 없다고 나옴 -> np.NaN으로 변환 -> missingno 패키지를 이용하여 결측값 시각화","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport missingno as msno\n\ntrain_copy = train.copy().replace(-1, np.NaN)                # 훈련 데이터를 직접 바꾸지 않고 복사본을 만들어서 바꿈\n\nmsno.bar(df = train_copy.iloc[:, 1:29], figsize = (13,6));   # 처음 28개만 결측값 시각화, 그래프 높이 낮을수록 결측값 많음(정상값을 표현하는 그래프)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T05:06:48.834228Z","iopub.execute_input":"2022-08-02T05:06:48.834637Z","iopub.status.idle":"2022-08-02T05:06:51.809988Z","shell.execute_reply.started":"2022-08-02T05:06:48.834604Z","shell.execute_reply":"2022-08-02T05:06:51.808172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"msno.bar(df = train_copy.iloc[:, 29:], figsize = (13,6)); ","metadata":{"execution":{"iopub.status.busy":"2022-08-02T05:08:22.677314Z","iopub.execute_input":"2022-08-02T05:08:22.677798Z","iopub.status.idle":"2022-08-02T05:08:25.272831Z","shell.execute_reply.started":"2022-08-02T05:08:22.677744Z","shell.execute_reply":"2022-08-02T05:08:25.271558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 매트릭스 형태로 결측값 시각화하기","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"msno.matrix(df = train_copy.iloc[:, 1:29], figsize = (13,6)); ","metadata":{"execution":{"iopub.status.busy":"2022-08-02T05:11:17.536538Z","iopub.execute_input":"2022-08-02T05:11:17.537023Z","iopub.status.idle":"2022-08-02T05:11:22.630965Z","shell.execute_reply.started":"2022-08-02T05:11:17.536985Z","shell.execute_reply":"2022-08-02T05:11:22.629389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 피처 요약표 만들기","metadata":{}},{"cell_type":"code","source":"def resumetable(df):\n    print(f'데이터셋 형상: {df.shape}')\n    summary = pd.DataFrame(df.dtypes, columns = ['데이터 타입'])\n    summary['결측값 개수'] = (df == -1).sum().values # 피처별 -1 개수\n    summary['고윳값 개수'] = df.nunique().values\n    summary['데이터 종류'] = None\n    for col in df.columns:\n        if 'bin' in col or col =='target':\n            summary.loc[col, '데이터 종류'] = '이진형'\n        elif 'cat' in col:\n            summary.loc[col, '데이터 종류'] = '명목형'\n        elif df[col].dtype == float:\n            summary.loc[col, '데이터 종류'] = '연속형'\n        elif df[col].dtype == int:\n            summary.loc[col, '데이터 종류'] = '순서형'\n            \n    return summary","metadata":{"execution":{"iopub.status.busy":"2022-08-02T05:38:16.220672Z","iopub.execute_input":"2022-08-02T05:38:16.221301Z","iopub.status.idle":"2022-08-02T05:38:16.234933Z","shell.execute_reply.started":"2022-08-02T05:38:16.221240Z","shell.execute_reply":"2022-08-02T05:38:16.233341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"summary = resumetable(train)\nsummary","metadata":{"execution":{"iopub.status.busy":"2022-08-02T05:38:18.396773Z","iopub.execute_input":"2022-08-02T05:38:18.397950Z","iopub.status.idle":"2022-08-02T05:38:18.758547Z","shell.execute_reply.started":"2022-08-02T05:38:18.397904Z","shell.execute_reply":"2022-08-02T05:38:18.757164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 요약표에서 원하는 정보만 추출하는 것도 가능\n\nsummary[summary['데이터 종류'] == '명목형'].index","metadata":{"execution":{"iopub.status.busy":"2022-08-02T05:45:52.945426Z","iopub.execute_input":"2022-08-02T05:45:52.945911Z","iopub.status.idle":"2022-08-02T05:45:52.960563Z","shell.execute_reply.started":"2022-08-02T05:45:52.945871Z","shell.execute_reply":"2022-08-02T05:45:52.959091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 적어둔 '데이터 종류'와 관계 없이 데이터 타입에 따라서 추출하는것도 당연히 가능\n\nsummary[summary['데이터 타입'] == 'float64'].index","metadata":{"execution":{"iopub.status.busy":"2022-08-02T05:51:03.551127Z","iopub.execute_input":"2022-08-02T05:51:03.551572Z","iopub.status.idle":"2022-08-02T05:51:03.560887Z","shell.execute_reply.started":"2022-08-02T05:51:03.551539Z","shell.execute_reply":"2022-08-02T05:51:03.559873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 데이터 시각화","metadata":{}},{"cell_type":"code","source":"# 라이브러리 호출\n\nimport seaborn as sns\nimport matplotlib as mpl\nimport matplotlib.pyplot as plt\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2022-08-02T06:37:44.854403Z","iopub.execute_input":"2022-08-02T06:37:44.854854Z","iopub.status.idle":"2022-08-02T06:37:44.865617Z","shell.execute_reply.started":"2022-08-02T06:37:44.854819Z","shell.execute_reply":"2022-08-02T06:37:44.864254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def write_percent(ax, total_size):\n    #도형 객체를 순회하며 막대그래프 상단에 타깃값 비율 표시\n    for patch in ax.patches:\n        height = patch.get_height()\n        width = patch.get_width()\n        left_coord = patch.get_x()\n        percent = height/total_size*100\n        \n        ax.text(left_coord + width/2.0, \n               height + total_size*0.001, \n               '{:1.1f}%'.format(percent), \n               ha = 'center')\n        \nmpl.rc('font', size = 15)\nplt.figure(figsize = (7, 6))\n\nax = sns.countplot(x = 'target', data = train)\nwrite_percent(ax, len(train))\nax.set_title('Target Distribution')\n","metadata":{"execution":{"iopub.status.busy":"2022-08-02T06:38:04.314224Z","iopub.execute_input":"2022-08-02T06:38:04.314676Z","iopub.status.idle":"2022-08-02T06:38:04.564188Z","shell.execute_reply.started":"2022-08-02T06:38:04.314637Z","shell.execute_reply":"2022-08-02T06:38:04.562950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"타깃값이 불균형이기 때문에 비율이 작은 타깃값 1을 잘 예측하는 것이 중요!\n\n-> 각 핓의 고윳값별 타깃값 1 비율 알아보기: 고윳값별로 타깃값 비율이 다르다면 이 피처는 유효한 것, 신뢰구간이 짧다면 유효한 것","metadata":{}},{"cell_type":"markdown","source":"# 이진 피처들의 고윳값별 타깃값 구하기","metadata":{}},{"cell_type":"code","source":"import matplotlib.gridspec as gridspec\n\ndef plot_target_ratio_by_features(df, features, num_rows, num_cols, size = (12, 18)):\n    mpl.rc('font', size = 9)\n    plt.figure(figsize = size)\n    grid = gridspec.GridSpec(num_rows, num_cols)\n    plt.subplots_adjust(wspace = 0.3, hspace = 0.3)\n    \n    for idx, feature in enumerate(features):\n        ax = plt.subplot(grid[idx])\n        sns.barplot(x = feature, y = 'target', data = df, palette = 'Set2', ax = ax)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T06:49:19.958699Z","iopub.execute_input":"2022-08-02T06:49:19.959176Z","iopub.status.idle":"2022-08-02T06:49:19.968246Z","shell.execute_reply.started":"2022-08-02T06:49:19.959139Z","shell.execute_reply":"2022-08-02T06:49:19.967121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bin_features = summary[summary['데이터 종류'] == '이진형'].index\nplot_target_ratio_by_features(train, bin_features, 6, 3)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T06:51:39.027149Z","iopub.execute_input":"2022-08-02T06:51:39.028420Z","iopub.status.idle":"2022-08-02T06:55:21.397916Z","shell.execute_reply.started":"2022-08-02T06:51:39.028371Z","shell.execute_reply":"2022-08-02T06:55:21.396638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"2번 피처 그래프의 경우 \n\n고윳값이 0일 때: 타깃값 0 = 96%, 1 = 4%\n\n고윳값이 1일 때: 타깃값 0 = 97.2%, 1 = 2.8%\n\n고윳값에 따라 타깃값이 1.2% 정도 차이가 나니 꽤나 유의미한 피처","metadata":{}},{"cell_type":"markdown","source":"ps_ind_10_bin ~ ps_ind_13_bin: 신뢰구간이 넓어서 통계적 유효성이 떨어짐\n\nps_calc_15_bin ~ ps_calc_20_bin: 고윳값별 타깃값 비율 차이가 없어 타깃값 예측력 없음 (그 외 아주 적게 차이나는 값들이라도 유의미하다고 보는 듯)","metadata":{}},{"cell_type":"markdown","source":" # 명목형 피처들의 고윳값별 타깃값 구하기","metadata":{}},{"cell_type":"code","source":"nom_features = summary[summary['데이터 종류'] == '명목형'].index\n\nplot_target_ratio_by_features(train, nom_features, 7, 2)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T08:40:58.038482Z","iopub.execute_input":"2022-08-02T08:40:58.039489Z","iopub.status.idle":"2022-08-02T08:43:31.158939Z","shell.execute_reply.started":"2022-08-02T08:40:58.039444Z","shell.execute_reply":"2022-08-02T08:43:31.157689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"ps_car_10_cat: 세 고윳값 간 비율이 비슷하기도 하고 고윳값 2의 신뢰구간이 유독 넓기도 하지만 그럼에도 제거하지 않을 경우 효율이 더 좋았다고 함.","metadata":{}},{"cell_type":"markdown","source":"# 순서형 피처","metadata":{}},{"cell_type":"code","source":"ord_features = summary[summary['데이터 종류'] == '순서형'].index\n\nplot_target_ratio_by_features(train, ord_features, 8, 2, (12, 20))","metadata":{"execution":{"iopub.status.busy":"2022-08-02T08:54:43.503508Z","iopub.execute_input":"2022-08-02T08:54:43.503980Z","iopub.status.idle":"2022-08-02T08:57:04.200536Z","shell.execute_reply.started":"2022-08-02T08:54:43.503943Z","shell.execute_reply":"2022-08-02T08:57:04.199328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"ps_ind_14: 고윳값 4의 타깃값 비율 신뢰구간이 넓어서 유효성 떨어짐, 게다가 나머지는 별 차이 없음\n\nps_calc_04 ~ ps_calc_14: 고윳값별 타깃값 비율 차이가 별로 없고, 차이나는 것은 신뢰구간이 너무 넓음","metadata":{}},{"cell_type":"markdown","source":"# 연속형 피처들의 고윳값별 타깃값 구하기","metadata":{}},{"cell_type":"code","source":"# 컷함수 사용 예시\n# 연속형은 판다스 컷 함수로 구간 만들어주기. 연속형 데이터를 범주형 데이터로 바꾸는 것\npd.cut([1.0, 1.5, 2.1, 2.7, 3.5, 4.0], 3) # 1.0, 1.5, 2.1, 2.7, 3.5, 4.0 이런 애들을 퉁쳐서 3개의 구간으로 만들기.. 대체 왜?","metadata":{"execution":{"iopub.status.busy":"2022-08-02T09:11:01.157428Z","iopub.execute_input":"2022-08-02T09:11:01.157962Z","iopub.status.idle":"2022-08-02T09:11:01.185060Z","shell.execute_reply.started":"2022-08-02T09:11:01.157924Z","shell.execute_reply":"2022-08-02T09:11:01.183623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cont_features = summary[summary['데이터 종류'] == '연속형'].index\n\nplt.figure(figsize = (12, 16))\ngrid = gridspec.GridSpec(5,2)\nplt.subplots_adjust(wspace = 0.2, hspace = 0.4)\n\nfor idx, cont_feature in enumerate(cont_features):\n    train[cont_feature] = pd.cut(train[cont_feature], 5) # 연속형 데이터들 5개의 구간으로\n    \n    ax = plt.subplot(grid[idx])\n    sns.barplot(x = cont_feature, y = 'target', data = train, palette = 'Set2', ax=ax)\n    ax.tick_params(axis = 'x', labelrotation = 10)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T09:16:14.232079Z","iopub.execute_input":"2022-08-02T09:16:14.232859Z","iopub.status.idle":"2022-08-02T09:17:50.508587Z","shell.execute_reply.started":"2022-08-02T09:16:14.232815Z","shell.execute_reply":"2022-08-02T09:17:50.507303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"ps_calc_01 ~ ps_calc_03: 타깃값 비율 차이가 별로 없어서 제거\n\n-> 이로서 피처 종류에 관계 없이 calc들은 모두 제거하는 것으로 판명!","metadata":{}},{"cell_type":"markdown","source":"# 연속형 피처 간 상관관계\n\n-> 상관관계가 높을 경우 두 피처 중 하나의 피처만 남겨두는 것이 좋음! \n\n-> 0.6 ~ 0.79: 강함, 0.8 ~ 1.0: 아주 강함","metadata":{}},{"cell_type":"markdown","source":"결측값 삭제 후 히트맵 그려서 알아보기\n\n결측값이 있으면 상관관계를 제대로 구하지 못한대!!! 왜지?","metadata":{}},{"cell_type":"code","source":"train_copy = train_copy.dropna()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T09:21:48.510014Z","iopub.execute_input":"2022-08-02T09:21:48.511250Z","iopub.status.idle":"2022-08-02T09:21:50.754366Z","shell.execute_reply.started":"2022-08-02T09:21:48.511186Z","shell.execute_reply":"2022-08-02T09:21:50.753117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (10, 8))\ncont_corr = train_copy[cont_features].corr()\nsns.heatmap(cont_corr, annot = True, cmap = 'OrRd');","metadata":{"execution":{"iopub.status.busy":"2022-08-02T09:25:26.428082Z","iopub.execute_input":"2022-08-02T09:25:26.428967Z","iopub.status.idle":"2022-08-02T09:25:27.201084Z","shell.execute_reply.started":"2022-08-02T09:25:26.428917Z","shell.execute_reply":"2022-08-02T09:25:27.200158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"12와 14가 가장 강한 상관계수를 보이지만 통상적으로 이 정도를 가지고 제거하지는 않음. 하지만 필자가 확인해본 결과 제거하는 편이 좋았음\n\n2와 3도 비슷한 상관계수이지만 이는 제거하지 않는 편이 좋았음. ","metadata":{}}]}