{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-01T22:17:45.528400Z","iopub.execute_input":"2022-08-01T22:17:45.528968Z","iopub.status.idle":"2022-08-01T22:17:45.563146Z","shell.execute_reply.started":"2022-08-01T22:17:45.528848Z","shell.execute_reply":"2022-08-01T22:17:45.562153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 8.2 탐색적 데이터 분석","metadata":{}},{"cell_type":"markdown","source":"### 8.2.1 데이터 둘러보기","metadata":{}},{"cell_type":"code","source":"import pandas as pd\n\ndata_path = '/kaggle/input/porto-seguro-safe-driver-prediction/'\n\ntrain = pd.read_csv(data_path + 'train.csv', index_col = 'id')\ntest = pd.read_csv(data_path + 'test.csv', index_col = 'id')\nsubmission = pd.read_csv(data_path + 'sample_submission.csv', index_col = 'id')","metadata":{"execution":{"iopub.status.busy":"2022-08-01T22:20:01.802093Z","iopub.execute_input":"2022-08-01T22:20:01.802551Z","iopub.status.idle":"2022-08-01T22:20:14.645530Z","shell.execute_reply.started":"2022-08-01T22:20:01.802518Z","shell.execute_reply":"2022-08-01T22:20:14.644240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape, test.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-01T22:20:15.767612Z","iopub.execute_input":"2022-08-01T22:20:15.768977Z","iopub.status.idle":"2022-08-01T22:20:15.780763Z","shell.execute_reply.started":"2022-08-01T22:20:15.768918Z","shell.execute_reply":"2022-08-01T22:20:15.779288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T22:20:24.920127Z","iopub.execute_input":"2022-08-01T22:20:24.921611Z","iopub.status.idle":"2022-08-01T22:20:24.953361Z","shell.execute_reply.started":"2022-08-01T22:20:24.921552Z","shell.execute_reply":"2022-08-01T22:20:24.952218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T22:20:38.072576Z","iopub.execute_input":"2022-08-01T22:20:38.072994Z","iopub.status.idle":"2022-08-01T22:20:38.098226Z","shell.execute_reply.started":"2022-08-01T22:20:38.072960Z","shell.execute_reply":"2022-08-01T22:20:38.097033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T22:20:59.033874Z","iopub.execute_input":"2022-08-01T22:20:59.034699Z","iopub.status.idle":"2022-08-01T22:20:59.051836Z","shell.execute_reply.started":"2022-08-01T22:20:59.034645Z","shell.execute_reply":"2022-08-01T22:20:59.048742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T22:21:28.184928Z","iopub.execute_input":"2022-08-01T22:21:28.185384Z","iopub.status.idle":"2022-08-01T22:21:28.290867Z","shell.execute_reply.started":"2022-08-01T22:21:28.185333Z","shell.execute_reply":"2022-08-01T22:21:28.288680Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport missingno as msno\n\ntrain_copy = train.copy().replace(-1, np.NaN)\n\nmsno.bar(df=train_copy.iloc[:, 1:29], figsize=(13,6))","metadata":{"execution":{"iopub.status.busy":"2022-08-01T22:24:49.568812Z","iopub.execute_input":"2022-08-01T22:24:49.569307Z","iopub.status.idle":"2022-08-01T22:24:52.581556Z","shell.execute_reply.started":"2022-08-01T22:24:49.569270Z","shell.execute_reply":"2022-08-01T22:24:52.580374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport missingno as msno\n\ntrain_copy = train.copy().replace(-1, np.NaN)\n\nmsno.bar(df=train_copy.iloc[:, 29:], figsize=(13,6))","metadata":{"execution":{"iopub.status.busy":"2022-08-01T22:26:37.451537Z","iopub.execute_input":"2022-08-01T22:26:37.452924Z","iopub.status.idle":"2022-08-01T22:26:40.374144Z","shell.execute_reply.started":"2022-08-01T22:26:37.452858Z","shell.execute_reply":"2022-08-01T22:26:40.372790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"결측값 매트릭스 형태로 시각화하기","metadata":{}},{"cell_type":"code","source":"msno.matrix(df=train_copy.iloc[:, 1:29], figsize=(13,6));","metadata":{"execution":{"iopub.status.busy":"2022-08-01T22:27:16.039179Z","iopub.execute_input":"2022-08-01T22:27:16.039619Z","iopub.status.idle":"2022-08-01T22:27:20.777068Z","shell.execute_reply.started":"2022-08-01T22:27:16.039587Z","shell.execute_reply":"2022-08-01T22:27:20.775821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"피처 요약표","metadata":{}},{"cell_type":"code","source":"def resumetable(df):\n    print(f'데이터셋 형상: {df.shape}')\n    summary = pd.DataFrame(df.dtypes, columns = ['데이터 타입'])\n    summary['결측값 개수'] = (df == -1).sum().values\n    summary['고윳값 개수'] = df.nunique().values\n    summary['데이터 종류'] = None\n    for col in df.columns :\n        if 'bin' in col or col == 'target':\n            summary.loc[col, '데이터 종류'] = '이진형'\n        elif 'cat' in col:\n            summary.loc[col, '데이터 종류'] = '명목형'\n        elif df[col].dtype == float:\n            summary.loc[col, '데이터 종류'] = '연속형'\n        elif df[col].dtype == int:\n            summary.loc[col, '데이터 종류'] = '순서형'\n            \n    return summary","metadata":{"execution":{"iopub.status.busy":"2022-08-01T22:31:52.652303Z","iopub.execute_input":"2022-08-01T22:31:52.652757Z","iopub.status.idle":"2022-08-01T22:31:52.663276Z","shell.execute_reply.started":"2022-08-01T22:31:52.652724Z","shell.execute_reply":"2022-08-01T22:31:52.662048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"summary = resumetable(train)\nsummary","metadata":{"execution":{"iopub.status.busy":"2022-08-01T22:32:11.645368Z","iopub.execute_input":"2022-08-01T22:32:11.645782Z","iopub.status.idle":"2022-08-01T22:32:12.020487Z","shell.execute_reply.started":"2022-08-01T22:32:11.645750Z","shell.execute_reply":"2022-08-01T22:32:12.019012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"summary[summary['데이터 종류'] == '명목형'].index","metadata":{"execution":{"iopub.status.busy":"2022-08-01T22:42:08.213007Z","iopub.execute_input":"2022-08-01T22:42:08.213489Z","iopub.status.idle":"2022-08-01T22:42:08.225716Z","shell.execute_reply.started":"2022-08-01T22:42:08.213445Z","shell.execute_reply":"2022-08-01T22:42:08.224445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"summary[summary['데이터 타입'] == 'float64'].index","metadata":{"execution":{"iopub.status.busy":"2022-08-01T22:42:40.063090Z","iopub.execute_input":"2022-08-01T22:42:40.064117Z","iopub.status.idle":"2022-08-01T22:42:40.073117Z","shell.execute_reply.started":"2022-08-01T22:42:40.064074Z","shell.execute_reply":"2022-08-01T22:42:40.072136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 8.2.2 데이터 시각화","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib as mpl\nimport matplotlib.pyplot as plt\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2022-08-01T22:43:46.335643Z","iopub.execute_input":"2022-08-01T22:43:46.336444Z","iopub.status.idle":"2022-08-01T22:43:46.348281Z","shell.execute_reply.started":"2022-08-01T22:43:46.336391Z","shell.execute_reply":"2022-08-01T22:43:46.346708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"타깃값 분포","metadata":{}},{"cell_type":"code","source":"def write_percent(ax, total_size):\n    for patch in ax.patches:\n        height = patch.get_height()\n        width = patch.get_width()\n        left_coord = patch.get_x()\n        percent = height/total_size * 100\n        \n        ax.text(left_coord + width/2.0,\n                height + total_size * 0.001,\n                '{:1.1f}%'.format(percent),\n                ha='center')\n        \nmpl.rc('font', size=15)\nplt.figure(figsize=(7,6))\n\nax = sns.countplot(x='target', data=train)\nwrite_percent(ax, len(train))\nax.set_title('Target Distribution')","metadata":{"execution":{"iopub.status.busy":"2022-08-01T22:45:50.060848Z","iopub.execute_input":"2022-08-01T22:45:50.061367Z","iopub.status.idle":"2022-08-01T22:45:51.147538Z","shell.execute_reply.started":"2022-08-01T22:45:50.061328Z","shell.execute_reply":"2022-08-01T22:45:51.146154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"이진 피처","metadata":{}},{"cell_type":"code","source":"import matplotlib.gridspec as gridspec\n\ndef plot_target_ratio_by_features(df, features, num_rows, num_cols, size=(12,18)):\n    mpl.rc('font', size=9)\n    plt.figure(figsize=size)\n    grid = gridspec.GridSpec(num_rows, num_cols) # 서브플롯 배치\n    plt.subplots_adjust(wspace=0.3, hspace=0.3)\n    \n    for idx, feature in enumerate(features):\n        ax = plt.subplot(grid[idx])\n        sns.barplot(x=feature, y='target', data=df, palette='Set2', ax=ax)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T22:52:15.394305Z","iopub.execute_input":"2022-08-01T22:52:15.394772Z","iopub.status.idle":"2022-08-01T22:52:15.404518Z","shell.execute_reply.started":"2022-08-01T22:52:15.394737Z","shell.execute_reply":"2022-08-01T22:52:15.402983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bin_features = summary[summary['데이터 종류'] == '이진형'].index\n\nplot_target_ratio_by_features(train, bin_features, 6, 3)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T22:53:20.162896Z","iopub.execute_input":"2022-08-01T22:53:20.163369Z","iopub.status.idle":"2022-08-01T22:56:38.843483Z","shell.execute_reply.started":"2022-08-01T22:53:20.163333Z","shell.execute_reply":"2022-08-01T22:56:38.842275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(bin_features)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T22:56:38.845572Z","iopub.execute_input":"2022-08-01T22:56:38.846600Z","iopub.status.idle":"2022-08-01T22:56:38.854625Z","shell.execute_reply.started":"2022-08-01T22:56:38.846541Z","shell.execute_reply":"2022-08-01T22:56:38.853049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"명목형 피처","metadata":{}},{"cell_type":"code","source":"nom_features = summary[summary['데이터 종류'] == '명목형'].index\n\nplot_target_ratio_by_features(train, nom_features, 7, 2)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T22:59:19.503748Z","iopub.execute_input":"2022-08-01T22:59:19.504230Z","iopub.status.idle":"2022-08-01T23:01:48.181562Z","shell.execute_reply.started":"2022-08-01T22:59:19.504178Z","shell.execute_reply":"2022-08-01T23:01:48.180170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"순서형 피처","metadata":{}},{"cell_type":"code","source":"ord_features = summary[summary['데이터 종류'] == '순서형'].index\n\nplot_target_ratio_by_features(train, ord_features, 8,2, (12,10))","metadata":{"execution":{"iopub.status.busy":"2022-08-01T23:04:59.636245Z","iopub.execute_input":"2022-08-01T23:04:59.636696Z","iopub.status.idle":"2022-08-01T23:07:12.008831Z","shell.execute_reply.started":"2022-08-01T23:04:59.636661Z","shell.execute_reply":"2022-08-01T23:07:12.007625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"연속형 피처","metadata":{}},{"cell_type":"code","source":"cont_features = summary[summary['데이터 종류'] == '연속형'].index\n\nplt.figure(figsize=(12,6))\ngrid = gridspec.GridSpec(5,2)\nplt.subplots_adjust(wspace=0.2, hspace=0.4)\n\nfor idx, cont_feature in enumerate(cont_features):\n    # 값을 5개 구간으로 나누기\n    train[cont_feature] = pd.cut(train[cont_feature], 5)\n    \n    ax = plt.subplot(grid[idx])\n    sns.barplot(x=cont_feature, y='target', data=train, palette='Set2', ax=ax)\n    ax.tick_params(axis='x', labelrotation=10)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T23:09:05.553656Z","iopub.execute_input":"2022-08-01T23:09:05.554124Z","iopub.status.idle":"2022-08-01T23:10:37.016650Z","shell.execute_reply.started":"2022-08-01T23:09:05.554092Z","shell.execute_reply":"2022-08-01T23:10:37.015348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.cut([1.0, 1.5, 2.1, 2.7, 3.5, 4.0], 3)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T23:10:37.108051Z","iopub.execute_input":"2022-08-01T23:10:37.108820Z","iopub.status.idle":"2022-08-01T23:10:37.120395Z","shell.execute_reply.started":"2022-08-01T23:10:37.108784Z","shell.execute_reply":"2022-08-01T23:10:37.119260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"연속형 피처2","metadata":{}},{"cell_type":"code","source":"train_copy = train_copy.dropna()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T23:12:56.535849Z","iopub.execute_input":"2022-08-01T23:12:56.536340Z","iopub.status.idle":"2022-08-01T23:12:56.680609Z","shell.execute_reply.started":"2022-08-01T23:12:56.536302Z","shell.execute_reply":"2022-08-01T23:12:56.679232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10,8))\ncont_corr = train_copy[cont_features].corr()\nsns.heatmap(cont_corr, annot=True, cmap='OrRd')","metadata":{"execution":{"iopub.status.busy":"2022-08-01T23:13:46.953447Z","iopub.execute_input":"2022-08-01T23:13:46.953946Z","iopub.status.idle":"2022-08-01T23:13:47.723179Z","shell.execute_reply.started":"2022-08-01T23:13:46.953911Z","shell.execute_reply":"2022-08-01T23:13:47.721752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}