{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-24T21:10:17.981259Z","iopub.execute_input":"2022-07-24T21:10:17.982162Z","iopub.status.idle":"2022-07-24T21:10:18.016904Z","shell.execute_reply.started":"2022-07-24T21:10:17.982027Z","shell.execute_reply":"2022-07-24T21:10:18.015565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 7.2.1 데이터 둘러보기","metadata":{}},{"cell_type":"code","source":"import pandas as pd\ndata_path = '../input/cat-in-the-dat/'\n\ntrain = pd.read_csv(data_path + 'train.csv', index_col = 'id')\ntest = pd.read_csv(data_path + 'test.csv', index_col = 'id')\nsubmission = pd.read_csv(data_path + 'sample_submission.csv', index_col = 'id')","metadata":{"execution":{"iopub.status.busy":"2022-07-24T21:10:18.019078Z","iopub.execute_input":"2022-07-24T21:10:18.019414Z","iopub.status.idle":"2022-07-24T21:10:20.941910Z","shell.execute_reply.started":"2022-07-24T21:10:18.019385Z","shell.execute_reply":"2022-07-24T21:10:20.940678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train","metadata":{"execution":{"iopub.status.busy":"2022-07-24T21:10:20.943446Z","iopub.execute_input":"2022-07-24T21:10:20.943940Z","iopub.status.idle":"2022-07-24T21:10:21.216877Z","shell.execute_reply.started":"2022-07-24T21:10:20.943901Z","shell.execute_reply":"2022-07-24T21:10:21.215540Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T21:10:21.219650Z","iopub.execute_input":"2022-07-24T21:10:21.220085Z","iopub.status.idle":"2022-07-24T21:10:21.827707Z","shell.execute_reply.started":"2022-07-24T21:10:21.220039Z","shell.execute_reply":"2022-07-24T21:10:21.826386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape, test.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-24T21:10:21.829396Z","iopub.execute_input":"2022-07-24T21:10:21.830106Z","iopub.status.idle":"2022-07-24T21:10:21.838125Z","shell.execute_reply.started":"2022-07-24T21:10:21.830059Z","shell.execute_reply":"2022-07-24T21:10:21.836873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test","metadata":{"execution":{"iopub.status.busy":"2022-07-24T21:10:21.841068Z","iopub.execute_input":"2022-07-24T21:10:21.841630Z","iopub.status.idle":"2022-07-24T21:10:22.020816Z","shell.execute_reply.started":"2022-07-24T21:10:21.841541Z","shell.execute_reply":"2022-07-24T21:10:22.019580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission","metadata":{"execution":{"iopub.status.busy":"2022-07-24T21:10:22.022547Z","iopub.execute_input":"2022-07-24T21:10:22.022896Z","iopub.status.idle":"2022-07-24T21:10:22.036366Z","shell.execute_reply.started":"2022-07-24T21:10:22.022858Z","shell.execute_reply":"2022-07-24T21:10:22.035029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head().T","metadata":{"execution":{"iopub.status.busy":"2022-07-24T21:10:22.038077Z","iopub.execute_input":"2022-07-24T21:10:22.038634Z","iopub.status.idle":"2022-07-24T21:10:22.061541Z","shell.execute_reply.started":"2022-07-24T21:10:22.038590Z","shell.execute_reply":"2022-07-24T21:10:22.060081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"피처 요약표 만들기","metadata":{}},{"cell_type":"code","source":"def resumetable(df):\n    print(f'데이터셋 형상 : {df.shape}')\n    summary = pd.DataFrame(df.dtypes, columns = ['데이터 타입'])\n#     summary = summary.reset_index()\n#     summary = summary.rename(columns = {'index':'피처'})\n#     summary['결측값 개수'] = df.isnull().sum().values\n#     summary['고윳값 개수'] = df.nunique().values\n#     summary['첫 번째 값'] = df.loc[0].values\n#     summary['두 번째 값'] = df.loc[1].values\n#     summary['세 번째 값'] = df.loc[2].values\n    \n    return summary\n\nresumetable(train)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T21:10:22.063108Z","iopub.execute_input":"2022-07-24T21:10:22.063523Z","iopub.status.idle":"2022-07-24T21:10:22.083817Z","shell.execute_reply.started":"2022-07-24T21:10:22.063460Z","shell.execute_reply":"2022-07-24T21:10:22.082401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def resumetable(df):\n    print(f'데이터셋 형상 : {df.shape}')\n    summary = pd.DataFrame(df.dtypes, columns = ['데이터 타입'])\n    summary = summary.reset_index()\n#     summary = summary.rename(columns = {'index':'피처'})\n#     summary['결측값 개수'] = df.isnull().sum().values\n#     summary['고윳값 개수'] = df.nunique().values\n#     summary['첫 번째 값'] = df.loc[0].values\n#     summary['두 번째 값'] = df.loc[1].values\n#     summary['세 번째 값'] = df.loc[2].values\n    \n    return summary\n\nresumetable(train)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T21:10:22.088275Z","iopub.execute_input":"2022-07-24T21:10:22.089473Z","iopub.status.idle":"2022-07-24T21:10:22.105140Z","shell.execute_reply.started":"2022-07-24T21:10:22.089434Z","shell.execute_reply":"2022-07-24T21:10:22.104063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def resumetable(df):\n    print(f'데이터셋 형상 : {df.shape}')\n    summary = pd.DataFrame(df.dtypes, columns = ['데이터 타입'])\n    summary = summary.reset_index()\n    summary = summary.rename(columns = {'index':'피처'})\n#     summary['결측값 개수'] = df.isnull().sum().values\n#     summary['고윳값 개수'] = df.nunique().values\n#     summary['첫 번째 값'] = df.loc[0].values\n#     summary['두 번째 값'] = df.loc[1].values\n#     summary['세 번째 값'] = df.loc[2].values\n    \n    return summary\n\nresumetable(train)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T21:10:22.106363Z","iopub.execute_input":"2022-07-24T21:10:22.106732Z","iopub.status.idle":"2022-07-24T21:10:22.126552Z","shell.execute_reply.started":"2022-07-24T21:10:22.106701Z","shell.execute_reply":"2022-07-24T21:10:22.125345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.isnull()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T21:10:22.128396Z","iopub.execute_input":"2022-07-24T21:10:22.129213Z","iopub.status.idle":"2022-07-24T21:10:22.716321Z","shell.execute_reply.started":"2022-07-24T21:10:22.129164Z","shell.execute_reply":"2022-07-24T21:10:22.714916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T21:10:22.717912Z","iopub.execute_input":"2022-07-24T21:10:22.719089Z","iopub.status.idle":"2022-07-24T21:10:23.270017Z","shell.execute_reply.started":"2022-07-24T21:10:22.719040Z","shell.execute_reply":"2022-07-24T21:10:23.268866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.isnull().sum().values","metadata":{"execution":{"iopub.status.busy":"2022-07-24T21:10:23.271424Z","iopub.execute_input":"2022-07-24T21:10:23.272116Z","iopub.status.idle":"2022-07-24T21:10:23.818400Z","shell.execute_reply.started":"2022-07-24T21:10:23.272071Z","shell.execute_reply":"2022-07-24T21:10:23.817249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def resumetable(df):\n    print(f'데이터셋 형상 : {df.shape}')\n    summary = pd.DataFrame(df.dtypes, columns = ['데이터 타입'])\n    summary = summary.reset_index()\n    summary = summary.rename(columns = {'index':'피처'})\n    summary['결측값 개수'] = df.isnull().sum().values\n#     summary['고윳값 개수'] = df.nunique().values\n#     summary['첫 번째 값'] = df.loc[0].values\n#     summary['두 번째 값'] = df.loc[1].values\n#     summary['세 번째 값'] = df.loc[2].values\n    \n    return summary\n\nresumetable(train)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T21:10:23.820054Z","iopub.execute_input":"2022-07-24T21:10:23.820733Z","iopub.status.idle":"2022-07-24T21:10:24.384699Z","shell.execute_reply.started":"2022-07-24T21:10:23.820687Z","shell.execute_reply":"2022-07-24T21:10:24.383556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def resumetable(df):\n    print(f'데이터셋 형상 : {df.shape}')\n    summary = pd.DataFrame(df.dtypes, columns = ['데이터 타입'])\n    summary = summary.reset_index()\n    summary = summary.rename(columns = {'index':'피처'})\n    summary['결측값 개수'] = df.isnull().sum().values\n    summary['고윳값 개수'] = df.nunique().values\n#     summary['첫 번째 값'] = df.loc[0].values\n#     summary['두 번째 값'] = df.loc[1].values\n#     summary['세 번째 값'] = df.loc[2].values\n    \n    return summary\n\nresumetable(train)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T21:10:24.386214Z","iopub.execute_input":"2022-07-24T21:10:24.387060Z","iopub.status.idle":"2022-07-24T21:10:25.359428Z","shell.execute_reply.started":"2022-07-24T21:10:24.387027Z","shell.execute_reply":"2022-07-24T21:10:25.358292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.nunique()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T21:10:25.360880Z","iopub.execute_input":"2022-07-24T21:10:25.361219Z","iopub.status.idle":"2022-07-24T21:10:25.783524Z","shell.execute_reply.started":"2022-07-24T21:10:25.361190Z","shell.execute_reply":"2022-07-24T21:10:25.782332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def resumetable(df):\n    print(f'데이터셋 형상 : {df.shape}')\n    summary = pd.DataFrame(df.dtypes, columns = ['데이터 타입'])\n    summary = summary.reset_index()\n    summary = summary.rename(columns = {'index':'피처'})\n    summary['결측값 개수'] = df.isnull().sum().values\n    summary['고윳값 개수'] = df.nunique().values\n    summary['첫 번째 값'] = df.loc[0].values\n    summary['두 번째 값'] = df.loc[1].values\n    summary['세 번째 값'] = df.loc[2].values\n    \n    return summary\n\nresumetable(train)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T21:10:25.784799Z","iopub.execute_input":"2022-07-24T21:10:25.785121Z","iopub.status.idle":"2022-07-24T21:10:26.769134Z","shell.execute_reply.started":"2022-07-24T21:10:25.785094Z","shell.execute_reply":"2022-07-24T21:10:26.767918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"피처 요약표 해석하기\n- 이진(binary) 피처 : bin_0 ~ bin_4\n- 명목형(nominal) 피처 : nom_0 ~ nom_9\n- 순서형(ordinal) 피처 : ord_0 ~ ord_5\n- 그 외 피처 : day, month, target","metadata":{}},{"cell_type":"code","source":"for i in range(3):\n    feature = 'ord_' + str(i)\n    print(f'{feature} 고윳값: {train[feature].unique()}')","metadata":{"execution":{"iopub.status.busy":"2022-07-24T21:10:26.770929Z","iopub.execute_input":"2022-07-24T21:10:26.771374Z","iopub.status.idle":"2022-07-24T21:10:26.829889Z","shell.execute_reply.started":"2022-07-24T21:10:26.771337Z","shell.execute_reply":"2022-07-24T21:10:26.828476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(3,6):\n    feature = 'ord_' + str(i)\n    print(f'{feature} 고윳값: {train[feature].unique()}')","metadata":{"execution":{"iopub.status.busy":"2022-07-24T21:10:26.831436Z","iopub.execute_input":"2022-07-24T21:10:26.831899Z","iopub.status.idle":"2022-07-24T21:10:26.901423Z","shell.execute_reply.started":"2022-07-24T21:10:26.831857Z","shell.execute_reply":"2022-07-24T21:10:26.900405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('day :', train['day'].unique())\nprint('month :', train['month'].unique())\nprint('target :', train['target'].unique())","metadata":{"execution":{"iopub.status.busy":"2022-07-24T21:10:26.902605Z","iopub.execute_input":"2022-07-24T21:10:26.903706Z","iopub.status.idle":"2022-07-24T21:10:26.916380Z","shell.execute_reply.started":"2022-07-24T21:10:26.903669Z","shell.execute_reply":"2022-07-24T21:10:26.915222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 7.2.2 데이터 시각화","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\nimport matplotlib as mpl\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2022-07-24T21:10:26.917923Z","iopub.execute_input":"2022-07-24T21:10:26.918366Z","iopub.status.idle":"2022-07-24T21:10:27.561787Z","shell.execute_reply.started":"2022-07-24T21:10:26.918326Z","shell.execute_reply":"2022-07-24T21:10:27.560468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"타깃값 분포\n- 수치형 데이터의 분포를 파악할 땐 주로 displot()을 사용\n- 범주형 데이터의 분포를 파악할 땐 countplot()을 사용","metadata":{}},{"cell_type":"code","source":"mpl.rc('font', size=15)\nplt.figure(figsize=(7,6))\n\nax = sns.countplot(x='target', data=train)\nax.set_title('Target Distribution')","metadata":{"execution":{"iopub.status.busy":"2022-07-24T21:10:27.565585Z","iopub.execute_input":"2022-07-24T21:10:27.566638Z","iopub.status.idle":"2022-07-24T21:10:27.805709Z","shell.execute_reply.started":"2022-07-24T21:10:27.566596Z","shell.execute_reply":"2022-07-24T21:10:27.804277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(ax.patches)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T21:10:27.807087Z","iopub.execute_input":"2022-07-24T21:10:27.807419Z","iopub.status.idle":"2022-07-24T21:10:27.812807Z","shell.execute_reply.started":"2022-07-24T21:10:27.807390Z","shell.execute_reply":"2022-07-24T21:10:27.811558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rectangle = ax.patches[0]\nprint('사각형 높이:', rectangle.get_height())\nprint('사각형 너비:', rectangle.get_width())\nprint('사각형 왼쪽 테두리의 x축 위치:', rectangle.get_x())","metadata":{"execution":{"iopub.status.busy":"2022-07-24T21:10:27.814599Z","iopub.execute_input":"2022-07-24T21:10:27.815080Z","iopub.status.idle":"2022-07-24T21:10:27.826317Z","shell.execute_reply.started":"2022-07-24T21:10:27.815037Z","shell.execute_reply":"2022-07-24T21:10:27.825359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('텍스트 위치의 x좌표:', rectangle.get_x() + rectangle.get_width()/2)\nprint('텍스트 위치의 y좌표:', rectangle.get_height() + len(train) * 0.001)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T21:11:12.976289Z","iopub.execute_input":"2022-07-24T21:11:12.976857Z","iopub.status.idle":"2022-07-24T21:11:12.984993Z","shell.execute_reply.started":"2022-07-24T21:11:12.976809Z","shell.execute_reply":"2022-07-24T21:11:12.983902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def write_percent(ax, total_size):\n    for patch in ax.patches:\n        height = patch.get_height()\n        width = patch.get_width()\n        left_coord = patch.get_x()\n        percent = height/total_size * 100\n        \n        ax.text(x = left_coord + width/2,\n                y = height + total_size * 0.001,\n                s = f'{percent:1.1f}%',\n                ha = 'center')\n        \nplt.figure(figsize=(7,6))\nax = sns.countplot(x='target', data=train)\nwrite_percent(ax, len(train))\nax.set_title('Target Districution')","metadata":{"execution":{"iopub.status.busy":"2022-07-24T21:25:49.636967Z","iopub.execute_input":"2022-07-24T21:25:49.637826Z","iopub.status.idle":"2022-07-24T21:25:49.858265Z","shell.execute_reply.started":"2022-07-24T21:25:49.637786Z","shell.execute_reply":"2022-07-24T21:25:49.856907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"이진 피처 분포","metadata":{}},{"cell_type":"code","source":"import matplotlib.gridspec as gridspec\n\nmpl.rc('font', size=12)\ngrid = gridspec.GridSpec(3,2) # 그래프를 3행 2열로 배치\nplt.figure(figsize=(10,16)) # 전체 Figure 설정\nplt.subplots_adjust(wspace=0.4, hspace=0.3) # 서브플롯 간 좌우/상하 여백 설정\n\nbin_features = ['bin_0', 'bin_1', 'bin_2', 'bin_3', 'bin_4']\n\nfor idx, feature in enumerate(bin_features):\n    ax = plt.subplot(grid[idx])\n    \n    sns.countplot(x=feature,\n                  data=train,\n                  hue='target',\n                  palette='pastel',\n                  ax=ax)\n    \n    ax.set_title(f'{feature} Distribution by Target')\n    write_percent(ax, len(train))","metadata":{"execution":{"iopub.status.busy":"2022-07-24T21:31:46.967162Z","iopub.execute_input":"2022-07-24T21:31:46.967574Z","iopub.status.idle":"2022-07-24T21:31:48.697697Z","shell.execute_reply.started":"2022-07-24T21:31:46.967542Z","shell.execute_reply":"2022-07-24T21:31:48.696545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"명목형 피처 분포","metadata":{}},{"cell_type":"markdown","source":"### 스텝1 : 교차분석표 생성함수 만들기\n- 교차표(cross-tabulation) 혹은 교차분석표는 범주형 데이터 2개를 비교 분석하는데 사용\n- 각 범주형 데이터의 빈도나 통계량을 행과 열로 결합해놓은 표","metadata":{}},{"cell_type":"code","source":"pd.crosstab(train['nom_0'], train['target'])","metadata":{"execution":{"iopub.status.busy":"2022-07-24T21:38:50.619219Z","iopub.execute_input":"2022-07-24T21:38:50.619644Z","iopub.status.idle":"2022-07-24T21:38:50.711139Z","shell.execute_reply.started":"2022-07-24T21:38:50.619610Z","shell.execute_reply":"2022-07-24T21:38:50.709697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"crosstab = pd.crosstab(train['nom_0'], train['target'], normalize='index') * 100\ncrosstab","metadata":{"execution":{"iopub.status.busy":"2022-07-24T21:41:43.759688Z","iopub.execute_input":"2022-07-24T21:41:43.760158Z","iopub.status.idle":"2022-07-24T21:41:43.834267Z","shell.execute_reply.started":"2022-07-24T21:41:43.760120Z","shell.execute_reply":"2022-07-24T21:41:43.832949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"crosstab = crosstab.reset_index()\ncrosstab","metadata":{"execution":{"iopub.status.busy":"2022-07-24T21:43:17.506320Z","iopub.execute_input":"2022-07-24T21:43:17.507187Z","iopub.status.idle":"2022-07-24T21:43:17.519887Z","shell.execute_reply.started":"2022-07-24T21:43:17.507142Z","shell.execute_reply":"2022-07-24T21:43:17.518569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_crosstab(df, feature):\n    crosstab = pd.crosstab(df[feature], df['target'], normalize='index') * 100\n    crosstab = crosstab.reset_index()\n    return crosstab","metadata":{"execution":{"iopub.status.busy":"2022-07-24T21:44:08.059424Z","iopub.execute_input":"2022-07-24T21:44:08.060651Z","iopub.status.idle":"2022-07-24T21:44:08.066630Z","shell.execute_reply.started":"2022-07-24T21:44:08.060609Z","shell.execute_reply":"2022-07-24T21:44:08.065229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"crosstab = get_crosstab(train, 'nom_0')\ncrosstab","metadata":{"execution":{"iopub.status.busy":"2022-07-24T21:44:20.986034Z","iopub.execute_input":"2022-07-24T21:44:20.986451Z","iopub.status.idle":"2022-07-24T21:44:21.066863Z","shell.execute_reply.started":"2022-07-24T21:44:20.986410Z","shell.execute_reply":"2022-07-24T21:44:21.065576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"crosstab[1]","metadata":{"execution":{"iopub.status.busy":"2022-07-24T21:44:37.090564Z","iopub.execute_input":"2022-07-24T21:44:37.090998Z","iopub.status.idle":"2022-07-24T21:44:37.100561Z","shell.execute_reply.started":"2022-07-24T21:44:37.090962Z","shell.execute_reply":"2022-07-24T21:44:37.099255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Step2 : 포인트플롯 생성 함수 만들기\n- ax : 포인트플롯을 그릴 축\n- feature : 포인트플롯으로 그릴 피처\n- crosstab : 교차분석표","metadata":{}},{"cell_type":"code","source":"def plot_pointplot(ax, feature, crosstab):\n    ax2 = ax.twinx() # x축은 공유하고, y축은 공유하지 않는 새로운 축 생성\n    ax2 = sns.pointplot(x=feature, y=1, data=crosstab, order=crosstab[feature].values,\n                        color='black', legend=False)\n    ax2.set_ylim(crosstab[1].min()-5, crosstab[1].max() * 1.1)\n    ax2.set_ylabel('Target 1 Ratio(%)')","metadata":{"execution":{"iopub.status.busy":"2022-07-24T21:48:23.552898Z","iopub.execute_input":"2022-07-24T21:48:23.553274Z","iopub.status.idle":"2022-07-24T21:48:23.561267Z","shell.execute_reply.started":"2022-07-24T21:48:23.553243Z","shell.execute_reply":"2022-07-24T21:48:23.559868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Step3 : 피처 분포도 및 피처별 타깃값 1의 비율 포인트플롯 생성 함수 만들기","metadata":{}},{"cell_type":"code","source":"def plot_cat_dist_with_true_ratio(df, features, num_rows, num_cols, size=(15,20)):\n    \n    plt.figure(figsize=size)\n    grid = gridspec.GridSpec(num_rows, num_cols)\n    plt.subplots_adjust(wspace=0.45, hspace=0.3)\n    \n    for idx, feature in enumerate(features):\n        ax = plt.subplot(grid[idx])\n        crosstab = get_crosstab(df, feature)\n        \n        sns.countplot(x=feature,\n                      data=df,\n                      order=crosstab[feature].values,\n                      color='skyblue',\n                      ax=ax)\n        \n        write_percent(ax, len(df))\n        \n        plot_pointplot(ax, feature, crosstab)\n        \n        ax.set_title(f'{feature} Distribution')","metadata":{"execution":{"iopub.status.busy":"2022-07-24T21:54:52.869996Z","iopub.execute_input":"2022-07-24T21:54:52.870400Z","iopub.status.idle":"2022-07-24T21:54:52.879457Z","shell.execute_reply.started":"2022-07-24T21:54:52.870367Z","shell.execute_reply":"2022-07-24T21:54:52.878546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nom_features = ['nom_0', 'nom_1', 'nom_2', 'nom_3', 'nom_4']\nplot_cat_dist_with_true_ratio(train, nom_features, num_rows=3, num_cols=2)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T21:54:53.930713Z","iopub.execute_input":"2022-07-24T21:54:53.931086Z","iopub.status.idle":"2022-07-24T21:54:56.767647Z","shell.execute_reply.started":"2022-07-24T21:54:53.931057Z","shell.execute_reply":"2022-07-24T21:54:56.766410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"순서형 피처 분포","metadata":{}},{"cell_type":"code","source":"ord_features = ['ord_0', 'ord_1', 'ord_2', 'ord_3']\nplot_cat_dist_with_true_ratio(train, ord_features,num_rows=2, num_cols=2, size=(15,12))","metadata":{"execution":{"iopub.status.busy":"2022-07-24T22:00:46.353840Z","iopub.execute_input":"2022-07-24T22:00:46.354275Z","iopub.status.idle":"2022-07-24T22:00:48.480162Z","shell.execute_reply.started":"2022-07-24T22:00:46.354243Z","shell.execute_reply":"2022-07-24T22:00:48.478799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pandas.api.types import CategoricalDtype\n\nord_1_value = ['Novice', 'Contributor', 'Expert', 'Master', 'Grandmaster']\nord_2_value = ['Freezing', 'Cold', 'Warm', 'Hot', 'Boiling Hot', 'Lava Hot']\n\n# 순서를 지정한 범주형 데이터 타입\nord_1_dtype = CategoricalDtype(categories=ord_1_value, ordered=True)\nord_2_dtype = CategoricalDtype(categories=ord_2_value, ordered=True)\n\n# 데이터 타입 변경\ntrain['ord_1'] = train['ord_1'].astype(ord_1_dtype)\ntrain['ord_2'] = train['ord_2'].astype(ord_2_dtype)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T22:08:35.563275Z","iopub.execute_input":"2022-07-24T22:08:35.563686Z","iopub.status.idle":"2022-07-24T22:08:35.573795Z","shell.execute_reply.started":"2022-07-24T22:08:35.563652Z","shell.execute_reply":"2022-07-24T22:08:35.572683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_cat_dist_with_true_ratio(train, ord_features, num_rows=2, num_cols=2, size=(15,12))","metadata":{"execution":{"iopub.status.busy":"2022-07-24T22:08:54.715234Z","iopub.execute_input":"2022-07-24T22:08:54.715778Z","iopub.status.idle":"2022-07-24T22:08:56.402275Z","shell.execute_reply.started":"2022-07-24T22:08:54.715731Z","shell.execute_reply":"2022-07-24T22:08:56.400936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_cat_dist_with_true_ratio(train, ['ord_4','ord_5'], num_rows=2, num_cols=1, size=(15,12))","metadata":{"execution":{"iopub.status.busy":"2022-07-24T22:10:28.605045Z","iopub.execute_input":"2022-07-24T22:10:28.605430Z","iopub.status.idle":"2022-07-24T22:10:32.828367Z","shell.execute_reply.started":"2022-07-24T22:10:28.605399Z","shell.execute_reply":"2022-07-24T22:10:32.827183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"날짜 피처 분포","metadata":{}},{"cell_type":"code","source":"date_features = ['day', 'month']\nplot_cat_dist_with_true_ratio(train, date_features, num_rows=2, num_cols=1, size=(10,10))","metadata":{"execution":{"iopub.status.busy":"2022-07-24T22:12:32.070571Z","iopub.execute_input":"2022-07-24T22:12:32.071006Z","iopub.status.idle":"2022-07-24T22:12:33.404647Z","shell.execute_reply.started":"2022-07-24T22:12:32.070974Z","shell.execute_reply":"2022-07-24T22:12:33.403851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}