{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-30T13:33:45.054079Z","iopub.execute_input":"2022-07-30T13:33:45.054508Z","iopub.status.idle":"2022-07-30T13:33:45.064004Z","shell.execute_reply.started":"2022-07-30T13:33:45.054463Z","shell.execute_reply":"2022-07-30T13:33:45.062597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 탐색적 데이터 분석, 데이터 특성에 따른 맞춤형 인코딩, \n# 이진분류(타깃값이 0, 1로 구성)\n\n범주형 피쳐 23개를 활용해 해당 데이터가 타깃값 1에 속할 확률을 예측하는 경진대회\n\n1) 인위적으로 만든 데이터 제공\n\n2) 각 피쳐와 타깃값의 의미를 알 수 없음 -> 배경지식 사용 불가능, 데이터만 보고 접근해야 함\n\n3) 제공 데이터는 모두 범주형 (bin_: 이진 피처,  nom_: 명목형 피처,  ord_: 순서형 피처,  day_ month_: 날짜 피쳐)","metadata":{}},{"cell_type":"code","source":"data_path = '/kaggle/input/cat-in-the-dat/'\n\ntrain = pd.read_csv(data_path + 'train.csv', index_col = 'id') # id라는 열을 인덱스로 사용\ntest = pd.read_csv(data_path + 'test.csv', index_col = 'id')\nsubmission = pd.read_csv(data_path + 'sample_submission.csv', index_col = 'id')","metadata":{"execution":{"iopub.status.busy":"2022-07-30T13:33:45.126659Z","iopub.execute_input":"2022-07-30T13:33:45.127482Z","iopub.status.idle":"2022-07-30T13:33:47.912629Z","shell.execute_reply.started":"2022-07-30T13:33:45.127423Z","shell.execute_reply":"2022-07-30T13:33:47.911310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape, test.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-30T13:33:47.914260Z","iopub.execute_input":"2022-07-30T13:33:47.914596Z","iopub.status.idle":"2022-07-30T13:33:47.924302Z","shell.execute_reply.started":"2022-07-30T13:33:47.914565Z","shell.execute_reply":"2022-07-30T13:33:47.923203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head().T # 피쳐 수가 많을 때는 .T 메서드를 이용하여 행과 열 위치를 바꾸어주기","metadata":{"execution":{"iopub.status.busy":"2022-07-30T13:33:47.926243Z","iopub.execute_input":"2022-07-30T13:33:47.927142Z","iopub.status.idle":"2022-07-30T13:33:47.953184Z","shell.execute_reply.started":"2022-07-30T13:33:47.927106Z","shell.execute_reply":"2022-07-30T13:33:47.952127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.head() \n# 타깃값이 0, 1 중 한가지인데, 각 id별로 타깃값이 1일 확룰을 예측해서 저장해주도록\n# 테스트 데이터 인덱스가 3000000부터 시작하므로 인덱스도 3000000부터","metadata":{"execution":{"iopub.status.busy":"2022-07-30T13:33:47.955325Z","iopub.execute_input":"2022-07-30T13:33:47.956446Z","iopub.status.idle":"2022-07-30T13:33:47.967410Z","shell.execute_reply.started":"2022-07-30T13:33:47.956406Z","shell.execute_reply":"2022-07-30T13:33:47.966242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def resumetable(df):\n    print(f'데이터셋 향상: {df.shape}')\n    summary = pd.DataFrame(df.dtypes, columns = ['데이터타입'])\n    summary = summary.reset_index()\n    summary = summary.rename(columns ={'index': '피처'})\n    summary['결측값 개수'] = df.isnull().sum().values\n    summary['고윳값 개수'] = df.nunique().values\n    summary['첫 번째 값'] = df.loc[0].values\n    summary['두 번째 값'] = df.loc[1].values\n    summary['세 번째 값'] = df.loc[2].values\n    \n    return summary\n\nresumetable(train)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T13:33:47.969491Z","iopub.execute_input":"2022-07-30T13:33:47.970372Z","iopub.status.idle":"2022-07-30T13:33:48.963833Z","shell.execute_reply.started":"2022-07-30T13:33:47.970324Z","shell.execute_reply":"2022-07-30T13:33:48.962817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"*왜 이렇게 하지 않는 것일까?*","metadata":{}},{"cell_type":"code","source":"def resumetable(df):\n    print(f'데이터셋 향상: {df.shape}')\n    summary = pd.DataFrame()\n    summary['피처'] = df.columns\n    summary['데이터타입'] = df.dtypes.values\n    summary['결측값 개수'] = df.isnull().sum().values\n    summary['고윳값 개수'] = df.nunique().values\n    summary['첫 번째 값'] = df.loc[0].values\n    summary['두 번째 값'] = df.loc[1].values\n    summary['세 번째 값'] = df.loc[2].values\n    \n    return summary\n\nresumetable(train)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T13:33:48.965376Z","iopub.execute_input":"2022-07-30T13:33:48.965757Z","iopub.status.idle":"2022-07-30T13:33:50.001565Z","shell.execute_reply.started":"2022-07-30T13:33:48.965725Z","shell.execute_reply":"2022-07-30T13:33:50.000246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"bin_0 ~ _4는 고윳값 개수가 모두 2개이므로 이진 피쳐\n\n하지만 첫째~셋째값은 모두 다르게 구성 -> 머신러닝 모델링을 위해서는 모두 숫자로 인코딩해주기","metadata":{}},{"cell_type":"markdown","source":"명목형 피쳐(nom_)는 모두 obect 타입","metadata":{}},{"cell_type":"markdown","source":"순서형 피처(ord_)들은 unique함수들을 이용해 입력된 값들의 종류(고윳값)을 출력해볼 수 있음","metadata":{}},{"cell_type":"code","source":"for i in range(3): #고윳값 적은 ord_1~ord_3까지만 먼저\n    feature = 'ord_' +str(i)\n    print(f'{feature} 고윳값: {train[feature].unique()}')","metadata":{"execution":{"iopub.status.busy":"2022-07-30T13:33:50.003147Z","iopub.execute_input":"2022-07-30T13:33:50.003857Z","iopub.status.idle":"2022-07-30T13:33:50.064452Z","shell.execute_reply.started":"2022-07-30T13:33:50.003816Z","shell.execute_reply":"2022-07-30T13:33:50.063078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(3, 6): #고윳값 많은 ord_4~ord_6\n    feature = 'ord_' +str(i)\n    print(f'{feature} 고윳값: {train[feature].unique()}')","metadata":{"execution":{"iopub.status.busy":"2022-07-30T13:33:50.066195Z","iopub.execute_input":"2022-07-30T13:33:50.066563Z","iopub.status.idle":"2022-07-30T13:33:50.141380Z","shell.execute_reply.started":"2022-07-30T13:33:50.066531Z","shell.execute_reply":"2022-07-30T13:33:50.139779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"추후에 이것들 알파벳순으로 인코딩하기","metadata":{}},{"cell_type":"markdown","source":"일, 월, target도 unique 함수로 고윳값 확인","metadata":{}},{"cell_type":"code","source":"print('day 고윳값: ',  train['day'].unique())\nprint('month 고윳값: ', train['month'].unique())\nprint('target 고윳값: ', train['target'].unique())","metadata":{"execution":{"iopub.status.busy":"2022-07-30T13:33:50.144056Z","iopub.execute_input":"2022-07-30T13:33:50.145197Z","iopub.status.idle":"2022-07-30T13:33:50.158771Z","shell.execute_reply.started":"2022-07-30T13:33:50.145160Z","shell.execute_reply":"2022-07-30T13:33:50.157590Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 데이터 시각화","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib as mpl\nimport matplotlib.pyplot as plt\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2022-07-30T13:33:50.162557Z","iopub.execute_input":"2022-07-30T13:33:50.163511Z","iopub.status.idle":"2022-07-30T13:33:50.843752Z","shell.execute_reply.started":"2022-07-30T13:33:50.163467Z","shell.execute_reply":"2022-07-30T13:33:50.842756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#타깃값 분포\n\nmpl.rc('font', size = 15)\nplt.figure(figsize = (7,6))\n\n#타깃값 분포 카운트플롯\nax = sns.countplot(x = 'target', data= train)\nax.set_title('Target Distribution')","metadata":{"execution":{"iopub.status.busy":"2022-07-30T13:33:50.845317Z","iopub.execute_input":"2022-07-30T13:33:50.845657Z","iopub.status.idle":"2022-07-30T13:33:51.090031Z","shell.execute_reply.started":"2022-07-30T13:33:50.845625Z","shell.execute_reply":"2022-07-30T13:33:51.088816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"타잇값 카운트플롯 상단에 비율을 표시 -> ax.patches","metadata":{}},{"cell_type":"markdown","source":"# **.patches","metadata":{}},{"cell_type":"code","source":"print(ax.patches)\n# ax 축을 구성하는 그래프 도형 객체 모두를 담은 리스트 출력\n# rectangle 객체 두 개(막대그래프 두 개)를 포함하는 리스트임을 알 수 있음","metadata":{"execution":{"iopub.status.busy":"2022-07-30T13:33:51.091892Z","iopub.execute_input":"2022-07-30T13:33:51.092286Z","iopub.status.idle":"2022-07-30T13:33:51.098762Z","shell.execute_reply.started":"2022-07-30T13:33:51.092254Z","shell.execute_reply":"2022-07-30T13:33:51.096972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rectangle = ax.patches[0]\nprint('사각형 높이: ', rectangle.get_height())\nprint('사각형 너비: ', rectangle.get_width())\nprint('사각형 사각형 왼쪽 테두리의 x축 위치: ', rectangle.get_x())","metadata":{"execution":{"iopub.status.busy":"2022-07-30T13:33:51.100855Z","iopub.execute_input":"2022-07-30T13:33:51.101937Z","iopub.status.idle":"2022-07-30T13:33:51.110562Z","shell.execute_reply.started":"2022-07-30T13:33:51.101886Z","shell.execute_reply":"2022-07-30T13:33:51.109431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('택스트 위치의 x좌표:', rectangle.get_x()+rectangle.get_width()/2.0)\nprint('택스트 위치의 y좌표:', rectangle.get_height() + len(train)*0.001)\n#len(train)+0.001: 여백","metadata":{"execution":{"iopub.status.busy":"2022-07-30T13:33:51.111950Z","iopub.execute_input":"2022-07-30T13:33:51.112316Z","iopub.status.idle":"2022-07-30T13:33:51.129727Z","shell.execute_reply.started":"2022-07-30T13:33:51.112284Z","shell.execute_reply":"2022-07-30T13:33:51.128819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"''' def write_percent(ax, total):\n    #도형 객채를 순회하며 막대 상단에 타깃값 비율 표시\n    \nfor patch in ax.patches:\n    height = patch.get_height()      # 도형 높이 = 데이터 개수 \n    width = patch.get_width()        # 도형 너비\n    left_coord = patch.get_x()       # 도형 왼쪽 테두리의 x축 위치\n    percent = height/total*100  # 타깃값 비율\n    \n    # 지정 좌표에 텍스트 입력\n    ax.text(x = left_coord + width/2.0, \n            y = height + total*0.001, \n            s = f'{percent:.1f}%',\n            ha = 'center')\n\n#카운트플롯 그리기    \nplt.figure(figsize = (7,6))\nax = sns.countplot(x = 'target', data = train)\nwrite_percent(ax, len(train))\nax.set_title('Target Distribution'); '''","metadata":{"execution":{"iopub.status.busy":"2022-07-30T13:33:51.131135Z","iopub.execute_input":"2022-07-30T13:33:51.131492Z","iopub.status.idle":"2022-07-30T13:33:51.140555Z","shell.execute_reply.started":"2022-07-30T13:33:51.131457Z","shell.execute_reply":"2022-07-30T13:33:51.138960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class write_percent:\n    def __init__(self, ax, total_size):\n        self.ax = ax\n        self.total_size = total_size\n        #도형 객채를 순회하며 막대 상단에 타깃값 비율 표시\n        \n        for patch in ax.patches:\n            height = patch.get_height()      # 도형 높이 = 데이터 개수 \n            width = patch.get_width()        # 도형 너비\n            left_coord = patch.get_x()       # 도형 왼쪽 테두리의 x축 위치\n            percent = height/total_size*100  # 타깃값 비율\n        \n            # 지정 좌표에 텍스트 입력\n            ax.text(x = left_coord + width/2.0, \n                    y = height + total_size*0.001, \n                    s = f'{percent:.1f}%',\n                    ha = 'center')\n\n#카운트플롯 그리기    \nplt.figure(figsize = (7,6))\nax = sns.countplot(x = 'target', data = train)\nwrite_percent(ax, len(train))\nax.set_title('Target Distribution');","metadata":{"execution":{"iopub.status.busy":"2022-07-30T13:33:51.142492Z","iopub.execute_input":"2022-07-30T13:33:51.143552Z","iopub.status.idle":"2022-07-30T13:33:51.304251Z","shell.execute_reply.started":"2022-07-30T13:33:51.143513Z","shell.execute_reply":"2022-07-30T13:33:51.303413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 이진 피처 분포\n\n이진 피처가 특정 타깃값에 치우쳤는지 확인\n\n즉, 0/1, T/F, Y/N 의 값을 갖는 데이터들이 각각 0/1중에 어떤 타깃값을 가지는지 확인","metadata":{}},{"cell_type":"code","source":"# 여러 그래프를 격자 형태로 준비\nimport matplotlib.gridspec as gridspec\n\n# 3행 2열 틀(figure) 준비\nmpl.rc('font', size = 12)\ngrid = gridspec.GridSpec(3, 2) # 나중에 grid[0]이런 식으로 원하는 서브플롯 지정할 수 있음\nplt.figure(figsize = (10, 16)) # 전체 피규어 사이즈\nplt.subplots_adjust(wspace = 0.4, hspace = 0.3)  # 서브플롯간의 여백\n\n# 서브플롯 그리기\nbin_features = ['bin_0', 'bin_1', 'bin_2', 'bin_3', 'bin_4'] # 이진 피쳐 목록\n\nfor idx, feature in enumerate(bin_features): \n    ax = plt.subplot(grid[idx])                 # for문 이용하여 그래프 그릴 ax 위치 지정\n    sns.countplot(x = feature,                  # 서브플롯 x축 피쳐\n                  data = train,                 # 서브플롯에서 사용할 데이터\n                  hue = 'target',               # 그래프 휴값\n                  palette = 'pastel',          # 색상\n                  ax = ax)                      # ax축 위에서 plt.subplot(grid[idx])로 지정한 것 끌고 옴\n    ax.set_title(f'{feature} Distribution by Target') # 그래프 제목\n    write_percent(ax, len(train))","metadata":{"execution":{"iopub.status.busy":"2022-07-30T13:33:51.305339Z","iopub.execute_input":"2022-07-30T13:33:51.305931Z","iopub.status.idle":"2022-07-30T13:33:53.141354Z","shell.execute_reply.started":"2022-07-30T13:33:51.305901Z","shell.execute_reply":"2022-07-30T13:33:53.140130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 명목형 피쳐 분포\n\n1) 교차 분석표 만들기(피처별로 타깃값 1을 가지는 비율을 포인트플롯으로 나타내기 위해서)","metadata":{}},{"cell_type":"code","source":"pd.crosstab(train['nom_0'], train['target'])","metadata":{"execution":{"iopub.status.busy":"2022-07-30T13:36:15.472072Z","iopub.execute_input":"2022-07-30T13:36:15.473409Z","iopub.status.idle":"2022-07-30T13:36:15.570787Z","shell.execute_reply.started":"2022-07-30T13:36:15.473368Z","shell.execute_reply":"2022-07-30T13:36:15.569583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"crosstab = pd.crosstab(train['nom_0'], train['target'], normalize = 'index')*100\ncrosstab\n\n# normalize에 index 전달하면 인덱스를 기준으로 정규화함, 여기에 columns 전달하면 열을 기준으로 정규화함 -> 비율을 구한다","metadata":{"execution":{"iopub.status.busy":"2022-07-30T13:45:42.416081Z","iopub.execute_input":"2022-07-30T13:45:42.416564Z","iopub.status.idle":"2022-07-30T13:45:42.516740Z","shell.execute_reply.started":"2022-07-30T13:45:42.416522Z","shell.execute_reply":"2022-07-30T13:45:42.515690Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"crosstab = crosstab.reset_index()\ncrosstab\n\n# 피처가 열로 와야 그래프 그리기 편하기 때문에 열로 가져오고 인덱스 재설정하기","metadata":{"execution":{"iopub.status.busy":"2022-07-30T13:54:56.311164Z","iopub.execute_input":"2022-07-30T13:54:56.311559Z","iopub.status.idle":"2022-07-30T13:54:56.323944Z","shell.execute_reply.started":"2022-07-30T13:54:56.311529Z","shell.execute_reply":"2022-07-30T13:54:56.323090Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 자주 사용하므로 함수로 만들어두기\ndef get_crosstab(df, feature):\n    crosstab = pd.crosstab(df[feature], df['target'], normalize = 'index')*100\n    crosstab = crosstab.reset_index()\n    return crosstab","metadata":{"execution":{"iopub.status.busy":"2022-07-30T14:08:20.434291Z","iopub.execute_input":"2022-07-30T14:08:20.434682Z","iopub.status.idle":"2022-07-30T14:08:20.440607Z","shell.execute_reply.started":"2022-07-30T14:08:20.434648Z","shell.execute_reply":"2022-07-30T14:08:20.439404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"crosstab = get_crosstab(train, 'nom_0')\ncrosstab","metadata":{"execution":{"iopub.status.busy":"2022-07-30T14:09:12.921300Z","iopub.execute_input":"2022-07-30T14:09:12.922135Z","iopub.status.idle":"2022-07-30T14:09:13.011763Z","shell.execute_reply.started":"2022-07-30T14:09:12.922096Z","shell.execute_reply":"2022-07-30T14:09:13.010552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 함수를 실행하고 타깃별로 피처 안에 그 타깃값이 얼마나 들어있는지 확인하려면 \"함수[열]\"\ncrosstab[1]","metadata":{"execution":{"iopub.status.busy":"2022-07-30T14:09:19.220469Z","iopub.execute_input":"2022-07-30T14:09:19.220862Z","iopub.status.idle":"2022-07-30T14:09:19.229460Z","shell.execute_reply.started":"2022-07-30T14:09:19.220830Z","shell.execute_reply":"2022-07-30T14:09:19.228101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"2) 포인트 플롯 생성 함수 만들기","metadata":{}},{"cell_type":"code","source":"def plot_pointplot(ax, feature, crosstab):\n    ax2 = ax.twinx()  # x축은 공유하고 y축은 공유하지 않는 새로운 축(ax2) 생성\n    \n    ax2 = sns.pointplot(x = feature,                           # x축 feature\n                        y = 1,                                 # y축 타깃값이 1인 비율(crosstab[1])\n                        data = crosstab,                       \n                        order = crosstab[feature].values,      # 교차분석표의 피쳐 순서대로 그리기\n                        color = 'black',                       # 포인트플롯 색상\n                        legend = False)                        # 범례 사용 안함\n    ax2.set_ylim(crosstab[1].min()-5, crosstab[1].max()*1.1)   # y축범위: \"타깃값이 1인 비율들 중 최솟값에서 5를 뺀 것\" 부터 \"타깃값이 1인 비율들 중 최댓값*1.1\"\n    ax2.set_ylabel('Target 1 Ratio(%)')","metadata":{"execution":{"iopub.status.busy":"2022-07-30T14:59:39.163046Z","iopub.execute_input":"2022-07-30T14:59:39.163500Z","iopub.status.idle":"2022-07-30T14:59:39.171511Z","shell.execute_reply.started":"2022-07-30T14:59:39.163465Z","shell.execute_reply":"2022-07-30T14:59:39.170235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"3) 피처 분포도 및 피처별 타깃값 1의 비율 포인트플롯 생성 함수 만들기\n\n\n앞서 만든 get_crosstab(), plot_pointplot()함수, 카운트플롯 처음 할때 사용한 write_percent 클래스 이용","metadata":{}},{"cell_type":"code","source":"class write_percent:\n    def __init__(self, ax, total_size):\n        self.ax = ax\n        self.total_size = total_size\n        #도형 객채를 순회하며 막대 상단에 타깃값 비율 표시\n        \n        for patch in ax.patches:\n            height = patch.get_height()      # 도형 높이 = 데이터 개수 \n            width = patch.get_width()        # 도형 너비\n            left_coord = patch.get_x()       # 도형 왼쪽 테두리의 x축 위치\n            percent = height/total_size*100  # 타깃값 비율\n        \n            # 지정 좌표에 텍스트 입력\n            ax.text(x = left_coord + width/2.0, \n                    y = height + total_size*0.001, \n                    s = f'{percent:.1f}%',\n                    ha = 'center')\n\ndef plot_pointplot(ax, feature, crosstab):\n    ax2 = ax.twinx()  # x축은 공유하고 y축은 공유하지 않는 새로운 축(ax2) 생성\n    \n    ax2 = sns.pointplot(x = feature,                           # x축 feature\n                        y = 1,                                 # y축 타깃값이 1인 비율(crosstab[1])\n                        data = crosstab,                       \n                        order = crosstab[feature].values,      # 교차분석표의 피쳐 순서대로 그리기\n                        color = 'black',                       # 포인트플롯 색상\n                        legend = False)                        # 범례 사용 안함\n    ax2.set_ylim(crosstab[1].min()-5, crosstab[1].max()*1.1)   # y축범위: \"타깃값이 1인 비율들 중 최솟값에서 5를 뺀 것\" 부터 \"타깃값이 1인 비율들 중 최댓값*1.1\"\n    ax2.set_ylabel('Target 1 Ratio(%)')\n\ndef plot_cat_dist_with_true_ratio(df, features, num_rows, num_cols, size = (15, 20)):\n    plt.figure(figsize = size)\n    grid = gridspec.GridSpec(num_rows, num_cols)     # 서브플롯 배치\n    plt.subplots_adjust(wspace = 0.45, hspace = 0.3) # 서브플롯 여백\n    \n    for idx, feature in enumerate(features):\n        ax = plt.subplot(grid[idx])\n        crosstab = get_crosstab(df, feature)\n    \n        sns.countplot(x=feature,\n                     data = df, \n                     order = crosstab[feature].values, \n                     color = 'skyblue', \n                     ax= ax)\n    \n        write_percent(ax,len(df))\n    \n        plot_pointplot(ax, feature, crosstab)\n    \n        ax.set_title(f'{feature} Distribution')","metadata":{"execution":{"iopub.status.busy":"2022-07-30T15:15:01.945970Z","iopub.execute_input":"2022-07-30T15:15:01.946408Z","iopub.status.idle":"2022-07-30T15:15:01.962665Z","shell.execute_reply.started":"2022-07-30T15:15:01.946374Z","shell.execute_reply":"2022-07-30T15:15:01.961626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nom_features = ['nom_0', 'nom_1', 'nom_2', 'nom_3', 'nom_4']\nplot_cat_dist_with_true_ratio(train, nom_features, num_rows=3, num_cols=2)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T15:08:12.011344Z","iopub.execute_input":"2022-07-30T15:08:12.011753Z","iopub.status.idle":"2022-07-30T15:08:14.783083Z","shell.execute_reply.started":"2022-07-30T15:08:12.011720Z","shell.execute_reply":"2022-07-30T15:08:14.781916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 순서형 피처 분포\n\nplot_cat_dist_with_true_ratio() 함수 이용하여 순서형 피쳐도 확인\n\n고윳값 개수가 적은 ord_0~3만 한 번에 그리고 4, 5는 따로 그리기","metadata":{}},{"cell_type":"code","source":"ord_features = ['ord_0', 'ord_1', 'ord_2', 'ord_3']\n\n# 그 전에 ord_1, ord_2 피쳐값 순서 정렬\n\nfrom pandas.api.types import CategoricalDtype\n\nord_1_value = ['Novice', 'Contributor', 'Expert', 'Master', 'Grandmaster']\nord_2_value = ['Freezing', 'Cold', 'Warm', 'Hot', 'Boiling Hot', 'Lava Hot']\n\nord_1_dtype = CategoricalDtype(categories = ord_1_value, ordered = True)\nord_2_dtype = CategoricalDtype(categories = ord_2_value, ordered = True)\n\ntrain['ord_1'] = train['ord_1'].astype(ord_1_dtype)\ntrain['ord_2'] = train['ord_2'].astype(ord_2_dtype)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T15:24:14.744953Z","iopub.execute_input":"2022-07-30T15:24:14.745398Z","iopub.status.idle":"2022-07-30T15:24:14.755760Z","shell.execute_reply.started":"2022-07-30T15:24:14.745365Z","shell.execute_reply":"2022-07-30T15:24:14.754450Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_cat_dist_with_true_ratio(train, ord_features, num_rows = 2, num_cols = 2, size=(15, 12))","metadata":{"execution":{"iopub.status.busy":"2022-07-30T15:24:19.904427Z","iopub.execute_input":"2022-07-30T15:24:19.905173Z","iopub.status.idle":"2022-07-30T15:24:21.501163Z","shell.execute_reply.started":"2022-07-30T15:24:19.905118Z","shell.execute_reply":"2022-07-30T15:24:21.500304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_cat_dist_with_true_ratio(train, ['ord_4', 'ord_5'], num_rows = 2, num_cols = 1, size=(15, 12))","metadata":{"execution":{"iopub.status.busy":"2022-07-30T15:27:24.353975Z","iopub.execute_input":"2022-07-30T15:27:24.355811Z","iopub.status.idle":"2022-07-30T15:27:28.616917Z","shell.execute_reply.started":"2022-07-30T15:27:24.355768Z","shell.execute_reply":"2022-07-30T15:27:28.616062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 날짜 피처 분포","metadata":{}},{"cell_type":"code","source":"date_features = ['day', 'month']\nplot_cat_dist_with_true_ratio(train, date_features, num_rows = 2, num_cols = 1, size=(10, 10))","metadata":{"execution":{"iopub.status.busy":"2022-07-30T15:30:09.414995Z","iopub.execute_input":"2022-07-30T15:30:09.416044Z","iopub.status.idle":"2022-07-30T15:30:10.195899Z","shell.execute_reply.started":"2022-07-30T15:30:09.415980Z","shell.execute_reply":"2022-07-30T15:30:10.194683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"교재 저자가 미리 인코딩해본 결과, \n\n1) 보편적으로는 일요일과 월요일, 1월과 12월 등 순환되는 부분에는 삼각함수를 사용해 인코딩하면, 머신러닝에서 값 차이가 적을수록 가까운 데이터라고 인식해버리는 탓에 1월과 2월의 간격보다 1월과 12월의 간격을 훨씬 크게 생각해버리는 문제를 해결할 수 있음. \n\n2) 하지만 여기의 데이터들은 종류가 많지 않기 때문에 삼각함수보다는 그냥 원핫인코딩이 성능이 더 좋았음. (1월이랑 12월을 멀게 인식하든 말든 그냥 내버려두기)","metadata":{}}]}