{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Categorical Feature Encoding Challenge\n* Data Analystics notebook\n\n> 머신러닝, 딥러닝 문제해결 전략 책(신백균 저) 참고","metadata":{}},{"cell_type":"code","source":"import pandas as pd\n\ndatapath = '/kaggle/input/cat-in-the-dat/'\n\ntrain = pd.read_csv(datapath + 'train.csv', index_col ='id')\ntest = pd.read_csv(datapath + 'test.csv', index_col ='id')\nsubmission = pd.read_csv(datapath + 'sample_submission.csv', index_col ='id')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-23T03:17:52.470095Z","iopub.execute_input":"2022-07-23T03:17:52.471116Z","iopub.status.idle":"2022-07-23T03:17:55.916398Z","shell.execute_reply.started":"2022-07-23T03:17:52.471000Z","shell.execute_reply":"2022-07-23T03:17:55.915243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape, test.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-23T03:17:55.919486Z","iopub.execute_input":"2022-07-23T03:17:55.919955Z","iopub.status.idle":"2022-07-23T03:17:55.932083Z","shell.execute_reply.started":"2022-07-23T03:17:55.919908Z","shell.execute_reply":"2022-07-23T03:17:55.930641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head().T","metadata":{"execution":{"iopub.status.busy":"2022-07-23T03:17:55.933842Z","iopub.execute_input":"2022-07-23T03:17:55.934315Z","iopub.status.idle":"2022-07-23T03:17:55.964079Z","shell.execute_reply.started":"2022-07-23T03:17:55.934246Z","shell.execute_reply":"2022-07-23T03:17:55.962767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T03:17:55.966823Z","iopub.execute_input":"2022-07-23T03:17:55.967148Z","iopub.status.idle":"2022-07-23T03:17:55.984288Z","shell.execute_reply.started":"2022-07-23T03:17:55.967120Z","shell.execute_reply":"2022-07-23T03:17:55.983135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T03:17:55.987162Z","iopub.execute_input":"2022-07-23T03:17:55.987894Z","iopub.status.idle":"2022-07-23T03:17:56.014976Z","shell.execute_reply.started":"2022-07-23T03:17:55.987815Z","shell.execute_reply":"2022-07-23T03:17:56.013627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 피처 요약표 만들기","metadata":{}},{"cell_type":"code","source":"def resumetable(df):\n    print(f'데이터셋 형상 : {df.shape}')\n    summary = pd.DataFrame(df.dtypes, columns=['데이터 타입'])\n    summary = summary.reset_index()\n    summary = summary.rename(columns={'index':'피처'})\n    summary['결측값 개수'] = df.isnull().sum().values\n    summary['고유값 개수'] = df.nunique().values\n    summary['첫 번째 값'] = df.loc[0].values\n    summary['두 번째 값'] = df.loc[1].values\n    summary['세 번째 값'] = df.loc[2].values\n    \n    return summary\n\nresumetable(train)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T03:17:56.016586Z","iopub.execute_input":"2022-07-23T03:17:56.017156Z","iopub.status.idle":"2022-07-23T03:17:57.058760Z","shell.execute_reply.started":"2022-07-23T03:17:56.017121Z","shell.execute_reply":"2022-07-23T03:17:57.056097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(3):\n    feature = 'ord_' + str(i)\n    print(f'{feature} 고윳값: {train[feature].unique()}')","metadata":{"execution":{"iopub.status.busy":"2022-07-23T03:17:57.059904Z","iopub.execute_input":"2022-07-23T03:17:57.060653Z","iopub.status.idle":"2022-07-23T03:17:57.116464Z","shell.execute_reply.started":"2022-07-23T03:17:57.060616Z","shell.execute_reply":"2022-07-23T03:17:57.115148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(3, 6):\n    feature = 'ord_' + str(i)\n    print(f'{feature} 고윳값: {train[feature].unique()}')","metadata":{"execution":{"iopub.status.busy":"2022-07-23T03:17:57.118036Z","iopub.execute_input":"2022-07-23T03:17:57.118365Z","iopub.status.idle":"2022-07-23T03:17:57.188740Z","shell.execute_reply.started":"2022-07-23T03:17:57.118337Z","shell.execute_reply":"2022-07-23T03:17:57.187348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('day 고윳값:',train['day'].unique())\nprint('month 고윳값:', train['month'].unique())\nprint('target 고윳값:', train['target'].unique())","metadata":{"execution":{"iopub.status.busy":"2022-07-23T03:17:57.190157Z","iopub.execute_input":"2022-07-23T03:17:57.190585Z","iopub.status.idle":"2022-07-23T03:17:57.205828Z","shell.execute_reply.started":"2022-07-23T03:17:57.190530Z","shell.execute_reply":"2022-07-23T03:17:57.204683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 데이터 시각화","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib as mpl\nimport matplotlib.pyplot as plt\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2022-07-23T03:17:57.207260Z","iopub.execute_input":"2022-07-23T03:17:57.207613Z","iopub.status.idle":"2022-07-23T03:17:57.887920Z","shell.execute_reply.started":"2022-07-23T03:17:57.207578Z","shell.execute_reply":"2022-07-23T03:17:57.886532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* 수치형 데이터의 분포를 파악할 땐 주로 displot()을 사용하고 범주형 데이터의 분포를 파악할 땐 countplot()을 사용","metadata":{}},{"cell_type":"markdown","source":"타겟값 분포\n","metadata":{}},{"cell_type":"code","source":"mpl.rc('font', size=15) # 폰트 크기 설정\nplt.figure(figsize=(7, 6)) # Figure 크기 설정\n\n# 타겟값 분포 카운트플롯\nax = sns.countplot(x='target', data=train)\nax.set_title('Target Distribution');","metadata":{"execution":{"iopub.status.busy":"2022-07-23T03:17:57.889309Z","iopub.execute_input":"2022-07-23T03:17:57.889738Z","iopub.status.idle":"2022-07-23T03:17:58.113831Z","shell.execute_reply.started":"2022-07-23T03:17:57.889691Z","shell.execute_reply":"2022-07-23T03:17:58.112527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(ax.patches)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T03:17:58.115257Z","iopub.execute_input":"2022-07-23T03:17:58.115685Z","iopub.status.idle":"2022-07-23T03:17:58.121418Z","shell.execute_reply.started":"2022-07-23T03:17:58.115653Z","shell.execute_reply":"2022-07-23T03:17:58.120357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rectangle = ax.patches[0] # 첫번째 Rectangle 객체\nprint('사각형의 높이:', rectangle.get_height())\nprint('사각형의 너비:', rectangle.get_width())\nprint('사각형 왼쪽 테두리의 x축 위치:', rectangle.get_x())","metadata":{"execution":{"iopub.status.busy":"2022-07-23T03:17:58.122968Z","iopub.execute_input":"2022-07-23T03:17:58.123708Z","iopub.status.idle":"2022-07-23T03:17:58.133538Z","shell.execute_reply.started":"2022-07-23T03:17:58.123649Z","shell.execute_reply":"2022-07-23T03:17:58.132568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('텍스트 위치의 x좌표:', rectangle.get_x() + rectangle.get_width() / 2.0)\nprint('텍스트 위치의 y좌표:', rectangle.get_height() + len(train) * 0.001)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T03:17:58.136946Z","iopub.execute_input":"2022-07-23T03:17:58.137504Z","iopub.status.idle":"2022-07-23T03:17:58.142712Z","shell.execute_reply.started":"2022-07-23T03:17:58.137474Z","shell.execute_reply":"2022-07-23T03:17:58.141934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def write_percent(ax, total_size):\n    '''도형 객체를 순회하며 막대 상단에 타겟값 비율 표시'''\n    for patch in ax.patches:\n        height = patch.get_height()     # 도형 높이(데이터 개수)\n        width = patch.get_width()       # 도형 너비\n        left_coord = patch.get_x()      # 도형 왼쪽 테두리의 x축 위치\n        percent = height/total_size*100 # 타겟값 비율\n        \n        # (x, y) 좌표에 텍스트 입력\n        ax.text(x=left_coord + width/2.0,    # x축 위치\n                y=height + total_size*0.001, # y축 위치\n                s=f'{percent:1.1f}%',        # 입력 텍스트\n                ha='center')                 # 가운데 정렬\n        \nplt.figure(figsize=(7, 6))\n\nax = sns.countplot(x='target', data=train)\nwrite_percent(ax, len(train))  # 비율 표시\nax.set_title('Target Distribution');","metadata":{"execution":{"iopub.status.busy":"2022-07-23T03:17:58.144099Z","iopub.execute_input":"2022-07-23T03:17:58.144675Z","iopub.status.idle":"2022-07-23T03:17:58.347066Z","shell.execute_reply.started":"2022-07-23T03:17:58.144644Z","shell.execute_reply":"2022-07-23T03:17:58.345640Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### 이진 피처 분포","metadata":{}},{"cell_type":"code","source":"import matplotlib.gridspec as gridspec # 여러 그래프를 격자 형태로 배치\n# 3행 2열 Figure(틀) 준비\nmpl.rc('font', size=12)\ngrid = gridspec.GridSpec(3, 2) # 그래프(서브플롯)를 3행 2열로 배치\nplt.figure(figsize=(10, 16))   # 전체 Figure 크기 설정\nplt.subplots_adjust(wspace=0.4, hspace=0.3) # 서브플롯 간 좌우/상하 여백 설정\n\n# 서브플롯 그리기\nbin_features = ['bin_0','bin_1', 'bin_2', 'bin_3', 'bin_4']  # 피처 목록\n\nfor idx, feature in enumerate(bin_features):\n    ax = plt.subplot(grid[idx])\n    \n    # ax축에 타겟값 분포 카운트플롯 그리기\n    sns.countplot(x=feature,\n                  data=train,\n                  hue='target',\n                  palette='pastel', # 그래프 색상 설정\n                  ax=ax)\n    \n    ax.set_title(f'{feature} Distribution by Target')  # 그래프 제목 설정\n    write_percent(ax, len(train))                      # 비율 표시","metadata":{"execution":{"iopub.status.busy":"2022-07-23T03:17:58.349671Z","iopub.execute_input":"2022-07-23T03:17:58.350705Z","iopub.status.idle":"2022-07-23T03:18:00.125304Z","shell.execute_reply.started":"2022-07-23T03:17:58.350664Z","shell.execute_reply":"2022-07-23T03:18:00.123889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"위의 결과로 이진 피처는 특정 타겟값에 치우치지 않았음을 확인가능","metadata":{}},{"cell_type":"markdown","source":"##### 명목형 피처 분포\n* norm_5 ~ 9까지는 고유값이 많고 의미를 알 수 없는 문자열이므로 생략<br>\n\n>Step1. 교차분석표 생성 함수 만들기<br>\nStep2. 포인트플롯 생성 함수 만들기<br>\nStep3. 피처 분포도 및 포인트플롯 생성 함수 만들기<br>","metadata":{}},{"cell_type":"markdown","source":"Step1. 교차분석표 생성 함수 만들기","metadata":{}},{"cell_type":"code","source":"pd.crosstab(train['nom_0'], train['target'])","metadata":{"execution":{"iopub.status.busy":"2022-07-23T03:46:04.943005Z","iopub.execute_input":"2022-07-23T03:46:04.943798Z","iopub.status.idle":"2022-07-23T03:46:05.061579Z","shell.execute_reply.started":"2022-07-23T03:46:04.943760Z","shell.execute_reply":"2022-07-23T03:46:05.060378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 정규화 후 비율을 백분율로 표현\ncrosstab = pd.crosstab(train['nom_0'], train['target'], normalize='index') * 100\ncrosstab","metadata":{"execution":{"iopub.status.busy":"2022-07-23T03:47:46.794404Z","iopub.execute_input":"2022-07-23T03:47:46.795260Z","iopub.status.idle":"2022-07-23T03:47:46.891630Z","shell.execute_reply.started":"2022-07-23T03:47:46.795208Z","shell.execute_reply":"2022-07-23T03:47:46.890408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"crosstab = crosstab.reset_index()\ncrosstab","metadata":{"execution":{"iopub.status.busy":"2022-07-23T03:48:55.410291Z","iopub.execute_input":"2022-07-23T03:48:55.410731Z","iopub.status.idle":"2022-07-23T03:48:55.425482Z","shell.execute_reply.started":"2022-07-23T03:48:55.410695Z","shell.execute_reply":"2022-07-23T03:48:55.424051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_crosstab(df, feature):\n    crosstab = pd.crosstab(df[feature], df['target'], normalize='index')*100\n    crosstab = crosstab.reset_index()\n    return crosstab","metadata":{"execution":{"iopub.status.busy":"2022-07-23T03:50:08.546040Z","iopub.execute_input":"2022-07-23T03:50:08.546441Z","iopub.status.idle":"2022-07-23T03:50:08.553682Z","shell.execute_reply.started":"2022-07-23T03:50:08.546409Z","shell.execute_reply":"2022-07-23T03:50:08.552479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"crosstab = get_crosstab(train, 'nom_0')\ncrosstab","metadata":{"execution":{"iopub.status.busy":"2022-07-23T03:50:32.676064Z","iopub.execute_input":"2022-07-23T03:50:32.676759Z","iopub.status.idle":"2022-07-23T03:50:32.774078Z","shell.execute_reply.started":"2022-07-23T03:50:32.676720Z","shell.execute_reply":"2022-07-23T03:50:32.772959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"crosstab[1]","metadata":{"execution":{"iopub.status.busy":"2022-07-23T03:50:57.008129Z","iopub.execute_input":"2022-07-23T03:50:57.008640Z","iopub.status.idle":"2022-07-23T03:50:57.017314Z","shell.execute_reply.started":"2022-07-23T03:50:57.008591Z","shell.execute_reply":"2022-07-23T03:50:57.015928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Step2. 포인트플롯 생성 함수 만들기<br>\n\nplot_pointplot() : 카운트플롯이 그려진 축에 포인트플롯을 중복으로 그려줌\n* ax : 포인트플롯을 그릴 축\n* feature : 포인트플롯으로 그릴 피처\n* crosstab : 교차분석표","metadata":{}},{"cell_type":"code","source":"def plot_pointplot(ax, feature, crosstab):\n    ax2 = ax.twinx() # x축은 공유하고 y축은 공유하지 않는 새로운 축 생성\n    # 새로운 축에 포인트플롯 그리기\n    ax2 = sns.pointplot(x=feature, y=1, data=crosstab,\n                        order=crosstab[feature].values, # 포인트플롯 순서\n                        color='black',                   # 포인트플롯 색상\n                        legend=False)                   # 범례 미표시\n    ax2.set_ylim(crosstab[1].min()-5, crosstab[1].max()*1.1) # y축 범위 설정\n    ax2.set_ylabel('Target 1 Ratio(%)')","metadata":{"execution":{"iopub.status.busy":"2022-07-23T03:57:05.741524Z","iopub.execute_input":"2022-07-23T03:57:05.741913Z","iopub.status.idle":"2022-07-23T03:57:05.750507Z","shell.execute_reply.started":"2022-07-23T03:57:05.741884Z","shell.execute_reply":"2022-07-23T03:57:05.749068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Step3. 피처 분포도 및 피처별 타겟값 1의 비율 포인트플롯 생성 함수 만들기\n* get_crosstab(), plot_pointplot()을 활용한 그래프 그리는 함수","metadata":{}},{"cell_type":"code","source":"def plot_cat_dist_with_true_ratio(df, features, num_rows, num_cols, size=(15, 20)):\n    plt.figure(figsize=size)  # 전체 Figure 크기 설정\n    grid = gridspec.GridSpec(num_rows, num_cols) # 서브플롯 배치\n    plt.subplots_adjust(wspace=0.45, hspace=0.3) # 서브플롯 좌우/상하 여백 설정\n    \n    for idx, feature in enumerate(features):\n        ax = plt.subplot(grid[idx])\n        crosstab = get_crosstab(df, feature) # 교차분석표 생성\n        \n        # ax축에 타겟값 분포 카운트플롯 그리기\n        sns.countplot(x=feature, data=df,\n                      order=crosstab[feature].values,\n                      color='skyblue',\n                      ax=ax)\n        \n        write_percent(ax, len(df)) # 비율 표시\n        \n        plot_pointplot(ax, feature, crosstab) # 포인트플롯 그리기\n        \n        ax.set_title(f'{feature} Distribution') # 그래프 제목 설정","metadata":{"execution":{"iopub.status.busy":"2022-07-23T04:06:00.301207Z","iopub.execute_input":"2022-07-23T04:06:00.302402Z","iopub.status.idle":"2022-07-23T04:06:00.313013Z","shell.execute_reply.started":"2022-07-23T04:06:00.302358Z","shell.execute_reply":"2022-07-23T04:06:00.311635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"norm_features = []\nfor i in range(5):   # 명목형 피처 nom_0 ~ 4까지 리스트로 만들기\n    norm_features.append(f'nom_{i}')\n\nplot_cat_dist_with_true_ratio(train, norm_features,\n                              num_rows=3, num_cols=2)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T04:16:11.726249Z","iopub.execute_input":"2022-07-23T04:16:11.726676Z","iopub.status.idle":"2022-07-23T04:16:14.922508Z","shell.execute_reply.started":"2022-07-23T04:16:11.726644Z","shell.execute_reply":"2022-07-23T04:16:14.921648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"norm의 고윳값마다 Target 1에 대한 비중이 다르므로 모두 의미있는 데이터","metadata":{}},{"cell_type":"markdown","source":"##### 순서형 피처 분포","metadata":{}},{"cell_type":"code","source":"ord_features = []\nfor i in range(4):\n    ord_features.append(f'ord_{i}')\n\nplot_cat_dist_with_true_ratio(train, ord_features,\n                              num_rows=2, num_cols=2, size=(15, 12))","metadata":{"execution":{"iopub.status.busy":"2022-07-23T04:17:39.863710Z","iopub.execute_input":"2022-07-23T04:17:39.864346Z","iopub.status.idle":"2022-07-23T04:17:42.144507Z","shell.execute_reply.started":"2022-07-23T04:17:39.864312Z","shell.execute_reply":"2022-07-23T04:17:42.143288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"ord_1, 2는 순서정렬이 되지 않음 => CategoricalDtype(0을 이용하여 피처에 순서를 지정할 수 있음\n* categories : 범주형 데이터 타입으로 인코딩할 값 목록\n* ordered : True로 설정하면 categories에 전달한 값의 순서가 유지됨","metadata":{}},{"cell_type":"code","source":"from pandas.api.types import CategoricalDtype\nord_1_value = ['Novice', 'Contributor', 'Expert', 'Master', 'Grandmaster']\nord_2_value = ['Freezing', 'Cold', 'Warm', 'Hot', 'Boiling Hot', 'Lava Hot']\n\n# 순서를 지정한 범주형 데이터 타입\nord_1_dtype = CategoricalDtype(categories=ord_1_value, ordered=True)\nord_2_dtype = CategoricalDtype(categories=ord_2_value, ordered=True)\n\n# 데이터 타입 변경\ntrain['ord_1'] = train['ord_1'].astype(ord_1_dtype)\ntrain['ord_2'] = train['ord_2'].astype(ord_2_dtype)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T04:32:49.669986Z","iopub.execute_input":"2022-07-23T04:32:49.670394Z","iopub.status.idle":"2022-07-23T04:32:49.960470Z","shell.execute_reply.started":"2022-07-23T04:32:49.670359Z","shell.execute_reply":"2022-07-23T04:32:49.959018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_cat_dist_with_true_ratio(train, ord_features,\n                              num_rows=2, num_cols=2, size=(15, 12))","metadata":{"execution":{"iopub.status.busy":"2022-07-23T04:41:28.543435Z","iopub.execute_input":"2022-07-23T04:41:28.543873Z","iopub.status.idle":"2022-07-23T04:41:30.255938Z","shell.execute_reply.started":"2022-07-23T04:41:28.543837Z","shell.execute_reply":"2022-07-23T04:41:30.254632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_cat_dist_with_true_ratio(train, ['ord_4', 'ord_5'],\n                              num_rows=2, num_cols=1, size=(15, 12))","metadata":{"execution":{"iopub.status.busy":"2022-07-23T04:42:32.942098Z","iopub.execute_input":"2022-07-23T04:42:32.942858Z","iopub.status.idle":"2022-07-23T04:42:37.550948Z","shell.execute_reply.started":"2022-07-23T04:42:32.942817Z","shell.execute_reply":"2022-07-23T04:42:37.549412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"ord feature들은 고윳값 순서에 따라 타겟값이 1인 비율이 증가하는 경향을 보임","metadata":{}},{"cell_type":"markdown","source":"날짜 피처 분포","metadata":{}},{"cell_type":"code","source":"date_features = ['day', 'month']\nplot_cat_dist_with_true_ratio(train, date_features,\n                              num_rows=2, num_cols=2, size=(10, 10))","metadata":{"execution":{"iopub.status.busy":"2022-07-23T04:44:05.049848Z","iopub.execute_input":"2022-07-23T04:44:05.050263Z","iopub.status.idle":"2022-07-23T04:44:05.866911Z","shell.execute_reply.started":"2022-07-23T04:44:05.050219Z","shell.execute_reply":"2022-07-23T04:44:05.865718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"요일과 월 데이터는 원-핫 인코딩 처리","metadata":{}},{"cell_type":"markdown","source":"### 분석 정리\n1. 결측값은 없음\n2. 모든 피처가 중요\n3. 이진 피처 인코딩 : 숫자가 아닌 이진 피처는 0과 1로 인코딩\n4. 명목형 피처 인코딩 : 전체 데이터가 크지 않으므로 모두 원-핫 인코딩\n5. 순서형 피처 인코딩 : 코윳값들의 순서에 맞게 인코딩 (이미 숫자로 되어 있다면 인코딩 X)\n6. 날짜 피처 인코딩 : 값의 크고 작음으로 해석되지 못하도록 원-핫 인코딩","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}