{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 7장 범주형 데이터 이진분류 경진대회","metadata":{"papermill":{"duration":0.021557,"end_time":"2021-07-31T03:06:28.409888","exception":false,"start_time":"2021-07-31T03:06:28.388331","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"## **데이터 설명**\n\n<피쳐> 의미는 알 수 없음. 아무 맥락이 없\nbin_* : 이진데이터(설명?)          \nnom_* : 명목데이터      \nord_* : 순서형데이터         \nday, month          \n<예측> 0, 1 형태로 예측함\nbin, nom, ord..각각 무엇을 말하는지?","metadata":{}},{"cell_type":"markdown","source":"## 7.2 범주형 데이터 이진분류 경진대회 탐색적 데이터 분석","metadata":{}},{"cell_type":"code","source":"import pandas as pd\n# 데이터 경로\ndata_path = '/kaggle/input/cat-in-the-dat/'\n\ntrain = pd.read_csv(data_path + 'train.csv', index_col='id')\ntest = pd.read_csv(data_path + 'test.csv', index_col='id')\nsubmission = pd.read_csv(data_path + 'sample_submission.csv', index_col='id')\ntrain #300000 *24col\ntest #200000 *23col, target 제","metadata":{"execution":{"iopub.status.busy":"2022-07-26T08:04:29.525421Z","iopub.execute_input":"2022-07-26T08:04:29.525742Z","iopub.status.idle":"2022-07-26T08:04:32.827770Z","shell.execute_reply.started":"2022-07-26T08:04:29.525710Z","shell.execute_reply":"2022-07-26T08:04:32.826679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def resumetable(df):\n    print(f'데이터 세트 형상: {df.shape}')\n    summary = pd.DataFrame(df.dtypes, columns=['데이터 타입'])\n    summary = summary.reset_index()\n    summary = summary.rename(columns={'index': '피처'})\n    summary['결측값 개수'] = df.isnull().sum().values\n    summary['고윳값 개수'] = df.nunique().values\n    #summary['첫 번째 값'] = df.loc[0].values\n    #summary['두 번째 값'] = df.loc[1].values\n    #summary['세 번째 값'] = df.loc[2].values\n\n    #고윳값도 한번에 출력하기\n    unique_feature = []\n    for i in range(len(df.columns)):\n        feature = train.columns[i]\n        unique_feature.append(train[feature].unique())\n    summary['고윳값'] = unique_feature\n    \n    return summary\n\nresumetable(train)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T08:04:32.829348Z","iopub.execute_input":"2022-07-26T08:04:32.829621Z","iopub.status.idle":"2022-07-26T08:04:34.318721Z","shell.execute_reply.started":"2022-07-26T08:04:32.829579Z","shell.execute_reply":"2022-07-26T08:04:34.317753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 탐색적 데이터 분석","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib as mpl\nimport matplotlib.pyplot as plt\n%matplotlib inline\n#\nsns.set_palette(\"pastel\")\n\nmpl.rc('font', size=15) # 폰트 크기 설정\nplt.figure(figsize=(7, 6)) # Figure 크기 설정\n\n# 타깃값 분포 카운트플롯\nax = sns.countplot(x='target', data=train)\nax.set(title='Target Distribution');","metadata":{"execution":{"iopub.status.busy":"2022-07-26T08:04:34.319990Z","iopub.execute_input":"2022-07-26T08:04:34.320306Z","iopub.status.idle":"2022-07-26T08:04:35.757308Z","shell.execute_reply.started":"2022-07-26T08:04:34.320266Z","shell.execute_reply":"2022-07-26T08:04:35.756232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## ax.patches 사용법","metadata":{}},{"cell_type":"code","source":"rectangle = ax.patches[0] # 첫 번째 Rectangle 객체\nprint('사각형 높이:', rectangle.get_height())\nprint('사각형 너비:', rectangle.get_width())\nprint('사각형 왼쪽 테두리의 x축 위치:', rectangle.get_x())\n\nprint('텍스트 위치의 x좌표:', rectangle.get_x() + rectangle.get_width()/2.0)\nprint('텍스트 위치의 y좌표:', rectangle.get_height() + len(train)*0.001)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T08:04:35.759887Z","iopub.execute_input":"2022-07-26T08:04:35.760155Z","iopub.status.idle":"2022-07-26T08:04:35.770697Z","shell.execute_reply.started":"2022-07-26T08:04:35.760127Z","shell.execute_reply":"2022-07-26T08:04:35.769127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def write_percent(ax, total_size):\n    '''도형 객체를 순회하며 막대 상단에 타깃값 비율 표시'''\n    for patch in ax.patches:\n        height = patch.get_height()     # 도형 높이(데이터 개수)\n        width = patch.get_width()       # 도형 너비\n        left_coord = patch.get_x()      # 도형 왼쪽 테두리의 x축 위치\n        percent = height/total_size*100 # 타깃값 비율\n        \n        # (x, y) 좌표에 텍스트 입력 \n        ax.text(x=left_coord + width/2.0,    # x축 위치\n                y=height + total_size*0.001, # y축 위치\n                s=f'{percent:1.1f}%',        # 입력 텍스트\n                ha='center')                 # 가운데 정렬\n\nplt.figure(figsize=(7, 6))\n\nax = sns.countplot(x='target', data=train)\nwrite_percent(ax, len(train)) # 비율 표시\nax.set_title('Target Distribution');","metadata":{"execution":{"iopub.status.busy":"2022-07-26T08:04:35.772284Z","iopub.execute_input":"2022-07-26T08:04:35.772760Z","iopub.status.idle":"2022-07-26T08:04:36.024806Z","shell.execute_reply.started":"2022-07-26T08:04:35.772693Z","shell.execute_reply":"2022-07-26T08:04:36.023515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 이진피쳐 분포","metadata":{}},{"cell_type":"code","source":"import matplotlib.gridspec as gridspec # 여러 그래프를 격자 형태로 배치\n\n# 3행 2열 틀(Figure) 준비\nmpl.rc('font', size=12)\ngrid = gridspec.GridSpec(3, 2) # 그래프(서브플롯)를 3행 2열로 배치\nplt.figure(figsize=(10, 16))   # 전체 Figure 크기 설정\nplt.subplots_adjust(wspace=0.4, hspace=0.3) # 서브플롯 간 좌우/상하 여백 설정\n\n# 서브플롯 그리기\nbin_features = ['bin_0', 'bin_1', 'bin_2', 'bin_3', 'bin_4'] # 피처 목록\n\nfor idx, feature in enumerate(bin_features): \n    ax = plt.subplot(grid[idx]) \n    \n    # ax축에 타깃값 분포 카운트플롯 그리기\n    sns.countplot(x=feature,\n                  data=train,\n                  hue='target',\n                  palette='pastel', # 그래프 색상 설정\n                  ax=ax)\n    \n    ax.set_title(f'{feature} Distribution by Target') # 그래프 제목 설정\n    write_percent(ax, len(train))                     # 비율 표시","metadata":{"execution":{"iopub.status.busy":"2022-07-26T08:04:36.026484Z","iopub.execute_input":"2022-07-26T08:04:36.026862Z","iopub.status.idle":"2022-07-26T08:04:38.162790Z","shell.execute_reply.started":"2022-07-26T08:04:36.026820Z","shell.execute_reply":"2022-07-26T08:04:38.161831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 명목형 피처 분포\n# 스텝 1 : 교차분석표 생성 함수 만들기","metadata":{}},{"cell_type":"code","source":"pd.crosstab(train['nom_0'], train['target'])","metadata":{"execution":{"iopub.status.busy":"2022-07-26T08:04:38.164448Z","iopub.execute_input":"2022-07-26T08:04:38.165167Z","iopub.status.idle":"2022-07-26T08:04:38.252867Z","shell.execute_reply.started":"2022-07-26T08:04:38.165117Z","shell.execute_reply":"2022-07-26T08:04:38.252016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 정규화 후 비율을 백분율로 표현\ncrosstab = pd.crosstab(train['nom_0'], train['target'], normalize='index')*100\ncrosstab # 각 행의 합이 100%가 됨","metadata":{"execution":{"iopub.status.busy":"2022-07-26T08:04:38.254228Z","iopub.execute_input":"2022-07-26T08:04:38.254902Z","iopub.status.idle":"2022-07-26T08:04:38.323187Z","shell.execute_reply.started":"2022-07-26T08:04:38.254868Z","shell.execute_reply":"2022-07-26T08:04:38.322185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"crosstab = crosstab.reset_index() # 인덱스 재설정\ncrosstab","metadata":{"execution":{"iopub.status.busy":"2022-07-26T08:04:38.324343Z","iopub.execute_input":"2022-07-26T08:04:38.324615Z","iopub.status.idle":"2022-07-26T08:04:38.338797Z","shell.execute_reply.started":"2022-07-26T08:04:38.324584Z","shell.execute_reply":"2022-07-26T08:04:38.337785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_crosstab(df, feature):\n    crosstab = pd.crosstab(df[feature], df['target'], normalize='index')*100\n    crosstab = crosstab.reset_index()\n    return crosstab\n\ncrosstab = get_crosstab(train, 'nom_0')\ncrosstab","metadata":{"execution":{"iopub.status.busy":"2022-07-26T08:04:38.342669Z","iopub.execute_input":"2022-07-26T08:04:38.342955Z","iopub.status.idle":"2022-07-26T08:04:38.415971Z","shell.execute_reply.started":"2022-07-26T08:04:38.342913Z","shell.execute_reply":"2022-07-26T08:04:38.415324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"crosstab[1]\n","metadata":{"execution":{"iopub.status.busy":"2022-07-26T08:04:38.417077Z","iopub.execute_input":"2022-07-26T08:04:38.417944Z","iopub.status.idle":"2022-07-26T08:04:38.424840Z","shell.execute_reply.started":"2022-07-26T08:04:38.417911Z","shell.execute_reply":"2022-07-26T08:04:38.424094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 스텝 2 : 포인트플롯 생성 함수 만들기\n# 스텝 3 : 피처 분포도 및 피처별 타깃값 1의 비율 포인트플롯 생성 함수 만들기\n","metadata":{}},{"cell_type":"code","source":"def plot_pointplot(ax, feature, crosstab):\n    ax2 = ax.twinx() # x축은 공유하고 y축은 공유하지 않는 새로운 축 생성\n    # 새로운 축에 포인트플롯 그리기\n    ax2 = sns.pointplot(x=feature, y=1, data=crosstab,\n                        order=crosstab[feature].values, # 포인트플롯 순서\n                        color='black',                  # 포인트플롯 색상\n                        legend=False)                   # 범례 미표시\n    ax2.set_ylim(crosstab[1].min()-5, crosstab[1].max()*1.1) # y축 범위 설정\n    ax2.set_ylabel('Target 1 Ratio(%)')\n    \ndef plot_cat_dist_with_true_ratio(df, features, num_rows, num_cols, \n                                  size=(15, 20)):\n    plt.figure(figsize=size)  # 전체 Figure 크기 설정\n    grid = gridspec.GridSpec(num_rows, num_cols) # 서브플롯 배치\n    plt.subplots_adjust(wspace=0.45, hspace=0.3) # 서브플롯 좌우/상하 여백 설정\n    \n    for idx, feature in enumerate(features): \n        ax = plt.subplot(grid[idx])\n        crosstab = get_crosstab(df, feature) # 교차분석표 생성\n\n        # ax축에 타깃값 분포 카운트플롯 그리기\n        sns.countplot(x=feature, data=df,\n                      order=crosstab[feature].values,\n                      color='skyblue',\n                      ax=ax)\n\n        write_percent(ax, len(df)) # 비율 표시\n       \n        plot_pointplot(ax, feature, crosstab) # 포인트플롯 그리기\n        \n        ax.set_title(f'{feature} Distribution') # 그래프 제목 설정\n","metadata":{"execution":{"iopub.status.busy":"2022-07-26T08:04:38.426227Z","iopub.execute_input":"2022-07-26T08:04:38.426796Z","iopub.status.idle":"2022-07-26T08:04:38.438666Z","shell.execute_reply.started":"2022-07-26T08:04:38.426757Z","shell.execute_reply":"2022-07-26T08:04:38.437701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nom_features = ['nom_0', 'nom_1', 'nom_2', 'nom_3', 'nom_4'] # 명목형 피처\nplot_cat_dist_with_true_ratio(train, nom_features, num_rows=3, num_cols=2)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T08:04:38.440146Z","iopub.execute_input":"2022-07-26T08:04:38.440688Z","iopub.status.idle":"2022-07-26T08:04:41.817899Z","shell.execute_reply.started":"2022-07-26T08:04:38.440652Z","shell.execute_reply":"2022-07-26T08:04:41.817208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#순서형 피처 분포\nord_features = ['ord_0', 'ord_1', 'ord_2', 'ord_3'] # 순서형 피처\nplot_cat_dist_with_true_ratio(train, ord_features, \n                              num_rows=2, num_cols=2, size=(15, 12))","metadata":{"execution":{"iopub.status.busy":"2022-07-26T08:04:41.819153Z","iopub.execute_input":"2022-07-26T08:04:41.819904Z","iopub.status.idle":"2022-07-26T08:04:44.587589Z","shell.execute_reply.started":"2022-07-26T08:04:41.819869Z","shell.execute_reply":"2022-07-26T08:04:44.586671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pandas.api.types import CategoricalDtype \n\nord_1_value = ['Novice', 'Contributor', 'Expert', 'Master', 'Grandmaster']\nord_2_value = ['Freezing', 'Cold', 'Warm', 'Hot', 'Boiling Hot', 'Lava Hot']\n\n# 순서를 지정한 범주형 데이터 타입\nord_1_dtype = CategoricalDtype(categories=ord_1_value, ordered=True)\nord_2_dtype = CategoricalDtype(categories=ord_2_value, ordered=True)\n\n# 데이터 타입 변경\ntrain['ord_1'] = train['ord_1'].astype(ord_1_dtype)\ntrain['ord_2'] = train['ord_2'].astype(ord_2_dtype)\nplot_cat_dist_with_true_ratio(train, ord_features, \n                              num_rows=2, num_cols=2, size=(15, 12))","metadata":{"execution":{"iopub.status.busy":"2022-07-26T08:04:44.588879Z","iopub.execute_input":"2022-07-26T08:04:44.589130Z","iopub.status.idle":"2022-07-26T08:04:47.042551Z","shell.execute_reply.started":"2022-07-26T08:04:44.589099Z","shell.execute_reply":"2022-07-26T08:04:47.041634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_cat_dist_with_true_ratio(train, ['ord_4', 'ord_5'], \n                              num_rows=2, num_cols=1, size=(15, 12))\n","metadata":{"execution":{"iopub.status.busy":"2022-07-26T08:04:47.043637Z","iopub.execute_input":"2022-07-26T08:04:47.043851Z","iopub.status.idle":"2022-07-26T08:04:53.125700Z","shell.execute_reply.started":"2022-07-26T08:04:47.043825Z","shell.execute_reply":"2022-07-26T08:04:53.124075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"date_features = ['day', 'month']\nplot_cat_dist_with_true_ratio(train, date_features, \n                              num_rows=2, num_cols=1, size=(10, 10))","metadata":{"execution":{"iopub.status.busy":"2022-07-26T08:04:53.127655Z","iopub.execute_input":"2022-07-26T08:04:53.128085Z","iopub.status.idle":"2022-07-26T08:04:54.178248Z","shell.execute_reply.started":"2022-07-26T08:04:53.128036Z","shell.execute_reply":"2022-07-26T08:04:54.177311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 7.3 범주형 데이터 이진분류 경진대회 베이스라인 모델","metadata":{}},{"cell_type":"code","source":"import pandas as pd\n# 데이터 경로\ndata_path = '/kaggle/input/cat-in-the-dat/'\n\ntrain = pd.read_csv(data_path + 'train.csv', index_col='id')\ntest = pd.read_csv(data_path + 'test.csv', index_col='id')\nsubmission = pd.read_csv(data_path + 'sample_submission.csv', index_col='id')","metadata":{"execution":{"iopub.status.busy":"2022-07-26T08:04:54.179637Z","iopub.execute_input":"2022-07-26T08:04:54.180006Z","iopub.status.idle":"2022-07-26T08:04:56.602936Z","shell.execute_reply.started":"2022-07-26T08:04:54.179960Z","shell.execute_reply":"2022-07-26T08:04:56.601839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_data = pd.concat([train, test]) # 훈련 데이터와 테스트 데이터 합치기 \nall_data = all_data.drop('target', axis=1) # 타깃값 제거\n","metadata":{"execution":{"iopub.status.busy":"2022-07-26T08:09:21.788946Z","iopub.execute_input":"2022-07-26T08:09:21.789239Z","iopub.status.idle":"2022-07-26T08:09:22.394128Z","shell.execute_reply.started":"2022-07-26T08:09:21.789209Z","shell.execute_reply":"2022-07-26T08:09:22.393176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample = all_data[:10]\ndisplay(sample.nom_0)\ndisplay(pd.get_dummies(sample.nom_0))\n\nfrom sklearn.preprocessing import OneHotEncoder\nencoder = OneHotEncoder() # 원-핫 인코더 생성\n# fit&transform 한꺼번에\nprint(encoder.fit_transform(sample[['nom_0']]).toarray())\nall_data_encoded = encoder.fit_transform(all_data) # 원-핫 인코딩 적용","metadata":{"execution":{"iopub.status.busy":"2022-07-26T08:09:22.396137Z","iopub.execute_input":"2022-07-26T08:09:22.396500Z","iopub.status.idle":"2022-07-26T08:09:27.181475Z","shell.execute_reply.started":"2022-07-26T08:09:22.396453Z","shell.execute_reply":"2022-07-26T08:09:27.180480Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_data_encoded\nencoder.transform(all_data)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T06:01:23.431448Z","iopub.execute_input":"2022-07-26T06:01:23.431732Z","iopub.status.idle":"2022-07-26T06:01:25.668585Z","shell.execute_reply.started":"2022-07-26T06:01:23.431699Z","shell.execute_reply":"2022-07-26T06:01:25.667731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#데이터 나누기\nnum_train = len(train) # 훈련 데이터 개수\n\n# 훈련 데이터와 테스트 데이터 나누기\nX_train = all_data_encoded[:num_train] # 0 ~ num_train - 1행\nX_test = all_data_encoded[num_train:] # num_train ~ 마지막 행\n\ny = train['target']\n\nfrom sklearn.model_selection import train_test_split\n\n# 훈련 데이터, 검증 데이터 분리\nX_train, X_valid, y_train, y_valid = train_test_split(X_train, y,\n                                                      test_size=0.1,\n                                                      stratify=y,\n                                                      random_state=10)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T08:21:07.878694Z","iopub.execute_input":"2022-07-26T08:21:07.879052Z","iopub.status.idle":"2022-07-26T08:21:08.643528Z","shell.execute_reply.started":"2022-07-26T08:21:07.879015Z","shell.execute_reply":"2022-07-26T08:21:08.642704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 7.3.2 모델훈련","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\n\nlogistic_model = LogisticRegression(max_iter=1000, random_state=42) # 모델 생성\nlogistic_model.fit(X_train, y_train) # 모델 훈련","metadata":{"execution":{"iopub.status.busy":"2022-07-26T08:21:15.419220Z","iopub.execute_input":"2022-07-26T08:21:15.419567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"logistic_model.predict_proba(X_valid)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-25T07:07:37.312598Z","iopub.execute_input":"2022-07-25T07:07:37.313190Z","iopub.status.idle":"2022-07-25T07:07:37.324568Z","shell.execute_reply.started":"2022-07-25T07:07:37.313131Z","shell.execute_reply":"2022-07-25T07:07:37.323703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"logistic_model.predict(X_valid)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-25T07:07:37.326239Z","iopub.execute_input":"2022-07-25T07:07:37.326868Z","iopub.status.idle":"2022-07-25T07:07:37.341398Z","shell.execute_reply.started":"2022-07-25T07:07:37.326792Z","shell.execute_reply":"2022-07-25T07:07:37.340039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 검증 데이터를 활용한 타깃 예측 \ny_valid_preds = logistic_model.predict_proba(X_valid)[:, 1]\n\nfrom sklearn.metrics import roc_auc_score # ROC AUC 점수 계산 함수\n\n# 검증 데이터 ROC AUC\nroc_auc = roc_auc_score(y_valid, y_valid_preds)\n\nprint(f'검증 데이터 ROC AUC : {roc_auc:.4f}')","metadata":{"execution":{"iopub.status.busy":"2022-07-25T07:07:45.503019Z","iopub.execute_input":"2022-07-25T07:07:45.503766Z","iopub.status.idle":"2022-07-25T07:07:45.524033Z","shell.execute_reply.started":"2022-07-25T07:07:45.503728Z","shell.execute_reply":"2022-07-25T07:07:45.522734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 타깃값 1일 확률 예측\ny_preds = logistic_model.predict_proba(X_test)[:, 1]\n\n# 제출 파일 생성\nsubmission['target'] = y_preds\nsubmission.to_csv('submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-25T07:07:53.607728Z","iopub.execute_input":"2022-07-25T07:07:53.608057Z","iopub.status.idle":"2022-07-25T07:07:54.465384Z","shell.execute_reply.started":"2022-07-25T07:07:53.608023Z","shell.execute_reply":"2022-07-25T07:07:54.464449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 7.5 범주형 데이터 이진분류 경진대회 모델 성능 개선 II","metadata":{}},{"cell_type":"code","source":"#이진 피처 인코딩\nall_data['bin_3'] = all_data['bin_3'].map({'F':0, 'T':1})\nall_data['bin_4'] = all_data['bin_4'].map({'N':0, 'Y':1})\n#순서형 피처 인코딩\nord1dict = {'Novice':0, 'Contributor':1, \n            'Expert':2, 'Master':3, 'Grandmaster':4}\nord2dict = {'Freezing':0, 'Cold':1, 'Warm':2, \n            'Hot':3, 'Boiling Hot':4, 'Lava Hot':5}\n\nall_data['ord_1'] = all_data['ord_1'].map(ord1dict)\nall_data['ord_2'] = all_data['ord_2'].map(ord2dict)\n\n\nfrom sklearn.preprocessing import OrdinalEncoder\n\nord_345 = ['ord_3', 'ord_4', 'ord_5']\n\nord_encoder = OrdinalEncoder() # OrdinalEncoder 객체 생성\n# ordinal 인코딩 적용\nall_data[ord_345] = ord_encoder.fit_transform(all_data[ord_345])\n\n# 피처별 인코딩 순서 출력\nfor feature, categories in zip(ord_345, ord_encoder.categories_):\n    print(feature)\n    print(categories)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T07:09:19.928346Z","iopub.execute_input":"2022-07-25T07:09:19.928663Z","iopub.status.idle":"2022-07-25T07:09:21.582375Z","shell.execute_reply.started":"2022-07-25T07:09:19.928630Z","shell.execute_reply":"2022-07-25T07:09:21.581488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#명목\nnom_features = ['nom_' + str(i) for i in range(10)] # 명목형 피처\n\n\nfrom sklearn.preprocessing import OneHotEncoder\n\nonehot_encoder = OneHotEncoder() # OneHotEncoder 객체 생성\n# 원-핫 인코딩 적용\nencoded_nom_matrix = onehot_encoder.fit_transform(all_data[nom_features])\n\nencoded_nom_matrix","metadata":{"execution":{"iopub.status.busy":"2022-07-25T07:20:34.873015Z","iopub.execute_input":"2022-07-25T07:20:34.874447Z","iopub.status.idle":"2022-07-25T07:20:37.715993Z","shell.execute_reply.started":"2022-07-25T07:20:34.874395Z","shell.execute_reply":"2022-07-25T07:20:37.715128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_data = all_data.drop(nom_features, axis=1) # 기존 명목형 피처 삭제\n","metadata":{"execution":{"iopub.status.busy":"2022-07-25T07:20:37.717647Z","iopub.execute_input":"2022-07-25T07:20:37.717903Z","iopub.status.idle":"2022-07-25T07:20:37.742641Z","shell.execute_reply.started":"2022-07-25T07:20:37.717872Z","shell.execute_reply":"2022-07-25T07:20:37.741599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#날짜 피처 인코딩\ndate_features  = ['day', 'month'] # 날짜 피처\n\n# 원-핫 인코딩 적용\nencoded_date_matrix = onehot_encoder.fit_transform(all_data[date_features])\n\nall_data = all_data.drop(date_features, axis=1) # 기존 날짜 피처 삭제\n\nencoded_date_matrix","metadata":{"execution":{"iopub.status.busy":"2022-07-25T07:20:44.039922Z","iopub.execute_input":"2022-07-25T07:20:44.042758Z","iopub.status.idle":"2022-07-25T07:20:44.203058Z","shell.execute_reply.started":"2022-07-25T07:20:44.042706Z","shell.execute_reply":"2022-07-25T07:20:44.202236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#피처 엔지니어링 II : 피처 스케일링\n#순서형 피처 스케일링\nfrom sklearn.preprocessing import MinMaxScaler\n\nord_features = ['ord_' + str(i) for i in range(6)] # 순서형 피처\n# min-max 정규화\nall_data[ord_features] = MinMaxScaler().fit_transform(all_data[ord_features])\n#인코딩 및 스케일링된 피처 합치기\n\nall_data","metadata":{"execution":{"iopub.status.busy":"2022-07-25T07:34:55.361929Z","iopub.execute_input":"2022-07-25T07:34:55.362275Z","iopub.status.idle":"2022-07-25T07:34:55.435108Z","shell.execute_reply.started":"2022-07-25T07:34:55.362242Z","shell.execute_reply":"2022-07-25T07:34:55.433817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy import sparse\n\n# 인코딩 및 스케일링된 피처 합치기\nall_data_sprs = sparse.hstack([sparse.csr_matrix(all_data),\n                               encoded_nom_matrix,\n                               encoded_date_matrix],\n                              format='csr')\n#데이터 나누기\nnum_train = len(train) # 훈련 데이터 개수\n\n# 훈련 데이터와 테스트 데이터 나누기\nX_train = all_data_sprs[:num_train] # 0 ~ num_train - 1행\nX_test = all_data_sprs[num_train:] # num_train ~ 마지막 행\n\ny = train['target']","metadata":{"execution":{"iopub.status.busy":"2022-07-25T08:04:26.707819Z","iopub.execute_input":"2022-07-25T08:04:26.708602Z","iopub.status.idle":"2022-07-25T08:04:27.800395Z","shell.execute_reply.started":"2022-07-25T08:04:26.708544Z","shell.execute_reply":"2022-07-25T08:04:27.799347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#하이퍼 파라미터 최적화\n\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.linear_model import LogisticRegression\n\n# 로지스틱 회귀 모델 생성\nlogistic_model = LogisticRegression()\n\n# 하이퍼파라미터 값 목록\nlr_params = {'C':[0.1, 0.125, 0.2], 'max_iter':[800, 900, 1000], \n             'solver':['liblinear'], 'random_state':[42]}\n\n# 그리드서치 객체 생성\ngridsearch_logistic_model = GridSearchCV(estimator=logistic_model,\n                                         param_grid=lr_params,\n                                         scoring='roc_auc', # 평가지표\n                                         cv=5)\n# 그리드서치 수행\ngridsearch_logistic_model.fit(X_train, y)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T08:11:21.211575Z","iopub.execute_input":"2022-07-25T08:11:21.212964Z","iopub.status.idle":"2022-07-25T08:20:35.813294Z","shell.execute_reply.started":"2022-07-25T08:11:21.212905Z","shell.execute_reply":"2022-07-25T08:20:35.812316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\nprint('최적 하이퍼파라미터:', gridsearch_logistic_model.best_params_)\n#최적 하이퍼파라미터: {'C': 0.125, 'max_iter': 800, 'random_state': 42, 'solver': 'liblinear'}\n#예측 및 결과 제출\n# 타깃값 1일 확률 예측\ny_preds = gridsearch_logistic_model.best_estimator_.predict_proba(X_test)[:,1]\n\n# 제출 파일 생성\nsubmission['target'] = y_preds\nsubmission.to_csv('submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-25T07:22:41.441601Z","iopub.execute_input":"2022-07-25T07:22:41.441973Z","iopub.status.idle":"2022-07-25T07:24:09.774379Z","shell.execute_reply.started":"2022-07-25T07:22:41.441923Z","shell.execute_reply":"2022-07-25T07:24:09.772655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission","metadata":{"execution":{"iopub.status.busy":"2022-07-25T07:24:09.775646Z","iopub.status.idle":"2022-07-25T07:24:09.777446Z","shell.execute_reply.started":"2022-07-25T07:24:09.776754Z","shell.execute_reply":"2022-07-25T07:24:09.776804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}