{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-29T11:15:44.454034Z","iopub.execute_input":"2022-07-29T11:15:44.454493Z","iopub.status.idle":"2022-07-29T11:15:44.464706Z","shell.execute_reply.started":"2022-07-29T11:15:44.454456Z","shell.execute_reply":"2022-07-29T11:15:44.463140Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\ndata_path = '/kaggle/input/cat-in-the-dat/'\n\ntrain = pd.read_csv(data_path + 'train.csv', index_col='id')\ntest = pd.read_csv(data_path + 'test.csv', index_col='id')\nsubmission = pd.read_csv(data_path + 'sample_submission.csv', index_col='id')","metadata":{"execution":{"iopub.status.busy":"2022-07-29T11:15:44.467968Z","iopub.execute_input":"2022-07-29T11:15:44.468857Z","iopub.status.idle":"2022-07-29T11:15:47.040991Z","shell.execute_reply.started":"2022-07-29T11:15:44.468805Z","shell.execute_reply":"2022-07-29T11:15:47.039879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape, test.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-29T11:15:47.042344Z","iopub.execute_input":"2022-07-29T11:15:47.042663Z","iopub.status.idle":"2022-07-29T11:15:47.052356Z","shell.execute_reply.started":"2022-07-29T11:15:47.042634Z","shell.execute_reply":"2022-07-29T11:15:47.051279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head().T","metadata":{"execution":{"iopub.status.busy":"2022-07-29T11:15:47.054662Z","iopub.execute_input":"2022-07-29T11:15:47.055370Z","iopub.status.idle":"2022-07-29T11:15:47.082616Z","shell.execute_reply.started":"2022-07-29T11:15:47.055332Z","shell.execute_reply":"2022-07-29T11:15:47.081453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def resumetable(df):\n    print(f'데이터셋 형상: {df.shape}')\n    summary = pd.DataFrame(df.dtypes, columns = ['데이터 타입'])\n    summary = summary.reset_index()\n    summary = summary.rename(columns={'index':'피처'})\n    summary['결측값 개수'] = df.isnull().sum().values\n    summary['고윳값 개수'] = df.nunique().values\n    summary['첫 번째 값'] = df.loc[0].values\n    summary['두 번째 값'] = df.loc[1].values\n    summary['세 번째 값'] = df.loc[2].values\n    \n    return summary\n\nresumetable(train)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T11:15:47.083777Z","iopub.execute_input":"2022-07-29T11:15:47.084107Z","iopub.status.idle":"2022-07-29T11:15:48.077888Z","shell.execute_reply.started":"2022-07-29T11:15:47.084071Z","shell.execute_reply":"2022-07-29T11:15:48.076809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(3):\n    feature = 'ord_' + str(i)\n    print(f'{feature} 고윳값: {train[feature].unique()}')","metadata":{"execution":{"iopub.status.busy":"2022-07-29T11:15:48.079237Z","iopub.execute_input":"2022-07-29T11:15:48.079637Z","iopub.status.idle":"2022-07-29T11:15:48.140984Z","shell.execute_reply.started":"2022-07-29T11:15:48.079607Z","shell.execute_reply":"2022-07-29T11:15:48.139854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(3, 6):\n    feature = 'ord_' + str(i)\n    print(f'{feature} 고윳값: {train[feature].unique()}')","metadata":{"execution":{"iopub.status.busy":"2022-07-29T11:15:48.142460Z","iopub.execute_input":"2022-07-29T11:15:48.142806Z","iopub.status.idle":"2022-07-29T11:15:48.219239Z","shell.execute_reply.started":"2022-07-29T11:15:48.142775Z","shell.execute_reply":"2022-07-29T11:15:48.218068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('day 고윳값:', train['day'].unique())\nprint('month 고윳값:', train['month'].unique())\nprint('target 고윳값:', train['target'].unique())","metadata":{"execution":{"iopub.status.busy":"2022-07-29T11:15:48.220718Z","iopub.execute_input":"2022-07-29T11:15:48.221201Z","iopub.status.idle":"2022-07-29T11:15:48.235828Z","shell.execute_reply.started":"2022-07-29T11:15:48.221158Z","shell.execute_reply":"2022-07-29T11:15:48.234362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib as mpl\nimport matplotlib.pyplot as plt\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2022-07-29T11:15:48.239159Z","iopub.execute_input":"2022-07-29T11:15:48.239619Z","iopub.status.idle":"2022-07-29T11:15:48.870265Z","shell.execute_reply.started":"2022-07-29T11:15:48.239578Z","shell.execute_reply":"2022-07-29T11:15:48.869244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mpl.rc('font', size=15)\nplt.figure(figsize=(7,6))\n\nax=sns.countplot(x='target', data=train)\nax.set_title('Target Distribution')","metadata":{"execution":{"iopub.status.busy":"2022-07-29T11:15:48.871206Z","iopub.execute_input":"2022-07-29T11:15:48.871470Z","iopub.status.idle":"2022-07-29T11:15:49.098746Z","shell.execute_reply.started":"2022-07-29T11:15:48.871445Z","shell.execute_reply":"2022-07-29T11:15:49.097749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rectangle = ax.patches[0]  #첫 번째 Rectangle 객체\nprint('사각형 높이:', rectangle.get_height())\nprint('사각형 너비:', rectangle.get_width())\nprint('사각형 왼쪽 테두리의 x축 위치:', rectangle.get_x())","metadata":{"execution":{"iopub.status.busy":"2022-07-29T11:15:49.099980Z","iopub.execute_input":"2022-07-29T11:15:49.100465Z","iopub.status.idle":"2022-07-29T11:15:49.106028Z","shell.execute_reply.started":"2022-07-29T11:15:49.100434Z","shell.execute_reply":"2022-07-29T11:15:49.105069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def write_percent(ax, total_size):\n    for patch in ax.patches:\n        height = patch.get_height()\n        width = patch.get_width()\n        left_coord = patch.get_x()  #도형 왼쪽 테두리의 x축 위치\n        percent = height/total_size*100  #타깃값 비율\n        \n        # (x, y) 좌표에 텍스트 입력\n        ax.text(x = left_coord + width/2.0,\n               y = height + total_size*0.001,\n               s = f'{percent:1.1f}%',\n               ha = 'center')\n        \nplt.figure(figsize=(7,6))\n\nax = sns.countplot(x='target', data=train)\nwrite_percent(ax, len(train))\nax.set_title('Target Distribution')","metadata":{"execution":{"iopub.status.busy":"2022-07-29T11:15:49.107329Z","iopub.execute_input":"2022-07-29T11:15:49.107788Z","iopub.status.idle":"2022-07-29T11:15:49.313176Z","shell.execute_reply.started":"2022-07-29T11:15:49.107759Z","shell.execute_reply":"2022-07-29T11:15:49.312063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.gridspec as gridspec\n# 3행 2열 틀(Figure) 준비\nmpl.rc('font', size=12)\ngrid = gridspec.GridSpec(3,2)  #그래프(서브플롯)를 3행 2열로 배치\nplt.figure(figsize=(10,16))    #전체 Figure 크기 설정\nplt.subplots_adjust(wspace=0.4, hspace=0.3) #서브플롯 간 좌우/상하 여백 설정\n\n#서브플롯 그리기\nbin_features = ['bin_0', 'bin_1', 'bin_2', 'bin_3', 'bin_4']\n\nfor idx, feature in enumerate(bin_features):\n    ax = plt.subplot(grid[idx])\n    \n    #ax축에 타깃값 분포 카운트플롯 그리기\n    sns.countplot(x=feature, data=train, hue='target', #타깃값으로 구분하기\n                  palette='pastel', ax=ax)\n    ax.set_title(f'{feature} Distribution by Target')\n    write_percent(ax, len(train))","metadata":{"execution":{"iopub.status.busy":"2022-07-29T11:15:49.314697Z","iopub.execute_input":"2022-07-29T11:15:49.315590Z","iopub.status.idle":"2022-07-29T11:15:51.017859Z","shell.execute_reply.started":"2022-07-29T11:15:49.315557Z","shell.execute_reply":"2022-07-29T11:15:51.017066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#교차분석표 만들기\npd.crosstab(train['nom_0'], train['target'])\n\n#index를 기준으로 정규화 후 비율을 백분율로 표현하기\ncrosstab = pd.crosstab(train['nom_0'], train['target'], normalize='index')*100\n\n#index를 열로 가져오기\ncrosstab = crosstab.reset_index()\ncrosstab","metadata":{"execution":{"iopub.status.busy":"2022-07-29T11:15:51.018728Z","iopub.execute_input":"2022-07-29T11:15:51.019007Z","iopub.status.idle":"2022-07-29T11:15:51.189217Z","shell.execute_reply.started":"2022-07-29T11:15:51.018981Z","shell.execute_reply":"2022-07-29T11:15:51.188075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_crosstab(df, feature):\n    crosstab = pd.crosstab(df[feature], df['target'], normalize='index')*100\n    crosstab = crosstab.reset_index()\n    return crosstab","metadata":{"execution":{"iopub.status.busy":"2022-07-29T11:15:51.190395Z","iopub.execute_input":"2022-07-29T11:15:51.191061Z","iopub.status.idle":"2022-07-29T11:15:51.195993Z","shell.execute_reply.started":"2022-07-29T11:15:51.191029Z","shell.execute_reply":"2022-07-29T11:15:51.195174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"crosstab = get_crosstab(train, 'nom_0')\ncrosstab","metadata":{"execution":{"iopub.status.busy":"2022-07-29T11:15:51.196992Z","iopub.execute_input":"2022-07-29T11:15:51.198059Z","iopub.status.idle":"2022-07-29T11:15:51.289827Z","shell.execute_reply.started":"2022-07-29T11:15:51.197980Z","shell.execute_reply":"2022-07-29T11:15:51.288749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_pointplot(ax, feature, crosstab):\n    ax2 = ax.twinx()  #x축은 공유하고 y축은 공유하지 않는 새로운 축 생성\n    #새로운 축에 포인트플롯 그리기\n    ax2 = sns.pointplot(x=feature, y=1, data=crosstab,\n                       order=crosstab[feature].values, color='black', legend=False)\n    ax2.set_ylim(crosstab[1].min()-5, crosstab[1].max()*1.1)\n    ax2.set_ylabel('Target 1 Ratio(%)')","metadata":{"execution":{"iopub.status.busy":"2022-07-29T11:15:51.291442Z","iopub.execute_input":"2022-07-29T11:15:51.292399Z","iopub.status.idle":"2022-07-29T11:15:51.300376Z","shell.execute_reply.started":"2022-07-29T11:15:51.292352Z","shell.execute_reply":"2022-07-29T11:15:51.299340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_cat_dist_with_true_ratio(df, features, num_rows, num_cols,\n                                 size=(15,20)):\n    plt.figure(figsize=size)\n    grid = gridspec.GridSpec(num_rows, num_cols)\n    plt.subplots_adjust(wspace=0.45, hspace=0.3)\n    \n    for idx, feature in enumerate(features):\n        ax = plt.subplot(grid[idx])\n        crosstab = get_crosstab(df, feature)  #교차분석표 생성\n        \n        #ax축에 타깃값 분포 카운트플롯 그리기\n        sns.countplot(x=feature, data=df,\n                     order=crosstab[feature].values,\n                     color='skyblue', ax=ax)\n        write_percent(ax, len(df))\n        plot_pointplot(ax, feature, crosstab)\n        ax.set_title(f'{feature} Distribution')","metadata":{"execution":{"iopub.status.busy":"2022-07-29T11:24:26.930301Z","iopub.execute_input":"2022-07-29T11:24:26.930689Z","iopub.status.idle":"2022-07-29T11:24:26.940062Z","shell.execute_reply.started":"2022-07-29T11:24:26.930656Z","shell.execute_reply":"2022-07-29T11:24:26.938799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nom_features = ['nom_0','nom_1','nom_2','nom_3','nom_4']\nplot_cat_dist_with_true_ratio(train, nom_features, num_rows=3, num_cols=2)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T11:24:54.366449Z","iopub.execute_input":"2022-07-29T11:24:54.366819Z","iopub.status.idle":"2022-07-29T11:24:57.236390Z","shell.execute_reply.started":"2022-07-29T11:24:54.366788Z","shell.execute_reply":"2022-07-29T11:24:57.235633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ord_features = ['ord_0', 'ord_1', 'ord_2', 'ord_3']\nplot_cat_dist_with_true_ratio(train, ord_features, num_rows=2, num_cols=2, size=(15,12))","metadata":{"execution":{"iopub.status.busy":"2022-07-29T11:29:44.078527Z","iopub.execute_input":"2022-07-29T11:29:44.079268Z","iopub.status.idle":"2022-07-29T11:29:46.437955Z","shell.execute_reply.started":"2022-07-29T11:29:44.079235Z","shell.execute_reply":"2022-07-29T11:29:46.436957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pandas.api.types import CategoricalDtype\n\nord_1_value = ['Novice', 'Contributor', 'Expert', 'Master', 'Grandmaster']\nord_2_value = ['Freezing', 'Cold', 'Warm', 'Hot', 'Boiling Hot', 'Lava Hot']\n\n#순서를 지정할 범주형 데이터 타입\nord_1_dtype = CategoricalDtype(categories=ord_1_value, ordered=True)\nord_2_dtype = CategoricalDtype(categories=ord_2_value, ordered=True)\n\n#데이터 타입 변경\ntrain['ord_1'] = train['ord_1'].astype(ord_1_dtype)\ntrain['ord_2'] = train['ord_2'].astype(ord_2_dtype)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T11:39:23.808106Z","iopub.execute_input":"2022-07-29T11:39:23.808528Z","iopub.status.idle":"2022-07-29T11:39:24.031068Z","shell.execute_reply.started":"2022-07-29T11:39:23.808500Z","shell.execute_reply":"2022-07-29T11:39:24.030174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_cat_dist_with_true_ratio(train, ord_features, num_rows=2, num_cols=2,\n                             size=(15,12))","metadata":{"execution":{"iopub.status.busy":"2022-07-29T11:40:05.726918Z","iopub.execute_input":"2022-07-29T11:40:05.727313Z","iopub.status.idle":"2022-07-29T11:40:07.325886Z","shell.execute_reply.started":"2022-07-29T11:40:05.727281Z","shell.execute_reply":"2022-07-29T11:40:07.324788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"date_features = ['day', 'month']\nplot_cat_dist_with_true_ratio(train, date_features, num_rows=2,\n                             num_cols=1, size=(10,10))","metadata":{"execution":{"iopub.status.busy":"2022-07-29T11:46:07.506556Z","iopub.execute_input":"2022-07-29T11:46:07.506954Z","iopub.status.idle":"2022-07-29T11:46:08.286794Z","shell.execute_reply.started":"2022-07-29T11:46:07.506919Z","shell.execute_reply":"2022-07-29T11:46:08.285697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\ndata_path='/kaggle/input/cat-in-the-dat/'\n\ntrain = pd.read_csv(data_path + 'train.csv', index_col='id')\ntest = pd.read_csv(data_path + 'test.csv', index_col='id')\nsubmission = pd.read_csv(data_path + 'sample_submission.csv', index_col='id')","metadata":{"execution":{"iopub.status.busy":"2022-07-31T06:12:43.823228Z","iopub.execute_input":"2022-07-31T06:12:43.823780Z","iopub.status.idle":"2022-07-31T06:12:47.038893Z","shell.execute_reply.started":"2022-07-31T06:12:43.823664Z","shell.execute_reply":"2022-07-31T06:12:47.037769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_data = pd.concat([train, test]) #훈련 데이터와 테스트 데이터 합치기\nall_data = all_data.drop('target', axis=1) #타깃값 제거하기\nall_data","metadata":{"execution":{"iopub.status.busy":"2022-07-31T06:15:26.691226Z","iopub.execute_input":"2022-07-31T06:15:26.691633Z","iopub.status.idle":"2022-07-31T06:15:27.722601Z","shell.execute_reply.started":"2022-07-31T06:15:26.691601Z","shell.execute_reply":"2022-07-31T06:15:27.721378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import OneHotEncoder\n\nencoder = OneHotEncoder() #원-핫 인코더 생성\nall_data_encoded = encoder.fit_transform(all_data)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T06:18:10.830582Z","iopub.execute_input":"2022-07-31T06:18:10.831088Z","iopub.status.idle":"2022-07-31T06:18:14.918343Z","shell.execute_reply.started":"2022-07-31T06:18:10.831048Z","shell.execute_reply":"2022-07-31T06:18:14.917337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_train = len(train) #훈련 데이터 개수\n\n#훈련 데이터와 테스트 데이터 행을 기준으로 나누기\nX_train = all_data_encoded[:num_train]\nX_test = all_data_encoded[num_train:]\n\ny=train['target']","metadata":{"execution":{"iopub.status.busy":"2022-07-31T06:24:25.727717Z","iopub.execute_input":"2022-07-31T06:24:25.728140Z","iopub.status.idle":"2022-07-31T06:24:25.939108Z","shell.execute_reply.started":"2022-07-31T06:24:25.728097Z","shell.execute_reply":"2022-07-31T06:24:25.937782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX_train, X_valid, y_train, y_valid = train_test_split(X_train, y, test_size=0.1,\n                                                     stratify=y, random_state=10)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T06:24:59.969894Z","iopub.execute_input":"2022-07-31T06:24:59.970338Z","iopub.status.idle":"2022-07-31T06:25:00.673018Z","shell.execute_reply.started":"2022-07-31T06:24:59.970302Z","shell.execute_reply":"2022-07-31T06:25:00.671781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\n\nlogistic_model = LogisticRegression(max_iter=1000, random_state=42)  #모델 생성\nlogistic_model.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T06:30:41.459732Z","iopub.execute_input":"2022-07-31T06:30:41.460168Z","iopub.status.idle":"2022-07-31T06:31:55.259258Z","shell.execute_reply.started":"2022-07-31T06:30:41.460134Z","shell.execute_reply":"2022-07-31T06:31:55.257998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"logistic_model.predict_proba(X_valid)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T06:33:44.502965Z","iopub.execute_input":"2022-07-31T06:33:44.503397Z","iopub.status.idle":"2022-07-31T06:33:44.514456Z","shell.execute_reply.started":"2022-07-31T06:33:44.503364Z","shell.execute_reply":"2022-07-31T06:33:44.512985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#검증 데이터를 활용한 타깃 예측\ny_valid_preds = logistic_model.predict_proba(X_valid)[:,1]","metadata":{"execution":{"iopub.status.busy":"2022-07-31T06:36:02.648159Z","iopub.execute_input":"2022-07-31T06:36:02.648869Z","iopub.status.idle":"2022-07-31T06:36:02.658612Z","shell.execute_reply.started":"2022-07-31T06:36:02.648826Z","shell.execute_reply":"2022-07-31T06:36:02.657187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import roc_auc_score\n\n#검증 데이터 ROC, AUC\nroc_auc = roc_auc_score(y_valid, y_valid_preds)\n\nprint(f'검증 데이터 ROC AUC : {roc_auc:.4f}')","metadata":{"execution":{"iopub.status.busy":"2022-07-31T06:40:27.847227Z","iopub.execute_input":"2022-07-31T06:40:27.847672Z","iopub.status.idle":"2022-07-31T06:40:27.867449Z","shell.execute_reply.started":"2022-07-31T06:40:27.847639Z","shell.execute_reply":"2022-07-31T06:40:27.866007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#타깃값 1일 확률 예측\ny_preds = logistic_model.predict_proba(X_test)[:,1]","metadata":{"execution":{"iopub.status.busy":"2022-07-31T06:42:55.752452Z","iopub.execute_input":"2022-07-31T06:42:55.752852Z","iopub.status.idle":"2022-07-31T06:42:55.773058Z","shell.execute_reply.started":"2022-07-31T06:42:55.752821Z","shell.execute_reply":"2022-07-31T06:42:55.771755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#제출 파일 완성하기\nsubmission['target'] = y_preds\nsubmission.to_csv('submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-31T06:43:29.923603Z","iopub.execute_input":"2022-07-31T06:43:29.924073Z","iopub.status.idle":"2022-07-31T06:43:30.776204Z","shell.execute_reply.started":"2022-07-31T06:43:29.924032Z","shell.execute_reply":"2022-07-31T06:43:30.774727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\ndata_path='/kaggle/input/cat-in-the-dat/'\n\ntrain = pd.read_csv(data_path + 'train.csv', index_col='id')\ntest = pd.read_csv(data_path + 'test.csv', index_col='id')\nsubmission = pd.read_csv(data_path + 'sample_submission.csv', index_col='id')","metadata":{"execution":{"iopub.status.busy":"2022-07-31T07:48:59.840044Z","iopub.execute_input":"2022-07-31T07:48:59.840352Z","iopub.status.idle":"2022-07-31T07:49:01.861359Z","shell.execute_reply.started":"2022-07-31T07:48:59.840329Z","shell.execute_reply":"2022-07-31T07:49:01.860123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_data = pd.concat([train, test])\nall_data = all_data.drop('target', axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T07:49:04.740056Z","iopub.execute_input":"2022-07-31T07:49:04.740395Z","iopub.status.idle":"2022-07-31T07:49:05.244492Z","shell.execute_reply.started":"2022-07-31T07:49:04.740372Z","shell.execute_reply":"2022-07-31T07:49:05.243294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_data['bin_3'] = all_data['bin_3'].map({'F':0, 'T':1})\nall_data['bin_4'] = all_data['bin_4'].map({'N':0, 'Y':1})","metadata":{"execution":{"iopub.status.busy":"2022-07-31T07:53:13.661804Z","iopub.execute_input":"2022-07-31T07:53:13.662101Z","iopub.status.idle":"2022-07-31T07:53:13.917108Z","shell.execute_reply.started":"2022-07-31T07:53:13.662078Z","shell.execute_reply":"2022-07-31T07:53:13.915695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ord1dict = {'Novice':0, 'Contributor':1, 'Expert':2,\n           'Master':3, 'Grandmaster':4}\nord2dict = {'Freezing':0, 'Cold':1, 'Warm':2,\n           'Hot':3, 'Boiling Hot':4, 'Lava Hot':5}\n\nall_data['ord_1'] = all_data['ord_1'].map(ord1dict)\nall_data['ord_2'] = all_data['ord_2'].map(ord2dict)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T07:57:07.944722Z","iopub.execute_input":"2022-07-31T07:57:07.945051Z","iopub.status.idle":"2022-07-31T07:57:08.132169Z","shell.execute_reply.started":"2022-07-31T07:57:07.945028Z","shell.execute_reply":"2022-07-31T07:57:08.131283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import OrdinalEncoder\n\nord_345 = ['ord_3', 'ord_4', 'ord_5']\n\nord_encoder = OrdinalEncoder()\nall_data[ord_345] = ord_encoder.fit_transform(all_data[ord_345])\n\n#피처별 인코딩 순서 출력\nfor feature, categories in zip(ord_345, ord_encoder.categories_):\n    print(feature)\n    print(categories)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T08:01:02.907322Z","iopub.execute_input":"2022-07-31T08:01:02.907685Z","iopub.status.idle":"2022-07-31T08:01:03.439130Z","shell.execute_reply.started":"2022-07-31T08:01:02.907657Z","shell.execute_reply":"2022-07-31T08:01:03.437949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#nom_0부터 nom_9까지 총 10개 피처리스트 만들기\nnom_features = ['nom_' + str(i) for i in range(10)]","metadata":{"execution":{"iopub.status.busy":"2022-07-31T08:10:35.283550Z","iopub.execute_input":"2022-07-31T08:10:35.283921Z","iopub.status.idle":"2022-07-31T08:10:35.288913Z","shell.execute_reply.started":"2022-07-31T08:10:35.283896Z","shell.execute_reply":"2022-07-31T08:10:35.287542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import OneHotEncoder\n\nonehot_encoder = OneHotEncoder()\nencoded_nom_matrix = onehot_encoder.fit_transform(all_data[nom_features])\nencoded_nom_matrix","metadata":{"execution":{"iopub.status.busy":"2022-07-31T08:11:32.632321Z","iopub.execute_input":"2022-07-31T08:11:32.632675Z","iopub.status.idle":"2022-07-31T08:11:34.047474Z","shell.execute_reply.started":"2022-07-31T08:11:32.632648Z","shell.execute_reply":"2022-07-31T08:11:34.046626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_data = all_data.drop(nom_features, axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T08:21:47.372916Z","iopub.execute_input":"2022-07-31T08:21:47.373259Z","iopub.status.idle":"2022-07-31T08:21:47.389831Z","shell.execute_reply.started":"2022-07-31T08:21:47.373229Z","shell.execute_reply":"2022-07-31T08:21:47.388809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"date_features = ['day', 'month']\n\n#원-핫 인코딩 적용\nencoded_date_matrix = onehot_encoder.fit_transform(all_data[date_features])\nall_data = all_data.drop(date_features, axis=1)\n\nencoded_date_matrix","metadata":{"execution":{"iopub.status.busy":"2022-07-31T08:23:49.629826Z","iopub.execute_input":"2022-07-31T08:23:49.630785Z","iopub.status.idle":"2022-07-31T08:23:49.773432Z","shell.execute_reply.started":"2022-07-31T08:23:49.630741Z","shell.execute_reply":"2022-07-31T08:23:49.771667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import MinMaxScaler\n\nord_features = ['ord_' +str(i) for i in range(6)]\nall_data[ord_features] = MinMaxScaler().fit_transform(all_data[ord_features])","metadata":{"execution":{"iopub.status.busy":"2022-07-31T08:27:12.174711Z","iopub.execute_input":"2022-07-31T08:27:12.175046Z","iopub.status.idle":"2022-07-31T08:27:12.255724Z","shell.execute_reply.started":"2022-07-31T08:27:12.175022Z","shell.execute_reply":"2022-07-31T08:27:12.254321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy import sparse\nall_data_sprs = sparse.hstack([sparse.csr_matrix(all_data),\n                              encoded_nom_matrix,\n                              encoded_date_matrix],\n                              format = 'csr')","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:02:10.244671Z","iopub.execute_input":"2022-07-31T09:02:10.245561Z","iopub.status.idle":"2022-07-31T09:02:10.790961Z","shell.execute_reply.started":"2022-07-31T09:02:10.245535Z","shell.execute_reply":"2022-07-31T09:02:10.789817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_train = len(train) #훈련 데이터 개수\n\n#훈련 데이터와 테스트 데이터 행을 기준으로 나누기\nX_train = all_data_sprs[:num_train]\nX_test = all_data_sprs[num_train:]\n\ny=train['target']","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:03:40.535847Z","iopub.execute_input":"2022-07-31T09:03:40.536173Z","iopub.status.idle":"2022-07-31T09:03:40.667585Z","shell.execute_reply.started":"2022-07-31T09:03:40.536150Z","shell.execute_reply":"2022-07-31T09:03:40.666339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX_train, X_valid, y_train, y_valid = train_test_split(X_train, y, test_size=0.1,\n                                                     stratify=y, random_state=10)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:04:42.677182Z","iopub.execute_input":"2022-07-31T09:04:42.677490Z","iopub.status.idle":"2022-07-31T09:04:42.903493Z","shell.execute_reply.started":"2022-07-31T09:04:42.677468Z","shell.execute_reply":"2022-07-31T09:04:42.902764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time  \n\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.linear_model import LogisticRegression\n\n#로지스틱 회귀 모델 생성\nlogistic_model = LogisticRegression()\n\n#하이퍼파라미터 값 목록\nlr_params = {'C':[0.1, 0.125, 0.2], 'max_iter':[800, 900, 1000],\n            'solver':['liblinear'], 'random_state':[42]}\n\ngridsearch_logistic_model = GridSearchCV(estimator = logistic_model,\n                                        param_grid = lr_params,\n                                        scoring = 'roc_auc', cv=5)\n\ngridsearch_logistic_model.fit(X_train, y_train)\nprint('최적 하이퍼파라미터:', gridsearch_logistic_model.best_params_)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:12:37.454428Z","iopub.execute_input":"2022-07-31T09:12:37.454798Z","iopub.status.idle":"2022-07-31T09:18:48.199818Z","shell.execute_reply.started":"2022-07-31T09:12:37.454768Z","shell.execute_reply":"2022-07-31T09:18:48.198770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_valid_preds = gridsearch_logistic_model.predict_proba(X_valid)[:,1]","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:20:20.201600Z","iopub.execute_input":"2022-07-31T09:20:20.201928Z","iopub.status.idle":"2022-07-31T09:20:20.208788Z","shell.execute_reply.started":"2022-07-31T09:20:20.201906Z","shell.execute_reply":"2022-07-31T09:20:20.208060Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import roc_auc_score\nroc_auc = roc_auc_score(y_valid, y_valid_preds)\nprint(f'검증 데이터 ROC AUC : {roc_auc:.4f}')","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:20:21.677912Z","iopub.execute_input":"2022-07-31T09:20:21.678922Z","iopub.status.idle":"2022-07-31T09:20:21.693510Z","shell.execute_reply.started":"2022-07-31T09:20:21.678863Z","shell.execute_reply":"2022-07-31T09:20:21.692532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#타깃값 1일 확률 예측\ny_preds = gridsearch_logistic_model.best_estimator_.predict_proba(X_test)[:,1]\n\n#제출 파일 생성\nsubmission['target'] = y_preds\nsubmission.to_csv('submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-31T09:21:57.364514Z","iopub.execute_input":"2022-07-31T09:21:57.364850Z","iopub.status.idle":"2022-07-31T09:21:57.798821Z","shell.execute_reply.started":"2022-07-31T09:21:57.364826Z","shell.execute_reply":"2022-07-31T09:21:57.798038Z"},"trusted":true},"execution_count":null,"outputs":[]}]}