{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## 1. EDA","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib as mpl\nimport matplotlib.pyplot as plt\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:06:06.940758Z","iopub.execute_input":"2022-08-03T06:06:06.941380Z","iopub.status.idle":"2022-08-03T06:06:08.330058Z","shell.execute_reply.started":"2022-08-03T06:06:06.941257Z","shell.execute_reply":"2022-08-03T06:06:08.328941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('../input/cat-in-the-dat/train.csv', index_col = 'id')\ntest = pd.read_csv('../input/cat-in-the-dat/test.csv', index_col = 'id')\nsubmission = pd.read_csv('../input/cat-in-the-dat/sample_submission.csv', index_col = 'id')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:06:08.331763Z","iopub.execute_input":"2022-08-03T06:06:08.332344Z","iopub.status.idle":"2022-08-03T06:06:11.936767Z","shell.execute_reply.started":"2022-08-03T06:06:08.332306Z","shell.execute_reply":"2022-08-03T06:06:11.935447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape, test.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:06:11.938506Z","iopub.execute_input":"2022-08-03T06:06:11.938959Z","iopub.status.idle":"2022-08-03T06:06:11.950867Z","shell.execute_reply.started":"2022-08-03T06:06:11.938916Z","shell.execute_reply":"2022-08-03T06:06:11.949135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head().T","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:06:11.954451Z","iopub.execute_input":"2022-08-03T06:06:11.954822Z","iopub.status.idle":"2022-08-03T06:06:11.977256Z","shell.execute_reply.started":"2022-08-03T06:06:11.954792Z","shell.execute_reply":"2022-08-03T06:06:11.976371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:06:11.978569Z","iopub.execute_input":"2022-08-03T06:06:11.979212Z","iopub.status.idle":"2022-08-03T06:06:11.990978Z","shell.execute_reply.started":"2022-08-03T06:06:11.979178Z","shell.execute_reply":"2022-08-03T06:06:11.989927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# feature summary table\ndef resumetable(df):\n    print(f'데이터형상: {df.shape}')\n    summary = pd.DataFrame(df.dtypes, columns = ['데이터타입'])\n    summary = summary.reset_index()\n    summary = summary.rename(columns = {'index': '피처'})\n    summary['결측값 개수'] = df.isnull().sum().values     \n    summary['고윳값 개수'] = df.nunique().values\n    summary['첫 번째 값'] = df.loc[0].values\n    summary['두 번째 값'] = df.loc[1].values\n    summary['세 번째 값'] = df.loc[2].values\n          \n    return summary\n        \nresumetable(train)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:06:11.992520Z","iopub.execute_input":"2022-08-03T06:06:11.993482Z","iopub.status.idle":"2022-08-03T06:06:12.672360Z","shell.execute_reply.started":"2022-08-03T06:06:11.993445Z","shell.execute_reply":"2022-08-03T06:06:12.671242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# data visualization(target distribution)\nmpl.rc('font', size = 15)\nplt.figure(figsize = (7,6))\n\nax = sns.countplot(x = 'target', data = train)\nax.set_title('Target Distribution')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:06:12.673606Z","iopub.execute_input":"2022-08-03T06:06:12.673930Z","iopub.status.idle":"2022-08-03T06:06:12.889980Z","shell.execute_reply.started":"2022-08-03T06:06:12.673902Z","shell.execute_reply":"2022-08-03T06:06:12.889116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(ax.patches)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:06:12.892097Z","iopub.execute_input":"2022-08-03T06:06:12.893487Z","iopub.status.idle":"2022-08-03T06:06:12.900945Z","shell.execute_reply.started":"2022-08-03T06:06:12.893432Z","shell.execute_reply":"2022-08-03T06:06:12.899471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rectangle = ax.patches[0]\nprint('사각형 높이:', rectangle.get_height())\nprint('사각형 너비:', rectangle.get_width())\nprint('사각형 왼쪽 테두리의 x축 위치:', rectangle.get_x())","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:06:12.902866Z","iopub.execute_input":"2022-08-03T06:06:12.904067Z","iopub.status.idle":"2022-08-03T06:06:12.914586Z","shell.execute_reply.started":"2022-08-03T06:06:12.903981Z","shell.execute_reply":"2022-08-03T06:06:12.913127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def write_percent(ax, total_size):\n    for patch in ax.patches:\n        height = patch.get_height()\n        width = patch.get_width()\n        left_coord = patch.get_x()\n        percent = height / total_size * 100\n        \n        ax.text(x = left_coord + width / 2,\n                y = height + total_size * 0.001,\n                s = f'{percent: 1.1f}%',\n                ha = 'center')\n        \nplt.figure(figsize = (7,6))\n\nax = sns.countplot(x = 'target', data = train)\nwrite_percent(ax, len(train))\nax.set_title('Target Distribution')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:06:12.917982Z","iopub.execute_input":"2022-08-03T06:06:12.919299Z","iopub.status.idle":"2022-08-03T06:06:13.162728Z","shell.execute_reply.started":"2022-08-03T06:06:12.919253Z","shell.execute_reply":"2022-08-03T06:06:13.161265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# data visualization(binary feature distribution)\nimport matplotlib.gridspec as gridspec\nmpl.rc('font', size = 12)\ngrid = gridspec.GridSpec(3,2)\nplt.figure(figsize = (10,16))\nplt.subplots_adjust(wspace = 0.4, hspace = 0.3)\n\nbin_features = ['bin_0', 'bin_1', 'bin_2', 'bin_3', 'bin_4']\n\nfor idx, feature in enumerate(bin_features):\n    ax = plt.subplot(grid[idx])\n    \n    sns.countplot(x = feature, data = train, hue = 'target', palette = 'pastel', ax = ax)\n    ax.set_title(f'{feature} Distribution by Target')\n    write_percent(ax, len(train))","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:06:13.164721Z","iopub.execute_input":"2022-08-03T06:06:13.166057Z","iopub.status.idle":"2022-08-03T06:06:14.631643Z","shell.execute_reply.started":"2022-08-03T06:06:13.165985Z","shell.execute_reply":"2022-08-03T06:06:14.630272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# data visualization(nominal feature distribution)\npd.crosstab(train['nom_0'], train['target'])\ncrosstab = pd.crosstab(train['nom_0'], train['target'], normalize = 'index') * 100\ncrosstab","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:06:14.632997Z","iopub.execute_input":"2022-08-03T06:06:14.633455Z","iopub.status.idle":"2022-08-03T06:06:14.780864Z","shell.execute_reply.started":"2022-08-03T06:06:14.633406Z","shell.execute_reply":"2022-08-03T06:06:14.779718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"crosstab = crosstab.reset_index()\ncrosstab","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:06:14.782228Z","iopub.execute_input":"2022-08-03T06:06:14.782610Z","iopub.status.idle":"2022-08-03T06:06:14.797863Z","shell.execute_reply.started":"2022-08-03T06:06:14.782578Z","shell.execute_reply":"2022-08-03T06:06:14.796115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_crosstab(df, feature):\n    crosstab = pd.crosstab(df[feature], df['target'], normalize = 'index') * 100\n    crosstab = crosstab.reset_index()\n    return crosstab","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:06:14.800295Z","iopub.execute_input":"2022-08-03T06:06:14.801006Z","iopub.status.idle":"2022-08-03T06:06:14.809169Z","shell.execute_reply.started":"2022-08-03T06:06:14.800945Z","shell.execute_reply":"2022-08-03T06:06:14.807772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"crosstab = get_crosstab(train, 'nom_0')\ncrosstab\ncrosstab[1]","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:06:14.810666Z","iopub.execute_input":"2022-08-03T06:06:14.811206Z","iopub.status.idle":"2022-08-03T06:06:14.889339Z","shell.execute_reply.started":"2022-08-03T06:06:14.811161Z","shell.execute_reply":"2022-08-03T06:06:14.888123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_pointplot(ax, feature, crosstab):\n    ax2 = ax.twinx()\n    ax2 = sns.pointplot(x = feature, y = 1, data = crosstab, order = crosstab[feature].values,\n                        color = 'black', legend = False)\n    ax2.set_ylim(crosstab[1].min() - 5, crosstab[1].max() * 1.1)\n    ax2.set_ylabel('Target 1 Ratio(%)')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:06:14.890805Z","iopub.execute_input":"2022-08-03T06:06:14.891192Z","iopub.status.idle":"2022-08-03T06:06:14.898541Z","shell.execute_reply.started":"2022-08-03T06:06:14.891159Z","shell.execute_reply":"2022-08-03T06:06:14.897539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_cat_dist_with_true_ratio(df, features, num_rows, num_cols, size = (15, 20)) :\n    plt.figure(figsize = size)\n    grid = gridspec.GridSpec(num_rows, num_cols)\n    plt.subplots_adjust(wspace = 0.45, hspace = 0.3)\n    \n    for idx, feature in enumerate(features):\n        ax = plt.subplot(grid[idx])\n        crosstab = get_crosstab(df, feature)\n        \n        sns.countplot(x = feature, data = df, order = crosstab[feature].values, color = 'skyblue', ax = ax)\n        write_percent(ax, len(df))\n        plot_pointplot(ax, feature, crosstab)\n        ax.set_title(f'{feature} Distribution')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:06:14.899687Z","iopub.execute_input":"2022-08-03T06:06:14.900063Z","iopub.status.idle":"2022-08-03T06:06:14.910832Z","shell.execute_reply.started":"2022-08-03T06:06:14.900000Z","shell.execute_reply":"2022-08-03T06:06:14.909914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nom_features = ['nom_0','nom_1','nom_2','nom_3','nom_4']\nplot_cat_dist_with_true_ratio(train, nom_features, num_rows = 3, num_cols = 2)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:06:14.912912Z","iopub.execute_input":"2022-08-03T06:06:14.913410Z","iopub.status.idle":"2022-08-03T06:06:17.209089Z","shell.execute_reply.started":"2022-08-03T06:06:14.913367Z","shell.execute_reply":"2022-08-03T06:06:17.207596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# data visualization(ordinary feature distribution)\nord_features = ['ord_0','ord_1','ord_2','ord_3']\nplot_cat_dist_with_true_ratio(train, ord_features, num_rows = 2, num_cols = 2, size = (15,12))","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:06:17.210763Z","iopub.execute_input":"2022-08-03T06:06:17.213335Z","iopub.status.idle":"2022-08-03T06:06:19.239285Z","shell.execute_reply.started":"2022-08-03T06:06:17.213293Z","shell.execute_reply":"2022-08-03T06:06:19.238319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pandas.api.types import CategoricalDtype\nord_1_value = ['Novice', 'Contributor', 'Expert', 'Master', 'Grandmaster']\nord_2_value = ['Freezing', 'Cold', 'Warm', 'Hot', 'Boiling Hot', 'Lave Hot']\n\nord_1_dtype = CategoricalDtype(categories = ord_1_value, ordered = True)\nord_2_dtype = CategoricalDtype(categories = ord_2_value, ordered = True)\n\ntrain['ord_1'] = train['ord_1'].astype(ord_1_dtype)\ntrain['ord_2'] = train['ord_2'].astype(ord_2_dtype)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:06:19.240399Z","iopub.execute_input":"2022-08-03T06:06:19.241454Z","iopub.status.idle":"2022-08-03T06:06:19.476972Z","shell.execute_reply.started":"2022-08-03T06:06:19.241410Z","shell.execute_reply":"2022-08-03T06:06:19.475889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_cat_dist_with_true_ratio(train, ord_features, num_rows = 2, num_cols = 2, size =(15,12))","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:06:19.478137Z","iopub.execute_input":"2022-08-03T06:06:19.479113Z","iopub.status.idle":"2022-08-03T06:06:21.117718Z","shell.execute_reply.started":"2022-08-03T06:06:19.479077Z","shell.execute_reply":"2022-08-03T06:06:21.116509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_cat_dist_with_true_ratio(train, ['ord_4', 'ord_5'], num_rows = 2, num_cols = 1, size =(15,12))","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:06:21.119344Z","iopub.execute_input":"2022-08-03T06:06:21.120273Z","iopub.status.idle":"2022-08-03T06:06:25.405691Z","shell.execute_reply.started":"2022-08-03T06:06:21.120236Z","shell.execute_reply":"2022-08-03T06:06:25.404538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"date_features = ['day', 'month']\nplot_cat_dist_with_true_ratio(train, date_features, num_rows = 2, num_cols = 1, size =(10,10))","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:06:25.406982Z","iopub.execute_input":"2022-08-03T06:06:25.407319Z","iopub.status.idle":"2022-08-03T06:06:26.217931Z","shell.execute_reply.started":"2022-08-03T06:06:25.407290Z","shell.execute_reply":"2022-08-03T06:06:26.217112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2. Baseline Model","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('../input/cat-in-the-dat/train.csv', index_col = 'id')\ntest = pd.read_csv('../input/cat-in-the-dat/test.csv', index_col = 'id')\nsubmission = pd.read_csv('../input/cat-in-the-dat/sample_submission.csv', index_col = 'id')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:16:30.172344Z","iopub.execute_input":"2022-08-03T06:16:30.172787Z","iopub.status.idle":"2022-08-03T06:16:32.246086Z","shell.execute_reply.started":"2022-08-03T06:16:30.172750Z","shell.execute_reply":"2022-08-03T06:16:32.244874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_data = pd.concat([train, test])\nall_data = all_data.drop('target', axis = 1)\nall_data","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:16:34.345479Z","iopub.execute_input":"2022-08-03T06:16:34.346641Z","iopub.status.idle":"2022-08-03T06:16:35.724847Z","shell.execute_reply.started":"2022-08-03T06:16:34.346593Z","shell.execute_reply":"2022-08-03T06:16:35.723695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import OneHotEncoder\nencoder = OneHotEncoder()\nall_data_encoded = encoder.fit_transform(all_data)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:17:59.804161Z","iopub.execute_input":"2022-08-03T06:17:59.804568Z","iopub.status.idle":"2022-08-03T06:18:03.071782Z","shell.execute_reply.started":"2022-08-03T06:17:59.804528Z","shell.execute_reply":"2022-08-03T06:18:03.070555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_train = len(train)\nX_train = all_data_encoded[:num_train]\nX_test = all_data_encoded[num_train:]\ny = train['target']","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:20:12.058143Z","iopub.execute_input":"2022-08-03T06:20:12.058608Z","iopub.status.idle":"2022-08-03T06:20:12.273560Z","shell.execute_reply.started":"2022-08-03T06:20:12.058572Z","shell.execute_reply":"2022-08-03T06:20:12.272409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_valid, y_train, y_valid = train_test_split(X_train, y, test_size = 0.1, stratify = y, random_state = 10)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:24:11.957785Z","iopub.execute_input":"2022-08-03T06:24:11.958278Z","iopub.status.idle":"2022-08-03T06:24:12.194417Z","shell.execute_reply.started":"2022-08-03T06:24:11.958243Z","shell.execute_reply":"2022-08-03T06:24:12.193239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nlogistic_model = LogisticRegression(max_iter = 1000, random_state = 42)\nlogistic_model.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:26:22.174934Z","iopub.execute_input":"2022-08-03T06:26:22.175831Z","iopub.status.idle":"2022-08-03T06:27:52.970189Z","shell.execute_reply.started":"2022-08-03T06:26:22.175787Z","shell.execute_reply":"2022-08-03T06:27:52.968337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"logistic_model.predict_proba(X_valid)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:28:42.242904Z","iopub.execute_input":"2022-08-03T06:28:42.243368Z","iopub.status.idle":"2022-08-03T06:28:42.255572Z","shell.execute_reply.started":"2022-08-03T06:28:42.243332Z","shell.execute_reply":"2022-08-03T06:28:42.254428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"logistic_model.predict(X_valid)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:28:59.982476Z","iopub.execute_input":"2022-08-03T06:28:59.982933Z","iopub.status.idle":"2022-08-03T06:28:59.992913Z","shell.execute_reply.started":"2022-08-03T06:28:59.982897Z","shell.execute_reply":"2022-08-03T06:28:59.992065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_valid_preds = logistic_model.predict_proba(X_valid)[: ,1]\nfrom sklearn.metrics import roc_auc_score\nroc_auc = roc_auc_score(y_valid, y_valid_preds)\nprint(f'검증 데이터 ROC AUC : {roc_auc:.4f}')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:34:46.722802Z","iopub.execute_input":"2022-08-03T06:34:46.723516Z","iopub.status.idle":"2022-08-03T06:34:46.745428Z","shell.execute_reply.started":"2022-08-03T06:34:46.723444Z","shell.execute_reply":"2022-08-03T06:34:46.744389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_preds = logistic_model.predict_proba(X_test)[:,1]\nsubmission['target'] = y_preds\nsubmission.to_csv('submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:41:30.004018Z","iopub.execute_input":"2022-08-03T06:41:30.004459Z","iopub.status.idle":"2022-08-03T06:41:30.537641Z","shell.execute_reply.started":"2022-08-03T06:41:30.004425Z","shell.execute_reply":"2022-08-03T06:41:30.535872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3. Model Improvment","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('../input/cat-in-the-dat/train.csv', index_col = 'id')\ntest = pd.read_csv('../input/cat-in-the-dat/test.csv', index_col = 'id')\nsubmission = pd.read_csv('../input/cat-in-the-dat/sample_submission.csv', index_col = 'id')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T07:02:39.754204Z","iopub.execute_input":"2022-08-03T07:02:39.754715Z","iopub.status.idle":"2022-08-03T07:02:42.243440Z","shell.execute_reply.started":"2022-08-03T07:02:39.754667Z","shell.execute_reply":"2022-08-03T07:02:42.241814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_data = pd.concat([train, test])\nall_data = all_data.drop('target', axis = 1)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T07:07:57.825226Z","iopub.execute_input":"2022-08-03T07:07:57.825727Z","iopub.status.idle":"2022-08-03T07:07:59.111491Z","shell.execute_reply.started":"2022-08-03T07:07:57.825684Z","shell.execute_reply":"2022-08-03T07:07:59.110563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# bin_3, bin_4 encoding\nall_data['bin_3'] = all_data['bin_3'].map({'F' :0, 'T': 1})\nall_data['bin_4'] = all_data['bin_4'].map({'N' :0, 'Y': 1})","metadata":{"execution":{"iopub.status.busy":"2022-08-03T07:09:20.274764Z","iopub.execute_input":"2022-08-03T07:09:20.275237Z","iopub.status.idle":"2022-08-03T07:09:20.648179Z","shell.execute_reply.started":"2022-08-03T07:09:20.275203Z","shell.execute_reply":"2022-08-03T07:09:20.646881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ord_1~5 encoding\nord1dict = {'Novice' : 0, 'Contributor' : 1, 'Expert' : 2, 'Master' : 3, 'Grandmaster' : 4}\nord2dict = {'Freezing' : 0, 'Cold' : 1, 'Warm' : 2, 'Hot' : 3, 'Boiling Hot' : 4, 'Lava Hot' : 5}\n\nall_data['ord_1'] = all_data['ord_1'].map(ord1dict)\nall_data['ord_2'] = all_data['ord_2'].map(ord2dict)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T07:14:53.862965Z","iopub.execute_input":"2022-08-03T07:14:53.863375Z","iopub.status.idle":"2022-08-03T07:14:54.143516Z","shell.execute_reply.started":"2022-08-03T07:14:53.863343Z","shell.execute_reply":"2022-08-03T07:14:54.142241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import OrdinalEncoder\nord_345 = ['ord_3', 'ord_4', 'ord_5']\nord_encoder = OrdinalEncoder()\nall_data[ord_345] = ord_encoder.fit_transform(all_data[ord_345])\nfor feature, categories in zip(ord_345, ord_encoder.categories_):\n    print(feature)\n    print(categories)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T07:21:04.688802Z","iopub.execute_input":"2022-08-03T07:21:04.689225Z","iopub.status.idle":"2022-08-03T07:21:05.473680Z","shell.execute_reply.started":"2022-08-03T07:21:04.689191Z","shell.execute_reply":"2022-08-03T07:21:05.472471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# nom_ encoding\nnom_features = ['nom_' + str(i) for i in range(10)]\nfrom sklearn.preprocessing import OneHotEncoder\nonehot_encoder = OneHotEncoder()\nencoded_nom_matrix = onehot_encoder.fit_transform(all_data[nom_features])\nencoded_nom_matrix","metadata":{"execution":{"iopub.status.busy":"2022-08-03T07:32:14.111254Z","iopub.execute_input":"2022-08-03T07:32:14.111691Z","iopub.status.idle":"2022-08-03T07:32:15.985238Z","shell.execute_reply.started":"2022-08-03T07:32:14.111657Z","shell.execute_reply":"2022-08-03T07:32:15.983884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_data = all_data.drop(nom_features, axis = 1)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T07:33:21.740179Z","iopub.execute_input":"2022-08-03T07:33:21.741228Z","iopub.status.idle":"2022-08-03T07:33:21.767640Z","shell.execute_reply.started":"2022-08-03T07:33:21.741187Z","shell.execute_reply":"2022-08-03T07:33:21.766131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# date feature encoding\ndate_features = ['day', 'month']\nencoded_date_matrix = onehot_encoder.fit_transform(all_data[date_features])\nall_data = all_data.drop(date_features, axis = 1)\nencoded_date_matrix","metadata":{"execution":{"iopub.status.busy":"2022-08-03T07:34:53.684581Z","iopub.execute_input":"2022-08-03T07:34:53.685012Z","iopub.status.idle":"2022-08-03T07:34:53.849768Z","shell.execute_reply.started":"2022-08-03T07:34:53.684980Z","shell.execute_reply":"2022-08-03T07:34:53.848515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# feature scaling(MinMaxScaler)\nfrom sklearn.preprocessing import MinMaxScaler\nord_features = ['ord_' + str(i) for i in range(6)]\nall_data[ord_features] = MinMaxScaler().fit_transform(all_data[ord_features])","metadata":{"execution":{"iopub.status.busy":"2022-08-03T07:38:59.076695Z","iopub.execute_input":"2022-08-03T07:38:59.077221Z","iopub.status.idle":"2022-08-03T07:38:59.193835Z","shell.execute_reply.started":"2022-08-03T07:38:59.077180Z","shell.execute_reply":"2022-08-03T07:38:59.192796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy import sparse\nall_data_sprs = sparse.hstack([sparse.csr_matrix(all_data),\n                               encoded_nom_matrix,\n                               encoded_date_matrix],\n                              format = 'csr')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T07:43:22.961802Z","iopub.execute_input":"2022-08-03T07:43:22.962366Z","iopub.status.idle":"2022-08-03T07:43:23.678290Z","shell.execute_reply.started":"2022-08-03T07:43:22.962324Z","shell.execute_reply":"2022-08-03T07:43:23.677089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_data_sprs","metadata":{"execution":{"iopub.status.busy":"2022-08-03T07:43:38.234467Z","iopub.execute_input":"2022-08-03T07:43:38.234907Z","iopub.status.idle":"2022-08-03T07:43:38.242893Z","shell.execute_reply.started":"2022-08-03T07:43:38.234873Z","shell.execute_reply":"2022-08-03T07:43:38.241623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_train = len(train)\nX_train = all_data_sprs[:num_train]\nX_test = all_data_sprs[num_train:]\ny = train['target']","metadata":{"execution":{"iopub.status.busy":"2022-08-03T07:47:34.229395Z","iopub.execute_input":"2022-08-03T07:47:34.230365Z","iopub.status.idle":"2022-08-03T07:47:34.373602Z","shell.execute_reply.started":"2022-08-03T07:47:34.230316Z","shell.execute_reply":"2022-08-03T07:47:34.372234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_valid, y_train, y_valid = train_test_split(X_train, y, test_size = 0.1, stratify = y, random_state = 10)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T07:48:57.891545Z","iopub.execute_input":"2022-08-03T07:48:57.891972Z","iopub.status.idle":"2022-08-03T07:48:58.063720Z","shell.execute_reply.started":"2022-08-03T07:48:57.891939Z","shell.execute_reply":"2022-08-03T07:48:58.062384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.linear_model import LogisticRegression\n\nlogistic_model = LogisticRegression()\nlr_params = {'C' : [0.1,0.125,0.2], 'max_iter' : [800,900,1000], 'solver' : ['liblinear'], 'random_state' : [42]}\ngridsearch_logistic_model = GridSearchCV(estimator = logistic_model, param_grid = lr_params, scoring = 'roc_auc', cv = 5)\ngridsearch_logistic_model.fit(X_train, y_train)\n\nprint('최적 하이퍼파라미터:', gridsearch_logistic_model.best_params_)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T07:56:27.563415Z","iopub.execute_input":"2022-08-03T07:56:27.563913Z","iopub.status.idle":"2022-08-03T08:06:10.597646Z","shell.execute_reply.started":"2022-08-03T07:56:27.563878Z","shell.execute_reply":"2022-08-03T08:06:10.596136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_valid_preds = gridsearch_logistic_model.predict_proba(X_valid)[:,1]\nfrom sklearn.metrics import roc_auc_score\nroc_auc = roc_auc_score(y_valid, y_valid_preds)\nprint(f'검증 데이터 ROC AUC : {roc_auc:.4f}')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T08:07:42.153731Z","iopub.execute_input":"2022-08-03T08:07:42.154302Z","iopub.status.idle":"2022-08-03T08:07:42.178599Z","shell.execute_reply.started":"2022-08-03T08:07:42.154260Z","shell.execute_reply":"2022-08-03T08:07:42.176987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_preds = gridsearch_logistic_model.best_estimator_.predict_proba(X_test)[:,1]\nsubmission['target'] = y_preds\nsubmission.to_csv('submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T08:09:58.927366Z","iopub.execute_input":"2022-08-03T08:09:58.928683Z","iopub.status.idle":"2022-08-03T08:09:59.462894Z","shell.execute_reply.started":"2022-08-03T08:09:58.928620Z","shell.execute_reply":"2022-08-03T08:09:59.461962Z"},"trusted":true},"execution_count":null,"outputs":[]}]}