{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-30T05:37:02.736220Z","iopub.execute_input":"2022-07-30T05:37:02.736649Z","iopub.status.idle":"2022-07-30T05:37:02.745504Z","shell.execute_reply.started":"2022-07-30T05:37:02.736619Z","shell.execute_reply":"2022-07-30T05:37:02.744534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_path='/kaggle/input/cat-in-the-dat/'\n\ntrain=pd.read_csv(data_path + 'train.csv', index_col='id')\ntest=pd.read_csv(data_path + 'test.csv', index_col='id')\nsubmission=pd.read_csv(data_path + 'sample_submission.csv', index_col='id')\n#index_col은 인덱스를 지정하는 파라미터, 인덱스에 id라는 열 이름을 전달하여 새로운 열 생성\n\ntrain.shape, test.shape\n# 데이터 크기를 (행, 열)로 안내함\n\ntrain.head().T\n#피처의 개수가 너무 많아 가로로 너무 김 > 데이터가 생략되므로 세로로 길게 나타내는 것이 좋음 > head().T\n","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:37:02.829816Z","iopub.execute_input":"2022-07-30T05:37:02.830396Z","iopub.status.idle":"2022-07-30T05:37:04.630388Z","shell.execute_reply.started":"2022-07-30T05:37:02.830367Z","shell.execute_reply":"2022-07-30T05:37:04.629219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:37:04.632042Z","iopub.execute_input":"2022-07-30T05:37:04.632925Z","iopub.status.idle":"2022-07-30T05:37:04.642348Z","shell.execute_reply.started":"2022-07-30T05:37:04.632899Z","shell.execute_reply":"2022-07-30T05:37:04.641428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def resumetable(df):\n    print(f'데이터셋 형상: {df.shape}')\n    summary = pd.DataFrame(df.dtypes, columns=['데이터 타입'])\n    summary = summary.reset_index()\n    summary = summary.rename(columns={'index':'피처'})\n    summary['결측값 개수'] = df.isnull().sum().values\n    summary['고윳값 개수'] = df.nunique().values\n    summary['첫 번째 값'] = df.loc[0].values\n    summary['두 번째 값'] = df.loc[1].values\n    summary['세 번째 값'] = df.loc[2].values\n    \n    return summary\n\n\nresumetable(train)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:37:04.643406Z","iopub.execute_input":"2022-07-30T05:37:04.643712Z","iopub.status.idle":"2022-07-30T05:37:05.782354Z","shell.execute_reply.started":"2022-07-30T05:37:04.643688Z","shell.execute_reply":"2022-07-30T05:37:05.781265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# bin_0~ bin_04는 고윳값이 2개인 이진 피처\n# 그 중에서 머신 러닝은 숫자만 인식하므로 T와 Y는 1로, F와 N은 0으로 인코딩 예정\n\n#norm_0~norm_9가 왜 명목형이고 ord_0~ord_5가 순서형인건 어떻게 알 수 있는지?\n\n#norm0~9는 모두 범주형 데이터로 인코딩 예정\n#ord0~5는 순서형 데이터이므로 (?) 순서에 유의하여 인코딩 예정\n\nfor i in range(3):\n    feature = 'ord_' + str(i)\n    print(f'{feature} 고윳값 : {train[feature].unique()}')\n\n# unique() 함수의 결과 값은 등장한 순서로 출력함\n# ord0~3의 고유값 확인했으니 순서대로 정리\n\nfor i in range(3,6):\n    feature = 'ord_' + str(i)\n    print(f'{feature} 고윳값 : {train[feature].unique()}')\n\n# ord3~5는 알파벳 순서로 정렬하면 될 것 같음","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:37:05.785776Z","iopub.execute_input":"2022-07-30T05:37:05.786121Z","iopub.status.idle":"2022-07-30T05:37:05.959285Z","shell.execute_reply.started":"2022-07-30T05:37:05.786094Z","shell.execute_reply":"2022-07-30T05:37:05.958305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('day 고윳값', train['day'].unique())\nprint('month 고윳값', train['month'].unique())\nprint('target 고윳값', train['target'].unique())","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:37:05.960706Z","iopub.execute_input":"2022-07-30T05:37:05.960943Z","iopub.status.idle":"2022-07-30T05:37:05.971773Z","shell.execute_reply.started":"2022-07-30T05:37:05.960921Z","shell.execute_reply":"2022-07-30T05:37:05.970537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **데이터 시각화**","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib as mpl\nimport matplotlib.pyplot as plt\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:37:05.973342Z","iopub.execute_input":"2022-07-30T05:37:05.973701Z","iopub.status.idle":"2022-07-30T05:37:05.981483Z","shell.execute_reply.started":"2022-07-30T05:37:05.973668Z","shell.execute_reply":"2022-07-30T05:37:05.980510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#타깃값 분포\n#타깃값 분포를 알면 데이터가 얼마나 불균형한지 알 수 있고, 부족한 타깃값에 더 집중해 모델링 수행 가능\nmpl.rc('font', size=15) \nplt.figure(figsize=(7, 6)) #책 오타\n\nax = sns.countplot(x='target', data=train)\nax.set_title('Target Distribution')","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:37:05.982699Z","iopub.execute_input":"2022-07-30T05:37:05.983201Z","iopub.status.idle":"2022-07-30T05:37:06.116547Z","shell.execute_reply.started":"2022-07-30T05:37:05.983158Z","shell.execute_reply":"2022-07-30T05:37:06.115533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(ax.patches)\nrectangle = ax.patches[0]\nprint('사각형 높이', rectangle.get_height())\nprint('사각형 너비', rectangle.get_width())\nprint('사각형 왼쪽 테두리의 x축 위치', rectangle.get_x())\nprint('텍스트 위치의 x좌표:', rectangle.get_x() + rectangle.get_width() / 2.0 )\nprint('텍스트 위치의 y좌표:', rectangle.get_height() + len(train) *0.001) #len(train)이 왜 여백?\n\n\ndef write_percent(ax, total_size) :\n    for patch in ax.patches:\n        height = patch.get_height()\n        width = patch.get_width()\n        left_coord = patch.get_x()\n        percent = height/total_size*100 # 타깃값 비율\n        \n    # (x, y) 좌표에 텍스트 입력\n    ax.text(x=left_coord + width/2.0, #x축 위치\n            y=height + total_size*0.001 , #y축 위치\n            s=f'{percent:1.1f}%', #입력 테스트\n            ha='center') #가운데 정렬\n\nplt.figure(figsize=(7,6))\n\nax=sns.countplot(x='target', data=train)\nwrite_percent(ax, len(train)) #비율 표시\nax.set_title('Target Distribution');","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:37:06.118456Z","iopub.execute_input":"2022-07-30T05:37:06.119100Z","iopub.status.idle":"2022-07-30T05:37:06.260300Z","shell.execute_reply.started":"2022-07-30T05:37:06.119066Z","shell.execute_reply":"2022-07-30T05:37:06.259529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#이진 피처 분포(bin_0~4)\nimport matplotlib.gridspec as gridspec #여러 그래프를 격자 형태로 배치\n# 3행 2열 틀(Figure) 준비\nmpl.rc('font', size=12)\ngrid=gridspec.GridSpec(3,2) # 그래프(서브플롯)를 3행 2열로 배치\nplt.figure(figsize=(10, 16)) # 전체 Figure 크기 설정\nplt.subplots_adjust(wspace=0.4, hspace=0.3) #서브플롯 간 좌우/상하 여백 설정\n\n#서브 플롯 그리기\nbin_features=['bin_0', 'bin_1', 'bin_2', 'bin_3', 'bin_4'] \nfor idx, feature in enumerate(bin_features) :\n    ax = plt.subplot(grid[idx]) \n    \n    sns.countplot(x=feature,\n                 data=train,\n                 hue='target',\n                 palette='pastel',\n                 ax=ax)\n    ax.set_title(f'{feature} Distribution by Target')\n    write_percent(ax, len(train))","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:37:06.261774Z","iopub.execute_input":"2022-07-30T05:37:06.262324Z","iopub.status.idle":"2022-07-30T05:37:07.248886Z","shell.execute_reply.started":"2022-07-30T05:37:06.262291Z","shell.execute_reply":"2022-07-30T05:37:07.248232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#명목형 피처 분포(nom_0~9)\n#nom5~9는 고유값 개수가 많고 의미 알 수 없는 문자열 나열되어 있으므로 nom0~4까지 시각화\n\n#1.교차분석표 생성 함수 만들기\n#2.포인트플롯 생성 함수 만들기\n#3.피처 분포도 및 포인트플롯 생성 함수 만들기\n\n\n#1.교차분석표 생성 함수 만들기\n\npd.crosstab(train['nom_0'], train['target'])\n# 개수로 나오는데 비율이 한 눈에 이해하기 쉬우므로 변환\n\n#정규화 후 비율을 백분율로 표현\ncrosstab= pd.crosstab(train['nom_0'], train['target'], normalize='index')*100 #normalize에 index를 전달하면 index 기준으로 비율 구해줌, columns 하면 columns 기준으로 구해줌\ncrosstab","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:37:07.252930Z","iopub.execute_input":"2022-07-30T05:37:07.253380Z","iopub.status.idle":"2022-07-30T05:37:07.354731Z","shell.execute_reply.started":"2022-07-30T05:37:07.253347Z","shell.execute_reply":"2022-07-30T05:37:07.353843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#위 표는 인덱스가 nom_0인데, 이를 열로 가져와야 그래프 그리기 편하다.\ncrosstab = crosstab.reset_index()\ncrosstab","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:37:07.356036Z","iopub.execute_input":"2022-07-30T05:37:07.356347Z","iopub.status.idle":"2022-07-30T05:37:07.367883Z","shell.execute_reply.started":"2022-07-30T05:37:07.356318Z","shell.execute_reply":"2022-07-30T05:37:07.367187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#교차분석표는 앞으로 계속 사용할 예정이므로 함수로 만들어두자\ndef get_crosstab(df, feature):\n    crosstab = pd.crosstab(df[feature], df['target'], normalize='index')*100\n    crosstab = crosstab.reset_index()\n    return crosstab","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:37:07.369166Z","iopub.execute_input":"2022-07-30T05:37:07.370406Z","iopub.status.idle":"2022-07-30T05:37:07.375212Z","shell.execute_reply.started":"2022-07-30T05:37:07.370380Z","shell.execute_reply":"2022-07-30T05:37:07.374144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"crosstab = get_crosstab(train, 'nom_0')\ncrosstab\n#타깃값 1 비율만 가져오려면 다음과 같이 열의 이름을 인수로 전달  >>>>>>>>> 그러면 행을 가져오는 방법은?\ncrosstab[1]","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:37:07.376246Z","iopub.execute_input":"2022-07-30T05:37:07.377223Z","iopub.status.idle":"2022-07-30T05:37:07.462392Z","shell.execute_reply.started":"2022-07-30T05:37:07.377195Z","shell.execute_reply":"2022-07-30T05:37:07.461688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#2.포인트플롯 생성 함수 만들기\n\n#파라미터 3개 (ax : 포인트플롯을 그릴 축, feature : 포인트플롯으로 그릴 피처, crosstab : 교차분석표)\n\ndef plot_pointplot(ax, feature, crosstab):\n    ax2=ax.twinx() #x축은 공유하고 y축은 공유하지 않는 새로운 축 생성\n    # 새로운 축에 포인트플롯 그리기\n    ax2=sns.pointplot(x=feature, y=1, data=crosstab,\n                     order=crosstab[feature].values, # 포인트플롯 순서\n                     color='black',\n                     legend=False) #범례 미표시\n    ax2.set_ylim(crosstab[1].min()-5, crosstab[1].max()*1.1) #y축 범위 설정\n    ax2.set_ylabel('Target 1 Ratio(%)')","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:37:07.463824Z","iopub.execute_input":"2022-07-30T05:37:07.464115Z","iopub.status.idle":"2022-07-30T05:37:07.472445Z","shell.execute_reply.started":"2022-07-30T05:37:07.464085Z","shell.execute_reply":"2022-07-30T05:37:07.471247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#3.피처 분포도 및 포인트플롯 생성 함수 만들기\n\ndef plot_cat_dist_with_true_ratio(df, features, num_rows, num_cols,\n                                 size=(15,20)) :\n    plt.figure(figsize=size)\n    grid = gridspec.GridSpec(num_rows, num_cols) #서브플롯 배치\n    plt.subplots_adjust(wspace=0.45, hspace=0.1) #서브플롯 좌우/상하 여백 설정\n    \n    for idx, feature in enumerate(features):\n        ax=plt.subplot(grid[idx])\n        crosstab=get_crosstab(df, feature)\n        \n        sns.countplot(x=feature, data=df,\n                     order=crosstab[feature].values,\n                     color='skyblue',\n                     ax=ax)\n        write_percent(ax,len(df))\n        \n        plot_pointplot(ax, feature, crosstab)\n        \n        ax.set_title(f'{feature} Distribution')\n        \nnom_features = ['nom_0','nom_1','nom_2','nom_3','nom_4']\nplot_cat_dist_with_true_ratio(train, nom_features, num_rows=3, num_cols=2)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:37:07.473697Z","iopub.execute_input":"2022-07-30T05:37:07.474000Z","iopub.status.idle":"2022-07-30T05:37:10.192931Z","shell.execute_reply.started":"2022-07-30T05:37:07.473973Z","shell.execute_reply":"2022-07-30T05:37:10.191816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 순서형 피처 분포\nord_features = ['ord_0', 'ord_1', 'ord_2', 'ord_3']\nplot_cat_dist_with_true_ratio(train, ord_features, num_rows=2, num_cols=2, size=(15,12))","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:37:10.194752Z","iopub.execute_input":"2022-07-30T05:37:10.195097Z","iopub.status.idle":"2022-07-30T05:37:11.478530Z","shell.execute_reply.started":"2022-07-30T05:37:10.195066Z","shell.execute_reply":"2022-07-30T05:37:11.477225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#ord_1와 ord_2는 피처 값의 순서가 정렬되지 않았음\n#CategoricalDtype()을 이용하면 피처의 순서를 지정할 수 있음\n\nfrom pandas.api.types import CategoricalDtype\n\nord_1_value = ['Novice', 'Contributor', 'Expert', 'Master', 'Grandmaster']\nord_2_value = ['Freezing', 'Cold', 'Warm', 'Hot', 'Boiling Hot', 'Lava Hot']\n\nord_1_dtype = CategoricalDtype(categories=ord_1_value, ordered=True)\nord_2_dtype = CategoricalDtype(categories=ord_2_value, ordered=True)\n\ntrain['ord_1'] = train['ord_1'].astype(ord_1_dtype)\ntrain['ord_2'] = train['ord_2'].astype(ord_2_dtype)\n\nplot_cat_dist_with_true_ratio(train, ord_features, num_rows=2, num_cols=2, size=(15,12))","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:37:11.480104Z","iopub.execute_input":"2022-07-30T05:37:11.480550Z","iopub.status.idle":"2022-07-30T05:37:13.188258Z","shell.execute_reply.started":"2022-07-30T05:37:11.480509Z","shell.execute_reply":"2022-07-30T05:37:13.186758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 고윳값 순서에 따라 타깃값 1의 비율도 비례해서 커진다는 것 알 수 있음!!\n# ord_4~5 분포는 고윳값이 많기 때문에 가로 길이를 늘려서 그려보기\nplot_cat_dist_with_true_ratio(train, ['ord_4', 'ord_5'], num_rows=2, num_cols=1, size=(15,12))","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:37:13.189735Z","iopub.execute_input":"2022-07-30T05:37:13.190023Z","iopub.status.idle":"2022-07-30T05:37:15.757939Z","shell.execute_reply.started":"2022-07-30T05:37:13.189999Z","shell.execute_reply":"2022-07-30T05:37:15.757301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ord_4~5 분포도 고윳값 순서에 따라 타깃밧 1의 비율이 증가한다!!\n# 마지막으로 날짜 피처 (day, month) 분포도 알아보기\ndate_features=['day', 'month']\nplot_cat_dist_with_true_ratio(train, date_features, num_rows=2, num_cols=1, size=(10,10))","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:37:15.758983Z","iopub.execute_input":"2022-07-30T05:37:15.759346Z","iopub.status.idle":"2022-07-30T05:37:16.430107Z","shell.execute_reply.started":"2022-07-30T05:37:15.759323Z","shell.execute_reply":"2022-07-30T05:37:16.429100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 순환형 데이터는 삼각함수를 이용해 인코딩하면 시작과 끝이 매끄럽게 연결됨(1,2의 간격, 끝과 처음의 간격은 같은 의미)\n\n# 분석 정리\n# 결측값 없음\n# 모든 피처가 중요함 (nom_5~9는 추정)\n# 원-핫 인코딩함\n# 순서형은 순서 주의하여 인코딩~~","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:37:16.431354Z","iopub.execute_input":"2022-07-30T05:37:16.431643Z","iopub.status.idle":"2022-07-30T05:37:16.437639Z","shell.execute_reply.started":"2022-07-30T05:37:16.431614Z","shell.execute_reply":"2022-07-30T05:37:16.436636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **베이스라인 모델**","metadata":{}},{"cell_type":"code","source":"import pandas as pd\n# 데이터 경로\ndata_path = '/kaggle/input/cat-in-the-dat/'\ntrain = pd.read_csv(data_path + 'train.csv', index_col='id')\ntest = pd.read_csv(data_path + 'test.csv', index_col='id')\nsubmission = pd.read_csv(data_path + 'sample_submission.csv', index_col='id')\n\n#데이터 합치기\nall_data=pd.concat([train, test]) #훈련 데이터와 테스트 데이터 합치기 , concat()함수로 합침\nall_data=all_data.drop('target', axis=1) #타깃값 제거\nall_data","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:37:16.439195Z","iopub.execute_input":"2022-07-30T05:37:16.439841Z","iopub.status.idle":"2022-07-30T05:37:19.102107Z","shell.execute_reply.started":"2022-07-30T05:37:16.439802Z","shell.execute_reply":"2022-07-30T05:37:19.101016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import OneHotEncoder\n\nencoder=OneHotEncoder()\nall_data_encoded = encoder.fit_transform(all_data)\n\n#원래 레이블 인코딩 후 원샷인코딩 하는게 아닌가?\n\n#데이터 나누기\nnum_train=len(train) #훈련 데이터 개수\n\n#훈련 데이터와 테스트 데이터 나누기\nX_train=all_data_encoded[:num_train]\nX_test=all_data_encoded[num_train:]\n\ny=train['target']","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:37:19.103872Z","iopub.execute_input":"2022-07-30T05:37:19.104806Z","iopub.status.idle":"2022-07-30T05:37:22.471253Z","shell.execute_reply.started":"2022-07-30T05:37:19.104778Z","shell.execute_reply":"2022-07-30T05:37:22.470243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\n# 훈련데이터, 검증 데이터 분리하기\nX_train, X_valid, y_train, y_valid = train_test_split(X_train, y, #첫번째 인수에 피처, 두번째 인수에 타깃값\n                                                     test_size=0.1, #검증데이터 크기 설정 - 정수면 갯수, 실수면 비율\n                                                     stratify=y, #stratify에 전달하는 값이 공평하게 분배되도록.. 되도록 타깃값 입력\n                                                     random_state=10) #시드값 고정, 다음에 실행에도 같은 결과","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:37:22.473062Z","iopub.execute_input":"2022-07-30T05:37:22.473880Z","iopub.status.idle":"2022-07-30T05:37:22.738872Z","shell.execute_reply.started":"2022-07-30T05:37:22.473839Z","shell.execute_reply":"2022-07-30T05:37:22.738095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 모델 훈련\nfrom sklearn.linear_model import LogisticRegression\n\nlogistic_model = LogisticRegression(max_iter=1000, random_state=42) #모델 생성 max_iter= 호귀 계수를 업데이트하는 반복 횟수\nlogistic_model.fit(X_train, y_train) #모델 훈련","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:37:22.740224Z","iopub.execute_input":"2022-07-30T05:37:22.741536Z","iopub.status.idle":"2022-07-30T05:38:28.767922Z","shell.execute_reply.started":"2022-07-30T05:37:22.741492Z","shell.execute_reply":"2022-07-30T05:38:28.765604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#모델 성능 검증\nlogistic_model.predict_proba(X_valid) #타깃값의 확률을 예측해서 출력 [0일 확률, 1일 확률]\n","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:38:28.769284Z","iopub.execute_input":"2022-07-30T05:38:28.769692Z","iopub.status.idle":"2022-07-30T05:38:28.791287Z","shell.execute_reply.started":"2022-07-30T05:38:28.769649Z","shell.execute_reply":"2022-07-30T05:38:28.790330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"logistic_model.predict(X_valid) #타깃값을 예측하여 출력","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:38:28.796842Z","iopub.execute_input":"2022-07-30T05:38:28.799502Z","iopub.status.idle":"2022-07-30T05:38:28.812321Z","shell.execute_reply.started":"2022-07-30T05:38:28.799426Z","shell.execute_reply":"2022-07-30T05:38:28.811442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#검증 데이터를 활용한 타깃 예측\ny_valid_preds = logistic_model.predict_proba(X_valid)[:, 1] # y_valid_preds에 검증 데이터 타깃값이 1일 확률 저장\n\n#타깃 예측값인 y_valid_preds와 실제 타깃값인 y_valid를 이용해 ROC AUC를 구해보기\nfrom sklearn.metrics import roc_auc_score\n\nroc_auc=roc_auc_score(y_valid, y_valid_preds)\n\nprint(f'검증 데이터 ROC AUC : {roc_auc:.4f}')","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:38:28.816402Z","iopub.execute_input":"2022-07-30T05:38:28.818578Z","iopub.status.idle":"2022-07-30T05:38:28.843852Z","shell.execute_reply.started":"2022-07-30T05:38:28.818541Z","shell.execute_reply":"2022-07-30T05:38:28.843025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 예측 및 결과 제출\ny_preds = logistic_model.predict_proba(X_test)[:, 1] #y_preds에 타깃값이 1일 확률 저장\n\n# 제출 파일 생성\nsubmission['target'] = y_preds\nsubmission.to_csv('submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:38:28.852846Z","iopub.execute_input":"2022-07-30T05:38:28.855033Z","iopub.status.idle":"2022-07-30T05:38:29.532566Z","shell.execute_reply.started":"2022-07-30T05:38:28.854995Z","shell.execute_reply":"2022-07-30T05:38:29.531646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **성능 개선**","metadata":{}},{"cell_type":"code","source":"\n# 1. 피처 맞춤 인코딩\n# 2. 피처 스케일링\n# 3. 하이퍼파라미터 최적화\n\n# 1. 피처 맞춤 인코딩\nimport pandas as pd\n\ndata_path = '/kaggle/input/cat-in-the-dat/'\n\ntrain = pd.read_csv(data_path + 'train.csv', index_col='id')\ntest = pd.read_csv(data_path + 'test.csv', index_col='id')\nsubmission = pd.read_csv(data_path + 'sample_submission.csv', index_col='id')","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:38:29.534315Z","iopub.execute_input":"2022-07-30T05:38:29.534636Z","iopub.status.idle":"2022-07-30T05:38:31.360360Z","shell.execute_reply.started":"2022-07-30T05:38:29.534607Z","shell.execute_reply":"2022-07-30T05:38:31.358770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_data = pd.concat([train, test])\nall_data = all_data.drop('target', axis =1)\n\n# 이진 피처 수작업 인코딩\n# bin_0~2는 이미 고윳값이 0, 1로만 이루어져 있으므로 인코딩 불필요\n# bin3~4는 고유값 T, F 또는 Y, N이라는 문자를 0, 1로 바꾸기\n\nall_data['bin_3'] = all_data['bin_3'].map({'F':0, 'T':1})\nall_data['bin_4'] = all_data['bin_4'].map({'N':0, 'Y':1})\n\n#순서형 피처 수작업 인코딩\nfor i in range(6):\n    feature = 'ord_' + str(i)\n    print(f'{feature} 고윳값 : {train[feature].unique()}')\n#ord_0은 이미 숫자이므로 인코딩 X\n#ord_1~2는 순서를 정해서 인코딩\n#ord_3~5는 알파벳 순서로 인코딩","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:38:31.363668Z","iopub.execute_input":"2022-07-30T05:38:31.364806Z","iopub.status.idle":"2022-07-30T05:38:32.237840Z","shell.execute_reply.started":"2022-07-30T05:38:31.364773Z","shell.execute_reply":"2022-07-30T05:38:32.236732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#ord_1~2 부터 인코딩\nord1dict = {'Novice':0, 'Contributor':1, 'Expert':2, 'Master':3, 'Grandmaster':4}\nord2dict = {'Freezing':0, 'Cold':1, 'Warm':2, 'Hot':3, 'Boiling Hot':4, 'Lava Hot':5}\n\nall_data['ord_1'] = all_data['ord_1'].map(ord1dict)\nall_data['ord_2'] = all_data['ord_2'].map(ord2dict)\n\n#ord3~5 알파벳 순으로 인코딩\n#알파벳 순서로 정렬하고 map()으로 인코딩가능하지만 고윳값 많아서 번거로움\n#사이킷런 OrdinalEncoder 사용\n\nfrom sklearn.preprocessing import OrdinalEncoder\n\nord_345 = ['ord_3', 'ord_4', 'ord_5']\n\nord_encoder = OrdinalEncoder() # 인코더 객체 생성\n# ordinal 인코딩 적용\nall_data[ord_345]= ord_encoder.fit_transform(all_data[ord_345])\n\n#피처별 인코딩 순서 출력\nfor feature, categories in zip(ord_345, ord_encoder.categories_):\n    print(feature)\n    print(categories)\n    \nall_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:38:32.239266Z","iopub.execute_input":"2022-07-30T05:38:32.239599Z","iopub.status.idle":"2022-07-30T05:38:33.438019Z","shell.execute_reply.started":"2022-07-30T05:38:32.239570Z","shell.execute_reply":"2022-07-30T05:38:33.436625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#명복형 피처 인코딩\n#순서 무시해도 되니까 원핫인코딩\nnom_features = ['nom_'+ str(i) for i in range(10)]\n\nfrom sklearn.preprocessing import OneHotEncoder\n\nonehot_encoder= OneHotEncoder() #인코더 객체 생성\n\nencoded_nom_matrix=onehot_encoder.fit_transform(all_data[nom_features])\n\nencoded_nom_matrix\n\nall_data=all_data.drop(nom_features, axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:38:33.439309Z","iopub.execute_input":"2022-07-30T05:38:33.439658Z","iopub.status.idle":"2022-07-30T05:38:36.384120Z","shell.execute_reply.started":"2022-07-30T05:38:33.439632Z","shell.execute_reply":"2022-07-30T05:38:36.383004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#날짜 피처 인코딩\ndate_feature= ['day', 'month']\n\nencoded_date_matrix= onehot_encoder.fit_transform(all_data[date_features])\nall_data=all_data.drop(date_features, axis=1)\n\nencoded_date_matrix","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:38:36.385889Z","iopub.execute_input":"2022-07-30T05:38:36.386292Z","iopub.status.idle":"2022-07-30T05:38:36.516049Z","shell.execute_reply.started":"2022-07-30T05:38:36.386252Z","shell.execute_reply":"2022-07-30T05:38:36.514940Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 2. 피처 스케일링\n# 이진, 명목형, 날짜 피처를 0,1 로 인코딩했으므로, 순서형 피처의 값도 0~1사이가 되도록 스케일링\n\nfrom sklearn.preprocessing import MinMaxScaler\n\nord_features = ['ord_' + str(i) for i in range(6)] # 순서형 피처\n# min-max 정규화\nall_data[ord_features] = MinMaxScaler().fit_transform(all_data[ord_features])\nall_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:38:36.517191Z","iopub.execute_input":"2022-07-30T05:38:36.517706Z","iopub.status.idle":"2022-07-30T05:38:36.587600Z","shell.execute_reply.started":"2022-07-30T05:38:36.517676Z","shell.execute_reply":"2022-07-30T05:38:36.586299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#인코딩 및 스케일링된 피처 합치기\n#all_data에 이진 피처(bin)와 순서형 피처(ord)만 들어있고, 명목형(nom)과 날짜형은 drop되어있고 encoded_{}_matrix에 저장되어 있음\n#all_data는 dataframe이고, encoded_matrix는 CSR형식이므로 맞춰줘야함\n#all_data를 CSR로 바꿈\n\nfrom scipy import sparse\n\nall_data_sprs=sparse.hstack([sparse.csr_matrix(all_data),\n                            encoded_nom_matrix,\n                            encoded_date_matrix],\n                            format= 'csr')\n\n#hstack()는 행렬을 수평 방향으로 합침\n\nall_data_sprs","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:38:36.589604Z","iopub.execute_input":"2022-07-30T05:38:36.590931Z","iopub.status.idle":"2022-07-30T05:38:37.202688Z","shell.execute_reply.started":"2022-07-30T05:38:36.590875Z","shell.execute_reply":"2022-07-30T05:38:37.201410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_train=len(train)\n\nX_train=all_data_sprs[:num_train]\nX_test=all_data_sprs[num_train:]\n\ny=train['target']","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:39:47.933097Z","iopub.execute_input":"2022-07-30T05:39:47.933539Z","iopub.status.idle":"2022-07-30T05:39:48.048160Z","shell.execute_reply.started":"2022-07-30T05:39:47.933502Z","shell.execute_reply":"2022-07-30T05:39:48.047183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:40:42.318884Z","iopub.execute_input":"2022-07-30T05:40:42.319242Z","iopub.status.idle":"2022-07-30T05:40:42.447805Z","shell.execute_reply.started":"2022-07-30T05:40:42.319214Z","shell.execute_reply":"2022-07-30T05:40:42.446242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# 3. 하이퍼파라미터 최적화\n\n\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.linear_model import LogisticRegression\n\nlogistic_model = LogisticRegression()\n\nlr_params = {'C':[0.1,0.125,0.2], 'max_iter':[800,900,1000], \n             'solver':['liblinear'], 'random_state':[42]}\n\ngridsearch_logistic_model = GridSearchCV(estimator=logistic_model,\n                                        param_grid=lr_params,\n                                        scoring='roc_auc',\n                                        cv=5)\n\ngridsearch_logistic_model.fit(X_train, y)\n\nprint('최적 하이퍼파라미터 :', gridsearch_logistic_model.best_params_)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:00:54.168436Z","iopub.execute_input":"2022-07-30T06:00:54.168811Z","iopub.status.idle":"2022-07-30T06:07:17.599582Z","shell.execute_reply.started":"2022-07-30T06:00:54.168783Z","shell.execute_reply":"2022-07-30T06:07:17.598606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_preds = gridsearch_logistic_model.predict_proba(X_test)[:, 1]\n\nfrom sklearn.metrics import roc_auc_score\n\nroc_auc = roc_auc_score(y_valid, y_valid_preds)\n\nprint(f'검증 데이터 ROC AUC : {roc_auc:.4f}')","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:31:43.260994Z","iopub.execute_input":"2022-07-30T06:31:43.261486Z","iopub.status.idle":"2022-07-30T06:31:43.301893Z","shell.execute_reply.started":"2022-07-30T06:31:43.261436Z","shell.execute_reply":"2022-07-30T06:31:43.300406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_preds = gridsearch_logistic_model.best_estimator_.predict_proba(X_test)[:,1]\n\nsubmission['target'] = y_preds\nsubmission.to_csv('submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:32:04.422175Z","iopub.execute_input":"2022-07-30T06:32:04.425161Z","iopub.status.idle":"2022-07-30T06:32:05.107440Z","shell.execute_reply.started":"2022-07-30T06:32:04.425096Z","shell.execute_reply":"2022-07-30T06:32:05.106604Z"},"trusted":true},"execution_count":null,"outputs":[]}]}