{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\ndata_path= \"/kaggle/input/cat-in-the-dat/\"\n\ntrain= pd.read_csv(data_path+ \"train.csv\", index_col=\"id\")","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:59:01.846683Z","iopub.execute_input":"2022-07-30T05:59:01.847753Z","iopub.status.idle":"2022-07-30T05:59:03.502133Z","shell.execute_reply.started":"2022-07-30T05:59:01.847711Z","shell.execute_reply":"2022-07-30T05:59:03.500897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test= pd.read_csv(data_path+ \"test.csv\", index_col=\"id\")","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:59:03.948338Z","iopub.execute_input":"2022-07-30T05:59:03.948826Z","iopub.status.idle":"2022-07-30T05:59:05.059366Z","shell.execute_reply.started":"2022-07-30T05:59:03.948786Z","shell.execute_reply":"2022-07-30T05:59:05.058236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission= pd.read_csv(data_path+ \"sample_submission.csv\", index_col=\"id\")","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:59:28.621192Z","iopub.execute_input":"2022-07-30T05:59:28.622330Z","iopub.status.idle":"2022-07-30T05:59:28.712134Z","shell.execute_reply.started":"2022-07-30T05:59:28.622271Z","shell.execute_reply":"2022-07-30T05:59:28.710826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 피처 맞춤형 인코딩","metadata":{}},{"cell_type":"code","source":"all_data= pd.concat([train, test])\nall_data= all_data.drop(\"target\", axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:01:45.522966Z","iopub.execute_input":"2022-07-30T06:01:45.523379Z","iopub.status.idle":"2022-07-30T06:01:46.181972Z","shell.execute_reply.started":"2022-07-30T06:01:45.523348Z","shell.execute_reply":"2022-07-30T06:01:46.180737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 이진 피처/ 수작업으로 인코딩","metadata":{}},{"cell_type":"code","source":"all_data[\"bin_3\"]= all_data[\"bin_3\"].map({\"F\":0, \"T\":1})","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:03:25.101463Z","iopub.execute_input":"2022-07-30T06:03:25.101896Z","iopub.status.idle":"2022-07-30T06:03:25.254737Z","shell.execute_reply.started":"2022-07-30T06:03:25.101863Z","shell.execute_reply":"2022-07-30T06:03:25.253608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_data[\"bin_4\"]= all_data[\"bin_4\"].map({\"N\":0, \"Y\":1})","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:03:23.563153Z","iopub.execute_input":"2022-07-30T06:03:23.563603Z","iopub.status.idle":"2022-07-30T06:03:23.722312Z","shell.execute_reply.started":"2022-07-30T06:03:23.563569Z","shell.execute_reply":"2022-07-30T06:03:23.720787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 순서형 피처 인코딩/ 수작업, ordinalEncoder","metadata":{}},{"cell_type":"code","source":"ord1dict= {\"Novice\": 0, \"Contributor\":1, \"Expert\":2, \"Master\":3, \"Grandmaster\":4}\nord2dict= {\"Freezing\":0, \"Cold\":1, \"Warm\":2, \"Hot\":3, \"Boiling Hot\":4, \"Lava Hot\":5}","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:06:18.830264Z","iopub.execute_input":"2022-07-30T06:06:18.831576Z","iopub.status.idle":"2022-07-30T06:06:18.836969Z","shell.execute_reply.started":"2022-07-30T06:06:18.831509Z","shell.execute_reply":"2022-07-30T06:06:18.835809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_data[\"ord_1\"]= all_data[\"ord_1\"].map(ord1dict)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:06:56.163213Z","iopub.execute_input":"2022-07-30T06:06:56.164522Z","iopub.status.idle":"2022-07-30T06:06:56.349335Z","shell.execute_reply.started":"2022-07-30T06:06:56.164448Z","shell.execute_reply":"2022-07-30T06:06:56.347850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_data[\"ord_2\"]= all_data[\"ord_2\"].map(ord2dict)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:07:19.477753Z","iopub.execute_input":"2022-07-30T06:07:19.478241Z","iopub.status.idle":"2022-07-30T06:07:19.656292Z","shell.execute_reply.started":"2022-07-30T06:07:19.478202Z","shell.execute_reply":"2022-07-30T06:07:19.654887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import OrdinalEncoder\n\n\nord_345= [\"ord_3\", \"ord_4\", \"ord_5\"]\n\nord_encoder= OrdinalEncoder()\n\nall_data[ord_345]= ord_encoder.fit_transform(all_data[ord_345])\n\nfor feature, categories in zip(ord_345, ord_encoder.categories_):\n    print(feature)\n    print(categories)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:10:32.603048Z","iopub.execute_input":"2022-07-30T06:10:32.603528Z","iopub.status.idle":"2022-07-30T06:10:33.389018Z","shell.execute_reply.started":"2022-07-30T06:10:32.603494Z","shell.execute_reply":"2022-07-30T06:10:33.387834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import OrdinalEncoder\n\n\nord_345= [\"ord_3\", \"ord_4\", \"ord_5\"]\n\nord_encoder= OrdinalEncoder()\n\nall_data[ord_345]= ord_encoder.fit_transform(all_data[ord_345])\n\nfor feature, categories in zip(ord_345, ord_encoder.categories_):\n    \n    print(categories)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:11:47.919265Z","iopub.execute_input":"2022-07-30T06:11:47.919752Z","iopub.status.idle":"2022-07-30T06:11:48.202986Z","shell.execute_reply.started":"2022-07-30T06:11:47.919713Z","shell.execute_reply":"2022-07-30T06:11:48.201718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 명목형 피처 인코딩/ 원 핫 인코딩","metadata":{}},{"cell_type":"code","source":"nom_features= [\"nom_\"+ str(i) for i in range(10)]","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:13:24.611038Z","iopub.execute_input":"2022-07-30T06:13:24.611514Z","iopub.status.idle":"2022-07-30T06:13:24.617525Z","shell.execute_reply.started":"2022-07-30T06:13:24.611474Z","shell.execute_reply":"2022-07-30T06:13:24.616396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import OneHotEncoder\n\nonehot_encoder= OneHotEncoder()\nencoded_nom_matrix= onehot_encoder.fit_transform(all_data[nom_features])\nencoded_nom_matrix","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:15:07.987905Z","iopub.execute_input":"2022-07-30T06:15:07.988324Z","iopub.status.idle":"2022-07-30T06:15:10.248486Z","shell.execute_reply.started":"2022-07-30T06:15:07.988292Z","shell.execute_reply":"2022-07-30T06:15:10.247314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_data= all_data.drop(nom_features, axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:16:31.124657Z","iopub.execute_input":"2022-07-30T06:16:31.125626Z","iopub.status.idle":"2022-07-30T06:16:31.153302Z","shell.execute_reply.started":"2022-07-30T06:16:31.125580Z","shell.execute_reply":"2022-07-30T06:16:31.152230Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 날짜 피처 인코딩/ 원 핫 인코딩","metadata":{}},{"cell_type":"code","source":"date_features= [\"day\", \"month\"]\n\nencoded_date_matrix= onehot_encoder.fit_transform(all_data[date_features])\nall_data= all_data.drop(date_features, axis=1)\n\nencoded_date_matrix","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:18:38.760088Z","iopub.execute_input":"2022-07-30T06:18:38.760630Z","iopub.status.idle":"2022-07-30T06:18:38.941311Z","shell.execute_reply.started":"2022-07-30T06:18:38.760573Z","shell.execute_reply":"2022-07-30T06:18:38.939931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# feature scaling: 서로 다른 피처들의 값의 범위가 일치하도록 조정하는 작업","metadata":{}},{"cell_type":"markdown","source":"#### min, max 정규화를 적용하면 피처 값의 범위가 0과 1 사이로 조정, 순서형 피처에 적용해야 한다. ","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import MinMaxScaler\n\nord_features= [\"ord_\"+ str(i) for i in range(6)]\n\nall_data[ord_features]= MinMaxScaler().fit_transform(all_data[ord_features])","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:22:16.649730Z","iopub.execute_input":"2022-07-30T06:22:16.650405Z","iopub.status.idle":"2022-07-30T06:22:16.719936Z","shell.execute_reply.started":"2022-07-30T06:22:16.650355Z","shell.execute_reply":"2022-07-30T06:22:16.718852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 모든 데이터 인코딩이 끝났으니 합칩시다. 그 전에 csr형식 행랼인 명목형, 날짜형 데이터 행렬형식 바꾸기","metadata":{}},{"cell_type":"code","source":"from scipy import sparse\n\nall_data_sprs= sparse.hstack([sparse.csr_matrix(all_data), \n                             encoded_nom_matrix, \n                             encoded_date_matrix], \n                             format= \"csr\")\n\nall_data_sprs","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:25:34.585422Z","iopub.execute_input":"2022-07-30T06:25:34.585864Z","iopub.status.idle":"2022-07-30T06:25:35.122246Z","shell.execute_reply.started":"2022-07-30T06:25:34.585825Z","shell.execute_reply":"2022-07-30T06:25:35.121427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 데이터 나누기","metadata":{}},{"cell_type":"code","source":"num_train= len(train)\n\nX_train= all_data_sprs[:num_train]\nX_test= all_data_sprs[num_train:]\n\ny= train[\"target\"]","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:27:06.446201Z","iopub.execute_input":"2022-07-30T06:27:06.446680Z","iopub.status.idle":"2022-07-30T06:27:06.680980Z","shell.execute_reply.started":"2022-07-30T06:27:06.446635Z","shell.execute_reply":"2022-07-30T06:27:06.679751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX_train, X_valid, y_train, y_valid= train_test_split(X_train, \n                                                    y, \n                                                    test_size=0.1, \n                                                    stratify= y, \n                                                    random_state=10)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:28:56.800388Z","iopub.execute_input":"2022-07-30T06:28:56.801427Z","iopub.status.idle":"2022-07-30T06:28:56.826635Z","shell.execute_reply.started":"2022-07-30T06:28:56.801377Z","shell.execute_reply":"2022-07-30T06:28:56.824884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 하이퍼파라미터 최적화","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import GridSearchCV\nfrom sklearn.linear_model import LogisticRegression\n\nlogistic_model= LogisticRegression()\n\nlr_params= {\"C\": [0.1, 0.125, 0.2], \"max_iter\": [800, 900, 1000], \"solver\": [\"liblinear\"], \"random_state\": [42]}","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:57:22.147689Z","iopub.execute_input":"2022-07-30T06:57:22.148105Z","iopub.status.idle":"2022-07-30T06:57:22.155474Z","shell.execute_reply.started":"2022-07-30T06:57:22.148074Z","shell.execute_reply":"2022-07-30T06:57:22.154309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\ngridsearch_logistic_model= GridSearchCV(estimator= logistic_model, \n                                       param_grid= lr_params, \n                                       scoring= \"roc_auc\", \n                                       cv=5)\n\ngridsearch_logistic_model.fit(X_train, y_train)\n\nprint(\"최적 하이퍼파라미터:\", gridsearch_logistic_model.best_params_)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:45:01.410427Z","iopub.execute_input":"2022-07-30T06:45:01.411879Z","iopub.status.idle":"2022-07-30T06:53:25.034195Z","shell.execute_reply.started":"2022-07-30T06:45:01.411827Z","shell.execute_reply":"2022-07-30T06:53:25.032865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 모델 성능 검증","metadata":{}},{"cell_type":"code","source":"y_valid_preds= gridsearch_logistic_model.predict_proba(X_valid)[:, 1]\n\n\nfrom sklearn.metrics import roc_auc_score\n\nroc_auc= roc_auc_score(y_valid, y_valid_preds)\nprint(f\"검증 데이터 ROC AUC: {roc_auc: 4f}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:56:08.717624Z","iopub.execute_input":"2022-07-30T06:56:08.718061Z","iopub.status.idle":"2022-07-30T06:56:08.740465Z","shell.execute_reply.started":"2022-07-30T06:56:08.718029Z","shell.execute_reply":"2022-07-30T06:56:08.739622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_preds = gridsearch_logistic_model.best_estimator_.predict_proba(X_test)[:,1]\n\nsubmission[\"target\"]= y_preds\nsubmission.to_csv(\"submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-30T06:58:26.519605Z","iopub.execute_input":"2022-07-30T06:58:26.521087Z","iopub.status.idle":"2022-07-30T06:58:27.322899Z","shell.execute_reply.started":"2022-07-30T06:58:26.521044Z","shell.execute_reply":"2022-07-30T06:58:27.321473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ","metadata":{}}]}