{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n        \n\ndata_path = '/kaggle/input/cat-in-the-dat/'\ntrain = pd.read_csv(data_path + 'train.csv',  index_col = 'id')\ntest = pd.read_csv(data_path + 'test.csv', index_col = 'id')\nsubmission = pd.read_csv(data_path + 'sample_submission.csv', index_col = 'id')\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-31T07:54:18.297579Z","iopub.execute_input":"2022-07-31T07:54:18.298005Z","iopub.status.idle":"2022-07-31T07:54:21.585532Z","shell.execute_reply.started":"2022-07-31T07:54:18.297970Z","shell.execute_reply":"2022-07-31T07:54:21.583981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 베이스라인 모델","metadata":{}},{"cell_type":"markdown","source":"# 데이터 합치기","metadata":{}},{"cell_type":"code","source":"all_data = pd.concat([train, test])             # 훈련 데이터와 테스트 데이터 합치기\nall_data = all_data.drop('target', axis = 1)    # 타깃값 제거\nall_data","metadata":{"execution":{"iopub.status.busy":"2022-07-31T07:54:21.587602Z","iopub.execute_input":"2022-07-31T07:54:21.588015Z","iopub.status.idle":"2022-07-31T07:54:22.856968Z","shell.execute_reply.started":"2022-07-31T07:54:21.587980Z","shell.execute_reply":"2022-07-31T07:54:22.855638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 인코딩 (원핫)","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import OneHotEncoder\n\nencoder = OneHotEncoder()\nall_data_encoded = encoder.fit_transform(all_data)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T07:54:22.858435Z","iopub.execute_input":"2022-07-31T07:54:22.858784Z","iopub.status.idle":"2022-07-31T07:54:26.965478Z","shell.execute_reply.started":"2022-07-31T07:54:22.858753Z","shell.execute_reply":"2022-07-31T07:54:26.964327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 데이터 나누기\n\n훈련 데이터와 테스트 데이터 나누기","metadata":{}},{"cell_type":"code","source":"num_train = len(train)                     # 훈련 데이터의 길이 numtrain에 저장\n\nX_train = all_data_encoded[:num_train]     # alldata의 0행부터 numtrain-1행까지 훈련 데이터로 분리 저장\nX_test = all_data_encoded[num_train:]     # alldata의 numtrain행부터 마지막행까지 테스트 데이터로 분리 저장\n\ny= train['target']","metadata":{"execution":{"iopub.status.busy":"2022-07-31T07:54:26.969408Z","iopub.execute_input":"2022-07-31T07:54:26.969964Z","iopub.status.idle":"2022-07-31T07:54:27.217009Z","shell.execute_reply.started":"2022-07-31T07:54:26.969916Z","shell.execute_reply":"2022-07-31T07:54:27.215779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"훈련 데이터에서 훈련데이터와 검증 데이터 나누기","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX_train, X_valid, y_train, y_valid = train_test_split(X_train, y,        # 피처, 타깃값\n                                                      test_size = 0.1,   # 검증 데이터 크기 지정하는 파라미터 (비율)\n                                                      stratify = y,      # y를 수평적 집단으로 나눈다 = 타깃값을 훈련 데이터와 검증 데이터에 같은 비율로 공정하게 배분한다\n                                                      random_state = 10) # 시드값 고정","metadata":{"execution":{"iopub.status.busy":"2022-07-31T07:54:27.218763Z","iopub.execute_input":"2022-07-31T07:54:27.219235Z","iopub.status.idle":"2022-07-31T07:54:27.466163Z","shell.execute_reply.started":"2022-07-31T07:54:27.219179Z","shell.execute_reply":"2022-07-31T07:54:27.464935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 모델 훈련\n\n선형회귀의 응용인 로지스틱 회귀 모델 이용","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\n\nlogistic_model = LogisticRegression(max_iter = 1000, random_state = 42) # 모델 생성\nlogistic_model.fit(X_train, y_train)                                    # 모델 훈련\n\n# max_iter: 모델의 회귀 계수를 업데이트하는 반복 회수, 즉 몇번이나 회귀 계수 트라이할 지\n# random_state: 값을 지정해놓으면 여러 번 실행해도 매번 같은 결과 나옴, 아무 값으로 해도 됨","metadata":{"execution":{"iopub.status.busy":"2022-07-31T07:54:27.468028Z","iopub.execute_input":"2022-07-31T07:54:27.468818Z","iopub.status.idle":"2022-07-31T07:55:43.423074Z","shell.execute_reply.started":"2022-07-31T07:54:27.468771Z","shell.execute_reply":"2022-07-31T07:55:43.421627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 모델 성능 검증","metadata":{}},{"cell_type":"code","source":"logistic_model.predict_proba(X_valid)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T07:55:43.424975Z","iopub.execute_input":"2022-07-31T07:55:43.425609Z","iopub.status.idle":"2022-07-31T07:55:43.451055Z","shell.execute_reply.started":"2022-07-31T07:55:43.425558Z","shell.execute_reply":"2022-07-31T07:55:43.449639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"첫번째 열은 타깃값이 0일 확률, 두번째 열은 타깃값이 1일 확률","metadata":{}},{"cell_type":"code","source":"logistic_model.predict(X_valid)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T07:55:43.452906Z","iopub.execute_input":"2022-07-31T07:55:43.454419Z","iopub.status.idle":"2022-07-31T07:55:43.480615Z","shell.execute_reply.started":"2022-07-31T07:55:43.454368Z","shell.execute_reply":"2022-07-31T07:55:43.479199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"위에서 확률이 많이 나온 쪽으로 예측","metadata":{}},{"cell_type":"markdown","source":"검증 데이터를 활용한 타깃값 예측: '타깃값이 1일 확률'","metadata":{}},{"cell_type":"code","source":"y_valid_preds = logistic_model.predict_proba(X_valid)[:,1]\n#y_valid_preds 변수에 y값(타깃값)이 1일 확률 저장","metadata":{"execution":{"iopub.status.busy":"2022-07-31T07:55:43.483061Z","iopub.execute_input":"2022-07-31T07:55:43.485507Z","iopub.status.idle":"2022-07-31T07:55:43.507651Z","shell.execute_reply.started":"2022-07-31T07:55:43.485454Z","shell.execute_reply":"2022-07-31T07:55:43.506140Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"검증데이터와 실제 데이터를 이용하여 성능 알아보기: ROC AUC\n\nROC 곡선에서 같은 거짓 양성 비율일 때 참 양성 비율이 높을수록 정확한 것이니 곡선이 y=x 그래프에서 멀어질 수록 정확한 것임. \n\n그렇다면 auc는 ROC이하의 넓이니까 1에 가까울수록 좋은 것임","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import roc_auc_score\n\nroc_auc = roc_auc_score(y_valid, y_valid_preds)\n\nprint(f'검증 데이터 ROC AUC: {roc_auc:.4f}')","metadata":{"execution":{"iopub.status.busy":"2022-07-31T07:55:43.517639Z","iopub.execute_input":"2022-07-31T07:55:43.521826Z","iopub.status.idle":"2022-07-31T07:55:43.547214Z","shell.execute_reply.started":"2022-07-31T07:55:43.521747Z","shell.execute_reply":"2022-07-31T07:55:43.545918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 예측 및 결과 제출\n\n실제 테스트 데이터를 활용해 타깃값이 1일 확류 예측하고 제출","metadata":{}},{"cell_type":"code","source":"y_preds = logistic_model.predict_proba(X_test)[:,1]","metadata":{"execution":{"iopub.status.busy":"2022-07-31T08:03:50.986685Z","iopub.execute_input":"2022-07-31T08:03:50.987467Z","iopub.status.idle":"2022-07-31T08:03:51.012730Z","shell.execute_reply.started":"2022-07-31T08:03:50.987417Z","shell.execute_reply":"2022-07-31T08:03:51.011175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission['target'] = y_preds\n\nsubmission.to_csv('submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-31T08:03:53.516783Z","iopub.execute_input":"2022-07-31T08:03:53.517298Z","iopub.status.idle":"2022-07-31T08:03:54.293972Z","shell.execute_reply.started":"2022-07-31T08:03:53.517262Z","shell.execute_reply":"2022-07-31T08:03:54.292477Z"},"trusted":true},"execution_count":null,"outputs":[]}]}