{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"raw","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-05-26T07:44:00.394556Z","iopub.execute_input":"2022-05-26T07:44:00.394964Z","iopub.status.idle":"2022-05-26T07:44:00.406862Z","shell.execute_reply.started":"2022-05-26T07:44:00.394934Z","shell.execute_reply":"2022-05-26T07:44:00.405409Z"}}},{"cell_type":"markdown","source":"# 한국어로 된 자료가 많이 없는 것 같아 작성 하게 되었습니다\n* 입문으로 가장 많이 하는 데이터가 타이타닉이니, 사소한 설명도 적어가며 작성 해 보았습니다\n* comment 남겨 주시면서 의견 공유 하면 좋을 거 같아요 ^^","metadata":{}},{"cell_type":"code","source":"# data set불러오기는 맨 위에 제공되는 코드 실행\n# 여기부터는 주로사용할 패키지 입력\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2022-05-26T07:44:00.41311Z","iopub.execute_input":"2022-05-26T07:44:00.413536Z","iopub.status.idle":"2022-05-26T07:44:00.428244Z","shell.execute_reply.started":"2022-05-26T07:44:00.413484Z","shell.execute_reply":"2022-05-26T07:44:00.426658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#맨위에서 제공한 데이터를 DataFrame으로 추출\ntitanic_train = pd.read_csv('/kaggle/input/titanic/train.csv') \n# 판다스의 read csv로 csv를 불러올 수 있고, 개인 pc에서도 위처럼 위치를 입력하면 불러올 수 있습니다\ntitanic_test = pd.read_csv('/kaggle/input/titanic/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-05-26T07:44:00.430894Z","iopub.execute_input":"2022-05-26T07:44:00.431388Z","iopub.status.idle":"2022-05-26T07:44:00.458922Z","shell.execute_reply.started":"2022-05-26T07:44:00.431341Z","shell.execute_reply":"2022-05-26T07:44:00.457374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* 머신러닝 모델링을 위해서는 아래 순서로 진행하면 편합니다\n    * 데이터확인\n    * 데이터전처리\n    * ML모델링\n    * 최적화 및 성능평가\n* 위 순서로 통상 진행하고, 성능평가에서 이슈가 있을 경우 맨 처음으로 돌아가서 재수행 하는 것도 좋습니다","metadata":{}},{"cell_type":"code","source":"titanic_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-26T07:44:00.462026Z","iopub.execute_input":"2022-05-26T07:44:00.462578Z","iopub.status.idle":"2022-05-26T07:44:00.484899Z","shell.execute_reply.started":"2022-05-26T07:44:00.462538Z","shell.execute_reply":"2022-05-26T07:44:00.482933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 데이터확인\ntitanic_train.info()","metadata":{"execution":{"iopub.status.busy":"2022-05-26T07:44:00.486497Z","iopub.execute_input":"2022-05-26T07:44:00.486844Z","iopub.status.idle":"2022-05-26T07:44:00.508207Z","shell.execute_reply.started":"2022-05-26T07:44:00.486817Z","shell.execute_reply":"2022-05-26T07:44:00.50705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* info()를 통해 데이터의 구성(컬럼,데이터수,결측여부,타입)을 확인 가능합니다\n* 12개의 컬럼으로 구성되어 있고 명목,수치형이 혼합된 데이터이며 결측값이 나이,객실(?),승선(?)에 존재합니다\n* 히스토그램으로 데이터의 분포를 확인 해 보겠습니다. 수치형만 히스토그램으로 도출이 가능하니, 히스토그램으로 산출되지 않는 데이터들은 수기로 확인합니다","metadata":{}},{"cell_type":"code","source":"# 히스토그램을 통한 데이터 분포 확인\ntitanic_train.hist(bins=10,figsize=(10,10))","metadata":{"execution":{"iopub.status.busy":"2022-05-26T07:44:00.509418Z","iopub.execute_input":"2022-05-26T07:44:00.509683Z","iopub.status.idle":"2022-05-26T07:44:01.962746Z","shell.execute_reply.started":"2022-05-26T07:44:00.509661Z","shell.execute_reply":"2022-05-26T07:44:01.961673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(titanic_train.Sex.value_counts(),titanic_train.Cabin.value_counts(),\n      titanic_train.Embarked.value_counts())\n# value_counts()를 통해서 각 변수의 구성요소와 개수들을 확인 가능합니다\n# 성별과 탑승여부는 특이사항 없어보이나, Cabin은 아주 여러개로 구성되어 있고,\n# 알파벳으로 좌석등급을 유추하는게 아닌가 추측 할 수 있습니다","metadata":{"execution":{"iopub.status.busy":"2022-05-26T07:44:01.964646Z","iopub.execute_input":"2022-05-26T07:44:01.964936Z","iopub.status.idle":"2022-05-26T07:44:01.975992Z","shell.execute_reply.started":"2022-05-26T07:44:01.964911Z","shell.execute_reply":"2022-05-26T07:44:01.974598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 결측제거 _ def로 함수를 만들어 한번에 처리할 수 있으나, 노트북의 장점을 살리기위해\n# 한줄 한줄 작성하며 실행하는 방식으로 하겠습니다 inplact를 True로 해주면 변환한 값을 바로 대치 해 줍니다\ntitanic_train.Age.fillna(titanic_train.Age.mean(),inplace=True)\n# 위의 나이도 사실 알 수 없으니 평균값으로 넣고(수치형이기 때문)\n# 아래 객실과 탑승여부는 명목형이니 n으로 넣습니다\ntitanic_train.Cabin.fillna('N',inplace=True) #\ntitanic_train.Embarked.fillna('N',inplace=True)\ntitanic_train.info() #결측이 모두 사라졌습니다.\n\n# 테스트셋에도 적용시킵니다\n# 추가로 test 셋에는 fare에 결측치가 있습니다\ntitanic_test.Age.fillna(titanic_test.Age.mean(),inplace=True)\ntitanic_test.Fare.fillna(titanic_test.Fare.mean(),inplace=True)\ntitanic_test.Cabin.fillna('N',inplace=True) #\ntitanic_test.Embarked.fillna('N',inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-05-26T07:44:01.977672Z","iopub.execute_input":"2022-05-26T07:44:01.978099Z","iopub.status.idle":"2022-05-26T07:44:02.011101Z","shell.execute_reply.started":"2022-05-26T07:44:01.978069Z","shell.execute_reply":"2022-05-26T07:44:02.009966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 전처리 _ Cabin 컬럼을 맨 앞 알파벳만 남기고 지우기\ntitanic_train.Cabin = titanic_train.Cabin.str[:1] # 데이터프레임의 컬럼에 str로 인덱싱을 하여\n#테스트셋에도 동일하게 적용합니다\ntitanic_test.Cabin = titanic_test.Cabin.str[:1]\n#맨 앞자리만 나오도록 전처리\ntitanic_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-26T07:44:02.013163Z","iopub.execute_input":"2022-05-26T07:44:02.013645Z","iopub.status.idle":"2022-05-26T07:44:02.032493Z","shell.execute_reply.started":"2022-05-26T07:44:02.01361Z","shell.execute_reply":"2022-05-26T07:44:02.031346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* 통계분석\n    * 통계분석은 데이터의 현황을 파악하기 위함입니다\n    * 이 과정에서 머신러닝 타겟값(생존여부)에 영향을 미치는 변수를 파악 할 수도 있고\n    * 모델링만이 아닌 인사이트 도출을 해 낼 수 있습니다\n* 대략적으로 생존에 영향을 미쳤을 변수들로만 확인을 해보겠습니다","metadata":{}},{"cell_type":"code","source":"print(titanic_train.groupby(['Sex','Survived'])['Survived'].count())\n#여성이 생존율이 월등히 높은걸 확인 가능합니다 시각화도 진행합니다\nsns.barplot(x='Sex',y='Survived',data=titanic_train)","metadata":{"execution":{"iopub.status.busy":"2022-05-26T07:44:02.033706Z","iopub.execute_input":"2022-05-26T07:44:02.034093Z","iopub.status.idle":"2022-05-26T07:44:02.232581Z","shell.execute_reply.started":"2022-05-26T07:44:02.034056Z","shell.execute_reply":"2022-05-26T07:44:02.231849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(titanic_train.groupby(['Cabin','Survived'])['Survived'].count())\n#N석이 가장 많은 것으로 보니 제일 저렴한 좌석이지 않을까 싶습니다\n#또한 해당 좌석의 생존율이 가장 낮은것으로 확인됩니다\nsns.barplot(x='Cabin',y='Survived',hue='Sex',data=titanic_train)\n#hue를 추가함으로써 각 막대들을 성별로 쪼개서 확인 가능합니다","metadata":{"execution":{"iopub.status.busy":"2022-05-26T07:44:02.233723Z","iopub.execute_input":"2022-05-26T07:44:02.234714Z","iopub.status.idle":"2022-05-26T07:44:03.005268Z","shell.execute_reply.started":"2022-05-26T07:44:02.234678Z","shell.execute_reply":"2022-05-26T07:44:03.003744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* 우선 성별과 좌석이 생존에 많은 영향을 끼칠 수 있겠다 라고 생각 가능합니다\n* 상위 좌석일수록 배의 위에 존재하지 않았나 추측도 가능하겠습니다","metadata":{}},{"cell_type":"markdown","source":"* 마지막으로 ML 모형에 학습시키기 위해서는 명목형이 아닌 수치형데이터 여야 합니다\n* 그래서 명목형인 변수들을 라벨링(Labeling) 해줍니다\n* labeling과 one hot encoding이 있는데\n    * labeing = 음식 이라는 컬럼에 '라면,치킨,국밥'을 '1,2,3'같은 식으로 변환\n    * one hot encoding = 위와같은 컬럼을 '라면''치킨''국밥'이라는 컬럼으로 각각 찢습니다\n    * 이후 라면에 해당하는 데이터는 1, 아니면 0 이런식으로 이진분류를 갖는 컬럼으로 찢어주는 방법입니다","metadata":{}},{"cell_type":"code","source":"# ****이때 test 데이터에는 transform만 해야 합니다.\n# fit을 해버리면 테스트 데이터 들 만의 새로운 라벨이 생깁니다. \n#train데이터로 학습을 할 것이기에 train에 fit된 라벨링을 그대로 test에 적용시키는 원리입니다\nfrom sklearn.preprocessing import LabelEncoder\nlabelen = LabelEncoder()\n\n# train,test 모두 라벨링을 해줘야 하니, def로 모듈을 구현해서 한번에 처리하도록 합니다\ndef label (train,test): # train은 fit transform을, test는 transform만 해야하니 두개를 입력받습니다\n    colum = ['Sex','Cabin','Embarked']\n    for x in colum :\n        train[x] = labelen.fit_transform(train[x])\n        test[x] = labelen.transform(test[x]) # 테스트는 transform만\n    return train,test\nlabel(titanic_train,titanic_test)\n        ","metadata":{"execution":{"iopub.status.busy":"2022-05-26T07:44:03.008137Z","iopub.execute_input":"2022-05-26T07:44:03.008466Z","iopub.status.idle":"2022-05-26T07:44:03.039847Z","shell.execute_reply.started":"2022-05-26T07:44:03.008437Z","shell.execute_reply":"2022-05-26T07:44:03.039094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train=titanic_train.drop(['PassengerId','Name','Ticket'],axis=1)\n# 승객번호,이름,티켓이름등은 예측에 무의미 할 것으로 보이니 drop으로 제거 해 주고,\n# inplace를 True로 하여 해당 컬럼들이 빠진 상태로 저장시킵니다\ntest=titanic_test.drop(['PassengerId','Name','Ticket'],axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-05-26T07:44:03.040963Z","iopub.execute_input":"2022-05-26T07:44:03.041372Z","iopub.status.idle":"2022-05-26T07:44:03.060781Z","shell.execute_reply.started":"2022-05-26T07:44:03.041342Z","shell.execute_reply":"2022-05-26T07:44:03.059948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ML 모델링 시작\n* 모델링을 하기 위해 train,test셋을 변수,타겟 별로 찢어줍니다\n* 모형을 선택하고, 하이퍼파라미터를 최적화시켜줍니다\n* 분류기에 적합한 평가방법으로 평가합니다","metadata":{}},{"cell_type":"markdown","source":"* Data Set 구성\n* X_train,X_test,y_train,y_test 로 구분합니다  ㅇ","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n# 그러나 이미 kaggle에서 train과 test를 나눈상태이니, 변수인X, 타겟인y로만 나눠주면 됩니다","metadata":{"execution":{"iopub.status.busy":"2022-05-26T07:44:03.062069Z","iopub.execute_input":"2022-05-26T07:44:03.062562Z","iopub.status.idle":"2022-05-26T07:44:03.079138Z","shell.execute_reply.started":"2022-05-26T07:44:03.062527Z","shell.execute_reply":"2022-05-26T07:44:03.078013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = train.drop('Survived',axis=1) # 학습을위한데이터이고 변수는 타겟인 생존을 제외한 변수\ny_train = train.Survived\nX_test = test # y_test가 없는 이유는 모델을 통해 산출된 결과물을 비교할 y_test를 \n#캐글에서 갖고있고, 해당 정답과 비교할 것 이기 때문입니다\n","metadata":{"execution":{"iopub.status.busy":"2022-05-26T07:44:03.080562Z","iopub.execute_input":"2022-05-26T07:44:03.081263Z","iopub.status.idle":"2022-05-26T07:44:03.099741Z","shell.execute_reply.started":"2022-05-26T07:44:03.081223Z","shell.execute_reply":"2022-05-26T07:44:03.098386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* 모형생성\n* 현재 생존유무를 가르는 '분류기'를 생성해야 합니다\n* 의사결정나무, 랜덤포레스트, gbm등 있지만 우선 랜덤포레스트로 해보겠습니다","metadata":{"execution":{"iopub.status.busy":"2022-05-26T06:24:31.61344Z","iopub.execute_input":"2022-05-26T06:24:31.615065Z","iopub.status.idle":"2022-05-26T06:24:31.630466Z","shell.execute_reply.started":"2022-05-26T06:24:31.614994Z","shell.execute_reply":"2022-05-26T06:24:31.628965Z"}}},{"cell_type":"code","source":"#train 데이터를 fit 하게되면 학습을 시킵니다. 이후 predict를통해 학습한 모형에 test데이터를 넣는식으로 확인합니다","metadata":{"execution":{"iopub.status.busy":"2022-05-26T07:44:03.101175Z","iopub.execute_input":"2022-05-26T07:44:03.101715Z","iopub.status.idle":"2022-05-26T07:44:03.121579Z","shell.execute_reply.started":"2022-05-26T07:44:03.101677Z","shell.execute_reply":"2022-05-26T07:44:03.120367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier # 많이들 익숙 하실 랜덤포레스트로 수행했습니다\nfrom sklearn.model_selection import GridSearchCV # 하이퍼파라미터를 최적 화 해줄 gridsearch를 활성화합니다\nrf = RandomForestClassifier()\nrf.fit(X_train,y_train)\nparams={'max_depth':[3,5,7,9,11],\n       'min_samples_split':[2,3,5],\n       'min_samples_leaf':[1,5,7,9]} # 의사결정나무를 기반으로하기때문에 의사결정나무와 파라미터가 유사합니다\nmodel = GridSearchCV(rf,cv=5,scoring='accuracy',param_grid=params)\nmodel.fit(X_train,y_train)\nmodel.best_score_ # 해당 모델로 테스트를 해보면 나올 정확도를 확인 할 수 있습니다","metadata":{"execution":{"iopub.status.busy":"2022-05-26T07:44:03.122565Z","iopub.execute_input":"2022-05-26T07:44:03.122861Z","iopub.status.idle":"2022-05-26T07:44:57.856957Z","shell.execute_reply.started":"2022-05-26T07:44:03.122837Z","shell.execute_reply":"2022-05-26T07:44:57.855626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred1 = model.best_estimator_.predict(X_test) # best estimator를 하면 최고의 성능을 보였던 파라미터를\n#모형에 적용시켜줍니다\nsubmission= pd.DataFrame({\"PassengerId\":titanic_test[\"PassengerId\"],\"Survived\":y_pred1})\nsubmission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-05-26T07:44:57.858142Z","iopub.execute_input":"2022-05-26T07:44:57.858429Z","iopub.status.idle":"2022-05-26T07:44:57.886179Z","shell.execute_reply.started":"2022-05-26T07:44:57.858404Z","shell.execute_reply":"2022-05-26T07:44:57.884683Z"},"trusted":true},"execution_count":null,"outputs":[]}]}