{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n%matplotlib inline \n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-23T09:01:59.521576Z","iopub.execute_input":"2022-07-23T09:01:59.521937Z","iopub.status.idle":"2022-07-23T09:01:59.536086Z","shell.execute_reply.started":"2022-07-23T09:01:59.521906Z","shell.execute_reply":"2022-07-23T09:01:59.534734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"titanic_df = pd.read_csv('/kaggle/input/titanic/train.csv')\ntitanic_df.head(10)\ntitanic_df['Cabin'].isna()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T09:01:59.538964Z","iopub.execute_input":"2022-07-23T09:01:59.539415Z","iopub.status.idle":"2022-07-23T09:01:59.560370Z","shell.execute_reply.started":"2022-07-23T09:01:59.539377Z","shell.execute_reply":"2022-07-23T09:01:59.559223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('\\n train data information \\n')\nprint(titanic_df.info())","metadata":{"execution":{"iopub.status.busy":"2022-07-23T09:01:59.561587Z","iopub.execute_input":"2022-07-23T09:01:59.562189Z","iopub.status.idle":"2022-07-23T09:01:59.576808Z","shell.execute_reply.started":"2022-07-23T09:01:59.562152Z","shell.execute_reply":"2022-07-23T09:01:59.575702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(titanic_df['Age'])","metadata":{"execution":{"iopub.status.busy":"2022-07-23T09:01:59.578414Z","iopub.execute_input":"2022-07-23T09:01:59.579096Z","iopub.status.idle":"2022-07-23T09:01:59.587172Z","shell.execute_reply.started":"2022-07-23T09:01:59.579046Z","shell.execute_reply":"2022-07-23T09:01:59.586059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"titanic_df['Age'].fillna(titanic_df['Age'].mean(), inplace=True)\ntitanic_df['Cabin'].fillna('N', inplace=True)\ntitanic_df['Embarked'].fillna('N', inplace=True)\nprint('data set null value count :', titanic_df.isnull().sum())","metadata":{"execution":{"iopub.status.busy":"2022-07-23T09:01:59.589498Z","iopub.execute_input":"2022-07-23T09:01:59.590160Z","iopub.status.idle":"2022-07-23T09:01:59.604245Z","shell.execute_reply.started":"2022-07-23T09:01:59.590122Z","shell.execute_reply":"2022-07-23T09:01:59.603349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#titanic_df.head(3)\nfor _ in titanic_df:\n    print(\"{count} value variation\".format(count = titanic_df[_].value_counts()))\nprint(type(titanic_df))","metadata":{"execution":{"iopub.status.busy":"2022-07-23T09:01:59.605775Z","iopub.execute_input":"2022-07-23T09:01:59.606280Z","iopub.status.idle":"2022-07-23T09:01:59.628332Z","shell.execute_reply.started":"2022-07-23T09:01:59.606249Z","shell.execute_reply":"2022-07-23T09:01:59.627499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(' Sex 값 분포 :\\n',titanic_df['Sex'].value_counts())\nprint('\\n Cabin 값 분포 :\\n',titanic_df['Cabin'].value_counts())\nprint('\\n Embarked 값 분포 :\\n',titanic_df['Embarked'].value_counts())","metadata":{"execution":{"iopub.status.busy":"2022-07-23T09:01:59.629481Z","iopub.execute_input":"2022-07-23T09:01:59.630048Z","iopub.status.idle":"2022-07-23T09:01:59.640992Z","shell.execute_reply.started":"2022-07-23T09:01:59.630013Z","shell.execute_reply":"2022-07-23T09:01:59.640071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"titanic_df['Cabin'] = titanic_df['Cabin'].str[:1]\nprint(titanic_df['Cabin'].head(3))","metadata":{"execution":{"iopub.status.busy":"2022-07-23T09:01:59.642089Z","iopub.execute_input":"2022-07-23T09:01:59.643141Z","iopub.status.idle":"2022-07-23T09:01:59.656880Z","shell.execute_reply.started":"2022-07-23T09:01:59.643108Z","shell.execute_reply":"2022-07-23T09:01:59.655858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"titanic_df.groupby(['Sex','Survived']).head(3)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T09:01:59.659259Z","iopub.execute_input":"2022-07-23T09:01:59.659944Z","iopub.status.idle":"2022-07-23T09:01:59.683967Z","shell.execute_reply.started":"2022-07-23T09:01:59.659909Z","shell.execute_reply":"2022-07-23T09:01:59.682768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"titanic_df.groupby(['Sex','Survived'])['Survived'].count()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T09:01:59.685526Z","iopub.execute_input":"2022-07-23T09:01:59.686598Z","iopub.status.idle":"2022-07-23T09:01:59.699317Z","shell.execute_reply.started":"2022-07-23T09:01:59.686551Z","shell.execute_reply":"2022-07-23T09:01:59.697864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## sex survived graph","metadata":{}},{"cell_type":"code","source":"sns.barplot(x='Sex', y='Survived', data=titanic_df)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T09:01:59.701525Z","iopub.execute_input":"2022-07-23T09:01:59.702746Z","iopub.status.idle":"2022-07-23T09:01:59.963250Z","shell.execute_reply.started":"2022-07-23T09:01:59.702689Z","shell.execute_reply":"2022-07-23T09:01:59.962095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Class, Sex, survived graph","metadata":{}},{"cell_type":"code","source":"sns.barplot(x='Pclass', y='Survived', hue='Sex', data=titanic_df)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T09:01:59.965791Z","iopub.execute_input":"2022-07-23T09:01:59.966228Z","iopub.status.idle":"2022-07-23T09:02:00.373485Z","shell.execute_reply.started":"2022-07-23T09:01:59.966193Z","shell.execute_reply":"2022-07-23T09:02:00.372140Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 입력 age에 따라 구분값을 반환하는 함수 설정. DataFrame의 apply lambda식에 사용. \ndef get_category(age):\n    cat = ''\n    if age <= -1: cat = 'Unknown'\n    elif age <= 5: cat = 'Baby'\n    elif age <= 12: cat = 'Child'\n    elif age <= 18: cat = 'Teenager'\n    elif age <= 25: cat = 'Student'\n    elif age <= 35: cat = 'Young Adult'\n    elif age <= 60: cat = 'Adult'\n    else : cat = 'Elderly'\n    \n    return cat\n\n# 막대그래프의 크기 figure를 더 크게 설정 \nplt.figure(figsize=(10,6))\n\n#X축의 값을 순차적으로 표시하기 위한 설정 \ngroup_names = ['Unknown', 'Baby', 'Child', 'Teenager', 'Student', 'Young Adult', 'Adult', 'Elderly']\n\n# lambda 식에 위에서 생성한 get_category( ) 함수를 반환값으로 지정. \n# get_category(X)는 입력값으로 'Age' 컬럼값을 받아서 해당하는 cat 반환\ntitanic_df['Age_cat'] = titanic_df['Age'].apply(lambda x : get_category(x))\nsns.barplot(x='Age_cat', y = 'Survived', hue='Sex', data=titanic_df, order=group_names)\ntitanic_df.drop('Age_cat', axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T09:02:00.374997Z","iopub.execute_input":"2022-07-23T09:02:00.375843Z","iopub.status.idle":"2022-07-23T09:02:01.168135Z","shell.execute_reply.started":"2022-07-23T09:02:00.375808Z","shell.execute_reply":"2022-07-23T09:02:01.166797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn import preprocessing\n\ndef encode_features(dataDF):\n    features = ['Cabin', 'Sex', 'Embarked']\n    for feature in features:\n        le = preprocessing.LabelEncoder()\n        le = le.fit(dataDF[feature])\n        dataDF[feature] = le.transform(dataDF[feature])\n        \n    return dataDF\n\ntitanic_df = encode_features(titanic_df)\ntitanic_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T09:02:01.169894Z","iopub.execute_input":"2022-07-23T09:02:01.170765Z","iopub.status.idle":"2022-07-23T09:02:01.192277Z","shell.execute_reply.started":"2022-07-23T09:02:01.170727Z","shell.execute_reply":"2022-07-23T09:02:01.190830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\n\n# Null 처리 함수\ndef fillna(df):\n    df['Age'].fillna(df['Age'].mean(), inplace=True)\n    df['Cabin'].fillna('N', inplace=True)\n    df['Embarked'].fillna('N', inplace=True)\n    df['Fare'].fillna(0, inplace=True)\n    return df\n\n# 머신러닝 알고리즘에 불필요한 피처 제거\ndef drop_features(df):\n    df.drop(['PassengerId', 'Name', 'Ticket'], axis=1, inplace=True)\n    return df\n\n# 레이블 인코딩 수행.\ndef format_features(df):\n    df['Cabin'] = df['Cabin'].str[:1]\n    features = ['Cabin', 'Sex', 'Embarked']\n    for feature in features:\n        le = LabelEncoder()\n        le = le.fit(df[feature])\n        df[feature] = le.transform(df[feature])\n    return df\n\n# 앞에서 설정한 데이터 전처리 함수 호출\ndef transform_features(df):\n    df = fillna(df)\n    df = drop_features(df)\n    df = format_features(df)\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-07-23T09:02:01.193730Z","iopub.execute_input":"2022-07-23T09:02:01.194448Z","iopub.status.idle":"2022-07-23T09:02:01.208111Z","shell.execute_reply.started":"2022-07-23T09:02:01.194404Z","shell.execute_reply":"2022-07-23T09:02:01.206796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 원본 데이터를 재로딩 하고, feature데이터 셋과 Label 데이터 셋 추출. \ntitanic_df = pd.read_csv('/kaggle/input/titanic/train.csv')\ny_titanic_df = titanic_df['Survived']\nX_titanic_df= titanic_df.drop('Survived',axis=1)\n\nX_titanic_df = transform_features(X_titanic_df)\n\nfrom sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test=train_test_split(X_titanic_df, y_titanic_df, \\\n                                                  test_size=0.2, random_state=11)\n\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import accuracy_score\n\n# 결정트리, Random Forest, 로지스틱 회귀를 위한 사이킷런 Classifier 클래스 생성\ndt_clf = DecisionTreeClassifier(random_state=11)\nrf_clf = RandomForestClassifier(random_state=11)\nlr_clf = LogisticRegression(solver='liblinear')\n\n# DecisionTreeClassifier 학습/예측/평가\ndt_clf.fit(X_train , y_train)\ndt_pred = dt_clf.predict(X_test)\nprint('DecisionTreeClassifier 정확도: {0:.4f}'.format(accuracy_score(y_test, dt_pred)))\n\n# RandomForestClassifier 학습/예측/평가\nrf_clf.fit(X_train , y_train)\nrf_pred = rf_clf.predict(X_test)\nprint('RandomForestClassifier 정확도:{0:.4f}'.format(accuracy_score(y_test, rf_pred)))\n\n# LogisticRegression 학습/예측/평가\nlr_clf.fit(X_train , y_train)\nlr_pred = lr_clf.predict(X_test)\nprint('LogisticRegression 정확도: {0:.4f}'.format(accuracy_score(y_test, lr_pred)))","metadata":{"execution":{"iopub.status.busy":"2022-07-23T09:02:01.209909Z","iopub.execute_input":"2022-07-23T09:02:01.210674Z","iopub.status.idle":"2022-07-23T09:02:01.529242Z","shell.execute_reply.started":"2022-07-23T09:02:01.210625Z","shell.execute_reply":"2022-07-23T09:02:01.527786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import KFold\n\ndef exec_kfold(clf, folds=5):\n    # 폴드 세트를 5개인 KFold객체를 생성, 폴드 수만큼 예측결과 저장을 위한  리스트 객체 생성.\n    kfold = KFold(n_splits=folds)\n    scores = []\n    \n    # KFold 교차 검증 수행. \n    for iter_count , (train_index, test_index) in enumerate(kfold.split(X_titanic_df)):\n        # X_titanic_df 데이터에서 교차 검증별로 학습과 검증 데이터를 가리키는 index 생성\n        X_train, X_test = X_titanic_df.values[train_index], X_titanic_df.values[test_index]\n        y_train, y_test = y_titanic_df.values[train_index], y_titanic_df.values[test_index]\n        \n        # Classifier 학습, 예측, 정확도 계산 \n        clf.fit(X_train, y_train) \n        predictions = clf.predict(X_test)\n        accuracy = accuracy_score(y_test, predictions)\n        scores.append(accuracy)\n        print(\"교차 검증 {0} 정확도: {1:.4f}\".format(iter_count, accuracy))     \n    \n    # 5개 fold에서의 평균 정확도 계산. \n    mean_score = np.mean(scores)\n    print(\"평균 정확도: {0:.4f}\".format(mean_score)) \n# exec_kfold 호출\nexec_kfold(dt_clf , folds=5) ","metadata":{"execution":{"iopub.status.busy":"2022-07-23T09:02:01.531069Z","iopub.execute_input":"2022-07-23T09:02:01.531544Z","iopub.status.idle":"2022-07-23T09:02:01.562612Z","shell.execute_reply.started":"2022-07-23T09:02:01.531498Z","shell.execute_reply":"2022-07-23T09:02:01.561374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import cross_val_score\n\nscores = cross_val_score(dt_clf, X_titanic_df, y_titanic_df, cv = 5)\nfor iter_count, accuracy in enumerate(scores):\n    print(\"교차 검증 {0} {1:.4f}\".format(iter_count, accuracy))\nprint(\"mean 정확도: {0:.4f}\".format(np.mean(scores)))","metadata":{"execution":{"iopub.status.busy":"2022-07-23T09:02:01.566657Z","iopub.execute_input":"2022-07-23T09:02:01.567875Z","iopub.status.idle":"2022-07-23T09:02:01.619211Z","shell.execute_reply.started":"2022-07-23T09:02:01.567826Z","shell.execute_reply":"2022-07-23T09:02:01.617769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import GridSearchCV\n\nparameters = {'max_depth':[2,3,5,10],\n             'min_samples_split':[2,3,5], 'min_samples_leaf':[1,5,8]}\n\ngrid_dclf = GridSearchCV(dt_clf , param_grid=parameters , scoring='accuracy' , cv=5)\ngrid_dclf.fit(X_train , y_train)\n\nprint('GridSearchCV 최적 하이퍼 파라미터 :',grid_dclf.best_params_)\nprint('GridSearchCV 최고 정확도: {0:.4f}'.format(grid_dclf.best_score_))\nbest_dclf = grid_dclf.best_estimator_\n\n# GridSearchCV의 최적 하이퍼 파라미터로 학습된 Estimator로 예측 및 평가 수행. \ndpredictions = best_dclf.predict(X_test)\naccuracy = accuracy_score(y_test , dpredictions)\nprint('테스트 세트에서의 DecisionTreeClassifier 정확도 : {0:.4f}'.format(accuracy))\nX_test.info()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-23T09:02:01.621096Z","iopub.execute_input":"2022-07-23T09:02:01.622048Z","iopub.status.idle":"2022-07-23T09:02:02.668329Z","shell.execute_reply.started":"2022-07-23T09:02:01.621993Z","shell.execute_reply":"2022-07-23T09:02:02.667158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"titanic_df = pd.read_csv('/kaggle/input/titanic/test.csv')\ntitanic_df.head(3)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T09:02:02.670057Z","iopub.execute_input":"2022-07-23T09:02:02.670421Z","iopub.status.idle":"2022-07-23T09:02:02.692762Z","shell.execute_reply.started":"2022-07-23T09:02:02.670388Z","shell.execute_reply":"2022-07-23T09:02:02.691593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 원본 데이터를 재로딩 하고, feature데이터 셋과 Label 데이터 셋 추출. \ntitanic_df = pd.read_csv('/kaggle/input/titanic/test.csv')\nsubmission_titanic_df = pd.read_csv('/kaggle/input/titanic/gender_submission.csv')\n\ny_test_titanic_df = submission_titanic_df['Survived']\n\ntitanic_df = transform_features(titanic_df)\n\n# GridSearchCV의 최적 하이퍼 파라미터로 학습된 Estimator로 예측 및 평가 수행. \n#dpredictions = best_dclf.predict(titanic_df)\ndpredictions = lr_clf.predict(titanic_df)\naccuracy = accuracy_score(y_test_titanic_df , dpredictions)\nprint('테스트 세트에서의 DecisionTreeClassifier 정확도 : {0:.4f}'.format(accuracy))\nbest_dclf.score(titanic_df, y_test_titanic_df)\nsubmission = pd.DataFrame({\"PassengerId\" : submission_titanic_df.PassengerId, \"Survived\" : dpredictions})\nsubmission.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T09:02:49.387251Z","iopub.execute_input":"2022-07-23T09:02:49.387680Z","iopub.status.idle":"2022-07-23T09:02:49.417146Z","shell.execute_reply.started":"2022-07-23T09:02:49.387648Z","shell.execute_reply":"2022-07-23T09:02:49.416015Z"},"trusted":true},"execution_count":null,"outputs":[]}]}