{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## 1. Imports","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-17T11:39:08.706995Z","iopub.execute_input":"2022-07-17T11:39:08.707895Z","iopub.status.idle":"2022-07-17T11:39:08.742286Z","shell.execute_reply.started":"2022-07-17T11:39:08.707779Z","shell.execute_reply":"2022-07-17T11:39:08.741269Z"}}},{"cell_type":"code","source":"# numpy and pandas for data manipulation\nimport numpy as np\nimport pandas as pd \n\n# sklearn preprocessing for dealing with categorical variables\nfrom sklearn.preprocessing import LabelEncoder\n\n# File system manangement\nimport os\n\n# Suppress warnings \nimport warnings\nwarnings.filterwarnings('ignore')\n\n# matplotlib and seaborn for plotting\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Scaler\nfrom sklearn.preprocessing import StandardScaler\n\n# model\nfrom sklearn.ensemble import RandomForestClassifier\nfrom xgboost import XGBClassifier\nfrom sklearn.svm import SVC\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.linear_model import LogisticRegression as LR\n\n# Ensemble Voting\nfrom sklearn.ensemble import VotingClassifier\n\n# metrics\nfrom sklearn.metrics import accuracy_score\n\nfrom sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2022-07-18T07:41:39.829842Z","iopub.execute_input":"2022-07-18T07:41:39.830594Z","iopub.status.idle":"2022-07-18T07:41:41.432819Z","shell.execute_reply.started":"2022-07-18T07:41:39.830446Z","shell.execute_reply":"2022-07-18T07:41:41.431777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2. Read in Data","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('../input/titanic/train.csv').drop(['PassengerId','Name'],axis=1) \nprint('train shape:\\n',train.shape)\ntrain.head(6)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T07:41:41.434450Z","iopub.execute_input":"2022-07-18T07:41:41.434736Z","iopub.status.idle":"2022-07-18T07:41:41.482727Z","shell.execute_reply.started":"2022-07-18T07:41:41.434708Z","shell.execute_reply":"2022-07-18T07:41:41.481736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv('../input/titanic/test.csv').drop(['PassengerId','Name'],axis=1) \nprint('test shape:\\n',test.shape)\ntest.head(6)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T07:41:41.484284Z","iopub.execute_input":"2022-07-18T07:41:41.485217Z","iopub.status.idle":"2022-07-18T07:41:41.510453Z","shell.execute_reply.started":"2022-07-18T07:41:41.485184Z","shell.execute_reply":"2022-07-18T07:41:41.509283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3. EDA","metadata":{}},{"cell_type":"markdown","source":"### 3.1 Summary","metadata":{}},{"cell_type":"code","source":"train.describe().T","metadata":{"execution":{"iopub.status.busy":"2022-07-18T07:41:41.513628Z","iopub.execute_input":"2022-07-18T07:41:41.513962Z","iopub.status.idle":"2022-07-18T07:41:41.549419Z","shell.execute_reply.started":"2022-07-18T07:41:41.513932Z","shell.execute_reply":"2022-07-18T07:41:41.548434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 3.2 Missing Values","metadata":{}},{"cell_type":"code","source":"# Function to calculate missing values by column# Funct \ndef missing_values_table(df):\n        # Total missing values\n        mis_val = df.isnull().sum()\n        \n        # Percentage of missing values\n        mis_val_percent = 100 * df.isnull().sum() / len(df)\n        \n        # Make a table with the results\n        mis_val_table = pd.concat([mis_val, mis_val_percent], axis=1)\n        \n        # Rename the columns\n        mis_val_table_ren_columns = mis_val_table.rename(\n        columns = {0 : 'Missing Values', 1 : '% of Total Values'})\n        \n        # Sort the table by percentage of missing descending\n        mis_val_table_ren_columns = mis_val_table_ren_columns[\n            mis_val_table_ren_columns.iloc[:,1] != 0].sort_values(\n        '% of Total Values', ascending=False).round(1)\n        \n        # Print some summary information\n        print (\"Your selected dataframe has \" + str(df.shape[1]) + \" columns.\\n\"      \n            \"There are \" + str(mis_val_table_ren_columns.shape[0]) +\n              \" columns that have missing values.\")\n        \n        # Return the dataframe with missing information\n        return mis_val_table_ren_columns","metadata":{"execution":{"iopub.status.busy":"2022-07-18T07:41:41.550392Z","iopub.execute_input":"2022-07-18T07:41:41.550707Z","iopub.status.idle":"2022-07-18T07:41:41.557635Z","shell.execute_reply.started":"2022-07-18T07:41:41.550681Z","shell.execute_reply":"2022-07-18T07:41:41.557044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Missing values statistics\nmissing_values = missing_values_table(train)\nmissing_values","metadata":{"execution":{"iopub.status.busy":"2022-07-18T07:41:41.558515Z","iopub.execute_input":"2022-07-18T07:41:41.558858Z","iopub.status.idle":"2022-07-18T07:41:41.584780Z","shell.execute_reply.started":"2022-07-18T07:41:41.558836Z","shell.execute_reply":"2022-07-18T07:41:41.583845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_values = missing_values_table(test)\nmissing_values","metadata":{"execution":{"iopub.status.busy":"2022-07-18T07:41:41.585932Z","iopub.execute_input":"2022-07-18T07:41:41.586758Z","iopub.status.idle":"2022-07-18T07:41:41.603672Z","shell.execute_reply.started":"2022-07-18T07:41:41.586728Z","shell.execute_reply":"2022-07-18T07:41:41.602898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 3.3 Column Types","metadata":{}},{"cell_type":"code","source":"# Number of each type of column\ntrain.dtypes.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T07:41:41.604925Z","iopub.execute_input":"2022-07-18T07:41:41.605466Z","iopub.status.idle":"2022-07-18T07:41:41.612316Z","shell.execute_reply.started":"2022-07-18T07:41:41.605436Z","shell.execute_reply":"2022-07-18T07:41:41.611491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.select_dtypes('object').apply(pd.Series.nunique, axis = 0)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T07:41:41.613411Z","iopub.execute_input":"2022-07-18T07:41:41.613687Z","iopub.status.idle":"2022-07-18T07:41:41.623850Z","shell.execute_reply.started":"2022-07-18T07:41:41.613632Z","shell.execute_reply":"2022-07-18T07:41:41.622915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.select_dtypes('object').apply(pd.Series.nunique, axis = 0)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T07:41:41.629399Z","iopub.execute_input":"2022-07-18T07:41:41.630052Z","iopub.status.idle":"2022-07-18T07:41:41.639154Z","shell.execute_reply.started":"2022-07-18T07:41:41.630021Z","shell.execute_reply":"2022-07-18T07:41:41.638330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 3.4 Survived","metadata":{}},{"cell_type":"code","source":"train['Survived'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T07:41:41.640125Z","iopub.execute_input":"2022-07-18T07:41:41.640857Z","iopub.status.idle":"2022-07-18T07:41:41.648295Z","shell.execute_reply.started":"2022-07-18T07:41:41.640817Z","shell.execute_reply":"2022-07-18T07:41:41.647449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['Survived'].plot.hist();","metadata":{"execution":{"iopub.status.busy":"2022-07-18T07:41:41.649533Z","iopub.execute_input":"2022-07-18T07:41:41.649969Z","iopub.status.idle":"2022-07-18T07:41:41.864647Z","shell.execute_reply.started":"2022-07-18T07:41:41.649941Z","shell.execute_reply":"2022-07-18T07:41:41.863666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 4. Data preprocessing","metadata":{}},{"cell_type":"markdown","source":"### 4.1 deal with the missing data","metadata":{}},{"cell_type":"code","source":"# 1. Cabin\ntrain['Cabin'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T07:41:41.865859Z","iopub.execute_input":"2022-07-18T07:41:41.866663Z","iopub.status.idle":"2022-07-18T07:41:41.875477Z","shell.execute_reply.started":"2022-07-18T07:41:41.866631Z","shell.execute_reply":"2022-07-18T07:41:41.874418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 2. age\ntrain['Age'].hist()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T07:41:41.876525Z","iopub.execute_input":"2022-07-18T07:41:41.876770Z","iopub.status.idle":"2022-07-18T07:41:42.064470Z","shell.execute_reply.started":"2022-07-18T07:41:41.876749Z","shell.execute_reply":"2022-07-18T07:41:42.063566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 3. Embarked\ntrain['Embarked'].value_counts()   ### common value is S","metadata":{"execution":{"iopub.status.busy":"2022-07-18T07:41:42.065820Z","iopub.execute_input":"2022-07-18T07:41:42.066544Z","iopub.status.idle":"2022-07-18T07:41:42.076595Z","shell.execute_reply.started":"2022-07-18T07:41:42.066500Z","shell.execute_reply":"2022-07-18T07:41:42.075650Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 4. Fare\ntrain['Fare'].value_counts() ","metadata":{"execution":{"iopub.status.busy":"2022-07-18T07:41:42.078533Z","iopub.execute_input":"2022-07-18T07:41:42.079233Z","iopub.status.idle":"2022-07-18T07:41:42.091234Z","shell.execute_reply.started":"2022-07-18T07:41:42.079194Z","shell.execute_reply":"2022-07-18T07:41:42.090273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['Fare'].hist()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T07:41:42.092922Z","iopub.execute_input":"2022-07-18T07:41:42.093551Z","iopub.status.idle":"2022-07-18T07:41:42.274864Z","shell.execute_reply.started":"2022-07-18T07:41:42.093511Z","shell.execute_reply":"2022-07-18T07:41:42.273780Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def clean(data):\n    data = data.drop(['Cabin','Ticket'],axis=1)\n    data['Age'] =  data['Age'].fillna(data['Age'].median())\n    data['Embarked']=data['Embarked'].fillna(\"S\")     \n    data['Fare'] = data['Fare'].fillna(data['Fare'].median())\n    return data\n    \ntrain = clean(train)\ntest  = clean(test)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T07:41:42.277346Z","iopub.execute_input":"2022-07-18T07:41:42.277993Z","iopub.status.idle":"2022-07-18T07:41:42.292368Z","shell.execute_reply.started":"2022-07-18T07:41:42.277950Z","shell.execute_reply":"2022-07-18T07:41:42.291066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 4.2 Label Encoding and One-Hot Encoding","metadata":{}},{"cell_type":"code","source":"# Create a label encoder object\nle = LabelEncoder()\nle_count = 0\n\n# Iterate through the columns\nfor col in train:\n    if train[col].dtype == 'object':\n        # If 2 or fewer unique categories\n        if len(list(train[col].unique())) <= 2:\n            # Train on the training data\n            le.fit(train[col])\n            # Transform both training and testing data\n            train[col] = le.transform(train[col])\n            test[col] = le.transform(test[col])\n            \n            # Keep track of how many columns were label encoded\n            le_count += 1\n            \nprint('%d columns were label encoded.' % le_count)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T07:41:42.294517Z","iopub.execute_input":"2022-07-18T07:41:42.295317Z","iopub.status.idle":"2022-07-18T07:41:42.308636Z","shell.execute_reply.started":"2022-07-18T07:41:42.295275Z","shell.execute_reply":"2022-07-18T07:41:42.307596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# one-hot encoding of categorical variables\ntrain = pd.get_dummies(train)\ntest = pd.get_dummies(test)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T07:41:42.309807Z","iopub.execute_input":"2022-07-18T07:41:42.310536Z","iopub.status.idle":"2022-07-18T07:41:42.323881Z","shell.execute_reply.started":"2022-07-18T07:41:42.310508Z","shell.execute_reply":"2022-07-18T07:41:42.322993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 4.3 StandardScaler","metadata":{}},{"cell_type":"code","source":"scaler = StandardScaler()\nscaler = scaler.fit(train.iloc[:,1:])\n\ntrain_new = scaler.transform(train.iloc[:,1:]) \ntest_new = scaler.transform(test) ","metadata":{"execution":{"iopub.status.busy":"2022-07-18T07:41:42.325200Z","iopub.execute_input":"2022-07-18T07:41:42.325914Z","iopub.status.idle":"2022-07-18T07:41:42.339848Z","shell.execute_reply.started":"2022-07-18T07:41:42.325881Z","shell.execute_reply":"2022-07-18T07:41:42.339065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 5. Model","metadata":{}},{"cell_type":"code","source":"# Xtrain, Xtest, Ytrain, Ytest = train_test_split(train.iloc[:,1:],train.iloc[:,0],test_size=0.3,random_state=420)\nXtrain, Xtest, Ytrain, Ytest = train_test_split(train_new,train.iloc[:,0],test_size=0.3,random_state=420)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T07:41:42.340927Z","iopub.execute_input":"2022-07-18T07:41:42.341357Z","iopub.status.idle":"2022-07-18T07:41:42.346804Z","shell.execute_reply.started":"2022-07-18T07:41:42.341332Z","shell.execute_reply":"2022-07-18T07:41:42.346035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 1. RandomForestClassifier\nrfc_model = RandomForestClassifier(n_estimators=38,max_depth=5,max_leaf_nodes=20,random_state=420)\n# 2. XGBClassifier\nxgb_model = XGBClassifier(n_estimators=60,eta=0.05,max_depth=5,gamma=0.5,eval_metric='logloss',random_state=420)\n# 3. SVC  ， kernel = \"linear\"\nsvc_model = SVC(kernel = \"linear\", gamma=\"auto\", cache_size=5000,random_state=420)\n# 4. GaussianNB\ngnb_model = GaussianNB()\n# 5. LogisticRegression\nlr_model = LR(penalty=\"l1\",solver=\"liblinear\",C=0.20,max_iter=1000,random_state=420)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T07:41:42.347988Z","iopub.execute_input":"2022-07-18T07:41:42.348467Z","iopub.status.idle":"2022-07-18T07:41:42.356951Z","shell.execute_reply.started":"2022-07-18T07:41:42.348438Z","shell.execute_reply":"2022-07-18T07:41:42.356044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vt_clf = VotingClassifier(estimators=[\n    ('rfc_model',rfc_model),\n    ('xgb_model', xgb_model),\n#     ('svc_model',svc_model),\n#     ('gnb_model',gnb_model),\n    ('lr_model',lr_model)], \n    voting = 'hard' \n)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T07:41:42.358444Z","iopub.execute_input":"2022-07-18T07:41:42.358719Z","iopub.status.idle":"2022-07-18T07:41:42.365696Z","shell.execute_reply.started":"2022-07-18T07:41:42.358690Z","shell.execute_reply":"2022-07-18T07:41:42.364802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for model in (rfc_model,xgb_model,lr_model,vt_clf):\n    model.fit(Xtrain,Ytrain)\n    y_pre = model.predict(Xtest)\n    print(model.__class__.__name__, accuracy_score(Ytest, y_pre))","metadata":{"execution":{"iopub.status.busy":"2022-07-18T07:41:42.366491Z","iopub.execute_input":"2022-07-18T07:41:42.366730Z","iopub.status.idle":"2022-07-18T07:41:44.423329Z","shell.execute_reply.started":"2022-07-18T07:41:42.366705Z","shell.execute_reply":"2022-07-18T07:41:44.420456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pre = vt_clf.predict(test_new)\ny_pre","metadata":{"execution":{"iopub.status.busy":"2022-07-18T07:41:44.425423Z","iopub.execute_input":"2022-07-18T07:41:44.426632Z","iopub.status.idle":"2022-07-18T07:41:44.456348Z","shell.execute_reply.started":"2022-07-18T07:41:44.426598Z","shell.execute_reply":"2022-07-18T07:41:44.455060Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 6. Submission","metadata":{}},{"cell_type":"code","source":"test = pd.read_csv('../input/titanic/test.csv')\nsubmission = pd.DataFrame({\n    \"PassengerId\" : test[\"PassengerId\"],\n    \"Survived\" : y_pre\n})\nsubmission.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T07:41:44.457885Z","iopub.execute_input":"2022-07-18T07:41:44.458300Z","iopub.status.idle":"2022-07-18T07:41:44.473454Z","shell.execute_reply.started":"2022-07-18T07:41:44.458267Z","shell.execute_reply":"2022-07-18T07:41:44.472547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T07:41:44.478488Z","iopub.execute_input":"2022-07-18T07:41:44.478773Z","iopub.status.idle":"2022-07-18T07:41:44.485412Z","shell.execute_reply.started":"2022-07-18T07:41:44.478746Z","shell.execute_reply":"2022-07-18T07:41:44.484409Z"},"trusted":true},"execution_count":null,"outputs":[]}]}