{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom matplotlib.pyplot import rcParams\nimport os\nfrom sklearn import tree\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.model_selection import GridSearchCV, cross_val_score\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import accuracy_score, f1_score, precision_score, recall_score\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-11T15:03:01.071651Z","iopub.execute_input":"2022-08-11T15:03:01.072074Z","iopub.status.idle":"2022-08-11T15:03:01.085274Z","shell.execute_reply.started":"2022-08-11T15:03:01.072041Z","shell.execute_reply":"2022-08-11T15:03:01.083937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data_ex = pd.read_csv('../input/titanic/train.csv')\ntest_data_ex = pd.read_csv('../input/titanic/test.csv')\n\n# deep copy to copy data and indeces\ntrain_ex = train_data_ex.copy(deep=True)\ntest_ex = test_data_ex.copy(deep=True)\n\n#Dealing with missing values\n#Dropping the column 'Cabin' as it has too many null values \n#Dropping the column 'Name'\ntrain_ex = train_ex.drop(['Name','Cabin'], axis=1)\ntest_ex = test_ex.drop(['Name','Cabin'], axis=1)\n\n# filling the nan values for Age and fare column with the mean \ncombined_data = [train_ex, test_ex]\nfor data in combined_data:\n    data.Age.fillna(data.Age.mean(), inplace = True)\n    data.Fare.fillna(data.Fare.mean(), inplace = True)\n    \n#filling the nan values of Embarked column with most_frequent value ('s')\ntrain_ex['Embarked'] = train_ex['Embarked'].fillna('S')\n    \n\n#Categorical variables:[Sex', 'Ticket', 'Embarked']\n\n#Let's start by converting Sex feature to categorical female=1 and male=0\n#train_ex.Sex = train_ex.Sex.map({'female':1, 'male':0})\n#test_ex.Sex = test_ex.Sex.map({'female':1, 'male':0})\n\n#using map funcion to change the Embarked column S = 1, C = 2, Q = 0\n#change = {'S':1,'C':2,'Q':0}\n#train_ex.Embarked = train_ex.Embarked.map(change)\n#test_ex.Embarked = test_ex.Embarked.map(change)\n\n#After using Hot Encoding , I got better results for all coming models\nobject_cols = ['Sex', 'Embarked']\n\nfrom sklearn.preprocessing import OneHotEncoder\n# Apply one-hot encoder to each column with categorical data\nOH_encoder = OneHotEncoder(handle_unknown='ignore', sparse=False)\nOH_cols_train = pd.DataFrame(OH_encoder.fit_transform(train_ex[object_cols]))\nOH_cols_test = pd.DataFrame(OH_encoder.transform(test_ex[object_cols]))\n# One-hot encoding removed index; put it back\nOH_cols_train.index = train_ex.index\nOH_cols_test.index = test_ex.index\n\n# Remove categorical columns (will replace with one-hot encoding)\nnum_X_train = train_ex.drop(object_cols, axis=1)\nnum_X_valid = test_ex.drop(object_cols, axis=1)\n\n# Add one-hot encoded columns to numerical features\ntrain_ex = pd.concat([num_X_train, OH_cols_train], axis=1)\ntest_ex = pd.concat([num_X_valid, OH_cols_test], axis=1)\n\nprint(train_ex.isnull().sum())\nprint(test_ex.isnull().sum())\n\n#Dropping PassengerId and Ticket column\ncolumns_to_drop = ['PassengerId','Ticket']\ntrain_ex.drop(columns_to_drop, axis = 1, inplace = True)\ntest_ex.drop(columns_to_drop[1], axis = 1, inplace = True)\n\nX_train_ex = train_ex.drop(\"Survived\", axis=1)\nY_train_ex = train_ex[\"Survived\"]\nX_test_ex = test_ex.drop(\"PassengerId\", axis = 1)\nprint(\"shape of X_train\",X_train_ex.shape)\nprint(\"Shape of Y_train\",Y_train_ex.shape)\nprint(\"Shape of x_test\",X_test_ex.shape)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T15:03:01.282985Z","iopub.execute_input":"2022-08-11T15:03:01.283540Z","iopub.status.idle":"2022-08-11T15:03:02.024187Z","shell.execute_reply.started":"2022-08-11T15:03:01.283496Z","shell.execute_reply":"2022-08-11T15:03:02.022662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nimport keras \nfrom keras.layers import Dense, Dropout, Input\nfrom keras.models import Sequential\nfrom tensorflow import keras\nfrom tensorflow.keras import layers, callbacks\n\n#Defining model\nmodel = Sequential()\nmodel.add(Dense(units = 32, input_shape = (10,), activation = 'relu'))\nmodel.add(Dense(units = 64, activation = 'relu', kernel_initializer = 'he_normal', use_bias = False))\nmodel.add(tf.keras.layers.BatchNormalization())\nmodel.add(Dense(units = 128, activation = 'relu',kernel_initializer = 'he_normal', use_bias = False))\nmodel.add(Dropout(0.1))\nmodel.add(Dense(units = 64, activation = 'relu',kernel_initializer = 'he_normal', use_bias = False))\nmodel.add(Dropout(0.1))\nmodel.add(Dense(units = 32, activation = 'relu'))\nmodel.add(Dropout(0.15))\nmodel.add(Dense(units = 16, activation = 'relu'))\nmodel.add(Dense(units = 8, activation = 'relu',kernel_initializer = 'he_normal', use_bias = False))\nmodel.add(Dense(units =1 , activation = 'sigmoid'))\n\nmodel.compile(loss = tf.keras.losses.binary_crossentropy, optimizer = tf.keras.optimizers.Adam(),metrics = ['acc'])\nmodel.fit(X_train_ex, Y_train_ex, batch_size = 32, verbose = 2, epochs = 50)\n\npredict_ex = model.predict(X_test_ex)\n#since we have use sigmoid activation function in output layer\npredict_ex = (predict_ex > 0.5).astype(int).ravel()\nprint(predict_ex)\n\nfrom sklearn import metrics\nY_pred_rand_ex= (model.predict(X_train_ex) > 0.5).astype(int)\nDeep_learning_acc= np.round(metrics.accuracy_score(Y_train_ex, Y_pred_rand_ex),2)\nprint('Accuracy with Hot encoding : ', Deep_learning_acc)\n\nsubmit = pd.DataFrame({\"PassengerId\":test_ex.PassengerId, 'Survived':predict_ex})\nsubmit.to_csv(\"submission.csv\",index = False)\nprint(\"Successfull Submission\")\n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T15:03:02.028986Z","iopub.execute_input":"2022-08-11T15:03:02.029631Z","iopub.status.idle":"2022-08-11T15:03:13.393638Z","shell.execute_reply.started":"2022-08-11T15:03:02.029577Z","shell.execute_reply":"2022-08-11T15:03:13.391940Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Decision Tree model\nfrom sklearn import tree\nX_train_ex = train_ex.drop(\"Survived\", axis=1)\nY_train_ex = train_ex[\"Survived\"]\nX_test_ex = test_ex.drop(\"PassengerId\", axis = 1)\ntrain_X, val_X, train_y, val_y = train_test_split(X_train_ex, Y_train_ex,test_size=0.15, random_state = 0)\nprint(train_X.shape)\nprint(train_y.shape)\ndecision_tree = tree.DecisionTreeClassifier(random_state = 0)\ndecision_tree.fit(train_X, train_y)\nY_pred = decision_tree.predict(val_X)\nacc_decision_tree = accuracy_score(val_y,Y_pred)\nprint('Accuracy score for validation data is:', acc_decision_tree)\n\nY_pred_test = decision_tree.predict(X_test_ex)\nsubmit = pd.DataFrame({\"PassengerId\":test_ex.PassengerId, 'Survived':Y_pred_test})\nsubmit.to_csv(\"submission.csv\",index = False)\nprint(\"Successfull Submission\")\n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T15:03:13.395841Z","iopub.execute_input":"2022-08-11T15:03:13.396377Z","iopub.status.idle":"2022-08-11T15:03:13.437542Z","shell.execute_reply.started":"2022-08-11T15:03:13.396315Z","shell.execute_reply":"2022-08-11T15:03:13.435994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Random Forest model\nfrom sklearn.tree import DecisionTreeRegressor\nX_train_ex = train_ex.drop(\"Survived\", axis=1)\nY_train_ex = train_ex[\"Survived\"]\nX_test_ex = test_ex.drop(\"PassengerId\", axis = 1)\nX_train_r, X_val_r, y_train_r, y_val_r = train_test_split(X_train_ex, Y_train_ex, test_size=0.17, random_state = 0)\nrdmf = RandomForestClassifier(n_estimators=20, criterion='entropy')\nrdmf.fit(X_train_r, y_train_r)\nY_pred = rdmf.predict(X_val_r)\nacc_random_forest = accuracy_score(y_val_r,Y_pred)\nprint('Accuracy score for validation data is:', acc_random_forest )\n\nY_pred_test = rdmf.predict(X_test_ex)\nsubmit = pd.DataFrame({\"PassengerId\":test_ex.PassengerId, 'Survived':Y_pred_test})\nsubmit.to_csv(\"submission.csv\",index = False)\nprint(\"Successfull Submission\")","metadata":{"execution":{"iopub.status.busy":"2022-08-11T15:03:13.440674Z","iopub.execute_input":"2022-08-11T15:03:13.441334Z","iopub.status.idle":"2022-08-11T15:03:13.544719Z","shell.execute_reply.started":"2022-08-11T15:03:13.441291Z","shell.execute_reply":"2022-08-11T15:03:13.542909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#K- Nearest Neighbours KNN\nfrom sklearn.neighbors import KNeighborsClassifier\nknn = KNeighborsClassifier(p=2, n_neighbors=17)\nknn.fit(X_train_r, y_train_r)\nY_pred = knn.predict(X_val_r)\nacc_knn =accuracy_score(y_val_r,Y_pred)\nprint('Accuracy score for validation data is:', acc_knn)\n\nY_pred_test = knn.predict(X_test_ex)\nsubmit = pd.DataFrame({\"PassengerId\":test_ex.PassengerId, 'Survived':Y_pred_test})\nsubmit.to_csv(\"submission.csv\",index = False)\nprint(\"Successfull Submission\")\n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T15:03:13.546789Z","iopub.execute_input":"2022-08-11T15:03:13.547523Z","iopub.status.idle":"2022-08-11T15:03:13.604898Z","shell.execute_reply.started":"2022-08-11T15:03:13.547468Z","shell.execute_reply":"2022-08-11T15:03:13.603273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from xgboost import XGBClassifier\nxgb = XGBClassifier()\nxgb.fit(X_train_r, y_train_r)\nY_pred = xgb.predict(X_val_r)\nxgb_acc= accuracy_score(y_val_r,Y_pred)\nprint('Accuracy score for validation data is:', xgb_acc)\n\nY_pred_test = xgb.predict(X_test_ex)\nsubmit = pd.DataFrame({\"PassengerId\":test_ex.PassengerId, 'Survived':Y_pred_test})\nsubmit.to_csv(\"submission.csv\",index = False)\nprint(\"Successfull Submission\")\n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T15:03:13.609306Z","iopub.execute_input":"2022-08-11T15:03:13.609811Z","iopub.status.idle":"2022-08-11T15:03:14.092334Z","shell.execute_reply.started":"2022-08-11T15:03:13.609773Z","shell.execute_reply":"2022-08-11T15:03:14.091041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Support Vector Machines\nfrom sklearn import svm\nSVC=svm.SVC(kernel='linear', C=1).fit(X_train_r, y_train_r)\nY_pred = SVC.predict(X_val_r)\nacc_linear_svc =accuracy_score(y_val_r,Y_pred)\nprint('Accuracy score for validation data is:',acc_linear_svc)\n\nY_pred_test = SVC.predict(X_test_ex)\nsubmit = pd.DataFrame({\"PassengerId\":test_ex.PassengerId, 'Survived':Y_pred_test})\nsubmit.to_csv(\"submission.csv\",index = False)\nprint(\"Successfull Submission\")","metadata":{"execution":{"iopub.status.busy":"2022-08-11T15:03:14.122200Z","iopub.execute_input":"2022-08-11T15:03:14.123443Z","iopub.status.idle":"2022-08-11T15:03:24.561476Z","shell.execute_reply.started":"2022-08-11T15:03:14.123390Z","shell.execute_reply":"2022-08-11T15:03:24.559893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Model Evaluation\nmodels = pd.DataFrame({\n    'Model': [ 'KNN','Random Forest','Linear SVC', 'Decision Tree','Deep learning','XGBoost'],\n    'Score': [acc_knn, acc_random_forest, acc_linear_svc, acc_decision_tree ,Deep_learning_acc ,xgb_acc]})\nmodels.sort_values(by='Score', ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T15:06:18.826924Z","iopub.execute_input":"2022-08-11T15:06:18.827444Z","iopub.status.idle":"2022-08-11T15:06:18.853068Z","shell.execute_reply.started":"2022-08-11T15:06:18.827402Z","shell.execute_reply":"2022-08-11T15:06:18.851847Z"},"trusted":true},"execution_count":null,"outputs":[]}]}