{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-03T21:54:39.950841Z","iopub.execute_input":"2022-08-03T21:54:39.951227Z","iopub.status.idle":"2022-08-03T21:54:39.963600Z","shell.execute_reply.started":"2022-08-03T21:54:39.951196Z","shell.execute_reply":"2022-08-03T21:54:39.962180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Import Libraries","metadata":{}},{"cell_type":"code","source":"import random\nimport pandas as pd \nimport numpy as np\nimport math as ma\nimport tensorflow as tf\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.metrics import confusion_matrix,recall_score\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.preprocessing import LabelEncoder\n\n#Read the Data\ndf=pd.read_csv('../input/titanic/train.csv')\ndf_test=pd.read_csv('../input/titanic/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T21:54:39.997112Z","iopub.execute_input":"2022-08-03T21:54:39.997656Z","iopub.status.idle":"2022-08-03T21:54:40.014410Z","shell.execute_reply.started":"2022-08-03T21:54:39.997628Z","shell.execute_reply":"2022-08-03T21:54:40.013406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Display our data","metadata":{}},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2022-08-03T21:54:40.032005Z","iopub.execute_input":"2022-08-03T21:54:40.032875Z","iopub.status.idle":"2022-08-03T21:54:40.055255Z","shell.execute_reply.started":"2022-08-03T21:54:40.032841Z","shell.execute_reply":"2022-08-03T21:54:40.054426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Do some analysis on our data\n* show the number of null values in every column in the dataset","metadata":{}},{"cell_type":"code","source":"print(df.isna().sum())\nprint(df_test.isna().sum())","metadata":{"execution":{"iopub.status.busy":"2022-08-03T21:54:40.087748Z","iopub.execute_input":"2022-08-03T21:54:40.090901Z","iopub.status.idle":"2022-08-03T21:54:40.102123Z","shell.execute_reply.started":"2022-08-03T21:54:40.090847Z","shell.execute_reply":"2022-08-03T21:54:40.100545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Describe the data\n* show some statistics about the data","metadata":{}},{"cell_type":"code","source":"df.describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T21:54:40.126179Z","iopub.execute_input":"2022-08-03T21:54:40.127151Z","iopub.status.idle":"2022-08-03T21:54:40.156600Z","shell.execute_reply.started":"2022-08-03T21:54:40.127121Z","shell.execute_reply":"2022-08-03T21:54:40.155496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data preprocessing\n* drop unuseful columns \n* fill null values of embarked column\n* divide my data to features and label\n* normalize the data for easy training \n","metadata":{}},{"cell_type":"code","source":"df.drop(columns=['Cabin','Age', 'PassengerId','Name','Ticket','SibSp','Parch'],inplace=True)\n\n\n\ndf[\"Embarked\"].fillna( random.choice(df[df['Embarked'] != np.nan][\"Embarked\"]), inplace =True)\n#divide data to training and label\nX=df.iloc[ : ,1:].values\ny=df.iloc[:,0]\n\nle = LabelEncoder()\nX[:, 1] = le.fit_transform(X[:, 1])\nX[:, 0] = le.fit_transform(X[:, 0])\nX[:, -1] = le.fit_transform(X[:, -1])\n\n\nsc = StandardScaler()\nX = sc.fit_transform(X)\nprint(X)\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-03T21:54:40.163327Z","iopub.execute_input":"2022-08-03T21:54:40.163622Z","iopub.status.idle":"2022-08-03T21:54:40.178126Z","shell.execute_reply.started":"2022-08-03T21:54:40.163597Z","shell.execute_reply":"2022-08-03T21:54:40.177329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T21:54:40.215831Z","iopub.execute_input":"2022-08-03T21:54:40.217050Z","iopub.status.idle":"2022-08-03T21:54:40.226499Z","shell.execute_reply.started":"2022-08-03T21:54:40.217001Z","shell.execute_reply":"2022-08-03T21:54:40.225682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Spliting the data\n* split my data to training and testing sets (.8 for training : .2 for testing)","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size = 0.2, random_state = 25)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-03T22:06:25.227057Z","iopub.execute_input":"2022-08-03T22:06:25.227427Z","iopub.status.idle":"2022-08-03T22:06:25.233977Z","shell.execute_reply.started":"2022-08-03T22:06:25.227398Z","shell.execute_reply":"2022-08-03T22:06:25.233134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Redo the same steps on test dataset (used only in submission)","metadata":{}},{"cell_type":"code","source":"\ndf_test.drop(columns=['Cabin','Age', 'PassengerId','Name','Ticket','SibSp','Parch'],inplace=True)\n\n\ndf_test['Fare'].fillna((df_test['Fare'].mean()), inplace=True)\ndf_test[\"Embarked\"].fillna( random.choice(df_test[df_test['Embarked'] != np.nan][\"Embarked\"]), inplace =True)\n\nsub_test=df_test\n\n\n\n\nle = LabelEncoder()\nsub_test.iloc[:, 1] = le.fit_transform(sub_test.iloc[:, 1])\nsub_test.iloc[:, 0] = le.fit_transform(sub_test.iloc[:, 0])\nsub_test.iloc[:, -1] = le.fit_transform(sub_test.iloc[:, -1])\n\n\n# Feature Scaling\nsc = StandardScaler()\nsub_test = sc.fit_transform(sub_test)\nprint(sub_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T21:54:40.328298Z","iopub.execute_input":"2022-08-03T21:54:40.328953Z","iopub.status.idle":"2022-08-03T21:54:40.346058Z","shell.execute_reply.started":"2022-08-03T21:54:40.328903Z","shell.execute_reply":"2022-08-03T21:54:40.344484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Visualization  \n* pairplots of every column in the data\n* boxplot of fare column to detect outliers\n* display correlation matrix of the data\n* use AutoViz library for more visualization","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\nsns.pairplot(df)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-03T21:54:40.796042Z","iopub.execute_input":"2022-08-03T21:54:40.797416Z","iopub.status.idle":"2022-08-03T21:54:42.495548Z","shell.execute_reply.started":"2022-08-03T21:54:40.797365Z","shell.execute_reply":"2022-08-03T21:54:42.493922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"boxplot = df.boxplot(column=[ 'Fare']) ","metadata":{"execution":{"iopub.status.busy":"2022-08-03T21:54:42.497840Z","iopub.execute_input":"2022-08-03T21:54:42.498615Z","iopub.status.idle":"2022-08-03T21:54:42.621592Z","shell.execute_reply.started":"2022-08-03T21:54:42.498580Z","shell.execute_reply":"2022-08-03T21:54:42.620814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nplt.figure(figsize=(40,15))\na=sns.heatmap(df.corr(),annot=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T21:54:42.623146Z","iopub.execute_input":"2022-08-03T21:54:42.623766Z","iopub.status.idle":"2022-08-03T21:54:42.937976Z","shell.execute_reply.started":"2022-08-03T21:54:42.623727Z","shell.execute_reply":"2022-08-03T21:54:42.936981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install autoviz\n","metadata":{"execution":{"iopub.status.busy":"2022-08-03T21:54:42.939606Z","iopub.execute_input":"2022-08-03T21:54:42.940005Z","iopub.status.idle":"2022-08-03T21:54:52.367155Z","shell.execute_reply.started":"2022-08-03T21:54:42.939967Z","shell.execute_reply":"2022-08-03T21:54:52.365987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from autoviz.AutoViz_Class import AutoViz_Class\nAV = AutoViz_Class()\ndf_av = AV.AutoViz('../input/titanic/train.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T21:54:52.371264Z","iopub.execute_input":"2022-08-03T21:54:52.371638Z","iopub.status.idle":"2022-08-03T21:54:56.493002Z","shell.execute_reply.started":"2022-08-03T21:54:52.371609Z","shell.execute_reply":"2022-08-03T21:54:56.491994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Fiting and testing data","metadata":{}},{"cell_type":"markdown","source":"# Use logistic regression\n<img src=\"https://pimages.toolbox.com/wp-content/uploads/2022/04/11040522/46-4.png\">","metadata":{}},{"cell_type":"code","source":"\nfrom sklearn.linear_model import LogisticRegression\n\nmodel=LogisticRegression()\n\nmodel.fit(X_train,y_train)\n\ny_pred = model.predict(X_test)\n\n#display model score on test data\nscore=model.score(X_test,y_test)\nprint(score)\n# Making the Confusion Matrix\ncm = confusion_matrix(y_test, y_pred)\nprint(cm)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-03T22:06:40.948530Z","iopub.execute_input":"2022-08-03T22:06:40.949496Z","iopub.status.idle":"2022-08-03T22:06:40.966059Z","shell.execute_reply.started":"2022-08-03T22:06:40.949408Z","shell.execute_reply":"2022-08-03T22:06:40.964783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Knearest neighbor \n<img src=\"https://static.javatpoint.com/tutorial/machine-learning/images/k-nearest-neighbor-algorithm-for-machine-learning3.png\">","metadata":{}},{"cell_type":"code","source":"from sklearn.neighbors import KNeighborsClassifier\nknn=KNeighborsClassifier(n_neighbors=12)\nknn.fit(X_train,y_train)\n\ny_pred_knn = knn.predict(X_test)\n#display score on test data\nscore=knn.score(X_test,y_test)\nprint(score)\n# Making the Confusion Matrix\ncm = confusion_matrix(y_test, y_pred_knn)\nprint(cm)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-03T22:07:46.424677Z","iopub.execute_input":"2022-08-03T22:07:46.425841Z","iopub.status.idle":"2022-08-03T22:07:46.445423Z","shell.execute_reply.started":"2022-08-03T22:07:46.425803Z","shell.execute_reply":"2022-08-03T22:07:46.444682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Support victor machine rbf kernel\n<img src=\"https://www.researchgate.net/publication/225415339/figure/fig1/AS:648614107955200@1531653062199/Decision-boundary-by-SVM-with-RBF-kernel-function.png\">","metadata":{}},{"cell_type":"code","source":"from sklearn.svm import SVC\nrbf_svc = SVC(kernel='rbf')\nrbf_svc.fit(X,y)\ny_pred_svm=rbf_svc.predict(X_test)\n\n#display score on test data\nscore=rbf_svc.score(X_test,y_test)\nprint(score)\n#confusion matrix\ncm = confusion_matrix(y_test, y_pred_svm)\nprint(cm)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T22:08:36.097360Z","iopub.execute_input":"2022-08-03T22:08:36.097731Z","iopub.status.idle":"2022-08-03T22:08:36.126512Z","shell.execute_reply.started":"2022-08-03T22:08:36.097700Z","shell.execute_reply":"2022-08-03T22:08:36.125581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Naive bayes \n<img src=\"https://www.ayush-mandowara.in/static/6ef9d890b17d0cf6a0a05c94d41cc426/4ca57/naive_bayes.png\">","metadata":{}},{"cell_type":"code","source":"from sklearn.naive_bayes import GaussianNB\nnb=GaussianNB()\nnb.fit(X_train,y_train)\ny_pred_nb=nb.predict(X_test)\n#display score on test data\nscore=nb.score(X_test,y_test)\nprint(score)\n#confusion matrix\ncm = confusion_matrix(y_test, y_pred_nb)\nprint(cm)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T22:08:48.183909Z","iopub.execute_input":"2022-08-03T22:08:48.184726Z","iopub.status.idle":"2022-08-03T22:08:48.193822Z","shell.execute_reply.started":"2022-08-03T22:08:48.184692Z","shell.execute_reply":"2022-08-03T22:08:48.192850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Decision Tree Classifier\n<img src='https://www.researchgate.net/profile/Nick-Bassiliades/publication/337413360/figure/fig2/AS:827514800836608@1574306311071/A-simple-decision-tree-classifier-with-4-features.ppm'>","metadata":{}},{"cell_type":"code","source":"from sklearn.tree import DecisionTreeClassifier\ndecision_tree = DecisionTreeClassifier()\ndecision_tree.fit(X_train, y_train)\ny_pred_dt = decision_tree.predict(X_test)\n#display score on test data\nscore=decision_tree.score(X_test,y_test)\nprint(score)\n\n#confusion matrix\ncm = confusion_matrix(y_test, y_pred_dt)\nprint(cm)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T22:08:54.746958Z","iopub.execute_input":"2022-08-03T22:08:54.747331Z","iopub.status.idle":"2022-08-03T22:08:54.758360Z","shell.execute_reply.started":"2022-08-03T22:08:54.747300Z","shell.execute_reply":"2022-08-03T22:08:54.756916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\nrandom_forest = RandomForestClassifier(n_estimators=7)\nrandom_forest.fit(X_train, y_train)\ny_pred_rf = random_forest.predict(X_test)\n#display score on test data\nscore=random_forest.score(X_test,y_test)\nprint(score)\n#confusion matrix\ncm = confusion_matrix(y_test, y_pred_rf)\nprint(cm)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T22:08:59.234915Z","iopub.execute_input":"2022-08-03T22:08:59.235834Z","iopub.status.idle":"2022-08-03T22:08:59.259443Z","shell.execute_reply.started":"2022-08-03T22:08:59.235784Z","shell.execute_reply":"2022-08-03T22:08:59.258502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import AdaBoostClassifier\nfrom sklearn.ensemble import BaggingClassifier\nbagging = BaggingClassifier(AdaBoostClassifier(n_estimators=3),max_samples=0.5, max_features=0.5)\nbagging.fit(X_train,y_train)\ny_pred = bagging.predict(X_test)\n\nscore=bagging.score(X_test,y_test)\nprint(score)\n# Making the Confusion Matrix\ncm = confusion_matrix(y_test, y_pred)\nprint(cm)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-03T22:09:12.717484Z","iopub.execute_input":"2022-08-03T22:09:12.717820Z","iopub.status.idle":"2022-08-03T22:09:12.790012Z","shell.execute_reply.started":"2022-08-03T22:09:12.717794Z","shell.execute_reply":"2022-08-03T22:09:12.788735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.svm import SVC\nfrom sklearn.ensemble import BaggingClassifier\nbagging_svm = BaggingClassifier(SVC(kernel='rbf'))\nbagging_svm.fit(X,y)\n\nscore=bagging_svm.score(X,y)\nprint(score)\n# Making the Confusion Matrix\n","metadata":{"execution":{"iopub.status.busy":"2022-08-03T22:09:16.370360Z","iopub.execute_input":"2022-08-03T22:09:16.370739Z","iopub.status.idle":"2022-08-03T22:09:16.561656Z","shell.execute_reply.started":"2022-08-03T22:09:16.370709Z","shell.execute_reply":"2022-08-03T22:09:16.560665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\nbagging = BaggingClassifier(RandomForestClassifier(n_estimators=11))\nbagging.fit(X_train,y_train)\ny_pred_bagsvm = bagging.predict(X_test)\n\nscore=bagging.score(X_test,y_test)\nprint(score)\n# Making the Confusion Matrix\ncm = confusion_matrix(y_test, y_pred_bagsvm)\nprint(cm)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T22:09:36.077064Z","iopub.execute_input":"2022-08-03T22:09:36.077819Z","iopub.status.idle":"2022-08-03T22:09:36.281532Z","shell.execute_reply.started":"2022-08-03T22:09:36.077788Z","shell.execute_reply":"2022-08-03T22:09:36.280564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# As we show the best score is support victor machine classifier","metadata":{}},{"cell_type":"code","source":"final=rbf_svc.predict(sub_test)\nfinal","metadata":{"execution":{"iopub.status.busy":"2022-08-03T22:09:45.781064Z","iopub.execute_input":"2022-08-03T22:09:45.781428Z","iopub.status.idle":"2022-08-03T22:09:45.795822Z","shell.execute_reply.started":"2022-08-03T22:09:45.781399Z","shell.execute_reply":"2022-08-03T22:09:45.794486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Save submission ","metadata":{}},{"cell_type":"code","source":"test = pd.read_csv(r\"../input/titanic/test.csv\")\nsubmission = pd.DataFrame({\n    \"PassengerId\" : test[\"PassengerId\"],\n    \"Survived\" : final\n})\nsubmission.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T22:09:50.069496Z","iopub.execute_input":"2022-08-03T22:09:50.069847Z","iopub.status.idle":"2022-08-03T22:09:50.086138Z","shell.execute_reply.started":"2022-08-03T22:09:50.069819Z","shell.execute_reply":"2022-08-03T22:09:50.085115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv', index = False)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-03T22:09:53.922526Z","iopub.execute_input":"2022-08-03T22:09:53.922952Z","iopub.status.idle":"2022-08-03T22:09:53.930207Z","shell.execute_reply.started":"2022-08-03T22:09:53.922900Z","shell.execute_reply":"2022-08-03T22:09:53.929096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# If you have any note on my notebook you are welcome","metadata":{}}]}