{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-10T23:36:50.812674Z","iopub.execute_input":"2022-08-10T23:36:50.815720Z","iopub.status.idle":"2022-08-10T23:36:50.826491Z","shell.execute_reply.started":"2022-08-10T23:36:50.815633Z","shell.execute_reply":"2022-08-10T23:36:50.825100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = pd.read_csv('/kaggle/input/titanic/train.csv') # read the train data\ntrain_data.tail() # display the data","metadata":{"execution":{"iopub.status.busy":"2022-08-10T23:36:50.832796Z","iopub.execute_input":"2022-08-10T23:36:50.833331Z","iopub.status.idle":"2022-08-10T23:36:50.874021Z","shell.execute_reply.started":"2022-08-10T23:36:50.833295Z","shell.execute_reply":"2022-08-10T23:36:50.872896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data = pd.read_csv('/kaggle/input/titanic/test.csv') # read the test data\ntest_data.head() # display the data","metadata":{"execution":{"iopub.status.busy":"2022-08-10T23:36:50.875756Z","iopub.execute_input":"2022-08-10T23:36:50.876568Z","iopub.status.idle":"2022-08-10T23:36:50.903063Z","shell.execute_reply.started":"2022-08-10T23:36:50.876514Z","shell.execute_reply":"2022-08-10T23:36:50.902055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.isnull().sum() # count nulls in train_data","metadata":{"execution":{"iopub.status.busy":"2022-08-10T23:36:50.904616Z","iopub.execute_input":"2022-08-10T23:36:50.904965Z","iopub.status.idle":"2022-08-10T23:36:50.915327Z","shell.execute_reply.started":"2022-08-10T23:36:50.904930Z","shell.execute_reply":"2022-08-10T23:36:50.914260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data.isnull().sum() # count nulls in test_data","metadata":{"execution":{"iopub.status.busy":"2022-08-10T23:36:50.916819Z","iopub.execute_input":"2022-08-10T23:36:50.917282Z","iopub.status.idle":"2022-08-10T23:36:50.931545Z","shell.execute_reply.started":"2022-08-10T23:36:50.917219Z","shell.execute_reply":"2022-08-10T23:36:50.930465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = train_data.drop(columns='Cabin', axis=1) # remove the cabin column as there is so many nulls \ntrain_data[\"Age\"].fillna(train_data[\"Age\"].mean(), inplace = True) # filling the empty values in \"Age\" column\ntrain_data.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T23:36:50.935480Z","iopub.execute_input":"2022-08-10T23:36:50.935833Z","iopub.status.idle":"2022-08-10T23:36:50.949303Z","shell.execute_reply.started":"2022-08-10T23:36:50.935798Z","shell.execute_reply":"2022-08-10T23:36:50.948310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data # print the train_data","metadata":{"execution":{"iopub.status.busy":"2022-08-10T23:36:50.950907Z","iopub.execute_input":"2022-08-10T23:36:50.951240Z","iopub.status.idle":"2022-08-10T23:36:50.983897Z","shell.execute_reply.started":"2022-08-10T23:36:50.951206Z","shell.execute_reply":"2022-08-10T23:36:50.982765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data = test_data.drop(columns='Cabin', axis=1) # remove the cabin column as there is so many nulls \ntest_data[\"Age\"].fillna(test_data[\"Age\"].mean(), inplace = True) # filling the empty values in \"Age\" column\ntest_data[\"Fare\"].fillna(test_data[\"Fare\"].mean(), inplace = True) # filling Fare N/A value with the mean of all values\ntest_data.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T23:36:50.985529Z","iopub.execute_input":"2022-08-10T23:36:50.985993Z","iopub.status.idle":"2022-08-10T23:36:51.003551Z","shell.execute_reply.started":"2022-08-10T23:36:50.985957Z","shell.execute_reply":"2022-08-10T23:36:51.002675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data","metadata":{"execution":{"iopub.status.busy":"2022-08-10T23:36:51.004634Z","iopub.execute_input":"2022-08-10T23:36:51.004973Z","iopub.status.idle":"2022-08-10T23:36:51.030950Z","shell.execute_reply.started":"2022-08-10T23:36:51.004938Z","shell.execute_reply":"2022-08-10T23:36:51.029749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = train_data.drop(columns = ['PassengerId','Name','Ticket','Survived'], axis=1) # drop unwanted columns\nX_train[\"Sex\"] = pd.get_dummies(train_data.Sex) # change Sex Column to numercal values\nX_train[\"Embarked\"] = pd.get_dummies(train_data.Embarked) # change Embarked Column to numercal values\nX_train","metadata":{"execution":{"iopub.status.busy":"2022-08-10T23:36:51.032496Z","iopub.execute_input":"2022-08-10T23:36:51.033311Z","iopub.status.idle":"2022-08-10T23:36:51.059726Z","shell.execute_reply.started":"2022-08-10T23:36:51.033273Z","shell.execute_reply":"2022-08-10T23:36:51.058855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test_submit = test_data.drop(columns = ['PassengerId','Name','Ticket'], axis=1) # drop unwanted columns\nX_test_submit[\"Sex\"] = pd.get_dummies(test_data.Sex) # change Sex Column to numercal values\nX_test_submit[\"Embarked\"] = pd.get_dummies(test_data.Embarked) # change Embarked Column to numercal values\nX_test_submit","metadata":{"execution":{"iopub.status.busy":"2022-08-10T23:36:51.060854Z","iopub.execute_input":"2022-08-10T23:36:51.061309Z","iopub.status.idle":"2022-08-10T23:36:51.089162Z","shell.execute_reply.started":"2022-08-10T23:36:51.061277Z","shell.execute_reply":"2022-08-10T23:36:51.087993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y = train_data[\"Survived\"] # get the Y in a separate variable \nY","metadata":{"execution":{"iopub.status.busy":"2022-08-10T23:36:52.179280Z","iopub.execute_input":"2022-08-10T23:36:52.179856Z","iopub.status.idle":"2022-08-10T23:36:52.187711Z","shell.execute_reply.started":"2022-08-10T23:36:52.179819Z","shell.execute_reply":"2022-08-10T23:36:52.186458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_test, Y_train, Y_test = train_test_split(X_train,Y, test_size=0.3, random_state=1) # split the data ","metadata":{"execution":{"iopub.status.busy":"2022-08-10T23:36:52.190275Z","iopub.execute_input":"2022-08-10T23:36:52.190612Z","iopub.status.idle":"2022-08-10T23:36:52.657612Z","shell.execute_reply.started":"2022-08-10T23:36:52.190579Z","shell.execute_reply":"2022-08-10T23:36:52.656264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# First using Decision Tree Classifier\n\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.metrics import accuracy_score\n\ntree_classifier = DecisionTreeClassifier(criterion=\"entropy\", random_state = 100, max_depth = 3) # define the classifier (max_depth = 3 gave the best accuracy)\ntree_classifier.fit(X_train,Y_train) # fit the classifier\nout = tree_classifier.predict(X_test) # calculate predictions\nprint(accuracy_score(Y_test,out))\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-10T23:36:52.659487Z","iopub.execute_input":"2022-08-10T23:36:52.659834Z","iopub.status.idle":"2022-08-10T23:36:52.718336Z","shell.execute_reply.started":"2022-08-10T23:36:52.659801Z","shell.execute_reply":"2022-08-10T23:36:52.716363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = tree_classifier.predict(X_test_submit) # calculate predictions for submission\noutput = pd.DataFrame({\"PassengerId\": test_data.PassengerId, \"Survived\" : predictions}) # calculate predictions for submission\noutput.to_csv('DecisionTreeClassifier.csv', index = False) # import the result into .csv file\n","metadata":{"execution":{"iopub.status.busy":"2022-08-10T23:36:52.721215Z","iopub.execute_input":"2022-08-10T23:36:52.722045Z","iopub.status.idle":"2022-08-10T23:36:52.734406Z","shell.execute_reply.started":"2022-08-10T23:36:52.721995Z","shell.execute_reply":"2022-08-10T23:36:52.733248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Second using Random Forest Classifier\n\nfrom sklearn.ensemble import RandomForestClassifier\n\n\nrandom_forest_classifier = RandomForestClassifier(n_estimators = 400, max_depth = 7, random_state= 1) # define the classifier (mac_depth = 7 gave the best accuracy)\nrandom_forest_classifier.fit(X_train,Y_train) # fit the classifier\nout2 = random_forest_classifier.predict(X_test) # make predictions\nprint(accuracy_score(Y_test,out2))\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-10T23:36:52.736966Z","iopub.execute_input":"2022-08-10T23:36:52.737481Z","iopub.status.idle":"2022-08-10T23:36:53.503988Z","shell.execute_reply.started":"2022-08-10T23:36:52.737434Z","shell.execute_reply":"2022-08-10T23:36:53.503040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions2 = random_forest_classifier.predict(X_test_submit) # calculate predictions for submission\noutput2 = pd.DataFrame({\"PassengerId\": test_data.PassengerId, \"Survived\" : predictions2}) # calculate predictions for submission\noutput2.to_csv('RandomForestClassifier.csv', index = False)# import the result into .csv file","metadata":{"execution":{"iopub.status.busy":"2022-08-10T23:36:53.505865Z","iopub.execute_input":"2022-08-10T23:36:53.506329Z","iopub.status.idle":"2022-08-10T23:36:53.518347Z","shell.execute_reply.started":"2022-08-10T23:36:53.506283Z","shell.execute_reply":"2022-08-10T23:36:53.517207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}