{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-11T14:11:59.866059Z","iopub.execute_input":"2022-07-11T14:11:59.866459Z","iopub.status.idle":"2022-07-11T14:11:59.877552Z","shell.execute_reply.started":"2022-07-11T14:11:59.866424Z","shell.execute_reply":"2022-07-11T14:11:59.876378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_csv = pd.read_csv(\"/kaggle/input/titanic/train.csv\")\ntest_csv = pd.read_csv(\"/kaggle/input/titanic/test.csv\")\nprint(data_csv.head())","metadata":{"execution":{"iopub.status.busy":"2022-07-11T14:11:59.879589Z","iopub.execute_input":"2022-07-11T14:11:59.880249Z","iopub.status.idle":"2022-07-11T14:11:59.904837Z","shell.execute_reply.started":"2022-07-11T14:11:59.880167Z","shell.execute_reply":"2022-07-11T14:11:59.903582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(data_csv.shape)\nprint(data_csv.isnull().sum())\nprint(test_csv.isnull().sum())","metadata":{"execution":{"iopub.status.busy":"2022-07-11T14:11:59.907674Z","iopub.execute_input":"2022-07-11T14:11:59.908191Z","iopub.status.idle":"2022-07-11T14:11:59.920647Z","shell.execute_reply.started":"2022-07-11T14:11:59.908141Z","shell.execute_reply":"2022-07-11T14:11:59.919560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#dropping Cabin\ndata_csv = data_csv.drop(\"Cabin\",axis=1)\ntest_csv = test_csv.drop(\"Cabin\",axis=1)\n\n#fill nan with mean for Age and mode for Ticket\ndata_csv[\"Age\"].fillna(data_csv[\"Age\"].mean(),inplace=True)\ntest_csv[\"Age\"].fillna(test_csv[\"Age\"].mean(),inplace=True)\ntest_csv[\"Fare\"].fillna(0,inplace=True)\n\n#fill na for categorical value\ndata_csv[\"Embarked\"].fillna(data_csv[\"Embarked\"].mode()[0],inplace=True)\ntest_csv[\"Embarked\"].fillna(test_csv[\"Embarked\"].mode()[0],inplace=True)\nprint(data_csv[\"Embarked\"])\nprint(data_csv.isnull().sum())\nprint(test_csv.isnull().sum())","metadata":{"execution":{"iopub.status.busy":"2022-07-11T14:11:59.923141Z","iopub.execute_input":"2022-07-11T14:11:59.925082Z","iopub.status.idle":"2022-07-11T14:11:59.948414Z","shell.execute_reply.started":"2022-07-11T14:11:59.925031Z","shell.execute_reply":"2022-07-11T14:11:59.947282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#split Family Name and drop Name \ndata_csv[\"FamilyName\"] = data_csv[\"Name\"].str.split(\",\",expand=True)[0]\ndata_csv = data_csv.drop(\"Name\",axis=1)\n\ntest_csv[\"FamilyName\"] = test_csv[\"Name\"].str.split(\",\",expand=True)[0]\ntest_csv = test_csv.drop(\"Name\",axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T14:11:59.951146Z","iopub.execute_input":"2022-07-11T14:11:59.951467Z","iopub.status.idle":"2022-07-11T14:11:59.967354Z","shell.execute_reply.started":"2022-07-11T14:11:59.951439Z","shell.execute_reply":"2022-07-11T14:11:59.966309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Custom function\ndef custom_split(str):\n    if str[0].isnumeric():\n        return str\n    else:\n        a=0\n        i=len(str)\n        while(i>1):\n            if str[i-1]== \" \":\n                a = len(str)-i+1\n                break\n            i = i-1\n        str1 = str[:-a:-1]\n        return str1[::-1]\n\n#Splitting Ticket to get the numeric values\ni=0\nwhile(i<len(data_csv[\"Ticket\"])):\n        data_csv.loc[i,\"Ticket\"] = custom_split(data_csv[\"Ticket\"][i])\n        i = i+1\ndata_csv[\"Ticket\"] = data_csv[\"Ticket\"].replace(\"INE\", \"0\")\n\n#convert to numeric type\ndata_csv[\"Ticket\"] = pd.to_numeric(data_csv[\"Ticket\"])\n\nj=0\nwhile(j<len(test_csv[\"Ticket\"])):\n        test_csv.loc[j,\"Ticket\"] = custom_split(test_csv[\"Ticket\"][j])\n        j = j+1\n        \n#convert to numeric type\ndata_csv[\"Ticket\"] = pd.to_numeric(data_csv[\"Ticket\"])\ntest_csv[\"Ticket\"] = pd.to_numeric(test_csv[\"Ticket\"])\n\nprint(data_csv[\"Ticket\"].head())\nprint(test_csv[\"Ticket\"].head())","metadata":{"execution":{"iopub.status.busy":"2022-07-11T14:11:59.969320Z","iopub.execute_input":"2022-07-11T14:11:59.970825Z","iopub.status.idle":"2022-07-11T14:12:00.431003Z","shell.execute_reply.started":"2022-07-11T14:11:59.970726Z","shell.execute_reply":"2022-07-11T14:12:00.429725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#check correleation\nprint(data_csv.corr())","metadata":{"execution":{"iopub.status.busy":"2022-07-11T14:12:00.433226Z","iopub.execute_input":"2022-07-11T14:12:00.433691Z","iopub.status.idle":"2022-07-11T14:12:00.448970Z","shell.execute_reply.started":"2022-07-11T14:12:00.433645Z","shell.execute_reply":"2022-07-11T14:12:00.447670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#split numeric and categoricl data\nnumeric_data = [column for column in data_csv.select_dtypes([\"int\", \"float\"])]\ncategoric_data = [column for column in data_csv.select_dtypes(exclude=[\"int\",\"float\"])]\n\ncategoric_data1 = [column for column in test_csv.select_dtypes(exclude=[\"int\",\"float\"])]\n\nprint(numeric_data)\nprint(categoric_data)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T14:12:00.450419Z","iopub.execute_input":"2022-07-11T14:12:00.450893Z","iopub.status.idle":"2022-07-11T14:12:00.464170Z","shell.execute_reply.started":"2022-07-11T14:12:00.450848Z","shell.execute_reply":"2022-07-11T14:12:00.463210Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Encoding categoric data\nfrom sklearn.preprocessing import OrdinalEncoder\nlabel = OrdinalEncoder()\ndata_csv[categoric_data] = label.fit_transform(data_csv[categoric_data])\ntest_csv[categoric_data1] = label.fit_transform(test_csv[categoric_data1])\nprint(data_csv[categoric_data])","metadata":{"execution":{"iopub.status.busy":"2022-07-11T14:12:00.465217Z","iopub.execute_input":"2022-07-11T14:12:00.466545Z","iopub.status.idle":"2022-07-11T14:12:01.037434Z","shell.execute_reply.started":"2022-07-11T14:12:00.466508Z","shell.execute_reply":"2022-07-11T14:12:01.036174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Scaling the data\nfrom sklearn.preprocessing import StandardScaler\nscaler = StandardScaler()\nnumeric_data.remove(\"Survived\")\ndata_csv[numeric_data] = scaler.fit_transform(data_csv[numeric_data])\nprint(data_csv[numeric_data])\n\ntest_csv[numeric_data] = scaler.fit_transform(test_csv[numeric_data])","metadata":{"execution":{"iopub.status.busy":"2022-07-11T14:12:01.039006Z","iopub.execute_input":"2022-07-11T14:12:01.039345Z","iopub.status.idle":"2022-07-11T14:12:01.073602Z","shell.execute_reply.started":"2022-07-11T14:12:01.039315Z","shell.execute_reply":"2022-07-11T14:12:01.072495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#check correleation\nprint(data_csv.corr())","metadata":{"execution":{"iopub.status.busy":"2022-07-11T14:12:01.074870Z","iopub.execute_input":"2022-07-11T14:12:01.075187Z","iopub.status.idle":"2022-07-11T14:12:01.088162Z","shell.execute_reply.started":"2022-07-11T14:12:01.075158Z","shell.execute_reply":"2022-07-11T14:12:01.087167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#split X and Y\nX = data_csv.drop(\"Survived\",axis=1)\nY = data_csv[\"Survived\"]\n\nfrom sklearn.model_selection import train_test_split\nx_train,x_test,y_train,y_test = train_test_split(X,Y,test_size=0.3)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T14:12:01.089473Z","iopub.execute_input":"2022-07-11T14:12:01.089831Z","iopub.status.idle":"2022-07-11T14:12:01.159631Z","shell.execute_reply.started":"2022-07-11T14:12:01.089798Z","shell.execute_reply":"2022-07-11T14:12:01.158209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Check accuracy of 3 models\nfrom sklearn.ensemble import RandomForestClassifier \nfrom sklearn.metrics import accuracy_score\nfrom xgboost import XGBClassifier\nfrom sklearn.tree import DecisionTreeClassifier\n\nforest_model = RandomForestClassifier()\nforest_model.fit(x_train,y_train)\nprint(\"Random Forest accuracy score: \",accuracy_score(y_test,forest_model.predict(x_test)))\n\nxgb_model = XGBClassifier(learning_rate =0.1,max_depth=5,colsample_bytree=0.8,seed=27)\nxgb_model.fit(x_train,y_train)\nprint(\"XGB Classifier accuracy score: \",accuracy_score(y_test,xgb_model.predict(x_test)))\n\ndecision_model = DecisionTreeClassifier()\ndecision_model.fit(x_train,y_train)\nprint(\"Decision Tree Classifier: \",accuracy_score(y_test,decision_model.predict(x_test)))","metadata":{"execution":{"iopub.status.busy":"2022-07-11T14:12:01.163811Z","iopub.execute_input":"2022-07-11T14:12:01.164369Z","iopub.status.idle":"2022-07-11T14:12:02.129638Z","shell.execute_reply.started":"2022-07-11T14:12:01.164327Z","shell.execute_reply":"2022-07-11T14:12:02.128666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Testing on real data taking good accuracy model\ny_predict = forest_model.predict(test_csv)\nprint(y_predict)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T14:12:02.133540Z","iopub.execute_input":"2022-07-11T14:12:02.135100Z","iopub.status.idle":"2022-07-11T14:12:02.168208Z","shell.execute_reply.started":"2022-07-11T14:12:02.135038Z","shell.execute_reply":"2022-07-11T14:12:02.166845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_csv1 = pd.read_csv(\"/kaggle/input/titanic/test.csv\")\noutputDF = pd.DataFrame({'PassengerId': test_csv1['PassengerId'], 'Survived': y_predict})\nprint(outputDF.head())","metadata":{"execution":{"iopub.status.busy":"2022-07-11T14:12:02.169651Z","iopub.execute_input":"2022-07-11T14:12:02.170077Z","iopub.status.idle":"2022-07-11T14:12:02.187460Z","shell.execute_reply.started":"2022-07-11T14:12:02.170046Z","shell.execute_reply":"2022-07-11T14:12:02.186574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"outputDF.to_csv(\"Survival_Prediction.csv\",index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T14:12:02.188716Z","iopub.execute_input":"2022-07-11T14:12:02.189252Z","iopub.status.idle":"2022-07-11T14:12:02.197453Z","shell.execute_reply.started":"2022-07-11T14:12:02.189221Z","shell.execute_reply":"2022-07-11T14:12:02.196316Z"},"trusted":true},"execution_count":null,"outputs":[]}]}