{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-11T08:05:43.588058Z","iopub.execute_input":"2022-07-11T08:05:43.589318Z","iopub.status.idle":"2022-07-11T08:05:43.602698Z","shell.execute_reply.started":"2022-07-11T08:05:43.589219Z","shell.execute_reply":"2022-07-11T08:05:43.601522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#read data\ndata_csv = pd.read_csv(\"/kaggle/input/spaceship-titanic/train.csv\")\ntest_data = pd.read_csv(\"/kaggle/input/spaceship-titanic/test.csv\")\n\n#check the data\nprint(data_csv.head())","metadata":{"execution":{"iopub.status.busy":"2022-07-11T08:05:43.607248Z","iopub.execute_input":"2022-07-11T08:05:43.607627Z","iopub.status.idle":"2022-07-11T08:05:43.667701Z","shell.execute_reply.started":"2022-07-11T08:05:43.607590Z","shell.execute_reply":"2022-07-11T08:05:43.666647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#split Name and Cabin\ndata_csv[\"FirstName\"] = data_csv[\"Name\"].str.split(\" \", expand=True)[0]\ndata_csv[\"Family\"] = data_csv[\"Name\"].str.split(\" \", expand=True)[1]\n\ntest_data[\"FirstName\"] = test_data[\"Name\"].str.split(\" \", expand=True)[0]\ntest_data[\"Family\"] = test_data[\"Name\"].str.split(\" \", expand=True)[1]\n\ndata_csv[\"CabinDeck\"] = data_csv[\"Cabin\"].str.split(\"/\", expand=True)[0]\ndata_csv[\"CabinNum\"] = data_csv[\"Cabin\"].str.split(\"/\", expand=True)[1]\ndata_csv[\"CabinSide\"] = data_csv[\"Cabin\"].str.split(\"/\", expand=True)[2]\n\ntest_data[\"CabinDeck\"] = test_data[\"Cabin\"].str.split(\"/\", expand=True)[0]\ntest_data[\"CabinNum\"] = test_data[\"Cabin\"].str.split(\"/\", expand=True)[1]\ntest_data[\"CabinSide\"] = test_data[\"Cabin\"].str.split(\"/\", expand=True)[2]\n\nprint(data_csv.head())","metadata":{"execution":{"iopub.status.busy":"2022-07-11T08:05:43.669764Z","iopub.execute_input":"2022-07-11T08:05:43.670744Z","iopub.status.idle":"2022-07-11T08:05:43.864421Z","shell.execute_reply.started":"2022-07-11T08:05:43.670696Z","shell.execute_reply":"2022-07-11T08:05:43.863107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Check for null values\nprint(data_csv.isnull().sum())","metadata":{"execution":{"iopub.status.busy":"2022-07-11T08:05:43.866189Z","iopub.execute_input":"2022-07-11T08:05:43.866615Z","iopub.status.idle":"2022-07-11T08:05:43.883260Z","shell.execute_reply.started":"2022-07-11T08:05:43.866574Z","shell.execute_reply":"2022-07-11T08:05:43.882171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#drop Name and Cabin\ndata_csv = data_csv.drop([\"Name\",\"Cabin\"],axis=1)\ntest_data = test_data.drop([\"Name\",\"Cabin\"],axis=1)\nprint(data_csv.head())","metadata":{"execution":{"iopub.status.busy":"2022-07-11T08:05:43.885487Z","iopub.execute_input":"2022-07-11T08:05:43.885792Z","iopub.status.idle":"2022-07-11T08:05:43.908792Z","shell.execute_reply.started":"2022-07-11T08:05:43.885766Z","shell.execute_reply":"2022-07-11T08:05:43.907566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_csv.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T08:05:43.911841Z","iopub.execute_input":"2022-07-11T08:05:43.912171Z","iopub.status.idle":"2022-07-11T08:05:43.930277Z","shell.execute_reply.started":"2022-07-11T08:05:43.912141Z","shell.execute_reply":"2022-07-11T08:05:43.929484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Change data type of Cabin Number\ndata_csv[\"CabinNum\"] = pd.to_numeric(data_csv[\"CabinNum\"])\ntest_data[\"CabinNum\"] = pd.to_numeric(test_data[\"CabinNum\"])","metadata":{"execution":{"iopub.status.busy":"2022-07-11T08:05:43.931378Z","iopub.execute_input":"2022-07-11T08:05:43.932185Z","iopub.status.idle":"2022-07-11T08:05:43.946792Z","shell.execute_reply.started":"2022-07-11T08:05:43.932144Z","shell.execute_reply":"2022-07-11T08:05:43.946000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_csv.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T08:05:43.947846Z","iopub.execute_input":"2022-07-11T08:05:43.948771Z","iopub.status.idle":"2022-07-11T08:05:43.965293Z","shell.execute_reply.started":"2022-07-11T08:05:43.948738Z","shell.execute_reply":"2022-07-11T08:05:43.964478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Seperate numeric and categoric data\nnumeric_data = [column for column in  data_csv.select_dtypes([\"int\", \"float\"])]\ncategoric_data = [column for column in data_csv.select_dtypes(exclude = [\"int\", \"float\"])]\ncategoric_data1 = [column for column in test_data.select_dtypes(exclude = [\"int\", \"float\"])]\nprint(categoric_data)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T08:05:43.966454Z","iopub.execute_input":"2022-07-11T08:05:43.967243Z","iopub.status.idle":"2022-07-11T08:05:43.980456Z","shell.execute_reply.started":"2022-07-11T08:05:43.967212Z","shell.execute_reply":"2022-07-11T08:05:43.979646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#fill na for numeric data\nfor i in numeric_data:\n    data_csv[i].fillna(data_csv[i].median(),inplace=True)\n    test_data[i].fillna(test_data[i].median(),inplace=True)\n    \n#fill na for categoric data\nfor i in categoric_data:\n    data_csv[i].fillna(data_csv[i].value_counts().index[0],inplace=True)\n    \nfor i in categoric_data1:\n    test_data[i].fillna(test_data[i].value_counts().index[0],inplace=True)\n    \nprint(data_csv.isnull().sum())","metadata":{"execution":{"iopub.status.busy":"2022-07-11T08:05:43.981711Z","iopub.execute_input":"2022-07-11T08:05:43.982253Z","iopub.status.idle":"2022-07-11T08:05:44.037158Z","shell.execute_reply.started":"2022-07-11T08:05:43.982217Z","shell.execute_reply":"2022-07-11T08:05:44.036332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#checking correleation\nprint(data_csv.corr())","metadata":{"execution":{"iopub.status.busy":"2022-07-11T08:05:44.039600Z","iopub.execute_input":"2022-07-11T08:05:44.040477Z","iopub.status.idle":"2022-07-11T08:05:44.055266Z","shell.execute_reply.started":"2022-07-11T08:05:44.040442Z","shell.execute_reply":"2022-07-11T08:05:44.054459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Encoding the categoric data\nfrom sklearn.preprocessing import OrdinalEncoder\nencoder = OrdinalEncoder()\ndata_csv[categoric_data] = encoder.fit_transform(data_csv[categoric_data])\ntest_data[categoric_data1] = encoder.fit_transform(test_data[categoric_data1])\n\nprint(data_csv[categoric_data])","metadata":{"execution":{"iopub.status.busy":"2022-07-11T08:05:44.056198Z","iopub.execute_input":"2022-07-11T08:05:44.056875Z","iopub.status.idle":"2022-07-11T08:05:44.528353Z","shell.execute_reply.started":"2022-07-11T08:05:44.056846Z","shell.execute_reply":"2022-07-11T08:05:44.527179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Scaling the data\nfrom sklearn.preprocessing import StandardScaler\nscaler = StandardScaler()\ndata_csv[numeric_data] = scaler.fit_transform(data_csv[numeric_data])\ntest_data[numeric_data]= scaler.fit_transform(test_data[numeric_data])\n\nprint(data_csv[numeric_data])","metadata":{"execution":{"iopub.status.busy":"2022-07-11T08:05:44.530009Z","iopub.execute_input":"2022-07-11T08:05:44.530702Z","iopub.status.idle":"2022-07-11T08:05:44.551372Z","shell.execute_reply.started":"2022-07-11T08:05:44.530666Z","shell.execute_reply":"2022-07-11T08:05:44.550538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Checking correlation\nprint(data_csv.corr())","metadata":{"execution":{"iopub.status.busy":"2022-07-11T08:05:44.552569Z","iopub.execute_input":"2022-07-11T08:05:44.553074Z","iopub.status.idle":"2022-07-11T08:05:44.576791Z","shell.execute_reply.started":"2022-07-11T08:05:44.553044Z","shell.execute_reply":"2022-07-11T08:05:44.575624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#split X and Y\nX = data_csv.drop(\"Transported\",axis=1)\nY = data_csv[\"Transported\"]\n\n#split train and test data\nfrom sklearn.model_selection import train_test_split\nx_train, x_test, y_train, y_test = train_test_split(X,Y,test_size=0.3)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T08:05:44.578119Z","iopub.execute_input":"2022-07-11T08:05:44.578450Z","iopub.status.idle":"2022-07-11T08:05:44.601377Z","shell.execute_reply.started":"2022-07-11T08:05:44.578421Z","shell.execute_reply":"2022-07-11T08:05:44.600245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Check accuracy score from 3 models\nfrom sklearn.ensemble import RandomForestClassifier \nfrom sklearn.metrics import accuracy_score\nfrom xgboost import XGBClassifier\n\nmodel = XGBClassifier(learning_rate =0.1,max_depth=5,colsample_bytree=0.8,seed=27)\nmodel.fit(x_train,y_train)\nprint(\"XGB \",accuracy_score(y_test,model.predict(x_test)))\n\nmodel1 = RandomForestClassifier(n_estimators=100)\nmodel1.fit(x_train,y_train)\nprint(\"Random Forest \",accuracy_score(y_test,model1.predict(x_test)))\n\nfrom sklearn.tree import DecisionTreeClassifier\nmodel3 = DecisionTreeClassifier()\nmodel3.fit(x_train,y_train)\nprint(\"Decision Tree \",accuracy_score(y_test,model3.predict(x_test)))","metadata":{"execution":{"iopub.status.busy":"2022-07-11T08:05:44.602720Z","iopub.execute_input":"2022-07-11T08:05:44.603040Z","iopub.status.idle":"2022-07-11T08:05:46.486575Z","shell.execute_reply.started":"2022-07-11T08:05:44.603013Z","shell.execute_reply":"2022-07-11T08:05:46.485378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#fit whole data for best model\nmodel_prod = XGBClassifier()\nmodel_prod.fit(X,Y)\ny_predict = pd.Series(model_prod.predict(test_data)).map({0:False,1:True})\nprint(y_predict)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T08:05:46.487803Z","iopub.execute_input":"2022-07-11T08:05:46.488092Z","iopub.status.idle":"2022-07-11T08:05:47.439413Z","shell.execute_reply.started":"2022-07-11T08:05:46.488067Z","shell.execute_reply":"2022-07-11T08:05:47.438466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#output the values to DF\ntest_data1 = pd.read_csv(\"/kaggle/input/spaceship-titanic/test.csv\")\noutput = pd.DataFrame({\"PassengerId\":test_data1.PassengerId,\"Transported\":y_predict})\nprint(output)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T08:05:47.440799Z","iopub.execute_input":"2022-07-11T08:05:47.441331Z","iopub.status.idle":"2022-07-11T08:05:47.463843Z","shell.execute_reply.started":"2022-07-11T08:05:47.441301Z","shell.execute_reply":"2022-07-11T08:05:47.463073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#output to CSV\noutput.to_csv(\"Submission.csv\",index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T08:05:47.464936Z","iopub.execute_input":"2022-07-11T08:05:47.465253Z","iopub.status.idle":"2022-07-11T08:05:47.477597Z","shell.execute_reply.started":"2022-07-11T08:05:47.465226Z","shell.execute_reply":"2022-07-11T08:05:47.476478Z"},"trusted":true},"execution_count":null,"outputs":[]}]}