{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-12T05:10:49.373055Z","iopub.execute_input":"2022-07-12T05:10:49.373871Z","iopub.status.idle":"2022-07-12T05:10:49.390235Z","shell.execute_reply.started":"2022-07-12T05:10:49.373742Z","shell.execute_reply":"2022-07-12T05:10:49.389120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def age_band(df):\n    l = []\n    for i in range(len(df)):\n        if df['Age'][i]>=0 and df['Age'][i]<=12:\n            l.append(\"Child\")\n        if df['Age'][i]>12 and df['Age'][i]<=19:\n            l.append(\"Adolescence\")\n        if df['Age'][i]>19 and df['Age'][i]<=59:\n            l.append(\"Adult\")\n        if df['Age'][i]>59:\n            l.append(\"Senior\")\n    return l \n\ndef skew(df):\n    print('Column \\t       Skewness  ')\n    print(\"=\"*30)\n    for i in df.columns:\n        if df[i].dtype == 'O':\n            continue\n        print(\"{}\\t{}\".format(i.ljust(15,' '),round(df[i].skew(),3)))\n\n# Function to calculate accuracy\ndef cal_accuracy(y_test, y_pred):\n      \n    print(\"Confusion Matrix: \",\n        confusion_matrix(y_test, y_pred))\n      \n    print (\"Accuracy : \",\n    accuracy_score(y_test,y_pred)*100)\n      \n    print(\"Report : \",\n    classification_report(y_test, y_pred))","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:10:49.391451Z","iopub.execute_input":"2022-07-12T05:10:49.391868Z","iopub.status.idle":"2022-07-12T05:10:49.404780Z","shell.execute_reply.started":"2022-07-12T05:10:49.391821Z","shell.execute_reply":"2022-07-12T05:10:49.403587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom scipy import stats\nfrom sklearn.metrics import confusion_matrix, accuracy_score, classification_report","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:10:50.481588Z","iopub.execute_input":"2022-07-12T05:10:50.482772Z","iopub.status.idle":"2022-07-12T05:10:51.150362Z","shell.execute_reply.started":"2022-07-12T05:10:50.482717Z","shell.execute_reply":"2022-07-12T05:10:51.149364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv(\"/kaggle/input/spaceship-titanic/train.csv\")\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:10:52.811770Z","iopub.execute_input":"2022-07-12T05:10:52.812766Z","iopub.status.idle":"2022-07-12T05:10:52.875871Z","shell.execute_reply.started":"2022-07-12T05:10:52.812711Z","shell.execute_reply":"2022-07-12T05:10:52.875072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Basic Info of the data","metadata":{}},{"cell_type":"code","source":"df_train.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:10:54.984444Z","iopub.execute_input":"2022-07-12T05:10:54.984949Z","iopub.status.idle":"2022-07-12T05:10:55.013990Z","shell.execute_reply.started":"2022-07-12T05:10:54.984909Z","shell.execute_reply":"2022-07-12T05:10:55.012346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = pd.read_csv(\"/kaggle/input/spaceship-titanic/test.csv\")\nid_pass= df_test['PassengerId']","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:10:55.915709Z","iopub.execute_input":"2022-07-12T05:10:55.916523Z","iopub.status.idle":"2022-07-12T05:10:55.942996Z","shell.execute_reply.started":"2022-07-12T05:10:55.916477Z","shell.execute_reply":"2022-07-12T05:10:55.942090Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:10:56.391308Z","iopub.execute_input":"2022-07-12T05:10:56.392220Z","iopub.status.idle":"2022-07-12T05:10:56.429300Z","shell.execute_reply.started":"2022-07-12T05:10:56.392155Z","shell.execute_reply":"2022-07-12T05:10:56.428109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:10:56.871772Z","iopub.execute_input":"2022-07-12T05:10:56.872267Z","iopub.status.idle":"2022-07-12T05:10:56.879743Z","shell.execute_reply.started":"2022-07-12T05:10:56.872230Z","shell.execute_reply":"2022-07-12T05:10:56.878543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:10:57.272442Z","iopub.execute_input":"2022-07-12T05:10:57.272952Z","iopub.status.idle":"2022-07-12T05:10:57.291966Z","shell.execute_reply.started":"2022-07-12T05:10:57.272914Z","shell.execute_reply":"2022-07-12T05:10:57.291014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('column \\t       Missing Percentage  ')   #Missing values in terms of percentage\nprint('='*30)\nfor col in df_train.columns:\n    percentage_col = (df_train[col].isnull().sum()/len(df_train))*100\n    \n    print('{}\\t{} % '.format(col.ljust(15,' '), round(percentage_col, 3)))","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:10:57.741480Z","iopub.execute_input":"2022-07-12T05:10:57.742417Z","iopub.status.idle":"2022-07-12T05:10:57.762921Z","shell.execute_reply.started":"2022-07-12T05:10:57.742364Z","shell.execute_reply":"2022-07-12T05:10:57.761508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['CryoSleep'].mode()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:10:58.245070Z","iopub.execute_input":"2022-07-12T05:10:58.245569Z","iopub.status.idle":"2022-07-12T05:10:58.255135Z","shell.execute_reply.started":"2022-07-12T05:10:58.245532Z","shell.execute_reply":"2022-07-12T05:10:58.254362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# filling missing values for train and test data set\n\ndf_train['Age'] = df_train['Age'].fillna(df_train['Age'].mean())\ndf_train['RoomService'] =df_train['RoomService'].fillna(df_train['RoomService'].mean())\ndf_train['FoodCourt']= df_train['FoodCourt'].fillna(df_train['FoodCourt'].mean())\ndf_train['ShoppingMall']=df_train['ShoppingMall'].fillna(df_train['ShoppingMall'].mean())\ndf_train['Spa']= df_train['Spa'].fillna(df_train['Spa'].mean())\ndf_train['VRDeck']= df_train['VRDeck'].fillna(df_train['VRDeck'].mean())\n\ndf_train['HomePlanet'] =df_train['HomePlanet'].fillna('Earth')\ndf_train['CryoSleep'] =df_train['CryoSleep'].fillna('False')\ndf_train['Cabin'] =df_train['Cabin'].fillna(\"G/734/S\")\ndf_train['Destination'] =df_train['Destination'].fillna(\"TRAPPIST-1e\")\ndf_train['VIP'] =df_train['VIP'].fillna('False')\n\ndf_train = df_train.drop(['PassengerId','Name'],axis=1)\n\n\n# Processing the test dataset\ndf_test = pd.read_csv(\"/kaggle/input/spaceship-titanic/test.csv\")\n# Building Baseline Model\ndf_test['Age'] = df_test['Age'].fillna(df_test['Age'].mean())\ndf_test['RoomService'] =df_test['RoomService'].fillna(df_test['RoomService'].mean())\ndf_test['FoodCourt']= df_test['FoodCourt'].fillna(df_test['FoodCourt'].mean())\ndf_test['ShoppingMall']=df_test['ShoppingMall'].fillna(df_test['ShoppingMall'].mean())\ndf_test['Spa']= df_test['Spa'].fillna(df_test['Spa'].mean())\ndf_test['VRDeck']= df_test['VRDeck'].fillna(df_test['VRDeck'].mean())\n\ndf_test['HomePlanet'] =df_test['HomePlanet'].fillna('Earth')\ndf_test['CryoSleep'] =df_test['CryoSleep'].fillna('False')\ndf_test['Cabin'] =df_test['Cabin'].fillna(\"G/734/S\")\ndf_test['Destination'] =df_test['Destination'].fillna(\"TRAPPIST-1e\")\ndf_test['VIP'] =df_test['VIP'].fillna('False')\n\ndf_test = df_test.drop(['PassengerId','Name'],axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:10:58.639684Z","iopub.execute_input":"2022-07-12T05:10:58.640797Z","iopub.status.idle":"2022-07-12T05:10:58.700338Z","shell.execute_reply.started":"2022-07-12T05:10:58.640732Z","shell.execute_reply":"2022-07-12T05:10:58.699374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:10:59.100096Z","iopub.execute_input":"2022-07-12T05:10:59.100866Z","iopub.status.idle":"2022-07-12T05:10:59.116983Z","shell.execute_reply.started":"2022-07-12T05:10:59.100802Z","shell.execute_reply":"2022-07-12T05:10:59.115865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.rcParams['figure.figsize'] = (18, 12)\nplt.subplot(2, 4, 1)\nsns.distplot(df_train['Age'], color = 'red')\nplt.grid()\n\nplt.subplot(2, 4, 2)\nsns.distplot(df_train['RoomService'], color = 'black')\nplt.grid()\n\nplt.subplot(2, 4, 3)\nsns.distplot(df_train['FoodCourt'], color = 'black')\nplt.grid()\nplt.subplot(2, 4, 4)\nsns.distplot(df_train['Spa'], color = 'red')\nplt.grid()\nplt.subplot(2, 4, 5)\nsns.distplot(df_train['ShoppingMall'], color = 'red')\nplt.grid()\nplt.subplot(2, 4, 6)\nsns.distplot(df_train['VRDeck'], color = 'red')\nplt.grid()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:10:59.990327Z","iopub.execute_input":"2022-07-12T05:10:59.990820Z","iopub.status.idle":"2022-07-12T05:11:01.713756Z","shell.execute_reply.started":"2022-07-12T05:10:59.990780Z","shell.execute_reply":"2022-07-12T05:11:01.712784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#visualizing test data set\nplt.rcParams['figure.figsize'] = (18, 12)\nplt.subplot(2, 4, 1)\nsns.distplot(df_test['Age'], color = 'red')\nplt.grid()\n\nplt.subplot(2, 4, 2)\nsns.distplot(df_test['RoomService'], color = 'black')\nplt.grid()\n\nplt.subplot(2, 4, 3)\nsns.distplot(df_test['FoodCourt'], color = 'black')\nplt.grid()\nplt.subplot(2, 4, 4)\nsns.distplot(df_test['Spa'], color = 'red')\nplt.grid()\nplt.subplot(2, 4, 5)\nsns.distplot(df_test['ShoppingMall'], color = 'red')\nplt.grid()\nplt.subplot(2, 4, 6)\nsns.distplot(df_test['VRDeck'], color = 'red')\nplt.grid()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:11:01.715369Z","iopub.execute_input":"2022-07-12T05:11:01.715936Z","iopub.status.idle":"2022-07-12T05:11:03.534129Z","shell.execute_reply.started":"2022-07-12T05:11:01.715897Z","shell.execute_reply":"2022-07-12T05:11:03.533040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nsns.catplot(y=\"HomePlanet\", hue=\"Transported\", kind=\"count\",\n            palette=\"pastel\", edgecolor=\".6\",\n            data=df_train)\n\nsns.catplot(y=\"CryoSleep\", hue=\"Transported\", kind=\"count\",\n            palette=\"pastel\", edgecolor=\".6\",\n            data=df_train)\n\nsns.catplot(y=\"Destination\", hue=\"Transported\", kind=\"count\",\n            palette=\"pastel\", edgecolor=\".6\",\n            data=df_train)\n\nsns.catplot(y=\"VIP\", hue=\"Transported\", kind=\"count\",\n            palette=\"pastel\", edgecolor=\".6\",\n            data=df_train)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:11:03.535267Z","iopub.execute_input":"2022-07-12T05:11:03.535603Z","iopub.status.idle":"2022-07-12T05:11:04.978627Z","shell.execute_reply.started":"2022-07-12T05:11:03.535574Z","shell.execute_reply":"2022-07-12T05:11:04.977380Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:11:04.981088Z","iopub.execute_input":"2022-07-12T05:11:04.981614Z","iopub.status.idle":"2022-07-12T05:11:04.998308Z","shell.execute_reply.started":"2022-07-12T05:11:04.981566Z","shell.execute_reply":"2022-07-12T05:11:04.997459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['Age1'] = age_band(df_train)\ndf_test['Age1']= age_band(df_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:11:05.631564Z","iopub.execute_input":"2022-07-12T05:11:05.632616Z","iopub.status.idle":"2022-07-12T05:11:06.350744Z","shell.execute_reply.started":"2022-07-12T05:11:05.632571Z","shell.execute_reply":"2022-07-12T05:11:06.349572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train and test sets have similar distrubution","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:11:06.827325Z","iopub.execute_input":"2022-07-12T05:11:06.827820Z","iopub.status.idle":"2022-07-12T05:11:06.832153Z","shell.execute_reply.started":"2022-07-12T05:11:06.827775Z","shell.execute_reply":"2022-07-12T05:11:06.831229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in df_train.columns:\n    if df_train[i].dtype == 'O':\n        continue\n    df_train[i] = np.sqrt(np.sqrt(df_train[i]))\n\nfor i in df_test.columns:\n    if df_test[i].dtype == 'O':\n        continue\n    df_test[i] = np.sqrt(np.sqrt(df_test[i]))\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:11:09.165662Z","iopub.execute_input":"2022-07-12T05:11:09.166702Z","iopub.status.idle":"2022-07-12T05:11:09.182361Z","shell.execute_reply.started":"2022-07-12T05:11:09.166653Z","shell.execute_reply":"2022-07-12T05:11:09.180785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Again ploting test data to check the differnece\n#visualizing test data set\nplt.rcParams['figure.figsize'] = (18, 12)\nplt.subplot(2, 4, 1)\nsns.distplot(df_test['Age'], color = 'red')\nplt.grid()\n\nplt.subplot(2, 4, 2)\nsns.distplot(df_test['RoomService'], color = 'black')\nplt.grid()\n\nplt.subplot(2, 4, 3)\nsns.distplot(df_test['FoodCourt'], color = 'black')\nplt.grid()\nplt.subplot(2, 4, 4)\nsns.distplot(df_test['Spa'], color = 'red')\nplt.grid()\nplt.subplot(2, 4, 5)\nsns.distplot(df_test['ShoppingMall'], color = 'red')\nplt.grid()\nplt.subplot(2, 4, 6)\nsns.distplot(df_test['VRDeck'], color = 'red')\nplt.grid()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:11:10.618997Z","iopub.execute_input":"2022-07-12T05:11:10.619871Z","iopub.status.idle":"2022-07-12T05:11:11.958440Z","shell.execute_reply.started":"2022-07-12T05:11:10.619823Z","shell.execute_reply":"2022-07-12T05:11:11.957658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['Destination'].unique()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:11:11.959997Z","iopub.execute_input":"2022-07-12T05:11:11.960462Z","iopub.status.idle":"2022-07-12T05:11:11.966915Z","shell.execute_reply.started":"2022-07-12T05:11:11.960430Z","shell.execute_reply":"2022-07-12T05:11:11.966017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['Age1'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:11:13.078856Z","iopub.execute_input":"2022-07-12T05:11:13.080055Z","iopub.status.idle":"2022-07-12T05:11:13.092105Z","shell.execute_reply.started":"2022-07-12T05:11:13.080010Z","shell.execute_reply":"2022-07-12T05:11:13.090739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = df_train.astype({\"CryoSleep\":'bool', \"VIP\":'bool'})\ndf_test = df_test.astype({\"CryoSleep\":'bool', \"VIP\":'bool'})","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:11:14.589730Z","iopub.execute_input":"2022-07-12T05:11:14.590862Z","iopub.status.idle":"2022-07-12T05:11:14.605936Z","shell.execute_reply.started":"2022-07-12T05:11:14.590816Z","shell.execute_reply":"2022-07-12T05:11:14.604703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bool_map = {True:1,False:0}\nplanet = {\"Europa\":1,'Earth':2,'Mars':3}\ndest  = {\"TRAPPIST-1e\":1,\"PSO J318.5-22\":2,\"55 Cancri e\":3}\nage = {\"Adult\":1,\"Adolescence\":2,\"Child\":3,\"Senior\":4}","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:11:15.638556Z","iopub.execute_input":"2022-07-12T05:11:15.639117Z","iopub.status.idle":"2022-07-12T05:11:15.646892Z","shell.execute_reply.started":"2022-07-12T05:11:15.639064Z","shell.execute_reply":"2022-07-12T05:11:15.645668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head(1)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:11:17.823906Z","iopub.execute_input":"2022-07-12T05:11:17.824376Z","iopub.status.idle":"2022-07-12T05:11:17.845291Z","shell.execute_reply.started":"2022-07-12T05:11:17.824339Z","shell.execute_reply":"2022-07-12T05:11:17.843923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['HomePlanet'] = df_train['HomePlanet'].map(planet)\ndf_train['CryoSleep'] = df_train['CryoSleep'].map(bool_map)\ndf_train['Destination'] =df_train['Destination'].map(dest)\ndf_train['VIP'] = df_train['VIP'].map(bool_map)\ndf_train['Age1']=df_train['Age1'].map(age)\n\ndf_test['HomePlanet'] = df_test['HomePlanet'].map(planet)\ndf_test['CryoSleep'] = df_test['CryoSleep'].map(bool_map)\ndf_test['Destination'] =df_test['Destination'].map(dest)\ndf_test['VIP'] = df_test['VIP'].map(bool_map)\ndf_test['Age1']=df_test['Age1'].map(age)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:11:18.339425Z","iopub.execute_input":"2022-07-12T05:11:18.340708Z","iopub.status.idle":"2022-07-12T05:11:18.369204Z","shell.execute_reply.started":"2022-07-12T05:11:18.340631Z","shell.execute_reply":"2022-07-12T05:11:18.367984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corr = df_train.corr()\nsns.heatmap(corr,annot =True)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:11:18.987340Z","iopub.execute_input":"2022-07-12T05:11:18.987829Z","iopub.status.idle":"2022-07-12T05:11:20.104387Z","shell.execute_reply.started":"2022-07-12T05:11:18.987792Z","shell.execute_reply":"2022-07-12T05:11:20.103388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corr['Transported']","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:11:20.106085Z","iopub.execute_input":"2022-07-12T05:11:20.106940Z","iopub.status.idle":"2022-07-12T05:11:20.114890Z","shell.execute_reply.started":"2022-07-12T05:11:20.106904Z","shell.execute_reply":"2022-07-12T05:11:20.113858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = df_train.drop('Cabin',axis=1)\ndf_test = df_test.drop('Cabin',axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:11:20.116237Z","iopub.execute_input":"2022-07-12T05:11:20.116961Z","iopub.status.idle":"2022-07-12T05:11:20.129208Z","shell.execute_reply.started":"2022-07-12T05:11:20.116916Z","shell.execute_reply":"2022-07-12T05:11:20.128432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:11:22.089182Z","iopub.execute_input":"2022-07-12T05:11:22.090044Z","iopub.status.idle":"2022-07-12T05:11:22.101629Z","shell.execute_reply.started":"2022-07-12T05:11:22.090007Z","shell.execute_reply":"2022-07-12T05:11:22.100414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# BaseLine Model Building\n#from sklearn import cross_validation\nX = df_train.drop(\"Transported\",axis=1)\ny = df_train['Transported']\n\nX_train, X_test, y_train, y_test = train_test_split(X,y,test_size = 0.25)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:11:24.023747Z","iopub.execute_input":"2022-07-12T05:11:24.025122Z","iopub.status.idle":"2022-07-12T05:11:24.036877Z","shell.execute_reply.started":"2022-07-12T05:11:24.025058Z","shell.execute_reply":"2022-07-12T05:11:24.035677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#got baseline result now we will work on improving the result","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:11:24.541564Z","iopub.execute_input":"2022-07-12T05:11:24.542121Z","iopub.status.idle":"2022-07-12T05:11:24.547147Z","shell.execute_reply.started":"2022-07-12T05:11:24.542080Z","shell.execute_reply":"2022-07-12T05:11:24.545771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:11:26.700232Z","iopub.execute_input":"2022-07-12T05:11:26.700773Z","iopub.status.idle":"2022-07-12T05:11:26.713045Z","shell.execute_reply.started":"2022-07-12T05:11:26.700733Z","shell.execute_reply":"2022-07-12T05:11:26.711740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\nfrom sklearn.model_selection import GridSearchCV","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:11:27.324225Z","iopub.execute_input":"2022-07-12T05:11:27.324764Z","iopub.status.idle":"2022-07-12T05:11:27.386724Z","shell.execute_reply.started":"2022-07-12T05:11:27.324723Z","shell.execute_reply":"2022-07-12T05:11:27.385873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rfc = RandomForestClassifier()\nforest_params = [{'max_depth': list(range(10, 15)), 'max_features': list(range(0,22))}]\n\nclf = GridSearchCV(rfc, forest_params, cv = 10, scoring='accuracy')\n\nclf.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:11:28.230397Z","iopub.execute_input":"2022-07-12T05:11:28.231198Z","iopub.status.idle":"2022-07-12T05:20:13.480809Z","shell.execute_reply.started":"2022-07-12T05:11:28.231157Z","shell.execute_reply":"2022-07-12T05:20:13.479582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_rf = clf.predict(X_test)\ncal_accuracy(y_test,y_pred_rf)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:20:13.482890Z","iopub.execute_input":"2022-07-12T05:20:13.483522Z","iopub.status.idle":"2022-07-12T05:20:13.563933Z","shell.execute_reply.started":"2022-07-12T05:20:13.483475Z","shell.execute_reply":"2022-07-12T05:20:13.562922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_rf","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:20:13.565104Z","iopub.execute_input":"2022-07-12T05:20:13.565500Z","iopub.status.idle":"2022-07-12T05:20:13.570925Z","shell.execute_reply.started":"2022-07-12T05:20:13.565471Z","shell.execute_reply":"2022-07-12T05:20:13.570205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = clf.predict(df_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:20:13.572926Z","iopub.execute_input":"2022-07-12T05:20:13.573437Z","iopub.status.idle":"2022-07-12T05:20:13.656299Z","shell.execute_reply.started":"2022-07-12T05:20:13.573402Z","shell.execute_reply":"2022-07-12T05:20:13.655430Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:20:13.657829Z","iopub.execute_input":"2022-07-12T05:20:13.658136Z","iopub.status.idle":"2022-07-12T05:20:13.664356Z","shell.execute_reply.started":"2022-07-12T05:20:13.658109Z","shell.execute_reply":"2022-07-12T05:20:13.663430Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bool_map = {1:True,0.0:False}","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:20:13.665873Z","iopub.execute_input":"2022-07-12T05:20:13.666482Z","iopub.status.idle":"2022-07-12T05:20:13.675589Z","shell.execute_reply.started":"2022-07-12T05:20:13.666440Z","shell.execute_reply":"2022-07-12T05:20:13.674707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output = pd.DataFrame({\"PassengerId\": id_pass, \"Transported\": test})\nprint(output.head())\noutput['Transported'] = output['Transported'].map(bool_map)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:20:13.676976Z","iopub.execute_input":"2022-07-12T05:20:13.677943Z","iopub.status.idle":"2022-07-12T05:20:13.693120Z","shell.execute_reply.started":"2022-07-12T05:20:13.677898Z","shell.execute_reply":"2022-07-12T05:20:13.692274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:21:18.205993Z","iopub.execute_input":"2022-07-12T05:21:18.206469Z","iopub.status.idle":"2022-07-12T05:21:18.220115Z","shell.execute_reply.started":"2022-07-12T05:21:18.206434Z","shell.execute_reply":"2022-07-12T05:21:18.219410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output.to_csv(\"./submission_1.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:21:19.274037Z","iopub.execute_input":"2022-07-12T05:21:19.274494Z","iopub.status.idle":"2022-07-12T05:21:19.293216Z","shell.execute_reply.started":"2022-07-12T05:21:19.274460Z","shell.execute_reply":"2022-07-12T05:21:19.292125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}