{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt \n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-15T08:13:05.188757Z","iopub.execute_input":"2022-07-15T08:13:05.189162Z","iopub.status.idle":"2022-07-15T08:13:05.197522Z","shell.execute_reply.started":"2022-07-15T08:13:05.189129Z","shell.execute_reply":"2022-07-15T08:13:05.196350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"space=pd.read_csv('../input/spaceship-titanic/train.csv')\nspace_test=pd.read_csv('../input/spaceship-titanic/test.csv')\nspace.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:13:05.596358Z","iopub.execute_input":"2022-07-15T08:13:05.597223Z","iopub.status.idle":"2022-07-15T08:13:05.666591Z","shell.execute_reply.started":"2022-07-15T08:13:05.597182Z","shell.execute_reply":"2022-07-15T08:13:05.665244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Before any model building ,we need to conduct some eda and data cleaning. From the above we can see that there are some nba values for some of the columns. Let's focus on the numerical set of dsta first. Before that jsut form looking at the list of columns, we wont be needing the name column. So let's drop that before continuing.   ","metadata":{}},{"cell_type":"code","source":"space.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:13:06.013464Z","iopub.execute_input":"2022-07-15T08:13:06.013866Z","iopub.status.idle":"2022-07-15T08:13:06.031543Z","shell.execute_reply.started":"2022-07-15T08:13:06.013832Z","shell.execute_reply":"2022-07-15T08:13:06.030396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"space=space.drop(['Name'],axis=1)\nspace_test=space_test.drop(['Name'],axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:13:06.421931Z","iopub.execute_input":"2022-07-15T08:13:06.422909Z","iopub.status.idle":"2022-07-15T08:13:06.431170Z","shell.execute_reply.started":"2022-07-15T08:13:06.422870Z","shell.execute_reply":"2022-07-15T08:13:06.430289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"space_num=space.select_dtypes(include=np.number)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:13:06.823669Z","iopub.execute_input":"2022-07-15T08:13:06.824462Z","iopub.status.idle":"2022-07-15T08:13:06.830543Z","shell.execute_reply.started":"2022-07-15T08:13:06.824423Z","shell.execute_reply":"2022-07-15T08:13:06.829751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"space_num.columns","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:13:07.315960Z","iopub.execute_input":"2022-07-15T08:13:07.316532Z","iopub.status.idle":"2022-07-15T08:13:07.324366Z","shell.execute_reply.started":"2022-07-15T08:13:07.316488Z","shell.execute_reply":"2022-07-15T08:13:07.322833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nimport warnings\n\nfor x in space_num.columns:\n    plt.hist(space_num[x],bins=100)\n    plt.show()\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:13:07.641023Z","iopub.execute_input":"2022-07-15T08:13:07.642017Z","iopub.status.idle":"2022-07-15T08:13:09.670859Z","shell.execute_reply.started":"2022-07-15T08:13:07.641977Z","shell.execute_reply":"2022-07-15T08:13:09.669956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Form the above, its probbaly clear that the Age column is the only one that is MCAR. Since it does not follow a normal dist either, we shall use the miedian to do our imputation. We will use thta same median value for the ets set too since the assumption is that the train set follows the population. ","metadata":{}},{"cell_type":"code","source":"space['Age']=space['Age'].fillna(space['Age'].median())\nspace_test['Age']=space_test['Age'].fillna(space['Age'].median())","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:13:09.672705Z","iopub.execute_input":"2022-07-15T08:13:09.672995Z","iopub.status.idle":"2022-07-15T08:13:09.681079Z","shell.execute_reply.started":"2022-07-15T08:13:09.672968Z","shell.execute_reply":"2022-07-15T08:13:09.679973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"as for the other few, it might not be MCAR since those columns were about the spending on the ship. let;s quickly gleam into one of the columns to investigate","metadata":{}},{"cell_type":"code","source":"temp=space[space['CryoSleep']==True]\ntemp[temp['RoomService']>0]","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:13:09.682587Z","iopub.execute_input":"2022-07-15T08:13:09.683410Z","iopub.status.idle":"2022-07-15T08:13:09.700370Z","shell.execute_reply.started":"2022-07-15T08:13:09.683378Z","shell.execute_reply":"2022-07-15T08:13:09.699335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"What we see here is that for those put in cryosleep, there will not be any cost for room serivce, and for that matter there shouldn't be any for the other costs as well. So let;s partially impute using this logic first.  ","metadata":{}},{"cell_type":"code","source":"space['RoomService']=space.apply(lambda row: 0 if row['CryoSleep']==True else row['RoomService'],axis=1)\nspace['FoodCourt']=space.apply(lambda row: 0 if row['CryoSleep']==True else row['FoodCourt'],axis=1)\nspace['ShoppingMall']=space.apply(lambda row: 0 if row['CryoSleep']==True else row['ShoppingMall'],axis=1)\nspace['Spa']=space.apply(lambda row: 0 if row['CryoSleep']==True else row['Spa'],axis=1)\nspace['VRDeck']=space.apply(lambda row: 0 if row['CryoSleep']==True else row['VRDeck'],axis=1)\n\nspace.isna().sum()\n#df['c'] = df.apply(\n#    lambda row: row['a']*row['b'] if np.isnan(row['c']) else row['c'],\n#    axis=1\n#)\n#space['RoomService'].fillna(0 if space['CryoSleep']==True else Nan)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:13:09.703549Z","iopub.execute_input":"2022-07-15T08:13:09.704285Z","iopub.status.idle":"2022-07-15T08:13:10.463187Z","shell.execute_reply.started":"2022-07-15T08:13:09.704240Z","shell.execute_reply":"2022-07-15T08:13:10.461930Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"space.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:13:10.464849Z","iopub.execute_input":"2022-07-15T08:13:10.465513Z","iopub.status.idle":"2022-07-15T08:13:10.502062Z","shell.execute_reply.started":"2022-07-15T08:13:10.465469Z","shell.execute_reply":"2022-07-15T08:13:10.500990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As the median for the rest are also 0, let;s jsut impute all the na values for the various costs with 0 then. ","metadata":{}},{"cell_type":"code","source":"space['RoomService']=space['RoomService'].fillna(0)\nspace['FoodCourt']=space['FoodCourt'].fillna(0)\nspace['ShoppingMall']=space['ShoppingMall'].fillna(0)\nspace['Spa']=space['Spa'].fillna(0)\nspace['VRDeck']=space['VRDeck'].fillna(0)\n\n\nspace_test['RoomService']=space_test['RoomService'].fillna(0)\nspace_test['FoodCourt']=space_test['FoodCourt'].fillna(0)\nspace_test['ShoppingMall']=space_test['ShoppingMall'].fillna(0)\nspace_test['Spa']=space_test['Spa'].fillna(0)\nspace_test['VRDeck']=space_test['VRDeck'].fillna(0)\n\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:13:10.504046Z","iopub.execute_input":"2022-07-15T08:13:10.505091Z","iopub.status.idle":"2022-07-15T08:13:10.516293Z","shell.execute_reply.started":"2022-07-15T08:13:10.505052Z","shell.execute_reply":"2022-07-15T08:13:10.515205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"space.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:13:10.518081Z","iopub.execute_input":"2022-07-15T08:13:10.519414Z","iopub.status.idle":"2022-07-15T08:13:10.534970Z","shell.execute_reply.started":"2022-07-15T08:13:10.519380Z","shell.execute_reply":"2022-07-15T08:13:10.534145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now that the missing values for our numerical have been sorted, Lets go the cateogrical variables.  ","metadata":{}},{"cell_type":"code","source":"space['HomePlanet'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:13:10.576297Z","iopub.execute_input":"2022-07-15T08:13:10.576868Z","iopub.status.idle":"2022-07-15T08:13:10.585207Z","shell.execute_reply.started":"2022-07-15T08:13:10.576837Z","shell.execute_reply":"2022-07-15T08:13:10.584219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There's tw things we can try here. One is straightforward and is to try and impute with the mode, or we can try to look at the passengerID to determine their home planet. The assumoption im amaking here is that people in the same group come from the same home country. ","metadata":{}},{"cell_type":"code","source":"space_temp=space[space['HomePlanet'].isna()==False]\nspace_temp.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:13:11.000345Z","iopub.execute_input":"2022-07-15T08:13:11.001516Z","iopub.status.idle":"2022-07-15T08:13:11.032468Z","shell.execute_reply.started":"2022-07-15T08:13:11.001459Z","shell.execute_reply":"2022-07-15T08:13:11.031341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"space['id_1']=space['PassengerId'].str[:4]\nspace['id_2']=space['PassengerId'].str[5:7]\nspace_temp=space[space['HomePlanet'].isna()==False]\nspace_temp.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:13:11.422013Z","iopub.execute_input":"2022-07-15T08:13:11.422432Z","iopub.status.idle":"2022-07-15T08:13:11.463553Z","shell.execute_reply.started":"2022-07-15T08:13:11.422397Z","shell.execute_reply":"2022-07-15T08:13:11.462588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#createa dictionary with the id_1 and homeplanet related info \nspace_dict=space[space['HomePlanet'].isna()==False]\nkey=space_dict['id_1']\nvalue=space_dict['HomePlanet']\nfinal_dict=dict(zip(key,value))\n\nfreq_dict=dict(space['id_1'].value_counts())\n","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:13:11.828107Z","iopub.execute_input":"2022-07-15T08:13:11.828788Z","iopub.status.idle":"2022-07-15T08:13:11.879560Z","shell.execute_reply.started":"2022-07-15T08:13:11.828746Z","shell.execute_reply":"2022-07-15T08:13:11.878522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def method(id):\n    if freq_dict[id]>1:\n        if id in final_dict:\n            return final_dict[id]\n    \n    \nspace['HomePlanet']=space.apply(lambda row: method(row['id_1']) if pd.isnull(row['HomePlanet'])==True else row['HomePlanet'] ,axis=1)\nspace['HomePlanet'].value_counts()\n    ","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:13:12.237183Z","iopub.execute_input":"2022-07-15T08:13:12.238014Z","iopub.status.idle":"2022-07-15T08:13:12.420730Z","shell.execute_reply.started":"2022-07-15T08:13:12.237970Z","shell.execute_reply":"2022-07-15T08:13:12.419664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"space.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:13:12.637066Z","iopub.execute_input":"2022-07-15T08:13:12.637465Z","iopub.status.idle":"2022-07-15T08:13:12.659087Z","shell.execute_reply.started":"2022-07-15T08:13:12.637431Z","shell.execute_reply":"2022-07-15T08:13:12.657955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"For the rest, we shall impute it with the mode, which is Earth. To repeat steps for the test set too. `","metadata":{}},{"cell_type":"code","source":"space['HomePlanet']=space['HomePlanet'].fillna('Earth')\nspace.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:13:13.042339Z","iopub.execute_input":"2022-07-15T08:13:13.043010Z","iopub.status.idle":"2022-07-15T08:13:13.065592Z","shell.execute_reply.started":"2022-07-15T08:13:13.042964Z","shell.execute_reply":"2022-07-15T08:13:13.064452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#repeat for test set \nspace_test['id_1']=space_test['PassengerId'].str[:4]\nspace_test['id_2']=space_test['PassengerId'].str[5:7]\nspace_temp=space_test[space_test['HomePlanet'].isna()==False]\n\n\n#createa dictionary with the id_1 and homeplanet related info \nspace_dict=space_test[space_test['HomePlanet'].isna()==False]\nkey=space_dict['id_1']\nvalue=space_dict['HomePlanet']\nfinal_dict=dict(zip(key,value))\n\nfreq_dict=dict(space_test['id_1'].value_counts())\n\n\nspace_test['HomePlanet']=space_test.apply(lambda row: method(row['id_1']) if pd.isnull(row['HomePlanet'])==True else row['HomePlanet'] ,axis=1)\nspace_test['HomePlanet']=space_test['HomePlanet'].fillna('Earth')\nspace_test.info()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:13:13.444621Z","iopub.execute_input":"2022-07-15T08:13:13.445289Z","iopub.status.idle":"2022-07-15T08:13:13.584050Z","shell.execute_reply.started":"2022-07-15T08:13:13.445236Z","shell.execute_reply":"2022-07-15T08:13:13.582945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's do some one hot encoding for the Homeplanet befoer we carry on. ","metadata":{}},{"cell_type":"code","source":"space=pd.get_dummies(space,prefix=['HomePlanet'],columns=['HomePlanet'],drop_first=True)\n#space=space.drop(['HomePlanet'],axis=1)\nspace_test=pd.get_dummies(space_test,prefix=['HomePlanet'],columns=['HomePlanet'],drop_first=True)\n#space_test=space_test.drop(['HomePlanet'],axis=1)\nspace.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:13:14.079011Z","iopub.execute_input":"2022-07-15T08:13:14.079428Z","iopub.status.idle":"2022-07-15T08:13:14.115564Z","shell.execute_reply.started":"2022-07-15T08:13:14.079394Z","shell.execute_reply":"2022-07-15T08:13:14.114488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Ok moving on to CryoSleep. Wem ight assume the opposite of what we did earlier. Where if there are - values for the different spendings then it is comfrimed to be cryopsleep. let;s prove if this is indeed correct first. ","metadata":{}},{"cell_type":"code","source":"space_trial=space[space['CryoSleep']==False]\nspace_trial[space_trial['RoomService']==0]","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:13:14.694981Z","iopub.execute_input":"2022-07-15T08:13:14.695407Z","iopub.status.idle":"2022-07-15T08:13:14.735652Z","shell.execute_reply.started":"2022-07-15T08:13:14.695371Z","shell.execute_reply":"2022-07-15T08:13:14.734629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Jsut by looking at the first row we can see that this is not th case all the time, so we cqant impute ased on the spending amounts. as the number of NA values are quite lwo for this, we shall do Mode imputation. Doing the same for the test set. Then we will encode it to 1 and 0 instead of True and False. ","metadata":{}},{"cell_type":"code","source":"space['CryoSleep']=space['CryoSleep'].fillna(False)\nspace_test['CryoSleep']=space_test['CryoSleep'].fillna(False)\nspace['CryoSleep']=space['CryoSleep'].replace([True,False],[1,0])\nspace_test['CryoSleep']=space_test['CryoSleep'].replace([True,False],[1,0])","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:13:15.095206Z","iopub.execute_input":"2022-07-15T08:13:15.095992Z","iopub.status.idle":"2022-07-15T08:13:15.114793Z","shell.execute_reply.started":"2022-07-15T08:13:15.095951Z","shell.execute_reply":"2022-07-15T08:13:15.113652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"space['Cabin'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:13:15.500564Z","iopub.execute_input":"2022-07-15T08:13:15.500932Z","iopub.status.idle":"2022-07-15T08:13:15.515479Z","shell.execute_reply.started":"2022-07-15T08:13:15.500902Z","shell.execute_reply":"2022-07-15T08:13:15.514139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"FOr cabin, the missing values are MCAR, but there's probaly mroe that we can take out from the cabin info. The number in the middle might not be as improtant, but the deck and side of the ship could be helpful. So let;s pre process this info further. For the Na values, what we can do is to create a new cateogry for bth the deck as well as th side of the ship to sifngify that it is missing info. At the same time, to better repprssent this missingness, we shall create one more column to signfiy if the cabin info was missing or not.   ","metadata":{}},{"cell_type":"code","source":"space['cabin_deck']=space['Cabin'].str[:1]\nspace['cabin_side']=space['Cabin'].str[-1]\nspace['cabin_deck']=space['cabin_deck'].fillna('U')\nspace['cabin_side']=space['cabin_side'].fillna('U')\nspace['cabin_missing']=np.where(space['cabin_side']=='U',1,0)\n\n\n\n#for the testing set too \nspace_test['cabin_deck']=space_test['Cabin'].str[:1]\nspace_test['cabin_side']=space_test['Cabin'].str[-1]\nspace_test['cabin_deck']=space_test['cabin_deck'].fillna('U')\nspace_test['cabin_side']=space_test['cabin_side'].fillna('U')\nspace_test['cabin_missing']=np.where(space_test['cabin_side']=='U',1,0)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:13:15.874034Z","iopub.execute_input":"2022-07-15T08:13:15.874434Z","iopub.status.idle":"2022-07-15T08:13:15.911063Z","shell.execute_reply.started":"2022-07-15T08:13:15.874399Z","shell.execute_reply":"2022-07-15T08:13:15.910276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"NOw we do a quick frequency encoding for the train and test set so that we can covnert them into numerical. We will do it for both the deck as well as the side of the ship. ","metadata":{}},{"cell_type":"code","source":"import category_encoders as ce\n\nfeatures=['cabin_deck','cabin_side']\n#count encoder \ncount_encoder = ce.CountEncoder(cols=features)\ncount_encoder.fit(space[features])\nspace = space.join(count_encoder.transform(space[features]).add_suffix('_count'))\nspace_test=space_test.join(count_encoder.transform(space_test[features]).add_suffix('_count'))\n\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:13:16.224198Z","iopub.execute_input":"2022-07-15T08:13:16.224924Z","iopub.status.idle":"2022-07-15T08:13:16.296160Z","shell.execute_reply.started":"2022-07-15T08:13:16.224870Z","shell.execute_reply":"2022-07-15T08:13:16.295329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"space.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:13:16.553409Z","iopub.execute_input":"2022-07-15T08:13:16.554077Z","iopub.status.idle":"2022-07-15T08:13:16.579502Z","shell.execute_reply.started":"2022-07-15T08:13:16.554028Z","shell.execute_reply":"2022-07-15T08:13:16.578276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now we focus on the Destination. again similar ot before the asusmption im making here is that people in the same group are travellign to the same destination as well. so let;s do a partial imputation again, follwoed by an impute with the mode. To repeatr for test set too.   ","metadata":{}},{"cell_type":"code","source":"\n#createa dictionary with the id_1 and destination related info \nspace_dict=space[space['Destination'].isna()==False]\nkey=space_dict['id_1']\nvalue=space_dict['Destination']\nfinal_dict=dict(zip(key,value))\n\nfreq_dict=dict(space['id_1'].value_counts())\n\n\nspace['Destination']=space.apply(lambda row: method(row['id_1']) if pd.isnull(row['Destination'])==True else row['Destination'] ,axis=1)\nspace['Destination']=space['Destination'].fillna('TRAPPIST-1e')\nspace.info()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:13:17.026858Z","iopub.execute_input":"2022-07-15T08:13:17.027525Z","iopub.status.idle":"2022-07-15T08:13:17.272908Z","shell.execute_reply.started":"2022-07-15T08:13:17.027473Z","shell.execute_reply":"2022-07-15T08:13:17.271794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#createa dictionary with the id_1 and destination related info \nspace_dict=space_test[space_test['Destination'].isna()==False]\nkey=space_dict['id_1']\nvalue=space_dict['Destination']\nfinal_dict=dict(zip(key,value))\n\nfreq_dict=dict(space_test['id_1'].value_counts())\n\n\nspace_test['Destination']=space_test.apply(lambda row: method(row['id_1']) if pd.isnull(row['Destination'])==True else row['Destination'] ,axis=1)\nspace_test['Destination']=space_test['Destination'].fillna('TRAPPIST-1e')\nspace_test.info()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:13:17.475282Z","iopub.execute_input":"2022-07-15T08:13:17.475703Z","iopub.status.idle":"2022-07-15T08:13:17.606923Z","shell.execute_reply.started":"2022-07-15T08:13:17.475666Z","shell.execute_reply":"2022-07-15T08:13:17.606012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let;s do some one hot encoding for the destination as well. ","metadata":{}},{"cell_type":"code","source":"space=pd.get_dummies(space,prefix=['Destination'],columns=['Destination'],drop_first=True)\n#space=space.drop(['HomePlanet'],axis=1)\nspace_test=pd.get_dummies(space_test,prefix=['Destination'],columns=['Destination'],drop_first=True)\n#space_test=space_test.drop(['HomePlanet'],axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:13:17.760081Z","iopub.execute_input":"2022-07-15T08:13:17.760504Z","iopub.status.idle":"2022-07-15T08:13:17.783542Z","shell.execute_reply.started":"2022-07-15T08:13:17.760467Z","shell.execute_reply":"2022-07-15T08:13:17.782477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"FInally, we have the VIP colulmn. This is most lilkely MCAR, so what might be good is to create a new value to signify the missing values. Then we will one hot encode it, repeated for the  ","metadata":{}},{"cell_type":"code","source":"space['VIP']=space['VIP'].fillna('U')\nspace=pd.get_dummies(space,prefix=['VIP'],columns=['VIP'],drop_first=True)\n\nspace_test['VIP']=space_test['VIP'].fillna('U')\nspace_test=pd.get_dummies(space_test,prefix=['VIP'],columns=['VIP'],drop_first=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:13:18.091518Z","iopub.execute_input":"2022-07-15T08:13:18.091912Z","iopub.status.idle":"2022-07-15T08:13:18.116940Z","shell.execute_reply.started":"2022-07-15T08:13:18.091877Z","shell.execute_reply":"2022-07-15T08:13:18.115979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"space.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:13:18.547424Z","iopub.execute_input":"2022-07-15T08:13:18.547798Z","iopub.status.idle":"2022-07-15T08:13:18.570934Z","shell.execute_reply.started":"2022-07-15T08:13:18.547767Z","shell.execute_reply":"2022-07-15T08:13:18.569750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Great! Let's d some further cleanup here by removign the coluimns we wont need anymore. ","metadata":{}},{"cell_type":"code","source":"space=space.drop(['PassengerId','Cabin'],axis=1)\nspace_test=space_test.drop(['Cabin'],axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:13:18.851570Z","iopub.execute_input":"2022-07-15T08:13:18.851924Z","iopub.status.idle":"2022-07-15T08:13:18.862958Z","shell.execute_reply.started":"2022-07-15T08:13:18.851893Z","shell.execute_reply":"2022-07-15T08:13:18.861457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let;s do some feature enginerring. In particualr we could create one column to signify wether a passenger is travelling solo or not. i.e only one in the group or more than one in the group. This could open up more possibilites wether people whoi travel in groups might have a higher chance of getting trasnported or not. We can use the frequency dicrtionary we created previously to do this.  ","metadata":{}},{"cell_type":"code","source":"freq_dict=dict(space['id_1'].value_counts())\n\nspace['solo']=space.apply(lambda row: 1 if freq_dict[row['id_1']]>1 else 0 ,axis=1)\n\nfreq_dict=dict(space_test['id_1'].value_counts())\n\nspace_test['solo']=space_test.apply(lambda row: 1 if freq_dict[row['id_1']]>1 else 0 ,axis=1)\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:13:19.132575Z","iopub.execute_input":"2022-07-15T08:13:19.133371Z","iopub.status.idle":"2022-07-15T08:13:19.422929Z","shell.execute_reply.started":"2022-07-15T08:13:19.133322Z","shell.execute_reply":"2022-07-15T08:13:19.422088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Ok let;s finally clean up our dataset even more. ","metadata":{}},{"cell_type":"code","source":"space=space.drop(['id_1','id_2','cabin_deck','cabin_side'],axis=1)\nspace_test=space_test.drop(['id_1','id_2','cabin_deck','cabin_side'],axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:13:19.424458Z","iopub.execute_input":"2022-07-15T08:13:19.424968Z","iopub.status.idle":"2022-07-15T08:13:19.432850Z","shell.execute_reply.started":"2022-07-15T08:13:19.424935Z","shell.execute_reply":"2022-07-15T08:13:19.431874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"space.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:13:19.598474Z","iopub.execute_input":"2022-07-15T08:13:19.598891Z","iopub.status.idle":"2022-07-15T08:13:19.622488Z","shell.execute_reply.started":"2022-07-15T08:13:19.598854Z","shell.execute_reply":"2022-07-15T08:13:19.621142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"space.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:13:19.767673Z","iopub.execute_input":"2022-07-15T08:13:19.768792Z","iopub.status.idle":"2022-07-15T08:13:19.783660Z","shell.execute_reply.started":"2022-07-15T08:13:19.768740Z","shell.execute_reply":"2022-07-15T08:13:19.782289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"space_test.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:13:20.073803Z","iopub.execute_input":"2022-07-15T08:13:20.074386Z","iopub.status.idle":"2022-07-15T08:13:20.089062Z","shell.execute_reply.started":"2022-07-15T08:13:20.074353Z","shell.execute_reply":"2022-07-15T08:13:20.088106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Logistic Rgeression with G-descent. \n\ni want to try and implement logistic rgression from scratch to better understand how the algorithm works so here we go. \n\nwe know that logistc regression uses a sigmooid fucntion as follows: \n\n$$ g(z)=\\frac{1}{1+exp^{-z}} $$\n\nhere z refers to our weights multiplied by our x matrix. \n\nthe cost fucntion that we are minimising here is w.r.t to the log likelihood fucntion whihc is as follows: \n\n$$J(\\beta)=-\\frac{1}{m}\\sum^{m}_{j}(y^{(j)}log(g(\\beta^tx^{(j)}))+(1-y^{(j)})log(1-g(\\beta^tx^{(j)})))     $$\n\n\nOur differntial of this is: \n\n$$\\frac{d}{d\\beta}=\\frac{1}{m}X^T(g(\\beta^TX)-y)   $$\n\nAnd finally the updating step usign g-descent is: \n\n$$ \\beta=\\beta-\\alpha\\frac{d}{d\\beta}$$\n\n\nOk so let;'s code our these steps. ","metadata":{}},{"cell_type":"code","source":"\nspace['Transported']=space['Transported'].replace([True,False],[1,0])\nxtrain=space.drop(['Transported'],axis=1)\nytrain=space['Transported']\n\nxtest=space_test.drop(['PassengerId'],axis=1)\n\n\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:13:20.378839Z","iopub.execute_input":"2022-07-15T08:13:20.379467Z","iopub.status.idle":"2022-07-15T08:13:20.392957Z","shell.execute_reply.started":"2022-07-15T08:13:20.379429Z","shell.execute_reply":"2022-07-15T08:13:20.392067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#define the sigmoid function \n\ndef sigmoid(z):\n    return 1/(1+np.exp(-z))\n\n\ndef prediction(beta,x):\n    return sigmoid(np.dot(x,beta))\n\n\n\n#define the cost function\ndef cost(beta,x,y):\n    return (-y * np.log(prediction(beta,x)) - (1 - y) * np.log(1 - prediction(beta,x))).mean()\n\n\ndef grad(beta,x,y):\n    return (np.dot(np.transpose(x),(prediction(beta,x)-y)))/y.shape[0]\n\n\n\ndef log_r(x,y,learning,iterations):\n    bias = np.ones((x.shape[0], 1))\n    x = np.concatenate((bias, x), axis=1)\n  \n  # Initialize the weights\n    weights = np.zeros(x.shape[1])\n    loss=[]  \n  # Training with gradient descent\n    for step in range(iterations):\n        weights=weights-(learning*grad(weights,x,y))\n    \n      \n      # Print log-likelihood every step\n        cost_1 =cost(weights,x,y)\n        loss.append(cost_1)\n\n    \n    return weights,loss\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:13:20.554675Z","iopub.execute_input":"2022-07-15T08:13:20.555476Z","iopub.status.idle":"2022-07-15T08:13:20.565925Z","shell.execute_reply.started":"2022-07-15T08:13:20.555440Z","shell.execute_reply":"2022-07-15T08:13:20.564624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xtrain=xtrain.to_numpy()\nytrain=ytrain.to_numpy()\n\n\n\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:13:20.668010Z","iopub.execute_input":"2022-07-15T08:13:20.668425Z","iopub.status.idle":"2022-07-15T08:13:20.673378Z","shell.execute_reply.started":"2022-07-15T08:13:20.668387Z","shell.execute_reply":"2022-07-15T08:13:20.672470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(xtrain.shape)\nprint(ytrain.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:13:20.826303Z","iopub.execute_input":"2022-07-15T08:13:20.826875Z","iopub.status.idle":"2022-07-15T08:13:20.831980Z","shell.execute_reply.started":"2022-07-15T08:13:20.826841Z","shell.execute_reply":"2022-07-15T08:13:20.830790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"beta,loss=log_r(xtrain,ytrain,0.000000005,300000)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:13:20.975401Z","iopub.execute_input":"2022-07-15T08:13:20.976101Z","iopub.status.idle":"2022-07-15T08:20:47.795296Z","shell.execute_reply.started":"2022-07-15T08:13:20.976064Z","shell.execute_reply":"2022-07-15T08:20:47.793854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(loss)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:20:47.797637Z","iopub.execute_input":"2022-07-15T08:20:47.798516Z","iopub.status.idle":"2022-07-15T08:20:48.062460Z","shell.execute_reply.started":"2022-07-15T08:20:47.798467Z","shell.execute_reply":"2022-07-15T08:20:48.061300Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xtest=xtest.to_numpy()\nbias = np.ones((xtest.shape[0], 1))\nx = np.concatenate((bias, xtest), axis=1)\n\nypred=prediction(beta,x)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:22:51.819013Z","iopub.execute_input":"2022-07-15T08:22:51.819468Z","iopub.status.idle":"2022-07-15T08:22:51.829994Z","shell.execute_reply.started":"2022-07-15T08:22:51.819431Z","shell.execute_reply":"2022-07-15T08:22:51.828621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ids=space_test['PassengerId']\nsubmission=pd.DataFrame(ids)\ny_pred = pd.Series(ypred, name='Transported')\nsubmission['Transported']=y_pred\nsubmission['Transported']=submission['Transported']>0.5\nsubmission.head()\nsubmission.to_csv('submission.csv',index= False)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:28:23.146203Z","iopub.execute_input":"2022-07-15T08:28:23.146641Z","iopub.status.idle":"2022-07-15T08:28:23.167437Z","shell.execute_reply.started":"2022-07-15T08:28:23.146606Z","shell.execute_reply":"2022-07-15T08:28:23.166282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## closing thoughts \n\nNot too bad for a first attempt i would say. Might have been able to perform better if we changed some of the daat cleaning methedologies. Let me know what you think about this notebook. THanks! ","metadata":{}}]}