{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt # plotting\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-05T20:05:34.153054Z","iopub.execute_input":"2022-08-05T20:05:34.153496Z","iopub.status.idle":"2022-08-05T20:05:34.164573Z","shell.execute_reply.started":"2022-08-05T20:05:34.153464Z","shell.execute_reply":"2022-08-05T20:05:34.162969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check out the data\n\ndata = pd.read_csv(\"/kaggle/input/spaceship-titanic/train.csv\")\ndata.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-05T20:05:34.187320Z","iopub.execute_input":"2022-08-05T20:05:34.188272Z","iopub.status.idle":"2022-08-05T20:05:34.240569Z","shell.execute_reply.started":"2022-08-05T20:05:34.188214Z","shell.execute_reply":"2022-08-05T20:05:34.239709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Explore passenger database home planets\n\n# Extracts home planet values\nHomePlanet = set(data[\"HomePlanet\"])\n#print(HomePlanet, len(HomePlanet))\n\n# Counts how many recurring values per unique key\nHomePlanet_stats = {i:np.count_nonzero(data[\"HomePlanet\"]==i) for i in data[\"HomePlanet\"]}\n\n# plot distributions in the dataset\nHomePlanet_stats_keys = HomePlanet_stats.keys()\nHomePlanet_stats_values = HomePlanet_stats.values()\nplt.pie(HomePlanet_stats_values, labels = HomePlanet_stats_keys, autopct='%1.1f%%')\nplt.show()","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2022-08-05T20:05:34.241997Z","iopub.execute_input":"2022-08-05T20:05:34.242954Z","iopub.status.idle":"2022-08-05T20:05:40.443989Z","shell.execute_reply.started":"2022-08-05T20:05:34.242920Z","shell.execute_reply":"2022-08-05T20:05:40.442048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Explore passenger database destination:\n\n# Extract destination values\nDestination = set(data['Destination']) # set function creates a set of values, not ordered in any way.\nprint(Destination)\n\n# Count recurring values\nDestination_stats = {i:np.count_nonzero(data['Destination']==i) for i in data['Destination']}\nprint(Destination_stats)\n\n# so apparently the counting above only works for dataframes\n# let me try something without a set\n\nDestination = data ['Destination']\nDestination_stats = {i:np.count_nonzero(Destination==i) for i in Destination}\nprint(Destination_stats)\n# awesome, i got the same thing\n\n# extract the x and y values for plotting\nDestination_stats_keys = Destination_stats.keys()\nDestination_stats_values = Destination_stats.values()\nplt.pie(Destination_stats_values, labels = Destination_stats.keys(), autopct = '%1.1f%%')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-05T20:05:40.446578Z","iopub.execute_input":"2022-08-05T20:05:40.447442Z","iopub.status.idle":"2022-08-05T20:05:52.607496Z","shell.execute_reply.started":"2022-08-05T20:05:40.447381Z","shell.execute_reply":"2022-08-05T20:05:52.605887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Now let's check how many were transported\n\nTransported = data[\"Transported\"]\nprint(Transported.describe())\n\nTransported_stats = {i:np.count_nonzero(Transported==i) for i in Transported}\nprint(Transported_stats)\n\nTransported_stats_keys = Transported_stats.keys()\nTransported_stats_values = Transported_stats.values()\nplt.pie(Transported_stats_values, labels = Transported_stats_keys, autopct = '%1.1f%%')\nplt.show()\n\n# ouch, almost half died. ","metadata":{"execution":{"iopub.status.busy":"2022-08-05T20:05:52.611702Z","iopub.execute_input":"2022-08-05T20:05:52.613099Z","iopub.status.idle":"2022-08-05T20:05:53.746294Z","shell.execute_reply.started":"2022-08-05T20:05:52.612939Z","shell.execute_reply":"2022-08-05T20:05:53.744837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Now let's figure out to which destination did ppl die\nSurvival_Destination = np.array([data[\"Destination\"],data[\"Transported\"]])\nprint(Survival_Destination)\n\n# Okay this does what I want, but lets see in the end what will be useful to implement\ntest= (zip(data[\"Destination\"],data[\"Transported\"]))\n#print(tuple(test))\n#print(test)\n\n#Survival_Destination_stats={Survival_Destination[0]:np.count_nonzero(Survival_Destination[]==i) for i in Survival_Destination}\n#Survival_Destination[0,1]\n#test[1].count(\"True\")\n\n# Okay this is very useful. Now how to plot?\nSurvival_Destination_table = data.groupby([\"Destination\", \"Transported\"]).size()\nprint(Survival_Destination_table)\ntype(Survival_Destination_table)\nprint(Survival_Destination_table[0:2]) # I don't quite understand this\n\n# okay so i can apply the same trick\nprint(Survival_Destination_table.index) # alright so it's a 2x1 index\nprint(Survival_Destination_table.values)\n\nprint(len(Survival_Destination_table.index))\n\n# Other ways of presenting the same thing above\n#data.value_counts([\"Destination\",\"Transported\"])\n#data.pivot_table(index=[\"Destination\", \"Transported\"],aggfunc='size')\n\n#data.groupby(['Destination',\"Transported\"]).size().plot(kind='pie',subplots=True)\n# oh this is a different way of doing","metadata":{"execution":{"iopub.status.busy":"2022-08-05T20:05:53.748301Z","iopub.execute_input":"2022-08-05T20:05:53.750104Z","iopub.status.idle":"2022-08-05T20:05:53.775502Z","shell.execute_reply.started":"2022-08-05T20:05:53.750036Z","shell.execute_reply":"2022-08-05T20:05:53.774107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Suggestion from stackoverflow.\nprint(Survival_Destination_table.reset_index())\nSurvival_Destination_table_new = Survival_Destination_table.reset_index()\n\n","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2022-08-05T20:05:53.778069Z","iopub.execute_input":"2022-08-05T20:05:53.779354Z","iopub.status.idle":"2022-08-05T20:05:53.797778Z","shell.execute_reply.started":"2022-08-05T20:05:53.779305Z","shell.execute_reply":"2022-08-05T20:05:53.796401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#TESTS\nprint(type(Survival_Destination_table_new)) # okay it's a dataframe object\nprint('\\n')\nprint(Survival_Destination_table_new.iloc[1]) # okay so this is how you call dataframe elements\nprint('\\n')\nprint(Survival_Destination_table_new.iloc[0:2]) # okay so it presents the related columns\nprint('\\n')\nprint(Survival_Destination_table_new.iloc[2]) # I am now a bit confused it lists the table I think\n# Okay I am now more confused on how this iloc presents data\nprint('\\n')\nprint(Survival_Destination_table_new.at[2,\"Destination\"]) #this approach only works for single elements. it doesn't for ranges.\n\n## spacer\nprint('\\n')\nprint(Survival_Destination_table_new[0]) # Yes! this works to get the column for the value only\n# let's try now to extract a specific value from the column.\nprint(Survival_Destination_table_new[0][1]) # It works!\n# now let's try an entire range of values\nprint(Survival_Destination_table_new[0][0:5]) #great! I think I'm ready.\n# what if i do this\nprint(Survival_Destination_table_new[0:2]) # rows\nprint(Survival_Destination_table_new[0:2][0]) # for column '0' print the two rows","metadata":{"execution":{"iopub.status.busy":"2022-08-05T20:05:53.803680Z","iopub.execute_input":"2022-08-05T20:05:53.804385Z","iopub.status.idle":"2022-08-05T20:05:53.838985Z","shell.execute_reply.started":"2022-08-05T20:05:53.804325Z","shell.execute_reply":"2022-08-05T20:05:53.837776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# convert dataframe to numpy\nSurvival_Destination_table_new = Survival_Destination_table_new.to_numpy()","metadata":{"execution":{"iopub.status.busy":"2022-08-05T20:05:53.840646Z","iopub.execute_input":"2022-08-05T20:05:53.840976Z","iopub.status.idle":"2022-08-05T20:05:53.846803Z","shell.execute_reply.started":"2022-08-05T20:05:53.840945Z","shell.execute_reply":"2022-08-05T20:05:53.846082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# okay now to plot it!\n\nfig, axes = (plt.subplots(nrows=1,ncols=3, sharex=True,sharey=True))\naxes[0].pie(Survival_Destination_table_new.T[2][0:2], labels = Survival_Destination_table_new.T[1][0:2],autopct ='%1.1f%%', frame=True)\naxes[0].set_title(Survival_Destination_table_new.T[0][0])\n# YAAAAAASSS\naxes[1].pie(Survival_Destination_table_new.T[2][2:4], labels = Survival_Destination_table_new.T[1][2:4],autopct ='%1.1f%%', frame=True)\naxes[1].set_title(Survival_Destination_table_new.T[0][2])\naxes[2].pie(Survival_Destination_table_new.T[2][4:6], labels = Survival_Destination_table_new.T[1][4:6],autopct ='%1.1f%%', frame=True)\naxes[2].set_title(Survival_Destination_table_new.T[0][4])\n\n\n#plt.title(#Survival_Destination_table_new.T[0])\n#fig, axes = plt.subplots(1,3)\n#axes[0].pie(Survival_Destination_table.values[0:2], labels = Survival_Destination_table.index[0:2], autopct = '%1.1f%%', radius=1)\n#axes[1].pie(Survival_Destination_table.values[2:4], labels = Survival_Destination_table.index[2:4], autopct = '%1.1f%%', radius=1)\n#axes[2].pie(Survival_Destination_table.values[4:6], labels = Survival_Destination_table.index[4:6], autopct = \"%1.1f%%\", radius=1)\nplt.plot()\n\n# okay that kind-of did something and the plot is now less messy.\n\n#########\n# but this doesn't really tell me anything. why did around half of the passengers didn't make it?\n# okay reading the long, the first destination is 55 Cancri e, and the ship was hit before they reached the first destination\n# so the fatalities are those who weren't able to escape \n# now what?\n# okay, so far we've looked at the destination and transported values. what about origin and transported values?","metadata":{"execution":{"iopub.status.busy":"2022-08-05T20:05:53.850043Z","iopub.execute_input":"2022-08-05T20:05:53.850884Z","iopub.status.idle":"2022-08-05T20:05:54.245702Z","shell.execute_reply.started":"2022-08-05T20:05:53.850852Z","shell.execute_reply":"2022-08-05T20:05:54.244591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Survival_Origin_table = data.groupby([\"HomePlanet\", \"Transported\"]).size()\nprint(Survival_Origin_table)\n\nSurvival_Origin_table = Survival_Origin_table.reset_index()\nprint(\"\\n\", Survival_Origin_table)\n\nSurvival_Origin_table = Survival_Origin_table.to_numpy()\nprint(\"\\n\", Survival_Origin_table)\n\nprint(\"\\n\", Survival_Origin_table.T)\n\n# Alright, the code now works as expected.\n\nfig, axes = plt.subplots(nrows=1, ncols=3, sharex=True, sharey=True)\n#fig, axes = plt.figure(frameon=True)\naxes[0].pie(Survival_Origin_table.T[2][0:2], autopct=\"%1.1f%%\", labels=Survival_Origin_table.T[1][0:2])\naxes[1].pie(Survival_Origin_table.T[2][2:4], autopct=\"%1.1f%%\", labels=Survival_Origin_table.T[1][2:4])\naxes[2].pie(Survival_Origin_table.T[2][4:6], autopct=\"%1.1f%%\", labels=Survival_Origin_table.T[1][4:6])\naxes[0].set_title (Survival_Origin_table.T[0][1])\naxes[1].set_title (Survival_Origin_table.T[0][3])\naxes[2].set_title (Survival_Origin_table.T[0][5])\nplt.plot()\n\n# okay this code is much more solid.","metadata":{"execution":{"iopub.status.busy":"2022-08-05T20:05:54.247183Z","iopub.execute_input":"2022-08-05T20:05:54.247512Z","iopub.status.idle":"2022-08-05T20:05:54.469257Z","shell.execute_reply.started":"2022-08-05T20:05:54.247482Z","shell.execute_reply":"2022-08-05T20:05:54.467824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check data again on what to look for next\n# too far to scroll up\ndata.head()\n\n# Correlation between the following variables and survival:\n# CryoSleep\n# Age\n# VIP\n\n# Can I do a multiplot pie chart? or a spider chart","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2022-08-05T20:05:54.471547Z","iopub.execute_input":"2022-08-05T20:05:54.472562Z","iopub.status.idle":"2022-08-05T20:05:54.514341Z","shell.execute_reply.started":"2022-08-05T20:05:54.472499Z","shell.execute_reply":"2022-08-05T20:05:54.512655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n","metadata":{}},{"cell_type":"code","source":"Multi_table = data.groupby(['CryoSleep', 'Age', 'VIP','Transported']).size()\nprint(Multi_table)\n\n# okay this isn't going to work becase the age is too broad\nMulti_table = data.groupby(['VIP','CryoSleep','Transported']).size()\nprint(\"\\n\", Multi_table)\n\n# not all VIPs were in cyrosleep\nMulti_table = data.groupby([\"VIP\", \"Transported\"]).size()\nprint(\"\\n\", Multi_table)\n# Okay so there's quite a low number of VIP and even then all of them didn't survive\n\nMulti_table = data.groupby([\"CryoSleep\", \"Transported\"]).size()\nprint(\"\\n\",Multi_table)\n\n# Well, this is interesting, those who were in cryosleep had a better survival ratio\n\n# now let me base on that \nMulti_table = data.groupby([\"CryoSleep\", \"VIP\", \"Transported\"]).size()\nprint(\"\\n\", Multi_table)\n# okay it seems that being VIP doesn't really guarantee survival\n# huh, playing around with the different data, the counts don't match\n# anyway, maybe next is to develop the model.","metadata":{"execution":{"iopub.status.busy":"2022-08-05T20:05:54.516730Z","iopub.execute_input":"2022-08-05T20:05:54.517434Z","iopub.status.idle":"2022-08-05T20:05:54.552088Z","shell.execute_reply.started":"2022-08-05T20:05:54.517400Z","shell.execute_reply":"2022-08-05T20:05:54.550991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# just for fun, let's check the age distribbution of the passengers\n\nAgeDist = data[\"Age\"]\nprint(AgeDist)\n\nplt.hist(AgeDist, bins=40, histtype='step', linewidth=5, density=True)\nplt.ylabel(\"Count / Percentage\")\nplt.xlabel(\"Age\")\nplt.show()","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2022-08-05T20:05:54.553658Z","iopub.execute_input":"2022-08-05T20:05:54.553973Z","iopub.status.idle":"2022-08-05T20:05:54.764319Z","shell.execute_reply.started":"2022-08-05T20:05:54.553942Z","shell.execute_reply":"2022-08-05T20:05:54.763235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# okay now let's do some model development ... which I have no idea about.\n\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import mean_absolute_error\nfrom sklearn.model_selection import train_test_split\nfrom sklearn import metrics","metadata":{"execution":{"iopub.status.busy":"2022-08-05T20:37:53.591111Z","iopub.execute_input":"2022-08-05T20:37:53.591527Z","iopub.status.idle":"2022-08-05T20:37:53.597767Z","shell.execute_reply.started":"2022-08-05T20:37:53.591494Z","shell.execute_reply":"2022-08-05T20:37:53.596431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# declare features that will be used for tranining the model\n\ndata.head() # quickly check the data\n#data_clean = data.dropna(axis=0)\ndata_clean = data.fillna(0)\nfeatures = [\"CryoSleep\"]\ndata_features = data_clean[features]\ndata_features = data_features\nprint(data_features.head())\nprint(\"\\n\", type(data_features))\n\n# declare target\ntarget = data_clean[\"Transported\"]\nprint(\"\\n\", target.head())\n\n# okay this is quickly turning into a problem because ... \n# what I need is the count ... but maybe not? hmm","metadata":{"execution":{"iopub.status.busy":"2022-08-05T20:39:55.092151Z","iopub.execute_input":"2022-08-05T20:39:55.092547Z","iopub.status.idle":"2022-08-05T20:39:55.116595Z","shell.execute_reply.started":"2022-08-05T20:39:55.092516Z","shell.execute_reply":"2022-08-05T20:39:55.115144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train model\ntrain_features,val_features,train_target,val_target = train_test_split(data_features, target,random_state=0)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T20:39:57.349708Z","iopub.execute_input":"2022-08-05T20:39:57.350262Z","iopub.status.idle":"2022-08-05T20:39:57.358933Z","shell.execute_reply.started":"2022-08-05T20:39:57.350226Z","shell.execute_reply":"2022-08-05T20:39:57.357505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"forest_model = RandomForestClassifier(random_state = 1)\nforest_model.fit(train_features,train_target)\nprediction = forest_model.predict(val_features)\n#print(mean_absolute_error(val_target, prediction)) #For RandomForestClassifier\nprint(prediction)\nprint(\"Accuracy:\", metrics.accuracy_score(val_target, prediction))","metadata":{"execution":{"iopub.status.busy":"2022-08-05T20:39:59.199172Z","iopub.execute_input":"2022-08-05T20:39:59.199616Z","iopub.status.idle":"2022-08-05T20:39:59.464209Z","shell.execute_reply.started":"2022-08-05T20:39:59.199581Z","shell.execute_reply":"2022-08-05T20:39:59.462888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Apply trained model to test data\n\nTestData = pd.read_csv(\"/kaggle/input/spaceship-titanic/test.csv\")\nprint(len(TestData)) # yep this is the 4277 rows\nTestData_clean = TestData.dropna(axis =0)\nprint(len(TestData_clean)) # okay this is a problem. \n#the cleaned data only has 3281 rows. I cannot use this then.\n\nprint(TestData[\"CryoSleep\"].head())\nprint(\"\\n\", TestData.head())\n\ndata_features_x = [\"CryoSleep\"]\ndata_features = TestData[data_features_x]\nprint(data_features.head())\nprint(type(data_features))\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T20:40:05.272786Z","iopub.execute_input":"2022-08-05T20:40:05.273251Z","iopub.status.idle":"2022-08-05T20:40:05.321134Z","shell.execute_reply.started":"2022-08-05T20:40:05.273217Z","shell.execute_reply":"2022-08-05T20:40:05.319435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Okay there is something wrong with how I extract data.\n# I figured it out. somehow I need to do the following to keep the data type as a DataFrame\n# data_features_x = [\"CryoSleep\"]\n# data_features = TestData[data_features_x]\n# if i do directly it will become a pandas.Series.\n\n# but now I have to use the cleaned data! maybe i can do fill NA\n\n# TestData = TestData[\"CryoSleep\"].to_numpy\n# print(type(TestData))\n#print(type(train_features))\n\nTestData_clean = TestData.fillna(False)\nTestDataFeatures = TestData_clean[data_features_x]\n\nRandomForest_prediction = forest_model.predict(TestDataFeatures)\nprint(\"\\n\", len(RandomForest_prediction))\nprint(\"\\n\", RandomForest_prediction)\n\n# damn why isn't the output boolean??\n# now it is","metadata":{"execution":{"iopub.status.busy":"2022-08-05T20:40:07.929367Z","iopub.execute_input":"2022-08-05T20:40:07.930569Z","iopub.status.idle":"2022-08-05T20:40:07.995612Z","shell.execute_reply.started":"2022-08-05T20:40:07.930523Z","shell.execute_reply":"2022-08-05T20:40:07.993323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compare before and after \nprint(TestData_clean[\"CryoSleep\"], data_clean[\"Transported\"])","metadata":{"execution":{"iopub.status.busy":"2022-08-05T20:40:10.405277Z","iopub.execute_input":"2022-08-05T20:40:10.405732Z","iopub.status.idle":"2022-08-05T20:40:10.415303Z","shell.execute_reply.started":"2022-08-05T20:40:10.405699Z","shell.execute_reply":"2022-08-05T20:40:10.414150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# The code below is now all unnecessary because\n# I figured out that I had to use a classifier instead of a regressor\n\n# print(type(RandomForest_prediction))\n# print(np.count_nonzero(RandomForest_prediction < 0.5))\n# print(np.count_nonzero(RandomForest_prediction < 0.8))\n# print(np.count_nonzero(RandomForest_prediction > 0.8))\n# # okay so it's either 0.8 or 0.3\n# print(np.array(RandomForest_prediction, dtype=bool))\n# b = RandomForest_prediction > 0.8\n# print(b)\n# # OMG","metadata":{"execution":{"iopub.status.busy":"2022-08-05T20:14:40.353391Z","iopub.execute_input":"2022-08-05T20:14:40.353896Z","iopub.status.idle":"2022-08-05T20:14:40.363236Z","shell.execute_reply.started":"2022-08-05T20:14:40.353857Z","shell.execute_reply":"2022-08-05T20:14:40.361572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#output = pd.DataFrame({\"PassengerId\":TestData_clean.PassengerId, \"Transported\":RandomForest_prediction})\n#output.to_csv(\"submission.csv\", index=False)\n\n## YAAASS it finally works\n#commenting out for now as i develop new outputs later in the code","metadata":{"execution":{"iopub.status.busy":"2022-08-05T20:23:17.470754Z","iopub.execute_input":"2022-08-05T20:23:17.471233Z","iopub.status.idle":"2022-08-05T20:23:17.486701Z","shell.execute_reply.started":"2022-08-05T20:23:17.471196Z","shell.execute_reply":"2022-08-05T20:23:17.485283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Okay now to add more complexity to the trianing model\n# reload everything just to be clear of any edits\n\ndata = pd.read_csv(\"/kaggle/input/spaceship-titanic/train.csv\")\n#data_clean = data.fillna(0)\ndata_clean = data.dropna(axis=0)\nfeatures = [\"CryoSleep\", \"VIP\", \"FoodCourt\",\"ShoppingMall\",\"Spa\",\"VRDeck\",\"RoomService\",\"Age\"]\ndata_features = data_clean[features]\n\n# declare target\ntarget = data_clean[\"Transported\"]\n\n# Train Test Split\ntrain_features, value_features, train_target, value_target = train_test_split(data_features, target, random_state=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T20:59:08.467805Z","iopub.execute_input":"2022-08-05T20:59:08.468328Z","iopub.status.idle":"2022-08-05T20:59:08.516057Z","shell.execute_reply.started":"2022-08-05T20:59:08.468290Z","shell.execute_reply":"2022-08-05T20:59:08.514887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_clean.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-05T20:56:35.870786Z","iopub.execute_input":"2022-08-05T20:56:35.871312Z","iopub.status.idle":"2022-08-05T20:56:35.894109Z","shell.execute_reply.started":"2022-08-05T20:56:35.871271Z","shell.execute_reply":"2022-08-05T20:56:35.892886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"forest_model = RandomForestClassifier(random_state=1)\nforest_model.fit(train_features, train_target)\nprediction = forest_model.predict(value_features)\nprint(\"Accuracy:\", metrics.accuracy_score(value_target, prediction))","metadata":{"execution":{"iopub.status.busy":"2022-08-05T20:59:18.659386Z","iopub.execute_input":"2022-08-05T20:59:18.659847Z","iopub.status.idle":"2022-08-05T20:59:19.309317Z","shell.execute_reply.started":"2022-08-05T20:59:18.659811Z","shell.execute_reply":"2022-08-05T20:59:19.307765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Apply on test data\nTestData = pd.read_csv(\"/kaggle/input/spaceship-titanic/test.csv\")\nTestData_clean = TestData.fillna(False)\n#features = [\"CryoSleep\", \"VIP\"] # no need to declare this because it was already done for training\nTestDataFeatures = TestData_clean[features]\nRandomForest_prediction = forest_model.predict(TestDataFeatures)\n\nprint(RandomForest_prediction)\nprint(type(RandomForest_prediction))\n\n#Output\noutput = pd.DataFrame({\"PassengerId\":TestData_clean.PassengerId, \"Transported\":RandomForest_prediction})\noutput.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T20:59:21.754559Z","iopub.execute_input":"2022-08-05T20:59:21.755052Z","iopub.status.idle":"2022-08-05T20:59:21.903780Z","shell.execute_reply.started":"2022-08-05T20:59:21.754990Z","shell.execute_reply":"2022-08-05T20:59:21.902342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}