{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Import packages, read CSVs and load into DFs","metadata":{"id":"D10PrQdg6i9C"}},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.metrics import accuracy_score\nfrom sklearn import svm\nfrom sklearn.ensemble import GradientBoostingClassifier\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.ensemble import VotingClassifier\nfrom xgboost import XGBClassifier\nfrom lightgbm import LGBMClassifier\nfrom sklearn.linear_model import RidgeClassifier, LogisticRegression\n%matplotlib inline","metadata":{"id":"x7pYF-AS5rD1","execution":{"iopub.status.busy":"2022-07-28T16:36:51.138610Z","iopub.execute_input":"2022-07-28T16:36:51.139078Z","iopub.status.idle":"2022-07-28T16:36:53.909414Z","shell.execute_reply.started":"2022-07-28T16:36:51.138990Z","shell.execute_reply":"2022-07-28T16:36:53.907999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Reading CSVs and creating Test and Train DFs\ndf_train = pd.read_csv(\"../input/spaceship-titanic/train.csv\")\ndf_test = pd.read_csv(\"../input/spaceship-titanic/test.csv\")","metadata":{"id":"F25-fXvW6CKk","execution":{"iopub.status.busy":"2022-07-28T16:36:53.911213Z","iopub.execute_input":"2022-07-28T16:36:53.911573Z","iopub.status.idle":"2022-07-28T16:36:54.020773Z","shell.execute_reply.started":"2022-07-28T16:36:53.911543Z","shell.execute_reply":"2022-07-28T16:36:54.017862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.info()","metadata":{"id":"7ac1rPTZCnt3","outputId":"994a8f43-8afc-4fdd-8c80-61002dbc71c4","execution":{"iopub.status.busy":"2022-07-28T16:36:54.022992Z","iopub.execute_input":"2022-07-28T16:36:54.023901Z","iopub.status.idle":"2022-07-28T16:36:54.068960Z","shell.execute_reply.started":"2022-07-28T16:36:54.023849Z","shell.execute_reply":"2022-07-28T16:36:54.067739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.info()","metadata":{"id":"rtDbxoon9sba","outputId":"b63ad4b9-21f7-4038-c52e-eee1a3a8969c","execution":{"iopub.status.busy":"2022-07-28T16:36:54.072752Z","iopub.execute_input":"2022-07-28T16:36:54.073752Z","iopub.status.idle":"2022-07-28T16:36:54.102831Z","shell.execute_reply.started":"2022-07-28T16:36:54.073699Z","shell.execute_reply":"2022-07-28T16:36:54.101440Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.sample()","metadata":{"id":"t-W149ZB7Erm","outputId":"ccd5862a-e78e-4314-f467-490b3e9b9ba1","execution":{"iopub.status.busy":"2022-07-28T16:36:54.104966Z","iopub.execute_input":"2022-07-28T16:36:54.105606Z","iopub.status.idle":"2022-07-28T16:36:54.136973Z","shell.execute_reply.started":"2022-07-28T16:36:54.105558Z","shell.execute_reply":"2022-07-28T16:36:54.135672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Cleaning:\n\n1. HomePlanet: One Hot Encoding.\n2. CryoSleep: Convert to 1 and 0. Replace NULL with Mode.\n3. Cabin: Replace NULL with [0/0/0]. Divide into 2 new columns - \"Deck\" and \"Side\"\n4. Destination: One Hot Encoding.\n5. Age: Replace NULL with Median.\n6. VIP: Convert to 1 and 0. Replace NULL with Mode.\n7. RoomService, FoodCourt, ShoppingMall, Spa, VRDeck: Replace NULL with 0.\n8. Name: Drop, not relevant\n9. Passenger: Only the 2nd half of the ID is important since it signifies whether they were travelling in groups or not, therefore create new column 'Group' that stores value of how many people were travelling together.\n\nConvert all columns into float dtypes.","metadata":{"id":"GhxBNkNI7ydf"}},{"cell_type":"code","source":"#Checking for missing values\ndf_train.isna().sum()","metadata":{"id":"3gv0c6OL48Mr","outputId":"5367bf86-65ab-4990-da79-7844c66448a2","execution":{"iopub.status.busy":"2022-07-28T16:36:54.138363Z","iopub.execute_input":"2022-07-28T16:36:54.139279Z","iopub.status.idle":"2022-07-28T16:36:54.158972Z","shell.execute_reply.started":"2022-07-28T16:36:54.139230Z","shell.execute_reply":"2022-07-28T16:36:54.157455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def clean_df(df):\n  #One Hot Encoding for HomePlanet\n  one_hot = pd.get_dummies(df['HomePlanet'])\n  df = df.drop('HomePlanet',axis = 1)\n  df = df.join(one_hot)\n  #One Hot Encoding for Destination\n  one_hot = pd.get_dummies(df['Destination'])\n  df = df.drop('Destination',axis = 1)\n  df = df.join(one_hot)\n  #Replace NULL in CryoSleep and VIP with mode of the column\n  df['CryoSleep'] = df['CryoSleep'].fillna(df.CryoSleep.mode()[0])\n  df['VIP'] = df['VIP'].fillna(df.CryoSleep.mode()[0])\n  #Drop Name since it's irrelevant \n  del df[\"Name\"]\n  #Replace NULL in Cabin with [0/0/0]\n  df.Cabin = df.Cabin.fillna('0/0/0')\n  #Create 2 new columns 'Deck' and 'Side' from Cabin and then dropping Cabin\n  df[\"Deck\"] = df[\"Cabin\"].apply(lambda x: x.split('/')[0])\n  df[\"Side\"] = df[\"Cabin\"].apply(lambda x: x.split('/')[2])\n  del df[\"Cabin\"]\n  #Convert CryoSleep from Object Dtype to Int\n  df[\"CryoSleep\"] = df[\"CryoSleep\"].astype(float)\n  #Replace NULL in RoomService, FoodCourt, ShoppingMall, Spa, VRDeck with 0\n  df['RoomService'] = df['RoomService'].fillna(0)\n  df['FoodCourt'] = df['FoodCourt'].fillna(0)\n  df['ShoppingMall'] = df['ShoppingMall'].fillna(0)\n  df['Spa'] = df['Spa'].fillna(0)\n  df['VRDeck'] = df['VRDeck'].fillna(0)\n  #Convert VIP from Object Dtype to Int\n  df[\"VIP\"] = df[\"VIP\"].astype(float)\n  #Creating new column 'Group' to count number of members\n  df[\"Group\"] = df[\"PassengerId\"].apply(lambda x: x.split('_')[1])\n  #Convert group from Object Dtype to Int\n  df[\"Group\"] = df[\"Group\"].astype(float)\n  #Drop PassengerId from train and test\n  del df[\"PassengerId\"]\n  #Replace NULL in 'Age' with median \n  df['Age'].fillna(df['Age'].median(), inplace=True)\n  return df","metadata":{"id":"HGSVZyWv7swN","execution":{"iopub.status.busy":"2022-07-28T16:36:54.160726Z","iopub.execute_input":"2022-07-28T16:36:54.161992Z","iopub.status.idle":"2022-07-28T16:36:54.178247Z","shell.execute_reply.started":"2022-07-28T16:36:54.161938Z","shell.execute_reply":"2022-07-28T16:36:54.176821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Label encoder for Deck and Side columns\nle=LabelEncoder()\ndef label(df):\n  df[\"Deck\"]=le.fit_transform(df[\"Deck\"])\n  df[\"Side\"]=le.fit_transform(df[\"Side\"])\n  return df","metadata":{"id":"Qx9g3MOUJ2sz","execution":{"iopub.status.busy":"2022-07-28T16:36:54.179715Z","iopub.execute_input":"2022-07-28T16:36:54.180366Z","iopub.status.idle":"2022-07-28T16:36:54.195131Z","shell.execute_reply.started":"2022-07-28T16:36:54.180330Z","shell.execute_reply":"2022-07-28T16:36:54.193700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = clean_df(df_train)\ndf_train = label(df_train)","metadata":{"id":"6JsH-uCk9E56","execution":{"iopub.status.busy":"2022-07-28T16:36:54.196837Z","iopub.execute_input":"2022-07-28T16:36:54.197239Z","iopub.status.idle":"2022-07-28T16:36:54.262330Z","shell.execute_reply.started":"2022-07-28T16:36:54.197206Z","shell.execute_reply":"2022-07-28T16:36:54.261337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Convert Transported from Object Dtype to Int\ndf_train[\"Transported\"] = df_train[\"Transported\"].astype(int)\ndf_train.sample()","metadata":{"id":"84XjyGy09CKD","outputId":"7bcd57bb-0a55-41e5-9835-e1259ee7e912","execution":{"iopub.status.busy":"2022-07-28T16:36:54.266005Z","iopub.execute_input":"2022-07-28T16:36:54.267074Z","iopub.status.idle":"2022-07-28T16:36:54.290613Z","shell.execute_reply.started":"2022-07-28T16:36:54.267036Z","shell.execute_reply":"2022-07-28T16:36:54.289719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.info()","metadata":{"id":"9aZFkq2DLft5","outputId":"b1bcacf6-2b41-4e6c-8bcc-cc316634eace","execution":{"iopub.status.busy":"2022-07-28T16:36:54.291855Z","iopub.execute_input":"2022-07-28T16:36:54.292401Z","iopub.status.idle":"2022-07-28T16:36:54.306148Z","shell.execute_reply.started":"2022-07-28T16:36:54.292370Z","shell.execute_reply":"2022-07-28T16:36:54.304978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Visualization:\n\n**Correlation Matrix / Heatmap**","metadata":{"id":"yAg0S4WuBqBU"}},{"cell_type":"code","source":"corr = df_train.corr()\ncorr.style.background_gradient(cmap='coolwarm')","metadata":{"id":"_JaAQbvYD1nU","outputId":"5d28c989-7794-4204-d150-2bc1d9857bc1","execution":{"iopub.status.busy":"2022-07-28T16:36:54.307689Z","iopub.execute_input":"2022-07-28T16:36:54.308432Z","iopub.status.idle":"2022-07-28T16:36:54.444950Z","shell.execute_reply.started":"2022-07-28T16:36:54.308389Z","shell.execute_reply":"2022-07-28T16:36:54.443849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Insights:**\n\n1. A lot of people from Earth had the same Deck.\n2. A lot of people who were in CryoSleep were Transported.\n3. A lot of people from Europa spent more money on different activities like Shopping, SPA, VR deck, Food Court and Room Service.","metadata":{"id":"MdOmipyjDnaI"}},{"cell_type":"markdown","source":"# Modeling:\n\n1. Data Preprocessing (Scaling - Standard Scaler)\n2. Train Test Split\n3. Train Data on Different Models\n4. Report of Model scores","metadata":{"id":"UaZwBIV0PqXs"}},{"cell_type":"code","source":"#Preprocessing of testing data\ndf_test = clean_df(df_test)\ndf_test = label(df_test)\ndf_test.sample()","metadata":{"id":"elATOS3pQ96E","outputId":"5e991c4f-c5f3-4d85-f3f0-017fce8bcd4b","execution":{"iopub.status.busy":"2022-07-28T16:36:54.446372Z","iopub.execute_input":"2022-07-28T16:36:54.447466Z","iopub.status.idle":"2022-07-28T16:36:54.510477Z","shell.execute_reply.started":"2022-07-28T16:36:54.447417Z","shell.execute_reply":"2022-07-28T16:36:54.509377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Split train df into test/train\nx=df_train[[\n 'CryoSleep', 'Age', 'VIP', 'RoomService', 'FoodCourt', 'ShoppingMall', 'Spa', 'VRDeck', 'Earth', 'Europa', 'Mars', '55 Cancri e', 'PSO J318.5-22', 'TRAPPIST-1e', 'Deck', 'Side', 'Group']]\ny=df_train['Transported']\nX_train, X_test, y_train, y_test= train_test_split(x, y, test_size=0.2, random_state=1)","metadata":{"id":"bn2MH2Aa2Y6S","execution":{"iopub.status.busy":"2022-07-28T16:36:54.512055Z","iopub.execute_input":"2022-07-28T16:36:54.513158Z","iopub.status.idle":"2022-07-28T16:36:54.526591Z","shell.execute_reply.started":"2022-07-28T16:36:54.513114Z","shell.execute_reply":"2022-07-28T16:36:54.525383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# #Scaling the data\n# X_train=StandardScaler().fit_transform(X_train)\n# X_test=StandardScaler().fit_transform(X_test)\n# df_test=StandardScaler().fit_transform(df_test)","metadata":{"id":"iF2xD45B96pR","execution":{"iopub.status.busy":"2022-07-28T16:36:54.528493Z","iopub.execute_input":"2022-07-28T16:36:54.529202Z","iopub.status.idle":"2022-07-28T16:36:54.536312Z","shell.execute_reply.started":"2022-07-28T16:36:54.529156Z","shell.execute_reply":"2022-07-28T16:36:54.535116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Interesting insight:** *When scaling was applied, the accuracy for each of the models reduced. Therefore, it is commented out.*","metadata":{"id":"NIHijg5H_onS"}},{"cell_type":"markdown","source":"**XG Boost**","metadata":{"id":"FTVp9fPhBVMq"}},{"cell_type":"code","source":"modelXG=XGBClassifier(verbose = 0)\nmodelXG.fit(X_train,y_train)","metadata":{"id":"E83QTf0JAseG","outputId":"7830fff5-e08c-4935-97c9-c821eff59c84","execution":{"iopub.status.busy":"2022-07-28T16:36:54.538100Z","iopub.execute_input":"2022-07-28T16:36:54.538842Z","iopub.status.idle":"2022-07-28T16:36:55.997720Z","shell.execute_reply.started":"2022-07-28T16:36:54.538796Z","shell.execute_reply":"2022-07-28T16:36:55.995357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(accuracy_score(modelXG.predict(X_test),y_test))","metadata":{"id":"GIB_A3tBCNKu","outputId":"c860c3a1-2bbb-470f-a3b8-7a9ebd5d1536","execution":{"iopub.status.busy":"2022-07-28T16:36:55.999769Z","iopub.execute_input":"2022-07-28T16:36:56.000159Z","iopub.status.idle":"2022-07-28T16:36:56.037897Z","shell.execute_reply.started":"2022-07-28T16:36:56.000125Z","shell.execute_reply":"2022-07-28T16:36:56.036096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Voting Classifier**","metadata":{"id":"EGQjP-apCbBQ"}},{"cell_type":"code","source":"models = {'XGboost':XGBClassifier(verbose = 0),\n           'gbc':GradientBoostingClassifier(),\n           'ridge':RidgeClassifier(),\n           'lr':LogisticRegression()}\n\nestimators = [('XGboost', XGBClassifier(verbose = 0)), ('gbc', GradientBoostingClassifier()),  ('lr', LogisticRegression())]\nmodelVC = VotingClassifier(estimators=estimators, voting='soft', weights=[1, 1, 1])\nmodelVC.fit(X_train,y_train)","metadata":{"id":"NxgEVea5CT36","outputId":"57ce8d0b-0a6a-49c6-94a7-be962ba24405","execution":{"iopub.status.busy":"2022-07-28T16:36:56.042692Z","iopub.execute_input":"2022-07-28T16:36:56.046220Z","iopub.status.idle":"2022-07-28T16:36:58.867549Z","shell.execute_reply.started":"2022-07-28T16:36:56.046168Z","shell.execute_reply":"2022-07-28T16:36:58.866448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(accuracy_score(modelVC.predict(X_test),y_test))","metadata":{"id":"_NMm3cFMCpV_","outputId":"07406920-206d-4134-d5d8-2b9e9d647a78","execution":{"iopub.status.busy":"2022-07-28T16:36:58.869179Z","iopub.execute_input":"2022-07-28T16:36:58.869872Z","iopub.status.idle":"2022-07-28T16:36:58.921348Z","shell.execute_reply.started":"2022-07-28T16:36:58.869828Z","shell.execute_reply":"2022-07-28T16:36:58.920465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**LGBM Classifier**","metadata":{"id":"HXsSC1n2MQ0-"}},{"cell_type":"code","source":"modelLGBM=LGBMClassifier(max_depth=6, random_state=314, silent=True, metric='None', n_jobs=6)\n\nmodelLGBM.fit(X_train,y_train)","metadata":{"id":"5p4WewSEMRVv","outputId":"182083ef-acb0-4892-824f-81d7b154cbb2","execution":{"iopub.status.busy":"2022-07-28T16:36:58.922723Z","iopub.execute_input":"2022-07-28T16:36:58.923308Z","iopub.status.idle":"2022-07-28T16:36:59.691700Z","shell.execute_reply.started":"2022-07-28T16:36:58.923274Z","shell.execute_reply":"2022-07-28T16:36:59.690674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(accuracy_score(modelLGBM.predict(X_test),y_test))","metadata":{"id":"84lrcFvVMSg1","outputId":"fc73f9f5-7a38-48a9-d300-30ab7b980e5c","execution":{"iopub.status.busy":"2022-07-28T16:36:59.693963Z","iopub.execute_input":"2022-07-28T16:36:59.694452Z","iopub.status.idle":"2022-07-28T16:36:59.713481Z","shell.execute_reply.started":"2022-07-28T16:36:59.694405Z","shell.execute_reply":"2022-07-28T16:36:59.712098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**SVM**","metadata":{"id":"gcsW1H0LMhqn"}},{"cell_type":"code","source":"modelSVM=svm.SVC(kernel='rbf')\nmodelSVM.fit(X_train, y_train)","metadata":{"id":"HesJpTvVMVB9","outputId":"dc196455-0ca7-4766-e721-647560ed1931","execution":{"iopub.status.busy":"2022-07-28T16:36:59.716567Z","iopub.execute_input":"2022-07-28T16:36:59.717103Z","iopub.status.idle":"2022-07-28T16:37:01.466474Z","shell.execute_reply.started":"2022-07-28T16:36:59.717055Z","shell.execute_reply":"2022-07-28T16:37:01.465341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(accuracy_score(modelSVM.predict(X_test),y_test))","metadata":{"id":"omfH5r_JMmIm","outputId":"1d2a35c2-d659-4e6d-eea6-79a87072972c","execution":{"iopub.status.busy":"2022-07-28T16:37:01.467707Z","iopub.execute_input":"2022-07-28T16:37:01.468137Z","iopub.status.idle":"2022-07-28T16:37:01.849503Z","shell.execute_reply.started":"2022-07-28T16:37:01.468100Z","shell.execute_reply":"2022-07-28T16:37:01.848348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"report = pd.DataFrame({\n    \"Model\" : [\"XG Boost\",\"Voting Classifier\", \"LGBM Classifier\", \"SVM\"],\n    \"Accuracy score\" : [accuracy_score(modelXG.predict(X_test),y_test),accuracy_score(modelVC.predict(X_test),y_test),accuracy_score(modelLGBM.predict(X_test),y_test), accuracy_score(modelSVM.predict(X_test),y_test)]\n})\nreport.sort_values(by = \"Accuracy score\")","metadata":{"id":"kk7TbJZeM4fR","outputId":"942439c9-7833-4c58-8bf8-27c0fe5a9f0c","execution":{"iopub.status.busy":"2022-07-28T16:37:01.851177Z","iopub.execute_input":"2022-07-28T16:37:01.852034Z","iopub.status.idle":"2022-07-28T16:37:02.358139Z","shell.execute_reply.started":"2022-07-28T16:37:01.851987Z","shell.execute_reply":"2022-07-28T16:37:02.356998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.sample()","metadata":{"id":"dH4XoxugI3gl","outputId":"fea11fae-6c54-49eb-eabb-9f033c331fc0","execution":{"iopub.status.busy":"2022-07-28T16:37:02.359602Z","iopub.execute_input":"2022-07-28T16:37:02.359968Z","iopub.status.idle":"2022-07-28T16:37:02.380043Z","shell.execute_reply.started":"2022-07-28T16:37:02.359936Z","shell.execute_reply":"2022-07-28T16:37:02.378782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.sample()","metadata":{"id":"pWKcrs1YI6Fd","outputId":"980704a0-15a3-4809-8f05-a206d6612ea0","execution":{"iopub.status.busy":"2022-07-28T16:37:02.381651Z","iopub.execute_input":"2022-07-28T16:37:02.382080Z","iopub.status.idle":"2022-07-28T16:37:02.403448Z","shell.execute_reply.started":"2022-07-28T16:37:02.382045Z","shell.execute_reply":"2022-07-28T16:37:02.401940Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predictions on Test Dataset","metadata":{"id":"S0qL-w0MMvuJ"}},{"cell_type":"code","source":"predframe=modelVC.predict(df_test)\n\npred=pd.DataFrame(predframe, columns=['Transported'])\n\npred['Transported']=pred['Transported'].replace(1, \"True\")\n\npred['Transported']=pred['Transported'].replace(0, \"False\")\n\npred","metadata":{"id":"Nq9DxixCMpJ0","outputId":"ffad62c0-23f3-400d-b55a-2499d7de4ec4","execution":{"iopub.status.busy":"2022-07-28T16:37:02.405699Z","iopub.execute_input":"2022-07-28T16:37:02.406194Z","iopub.status.idle":"2022-07-28T16:37:02.458387Z","shell.execute_reply.started":"2022-07-28T16:37:02.406156Z","shell.execute_reply":"2022-07-28T16:37:02.457070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Creating submission df and file\npred1=pd.read_csv(\"../input/spaceship-titanic/test.csv\")\n\npred2=pred1[\"PassengerId\"]\n\nfinalPred=pd.DataFrame(pred2, columns=['PassengerId'])\nsub=pd.concat([finalPred,pred], axis=1)\nsub.to_csv(\"submission.csv\", index=False)","metadata":{"id":"UaVU2dkd-0ZC","execution":{"iopub.status.busy":"2022-07-28T16:37:02.466588Z","iopub.execute_input":"2022-07-28T16:37:02.467306Z","iopub.status.idle":"2022-07-28T16:37:02.539269Z","shell.execute_reply.started":"2022-07-28T16:37:02.467245Z","shell.execute_reply":"2022-07-28T16:37:02.537704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub=pd.concat([finalPred,pred], axis=1)\nsub","metadata":{"id":"6sSgil6QFasA","outputId":"abb0d19c-a1b5-4f05-c7a8-eed47f3063cb","execution":{"iopub.status.busy":"2022-07-28T16:37:02.541648Z","iopub.execute_input":"2022-07-28T16:37:02.542565Z","iopub.status.idle":"2022-07-28T16:37:02.566109Z","shell.execute_reply.started":"2022-07-28T16:37:02.542496Z","shell.execute_reply":"2022-07-28T16:37:02.564960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.to_csv(\"submission.csv\", index=False)","metadata":{"id":"PDkx2j2HFcdL","execution":{"iopub.status.busy":"2022-07-28T16:37:02.567672Z","iopub.execute_input":"2022-07-28T16:37:02.568052Z","iopub.status.idle":"2022-07-28T16:37:02.585672Z","shell.execute_reply.started":"2022-07-28T16:37:02.568020Z","shell.execute_reply":"2022-07-28T16:37:02.584579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Final Thoughts:**\n\nThank you for viewing this notebook. One of the first ever submissions to a competition with a team. Fun dataset. \n\nSomething we did not explore but might be worth exploring:\n1. Whether the people transported were determined by alphabetic order ('Name' column)\n\nCould someone help answer these questions:\n1. Why did scaling the data affect the accuracy negatively?\n2. What are other better ways to visualize the data? (Which columns required more visualization / inspection?)\n3. What other parameters for the models or other models should we try to improve the accuracy?\n\nWe would appreciate any and every feedback. Thank you once again.","metadata":{"id":"ozRVGQjBFm7E"}}]}