{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport regex as re\nimport scipy as sc\n\nfrom sklearn.model_selection import train_test_split ,GridSearchCV\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import accuracy_score , confusion_matrix\n\n%matplotlib inline\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-08T04:01:12.724580Z","iopub.execute_input":"2022-08-08T04:01:12.725202Z","iopub.status.idle":"2022-08-08T04:01:12.737431Z","shell.execute_reply.started":"2022-08-08T04:01:12.725168Z","shell.execute_reply":"2022-08-08T04:01:12.736390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/spaceship-titanic/train.csv')\ntest = pd.read_csv('/kaggle/input/spaceship-titanic/test.csv')\n\n### Visualizing the ages of the survived vs Transported people ###\nsurvived_age = train[train['Transported'] == False]['Age']\ntransported_age = train[train['Transported'] == True]['Age']\n\nfig = plt.figure(figsize = (16,8))\nplt.hist(survived_age , bins = np.arange(0,85,5) ,density=True, label='Survived' , alpha= 0.5 , color='#3c9db0')\nplt.hist(transported_age , bins = np.arange(0,85,5) ,density=True, label='Transported' , alpha= 0.35 , color='#D7503C')\nplt.legend(frameon = False , fontsize= 'large')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T04:01:16.016735Z","iopub.execute_input":"2022-08-08T04:01:16.017465Z","iopub.status.idle":"2022-08-08T04:01:16.417041Z","shell.execute_reply.started":"2022-08-08T04:01:16.017426Z","shell.execute_reply":"2022-08-08T04:01:16.415854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_copy = train.copy()\ntrain_copy[['deck' , 'num' , 'side']] = train_copy['Cabin'].str.split('/' , expand = True)\ntrain_copy['HomePlanet']=train_copy['HomePlanet'].astype('category').cat.codes\ntrain_copy['Destination'] =train_copy['Destination'].astype('category').cat.codes\ntrain_copy['deck'] =train_copy['deck'].astype('category').cat.codes\ntrain_copy['side'] =train_copy['side'].astype('category').cat.codes\ntrain_copy.drop(['PassengerId' , 'Name' , 'Cabin'] , axis=1 , inplace=True)\n\nsns.set(rc = {'figure.figsize':(16,8)})\nsns.heatmap(train_copy.corr(), annot = True, fmt='.2g',cmap= 'coolwarm')","metadata":{"execution":{"iopub.status.busy":"2022-08-08T04:01:41.429264Z","iopub.execute_input":"2022-08-08T04:01:41.429724Z","iopub.status.idle":"2022-08-08T04:01:42.307159Z","shell.execute_reply.started":"2022-08-08T04:01:41.429685Z","shell.execute_reply":"2022-08-08T04:01:42.306014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### Preprocessing Training data ###\n# Splitting the cabin into different its 3 components #\ntrain[['deck' , 'num' , 'side']] = train['Cabin'].str.split('/' , expand = True)\n\n# Keeping the important features only #\ncolumns = ['CryoSleep' , 'Age' , 'HomePlanet' , 'Spa' , 'VRDeck' , 'RoomService' , 'Destination', 'deck' , 'num' , 'side' , 'Transported']\ntrain_cleaned = train[columns].dropna()\n\n# converting labeled featuires into numerical ones #\n\nhome_encoder = LabelEncoder().fit(train_cleaned.HomePlanet.unique())\ntrain_cleaned.HomePlanet = home_encoder.transform(train_cleaned.HomePlanet)\n\ndestination_encoder = LabelEncoder().fit(train_cleaned.Destination.unique())\ntrain_cleaned.Destination = destination_encoder.transform(train_cleaned.Destination)\n\ndeck_encoder = LabelEncoder().fit(train_cleaned.deck)\ntrain_cleaned.deck = deck_encoder.transform(train_cleaned.deck)\n\nside_encoder = LabelEncoder().fit(train_cleaned.side)\ntrain_cleaned.side = side_encoder.transform(train_cleaned.side)\n\ntrain_cleaned.sample(10)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T04:01:55.719170Z","iopub.execute_input":"2022-08-08T04:01:55.720037Z","iopub.status.idle":"2022-08-08T04:01:55.782903Z","shell.execute_reply.started":"2022-08-08T04:01:55.719991Z","shell.execute_reply":"2022-08-08T04:01:55.781786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features = train_cleaned.iloc[:,:-1]\ntarget = train_cleaned.Transported\n\nX_train , X_test , y_train , y_test = train_test_split(features, target , random_state=34)\nparams = {'n_estimators':[20 , 30 , 35 , 40 , 50 , 60],\n          'max_depth':[7 , 8 , 9 , 10]}\nmodel = GridSearchCV(RandomForestClassifier(random_state = 34) , param_grid= params , scoring='accuracy').fit(X_train , y_train)\n\nprint('Best accuracy is {0:.5f}'.format(model.best_score_))\nprint('Best Parameters are {}'.format(model.best_params_))","metadata":{"execution":{"iopub.status.busy":"2022-08-08T04:02:50.343790Z","iopub.execute_input":"2022-08-08T04:02:50.344843Z","iopub.status.idle":"2022-08-08T04:03:15.024503Z","shell.execute_reply.started":"2022-08-08T04:02:50.344802Z","shell.execute_reply":"2022-08-08T04:03:15.023173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rfc = model.best_estimator_\ny_predicted = rfc.predict(X_test)\n\nprint('Accuracy on our test samples = {0:.5f}'.format(accuracy_score(y_test , y_predicted)))\nprint('Confusion Matrix:\\n{}'.format(confusion_matrix(y_test , y_predicted)))","metadata":{"execution":{"iopub.status.busy":"2022-08-08T04:03:19.249392Z","iopub.execute_input":"2022-08-08T04:03:19.249834Z","iopub.status.idle":"2022-08-08T04:03:19.277268Z","shell.execute_reply.started":"2022-08-08T04:03:19.249796Z","shell.execute_reply":"2022-08-08T04:03:19.276385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### Preprocessing Test Data ###\n\ntest[['deck' , 'num' , 'side']] = test['Cabin'].str.split('/' , expand = True)\n\ncolumns = ['PassengerId' , 'CryoSleep' , 'Age' , 'HomePlanet' , 'Spa' , 'VRDeck' , 'RoomService' , 'Destination', 'deck' , 'num' , 'side' ]\ntest_cleaned = test[columns].fillna(method = 'bfill')\n\ntest_cleaned.HomePlanet = home_encoder.transform(test_cleaned.HomePlanet)\ntest_cleaned.Destination = destination_encoder.transform(test_cleaned.Destination)\ntest_cleaned.deck = deck_encoder.transform(test_cleaned.deck)\ntest_cleaned.side = side_encoder.transform(test_cleaned.side)\n\n### Predicting the output ###\n\nTransported_predictions = rfc.predict(test_cleaned.drop(['PassengerId'] , axis=1))\noutput = pd.DataFrame({\n    'PassengerId': test_cleaned.PassengerId,\n    'Transported': Transported_predictions\n})\noutput.to_csv('output.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T04:04:34.070965Z","iopub.execute_input":"2022-08-08T04:04:34.072054Z","iopub.status.idle":"2022-08-08T04:04:34.138554Z","shell.execute_reply.started":"2022-08-08T04:04:34.072009Z","shell.execute_reply":"2022-08-08T04:04:34.137618Z"},"trusted":true},"execution_count":null,"outputs":[]}]}