{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-02T11:21:30.487653Z","iopub.execute_input":"2022-07-02T11:21:30.488713Z","iopub.status.idle":"2022-07-02T11:21:30.502105Z","shell.execute_reply.started":"2022-07-02T11:21:30.488657Z","shell.execute_reply":"2022-07-02T11:21:30.500896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/spaceship-titanic/train.csv')\ntrain","metadata":{"execution":{"iopub.status.busy":"2022-07-02T11:21:30.609822Z","iopub.execute_input":"2022-07-02T11:21:30.611491Z","iopub.status.idle":"2022-07-02T11:21:30.692678Z","shell.execute_reply.started":"2022-07-02T11:21:30.611436Z","shell.execute_reply":"2022-07-02T11:21:30.691828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-02T11:21:30.703091Z","iopub.execute_input":"2022-07-02T11:21:30.703522Z","iopub.status.idle":"2022-07-02T11:21:30.728633Z","shell.execute_reply.started":"2022-07-02T11:21:30.703491Z","shell.execute_reply":"2022-07-02T11:21:30.726529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**We're going to work on these features to find significance relations:**\n*  HomePlanet \n*  CryoSleep \n*  Cabin \n*  Destination \n*  Age \n*  VIP \n*  RoomService,\n*  FoodCourt,\n*  ShoppingMall,\n*  Spa,\n*  VRDeck\n\nAfter we check each features type, we change dummy variables.","metadata":{}},{"cell_type":"code","source":"df_dc = pd.get_dummies(train, columns=['HomePlanet','CryoSleep','Destination','VIP'])\ndf_dc.drop(['Name','Cabin','PassengerId'], axis = 1, inplace = True)\ndf_dc","metadata":{"execution":{"iopub.status.busy":"2022-07-02T11:21:30.791392Z","iopub.execute_input":"2022-07-02T11:21:30.791765Z","iopub.status.idle":"2022-07-02T11:21:30.837584Z","shell.execute_reply.started":"2022-07-02T11:21:30.791734Z","shell.execute_reply":"2022-07-02T11:21:30.836174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"first_column = df_dc.pop('Transported')\n  \n# insert column using insert(position,column_name,\n# first_column) function\ndf_dc.insert(16, 'Transported', first_column)","metadata":{"execution":{"iopub.status.busy":"2022-07-02T11:21:30.874927Z","iopub.execute_input":"2022-07-02T11:21:30.875369Z","iopub.status.idle":"2022-07-02T11:21:30.883432Z","shell.execute_reply.started":"2022-07-02T11:21:30.875337Z","shell.execute_reply":"2022-07-02T11:21:30.881866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Import label encoder\nfrom sklearn import preprocessing\n  \n# label_encoder object knows how to understand word labels.\nlabel_encoder = preprocessing.LabelEncoder()\n  \n# Encode labels in column 'species'.\ndf_dc['Transported']= label_encoder.fit_transform(df_dc['Transported'])\n  \ndf_dc","metadata":{"execution":{"iopub.status.busy":"2022-07-02T11:21:30.960061Z","iopub.execute_input":"2022-07-02T11:21:30.961337Z","iopub.status.idle":"2022-07-02T11:21:30.995064Z","shell.execute_reply.started":"2022-07-02T11:21:30.961292Z","shell.execute_reply":"2022-07-02T11:21:30.994271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_dc.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-02T11:21:31.031164Z","iopub.execute_input":"2022-07-02T11:21:31.03172Z","iopub.status.idle":"2022-07-02T11:21:31.04504Z","shell.execute_reply.started":"2022-07-02T11:21:31.031689Z","shell.execute_reply":"2022-07-02T11:21:31.044117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_dc.fillna(df_dc[['Age','RoomService','FoodCourt','ShoppingMall','Spa','VRDeck']].median(),\n                inplace=True)                   ","metadata":{"execution":{"iopub.status.busy":"2022-07-02T11:21:31.132942Z","iopub.execute_input":"2022-07-02T11:21:31.133663Z","iopub.status.idle":"2022-07-02T11:21:31.143732Z","shell.execute_reply.started":"2022-07-02T11:21:31.133629Z","shell.execute_reply":"2022-07-02T11:21:31.142237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_dc.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-02T11:21:31.216201Z","iopub.execute_input":"2022-07-02T11:21:31.216638Z","iopub.status.idle":"2022-07-02T11:21:31.231434Z","shell.execute_reply.started":"2022-07-02T11:21:31.216602Z","shell.execute_reply":"2022-07-02T11:21:31.230056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# define X, y\nX=df_dc.iloc[:,:-1]\ny=df_dc.iloc[:,-1]\n","metadata":{"execution":{"iopub.status.busy":"2022-07-02T11:21:31.280236Z","iopub.execute_input":"2022-07-02T11:21:31.280704Z","iopub.status.idle":"2022-07-02T11:21:31.288053Z","shell.execute_reply.started":"2022-07-02T11:21:31.280671Z","shell.execute_reply":"2022-07-02T11:21:31.286824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Import the necessary modules\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import confusion_matrix, classification_report\nfrom sklearn.model_selection import train_test_split\n\n# Create training and test sets\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size = 0.4, random_state=42)\n\n# Create the classifier: logreg\nlogreg = LogisticRegression()\n\n# Fit the classifier to the training data\nlogreg.fit(X_train, y_train)\n\n# Predict the labels of the test set: y_pred\ny_pred = logreg.predict(X_test)\n\n# Compute and print the confusion matrix and classification report\nprint(confusion_matrix(y_test, y_pred))\nprint(classification_report(y_test, y_pred))","metadata":{"execution":{"iopub.status.busy":"2022-07-02T11:21:31.362946Z","iopub.execute_input":"2022-07-02T11:21:31.363926Z","iopub.status.idle":"2022-07-02T11:21:31.521078Z","shell.execute_reply.started":"2022-07-02T11:21:31.363888Z","shell.execute_reply":"2022-07-02T11:21:31.519836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv('/kaggle/input/spaceship-titanic/test.csv')\ntest","metadata":{"execution":{"iopub.status.busy":"2022-07-02T11:21:31.523284Z","iopub.execute_input":"2022-07-02T11:21:31.524259Z","iopub.status.idle":"2022-07-02T11:21:31.608959Z","shell.execute_reply.started":"2022-07-02T11:21:31.524206Z","shell.execute_reply":"2022-07-02T11:21:31.607567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.drop(['Name','Cabin','PassengerId'], axis = 1, inplace = True)\ntest","metadata":{"execution":{"iopub.status.busy":"2022-07-02T11:21:31.611958Z","iopub.execute_input":"2022-07-02T11:21:31.612899Z","iopub.status.idle":"2022-07-02T11:21:31.64554Z","shell.execute_reply.started":"2022-07-02T11:21:31.612844Z","shell.execute_reply":"2022-07-02T11:21:31.644557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.get_dummies(test, columns=['HomePlanet','CryoSleep','Destination','VIP'])","metadata":{"execution":{"iopub.status.busy":"2022-07-02T11:21:31.647139Z","iopub.execute_input":"2022-07-02T11:21:31.647626Z","iopub.status.idle":"2022-07-02T11:21:31.660958Z","shell.execute_reply.started":"2022-07-02T11:21:31.647594Z","shell.execute_reply":"2022-07-02T11:21:31.659888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.fillna(df_dc[['Age','RoomService','FoodCourt','ShoppingMall','Spa','VRDeck']].median(),\n                inplace=True)   \n\nX_test=test.iloc[:,:]\ny_pred = logreg.predict(X_test)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-02T11:21:31.687949Z","iopub.execute_input":"2022-07-02T11:21:31.688389Z","iopub.status.idle":"2022-07-02T11:21:31.705096Z","shell.execute_reply.started":"2022-07-02T11:21:31.688358Z","shell.execute_reply":"2022-07-02T11:21:31.702951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_new = pd.concat([test, pd.DataFrame(y_pred)], axis=1)\ntest_new","metadata":{"execution":{"iopub.status.busy":"2022-07-02T11:21:31.752718Z","iopub.execute_input":"2022-07-02T11:21:31.753413Z","iopub.status.idle":"2022-07-02T11:21:31.803541Z","shell.execute_reply.started":"2022-07-02T11:21:31.753359Z","shell.execute_reply":"2022-07-02T11:21:31.802207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_new.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-02T11:21:31.809576Z","iopub.execute_input":"2022-07-02T11:21:31.810494Z","iopub.status.idle":"2022-07-02T11:21:31.830848Z","shell.execute_reply.started":"2022-07-02T11:21:31.810437Z","shell.execute_reply":"2022-07-02T11:21:31.829885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_new.rename(columns={0: 'Transported'},\n          inplace=True, errors='raise')\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-02T11:21:31.832812Z","iopub.execute_input":"2022-07-02T11:21:31.833402Z","iopub.status.idle":"2022-07-02T11:21:31.839381Z","shell.execute_reply.started":"2022-07-02T11:21:31.833365Z","shell.execute_reply":"2022-07-02T11:21:31.838313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_new","metadata":{"execution":{"iopub.status.busy":"2022-07-02T11:21:31.841244Z","iopub.execute_input":"2022-07-02T11:21:31.842411Z","iopub.status.idle":"2022-07-02T11:21:31.876511Z","shell.execute_reply.started":"2022-07-02T11:21:31.84237Z","shell.execute_reply":"2022-07-02T11:21:31.875627Z"},"trusted":true},"execution_count":null,"outputs":[]}]}