{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":" # This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\n# data analysis and wrangling\nimport pandas as pd\nimport numpy as np\nimport random as rnd\n\n# visualization\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n%matplotlib inline\nfrom plotly import figure_factory as figfac\nimport missingno as msno\n\n# machine learning\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.svm import SVC, LinearSVC\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.linear_model import Perceptron\nfrom sklearn.linear_model import SGDClassifier\nfrom sklearn.tree import DecisionTreeClassifier\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-15T02:09:58.433919Z","iopub.execute_input":"2022-08-15T02:09:58.434382Z","iopub.status.idle":"2022-08-15T02:09:58.453028Z","shell.execute_reply.started":"2022-08-15T02:09:58.434345Z","shell.execute_reply":"2022-08-15T02:09:58.451422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(\"../input/spaceship-titanic/train.csv\")\ntest = pd.read_csv(\"../input/spaceship-titanic/test.csv\")\nSubmission = test['PassengerId'].copy()\n#train\ntest","metadata":{"execution":{"iopub.status.busy":"2022-08-15T02:09:58.472348Z","iopub.execute_input":"2022-08-15T02:09:58.472749Z","iopub.status.idle":"2022-08-15T02:09:58.540747Z","shell.execute_reply.started":"2022-08-15T02:09:58.472714Z","shell.execute_reply":"2022-08-15T02:09:58.539565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2022-08-15T02:09:58.545736Z","iopub.execute_input":"2022-08-15T02:09:58.546496Z","iopub.status.idle":"2022-08-15T02:09:58.564678Z","shell.execute_reply.started":"2022-08-15T02:09:58.546455Z","shell.execute_reply":"2022-08-15T02:09:58.563341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Cabin is categorized as numerical variable\n\nAccording to data description, \n* Cabin - The cabin number where the passenger is staying. Takes the form **deck/num/side**, where side can be either P for Port or S for Starboard.\n\nI gonna divde Cabin for 2 columns;Deck, Side|","metadata":{}},{"cell_type":"markdown","source":"Also, PassengerID includes another inform; Whether the passenger is travelling with someone.\nAccording to description,\n* A unique Id for each passenger. Each Id takes the form gggg_pp where gggg indicates a group the passenger is travelling with and **pp is their number within the group**. People in a group are often family members, but not always.","metadata":{}},{"cell_type":"code","source":"combine = [train, test]\n\n# False/True to binary\nfor dataset in combine:\n    dataset['Cryosleep'] = 0\n    dataset.loc[dataset['CryoSleep'] == True, 'Cryosleep'] = 1\n\ndel train['CryoSleep']\ndel test['CryoSleep']\n\n# False/True to binary\nfor dataset in combine:\n    dataset['Vip'] = 0\n    dataset.loc[dataset['VIP'] == True, 'Vip'] = 1\n\ndel train['VIP']\ndel test['VIP']\n\n# False/True to binary\ntrain['Transported_binary'] = 0\ntrain.loc[train['Transported'] == True, 'Transported_binary'] = 1\n\ndel train['Transported']\n\n\n# Divide Cabin into 2 different columns\nfor dataset in combine:\n    dataset.Cabin.str.split('/')\n    dataset['Deck'] = dataset.Cabin.str.split('/').str[0]\n    dataset['Side'] = dataset.Cabin.str.split('/').str[2]\n    \ndel train['Cabin']\ndel test['Cabin']\n\n# Extract companion from passengerid\nfor dataset in combine:\n    dataset.PassengerId.str.split('_')\n    A =  dataset.PassengerId.str.split('_').str[1]\n    for idx in range(len(A)):\n        if A[idx].startswith('0'):\n            A[idx] = A[idx][1:]\n    dataset['Companion'] = A\n\ndel train['PassengerId']\ndel test['PassengerId']\n\n# Delete Name\ndel train['Name']\ndel test['Name']\n\ntrain['Age'] = train['Age'].fillna(train['Age'].median())\ntest['Age'] = test['Age'].fillna(test['Age'].median())\n\ntrain['RoomService'] = train['RoomService'].fillna(train['RoomService'].median())\ntest['RoomService'] = test['RoomService'].fillna(test['RoomService'].median())\n\ntrain['FoodCourt'] = train['FoodCourt'].fillna(train['FoodCourt'].median())\ntest['FoodCourt'] = test['FoodCourt'].fillna(test['FoodCourt'].median())\n\ntrain['ShoppingMall'] = train['ShoppingMall'].fillna(train['ShoppingMall'].median())\ntest['ShoppingMall'] = test['ShoppingMall'].fillna(test['ShoppingMall'].median())\n\ntrain['Spa'] = train['Spa'].fillna(train['Spa'].median())\ntest['Spa'] = test['Spa'].fillna(test['Spa'].median())\n\ntrain['VRDeck'] = train['VRDeck'].fillna(train['VRDeck'].median())\ntest['VRDeck'] = test['VRDeck'].fillna(test['VRDeck'].median())","metadata":{"execution":{"iopub.status.busy":"2022-08-15T02:09:58.596121Z","iopub.execute_input":"2022-08-15T02:09:58.597390Z","iopub.status.idle":"2022-08-15T02:09:58.903734Z","shell.execute_reply.started":"2022-08-15T02:09:58.597341Z","shell.execute_reply":"2022-08-15T02:09:58.902362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in train.columns:\n    print(i, train[i].nunique())\n    print(\"==================================\")\n\ncat_cols = [col for col in train.columns if train[col].nunique() < 10 or train[col].dtype == '0']\nnum_cols = [col for col in train.columns if train[col].nunique() > 10 and train[col].dtype != \"0\"]\ncat_but_car = [col for col in train.columns if train[col].dtype == '0' and train[col].nunique() > 10]\ncat_cols = [col for col in cat_cols if col not in cat_but_car]\n\nprint(\"Categorical Varibales: \", cat_cols)\nprint(\"==================================\")\nprint(\"Numerical Varibales: \", num_cols)","metadata":{"execution":{"iopub.status.busy":"2022-08-15T02:09:58.906231Z","iopub.execute_input":"2022-08-15T02:09:58.907501Z","iopub.status.idle":"2022-08-15T02:09:58.936081Z","shell.execute_reply.started":"2022-08-15T02:09:58.907448Z","shell.execute_reply":"2022-08-15T02:09:58.935069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def car_vars_histogram(data, cat_cols, plot=False):\n    print(pd.DataFrame({cat_cols: data[cat_cols].value_counts(),\n                      \"Ratio\": 100 * data[cat_cols].value_counts() / len(data)}))\n    print(\"--------------------------------------\")\n    \n    if plot:\n        sns.histplot(data = data, x = data[cat_cols])\n        plt.show()\n        print(\"--------------------------------------\")\n\nfor i in cat_cols:\n    car_vars_histogram(train, i, plot=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-15T02:09:58.937660Z","iopub.execute_input":"2022-08-15T02:09:58.938344Z","iopub.status.idle":"2022-08-15T02:10:00.837552Z","shell.execute_reply.started":"2022-08-15T02:09:58.938305Z","shell.execute_reply":"2022-08-15T02:10:00.836295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def num_vars_distplot(data, num_cols, plot=False):\n    print(data[num_cols].describe([0.92,0.94, 0.95, 0.96, 0.98, 0.99]).T)\n    for i in num_cols:\n        print(\"===========================================\")\n        if plot:\n            sns.displot(data[i])\n            plt.title(\"Ditstribution of Varaible\")\n            plt.xlabel(i)\n            plt.show()\n\nnum_vars_distplot(train, num_cols, plot=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-15T02:10:00.840275Z","iopub.execute_input":"2022-08-15T02:10:00.840648Z","iopub.status.idle":"2022-08-15T02:11:02.104393Z","shell.execute_reply.started":"2022-08-15T02:10:00.840617Z","shell.execute_reply":"2022-08-15T02:11:02.103395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Binarize categorical varibles - HomePlanet, Destination, Deck, Side\nfor dataset in combine:\n    dataset['Homeplanet'] = 0\n    dataset.loc[dataset['HomePlanet'] == 'Earth', 'Homeplanet'] = 1\n\ndel train['HomePlanet']\ndel test['HomePlanet']\n\nfor dataset in combine:\n    dataset['Destination_binary'] = 0\n    dataset.loc[dataset['Destination'] == 'TRAPPIST-1e', 'Destination_binary'] = 1\n\ndel train['Destination']\ndel test['Destination']\n\nfor dataset in combine:\n    dataset['Deck_binary'] = 0\n    dataset.loc[(dataset['Deck'] == 'F') | (dataset['Deck'] == 'G'), 'Deck_binary'] = 1\n\ndel train['Deck']\ndel test['Deck']\n\nfor dataset in combine:\n    dataset['Side_binary'] = 0\n    dataset.loc[dataset['Side'] == 'S', 'Side_binary'] = 1\n\ndel train['Side']\ndel test['Side']","metadata":{"execution":{"iopub.status.busy":"2022-08-15T02:11:02.105630Z","iopub.execute_input":"2022-08-15T02:11:02.106665Z","iopub.status.idle":"2022-08-15T02:11:02.136516Z","shell.execute_reply.started":"2022-08-15T02:11:02.106625Z","shell.execute_reply":"2022-08-15T02:11:02.135285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corr = train.corr()\nplt.figure(figsize=(15,15))\nsns.set(font_scale=1.25) # size of font\nsns.heatmap(corr, cbar=True, annot=True, square=True, fmt='.2f')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-15T02:11:02.138136Z","iopub.execute_input":"2022-08-15T02:11:02.138748Z","iopub.status.idle":"2022-08-15T02:11:03.208073Z","shell.execute_reply.started":"2022-08-15T02:11:02.138715Z","shell.execute_reply":"2022-08-15T02:11:03.206913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test","metadata":{"execution":{"iopub.status.busy":"2022-08-15T02:11:03.209489Z","iopub.execute_input":"2022-08-15T02:11:03.209833Z","iopub.status.idle":"2022-08-15T02:11:03.238108Z","shell.execute_reply.started":"2022-08-15T02:11:03.209801Z","shell.execute_reply":"2022-08-15T02:11:03.236742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = train.drop('Transported_binary', axis = 1)\nY_train = train['Transported_binary']\nX_test = test.copy()\nX_train.shape, Y_train.shape, X_test.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-15T02:11:03.239819Z","iopub.execute_input":"2022-08-15T02:11:03.240812Z","iopub.status.idle":"2022-08-15T02:11:03.252258Z","shell.execute_reply.started":"2022-08-15T02:11:03.240750Z","shell.execute_reply":"2022-08-15T02:11:03.251231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"logreg = LogisticRegression()\nlogreg.fit(X_train, Y_train)\nY_pred = logreg.predict(X_test)\nacc_log = round(logreg.score(X_train, Y_train) * 100, 2)\nacc_log","metadata":{"execution":{"iopub.status.busy":"2022-08-15T02:11:03.253704Z","iopub.execute_input":"2022-08-15T02:11:03.254185Z","iopub.status.idle":"2022-08-15T02:11:03.439468Z","shell.execute_reply.started":"2022-08-15T02:11:03.254144Z","shell.execute_reply":"2022-08-15T02:11:03.438123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"coeff_df = pd.DataFrame(X_train.columns.delete(0))\ncoeff_df.columns = ['Feature']\ncoeff_df['Correlation'] = pd.Series(logreg.coef_[0])\n\ncoeff_df.sort_values(by='Correlation', ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-15T02:11:03.445231Z","iopub.execute_input":"2022-08-15T02:11:03.446341Z","iopub.status.idle":"2022-08-15T02:11:03.469231Z","shell.execute_reply.started":"2022-08-15T02:11:03.446281Z","shell.execute_reply":"2022-08-15T02:11:03.467944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"svc = SVC()\nsvc.fit(X_train, Y_train)\nY_pred = svc.predict(X_test)\nacc_svc = round(svc.score(X_train, Y_train) * 100, 2)\nacc_svc","metadata":{"execution":{"iopub.status.busy":"2022-08-15T02:11:03.471450Z","iopub.execute_input":"2022-08-15T02:11:03.472372Z","iopub.status.idle":"2022-08-15T02:11:10.444011Z","shell.execute_reply.started":"2022-08-15T02:11:03.472320Z","shell.execute_reply":"2022-08-15T02:11:10.442912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"knn = KNeighborsClassifier(n_neighbors = 5)\nknn.fit(X_train, Y_train)\nY_pred = knn.predict(X_test)\nacc_knn = round(knn.score(X_train, Y_train) * 100, 2)\nacc_knn","metadata":{"execution":{"iopub.status.busy":"2022-08-15T02:11:10.445565Z","iopub.execute_input":"2022-08-15T02:11:10.446762Z","iopub.status.idle":"2022-08-15T02:11:11.473698Z","shell.execute_reply.started":"2022-08-15T02:11:10.446719Z","shell.execute_reply":"2022-08-15T02:11:11.472579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gaussian = GaussianNB()\ngaussian.fit(X_train, Y_train)\nY_pred = gaussian.predict(X_test)\nacc_gaussian = round(gaussian.score(X_train, Y_train) * 100, 2)\nacc_gaussian","metadata":{"execution":{"iopub.status.busy":"2022-08-15T02:11:11.475062Z","iopub.execute_input":"2022-08-15T02:11:11.475407Z","iopub.status.idle":"2022-08-15T02:11:11.515440Z","shell.execute_reply.started":"2022-08-15T02:11:11.475376Z","shell.execute_reply":"2022-08-15T02:11:11.514492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"perception = Perceptron()\nperception.fit(X_train, Y_train)\nY_pred = perception.predict(X_test)\nacc_perceptron = round(perception.score(X_train, Y_train) * 100, 2)\nacc_perceptron","metadata":{"execution":{"iopub.status.busy":"2022-08-15T02:11:11.516719Z","iopub.execute_input":"2022-08-15T02:11:11.517258Z","iopub.status.idle":"2022-08-15T02:11:11.575571Z","shell.execute_reply.started":"2022-08-15T02:11:11.517223Z","shell.execute_reply":"2022-08-15T02:11:11.572799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"linear_svc = LinearSVC()\nlinear_svc.fit(X_train, Y_train)\nY_pred = linear_svc.predict(X_test)\nacc_linear_svc = round(linear_svc.score(X_train, Y_train) * 100, 2)\nacc_linear_svc","metadata":{"execution":{"iopub.status.busy":"2022-08-15T02:11:11.577106Z","iopub.execute_input":"2022-08-15T02:11:11.577817Z","iopub.status.idle":"2022-08-15T02:11:12.352574Z","shell.execute_reply.started":"2022-08-15T02:11:11.577769Z","shell.execute_reply":"2022-08-15T02:11:12.351344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sgd = SGDClassifier()\nsgd.fit(X_train, Y_train)\nY_pred = sgd.predict(X_test)\nacc_sgd = round(sgd.score(X_train, Y_train) * 100, 2)\nacc_sgd","metadata":{"execution":{"iopub.status.busy":"2022-08-15T02:11:12.354577Z","iopub.execute_input":"2022-08-15T02:11:12.355447Z","iopub.status.idle":"2022-08-15T02:11:12.472626Z","shell.execute_reply.started":"2022-08-15T02:11:12.355383Z","shell.execute_reply":"2022-08-15T02:11:12.471222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"decision_tree = DecisionTreeClassifier()\ndecision_tree.fit(X_train, Y_train)\nY_pred = decision_tree.predict(X_test)\nacc_decision_tree = round(decision_tree.score(X_train, Y_train) * 100, 2)\nacc_decision_tree","metadata":{"execution":{"iopub.status.busy":"2022-08-15T02:11:12.475092Z","iopub.execute_input":"2022-08-15T02:11:12.476094Z","iopub.status.idle":"2022-08-15T02:11:12.593814Z","shell.execute_reply.started":"2022-08-15T02:11:12.476044Z","shell.execute_reply":"2022-08-15T02:11:12.592999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"random_forest = RandomForestClassifier(n_estimators=100)\nrandom_forest.fit(X_train, Y_train)\nY_pred = random_forest.predict(X_test)\nrandom_forest.score(X_train, Y_train)\nacc_random_forest = round(random_forest.score(X_train, Y_train) * 100, 2)\nacc_random_forest","metadata":{"execution":{"iopub.status.busy":"2022-08-15T02:11:12.595245Z","iopub.execute_input":"2022-08-15T02:11:12.595833Z","iopub.status.idle":"2022-08-15T02:11:14.152784Z","shell.execute_reply.started":"2022-08-15T02:11:12.595799Z","shell.execute_reply":"2022-08-15T02:11:14.151597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models = pd.DataFrame({\n    'Model': ['Support Vector Machines', 'KNN', 'Logistic Regression', \n              'Random Forest', 'Naive Bayes', 'Perceptron', \n              'Stochastic Gradient Decent', 'Linear SVC', \n              'Decision Tree'],\n    'Score': [acc_svc, acc_knn, acc_log, \n              acc_random_forest, acc_gaussian, acc_perceptron, \n              acc_sgd, acc_linear_svc, acc_decision_tree]})\nmodels.sort_values(by='Score', ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-15T02:11:14.154283Z","iopub.execute_input":"2022-08-15T02:11:14.154649Z","iopub.status.idle":"2022-08-15T02:11:14.171459Z","shell.execute_reply.started":"2022-08-15T02:11:14.154615Z","shell.execute_reply":"2022-08-15T02:11:14.169939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = pd.Series(random_forest.predict(X_test).astype(bool), name=\"Transported\")\n\nresults = pd.concat([Submission,predictions],axis=1)\n\nresults.to_csv(\"submission_2.csv\",index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-15T02:16:55.294754Z","iopub.execute_input":"2022-08-15T02:16:55.295774Z","iopub.status.idle":"2022-08-15T02:16:55.418539Z","shell.execute_reply.started":"2022-08-15T02:16:55.295727Z","shell.execute_reply":"2022-08-15T02:16:55.417429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}