{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-17T13:47:53.840541Z","iopub.execute_input":"2022-07-17T13:47:53.841746Z","iopub.status.idle":"2022-07-17T13:47:53.855072Z","shell.execute_reply.started":"2022-07-17T13:47:53.841703Z","shell.execute_reply":"2022-07-17T13:47:53.853775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport io\n%matplotlib inline\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn import metrics\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\nfrom statsmodels.stats.outliers_influence import variance_inflation_factor\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.metrics import classification_report, confusion_matrix","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:47:53.857396Z","iopub.execute_input":"2022-07-17T13:47:53.858287Z","iopub.status.idle":"2022-07-17T13:47:53.870629Z","shell.execute_reply.started":"2022-07-17T13:47:53.858235Z","shell.execute_reply":"2022-07-17T13:47:53.869597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load data\ndf_train = pd.read_csv(\"/kaggle/input/spaceship-titanic/train.csv\") \ndf_test = pd.read_csv(\"/kaggle/input/spaceship-titanic/test.csv\") ","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:47:53.872388Z","iopub.execute_input":"2022-07-17T13:47:53.873201Z","iopub.status.idle":"2022-07-17T13:47:53.926727Z","shell.execute_reply.started":"2022-07-17T13:47:53.873158Z","shell.execute_reply":"2022-07-17T13:47:53.925585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head(3).style.background_gradient(axis=None) ","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:47:53.929666Z","iopub.execute_input":"2022-07-17T13:47:53.931479Z","iopub.status.idle":"2022-07-17T13:47:53.953098Z","shell.execute_reply.started":"2022-07-17T13:47:53.931418Z","shell.execute_reply":"2022-07-17T13:47:53.951814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.head(3).style.background_gradient(axis=None) ","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:47:53.955193Z","iopub.execute_input":"2022-07-17T13:47:53.956106Z","iopub.status.idle":"2022-07-17T13:47:53.980743Z","shell.execute_reply.started":"2022-07-17T13:47:53.956054Z","shell.execute_reply":"2022-07-17T13:47:53.979457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Train data shape :- \" + str(df_train.shape))\nprint(\"Test data shape :- \" + str(df_test.shape))","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:47:53.982502Z","iopub.execute_input":"2022-07-17T13:47:53.982940Z","iopub.status.idle":"2022-07-17T13:47:53.989933Z","shell.execute_reply.started":"2022-07-17T13:47:53.982886Z","shell.execute_reply":"2022-07-17T13:47:53.988826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Combine both train and test dataset to perform Preprocessing \ntrain_transported = df_train['Transported']\ndf_preprocess = pd.concat((df_train, df_test)).reset_index(drop = True)\ndf_preprocess.drop(['Transported'], axis = 1, inplace = True) # Removing Transported as it is dependent (output) variable.\ndf_preprocess.drop(['PassengerId'], axis = 1, inplace = True) # Removing PassengerId as it is not required for analysis.\ndf_preprocess.drop(['Name'], axis = 1, inplace = True) # Removing Name as it is not required for analysis.\nprint(\"Combined data shape :- \" + str(df_preprocess.shape))\ndf_preprocess.head(3).style.background_gradient(axis=None) ","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:47:53.991616Z","iopub.execute_input":"2022-07-17T13:47:53.992314Z","iopub.status.idle":"2022-07-17T13:47:54.032387Z","shell.execute_reply.started":"2022-07-17T13:47:53.992270Z","shell.execute_reply":"2022-07-17T13:47:54.031534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Extract categorical and numerical columns\ncategorical_columns = df_preprocess.dtypes[df_preprocess.dtypes == \"object\"].index\nprint(\"Categorical Columns :- \")\nprint(categorical_columns,\"\\n\\n\")\nnumerical_columns = df_preprocess.dtypes[df_preprocess.dtypes != \"object\"].index\nprint(\"Numerical Columns :- \")\nprint(numerical_columns,\"\\n\\n\")","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:47:54.033622Z","iopub.execute_input":"2022-07-17T13:47:54.034793Z","iopub.status.idle":"2022-07-17T13:47:54.047596Z","shell.execute_reply.started":"2022-07-17T13:47:54.034756Z","shell.execute_reply":"2022-07-17T13:47:54.046131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##### Reusebale functions -\n\ndef get_missing_values_percent(input_dataframe):\n  percent_missing = round(input_dataframe.isnull().sum() * 100 / len(input_dataframe),2)\n  missing_value_df = pd.DataFrame({'column_name': input_dataframe.columns,\n                                  'percent_missing': percent_missing})\n  missing_value_df = missing_value_df[missing_value_df['percent_missing'] > 0]\n  missing_value_df = missing_value_df.sort_values(by='percent_missing',ascending=False)\n  missing_value_df.set_index('column_name')\n  return missing_value_df\n","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:47:54.050895Z","iopub.execute_input":"2022-07-17T13:47:54.051495Z","iopub.status.idle":"2022-07-17T13:47:54.059166Z","shell.execute_reply.started":"2022-07-17T13:47:54.051449Z","shell.execute_reply":"2022-07-17T13:47:54.058279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Handle Missing Values - \nmissing_value_df = get_missing_values_percent(df_preprocess)\nmissing_value_df","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:47:54.060257Z","iopub.execute_input":"2022-07-17T13:47:54.060916Z","iopub.status.idle":"2022-07-17T13:47:54.090003Z","shell.execute_reply.started":"2022-07-17T13:47:54.060887Z","shell.execute_reply":"2022-07-17T13:47:54.088533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Replace nulls with MEAN for numerical columns -\nfor col in numerical_columns:\n  df_preprocess[col].fillna(df_preprocess[col].mean(),inplace=True)\n\n# Replace nulls with MODE for categorical columns -\nfor col in categorical_columns:\n  df_preprocess[col].fillna(df_preprocess[col].mode(),inplace=True)\n\nmissing_value_df = get_missing_values_percent(df_preprocess)\nmissing_value_df","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:47:54.091430Z","iopub.execute_input":"2022-07-17T13:47:54.093177Z","iopub.status.idle":"2022-07-17T13:47:54.147970Z","shell.execute_reply.started":"2022-07-17T13:47:54.093131Z","shell.execute_reply":"2022-07-17T13:47:54.147164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"CryoSleep :- \" + str(df_preprocess[\"CryoSleep\"].mode()),\"\\n\")\nprint(\"Cabin :- \" + str(df_preprocess[\"Cabin\"].mode()),\"\\n\")\nprint(\"VIP :- \" + str(df_preprocess[\"VIP\"].mode()),\"\\n\")\n#print(\"Name :- \" + str(df_preprocess[\"Name\"].mode()),\"\\n\")\nprint(\"HomePlanet :- \" + str(df_preprocess[\"HomePlanet\"].mode()),\"\\n\")\nprint(\"Destination :- \" + str(df_preprocess[\"Destination\"].mode()),\"\\n\")","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:47:54.149152Z","iopub.execute_input":"2022-07-17T13:47:54.149601Z","iopub.status.idle":"2022-07-17T13:47:54.172050Z","shell.execute_reply.started":"2022-07-17T13:47:54.149571Z","shell.execute_reply":"2022-07-17T13:47:54.170930Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_preprocess[\"CryoSleep\"].fillna(\"False\",inplace=True)\ndf_preprocess[\"Cabin\"].fillna(\"G/160/P\",inplace=True)\ndf_preprocess[\"VIP\"].fillna(\"False\",inplace=True)\ndf_preprocess[\"HomePlanet\"].fillna(\"Earth\",inplace=True)\ndf_preprocess[\"Destination\"].fillna(\"TRAPPIST-1e\",inplace=True)\n\nmissing_value_df = get_missing_values_percent(df_preprocess)\nmissing_value_df","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:47:54.173012Z","iopub.execute_input":"2022-07-17T13:47:54.173963Z","iopub.status.idle":"2022-07-17T13:47:54.210517Z","shell.execute_reply.started":"2022-07-17T13:47:54.173925Z","shell.execute_reply":"2022-07-17T13:47:54.209458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Encoding for categorical variables -\n\nfor col in categorical_columns:\n    label_encoder_obj = LabelEncoder() \n    label_encoder_obj.fit(list(df_preprocess[col].values)) \n    df_preprocess[col] = label_encoder_obj.transform(list(df_preprocess[col].values))\n\ndf_preprocess[categorical_columns].head(5).style.background_gradient(axis=None) ","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:47:54.211895Z","iopub.execute_input":"2022-07-17T13:47:54.212215Z","iopub.status.idle":"2022-07-17T13:47:54.356124Z","shell.execute_reply.started":"2022-07-17T13:47:54.212186Z","shell.execute_reply":"2022-07-17T13:47:54.355057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(nrows = 3, ncols = 3)    # axes is 2d array (3x3)\naxes = axes.flatten()         # Convert axes to 1d array of length 9\nfig.set_size_inches(15, 15)\n\nfor ax, col in zip(axes, df_preprocess.columns):\n  sns.histplot(df_preprocess[col], ax = ax)\n  ax.set_title(col)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:47:54.357565Z","iopub.execute_input":"2022-07-17T13:47:54.358358Z","iopub.status.idle":"2022-07-17T13:48:15.485602Z","shell.execute_reply.started":"2022-07-17T13:47:54.358323Z","shell.execute_reply":"2022-07-17T13:48:15.484369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Split back the data into train and test\nprint(\"Original Train data shape :- \" + str(df_train.shape))\nprint(\"Original Test data shape :- \" + str(df_test.shape))\nprint(\"Combined data shape :- \" + str(df_preprocess.shape))\ntrain_preprocessed = df_preprocess[0:8693]\ntest_preprocessed = df_preprocess[8693:]\nprint(\"Preprocessed Train data shape :- \" + str(train_preprocessed.shape))\nprint(\"Preprocessed Test data shape :- \" + str(test_preprocessed.shape))","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:48:15.487079Z","iopub.execute_input":"2022-07-17T13:48:15.487491Z","iopub.status.idle":"2022-07-17T13:48:15.494620Z","shell.execute_reply.started":"2022-07-17T13:48:15.487458Z","shell.execute_reply":"2022-07-17T13:48:15.493890Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = train_preprocessed\ny = train_transported\nfeature_columns = train_preprocessed.columns\n\ny = y.replace({False: 0, True: 1})\n#print(y.unique())\n\n# Splitting the train data \ntrain_X, val_X, train_y, val_y = train_test_split(X, y, test_size=0.30, random_state=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:48:15.495833Z","iopub.execute_input":"2022-07-17T13:48:15.496403Z","iopub.status.idle":"2022-07-17T13:48:15.518933Z","shell.execute_reply.started":"2022-07-17T13:48:15.496368Z","shell.execute_reply":"2022-07-17T13:48:15.517799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"classifier = LogisticRegression(#solver='liblinear', \n                                #class_weight = 'balanced', \n                                random_state = 0, \n                                max_iter=3000)\nclassifier.fit(train_X, train_y)\ny_pred = classifier.predict(val_X)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:48:15.520436Z","iopub.execute_input":"2022-07-17T13:48:15.520727Z","iopub.status.idle":"2022-07-17T13:48:15.994973Z","shell.execute_reply.started":"2022-07-17T13:48:15.520699Z","shell.execute_reply":"2022-07-17T13:48:15.993637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print (\"Accuracy : \", accuracy_score(val_y, y_pred))\nprint('Training accuracy = {}'.format(round(classifier.score(train_X,train_y)*100,2)))\nprint('Testing accuracy = {}'.format(round(classifier.score(val_X,val_y)*100,2)),\"\\n\")\nprint(\"Classification Report :-\")\nprint(classification_report(val_y, y_pred))","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:48:15.996918Z","iopub.execute_input":"2022-07-17T13:48:15.997703Z","iopub.status.idle":"2022-07-17T13:48:16.032820Z","shell.execute_reply.started":"2022-07-17T13:48:15.997657Z","shell.execute_reply":"2022-07-17T13:48:16.031583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_predictions = classifier.predict(test_preprocessed[feature_columns])\nresults = pd.DataFrame({'PassengerId': df_test['PassengerId'], 'Transported': model_predictions})\nresults.Transported = results.Transported.replace({0: False, 1: True})\nresults.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:48:16.034739Z","iopub.execute_input":"2022-07-17T13:48:16.035547Z","iopub.status.idle":"2022-07-17T13:48:16.062367Z","shell.execute_reply.started":"2022-07-17T13:48:16.035496Z","shell.execute_reply":"2022-07-17T13:48:16.061046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:48:16.064384Z","iopub.execute_input":"2022-07-17T13:48:16.065230Z","iopub.status.idle":"2022-07-17T13:48:16.090695Z","shell.execute_reply.started":"2022-07-17T13:48:16.065180Z","shell.execute_reply":"2022-07-17T13:48:16.089233Z"},"trusted":true},"execution_count":null,"outputs":[]}]}