{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas  as pd\nimport seaborn as sns\nimport matplotlib as plt\nimport numpy as np\n\n\ntrain_path = \"../input/spaceship-titanic/train.csv\"\ntest_path = \"../input/spaceship-titanic/test.csv\"\ntrain_data = pd.read_csv(\"../input/spaceship-titanic/train.csv\")\ntest_data = pd.read_csv(\"../input/spaceship-titanic/test.csv\")\ndata = train_data.append(test_data)\n#data = pd.concat([test_data, train_data])\ndata.head(4)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T11:35:23.989488Z","iopub.execute_input":"2022-08-12T11:35:23.990292Z","iopub.status.idle":"2022-08-12T11:35:25.040060Z","shell.execute_reply.started":"2022-08-12T11:35:23.990249Z","shell.execute_reply":"2022-08-12T11:35:25.038909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### *Describing the data*\n### *Checking the data information: data types, name of columns, numbers of columns and rows*\n### *Checking the shape of our data*\n### *Checking if there are any duplicate in our data*","metadata":{}},{"cell_type":"code","source":"data.describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T11:36:38.837736Z","iopub.execute_input":"2022-08-12T11:36:38.838583Z","iopub.status.idle":"2022-08-12T11:36:38.878835Z","shell.execute_reply.started":"2022-08-12T11:36:38.838549Z","shell.execute_reply":"2022-08-12T11:36:38.877739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T11:36:50.672339Z","iopub.execute_input":"2022-08-12T11:36:50.672747Z","iopub.status.idle":"2022-08-12T11:36:50.696401Z","shell.execute_reply.started":"2022-08-12T11:36:50.672714Z","shell.execute_reply":"2022-08-12T11:36:50.695185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.shape","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.duplicated()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Checking if there are any missing values in our data set.\n### if there are, we will first of all display columns that have missing values in a tablelar format using list comprehension and the blow code","metadata":{}},{"cell_type":"code","source":"\nmissing_values = [missing_values for missing_values in data if data[missing_values].isnull().sum()>0]\nmissing_data = data[missing_values]\n\nmissing_data.head(3)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-12T11:37:55.708219Z","iopub.execute_input":"2022-08-12T11:37:55.708634Z","iopub.status.idle":"2022-08-12T11:37:55.740034Z","shell.execute_reply.started":"2022-08-12T11:37:55.708591Z","shell.execute_reply":"2022-08-12T11:37:55.739295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### After displaying columns with missing values in a tablelar format, we will use the isnull() and sum() functions to display the numbers of missing values with their respective columns and ordering them.","metadata":{}},{"cell_type":"code","source":"null_values = data.isnull().sum().sort_values(ascending=False)\nnull_values","metadata":{"execution":{"iopub.status.busy":"2022-08-12T11:39:08.221546Z","iopub.execute_input":"2022-08-12T11:39:08.222254Z","iopub.status.idle":"2022-08-12T11:39:08.236926Z","shell.execute_reply.started":"2022-08-12T11:39:08.222218Z","shell.execute_reply":"2022-08-12T11:39:08.236198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### HANDLING OUR MISSING VALUES.\n### We are using the SimpleImputer function to fill in the missing values\n###  ***If the values are of the object(categorical) data types, fill them in with \"most_frequent\"***\n### ***If the values are of the numerical data types, fill them in with the median.***\n### ***We are filling these missing values because the numbers of missing values are high and for a machine model to work well, it needs to be fitted with enough data***","metadata":{}},{"cell_type":"code","source":"from sklearn.impute import SimpleImputer\n#for values in data.columns:\n    \n    #if data[values].dtypes == \"object\" and data[values].isnull().sum()>0:\n        #Imputer = SimpleImputer(missing_values = np.nan, strategy=\"most_frequent\")\n       # data_cat = data.select_dtypes(exclude = np.number)\n        #data_cat_copy = data_cat.copy()\n        #data_cat = Imputer.fit_transform(data_cat)\n    #data_cat_frame = pd.DataFrame(data_cat, columns= data_cat_copy.columns)\n    #if data[values].dtypes in [\"int32\", \"int64\", \"float64\"] and data[values].isnull().sum()>0:\n        #Imputer = SimpleImputer(missing_values = np.nan, strategy=\"median\")\n        #data_num = data.select_dtypes(include = np.number)\n        #data_num_copy = data_num.copy()\n        #data_num = Imputer.fit_transform(data_num)\n    #data_num_frame = pd.DataFrame(data_num, columns = data_num_copy.columns)\n    \nfor values in missing_data.columns:\n    \n    if missing_data[values].dtypes == \"object\" and missing_data[values].isnull().sum()>0:\n        Imputer = SimpleImputer(missing_values = np.nan, strategy=\"most_frequent\")\n        cat_data = missing_data.select_dtypes(exclude = np.number)\n        data[cat_data.columns] = Imputer.fit_transform(cat_data)\n    \n    if missing_data[values].dtypes in [\"int32\", \"int64\", \"float64\"] and missing_data[values].isnull().sum()>0:\n        Imputer = SimpleImputer(missing_values = np.nan, strategy=\"median\")\n        num_data = missing_data.select_dtypes(include = np.number)\n        data[num_data.columns] = Imputer.fit_transform(num_data)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T11:39:56.303419Z","iopub.execute_input":"2022-08-12T11:39:56.304152Z","iopub.status.idle":"2022-08-12T11:39:56.881205Z","shell.execute_reply.started":"2022-08-12T11:39:56.304111Z","shell.execute_reply":"2022-08-12T11:39:56.880015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### After using SimpleImputer to fill in the missing values, we will check if there are still any missing value in the data","metadata":{}},{"cell_type":"code","source":"data.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T11:40:28.724992Z","iopub.execute_input":"2022-08-12T11:40:28.725998Z","iopub.status.idle":"2022-08-12T11:40:28.739940Z","shell.execute_reply.started":"2022-08-12T11:40:28.725932Z","shell.execute_reply":"2022-08-12T11:40:28.738916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.head(3)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T11:40:44.603430Z","iopub.execute_input":"2022-08-12T11:40:44.604576Z","iopub.status.idle":"2022-08-12T11:40:44.623453Z","shell.execute_reply.started":"2022-08-12T11:40:44.604534Z","shell.execute_reply":"2022-08-12T11:40:44.622224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Dropping the PassengerId column because we won't be needing it for the model","metadata":{}},{"cell_type":"code","source":"data_copy = data.copy()\ndata_copy = data.drop(\"PassengerId\", axis = 1)\ndata_copy.head(3)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T11:41:29.388180Z","iopub.execute_input":"2022-08-12T11:41:29.388816Z","iopub.status.idle":"2022-08-12T11:41:29.410642Z","shell.execute_reply.started":"2022-08-12T11:41:29.388784Z","shell.execute_reply":"2022-08-12T11:41:29.409595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}