{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Topic: Project 1: Titanic\n# Course: Analytical Information Systems\n# Author: Sampat Ramesh Acharya\n# Professor: Gefei Zhang\n# Date: 6th July 2022\n\n################## Importing Libraries ##################\n\n# pandas: To be used for handling the datasets in the pandas dataframe for data processing and analysis\nimport pandas as pd\n\n# matplotlib: To be used for standard library to create visualizations\nimport matplotlib.pyplot as plt \n%matplotlib inline\n\n# seaborn: To be used for advanced visualization library to create more advanced charts\nimport seaborn as sn \n\n# numpy: To be used for linear algebra\nimport numpy as np \n \n# os: To be used for reading and writing the file\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-06T17:16:12.875322Z","iopub.execute_input":"2022-07-06T17:16:12.875741Z","iopub.status.idle":"2022-07-06T17:16:12.886647Z","shell.execute_reply.started":"2022-07-06T17:16:12.875708Z","shell.execute_reply":"2022-07-06T17:16:12.885292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Analysing the Provided Datasets**","metadata":{}},{"cell_type":"code","source":"# Reading the train.csv and test.csv files using pandas library\ntrain             = pd.read_csv('/kaggle/input/titanic/train.csv')\ntest              = pd.read_csv('/kaggle/input/titanic/test.csv')\n\n# Converting the male female data into Numeric data\n# This column can then be used to calculate probabilty\ntrain[\"Sex\"]      = np.where(train[\"Sex\"]==\"male\",0,1)\ntest[\"Sex\"]       = np.where(test[\"Sex\"]==\"male\",0,1)\n\n# Checking for NA values\nfor column in train.columns:\n    print(column, train[column].isna().sum())","metadata":{"execution":{"iopub.status.busy":"2022-07-06T17:16:12.891527Z","iopub.execute_input":"2022-07-06T17:16:12.892569Z","iopub.status.idle":"2022-07-06T17:16:12.913719Z","shell.execute_reply.started":"2022-07-06T17:16:12.892525Z","shell.execute_reply":"2022-07-06T17:16:12.912936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**As per the observation, there are NA values in \"Age\", \"Cabin\" and Embarked Column**","metadata":{}},{"cell_type":"code","source":"# No need to fill the na values for Cabin since it won't be considered for Prediction\n\n# Filling the NA values of age column with average age\naverage_age_train = train['Age'].mean(skipna = True)\ntrain['Age']      = train['Age'].fillna(value = average_age_train)\naverage_age_test  = test['Age'].mean(skipna = True)\ntest['Age']       = test['Age'].fillna(value = average_age_test)\n\n# Filling the NA values of embarked column with majority value\ntrain[\"Embarked\"] = np.where(train[\"Embarked\"]==\"S\",1, np.where(train[\"Embarked\"]==\"C\",2,np.where(train[\"Embarked\"]==\"Q\",3,4)))\ntrain['Embarked'] = train['Embarked'].fillna(value = 2)\ntest[\"Embarked\"]  = np.where(test[\"Embarked\"]==\"S\",1, np.where(test[\"Embarked\"]==\"C\",2,np.where(test[\"Embarked\"]==\"Q\",3,4)))\ntest['Embarked']  = test['Embarked'].fillna(value = 2)\n\n# Dropping unwanted columns from the training and test data\ntrain             = train.drop(columns = [\"PassengerId\",\"Name\",\"Ticket\",\"Cabin\"])\ntest_Ids          = test\ntest              = test.drop(columns = [\"PassengerId\",\"Name\",\"Ticket\",\"Cabin\"])\n\n# Reindexing the Features of the data set\ntrain             = train.reindex(columns = [\"Sex\",\"Age\",\"Pclass\",\"Fare\",\"Embarked\",\"SibSp\",\"Parch\",\"Survived\"])\ntest              = test.reindex(columns = [\"Sex\",\"Age\",\"Pclass\",\"Fare\",\"Embarked\",\"SibSp\",\"Parch\"])\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-06T17:16:12.915128Z","iopub.execute_input":"2022-07-06T17:16:12.915593Z","iopub.status.idle":"2022-07-06T17:16:12.939767Z","shell.execute_reply.started":"2022-07-06T17:16:12.915565Z","shell.execute_reply":"2022-07-06T17:16:12.938770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**For Gaussian classification, the classes should be continuous or normally distributed and also the two classes shouldn't be highly related**","metadata":{}},{"cell_type":"code","source":"# Checking for correlation between the classes\ncorr = train.iloc[:,:-1].corr(method='pearson')\ncmap = sn.diverging_palette(250,354,80,60,center='dark',as_cmap=True)\nsn.heatmap(corr, vmax=1, vmin=-.5, cmap=cmap, square=True, linewidths = .2, annot=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-06T17:16:12.941134Z","iopub.execute_input":"2022-07-06T17:16:12.941447Z","iopub.status.idle":"2022-07-06T17:16:13.504671Z","shell.execute_reply.started":"2022-07-06T17:16:12.941420Z","shell.execute_reply":"2022-07-06T17:16:13.503856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**As per the the above correlation between the features, the \"Fare\" and \"Pclass\" seems to be highly related. So Pclass can be dropped**","metadata":{}},{"cell_type":"code","source":"# Dropping Pclass from train and test data set\ntrain             = train.drop(columns = [\"Pclass\"])\ntest              = test.drop(columns = [\"Pclass\"])\n\n# checking for Normal distribution of the different features in the train data\ncontinuous_numeric_features = ['Age', 'Fare', 'Parch', 'SibSp', 'Embarked']\nfor feature in continuous_numeric_features:\n    sn.histplot(train[feature], kde=True)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-06T17:16:13.506314Z","iopub.execute_input":"2022-07-06T17:16:13.507233Z","iopub.status.idle":"2022-07-06T17:16:14.740269Z","shell.execute_reply.started":"2022-07-06T17:16:13.507197Z","shell.execute_reply":"2022-07-06T17:16:14.739008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**As per the above graphs, it seems like the data is normally distributed. These features can be considered for Gaussian Classification**","metadata":{}},{"cell_type":"markdown","source":"**Creating Class for Gaussian Classifier**","metadata":{}},{"cell_type":"code","source":"class Gaussian_Naive_Bayes:\n    def __init__(self, prior_value = None, class_num = None, \n               mean_value = None, variance_value = None, unique_classes = None):\n        self.prior_value    = prior_value\n        # how many unique classes\n        self.class_num      = class_num\n        # mean of x values\n        self.mean_value     = mean_value\n        # variance of x values\n        self.variance_value = variance_value\n        # the unique classes present\n        self.unique_classes = unique_classes\n    \n  # get the mean and variance of the x values\n  \n    def fit_dataSet(self, features, target):\n    # get the mean and variance of the x values\n        self.features       = features\n        self.target         = target\n        self.mean_value     = np.array(features.groupby(by=target).mean())\n        self.variance_value = np.array(features.groupby(by=target).var())\n        self.class_num      = len(np.unique(target))\n        self.unique_classes = np.unique(target)\n        self.prior_value    = 1 / self.class_num\n        return self\n\n    def mean_var_func(self):\n        # mean and variance from the trainig data\n        mean_val            = np.array(self.mean_value)\n        variance            = np.array(self.variance_value)\n\n    # pull and combine the corresponding mean and variance\n        self.mean_var       = []\n        for i in range(len(mean_val)):\n            m_row           = mean_val[i]\n            v_row           = variance[i]\n            for a, b in enumerate(m_row):\n                mean        = b\n                var         = v_row[a]\n                self.mean_var.append([mean, var])\n\n        return self.mean_var\n\n    def split_func(self):\n        split_value         = np.vsplit(np.array(self.mean_var_func()), self.class_num)\n        return split_value\n\n    def Gaussian_Classifier_Func(self, x_val, x_mean, x_var):\n  \n    # define the base formula for prediction probabilities\n    # Variance of the x value in question\n        self.x_val          = x_val\n\n        # x mean value\n        self.x_mean         = x_mean\n\n        # the x value that is being used for computation\n        self.x_var          = x_var\n\n        # natural log\n        e                   = np.e\n\n        # pi\n        pi                  = np.pi\n\n        # first part of the equation\n        # 1 divided by the sqrt of 2 * pi * y_variance\n        equation_1          = 1 / (np.sqrt(2 * pi * x_var))\n    \n        # second part of equation implementation\n        # denominator of equation\n        denom               = 2 * x_var\n\n        # numerator calculation\n        numerator           = (x_val - x_mean) ** 2\n\n        # the exponent\n        expo                = np.exp(-(numerator/denom))\n\n        prob                = equation_1 * expo\n\n        return prob\n\n    def predict_survived(self, test_data):\n        self.test_data = test_data\n        # calculate the probabilities using base formula above\n        # defining the mean and variance that has being split into\n        # various classes.\n\n        split_class = self.split_func()\n        prob = []\n        for i in range(self.class_num):\n            class_one = split_class[i]\n            for i in range(len(class_one)):\n                class_one_x_mean = class_one[i][0]\n                class_one_x_var = class_one[i][1]\n                x_value = test_data[i]\n                # now calculate the probabilities of each class. \n                prob.append([self.Gaussian_Classifier_Func(x_value, class_one_x_mean, \n                                                       class_one_x_var)])\n\n        # turn prob into an array\n\n        prob_array = np.array(prob)\n\n        # split the probability into various classes again\n\n        prob_split = np.vsplit(prob_array, self.class_num)\n\n        # calculate the final probabilities\n\n        final_probabilities = []\n\n        for i in prob_split:\n            class_prob = np.prod(i) * self.prior_value\n            final_probabilities.append(class_prob)\n\n        # determining the maximum probability \n        maximum_prob = max(final_probabilities)\n\n        # getting the index that corresponds to maximum probability\n        prob_index = final_probabilities.index(maximum_prob)\n\n        # using the index of the maximum probability to get\n        # the class that corresponds to the maximum probability\n        prediction = self.unique_classes[prob_index]\n\n        return prediction\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-06T17:16:14.742046Z","iopub.execute_input":"2022-07-06T17:16:14.742820Z","iopub.status.idle":"2022-07-06T17:16:14.760184Z","shell.execute_reply.started":"2022-07-06T17:16:14.742777Z","shell.execute_reply":"2022-07-06T17:16:14.759187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction = []\nGaussian_Classifier = Gaussian_Naive_Bayes()\nused_feature = [\"Sex\",\"Age\", \"Fare\", \"SibSp\",\"Embarked\"]\nGaussian_Classifier.fit_dataSet(train[used_feature], train['Survived'])\nfor i in range(len(test.index)):\n    test_first_data = np.array(test.iloc[i])\n    result = Gaussian_Classifier.predict_survived(test_first_data)\n    prediction.append(result)\n\ntest_Passenger_Id = test_Ids[\"PassengerId\"]\nprediction_data = {\"PassengerId\": test_Passenger_Id, \"Survived\": prediction }\nSubmission_data = pd.DataFrame(prediction_data)\nSubmission_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-06T17:16:14.763226Z","iopub.execute_input":"2022-07-06T17:16:14.763582Z","iopub.status.idle":"2022-07-06T17:16:14.911205Z","shell.execute_reply.started":"2022-07-06T17:16:14.763551Z","shell.execute_reply":"2022-07-06T17:16:14.910059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Submission_data.to_csv(\"Submission_data.csv\", index = False)","metadata":{"execution":{"iopub.status.busy":"2022-07-06T17:16:17.583824Z","iopub.execute_input":"2022-07-06T17:16:17.584182Z","iopub.status.idle":"2022-07-06T17:16:17.590263Z","shell.execute_reply.started":"2022-07-06T17:16:17.584152Z","shell.execute_reply":"2022-07-06T17:16:17.589329Z"},"trusted":true},"execution_count":null,"outputs":[]}]}