{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This code I have developed is not perfect. There are a lot of improvements which can be done. Any improvements would be welcome.\n# I have tried to explain it as much as possible. If anyone has any doubts, please feel free to ask in the comments section.\n# I have obtained an accuracy of 79.9% using this code.\n# Importing the libraries.\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport pandas as pd\nfrom sklearn.metrics import confusion_matrix # For construction of the Confusion Matrix.\nimport seaborn as sns # I have used this library for visualising the Confusion Matrix.\n# The Keras library is for the implementation of the Neural Network. \nimport keras \nfrom keras.models import Sequential # For defining the type of the Neural Network\nfrom keras.layers import Dense  # For defining the layers of the Neural Network\nfrom keras.layers import Dropout # For Dropout Regularization\nfrom keras.optimizers import RMSprop # For the RMSprop optimizer \nfrom keras.callbacks import ReduceLROnPlateau # For Simulated Annealing\nfrom keras.wrappers.scikit_learn import KerasClassifier\n\n# Importing the dataset.\ntrain = pd.read_csv('../input/train.csv')\ntest = pd.read_csv('../input/test.csv')\n# Dropping the unncessary columns.\ntrain = train.drop([\"Name\", \"Cabin\", \"Embarked\", \"PassengerId\"], axis = 1)\ntest1 = test.drop([\"Name\", \"Cabin\", \"Embarked\", \"PassengerId\"], axis = 1) #Returns a dataframe.\ntest = test.drop([\"Name\", \"Cabin\", \"Embarked\", \"PassengerId\"], axis = 1).values #Returns a numpy object.\n# X is a set of independent variables using which we will predict the dependent variable. \nX1 = train.drop([\"Survived\"], axis = 1) # Returns a dataframe.\nX = train.drop([\"Survived\"], axis = 1).values # Returns a numpy object.\ny = train.iloc[:, 0].values # y is the dependent variable that is whether the passengers survived  or not.\n\n# Combining the Training and the Test Data.\nframes = [X1, test1]\nresult = pd.concat(frames ,keys = ['X','test'])\n# We converted the result dataframe to a numpy object. \n# But why? ->Because the Imputer class cannot work on a dataframe.\nresult = result.values\n\n# Checking the dataset for missing values.\n# result.isnull().describe\n# Handling the missing values.\nfrom sklearn.preprocessing import Imputer\n#Here we define an object of the Imputer class and our strategy is to replace the missing values (NaN) by the mean of the respective column.\nimputer = Imputer(missing_values = 'NaN', strategy = 'mean', axis = 0) \nimputer = imputer.fit(result[:, 2:3])\nresult[:, 2:3] = imputer.transform(result[:, 2:3])\nimputer = imputer.fit(result[:, -1:])\nresult[:, -1:] = imputer.transform(result[:, -1:])\n\n# Encoding the values that is mapping the strings to integers.\nfrom sklearn.preprocessing import LabelEncoder,OneHotEncoder\nlabelencoder = LabelEncoder()\nresult[:,5] = labelencoder.fit_transform(result[:,5])\nresult[:,1] = labelencoder.fit_transform(result[:,1])\n# Here we perform One Hot Encoding to remove any dependencies between the Encoded Values.\n# For example, if we have a column named country with three countries A,B,C.\n# If the model maps A to 0, B to 1, C to 2 then since 1>2, the model should not consider the country B greater than C.\nonehotencoder = OneHotEncoder(categorical_features = [1])\nresult = onehotencoder.fit_transform(result).toarray()\nonehotencoder = OneHotEncoder(categorical_features = [5])\nresult = onehotencoder.fit_transform(result).toarray()\n\n# Feature Scaling (Reuqired for Neural Network to reduce computation).\nfrom sklearn.preprocessing import StandardScaler\nsc = StandardScaler()\nresult = sc.fit_transform(result)\n\n# Splitting the combined dataframe back to the originial dataframes\nresult1 = pd.DataFrame(result)\nX1 = result1.iloc[0:891, :].values\ntest1 = result1.iloc[891:, :].values \n#I have used the used following formula for the number of nodes in the hidden layer.\n#Nh=Ns/(α∗(Ni+No))\n#Ni  = number of input neurons.\n#No = number of output neurons.\n#Ns = number of samples in training data set.\n#α = an arbitrary scaling factor usually 2-10.\nNh = int(891/32)\n# Initialising the ANN\nclassifier = Sequential()\n\n# Adding the input layer and the first hidden layer\nclassifier.add(Dense(units = Nh, kernel_initializer = 'uniform', activation = 'relu', input_dim = 15))\n#Adding dropout regularization to prevent overfitting without dropout I got an accuracy of 75 and with dropout it increased to 78.\nclassifier.add(Dropout(0.01))\n\n# Adding the second hidden layer\nclassifier.add(Dense(units = Nh, kernel_initializer = 'uniform', activation = 'relu'))\nclassifier.add(Dropout(0.01))\n\n# Adding the output layer\nclassifier.add(Dense(units = 1, kernel_initializer = 'uniform', activation = 'sigmoid'))\nclassifier.add(Dropout(0.01))\n\n# Define the optimizer\noptimizer = RMSprop(lr=0.001, rho=0.9, epsilon=1e-08, decay=0.0)\n# Compile the model\nclassifier.compile(optimizer = optimizer , loss = \"binary_crossentropy\", metrics=[\"accuracy\"])\n\n#In order to make the optimizer converge faster and closest to the global minimum of the loss function, i used an annealing method of the learning rate (LR).\n#The LR is the step by which the optimizer walks through the 'loss landscape'. The higher LR, the bigger are the steps and the quicker is the convergence. However the sampling is very poor with an high LR and the optimizer could probably fall into a local minima.\n#Its better to have a decreasing learning rate during the training to reach efficiently the global minimum of the loss function.\n#To keep the advantage of the fast computation time with a high LR, i decreased the LR dynamically every X steps (epochs) depending if it is necessary (when accuracy is not improved).\n#With the ReduceLROnPlateau function from Keras.callbacks, i choose to reduce the LR by half if the accuracy is not improved after 3 epochs.\n# Set a learning rate annealer\nlearning_rate_reduction = ReduceLROnPlateau(monitor='acc', \n                                            patience=3, \n                                            verbose=1, \n                                            factor=0.5, \n                                            min_lr=0.00001)\n\n# Fitting the ANN to the Training set\nhistory = classifier.fit(X1, y, batch_size = 25, epochs = 1000, callbacks = [learning_rate_reduction])\n# I have commented the Confusion Matrix code as I have directly predicted the test set.\n# We create a Confusion Matrix to check the performance of our model that is how many correct and incorrect predictions it has made.\n# Creating the Confusion Matrix\n#confusion_mtx = confusion_matrix(test1, final) \n# Visualise the Confusion Matrix \n#sns.heatmap(confusion_mtx, annot=True, fmt='d')\n# Predicting the Test set results\nfinal = classifier.predict(test1)\n# Creating the final dataframe in the required format\nfinal = (final > 0.5)\nfinal = final.astype(int)\nfinal = pd.DataFrame(final)\nfinal['PassengerId'] = pd.Series(data = np.arange(892,1310), index=final.index)\nfinal.columns = ['Survived','PassengerId']\ncolumnsTitles=[\"PassengerId\",\"Survived\"]\nfinal=final.reindex(columns=columnsTitles)\n\n# Exporting the dataframe\nfinal.to_csv('Predictions_ANN.csv', index = False)\n\n","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}