{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e33643ed343aaa7b21e7489eed4eb598877d1a00"},"cell_type":"markdown","source":"Load the dataset"},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"dataset = pd.read_csv('../input/train.csv')\nX = dataset.iloc[:,2:12]\ny = dataset.iloc[:,1]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"b484a557553cc698546b0509e74c4a659c50136c"},"cell_type":"markdown","source":"Remove the columns \"Name\", Cabin and \"Ticket\", since I will not work with them now."},{"metadata":{"trusted":true,"_uuid":"cf692615381344dc7c34cefd54e3d0433031a6e2","scrolled":false},"cell_type":"code","source":"X = X.drop([\"Ticket\",\"Cabin\",\"Name\"], axis=1)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"c912a755ff2f12d2617eda8e0dbb5e85edde2175"},"cell_type":"markdown","source":"Which columns have missing values?"},{"metadata":{"trusted":true,"_uuid":"0fc11117411341f2034ce444dd4c20777a48fb92"},"cell_type":"code","source":"missing_val_count_by_column = (X.isnull().sum())\nprint(missing_val_count_by_column[missing_val_count_by_column > 0])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f16f98847ff8cc73312bf3358f74115cfcd26bc1"},"cell_type":"markdown","source":"Fill the missing values with the mean value. This is not the best option, \nwe could make a better approximation with the other parameters, but for \ntoday, I will use the simplest one. \nThe SimpleImputer class can only fill the missing features at once only if they are of the same type. In this case we have missing age and Embarked, which are different types. For this reason we have to fill them separatelly."},{"metadata":{"trusted":true,"_uuid":"2a8206e1e72f9eafde07c46652755689c9a5da27"},"cell_type":"code","source":"from sklearn.impute import SimpleImputer\nmy_imputer = SimpleImputer(strategy='most_frequent')\nX_with_column_names = X\nembarked = X.iloc[:,5:7]\nmissing_val_count_by_column = (embarked.isnull().sum())\nembarked = my_imputer.fit_transform(embarked)\nX.iloc[:,6]=embarked[:,1]\n\nimputer_age = SimpleImputer()\nage = X.iloc[:,2:3]\nage = imputer_age.fit_transform(age)\nX.iloc[:,2]=age","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a42fd275d58dd7284d20fb406bbdbf7819c71046"},"cell_type":"markdown","source":"Encode categorical Sex"},{"metadata":{"trusted":true,"_uuid":"20a24321e27981221d0eec67aed065bf5ff5b56d"},"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder, OneHotEncoder\nlabelencoder_X_1 = LabelEncoder()\nX.iloc[:,1] = labelencoder_X_1.fit_transform(X.iloc[:, 1])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f99100da20cc564d3b240fb2c3848947f09e8dd9"},"cell_type":"markdown","source":"Encode Embarked"},{"metadata":{"trusted":true,"_uuid":"633c48eab81c5cbe6cc658454597f9ca43f6c18e"},"cell_type":"code","source":"labelencoder_X_2 = LabelEncoder()\nX.iloc[:,6]= labelencoder_X_2.fit_transform(X.iloc[:,6].astype(str))\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"384713dbb1d128734ff1c1a9d136070c6ec4ce0a"},"cell_type":"markdown","source":"Encoding categorical data (Pclass)"},{"metadata":{"trusted":true,"_uuid":"3b66c3ef7fdeaa203f2463d960ca0ba3c456ee25"},"cell_type":"code","source":"onehotencoder = OneHotEncoder(categorical_features = [0])\nX = onehotencoder.fit_transform(X).toarray()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4781064cecef269a8994076b27eb8ea41ffb2af1"},"cell_type":"code","source":"Remove the dummy variable:","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"06f14c1a217584393ff6c57faef9eb8b403cf8b0"},"cell_type":"code","source":"X = X[:, 1:]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"1cde54ed4b73afe739b0850e017abc31374d1ac8"},"cell_type":"markdown","source":"The same with \"Embarked\""},{"metadata":{"_uuid":"ab32c5eb599027450f90cb33256b3cd285172259","trusted":true,"scrolled":true},"cell_type":"code","source":"onehotencoder = OneHotEncoder(categorical_features = [7])\nX = onehotencoder.fit_transform(X).toarray()\nX = X[:, 1:]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"035267af7cec586c4a7643c97a01c71ef3900212"},"cell_type":"markdown","source":"I'll try to build a ANN, even if there are only few data in the dataset."},{"metadata":{"_uuid":"ddac73c49661a14e5589ced07c2cbf17a0607465"},"cell_type":"markdown","source":"Splitting the dataset into the Training set and Test set"},{"metadata":{"trusted":true,"_uuid":"ce923493bfd9dadaa970ae3c6f5c0645e502f998"},"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size = 0.2, random_state = 0)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"c7f2194dfa89412ffc5b643179bebb88f0c45565"},"cell_type":"markdown","source":"Feature Scaling"},{"metadata":{"trusted":true,"_uuid":"5736a9e12759a8e15187bf8a278bbc9db3b89b31"},"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\nsc = StandardScaler()\nX_train = sc.fit_transform(X_train)\nX_test = sc.transform(X_test)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"c91ce16779691adb697da6a72f2b23828dbeee31"},"cell_type":"markdown","source":"## Part 2 - Now let's make the ANN!\n\nImporting the Keras libraries and packages"},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"855894f1fd36d14495d48956ba45d993042c2aa3"},"cell_type":"code","source":"import tensorflow\nimport keras\nfrom keras.models import Sequential\nfrom keras.layers import Dense","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"7740a09dac714b246943b58c3651321026ce75ae"},"cell_type":"markdown","source":"Initialising the ANN"},{"metadata":{"trusted":true,"_uuid":"e36d1b369456f3547fe8c9d647f2ce9e80720a44"},"cell_type":"code","source":"classifier = Sequential()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"b004abf72bfbf988783e97ca8633cc1f8b3c674b"},"cell_type":"markdown","source":"Adding the input layer and the first hidden layer"},{"metadata":{"trusted":true,"_uuid":"423df720bb813a305e1b1fc5f106551b8f50c9bc"},"cell_type":"code","source":"classifier.add(Dense(output_dim = 5, init = 'uniform', activation = 'relu', input_dim = 9))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"07a4c3aaec7d9e7b20842f78bdcafb8efb26e05b"},"cell_type":"code","source":"classifier.add(Dense(output_dim = 5, init = 'uniform', activation = 'relu'))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a7137554d01de3ffa8b49dbdc3d8a464ca913000"},"cell_type":"markdown","source":"Adding the output layer"},{"metadata":{"trusted":true,"_uuid":"88bfde465d0db1a45dd58c82f7ca244af53c5f0a"},"cell_type":"code","source":"classifier.add(Dense(output_dim = 1, init = 'uniform', activation = 'sigmoid'))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e4f69cc6157b8d0b6a51eea3a0f21251da5d7251"},"cell_type":"markdown","source":"Compiling the ANN"},{"metadata":{"trusted":true,"_uuid":"14f52a6a60bd8321d537e631ea7f5dc661a43a0d"},"cell_type":"code","source":"classifier.compile(optimizer = 'adam', loss = 'binary_crossentropy', metrics = ['accuracy'])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"71b93eac6cb10a0b20f65336b1927533af82df22"},"cell_type":"markdown","source":"Fitting the ANN to the Training set"},{"metadata":{"trusted":true,"_uuid":"9bc5f9fbdbf3d80fdcf8943a2c6f6b7384ef5e23"},"cell_type":"code","source":"classifier.fit(X_train, y_train, batch_size = 5, nb_epoch = 100)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"66a2beb17abcb7978a72d2bfd2b52c081db82720"},"cell_type":"markdown","source":"Predict the test set (from the splitting) and make the confusion matrix"},{"metadata":{"trusted":true,"_uuid":"7a26370b74bbd1d26ab4625e4e2425e4233d8069"},"cell_type":"code","source":"y_pred = classifier.predict(X_test)\ny_pred = (y_pred > 0.5)\n\nfrom sklearn.metrics import confusion_matrix\ncm = confusion_matrix(y_test, y_pred)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"dcd34dfe71fed825c9895923f62b529b1f02cedb"},"cell_type":"code","source":"cm","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"6bd447ed84a4003b81c47e16b89d49c0ddf6ed45"},"cell_type":"markdown","source":"# Let's see the results with the test.csv"},{"metadata":{"trusted":true,"_uuid":"e1298ffc59452a6827e5632de9e1724d01ac70c9"},"cell_type":"code","source":"test_dataset = pd.read_csv('../input/test.csv')\nX_real_test = test_dataset.iloc[:,1:12]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"045a076b274417566c875b10137ca94f7e955df2"},"cell_type":"code","source":"# Remove the columns \"Name\" and \"Ticket\", since I will not work with them now.\nX_real_test = X_real_test.drop([\"Ticket\",\"Cabin\",\"Name\"], axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a43bcdad5d2f57862593450ef4e403536ba33d09"},"cell_type":"code","source":"#Encode categorical Sex\nfrom sklearn.preprocessing import LabelEncoder, OneHotEncoder\nlabelencoder_X_1 = LabelEncoder()\nX_real_test.iloc[:,1] = labelencoder_X_1.fit_transform(X_real_test.iloc[:, 1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"444eec721460596d56d835bfeddcd1201c9a9e6a"},"cell_type":"code","source":"#Encode categorical Embarked\nlabelencoder_X_2 = LabelEncoder()\nX_real_test.iloc[:,6]= labelencoder_X_2.fit_transform(X_real_test.iloc[:,6].astype(str))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"570dc4676369f48221c12a3cb35c70f267d3dc67"},"cell_type":"code","source":"#Which columns have missing values?\nmissing_val_count_by_column = (X_real_test.isnull().sum())\nprint(missing_val_count_by_column[missing_val_count_by_column > 0])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"91bd8fd57172a3f9a52c4c1f8ba50e6dd1531bab"},"cell_type":"code","source":"#Fill the missing values with the mean value. This is not the best option, \n#we could make a better approximation with the other parameters, but for \n#today, I will use the simplest one.\nfrom sklearn.impute import SimpleImputer\nmy_imputer = SimpleImputer()\nX_with_column_names = X_real_test\nX_real_test = my_imputer.fit_transform(X_real_test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7bcbf1a97ad7df5a769e9783867dcbb24db72a31"},"cell_type":"code","source":"#Encoding categorical data (Pclass)\nonehotencoder = OneHotEncoder(categorical_features = [0])\nX_real_test = onehotencoder.fit_transform(X_real_test).toarray()\n#Remove the dummy variable:\nX_real_test = X_real_test[:, 1:] ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c01c4ea882f490e945eb75d7b3ab7b229cfe1c29"},"cell_type":"code","source":"#The same with \"Embarked\"\nonehotencoder = OneHotEncoder(categorical_features = [7])\nX_real_test = onehotencoder.fit_transform(X_real_test).toarray()\nX_real_test = X_real_test[:, 1:]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"2068ac08ba2fe3e2f771a9ac014705f42c101cc5"},"cell_type":"markdown","source":"Now we have the test dataset"},{"metadata":{"trusted":true,"_uuid":"58f2f40bf9a965937de5d5aca95ec0a38783b22f"},"cell_type":"code","source":"X_test_dataset = X_real_test","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d4974eda33974862aca6d687cf5a83a4a526f243"},"cell_type":"markdown","source":"##Predict the test set"},{"metadata":{"trusted":true,"_uuid":"a6044736d68c8c6d2beb4ee7c6722b2e8bbef12e"},"cell_type":"code","source":"survived_pred = classifier.predict(X_test_dataset)\nsurvived_pred =  (survived_pred > 0.5)\nsurvived_pred =  survived_pred*1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b881659f3a5032dac906eebc25ec509b88598671"},"cell_type":"code","source":"survived = pd.DataFrame(columns=['PassengerId','Survived'])\nsurvived['PassengerId'] = test_dataset.iloc[:,0]\nsurvived['Survived'] = survived_pred\nsubmission = pd.DataFrame({\n        \"PassengerId\": survived[\"PassengerId\"],\n        \"Survived\": survived[\"Survived\"]\n    })\nsubmission.to_csv('../output/submission.csv', index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}