{"cells":[{"metadata":{"_uuid":"d3d418ce6a43130d94b9f4f0fa3cfff318c0d142"},"cell_type":"markdown","source":"## I tried to simplify the analysis of Titanic dataset as much as possible, using Keras model and Deep Learning. Hope it will be useful\n##  Ho cercato di semplificare il più possibile l'analisi del dataset del Titanic, utilizzando Keras e deep learning. Spero vi sia utile"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# Import libraries\n# Importo le librerie\n%matplotlib inline\nimport matplotlib.pyplot as plt\nimport pandas as pd\nimport numpy as np\nimport seaborn as sns\nfrom sklearn.model_selection import train_test_split","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"38308cf657cf98dc4633acdb7081e29b36b59036"},"cell_type":"markdown","source":"## **Read data from files - leggo i dati dai files**"},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"# Read csv files\n# Leggo i files .csv\n\ndf_train = pd.read_csv('../input/train.csv', index_col=0)    #train dataset import. Set first column 'passengerId' as index\ndf_test = pd.read_csv('../input/test.csv', index_col=0)    #test dataset import. Set first column 'passengerId' as index\n\n# Concatenate two dataframes without 'Survived' column, to better management.\n# concateno i due dataframe senza la colonna 'Survived', per una miglior gestione\n\ndf_all = pd.concat([df_train, df_test], axis=0, sort=True).drop(['Survived'], axis = 1)\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"5253f4833a099d6c2e3c3204b4d1c88c3a8fa032"},"cell_type":"markdown","source":"## **Data visualization - Visualizzazione dei dati**"},{"metadata":{"trusted":true,"_uuid":"b15a007de6e576dc4d5f93d1aeb7afa605509020"},"cell_type":"code","source":"# First 5 train rows\n# Prime 5 righe dela df train\n\ndf_all.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"c06ef3394f0794614f27c8e0b466961b23444b77"},"cell_type":"markdown","source":"## **Manage Nan values - Gestione dei valori Nan**"},{"metadata":{"trusted":true,"_uuid":"c6229e197f5ad154c294cbae7eabffce789541af"},"cell_type":"code","source":"# Plot a heatmap to show how nan are distributed, first on train dataset....\n# Plotto una heatmap  per visualizzare come sono distribuiti i Nan, primo sul dataset di train...\n\nsns.heatmap(data=df_all.isnull())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8644468ce4c99153dc1465c6c5193b6761029f16"},"cell_type":"code","source":"# fill age's nan with the median of age's values grouped by 'Class' and 'Sex'\n# Riempio i nan della colonna 'age' con la mediana dei valori di 'age' raggruppati per 'Class' e 'Sex'\n\ndf_all['Age'] = df_all.groupby(['Pclass','Sex'], sort=False)['Age'].apply(lambda x: x.fillna(x.median()))\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f06ee1a3032b2e802c7dc718747d7ae11ee840ce"},"cell_type":"code","source":"# Count 'Embarked' values...\n# Conto i valori della colonna 'Embarked'...\n\ndf_all['Embarked'].value_counts(dropna=False).plot(kind='bar', figsize=(6,4), title='Embarked')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d212683efe6c613c40b088a0ae3b7bfddc385d2c"},"cell_type":"code","source":"# ... and fill Nan with the most frequent value ('S')\n# ... e riempi i Nan con il valore più frequente ('S')\n\ndf_all['Embarked'] = df_all['Embarked'].fillna('S')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9581c609c2873d80244726317fd35dd96b923447"},"cell_type":"code","source":"# fill also the unique Nan in 'Fare' test DF's column with the median of the class\n# riempi l'unico 'Nan' nella colonna 'Fare' del dataframe DF\n\ndf_all['Fare'] = df_all.groupby(['Pclass'], sort=False)['Fare'].apply(lambda x: x.fillna(x.median()))\n\ndf_all.info()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"29d586bc77c82f7fe60a6f4e3a22b91aba59e2e0"},"cell_type":"markdown","source":"* Nan remains only in column 'Cabin' , but doesn't matter, i'll drop this column later\n* Rimangono Nan values solo nella colonna Cabin ma non importa, verrà droppata più tardi"},{"metadata":{"_uuid":"54aef151e5310f78f23a9a06067851a2d09c7bed"},"cell_type":"markdown","source":"## **Data Manipulation - Manipolazione dei dati**"},{"metadata":{"trusted":true,"_uuid":"99fd123638b4b274941c170949049b784bb2b26f"},"cell_type":"code","source":"# From 'Name' column, Extract titles name\n# Estraggo il titolo dalla colonna 'Nome'\n\ndf_all['Title'] = df_all['Name'].apply(lambda x: x[x.find(',')+2:x.find('.')])\n\n# Try to define the correct title classification for un-classified values ..\n# Cerco di definire la corretta classificazione per i titoli..\n\n\ndf_all['Title'] = df_all['Title'].replace(['Mme','Ms'], 'Mrs')\ndf_all['Title'] = df_all['Title'].replace(['Mlle','Lady'], 'Miss')\n\n# ...and group the less common titles under 'other' label\n# ... \n\ndf_all['Title'] = df_all['Title'].replace(['the Countess',\n                                               'Capt', 'Col','Don', \n                                               'Dr', 'Major', 'Rev', \n                                               'Sir', 'Jonkheer', 'Dona'], 'Others')\n\ndf_all['Title'].value_counts().plot(kind='bar', figsize=(6,4), title='Title')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1555f5b3aff4c4135e08c21a6c53fe12c4c74ad3"},"cell_type":"code","source":"# The get_dummies converts categorical variables in dummies variables (boolean)\ndf_all = pd.get_dummies(df_all,columns=['Sex','Embarked','Title'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"852dd94d1d445824ed94785f1c38f0dbed8e71b9"},"cell_type":"code","source":"# Remove Cabin, Name and Ticket Columns because are unuseful\ndf_all.drop(['Cabin','Name','Ticket'], axis=1, inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"5d6f7e9869789f26e25517534cd7ac226db02d0f"},"cell_type":"markdown","source":"## **Try to find correlations between features - Cerco correlazioni tra le features**"},{"metadata":{"trusted":true,"_uuid":"ec3530cb41ba6b965ede8fe40cbd04742c49bc6d"},"cell_type":"code","source":"# With heatmap we try to find correlation among features\n# Con heatmap cerchiamo correlazioni tra le features\n\nplt.figure(figsize= (10,10), dpi=100)\nsns.heatmap(df_all.corr(), square=True, annot=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"672db86452a665a930ab15ab13383ecd22ff5255"},"cell_type":"markdown","source":"## **Data Analysis**"},{"metadata":{"trusted":true,"_uuid":"75a7972ca8b6ac354953a1944bcbd9f3bbaac28e"},"cell_type":"code","source":"# I use my train dataframe to assign values to y (predicted label) and x\ny = df_train['Survived']\nx = df_all.iloc[:891,1:] # original 891 train rows\n# con iloc assegno ad una variabile le colonne dalla 1 (seconda) in poi (i primi : significano tutte le righe)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"877b308fb187d5758c905ca2f7d38d4a491c64fb"},"cell_type":"code","source":"from sklearn.feature_selection import SelectKBest\nfrom sklearn.feature_selection import chi2\n\nbestfeatures = SelectKBest(score_func=chi2, k='all')\nfit = bestfeatures.fit(x,y)\ndfscores = pd.DataFrame(fit.scores_)\ndfcolumns = pd.DataFrame(x.columns)\n#concat two dataframes for better visualization \nfeatureScores = pd.concat([dfcolumns,dfscores],axis=1)\nfeatureScores.columns = ['Specs','Score']  #naming the dataframe columns\nprint(featureScores.nlargest(15,'Score'))  #print 10 best features","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8a88384f0c16e0e207f7e2400722cacce4abe293"},"cell_type":"code","source":"# With ExtraTreesClassifier i try to find correlations between features\n# Con l'ExtraTreesClassifier cercco di trovare correlazioni tra le features\n\nfrom sklearn.ensemble import ExtraTreesClassifier\nimport matplotlib.pyplot as plt\nmodel = ExtraTreesClassifier()\nmodel.fit(x,y)\nprint(model.feature_importances_) #use inbuilt class feature_importances of tree based classifiers\n#plot graph of feature importances for better visualization\nfeat_importances = pd.Series(model.feature_importances_, index=x.columns)\nfeat_importances.nlargest(15).plot(kind='barh')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"bddd4f0db10cda9e20d7f9a834acb7e8d56938c0"},"cell_type":"markdown","source":"Fare, Title_Others, Embarked_S, Embarked_C, , Embarked_Q are poorly correlated, so i drop them"},{"metadata":{"_uuid":"11c80f85fba7698b6e7f2d08e8d49903a9ceb919"},"cell_type":"markdown","source":"## **Prediction with deep learning**"},{"metadata":{"trusted":true,"_uuid":"97620a40d04a5dc85816061802308b5bad471c01"},"cell_type":"code","source":"# With train_test_split i divide my train df in two parts, so i can run my test and show how my predictions works\nX_train, X_test, Y_train, Y_test = train_test_split(x.drop(['Fare','Title_Others','Embarked_C','Embarked_S','Embarked_Q'], axis=1), y, test_size=0.3, random_state=42)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"25a31360f75d563df66607bbb6cfae77751df7fd"},"cell_type":"code","source":"# Test with deep learning with 2 model, activation sigmoid, optimizer adam learning rate .1 (0.799 in kaggle )\n# Test con deep learning con 2 modelli, activation sigmoid, optimizer adam e learning rate .1 (0.799 in kaggle )\n\n# Test with deep learning\n\nfrom keras import models\nfrom keras import layers\nfrom keras import optimizers\n\n# initialize the network\nmodel = models.Sequential()\n\n# add nodes to the network\nmodel.add( \n    layers.Dense(1,                    # no. of neurons\n                 activation='sigmoid', # activation function\n                 input_shape=(14,)      # shape of the input\n                ))\n\nmodel.add( \n    layers.Dense(1,                    # no. of neurons\n                 activation='sigmoid', # activation function\n                ))\n\n# finalize the network\nmodel.compile(\n    optimizers.Adam(lr=.01)\n, # lr is the learning rate\n    loss='binary_crossentropy',       # loss function\n    metrics=['accuracy'] )   # additional quality measure\n\n# train the network\nhist = model.fit( x=x, # training examples\n           y=y, # desired output\n           epochs=200, # number of training epochs \n           verbose=1)\n# 0.79425 in kaggle with 2 model, activation sigmoid, optimizer adam learning rate .1\nmodel.evaluate(x, y)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"2f9ce59f6da5afd0241b6157c5413a94fe553b19","trusted":true},"cell_type":"code","source":"# Plot accuracy and loss\n# Plotto accuracy and loss\n\nfig, axes = plt.subplots(figsize=(6,6))\n\naxes.plot(hist.history['loss'], label='Loss')\naxes.plot(hist.history['acc'], label='Accuracy')\n\naxes.set_title(\"Training History\", fontsize=18)\naxes.set_xlabel(\"Epochs\", fontsize=18)\naxes.legend(fontsize=20)\n\n# Final accuracy\nprint (\"Accuracy:\", hist.history['acc'][-1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"77824fff38dc01fdcb070f2541e413bc2c0f1f60"},"cell_type":"code","source":"# write results in a new 'Survived' column df test dataframe. \n# I take only last 418 rows of df_all dataframe, because they represent my test df\ndf_test['Survived'] = model.predict_classes(df_all.iloc[891:,1:])\ndf_test['Survived'].to_csv(\"../prediction.csv\", header=\"PassengerId,Survived\")\ndf_test.drop(['Survived'], axis=1, inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"1f399df6601685c0257e9437ab95ee79ff2ea4c9"},"cell_type":"markdown","source":""}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}