{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"scrolled":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\nimport matplotlib as mpl\nimport matplotlib.pyplot as plt\nplt.style.use('fivethirtyeight')\n%matplotlib notebook\n# Any results you write to the current directory are saved as output.\n\nfrom sklearn.model_selection import GridSearchCV","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"titanic = pd.read_csv('../input/train.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"31b6c941b5525a512c8ca10d0d9ef9c999ece7e7"},"cell_type":"code","source":"#Wrangling the data set into usable varaibles\nlastname = titanic['Name'].str.split(\",\", n=1, expand=True)\ntitanic['LastName'] = lastname[0]\ntitanic['FirstName'] = lastname[1]\ntitanic['TravelCompanion'] = titanic['Ticket'].duplicated()\ntitanic['SameLastName'] = titanic['LastName'].duplicated()\ntitanic['ConfirmedFamily'] = (titanic['TravelCompanion'] == True) & (titanic['SameLastName'] == True)\ntitanic.drop(columns = ['Name','Ticket', 'Cabin', 'FirstName','LastName', 'PassengerId','SameLastName'], inplace = True)\n\n#Age has ~177 missing values here - do I drop the column, or should I fill it later on?\ntitanic.dropna(subset=['Embarked','Age'], inplace=True) \n#Age has ~177 missing values here - do I drop the column, or should I fill it later on?\n\ntitanic['Pclass'] = titanic['Pclass'].astype(str)\ntitanic['TravelCompanion'] = titanic['TravelCompanion'].astype(int)\ntitanic['ConfirmedFamily'] = titanic['ConfirmedFamily'].astype(int)\ntitanic.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"264b8284cbd4184a68cbe697b29a14049c906917"},"cell_type":"code","source":"#Split Data into x and y\ntitanic.dtypes\ndata_X = titanic.drop(columns = 'Survived')\ndata_Y = titanic['Survived']\ndata_X = pd.get_dummies(data_X, drop_first=True)\ndata_X.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"54a3c0e5d0499d28a2314001612ce1ca1cb58a8e"},"cell_type":"code","source":"# Split my Training into training and test data\ntitanic_train_X, titanic_test_X, titanic_train_Y, titanic_test_Y = train_test_split(data_X, data_Y, \n                                                                                       random_state=37,\n                                                                                       train_size = 0.7)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":false,"_uuid":"c473eaa0b5e76b46b0da247ab391db4b0ba94d1e"},"cell_type":"code","source":"#Find right parameters\n# The model you want to set the parameters for\nmodel = DecisionTreeClassifier(class_weight='balanced')\n\n# The parameters to search over for the model\nparams = {'max_depth':[2,3,4],\n          'max_features':['auto','log2',None]}\n\n\n# Prepare the GridSearch for cross validation\ngrid_search_dec_tree = GridSearchCV(model, # Note the model is DecisionTreeClassifier as stated above\n                                    param_grid=params, # The parameters to search over. \n                                   cv=10, # How many hold out sets to use\n                                   n_jobs = 1 # Number of parallel processes to run. \n                                   )\n\n# Do the cross validation on the training data \ngrid_search_dec_tree.fit(titanic_train_X, titanic_train_Y)\n\n# Select the best model\n\nbest_dec_tree_cv = grid_search_dec_tree.best_estimator_\n\n# Print the best parameter combination \nprint(grid_search_dec_tree.best_params_)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f55ccd23230fba030bc5663779d0b7a14b8caf5c"},"cell_type":"code","source":"# Finally test the performance of the best model on the test data\npred_Y = best_dec_tree_cv.predict(titanic_test_X)\n\n#Print the accuracy \nprint(sklmetrics.accuracy_score(titanic_test_Y, pred_Y))\nconf_mat = sklmetrics.confusion_matrix(titanic_test_Y, pred_Y)\nprint(conf_mat)\n\n# Confusion matrix\nsns.heatmap(conf_mat, fmt='d',square=True, annot=True, cbar = False, xticklabels = ['Failure','Success'], \n                                                          yticklabels = ['Failure','Success'])\nplt.xlabel(\"Predicted Value\")\nplt.ylabel(\"True Value\")","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}