{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Importing data","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np","metadata":{"execution":{"iopub.status.busy":"2022-07-07T00:59:35.228354Z","iopub.execute_input":"2022-07-07T00:59:35.228896Z","iopub.status.idle":"2022-07-07T00:59:35.233519Z","shell.execute_reply.started":"2022-07-07T00:59:35.228861Z","shell.execute_reply":"2022-07-07T00:59:35.232236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('../input/titanic/train.csv')\ndf_test = pd.read_csv('../input/titanic/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-07T00:59:35.240782Z","iopub.execute_input":"2022-07-07T00:59:35.241442Z","iopub.status.idle":"2022-07-07T00:59:35.255907Z","shell.execute_reply.started":"2022-07-07T00:59:35.241410Z","shell.execute_reply":"2022-07-07T00:59:35.254787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.set_index('PassengerId', inplace=True)\ndf_test.set_index('PassengerId', inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T00:59:35.257151Z","iopub.execute_input":"2022-07-07T00:59:35.258081Z","iopub.status.idle":"2022-07-07T00:59:35.271496Z","shell.execute_reply.started":"2022-07-07T00:59:35.258049Z","shell.execute_reply":"2022-07-07T00:59:35.270115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2022-07-07T00:59:35.273841Z","iopub.execute_input":"2022-07-07T00:59:35.274493Z","iopub.status.idle":"2022-07-07T00:59:35.313340Z","shell.execute_reply.started":"2022-07-07T00:59:35.274453Z","shell.execute_reply":"2022-07-07T00:59:35.312021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Exploration","metadata":{}},{"cell_type":"code","source":"# Number of na values for each independent variable \ndef custom_key(na_values):\n    return na_values[1]  # second parameter denotes the proportion of na value for the variable\nna_lst = []\nlength = df.shape[0]\nfor column in df.columns:\n    na_lst.append([column,df[column].isna().sum()/length])\n#na_lst.sort(key = custom_key, reverse = True)\nna_lst","metadata":{"execution":{"iopub.status.busy":"2022-07-07T00:59:35.314468Z","iopub.execute_input":"2022-07-07T00:59:35.315623Z","iopub.status.idle":"2022-07-07T00:59:35.328182Z","shell.execute_reply.started":"2022-07-07T00:59:35.315578Z","shell.execute_reply":"2022-07-07T00:59:35.327049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['Cabin'].unique()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T00:59:35.330394Z","iopub.execute_input":"2022-07-07T00:59:35.330797Z","iopub.status.idle":"2022-07-07T00:59:35.350625Z","shell.execute_reply.started":"2022-07-07T00:59:35.330762Z","shell.execute_reply":"2022-07-07T00:59:35.349951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#demographics by gender, class, embark","metadata":{"execution":{"iopub.status.busy":"2022-07-07T00:59:35.351509Z","iopub.execute_input":"2022-07-07T00:59:35.352256Z","iopub.status.idle":"2022-07-07T00:59:35.360934Z","shell.execute_reply.started":"2022-07-07T00:59:35.352230Z","shell.execute_reply":"2022-07-07T00:59:35.360072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#survival proportion by gender","metadata":{"execution":{"iopub.status.busy":"2022-07-07T00:59:35.362139Z","iopub.execute_input":"2022-07-07T00:59:35.363028Z","iopub.status.idle":"2022-07-07T00:59:35.375385Z","shell.execute_reply.started":"2022-07-07T00:59:35.362967Z","shell.execute_reply":"2022-07-07T00:59:35.373795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#survival proportion by gender and class","metadata":{"execution":{"iopub.status.busy":"2022-07-07T00:59:35.377173Z","iopub.execute_input":"2022-07-07T00:59:35.377989Z","iopub.status.idle":"2022-07-07T00:59:35.384760Z","shell.execute_reply.started":"2022-07-07T00:59:35.377948Z","shell.execute_reply":"2022-07-07T00:59:35.383960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data pre-processing","metadata":{}},{"cell_type":"code","source":"df['Cabin'] = df['Cabin'].astype(str)\ndf_test['Cabin'] = df_test['Cabin'].astype(str)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T00:59:35.386019Z","iopub.execute_input":"2022-07-07T00:59:35.386549Z","iopub.status.idle":"2022-07-07T00:59:35.399359Z","shell.execute_reply.started":"2022-07-07T00:59:35.386514Z","shell.execute_reply":"2022-07-07T00:59:35.398061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#cabin -> first char -> dummy\n#embarked -> dummy\n#sex -> dummy\ndf['Cabin_section'] = df.apply(lambda x: x['Cabin'][0] if x['Cabin'] != np.nan else np.nan, axis = 1)\ndf_test['Cabin_section'] = df_test.apply(lambda x: x['Cabin'][0] if x['Cabin'] != np.nan else np.nan, axis = 1)\n\ndf_cabin = pd.get_dummies(df['Cabin_section'], prefix = 'cabin_')\ndf_embarked = pd.get_dummies(df['Embarked'])\ndf_sex = pd.get_dummies(df['Sex'], drop_first = True)\n\ndf_test_cabin = pd.get_dummies(df_test['Cabin_section'], prefix = 'cabin_')\ndf_test_embarked = pd.get_dummies(df_test['Embarked'])\ndf_test_sex = pd.get_dummies(df_test['Sex'], drop_first = True)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T00:59:35.416753Z","iopub.execute_input":"2022-07-07T00:59:35.417917Z","iopub.status.idle":"2022-07-07T00:59:35.466851Z","shell.execute_reply.started":"2022-07-07T00:59:35.417865Z","shell.execute_reply":"2022-07-07T00:59:35.465085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_merged = pd.concat([df, df_cabin, df_embarked, df_sex], axis=1)\ndf_test_merged = pd.concat([df_test, df_test_cabin, df_test_embarked, df_test_sex], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T00:59:35.469439Z","iopub.execute_input":"2022-07-07T00:59:35.469856Z","iopub.status.idle":"2022-07-07T00:59:35.481314Z","shell.execute_reply.started":"2022-07-07T00:59:35.469826Z","shell.execute_reply":"2022-07-07T00:59:35.480230Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_merged","metadata":{"execution":{"iopub.status.busy":"2022-07-07T00:59:35.482376Z","iopub.execute_input":"2022-07-07T00:59:35.483167Z","iopub.status.idle":"2022-07-07T00:59:35.509426Z","shell.execute_reply.started":"2022-07-07T00:59:35.483121Z","shell.execute_reply":"2022-07-07T00:59:35.508391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_merged.drop(['Sex','Embarked','Cabin','Cabin_section','Name', 'Ticket'], axis=1, inplace=True)\ndf_test_merged.drop(['Sex','Embarked','Cabin','Cabin_section','Name', 'Ticket'], axis=1, inplace=True)\n#we try regex ticket number later","metadata":{"execution":{"iopub.status.busy":"2022-07-07T00:59:35.510743Z","iopub.execute_input":"2022-07-07T00:59:35.511069Z","iopub.status.idle":"2022-07-07T00:59:35.519432Z","shell.execute_reply.started":"2022-07-07T00:59:35.511037Z","shell.execute_reply":"2022-07-07T00:59:35.518768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Replace nan with median\nfrom sklearn.impute import SimpleImputer\nimp_median = SimpleImputer(missing_values = np.nan, strategy = 'median')\nidf_merged=pd.DataFrame(imp_median.fit_transform(df_merged))\nidf_merged.columns=df_merged.columns\nidf_merged.index=df_merged.index\nidf_test_merged=pd.DataFrame(imp_median.fit_transform(df_test_merged))\nidf_test_merged.columns=df_test_merged.columns\nidf_test_merged.index=df_test_merged.index","metadata":{"execution":{"iopub.status.busy":"2022-07-07T00:59:35.522256Z","iopub.execute_input":"2022-07-07T00:59:35.523120Z","iopub.status.idle":"2022-07-07T00:59:36.095057Z","shell.execute_reply.started":"2022-07-07T00:59:35.523095Z","shell.execute_reply":"2022-07-07T00:59:36.093960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"idf_merged","metadata":{"execution":{"iopub.status.busy":"2022-07-07T00:59:36.096249Z","iopub.execute_input":"2022-07-07T00:59:36.096511Z","iopub.status.idle":"2022-07-07T00:59:36.131305Z","shell.execute_reply.started":"2022-07-07T00:59:36.096487Z","shell.execute_reply":"2022-07-07T00:59:36.130450Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Scaling of age and fare\nfrom sklearn.preprocessing import StandardScaler\nfeatures = ['Age', 'Fare']\nautoscaler = StandardScaler()\nidf_merged[features] = autoscaler.fit_transform(idf_merged[features])\nidf_test_merged[features] = autoscaler.fit_transform(idf_test_merged[features])","metadata":{"execution":{"iopub.status.busy":"2022-07-07T00:59:36.132449Z","iopub.execute_input":"2022-07-07T00:59:36.133471Z","iopub.status.idle":"2022-07-07T00:59:36.152239Z","shell.execute_reply.started":"2022-07-07T00:59:36.133311Z","shell.execute_reply":"2022-07-07T00:59:36.151219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = idf_merged[set(idf_merged.columns)-set(['Survived'])]\ny_train = idf_merged['Survived']","metadata":{"execution":{"iopub.status.busy":"2022-07-07T00:59:36.153758Z","iopub.execute_input":"2022-07-07T00:59:36.154087Z","iopub.status.idle":"2022-07-07T00:59:36.167518Z","shell.execute_reply.started":"2022-07-07T00:59:36.154053Z","shell.execute_reply":"2022-07-07T00:59:36.166823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_,X_test = X_train.align(idf_test_merged, join='left', axis=1, fill_value=0)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T00:59:36.168692Z","iopub.execute_input":"2022-07-07T00:59:36.169872Z","iopub.status.idle":"2022-07-07T00:59:36.185933Z","shell.execute_reply.started":"2022-07-07T00:59:36.169836Z","shell.execute_reply":"2022-07-07T00:59:36.184907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split, GridSearchCV, cross_val_score\nX_train_train, X_train_test, y_train_train, y_train_test = train_test_split(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T00:59:36.187235Z","iopub.execute_input":"2022-07-07T00:59:36.187969Z","iopub.status.idle":"2022-07-07T00:59:36.197860Z","shell.execute_reply.started":"2022-07-07T00:59:36.187936Z","shell.execute_reply":"2022-07-07T00:59:36.196921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# SVM","metadata":{}},{"cell_type":"code","source":"from sklearn import svm\nclf = svm.SVC(kernel = 'rbf')","metadata":{"execution":{"iopub.status.busy":"2022-07-07T00:59:36.198818Z","iopub.execute_input":"2022-07-07T00:59:36.199870Z","iopub.status.idle":"2022-07-07T00:59:36.208072Z","shell.execute_reply.started":"2022-07-07T00:59:36.199845Z","shell.execute_reply":"2022-07-07T00:59:36.206669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf_fitted = clf.fit(X_train_train, y_train_train)\ny_train_pred = clf_fitted.predict(X_train_test)\nclf_fitted.score(X_train_test,y_train_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T00:59:36.209525Z","iopub.execute_input":"2022-07-07T00:59:36.209878Z","iopub.status.idle":"2022-07-07T00:59:36.257306Z","shell.execute_reply.started":"2022-07-07T00:59:36.209850Z","shell.execute_reply":"2022-07-07T00:59:36.256707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.dummy import DummyClassifier\n\n# Negative class (0) is most frequent\ndummy_majority = DummyClassifier(strategy = 'most_frequent').fit(X_train, y_train)\n# Therefore the dummy 'most_frequent' classifier always predicts class 0\ny_dummy_predictions = dummy_majority.predict(X_test)\n\n\n# accuracy is the default scoring metric\nprint('Cross-validation (accuracy)', cross_val_score(dummy_majority, X_train, y_train, cv=5))\n# use AUC as scoring metric\nprint('Cross-validation (AUC)', cross_val_score(dummy_majority, X_train, y_train, cv=5, scoring = 'roc_auc'))\n# use recall as scoring metric\nprint('Cross-validation (recall)', cross_val_score(dummy_majority, X_train, y_train, cv=5, scoring = 'recall'))","metadata":{"execution":{"iopub.status.busy":"2022-07-07T00:59:36.258617Z","iopub.execute_input":"2022-07-07T00:59:36.258993Z","iopub.status.idle":"2022-07-07T00:59:36.302195Z","shell.execute_reply.started":"2022-07-07T00:59:36.258955Z","shell.execute_reply":"2022-07-07T00:59:36.300269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# accuracy is the default scoring metric\nprint('Cross-validation (accuracy)', cross_val_score(clf, X_train, y_train, cv=5))\n# use AUC as scoring metric\nprint('Cross-validation (AUC)', cross_val_score(clf, X_train, y_train, cv=5, scoring = 'roc_auc'))\n# use recall as scoring metric\nprint('Cross-validation (recall)', cross_val_score(clf, X_train, y_train, cv=5, scoring = 'recall'))","metadata":{"execution":{"iopub.status.busy":"2022-07-07T00:59:36.307559Z","iopub.execute_input":"2022-07-07T00:59:36.307941Z","iopub.status.idle":"2022-07-07T00:59:36.646838Z","shell.execute_reply.started":"2022-07-07T00:59:36.307917Z","shell.execute_reply":"2022-07-07T00:59:36.645690Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"svm_linear = {'C': [0.001,0.1, 1, 10], \n              'kernel': ['linear']} \nsvm_others = {'C': [0.001, 0.1, 1, 10,],\n              'gamma': [0.001, 0.1, 1, 10], \n              'kernel': ['rbf']}\nparameters = [svm_linear, svm_others]\ngrid_clf_acc = GridSearchCV(clf, param_grid = parameters, scoring = 'accuracy', n_jobs = 4, verbose = 4)\ngrid_clf_acc.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T00:59:36.647887Z","iopub.execute_input":"2022-07-07T00:59:36.648101Z","iopub.status.idle":"2022-07-07T00:59:39.279014Z","shell.execute_reply.started":"2022-07-07T00:59:36.648079Z","shell.execute_reply":"2022-07-07T00:59:39.277984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Grid best parameter (max. accuracy): ', grid_clf_acc.best_params_)\nprint('Grid best score (accuracy): ', grid_clf_acc.best_score_)\nprint(grid_clf_acc.cv_results_['mean_test_score'])","metadata":{"execution":{"iopub.status.busy":"2022-07-07T00:59:39.280048Z","iopub.execute_input":"2022-07-07T00:59:39.280255Z","iopub.status.idle":"2022-07-07T00:59:39.286266Z","shell.execute_reply.started":"2022-07-07T00:59:39.280233Z","shell.execute_reply":"2022-07-07T00:59:39.285427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf_chosen = svm.SVC(kernel = 'rbf', C = 1, gamma = 0.1)\nsvc_fit = clf_chosen.fit(X_train,y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T00:59:39.287281Z","iopub.execute_input":"2022-07-07T00:59:39.287504Z","iopub.status.idle":"2022-07-07T00:59:39.319539Z","shell.execute_reply.started":"2022-07-07T00:59:39.287482Z","shell.execute_reply":"2022-07-07T00:59:39.318846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_test_pred = svc_fit.predict(X_test)\noutput = pd.DataFrame({\"PassengerId\": X_test.index, \"Survived\": y_test_pred})\noutput['Survived'] = pd.to_numeric(output['Survived'], downcast='integer')\noutput.to_csv(\"svm.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T00:59:39.320976Z","iopub.execute_input":"2022-07-07T00:59:39.321686Z","iopub.status.idle":"2022-07-07T00:59:39.344423Z","shell.execute_reply.started":"2022-07-07T00:59:39.321623Z","shell.execute_reply":"2022-07-07T00:59:39.343421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Logistic Regression","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression","metadata":{"execution":{"iopub.status.busy":"2022-07-07T00:59:39.346177Z","iopub.execute_input":"2022-07-07T00:59:39.346809Z","iopub.status.idle":"2022-07-07T00:59:39.351796Z","shell.execute_reply.started":"2022-07-07T00:59:39.346769Z","shell.execute_reply":"2022-07-07T00:59:39.350439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"log_reg = LogisticRegression(max_iter = 1000)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T00:59:39.355439Z","iopub.execute_input":"2022-07-07T00:59:39.355795Z","iopub.status.idle":"2022-07-07T00:59:39.365474Z","shell.execute_reply.started":"2022-07-07T00:59:39.355770Z","shell.execute_reply":"2022-07-07T00:59:39.364493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"log_reg_fit = log_reg.fit(X_train_train, y_train_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T00:59:39.366492Z","iopub.execute_input":"2022-07-07T00:59:39.367429Z","iopub.status.idle":"2022-07-07T00:59:39.404695Z","shell.execute_reply.started":"2022-07-07T00:59:39.367401Z","shell.execute_reply":"2022-07-07T00:59:39.403864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"log_reg_fit.score(X_train_test,y_train_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T00:59:39.406209Z","iopub.execute_input":"2022-07-07T00:59:39.406777Z","iopub.status.idle":"2022-07-07T00:59:39.416008Z","shell.execute_reply.started":"2022-07-07T00:59:39.406745Z","shell.execute_reply":"2022-07-07T00:59:39.415171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# accuracy is the default scoring metric\nprint('Cross-validation (accuracy)', cross_val_score(log_reg, X_train, y_train, cv=5))\n# use AUC as scoring metric\nprint('Cross-validation (AUC)', cross_val_score(log_reg, X_train, y_train, cv=5, scoring = 'roc_auc'))\n# use recall as scoring metric\nprint('Cross-validation (recall)', cross_val_score(log_reg, X_train, y_train, cv=5, scoring = 'recall'))","metadata":{"execution":{"iopub.status.busy":"2022-07-07T00:59:39.417605Z","iopub.execute_input":"2022-07-07T00:59:39.418180Z","iopub.status.idle":"2022-07-07T00:59:39.777190Z","shell.execute_reply.started":"2022-07-07T00:59:39.418148Z","shell.execute_reply":"2022-07-07T00:59:39.776329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"grid_dict = {\n    'C': [0.001,0.01,0.1,1,5,10,100]\n}\n\ngrid_clf_acc = GridSearchCV(log_reg, param_grid = grid_dict, scoring = 'accuracy')\ngrid_clf_acc.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T00:59:39.778700Z","iopub.execute_input":"2022-07-07T00:59:39.779280Z","iopub.status.idle":"2022-07-07T00:59:40.778226Z","shell.execute_reply.started":"2022-07-07T00:59:39.779234Z","shell.execute_reply":"2022-07-07T00:59:40.777431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Grid best parameter (max. accuracy): ', grid_clf_acc.best_params_)\nprint('Grid best score (accuracy): ', grid_clf_acc.best_score_)\nprint(grid_clf_acc.cv_results_['mean_test_score'])","metadata":{"execution":{"iopub.status.busy":"2022-07-07T00:59:40.779573Z","iopub.execute_input":"2022-07-07T00:59:40.780144Z","iopub.status.idle":"2022-07-07T00:59:40.785970Z","shell.execute_reply.started":"2022-07-07T00:59:40.780111Z","shell.execute_reply":"2022-07-07T00:59:40.785217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"log_reg_chosen = LogisticRegression(C=1,max_iter=1000)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T00:59:40.787348Z","iopub.execute_input":"2022-07-07T00:59:40.787973Z","iopub.status.idle":"2022-07-07T00:59:40.798220Z","shell.execute_reply.started":"2022-07-07T00:59:40.787939Z","shell.execute_reply":"2022-07-07T00:59:40.797364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"log_reg_chosen_fit = log_reg_chosen.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T00:59:40.799690Z","iopub.execute_input":"2022-07-07T00:59:40.800267Z","iopub.status.idle":"2022-07-07T00:59:40.830891Z","shell.execute_reply.started":"2022-07-07T00:59:40.800236Z","shell.execute_reply":"2022-07-07T00:59:40.830095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_test_pred = log_reg_chosen_fit.predict(X_test)\noutput = pd.DataFrame({\"PassengerId\": X_test.index, \"Survived\": y_test_pred})\noutput['Survived'] = pd.to_numeric(output['Survived'], downcast='integer')\noutput.to_csv(\"logreg.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T00:59:40.832324Z","iopub.execute_input":"2022-07-07T00:59:40.832909Z","iopub.status.idle":"2022-07-07T00:59:40.843028Z","shell.execute_reply.started":"2022-07-07T00:59:40.832877Z","shell.execute_reply":"2022-07-07T00:59:40.842239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output","metadata":{"execution":{"iopub.status.busy":"2022-07-07T00:59:40.844460Z","iopub.execute_input":"2022-07-07T00:59:40.845112Z","iopub.status.idle":"2022-07-07T00:59:40.859739Z","shell.execute_reply.started":"2022-07-07T00:59:40.845080Z","shell.execute_reply":"2022-07-07T00:59:40.858925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Random Forest","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier","metadata":{"execution":{"iopub.status.busy":"2022-07-07T00:59:40.861341Z","iopub.execute_input":"2022-07-07T00:59:40.862003Z","iopub.status.idle":"2022-07-07T00:59:40.953130Z","shell.execute_reply.started":"2022-07-07T00:59:40.861969Z","shell.execute_reply":"2022-07-07T00:59:40.952447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rfc = RandomForestClassifier()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T00:59:40.954328Z","iopub.execute_input":"2022-07-07T00:59:40.955022Z","iopub.status.idle":"2022-07-07T00:59:40.959450Z","shell.execute_reply.started":"2022-07-07T00:59:40.954989Z","shell.execute_reply":"2022-07-07T00:59:40.958544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"grid_dict = [{'n_estimators':[10,50,100,200,400],'max_depth':[None,3,4,5,6],'criterion':['gini','entropy']}]\ngrid_clf_acc = GridSearchCV(rfc, param_grid = grid_dict, scoring = 'accuracy', n_jobs = 4, verbose = 4)\ngrid_clf_acc.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T00:59:40.960718Z","iopub.execute_input":"2022-07-07T00:59:40.960952Z","iopub.status.idle":"2022-07-07T01:00:05.909138Z","shell.execute_reply.started":"2022-07-07T00:59:40.960930Z","shell.execute_reply":"2022-07-07T01:00:05.907672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Grid best parameter (max. accuracy): ', grid_clf_acc.best_params_)\nprint('Grid best score (accuracy): ', grid_clf_acc.best_score_)\nprint(grid_clf_acc.cv_results_['mean_test_score'])","metadata":{"execution":{"iopub.status.busy":"2022-07-07T01:00:05.911033Z","iopub.execute_input":"2022-07-07T01:00:05.911378Z","iopub.status.idle":"2022-07-07T01:00:05.920323Z","shell.execute_reply.started":"2022-07-07T01:00:05.911345Z","shell.execute_reply":"2022-07-07T01:00:05.919064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rfc_chosen = RandomForestClassifier(criterion = 'entropy', max_depth = 6, n_estimators = 200)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T01:00:05.921946Z","iopub.execute_input":"2022-07-07T01:00:05.922714Z","iopub.status.idle":"2022-07-07T01:00:05.930074Z","shell.execute_reply.started":"2022-07-07T01:00:05.922672Z","shell.execute_reply":"2022-07-07T01:00:05.929037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rfc_chosen_fit = rfc_chosen.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T01:00:05.932059Z","iopub.execute_input":"2022-07-07T01:00:05.933003Z","iopub.status.idle":"2022-07-07T01:00:06.259770Z","shell.execute_reply.started":"2022-07-07T01:00:05.932958Z","shell.execute_reply":"2022-07-07T01:00:06.258522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_test_pred = rfc_chosen_fit.predict(X_test)\noutput = pd.DataFrame({\"PassengerId\": X_test.index, \"Survived\": y_test_pred})\noutput['Survived'] = pd.to_numeric(output['Survived'], downcast='integer')\noutput.to_csv(\"rfc.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T01:00:06.261047Z","iopub.execute_input":"2022-07-07T01:00:06.261359Z","iopub.status.idle":"2022-07-07T01:00:06.306097Z","shell.execute_reply.started":"2022-07-07T01:00:06.261328Z","shell.execute_reply":"2022-07-07T01:00:06.305164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}