{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"So, let's start with the Titanic Challenge. I'm just learning and I'll try different types of classifiers...","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd ","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-07T17:43:22.246823Z","iopub.execute_input":"2022-08-07T17:43:22.247294Z","iopub.status.idle":"2022-08-07T17:43:22.260921Z","shell.execute_reply.started":"2022-08-07T17:43:22.247143Z","shell.execute_reply":"2022-08-07T17:43:22.259493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Loading the data","metadata":{}},{"cell_type":"code","source":"df_train = pd.read_csv(\"/kaggle/input/titanic/train.csv\", index_col = 'PassengerId')\ndf_train.sample(10)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:43:22.262392Z","iopub.execute_input":"2022-08-07T17:43:22.262649Z","iopub.status.idle":"2022-08-07T17:43:22.301646Z","shell.execute_reply.started":"2022-08-07T17:43:22.262618Z","shell.execute_reply":"2022-08-07T17:43:22.300332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = pd.read_csv(\"/kaggle/input/titanic/test.csv\", index_col = 'PassengerId')\ndf_test.sample(10)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:43:22.303782Z","iopub.execute_input":"2022-08-07T17:43:22.304170Z","iopub.status.idle":"2022-08-07T17:43:22.329737Z","shell.execute_reply.started":"2022-08-07T17:43:22.304134Z","shell.execute_reply":"2022-08-07T17:43:22.328746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n%matplotlib inline\nimport seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:43:22.333099Z","iopub.execute_input":"2022-08-07T17:43:22.334090Z","iopub.status.idle":"2022-08-07T17:43:22.883201Z","shell.execute_reply.started":"2022-08-07T17:43:22.334031Z","shell.execute_reply":"2022-08-07T17:43:22.881319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:43:22.885826Z","iopub.execute_input":"2022-08-07T17:43:22.886207Z","iopub.status.idle":"2022-08-07T17:43:22.905618Z","shell.execute_reply.started":"2022-08-07T17:43:22.886158Z","shell.execute_reply":"2022-08-07T17:43:22.904385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:43:22.907464Z","iopub.execute_input":"2022-08-07T17:43:22.908165Z","iopub.status.idle":"2022-08-07T17:43:22.924015Z","shell.execute_reply.started":"2022-08-07T17:43:22.908127Z","shell.execute_reply":"2022-08-07T17:43:22.922723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from https://medium.datadriveninvestor.com/step-by-step-exploratory-data-analysis-of-titanic-dataset-2d0fb09b0e86\nsns.heatmap(df_train.isnull(),yticklabels=False,cbar=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:43:22.925839Z","iopub.execute_input":"2022-08-07T17:43:22.932483Z","iopub.status.idle":"2022-08-07T17:43:23.241543Z","shell.execute_reply.started":"2022-08-07T17:43:22.932401Z","shell.execute_reply":"2022-08-07T17:43:23.240373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"So, the columns 'Age', 'Embarked' and 'Cabin' have missing values (and Cabin is missing for more than 3/4 of the passengers, in fact)","metadata":{}},{"cell_type":"code","source":"df_test.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:43:23.243332Z","iopub.execute_input":"2022-08-07T17:43:23.243853Z","iopub.status.idle":"2022-08-07T17:43:23.263689Z","shell.execute_reply.started":"2022-08-07T17:43:23.243805Z","shell.execute_reply":"2022-08-07T17:43:23.262673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in ['Survived', 'Pclass', 'Sex']:\n    print(df_train[[col]].value_counts())","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:43:23.265398Z","iopub.execute_input":"2022-08-07T17:43:23.265870Z","iopub.status.idle":"2022-08-07T17:43:23.289091Z","shell.execute_reply.started":"2022-08-07T17:43:23.265827Z","shell.execute_reply":"2022-08-07T17:43:23.288116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's look at titles","metadata":{}},{"cell_type":"code","source":"df_train['Title'] = df_train['Name'].str.extract('([A-Za-z]+)\\.')\ndf_train","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:43:23.290717Z","iopub.execute_input":"2022-08-07T17:43:23.291241Z","iopub.status.idle":"2022-08-07T17:43:23.333704Z","shell.execute_reply.started":"2022-08-07T17:43:23.291178Z","shell.execute_reply":"2022-08-07T17:43:23.332607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test['Title'] = df_test['Name'].str.extract('([A-Za-z]+)\\.')\ndf_test","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:43:23.335849Z","iopub.execute_input":"2022-08-07T17:43:23.336256Z","iopub.status.idle":"2022-08-07T17:43:23.383196Z","shell.execute_reply.started":"2022-08-07T17:43:23.336194Z","shell.execute_reply":"2022-08-07T17:43:23.382016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['Title'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:43:23.384911Z","iopub.execute_input":"2022-08-07T17:43:23.385464Z","iopub.status.idle":"2022-08-07T17:43:23.397348Z","shell.execute_reply.started":"2022-08-07T17:43:23.385417Z","shell.execute_reply":"2022-08-07T17:43:23.396063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.loc[df_train['Title'].isin(['Master'])]","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:43:23.403554Z","iopub.execute_input":"2022-08-07T17:43:23.404156Z","iopub.status.idle":"2022-08-07T17:43:23.453832Z","shell.execute_reply.started":"2022-08-07T17:43:23.404107Z","shell.execute_reply":"2022-08-07T17:43:23.452329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.loc[df_train['Title'] == 'Master']['Age'].max()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:43:23.455373Z","iopub.execute_input":"2022-08-07T17:43:23.455725Z","iopub.status.idle":"2022-08-07T17:43:23.470343Z","shell.execute_reply.started":"2022-08-07T17:43:23.455678Z","shell.execute_reply":"2022-08-07T17:43:23.469342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.loc[df_train['Title'] == 'Mr']['Age'].min()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:43:23.471903Z","iopub.execute_input":"2022-08-07T17:43:23.472988Z","iopub.status.idle":"2022-08-07T17:43:23.489856Z","shell.execute_reply.started":"2022-08-07T17:43:23.472935Z","shell.execute_reply":"2022-08-07T17:43:23.488908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"So, 'Master' is a title for young boys.","metadata":{}},{"cell_type":"code","source":"df_train.loc[df_train['Title'].isin(['Dr', 'Rev', 'Jonkheer'])]","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:43:23.492405Z","iopub.execute_input":"2022-08-07T17:43:23.492845Z","iopub.status.idle":"2022-08-07T17:43:23.520527Z","shell.execute_reply.started":"2022-08-07T17:43:23.492738Z","shell.execute_reply":"2022-08-07T17:43:23.519643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Wikipedia: Jonkheer (female equivalent: jonkvrouw; French: Écuyer; English: Squire) is an honorific in the Low Countries denoting the lowest rank within the nobility. In the Netherlands, this in general concerns a prefix used by the untitled nobility. In Belgium, this is the lowest title within the nobility system, recognised by the Court of Cassation.","metadata":{}},{"cell_type":"code","source":"df_train.loc[df_train['Title'].isin(['Major', 'Col', 'Capt'])]","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:43:23.522310Z","iopub.execute_input":"2022-08-07T17:43:23.522881Z","iopub.status.idle":"2022-08-07T17:43:23.544331Z","shell.execute_reply.started":"2022-08-07T17:43:23.522832Z","shell.execute_reply":"2022-08-07T17:43:23.543214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test['Title'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:43:23.545902Z","iopub.execute_input":"2022-08-07T17:43:23.546766Z","iopub.status.idle":"2022-08-07T17:43:23.557287Z","shell.execute_reply.started":"2022-08-07T17:43:23.546717Z","shell.execute_reply":"2022-08-07T17:43:23.556312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.loc[df_test['Title'].isin(['Dr', 'Rev', 'Dona'])]","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:43:23.559178Z","iopub.execute_input":"2022-08-07T17:43:23.559775Z","iopub.status.idle":"2022-08-07T17:43:23.587430Z","shell.execute_reply.started":"2022-08-07T17:43:23.559727Z","shell.execute_reply":"2022-08-07T17:43:23.586501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# (Attempt of) feature engineering","metadata":{}},{"cell_type":"code","source":"# idea from CORAZON17 \ndef convert_title(title):\n    if title in [\"Ms\", \"Mlle\", \"Miss\"]:\n        return \"Miss\"\n    elif title in [\"Mme\", \"Mrs\", \"Countess\", \"Lady\", \"Dona\"]:\n        return \"Mrs\"\n    elif title in [\"Mr\", \"Major\", \"Col\", \"Capt\", \"Sir\", \"Don\", \"Jonkheer\"]:\n        return \"Mr\"\n    elif title == \"Master\":\n        return \"Master\"\n    elif title == \"Rev\":\n        return \"Rev\"\n    else:\n        return title       \n","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:43:23.588935Z","iopub.execute_input":"2022-08-07T17:43:23.589767Z","iopub.status.idle":"2022-08-07T17:43:23.597492Z","shell.execute_reply.started":"2022-08-07T17:43:23.589721Z","shell.execute_reply":"2022-08-07T17:43:23.596555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train = df_train.Survived\nX_train_full = df_train.drop(['Survived'], axis=1)\nX_train = X_train_full.drop(['Name','Ticket', 'Cabin'], axis=1)\nX_train","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:43:23.599279Z","iopub.execute_input":"2022-08-07T17:43:23.599839Z","iopub.status.idle":"2022-08-07T17:43:23.633777Z","shell.execute_reply.started":"2022-08-07T17:43:23.599779Z","shell.execute_reply":"2022-08-07T17:43:23.632854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:43:23.635574Z","iopub.execute_input":"2022-08-07T17:43:23.636261Z","iopub.status.idle":"2022-08-07T17:43:23.650036Z","shell.execute_reply.started":"2022-08-07T17:43:23.636185Z","shell.execute_reply":"2022-08-07T17:43:23.649336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test = df_test.drop(['Name','Ticket', 'Cabin'], axis=1)\nX_test","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:43:23.653867Z","iopub.execute_input":"2022-08-07T17:43:23.654331Z","iopub.status.idle":"2022-08-07T17:43:23.680199Z","shell.execute_reply.started":"2022-08-07T17:43:23.654293Z","shell.execute_reply":"2022-08-07T17:43:23.679320Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train[\"Title\"] = df_train[\"Title\"].map(convert_title)\nX_train[\"Title\"].value_counts()\n","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:43:23.682103Z","iopub.execute_input":"2022-08-07T17:43:23.682863Z","iopub.status.idle":"2022-08-07T17:43:23.694894Z","shell.execute_reply.started":"2022-08-07T17:43:23.682817Z","shell.execute_reply":"2022-08-07T17:43:23.694155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test[\"Title\"] = df_test[\"Title\"].map(convert_title)\nX_test[\"Title\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:43:23.696642Z","iopub.execute_input":"2022-08-07T17:43:23.697447Z","iopub.status.idle":"2022-08-07T17:43:23.711626Z","shell.execute_reply.started":"2022-08-07T17:43:23.697391Z","shell.execute_reply":"2022-08-07T17:43:23.710632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for title in X_train[\"Title\"].value_counts().keys():\n    m = X_train[X_train[\"Title\"]==title]['Age'].median()\n    print(title, m)\n    X_train.loc[(X_train.Age.isnull()) & (X_train[\"Title\"]==title), 'Age'] = m\n    X_test.loc[(X_test.Age.isnull()) & (X_test[\"Title\"]==title), 'Age'] = m\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:43:23.713288Z","iopub.execute_input":"2022-08-07T17:43:23.713874Z","iopub.status.idle":"2022-08-07T17:43:23.744172Z","shell.execute_reply.started":"2022-08-07T17:43:23.713828Z","shell.execute_reply":"2022-08-07T17:43:23.743264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:43:23.745833Z","iopub.execute_input":"2022-08-07T17:43:23.746357Z","iopub.status.idle":"2022-08-07T17:43:23.757170Z","shell.execute_reply.started":"2022-08-07T17:43:23.746311Z","shell.execute_reply":"2022-08-07T17:43:23.756082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:43:23.758663Z","iopub.execute_input":"2022-08-07T17:43:23.759394Z","iopub.status.idle":"2022-08-07T17:43:23.771862Z","shell.execute_reply.started":"2022-08-07T17:43:23.759358Z","shell.execute_reply":"2022-08-07T17:43:23.770995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train['Embarked'].fillna(value=X_train['Embarked'].mode()[0],inplace=True)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:43:23.773446Z","iopub.execute_input":"2022-08-07T17:43:23.775102Z","iopub.status.idle":"2022-08-07T17:43:23.782583Z","shell.execute_reply.started":"2022-08-07T17:43:23.775052Z","shell.execute_reply":"2022-08-07T17:43:23.781568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#X_train['FareLog'] = np.log(X_train['Fare'] + 1)\n#X_test['FareLog'] = np.log(X_test['Fare'] + 1)\n#X_train.drop(['Fare'], axis=1, inplace=True)\n#X_test.drop(['Fare'], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:43:23.784286Z","iopub.execute_input":"2022-08-07T17:43:23.785405Z","iopub.status.idle":"2022-08-07T17:43:23.792069Z","shell.execute_reply.started":"2022-08-07T17:43:23.785356Z","shell.execute_reply":"2022-08-07T17:43:23.791024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test.loc[(X_test['Fare'].isnull())]","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:43:23.793620Z","iopub.execute_input":"2022-08-07T17:43:23.794885Z","iopub.status.idle":"2022-08-07T17:43:23.814942Z","shell.execute_reply.started":"2022-08-07T17:43:23.794844Z","shell.execute_reply":"2022-08-07T17:43:23.814320Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test.loc[(X_test['Fare'].isnull()), 'Fare'] = X_train.loc[(X_train['Pclass']==3)]['Fare'].median()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:43:23.816530Z","iopub.execute_input":"2022-08-07T17:43:23.817068Z","iopub.status.idle":"2022-08-07T17:43:23.825372Z","shell.execute_reply.started":"2022-08-07T17:43:23.817011Z","shell.execute_reply":"2022-08-07T17:43:23.824472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:43:23.827469Z","iopub.execute_input":"2022-08-07T17:43:23.828291Z","iopub.status.idle":"2022-08-07T17:43:23.840746Z","shell.execute_reply.started":"2022-08-07T17:43:23.827994Z","shell.execute_reply":"2022-08-07T17:43:23.839751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:43:23.842725Z","iopub.execute_input":"2022-08-07T17:43:23.843359Z","iopub.status.idle":"2022-08-07T17:43:23.855852Z","shell.execute_reply.started":"2022-08-07T17:43:23.843310Z","shell.execute_reply":"2022-08-07T17:43:23.854578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# some ideas from others' notebooks\n#X_train['FamilySize'] = X_train['SibSp'] + X_train['Parch']\n#X_train.drop(['SibSp', 'Parch'], axis=1, inplace=True)\n#X_train['AgePclass'] =  X_train['Age'] * X_train['Pclass']\n#X_test['AgePclass'] =  X_test['Age'] * X_test['Pclass']\n","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:43:23.858664Z","iopub.execute_input":"2022-08-07T17:43:23.859403Z","iopub.status.idle":"2022-08-07T17:43:23.864011Z","shell.execute_reply.started":"2022-08-07T17:43:23.859351Z","shell.execute_reply":"2022-08-07T17:43:23.863040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = pd.get_dummies(X_train)\nX_train","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:43:23.866084Z","iopub.execute_input":"2022-08-07T17:43:23.866693Z","iopub.status.idle":"2022-08-07T17:43:23.915622Z","shell.execute_reply.started":"2022-08-07T17:43:23.866647Z","shell.execute_reply":"2022-08-07T17:43:23.914775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test = pd.get_dummies(X_test)\nX_test","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:43:23.916869Z","iopub.execute_input":"2022-08-07T17:43:23.917802Z","iopub.status.idle":"2022-08-07T17:43:23.957642Z","shell.execute_reply.started":"2022-08-07T17:43:23.917757Z","shell.execute_reply":"2022-08-07T17:43:23.956797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Random Forest","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import KFold, cross_val_score\n\nfrom sklearn.ensemble import RandomForestClassifier\n\nbest_parameters = 0, 0\nbest_score = 0\n\nkf = KFold(n_splits=5, shuffle=True, random_state=42)\nfor k in range(100, 500, 100):\n    for d in range(5, 20, 5):\n        print(f'Number of trees = {k}, max depth = {d}')    \n        clf = RandomForestClassifier(n_estimators=k, max_depth=d, random_state=42)\n        score = round(cross_val_score(clf, X_train, y_train, cv = kf, scoring='accuracy').mean(), 3)\n        print(f'Accuracy = {score}')\n        if score > best_score:\n            best_score = score\n            best_parameters = k, d            \n","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:43:23.958914Z","iopub.execute_input":"2022-08-07T17:43:23.959829Z","iopub.status.idle":"2022-08-07T17:43:53.696077Z","shell.execute_reply.started":"2022-08-07T17:43:23.959783Z","shell.execute_reply":"2022-08-07T17:43:53.694533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_score, best_parameters","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:43:53.697308Z","iopub.execute_input":"2022-08-07T17:43:53.697540Z","iopub.status.idle":"2022-08-07T17:43:53.703948Z","shell.execute_reply.started":"2022-08-07T17:43:53.697510Z","shell.execute_reply":"2022-08-07T17:43:53.702963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf_rf = RandomForestClassifier(n_estimators=best_parameters[0], max_depth=best_parameters[1], random_state=42)\nclf_rf.fit(X_train, y_train)\npredictions_rf = clf_rf.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:43:53.709791Z","iopub.execute_input":"2022-08-07T17:43:53.710214Z","iopub.status.idle":"2022-08-07T17:43:54.172638Z","shell.execute_reply.started":"2022-08-07T17:43:53.710177Z","shell.execute_reply":"2022-08-07T17:43:54.171728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output_rf = pd.DataFrame({'PassengerId': df_test.index, 'Survived': predictions_rf})\noutput_rf.to_csv('submission_rf.csv', index=False)\nprint(\"Your submission was successfully saved!\")","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:43:54.173980Z","iopub.execute_input":"2022-08-07T17:43:54.174327Z","iopub.status.idle":"2022-08-07T17:43:54.184848Z","shell.execute_reply.started":"2022-08-07T17:43:54.174284Z","shell.execute_reply":"2022-08-07T17:43:54.183319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Logistic Regression","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\nscaler = StandardScaler()\nX_train_scaled = scaler.fit_transform(X_train)\nX_train_scaled","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:43:54.186527Z","iopub.execute_input":"2022-08-07T17:43:54.186900Z","iopub.status.idle":"2022-08-07T17:43:54.201391Z","shell.execute_reply.started":"2022-08-07T17:43:54.186864Z","shell.execute_reply":"2022-08-07T17:43:54.200390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test_scaled = scaler.transform(X_test)\nX_test_scaled","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:43:54.203142Z","iopub.execute_input":"2022-08-07T17:43:54.203415Z","iopub.status.idle":"2022-08-07T17:43:54.215088Z","shell.execute_reply.started":"2022-08-07T17:43:54.203385Z","shell.execute_reply":"2022-08-07T17:43:54.214080Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nbest_coeff = 0\nbest_score = 0\nfor reg_coeff in [0.1, 1, 5, 10, 20, 50, 100, 500, 1000]:\n    print(f'Regularization coefficient = {reg_coeff}')    \n    clf = LogisticRegression(penalty='l2', C=reg_coeff, random_state=42)\n    score = round(cross_val_score(clf, X_train_scaled, y_train, cv = kf, scoring='accuracy').mean(), 3)\n    print(f'Accuracy = {score}')\n    if score > best_score:\n        best_score, best_coeff = score, reg_coeff\n","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:43:54.216783Z","iopub.execute_input":"2022-08-07T17:43:54.217222Z","iopub.status.idle":"2022-08-07T17:43:55.506472Z","shell.execute_reply.started":"2022-08-07T17:43:54.217171Z","shell.execute_reply":"2022-08-07T17:43:55.505130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_score, best_coeff","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:43:55.508708Z","iopub.execute_input":"2022-08-07T17:43:55.509488Z","iopub.status.idle":"2022-08-07T17:43:55.518765Z","shell.execute_reply.started":"2022-08-07T17:43:55.509421Z","shell.execute_reply":"2022-08-07T17:43:55.517531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf_lr = LogisticRegression(penalty='l2', C=best_coeff, random_state=42)\nclf_lr.fit(X_train_scaled, y_train)\npredictions_lr = clf_lr.predict(X_test_scaled)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:43:55.521047Z","iopub.execute_input":"2022-08-07T17:43:55.521830Z","iopub.status.idle":"2022-08-07T17:43:55.558842Z","shell.execute_reply.started":"2022-08-07T17:43:55.521774Z","shell.execute_reply":"2022-08-07T17:43:55.557604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output_lr = pd.DataFrame({'PassengerId': df_test.index, 'Survived': predictions_lr})\noutput_lr.to_csv('submission_log.csv', index=False)\nprint(\"Your submission was successfully saved!\")","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:43:55.566347Z","iopub.execute_input":"2022-08-07T17:43:55.567093Z","iopub.status.idle":"2022-08-07T17:43:55.589969Z","shell.execute_reply.started":"2022-08-07T17:43:55.567033Z","shell.execute_reply":"2022-08-07T17:43:55.588720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"(Gives the best score on test set now, 0.77511)","metadata":{}},{"cell_type":"markdown","source":"Logistic regression requires features to be independent (https://towardsdatascience.com/assumptions-of-logistic-regression-clearly-explained-44d85a22b290#:~:text=Logistic%20regression%20does%20not%20require,but%20not%20for%20logistic%20regression), but can we say this about 'Sex' and 'Title' It seems we cannot: after get_dummies, we have X_train['Sex_female'] = X_train['Title_Miss'] + X_train['Title_Mrs'] (okay, there is also one female Dr.) and X_train['Sex_male'] is the sum of the columns corresponding to other titles. Title, in fact (with a single exception), gives us information about a passenger's sex. So, let's drop dependent columns and look at the result.","metadata":{}},{"cell_type":"code","source":"X_train_scaled_independent = scaler.fit_transform(X_train.drop(columns=['Sex_female', 'Sex_male']))\nX_train_scaled_independent","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:43:55.598272Z","iopub.execute_input":"2022-08-07T17:43:55.599157Z","iopub.status.idle":"2022-08-07T17:43:55.630721Z","shell.execute_reply.started":"2022-08-07T17:43:55.599091Z","shell.execute_reply":"2022-08-07T17:43:55.629529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test_scaled_independent = scaler.transform(X_test.drop(columns=['Sex_female', 'Sex_male']))\nX_test_scaled_independent","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:43:55.637780Z","iopub.execute_input":"2022-08-07T17:43:55.641502Z","iopub.status.idle":"2022-08-07T17:43:55.667130Z","shell.execute_reply.started":"2022-08-07T17:43:55.641406Z","shell.execute_reply":"2022-08-07T17:43:55.665905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nbest_coeff = 0\nbest_score = 0\nfor reg_coeff in [0.1, 1, 5, 10, 20, 50, 100, 500, 1000]:\n    print(f'Regularization coefficient = {reg_coeff}')    \n    clf = LogisticRegression(penalty='l2', C=reg_coeff, random_state=42)\n    score = round(cross_val_score(clf, X_train_scaled_independent, y_train, cv = kf, scoring='accuracy').mean(), 3)\n    print(f'Accuracy = {score}')\n    if score > best_score:\n        best_score, best_coeff = score, reg_coeff","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:43:55.674050Z","iopub.execute_input":"2022-08-07T17:43:55.676676Z","iopub.status.idle":"2022-08-07T17:43:56.590121Z","shell.execute_reply.started":"2022-08-07T17:43:55.674759Z","shell.execute_reply":"2022-08-07T17:43:56.588760Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_score, best_coeff","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:43:56.592442Z","iopub.execute_input":"2022-08-07T17:43:56.593901Z","iopub.status.idle":"2022-08-07T17:43:56.610380Z","shell.execute_reply.started":"2022-08-07T17:43:56.593821Z","shell.execute_reply":"2022-08-07T17:43:56.606728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf_lr2 = LogisticRegression(penalty='l2', C=best_coeff, random_state=42)\nclf_lr2.fit(X_train_scaled_independent, y_train)\npredictions_lr2 = clf_lr2.predict(X_test_scaled_independent)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:43:56.617812Z","iopub.execute_input":"2022-08-07T17:43:56.621671Z","iopub.status.idle":"2022-08-07T17:43:56.645435Z","shell.execute_reply.started":"2022-08-07T17:43:56.621581Z","shell.execute_reply":"2022-08-07T17:43:56.644153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output_lr2 = pd.DataFrame({'PassengerId': df_test.index, 'Survived': predictions_lr2})\noutput_lr2.to_csv('submission_log_ind.csv', index=False)\nprint(\"Your submission was successfully saved!\")","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:43:56.653164Z","iopub.execute_input":"2022-08-07T17:43:56.657204Z","iopub.status.idle":"2022-08-07T17:43:56.677897Z","shell.execute_reply.started":"2022-08-07T17:43:56.657093Z","shell.execute_reply":"2022-08-07T17:43:56.676695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Test score is 0.77272 -- even a bit worse than with dependent columns..","metadata":{}},{"cell_type":"markdown","source":"# KNN","metadata":{}},{"cell_type":"code","source":"from sklearn.neighbors import KNeighborsClassifier\n\nbest_score = 0\nbest_k = 0\nM = 50\nfor k in range(1, M, 2):\n    clf = KNeighborsClassifier(n_neighbors=k)\n    score = cross_val_score(clf, X_train_scaled, y_train, cv=kf, scoring='accuracy').mean()\n    print(f'{k}: Accuracy = {score}') \n    if score > best_score:\n        best_score = score\n        best_k = k","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:43:56.685212Z","iopub.execute_input":"2022-08-07T17:43:56.689113Z","iopub.status.idle":"2022-08-07T17:43:59.646134Z","shell.execute_reply.started":"2022-08-07T17:43:56.689024Z","shell.execute_reply":"2022-08-07T17:43:59.644319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_score, best_k","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:43:59.654798Z","iopub.execute_input":"2022-08-07T17:43:59.659064Z","iopub.status.idle":"2022-08-07T17:43:59.674873Z","shell.execute_reply.started":"2022-08-07T17:43:59.658953Z","shell.execute_reply":"2022-08-07T17:43:59.673646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf_knn = KNeighborsClassifier(n_neighbors=best_k)\nclf_knn.fit(X_train_scaled, y_train)\npredictions_knn = clf_knn.predict(X_test_scaled)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:43:59.680577Z","iopub.execute_input":"2022-08-07T17:43:59.685981Z","iopub.status.idle":"2022-08-07T17:43:59.750292Z","shell.execute_reply.started":"2022-08-07T17:43:59.685889Z","shell.execute_reply":"2022-08-07T17:43:59.749079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output_knn = pd.DataFrame({'PassengerId': df_test.index, 'Survived': predictions_knn})\noutput_knn.to_csv('submission_knn.csv', index=False)\nprint(\"Your submission was successfully saved!\")","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:43:59.757824Z","iopub.execute_input":"2022-08-07T17:43:59.762587Z","iopub.status.idle":"2022-08-07T17:43:59.783975Z","shell.execute_reply.started":"2022-08-07T17:43:59.762491Z","shell.execute_reply":"2022-08-07T17:43:59.782219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# XGB","metadata":{}},{"cell_type":"code","source":"from xgboost import XGBClassifier\n\nbest_score = 0\nbest_k = 0\n\nfor k in range(100, 1000, 200):\n    print(f'{k} models')\n    clf = XGBClassifier(n_estimators=k, learning_rate=0.05, n_jobs=4)\n    clf.fit(X_train, y_train)\n    score = cross_val_score(clf, X_train, y_train, cv=kf, scoring='accuracy').mean()\n    print(f'{k}: Accuracy = {score}')\n    if score > best_score:\n        best_score = score\n        best_k = k\n    ","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:43:59.792520Z","iopub.execute_input":"2022-08-07T17:43:59.796546Z","iopub.status.idle":"2022-08-07T17:44:42.870733Z","shell.execute_reply.started":"2022-08-07T17:43:59.796443Z","shell.execute_reply":"2022-08-07T17:44:42.869882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_score, best_k","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:44:42.874939Z","iopub.execute_input":"2022-08-07T17:44:42.877064Z","iopub.status.idle":"2022-08-07T17:44:42.885416Z","shell.execute_reply.started":"2022-08-07T17:44:42.877004Z","shell.execute_reply":"2022-08-07T17:44:42.884402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf_xgb = XGBClassifier(n_estimators=best_k, learning_rate=0.05, n_jobs=4)\nclf_xgb.fit(X_train, y_train)\npredictions_xgb = clf_xgb.predict(X_test)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:44:42.886882Z","iopub.execute_input":"2022-08-07T17:44:42.887275Z","iopub.status.idle":"2022-08-07T17:44:43.203804Z","shell.execute_reply.started":"2022-08-07T17:44:42.887210Z","shell.execute_reply":"2022-08-07T17:44:43.202891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output_xgb = pd.DataFrame({'PassengerId': df_test.index, 'Survived': predictions_xgb})\noutput_xgb.to_csv('submission_xgb.csv', index=False)\nprint(\"Your submission was successfully saved!\")","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:44:43.205678Z","iopub.execute_input":"2022-08-07T17:44:43.206328Z","iopub.status.idle":"2022-08-07T17:44:43.216566Z","shell.execute_reply.started":"2022-08-07T17:44:43.206279Z","shell.execute_reply":"2022-08-07T17:44:43.215660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Support Vectors","metadata":{}},{"cell_type":"code","source":"from sklearn.svm import SVC\n\nfor g in ['auto', 'scale']:\n    clf = SVC(gamma=g)\n    clf.fit(X_train, y_train)\n    score = cross_val_score(clf, X_train_scaled, y_train, cv=kf, scoring='accuracy').mean()\n    print(f'{g}: Accuracy = {score}')","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:44:43.218685Z","iopub.execute_input":"2022-08-07T17:44:43.219555Z","iopub.status.idle":"2022-08-07T17:44:43.523189Z","shell.execute_reply.started":"2022-08-07T17:44:43.219507Z","shell.execute_reply":"2022-08-07T17:44:43.522146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Almost no difference...","metadata":{}},{"cell_type":"code","source":"clf_svc = SVC(probability=True)\nclf_svc.fit(X_train_scaled, y_train)\npredictions_svc = clf_svc.predict(X_test_scaled)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:44:43.524730Z","iopub.execute_input":"2022-08-07T17:44:43.525008Z","iopub.status.idle":"2022-08-07T17:44:43.671480Z","shell.execute_reply.started":"2022-08-07T17:44:43.524976Z","shell.execute_reply":"2022-08-07T17:44:43.669698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output_svc = pd.DataFrame({'PassengerId': df_test.index, 'Survived': predictions_svc})\noutput_svc.to_csv('submission_svc.csv', index=False)\nprint(\"Your submission was successfully saved!\")","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:44:43.673081Z","iopub.execute_input":"2022-08-07T17:44:43.673381Z","iopub.status.idle":"2022-08-07T17:44:43.683486Z","shell.execute_reply.started":"2022-08-07T17:44:43.673347Z","shell.execute_reply":"2022-08-07T17:44:43.682149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"(Yeah, score of 0.78468 !)","metadata":{}},{"cell_type":"markdown","source":"# Voting Classifier\nCombine the results of all previously used classifiers (idea from https://www.kaggle.com/code/koustavghosh149/a-simple-approach-to-titanic-dataset-83-2-accuracy)","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import VotingClassifier\nvoting_clf =  VotingClassifier(estimators=\n                               [('knn', clf_knn), ('lr', clf_lr), ('svc', clf_svc)], \n                               voting='soft')","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:44:43.685577Z","iopub.execute_input":"2022-08-07T17:44:43.686513Z","iopub.status.idle":"2022-08-07T17:44:43.692786Z","shell.execute_reply.started":"2022-08-07T17:44:43.686461Z","shell.execute_reply":"2022-08-07T17:44:43.691808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"voting_clf.fit(X_train_scaled, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:44:43.694736Z","iopub.execute_input":"2022-08-07T17:44:43.695136Z","iopub.status.idle":"2022-08-07T17:44:43.955488Z","shell.execute_reply.started":"2022-08-07T17:44:43.695090Z","shell.execute_reply":"2022-08-07T17:44:43.954312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = voting_clf.predict(X_test_scaled)\noutput = pd.DataFrame({'PassengerId': df_test.index, 'Survived': predictions})\noutput.to_csv('submission_voting.csv', index=False)\nprint(\"Your submission was successfully saved!\")","metadata":{"execution":{"iopub.status.busy":"2022-08-07T17:44:43.958908Z","iopub.execute_input":"2022-08-07T17:44:43.959616Z","iopub.status.idle":"2022-08-07T17:44:44.028288Z","shell.execute_reply.started":"2022-08-07T17:44:43.959555Z","shell.execute_reply":"2022-08-07T17:44:44.026938Z"},"trusted":true},"execution_count":null,"outputs":[]}]}