{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Titanic Dataset\n\nMy goal is build a new version [^1] of predictive model that predicts which passengers survived the Titanic shipwreck; and try to use some \"advanced\" (for me :smile:) techniques such as cross validation, grid search and an ensemble algorithm like \"Random Forest Classifier\". Let's see what happen :smile:\n\n[^1]: See https://www.kaggle.com/code/francescopaolol/logisticregression-on-complete-titanic-dataset","metadata":{"_uuid":"b8af4a59-b629-44e4-9eb3-22496fd35657","_cell_guid":"c62392b2-c4c8-420e-a1dd-a2648c1c7ebb","trusted":true}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"62018a59-09c6-4753-a159-8ad6cbfd211c","_cell_guid":"393fb966-e7ea-4041-9078-cc7e47e0e68c","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-08-11T17:17:54.633406Z","iopub.execute_input":"2022-08-11T17:17:54.633877Z","iopub.status.idle":"2022-08-11T17:17:54.661831Z","shell.execute_reply.started":"2022-08-11T17:17:54.633786Z","shell.execute_reply":"2022-08-11T17:17:54.660999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def showPercentageNan(colName, dtFrame):\n    perc = (dtFrame[colName].isnull().sum() / dtFrame.shape[0])\n    print(f\"Percent of missing '{colName}' records is {round(perc * 100,3)} %\")\n    return\n\ndef makeDummies(df, colummnName):\n    dummy = pd.get_dummies(df[colummnName], prefix=f\"{colummnName}_\")\n    return pd.concat([df, dummy], axis=1).drop(colummnName, axis=1)","metadata":{"_uuid":"54f2a1a4-1672-4579-8714-2a28fa47a32f","_cell_guid":"a0a9917e-4eb2-4b65-9ef4-edc1232282b4","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-08-11T17:17:54.707537Z","iopub.execute_input":"2022-08-11T17:17:54.708464Z","iopub.status.idle":"2022-08-11T17:17:54.715598Z","shell.execute_reply.started":"2022-08-11T17:17:54.708420Z","shell.execute_reply":"2022-08-11T17:17:54.714221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from IPython.display import display, Markdown\n\ndef basicEDA(df, title):\n    display(Markdown(\"**Just first five rows**\"))\n    display(df.head())\n    display(f\"The {title} data set consists of {df.shape[1]} different features which for {df.shape[0]} samples.\")\n    display(Markdown(\"**Info about the index dtype and columns, non-null values and memory usage.**\"))\n    display(df.info())\n    display(Markdown(\"**Check all of the values in the data frame which holds my data.**\"))\n    display(df.applymap(type))\n\n    # isnull() is just an alias of the isna() method as shown in pandas source code.\n    nrNa = df.isna().sum()\n    display(Markdown(\"**Count na values**\"))\n    display(nrNa)\n\n    if nrNa.any() > 0:\n        colNan = list(df[df.columns[df.isna().any()]].columns)\n        display(f\"The columns with missing data are: {colNan}\")\n\n        for col in colNan:\n            showPercentageNan(col, df)\n\n    display(Markdown(\"**Show the statistic report of the numeric features of the dataset**\"))\n    display(df.describe().transpose())\n\n    display(Markdown(\"**Show the statistic report of the categorical features of the dataset**\"))\n    display(df.describe(include=['O']).transpose())\n    return","metadata":{"_uuid":"be3f9e89-1f47-4970-8f35-5fa9ddbf68d1","_cell_guid":"8c93b858-c65d-4bd8-85da-72b7348d5583","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-08-11T17:17:54.777862Z","iopub.execute_input":"2022-08-11T17:17:54.778598Z","iopub.status.idle":"2022-08-11T17:17:54.788328Z","shell.execute_reply.started":"2022-08-11T17:17:54.778554Z","shell.execute_reply":"2022-08-11T17:17:54.787015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def delColumn(df, name):\n    if name in df.columns:\n        df.drop(name, axis=1, inplace=True)\n    return","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:17:54.834844Z","iopub.execute_input":"2022-08-11T17:17:54.835729Z","iopub.status.idle":"2022-08-11T17:17:54.840564Z","shell.execute_reply.started":"2022-08-11T17:17:54.835692Z","shell.execute_reply":"2022-08-11T17:17:54.839595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#  Import data set","metadata":{"_uuid":"61953d80-19f5-407a-8fa4-71c03408aef7","_cell_guid":"885e8290-8019-41f5-bcb9-1cd388ea5d59","trusted":true}},{"cell_type":"code","source":"# We'll use a dataset taken from: https://www.kaggle.com/competitions/titanic\ndfTrain = pd.read_csv(\"../input/titanic/train.csv\", sep=',')\ndfTest = pd.read_csv(\"../input/titanic/test.csv\", sep=\",\")","metadata":{"_uuid":"27ad7b71-f0d2-4645-b8d1-e5d1718963de","_cell_guid":"87069b6a-3e3c-4c35-b7ea-78eb4c2856ea","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-08-11T17:17:54.891710Z","iopub.execute_input":"2022-08-11T17:17:54.892511Z","iopub.status.idle":"2022-08-11T17:17:54.927885Z","shell.execute_reply.started":"2022-08-11T17:17:54.892468Z","shell.execute_reply":"2022-08-11T17:17:54.926470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Basic EDA and cleaning data","metadata":{"_uuid":"96667b5b-24ed-4e0f-901c-b1c7cce42d96","_cell_guid":"1acb39fa-1e66-4a91-bdc9-ae56dc6e5b16","trusted":true}},{"cell_type":"code","source":"basicEDA(dfTrain, \"Titanic Train\")","metadata":{"_uuid":"95498277-36aa-4d98-9b04-324ce377f298","_cell_guid":"b0f7011d-6dc2-47c0-8fb6-e69fa7df2140","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-08-11T17:17:54.961737Z","iopub.execute_input":"2022-08-11T17:17:54.962805Z","iopub.status.idle":"2022-08-11T17:17:55.101229Z","shell.execute_reply.started":"2022-08-11T17:17:54.962759Z","shell.execute_reply":"2022-08-11T17:17:55.100153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"basicEDA(dfTest, \"Titanic Test\")","metadata":{"_uuid":"d3d1c9e0-8492-47db-bbdd-b992a4c05fc8","_cell_guid":"55a98f07-0dcb-4c7b-9998-aaac4078fae8","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-08-11T17:17:55.103705Z","iopub.execute_input":"2022-08-11T17:17:55.104190Z","iopub.status.idle":"2022-08-11T17:17:55.223507Z","shell.execute_reply.started":"2022-08-11T17:17:55.104145Z","shell.execute_reply":"2022-08-11T17:17:55.222331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Some considerations: first of all, we can see how \"Survived\" is our target:\n- Survived 0 = no, 1 = yes\n\nAs regards the other features we have:\n- PassengerID: \n- Pclass: 1 = 1st, 2 = 2nd, 3 = 3rd\n- Name: self explanatory\n- Sex: self explanatory\n- Age: self explanatory\n- SibSp = nr of sibilings / spouses abroad\n- Parch = nr of parents / children abroad\n- Ticket = self explanatory\n- Fare = passenger fare\n- Cabin = self explanatory\n- Embarked = port of embrarkation --> C = Chernourg, Q = Queenstown, S = Southhampton\n\nI think I can do something in order to slim down this dataset.","metadata":{"_uuid":"36f6ad02-ba2e-490e-8932-ce99065e0f43","_cell_guid":"1e014931-ee38-4126-b2be-39c3b5e37af6","trusted":true}},{"cell_type":"markdown","source":"We can see that the two dataset are similar.","metadata":{"_uuid":"adec02fa-5f60-429f-8c15-53b813b9947b","_cell_guid":"68859296-27b2-4619-a4dd-9a6910889d08","trusted":true}},{"cell_type":"markdown","source":"## Feature engineering","metadata":{"_uuid":"88429990-dbf6-4b8f-a6a5-c55fa93d728c","_cell_guid":"4eadd4b9-751e-4e6f-8c82-a44bbcaaf114","trusted":true}},{"cell_type":"markdown","source":"So, I think that 'Name', 'Embarked', 'Cabin' and 'Ticket' features can be dropped because I don't believe that be called \"Nicholas\" or \"Augusta\" increases the possibility to survive. Same reasoning for the others features.\nThen, let's start to drop useless features.","metadata":{"_uuid":"8bb7b3af-144d-45ab-ad1c-a3866e86a8aa","_cell_guid":"45bb53ca-4f4a-46bb-a4e8-c1c86087b3a3","trusted":true}},{"cell_type":"code","source":"delColumn(dfTrain, \"Name\")\ndelColumn(dfTrain, \"Ticket\")\ndelColumn(dfTrain, \"Cabin\")\ndelColumn(dfTrain, \"Embarked\")","metadata":{"_uuid":"3a449015-29ce-4558-830b-1f816ab2d20b","_cell_guid":"72caefc2-14cd-40d4-a6af-50f88d6e92ab","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-08-11T17:17:55.224891Z","iopub.execute_input":"2022-08-11T17:17:55.225248Z","iopub.status.idle":"2022-08-11T17:17:55.234994Z","shell.execute_reply.started":"2022-08-11T17:17:55.225215Z","shell.execute_reply":"2022-08-11T17:17:55.233664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As regards \"Fare\", let's check out if the fare is related to the better chances to be survive.","metadata":{"_uuid":"edce68a8-5940-4a01-b6ee-50f3818afafb","_cell_guid":"2a99a3f2-a237-431f-9b26-6630b72c7b6e","trusted":true}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\nif \"Fare\" in dfTrain.columns:\n    dfTmp = dfTrain[[\"Fare\", \"Survived\"]]\n    dfSurvivedYes = dfTmp[dfTmp.Survived == 1]\n    dfSurvivedNo = dfTmp[dfTmp.Survived == 0]\n\n    plt.figure(figsize = (15,8))\n    plt.title(\"Fare related to stay alive.\")\n    sns.histplot(data = dfSurvivedYes[dfSurvivedYes.Fare > 0],\n                 x = 'Fare',\n                 color = 'navy'\n                )\n\n    plt.figure(figsize = (15,8))\n    plt.title(\"Fare related to not survive.\")\n    sns.histplot(data = dfSurvivedNo[dfSurvivedNo.Fare > 0],\n                 x = 'Fare',\n                 color = 'navy'\n                )\n\n\n\n    plt.show()","metadata":{"_uuid":"df50ca60-d321-4de5-b9c5-a49d779ce989","_cell_guid":"e9ededfd-9ba4-4b1f-aa1e-1d98133d4c6c","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-08-11T17:17:55.237125Z","iopub.execute_input":"2022-08-11T17:17:55.237824Z","iopub.status.idle":"2022-08-11T17:17:56.402421Z","shell.execute_reply.started":"2022-08-11T17:17:55.237788Z","shell.execute_reply":"2022-08-11T17:17:56.401613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"So one can be dead or alive, no matter how much he paid as fare: we can also drop this feature.","metadata":{"_uuid":"ab6f1992-7c12-4521-be10-a1db7ea25a75","_cell_guid":"a8d76786-29dc-4871-a449-9da520039e6c","trusted":true}},{"cell_type":"code","source":"delColumn(dfTrain, \"Fare\")","metadata":{"_uuid":"bb4a9419-fbed-4a25-b4a0-bf9297f49789","_cell_guid":"d1edb32c-aa65-44cf-aa3f-5d8685d4716d","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-08-11T17:17:56.403499Z","iopub.execute_input":"2022-08-11T17:17:56.404268Z","iopub.status.idle":"2022-08-11T17:17:56.409102Z","shell.execute_reply.started":"2022-08-11T17:17:56.404234Z","shell.execute_reply":"2022-08-11T17:17:56.408126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"At this point we have few but good features. Remains to resolve the missing 'Age' records (that is 19.865 %).\nSo, let's find out correlations with \"Age\" feature.","metadata":{"_uuid":"42de6138-3781-4267-9bfe-66deb1e65168","_cell_guid":"e6cd906b-0b76-4fad-b402-98b035677bf5","trusted":true}},{"cell_type":"code","source":"dfTrain.corr()","metadata":{"_uuid":"6a1017ee-03ef-4039-a78f-2726b9451e5c","_cell_guid":"717db10f-202a-4219-920f-8b56f73e1c70","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-08-11T17:17:56.411392Z","iopub.execute_input":"2022-08-11T17:17:56.411778Z","iopub.status.idle":"2022-08-11T17:17:56.435300Z","shell.execute_reply.started":"2022-08-11T17:17:56.411747Z","shell.execute_reply":"2022-08-11T17:17:56.434302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We have three feature correlate with \"Age\": Pclass (PCC: -0.369226), SibSp (PCC: -0.308247), Parch (PCC: -0.189119).\nI'm going to use \"IterativeImputer\" (which is a multivariate imputer that estimates each feature from all the others) with RandomForestRegressor...","metadata":{"_uuid":"df3b903a-c9cc-4ada-8164-d513f49f859f","_cell_guid":"daa886b9-a917-45d2-b853-6c991ae2e16e","trusted":true}},{"cell_type":"code","source":"from sklearn.experimental import enable_iterative_imputer  \nfrom sklearn.impute import IterativeImputer\n\nfrom sklearn.ensemble import RandomForestRegressor\nimport pandas as pd\n\ndftmp = dfTrain.loc[:, [\"Age\"]]\n\nimp = IterativeImputer(RandomForestRegressor(), \n                       max_iter=10, \n                       tol=0.001, \n                       random_state=0, \n                       sample_posterior=False, \n                       verbose=True)\ndftmp = pd.DataFrame(imp.fit_transform(dftmp), \n                     columns=dftmp.columns)","metadata":{"_uuid":"88d8a614-f1d8-4dc7-a5cc-bfb62088f0fe","_cell_guid":"c3584e65-2272-414e-a7f4-702f4857ec75","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-08-11T17:17:56.436851Z","iopub.execute_input":"2022-08-11T17:17:56.437245Z","iopub.status.idle":"2022-08-11T17:17:56.764500Z","shell.execute_reply.started":"2022-08-11T17:17:56.437204Z","shell.execute_reply":"2022-08-11T17:17:56.763583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"...check if all data are property filled...","metadata":{"_uuid":"403ab6c6-4e58-4402-9a92-4499a2fe5cba","_cell_guid":"27349567-0842-4a07-a604-e85b07041ef0","trusted":true}},{"cell_type":"code","source":"print(\"\\nNumber of rows where 'Age' are null or empty\")\nprint(dftmp.isnull().sum())","metadata":{"_uuid":"79a9867c-bfea-4b8d-a640-4dd3c4c3de5f","_cell_guid":"91d6df99-573f-469e-8073-cd5b3cb0718c","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-08-11T17:17:56.765929Z","iopub.execute_input":"2022-08-11T17:17:56.766957Z","iopub.status.idle":"2022-08-11T17:17:56.774211Z","shell.execute_reply.started":"2022-08-11T17:17:56.766920Z","shell.execute_reply":"2022-08-11T17:17:56.772947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"...and finally refill the missing Age values.","metadata":{"_uuid":"106c9cb1-3906-4656-abe3-e491d67e6810","_cell_guid":"425aad6a-b354-4a39-b86a-468805117648","trusted":true}},{"cell_type":"code","source":"delColumn(dfTrain, \"Age\")\ndfTrain = dfTrain.join(dftmp)","metadata":{"_uuid":"d2e0adae-91da-403d-bd4f-ece3c6053b6b","_cell_guid":"b4f1d0f7-e40d-4cac-a7db-951d60f09867","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-08-11T17:17:56.775812Z","iopub.execute_input":"2022-08-11T17:17:56.776577Z","iopub.status.idle":"2022-08-11T17:17:56.793381Z","shell.execute_reply.started":"2022-08-11T17:17:56.776542Z","shell.execute_reply":"2022-08-11T17:17:56.792318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Remains to encode the \"Sex\" feature.","metadata":{"_uuid":"3eae6436-de2a-4df3-b9e1-8e02c1ee472c","_cell_guid":"7e597f11-55c8-47ca-b2e2-bb2e213f2b0b","trusted":true}},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\n\nlabelencoder_X = LabelEncoder()\n\ndfTrain[\"Sex\"] = labelencoder_X.fit_transform(dfTrain[\"Sex\"])","metadata":{"_uuid":"30773899-e8f5-4ac5-a763-5f82b868e8c3","_cell_guid":"242f9fff-0a7e-42cf-af89-3e948bec5788","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-08-11T17:17:56.794996Z","iopub.execute_input":"2022-08-11T17:17:56.795370Z","iopub.status.idle":"2022-08-11T17:17:56.801716Z","shell.execute_reply.started":"2022-08-11T17:17:56.795337Z","shell.execute_reply":"2022-08-11T17:17:56.800616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"So this is our starting dataset.","metadata":{"_uuid":"6451c0bf-ee8c-4611-a2d2-d0e31721c686","_cell_guid":"1735d9b5-4692-4848-9254-baa000e235f0","trusted":true}},{"cell_type":"code","source":"dfTrain.head()","metadata":{"_uuid":"fd2ae835-4b5d-4712-ac5b-51a28a4a4013","_cell_guid":"e5336916-f28b-4c99-90b4-e1301abbdc18","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-08-11T17:17:56.805662Z","iopub.execute_input":"2022-08-11T17:17:56.806395Z","iopub.status.idle":"2022-08-11T17:17:56.822791Z","shell.execute_reply.started":"2022-08-11T17:17:56.806345Z","shell.execute_reply":"2022-08-11T17:17:56.821541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Same things to test dataset","metadata":{}},{"cell_type":"code","source":"delColumn(dfTest, \"Name\")\ndelColumn(dfTest, \"Ticket\")\ndelColumn(dfTest, \"Cabin\")\ndelColumn(dfTest, \"Embarked\")\ndelColumn(dfTest, \"Fare\")\n\ndftmp = dfTest.loc[:, [\"Age\"]]\n\nimp = IterativeImputer(RandomForestRegressor(), \n                       max_iter=10, \n                       tol=0.001, \n                       random_state=0, \n                       sample_posterior=False, \n                       verbose=True)\ndftmp = pd.DataFrame(imp.fit_transform(dftmp), \n                     columns=dftmp.columns)\ndfTest.drop(\"Age\", axis=1, inplace=True)\ndfTest = dfTest.join(dftmp)\n\ndfTest[\"Sex\"] = labelencoder_X.fit_transform(dfTest[\"Sex\"])\n\ndfTest.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:17:56.824178Z","iopub.execute_input":"2022-08-11T17:17:56.825273Z","iopub.status.idle":"2022-08-11T17:17:56.854054Z","shell.execute_reply.started":"2022-08-11T17:17:56.825227Z","shell.execute_reply":"2022-08-11T17:17:56.852775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train and test the model","metadata":{"_uuid":"be1d2982-45db-4f16-a610-a205587bf2b9","_cell_guid":"2a257331-7613-4ffc-87cb-4c4f4af936d6","trusted":true}},{"cell_type":"markdown","source":"Once prepared data, we can split data in train and test, as usual.","metadata":{"_uuid":"7738daa1-1fda-46e3-bbc1-0659f5ac6c89","_cell_guid":"b789e385-eab2-461d-9c36-c7363889faff","trusted":true}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX = dfTrain.drop(\"Survived\", axis=1)\ny = dfTrain[\"Survived\"]\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.4)","metadata":{"_uuid":"b79b647d-be73-4e19-926a-f53cdce97528","_cell_guid":"1a09e0db-e79c-46ca-9fc6-e1f3ff73aa1d","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-08-11T17:17:56.858062Z","iopub.execute_input":"2022-08-11T17:17:56.858431Z","iopub.status.idle":"2022-08-11T17:17:56.867861Z","shell.execute_reply.started":"2022-08-11T17:17:56.858391Z","shell.execute_reply":"2022-08-11T17:17:56.866804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"And prepare the grid search with RandomForestClassifier model, and return","metadata":{"_uuid":"df94be86-8d9c-412b-a48c-c097fb44225d","_cell_guid":"ca6c39e2-3cf2-487f-a5f2-fee1165e3b63","trusted":true}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\nfrom sklearn.model_selection import GridSearchCV, KFold\n\nrndForestParams = { \n    \"criterion\" : [\"gini\", \"entropy\"],                  # {“gini”, “entropy”, “log_loss”},\n    \"min_samples_leaf\" : [1, 5, 10],                    # The minimum number of samples required to be at a leaf node. \n    \"min_samples_split\" : [4, 10, 14, 16],              # The minimum number of samples required to split an internal node\n    \"n_estimators\": [150, 300, 700, 1000]               # The number of trees in the forest.\n}\n\nrfModel = RandomForestClassifier(\n    max_features = \"sqrt\",                              # The number of features to consider when looking for the best split\n    oob_score = True,                                   # Whether to use out-of-bag samples to estimate the generalization score. \n                                                        # Only available if 'bootstrap = True' (that's default value!)\n    random_state = 1,                                   # Controls both the randomness of the bootstrapping of the samples used when building trees \n    n_jobs = -1                                         # '-1' means using all processors.\n)                                        \n\ncv_method = KFold(n_splits = 10, shuffle = True)\n\n\ngs = GridSearchCV(\n    estimator = rfModel, \n    param_grid = rndForestParams, \n    scoring='accuracy', \n    cv = cv_method, \n    n_jobs=-1\n)","metadata":{"_uuid":"6eddaf70-3370-4817-9fae-226c7322f025","_cell_guid":"86dfcc0a-c438-4b9d-8fc7-341ca58e0b3d","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-08-11T17:17:56.868935Z","iopub.execute_input":"2022-08-11T17:17:56.869971Z","iopub.status.idle":"2022-08-11T17:17:56.878223Z","shell.execute_reply.started":"2022-08-11T17:17:56.869935Z","shell.execute_reply":"2022-08-11T17:17:56.877168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now we can fit the gridsearch object (it will take a while...).","metadata":{"_uuid":"43541124-e8c1-4660-ba92-65ce2f7d9bb9","_cell_guid":"4442c1c5-9730-40c4-aa49-aabb5d85961f","trusted":true}},{"cell_type":"code","source":"gs.fit(X_train, y_train)","metadata":{"_uuid":"cbd9875a-50f0-4342-9329-d468e2e930f2","_cell_guid":"103ee163-c284-4d82-a664-392c4d9a82ba","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-08-11T17:17:56.879758Z","iopub.execute_input":"2022-08-11T17:17:56.880530Z","iopub.status.idle":"2022-08-11T17:26:33.041481Z","shell.execute_reply.started":"2022-08-11T17:17:56.880495Z","shell.execute_reply":"2022-08-11T17:26:33.040257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can see the parameter setting that gave the best results on the hold out data...","metadata":{"_uuid":"4c2a3a47-06c7-4aba-b908-454fae37a129","_cell_guid":"48a7dcf6-6578-46cf-a338-98742f27965d","trusted":true}},{"cell_type":"code","source":"gs.best_params_","metadata":{"_uuid":"f0eaa027-6f34-40cb-b753-197a0f775caf","_cell_guid":"0cb1ec51-6293-49e2-a90f-6e932a3b9f15","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-08-11T17:26:33.044110Z","iopub.execute_input":"2022-08-11T17:26:33.044627Z","iopub.status.idle":"2022-08-11T17:26:33.051774Z","shell.execute_reply.started":"2022-08-11T17:26:33.044572Z","shell.execute_reply":"2022-08-11T17:26:33.050714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"...and set up a model with the estimator that was chosen by the search.","metadata":{"_uuid":"76edcc11-6901-4184-b709-055fe43c201f","_cell_guid":"da022e19-1885-4efd-86da-82e59d4bfcab","trusted":true}},{"cell_type":"code","source":"RFC_Model = gs.best_estimator_","metadata":{"_uuid":"42c5d138-fffd-4a6e-8f3d-8993b9ba2f1e","_cell_guid":"d8f21a27-6e52-4aba-a053-596692e91e59","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-08-11T17:26:33.053561Z","iopub.execute_input":"2022-08-11T17:26:33.053911Z","iopub.status.idle":"2022-08-11T17:26:33.061894Z","shell.execute_reply.started":"2022-08-11T17:26:33.053880Z","shell.execute_reply":"2022-08-11T17:26:33.060742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"And show what is the average of all cv folds for a single combination of the parameters you specify in the tuned_params.","metadata":{"_uuid":"931b5b33-d60a-4a16-bc46-0fd9622b2edc","_cell_guid":"33b5fb2f-9a27-4d52-890f-3560431c01b5","trusted":true}},{"cell_type":"code","source":"gs.best_score_                                  #Mean cross-validated score of the best_estimator","metadata":{"_uuid":"467d5850-cb2a-47dd-a238-6338ee794226","_cell_guid":"ae99816f-41ba-4c21-a969-101b267c9dd0","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-08-11T17:26:33.063912Z","iopub.execute_input":"2022-08-11T17:26:33.064455Z","iopub.status.idle":"2022-08-11T17:26:33.076352Z","shell.execute_reply.started":"2022-08-11T17:26:33.064421Z","shell.execute_reply":"2022-08-11T17:26:33.074593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's predict on train data.","metadata":{"_uuid":"64409213-a949-4d0e-a630-2d8b322eeb97","_cell_guid":"c28445be-5473-4533-bd3e-165678a0e8d7","trusted":true}},{"cell_type":"code","source":"RFC_Model.predict(X_train)","metadata":{"_uuid":"bb454724-3bec-47f8-946c-8175ea0e6f7e","_cell_guid":"a7927b76-6f8e-4059-859b-2063440f268b","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-08-11T17:26:33.077720Z","iopub.execute_input":"2022-08-11T17:26:33.078564Z","iopub.status.idle":"2022-08-11T17:26:33.292326Z","shell.execute_reply.started":"2022-08-11T17:26:33.078527Z","shell.execute_reply":"2022-08-11T17:26:33.291536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"And make the prediction on test data.","metadata":{"_uuid":"1320d7df-8116-4a92-b8f5-7a5dffe900de","_cell_guid":"70e2433b-cf94-4922-a755-9fd31adbe5f1","trusted":true}},{"cell_type":"code","source":"RFC_Model.score(X_test, y_test)                 #Return the mean accuracy on the given test data and labels","metadata":{"_uuid":"1cbb9c94-b7c7-4a05-8bfc-94b02b366c0b","_cell_guid":"7a76f0af-2cb6-4988-8bad-541783de5c4a","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-08-11T17:26:33.293509Z","iopub.execute_input":"2022-08-11T17:26:33.293997Z","iopub.status.idle":"2022-08-11T17:26:33.404465Z","shell.execute_reply.started":"2022-08-11T17:26:33.293968Z","shell.execute_reply":"2022-08-11T17:26:33.403439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model Performance Analysis","metadata":{"_uuid":"3cce88f8-df7f-48e9-ab70-533906226e4c","_cell_guid":"6b44f96a-d712-4b7b-a8ed-4e39dcc23677","trusted":true}},{"cell_type":"markdown","source":"Displaying the learning curve, we note that we have enough data to try to make a model.","metadata":{"_uuid":"d283fc0d-e249-4ff7-a5d6-12769b9eb558","_cell_guid":"28bce994-8c9f-42d0-8946-3492f61d336a","trusted":true}},{"cell_type":"code","source":"from yellowbrick.model_selection import learning_curve\n\nlearning_curve(RFC_Model, X_test, y_test, scoring='accuracy')\nplt.show()","metadata":{"_uuid":"b27f296d-106c-4926-b340-ac3c7bf4fdd2","_cell_guid":"1f28a505-7980-48c8-b41f-ab75daf498db","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-08-11T17:26:33.405991Z","iopub.execute_input":"2022-08-11T17:26:33.406326Z","iopub.status.idle":"2022-08-11T17:26:42.106569Z","shell.execute_reply.started":"2022-08-11T17:26:33.406298Z","shell.execute_reply":"2022-08-11T17:26:42.105438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import classification_report\n\ndfReport = pd.DataFrame(classification_report(y_test, RFC_Model.predict(X_test), output_dict=True))\ndfReport","metadata":{"_uuid":"dd5e4561-67fe-4c65-a7c0-7400a62c14fc","_cell_guid":"24d9926e-4abf-4ed8-a69d-a57c4fd25d52","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-08-11T17:26:42.108657Z","iopub.execute_input":"2022-08-11T17:26:42.109135Z","iopub.status.idle":"2022-08-11T17:26:42.233813Z","shell.execute_reply.started":"2022-08-11T17:26:42.109087Z","shell.execute_reply":"2022-08-11T17:26:42.232388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix, ConfusionMatrixDisplay\n\npredictions = RFC_Model.predict(X_test)\ncm = confusion_matrix(y_test, predictions, labels = RFC_Model.classes_)\ndisp = ConfusionMatrixDisplay(confusion_matrix = cm,\n                              display_labels = RFC_Model.classes_)\ndisp.plot()\nplt.grid(visible=None)\nplt.show()","metadata":{"_uuid":"ad151247-878a-4ab9-b050-9b7fad1953d1","_cell_guid":"6b9a4d47-45cb-4d8d-86f2-5db58dd753f7","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-08-11T17:26:42.235362Z","iopub.execute_input":"2022-08-11T17:26:42.236353Z","iopub.status.idle":"2022-08-11T17:26:42.683623Z","shell.execute_reply.started":"2022-08-11T17:26:42.236315Z","shell.execute_reply":"2022-08-11T17:26:42.682460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{"_uuid":"205953f1-7612-4fd8-b641-8a54973d255d","_cell_guid":"66b72eb3-48e4-420c-a0fd-2576f908b189","trusted":true}},{"cell_type":"code","source":"predictions = RFC_Model.predict(dfTest)\npredictions","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:26:42.684834Z","iopub.execute_input":"2022-08-11T17:26:42.685928Z","iopub.status.idle":"2022-08-11T17:26:42.799562Z","shell.execute_reply.started":"2022-08-11T17:26:42.685885Z","shell.execute_reply":"2022-08-11T17:26:42.798327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PassengerId = dfTest['PassengerId']\nsubmission = pd.DataFrame({\"PassengerId\": PassengerId,\"Survived\": predictions})\nsubmission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:26:42.800817Z","iopub.execute_input":"2022-08-11T17:26:42.801214Z","iopub.status.idle":"2022-08-11T17:26:42.811099Z","shell.execute_reply.started":"2022-08-11T17:26:42.801181Z","shell.execute_reply":"2022-08-11T17:26:42.810040Z"},"trusted":true},"execution_count":null,"outputs":[]}]}