{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Hello! #\n\n### This notebook aims to examine the effects of an inbalanced dataset, and possible methods of effective sampling. ###","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom imblearn.over_sampling import SMOTE \nfrom sklearn.model_selection import train_test_split,cross_val_score\nfrom xgboost import XGBClassifier\nimport matplotlib.pyplot as plt\n\n# dicts for scores\ncv_score = {}\ntest_score = {}","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-13T23:57:44.946021Z","iopub.execute_input":"2022-08-13T23:57:44.946997Z","iopub.status.idle":"2022-08-13T23:57:44.953330Z","shell.execute_reply.started":"2022-08-13T23:57:44.946953Z","shell.execute_reply":"2022-08-13T23:57:44.952145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Baseline #","metadata":{}},{"cell_type":"code","source":"def load_data():\n    testing = pd.read_csv('../input/tabular-playground-series-aug-2022/test.csv')\n    training = pd.read_csv('/kaggle/input/tabular-playground-series-aug-2022/train.csv')\n\n    training.drop('id', inplace=True, axis=1)\n    testing.drop('id', inplace=True, axis=1)\n    \n    # might as well replace NaN values with 0.\n    training.fillna(0, inplace=True)\n    testing.fillna(0, inplace=True)\n    \n    # not important, just so the model doesnt throw errors due to dtypes\n    training = pd.get_dummies(training)\n    testing = pd.get_dummies(testing)\n    \n    return training, testing","metadata":{"execution":{"iopub.status.busy":"2022-08-13T23:57:45.040123Z","iopub.execute_input":"2022-08-13T23:57:45.041297Z","iopub.status.idle":"2022-08-13T23:57:45.054118Z","shell.execute_reply.started":"2022-08-13T23:57:45.041248Z","shell.execute_reply":"2022-08-13T23:57:45.052506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training, testing = load_data()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T23:57:45.162143Z","iopub.execute_input":"2022-08-13T23:57:45.162855Z","iopub.status.idle":"2022-08-13T23:57:45.398298Z","shell.execute_reply.started":"2022-08-13T23:57:45.162814Z","shell.execute_reply":"2022-08-13T23:57:45.396938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's plot the inbalance of the target variable.","metadata":{}},{"cell_type":"code","source":"training['failure'].value_counts().plot.pie(autopct='%.2f')\nprint(f'There are {training.shape[0]} rows.')","metadata":{"execution":{"iopub.status.busy":"2022-08-13T23:57:45.400493Z","iopub.execute_input":"2022-08-13T23:57:45.400981Z","iopub.status.idle":"2022-08-13T23:57:45.514346Z","shell.execute_reply.started":"2022-08-13T23:57:45.400926Z","shell.execute_reply":"2022-08-13T23:57:45.511945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"First lets get a baseline. We shall split the training data into features and target variables, and then into train and test sets.","metadata":{}},{"cell_type":"code","source":"Y = training['failure']\nX = training.drop(['failure'], axis=1)\nx_train,x_test,y_train,y_test = train_test_split(X, Y,test_size=0.15)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T23:57:45.517258Z","iopub.execute_input":"2022-08-13T23:57:45.517958Z","iopub.status.idle":"2022-08-13T23:57:45.557867Z","shell.execute_reply.started":"2022-08-13T23:57:45.517850Z","shell.execute_reply":"2022-08-13T23:57:45.556087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The following code block creates an XGBoost model, fits it to the data and then cross validates. The model is then scored on testing data. The cross validation scores and the testing scores for each method has been examined at the end of the notebook.","metadata":{}},{"cell_type":"code","source":"model = XGBClassifier()\nmodel.fit(x_train, y_train)\ncv_score['baseline'] = cross_val_score(model, X, Y, cv=10)\ntest_score['baseline'] = model.score(x_test, y_test)\nprint(cv_score['baseline'])\nprint(test_score['baseline'])","metadata":{"execution":{"iopub.status.busy":"2022-08-13T23:57:45.561039Z","iopub.execute_input":"2022-08-13T23:57:45.562500Z","iopub.status.idle":"2022-08-13T23:58:29.247192Z","shell.execute_reply.started":"2022-08-13T23:57:45.562434Z","shell.execute_reply":"2022-08-13T23:58:29.246248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Over-sampling with replacement #","metadata":{}},{"cell_type":"markdown","source":"Now let's oversample with replacement of the minority class.\nRows from the minority class are sampled until the classes are balanced. This leads to duplicate examples, and possible overfitting issues.","metadata":{}},{"cell_type":"code","source":"# re-import the data\ntraining, testing = load_data()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T23:58:29.251691Z","iopub.execute_input":"2022-08-13T23:58:29.254234Z","iopub.status.idle":"2022-08-13T23:58:29.487525Z","shell.execute_reply.started":"2022-08-13T23:58:29.254188Z","shell.execute_reply":"2022-08-13T23:58:29.486050Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# taken from my other notebook found here: https://www.kaggle.com/code/cameron858/tps-aug22\n\ndef sample(df, target_column: str, sample_method: str) -> pd.DataFrame:\n    \"\"\"Over / Under samping implementation for TPS-Aug22.\n    The majority / minority classes are hard coded in according to the competition.\n    \"\"\"\n    assert sample_method in ['over', 'under'], 'Samping method should be either \"over\" or \"under\"'\n    \n    if sample_method == 'over':\n        majority_count = df[target_column].value_counts().sort_values()[0]\n        minority = df[df[target_column] == 1].sample(n=majority_count, replace=True)    # sample the minority to the same amount as the majority class\n        majority = df[df[target_column] == 0]\n    elif sample_method == 'under':\n        minority_count = df[target_column].value_counts().sort_values()[1]\n        minority = df[df[target_column] == 1]\n        majority = df[df[target_column] == 0].sample(n=minority_count)\n    \n    return majority.append(minority, ignore_index=True)\n\ntraining = sample(df=training, target_column='failure', sample_method='over')","metadata":{"execution":{"iopub.status.busy":"2022-08-13T23:58:29.489397Z","iopub.execute_input":"2022-08-13T23:58:29.489904Z","iopub.status.idle":"2022-08-13T23:58:29.522008Z","shell.execute_reply.started":"2022-08-13T23:58:29.489856Z","shell.execute_reply":"2022-08-13T23:58:29.520650Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training['failure'].value_counts().plot.pie(autopct='%.2f')\nprint(f'There are {training.shape[0]} rows.')","metadata":{"execution":{"iopub.status.busy":"2022-08-13T23:58:29.525391Z","iopub.execute_input":"2022-08-13T23:58:29.525802Z","iopub.status.idle":"2022-08-13T23:58:29.625760Z","shell.execute_reply.started":"2022-08-13T23:58:29.525768Z","shell.execute_reply":"2022-08-13T23:58:29.623677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The dataset is now balanced. Notice the number of rows has of course increased.","metadata":{}},{"cell_type":"code","source":"Y = training['failure']\nX = training.drop(['failure'], axis=1)\nx_train,x_test,y_train,y_test = train_test_split(X, Y,test_size=0.15)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T23:58:29.629035Z","iopub.execute_input":"2022-08-13T23:58:29.630156Z","iopub.status.idle":"2022-08-13T23:58:29.682159Z","shell.execute_reply.started":"2022-08-13T23:58:29.630087Z","shell.execute_reply":"2022-08-13T23:58:29.680048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = XGBClassifier()\nmodel.fit(x_train, y_train)\ncv_score['over-sampled'] = cross_val_score(model, X, Y, cv=10)\ntest_score['over-sampled'] = model.score(x_test, y_test)\nprint(cv_score['over-sampled'])\nprint(test_score['over-sampled'])","metadata":{"execution":{"iopub.status.busy":"2022-08-13T23:58:29.685319Z","iopub.execute_input":"2022-08-13T23:58:29.686136Z","iopub.status.idle":"2022-08-13T23:59:26.751308Z","shell.execute_reply.started":"2022-08-13T23:58:29.686075Z","shell.execute_reply":"2022-08-13T23:59:26.750425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Under-sampling # ","metadata":{}},{"cell_type":"markdown","source":"Rows from the majority class are deleted until the classes are balanced. This can result in invaluable information being lost/","metadata":{}},{"cell_type":"code","source":"# re-import the data\ntraining, testing = load_data()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T23:59:26.755521Z","iopub.execute_input":"2022-08-13T23:59:26.757694Z","iopub.status.idle":"2022-08-13T23:59:26.963617Z","shell.execute_reply.started":"2022-08-13T23:59:26.757644Z","shell.execute_reply":"2022-08-13T23:59:26.962468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training = sample(df=training, target_column='failure', sample_method='under')","metadata":{"execution":{"iopub.status.busy":"2022-08-13T23:59:26.965105Z","iopub.execute_input":"2022-08-13T23:59:26.965569Z","iopub.status.idle":"2022-08-13T23:59:26.991119Z","shell.execute_reply.started":"2022-08-13T23:59:26.965523Z","shell.execute_reply":"2022-08-13T23:59:26.989956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training['failure'].value_counts().plot.pie(autopct='%.2f')\nprint(f'There are {training.shape[0]} rows.')","metadata":{"execution":{"iopub.status.busy":"2022-08-13T23:59:26.993647Z","iopub.execute_input":"2022-08-13T23:59:26.994331Z","iopub.status.idle":"2022-08-13T23:59:27.088373Z","shell.execute_reply.started":"2022-08-13T23:59:26.994285Z","shell.execute_reply":"2022-08-13T23:59:27.086687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y = training['failure']\nX = training.drop(['failure'], axis=1)\nx_train,x_test,y_train,y_test = train_test_split(X, Y,test_size=0.15)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T23:59:27.096065Z","iopub.execute_input":"2022-08-13T23:59:27.096865Z","iopub.status.idle":"2022-08-13T23:59:27.119143Z","shell.execute_reply.started":"2022-08-13T23:59:27.096799Z","shell.execute_reply":"2022-08-13T23:59:27.117482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = XGBClassifier()\nmodel.fit(x_train, y_train)\ncv_score['under-sampled'] = cross_val_score(model, X, Y, cv=10)\ntest_score['under-sampled'] = model.score(x_test, y_test)\nprint(cv_score['under-sampled'])\nprint(test_score['under-sampled'])","metadata":{"execution":{"iopub.status.busy":"2022-08-13T23:59:27.121599Z","iopub.execute_input":"2022-08-13T23:59:27.122622Z","iopub.status.idle":"2022-08-13T23:59:51.053099Z","shell.execute_reply.started":"2022-08-13T23:59:27.122557Z","shell.execute_reply":"2022-08-13T23:59:51.052164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# SMOTE #","metadata":{}},{"cell_type":"markdown","source":"[SMOTE ](http://machinelearningmastery.com/smote-oversampling-for-imbalanced-classification/) shall be employed to provide a better sampling of the data.","metadata":{}},{"cell_type":"code","source":"# re-import the data\ntraining, testing = load_data()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T23:59:51.057225Z","iopub.execute_input":"2022-08-13T23:59:51.059380Z","iopub.status.idle":"2022-08-13T23:59:51.278787Z","shell.execute_reply.started":"2022-08-13T23:59:51.059335Z","shell.execute_reply":"2022-08-13T23:59:51.277505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# the features and target variables need to first be split.\nY = training['failure']\nX = training.drop(['failure'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T23:59:51.280298Z","iopub.execute_input":"2022-08-13T23:59:51.280781Z","iopub.status.idle":"2022-08-13T23:59:51.291558Z","shell.execute_reply.started":"2022-08-13T23:59:51.280735Z","shell.execute_reply":"2022-08-13T23:59:51.290466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"smt = SMOTE()\nx_res, y_res = smt.fit_resample(X, Y)   # x_res -> x_resampled\ny_res.value_counts().plot.pie(autopct='%.2f')\nprint(f'There are {y_res.shape[0]} rows.')","metadata":{"execution":{"iopub.status.busy":"2022-08-13T23:59:51.293027Z","iopub.execute_input":"2022-08-13T23:59:51.293363Z","iopub.status.idle":"2022-08-13T23:59:52.198353Z","shell.execute_reply.started":"2022-08-13T23:59:51.293332Z","shell.execute_reply":"2022-08-13T23:59:52.196258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The data is now balanced.","metadata":{}},{"cell_type":"code","source":"x_train,x_test,y_train,y_test = train_test_split(x_res, y_res,test_size=0.15)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T23:59:52.202371Z","iopub.execute_input":"2022-08-13T23:59:52.203574Z","iopub.status.idle":"2022-08-13T23:59:52.229941Z","shell.execute_reply.started":"2022-08-13T23:59:52.203522Z","shell.execute_reply":"2022-08-13T23:59:52.228849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = XGBClassifier()\nmodel.fit(x_train, y_train)\ncv_score['SMOTE'] = cross_val_score(model, X, Y, cv=10)\ntest_score['SMOTE'] = model.score(x_test, y_test)\nprint(cv_score['SMOTE'])\nprint(test_score['SMOTE'])","metadata":{"execution":{"iopub.status.busy":"2022-08-13T23:59:52.232924Z","iopub.execute_input":"2022-08-13T23:59:52.233434Z","iopub.status.idle":"2022-08-14T00:00:39.829313Z","shell.execute_reply.started":"2022-08-13T23:59:52.233368Z","shell.execute_reply":"2022-08-14T00:00:39.828474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots()\nax.boxplot(cv_score.values())\nax.set_xticklabels(cv_score.keys())","metadata":{"execution":{"iopub.status.busy":"2022-08-14T00:00:39.830670Z","iopub.execute_input":"2022-08-14T00:00:39.831213Z","iopub.status.idle":"2022-08-14T00:00:40.055469Z","shell.execute_reply.started":"2022-08-14T00:00:39.831179Z","shell.execute_reply":"2022-08-14T00:00:40.054309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get the test scores and methods as lists\nvalues = list(test_score.values())\nlabels = list(test_score.keys())\n\n# plot bar chart\nfig, ax = plt.subplots()\nplt.bar(range(len(test_score)), values, tick_label=labels)\nplt.ylim(min(values) - 0.01, max(values) + 0.01)   # auto-scales the yaxis for presentation\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-14T00:00:40.057115Z","iopub.execute_input":"2022-08-14T00:00:40.057510Z","iopub.status.idle":"2022-08-14T00:00:40.246675Z","shell.execute_reply.started":"2022-08-14T00:00:40.057474Z","shell.execute_reply":"2022-08-14T00:00:40.245475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Using SMOTE has a clear advantage over other methods.","metadata":{}}]}