{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-08-13T08:58:25.751360Z","iopub.execute_input":"2023-08-13T08:58:25.751793Z","iopub.status.idle":"2023-08-13T08:58:25.760466Z","shell.execute_reply.started":"2023-08-13T08:58:25.751758Z","shell.execute_reply":"2023-08-13T08:58:25.759281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport warnings\nwarnings.filterwarnings(\"ignore\")\nfrom sklearn.metrics import classification_report","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:47:03.653361Z","iopub.execute_input":"2023-11-12T12:47:03.653900Z","iopub.status.idle":"2023-11-12T12:47:05.509641Z","shell.execute_reply.started":"2023-11-12T12:47:03.653843Z","shell.execute_reply":"2023-11-12T12:47:05.508409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **EDA**","metadata":{}},{"cell_type":"code","source":"# importing the training dataset\ntrain_df = pd.read_csv('/kaggle/input/amex-default-prediction/train_data.csv' , nrows = 150000)\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:47:05.511934Z","iopub.execute_input":"2023-11-12T12:47:05.512333Z","iopub.status.idle":"2023-11-12T12:47:17.249997Z","shell.execute_reply.started":"2023-11-12T12:47:05.512301Z","shell.execute_reply":"2023-11-12T12:47:17.248909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.shape","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:47:17.251126Z","iopub.execute_input":"2023-11-12T12:47:17.251421Z","iopub.status.idle":"2023-11-12T12:47:17.257852Z","shell.execute_reply.started":"2023-11-12T12:47:17.251396Z","shell.execute_reply":"2023-11-12T12:47:17.256864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.describe()","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:47:17.260686Z","iopub.execute_input":"2023-11-12T12:47:17.261086Z","iopub.status.idle":"2023-11-12T12:47:18.893952Z","shell.execute_reply.started":"2023-11-12T12:47:17.261033Z","shell.execute_reply":"2023-11-12T12:47:18.892777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# information about the dataset\ntrain_df.info()","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:47:18.895306Z","iopub.execute_input":"2023-11-12T12:47:18.895645Z","iopub.status.idle":"2023-11-12T12:47:18.922852Z","shell.execute_reply.started":"2023-11-12T12:47:18.895616Z","shell.execute_reply":"2023-11-12T12:47:18.921751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get the number of unique elements in the dataset\ntrain_df.nunique()","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:47:18.924413Z","iopub.execute_input":"2023-11-12T12:47:18.925161Z","iopub.status.idle":"2023-11-12T12:47:20.144008Z","shell.execute_reply.started":"2023-11-12T12:47:18.925119Z","shell.execute_reply":"2023-11-12T12:47:20.142933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# it is noticed that out of 150000 samples only 12441 unique customers are available","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:47:20.145885Z","iopub.execute_input":"2023-11-12T12:47:20.146360Z","iopub.status.idle":"2023-11-12T12:47:20.151635Z","shell.execute_reply.started":"2023-11-12T12:47:20.146318Z","shell.execute_reply":"2023-11-12T12:47:20.150380Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels = pd.read_csv('/kaggle/input/amex-default-prediction/train_labels.csv' , nrows = 150000)\ntrain_labels.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:47:20.153312Z","iopub.execute_input":"2023-11-12T12:47:20.153825Z","iopub.status.idle":"2023-11-12T12:47:20.475362Z","shell.execute_reply.started":"2023-11-12T12:47:20.153770Z","shell.execute_reply":"2023-11-12T12:47:20.474204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get the number of unique elements in the dataset\ntrain_labels.nunique()","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:47:20.476832Z","iopub.execute_input":"2023-11-12T12:47:20.477187Z","iopub.status.idle":"2023-11-12T12:47:20.558447Z","shell.execute_reply.started":"2023-11-12T12:47:20.477156Z","shell.execute_reply":"2023-11-12T12:47:20.557326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check target value distribution\nplt.figure(dpi=100)\nsns.countplot(data=train_labels, x='target')\nplt.xlabel('Target')\nplt.ylabel('Target Score')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:47:20.562459Z","iopub.execute_input":"2023-11-12T12:47:20.563105Z","iopub.status.idle":"2023-11-12T12:47:20.810126Z","shell.execute_reply.started":"2023-11-12T12:47:20.563062Z","shell.execute_reply":"2023-11-12T12:47:20.809123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels['target'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:47:20.811616Z","iopub.execute_input":"2023-11-12T12:47:20.811960Z","iopub.status.idle":"2023-11-12T12:47:20.822101Z","shell.execute_reply.started":"2023-11-12T12:47:20.811930Z","shell.execute_reply":"2023-11-12T12:47:20.821078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# merge the two dataframes based on customer_ID\ntrain_df = pd.merge(train_df , train_labels , on='customer_ID')\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:47:20.823685Z","iopub.execute_input":"2023-11-12T12:47:20.824060Z","iopub.status.idle":"2023-11-12T12:47:21.123149Z","shell.execute_reply.started":"2023-11-12T12:47:20.824004Z","shell.execute_reply":"2023-11-12T12:47:21.122103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get the number of unique elements in the dataset\ntrain_df.nunique()","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:47:21.124937Z","iopub.execute_input":"2023-11-12T12:47:21.125545Z","iopub.status.idle":"2023-11-12T12:47:22.362459Z","shell.execute_reply.started":"2023-11-12T12:47:21.125501Z","shell.execute_reply":"2023-11-12T12:47:22.361401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Correlation matrix\n# Calculate the correlation matrix\ncorr_matrix = train_df.corr()\ncorr_matrix","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:47:22.363973Z","iopub.execute_input":"2023-11-12T12:47:22.364434Z","iopub.status.idle":"2023-11-12T12:47:34.315315Z","shell.execute_reply.started":"2023-11-12T12:47:22.364395Z","shell.execute_reply":"2023-11-12T12:47:34.314143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# dropping the highly correlated features\n# Create correlation matrix\ncorr_matrix = train_df.drop(['target'] , axis = 1).corr().abs()\n\n# Select upper triangle of correlation matrix\nupper = corr_matrix.where(np.triu(np.ones(corr_matrix.shape), k=1).astype(bool))\n\n# Find features with correlation greater than 0.8\nto_drop = [column for column in upper.columns if any(upper[column] > 0.8)]\n\n# Drop features \ntrain_df.drop(to_drop, axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:48:08.359572Z","iopub.execute_input":"2023-11-12T12:48:08.360125Z","iopub.status.idle":"2023-11-12T12:48:20.308775Z","shell.execute_reply.started":"2023-11-12T12:48:08.360082Z","shell.execute_reply":"2023-11-12T12:48:20.307470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"to_drop","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:48:23.349537Z","iopub.execute_input":"2023-11-12T12:48:23.349923Z","iopub.status.idle":"2023-11-12T12:48:23.358566Z","shell.execute_reply.started":"2023-11-12T12:48:23.349892Z","shell.execute_reply":"2023-11-12T12:48:23.357244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(to_drop)","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:48:31.549549Z","iopub.execute_input":"2023-11-12T12:48:31.550041Z","iopub.status.idle":"2023-11-12T12:48:31.558171Z","shell.execute_reply.started":"2023-11-12T12:48:31.550007Z","shell.execute_reply":"2023-11-12T12:48:31.556818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# visulalize the relationship between each pair of attributes using a scatter matrix\n# scatter matrix will get different colors based on the target value\n# sns.pairplot(train_df , hue='target' , palette='husl')","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:48:49.494667Z","iopub.execute_input":"2023-11-12T12:48:49.495917Z","iopub.status.idle":"2023-11-12T12:48:49.500960Z","shell.execute_reply.started":"2023-11-12T12:48:49.495865Z","shell.execute_reply":"2023-11-12T12:48:49.499834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualize the distribution of data for each attribute using a box plot for each class\n# we need to group by the quality and count the no of records with that relevant quality\n\n# we need to loop through the all of the columns except 'qaulity' column\n# for i in range(len(train_df.columns) - 1):\n\n#   # to seperate the plots from overlapping\n#   print('\\n')\n\n#   # set the figure size\n#   plt.figure(figsize=(10, 6))\n\n#   # use quality column for color encoding(as legend) by setting the hue value\n#   sns.boxplot(x = 'target' , y = train_df.columns[i] , data = train_df , palette = sns.color_palette('bright'))\n\n#   # set the title\n#   plt.title(f'Target distribution over {train_df.columns[i]}')\n\n#   # set legend\n#   # NOTE : setting the hue value will show strange behaviour in the plot so it is commented\n#   # plt.legend(title=\"Wine Quality\", loc=\"upper right\" , fontsize=8)\n\n#   # set the label of the x-axis\n#   plt.xlabel('Target')\n\n#   # set the label for the y-axis\n#   plt.ylabel(f'{train_df.columns[i]}')\n\n#   # show the plot\n#   plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:48:50.284324Z","iopub.execute_input":"2023-11-12T12:48:50.285356Z","iopub.status.idle":"2023-11-12T12:48:50.291833Z","shell.execute_reply.started":"2023-11-12T12:48:50.285312Z","shell.execute_reply":"2023-11-12T12:48:50.290595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# find the mean of each and every column\ntrain_df.mean()","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:48:51.370393Z","iopub.execute_input":"2023-11-12T12:48:51.371003Z","iopub.status.idle":"2023-11-12T12:50:07.078730Z","shell.execute_reply.started":"2023-11-12T12:48:51.370971Z","shell.execute_reply":"2023-11-12T12:50:07.077452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !pip install pandas-profiling","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-11-12T12:50:07.081208Z","iopub.execute_input":"2023-11-12T12:50:07.081663Z","iopub.status.idle":"2023-11-12T12:50:07.086611Z","shell.execute_reply.started":"2023-11-12T12:50:07.081611Z","shell.execute_reply":"2023-11-12T12:50:07.085427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # import the required libraries\n# import pandas_profiling\n\n# # Generate a data profiling report\n# profile = train_df.profile_report()\n\n# # call this function to start generating the report\n# profile","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:50:07.088013Z","iopub.execute_input":"2023-11-12T12:50:07.088818Z","iopub.status.idle":"2023-11-12T12:50:07.100971Z","shell.execute_reply.started":"2023-11-12T12:50:07.088778Z","shell.execute_reply":"2023-11-12T12:50:07.099954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Handling the missing values**","metadata":{}},{"cell_type":"code","source":"# checking the columns for missing values\ntrain_df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:50:07.103086Z","iopub.execute_input":"2023-11-12T12:50:07.103426Z","iopub.status.idle":"2023-11-12T12:50:07.346113Z","shell.execute_reply.started":"2023-11-12T12:50:07.103397Z","shell.execute_reply":"2023-11-12T12:50:07.345161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# calculate the missing value percentage\ntrain_df.isna().mean() * 100","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:50:29.310547Z","iopub.execute_input":"2023-11-12T12:50:29.311025Z","iopub.status.idle":"2023-11-12T12:50:29.550849Z","shell.execute_reply.started":"2023-11-12T12:50:29.310994Z","shell.execute_reply":"2023-11-12T12:50:29.549668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# we can drop the columns that higher missing value percentage\n# Assuming 'df' is your DataFrame\nthreshold = 0.3  # 30% threshold\n\n# Calculate the missing value percentage for each column\nmissing_percentage = train_df.isnull().mean() * 100\n\n# Get the columns with missing percentage greater than the threshold\ncolumns_to_drop = missing_percentage[missing_percentage > threshold].index\n\n# Drop the columns from the DataFrame\ntrain_df = train_df.drop(columns=columns_to_drop)\n\n# Now 'df' contains the DataFrame with columns dropped if missing percentage > 30%","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:50:31.455903Z","iopub.execute_input":"2023-11-12T12:50:31.456398Z","iopub.status.idle":"2023-11-12T12:50:31.753890Z","shell.execute_reply.started":"2023-11-12T12:50:31.456363Z","shell.execute_reply":"2023-11-12T12:50:31.752596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"columns_to_drop","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:50:33.025211Z","iopub.execute_input":"2023-11-12T12:50:33.025707Z","iopub.status.idle":"2023-11-12T12:50:33.035160Z","shell.execute_reply.started":"2023-11-12T12:50:33.025671Z","shell.execute_reply":"2023-11-12T12:50:33.033603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# number of columns dropped\nlen(columns_to_drop)","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:50:34.385036Z","iopub.execute_input":"2023-11-12T12:50:34.385531Z","iopub.status.idle":"2023-11-12T12:50:34.393972Z","shell.execute_reply.started":"2023-11-12T12:50:34.385495Z","shell.execute_reply":"2023-11-12T12:50:34.392541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get the remaining categorical columns after dropping the rows\ncategorical_cols = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120','D_126','D_63', 'D_64', 'D_66']\n\nremaining_categorical = []\n\nfor col in categorical_cols:\n    if col in train_df.columns:\n        remaining_categorical.append(col)\n\nremaining_categorical","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:50:39.419460Z","iopub.execute_input":"2023-11-12T12:50:39.419953Z","iopub.status.idle":"2023-11-12T12:50:39.431438Z","shell.execute_reply.started":"2023-11-12T12:50:39.419921Z","shell.execute_reply":"2023-11-12T12:50:39.430007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['B_30'].unique()","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:50:42.291575Z","iopub.execute_input":"2023-11-12T12:50:42.292040Z","iopub.status.idle":"2023-11-12T12:50:42.302203Z","shell.execute_reply.started":"2023-11-12T12:50:42.292007Z","shell.execute_reply":"2023-11-12T12:50:42.300876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['B_38'].unique()","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:50:43.156641Z","iopub.execute_input":"2023-11-12T12:50:43.157428Z","iopub.status.idle":"2023-11-12T12:50:43.167991Z","shell.execute_reply.started":"2023-11-12T12:50:43.157372Z","shell.execute_reply":"2023-11-12T12:50:43.166486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['D_63'].unique()","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:50:43.881368Z","iopub.execute_input":"2023-11-12T12:50:43.881834Z","iopub.status.idle":"2023-11-12T12:50:43.902546Z","shell.execute_reply.started":"2023-11-12T12:50:43.881794Z","shell.execute_reply":"2023-11-12T12:50:43.901095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# It is observed that B_30 and B_38 contains Null values","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:50:46.731277Z","iopub.execute_input":"2023-11-12T12:50:46.731710Z","iopub.status.idle":"2023-11-12T12:50:46.736936Z","shell.execute_reply.started":"2023-11-12T12:50:46.731676Z","shell.execute_reply":"2023-11-12T12:50:46.735481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# before standardization\ntrain_df.mean()","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:50:47.514425Z","iopub.execute_input":"2023-11-12T12:50:47.514890Z","iopub.status.idle":"2023-11-12T12:52:11.137762Z","shell.execute_reply.started":"2023-11-12T12:50:47.514858Z","shell.execute_reply":"2023-11-12T12:52:11.136528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# before standardization\ntrain_df.std()","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:52:11.140222Z","iopub.execute_input":"2023-11-12T12:52:11.140681Z","iopub.status.idle":"2023-11-12T12:52:11.499219Z","shell.execute_reply.started":"2023-11-12T12:52:11.140639Z","shell.execute_reply":"2023-11-12T12:52:11.498369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Data Preprocessing**","metadata":{}},{"cell_type":"code","source":"# removing customer_ID and S_2 columns before standardization\ntrain_df.drop(['customer_ID' , 'S_2'] , axis = 1 , inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:52:11.500408Z","iopub.execute_input":"2023-11-12T12:52:11.500898Z","iopub.status.idle":"2023-11-12T12:52:11.545379Z","shell.execute_reply.started":"2023-11-12T12:52:11.500869Z","shell.execute_reply":"2023-11-12T12:52:11.544294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# fill the columns with mode of that column\nfor column in ['B_30','B_38']:\n    mode_value = train_df[column].mode()[0]\n    train_df[column].fillna(mode_value, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:52:26.644115Z","iopub.execute_input":"2023-11-12T12:52:26.644567Z","iopub.status.idle":"2023-11-12T12:52:26.659585Z","shell.execute_reply.started":"2023-11-12T12:52:26.644535Z","shell.execute_reply":"2023-11-12T12:52:26.657876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# do label encoding\n# Import label encoder\nfrom sklearn import preprocessing\n\nlabel_encoder = preprocessing.LabelEncoder()\n  \n# Encode labels in column 'species'.\ntrain_df['D_63']= label_encoder.fit_transform(train_df['D_63'])\n  \ntrain_df['D_63'].unique()","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:52:27.661705Z","iopub.execute_input":"2023-11-12T12:52:27.662208Z","iopub.status.idle":"2023-11-12T12:52:27.729928Z","shell.execute_reply.started":"2023-11-12T12:52:27.662173Z","shell.execute_reply":"2023-11-12T12:52:27.728472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:52:28.609142Z","iopub.execute_input":"2023-11-12T12:52:28.609578Z","iopub.status.idle":"2023-11-12T12:52:28.643793Z","shell.execute_reply.started":"2023-11-12T12:52:28.609545Z","shell.execute_reply":"2023-11-12T12:52:28.642304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# upto this point all the categorical columns are label encoded or dropped \n# now we can process the columns with numerical data\ntrain_df = train_df.fillna(train_df.mean())\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:52:31.044973Z","iopub.execute_input":"2023-11-12T12:52:31.045412Z","iopub.status.idle":"2023-11-12T12:52:33.014970Z","shell.execute_reply.started":"2023-11-12T12:52:31.045380Z","shell.execute_reply":"2023-11-12T12:52:33.013845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get all columns except the categorical ones\nto_be_std = train_df.loc[: , ~train_df.columns.isin(['B_30', 'B_38', 'D_63'])]\nto_be_std","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:52:37.921428Z","iopub.execute_input":"2023-11-12T12:52:37.921904Z","iopub.status.idle":"2023-11-12T12:52:38.159907Z","shell.execute_reply.started":"2023-11-12T12:52:37.921872Z","shell.execute_reply":"2023-11-12T12:52:38.158353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import the libraries required for standardization\nfrom sklearn.preprocessing import StandardScaler\n\n# get all columns except the categorical ones\nto_be_std = train_df.loc[: , ~train_df.columns.isin(['B_30', 'B_38', 'D_63','target'])]\n\n# initialize scaler\nscaler = StandardScaler()\n\n# standardize data\nstandardized_data = scaler.fit_transform(to_be_std)\n\n# convert numpy array back to pandas dataframe\nto_be_std = pd.DataFrame(standardized_data, columns=to_be_std.columns)","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:53:14.929162Z","iopub.execute_input":"2023-11-12T12:53:14.929587Z","iopub.status.idle":"2023-11-12T12:53:15.256642Z","shell.execute_reply.started":"2023-11-12T12:53:14.929555Z","shell.execute_reply":"2023-11-12T12:53:15.255423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.loc[: , ~train_df.columns.isin(['B_30', 'B_38', 'D_63','target'])] = to_be_std","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:53:16.919255Z","iopub.execute_input":"2023-11-12T12:53:16.919732Z","iopub.status.idle":"2023-11-12T12:53:17.006685Z","shell.execute_reply.started":"2023-11-12T12:53:16.919695Z","shell.execute_reply":"2023-11-12T12:53:17.005337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"to_be_std.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:53:17.926702Z","iopub.execute_input":"2023-11-12T12:53:17.927174Z","iopub.status.idle":"2023-11-12T12:53:17.955803Z","shell.execute_reply.started":"2023-11-12T12:53:17.927142Z","shell.execute_reply":"2023-11-12T12:53:17.954825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# do onehot encoding\ntrain_df = pd.get_dummies(train_df , columns=['B_30', 'B_38', 'D_63'])","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:53:19.239575Z","iopub.execute_input":"2023-11-12T12:53:19.240029Z","iopub.status.idle":"2023-11-12T12:53:19.451363Z","shell.execute_reply.started":"2023-11-12T12:53:19.239996Z","shell.execute_reply":"2023-11-12T12:53:19.450035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.info()","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:53:20.216347Z","iopub.execute_input":"2023-11-12T12:53:20.217125Z","iopub.status.idle":"2023-11-12T12:53:20.232613Z","shell.execute_reply.started":"2023-11-12T12:53:20.217091Z","shell.execute_reply":"2023-11-12T12:53:20.231478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get all the columns that has null values\ntrain_df.columns[train_df.isnull().any()]","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:53:21.423015Z","iopub.execute_input":"2023-11-12T12:53:21.423525Z","iopub.status.idle":"2023-11-12T12:53:21.454198Z","shell.execute_reply.started":"2023-11-12T12:53:21.423488Z","shell.execute_reply":"2023-11-12T12:53:21.452703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get the X , y values\nX = train_df.drop(['target'] , axis = 1)\ny = train_df['target']","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:53:22.399448Z","iopub.execute_input":"2023-11-12T12:53:22.399935Z","iopub.status.idle":"2023-11-12T12:53:22.463200Z","shell.execute_reply.started":"2023-11-12T12:53:22.399899Z","shell.execute_reply":"2023-11-12T12:53:22.460359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:53:23.299187Z","iopub.execute_input":"2023-11-12T12:53:23.299680Z","iopub.status.idle":"2023-11-12T12:53:23.327003Z","shell.execute_reply.started":"2023-11-12T12:53:23.299645Z","shell.execute_reply":"2023-11-12T12:53:23.325621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:53:24.015594Z","iopub.execute_input":"2023-11-12T12:53:24.016135Z","iopub.status.idle":"2023-11-12T12:53:24.025713Z","shell.execute_reply.started":"2023-11-12T12:53:24.016098Z","shell.execute_reply":"2023-11-12T12:53:24.024452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# after standardization\nX.mean()","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:53:25.680116Z","iopub.execute_input":"2023-11-12T12:53:25.680520Z","iopub.status.idle":"2023-11-12T12:53:25.725318Z","shell.execute_reply.started":"2023-11-12T12:53:25.680488Z","shell.execute_reply":"2023-11-12T12:53:25.723975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# after standardization\nX.std()","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:53:27.672361Z","iopub.execute_input":"2023-11-12T12:53:27.673608Z","iopub.status.idle":"2023-11-12T12:53:27.911059Z","shell.execute_reply.started":"2023-11-12T12:53:27.673551Z","shell.execute_reply":"2023-11-12T12:53:27.909591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Logistic Regression**","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\n# # Split the data into training and testing sets (80% training, 20% testing)\n# The random_state parameter is used to ensure reproducibility of the results.\n# By setting it to a specific value (in this case, 42),\n# the random shuffling and splitting of the data will be the same every time the code is executed.\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.20, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:53:31.359731Z","iopub.execute_input":"2023-11-12T12:53:31.361042Z","iopub.status.idle":"2023-11-12T12:53:31.524838Z","shell.execute_reply.started":"2023-11-12T12:53:31.360990Z","shell.execute_reply":"2023-11-12T12:53:31.523700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\n\n# Create an instance of the logistic regression model\nlogreg_model = LogisticRegression()\n\n# Fit the model to the training data\nlogreg_model.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:53:33.609794Z","iopub.execute_input":"2023-11-12T12:53:33.610550Z","iopub.status.idle":"2023-11-12T12:53:36.730084Z","shell.execute_reply.started":"2023-11-12T12:53:33.610502Z","shell.execute_reply":"2023-11-12T12:53:36.728574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import classification_report, confusion_matrix\n\n# Predict the labels of the test set\ny_pred = logreg_model.predict(X_test)\n\n# Create confusion matrix\nconfusion_mat = confusion_matrix(y_test, y_pred)\n\n# plot confusion matrix using seaborn\nplt.figure(dpi=100)\nsns.heatmap(confusion_mat, annot=True, cmap='Blues', fmt='g')\nplt.xlabel('Predicted')\nplt.ylabel('Actual')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:53:38.329527Z","iopub.execute_input":"2023-11-12T12:53:38.329999Z","iopub.status.idle":"2023-11-12T12:53:38.732832Z","shell.execute_reply.started":"2023-11-12T12:53:38.329957Z","shell.execute_reply":"2023-11-12T12:53:38.731198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluate the model on the test set\nprint(classification_report(y_test, y_pred))","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:53:42.509648Z","iopub.execute_input":"2023-11-12T12:53:42.510145Z","iopub.status.idle":"2023-11-12T12:53:42.574612Z","shell.execute_reply.started":"2023-11-12T12:53:42.510109Z","shell.execute_reply":"2023-11-12T12:53:42.573128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score\n\n# Define the evaluation metric\ndef evaluate_model(X_train, X_test, y_train, y_test):\n    model = LogisticRegression()\n    model.fit(X_train, y_train)\n    y_pred = model.predict(X_test)\n    return accuracy_score(y_test, y_pred)\n\nsplit_ratios = [0.1, 0.15, 0.2, 0.25, 0.3, 0.35, 0.4 , 0.45 , 0.5 , 0.55 , 0.6 , 0.65 , 0.7 , 0.75 , 0.8]\n\nresults = []\nfor split_ratio in split_ratios:\n    X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=split_ratio, random_state=42)\n    accuracy = evaluate_model(X_train, X_test, y_train, y_test)\n    results.append((split_ratio, accuracy))\n\nprint(\"Split Ratio\\tAccuracy\")\nfor split_ratio, accuracy in results:\n    print(f\"{split_ratio}\\t\\t{accuracy}\")\n\n# Find the optimal split ratio with the highest accuracy\noptimal_split = max(results, key=lambda x: x[1])\nprint(\"Optimal Split Ratio:\", optimal_split[0])","metadata":{"execution":{"iopub.status.busy":"2023-08-26T18:38:20.239050Z","iopub.execute_input":"2023-08-26T18:38:20.239732Z","iopub.status.idle":"2023-08-26T18:38:54.612158Z","shell.execute_reply.started":"2023-08-26T18:38:20.239698Z","shell.execute_reply":"2023-08-26T18:38:54.610514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.model_selection import GridSearchCV\n\n# # Split the data into a training set and a test set\n# X_train, X_test, y_train, y_test = train_test_split(X , y, test_size=0.3 , random_state=42)\n\n# # Create a list of hyperparameters to tune\n# hyperparameters = {\n#     'C': [0.1, 1 , 10 , 100],\n#     'penalty': ['l1', 'l2' , 'elasticnet', 'none'],\n#     'solver': ['liblinear', 'lbfgs' , 'saga' , 'newton-cg' , 'sag'],\n#     'max_iter': [100, 200 , 300]\n# }\n\n# # Create a grid of hyperparameters\n# # Perform grid search using cross-validation\n# grid = GridSearchCV(LogisticRegression(), hyperparameters, cv=5)\n\n# # Train a model for each combination of hyperparameters in the grid\n# grid.fit(X_train, y_train)\n\n# # Predict the labels of the test set\n# y_pred = grid.predict(X_test)\n\n# # Evaluate the models on the test set\n# accuracy = accuracy_score(y_test , y_pred)\n\n# # Choose the model with the best performance on the test set\n# # best_model = grid.best_estimator_\n# best_model = grid.best_params_\n# best_score = grid.best_score_\n\n# print(f'Best model: {best_model}')\n# print(f'Accuracy: {accuracy}')","metadata":{"execution":{"iopub.status.busy":"2023-08-26T18:38:54.614419Z","iopub.execute_input":"2023-08-26T18:38:54.615895Z","iopub.status.idle":"2023-08-26T18:38:54.625679Z","shell.execute_reply.started":"2023-08-26T18:38:54.615829Z","shell.execute_reply":"2023-08-26T18:38:54.624143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import f1_score\n\n# Split the data into a training set and a test set\nX_train, X_test, y_train, y_test = train_test_split(X , y, test_size=0.3 , random_state=42)\n\n# Create a linear regression model\nmodel = LogisticRegression(penalty='l2' , C = 1 , solver = 'saga' , max_iter = 200)\n\n# Train the model on the training set\nmodel.fit(X_train , y_train)\n\n# Predict the labels of the test set\ny_pred = model.predict(X_test)\n\n# Evaluate the model on the test set\nprint(classification_report(y_test, y_pred))","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:53:49.070555Z","iopub.execute_input":"2023-11-12T12:53:49.071038Z","iopub.status.idle":"2023-11-12T12:54:37.673162Z","shell.execute_reply.started":"2023-11-12T12:53:49.071004Z","shell.execute_reply":"2023-11-12T12:54:37.671959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create confusion matrix\nconfusion_mat = confusion_matrix(y_test, y_pred)\n\n# plot confusion matrix using seaborn\nplt.figure(dpi=100)\nsns.heatmap(confusion_mat, annot=True, cmap='Blues', fmt='g')\nplt.xlabel('Predicted')\nplt.ylabel('Actual')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:54:37.675372Z","iopub.execute_input":"2023-11-12T12:54:37.675730Z","iopub.status.idle":"2023-11-12T12:54:37.972392Z","shell.execute_reply.started":"2023-11-12T12:54:37.675699Z","shell.execute_reply":"2023-11-12T12:54:37.971142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Balanced Random Forest Classifier**","metadata":{}},{"cell_type":"code","source":"from imblearn.ensemble import BalancedRandomForestClassifier\n\n# create the randomForestClassifier with default params\nclf = BalancedRandomForestClassifier()\n\n# fit the classifier for \nclf.fit(X_train , y_train)\ny_pred = clf.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:54:37.973757Z","iopub.execute_input":"2023-11-12T12:54:37.974125Z","iopub.status.idle":"2023-11-12T12:56:11.517628Z","shell.execute_reply.started":"2023-11-12T12:54:37.974094Z","shell.execute_reply":"2023-11-12T12:56:11.516578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create confusion matrix\nconfusion_mat = confusion_matrix(y_test, y_pred)\n\n# plot confusion matrix using seaborn\nplt.figure(dpi=100)\nsns.heatmap(confusion_mat, annot=True, cmap='Blues', fmt='g')\nplt.xlabel('Predicted')\nplt.ylabel('Actual')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:56:11.520488Z","iopub.execute_input":"2023-11-12T12:56:11.520867Z","iopub.status.idle":"2023-11-12T12:56:11.808303Z","shell.execute_reply.started":"2023-11-12T12:56:11.520835Z","shell.execute_reply":"2023-11-12T12:56:11.807118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluate the model on the test set\nprint(classification_report(y_test, y_pred))","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:56:11.809746Z","iopub.execute_input":"2023-11-12T12:56:11.810203Z","iopub.status.idle":"2023-11-12T12:56:11.900525Z","shell.execute_reply.started":"2023-11-12T12:56:11.810170Z","shell.execute_reply":"2023-11-12T12:56:11.899516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **XGBoost Classifier**","metadata":{}},{"cell_type":"code","source":"import xgboost as xgb\n\n# declare parameters\nparams = {\n            'objective':'reg:logistic',\n            'random_state': 42,\n        }\n\nxgb_model = xgb.XGBClassifier(**params)\nxgb_model.fit(X_train, y_train)\n\n# do the prediction\ny_pred = xgb_model.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:56:11.901802Z","iopub.execute_input":"2023-11-12T12:56:11.902329Z","iopub.status.idle":"2023-11-12T12:57:50.297169Z","shell.execute_reply.started":"2023-11-12T12:56:11.902298Z","shell.execute_reply":"2023-11-12T12:57:50.296123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create confusion matrix\nconfusion_mat = confusion_matrix(y_test, y_pred)\n\n# plot confusion matrix using seaborn\nplt.figure(dpi=100)\nsns.heatmap(confusion_mat, annot=True, cmap='Blues', fmt='g')\nplt.xlabel('Predicted')\nplt.ylabel('Actual')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:57:50.298996Z","iopub.execute_input":"2023-11-12T12:57:50.299764Z","iopub.status.idle":"2023-11-12T12:57:50.594327Z","shell.execute_reply.started":"2023-11-12T12:57:50.299727Z","shell.execute_reply":"2023-11-12T12:57:50.593442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluate the model on the test set\nprint(classification_report(y_test, y_pred))","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:57:50.595760Z","iopub.execute_input":"2023-11-12T12:57:50.596373Z","iopub.status.idle":"2023-11-12T12:57:50.682763Z","shell.execute_reply.started":"2023-11-12T12:57:50.596342Z","shell.execute_reply":"2023-11-12T12:57:50.681557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Handling the class imbalance**","metadata":{}},{"cell_type":"code","source":"# imbalance percentage\ntrain_df['target'].value_counts().min()","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:57:50.683950Z","iopub.execute_input":"2023-11-12T12:57:50.684305Z","iopub.status.idle":"2023-11-12T12:57:50.694486Z","shell.execute_reply.started":"2023-11-12T12:57:50.684275Z","shell.execute_reply":"2023-11-12T12:57:50.693363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# imbalance percentage\nprint(f\"{train_df['target'].value_counts().min() / train_df['target'].count() * 100 : .3f}%\")","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:57:50.698530Z","iopub.execute_input":"2023-11-12T12:57:50.698958Z","iopub.status.idle":"2023-11-12T12:57:50.707741Z","shell.execute_reply.started":"2023-11-12T12:57:50.698927Z","shell.execute_reply":"2023-11-12T12:57:50.706689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# decided to go with Oversampling(creating more duplicates from the minority class)\nfrom imblearn.over_sampling import SMOTE\n\n# Apply SMOTE oversampling to the training data\nsm = SMOTE(random_state=42)\nX_train_resampled, y_train_resampled = sm.fit_resample(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:57:50.708968Z","iopub.execute_input":"2023-11-12T12:57:50.709361Z","iopub.status.idle":"2023-11-12T12:57:54.576366Z","shell.execute_reply.started":"2023-11-12T12:57:50.709329Z","shell.execute_reply":"2023-11-12T12:57:54.575202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train_resampled.shape[0]","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:57:54.577586Z","iopub.execute_input":"2023-11-12T12:57:54.577912Z","iopub.status.idle":"2023-11-12T12:57:54.584367Z","shell.execute_reply.started":"2023-11-12T12:57:54.577884Z","shell.execute_reply":"2023-11-12T12:57:54.583272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train_resampled.shape[0]","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:57:54.585516Z","iopub.execute_input":"2023-11-12T12:57:54.585855Z","iopub.status.idle":"2023-11-12T12:57:54.596675Z","shell.execute_reply.started":"2023-11-12T12:57:54.585827Z","shell.execute_reply":"2023-11-12T12:57:54.595560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# after oversampling tried to get the results from Logistic regression model\nmodel.fit(X_train_resampled, y_train_resampled)\ny_pred = model.predict(X_test)\n\n# Evaluate the performance of the model on the resampled test data\nprint(classification_report(y_test, y_pred))","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:57:54.597874Z","iopub.execute_input":"2023-11-12T12:57:54.598227Z","iopub.status.idle":"2023-11-12T12:59:07.919473Z","shell.execute_reply.started":"2023-11-12T12:57:54.598197Z","shell.execute_reply":"2023-11-12T12:59:07.918284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create confusion matrix\nconfusion_mat = confusion_matrix(y_test, y_pred)\n\n# plot confusion matrix using seaborn\nplt.figure(dpi=100)\nsns.heatmap(confusion_mat, annot=True, cmap='Blues', fmt='g')\nplt.xlabel('Predicted')\nplt.ylabel('Actual')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:59:07.920965Z","iopub.execute_input":"2023-11-12T12:59:07.921374Z","iopub.status.idle":"2023-11-12T12:59:08.209639Z","shell.execute_reply.started":"2023-11-12T12:59:07.921343Z","shell.execute_reply":"2023-11-12T12:59:08.208644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# after oversampling tried to get the results from Random Forest Classifier\nclf.fit(X_train_resampled, y_train_resampled)\ny_pred = clf.predict(X_test)\n\n# Evaluate the performance of the model on the resampled test data\nprint(classification_report(y_test, y_pred))","metadata":{"execution":{"iopub.status.busy":"2023-11-12T12:59:08.210945Z","iopub.execute_input":"2023-11-12T12:59:08.212965Z","iopub.status.idle":"2023-11-12T13:04:39.307538Z","shell.execute_reply.started":"2023-11-12T12:59:08.212933Z","shell.execute_reply":"2023-11-12T13:04:39.306338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create confusion matrix\nconfusion_mat = confusion_matrix(y_test, y_pred)\n\n# plot confusion matrix using seaborn\nplt.figure(dpi=100)\nsns.heatmap(confusion_mat, annot=True, cmap='Blues', fmt='g')\nplt.xlabel('Predicted')\nplt.ylabel('Actual')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-12T13:04:39.309089Z","iopub.execute_input":"2023-11-12T13:04:39.310066Z","iopub.status.idle":"2023-11-12T13:04:39.599945Z","shell.execute_reply.started":"2023-11-12T13:04:39.310011Z","shell.execute_reply":"2023-11-12T13:04:39.598812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# after oversampling tried to get the results from Random Forest Classifier\nxgb_model.fit(X_train_resampled, y_train_resampled)\ny_pred = xgb_model.predict(X_test)\n\n# Evaluate the performance of the model on the resampled test data\nprint(classification_report(y_test, y_pred))","metadata":{"execution":{"iopub.status.busy":"2023-11-12T13:04:39.601096Z","iopub.execute_input":"2023-11-12T13:04:39.601407Z","iopub.status.idle":"2023-11-12T13:07:12.316918Z","shell.execute_reply.started":"2023-11-12T13:04:39.601380Z","shell.execute_reply":"2023-11-12T13:07:12.315752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create confusion matrix\nconfusion_mat = confusion_matrix(y_test, y_pred)\n\n# plot confusion matrix using seaborn\nplt.figure(dpi=100)\nsns.heatmap(confusion_mat, annot=True, cmap='Blues', fmt='g')\nplt.xlabel('Predicted')\nplt.ylabel('Actual')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-12T13:07:12.318441Z","iopub.execute_input":"2023-11-12T13:07:12.319527Z","iopub.status.idle":"2023-11-12T13:07:12.620323Z","shell.execute_reply.started":"2023-11-12T13:07:12.319484Z","shell.execute_reply":"2023-11-12T13:07:12.619125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Checking for Overfitting","metadata":{}},{"cell_type":"code","source":"# logistic regression model\ny_train_pred = model.predict(X_train)\nprint(classification_report(y_train, y_train_pred))","metadata":{"execution":{"iopub.status.busy":"2023-11-12T13:07:12.621732Z","iopub.execute_input":"2023-11-12T13:07:12.622107Z","iopub.status.idle":"2023-11-12T13:07:12.935723Z","shell.execute_reply.started":"2023-11-12T13:07:12.622044Z","shell.execute_reply":"2023-11-12T13:07:12.934592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# balanced random forest classifier\ny_train_pred = clf.predict(X_train)\nprint(classification_report(y_train, y_train_pred))","metadata":{"execution":{"iopub.status.busy":"2023-11-12T13:07:12.937392Z","iopub.execute_input":"2023-11-12T13:07:12.938381Z","iopub.status.idle":"2023-11-12T13:07:17.332275Z","shell.execute_reply.started":"2023-11-12T13:07:12.938345Z","shell.execute_reply":"2023-11-12T13:07:17.330933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# xgb\ny_train_pred = xgb_model.predict(X_train)\nprint(classification_report(y_train, y_train_pred))","metadata":{"execution":{"iopub.status.busy":"2023-11-12T13:07:17.333726Z","iopub.execute_input":"2023-11-12T13:07:17.334255Z","iopub.status.idle":"2023-11-12T13:07:17.724525Z","shell.execute_reply.started":"2023-11-12T13:07:17.334216Z","shell.execute_reply":"2023-11-12T13:07:17.723670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Save the Model**","metadata":{}},{"cell_type":"code","source":"import joblib\n\n# save\njoblib.dump(model, \"model_logreg.pkl\" , compress=9) \njoblib.dump(clf, \"model_clf.pkl\" , compress=9) \njoblib.dump(xgb_model, \"model_xgb.pkl\" , compress=9) \n","metadata":{"execution":{"iopub.status.busy":"2023-11-12T13:07:17.725811Z","iopub.execute_input":"2023-11-12T13:07:17.726161Z","iopub.status.idle":"2023-11-12T13:14:36.330374Z","shell.execute_reply.started":"2023-11-12T13:07:17.726131Z","shell.execute_reply":"2023-11-12T13:14:36.329135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}