{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-06-10T09:04:43.206736Z","iopub.execute_input":"2022-06-10T09:04:43.207576Z","iopub.status.idle":"2022-06-10T09:04:43.246516Z","shell.execute_reply.started":"2022-06-10T09:04:43.207469Z","shell.execute_reply":"2022-06-10T09:04:43.245777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The objective of this competition is to predict the probability that a customer does not pay back their credit card balance amount in the future based on their monthly customer profile. The target binary variable is calculated by observing 18 months performance window after the latest credit card statement, and if the customer does not pay due amount in 120 days after their latest statement date it is considered a default event.\n\nThe dataset contains aggregated profile features for each customer at each statement date. Features are anonymized and normalized, and fall into the following general categories:\n\nD_* = Delinquency variables\nS_* = Spend variables\nP_* = Payment variables\nB_* = Balance variables\nR_* = Risk variables\nwith the following features being categorical:\n\n['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\n\nYour task is to predict, for each customer_ID, the probability of a future payment default (target = 1).\n\nNote that the negative class has been subsampled for this dataset at 5%, and thus receives a 20x weighting in the scoring metric.","metadata":{}},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings(\"ignore\")\nfrom pathlib import Path\nimport seaborn as sb\nimport random\nimport gc\nsb.color_palette(\"Spectral\", as_cmap=True)\n\n\n\nTRAIN_DATA_PATH = \"/kaggle/input/parquet-files-amexdefault-prediction/\"\nTRAIN_FILE = \"train_data\"\nTRAIN_LABELS = \"train_labels\"\n\n","metadata":{"execution":{"iopub.status.busy":"2022-06-10T09:04:43.298902Z","iopub.execute_input":"2022-06-10T09:04:43.299483Z","iopub.status.idle":"2022-06-10T09:04:44.584221Z","shell.execute_reply.started":"2022-06-10T09:04:43.299449Z","shell.execute_reply":"2022-06-10T09:04:44.583297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Convert .csv to feather and save\nCSV_DATA_PATH = \"/kaggle/input/amex-default-prediction/\"\ndf = pd.read_csv(os.path.join(CSV_DATA_PATH,TRAIN_LABELS+\".csv\"))\npath=os.path.join(\"/kaggle/working/\",TRAIN_LABELS+\".ftr\")\nif Path(path).exists():\n   os.remove(\"/kaggle/working/train_labels.ftr\")\nprint(f\"Location of Feather File {path}\")\ndf.to_feather(path)","metadata":{"execution":{"iopub.status.busy":"2022-06-10T09:04:44.585619Z","iopub.execute_input":"2022-06-10T09:04:44.586385Z","iopub.status.idle":"2022-06-10T09:04:45.960989Z","shell.execute_reply.started":"2022-06-10T09:04:44.586346Z","shell.execute_reply":"2022-06-10T09:04:45.959830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_data(fpath,fname,flabels):\n    file_path = os.path.join(fpath,fname+\".ftr\")\n    file_path = Path(file_path)\n    file_labels = Path(os.path.join(\"/kaggle/working/\",flabels+\".ftr\"))\n    if file_path.exists() and file_labels.exists():\n        print(f\"{file_path}, and {file_labels} are available\")\n        df1 = pd.read_feather(file_path,use_threads=True)\n        df2 = pd.read_feather(file_labels,use_threads=True)\n        \n        return df1,df2\n    else:\n        print(\"No Such File\")\n        return\n ","metadata":{"execution":{"iopub.status.busy":"2022-06-10T09:04:45.962886Z","iopub.execute_input":"2022-06-10T09:04:45.963369Z","iopub.status.idle":"2022-06-10T09:04:45.976294Z","shell.execute_reply.started":"2022-06-10T09:04:45.963326Z","shell.execute_reply":"2022-06-10T09:04:45.975005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain_x,train_y = load_data(TRAIN_DATA_PATH,TRAIN_FILE,TRAIN_LABELS )\ntrain_x.info()","metadata":{"execution":{"iopub.status.busy":"2022-06-10T09:04:45.980359Z","iopub.execute_input":"2022-06-10T09:04:45.981423Z","iopub.status.idle":"2022-06-10T09:05:12.917108Z","shell.execute_reply.started":"2022-06-10T09:04:45.981366Z","shell.execute_reply":"2022-06-10T09:05:12.915901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categorical_var = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\ndiscard = [\"customer_ID\",\"S_2\"] + categorical_var\nnumeric_cols = list(set(train_x.columns)-set(discard))\n","metadata":{"execution":{"iopub.status.busy":"2022-06-10T09:05:12.918766Z","iopub.execute_input":"2022-06-10T09:05:12.919203Z","iopub.status.idle":"2022-06-10T09:05:12.925569Z","shell.execute_reply.started":"2022-06-10T09:05:12.919166Z","shell.execute_reply":"2022-06-10T09:05:12.924344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nfor col in train_x.columns:\n    if col in categorical_var:\n        print(col)\n\n#num_cols = train_x._get_numeric_data().columns\n#cat_var = list(set(train_x.columns)-set(num_cols))\n#cat_var","metadata":{"execution":{"iopub.status.busy":"2022-06-10T09:05:12.927048Z","iopub.execute_input":"2022-06-10T09:05:12.928148Z","iopub.status.idle":"2022-06-10T09:05:12.940255Z","shell.execute_reply.started":"2022-06-10T09:05:12.928095Z","shell.execute_reply":"2022-06-10T09:05:12.938897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let us conduct the Chi-Square Test to find the correlation between the categorical variables.","metadata":{}},{"cell_type":"code","source":"%%time\nfrom scipy.stats import chi2_contingency\n'''\nH0: Categorical variables are not correlated\nH1: Categorical variables are highly correlated\n'''\ndrop_catvar=[]\ncols = [i for i in train_x.columns.to_list()]\nfor i in range(len(train_x.columns)-1):\n    col1 = cols[i]\n    col2 = cols[i+1]\n   \n    if col1 in categorical_var:\n        if col2 in categorical_var:\n             \n             result = pd.crosstab(index=train_x[col1],columns=train_x[col2])\n             chi2 = chi2_contingency(result)\n                \n             if chi2[1] >= 0.05:\n                 print(f\"{col1} and {col2} are not correlated, p-value: {chi2[1]}\")\n             else:\n                 print(f\"{col1} and {col2} are correlated, p-value: {chi2[1]}\")\n             drop_catvar.append(col2)\n\ncategorical_var = list(set(categorical_var)-set(drop_catvar))","metadata":{"execution":{"iopub.status.busy":"2022-06-10T09:05:12.941505Z","iopub.execute_input":"2022-06-10T09:05:12.942046Z","iopub.status.idle":"2022-06-10T09:05:14.344820Z","shell.execute_reply.started":"2022-06-10T09:05:12.942003Z","shell.execute_reply":"2022-06-10T09:05:14.343684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We need to drop highly correlated data.","metadata":{}},{"cell_type":"code","source":"df_temp = pd.merge(train_x,train_y,on=[\"customer_ID\"])\n","metadata":{"execution":{"iopub.status.busy":"2022-06-10T09:05:14.346105Z","iopub.execute_input":"2022-06-10T09:05:14.346471Z","iopub.status.idle":"2022-06-10T09:06:59.207223Z","shell.execute_reply.started":"2022-06-10T09:05:14.346439Z","shell.execute_reply":"2022-06-10T09:06:59.206123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Two-Way Table - Conditional Probability for Categorical Variable\nfor col in categorical_var:\n    result = pd.crosstab(index=df_temp['target'],columns=train_x[col],normalize=\"index\",margins=True,dropna=True)\n    print(result)\n    print(\"*\"*20)\ndel df_temp\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-10T09:06:59.208627Z","iopub.execute_input":"2022-06-10T09:06:59.208998Z","iopub.status.idle":"2022-06-10T09:07:12.982301Z","shell.execute_reply.started":"2022-06-10T09:06:59.208965Z","shell.execute_reply":"2022-06-10T09:07:12.981596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Column 'B_30' label '0.0' is influencing most the target column.Column 'B_38' label '2.0' is influencing most the target column.","metadata":{}},{"cell_type":"code","source":"print(train_x[\"D_63\"].iloc[0:5])","metadata":{"execution":{"iopub.status.busy":"2022-06-10T09:07:12.984944Z","iopub.execute_input":"2022-06-10T09:07:12.985758Z","iopub.status.idle":"2022-06-10T09:07:12.992371Z","shell.execute_reply.started":"2022-06-10T09:07:12.985720Z","shell.execute_reply":"2022-06-10T09:07:12.991431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in train_x.columns:\n    if col in categorical_var:\n        dummy = pd.get_dummies(train_x[col],prefix=col)\n        train_x = train_x.join(dummy)\n        train_x.drop(col,axis=1,inplace=True)\ntrain_x.head()   ","metadata":{"execution":{"iopub.status.busy":"2022-06-10T09:07:12.993513Z","iopub.execute_input":"2022-06-10T09:07:12.993854Z","iopub.status.idle":"2022-06-10T09:08:46.506989Z","shell.execute_reply.started":"2022-06-10T09:07:12.993818Z","shell.execute_reply":"2022-06-10T09:08:46.506176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain_x[\"S_2\"]=pd.to_datetime(train_x[\"S_2\"])\n\nfor col in numeric_cols:\n           \n        try:\n            train_x[col]=pd.to_numeric(train_x[col])\n    \n        except:\n               print(\"Casting Error\")\n    ","metadata":{"execution":{"iopub.status.busy":"2022-06-10T09:08:46.508279Z","iopub.execute_input":"2022-06-10T09:08:46.508780Z","iopub.status.idle":"2022-06-10T09:08:48.405925Z","shell.execute_reply.started":"2022-06-10T09:08:46.508748Z","shell.execute_reply":"2022-06-10T09:08:48.404885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Missing Value \ncol_value=[]\ncol_name = []\nfor col in train_x.columns:\n    missing_val = train_x[col].isna().sum()\n    missing_val_per = round(missing_val*100/len(train_x),2)\n    col_value.append(float(missing_val_per))\n    col_name.append(col)\n    #print(f\"Column {col} has {missing_val} number of missing values i.e.{missing_val_per}%\")\n\n\nmissing_val_df = pd.DataFrame({\"Col\":col_name,\"Missing Value in Percentage\":col_value})\nmissing_val_df = missing_val_df.sort_values(\"Missing Value in Percentage\",ascending=False)\n","metadata":{"execution":{"iopub.status.busy":"2022-06-10T09:08:48.407244Z","iopub.execute_input":"2022-06-10T09:08:48.407621Z","iopub.status.idle":"2022-06-10T09:08:54.048297Z","shell.execute_reply.started":"2022-06-10T09:08:48.407587Z","shell.execute_reply":"2022-06-10T09:08:54.047200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nplt.figure(figsize=(40,10))\nplt.bar(missing_val_df[\"Col\"],missing_val_df[\"Missing Value in Percentage\"],width=0.8)\nplt.xticks(rotation=90)\nplt.xlabel(\"Column Name\", fontsize=18)\nplt.ylabel(\"Percentage\",fontsize=18)\nplt.title(\"Missing Value\",fontsize=20)\nplt.legend(\"AMEX\",fontsize=18)\nplt.show()\ndel missing_val_df\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-10T09:08:54.049776Z","iopub.execute_input":"2022-06-10T09:08:54.050170Z","iopub.status.idle":"2022-06-10T09:08:56.701776Z","shell.execute_reply.started":"2022-06-10T09:08:54.050135Z","shell.execute_reply":"2022-06-10T09:08:56.700666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_x=train_x.dropna(axis=1)\ntrain_x.head()\n","metadata":{"execution":{"iopub.status.busy":"2022-06-10T09:08:56.703624Z","iopub.execute_input":"2022-06-10T09:08:56.704477Z","iopub.status.idle":"2022-06-10T09:09:04.609558Z","shell.execute_reply.started":"2022-06-10T09:08:56.704423Z","shell.execute_reply":"2022-06-10T09:09:04.608472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Number of columns\ncatcols=[]\nfor col in train_x.columns:\n    for cat_var in categorical_var:\n        if col.startswith(cat_var):\n            catcols.append(col)\n            \nnumeric_cols = list(set(train_x.columns)-set(discard)-set(catcols))\nlen(numeric_cols)","metadata":{"execution":{"iopub.status.busy":"2022-06-10T09:09:04.610878Z","iopub.execute_input":"2022-06-10T09:09:04.611239Z","iopub.status.idle":"2022-06-10T09:09:04.620179Z","shell.execute_reply.started":"2022-06-10T09:09:04.611199Z","shell.execute_reply":"2022-06-10T09:09:04.619125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Detection of Class Imbalance\ntarget = train_y.drop(\"customer_ID\",axis=1)\ntarget.tail()","metadata":{"execution":{"iopub.status.busy":"2022-06-10T09:09:04.621637Z","iopub.execute_input":"2022-06-10T09:09:04.622022Z","iopub.status.idle":"2022-06-10T09:09:04.639322Z","shell.execute_reply.started":"2022-06-10T09:09:04.621988Z","shell.execute_reply":"2022-06-10T09:09:04.638442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def class_imbalance(df):\n    yes = df[df[\"target\"]==1]\n    no = df[df[\"target\"]==0]\n    pyes = len(yes)*100/(len(no)+len(yes))\n    pno = 100-pyes\n    print(\"Percentage of Class 1:\",round(pyes,2),\"%\")\n    print(\"Percentage of Class 0:\",round(pno,2),\"%\")\n    if (pyes != pno):\n        print(\"Class Imbalance Exsists\\n\")\n    else:\n        print(\"No Class Imbalance Exsists\\n\")\n    plt.figure(figsize=(4,4))\n    xlab = [\"1\",\"0\"]\n    xpos =np.arange(len(xlab))\n    ylab=[pyes/100,pno/100]\n    plt.bar(xpos,ylab,width=0.4,alpha=0.7)\n    plt.xticks(xpos,xlab)\n    plt.title(\"Class Imblance\")\n    plt.legend()\n    plt.show()\n    ","metadata":{"execution":{"iopub.status.busy":"2022-06-10T09:09:04.640980Z","iopub.execute_input":"2022-06-10T09:09:04.641659Z","iopub.status.idle":"2022-06-10T09:09:04.651121Z","shell.execute_reply.started":"2022-06-10T09:09:04.641622Z","shell.execute_reply":"2022-06-10T09:09:04.650190Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_imbalance(target)","metadata":{"execution":{"iopub.status.busy":"2022-06-10T09:09:04.653022Z","iopub.execute_input":"2022-06-10T09:09:04.653907Z","iopub.status.idle":"2022-06-10T09:09:04.840206Z","shell.execute_reply.started":"2022-06-10T09:09:04.653811Z","shell.execute_reply":"2022-06-10T09:09:04.839304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Down sampling factor 3\ndef down_sampling(trainx,trainy):\n    df_merge = pd.merge(trainx,trainy,on=\"customer_ID\")\n    df_merge_1 = df_merge[df_merge['target']==1]\n    df_merge_0 = df_merge[df_merge['target']==0]\n    df_merge_dsample0 = df_merge_0.sample(n=len(df_merge_1))\n    df_dsample = pd.concat([df_merge_dsample0,df_merge_1])\n    train_y_dsample =df_dsample[\"target\"] \n    train_x_dsample = df_dsample.drop(\"target\",axis=1)\n    del df_merge,df_merge_1,df_merge_0,trainx,trainy\n    gc.collect()\n    return train_x_dsample,train_y_dsample","metadata":{"execution":{"iopub.status.busy":"2022-06-10T09:09:04.841256Z","iopub.execute_input":"2022-06-10T09:09:04.842223Z","iopub.status.idle":"2022-06-10T09:09:04.850536Z","shell.execute_reply.started":"2022-06-10T09:09:04.842171Z","shell.execute_reply":"2022-06-10T09:09:04.849259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainx,trainy = down_sampling(train_x,train_y)","metadata":{"execution":{"iopub.status.busy":"2022-06-10T09:09:04.852133Z","iopub.execute_input":"2022-06-10T09:09:04.853107Z","iopub.status.idle":"2022-06-10T09:09:26.617601Z","shell.execute_reply.started":"2022-06-10T09:09:04.853038Z","shell.execute_reply":"2022-06-10T09:09:26.616658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainy_df = pd.DataFrame({\"target\":trainy.to_list()})\nclass_imbalance(trainy_df)","metadata":{"execution":{"iopub.status.busy":"2022-06-10T09:09:26.618881Z","iopub.execute_input":"2022-06-10T09:09:26.619235Z","iopub.status.idle":"2022-06-10T09:09:27.963636Z","shell.execute_reply.started":"2022-06-10T09:09:26.619206Z","shell.execute_reply":"2022-06-10T09:09:27.962842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nplt.figure(figsize=(30,30))\ncol_name = train_x.columns.to_list()\n#numeric_cols = list(set(col_name)-set(discard))\ncorr = trainx[numeric_cols].corr()\nmatrix_mask=np.triu(corr)\nsb.heatmap(corr,annot=True,fmt=\"0.1g\",cmap=\"viridis\",mask=matrix_mask)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-10T09:09:27.964691Z","iopub.execute_input":"2022-06-10T09:09:27.965545Z","iopub.status.idle":"2022-06-10T09:10:10.244195Z","shell.execute_reply.started":"2022-06-10T09:09:27.965506Z","shell.execute_reply":"2022-06-10T09:10:10.242937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Highy correlated columns should be dropped. ","metadata":{}},{"cell_type":"code","source":"#update numeric columns\ndrop_numeric_cols=[]\npair=[]\nfor col in numeric_cols:\n    for i in range(len(corr)):\n        if abs(corr[col].iloc[i]) >= 0.9 and col != numeric_cols[i] :\n            print(f\"{col} and {numeric_cols[i]} are highly correlated...\") \n            if col not in pair:\n                pair.append(col)\n                pair.append(numeric_cols[i])\n                drop_numeric_cols.append(col) \nnumeric_cols = list(set(numeric_cols)-set(drop_numeric_cols))\nprint(f\"Dropping columns : {drop_numeric_cols}\")\ndel drop_numeric_cols\ndel pair\ngc.collect()\n","metadata":{"execution":{"iopub.status.busy":"2022-06-10T09:10:10.246107Z","iopub.execute_input":"2022-06-10T09:10:10.246584Z","iopub.status.idle":"2022-06-10T09:10:10.561591Z","shell.execute_reply.started":"2022-06-10T09:10:10.246539Z","shell.execute_reply":"2022-06-10T09:10:10.560684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfig,axes=plt.subplots(int(len(numeric_cols)/5),5)\nnrow=0\nncol=0\ndf = trainx.sample(1000)\nfor col in numeric_cols:\n    if ncol <5:\n       fig.set_figheight(10)\n       fig.set_figwidth(10)\n       g=sb.distplot(df[col],hist=True,kde=True,ax=axes[nrow,ncol])\n       g.set(title=col)\n       g.set(xlabel=None)\n       g.set(ylabel=None)\n        \n       ncol +=1\n    else:\n        nrow +=1\n        ncol=0\nfig.subplots_adjust(hspace=1.5) \nfig.subplots_adjust(wspace=1) \nplt.show()\ndel df\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-10T09:10:10.562701Z","iopub.execute_input":"2022-06-10T09:10:10.563044Z","iopub.status.idle":"2022-06-10T09:10:20.982982Z","shell.execute_reply.started":"2022-06-10T09:10:10.563014Z","shell.execute_reply":"2022-06-10T09:10:20.981813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ndf= trainx.sample(1000)\nfor col1,col2 in zip(numeric_cols,numeric_cols[1:]):\n    if col1.startswith(\"S\") and col2.startswith(\"R\"):\n        plt.figure(figsize=(5,5))\n        sb.jointplot(x=col1,y=col2,data=df,kind=\"kde\")\n        plt.show()\ndel df\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-10T09:10:20.984593Z","iopub.execute_input":"2022-06-10T09:10:20.985401Z","iopub.status.idle":"2022-06-10T09:10:26.198765Z","shell.execute_reply.started":"2022-06-10T09:10:20.985346Z","shell.execute_reply":"2022-06-10T09:10:26.197722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There are different approaches to detect outliers. Isolation Forest is one of them. I have used an alternative approach instead of using Isolation Forest for the large dataset.","metadata":{}},{"cell_type":"code","source":"#Detection of Outliers\nfrom sklearn.ensemble import IsolationForest\ndef detect_outliers(train_data,col):\n    cf = IsolationForest(random_state=224,n_jobs=-1).fit(np.array(train_data[col].to_list()).reshape(-1,1))\n    predict = cf.predict(np.array(train_data[col].to_list()).reshape(-1,1))\n    colors={1:\"blue\",-1:\"black\"}\n    df = pd.DataFrame({\"Colors\":predict})\n    plt.figure(figsize=(5,5))\n    train_data[col].plot(style='.',color=df[\"Colors\"].map(colors),alpha=0.6)\n    plt.title(f\"Outlier Detection of {col} \")\n    plt.show()\n    del df\n    gc.collect()\n    return predict","metadata":{"execution":{"iopub.status.busy":"2022-06-10T09:10:26.200142Z","iopub.execute_input":"2022-06-10T09:10:26.200626Z","iopub.status.idle":"2022-06-10T09:10:26.627418Z","shell.execute_reply.started":"2022-06-10T09:10:26.200577Z","shell.execute_reply":"2022-06-10T09:10:26.626341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Imputation\ndef data_impute(data,col):\n    q1 = data[col].quantile(0.25)\n    q3 = data[col].quantile(0.75)\n    IQR = q3-q1\n    rng = 3*IQR\n    data[col]=np.where(data[col] >= q3+rng,data[col].median(),data[col])\n    data[col]=np.where(data[col] <= q1-rng,data[col].median(),data[col])\n","metadata":{"execution":{"iopub.status.busy":"2022-06-10T09:10:26.631077Z","iopub.execute_input":"2022-06-10T09:10:26.631459Z","iopub.status.idle":"2022-06-10T09:10:26.638907Z","shell.execute_reply.started":"2022-06-10T09:10:26.631425Z","shell.execute_reply":"2022-06-10T09:10:26.637804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Impute Numerical Values:\nfor col in numeric_cols:\n    data_impute(trainx,col)","metadata":{"execution":{"iopub.status.busy":"2022-06-10T09:10:26.640879Z","iopub.execute_input":"2022-06-10T09:10:26.641399Z","iopub.status.idle":"2022-06-10T09:10:50.777874Z","shell.execute_reply.started":"2022-06-10T09:10:26.641350Z","shell.execute_reply":"2022-06-10T09:10:50.776724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Feature Engineering\ntrainx[\"Days\"]=train_x[\"S_2\"].dt.day\ntrainx[\"Month\"]=train_x[\"S_2\"].dt.month\ntrainx[\"Year\"]=train_x[\"S_2\"].dt.year\n\n","metadata":{"execution":{"iopub.status.busy":"2022-06-10T09:10:50.779627Z","iopub.execute_input":"2022-06-10T09:10:50.780128Z","iopub.status.idle":"2022-06-10T09:10:52.567410Z","shell.execute_reply.started":"2022-06-10T09:10:50.780084Z","shell.execute_reply":"2022-06-10T09:10:52.566303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"customer_id = [ i for i in trainx[\"customer_ID\"].to_list()]\ntrainx.drop(\"customer_ID\",axis=1,inplace=True)\ntrainx.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-06-10T09:10:52.569140Z","iopub.execute_input":"2022-06-10T09:10:52.569521Z","iopub.status.idle":"2022-06-10T09:10:53.857582Z","shell.execute_reply.started":"2022-06-10T09:10:52.569487Z","shell.execute_reply":"2022-06-10T09:10:53.856826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Scale Numerical Values\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.decomposition import PCA\nscaler = StandardScaler()\nnum_cols = list(set(trainx.columns)-set(discard))\nscaler.fit(trainx[num_cols])\ntrainx_scaled = scaler.transform(trainx[num_cols])\npca = PCA()\ncomp = pca.fit(trainx_scaled)\nplt.plot(np.cumsum(comp.explained_variance_ratio_))\nplt.grid(axis=\"both\")\nplt.xlabel(\"PRINCIPAL COMPONENTS\")\nplt.ylabel(\"VARIANCE\")\nsb.despine()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-10T09:10:53.859224Z","iopub.execute_input":"2022-06-10T09:10:53.859606Z","iopub.status.idle":"2022-06-10T09:11:40.134371Z","shell.execute_reply.started":"2022-06-10T09:10:53.859570Z","shell.execute_reply":"2022-06-10T09:11:40.132713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The first 60 components explains 80% of variation and the first 80 components explains more than 95% variation.\n\n\nWork is going on.Going to add more work related feature engineering. Thanks to **AMEX.**","metadata":{}}]}