{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-05T12:17:50.631375Z","iopub.execute_input":"2022-07-05T12:17:50.632003Z","iopub.status.idle":"2022-07-05T12:17:50.665563Z","shell.execute_reply.started":"2022-07-05T12:17:50.631911Z","shell.execute_reply":"2022-07-05T12:17:50.664790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nimport plotly.graph_objects as go\nimport gc","metadata":{"execution":{"iopub.status.busy":"2022-07-05T12:17:50.765264Z","iopub.execute_input":"2022-07-05T12:17:50.766320Z","iopub.status.idle":"2022-07-05T12:17:52.018511Z","shell.execute_reply.started":"2022-07-05T12:17:50.766282Z","shell.execute_reply":"2022-07-05T12:17:52.017450Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train=pd.read_parquet(\"../input/amex-data-integer-dtypes-parquet-format/train.parquet\")\n\n%time","metadata":{"execution":{"iopub.status.busy":"2022-07-05T12:17:52.020255Z","iopub.execute_input":"2022-07-05T12:17:52.020529Z","iopub.status.idle":"2022-07-05T12:18:10.786872Z","shell.execute_reply.started":"2022-07-05T12:17:52.020504Z","shell.execute_reply":"2022-07-05T12:18:10.786066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels=pd.read_csv(\"../input/amex-default-prediction/train_labels.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-05T12:18:10.787999Z","iopub.execute_input":"2022-07-05T12:18:10.788515Z","iopub.status.idle":"2022-07-05T12:18:11.722963Z","shell.execute_reply.started":"2022-07-05T12:18:10.788482Z","shell.execute_reply":"2022-07-05T12:18:11.722069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train=df_train.merge(labels,left_on='customer_ID',right_on='customer_ID')","metadata":{"execution":{"iopub.status.busy":"2022-07-05T12:18:11.724967Z","iopub.execute_input":"2022-07-05T12:18:11.725486Z","iopub.status.idle":"2022-07-05T12:19:38.541896Z","shell.execute_reply.started":"2022-07-05T12:18:11.725454Z","shell.execute_reply":"2022-07-05T12:19:38.540685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**#checking for null values **","metadata":{}},{"cell_type":"code","source":"null_val=df_train.isna().sum().sort_values(ascending=False)\n\nplt.title(\"Distribution of null values\")\nnull_val[null_val>0].plot(kind=\"hist\")","metadata":{"execution":{"iopub.status.busy":"2022-07-05T12:19:38.543169Z","iopub.execute_input":"2022-07-05T12:19:38.543516Z","iopub.status.idle":"2022-07-05T12:19:41.592975Z","shell.execute_reply.started":"2022-07-05T12:19:38.543485Z","shell.execute_reply":"2022-07-05T12:19:41.592262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T12:19:41.594040Z","iopub.execute_input":"2022-07-05T12:19:41.595061Z","iopub.status.idle":"2022-07-05T12:19:41.716086Z","shell.execute_reply.started":"2022-07-05T12:19:41.595019Z","shell.execute_reply":"2022-07-05T12:19:41.715319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(40,10))\nplt.title(\"Null value count\")\nplt.xlabel(\"Columns\")\nplt.ylabel(\"Count\")\nnull_val[null_val > 0 ].plot(kind=\"bar\");\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T12:19:41.717415Z","iopub.execute_input":"2022-07-05T12:19:41.718365Z","iopub.status.idle":"2022-07-05T12:19:42.614250Z","shell.execute_reply.started":"2022-07-05T12:19:41.718335Z","shell.execute_reply":"2022-07-05T12:19:42.613237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**DROPPING COLUMNS WITH LOSS>70%**","metadata":{}},{"cell_type":"code","source":"tmp=df_train.isna().sum().mul(100).div(len(df_train)).sort_values(ascending=False)\nmssing_df=pd.DataFrame(tmp).reset_index()\ndrop_colm=mssing_df[mssing_df[0]>70][\"index\"].values\nprint(drop_colm)\ndf_train.drop(columns=drop_colm,axis=1,inplace=True)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-05T12:19:42.615843Z","iopub.execute_input":"2022-07-05T12:19:42.616532Z","iopub.status.idle":"2022-07-05T12:19:48.672956Z","shell.execute_reply.started":"2022-07-05T12:19:42.616496Z","shell.execute_reply":"2022-07-05T12:19:48.672030Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> LETs check the imbalancing in target variable","metadata":{}},{"cell_type":"code","source":"sns.countplot(\n                df_train[\"target\"].values,\n              ).set_xlabel(\"Target\");\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T12:19:48.674144Z","iopub.execute_input":"2022-07-05T12:19:48.674433Z","iopub.status.idle":"2022-07-05T12:19:49.312529Z","shell.execute_reply.started":"2022-07-05T12:19:48.674407Z","shell.execute_reply":"2022-07-05T12:19:49.311274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Number of unique customer :\",len(df_train[\"customer_ID\"].unique()))\n#Lets Visualize data for any unqiue customer\ncust_id = np.random.choice(df_train[\"customer_ID\"])\npd.set_option('display.max_columns',200)\ndf_train[df_train[\"customer_ID\"] == cust_id]","metadata":{"execution":{"iopub.status.busy":"2022-07-05T12:19:49.317729Z","iopub.execute_input":"2022-07-05T12:19:49.318811Z","iopub.status.idle":"2022-07-05T12:19:51.168366Z","shell.execute_reply.started":"2022-07-05T12:19:49.318766Z","shell.execute_reply":"2022-07-05T12:19:51.167237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**check the data for each customer**","metadata":{}},{"cell_type":"code","source":"rand_customers=np.unique(df_train[\"customer_ID\"])[:100] #for 100 customer\nid_counts= df_train[df_train[\"customer_ID\"].isin(rand_customers)].groupby('customer_ID').agg(\"count\")\nplt.figure(figsize=(20,10))\n#no of times data for a customer is in the table\nid_counts[\"S_2\"].plot(kind=\"bar\")\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T12:19:51.169675Z","iopub.execute_input":"2022-07-05T12:19:51.170266Z","iopub.status.idle":"2022-07-05T12:20:01.668349Z","shell.execute_reply.started":"2022-07-05T12:19:51.170234Z","shell.execute_reply":"2022-07-05T12:20:01.667268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#function to return the columns and dataset of the given on the basis of types of variables\ndef Analyze(s):\n    var = [col for col in df_train.columns if col.startswith(s)]\n    dataframe=pd.DataFrame(data=df_train,columns=var)\n    for col in var:\n        print(\"{} has {} unique values\".format(col,df_train[col].nunique()))\n    return var, dataframe   #return list of variable and a dataframe\n       ","metadata":{"execution":{"iopub.status.busy":"2022-07-05T12:20:01.669881Z","iopub.execute_input":"2022-07-05T12:20:01.670277Z","iopub.status.idle":"2022-07-05T12:20:01.678079Z","shell.execute_reply.started":"2022-07-05T12:20:01.670240Z","shell.execute_reply":"2022-07-05T12:20:01.677137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#counting each type of variables\nvar_count={}\nfor col in df_train.columns:\n    if(col.startswith(\"S_\")):\n        var_count[\"Spend Variables\"]=var_count.get(\"Spend Variable\",0) +1\n    if(col.startswith(\"D_\")):\n        var_count[\"Deliquency variables\"]=var_count.get(\"Deliquency variables\",0) +1\n    if(col.startswith(\"B_\")):\n        var_count[\"Balance variables\"]=var_count.get(\"Balance variables\",0) +1\n    if(col.startswith(\"P_\")):\n        var_count[\"Payment variables\"]=var_count.get(\"Payment variables\",0) +1\n    if(col.startswith(\"R_\")):\n         var_count[\"Risk variables\"]=var_count.get(\"Risk variables\",0) +1\nplt.figure(figsize=(15,5))\nsns.barplot(x=list(var_count.keys()),y=list(var_count.values()))\n","metadata":{"execution":{"iopub.status.busy":"2022-07-05T12:20:01.679812Z","iopub.execute_input":"2022-07-05T12:20:01.680319Z","iopub.status.idle":"2022-07-05T12:20:01.868934Z","shell.execute_reply.started":"2022-07-05T12:20:01.680274Z","shell.execute_reply":"2022-07-05T12:20:01.867703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**LETS ANALYSE PAYMENT VARIABLES**","metadata":{}},{"cell_type":"code","source":"payment_vars = [col for col in df_train.columns if col.startswith(\"P_\")]\ncorr=df_train[payment_vars+[\"target\"]].corr()\nsns.heatmap(corr, annot=True, cmap=\"Purples\");","metadata":{"execution":{"iopub.status.busy":"2022-07-05T12:20:01.870375Z","iopub.execute_input":"2022-07-05T12:20:01.870698Z","iopub.status.idle":"2022-07-05T12:20:02.577420Z","shell.execute_reply.started":"2022-07-05T12:20:01.870669Z","shell.execute_reply":"2022-07-05T12:20:02.576572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes= plt.subplots(1,3,figsize=(20,5))\naxes= axes.ravel()  #returns 1 d array\nfor i,col in enumerate(payment_vars):\n    sns.histplot(data=df_train,x=col,hue=\"target\",ax=axes[i])\n\nfig.suptitle(\"Distribution of Payment Variables w.r.t target\")\nfig.tight_layout()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-05T12:20:02.579011Z","iopub.execute_input":"2022-07-05T12:20:02.579920Z","iopub.status.idle":"2022-07-05T12:20:22.829472Z","shell.execute_reply.started":"2022-07-05T12:20:02.579881Z","shell.execute_reply":"2022-07-05T12:20:22.828457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train[\"P_4\"].value_counts()\ndf_train[\"P_4\"]=df_train[\"P_4\"].apply(lambda x: 0 if x==0 else 1)\nsns.countplot(data=df_train, x=\"P_4\",hue=\"target\")\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T12:20:22.830594Z","iopub.execute_input":"2022-07-05T12:20:22.830916Z","iopub.status.idle":"2022-07-05T12:20:28.690401Z","shell.execute_reply.started":"2022-07-05T12:20:22.830889Z","shell.execute_reply":"2022-07-05T12:20:28.689402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Higher p_2 value lower the chances of default\n* target=1  is following normal distribution in p_2 and p_3\n* when p4=1 then there are 0.5 probality of a customer going default\n","metadata":{}},{"cell_type":"markdown","source":"**LETS ANALYZE FOR SPEND VARIABLES**","metadata":{}},{"cell_type":"code","source":"# Handling date column\ndf_train['S_2'] = pd.to_datetime(df_train['S_2'], errors='coerce')\ndf_train[\"S_2_day\"] =df_train[\"S_2\"].dt.day\ndf_train[\"S_2_month\"] = df_train[\"S_2\"].dt.month\ndf_train[\"S_2_year\"] = df_train[\"S_2\"].dt.year\ndf_train.drop(columns=[\"S_2\"],axis=1,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T12:20:28.691539Z","iopub.execute_input":"2022-07-05T12:20:28.691765Z","iopub.status.idle":"2022-07-05T12:20:32.331965Z","shell.execute_reply.started":"2022-07-05T12:20:28.691743Z","shell.execute_reply":"2022-07-05T12:20:32.330939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Spend_vars,Spend_df= Analyze(\"S_\")\n","metadata":{"execution":{"iopub.status.busy":"2022-07-05T12:20:32.333306Z","iopub.execute_input":"2022-07-05T12:20:32.333709Z","iopub.status.idle":"2022-07-05T12:20:38.889589Z","shell.execute_reply.started":"2022-07-05T12:20:32.333678Z","shell.execute_reply":"2022-07-05T12:20:38.888592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corr = Spend_df.corrwith(df_train[\"target\"], axis=0)\nval = [str(round(v ,2) *100) + '%' for v in corr.values]\n\nfig = go.Figure()\nfig.add_trace(go.Bar(y=corr.index, x= corr.values,\n                     orientation='h',\n                     marker_color = '#9900cc',\n                     text = val,\n                     textposition = 'outside',\n                     textfont_color = '#ffff80'))\nfig.update_layout(template = 'plotly_dark',\n                  title = \"Spend_Variables Correlation with Target\",\n                  width = 800,\n                  height = 3000)\nfig.update_xaxes(range=[-2,2])","metadata":{"execution":{"iopub.status.busy":"2022-07-05T12:20:38.890979Z","iopub.execute_input":"2022-07-05T12:20:38.891460Z","iopub.status.idle":"2022-07-05T12:20:41.666752Z","shell.execute_reply.started":"2022-07-05T12:20:38.891416Z","shell.execute_reply":"2022-07-05T12:20:41.665730Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#checking correlation below threshold and returning the columnn names\nl=[]\nfor i ,j  in enumerate(corr):\n    if(abs(j)<0.08):  #threshold=0.08 \n        print(Spend_vars[i],j)\n        l.append(Spend_vars[i])\n","metadata":{"execution":{"iopub.status.busy":"2022-07-05T12:20:41.667881Z","iopub.execute_input":"2022-07-05T12:20:41.668149Z","iopub.status.idle":"2022-07-05T12:20:41.675219Z","shell.execute_reply.started":"2022-07-05T12:20:41.668124Z","shell.execute_reply":"2022-07-05T12:20:41.674241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#correlation with target: \n\"\"\"S12=0.07 ,s_19=0.009, s_16=0.03,s_17=0.03,s_18,s_27=-0.02 s_26=-0.04,s_23=0.05,s_5=0.04,S_9=0.07\"\"\"","metadata":{}},{"cell_type":"code","source":"#dropping column below threshold\ndf_train.drop(columns=l,axis=1,inplace=True)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-05T12:20:41.676775Z","iopub.execute_input":"2022-07-05T12:20:41.677461Z","iopub.status.idle":"2022-07-05T12:20:42.842294Z","shell.execute_reply.started":"2022-07-05T12:20:41.677422Z","shell.execute_reply":"2022-07-05T12:20:42.841499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"****Checking Categorical Columns  and numerical columns**\n","metadata":{}},{"cell_type":"markdown","source":"**categorical columns=['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68'] #given with the dataset**","metadata":{}},{"cell_type":"code","source":"cat=['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\ncols = df_train.columns\nnum_cols = df_train._get_numeric_data().columns\nprint(num_cols)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-05T12:20:42.843668Z","iopub.execute_input":"2022-07-05T12:20:42.844176Z","iopub.status.idle":"2022-07-05T12:20:42.851029Z","shell.execute_reply.started":"2022-07-05T12:20:42.844147Z","shell.execute_reply":"2022-07-05T12:20:42.849990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"numerical_columns = list(set(cols) - set(cat))\nfiltered_numerical_columns = list(set(df_train[numerical_columns])-{\"S_2\",\"customer_ID\"})\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T12:20:42.852227Z","iopub.execute_input":"2022-07-05T12:20:42.852806Z","iopub.status.idle":"2022-07-05T12:20:44.155187Z","shell.execute_reply.started":"2022-07-05T12:20:42.852766Z","shell.execute_reply":"2022-07-05T12:20:44.154183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(filtered_numerical_columns)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-05T12:20:44.156516Z","iopub.execute_input":"2022-07-05T12:20:44.156792Z","iopub.status.idle":"2022-07-05T12:20:44.163566Z","shell.execute_reply.started":"2022-07-05T12:20:44.156767Z","shell.execute_reply":"2022-07-05T12:20:44.162426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in cat:\n    print(i + \" Attribute is  of Data Type : \"+ str(df_train[i].dtypes))\n#converting into object\nfor i in cat:\n    df_train[i] = df_train[i].astype(\"object\")\n    print(i + \" Attribute is  of Data Type : \"+ str(df_train[i].dtypes))","metadata":{"execution":{"iopub.status.busy":"2022-07-05T12:20:44.165041Z","iopub.execute_input":"2022-07-05T12:20:44.165331Z","iopub.status.idle":"2022-07-05T12:20:47.411331Z","shell.execute_reply.started":"2022-07-05T12:20:44.165304Z","shell.execute_reply":"2022-07-05T12:20:47.410121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train[cat].nunique()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T12:20:47.412608Z","iopub.execute_input":"2022-07-05T12:20:47.412997Z","iopub.status.idle":"2022-07-05T12:20:53.674129Z","shell.execute_reply.started":"2022-07-05T12:20:47.412966Z","shell.execute_reply":"2022-07-05T12:20:53.673062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(20,20))\nfor i,feature in enumerate(cat):\n    plt.subplot(4,3,i+1)\n    sns.countplot(x=df_train[feature],hue=df_train['target'])\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T12:20:53.675092Z","iopub.execute_input":"2022-07-05T12:20:53.675371Z","iopub.status.idle":"2022-07-05T12:23:59.011903Z","shell.execute_reply.started":"2022-07-05T12:20:53.675346Z","shell.execute_reply":"2022-07-05T12:23:59.011135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# For numeric columns filling null values\nfiltered_numerical_columns = df_train.select_dtypes(np.number).columns\ndf_train[filtered_numerical_columns] = df_train[filtered_numerical_columns].fillna(df_train[filtered_numerical_columns].mean())","metadata":{"execution":{"iopub.status.busy":"2022-07-05T12:23:59.018558Z","iopub.execute_input":"2022-07-05T12:23:59.019525Z","iopub.status.idle":"2022-07-05T12:24:08.254710Z","shell.execute_reply.started":"2022-07-05T12:23:59.019489Z","shell.execute_reply":"2022-07-05T12:24:08.253457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.isnull().sum()\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T12:24:08.255923Z","iopub.execute_input":"2022-07-05T12:24:08.256281Z","iopub.status.idle":"2022-07-05T12:24:16.679340Z","shell.execute_reply.started":"2022-07-05T12:24:08.256248Z","shell.execute_reply":"2022-07-05T12:24:16.678153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Performing the Feature Encoding\nMachine learning models can only work with numerical values. For this reason, it is necessary to transform the categorical values of the relevant features into numerical ones. This process is called feature encoding.","metadata":{}},{"cell_type":"code","source":"for col in cat:\n    print('{} has {} categories'.format(col,df_train[col].nunique()))","metadata":{"execution":{"iopub.status.busy":"2022-07-05T12:24:16.680640Z","iopub.execute_input":"2022-07-05T12:24:16.680936Z","iopub.status.idle":"2022-07-05T12:24:20.756037Z","shell.execute_reply.started":"2022-07-05T12:24:16.680909Z","shell.execute_reply":"2022-07-05T12:24:20.754950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Visualize data for a random customer\n\"\"\"\nprint(\"Number of unique customer :\",len(df_train[\"customer_ID\"].unique()))\n#Lets Visualize data for any unqiue customer\ncust_id = np.random.choice(df_train[\"customer_ID\"])\npd.set_option('display.max_columns',200)\ndf_train[df_train[\"customer_ID\"] == cust_id]\ngc.collect()\"\"\"","metadata":{"execution":{"iopub.status.busy":"2022-07-05T12:24:20.757322Z","iopub.execute_input":"2022-07-05T12:24:20.757726Z","iopub.status.idle":"2022-07-05T12:24:20.763997Z","shell.execute_reply.started":"2022-07-05T12:24:20.757697Z","shell.execute_reply":"2022-07-05T12:24:20.762991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nle=LabelEncoder()\nfor col in cat:\n    df_train[col]=le.fit_transform(df_train[col])\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T12:24:20.765216Z","iopub.execute_input":"2022-07-05T12:24:20.765508Z","iopub.status.idle":"2022-07-05T12:24:43.011788Z","shell.execute_reply.started":"2022-07-05T12:24:20.765481Z","shell.execute_reply":"2022-07-05T12:24:43.010668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**LETS ANALYSE RISK VARIABLES**","metadata":{}},{"cell_type":"code","source":"Risk_var,Risk_df=Analyze(\"R_\")\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T12:24:43.013085Z","iopub.execute_input":"2022-07-05T12:24:43.013431Z","iopub.status.idle":"2022-07-05T12:24:46.326692Z","shell.execute_reply.started":"2022-07-05T12:24:43.013401Z","shell.execute_reply":"2022-07-05T12:24:46.325620Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" Correlation Analysis With Target","metadata":{}},{"cell_type":"code","source":"corr = Risk_df.corrwith(df_train[\"target\"], axis=0)\nval = [str(round(v ,2) *100) + '%' for v in corr.values]\n\nfig = go.Figure()\nfig.add_trace(go.Bar(y=corr.index, x= corr.values,\n                     orientation='h',\n                     marker_color = '#9900cc',\n                     text = val,\n                     textposition = 'outside',\n                     textfont_color = '#ffff80'))\nfig.update_layout(template = 'plotly_dark',\n                  title = \"Risk_Variables Correlation with Target\",\n                  width = 800,\n                  height = 3000)\nfig.update_xaxes(range=[-2,2])\n","metadata":{"execution":{"iopub.status.busy":"2022-07-05T12:24:46.328016Z","iopub.execute_input":"2022-07-05T12:24:46.328324Z","iopub.status.idle":"2022-07-05T12:24:49.105368Z","shell.execute_reply.started":"2022-07-05T12:24:46.328297Z","shell.execute_reply":"2022-07-05T12:24:49.104250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T12:24:49.106903Z","iopub.execute_input":"2022-07-05T12:24:49.107903Z","iopub.status.idle":"2022-07-05T12:24:49.332813Z","shell.execute_reply.started":"2022-07-05T12:24:49.107852Z","shell.execute_reply":"2022-07-05T12:24:49.331945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"1. R_18 has ~ 0 correlation with the target variable\n2. R_23 has ~ 0.01\n3. R_14 has ~ 0.03 correlation\nLETS DROP THESE COLUMN FOR A WHILE ","metadata":{}},{"cell_type":"code","source":"l=[]\nfor i ,j  in enumerate(corr):\n    if(abs(j)<0.05):\n        l.append(Risk_var[i])\n\ndf_train.drop(l,axis=1,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T12:24:49.334300Z","iopub.execute_input":"2022-07-05T12:24:49.334593Z","iopub.status.idle":"2022-07-05T12:24:50.641920Z","shell.execute_reply.started":"2022-07-05T12:24:49.334566Z","shell.execute_reply":"2022-07-05T12:24:50.640947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-05T12:24:50.643345Z","iopub.execute_input":"2022-07-05T12:24:50.644335Z","iopub.status.idle":"2022-07-05T12:24:50.651803Z","shell.execute_reply.started":"2022-07-05T12:24:50.644287Z","shell.execute_reply":"2022-07-05T12:24:50.650625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**LETS ANALYSE DELIQUENCY VARIABLES**","metadata":{}},{"cell_type":"code","source":"Deli_var ,Deli_df=Analyze(\"D_\")\nprint(Deli_var)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T12:24:50.653362Z","iopub.execute_input":"2022-07-05T12:24:50.653678Z","iopub.status.idle":"2022-07-05T12:25:11.167272Z","shell.execute_reply.started":"2022-07-05T12:24:50.653640Z","shell.execute_reply":"2022-07-05T12:25:11.165551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"d_corr = Deli_df.corrwith(df_train[\"target\"], axis=0)\nval = [str(round(v ,2) *100) + '%' for v in d_corr.values]\n\nfig = go.Figure()\nfig.add_trace(go.Bar(y=d_corr.index, x= d_corr.values,\n                     orientation='h',\n                     marker_color = '#9900cc',\n                     text = val,\n                     textposition = 'outside',\n                     textfont_color = '#ffff80'))\nfig.update_layout(template = 'plotly_dark',\n                  title = \"Deliquncy Correlation with Target\",\n                  width = 800,\n                  height = 3000)\nfig.update_xaxes(range=[-2,2])\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-05T12:25:11.169240Z","iopub.execute_input":"2022-07-05T12:25:11.169656Z","iopub.status.idle":"2022-07-05T12:25:19.393022Z","shell.execute_reply.started":"2022-07-05T12:25:11.169614Z","shell.execute_reply":"2022-07-05T12:25:19.392219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i ,j  in enumerate(d_corr):\n    if(abs(j)<0.05):\n        print(Deli_var[i])","metadata":{"execution":{"iopub.status.busy":"2022-07-05T12:25:19.394259Z","iopub.execute_input":"2022-07-05T12:25:19.395143Z","iopub.status.idle":"2022-07-05T12:25:19.399898Z","shell.execute_reply.started":"2022-07-05T12:25:19.395112Z","shell.execute_reply":"2022-07-05T12:25:19.399221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"l=[]\nfor i ,j  in enumerate(d_corr):\n    if(abs(j)<0.05):\n        l.append(Deli_var[i])","metadata":{"execution":{"iopub.status.busy":"2022-07-05T12:25:19.401013Z","iopub.execute_input":"2022-07-05T12:25:19.401908Z","iopub.status.idle":"2022-07-05T12:25:19.412521Z","shell.execute_reply.started":"2022-07-05T12:25:19.401877Z","shell.execute_reply":"2022-07-05T12:25:19.411683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.drop(columns=l,axis=1,inplace=True)\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T12:25:19.413918Z","iopub.execute_input":"2022-07-05T12:25:19.414475Z","iopub.status.idle":"2022-07-05T12:25:21.687772Z","shell.execute_reply.started":"2022-07-05T12:25:19.414445Z","shell.execute_reply":"2022-07-05T12:25:21.686656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-05T12:25:21.689656Z","iopub.execute_input":"2022-07-05T12:25:21.690004Z","iopub.status.idle":"2022-07-05T12:25:21.696808Z","shell.execute_reply.started":"2022-07-05T12:25:21.689974Z","shell.execute_reply":"2022-07-05T12:25:21.695763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**LETS ANALYSE Balance VARIABLES**","metadata":{}},{"cell_type":"code","source":"Bal_var,Bal_df = Analyze(\"B_\")","metadata":{"execution":{"iopub.status.busy":"2022-07-05T12:25:21.698574Z","iopub.execute_input":"2022-07-05T12:25:21.699510Z","iopub.status.idle":"2022-07-05T12:25:42.164752Z","shell.execute_reply.started":"2022-07-05T12:25:21.699463Z","shell.execute_reply":"2022-07-05T12:25:42.163479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corr = Bal_df.corrwith(df_train[\"target\"], axis=0)\nval = [str(round(v ,2) *100) + '%' for v in corr.values]\n\nfig = go.Figure()\nfig.add_trace(go.Bar(y=corr.index, x= corr.values,\n                     orientation='h',\n                     marker_color = '#9900cc',\n                     text = val,\n                     textposition = 'outside',\n                     textfont_color = '#ffff80'))\nfig.update_layout(template = 'plotly_dark',\n                  title = \"Balance variables Correlation with Target\",\n                  width = 800,\n                  height = 3000)\nfig.update_xaxes(range=[-2,2])","metadata":{"execution":{"iopub.status.busy":"2022-07-05T12:25:42.167276Z","iopub.execute_input":"2022-07-05T12:25:42.168407Z","iopub.status.idle":"2022-07-05T12:25:45.965285Z","shell.execute_reply.started":"2022-07-05T12:25:42.168358Z","shell.execute_reply":"2022-07-05T12:25:45.964246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"l=[]\nfor i ,j  in enumerate(corr):\n    if(abs(j)<0.05):\n        l.append(Bal_var[i])\nprint(*l)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T12:25:45.966456Z","iopub.execute_input":"2022-07-05T12:25:45.966765Z","iopub.status.idle":"2022-07-05T12:25:45.974070Z","shell.execute_reply.started":"2022-07-05T12:25:45.966738Z","shell.execute_reply":"2022-07-05T12:25:45.972367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.drop(l,axis=1,inplace=True)\ndf_train.shape\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T12:25:45.975536Z","iopub.execute_input":"2022-07-05T12:25:45.976192Z","iopub.status.idle":"2022-07-05T12:25:47.450258Z","shell.execute_reply.started":"2022-07-05T12:25:45.976158Z","shell.execute_reply":"2022-07-05T12:25:47.449243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T12:29:00.445746Z","iopub.execute_input":"2022-07-05T12:29:00.446128Z","iopub.status.idle":"2022-07-05T12:29:00.535390Z","shell.execute_reply.started":"2022-07-05T12:29:00.446096Z","shell.execute_reply":"2022-07-05T12:29:00.534285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(\n                df_train[\"target\"].values,\n              ).set_xlabel(\"Target\");\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T12:30:59.156314Z","iopub.execute_input":"2022-07-05T12:30:59.156738Z","iopub.status.idle":"2022-07-05T12:31:00.196655Z","shell.execute_reply.started":"2022-07-05T12:30:59.156703Z","shell.execute_reply":"2022-07-05T12:31:00.195486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2022-07-05T12:47:21.955807Z","iopub.execute_input":"2022-07-05T12:47:21.956704Z","iopub.status.idle":"2022-07-05T12:47:31.545158Z","shell.execute_reply.started":"2022-07-05T12:47:21.956661Z","shell.execute_reply":"2022-07-05T12:47:31.543942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2022-07-05T12:49:43.169015Z","iopub.execute_input":"2022-07-05T12:49:43.169368Z","iopub.status.idle":"2022-07-05T12:49:45.349228Z","shell.execute_reply.started":"2022-07-05T12:49:43.169339Z","shell.execute_reply":"2022-07-05T12:49:45.348301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2022-07-05T12:49:40.177190Z","iopub.execute_input":"2022-07-05T12:49:40.177653Z","iopub.status.idle":"2022-07-05T12:49:41.943969Z","shell.execute_reply.started":"2022-07-05T12:49:40.177614Z","shell.execute_reply":"2022-07-05T12:49:41.943060Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}