{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-03T01:35:51.692240Z","iopub.execute_input":"2022-08-03T01:35:51.693243Z","iopub.status.idle":"2022-08-03T01:35:51.746662Z","shell.execute_reply.started":"2022-08-03T01:35:51.693057Z","shell.execute_reply":"2022-08-03T01:35:51.745148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore')\n\ndf = pd.read_feather('../input/amexfeather/train_data.ftr')\n\n#df.info(verbose=True,show_counts=True)\n\ndf_train =  (df\n            .groupby('customer_ID')\n            .tail(1))\n\n# Identifying label data from the whole datset and visualising label counts and display % of data as annotations\nimport seaborn as sn\nimport matplotlib.pyplot as plt\nplt.figure(figsize=(5,5))\n\nax = sn.countplot(x=\"target\", data=df_train)\nfor p in ax.patches:#displaying % as annotations\n        ax.annotate(str(round(100*p.get_height()/len(df_train),2))+\"%\", (p.get_x()+0.3, p.get_height()+2))\n\nprint(df_train.target.value_counts())\nprint(\"\\nTotal Number:\",sum(df_train.target.value_counts()))  ","metadata":{"execution":{"iopub.status.busy":"2022-08-03T01:35:51.749254Z","iopub.execute_input":"2022-08-03T01:35:51.750180Z","iopub.status.idle":"2022-08-03T01:36:19.961558Z","shell.execute_reply.started":"2022-08-03T01:35:51.750124Z","shell.execute_reply":"2022-08-03T01:36:19.960278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#df_train.info(verbose=True,show_counts=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T01:36:19.965478Z","iopub.execute_input":"2022-08-03T01:36:19.965899Z","iopub.status.idle":"2022-08-03T01:36:19.971405Z","shell.execute_reply.started":"2022-08-03T01:36:19.965861Z","shell.execute_reply":"2022-08-03T01:36:19.969864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_f = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68'] \nall_f = list(df_train.columns)\nall_f.remove(\"customer_ID\")\nall_f.remove(\"S_2\")\nall_f.remove(\"target\")\n\nnum_f = list(set(all_f) - set(cat_f))\nnum_f.append(\"target\")\ndfnum = df_train[num_f]\n\n#print(dfnum_train.info(verbose=True,show_counts=True))\n\ndfnum","metadata":{"execution":{"iopub.status.busy":"2022-08-03T01:36:19.975256Z","iopub.execute_input":"2022-08-03T01:36:19.976654Z","iopub.status.idle":"2022-08-03T01:36:20.413843Z","shell.execute_reply.started":"2022-08-03T01:36:19.976553Z","shell.execute_reply":"2022-08-03T01:36:20.412596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#drop columns with less than 20 % data\nperc = 20.0 # Like N %\nmin_count =  int(((100-perc)/100)*dfnum.shape[0] + 1)\ndfnum = dfnum.dropna( axis=1, \n                thresh=min_count)\nall_f = list(dfnum.columns)\nnum_f = list(set(all_f) - set(cat_f))","metadata":{"execution":{"iopub.status.busy":"2022-08-03T01:36:20.415677Z","iopub.execute_input":"2022-08-03T01:36:20.416927Z","iopub.status.idle":"2022-08-03T01:36:21.143909Z","shell.execute_reply.started":"2022-08-03T01:36:20.416872Z","shell.execute_reply":"2022-08-03T01:36:21.142781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfnum = dfnum[num_f]\ndfnum.target","metadata":{"execution":{"iopub.status.busy":"2022-08-03T01:36:21.145459Z","iopub.execute_input":"2022-08-03T01:36:21.145834Z","iopub.status.idle":"2022-08-03T01:36:21.433977Z","shell.execute_reply.started":"2022-08-03T01:36:21.145800Z","shell.execute_reply":"2022-08-03T01:36:21.432419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = dfnum.dropna()\n\ntest = test.reset_index()\ntest = test.drop(\"index\",axis=1)\ntest","metadata":{"execution":{"iopub.status.busy":"2022-08-03T01:36:21.436230Z","iopub.execute_input":"2022-08-03T01:36:21.436781Z","iopub.status.idle":"2022-08-03T01:36:22.462441Z","shell.execute_reply.started":"2022-08-03T01:36:21.436731Z","shell.execute_reply":"2022-08-03T01:36:22.461044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in num_f:\n    dfnum[col] = dfnum[col].fillna(test[col].median())\n    \n\n\nY = dfnum.target\nnum_f.remove(\"target\")\nX= dfnum[num_f]\n\n\n\n## Importing resample from *sklearn.utils* package.\nfrom sklearn.utils import resample\n\n# Separate the case of yes-subscribes and no-subscribes\ndfnum_0 = dfnum[dfnum.target == 0]\ndfnum_1 = dfnum[dfnum.target == 1]\n\n##Upsample the yes-subscribed cases.\ndfnum_minority_upsampled = resample(dfnum_1, \n                                 replace=True,     # sample with replacement\n                                 n_samples=300000) \n\n# Combine majority class with upsampled minority class\nnew_dfnum = pd.concat([dfnum_0, dfnum_minority_upsampled])\nnew_dfnum\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-03T01:36:22.463990Z","iopub.execute_input":"2022-08-03T01:36:22.464355Z","iopub.status.idle":"2022-08-03T01:36:26.295057Z","shell.execute_reply.started":"2022-08-03T01:36:22.464321Z","shell.execute_reply":"2022-08-03T01:36:26.293915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.utils import shuffle\nnew_dfnum = shuffle(new_dfnum)\nnew_dfnum","metadata":{"execution":{"iopub.status.busy":"2022-08-03T01:36:26.296524Z","iopub.execute_input":"2022-08-03T01:36:26.296967Z","iopub.status.idle":"2022-08-03T01:36:28.146025Z","shell.execute_reply.started":"2022-08-03T01:36:26.296934Z","shell.execute_reply":"2022-08-03T01:36:28.144828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\nplt.figure(figsize=(5,5))\n\nax = sn.countplot(x=\"target\", data=new_dfnum)\nfor p in ax.patches:#displaying % as annotations\n        ax.annotate(str(round(100*p.get_height()/len(new_dfnum),2))+\"%\", (p.get_x()+0.3, p.get_height()+2))\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-03T01:36:28.149528Z","iopub.execute_input":"2022-08-03T01:36:28.150002Z","iopub.status.idle":"2022-08-03T01:36:28.452933Z","shell.execute_reply.started":"2022-08-03T01:36:28.149966Z","shell.execute_reply":"2022-08-03T01:36:28.451565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\nY = new_dfnum.target\n#num_f.remove(\"target\")\nX =new_dfnum[num_f]\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-03T01:36:28.454903Z","iopub.execute_input":"2022-08-03T01:36:28.455390Z","iopub.status.idle":"2022-08-03T01:36:28.835209Z","shell.execute_reply.started":"2022-08-03T01:36:28.455345Z","shell.execute_reply":"2022-08-03T01:36:28.833857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\ntrain_X, test_X, train_y, test_y = train_test_split( X,\n                                                    Y,\n                                                    test_size = 0.3,\n                                                    random_state = 42 )","metadata":{"execution":{"iopub.status.busy":"2022-08-03T01:36:28.839541Z","iopub.execute_input":"2022-08-03T01:36:28.840013Z","iopub.status.idle":"2022-08-03T01:36:30.317067Z","shell.execute_reply.started":"2022-08-03T01:36:28.839979Z","shell.execute_reply":"2022-08-03T01:36:30.315732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Importing Random Forest Classifier from the sklearn.ensemble\nfrom sklearn.ensemble import RandomForestClassifier\n\n## Initializing the Random Forest Classifier with max_dept and n_estimators\nradm_clf = RandomForestClassifier( max_depth=10, n_estimators=10)\nradm_clf.fit( train_X, train_y )","metadata":{"execution":{"iopub.status.busy":"2022-08-03T01:36:30.318594Z","iopub.execute_input":"2022-08-03T01:36:30.319651Z","iopub.status.idle":"2022-08-03T01:37:32.754984Z","shell.execute_reply.started":"2022-08-03T01:36:30.319581Z","shell.execute_reply":"2022-08-03T01:37:32.753580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a dataframe to store the featues and their corresponding importances\nfeature_rank = pd.DataFrame( { 'feature': train_X.columns,\n                               'importance': radm_clf.feature_importances_ } )\n\n## Sorting the features based on their importances with most important feature at top.\nfeature_rank = feature_rank.sort_values('importance', ascending = False)\n\nplt.figure(figsize=(8, 40))\n# plot the values\nsn.barplot( y = 'feature', x = 'importance', data = feature_rank );","metadata":{"execution":{"iopub.status.busy":"2022-08-03T01:37:32.756677Z","iopub.execute_input":"2022-08-03T01:37:32.757516Z","iopub.status.idle":"2022-08-03T01:37:34.683275Z","shell.execute_reply.started":"2022-08-03T01:37:32.757460Z","shell.execute_reply":"2022-08-03T01:37:34.682055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_rank['cumsum'] = feature_rank.importance.cumsum() * 100\nfeature_df = feature_rank.head(43)\nfeature_df","metadata":{"execution":{"iopub.status.busy":"2022-08-03T01:41:41.120108Z","iopub.execute_input":"2022-08-03T01:41:41.120528Z","iopub.status.idle":"2022-08-03T01:41:41.138888Z","shell.execute_reply.started":"2022-08-03T01:41:41.120493Z","shell.execute_reply":"2022-08-03T01:41:41.137954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"list(feature_df.feature)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T01:41:45.654781Z","iopub.execute_input":"2022-08-03T01:41:45.655672Z","iopub.status.idle":"2022-08-03T01:41:45.689228Z","shell.execute_reply.started":"2022-08-03T01:41:45.655504Z","shell.execute_reply":"2022-08-03T01:41:45.682414Z"},"trusted":true},"execution_count":null,"outputs":[]}]}