{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"From previous submission with XGB model observed that accuracy improved\n\nUndersampling of majority class,over sampling of minority class has improved the recall precision and accuracy scores.But had no effect on the amex_metric.For the test data amex metric has infact reduced from .52 to .47\n\nIn view of this current notebook is created to check whether the metric would improve by considering only the important features.\n\nCheck if the the chunked data would give the same important features  for RandomForest,GradientBoost\n\nStep 1: load multiple data.csv files\n","metadata":{}},{"cell_type":"code","source":"#-------------------------------------------------------------------------------------------------------------------------------\nimport pandas as pd                                                 # Importing for panel data analysis\n#from pandas_profiling import ProfileReport                          # Import Pandas Profiling (To generate Univariate Analysis)\n\npd.set_option('display.max_columns', None)                          # Unfolding hidden features if the cardinality is high      \npd.set_option('display.max_colwidth', None)                         # Unfolding the max feature width for better clearity      \npd.set_option('display.max_rows', None)                             # Unfolding hidden data points if the cardinality is high\npd.set_option('mode.chained_assignment', None)                      # Removing restriction over chained assignments operations\npd.set_option('display.float_format', lambda x: '%.5f' % x)         # To suppress scientific notation over exponential values\n#-------------------------------------------------------------------------------------------------------------------------------\nimport numpy as np                                                  # Importing package numpys (For Numerical Python)\n#-------------------------------------------------------------------------------------------------------------------------------\nimport matplotlib.pyplot as plt                                     # Importing pyplot interface using matplotlib\nfrom matplotlib.pylab import rcParams                               # Backend used for rendering and GUI integration                                               \nimport seaborn as sns                                               # Importin seaborm library for interactive visualization\n%matplotlib inline\n#-------------------------------------------------------------------------------------------------------------------------------\nfrom sklearn.metrics import accuracy_score                          # For calculating the accuracy for the model\nfrom sklearn.metrics import precision_score                         # For calculating the Precision of the model\nfrom sklearn.metrics import recall_score                            # For calculating the recall of the model\nfrom sklearn.metrics import precision_recall_curve                  # For precision and recall metric estimation\nfrom sklearn.metrics import confusion_matrix                        # For verifying model performance using confusion matrix\nfrom sklearn.metrics import f1_score                                # For Checking the F1-Score of our model  \nfrom sklearn.metrics import roc_curve                               # For Roc-Auc metric estimation\n#-------------------------------------------------------------------------------------------------------------------------------\nfrom sklearn.model_selection import train_test_split                # To split the data in training and testing part     \nfrom sklearn.linear_model import LogisticRegression                 # To create the Logistic Regression Model\nfrom sklearn.ensemble import RandomForestClassifier, GradientBoostingClassifier\nfrom sklearn.metrics import confusion_matrix,classification_report\nfrom sklearn.model_selection import cross_val_score\n#-------------------------------------------------------------------------------------------------------------------------------\nimport warnings                                                     # Importing warning to disable runtime warnings\nwarnings.filterwarnings(\"ignore\")                                   # Warnings will appear only once\n","metadata":{"execution":{"iopub.status.busy":"2022-07-23T14:23:31.821777Z","iopub.execute_input":"2022-07-23T14:23:31.822220Z","iopub.status.idle":"2022-07-23T14:23:31.841979Z","shell.execute_reply.started":"2022-07-23T14:23:31.822184Z","shell.execute_reply":"2022-07-23T14:23:31.840846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\n\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"headers = [*pd.read_csv('../input/amex-default-prediction/train_data.csv', nrows=1)]\nprint(headers)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T14:28:22.422709Z","iopub.execute_input":"2022-07-23T14:28:22.423109Z","iopub.status.idle":"2022-07-23T14:28:22.453836Z","shell.execute_reply.started":"2022-07-23T14:28:22.423075Z","shell.execute_reply":"2022-07-23T14:28:22.453073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del_col=['B_3', 'D_48', 'B_8', 'B_11', 'R_4', 'S_7', 'D_55', 'B_13', 'D_58', 'D_61', 'B_15', 'B_16', 'B_18', 'B_19', 'B_20', 'B_22', 'S_15', 'B_23', 'D_74', 'D_75', 'D_77', 'R_8', 'S_16', 'B_28', 'B_30', 'R_21', 'B_33', 'S_24', 'D_103', 'D_104', 'D_107', 'B_37', 'B_38', 'D_113', 'D_118', 'D_119', 'D_121', 'D_129', 'D_131', 'D_133', 'D_141', 'D_143','D_42','D_49','D_50','D_53','D_56','S_9','B_17','D_66','D_73','D_76','D_77','R_9','D_82','B_29','D_87','D_88',\n 'D_105','D_106','R_26','D_108','D_110', 'D_111', 'B_39', 'B_42', 'D_132','D_134','D_135','D_136','D_137','D_138','D_142']\n","metadata":{"execution":{"iopub.status.busy":"2022-07-23T14:28:26.627905Z","iopub.execute_input":"2022-07-23T14:28:26.629089Z","iopub.status.idle":"2022-07-23T14:28:26.635946Z","shell.execute_reply.started":"2022-07-23T14:28:26.629047Z","shell.execute_reply":"2022-07-23T14:28:26.635139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_col=list ((set(headers) - set (del_col)))","metadata":{"execution":{"iopub.status.busy":"2022-07-23T14:28:29.484946Z","iopub.execute_input":"2022-07-23T14:28:29.485322Z","iopub.status.idle":"2022-07-23T14:28:29.490337Z","shell.execute_reply.started":"2022-07-23T14:28:29.485292Z","shell.execute_reply":"2022-07-23T14:28:29.489080Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.feature_selection import SelectFromModel","metadata":{"execution":{"iopub.status.busy":"2022-07-23T14:28:35.639647Z","iopub.execute_input":"2022-07-23T14:28:35.640050Z","iopub.status.idle":"2022-07-23T14:28:35.645131Z","shell.execute_reply.started":"2022-07-23T14:28:35.640019Z","shell.execute_reply":"2022-07-23T14:28:35.644137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def find_features_rf(df_model,file_name,df_tar):\n    \n    df_model.drop('target',axis=1,inplace=True)\n    y=df_tar.target\n    X=df_model\n    feat_labels=X.columns\n    X_train, X_test, y_train, y_test = train_test_split(X, y, test_size = 0.3, random_state = 42, stratify = y)\n\n    print('Training Data Shape:', X_train.shape, y_train.shape)\n    print('Testing Data Shape:', X_test.shape, y_test.shape)\n    RC = RandomForestClassifier(n_estimators=10,class_weight='balanced',verbose=True)\n    RC.fit(X_train,y_train)\n    for feature in zip(feat_labels, RC.feature_importances_):\n        print(feature)\n      \n\n    sfm = SelectFromModel(RC, threshold=0.01)\n\n# Train the selector\n    sfm.fit(X_train, y_train)  \n    for feature_list_index in sfm.get_support(indices=True):\n        print(feat_labels[feature_list_index])","metadata":{"execution":{"iopub.status.busy":"2022-07-23T14:28:37.921747Z","iopub.execute_input":"2022-07-23T14:28:37.922680Z","iopub.status.idle":"2022-07-23T14:28:37.930867Z","shell.execute_reply.started":"2022-07-23T14:28:37.922644Z","shell.execute_reply":"2022-07-23T14:28:37.929743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def data_pre(df_passed,file_name):\n    print(file_name)\n    print('///...preparing data for model:',file_name)\n    df1=df_passed.copy()\n    \n    df1.drop('customer_ID',axis=1,inplace=True)\n    df1.drop('S_2',axis=1,inplace=True)\n    df1.fillna(method=\"ffill\",inplace=True)\n    df1.fillna(method=\"bfill\",inplace=True)\n    #df1.info(verbose=True)\n    df1['D_64'] = df1['D_64'].str.replace('-1','R')\n    df1_dummies = pd.get_dummies(df1, drop_first=True)\n    find_features_rf(df1_dummies,file_name,df1)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T14:28:44.112733Z","iopub.execute_input":"2022-07-23T14:28:44.113162Z","iopub.status.idle":"2022-07-23T14:28:44.120417Z","shell.execute_reply.started":"2022-07-23T14:28:44.113130Z","shell.execute_reply":"2022-07-23T14:28:44.119617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Reading dataset with 500000 rows\n# List of filenames to read in\nDATA_FILES = [\n    '../input/amex-reading-data/data1.csv',\n    '../input/amex-reading-data/data2.csv',\n\t'../input/amex-reading-data/data3.csv',\n\t'../input/amex-reading-data/data4.csv',\n\t'../input/amex-reading-data/data5.csv',\n    '../input/amex-reading-data/data6.csv',\n    '../input/amex-reading-data/data7.csv',\n    '../input/amex-reading-data/data8.csv',\n\t'../input/amex-reading-data/data9.csv',\n\t'../input/amex-reading-data/data10.csv',\n\t'../input/amex-reading-data/data11.csv',\n    '../input/amex-reading-data/data12.csv',\n    \n    \n    \n\t# ...\n]\n\n# This is the actual generator\ndef load_files(filenames):\n  df_target=pd.read_csv('../input/amex-default-prediction/train_labels.csv')   \n  for filename in filenames:\n        \n        df= pd.read_csv(filename,usecols =final_col)\n        df1=pd.merge(df,df_target,on='customer_ID',how='inner')\n        data_pre(df1,filename) \n\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-23T14:28:49.129917Z","iopub.execute_input":"2022-07-23T14:28:49.130300Z","iopub.status.idle":"2022-07-23T14:28:49.137647Z","shell.execute_reply.started":"2022-07-23T14:28:49.130269Z","shell.execute_reply":"2022-07-23T14:28:49.136426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"load_files(DATA_FILES)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T14:28:54.685147Z","iopub.execute_input":"2022-07-23T14:28:54.685537Z","iopub.status.idle":"2022-07-23T14:35:02.011633Z","shell.execute_reply.started":"2022-07-23T14:28:54.685505Z","shell.execute_reply":"2022-07-23T14:35:02.010282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}