{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"The following steps were followed to create the final submission file \nThis model was build on RandomForest classifier reading data from the csv files give in the competition\n\nStep1: Read the train_data.csv with chunk_size=5lakh\n\nStep2: create individual files for each chunk,12 such files were generated\n\nstep3: Checked individual dataframes for columns with more than 40% missing values\n\nStep4. All the 12 dataframes have same columns for missing values\n\nstep5. Checked individual dataframes for columns with correlations coeff.75\n\nstep6. same columns were found in all 12 dataframes for correlation coeff .75\n\nstep7. Hence selected one  dataframe and deleted the above columns\n\nstep8. Filled nan values with bfill,ffill\n\nstep9. Build the model with randomn forest as this gave a better amex_metric\n\nstep10. Read the test_data.csv with chunk_size=5lak\n\nstep11. followed steps for data imputation as train data\n\nstep12.predicted submitted the file\n\nstep13.The following procedure gave an amex_metric .477\n\n","metadata":{}},{"cell_type":"code","source":"#-------------------------------------------------------------------------------------------------------------------------------\nimport pandas as pd                                                 # Importing for panel data analysis\n#from pandas_profiling import ProfileReport                          # Import Pandas Profiling (To generate Univariate Analysis)\n\npd.set_option('display.max_columns', None)                          # Unfolding hidden features if the cardinality is high      \npd.set_option('display.max_colwidth', None)                         # Unfolding the max feature width for better clearity      \npd.set_option('display.max_rows', None)                             # Unfolding hidden data points if the cardinality is high\npd.set_option('mode.chained_assignment', None)                      # Removing restriction over chained assignments operations\npd.set_option('display.float_format', lambda x: '%.5f' % x)         # To suppress scientific notation over exponential values\n#-------------------------------------------------------------------------------------------------------------------------------\nimport numpy as np                                                  # Importing package numpys (For Numerical Python)\n#-------------------------------------------------------------------------------------------------------------------------------\nimport matplotlib.pyplot as plt                                     # Importing pyplot interface using matplotlib\nfrom matplotlib.pylab import rcParams                               # Backend used for rendering and GUI integration                                               \nimport seaborn as sns                                               # Importin seaborm library for interactive visualization\n%matplotlib inline\n#-------------------------------------------------------------------------------------------------------------------------------\nfrom sklearn.metrics import accuracy_score                          # For calculating the accuracy for the model\nfrom sklearn.metrics import precision_score                         # For calculating the Precision of the model\nfrom sklearn.metrics import recall_score                            # For calculating the recall of the model\nfrom sklearn.metrics import precision_recall_curve                  # For precision and recall metric estimation\nfrom sklearn.metrics import confusion_matrix                        # For verifying model performance using confusion matrix\nfrom sklearn.metrics import f1_score                                # For Checking the F1-Score of our model  \nfrom sklearn.metrics import roc_curve                               # For Roc-Auc metric estimation\n#-------------------------------------------------------------------------------------------------------------------------------\nfrom sklearn.model_selection import train_test_split                # To split the data in training and testing part     \nfrom sklearn.linear_model import LogisticRegression                 # To create the Logistic Regression Model\nfrom sklearn.ensemble import RandomForestClassifier, GradientBoostingClassifier\nfrom sklearn.metrics import confusion_matrix,classification_report\nfrom sklearn.model_selection import cross_val_score\n#-------------------------------------------------------------------------------------------------------------------------------\nimport warnings                                                     # Importing warning to disable runtime warnings\nwarnings.filterwarnings(\"ignore\")                                   # Warnings will appear only once\n","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:38:05.295646Z","iopub.execute_input":"2022-07-22T17:38:05.296138Z","iopub.status.idle":"2022-07-22T17:38:05.319393Z","shell.execute_reply.started":"2022-07-22T17:38:05.296099Z","shell.execute_reply":"2022-07-22T17:38:05.317707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create correlation matrix\nto_drop=[]\nimport numpy as np\ndef drop_corr_columns(df,i):\n\n    corr_matrix = df.corr().abs()\n\n# Select upper triangle of correlation matrix\n    upper = corr_matrix.where(np.triu(np.ones(corr_matrix.shape), k=1).astype(np.bool))\n\n# Find index of feature columns with correlatgc.collect()ion greater than 0.95\n    to_drop = [column for column in upper.columns if any(upper[column] > 0.7)]\n    print('columns to drop in ',{i},'dataset',to_drop)\n    df1.drop(axis=1,columns= to_drop,inplace=True)\n   ","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:38:05.333183Z","iopub.execute_input":"2022-07-22T17:38:05.333699Z","iopub.status.idle":"2022-07-22T17:38:05.342295Z","shell.execute_reply.started":"2022-07-22T17:38:05.333659Z","shell.execute_reply":"2022-07-22T17:38:05.341227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Reading dataset with 500000 rows\n\ndf=pd.read_csv('../input/amex-reading-data/data1.csv')\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:38:05.368047Z","iopub.execute_input":"2022-07-22T17:38:05.368799Z","iopub.status.idle":"2022-07-22T17:38:42.308987Z","shell.execute_reply.started":"2022-07-22T17:38:05.368753Z","shell.execute_reply":"2022-07-22T17:38:42.307634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_target=pd.read_csv('../input/amex-default-prediction/train_labels.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:38:42.311919Z","iopub.execute_input":"2022-07-22T17:38:42.312483Z","iopub.status.idle":"2022-07-22T17:38:43.181320Z","shell.execute_reply.started":"2022-07-22T17:38:42.312414Z","shell.execute_reply":"2022-07-22T17:38:43.179918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1=pd.merge(df,df_target,on='customer_ID',how='inner')","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:38:43.183369Z","iopub.execute_input":"2022-07-22T17:38:43.183778Z","iopub.status.idle":"2022-07-22T17:38:44.908343Z","shell.execute_reply.started":"2022-07-22T17:38:43.183745Z","shell.execute_reply":"2022-07-22T17:38:44.906857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"list_miss=df1.isnull().sum()/len(df1)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:38:44.911132Z","iopub.execute_input":"2022-07-22T17:38:44.911546Z","iopub.status.idle":"2022-07-22T17:38:45.328751Z","shell.execute_reply.started":"2022-07-22T17:38:44.911501Z","shell.execute_reply":"2022-07-22T17:38:45.327150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#finding columns with more than 40% missing values\nlist_missing=[]\nfor i in range(len(list_miss)):\n    if list_miss[i]>0.4 :\n        list_missing.append(list_miss.index[i])","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:38:45.330269Z","iopub.execute_input":"2022-07-22T17:38:45.330707Z","iopub.status.idle":"2022-07-22T17:38:45.339659Z","shell.execute_reply.started":"2022-07-22T17:38:45.330671Z","shell.execute_reply":"2022-07-22T17:38:45.337642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"list_missing","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:38:45.341564Z","iopub.execute_input":"2022-07-22T17:38:45.341956Z","iopub.status.idle":"2022-07-22T17:38:45.355697Z","shell.execute_reply.started":"2022-07-22T17:38:45.341921Z","shell.execute_reply":"2022-07-22T17:38:45.354644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1.drop(columns=list_missing,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:38:45.357976Z","iopub.execute_input":"2022-07-22T17:38:45.358349Z","iopub.status.idle":"2022-07-22T17:38:46.664756Z","shell.execute_reply.started":"2022-07-22T17:38:45.358317Z","shell.execute_reply":"2022-07-22T17:38:46.663364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"drop_corr_columns(df1,1)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:38:46.666979Z","iopub.execute_input":"2022-07-22T17:38:46.667558Z","iopub.status.idle":"2022-07-22T17:39:21.110921Z","shell.execute_reply.started":"2022-07-22T17:38:46.667498Z","shell.execute_reply":"2022-07-22T17:39:21.109565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"to_drop","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:39:21.113035Z","iopub.execute_input":"2022-07-22T17:39:21.113465Z","iopub.status.idle":"2022-07-22T17:39:21.122005Z","shell.execute_reply.started":"2022-07-22T17:39:21.113413Z","shell.execute_reply":"2022-07-22T17:39:21.120293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:39:21.128015Z","iopub.execute_input":"2022-07-22T17:39:21.128645Z","iopub.status.idle":"2022-07-22T17:39:21.153320Z","shell.execute_reply.started":"2022-07-22T17:39:21.128605Z","shell.execute_reply":"2022-07-22T17:39:21.151879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1.drop('Unnamed: 0',axis=1,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:39:21.155236Z","iopub.execute_input":"2022-07-22T17:39:21.156276Z","iopub.status.idle":"2022-07-22T17:39:21.328359Z","shell.execute_reply.started":"2022-07-22T17:39:21.156226Z","shell.execute_reply":"2022-07-22T17:39:21.326945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1.drop('customer_ID',axis=1,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:39:21.330013Z","iopub.execute_input":"2022-07-22T17:39:21.330424Z","iopub.status.idle":"2022-07-22T17:39:21.584946Z","shell.execute_reply.started":"2022-07-22T17:39:21.330390Z","shell.execute_reply":"2022-07-22T17:39:21.583581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1.drop('S_2',axis=1,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:39:21.586602Z","iopub.execute_input":"2022-07-22T17:39:21.586946Z","iopub.status.idle":"2022-07-22T17:39:21.827324Z","shell.execute_reply.started":"2022-07-22T17:39:21.586916Z","shell.execute_reply":"2022-07-22T17:39:21.826088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1.fillna(method=\"ffill\",inplace=True)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:39:21.828765Z","iopub.execute_input":"2022-07-22T17:39:21.829089Z","iopub.status.idle":"2022-07-22T17:39:22.098862Z","shell.execute_reply.started":"2022-07-22T17:39:21.829058Z","shell.execute_reply":"2022-07-22T17:39:22.097518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1.fillna(method=\"bfill\",inplace=True)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:39:22.100603Z","iopub.execute_input":"2022-07-22T17:39:22.101336Z","iopub.status.idle":"2022-07-22T17:39:22.396224Z","shell.execute_reply.started":"2022-07-22T17:39:22.101293Z","shell.execute_reply":"2022-07-22T17:39:22.394790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:39:22.398077Z","iopub.execute_input":"2022-07-22T17:39:22.398473Z","iopub.status.idle":"2022-07-22T17:39:22.654366Z","shell.execute_reply.started":"2022-07-22T17:39:22.398419Z","shell.execute_reply":"2022-07-22T17:39:22.652918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1.info(verbose=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:39:22.656821Z","iopub.execute_input":"2022-07-22T17:39:22.657317Z","iopub.status.idle":"2022-07-22T17:39:22.680012Z","shell.execute_reply.started":"2022-07-22T17:39:22.657269Z","shell.execute_reply":"2022-07-22T17:39:22.678968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"S_2,D_63,D_64 object type","metadata":{}},{"cell_type":"code","source":"df1.D_63.unique()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:39:22.681639Z","iopub.execute_input":"2022-07-22T17:39:22.682543Z","iopub.status.idle":"2022-07-22T17:39:22.724856Z","shell.execute_reply.started":"2022-07-22T17:39:22.682504Z","shell.execute_reply":"2022-07-22T17:39:22.723693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1['D_64'] = df1['D_64'].str.replace('-1','R')\n","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:39:22.726410Z","iopub.execute_input":"2022-07-22T17:39:22.727046Z","iopub.status.idle":"2022-07-22T17:39:23.135931Z","shell.execute_reply.started":"2022-07-22T17:39:22.727010Z","shell.execute_reply":"2022-07-22T17:39:23.134315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Model building","metadata":{}},{"cell_type":"markdown","source":"Step 1: one hot encoding","metadata":{}},{"cell_type":"code","source":"df1_dummies = pd.get_dummies(df1, drop_first=True)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:39:23.138016Z","iopub.execute_input":"2022-07-22T17:39:23.138417Z","iopub.status.idle":"2022-07-22T17:39:23.724714Z","shell.execute_reply.started":"2022-07-22T17:39:23.138382Z","shell.execute_reply":"2022-07-22T17:39:23.723423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1_dummies.drop('target',axis=1,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:39:23.726541Z","iopub.execute_input":"2022-07-22T17:39:23.727321Z","iopub.status.idle":"2022-07-22T17:39:24.374210Z","shell.execute_reply.started":"2022-07-22T17:39:23.727281Z","shell.execute_reply":"2022-07-22T17:39:24.372947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1_dummies.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:39:24.375859Z","iopub.execute_input":"2022-07-22T17:39:24.376242Z","iopub.status.idle":"2022-07-22T17:39:24.396786Z","shell.execute_reply.started":"2022-07-22T17:39:24.376208Z","shell.execute_reply":"2022-07-22T17:39:24.395252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y=df1.target","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:39:24.398860Z","iopub.execute_input":"2022-07-22T17:39:24.399179Z","iopub.status.idle":"2022-07-22T17:39:24.405039Z","shell.execute_reply.started":"2022-07-22T17:39:24.399152Z","shell.execute_reply":"2022-07-22T17:39:24.404053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:39:24.406502Z","iopub.execute_input":"2022-07-22T17:39:24.407573Z","iopub.status.idle":"2022-07-22T17:39:24.420473Z","shell.execute_reply.started":"2022-07-22T17:39:24.407534Z","shell.execute_reply":"2022-07-22T17:39:24.418789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X=df1_dummies","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:39:24.423421Z","iopub.execute_input":"2022-07-22T17:39:24.424775Z","iopub.status.idle":"2022-07-22T17:39:24.430388Z","shell.execute_reply.started":"2022-07-22T17:39:24.424722Z","shell.execute_reply":"2022-07-22T17:39:24.429366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:39:24.431736Z","iopub.execute_input":"2022-07-22T17:39:24.432059Z","iopub.status.idle":"2022-07-22T17:39:24.442398Z","shell.execute_reply.started":"2022-07-22T17:39:24.432020Z","shell.execute_reply":"2022-07-22T17:39:24.441626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:39:24.443401Z","iopub.execute_input":"2022-07-22T17:39:24.443752Z","iopub.status.idle":"2022-07-22T17:39:24.456730Z","shell.execute_reply.started":"2022-07-22T17:39:24.443723Z","shell.execute_reply":"2022-07-22T17:39:24.455553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X, y, test_size = 0.3, random_state = 42, stratify = y)\n\nprint('Training Data Shape:', X_train.shape, y_train.shape)\nprint('Testing Data Shape:', X_test.shape, y_test.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:39:24.464679Z","iopub.execute_input":"2022-07-22T17:39:24.465030Z","iopub.status.idle":"2022-07-22T17:39:25.636891Z","shell.execute_reply.started":"2022-07-22T17:39:24.464999Z","shell.execute_reply":"2022-07-22T17:39:25.635137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"logreg = LogisticRegression()\nlogreg.fit(X_train,y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:39:25.638612Z","iopub.execute_input":"2022-07-22T17:39:25.639070Z","iopub.status.idle":"2022-07-22T17:39:39.390838Z","shell.execute_reply.started":"2022-07-22T17:39:25.639025Z","shell.execute_reply":"2022-07-22T17:39:39.389600Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#predicting on train data\ny_pred_train = logreg.predict(X_train)\n\n#predicting on test data\ny_pred_test = logreg.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:39:39.392877Z","iopub.execute_input":"2022-07-22T17:39:39.393693Z","iopub.status.idle":"2022-07-22T17:39:39.782693Z","shell.execute_reply.started":"2022-07-22T17:39:39.393642Z","shell.execute_reply":"2022-07-22T17:39:39.781148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nprint(confusion_matrix(y_test, y_pred_test))\nconfusion_matrix.index = ['Actual D','Actual nd']\nconfusion_matrix.columns = ['Predicted D','Predicted nd']\nprint(confusion_matrix)\nprint(classification_report(y_test,y_pred_test))","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:39:39.784977Z","iopub.execute_input":"2022-07-22T17:39:39.785871Z","iopub.status.idle":"2022-07-22T17:39:40.184419Z","shell.execute_reply.started":"2022-07-22T17:39:39.785815Z","shell.execute_reply":"2022-07-22T17:39:40.183231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = logreg.predict(X_test)\nprint('Accuracy score for test data is:', accuracy_score(y_test,pred))","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:39:40.185999Z","iopub.execute_input":"2022-07-22T17:39:40.186364Z","iopub.status.idle":"2022-07-22T17:39:40.328807Z","shell.execute_reply.started":"2022-07-22T17:39:40.186323Z","shell.execute_reply":"2022-07-22T17:39:40.325366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scoreslg = cross_val_score(logreg,X_train,y_train,cv=5,scoring='f1')\nprint(scoreslg)\nprint(\"Average f1\")\nprint(np.mean(scoreslg))","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:39:40.335888Z","iopub.execute_input":"2022-07-22T17:39:40.336545Z","iopub.status.idle":"2022-07-22T17:40:34.663199Z","shell.execute_reply.started":"2022-07-22T17:39:40.336490Z","shell.execute_reply":"2022-07-22T17:40:34.661778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.tree import DecisionTreeClassifier\ndtree=DecisionTreeClassifier()\n\ndtree.fit(X_train,y_train)\n#predicting on train data\ny_pred_train = dtree.predict(X_train)\n\n#predicting on test data\ny_pred_test = dtree.predict(X_test)\n\nprint(\"Testing Accuracy for dtree\")\nprint(dtree.score(X_test,y_test))\n\nprint(\"Training Accuracy for dtree\")\nprint(dtree.score(X_train,y_train))\nprint(confusion_matrix(y_test,y_pred_test))\nprint(classification_report(y_test,y_pred_test))\n","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:40:34.665174Z","iopub.execute_input":"2022-07-22T17:40:34.666216Z","iopub.status.idle":"2022-07-22T17:44:04.322979Z","shell.execute_reply.started":"2022-07-22T17:40:34.666155Z","shell.execute_reply":"2022-07-22T17:44:04.321284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import GradientBoostingClassifier\nXB=GradientBoostingClassifier(max_depth = 10,n_estimators=20,verbose=True)\nXB.fit(X_train,y_train)\nprint(\"Training Accuracy\")\nprint(XB.score(X_train,y_train))\nprint(\"Testing Accuracy\")\nprint(XB.score(X_test,y_test))\npredicted = XB.predict(X_test)\nprint(confusion_matrix(y_test,predicted))\nprint(classification_report(y_test,predicted))\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:47:40.669287Z","iopub.execute_input":"2022-07-22T17:47:40.670119Z","iopub.status.idle":"2022-07-22T17:59:05.079348Z","shell.execute_reply.started":"2022-07-22T17:47:40.670078Z","shell.execute_reply":"2022-07-22T17:59:05.077360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def amex_metric_mod(y_true, y_pred):\n\n    labels     = np.transpose(np.array([y_true, y_pred]))\n    labels     = labels[labels[:, 1].argsort()[::-1]]\n    weights    = np.where(labels[:,0]==0, 20, 1)\n    cut_vals   = labels[np.cumsum(weights) <= int(0.04 * np.sum(weights))]\n    top_four   = np.sum(cut_vals[:,0]) / np.sum(labels[:,0])\n\n    gini = [0,0]\n    for i in [1,0]:\n        labels         = np.transpose(np.array([y_true, y_pred]))\n        labels         = labels[labels[:, i].argsort()[::-1]]\n        weight         = np.where(labels[:,0]==0, 20, 1)\n        weight_random  = np.cumsum(weight / np.sum(weight))\n        total_pos      = np.sum(labels[:, 0] *  weight)\n        cum_pos_found  = np.cumsum(labels[:, 0] * weight)\n        lorentz        = cum_pos_found / total_pos\n        gini[i]        = np.sum((lorentz - weight_random) * weight)\n\n    return 0.5 * (gini[1]/gini[0] + top_four)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:59:05.082155Z","iopub.execute_input":"2022-07-22T17:59:05.082740Z","iopub.status.idle":"2022-07-22T17:59:05.096490Z","shell.execute_reply.started":"2022-07-22T17:59:05.082694Z","shell.execute_reply":"2022-07-22T17:59:05.095435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(amex_metric_mod(y_test, predicted)) ","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:59:14.273417Z","iopub.execute_input":"2022-07-22T17:59:14.273868Z","iopub.status.idle":"2022-07-22T17:59:14.318734Z","shell.execute_reply.started":"2022-07-22T17:59:14.273833Z","shell.execute_reply":"2022-07-22T17:59:14.317210Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndef create_model(df_test,subm_test):\n   \n    \n    df_test.drop('customer_ID',axis=1,inplace=True)\n    df_test.drop('S_2',axis=1,inplace=True)\n    df_test.fillna(method='ffill',inplace=True)\n    df_test.fillna(method='bfill',inplace=True)\n    df_test_dummies=pd.get_dummies(df_test, drop_first=True)\n  \n    subm_test['pred']=XB.predict(df_test_dummies)\n    #print(subm_test.head())\n    #subm_test2=subm_test2.append(subm_test1)\n    subm_test.to_csv('submssion.csv',mode='a',header=True,index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:59:28.377541Z","iopub.execute_input":"2022-07-22T17:59:28.378015Z","iopub.status.idle":"2022-07-22T17:59:28.386127Z","shell.execute_reply.started":"2022-07-22T17:59:28.377979Z","shell.execute_reply":"2022-07-22T17:59:28.384771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del_col=['B_3', 'D_48', 'B_8', 'B_11', 'R_4', 'S_7', 'D_55', 'B_13', 'D_58', 'D_61', 'B_15', 'B_16', 'B_18', 'B_19', 'B_20', 'B_22', 'S_15', 'B_23', 'D_74', 'D_75', 'D_77', 'R_8', 'S_16', 'B_28', 'B_30', 'R_21', 'B_33', 'S_24', 'D_103', 'D_104', 'D_107', 'B_37', 'B_38', 'D_113', 'D_118', 'D_119', 'D_121', 'D_129', 'D_131', 'D_133', 'D_141', 'D_143','D_42',\n 'D_49',\n 'D_50',\n 'D_53',\n 'D_56',\n 'S_9',\n 'B_17',\n 'D_66',\n 'D_73',\n 'D_76',\n 'D_77',\n 'R_9',\n 'D_82',\n 'B_29',\n 'D_87',\n 'D_88',\n 'D_105',\n 'D_106',\n 'R_26',\n 'D_108',\n 'D_110',\n 'D_111',\n 'B_39',\n 'B_42',\n 'D_132',\n 'D_134',\n 'D_135',\n 'D_136',\n 'D_137',\n 'D_138',\n 'D_142']\n","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:59:35.194819Z","iopub.execute_input":"2022-07-22T17:59:35.195199Z","iopub.status.idle":"2022-07-22T17:59:35.204907Z","shell.execute_reply.started":"2022-07-22T17:59:35.195167Z","shell.execute_reply":"2022-07-22T17:59:35.203283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"headers = [*pd.read_csv('../input/amex-default-prediction/test_data.csv', nrows=1)]\n#cols = list(pd.read_csv(\"test_data.csv\", nrows =1))\nprint(headers)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:59:46.380576Z","iopub.execute_input":"2022-07-22T17:59:46.381468Z","iopub.status.idle":"2022-07-22T17:59:46.418154Z","shell.execute_reply.started":"2022-07-22T17:59:46.381389Z","shell.execute_reply":"2022-07-22T17:59:46.416781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_col=list ((set(headers) - set (del_col)))","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:59:46.444847Z","iopub.execute_input":"2022-07-22T17:59:46.445272Z","iopub.status.idle":"2022-07-22T17:59:46.451395Z","shell.execute_reply.started":"2022-07-22T17:59:46.445234Z","shell.execute_reply":"2022-07-22T17:59:46.449549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_testdata = pd.read_csv('../input/amex-default-prediction/test_data.csv', chunksize=500000, iterator=True, usecols =final_col)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:59:46.474985Z","iopub.execute_input":"2022-07-22T17:59:46.475637Z","iopub.status.idle":"2022-07-22T17:59:46.483793Z","shell.execute_reply.started":"2022-07-22T17:59:46.475601Z","shell.execute_reply":"2022-07-22T17:59:46.482011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\nfor iter_num, chunk in enumerate(df_testdata, 1):\n    print(iter_num)\n    subm_t=pd.DataFrame(columns=['customer_ID','pred'])\n    subm_t[\"customer_ID\"]=  chunk[\"customer_ID\"]\n   \n    #print(chunk.info())\n    #print(subm_t.info())\n    create_model(chunk,subm_t)\n    \n","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:59:46.509256Z","iopub.execute_input":"2022-07-22T17:59:46.510351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\ngc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndf_sub=pd.read_csv('submssion.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sub.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sub.customer_ID.nunique()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#df.drop(df.loc[df['Stock']=='Yes'].index, inplace=True)\ndf_sub.drop(df_sub.loc[df_sub['customer_ID']=='customer_ID'].index,inplace=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sub.customer_ID.nunique()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_final_sub=df_sub.groupby('customer_ID').tail(1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_final_sub.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_final_sub.pred.nunique()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_final_sub.pred= df_final_sub.pred.astype (float)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_final_sub.pred= df_final_sub.pred.astype ('Int64')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_final_sub.pred.value_counts()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_final_sub.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndf_final_sub.rename(columns = {'pred':'prediction'}, inplace = True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_final_sub.to_csv(\"submission.csv\",header=True,index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}