{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-28T08:43:44.508272Z","iopub.execute_input":"2022-07-28T08:43:44.509427Z","iopub.status.idle":"2022-07-28T08:43:44.546563Z","shell.execute_reply.started":"2022-07-28T08:43:44.509288Z","shell.execute_reply":"2022-07-28T08:43:44.545474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore')\n\ndf = pd.read_feather('../input/amexfeather/train_data.ftr')\n\ndf.info(verbose=True,show_counts=True)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-28T08:43:44.623552Z","iopub.execute_input":"2022-07-28T08:43:44.624031Z","iopub.status.idle":"2022-07-28T08:44:10.738660Z","shell.execute_reply.started":"2022-07-28T08:43:44.623990Z","shell.execute_reply":"2022-07-28T08:44:10.737439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The dataset has 190 fetures out of which certain features are categorical,\n['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\n\nrest are numerical it seems,\nwe also observe a lot of missing data for certain features, we will need to identify those features and select features basis data availability first then again basis statistical importance , as we may discover little later through this notebook","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sn\nplt.figure(figsize=(5,5))\n\nax = sn.countplot(x=\"target\", data=df)\n\nfor p in ax.patches:#displaying % as annotations\n        ax.annotate(round(100*p.get_height()/len(df),2), (p.get_x()+0.3, p.get_height()+2))","metadata":{"execution":{"iopub.status.busy":"2022-07-28T08:44:10.740980Z","iopub.execute_input":"2022-07-28T08:44:10.741517Z","iopub.status.idle":"2022-07-28T08:44:12.270589Z","shell.execute_reply.started":"2022-07-28T08:44:10.741468Z","shell.execute_reply":"2022-07-28T08:44:12.269610Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Plotting class distribution we keep in mind that negative class here is 0 and positive class is 1, we may need to use some technique for balanced learning if the results from our Base Model is not fine. Meanwhile we prepare our dataset for machine learning.","metadata":{}},{"cell_type":"code","source":"cat_f = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126','D_63','D_64', 'D_66', 'D_68'] \n\nall_f = list(df.columns)\nall_f.remove(\"customer_ID\")\nall_f.remove(\"S_2\")\nall_f.remove(\"D_142\")\n\n#finding set of numerical features by cosnducting simple set operations\nnum_f = list(set(all_f) - set(cat_f))\n\nprint(num_f[0:5])","metadata":{"execution":{"iopub.status.busy":"2022-07-28T08:44:12.272321Z","iopub.execute_input":"2022-07-28T08:44:12.273004Z","iopub.status.idle":"2022-07-28T08:44:12.281488Z","shell.execute_reply.started":"2022-07-28T08:44:12.272963Z","shell.execute_reply":"2022-07-28T08:44:12.280229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df[all_f]\n\n#find null data\nprint(df.isnull().sum())\n\n#drop columns with less than 20 % data\nperc = 20.0 # Like N %\nmin_count =  int(((100-perc)/100)*df.shape[0] + 1)\ndf = df.dropna( axis=1, \n                thresh=min_count)\ndf","metadata":{"execution":{"iopub.status.busy":"2022-07-28T08:44:12.283962Z","iopub.execute_input":"2022-07-28T08:44:12.284443Z","iopub.status.idle":"2022-07-28T08:44:32.696618Z","shell.execute_reply.started":"2022-07-28T08:44:12.284375Z","shell.execute_reply":"2022-07-28T08:44:32.695474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"After removing certain features from the dataframe we clean it further by dropping all rows with NA , we have lost some data but still have a lot of data for model building and testing.","metadata":{}},{"cell_type":"code","source":"df=df.dropna()\ndf=df.reset_index()\ndf=df.drop(\"index\",axis=1)\ndf","metadata":{"execution":{"iopub.status.busy":"2022-07-28T08:44:32.698258Z","iopub.execute_input":"2022-07-28T08:44:32.698790Z","iopub.status.idle":"2022-07-28T08:44:47.623125Z","shell.execute_reply.started":"2022-07-28T08:44:32.698745Z","shell.execute_reply":"2022-07-28T08:44:47.622100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_f = list(df.columns)\n\n#finding set of numerical features by cosnducting simple set operations\nnum_f = list(set(all_f) - set(cat_f))\n\ncat_f = list(set(all_f) - set(num_f))\n\nnum_f[0:5]","metadata":{"execution":{"iopub.status.busy":"2022-07-28T08:44:47.624786Z","iopub.execute_input":"2022-07-28T08:44:47.625458Z","iopub.status.idle":"2022-07-28T08:44:47.633957Z","shell.execute_reply.started":"2022-07-28T08:44:47.625411Z","shell.execute_reply":"2022-07-28T08:44:47.633254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_f=list(df.columns)\nall_f.remove(\"target\")","metadata":{"execution":{"iopub.status.busy":"2022-07-28T08:44:47.635368Z","iopub.execute_input":"2022-07-28T08:44:47.635970Z","iopub.status.idle":"2022-07-28T08:44:47.654617Z","shell.execute_reply.started":"2022-07-28T08:44:47.635928Z","shell.execute_reply":"2022-07-28T08:44:47.653627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"encoded_df = pd.get_dummies( df[all_f], \n                                        columns = cat_f,\n                                        drop_first = True )\n\nencoded_df","metadata":{"execution":{"iopub.status.busy":"2022-07-28T08:44:47.655855Z","iopub.execute_input":"2022-07-28T08:44:47.656697Z","iopub.status.idle":"2022-07-28T08:44:54.461540Z","shell.execute_reply.started":"2022-07-28T08:44:47.656663Z","shell.execute_reply":"2022-07-28T08:44:54.460814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = encoded_df\nY = df['target']","metadata":{"execution":{"iopub.status.busy":"2022-07-28T08:44:54.462557Z","iopub.execute_input":"2022-07-28T08:44:54.463225Z","iopub.status.idle":"2022-07-28T08:44:54.467875Z","shell.execute_reply.started":"2022-07-28T08:44:54.463193Z","shell.execute_reply":"2022-07-28T08:44:54.466870Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\ntrain_X, test_X, train_y, test_y = train_test_split( X,\n                                                    Y,\n                                                    test_size = 0.3,\n                                                    random_state = 42 )","metadata":{"execution":{"iopub.status.busy":"2022-07-28T08:44:54.472006Z","iopub.execute_input":"2022-07-28T08:44:54.473077Z","iopub.status.idle":"2022-07-28T08:45:08.556638Z","shell.execute_reply.started":"2022-07-28T08:44:54.472995Z","shell.execute_reply":"2022-07-28T08:45:08.555484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#We need to scale our feature in a standard scale, it's our pre-processor\nfrom sklearn.preprocessing import StandardScaler\n\nsc = StandardScaler()\ntrain_X = sc.fit_transform(train_X)\ntest_X = sc.transform(test_X)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T08:45:08.558469Z","iopub.execute_input":"2022-07-28T08:45:08.558985Z","iopub.status.idle":"2022-07-28T08:45:34.516879Z","shell.execute_reply.started":"2022-07-28T08:45:08.558934Z","shell.execute_reply":"2022-07-28T08:45:34.515909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\n# grid search solver to find best fit\nfrom sklearn.datasets import make_classification\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.model_selection import RepeatedStratifiedKFold\nfrom sklearn.linear_model import LogisticRegression\n\n#define class weight dictionary, negative class has 20x weight\nw = {0:20, 1:1}\n\n# define dataset\nX, y = make_classification(n_samples=10000, n_features=176, n_redundant=0, random_state=1)\n# define model\nmodel = LogisticRegression(random_state=1, class_weight=w)\n# define model evaluation method\ncv = RepeatedStratifiedKFold(n_splits=7, n_repeats=3, random_state=1)\n# define grid\ngrid = dict()\ngrid['solver'] = ['liblinear', 'newton-cg', 'lbfgs', 'sag', 'saga']\n# define search\nsearch = GridSearchCV(model, grid, scoring='roc_auc', cv=cv, n_jobs=1)\n# perform the search\nresults = search.fit(X, y)\n# summarize\nprint('Mean Accuracy: %.3f' % results.best_score_)\nprint('Config: %s' % results.best_params_)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T08:45:34.518220Z","iopub.execute_input":"2022-07-28T08:45:34.518579Z","iopub.status.idle":"2022-07-28T08:47:12.565799Z","shell.execute_reply.started":"2022-07-28T08:45:34.518547Z","shell.execute_reply":"2022-07-28T08:47:12.562159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"probs1= search.predict_proba(X)[:,1]\nprobs1","metadata":{"execution":{"iopub.status.busy":"2022-07-28T08:47:12.567882Z","iopub.execute_input":"2022-07-28T08:47:12.568669Z","iopub.status.idle":"2022-07-28T08:47:12.585809Z","shell.execute_reply.started":"2022-07-28T08:47:12.568620Z","shell.execute_reply":"2022-07-28T08:47:12.584595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score, roc_curve, auc\n\ndef evaluate_roc(probs, y_true):\n    \"\"\"\n    - Print AUC and accuracy on the test set\n    - Plot ROC\n  \n    \"\"\"\n    preds = probs1\n    fpr, tpr, threshold = roc_curve(y_true, preds)\n    roc_auc = auc(fpr, tpr)\n    print(f'AUC: {roc_auc:.4f}')\n       \n    # Get accuracy over the test set\n    y_pred = np.where(preds >= 0.5, 1, 0)\n    accuracy = accuracy_score(y_true, y_pred)\n    print(f'Accuracy: {accuracy*100:.2f}%')\n    \n    # Plot ROC AUC\n    plt.title('Receiver Operating Characteristic')\n    plt.plot(fpr, tpr, 'b', label = 'AUC = %0.2f' % roc_auc)\n    plt.legend(loc = 'lower right')\n    plt.plot([0, 1], [0, 1],'r--')\n    plt.xlim([0, 1])\n    plt.ylim([0, 1])\n    plt.ylabel('True Positive Rate')\n    plt.xlabel('False Positive Rate')\n    plt.show()\n    \n# Evaluate the classifier\nevaluate_roc(probs1,y)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T08:47:12.591895Z","iopub.execute_input":"2022-07-28T08:47:12.595122Z","iopub.status.idle":"2022-07-28T08:47:12.799575Z","shell.execute_reply.started":"2022-07-28T08:47:12.595056Z","shell.execute_reply":"2022-07-28T08:47:12.796801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\n\ndel df,encoded_df\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T08:47:12.801094Z","iopub.execute_input":"2022-07-28T08:47:12.801600Z","iopub.status.idle":"2022-07-28T08:47:12.991918Z","shell.execute_reply.started":"2022-07-28T08:47:12.801561Z","shell.execute_reply":"2022-07-28T08:47:12.990890Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del X,train_X,test_X\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T08:47:12.994148Z","iopub.execute_input":"2022-07-28T08:47:12.994866Z","iopub.status.idle":"2022-07-28T08:47:13.181980Z","shell.execute_reply.started":"2022-07-28T08:47:12.994816Z","shell.execute_reply":"2022-07-28T08:47:13.180780Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.read_feather('/kaggle/input/amexfeather/test_data.ftr')\n\ndf_test =  (test_df\n            .groupby('customer_ID')\n            .tail(1))\ndel test_df\ngc.collect()\n\n#num_f.append(\"customer_ID\")\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-28T08:47:13.183920Z","iopub.execute_input":"2022-07-28T08:47:13.184485Z","iopub.status.idle":"2022-07-28T08:47:54.720018Z","shell.execute_reply.started":"2022-07-28T08:47:13.184434Z","shell.execute_reply":"2022-07-28T08:47:54.718876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#all_f.remove('customer_ID')","metadata":{"execution":{"iopub.status.busy":"2022-07-28T08:47:54.721906Z","iopub.execute_input":"2022-07-28T08:47:54.722439Z","iopub.status.idle":"2022-07-28T08:47:54.727833Z","shell.execute_reply.started":"2022-07-28T08:47:54.722361Z","shell.execute_reply":"2022-07-28T08:47:54.726806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfn_test = df_test[all_f]\ndfn_test","metadata":{"execution":{"iopub.status.busy":"2022-07-28T08:47:54.730106Z","iopub.execute_input":"2022-07-28T08:47:54.730651Z","iopub.status.idle":"2022-07-28T08:47:55.681027Z","shell.execute_reply.started":"2022-07-28T08:47:54.730605Z","shell.execute_reply":"2022-07-28T08:47:55.679799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\n\n\ndel df_test\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T08:47:55.682924Z","iopub.execute_input":"2022-07-28T08:47:55.683470Z","iopub.status.idle":"2022-07-28T08:47:55.931581Z","shell.execute_reply.started":"2022-07-28T08:47:55.683420Z","shell.execute_reply":"2022-07-28T08:47:55.930475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cid = dfn_test.customer_ID\n\nall_f.remove(\"customer_ID\")\n\ndfn_test = dfn_test[all_f]\ndfn_test","metadata":{"execution":{"iopub.status.busy":"2022-07-28T08:47:55.933492Z","iopub.execute_input":"2022-07-28T08:47:55.934013Z","iopub.status.idle":"2022-07-28T08:47:56.071057Z","shell.execute_reply.started":"2022-07-28T08:47:55.933965Z","shell.execute_reply":"2022-07-28T08:47:56.069354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_f.remove('target')","metadata":{"execution":{"iopub.status.busy":"2022-07-28T08:47:56.072573Z","iopub.status.idle":"2022-07-28T08:47:56.073029Z","shell.execute_reply.started":"2022-07-28T08:47:56.072833Z","shell.execute_reply":"2022-07-28T08:47:56.072853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = dfn_test.dropna()\n\nfor col in num_f:\n    dfn_test[col] = dfn_test[col].fillna(test[col].median())\ndel test\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T08:47:56.074071Z","iopub.status.idle":"2022-07-28T08:47:56.074999Z","shell.execute_reply.started":"2022-07-28T08:47:56.074779Z","shell.execute_reply":"2022-07-28T08:47:56.074803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in cat_f:\n    dfn_test[col] = dfn_test[col].fillna(dfn_test[col].value_counts().idxmax())\nprint(dfn_test.isnull().sum())","metadata":{"execution":{"iopub.status.busy":"2022-07-28T08:47:56.075936Z","iopub.status.idle":"2022-07-28T08:47:56.076260Z","shell.execute_reply.started":"2022-07-28T08:47:56.076105Z","shell.execute_reply":"2022-07-28T08:47:56.076121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"encoded_dft = pd.get_dummies(  dfn_test, \n                                    columns = cat_f,\n                                    drop_first = True )\n\nencoded_dft","metadata":{"execution":{"iopub.status.busy":"2022-07-28T08:47:56.077073Z","iopub.status.idle":"2022-07-28T08:47:56.077425Z","shell.execute_reply.started":"2022-07-28T08:47:56.077246Z","shell.execute_reply":"2022-07-28T08:47:56.077263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nencoded_dft = pd.get_dummies( test_dfn, \n                                        columns = cat_f,\n                                        drop_first = True )\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T08:47:56.078979Z","iopub.status.idle":"2022-07-28T08:47:56.079323Z","shell.execute_reply.started":"2022-07-28T08:47:56.079165Z","shell.execute_reply":"2022-07-28T08:47:56.079181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_yt = search.predict(encoded_dft)\nfinal_probs = search.predict_proba(encoded_dft)[:,1]\ndel encoded_dft\ngc.collect()\ntest_dfn['prediction'] = final_probs\n\ntest_dfn['customer_ID'] = c_id\n\ntest_dfn = test_dfn[['customer_ID','prediction']]","metadata":{"execution":{"iopub.status.busy":"2022-07-28T08:47:56.080133Z","iopub.status.idle":"2022-07-28T08:47:56.080528Z","shell.execute_reply.started":"2022-07-28T08:47:56.080295Z","shell.execute_reply":"2022-07-28T08:47:56.080310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dfn.prediction.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T08:47:56.081485Z","iopub.status.idle":"2022-07-28T08:47:56.081809Z","shell.execute_reply.started":"2022-07-28T08:47:56.081653Z","shell.execute_reply":"2022-07-28T08:47:56.081669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dfn['predicted_b'] = test_dfn.prediction.map( \n                            lambda x: 1 if x > 0.5 else 0)\n\ntest_dfn.sample(100, random_state = 42)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T08:47:56.082827Z","iopub.status.idle":"2022-07-28T08:47:56.083157Z","shell.execute_reply.started":"2022-07-28T08:47:56.082999Z","shell.execute_reply":"2022-07-28T08:47:56.083016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dfn.predicted_b.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T08:47:56.084013Z","iopub.status.idle":"2022-07-28T08:47:56.084346Z","shell.execute_reply.started":"2022-07-28T08:47:56.084192Z","shell.execute_reply":"2022-07-28T08:47:56.084207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dfn.predictions = test_dfn.predicted_b\n\ntest_dfn = test_dfn[['customer_ID','prediction']]\n\ntest_dfn.to_csv(\"submission.csv\",index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T08:47:56.085552Z","iopub.status.idle":"2022-07-28T08:47:56.085907Z","shell.execute_reply.started":"2022-07-28T08:47:56.085722Z","shell.execute_reply":"2022-07-28T08:47:56.085739Z"},"trusted":true},"execution_count":null,"outputs":[]}]}