{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-13T13:28:13.913675Z","iopub.execute_input":"2022-07-13T13:28:13.914189Z","iopub.status.idle":"2022-07-13T13:28:13.958828Z","shell.execute_reply.started":"2022-07-13T13:28:13.914081Z","shell.execute_reply":"2022-07-13T13:28:13.957543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore')\n\ndf = pd.read_feather('../input/amexfeather/train_data.ftr')\n\n#EDA\n#print(df.info(verbose=True,show_counts=True))\n\nimport matplotlib.pyplot as plt\nimport seaborn as sn\nplt.figure(figsize=(5,5))\n\nax = sn.countplot(x=\"target\", data=df)\nfor p in ax.patches:#displaying % as annotations\n        ax.annotate(str(round(100*p.get_height()/len(df),2))+\"%\", (p.get_x()+0.3, p.get_height()+2))","metadata":{"execution":{"iopub.status.busy":"2022-07-13T13:28:13.973297Z","iopub.execute_input":"2022-07-13T13:28:13.974145Z","iopub.status.idle":"2022-07-13T13:28:34.486280Z","shell.execute_reply.started":"2022-07-13T13:28:13.974103Z","shell.execute_reply":"2022-07-13T13:28:34.484996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Categorical features are given with the dataset\n\ncat_f = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68'] \nall_f = list(df.columns)\nall_f.remove(\"customer_ID\")\nall_f.remove(\"S_2\")\nall_f.remove(\"D_142\")\n\n#finding set of numerical features by cosnducting simple set operations\n\nnum_f = list(set(all_f) - set(cat_f))\nprint(num_f[0:5])\ndf = df[all_f]\n\n#find null data\nprint(df.isnull().sum())\n\n#Feature selection,drop columns with less than 20 % data\nperc = 20.0 # Like N %\nmin_count =  int(((100-perc)/100)*df.shape[0] + 1)\ndf = df.dropna( axis=1, \n                thresh=min_count)\n\ndf=df.dropna()\ndf=df.reset_index()\ndf=df.drop(\"index\",axis=1)\n\nall_f = list(df.columns)\n\n#finding set of numerical features by cosnducting simple set operations\nnum_f = list(set(all_f) - set(cat_f))\n\ncat_f = list(set(all_f) - set(num_f))\n\nall_f=list(df.columns)\nall_f.remove(\"target\")\n\nencoded_df = pd.get_dummies( df[all_f], \n                                        columns = cat_f,\n                                        drop_first = True )\n\nX = encoded_df\nY = df['target']\n\n#Define train and test data for model building and testing\nfrom sklearn.model_selection import train_test_split\n\ntrain_X, test_X, train_y, test_y = train_test_split( X,\n                                                    Y,\n                                                    test_size = 0.3,\n                                                    random_state = 42 )\n\n#We need to scale our feature in a standard scale, it's our pre-processor\nfrom sklearn.preprocessing import StandardScaler\n\nsc = StandardScaler()\ntrain_X = sc.fit_transform(train_X)\ntest_X = sc.transform(test_X)\n\n#import model dependencies\nfrom sklearn.linear_model import LogisticRegression\n\n# grid search solver to find best fit\nfrom sklearn.datasets import make_classification\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.model_selection import RepeatedStratifiedKFold\n\n#define class weight dictionary, negative class has 20x weight\nw = {0:20, 1:1}\n\n# define datase\nX, y = make_classification(n_samples=1000, n_features=176, n_redundant=0, random_state=1)\n# define model\nmodel = LogisticRegression(random_state=1, class_weight=w)\n# define model evaluation method\ncv = RepeatedStratifiedKFold(n_splits=7, n_repeats=3, random_state=1)\n# define grid\ngrid = dict()\ngrid['solver'] = ['liblinear', 'newton-cg', 'lbfgs', 'sag', 'saga']\n# define search\nsearch = GridSearchCV(model, grid, scoring='accuracy', cv=cv, n_jobs=1)\n# perform the search\nresults = search.fit(X, y)\n# summarize\nprint('Mean Accuracy: %.3f' % results.best_score_)\nprint('Config: %s' % results.best_params_)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-13T13:28:34.488638Z","iopub.execute_input":"2022-07-13T13:28:34.489459Z","iopub.status.idle":"2022-07-13T13:29:55.465117Z","shell.execute_reply.started":"2022-07-13T13:28:34.489426Z","shell.execute_reply":"2022-07-13T13:29:55.463838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"probs1= search.predict_proba(X)[:,1]\n\nfrom sklearn.metrics import accuracy_score, roc_curve, auc\n\ndef evaluate_roc(probs, y_true):\n    \"\"\"\n    - Print AUC and accuracy on the test set\n    - Plot ROC\n  \n    \"\"\"\n    preds = probs1\n    fpr, tpr, threshold = roc_curve(y_true, preds)\n    roc_auc = auc(fpr, tpr)\n    print(f'AUC: {roc_auc:.4f}')\n       \n    # Get accuracy over the test set\n    y_pred = np.where(preds >= 0.5, 1, 0)\n    accuracy = accuracy_score(y_true, y_pred)\n    print(f'Accuracy: {accuracy*100:.2f}%')\n    \n    # Plot ROC AUC\n    plt.title('Receiver Operating Characteristic')\n    plt.plot(fpr, tpr, 'b', label = 'AUC = %0.2f' % roc_auc)\n    plt.legend(loc = 'lower right')\n    plt.plot([0, 1], [0, 1],'r--')\n    plt.xlim([0, 1])\n    plt.ylim([0, 1])\n    plt.ylabel('True Positive Rate')\n    plt.xlabel('False Positive Rate')\n    plt.show()\n    \n# Evaluate the classifier\nevaluate_roc(probs1,y)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T13:29:55.466894Z","iopub.execute_input":"2022-07-13T13:29:55.467655Z","iopub.status.idle":"2022-07-13T13:29:55.660150Z","shell.execute_reply.started":"2022-07-13T13:29:55.467605Z","shell.execute_reply":"2022-07-13T13:29:55.659200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\n\ndel df,encoded_df\n\ndel X,train_X,test_X\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T13:29:55.662386Z","iopub.execute_input":"2022-07-13T13:29:55.662721Z","iopub.status.idle":"2022-07-13T13:29:55.859971Z","shell.execute_reply.started":"2022-07-13T13:29:55.662693Z","shell.execute_reply":"2022-07-13T13:29:55.858646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ntest_df = pd.read_feather('/kaggle/input/amexfeather/test_data.ftr')\n\nall_f.append(\"customer_ID\")\n\ntest_dfn = test_df[all_f]\n\ndel test_df\n\ngc.collect()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-13T13:37:39.840760Z","iopub.execute_input":"2022-07-13T13:37:39.842507Z","iopub.status.idle":"2022-07-13T13:38:04.340810Z","shell.execute_reply.started":"2022-07-13T13:37:39.842453Z","shell.execute_reply":"2022-07-13T13:38:04.339318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ntest_dfn = test_dfn.dropna().sample(700000)\n\nc_id = test_dfn.customer_ID\n\nall_f.remove(\"customer_ID\")\n\ntest_dfn = test_dfn[all_f]\n\nencoded_dft = pd.get_dummies( test_dfn, \n                                        columns = cat_f,\n                                        drop_first = True )\n\ngc.collect()\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-13T13:39:28.398368Z","iopub.execute_input":"2022-07-13T13:39:28.398819Z","iopub.status.idle":"2022-07-13T13:39:54.301739Z","shell.execute_reply.started":"2022-07-13T13:39:28.398782Z","shell.execute_reply":"2022-07-13T13:39:54.300478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\npred_yt = search.predict(encoded_dft)\nfinal_probs = search.predict_proba(encoded_dft)[:,1]\ndel encoded_dft\ngc.collect()\ntest_dfn['prediction'] = final_probs\n","metadata":{"execution":{"iopub.status.busy":"2022-07-13T13:39:54.303640Z","iopub.execute_input":"2022-07-13T13:39:54.303983Z","iopub.status.idle":"2022-07-13T13:39:57.527830Z","shell.execute_reply.started":"2022-07-13T13:39:54.303954Z","shell.execute_reply":"2022-07-13T13:39:57.526524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dfn['customer_ID'] = c_id\n\ntest_dfn","metadata":{"execution":{"iopub.status.busy":"2022-07-13T13:39:57.529231Z","iopub.execute_input":"2022-07-13T13:39:57.529596Z","iopub.status.idle":"2022-07-13T13:39:57.847707Z","shell.execute_reply.started":"2022-07-13T13:39:57.529562Z","shell.execute_reply":"2022-07-13T13:39:57.845985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ntest_dfn = test_dfn[['customer_ID','prediction']]\n\ntest_dfn['predicted_b'] = test_dfn.prediction.map( \n                            lambda x: 1 if x > 0.5 else 0)\n\ntest_dfn.prediction = test_dfn.predicted_b\n\ntest_dfn = test_dfn[['customer_ID','prediction']]\n","metadata":{"execution":{"iopub.status.busy":"2022-07-13T13:39:57.851035Z","iopub.execute_input":"2022-07-13T13:39:57.851532Z","iopub.status.idle":"2022-07-13T13:39:58.207705Z","shell.execute_reply.started":"2022-07-13T13:39:57.851482Z","shell.execute_reply":"2022-07-13T13:39:58.206323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dfn.columns","metadata":{"execution":{"iopub.status.busy":"2022-07-13T13:39:58.209513Z","iopub.execute_input":"2022-07-13T13:39:58.210034Z","iopub.status.idle":"2022-07-13T13:39:58.216846Z","shell.execute_reply.started":"2022-07-13T13:39:58.209988Z","shell.execute_reply":"2022-07-13T13:39:58.216063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dfn=test_dfn.reset_index()\ntest_dfn=test_dfn.drop(\"index\",axis=1)\ntest_dfn","metadata":{"execution":{"iopub.status.busy":"2022-07-13T13:39:58.217938Z","iopub.execute_input":"2022-07-13T13:39:58.218980Z","iopub.status.idle":"2022-07-13T13:39:58.347218Z","shell.execute_reply.started":"2022-07-13T13:39:58.218944Z","shell.execute_reply":"2022-07-13T13:39:58.346060Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dfn = test_dfn.drop_duplicates(['customer_ID'])","metadata":{"execution":{"iopub.status.busy":"2022-07-13T13:40:01.723867Z","iopub.execute_input":"2022-07-13T13:40:01.724283Z","iopub.status.idle":"2022-07-13T13:40:02.122740Z","shell.execute_reply.started":"2022-07-13T13:40:01.724251Z","shell.execute_reply":"2022-07-13T13:40:02.121422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dfn ","metadata":{"execution":{"iopub.status.busy":"2022-07-13T13:37:02.327652Z","iopub.execute_input":"2022-07-13T13:37:02.328059Z","iopub.status.idle":"2022-07-13T13:37:02.346987Z","shell.execute_reply.started":"2022-07-13T13:37:02.328025Z","shell.execute_reply":"2022-07-13T13:37:02.345782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dfn.to_csv(\"submission.csv\",index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T13:40:53.173647Z","iopub.execute_input":"2022-07-13T13:40:53.174042Z","iopub.status.idle":"2022-07-13T13:40:54.375063Z","shell.execute_reply.started":"2022-07-13T13:40:53.174012Z","shell.execute_reply":"2022-07-13T13:40:54.373492Z"},"trusted":true},"execution_count":null,"outputs":[]}]}