{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-12T11:07:00.809376Z","iopub.execute_input":"2022-07-12T11:07:00.809852Z","iopub.status.idle":"2022-07-12T11:07:00.852290Z","shell.execute_reply.started":"2022-07-12T11:07:00.809763Z","shell.execute_reply":"2022-07-12T11:07:00.851169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore')\n\ndf = pd.read_feather('../input/amexfeather/train_data.ftr')\n\n#df.info(verbose=True,show_counts=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:07:00.854392Z","iopub.execute_input":"2022-07-12T11:07:00.855176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sn\nplt.figure(figsize=(5,5))\n\nax = sn.countplot(x=\"target\", data=df)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_f = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68'] \nall_f = list(df.columns)\nall_f.remove(\"customer_ID\")\nall_f.remove(\"S_2\")\nall_f.remove(\"D_142\")\n#finding set of numerical features by cosnducting simple set operations\nnum_f = list(set(all_f) - set(cat_f))\n\nprint(num_f[0:5])\n\ndf = df[all_f]\n\n#find null data\nprint(df.isnull().sum())\n\n#drop columns with less than 20 % data\nperc = 20.0 # Like N %\nmin_count =  int(((100-perc)/100)*df.shape[0] + 1)\ndf = df.dropna( axis=1, \n                thresh=min_count)\n\ndf=df.dropna()\ndf=df.reset_index()\ndf=df.drop(\"index\",axis=1)\n\nall_f = list(df.columns)\n\n#finding set of numerical features by cosnducting simple set operations\nnum_f = list(set(all_f) - set(cat_f))\n\ncat_f = list(set(all_f) - set(num_f))\n\nall_f=list(df.columns)\nall_f.remove(\"target\")\n\nencoded_df = pd.get_dummies( df[all_f], \n                                        columns = cat_f,\n                                        drop_first = True )\n\nX = encoded_df\nY = df['target']\n\nfrom sklearn.model_selection import train_test_split\n\ntrain_X, test_X, train_y, test_y = train_test_split( X,\n                                                    Y,\n                                                    test_size = 0.3,\n                                                    random_state = 42 )\n\n#We need to scale our feature in a standard scale, it's our pre-processor\nfrom sklearn.preprocessing import StandardScaler\n\nsc = StandardScaler()\ntrain_X = sc.fit_transform(train_X)\ntest_X = sc.transform(test_X)\n\nfrom sklearn.linear_model import LogisticRegression\n\n# grid search solver to find best fit\nfrom sklearn.datasets import make_classification\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.model_selection import RepeatedStratifiedKFold\n\n#define class weight dictionary, negative class has 20x weight\nw = {0:20, 1:1}\n\n# define dataset\nX, y = make_classification(n_samples=100000, n_features=176, n_redundant=0, random_state=1)\n# define model\nmodel = LogisticRegression(random_state=1, class_weight=w)\n# define model evaluation method\ncv = RepeatedStratifiedKFold(n_splits=7, n_repeats=3, random_state=1)\n# define grid\ngrid = dict()\ngrid['solver'] = ['liblinear', 'newton-cg', 'lbfgs', 'sag', 'saga']\n# define search\nsearch = GridSearchCV(model, grid, scoring='roc_auc', cv=cv, n_jobs=1)\n# perform the search\nresults = search.fit(X, y)\n# summarize\nprint('Mean Accuracy: %.3f' % results.best_score_)\nprint('Config: %s' % results.best_params_)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"probs1= search.predict_proba(X)[:,1]\nprobs1","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score, roc_curve, auc\n\ndef evaluate_roc(probs, y_true):\n    \"\"\"\n    - Print AUC and accuracy on the test set\n    - Plot ROC\n  \n    \"\"\"\n    preds = probs1\n    fpr, tpr, threshold = roc_curve(y_true, preds)\n    roc_auc = auc(fpr, tpr)\n    print(f'AUC: {roc_auc:.4f}')\n       \n    # Get accuracy over the test set\n    y_pred = np.where(preds >= 0.5, 1, 0)\n    accuracy = accuracy_score(y_true, y_pred)\n    print(f'Accuracy: {accuracy*100:.2f}%')\n    \n    # Plot ROC AUC\n    plt.title('Receiver Operating Characteristic')\n    plt.plot(fpr, tpr, 'b', label = 'AUC = %0.2f' % roc_auc)\n    plt.legend(loc = 'lower right')\n    plt.plot([0, 1], [0, 1],'r--')\n    plt.xlim([0, 1])\n    plt.ylim([0, 1])\n    plt.ylabel('True Positive Rate')\n    plt.xlabel('False Positive Rate')\n    plt.show()\n    \n# Evaluate the classifier\nevaluate_roc(probs1,y)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\n\ndel df,encoded_df\n\ngc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del X,train_X,test_X\n\ngc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.read_feather('/kaggle/input/amexfeather/test_data.ftr')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_f.append(\"customer_ID\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dfn = test_df[all_f]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del test_df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dfn = test_dfn.dropna().sample(1000000)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"c_id = test_dfn.customer_ID","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_f.remove(\"customer_ID\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dfn = test_dfn[all_f]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"encoded_dft = pd.get_dummies( test_dfn, \n                                        columns = cat_f,\n                                        drop_first = True )","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ngc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_yt = search.predict(encoded_dft)\nfinal_probs = search.predict_proba(encoded_dft)[:,1]\ndel encoded_dft\ngc.collect()\ntest_dfn['prediction'] = final_probs","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dfn['customer_ID'] = c_id","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dfn","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dfn = test_dfn[['customer_ID','prediction']]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dfn","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dfn.prediction.describe()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dfn['predicted_b'] = test_dfn.prediction.map( \n                            lambda x: 1 if x > 0.5 else 0)\n\ntest_dfn.sample(100, random_state = 42)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dfn.predicted_b.value_counts()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dfn.prediction = test_dfn.predicted_b","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dfn = test_dfn[['customer_ID','prediction']]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dfn","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dfn.to_csv(\"submission.csv\",index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}