{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-08T20:31:36.342229Z","iopub.execute_input":"2023-05-08T20:31:36.343111Z","iopub.status.idle":"2023-05-08T20:31:36.357430Z","shell.execute_reply.started":"2023-05-08T20:31:36.343074Z","shell.execute_reply":"2023-05-08T20:31:36.356600Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.utils import resample\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.tree import DecisionTreeClassifier\nfrom xgboost import XGBClassifier\nfrom sklearn.model_selection import cross_validate\nfrom sklearn.metrics import confusion_matrix, classification_report, accuracy_score","metadata":{"execution":{"iopub.status.busy":"2023-05-08T20:31:36.361476Z","iopub.execute_input":"2023-05-08T20:31:36.361732Z","iopub.status.idle":"2023-05-08T20:31:37.454204Z","shell.execute_reply.started":"2023-05-08T20:31:36.361710Z","shell.execute_reply":"2023-05-08T20:31:37.453241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.set_option('display.max_columns', None)","metadata":{"execution":{"iopub.status.busy":"2023-05-08T20:31:37.456037Z","iopub.execute_input":"2023-05-08T20:31:37.456383Z","iopub.status.idle":"2023-05-08T20:31:37.461679Z","shell.execute_reply.started":"2023-05-08T20:31:37.456339Z","shell.execute_reply":"2023-05-08T20:31:37.460622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Data Preprocessing\ndf_train = pd.read_csv('/kaggle/input/amex-default-prediction/train_data.csv',nrows=100000)\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-08T20:31:37.463371Z","iopub.execute_input":"2023-05-08T20:31:37.464036Z","iopub.status.idle":"2023-05-08T20:31:44.737406Z","shell.execute_reply.started":"2023-05-08T20:31:37.464005Z","shell.execute_reply":"2023-05-08T20:31:44.736333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-08T20:31:44.740313Z","iopub.execute_input":"2023-05-08T20:31:44.740963Z","iopub.status.idle":"2023-05-08T20:31:44.747400Z","shell.execute_reply.started":"2023-05-08T20:31:44.740929Z","shell.execute_reply":"2023-05-08T20:31:44.746408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_label = pd.read_csv('/kaggle/input/amex-default-prediction/train_labels.csv',nrows=100000)\ndf_train_label.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-08T20:31:44.748620Z","iopub.execute_input":"2023-05-08T20:31:44.749605Z","iopub.status.idle":"2023-05-08T20:31:44.945546Z","shell.execute_reply.started":"2023-05-08T20:31:44.749544Z","shell.execute_reply":"2023-05-08T20:31:44.944671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_label.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-08T20:31:44.946922Z","iopub.execute_input":"2023-05-08T20:31:44.947261Z","iopub.status.idle":"2023-05-08T20:31:44.955653Z","shell.execute_reply.started":"2023-05-08T20:31:44.947232Z","shell.execute_reply":"2023-05-08T20:31:44.954565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#merge\ndf_train = pd.merge(df_train,df_train_label,how='inner',on=['customer_ID'])\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-08T20:31:44.957270Z","iopub.execute_input":"2023-05-08T20:31:44.957838Z","iopub.status.idle":"2023-05-08T20:31:45.246417Z","shell.execute_reply.started":"2023-05-08T20:31:44.957796Z","shell.execute_reply":"2023-05-08T20:31:45.245543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-08T20:31:45.247781Z","iopub.execute_input":"2023-05-08T20:31:45.248379Z","iopub.status.idle":"2023-05-08T20:31:45.254719Z","shell.execute_reply.started":"2023-05-08T20:31:45.248329Z","shell.execute_reply":"2023-05-08T20:31:45.253648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.drop(columns=['customer_ID'],inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-05-08T20:51:24.221642Z","iopub.execute_input":"2023-05-08T20:51:24.222389Z","iopub.status.idle":"2023-05-08T20:51:24.259546Z","shell.execute_reply.started":"2023-05-08T20:51:24.222334Z","shell.execute_reply":"2023-05-08T20:51:24.258422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2023-05-08T20:31:45.256423Z","iopub.execute_input":"2023-05-08T20:31:45.256899Z","iopub.status.idle":"2023-05-08T20:31:45.341684Z","shell.execute_reply.started":"2023-05-08T20:31:45.256868Z","shell.execute_reply":"2023-05-08T20:31:45.340675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"total = df_train.isnull().sum().sort_values(ascending=False)\npercent = (df_train.isnull().sum()/df_train.isnull().count()).sort_values(ascending=False)\nmissing_data = pd.concat([total, percent*100], axis=1, keys=['Total', 'Percent'])\nmissing_data.head(50)","metadata":{"execution":{"iopub.status.busy":"2023-05-08T20:31:45.345796Z","iopub.execute_input":"2023-05-08T20:31:45.346059Z","iopub.status.idle":"2023-05-08T20:31:45.595261Z","shell.execute_reply.started":"2023-05-08T20:31:45.346036Z","shell.execute_reply":"2023-05-08T20:31:45.594254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df_train = df_train.dropna()\n\n#drop variables with missing values >=2% in the train dataframe\ni=0\nfor col in df_train.columns:\n    if (df_train[col].isnull().sum()/len(df_train[col])*100) >=2:\n        print(\"Dropping column\", col)\n        df_train.drop(labels=col,axis=1,inplace=True)\n        i=i+1\n        \nprint(\"Total number of columns dropped in train dataframe\", i)","metadata":{"execution":{"iopub.status.busy":"2023-05-08T20:31:45.596846Z","iopub.execute_input":"2023-05-08T20:31:45.597166Z","iopub.status.idle":"2023-05-08T20:31:48.319124Z","shell.execute_reply.started":"2023-05-08T20:31:45.597138Z","shell.execute_reply":"2023-05-08T20:31:48.318092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"total = df_train.isnull().sum().sort_values(ascending=False)\npercent = (df_train.isnull().sum()/df_train.isnull().count()).sort_values(ascending=False)\nmissing_data = pd.concat([total, percent*100], axis=1, keys=['Total', 'Percent'])\nmissing_data.head(50)","metadata":{"execution":{"iopub.status.busy":"2023-05-08T20:31:48.320503Z","iopub.execute_input":"2023-05-08T20:31:48.321056Z","iopub.status.idle":"2023-05-08T20:31:48.500324Z","shell.execute_reply.started":"2023-05-08T20:31:48.321020Z","shell.execute_reply":"2023-05-08T20:31:48.499322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.isna().any().count()","metadata":{"execution":{"iopub.status.busy":"2023-05-08T20:31:48.501567Z","iopub.execute_input":"2023-05-08T20:31:48.502000Z","iopub.status.idle":"2023-05-08T20:31:48.557174Z","shell.execute_reply.started":"2023-05-08T20:31:48.501969Z","shell.execute_reply":"2023-05-08T20:31:48.556300Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-08T20:31:48.558601Z","iopub.execute_input":"2023-05-08T20:31:48.558933Z","iopub.status.idle":"2023-05-08T20:31:48.566435Z","shell.execute_reply.started":"2023-05-08T20:31:48.558902Z","shell.execute_reply":"2023-05-08T20:31:48.565378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train=df_train.fillna(0)","metadata":{"execution":{"iopub.status.busy":"2023-05-08T20:31:48.568121Z","iopub.execute_input":"2023-05-08T20:31:48.568811Z","iopub.status.idle":"2023-05-08T20:31:48.733196Z","shell.execute_reply.started":"2023-05-08T20:31:48.568781Z","shell.execute_reply":"2023-05-08T20:31:48.732228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.isna().any().count()","metadata":{"execution":{"iopub.status.busy":"2023-05-08T20:31:48.736274Z","iopub.execute_input":"2023-05-08T20:31:48.736901Z","iopub.status.idle":"2023-05-08T20:31:48.789911Z","shell.execute_reply.started":"2023-05-08T20:31:48.736867Z","shell.execute_reply":"2023-05-08T20:31:48.788919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.drop(columns=['S_2'],inplace=True)\n# df_train.drop(columns=['S_2','D_63'],inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-05-08T20:31:48.791054Z","iopub.execute_input":"2023-05-08T20:31:48.791393Z","iopub.status.idle":"2023-05-08T20:31:48.832691Z","shell.execute_reply.started":"2023-05-08T20:31:48.791344Z","shell.execute_reply":"2023-05-08T20:31:48.831662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"D_63 = {'CO':0, 'CR':1, 'CL':2, 'XZ':3, 'XM':4, 'XL':5}\ndf_train['D_63'].replace(D_63, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-05-08T20:31:48.834232Z","iopub.execute_input":"2023-05-08T20:31:48.834774Z","iopub.status.idle":"2023-05-08T20:31:48.879994Z","shell.execute_reply.started":"2023-05-08T20:31:48.834730Z","shell.execute_reply":"2023-05-08T20:31:48.879156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train\n","metadata":{"execution":{"iopub.status.busy":"2023-05-08T20:31:48.881181Z","iopub.execute_input":"2023-05-08T20:31:48.881606Z","iopub.status.idle":"2023-05-08T20:31:48.978972Z","shell.execute_reply.started":"2023-05-08T20:31:48.881575Z","shell.execute_reply":"2023-05-08T20:31:48.978174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-08T20:31:48.981031Z","iopub.execute_input":"2023-05-08T20:31:48.982013Z","iopub.status.idle":"2023-05-08T20:31:48.989029Z","shell.execute_reply.started":"2023-05-08T20:31:48.981981Z","shell.execute_reply":"2023-05-08T20:31:48.988075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = pd.read_csv('/kaggle/input/amex-default-prediction/test_data.csv',nrows=10000,index_col='customer_ID')\n","metadata":{"execution":{"iopub.status.busy":"2023-05-08T20:31:48.990144Z","iopub.execute_input":"2023-05-08T20:31:48.990452Z","iopub.status.idle":"2023-05-08T20:31:49.701332Z","shell.execute_reply.started":"2023-05-08T20:31:48.990430Z","shell.execute_reply":"2023-05-08T20:31:49.700402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-08T20:31:49.702753Z","iopub.execute_input":"2023-05-08T20:31:49.703080Z","iopub.status.idle":"2023-05-08T20:31:49.709475Z","shell.execute_reply.started":"2023-05-08T20:31:49.703049Z","shell.execute_reply":"2023-05-08T20:31:49.708556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-08T20:31:49.710762Z","iopub.execute_input":"2023-05-08T20:31:49.711745Z","iopub.status.idle":"2023-05-08T20:31:49.825332Z","shell.execute_reply.started":"2023-05-08T20:31:49.711713Z","shell.execute_reply":"2023-05-08T20:31:49.824545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2023-05-08T20:31:49.826222Z","iopub.execute_input":"2023-05-08T20:31:49.826530Z","iopub.status.idle":"2023-05-08T20:31:49.843376Z","shell.execute_reply.started":"2023-05-08T20:31:49.826495Z","shell.execute_reply":"2023-05-08T20:31:49.842360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df_test = df_test.dropna()\n\n#drop variables with missing values >=2% in the test dataframe\ni=0\nfor col in df_test.columns:\n    if (df_test[col].isnull().sum()/len(df_test[col])*100) >=2:\n        print(\"Dropping column\", col)\n        df_test.drop(labels=col,axis=1,inplace=True)\n        i=i+1\n        \nprint(\"Total number of columns dropped in train dataframe\", i)","metadata":{"execution":{"iopub.status.busy":"2023-05-08T20:31:49.844821Z","iopub.execute_input":"2023-05-08T20:31:49.845221Z","iopub.status.idle":"2023-05-08T20:31:50.016976Z","shell.execute_reply.started":"2023-05-08T20:31:49.845192Z","shell.execute_reply":"2023-05-08T20:31:50.015879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.isna().any().count()","metadata":{"execution":{"iopub.status.busy":"2023-05-08T20:31:50.018554Z","iopub.execute_input":"2023-05-08T20:31:50.018914Z","iopub.status.idle":"2023-05-08T20:31:50.030485Z","shell.execute_reply.started":"2023-05-08T20:31:50.018881Z","shell.execute_reply":"2023-05-08T20:31:50.029297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test=df_test.fillna(0)","metadata":{"execution":{"iopub.status.busy":"2023-05-08T20:31:50.031906Z","iopub.execute_input":"2023-05-08T20:31:50.032927Z","iopub.status.idle":"2023-05-08T20:31:50.045376Z","shell.execute_reply.started":"2023-05-08T20:31:50.032895Z","shell.execute_reply":"2023-05-08T20:31:50.044337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df_test.drop(columns=['customer_ID'],inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-05-08T20:53:12.886625Z","iopub.execute_input":"2023-05-08T20:53:12.886976Z","iopub.status.idle":"2023-05-08T20:53:12.890891Z","shell.execute_reply.started":"2023-05-08T20:53:12.886951Z","shell.execute_reply":"2023-05-08T20:53:12.889938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.drop(columns=['S_2'],inplace=True)\n\nD_63 = {'CO':0, 'CR':1, 'CL':2, 'XZ':3, 'XM':4, 'XL':5}\ndf_test['D_63'].replace(D_63, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-05-08T20:31:50.050497Z","iopub.execute_input":"2023-05-08T20:31:50.050751Z","iopub.status.idle":"2023-05-08T20:31:50.064684Z","shell.execute_reply.started":"2023-05-08T20:31:50.050729Z","shell.execute_reply":"2023-05-08T20:31:50.063810Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.shape, df_test.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-08T20:31:50.066233Z","iopub.execute_input":"2023-05-08T20:31:50.066583Z","iopub.status.idle":"2023-05-08T20:31:50.074108Z","shell.execute_reply.started":"2023-05-08T20:31:50.066553Z","shell.execute_reply":"2023-05-08T20:31:50.073169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-08T20:53:30.513597Z","iopub.execute_input":"2023-05-08T20:53:30.513962Z","iopub.status.idle":"2023-05-08T20:53:30.593243Z","shell.execute_reply.started":"2023-05-08T20:53:30.513934Z","shell.execute_reply":"2023-05-08T20:53:30.591551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-08T20:31:50.152820Z","iopub.execute_input":"2023-05-08T20:31:50.153146Z","iopub.status.idle":"2023-05-08T20:31:50.231603Z","shell.execute_reply.started":"2023-05-08T20:31:50.153117Z","shell.execute_reply":"2023-05-08T20:31:50.230430Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corrmat = df_train.corr().abs()\nf, ax = plt.subplots(figsize=(30, 30))\ncols = corrmat.nlargest(50, 'target')['target'].index\ncm = np.corrcoef(df_train[cols].values.T)\nsns.set(font_scale=1.2)\nhm = sns.heatmap(cm, cbar=True, annot=True, square=True, fmt='.2f', annot_kws={'size': 10},yticklabels=cols.values, xticklabels=cols.values)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-08T20:31:50.232693Z","iopub.execute_input":"2023-05-08T20:31:50.233540Z","iopub.status.idle":"2023-05-08T20:31:58.838211Z","shell.execute_reply.started":"2023-05-08T20:31:50.233507Z","shell.execute_reply":"2023-05-08T20:31:58.837403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(4,3))\nsns.countplot(x=df_train['target'])","metadata":{"execution":{"iopub.status.busy":"2023-05-08T20:31:58.839461Z","iopub.execute_input":"2023-05-08T20:31:58.839933Z","iopub.status.idle":"2023-05-08T20:31:59.048446Z","shell.execute_reply.started":"2023-05-08T20:31:58.839903Z","shell.execute_reply":"2023-05-08T20:31:59.047266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(20,20))\nfor i in range(len(cols[:25])):\n    plt.subplot(5,5, i+1) #the figure has 5 row, 5 columns, and this plot is the i-th plot.\n    sns.scatterplot(x=df_train[cols[i]], y=df_train['target'])\nplt.suptitle('Relation between target and 25 important features')","metadata":{"execution":{"iopub.status.busy":"2023-05-08T20:31:59.050086Z","iopub.execute_input":"2023-05-08T20:31:59.050844Z","iopub.status.idle":"2023-05-08T20:32:11.486692Z","shell.execute_reply.started":"2023-05-08T20:31:59.050807Z","shell.execute_reply":"2023-05-08T20:32:11.485838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(20,20))\nfor i in range(len(cols[:25])):\n    plt.subplot(5,5, i+1) #the figure has 3 row, 3 columns, and this plot is the i-th plot.\n    sns.boxplot(x=df_train[cols[i]])\nplt.suptitle('Values for 25 important features')","metadata":{"execution":{"iopub.status.busy":"2023-05-08T20:32:11.487914Z","iopub.execute_input":"2023-05-08T20:32:11.489031Z","iopub.status.idle":"2023-05-08T20:32:15.331151Z","shell.execute_reply.started":"2023-05-08T20:32:11.488999Z","shell.execute_reply":"2023-05-08T20:32:15.330269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# #removing outliers\n# #print(len(df_train[(df_train['B_9'] > 15)]))\n# df_train.drop(df_train[(df_train['B_9'] > 15)].index, inplace=True)\n# #print(len(df_train[(df_train['D_75'] > 3.5)]))\n# df_train.drop(df_train[(df_train['D_75'] > 3.5)].index, inplace=True)\n# #print(len(df_train[(df_train['B_7'] < -0.5)]))\n# df_train.drop(df_train[(df_train['B_7'] < -0.5)].index, inplace=True)\n# #print(len(df_train[(df_train['B_23'] > 1.5)]))\n# df_train.drop(df_train[(df_train['B_23'] > 1.5)].index, inplace=True)\n# #print(len(df_train[(df_train['B_4'] > 4)]))\n# df_train.drop(df_train[(df_train['B_4'] > 4)].index, inplace=True)\n# #print(len(df_train[(df_train['B_1'] < -0.5)]))\n# df_train.drop(df_train[(df_train['B_1'] < -0.5)].index, inplace=True)\n# #print(len(df_train[(df_train['B_11'] > 1.6)]))\n# df_train.drop(df_train[(df_train['B_11'] > 1.6)].index, inplace=True)\n# #print(len(df_train[(df_train['R_1'] > 2.6)]))\n# df_train.drop(df_train[(df_train['R_1'] > 2.6)].index, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-05-08T20:32:15.332556Z","iopub.execute_input":"2023-05-08T20:32:15.333555Z","iopub.status.idle":"2023-05-08T20:32:15.338264Z","shell.execute_reply.started":"2023-05-08T20:32:15.333522Z","shell.execute_reply":"2023-05-08T20:32:15.337388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#balance the data\n\n# df_majority_0 = df_train[(df_train['target']==0)] \n# df_minority_1 = df_train[(df_train['target']==1)] \n\n# df_minority_upsampled = resample(df_minority_1, \n#                                  replace=True,    \n#                                  n_samples= 4153544, \n#                                  random_state=44) \n\n# df_upsampled = pd.concat([df_minority_upsampled, df_majority_0])\n# # chart\n# plt.figure(figsize=(4,3))\n# sns.countplot(x=df_upsampled['target'])","metadata":{"execution":{"iopub.status.busy":"2023-05-08T20:32:15.339574Z","iopub.execute_input":"2023-05-08T20:32:15.340537Z","iopub.status.idle":"2023-05-08T20:32:15.353345Z","shell.execute_reply.started":"2023-05-08T20:32:15.340503Z","shell.execute_reply":"2023-05-08T20:32:15.352256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Creating independent X and dependent y variable","metadata":{}},{"cell_type":"code","source":"X = df_train.loc[:,df_train.columns != 'target']\ny = df_train['target']","metadata":{"execution":{"iopub.status.busy":"2023-05-08T20:53:57.368697Z","iopub.execute_input":"2023-05-08T20:53:57.369070Z","iopub.status.idle":"2023-05-08T20:53:57.407472Z","shell.execute_reply.started":"2023-05-08T20:53:57.369042Z","shell.execute_reply":"2023-05-08T20:53:57.405669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-08T20:54:04.613657Z","iopub.execute_input":"2023-05-08T20:54:04.614021Z","iopub.status.idle":"2023-05-08T20:54:04.695279Z","shell.execute_reply.started":"2023-05-08T20:54:04.613994Z","shell.execute_reply":"2023-05-08T20:54:04.694300Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-08T20:54:08.633739Z","iopub.execute_input":"2023-05-08T20:54:08.634335Z","iopub.status.idle":"2023-05-08T20:54:08.643263Z","shell.execute_reply.started":"2023-05-08T20:54:08.634301Z","shell.execute_reply":"2023-05-08T20:54:08.642228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-08T20:54:12.008483Z","iopub.execute_input":"2023-05-08T20:54:12.008885Z","iopub.status.idle":"2023-05-08T20:54:12.021200Z","shell.execute_reply.started":"2023-05-08T20:54:12.008856Z","shell.execute_reply":"2023-05-08T20:54:12.020391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-08T20:54:14.954629Z","iopub.execute_input":"2023-05-08T20:54:14.955068Z","iopub.status.idle":"2023-05-08T20:54:14.967148Z","shell.execute_reply.started":"2023-05-08T20:54:14.955031Z","shell.execute_reply":"2023-05-08T20:54:14.966192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Test Val Test split","metadata":{}},{"cell_type":"code","source":"X_train, X_val_test, y_train, y_val_test = train_test_split(X, y, train_size=0.98, shuffle=True, random_state=100)","metadata":{"execution":{"iopub.status.busy":"2023-05-08T20:54:17.888140Z","iopub.execute_input":"2023-05-08T20:54:17.888525Z","iopub.status.idle":"2023-05-08T20:54:17.971629Z","shell.execute_reply.started":"2023-05-08T20:54:17.888496Z","shell.execute_reply":"2023-05-08T20:54:17.970642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_val, X_test, y_val, y_test = train_test_split(X_val_test, y_val_test, train_size=0.5,shuffle=True,random_state=100)","metadata":{"execution":{"iopub.status.busy":"2023-05-08T20:54:22.082497Z","iopub.execute_input":"2023-05-08T20:54:22.083529Z","iopub.status.idle":"2023-05-08T20:54:22.092164Z","shell.execute_reply.started":"2023-05-08T20:54:22.083478Z","shell.execute_reply":"2023-05-08T20:54:22.091190Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Logistic regression","metadata":{}},{"cell_type":"code","source":"LogisticRegressionModel = LogisticRegression(penalty='l2',solver='sag',C=2.0,random_state=44)\nLogisticRegressionModel.fit(X_train, y_train)\n\n#Calculating Details\nprint('LogisticRegressionModel Train Score is : ' , LogisticRegressionModel.score(X_train, y_train))\nprint('LogisticRegressionModel Test Score is : ' , LogisticRegressionModel.score(X_test, y_test))","metadata":{"execution":{"iopub.status.busy":"2023-05-08T20:54:24.582258Z","iopub.execute_input":"2023-05-08T20:54:24.583593Z","iopub.status.idle":"2023-05-08T20:54:46.362833Z","shell.execute_reply.started":"2023-05-08T20:54:24.583548Z","shell.execute_reply":"2023-05-08T20:54:46.357400Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CrossValidateValues2 = cross_validate(LogisticRegressionModel, X_val, y_val, cv=5, return_train_score = True)\n\n# Showing Results\nprint('Train Score Value : ', CrossValidateValues2['train_score'])\nprint('Test Score Value : ', CrossValidateValues2['test_score'])\nprint('Fit Time : ', CrossValidateValues2['fit_time'])\nprint('Score Time : ', CrossValidateValues2['score_time'])","metadata":{"execution":{"iopub.status.busy":"2023-05-08T20:55:06.903745Z","iopub.execute_input":"2023-05-08T20:55:06.904117Z","iopub.status.idle":"2023-05-08T20:55:07.982601Z","shell.execute_reply.started":"2023-05-08T20:55:06.904088Z","shell.execute_reply":"2023-05-08T20:55:07.981494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = LogisticRegressionModel.predict(X_test)\n\nplt.figure(figsize=(4,3))\nCM = confusion_matrix(y_test, y_pred)\nsns.heatmap(CM, center=True)\nplt.show()\n\nprint('Confusion Matrix is\\n', CM)","metadata":{"execution":{"iopub.status.busy":"2023-05-08T20:55:31.595099Z","iopub.execute_input":"2023-05-08T20:55:31.596106Z","iopub.status.idle":"2023-05-08T20:55:31.854696Z","shell.execute_reply.started":"2023-05-08T20:55:31.596058Z","shell.execute_reply":"2023-05-08T20:55:31.853661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(classification_report(y_test, y_pred))\nprint(accuracy_score(y_test, y_pred))","metadata":{"execution":{"iopub.status.busy":"2023-05-08T20:55:56.615809Z","iopub.execute_input":"2023-05-08T20:55:56.616168Z","iopub.status.idle":"2023-05-08T20:55:56.632497Z","shell.execute_reply.started":"2023-05-08T20:55:56.616140Z","shell.execute_reply":"2023-05-08T20:55:56.631426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Decision Tree","metadata":{}},{"cell_type":"code","source":"DecisionTreeClassifierModel = DecisionTreeClassifier(criterion='entropy',max_depth=10,random_state=44)\nDecisionTreeClassifierModel.fit(X_train, y_train)\n\n#Calculating Details\nprint('DecisionTreeClassifierModel Train Score is : ' , DecisionTreeClassifierModel.score(X_train, y_train))\nprint('DecisionTreeClassifierModel Test Score is : ' , DecisionTreeClassifierModel.score(X_test, y_test))","metadata":{"execution":{"iopub.status.busy":"2023-05-08T20:56:27.343692Z","iopub.execute_input":"2023-05-08T20:56:27.344090Z","iopub.status.idle":"2023-05-08T20:56:45.914733Z","shell.execute_reply.started":"2023-05-08T20:56:27.344059Z","shell.execute_reply":"2023-05-08T20:56:45.913665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CrossValidateValues3 = cross_validate(DecisionTreeClassifierModel, X_val, y_val, cv=5, return_train_score = True)\n\n# Showing Results\nprint('Train Score Value : ', CrossValidateValues3['train_score'])\nprint('Test Score Value : ', CrossValidateValues3['test_score'])\nprint('Fit Time : ', CrossValidateValues3['fit_time'])\nprint('Score Time : ', CrossValidateValues3['score_time'])","metadata":{"execution":{"iopub.status.busy":"2023-05-08T20:56:53.633584Z","iopub.execute_input":"2023-05-08T20:56:53.634195Z","iopub.status.idle":"2023-05-08T20:56:54.018779Z","shell.execute_reply.started":"2023-05-08T20:56:53.634159Z","shell.execute_reply":"2023-05-08T20:56:54.017599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_DT = DecisionTreeClassifierModel.predict(X_test)\n\nplt.figure(figsize=(4,3))\n\nCM_DT = confusion_matrix(y_test, y_pred_DT)\nsns.heatmap(CM_DT, center=True)\nplt.show()\n\nprint('Confusion Matrix is\\n', CM_DT)","metadata":{"execution":{"iopub.status.busy":"2023-05-08T20:56:59.446069Z","iopub.execute_input":"2023-05-08T20:56:59.446453Z","iopub.status.idle":"2023-05-08T20:56:59.684696Z","shell.execute_reply.started":"2023-05-08T20:56:59.446422Z","shell.execute_reply":"2023-05-08T20:56:59.683642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(classification_report(y_test, y_pred_DT))\nprint(accuracy_score(y_test, y_pred_DT))","metadata":{"execution":{"iopub.status.busy":"2023-05-08T20:57:17.281689Z","iopub.execute_input":"2023-05-08T20:57:17.282262Z","iopub.status.idle":"2023-05-08T20:57:17.305018Z","shell.execute_reply.started":"2023-05-08T20:57:17.282220Z","shell.execute_reply":"2023-05-08T20:57:17.304003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"XG Boost","metadata":{}},{"cell_type":"code","source":"XGBClassifierModel = XGBClassifier(n_estimators=100, max_depth=15, eta=0.01, subsample=0.6, colsample_bytree=0.8) \nXGBClassifierModel.fit(X_train, y_train)\n\n#Calculating Details\nprint('XGBClassifierModel Train Score is : ' , XGBClassifierModel.score(X_train, y_train))\nprint('XGBClassifierModel Test Score is : ' , XGBClassifierModel.score(X_test, y_test))","metadata":{"execution":{"iopub.status.busy":"2023-05-08T20:59:14.624428Z","iopub.execute_input":"2023-05-08T20:59:14.624843Z","iopub.status.idle":"2023-05-08T21:05:01.439175Z","shell.execute_reply.started":"2023-05-08T20:59:14.624815Z","shell.execute_reply":"2023-05-08T21:05:01.438380Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CrossValidateValues4 = cross_validate(XGBClassifierModel, X_val, y_val, cv=5, return_train_score = True)\n\n# Showing Results\nprint('Train Score Value : ', CrossValidateValues4['train_score'])\nprint('Test Score Value : ', CrossValidateValues4['test_score'])\nprint('Fit Time : ', CrossValidateValues4['fit_time'])\nprint('Score Time : ', CrossValidateValues4['score_time'])","metadata":{"execution":{"iopub.status.busy":"2023-05-08T21:05:41.774503Z","iopub.execute_input":"2023-05-08T21:05:41.774875Z","iopub.status.idle":"2023-05-08T21:05:48.082966Z","shell.execute_reply.started":"2023-05-08T21:05:41.774848Z","shell.execute_reply":"2023-05-08T21:05:48.082267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_x = XGBClassifierModel.predict(X_test)\nCM_x = confusion_matrix(y_test, y_pred_x)\n\nplt.figure(figsize=(4,3))\nsns.heatmap(CM_x, center=True)\nplt.show()\n\nprint('Confusion Matrix is\\n', CM_x)","metadata":{"execution":{"iopub.status.busy":"2023-05-08T21:05:50.709309Z","iopub.execute_input":"2023-05-08T21:05:50.709685Z","iopub.status.idle":"2023-05-08T21:05:50.945209Z","shell.execute_reply.started":"2023-05-08T21:05:50.709651Z","shell.execute_reply":"2023-05-08T21:05:50.944301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(classification_report(y_test, y_pred_x))\nprint(accuracy_score(y_test, y_pred_x))","metadata":{"execution":{"iopub.status.busy":"2023-05-08T21:05:57.421292Z","iopub.execute_input":"2023-05-08T21:05:57.421666Z","iopub.status.idle":"2023-05-08T21:05:57.437274Z","shell.execute_reply.started":"2023-05-08T21:05:57.421637Z","shell.execute_reply":"2023-05-08T21:05:57.436270Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Testing","metadata":{}},{"cell_type":"code","source":"# x_test = df_test[cols] \n\n# y_pred_gbr = GBRModel.predict(x_test)\n# y_pred_rf = RandomForestRegressorModel.predict(x_test)\n# y_pred_x = XGBClassifierModel.predict(x_test)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#submission = df_test[[\"customer_ID\"]] \n#submission[\"prediction\"] = y_pred_x\n#submission.to_csv('output.csv', index=False)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from wordcloud import WordCloud \nthank_you_str=\"Pandas,SciKit,Machine Learning,Thanks,Happy Learning,Collaboration,Thankyou,Keep Learning,pattern recognition,unsupervised learning,automaton,computational learning theory,computer,connectionism,probability theory,statistics,machine,learning,data,robot,simulator,computer science\"\n# create WordCloud with converted string\nwordcloud = WordCloud(width = 1000, height = 500, colormap='rainbow', random_state=1, background_color='white', collocations=True).generate(thank_you_str)\nplt.figure(figsize=(20, 20))\nplt.imshow(wordcloud) \nplt.axis(\"off\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-08T21:30:18.673262Z","iopub.execute_input":"2023-05-08T21:30:18.673651Z","iopub.status.idle":"2023-05-08T21:30:19.305125Z","shell.execute_reply.started":"2023-05-08T21:30:18.673621Z","shell.execute_reply":"2023-05-08T21:30:19.304239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}