{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-24T18:11:52.428345Z","iopub.execute_input":"2022-08-24T18:11:52.428788Z","iopub.status.idle":"2022-08-24T18:11:52.441275Z","shell.execute_reply.started":"2022-08-24T18:11:52.428753Z","shell.execute_reply":"2022-08-24T18:11:52.439702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Import libraries","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\nimport matplotlib.pylab as plt\n%matplotlib inline\n","metadata":{"execution":{"iopub.status.busy":"2022-08-24T18:11:52.879172Z","iopub.execute_input":"2022-08-24T18:11:52.879974Z","iopub.status.idle":"2022-08-24T18:11:52.887735Z","shell.execute_reply.started":"2022-08-24T18:11:52.879929Z","shell.execute_reply":"2022-08-24T18:11:52.886125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn import preprocessing\nfrom sklearn.decomposition import PCA","metadata":{"execution":{"iopub.status.busy":"2022-08-24T18:11:53.182137Z","iopub.execute_input":"2022-08-24T18:11:53.182959Z","iopub.status.idle":"2022-08-24T18:11:53.188390Z","shell.execute_reply.started":"2022-08-24T18:11:53.182918Z","shell.execute_reply":"2022-08-24T18:11:53.186897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.impute import SimpleImputer\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.ensemble import GradientBoostingClassifier\nfrom sklearn.metrics import accuracy_score, precision_score, recall_score\nfrom time import time","metadata":{"execution":{"iopub.status.busy":"2022-08-24T18:11:53.858599Z","iopub.execute_input":"2022-08-24T18:11:53.859066Z","iopub.status.idle":"2022-08-24T18:11:53.865710Z","shell.execute_reply.started":"2022-08-24T18:11:53.859029Z","shell.execute_reply":"2022-08-24T18:11:53.864299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Read and Explore data\nrequires more memory than available, so lets start with only 1000 rows","metadata":{"execution":{"iopub.status.busy":"2022-08-08T14:21:22.862413Z","iopub.execute_input":"2022-08-08T14:21:22.862864Z","iopub.status.idle":"2022-08-08T14:21:22.868480Z","shell.execute_reply.started":"2022-08-08T14:21:22.862827Z","shell.execute_reply":"2022-08-08T14:21:22.867159Z"}}},{"cell_type":"code","source":"train_data= pd.read_csv('/kaggle/input/amex-default-prediction/train_data.csv', nrows=1000)\ntest_data= pd.read_csv('/kaggle/input/amex-default-prediction/test_data.csv', nrows=1000)\ntrain_labels= pd.read_csv('/kaggle/input/amex-default-prediction/train_labels.csv', nrows=1000)","metadata":{"execution":{"iopub.status.busy":"2022-08-24T18:11:54.511568Z","iopub.execute_input":"2022-08-24T18:11:54.512359Z","iopub.status.idle":"2022-08-24T18:11:54.635207Z","shell.execute_reply.started":"2022-08-24T18:11:54.512305Z","shell.execute_reply":"2022-08-24T18:11:54.634160Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-24T18:11:55.693516Z","iopub.execute_input":"2022-08-24T18:11:55.694706Z","iopub.status.idle":"2022-08-24T18:11:55.728908Z","shell.execute_reply.started":"2022-08-24T18:11:55.694651Z","shell.execute_reply":"2022-08-24T18:11:55.727586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Correlation of various features without target variable as well as correlation with other features. ","metadata":{}},{"cell_type":"code","source":"train_data.corr()","metadata":{"execution":{"iopub.status.busy":"2022-08-24T18:12:45.018116Z","iopub.execute_input":"2022-08-24T18:12:45.018630Z","iopub.status.idle":"2022-08-24T18:12:45.130597Z","shell.execute_reply.started":"2022-08-24T18:12:45.018591Z","shell.execute_reply":"2022-08-24T18:12:45.129207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### remove categorical columns","metadata":{"execution":{"iopub.status.busy":"2022-08-08T15:31:24.680445Z","iopub.execute_input":"2022-08-08T15:31:24.680918Z","iopub.status.idle":"2022-08-08T15:31:24.686406Z","shell.execute_reply.started":"2022-08-08T15:31:24.680883Z","shell.execute_reply":"2022-08-08T15:31:24.684954Z"}}},{"cell_type":"code","source":"cols_without_categorical_data = [col for col in train_data.columns if train_data[col].dtypes != 'object'] \ntrain_set = train_data[cols_without_categorical_data]\ntest_set = test_data[cols_without_categorical_data]","metadata":{"execution":{"iopub.status.busy":"2022-08-24T18:12:45.399221Z","iopub.execute_input":"2022-08-24T18:12:45.400024Z","iopub.status.idle":"2022-08-24T18:12:45.421235Z","shell.execute_reply.started":"2022-08-24T18:12:45.399962Z","shell.execute_reply":"2022-08-24T18:12:45.419971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Alternate way of removing categorical data:\n#X_train = train_data.select_dtypes(exclude=['object'])","metadata":{"execution":{"iopub.status.busy":"2022-08-24T18:12:45.552878Z","iopub.execute_input":"2022-08-24T18:12:45.553998Z","iopub.status.idle":"2022-08-24T18:12:45.559415Z","shell.execute_reply.started":"2022-08-24T18:12:45.553941Z","shell.execute_reply":"2022-08-24T18:12:45.558059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#find if there are missing values\ntrain_set.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-24T18:12:45.761086Z","iopub.execute_input":"2022-08-24T18:12:45.761863Z","iopub.status.idle":"2022-08-24T18:12:45.775079Z","shell.execute_reply.started":"2022-08-24T18:12:45.761804Z","shell.execute_reply":"2022-08-24T18:12:45.773713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#find if there are flatline features\nstds= pd.DataFrame(train_data.describe().loc['std'])\nstds[stds['std']==0 ]","metadata":{"execution":{"iopub.status.busy":"2022-08-24T18:12:46.590924Z","iopub.execute_input":"2022-08-24T18:12:46.591363Z","iopub.status.idle":"2022-08-24T18:12:47.074113Z","shell.execute_reply.started":"2022-08-24T18:12:46.591329Z","shell.execute_reply":"2022-08-24T18:12:47.072526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#find if there are features with no values\nstds.isnull()[stds.isnull()['std']==True].index","metadata":{"execution":{"iopub.status.busy":"2022-08-24T18:12:47.076778Z","iopub.execute_input":"2022-08-24T18:12:47.077216Z","iopub.status.idle":"2022-08-24T18:12:47.086995Z","shell.execute_reply.started":"2022-08-24T18:12:47.077179Z","shell.execute_reply":"2022-08-24T18:12:47.085955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# drop the features that have 0 std (meaning flatline) or have null std (meaning no values present)\ntrain_set.drop(columns=['D_66', 'D_116', 'D_73', 'D_87', 'D_88', 'D_110', 'D_111', 'B_39'], inplace=True)\ntest_set.drop(columns=['D_66', 'D_116', 'D_73', 'D_87', 'D_88', 'D_110', 'D_111', 'B_39'], inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-24T18:12:47.333770Z","iopub.execute_input":"2022-08-24T18:12:47.334270Z","iopub.status.idle":"2022-08-24T18:12:47.347212Z","shell.execute_reply.started":"2022-08-24T18:12:47.334231Z","shell.execute_reply":"2022-08-24T18:12:47.345453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Impute missing values","metadata":{"execution":{"iopub.status.busy":"2022-08-24T18:14:00.246049Z","iopub.execute_input":"2022-08-24T18:14:00.246479Z","iopub.status.idle":"2022-08-24T18:14:00.252031Z","shell.execute_reply.started":"2022-08-24T18:14:00.246445Z","shell.execute_reply":"2022-08-24T18:14:00.250567Z"}}},{"cell_type":"code","source":"#impute nans with mean\nimputer = SimpleImputer(missing_values=np.nan, strategy='mean')\nimputer = imputer.fit(train_set)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-24T18:20:26.491652Z","iopub.execute_input":"2022-08-24T18:20:26.492188Z","iopub.status.idle":"2022-08-24T18:20:26.507408Z","shell.execute_reply.started":"2022-08-24T18:20:26.492148Z","shell.execute_reply":"2022-08-24T18:20:26.505439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_set = pd.DataFrame(imputer.transform(train_set), columns=train_set.columns)\ntest_input= pd.DataFrame(imputer.transform(test_set), columns=test_set.columns)","metadata":{"execution":{"iopub.status.busy":"2022-08-24T18:20:27.533334Z","iopub.execute_input":"2022-08-24T18:20:27.533803Z","iopub.status.idle":"2022-08-24T18:20:27.556955Z","shell.execute_reply.started":"2022-08-24T18:20:27.533763Z","shell.execute_reply":"2022-08-24T18:20:27.555213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Normalize the data","metadata":{}},{"cell_type":"code","source":"scaler = preprocessing.MinMaxScaler()\nscaler = scaler.fit(train_set)","metadata":{"execution":{"iopub.status.busy":"2022-08-24T18:20:30.911545Z","iopub.execute_input":"2022-08-24T18:20:30.912404Z","iopub.status.idle":"2022-08-24T18:20:30.924355Z","shell.execute_reply.started":"2022-08-24T18:20:30.912359Z","shell.execute_reply":"2022-08-24T18:20:30.922907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_set=pd.DataFrame(scaler.transform(train_set), columns=train_set.columns)\ntest_input=pd.DataFrame(scaler.transform(test_input), columns=test_set.columns)","metadata":{"execution":{"iopub.status.busy":"2022-08-24T18:20:34.630267Z","iopub.execute_input":"2022-08-24T18:20:34.631562Z","iopub.status.idle":"2022-08-24T18:20:34.649564Z","shell.execute_reply.started":"2022-08-24T18:20:34.631507Z","shell.execute_reply":"2022-08-24T18:20:34.647823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### PCA to visualize data in 2 dimension","metadata":{"execution":{"iopub.status.busy":"2022-08-24T18:21:30.364179Z","iopub.execute_input":"2022-08-24T18:21:30.364774Z","iopub.status.idle":"2022-08-24T18:21:30.370374Z","shell.execute_reply.started":"2022-08-24T18:21:30.364732Z","shell.execute_reply":"2022-08-24T18:21:30.369276Z"}}},{"cell_type":"code","source":"pca = PCA(2)\nx_pca = pca.fit_transform(train_set)\nx_pca = pd.DataFrame(x_pca)\nx_pca.columns=['PC1','PC2']","metadata":{"execution":{"iopub.status.busy":"2022-08-24T18:24:33.817102Z","iopub.execute_input":"2022-08-24T18:24:33.817586Z","iopub.status.idle":"2022-08-24T18:24:33.876212Z","shell.execute_reply.started":"2022-08-24T18:24:33.817547Z","shell.execute_reply":"2022-08-24T18:24:33.874857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"di={'0':'blue','1':'orange','-1':'green'}\nplt.scatter(x_pca['PC1'], x_pca['PC2'], alpha=0.8, color=[di[str(i)] for i in train_labels['target']] )\nplt.title('Scatter plot')\nplt.xlabel('PC1')\nplt.ylabel('PC2')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-24T18:27:01.715690Z","iopub.execute_input":"2022-08-24T18:27:01.716440Z","iopub.status.idle":"2022-08-24T18:27:02.029481Z","shell.execute_reply.started":"2022-08-24T18:27:01.716369Z","shell.execute_reply":"2022-08-24T18:27:02.028082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Above plot indicates that there is no clear separation between the customers that default the loan from those that do not.","metadata":{"execution":{"iopub.status.busy":"2022-08-24T18:25:51.004412Z","iopub.execute_input":"2022-08-24T18:25:51.004878Z","iopub.status.idle":"2022-08-24T18:25:51.012883Z","shell.execute_reply.started":"2022-08-24T18:25:51.004828Z","shell.execute_reply":"2022-08-24T18:25:51.011350Z"}}},{"cell_type":"markdown","source":"## Baseline Model","metadata":{}},{"cell_type":"code","source":"submission_df=pd.DataFrame({'customer_ID': test_data.customer_ID})","metadata":{"execution":{"iopub.status.busy":"2022-08-24T17:52:59.257971Z","iopub.execute_input":"2022-08-24T17:52:59.259026Z","iopub.status.idle":"2022-08-24T17:52:59.265359Z","shell.execute_reply.started":"2022-08-24T17:52:59.258973Z","shell.execute_reply":"2022-08-24T17:52:59.264032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y=train_labels['target']\nX= train_set\nX_train, X_test, y_train, y_test = train_test_split(X, y, random_state=18)\nprint (len(X_train), len(X_test))","metadata":{"execution":{"iopub.status.busy":"2022-08-24T17:53:00.980396Z","iopub.execute_input":"2022-08-24T17:53:00.980878Z","iopub.status.idle":"2022-08-24T17:53:00.994360Z","shell.execute_reply.started":"2022-08-24T17:53:00.980826Z","shell.execute_reply":"2022-08-24T17:53:00.993143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Model Training","metadata":{}},{"cell_type":"code","source":"clf = RandomForestClassifier()\nclf = clf.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-24T17:53:03.419000Z","iopub.execute_input":"2022-08-24T17:53:03.419656Z","iopub.status.idle":"2022-08-24T17:53:04.300670Z","shell.execute_reply.started":"2022-08-24T17:53:03.419620Z","shell.execute_reply":"2022-08-24T17:53:04.299577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"XB = GradientBoostingClassifier(max_depth = 10,n_estimators=20,verbose=True)\n%time XB.fit(X_train,y_train)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Model Accuracy (Validation)","metadata":{}},{"cell_type":"code","source":"predict = XB.predict(X_test)\nprint(\"Accuracy Score for GradientBoostingClassifier \",accuracy_score(predict,y_test))\npredict = clf.predict(X_test)\nprint(\"Accuracy Score for RandomForestClassifier \",accuracy_score(predict,y_test))\n","metadata":{"execution":{"iopub.status.busy":"2022-08-24T17:58:35.526715Z","iopub.execute_input":"2022-08-24T17:58:35.527175Z","iopub.status.idle":"2022-08-24T17:58:35.564310Z","shell.execute_reply.started":"2022-08-24T17:58:35.527139Z","shell.execute_reply":"2022-08-24T17:58:35.562519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Save results","metadata":{}},{"cell_type":"code","source":"result=clf.predict(test_input)\nsubmission_df['label']=result\nsubmission_df.to_csv('submission.csv', index=False)","metadata":{},"execution_count":null,"outputs":[]}]}