{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-09-22T12:11:46.695314Z","iopub.execute_input":"2022-09-22T12:11:46.695958Z","iopub.status.idle":"2022-09-22T12:11:46.711041Z","shell.execute_reply.started":"2022-09-22T12:11:46.695876Z","shell.execute_reply":"2022-09-22T12:11:46.709706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# importing SimpleImputer for handling missing value\nfrom sklearn.impute import SimpleImputer\n# importing MissingIndicator for handling missing value\nfrom sklearn.impute import MissingIndicator\n# importing StandardScaler for standardization\nfrom sklearn.preprocessing import StandardScaler\n# importing OnHotEncoder for encoding categorical variable\nfrom sklearn.preprocessing import OneHotEncoder, LabelEncoder\n# importing for transformation\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.compose import make_column_transformer\nfrom sklearn.compose import make_column_selector\n# importing PCA for handling dimensonality reduction\nfrom sklearn.decomposition import PCA\n\n# importing pipeline for chaining model building activities\n#from sklearn.pipeline import Pipeline\n#from sklearn.pipeline import make_pipeline\nfrom imblearn.pipeline import Pipeline\nfrom imblearn.pipeline import make_pipeline as mp\n# importing FeatureUnion for combining transformers\nfrom sklearn.pipeline import FeatureUnion\n\n# importing samplers for handling data imbalance\nfrom imblearn.combine import SMOTEENN \nfrom imblearn.over_sampling import SMOTE\nfrom imblearn.over_sampling import RandomOverSampler \nfrom imblearn.under_sampling import RandomUnderSampler \n\n# importing train_test_split for train and validation split\nfrom sklearn.model_selection import train_test_split\n# importing SelectFromModel to select features from model \nfrom sklearn.feature_selection import SelectFromModel   ","metadata":{"execution":{"iopub.status.busy":"2022-09-22T12:11:46.714074Z","iopub.execute_input":"2022-09-22T12:11:46.714586Z","iopub.status.idle":"2022-09-22T12:11:46.727097Z","shell.execute_reply.started":"2022-09-22T12:11:46.714547Z","shell.execute_reply":"2022-09-22T12:11:46.725588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Loading dataset train_data.csv\ntrain_df_sample = pd.read_csv('../input/amex-default-prediction/train_data.csv', nrows=100000)\ntrain_df_sample.head()","metadata":{"execution":{"iopub.status.busy":"2022-09-22T12:11:46.728996Z","iopub.execute_input":"2022-09-22T12:11:46.729398Z","iopub.status.idle":"2022-09-22T12:11:52.816924Z","shell.execute_reply.started":"2022-09-22T12:11:46.729366Z","shell.execute_reply":"2022-09-22T12:11:52.815771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df_sample.info()","metadata":{"execution":{"iopub.status.busy":"2022-09-22T12:11:52.819441Z","iopub.execute_input":"2022-09-22T12:11:52.819804Z","iopub.status.idle":"2022-09-22T12:11:52.846004Z","shell.execute_reply.started":"2022-09-22T12:11:52.819770Z","shell.execute_reply":"2022-09-22T12:11:52.844687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Loading dataset test_data.csv\ntest_df = pd.read_csv('../input/amex-default-prediction/test_data.csv', nrows=100000, index_col='customer_ID')\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-09-22T12:11:52.847789Z","iopub.execute_input":"2022-09-22T12:11:52.848223Z","iopub.status.idle":"2022-09-22T12:11:56.529596Z","shell.execute_reply.started":"2022-09-22T12:11:52.848180Z","shell.execute_reply":"2022-09-22T12:11:56.528640Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df=test_df.reset_index()\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-09-22T12:11:56.530911Z","iopub.execute_input":"2022-09-22T12:11:56.531252Z","iopub.status.idle":"2022-09-22T12:11:56.622659Z","shell.execute_reply.started":"2022-09-22T12:11:56.531220Z","shell.execute_reply":"2022-09-22T12:11:56.621488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Loading dataset train_labels.csv\ntrain_label_df = pd.read_csv('../input/amex-default-prediction/train_labels.csv')","metadata":{"execution":{"iopub.status.busy":"2022-09-22T12:11:56.624504Z","iopub.execute_input":"2022-09-22T12:11:56.625272Z","iopub.status.idle":"2022-09-22T12:11:57.230355Z","shell.execute_reply.started":"2022-09-22T12:11:56.625225Z","shell.execute_reply":"2022-09-22T12:11:57.229061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_label_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-09-22T12:11:57.231888Z","iopub.execute_input":"2022-09-22T12:11:57.232258Z","iopub.status.idle":"2022-09-22T12:11:57.243300Z","shell.execute_reply.started":"2022-09-22T12:11:57.232223Z","shell.execute_reply":"2022-09-22T12:11:57.241973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"raw","source":"","metadata":{}},{"cell_type":"code","source":"# Merge of train_df_sample and train_label_df dataframe using key as customer_ID\ntrain_df = pd.merge(train_df_sample, train_label_df, how=\"inner\", on=[\"customer_ID\"])","metadata":{"execution":{"iopub.status.busy":"2022-09-22T12:11:57.244853Z","iopub.execute_input":"2022-09-22T12:11:57.245269Z","iopub.status.idle":"2022-09-22T12:11:57.822174Z","shell.execute_reply.started":"2022-09-22T12:11:57.245232Z","shell.execute_reply":"2022-09-22T12:11:57.820957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print summary of merged dataframe\ntrain_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-09-22T12:11:57.826585Z","iopub.execute_input":"2022-09-22T12:11:57.826954Z","iopub.status.idle":"2022-09-22T12:11:57.852314Z","shell.execute_reply.started":"2022-09-22T12:11:57.826922Z","shell.execute_reply":"2022-09-22T12:11:57.851017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['target'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-09-22T12:11:57.854536Z","iopub.execute_input":"2022-09-22T12:11:57.855000Z","iopub.status.idle":"2022-09-22T12:11:57.867027Z","shell.execute_reply.started":"2022-09-22T12:11:57.854956Z","shell.execute_reply":"2022-09-22T12:11:57.865797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-09-22T12:11:57.868965Z","iopub.execute_input":"2022-09-22T12:11:57.869460Z","iopub.status.idle":"2022-09-22T12:11:57.900107Z","shell.execute_reply.started":"2022-09-22T12:11:57.869414Z","shell.execute_reply":"2022-09-22T12:11:57.898595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#dataset= train_df.merge(test_df, on= 'customer_ID', how='left')","metadata":{"execution":{"iopub.status.busy":"2022-09-22T12:11:57.901981Z","iopub.execute_input":"2022-09-22T12:11:57.902341Z","iopub.status.idle":"2022-09-22T12:11:57.909792Z","shell.execute_reply.started":"2022-09-22T12:11:57.902307Z","shell.execute_reply":"2022-09-22T12:11:57.908587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#dataset.shape","metadata":{"execution":{"iopub.status.busy":"2022-09-22T12:11:57.911490Z","iopub.execute_input":"2022-09-22T12:11:57.911938Z","iopub.status.idle":"2022-09-22T12:11:57.922928Z","shell.execute_reply.started":"2022-09-22T12:11:57.911888Z","shell.execute_reply":"2022-09-22T12:11:57.921448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#dataset['S_2']","metadata":{"execution":{"iopub.status.busy":"2022-09-22T12:11:57.924779Z","iopub.execute_input":"2022-09-22T12:11:57.925204Z","iopub.status.idle":"2022-09-22T12:11:57.933055Z","shell.execute_reply.started":"2022-09-22T12:11:57.925151Z","shell.execute_reply":"2022-09-22T12:11:57.931826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#dataset.head()","metadata":{"execution":{"iopub.status.busy":"2022-09-22T12:11:57.934174Z","iopub.execute_input":"2022-09-22T12:11:57.934551Z","iopub.status.idle":"2022-09-22T12:11:57.945676Z","shell.execute_reply.started":"2022-09-22T12:11:57.934496Z","shell.execute_reply":"2022-09-22T12:11:57.944468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process_data(df):\n   \n    #drop variables with missing values >=75% in the train dataframe\n    i=0\n    \n    for col in df.columns:\n        if (df[col].isnull().sum()/len(df[col])*100) >=75:\n            #print(\"Dropping column\", col)\n            df.drop(labels=col,axis=1,inplace=True)\n            i=i+1\n            #print(\"Total number of columns dropped in train dataframe\", i)\n    \n    data=df\n        \n    data['Year']=data['S_2'].apply(lambda x:np.int(x[0:4]))\n    data['Month']=data['S_2'].apply(lambda x:np.int(x[5:7]))\n    data['Day']=data['S_2'].apply(lambda x:np.int(x[8:10]))\n    data=data.drop('S_2', axis=1)\n    \n    \n    data_na=data\n    data_na= data_na.fillna(0)\n    data2=data_na.drop('customer_ID', axis=1)\n    \n    cat_data={column: len(data2[column].unique()) for column in data2.select_dtypes('object').columns}\n    \n   \n    onehotencoder=OneHotEncoder()\n    categorical_cols=['D_63','D_64','D_68','B_30','B_38','D_114','D_116','D_117','D_120','D_126']\n    dummies=pd.get_dummies(data2, columns=categorical_cols)\n\n    #the above transformed data is an array so convert it to a dataframe \n    dummy_data=pd.DataFrame(dummies, index= data2.index)\n\n    #now concatenate the original data and the dummified data using pandas\n    concatenated_data= pd.concat([data2, dummy_data], axis=1)\n    \n    data3= concatenated_data.drop(['D_63','D_64','D_68','B_30','B_38','D_114','D_116','D_117','D_120','D_126'], axis=1)\n    \n    data3= data3.astype(float)\n    data3=data3.astype(int)\n    \n    data3=data3.loc[:, ~data3.T.duplicated(keep='first')]\n    \n    return data3","metadata":{"execution":{"iopub.status.busy":"2022-09-22T12:11:57.946799Z","iopub.execute_input":"2022-09-22T12:11:57.947186Z","iopub.status.idle":"2022-09-22T12:11:57.966767Z","shell.execute_reply.started":"2022-09-22T12:11:57.947153Z","shell.execute_reply":"2022-09-22T12:11:57.965359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#dataset['S_2']","metadata":{"execution":{"iopub.status.busy":"2022-09-22T12:11:57.968545Z","iopub.execute_input":"2022-09-22T12:11:57.969014Z","iopub.status.idle":"2022-09-22T12:11:57.979348Z","shell.execute_reply.started":"2022-09-22T12:11:57.968978Z","shell.execute_reply":"2022-09-22T12:11:57.978083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"a=process_data(train_df)\na.shape","metadata":{"execution":{"iopub.status.busy":"2022-09-22T12:11:57.980904Z","iopub.execute_input":"2022-09-22T12:11:57.981657Z","iopub.status.idle":"2022-09-22T12:12:28.357798Z","shell.execute_reply.started":"2022-09-22T12:11:57.981616Z","shell.execute_reply":"2022-09-22T12:12:28.356585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"b=process_data(test_df)\nb.shape","metadata":{"execution":{"iopub.status.busy":"2022-09-22T12:12:28.359085Z","iopub.execute_input":"2022-09-22T12:12:28.359421Z","iopub.status.idle":"2022-09-22T12:13:01.082964Z","shell.execute_reply.started":"2022-09-22T12:12:28.359388Z","shell.execute_reply":"2022-09-22T12:13:01.081816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"a.head()","metadata":{"execution":{"iopub.status.busy":"2022-09-22T12:13:01.084789Z","iopub.execute_input":"2022-09-22T12:13:01.085248Z","iopub.status.idle":"2022-09-22T12:13:01.106266Z","shell.execute_reply.started":"2022-09-22T12:13:01.085204Z","shell.execute_reply":"2022-09-22T12:13:01.104822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"a['target']","metadata":{"execution":{"iopub.status.busy":"2022-09-22T12:13:01.107998Z","iopub.execute_input":"2022-09-22T12:13:01.108331Z","iopub.status.idle":"2022-09-22T12:13:01.118449Z","shell.execute_reply.started":"2022-09-22T12:13:01.108301Z","shell.execute_reply":"2022-09-22T12:13:01.116991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"a=a.loc[:, ~a.T.duplicated(keep='first')]\na['target']\n","metadata":{"execution":{"iopub.status.busy":"2022-09-22T12:13:01.119985Z","iopub.execute_input":"2022-09-22T12:13:01.120332Z","iopub.status.idle":"2022-09-22T12:13:28.693649Z","shell.execute_reply.started":"2022-09-22T12:13:01.120303Z","shell.execute_reply":"2022-09-22T12:13:28.692458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"a.head()","metadata":{"execution":{"iopub.status.busy":"2022-09-22T12:13:28.695218Z","iopub.execute_input":"2022-09-22T12:13:28.695581Z","iopub.status.idle":"2022-09-22T12:13:28.716605Z","shell.execute_reply.started":"2022-09-22T12:13:28.695543Z","shell.execute_reply":"2022-09-22T12:13:28.715374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"b.head()","metadata":{"execution":{"iopub.status.busy":"2022-09-22T12:13:28.719695Z","iopub.execute_input":"2022-09-22T12:13:28.721239Z","iopub.status.idle":"2022-09-22T12:13:28.744803Z","shell.execute_reply.started":"2022-09-22T12:13:28.721111Z","shell.execute_reply":"2022-09-22T12:13:28.743390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(a.columns.difference(b.columns))","metadata":{"execution":{"iopub.status.busy":"2022-09-22T12:13:28.746609Z","iopub.execute_input":"2022-09-22T12:13:28.746992Z","iopub.status.idle":"2022-09-22T12:13:28.758260Z","shell.execute_reply.started":"2022-09-22T12:13:28.746958Z","shell.execute_reply":"2022-09-22T12:13:28.757213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(b.columns.difference(a.columns))","metadata":{"execution":{"iopub.status.busy":"2022-09-22T12:13:28.760909Z","iopub.execute_input":"2022-09-22T12:13:28.762222Z","iopub.status.idle":"2022-09-22T12:13:28.769915Z","shell.execute_reply.started":"2022-09-22T12:13:28.762174Z","shell.execute_reply":"2022-09-22T12:13:28.768756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"a['D_64_-1'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-09-22T12:13:28.777697Z","iopub.execute_input":"2022-09-22T12:13:28.778122Z","iopub.status.idle":"2022-09-22T12:13:28.788441Z","shell.execute_reply.started":"2022-09-22T12:13:28.778071Z","shell.execute_reply":"2022-09-22T12:13:28.787086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"a1=a.copy()","metadata":{"execution":{"iopub.status.busy":"2022-09-22T12:13:28.791448Z","iopub.execute_input":"2022-09-22T12:13:28.792107Z","iopub.status.idle":"2022-09-22T12:13:28.880500Z","shell.execute_reply.started":"2022-09-22T12:13:28.792050Z","shell.execute_reply":"2022-09-22T12:13:28.879435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#a1.drop(['D_64_-1'])","metadata":{"execution":{"iopub.status.busy":"2022-09-22T12:13:28.881798Z","iopub.execute_input":"2022-09-22T12:13:28.882123Z","iopub.status.idle":"2022-09-22T12:13:28.887394Z","shell.execute_reply.started":"2022-09-22T12:13:28.882092Z","shell.execute_reply":"2022-09-22T12:13:28.886171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"a1['D_64_-1']\na1=a1.drop(['D_64_-1'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-09-22T12:13:28.889166Z","iopub.execute_input":"2022-09-22T12:13:28.889717Z","iopub.status.idle":"2022-09-22T12:13:28.975024Z","shell.execute_reply.started":"2022-09-22T12:13:28.889671Z","shell.execute_reply":"2022-09-22T12:13:28.973934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train=a1[\"target\"]\nX_train=a1.drop(\"target\", axis=1)\nX_test= b.copy()","metadata":{"execution":{"iopub.status.busy":"2022-09-22T12:13:28.976848Z","iopub.execute_input":"2022-09-22T12:13:28.977368Z","iopub.status.idle":"2022-09-22T12:13:29.161375Z","shell.execute_reply.started":"2022-09-22T12:13:28.977309Z","shell.execute_reply":"2022-09-22T12:13:29.160176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Model Building/Evaluation","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import ExtraTreesClassifier,RandomForestClassifier,AdaBoostClassifier\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.linear_model import LinearRegression, Ridge, Lasso, ElasticNet, SGDClassifier\nfrom sklearn.model_selection import GridSearchCV, cross_val_score,StratifiedKFold\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.svm import SVC\nfrom sklearn import linear_model\nfrom sklearn.tree import DecisionTreeClassifier\nimport xgboost as xgb\nfrom sklearn import metrics\nimport random as rd\nmodels = [#(\"LR\", LinearRegression()),\n          (\"Naive Bayes\", GaussianNB()),\n          (\"KNN\", KNeighborsClassifier()),\n          (\"DTC\", DecisionTreeClassifier()),\n          (\"SGD\", SGDClassifier()),\n          (\"Ada\", AdaBoostClassifier()),\n          (\"RFC\",RandomForestClassifier()),\n          (\"Extra:\",ExtraTreesClassifier()),\n          #(\"SVC\", SVC()),#has no feature importances\n          (\"XGBoost\", xgb.XGBClassifier())\n         ]\n","metadata":{"execution":{"iopub.status.busy":"2022-09-22T12:13:29.162736Z","iopub.execute_input":"2022-09-22T12:13:29.163083Z","iopub.status.idle":"2022-09-22T12:13:29.173959Z","shell.execute_reply.started":"2022-09-22T12:13:29.163025Z","shell.execute_reply":"2022-09-22T12:13:29.172273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"def model_evaluation(X,X2,y):\n    X= X_train\n    X2= X_test\n    y=y_train\n    models = [#(\"LR\", LinearRegression()),\n        (\"Naive Bayes\", GaussianNB()),\n          (\"KNN\", KNeighborsClassifier()),\n          (\"DTC\", DecisionTreeClassifier()),\n          (\"SGD\", SGDClassifier()),\n          (\"Ada\", AdaBoostClassifier()),\n          (\"RFC\",RandomForestClassifier()),\n          (\"Extra:\",ExtraTreesClassifier()),\n          (\"XGB:\", xgb.XGBClassifier())\n              \n              \n         ]\n    means = []\n    stds = []\n\n    for name, classifier in models:\n        scores = []\n        for _ in range(10):\n            \n            model = classifier\n            model.fit(X_train, y_train)\n            y_pred = model.predict(X_test)\n            Accuracy=model.score(X_train, y_train)\n\n            scores.append(Accuracy)\n            \n\n        means.append(np.mean(scores))\n        stds.append(np.std(scores))\n        \n    Total = list(zip(means,stds))\n    hsh = {}\n    for j in range(len(models)):\n        hsh[models[j][0]] = list(Total[j])\n\n    data4 = pd.DataFrame(hsh)\n    db=data4.transpose()\n    db.columns=['mean','std']\n    \n    return db","metadata":{"execution":{"iopub.status.busy":"2022-09-21T15:03:55.066817Z","iopub.execute_input":"2022-09-21T15:03:55.067308Z","iopub.status.idle":"2022-09-21T15:03:55.079610Z","shell.execute_reply.started":"2022-09-21T15:03:55.067255Z","shell.execute_reply":"2022-09-21T15:03:55.078065Z"}}},{"cell_type":"code","source":"#model_a=model_evaluation(X_train, X_test, y_train)\n#print(model_a)","metadata":{"execution":{"iopub.status.busy":"2022-09-22T12:13:29.175566Z","iopub.execute_input":"2022-09-22T12:13:29.176055Z","iopub.status.idle":"2022-09-22T12:13:29.187715Z","shell.execute_reply.started":"2022-09-22T12:13:29.176005Z","shell.execute_reply":"2022-09-22T12:13:29.186034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#o=====","metadata":{"execution":{"iopub.status.busy":"2022-09-22T12:13:29.190621Z","iopub.execute_input":"2022-09-22T12:13:29.191515Z","iopub.status.idle":"2022-09-22T12:13:29.199762Z","shell.execute_reply.started":"2022-09-22T12:13:29.191473Z","shell.execute_reply":"2022-09-22T12:13:29.198558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" #X_train,X_test, y_train, y_test= train_test_split(X1, y1, test_size= 0.3, random_state= 48 )","metadata":{"execution":{"iopub.status.busy":"2022-09-22T12:13:29.201102Z","iopub.execute_input":"2022-09-22T12:13:29.201451Z","iopub.status.idle":"2022-09-22T12:13:29.210560Z","shell.execute_reply.started":"2022-09-22T12:13:29.201419Z","shell.execute_reply":"2022-09-22T12:13:29.209359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_xgb=xgb.XGBClassifier()\nmodel_xgb.fit(X_train, y_train)\naccuracy_xgb=model_xgb.score(X_train, y_train)\nprint(accuracy_xgb)\n#print(\"Accuracy:\", metrics.accuracy_score(y_test, y_pred))\n","metadata":{"execution":{"iopub.status.busy":"2022-09-22T12:13:29.212317Z","iopub.execute_input":"2022-09-22T12:13:29.213314Z","iopub.status.idle":"2022-09-22T12:14:09.916812Z","shell.execute_reply.started":"2022-09-22T12:13:29.213275Z","shell.execute_reply":"2022-09-22T12:14:09.915505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_Extra=ExtraTreesClassifier()\nmodel_Extra.fit(X_train, y_train)\naccuracy_Extra=model_Extra.score(X_train,y_train)\nprint(accuracy_Extra)\n#print(\"Accuracy:\", metrics.accuracy_score(y_test, y_pred))","metadata":{"execution":{"iopub.status.busy":"2022-09-22T12:14:09.918414Z","iopub.execute_input":"2022-09-22T12:14:09.919083Z","iopub.status.idle":"2022-09-22T12:15:01.300101Z","shell.execute_reply.started":"2022-09-22T12:14:09.919046Z","shell.execute_reply":"2022-09-22T12:15:01.298874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#g===","metadata":{"execution":{"iopub.status.busy":"2022-09-22T12:15:01.303757Z","iopub.execute_input":"2022-09-22T12:15:01.304134Z","iopub.status.idle":"2022-09-22T12:15:01.309980Z","shell.execute_reply.started":"2022-09-22T12:15:01.304098Z","shell.execute_reply":"2022-09-22T12:15:01.308704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"def model_evaluation2(X1,y1):\n    X_train1, X_test1, y_train1, y_test1 = train_test_split(X1, y1, test_size=0.25, stratify=y1, random_state=42)\n    models = [#(\"LR\", LinearRegression()),\n        (\"Naive Bayes\", GaussianNB()),\n          (\"KNN\", KNeighborsClassifier()),\n          (\"DTC\", DecisionTreeClassifier()),\n          (\"SGD\", SGDClassifier()),\n          (\"Ada\", AdaBoostClassifier()),\n          (\"RFC\",RandomForestClassifier()),\n          (\"Extra:\",ExtraTreesClassifier()),\n          (\"XGB:\", xgb.XGBClassifier())\n              \n              \n         ]\n    means = []\n    stds = []\n\n    for name, classifier in models:\n        scores = []\n        for _ in range(10):\n            \n            model = classifier\n            model.fit(X_train, y_train)\n            y_pred = model.predict(X_test)\n            Accuracy=metrics.accuracy_score(y_test,y_pred)\n\n            scores.append(Accuracy)\n            \n\n        means.append(np.mean(scores))\n        stds.append(np.std(scores))\n        \n    Total = list(zip(means,stds))\n    hsh = {}\n    for j in range(len(models)):\n        hsh[models[j][0]] = list(Total[j])\n\n    data4 = pd.DataFrame(hsh)\n    db2=data4.transpose()\n    db2.columns=['mean','std']\n    \n    return db2","metadata":{"execution":{"iopub.status.busy":"2022-09-21T14:55:18.583467Z","iopub.status.idle":"2022-09-21T14:55:18.583865Z","shell.execute_reply.started":"2022-09-21T14:55:18.583661Z","shell.execute_reply":"2022-09-21T14:55:18.583680Z"}}},{"cell_type":"code","source":"#model_b=model_evaluation2(X1, y1)\n#print(model_b)","metadata":{"execution":{"iopub.status.busy":"2022-09-22T12:15:01.312488Z","iopub.execute_input":"2022-09-22T12:15:01.312941Z","iopub.status.idle":"2022-09-22T12:15:01.323146Z","shell.execute_reply.started":"2022-09-22T12:15:01.312895Z","shell.execute_reply":"2022-09-22T12:15:01.321807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#splitting training data into training and testing (validation) set\n#X_train1, X_test1, y_train1, y_test1 = train_test_split(X1, y1, test_size=0.25, stratify=y1, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-09-22T12:15:01.324939Z","iopub.execute_input":"2022-09-22T12:15:01.325769Z","iopub.status.idle":"2022-09-22T12:15:01.336320Z","shell.execute_reply.started":"2022-09-22T12:15:01.325717Z","shell.execute_reply":"2022-09-22T12:15:01.334865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_Extra1=ExtraTreesClassifier()\nmodel_Extra1.fit(X_train, y_train)\ny_pred1=model_Extra1.predict(X_test)\n#print(\"Accuracy:\", metrics.accuracy_score(y_test1, y_pred1))\n","metadata":{"execution":{"iopub.status.busy":"2022-09-22T12:15:01.337864Z","iopub.execute_input":"2022-09-22T12:15:01.338967Z","iopub.status.idle":"2022-09-22T12:15:52.917130Z","shell.execute_reply.started":"2022-09-22T12:15:01.338921Z","shell.execute_reply":"2022-09-22T12:15:52.915964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#model_xgb1=xgb.XGBClassifier()\n#model_xgb1.fit(X_train1, y_train1)\n#y_pred2=model_xgb1.predict(X_test1)\n#print(\"Accuracy:\", metrics.accuracy_score(y_test1, y_pred2))\n","metadata":{"execution":{"iopub.status.busy":"2022-09-22T12:15:52.919093Z","iopub.execute_input":"2022-09-22T12:15:52.919484Z","iopub.status.idle":"2022-09-22T12:15:52.925098Z","shell.execute_reply.started":"2022-09-22T12:15:52.919450Z","shell.execute_reply":"2022-09-22T12:15:52.923678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#test= pd.read_csv('../input/amex-default-prediction/test_data.csv',  nrows=100000)\n#testCustomer  = pd.read_csv('/kaggle/input/amex-default-prediction/sample_submission.csv', usecols=['customer_ID'], low_memory=True)","metadata":{"execution":{"iopub.status.busy":"2022-09-22T12:15:52.926664Z","iopub.execute_input":"2022-09-22T12:15:52.927110Z","iopub.status.idle":"2022-09-22T12:15:52.937033Z","shell.execute_reply.started":"2022-09-22T12:15:52.927075Z","shell.execute_reply":"2022-09-22T12:15:52.935703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#test.shape","metadata":{"execution":{"iopub.status.busy":"2022-09-22T12:15:52.938920Z","iopub.execute_input":"2022-09-22T12:15:52.939395Z","iopub.status.idle":"2022-09-22T12:15:52.949731Z","shell.execute_reply.started":"2022-09-22T12:15:52.939344Z","shell.execute_reply":"2022-09-22T12:15:52.948454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#testData = process_data(test)","metadata":{"execution":{"iopub.status.busy":"2022-09-22T12:15:52.953969Z","iopub.execute_input":"2022-09-22T12:15:52.954514Z","iopub.status.idle":"2022-09-22T12:15:52.962818Z","shell.execute_reply.started":"2022-09-22T12:15:52.954476Z","shell.execute_reply":"2022-09-22T12:15:52.961445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#test.head()","metadata":{"execution":{"iopub.status.busy":"2022-09-22T12:15:52.964272Z","iopub.execute_input":"2022-09-22T12:15:52.964726Z","iopub.status.idle":"2022-09-22T12:15:52.973829Z","shell.execute_reply.started":"2022-09-22T12:15:52.964690Z","shell.execute_reply":"2022-09-22T12:15:52.972840Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#a['target']","metadata":{"execution":{"iopub.status.busy":"2022-09-22T12:15:52.975382Z","iopub.execute_input":"2022-09-22T12:15:52.976221Z","iopub.status.idle":"2022-09-22T12:15:52.987219Z","shell.execute_reply.started":"2022-09-22T12:15:52.976182Z","shell.execute_reply":"2022-09-22T12:15:52.985816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_final=model_xgb.predict(X_test)\ny_pred_final=pd.DataFrame(y_pred_final, columns=['prediction'])\n#Retrieve the probability of default\ny_pred_final.head()","metadata":{"execution":{"iopub.status.busy":"2022-09-22T12:15:52.988816Z","iopub.execute_input":"2022-09-22T12:15:52.989225Z","iopub.status.idle":"2022-09-22T12:15:53.233879Z","shell.execute_reply.started":"2022-09-22T12:15:52.989189Z","shell.execute_reply":"2022-09-22T12:15:53.232978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_final.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-09-22T12:15:53.238299Z","iopub.execute_input":"2022-09-22T12:15:53.240546Z","iopub.status.idle":"2022-09-22T12:15:53.255118Z","shell.execute_reply.started":"2022-09-22T12:15:53.240487Z","shell.execute_reply":"2022-09-22T12:15:53.253949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission=pd.read_csv('../input/amex-default-prediction/sample_submission.csv')\nsample_submission.head()","metadata":{"execution":{"iopub.status.busy":"2022-09-22T12:21:57.826807Z","iopub.execute_input":"2022-09-22T12:21:57.827367Z","iopub.status.idle":"2022-09-22T12:22:00.653997Z","shell.execute_reply.started":"2022-09-22T12:21:57.827325Z","shell.execute_reply":"2022-09-22T12:22:00.652686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample=sample_submission[:100000]\nsample.info()","metadata":{"execution":{"iopub.status.busy":"2022-09-22T12:46:55.562723Z","iopub.execute_input":"2022-09-22T12:46:55.563557Z","iopub.status.idle":"2022-09-22T12:46:55.583923Z","shell.execute_reply.started":"2022-09-22T12:46:55.563504Z","shell.execute_reply":"2022-09-22T12:46:55.582695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred= model_Extra.predict(X_test)\ny_pred= pd.DataFrame(y_pred, columns=['prediction'])\ny_pred.info()","metadata":{"execution":{"iopub.status.busy":"2022-09-22T12:55:16.858275Z","iopub.execute_input":"2022-09-22T12:55:16.858789Z","iopub.status.idle":"2022-09-22T12:55:20.400720Z","shell.execute_reply.started":"2022-09-22T12:55:16.858750Z","shell.execute_reply":"2022-09-22T12:55:20.399507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample['prediction']=y_pred\nsample.head()","metadata":{"execution":{"iopub.status.busy":"2022-09-22T12:56:24.171393Z","iopub.execute_input":"2022-09-22T12:56:24.171885Z","iopub.status.idle":"2022-09-22T12:56:24.183912Z","shell.execute_reply.started":"2022-09-22T12:56:24.171848Z","shell.execute_reply":"2022-09-22T12:56:24.182919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-09-22T13:03:45.507100Z","iopub.execute_input":"2022-09-22T13:03:45.507648Z","iopub.status.idle":"2022-09-22T13:03:45.792353Z","shell.execute_reply.started":"2022-09-22T13:03:45.507608Z","shell.execute_reply":"2022-09-22T13:03:45.791038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}