{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport warnings\nimport seaborn as sns\nwarnings.filterwarnings('ignore')\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-09T17:46:57.523669Z","iopub.execute_input":"2022-08-09T17:46:57.524066Z","iopub.status.idle":"2022-08-09T17:46:58.076867Z","shell.execute_reply.started":"2022-08-09T17:46:57.523990Z","shell.execute_reply":"2022-08-09T17:46:58.075388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df_orig = pd.read_csv('../input/tabular-playground-series-aug-2022/train.csv')\ntrain_df = train_df_orig.copy()\n\ntest_df_orig = pd.read_csv('../input/tabular-playground-series-aug-2022/test.csv')\ntest_df = test_df_orig.copy()","metadata":{"execution":{"iopub.status.busy":"2022-08-09T17:46:58.078239Z","iopub.execute_input":"2022-08-09T17:46:58.078506Z","iopub.status.idle":"2022-08-09T17:46:58.276434Z","shell.execute_reply.started":"2022-08-09T17:46:58.078482Z","shell.execute_reply":"2022-08-09T17:46:58.275101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head(2)","metadata":{"execution":{"iopub.status.busy":"2022-08-09T17:46:58.278185Z","iopub.execute_input":"2022-08-09T17:46:58.278524Z","iopub.status.idle":"2022-08-09T17:46:58.314267Z","shell.execute_reply.started":"2022-08-09T17:46:58.278491Z","shell.execute_reply":"2022-08-09T17:46:58.313107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Transform String value to integer values","metadata":{}},{"cell_type":"code","source":"string_column = ['product_code', 'attribute_0', 'attribute_1']\nfor i in string_column:\n    print(\"Unique value of \"+str(i) +\" is \" + str(train_df[i].unique()))","metadata":{"execution":{"iopub.status.busy":"2022-08-09T17:46:58.366545Z","iopub.execute_input":"2022-08-09T17:46:58.366853Z","iopub.status.idle":"2022-08-09T17:46:58.383764Z","shell.execute_reply.started":"2022-08-09T17:46:58.366830Z","shell.execute_reply":"2022-08-09T17:46:58.383105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head(1)","metadata":{"execution":{"iopub.status.busy":"2022-08-09T17:46:58.564346Z","iopub.execute_input":"2022-08-09T17:46:58.565229Z","iopub.status.idle":"2022-08-09T17:46:58.587694Z","shell.execute_reply.started":"2022-08-09T17:46:58.565201Z","shell.execute_reply":"2022-08-09T17:46:58.585839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.preprocessing import OrdinalEncoder\n\n# oenc = OrdinalEncoder()\n# oenc.fit(train_df[['product_code']])\n# train_df[['product_code']] = oenc.transform(train_df[['product_code']])","metadata":{"execution":{"iopub.status.busy":"2022-08-09T17:46:58.717913Z","iopub.execute_input":"2022-08-09T17:46:58.718312Z","iopub.status.idle":"2022-08-09T17:46:58.722612Z","shell.execute_reply.started":"2022-08-09T17:46:58.718282Z","shell.execute_reply":"2022-08-09T17:46:58.721765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import OrdinalEncoder\n\noenc = OrdinalEncoder()\n\nfor i in string_column:\n    oenc.fit(train_df[[i]])\n    train_df[[i]] = oenc.transform(train_df[[i]])\n    oenc.fit(test_df[[i]])\n    test_df[[i]] = oenc.transform(test_df[[i]])\n\n# le = preprocessing.LabelEncoder()\n# for i in string_column:\n#     le.fit(train_df[i])\n#     train_df[i] = le.transform(train_df[i])\n#     le.fit(test_df[i])\n#     test_df[i] = le.transform(test_df[i])","metadata":{"execution":{"iopub.status.busy":"2022-08-09T17:46:58.878979Z","iopub.execute_input":"2022-08-09T17:46:58.879287Z","iopub.status.idle":"2022-08-09T17:46:59.006092Z","shell.execute_reply.started":"2022-08-09T17:46:58.879263Z","shell.execute_reply":"2022-08-09T17:46:59.004771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head(1)","metadata":{"execution":{"iopub.status.busy":"2022-08-09T17:46:59.101565Z","iopub.execute_input":"2022-08-09T17:46:59.101932Z","iopub.status.idle":"2022-08-09T17:46:59.125770Z","shell.execute_reply.started":"2022-08-09T17:46:59.101907Z","shell.execute_reply":"2022-08-09T17:46:59.124290Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Check Null values in the dataframe","metadata":{}},{"cell_type":"code","source":"print(\"Train Dataset\")\nprint(train_df.columns[train_df.isna().any()].tolist())\nprint(\"Test Dataset\")\nprint(test_df.columns[test_df.isna().any()].tolist())","metadata":{"execution":{"iopub.status.busy":"2022-08-09T17:46:59.269616Z","iopub.execute_input":"2022-08-09T17:46:59.270145Z","iopub.status.idle":"2022-08-09T17:46:59.280553Z","shell.execute_reply.started":"2022-08-09T17:46:59.270111Z","shell.execute_reply":"2022-08-09T17:46:59.279179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i, j in zip(train_df.columns, test_df.columns):\n    if train_df[i].isnull().sum() !=0:\n        train_df[i].fillna(value = train_df[i].mean(), inplace=True)\n    if test_df[j].isnull().sum() !=0:\n        test_df[i].fillna(value = test_df[i].mean(), inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-09T17:46:59.516107Z","iopub.execute_input":"2022-08-09T17:46:59.516641Z","iopub.status.idle":"2022-08-09T17:46:59.545374Z","shell.execute_reply.started":"2022-08-09T17:46:59.516616Z","shell.execute_reply":"2022-08-09T17:46:59.544391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Train Dataset\")\nprint(train_df.columns[train_df.isna().any()].tolist())\nprint(\"Test Dataset\")\nprint(test_df.columns[test_df.isna().any()].tolist())","metadata":{"execution":{"iopub.status.busy":"2022-08-09T17:46:59.810979Z","iopub.execute_input":"2022-08-09T17:46:59.812017Z","iopub.status.idle":"2022-08-09T17:46:59.821411Z","shell.execute_reply.started":"2022-08-09T17:46:59.811983Z","shell.execute_reply":"2022-08-09T17:46:59.820207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Distribution of Dataset","metadata":{}},{"cell_type":"code","source":"fig, axes = plt.subplots(nrows=5, ncols=5, figsize=(20, 9), sharex = False, sharey = False)\naxes = axes.ravel()  \ncols = train_df.columns[:-1]\n\nfor col, ax in zip(cols, axes):\n    data = train_df\n    sns.kdeplot(data=data, x=col, shade=True, ax=ax, hue='failure')\n    \nfig.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-09T17:47:00.011631Z","iopub.execute_input":"2022-08-09T17:47:00.011969Z","iopub.status.idle":"2022-08-09T17:47:06.174896Z","shell.execute_reply.started":"2022-08-09T17:47:00.011931Z","shell.execute_reply":"2022-08-09T17:47:06.174122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (10,8))\ncorr = train_df.corr()\nsns.heatmap(corr,xticklabels=corr.columns,yticklabels=corr.columns,linewidths=.1,cmap=\"Blues\", square=True, robust=True)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-09T17:47:06.176239Z","iopub.execute_input":"2022-08-09T17:47:06.176586Z","iopub.status.idle":"2022-08-09T17:47:06.739562Z","shell.execute_reply.started":"2022-08-09T17:47:06.176563Z","shell.execute_reply":"2022-08-09T17:47:06.738031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.groupby('product_code').mean()","metadata":{"execution":{"iopub.status.busy":"2022-08-09T17:47:06.741233Z","iopub.execute_input":"2022-08-09T17:47:06.741548Z","iopub.status.idle":"2022-08-09T17:47:06.779004Z","shell.execute_reply.started":"2022-08-09T17:47:06.741523Z","shell.execute_reply":"2022-08-09T17:47:06.778242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Modeling","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX = train_df[train_df.columns[:-1]]\ny = train_df['failure']\n\nX_train, X_test, y_train, y_test = train_test_split(X,y, test_size=0.3, random_state=100)","metadata":{"execution":{"iopub.status.busy":"2022-08-09T17:47:06.781733Z","iopub.execute_input":"2022-08-09T17:47:06.782116Z","iopub.status.idle":"2022-08-09T17:47:06.877826Z","shell.execute_reply.started":"2022-08-09T17:47:06.782089Z","shell.execute_reply":"2022-08-09T17:47:06.876548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Logistic Regression","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\n\nmodel_log = LogisticRegression()\nmodel_log.fit(X_train,y_train)\npred_log=model_log.predict(X_test)\nlog_score =model_log.score(X_train,y_train)\nlog_pred_score =round(log_score*100,2)\nprint(log_pred_score)","metadata":{"execution":{"iopub.status.busy":"2022-08-09T17:47:06.879006Z","iopub.execute_input":"2022-08-09T17:47:06.879454Z","iopub.status.idle":"2022-08-09T17:47:07.057715Z","shell.execute_reply.started":"2022-08-09T17:47:06.879427Z","shell.execute_reply":"2022-08-09T17:47:07.056995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### XGB","metadata":{}},{"cell_type":"code","source":"import xgboost as xgb\n\nxgb_model = xgb.XGBClassifier()\nxgb_model.fit(X_train, np.ravel(y_train))\n\npredict_xgb = xgb_model.predict_proba(X_test)\npredict_xgb_prob = pd.DataFrame(predict_xgb[:,1],columns = ['Default Probability'])\nxgb_probability = pd.concat([predict_xgb_prob, y_test.reset_index(drop=True)],axis=1)\nxgb_probability.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-09T17:47:07.058937Z","iopub.execute_input":"2022-08-09T17:47:07.059396Z","iopub.status.idle":"2022-08-09T17:47:09.720914Z","shell.execute_reply.started":"2022-08-09T17:47:07.059367Z","shell.execute_reply":"2022-08-09T17:47:09.720196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xgb.plot_importance(xgb_model,importance_type='weight')","metadata":{"execution":{"iopub.status.busy":"2022-08-09T17:47:09.723529Z","iopub.execute_input":"2022-08-09T17:47:09.723957Z","iopub.status.idle":"2022-08-09T17:47:10.041163Z","shell.execute_reply.started":"2022-08-09T17:47:09.723908Z","shell.execute_reply":"2022-08-09T17:47:10.040211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_xgb=xgb_model.predict(X_test)\nxgb_score =xgb_model.score(X_train,y_train)\nxgb_pred_score =round(xgb_score*100,2)\nprint(xgb_pred_score)","metadata":{"execution":{"iopub.status.busy":"2022-08-09T17:47:10.042200Z","iopub.execute_input":"2022-08-09T17:47:10.042542Z","iopub.status.idle":"2022-08-09T17:47:10.098096Z","shell.execute_reply.started":"2022-08-09T17:47:10.042510Z","shell.execute_reply":"2022-08-09T17:47:10.097059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### KNN","metadata":{}},{"cell_type":"code","source":"from sklearn.neighbors import KNeighborsClassifier\n\nmodel_knn=KNeighborsClassifier()\nmodel_knn.fit(X_train,y_train)\npred_knn=model_knn.predict(X_test)\nknn_score =model_knn.score(X_train,y_train)\nknn_pred_score =round(knn_score*100,2)\nprint(knn_pred_score)","metadata":{"execution":{"iopub.status.busy":"2022-08-09T17:47:10.099115Z","iopub.execute_input":"2022-08-09T17:47:10.099535Z","iopub.status.idle":"2022-08-09T17:47:19.070498Z","shell.execute_reply.started":"2022-08-09T17:47:10.099509Z","shell.execute_reply":"2022-08-09T17:47:19.069332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Gradient Boosting","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import GradientBoostingClassifier\n\nmodel_gbdt = GradientBoostingClassifier(n_estimators=100, learning_rate=1.0, max_depth=1, random_state=0).fit(X_train, y_train)\npred_gbdt = model_gbdt.predict(X_test)\ngbdt_score = model_gbdt.score(X_test, y_test)\ngbdt_pred_score  =round(gbdt_score*100,2)\ngbdt_pred_score","metadata":{"execution":{"iopub.status.busy":"2022-08-09T17:47:19.073382Z","iopub.execute_input":"2022-08-09T17:47:19.073688Z","iopub.status.idle":"2022-08-09T17:47:23.122453Z","shell.execute_reply.started":"2022-08-09T17:47:19.073663Z","shell.execute_reply":"2022-08-09T17:47:23.121231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### LinearDiscriminant (LDA)","metadata":{}},{"cell_type":"code","source":"from sklearn.discriminant_analysis import LinearDiscriminantAnalysis\n\nmodel_lda = LinearDiscriminantAnalysis()\nmodel_lda.fit(X_train,y_train)\npred_lda=model_lda.predict(X_test)\nlda_score =model_lda.score(X_train,y_train)\nlda_pred_score =round(lda_score*100,2)\nprint(lda_pred_score)","metadata":{"execution":{"iopub.status.busy":"2022-08-09T17:47:23.124035Z","iopub.execute_input":"2022-08-09T17:47:23.124607Z","iopub.status.idle":"2022-08-09T17:47:23.222589Z","shell.execute_reply.started":"2022-08-09T17:47:23.124582Z","shell.execute_reply":"2022-08-09T17:47:23.220795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### GaussianNB","metadata":{}},{"cell_type":"code","source":"from sklearn.naive_bayes import GaussianNB\n\nmodel_gnb =GaussianNB()\nmodel_gnb.fit(X_train,y_train)\npred_gnb=model_gnb.predict(X_test)\ngnb_score =model_gnb.score(X_train,y_train)\ngnb_pred_score =round(gnb_score*100,2)\nprint(gnb_pred_score)","metadata":{"execution":{"iopub.status.busy":"2022-08-09T17:47:23.223748Z","iopub.execute_input":"2022-08-09T17:47:23.225324Z","iopub.status.idle":"2022-08-09T17:47:23.294856Z","shell.execute_reply.started":"2022-08-09T17:47:23.225295Z","shell.execute_reply":"2022-08-09T17:47:23.293905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### RandomForest","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\n\nmodel_rfc =RandomForestClassifier(max_depth=2, random_state=0)\nmodel_rfc.fit(X_train,y_train)\npred_rfc=model_rfc.predict(X_test)\nrfc_score =model_rfc.score(X_train,y_train)\nrfc_pred_score =round(rfc_score*100,2)\nprint(rfc_pred_score)","metadata":{"execution":{"iopub.status.busy":"2022-08-09T17:47:23.296791Z","iopub.execute_input":"2022-08-09T17:47:23.297649Z","iopub.status.idle":"2022-08-09T17:47:24.882207Z","shell.execute_reply.started":"2022-08-09T17:47:23.297619Z","shell.execute_reply":"2022-08-09T17:47:24.881672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### VotingClassifier","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import VotingClassifier\n#https://scikit-learn.org/stable/modules/generated/sklearn.ensemble.VotingClassifier.html\n\nlr1 = LogisticRegression(max_iter = 200, C=0.05, penalty='l1', solver='liblinear')\n    #multi_class='multinomial', random_state=1)\nrfc2 = RandomForestClassifier(n_estimators=50, random_state=1)\ngnb3 = GaussianNB()\nmodel_vtc = VotingClassifier(estimators=[('lr', lr1), ('rf', rfc2), ('gnb', gnb3)], voting='soft')\n\nmodel_vtc.fit(X_train,y_train)\npred_vtc=model_vtc.predict(X_test)\nvtc_score =model_vtc.score(X_train,y_train)\nvtc_pred_score =round(vtc_score*100,2)\nprint(vtc_pred_score)","metadata":{"execution":{"iopub.status.busy":"2022-08-09T17:47:24.883193Z","iopub.execute_input":"2022-08-09T17:47:24.883523Z","iopub.status.idle":"2022-08-09T17:47:31.700842Z","shell.execute_reply.started":"2022-08-09T17:47:24.883502Z","shell.execute_reply":"2022-08-09T17:47:31.700009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Result","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import classification_report\n\nprint('Logistic Regression')\nprint(classification_report(y_test,pred_log))\n\nprint('XGBoost')\nprint(classification_report(y_test,pred_xgb))\n\nprint('KNN')\nprint(classification_report(y_test,pred_knn))\n\nprint('GradientBoosting')\nprint(classification_report(y_test,pred_gbdt))\n\nprint('LinearDiscriminant')\nprint(classification_report(y_test,pred_lda))\n\nprint('GaussianNB')\nprint(classification_report(y_test,pred_gnb))\n      \nprint('Random Forest')\nprint(classification_report(y_test,pred_rfc))\n      \nprint('Voting Classifier')\nprint(classification_report(y_test,pred_vtc))","metadata":{"execution":{"iopub.status.busy":"2022-08-09T17:47:58.339786Z","iopub.execute_input":"2022-08-09T17:47:58.340214Z","iopub.status.idle":"2022-08-09T17:47:58.430488Z","shell.execute_reply.started":"2022-08-09T17:47:58.340138Z","shell.execute_reply":"2022-08-09T17:47:58.429302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## ROC Curve","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import roc_curve, roc_auc_score\n\nplt.figure(figsize=(15,7))\nplt.subplot(1,2,2)\nplt.plot([0, 1], [0, 1], 'k--')\n\nmodel_log_proba=model_log.predict_proba(X_test)[:,1]\nfpr, tpr, thresholds  = roc_curve(y_test,model_log_proba)\nplt.plot(fpr, tpr, label='Logistic Regression')\n\nxgb_model_proba=xgb_model.predict_proba(X_test)[:,1]\nfpr, tpr, thresholds  = roc_curve(y_test,xgb_model_proba)\nplt.plot(fpr, tpr, label='XGBoosts')\n\nmodel_knn_proba=model_knn.predict_proba(X_test)[:,1]\nfpr, tpr, thresholds  = roc_curve(y_test,model_knn_proba)\nplt.plot(fpr, tpr, label='KNN')\n\nmodel_gbdt_proba=model_gbdt.predict_proba(X_test)[:,1]\nfpr, tpr, thresholds  = roc_curve(y_test,model_gbdt_proba)\nplt.plot(fpr, tpr, label='GradientBoosting')\n\nmodel_lda_proba=model_lda.predict_proba(X_test)[:,1]\nfpr, tpr, thresholds  = roc_curve(y_test,model_lda_proba)\nplt.plot(fpr, tpr, label='LinearDiscriminant')\n\nmodel_gnb_proba=model_gnb.predict_proba(X_test)[:,1]\nfpr, tpr, thresholds  = roc_curve(y_test,model_gnb_proba)\nplt.plot(fpr, tpr, label='GaussianNB')\n\nmodel_rfc_proba=model_rfc.predict_proba(X_test)[:,1]\nfpr, tpr, thresholds  = roc_curve(y_test,model_rfc_proba)\nplt.plot(fpr, tpr, label='RandomForest')\n\nmodel_vtc_proba=model_vtc.predict_proba(X_test)[:,1]\nfpr, tpr, thresholds  = roc_curve(y_test,model_vtc_proba)\nplt.plot(fpr, tpr, label='VotingClassifier')\n\n\nplt.xlabel('False Positive Rate',fontsize=14)\nplt.ylabel('True Positive Rate',fontsize=14)\nplt.title('ROC Curve',fontsize=15)\nplt.legend()","metadata":{"execution":{"iopub.status.busy":"2022-08-09T17:48:06.334621Z","iopub.execute_input":"2022-08-09T17:48:06.334994Z","iopub.status.idle":"2022-08-09T17:48:09.329355Z","shell.execute_reply.started":"2022-08-09T17:48:06.334968Z","shell.execute_reply":"2022-08-09T17:48:09.328242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Based on the ROC Curve, it shows that the Logistic Regression and LDA performs better than other model. ","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score\n\nprint(accuracy_score(y_test, pred_log))\nprint(accuracy_score(y_test, pred_lda))\nprint(accuracy_score(y_test, pred_gnb))\nprint(accuracy_score(y_test, pred_gbdt))\nprint(accuracy_score(y_test, pred_rfc))\nprint(accuracy_score(y_test, pred_vtc))","metadata":{"execution":{"iopub.status.busy":"2022-08-09T17:49:34.938405Z","iopub.execute_input":"2022-08-09T17:49:34.938777Z","iopub.status.idle":"2022-08-09T17:49:34.951884Z","shell.execute_reply.started":"2022-08-09T17:49:34.938751Z","shell.execute_reply":"2022-08-09T17:49:34.950623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import roc_curve, roc_auc_score\n\nplt.figure(figsize=(15,7))\nplt.subplot(1,2,2)\nplt.plot([0, 1], [0, 1], 'k--')\n\nmodel_log_proba=model_log.predict_proba(X_test)[:,1]\nfpr, tpr, thresholds  = roc_curve(y_test,model_log_proba)\nplt.plot(fpr, tpr, label='Logistic Regression')\n\nmodel_lda_proba=model_lda.predict_proba(X_test)[:,1]\nfpr, tpr, thresholds  = roc_curve(y_test,model_lda_proba)\nplt.plot(fpr, tpr, label='LinearDiscriminant')\n\nmodel_rfc_proba=model_rfc.predict_proba(X_test)[:,1]\nfpr, tpr, thresholds  = roc_curve(y_test,model_rfc_proba)\nplt.plot(fpr, tpr, label='RandomForest')\n\nplt.xlabel('False Positive Rate',fontsize=14)\nplt.ylabel('True Positive Rate',fontsize=14)\nplt.title('ROC Curve',fontsize=15)\nplt.legend()","metadata":{"execution":{"iopub.status.busy":"2022-08-09T17:49:44.084864Z","iopub.execute_input":"2022-08-09T17:49:44.085229Z","iopub.status.idle":"2022-08-09T17:49:44.343416Z","shell.execute_reply.started":"2022-08-09T17:49:44.085204Z","shell.execute_reply":"2022-08-09T17:49:44.342413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It seems Logistic Regression performs better. ","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import cluster\n\nlog_table = cluster.contingency_matrix(y_test, pred_log)\nprint(log_table)\n\nsns.heatmap(log_table, annot=True, fmt='.2f',cmap=\"BuPu\", vmin=0.0, vmax=100.0)","metadata":{"execution":{"iopub.status.busy":"2022-08-09T17:49:49.807370Z","iopub.execute_input":"2022-08-09T17:49:49.807681Z","iopub.status.idle":"2022-08-09T17:49:49.967750Z","shell.execute_reply.started":"2022-08-09T17:49:49.807647Z","shell.execute_reply":"2022-08-09T17:49:49.966919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"log_prediction = model_log.predict_proba(test_df)[:,1]\n\nlabels = test_df['id']\nlog_submission = pd.DataFrame(np.array([labels, log_prediction]).T, columns = ['id', 'failure'])\nlog_submission['id'] = log_submission['id'].astype(int)\nlog_submission.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-09T17:49:50.894919Z","iopub.execute_input":"2022-08-09T17:49:50.896427Z","iopub.status.idle":"2022-08-09T17:49:50.912566Z","shell.execute_reply.started":"2022-08-09T17:49:50.896378Z","shell.execute_reply":"2022-08-09T17:49:50.911699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"log_submission.to_csv('log_submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-09T17:47:31.823543Z","iopub.status.idle":"2022-08-09T17:47:31.824023Z","shell.execute_reply.started":"2022-08-09T17:47:31.823848Z","shell.execute_reply":"2022-08-09T17:47:31.823865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# rfc_prediction = model_rfc.predict_proba(test_df)[:,1]\n\n# labels = test_df['id']\n# rfc_submission = pd.DataFrame(np.array([labels, rfc_prediction]).T, columns = ['id', 'failure'])\n# rfc_submission['id'] = rfc_submission['id'].astype(int)\n# rfc_submission.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-09T17:49:54.704175Z","iopub.execute_input":"2022-08-09T17:49:54.704569Z","iopub.status.idle":"2022-08-09T17:49:54.833741Z","shell.execute_reply.started":"2022-08-09T17:49:54.704539Z","shell.execute_reply":"2022-08-09T17:49:54.832695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# rfc_submission.to_csv('rfc_submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-09T17:49:56.975665Z","iopub.execute_input":"2022-08-09T17:49:56.976542Z","iopub.status.idle":"2022-08-09T17:49:57.027323Z","shell.execute_reply.started":"2022-08-09T17:49:56.976514Z","shell.execute_reply":"2022-08-09T17:49:57.026297Z"},"trusted":true},"execution_count":null,"outputs":[]}]}