{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.compose import make_column_transformer,make_column_selector\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.preprocessing import StandardScaler,OneHotEncoder\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.feature_selection import mutual_info_classif\nfrom sklearn.feature_selection import GenericUnivariateSelect\nfrom sklearn.ensemble import RandomForestClassifier\nfrom xgboost import XGBClassifier\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.model_selection import cross_val_score\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-08T10:53:17.731525Z","iopub.execute_input":"2022-08-08T10:53:17.731915Z","iopub.status.idle":"2022-08-08T10:53:17.739898Z","shell.execute_reply.started":"2022-08-08T10:53:17.731884Z","shell.execute_reply":"2022-08-08T10:53:17.738979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = pd.read_csv('../input/tabular-playground-series-aug-2022/train.csv',index_col='id')\ntest_data = pd.read_csv('../input/tabular-playground-series-aug-2022/test.csv',index_col ='id')\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T10:53:20.716688Z","iopub.execute_input":"2022-08-08T10:53:20.717127Z","iopub.status.idle":"2022-08-08T10:53:20.920265Z","shell.execute_reply.started":"2022-08-08T10:53:20.717090Z","shell.execute_reply":"2022-08-08T10:53:20.919142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T09:42:45.008487Z","iopub.execute_input":"2022-08-08T09:42:45.008890Z","iopub.status.idle":"2022-08-08T09:42:45.045358Z","shell.execute_reply.started":"2022-08-08T09:42:45.008856Z","shell.execute_reply":"2022-08-08T09:42:45.044152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data['failure'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T09:42:45.047168Z","iopub.execute_input":"2022-08-08T09:42:45.048687Z","iopub.status.idle":"2022-08-08T09:42:45.061500Z","shell.execute_reply.started":"2022-08-08T09:42:45.048635Z","shell.execute_reply":"2022-08-08T09:42:45.059914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def count_missing_values(df):\n    total_col = df.shape[0]\n    for col in df.columns:\n        missing_col = df[col].isnull().sum()\n        perc_missing = (missing_col/total_col)*100\n        if perc_missing >0:\n            print(\"{}:{:.2f}%\".format(col,perc_missing))\ncount_missing_values(train_data)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T09:42:45.064428Z","iopub.execute_input":"2022-08-08T09:42:45.064794Z","iopub.status.idle":"2022-08-08T09:42:45.092069Z","shell.execute_reply.started":"2022-08-08T09:42:45.064764Z","shell.execute_reply":"2022-08-08T09:42:45.090856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#i attempt with deep learning\n# column_num = [col for col in data.columns if data.dtypes !='object']\ntrain_data.dtypes\n","metadata":{"execution":{"iopub.status.busy":"2022-08-08T09:42:45.093920Z","iopub.execute_input":"2022-08-08T09:42:45.094383Z","iopub.status.idle":"2022-08-08T09:42:45.105492Z","shell.execute_reply.started":"2022-08-08T09:42:45.094345Z","shell.execute_reply":"2022-08-08T09:42:45.104054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = train_data.copy()\ny = df[['failure']]\ndf.drop('failure',axis=1,inplace=True)\ncol_num = [col for col in df if df[col].dtypes =='float64']\n# list(col_num)\n\ncol_cat = [col for col in df if col not in col_num]\n# col_cat\ntotal_col_num = col_num +col_cat\ncol_cat_obj = ['product_code','attribute_0','attribute_1']\nfinal_col_num = [col for col in total_col_num if col not in col_cat_obj]\nfinal_col_num","metadata":{"execution":{"iopub.status.busy":"2022-08-08T10:53:27.717331Z","iopub.execute_input":"2022-08-08T10:53:27.718105Z","iopub.status.idle":"2022-08-08T10:53:27.737090Z","shell.execute_reply.started":"2022-08-08T10:53:27.718068Z","shell.execute_reply":"2022-08-08T10:53:27.735729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df.head()\ny","metadata":{"execution":{"iopub.status.busy":"2022-08-08T10:53:31.573186Z","iopub.execute_input":"2022-08-08T10:53:31.573894Z","iopub.status.idle":"2022-08-08T10:53:31.586222Z","shell.execute_reply.started":"2022-08-08T10:53:31.573845Z","shell.execute_reply":"2022-08-08T10:53:31.584935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transformer_num = make_pipeline(SimpleImputer(strategy='constant'),StandardScaler(),)\ntransformer_cat = make_pipeline(SimpleImputer(strategy='constant',fill_value='NA'),\n                                OneHotEncoder(handle_unknown='ignore'),)\n\npreprocessor = make_column_transformer((transformer_num,final_col_num),\n                                      (transformer_cat,col_cat_obj),)\n\n#stratify - make sure classes are evenlly represented across splits\nX_train, X_valid, y_train, y_valid = \\\n    train_test_split(df, y, train_size=0.8)\n\nX_train = preprocessor.fit_transform(X_train)\nX_valid = preprocessor.transform(X_valid)\n\n\ninput_shape = [X_train.shape[1]]\nprint(input_shape)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-08T11:29:12.784483Z","iopub.execute_input":"2022-08-08T11:29:12.784994Z","iopub.status.idle":"2022-08-08T11:29:12.881444Z","shell.execute_reply.started":"2022-08-08T11:29:12.784939Z","shell.execute_reply":"2022-08-08T11:29:12.879559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# X_train[:2]","metadata":{"execution":{"iopub.status.busy":"2022-08-08T09:42:45.407145Z","iopub.execute_input":"2022-08-08T09:42:45.407737Z","iopub.status.idle":"2022-08-08T09:42:45.415056Z","shell.execute_reply.started":"2022-08-08T09:42:45.407663Z","shell.execute_reply":"2022-08-08T09:42:45.413226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Matric evaluation\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.metrics import accuracy_score\n# train_test_split: 80%, 20%\n# X_train, X_val, y_train, y_val = train_test_split(df, y, train_size=0.8, test_size=0.2)\n\n# modeling\nrf_clf = RandomForestClassifier(random_state=42)\n\n# Bundle preprocessing and modeling code in a pipeline\n# clf = Pipeline(steps=[\n#     ('model', rf_clf)\n# ])\n\n# train\nrf_clf.fit(X_train, y_train)\npreds = rf_clf.predict(X_valid)\n\n\n# confusion_matrix\ncm = confusion_matrix(y_valid, preds)\nsns.heatmap(cm, annot=True, fmt=\"d\")","metadata":{"execution":{"iopub.status.busy":"2022-08-08T11:08:51.808517Z","iopub.execute_input":"2022-08-08T11:08:51.810062Z","iopub.status.idle":"2022-08-08T11:09:03.102624Z","shell.execute_reply.started":"2022-08-08T11:08:51.810003Z","shell.execute_reply":"2022-08-08T11:09:03.101406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # from sklearn.model_selection import GridSearchCV\n# xgb = XGBClassifier(random_state = 42)\n# hyperparameters = { #'features__text__tfidf__max_df': [0.9, 0.95],\n# #                     #'features__text__tfidf__ngram_range': [(1,1), (1,2)],\n#                     'learning_rate': [0.1, 0.2],\n#                     'n_estimators': [20, 30, 50,100,150],\n#                     'max_depth': [2, 4],\n# #                     'min_samples_leaf': [2, 4],\n#                     'eval_metric' : ['rmse']\n#                    }\n# clf = GridSearchCV(xgb, hyperparameters, cv = 5)\n \n# # Fit and tune model\n# clf.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T10:54:03.352569Z","iopub.execute_input":"2022-08-08T10:54:03.353023Z","iopub.status.idle":"2022-08-08T10:56:05.376565Z","shell.execute_reply.started":"2022-08-08T10:54:03.352959Z","shell.execute_reply":"2022-08-08T10:56:05.375494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(clf.best_params_)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T10:56:16.801960Z","iopub.execute_input":"2022-08-08T10:56:16.803554Z","iopub.status.idle":"2022-08-08T10:56:16.810335Z","shell.execute_reply.started":"2022-08-08T10:56:16.803502Z","shell.execute_reply":"2022-08-08T10:56:16.809020Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xgb = XGBClassifier()\nxgb.fit(X_train,y_train)\nprint(\"XGB Score:\",xgb.score(X_valid,y_valid))","metadata":{"execution":{"iopub.status.busy":"2022-08-08T11:35:54.329152Z","iopub.execute_input":"2022-08-08T11:35:54.329764Z","iopub.status.idle":"2022-08-08T11:35:58.156653Z","shell.execute_reply.started":"2022-08-08T11:35:54.329720Z","shell.execute_reply":"2022-08-08T11:35:58.155244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#different approach for feature selection\n!pip install feature-engine","metadata":{"execution":{"iopub.status.busy":"2022-08-08T10:58:15.482745Z","iopub.execute_input":"2022-08-08T10:58:15.484253Z","iopub.status.idle":"2022-08-08T10:58:28.911594Z","shell.execute_reply.started":"2022-08-08T10:58:15.484167Z","shell.execute_reply":"2022-08-08T10:58:28.909936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from feature_engine.selection import (\n    DropDuplicateFeatures,\n    DropConstantFeatures,\n    DropDuplicateFeatures,\n    DropCorrelatedFeatures,\n    SmartCorrelatedSelection,\n    SelectByShuffling,\n    SelectBySingleFeaturePerformance,\n    RecursiveFeatureElimination,\n)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T10:59:07.192681Z","iopub.execute_input":"2022-08-08T10:59:07.193163Z","iopub.status.idle":"2022-08-08T10:59:07.221724Z","shell.execute_reply.started":"2022-08-08T10:59:07.193125Z","shell.execute_reply":"2022-08-08T10:59:07.220503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pipe = Pipeline([\n    # ======== FEATURE SELECTION =======\n    ('constant', DropConstantFeatures(tol=0.998)), # drops constand and quasi-constant altogether\n    ('duplicated', DropDuplicateFeatures()), # drop duplicated\n    ('shuffle', SelectByShuffling( # select by feature shuffling\n#         estimator = RandomForestClassifier(n_estimators=20, max_depth=2, random_state=1), # the model\n        estimator = XGBClassifier(n_estimators=20,learning_rate=0.1,max_depth=2,random_state = 42),\n        scoring=\"roc_auc\", # the metric to determine model performance\n        cv=3, # the cross-validation fold\n    )),\n    \n    # =====  the machine learning model ====\n#     ('random_forest', RandomForestClassifier(n_estimators=10, max_depth=2, random_state=1)),\n    ('xgbModel',XGBClassifier(n_estimators=20,learning_rate=0.1,max_depth=2,random_state = 42)),\n])\n\n# find features to remove\npipe.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T11:30:11.732217Z","iopub.execute_input":"2022-08-08T11:30:11.732697Z","iopub.status.idle":"2022-08-08T11:30:14.182416Z","shell.execute_reply.started":"2022-08-08T11:30:11.732663Z","shell.execute_reply":"2022-08-08T11:30:14.181068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import roc_auc_score\npred = pipe.predict_proba(X_valid)\nprint('Train roc-auc: {}'.format(roc_auc_score(y_valid, pred[:,1])))","metadata":{"execution":{"iopub.status.busy":"2022-08-08T11:37:16.994869Z","iopub.execute_input":"2022-08-08T11:37:16.995326Z","iopub.status.idle":"2022-08-08T11:37:17.022938Z","shell.execute_reply.started":"2022-08-08T11:37:16.995291Z","shell.execute_reply":"2022-08-08T11:37:17.021906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # 1. Approach with Random Forest\n\n# rf = RandomForestClassifier(n_estimators = 20 , max_depth = 2,\n#                            min_samples_leaf=2,random_state = 42,)\n# rf.fit(X_train, y_train)\n\n# print(\"Random forest algorithm result:\" , rf.score(X_valid, y_valid))","metadata":{"execution":{"iopub.status.busy":"2022-08-08T11:10:50.442238Z","iopub.execute_input":"2022-08-08T11:10:50.442649Z","iopub.status.idle":"2022-08-08T11:10:50.769496Z","shell.execute_reply.started":"2022-08-08T11:10:50.442617Z","shell.execute_reply":"2022-08-08T11:10:50.768044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = preprocessor.transform(test_data)\ntest_df.shape\n# preds = clf.predict(test_df)\n# probs = clf.predict_proba(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T09:42:55.616563Z","iopub.execute_input":"2022-08-08T09:42:55.617296Z","iopub.status.idle":"2022-08-08T09:42:55.681489Z","shell.execute_reply.started":"2022-08-08T09:42:55.617251Z","shell.execute_reply":"2022-08-08T09:42:55.680250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# preds = xgb.predict(test_df)\n# preds.shape\npipe_preds = pipe.predict(test_df)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T11:30:44.441509Z","iopub.execute_input":"2022-08-08T11:30:44.441951Z","iopub.status.idle":"2022-08-08T11:30:44.468354Z","shell.execute_reply.started":"2022-08-08T11:30:44.441917Z","shell.execute_reply":"2022-08-08T11:30:44.467401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample = pd.read_csv('../input/tabular-playground-series-aug-2022/sample_submission.csv')\nsample['failure'] = pipe_preds\nsample.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-08T11:31:21.886580Z","iopub.execute_input":"2022-08-08T11:31:21.887055Z","iopub.status.idle":"2022-08-08T11:31:21.907168Z","shell.execute_reply.started":"2022-08-08T11:31:21.886997Z","shell.execute_reply":"2022-08-08T11:31:21.906103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample.to_csv('six_submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T11:31:29.149797Z","iopub.execute_input":"2022-08-08T11:31:29.151491Z","iopub.status.idle":"2022-08-08T11:31:29.182635Z","shell.execute_reply.started":"2022-08-08T11:31:29.151438Z","shell.execute_reply":"2022-08-08T11:31:29.181500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}