{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import GroupKFold\nfrom sklearn.metrics import roc_auc_score\nfrom lightgbm import LGBMClassifier\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.impute import KNNImputer\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.model_selection import StratifiedKFold\nimport warnings\nwarnings.filterwarnings('ignore')\nplt.style.use('seaborn-darkgrid')","metadata":{"execution":{"iopub.status.busy":"2022-08-12T03:32:09.968493Z","iopub.execute_input":"2022-08-12T03:32:09.969026Z","iopub.status.idle":"2022-08-12T03:32:09.977458Z","shell.execute_reply.started":"2022-08-12T03:32:09.968977Z","shell.execute_reply":"2022-08-12T03:32:09.976320Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = pd.read_csv('../input/tabular-playground-series-aug-2022/train.csv')\ntest_data = pd.read_csv('../input/tabular-playground-series-aug-2022/test.csv')\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T03:32:30.022106Z","iopub.execute_input":"2022-08-12T03:32:30.022539Z","iopub.status.idle":"2022-08-12T03:32:30.323140Z","shell.execute_reply.started":"2022-08-12T03:32:30.022501Z","shell.execute_reply":"2022-08-12T03:32:30.322085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Some EDA","metadata":{}},{"cell_type":"code","source":"# target distribution pie chart\ntrain_data.failure.value_counts().plot(kind='pie', autopct='%1.1f%%')","metadata":{"execution":{"iopub.status.busy":"2022-08-12T03:32:32.981985Z","iopub.execute_input":"2022-08-12T03:32:32.982409Z","iopub.status.idle":"2022-08-12T03:32:33.142260Z","shell.execute_reply.started":"2022-08-12T03:32:32.982374Z","shell.execute_reply":"2022-08-12T03:32:33.141110Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"numerical_cols = train_data.select_dtypes(include=[ 'float64']).columns\n# distribution of numerical columns\ntrain_data[numerical_cols].hist(bins=50, figsize=(20,20))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T03:32:34.955868Z","iopub.execute_input":"2022-08-12T03:32:34.956293Z","iopub.status.idle":"2022-08-12T03:32:38.236071Z","shell.execute_reply.started":"2022-08-12T03:32:34.956257Z","shell.execute_reply":"2022-08-12T03:32:38.235202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# categorical columns\ncategorical_cols = train_data.select_dtypes(include=[ 'object']).columns\n# distribution of categorical columns\nfor col in categorical_cols:\n    train_data[col].value_counts().plot(kind='bar')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T03:32:42.470891Z","iopub.execute_input":"2022-08-12T03:32:42.471766Z","iopub.status.idle":"2022-08-12T03:32:42.947160Z","shell.execute_reply.started":"2022-08-12T03:32:42.471718Z","shell.execute_reply":"2022-08-12T03:32:42.946010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# categorical_cols in test_data\nfor col in categorical_cols:\n    test_data[col].value_counts().plot(kind='bar')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T03:32:45.745071Z","iopub.execute_input":"2022-08-12T03:32:45.745454Z","iopub.status.idle":"2022-08-12T03:32:46.299689Z","shell.execute_reply.started":"2022-08-12T03:32:45.745422Z","shell.execute_reply":"2022-08-12T03:32:46.298526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We have to predict the failure for product F , I, G and H which are not in trainig data. So we have to use groupK fold cross validation.","metadata":{}},{"cell_type":"code","source":"# missing values\ntrain_data.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T03:32:49.029100Z","iopub.execute_input":"2022-08-12T03:32:49.029781Z","iopub.status.idle":"2022-08-12T03:32:49.043168Z","shell.execute_reply.started":"2022-08-12T03:32:49.029743Z","shell.execute_reply":"2022-08-12T03:32:49.041996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T03:32:51.307163Z","iopub.execute_input":"2022-08-12T03:32:51.307864Z","iopub.status.idle":"2022-08-12T03:32:51.321282Z","shell.execute_reply.started":"2022-08-12T03:32:51.307826Z","shell.execute_reply":"2022-08-12T03:32:51.320256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def fillna(df, cols, values):    \n#     new_df = df.copy()\n#     for col in cols:\n#         new_df[col] = new_df[col].fillna(values[col])\n#     return new_df\n# na_cols = train_data.columns[train_data.isnull().any()]\n# median = train_data[na_cols].median()\n# train = fillna(train_data, na_cols, median)\n# test = fillna(test_data, na_cols, median)\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from lightgbm import LGBMRegressor\nclass LightGBMInputer:\n    def __init__(self,df,**lgb_params):\n        self.df = df\n        self.verbose = False\n        self.lgb_params = lgb_params\n        self.nan_cols = df.columns[df.isnull().any()]\n    def impute(self):\n        for col in self.nan_cols:\n            if self.verbose:\n                print('Imputing %s' % col)\n            lightgbm = LGBMRegressor(**self.lgb_params)\n            nan_index = self.df[col].isnull()\n            train_index = ~nan_index\n            X_train = self.df.loc[train_index,self.df.columns != col]\n            y_train = self.df.loc[train_index,col]\n            X_test = self.df.loc[nan_index,self.df.columns != col]\n            lightgbm.fit(X_train,y_train,verbose=False)\n\n            self.df.loc[nan_index,col] = lightgbm.predict(X_test)\n        return self.df\n\n\n\n    \n","metadata":{"execution":{"iopub.status.busy":"2022-08-12T03:33:09.022626Z","iopub.execute_input":"2022-08-12T03:33:09.023062Z","iopub.status.idle":"2022-08-12T03:33:09.033166Z","shell.execute_reply.started":"2022-08-12T03:33:09.023027Z","shell.execute_reply":"2022-08-12T03:33:09.031794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train_data.copy()\ntest = test_data.copy()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T03:33:12.895838Z","iopub.execute_input":"2022-08-12T03:33:12.896250Z","iopub.status.idle":"2022-08-12T03:33:12.908823Z","shell.execute_reply.started":"2022-08-12T03:33:12.896215Z","shell.execute_reply":"2022-08-12T03:33:12.907741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# idea by ambrose\ntrain['m3_isna'] = train['measurement_3'].isna().astype(int)\ntest['m3_isna'] = test['measurement_3'].isna().astype(int)\ntrain['m5_isna'] = train['measurement_5'].isna().astype(int)\ntest['m5_isna'] = test['measurement_5'].isna().astype(int)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T03:33:23.029586Z","iopub.execute_input":"2022-08-12T03:33:23.029991Z","iopub.status.idle":"2022-08-12T03:33:23.041658Z","shell.execute_reply.started":"2022-08-12T03:33:23.029959Z","shell.execute_reply":"2022-08-12T03:33:23.040627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Feature engineering idea by \nhttps://www.kaggle.com/code/desalegngeb/tps08-logisticregression-and-some-fe","metadata":{}},{"cell_type":"code","source":"train['attribute2*3'] = train['attribute_2'] * train[   'attribute_3']\ntest['attribute2*3'] = test['attribute_2'] * test['attribute_3']\ntrain.drop(['attribute_2', 'attribute_3'], axis=1, inplace=True)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-12T03:35:06.755588Z","iopub.execute_input":"2022-08-12T03:35:06.756727Z","iopub.status.idle":"2022-08-12T03:35:06.773769Z","shell.execute_reply.started":"2022-08-12T03:35:06.756685Z","shell.execute_reply":"2022-08-12T03:35:06.772531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# impute\ntrain['attribute_0'] = train['attribute_0'].astype('category')\ntrain['attribute_1'] = train['attribute_1'].astype('category')\ntest['attribute_0'] = test['attribute_0'].astype('category')\ntest['attribute_1'] = test['attribute_1'].astype('category')\nfeature = [col for col in train if col not in ['failure','product_code','id']]\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-12T03:35:08.522967Z","iopub.execute_input":"2022-08-12T03:35:08.523388Z","iopub.status.idle":"2022-08-12T03:35:08.543499Z","shell.execute_reply.started":"2022-08-12T03:35:08.523353Z","shell.execute_reply":"2022-08-12T03:35:08.542377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meas_gr1_cols = [f\"measurement_{i:d}\" for i in list(range(3, 5)) + list(range(9, 17))]\ntrain['meas_gr1'] = np.mean(train[meas_gr1_cols], axis=1)\ntest['meas_gr1'] = np.mean(test[meas_gr1_cols], axis=1)\ntrain['std_gr1'] = np.std(train[meas_gr1_cols], axis=1)\ntest['std_gr1'] = np.std(test[meas_gr1_cols], axis=1)\ntrain.drop(meas_gr1_cols, axis=1, inplace=True)\ntest.drop(meas_gr1_cols, axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T03:35:14.471470Z","iopub.execute_input":"2022-08-12T03:35:14.471931Z","iopub.status.idle":"2022-08-12T03:35:14.522536Z","shell.execute_reply.started":"2022-08-12T03:35:14.471894Z","shell.execute_reply":"2022-08-12T03:35:14.521430Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndef fit(X_train, y_train, X_test, y_test,test_df,model):\n    test =test_df.copy()\n    # knn = KNNImputer(n_neighbors=3)\n    # na_features = X_train.columns[X_train.isnull().any()]\n    # knn.fit(X_train[na_features])\n    # X_train[na_features] = knn.transform(X_train[na_features])\n    # X_test[na_features] = knn.transform(X_test[na_features])\n    features = [col for col in X_train.columns if col not in ['product_code','id']]\n    cols_to_transform = ['attribute_0', 'attribute_1']\n    X_train[cols_to_transform] = X_train[cols_to_transform].astype('category')\n    X_test[cols_to_transform] = X_test[cols_to_transform].astype('category')\n    lgb_params ={\n        'boosting_type': 'gbdt',\n        'max_depth': 3,\n        'num_leaves': 20,\n    }\n    X_train = LightGBMInputer(X_train[features],**lgb_params).impute()\n    X_test = LightGBMInputer(X_test[features],**lgb_params).impute()\n    ohe = OneHotEncoder(sparse=False,handle_unknown='ignore')\n    # tranforms\n    train_ohe = ohe.fit_transform(X_train[cols_to_transform])\n\n    test_ohe = ohe.transform(X_test[cols_to_transform])\n    X_train.drop(cols_to_transform, axis=1, inplace=True)\n    X_test.drop(cols_to_transform, axis=1, inplace=True)\n    X_train[ohe.get_feature_names(cols_to_transform)] = train_ohe\n    X_test[ohe.get_feature_names(cols_to_transform)] = test_ohe\n    \n    features2 = [col for col in X_train.columns if col not in ['product_code','id']]\n    numerical_cols = X_train.select_dtypes(include=[ 'float64']).columns\n    scaler = StandardScaler()\n    X_train[numerical_cols] = scaler.fit_transform(X_train[numerical_cols])\n    X_test [numerical_cols] = scaler.transform(X_test [numerical_cols])\n    model.fit(X_train[features2], y_train)\n\n    y_pred = model.predict_proba(X_test[features2])[:,1]\n    print('AUC:', roc_auc_score(y_test, y_pred))\n    # test[na_features] = knn.transform(test[na_features])\n    test[cols_to_transform] = test[cols_to_transform].astype('category')\n    test = LightGBMInputer(test[features]).impute()\n    test_ohe = ohe.transform(test[cols_to_transform])\n    test_ohe = pd.DataFrame(test_ohe, columns=ohe.get_feature_names(cols_to_transform))\n    test.drop(cols_to_transform, axis=1, inplace=True)\n    test[ohe.get_feature_names(cols_to_transform)] = test_ohe\n\n    test[numerical_cols] = scaler.transform(test[numerical_cols])\n    \n    test_pred = model.predict_proba(test[features2])[:,1]\n\n    # importance = model.coef_.ravel()\n    return test_pred,roc_auc_score(y_test, y_pred), y_pred\n\ndef fit_lightgbm(X_train, y_train, X_test, y_test,test_df, model):\n    test =test_df.copy()\n    \n    # na_features = X_train.columns[X_train.isnull().any()]\n    # knn.fit(X_train[na_features])\n    # X_train[na_features] = knn.transform(X_train[na_features])\n    # X_test[na_features] = knn.transform(X_test[na_features])\n    features = [col for col in X_train.columns if col not in ['product_code','id']]\n    cols_to_transform = ['attribute_0', 'attribute_1']\n    X_train[cols_to_transform] = X_train[cols_to_transform].astype('category')\n    X_test[cols_to_transform] = X_test[cols_to_transform].astype('category')\n    lgb_params ={\n        'boosting_type': 'gbdt',\n        'max_depth': 2,\n        'num_leaves': 20,\n        'n_estimators':30\n    }\n    X_train = LightGBMInputer(X_train[features],**lgb_params).impute()\n    X_test = LightGBMInputer(X_test[features],**lgb_params).impute()\n    ohe = OneHotEncoder(sparse=False,handle_unknown='ignore')\n    # tranforms\n    \n   # for lightgbm\n    categorical_cols = train_data.select_dtypes(include=['category']).columns\n    model.fit(X_train[features], y_train, eval_set=[(X_train[features], y_train), (X_test[features], y_test)],early_stopping_rounds=120, verbose=False)\n    y_pred = model.predict_proba(X_test[features])[:,1]\n    print('AUC:', roc_auc_score(y_test, y_pred))\n    # test[na_features] = knn.transform(test[na_features])\n    test = LightGBMInputer(test[features]).impute()\n\n    test_pred = model.predict_proba(test[features])[:,1]\n    # importance = model.feature_importances_\n    return test_pred,roc_auc_score(y_test, y_pred), y_pred\n\ndef fit_fold(train,target, test,fit_logreg=True):\n    \"\"\"\n    train: train data\n    target: target\n    test: test data\n    fit_logreg: boolean, whether to fit logistic regression or Lightgbm\n    \"\"\"\n    kfold = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\n    score=0\n    test_pred = np.zeros(test.shape[0])\n    y_pred = 0\n   \n    for train_index, test_index in kfold.split(train, target, groups=train['product_code']):\n        X_train, X_test = train.iloc[train_index], train.iloc[test_index]\n        y_train, y_test = target.iloc[train_index], target.iloc[test_index]\n#         params = {\n#     'n_estimators': 350,\n#     'objective': 'binary',\n#     'metric': 'auc',\n#     'num_leaves': 34,\n#     'learning_rate': 0.06,\n#     'colsample_bytree':0.5,\n#     'subsample':0.5,\n#     'verbose': -1\n# }\n#         lgbm = LGBMClassifier(**params)\n        if fit_logreg:\n            logreg = LogisticRegression(penalty='l1', random_state=1, solver='liblinear',C= 0.01,class_weight = 'balanced') \n            test_pred_, score_, y_pred_ = fit(X_train, y_train, X_test, y_test,test,logreg)\n        else:\n            lgb_params ={\n                 'boosting_type': 'gbdt',\n                    'max_depth': 2,\n                    'num_leaves': 20,\n                    'learning_rate': 0.08,\n                    'n_estimators': 20\n                \n            }\n\n            model = LGBMClassifier(**lgb_params)\n            test_pred_, score_, y_pred_ = fit_lightgbm(X_train, y_train, X_test, y_test,test,model)\n\n\n\n        \n        test_pred += test_pred_\n        score += score_\n        # y_pred += y_pred_\n   \n    \n    return score/5, test_pred/5,y_pred/5\n\n\n\n    \n","metadata":{"execution":{"iopub.status.busy":"2022-08-12T03:35:19.862604Z","iopub.execute_input":"2022-08-12T03:35:19.863009Z","iopub.status.idle":"2022-08-12T03:35:19.891306Z","shell.execute_reply.started":"2022-08-12T03:35:19.862976Z","shell.execute_reply":"2022-08-12T03:35:19.889983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lightgbm_score , test_pred_lgb,y_pred = fit_fold(train.drop(['failure'],axis=1), train.failure, test,fit_logreg=False)\nprint('Lightgbm score:', lightgbm_score)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-12T03:35:23.003199Z","iopub.execute_input":"2022-08-12T03:35:23.003609Z","iopub.status.idle":"2022-08-12T03:35:35.675411Z","shell.execute_reply.started":"2022-08-12T03:35:23.003566Z","shell.execute_reply":"2022-08-12T03:35:35.673984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"logreg_score , test_pred_logreg,y_pred = fit_fold( train.drop(['failure'],axis=1), train.failure, test)\nprint('logreg_score:', logreg_score)\n# 0.5900","metadata":{"execution":{"iopub.status.busy":"2022-08-12T03:35:43.035479Z","iopub.execute_input":"2022-08-12T03:35:43.036445Z","iopub.status.idle":"2022-08-12T03:36:01.458393Z","shell.execute_reply.started":"2022-08-12T03:35:43.036400Z","shell.execute_reply":"2022-08-12T03:36:01.457099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"using logistic regression as final submission","metadata":{}},{"cell_type":"code","source":"sample_submission = pd.read_csv('../input/tabular-playground-series-aug-2022/sample_submission.csv')\nsample_submission.failure = test_pred_logreg\nsample_submission.to_csv('submission.csv', index=False)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Next ensembling multiple model","metadata":{}}]}