{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Import libraries (A couple of them may be useless)","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport seaborn as sns\nfrom IPython.display import display\n\nfrom sklearn.decomposition import PCA\nfrom sklearn.experimental import enable_iterative_imputer\nfrom sklearn.preprocessing import RobustScaler, OrdinalEncoder, OneHotEncoder, LabelEncoder, StandardScaler, Normalizer, PowerTransformer\nfrom sklearn.impute import SimpleImputer, IterativeImputer, KNNImputer\nfrom sklearn.metrics import roc_auc_score, accuracy_score, f1_score\nfrom sklearn.model_selection import cross_val_score, KFold, StratifiedKFold, train_test_split, GroupKFold\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.svm import SVC\nfrom sklearn.naive_bayes import GaussianNB, MultinomialNB\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.base import TransformerMixin, BaseEstimator\nimport optuna\n\nfrom lightgbm import LGBMClassifier, LGBMRegressor\nfrom xgboost import XGBClassifier, XGBRegressor\n\nfrom eli5.sklearn import PermutationImportance\nimport eli5\n\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:39:59.762556Z","iopub.execute_input":"2022-08-04T14:39:59.763740Z","iopub.status.idle":"2022-08-04T14:40:09.547922Z","shell.execute_reply.started":"2022-08-04T14:39:59.763687Z","shell.execute_reply":"2022-08-04T14:40:09.546501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's examine data","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(\"../input/tabular-playground-series-aug-2022/train.csv\", index_col=\"id\")\ndf_test = pd.read_csv(\"../input/tabular-playground-series-aug-2022/test.csv\", index_col=\"id\")\n\ny_col = \"failure\"\n\nprint(\"Train: \")\ndisplay(df.head())\nprint(\"Test: \")\ndisplay(df_test.head())","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:40:09.550066Z","iopub.execute_input":"2022-08-04T14:40:09.550905Z","iopub.status.idle":"2022-08-04T14:40:09.887524Z","shell.execute_reply.started":"2022-08-04T14:40:09.550868Z","shell.execute_reply":"2022-08-04T14:40:09.886236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:40:09.889099Z","iopub.execute_input":"2022-08-04T14:40:09.889488Z","iopub.status.idle":"2022-08-04T14:40:09.998538Z","shell.execute_reply.started":"2022-08-04T14:40:09.889453Z","shell.execute_reply":"2022-08-04T14:40:09.997046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:40:10.000915Z","iopub.execute_input":"2022-08-04T14:40:10.001249Z","iopub.status.idle":"2022-08-04T14:40:10.096892Z","shell.execute_reply.started":"2022-08-04T14:40:10.001218Z","shell.execute_reply":"2022-08-04T14:40:10.095657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"object_cols = []\nfloat_cols = []\nint_cols = []\nnum_cols = []\nattr_cols = []\nmeasures_cols = []\nnan_cols = []\n\ndef reset_cols_arr(df):\n    global object_cols\n    global float_cols\n    global int_cols\n    global num_cols\n    global attr_cols\n    global measures_cols\n    global nan_cols\n    df = df.drop(y_col, axis=1)\n    \n    try:\n        df = df.drop(\"kfold\", axis=1)\n    except:\n        pass\n    \n    object_cols = np.array(df.select_dtypes(include=['object']).columns)\n    float_cols = np.array(df.select_dtypes(include=['float']).columns)\n    int_cols = np.array(df.select_dtypes(include=['int']).columns)\n    num_cols = np.array(df.select_dtypes(include=['float', 'int']).columns)\n    attr_cols = np.array([i for i in df.columns if i[:9] == \"attribute\"])\n    measures_cols = np.array([i for i in df.columns if i[:11] == \"measurement\"])\n    nan_cols = np.array([i for i in df.columns if df[i].isna().sum() > 0.0])\n    \nreset_cols_arr(df)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:40:10.098249Z","iopub.execute_input":"2022-08-04T14:40:10.098628Z","iopub.status.idle":"2022-08-04T14:40:10.136019Z","shell.execute_reply.started":"2022-08-04T14:40:10.098595Z","shell.execute_reply":"2022-08-04T14:40:10.134880Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def describe_df(title, train, test):\n    print(f\"{title} \\n\")\n    print(\"Train: \", train)\n    print(\"Test: \", test)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:40:10.137618Z","iopub.execute_input":"2022-08-04T14:40:10.138263Z","iopub.status.idle":"2022-08-04T14:40:10.143059Z","shell.execute_reply.started":"2022-08-04T14:40:10.138228Z","shell.execute_reply":"2022-08-04T14:40:10.141928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"describe_df(\"Shape\", df.shape, df_test.shape)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:40:10.144456Z","iopub.execute_input":"2022-08-04T14:40:10.144825Z","iopub.status.idle":"2022-08-04T14:40:10.154870Z","shell.execute_reply.started":"2022-08-04T14:40:10.144783Z","shell.execute_reply":"2022-08-04T14:40:10.153964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"describe_df(\"The number of NaN\", df.isna().sum().sum(), df_test.isna().sum().sum())","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:40:10.156137Z","iopub.execute_input":"2022-08-04T14:40:10.157055Z","iopub.status.idle":"2022-08-04T14:40:10.179408Z","shell.execute_reply.started":"2022-08-04T14:40:10.157019Z","shell.execute_reply":"2022-08-04T14:40:10.178585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"describe_df(\"The number of NaN\", df.isna().sum(), df_test.isna().sum())","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:40:10.180675Z","iopub.execute_input":"2022-08-04T14:40:10.181175Z","iopub.status.idle":"2022-08-04T14:40:10.200128Z","shell.execute_reply.started":"2022-08-04T14:40:10.181143Z","shell.execute_reply":"2022-08-04T14:40:10.199191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:40:10.204764Z","iopub.execute_input":"2022-08-04T14:40:10.205149Z","iopub.status.idle":"2022-08-04T14:40:10.229994Z","shell.execute_reply.started":"2022-08-04T14:40:10.205115Z","shell.execute_reply":"2022-08-04T14:40:10.228832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There is NaN values only in float cols","metadata":{}},{"cell_type":"code","source":"df[y_col].hist()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:41:18.164282Z","iopub.execute_input":"2022-08-04T14:41:18.164744Z","iopub.status.idle":"2022-08-04T14:41:18.447221Z","shell.execute_reply.started":"2022-08-04T14:41:18.164700Z","shell.execute_reply":"2022-08-04T14:41:18.445999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Since our target isn't balanced, we'll use a stratify to split our data on train and validation set (Cross Validation)","metadata":{}},{"cell_type":"code","source":"for i in num_cols:\n    plt.figure(figsize=(5, 5))\n    plt.subplots()\n    sns.distplot(df[i])","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:41:57.048514Z","iopub.execute_input":"2022-08-04T14:41:57.048938Z","iopub.status.idle":"2022-08-04T14:42:06.705823Z","shell.execute_reply.started":"2022-08-04T14:41:57.048899Z","shell.execute_reply":"2022-08-04T14:42:06.704842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The all measurements are normal distributed","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(16, 9))\n\nsns.heatmap(df.corr().abs(), annot=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:42:06.707470Z","iopub.execute_input":"2022-08-04T14:42:06.707824Z","iopub.status.idle":"2022-08-04T14:42:09.274148Z","shell.execute_reply.started":"2022-08-04T14:42:06.707794Z","shell.execute_reply":"2022-08-04T14:42:09.273187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The measurement_17 has a high correlation with the measurement_8, measurement_7, measurement_6 and measurement_5","metadata":{}},{"cell_type":"markdown","source":"Now we'll create folds for our dataset","metadata":{}},{"cell_type":"code","source":"# Creating folds\n\nkfold = GroupKFold(n_splits=5)\n\ndf[\"kfold\"] = -1\n\nfor fold, (_, valid) in enumerate(kfold.split(X=df.drop(y_col, axis=1), y=df[y_col], groups=df.product_code)):\n    df.loc[valid, \"kfold\"] = fold\n\n# index_k = {\n#     0: df[df.kfold == 0].index,\n#     1: df[df.kfold == 1].index,\n#     2: df[df.kfold == 2].index,\n#     3: df[df.kfold == 3].index,\n#     4: df[df.kfold == 4].index\n# }\n    \ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:44:53.420746Z","iopub.execute_input":"2022-08-04T14:44:53.421162Z","iopub.status.idle":"2022-08-04T14:44:53.472158Z","shell.execute_reply.started":"2022-08-04T14:44:53.421129Z","shell.execute_reply":"2022-08-04T14:44:53.470947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Encoding Category Features","metadata":{}},{"cell_type":"code","source":"[df[i].unique() for i in object_cols], [df_test[i].unique() for i in object_cols]","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:44:56.028693Z","iopub.execute_input":"2022-08-04T14:44:56.029170Z","iopub.status.idle":"2022-08-04T14:44:56.048078Z","shell.execute_reply.started":"2022-08-04T14:44:56.029136Z","shell.execute_reply":"2022-08-04T14:44:56.046936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Since the uniques of product_code in train set are different from the unique in test set, we can't encode it, we'll drop it. But we need a some information that tells us that the product_code is different (For example product_code was **A** and become **B**, and when product_code is becoming **B**, wee need some information that it was changed). And that is the reason why we use a GroupKFold instead KFold, StratifiedKFold and etc. So, now we must drop it","metadata":{}},{"cell_type":"code","source":"df1 = df.drop(\"product_code\", axis=1)\ndf_test_1 = df_test.drop(\"product_code\", axis=1)\ndf1","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:51:24.734381Z","iopub.execute_input":"2022-08-04T14:51:24.734778Z","iopub.status.idle":"2022-08-04T14:51:24.784611Z","shell.execute_reply.started":"2022-08-04T14:51:24.734748Z","shell.execute_reply":"2022-08-04T14:51:24.783309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df2 = df1.copy()\ndf_test_2 = df_test_1.copy()\n\ndf2.attribute_0 = df2.attribute_0.str.split(\"_\", expand=True)[1].astype(\"int\")\ndf2.attribute_1 = df2.attribute_1.str.split(\"_\", expand=True)[1].astype(\"int\")\n\ndf_test_2.attribute_0 = df_test_2.attribute_0.str.split(\"_\", expand=True)[1].astype(\"int\")\ndf_test_2.attribute_1 = df_test_2.attribute_1.str.split(\"_\", expand=True)[1].astype(\"int\")\n\nreset_cols_arr(df2)\ndf2.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:51:27.805086Z","iopub.execute_input":"2022-08-04T14:51:27.805505Z","iopub.status.idle":"2022-08-04T14:51:28.021775Z","shell.execute_reply.started":"2022-08-04T14:51:27.805467Z","shell.execute_reply":"2022-08-04T14:51:28.020475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_cols","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:51:28.716635Z","iopub.execute_input":"2022-08-04T14:51:28.717552Z","iopub.status.idle":"2022-08-04T14:51:28.725137Z","shell.execute_reply.started":"2022-08-04T14:51:28.717508Z","shell.execute_reply":"2022-08-04T14:51:28.723951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Write usefull functions","metadata":{}},{"cell_type":"markdown","source":"We'll write some functions, to use them in our GroupKFold","metadata":{}},{"cell_type":"code","source":"def scale(df, df_valid, df_test, cols, scaler=StandardScaler()):\n    df = df.copy()\n    df_valid = df_valid.copy()\n    df_test = df_test.copy()\n    \n    sc = scaler\n    sc.fit(df[cols])\n\n    df[cols] = pd.DataFrame(sc.transform(df[cols]), columns=cols, index=df.index)\n    df_valid[cols] = pd.DataFrame(sc.transform(df_valid[cols]), columns=cols, index=df_valid.index)\n    df_test[cols] = pd.DataFrame(sc.transform(df_test[cols]), columns=cols, index=df_test.index)\n    \n    return {\n        \"train\": df,\n        'valid': df_valid,\n        \"test\": df_test\n    }\n\nscale(df2, df_test_2, df_test_2, num_cols)[\"valid\"]","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:52:24.477569Z","iopub.execute_input":"2022-08-04T14:52:24.477959Z","iopub.status.idle":"2022-08-04T14:52:24.573875Z","shell.execute_reply.started":"2022-08-04T14:52:24.477928Z","shell.execute_reply":"2022-08-04T14:52:24.572612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def impute(df, df_valid, df_test, cols, imputer=IterativeImputer()):\n    df = df.copy()\n    df_valid = df_valid.copy()\n    df_test = df_test.copy()\n    \n    df[num_cols] = pd.DataFrame(imputer.fit_transform(df[num_cols]), columns=num_cols, index=df.index)\n    df_valid[num_cols] = pd.DataFrame(imputer.fit_transform(df_valid[num_cols]), columns=num_cols, index=df_valid.index)\n    df_test.loc[:, num_cols] = pd.DataFrame(imputer.transform(df_test[num_cols]), columns=num_cols, index=df_test.index)\n\n    return {\n        \"train\": df,\n        \"valid\": df_valid,\n        \"test\": df_test\n    }\n\nimpute(df2, df_test_2, df_test_2, num_cols)[\"valid\"]","metadata":{"execution":{"iopub.status.busy":"2022-08-04T15:01:07.557164Z","iopub.execute_input":"2022-08-04T15:01:07.557661Z","iopub.status.idle":"2022-08-04T15:01:33.087298Z","shell.execute_reply.started":"2022-08-04T15:01:07.557619Z","shell.execute_reply":"2022-08-04T15:01:33.085581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Modeling (BaseLine)","metadata":{}},{"cell_type":"code","source":"X = df2.copy()\nX_test = df_test_2.copy()\ny = X.pop(y_col)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:53:06.576082Z","iopub.execute_input":"2022-08-04T14:53:06.577182Z","iopub.status.idle":"2022-08-04T14:53:06.586408Z","shell.execute_reply.started":"2022-08-04T14:53:06.577133Z","shell.execute_reply":"2022-08-04T14:53:06.585149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now we'll build our model using a cv technique. If some scaling, imputing and etc. won't help us, we'll comment them","metadata":{}},{"cell_type":"code","source":"scores = []\nvalid_preds_1 = []\ntest_preds_1 = []\n\nfor i in range(5):\n    Xtest = X_test.copy()\n    X_train = X[X.kfold != i].drop(\"kfold\", axis=1)\n    X_valid = X[X.kfold == i].drop(\"kfold\", axis=1)\n    y_train = y.loc[X_train.index]\n    y_valid = y.loc[X_valid.index]\n    \n#     scaled = scale(X_train, X_valid, X_test, num_cols, scaler=RobustScaler())\n#     X_train = scaled[\"train\"]\n#     X_valid = scaled[\"valid\"]\n#     X_test = scaled[\"test\"]\n    \n#     imputed = impute(X_train, X_valid, X_test, num_cols, imputer=SimpleImputer(strategy=\"median\"))\n#     X_train = imputed[\"train\"]\n#     X_valid = imputed[\"valid\"]\n#     X_test = imputed[\"test\"]\n    \n    model = LGBMClassifier(n_estimators=1000, learning_rate=0.007, random_state=0)\n    model.fit(X_train, y_train)\n    \n    test_preds = model.predict_proba(Xtest)[:, 1]\n    valid_preds = model.predict_proba(X_valid)[:, 1]\n    \n    test_preds_1.append(test_preds)\n    valid_preds_1.append(valid_preds)\n    \n    roc_auc = roc_auc_score(y_valid, valid_preds)\n    score = accuracy_score(y_valid, valid_preds.round())\n    scores.append(score)\n    \n    print(roc_auc, score)\n\nprint(\"=================================\")\nprint(np.mean(scores))","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:55:11.538452Z","iopub.execute_input":"2022-08-04T14:55:11.538838Z","iopub.status.idle":"2022-08-04T14:55:37.076302Z","shell.execute_reply.started":"2022-08-04T14:55:11.538802Z","shell.execute_reply":"2022-08-04T14:55:37.075257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## And submitting!","metadata":{}},{"cell_type":"code","source":"pred = np.mean(np.column_stack(test_preds_1), axis=1)\npred","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:57:22.835402Z","iopub.execute_input":"2022-08-04T14:57:22.835834Z","iopub.status.idle":"2022-08-04T14:57:22.845203Z","shell.execute_reply.started":"2022-08-04T14:57:22.835800Z","shell.execute_reply":"2022-08-04T14:57:22.844207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ss = pd.read_csv(\"../input/tabular-playground-series-aug-2022/sample_submission.csv\")\nss.failure = pred\nss","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:57:25.530332Z","iopub.execute_input":"2022-08-04T14:57:25.531488Z","iopub.status.idle":"2022-08-04T14:57:25.560214Z","shell.execute_reply.started":"2022-08-04T14:57:25.531435Z","shell.execute_reply":"2022-08-04T14:57:25.559091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ss.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:57:27.045451Z","iopub.execute_input":"2022-08-04T14:57:27.045898Z","iopub.status.idle":"2022-08-04T14:57:27.103294Z","shell.execute_reply.started":"2022-08-04T14:57:27.045860Z","shell.execute_reply":"2022-08-04T14:57:27.101960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"*In processing...*","metadata":{}}]}