{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8d60e8b3-6c7c-4d40-9f54-964ded8246ad","_cell_guid":"c4d04121-be18-49f4-9bd8-de06b65df44c","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-08-06T08:30:39.664616Z","iopub.execute_input":"2022-08-06T08:30:39.665086Z","iopub.status.idle":"2022-08-06T08:30:39.676074Z","shell.execute_reply.started":"2022-08-06T08:30:39.665039Z","shell.execute_reply":"2022-08-06T08:30:39.674381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# load the data\nroot = os.path.join(\"..\",\"input\",\"tabular-playground-series-aug-2022\")\ndf_train = pd.read_csv(os.path.join(root, \"train.csv\"), index_col=0)\ndf_test = pd.read_csv(os.path.join(root, \"test.csv\"), index_col=0)\n\ndf_all = pd.concat([df_train.copy(), df_test.copy()])\n\ndf_test","metadata":{"execution":{"iopub.status.busy":"2022-08-06T08:30:44.119867Z","iopub.execute_input":"2022-08-06T08:30:44.121396Z","iopub.status.idle":"2022-08-06T08:30:44.482879Z","shell.execute_reply.started":"2022-08-06T08:30:44.121328Z","shell.execute_reply":"2022-08-06T08:30:44.481088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# impute by product code\n\nfrom sklearn.experimental import enable_iterative_imputer  # noqa\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.impute import KNNImputer\nfrom sklearn.impute import IterativeImputer\n\nproduct_codes = df_all[\"product_code\"].unique().tolist()\nfor p in product_codes:\n    print(p)\n    \n    # imputer = SimpleImputer(strategy=\"median\")\n    imputer = KNNImputer(n_neighbors=5)\n    # imputer = IterativeImputer(random_state=0)\n\n    features = [c for c in df_all.columns if c != \"failure\"]\n    numerical_features = [c for c in features if df_all[c].dtype == \"float64\" or df_all[c].dtype == \"int64\"]\n    \n    mask = (df_all[\"product_code\"] == p)\n    \n    X = df_all.loc[mask, numerical_features].values\n    print(mask.sum(), X.shape)\n\n    print(\"df_all.isna().sum()\", df_all.loc[mask, numerical_features].isna().sum().sum())\n    df_all.loc[mask, numerical_features] = imputer.fit_transform(X)\n    print(\"df_all.isna().sum()\", df_all.loc[mask, numerical_features].isna().sum().sum())\n\ndf_train = df_all[~df_all[\"failure\"].isna()].copy()\ndf_test  = df_all[ df_all[\"failure\"].isna()].copy()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T08:30:48.988822Z","iopub.execute_input":"2022-08-06T08:30:48.989283Z","iopub.status.idle":"2022-08-06T08:31:02.890213Z","shell.execute_reply.started":"2022-08-06T08:30:48.989243Z","shell.execute_reply":"2022-08-06T08:31:02.888936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pre-process\n\nfrom sklearn.preprocessing import StandardScaler, RobustScaler, PowerTransformer\n\nfeatures = [c for c in df_all.columns if c != \"failure\"]\nnumerical_features = [c for c in features if df_all[c].dtype == \"float64\" or df_all[c].dtype == \"int64\"]\n\ntransformer = StandardScaler()\ntransformer.fit(df_all[numerical_features])\n\ndata = transformer.transform(df_all[numerical_features])\nfor n,f in enumerate(numerical_features):\n    df_all[f] = data[:, n]\n    \ndf_train = df_all[~df_all[\"failure\"].isna()].copy()\ndf_test  = df_all[ df_all[\"failure\"].isna()].copy()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T08:31:06.017037Z","iopub.execute_input":"2022-08-06T08:31:06.018044Z","iopub.status.idle":"2022-08-06T08:31:06.080606Z","shell.execute_reply.started":"2022-08-06T08:31:06.017978Z","shell.execute_reply":"2022-08-06T08:31:06.078855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_all","metadata":{"execution":{"iopub.status.busy":"2022-08-06T08:31:15.183846Z","iopub.execute_input":"2022-08-06T08:31:15.184329Z","iopub.status.idle":"2022-08-06T08:31:15.235183Z","shell.execute_reply.started":"2022-08-06T08:31:15.184290Z","shell.execute_reply":"2022-08-06T08:31:15.234077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# quick label encode\n\n# attributes_0 = df_all[\"attribute_0\"].unique().tolist()\n# df_all[\"attribute_0\"] = df_all[\"attribute_0\"].apply(lambda x : attributes_0.index(x))\ndf_all[\"attribute_0\"] =  df_all[\"attribute_0\"].astype('category')\n\n# attributes_1 = df_all[\"attribute_1\"].unique().tolist()\n# df_all[\"attribute_1\"] = df_all[\"attribute_1\"].apply(lambda x : attributes_1.index(x))\ndf_all[\"attribute_1\"] =  df_all[\"attribute_1\"].astype('category')\n\ndf_train = df_all[~df_all[\"failure\"].isna()].copy()\ndf_test  = df_all[ df_all[\"failure\"].isna()].copy()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T08:31:19.726831Z","iopub.execute_input":"2022-08-06T08:31:19.727330Z","iopub.status.idle":"2022-08-06T08:31:19.762211Z","shell.execute_reply.started":"2022-08-06T08:31:19.727288Z","shell.execute_reply":"2022-08-06T08:31:19.760726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#list features\nfeatures = [c for c in df_all.columns if c != \"failure\" and c != \"product_code\"]","metadata":{"execution":{"iopub.status.busy":"2022-08-06T08:31:22.052107Z","iopub.execute_input":"2022-08-06T08:31:22.052608Z","iopub.status.idle":"2022-08-06T08:31:22.058258Z","shell.execute_reply.started":"2022-08-06T08:31:22.052569Z","shell.execute_reply":"2022-08-06T08:31:22.057001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# function to generate the query counts \ndef generate_qid_groups(qid):\n    qid_groups = []\n    memo = \"\"\n    count = 0\n    for q in qid:\n        if q != memo: \n            if memo != \"\": qid_groups.append(count) \n            memo = q\n            count = 1\n        else: count += 1\n    qid_groups.append(count)\n    return qid_groups\n\nqueries = df_train[\"product_code\"]\ngenerate_qid_groups(queries)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T08:40:29.354023Z","iopub.execute_input":"2022-08-06T08:40:29.354561Z","iopub.status.idle":"2022-08-06T08:40:29.373058Z","shell.execute_reply.started":"2022-08-06T08:40:29.354513Z","shell.execute_reply":"2022-08-06T08:40:29.371421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#optimized params for LGBM\n\nparams_LGB = {\n 'boosting_type': 'gbdt',\n 'class_weight': None,\n 'colsample_bytree': 1.0,\n 'importance_type': 'gain',\n 'learning_rate': 0.06,\n 'max_depth': 2,\n 'min_child_samples': 20,\n 'min_child_weight': 0.001,\n 'min_split_gain': 0.0,\n 'n_estimators': 500,\n 'n_jobs': -1,\n 'num_leaves': 4,\n 'objective': 'binary',\n 'random_state': None,\n 'reg_alpha': 0.0,\n 'reg_lambda': 0.0,\n 'silent': 'warn',\n 'subsample': 1.0,\n 'subsample_freq': 0,\n 'max_bins': 187,\n 'metric': 'auc', #'binary_logloss', #'auc',\n 'scale_pos_weight': 1,\n 'seed': 42}","metadata":{"execution":{"iopub.status.busy":"2022-08-06T08:38:10.222983Z","iopub.execute_input":"2022-08-06T08:38:10.223438Z","iopub.status.idle":"2022-08-06T08:38:10.233172Z","shell.execute_reply.started":"2022-08-06T08:38:10.223404Z","shell.execute_reply":"2022-08-06T08:38:10.231455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# build model, CV over product_codes\n\nfrom lightgbm import LGBMRanker\n\nimport warnings\nwarnings.filterwarnings('ignore')\n\nmodels = []\nproduct_codes = df_train[\"product_code\"].unique().tolist()\nfor product_code in product_codes: \n    \n    print(\"product code\", product_code)\n\n    features = [c for c in df_all.columns if c != \"failure\" and c != \"product_code\"]\n\n    train_mask = (df_train[\"product_code\"] != product_code)\n    test_mask = (df_train[\"product_code\"] == product_code)\n\n    X_train = df_train.loc[train_mask, features]\n    y_train = df_train.loc[train_mask][\"failure\"]\n    queries_train = df_train.loc[train_mask][\"product_code\"]\n\n    X_test = df_train.loc[test_mask, features]\n    y_test = df_train.loc[test_mask][\"failure\"]\n    queries_test = df_train.loc[test_mask][\"product_code\"]\n    \n    models.append(LGBMRanker(**params_LGB))\n    models[-1] = models[-1].fit(\n        X_train, y_train, \n        group=generate_qid_groups(queries_train), \n        eval_set=[(X_test, y_test)],\n        eval_group=[generate_qid_groups(queries_test)],\n        verbose=-1,\n        early_stopping_rounds=100\n    )\n    print(\"auc\", models[-1].best_score_[\"valid_0\"][\"auc\"])","metadata":{"execution":{"iopub.status.busy":"2022-08-06T08:38:12.166879Z","iopub.execute_input":"2022-08-06T08:38:12.167984Z","iopub.status.idle":"2022-08-06T08:38:14.165507Z","shell.execute_reply.started":"2022-08-06T08:38:12.167934Z","shell.execute_reply":"2022-08-06T08:38:14.164114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check the models, OOF AUC\n\ndf_train[\"pred\"] = 0.0\n\nfor i, m in enumerate(models):\n    print(\"model\", i)\n    product_code = product_codes[i] \n    \n    print(\"product_code\", product_code)\n    test_mask = (df_train[\"product_code\"] == product_code)\n\n    X_test = df_train.loc[test_mask, features]\n    y_test = df_train.loc[test_mask][\"failure\"]\n\n    preds = m.predict(X_test)\n    df_train.loc[test_mask, \"pred\"] = df_train.loc[test_mask, \"pred\"] + preds\n","metadata":{"execution":{"iopub.status.busy":"2022-08-06T08:38:16.821585Z","iopub.execute_input":"2022-08-06T08:38:16.821984Z","iopub.status.idle":"2022-08-06T08:38:16.930488Z","shell.execute_reply.started":"2022-08-06T08:38:16.821950Z","shell.execute_reply":"2022-08-06T08:38:16.929314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom sklearn.metrics import roc_curve, auc, roc_auc_score, RocCurveDisplay, log_loss\nprint(\"auc\", roc_auc_score(df_train[\"failure\"].values, df_train[\"pred\"].values))    \nprint(\"log loss\", log_loss(df_train[\"failure\"].values, df_train[\"pred\"].values))    \n\n\nfpr, tpr, thresholds = roc_curve(df_train[\"failure\"].values, df_train[\"pred\"].values)\nroc_auc = auc(fpr, tpr)\ndisplay = RocCurveDisplay(fpr=fpr, tpr=tpr, roc_auc=roc_auc, estimator_name='example estimator')\ndisplay.plot()\nplt.show()\n\n# 0.5852832565746557","metadata":{"execution":{"iopub.status.busy":"2022-08-06T08:38:20.235762Z","iopub.execute_input":"2022-08-06T08:38:20.236280Z","iopub.status.idle":"2022-08-06T08:38:20.423098Z","shell.execute_reply.started":"2022-08-06T08:38:20.236242Z","shell.execute_reply":"2022-08-06T08:38:20.422011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# compute predictions for test data\n\nfrom scipy.stats import rankdata\n\ndf_test[\"pred\"] = 0.0\nproduct_codes = df_test[\"product_code\"].unique().tolist()\nfor product_code in product_codes: \n    print(product_code)\n    \n    for i, m in enumerate(models):\n        print(\"model\", i)\n        \n        test_mask = (df_test[\"product_code\"] == product_code)\n\n        X_test = df_test.loc[test_mask, features]\n\n        preds = m.predict(X_test)\n        df_test.loc[test_mask, \"pred\"] = df_test.loc[test_mask, \"pred\"] + preds\n        \n    df_test.loc[test_mask, \"pred\"] = ( (rankdata(df_test.loc[test_mask, \"pred\"]) - 1) / (df_test.loc[test_mask, \"pred\"].shape[0] - 1))\n\ndf_test[\"pred\"] = ( (rankdata(df_test[\"pred\"]) - 1) / (df_test[\"pred\"].shape[0] - 1))","metadata":{"execution":{"iopub.status.busy":"2022-08-06T08:27:52.654548Z","iopub.execute_input":"2022-08-06T08:27:52.655012Z","iopub.status.idle":"2022-08-06T08:27:52.925820Z","shell.execute_reply.started":"2022-08-06T08:27:52.654972Z","shell.execute_reply":"2022-08-06T08:27:52.924666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test[\"pred\"].hist(bins=100)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T08:27:55.787560Z","iopub.execute_input":"2022-08-06T08:27:55.788041Z","iopub.status.idle":"2022-08-06T08:27:56.153932Z","shell.execute_reply.started":"2022-08-06T08:27:55.788002Z","shell.execute_reply":"2022-08-06T08:27:56.152645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create submission file \n\ndf_submission = pd.read_csv(os.path.join(root, \"sample_submission.csv\"), index_col=0)\n\ndf_submission = df_submission.join(df_test[\"pred\"])\ndf_submission[\"failure\"] = df_submission[\"pred\"]\ndf_submission[\"failure\"].to_csv(\"submission.csv\")\ndf_submission[\"failure\"]\n","metadata":{"execution":{"iopub.status.busy":"2022-08-06T08:27:59.704161Z","iopub.execute_input":"2022-08-06T08:27:59.704642Z","iopub.status.idle":"2022-08-06T08:27:59.785704Z","shell.execute_reply.started":"2022-08-06T08:27:59.704600Z","shell.execute_reply":"2022-08-06T08:27:59.784285Z"},"trusted":true},"execution_count":null,"outputs":[]}]}