{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Introduction #\n\n* [1/3] [Data Preparation][1]\n* [2/3] [CatBoost Ranker Training][2]\n* **[3/3] [CatBoost Ranker Inference][3] ← (this kernel)**\n\n### About Solution\n\n- Feature data\n    - **Markdown string (64 tokens) & No code string**\n    - From markdown string to Distilbert feature vector (768 vector)\n- Target data\n    - Set code cell position target value\n    - Set markdown cell posision target value with increasing the target value by linearly\n    - **So markdown cell's target values can be over 100 (Normalizing is not applied)**\n    - This way can be helpful for calculating unbiased target value on markdown cell\n    - Since this way doesn't refer to exact position of markdown cell\n- Model and hyperparameters\n    - Model : CatBoost Ranker\n    - Loss : RMSE (by groups)\n    - Hyperparameters : Tuning with Optuna\n    - ~~**Log transformation on markdown cell position target value (for normal distribution on the target value)**~~\n   \nI refer to the form of **[NICK KUZMENKOV][7]'s kernels** which are listed below :)<br>\n* [1/3] [Data Preparation][4]\n* [2/3] [TPU Training][5] (~4 hours)\n* [3/3] [GPU Inference][6] (~2 hours)\n\n[1]: https://www.kaggle.com/cafelatte1/catboost-ranker-data-preparation\n[2]: https://www.kaggle.com/cafelatte1/catboost-ranker-training\n[3]: https://www.kaggle.com/cafelatte1/catboost-ranker-inference\n[4]: https://www.kaggle.com/nickuzmenkov/ai4code-tf-tpu-codebert-data-preparation/notebook\n[5]: https://www.kaggle.com/nickuzmenkov/ai4code-tf-tpu-codebert-training\n[6]: https://www.kaggle.com/nickuzmenkov/ai4code-tf-tpu-codebert-inference\n[7]: https://www.kaggle.com/nickuzmenkov","metadata":{}},{"cell_type":"markdown","source":"# Setup #","metadata":{}},{"cell_type":"code","source":"# import sys\n# !cp ../input/rapids/rapids.21.06 /opt/conda/envs/rapids.tar.gz\n# !cd /opt/conda/envs/ && tar -xzvf rapids.tar.gz > /dev/null\n# sys.path = [\"/opt/conda/envs/rapids/lib/python3.7/site-packages\"] + sys.path\n# sys.path = [\"/opt/conda/envs/rapids/lib/python3.7\"] + sys.path\n# sys.path = [\"/opt/conda/envs/rapids/lib\"] + sys.path \n# !cp /opt/conda/envs/rapids/lib/libxgboost.so /opt/conda/lib/","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-13T03:41:39.169184Z","iopub.execute_input":"2022-07-13T03:41:39.169806Z","iopub.status.idle":"2022-07-13T03:41:39.193240Z","shell.execute_reply.started":"2022-07-13T03:41:39.169712Z","shell.execute_reply":"2022-07-13T03:41:39.192502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport sys\nimport shutil\nfrom glob import glob\nimport multiprocessing as mp\nimport gc\nfrom pathlib import Path\nfrom scipy import stats\nfrom scipy.special import boxcox, softmax\nfrom scipy import sparse\n\nfrom multiprocessing import cpu_count\nimport copy\nimport pickle\nimport warnings\nfrom datetime import datetime, timedelta\nfrom time import time, sleep, mktime\nfrom matplotlib import font_manager as fm, rc, rcParams\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom tqdm import tqdm\nimport re\nimport random as rnd\nimport psutil\nfrom optuna import Trial, create_study\nfrom optuna.samplers import TPESampler\n\nimport numpy as np\nfrom numpy import array, nan, random as np_rnd, where\nimport pandas as pd\nfrom pandas import DataFrame as dataframe, Series as series, isna, read_csv\nfrom pandas.tseries.offsets import DateOffset\n\nfrom sklearn.model_selection import train_test_split as tts, StratifiedKFold, StratifiedShuffleSplit\nfrom sklearn.preprocessing import LabelEncoder, OneHotEncoder, StandardScaler, MinMaxScaler, RobustScaler, KBinsDiscretizer\nfrom sklearn import metrics\nfrom sklearn.compose import ColumnTransformer\n# config_missingpy(); from missingpy import MissForest\nfrom sklearn.impute import KNNImputer\nfrom optuna import Trial, create_study\nfrom sklearn.model_selection import GroupKFold, GroupShuffleSplit, StratifiedGroupKFold\nfrom sklearn.feature_extraction.text import TfidfVectorizer\n\ntry:\n    import cudf as cd\n    import cupy as cp\n    from cuml.cluster import KMeans\n    from cuml.neighbors import NearestNeighbors\n    from cuml.metrics.cluster import silhouette_score\nexcept:\n    print(\"RAPIDS Import ERROR\")\n\nimport catboost as cat\n\n# # ===== tensorflow =====\n# import tensorflow as tf\n# from tensorflow import random as tf_rnd\n# from tensorflow.keras.models import Model\n# from tensorflow.keras.models import Sequential\n# from tensorflow.keras import layers\n# from tensorflow.keras import activations\n# from tensorflow.keras import optimizers\n# from tensorflow.keras import metrics as tf_metrics\n# from tensorflow.keras import callbacks as tf_callbacks\n# from tqdm.keras import TqdmCallback\n# import tensorflow_addons as tfa\n# from tensorflow.keras.utils import plot_model\n# from keras.utils.layer_utils import count_params\n\n# import keras_tuner as kt\n# from keras_tuner import HyperModel\n# import tensorflow_hub as tf_hub\n# import tensorflow_recommenders as tfrs\n\n# # GPU check\n# if tf.test.gpu_device_name() != '/device:GPU:0':\n#     print('GPU device not found')\n# else:\n#     print('Found GPU')\n\n# # GPU memory setting\n# gpus = tf.config.list_physical_devices('GPU')\n# if gpus:\n#   try:\n#     tf.config.experimental.set_memory_growth(gpus[0], True)\n#   except RuntimeError as e:\n#     print(e)\n\nwarnings.filterwarnings(action='ignore')\nrcParams['axes.unicode_minus'] = False\npd.set_option('display.max_columns', 100)\npd.set_option('display.max_rows', 100)\npd.set_option('display.width', 1000)\npd.set_option('max_colwidth', 200)\n# plt.rc('font', family='NanumSquareB')\n\ndata_dir = Path('../input/AI4Code')","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-13T03:41:39.194822Z","iopub.execute_input":"2022-07-13T03:41:39.195082Z","iopub.status.idle":"2022-07-13T03:41:44.994550Z","shell.execute_reply.started":"2022-07-13T03:41:39.195048Z","shell.execute_reply":"2022-07-13T03:41:44.993702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ===== utility functions =====\n# label encoding for categorical column with excepting na value\ndef seed_everything(seed=42):\n    # python random module\n    rnd.seed(seed)\n    # numpy random\n    np_rnd.seed(seed)\n#     # tf random\n#     tf_rnd.set_seed(seed)\n    # RAPIDS random\n    try:\n        cp.random.seed(seed)\n    except:\n        pass\ndef which(bool_list):\n    return where(bool_list)[0]\ndef easyIO(x=None, path=None, op=\"r\"):\n    tmp = None\n    if op == \"r\":\n        with open(path, \"rb\") as f:\n            tmp = pickle.load(f)\n        return tmp\n    elif op == \"w\":\n        with open(path, \"wb\") as f:\n            pickle.dump(x, f)\n    else:\n        print(\"Unknown operation type\")\ndef diff(first, second):\n    second = set(second)\n    return [item for item in first if item not in second]\ndef findIdx(data_x, col_names):\n    return [int(i) for i, j in enumerate(data_x) if j in col_names]\ndef orderElems(for_order, using_ref):\n    return [i for i in using_ref if i in for_order]\n# concatenate by row\ndef cbr(df1, df2):\n    if type(df1) == series:\n        tmp_concat = series(pd.concat([dataframe(df1), dataframe(df2)], axis=0, ignore_index=True).iloc[:,0])\n        tmp_concat.reset_index(drop=True, inplace=True)\n    elif type(df1) == dataframe:\n        tmp_concat = pd.concat([df1, df2], axis=0, ignore_index=True)\n        tmp_concat.reset_index(drop=True, inplace=True)\n    elif type(df1) == np.ndarray:\n        tmp_concat = np.concatenate([df1, df2], axis=0)\n    else:\n        print(\"Unknown Type: return 1st argument\")\n        tmp_concat = df1\n    return tmp_concat\ndef change_width(ax, new_value):\n    for patch in ax.patches :\n        current_width = patch.get_width()\n        adj_value = current_width - new_value\n        # we change the bar width\n        patch.set_width(new_value)\n        # we recenter the bar\n        patch.set_x(patch.get_x() + adj_value * .5)\ndef week_of_month(date):\n    month = date.month\n    week = 0\n    while date.month == month:\n        week += 1\n        date -= timedelta(days=7)\n    return week\ndef getSeason(date):\n    month = date.month\n    if month in [3, 4, 5]:\n        return \"Spring\"\n    elif month in [6, 7, 8]:\n        return \"Summer\"\n    elif month in [9, 10, 11]:\n        return \"Fall\"\n    else:\n        return \"Winter\"\ndef createFolder(directory):\n    try:\n        if not os.path.exists(directory):\n            os.makedirs(directory)\n    except OSError:\n        print('Error: Creating directory. ' + directory)\n# def softmax(x):\n#     max = np.max(x, axis=1, keepdims=True)  # returns max of each row and keeps same dims\n#     e_x = np.exp(x - max)  # subtracts each row with its max value\n#     sum = np.sum(e_x, axis=1, keepdims=True)  # returns sum of each row and keeps same dims\n#     f_x = e_x / sum\n#     return f_x\ndef sigmoid(x):\n    return 1/(1 + np.exp(-x))\ndef dispPerformance(result_dic):\n    perf_table = dataframe()\n    index_names = []\n    for k, v in result_dic.items():\n        index_names.append(k)\n        perf_table = pd.concat([perf_table, series(v[\"performance\"]).to_frame().T], ignore_index=True, axis=0)\n    perf_table.index = index_names\n    perf_table.sort_values(perf_table.columns[0], inplace=True)\n    print(perf_table)\n    return perf_table\ndef powspace(start, stop, power, num):\n    start = np.power(start, 1/float(power))\n    stop = np.power(stop, 1/float(power))\n    return np.power(np.linspace(start, stop, num=num), power)\ndef xgb_custom_lossfunction(alpha = 1):\n    def support_under_mse(label, pred):\n        # grad : 1차 미분\n        # hess : 2차 미분\n        residual = (label - pred).astype(\"float\")\n        grad = np.where(residual > 0, -2 * alpha * residual, -2 * residual)\n        hess = np.where(residual > 0, 2 * alpha, 2.0)\n        return grad, hess\n    return support_under_mse\ndef pd_flatten(df):\n    df = df.unstack()\n    df.index = [str(i) + \"_\" + str(j) for i, j in df.index]\n    return df\ndef tf_losses_rmse(y_true, y_pred, sample_weight=None):\n    return tf.sqrt(tf.reduce_mean((y_true - y_pred) ** 2)) if sample_weight is None else tf.sqrt(tf.reduce_mean(((y_true - y_pred) ** 2) * sample_weight))\ndef tf_loss_nmae(y_true, y_pred, sample_weight=False):\n    mae = tf.reduce_mean(tf.math.abs(y_true - y_pred))\n    score = tf.math.divide(mae, tf.reduce_mean(tf.math.abs(y_true)))\n    return score\ndef text_extractor(string, lang=\"eng\", spacing=True):\n    # # 괄호를 포함한 괄호 안 문자 제거 정규식\n    # re.sub(r'\\([^)]*\\)', '', remove_text)\n    # # <>를 포함한 <> 안 문자 제거 정규식\n    # re.sub(r'\\<[^)]*\\>', '', remove_text)\n    if lang == \"eng\":\n        text_finder = re.compile('[^ A-Za-z]') if spacing else re.compile('[^A-Za-z]')\n    elif lang == \"kor\":\n        text_finder = re.compile('[^ ㄱ-ㅣ가-힣+]') if spacing else re.compile('[^ㄱ-ㅣ가-힣+]')\n    # default : kor + eng\n    else:\n        text_finder = re.compile('[^ A-Za-zㄱ-ㅣ가-힣+]') if spacing else re.compile('[^A-Za-zㄱ-ㅣ가-힣+]')\n    return text_finder.sub('', string)\ndef memory_usage(message='debug'):\n    # current process RAM usage\n    p = psutil.Process()\n    rss = p.memory_info().rss / 2 ** 20 # Bytes to MB\n    print(f\"[{message}] memory usage: {rss: 10.3f} MB\")\n    return rss\nclass MyLabelEncoder:\n    def __init__(self, preset={}):\n        # dic_cat format -> {\"col_name\": {\"value\": replace}}\n        self.dic_cat = preset\n    def fit_transform(self, data_x, col_names):\n        tmp_x = copy.deepcopy(data_x)\n        for i in col_names:\n            # if key is not in dic, update dic\n            if i not in self.dic_cat.keys():\n                tmp_dic = dict.fromkeys(sorted(set(tmp_x[i]).difference([nan])))\n                label_cnt = 0\n                for j in tmp_dic.keys():\n                    tmp_dic[j] = label_cnt\n                    label_cnt += 1\n                self.dic_cat[i] = tmp_dic\n            # transform value which is not in dic to nan\n            tmp_x[i] = tmp_x[i].astype(\"object\")\n            conv = tmp_x[i].replace(self.dic_cat[i])\n            for conv_idx, j in enumerate(conv):\n                if j not in self.dic_cat[i].values():\n                    conv[conv_idx] = nan\n            # final return\n            tmp_x[i] = conv.astype(\"float\")\n        return tmp_x\n    def transform(self, data_x):\n        tmp_x = copy.deepcopy(data_x)\n        for i in self.dic_cat.keys():\n            # transform value which is not in dic to nan\n            tmp_x[i] = tmp_x[i].astype(\"object\")\n            conv = tmp_x[i].replace(self.dic_cat[i])\n            for conv_idx, j in enumerate(conv):\n                if j not in self.dic_cat[i].values():\n                    conv[conv_idx] = nan\n            # final return\n            tmp_x[i] = conv.astype(\"float\")\n        return tmp_x\n    def clear(self):\n        self.dic_cat = {}\nclass MyOneHotEncoder:\n    def __init__(self, label_preset={}):\n        self.dic_cat = {}\n        self.label_preset = label_preset\n    def fit_transform(self, data_x, col_names):\n        tmp_x = dataframe()\n        for i in data_x:\n            if i not in col_names:\n                tmp_x = pd.concat([tmp_x, dataframe(data_x[i])], axis=1)\n            else:\n                if not ((data_x[i].dtype.name == \"object\") or (data_x[i].dtype.name == \"category\")):\n                    print(F\"WARNING : {i} is not object or category\")\n                self.dic_cat[i] = OneHotEncoder(sparse=False, handle_unknown=\"ignore\")\n                conv = self.dic_cat[i].fit_transform(dataframe(data_x[i])).astype(\"int\")\n                col_list = []\n                for j in self.dic_cat[i].categories_[0]:\n                    if i in self.label_preset.keys():\n                        for k, v in self.label_preset[i].items():\n                            if v == j:\n                                col_list.append(str(i) + \"_\" + str(k))\n                    else:\n                        col_list.append(str(i) + \"_\" + str(j))\n                conv = dataframe(conv, columns=col_list)\n                tmp_x = pd.concat([tmp_x, conv], axis=1)\n        return tmp_x\n    def transform(self, data_x):\n        tmp_x = dataframe()\n        for i in data_x:\n            if not i in list(self.dic_cat.keys()):\n                tmp_x = pd.concat([tmp_x, dataframe(data_x[i])], axis=1)\n            else:\n                if not ((data_x[i].dtype.name == \"object\") or (data_x[i].dtype.name == \"category\")):\n                    print(F\"WARNING : {i} is not object or category\")\n                conv = self.dic_cat[i].transform(dataframe(data_x[i])).astype(\"int\")\n                col_list = []\n                for j in self.dic_cat[i].categories_[0]:\n                    if i in self.label_preset.keys():\n                        for k, v in self.label_preset[i].items():\n                            if v == j: col_list.append(str(i) + \"_\" + str(k))\n                    else:\n                        col_list.append(str(i) + \"_\" + str(j))\n                conv = dataframe(conv, columns=col_list)\n                tmp_x = pd.concat([tmp_x, conv], axis=1)\n        return tmp_x\n    def clear(self):\n        self.dic_cat = {}\n        self.label_preset = {}\nclass MyKNNImputer:\n    def __init__(self, k=5):\n        self.imputer = KNNImputer(n_neighbors=k)\n        self.dic_cat = {}\n    def fit_transform(self, x, cat_vars=None):\n        if cat_vars is None:\n            x_imp = dataframe(self.imputer.fit_transform(x), columns=x.columns)\n        else:\n            naIdx = dict.fromkeys(cat_vars)\n            for i in cat_vars:\n                self.dic_cat[i] = diff(list(sorted(set(x[i]))), [nan])\n                naIdx[i] = list(which(array(x[i].isna())))\n            x_imp = dataframe(self.imputer.fit_transform(x), columns=x.columns)\n\n            # if imputed categorical value are not in the range, adjust the value\n            for i in cat_vars:\n                x_imp[i] = x_imp[i].apply(lambda x: int(round(x, 0)))\n                for j in naIdx[i]:\n                    if x_imp[i][j] not in self.dic_cat[i]:\n                        if x_imp[i][j] < self.dic_cat[i][0]:\n                            x_imp[i][naIdx[i]] = self.dic_cat[i][0]\n                        elif x_imp[i][j] > self.dic_cat[i][0]:\n                            x_imp[i][naIdx[i]] = self.dic_cat[i][len(self.dic_cat[i]) - 1]\n        return x_imp\n    def transform(self, x):\n        if len(self.dic_cat.keys()) == 0:\n            x_imp = dataframe(self.imputer.transform(x), columns=x.columns)\n        else:\n            naIdx = dict.fromkeys(self.dic_cat.keys())\n            for i in self.dic_cat.keys():\n                naIdx[i] = list(which(array(x[i].isna())))\n            x_imp = dataframe(self.imputer.transform(x), columns=x.columns)\n\n            # if imputed categorical value are not in the range, adjust the value\n            for i in self.dic_cat.keys():\n                x_imp[i] = x_imp[i].apply(lambda x: int(round(x, 0)))\n                for j in naIdx[i]:\n                    if x_imp[i][j] not in self.dic_cat[i]:\n                        if x_imp[i][j] < self.dic_cat[i][0]:\n                            x_imp[i][naIdx[i]] = self.dic_cat[i][0]\n                        elif x_imp[i][j] > self.dic_cat[i][0]:\n                            x_imp[i][naIdx[i]] = self.dic_cat[i][len(self.dic_cat[i]) - 1]\n        return x_imp\n    def clear(self):\n        self.imputer = None\n        self.dic_cat = {}\ndef remove_outlier(df, std=3, mode=\"remove\"):\n    tmp_df = df.copy()\n    if mode == \"remove\":\n        outlier_mask = (np.abs(stats.zscore(tmp_df)) > std).all(axis=1)\n        print(\"found outlier :\", outlier_mask.sum())\n        tmp_df = tmp_df[~outlier_mask]\n    elif mode == \"interpolate\":\n        tmp_outlier = []\n        for i in tmp_df:\n            outlier_mask = (np.abs(stats.zscore(tmp_df[i])) > std)\n            tmp_outlier.append(outlier_mask.sum())\n            if tmp_outlier[-1] == 0:\n                continue\n            tmp_df[i][outlier_mask] = np.nan\n            tmp_df[i] = tmp_df[i].interpolate(method='linear').bfill()\n        print(\"found outlier :\", np.sum(outlier_mask))\n    return tmp_df\n\nseed_everything()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-13T03:41:44.996512Z","iopub.execute_input":"2022-07-13T03:41:44.996734Z","iopub.status.idle":"2022-07-13T03:41:45.308759Z","shell.execute_reply.started":"2022-07-13T03:41:44.996701Z","shell.execute_reply":"2022-07-13T03:41:45.307990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Data #","metadata":{}},{"cell_type":"code","source":"# define function reading the train data\ndef read_notebook(path):\n    return (\n        pd.read_json(\n            path,\n            dtype={'cell_type': 'category', 'source': 'str'})\n        .assign(id=path.stem)\n        .rename_axis('cell_id')\n    )\ndef update_rawdata_dic(x, proc_id):\n    rawdata = [read_notebook(path) for path in x]\n    tmp_dic.update({proc_id: rawdata})\n    print(\"job finished :\", proc_id, \"\\n\")","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:41:45.313170Z","iopub.execute_input":"2022-07-13T03:41:45.315144Z","iopub.status.idle":"2022-07-13T03:41:45.323214Z","shell.execute_reply.started":"2022-07-13T03:41:45.315104Z","shell.execute_reply":"2022-07-13T03:41:45.322577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"paths_test = list((data_dir / 'test').glob('*.json'))\nnotebooks_test = [\n    read_notebook(path) for path in tqdm(paths_test, desc='Test NBs')\n]","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:41:45.329731Z","iopub.execute_input":"2022-07-13T03:41:45.330459Z","iopub.status.idle":"2022-07-13T03:41:45.396208Z","shell.execute_reply.started":"2022-07-13T03:41:45.330410Z","shell.execute_reply":"2022-07-13T03:41:45.395517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = (\n    pd.concat(notebooks_test)\n    .set_index('id', append=True)\n    .swaplevel()\n    .sort_index(level='id', sort_remaining=False)\n)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:41:45.397586Z","iopub.execute_input":"2022-07-13T03:41:45.398023Z","iopub.status.idle":"2022-07-13T03:41:45.414433Z","shell.execute_reply.started":"2022-07-13T03:41:45.397987Z","shell.execute_reply":"2022-07-13T03:41:45.413789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Ordering the cells & Labeling the target value","metadata":{}},{"cell_type":"code","source":"def get_ranks(orderd, unordered, normalize=False, score_by_position=True):\n    if score_by_position:\n        scores = pd.Series([orderd.index(d) for d in unordered]) + 1\n        return (scores / scores.max()).tolist() if normalize else scores.tolist()\n    else:\n        scores = pd.Series([len(orderd)-1 - orderd.index(d) for d in unordered]) + 1\n        return (scores / scores.max()).tolist() if normalize else scores.tolist()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:41:45.415894Z","iopub.execute_input":"2022-07-13T03:41:45.416161Z","iopub.status.idle":"2022-07-13T03:41:45.422639Z","shell.execute_reply.started":"2022-07-13T03:41:45.416124Z","shell.execute_reply":"2022-07-13T03:41:45.421929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def restruct_rank_values(x):\n    input_df = x.copy()\n    for i in input_df.index.get_level_values(0).unique():\n        tmp_df = input_df.loc[i].copy()\n        tmp_md = []\n        last_code_rank_value = 0\n        interval_rank_value = tmp_df.loc[(tmp_df[\"cell_type\"] == \"code\"), \"rank\"].diff().min()\n        interval_rank_value = 10.0 if np.isnan(interval_rank_value) else interval_rank_value\n        for idx, value in enumerate(tmp_df[\"cell_type\"]):\n            if value == \"markdown\":\n                tmp_md.append(idx)\n            elif value == \"code\":\n                if len(tmp_md) > 0:\n                    tmp_df[\"rank\"].iloc[tmp_md] = np.linspace(last_code_rank_value, tmp_df[\"rank\"].iloc[idx], len(tmp_md)+2)[1:-1]\n                    tmp_md = []\n                else:\n                    pass\n                last_code_rank_value = tmp_df[\"rank\"].iloc[idx]\n        # if markdown is last cell\n        if len(tmp_md) > 0:\n            for idx, value in enumerate(tmp_md):    \n                tmp_df[\"rank\"].iloc[value] = last_code_rank_value + (interval_rank_value * (idx + 1))\n            tmp_md = []\n        input_df.loc[i] = tmp_df.values\n    return input_df","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:41:45.424170Z","iopub.execute_input":"2022-07-13T03:41:45.424741Z","iopub.status.idle":"2022-07-13T03:41:45.436800Z","shell.execute_reply.started":"2022-07-13T03:41:45.424705Z","shell.execute_reply":"2022-07-13T03:41:45.436027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test[\"rank\"] = df_test.groupby([\"id\", \"cell_type\"]).cumcount()\ndf_test[\"rank\"] = df_test.groupby([\"id\", \"cell_type\"]).apply(lambda x: x[\"rank\"][::-1]).values\ndf_test[\"rank\"] = df_test.groupby([\"id\", \"cell_type\"])[\"rank\"].rank(pct=True) * 100","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:41:45.438495Z","iopub.execute_input":"2022-07-13T03:41:45.439293Z","iopub.status.idle":"2022-07-13T03:41:45.470607Z","shell.execute_reply.started":"2022-07-13T03:41:45.439245Z","shell.execute_reply":"2022-07-13T03:41:45.469938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.loc[(df_test[\"cell_type\"] == \"markdown\"), \"rank\"] = 0","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:41:45.471857Z","iopub.execute_input":"2022-07-13T03:41:45.472346Z","iopub.status.idle":"2022-07-13T03:41:45.477790Z","shell.execute_reply.started":"2022-07-13T03:41:45.472302Z","shell.execute_reply":"2022-07-13T03:41:45.477120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df_test.head(20)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:41:45.481015Z","iopub.execute_input":"2022-07-13T03:41:45.481405Z","iopub.status.idle":"2022-07-13T03:41:45.487308Z","shell.execute_reply.started":"2022-07-13T03:41:45.481374Z","shell.execute_reply":"2022-07-13T03:41:45.486430Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preprocessing & Text Cleansing","metadata":{}},{"cell_type":"code","source":"df_test[\"cell_type\"] = df_test[\"cell_type\"].apply(lambda x: 1.0 if x == \"markdown\" else 0.0)\ndf_test[\"cell_type\"] = df_test[\"cell_type\"].astype(\"float32\")\ndf_test[\"rank\"] = df_test[\"rank\"].astype(\"float32\")","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:41:45.488651Z","iopub.execute_input":"2022-07-13T03:41:45.489158Z","iopub.status.idle":"2022-07-13T03:41:45.500098Z","shell.execute_reply.started":"2022-07-13T03:41:45.489122Z","shell.execute_reply":"2022-07-13T03:41:45.499404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def str_cleansing(x):\n    preproc_lines = []\n    for line in x.split(\"\\n\"):\n        line = line.replace('\\n','')\n        line = line.replace('\\t',' ')\n        \n        # remove alias names\n        tmp = []\n        for idx, value in enumerate(line.split()):\n            if value == \"as\":\n                tmp.append(idx)\n                try:\n                    tmp.append(idx + 1)\n                except:\n                    pass\n        line = \" \".join(series(line.split()).iloc[diff(list(range(len(line.split()))), tmp)].tolist())\n\n        # replace 'import numpy as np' -> 'import numpy' to be easily tokenized\n        # but, below codes are not much effective for codebert tokenizer (codebert seems not to be able to interpret 'import ~') \n        tmp = []\n        skip_flag = True\n        for idx, value in enumerate(line.split()):\n            if skip_flag:\n                if value == \"import\":\n                    try:\n                        tmp.append(\"import \" + line.split()[idx+1])\n                        skip_flag=False\n                    except:\n                        tmp.append(\"import \")\n                elif value == \"from\":\n                    try:\n                        tmp.append(\"from \" + line.split()[idx+1])\n                        skip_flag=False\n                    except:\n                        tmp.append(\"from \") \n                else:\n                    tmp.append(value)\n            else:\n                skip_flag = True\n                continue\n        line = \" \".join(tmp)\n        \n        preproc_lines.append(line)\n        text_finder = re.compile('[^A-Za-z]')\n        preproc_lines[-1] = \" \".join(text_finder.sub(' ', preproc_lines[-1]).split()).lower()\n    preprocessed_script = ' '.join(preproc_lines)\n    preprocessed_script = ' '.join(preprocessed_script.split())\n    return preprocessed_script","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:41:45.502864Z","iopub.execute_input":"2022-07-13T03:41:45.504703Z","iopub.status.idle":"2022-07-13T03:41:45.516433Z","shell.execute_reply.started":"2022-07-13T03:41:45.504674Z","shell.execute_reply":"2022-07-13T03:41:45.515738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test[\"source\"] = df_test[\"source\"].apply(lambda x: str_cleansing(x))","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:41:45.517468Z","iopub.execute_input":"2022-07-13T03:41:45.518038Z","iopub.status.idle":"2022-07-13T03:41:45.763888Z","shell.execute_reply.started":"2022-07-13T03:41:45.518001Z","shell.execute_reply":"2022-07-13T03:41:45.763141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:41:45.765109Z","iopub.execute_input":"2022-07-13T03:41:45.765382Z","iopub.status.idle":"2022-07-13T03:41:45.786529Z","shell.execute_reply.started":"2022-07-13T03:41:45.765347Z","shell.execute_reply":"2022-07-13T03:41:45.785713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from transformers import AutoTokenizer, AutoModel\nimport torch\nfrom torch.utils.data import DataLoader, TensorDataset\ntorch_device = torch.device('cuda') if torch.cuda.is_available() else None\n\ncodebert_tokenizer_path = \"../input/ai4code-ms-codebert/microsoft-codebert-base/tokenizer/\"\ncodebert_model_path = \"../input/ai4code-ms-codebert/microsoft-codebert-base/model/\"\n\nbert_tokenizer_path = \"../input/ai4code-bert/bert-base-uncased/tokenizer/\"\nbert_model_path = \"../input/ai4code-bert/bert-base-uncased/model/\"\n\ndistilbert_tokenizer_path = \"../input/ai4code-distilbert/distilbert-base-uncased/tokenizer/\"\ndistilbert_model_path = \"../input/ai4code-distilbert/distilbert-base-uncased/model/\"\n\ntokenizer = AutoTokenizer.from_pretrained(distilbert_tokenizer_path)\nif torch.cuda.is_available():\n    model = AutoModel.from_pretrained(distilbert_model_path).to(torch_device)\nelse:\n    model = AutoModel.from_pretrained(distilbert_model_path)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:41:45.787825Z","iopub.execute_input":"2022-07-13T03:41:45.788068Z","iopub.status.idle":"2022-07-13T03:42:04.571097Z","shell.execute_reply.started":"2022-07-13T03:41:45.788036Z","shell.execute_reply":"2022-07-13T03:42:04.570302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"MAX_LEN = 64\ntokenizer_output = {\n    \"input_ids\": np.empty(shape=(df_test[(df_test[\"cell_type\"] == 1.0)].shape[0], MAX_LEN), dtype=\"int64\"),\n    \"attention_mask\": np.empty(shape=(df_test[(df_test[\"cell_type\"] == 1.0)].shape[0], MAX_LEN), dtype=\"int64\")\n}\n\nfor idx, value in enumerate(df_test.loc[(df_test[\"cell_type\"] == 1.0), \"source\"].to_list()):\n    tokens = tokenizer(\n        value,\n        max_length=MAX_LEN,\n        padding=\"max_length\",\n        truncation=True,\n        return_token_type_ids=False,\n    )\n    tokenizer_output[\"input_ids\"][idx] = tokens[\"input_ids\"]\n    tokenizer_output[\"attention_mask\"][idx] = tokens[\"attention_mask\"]","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:42:04.575708Z","iopub.execute_input":"2022-07-13T03:42:04.578142Z","iopub.status.idle":"2022-07-13T03:42:04.639015Z","shell.execute_reply.started":"2022-07-13T03:42:04.578102Z","shell.execute_reply":"2022-07-13T03:42:04.638339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_size = 32\ndf_test_fv = []\n\nif torch.cuda.is_available():\n    loader = DataLoader(\n        TensorDataset(torch.from_numpy(tokenizer_output[\"input_ids\"]).to(torch_device),\n                      torch.from_numpy(tokenizer_output[\"attention_mask\"]).to(torch_device)),\n        batch_size=batch_size,\n    )\n    for batch, (input_ids, attention_mask) in enumerate(loader):\n        outputs = model(input_ids=input_ids, attention_mask=attention_mask)\n        df_test_fv.append(outputs.last_hidden_state.detach().cpu().numpy().mean(axis=1))\nelse:\n    loader = DataLoader(\n        TensorDataset(torch.from_numpy(tokenizer_output[\"input_ids\"]),\n                      torch.from_numpy(tokenizer_output[\"attention_mask\"])),\n        batch_size=batch_size,\n    )\n    for batch, (input_ids, attention_mask) in enumerate(loader):\n        outputs = model(input_ids=input_ids, attention_mask=attention_mask)\n        df_test_fv.append(outputs.last_hidden_state.detach().cpu().numpy().mean(axis=1))\n        \ndf_test_fv = np.concatenate(df_test_fv, axis=0)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:42:04.640355Z","iopub.execute_input":"2022-07-13T03:42:04.640808Z","iopub.status.idle":"2022-07-13T03:42:05.886980Z","shell.execute_reply.started":"2022-07-13T03:42:04.640766Z","shell.execute_reply":"2022-07-13T03:42:05.886283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test_fv.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:42:05.888214Z","iopub.execute_input":"2022-07-13T03:42:05.888484Z","iopub.status.idle":"2022-07-13T03:42:05.895017Z","shell.execute_reply.started":"2022-07-13T03:42:05.888451Z","shell.execute_reply":"2022-07-13T03:42:05.894230Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Inference","metadata":{}},{"cell_type":"code","source":"def convert_rank_ids(x):\n    # Convert the cell_id index into a column.\n    x = x.reset_index('cell_id')\n    # Group the cell_ids for each notebook into a list.\n    x = x.groupby('id')['cell_id'].apply(list)\n    return x","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:42:05.896604Z","iopub.execute_input":"2022-07-13T03:42:05.896875Z","iopub.status.idle":"2022-07-13T03:42:05.904447Z","shell.execute_reply.started":"2022-07-13T03:42:05.896840Z","shell.execute_reply":"2022-07-13T03:42:05.903676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy.special import boxcox1p\nfrom scipy.special import inv_boxcox1p\nboxcox_lambda = 0.75","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:42:05.905163Z","iopub.execute_input":"2022-07-13T03:42:05.905357Z","iopub.status.idle":"2022-07-13T03:42:05.915212Z","shell.execute_reply.started":"2022-07-13T03:42:05.905334Z","shell.execute_reply":"2022-07-13T03:42:05.914456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_x = df_test_fv[:]\ntest_y = None\ntest_groups = df_test[df_test[\"cell_type\"] == 1.0].index.get_level_values(0).copy()\n\ntest_pool = cat.Pool(\n    data=test_x,\n    label=test_y,\n    group_id=test_groups.to_list()\n)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:42:05.916621Z","iopub.execute_input":"2022-07-13T03:42:05.916940Z","iopub.status.idle":"2022-07-13T03:42:05.954239Z","shell.execute_reply.started":"2022-07-13T03:42:05.916902Z","shell.execute_reply":"2022-07-13T03:42:05.953457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fold_metrics = []\n\nn_folds = 5\n\nstart_time_training = time()\n# fold training\nfor fold in range(n_folds):\n    print(\"\\n===== Fold\", fold, \"=====\\n\")\n    start_mem = memory_usage()   \n    \n    model = cat.CatBoostRanker().load_model(\"../input/catboost-ranker-training/models/model_fold_\" + str(fold) + \".cbm\")\n    print(model.get_params())\n    \n    #     df_test.loc[df_test[\"cell_type\"] == 1.0, \"rank\"] += inv_boxcox1p(model.predict(test_pool)[df_test[\"cell_type\"] == 1.0], boxcox_lambda) / n_folds\n    #     print(model.predict(test_pool)[:5])\n#     df_test.loc[df_test[\"cell_type\"] == 1.0, \"rank\"] += inv_boxcox1p(model.predict(test_pool), boxcox_lambda) / n_folds\n    df_test.loc[df_test[\"cell_type\"] == 1.0, \"rank\"] += model.predict(test_pool) / n_folds\n    \n    gc.collect()\n    end_mem = memory_usage()\n    print(\"@Memory leaked :\", end_mem - start_mem, \"\\n\")\nend_time_training = time()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:42:05.955774Z","iopub.execute_input":"2022-07-13T03:42:05.956017Z","iopub.status.idle":"2022-07-13T03:42:08.000050Z","shell.execute_reply.started":"2022-07-13T03:42:05.955985Z","shell.execute_reply":"2022-07-13T03:42:07.998912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(12, 6))\ngraph = sns.histplot(df_test.loc[df_test[\"cell_type\"] == 1.0, \"rank\"], bins=50, color=\"orange\")\nplt.title(\"Inferenced distribution of markdown cells' rank (code cell is scaled 0 ~ 100)\", fontsize=15, fontweight=\"bold\", pad=15)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:42:08.001491Z","iopub.execute_input":"2022-07-13T03:42:08.001738Z","iopub.status.idle":"2022-07-13T03:42:08.327194Z","shell.execute_reply.started":"2022-07-13T03:42:08.001704Z","shell.execute_reply":"2022-07-13T03:42:08.326464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"df_test = df_test.sort_values([\"id\", \"rank\"], ascending=[True, False])\ndf_test.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:42:08.328806Z","iopub.execute_input":"2022-07-13T03:42:08.329086Z","iopub.status.idle":"2022-07-13T03:42:08.346217Z","shell.execute_reply.started":"2022-07-13T03:42:08.329046Z","shell.execute_reply":"2022-07-13T03:42:08.345598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_submit = (\n    convert_rank_ids(df_test)\n    .apply(' '.join)  # list of ids -> string of ids\n    .rename_axis('id')\n    .rename('cell_order')\n)\ny_submit.iloc[0]","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:42:08.347502Z","iopub.execute_input":"2022-07-13T03:42:08.347750Z","iopub.status.idle":"2022-07-13T03:42:08.357043Z","shell.execute_reply.started":"2022-07-13T03:42:08.347717Z","shell.execute_reply":"2022-07-13T03:42:08.355876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_submit.to_csv('submission.csv')","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-13T03:42:08.358440Z","iopub.execute_input":"2022-07-13T03:42:08.358816Z","iopub.status.idle":"2022-07-13T03:42:08.369945Z","shell.execute_reply.started":"2022-07-13T03:42:08.358780Z","shell.execute_reply":"2022-07-13T03:42:08.369095Z"},"trusted":true},"execution_count":null,"outputs":[]}]}