{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Introduction #\n\n* **[1/3] [Data Preparation][1] ← (this kernel)**\n* [2/3] [CatBoost Ranker Training][2]\n* [3/3] [CatBoost Ranker Inference][3]\n\n### About Solution\n\n- Feature data\n    - **Markdown string (64 tokens) & No code string**\n    - From markdown string to Distilbert feature vector (768 vector)\n- Target data\n    - Set code cell position target value\n    - Set markdown cell posision target value with increasing the target value by linearly\n    - **So markdown cell's target values can be over 100 (Normalizing is not applied)**\n    - This way can be helpful for calculating unbiased target value on markdown cell\n    - Since this way doesn't refer to exact position of markdown cell\n- Model and hyperparameters\n    - Model : CatBoost Ranker\n    - Loss : RMSE (by groups)\n    - Hyperparameters : Tuning with Optuna\n    - **Log transformation on markdown cell position target value (for normal distribution on the target value)**\n   \nI refer to the form of **[NICK KUZMENKOV][4]'s kernels** which are listed below :)<br>\n* [1/3] [Data Preparation][1]\n* [2/3] [TPU Training][2] (~4 hours)\n* [3/3] [GPU Inference][3] (~2 hours)\n\n[1]: https://www.kaggle.com/cafelatte1/catboost-ranker-data-preparation\n[2]: https://www.kaggle.com/cafelatte1/catboost-ranker-training\n[3]: https://www.kaggle.com/cafelatte1/catboost-ranker-inference\n[4]: https://www.kaggle.com/nickuzmenkov/ai4code-tf-tpu-codebert-data-preparation/notebook\n[5]: https://www.kaggle.com/nickuzmenkov/ai4code-tf-tpu-codebert-training\n[6]: https://www.kaggle.com/nickuzmenkov/ai4code-tf-tpu-codebert-inference\n[7]: https://www.kaggle.com/nickuzmenkov","metadata":{}},{"cell_type":"markdown","source":"# Setup #","metadata":{}},{"cell_type":"code","source":"# import sys\n# !cp ../input/rapids/rapids.21.06 /opt/conda/envs/rapids.tar.gz\n# !cd /opt/conda/envs/ && tar -xzvf rapids.tar.gz > /dev/null\n# sys.path = [\"/opt/conda/envs/rapids/lib/python3.7/site-packages\"] + sys.path\n# sys.path = [\"/opt/conda/envs/rapids/lib/python3.7\"] + sys.path\n# sys.path = [\"/opt/conda/envs/rapids/lib\"] + sys.path \n# !cp /opt/conda/envs/rapids/lib/libxgboost.so /opt/conda/lib/","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2022-07-06T06:41:07.294492Z","iopub.execute_input":"2022-07-06T06:41:07.294784Z","iopub.status.idle":"2022-07-06T06:41:07.299249Z","shell.execute_reply.started":"2022-07-06T06:41:07.294751Z","shell.execute_reply":"2022-07-06T06:41:07.298168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport sys\nimport shutil\nfrom glob import glob\nimport multiprocessing as mp\nimport gc\nfrom pathlib import Path\nfrom scipy import stats\nfrom scipy.special import boxcox, softmax\nfrom scipy import sparse\n\nfrom multiprocessing import cpu_count\nimport copy\nimport pickle\nimport warnings\nfrom datetime import datetime, timedelta\nfrom time import time, sleep, mktime\nfrom matplotlib import font_manager as fm, rc, rcParams\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom tqdm import tqdm\nimport re\nimport random as rnd\nimport psutil\nfrom optuna import Trial, create_study\nfrom optuna.samplers import TPESampler\n\nimport numpy as np\nfrom numpy import array, nan, random as np_rnd, where\nimport pandas as pd\nfrom pandas import DataFrame as dataframe, Series as series, isna, read_csv\nfrom pandas.tseries.offsets import DateOffset\n\nfrom sklearn.model_selection import train_test_split as tts, StratifiedKFold, StratifiedShuffleSplit\nfrom sklearn.preprocessing import LabelEncoder, OneHotEncoder, StandardScaler, MinMaxScaler, RobustScaler, KBinsDiscretizer\nfrom sklearn import metrics\nfrom sklearn.compose import ColumnTransformer\n# config_missingpy(); from missingpy import MissForest\nfrom sklearn.impute import KNNImputer\nfrom optuna import Trial, create_study\nfrom sklearn.model_selection import GroupKFold, GroupShuffleSplit, StratifiedGroupKFold\nfrom sklearn.feature_extraction.text import TfidfVectorizer\n\ntry:\n    import cudf as cd\n    import cupy as cp\n    from cuml.cluster import KMeans\n    from cuml.neighbors import NearestNeighbors\n    from cuml.metrics.cluster import silhouette_score\nexcept:\n    print(\"RAPIDS Import ERROR\")\n\nimport catboost as cat\n\n# # ===== tensorflow =====\n# import tensorflow as tf\n# from tensorflow import random as tf_rnd\n# from tensorflow.keras.models import Model\n# from tensorflow.keras.models import Sequential\n# from tensorflow.keras import layers\n# from tensorflow.keras import activations\n# from tensorflow.keras import optimizers\n# from tensorflow.keras import metrics as tf_metrics\n# from tensorflow.keras import callbacks as tf_callbacks\n# from tqdm.keras import TqdmCallback\n# import tensorflow_addons as tfa\n# from tensorflow.keras.utils import plot_model\n# from keras.utils.layer_utils import count_params\n\n# import keras_tuner as kt\n# from keras_tuner import HyperModel\n# import tensorflow_hub as tf_hub\n# import tensorflow_recommenders as tfrs\n\n# # GPU check\n# if tf.test.gpu_device_name() != '/device:GPU:0':\n#     print('GPU device not found')\n# else:\n#     print('Found GPU')\n\n# # GPU memory setting\n# gpus = tf.config.list_physical_devices('GPU')\n# if gpus:\n#   try:\n#     tf.config.experimental.set_memory_growth(gpus[0], True)\n#   except RuntimeError as e:\n#     print(e)\n\nwarnings.filterwarnings(action='ignore')\nrcParams['axes.unicode_minus'] = False\npd.set_option('display.max_columns', 100)\npd.set_option('display.max_rows', 100)\npd.set_option('display.width', 1000)\npd.set_option('max_colwidth', 200)\n# plt.rc('font', family='NanumSquareB')\n\ndata_dir = Path('../input/AI4Code')","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2022-07-06T06:41:07.465358Z","iopub.execute_input":"2022-07-06T06:41:07.465747Z","iopub.status.idle":"2022-07-06T06:41:07.483312Z","shell.execute_reply.started":"2022-07-06T06:41:07.465714Z","shell.execute_reply":"2022-07-06T06:41:07.482225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ===== utility functions =====\n# label encoding for categorical column with excepting na value\ndef seed_everything(seed=42):\n    # python random module\n    rnd.seed(seed)\n    # numpy random\n    np_rnd.seed(seed)\n#     # tf random\n#     tf_rnd.set_seed(seed)\n    # RAPIDS random\n    try:\n        cp.random.seed(seed)\n    except:\n        pass\ndef which(bool_list):\n    return where(bool_list)[0]\ndef easyIO(x=None, path=None, op=\"r\"):\n    tmp = None\n    if op == \"r\":\n        with open(path, \"rb\") as f:\n            tmp = pickle.load(f)\n        return tmp\n    elif op == \"w\":\n        with open(path, \"wb\") as f:\n            pickle.dump(x, f)\n    else:\n        print(\"Unknown operation type\")\ndef diff(first, second):\n    second = set(second)\n    return [item for item in first if item not in second]\ndef findIdx(data_x, col_names):\n    return [int(i) for i, j in enumerate(data_x) if j in col_names]\ndef orderElems(for_order, using_ref):\n    return [i for i in using_ref if i in for_order]\n# concatenate by row\ndef cbr(df1, df2):\n    if type(df1) == series:\n        tmp_concat = series(pd.concat([dataframe(df1), dataframe(df2)], axis=0, ignore_index=True).iloc[:,0])\n        tmp_concat.reset_index(drop=True, inplace=True)\n    elif type(df1) == dataframe:\n        tmp_concat = pd.concat([df1, df2], axis=0, ignore_index=True)\n        tmp_concat.reset_index(drop=True, inplace=True)\n    elif type(df1) == np.ndarray:\n        tmp_concat = np.concatenate([df1, df2], axis=0)\n    else:\n        print(\"Unknown Type: return 1st argument\")\n        tmp_concat = df1\n    return tmp_concat\ndef change_width(ax, new_value):\n    for patch in ax.patches :\n        current_width = patch.get_width()\n        adj_value = current_width - new_value\n        # we change the bar width\n        patch.set_width(new_value)\n        # we recenter the bar\n        patch.set_x(patch.get_x() + adj_value * .5)\ndef week_of_month(date):\n    month = date.month\n    week = 0\n    while date.month == month:\n        week += 1\n        date -= timedelta(days=7)\n    return week\ndef getSeason(date):\n    month = date.month\n    if month in [3, 4, 5]:\n        return \"Spring\"\n    elif month in [6, 7, 8]:\n        return \"Summer\"\n    elif month in [9, 10, 11]:\n        return \"Fall\"\n    else:\n        return \"Winter\"\ndef createFolder(directory):\n    try:\n        if not os.path.exists(directory):\n            os.makedirs(directory)\n    except OSError:\n        print('Error: Creating directory. ' + directory)\n# def softmax(x):\n#     max = np.max(x, axis=1, keepdims=True)  # returns max of each row and keeps same dims\n#     e_x = np.exp(x - max)  # subtracts each row with its max value\n#     sum = np.sum(e_x, axis=1, keepdims=True)  # returns sum of each row and keeps same dims\n#     f_x = e_x / sum\n#     return f_x\ndef sigmoid(x):\n    return 1/(1 + np.exp(-x))\ndef dispPerformance(result_dic):\n    perf_table = dataframe()\n    index_names = []\n    for k, v in result_dic.items():\n        index_names.append(k)\n        perf_table = pd.concat([perf_table, series(v[\"performance\"]).to_frame().T], ignore_index=True, axis=0)\n    perf_table.index = index_names\n    perf_table.sort_values(perf_table.columns[0], inplace=True)\n    print(perf_table)\n    return perf_table\ndef powspace(start, stop, power, num):\n    start = np.power(start, 1/float(power))\n    stop = np.power(stop, 1/float(power))\n    return np.power(np.linspace(start, stop, num=num), power)\ndef xgb_custom_lossfunction(alpha = 1):\n    def support_under_mse(label, pred):\n        # grad : 1차 미분\n        # hess : 2차 미분\n        residual = (label - pred).astype(\"float\")\n        grad = np.where(residual > 0, -2 * alpha * residual, -2 * residual)\n        hess = np.where(residual > 0, 2 * alpha, 2.0)\n        return grad, hess\n    return support_under_mse\ndef pd_flatten(df):\n    df = df.unstack()\n    df.index = [str(i) + \"_\" + str(j) for i, j in df.index]\n    return df\ndef tf_losses_rmse(y_true, y_pred, sample_weight=None):\n    return tf.sqrt(tf.reduce_mean((y_true - y_pred) ** 2)) if sample_weight is None else tf.sqrt(tf.reduce_mean(((y_true - y_pred) ** 2) * sample_weight))\ndef tf_loss_nmae(y_true, y_pred, sample_weight=False):\n    mae = tf.reduce_mean(tf.math.abs(y_true - y_pred))\n    score = tf.math.divide(mae, tf.reduce_mean(tf.math.abs(y_true)))\n    return score\ndef text_extractor(string, lang=\"eng\", spacing=True):\n    # # 괄호를 포함한 괄호 안 문자 제거 정규식\n    # re.sub(r'\\([^)]*\\)', '', remove_text)\n    # # <>를 포함한 <> 안 문자 제거 정규식\n    # re.sub(r'\\<[^)]*\\>', '', remove_text)\n    if lang == \"eng\":\n        text_finder = re.compile('[^ A-Za-z]') if spacing else re.compile('[^A-Za-z]')\n    elif lang == \"kor\":\n        text_finder = re.compile('[^ ㄱ-ㅣ가-힣+]') if spacing else re.compile('[^ㄱ-ㅣ가-힣+]')\n    # default : kor + eng\n    else:\n        text_finder = re.compile('[^ A-Za-zㄱ-ㅣ가-힣+]') if spacing else re.compile('[^A-Za-zㄱ-ㅣ가-힣+]')\n    return text_finder.sub('', string)\ndef memory_usage(message='debug'):\n    # current process RAM usage\n    p = psutil.Process()\n    rss = p.memory_info().rss / 2 ** 20 # Bytes to MB\n    print(f\"[{message}] memory usage: {rss: 10.3f} MB\")\n    return rss\nclass MyLabelEncoder:\n    def __init__(self, preset={}):\n        # dic_cat format -> {\"col_name\": {\"value\": replace}}\n        self.dic_cat = preset\n    def fit_transform(self, data_x, col_names):\n        tmp_x = copy.deepcopy(data_x)\n        for i in col_names:\n            # if key is not in dic, update dic\n            if i not in self.dic_cat.keys():\n                tmp_dic = dict.fromkeys(sorted(set(tmp_x[i]).difference([nan])))\n                label_cnt = 0\n                for j in tmp_dic.keys():\n                    tmp_dic[j] = label_cnt\n                    label_cnt += 1\n                self.dic_cat[i] = tmp_dic\n            # transform value which is not in dic to nan\n            tmp_x[i] = tmp_x[i].astype(\"object\")\n            conv = tmp_x[i].replace(self.dic_cat[i])\n            for conv_idx, j in enumerate(conv):\n                if j not in self.dic_cat[i].values():\n                    conv[conv_idx] = nan\n            # final return\n            tmp_x[i] = conv.astype(\"float\")\n        return tmp_x\n    def transform(self, data_x):\n        tmp_x = copy.deepcopy(data_x)\n        for i in self.dic_cat.keys():\n            # transform value which is not in dic to nan\n            tmp_x[i] = tmp_x[i].astype(\"object\")\n            conv = tmp_x[i].replace(self.dic_cat[i])\n            for conv_idx, j in enumerate(conv):\n                if j not in self.dic_cat[i].values():\n                    conv[conv_idx] = nan\n            # final return\n            tmp_x[i] = conv.astype(\"float\")\n        return tmp_x\n    def clear(self):\n        self.dic_cat = {}\nclass MyOneHotEncoder:\n    def __init__(self, label_preset={}):\n        self.dic_cat = {}\n        self.label_preset = label_preset\n    def fit_transform(self, data_x, col_names):\n        tmp_x = dataframe()\n        for i in data_x:\n            if i not in col_names:\n                tmp_x = pd.concat([tmp_x, dataframe(data_x[i])], axis=1)\n            else:\n                if not ((data_x[i].dtype.name == \"object\") or (data_x[i].dtype.name == \"category\")):\n                    print(F\"WARNING : {i} is not object or category\")\n                self.dic_cat[i] = OneHotEncoder(sparse=False, handle_unknown=\"ignore\")\n                conv = self.dic_cat[i].fit_transform(dataframe(data_x[i])).astype(\"int\")\n                col_list = []\n                for j in self.dic_cat[i].categories_[0]:\n                    if i in self.label_preset.keys():\n                        for k, v in self.label_preset[i].items():\n                            if v == j:\n                                col_list.append(str(i) + \"_\" + str(k))\n                    else:\n                        col_list.append(str(i) + \"_\" + str(j))\n                conv = dataframe(conv, columns=col_list)\n                tmp_x = pd.concat([tmp_x, conv], axis=1)\n        return tmp_x\n    def transform(self, data_x):\n        tmp_x = dataframe()\n        for i in data_x:\n            if not i in list(self.dic_cat.keys()):\n                tmp_x = pd.concat([tmp_x, dataframe(data_x[i])], axis=1)\n            else:\n                if not ((data_x[i].dtype.name == \"object\") or (data_x[i].dtype.name == \"category\")):\n                    print(F\"WARNING : {i} is not object or category\")\n                conv = self.dic_cat[i].transform(dataframe(data_x[i])).astype(\"int\")\n                col_list = []\n                for j in self.dic_cat[i].categories_[0]:\n                    if i in self.label_preset.keys():\n                        for k, v in self.label_preset[i].items():\n                            if v == j: col_list.append(str(i) + \"_\" + str(k))\n                    else:\n                        col_list.append(str(i) + \"_\" + str(j))\n                conv = dataframe(conv, columns=col_list)\n                tmp_x = pd.concat([tmp_x, conv], axis=1)\n        return tmp_x\n    def clear(self):\n        self.dic_cat = {}\n        self.label_preset = {}\nclass MyKNNImputer:\n    def __init__(self, k=5):\n        self.imputer = KNNImputer(n_neighbors=k)\n        self.dic_cat = {}\n    def fit_transform(self, x, cat_vars=None):\n        if cat_vars is None:\n            x_imp = dataframe(self.imputer.fit_transform(x), columns=x.columns)\n        else:\n            naIdx = dict.fromkeys(cat_vars)\n            for i in cat_vars:\n                self.dic_cat[i] = diff(list(sorted(set(x[i]))), [nan])\n                naIdx[i] = list(which(array(x[i].isna())))\n            x_imp = dataframe(self.imputer.fit_transform(x), columns=x.columns)\n\n            # if imputed categorical value are not in the range, adjust the value\n            for i in cat_vars:\n                x_imp[i] = x_imp[i].apply(lambda x: int(round(x, 0)))\n                for j in naIdx[i]:\n                    if x_imp[i][j] not in self.dic_cat[i]:\n                        if x_imp[i][j] < self.dic_cat[i][0]:\n                            x_imp[i][naIdx[i]] = self.dic_cat[i][0]\n                        elif x_imp[i][j] > self.dic_cat[i][0]:\n                            x_imp[i][naIdx[i]] = self.dic_cat[i][len(self.dic_cat[i]) - 1]\n        return x_imp\n    def transform(self, x):\n        if len(self.dic_cat.keys()) == 0:\n            x_imp = dataframe(self.imputer.transform(x), columns=x.columns)\n        else:\n            naIdx = dict.fromkeys(self.dic_cat.keys())\n            for i in self.dic_cat.keys():\n                naIdx[i] = list(which(array(x[i].isna())))\n            x_imp = dataframe(self.imputer.transform(x), columns=x.columns)\n\n            # if imputed categorical value are not in the range, adjust the value\n            for i in self.dic_cat.keys():\n                x_imp[i] = x_imp[i].apply(lambda x: int(round(x, 0)))\n                for j in naIdx[i]:\n                    if x_imp[i][j] not in self.dic_cat[i]:\n                        if x_imp[i][j] < self.dic_cat[i][0]:\n                            x_imp[i][naIdx[i]] = self.dic_cat[i][0]\n                        elif x_imp[i][j] > self.dic_cat[i][0]:\n                            x_imp[i][naIdx[i]] = self.dic_cat[i][len(self.dic_cat[i]) - 1]\n        return x_imp\n    def clear(self):\n        self.imputer = None\n        self.dic_cat = {}\ndef remove_outlier(df, std=3, mode=\"remove\"):\n    tmp_df = df.copy()\n    if mode == \"remove\":\n        outlier_mask = (np.abs(stats.zscore(tmp_df)) > std).all(axis=1)\n        print(\"found outlier :\", outlier_mask.sum())\n        tmp_df = tmp_df[~outlier_mask]\n    elif mode == \"interpolate\":\n        tmp_outlier = []\n        for i in tmp_df:\n            outlier_mask = (np.abs(stats.zscore(tmp_df[i])) > std)\n            tmp_outlier.append(outlier_mask.sum())\n            if tmp_outlier[-1] == 0:\n                continue\n            tmp_df[i][outlier_mask] = np.nan\n            tmp_df[i] = tmp_df[i].interpolate(method='linear').bfill()\n        print(\"found outlier :\", np.sum(outlier_mask))\n    return tmp_df\n\nseed_everything()","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2022-07-06T06:41:07.715373Z","iopub.execute_input":"2022-07-06T06:41:07.715773Z","iopub.status.idle":"2022-07-06T06:41:07.795201Z","shell.execute_reply.started":"2022-07-06T06:41:07.715737Z","shell.execute_reply":"2022-07-06T06:41:07.794033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Data #","metadata":{}},{"cell_type":"code","source":"# define function reading the train data\ndef read_notebook(path):\n    return (\n        pd.read_json(\n            path,\n            dtype={'cell_type': 'category', 'source': 'str'})\n        .assign(id=path.stem)\n        .rename_axis('cell_id')\n    )\ndef update_rawdata_dic(x, proc_id):\n    rawdata = [read_notebook(path) for path in x]\n    tmp_dic.update({proc_id: rawdata})\n    print(\"job finished :\", proc_id, \"\\n\")","metadata":{"execution":{"iopub.status.busy":"2022-07-06T06:41:07.798512Z","iopub.execute_input":"2022-07-06T06:41:07.799499Z","iopub.status.idle":"2022-07-06T06:41:07.811903Z","shell.execute_reply.started":"2022-07-06T06:41:07.799352Z","shell.execute_reply":"2022-07-06T06:41:07.811020Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# TRAIN_NB_LENGTH = 100\n# TRAIN_NB_LENGTH = 1000\nTRAIN_NB_LENGTH = 10000\n# TRAIN_NB_LENGTH = 25000\n\n# read one by one alogn with the notebook ids\n# random select train dataset\npaths_train = list((data_dir / 'train').glob('*.json'))\nnp_rnd.seed(1)\npaths_train = list(array(paths_train)[np_rnd.randint(0, len(paths_train)-1, size=TRAIN_NB_LENGTH, dtype=int)])\n\nstart_time = time()\nnotebooks_train = [\n    read_notebook(path) for path in tqdm(paths_train, desc='Train NBs')\n]\nprint(time() - start_time)","metadata":{"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-06T06:41:07.813608Z","iopub.execute_input":"2022-07-06T06:41:07.814810Z","iopub.status.idle":"2022-07-06T06:41:09.918400Z","shell.execute_reply.started":"2022-07-06T06:41:07.814648Z","shell.execute_reply":"2022-07-06T06:41:09.917392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# concatenate the data by row\ndf_full_raw = (\n    pd.concat(notebooks_train)\n    .set_index('id', append=True)\n    .swaplevel()\n    .sort_index(level='id', sort_remaining=False)\n)\nprint(df_full_raw)","metadata":{"execution":{"iopub.status.busy":"2022-07-06T06:41:09.920303Z","iopub.execute_input":"2022-07-06T06:41:09.921029Z","iopub.status.idle":"2022-07-06T06:41:09.994700Z","shell.execute_reply.started":"2022-07-06T06:41:09.920956Z","shell.execute_reply":"2022-07-06T06:41:09.993560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del notebooks_train, paths_train","metadata":{"execution":{"iopub.status.busy":"2022-07-06T06:41:09.997975Z","iopub.execute_input":"2022-07-06T06:41:09.998316Z","iopub.status.idle":"2022-07-06T06:41:10.004950Z","shell.execute_reply.started":"2022-07-06T06:41:09.998273Z","shell.execute_reply":"2022-07-06T06:41:10.003621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Ordering the cells & Labeling the target value","metadata":{}},{"cell_type":"code","source":"def get_ranks(orderd, unordered, normalize=False, score_by_position=True):\n    if score_by_position:\n        scores = pd.Series([orderd.index(d) for d in unordered]) + 1\n        return (scores / scores.max()).tolist() if normalize else scores.tolist()\n    else:\n        scores = pd.Series([len(orderd)-1 - orderd.index(d) for d in unordered]) + 1\n        return (scores / scores.max()).tolist() if normalize else scores.tolist()","metadata":{"execution":{"iopub.status.busy":"2022-07-06T06:41:10.007281Z","iopub.execute_input":"2022-07-06T06:41:10.007957Z","iopub.status.idle":"2022-07-06T06:41:10.018507Z","shell.execute_reply.started":"2022-07-06T06:41:10.007900Z","shell.execute_reply":"2022-07-06T06:41:10.017499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_orders = pd.read_csv(\n    data_dir / 'train_orders.csv',\n    index_col='id',\n    squeeze=True,\n).str.split()  # Split the string representation of cell_ids into a list\n\nprint(df_orders)","metadata":{"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-06T06:41:10.019655Z","iopub.execute_input":"2022-07-06T06:41:10.021455Z","iopub.status.idle":"2022-07-06T06:41:11.706101Z","shell.execute_reply.started":"2022-07-06T06:41:10.021410Z","shell.execute_reply":"2022-07-06T06:41:11.705027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# load the data including correct order\n# join the unordered raw train data to the correctly ordered data\ndf_ranks = df_orders.to_frame().join(\n    df_full_raw.reset_index('cell_id').groupby('id')['cell_id'].apply(list), how='right',\n)\n# columns order : NB_id, correct cell order, incorrect cell order\nprint(df_ranks.head())","metadata":{"execution":{"iopub.status.busy":"2022-07-06T06:41:11.707727Z","iopub.execute_input":"2022-07-06T06:41:11.708797Z","iopub.status.idle":"2022-07-06T06:41:11.830963Z","shell.execute_reply.started":"2022-07-06T06:41:11.708749Z","shell.execute_reply":"2022-07-06T06:41:11.830017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get the correct rank score and save to the dictionary\nranks = {}\nfor id_, cell_order, cell_id in df_ranks.itertuples():\n    ranks[id_] = {'cell_id': cell_id, 'rank': get_ranks(cell_order, cell_id, normalize=True, score_by_position=True)}","metadata":{"execution":{"iopub.status.busy":"2022-07-06T06:41:11.832447Z","iopub.execute_input":"2022-07-06T06:41:11.833344Z","iopub.status.idle":"2022-07-06T06:41:11.925701Z","shell.execute_reply.started":"2022-07-06T06:41:11.833279Z","shell.execute_reply":"2022-07-06T06:41:11.924462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_full = (\n    # create dataframe from the dictionary including rank score   \n    pd.DataFrame.from_dict(ranks, orient='index')\n    # rename the index\n    .rename_axis('id')\n    # flatten to the row from the list-typed correctly ordered cell_id object \n    .apply(pd.Series.explode)\n    # set 2nd index as cell_id\n    .set_index('cell_id', append=True)\n# join the unordered dataframe to the ordered dataframe\n).join(df_full_raw, on=[\"id\", \"cell_id\"], how=\"right\")\n\ndf_full = df_full.sort_values([\"id\", \"rank\"], ascending=[True, False])","metadata":{"execution":{"iopub.status.busy":"2022-07-06T06:41:11.928507Z","iopub.execute_input":"2022-07-06T06:41:11.929992Z","iopub.status.idle":"2022-07-06T06:41:11.981604Z","shell.execute_reply.started":"2022-07-06T06:41:11.929946Z","shell.execute_reply":"2022-07-06T06:41:11.980722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def restruct_rank_values(x):\n    input_df = x.copy()\n    for i in input_df.index.get_level_values(0).unique():\n        tmp_df = input_df.loc[i].copy()\n        tmp_md = []\n        last_code_rank_value = 0\n        interval_rank_value = tmp_df.loc[(tmp_df[\"cell_type\"] == \"code\"), \"rank\"].diff().min()\n        interval_rank_value = 10.0 if np.isnan(interval_rank_value) else interval_rank_value\n        for idx, value in enumerate(tmp_df[\"cell_type\"]):\n            if value == \"markdown\":\n                tmp_md.append(idx)\n            elif value == \"code\":\n                if len(tmp_md) > 0:\n                    tmp_df[\"rank\"].iloc[tmp_md] = np.linspace(last_code_rank_value, tmp_df[\"rank\"].iloc[idx], len(tmp_md)+2)[1:-1]\n                    tmp_md = []\n                else:\n                    pass\n                last_code_rank_value = tmp_df[\"rank\"].iloc[idx]\n        # if markdown is last cell\n        if len(tmp_md) > 0:\n            for idx, value in enumerate(tmp_md):    \n                tmp_df[\"rank\"].iloc[value] = last_code_rank_value + (interval_rank_value * (idx + 1))\n            tmp_md = []\n        input_df.loc[i] = tmp_df.values\n    return input_df","metadata":{"execution":{"iopub.status.busy":"2022-07-06T06:41:11.983066Z","iopub.execute_input":"2022-07-06T06:41:11.983459Z","iopub.status.idle":"2022-07-06T06:41:11.994814Z","shell.execute_reply.started":"2022-07-06T06:41:11.983416Z","shell.execute_reply":"2022-07-06T06:41:11.993656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_full[\"rank\"] = df_full.groupby([\"id\", \"cell_type\"]).cumcount()\ndf_full[\"rank\"] = df_full.groupby([\"id\", \"cell_type\"])[\"rank\"].rank(pct=True) * 100","metadata":{"execution":{"iopub.status.busy":"2022-07-06T06:41:11.999730Z","iopub.execute_input":"2022-07-06T06:41:12.000350Z","iopub.status.idle":"2022-07-06T06:41:12.017489Z","shell.execute_reply.started":"2022-07-06T06:41:12.000289Z","shell.execute_reply":"2022-07-06T06:41:12.016631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_full.loc[(df_full[\"cell_type\"] == \"markdown\"), \"rank\"] = 0","metadata":{"execution":{"iopub.status.busy":"2022-07-06T06:41:12.019127Z","iopub.execute_input":"2022-07-06T06:41:12.019468Z","iopub.status.idle":"2022-07-06T06:41:12.026603Z","shell.execute_reply.started":"2022-07-06T06:41:12.019430Z","shell.execute_reply":"2022-07-06T06:41:12.025446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get positional target value on markdown cell\ndf_full = restruct_rank_values(df_full)","metadata":{"execution":{"iopub.status.busy":"2022-07-06T06:41:12.028518Z","iopub.execute_input":"2022-07-06T06:41:12.029233Z","iopub.status.idle":"2022-07-06T06:41:13.190276Z","shell.execute_reply.started":"2022-07-06T06:41:12.029192Z","shell.execute_reply":"2022-07-06T06:41:13.189257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_full = df_full.sort_values([\"id\", \"rank\"], ascending=[True, False])","metadata":{"execution":{"iopub.status.busy":"2022-07-06T06:41:13.192152Z","iopub.execute_input":"2022-07-06T06:41:13.192477Z","iopub.status.idle":"2022-07-06T06:41:13.203620Z","shell.execute_reply.started":"2022-07-06T06:41:13.192437Z","shell.execute_reply":"2022-07-06T06:41:13.202391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_full.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-06T06:41:13.206771Z","iopub.execute_input":"2022-07-06T06:41:13.207482Z","iopub.status.idle":"2022-07-06T06:41:13.224295Z","shell.execute_reply.started":"2022-07-06T06:41:13.207399Z","shell.execute_reply":"2022-07-06T06:41:13.223125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_full.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-07-06T06:41:13.225706Z","iopub.execute_input":"2022-07-06T06:41:13.226093Z","iopub.status.idle":"2022-07-06T06:41:13.241933Z","shell.execute_reply.started":"2022-07-06T06:41:13.226051Z","shell.execute_reply":"2022-07-06T06:41:13.240848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_full.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-06T06:41:13.243222Z","iopub.execute_input":"2022-07-06T06:41:13.243674Z","iopub.status.idle":"2022-07-06T06:41:13.259373Z","shell.execute_reply.started":"2022-07-06T06:41:13.243629Z","shell.execute_reply":"2022-07-06T06:41:13.258360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df_full[\"rank\"].min())\nprint(df_full[\"rank\"].max())","metadata":{"execution":{"iopub.status.busy":"2022-07-06T06:41:13.261608Z","iopub.execute_input":"2022-07-06T06:41:13.261825Z","iopub.status.idle":"2022-07-06T06:41:13.271663Z","shell.execute_reply.started":"2022-07-06T06:41:13.261799Z","shell.execute_reply":"2022-07-06T06:41:13.270239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del df_ranks, df_full_raw, ranks","metadata":{"execution":{"iopub.status.busy":"2022-07-06T06:41:13.273448Z","iopub.execute_input":"2022-07-06T06:41:13.274379Z","iopub.status.idle":"2022-07-06T06:41:13.281593Z","shell.execute_reply.started":"2022-07-06T06:41:13.274334Z","shell.execute_reply":"2022-07-06T06:41:13.280552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preprocessing","metadata":{}},{"cell_type":"code","source":"df_full[\"cell_type\"] = df_full[\"cell_type\"].apply(lambda x: 1.0 if x == \"markdown\" else 0.0)","metadata":{"execution":{"iopub.status.busy":"2022-07-06T06:41:13.283200Z","iopub.execute_input":"2022-07-06T06:41:13.283959Z","iopub.status.idle":"2022-07-06T06:41:13.295250Z","shell.execute_reply.started":"2022-07-06T06:41:13.283783Z","shell.execute_reply":"2022-07-06T06:41:13.294156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_full.groupby(\"id\").size().describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-06T06:41:13.298080Z","iopub.execute_input":"2022-07-06T06:41:13.298406Z","iopub.status.idle":"2022-07-06T06:41:13.313663Z","shell.execute_reply.started":"2022-07-06T06:41:13.298308Z","shell.execute_reply":"2022-07-06T06:41:13.312312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# adjust the cell length\nMIN_CELL_LEN = 8\nMAX_CELL_LEN = 256\ndf_full = df_full.loc[df_full.groupby(\"id\").size().index[(df_full.groupby(\"id\").size() >= MIN_CELL_LEN) & (df_full.groupby(\"id\").size() <= MAX_CELL_LEN)]]\n\n# select the train data which has at least 1 cell for both code & markdown\ndf_full = df_full.loc[df_full.index.get_level_values(0).unique()[df_full.groupby(\"id\")[\"cell_type\"].apply(lambda x: True if ((x==0.0).sum() > 1) & ((x==1.0).sum() > 1) else False)]]","metadata":{"execution":{"iopub.status.busy":"2022-07-06T06:41:13.315169Z","iopub.execute_input":"2022-07-06T06:41:13.315500Z","iopub.status.idle":"2022-07-06T06:41:13.382655Z","shell.execute_reply.started":"2022-07-06T06:41:13.315460Z","shell.execute_reply":"2022-07-06T06:41:13.381800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_full.groupby(\"id\").size().describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-06T06:41:13.383885Z","iopub.execute_input":"2022-07-06T06:41:13.385474Z","iopub.status.idle":"2022-07-06T06:41:13.399264Z","shell.execute_reply.started":"2022-07-06T06:41:13.385430Z","shell.execute_reply":"2022-07-06T06:41:13.397996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_full[\"cell_type\"] = df_full[\"cell_type\"].astype(\"float32\")\ndf_full[\"rank\"] = df_full[\"rank\"].astype(\"float32\")","metadata":{"execution":{"iopub.status.busy":"2022-07-06T06:41:13.401031Z","iopub.execute_input":"2022-07-06T06:41:13.401487Z","iopub.status.idle":"2022-07-06T06:41:13.409500Z","shell.execute_reply.started":"2022-07-06T06:41:13.401443Z","shell.execute_reply":"2022-07-06T06:41:13.408076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Text Cleansing","metadata":{}},{"cell_type":"code","source":"def str_cleansing(x):\n    preproc_lines = []\n    for line in x.split(\"\\n\"):\n        line = line.replace('\\n','')\n        line = line.replace('\\t',' ')\n        \n        # remove alias names\n        tmp = []\n        for idx, value in enumerate(line.split()):\n            if value == \"as\":\n                tmp.append(idx)\n                try:\n                    tmp.append(idx + 1)\n                except:\n                    pass\n        line = \" \".join(series(line.split()).iloc[diff(list(range(len(line.split()))), tmp)].tolist())\n\n        # replace 'import numpy as np' -> 'import numpy' to be easily tokenized\n        # but, below codes are not much effective for codebert tokenizer (codebert seems not to be able to interpret 'import ~') \n        tmp = []\n        skip_flag = True\n        for idx, value in enumerate(line.split()):\n            if skip_flag:\n                if value == \"import\":\n                    try:\n                        tmp.append(\"import \" + line.split()[idx+1])\n                        skip_flag=False\n                    except:\n                        tmp.append(\"import \")\n                elif value == \"from\":\n                    try:\n                        tmp.append(\"from \" + line.split()[idx+1])\n                        skip_flag=False\n                    except:\n                        tmp.append(\"from \") \n                else:\n                    tmp.append(value)\n            else:\n                skip_flag = True\n                continue\n        line = \" \".join(tmp)\n        \n        preproc_lines.append(line)\n        text_finder = re.compile('[^A-Za-z]')\n        preproc_lines[-1] = \" \".join(text_finder.sub(' ', preproc_lines[-1]).split()).lower()\n    preprocessed_script = ' '.join(preproc_lines)\n    preprocessed_script = ' '.join(preprocessed_script.split())\n    return preprocessed_script","metadata":{"execution":{"iopub.status.busy":"2022-07-06T06:41:13.411070Z","iopub.execute_input":"2022-07-06T06:41:13.412800Z","iopub.status.idle":"2022-07-06T06:41:13.427766Z","shell.execute_reply.started":"2022-07-06T06:41:13.411578Z","shell.execute_reply":"2022-07-06T06:41:13.426692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df_full[\"source\"][df_full[\"cell_type\"] == 0.0] = df_full[\"source\"][df_full[\"cell_type\"] == 0.0].apply(lambda x: str_cleansing(x, 0))\n# df_full[\"source\"][df_full[\"cell_type\"] == 1.0] = df_full[\"source\"][df_full[\"cell_type\"] == 1.0].apply(lambda x: str_cleansing(x, 1))\n\ndf_full[\"source\"] = df_full[\"source\"].apply(lambda x: str_cleansing(x))","metadata":{"execution":{"iopub.status.busy":"2022-07-06T06:41:13.429334Z","iopub.execute_input":"2022-07-06T06:41:13.430089Z","iopub.status.idle":"2022-07-06T06:41:20.610025Z","shell.execute_reply.started":"2022-07-06T06:41:13.430002Z","shell.execute_reply":"2022-07-06T06:41:20.608957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# remove same string\ndf_full = df_full.groupby(\"id\").apply(lambda x: x.drop_duplicates([\"source\"], keep=\"last\")).reset_index(0, drop=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-06T06:41:20.611461Z","iopub.execute_input":"2022-07-06T06:41:20.611751Z","iopub.status.idle":"2022-07-06T06:41:20.792652Z","shell.execute_reply.started":"2022-07-06T06:41:20.611710Z","shell.execute_reply":"2022-07-06T06:41:20.791804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# remove markdown cells of which rank value is over 150 (so many markdown cell in head section)\nover_md_cell = df_full.loc[df_full[\"cell_type\"] == 1.0, \"rank\"].index[df_full.loc[df_full[\"cell_type\"] == 1.0, \"rank\"] > 150].get_level_values(0).unique()\ndf_full = df_full.drop(over_md_cell)","metadata":{"execution":{"iopub.status.busy":"2022-07-06T06:41:20.793961Z","iopub.execute_input":"2022-07-06T06:41:20.794750Z","iopub.status.idle":"2022-07-06T06:41:20.807648Z","shell.execute_reply.started":"2022-07-06T06:41:20.794675Z","shell.execute_reply":"2022-07-06T06:41:20.806745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_full.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-07-06T06:41:20.809662Z","iopub.execute_input":"2022-07-06T06:41:20.809983Z","iopub.status.idle":"2022-07-06T06:41:20.826274Z","shell.execute_reply.started":"2022-07-06T06:41:20.809942Z","shell.execute_reply":"2022-07-06T06:41:20.825381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**get only markdown cell**","metadata":{}},{"cell_type":"code","source":"createFolder(\"./df_full/raw/\")\ncreateFolder(\"./df_full/fold/\")\neasyIO(df_full, \"./df_full/raw/df_full_ori.pkl\", \"w\")","metadata":{"execution":{"iopub.status.busy":"2022-07-06T06:33:45.753492Z","iopub.execute_input":"2022-07-06T06:33:45.754221Z","iopub.status.idle":"2022-07-06T06:33:45.766684Z","shell.execute_reply.started":"2022-07-06T06:33:45.754178Z","shell.execute_reply":"2022-07-06T06:33:45.765735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_full = df_full[(df_full[\"cell_type\"] == 1.0)]\ndf_full.drop(\"cell_type\", axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-06T06:33:45.768063Z","iopub.execute_input":"2022-07-06T06:33:45.772083Z","iopub.status.idle":"2022-07-06T06:33:45.779213Z","shell.execute_reply.started":"2022-07-06T06:33:45.772051Z","shell.execute_reply":"2022-07-06T06:33:45.778268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Extract distilbert feature vector","metadata":{}},{"cell_type":"code","source":"from transformers import AutoTokenizer, AutoModel\nimport torch\nfrom torch.utils.data import DataLoader, TensorDataset\ntorch_device = torch.device('cuda') if torch.cuda.is_available() else None\n\ncodebert_tokenizer_path = \"../input/ai4code-ms-codebert/microsoft-codebert-base/tokenizer/\"\ncodebert_model_path = \"../input/ai4code-ms-codebert/microsoft-codebert-base/model/\"\n\nbert_tokenizer_path = \"../input/ai4code-bert/bert-base-uncased/tokenizer/\"\nbert_model_path = \"../input/ai4code-bert/bert-base-uncased/model/\"\n\ndistilbert_tokenizer_path = \"../input/ai4code-distilbert/distilbert-base-uncased/tokenizer/\"\ndistilbert_model_path = \"../input/ai4code-distilbert/distilbert-base-uncased/model/\"\n\ntokenizer = AutoTokenizer.from_pretrained(distilbert_tokenizer_path)\nif torch.cuda.is_available():\n    model = AutoModel.from_pretrained(distilbert_model_path).to(torch_device)\nelse:\n    model = AutoModel.from_pretrained(distilbert_model_path)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-05T07:36:41.287506Z","iopub.execute_input":"2022-07-05T07:36:41.287777Z","iopub.status.idle":"2022-07-05T07:36:42.270799Z","shell.execute_reply.started":"2022-07-05T07:36:41.287747Z","shell.execute_reply":"2022-07-05T07:36:42.269784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**checking the number of tokens**","metadata":{}},{"cell_type":"code","source":"MAX_LEN = 512\ntokenizer_output = {\n    \"input_ids\": np.empty(shape=(df_full.shape[0], MAX_LEN), dtype=\"int64\"),\n    \"attention_mask\": np.empty(shape=(df_full.shape[0], MAX_LEN), dtype=\"int64\")\n}\n\nfor idx, value in enumerate(df_full[\"source\"].to_list()):\n    tokens = tokenizer(\n        value,\n        max_length=MAX_LEN,\n        padding=\"max_length\",\n        truncation=True,\n        return_token_type_ids=False,\n    )\n    tokenizer_output[\"input_ids\"][idx] = tokens[\"input_ids\"]\n    tokenizer_output[\"attention_mask\"][idx] = tokens[\"attention_mask\"]","metadata":{"execution":{"iopub.status.busy":"2022-07-05T07:36:44.317279Z","iopub.execute_input":"2022-07-05T07:36:44.317548Z","iopub.status.idle":"2022-07-05T07:36:45.427152Z","shell.execute_reply.started":"2022-07-05T07:36:44.317518Z","shell.execute_reply":"2022-07-05T07:36:45.426124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_size = 16\ndf_full_fv = []\n\nif torch.cuda.is_available():\n    loader = DataLoader(\n        TensorDataset(torch.from_numpy(tokenizer_output[\"input_ids\"]).to(torch_device),\n                      torch.from_numpy(tokenizer_output[\"attention_mask\"]).to(torch_device)),\n        batch_size=batch_size,\n    )\n    for batch, (input_ids, attention_mask) in enumerate(tqdm(loader)):\n        outputs = model(input_ids=input_ids, attention_mask=attention_mask)\n        df_full_fv.append(outputs.last_hidden_state.detach().cpu().numpy().mean(axis=1))\nelse:\n    loader = DataLoader(\n        TensorDataset(torch.from_numpy(tokenizer_output[\"input_ids\"]),\n                      torch.from_numpy(tokenizer_output[\"attention_mask\"])),\n        batch_size=batch_size,\n    )\n    for batch, (input_ids, attention_mask) in enumerate(tqdm(loader)):\n        outputs = model(input_ids=input_ids, attention_mask=attention_mask)\n        df_full_fv.append(outputs.last_hidden_state.detach().cpu().numpy().mean(axis=1))\n\ntorch.cuda.empty_cache()\ndf_full_fv = np.concatenate(df_full_fv, axis=0)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T07:36:47.406738Z","iopub.execute_input":"2022-07-05T07:36:47.407185Z","iopub.status.idle":"2022-07-05T07:37:05.577914Z","shell.execute_reply.started":"2022-07-05T07:36:47.407152Z","shell.execute_reply":"2022-07-05T07:37:05.576925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cell_length = series([(series(i) == 1).sum() for i in tokenizer_output[\"attention_mask\"]])\nprint(\"valid token length summary\")\ncell_length.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T07:37:34.823537Z","iopub.execute_input":"2022-07-05T07:37:34.82381Z","iopub.status.idle":"2022-07-05T07:37:35.772822Z","shell.execute_reply.started":"2022-07-05T07:37:34.82378Z","shell.execute_reply":"2022-07-05T07:37:35.771939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(12, 6))\ngraph = sns.histplot(cell_length, bins=50, color=\"orange\")\nplt.title(\"Distribution on the number of tokens (\" + str(MAX_LEN) + \")\", fontsize=15, fontweight=\"bold\", pad=15)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T07:38:40.673424Z","iopub.execute_input":"2022-07-05T07:38:40.673712Z","iopub.status.idle":"2022-07-05T07:38:41.079779Z","shell.execute_reply.started":"2022-07-05T07:38:40.673667Z","shell.execute_reply":"2022-07-05T07:38:41.078903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**shuffle train data and set proper number of token (64)**","metadata":{}},{"cell_type":"code","source":"nb_group = df_full.index.get_level_values(0).unique()\nlen(nb_group)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T07:38:45.25669Z","iopub.execute_input":"2022-07-05T07:38:45.257373Z","iopub.status.idle":"2022-07-05T07:38:45.266045Z","shell.execute_reply.started":"2022-07-05T07:38:45.257338Z","shell.execute_reply":"2022-07-05T07:38:45.264835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np_rnd.seed(42)\nshuffled_idx = np_rnd.choice(len(nb_group), size=len(nb_group), replace=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T07:38:48.495612Z","iopub.execute_input":"2022-07-05T07:38:48.496038Z","iopub.status.idle":"2022-07-05T07:38:48.504071Z","shell.execute_reply.started":"2022-07-05T07:38:48.495992Z","shell.execute_reply":"2022-07-05T07:38:48.502892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_full = df_full.loc[nb_group[shuffled_idx]]","metadata":{"execution":{"iopub.status.busy":"2022-07-05T07:38:50.198838Z","iopub.execute_input":"2022-07-05T07:38:50.199237Z","iopub.status.idle":"2022-07-05T07:38:50.208038Z","shell.execute_reply.started":"2022-07-05T07:38:50.19918Z","shell.execute_reply":"2022-07-05T07:38:50.206932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"MAX_LEN = 64\ntokenizer_output = {\n    \"input_ids\": np.empty(shape=(df_full.shape[0], MAX_LEN), dtype=\"int64\"),\n    \"attention_mask\": np.empty(shape=(df_full.shape[0], MAX_LEN), dtype=\"int64\")\n}\n\nfor idx, value in enumerate(df_full[\"source\"].to_list()):\n    tokens = tokenizer(\n        value,\n        max_length=MAX_LEN,\n        padding=\"max_length\",\n        truncation=True,\n        return_token_type_ids=False,\n    )\n    tokenizer_output[\"input_ids\"][idx] = tokens[\"input_ids\"]\n    tokenizer_output[\"attention_mask\"][idx] = tokens[\"attention_mask\"]","metadata":{"execution":{"iopub.status.busy":"2022-07-05T07:39:00.460421Z","iopub.execute_input":"2022-07-05T07:39:00.460733Z","iopub.status.idle":"2022-07-05T07:39:01.361139Z","shell.execute_reply.started":"2022-07-05T07:39:00.460701Z","shell.execute_reply":"2022-07-05T07:39:01.360186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_size = 16\ndf_full_fv = []\n\nif torch.cuda.is_available():\n    loader = DataLoader(\n        TensorDataset(torch.from_numpy(tokenizer_output[\"input_ids\"]).to(torch_device),\n                      torch.from_numpy(tokenizer_output[\"attention_mask\"]).to(torch_device)),\n        batch_size=batch_size,\n    )\n    for batch, (input_ids, attention_mask) in enumerate(tqdm(loader)):\n        outputs = model(input_ids=input_ids, attention_mask=attention_mask)\n        df_full_fv.append(outputs.last_hidden_state.detach().cpu().numpy().mean(axis=1))\nelse:\n    loader = DataLoader(\n        TensorDataset(torch.from_numpy(tokenizer_output[\"input_ids\"]),\n                      torch.from_numpy(tokenizer_output[\"attention_mask\"])),\n        batch_size=batch_size,\n    )\n    for batch, (input_ids, attention_mask) in enumerate(tqdm(loader)):\n        outputs = model(input_ids=input_ids, attention_mask=attention_mask)\n        df_full_fv.append(outputs.last_hidden_state.detach().cpu().numpy().mean(axis=1))\n\ntorch.cuda.empty_cache()\ndf_full_fv = np.concatenate(df_full_fv, axis=0)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T07:39:02.682083Z","iopub.execute_input":"2022-07-05T07:39:02.682538Z","iopub.status.idle":"2022-07-05T07:39:05.386633Z","shell.execute_reply.started":"2022-07-05T07:39:02.682505Z","shell.execute_reply":"2022-07-05T07:39:05.385647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cell_length = series([(series(i) == 1).sum() for i in tokenizer_output[\"attention_mask\"]])\nprint(\"valid token length summary\")\ncell_length.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T07:39:15.994668Z","iopub.execute_input":"2022-07-05T07:39:15.994948Z","iopub.status.idle":"2022-07-05T07:39:16.561619Z","shell.execute_reply.started":"2022-07-05T07:39:15.994919Z","shell.execute_reply":"2022-07-05T07:39:16.560585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(12, 6))\ngraph = sns.histplot(cell_length, bins=50, color=\"green\")\nplt.title(\"Distribution on the number of tokens (\" + str(MAX_LEN) + \")\", fontsize=15, fontweight=\"bold\", pad=15)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T07:39:22.33623Z","iopub.execute_input":"2022-07-05T07:39:22.338265Z","iopub.status.idle":"2022-07-05T07:39:23.046692Z","shell.execute_reply.started":"2022-07-05T07:39:22.338216Z","shell.execute_reply":"2022-07-05T07:39:23.045868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_full.drop(\"source\", axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T07:39:53.136678Z","iopub.execute_input":"2022-07-05T07:39:53.137286Z","iopub.status.idle":"2022-07-05T07:39:53.146049Z","shell.execute_reply.started":"2022-07-05T07:39:53.137246Z","shell.execute_reply":"2022-07-05T07:39:53.142601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_full_fv.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-05T07:39:54.369184Z","iopub.execute_input":"2022-07-05T07:39:54.369525Z","iopub.status.idle":"2022-07-05T07:39:54.37674Z","shell.execute_reply.started":"2022-07-05T07:39:54.369478Z","shell.execute_reply":"2022-07-05T07:39:54.375871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_full_fv[:5]","metadata":{"execution":{"iopub.status.busy":"2022-07-05T07:39:55.502085Z","iopub.execute_input":"2022-07-05T07:39:55.502466Z","iopub.status.idle":"2022-07-05T07:39:55.546498Z","shell.execute_reply.started":"2022-07-05T07:39:55.502419Z","shell.execute_reply":"2022-07-05T07:39:55.5434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Save","metadata":{}},{"cell_type":"code","source":"df_full_groups = (df_full.index.get_level_values(0), df_full.index.get_level_values(1))\nids = df_full.index.unique('id')\n\n# for group fold split\ndf_ancestors = pd.read_csv(data_dir / 'train_ancestors.csv', index_col='id')\nancestors = df_ancestors.loc[ids, 'ancestor_id']\n\n# stratified sampling by each length of code blocks\nnb_length_spliter = KBinsDiscretizer(n_bins=5, strategy=\"quantile\", encode=\"ordinal\")\nstrat_y = nb_length_spliter.fit_transform(df_full.groupby('id').size().to_frame()).flatten().astype(\"int32\")","metadata":{"execution":{"iopub.status.busy":"2022-07-05T07:40:26.516056Z","iopub.execute_input":"2022-07-05T07:40:26.516455Z","iopub.status.idle":"2022-07-05T07:40:26.86409Z","shell.execute_reply.started":"2022-07-05T07:40:26.516414Z","shell.execute_reply":"2022-07-05T07:40:26.863017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"easyIO(df_full, \"./df_full/raw/df_full.pkl\", \"w\")\neasyIO(df_orders, \"./df_full/raw/df_orders.pkl\", \"w\")\nnp.savez_compressed(\"./df_full/raw/df_full_fv.npz\", value=df_full_fv)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T07:40:28.153203Z","iopub.execute_input":"2022-07-05T07:40:28.153557Z","iopub.status.idle":"2022-07-05T07:40:31.659912Z","shell.execute_reply.started":"2022-07-05T07:40:28.153518Z","shell.execute_reply":"2022-07-05T07:40:31.658931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_folds = 5\nkfolds_spliter = StratifiedGroupKFold(n_folds, shuffle=True, random_state=42)\n\nfor fold, (train_idx, val_idx) in enumerate(kfolds_spliter.split(range(len(ids)), y=strat_y, groups=ancestors)):   \n    fold_save_path = \"./df_full/fold/train_fold_\" + str(fold) + \".npz\" \n    np.savez_compressed(\n        fold_save_path,\n        x=df_full_fv[findIdx(df_full_groups[0], ids[train_idx])],\n        y=(df_full[\"rank\"][findIdx(df_full_groups[0], ids[train_idx])]).to_numpy(),\n        groups=(df_full_groups[0][findIdx(df_full_groups[0], ids[train_idx])]).to_numpy()\n    )\n    \n    fold_save_path = \"./df_full/fold/val_fold_\" + str(fold) + \".npz\" \n    np.savez_compressed(\n        fold_save_path,\n        x=df_full_fv[findIdx(df_full_groups[0], ids[val_idx])],\n        y=(df_full[\"rank\"][findIdx(df_full_groups[0], ids[val_idx])]).to_numpy(),\n        groups=(df_full_groups[0][findIdx(df_full_groups[0], ids[val_idx])]).to_numpy()\n    )","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-05T07:40:33.456786Z","iopub.execute_input":"2022-07-05T07:40:33.45716Z","iopub.status.idle":"2022-07-05T07:40:35.243778Z","shell.execute_reply.started":"2022-07-05T07:40:33.457129Z","shell.execute_reply":"2022-07-05T07:40:35.242872Z"},"trusted":true},"execution_count":null,"outputs":[]}]}