{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Introduction #\n\n* [1/3] [Data Preparation][1]\n* **[2/3] [CatBoost Ranker Training][2] ← (this kernel)**\n* [3/3] [CatBoost Ranker Inference][3]\n\n### About Solution\n\n- Feature data\n    - **Markdown string (64 tokens) & No code string**\n    - From markdown string to Distilbert feature vector (768 vector)\n- Target data\n    - Set code cell position target value\n    - Set markdown cell posision target value with increasing the target value by linearly\n    - **So markdown cell's target values can be over 100 (Normalizing is not applied)**\n    - This way can be helpful for calculating unbiased target value on markdown cell\n    - Since this way doesn't refer to exact position of markdown cell\n- Model and hyperparameters\n    - Model : CatBoost Ranker\n    - Loss : RMSE (by groups)\n    - Hyperparameters : Tuning with Optuna\n    - **Log transformation on markdown cell position target value (for normal distribution on the target value)**\n   \nI refer to the form of **[NICK KUZMENKOV][4]'s kernels** which are listed below :)<br>\n* [1/3] [Data Preparation][1]\n* [2/3] [TPU Training][2] (~4 hours)\n* [3/3] [GPU Inference][3] (~2 hours)\n\n[1]: https://www.kaggle.com/cafelatte1/catboost-ranker-data-preparation\n[2]: https://www.kaggle.com/cafelatte1/catboost-ranker-training\n[3]: https://www.kaggle.com/cafelatte1/catboost-ranker-inference\n[4]: https://www.kaggle.com/nickuzmenkov/ai4code-tf-tpu-codebert-data-preparation/notebook\n[5]: https://www.kaggle.com/nickuzmenkov/ai4code-tf-tpu-codebert-training\n[6]: https://www.kaggle.com/nickuzmenkov/ai4code-tf-tpu-codebert-inference\n[7]: https://www.kaggle.com/nickuzmenkov","metadata":{}},{"cell_type":"markdown","source":"# Setup #","metadata":{}},{"cell_type":"code","source":"# import sys\n# !cp ../input/rapids/rapids.21.06 /opt/conda/envs/rapids.tar.gz\n# !cd /opt/conda/envs/ && tar -xzvf rapids.tar.gz > /dev/null\n# sys.path = [\"/opt/conda/envs/rapids/lib/python3.7/site-packages\"] + sys.path\n# sys.path = [\"/opt/conda/envs/rapids/lib/python3.7\"] + sys.path\n# sys.path = [\"/opt/conda/envs/rapids/lib\"] + sys.path \n# !cp /opt/conda/envs/rapids/lib/libxgboost.so /opt/conda/lib/","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-12T07:39:10.400013Z","iopub.execute_input":"2022-07-12T07:39:10.400342Z","iopub.status.idle":"2022-07-12T07:39:10.420589Z","shell.execute_reply.started":"2022-07-12T07:39:10.400258Z","shell.execute_reply":"2022-07-12T07:39:10.419874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport sys\nimport shutil\nfrom glob import glob\nimport multiprocessing as mp\nimport gc\nfrom pathlib import Path\nfrom scipy import stats\nfrom scipy.special import boxcox, softmax\nfrom scipy import sparse\n\nfrom multiprocessing import cpu_count\nimport copy\nimport pickle\nimport warnings\nfrom datetime import datetime, timedelta\nfrom time import time, sleep, mktime\nfrom matplotlib import font_manager as fm, rc, rcParams\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom tqdm import tqdm\nimport re\nimport random as rnd\nimport psutil\nfrom optuna import Trial, create_study\nfrom optuna.samplers import TPESampler\n\nimport numpy as np\nfrom numpy import array, nan, random as np_rnd, where\nimport pandas as pd\nfrom pandas import DataFrame as dataframe, Series as series, isna, read_csv\nfrom pandas.tseries.offsets import DateOffset\n\nfrom sklearn.model_selection import train_test_split as tts, StratifiedKFold, StratifiedShuffleSplit\nfrom sklearn.preprocessing import LabelEncoder, OneHotEncoder, StandardScaler, MinMaxScaler, RobustScaler, KBinsDiscretizer\nfrom sklearn import metrics\nfrom sklearn.compose import ColumnTransformer\n# config_missingpy(); from missingpy import MissForest\nfrom sklearn.impute import KNNImputer\nfrom optuna import Trial, create_study\nfrom sklearn.model_selection import GroupKFold, GroupShuffleSplit, StratifiedGroupKFold\nfrom sklearn.feature_extraction.text import TfidfVectorizer\n\ntry:\n    import cudf as cd\n    import cupy as cp\n    from cuml.cluster import KMeans\n    from cuml.neighbors import NearestNeighbors\n    from cuml.metrics.cluster import silhouette_score\nexcept:\n    print(\"RAPIDS Import ERROR\")\n\nimport catboost as cat\nimport xgboost as xgb\n\n# # ===== tensorflow =====\n# import tensorflow as tf\n# from tensorflow import random as tf_rnd\n# from tensorflow.keras.models import Model\n# from tensorflow.keras.models import Sequential\n# from tensorflow.keras import layers\n# from tensorflow.keras import activations\n# from tensorflow.keras import optimizers\n# from tensorflow.keras import metrics as tf_metrics\n# from tensorflow.keras import callbacks as tf_callbacks\n# from tqdm.keras import TqdmCallback\n# import tensorflow_addons as tfa\n# from tensorflow.keras.utils import plot_model\n# from keras.utils.layer_utils import count_params\n\n# import keras_tuner as kt\n# from keras_tuner import HyperModel\n# import tensorflow_hub as tf_hub\n# import tensorflow_recommenders as tfrs\n\n# # GPU check\n# if tf.test.gpu_device_name() != '/device:GPU:0':\n#     print('GPU device not found')\n# else:\n#     print('Found GPU')\n\n# # GPU memory setting\n# gpus = tf.config.list_physical_devices('GPU')\n# if gpus:\n#   try:\n#     tf.config.experimental.set_memory_growth(gpus[0], True)\n#   except RuntimeError as e:\n#     print(e)\n\nwarnings.filterwarnings(action='ignore')\nrcParams['axes.unicode_minus'] = False\npd.set_option('display.max_columns', 100)\npd.set_option('display.max_rows', 100)\npd.set_option('display.width', 1000)\npd.set_option('max_colwidth', 200)\n# plt.rc('font', family='NanumSquareB')\n\ndata_dir = Path('../input/AI4Code')","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-12T07:39:10.505829Z","iopub.execute_input":"2022-07-12T07:39:10.506148Z","iopub.status.idle":"2022-07-12T07:39:14.845721Z","shell.execute_reply.started":"2022-07-12T07:39:10.506114Z","shell.execute_reply":"2022-07-12T07:39:14.844827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ===== utility functions =====\n# label encoding for categorical column with excepting na value\ndef seed_everything(seed=42):\n    # python random module\n    rnd.seed(seed)\n    # numpy random\n    np_rnd.seed(seed)\n#     # tf random\n#     tf_rnd.set_seed(seed)\n    # RAPIDS random\n    try:\n        cp.random.seed(seed)\n    except:\n        pass\ndef which(bool_list):\n    return where(bool_list)[0]\ndef easyIO(x=None, path=None, op=\"r\"):\n    tmp = None\n    if op == \"r\":\n        with open(path, \"rb\") as f:\n            tmp = pickle.load(f)\n        return tmp\n    elif op == \"w\":\n        with open(path, \"wb\") as f:\n            pickle.dump(x, f)\n    else:\n        print(\"Unknown operation type\")\ndef diff(first, second):\n    second = set(second)\n    return [item for item in first if item not in second]\ndef findIdx(data_x, col_names):\n    return [int(i) for i, j in enumerate(data_x) if j in col_names]\ndef orderElems(for_order, using_ref):\n    return [i for i in using_ref if i in for_order]\n# concatenate by row\ndef cbr(df1, df2):\n    if type(df1) == series:\n        tmp_concat = series(pd.concat([dataframe(df1), dataframe(df2)], axis=0, ignore_index=True).iloc[:,0])\n        tmp_concat.reset_index(drop=True, inplace=True)\n    elif type(df1) == dataframe:\n        tmp_concat = pd.concat([df1, df2], axis=0, ignore_index=True)\n        tmp_concat.reset_index(drop=True, inplace=True)\n    elif type(df1) == np.ndarray:\n        tmp_concat = np.concatenate([df1, df2], axis=0)\n    else:\n        print(\"Unknown Type: return 1st argument\")\n        tmp_concat = df1\n    return tmp_concat\ndef change_width(ax, new_value):\n    for patch in ax.patches :\n        current_width = patch.get_width()\n        adj_value = current_width - new_value\n        # we change the bar width\n        patch.set_width(new_value)\n        # we recenter the bar\n        patch.set_x(patch.get_x() + adj_value * .5)\ndef week_of_month(date):\n    month = date.month\n    week = 0\n    while date.month == month:\n        week += 1\n        date -= timedelta(days=7)\n    return week\ndef getSeason(date):\n    month = date.month\n    if month in [3, 4, 5]:\n        return \"Spring\"\n    elif month in [6, 7, 8]:\n        return \"Summer\"\n    elif month in [9, 10, 11]:\n        return \"Fall\"\n    else:\n        return \"Winter\"\ndef createFolder(directory):\n    try:\n        if not os.path.exists(directory):\n            os.makedirs(directory)\n    except OSError:\n        print('Error: Creating directory. ' + directory)\n# def softmax(x):\n#     max = np.max(x, axis=1, keepdims=True)  # returns max of each row and keeps same dims\n#     e_x = np.exp(x - max)  # subtracts each row with its max value\n#     sum = np.sum(e_x, axis=1, keepdims=True)  # returns sum of each row and keeps same dims\n#     f_x = e_x / sum\n#     return f_x\ndef sigmoid(x):\n    return 1/(1 + np.exp(-x))\ndef dispPerformance(result_dic):\n    perf_table = dataframe()\n    index_names = []\n    for k, v in result_dic.items():\n        index_names.append(k)\n        perf_table = pd.concat([perf_table, series(v[\"performance\"]).to_frame().T], ignore_index=True, axis=0)\n    perf_table.index = index_names\n    perf_table.sort_values(perf_table.columns[0], inplace=True)\n    print(perf_table)\n    return perf_table\ndef powspace(start, stop, power, num):\n    start = np.power(start, 1/float(power))\n    stop = np.power(stop, 1/float(power))\n    return np.power(np.linspace(start, stop, num=num), power)\ndef xgb_custom_lossfunction(alpha = 1):\n    def support_under_mse(label, pred):\n        # grad : 1차 미분\n        # hess : 2차 미분\n        residual = (label - pred).astype(\"float\")\n        grad = np.where(residual > 0, -2 * alpha * residual, -2 * residual)\n        hess = np.where(residual > 0, 2 * alpha, 2.0)\n        return grad, hess\n    return support_under_mse\ndef pd_flatten(df):\n    df = df.unstack()\n    df.index = [str(i) + \"_\" + str(j) for i, j in df.index]\n    return df\ndef tf_losses_rmse(y_true, y_pred, sample_weight=None):\n    return tf.sqrt(tf.reduce_mean((y_true - y_pred) ** 2)) if sample_weight is None else tf.sqrt(tf.reduce_mean(((y_true - y_pred) ** 2) * sample_weight))\ndef tf_loss_nmae(y_true, y_pred, sample_weight=False):\n    mae = tf.reduce_mean(tf.math.abs(y_true - y_pred))\n    score = tf.math.divide(mae, tf.reduce_mean(tf.math.abs(y_true)))\n    return score\ndef text_extractor(string, lang=\"eng\", spacing=True):\n    # # 괄호를 포함한 괄호 안 문자 제거 정규식\n    # re.sub(r'\\([^)]*\\)', '', remove_text)\n    # # <>를 포함한 <> 안 문자 제거 정규식\n    # re.sub(r'\\<[^)]*\\>', '', remove_text)\n    if lang == \"eng\":\n        text_finder = re.compile('[^ A-Za-z]') if spacing else re.compile('[^A-Za-z]')\n    elif lang == \"kor\":\n        text_finder = re.compile('[^ ㄱ-ㅣ가-힣+]') if spacing else re.compile('[^ㄱ-ㅣ가-힣+]')\n    # default : kor + eng\n    else:\n        text_finder = re.compile('[^ A-Za-zㄱ-ㅣ가-힣+]') if spacing else re.compile('[^A-Za-zㄱ-ㅣ가-힣+]')\n    return text_finder.sub('', string)\ndef memory_usage(message='debug'):\n    # current process RAM usage\n    p = psutil.Process()\n    rss = p.memory_info().rss / 2 ** 20 # Bytes to MB\n    print(f\"[{message}] memory usage: {rss: 10.3f} MB\")\n    return rss\nclass MyLabelEncoder:\n    def __init__(self, preset={}):\n        # dic_cat format -> {\"col_name\": {\"value\": replace}}\n        self.dic_cat = preset\n    def fit_transform(self, data_x, col_names):\n        tmp_x = copy.deepcopy(data_x)\n        for i in col_names:\n            # if key is not in dic, update dic\n            if i not in self.dic_cat.keys():\n                tmp_dic = dict.fromkeys(sorted(set(tmp_x[i]).difference([nan])))\n                label_cnt = 0\n                for j in tmp_dic.keys():\n                    tmp_dic[j] = label_cnt\n                    label_cnt += 1\n                self.dic_cat[i] = tmp_dic\n            # transform value which is not in dic to nan\n            tmp_x[i] = tmp_x[i].astype(\"object\")\n            conv = tmp_x[i].replace(self.dic_cat[i])\n            for conv_idx, j in enumerate(conv):\n                if j not in self.dic_cat[i].values():\n                    conv[conv_idx] = nan\n            # final return\n            tmp_x[i] = conv.astype(\"float\")\n        return tmp_x\n    def transform(self, data_x):\n        tmp_x = copy.deepcopy(data_x)\n        for i in self.dic_cat.keys():\n            # transform value which is not in dic to nan\n            tmp_x[i] = tmp_x[i].astype(\"object\")\n            conv = tmp_x[i].replace(self.dic_cat[i])\n            for conv_idx, j in enumerate(conv):\n                if j not in self.dic_cat[i].values():\n                    conv[conv_idx] = nan\n            # final return\n            tmp_x[i] = conv.astype(\"float\")\n        return tmp_x\n    def clear(self):\n        self.dic_cat = {}\nclass MyOneHotEncoder:\n    def __init__(self, label_preset={}):\n        self.dic_cat = {}\n        self.label_preset = label_preset\n    def fit_transform(self, data_x, col_names):\n        tmp_x = dataframe()\n        for i in data_x:\n            if i not in col_names:\n                tmp_x = pd.concat([tmp_x, dataframe(data_x[i])], axis=1)\n            else:\n                if not ((data_x[i].dtype.name == \"object\") or (data_x[i].dtype.name == \"category\")):\n                    print(F\"WARNING : {i} is not object or category\")\n                self.dic_cat[i] = OneHotEncoder(sparse=False, handle_unknown=\"ignore\")\n                conv = self.dic_cat[i].fit_transform(dataframe(data_x[i])).astype(\"int\")\n                col_list = []\n                for j in self.dic_cat[i].categories_[0]:\n                    if i in self.label_preset.keys():\n                        for k, v in self.label_preset[i].items():\n                            if v == j:\n                                col_list.append(str(i) + \"_\" + str(k))\n                    else:\n                        col_list.append(str(i) + \"_\" + str(j))\n                conv = dataframe(conv, columns=col_list)\n                tmp_x = pd.concat([tmp_x, conv], axis=1)\n        return tmp_x\n    def transform(self, data_x):\n        tmp_x = dataframe()\n        for i in data_x:\n            if not i in list(self.dic_cat.keys()):\n                tmp_x = pd.concat([tmp_x, dataframe(data_x[i])], axis=1)\n            else:\n                if not ((data_x[i].dtype.name == \"object\") or (data_x[i].dtype.name == \"category\")):\n                    print(F\"WARNING : {i} is not object or category\")\n                conv = self.dic_cat[i].transform(dataframe(data_x[i])).astype(\"int\")\n                col_list = []\n                for j in self.dic_cat[i].categories_[0]:\n                    if i in self.label_preset.keys():\n                        for k, v in self.label_preset[i].items():\n                            if v == j: col_list.append(str(i) + \"_\" + str(k))\n                    else:\n                        col_list.append(str(i) + \"_\" + str(j))\n                conv = dataframe(conv, columns=col_list)\n                tmp_x = pd.concat([tmp_x, conv], axis=1)\n        return tmp_x\n    def clear(self):\n        self.dic_cat = {}\n        self.label_preset = {}\nclass MyKNNImputer:\n    def __init__(self, k=5):\n        self.imputer = KNNImputer(n_neighbors=k)\n        self.dic_cat = {}\n    def fit_transform(self, x, cat_vars=None):\n        if cat_vars is None:\n            x_imp = dataframe(self.imputer.fit_transform(x), columns=x.columns)\n        else:\n            naIdx = dict.fromkeys(cat_vars)\n            for i in cat_vars:\n                self.dic_cat[i] = diff(list(sorted(set(x[i]))), [nan])\n                naIdx[i] = list(which(array(x[i].isna())))\n            x_imp = dataframe(self.imputer.fit_transform(x), columns=x.columns)\n\n            # if imputed categorical value are not in the range, adjust the value\n            for i in cat_vars:\n                x_imp[i] = x_imp[i].apply(lambda x: int(round(x, 0)))\n                for j in naIdx[i]:\n                    if x_imp[i][j] not in self.dic_cat[i]:\n                        if x_imp[i][j] < self.dic_cat[i][0]:\n                            x_imp[i][naIdx[i]] = self.dic_cat[i][0]\n                        elif x_imp[i][j] > self.dic_cat[i][0]:\n                            x_imp[i][naIdx[i]] = self.dic_cat[i][len(self.dic_cat[i]) - 1]\n        return x_imp\n    def transform(self, x):\n        if len(self.dic_cat.keys()) == 0:\n            x_imp = dataframe(self.imputer.transform(x), columns=x.columns)\n        else:\n            naIdx = dict.fromkeys(self.dic_cat.keys())\n            for i in self.dic_cat.keys():\n                naIdx[i] = list(which(array(x[i].isna())))\n            x_imp = dataframe(self.imputer.transform(x), columns=x.columns)\n\n            # if imputed categorical value are not in the range, adjust the value\n            for i in self.dic_cat.keys():\n                x_imp[i] = x_imp[i].apply(lambda x: int(round(x, 0)))\n                for j in naIdx[i]:\n                    if x_imp[i][j] not in self.dic_cat[i]:\n                        if x_imp[i][j] < self.dic_cat[i][0]:\n                            x_imp[i][naIdx[i]] = self.dic_cat[i][0]\n                        elif x_imp[i][j] > self.dic_cat[i][0]:\n                            x_imp[i][naIdx[i]] = self.dic_cat[i][len(self.dic_cat[i]) - 1]\n        return x_imp\n    def clear(self):\n        self.imputer = None\n        self.dic_cat = {}\ndef remove_outlier(df, std=3, mode=\"remove\"):\n    tmp_df = df.copy()\n    if mode == \"remove\":\n        outlier_mask = (np.abs(stats.zscore(tmp_df)) > std).all(axis=1)\n        print(\"found outlier :\", outlier_mask.sum())\n        tmp_df = tmp_df[~outlier_mask]\n    elif mode == \"interpolate\":\n        tmp_outlier = []\n        for i in tmp_df:\n            outlier_mask = (np.abs(stats.zscore(tmp_df[i])) > std)\n            tmp_outlier.append(outlier_mask.sum())\n            if tmp_outlier[-1] == 0:\n                continue\n            tmp_df[i][outlier_mask] = np.nan\n            tmp_df[i] = tmp_df[i].interpolate(method='linear').bfill()\n        print(\"found outlier :\", np.sum(outlier_mask))\n    return tmp_df\n\nseed_everything()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-12T07:39:14.847754Z","iopub.execute_input":"2022-07-12T07:39:14.848157Z","iopub.status.idle":"2022-07-12T07:39:15.132361Z","shell.execute_reply.started":"2022-07-12T07:39:14.848117Z","shell.execute_reply":"2022-07-12T07:39:15.131500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training - CatBoost Ranker","metadata":{}},{"cell_type":"code","source":"from bisect import bisect\n\ndef create_dataset(x, y, batch_size, shuffle=True, sample_weight=None):\n    if sample_weight is not None:\n        dataset = tf.data.Dataset.from_tensor_slices((x, y, sample_weight))\n    else:\n        dataset = tf.data.Dataset.from_tensor_slices((x, y))\n    dataset = dataset.shuffle(int(batch_size * 4), reshuffle_each_iteration=True) if shuffle else dataset\n    dataset = dataset.batch(batch_size)\n    dataset = dataset.prefetch(4)\n    return dataset\ndef count_inversions(a):\n    inversions = 0\n    sorted_so_far = []\n    for i, u in enumerate(a):\n        j = bisect(sorted_so_far, u)\n        inversions += i - j\n        sorted_so_far.insert(j, u)\n    return inversions\ndef kendall_tau(ground_truth, predictions):\n    total_inversions = 0\n    total_2max = 0  # twice the maximum possible inversions across all instances\n    for gt, pred in zip(ground_truth, predictions):\n        ranks = [gt.index(x) for x in pred]  # rank predicted order in terms of ground truth\n        total_inversions += count_inversions(ranks)\n        n = len(gt)\n        total_2max += n * (n - 1)\n    return 1 - 4 * total_inversions / total_2max\ndef convert_rank_ids(x):\n    # Convert the cell_id index into a column.\n    x = x.reset_index('cell_id')\n    # Group the cell_ids for each notebook into a list.\n    x = x.groupby('id')['cell_id'].apply(list)\n    return x\ndef flatten_list(xss):\n    return [x for xs in xss for x in xs]","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-12T07:39:15.135926Z","iopub.execute_input":"2022-07-12T07:39:15.136144Z","iopub.status.idle":"2022-07-12T07:39:15.150171Z","shell.execute_reply.started":"2022-07-12T07:39:15.136120Z","shell.execute_reply":"2022-07-12T07:39:15.149373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# optuna function\ndef optuna_objective_function(trial: Trial, fold, train_x, train_y, train_groups, val_x, val_y, val_groups, categoIdx,\n                              model_name, output_container, ntrees=1000, eta=1e-2, best_ntrees=[None]):\n    if model_name == \"LGB_GOSS\":\n        # objective\n        # regession : \"mae\", \"mse\"\n        # classification - binary : \"binary\"\n        # classification - binary : \"multiclass\" (num_class=n)\n        # ranking : \"xe_ndcg_mart\"\n\n        # metric\n        # regession : \"mae\", \"mse\", \"rmse\"\n        # classification - binary : \"binary_logloss\", \"binary_error\", \"auc\"\n        # classification - muticlass : \"multi_logloss\", \"multi_error\"\n        # ranking : \"ndcg\", \"map\"\n\n        tuning_params = {\n            \"learning_rate\": trial.suggest_categorical(\"learning_rate\", [1e-2, 5e-3, 1e-3]),\n            \"num_leaves\": trial.suggest_categorical(\"num_leaves\", [pow(2, i) - 1 for i in [4, 5, 6, 7, 8]]),\n            # goss sampling hyper-parameter replacing the \"sumample\"\n            \"top_rate\": trial.suggest_float(\"top_rate\", 0.2, 0.5, step=0.1),\n            \"other_rate\": trial.suggest_float(\"other_rate\", 0.2, 0.5, step=0.1),\n            \"colsample_bytree\": trial.suggest_float(\"colsample_bytree\", 0.5, 0.8, step=0.1),\n            \"reg_lambda\": trial.suggest_categorical(\"reg_lambda\", list(np.linspace(1e-3, 1.0, num=150, endpoint=False)) + list(np.linspace(1.0, 1e+2, num=50, endpoint=True))),\n            \"min_child_weight\": trial.suggest_categorical(\"min_child_weight\", list(np.linspace(1e-3, 1.0, num=150, endpoint=False)) + list(np.linspace(1.0, 1e+2, num=50, endpoint=True))),\n            \"min_child_samples\": trial.suggest_int(\"min_child_samples\", 1, 51, step=2),\n            \"min_gain_to_split\": trial.suggest_categorical(\"min_gain_to_split\", list(np.linspace(1e-3, 1.0, num=150, endpoint=False)) + list(np.linspace(1.0, 1e+2, num=50, endpoint=True))),\n            # for binary\n            \"scale_pos_weight\": trial.suggest_categorical(\"scale_pos_weight\", list(np.linspace(1e-3, 1.0, num=150, endpoint=False)) + list(np.linspace(1.0, 1e+2, num=50, endpoint=True))),\n        }\n        model = lgb.LGBMClassifier(boosting_type=\"goss\", objective=\"binary\",\n                                   n_estimators=ntrees, device_type=\"gpu\",\n                                   random_state=fold, verbose=-1, **tuning_params)\n        cb_list = [\n            lgb.early_stopping(stopping_rounds=int(ntrees * 0.2), first_metric_only=True, verbose=False, min_delta=0.001),\n        ]\n        model.fit(train_x, train_y, categorical_feature=categoIdx,\n                  eval_set=(val_x,val_y), eval_metric=\"auc\", callbacks=cb_list)\n        best_ntrees[0] = model.best_iteration_\n    elif model_name == \"LGB_GBM\":\n        # objective\n        # regession : \"mae\", \"mse\"\n        # classification - binary : \"binary\"\n        # classification - binary : \"multiclass\" (num_class=n)\n        # ranking : \"xe_ndcg_mart\"\n\n        # metric\n        # regession : \"mae\", \"mse\", \"rmse\"\n        # classification - binary : \"binary_logloss\", \"binary_error\", \"auc\"\n        # classification - muticlass : \"multi_logloss\", \"multi_error\"\n        # ranking : \"ndcg\", \"map\"\n\n        tuning_params = {\n            \"learning_rate\": trial.suggest_categorical(\"learning_rate\", [1e-2, 5e-3, 1e-3]),\n            \"num_leaves\": trial.suggest_categorical(\"num_leaves\", [pow(2, i) - 1 for i in [4, 5, 6, 7, 8]]),\n            \"subsample\": trial.suggest_float(\"subsample\", 0.5, 0.8, step=0.1),\n            \"colsample_bytree\": trial.suggest_float(\"colsample_bytree\", 0.5, 0.8, step=0.1),\n            \"reg_lambda\": trial.suggest_categorical(\"reg_lambda\", list(np.linspace(1e-3, 1.0, num=150, endpoint=False)) + list(np.linspace(1.0, 1e+2, num=50, endpoint=True))),\n            \"min_child_weight\": trial.suggest_categorical(\"min_child_weight\", list(np.linspace(1e-3, 1.0, num=150, endpoint=False)) + list(np.linspace(1.0, 1e+2, num=50, endpoint=True))),\n            \"min_child_samples\": trial.suggest_int(\"min_child_samples\", 1, 51, step=2),\n            \"min_gain_to_split\": trial.suggest_categorical(\"min_gain_to_split\", list(np.linspace(1e-3, 1.0, num=150, endpoint=False)) + list(np.linspace(1.0, 1e+2, num=50, endpoint=True))),\n            # for binary\n            \"scale_pos_weight\": trial.suggest_categorical(\"scale_pos_weight\", list(np.linspace(1e-3, 1.0, num=150, endpoint=False)) + list(np.linspace(1.0, 1e+2, num=50, endpoint=True))),\n        }\n\n        model = lgb.LGBMClassifier(boosting_type=\"gbdt\", objective=\"binary\",\n                                   n_estimators=ntrees, device_type=\"gpu\",\n                                   random_state=fold, verbose=-1, **tuning_params)\n        cb_list = [\n            lgb.early_stopping(stopping_rounds=int(ntrees * 0.2), first_metric_only=True, verbose=False, min_delta=0.001),\n        ]\n        model.fit(train_x, train_y, categorical_feature=categoIdx,\n                  eval_set=(val_x,val_y), eval_metric=\"auc\", callbacks=cb_list)\n        best_ntrees[0] = model.best_iteration_\n    elif model_name == \"XGB_GBT\":\n        # objective\n        # regession : \"reg:absoluteerror\", \"reg:squarederror\"\n        # classification - binary : \"binary:logistic\"\n        # classification - multicalss :\"multi:softmax\" (num_class=n)\n        # ranking : \"rank:ndcg\"\n\n        # metric\n        # regession : \"mae\", \"rmse\"\n        # classification - binary : \"logloss\", \"error@t\" (t=threshold), \"auc\"\n        # classification - multicalss : \"mlogloss\", \"merror\"\n        # ranking : \"ndcg\", \"map\"\n\n        tuning_params = {\n            \"learning_rate\": trial.suggest_categorical(\"learning_rate\", [1e-2, 5e-3, 1e-3]),\n            \"max_depth\": trial.suggest_categorical(\"max_depth\", [4, 5, 6, 7, 8]),\n            \"subsample\": trial.suggest_float(\"subsample\", 0.5, 0.8, step=0.1),\n            \"colsample_bytree\": trial.suggest_float(\"colsample_bytree\", 0.5, 0.8, step=0.1),\n            \"reg_lambda\": trial.suggest_categorical(\"reg_lambda\", list(np.linspace(1e-3, 1.0, num=150, endpoint=False)) + list(np.linspace(1.0, 1e+2, num=50, endpoint=True))),\n            \"min_child_weight\": trial.suggest_categorical(\"min_child_weight\", list(np.linspace(1e-3, 1.0, num=150, endpoint=False)) + list(np.linspace(1.0, 1e+2, num=50, endpoint=True))),\n            \"gamma\": trial.suggest_categorical(\"gamma\", list(np.linspace(1e-3, 1.0, num=150, endpoint=False)) + list(np.linspace(1.0, 1e+2, num=50, endpoint=True))),\n            # for binary\n            \"scale_pos_weight\": trial.suggest_categorical(\"scale_pos_weight\", list(np.linspace(1e-3, 1.0, num=150, endpoint=False)) + list(np.linspace(1.0, 1e+2, num=50, endpoint=True))),\n        }\n        model = xgb.XGBClassifier(booster=\"gbtree\", objective=\"binary:logistic\",\n                            n_estimators=ntrees, tree_method=\"gpu_hist\",\n                            random_state=fold, verbosity=0, **tuning_params)\n        model.fit(train_x, train_y,\n                  eval_set=[(val_x, val_y)], eval_metric=\"auc\",\n                  early_stopping_rounds=int(ntrees * 0.2), verbose=False)\n        best_ntrees[0] = model.best_iteration\n    elif model_name == \"CAT_GBM\":\n#         CatBoostRanker, Pool, MetricVisualizer\n        # objective\n        # regession : \"MAE\", \"RMSE\"\n        # classification - binary : \"Logloss\"\n        # classification - multicalss :\"MultiClass\"\n        # ranking :\n        # past -> slow, low quality -> high quality\n        # RMSE\n        # QueryRMSE\n        # PairLogit\n        # PairLogitPairwise\n        # YetiRank\n        # YetiRankPairwise\n\n        # metric\n        # regession : \"MAE\", \"RMSE\", \"R2\"\n        # classification - binary : \"Logloss\", \"Accuracy\", \"AUC\", \"F1\"\n        # classification - multicalss : \"MultiClass\", \"Accuracy\", \"TotalF1\" (average=Weighted,Macro,Micro)\n        # ranking : \"PairLogit\", \"YetiRank\", \"NDCG\", \"MAP\"\n\n        tuning_params = {\n            \"learning_rate\": trial.suggest_categorical(\"learning_rate\", [5e-3]),\n            \"max_depth\": trial.suggest_categorical(\"max_depth\", [4, 6, 8]),\n            \"bagging_temperature\": trial.suggest_categorical(\"bagging_temperature\", list(np.linspace(1e-3, 1.0, num=150, endpoint=False)) + list(np.linspace(1.0, 1e+2, num=50, endpoint=True))),\n            # rsm = colsample_bylevel (not supported for GPU)\n            # \"rsm\": trial.suggest_float(\"rsm\", 0.5, 0.8, step=0.1)\n            \"random_strength\": trial.suggest_categorical(\"random_strength\", [0.01, 0.05, 0.1, 0.5, 1.0, 1.5, 2.0, 2.5, 3.0]),\n            \"reg_lambda\": trial.suggest_categorical(\"reg_lambda\", list(np.linspace(1e-3, 1.0, num=150, endpoint=False)) + list(np.linspace(1.0, 1e+2, num=50, endpoint=True))),\n            \"min_child_samples\": trial.suggest_float(\"min_child_samples\", 1, 51, step=2),\n            # for binary\n            # \"scale_pos_weight\": trial.suggest_categorical(\"scale_pos_weight\", list(np.linspace(1e-3, 1.0, num=150, endpoint=False)) + list(np.linspace(1.0, 1e+2, num=50, endpoint=True))),\n        }\n\n        model = cat.CatBoostRanker(boosting_type=\"Plain\", loss_function=\"RMSE\", eval_metric=\"RMSE\",\n                            n_estimators=ntrees, task_type=\"GPU\", bootstrap_type=\"Bayesian\",\n                            verbose=False, random_state=fold, **tuning_params)\n    \n        train_pool = cat.Pool(\n            data=train_x,\n            label=train_y,\n            group_id=train_groups\n        )\n\n        val_pool = cat.Pool(\n            data=val_x,\n            label=val_y,\n            group_id=val_groups\n        )\n\n        model.fit(train_pool, cat_features=categoIdx,\n                eval_set=val_pool, early_stopping_rounds=int(ntrees * 0.2), use_best_model=True,\n                verbose=False)\n        best_ntrees[0] = model.best_iteration_\n    elif model_name == \"CAT_ORD\":\n        # objective\n        # regession : \"MAE\", \"RMSE\"\n        # classification - binary : \"Logloss\"\n        # classification - multicalss :\"MultiClass\"\n        # ranking : \"PairLogit\", \"YetiRank\"\n\n        # metric\n        # regession : \"MAE\", \"RMSE\", \"R2\"\n        # classification - binary : \"Logloss\", \"Accuracy\", \"AUC\", \"F1\"\n        # classification - multicalss : \"MultiClass\", \"Accuracy\", \"TotalF1\" (average=Weighted,Macro,Micro)\n        # ranking : \"PairLogit\", \"YetiRank\", \"NDCG\", \"MAP\"\n\n        tuning_params = {\n            \"learning_rate\": trial.suggest_categorical(\"learning_rate\", [5e-3]),\n            \"max_depth\": trial.suggest_categorical(\"max_depth\", [4, 6, 8]),\n            # \"bagging_temperature\": trial.suggest_categorical(\"bagging_temperature\", list(np.linspace(1e-3, 1.0, num=75, endpoint=False)) + list(np.linspace(1.0, 1e+2, num=25, endpoint=True))),\n            # rsm = colsample_bylevel (not supported for GPU)\n            # \"rsm\": trial.suggest_float(\"rsm\", 0.5, 0.8, step=0.1),\n            \"random_strength\": trial.suggest_categorical(\"random_strength\", [0.01, 0.1, 1.0, 2.0, 3.0]),\n            \"reg_lambda\": trial.suggest_categorical(\"reg_lambda\", list(np.linspace(1e-3, 1.0, num=150, endpoint=False)) + list(np.linspace(1.0, 1e+2, num=50, endpoint=True))),\n            # \"min_child_samples\": trial.suggest_float(\"min_child_samples\", 1, 51, step=2),\n            # for binary\n            # \"scale_pos_weight\": trial.suggest_categorical(\"scale_pos_weight\", list(np.linspace(1e-3, 1.0, num=150, endpoint=False)) + list(np.linspace(1.0, 1e+2, num=50, endpoint=True))),\n        }\n\n        model = cat.CatBoostClassifier(boosting_type=\"Ordered\", loss_function=\"RMSE\", eval_metric=\"RMSE\",\n                                    n_estimators=ntrees, task_type=\"GPU\", bootstrap_type=\"Bayesian\",\n                                    verbose=False, random_state=fold, **tuning_params)\n        model.fit(train_x, train_y, cat_features=categoIdx,\n                eval_set=[(val_x, val_y)], early_stopping_rounds=int(ntrees * 0.2), use_best_model=True,\n                verbose=False)\n        best_ntrees[0] = model.best_iteration_\n    else:\n        print(\"unknown\")\n        return -1\n    \n    optuna_score = model.best_score_[\"validation\"][\"RMSE\"]\n#     print(\"kendall_tau score :\", kendall_tau(df_orders.loc[np.unique(val_groups)].sort_index(), convert_rank_ids(tmp.sort_values([\"id\", \"rank\"], ascending=[True, False]))))\n    \n#     tmp = val_pred.loc[np.unique(val_groups)]\n#     tmp.loc[tmp[\"cell_type\"] == 1.0, \"rank\"] = inv_boxcox1p(model.predict(val_pool), boxcox_lambda)\n#     optuna_score = kendall_tau(df_orders.loc[np.unique(val_groups)].sort_index(), convert_rank_ids(tmp.sort_values([\"id\", \"rank\"], ascending=[True, False])))\n# #     print(\"kendall_tau score :\", optuna_score)\n    \n    if optuna_score < output_container[\"score\"]:\n        print(\"found the best !\")\n        if best_ntrees[0] is not None:\n            print(\"number of trees :\", best_ntrees)\n        output_container[\"model\"] = model\n        tmp = val_pred.loc[np.unique(val_groups)]\n#         tmp.loc[tmp[\"cell_type\"] == 1.0, \"rank\"] = inv_boxcox1p(model.predict(val_pool), boxcox_lambda)\n        tmp.loc[tmp[\"cell_type\"] == 1.0, \"rank\"] = model.predict(val_pool)\n        output_container[\"pred\"] = tmp\n        output_container[\"score\"] = optuna_score\n\n    return optuna_score","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-12T07:39:15.153097Z","iopub.execute_input":"2022-07-12T07:39:15.153454Z","iopub.status.idle":"2022-07-12T07:39:15.203197Z","shell.execute_reply.started":"2022-07-12T07:39:15.153415Z","shell.execute_reply":"2022-07-12T07:39:15.202182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# fixed parameter setting\n\nntrees = 5000\neta = 5e-3\n\n# ntrees = 1000\n# eta = 5e-3\n\n# ntrees = 100\n# eta = 1e-2","metadata":{"execution":{"iopub.status.busy":"2022-07-12T07:39:15.205761Z","iopub.execute_input":"2022-07-12T07:39:15.206438Z","iopub.status.idle":"2022-07-12T07:39:15.217815Z","shell.execute_reply.started":"2022-07-12T07:39:15.206400Z","shell.execute_reply":"2022-07-12T07:39:15.217083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_full_ori = easyIO(None, \"../input/catboost-ranker-data-preparation/df_full/raw/df_full_ori.pkl\", \"r\")\ndf_orders = easyIO(None, \"../input/catboost-ranker-data-preparation/df_full/raw/df_orders.pkl\", \"r\")","metadata":{"execution":{"iopub.status.busy":"2022-07-12T07:39:15.219217Z","iopub.execute_input":"2022-07-12T07:39:15.219552Z","iopub.status.idle":"2022-07-12T07:39:18.898389Z","shell.execute_reply.started":"2022-07-12T07:39:15.219516Z","shell.execute_reply":"2022-07-12T07:39:18.897628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import seaborn as sns\n# from scipy.special import boxcox1p\n# from scipy.special import inv_boxcox1p\n# boxcox_lambda = 0.75","metadata":{"execution":{"iopub.status.busy":"2022-07-12T07:39:18.899794Z","iopub.execute_input":"2022-07-12T07:39:18.900070Z","iopub.status.idle":"2022-07-12T07:39:18.906185Z","shell.execute_reply.started":"2022-07-12T07:39:18.900032Z","shell.execute_reply":"2022-07-12T07:39:18.903547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(12, 6))\ngraph = sns.histplot(df_full_ori.loc[df_full_ori[\"cell_type\"] == 1.0, \"rank\"], bins=50, color=\"green\")\nplt.title(\"Original distribution of markdown cells' rank (code cell is scaled 0 ~ 100)\", fontsize=15, fontweight=\"bold\", pad=15)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T07:39:18.907621Z","iopub.execute_input":"2022-07-12T07:39:18.908110Z","iopub.status.idle":"2022-07-12T07:39:19.392265Z","shell.execute_reply.started":"2022-07-12T07:39:18.908063Z","shell.execute_reply":"2022-07-12T07:39:19.391535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import seaborn as sns\n\n# fig, ax = plt.subplots(figsize=(12, 6))\n# graph = sns.histplot(boxcox1p(df_full_ori.loc[df_full_ori[\"cell_type\"] == 1.0, \"rank\"], boxcox_lambda), bins=50, color=\"orange\")\n# plt.title(\"Boxcox Transformation\", fontsize=15, fontweight=\"bold\", pad=15)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T07:39:19.393735Z","iopub.execute_input":"2022-07-12T07:39:19.393996Z","iopub.status.idle":"2022-07-12T07:39:19.397641Z","shell.execute_reply.started":"2022-07-12T07:39:19.393960Z","shell.execute_reply":"2022-07-12T07:39:19.396876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fold_metrics = []\n\nval_pred = df_full_ori[[\"cell_type\", \"rank\"]].copy()\nval_pred.loc[(val_pred[\"cell_type\"] == 1.0), \"rank\"] = 0\n\nn_folds = 5\n\nstart_time_training = time()\n# fold training\nfor fold in range(n_folds):\n    print(\"\\n===== Fold\", fold, \"=====\\n\")\n    \n    createFolder(\"./models/\")\n    start_mem = memory_usage()   \n    \n    # train only markdown cell\n    tmp_obj = np.load(\"../input/catboost-ranker-data-preparation/df_full/fold/train_fold_\" + str(fold) + \".npz\", allow_pickle=True)\n    train_x = tmp_obj[\"x\"]\n#     train_y = boxcox1p(tmp_obj[\"y\"], boxcox_lambda)\n    train_y = tmp_obj[\"y\"]\n    train_groups = tmp_obj[\"groups\"]\n    \n    tmp_obj = np.load(\"../input/catboost-ranker-data-preparation/df_full/fold/val_fold_\" + str(fold) + \".npz\", allow_pickle=True)\n    val_x = tmp_obj[\"x\"]\n#     val_y = boxcox1p(tmp_obj[\"y\"], boxcox_lambda)\n    val_y = tmp_obj[\"y\"]\n    val_groups = tmp_obj[\"groups\"]\n    del tmp_obj\n    \n    output_container = {\"model\": None, \"pred\": None, \"score\": np.inf}\n    optuna_direction = \"minimize\"\n    optuna_trials = 1000\n    optuna_timout = int(10 * 3600 / n_folds)\n    optuna_study = create_study(direction=optuna_direction, sampler=TPESampler())\n    \n    best_ntrees = [0]\n    optuna_study.optimize(\n        lambda trial: optuna_objective_function(\n            trial, fold, train_x, train_y, train_groups, val_x, val_y, val_groups, categoIdx=None, model_name=\"CAT_GBM\", output_container=output_container,\n            ntrees=ntrees, eta=eta, best_ntrees=best_ntrees\n        ),\n        n_jobs=1, n_trials=optuna_trials, timeout=optuna_timout\n    )\n    \n    best_params = copy.deepcopy(optuna_study.best_params)\n    output_container[\"model\"].save_model(\"./models/model_fold_\" + str(fold) + \".cbm\", format=\"cbm\")\n    if best_ntrees[0] is not None:\n        best_params[\"ntrees\"] = best_ntrees[0]\n    print(\"fold\", fold, \"best params :\", best_params)\n#     easyIO(best_params, \"./models/params_fold_\" + str(fold) + \".pkl\", \"w\")\n\n    val_pred.loc[np.unique(val_groups)] = output_container[\"pred\"]\n    fold_metrics.append(kendall_tau(df_orders.loc[np.unique(val_groups)].sort_index(), convert_rank_ids(val_pred.loc[np.unique(val_groups)].sort_values([\"id\", \"rank\"], ascending=[True, False]))))\n#     fold_metrics.append(output_container[\"score\"])\n    print(\"fold\", fold, \"kendall_tau score :\", fold_metrics[-1])\n    \n    gc.collect()\n    end_mem = memory_usage()\n    print(\"@Memory leaked :\", end_mem - start_mem, \"\\n\")\n    \nend_time_training = time()","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-12T07:39:19.400623Z","iopub.execute_input":"2022-07-12T07:39:19.400911Z","iopub.status.idle":"2022-07-12T07:42:57.736559Z","shell.execute_reply.started":"2022-07-12T07:39:19.400873Z","shell.execute_reply":"2022-07-12T07:42:57.733943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_pred = val_pred.sort_values([\"id\", \"rank\"], ascending=[True, False])\n\nprint(\"sample prediction\")\nval_pred.loc[np.unique(val_groups)[0]].iloc[:30]","metadata":{"execution":{"iopub.status.busy":"2022-07-12T07:42:57.737931Z","iopub.status.idle":"2022-07-12T07:42:57.738859Z","shell.execute_reply.started":"2022-07-12T07:42:57.738580Z","shell.execute_reply":"2022-07-12T07:42:57.738607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for idx, value in enumerate(fold_metrics):\n    print(\"fold\", idx, \"score :\", value)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T07:42:57.746836Z","iopub.status.idle":"2022-07-12T07:42:57.747513Z","shell.execute_reply.started":"2022-07-12T07:42:57.747266Z","shell.execute_reply":"2022-07-12T07:42:57.747293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"fold average score :\", np.mean([i for i in fold_metrics]))","metadata":{"execution":{"iopub.status.busy":"2022-07-12T07:42:57.748880Z","iopub.status.idle":"2022-07-12T07:42:57.749529Z","shell.execute_reply.started":"2022-07-12T07:42:57.749279Z","shell.execute_reply":"2022-07-12T07:42:57.749305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(12, 6))\ngraph = sns.histplot(val_pred.loc[val_pred[\"cell_type\"] == 1.0, \"rank\"], bins=50, color=\"green\")\nplt.title(\"Predicted distribution of markdown cells' rank (code cell is scaled 0 ~ 100)\", fontsize=15, fontweight=\"bold\", pad=15)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T07:42:57.750838Z","iopub.status.idle":"2022-07-12T07:42:57.751458Z","shell.execute_reply.started":"2022-07-12T07:42:57.751225Z","shell.execute_reply":"2022-07-12T07:42:57.751250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# fig, ax = plt.subplots(figsize=(12, 6))\n# graph = sns.histplot(boxcox1p(val_pred.loc[val_pred[\"cell_type\"] == 1.0, \"rank\"], boxcox_lambda), bins=50, color=\"orange\")\n# plt.title(\"Boxcox Transformation\", fontsize=15, fontweight=\"bold\", pad=15)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T07:42:57.752639Z","iopub.status.idle":"2022-07-12T07:42:57.753281Z","shell.execute_reply.started":"2022-07-12T07:42:57.753045Z","shell.execute_reply":"2022-07-12T07:42:57.753069Z"},"trusted":true},"execution_count":null,"outputs":[]}]}