{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 🧑‍💻 __AI4Code Submission: TFRankEncoder__\n---\n### <a href='#postprocessing'> 🏤 Postprocessing </a> ","metadata":{"papermill":{"duration":0.007485,"end_time":"2022-07-05T16:21:11.438590","exception":false,"start_time":"2022-07-05T16:21:11.431105","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# todo: Change max marknown to 128??","metadata":{"execution":{"iopub.execute_input":"2022-07-05T16:21:11.451773Z","iopub.status.busy":"2022-07-05T16:21:11.451352Z","iopub.status.idle":"2022-07-05T16:21:11.456395Z","shell.execute_reply":"2022-07-05T16:21:11.455731Z"},"papermill":{"duration":0.014392,"end_time":"2022-07-05T16:21:11.458805","exception":false,"start_time":"2022-07-05T16:21:11.444413","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Global Hyperparameters ##\nVALID_FOLD = 0\nHIDE_LB_SCORE = False","metadata":{"execution":{"iopub.execute_input":"2022-07-05T16:21:11.471130Z","iopub.status.busy":"2022-07-05T16:21:11.470801Z","iopub.status.idle":"2022-07-05T16:21:11.479129Z","shell.execute_reply":"2022-07-05T16:21:11.478468Z"},"papermill":{"duration":0.016201,"end_time":"2022-07-05T16:21:11.480750","exception":false,"start_time":"2022-07-05T16:21:11.464549","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nimport sys\nsys.path.extend(['/kaggle/input/github-ai4code/ai4code', 'ai4code'])\nsys.path.extend(['/kaggle/input/github-ai4code/fast-nlp', 'fast-nlp'])\nsys.path.append('/kaggle/input/omegaconf')\n\nimport src\nfrom src import *\n\nimport ai4c\nimport ai4c.process_df\n#import ai4c.tf_rankencoder\n\nimport ai4c.submission\nfrom ai4c.submission import *\n#import ai4c.submission.tf_rankencoder","metadata":{"execution":{"iopub.execute_input":"2022-07-05T16:21:11.493047Z","iopub.status.busy":"2022-07-05T16:21:11.492809Z","iopub.status.idle":"2022-07-05T16:21:52.840213Z","shell.execute_reply":"2022-07-05T16:21:52.839259Z"},"papermill":{"duration":41.355758,"end_time":"2022-07-05T16:21:52.842127","exception":false,"start_time":"2022-07-05T16:21:11.486369","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow_addons as tfa \nimport tensorflow as tf\n\ndef enable_mixed_precision(): \n    from tensorflow.keras.mixed_precision import experimental as mixed_precision\n    policy = mixed_precision.Policy('mixed_float16')\n    mixed_precision.set_policy(policy)\n\nenable_mixed_precision()\n# tf.config.optimizer.set_jit(True)","metadata":{"execution":{"iopub.execute_input":"2022-07-05T16:21:52.856343Z","iopub.status.busy":"2022-07-05T16:21:52.855395Z","iopub.status.idle":"2022-07-05T16:22:00.841017Z","shell.execute_reply":"2022-07-05T16:22:00.840216Z"},"papermill":{"duration":7.994734,"end_time":"2022-07-05T16:22:00.843125","exception":false,"start_time":"2022-07-05T16:21:52.848391","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cell_df = ai4c.process_df.build_cell_df(\n    '/kaggle/input/AI4Code/test',\n    max_cell_char_count=10000,\n)\nnotebooks_df = ai4c.process_df.build_notebooks_df(cell_df)\n\nif IS_INTERACTIVE:\n    notebooks_df = pd.read_csv('/kaggle/input/ai4code-dataframes/notebooks_df.csv')\n    notebooks_df = notebooks_df[notebooks_df.notebook_fold==VALID_FOLD].sample(10000)\ntotal_notebook_count = len(notebooks_df)","metadata":{"execution":{"iopub.execute_input":"2022-07-05T16:22:00.857446Z","iopub.status.busy":"2022-07-05T16:22:00.857156Z","iopub.status.idle":"2022-07-05T16:22:01.074665Z","shell.execute_reply":"2022-07-05T16:22:01.073758Z"},"papermill":{"duration":0.22638,"end_time":"2022-07-05T16:22:01.076429","exception":false,"start_time":"2022-07-05T16:22:00.850049","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_backbone_and_tokenizer(backbone_folder, backbone_name, tf_backbone=True): \n    backbone_code = backbone_name.replace('/', '_')\n    backbone_dir = f'/kaggle/input/{backbone_folder}/{backbone_code}'\n    print(f'Loading {backbone_name} from {backbone_dir}')\n    if tf_backbone:\n        backbone = transformers.TFAutoModel.from_pretrained(backbone_dir, from_pt=True)\n    else:\n        backbone = transformers.AutoModel.from_pretrained(backbone_dir)\n    tokenizer = transformers.AutoTokenizer.from_pretrained(backbone_dir)\n    return backbone, tokenizer\n\ndef build_hidden_layer(hidden_layer_units=[], hidden_dropout=0.10, activation_str='mish', l2_regularization=0, name='hidden_layer'): \n    if not hidden_layer_units: \n        return tf.keras.layers.Lambda(lambda x: x)\n    activation_fn = {'mish': tfa.activations.mish, None: None, 'gelu': tf.keras.activations.gelu}[activation_str]\n    hidden_layers = []\n    for units in hidden_layer_units: \n        hidden_layers.append(tf.keras.layers.Dropout(hidden_dropout))\n        hidden_layers.append(tf.keras.layers.Dense(\n            units=units, \n            activation=activation_fn, \n            kernel_regularizer=tf.keras.regularizers.l2(l2=l2_regularization)\n        ))\n    return tf.keras.Sequential(hidden_layers, name=name)","metadata":{"execution":{"iopub.execute_input":"2022-07-05T16:22:01.091051Z","iopub.status.busy":"2022-07-05T16:22:01.090453Z","iopub.status.idle":"2022-07-05T16:22:01.171536Z","shell.execute_reply":"2022-07-05T16:22:01.170828Z"},"papermill":{"duration":0.090061,"end_time":"2022-07-05T16:22:01.173144","exception":false,"start_time":"2022-07-05T16:22:01.083083","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def prune_cell_tokens(cell_token_ids, max_seq_len):\n    \"\"\"\n    Prunes cells that take too many tokens to fit in max_seq_len.\n    \"\"\"\n    cell_token_counts = [len(token_ids) for token_ids in cell_token_ids]\n    total_number_of_cells = len(cell_token_counts)\n    total_tokens_to_prune = max(sum(cell_token_counts)-max_seq_len, 0)\n\n    tokens_to_prune_per_cell = [0]*total_number_of_cells\n    total_pruned_tokens = 0\n    while total_tokens_to_prune > 0:\n        cur_max_cell_token_count = max(cell_token_counts)\n        second_max_cell_token_count = sorted(cell_token_counts)[-2]\n        for cell_idx, cell_token_count in enumerate(cell_token_counts):\n            if not cell_token_count == cur_max_cell_token_count: \n                continue\n            num_tokens_to_pop = min(cell_token_count-second_max_cell_token_count+1, total_tokens_to_prune)\n            tokens_to_prune_per_cell[cell_idx] += num_tokens_to_pop\n            total_pruned_tokens += num_tokens_to_pop\n            total_tokens_to_prune -= num_tokens_to_pop\n            cell_token_counts[cell_idx] -= num_tokens_to_pop\n            break\n    \n    # Prune the cell tokens\n    pruned_cell_token_ids = []\n    for cell_token_ids, num_tokens_to_pop in zip(cell_token_ids, tokens_to_prune_per_cell):\n        if num_tokens_to_pop == 0:\n            pruned_cell_token_ids.append(cell_token_ids)\n            continue\n        pruned_cell_token_ids.append(cell_token_ids[:-num_tokens_to_pop])\n    return pruned_cell_token_ids\n\n\ndef convert_to_features_tf_rankencoder(notebook, tokenizer, max_seq_len, max_markdown_seq_len, max_tokens_per_cell):\n    '''Converts a notebook to a dataset for training the model.'''\n\n    markdown_cell_sources = notebook['merged_markdown_cell_sources'].split(CELL_SEP)\n    code_cell_sources = notebook['merged_code_cell_sources'].split(CELL_SEP)\n\n    # Remove cells from the end of the notebook so that all cells have at least one representative token\n    max_markdown_cells = max_markdown_seq_len//2\n    max_code_cells = (max_seq_len-max_markdown_seq_len)//2\n    if len(markdown_cell_sources) > max_markdown_cells:\n        markdown_cell_sources = markdown_cell_sources[:max_markdown_cells]\n    if len(code_cell_sources) > max_code_cells:\n        code_cell_sources = code_cell_sources[:max_code_cells]\n    markdown_cell_count = len(markdown_cell_sources)\n    code_cell_count = len(code_cell_sources)\n\n    max_tokens_per_markdown_cell = max(max_tokens_per_cell, max_markdown_seq_len//markdown_cell_count)\n    markdown_cell_token_ids = tokenizer(\n        markdown_cell_sources,\n        max_length=max_tokens_per_markdown_cell,\n        truncation=True,\n    )['input_ids']\n    markdown_cell_token_ids = prune_cell_tokens(markdown_cell_token_ids, max_markdown_seq_len)\n    total_markdown_cell_tokens = sum([len(token_ids) for token_ids in markdown_cell_token_ids])\n\n    max_code_seq_len = max_seq_len - total_markdown_cell_tokens\n    max_tokens_per_code_cell = max(max_tokens_per_cell, max_code_seq_len//code_cell_count)\n    code_cell_token_ids = tokenizer(\n        code_cell_sources, \n        max_length=max_tokens_per_code_cell, \n        truncation=True, \n    )['input_ids']\n    code_cell_token_ids = prune_cell_tokens(code_cell_token_ids, max_seq_len-total_markdown_cell_tokens)\n\n    # Merge the tokenized cells and create the model features\n    cell_token_ids = markdown_cell_token_ids + code_cell_token_ids\n    notebook_cell_count = len(cell_token_ids)\n\n    # Create the model features\n    cell_pct_ranks = [None]*notebook_cell_count\n    \n    input_ids = []\n    token_cell_ids, token_type_ids = [], []\n    for cur_cell_idx, cell_token_ids in enumerate(cell_token_ids):\n        token_count_for_cell = len(cell_token_ids)\n        if cur_cell_idx < markdown_cell_count:\n            token_cell_id = 1\n        else: \n            token_cell_id = 0\n        \n        input_ids += cell_token_ids\n        token_cell_ids += [token_cell_id] * token_count_for_cell\n        token_type_id = 0 if cur_cell_idx % 2 == 0 else 1\n        token_type_ids += [token_type_id] * token_count_for_cell\n\n    # Pad the features to match max_seq_len #\n    num_pad_tokens = max_seq_len-len(input_ids)\n    attention_mask = [1]*len(input_ids) + [0]*num_pad_tokens\n    input_ids += [0]*num_pad_tokens\n    token_type_ids += [-100]*num_pad_tokens\n    token_cell_ids += [-100]*num_pad_tokens\n    \n    # Check for bugs\n    assert len(input_ids) == len(attention_mask) == len(token_type_ids)== len(token_cell_ids) == max_seq_len\n    \n    # Build the feature dict for the input \n    notebook_features = {\n        'input_ids': input_ids, \n        'attention_mask': attention_mask,\n        'token_type_ids': token_type_ids,\n        'token_cell_ids': token_cell_ids,\n    }\n    return notebook_features\n\n\ndef df_to_dataset_tfrankencoder(df, tokenizer, max_seq_len, max_markdown_seq_len, max_tokens_per_cell):\n    convert_to_features = partial(\n        convert_to_features_tf_rankencoder, \n        tokenizer=tokenizer, \n        max_seq_len=max_seq_len,\n        max_markdown_seq_len=max_markdown_seq_len,\n        max_tokens_per_cell=max_tokens_per_cell,\n    )\n    raw_dataset = datasets.Dataset.from_pandas(df)\n    processed_dataset = raw_dataset.map(\n        convert_to_features,\n        remove_columns=raw_dataset.column_names\n    )\n    processed_dataset.set_format(type='numpy')\n    return processed_dataset\n\n\ndef build_model_tf_rankencoder(backbone, max_seq_len, hidden_layer_units, hidden_layer_activation):\n    input_ids = tf.keras.Input(shape=(max_seq_len,), dtype=tf.int32, name='input_ids')\n    attention_mask = tf.keras.Input(shape=(max_seq_len,), dtype=tf.float32, name='attention_mask')\n    token_type_ids = tf.keras.Input(shape=(max_seq_len,), dtype=tf.int32, name='token_type_ids')\n    model_inputs = [input_ids, attention_mask, token_type_ids]\n    \n    hidden_layers = build_hidden_layer(\n        hidden_layer_units = hidden_layer_units, \n        hidden_dropout = backbone.config.hidden_dropout_prob, \n        activation_str = hidden_layer_activation,\n    )\n    ranker = tf.keras.Sequential([hidden_layers, tf.keras.layers.Reshape((max_seq_len,))], name='token_labels')\n\n    backbone_outputs = backbone(\n        input_ids=input_ids,\n        attention_mask=attention_mask,\n        token_type_ids=token_type_ids,\n        training=False,\n    )\n    x = backbone_outputs.last_hidden_state\n    token_preds = ranker(x)\n    return tf.keras.Model(inputs=model_inputs, outputs=token_preds)\n\ndef convert_hf_dataset_to_test_ds(dataset, batch_size):\n    dataset.set_format(type='numpy')\n\n    input_id_ds = tf.data.Dataset.from_tensor_slices(dataset['input_ids'].astype(np.int32))\n    input_mask_ds = tf.data.Dataset.from_tensor_slices(dataset['attention_mask'].astype(np.int32))\n    token_type_ids_ds = tf.data.Dataset.from_tensor_slices(dataset['token_type_ids'].astype(np.int32))\n    input_ds = tf.data.Dataset.zip((input_id_ds, input_mask_ds, token_type_ids_ds))\n\n    ds = tf.data.Dataset.zip((input_ds, input_ds))\n    return ds.batch(batch_size).prefetch(tf.data.AUTOTUNE)\n\ndef process_dense_predictions_tfrankencoder(all_notebook_token_preds, dataset, df):\n    '''Process notebook token predictions to give predicted cell ranks'''\n    # TODO: Debug more throughly\n    \n    # Store in numpy array for fast access\n    all_notebooks_token_type_ids = dataset['token_type_ids']\n    all_notebooks_cell_ids = df.merged_cell_ids.values\n    \n    cell_id_to_rank_preds = defaultdict(list)\n    for notebook_idx in tqdm(range(len(df))):\n        notebook_token_type_ids = all_notebooks_token_type_ids[notebook_idx]\n        notebook_cell_ids = all_notebooks_cell_ids[notebook_idx].split(CELL_SEP)\n        notebook_token_preds = all_notebook_token_preds[notebook_idx]\n        \n        cur_cell_idx = 0\n        prev_token_type_id = 0\n        for token_pred, token_type_id in zip(notebook_token_preds, notebook_token_type_ids):\n            if token_type_id == -100:\n                continue\n            \n            if token_type_id == prev_token_type_id:\n                cur_cell_id = notebook_cell_ids[cur_cell_idx]\n                cell_id_to_rank_preds[cur_cell_id].append(token_pred)\n            else: \n                cur_cell_idx += 1\n            prev_token_type_id = token_type_id\n            \n    cell_id_to_rank_pred = {cell_id: sum(preds)/len(preds) for cell_id, preds in cell_id_to_rank_preds.items()}\n    return cell_id_to_rank_pred","metadata":{"execution":{"iopub.execute_input":"2022-07-05T16:22:01.187795Z","iopub.status.busy":"2022-07-05T16:22:01.187509Z","iopub.status.idle":"2022-07-05T16:22:01.291150Z","shell.execute_reply":"2022-07-05T16:22:01.290497Z"},"papermill":{"duration":0.112867,"end_time":"2022-07-05T16:22:01.292843","exception":false,"start_time":"2022-07-05T16:22:01.179976","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model Family #1: RoBERTa Large w. Max Seq Len 512\n---\n\n~20 mins for 20k notebooks","metadata":{"papermill":{"duration":0.006803,"end_time":"2022-07-05T16:22:01.306458","exception":false,"start_time":"2022-07-05T16:22:01.299655","status":"completed"},"tags":[]}},{"cell_type":"code","source":"ROBERTA_LARGE_PREDS = defaultdict(list)","metadata":{"execution":{"iopub.execute_input":"2022-07-05T16:22:01.320590Z","iopub.status.busy":"2022-07-05T16:22:01.319958Z","iopub.status.idle":"2022-07-05T16:22:01.393914Z","shell.execute_reply":"2022-07-05T16:22:01.393247Z"},"papermill":{"duration":0.082805,"end_time":"2022-07-05T16:22:01.395613","exception":false,"start_time":"2022-07-05T16:22:01.312808","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%hyperparameters\n\n# 10k notebooks: 3:25 to tokenize + 10 min to infer (float32)\n\nbackbone_name: 'roberta-large'\nbackbone_folder: 'nlp-backbones-roberta'\n\nattention_probs_dropout_prob: 0.00\nhidden_dropout_prob: 0.10\n\nmax_seq_len: 512\nhidden_layer_units: [256, 16, 1]\nhidden_layer_activation: 'gelu'\n\nmax_markdown_seq_len: 256\nmax_tokens_per_cell: 128\n\nbatch_size: 256\n\nmodel_weights_folder: 'ai4code-wandb-tfrankencoder-roberta-large'\n\nmodel_weight_files: [\n    'roberta-large-tau875605-tau16_793-seq512-fold0.h5',\n]\n\n# Average kendall tau for notebooks with 4+ cells is: 0.8603496112505171\n# Average kendall tau for notebooks with 16+ cells is: 0.7934952217653062\n# Average kendall tau for notebooks with 32+ cells is: 0.7074705906869438\n# Average kendall tau for notebooks with 64+ cells is: 0.5910284344313729\n# Average kendall tau for notebooks with 128+ cells is: 0.4454556141475093\n# Average Kendall Tau: 0.87574618687465411","metadata":{"execution":{"iopub.execute_input":"2022-07-05T16:22:01.409828Z","iopub.status.busy":"2022-07-05T16:22:01.409201Z","iopub.status.idle":"2022-07-05T16:22:01.492208Z","shell.execute_reply":"2022-07-05T16:22:01.491541Z"},"papermill":{"duration":0.09211,"end_time":"2022-07-05T16:22:01.494032","exception":false,"start_time":"2022-07-05T16:22:01.401922","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with tf.device('/gpu:0'):\n    backbone_code = HP.backbone_name.replace('/', '_')\n    backbone_dir = f'/kaggle/input/{HP.backbone_folder}/{backbone_code}'\n    backbone = transformers.TFAutoModel.from_pretrained(\n        backbone_dir, \n        from_pt=True,\n        hidden_dropout_prob=HP.hidden_dropout_prob,\n        attention_probs_dropout_prob=HP.attention_probs_dropout_prob,\n    )\n    tokenizer = transformers.AutoTokenizer.from_pretrained(backbone_dir)\n    \n    model = build_model_tf_rankencoder(\n        backbone=backbone,\n        max_seq_len=HP.max_seq_len,\n        hidden_layer_units=HP.hidden_layer_units,\n        hidden_layer_activation=HP.hidden_layer_activation,\n    )\n\ndf = notebooks_df[notebooks_df.markdown_cell_count < 4]\ntest_dataset = df_to_dataset_tfrankencoder(\n    df=df,\n    tokenizer=tokenizer,\n    max_seq_len=HP.max_seq_len,\n    max_markdown_seq_len=HP.max_markdown_seq_len,\n    max_tokens_per_cell=HP.max_tokens_per_cell,\n)\n\nfor weights_file in tqdm(HP.model_weight_files):\n    model.load_weights(f'/kaggle/input/{HP.model_weights_folder}/{weights_file}')\n    test_ds = convert_hf_dataset_to_test_ds(test_dataset, batch_size=HP.batch_size)\n    \n    with tf.device('/gpu:0'):\n        all_notebook_token_preds = model.predict(test_ds, verbose=True)\n    \n    cell_id_to_pred_rank = process_dense_predictions_tfrankencoder(all_notebook_token_preds, test_dataset, df)\n    for cell_id, pred in cell_id_to_pred_rank.items():\n        ROBERTA_LARGE_PREDS[cell_id].append(pred)\n    \n    tf.keras.backend.clear_session()\n    gc.collect()","metadata":{"execution":{"iopub.execute_input":"2022-07-05T16:22:01.507886Z","iopub.status.busy":"2022-07-05T16:22:01.507465Z","iopub.status.idle":"2022-07-05T16:22:49.390866Z","shell.execute_reply":"2022-07-05T16:22:49.390036Z"},"papermill":{"duration":47.892246,"end_time":"2022-07-05T16:22:49.392655","exception":false,"start_time":"2022-07-05T16:22:01.500409","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROBERTA_LARGE_PREDS = {\n    cell_id: sum(cell_preds)/len(cell_preds) \n    for cell_id, cell_preds in ROBERTA_LARGE_PREDS.items()\n} ","metadata":{"execution":{"iopub.execute_input":"2022-07-05T16:22:49.408564Z","iopub.status.busy":"2022-07-05T16:22:49.407978Z","iopub.status.idle":"2022-07-05T16:22:49.495608Z","shell.execute_reply":"2022-07-05T16:22:49.494916Z"},"papermill":{"duration":0.09722,"end_time":"2022-07-05T16:22:49.497263","exception":false,"start_time":"2022-07-05T16:22:49.400043","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Postprocessing\n---\n\n<a name='postprocessing'>","metadata":{"papermill":{"duration":0.007073,"end_time":"2022-07-05T16:22:49.511847","exception":false,"start_time":"2022-07-05T16:22:49.504774","status":"completed"},"tags":[]}},{"cell_type":"code","source":"cell_id_to_pred_rank = ROBERTA_LARGE_PREDS","metadata":{"execution":{"iopub.execute_input":"2022-07-05T16:22:49.527155Z","iopub.status.busy":"2022-07-05T16:22:49.526888Z","iopub.status.idle":"2022-07-05T16:22:49.603996Z","shell.execute_reply":"2022-07-05T16:22:49.603308Z"},"papermill":{"duration":0.086973,"end_time":"2022-07-05T16:22:49.605886","exception":false,"start_time":"2022-07-05T16:22:49.518913","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predicted_cell_orders = []\nfor notebook_idx in range(len(notebooks_df)):\n    cell_ids = notebooks_df.iloc[notebook_idx].merged_cell_ids.split(CELL_SEP)\n    pred_cell_ranks = [\n        cell_id_to_pred_rank.get(cell_id, random.random()) \n        for cell_idx, cell_id in enumerate(cell_ids)\n    ]\n    ordered_cell_ids_list = [cell_id for cell_pred, cell_id in sorted(zip(pred_cell_ranks, cell_ids), key=lambda pairs: pairs[0])]\n    ordered_cell_ids = ' '.join(ordered_cell_ids_list)\n    predicted_cell_orders.append(ordered_cell_ids)\nnotebooks_df['cell_order'] = predicted_cell_orders\n\nif HIDE_LB_SCORE: \n    notebooks_df.cell_order = notebooks_df.cell_order.apply(\n        lambda cell_order: ' '.join(cell_order.split()[::-1])\n    )\nnotebooks_df['id'] = notebooks_df.notebook_id\nnotebooks_df[['id', 'cell_order']].to_csv('/kaggle/working/submission.csv', index=False)","metadata":{"execution":{"iopub.execute_input":"2022-07-05T16:22:49.621454Z","iopub.status.busy":"2022-07-05T16:22:49.621196Z","iopub.status.idle":"2022-07-05T16:22:49.706316Z","shell.execute_reply":"2022-07-05T16:22:49.705644Z"},"papermill":{"duration":0.095043,"end_time":"2022-07-05T16:22:49.708399","exception":false,"start_time":"2022-07-05T16:22:49.613356","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 👽 Paranoid Validation\n---","metadata":{"papermill":{"duration":0.006933,"end_time":"2022-07-05T16:22:49.722483","exception":false,"start_time":"2022-07-05T16:22:49.715550","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# all_notebook_cell_pct_ranks = notebooks_df.merged_cell_pct_ranks.values\n# all_notebook_cell_ids = notebooks_df.merged_cell_ids.values\n# all_notebook_pred_cell_ids = notebooks_df.cell_order.values\n\n# all_notebook_taus = []\n# for notebook_idx in range(len(notebooks_df)):\n#     true_cell_ranks = [float(rank) for rank in all_notebook_cell_pct_ranks[notebook_idx].split(CELL_SEP)]\n    \n#     pred_cell_ids = all_notebook_pred_cell_ids[notebook_idx].split()\n#     cell_id_to_pred = {cell_id: pred for pred, cell_id in enumerate(pred_cell_ids)}\n#     pred_cell_ranks = [cell_id_to_pred[cell_id] for cell_id in all_notebook_cell_ids[notebook_idx].split(CELL_SEP)]\n    \n#     notebook_tau = scipy.stats.kendalltau(true_cell_ranks, pred_cell_ranks, method='asymptotic')[0]\n#     all_notebook_taus.append(notebook_tau)\n\n# notebooks_df['kendall_tau'] = all_notebook_taus\n# for cell_cutoff in [4, 16, 32, 64, 128]:\n#     tau = notebooks_df[notebooks_df.markdown_cell_count>=cell_cutoff].kendall_tau.mean()\n#     print(f\"Average kendall tau for notebooks with {colored(cell_cutoff, 'yellow')}+ cells is: {colored(tau, 'blue')}\")\n# print(f'Average Kendall Tau:', colored(str(notebooks_df.kendall_tau.mean()), 'red'))","metadata":{"execution":{"iopub.execute_input":"2022-07-05T16:22:49.737919Z","iopub.status.busy":"2022-07-05T16:22:49.737667Z","iopub.status.idle":"2022-07-05T16:22:49.813927Z","shell.execute_reply":"2022-07-05T16:22:49.813079Z"},"papermill":{"duration":0.08592,"end_time":"2022-07-05T16:22:49.815681","exception":false,"start_time":"2022-07-05T16:22:49.729761","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# cell_id_to_cell_source = {}\n# for merged_cell_ids, merged_cell_sources in zip(notebooks_df.merged_cell_ids, notebooks_df.merged_cell_sources):\n#     for cell_id, cell_source in zip(merged_cell_ids.split(CELL_SEP), merged_cell_sources.split(CELL_SEP)):\n#         cell_id_to_cell_source[cell_id] = cell_source\n        \n# for notebook_idx in range(len(notebooks_df)):\n#     pred_cell_orders = notebooks_df.iloc[notebook_idx].cell_order.split()\n#     for cell_id in pred_cell_orders:\n#         print('cell_id:', cell_id)\n#         print(cell_id_to_cell_source[cell_id])\n#         print('-'*100)\n#     print('\\n\\n\\n')\n#     print('0'*100)\n#     print('\\n\\n\\n')","metadata":{"execution":{"iopub.execute_input":"2022-07-05T16:22:49.831197Z","iopub.status.busy":"2022-07-05T16:22:49.830599Z","iopub.status.idle":"2022-07-05T16:22:49.905176Z","shell.execute_reply":"2022-07-05T16:22:49.904542Z"},"papermill":{"duration":0.08402,"end_time":"2022-07-05T16:22:49.906767","exception":false,"start_time":"2022-07-05T16:22:49.822747","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# i = random.randint(0, len(test_dataset)-1)\n# print('Index:', i)\n# input_ids, cell_ids = test_dataset['input_ids'][i], test_dataset['cell_ids'][i]\n# token_cell_types = test_dataset['token_cell_types'][i]\n# cell_ids = cell_ids.split(SEP)\n\n# notebook_df = df[df.cell_id.isin(cell_ids)]\n# notebook_id = notebook_df.index.unique()[0]\n# print('Notebook Id:', notebook_id, '\\n\\n')\n\n# print(tokenizer.decode(input_ids))\n\n# train_notebook_df = train.loc[notebook_id].sort_values(by='pct_rank')\n# train_notebook_df\n\n\n# from termcolor import colored\n\n# cell_idx = 0\n# for token_idx, (token_id, token_cell_type) in enumerate(zip(input_ids, token_cell_types)):\n#     if token_cell_type is None:\n#         continue\n#     print(colored('-', 'red')*100)\n#     print(token_cell_type)\n    \n#     for next_token_idx, (token_id, token_cell_type) in enumerate(zip(input_ids, token_cell_types)):\n#         if token_cell_type is None or next_token_idx <= token_idx:\n#             continue\n#         break\n#     print('token_idx:', token_idx, next_token_idx)\n        \n#     cell_id = cell_ids[cell_idx]\n#     print('Cell ID:', cell_id)\n#     print('Prediction:', CELL_ID_TO_PRED[cell_id])\n#     print('Label:', train_notebook_df[train_notebook_df.source.apply(lambda x: text in x)].pct_rank.iloc[0])\n    \n#     print(tokenizer.decode(input_ids[token_idx:next_token_idx]))\n#     text = tokenizer.decode(input_ids[token_idx+2:next_token_idx-1])\n    \n    \n    \n#     cell_idx += 1","metadata":{"execution":{"iopub.execute_input":"2022-07-05T16:22:49.922421Z","iopub.status.busy":"2022-07-05T16:22:49.922175Z","iopub.status.idle":"2022-07-05T16:22:49.996396Z","shell.execute_reply":"2022-07-05T16:22:49.995722Z"},"papermill":{"duration":0.083796,"end_time":"2022-07-05T16:22:49.998147","exception":false,"start_time":"2022-07-05T16:22:49.914351","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Temp Public Notebook\n---","metadata":{"papermill":{"duration":0.006926,"end_time":"2022-07-05T16:22:50.012097","exception":false,"start_time":"2022-07-05T16:22:50.005171","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# import glob\n# import os\n# from typing import List\n\n# import numpy as np\n# import pandas as pd\n# import tensorflow as tf\n# import transformers\n# from tqdm.notebook import tqdm\n# BATCH_SIZE = 32\n# SLICES = 8\n# MD_MAX_LEN = 64\n# TOTAL_MAX_LEN = 512\n# STRATEGY = tf.distribute.get_strategy()\n# BASE_MODEL = \"../input/codebert-base/codebert-base\"\n# TOKENIZER = transformers.AutoTokenizer.from_pretrained(BASE_MODEL)\n# INPUT_PATH = \"../input/AI4Code\"\n# def read_notebook(path: str) -> pd.DataFrame:\n#     return (\n#         pd.read_json(path, dtype={\"cell_type\": \"category\", \"source\": \"str\"})\n#         .assign(id=os.path.basename(path).split(\".\")[0])\n#         .rename_axis(\"cell_id\")\n#     )\n\n\n# def clean_code(cell: str) -> str:\n#     return str(cell).replace(\"\\\\n\", \"\\n\")\n\n\n# def sample_cells(cells: List[str], n: int) -> List[str]:\n#     cells = [clean_code(cell) for cell in cells]\n#     if n >= len(cells):\n#         return cells\n#     else:\n#         results = []\n#         step = len(cells) / n\n#         idx = 0\n#         while int(np.round(idx)) < len(cells):\n#             results.append(cells[int(np.round(idx))])\n#             idx += step\n#         if cells[-1] not in results:\n#             results[-1] = cells[-1]\n#         return results\n\n\n# def get_features(df: pd.DataFrame) -> dict:\n#     features = {}\n#     for i, sub_df in tqdm(df.groupby(\"id\"), desc=\"Features\"):\n#         features[i] = {}\n#         total_md = sub_df[sub_df.cell_type == \"markdown\"].shape[0]\n#         code_sub_df = sub_df[sub_df.cell_type == \"code\"]\n#         total_code = code_sub_df.shape[0]\n#         codes = sample_cells(code_sub_df.source.values, 20)\n#         features[i][\"total_code\"] = total_code\n#         features[i][\"total_md\"] = total_md\n#         features[i][\"codes\"] = codes\n#     return features\n\n\n# def tokenize(df: pd.DataFrame, fts: dict) -> dict:\n#     input_ids = np.zeros((len(df), TOTAL_MAX_LEN), dtype=np.int32)\n#     attention_mask = np.zeros((len(df), TOTAL_MAX_LEN), dtype=np.int32)\n#     features = np.zeros((len(df),), dtype=np.float32)\n\n#     for i, row in tqdm(\n#         df.reset_index(drop=True).iterrows(), desc=\"Tokens\", total=len(df)\n#     ):\n#         row_fts = fts[row.id]\n\n#         inputs = TOKENIZER.encode_plus(\n#             row.source,\n#             None,\n#             add_special_tokens=True,\n#             max_length=MD_MAX_LEN,\n#             padding=\"max_length\",\n#             return_token_type_ids=True,\n#             truncation=True,\n#         )\n#         code_inputs = TOKENIZER.batch_encode_plus(\n#             [str(x) for x in row_fts[\"codes\"]] or [\"\"],\n#             add_special_tokens=True,\n#             max_length=23,\n#             padding=\"max_length\",\n#             truncation=True,\n#         )\n\n#         ids = inputs[\"input_ids\"]\n#         for x in code_inputs[\"input_ids\"]:\n#             ids.extend(x[:-1])\n#         ids = ids[:TOTAL_MAX_LEN]\n#         if len(ids) != TOTAL_MAX_LEN:\n#             ids = ids + [\n#                 TOKENIZER.pad_token_id,\n#             ] * (TOTAL_MAX_LEN - len(ids))\n\n#         mask = inputs[\"attention_mask\"]\n#         for x in code_inputs[\"attention_mask\"]:\n#             mask.extend(x[:-1])\n#         mask = mask[:TOTAL_MAX_LEN]\n#         if len(mask) != TOTAL_MAX_LEN:\n#             mask = mask + [\n#                 TOKENIZER.pad_token_id,\n#             ] * (TOTAL_MAX_LEN - len(mask))\n\n#         input_ids[i] = ids\n#         attention_mask[i] = mask\n#         features[i] = (\n#             row_fts[\"total_md\"] / (row_fts[\"total_md\"] + row_fts[\"total_code\"]) or 1\n#         )\n\n#     return {\n#         \"input_ids\": input_ids,\n#         \"attention_mask\": attention_mask,\n#         \"features\": features,\n#     }\n\n\n# def get_ranks(base: pd.Series, derived: List[str]) -> List[str]:\n#     return [base.index(d) for d in derived]\n\n\n# def get_dataset(\n#     input_ids: np.array,\n#     attention_mask: np.array,\n#     feature: np.array,\n# ) -> tf.data.Dataset:\n#     dataset = tf.data.Dataset.from_tensor_slices(\n#         {\"input_ids\": input_ids, \"attention_mask\": attention_mask, \"feature\": feature}\n#     )\n#     dataset = dataset.batch(BATCH_SIZE)\n#     return dataset.prefetch(tf.data.AUTOTUNE)\n\n\n# def get_model() -> tf.keras.Model:\n#     backbone = transformers.TFAutoModel.from_pretrained(BASE_MODEL)\n#     input_ids = tf.keras.layers.Input(\n#         shape=(TOTAL_MAX_LEN,),\n#         dtype=tf.int32,\n#         name=\"input_ids\",\n#     )\n#     attention_mask = tf.keras.layers.Input(\n#         shape=(TOTAL_MAX_LEN,),\n#         dtype=tf.int32,\n#         name=\"attention_mask\",\n#     )\n#     feature = tf.keras.layers.Input(\n#         shape=(1,),\n#         dtype=tf.float32,\n#         name=\"feature\",\n#     )\n#     x = backbone({\"input_ids\": input_ids, \"attention_mask\": attention_mask})[0]\n#     x = tf.concat([x[:, 0, :], feature], axis=1)\n#     outputs = tf.keras.layers.Dense(1, activation=\"linear\", dtype=\"float32\")(x)\n#     return tf.keras.Model(\n#         inputs=[input_ids, attention_mask, feature],\n#         outputs=outputs,\n#     )\n\n# paths = glob.glob(os.path.join(INPUT_PATH, \"test\", \"*.json\"))\n# df = (\n#     pd.concat([read_notebook(x) for x in tqdm(paths, desc=\"Concat\")])\n#     .set_index(\"id\", append=True)\n#     .swaplevel()\n#     .sort_index(level=\"id\", sort_remaining=False)\n# ).reset_index()\n# df[\"source\"] = df[\"source\"].str.slice(0, MD_MAX_LEN)\n# df[\"rank\"] = df.groupby([\"id\", \"cell_type\"]).cumcount()\n# df[\"pct_rank\"] = df.groupby([\"id\", \"cell_type\"])[\"rank\"].rank(pct=True)\n\n# fts = get_features(df)\n\n# with STRATEGY.scope():\n#     model = get_model()\n#     model.load_weights(\"../input/ai4code-codebert-weights/model_0.h5\")\n\n# predict = np.array([], dtype=np.float32)\n\n# for chunk in tqdm(\n#     np.array_split(df[df[\"cell_type\"] == \"markdown\"], SLICES), total=SLICES\n# ):\n#     if chunk.empty:\n#         continue\n\n#     data = tokenize(chunk, fts)\n\n#     dataset = get_dataset(data[\"input_ids\"], data[\"attention_mask\"], data[\"features\"])\n#     predict = np.r_[\n#         predict,\n#         model.predict(dataset).reshape(\n#             -1,\n#         ),\n#     ]\n# df.loc[df[\"cell_type\"] == \"markdown\", \"pct_rank\"] = predict\n# df = df.sort_values(\"pct_rank\").groupby(\"id\")[\"cell_id\"].apply(\" \".join)\n# df.name = \"cell_order\"\n# df.to_csv(\"submission.csv\")","metadata":{"execution":{"iopub.execute_input":"2022-07-05T16:22:50.028017Z","iopub.status.busy":"2022-07-05T16:22:50.027768Z","iopub.status.idle":"2022-07-05T16:22:50.107303Z","shell.execute_reply":"2022-07-05T16:22:50.106647Z"},"papermill":{"duration":0.08946,"end_time":"2022-07-05T16:22:50.108963","exception":false,"start_time":"2022-07-05T16:22:50.019503","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# MARKDOWN_COUNT_CUTOFF = 40\n\n# notebook_id_to_cell_count = notebooks_df.set_index('id').to_dict()['markdown_cell_count']\n# df = pd.read_csv('submission.csv')\n# df['markdown_cell_count'] = df['id'].map(notebook_id_to_cell_count)\n\n\n# mine = notebooks_df[notebooks_df.markdown_cell_count <= MARKDOWN_COUNT_CUTOFF]\n# public = df[df.markdown_cell_count > MARKDOWN_COUNT_CUTOFF]\n# sub = pd.concat([mine, public])\n# sub[['id', 'cell_order']].to_csv('/kaggle/working/submission.csv', index=False)","metadata":{"execution":{"iopub.execute_input":"2022-07-05T16:22:50.124082Z","iopub.status.busy":"2022-07-05T16:22:50.123837Z","iopub.status.idle":"2022-07-05T16:22:50.196098Z","shell.execute_reply":"2022-07-05T16:22:50.195436Z"},"papermill":{"duration":0.082023,"end_time":"2022-07-05T16:22:50.198041","exception":false,"start_time":"2022-07-05T16:22:50.116018","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]}]}