{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"tpu1vmV38","dataSources":[{"sourceId":51294,"databundleVersionId":6923401,"sourceType":"competition"},{"sourceId":6933839,"sourceType":"datasetVersion","datasetId":3981418}],"dockerImageVersionId":30587,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"https://www.kaggle.com/code/iafoss/rna-starter-0-186-lb","metadata":{}},{"cell_type":"markdown","source":"https://www.kaggle.com/code/pranshubahadur/esm2-rmdb-rna-dataset/notebook","metadata":{}},{"cell_type":"code","source":"import warnings\nwarnings.simplefilter('ignore')\n\nimport pandas as pd\npd.set_option('display.max_columns', 30)","metadata":{"execution":{"iopub.status.busy":"2023-12-06T04:56:33.392697Z","iopub.execute_input":"2023-12-06T04:56:33.393076Z","iopub.status.idle":"2023-12-06T04:56:34.296055Z","shell.execute_reply.started":"2023-12-06T04:56:33.393044Z","shell.execute_reply":"2023-12-06T04:56:34.295336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\n\n\nresolver = tf.distribute.cluster_resolver.TPUClusterResolver()\ntf.config.experimental_connect_to_cluster(resolver)\ntf.tpu.experimental.initialize_tpu_system(resolver)\nstrategy = tf.distribute.experimental.TPUStrategy(resolver)","metadata":{"execution":{"iopub.status.busy":"2023-12-06T04:56:34.297477Z","iopub.execute_input":"2023-12-06T04:56:34.297832Z","iopub.status.idle":"2023-12-06T04:56:55.4323Z","shell.execute_reply.started":"2023-12-06T04:56:34.297802Z","shell.execute_reply":"2023-12-06T04:56:55.431533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rmdb = pd.read_csv('/kaggle/input/rmdb-rna-mapping-database-2023-data/rmdb_data.v1.3.0.csv')\nrmdb","metadata":{"execution":{"iopub.status.busy":"2023-12-06T04:56:55.433307Z","iopub.execute_input":"2023-12-06T04:56:55.433567Z","iopub.status.idle":"2023-12-06T04:57:07.581961Z","shell.execute_reply.started":"2023-12-06T04:56:55.433538Z","shell.execute_reply":"2023-12-06T04:57:07.581009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rmdb_SN_filter = rmdb[rmdb.SN_filter == 1]\nerror_feats = [feat for feat in rmdb_SN_filter.columns if 'error' in feat]\nrmdb_SN_filter_no_error = rmdb_SN_filter.drop(columns = error_feats)","metadata":{"execution":{"iopub.status.busy":"2023-12-06T04:57:07.58393Z","iopub.execute_input":"2023-12-06T04:57:07.584244Z","iopub.status.idle":"2023-12-06T04:57:07.967098Z","shell.execute_reply.started":"2023-12-06T04:57:07.584214Z","shell.execute_reply":"2023-12-06T04:57:07.966202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# rmdb_SN_filter_no_error","metadata":{"execution":{"iopub.status.busy":"2023-12-06T04:57:07.968138Z","iopub.execute_input":"2023-12-06T04:57:07.96844Z","iopub.status.idle":"2023-12-06T04:57:07.972219Z","shell.execute_reply.started":"2023-12-06T04:57:07.968413Z","shell.execute_reply":"2023-12-06T04:57:07.971491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"experiment_type = rmdb_SN_filter_no_error.experiment_type.unique()\n\nrmdb_etype_2A3 = [experiment_type[0]]\nrmdb_etype_DMS = (experiment_type[1:]).tolist()","metadata":{"execution":{"iopub.status.busy":"2023-12-06T04:57:07.973261Z","iopub.execute_input":"2023-12-06T04:57:07.973516Z","iopub.status.idle":"2023-12-06T04:57:07.988871Z","shell.execute_reply.started":"2023-12-06T04:57:07.97349Z","shell.execute_reply":"2023-12-06T04:57:07.988219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# rmdb_etype_2A3 = ['1M7','NMIA','BzCN']\n# rmdb_etype_DMS = ['BzCN_cotx', 'DMS_cotx', 'DMS_M2_seq', 'DMS']","metadata":{"execution":{"iopub.status.busy":"2023-12-06T04:57:07.989784Z","iopub.execute_input":"2023-12-06T04:57:07.990047Z","iopub.status.idle":"2023-12-06T04:57:07.997372Z","shell.execute_reply.started":"2023-12-06T04:57:07.990021Z","shell.execute_reply":"2023-12-06T04:57:07.996707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_quick = pd.read_csv('/kaggle/input/stanford-ribonanza-rna-folding/train_data_QUICK_START.csv')\ntrain_quick_2A3 = train_quick[train_quick.experiment_type == '2A3_MaP'].reset_index(drop=True)\ntrain_quick_DMS = train_quick[train_quick.experiment_type == 'DMS_MaP'].reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2023-12-06T04:57:07.998228Z","iopub.execute_input":"2023-12-06T04:57:07.998453Z","iopub.status.idle":"2023-12-06T04:57:24.2099Z","shell.execute_reply.started":"2023-12-06T04:57:07.99843Z","shell.execute_reply":"2023-12-06T04:57:24.208975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rmdb_2A3 = rmdb_SN_filter_no_error[rmdb_SN_filter_no_error.experiment_type.isin(rmdb_etype_2A3)].reset_index(drop=True)\nrmdb_2A3 = pd.concat([train_quick_2A3, rmdb_2A3]).reset_index(drop=True)\n\nrmdb_DMS = rmdb_SN_filter_no_error[rmdb_SN_filter_no_error.experiment_type.isin(rmdb_etype_DMS)].reset_index(drop=True)\nrmdb_DMS = pd.concat([train_quick_DMS, rmdb_DMS]).reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2023-12-06T04:57:24.210987Z","iopub.execute_input":"2023-12-06T04:57:24.211288Z","iopub.status.idle":"2023-12-06T04:57:27.535467Z","shell.execute_reply.started":"2023-12-06T04:57:24.211257Z","shell.execute_reply":"2023-12-06T04:57:27.53451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_2A3 = rmdb_2A3.sequence\nrmdb_2A3 = rmdb_2A3.filter(regex='reactivity_[0-9]')\nrmdb_2A3 = rmdb_2A3.fillna(0)\nX_2A3 = X_2A3.iloc[rmdb_2A3.index].reset_index(drop=True)\nY_2A3 = rmdb_2A3.reset_index(drop=True)\n\nX_DMS = rmdb_DMS.sequence\nrmdb_DMS = rmdb_DMS.filter(regex='reactivity_[0-9]')\nrmdb_DMS = rmdb_DMS.fillna(0)\nX_DMS = X_DMS.iloc[rmdb_DMS.index].reset_index(drop=True)\nY_DMS = rmdb_DMS.reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2023-12-06T04:57:27.538129Z","iopub.execute_input":"2023-12-06T04:57:27.538425Z","iopub.status.idle":"2023-12-06T04:57:29.471629Z","shell.execute_reply.started":"2023-12-06T04:57:27.538396Z","shell.execute_reply":"2023-12-06T04:57:29.470647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\n\ndel train_quick\ndel train_quick_2A3\ndel train_quick_DMS\n\ndel rmdb\ndel rmdb_SN_filter\ndel error_feats\ndel rmdb_SN_filter_no_error\ndel rmdb_etype_2A3\ndel rmdb_etype_DMS\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-12-06T04:57:29.472711Z","iopub.execute_input":"2023-12-06T04:57:29.473023Z","iopub.status.idle":"2023-12-06T04:57:29.711981Z","shell.execute_reply.started":"2023-12-06T04:57:29.472997Z","shell.execute_reply":"2023-12-06T04:57:29.711137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras import Model\nfrom transformers import TFAutoModel\n\nclass RNA(Model):\n    \n    def __init__(self):\n        \n        super().__init__()\n        self.encoder = TFAutoModel.from_pretrained('AmelieSchreiber/esm2_t6_8M_UR50D_rna_binding_site_predictor')\n        self.dropout = tf.keras.layers.Dropout(0.2)\n        self.dense = tf.keras.layers.Dense(1)\n        \n    def call(self, x):\n        \n        x = self.encoder(x).last_hidden_state\n        \n        return tf.squeeze(self.dense(x), -1)","metadata":{"execution":{"iopub.status.busy":"2023-12-06T04:57:29.713002Z","iopub.execute_input":"2023-12-06T04:57:29.71327Z","iopub.status.idle":"2023-12-06T04:57:51.153723Z","shell.execute_reply.started":"2023-12-06T04:57:29.713246Z","shell.execute_reply":"2023-12-06T04:57:51.152741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def loss(X, Y):\n    \n    X_mask = tf.math.is_nan(X)\n    X = tf.where(X_mask, tf.zeros_like(X), X)\n    sum_mask = tf.math.reduce_sum(tf.where(X_mask, tf.zeros_like(X), tf.ones_like(X)))\n    loss = tf.math.abs(X - Y)\n    loss = tf.where(X_mask, tf.zeros_like(loss), loss)\n    loss = tf.math.reduce_sum(loss) / (sum_mask if sum_mask != 0.0 else 1.0)\n    \n    return loss","metadata":{"execution":{"iopub.status.busy":"2023-12-06T04:57:51.154884Z","iopub.execute_input":"2023-12-06T04:57:51.155396Z","iopub.status.idle":"2023-12-06T04:57:51.16065Z","shell.execute_reply.started":"2023-12-06T04:57:51.155366Z","shell.execute_reply":"2023-12-06T04:57:51.15995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text_vectorization = tf.keras.layers.TextVectorization(output_mode='int',\n                                                       ngrams=1,\n                                                       output_sequence_length=Y_DMS.shape[1],\n                                                       split='character',\n                                                       vocabulary=['a', 'c', 'g', 'u'])","metadata":{"execution":{"iopub.status.busy":"2023-12-06T04:57:51.161605Z","iopub.execute_input":"2023-12-06T04:57:51.161845Z","iopub.status.idle":"2023-12-06T04:57:51.201088Z","shell.execute_reply.started":"2023-12-06T04:57:51.16182Z","shell.execute_reply":"2023-12-06T04:57:51.200238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with strategy.scope():\n    \n    model_2A3 = RNA()\n    model_2A3.compile(optimizer=tf.keras.optimizers.Adam(learning_rate=5e-4), loss=loss)\n    \n    train_ds = tf.data.Dataset.from_tensor_slices((X_2A3.values, Y_2A3.values)) \\\n        .batch(128) \\\n        .map(lambda x, y: (text_vectorization(x), tf.clip_by_value(y, 0, 1)))\n    train_ds = train_ds.shuffle(train_ds.cardinality())\n    train_data = train_ds.take(int(len(train_ds) * 0.8))\n    validation_data = train_ds.skip(int(len(train_ds) * 0.8)).take(int(len(train_ds) * 0.2))\n\n    model_2A3.fit(train_data, validation_data=validation_data, epochs=25, batch_size=128)","metadata":{"execution":{"iopub.status.busy":"2023-12-06T04:57:51.202144Z","iopub.execute_input":"2023-12-06T04:57:51.202461Z","iopub.status.idle":"2023-12-06T06:20:59.602081Z","shell.execute_reply.started":"2023-12-06T04:57:51.202422Z","shell.execute_reply":"2023-12-06T06:20:59.600914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model_2A3 = RNA()\n# model_2A3.compile(optimizer=tf.keras.optimizers.Adam(learning_rate=5e-4), loss=loss)","metadata":{"execution":{"iopub.status.busy":"2023-12-06T06:20:59.604201Z","iopub.execute_input":"2023-12-06T06:20:59.60479Z","iopub.status.idle":"2023-12-06T06:20:59.608343Z","shell.execute_reply.started":"2023-12-06T06:20:59.604761Z","shell.execute_reply":"2023-12-06T06:20:59.607641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_ds = tf.data.Dataset.from_tensor_slices((X_2A3.values, Y_2A3.values)) \\\n#     .batch(128) \\\n#     .map(lambda x, y: (text_vectorization(x), tf.clip_by_value(y, 0, 1)))\n# train_ds = train_ds.shuffle(train_ds.cardinality())\n# train_data = train_ds.take(int(len(train_ds) * 0.8))\n# validation_data = train_ds.skip(int(len(train_ds) * 0.8)).take(int(len(train_ds) * 0.2))\n\n# model_2A3.fit(train_data, validation_data=validation_data, epochs=25, batch_size=128)","metadata":{"execution":{"iopub.status.busy":"2023-12-06T06:20:59.609249Z","iopub.execute_input":"2023-12-06T06:20:59.609507Z","iopub.status.idle":"2023-12-06T06:20:59.619506Z","shell.execute_reply.started":"2023-12-06T06:20:59.609472Z","shell.execute_reply":"2023-12-06T06:20:59.618776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text_vectorization = tf.keras.layers.TextVectorization(output_mode='int',\n                                                       ngrams=1,\n                                                       output_sequence_length=457,\n                                                       split='character',\n                                                       vocabulary=['a', 'c', 'g', 'u'])","metadata":{"execution":{"iopub.status.busy":"2023-12-06T06:20:59.620434Z","iopub.execute_input":"2023-12-06T06:20:59.620703Z","iopub.status.idle":"2023-12-06T06:20:59.637083Z","shell.execute_reply.started":"2023-12-06T06:20:59.620676Z","shell.execute_reply":"2023-12-06T06:20:59.636388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv('/kaggle/input/stanford-ribonanza-rna-folding/test_sequences.csv')","metadata":{"execution":{"iopub.status.busy":"2023-12-06T06:20:59.637892Z","iopub.execute_input":"2023-12-06T06:20:59.638132Z","iopub.status.idle":"2023-12-06T06:21:07.035815Z","shell.execute_reply.started":"2023-12-06T06:20:59.638107Z","shell.execute_reply":"2023-12-06T06:21:07.034676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds_2A3 = model_2A3.predict(tf.data.Dataset.from_tensor_slices((test.sequence)) \\\n    .batch(128) \\\n    .map(lambda x: text_vectorization(x)))","metadata":{"execution":{"iopub.status.busy":"2023-12-06T06:21:07.036843Z","iopub.execute_input":"2023-12-06T06:21:07.037098Z","iopub.status.idle":"2023-12-06T06:28:32.975503Z","shell.execute_reply.started":"2023-12-06T06:21:07.037072Z","shell.execute_reply":"2023-12-06T06:28:32.97424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_2A3 = pd.DataFrame(preds_2A3)\ndf_2A3.to_csv('preds_2A3.csv')","metadata":{"execution":{"iopub.status.busy":"2023-12-06T06:30:32.78203Z","iopub.execute_input":"2023-12-06T06:30:32.782888Z","iopub.status.idle":"2023-12-06T06:40:33.95587Z","shell.execute_reply.started":"2023-12-06T06:30:32.782849Z","shell.execute_reply":"2023-12-06T06:40:33.954583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with strategy.scope():\n    \n    model_DMS = RNA()\n    model_DMS.compile(optimizer=tf.keras.optimizers.Adam(learning_rate=5e-4), loss=loss)\n    \n    train_ds = tf.data.Dataset.from_tensor_slices((X_DMS.values, Y_DMS.values)) \\\n        .batch(128) \\\n        .map(lambda x, y: (text_vectorization(x), tf.clip_by_value(y, 0, 1)))\n    train_ds = train_ds.shuffle(train_ds.cardinality())\n    train_data = train_ds.take(int(len(train_ds) * 0.8))\n    validation_data = train_ds.skip(int(len(train_ds) * 0.8)).take(int(len(train_ds) * 0.2))\n\n    model_DMS.fit(train_data, validation_data=validation_data, epochs=25, batch_size=128)","metadata":{"execution":{"iopub.status.busy":"2023-12-06T06:40:33.957676Z","iopub.execute_input":"2023-12-06T06:40:33.958074Z","iopub.status.idle":"2023-12-06T06:40:43.502218Z","shell.execute_reply.started":"2023-12-06T06:40:33.958043Z","shell.execute_reply":"2023-12-06T06:40:43.50016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model_DMS = RNA()\n# model_DMS.compile(optimizer=tf.keras.optimizers.Adam(learning_rate=5e-4), loss=loss)","metadata":{"execution":{"iopub.status.busy":"2023-12-06T06:40:43.5029Z","iopub.status.idle":"2023-12-06T06:40:43.503214Z","shell.execute_reply.started":"2023-12-06T06:40:43.503049Z","shell.execute_reply":"2023-12-06T06:40:43.503063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_ds = tf.data.Dataset.from_tensor_slices((X_DMS.values, Y_DMS.values)) \\\n#     .batch(128) \\\n#     .map(lambda x, y: (text_vectorization(x), tf.clip_by_value(y, 0, 1)))\n# train_ds = train_ds.shuffle(train_ds.cardinality())\n# train_data = train_ds.take(int(len(train_ds) * 0.8))\n# validation_data = train_ds.skip(int(len(train_ds) * 0.8)).take(int(len(train_ds) * 0.2))\n\n# model_DMS.fit(train_data, validation_data=validation_data, epochs=25, batch_size=128)","metadata":{"execution":{"iopub.status.busy":"2023-12-06T06:40:43.504437Z","iopub.status.idle":"2023-12-06T06:40:43.504728Z","shell.execute_reply.started":"2023-12-06T06:40:43.504585Z","shell.execute_reply":"2023-12-06T06:40:43.5046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds_DMS = model_DMS.predict(tf.data.Dataset.from_tensor_slices((test.sequence)) \\\n    .batch(128) \\\n    .map(lambda x: text_vectorization(x)))","metadata":{"execution":{"iopub.status.busy":"2023-12-06T06:40:43.505582Z","iopub.status.idle":"2023-12-06T06:40:43.505909Z","shell.execute_reply.started":"2023-12-06T06:40:43.505738Z","shell.execute_reply":"2023-12-06T06:40:43.505755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_DMS = pd.DataFrame(preds_DMS)\ndf_DMS.to_csv('preds_DMS.csv')","metadata":{"execution":{"iopub.status.busy":"2023-12-06T06:40:43.548523Z","iopub.status.idle":"2023-12-06T06:40:43.548833Z","shell.execute_reply.started":"2023-12-06T06:40:43.548683Z","shell.execute_reply":"2023-12-06T06:40:43.548697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_seq_lengths = test.sequence.str.len()","metadata":{"execution":{"iopub.status.busy":"2023-12-06T06:40:43.549843Z","iopub.status.idle":"2023-12-06T06:40:43.550122Z","shell.execute_reply.started":"2023-12-06T06:40:43.549983Z","shell.execute_reply":"2023-12-06T06:40:43.549997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\n\npreds_2A3_bylength = []\n\nfor i in range(test_seq_lengths.size):\n    \n    x = np.reshape(preds_2A3[i, :test_seq_lengths[i]], (-1, 1))\n    x = np.clip(x, 0, 1)\n    \n    preds_2A3_bylength.append(x)","metadata":{"execution":{"iopub.status.busy":"2023-12-06T06:40:43.550905Z","iopub.status.idle":"2023-12-06T06:40:43.551174Z","shell.execute_reply.started":"2023-12-06T06:40:43.551041Z","shell.execute_reply":"2023-12-06T06:40:43.551054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds_2A3 = np.concatenate(preds_2A3_bylength, 0)","metadata":{"execution":{"iopub.status.busy":"2023-12-06T06:40:43.552496Z","iopub.status.idle":"2023-12-06T06:40:43.55278Z","shell.execute_reply.started":"2023-12-06T06:40:43.552641Z","shell.execute_reply":"2023-12-06T06:40:43.552655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds_DMS_bylength = []\n\nfor i in range(test_seq_lengths.size):\n    \n    x = np.reshape(preds_DMS[i, :test_seq_lengths[i]], (-1, 1))\n    x = np.clip(x, 0, 1)\n    \n    preds_DMS_bylength.append(x)","metadata":{"execution":{"iopub.status.busy":"2023-12-06T06:40:43.553464Z","iopub.status.idle":"2023-12-06T06:40:43.553729Z","shell.execute_reply.started":"2023-12-06T06:40:43.553597Z","shell.execute_reply":"2023-12-06T06:40:43.553611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds_DMS = np.concatenate(preds_DMS_bylength, 0)","metadata":{"execution":{"iopub.status.busy":"2023-12-06T06:40:43.55463Z","iopub.status.idle":"2023-12-06T06:40:43.55491Z","shell.execute_reply.started":"2023-12-06T06:40:43.554769Z","shell.execute_reply":"2023-12-06T06:40:43.554782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds_DMS.shape","metadata":{"execution":{"iopub.status.busy":"2023-12-06T06:40:43.556291Z","iopub.status.idle":"2023-12-06T06:40:43.556621Z","shell.execute_reply.started":"2023-12-06T06:40:43.55646Z","shell.execute_reply":"2023-12-06T06:40:43.556477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFRame({'id': np.arrange(preds_DMS.shape[0]),\n                           'reactivity_DMS_MaP': preds_DMS[:, 0],\n                           'reactivity_2A3_MaP': preds_2A3[:, 0]})","metadata":{"execution":{"iopub.status.busy":"2023-12-06T06:40:43.557855Z","iopub.status.idle":"2023-12-06T06:40:43.558133Z","shell.execute_reply.started":"2023-12-06T06:40:43.557997Z","shell.execute_reply":"2023-12-06T06:40:43.558011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_parquet('submission.parquet', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-12-06T06:40:43.558918Z","iopub.status.idle":"2023-12-06T06:40:43.559204Z","shell.execute_reply.started":"2023-12-06T06:40:43.559055Z","shell.execute_reply":"2023-12-06T06:40:43.559069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}