{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":51294,"databundleVersionId":7331882,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"## EDA\nimport pandas as pd\nimport plotly.express as px\nimport matplotlib.pyplot as plt\nfrom pandas import read_csv\nimport os\n\n#Data prep\nfrom keras.utils import to_categorical\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.preprocessing.sequence import pad_sequences\n\n#Model\nimport numpy as np\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Embedding, GRU, Dense, Flatten, Masking\n\n##embedding\nfrom transformers import BertTokenizer, BertModel\n","metadata":{"execution":{"iopub.status.busy":"2024-10-21T10:42:29.771533Z","iopub.execute_input":"2024-10-21T10:42:29.771977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.DataFrame(read_csv('/kaggle/input/stanford-ribonanza-rna-folding/train_data_QUICK_START.csv'))\ndf[0:5]","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Data prep\ntrain_data = df.drop(['sequence_id', 'dataset_name'], axis=1)\ntrain_data_2A3 = df.loc[df['experiment_type'] == '2A3_MaP']\ntrain_data_DMS = df.loc[df['experiment_type'] == 'DMS_MaP']\ntrain_data_2A3_shuffle = train_data_2A3.sample(frac=1).reset_index(drop=True)\ntrain_data_DMS_shuffle = train_data_DMS.sample(frac=1).reset_index(drop=True)\ntrain_data_DMS_shuffle[0:5]","metadata":{"execution":{"iopub.status.busy":"2024-10-21T07:44:45.830722Z","iopub.execute_input":"2024-10-21T07:44:45.831219Z","iopub.status.idle":"2024-10-21T07:44:48.320030Z","shell.execute_reply.started":"2024-10-21T07:44:45.831176Z","shell.execute_reply":"2024-10-21T07:44:48.318622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#encode sequences 2A3\n# Define mapping for nucleotide to integer (A=1, U=2, G=3, C=4)\nnucleotide_to_int = {'A': 1, 'U': 2, 'G': 3, 'C': 4}\n\n# Convert RNA sequences to integer sequences\ndef encode_rna_sequence(seq):\n    return [nucleotide_to_int[nuc] for nuc in seq]\n\nencoded_sequences = [encode_rna_sequence(seq) for seq in train_data_2A3_shuffle['sequence']]\n#padding des séquences plus courtes\nmax_sequence_length = max([len(seq) for seq in encoded_sequences])\nX_train_2A3 = np.array(pad_sequences(encoded_sequences, maxlen=max_sequence_length, padding='post'))\ndel encoded_sequences\n","metadata":{"execution":{"iopub.status.busy":"2024-10-21T07:44:48.322378Z","iopub.execute_input":"2024-10-21T07:44:48.322968Z","iopub.status.idle":"2024-10-21T07:44:54.932198Z","shell.execute_reply.started":"2024-10-21T07:44:48.322818Z","shell.execute_reply":"2024-10-21T07:44:54.930764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"encoded_sequences = [encode_rna_sequence(seq) for seq in train_data_DMS_shuffle['sequence']]\n#padding des séquences plus courtes\nmax_sequence_length = max([len(seq) for seq in encoded_sequences])\nX_train_DMS = np.array(pad_sequences(encoded_sequences, maxlen=max_sequence_length, padding='post'))\ndel encoded_sequences\n","metadata":{"execution":{"iopub.status.busy":"2024-10-21T07:44:54.934503Z","iopub.execute_input":"2024-10-21T07:44:54.934969Z","iopub.status.idle":"2024-10-21T07:45:01.052112Z","shell.execute_reply.started":"2024-10-21T07:44:54.934924Z","shell.execute_reply":"2024-10-21T07:45:01.050664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train_DMS = to_categorical(X_train_DMS, 5)\nX_train_2A3 = to_categorical(X_train_2A3, 5)","metadata":{"execution":{"iopub.status.busy":"2024-10-21T07:45:01.053723Z","iopub.execute_input":"2024-10-21T07:45:01.054273Z","iopub.status.idle":"2024-10-21T07:45:03.918118Z","shell.execute_reply.started":"2024-10-21T07:45:01.054213Z","shell.execute_reply":"2024-10-21T07:45:03.916807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train_DMS = np.array(train_data_DMS_shuffle.iloc[:,4:210])\ny_train_2A3 = np.array(train_data_2A3_shuffle.iloc[:,4:210])","metadata":{"execution":{"iopub.status.busy":"2024-10-21T07:45:03.921112Z","iopub.execute_input":"2024-10-21T07:45:03.921539Z","iopub.status.idle":"2024-10-21T07:45:04.316172Z","shell.execute_reply.started":"2024-10-21T07:45:03.921496Z","shell.execute_reply":"2024-10-21T07:45:04.314867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train_DMS[np.isnan(y_train_DMS)] = 0\ny_train_2A3[np.isnan(y_train_2A3)] = 0","metadata":{"execution":{"iopub.status.busy":"2024-10-21T07:45:04.317708Z","iopub.execute_input":"2024-10-21T07:45:04.318262Z","iopub.status.idle":"2024-10-21T07:45:04.600939Z","shell.execute_reply.started":"2024-10-21T07:45:04.318203Z","shell.execute_reply":"2024-10-21T07:45:04.599525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Custom loss function to mask NaN reactivity values\ndef masked_mae(y_true, y_pred):\n    # Create a mask where the NaN values are replaced by 0 (not included in loss)\n    mask = tf.cast(~tf.math.is_nan(y_true), dtype=tf.float32)\n    \n    # Replace NaN values in y_true with 0 (or any placeholder, since they are ignored)\n    y_true_masked = tf.where(tf.math.is_nan(y_true), tf.zeros_like(y_true), y_true)\n    \n    # Compute the squared differences\n    abs_diff = tf.abs(y_true_masked - y_pred)\n    \n    # Multiply by the mask to ignore NaN timesteps (zero-out loss for NaN values)\n    masked_abs_difference = abs_diff * mask\n    \n    # Return the mean loss (only for valid values)\n    return tf.reduce_sum(masked_abs_difference) / tf.reduce_sum(mask)\n\n# Now, compile the model with the custom masked loss function","metadata":{"execution":{"iopub.status.busy":"2024-10-21T07:45:04.602871Z","iopub.execute_input":"2024-10-21T07:45:04.603565Z","iopub.status.idle":"2024-10-21T07:45:04.611369Z","shell.execute_reply.started":"2024-10-21T07:45:04.603506Z","shell.execute_reply":"2024-10-21T07:45:04.610011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# GRU\ndef GRU_model():\n    hidden_units = 32\n    input_shape  = (206, 5)\n    dense_units  = 206\n    model = Sequential()\n    model.add(GRU(hidden_units, input_shape=input_shape, return_sequences=True))\n    model.add(Flatten())\n    model.add(Dense(units=dense_units, activation='tanh'))\n\n    # Compile model\n    model.compile(loss=masked_mae, optimizer='adam')\n    return model","metadata":{"execution":{"iopub.status.busy":"2024-10-21T07:45:04.613174Z","iopub.execute_input":"2024-10-21T07:45:04.613614Z","iopub.status.idle":"2024-10-21T07:45:04.627407Z","shell.execute_reply.started":"2024-10-21T07:45:04.613570Z","shell.execute_reply":"2024-10-21T07:45:04.625955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_GRU = GRU_model()\nprint(f'{model_GRU.summary()}')","metadata":{"execution":{"iopub.status.busy":"2024-10-21T07:45:08.302070Z","iopub.execute_input":"2024-10-21T07:45:08.302540Z","iopub.status.idle":"2024-10-21T07:45:08.379361Z","shell.execute_reply.started":"2024-10-21T07:45:08.302495Z","shell.execute_reply":"2024-10-21T07:45:08.378291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model_GRU.fit(X_train_DMS, y_train_DMS, validation_split = 0.2, epochs=30, batch_size=128)","metadata":{"execution":{"iopub.status.busy":"2024-10-21T07:45:11.556537Z","iopub.execute_input":"2024-10-21T07:45:11.557037Z","iopub.status.idle":"2024-10-21T09:08:05.160921Z","shell.execute_reply.started":"2024-10-21T07:45:11.556989Z","shell.execute_reply":"2024-10-21T09:08:05.158980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# \"Loss\"\nplt.plot(history.history['loss'])\nplt.plot(history.history['val_loss'])\nplt.title('model loss')\nplt.ylabel('loss')\nplt.xlabel('epoch')\nplt.legend(['train', 'validation'], loc='upper left')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-21T09:08:05.164623Z","iopub.execute_input":"2024-10-21T09:08:05.165160Z","iopub.status.idle":"2024-10-21T09:08:05.515245Z","shell.execute_reply.started":"2024-10-21T09:08:05.165108Z","shell.execute_reply":"2024-10-21T09:08:05.513812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_GRU.save(\"/kaggle/working/GRU_simple_1data_DMS.keras\")","metadata":{"execution":{"iopub.status.busy":"2024-10-21T09:08:05.516513Z","iopub.execute_input":"2024-10-21T09:08:05.516946Z","iopub.status.idle":"2024-10-21T09:08:05.656780Z","shell.execute_reply.started":"2024-10-21T09:08:05.516902Z","shell.execute_reply":"2024-10-21T09:08:05.655444Z"},"trusted":true},"execution_count":null,"outputs":[]}]}