{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":51294,"databundleVersionId":6923401,"sourceType":"competition"},{"sourceId":7074723,"sourceType":"datasetVersion","datasetId":4074776}],"dockerImageVersionId":30588,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from shutil import copyfile\n\ncopyfile(src = \"/kaggle/input/rna-transformer-model/TransformerNoExperiment.py\", dst = \"../working/TransformerNoExperiment.py\")\ncopyfile(src = \"/kaggle/input/rna-transformer-model/TransformerExperimentToken.py\", dst = \"../working/TransformerExperimentToken.py\")\ncopyfile(src = \"/kaggle/input/rna-transformer-model/TransformerExperimentAdditive.py\", dst = \"../working/TransformerExperimentAdditive.py\")","metadata":{"execution":{"iopub.status.busy":"2023-11-28T05:58:47.370405Z","iopub.execute_input":"2023-11-28T05:58:47.371005Z","iopub.status.idle":"2023-11-28T05:58:47.393756Z","shell.execute_reply.started":"2023-11-28T05:58:47.370976Z","shell.execute_reply":"2023-11-28T05:58:47.392896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nimport pandas as pd\n\nfrom sklearn.model_selection import train_test_split\nfrom matplotlib import pyplot as plt\n\nfrom TransformerNoExperiment import run_experiment as run_experiment_no\nfrom TransformerExperimentAdditive import run_experiment as run_experiment_additive\nfrom TransformerExperimentToken import run_experiment as run_experiment_token","metadata":{"execution":{"iopub.status.busy":"2023-11-28T05:58:56.466508Z","iopub.execute_input":"2023-11-28T05:58:56.466948Z","iopub.status.idle":"2023-11-28T05:59:00.236950Z","shell.execute_reply.started":"2023-11-28T05:58:56.466920Z","shell.execute_reply":"2023-11-28T05:59:00.235915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DEVICE = \"cuda\" if torch.cuda.is_available() else \"cpu\"\nDEVICE","metadata":{"execution":{"iopub.status.busy":"2023-11-28T05:59:09.634038Z","iopub.execute_input":"2023-11-28T05:59:09.635169Z","iopub.status.idle":"2023-11-28T05:59:09.641913Z","shell.execute_reply.started":"2023-11-28T05:59:09.635125Z","shell.execute_reply":"2023-11-28T05:59:09.640922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Data Constants\nFILENAME = \"/kaggle/input/stanford-ribonanza-rna-folding/train_data_QUICK_START.csv\"\n\nDATA_ROWS = 16000\nTEST_SIZE = 0.2","metadata":{"execution":{"iopub.status.busy":"2023-11-28T05:59:22.529860Z","iopub.execute_input":"2023-11-28T05:59:22.530817Z","iopub.status.idle":"2023-11-28T05:59:22.535165Z","shell.execute_reply.started":"2023-11-28T05:59:22.530775Z","shell.execute_reply":"2023-11-28T05:59:22.534300Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Training Constants\nEPOCHS = 75\nBATCH_SIZE = 32\nLEARNING_RATE = 1e-3","metadata":{"execution":{"iopub.status.busy":"2023-11-28T05:59:22.537248Z","iopub.execute_input":"2023-11-28T05:59:22.537783Z","iopub.status.idle":"2023-11-28T05:59:22.554451Z","shell.execute_reply.started":"2023-11-28T05:59:22.537752Z","shell.execute_reply":"2023-11-28T05:59:22.553488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_csv_data(filename=FILENAME, test_size=TEST_SIZE, num_rows=1000):\n    df = pd.read_csv(filename)\n    df = df.iloc[:num_rows, :]\n\n    # Reformat the reactivity columns\n    reactivity_columns = [col for col in df.columns if col.startswith('reactivity_0')]\n    df['reactivity'] = df[reactivity_columns].values.tolist()\n\n    # Select the relevant columns\n    clean_df = df.loc[:, ['sequence', 'experiment_type', 'reactivity']]\n\n    # Split into train and test sets\n    train_df, test_df = train_test_split(clean_df, test_size=test_size)\n    return train_df, test_df","metadata":{"execution":{"iopub.status.busy":"2023-11-28T05:59:22.555451Z","iopub.execute_input":"2023-11-28T05:59:22.555714Z","iopub.status.idle":"2023-11-28T05:59:22.564473Z","shell.execute_reply.started":"2023-11-28T05:59:22.555693Z","shell.execute_reply":"2023-11-28T05:59:22.563621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_losses(train_losses, test_losses):\n    fig, axs = plt.subplots(1, 1, figsize=(12, 4))\n    axs.plot(train_losses, c=\"r\", label=\"Train loss\")\n    axs.plot(test_losses, c=\"b\", label=\"Test loss\")\n    axs.legend()\n    axs.set_xlabel(\"Epochs\")\n\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-28T05:59:22.566705Z","iopub.execute_input":"2023-11-28T05:59:22.566955Z","iopub.status.idle":"2023-11-28T05:59:22.573793Z","shell.execute_reply.started":"2023-11-28T05:59:22.566933Z","shell.execute_reply":"2023-11-28T05:59:22.572921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df, test_df = load_csv_data(num_rows=DATA_ROWS)","metadata":{"execution":{"iopub.status.busy":"2023-11-28T05:59:22.574754Z","iopub.execute_input":"2023-11-28T05:59:22.575005Z","iopub.status.idle":"2023-11-28T05:59:42.240054Z","shell.execute_reply.started":"2023-11-28T05:59:22.574983Z","shell.execute_reply":"2023-11-28T05:59:42.239246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Experiment parameters\nNUM_LAYERS = [3, 5, 10, 20]","metadata":{"execution":{"iopub.status.busy":"2023-11-28T05:59:42.241150Z","iopub.execute_input":"2023-11-28T05:59:42.241403Z","iopub.status.idle":"2023-11-28T05:59:42.245702Z","shell.execute_reply.started":"2023-11-28T05:59:42.241381Z","shell.execute_reply":"2023-11-28T05:59:42.244775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Use model with no experiment\nprint(\"Running experiment with no experiment\")\nfinal_test_losses = []\nfor num_layers in NUM_LAYERS:\n    print(f\"Running experiment with {num_layers} layers\")\n    train_losses, test_losses = run_experiment_no(train_df, test_df, n_layers=num_layers, n_head = 4, d_model = 128, epochs=EPOCHS, batch_size=BATCH_SIZE, learning_rate=LEARNING_RATE)\n    final_test_losses.append(test_losses[-1])\n    plot_losses(train_losses, test_losses)\nprint(final_test_losses)","metadata":{"execution":{"iopub.status.busy":"2023-11-28T05:59:42.247024Z","iopub.execute_input":"2023-11-28T05:59:42.247683Z","iopub.status.idle":"2023-11-28T07:49:31.789540Z","shell.execute_reply.started":"2023-11-28T05:59:42.247644Z","shell.execute_reply":"2023-11-28T07:49:31.788620Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Use model with additive experiment embedding\nprint(\"Running experiment with additive experiment embedding\")\nfinal_test_losses = []\nfor num_layers in NUM_LAYERS:\n    print(f\"Running experiment with {num_layers} layers\")\n    train_losses, test_losses = run_experiment_additive(train_df, test_df, n_layers=num_layers, n_head = 4, d_model = 128, epochs=EPOCHS, batch_size=BATCH_SIZE, learning_rate=LEARNING_RATE)\n    final_test_losses.append(test_losses[-1])\n    plot_losses(train_losses, test_losses)\nprint(final_test_losses)","metadata":{"execution":{"iopub.status.busy":"2023-11-28T07:49:31.790894Z","iopub.execute_input":"2023-11-28T07:49:31.791523Z","iopub.status.idle":"2023-11-28T09:41:25.445543Z","shell.execute_reply.started":"2023-11-28T07:49:31.791486Z","shell.execute_reply":"2023-11-28T09:41:25.444655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Use model with token experiment embedding\nprint(\"Running experiment with token experiment embedding\")\nfinal_test_losses = []\nfor num_layers in NUM_LAYERS:\n    print(f\"Running experiment with {num_layers} layers\")\n    train_losses, test_losses = run_experiment_token(train_df, test_df, n_layers=num_layers, n_head = 4, d_model = 128, epochs=EPOCHS, batch_size=BATCH_SIZE, learning_rate=LEARNING_RATE)\n    final_test_losses.append(test_losses[-1])\n    plot_losses(train_losses, test_losses)\nprint(final_test_losses)","metadata":{"execution":{"iopub.status.busy":"2023-11-28T09:41:25.446963Z","iopub.execute_input":"2023-11-28T09:41:25.447665Z","iopub.status.idle":"2023-11-28T11:28:37.515553Z","shell.execute_reply.started":"2023-11-28T09:41:25.447620Z","shell.execute_reply":"2023-11-28T11:28:37.514648Z"},"trusted":true},"execution_count":null,"outputs":[]}]}