{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":51294,"databundleVersionId":6923401,"sourceType":"competition"}],"dockerImageVersionId":30587,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Quick notebook to identify the sequences that are shared between the training set and the public leaderboard.\n\n#### From the additional notes on the data of this competition:\n*     https://www.kaggle.com/competitions/stanford-ribonanza-rna-folding/data\n*     We note that 37,828 of the test sequences, derived from the RFAM database, are identical or near-identical to the train set; for these cases the leaderboard test set contains higher signal-to-noise measurements than the train_data and serve as a test of model ability to 'denoise' chemical mapping data.\n\nI'm learning polars, so if you know of a more efficient way of doing it with polars, please show me.","metadata":{}},{"cell_type":"code","source":"from os import path\nimport gc\nimport polars as pl","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-11-17T16:51:34.926669Z","iopub.execute_input":"2023-11-17T16:51:34.927230Z","iopub.status.idle":"2023-11-17T16:51:34.933305Z","shell.execute_reply.started":"2023-11-17T16:51:34.927190Z","shell.execute_reply":"2023-11-17T16:51:34.932142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CFG:\n    base_path = \"../input/stanford-ribonanza-rna-folding\"\n    columns_oi = ['sequence_id', 'sequence']\n    csv_train_qs = \"train_data_QUICK_START.csv\"\n    csv_train = \"train_data.csv\"\n    csv_inf = \"test_sequences.csv\"\n    max_seq_len = 206  # from the data info of the competition","metadata":{"execution":{"iopub.status.busy":"2023-11-17T16:51:35.864256Z","iopub.execute_input":"2023-11-17T16:51:35.864753Z","iopub.status.idle":"2023-11-17T16:51:35.870963Z","shell.execute_reply.started":"2023-11-17T16:51:35.864718Z","shell.execute_reply":"2023-11-17T16:51:35.869767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Training dataset\ndf_train = (\n    pl.scan_csv(path.join(CFG.base_path, CFG.csv_train))\n    .select(CFG.columns_oi)\n    .unique(subset=[\"sequence\"], maintain_order=True)\n)\n\n# Inference dataset\ndf_inf = (\n    pl.scan_csv(path.join(CFG.base_path, CFG.csv_inf))\n    .select(CFG.columns_oi)\n    .filter(\n        pl.col('sequence')\n        .str.len_bytes()\n        .le(CFG.max_seq_len)\n    )\n)\n\n# Sequences that are in both datasets.\ndf = (\n    pl.concat([df_train, df_inf], how=\"vertical\", parallel=True)\n    .filter(\n        pl.col('sequence')\n        .is_duplicated()\n    )\n    .unique(subset=['sequence'], maintain_order=True)\n).collect()\n\ndel df_train, df_inf\ngc.collect();","metadata":{"execution":{"iopub.status.busy":"2023-11-17T16:51:36.293377Z","iopub.execute_input":"2023-11-17T16:51:36.293876Z","iopub.status.idle":"2023-11-17T16:51:42.277473Z","shell.execute_reply.started":"2023-11-17T16:51:36.293843Z","shell.execute_reply":"2023-11-17T16:51:42.276242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Number of identified sequences\ndf.select(pl.count()).item()","metadata":{"execution":{"iopub.status.busy":"2023-11-17T16:51:42.279736Z","iopub.execute_input":"2023-11-17T16:51:42.280169Z","iopub.status.idle":"2023-11-17T16:51:42.289435Z","shell.execute_reply.started":"2023-11-17T16:51:42.280111Z","shell.execute_reply":"2023-11-17T16:51:42.288285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Number of non identified sequences\n37_828 - 37_677","metadata":{"execution":{"iopub.status.busy":"2023-11-17T16:51:42.290737Z","iopub.execute_input":"2023-11-17T16:51:42.291157Z","iopub.status.idle":"2023-11-17T16:51:42.301164Z","shell.execute_reply.started":"2023-11-17T16:51:42.291128Z","shell.execute_reply":"2023-11-17T16:51:42.300096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Ideas to find the last 151 sequences, remove the markers and check again for identical sequences, if not luck with that -> clustering without the markers.","metadata":{}},{"cell_type":"markdown","source":"## Filtering the identified sequences from the train_data_QUICK_START.csv","metadata":{"execution":{"iopub.status.busy":"2023-11-17T16:20:58.361625Z","iopub.execute_input":"2023-11-17T16:20:58.362126Z","iopub.status.idle":"2023-11-17T16:20:58.370703Z","shell.execute_reply.started":"2023-11-17T16:20:58.362089Z","shell.execute_reply":"2023-11-17T16:20:58.369260Z"}}},{"cell_type":"code","source":"# Quickstart dataset mostly clean\ndf_qs_clean = (\n    pl.scan_csv(path.join(CFG.base_path, CFG.csv_train_qs))\n    .filter(\n        pl.col('sequence')\n        .is_in(df['sequence'])\n        .not_()\n    )\n).collect()\n\n# Shared sequences\nshared_seq = (\n    pl.scan_csv(path.join(CFG.base_path, CFG.csv_train_qs))\n    .filter(\n        pl.col('sequence')\n        .is_in(df['sequence'])\n    )\n).collect()","metadata":{"execution":{"iopub.status.busy":"2023-11-17T16:56:26.605931Z","iopub.execute_input":"2023-11-17T16:56:26.606520Z","iopub.status.idle":"2023-11-17T16:56:33.892367Z","shell.execute_reply.started":"2023-11-17T16:56:26.606479Z","shell.execute_reply":"2023-11-17T16:56:33.891325Z"},"trusted":true},"execution_count":null,"outputs":[]}]}